@houtini/seo-audit-console 0.8.0 → 0.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (207) hide show
  1. package/README.md +308 -306
  2. package/dist/audit/checks.d.ts +0 -7
  3. package/dist/audit/checks.d.ts.map +1 -1
  4. package/dist/audit/checks.js +130 -242
  5. package/dist/audit/checks.js.map +1 -1
  6. package/dist/audit/drift.d.ts +0 -12
  7. package/dist/audit/drift.d.ts.map +1 -1
  8. package/dist/audit/drift.js +0 -15
  9. package/dist/audit/drift.js.map +1 -1
  10. package/dist/audit/engine.d.ts +0 -3
  11. package/dist/audit/engine.d.ts.map +1 -1
  12. package/dist/audit/engine.js +7 -42
  13. package/dist/audit/engine.js.map +1 -1
  14. package/dist/audit/keywordList.d.ts +0 -14
  15. package/dist/audit/keywordList.d.ts.map +1 -1
  16. package/dist/audit/keywordList.js +0 -5
  17. package/dist/audit/keywordList.js.map +1 -1
  18. package/dist/audit/opportunities.js +3 -26
  19. package/dist/audit/opportunities.js.map +1 -1
  20. package/dist/audit/recon.d.ts +0 -31
  21. package/dist/audit/recon.d.ts.map +1 -1
  22. package/dist/audit/recon.js +0 -27
  23. package/dist/audit/recon.js.map +1 -1
  24. package/dist/audit/report.d.ts +0 -4
  25. package/dist/audit/report.d.ts.map +1 -1
  26. package/dist/audit/report.js +0 -7
  27. package/dist/audit/report.js.map +1 -1
  28. package/dist/audit/schema-validate.d.ts +0 -26
  29. package/dist/audit/schema-validate.d.ts.map +1 -1
  30. package/dist/audit/schema-validate.js +8 -64
  31. package/dist/audit/schema-validate.js.map +1 -1
  32. package/dist/audit/templates.d.ts +0 -24
  33. package/dist/audit/templates.d.ts.map +1 -1
  34. package/dist/audit/templates.js +1 -21
  35. package/dist/audit/templates.js.map +1 -1
  36. package/dist/audit/topicGaps.d.ts +0 -1
  37. package/dist/audit/topicGaps.d.ts.map +1 -1
  38. package/dist/audit/topicGaps.js +3 -24
  39. package/dist/audit/topicGaps.js.map +1 -1
  40. package/dist/core/AuditDatabase.d.ts +0 -14
  41. package/dist/core/AuditDatabase.d.ts.map +1 -1
  42. package/dist/core/AuditDatabase.js +8 -61
  43. package/dist/core/AuditDatabase.js.map +1 -1
  44. package/dist/core/Backlinks.d.ts +0 -7
  45. package/dist/core/Backlinks.d.ts.map +1 -1
  46. package/dist/core/Backlinks.js +1 -23
  47. package/dist/core/Backlinks.js.map +1 -1
  48. package/dist/core/Crawler.d.ts +0 -1
  49. package/dist/core/Crawler.d.ts.map +1 -1
  50. package/dist/core/Crawler.js +24 -119
  51. package/dist/core/Crawler.js.map +1 -1
  52. package/dist/core/DataForSeoClient.d.ts +0 -74
  53. package/dist/core/DataForSeoClient.d.ts.map +1 -1
  54. package/dist/core/DataForSeoClient.js +1 -72
  55. package/dist/core/DataForSeoClient.js.map +1 -1
  56. package/dist/core/Entities.d.ts +0 -6
  57. package/dist/core/Entities.d.ts.map +1 -1
  58. package/dist/core/Entities.js +0 -8
  59. package/dist/core/Entities.js.map +1 -1
  60. package/dist/core/FirecrawlClient.d.ts +0 -18
  61. package/dist/core/FirecrawlClient.d.ts.map +1 -1
  62. package/dist/core/FirecrawlClient.js +1 -13
  63. package/dist/core/FirecrawlClient.js.map +1 -1
  64. package/dist/core/GscClient.d.ts +0 -7
  65. package/dist/core/GscClient.d.ts.map +1 -1
  66. package/dist/core/GscClient.js +1 -10
  67. package/dist/core/GscClient.js.map +1 -1
  68. package/dist/core/GscSync.d.ts +0 -4
  69. package/dist/core/GscSync.d.ts.map +1 -1
  70. package/dist/core/GscSync.js +2 -33
  71. package/dist/core/GscSync.js.map +1 -1
  72. package/dist/core/JobManager.d.ts +0 -5
  73. package/dist/core/JobManager.d.ts.map +1 -1
  74. package/dist/core/JobManager.js +0 -8
  75. package/dist/core/JobManager.js.map +1 -1
  76. package/dist/core/LinkIntersect.d.ts +0 -1
  77. package/dist/core/LinkIntersect.d.ts.map +1 -1
  78. package/dist/core/LinkIntersect.js +1 -37
  79. package/dist/core/LinkIntersect.js.map +1 -1
  80. package/dist/core/MajesticClient.d.ts +0 -26
  81. package/dist/core/MajesticClient.d.ts.map +1 -1
  82. package/dist/core/MajesticClient.js +0 -15
  83. package/dist/core/MajesticClient.js.map +1 -1
  84. package/dist/core/RankTracker.d.ts +0 -4
  85. package/dist/core/RankTracker.d.ts.map +1 -1
  86. package/dist/core/RankTracker.js +0 -9
  87. package/dist/core/RankTracker.js.map +1 -1
  88. package/dist/core/Refresh.d.ts +0 -6
  89. package/dist/core/Refresh.d.ts.map +1 -1
  90. package/dist/core/Refresh.js +0 -6
  91. package/dist/core/Refresh.js.map +1 -1
  92. package/dist/core/SupadataClient.d.ts +0 -11
  93. package/dist/core/SupadataClient.d.ts.map +1 -1
  94. package/dist/core/SupadataClient.js +0 -5
  95. package/dist/core/SupadataClient.js.map +1 -1
  96. package/dist/core/UrlInspector.d.ts +0 -5
  97. package/dist/core/UrlInspector.d.ts.map +1 -1
  98. package/dist/core/UrlInspector.js +0 -6
  99. package/dist/core/UrlInspector.js.map +1 -1
  100. package/dist/core/WikidataClient.d.ts +0 -2
  101. package/dist/core/WikidataClient.d.ts.map +1 -1
  102. package/dist/core/WikidataClient.js +0 -7
  103. package/dist/core/WikidataClient.js.map +1 -1
  104. package/dist/core/agentReadiness.d.ts +0 -9
  105. package/dist/core/agentReadiness.d.ts.map +1 -1
  106. package/dist/core/agentReadiness.js +0 -13
  107. package/dist/core/agentReadiness.js.map +1 -1
  108. package/dist/core/ctrModel.js +0 -3
  109. package/dist/core/ctrModel.js.map +1 -1
  110. package/dist/core/dashboardData.d.ts +0 -1
  111. package/dist/core/dashboardData.d.ts.map +1 -1
  112. package/dist/core/dashboardData.js +13 -95
  113. package/dist/core/dashboardData.js.map +1 -1
  114. package/dist/core/dataStorage.d.ts +0 -2
  115. package/dist/core/dataStorage.d.ts.map +1 -1
  116. package/dist/core/dataStorage.js +6 -19
  117. package/dist/core/dataStorage.js.map +1 -1
  118. package/dist/core/draftBrief.d.ts +0 -7
  119. package/dist/core/draftBrief.d.ts.map +1 -1
  120. package/dist/core/draftBrief.js +0 -10
  121. package/dist/core/draftBrief.js.map +1 -1
  122. package/dist/core/extract.d.ts +0 -11
  123. package/dist/core/extract.d.ts.map +1 -1
  124. package/dist/core/extract.js +1 -38
  125. package/dist/core/extract.js.map +1 -1
  126. package/dist/core/googleNews.d.ts +0 -11
  127. package/dist/core/googleNews.d.ts.map +1 -1
  128. package/dist/core/googleNews.js +0 -7
  129. package/dist/core/googleNews.js.map +1 -1
  130. package/dist/core/gscFreshness.d.ts +0 -10
  131. package/dist/core/gscFreshness.d.ts.map +1 -1
  132. package/dist/core/gscFreshness.js +0 -11
  133. package/dist/core/gscFreshness.js.map +1 -1
  134. package/dist/core/linkGraph.d.ts +0 -15
  135. package/dist/core/linkGraph.d.ts.map +1 -1
  136. package/dist/core/linkGraph.js +1 -23
  137. package/dist/core/linkGraph.js.map +1 -1
  138. package/dist/core/marketSizing.d.ts +0 -13
  139. package/dist/core/marketSizing.d.ts.map +1 -1
  140. package/dist/core/marketSizing.js +1 -1
  141. package/dist/core/marketSizing.js.map +1 -1
  142. package/dist/core/passageScore.d.ts +0 -13
  143. package/dist/core/passageScore.d.ts.map +1 -1
  144. package/dist/core/passageScore.js +1 -16
  145. package/dist/core/passageScore.js.map +1 -1
  146. package/dist/core/paths.d.ts +0 -3
  147. package/dist/core/paths.d.ts.map +1 -1
  148. package/dist/core/paths.js +0 -0
  149. package/dist/core/paths.js.map +1 -1
  150. package/dist/core/queryData.d.ts +0 -1
  151. package/dist/core/queryData.d.ts.map +1 -1
  152. package/dist/core/queryData.js +0 -18
  153. package/dist/core/queryData.js.map +1 -1
  154. package/dist/core/reconFetch.d.ts +0 -6
  155. package/dist/core/reconFetch.d.ts.map +1 -1
  156. package/dist/core/reconFetch.js +1 -4
  157. package/dist/core/reconFetch.js.map +1 -1
  158. package/dist/core/reconResearch.d.ts +0 -20
  159. package/dist/core/reconResearch.d.ts.map +1 -1
  160. package/dist/core/reconResearch.js +1 -16
  161. package/dist/core/reconResearch.js.map +1 -1
  162. package/dist/core/reranker.d.ts +0 -2
  163. package/dist/core/reranker.d.ts.map +1 -1
  164. package/dist/core/reranker.js +1 -11
  165. package/dist/core/reranker.js.map +1 -1
  166. package/dist/core/robots.d.ts +0 -6
  167. package/dist/core/robots.d.ts.map +1 -1
  168. package/dist/core/robots.js +1 -7
  169. package/dist/core/robots.js.map +1 -1
  170. package/dist/core/serpFootprint.d.ts +0 -12
  171. package/dist/core/serpFootprint.d.ts.map +1 -1
  172. package/dist/core/serpFootprint.js +0 -1
  173. package/dist/core/serpFootprint.js.map +1 -1
  174. package/dist/core/serpRecon.d.ts +0 -14
  175. package/dist/core/serpRecon.d.ts.map +1 -1
  176. package/dist/core/serpRecon.js +1 -13
  177. package/dist/core/serpRecon.js.map +1 -1
  178. package/dist/core/sitemap.js +3 -16
  179. package/dist/core/sitemap.js.map +1 -1
  180. package/dist/core/sql.d.ts +0 -4
  181. package/dist/core/sql.d.ts.map +1 -1
  182. package/dist/core/sql.js +0 -4
  183. package/dist/core/sql.js.map +1 -1
  184. package/dist/core/url-key.d.ts +0 -31
  185. package/dist/core/url-key.d.ts.map +1 -1
  186. package/dist/core/url-key.js +1 -38
  187. package/dist/core/url-key.js.map +1 -1
  188. package/dist/core/webHandlers.d.ts +0 -6
  189. package/dist/core/webHandlers.d.ts.map +1 -1
  190. package/dist/core/webHandlers.js +2 -5
  191. package/dist/core/webHandlers.js.map +1 -1
  192. package/dist/core/webServer.d.ts +0 -16
  193. package/dist/core/webServer.d.ts.map +1 -1
  194. package/dist/core/webServer.js +3 -17
  195. package/dist/core/webServer.js.map +1 -1
  196. package/dist/dashboard.js +0 -23
  197. package/dist/dashboard.js.map +1 -1
  198. package/dist/generators/index.d.ts +0 -7
  199. package/dist/generators/index.d.ts.map +1 -1
  200. package/dist/generators/index.js +1 -26
  201. package/dist/generators/index.js.map +1 -1
  202. package/dist/index.js +0 -2
  203. package/dist/index.js.map +1 -1
  204. package/dist/server.js +16 -166
  205. package/dist/server.js.map +1 -1
  206. package/package.json +2 -2
  207. package/server.json +2 -2
@@ -5,33 +5,17 @@ import { expectedCtr } from '../core/ctrModel.js';
5
5
  import { latestTwoCrawls } from './drift.js';
6
6
  import { HTML_CT } from '../core/sql.js';
7
7
  const rows = (ctx, sql, ...args) => ctx.db.prepare(sql).all(...args);
8
- // Streaming variant — yields one row at a time instead of materialising the whole result set. Use for
9
- // checks that scan a large/fat column (e.g. body_chunks) and only need independent per-row work, so a
10
- // big content site doesn't load every page's body text into one array.
11
8
  const iterRows = (ctx, sql, ...args) => ctx.db.prepare(sql).iterate(...args);
12
- // d is the finalised GSC date (partial trailing days already trimmed upstream); cap the window at
13
- // it so checks never count unfinalised days. winPrev already caps below d, so it's unaffected.
14
9
  const win = (d) => `date > date('${d}', '-28 days') AND date <= '${d}'`;
15
- // Prior 28-day window (the 28 days BEFORE the current window) — for period-over-period checks.
16
10
  const winPrev = (d) => `date <= date('${d}', '-28 days') AND date > date('${d}', '-56 days')`;
17
- // SQL clause to drop branded queries (whole-token match). Branded multi-URL ranking is sitelinks,
18
- // not cannibalisation; branded top-queries aren't anchor targets. The brand is re-sanitised to
19
- // alnum HERE (not trusting the caller) so inlining it into SQL is unconditionally injection-safe.
20
11
  const brandExcl = (c) => {
21
12
  const b = (c.brand ?? '').replace(/[^a-z0-9]/g, '');
22
13
  return b ? `AND (' ' || LOWER(query) || ' ') NOT LIKE '% ${b} %'` : '';
23
14
  };
24
- // Days of GSC history actually held — period-over-period checks need enough span to be meaningful.
25
15
  const spanDays = (c) => {
26
- // Span must be measured up to the FINALISED max date the windows actually key off
27
- // (gscMaxDate is trimmed ~3 days below raw MAX(date)) — measuring the raw span lets the
28
- // ≥56-day guard pass while the previous-28d window still reaches before MIN(date),
29
- // undercounting the prior period (rising-pages FPs, traffic-decay FNs).
30
16
  const r = c.db.prepare(`SELECT julianday(?) - julianday(MIN(date)) d FROM search_analytics`).get(c.gscMaxDate ?? null);
31
17
  return r?.d ?? 0;
32
18
  };
33
- // Newest dateModified/datePublished anywhere in a page's JSON-LD (recurses @graph/arrays) — the
34
- // effective "last meaningfully updated" date. json_ld is a JSON array of raw block strings.
35
19
  const newestSchemaDate = (jl) => {
36
20
  if (!jl)
37
21
  return null;
@@ -58,27 +42,15 @@ const newestSchemaDate = (jl) => {
58
42
  try {
59
43
  scan(JSON.parse(block));
60
44
  }
61
- catch { /* skip block */ }
45
+ catch { }
62
46
  }
63
47
  }
64
- catch { /* skip */ }
48
+ catch { }
65
49
  return bestStr;
66
50
  };
67
- // Paginated archive URLs (/page/2, ?page=3, ?paged=2). They legitimately share titles/metas with
68
- // page 1 and are intentionally absent from sitemaps, so they must NOT generate duplicate-title /
69
- // duplicate-meta / missing-meta / not-in-sitemap false positives. Pass the column reference
70
- // (e.g. 'url_key' or 'p.url_key') so it composes with table aliases.
71
- const notPagination = (col = 'url_key') =>
72
- // Anchor the param name to ?/& — a bare LIKE '%page=%' also matches per_page=/on_page=/
73
- // homepage=, wrongly exempting those URLs from duplicate-title/meta/sitemap checks.
74
- `${col} NOT GLOB '*/page/[0-9]*' AND ${col} NOT GLOB '*[?&]page=[0-9]*' AND ${col} NOT GLOB '*[?&]paged=[0-9]*'`;
75
- // Significant query terms (drop stopwords; keep ≥2 chars so "vr"/"pc"/"ai" count).
51
+ const notPagination = (col = 'url_key') => `${col} NOT GLOB '*/page/[0-9]*' AND ${col} NOT GLOB '*[?&]page=[0-9]*' AND ${col} NOT GLOB '*[?&]paged=[0-9]*'`;
76
52
  const STOP = new Set(['the', 'a', 'an', 'and', 'or', 'of', 'for', 'to', 'in', 'on', 'with', 'your', 'you', 'is', 'are', 'best', 'how', 'what', 'vs', 'why', 'can']);
77
53
  const terms = (s) => (s || '').toLowerCase().replace(/[^a-z0-9 ]+/g, ' ').split(/\s+/).filter(t => t.length >= 2 && !STOP.has(t));
78
- // A query term is "present" in the title if a TITLE WORD matches it (exact, or a
79
- // plural/stem prefix either way for ≥4-char tokens), or the space-collapsed title
80
- // contains it (for multi-word brands like "sync mesh" ≈ "syncmesh", ≥5 chars).
81
- // Word-level avoids substring false-matches (e.g. "art" inside "smart").
82
54
  const titleHasTerm = (title, term) => {
83
55
  const t = title.toLowerCase();
84
56
  const words = t.split(/[^a-z0-9]+/).filter(Boolean);
@@ -86,7 +58,6 @@ const titleHasTerm = (title, term) => {
86
58
  return true;
87
59
  return term.length >= 5 && t.replace(/[^a-z0-9]+/g, '').includes(term);
88
60
  };
89
- // Validate captured JSON-LD per page, keeping findings whose issue-kinds match `kinds`.
90
61
  function schemaFindings(ctx, kinds) {
91
62
  const set = new Set(kinds);
92
63
  const out = [];
@@ -97,11 +68,8 @@ function schemaFindings(ctx, kinds) {
97
68
  }
98
69
  return out;
99
70
  }
100
- // Position→expected-CTR curve lives in core/ctrModel (shared with the dashboard). Re-export it so
101
- // existing `import { expectedCtr } from './checks.js'` call sites keep working.
102
71
  export { expectedCtr };
103
72
  export const CHECKS = [
104
- // ── On-page (crawl, deterministic) ──────────────────────────────────────
105
73
  {
106
74
  id: 'missing-title', category: 'onpage', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
107
75
  title: 'Missing title tag', fix: 'Add a unique, descriptive <title> (~50–60 chars).',
@@ -132,19 +100,11 @@ export const CHECKS = [
132
100
  title: 'Thin content', fix: 'Expand or consolidate — under ~200 words of body text.',
133
101
  run: (c) => rows(c, `SELECT url_key urlKey, word_count FROM pages WHERE status_code=200 AND indexable=1 AND word_count < 200`).map(r => ({ urlKey: r.urlKey, evidence: { wordCount: r.word_count } })),
134
102
  },
135
- // ── Indexation / crawlability ───────────────────────────────────────────
136
103
  {
137
- // RETIRED: 'canonical-mismatch' (was HIGH). A 200 page whose canonical points to a *healthy*
138
- // 200 indexable URL is intentional consolidation (slug variants, category merges) — normal SEO,
139
- // not an issue, yet it fired HIGH on every such page (pure noise). Every actionable case is
140
- // already covered by a higher-signal check: broken-canonical-target (unhealthy target),
141
- // canonical-ignored (Google ranks the non-canonical page), canonical-conflict (GSC disagrees).
142
- // So plain "canonical points elsewhere" has no high-confidence residual — removed.
143
104
  id: 'broken-internal-links', category: 'crawlability', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
144
105
  title: 'Internal links to 4xx/5xx', fix: 'Repoint internal links to a live, canonical URL.',
145
106
  run: (c) => rows(c, `SELECT l.target_key urlKey, p.status_code status, COUNT(DISTINCT l.source_key) sources FROM links l JOIN pages p ON p.url_key=l.target_key WHERE l.is_internal=1 AND p.status_code >= 400 AND p.status_code NOT IN (429,503) GROUP BY l.target_key`).map(r => ({ urlKey: r.urlKey, evidence: { status: r.status, linkingPages: r.sources } })),
146
107
  },
147
- // ── Extractor-dependent (images + canonical shape) ──────────────────────
148
108
  {
149
109
  id: 'image-alt', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
150
110
  title: 'Images missing alt text', fix: 'Add descriptive alt text to content images (alt="" only for decorative).',
@@ -160,13 +120,8 @@ export const CHECKS = [
160
120
  title: 'Multiple canonical tags', fix: 'Keep exactly one rel=canonical — conflicting canonicals let Google pick (or ignore) one.',
161
121
  run: (c) => rows(c, `SELECT url_key urlKey, canonical_count cnt FROM pages WHERE status_code=200 AND canonical_count > 1`).map(r => ({ urlKey: r.urlKey, evidence: { canonicalCount: r.cnt } })),
162
122
  },
163
- // ── Extractor additions (CLS, headings, mixed content, directives, social) ───
164
123
  {
165
124
  id: 'images-missing-dimensions', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
166
- // Lint, NOT a measured Core Web Vital. Missing width/height attributes only cause CLS if the
167
- // CSS doesn't already reserve space — modern themes using aspect-ratio / fixed boxes have ~0
168
- // measured CLS despite missing attributes. So we report it as a hygiene lint and explicitly say
169
- // it's not a confirmed CWV issue; escalate only against field CLS (high-yield-cwv-fail does that).
170
125
  title: 'Images missing width/height attributes', fix: 'Add width & height (or rely on CSS aspect-ratio) so the browser reserves space. NOTE: this is a lint — if your CSS already reserves space (aspect-ratio / fixed box) measured CLS is likely ~0 and there is nothing to fix. Confirm with field CLS before prioritising.',
171
126
  run: (c) => rows(c, `SELECT url_key urlKey, images_missing_dimensions n, image_count total FROM pages WHERE status_code=200 AND indexable=1 AND images_missing_dimensions > 0`).map(r => ({ urlKey: r.urlKey, evidence: { missingDimensions: r.n, total: r.total, note: 'lint only — no measured CLS impact unless field data shows layout shift' } })),
172
127
  },
@@ -190,13 +145,11 @@ export const CHECKS = [
190
145
  title: 'No social share tags', fix: 'Add Open Graph (og:title/og:image) and/or Twitter Card tags so shared links render a rich preview.',
191
146
  run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND (og_tags IS NULL OR og_tags='') AND (twitter_tags IS NULL OR twitter_tags='')`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
192
147
  },
193
- // ── Security / war-stories (headers now captured) ───────────────────────
194
148
  {
195
149
  id: 'missing-hsts', category: 'security', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
196
150
  title: 'Missing HSTS header', fix: 'Add Strict-Transport-Security with a sensible max-age.',
197
151
  run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (security_headers IS NULL OR security_headers NOT LIKE '%hsts%') LIMIT 1`).map(r => ({ urlKey: r.urlKey, evidence: { note: 'representative page; HSTS is site-wide' } })),
198
152
  },
199
- // ── Merged GSC × crawl (the differentiator) ─────────────────────────────
200
153
  {
201
154
  id: 'noindex-with-traffic', category: 'indexation', severity: 'crit', labels: ['D', 'G'], certainty: 1, effortBase: 1, fixType: 'per-page',
202
155
  title: 'Noindex page still getting clicks', fix: 'Remove noindex if the page should rank — it earns clicks.',
@@ -245,7 +198,6 @@ export const CHECKS = [
245
198
  title: 'Striking-distance query (page 2)', fix: 'Small on-page + internal-link push could reach page 1.',
246
199
  run: (c) => c.gscMaxDate ? rows(c, `SELECT page_key urlKey, query, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) position, SUM(impressions) impressions FROM search_analytics WHERE query IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY query, page_key HAVING SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0)>10 AND SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0)<=20 AND SUM(impressions)>=20 ORDER BY impressions DESC LIMIT 50`).map(r => ({ urlKey: r.urlKey, evidence: { query: r.query, position: Math.round(r.position * 10) / 10, impressions: r.impressions } })) : [],
247
200
  },
248
- // ── Additions from industry checklist review (buildable on current data) ──
249
201
  {
250
202
  id: 'title-too-long', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
251
203
  title: 'Title over ~60 chars', fix: 'Trim the title so the primary keyword sits within ~60 chars.',
@@ -264,8 +216,6 @@ export const CHECKS = [
264
216
  {
265
217
  id: 'redirect-chain', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
266
218
  title: 'Redirect chain (2+ hops)', fix: 'Collapse to a single hop to the final URL.',
267
- // Count hops in SQL (json_array_length, guarded by json_valid) and filter to >=2 there — so we
268
- // never pull every redirect-bearing page into JS just to count + drop most of them.
269
219
  run: (c) => rows(c, `SELECT url_key urlKey, json_array_length(redirects) hops FROM pages
270
220
  WHERE redirects IS NOT NULL AND json_valid(redirects) AND json_array_length(redirects) >= 2`)
271
221
  .map(r => ({ urlKey: r.urlKey, evidence: { hops: r.hops } })),
@@ -273,13 +223,6 @@ export const CHECKS = [
273
223
  {
274
224
  id: 'internal-links-to-redirects', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
275
225
  title: 'Internal links pointing through redirects', fix: 'Repoint internal links to the final URL (saves crawl + equity).',
276
- // Flag a link ONLY if the raw href it uses is itself a redirect source — i.e. that exact URL
277
- // appears as a `from` hop in some redirect chain. The old query flagged any link whose target
278
- // page merely HAD a `redirects` entry, which fired site-wide on www-canonical properties: every
279
- // page records the seed apex→www hop, yet the actual hrefs already use the final (www) URL and
280
- // never redirect. Matching the href (fragment-stripped) against the real redirect-source set is
281
- // robust to that artefact. Kept entirely in SQL (json_each over pages.redirects) so we never
282
- // materialise the whole links table in JS — the links table is the largest in the DB.
283
226
  run: (c) => {
284
227
  return rows(c, `
285
228
  WITH redir_src AS (
@@ -302,11 +245,8 @@ export const CHECKS = [
302
245
  {
303
246
  id: 'missing-structured-data', category: 'schema', severity: 'low', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'per-page',
304
247
  title: 'No structured data', fix: 'Add relevant JSON-LD (Article, Product, Organization…).',
305
- // Only "no structured data" if there's no JSON-LD AND no Microdata/RDFa either — else a
306
- // page using valid Microdata (common on older themes) is falsely flagged.
307
248
  run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND (json_ld IS NULL OR json_ld='') AND COALESCE(has_microdata,0)=0 AND COALESCE(has_rdfa,0)=0`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
308
249
  },
309
- // ── Schema validation (validate captured json_ld vs maintained Rich-Results map) ──
310
250
  {
311
251
  id: 'invalid-schema', category: 'schema', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
312
252
  title: 'Invalid structured data', fix: 'Fix the JSON-LD so each block parses and carries @context (https://schema.org) + a valid @type.',
@@ -335,15 +275,6 @@ export const CHECKS = [
335
275
  {
336
276
  id: 'keyword-cannibalisation', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
337
277
  title: 'Keyword cannibalisation', fix: 'Consolidate or differentiate — multiple URLs compete for one query.',
338
- // A URL only "competes" if it ranks for the query (impression-weighted pos < 20) AND holds a
339
- // non-trivial share of the leader's impressions — ≥10% of the leader OR ≥500 impressions in its
340
- // own right, and ≥10 impressions minimum. Without that floor, incidental long-tail appearances
341
- // (a page picking up 1–7 impressions for the query) counted as competitors: e.g. "best vr headset"
342
- // reported 10 URLs when ONE page held 59,570 impressions at pos 1.1 and the other nine had 1–7
343
- // each — not cannibalisation, Google had decided. The absolute ≥500 backstop keeps a genuine
344
- // mid-volume rival under a dominant leader (e.g. 60k leader + a real 4k second page = 6.7%, below
345
- // the 10% bar) from being silently dropped. We also exclude "dominance" where the best pages both
346
- // sit at pos 1–2 (indented/double results are good). Branded queries are dropped via brandExcl.
347
278
  run: (c) => {
348
279
  if (!c.gscMaxDate)
349
280
  return [];
@@ -368,9 +299,6 @@ export const CHECKS = [
368
299
  const keys = String(r.pk || '').split('\x1f'), imps = String(r.im || '').split('\x1f'), poss = String(r.ps || '').split('\x1f');
369
300
  const items = keys.map((u, i) => ({ url: u, impressions: Number(imps[i]) || 0, position: Number(poss[i]) || 0, title: titleOf.get(u)?.title ?? null }))
370
301
  .sort((a, b) => b.impressions - a.impressions);
371
- // Differentiation signal: do the top-2 competing pages share significant title terms beyond the
372
- // query itself? If not (and both are titled), they're likely intentionally distinct pages (e.g.
373
- // pairwise comparisons), not true duplicates competing for one intent — annotate, don't suppress.
374
302
  const q = new Set(terms(r.query));
375
303
  const sig = items.slice(0, 2).map(it => terms(it.title ?? '').filter((w) => !q.has(w)));
376
304
  const differentiated = sig.length === 2 && items[0].title != null && items[1].title != null
@@ -389,22 +317,16 @@ export const CHECKS = [
389
317
  {
390
318
  id: 'ctr-below-expected', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 3, fixType: 'per-page',
391
319
  title: 'CTR far below position-expected', fix: 'Rewrite title/meta — ranking well but under-clicked (snippet opportunity).',
392
- // High-confidence floor: ≥500 impressions/28d. A title/meta rewrite (HIGH, ~3h) is only
393
- // worth flagging where the snippet earns enough visibility for a CTR lift to pay back — a
394
- // 100-impression page at 1% vs 3% expected is a 2-clicks gap, not a HIGH issue.
395
320
  run: (c) => c.gscMaxDate ? rows(c, `SELECT page_key urlKey, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) position, SUM(clicks) clicks, SUM(impressions) impressions FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key HAVING SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) <= 10 AND SUM(impressions) >= 500`)
396
321
  .map(r => { const ctr = r.clicks / r.impressions; const exp = expectedCtr(r.position); return { urlKey: r.urlKey, ctr, exp, position: r.position, impressions: r.impressions }; })
397
322
  .filter(x => x.ctr < x.exp * 0.5)
398
323
  .map(x => {
399
324
  const ev = { position: Math.round(x.position * 10) / 10, ctr: Math.round(x.ctr * 1000) / 10 + '%', expectedCtr: Math.round(x.exp * 1000) / 10 + '%', impressions: x.impressions };
400
- // Extreme case: near-zero CTR at a strong position isn't a title problem — a SERP feature
401
- // (image/video/AI overview) or navigational intent is taking the clicks. Different fix.
402
325
  if (x.position <= 5 && x.ctr < x.exp * 0.15)
403
326
  ev.note = 'near-zero CTR for the position — likely a SERP feature or navigational intent taking the clicks; check the live SERP before rewriting the title/meta';
404
327
  return { urlKey: x.urlKey, evidence: ev };
405
328
  }) : [],
406
329
  },
407
- // ── Period-over-period (GSC history by date) — the trend questions SEOs live in ──
408
330
  {
409
331
  id: 'traffic-decay', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
410
332
  title: 'Page losing clicks (period-over-period)', fix: 'Refresh and expand the content, and check for lost rankings — this page’s Search Console clicks fell sharply against the previous 28 days.',
@@ -416,7 +338,6 @@ export const CHECKS = [
416
338
  FROM prev LEFT JOIN cur ON cur.page_key=prev.page_key
417
339
  WHERE prev.c >= 30 AND COALESCE(cur.c,0) < prev.c * 0.6
418
340
  ORDER BY (prev.c - COALESCE(cur.c,0)) DESC LIMIT 40`)
419
- // page_key IS a url_key — carry it in the joinable column, not just evidence.
420
341
  .map(r => ({ urlKey: r.url, evidence: { url: r.url, previousClicks: r.prevC, currentClicks: r.curC, clicksLost: r.prevC - r.curC, dropPercent: Math.round((1 - r.curC / r.prevC) * 100) + '%', clicks: r.prevC - r.curC, impressions: r.curI, position: Math.round(r.pos * 10) / 10 } })),
421
342
  },
422
343
  {
@@ -447,7 +368,6 @@ export const CHECKS = [
447
368
  title: 'Indexable page with no search traffic', fix: 'No impressions in 90 days despite being indexable — consolidate, improve, or noindex/prune to concentrate crawl budget and internal authority (confirm it isn’t seasonal or brand-new first).',
448
369
  run: (c) => (!c.gscMaxDate || spanDays(c) < 90) ? [] : rows(c, `SELECT url_key urlKey, ipr FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND COALESCE(click_depth, 999) >= 1 AND url_key NOT IN (SELECT DISTINCT page_key FROM search_analytics WHERE page_key IS NOT NULL AND date <= '${c.gscMaxDate}' AND date > date('${c.gscMaxDate}','-90 days') AND impressions > 0) ORDER BY ipr DESC LIMIT 100`).map(r => ({ urlKey: r.urlKey, evidence: { note: 'indexable but zero impressions in 90 days', ipr: Math.round(r.ipr) } })),
449
370
  },
450
- // ── On-page parity (crawl-only, deterministic) ──
451
371
  {
452
372
  id: 'duplicate-meta-description', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
453
373
  title: 'Duplicate meta description', fix: 'Give each indexable page a unique meta description.',
@@ -506,7 +426,6 @@ export const CHECKS = [
506
426
  .map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, h1: x.h1 } }));
507
427
  },
508
428
  },
509
- // ── Content cluster (AI-era / RAG layer): exploit the chunked body text captured at crawl ──
510
429
  {
511
430
  id: 'body-missing-top-query', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
512
431
  title: 'Top query missing from the page body', fix: 'The page ranks for this query yet its terms appear nowhere — not the title, H1, or body copy. Add a section that actually covers the topic; if it can’t, the page is too thin to hold the ranking and a stronger page should target it.',
@@ -527,8 +446,8 @@ export const CHECKS = [
527
446
  for (const ch of JSON.parse(x.bc))
528
447
  body += ` ${ch.heading || ''} ${ch.text || ''}`;
529
448
  }
530
- catch { /* skip */ }
531
- return !q.some((w) => titleHasTerm(body, w)); // none of the query's terms appear on the page
449
+ catch { }
450
+ return !q.some((w) => titleHasTerm(body, w));
532
451
  }).map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, note: 'query terms absent from title, H1 and body' } }));
533
452
  },
534
453
  },
@@ -565,7 +484,7 @@ export const CHECKS = [
565
484
  return r.filter(x => {
566
485
  const q = terms(x.query);
567
486
  if (q.length < 2)
568
- return false; // only multi-term queries can be "scattered"
487
+ return false;
569
488
  let chunks;
570
489
  try {
571
490
  chunks = JSON.parse(x.bc);
@@ -575,8 +494,8 @@ export const CHECKS = [
575
494
  }
576
495
  const whole = `${x.title || ''} ${chunks.map(ch => `${ch.heading || ''} ${ch.text || ''}`).join(' ')}`;
577
496
  if (!q.every((w) => titleHasTerm(whole, w)))
578
- return false; // page must contain all terms (else it's body-missing)
579
- return !chunks.some(ch => { const h = `${ch.heading || ''} ${ch.text || ''}`; return q.every((w) => titleHasTerm(h, w)); }); // but no single chunk does
497
+ return false;
498
+ return !chunks.some(ch => { const h = `${ch.heading || ''} ${ch.text || ''}`; return q.every((w) => titleHasTerm(h, w)); });
580
499
  }).map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, note: 'terms present but never together in one passage' } }));
581
500
  },
582
501
  },
@@ -605,10 +524,6 @@ export const CHECKS = [
605
524
  },
606
525
  },
607
526
  {
608
- // Hobo "Signal Coherence" / Goldmine; leak: anchor_mismatch. Google leans on internal anchors to
609
- // understand a page's topic — if the IN-CONTENT inbound anchors never mention the query the page
610
- // actually ranks for, that's an incoherent internal signal. Guard against boilerplate FPs by using
611
- // ONLY placement='body' anchors (nav/footer/aside excluded) and requiring ≥3 of them.
612
527
  id: 'anchor-text-incoherent', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
613
528
  title: 'Internal anchors don’t mention the page’s top query', fix: 'The in-content internal links pointing at this page never use its top-ranking query in their anchor text — and Google leans on internal anchors to understand what a page is about. Re-anchor the key internal links with descriptive, query-relevant text instead of generic “read more” / brand-only labels.',
614
529
  run: (c) => {
@@ -619,12 +534,6 @@ export const CHECKS = [
619
534
  FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} ${brandExcl(c)} GROUP BY page_key, query) s
620
535
  JOIN pages p ON p.url_key = s.page_key
621
536
  WHERE s.rn = 1 AND p.indexable = 1 AND s.impr >= 100`);
622
- // Pool only GENUINE editorial anchors. "Chrome" (nav/footer/breadcrumb/CTA) is detected
623
- // STRUCTURALLY, not via an English word-list: an anchor text reused across a large share of
624
- // the site's pages is templated boilerplate — e.g. the EHI homepage's 2,940 inbound "Home"
625
- // breadcrumb links — and that holds in any language ("Startseite", "Accueil"…). We also drop
626
- // self-links and anchors with no letters ("(0)", page numbers, arrows). Editorial in-content
627
- // anchors recur on only a handful of pages, so they survive.
628
537
  const anchors = new Map();
629
538
  for (const a of rows(c, `
630
539
  WITH body_anchors AS (
@@ -645,18 +554,13 @@ export const CHECKS = [
645
554
  return top.filter(x => {
646
555
  const a = anchors.get(x.urlKey);
647
556
  if (!a || a.n < 3)
648
- return false; // need enough genuine in-content inbound links to judge
557
+ return false;
649
558
  const q = terms(x.query);
650
559
  return q.length > 0 && !q.some((w) => titleHasTerm(a.pool, w));
651
560
  }).map(x => { const a = anchors.get(x.urlKey); return { urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, inboundInContentLinks: a.n } }; });
652
561
  },
653
562
  },
654
563
  {
655
- // The "RAG snippetability" test. A local cross-encoder (the kind AI search uses to re-rank) scored
656
- // every chunk against the page's top query; we persisted the single best-passage score. A low max
657
- // means no dense, extractable answer anywhere on the page — it will lose in AI/passage search even
658
- // if it keyword-matches. Model-derived (not deterministic truth) → N label, includeJudgement-gated.
659
- // Requires `score_passages` to have run (like CWV needs page_lighthouse).
660
564
  id: 'weak-passage-answer', category: 'merged', severity: 'high', labels: ['G', 'N'], certainty: 0.8, effortBase: 5, fixType: 'per-page',
661
565
  title: 'No passage strongly answers the ranking query (AI-search risk)', fix: 'A local neural reranker found no single passage on this page that confidently answers its top query — the page covers the topic loosely but offers no dense, extractable answer, so AI/passage search will prefer a clearer source. Add a focused, self-contained passage: a heading that states the question + a direct ~50-word answer up top. Run `score_passages` to (re)populate.',
662
566
  run: (c) => rows(c, `SELECT url_key urlKey, max_passage_score mps, max_passage_query q, max_passage_impr impr FROM pages
@@ -664,10 +568,6 @@ export const CHECKS = [
664
568
  .map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.q, maxPassageScore: x.mps, impressions: x.impr ?? 0, note: 'best passage scores below the reranker confidence threshold' } })),
665
569
  },
666
570
  {
667
- // Dejan: search weights the opening heavily and AI answers front-load. If the ranking query's
668
- // terms are present LATER in the page but absent from the opening (~first 2 chunks / ~200 words),
669
- // the answer is buried. (body-missing-top-query handles total absence; this is the buried case.)
670
- // Informational intent only — front-loading matters less for navigational/transactional queries.
671
571
  id: 'answer-not-front-loaded', category: 'merged', severity: 'med', labels: ['G', 'N'], certainty: 0.6, effortBase: 3, fixType: 'per-page',
672
572
  title: 'Answer to the ranking query is buried, not front-loaded', fix: 'The page covers its top query but the terms don’t appear up top (the intro / first section). Google weights the opening heavily and AI answers front-load — move a direct ~50–100-word answer to the first section. (Heuristic — informational queries.)',
673
573
  run: (c) => {
@@ -699,14 +599,10 @@ export const CHECKS = [
699
599
  },
700
600
  },
701
601
  {
702
- // Dejan "density beats length": AI grounds ~370 words/page, diminishing past ~1,500. A very long
703
- // page with an over-long unbroken section grounds poorly — split it into focused, headed passages.
704
602
  id: 'content-bloat', category: 'content', severity: 'low', labels: ['D', 'N'], certainty: 0.6, effortBase: 5, fixType: 'per-page',
705
603
  title: 'Over-long section dilutes AI-grounding (density beats length)', fix: 'This page has a very long unbroken section. AI search grounds only ~370 words per page with sharp diminishing returns past ~1,500 — break the long section into focused, headed passages (or tighten it) so each answers one thing cleanly.',
706
604
  run: (c) => {
707
605
  const out = [];
708
- // Use the real (uncapped) word_count ÷ number of headed sections — chunk TEXT is capped at
709
- // extraction, so we infer over-long sections from words-per-heading, not from chunk length.
710
606
  for (const x of iterRows(c, `SELECT url_key urlKey, word_count wc, body_chunks bc FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND word_count >= 2500 AND body_chunks IS NOT NULL`)) {
711
607
  let chunks;
712
608
  try {
@@ -724,15 +620,11 @@ export const CHECKS = [
724
620
  },
725
621
  },
726
622
  {
727
- // Hobo Level 3 freshness / lastSignificantUpdate. Gemini guard: YoY windows (negate seasonality
728
- // + zero-click-SERP CTR loss). Flag when the page hasn't been meaningfully re-dated in >12 months
729
- // AND clicks are down >25% YoY AND impressions down >15% YoY (impressions confirm ranking decay,
730
- // not just CTR). Needs ~13 months of GSC — guarded by spanDays so it stays silent on shallow syncs.
731
623
  id: 'stale-content', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
732
624
  title: 'Stale page declining year-on-year', fix: 'This page hasn’t been meaningfully updated in over a year and its Search Console clicks are down sharply versus the same period last year — refresh and expand the content (and honestly re-date it) to rebuild the freshness signal Google rewards.',
733
625
  run: (c) => {
734
626
  if (!c.gscMaxDate || spanDays(c) < 455)
735
- return []; // YoY window reaches back 455 days — anything less truncates the prior-year period
627
+ return [];
736
628
  const d = c.gscMaxDate;
737
629
  const out = [];
738
630
  const rs = rows(c, `
@@ -748,7 +640,7 @@ export const CHECKS = [
748
640
  continue;
749
641
  const ageDays = (Date.parse(d) - Date.parse(dm)) / 86400000;
750
642
  if (!(ageDays >= 365))
751
- continue; // only genuinely stale pages (>12 months since last schema date)
643
+ continue;
752
644
  out.push({ urlKey: x.url, evidence: { url: x.url, dateModified: dm.slice(0, 10), clicksYoY: `${x.pyC}→${x.curC} (-${Math.round((1 - x.curC / x.pyC) * 100)}%)`, impressionsYoY: `${x.pyI}→${x.curI}`, clicks: x.pyC - x.curC, impressions: x.pyI } });
753
645
  }
754
646
  return out;
@@ -790,23 +682,17 @@ export const CHECKS = [
790
682
  without.push(p.urlKey);
791
683
  }
792
684
  if (pages.length === 0 || withBc / pages.length < 0.4)
793
- return []; // site doesn't use breadcrumbs → design choice, not a bug
685
+ return [];
794
686
  return without.map(u => ({ urlKey: u, evidence: { note: 'site uses BreadcrumbList elsewhere; missing here' } }));
795
687
  },
796
688
  },
797
- // ── Merged crawl × Search Console — the "expert questions" that need both datasets ──
798
689
  {
799
690
  id: 'ghost-pages', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
800
691
  title: 'Ranking page the crawl can’t reach', fix: 'Google sends impressions/clicks to this URL but the site crawl never reached it — add internal links so it’s discoverable (or confirm it should exist and isn’t blocked).',
801
692
  run: (c) => {
802
693
  if (!c.gscMaxDate)
803
694
  return [];
804
- // Only meaningful on a COMPLETE crawl — if the crawl hit its maxPages cap, "absent
805
- // from crawl" is unreliable. Tie the guard to the crawl that produced the CURRENT
806
- // pages (not the latest crawl_metadata row, which may be a later failed crawl).
807
695
  const m = c.db.prepare('SELECT urls_crawled c, max_pages m FROM crawl_metadata WHERE crawl_id = (SELECT crawl_id FROM pages LIMIT 1)').get();
808
- // max_pages is nullable (NULL = no cap = complete crawl) — `c >= null` coerces to
809
- // `c >= 0`, which would silently disable the check forever on capless crawls.
810
696
  if (!m || m.c === 0 || (m.m != null && m.c >= m.m))
811
697
  return [];
812
698
  return rows(c, `SELECT page_key urlKey, SUM(clicks) clicks, SUM(impressions) impressions FROM search_analytics
@@ -831,7 +717,6 @@ export const CHECKS = [
831
717
  .map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, title: x.title } }));
832
718
  },
833
719
  },
834
- // ── Internal link graph (iPR + click-depth + anchor text, from the crawl `links` table) ──
835
720
  {
836
721
  id: 'deep-pages', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
837
722
  title: 'Page buried deep in the structure', fix: 'Add in-content (body) links from higher-level pages — this page is 4+ clicks from the homepage via body links.',
@@ -843,10 +728,6 @@ export const CHECKS = [
843
728
  run: (c) => c.gscMaxDate ? rows(c, `SELECT p.url_key urlKey, p.ipr ipr, p.inlink_count inl, SUM(sa.impressions) impressions FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key WHERE p.indexable=1 AND p.ipr < 30 AND p.inlink_count BETWEEN 1 AND 3 AND sa.${win(c.gscMaxDate)} GROUP BY p.url_key HAVING SUM(sa.impressions) >= 300 ORDER BY impressions DESC`).map(r => ({ urlKey: r.urlKey, evidence: { impressions: r.impressions, ipr: Math.round(r.ipr), inlinks: r.inl } })) : [],
844
729
  },
845
730
  {
846
- // The tier BETWEEN "orphan" (0 inlinks) and "fine": pages reached almost only via nav/footer.
847
- // inlink_count counts ALL internal links, so a page sitting in the global nav looks well-linked
848
- // even with zero EDITORIAL links — yet Google leans on in-content links for topic + equity. We
849
- // count distinct in-content (placement='body') inbound sources; ≤2 + real demand = under-linked.
850
731
  id: 'underlinked-editorial', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
851
732
  title: 'High demand, almost no in-content internal links', fix: 'This page earns real impressions but is reached mainly via nav/footer — add descriptive in-content links to it from related articles. Editorial body links pass more topical context and equity than templated nav links.',
852
733
  run: (c) => c.gscMaxDate ? rows(c, `
@@ -858,18 +739,6 @@ export const CHECKS = [
858
739
  HAVING SUM(sa.impressions) >= 300 AND bodyLinks <= 2
859
740
  ORDER BY impressions DESC LIMIT 40`).map(r => ({ urlKey: r.urlKey, evidence: { impressions: r.impressions, inContentLinks: r.bodyLinks, note: 'reached mainly via nav/footer — thin on editorial (in-content) links' } })) : [],
860
741
  },
861
- // NOTE: internal anchor-text checks (over-optimisation + generic/empty anchors) prototyped
862
- // and PULLED twice. Re-evaluated 2026-06-22 against the AgricIDaniel/claude-seo and
863
- // Bhanunamikaze/Agentic-SEO-Skill repos, this time using the links.placement='body' filter
864
- // plus excluding anchors that match the target's own title/H1. On real data (ehi.com.au) the
865
- // dominant survivors are still false positives: sitewide template CTAs ("home" 864/865,
866
- // "contact us", "apply today") and category links whose anchor IS the page title. The
867
- // page-level "mostly generic-anchored" variant returned 0 signal; raw empty anchors are
868
- // image/thumbnail-link noise (3,764/23,435 body links). Reliable detection needs an
869
- // editorial-vs-template link classifier we don't store (placement='body' still includes
870
- // in-template CTAs and product grids). Keep pulled — a wrong finding is worse than none.
871
- // ── Backlinks (need page_backlinks populated via pull_backlinks; gate so we never assert
872
- // "no external links" without data) ──
873
742
  {
874
743
  id: 'backlinks-to-404', category: 'crawlability', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
875
744
  title: 'External backlinks pointing to a dead page', fix: '301-redirect this URL to the best live equivalent — external link equity is hitting a 4xx/5xx page and being wasted.',
@@ -882,17 +751,14 @@ export const CHECKS = [
882
751
  run: (c) => {
883
752
  const has = c.db.prepare('SELECT COUNT(*) n FROM page_backlinks').get().n;
884
753
  if (!has)
885
- return []; // backlinks not pulled — can't credibly assert "no external links"
754
+ return [];
886
755
  return rows(c, `SELECT p.url_key urlKey FROM pages p LEFT JOIN page_backlinks b ON b.url_key=p.url_key WHERE p.status_code=200 AND p.indexable=1 AND p.inlink_count=0 AND COALESCE(b.backlinks,0)=0`)
887
756
  .map(r => ({ urlKey: r.urlKey, evidence: { inlinks: 0, backlinks: 0 } }));
888
757
  },
889
758
  },
890
- // ── Phase 6a — expert questions where crawl (intent) and reality diverge ──
891
759
  {
892
760
  id: 'ipr-bleed-by-status', category: 'crawlability', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
893
761
  title: 'Internal link equity flowing into dead URLs', fix: 'Repoint or 301 these internal links — they target non-200 URLs and waste the internal PageRank of the (often high-authority) pages linking to them.',
894
- // Sum the iPR of the SOURCE pages linking to each non-200 internal target. "Found a 404" is
895
- // junior; "this 404 drains the equity of N high-iPR pages" is the director-level find.
896
762
  run: (c) => rows(c, `SELECT l.target_key urlKey, COUNT(DISTINCT l.source_key) linkingPages,
897
763
  ROUND(SUM(src.ipr), 1) wastedIpr, t.status_code st
898
764
  FROM links l
@@ -905,9 +771,6 @@ export const CHECKS = [
905
771
  {
906
772
  id: 'broken-canonical-target', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
907
773
  title: 'Canonical points to a broken or unhealthy URL', fix: 'Point the canonical at a live, indexable, self-canonical HTTPS URL — Google ignores a canonical whose target is a 4xx/5xx/redirect, noindex, itself canonicalised elsewhere (a chain/loop), or an HTTPS→HTTP downgrade.',
908
- // Join the declared canonical_key back to the crawl. Only flag when we crawled the target.
909
- // Skip self-canonicals. Covers: non-200 target, noindex target, canonical chain/loop (target
910
- // canonicalises onward), and HTTPS→HTTP downgrade (research: Sitebulb indexability hints).
911
774
  run: (c) => rows(c, `SELECT p.url_key urlKey, p.canonical_url canon, p.canonical_key ck, t.status_code st, t.noindex ni, t.canonical_key tck
912
775
  FROM pages p JOIN pages t ON t.url_key = p.canonical_key
913
776
  WHERE p.canonical_key IS NOT NULL AND p.canonical_key != p.url_key AND p.status_code = 200
@@ -925,8 +788,6 @@ export const CHECKS = [
925
788
  {
926
789
  id: 'faceted-spider-trap', category: 'crawlability', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'global',
927
790
  title: 'Indexable faceted URLs burning crawl budget', fix: 'noindex (or robots-disallow / canonicalise) multi-parameter filter URLs — they are indexable but earn zero search traffic, so they only waste crawl budget and risk index bloat.',
928
- // Multi-parameter (>=2 params), indexable, zero GSC impressions in-window = classic facet trap.
929
- // Gate on GSC so "zero search value" is a real claim, not just "no data".
930
791
  run: (c) => {
931
792
  if (!c.gscMaxDate)
932
793
  return [];
@@ -939,8 +800,6 @@ export const CHECKS = [
939
800
  {
940
801
  id: 'soft-404-shell', category: 'indexation', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
941
802
  title: 'Soft 404 — 200 OK but Google treats it as not-found', fix: 'Either populate the page with real content, or return a true 404/410 (or noindex) — it serves 200 but Google has flagged it as a soft 404.',
942
- // URL Inspection page-fetch-state = soft 404 while the crawler sees a 200. The crawler alone
943
- // would call this page fine; Google disagrees. Needs URL Inspection populated.
944
803
  run: (c) => rows(c, `SELECT i.url_key urlKey, p.word_count wc, p.bytes bytes
945
804
  FROM url_inspection i JOIN pages p ON p.url_key = i.url_key
946
805
  WHERE LOWER(i.page_fetch_state) LIKE '%soft%' AND p.status_code = 200`).map(r => ({ urlKey: r.urlKey, evidence: { pageFetchState: 'soft 404', wordCount: r.wc, bytes: r.bytes } })),
@@ -948,9 +807,6 @@ export const CHECKS = [
948
807
  {
949
808
  id: 'rich-result-issues', category: 'schema', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
950
809
  title: 'Google-verified rich result issues (URL Inspection)', fix: 'Fix the structured-data issues Google itself reports for this page — these come from the URL Inspection API (Google’s own validation), not our local validator, so they are authoritative. Address the listed issue messages per rich-result type.',
951
- // Parses the stored richResultsResult JSON (url_inspection.rich_results) that inspect_urls
952
- // already captures: detectedItems[].items[].issues[] carries Google's severity + message.
953
- // Needs URL Inspection populated (run inspect_urls). Zero API cost — data is already in the DB.
954
810
  run: (c) => {
955
811
  const out = [];
956
812
  for (const r of iterRows(c, `SELECT url_key urlKey, rich_results rr FROM url_inspection WHERE rich_results IS NOT NULL AND rich_results != ''`)) {
@@ -961,8 +817,6 @@ export const CHECKS = [
961
817
  catch {
962
818
  continue;
963
819
  }
964
- // Dedup per (type, severity, message) with a count — a listicle repeats the same
965
- // "Missing field review" warning per product item; keep one sample item per group.
966
820
  const groups = new Map();
967
821
  for (const det of parsed?.detectedItems ?? []) {
968
822
  for (const item of det?.items ?? []) {
@@ -994,17 +848,14 @@ export const CHECKS = [
994
848
  title: 'hreflang missing return tag (not reciprocated)', fix: 'Add the reciprocal hreflang on the target page — Google ignores one-way hreflang annotations that don’t link back.',
995
849
  run: (c) => hreflangFindings(c).noReturn,
996
850
  },
997
- // ── 6a finishers — consume persisted DataForSEO enrichments (gated; need the data pulled) ──
998
851
  {
999
852
  id: 'intent-vs-pagetype-mismatch', category: 'merged', severity: 'med', labels: ['G', 'N'], certainty: 0.7, effortBase: 8, fixType: 'per-page',
1000
853
  title: 'Page type mismatches its query intent', fix: 'Reformat or retarget the page — its template doesn’t match the SERP intent for its top query (e.g. a product page ranking for an informational query, or an article for a transactional one). Run search_intent siteUrl:<property> to populate intents.',
1001
- // Join each page's top GSC query → persisted keyword_intent → page schema flavour (from json_ld).
1002
- // Judgement (N, 0.7): intent + schema-type inference is heuristic, so it only runs with includeJudgement.
1003
854
  run: (c) => {
1004
855
  if (!c.gscMaxDate)
1005
856
  return [];
1006
857
  if ((c.db.prepare('SELECT COUNT(*) n FROM keyword_intent').get().n) === 0)
1007
- return []; // intents not pulled
858
+ return [];
1008
859
  const r = rows(c, `SELECT s.page_key urlKey, s.query query, ki.intent intent, p.json_ld jsonLd FROM
1009
860
  (SELECT page_key, query, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
1010
861
  FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
@@ -1035,7 +886,7 @@ export const CHECKS = [
1035
886
  if (!c.gscMaxDate)
1036
887
  return [];
1037
888
  if ((c.db.prepare('SELECT COUNT(*) n FROM page_cwv').get().n) === 0)
1038
- return []; // CWV not pulled
889
+ return [];
1039
890
  return rows(c, `SELECT cw.url_key urlKey, cw.lcp_ms lcp, cw.cls cls, cw.performance perf,
1040
891
  (SELECT COALESCE(SUM(clicks),0) FROM search_analytics sa WHERE sa.page_key = cw.url_key AND ${win(c.gscMaxDate)}) clicks
1041
892
  FROM page_cwv cw
@@ -1044,14 +895,13 @@ export const CHECKS = [
1044
895
  .map(r => ({ urlKey: r.urlKey, evidence: { clicks: r.clicks, lcpMs: r.lcp != null ? Math.round(r.lcp) : null, cls: r.cls != null ? Math.round(r.cls * 1000) / 1000 : null, performance: r.perf != null ? Math.round(r.perf * 100) : null } }));
1045
896
  },
1046
897
  },
1047
- // ── 6b — per-template systemic issues (template-typed, deterministic) ──
1048
898
  {
1049
899
  id: 'pagination-canonical-to-page-1', category: 'crawlability', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
1050
900
  title: 'Paginated pages canonicalising away from themselves', fix: 'Make each paginated page (page 2, 3, …) self-canonical. Canonicalising page 2+ back to page 1 tells Google the deeper pages are duplicates, so products/articles linked only from page 2+ drop out of the crawl.',
1051
901
  run: (c) => rows(c, `SELECT url_key urlKey, url, canonical_url canon FROM pages
1052
902
  WHERE status_code=200 AND canonical_count>0 AND canonical_key IS NOT NULL AND canonical_key != url_key
1053
903
  AND (rel_prev=1 OR url LIKE '%/page/%' OR url GLOB '*[?&]page=[0-9]*' OR url GLOB '*[?&]paged=[0-9]*' OR url GLOB '*[?&]p=[0-9]*')`)
1054
- .filter(r => !/[?&](page|p)=1(\b|&|$)/.test(r.url) && !/\/page\/1(\/?$|\?)/.test(r.url)) // page 1 self-canonicalising to base is fine
904
+ .filter(r => !/[?&](page|p)=1(\b|&|$)/.test(r.url) && !/\/page\/1(\/?$|\?)/.test(r.url))
1055
905
  .map(r => ({ urlKey: r.urlKey, evidence: { url: r.url, canonical: r.canon } })),
1056
906
  },
1057
907
  {
@@ -1074,13 +924,12 @@ export const CHECKS = [
1074
924
  return out;
1075
925
  },
1076
926
  },
1077
- // ── 6d — Wikidata entity layer (heuristic H1→QID; N/judgement, gated on resolve_entities) ──
1078
927
  {
1079
928
  id: 'entity-internal-link-gap', category: 'crawlability', severity: 'low', labels: ['N'], certainty: 0.5, effortBase: 3, fixType: 'per-page',
1080
929
  title: 'Topically related pages not internally linked', fix: 'Add an internal link from the broader page to the more specific one — Wikidata says their entities are related (subclass-of / part-of) but no internal link connects them, leaving a gap in the topical mesh. Run resolve_entities first; verify the entity match before acting (heuristic).',
1081
930
  run: (c) => {
1082
931
  if ((c.db.prepare('SELECT COUNT(*) n FROM page_entity').get().n) === 0)
1083
- return []; // not resolved
932
+ return [];
1084
933
  return rows(c, `SELECT parent.url_key urlKey, child.url_key target, parent.label pl, child.label cl, ee.relation rel
1085
934
  FROM entity_edge ee
1086
935
  JOIN page_entity child ON child.qid = ee.qid
@@ -1090,7 +939,6 @@ export const CHECKS = [
1090
939
  .map(r => ({ urlKey: r.urlKey, evidence: { suggestLinkTo: r.target, parentEntity: r.pl, childEntity: r.cl, relation: r.rel } }));
1091
940
  },
1092
941
  },
1093
- // ── Sitemap ↔ crawl reconciliation (gated on a sitemap having been fetched) ──
1094
942
  {
1095
943
  id: 'sitemap-non-indexable', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
1096
944
  title: 'Sitemap lists non-indexable URLs', fix: 'Remove URLs from the XML sitemap that are 4xx/5xx, redirected, noindex, canonicalised or robots-blocked — the sitemap should list only canonical, indexable pages, or Google loses trust in it.',
@@ -1121,7 +969,6 @@ export const CHECKS = [
1121
969
  .map(r => ({ urlKey: r.urlKey, evidence: { inlinks: 0, note: 'in sitemap, no internal links' } }));
1122
970
  },
1123
971
  },
1124
- // ── Cheap performance proxies (from data captured at crawl time — no extra fetch) ──
1125
972
  {
1126
973
  id: 'slow-response', category: 'performance', severity: 'med', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'global',
1127
974
  title: 'Slow server response (TTFB proxy)', fix: 'Investigate slow server/TTFB — caching, CDN, or backend. Response time over ~1.5s hurts Core Web Vitals and crawl rate.',
@@ -1131,11 +978,6 @@ export const CHECKS = [
1131
978
  {
1132
979
  id: 'large-html', category: 'performance', severity: 'low', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'per-page',
1133
980
  title: 'Large HTML document (transferred)', fix: 'Trim the HTML payload — bloated markup slows render and First Contentful Paint (often huge inline SVG/CSS/JSON or unminified output). Judged on TRANSFERRED bytes, not raw: behind a compressing CDN (brotli/gzip) raw size matters far less.',
1134
- // Severity is on what the browser actually downloads, not raw bytes: a 200KB page served brotli
1135
- // is ~45KB over the wire and is NOT a real perf problem. We estimate transfer size (raw × ~0.22
1136
- // for br/gzip, else raw) and only flag pages whose ESTIMATED transferred HTML exceeds ~60KB.
1137
- // Prefilter at the 60KB threshold itself, not higher — an UNCOMPRESSED 60–150KB page
1138
- // (est = raw) is exactly the case that matters most and must reach the est filter.
1139
981
  run: (c) => rows(c, `SELECT url_key urlKey, bytes, content_encoding enc FROM pages WHERE status_code=200 AND ${HTML_CT} AND bytes > 60000 ORDER BY bytes DESC`)
1140
982
  .map(r => { const compressed = /br|gzip|deflate|zstd/i.test(r.enc || ''); const est = compressed ? Math.round(r.bytes * 0.22) : r.bytes; return { urlKey: r.urlKey, est, compressed, raw: r.bytes }; })
1141
983
  .filter(x => x.est > 60000)
@@ -1147,7 +989,6 @@ export const CHECKS = [
1147
989
  run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (content_encoding IS NULL OR content_encoding='')`)
1148
990
  .map(r => ({ urlKey: r.urlKey, evidence: {} })),
1149
991
  },
1150
- // ── Checklist-coverage additions (2026-07-20 — see plan/checklist-coverage.md) ──
1151
992
  {
1152
993
  id: 'robots-blocked-with-traffic', category: 'crawlability', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 1, fixType: 'per-page',
1153
994
  title: 'Robots-blocked page still earning search traffic', fix: 'This URL is disallowed in robots.txt yet Google still shows it (usually as a bare "no information" result) and users still land on it. Either unblock it so it can be crawled and ranked properly, or — if it genuinely shouldn\'t be found — unblock it AND add noindex (a robots-blocked page can never see the noindex).',
@@ -1184,8 +1025,6 @@ export const CHECKS = [
1184
1025
  {
1185
1026
  id: 'favicon-missing', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
1186
1027
  title: 'No favicon declared', fix: 'Add a favicon (<link rel="icon" …>) — Google shows it next to your result on mobile, and a missing one costs a little trust/recognition on every SERP appearance. One line in the template.',
1187
- // Gated on the column being populated (has_favicon is NULL on crawls from before this
1188
- // was captured) — never flag from a pre-feature crawl.
1189
1028
  run: (c) => {
1190
1029
  const hp = c.db.prepare(`SELECT url_key, has_favicon hf FROM pages WHERE status_code=200 AND has_favicon IS NOT NULL ORDER BY (click_depth=0) DESC, inlink_count DESC LIMIT 1`).get();
1191
1030
  return hp && hp.hf === 0 ? [{ urlKey: hp.url_key, evidence: { note: 'no <link rel="icon"> on the homepage' } }] : [];
@@ -1197,10 +1036,9 @@ export const CHECKS = [
1197
1036
  run: (c) => {
1198
1037
  const all = c.db.prepare(`SELECT url_key, lastmod FROM sitemap_urls WHERE lastmod IS NOT NULL AND lastmod <> ''`).all();
1199
1038
  if (all.length < 20)
1200
- return []; // too few dated URLs to judge the pattern
1039
+ return [];
1201
1040
  const ev = { urlsWithLastmod: all.length };
1202
1041
  let tripped = false;
1203
- // Tell 1: a generator stamping every URL with "now" — >90% share one date, and that date is recent.
1204
1042
  const byDay = new Map();
1205
1043
  for (const r of all)
1206
1044
  byDay.set(r.lastmod.slice(0, 10), (byDay.get(r.lastmod.slice(0, 10)) ?? 0) + 1);
@@ -1210,19 +1048,13 @@ export const CHECKS = [
1210
1048
  tripped = true;
1211
1049
  ev.sharedStamp = `${Math.round(topN / all.length * 100)}% of URLs claim ${topDay} — a generation timestamp, not a change date`;
1212
1050
  }
1213
- // Tell 2: dates in the future.
1214
1051
  const future = all.filter(r => Date.parse(r.lastmod) > Date.now() + 86400000).length;
1215
1052
  if (future > 0) {
1216
1053
  tripped = true;
1217
1054
  ev.futureDates = future;
1218
1055
  }
1219
- // Tell 3: lastmod claims a change between our two most recent crawls, yet none of the
1220
- // tracked page fields (status, title, meta, H1, word count, schema types) changed.
1221
1056
  const crawls = latestTwoCrawls(c.db);
1222
1057
  if (crawls.length === 2) {
1223
- // Compare DATE prefixes on both sides — lastmod is stored verbatim and often a full
1224
- // timestamp; compared raw against a 10-char date, same-day stamps sort "after" the
1225
- // boundary and are silently excluded (exactly the stamp-everything-today pattern).
1226
1058
  const phantom = c.db.prepare(`SELECT s.url_key FROM sitemap_urls s
1227
1059
  JOIN page_snapshots n ON n.url_key = s.url_key AND n.crawl_id = ?
1228
1060
  JOIN page_snapshots o ON o.url_key = s.url_key AND o.crawl_id = ?
@@ -1242,8 +1074,6 @@ export const CHECKS = [
1242
1074
  {
1243
1075
  id: 'no-304-revalidation', category: 'performance', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
1244
1076
  title: 'Server ignores conditional requests (no 304)', fix: 'Pages advertise Last-Modified/ETag but the server re-serves a full 200 when asked "has this changed?" (If-Modified-Since / If-None-Match). A properly configured server answers 304 Not Modified — it saves bandwidth on every revalidating crawler and cache, and signals stability to Googlebot. Usually a server/CDN setting.',
1245
- // Populated by the crawler's post-crawl probe (pages.conditional_304); NULL-gated so
1246
- // pre-feature crawls never flag. Fires only when NO probed page honoured the request.
1247
1077
  run: (c) => {
1248
1078
  const r = c.db.prepare(`SELECT COUNT(*) probed, COALESCE(SUM(conditional_304),0) ok FROM pages WHERE conditional_304 IS NOT NULL`).get();
1249
1079
  if (r.probed < 5 || r.ok > 0)
@@ -1255,45 +1085,12 @@ export const CHECKS = [
1255
1085
  {
1256
1086
  id: 'analytics-missing', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
1257
1087
  title: 'No client-side analytics detected', fix: 'No analytics or tag-manager snippet was found on any crawled page (GA4, GTM, Plausible, Matomo, Fathom, Clarity…). If you measure server-side, ignore this; otherwise you\'re flying blind — install an analytics package before making SEO decisions.',
1258
- // Fires only when EVERY populated page lacks a snippet — one page with analytics = installed.
1259
1088
  run: (c) => {
1260
1089
  const r = c.db.prepare(`SELECT COUNT(*) total, COALESCE(SUM(has_analytics),0) withA FROM pages WHERE status_code=200 AND ${HTML_CT} AND has_analytics IS NOT NULL`).get();
1261
1090
  return r.total >= 3 && r.withA === 0 ? [{ urlKey: null, evidence: { pagesChecked: r.total, note: 'no known analytics snippet on any crawled page (server-side measurement is invisible to a crawl)' } }] : [];
1262
1091
  },
1263
1092
  },
1264
- // ── Google "AI features / succeeding in AI search" guide (developers.google.com/search/docs/
1265
- // fundamentals/ai-optimization-guide, read 2026-08-02) — principle → check mapping:
1266
- // • Unique/compelling/non-commodity, people-first content .... thin-content, content-bloat,
1267
- // ai-slop-signals (NEW below — mechanical/templated prose is the anti-signal of "unique take")
1268
- // • Unique point of view / first-hand experience .............. article-no-author (NEW below —
1269
- // unattributed articles are the deterministically checkable slice), stale-content
1270
- // • Clear organisation (paragraphs/sections/headings) ......... poor-chunkability, content-bloat,
1271
- // heading-hierarchy; answer coverage: body-missing-top-query, rag-answer-gap,
1272
- // answer-not-front-loaded, weak-passage-answer, low-extractability
1273
- // • Indexed + snippet-eligible / technical requirements ....... indexation family (noindex,
1274
- // canonical, robots-blocked-with-traffic), soft-404-shell, sitemap reconciliation
1275
- // • Crawlable public content .................................. crawlability family, broken links,
1276
- // redirect chains; freshness-honesty (sitemap-lastmod-untrustworthy, no-304-revalidation)
1277
- // • Semantic HTML / parseable ................................. heading-hierarchy, missing-h1,
1278
- // multiple-h1, missing-lang
1279
- // • Reduce duplicate content .................................. duplicate-title/meta, canonical
1280
- // family, faceted-spider-trap, keyword-cannibalisation
1281
- // • Images/video supporting text .............................. image-alt, images-missing-dimensions
1282
- // • Page experience / latency ................................. performance proxies, high-yield-cwv-fail
1283
- // • Structured data honesty (not required, but keep it valid) . schema-validate family,
1284
- // article-date-illogical
1285
- // • Don't chunk artificially / rewrite for AI ................. covered by NOT having such checks;
1286
- // our chunk checks reward structure for humans, not tiny AI fragments
1287
- // • "Don't create llms.txt" ................................... tension with agent-readiness probes,
1288
- // which score llms.txt for *agent* (not Google AI-search) consumption — left as-is, different audience
1289
- // • Merchant Center / Business Profile / GenAI report ......... out of scope (not crawl/GSC data)
1290
- {
1291
- // Anti-signal of the guide's "unique, non-commodity, people-first content": prose that reads
1292
- // machine-generated. Three heuristics over body_chunks — slop-lexicon density, sentence-length
1293
- // uniformity (low variance = mechanical), repeated chunk openers (page-internal boilerplate).
1294
- // Precision over recall: absolute floors on every signal, ≥2 signals required, AND the composite
1295
- // score must sit in the top decile of pages showing any signal. Judgement-gated — a human wrote
1296
- // "delve" long before LLMs did.
1093
+ {
1297
1094
  id: 'ai-slop-signals', category: 'content', severity: 'low', labels: ['N'], certainty: 0.5, effortBase: 5, fixType: 'per-page',
1298
1095
  title: 'Prose shows machine-generated (slop) signals', fix: 'This page’s copy trips several statistical tells of generic AI-generated text: stock filler phrases, unusually uniform sentence lengths, and/or sections that all open the same way. Google’s AI-search guidance rewards unique, people-first content with a first-hand point of view — rewrite the flagged sections with specifics only you can supply (real experience, real numbers, real opinions) and cut the filler. (Heuristic — verify by reading the page; competent human writing can trip these tells.)',
1299
1096
  run: (c) => {
@@ -1322,7 +1119,6 @@ export const CHECKS = [
1322
1119
  const bodyWords = body.split(/\s+/).filter(Boolean).length;
1323
1120
  if (bodyWords < 250)
1324
1121
  continue;
1325
- // (i) slop-lexicon density (hits per 1000 words, ≥3 distinct terms required)
1326
1122
  const hits = new Map();
1327
1123
  for (const p of PHRASES) {
1328
1124
  let n = 0, i = -1;
@@ -1334,7 +1130,6 @@ export const CHECKS = [
1334
1130
  const totalHits = [...hits.values()].reduce((a, b) => a + b, 0);
1335
1131
  const density = totalHits / bodyWords * 1000;
1336
1132
  const lexSignal = hits.size >= 3 && density >= 2.5;
1337
- // (ii) sentence-length uniformity — coefficient of variation over sentence word-counts
1338
1133
  const sentences = body.split(/[.!?]+\s/).map(s => s.split(/\s+/).filter(Boolean).length).filter(n => n >= 5 && n <= 60);
1339
1134
  let cv = null;
1340
1135
  if (sentences.length >= 12) {
@@ -1342,11 +1137,9 @@ export const CHECKS = [
1342
1137
  cv = Math.sqrt(sentences.reduce((a, b) => a + (b - mean) ** 2, 0) / sentences.length) / mean;
1343
1138
  }
1344
1139
  const uniformSignal = cv !== null && cv < 0.28;
1345
- // (iii) repeated openers across the page's own chunks (first 3 words, ≥3 chunks sharing one)
1346
1140
  const openers = new Map();
1347
1141
  if (texts.length >= 5)
1348
1142
  for (const t of texts) {
1349
- // Letters only — numeric/UI-chrome openers ("Show 10 20…") are widget text, not prose.
1350
1143
  const o = t.toLowerCase().replace(/[^a-z\s]/g, ' ').split(/\s+/).filter(w => w.length >= 2).slice(0, 3).join(' ');
1351
1144
  if (o.split(' ').length === 3)
1352
1145
  openers.set(o, (openers.get(o) ?? 0) + 1);
@@ -1361,10 +1154,6 @@ export const CHECKS = [
1361
1154
  }
1362
1155
  if (!metrics.length)
1363
1156
  return [];
1364
- // Outliers only: the LEXICON signal is mandatory (uniform sentences + repeated openers
1365
- // without a single slop phrase is template chrome, not slop — live-verified on simracing),
1366
- // plus ≥1 corroborating signal, AND composite score in the top decile of pages that showed
1367
- // any signal at all.
1368
1157
  const sorted = metrics.map(m => m.score).sort((a, b) => a - b);
1369
1158
  const p90 = sorted[Math.min(sorted.length - 1, Math.floor(sorted.length * 0.9))];
1370
1159
  return metrics
@@ -1381,11 +1170,6 @@ export const CHECKS = [
1381
1170
  },
1382
1171
  },
1383
1172
  {
1384
- // The guide's "unique point of view based on personal experience or expertise", cut down to its
1385
- // deterministically checkable slice: an Article/BlogPosting that carries schema yet names no
1386
- // author. Attribution is the machine-readable experience/expertise signal; a byline-less article
1387
- // is the commodity-content default. Only fires where Article schema EXISTS (no schema at all is
1388
- // schema-opportunity territory, not this check).
1389
1173
  id: 'article-no-author', category: 'schema', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
1390
1174
  title: 'Article schema with no author attribution', fix: 'This page marks itself up as an Article/BlogPosting but declares no author. Google’s AI-search guidance rewards content with a demonstrable first-hand point of view — add `author` (a Person with a real name, ideally linking to an author page) to the Article schema and a visible byline to match.',
1391
1175
  run: (c) => {
@@ -1407,7 +1191,117 @@ export const CHECKS = [
1407
1191
  return out;
1408
1192
  },
1409
1193
  },
1194
+ {
1195
+ id: 'discover-max-image-preview', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global', yieldCoef: 0.15,
1196
+ title: 'Article missing max-image-preview:large (no large Discover card)',
1197
+ fix: 'Add `<meta name="robots" content="max-image-preview:large">` (or the same directive in an X-Robots-Tag header). Without it Google cannot feature the article with the large, high-CTR image card in Google Discover and image search — you get a thumbnail or nothing. This is usually one site-wide template/config change (WordPress core already emits it; a plugin or theme has overridden it here).',
1198
+ run: (c) => rows(c, `SELECT url_key urlKey, json_ld jsonLd, og_tags ogTags, robots, x_robots_tag xrt FROM pages WHERE ${DISCOVER_PREFILTER}`)
1199
+ .filter(r => isDiscoverEligible(r.jsonLd, r.ogTags) && !hasMaxImagePreviewLarge(r.robots, r.xrt) && !hasPreviewRestriction(r.robots, r.xrt))
1200
+ .map(r => ({ urlKey: r.urlKey, evidence: { robots: r.robots ?? null, xRobotsTag: r.xrt ?? null, note: 'article-type page without max-image-preview:large' } })),
1201
+ },
1202
+ {
1203
+ id: 'discover-missing-og', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page', yieldCoef: 0.12,
1204
+ title: 'Article missing og:title or og:image (Discover cards use these)',
1205
+ fix: 'Add both `og:title` and `og:image` — Google frequently builds the Discover headline and card image straight from Open Graph. Make og:image ≥1200px wide at 16:9, and write og:title for click-through, not just for social.',
1206
+ run: (c) => rows(c, `SELECT url_key urlKey, json_ld jsonLd, og_tags ogTags FROM pages WHERE ${DISCOVER_PREFILTER}`)
1207
+ .filter(r => isDiscoverEligible(r.jsonLd, r.ogTags))
1208
+ .map(r => ({ urlKey: r.urlKey, missing: ['og:title', 'og:image'].filter(k => { const v = parseOg(r.ogTags)[k]; return !(v && String(v).trim()); }) }))
1209
+ .filter(x => x.missing.length)
1210
+ .map(x => ({ urlKey: x.urlKey, evidence: { missing: x.missing } })),
1211
+ },
1212
+ {
1213
+ id: 'discover-slow-response', category: 'performance', severity: 'med', labels: ['D', 'N'], certainty: 0.5, effortBase: 5, fixType: 'global', yieldCoef: 0.1,
1214
+ title: 'Article response looks slow for Discover (>600ms wall-time)',
1215
+ fix: 'Aim for server response (TTFB) under 600ms — under 200ms is optimal — since Discover favours freshly-crawlable content. NOTE: the figure here is whole-fetch WALL TIME (it includes the HTML download and any retry/back-off), not a clean server TTFB, so confirm against a real TTFB or field measurement before prioritising. Caching, a CDN, or backend work are the usual levers.',
1216
+ run: (c) => rows(c, `SELECT url_key urlKey, json_ld jsonLd, og_tags ogTags, response_time_ms ms FROM pages WHERE ${DISCOVER_PREFILTER} AND response_time_ms > 600 AND (redirects IS NULL OR redirects='') ORDER BY response_time_ms DESC`)
1217
+ .filter(r => isDiscoverEligible(r.jsonLd, r.ogTags))
1218
+ .map(r => ({ urlKey: r.urlKey, evidence: { wallTimeMs: r.ms, target: '<600ms (ideal <200ms)', note: 'whole-fetch wall time incl. HTML download / retries — a proxy, not a measured TTFB' } })),
1219
+ },
1220
+ {
1221
+ id: 'discover-crawl-waste', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'global', yieldCoef: 0.1,
1222
+ title: 'High crawl waste (budget spent on redirects / non-indexable URLs)',
1223
+ fix: 'Google allocates a finite crawl budget per host; spending it on redirects, error pages and non-indexable URLs slows discovery of your real content (and Discover leans on fast discovery). Aim for almost everything Google crawls to be an indexable 200 — repoint internal links off redirects/404s, prune faceted/parameter URLs, and keep the sitemap to canonical indexable pages.',
1224
+ run: (c) => {
1225
+ const clause = `status_code IS NOT NULL AND status_code NOT IN (429,503)`;
1226
+ const tot = (rows(c, `SELECT COUNT(*) n FROM pages WHERE ${clause}`)[0]?.n) || 0;
1227
+ if (tot < 50)
1228
+ return [];
1229
+ const byReason = rows(c, `SELECT COALESCE(indexable_reason,'(other)') reason, COUNT(*) n FROM pages WHERE ${clause} AND indexable=0 AND COALESCE(indexable_reason,'') != 'non-html' GROUP BY 1 ORDER BY 2 DESC`);
1230
+ const nonIndexable = byReason.reduce((s, r) => s + r.n, 0);
1231
+ const redirected = (rows(c, `SELECT COUNT(*) n FROM pages WHERE ${clause} AND redirects IS NOT NULL AND redirects != ''`)[0]?.n) || 0;
1232
+ const waste = nonIndexable + redirected;
1233
+ const ratio = waste / tot;
1234
+ if (ratio < 0.3)
1235
+ return [];
1236
+ return [{ urlKey: null, evidence: { crawledUrls: tot, wasted: waste, wasteRatio: Math.round(ratio * 1000) / 10 + '%', redirected, nonIndexableByReason: Object.fromEntries(byReason.map(r => [r.reason, r.n])), note: 'crawled URLs that redirect or cannot be indexed (429/503 throttling and non-HTML assets excluded)' } }];
1237
+ },
1238
+ },
1239
+ {
1240
+ id: 'discover-generic-article-type', category: 'schema', severity: 'low', labels: ['D', 'N'], certainty: 0.5, effortBase: 1, fixType: 'global',
1241
+ title: 'Generic Article type where a more specific one fits',
1242
+ fix: 'The page’s Article markup uses only the generic `Article` type. Google recommends the most specific applicable type — `NewsArticle` for news, `LiveBlogPosting` for live coverage, `ProfilePage` for a person/creator profile — which unlocks richer Discover/News treatment. Usually one template/plugin setting. Only change it if the more specific type genuinely fits (judgement).',
1243
+ run: (c) => rows(c, `SELECT url_key urlKey, json_ld jsonLd FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND json_ld LIKE '%Article%'`)
1244
+ .map(r => { const present = new Set(); for (const n of discoverArticleNodes(r.jsonLd))
1245
+ for (const t of nodeTypeList(n))
1246
+ if (DISCOVER_TYPES.includes(t))
1247
+ present.add(t); return { urlKey: r.urlKey, present }; })
1248
+ .filter(x => x.present.has('article') && x.present.size === 1)
1249
+ .map(x => ({ urlKey: x.urlKey, evidence: { articleTypesDeclared: ['Article'], note: 'the Article schema declares only the generic Article type — no more specific subtype' } })),
1250
+ },
1251
+ {
1252
+ id: 'discover-image-schema', category: 'schema', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page', yieldCoef: 0.12,
1253
+ title: 'Article schema declares no image (Discover needs a large image)',
1254
+ fix: 'Add an `image` to the Article structured data — Discover cards need a big image and cannot build one from the page alone. Google wants a high-res image (≥1200px wide), ideally in three crops: 16:9 (1200×675), 4:3 (1200×900) and 1:1 (1200×1200). The main visible image at the top of the article should match the 16:9 schema image.',
1255
+ run: (c) => rows(c, `SELECT url_key urlKey, json_ld jsonLd, og_tags ogTags FROM pages WHERE ${DISCOVER_PREFILTER}`)
1256
+ .map(r => ({ urlKey: r.urlKey, nodes: discoverArticleNodes(r.jsonLd) }))
1257
+ .filter(x => x.nodes.length > 0 && !x.nodes.some(hasSchemaImage))
1258
+ .map(x => ({ urlKey: x.urlKey, evidence: { note: 'no `image` on the Article schema — Discover cannot build a large image card from schema alone' } })),
1259
+ },
1410
1260
  ];
1261
+ const DISCOVER_TYPES = ['article', 'newsarticle', 'blogposting', 'liveblogposting', 'profilepage', 'reportagenewsarticle', 'opinionnewsarticle', 'reviewnewsarticle', 'techarticle', 'scholarlyarticle'];
1262
+ const DISCOVER_PREFILTER = `status_code=200 AND indexable=1 AND ${HTML_CT} AND (json_ld LIKE '%Article%' OR json_ld LIKE '%BlogPosting%' OR json_ld LIKE '%ProfilePage%' OR og_tags LIKE '%article%')`;
1263
+ function nodeTypeList(n) {
1264
+ const t = n?.['@type'];
1265
+ if (typeof t === 'string')
1266
+ return [t.toLowerCase()];
1267
+ if (Array.isArray(t))
1268
+ return t.filter((x) => typeof x === 'string').map(x => x.toLowerCase());
1269
+ return [];
1270
+ }
1271
+ function discoverArticleNodes(jsonLd) {
1272
+ return parseJsonLdNodes(jsonLd).filter(n => nodeTypeList(n).some(t => DISCOVER_TYPES.includes(t)));
1273
+ }
1274
+ function parseOg(ogTags) {
1275
+ if (!ogTags)
1276
+ return {};
1277
+ try {
1278
+ const o = JSON.parse(ogTags);
1279
+ return o && typeof o === 'object' && !Array.isArray(o) ? o : {};
1280
+ }
1281
+ catch {
1282
+ return {};
1283
+ }
1284
+ }
1285
+ function isDiscoverEligible(jsonLd, ogTags) {
1286
+ if (discoverArticleNodes(jsonLd).length)
1287
+ return true;
1288
+ if (parseJsonLdNodes(jsonLd).some(n => nodeTypeList(n).length))
1289
+ return false;
1290
+ return (parseOg(ogTags)['og:type'] ?? '').toLowerCase() === 'article';
1291
+ }
1292
+ function hasMaxImagePreviewLarge(robots, xrt) {
1293
+ return `${robots ?? ''} ${xrt ?? ''}`.toLowerCase().replace(/\s+/g, '').includes('max-image-preview:large');
1294
+ }
1295
+ function hasPreviewRestriction(robots, xrt) {
1296
+ const s = `${robots ?? ''} ${xrt ?? ''}`.toLowerCase().replace(/\s+/g, '');
1297
+ return s.includes('max-image-preview:none') || s.includes('max-image-preview:standard') || s.includes('nosnippet');
1298
+ }
1299
+ function hasSchemaImage(node) {
1300
+ const img = node?.image;
1301
+ const one = (v) => typeof v === 'string' ? v.trim().length > 0
1302
+ : !!v && typeof v === 'object' && !Array.isArray(v) && !!(v.url || v.contentUrl || v['@id']);
1303
+ return Array.isArray(img) ? img.some(one) : one(img);
1304
+ }
1411
1305
  const sitemapHasRows = (c) => (c.db.prepare('SELECT COUNT(*) n FROM sitemap_urls').get().n) > 0;
1412
1306
  let _hreflangCache = null;
1413
1307
  function hreflangFindings(ctx) {
@@ -1419,7 +1313,6 @@ function hreflangFindings(ctx) {
1419
1313
  const status = new Map();
1420
1314
  for (const p of ctx.db.prepare(`SELECT url_key, status_code, noindex FROM pages`).all())
1421
1315
  status.set(p.url_key, { status: p.status_code, noindex: p.noindex });
1422
- // declared[sourceKey] = set of internal alternate targetKeys (excluding self)
1423
1316
  const declared = new Map();
1424
1317
  const parsed = [];
1425
1318
  for (const p of pageRows) {
@@ -1442,13 +1335,10 @@ function hreflangFindings(ctx) {
1442
1335
  continue;
1443
1336
  }
1444
1337
  if (key === p.url_key)
1445
- continue; // self-reference
1338
+ continue;
1446
1339
  targets.push({ key, lang: (lang || '').toLowerCase() });
1447
1340
  }
1448
1341
  parsed.push({ srcKey: p.url_key, targets });
1449
- // A return link via x-default is a VALID return tag (common when x-default is the
1450
- // homepage) — include all targets here; x-default is only excluded as a reciprocation
1451
- // *requirement* in the loop below, never as a way of satisfying one.
1452
1342
  declared.set(p.url_key, new Set(targets.map(t => t.key)));
1453
1343
  }
1454
1344
  const broken = [];
@@ -1460,8 +1350,6 @@ function hreflangFindings(ctx) {
1460
1350
  const st = status.get(t.key);
1461
1351
  if (st && (st.status !== 200 || st.noindex === 1))
1462
1352
  brokenTargets.push(t.key);
1463
- // reciprocation only for internal, live (200) targets we crawled, excluding x-default
1464
- // (a broken target is already reported by broken-hreflang-target — don't double-flag)
1465
1353
  if (t.lang !== 'x-default' && st && st.status === 200 && st.noindex !== 1 && !(declared.get(t.key)?.has(srcKey)))
1466
1354
  missingReturn.push(t.key);
1467
1355
  }