@houtini/seo-audit-console 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (166) hide show
  1. package/LICENSE +92 -0
  2. package/README.md +211 -0
  3. package/dist/audit/checks.d.ts +35 -0
  4. package/dist/audit/checks.d.ts.map +1 -0
  5. package/dist/audit/checks.js +1475 -0
  6. package/dist/audit/checks.js.map +1 -0
  7. package/dist/audit/drift.d.ts +40 -0
  8. package/dist/audit/drift.d.ts.map +1 -0
  9. package/dist/audit/drift.js +148 -0
  10. package/dist/audit/drift.js.map +1 -0
  11. package/dist/audit/engine.d.ts +33 -0
  12. package/dist/audit/engine.d.ts.map +1 -0
  13. package/dist/audit/engine.js +186 -0
  14. package/dist/audit/engine.js.map +1 -0
  15. package/dist/audit/opportunities.d.ts +23 -0
  16. package/dist/audit/opportunities.d.ts.map +1 -0
  17. package/dist/audit/opportunities.js +149 -0
  18. package/dist/audit/opportunities.js.map +1 -0
  19. package/dist/audit/report.d.ts +11 -0
  20. package/dist/audit/report.d.ts.map +1 -0
  21. package/dist/audit/report.js +59 -0
  22. package/dist/audit/report.js.map +1 -0
  23. package/dist/audit/schema-validate.d.ts +35 -0
  24. package/dist/audit/schema-validate.d.ts.map +1 -0
  25. package/dist/audit/schema-validate.js +293 -0
  26. package/dist/audit/schema-validate.js.map +1 -0
  27. package/dist/audit/templates.d.ts +43 -0
  28. package/dist/audit/templates.d.ts.map +1 -0
  29. package/dist/audit/templates.js +129 -0
  30. package/dist/audit/templates.js.map +1 -0
  31. package/dist/audit/topicGaps.d.ts +42 -0
  32. package/dist/audit/topicGaps.d.ts.map +1 -0
  33. package/dist/audit/topicGaps.js +181 -0
  34. package/dist/audit/topicGaps.js.map +1 -0
  35. package/dist/core/AuditDatabase.d.ts +33 -0
  36. package/dist/core/AuditDatabase.d.ts.map +1 -0
  37. package/dist/core/AuditDatabase.js +481 -0
  38. package/dist/core/AuditDatabase.js.map +1 -0
  39. package/dist/core/Backlinks.d.ts +33 -0
  40. package/dist/core/Backlinks.d.ts.map +1 -0
  41. package/dist/core/Backlinks.js +110 -0
  42. package/dist/core/Backlinks.js.map +1 -0
  43. package/dist/core/Crawler.d.ts +23 -0
  44. package/dist/core/Crawler.d.ts.map +1 -0
  45. package/dist/core/Crawler.js +588 -0
  46. package/dist/core/Crawler.js.map +1 -0
  47. package/dist/core/DataForSeoClient.d.ts +86 -0
  48. package/dist/core/DataForSeoClient.d.ts.map +1 -0
  49. package/dist/core/DataForSeoClient.js +232 -0
  50. package/dist/core/DataForSeoClient.js.map +1 -0
  51. package/dist/core/Entities.d.ts +23 -0
  52. package/dist/core/Entities.d.ts.map +1 -0
  53. package/dist/core/Entities.js +62 -0
  54. package/dist/core/Entities.js.map +1 -0
  55. package/dist/core/GscClient.d.ts +22 -0
  56. package/dist/core/GscClient.d.ts.map +1 -0
  57. package/dist/core/GscClient.js +93 -0
  58. package/dist/core/GscClient.js.map +1 -0
  59. package/dist/core/GscSync.d.ts +20 -0
  60. package/dist/core/GscSync.d.ts.map +1 -0
  61. package/dist/core/GscSync.js +133 -0
  62. package/dist/core/GscSync.js.map +1 -0
  63. package/dist/core/JobManager.d.ts +30 -0
  64. package/dist/core/JobManager.d.ts.map +1 -0
  65. package/dist/core/JobManager.js +68 -0
  66. package/dist/core/JobManager.js.map +1 -0
  67. package/dist/core/RankTracker.d.ts +25 -0
  68. package/dist/core/RankTracker.d.ts.map +1 -0
  69. package/dist/core/RankTracker.js +78 -0
  70. package/dist/core/RankTracker.js.map +1 -0
  71. package/dist/core/Refresh.d.ts +32 -0
  72. package/dist/core/Refresh.d.ts.map +1 -0
  73. package/dist/core/Refresh.js +73 -0
  74. package/dist/core/Refresh.js.map +1 -0
  75. package/dist/core/UrlInspector.d.ts +22 -0
  76. package/dist/core/UrlInspector.d.ts.map +1 -0
  77. package/dist/core/UrlInspector.js +92 -0
  78. package/dist/core/UrlInspector.js.map +1 -0
  79. package/dist/core/WikidataClient.d.ts +19 -0
  80. package/dist/core/WikidataClient.d.ts.map +1 -0
  81. package/dist/core/WikidataClient.js +59 -0
  82. package/dist/core/WikidataClient.js.map +1 -0
  83. package/dist/core/agentReadiness.d.ts +34 -0
  84. package/dist/core/agentReadiness.d.ts.map +1 -0
  85. package/dist/core/agentReadiness.js +119 -0
  86. package/dist/core/agentReadiness.js.map +1 -0
  87. package/dist/core/ctrModel.d.ts +2 -0
  88. package/dist/core/ctrModel.d.ts.map +1 -0
  89. package/dist/core/ctrModel.js +6 -0
  90. package/dist/core/ctrModel.js.map +1 -0
  91. package/dist/core/dashboardData.d.ts +312 -0
  92. package/dist/core/dashboardData.d.ts.map +1 -0
  93. package/dist/core/dashboardData.js +550 -0
  94. package/dist/core/dashboardData.js.map +1 -0
  95. package/dist/core/dataStorage.d.ts +42 -0
  96. package/dist/core/dataStorage.d.ts.map +1 -0
  97. package/dist/core/dataStorage.js +193 -0
  98. package/dist/core/dataStorage.js.map +1 -0
  99. package/dist/core/draftBrief.d.ts +27 -0
  100. package/dist/core/draftBrief.d.ts.map +1 -0
  101. package/dist/core/draftBrief.js +69 -0
  102. package/dist/core/draftBrief.js.map +1 -0
  103. package/dist/core/extract.d.ts +67 -0
  104. package/dist/core/extract.d.ts.map +1 -0
  105. package/dist/core/extract.js +262 -0
  106. package/dist/core/extract.js.map +1 -0
  107. package/dist/core/gscFreshness.d.ts +18 -0
  108. package/dist/core/gscFreshness.d.ts.map +1 -0
  109. package/dist/core/gscFreshness.js +32 -0
  110. package/dist/core/gscFreshness.js.map +1 -0
  111. package/dist/core/linkGraph.d.ts +19 -0
  112. package/dist/core/linkGraph.d.ts.map +1 -0
  113. package/dist/core/linkGraph.js +125 -0
  114. package/dist/core/linkGraph.js.map +1 -0
  115. package/dist/core/passageScore.d.ts +19 -0
  116. package/dist/core/passageScore.d.ts.map +1 -0
  117. package/dist/core/passageScore.js +59 -0
  118. package/dist/core/passageScore.js.map +1 -0
  119. package/dist/core/paths.d.ts +5 -0
  120. package/dist/core/paths.d.ts.map +1 -0
  121. package/dist/core/paths.js +15 -0
  122. package/dist/core/paths.js.map +1 -0
  123. package/dist/core/queryData.d.ts +35 -0
  124. package/dist/core/queryData.d.ts.map +1 -0
  125. package/dist/core/queryData.js +200 -0
  126. package/dist/core/queryData.js.map +1 -0
  127. package/dist/core/reranker.d.ts +8 -0
  128. package/dist/core/reranker.d.ts.map +1 -0
  129. package/dist/core/reranker.js +68 -0
  130. package/dist/core/reranker.js.map +1 -0
  131. package/dist/core/robots.d.ts +11 -0
  132. package/dist/core/robots.d.ts.map +1 -0
  133. package/dist/core/robots.js +76 -0
  134. package/dist/core/robots.js.map +1 -0
  135. package/dist/core/sitemap.d.ts +15 -0
  136. package/dist/core/sitemap.d.ts.map +1 -0
  137. package/dist/core/sitemap.js +100 -0
  138. package/dist/core/sitemap.js.map +1 -0
  139. package/dist/core/sql.d.ts +6 -0
  140. package/dist/core/sql.d.ts.map +1 -0
  141. package/dist/core/sql.js +6 -0
  142. package/dist/core/sql.js.map +1 -0
  143. package/dist/core/types.d.ts +22 -0
  144. package/dist/core/types.d.ts.map +1 -0
  145. package/dist/core/types.js +2 -0
  146. package/dist/core/types.js.map +1 -0
  147. package/dist/core/url-key.d.ts +39 -0
  148. package/dist/core/url-key.d.ts.map +1 -0
  149. package/dist/core/url-key.js +106 -0
  150. package/dist/core/url-key.js.map +1 -0
  151. package/dist/generators/index.d.ts +45 -0
  152. package/dist/generators/index.d.ts.map +1 -0
  153. package/dist/generators/index.js +184 -0
  154. package/dist/generators/index.js.map +1 -0
  155. package/dist/index.d.ts +3 -0
  156. package/dist/index.d.ts.map +1 -0
  157. package/dist/index.js +10 -0
  158. package/dist/index.js.map +1 -0
  159. package/dist/server.d.ts +7 -0
  160. package/dist/server.d.ts.map +1 -0
  161. package/dist/server.js +1387 -0
  162. package/dist/server.js.map +1 -0
  163. package/dist/src/ui/dashboard.html +347 -0
  164. package/dist/src/ui/sync-progress.html +104 -0
  165. package/package.json +101 -0
  166. package/server.json +57 -0
@@ -0,0 +1,1475 @@
1
+ import { validateJsonLdColumn } from './schema-validate.js';
2
+ import { urlKey } from '../core/url-key.js';
3
+ import { parseJsonLdNodes, nodeType } from './templates.js';
4
+ import { expectedCtr } from '../core/ctrModel.js';
5
+ import { latestTwoCrawls } from './drift.js';
6
+ import { HTML_CT } from '../core/sql.js';
7
+ const rows = (ctx, sql, ...args) => ctx.db.prepare(sql).all(...args);
8
+ // Streaming variant — yields one row at a time instead of materialising the whole result set. Use for
9
+ // checks that scan a large/fat column (e.g. body_chunks) and only need independent per-row work, so a
10
+ // big content site doesn't load every page's body text into one array.
11
+ const iterRows = (ctx, sql, ...args) => ctx.db.prepare(sql).iterate(...args);
12
+ // d is the finalised GSC date (partial trailing days already trimmed upstream); cap the window at
13
+ // it so checks never count unfinalised days. winPrev already caps below d, so it's unaffected.
14
+ const win = (d) => `date > date('${d}', '-28 days') AND date <= '${d}'`;
15
+ // Prior 28-day window (the 28 days BEFORE the current window) — for period-over-period checks.
16
+ const winPrev = (d) => `date <= date('${d}', '-28 days') AND date > date('${d}', '-56 days')`;
17
+ // SQL clause to drop branded queries (whole-token match). Branded multi-URL ranking is sitelinks,
18
+ // not cannibalisation; branded top-queries aren't anchor targets. The brand is re-sanitised to
19
+ // alnum HERE (not trusting the caller) so inlining it into SQL is unconditionally injection-safe.
20
+ const brandExcl = (c) => {
21
+ const b = (c.brand ?? '').replace(/[^a-z0-9]/g, '');
22
+ return b ? `AND (' ' || LOWER(query) || ' ') NOT LIKE '% ${b} %'` : '';
23
+ };
24
+ // Days of GSC history actually held — period-over-period checks need enough span to be meaningful.
25
+ const spanDays = (c) => {
26
+ // Span must be measured up to the FINALISED max date the windows actually key off
27
+ // (gscMaxDate is trimmed ~3 days below raw MAX(date)) — measuring the raw span lets the
28
+ // ≥56-day guard pass while the previous-28d window still reaches before MIN(date),
29
+ // undercounting the prior period (rising-pages FPs, traffic-decay FNs).
30
+ const r = c.db.prepare(`SELECT julianday(?) - julianday(MIN(date)) d FROM search_analytics`).get(c.gscMaxDate ?? null);
31
+ return r?.d ?? 0;
32
+ };
33
+ // Newest dateModified/datePublished anywhere in a page's JSON-LD (recurses @graph/arrays) — the
34
+ // effective "last meaningfully updated" date. json_ld is a JSON array of raw block strings.
35
+ const newestSchemaDate = (jl) => {
36
+ if (!jl)
37
+ return null;
38
+ let best = -Infinity, bestStr = null;
39
+ const scan = (o) => {
40
+ if (!o || typeof o !== 'object')
41
+ return;
42
+ for (const key of ['dateModified', 'datePublished']) {
43
+ const v = o[key];
44
+ if (typeof v === 'string') {
45
+ const t = Date.parse(v);
46
+ if (!isNaN(t) && t > best) {
47
+ best = t;
48
+ bestStr = v;
49
+ }
50
+ }
51
+ }
52
+ for (const k in o)
53
+ if (o[k] && typeof o[k] === 'object')
54
+ scan(o[k]);
55
+ };
56
+ try {
57
+ for (const block of JSON.parse(jl)) {
58
+ try {
59
+ scan(JSON.parse(block));
60
+ }
61
+ catch { /* skip block */ }
62
+ }
63
+ }
64
+ catch { /* skip */ }
65
+ return bestStr;
66
+ };
67
+ // Paginated archive URLs (/page/2, ?page=3, ?paged=2). They legitimately share titles/metas with
68
+ // page 1 and are intentionally absent from sitemaps, so they must NOT generate duplicate-title /
69
+ // duplicate-meta / missing-meta / not-in-sitemap false positives. Pass the column reference
70
+ // (e.g. 'url_key' or 'p.url_key') so it composes with table aliases.
71
+ const notPagination = (col = 'url_key') =>
72
+ // Anchor the param name to ?/& — a bare LIKE '%page=%' also matches per_page=/on_page=/
73
+ // homepage=, wrongly exempting those URLs from duplicate-title/meta/sitemap checks.
74
+ `${col} NOT GLOB '*/page/[0-9]*' AND ${col} NOT GLOB '*[?&]page=[0-9]*' AND ${col} NOT GLOB '*[?&]paged=[0-9]*'`;
75
+ // Significant query terms (drop stopwords; keep ≥2 chars so "vr"/"pc"/"ai" count).
76
+ const STOP = new Set(['the', 'a', 'an', 'and', 'or', 'of', 'for', 'to', 'in', 'on', 'with', 'your', 'you', 'is', 'are', 'best', 'how', 'what', 'vs', 'why', 'can']);
77
+ const terms = (s) => (s || '').toLowerCase().replace(/[^a-z0-9 ]+/g, ' ').split(/\s+/).filter(t => t.length >= 2 && !STOP.has(t));
78
+ // A query term is "present" in the title if a TITLE WORD matches it (exact, or a
79
+ // plural/stem prefix either way for ≥4-char tokens), or the space-collapsed title
80
+ // contains it (for multi-word brands like "sync mesh" ≈ "syncmesh", ≥5 chars).
81
+ // Word-level avoids substring false-matches (e.g. "art" inside "smart").
82
+ const titleHasTerm = (title, term) => {
83
+ const t = title.toLowerCase();
84
+ const words = t.split(/[^a-z0-9]+/).filter(Boolean);
85
+ if (words.some(w => w === term || (term.length >= 4 && w.startsWith(term)) || (w.length >= 4 && term.startsWith(w))))
86
+ return true;
87
+ return term.length >= 5 && t.replace(/[^a-z0-9]+/g, '').includes(term);
88
+ };
89
+ // Validate captured JSON-LD per page, keeping findings whose issue-kinds match `kinds`.
90
+ function schemaFindings(ctx, kinds) {
91
+ const set = new Set(kinds);
92
+ const out = [];
93
+ for (const r of rows(ctx, `SELECT url_key urlKey, json_ld jsonLd FROM pages WHERE status_code=200 AND json_ld IS NOT NULL AND json_ld != ''`)) {
94
+ const hits = validateJsonLdColumn(r.jsonLd).filter(i => set.has(i.kind));
95
+ if (hits.length)
96
+ out.push({ urlKey: r.urlKey, evidence: { issues: hits.map(h => ({ type: h.type, detail: h.detail, fields: h.fields })) } });
97
+ }
98
+ return out;
99
+ }
100
+ // Position→expected-CTR curve lives in core/ctrModel (shared with the dashboard). Re-export it so
101
+ // existing `import { expectedCtr } from './checks.js'` call sites keep working.
102
+ export { expectedCtr };
103
+ export const CHECKS = [
104
+ // ── On-page (crawl, deterministic) ──────────────────────────────────────
105
+ {
106
+ id: 'missing-title', category: 'onpage', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
107
+ title: 'Missing title tag', fix: 'Add a unique, descriptive <title> (~50–60 chars).',
108
+ run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (title IS NULL OR TRIM(title)='')`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
109
+ },
110
+ {
111
+ id: 'duplicate-title', category: 'onpage', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
112
+ title: 'Duplicate title tag', fix: 'Make each indexable page’s title unique.',
113
+ run: (c) => rows(c, `SELECT url_key urlKey, title FROM pages WHERE status_code=200 AND indexable=1 AND ${notPagination()} AND title IS NOT NULL AND TRIM(title)!='' AND LOWER(TRIM(title)) IN (SELECT LOWER(TRIM(title)) FROM pages WHERE status_code=200 AND indexable=1 AND ${notPagination()} AND title IS NOT NULL GROUP BY LOWER(TRIM(title)) HAVING COUNT(*)>1)`).map(r => ({ urlKey: r.urlKey, evidence: { title: r.title } })),
114
+ },
115
+ {
116
+ id: 'missing-meta-description', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
117
+ title: 'Missing meta description', fix: 'Add a unique meta description (~120–155 chars).',
118
+ run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND ${notPagination()} AND (meta_description IS NULL OR TRIM(meta_description)='')`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
119
+ },
120
+ {
121
+ id: 'missing-h1', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
122
+ title: 'Missing H1', fix: 'Add a single descriptive <h1>.',
123
+ run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND (h1_count IS NULL OR h1_count=0)`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
124
+ },
125
+ {
126
+ id: 'multiple-h1', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
127
+ title: 'Multiple H1s', fix: 'Use one H1 per page.',
128
+ run: (c) => rows(c, `SELECT url_key urlKey, h1_count FROM pages WHERE status_code=200 AND h1_count>1`).map(r => ({ urlKey: r.urlKey, evidence: { h1Count: r.h1_count } })),
129
+ },
130
+ {
131
+ id: 'thin-content', category: 'content', severity: 'med', labels: ['D', 'N'], certainty: 0.6, effortBase: 5, fixType: 'per-page',
132
+ title: 'Thin content', fix: 'Expand or consolidate — under ~200 words of body text.',
133
+ run: (c) => rows(c, `SELECT url_key urlKey, word_count FROM pages WHERE status_code=200 AND indexable=1 AND word_count < 200`).map(r => ({ urlKey: r.urlKey, evidence: { wordCount: r.word_count } })),
134
+ },
135
+ // ── Indexation / crawlability ───────────────────────────────────────────
136
+ {
137
+ // RETIRED: 'canonical-mismatch' (was HIGH). A 200 page whose canonical points to a *healthy*
138
+ // 200 indexable URL is intentional consolidation (slug variants, category merges) — normal SEO,
139
+ // not an issue, yet it fired HIGH on every such page (pure noise). Every actionable case is
140
+ // already covered by a higher-signal check: broken-canonical-target (unhealthy target),
141
+ // canonical-ignored (Google ranks the non-canonical page), canonical-conflict (GSC disagrees).
142
+ // So plain "canonical points elsewhere" has no high-confidence residual — removed.
143
+ id: 'broken-internal-links', category: 'crawlability', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
144
+ title: 'Internal links to 4xx/5xx', fix: 'Repoint internal links to a live, canonical URL.',
145
+ run: (c) => rows(c, `SELECT l.target_key urlKey, p.status_code status, COUNT(DISTINCT l.source_key) sources FROM links l JOIN pages p ON p.url_key=l.target_key WHERE l.is_internal=1 AND p.status_code >= 400 AND p.status_code NOT IN (429,503) GROUP BY l.target_key`).map(r => ({ urlKey: r.urlKey, evidence: { status: r.status, linkingPages: r.sources } })),
146
+ },
147
+ // ── Extractor-dependent (images + canonical shape) ──────────────────────
148
+ {
149
+ id: 'image-alt', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
150
+ title: 'Images missing alt text', fix: 'Add descriptive alt text to content images (alt="" only for decorative).',
151
+ run: (c) => rows(c, `SELECT url_key urlKey, images_without_alt missing, image_count total FROM pages WHERE status_code=200 AND indexable=1 AND images_without_alt > 0`).map(r => ({ urlKey: r.urlKey, evidence: { missing: r.missing, total: r.total } })),
152
+ },
153
+ {
154
+ id: 'canonical-relative', category: 'indexation', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
155
+ title: 'Canonical declared as a relative URL', fix: 'Use an absolute https URL in rel=canonical — relative canonicals are error-prone.',
156
+ run: (c) => rows(c, `SELECT url_key urlKey, canonical_url canonical FROM pages WHERE status_code=200 AND canonical_relative=1`).map(r => ({ urlKey: r.urlKey, evidence: { canonical: r.canonical } })),
157
+ },
158
+ {
159
+ id: 'multiple-canonical', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
160
+ title: 'Multiple canonical tags', fix: 'Keep exactly one rel=canonical — conflicting canonicals let Google pick (or ignore) one.',
161
+ run: (c) => rows(c, `SELECT url_key urlKey, canonical_count cnt FROM pages WHERE status_code=200 AND canonical_count > 1`).map(r => ({ urlKey: r.urlKey, evidence: { canonicalCount: r.cnt } })),
162
+ },
163
+ // ── Extractor additions (CLS, headings, mixed content, directives, social) ───
164
+ {
165
+ id: 'images-missing-dimensions', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
166
+ // Lint, NOT a measured Core Web Vital. Missing width/height attributes only cause CLS if the
167
+ // CSS doesn't already reserve space — modern themes using aspect-ratio / fixed boxes have ~0
168
+ // measured CLS despite missing attributes. So we report it as a hygiene lint and explicitly say
169
+ // it's not a confirmed CWV issue; escalate only against field CLS (high-yield-cwv-fail does that).
170
+ title: 'Images missing width/height attributes', fix: 'Add width & height (or rely on CSS aspect-ratio) so the browser reserves space. NOTE: this is a lint — if your CSS already reserves space (aspect-ratio / fixed box) measured CLS is likely ~0 and there is nothing to fix. Confirm with field CLS before prioritising.',
171
+ run: (c) => rows(c, `SELECT url_key urlKey, images_missing_dimensions n, image_count total FROM pages WHERE status_code=200 AND indexable=1 AND images_missing_dimensions > 0`).map(r => ({ urlKey: r.urlKey, evidence: { missingDimensions: r.n, total: r.total, note: 'lint only — no measured CLS impact unless field data shows layout shift' } })),
172
+ },
173
+ {
174
+ id: 'heading-hierarchy', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
175
+ title: 'Skipped heading levels', fix: 'Use headings in order (don’t jump e.g. h1→h3) — keeps the document outline accessible and parseable.',
176
+ run: (c) => rows(c, `SELECT url_key urlKey, heading_skips n FROM pages WHERE status_code=200 AND indexable=1 AND heading_skips > 0`).map(r => ({ urlKey: r.urlKey, evidence: { skippedLevels: r.n } })),
177
+ },
178
+ {
179
+ id: 'mixed-content', category: 'security', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
180
+ title: 'Mixed content (http on https)', fix: 'Serve every subresource over https — browsers block or warn on insecure resources.',
181
+ run: (c) => rows(c, `SELECT url_key urlKey, mixed_content_count n FROM pages WHERE status_code=200 AND url LIKE 'https://%' AND mixed_content_count > 0`).map(r => ({ urlKey: r.urlKey, evidence: { insecureResources: r.n } })),
182
+ },
183
+ {
184
+ id: 'meta-nofollow', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
185
+ title: 'Meta robots nofollow', fix: 'Remove nofollow from meta robots unless you intend to drop all link equity from this page.',
186
+ run: (c) => rows(c, `SELECT url_key urlKey, robots FROM pages WHERE status_code=200 AND indexable=1 AND robots LIKE '%nofollow%'`).map(r => ({ urlKey: r.urlKey, evidence: { robots: r.robots } })),
187
+ },
188
+ {
189
+ id: 'missing-social-tags', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
190
+ title: 'No social share tags', fix: 'Add Open Graph (og:title/og:image) and/or Twitter Card tags so shared links render a rich preview.',
191
+ run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND (og_tags IS NULL OR og_tags='') AND (twitter_tags IS NULL OR twitter_tags='')`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
192
+ },
193
+ // ── Security / war-stories (headers now captured) ───────────────────────
194
+ {
195
+ id: 'missing-hsts', category: 'security', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
196
+ title: 'Missing HSTS header', fix: 'Add Strict-Transport-Security with a sensible max-age.',
197
+ run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (security_headers IS NULL OR security_headers NOT LIKE '%hsts%') LIMIT 1`).map(r => ({ urlKey: r.urlKey, evidence: { note: 'representative page; HSTS is site-wide' } })),
198
+ },
199
+ // ── Merged GSC × crawl (the differentiator) ─────────────────────────────
200
+ {
201
+ id: 'noindex-with-traffic', category: 'indexation', severity: 'crit', labels: ['D', 'G'], certainty: 1, effortBase: 1, fixType: 'per-page',
202
+ title: 'Noindex page still getting clicks', fix: 'Remove noindex if the page should rank — it earns clicks.',
203
+ run: (c) => c.gscMaxDate ? rows(c, `SELECT p.url_key urlKey, SUM(sa.clicks) clicks FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key WHERE p.noindex=1 AND sa.${win(c.gscMaxDate)} GROUP BY p.url_key HAVING SUM(sa.clicks)>0`).map(r => ({ urlKey: r.urlKey, evidence: { clicks: r.clicks } })) : [],
204
+ },
205
+ {
206
+ id: 'orphan-with-impressions', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
207
+ title: 'Orphan page earning impressions', fix: 'Add internal links — Google ranks it but the site barely links to it. If it drives a large traffic share, protect it BEFORE any cleanup or migration (it is load-bearing).',
208
+ run: (c) => {
209
+ if (!c.gscMaxDate)
210
+ return [];
211
+ const total = (rows(c, `SELECT SUM(clicks) c FROM search_analytics WHERE ${win(c.gscMaxDate)}`)[0]?.c) || 1;
212
+ return rows(c, `SELECT p.url_key urlKey, SUM(sa.impressions) impressions, SUM(sa.clicks) clicks FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key WHERE p.inlink_count=0 AND p.indexable=1 AND sa.${win(c.gscMaxDate)} GROUP BY p.url_key HAVING SUM(sa.impressions)>0 ORDER BY clicks DESC, impressions DESC`)
213
+ .map(r => { const share = Math.round(r.clicks / total * 1000) / 10; return { urlKey: r.urlKey, evidence: { impressions: r.impressions, clicks: r.clicks, inlinks: 0, trafficShare: share + '%', ...(share >= 5 ? { note: `LOAD-BEARING orphan: drives ${share}% of site clicks with zero internal links — protect before any cleanup/migration` } : {}) } }; });
214
+ },
215
+ },
216
+ {
217
+ id: 'canonical-ignored', category: 'indexation', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
218
+ title: 'Google may be ignoring the declared canonical', fix: 'This page declares a canonical pointing elsewhere, yet Google still ranks IT (real impressions) — Google is overriding the canonical, usually because the target is weaker or internal links favour this URL. Decide which URL you actually want indexed, then align both the canonical and the internal links to it.',
219
+ run: (c) => c.gscMaxDate ? rows(c, `SELECT p.url_key urlKey, p.canonical_url canonical, SUM(sa.impressions) impressions, SUM(sa.clicks) clicks FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key WHERE p.canonical_key IS NOT NULL AND p.canonical_key != p.url_key AND sa.${win(c.gscMaxDate)} GROUP BY p.url_key HAVING SUM(sa.impressions) >= 50 ORDER BY impressions DESC LIMIT 40`).map(r => ({ urlKey: r.urlKey, evidence: { declaredCanonical: r.canonical, impressions: r.impressions, clicks: r.clicks, note: 'ranks despite pointing its canonical elsewhere' } })) : [],
220
+ },
221
+ {
222
+ id: 'indexed-junk-url', category: 'indexation', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'global',
223
+ title: 'Internal-search / faceted URL is indexed and ranking', fix: 'A URL whose signature is internal site-search, a faceted filter, or a tracking-param variant is earning Google impressions — it has been accidentally indexed. noindex or robots-block these and canonicalise filter URLs, so thin/duplicate pages stop bleeding index quality. (The query footprint proves it even when the DOM looks fine.)',
224
+ run: (c) => {
225
+ if (!c.gscMaxDate)
226
+ return [];
227
+ const junk = /[?&](q|s|search|keyword|orderby|sort_by|filter|variant|pf_|dppref|replytocom)=|\/search(-results)?\//i;
228
+ return rows(c, `SELECT page_key urlKey, SUM(impressions) impressions, SUM(clicks) clicks FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key HAVING SUM(impressions) >= 30 ORDER BY SUM(impressions) DESC`)
229
+ .filter(r => junk.test(r.urlKey)).slice(0, 40)
230
+ .map(r => ({ urlKey: r.urlKey, evidence: { impressions: r.impressions, clicks: r.clicks, note: 'URL signature = internal search / facet / param — likely accidental indexation' } }));
231
+ },
232
+ },
233
+ {
234
+ id: 'coverage-not-indexed', category: 'indexation', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
235
+ title: 'Crawled but not indexed', fix: 'Investigate quality/duplication — Google crawled it and chose not to index.',
236
+ run: (c) => rows(c, `SELECT url_key urlKey, coverage_state state FROM url_inspection WHERE coverage_state LIKE '%not indexed%'`).map(r => ({ urlKey: r.urlKey, evidence: { coverageState: r.state } })),
237
+ },
238
+ {
239
+ id: 'canonical-conflict', category: 'indexation', severity: 'high', labels: ['G'], certainty: 1, effortBase: 3, fixType: 'per-page',
240
+ title: 'Google chose a different canonical', fix: 'Align your declared canonical with the page Google actually indexes.',
241
+ run: (c) => rows(c, `SELECT url_key urlKey, google_canonical g, user_canonical u FROM url_inspection WHERE user_canonical IS NOT NULL AND google_canonical IS NOT NULL AND google_canonical != user_canonical`).map(r => ({ urlKey: r.urlKey, evidence: { googleCanonical: r.g, userCanonical: r.u } })),
242
+ },
243
+ {
244
+ id: 'striking-distance', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
245
+ title: 'Striking-distance query (page 2)', fix: 'Small on-page + internal-link push could reach page 1.',
246
+ run: (c) => c.gscMaxDate ? rows(c, `SELECT page_key urlKey, query, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) position, SUM(impressions) impressions FROM search_analytics WHERE query IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY query, page_key HAVING SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0)>10 AND SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0)<=20 AND SUM(impressions)>=20 ORDER BY impressions DESC LIMIT 50`).map(r => ({ urlKey: r.urlKey, evidence: { query: r.query, position: Math.round(r.position * 10) / 10, impressions: r.impressions } })) : [],
247
+ },
248
+ // ── Additions from industry checklist review (buildable on current data) ──
249
+ {
250
+ id: 'title-too-long', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
251
+ title: 'Title over ~60 chars', fix: 'Trim the title so the primary keyword sits within ~60 chars.',
252
+ run: (c) => rows(c, `SELECT url_key urlKey, title_length len FROM pages WHERE status_code=200 AND indexable=1 AND title_length > 60`).map(r => ({ urlKey: r.urlKey, evidence: { titleLength: r.len } })),
253
+ },
254
+ {
255
+ id: 'meta-description-length', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
256
+ title: 'Meta description over ~160 chars', fix: 'Tighten to ~150–160 chars so it isn’t truncated.',
257
+ run: (c) => rows(c, `SELECT url_key urlKey, meta_description_length len FROM pages WHERE status_code=200 AND indexable=1 AND meta_description_length > 160`).map(r => ({ urlKey: r.urlKey, evidence: { length: r.len } })),
258
+ },
259
+ {
260
+ id: 'non-https', category: 'security', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
261
+ title: 'Page served over HTTP', fix: 'Serve over HTTPS and 301 the HTTP version.',
262
+ run: (c) => rows(c, `SELECT url_key urlKey, url FROM pages WHERE status_code=200 AND url LIKE 'http://%'`).map(r => ({ urlKey: r.urlKey, evidence: { url: r.url } })),
263
+ },
264
+ {
265
+ id: 'redirect-chain', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
266
+ title: 'Redirect chain (2+ hops)', fix: 'Collapse to a single hop to the final URL.',
267
+ // Count hops in SQL (json_array_length, guarded by json_valid) and filter to >=2 there — so we
268
+ // never pull every redirect-bearing page into JS just to count + drop most of them.
269
+ run: (c) => rows(c, `SELECT url_key urlKey, json_array_length(redirects) hops FROM pages
270
+ WHERE redirects IS NOT NULL AND json_valid(redirects) AND json_array_length(redirects) >= 2`)
271
+ .map(r => ({ urlKey: r.urlKey, evidence: { hops: r.hops } })),
272
+ },
273
+ {
274
+ id: 'internal-links-to-redirects', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
275
+ title: 'Internal links pointing through redirects', fix: 'Repoint internal links to the final URL (saves crawl + equity).',
276
+ // Flag a link ONLY if the raw href it uses is itself a redirect source — i.e. that exact URL
277
+ // appears as a `from` hop in some redirect chain. The old query flagged any link whose target
278
+ // page merely HAD a `redirects` entry, which fired site-wide on www-canonical properties: every
279
+ // page records the seed apex→www hop, yet the actual hrefs already use the final (www) URL and
280
+ // never redirect. Matching the href (fragment-stripped) against the real redirect-source set is
281
+ // robust to that artefact. Kept entirely in SQL (json_each over pages.redirects) so we never
282
+ // materialise the whole links table in JS — the links table is the largest in the DB.
283
+ run: (c) => {
284
+ return rows(c, `
285
+ WITH redir_src AS (
286
+ -- Feed json_each a CASE that yields '[]' for any null/invalid value, so it never receives
287
+ -- malformed JSON regardless of how SQLite orders the scan vs the WHERE (a plain WHERE
288
+ -- json_valid() guard gets defeated by subquery flattening). Mirrors the old JS try/catch.
289
+ SELECT DISTINCT json_extract(j.value, '$.from') src
290
+ FROM pages, json_each(CASE WHEN json_valid(pages.redirects) THEN pages.redirects ELSE '[]' END) j
291
+ WHERE pages.redirects IS NOT NULL AND pages.redirects <> '[]' AND j.type = 'object'
292
+ )
293
+ SELECT l.target_key urlKey, COUNT(DISTINCT l.source_key) sources
294
+ FROM links l
295
+ JOIN redir_src r ON r.src = (CASE WHEN instr(l.target_url, '#') > 0
296
+ THEN substr(l.target_url, 1, instr(l.target_url, '#') - 1)
297
+ ELSE l.target_url END)
298
+ WHERE l.is_internal = 1
299
+ GROUP BY l.target_key`).map(r => ({ urlKey: r.urlKey, evidence: { linkingPages: r.sources } }));
300
+ },
301
+ },
302
+ {
303
+ id: 'missing-structured-data', category: 'schema', severity: 'low', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'per-page',
304
+ title: 'No structured data', fix: 'Add relevant JSON-LD (Article, Product, Organization…).',
305
+ // Only "no structured data" if there's no JSON-LD AND no Microdata/RDFa either — else a
306
+ // page using valid Microdata (common on older themes) is falsely flagged.
307
+ run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND (json_ld IS NULL OR json_ld='') AND COALESCE(has_microdata,0)=0 AND COALESCE(has_rdfa,0)=0`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
308
+ },
309
+ // ── Schema validation (validate captured json_ld vs maintained Rich-Results map) ──
310
+ {
311
+ id: 'invalid-schema', category: 'schema', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
312
+ title: 'Invalid structured data', fix: 'Fix the JSON-LD so each block parses and carries @context (https://schema.org) + a valid @type.',
313
+ run: (c) => schemaFindings(c, ['parse', 'context', 'type']),
314
+ },
315
+ {
316
+ id: 'missing-required-fields', category: 'schema', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
317
+ title: 'Structured data missing required fields', fix: 'Add the Google-required properties for the detected schema type (cited per finding).',
318
+ run: (c) => schemaFindings(c, ['required']),
319
+ },
320
+ {
321
+ id: 'schema-value-errors', category: 'schema', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
322
+ title: 'Structured-data value errors', fix: 'Use absolute, indexable image/URL values and ISO-8601 dates in JSON-LD.',
323
+ run: (c) => schemaFindings(c, ['value']),
324
+ },
325
+ {
326
+ id: 'forbidden-schema', category: 'schema', severity: 'high', labels: ['D', 'N'], certainty: 0.7, effortBase: 1, fixType: 'per-page',
327
+ title: 'Restricted schema type in use', fix: 'Remove FAQPage/HowTo markup unless the page qualifies for the narrow remaining eligibility — it risks no benefit or a manual action.',
328
+ run: (c) => schemaFindings(c, ['forbidden']),
329
+ },
330
+ {
331
+ id: 'missing-viewport', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
332
+ title: 'Missing viewport meta (mobile)', fix: 'Add <meta name="viewport" content="width=device-width, initial-scale=1">.',
333
+ run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (viewport IS NULL OR viewport='')`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
334
+ },
335
+ {
336
+ id: 'keyword-cannibalisation', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
337
+ title: 'Keyword cannibalisation', fix: 'Consolidate or differentiate — multiple URLs compete for one query.',
338
+ // A URL only "competes" if it ranks for the query (impression-weighted pos < 20) AND holds a
339
+ // non-trivial share of the leader's impressions — ≥10% of the leader OR ≥500 impressions in its
340
+ // own right, and ≥10 impressions minimum. Without that floor, incidental long-tail appearances
341
+ // (a page picking up 1–7 impressions for the query) counted as competitors: e.g. "best vr headset"
342
+ // reported 10 URLs when ONE page held 59,570 impressions at pos 1.1 and the other nine had 1–7
343
+ // each — not cannibalisation, Google had decided. The absolute ≥500 backstop keeps a genuine
344
+ // mid-volume rival under a dominant leader (e.g. 60k leader + a real 4k second page = 6.7%, below
345
+ // the 10% bar) from being silently dropped. We also exclude "dominance" where the best pages both
346
+ // sit at pos 1–2 (indented/double results are good). Branded queries are dropped via brandExcl.
347
+ run: (c) => {
348
+ if (!c.gscMaxDate)
349
+ return [];
350
+ const flagged = rows(c, `
351
+ WITH per_page AS (
352
+ SELECT query, page_key, SUM(clicks) clicks, SUM(impressions) impressions,
353
+ SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos
354
+ FROM search_analytics
355
+ WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} ${brandExcl(c)}
356
+ GROUP BY query, page_key
357
+ HAVING pos < 20 AND SUM(impressions) >= 10
358
+ ),
359
+ ranked AS (SELECT *, MAX(impressions) OVER (PARTITION BY query) topImpr FROM per_page),
360
+ competing AS (SELECT * FROM ranked WHERE impressions >= topImpr * 0.1 OR impressions >= 500)
361
+ SELECT query, COUNT(*) urls, SUM(clicks) clicks, SUM(impressions) impressions, MIN(pos) bestPos, MAX(pos) worstPos,
362
+ GROUP_CONCAT(page_key, char(31)) pk, GROUP_CONCAT(CAST(ROUND(impressions) AS INT), char(31)) im, GROUP_CONCAT(ROUND(pos,1), char(31)) ps
363
+ FROM competing GROUP BY query
364
+ HAVING COUNT(*) >= 2 AND SUM(impressions) >= 50 AND NOT (MAX(pos) <= 2)
365
+ ORDER BY impressions DESC LIMIT 40`);
366
+ const titleOf = c.db.prepare('SELECT title FROM pages WHERE url_key = ?');
367
+ return flagged.map(r => {
368
+ const keys = String(r.pk || '').split('\x1f'), imps = String(r.im || '').split('\x1f'), poss = String(r.ps || '').split('\x1f');
369
+ const items = keys.map((u, i) => ({ url: u, impressions: Number(imps[i]) || 0, position: Number(poss[i]) || 0, title: titleOf.get(u)?.title ?? null }))
370
+ .sort((a, b) => b.impressions - a.impressions);
371
+ // Differentiation signal: do the top-2 competing pages share significant title terms beyond the
372
+ // query itself? If not (and both are titled), they're likely intentionally distinct pages (e.g.
373
+ // pairwise comparisons), not true duplicates competing for one intent — annotate, don't suppress.
374
+ const q = new Set(terms(r.query));
375
+ const sig = items.slice(0, 2).map(it => terms(it.title ?? '').filter((w) => !q.has(w)));
376
+ const differentiated = sig.length === 2 && items[0].title != null && items[1].title != null
377
+ && sig[0].filter((w) => sig[1].includes(w)).length < 2;
378
+ const ev = {
379
+ query: r.query, competingUrls: r.urls, clicks: r.clicks, impressions: r.impressions,
380
+ positions: `${Math.round(r.bestPos * 10) / 10}–${Math.round(r.worstPos * 10) / 10}`,
381
+ urls: items.slice(0, 4).map(it => ({ url: it.url, impressions: it.impressions, position: it.position, title: it.title })),
382
+ };
383
+ if (differentiated)
384
+ ev.note = 'competing URLs have distinct titles — likely intentional differentiation (e.g. pairwise comparisons), not true cannibalisation; verify before consolidating';
385
+ return { urlKey: null, evidence: ev };
386
+ });
387
+ },
388
+ },
389
+ {
390
+ id: 'ctr-below-expected', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 3, fixType: 'per-page',
391
+ title: 'CTR far below position-expected', fix: 'Rewrite title/meta — ranking well but under-clicked (snippet opportunity).',
392
+ // High-confidence floor: ≥500 impressions/28d. A title/meta rewrite (HIGH, ~3h) is only
393
+ // worth flagging where the snippet earns enough visibility for a CTR lift to pay back — a
394
+ // 100-impression page at 1% vs 3% expected is a 2-clicks gap, not a HIGH issue.
395
+ run: (c) => c.gscMaxDate ? rows(c, `SELECT page_key urlKey, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) position, SUM(clicks) clicks, SUM(impressions) impressions FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key HAVING SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) <= 10 AND SUM(impressions) >= 500`)
396
+ .map(r => { const ctr = r.clicks / r.impressions; const exp = expectedCtr(r.position); return { urlKey: r.urlKey, ctr, exp, position: r.position, impressions: r.impressions }; })
397
+ .filter(x => x.ctr < x.exp * 0.5)
398
+ .map(x => {
399
+ const ev = { position: Math.round(x.position * 10) / 10, ctr: Math.round(x.ctr * 1000) / 10 + '%', expectedCtr: Math.round(x.exp * 1000) / 10 + '%', impressions: x.impressions };
400
+ // Extreme case: near-zero CTR at a strong position isn't a title problem — a SERP feature
401
+ // (image/video/AI overview) or navigational intent is taking the clicks. Different fix.
402
+ if (x.position <= 5 && x.ctr < x.exp * 0.15)
403
+ ev.note = 'near-zero CTR for the position — likely a SERP feature or navigational intent taking the clicks; check the live SERP before rewriting the title/meta';
404
+ return { urlKey: x.urlKey, evidence: ev };
405
+ }) : [],
406
+ },
407
+ // ── Period-over-period (GSC history by date) — the trend questions SEOs live in ──
408
+ {
409
+ id: 'traffic-decay', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
410
+ title: 'Page losing clicks (period-over-period)', fix: 'Refresh and expand the content, and check for lost rankings — this page’s Search Console clicks fell sharply against the previous 28 days.',
411
+ run: (c) => (!c.gscMaxDate || spanDays(c) < 56) ? [] : rows(c, `
412
+ WITH cur AS (SELECT page_key, SUM(clicks) c, SUM(impressions) i, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos
413
+ FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key),
414
+ prev AS (SELECT page_key, SUM(clicks) c FROM search_analytics WHERE page_key IS NOT NULL AND ${winPrev(c.gscMaxDate)} GROUP BY page_key)
415
+ SELECT prev.page_key url, prev.c prevC, COALESCE(cur.c,0) curC, COALESCE(cur.i,0) curI, COALESCE(cur.pos,10) pos
416
+ FROM prev LEFT JOIN cur ON cur.page_key=prev.page_key
417
+ WHERE prev.c >= 30 AND COALESCE(cur.c,0) < prev.c * 0.6
418
+ ORDER BY (prev.c - COALESCE(cur.c,0)) DESC LIMIT 40`)
419
+ .map(r => ({ urlKey: null, evidence: { url: r.url, previousClicks: r.prevC, currentClicks: r.curC, clicksLost: r.prevC - r.curC, dropPercent: Math.round((1 - r.curC / r.prevC) * 100) + '%', clicks: r.prevC - r.curC, impressions: r.curI, position: Math.round(r.pos * 10) / 10 } })),
420
+ },
421
+ {
422
+ id: 'lost-queries', category: 'merged', severity: 'med', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
423
+ title: 'Query dropped out of rankings', fix: 'This query drove clicks last period and drives none now — find the page that ranked, check for de-indexing or lost rankings, and win it back.',
424
+ run: (c) => (!c.gscMaxDate || spanDays(c) < 56) ? [] : rows(c, `
425
+ WITH cur AS (SELECT query, SUM(clicks) c FROM search_analytics WHERE query IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY query),
426
+ prev AS (SELECT query, SUM(clicks) c, SUM(impressions) i FROM search_analytics WHERE query IS NOT NULL AND ${winPrev(c.gscMaxDate)} GROUP BY query)
427
+ SELECT prev.query query, prev.c prevC, prev.i prevI FROM prev LEFT JOIN cur ON cur.query=prev.query
428
+ WHERE prev.c >= 10 AND COALESCE(cur.c,0)=0
429
+ ORDER BY prev.c DESC LIMIT 40`)
430
+ .map(r => ({ urlKey: null, evidence: { query: r.query, previousClicks: r.prevC, currentClicks: 0, clicks: r.prevC, impressions: r.prevI } })),
431
+ },
432
+ {
433
+ id: 'position-slipping', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 3, fixType: 'per-page',
434
+ title: 'Page-1 ranking slipping', fix: 'Average position for this query worsened by 3+ spots vs the previous 28 days while it was still on page one — investigate the ranking loss before the clicks follow it down.',
435
+ run: (c) => (!c.gscMaxDate || spanDays(c) < 56) ? [] : rows(c, `
436
+ WITH cur AS (SELECT query, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos, SUM(impressions) i, SUM(clicks) c
437
+ FROM search_analytics WHERE query IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY query HAVING SUM(impressions) >= 100),
438
+ prev AS (SELECT query, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos FROM search_analytics WHERE query IS NOT NULL AND ${winPrev(c.gscMaxDate)} GROUP BY query)
439
+ SELECT cur.query query, prev.pos prevPos, cur.pos curPos, cur.i i, cur.c c FROM cur JOIN prev ON prev.query=cur.query
440
+ WHERE prev.pos <= 10 AND cur.pos - prev.pos >= 3
441
+ ORDER BY (cur.pos - prev.pos) * cur.i DESC LIMIT 40`)
442
+ .map(r => ({ urlKey: null, evidence: { query: r.query, previousPosition: Math.round(r.prevPos * 10) / 10, currentPosition: Math.round(r.curPos * 10) / 10, slippedBy: Math.round((r.curPos - r.prevPos) * 10) / 10, impressions: r.i, clicks: r.c, position: Math.round(r.curPos * 10) / 10 } })),
443
+ },
444
+ {
445
+ id: 'index-bloat', category: 'indexation', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
446
+ title: 'Indexable page with no search traffic', fix: 'No impressions in 90 days despite being indexable — consolidate, improve, or noindex/prune to concentrate crawl budget and internal authority (confirm it isn’t seasonal or brand-new first).',
447
+ run: (c) => (!c.gscMaxDate || spanDays(c) < 90) ? [] : rows(c, `SELECT url_key urlKey, ipr FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND COALESCE(click_depth, 999) >= 1 AND url_key NOT IN (SELECT DISTINCT page_key FROM search_analytics WHERE page_key IS NOT NULL AND date <= '${c.gscMaxDate}' AND date > date('${c.gscMaxDate}','-90 days') AND impressions > 0) ORDER BY ipr DESC LIMIT 100`).map(r => ({ urlKey: r.urlKey, evidence: { note: 'indexable but zero impressions in 90 days', ipr: Math.round(r.ipr) } })),
448
+ },
449
+ // ── On-page parity (crawl-only, deterministic) ──
450
+ {
451
+ id: 'duplicate-meta-description', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
452
+ title: 'Duplicate meta description', fix: 'Give each indexable page a unique meta description.',
453
+ run: (c) => rows(c, `SELECT url_key urlKey, meta_description md FROM pages WHERE status_code=200 AND indexable=1 AND ${notPagination()} AND meta_description IS NOT NULL AND TRIM(meta_description)!='' AND LOWER(TRIM(meta_description)) IN (SELECT LOWER(TRIM(meta_description)) FROM pages WHERE status_code=200 AND indexable=1 AND ${notPagination()} AND meta_description IS NOT NULL AND TRIM(meta_description)!='' GROUP BY LOWER(TRIM(meta_description)) HAVING COUNT(*)>1)`).map(r => ({ urlKey: r.urlKey, evidence: { metaDescription: r.md } })),
454
+ },
455
+ {
456
+ id: 'title-h1-mismatch', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
457
+ title: 'Title and H1 share no significant words', fix: 'Align the <title> and <h1> — they currently have no significant words in common, which blurs the page’s topic signal.',
458
+ run: (c) => rows(c, `SELECT url_key urlKey, title, h1 FROM pages WHERE status_code=200 AND indexable=1 AND title IS NOT NULL AND TRIM(title)!='' AND h1 IS NOT NULL AND TRIM(h1)!=''`)
459
+ .filter(r => { const tt = terms(r.title), th = terms(r.h1); if (tt.length < 2 || th.length < 2)
460
+ return false; const set = new Set(th); return !tt.some(w => set.has(w)); })
461
+ .map(r => ({ urlKey: r.urlKey, evidence: { title: r.title, h1: r.h1 } })),
462
+ },
463
+ {
464
+ id: 'rising-pages', category: 'merged', severity: 'low', labels: ['G'], certainty: 1, effortBase: 2, fixType: 'per-page',
465
+ title: 'Page gaining clicks fast (double down)', fix: 'Clicks jumped vs the previous 28 days — reinforce it with internal links and related content while the momentum is there.',
466
+ run: (c) => (!c.gscMaxDate || spanDays(c) < 56) ? [] : rows(c, `
467
+ WITH cur AS (SELECT page_key, SUM(clicks) c, SUM(impressions) i, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key),
468
+ prev AS (SELECT page_key, SUM(clicks) c FROM search_analytics WHERE page_key IS NOT NULL AND ${winPrev(c.gscMaxDate)} GROUP BY page_key)
469
+ SELECT cur.page_key url, COALESCE(prev.c,0) prevC, cur.c curC, cur.i curI, COALESCE(cur.pos,10) pos
470
+ FROM cur LEFT JOIN prev ON prev.page_key=cur.page_key
471
+ WHERE cur.c >= 30 AND cur.c >= COALESCE(prev.c,0) * 1.5
472
+ ORDER BY (cur.c - COALESCE(prev.c,0)) DESC LIMIT 25`)
473
+ .map(r => ({ urlKey: null, evidence: { url: r.url, previousClicks: r.prevC, currentClicks: r.curC, clicksGained: r.curC - r.prevC, clicks: r.curC, impressions: r.curI, position: Math.round(r.pos * 10) / 10 } })),
474
+ },
475
+ {
476
+ id: 'traffic-to-dead-url', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
477
+ title: 'Search traffic to a dead (non-200) URL', fix: 'Google still sends clicks/impressions to this URL but the crawl returns a 4xx/5xx — recover the page or 301 it to the best live equivalent so the demand isn’t lost. (The "directive contradicts reality" join: you rank for a page that no longer works.)',
478
+ run: (c) => c.gscMaxDate ? rows(c, `SELECT p.url_key urlKey, p.status_code st, SUM(sa.clicks) clicks, SUM(sa.impressions) impressions FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key WHERE p.status_code >= 400 AND sa.${win(c.gscMaxDate)} GROUP BY p.url_key HAVING SUM(sa.impressions) >= 10 ORDER BY clicks DESC, impressions DESC LIMIT 40`)
479
+ .map(r => ({ urlKey: r.urlKey, evidence: { status: r.st, clicks: r.clicks, impressions: r.impressions, note: `HTTP ${r.st} but still earning search traffic` } })) : [],
480
+ },
481
+ {
482
+ id: 'impressions-rising-clicks-flat', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 3, fixType: 'per-page',
483
+ title: 'Impressions rising but clicks flat (CTR erosion)', fix: 'Google is showing this page MORE than it used to, yet you’re not winning more clicks — a stale title/meta, or a SERP feature (AI overview, snippet, pack) is taking them. Rewrite the snippet or target the feature.',
484
+ run: (c) => (!c.gscMaxDate || spanDays(c) < 56) ? [] : rows(c, `
485
+ WITH cur AS (SELECT page_key, SUM(clicks) c, SUM(impressions) i, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key),
486
+ prev AS (SELECT page_key, SUM(clicks) c, SUM(impressions) i FROM search_analytics WHERE page_key IS NOT NULL AND ${winPrev(c.gscMaxDate)} GROUP BY page_key)
487
+ SELECT cur.page_key url, prev.i prevImpr, cur.i curImpr, prev.c prevClicks, cur.c curClicks, cur.pos pos
488
+ FROM cur JOIN prev ON prev.page_key=cur.page_key
489
+ WHERE prev.i >= 200 AND cur.i >= prev.i * 1.3 AND cur.c <= prev.c
490
+ ORDER BY (cur.i - prev.i) DESC LIMIT 40`)
491
+ .map(r => ({ urlKey: null, evidence: { url: r.url, previousImpressions: r.prevImpr, currentImpressions: r.curImpr, impressionsChange: '+' + Math.round((r.curImpr / r.prevImpr - 1) * 100) + '%', previousClicks: r.prevClicks, currentClicks: r.curClicks, impressions: r.curImpr, clicks: r.curClicks, position: Math.round(r.pos * 10) / 10 } })),
492
+ },
493
+ {
494
+ id: 'h1-missing-top-query', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
495
+ title: 'Top query missing from the H1', fix: 'Work the page’s top-performing query into the <h1> — it ranks for this term but the main heading doesn’t mention it.',
496
+ run: (c) => {
497
+ if (!c.gscMaxDate)
498
+ return [];
499
+ const r = rows(c, `SELECT s.page_key urlKey, s.query query, s.impr impressions, p.h1 h1 FROM
500
+ (SELECT page_key, query, SUM(impressions) impr, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
501
+ FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
502
+ JOIN pages p ON p.url_key = s.page_key
503
+ WHERE s.rn = 1 AND p.indexable = 1 AND p.h1 IS NOT NULL AND p.h1 != '' AND s.impr >= 100`);
504
+ return r.filter(x => { const q = terms(x.query); return q.length > 0 && !q.some(w => titleHasTerm(x.h1, w)); })
505
+ .map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, h1: x.h1 } }));
506
+ },
507
+ },
508
+ // ── Content cluster (AI-era / RAG layer): exploit the chunked body text captured at crawl ──
509
+ {
510
+ id: 'body-missing-top-query', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
511
+ title: 'Top query missing from the page body', fix: 'The page ranks for this query yet its terms appear nowhere — not the title, H1, or body copy. Add a section that actually covers the topic; if it can’t, the page is too thin to hold the ranking and a stronger page should target it.',
512
+ run: (c) => {
513
+ if (!c.gscMaxDate)
514
+ return [];
515
+ const r = rows(c, `SELECT s.page_key urlKey, s.query query, s.impr impressions, p.title title, p.h1 h1, p.body_chunks bc FROM
516
+ (SELECT page_key, query, SUM(impressions) impr, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
517
+ FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
518
+ JOIN pages p ON p.url_key = s.page_key
519
+ WHERE s.rn = 1 AND p.indexable = 1 AND p.body_chunks IS NOT NULL AND s.impr >= 100`);
520
+ return r.filter(x => {
521
+ const q = terms(x.query);
522
+ if (!q.length)
523
+ return false;
524
+ let body = `${x.title || ''} ${x.h1 || ''}`;
525
+ try {
526
+ for (const ch of JSON.parse(x.bc))
527
+ body += ` ${ch.heading || ''} ${ch.text || ''}`;
528
+ }
529
+ catch { /* skip */ }
530
+ return !q.some((w) => titleHasTerm(body, w)); // none of the query's terms appear on the page
531
+ }).map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, note: 'query terms absent from title, H1 and body' } }));
532
+ },
533
+ },
534
+ {
535
+ id: 'poor-chunkability', category: 'content', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
536
+ title: 'Content has no heading structure (poor RAG boundaries)', fix: 'Break the copy into headed sections (h2/h3). AI search and featured snippets lift self-contained, headed passages — a wall of text with no subheadings gives them no clean chunk to quote.',
537
+ run: (c) => {
538
+ const out = [];
539
+ for (const x of iterRows(c, `SELECT url_key urlKey, word_count wc, body_chunks bc FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND word_count >= 400 AND body_chunks IS NOT NULL`)) {
540
+ let headed;
541
+ try {
542
+ headed = JSON.parse(x.bc).filter((k) => k.heading && k.level >= 2).length;
543
+ }
544
+ catch {
545
+ continue;
546
+ }
547
+ if (headed <= 1)
548
+ out.push({ urlKey: x.urlKey, evidence: { wordCount: x.wc, note: '400+ words with ≤1 subheading — one undifferentiated block' } });
549
+ }
550
+ return out;
551
+ },
552
+ },
553
+ {
554
+ id: 'rag-answer-gap', category: 'merged', severity: 'med', labels: ['G', 'N'], certainty: 0.6, effortBase: 5, fixType: 'per-page',
555
+ title: 'No single passage answers the ranking query (RAG gap)', fix: 'The page contains the query’s terms but scattered across sections — no one chunk (heading + paragraph) holds them together. AI answers lift a single self-contained passage, so add one that directly answers the query in ~50 words. (Heuristic — confirm the query’s intent first.)',
556
+ run: (c) => {
557
+ if (!c.gscMaxDate)
558
+ return [];
559
+ const r = rows(c, `SELECT s.page_key urlKey, s.query query, s.impr impressions, p.title title, p.body_chunks bc FROM
560
+ (SELECT page_key, query, SUM(impressions) impr, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
561
+ FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
562
+ JOIN pages p ON p.url_key = s.page_key
563
+ WHERE s.rn = 1 AND p.indexable = 1 AND p.body_chunks IS NOT NULL AND s.impr >= 200`);
564
+ return r.filter(x => {
565
+ const q = terms(x.query);
566
+ if (q.length < 2)
567
+ return false; // only multi-term queries can be "scattered"
568
+ let chunks;
569
+ try {
570
+ chunks = JSON.parse(x.bc);
571
+ }
572
+ catch {
573
+ return false;
574
+ }
575
+ const whole = `${x.title || ''} ${chunks.map(ch => `${ch.heading || ''} ${ch.text || ''}`).join(' ')}`;
576
+ if (!q.every((w) => titleHasTerm(whole, w)))
577
+ return false; // page must contain all terms (else it's body-missing)
578
+ return !chunks.some(ch => { const h = `${ch.heading || ''} ${ch.text || ''}`; return q.every((w) => titleHasTerm(h, w)); }); // but no single chunk does
579
+ }).map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, note: 'terms present but never together in one passage' } }));
580
+ },
581
+ },
582
+ {
583
+ id: 'low-extractability', category: 'content', severity: 'low', labels: ['D', 'N'], certainty: 0.5, effortBase: 3, fixType: 'per-page',
584
+ title: 'Passages depend on context (low answer-extractability)', fix: 'Half or more of this page’s sections open with “it / this / they / there…” or never name the subject — lifted out of the page by an LLM they read as meaningless. Open key sections by naming the entity. (Heuristic.)',
585
+ run: (c) => {
586
+ const out = [];
587
+ const PRON = /^(it|this|that|these|those|they|he|she|there|here|such|one)\b/i;
588
+ for (const x of iterRows(c, `SELECT url_key urlKey, body_chunks bc FROM pages WHERE status_code=200 AND indexable=1 AND word_count >= 400 AND body_chunks IS NOT NULL`)) {
589
+ let chunks;
590
+ try {
591
+ chunks = JSON.parse(x.bc);
592
+ }
593
+ catch {
594
+ continue;
595
+ }
596
+ const bodied = chunks.filter((k) => k.text && k.text.length > 60);
597
+ if (bodied.length < 3)
598
+ continue;
599
+ const dep = bodied.filter((k) => PRON.test(String(k.text).trim())).length;
600
+ if (dep / bodied.length >= 0.5)
601
+ out.push({ urlKey: x.urlKey, evidence: { dependentSections: dep, totalSections: bodied.length, note: `${Math.round(dep / bodied.length * 100)}% of sections open context-dependent` } });
602
+ }
603
+ return out;
604
+ },
605
+ },
606
+ {
607
+ // Hobo "Signal Coherence" / Goldmine; leak: anchor_mismatch. Google leans on internal anchors to
608
+ // understand a page's topic — if the IN-CONTENT inbound anchors never mention the query the page
609
+ // actually ranks for, that's an incoherent internal signal. Guard against boilerplate FPs by using
610
+ // ONLY placement='body' anchors (nav/footer/aside excluded) and requiring ≥3 of them.
611
+ id: 'anchor-text-incoherent', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
612
+ title: 'Internal anchors don’t mention the page’s top query', fix: 'The in-content internal links pointing at this page never use its top-ranking query in their anchor text — and Google leans on internal anchors to understand what a page is about. Re-anchor the key internal links with descriptive, query-relevant text instead of generic “read more” / brand-only labels.',
613
+ run: (c) => {
614
+ if (!c.gscMaxDate)
615
+ return [];
616
+ const top = rows(c, `SELECT s.page_key urlKey, s.query query, s.impr impressions FROM
617
+ (SELECT page_key, query, SUM(impressions) impr, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
618
+ FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} ${brandExcl(c)} GROUP BY page_key, query) s
619
+ JOIN pages p ON p.url_key = s.page_key
620
+ WHERE s.rn = 1 AND p.indexable = 1 AND s.impr >= 100`);
621
+ // Pool only GENUINE editorial anchors. "Chrome" (nav/footer/breadcrumb/CTA) is detected
622
+ // STRUCTURALLY, not via an English word-list: an anchor text reused across a large share of
623
+ // the site's pages is templated boilerplate — e.g. the EHI homepage's 2,940 inbound "Home"
624
+ // breadcrumb links — and that holds in any language ("Startseite", "Accueil"…). We also drop
625
+ // self-links and anchors with no letters ("(0)", page numbers, arrows). Editorial in-content
626
+ // anchors recur on only a handful of pages, so they survive.
627
+ const anchors = new Map();
628
+ for (const a of rows(c, `
629
+ WITH body_anchors AS (
630
+ SELECT target_key, anchor_text, source_key, LOWER(TRIM(anchor_text)) atext
631
+ FROM links
632
+ WHERE is_internal = 1 AND placement = 'body' AND source_key <> target_key
633
+ AND anchor_text IS NOT NULL AND TRIM(anchor_text) != '' AND LOWER(TRIM(anchor_text)) GLOB '*[a-z]*'
634
+ ),
635
+ templated AS (
636
+ SELECT atext FROM body_anchors GROUP BY atext
637
+ HAVING COUNT(DISTINCT source_key) > MAX(20, (SELECT COUNT(*) FROM pages WHERE status_code = 200 AND ${HTML_CT}) * 0.15)
638
+ )
639
+ SELECT target_key tk, GROUP_CONCAT(anchor_text, ' ') pool, COUNT(*) n
640
+ FROM body_anchors
641
+ WHERE atext NOT IN (SELECT atext FROM templated)
642
+ GROUP BY target_key`))
643
+ anchors.set(a.tk, { pool: a.pool, n: a.n });
644
+ return top.filter(x => {
645
+ const a = anchors.get(x.urlKey);
646
+ if (!a || a.n < 3)
647
+ return false; // need enough genuine in-content inbound links to judge
648
+ const q = terms(x.query);
649
+ return q.length > 0 && !q.some((w) => titleHasTerm(a.pool, w));
650
+ }).map(x => { const a = anchors.get(x.urlKey); return { urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, inboundInContentLinks: a.n } }; });
651
+ },
652
+ },
653
+ {
654
+ // The "RAG snippetability" test. A local cross-encoder (the kind AI search uses to re-rank) scored
655
+ // every chunk against the page's top query; we persisted the single best-passage score. A low max
656
+ // means no dense, extractable answer anywhere on the page — it will lose in AI/passage search even
657
+ // if it keyword-matches. Model-derived (not deterministic truth) → N label, includeJudgement-gated.
658
+ // Requires `score_passages` to have run (like CWV needs page_lighthouse).
659
+ id: 'weak-passage-answer', category: 'merged', severity: 'high', labels: ['G', 'N'], certainty: 0.8, effortBase: 5, fixType: 'per-page',
660
+ title: 'No passage strongly answers the ranking query (AI-search risk)', fix: 'A local neural reranker found no single passage on this page that confidently answers its top query — the page covers the topic loosely but offers no dense, extractable answer, so AI/passage search will prefer a clearer source. Add a focused, self-contained passage: a heading that states the question + a direct ~50-word answer up top. Run `score_passages` to (re)populate.',
661
+ run: (c) => rows(c, `SELECT url_key urlKey, max_passage_score mps, max_passage_query q, max_passage_impr impr FROM pages
662
+ WHERE indexable=1 AND max_passage_score IS NOT NULL AND max_passage_score < 3`)
663
+ .map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.q, maxPassageScore: x.mps, impressions: x.impr ?? 0, note: 'best passage scores below the reranker confidence threshold' } })),
664
+ },
665
+ {
666
+ // Dejan: search weights the opening heavily and AI answers front-load. If the ranking query's
667
+ // terms are present LATER in the page but absent from the opening (~first 2 chunks / ~200 words),
668
+ // the answer is buried. (body-missing-top-query handles total absence; this is the buried case.)
669
+ // Informational intent only — front-loading matters less for navigational/transactional queries.
670
+ id: 'answer-not-front-loaded', category: 'merged', severity: 'med', labels: ['G', 'N'], certainty: 0.6, effortBase: 3, fixType: 'per-page',
671
+ title: 'Answer to the ranking query is buried, not front-loaded', fix: 'The page covers its top query but the terms don’t appear up top (the intro / first section). Google weights the opening heavily and AI answers front-load — move a direct ~50–100-word answer to the first section. (Heuristic — informational queries.)',
672
+ run: (c) => {
673
+ if (!c.gscMaxDate)
674
+ return [];
675
+ const INFO = /\b(how|what|why|when|which|who|guide|tutorial|best|vs|versus|is|are|does|do|can|should|tips|ideas|examples|meaning|definition|setup|settings)\b/i;
676
+ const r = rows(c, `SELECT s.page_key urlKey, s.query query, s.impr impressions, p.title title, p.body_chunks bc FROM
677
+ (SELECT page_key, query, SUM(impressions) impr, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
678
+ FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
679
+ JOIN pages p ON p.url_key = s.page_key
680
+ WHERE s.rn = 1 AND p.indexable = 1 AND p.body_chunks IS NOT NULL AND p.word_count >= 800 AND s.impr >= 100`);
681
+ return r.filter(x => {
682
+ if (!INFO.test(x.query))
683
+ return false;
684
+ const q = terms(x.query);
685
+ if (!q.length)
686
+ return false;
687
+ let chunks;
688
+ try {
689
+ chunks = JSON.parse(x.bc);
690
+ }
691
+ catch {
692
+ return false;
693
+ }
694
+ const front = `${x.title || ''} ${chunks.slice(0, 2).map((k) => `${k.heading || ''} ${k.text || ''}`).join(' ')}`.slice(0, 1200);
695
+ const whole = `${x.title || ''} ${chunks.map((k) => `${k.heading || ''} ${k.text || ''}`).join(' ')}`;
696
+ return q.some((w) => titleHasTerm(whole, w)) && !q.some((w) => titleHasTerm(front, w));
697
+ }).map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, note: 'query terms appear later in the page but not in the opening' } }));
698
+ },
699
+ },
700
+ {
701
+ // Dejan "density beats length": AI grounds ~370 words/page, diminishing past ~1,500. A very long
702
+ // page with an over-long unbroken section grounds poorly — split it into focused, headed passages.
703
+ id: 'content-bloat', category: 'content', severity: 'low', labels: ['D', 'N'], certainty: 0.6, effortBase: 5, fixType: 'per-page',
704
+ title: 'Over-long section dilutes AI-grounding (density beats length)', fix: 'This page has a very long unbroken section. AI search grounds only ~370 words per page with sharp diminishing returns past ~1,500 — break the long section into focused, headed passages (or tighten it) so each answers one thing cleanly.',
705
+ run: (c) => {
706
+ const out = [];
707
+ // Use the real (uncapped) word_count ÷ number of headed sections — chunk TEXT is capped at
708
+ // extraction, so we infer over-long sections from words-per-heading, not from chunk length.
709
+ for (const x of iterRows(c, `SELECT url_key urlKey, word_count wc, body_chunks bc FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND word_count >= 2500 AND body_chunks IS NOT NULL`)) {
710
+ let chunks;
711
+ try {
712
+ chunks = JSON.parse(x.bc);
713
+ }
714
+ catch {
715
+ continue;
716
+ }
717
+ const headed = chunks.filter((k) => k.heading && k.level >= 2).length;
718
+ const wordsPerSection = Math.round(x.wc / Math.max(1, headed));
719
+ if (wordsPerSection > 500)
720
+ out.push({ urlKey: x.urlKey, evidence: { wordCount: x.wc, headedSections: headed, wordsPerSection, note: 'long page with sparse headings — sections average >500 words, too large to ground cleanly' } });
721
+ }
722
+ return out;
723
+ },
724
+ },
725
+ {
726
+ // Hobo Level 3 freshness / lastSignificantUpdate. Gemini guard: YoY windows (negate seasonality
727
+ // + zero-click-SERP CTR loss). Flag when the page hasn't been meaningfully re-dated in >12 months
728
+ // AND clicks are down >25% YoY AND impressions down >15% YoY (impressions confirm ranking decay,
729
+ // not just CTR). Needs ~13 months of GSC — guarded by spanDays so it stays silent on shallow syncs.
730
+ id: 'stale-content', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
731
+ title: 'Stale page declining year-on-year', fix: 'This page hasn’t been meaningfully updated in over a year and its Search Console clicks are down sharply versus the same period last year — refresh and expand the content (and honestly re-date it) to rebuild the freshness signal Google rewards.',
732
+ run: (c) => {
733
+ if (!c.gscMaxDate || spanDays(c) < 455)
734
+ return []; // YoY window reaches back 455 days — anything less truncates the prior-year period
735
+ const d = c.gscMaxDate;
736
+ const out = [];
737
+ const rs = rows(c, `
738
+ WITH cur AS (SELECT page_key, SUM(clicks) c, SUM(impressions) i FROM search_analytics WHERE page_key IS NOT NULL AND date <= '${d}' AND date > date('${d}','-90 days') GROUP BY page_key),
739
+ py AS (SELECT page_key, SUM(clicks) c, SUM(impressions) i FROM search_analytics WHERE page_key IS NOT NULL AND date <= date('${d}','-365 days') AND date > date('${d}','-455 days') GROUP BY page_key)
740
+ SELECT cur.page_key url, cur.c curC, cur.i curI, py.c pyC, py.i pyI, p.json_ld jl
741
+ FROM cur JOIN py ON py.page_key = cur.page_key JOIN pages p ON p.url_key = cur.page_key
742
+ WHERE p.indexable = 1 AND py.c >= 50 AND cur.c < py.c * 0.75 AND cur.i < py.i * 0.85
743
+ ORDER BY (py.c - cur.c) DESC LIMIT 40`);
744
+ for (const x of rs) {
745
+ const dm = newestSchemaDate(x.jl);
746
+ if (!dm)
747
+ continue;
748
+ const ageDays = (Date.parse(d) - Date.parse(dm)) / 86400000;
749
+ if (!(ageDays >= 365))
750
+ continue; // only genuinely stale pages (>12 months since last schema date)
751
+ out.push({ urlKey: null, evidence: { url: x.url, dateModified: dm.slice(0, 10), clicksYoY: `${x.pyC}→${x.curC} (-${Math.round((1 - x.curC / x.pyC) * 100)}%)`, impressionsYoY: `${x.pyI}→${x.curI}`, clicks: x.pyC - x.curC, impressions: x.pyI } });
752
+ }
753
+ return out;
754
+ },
755
+ },
756
+ {
757
+ id: 'high-ipr-no-traffic', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
758
+ title: 'Internal authority wasted on a no-traffic page', fix: 'High internal link equity (iPR) and Google does rank it (it earns impressions), yet it gets zero clicks — rewrite the title/snippet or improve the page, or repoint that authority to pages that convert it. (Requires impressions, so functional pages with no search demand are excluded.)',
759
+ run: (c) => (!c.gscMaxDate || spanDays(c) < 90) ? [] : rows(c, `SELECT p.url_key urlKey, p.ipr ipr, p.inlink_count inl, SUM(sa.impressions) impressions
760
+ FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key
761
+ WHERE p.indexable=1 AND p.ipr >= 50 AND COALESCE(p.click_depth, 999) >= 1 AND sa.date <= '${c.gscMaxDate}' AND sa.date > date('${c.gscMaxDate}','-90 days')
762
+ GROUP BY p.url_key HAVING SUM(sa.clicks)=0 AND SUM(sa.impressions) >= 100
763
+ ORDER BY p.ipr DESC LIMIT 50`).map(r => ({ urlKey: r.urlKey, evidence: { ipr: Math.round(r.ipr), inlinks: r.inl, impressions: r.impressions, clicks: 0, note: 'high internal authority + impressions, zero clicks in 90 days' } })),
764
+ },
765
+ {
766
+ id: 'homepage-missing-org-schema', category: 'schema', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
767
+ title: 'Homepage missing Organization / WebSite schema', fix: 'Add Organization (or LocalBusiness) and WebSite JSON-LD to the homepage — it underpins the knowledge panel, logo and sitelinks search box.',
768
+ run: (c) => {
769
+ const hp = (rows(c, `SELECT url_key urlKey, json_ld jsonLd FROM pages WHERE status_code=200 AND click_depth=0 LIMIT 1`)[0]
770
+ ?? rows(c, `SELECT url_key urlKey, json_ld jsonLd FROM pages WHERE status_code=200 ORDER BY inlink_count DESC LIMIT 1`)[0]);
771
+ if (!hp)
772
+ return [];
773
+ const types = new Set(parseJsonLdNodes(hp.jsonLd).map(nodeType).filter(Boolean));
774
+ const ok = ['Organization', 'LocalBusiness', 'Corporation', 'OnlineStore', 'WebSite'].some(t => types.has(t));
775
+ return ok ? [] : [{ urlKey: hp.urlKey, evidence: { found: [...types].join(', ') || 'none', note: 'no Organization/WebSite node on the homepage' } }];
776
+ },
777
+ },
778
+ {
779
+ id: 'breadcrumb-schema-inconsistent', category: 'schema', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
780
+ title: 'Breadcrumb schema missing on some pages', fix: 'Add BreadcrumbList JSON-LD — most of the site has it, so these pages are inconsistent and miss breadcrumb rich results.',
781
+ run: (c) => {
782
+ const pages = rows(c, `SELECT url_key urlKey, json_ld jsonLd, click_depth cd FROM pages WHERE status_code=200 AND indexable=1`);
783
+ let withBc = 0;
784
+ const without = [];
785
+ for (const p of pages) {
786
+ if (parseJsonLdNodes(p.jsonLd).some(n => nodeType(n) === 'BreadcrumbList'))
787
+ withBc++;
788
+ else if ((p.cd ?? 0) >= 2)
789
+ without.push(p.urlKey);
790
+ }
791
+ if (pages.length === 0 || withBc / pages.length < 0.4)
792
+ return []; // site doesn't use breadcrumbs → design choice, not a bug
793
+ return without.map(u => ({ urlKey: u, evidence: { note: 'site uses BreadcrumbList elsewhere; missing here' } }));
794
+ },
795
+ },
796
+ // ── Merged crawl × Search Console — the "expert questions" that need both datasets ──
797
+ {
798
+ id: 'ghost-pages', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
799
+ title: 'Ranking page the crawl can’t reach', fix: 'Google sends impressions/clicks to this URL but the site crawl never reached it — add internal links so it’s discoverable (or confirm it should exist and isn’t blocked).',
800
+ run: (c) => {
801
+ if (!c.gscMaxDate)
802
+ return [];
803
+ // Only meaningful on a COMPLETE crawl — if the crawl hit its maxPages cap, "absent
804
+ // from crawl" is unreliable. Tie the guard to the crawl that produced the CURRENT
805
+ // pages (not the latest crawl_metadata row, which may be a later failed crawl).
806
+ const m = c.db.prepare('SELECT urls_crawled c, max_pages m FROM crawl_metadata WHERE crawl_id = (SELECT crawl_id FROM pages LIMIT 1)').get();
807
+ // max_pages is nullable (NULL = no cap = complete crawl) — `c >= null` coerces to
808
+ // `c >= 0`, which would silently disable the check forever on capless crawls.
809
+ if (!m || m.c === 0 || (m.m != null && m.c >= m.m))
810
+ return [];
811
+ return rows(c, `SELECT page_key urlKey, SUM(clicks) clicks, SUM(impressions) impressions FROM search_analytics
812
+ WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key
813
+ HAVING SUM(impressions) >= 50 AND page_key NOT IN (SELECT url_key FROM pages)
814
+ ORDER BY impressions DESC`).map(r => ({ urlKey: r.urlKey, evidence: { clicks: r.clicks, impressions: r.impressions, note: 'earns GSC traffic but absent from the crawl' } }));
815
+ },
816
+ },
817
+ {
818
+ id: 'title-missing-top-query', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
819
+ title: 'Top query missing from the title', fix: 'Work the page’s top-performing query into the <title> — it already ranks for this term but the title doesn’t mention it.',
820
+ run: (c) => {
821
+ if (!c.gscMaxDate)
822
+ return [];
823
+ const r = rows(c, `SELECT s.page_key urlKey, s.query query, s.impr impressions, p.title title FROM
824
+ (SELECT page_key, query, SUM(impressions) impr, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
825
+ FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
826
+ JOIN pages p ON p.url_key = s.page_key
827
+ WHERE s.rn = 1 AND p.indexable = 1 AND p.title IS NOT NULL AND p.title != '' AND s.impr >= 100`);
828
+ return r
829
+ .filter(x => { const q = terms(x.query); return q.length > 0 && !q.some(w => titleHasTerm(x.title, w)); })
830
+ .map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, title: x.title } }));
831
+ },
832
+ },
833
+ // ── Internal link graph (iPR + click-depth + anchor text, from the crawl `links` table) ──
834
+ {
835
+ id: 'deep-pages', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
836
+ title: 'Page buried deep in the structure', fix: 'Add in-content (body) links from higher-level pages — this page is 4+ clicks from the homepage via body links.',
837
+ run: (c) => rows(c, `SELECT url_key urlKey, click_depth d FROM pages WHERE status_code=200 AND indexable=1 AND click_depth >= 4 ORDER BY click_depth DESC`).map(r => ({ urlKey: r.urlKey, evidence: { clickDepth: r.d } })),
838
+ },
839
+ {
840
+ id: 'underlinked-high-demand', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
841
+ title: 'High search demand, low internal authority', fix: 'Add internal links from high-authority (high-iPR) pages — this earns impressions but the site gives it little internal link equity.',
842
+ run: (c) => c.gscMaxDate ? rows(c, `SELECT p.url_key urlKey, p.ipr ipr, p.inlink_count inl, SUM(sa.impressions) impressions FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key WHERE p.indexable=1 AND p.ipr < 30 AND p.inlink_count BETWEEN 1 AND 3 AND sa.${win(c.gscMaxDate)} GROUP BY p.url_key HAVING SUM(sa.impressions) >= 300 ORDER BY impressions DESC`).map(r => ({ urlKey: r.urlKey, evidence: { impressions: r.impressions, ipr: Math.round(r.ipr), inlinks: r.inl } })) : [],
843
+ },
844
+ {
845
+ // The tier BETWEEN "orphan" (0 inlinks) and "fine": pages reached almost only via nav/footer.
846
+ // inlink_count counts ALL internal links, so a page sitting in the global nav looks well-linked
847
+ // even with zero EDITORIAL links — yet Google leans on in-content links for topic + equity. We
848
+ // count distinct in-content (placement='body') inbound sources; ≤2 + real demand = under-linked.
849
+ id: 'underlinked-editorial', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
850
+ title: 'High demand, almost no in-content internal links', fix: 'This page earns real impressions but is reached mainly via nav/footer — add descriptive in-content links to it from related articles. Editorial body links pass more topical context and equity than templated nav links.',
851
+ run: (c) => c.gscMaxDate ? rows(c, `
852
+ SELECT p.url_key urlKey, SUM(sa.impressions) impressions,
853
+ (SELECT COUNT(DISTINCT l.source_key) FROM links l WHERE l.target_key = p.url_key AND l.is_internal = 1 AND l.placement = 'body' AND l.source_key <> p.url_key) bodyLinks
854
+ FROM pages p JOIN search_analytics sa ON sa.page_key = p.url_key
855
+ WHERE p.indexable = 1 AND sa.${win(c.gscMaxDate)}
856
+ GROUP BY p.url_key
857
+ HAVING SUM(sa.impressions) >= 300 AND bodyLinks <= 2
858
+ ORDER BY impressions DESC LIMIT 40`).map(r => ({ urlKey: r.urlKey, evidence: { impressions: r.impressions, inContentLinks: r.bodyLinks, note: 'reached mainly via nav/footer — thin on editorial (in-content) links' } })) : [],
859
+ },
860
+ // NOTE: internal anchor-text checks (over-optimisation + generic/empty anchors) prototyped
861
+ // and PULLED twice. Re-evaluated 2026-06-22 against the AgricIDaniel/claude-seo and
862
+ // Bhanunamikaze/Agentic-SEO-Skill repos, this time using the links.placement='body' filter
863
+ // plus excluding anchors that match the target's own title/H1. On real data (ehi.com.au) the
864
+ // dominant survivors are still false positives: sitewide template CTAs ("home" 864/865,
865
+ // "contact us", "apply today") and category links whose anchor IS the page title. The
866
+ // page-level "mostly generic-anchored" variant returned 0 signal; raw empty anchors are
867
+ // image/thumbnail-link noise (3,764/23,435 body links). Reliable detection needs an
868
+ // editorial-vs-template link classifier we don't store (placement='body' still includes
869
+ // in-template CTAs and product grids). Keep pulled — a wrong finding is worse than none.
870
+ // ── Backlinks (need page_backlinks populated via pull_backlinks; gate so we never assert
871
+ // "no external links" without data) ──
872
+ {
873
+ id: 'backlinks-to-404', category: 'crawlability', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
874
+ title: 'External backlinks pointing to a dead page', fix: '301-redirect this URL to the best live equivalent — external link equity is hitting a 4xx/5xx page and being wasted.',
875
+ run: (c) => rows(c, `SELECT url_key urlKey, backlinks, referring_domains rd, status_code st FROM page_backlinks WHERE status_code >= 400 AND backlinks > 0 ORDER BY backlinks DESC`)
876
+ .map(r => ({ urlKey: r.urlKey, evidence: { status: r.st, backlinks: r.backlinks, referringDomains: r.rd } })),
877
+ },
878
+ {
879
+ id: 'orphan-no-links', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'per-page',
880
+ title: 'Orphan page — no internal or external links', fix: 'Add internal links (and earn external ones) — this indexable page has zero inlinks and no backlinks, so it depends on the sitemap alone.',
881
+ run: (c) => {
882
+ const has = c.db.prepare('SELECT COUNT(*) n FROM page_backlinks').get().n;
883
+ if (!has)
884
+ return []; // backlinks not pulled — can't credibly assert "no external links"
885
+ return rows(c, `SELECT p.url_key urlKey FROM pages p LEFT JOIN page_backlinks b ON b.url_key=p.url_key WHERE p.status_code=200 AND p.indexable=1 AND p.inlink_count=0 AND COALESCE(b.backlinks,0)=0`)
886
+ .map(r => ({ urlKey: r.urlKey, evidence: { inlinks: 0, backlinks: 0 } }));
887
+ },
888
+ },
889
+ // ── Phase 6a — expert questions where crawl (intent) and reality diverge ──
890
+ {
891
+ id: 'ipr-bleed-by-status', category: 'crawlability', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
892
+ title: 'Internal link equity flowing into dead URLs', fix: 'Repoint or 301 these internal links — they target non-200 URLs and waste the internal PageRank of the (often high-authority) pages linking to them.',
893
+ // Sum the iPR of the SOURCE pages linking to each non-200 internal target. "Found a 404" is
894
+ // junior; "this 404 drains the equity of N high-iPR pages" is the director-level find.
895
+ run: (c) => rows(c, `SELECT l.target_key urlKey, COUNT(DISTINCT l.source_key) linkingPages,
896
+ ROUND(SUM(src.ipr), 1) wastedIpr, t.status_code st
897
+ FROM links l
898
+ JOIN pages src ON src.url_key = l.source_key
899
+ JOIN pages t ON t.url_key = l.target_key
900
+ WHERE l.is_internal = 1 AND t.status_code >= 400 AND t.status_code NOT IN (429,503)
901
+ GROUP BY l.target_key HAVING SUM(src.ipr) > 0
902
+ ORDER BY wastedIpr DESC`).map(r => ({ urlKey: r.urlKey, evidence: { status: r.st, linkingPages: r.linkingPages, wastedIpr: r.wastedIpr } })),
903
+ },
904
+ {
905
+ id: 'broken-canonical-target', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
906
+ title: 'Canonical points to a broken or unhealthy URL', fix: 'Point the canonical at a live, indexable, self-canonical HTTPS URL — Google ignores a canonical whose target is a 4xx/5xx/redirect, noindex, itself canonicalised elsewhere (a chain/loop), or an HTTPS→HTTP downgrade.',
907
+ // Join the declared canonical_key back to the crawl. Only flag when we crawled the target.
908
+ // Skip self-canonicals. Covers: non-200 target, noindex target, canonical chain/loop (target
909
+ // canonicalises onward), and HTTPS→HTTP downgrade (research: Sitebulb indexability hints).
910
+ run: (c) => rows(c, `SELECT p.url_key urlKey, p.canonical_url canon, p.canonical_key ck, t.status_code st, t.noindex ni, t.canonical_key tck
911
+ FROM pages p JOIN pages t ON t.url_key = p.canonical_key
912
+ WHERE p.canonical_key IS NOT NULL AND p.canonical_key != p.url_key AND p.status_code = 200
913
+ AND (t.status_code != 200 OR t.noindex = 1
914
+ OR (t.canonical_key IS NOT NULL AND t.canonical_key != t.url_key)
915
+ OR (p.url_key LIKE 'https://%' AND p.canonical_url LIKE 'http://%'))`)
916
+ .map(r => {
917
+ const reason = r.st !== 200 ? `target returns HTTP ${r.st}`
918
+ : r.ni ? 'target is noindex'
919
+ : (r.tck && r.tck !== r.ck) ? (r.tck === r.urlKey ? 'canonical loop (target points back here)' : 'canonical chain (target canonicalises onward)')
920
+ : 'HTTPS page canonicalises to an HTTP URL';
921
+ return { urlKey: r.urlKey, evidence: { canonical: r.canon, targetStatus: r.st, reason } };
922
+ }),
923
+ },
924
+ {
925
+ id: 'faceted-spider-trap', category: 'crawlability', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'global',
926
+ title: 'Indexable faceted URLs burning crawl budget', fix: 'noindex (or robots-disallow / canonicalise) multi-parameter filter URLs — they are indexable but earn zero search traffic, so they only waste crawl budget and risk index bloat.',
927
+ // Multi-parameter (>=2 params), indexable, zero GSC impressions in-window = classic facet trap.
928
+ // Gate on GSC so "zero search value" is a real claim, not just "no data".
929
+ run: (c) => {
930
+ if (!c.gscMaxDate)
931
+ return [];
932
+ return rows(c, `SELECT url_key urlKey, url FROM pages
933
+ WHERE status_code = 200 AND indexable = 1 AND url LIKE '%?%' AND url LIKE '%&%'
934
+ AND url_key NOT IN (SELECT page_key FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} AND impressions > 0)
935
+ ORDER BY url`).map(r => ({ urlKey: r.urlKey, evidence: { url: r.url, params: (r.url.split('?')[1] ?? '').split('&').map((kv) => kv.split('=')[0]).join(', '), note: 'indexable, multi-parameter, zero GSC impressions' } }));
936
+ },
937
+ },
938
+ {
939
+ id: 'soft-404-shell', category: 'indexation', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
940
+ title: 'Soft 404 — 200 OK but Google treats it as not-found', fix: 'Either populate the page with real content, or return a true 404/410 (or noindex) — it serves 200 but Google has flagged it as a soft 404.',
941
+ // URL Inspection page-fetch-state = soft 404 while the crawler sees a 200. The crawler alone
942
+ // would call this page fine; Google disagrees. Needs URL Inspection populated.
943
+ run: (c) => rows(c, `SELECT i.url_key urlKey, p.word_count wc, p.bytes bytes
944
+ FROM url_inspection i JOIN pages p ON p.url_key = i.url_key
945
+ WHERE LOWER(i.page_fetch_state) LIKE '%soft%' AND p.status_code = 200`).map(r => ({ urlKey: r.urlKey, evidence: { pageFetchState: 'soft 404', wordCount: r.wc, bytes: r.bytes } })),
946
+ },
947
+ {
948
+ id: 'rich-result-issues', category: 'schema', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
949
+ title: 'Google-verified rich result issues (URL Inspection)', fix: 'Fix the structured-data issues Google itself reports for this page — these come from the URL Inspection API (Google’s own validation), not our local validator, so they are authoritative. Address the listed issue messages per rich-result type.',
950
+ // Parses the stored richResultsResult JSON (url_inspection.rich_results) that inspect_urls
951
+ // already captures: detectedItems[].items[].issues[] carries Google's severity + message.
952
+ // Needs URL Inspection populated (run inspect_urls). Zero API cost — data is already in the DB.
953
+ run: (c) => {
954
+ const out = [];
955
+ for (const r of iterRows(c, `SELECT url_key urlKey, rich_results rr FROM url_inspection WHERE rich_results IS NOT NULL AND rich_results != ''`)) {
956
+ let parsed;
957
+ try {
958
+ parsed = JSON.parse(r.rr);
959
+ }
960
+ catch {
961
+ continue;
962
+ }
963
+ // Dedup per (type, severity, message) with a count — a listicle repeats the same
964
+ // "Missing field review" warning per product item; keep one sample item per group.
965
+ const groups = new Map();
966
+ for (const det of parsed?.detectedItems ?? []) {
967
+ for (const item of det?.items ?? []) {
968
+ for (const iss of item?.issues ?? []) {
969
+ if (!iss?.issueMessage)
970
+ continue;
971
+ const key = `${det.richResultType ?? 'unknown'}|${iss.severity ?? ''}|${iss.issueMessage}`;
972
+ const g = groups.get(key);
973
+ if (g)
974
+ g.count++;
975
+ else
976
+ groups.set(key, { richResultType: det.richResultType ?? 'unknown', severity: iss.severity ?? null, message: iss.issueMessage, count: 1, sampleItem: item.name ?? null });
977
+ }
978
+ }
979
+ }
980
+ if (groups.size)
981
+ out.push({ urlKey: r.urlKey, evidence: { verdict: parsed?.verdict ?? null, issues: [...groups.values()] } });
982
+ }
983
+ return out;
984
+ },
985
+ },
986
+ {
987
+ id: 'broken-hreflang-target', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
988
+ title: 'hreflang points to a broken/non-indexable URL', fix: 'Point each hreflang alternate at a live, indexable URL — Google drops the whole cluster if an alternate is 4xx/5xx/redirect/noindex.',
989
+ run: (c) => hreflangFindings(c).broken,
990
+ },
991
+ {
992
+ id: 'hreflang-no-return-tag', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
993
+ title: 'hreflang missing return tag (not reciprocated)', fix: 'Add the reciprocal hreflang on the target page — Google ignores one-way hreflang annotations that don’t link back.',
994
+ run: (c) => hreflangFindings(c).noReturn,
995
+ },
996
+ // ── 6a finishers — consume persisted DataForSEO enrichments (gated; need the data pulled) ──
997
+ {
998
+ id: 'intent-vs-pagetype-mismatch', category: 'merged', severity: 'med', labels: ['G', 'N'], certainty: 0.7, effortBase: 8, fixType: 'per-page',
999
+ title: 'Page type mismatches its query intent', fix: 'Reformat or retarget the page — its template doesn’t match the SERP intent for its top query (e.g. a product page ranking for an informational query, or an article for a transactional one). Run search_intent siteUrl:<property> to populate intents.',
1000
+ // Join each page's top GSC query → persisted keyword_intent → page schema flavour (from json_ld).
1001
+ // Judgement (N, 0.7): intent + schema-type inference is heuristic, so it only runs with includeJudgement.
1002
+ run: (c) => {
1003
+ if (!c.gscMaxDate)
1004
+ return [];
1005
+ if ((c.db.prepare('SELECT COUNT(*) n FROM keyword_intent').get().n) === 0)
1006
+ return []; // intents not pulled
1007
+ const r = rows(c, `SELECT s.page_key urlKey, s.query query, ki.intent intent, p.json_ld jsonLd FROM
1008
+ (SELECT page_key, query, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
1009
+ FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
1010
+ JOIN pages p ON p.url_key = s.page_key
1011
+ JOIN keyword_intent ki ON ki.keyword = LOWER(s.query)
1012
+ WHERE s.rn = 1 AND p.status_code = 200 AND p.json_ld IS NOT NULL AND p.json_ld != ''`);
1013
+ const out = [];
1014
+ for (const x of r) {
1015
+ const types = parseJsonLdNodes(x.jsonLd).map(n => (nodeType(n) ?? '').toLowerCase());
1016
+ const productish = types.some(t => t === 'product' || t === 'offer');
1017
+ const articleish = types.some(t => /article|blogposting|newsarticle/.test(t));
1018
+ const intent = (x.intent || '').toLowerCase();
1019
+ let mismatch = null;
1020
+ if (productish && intent === 'informational')
1021
+ mismatch = 'product/offer page ranking for an informational query';
1022
+ else if (articleish && intent === 'transactional')
1023
+ mismatch = 'article page ranking for a transactional query';
1024
+ if (mismatch)
1025
+ out.push({ urlKey: x.urlKey, evidence: { topQuery: x.query, queryIntent: intent, mismatch } });
1026
+ }
1027
+ return out;
1028
+ },
1029
+ },
1030
+ {
1031
+ id: 'high-yield-cwv-fail', category: 'performance', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 8, fixType: 'per-page',
1032
+ title: 'High-traffic page failing Core Web Vitals', fix: 'Prioritise CWV work here — this page earns real clicks but fails lab Core Web Vitals (LCP > 2.5s, CLS > 0.1, or performance < 50), so engineering effort has clear ROI. Run page_lighthouse siteUrl:<property> on key URLs to populate CWV.',
1033
+ run: (c) => {
1034
+ if (!c.gscMaxDate)
1035
+ return [];
1036
+ if ((c.db.prepare('SELECT COUNT(*) n FROM page_cwv').get().n) === 0)
1037
+ return []; // CWV not pulled
1038
+ return rows(c, `SELECT cw.url_key urlKey, cw.lcp_ms lcp, cw.cls cls, cw.performance perf,
1039
+ (SELECT COALESCE(SUM(clicks),0) FROM search_analytics sa WHERE sa.page_key = cw.url_key AND ${win(c.gscMaxDate)}) clicks
1040
+ FROM page_cwv cw
1041
+ WHERE (cw.lcp_ms > 2500 OR cw.cls > 0.1 OR cw.performance < 0.5)`)
1042
+ .filter(r => r.clicks > 0)
1043
+ .map(r => ({ urlKey: r.urlKey, evidence: { clicks: r.clicks, lcpMs: r.lcp != null ? Math.round(r.lcp) : null, cls: r.cls != null ? Math.round(r.cls * 1000) / 1000 : null, performance: r.perf != null ? Math.round(r.perf * 100) : null } }));
1044
+ },
1045
+ },
1046
+ // ── 6b — per-template systemic issues (template-typed, deterministic) ──
1047
+ {
1048
+ id: 'pagination-canonical-to-page-1', category: 'crawlability', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
1049
+ title: 'Paginated pages canonicalising away from themselves', fix: 'Make each paginated page (page 2, 3, …) self-canonical. Canonicalising page 2+ back to page 1 tells Google the deeper pages are duplicates, so products/articles linked only from page 2+ drop out of the crawl.',
1050
+ run: (c) => rows(c, `SELECT url_key urlKey, url, canonical_url canon FROM pages
1051
+ WHERE status_code=200 AND canonical_count>0 AND canonical_key IS NOT NULL AND canonical_key != url_key
1052
+ AND (rel_prev=1 OR url LIKE '%/page/%' OR url GLOB '*[?&]page=[0-9]*' OR url GLOB '*[?&]paged=[0-9]*' OR url GLOB '*[?&]p=[0-9]*')`)
1053
+ .filter(r => !/[?&](page|p)=1(\b|&|$)/.test(r.url) && !/\/page\/1(\/?$|\?)/.test(r.url)) // page 1 self-canonicalising to base is fine
1054
+ .map(r => ({ urlKey: r.urlKey, evidence: { url: r.url, canonical: r.canon } })),
1055
+ },
1056
+ {
1057
+ id: 'article-date-illogical', category: 'schema', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
1058
+ title: 'Article dateModified earlier than datePublished', fix: 'Fix the Article schema dates — dateModified must be on or after datePublished. An impossible date undermines trust in the markup and can suppress the freshness signal.',
1059
+ run: (c) => {
1060
+ const out = [];
1061
+ for (const r of rows(c, `SELECT url_key urlKey, json_ld jsonLd FROM pages WHERE status_code=200 AND json_ld LIKE '%Article%' AND json_ld LIKE '%date%'`)) {
1062
+ for (const n of parseJsonLdNodes(r.jsonLd)) {
1063
+ const t = nodeType(n);
1064
+ if (!t || !/article|blogposting|newsarticle/i.test(t))
1065
+ continue;
1066
+ const pub = Date.parse(n.datePublished), mod = Date.parse(n.dateModified);
1067
+ if (!Number.isNaN(pub) && !Number.isNaN(mod) && mod < pub) {
1068
+ out.push({ urlKey: r.urlKey, evidence: { datePublished: n.datePublished, dateModified: n.dateModified } });
1069
+ break;
1070
+ }
1071
+ }
1072
+ }
1073
+ return out;
1074
+ },
1075
+ },
1076
+ // ── 6d — Wikidata entity layer (heuristic H1→QID; N/judgement, gated on resolve_entities) ──
1077
+ {
1078
+ id: 'entity-internal-link-gap', category: 'crawlability', severity: 'low', labels: ['N'], certainty: 0.5, effortBase: 3, fixType: 'per-page',
1079
+ title: 'Topically related pages not internally linked', fix: 'Add an internal link from the broader page to the more specific one — Wikidata says their entities are related (subclass-of / part-of) but no internal link connects them, leaving a gap in the topical mesh. Run resolve_entities first; verify the entity match before acting (heuristic).',
1080
+ run: (c) => {
1081
+ if ((c.db.prepare('SELECT COUNT(*) n FROM page_entity').get().n) === 0)
1082
+ return []; // not resolved
1083
+ return rows(c, `SELECT parent.url_key urlKey, child.url_key target, parent.label pl, child.label cl, ee.relation rel
1084
+ FROM entity_edge ee
1085
+ JOIN page_entity child ON child.qid = ee.qid
1086
+ JOIN page_entity parent ON parent.qid = ee.related_qid
1087
+ WHERE parent.url_key != child.url_key
1088
+ AND NOT EXISTS (SELECT 1 FROM links l WHERE l.source_key = parent.url_key AND l.target_key = child.url_key AND l.is_internal = 1)`)
1089
+ .map(r => ({ urlKey: r.urlKey, evidence: { suggestLinkTo: r.target, parentEntity: r.pl, childEntity: r.cl, relation: r.rel } }));
1090
+ },
1091
+ },
1092
+ // ── Sitemap ↔ crawl reconciliation (gated on a sitemap having been fetched) ──
1093
+ {
1094
+ id: 'sitemap-non-indexable', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
1095
+ title: 'Sitemap lists non-indexable URLs', fix: 'Remove URLs from the XML sitemap that are 4xx/5xx, redirected, noindex, canonicalised or robots-blocked — the sitemap should list only canonical, indexable pages, or Google loses trust in it.',
1096
+ run: (c) => {
1097
+ if (!sitemapHasRows(c))
1098
+ return [];
1099
+ return rows(c, `SELECT s.url_key urlKey, p.indexable_reason reason, p.status_code st FROM sitemap_urls s JOIN pages p ON p.url_key = s.url_key WHERE p.indexable = 0`)
1100
+ .map(r => ({ urlKey: r.urlKey, evidence: { reason: r.reason ?? 'not-indexable', status: r.st } }));
1101
+ },
1102
+ },
1103
+ {
1104
+ id: 'indexable-not-in-sitemap', category: 'indexation', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
1105
+ title: 'Indexable pages missing from the sitemap', fix: 'Add these indexable pages to the XML sitemap so Google discovers and prioritises them.',
1106
+ run: (c) => {
1107
+ if (!sitemapHasRows(c))
1108
+ return [];
1109
+ return rows(c, `SELECT p.url_key urlKey FROM pages p LEFT JOIN sitemap_urls s ON s.url_key = p.url_key WHERE p.status_code = 200 AND p.indexable = 1 AND ${notPagination('p.url_key')} AND s.url_key IS NULL`)
1110
+ .map(r => ({ urlKey: r.urlKey, evidence: {} }));
1111
+ },
1112
+ },
1113
+ {
1114
+ id: 'sitemap-orphan', category: 'crawlability', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
1115
+ title: 'Sitemap page with no internal links', fix: 'Add internal links — this page is in the sitemap but nothing links to it, so it relies on the sitemap alone for discovery and earns little internal authority.',
1116
+ run: (c) => {
1117
+ if (!sitemapHasRows(c))
1118
+ return [];
1119
+ return rows(c, `SELECT p.url_key urlKey FROM pages p JOIN sitemap_urls s ON s.url_key = p.url_key WHERE p.status_code = 200 AND p.indexable = 1 AND p.inlink_count = 0`)
1120
+ .map(r => ({ urlKey: r.urlKey, evidence: { inlinks: 0, note: 'in sitemap, no internal links' } }));
1121
+ },
1122
+ },
1123
+ // ── Cheap performance proxies (from data captured at crawl time — no extra fetch) ──
1124
+ {
1125
+ id: 'slow-response', category: 'performance', severity: 'med', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'global',
1126
+ title: 'Slow server response (TTFB proxy)', fix: 'Investigate slow server/TTFB — caching, CDN, or backend. Response time over ~1.5s hurts Core Web Vitals and crawl rate.',
1127
+ run: (c) => rows(c, `SELECT url_key urlKey, response_time_ms ms FROM pages WHERE status_code=200 AND ${HTML_CT} AND response_time_ms > 1500 ORDER BY response_time_ms DESC`)
1128
+ .map(r => ({ urlKey: r.urlKey, evidence: { responseMs: r.ms } })),
1129
+ },
1130
+ {
1131
+ id: 'large-html', category: 'performance', severity: 'low', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'per-page',
1132
+ title: 'Large HTML document (transferred)', fix: 'Trim the HTML payload — bloated markup slows render and First Contentful Paint (often huge inline SVG/CSS/JSON or unminified output). Judged on TRANSFERRED bytes, not raw: behind a compressing CDN (brotli/gzip) raw size matters far less.',
1133
+ // Severity is on what the browser actually downloads, not raw bytes: a 200KB page served brotli
1134
+ // is ~45KB over the wire and is NOT a real perf problem. We estimate transfer size (raw × ~0.22
1135
+ // for br/gzip, else raw) and only flag pages whose ESTIMATED transferred HTML exceeds ~60KB.
1136
+ // Prefilter at the 60KB threshold itself, not higher — an UNCOMPRESSED 60–150KB page
1137
+ // (est = raw) is exactly the case that matters most and must reach the est filter.
1138
+ run: (c) => rows(c, `SELECT url_key urlKey, bytes, content_encoding enc FROM pages WHERE status_code=200 AND ${HTML_CT} AND bytes > 60000 ORDER BY bytes DESC`)
1139
+ .map(r => { const compressed = /br|gzip|deflate|zstd/i.test(r.enc || ''); const est = compressed ? Math.round(r.bytes * 0.22) : r.bytes; return { urlKey: r.urlKey, est, compressed, raw: r.bytes }; })
1140
+ .filter(x => x.est > 60000)
1141
+ .map(x => ({ urlKey: x.urlKey, evidence: { rawBytes: x.raw, estTransferBytes: x.est, compressed: x.compressed, note: x.compressed ? 'raw HTML; estimated transferred size after CDN compression' : 'served UNCOMPRESSED — enable brotli/gzip' } })),
1142
+ },
1143
+ {
1144
+ id: 'uncompressed-html', category: 'performance', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
1145
+ title: 'HTML served without compression', fix: 'Enable gzip or brotli for HTML responses — uncompressed HTML wastes bandwidth and slows load. Usually a one-line server/CDN setting.',
1146
+ run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (content_encoding IS NULL OR content_encoding='')`)
1147
+ .map(r => ({ urlKey: r.urlKey, evidence: {} })),
1148
+ },
1149
+ // ── Checklist-coverage additions (2026-07-20 — see plan/checklist-coverage.md) ──
1150
+ {
1151
+ id: 'robots-blocked-with-traffic', category: 'crawlability', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 1, fixType: 'per-page',
1152
+ title: 'Robots-blocked page still earning search traffic', fix: 'This URL is disallowed in robots.txt yet Google still shows it (usually as a bare "no information" result) and users still land on it. Either unblock it so it can be crawled and ranked properly, or — if it genuinely shouldn\'t be found — unblock it AND add noindex (a robots-blocked page can never see the noindex).',
1153
+ run: (c) => !c.gscMaxDate ? [] : rows(c, `SELECT p.url_key urlKey, SUM(sa.clicks) clicks, SUM(sa.impressions) impressions
1154
+ FROM pages p JOIN search_analytics sa ON sa.page_key = p.url_key
1155
+ WHERE p.indexable_reason='robots-disallowed' AND ${win(c.gscMaxDate)}
1156
+ GROUP BY p.url_key HAVING SUM(sa.impressions) >= 10
1157
+ ORDER BY SUM(sa.impressions) DESC LIMIT 50`)
1158
+ .map(r => ({ urlKey: r.urlKey, evidence: { clicks: r.clicks, impressions: r.impressions, note: 'disallowed in robots.txt yet earning impressions' } })),
1159
+ },
1160
+ {
1161
+ id: 'missing-lang', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
1162
+ title: 'Pages missing an html lang attribute', fix: 'Declare the page language on the <html> element (e.g. lang="en-GB") — it helps search engines, screen readers and translation systems know what language they\'re reading. Usually one template edit.',
1163
+ run: (c) => {
1164
+ const n = c.db.prepare(`SELECT COUNT(*) n FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND (lang IS NULL OR TRIM(lang)='')`).get().n;
1165
+ return n > 0 ? [{ urlKey: null, evidence: { pages: n, note: 'indexable pages with no lang attribute' } }] : [];
1166
+ },
1167
+ },
1168
+ {
1169
+ id: 'image-preview-restricted', category: 'indexation', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
1170
+ title: 'max-image-preview restricted below large', fix: 'The robots meta tag caps image previews at none/standard. Google Discover strongly favours large image previews — set max-image-preview:large (or remove the restriction) unless there\'s a licensing reason not to.',
1171
+ run: (c) => rows(c, `SELECT url_key urlKey, robots FROM pages WHERE status_code=200 AND indexable=1
1172
+ AND (LOWER(REPLACE(robots,' ','')) LIKE '%max-image-preview:none%' OR LOWER(REPLACE(robots,' ','')) LIKE '%max-image-preview:standard%') LIMIT 50`)
1173
+ .map(r => ({ urlKey: r.urlKey, evidence: { robots: r.robots } })),
1174
+ },
1175
+ {
1176
+ id: 'excessive-links', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
1177
+ title: 'Excessive number of links on the page', fix: 'Hundreds of links on one page dilute the equity each one passes and drown the ones that matter. Trim boilerplate link blocks (mega-menus, tag clouds, footer sprawl) so the important links stand out.',
1178
+ run: (c) => rows(c, `SELECT url_key urlKey, internal_links il, external_links el FROM pages
1179
+ WHERE status_code=200 AND indexable=1 AND (internal_links + external_links) > 300
1180
+ ORDER BY (internal_links + external_links) DESC LIMIT 30`)
1181
+ .map(r => ({ urlKey: r.urlKey, evidence: { internalLinks: r.il, externalLinks: r.el, total: r.il + r.el } })),
1182
+ },
1183
+ {
1184
+ id: 'favicon-missing', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
1185
+ title: 'No favicon declared', fix: 'Add a favicon (<link rel="icon" …>) — Google shows it next to your result on mobile, and a missing one costs a little trust/recognition on every SERP appearance. One line in the template.',
1186
+ // Gated on the column being populated (has_favicon is NULL on crawls from before this
1187
+ // was captured) — never flag from a pre-feature crawl.
1188
+ run: (c) => {
1189
+ const hp = c.db.prepare(`SELECT url_key, has_favicon hf FROM pages WHERE status_code=200 AND has_favicon IS NOT NULL ORDER BY (click_depth=0) DESC, inlink_count DESC LIMIT 1`).get();
1190
+ return hp && hp.hf === 0 ? [{ urlKey: hp.url_key, evidence: { note: 'no <link rel="icon"> on the homepage' } }] : [];
1191
+ },
1192
+ },
1193
+ {
1194
+ id: 'sitemap-lastmod-untrustworthy', category: 'indexation', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
1195
+ title: 'Sitemap lastmod dates are not trustworthy', fix: 'Google uses <lastmod> to prioritise recrawling — but only while it stays honest; a sitemap that stamps everything with the generation date (or future dates) teaches Google to ignore yours. Make lastmod reflect the last genuine content change, or drop it entirely.',
1196
+ run: (c) => {
1197
+ const all = c.db.prepare(`SELECT url_key, lastmod FROM sitemap_urls WHERE lastmod IS NOT NULL AND lastmod <> ''`).all();
1198
+ if (all.length < 20)
1199
+ return []; // too few dated URLs to judge the pattern
1200
+ const ev = { urlsWithLastmod: all.length };
1201
+ let tripped = false;
1202
+ // Tell 1: a generator stamping every URL with "now" — >90% share one date, and that date is recent.
1203
+ const byDay = new Map();
1204
+ for (const r of all)
1205
+ byDay.set(r.lastmod.slice(0, 10), (byDay.get(r.lastmod.slice(0, 10)) ?? 0) + 1);
1206
+ const [topDay, topN] = [...byDay.entries()].sort((a, b) => b[1] - a[1])[0];
1207
+ const ageDays = (Date.now() - Date.parse(topDay)) / 86400000;
1208
+ if (topN / all.length > 0.9 && Number.isFinite(ageDays) && ageDays < 35) {
1209
+ tripped = true;
1210
+ ev.sharedStamp = `${Math.round(topN / all.length * 100)}% of URLs claim ${topDay} — a generation timestamp, not a change date`;
1211
+ }
1212
+ // Tell 2: dates in the future.
1213
+ const future = all.filter(r => Date.parse(r.lastmod) > Date.now() + 86400000).length;
1214
+ if (future > 0) {
1215
+ tripped = true;
1216
+ ev.futureDates = future;
1217
+ }
1218
+ // Tell 3: lastmod claims a change between our two most recent crawls, yet none of the
1219
+ // tracked page fields (status, title, meta, H1, word count, schema types) changed.
1220
+ const crawls = latestTwoCrawls(c.db);
1221
+ if (crawls.length === 2) {
1222
+ // Compare DATE prefixes on both sides — lastmod is stored verbatim and often a full
1223
+ // timestamp; compared raw against a 10-char date, same-day stamps sort "after" the
1224
+ // boundary and are silently excluded (exactly the stamp-everything-today pattern).
1225
+ const phantom = c.db.prepare(`SELECT s.url_key FROM sitemap_urls s
1226
+ JOIN page_snapshots n ON n.url_key = s.url_key AND n.crawl_id = ?
1227
+ JOIN page_snapshots o ON o.url_key = s.url_key AND o.crawl_id = ?
1228
+ WHERE substr(s.lastmod,1,10) > substr(?,1,10) AND substr(s.lastmod,1,10) <= substr(?,1,10)
1229
+ AND n.status_code IS o.status_code AND n.title IS o.title AND n.meta_description IS o.meta_description
1230
+ AND n.h1 IS o.h1 AND n.word_count IS o.word_count AND n.schema_types IS o.schema_types
1231
+ LIMIT 200`).all(crawls[0].crawl_id, crawls[1].crawl_id, crawls[1].at, crawls[0].at);
1232
+ if (phantom.length >= 5) {
1233
+ tripped = true;
1234
+ ev.phantomChanges = phantom.length;
1235
+ ev.phantomExamples = phantom.slice(0, 5).map(p => p.url_key);
1236
+ }
1237
+ }
1238
+ return tripped ? [{ urlKey: null, evidence: ev }] : [];
1239
+ },
1240
+ },
1241
+ {
1242
+ id: 'no-304-revalidation', category: 'performance', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
1243
+ title: 'Server ignores conditional requests (no 304)', fix: 'Pages advertise Last-Modified/ETag but the server re-serves a full 200 when asked "has this changed?" (If-Modified-Since / If-None-Match). A properly configured server answers 304 Not Modified — it saves bandwidth on every revalidating crawler and cache, and signals stability to Googlebot. Usually a server/CDN setting.',
1244
+ // Populated by the crawler's post-crawl probe (pages.conditional_304); NULL-gated so
1245
+ // pre-feature crawls never flag. Fires only when NO probed page honoured the request.
1246
+ run: (c) => {
1247
+ const r = c.db.prepare(`SELECT COUNT(*) probed, COALESCE(SUM(conditional_304),0) ok FROM pages WHERE conditional_304 IS NOT NULL`).get();
1248
+ if (r.probed < 5 || r.ok > 0)
1249
+ return [];
1250
+ const noValidators = c.db.prepare(`SELECT COUNT(*) n FROM pages WHERE status_code=200 AND ${HTML_CT} AND last_modified IS NULL AND etag IS NULL`).get().n;
1251
+ return [{ urlKey: null, evidence: { probed: r.probed, honoured304: 0, pagesWithoutValidators: noValidators, note: 'every conditional re-request returned a full 200' } }];
1252
+ },
1253
+ },
1254
+ {
1255
+ id: 'analytics-missing', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
1256
+ title: 'No client-side analytics detected', fix: 'No analytics or tag-manager snippet was found on any crawled page (GA4, GTM, Plausible, Matomo, Fathom, Clarity…). If you measure server-side, ignore this; otherwise you\'re flying blind — install an analytics package before making SEO decisions.',
1257
+ // Fires only when EVERY populated page lacks a snippet — one page with analytics = installed.
1258
+ run: (c) => {
1259
+ const r = c.db.prepare(`SELECT COUNT(*) total, COALESCE(SUM(has_analytics),0) withA FROM pages WHERE status_code=200 AND ${HTML_CT} AND has_analytics IS NOT NULL`).get();
1260
+ return r.total >= 3 && r.withA === 0 ? [{ urlKey: null, evidence: { pagesChecked: r.total, note: 'no known analytics snippet on any crawled page (server-side measurement is invisible to a crawl)' } }] : [];
1261
+ },
1262
+ },
1263
+ // ── Google "AI features / succeeding in AI search" guide (developers.google.com/search/docs/
1264
+ // fundamentals/ai-optimization-guide, read 2026-08-02) — principle → check mapping:
1265
+ // • Unique/compelling/non-commodity, people-first content .... thin-content, content-bloat,
1266
+ // ai-slop-signals (NEW below — mechanical/templated prose is the anti-signal of "unique take")
1267
+ // • Unique point of view / first-hand experience .............. article-no-author (NEW below —
1268
+ // unattributed articles are the deterministically checkable slice), stale-content
1269
+ // • Clear organisation (paragraphs/sections/headings) ......... poor-chunkability, content-bloat,
1270
+ // heading-hierarchy; answer coverage: body-missing-top-query, rag-answer-gap,
1271
+ // answer-not-front-loaded, weak-passage-answer, low-extractability
1272
+ // • Indexed + snippet-eligible / technical requirements ....... indexation family (noindex,
1273
+ // canonical, robots-blocked-with-traffic), soft-404-shell, sitemap reconciliation
1274
+ // • Crawlable public content .................................. crawlability family, broken links,
1275
+ // redirect chains; freshness-honesty (sitemap-lastmod-untrustworthy, no-304-revalidation)
1276
+ // • Semantic HTML / parseable ................................. heading-hierarchy, missing-h1,
1277
+ // multiple-h1, missing-lang
1278
+ // • Reduce duplicate content .................................. duplicate-title/meta, canonical
1279
+ // family, faceted-spider-trap, keyword-cannibalisation
1280
+ // • Images/video supporting text .............................. image-alt, images-missing-dimensions
1281
+ // • Page experience / latency ................................. performance proxies, high-yield-cwv-fail
1282
+ // • Structured data honesty (not required, but keep it valid) . schema-validate family,
1283
+ // article-date-illogical
1284
+ // • Don't chunk artificially / rewrite for AI ................. covered by NOT having such checks;
1285
+ // our chunk checks reward structure for humans, not tiny AI fragments
1286
+ // • "Don't create llms.txt" ................................... tension with agent-readiness probes,
1287
+ // which score llms.txt for *agent* (not Google AI-search) consumption — left as-is, different audience
1288
+ // • Merchant Center / Business Profile / GenAI report ......... out of scope (not crawl/GSC data)
1289
+ {
1290
+ // Anti-signal of the guide's "unique, non-commodity, people-first content": prose that reads
1291
+ // machine-generated. Three heuristics over body_chunks — slop-lexicon density, sentence-length
1292
+ // uniformity (low variance = mechanical), repeated chunk openers (page-internal boilerplate).
1293
+ // Precision over recall: absolute floors on every signal, ≥2 signals required, AND the composite
1294
+ // score must sit in the top decile of pages showing any signal. Judgement-gated — a human wrote
1295
+ // "delve" long before LLMs did.
1296
+ id: 'ai-slop-signals', category: 'content', severity: 'low', labels: ['N'], certainty: 0.5, effortBase: 5, fixType: 'per-page',
1297
+ title: 'Prose shows machine-generated (slop) signals', fix: 'This page’s copy trips several statistical tells of generic AI-generated text: stock filler phrases, unusually uniform sentence lengths, and/or sections that all open the same way. Google’s AI-search guidance rewards unique, people-first content with a first-hand point of view — rewrite the flagged sections with specifics only you can supply (real experience, real numbers, real opinions) and cut the filler. (Heuristic — verify by reading the page; competent human writing can trip these tells.)',
1298
+ run: (c) => {
1299
+ const PHRASES = [
1300
+ 'in today’s fast-paced world', "in today's fast-paced world", 'in today’s digital age', "in today's digital age",
1301
+ 'it’s important to note', "it's important to note", 'it is important to note', 'in conclusion',
1302
+ 'harness the power', 'unlock the potential', 'look no further', 'in the realm of', 'navigate the complexities',
1303
+ 'a testament to', 'plays a crucial role', 'a wide range of', 'when it comes to', 'at the end of the day',
1304
+ 'whether you’re a', "whether you're a", 'let’s dive', "let's dive", 'dive into the world of',
1305
+ 'elevate your', 'take your * to the next level', 'game-changer', 'game changer', 'cutting-edge', 'ever-evolving',
1306
+ 'treasure trove', 'rich tapestry', 'delve', 'delving', 'seamlessly', 'revolutionize', 'revolutionise', 'unleash',
1307
+ ].filter(p => !p.includes('*'));
1308
+ const metrics = [];
1309
+ for (const x of iterRows(c, `SELECT url_key urlKey, word_count wc, body_chunks bc FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND word_count >= 300 AND body_chunks IS NOT NULL`)) {
1310
+ let chunks;
1311
+ try {
1312
+ chunks = JSON.parse(x.bc);
1313
+ }
1314
+ catch {
1315
+ continue;
1316
+ }
1317
+ const texts = chunks.map((k) => String(k.text || '')).filter(t => t.length > 0);
1318
+ if (!texts.length)
1319
+ continue;
1320
+ const body = texts.join(' ').toLowerCase();
1321
+ const bodyWords = body.split(/\s+/).filter(Boolean).length;
1322
+ if (bodyWords < 250)
1323
+ continue;
1324
+ // (i) slop-lexicon density (hits per 1000 words, ≥3 distinct terms required)
1325
+ const hits = new Map();
1326
+ for (const p of PHRASES) {
1327
+ let n = 0, i = -1;
1328
+ while ((i = body.indexOf(p, i + 1)) !== -1)
1329
+ n++;
1330
+ if (n)
1331
+ hits.set(p, n);
1332
+ }
1333
+ const totalHits = [...hits.values()].reduce((a, b) => a + b, 0);
1334
+ const density = totalHits / bodyWords * 1000;
1335
+ const lexSignal = hits.size >= 3 && density >= 2.5;
1336
+ // (ii) sentence-length uniformity — coefficient of variation over sentence word-counts
1337
+ const sentences = body.split(/[.!?]+\s/).map(s => s.split(/\s+/).filter(Boolean).length).filter(n => n >= 5 && n <= 60);
1338
+ let cv = null;
1339
+ if (sentences.length >= 12) {
1340
+ const mean = sentences.reduce((a, b) => a + b, 0) / sentences.length;
1341
+ cv = Math.sqrt(sentences.reduce((a, b) => a + (b - mean) ** 2, 0) / sentences.length) / mean;
1342
+ }
1343
+ const uniformSignal = cv !== null && cv < 0.28;
1344
+ // (iii) repeated openers across the page's own chunks (first 3 words, ≥3 chunks sharing one)
1345
+ const openers = new Map();
1346
+ if (texts.length >= 5)
1347
+ for (const t of texts) {
1348
+ // Letters only — numeric/UI-chrome openers ("Show 10 20…") are widget text, not prose.
1349
+ const o = t.toLowerCase().replace(/[^a-z\s]/g, ' ').split(/\s+/).filter(w => w.length >= 2).slice(0, 3).join(' ');
1350
+ if (o.split(' ').length === 3)
1351
+ openers.set(o, (openers.get(o) ?? 0) + 1);
1352
+ }
1353
+ const topOpener = [...openers.entries()].sort((a, b) => b[1] - a[1])[0];
1354
+ const boilerSignal = !!topOpener && topOpener[1] >= 3;
1355
+ const signals = (lexSignal ? 1 : 0) + (uniformSignal ? 1 : 0) + (boilerSignal ? 1 : 0);
1356
+ if (signals === 0)
1357
+ continue;
1358
+ const score = density + (uniformSignal ? 3 : 0) + (boilerSignal ? 2 : 0);
1359
+ metrics.push({ urlKey: x.urlKey, score, signals, hits, density, cv, opener: boilerSignal ? { text: topOpener[0], n: topOpener[1] } : null, sentences: sentences.length });
1360
+ }
1361
+ if (!metrics.length)
1362
+ return [];
1363
+ // Outliers only: the LEXICON signal is mandatory (uniform sentences + repeated openers
1364
+ // without a single slop phrase is template chrome, not slop — live-verified on simracing),
1365
+ // plus ≥1 corroborating signal, AND composite score in the top decile of pages that showed
1366
+ // any signal at all.
1367
+ const sorted = metrics.map(m => m.score).sort((a, b) => a - b);
1368
+ const p90 = sorted[Math.min(sorted.length - 1, Math.floor(sorted.length * 0.9))];
1369
+ return metrics
1370
+ .filter(m => m.signals >= 2 && m.hits.size >= 3 && m.density >= 2.5 && m.score >= p90)
1371
+ .sort((a, b) => b.score - a.score).slice(0, 30)
1372
+ .map(m => ({ urlKey: m.urlKey, evidence: {
1373
+ slopPhrases: [...m.hits.entries()].sort((a, b) => b[1] - a[1]).slice(0, 6).map(([p, n]) => `${p} ×${n}`),
1374
+ lexDensityPer1000: Math.round(m.density * 10) / 10,
1375
+ sentenceLengthCV: m.cv !== null ? Math.round(m.cv * 100) / 100 : null,
1376
+ repeatedOpener: m.opener ? `"${m.opener.text}…" opens ${m.opener.n} sections` : null,
1377
+ sentencesMeasured: m.sentences,
1378
+ note: 'multiple statistical slop signals — verify by reading before rewriting',
1379
+ } }));
1380
+ },
1381
+ },
1382
+ {
1383
+ // The guide's "unique point of view based on personal experience or expertise", cut down to its
1384
+ // deterministically checkable slice: an Article/BlogPosting that carries schema yet names no
1385
+ // author. Attribution is the machine-readable experience/expertise signal; a byline-less article
1386
+ // is the commodity-content default. Only fires where Article schema EXISTS (no schema at all is
1387
+ // schema-opportunity territory, not this check).
1388
+ id: 'article-no-author', category: 'schema', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
1389
+ title: 'Article schema with no author attribution', fix: 'This page marks itself up as an Article/BlogPosting but declares no author. Google’s AI-search guidance rewards content with a demonstrable first-hand point of view — add `author` (a Person with a real name, ideally linking to an author page) to the Article schema and a visible byline to match.',
1390
+ run: (c) => {
1391
+ const out = [];
1392
+ for (const r of rows(c, `SELECT url_key urlKey, json_ld jsonLd FROM pages WHERE status_code=200 AND indexable=1 AND json_ld LIKE '%Article%'`)) {
1393
+ for (const n of parseJsonLdNodes(r.jsonLd)) {
1394
+ const t = nodeType(n);
1395
+ if (!t || !/^(article|blogposting|newsarticle|techarticle|scholarlyarticle)$/i.test(t))
1396
+ continue;
1397
+ const a = n.author;
1398
+ const named = (v) => !!v && (typeof v === 'string' ? v.trim().length > 0
1399
+ : Array.isArray(v) ? v.some(named) : typeof v === 'object' && typeof v.name === 'string' && v.name.trim().length > 0);
1400
+ if (!named(a)) {
1401
+ out.push({ urlKey: r.urlKey, evidence: { schemaType: t, author: a ?? null, note: 'Article markup present, author absent or unnamed' } });
1402
+ break;
1403
+ }
1404
+ }
1405
+ }
1406
+ return out;
1407
+ },
1408
+ },
1409
+ ];
1410
+ const sitemapHasRows = (c) => (c.db.prepare('SELECT COUNT(*) n FROM sitemap_urls').get().n) > 0;
1411
+ let _hreflangCache = null;
1412
+ function hreflangFindings(ctx) {
1413
+ if (_hreflangCache && _hreflangCache.db === ctx.db)
1414
+ return _hreflangCache.out;
1415
+ const hostForm = ctx.db.prepare(`SELECT host_form h FROM property_meta LIMIT 1`).get()?.h;
1416
+ const opts = { hostForm: hostForm ?? 'asis' };
1417
+ const pageRows = ctx.db.prepare(`SELECT url_key, url, status_code, noindex, hreflang FROM pages WHERE hreflang IS NOT NULL AND hreflang != ''`).all();
1418
+ const status = new Map();
1419
+ for (const p of ctx.db.prepare(`SELECT url_key, status_code, noindex FROM pages`).all())
1420
+ status.set(p.url_key, { status: p.status_code, noindex: p.noindex });
1421
+ // declared[sourceKey] = set of internal alternate targetKeys (excluding self)
1422
+ const declared = new Map();
1423
+ const parsed = [];
1424
+ for (const p of pageRows) {
1425
+ let arr;
1426
+ try {
1427
+ arr = JSON.parse(p.hreflang);
1428
+ }
1429
+ catch {
1430
+ continue;
1431
+ }
1432
+ const targets = [];
1433
+ for (const { lang, href } of arr) {
1434
+ if (!href)
1435
+ continue;
1436
+ let key;
1437
+ try {
1438
+ key = urlKey(new URL(href, p.url).toString(), opts);
1439
+ }
1440
+ catch {
1441
+ continue;
1442
+ }
1443
+ if (key === p.url_key)
1444
+ continue; // self-reference
1445
+ targets.push({ key, lang: (lang || '').toLowerCase() });
1446
+ }
1447
+ parsed.push({ srcKey: p.url_key, targets });
1448
+ // A return link via x-default is a VALID return tag (common when x-default is the
1449
+ // homepage) — include all targets here; x-default is only excluded as a reciprocation
1450
+ // *requirement* in the loop below, never as a way of satisfying one.
1451
+ declared.set(p.url_key, new Set(targets.map(t => t.key)));
1452
+ }
1453
+ const broken = [];
1454
+ const noReturn = [];
1455
+ for (const { srcKey, targets } of parsed) {
1456
+ const brokenTargets = [];
1457
+ const missingReturn = [];
1458
+ for (const t of targets) {
1459
+ const st = status.get(t.key);
1460
+ if (st && (st.status !== 200 || st.noindex === 1))
1461
+ brokenTargets.push(t.key);
1462
+ // reciprocation only for internal, live (200) targets we crawled, excluding x-default
1463
+ // (a broken target is already reported by broken-hreflang-target — don't double-flag)
1464
+ if (t.lang !== 'x-default' && st && st.status === 200 && st.noindex !== 1 && !(declared.get(t.key)?.has(srcKey)))
1465
+ missingReturn.push(t.key);
1466
+ }
1467
+ if (brokenTargets.length)
1468
+ broken.push({ urlKey: srcKey, evidence: { brokenAlternates: brokenTargets.slice(0, 10), count: brokenTargets.length } });
1469
+ if (missingReturn.length)
1470
+ noReturn.push({ urlKey: srcKey, evidence: { missingReturnFrom: missingReturn.slice(0, 10), count: missingReturn.length } });
1471
+ }
1472
+ _hreflangCache = { db: ctx.db, out: { broken, noReturn } };
1473
+ return { broken, noReturn };
1474
+ }
1475
+ //# sourceMappingURL=checks.js.map