@houtini/seo-audit-console 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +92 -0
- package/README.md +211 -0
- package/dist/audit/checks.d.ts +35 -0
- package/dist/audit/checks.d.ts.map +1 -0
- package/dist/audit/checks.js +1475 -0
- package/dist/audit/checks.js.map +1 -0
- package/dist/audit/drift.d.ts +40 -0
- package/dist/audit/drift.d.ts.map +1 -0
- package/dist/audit/drift.js +148 -0
- package/dist/audit/drift.js.map +1 -0
- package/dist/audit/engine.d.ts +33 -0
- package/dist/audit/engine.d.ts.map +1 -0
- package/dist/audit/engine.js +186 -0
- package/dist/audit/engine.js.map +1 -0
- package/dist/audit/opportunities.d.ts +23 -0
- package/dist/audit/opportunities.d.ts.map +1 -0
- package/dist/audit/opportunities.js +149 -0
- package/dist/audit/opportunities.js.map +1 -0
- package/dist/audit/report.d.ts +11 -0
- package/dist/audit/report.d.ts.map +1 -0
- package/dist/audit/report.js +59 -0
- package/dist/audit/report.js.map +1 -0
- package/dist/audit/schema-validate.d.ts +35 -0
- package/dist/audit/schema-validate.d.ts.map +1 -0
- package/dist/audit/schema-validate.js +293 -0
- package/dist/audit/schema-validate.js.map +1 -0
- package/dist/audit/templates.d.ts +43 -0
- package/dist/audit/templates.d.ts.map +1 -0
- package/dist/audit/templates.js +129 -0
- package/dist/audit/templates.js.map +1 -0
- package/dist/audit/topicGaps.d.ts +42 -0
- package/dist/audit/topicGaps.d.ts.map +1 -0
- package/dist/audit/topicGaps.js +181 -0
- package/dist/audit/topicGaps.js.map +1 -0
- package/dist/core/AuditDatabase.d.ts +33 -0
- package/dist/core/AuditDatabase.d.ts.map +1 -0
- package/dist/core/AuditDatabase.js +481 -0
- package/dist/core/AuditDatabase.js.map +1 -0
- package/dist/core/Backlinks.d.ts +33 -0
- package/dist/core/Backlinks.d.ts.map +1 -0
- package/dist/core/Backlinks.js +110 -0
- package/dist/core/Backlinks.js.map +1 -0
- package/dist/core/Crawler.d.ts +23 -0
- package/dist/core/Crawler.d.ts.map +1 -0
- package/dist/core/Crawler.js +588 -0
- package/dist/core/Crawler.js.map +1 -0
- package/dist/core/DataForSeoClient.d.ts +86 -0
- package/dist/core/DataForSeoClient.d.ts.map +1 -0
- package/dist/core/DataForSeoClient.js +232 -0
- package/dist/core/DataForSeoClient.js.map +1 -0
- package/dist/core/Entities.d.ts +23 -0
- package/dist/core/Entities.d.ts.map +1 -0
- package/dist/core/Entities.js +62 -0
- package/dist/core/Entities.js.map +1 -0
- package/dist/core/GscClient.d.ts +22 -0
- package/dist/core/GscClient.d.ts.map +1 -0
- package/dist/core/GscClient.js +93 -0
- package/dist/core/GscClient.js.map +1 -0
- package/dist/core/GscSync.d.ts +20 -0
- package/dist/core/GscSync.d.ts.map +1 -0
- package/dist/core/GscSync.js +133 -0
- package/dist/core/GscSync.js.map +1 -0
- package/dist/core/JobManager.d.ts +30 -0
- package/dist/core/JobManager.d.ts.map +1 -0
- package/dist/core/JobManager.js +68 -0
- package/dist/core/JobManager.js.map +1 -0
- package/dist/core/RankTracker.d.ts +25 -0
- package/dist/core/RankTracker.d.ts.map +1 -0
- package/dist/core/RankTracker.js +78 -0
- package/dist/core/RankTracker.js.map +1 -0
- package/dist/core/Refresh.d.ts +32 -0
- package/dist/core/Refresh.d.ts.map +1 -0
- package/dist/core/Refresh.js +73 -0
- package/dist/core/Refresh.js.map +1 -0
- package/dist/core/UrlInspector.d.ts +22 -0
- package/dist/core/UrlInspector.d.ts.map +1 -0
- package/dist/core/UrlInspector.js +92 -0
- package/dist/core/UrlInspector.js.map +1 -0
- package/dist/core/WikidataClient.d.ts +19 -0
- package/dist/core/WikidataClient.d.ts.map +1 -0
- package/dist/core/WikidataClient.js +59 -0
- package/dist/core/WikidataClient.js.map +1 -0
- package/dist/core/agentReadiness.d.ts +34 -0
- package/dist/core/agentReadiness.d.ts.map +1 -0
- package/dist/core/agentReadiness.js +119 -0
- package/dist/core/agentReadiness.js.map +1 -0
- package/dist/core/ctrModel.d.ts +2 -0
- package/dist/core/ctrModel.d.ts.map +1 -0
- package/dist/core/ctrModel.js +6 -0
- package/dist/core/ctrModel.js.map +1 -0
- package/dist/core/dashboardData.d.ts +312 -0
- package/dist/core/dashboardData.d.ts.map +1 -0
- package/dist/core/dashboardData.js +550 -0
- package/dist/core/dashboardData.js.map +1 -0
- package/dist/core/dataStorage.d.ts +42 -0
- package/dist/core/dataStorage.d.ts.map +1 -0
- package/dist/core/dataStorage.js +193 -0
- package/dist/core/dataStorage.js.map +1 -0
- package/dist/core/draftBrief.d.ts +27 -0
- package/dist/core/draftBrief.d.ts.map +1 -0
- package/dist/core/draftBrief.js +69 -0
- package/dist/core/draftBrief.js.map +1 -0
- package/dist/core/extract.d.ts +67 -0
- package/dist/core/extract.d.ts.map +1 -0
- package/dist/core/extract.js +262 -0
- package/dist/core/extract.js.map +1 -0
- package/dist/core/gscFreshness.d.ts +18 -0
- package/dist/core/gscFreshness.d.ts.map +1 -0
- package/dist/core/gscFreshness.js +32 -0
- package/dist/core/gscFreshness.js.map +1 -0
- package/dist/core/linkGraph.d.ts +19 -0
- package/dist/core/linkGraph.d.ts.map +1 -0
- package/dist/core/linkGraph.js +125 -0
- package/dist/core/linkGraph.js.map +1 -0
- package/dist/core/passageScore.d.ts +19 -0
- package/dist/core/passageScore.d.ts.map +1 -0
- package/dist/core/passageScore.js +59 -0
- package/dist/core/passageScore.js.map +1 -0
- package/dist/core/paths.d.ts +5 -0
- package/dist/core/paths.d.ts.map +1 -0
- package/dist/core/paths.js +15 -0
- package/dist/core/paths.js.map +1 -0
- package/dist/core/queryData.d.ts +35 -0
- package/dist/core/queryData.d.ts.map +1 -0
- package/dist/core/queryData.js +200 -0
- package/dist/core/queryData.js.map +1 -0
- package/dist/core/reranker.d.ts +8 -0
- package/dist/core/reranker.d.ts.map +1 -0
- package/dist/core/reranker.js +68 -0
- package/dist/core/reranker.js.map +1 -0
- package/dist/core/robots.d.ts +11 -0
- package/dist/core/robots.d.ts.map +1 -0
- package/dist/core/robots.js +76 -0
- package/dist/core/robots.js.map +1 -0
- package/dist/core/sitemap.d.ts +15 -0
- package/dist/core/sitemap.d.ts.map +1 -0
- package/dist/core/sitemap.js +100 -0
- package/dist/core/sitemap.js.map +1 -0
- package/dist/core/sql.d.ts +6 -0
- package/dist/core/sql.d.ts.map +1 -0
- package/dist/core/sql.js +6 -0
- package/dist/core/sql.js.map +1 -0
- package/dist/core/types.d.ts +22 -0
- package/dist/core/types.d.ts.map +1 -0
- package/dist/core/types.js +2 -0
- package/dist/core/types.js.map +1 -0
- package/dist/core/url-key.d.ts +39 -0
- package/dist/core/url-key.d.ts.map +1 -0
- package/dist/core/url-key.js +106 -0
- package/dist/core/url-key.js.map +1 -0
- package/dist/generators/index.d.ts +45 -0
- package/dist/generators/index.d.ts.map +1 -0
- package/dist/generators/index.js +184 -0
- package/dist/generators/index.js.map +1 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +10 -0
- package/dist/index.js.map +1 -0
- package/dist/server.d.ts +7 -0
- package/dist/server.d.ts.map +1 -0
- package/dist/server.js +1387 -0
- package/dist/server.js.map +1 -0
- package/dist/src/ui/dashboard.html +347 -0
- package/dist/src/ui/sync-progress.html +104 -0
- package/package.json +101 -0
- package/server.json +57 -0
|
@@ -0,0 +1,1475 @@
|
|
|
1
|
+
import { validateJsonLdColumn } from './schema-validate.js';
|
|
2
|
+
import { urlKey } from '../core/url-key.js';
|
|
3
|
+
import { parseJsonLdNodes, nodeType } from './templates.js';
|
|
4
|
+
import { expectedCtr } from '../core/ctrModel.js';
|
|
5
|
+
import { latestTwoCrawls } from './drift.js';
|
|
6
|
+
import { HTML_CT } from '../core/sql.js';
|
|
7
|
+
const rows = (ctx, sql, ...args) => ctx.db.prepare(sql).all(...args);
|
|
8
|
+
// Streaming variant — yields one row at a time instead of materialising the whole result set. Use for
|
|
9
|
+
// checks that scan a large/fat column (e.g. body_chunks) and only need independent per-row work, so a
|
|
10
|
+
// big content site doesn't load every page's body text into one array.
|
|
11
|
+
const iterRows = (ctx, sql, ...args) => ctx.db.prepare(sql).iterate(...args);
|
|
12
|
+
// d is the finalised GSC date (partial trailing days already trimmed upstream); cap the window at
|
|
13
|
+
// it so checks never count unfinalised days. winPrev already caps below d, so it's unaffected.
|
|
14
|
+
const win = (d) => `date > date('${d}', '-28 days') AND date <= '${d}'`;
|
|
15
|
+
// Prior 28-day window (the 28 days BEFORE the current window) — for period-over-period checks.
|
|
16
|
+
const winPrev = (d) => `date <= date('${d}', '-28 days') AND date > date('${d}', '-56 days')`;
|
|
17
|
+
// SQL clause to drop branded queries (whole-token match). Branded multi-URL ranking is sitelinks,
|
|
18
|
+
// not cannibalisation; branded top-queries aren't anchor targets. The brand is re-sanitised to
|
|
19
|
+
// alnum HERE (not trusting the caller) so inlining it into SQL is unconditionally injection-safe.
|
|
20
|
+
const brandExcl = (c) => {
|
|
21
|
+
const b = (c.brand ?? '').replace(/[^a-z0-9]/g, '');
|
|
22
|
+
return b ? `AND (' ' || LOWER(query) || ' ') NOT LIKE '% ${b} %'` : '';
|
|
23
|
+
};
|
|
24
|
+
// Days of GSC history actually held — period-over-period checks need enough span to be meaningful.
|
|
25
|
+
const spanDays = (c) => {
|
|
26
|
+
// Span must be measured up to the FINALISED max date the windows actually key off
|
|
27
|
+
// (gscMaxDate is trimmed ~3 days below raw MAX(date)) — measuring the raw span lets the
|
|
28
|
+
// ≥56-day guard pass while the previous-28d window still reaches before MIN(date),
|
|
29
|
+
// undercounting the prior period (rising-pages FPs, traffic-decay FNs).
|
|
30
|
+
const r = c.db.prepare(`SELECT julianday(?) - julianday(MIN(date)) d FROM search_analytics`).get(c.gscMaxDate ?? null);
|
|
31
|
+
return r?.d ?? 0;
|
|
32
|
+
};
|
|
33
|
+
// Newest dateModified/datePublished anywhere in a page's JSON-LD (recurses @graph/arrays) — the
|
|
34
|
+
// effective "last meaningfully updated" date. json_ld is a JSON array of raw block strings.
|
|
35
|
+
const newestSchemaDate = (jl) => {
|
|
36
|
+
if (!jl)
|
|
37
|
+
return null;
|
|
38
|
+
let best = -Infinity, bestStr = null;
|
|
39
|
+
const scan = (o) => {
|
|
40
|
+
if (!o || typeof o !== 'object')
|
|
41
|
+
return;
|
|
42
|
+
for (const key of ['dateModified', 'datePublished']) {
|
|
43
|
+
const v = o[key];
|
|
44
|
+
if (typeof v === 'string') {
|
|
45
|
+
const t = Date.parse(v);
|
|
46
|
+
if (!isNaN(t) && t > best) {
|
|
47
|
+
best = t;
|
|
48
|
+
bestStr = v;
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
for (const k in o)
|
|
53
|
+
if (o[k] && typeof o[k] === 'object')
|
|
54
|
+
scan(o[k]);
|
|
55
|
+
};
|
|
56
|
+
try {
|
|
57
|
+
for (const block of JSON.parse(jl)) {
|
|
58
|
+
try {
|
|
59
|
+
scan(JSON.parse(block));
|
|
60
|
+
}
|
|
61
|
+
catch { /* skip block */ }
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
catch { /* skip */ }
|
|
65
|
+
return bestStr;
|
|
66
|
+
};
|
|
67
|
+
// Paginated archive URLs (/page/2, ?page=3, ?paged=2). They legitimately share titles/metas with
|
|
68
|
+
// page 1 and are intentionally absent from sitemaps, so they must NOT generate duplicate-title /
|
|
69
|
+
// duplicate-meta / missing-meta / not-in-sitemap false positives. Pass the column reference
|
|
70
|
+
// (e.g. 'url_key' or 'p.url_key') so it composes with table aliases.
|
|
71
|
+
const notPagination = (col = 'url_key') =>
|
|
72
|
+
// Anchor the param name to ?/& — a bare LIKE '%page=%' also matches per_page=/on_page=/
|
|
73
|
+
// homepage=, wrongly exempting those URLs from duplicate-title/meta/sitemap checks.
|
|
74
|
+
`${col} NOT GLOB '*/page/[0-9]*' AND ${col} NOT GLOB '*[?&]page=[0-9]*' AND ${col} NOT GLOB '*[?&]paged=[0-9]*'`;
|
|
75
|
+
// Significant query terms (drop stopwords; keep ≥2 chars so "vr"/"pc"/"ai" count).
|
|
76
|
+
const STOP = new Set(['the', 'a', 'an', 'and', 'or', 'of', 'for', 'to', 'in', 'on', 'with', 'your', 'you', 'is', 'are', 'best', 'how', 'what', 'vs', 'why', 'can']);
|
|
77
|
+
const terms = (s) => (s || '').toLowerCase().replace(/[^a-z0-9 ]+/g, ' ').split(/\s+/).filter(t => t.length >= 2 && !STOP.has(t));
|
|
78
|
+
// A query term is "present" in the title if a TITLE WORD matches it (exact, or a
|
|
79
|
+
// plural/stem prefix either way for ≥4-char tokens), or the space-collapsed title
|
|
80
|
+
// contains it (for multi-word brands like "sync mesh" ≈ "syncmesh", ≥5 chars).
|
|
81
|
+
// Word-level avoids substring false-matches (e.g. "art" inside "smart").
|
|
82
|
+
const titleHasTerm = (title, term) => {
|
|
83
|
+
const t = title.toLowerCase();
|
|
84
|
+
const words = t.split(/[^a-z0-9]+/).filter(Boolean);
|
|
85
|
+
if (words.some(w => w === term || (term.length >= 4 && w.startsWith(term)) || (w.length >= 4 && term.startsWith(w))))
|
|
86
|
+
return true;
|
|
87
|
+
return term.length >= 5 && t.replace(/[^a-z0-9]+/g, '').includes(term);
|
|
88
|
+
};
|
|
89
|
+
// Validate captured JSON-LD per page, keeping findings whose issue-kinds match `kinds`.
|
|
90
|
+
function schemaFindings(ctx, kinds) {
|
|
91
|
+
const set = new Set(kinds);
|
|
92
|
+
const out = [];
|
|
93
|
+
for (const r of rows(ctx, `SELECT url_key urlKey, json_ld jsonLd FROM pages WHERE status_code=200 AND json_ld IS NOT NULL AND json_ld != ''`)) {
|
|
94
|
+
const hits = validateJsonLdColumn(r.jsonLd).filter(i => set.has(i.kind));
|
|
95
|
+
if (hits.length)
|
|
96
|
+
out.push({ urlKey: r.urlKey, evidence: { issues: hits.map(h => ({ type: h.type, detail: h.detail, fields: h.fields })) } });
|
|
97
|
+
}
|
|
98
|
+
return out;
|
|
99
|
+
}
|
|
100
|
+
// Position→expected-CTR curve lives in core/ctrModel (shared with the dashboard). Re-export it so
|
|
101
|
+
// existing `import { expectedCtr } from './checks.js'` call sites keep working.
|
|
102
|
+
export { expectedCtr };
|
|
103
|
+
export const CHECKS = [
|
|
104
|
+
// ── On-page (crawl, deterministic) ──────────────────────────────────────
|
|
105
|
+
{
|
|
106
|
+
id: 'missing-title', category: 'onpage', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
107
|
+
title: 'Missing title tag', fix: 'Add a unique, descriptive <title> (~50–60 chars).',
|
|
108
|
+
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (title IS NULL OR TRIM(title)='')`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
|
|
109
|
+
},
|
|
110
|
+
{
|
|
111
|
+
id: 'duplicate-title', category: 'onpage', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
112
|
+
title: 'Duplicate title tag', fix: 'Make each indexable page’s title unique.',
|
|
113
|
+
run: (c) => rows(c, `SELECT url_key urlKey, title FROM pages WHERE status_code=200 AND indexable=1 AND ${notPagination()} AND title IS NOT NULL AND TRIM(title)!='' AND LOWER(TRIM(title)) IN (SELECT LOWER(TRIM(title)) FROM pages WHERE status_code=200 AND indexable=1 AND ${notPagination()} AND title IS NOT NULL GROUP BY LOWER(TRIM(title)) HAVING COUNT(*)>1)`).map(r => ({ urlKey: r.urlKey, evidence: { title: r.title } })),
|
|
114
|
+
},
|
|
115
|
+
{
|
|
116
|
+
id: 'missing-meta-description', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
117
|
+
title: 'Missing meta description', fix: 'Add a unique meta description (~120–155 chars).',
|
|
118
|
+
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND ${notPagination()} AND (meta_description IS NULL OR TRIM(meta_description)='')`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
|
|
119
|
+
},
|
|
120
|
+
{
|
|
121
|
+
id: 'missing-h1', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
122
|
+
title: 'Missing H1', fix: 'Add a single descriptive <h1>.',
|
|
123
|
+
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND (h1_count IS NULL OR h1_count=0)`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
|
|
124
|
+
},
|
|
125
|
+
{
|
|
126
|
+
id: 'multiple-h1', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
127
|
+
title: 'Multiple H1s', fix: 'Use one H1 per page.',
|
|
128
|
+
run: (c) => rows(c, `SELECT url_key urlKey, h1_count FROM pages WHERE status_code=200 AND h1_count>1`).map(r => ({ urlKey: r.urlKey, evidence: { h1Count: r.h1_count } })),
|
|
129
|
+
},
|
|
130
|
+
{
|
|
131
|
+
id: 'thin-content', category: 'content', severity: 'med', labels: ['D', 'N'], certainty: 0.6, effortBase: 5, fixType: 'per-page',
|
|
132
|
+
title: 'Thin content', fix: 'Expand or consolidate — under ~200 words of body text.',
|
|
133
|
+
run: (c) => rows(c, `SELECT url_key urlKey, word_count FROM pages WHERE status_code=200 AND indexable=1 AND word_count < 200`).map(r => ({ urlKey: r.urlKey, evidence: { wordCount: r.word_count } })),
|
|
134
|
+
},
|
|
135
|
+
// ── Indexation / crawlability ───────────────────────────────────────────
|
|
136
|
+
{
|
|
137
|
+
// RETIRED: 'canonical-mismatch' (was HIGH). A 200 page whose canonical points to a *healthy*
|
|
138
|
+
// 200 indexable URL is intentional consolidation (slug variants, category merges) — normal SEO,
|
|
139
|
+
// not an issue, yet it fired HIGH on every such page (pure noise). Every actionable case is
|
|
140
|
+
// already covered by a higher-signal check: broken-canonical-target (unhealthy target),
|
|
141
|
+
// canonical-ignored (Google ranks the non-canonical page), canonical-conflict (GSC disagrees).
|
|
142
|
+
// So plain "canonical points elsewhere" has no high-confidence residual — removed.
|
|
143
|
+
id: 'broken-internal-links', category: 'crawlability', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
|
|
144
|
+
title: 'Internal links to 4xx/5xx', fix: 'Repoint internal links to a live, canonical URL.',
|
|
145
|
+
run: (c) => rows(c, `SELECT l.target_key urlKey, p.status_code status, COUNT(DISTINCT l.source_key) sources FROM links l JOIN pages p ON p.url_key=l.target_key WHERE l.is_internal=1 AND p.status_code >= 400 AND p.status_code NOT IN (429,503) GROUP BY l.target_key`).map(r => ({ urlKey: r.urlKey, evidence: { status: r.status, linkingPages: r.sources } })),
|
|
146
|
+
},
|
|
147
|
+
// ── Extractor-dependent (images + canonical shape) ──────────────────────
|
|
148
|
+
{
|
|
149
|
+
id: 'image-alt', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
150
|
+
title: 'Images missing alt text', fix: 'Add descriptive alt text to content images (alt="" only for decorative).',
|
|
151
|
+
run: (c) => rows(c, `SELECT url_key urlKey, images_without_alt missing, image_count total FROM pages WHERE status_code=200 AND indexable=1 AND images_without_alt > 0`).map(r => ({ urlKey: r.urlKey, evidence: { missing: r.missing, total: r.total } })),
|
|
152
|
+
},
|
|
153
|
+
{
|
|
154
|
+
id: 'canonical-relative', category: 'indexation', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
155
|
+
title: 'Canonical declared as a relative URL', fix: 'Use an absolute https URL in rel=canonical — relative canonicals are error-prone.',
|
|
156
|
+
run: (c) => rows(c, `SELECT url_key urlKey, canonical_url canonical FROM pages WHERE status_code=200 AND canonical_relative=1`).map(r => ({ urlKey: r.urlKey, evidence: { canonical: r.canonical } })),
|
|
157
|
+
},
|
|
158
|
+
{
|
|
159
|
+
id: 'multiple-canonical', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
160
|
+
title: 'Multiple canonical tags', fix: 'Keep exactly one rel=canonical — conflicting canonicals let Google pick (or ignore) one.',
|
|
161
|
+
run: (c) => rows(c, `SELECT url_key urlKey, canonical_count cnt FROM pages WHERE status_code=200 AND canonical_count > 1`).map(r => ({ urlKey: r.urlKey, evidence: { canonicalCount: r.cnt } })),
|
|
162
|
+
},
|
|
163
|
+
// ── Extractor additions (CLS, headings, mixed content, directives, social) ───
|
|
164
|
+
{
|
|
165
|
+
id: 'images-missing-dimensions', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
166
|
+
// Lint, NOT a measured Core Web Vital. Missing width/height attributes only cause CLS if the
|
|
167
|
+
// CSS doesn't already reserve space — modern themes using aspect-ratio / fixed boxes have ~0
|
|
168
|
+
// measured CLS despite missing attributes. So we report it as a hygiene lint and explicitly say
|
|
169
|
+
// it's not a confirmed CWV issue; escalate only against field CLS (high-yield-cwv-fail does that).
|
|
170
|
+
title: 'Images missing width/height attributes', fix: 'Add width & height (or rely on CSS aspect-ratio) so the browser reserves space. NOTE: this is a lint — if your CSS already reserves space (aspect-ratio / fixed box) measured CLS is likely ~0 and there is nothing to fix. Confirm with field CLS before prioritising.',
|
|
171
|
+
run: (c) => rows(c, `SELECT url_key urlKey, images_missing_dimensions n, image_count total FROM pages WHERE status_code=200 AND indexable=1 AND images_missing_dimensions > 0`).map(r => ({ urlKey: r.urlKey, evidence: { missingDimensions: r.n, total: r.total, note: 'lint only — no measured CLS impact unless field data shows layout shift' } })),
|
|
172
|
+
},
|
|
173
|
+
{
|
|
174
|
+
id: 'heading-hierarchy', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
175
|
+
title: 'Skipped heading levels', fix: 'Use headings in order (don’t jump e.g. h1→h3) — keeps the document outline accessible and parseable.',
|
|
176
|
+
run: (c) => rows(c, `SELECT url_key urlKey, heading_skips n FROM pages WHERE status_code=200 AND indexable=1 AND heading_skips > 0`).map(r => ({ urlKey: r.urlKey, evidence: { skippedLevels: r.n } })),
|
|
177
|
+
},
|
|
178
|
+
{
|
|
179
|
+
id: 'mixed-content', category: 'security', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
180
|
+
title: 'Mixed content (http on https)', fix: 'Serve every subresource over https — browsers block or warn on insecure resources.',
|
|
181
|
+
run: (c) => rows(c, `SELECT url_key urlKey, mixed_content_count n FROM pages WHERE status_code=200 AND url LIKE 'https://%' AND mixed_content_count > 0`).map(r => ({ urlKey: r.urlKey, evidence: { insecureResources: r.n } })),
|
|
182
|
+
},
|
|
183
|
+
{
|
|
184
|
+
id: 'meta-nofollow', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
185
|
+
title: 'Meta robots nofollow', fix: 'Remove nofollow from meta robots unless you intend to drop all link equity from this page.',
|
|
186
|
+
run: (c) => rows(c, `SELECT url_key urlKey, robots FROM pages WHERE status_code=200 AND indexable=1 AND robots LIKE '%nofollow%'`).map(r => ({ urlKey: r.urlKey, evidence: { robots: r.robots } })),
|
|
187
|
+
},
|
|
188
|
+
{
|
|
189
|
+
id: 'missing-social-tags', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
190
|
+
title: 'No social share tags', fix: 'Add Open Graph (og:title/og:image) and/or Twitter Card tags so shared links render a rich preview.',
|
|
191
|
+
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND (og_tags IS NULL OR og_tags='') AND (twitter_tags IS NULL OR twitter_tags='')`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
|
|
192
|
+
},
|
|
193
|
+
// ── Security / war-stories (headers now captured) ───────────────────────
|
|
194
|
+
{
|
|
195
|
+
id: 'missing-hsts', category: 'security', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
196
|
+
title: 'Missing HSTS header', fix: 'Add Strict-Transport-Security with a sensible max-age.',
|
|
197
|
+
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (security_headers IS NULL OR security_headers NOT LIKE '%hsts%') LIMIT 1`).map(r => ({ urlKey: r.urlKey, evidence: { note: 'representative page; HSTS is site-wide' } })),
|
|
198
|
+
},
|
|
199
|
+
// ── Merged GSC × crawl (the differentiator) ─────────────────────────────
|
|
200
|
+
{
|
|
201
|
+
id: 'noindex-with-traffic', category: 'indexation', severity: 'crit', labels: ['D', 'G'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
202
|
+
title: 'Noindex page still getting clicks', fix: 'Remove noindex if the page should rank — it earns clicks.',
|
|
203
|
+
run: (c) => c.gscMaxDate ? rows(c, `SELECT p.url_key urlKey, SUM(sa.clicks) clicks FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key WHERE p.noindex=1 AND sa.${win(c.gscMaxDate)} GROUP BY p.url_key HAVING SUM(sa.clicks)>0`).map(r => ({ urlKey: r.urlKey, evidence: { clicks: r.clicks } })) : [],
|
|
204
|
+
},
|
|
205
|
+
{
|
|
206
|
+
id: 'orphan-with-impressions', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
207
|
+
title: 'Orphan page earning impressions', fix: 'Add internal links — Google ranks it but the site barely links to it. If it drives a large traffic share, protect it BEFORE any cleanup or migration (it is load-bearing).',
|
|
208
|
+
run: (c) => {
|
|
209
|
+
if (!c.gscMaxDate)
|
|
210
|
+
return [];
|
|
211
|
+
const total = (rows(c, `SELECT SUM(clicks) c FROM search_analytics WHERE ${win(c.gscMaxDate)}`)[0]?.c) || 1;
|
|
212
|
+
return rows(c, `SELECT p.url_key urlKey, SUM(sa.impressions) impressions, SUM(sa.clicks) clicks FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key WHERE p.inlink_count=0 AND p.indexable=1 AND sa.${win(c.gscMaxDate)} GROUP BY p.url_key HAVING SUM(sa.impressions)>0 ORDER BY clicks DESC, impressions DESC`)
|
|
213
|
+
.map(r => { const share = Math.round(r.clicks / total * 1000) / 10; return { urlKey: r.urlKey, evidence: { impressions: r.impressions, clicks: r.clicks, inlinks: 0, trafficShare: share + '%', ...(share >= 5 ? { note: `LOAD-BEARING orphan: drives ${share}% of site clicks with zero internal links — protect before any cleanup/migration` } : {}) } }; });
|
|
214
|
+
},
|
|
215
|
+
},
|
|
216
|
+
{
|
|
217
|
+
id: 'canonical-ignored', category: 'indexation', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
218
|
+
title: 'Google may be ignoring the declared canonical', fix: 'This page declares a canonical pointing elsewhere, yet Google still ranks IT (real impressions) — Google is overriding the canonical, usually because the target is weaker or internal links favour this URL. Decide which URL you actually want indexed, then align both the canonical and the internal links to it.',
|
|
219
|
+
run: (c) => c.gscMaxDate ? rows(c, `SELECT p.url_key urlKey, p.canonical_url canonical, SUM(sa.impressions) impressions, SUM(sa.clicks) clicks FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key WHERE p.canonical_key IS NOT NULL AND p.canonical_key != p.url_key AND sa.${win(c.gscMaxDate)} GROUP BY p.url_key HAVING SUM(sa.impressions) >= 50 ORDER BY impressions DESC LIMIT 40`).map(r => ({ urlKey: r.urlKey, evidence: { declaredCanonical: r.canonical, impressions: r.impressions, clicks: r.clicks, note: 'ranks despite pointing its canonical elsewhere' } })) : [],
|
|
220
|
+
},
|
|
221
|
+
{
|
|
222
|
+
id: 'indexed-junk-url', category: 'indexation', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'global',
|
|
223
|
+
title: 'Internal-search / faceted URL is indexed and ranking', fix: 'A URL whose signature is internal site-search, a faceted filter, or a tracking-param variant is earning Google impressions — it has been accidentally indexed. noindex or robots-block these and canonicalise filter URLs, so thin/duplicate pages stop bleeding index quality. (The query footprint proves it even when the DOM looks fine.)',
|
|
224
|
+
run: (c) => {
|
|
225
|
+
if (!c.gscMaxDate)
|
|
226
|
+
return [];
|
|
227
|
+
const junk = /[?&](q|s|search|keyword|orderby|sort_by|filter|variant|pf_|dppref|replytocom)=|\/search(-results)?\//i;
|
|
228
|
+
return rows(c, `SELECT page_key urlKey, SUM(impressions) impressions, SUM(clicks) clicks FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key HAVING SUM(impressions) >= 30 ORDER BY SUM(impressions) DESC`)
|
|
229
|
+
.filter(r => junk.test(r.urlKey)).slice(0, 40)
|
|
230
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { impressions: r.impressions, clicks: r.clicks, note: 'URL signature = internal search / facet / param — likely accidental indexation' } }));
|
|
231
|
+
},
|
|
232
|
+
},
|
|
233
|
+
{
|
|
234
|
+
id: 'coverage-not-indexed', category: 'indexation', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
235
|
+
title: 'Crawled but not indexed', fix: 'Investigate quality/duplication — Google crawled it and chose not to index.',
|
|
236
|
+
run: (c) => rows(c, `SELECT url_key urlKey, coverage_state state FROM url_inspection WHERE coverage_state LIKE '%not indexed%'`).map(r => ({ urlKey: r.urlKey, evidence: { coverageState: r.state } })),
|
|
237
|
+
},
|
|
238
|
+
{
|
|
239
|
+
id: 'canonical-conflict', category: 'indexation', severity: 'high', labels: ['G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
240
|
+
title: 'Google chose a different canonical', fix: 'Align your declared canonical with the page Google actually indexes.',
|
|
241
|
+
run: (c) => rows(c, `SELECT url_key urlKey, google_canonical g, user_canonical u FROM url_inspection WHERE user_canonical IS NOT NULL AND google_canonical IS NOT NULL AND google_canonical != user_canonical`).map(r => ({ urlKey: r.urlKey, evidence: { googleCanonical: r.g, userCanonical: r.u } })),
|
|
242
|
+
},
|
|
243
|
+
{
|
|
244
|
+
id: 'striking-distance', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
245
|
+
title: 'Striking-distance query (page 2)', fix: 'Small on-page + internal-link push could reach page 1.',
|
|
246
|
+
run: (c) => c.gscMaxDate ? rows(c, `SELECT page_key urlKey, query, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) position, SUM(impressions) impressions FROM search_analytics WHERE query IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY query, page_key HAVING SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0)>10 AND SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0)<=20 AND SUM(impressions)>=20 ORDER BY impressions DESC LIMIT 50`).map(r => ({ urlKey: r.urlKey, evidence: { query: r.query, position: Math.round(r.position * 10) / 10, impressions: r.impressions } })) : [],
|
|
247
|
+
},
|
|
248
|
+
// ── Additions from industry checklist review (buildable on current data) ──
|
|
249
|
+
{
|
|
250
|
+
id: 'title-too-long', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
251
|
+
title: 'Title over ~60 chars', fix: 'Trim the title so the primary keyword sits within ~60 chars.',
|
|
252
|
+
run: (c) => rows(c, `SELECT url_key urlKey, title_length len FROM pages WHERE status_code=200 AND indexable=1 AND title_length > 60`).map(r => ({ urlKey: r.urlKey, evidence: { titleLength: r.len } })),
|
|
253
|
+
},
|
|
254
|
+
{
|
|
255
|
+
id: 'meta-description-length', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
256
|
+
title: 'Meta description over ~160 chars', fix: 'Tighten to ~150–160 chars so it isn’t truncated.',
|
|
257
|
+
run: (c) => rows(c, `SELECT url_key urlKey, meta_description_length len FROM pages WHERE status_code=200 AND indexable=1 AND meta_description_length > 160`).map(r => ({ urlKey: r.urlKey, evidence: { length: r.len } })),
|
|
258
|
+
},
|
|
259
|
+
{
|
|
260
|
+
id: 'non-https', category: 'security', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
|
|
261
|
+
title: 'Page served over HTTP', fix: 'Serve over HTTPS and 301 the HTTP version.',
|
|
262
|
+
run: (c) => rows(c, `SELECT url_key urlKey, url FROM pages WHERE status_code=200 AND url LIKE 'http://%'`).map(r => ({ urlKey: r.urlKey, evidence: { url: r.url } })),
|
|
263
|
+
},
|
|
264
|
+
{
|
|
265
|
+
id: 'redirect-chain', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
|
|
266
|
+
title: 'Redirect chain (2+ hops)', fix: 'Collapse to a single hop to the final URL.',
|
|
267
|
+
// Count hops in SQL (json_array_length, guarded by json_valid) and filter to >=2 there — so we
|
|
268
|
+
// never pull every redirect-bearing page into JS just to count + drop most of them.
|
|
269
|
+
run: (c) => rows(c, `SELECT url_key urlKey, json_array_length(redirects) hops FROM pages
|
|
270
|
+
WHERE redirects IS NOT NULL AND json_valid(redirects) AND json_array_length(redirects) >= 2`)
|
|
271
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { hops: r.hops } })),
|
|
272
|
+
},
|
|
273
|
+
{
|
|
274
|
+
id: 'internal-links-to-redirects', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
|
|
275
|
+
title: 'Internal links pointing through redirects', fix: 'Repoint internal links to the final URL (saves crawl + equity).',
|
|
276
|
+
// Flag a link ONLY if the raw href it uses is itself a redirect source — i.e. that exact URL
|
|
277
|
+
// appears as a `from` hop in some redirect chain. The old query flagged any link whose target
|
|
278
|
+
// page merely HAD a `redirects` entry, which fired site-wide on www-canonical properties: every
|
|
279
|
+
// page records the seed apex→www hop, yet the actual hrefs already use the final (www) URL and
|
|
280
|
+
// never redirect. Matching the href (fragment-stripped) against the real redirect-source set is
|
|
281
|
+
// robust to that artefact. Kept entirely in SQL (json_each over pages.redirects) so we never
|
|
282
|
+
// materialise the whole links table in JS — the links table is the largest in the DB.
|
|
283
|
+
run: (c) => {
|
|
284
|
+
return rows(c, `
|
|
285
|
+
WITH redir_src AS (
|
|
286
|
+
-- Feed json_each a CASE that yields '[]' for any null/invalid value, so it never receives
|
|
287
|
+
-- malformed JSON regardless of how SQLite orders the scan vs the WHERE (a plain WHERE
|
|
288
|
+
-- json_valid() guard gets defeated by subquery flattening). Mirrors the old JS try/catch.
|
|
289
|
+
SELECT DISTINCT json_extract(j.value, '$.from') src
|
|
290
|
+
FROM pages, json_each(CASE WHEN json_valid(pages.redirects) THEN pages.redirects ELSE '[]' END) j
|
|
291
|
+
WHERE pages.redirects IS NOT NULL AND pages.redirects <> '[]' AND j.type = 'object'
|
|
292
|
+
)
|
|
293
|
+
SELECT l.target_key urlKey, COUNT(DISTINCT l.source_key) sources
|
|
294
|
+
FROM links l
|
|
295
|
+
JOIN redir_src r ON r.src = (CASE WHEN instr(l.target_url, '#') > 0
|
|
296
|
+
THEN substr(l.target_url, 1, instr(l.target_url, '#') - 1)
|
|
297
|
+
ELSE l.target_url END)
|
|
298
|
+
WHERE l.is_internal = 1
|
|
299
|
+
GROUP BY l.target_key`).map(r => ({ urlKey: r.urlKey, evidence: { linkingPages: r.sources } }));
|
|
300
|
+
},
|
|
301
|
+
},
|
|
302
|
+
{
|
|
303
|
+
id: 'missing-structured-data', category: 'schema', severity: 'low', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
304
|
+
title: 'No structured data', fix: 'Add relevant JSON-LD (Article, Product, Organization…).',
|
|
305
|
+
// Only "no structured data" if there's no JSON-LD AND no Microdata/RDFa either — else a
|
|
306
|
+
// page using valid Microdata (common on older themes) is falsely flagged.
|
|
307
|
+
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND (json_ld IS NULL OR json_ld='') AND COALESCE(has_microdata,0)=0 AND COALESCE(has_rdfa,0)=0`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
|
|
308
|
+
},
|
|
309
|
+
// ── Schema validation (validate captured json_ld vs maintained Rich-Results map) ──
|
|
310
|
+
{
|
|
311
|
+
id: 'invalid-schema', category: 'schema', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
312
|
+
title: 'Invalid structured data', fix: 'Fix the JSON-LD so each block parses and carries @context (https://schema.org) + a valid @type.',
|
|
313
|
+
run: (c) => schemaFindings(c, ['parse', 'context', 'type']),
|
|
314
|
+
},
|
|
315
|
+
{
|
|
316
|
+
id: 'missing-required-fields', category: 'schema', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
317
|
+
title: 'Structured data missing required fields', fix: 'Add the Google-required properties for the detected schema type (cited per finding).',
|
|
318
|
+
run: (c) => schemaFindings(c, ['required']),
|
|
319
|
+
},
|
|
320
|
+
{
|
|
321
|
+
id: 'schema-value-errors', category: 'schema', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
322
|
+
title: 'Structured-data value errors', fix: 'Use absolute, indexable image/URL values and ISO-8601 dates in JSON-LD.',
|
|
323
|
+
run: (c) => schemaFindings(c, ['value']),
|
|
324
|
+
},
|
|
325
|
+
{
|
|
326
|
+
id: 'forbidden-schema', category: 'schema', severity: 'high', labels: ['D', 'N'], certainty: 0.7, effortBase: 1, fixType: 'per-page',
|
|
327
|
+
title: 'Restricted schema type in use', fix: 'Remove FAQPage/HowTo markup unless the page qualifies for the narrow remaining eligibility — it risks no benefit or a manual action.',
|
|
328
|
+
run: (c) => schemaFindings(c, ['forbidden']),
|
|
329
|
+
},
|
|
330
|
+
{
|
|
331
|
+
id: 'missing-viewport', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
|
|
332
|
+
title: 'Missing viewport meta (mobile)', fix: 'Add <meta name="viewport" content="width=device-width, initial-scale=1">.',
|
|
333
|
+
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (viewport IS NULL OR viewport='')`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
|
|
334
|
+
},
|
|
335
|
+
{
|
|
336
|
+
id: 'keyword-cannibalisation', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
337
|
+
title: 'Keyword cannibalisation', fix: 'Consolidate or differentiate — multiple URLs compete for one query.',
|
|
338
|
+
// A URL only "competes" if it ranks for the query (impression-weighted pos < 20) AND holds a
|
|
339
|
+
// non-trivial share of the leader's impressions — ≥10% of the leader OR ≥500 impressions in its
|
|
340
|
+
// own right, and ≥10 impressions minimum. Without that floor, incidental long-tail appearances
|
|
341
|
+
// (a page picking up 1–7 impressions for the query) counted as competitors: e.g. "best vr headset"
|
|
342
|
+
// reported 10 URLs when ONE page held 59,570 impressions at pos 1.1 and the other nine had 1–7
|
|
343
|
+
// each — not cannibalisation, Google had decided. The absolute ≥500 backstop keeps a genuine
|
|
344
|
+
// mid-volume rival under a dominant leader (e.g. 60k leader + a real 4k second page = 6.7%, below
|
|
345
|
+
// the 10% bar) from being silently dropped. We also exclude "dominance" where the best pages both
|
|
346
|
+
// sit at pos 1–2 (indented/double results are good). Branded queries are dropped via brandExcl.
|
|
347
|
+
run: (c) => {
|
|
348
|
+
if (!c.gscMaxDate)
|
|
349
|
+
return [];
|
|
350
|
+
const flagged = rows(c, `
|
|
351
|
+
WITH per_page AS (
|
|
352
|
+
SELECT query, page_key, SUM(clicks) clicks, SUM(impressions) impressions,
|
|
353
|
+
SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos
|
|
354
|
+
FROM search_analytics
|
|
355
|
+
WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} ${brandExcl(c)}
|
|
356
|
+
GROUP BY query, page_key
|
|
357
|
+
HAVING pos < 20 AND SUM(impressions) >= 10
|
|
358
|
+
),
|
|
359
|
+
ranked AS (SELECT *, MAX(impressions) OVER (PARTITION BY query) topImpr FROM per_page),
|
|
360
|
+
competing AS (SELECT * FROM ranked WHERE impressions >= topImpr * 0.1 OR impressions >= 500)
|
|
361
|
+
SELECT query, COUNT(*) urls, SUM(clicks) clicks, SUM(impressions) impressions, MIN(pos) bestPos, MAX(pos) worstPos,
|
|
362
|
+
GROUP_CONCAT(page_key, char(31)) pk, GROUP_CONCAT(CAST(ROUND(impressions) AS INT), char(31)) im, GROUP_CONCAT(ROUND(pos,1), char(31)) ps
|
|
363
|
+
FROM competing GROUP BY query
|
|
364
|
+
HAVING COUNT(*) >= 2 AND SUM(impressions) >= 50 AND NOT (MAX(pos) <= 2)
|
|
365
|
+
ORDER BY impressions DESC LIMIT 40`);
|
|
366
|
+
const titleOf = c.db.prepare('SELECT title FROM pages WHERE url_key = ?');
|
|
367
|
+
return flagged.map(r => {
|
|
368
|
+
const keys = String(r.pk || '').split('\x1f'), imps = String(r.im || '').split('\x1f'), poss = String(r.ps || '').split('\x1f');
|
|
369
|
+
const items = keys.map((u, i) => ({ url: u, impressions: Number(imps[i]) || 0, position: Number(poss[i]) || 0, title: titleOf.get(u)?.title ?? null }))
|
|
370
|
+
.sort((a, b) => b.impressions - a.impressions);
|
|
371
|
+
// Differentiation signal: do the top-2 competing pages share significant title terms beyond the
|
|
372
|
+
// query itself? If not (and both are titled), they're likely intentionally distinct pages (e.g.
|
|
373
|
+
// pairwise comparisons), not true duplicates competing for one intent — annotate, don't suppress.
|
|
374
|
+
const q = new Set(terms(r.query));
|
|
375
|
+
const sig = items.slice(0, 2).map(it => terms(it.title ?? '').filter((w) => !q.has(w)));
|
|
376
|
+
const differentiated = sig.length === 2 && items[0].title != null && items[1].title != null
|
|
377
|
+
&& sig[0].filter((w) => sig[1].includes(w)).length < 2;
|
|
378
|
+
const ev = {
|
|
379
|
+
query: r.query, competingUrls: r.urls, clicks: r.clicks, impressions: r.impressions,
|
|
380
|
+
positions: `${Math.round(r.bestPos * 10) / 10}–${Math.round(r.worstPos * 10) / 10}`,
|
|
381
|
+
urls: items.slice(0, 4).map(it => ({ url: it.url, impressions: it.impressions, position: it.position, title: it.title })),
|
|
382
|
+
};
|
|
383
|
+
if (differentiated)
|
|
384
|
+
ev.note = 'competing URLs have distinct titles — likely intentional differentiation (e.g. pairwise comparisons), not true cannibalisation; verify before consolidating';
|
|
385
|
+
return { urlKey: null, evidence: ev };
|
|
386
|
+
});
|
|
387
|
+
},
|
|
388
|
+
},
|
|
389
|
+
{
|
|
390
|
+
id: 'ctr-below-expected', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
391
|
+
title: 'CTR far below position-expected', fix: 'Rewrite title/meta — ranking well but under-clicked (snippet opportunity).',
|
|
392
|
+
// High-confidence floor: ≥500 impressions/28d. A title/meta rewrite (HIGH, ~3h) is only
|
|
393
|
+
// worth flagging where the snippet earns enough visibility for a CTR lift to pay back — a
|
|
394
|
+
// 100-impression page at 1% vs 3% expected is a 2-clicks gap, not a HIGH issue.
|
|
395
|
+
run: (c) => c.gscMaxDate ? rows(c, `SELECT page_key urlKey, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) position, SUM(clicks) clicks, SUM(impressions) impressions FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key HAVING SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) <= 10 AND SUM(impressions) >= 500`)
|
|
396
|
+
.map(r => { const ctr = r.clicks / r.impressions; const exp = expectedCtr(r.position); return { urlKey: r.urlKey, ctr, exp, position: r.position, impressions: r.impressions }; })
|
|
397
|
+
.filter(x => x.ctr < x.exp * 0.5)
|
|
398
|
+
.map(x => {
|
|
399
|
+
const ev = { position: Math.round(x.position * 10) / 10, ctr: Math.round(x.ctr * 1000) / 10 + '%', expectedCtr: Math.round(x.exp * 1000) / 10 + '%', impressions: x.impressions };
|
|
400
|
+
// Extreme case: near-zero CTR at a strong position isn't a title problem — a SERP feature
|
|
401
|
+
// (image/video/AI overview) or navigational intent is taking the clicks. Different fix.
|
|
402
|
+
if (x.position <= 5 && x.ctr < x.exp * 0.15)
|
|
403
|
+
ev.note = 'near-zero CTR for the position — likely a SERP feature or navigational intent taking the clicks; check the live SERP before rewriting the title/meta';
|
|
404
|
+
return { urlKey: x.urlKey, evidence: ev };
|
|
405
|
+
}) : [],
|
|
406
|
+
},
|
|
407
|
+
// ── Period-over-period (GSC history by date) — the trend questions SEOs live in ──
|
|
408
|
+
{
|
|
409
|
+
id: 'traffic-decay', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
410
|
+
title: 'Page losing clicks (period-over-period)', fix: 'Refresh and expand the content, and check for lost rankings — this page’s Search Console clicks fell sharply against the previous 28 days.',
|
|
411
|
+
run: (c) => (!c.gscMaxDate || spanDays(c) < 56) ? [] : rows(c, `
|
|
412
|
+
WITH cur AS (SELECT page_key, SUM(clicks) c, SUM(impressions) i, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos
|
|
413
|
+
FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key),
|
|
414
|
+
prev AS (SELECT page_key, SUM(clicks) c FROM search_analytics WHERE page_key IS NOT NULL AND ${winPrev(c.gscMaxDate)} GROUP BY page_key)
|
|
415
|
+
SELECT prev.page_key url, prev.c prevC, COALESCE(cur.c,0) curC, COALESCE(cur.i,0) curI, COALESCE(cur.pos,10) pos
|
|
416
|
+
FROM prev LEFT JOIN cur ON cur.page_key=prev.page_key
|
|
417
|
+
WHERE prev.c >= 30 AND COALESCE(cur.c,0) < prev.c * 0.6
|
|
418
|
+
ORDER BY (prev.c - COALESCE(cur.c,0)) DESC LIMIT 40`)
|
|
419
|
+
.map(r => ({ urlKey: null, evidence: { url: r.url, previousClicks: r.prevC, currentClicks: r.curC, clicksLost: r.prevC - r.curC, dropPercent: Math.round((1 - r.curC / r.prevC) * 100) + '%', clicks: r.prevC - r.curC, impressions: r.curI, position: Math.round(r.pos * 10) / 10 } })),
|
|
420
|
+
},
|
|
421
|
+
{
|
|
422
|
+
id: 'lost-queries', category: 'merged', severity: 'med', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
423
|
+
title: 'Query dropped out of rankings', fix: 'This query drove clicks last period and drives none now — find the page that ranked, check for de-indexing or lost rankings, and win it back.',
|
|
424
|
+
run: (c) => (!c.gscMaxDate || spanDays(c) < 56) ? [] : rows(c, `
|
|
425
|
+
WITH cur AS (SELECT query, SUM(clicks) c FROM search_analytics WHERE query IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY query),
|
|
426
|
+
prev AS (SELECT query, SUM(clicks) c, SUM(impressions) i FROM search_analytics WHERE query IS NOT NULL AND ${winPrev(c.gscMaxDate)} GROUP BY query)
|
|
427
|
+
SELECT prev.query query, prev.c prevC, prev.i prevI FROM prev LEFT JOIN cur ON cur.query=prev.query
|
|
428
|
+
WHERE prev.c >= 10 AND COALESCE(cur.c,0)=0
|
|
429
|
+
ORDER BY prev.c DESC LIMIT 40`)
|
|
430
|
+
.map(r => ({ urlKey: null, evidence: { query: r.query, previousClicks: r.prevC, currentClicks: 0, clicks: r.prevC, impressions: r.prevI } })),
|
|
431
|
+
},
|
|
432
|
+
{
|
|
433
|
+
id: 'position-slipping', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
434
|
+
title: 'Page-1 ranking slipping', fix: 'Average position for this query worsened by 3+ spots vs the previous 28 days while it was still on page one — investigate the ranking loss before the clicks follow it down.',
|
|
435
|
+
run: (c) => (!c.gscMaxDate || spanDays(c) < 56) ? [] : rows(c, `
|
|
436
|
+
WITH cur AS (SELECT query, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos, SUM(impressions) i, SUM(clicks) c
|
|
437
|
+
FROM search_analytics WHERE query IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY query HAVING SUM(impressions) >= 100),
|
|
438
|
+
prev AS (SELECT query, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos FROM search_analytics WHERE query IS NOT NULL AND ${winPrev(c.gscMaxDate)} GROUP BY query)
|
|
439
|
+
SELECT cur.query query, prev.pos prevPos, cur.pos curPos, cur.i i, cur.c c FROM cur JOIN prev ON prev.query=cur.query
|
|
440
|
+
WHERE prev.pos <= 10 AND cur.pos - prev.pos >= 3
|
|
441
|
+
ORDER BY (cur.pos - prev.pos) * cur.i DESC LIMIT 40`)
|
|
442
|
+
.map(r => ({ urlKey: null, evidence: { query: r.query, previousPosition: Math.round(r.prevPos * 10) / 10, currentPosition: Math.round(r.curPos * 10) / 10, slippedBy: Math.round((r.curPos - r.prevPos) * 10) / 10, impressions: r.i, clicks: r.c, position: Math.round(r.curPos * 10) / 10 } })),
|
|
443
|
+
},
|
|
444
|
+
{
|
|
445
|
+
id: 'index-bloat', category: 'indexation', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
446
|
+
title: 'Indexable page with no search traffic', fix: 'No impressions in 90 days despite being indexable — consolidate, improve, or noindex/prune to concentrate crawl budget and internal authority (confirm it isn’t seasonal or brand-new first).',
|
|
447
|
+
run: (c) => (!c.gscMaxDate || spanDays(c) < 90) ? [] : rows(c, `SELECT url_key urlKey, ipr FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND COALESCE(click_depth, 999) >= 1 AND url_key NOT IN (SELECT DISTINCT page_key FROM search_analytics WHERE page_key IS NOT NULL AND date <= '${c.gscMaxDate}' AND date > date('${c.gscMaxDate}','-90 days') AND impressions > 0) ORDER BY ipr DESC LIMIT 100`).map(r => ({ urlKey: r.urlKey, evidence: { note: 'indexable but zero impressions in 90 days', ipr: Math.round(r.ipr) } })),
|
|
448
|
+
},
|
|
449
|
+
// ── On-page parity (crawl-only, deterministic) ──
|
|
450
|
+
{
|
|
451
|
+
id: 'duplicate-meta-description', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
452
|
+
title: 'Duplicate meta description', fix: 'Give each indexable page a unique meta description.',
|
|
453
|
+
run: (c) => rows(c, `SELECT url_key urlKey, meta_description md FROM pages WHERE status_code=200 AND indexable=1 AND ${notPagination()} AND meta_description IS NOT NULL AND TRIM(meta_description)!='' AND LOWER(TRIM(meta_description)) IN (SELECT LOWER(TRIM(meta_description)) FROM pages WHERE status_code=200 AND indexable=1 AND ${notPagination()} AND meta_description IS NOT NULL AND TRIM(meta_description)!='' GROUP BY LOWER(TRIM(meta_description)) HAVING COUNT(*)>1)`).map(r => ({ urlKey: r.urlKey, evidence: { metaDescription: r.md } })),
|
|
454
|
+
},
|
|
455
|
+
{
|
|
456
|
+
id: 'title-h1-mismatch', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
457
|
+
title: 'Title and H1 share no significant words', fix: 'Align the <title> and <h1> — they currently have no significant words in common, which blurs the page’s topic signal.',
|
|
458
|
+
run: (c) => rows(c, `SELECT url_key urlKey, title, h1 FROM pages WHERE status_code=200 AND indexable=1 AND title IS NOT NULL AND TRIM(title)!='' AND h1 IS NOT NULL AND TRIM(h1)!=''`)
|
|
459
|
+
.filter(r => { const tt = terms(r.title), th = terms(r.h1); if (tt.length < 2 || th.length < 2)
|
|
460
|
+
return false; const set = new Set(th); return !tt.some(w => set.has(w)); })
|
|
461
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { title: r.title, h1: r.h1 } })),
|
|
462
|
+
},
|
|
463
|
+
{
|
|
464
|
+
id: 'rising-pages', category: 'merged', severity: 'low', labels: ['G'], certainty: 1, effortBase: 2, fixType: 'per-page',
|
|
465
|
+
title: 'Page gaining clicks fast (double down)', fix: 'Clicks jumped vs the previous 28 days — reinforce it with internal links and related content while the momentum is there.',
|
|
466
|
+
run: (c) => (!c.gscMaxDate || spanDays(c) < 56) ? [] : rows(c, `
|
|
467
|
+
WITH cur AS (SELECT page_key, SUM(clicks) c, SUM(impressions) i, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key),
|
|
468
|
+
prev AS (SELECT page_key, SUM(clicks) c FROM search_analytics WHERE page_key IS NOT NULL AND ${winPrev(c.gscMaxDate)} GROUP BY page_key)
|
|
469
|
+
SELECT cur.page_key url, COALESCE(prev.c,0) prevC, cur.c curC, cur.i curI, COALESCE(cur.pos,10) pos
|
|
470
|
+
FROM cur LEFT JOIN prev ON prev.page_key=cur.page_key
|
|
471
|
+
WHERE cur.c >= 30 AND cur.c >= COALESCE(prev.c,0) * 1.5
|
|
472
|
+
ORDER BY (cur.c - COALESCE(prev.c,0)) DESC LIMIT 25`)
|
|
473
|
+
.map(r => ({ urlKey: null, evidence: { url: r.url, previousClicks: r.prevC, currentClicks: r.curC, clicksGained: r.curC - r.prevC, clicks: r.curC, impressions: r.curI, position: Math.round(r.pos * 10) / 10 } })),
|
|
474
|
+
},
|
|
475
|
+
{
|
|
476
|
+
id: 'traffic-to-dead-url', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
477
|
+
title: 'Search traffic to a dead (non-200) URL', fix: 'Google still sends clicks/impressions to this URL but the crawl returns a 4xx/5xx — recover the page or 301 it to the best live equivalent so the demand isn’t lost. (The "directive contradicts reality" join: you rank for a page that no longer works.)',
|
|
478
|
+
run: (c) => c.gscMaxDate ? rows(c, `SELECT p.url_key urlKey, p.status_code st, SUM(sa.clicks) clicks, SUM(sa.impressions) impressions FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key WHERE p.status_code >= 400 AND sa.${win(c.gscMaxDate)} GROUP BY p.url_key HAVING SUM(sa.impressions) >= 10 ORDER BY clicks DESC, impressions DESC LIMIT 40`)
|
|
479
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { status: r.st, clicks: r.clicks, impressions: r.impressions, note: `HTTP ${r.st} but still earning search traffic` } })) : [],
|
|
480
|
+
},
|
|
481
|
+
{
|
|
482
|
+
id: 'impressions-rising-clicks-flat', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
483
|
+
title: 'Impressions rising but clicks flat (CTR erosion)', fix: 'Google is showing this page MORE than it used to, yet you’re not winning more clicks — a stale title/meta, or a SERP feature (AI overview, snippet, pack) is taking them. Rewrite the snippet or target the feature.',
|
|
484
|
+
run: (c) => (!c.gscMaxDate || spanDays(c) < 56) ? [] : rows(c, `
|
|
485
|
+
WITH cur AS (SELECT page_key, SUM(clicks) c, SUM(impressions) i, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) pos FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key),
|
|
486
|
+
prev AS (SELECT page_key, SUM(clicks) c, SUM(impressions) i FROM search_analytics WHERE page_key IS NOT NULL AND ${winPrev(c.gscMaxDate)} GROUP BY page_key)
|
|
487
|
+
SELECT cur.page_key url, prev.i prevImpr, cur.i curImpr, prev.c prevClicks, cur.c curClicks, cur.pos pos
|
|
488
|
+
FROM cur JOIN prev ON prev.page_key=cur.page_key
|
|
489
|
+
WHERE prev.i >= 200 AND cur.i >= prev.i * 1.3 AND cur.c <= prev.c
|
|
490
|
+
ORDER BY (cur.i - prev.i) DESC LIMIT 40`)
|
|
491
|
+
.map(r => ({ urlKey: null, evidence: { url: r.url, previousImpressions: r.prevImpr, currentImpressions: r.curImpr, impressionsChange: '+' + Math.round((r.curImpr / r.prevImpr - 1) * 100) + '%', previousClicks: r.prevClicks, currentClicks: r.curClicks, impressions: r.curImpr, clicks: r.curClicks, position: Math.round(r.pos * 10) / 10 } })),
|
|
492
|
+
},
|
|
493
|
+
{
|
|
494
|
+
id: 'h1-missing-top-query', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
495
|
+
title: 'Top query missing from the H1', fix: 'Work the page’s top-performing query into the <h1> — it ranks for this term but the main heading doesn’t mention it.',
|
|
496
|
+
run: (c) => {
|
|
497
|
+
if (!c.gscMaxDate)
|
|
498
|
+
return [];
|
|
499
|
+
const r = rows(c, `SELECT s.page_key urlKey, s.query query, s.impr impressions, p.h1 h1 FROM
|
|
500
|
+
(SELECT page_key, query, SUM(impressions) impr, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
|
|
501
|
+
FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
|
|
502
|
+
JOIN pages p ON p.url_key = s.page_key
|
|
503
|
+
WHERE s.rn = 1 AND p.indexable = 1 AND p.h1 IS NOT NULL AND p.h1 != '' AND s.impr >= 100`);
|
|
504
|
+
return r.filter(x => { const q = terms(x.query); return q.length > 0 && !q.some(w => titleHasTerm(x.h1, w)); })
|
|
505
|
+
.map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, h1: x.h1 } }));
|
|
506
|
+
},
|
|
507
|
+
},
|
|
508
|
+
// ── Content cluster (AI-era / RAG layer): exploit the chunked body text captured at crawl ──
|
|
509
|
+
{
|
|
510
|
+
id: 'body-missing-top-query', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
511
|
+
title: 'Top query missing from the page body', fix: 'The page ranks for this query yet its terms appear nowhere — not the title, H1, or body copy. Add a section that actually covers the topic; if it can’t, the page is too thin to hold the ranking and a stronger page should target it.',
|
|
512
|
+
run: (c) => {
|
|
513
|
+
if (!c.gscMaxDate)
|
|
514
|
+
return [];
|
|
515
|
+
const r = rows(c, `SELECT s.page_key urlKey, s.query query, s.impr impressions, p.title title, p.h1 h1, p.body_chunks bc FROM
|
|
516
|
+
(SELECT page_key, query, SUM(impressions) impr, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
|
|
517
|
+
FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
|
|
518
|
+
JOIN pages p ON p.url_key = s.page_key
|
|
519
|
+
WHERE s.rn = 1 AND p.indexable = 1 AND p.body_chunks IS NOT NULL AND s.impr >= 100`);
|
|
520
|
+
return r.filter(x => {
|
|
521
|
+
const q = terms(x.query);
|
|
522
|
+
if (!q.length)
|
|
523
|
+
return false;
|
|
524
|
+
let body = `${x.title || ''} ${x.h1 || ''}`;
|
|
525
|
+
try {
|
|
526
|
+
for (const ch of JSON.parse(x.bc))
|
|
527
|
+
body += ` ${ch.heading || ''} ${ch.text || ''}`;
|
|
528
|
+
}
|
|
529
|
+
catch { /* skip */ }
|
|
530
|
+
return !q.some((w) => titleHasTerm(body, w)); // none of the query's terms appear on the page
|
|
531
|
+
}).map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, note: 'query terms absent from title, H1 and body' } }));
|
|
532
|
+
},
|
|
533
|
+
},
|
|
534
|
+
{
|
|
535
|
+
id: 'poor-chunkability', category: 'content', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
536
|
+
title: 'Content has no heading structure (poor RAG boundaries)', fix: 'Break the copy into headed sections (h2/h3). AI search and featured snippets lift self-contained, headed passages — a wall of text with no subheadings gives them no clean chunk to quote.',
|
|
537
|
+
run: (c) => {
|
|
538
|
+
const out = [];
|
|
539
|
+
for (const x of iterRows(c, `SELECT url_key urlKey, word_count wc, body_chunks bc FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND word_count >= 400 AND body_chunks IS NOT NULL`)) {
|
|
540
|
+
let headed;
|
|
541
|
+
try {
|
|
542
|
+
headed = JSON.parse(x.bc).filter((k) => k.heading && k.level >= 2).length;
|
|
543
|
+
}
|
|
544
|
+
catch {
|
|
545
|
+
continue;
|
|
546
|
+
}
|
|
547
|
+
if (headed <= 1)
|
|
548
|
+
out.push({ urlKey: x.urlKey, evidence: { wordCount: x.wc, note: '400+ words with ≤1 subheading — one undifferentiated block' } });
|
|
549
|
+
}
|
|
550
|
+
return out;
|
|
551
|
+
},
|
|
552
|
+
},
|
|
553
|
+
{
|
|
554
|
+
id: 'rag-answer-gap', category: 'merged', severity: 'med', labels: ['G', 'N'], certainty: 0.6, effortBase: 5, fixType: 'per-page',
|
|
555
|
+
title: 'No single passage answers the ranking query (RAG gap)', fix: 'The page contains the query’s terms but scattered across sections — no one chunk (heading + paragraph) holds them together. AI answers lift a single self-contained passage, so add one that directly answers the query in ~50 words. (Heuristic — confirm the query’s intent first.)',
|
|
556
|
+
run: (c) => {
|
|
557
|
+
if (!c.gscMaxDate)
|
|
558
|
+
return [];
|
|
559
|
+
const r = rows(c, `SELECT s.page_key urlKey, s.query query, s.impr impressions, p.title title, p.body_chunks bc FROM
|
|
560
|
+
(SELECT page_key, query, SUM(impressions) impr, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
|
|
561
|
+
FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
|
|
562
|
+
JOIN pages p ON p.url_key = s.page_key
|
|
563
|
+
WHERE s.rn = 1 AND p.indexable = 1 AND p.body_chunks IS NOT NULL AND s.impr >= 200`);
|
|
564
|
+
return r.filter(x => {
|
|
565
|
+
const q = terms(x.query);
|
|
566
|
+
if (q.length < 2)
|
|
567
|
+
return false; // only multi-term queries can be "scattered"
|
|
568
|
+
let chunks;
|
|
569
|
+
try {
|
|
570
|
+
chunks = JSON.parse(x.bc);
|
|
571
|
+
}
|
|
572
|
+
catch {
|
|
573
|
+
return false;
|
|
574
|
+
}
|
|
575
|
+
const whole = `${x.title || ''} ${chunks.map(ch => `${ch.heading || ''} ${ch.text || ''}`).join(' ')}`;
|
|
576
|
+
if (!q.every((w) => titleHasTerm(whole, w)))
|
|
577
|
+
return false; // page must contain all terms (else it's body-missing)
|
|
578
|
+
return !chunks.some(ch => { const h = `${ch.heading || ''} ${ch.text || ''}`; return q.every((w) => titleHasTerm(h, w)); }); // but no single chunk does
|
|
579
|
+
}).map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, note: 'terms present but never together in one passage' } }));
|
|
580
|
+
},
|
|
581
|
+
},
|
|
582
|
+
{
|
|
583
|
+
id: 'low-extractability', category: 'content', severity: 'low', labels: ['D', 'N'], certainty: 0.5, effortBase: 3, fixType: 'per-page',
|
|
584
|
+
title: 'Passages depend on context (low answer-extractability)', fix: 'Half or more of this page’s sections open with “it / this / they / there…” or never name the subject — lifted out of the page by an LLM they read as meaningless. Open key sections by naming the entity. (Heuristic.)',
|
|
585
|
+
run: (c) => {
|
|
586
|
+
const out = [];
|
|
587
|
+
const PRON = /^(it|this|that|these|those|they|he|she|there|here|such|one)\b/i;
|
|
588
|
+
for (const x of iterRows(c, `SELECT url_key urlKey, body_chunks bc FROM pages WHERE status_code=200 AND indexable=1 AND word_count >= 400 AND body_chunks IS NOT NULL`)) {
|
|
589
|
+
let chunks;
|
|
590
|
+
try {
|
|
591
|
+
chunks = JSON.parse(x.bc);
|
|
592
|
+
}
|
|
593
|
+
catch {
|
|
594
|
+
continue;
|
|
595
|
+
}
|
|
596
|
+
const bodied = chunks.filter((k) => k.text && k.text.length > 60);
|
|
597
|
+
if (bodied.length < 3)
|
|
598
|
+
continue;
|
|
599
|
+
const dep = bodied.filter((k) => PRON.test(String(k.text).trim())).length;
|
|
600
|
+
if (dep / bodied.length >= 0.5)
|
|
601
|
+
out.push({ urlKey: x.urlKey, evidence: { dependentSections: dep, totalSections: bodied.length, note: `${Math.round(dep / bodied.length * 100)}% of sections open context-dependent` } });
|
|
602
|
+
}
|
|
603
|
+
return out;
|
|
604
|
+
},
|
|
605
|
+
},
|
|
606
|
+
{
|
|
607
|
+
// Hobo "Signal Coherence" / Goldmine; leak: anchor_mismatch. Google leans on internal anchors to
|
|
608
|
+
// understand a page's topic — if the IN-CONTENT inbound anchors never mention the query the page
|
|
609
|
+
// actually ranks for, that's an incoherent internal signal. Guard against boilerplate FPs by using
|
|
610
|
+
// ONLY placement='body' anchors (nav/footer/aside excluded) and requiring ≥3 of them.
|
|
611
|
+
id: 'anchor-text-incoherent', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
612
|
+
title: 'Internal anchors don’t mention the page’s top query', fix: 'The in-content internal links pointing at this page never use its top-ranking query in their anchor text — and Google leans on internal anchors to understand what a page is about. Re-anchor the key internal links with descriptive, query-relevant text instead of generic “read more” / brand-only labels.',
|
|
613
|
+
run: (c) => {
|
|
614
|
+
if (!c.gscMaxDate)
|
|
615
|
+
return [];
|
|
616
|
+
const top = rows(c, `SELECT s.page_key urlKey, s.query query, s.impr impressions FROM
|
|
617
|
+
(SELECT page_key, query, SUM(impressions) impr, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
|
|
618
|
+
FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} ${brandExcl(c)} GROUP BY page_key, query) s
|
|
619
|
+
JOIN pages p ON p.url_key = s.page_key
|
|
620
|
+
WHERE s.rn = 1 AND p.indexable = 1 AND s.impr >= 100`);
|
|
621
|
+
// Pool only GENUINE editorial anchors. "Chrome" (nav/footer/breadcrumb/CTA) is detected
|
|
622
|
+
// STRUCTURALLY, not via an English word-list: an anchor text reused across a large share of
|
|
623
|
+
// the site's pages is templated boilerplate — e.g. the EHI homepage's 2,940 inbound "Home"
|
|
624
|
+
// breadcrumb links — and that holds in any language ("Startseite", "Accueil"…). We also drop
|
|
625
|
+
// self-links and anchors with no letters ("(0)", page numbers, arrows). Editorial in-content
|
|
626
|
+
// anchors recur on only a handful of pages, so they survive.
|
|
627
|
+
const anchors = new Map();
|
|
628
|
+
for (const a of rows(c, `
|
|
629
|
+
WITH body_anchors AS (
|
|
630
|
+
SELECT target_key, anchor_text, source_key, LOWER(TRIM(anchor_text)) atext
|
|
631
|
+
FROM links
|
|
632
|
+
WHERE is_internal = 1 AND placement = 'body' AND source_key <> target_key
|
|
633
|
+
AND anchor_text IS NOT NULL AND TRIM(anchor_text) != '' AND LOWER(TRIM(anchor_text)) GLOB '*[a-z]*'
|
|
634
|
+
),
|
|
635
|
+
templated AS (
|
|
636
|
+
SELECT atext FROM body_anchors GROUP BY atext
|
|
637
|
+
HAVING COUNT(DISTINCT source_key) > MAX(20, (SELECT COUNT(*) FROM pages WHERE status_code = 200 AND ${HTML_CT}) * 0.15)
|
|
638
|
+
)
|
|
639
|
+
SELECT target_key tk, GROUP_CONCAT(anchor_text, ' ') pool, COUNT(*) n
|
|
640
|
+
FROM body_anchors
|
|
641
|
+
WHERE atext NOT IN (SELECT atext FROM templated)
|
|
642
|
+
GROUP BY target_key`))
|
|
643
|
+
anchors.set(a.tk, { pool: a.pool, n: a.n });
|
|
644
|
+
return top.filter(x => {
|
|
645
|
+
const a = anchors.get(x.urlKey);
|
|
646
|
+
if (!a || a.n < 3)
|
|
647
|
+
return false; // need enough genuine in-content inbound links to judge
|
|
648
|
+
const q = terms(x.query);
|
|
649
|
+
return q.length > 0 && !q.some((w) => titleHasTerm(a.pool, w));
|
|
650
|
+
}).map(x => { const a = anchors.get(x.urlKey); return { urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, inboundInContentLinks: a.n } }; });
|
|
651
|
+
},
|
|
652
|
+
},
|
|
653
|
+
{
|
|
654
|
+
// The "RAG snippetability" test. A local cross-encoder (the kind AI search uses to re-rank) scored
|
|
655
|
+
// every chunk against the page's top query; we persisted the single best-passage score. A low max
|
|
656
|
+
// means no dense, extractable answer anywhere on the page — it will lose in AI/passage search even
|
|
657
|
+
// if it keyword-matches. Model-derived (not deterministic truth) → N label, includeJudgement-gated.
|
|
658
|
+
// Requires `score_passages` to have run (like CWV needs page_lighthouse).
|
|
659
|
+
id: 'weak-passage-answer', category: 'merged', severity: 'high', labels: ['G', 'N'], certainty: 0.8, effortBase: 5, fixType: 'per-page',
|
|
660
|
+
title: 'No passage strongly answers the ranking query (AI-search risk)', fix: 'A local neural reranker found no single passage on this page that confidently answers its top query — the page covers the topic loosely but offers no dense, extractable answer, so AI/passage search will prefer a clearer source. Add a focused, self-contained passage: a heading that states the question + a direct ~50-word answer up top. Run `score_passages` to (re)populate.',
|
|
661
|
+
run: (c) => rows(c, `SELECT url_key urlKey, max_passage_score mps, max_passage_query q, max_passage_impr impr FROM pages
|
|
662
|
+
WHERE indexable=1 AND max_passage_score IS NOT NULL AND max_passage_score < 3`)
|
|
663
|
+
.map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.q, maxPassageScore: x.mps, impressions: x.impr ?? 0, note: 'best passage scores below the reranker confidence threshold' } })),
|
|
664
|
+
},
|
|
665
|
+
{
|
|
666
|
+
// Dejan: search weights the opening heavily and AI answers front-load. If the ranking query's
|
|
667
|
+
// terms are present LATER in the page but absent from the opening (~first 2 chunks / ~200 words),
|
|
668
|
+
// the answer is buried. (body-missing-top-query handles total absence; this is the buried case.)
|
|
669
|
+
// Informational intent only — front-loading matters less for navigational/transactional queries.
|
|
670
|
+
id: 'answer-not-front-loaded', category: 'merged', severity: 'med', labels: ['G', 'N'], certainty: 0.6, effortBase: 3, fixType: 'per-page',
|
|
671
|
+
title: 'Answer to the ranking query is buried, not front-loaded', fix: 'The page covers its top query but the terms don’t appear up top (the intro / first section). Google weights the opening heavily and AI answers front-load — move a direct ~50–100-word answer to the first section. (Heuristic — informational queries.)',
|
|
672
|
+
run: (c) => {
|
|
673
|
+
if (!c.gscMaxDate)
|
|
674
|
+
return [];
|
|
675
|
+
const INFO = /\b(how|what|why|when|which|who|guide|tutorial|best|vs|versus|is|are|does|do|can|should|tips|ideas|examples|meaning|definition|setup|settings)\b/i;
|
|
676
|
+
const r = rows(c, `SELECT s.page_key urlKey, s.query query, s.impr impressions, p.title title, p.body_chunks bc FROM
|
|
677
|
+
(SELECT page_key, query, SUM(impressions) impr, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
|
|
678
|
+
FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
|
|
679
|
+
JOIN pages p ON p.url_key = s.page_key
|
|
680
|
+
WHERE s.rn = 1 AND p.indexable = 1 AND p.body_chunks IS NOT NULL AND p.word_count >= 800 AND s.impr >= 100`);
|
|
681
|
+
return r.filter(x => {
|
|
682
|
+
if (!INFO.test(x.query))
|
|
683
|
+
return false;
|
|
684
|
+
const q = terms(x.query);
|
|
685
|
+
if (!q.length)
|
|
686
|
+
return false;
|
|
687
|
+
let chunks;
|
|
688
|
+
try {
|
|
689
|
+
chunks = JSON.parse(x.bc);
|
|
690
|
+
}
|
|
691
|
+
catch {
|
|
692
|
+
return false;
|
|
693
|
+
}
|
|
694
|
+
const front = `${x.title || ''} ${chunks.slice(0, 2).map((k) => `${k.heading || ''} ${k.text || ''}`).join(' ')}`.slice(0, 1200);
|
|
695
|
+
const whole = `${x.title || ''} ${chunks.map((k) => `${k.heading || ''} ${k.text || ''}`).join(' ')}`;
|
|
696
|
+
return q.some((w) => titleHasTerm(whole, w)) && !q.some((w) => titleHasTerm(front, w));
|
|
697
|
+
}).map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, note: 'query terms appear later in the page but not in the opening' } }));
|
|
698
|
+
},
|
|
699
|
+
},
|
|
700
|
+
{
|
|
701
|
+
// Dejan "density beats length": AI grounds ~370 words/page, diminishing past ~1,500. A very long
|
|
702
|
+
// page with an over-long unbroken section grounds poorly — split it into focused, headed passages.
|
|
703
|
+
id: 'content-bloat', category: 'content', severity: 'low', labels: ['D', 'N'], certainty: 0.6, effortBase: 5, fixType: 'per-page',
|
|
704
|
+
title: 'Over-long section dilutes AI-grounding (density beats length)', fix: 'This page has a very long unbroken section. AI search grounds only ~370 words per page with sharp diminishing returns past ~1,500 — break the long section into focused, headed passages (or tighten it) so each answers one thing cleanly.',
|
|
705
|
+
run: (c) => {
|
|
706
|
+
const out = [];
|
|
707
|
+
// Use the real (uncapped) word_count ÷ number of headed sections — chunk TEXT is capped at
|
|
708
|
+
// extraction, so we infer over-long sections from words-per-heading, not from chunk length.
|
|
709
|
+
for (const x of iterRows(c, `SELECT url_key urlKey, word_count wc, body_chunks bc FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND word_count >= 2500 AND body_chunks IS NOT NULL`)) {
|
|
710
|
+
let chunks;
|
|
711
|
+
try {
|
|
712
|
+
chunks = JSON.parse(x.bc);
|
|
713
|
+
}
|
|
714
|
+
catch {
|
|
715
|
+
continue;
|
|
716
|
+
}
|
|
717
|
+
const headed = chunks.filter((k) => k.heading && k.level >= 2).length;
|
|
718
|
+
const wordsPerSection = Math.round(x.wc / Math.max(1, headed));
|
|
719
|
+
if (wordsPerSection > 500)
|
|
720
|
+
out.push({ urlKey: x.urlKey, evidence: { wordCount: x.wc, headedSections: headed, wordsPerSection, note: 'long page with sparse headings — sections average >500 words, too large to ground cleanly' } });
|
|
721
|
+
}
|
|
722
|
+
return out;
|
|
723
|
+
},
|
|
724
|
+
},
|
|
725
|
+
{
|
|
726
|
+
// Hobo Level 3 freshness / lastSignificantUpdate. Gemini guard: YoY windows (negate seasonality
|
|
727
|
+
// + zero-click-SERP CTR loss). Flag when the page hasn't been meaningfully re-dated in >12 months
|
|
728
|
+
// AND clicks are down >25% YoY AND impressions down >15% YoY (impressions confirm ranking decay,
|
|
729
|
+
// not just CTR). Needs ~13 months of GSC — guarded by spanDays so it stays silent on shallow syncs.
|
|
730
|
+
id: 'stale-content', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
731
|
+
title: 'Stale page declining year-on-year', fix: 'This page hasn’t been meaningfully updated in over a year and its Search Console clicks are down sharply versus the same period last year — refresh and expand the content (and honestly re-date it) to rebuild the freshness signal Google rewards.',
|
|
732
|
+
run: (c) => {
|
|
733
|
+
if (!c.gscMaxDate || spanDays(c) < 455)
|
|
734
|
+
return []; // YoY window reaches back 455 days — anything less truncates the prior-year period
|
|
735
|
+
const d = c.gscMaxDate;
|
|
736
|
+
const out = [];
|
|
737
|
+
const rs = rows(c, `
|
|
738
|
+
WITH cur AS (SELECT page_key, SUM(clicks) c, SUM(impressions) i FROM search_analytics WHERE page_key IS NOT NULL AND date <= '${d}' AND date > date('${d}','-90 days') GROUP BY page_key),
|
|
739
|
+
py AS (SELECT page_key, SUM(clicks) c, SUM(impressions) i FROM search_analytics WHERE page_key IS NOT NULL AND date <= date('${d}','-365 days') AND date > date('${d}','-455 days') GROUP BY page_key)
|
|
740
|
+
SELECT cur.page_key url, cur.c curC, cur.i curI, py.c pyC, py.i pyI, p.json_ld jl
|
|
741
|
+
FROM cur JOIN py ON py.page_key = cur.page_key JOIN pages p ON p.url_key = cur.page_key
|
|
742
|
+
WHERE p.indexable = 1 AND py.c >= 50 AND cur.c < py.c * 0.75 AND cur.i < py.i * 0.85
|
|
743
|
+
ORDER BY (py.c - cur.c) DESC LIMIT 40`);
|
|
744
|
+
for (const x of rs) {
|
|
745
|
+
const dm = newestSchemaDate(x.jl);
|
|
746
|
+
if (!dm)
|
|
747
|
+
continue;
|
|
748
|
+
const ageDays = (Date.parse(d) - Date.parse(dm)) / 86400000;
|
|
749
|
+
if (!(ageDays >= 365))
|
|
750
|
+
continue; // only genuinely stale pages (>12 months since last schema date)
|
|
751
|
+
out.push({ urlKey: null, evidence: { url: x.url, dateModified: dm.slice(0, 10), clicksYoY: `${x.pyC}→${x.curC} (-${Math.round((1 - x.curC / x.pyC) * 100)}%)`, impressionsYoY: `${x.pyI}→${x.curI}`, clicks: x.pyC - x.curC, impressions: x.pyI } });
|
|
752
|
+
}
|
|
753
|
+
return out;
|
|
754
|
+
},
|
|
755
|
+
},
|
|
756
|
+
{
|
|
757
|
+
id: 'high-ipr-no-traffic', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
758
|
+
title: 'Internal authority wasted on a no-traffic page', fix: 'High internal link equity (iPR) and Google does rank it (it earns impressions), yet it gets zero clicks — rewrite the title/snippet or improve the page, or repoint that authority to pages that convert it. (Requires impressions, so functional pages with no search demand are excluded.)',
|
|
759
|
+
run: (c) => (!c.gscMaxDate || spanDays(c) < 90) ? [] : rows(c, `SELECT p.url_key urlKey, p.ipr ipr, p.inlink_count inl, SUM(sa.impressions) impressions
|
|
760
|
+
FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key
|
|
761
|
+
WHERE p.indexable=1 AND p.ipr >= 50 AND COALESCE(p.click_depth, 999) >= 1 AND sa.date <= '${c.gscMaxDate}' AND sa.date > date('${c.gscMaxDate}','-90 days')
|
|
762
|
+
GROUP BY p.url_key HAVING SUM(sa.clicks)=0 AND SUM(sa.impressions) >= 100
|
|
763
|
+
ORDER BY p.ipr DESC LIMIT 50`).map(r => ({ urlKey: r.urlKey, evidence: { ipr: Math.round(r.ipr), inlinks: r.inl, impressions: r.impressions, clicks: 0, note: 'high internal authority + impressions, zero clicks in 90 days' } })),
|
|
764
|
+
},
|
|
765
|
+
{
|
|
766
|
+
id: 'homepage-missing-org-schema', category: 'schema', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
|
|
767
|
+
title: 'Homepage missing Organization / WebSite schema', fix: 'Add Organization (or LocalBusiness) and WebSite JSON-LD to the homepage — it underpins the knowledge panel, logo and sitelinks search box.',
|
|
768
|
+
run: (c) => {
|
|
769
|
+
const hp = (rows(c, `SELECT url_key urlKey, json_ld jsonLd FROM pages WHERE status_code=200 AND click_depth=0 LIMIT 1`)[0]
|
|
770
|
+
?? rows(c, `SELECT url_key urlKey, json_ld jsonLd FROM pages WHERE status_code=200 ORDER BY inlink_count DESC LIMIT 1`)[0]);
|
|
771
|
+
if (!hp)
|
|
772
|
+
return [];
|
|
773
|
+
const types = new Set(parseJsonLdNodes(hp.jsonLd).map(nodeType).filter(Boolean));
|
|
774
|
+
const ok = ['Organization', 'LocalBusiness', 'Corporation', 'OnlineStore', 'WebSite'].some(t => types.has(t));
|
|
775
|
+
return ok ? [] : [{ urlKey: hp.urlKey, evidence: { found: [...types].join(', ') || 'none', note: 'no Organization/WebSite node on the homepage' } }];
|
|
776
|
+
},
|
|
777
|
+
},
|
|
778
|
+
{
|
|
779
|
+
id: 'breadcrumb-schema-inconsistent', category: 'schema', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
780
|
+
title: 'Breadcrumb schema missing on some pages', fix: 'Add BreadcrumbList JSON-LD — most of the site has it, so these pages are inconsistent and miss breadcrumb rich results.',
|
|
781
|
+
run: (c) => {
|
|
782
|
+
const pages = rows(c, `SELECT url_key urlKey, json_ld jsonLd, click_depth cd FROM pages WHERE status_code=200 AND indexable=1`);
|
|
783
|
+
let withBc = 0;
|
|
784
|
+
const without = [];
|
|
785
|
+
for (const p of pages) {
|
|
786
|
+
if (parseJsonLdNodes(p.jsonLd).some(n => nodeType(n) === 'BreadcrumbList'))
|
|
787
|
+
withBc++;
|
|
788
|
+
else if ((p.cd ?? 0) >= 2)
|
|
789
|
+
without.push(p.urlKey);
|
|
790
|
+
}
|
|
791
|
+
if (pages.length === 0 || withBc / pages.length < 0.4)
|
|
792
|
+
return []; // site doesn't use breadcrumbs → design choice, not a bug
|
|
793
|
+
return without.map(u => ({ urlKey: u, evidence: { note: 'site uses BreadcrumbList elsewhere; missing here' } }));
|
|
794
|
+
},
|
|
795
|
+
},
|
|
796
|
+
// ── Merged crawl × Search Console — the "expert questions" that need both datasets ──
|
|
797
|
+
{
|
|
798
|
+
id: 'ghost-pages', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
799
|
+
title: 'Ranking page the crawl can’t reach', fix: 'Google sends impressions/clicks to this URL but the site crawl never reached it — add internal links so it’s discoverable (or confirm it should exist and isn’t blocked).',
|
|
800
|
+
run: (c) => {
|
|
801
|
+
if (!c.gscMaxDate)
|
|
802
|
+
return [];
|
|
803
|
+
// Only meaningful on a COMPLETE crawl — if the crawl hit its maxPages cap, "absent
|
|
804
|
+
// from crawl" is unreliable. Tie the guard to the crawl that produced the CURRENT
|
|
805
|
+
// pages (not the latest crawl_metadata row, which may be a later failed crawl).
|
|
806
|
+
const m = c.db.prepare('SELECT urls_crawled c, max_pages m FROM crawl_metadata WHERE crawl_id = (SELECT crawl_id FROM pages LIMIT 1)').get();
|
|
807
|
+
// max_pages is nullable (NULL = no cap = complete crawl) — `c >= null` coerces to
|
|
808
|
+
// `c >= 0`, which would silently disable the check forever on capless crawls.
|
|
809
|
+
if (!m || m.c === 0 || (m.m != null && m.c >= m.m))
|
|
810
|
+
return [];
|
|
811
|
+
return rows(c, `SELECT page_key urlKey, SUM(clicks) clicks, SUM(impressions) impressions FROM search_analytics
|
|
812
|
+
WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key
|
|
813
|
+
HAVING SUM(impressions) >= 50 AND page_key NOT IN (SELECT url_key FROM pages)
|
|
814
|
+
ORDER BY impressions DESC`).map(r => ({ urlKey: r.urlKey, evidence: { clicks: r.clicks, impressions: r.impressions, note: 'earns GSC traffic but absent from the crawl' } }));
|
|
815
|
+
},
|
|
816
|
+
},
|
|
817
|
+
{
|
|
818
|
+
id: 'title-missing-top-query', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
819
|
+
title: 'Top query missing from the title', fix: 'Work the page’s top-performing query into the <title> — it already ranks for this term but the title doesn’t mention it.',
|
|
820
|
+
run: (c) => {
|
|
821
|
+
if (!c.gscMaxDate)
|
|
822
|
+
return [];
|
|
823
|
+
const r = rows(c, `SELECT s.page_key urlKey, s.query query, s.impr impressions, p.title title FROM
|
|
824
|
+
(SELECT page_key, query, SUM(impressions) impr, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
|
|
825
|
+
FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
|
|
826
|
+
JOIN pages p ON p.url_key = s.page_key
|
|
827
|
+
WHERE s.rn = 1 AND p.indexable = 1 AND p.title IS NOT NULL AND p.title != '' AND s.impr >= 100`);
|
|
828
|
+
return r
|
|
829
|
+
.filter(x => { const q = terms(x.query); return q.length > 0 && !q.some(w => titleHasTerm(x.title, w)); })
|
|
830
|
+
.map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, title: x.title } }));
|
|
831
|
+
},
|
|
832
|
+
},
|
|
833
|
+
// ── Internal link graph (iPR + click-depth + anchor text, from the crawl `links` table) ──
|
|
834
|
+
{
|
|
835
|
+
id: 'deep-pages', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
836
|
+
title: 'Page buried deep in the structure', fix: 'Add in-content (body) links from higher-level pages — this page is 4+ clicks from the homepage via body links.',
|
|
837
|
+
run: (c) => rows(c, `SELECT url_key urlKey, click_depth d FROM pages WHERE status_code=200 AND indexable=1 AND click_depth >= 4 ORDER BY click_depth DESC`).map(r => ({ urlKey: r.urlKey, evidence: { clickDepth: r.d } })),
|
|
838
|
+
},
|
|
839
|
+
{
|
|
840
|
+
id: 'underlinked-high-demand', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
841
|
+
title: 'High search demand, low internal authority', fix: 'Add internal links from high-authority (high-iPR) pages — this earns impressions but the site gives it little internal link equity.',
|
|
842
|
+
run: (c) => c.gscMaxDate ? rows(c, `SELECT p.url_key urlKey, p.ipr ipr, p.inlink_count inl, SUM(sa.impressions) impressions FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key WHERE p.indexable=1 AND p.ipr < 30 AND p.inlink_count BETWEEN 1 AND 3 AND sa.${win(c.gscMaxDate)} GROUP BY p.url_key HAVING SUM(sa.impressions) >= 300 ORDER BY impressions DESC`).map(r => ({ urlKey: r.urlKey, evidence: { impressions: r.impressions, ipr: Math.round(r.ipr), inlinks: r.inl } })) : [],
|
|
843
|
+
},
|
|
844
|
+
{
|
|
845
|
+
// The tier BETWEEN "orphan" (0 inlinks) and "fine": pages reached almost only via nav/footer.
|
|
846
|
+
// inlink_count counts ALL internal links, so a page sitting in the global nav looks well-linked
|
|
847
|
+
// even with zero EDITORIAL links — yet Google leans on in-content links for topic + equity. We
|
|
848
|
+
// count distinct in-content (placement='body') inbound sources; ≤2 + real demand = under-linked.
|
|
849
|
+
id: 'underlinked-editorial', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
850
|
+
title: 'High demand, almost no in-content internal links', fix: 'This page earns real impressions but is reached mainly via nav/footer — add descriptive in-content links to it from related articles. Editorial body links pass more topical context and equity than templated nav links.',
|
|
851
|
+
run: (c) => c.gscMaxDate ? rows(c, `
|
|
852
|
+
SELECT p.url_key urlKey, SUM(sa.impressions) impressions,
|
|
853
|
+
(SELECT COUNT(DISTINCT l.source_key) FROM links l WHERE l.target_key = p.url_key AND l.is_internal = 1 AND l.placement = 'body' AND l.source_key <> p.url_key) bodyLinks
|
|
854
|
+
FROM pages p JOIN search_analytics sa ON sa.page_key = p.url_key
|
|
855
|
+
WHERE p.indexable = 1 AND sa.${win(c.gscMaxDate)}
|
|
856
|
+
GROUP BY p.url_key
|
|
857
|
+
HAVING SUM(sa.impressions) >= 300 AND bodyLinks <= 2
|
|
858
|
+
ORDER BY impressions DESC LIMIT 40`).map(r => ({ urlKey: r.urlKey, evidence: { impressions: r.impressions, inContentLinks: r.bodyLinks, note: 'reached mainly via nav/footer — thin on editorial (in-content) links' } })) : [],
|
|
859
|
+
},
|
|
860
|
+
// NOTE: internal anchor-text checks (over-optimisation + generic/empty anchors) prototyped
|
|
861
|
+
// and PULLED twice. Re-evaluated 2026-06-22 against the AgricIDaniel/claude-seo and
|
|
862
|
+
// Bhanunamikaze/Agentic-SEO-Skill repos, this time using the links.placement='body' filter
|
|
863
|
+
// plus excluding anchors that match the target's own title/H1. On real data (ehi.com.au) the
|
|
864
|
+
// dominant survivors are still false positives: sitewide template CTAs ("home" 864/865,
|
|
865
|
+
// "contact us", "apply today") and category links whose anchor IS the page title. The
|
|
866
|
+
// page-level "mostly generic-anchored" variant returned 0 signal; raw empty anchors are
|
|
867
|
+
// image/thumbnail-link noise (3,764/23,435 body links). Reliable detection needs an
|
|
868
|
+
// editorial-vs-template link classifier we don't store (placement='body' still includes
|
|
869
|
+
// in-template CTAs and product grids). Keep pulled — a wrong finding is worse than none.
|
|
870
|
+
// ── Backlinks (need page_backlinks populated via pull_backlinks; gate so we never assert
|
|
871
|
+
// "no external links" without data) ──
|
|
872
|
+
{
|
|
873
|
+
id: 'backlinks-to-404', category: 'crawlability', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
874
|
+
title: 'External backlinks pointing to a dead page', fix: '301-redirect this URL to the best live equivalent — external link equity is hitting a 4xx/5xx page and being wasted.',
|
|
875
|
+
run: (c) => rows(c, `SELECT url_key urlKey, backlinks, referring_domains rd, status_code st FROM page_backlinks WHERE status_code >= 400 AND backlinks > 0 ORDER BY backlinks DESC`)
|
|
876
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { status: r.st, backlinks: r.backlinks, referringDomains: r.rd } })),
|
|
877
|
+
},
|
|
878
|
+
{
|
|
879
|
+
id: 'orphan-no-links', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
880
|
+
title: 'Orphan page — no internal or external links', fix: 'Add internal links (and earn external ones) — this indexable page has zero inlinks and no backlinks, so it depends on the sitemap alone.',
|
|
881
|
+
run: (c) => {
|
|
882
|
+
const has = c.db.prepare('SELECT COUNT(*) n FROM page_backlinks').get().n;
|
|
883
|
+
if (!has)
|
|
884
|
+
return []; // backlinks not pulled — can't credibly assert "no external links"
|
|
885
|
+
return rows(c, `SELECT p.url_key urlKey FROM pages p LEFT JOIN page_backlinks b ON b.url_key=p.url_key WHERE p.status_code=200 AND p.indexable=1 AND p.inlink_count=0 AND COALESCE(b.backlinks,0)=0`)
|
|
886
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { inlinks: 0, backlinks: 0 } }));
|
|
887
|
+
},
|
|
888
|
+
},
|
|
889
|
+
// ── Phase 6a — expert questions where crawl (intent) and reality diverge ──
|
|
890
|
+
{
|
|
891
|
+
id: 'ipr-bleed-by-status', category: 'crawlability', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
|
|
892
|
+
title: 'Internal link equity flowing into dead URLs', fix: 'Repoint or 301 these internal links — they target non-200 URLs and waste the internal PageRank of the (often high-authority) pages linking to them.',
|
|
893
|
+
// Sum the iPR of the SOURCE pages linking to each non-200 internal target. "Found a 404" is
|
|
894
|
+
// junior; "this 404 drains the equity of N high-iPR pages" is the director-level find.
|
|
895
|
+
run: (c) => rows(c, `SELECT l.target_key urlKey, COUNT(DISTINCT l.source_key) linkingPages,
|
|
896
|
+
ROUND(SUM(src.ipr), 1) wastedIpr, t.status_code st
|
|
897
|
+
FROM links l
|
|
898
|
+
JOIN pages src ON src.url_key = l.source_key
|
|
899
|
+
JOIN pages t ON t.url_key = l.target_key
|
|
900
|
+
WHERE l.is_internal = 1 AND t.status_code >= 400 AND t.status_code NOT IN (429,503)
|
|
901
|
+
GROUP BY l.target_key HAVING SUM(src.ipr) > 0
|
|
902
|
+
ORDER BY wastedIpr DESC`).map(r => ({ urlKey: r.urlKey, evidence: { status: r.st, linkingPages: r.linkingPages, wastedIpr: r.wastedIpr } })),
|
|
903
|
+
},
|
|
904
|
+
{
|
|
905
|
+
id: 'broken-canonical-target', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
906
|
+
title: 'Canonical points to a broken or unhealthy URL', fix: 'Point the canonical at a live, indexable, self-canonical HTTPS URL — Google ignores a canonical whose target is a 4xx/5xx/redirect, noindex, itself canonicalised elsewhere (a chain/loop), or an HTTPS→HTTP downgrade.',
|
|
907
|
+
// Join the declared canonical_key back to the crawl. Only flag when we crawled the target.
|
|
908
|
+
// Skip self-canonicals. Covers: non-200 target, noindex target, canonical chain/loop (target
|
|
909
|
+
// canonicalises onward), and HTTPS→HTTP downgrade (research: Sitebulb indexability hints).
|
|
910
|
+
run: (c) => rows(c, `SELECT p.url_key urlKey, p.canonical_url canon, p.canonical_key ck, t.status_code st, t.noindex ni, t.canonical_key tck
|
|
911
|
+
FROM pages p JOIN pages t ON t.url_key = p.canonical_key
|
|
912
|
+
WHERE p.canonical_key IS NOT NULL AND p.canonical_key != p.url_key AND p.status_code = 200
|
|
913
|
+
AND (t.status_code != 200 OR t.noindex = 1
|
|
914
|
+
OR (t.canonical_key IS NOT NULL AND t.canonical_key != t.url_key)
|
|
915
|
+
OR (p.url_key LIKE 'https://%' AND p.canonical_url LIKE 'http://%'))`)
|
|
916
|
+
.map(r => {
|
|
917
|
+
const reason = r.st !== 200 ? `target returns HTTP ${r.st}`
|
|
918
|
+
: r.ni ? 'target is noindex'
|
|
919
|
+
: (r.tck && r.tck !== r.ck) ? (r.tck === r.urlKey ? 'canonical loop (target points back here)' : 'canonical chain (target canonicalises onward)')
|
|
920
|
+
: 'HTTPS page canonicalises to an HTTP URL';
|
|
921
|
+
return { urlKey: r.urlKey, evidence: { canonical: r.canon, targetStatus: r.st, reason } };
|
|
922
|
+
}),
|
|
923
|
+
},
|
|
924
|
+
{
|
|
925
|
+
id: 'faceted-spider-trap', category: 'crawlability', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'global',
|
|
926
|
+
title: 'Indexable faceted URLs burning crawl budget', fix: 'noindex (or robots-disallow / canonicalise) multi-parameter filter URLs — they are indexable but earn zero search traffic, so they only waste crawl budget and risk index bloat.',
|
|
927
|
+
// Multi-parameter (>=2 params), indexable, zero GSC impressions in-window = classic facet trap.
|
|
928
|
+
// Gate on GSC so "zero search value" is a real claim, not just "no data".
|
|
929
|
+
run: (c) => {
|
|
930
|
+
if (!c.gscMaxDate)
|
|
931
|
+
return [];
|
|
932
|
+
return rows(c, `SELECT url_key urlKey, url FROM pages
|
|
933
|
+
WHERE status_code = 200 AND indexable = 1 AND url LIKE '%?%' AND url LIKE '%&%'
|
|
934
|
+
AND url_key NOT IN (SELECT page_key FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} AND impressions > 0)
|
|
935
|
+
ORDER BY url`).map(r => ({ urlKey: r.urlKey, evidence: { url: r.url, params: (r.url.split('?')[1] ?? '').split('&').map((kv) => kv.split('=')[0]).join(', '), note: 'indexable, multi-parameter, zero GSC impressions' } }));
|
|
936
|
+
},
|
|
937
|
+
},
|
|
938
|
+
{
|
|
939
|
+
id: 'soft-404-shell', category: 'indexation', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
940
|
+
title: 'Soft 404 — 200 OK but Google treats it as not-found', fix: 'Either populate the page with real content, or return a true 404/410 (or noindex) — it serves 200 but Google has flagged it as a soft 404.',
|
|
941
|
+
// URL Inspection page-fetch-state = soft 404 while the crawler sees a 200. The crawler alone
|
|
942
|
+
// would call this page fine; Google disagrees. Needs URL Inspection populated.
|
|
943
|
+
run: (c) => rows(c, `SELECT i.url_key urlKey, p.word_count wc, p.bytes bytes
|
|
944
|
+
FROM url_inspection i JOIN pages p ON p.url_key = i.url_key
|
|
945
|
+
WHERE LOWER(i.page_fetch_state) LIKE '%soft%' AND p.status_code = 200`).map(r => ({ urlKey: r.urlKey, evidence: { pageFetchState: 'soft 404', wordCount: r.wc, bytes: r.bytes } })),
|
|
946
|
+
},
|
|
947
|
+
{
|
|
948
|
+
id: 'rich-result-issues', category: 'schema', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
949
|
+
title: 'Google-verified rich result issues (URL Inspection)', fix: 'Fix the structured-data issues Google itself reports for this page — these come from the URL Inspection API (Google’s own validation), not our local validator, so they are authoritative. Address the listed issue messages per rich-result type.',
|
|
950
|
+
// Parses the stored richResultsResult JSON (url_inspection.rich_results) that inspect_urls
|
|
951
|
+
// already captures: detectedItems[].items[].issues[] carries Google's severity + message.
|
|
952
|
+
// Needs URL Inspection populated (run inspect_urls). Zero API cost — data is already in the DB.
|
|
953
|
+
run: (c) => {
|
|
954
|
+
const out = [];
|
|
955
|
+
for (const r of iterRows(c, `SELECT url_key urlKey, rich_results rr FROM url_inspection WHERE rich_results IS NOT NULL AND rich_results != ''`)) {
|
|
956
|
+
let parsed;
|
|
957
|
+
try {
|
|
958
|
+
parsed = JSON.parse(r.rr);
|
|
959
|
+
}
|
|
960
|
+
catch {
|
|
961
|
+
continue;
|
|
962
|
+
}
|
|
963
|
+
// Dedup per (type, severity, message) with a count — a listicle repeats the same
|
|
964
|
+
// "Missing field review" warning per product item; keep one sample item per group.
|
|
965
|
+
const groups = new Map();
|
|
966
|
+
for (const det of parsed?.detectedItems ?? []) {
|
|
967
|
+
for (const item of det?.items ?? []) {
|
|
968
|
+
for (const iss of item?.issues ?? []) {
|
|
969
|
+
if (!iss?.issueMessage)
|
|
970
|
+
continue;
|
|
971
|
+
const key = `${det.richResultType ?? 'unknown'}|${iss.severity ?? ''}|${iss.issueMessage}`;
|
|
972
|
+
const g = groups.get(key);
|
|
973
|
+
if (g)
|
|
974
|
+
g.count++;
|
|
975
|
+
else
|
|
976
|
+
groups.set(key, { richResultType: det.richResultType ?? 'unknown', severity: iss.severity ?? null, message: iss.issueMessage, count: 1, sampleItem: item.name ?? null });
|
|
977
|
+
}
|
|
978
|
+
}
|
|
979
|
+
}
|
|
980
|
+
if (groups.size)
|
|
981
|
+
out.push({ urlKey: r.urlKey, evidence: { verdict: parsed?.verdict ?? null, issues: [...groups.values()] } });
|
|
982
|
+
}
|
|
983
|
+
return out;
|
|
984
|
+
},
|
|
985
|
+
},
|
|
986
|
+
{
|
|
987
|
+
id: 'broken-hreflang-target', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
988
|
+
title: 'hreflang points to a broken/non-indexable URL', fix: 'Point each hreflang alternate at a live, indexable URL — Google drops the whole cluster if an alternate is 4xx/5xx/redirect/noindex.',
|
|
989
|
+
run: (c) => hreflangFindings(c).broken,
|
|
990
|
+
},
|
|
991
|
+
{
|
|
992
|
+
id: 'hreflang-no-return-tag', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
993
|
+
title: 'hreflang missing return tag (not reciprocated)', fix: 'Add the reciprocal hreflang on the target page — Google ignores one-way hreflang annotations that don’t link back.',
|
|
994
|
+
run: (c) => hreflangFindings(c).noReturn,
|
|
995
|
+
},
|
|
996
|
+
// ── 6a finishers — consume persisted DataForSEO enrichments (gated; need the data pulled) ──
|
|
997
|
+
{
|
|
998
|
+
id: 'intent-vs-pagetype-mismatch', category: 'merged', severity: 'med', labels: ['G', 'N'], certainty: 0.7, effortBase: 8, fixType: 'per-page',
|
|
999
|
+
title: 'Page type mismatches its query intent', fix: 'Reformat or retarget the page — its template doesn’t match the SERP intent for its top query (e.g. a product page ranking for an informational query, or an article for a transactional one). Run search_intent siteUrl:<property> to populate intents.',
|
|
1000
|
+
// Join each page's top GSC query → persisted keyword_intent → page schema flavour (from json_ld).
|
|
1001
|
+
// Judgement (N, 0.7): intent + schema-type inference is heuristic, so it only runs with includeJudgement.
|
|
1002
|
+
run: (c) => {
|
|
1003
|
+
if (!c.gscMaxDate)
|
|
1004
|
+
return [];
|
|
1005
|
+
if ((c.db.prepare('SELECT COUNT(*) n FROM keyword_intent').get().n) === 0)
|
|
1006
|
+
return []; // intents not pulled
|
|
1007
|
+
const r = rows(c, `SELECT s.page_key urlKey, s.query query, ki.intent intent, p.json_ld jsonLd FROM
|
|
1008
|
+
(SELECT page_key, query, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
|
|
1009
|
+
FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
|
|
1010
|
+
JOIN pages p ON p.url_key = s.page_key
|
|
1011
|
+
JOIN keyword_intent ki ON ki.keyword = LOWER(s.query)
|
|
1012
|
+
WHERE s.rn = 1 AND p.status_code = 200 AND p.json_ld IS NOT NULL AND p.json_ld != ''`);
|
|
1013
|
+
const out = [];
|
|
1014
|
+
for (const x of r) {
|
|
1015
|
+
const types = parseJsonLdNodes(x.jsonLd).map(n => (nodeType(n) ?? '').toLowerCase());
|
|
1016
|
+
const productish = types.some(t => t === 'product' || t === 'offer');
|
|
1017
|
+
const articleish = types.some(t => /article|blogposting|newsarticle/.test(t));
|
|
1018
|
+
const intent = (x.intent || '').toLowerCase();
|
|
1019
|
+
let mismatch = null;
|
|
1020
|
+
if (productish && intent === 'informational')
|
|
1021
|
+
mismatch = 'product/offer page ranking for an informational query';
|
|
1022
|
+
else if (articleish && intent === 'transactional')
|
|
1023
|
+
mismatch = 'article page ranking for a transactional query';
|
|
1024
|
+
if (mismatch)
|
|
1025
|
+
out.push({ urlKey: x.urlKey, evidence: { topQuery: x.query, queryIntent: intent, mismatch } });
|
|
1026
|
+
}
|
|
1027
|
+
return out;
|
|
1028
|
+
},
|
|
1029
|
+
},
|
|
1030
|
+
{
|
|
1031
|
+
id: 'high-yield-cwv-fail', category: 'performance', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 8, fixType: 'per-page',
|
|
1032
|
+
title: 'High-traffic page failing Core Web Vitals', fix: 'Prioritise CWV work here — this page earns real clicks but fails lab Core Web Vitals (LCP > 2.5s, CLS > 0.1, or performance < 50), so engineering effort has clear ROI. Run page_lighthouse siteUrl:<property> on key URLs to populate CWV.',
|
|
1033
|
+
run: (c) => {
|
|
1034
|
+
if (!c.gscMaxDate)
|
|
1035
|
+
return [];
|
|
1036
|
+
if ((c.db.prepare('SELECT COUNT(*) n FROM page_cwv').get().n) === 0)
|
|
1037
|
+
return []; // CWV not pulled
|
|
1038
|
+
return rows(c, `SELECT cw.url_key urlKey, cw.lcp_ms lcp, cw.cls cls, cw.performance perf,
|
|
1039
|
+
(SELECT COALESCE(SUM(clicks),0) FROM search_analytics sa WHERE sa.page_key = cw.url_key AND ${win(c.gscMaxDate)}) clicks
|
|
1040
|
+
FROM page_cwv cw
|
|
1041
|
+
WHERE (cw.lcp_ms > 2500 OR cw.cls > 0.1 OR cw.performance < 0.5)`)
|
|
1042
|
+
.filter(r => r.clicks > 0)
|
|
1043
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { clicks: r.clicks, lcpMs: r.lcp != null ? Math.round(r.lcp) : null, cls: r.cls != null ? Math.round(r.cls * 1000) / 1000 : null, performance: r.perf != null ? Math.round(r.perf * 100) : null } }));
|
|
1044
|
+
},
|
|
1045
|
+
},
|
|
1046
|
+
// ── 6b — per-template systemic issues (template-typed, deterministic) ──
|
|
1047
|
+
{
|
|
1048
|
+
id: 'pagination-canonical-to-page-1', category: 'crawlability', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
|
|
1049
|
+
title: 'Paginated pages canonicalising away from themselves', fix: 'Make each paginated page (page 2, 3, …) self-canonical. Canonicalising page 2+ back to page 1 tells Google the deeper pages are duplicates, so products/articles linked only from page 2+ drop out of the crawl.',
|
|
1050
|
+
run: (c) => rows(c, `SELECT url_key urlKey, url, canonical_url canon FROM pages
|
|
1051
|
+
WHERE status_code=200 AND canonical_count>0 AND canonical_key IS NOT NULL AND canonical_key != url_key
|
|
1052
|
+
AND (rel_prev=1 OR url LIKE '%/page/%' OR url GLOB '*[?&]page=[0-9]*' OR url GLOB '*[?&]paged=[0-9]*' OR url GLOB '*[?&]p=[0-9]*')`)
|
|
1053
|
+
.filter(r => !/[?&](page|p)=1(\b|&|$)/.test(r.url) && !/\/page\/1(\/?$|\?)/.test(r.url)) // page 1 self-canonicalising to base is fine
|
|
1054
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { url: r.url, canonical: r.canon } })),
|
|
1055
|
+
},
|
|
1056
|
+
{
|
|
1057
|
+
id: 'article-date-illogical', category: 'schema', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
1058
|
+
title: 'Article dateModified earlier than datePublished', fix: 'Fix the Article schema dates — dateModified must be on or after datePublished. An impossible date undermines trust in the markup and can suppress the freshness signal.',
|
|
1059
|
+
run: (c) => {
|
|
1060
|
+
const out = [];
|
|
1061
|
+
for (const r of rows(c, `SELECT url_key urlKey, json_ld jsonLd FROM pages WHERE status_code=200 AND json_ld LIKE '%Article%' AND json_ld LIKE '%date%'`)) {
|
|
1062
|
+
for (const n of parseJsonLdNodes(r.jsonLd)) {
|
|
1063
|
+
const t = nodeType(n);
|
|
1064
|
+
if (!t || !/article|blogposting|newsarticle/i.test(t))
|
|
1065
|
+
continue;
|
|
1066
|
+
const pub = Date.parse(n.datePublished), mod = Date.parse(n.dateModified);
|
|
1067
|
+
if (!Number.isNaN(pub) && !Number.isNaN(mod) && mod < pub) {
|
|
1068
|
+
out.push({ urlKey: r.urlKey, evidence: { datePublished: n.datePublished, dateModified: n.dateModified } });
|
|
1069
|
+
break;
|
|
1070
|
+
}
|
|
1071
|
+
}
|
|
1072
|
+
}
|
|
1073
|
+
return out;
|
|
1074
|
+
},
|
|
1075
|
+
},
|
|
1076
|
+
// ── 6d — Wikidata entity layer (heuristic H1→QID; N/judgement, gated on resolve_entities) ──
|
|
1077
|
+
{
|
|
1078
|
+
id: 'entity-internal-link-gap', category: 'crawlability', severity: 'low', labels: ['N'], certainty: 0.5, effortBase: 3, fixType: 'per-page',
|
|
1079
|
+
title: 'Topically related pages not internally linked', fix: 'Add an internal link from the broader page to the more specific one — Wikidata says their entities are related (subclass-of / part-of) but no internal link connects them, leaving a gap in the topical mesh. Run resolve_entities first; verify the entity match before acting (heuristic).',
|
|
1080
|
+
run: (c) => {
|
|
1081
|
+
if ((c.db.prepare('SELECT COUNT(*) n FROM page_entity').get().n) === 0)
|
|
1082
|
+
return []; // not resolved
|
|
1083
|
+
return rows(c, `SELECT parent.url_key urlKey, child.url_key target, parent.label pl, child.label cl, ee.relation rel
|
|
1084
|
+
FROM entity_edge ee
|
|
1085
|
+
JOIN page_entity child ON child.qid = ee.qid
|
|
1086
|
+
JOIN page_entity parent ON parent.qid = ee.related_qid
|
|
1087
|
+
WHERE parent.url_key != child.url_key
|
|
1088
|
+
AND NOT EXISTS (SELECT 1 FROM links l WHERE l.source_key = parent.url_key AND l.target_key = child.url_key AND l.is_internal = 1)`)
|
|
1089
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { suggestLinkTo: r.target, parentEntity: r.pl, childEntity: r.cl, relation: r.rel } }));
|
|
1090
|
+
},
|
|
1091
|
+
},
|
|
1092
|
+
// ── Sitemap ↔ crawl reconciliation (gated on a sitemap having been fetched) ──
|
|
1093
|
+
{
|
|
1094
|
+
id: 'sitemap-non-indexable', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
|
|
1095
|
+
title: 'Sitemap lists non-indexable URLs', fix: 'Remove URLs from the XML sitemap that are 4xx/5xx, redirected, noindex, canonicalised or robots-blocked — the sitemap should list only canonical, indexable pages, or Google loses trust in it.',
|
|
1096
|
+
run: (c) => {
|
|
1097
|
+
if (!sitemapHasRows(c))
|
|
1098
|
+
return [];
|
|
1099
|
+
return rows(c, `SELECT s.url_key urlKey, p.indexable_reason reason, p.status_code st FROM sitemap_urls s JOIN pages p ON p.url_key = s.url_key WHERE p.indexable = 0`)
|
|
1100
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { reason: r.reason ?? 'not-indexable', status: r.st } }));
|
|
1101
|
+
},
|
|
1102
|
+
},
|
|
1103
|
+
{
|
|
1104
|
+
id: 'indexable-not-in-sitemap', category: 'indexation', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
|
|
1105
|
+
title: 'Indexable pages missing from the sitemap', fix: 'Add these indexable pages to the XML sitemap so Google discovers and prioritises them.',
|
|
1106
|
+
run: (c) => {
|
|
1107
|
+
if (!sitemapHasRows(c))
|
|
1108
|
+
return [];
|
|
1109
|
+
return rows(c, `SELECT p.url_key urlKey FROM pages p LEFT JOIN sitemap_urls s ON s.url_key = p.url_key WHERE p.status_code = 200 AND p.indexable = 1 AND ${notPagination('p.url_key')} AND s.url_key IS NULL`)
|
|
1110
|
+
.map(r => ({ urlKey: r.urlKey, evidence: {} }));
|
|
1111
|
+
},
|
|
1112
|
+
},
|
|
1113
|
+
{
|
|
1114
|
+
id: 'sitemap-orphan', category: 'crawlability', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
1115
|
+
title: 'Sitemap page with no internal links', fix: 'Add internal links — this page is in the sitemap but nothing links to it, so it relies on the sitemap alone for discovery and earns little internal authority.',
|
|
1116
|
+
run: (c) => {
|
|
1117
|
+
if (!sitemapHasRows(c))
|
|
1118
|
+
return [];
|
|
1119
|
+
return rows(c, `SELECT p.url_key urlKey FROM pages p JOIN sitemap_urls s ON s.url_key = p.url_key WHERE p.status_code = 200 AND p.indexable = 1 AND p.inlink_count = 0`)
|
|
1120
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { inlinks: 0, note: 'in sitemap, no internal links' } }));
|
|
1121
|
+
},
|
|
1122
|
+
},
|
|
1123
|
+
// ── Cheap performance proxies (from data captured at crawl time — no extra fetch) ──
|
|
1124
|
+
{
|
|
1125
|
+
id: 'slow-response', category: 'performance', severity: 'med', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'global',
|
|
1126
|
+
title: 'Slow server response (TTFB proxy)', fix: 'Investigate slow server/TTFB — caching, CDN, or backend. Response time over ~1.5s hurts Core Web Vitals and crawl rate.',
|
|
1127
|
+
run: (c) => rows(c, `SELECT url_key urlKey, response_time_ms ms FROM pages WHERE status_code=200 AND ${HTML_CT} AND response_time_ms > 1500 ORDER BY response_time_ms DESC`)
|
|
1128
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { responseMs: r.ms } })),
|
|
1129
|
+
},
|
|
1130
|
+
{
|
|
1131
|
+
id: 'large-html', category: 'performance', severity: 'low', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
1132
|
+
title: 'Large HTML document (transferred)', fix: 'Trim the HTML payload — bloated markup slows render and First Contentful Paint (often huge inline SVG/CSS/JSON or unminified output). Judged on TRANSFERRED bytes, not raw: behind a compressing CDN (brotli/gzip) raw size matters far less.',
|
|
1133
|
+
// Severity is on what the browser actually downloads, not raw bytes: a 200KB page served brotli
|
|
1134
|
+
// is ~45KB over the wire and is NOT a real perf problem. We estimate transfer size (raw × ~0.22
|
|
1135
|
+
// for br/gzip, else raw) and only flag pages whose ESTIMATED transferred HTML exceeds ~60KB.
|
|
1136
|
+
// Prefilter at the 60KB threshold itself, not higher — an UNCOMPRESSED 60–150KB page
|
|
1137
|
+
// (est = raw) is exactly the case that matters most and must reach the est filter.
|
|
1138
|
+
run: (c) => rows(c, `SELECT url_key urlKey, bytes, content_encoding enc FROM pages WHERE status_code=200 AND ${HTML_CT} AND bytes > 60000 ORDER BY bytes DESC`)
|
|
1139
|
+
.map(r => { const compressed = /br|gzip|deflate|zstd/i.test(r.enc || ''); const est = compressed ? Math.round(r.bytes * 0.22) : r.bytes; return { urlKey: r.urlKey, est, compressed, raw: r.bytes }; })
|
|
1140
|
+
.filter(x => x.est > 60000)
|
|
1141
|
+
.map(x => ({ urlKey: x.urlKey, evidence: { rawBytes: x.raw, estTransferBytes: x.est, compressed: x.compressed, note: x.compressed ? 'raw HTML; estimated transferred size after CDN compression' : 'served UNCOMPRESSED — enable brotli/gzip' } })),
|
|
1142
|
+
},
|
|
1143
|
+
{
|
|
1144
|
+
id: 'uncompressed-html', category: 'performance', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
1145
|
+
title: 'HTML served without compression', fix: 'Enable gzip or brotli for HTML responses — uncompressed HTML wastes bandwidth and slows load. Usually a one-line server/CDN setting.',
|
|
1146
|
+
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (content_encoding IS NULL OR content_encoding='')`)
|
|
1147
|
+
.map(r => ({ urlKey: r.urlKey, evidence: {} })),
|
|
1148
|
+
},
|
|
1149
|
+
// ── Checklist-coverage additions (2026-07-20 — see plan/checklist-coverage.md) ──
|
|
1150
|
+
{
|
|
1151
|
+
id: 'robots-blocked-with-traffic', category: 'crawlability', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
1152
|
+
title: 'Robots-blocked page still earning search traffic', fix: 'This URL is disallowed in robots.txt yet Google still shows it (usually as a bare "no information" result) and users still land on it. Either unblock it so it can be crawled and ranked properly, or — if it genuinely shouldn\'t be found — unblock it AND add noindex (a robots-blocked page can never see the noindex).',
|
|
1153
|
+
run: (c) => !c.gscMaxDate ? [] : rows(c, `SELECT p.url_key urlKey, SUM(sa.clicks) clicks, SUM(sa.impressions) impressions
|
|
1154
|
+
FROM pages p JOIN search_analytics sa ON sa.page_key = p.url_key
|
|
1155
|
+
WHERE p.indexable_reason='robots-disallowed' AND ${win(c.gscMaxDate)}
|
|
1156
|
+
GROUP BY p.url_key HAVING SUM(sa.impressions) >= 10
|
|
1157
|
+
ORDER BY SUM(sa.impressions) DESC LIMIT 50`)
|
|
1158
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { clicks: r.clicks, impressions: r.impressions, note: 'disallowed in robots.txt yet earning impressions' } })),
|
|
1159
|
+
},
|
|
1160
|
+
{
|
|
1161
|
+
id: 'missing-lang', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
1162
|
+
title: 'Pages missing an html lang attribute', fix: 'Declare the page language on the <html> element (e.g. lang="en-GB") — it helps search engines, screen readers and translation systems know what language they\'re reading. Usually one template edit.',
|
|
1163
|
+
run: (c) => {
|
|
1164
|
+
const n = c.db.prepare(`SELECT COUNT(*) n FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND (lang IS NULL OR TRIM(lang)='')`).get().n;
|
|
1165
|
+
return n > 0 ? [{ urlKey: null, evidence: { pages: n, note: 'indexable pages with no lang attribute' } }] : [];
|
|
1166
|
+
},
|
|
1167
|
+
},
|
|
1168
|
+
{
|
|
1169
|
+
id: 'image-preview-restricted', category: 'indexation', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
1170
|
+
title: 'max-image-preview restricted below large', fix: 'The robots meta tag caps image previews at none/standard. Google Discover strongly favours large image previews — set max-image-preview:large (or remove the restriction) unless there\'s a licensing reason not to.',
|
|
1171
|
+
run: (c) => rows(c, `SELECT url_key urlKey, robots FROM pages WHERE status_code=200 AND indexable=1
|
|
1172
|
+
AND (LOWER(REPLACE(robots,' ','')) LIKE '%max-image-preview:none%' OR LOWER(REPLACE(robots,' ','')) LIKE '%max-image-preview:standard%') LIMIT 50`)
|
|
1173
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { robots: r.robots } })),
|
|
1174
|
+
},
|
|
1175
|
+
{
|
|
1176
|
+
id: 'excessive-links', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
1177
|
+
title: 'Excessive number of links on the page', fix: 'Hundreds of links on one page dilute the equity each one passes and drown the ones that matter. Trim boilerplate link blocks (mega-menus, tag clouds, footer sprawl) so the important links stand out.',
|
|
1178
|
+
run: (c) => rows(c, `SELECT url_key urlKey, internal_links il, external_links el FROM pages
|
|
1179
|
+
WHERE status_code=200 AND indexable=1 AND (internal_links + external_links) > 300
|
|
1180
|
+
ORDER BY (internal_links + external_links) DESC LIMIT 30`)
|
|
1181
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { internalLinks: r.il, externalLinks: r.el, total: r.il + r.el } })),
|
|
1182
|
+
},
|
|
1183
|
+
{
|
|
1184
|
+
id: 'favicon-missing', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
1185
|
+
title: 'No favicon declared', fix: 'Add a favicon (<link rel="icon" …>) — Google shows it next to your result on mobile, and a missing one costs a little trust/recognition on every SERP appearance. One line in the template.',
|
|
1186
|
+
// Gated on the column being populated (has_favicon is NULL on crawls from before this
|
|
1187
|
+
// was captured) — never flag from a pre-feature crawl.
|
|
1188
|
+
run: (c) => {
|
|
1189
|
+
const hp = c.db.prepare(`SELECT url_key, has_favicon hf FROM pages WHERE status_code=200 AND has_favicon IS NOT NULL ORDER BY (click_depth=0) DESC, inlink_count DESC LIMIT 1`).get();
|
|
1190
|
+
return hp && hp.hf === 0 ? [{ urlKey: hp.url_key, evidence: { note: 'no <link rel="icon"> on the homepage' } }] : [];
|
|
1191
|
+
},
|
|
1192
|
+
},
|
|
1193
|
+
{
|
|
1194
|
+
id: 'sitemap-lastmod-untrustworthy', category: 'indexation', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
1195
|
+
title: 'Sitemap lastmod dates are not trustworthy', fix: 'Google uses <lastmod> to prioritise recrawling — but only while it stays honest; a sitemap that stamps everything with the generation date (or future dates) teaches Google to ignore yours. Make lastmod reflect the last genuine content change, or drop it entirely.',
|
|
1196
|
+
run: (c) => {
|
|
1197
|
+
const all = c.db.prepare(`SELECT url_key, lastmod FROM sitemap_urls WHERE lastmod IS NOT NULL AND lastmod <> ''`).all();
|
|
1198
|
+
if (all.length < 20)
|
|
1199
|
+
return []; // too few dated URLs to judge the pattern
|
|
1200
|
+
const ev = { urlsWithLastmod: all.length };
|
|
1201
|
+
let tripped = false;
|
|
1202
|
+
// Tell 1: a generator stamping every URL with "now" — >90% share one date, and that date is recent.
|
|
1203
|
+
const byDay = new Map();
|
|
1204
|
+
for (const r of all)
|
|
1205
|
+
byDay.set(r.lastmod.slice(0, 10), (byDay.get(r.lastmod.slice(0, 10)) ?? 0) + 1);
|
|
1206
|
+
const [topDay, topN] = [...byDay.entries()].sort((a, b) => b[1] - a[1])[0];
|
|
1207
|
+
const ageDays = (Date.now() - Date.parse(topDay)) / 86400000;
|
|
1208
|
+
if (topN / all.length > 0.9 && Number.isFinite(ageDays) && ageDays < 35) {
|
|
1209
|
+
tripped = true;
|
|
1210
|
+
ev.sharedStamp = `${Math.round(topN / all.length * 100)}% of URLs claim ${topDay} — a generation timestamp, not a change date`;
|
|
1211
|
+
}
|
|
1212
|
+
// Tell 2: dates in the future.
|
|
1213
|
+
const future = all.filter(r => Date.parse(r.lastmod) > Date.now() + 86400000).length;
|
|
1214
|
+
if (future > 0) {
|
|
1215
|
+
tripped = true;
|
|
1216
|
+
ev.futureDates = future;
|
|
1217
|
+
}
|
|
1218
|
+
// Tell 3: lastmod claims a change between our two most recent crawls, yet none of the
|
|
1219
|
+
// tracked page fields (status, title, meta, H1, word count, schema types) changed.
|
|
1220
|
+
const crawls = latestTwoCrawls(c.db);
|
|
1221
|
+
if (crawls.length === 2) {
|
|
1222
|
+
// Compare DATE prefixes on both sides — lastmod is stored verbatim and often a full
|
|
1223
|
+
// timestamp; compared raw against a 10-char date, same-day stamps sort "after" the
|
|
1224
|
+
// boundary and are silently excluded (exactly the stamp-everything-today pattern).
|
|
1225
|
+
const phantom = c.db.prepare(`SELECT s.url_key FROM sitemap_urls s
|
|
1226
|
+
JOIN page_snapshots n ON n.url_key = s.url_key AND n.crawl_id = ?
|
|
1227
|
+
JOIN page_snapshots o ON o.url_key = s.url_key AND o.crawl_id = ?
|
|
1228
|
+
WHERE substr(s.lastmod,1,10) > substr(?,1,10) AND substr(s.lastmod,1,10) <= substr(?,1,10)
|
|
1229
|
+
AND n.status_code IS o.status_code AND n.title IS o.title AND n.meta_description IS o.meta_description
|
|
1230
|
+
AND n.h1 IS o.h1 AND n.word_count IS o.word_count AND n.schema_types IS o.schema_types
|
|
1231
|
+
LIMIT 200`).all(crawls[0].crawl_id, crawls[1].crawl_id, crawls[1].at, crawls[0].at);
|
|
1232
|
+
if (phantom.length >= 5) {
|
|
1233
|
+
tripped = true;
|
|
1234
|
+
ev.phantomChanges = phantom.length;
|
|
1235
|
+
ev.phantomExamples = phantom.slice(0, 5).map(p => p.url_key);
|
|
1236
|
+
}
|
|
1237
|
+
}
|
|
1238
|
+
return tripped ? [{ urlKey: null, evidence: ev }] : [];
|
|
1239
|
+
},
|
|
1240
|
+
},
|
|
1241
|
+
{
|
|
1242
|
+
id: 'no-304-revalidation', category: 'performance', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
1243
|
+
title: 'Server ignores conditional requests (no 304)', fix: 'Pages advertise Last-Modified/ETag but the server re-serves a full 200 when asked "has this changed?" (If-Modified-Since / If-None-Match). A properly configured server answers 304 Not Modified — it saves bandwidth on every revalidating crawler and cache, and signals stability to Googlebot. Usually a server/CDN setting.',
|
|
1244
|
+
// Populated by the crawler's post-crawl probe (pages.conditional_304); NULL-gated so
|
|
1245
|
+
// pre-feature crawls never flag. Fires only when NO probed page honoured the request.
|
|
1246
|
+
run: (c) => {
|
|
1247
|
+
const r = c.db.prepare(`SELECT COUNT(*) probed, COALESCE(SUM(conditional_304),0) ok FROM pages WHERE conditional_304 IS NOT NULL`).get();
|
|
1248
|
+
if (r.probed < 5 || r.ok > 0)
|
|
1249
|
+
return [];
|
|
1250
|
+
const noValidators = c.db.prepare(`SELECT COUNT(*) n FROM pages WHERE status_code=200 AND ${HTML_CT} AND last_modified IS NULL AND etag IS NULL`).get().n;
|
|
1251
|
+
return [{ urlKey: null, evidence: { probed: r.probed, honoured304: 0, pagesWithoutValidators: noValidators, note: 'every conditional re-request returned a full 200' } }];
|
|
1252
|
+
},
|
|
1253
|
+
},
|
|
1254
|
+
{
|
|
1255
|
+
id: 'analytics-missing', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
1256
|
+
title: 'No client-side analytics detected', fix: 'No analytics or tag-manager snippet was found on any crawled page (GA4, GTM, Plausible, Matomo, Fathom, Clarity…). If you measure server-side, ignore this; otherwise you\'re flying blind — install an analytics package before making SEO decisions.',
|
|
1257
|
+
// Fires only when EVERY populated page lacks a snippet — one page with analytics = installed.
|
|
1258
|
+
run: (c) => {
|
|
1259
|
+
const r = c.db.prepare(`SELECT COUNT(*) total, COALESCE(SUM(has_analytics),0) withA FROM pages WHERE status_code=200 AND ${HTML_CT} AND has_analytics IS NOT NULL`).get();
|
|
1260
|
+
return r.total >= 3 && r.withA === 0 ? [{ urlKey: null, evidence: { pagesChecked: r.total, note: 'no known analytics snippet on any crawled page (server-side measurement is invisible to a crawl)' } }] : [];
|
|
1261
|
+
},
|
|
1262
|
+
},
|
|
1263
|
+
// ── Google "AI features / succeeding in AI search" guide (developers.google.com/search/docs/
|
|
1264
|
+
// fundamentals/ai-optimization-guide, read 2026-08-02) — principle → check mapping:
|
|
1265
|
+
// • Unique/compelling/non-commodity, people-first content .... thin-content, content-bloat,
|
|
1266
|
+
// ai-slop-signals (NEW below — mechanical/templated prose is the anti-signal of "unique take")
|
|
1267
|
+
// • Unique point of view / first-hand experience .............. article-no-author (NEW below —
|
|
1268
|
+
// unattributed articles are the deterministically checkable slice), stale-content
|
|
1269
|
+
// • Clear organisation (paragraphs/sections/headings) ......... poor-chunkability, content-bloat,
|
|
1270
|
+
// heading-hierarchy; answer coverage: body-missing-top-query, rag-answer-gap,
|
|
1271
|
+
// answer-not-front-loaded, weak-passage-answer, low-extractability
|
|
1272
|
+
// • Indexed + snippet-eligible / technical requirements ....... indexation family (noindex,
|
|
1273
|
+
// canonical, robots-blocked-with-traffic), soft-404-shell, sitemap reconciliation
|
|
1274
|
+
// • Crawlable public content .................................. crawlability family, broken links,
|
|
1275
|
+
// redirect chains; freshness-honesty (sitemap-lastmod-untrustworthy, no-304-revalidation)
|
|
1276
|
+
// • Semantic HTML / parseable ................................. heading-hierarchy, missing-h1,
|
|
1277
|
+
// multiple-h1, missing-lang
|
|
1278
|
+
// • Reduce duplicate content .................................. duplicate-title/meta, canonical
|
|
1279
|
+
// family, faceted-spider-trap, keyword-cannibalisation
|
|
1280
|
+
// • Images/video supporting text .............................. image-alt, images-missing-dimensions
|
|
1281
|
+
// • Page experience / latency ................................. performance proxies, high-yield-cwv-fail
|
|
1282
|
+
// • Structured data honesty (not required, but keep it valid) . schema-validate family,
|
|
1283
|
+
// article-date-illogical
|
|
1284
|
+
// • Don't chunk artificially / rewrite for AI ................. covered by NOT having such checks;
|
|
1285
|
+
// our chunk checks reward structure for humans, not tiny AI fragments
|
|
1286
|
+
// • "Don't create llms.txt" ................................... tension with agent-readiness probes,
|
|
1287
|
+
// which score llms.txt for *agent* (not Google AI-search) consumption — left as-is, different audience
|
|
1288
|
+
// • Merchant Center / Business Profile / GenAI report ......... out of scope (not crawl/GSC data)
|
|
1289
|
+
{
|
|
1290
|
+
// Anti-signal of the guide's "unique, non-commodity, people-first content": prose that reads
|
|
1291
|
+
// machine-generated. Three heuristics over body_chunks — slop-lexicon density, sentence-length
|
|
1292
|
+
// uniformity (low variance = mechanical), repeated chunk openers (page-internal boilerplate).
|
|
1293
|
+
// Precision over recall: absolute floors on every signal, ≥2 signals required, AND the composite
|
|
1294
|
+
// score must sit in the top decile of pages showing any signal. Judgement-gated — a human wrote
|
|
1295
|
+
// "delve" long before LLMs did.
|
|
1296
|
+
id: 'ai-slop-signals', category: 'content', severity: 'low', labels: ['N'], certainty: 0.5, effortBase: 5, fixType: 'per-page',
|
|
1297
|
+
title: 'Prose shows machine-generated (slop) signals', fix: 'This page’s copy trips several statistical tells of generic AI-generated text: stock filler phrases, unusually uniform sentence lengths, and/or sections that all open the same way. Google’s AI-search guidance rewards unique, people-first content with a first-hand point of view — rewrite the flagged sections with specifics only you can supply (real experience, real numbers, real opinions) and cut the filler. (Heuristic — verify by reading the page; competent human writing can trip these tells.)',
|
|
1298
|
+
run: (c) => {
|
|
1299
|
+
const PHRASES = [
|
|
1300
|
+
'in today’s fast-paced world', "in today's fast-paced world", 'in today’s digital age', "in today's digital age",
|
|
1301
|
+
'it’s important to note', "it's important to note", 'it is important to note', 'in conclusion',
|
|
1302
|
+
'harness the power', 'unlock the potential', 'look no further', 'in the realm of', 'navigate the complexities',
|
|
1303
|
+
'a testament to', 'plays a crucial role', 'a wide range of', 'when it comes to', 'at the end of the day',
|
|
1304
|
+
'whether you’re a', "whether you're a", 'let’s dive', "let's dive", 'dive into the world of',
|
|
1305
|
+
'elevate your', 'take your * to the next level', 'game-changer', 'game changer', 'cutting-edge', 'ever-evolving',
|
|
1306
|
+
'treasure trove', 'rich tapestry', 'delve', 'delving', 'seamlessly', 'revolutionize', 'revolutionise', 'unleash',
|
|
1307
|
+
].filter(p => !p.includes('*'));
|
|
1308
|
+
const metrics = [];
|
|
1309
|
+
for (const x of iterRows(c, `SELECT url_key urlKey, word_count wc, body_chunks bc FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND word_count >= 300 AND body_chunks IS NOT NULL`)) {
|
|
1310
|
+
let chunks;
|
|
1311
|
+
try {
|
|
1312
|
+
chunks = JSON.parse(x.bc);
|
|
1313
|
+
}
|
|
1314
|
+
catch {
|
|
1315
|
+
continue;
|
|
1316
|
+
}
|
|
1317
|
+
const texts = chunks.map((k) => String(k.text || '')).filter(t => t.length > 0);
|
|
1318
|
+
if (!texts.length)
|
|
1319
|
+
continue;
|
|
1320
|
+
const body = texts.join(' ').toLowerCase();
|
|
1321
|
+
const bodyWords = body.split(/\s+/).filter(Boolean).length;
|
|
1322
|
+
if (bodyWords < 250)
|
|
1323
|
+
continue;
|
|
1324
|
+
// (i) slop-lexicon density (hits per 1000 words, ≥3 distinct terms required)
|
|
1325
|
+
const hits = new Map();
|
|
1326
|
+
for (const p of PHRASES) {
|
|
1327
|
+
let n = 0, i = -1;
|
|
1328
|
+
while ((i = body.indexOf(p, i + 1)) !== -1)
|
|
1329
|
+
n++;
|
|
1330
|
+
if (n)
|
|
1331
|
+
hits.set(p, n);
|
|
1332
|
+
}
|
|
1333
|
+
const totalHits = [...hits.values()].reduce((a, b) => a + b, 0);
|
|
1334
|
+
const density = totalHits / bodyWords * 1000;
|
|
1335
|
+
const lexSignal = hits.size >= 3 && density >= 2.5;
|
|
1336
|
+
// (ii) sentence-length uniformity — coefficient of variation over sentence word-counts
|
|
1337
|
+
const sentences = body.split(/[.!?]+\s/).map(s => s.split(/\s+/).filter(Boolean).length).filter(n => n >= 5 && n <= 60);
|
|
1338
|
+
let cv = null;
|
|
1339
|
+
if (sentences.length >= 12) {
|
|
1340
|
+
const mean = sentences.reduce((a, b) => a + b, 0) / sentences.length;
|
|
1341
|
+
cv = Math.sqrt(sentences.reduce((a, b) => a + (b - mean) ** 2, 0) / sentences.length) / mean;
|
|
1342
|
+
}
|
|
1343
|
+
const uniformSignal = cv !== null && cv < 0.28;
|
|
1344
|
+
// (iii) repeated openers across the page's own chunks (first 3 words, ≥3 chunks sharing one)
|
|
1345
|
+
const openers = new Map();
|
|
1346
|
+
if (texts.length >= 5)
|
|
1347
|
+
for (const t of texts) {
|
|
1348
|
+
// Letters only — numeric/UI-chrome openers ("Show 10 20…") are widget text, not prose.
|
|
1349
|
+
const o = t.toLowerCase().replace(/[^a-z\s]/g, ' ').split(/\s+/).filter(w => w.length >= 2).slice(0, 3).join(' ');
|
|
1350
|
+
if (o.split(' ').length === 3)
|
|
1351
|
+
openers.set(o, (openers.get(o) ?? 0) + 1);
|
|
1352
|
+
}
|
|
1353
|
+
const topOpener = [...openers.entries()].sort((a, b) => b[1] - a[1])[0];
|
|
1354
|
+
const boilerSignal = !!topOpener && topOpener[1] >= 3;
|
|
1355
|
+
const signals = (lexSignal ? 1 : 0) + (uniformSignal ? 1 : 0) + (boilerSignal ? 1 : 0);
|
|
1356
|
+
if (signals === 0)
|
|
1357
|
+
continue;
|
|
1358
|
+
const score = density + (uniformSignal ? 3 : 0) + (boilerSignal ? 2 : 0);
|
|
1359
|
+
metrics.push({ urlKey: x.urlKey, score, signals, hits, density, cv, opener: boilerSignal ? { text: topOpener[0], n: topOpener[1] } : null, sentences: sentences.length });
|
|
1360
|
+
}
|
|
1361
|
+
if (!metrics.length)
|
|
1362
|
+
return [];
|
|
1363
|
+
// Outliers only: the LEXICON signal is mandatory (uniform sentences + repeated openers
|
|
1364
|
+
// without a single slop phrase is template chrome, not slop — live-verified on simracing),
|
|
1365
|
+
// plus ≥1 corroborating signal, AND composite score in the top decile of pages that showed
|
|
1366
|
+
// any signal at all.
|
|
1367
|
+
const sorted = metrics.map(m => m.score).sort((a, b) => a - b);
|
|
1368
|
+
const p90 = sorted[Math.min(sorted.length - 1, Math.floor(sorted.length * 0.9))];
|
|
1369
|
+
return metrics
|
|
1370
|
+
.filter(m => m.signals >= 2 && m.hits.size >= 3 && m.density >= 2.5 && m.score >= p90)
|
|
1371
|
+
.sort((a, b) => b.score - a.score).slice(0, 30)
|
|
1372
|
+
.map(m => ({ urlKey: m.urlKey, evidence: {
|
|
1373
|
+
slopPhrases: [...m.hits.entries()].sort((a, b) => b[1] - a[1]).slice(0, 6).map(([p, n]) => `${p} ×${n}`),
|
|
1374
|
+
lexDensityPer1000: Math.round(m.density * 10) / 10,
|
|
1375
|
+
sentenceLengthCV: m.cv !== null ? Math.round(m.cv * 100) / 100 : null,
|
|
1376
|
+
repeatedOpener: m.opener ? `"${m.opener.text}…" opens ${m.opener.n} sections` : null,
|
|
1377
|
+
sentencesMeasured: m.sentences,
|
|
1378
|
+
note: 'multiple statistical slop signals — verify by reading before rewriting',
|
|
1379
|
+
} }));
|
|
1380
|
+
},
|
|
1381
|
+
},
|
|
1382
|
+
{
|
|
1383
|
+
// The guide's "unique point of view based on personal experience or expertise", cut down to its
|
|
1384
|
+
// deterministically checkable slice: an Article/BlogPosting that carries schema yet names no
|
|
1385
|
+
// author. Attribution is the machine-readable experience/expertise signal; a byline-less article
|
|
1386
|
+
// is the commodity-content default. Only fires where Article schema EXISTS (no schema at all is
|
|
1387
|
+
// schema-opportunity territory, not this check).
|
|
1388
|
+
id: 'article-no-author', category: 'schema', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
1389
|
+
title: 'Article schema with no author attribution', fix: 'This page marks itself up as an Article/BlogPosting but declares no author. Google’s AI-search guidance rewards content with a demonstrable first-hand point of view — add `author` (a Person with a real name, ideally linking to an author page) to the Article schema and a visible byline to match.',
|
|
1390
|
+
run: (c) => {
|
|
1391
|
+
const out = [];
|
|
1392
|
+
for (const r of rows(c, `SELECT url_key urlKey, json_ld jsonLd FROM pages WHERE status_code=200 AND indexable=1 AND json_ld LIKE '%Article%'`)) {
|
|
1393
|
+
for (const n of parseJsonLdNodes(r.jsonLd)) {
|
|
1394
|
+
const t = nodeType(n);
|
|
1395
|
+
if (!t || !/^(article|blogposting|newsarticle|techarticle|scholarlyarticle)$/i.test(t))
|
|
1396
|
+
continue;
|
|
1397
|
+
const a = n.author;
|
|
1398
|
+
const named = (v) => !!v && (typeof v === 'string' ? v.trim().length > 0
|
|
1399
|
+
: Array.isArray(v) ? v.some(named) : typeof v === 'object' && typeof v.name === 'string' && v.name.trim().length > 0);
|
|
1400
|
+
if (!named(a)) {
|
|
1401
|
+
out.push({ urlKey: r.urlKey, evidence: { schemaType: t, author: a ?? null, note: 'Article markup present, author absent or unnamed' } });
|
|
1402
|
+
break;
|
|
1403
|
+
}
|
|
1404
|
+
}
|
|
1405
|
+
}
|
|
1406
|
+
return out;
|
|
1407
|
+
},
|
|
1408
|
+
},
|
|
1409
|
+
];
|
|
1410
|
+
const sitemapHasRows = (c) => (c.db.prepare('SELECT COUNT(*) n FROM sitemap_urls').get().n) > 0;
|
|
1411
|
+
let _hreflangCache = null;
|
|
1412
|
+
function hreflangFindings(ctx) {
|
|
1413
|
+
if (_hreflangCache && _hreflangCache.db === ctx.db)
|
|
1414
|
+
return _hreflangCache.out;
|
|
1415
|
+
const hostForm = ctx.db.prepare(`SELECT host_form h FROM property_meta LIMIT 1`).get()?.h;
|
|
1416
|
+
const opts = { hostForm: hostForm ?? 'asis' };
|
|
1417
|
+
const pageRows = ctx.db.prepare(`SELECT url_key, url, status_code, noindex, hreflang FROM pages WHERE hreflang IS NOT NULL AND hreflang != ''`).all();
|
|
1418
|
+
const status = new Map();
|
|
1419
|
+
for (const p of ctx.db.prepare(`SELECT url_key, status_code, noindex FROM pages`).all())
|
|
1420
|
+
status.set(p.url_key, { status: p.status_code, noindex: p.noindex });
|
|
1421
|
+
// declared[sourceKey] = set of internal alternate targetKeys (excluding self)
|
|
1422
|
+
const declared = new Map();
|
|
1423
|
+
const parsed = [];
|
|
1424
|
+
for (const p of pageRows) {
|
|
1425
|
+
let arr;
|
|
1426
|
+
try {
|
|
1427
|
+
arr = JSON.parse(p.hreflang);
|
|
1428
|
+
}
|
|
1429
|
+
catch {
|
|
1430
|
+
continue;
|
|
1431
|
+
}
|
|
1432
|
+
const targets = [];
|
|
1433
|
+
for (const { lang, href } of arr) {
|
|
1434
|
+
if (!href)
|
|
1435
|
+
continue;
|
|
1436
|
+
let key;
|
|
1437
|
+
try {
|
|
1438
|
+
key = urlKey(new URL(href, p.url).toString(), opts);
|
|
1439
|
+
}
|
|
1440
|
+
catch {
|
|
1441
|
+
continue;
|
|
1442
|
+
}
|
|
1443
|
+
if (key === p.url_key)
|
|
1444
|
+
continue; // self-reference
|
|
1445
|
+
targets.push({ key, lang: (lang || '').toLowerCase() });
|
|
1446
|
+
}
|
|
1447
|
+
parsed.push({ srcKey: p.url_key, targets });
|
|
1448
|
+
// A return link via x-default is a VALID return tag (common when x-default is the
|
|
1449
|
+
// homepage) — include all targets here; x-default is only excluded as a reciprocation
|
|
1450
|
+
// *requirement* in the loop below, never as a way of satisfying one.
|
|
1451
|
+
declared.set(p.url_key, new Set(targets.map(t => t.key)));
|
|
1452
|
+
}
|
|
1453
|
+
const broken = [];
|
|
1454
|
+
const noReturn = [];
|
|
1455
|
+
for (const { srcKey, targets } of parsed) {
|
|
1456
|
+
const brokenTargets = [];
|
|
1457
|
+
const missingReturn = [];
|
|
1458
|
+
for (const t of targets) {
|
|
1459
|
+
const st = status.get(t.key);
|
|
1460
|
+
if (st && (st.status !== 200 || st.noindex === 1))
|
|
1461
|
+
brokenTargets.push(t.key);
|
|
1462
|
+
// reciprocation only for internal, live (200) targets we crawled, excluding x-default
|
|
1463
|
+
// (a broken target is already reported by broken-hreflang-target — don't double-flag)
|
|
1464
|
+
if (t.lang !== 'x-default' && st && st.status === 200 && st.noindex !== 1 && !(declared.get(t.key)?.has(srcKey)))
|
|
1465
|
+
missingReturn.push(t.key);
|
|
1466
|
+
}
|
|
1467
|
+
if (brokenTargets.length)
|
|
1468
|
+
broken.push({ urlKey: srcKey, evidence: { brokenAlternates: brokenTargets.slice(0, 10), count: brokenTargets.length } });
|
|
1469
|
+
if (missingReturn.length)
|
|
1470
|
+
noReturn.push({ urlKey: srcKey, evidence: { missingReturnFrom: missingReturn.slice(0, 10), count: missingReturn.length } });
|
|
1471
|
+
}
|
|
1472
|
+
_hreflangCache = { db: ctx.db, out: { broken, noReturn } };
|
|
1473
|
+
return { broken, noReturn };
|
|
1474
|
+
}
|
|
1475
|
+
//# sourceMappingURL=checks.js.map
|