@houtini/seo-audit-console 0.8.0 → 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +308 -306
- package/dist/audit/checks.d.ts +0 -7
- package/dist/audit/checks.d.ts.map +1 -1
- package/dist/audit/checks.js +130 -242
- package/dist/audit/checks.js.map +1 -1
- package/dist/audit/drift.d.ts +0 -12
- package/dist/audit/drift.d.ts.map +1 -1
- package/dist/audit/drift.js +0 -15
- package/dist/audit/drift.js.map +1 -1
- package/dist/audit/engine.d.ts +0 -3
- package/dist/audit/engine.d.ts.map +1 -1
- package/dist/audit/engine.js +7 -42
- package/dist/audit/engine.js.map +1 -1
- package/dist/audit/keywordList.d.ts +0 -14
- package/dist/audit/keywordList.d.ts.map +1 -1
- package/dist/audit/keywordList.js +0 -5
- package/dist/audit/keywordList.js.map +1 -1
- package/dist/audit/opportunities.js +3 -26
- package/dist/audit/opportunities.js.map +1 -1
- package/dist/audit/recon.d.ts +0 -31
- package/dist/audit/recon.d.ts.map +1 -1
- package/dist/audit/recon.js +0 -27
- package/dist/audit/recon.js.map +1 -1
- package/dist/audit/report.d.ts +0 -4
- package/dist/audit/report.d.ts.map +1 -1
- package/dist/audit/report.js +0 -7
- package/dist/audit/report.js.map +1 -1
- package/dist/audit/schema-validate.d.ts +0 -26
- package/dist/audit/schema-validate.d.ts.map +1 -1
- package/dist/audit/schema-validate.js +8 -64
- package/dist/audit/schema-validate.js.map +1 -1
- package/dist/audit/templates.d.ts +0 -24
- package/dist/audit/templates.d.ts.map +1 -1
- package/dist/audit/templates.js +1 -21
- package/dist/audit/templates.js.map +1 -1
- package/dist/audit/topicGaps.d.ts +0 -1
- package/dist/audit/topicGaps.d.ts.map +1 -1
- package/dist/audit/topicGaps.js +3 -24
- package/dist/audit/topicGaps.js.map +1 -1
- package/dist/core/AuditDatabase.d.ts +0 -14
- package/dist/core/AuditDatabase.d.ts.map +1 -1
- package/dist/core/AuditDatabase.js +8 -61
- package/dist/core/AuditDatabase.js.map +1 -1
- package/dist/core/Backlinks.d.ts +0 -7
- package/dist/core/Backlinks.d.ts.map +1 -1
- package/dist/core/Backlinks.js +1 -23
- package/dist/core/Backlinks.js.map +1 -1
- package/dist/core/Crawler.d.ts +0 -1
- package/dist/core/Crawler.d.ts.map +1 -1
- package/dist/core/Crawler.js +24 -119
- package/dist/core/Crawler.js.map +1 -1
- package/dist/core/DataForSeoClient.d.ts +0 -74
- package/dist/core/DataForSeoClient.d.ts.map +1 -1
- package/dist/core/DataForSeoClient.js +1 -72
- package/dist/core/DataForSeoClient.js.map +1 -1
- package/dist/core/Entities.d.ts +0 -6
- package/dist/core/Entities.d.ts.map +1 -1
- package/dist/core/Entities.js +0 -8
- package/dist/core/Entities.js.map +1 -1
- package/dist/core/FirecrawlClient.d.ts +0 -18
- package/dist/core/FirecrawlClient.d.ts.map +1 -1
- package/dist/core/FirecrawlClient.js +1 -13
- package/dist/core/FirecrawlClient.js.map +1 -1
- package/dist/core/GscClient.d.ts +0 -7
- package/dist/core/GscClient.d.ts.map +1 -1
- package/dist/core/GscClient.js +1 -10
- package/dist/core/GscClient.js.map +1 -1
- package/dist/core/GscSync.d.ts +0 -4
- package/dist/core/GscSync.d.ts.map +1 -1
- package/dist/core/GscSync.js +2 -33
- package/dist/core/GscSync.js.map +1 -1
- package/dist/core/JobManager.d.ts +0 -5
- package/dist/core/JobManager.d.ts.map +1 -1
- package/dist/core/JobManager.js +0 -8
- package/dist/core/JobManager.js.map +1 -1
- package/dist/core/LinkIntersect.d.ts +0 -1
- package/dist/core/LinkIntersect.d.ts.map +1 -1
- package/dist/core/LinkIntersect.js +1 -37
- package/dist/core/LinkIntersect.js.map +1 -1
- package/dist/core/MajesticClient.d.ts +0 -26
- package/dist/core/MajesticClient.d.ts.map +1 -1
- package/dist/core/MajesticClient.js +0 -15
- package/dist/core/MajesticClient.js.map +1 -1
- package/dist/core/RankTracker.d.ts +0 -4
- package/dist/core/RankTracker.d.ts.map +1 -1
- package/dist/core/RankTracker.js +0 -9
- package/dist/core/RankTracker.js.map +1 -1
- package/dist/core/Refresh.d.ts +0 -6
- package/dist/core/Refresh.d.ts.map +1 -1
- package/dist/core/Refresh.js +0 -6
- package/dist/core/Refresh.js.map +1 -1
- package/dist/core/SupadataClient.d.ts +0 -11
- package/dist/core/SupadataClient.d.ts.map +1 -1
- package/dist/core/SupadataClient.js +0 -5
- package/dist/core/SupadataClient.js.map +1 -1
- package/dist/core/UrlInspector.d.ts +0 -5
- package/dist/core/UrlInspector.d.ts.map +1 -1
- package/dist/core/UrlInspector.js +0 -6
- package/dist/core/UrlInspector.js.map +1 -1
- package/dist/core/WikidataClient.d.ts +0 -2
- package/dist/core/WikidataClient.d.ts.map +1 -1
- package/dist/core/WikidataClient.js +0 -7
- package/dist/core/WikidataClient.js.map +1 -1
- package/dist/core/agentReadiness.d.ts +0 -9
- package/dist/core/agentReadiness.d.ts.map +1 -1
- package/dist/core/agentReadiness.js +0 -13
- package/dist/core/agentReadiness.js.map +1 -1
- package/dist/core/ctrModel.js +0 -3
- package/dist/core/ctrModel.js.map +1 -1
- package/dist/core/dashboardData.d.ts +0 -1
- package/dist/core/dashboardData.d.ts.map +1 -1
- package/dist/core/dashboardData.js +13 -95
- package/dist/core/dashboardData.js.map +1 -1
- package/dist/core/dataStorage.d.ts +0 -2
- package/dist/core/dataStorage.d.ts.map +1 -1
- package/dist/core/dataStorage.js +6 -19
- package/dist/core/dataStorage.js.map +1 -1
- package/dist/core/draftBrief.d.ts +0 -7
- package/dist/core/draftBrief.d.ts.map +1 -1
- package/dist/core/draftBrief.js +0 -10
- package/dist/core/draftBrief.js.map +1 -1
- package/dist/core/extract.d.ts +0 -11
- package/dist/core/extract.d.ts.map +1 -1
- package/dist/core/extract.js +1 -38
- package/dist/core/extract.js.map +1 -1
- package/dist/core/googleNews.d.ts +0 -11
- package/dist/core/googleNews.d.ts.map +1 -1
- package/dist/core/googleNews.js +0 -7
- package/dist/core/googleNews.js.map +1 -1
- package/dist/core/gscFreshness.d.ts +0 -10
- package/dist/core/gscFreshness.d.ts.map +1 -1
- package/dist/core/gscFreshness.js +0 -11
- package/dist/core/gscFreshness.js.map +1 -1
- package/dist/core/linkGraph.d.ts +0 -15
- package/dist/core/linkGraph.d.ts.map +1 -1
- package/dist/core/linkGraph.js +1 -23
- package/dist/core/linkGraph.js.map +1 -1
- package/dist/core/marketSizing.d.ts +0 -13
- package/dist/core/marketSizing.d.ts.map +1 -1
- package/dist/core/marketSizing.js +1 -1
- package/dist/core/marketSizing.js.map +1 -1
- package/dist/core/passageScore.d.ts +0 -13
- package/dist/core/passageScore.d.ts.map +1 -1
- package/dist/core/passageScore.js +1 -16
- package/dist/core/passageScore.js.map +1 -1
- package/dist/core/paths.d.ts +0 -3
- package/dist/core/paths.d.ts.map +1 -1
- package/dist/core/paths.js +0 -0
- package/dist/core/paths.js.map +1 -1
- package/dist/core/queryData.d.ts +0 -1
- package/dist/core/queryData.d.ts.map +1 -1
- package/dist/core/queryData.js +0 -18
- package/dist/core/queryData.js.map +1 -1
- package/dist/core/reconFetch.d.ts +0 -6
- package/dist/core/reconFetch.d.ts.map +1 -1
- package/dist/core/reconFetch.js +1 -4
- package/dist/core/reconFetch.js.map +1 -1
- package/dist/core/reconResearch.d.ts +0 -20
- package/dist/core/reconResearch.d.ts.map +1 -1
- package/dist/core/reconResearch.js +1 -16
- package/dist/core/reconResearch.js.map +1 -1
- package/dist/core/reranker.d.ts +0 -2
- package/dist/core/reranker.d.ts.map +1 -1
- package/dist/core/reranker.js +1 -11
- package/dist/core/reranker.js.map +1 -1
- package/dist/core/robots.d.ts +0 -6
- package/dist/core/robots.d.ts.map +1 -1
- package/dist/core/robots.js +1 -7
- package/dist/core/robots.js.map +1 -1
- package/dist/core/serpFootprint.d.ts +0 -12
- package/dist/core/serpFootprint.d.ts.map +1 -1
- package/dist/core/serpFootprint.js +0 -1
- package/dist/core/serpFootprint.js.map +1 -1
- package/dist/core/serpRecon.d.ts +0 -14
- package/dist/core/serpRecon.d.ts.map +1 -1
- package/dist/core/serpRecon.js +1 -13
- package/dist/core/serpRecon.js.map +1 -1
- package/dist/core/sitemap.js +3 -16
- package/dist/core/sitemap.js.map +1 -1
- package/dist/core/sql.d.ts +0 -4
- package/dist/core/sql.d.ts.map +1 -1
- package/dist/core/sql.js +0 -4
- package/dist/core/sql.js.map +1 -1
- package/dist/core/url-key.d.ts +0 -31
- package/dist/core/url-key.d.ts.map +1 -1
- package/dist/core/url-key.js +1 -38
- package/dist/core/url-key.js.map +1 -1
- package/dist/core/webHandlers.d.ts +0 -6
- package/dist/core/webHandlers.d.ts.map +1 -1
- package/dist/core/webHandlers.js +2 -5
- package/dist/core/webHandlers.js.map +1 -1
- package/dist/core/webServer.d.ts +0 -16
- package/dist/core/webServer.d.ts.map +1 -1
- package/dist/core/webServer.js +3 -17
- package/dist/core/webServer.js.map +1 -1
- package/dist/dashboard.js +0 -23
- package/dist/dashboard.js.map +1 -1
- package/dist/generators/index.d.ts +0 -7
- package/dist/generators/index.d.ts.map +1 -1
- package/dist/generators/index.js +1 -26
- package/dist/generators/index.js.map +1 -1
- package/dist/index.js +0 -2
- package/dist/index.js.map +1 -1
- package/dist/server.js +16 -166
- package/dist/server.js.map +1 -1
- package/package.json +2 -2
- package/server.json +2 -2
package/dist/audit/checks.js
CHANGED
|
@@ -5,33 +5,17 @@ import { expectedCtr } from '../core/ctrModel.js';
|
|
|
5
5
|
import { latestTwoCrawls } from './drift.js';
|
|
6
6
|
import { HTML_CT } from '../core/sql.js';
|
|
7
7
|
const rows = (ctx, sql, ...args) => ctx.db.prepare(sql).all(...args);
|
|
8
|
-
// Streaming variant — yields one row at a time instead of materialising the whole result set. Use for
|
|
9
|
-
// checks that scan a large/fat column (e.g. body_chunks) and only need independent per-row work, so a
|
|
10
|
-
// big content site doesn't load every page's body text into one array.
|
|
11
8
|
const iterRows = (ctx, sql, ...args) => ctx.db.prepare(sql).iterate(...args);
|
|
12
|
-
// d is the finalised GSC date (partial trailing days already trimmed upstream); cap the window at
|
|
13
|
-
// it so checks never count unfinalised days. winPrev already caps below d, so it's unaffected.
|
|
14
9
|
const win = (d) => `date > date('${d}', '-28 days') AND date <= '${d}'`;
|
|
15
|
-
// Prior 28-day window (the 28 days BEFORE the current window) — for period-over-period checks.
|
|
16
10
|
const winPrev = (d) => `date <= date('${d}', '-28 days') AND date > date('${d}', '-56 days')`;
|
|
17
|
-
// SQL clause to drop branded queries (whole-token match). Branded multi-URL ranking is sitelinks,
|
|
18
|
-
// not cannibalisation; branded top-queries aren't anchor targets. The brand is re-sanitised to
|
|
19
|
-
// alnum HERE (not trusting the caller) so inlining it into SQL is unconditionally injection-safe.
|
|
20
11
|
const brandExcl = (c) => {
|
|
21
12
|
const b = (c.brand ?? '').replace(/[^a-z0-9]/g, '');
|
|
22
13
|
return b ? `AND (' ' || LOWER(query) || ' ') NOT LIKE '% ${b} %'` : '';
|
|
23
14
|
};
|
|
24
|
-
// Days of GSC history actually held — period-over-period checks need enough span to be meaningful.
|
|
25
15
|
const spanDays = (c) => {
|
|
26
|
-
// Span must be measured up to the FINALISED max date the windows actually key off
|
|
27
|
-
// (gscMaxDate is trimmed ~3 days below raw MAX(date)) — measuring the raw span lets the
|
|
28
|
-
// ≥56-day guard pass while the previous-28d window still reaches before MIN(date),
|
|
29
|
-
// undercounting the prior period (rising-pages FPs, traffic-decay FNs).
|
|
30
16
|
const r = c.db.prepare(`SELECT julianday(?) - julianday(MIN(date)) d FROM search_analytics`).get(c.gscMaxDate ?? null);
|
|
31
17
|
return r?.d ?? 0;
|
|
32
18
|
};
|
|
33
|
-
// Newest dateModified/datePublished anywhere in a page's JSON-LD (recurses @graph/arrays) — the
|
|
34
|
-
// effective "last meaningfully updated" date. json_ld is a JSON array of raw block strings.
|
|
35
19
|
const newestSchemaDate = (jl) => {
|
|
36
20
|
if (!jl)
|
|
37
21
|
return null;
|
|
@@ -58,27 +42,15 @@ const newestSchemaDate = (jl) => {
|
|
|
58
42
|
try {
|
|
59
43
|
scan(JSON.parse(block));
|
|
60
44
|
}
|
|
61
|
-
catch {
|
|
45
|
+
catch { }
|
|
62
46
|
}
|
|
63
47
|
}
|
|
64
|
-
catch {
|
|
48
|
+
catch { }
|
|
65
49
|
return bestStr;
|
|
66
50
|
};
|
|
67
|
-
|
|
68
|
-
// page 1 and are intentionally absent from sitemaps, so they must NOT generate duplicate-title /
|
|
69
|
-
// duplicate-meta / missing-meta / not-in-sitemap false positives. Pass the column reference
|
|
70
|
-
// (e.g. 'url_key' or 'p.url_key') so it composes with table aliases.
|
|
71
|
-
const notPagination = (col = 'url_key') =>
|
|
72
|
-
// Anchor the param name to ?/& — a bare LIKE '%page=%' also matches per_page=/on_page=/
|
|
73
|
-
// homepage=, wrongly exempting those URLs from duplicate-title/meta/sitemap checks.
|
|
74
|
-
`${col} NOT GLOB '*/page/[0-9]*' AND ${col} NOT GLOB '*[?&]page=[0-9]*' AND ${col} NOT GLOB '*[?&]paged=[0-9]*'`;
|
|
75
|
-
// Significant query terms (drop stopwords; keep ≥2 chars so "vr"/"pc"/"ai" count).
|
|
51
|
+
const notPagination = (col = 'url_key') => `${col} NOT GLOB '*/page/[0-9]*' AND ${col} NOT GLOB '*[?&]page=[0-9]*' AND ${col} NOT GLOB '*[?&]paged=[0-9]*'`;
|
|
76
52
|
const STOP = new Set(['the', 'a', 'an', 'and', 'or', 'of', 'for', 'to', 'in', 'on', 'with', 'your', 'you', 'is', 'are', 'best', 'how', 'what', 'vs', 'why', 'can']);
|
|
77
53
|
const terms = (s) => (s || '').toLowerCase().replace(/[^a-z0-9 ]+/g, ' ').split(/\s+/).filter(t => t.length >= 2 && !STOP.has(t));
|
|
78
|
-
// A query term is "present" in the title if a TITLE WORD matches it (exact, or a
|
|
79
|
-
// plural/stem prefix either way for ≥4-char tokens), or the space-collapsed title
|
|
80
|
-
// contains it (for multi-word brands like "sync mesh" ≈ "syncmesh", ≥5 chars).
|
|
81
|
-
// Word-level avoids substring false-matches (e.g. "art" inside "smart").
|
|
82
54
|
const titleHasTerm = (title, term) => {
|
|
83
55
|
const t = title.toLowerCase();
|
|
84
56
|
const words = t.split(/[^a-z0-9]+/).filter(Boolean);
|
|
@@ -86,7 +58,6 @@ const titleHasTerm = (title, term) => {
|
|
|
86
58
|
return true;
|
|
87
59
|
return term.length >= 5 && t.replace(/[^a-z0-9]+/g, '').includes(term);
|
|
88
60
|
};
|
|
89
|
-
// Validate captured JSON-LD per page, keeping findings whose issue-kinds match `kinds`.
|
|
90
61
|
function schemaFindings(ctx, kinds) {
|
|
91
62
|
const set = new Set(kinds);
|
|
92
63
|
const out = [];
|
|
@@ -97,11 +68,8 @@ function schemaFindings(ctx, kinds) {
|
|
|
97
68
|
}
|
|
98
69
|
return out;
|
|
99
70
|
}
|
|
100
|
-
// Position→expected-CTR curve lives in core/ctrModel (shared with the dashboard). Re-export it so
|
|
101
|
-
// existing `import { expectedCtr } from './checks.js'` call sites keep working.
|
|
102
71
|
export { expectedCtr };
|
|
103
72
|
export const CHECKS = [
|
|
104
|
-
// ── On-page (crawl, deterministic) ──────────────────────────────────────
|
|
105
73
|
{
|
|
106
74
|
id: 'missing-title', category: 'onpage', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
107
75
|
title: 'Missing title tag', fix: 'Add a unique, descriptive <title> (~50–60 chars).',
|
|
@@ -132,19 +100,11 @@ export const CHECKS = [
|
|
|
132
100
|
title: 'Thin content', fix: 'Expand or consolidate — under ~200 words of body text.',
|
|
133
101
|
run: (c) => rows(c, `SELECT url_key urlKey, word_count FROM pages WHERE status_code=200 AND indexable=1 AND word_count < 200`).map(r => ({ urlKey: r.urlKey, evidence: { wordCount: r.word_count } })),
|
|
134
102
|
},
|
|
135
|
-
// ── Indexation / crawlability ───────────────────────────────────────────
|
|
136
103
|
{
|
|
137
|
-
// RETIRED: 'canonical-mismatch' (was HIGH). A 200 page whose canonical points to a *healthy*
|
|
138
|
-
// 200 indexable URL is intentional consolidation (slug variants, category merges) — normal SEO,
|
|
139
|
-
// not an issue, yet it fired HIGH on every such page (pure noise). Every actionable case is
|
|
140
|
-
// already covered by a higher-signal check: broken-canonical-target (unhealthy target),
|
|
141
|
-
// canonical-ignored (Google ranks the non-canonical page), canonical-conflict (GSC disagrees).
|
|
142
|
-
// So plain "canonical points elsewhere" has no high-confidence residual — removed.
|
|
143
104
|
id: 'broken-internal-links', category: 'crawlability', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
|
|
144
105
|
title: 'Internal links to 4xx/5xx', fix: 'Repoint internal links to a live, canonical URL.',
|
|
145
106
|
run: (c) => rows(c, `SELECT l.target_key urlKey, p.status_code status, COUNT(DISTINCT l.source_key) sources FROM links l JOIN pages p ON p.url_key=l.target_key WHERE l.is_internal=1 AND p.status_code >= 400 AND p.status_code NOT IN (429,503) GROUP BY l.target_key`).map(r => ({ urlKey: r.urlKey, evidence: { status: r.status, linkingPages: r.sources } })),
|
|
146
107
|
},
|
|
147
|
-
// ── Extractor-dependent (images + canonical shape) ──────────────────────
|
|
148
108
|
{
|
|
149
109
|
id: 'image-alt', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
150
110
|
title: 'Images missing alt text', fix: 'Add descriptive alt text to content images (alt="" only for decorative).',
|
|
@@ -160,13 +120,8 @@ export const CHECKS = [
|
|
|
160
120
|
title: 'Multiple canonical tags', fix: 'Keep exactly one rel=canonical — conflicting canonicals let Google pick (or ignore) one.',
|
|
161
121
|
run: (c) => rows(c, `SELECT url_key urlKey, canonical_count cnt FROM pages WHERE status_code=200 AND canonical_count > 1`).map(r => ({ urlKey: r.urlKey, evidence: { canonicalCount: r.cnt } })),
|
|
162
122
|
},
|
|
163
|
-
// ── Extractor additions (CLS, headings, mixed content, directives, social) ───
|
|
164
123
|
{
|
|
165
124
|
id: 'images-missing-dimensions', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
166
|
-
// Lint, NOT a measured Core Web Vital. Missing width/height attributes only cause CLS if the
|
|
167
|
-
// CSS doesn't already reserve space — modern themes using aspect-ratio / fixed boxes have ~0
|
|
168
|
-
// measured CLS despite missing attributes. So we report it as a hygiene lint and explicitly say
|
|
169
|
-
// it's not a confirmed CWV issue; escalate only against field CLS (high-yield-cwv-fail does that).
|
|
170
125
|
title: 'Images missing width/height attributes', fix: 'Add width & height (or rely on CSS aspect-ratio) so the browser reserves space. NOTE: this is a lint — if your CSS already reserves space (aspect-ratio / fixed box) measured CLS is likely ~0 and there is nothing to fix. Confirm with field CLS before prioritising.',
|
|
171
126
|
run: (c) => rows(c, `SELECT url_key urlKey, images_missing_dimensions n, image_count total FROM pages WHERE status_code=200 AND indexable=1 AND images_missing_dimensions > 0`).map(r => ({ urlKey: r.urlKey, evidence: { missingDimensions: r.n, total: r.total, note: 'lint only — no measured CLS impact unless field data shows layout shift' } })),
|
|
172
127
|
},
|
|
@@ -190,13 +145,11 @@ export const CHECKS = [
|
|
|
190
145
|
title: 'No social share tags', fix: 'Add Open Graph (og:title/og:image) and/or Twitter Card tags so shared links render a rich preview.',
|
|
191
146
|
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND (og_tags IS NULL OR og_tags='') AND (twitter_tags IS NULL OR twitter_tags='')`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
|
|
192
147
|
},
|
|
193
|
-
// ── Security / war-stories (headers now captured) ───────────────────────
|
|
194
148
|
{
|
|
195
149
|
id: 'missing-hsts', category: 'security', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
196
150
|
title: 'Missing HSTS header', fix: 'Add Strict-Transport-Security with a sensible max-age.',
|
|
197
151
|
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (security_headers IS NULL OR security_headers NOT LIKE '%hsts%') LIMIT 1`).map(r => ({ urlKey: r.urlKey, evidence: { note: 'representative page; HSTS is site-wide' } })),
|
|
198
152
|
},
|
|
199
|
-
// ── Merged GSC × crawl (the differentiator) ─────────────────────────────
|
|
200
153
|
{
|
|
201
154
|
id: 'noindex-with-traffic', category: 'indexation', severity: 'crit', labels: ['D', 'G'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
202
155
|
title: 'Noindex page still getting clicks', fix: 'Remove noindex if the page should rank — it earns clicks.',
|
|
@@ -245,7 +198,6 @@ export const CHECKS = [
|
|
|
245
198
|
title: 'Striking-distance query (page 2)', fix: 'Small on-page + internal-link push could reach page 1.',
|
|
246
199
|
run: (c) => c.gscMaxDate ? rows(c, `SELECT page_key urlKey, query, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) position, SUM(impressions) impressions FROM search_analytics WHERE query IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY query, page_key HAVING SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0)>10 AND SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0)<=20 AND SUM(impressions)>=20 ORDER BY impressions DESC LIMIT 50`).map(r => ({ urlKey: r.urlKey, evidence: { query: r.query, position: Math.round(r.position * 10) / 10, impressions: r.impressions } })) : [],
|
|
247
200
|
},
|
|
248
|
-
// ── Additions from industry checklist review (buildable on current data) ──
|
|
249
201
|
{
|
|
250
202
|
id: 'title-too-long', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
251
203
|
title: 'Title over ~60 chars', fix: 'Trim the title so the primary keyword sits within ~60 chars.',
|
|
@@ -264,8 +216,6 @@ export const CHECKS = [
|
|
|
264
216
|
{
|
|
265
217
|
id: 'redirect-chain', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
|
|
266
218
|
title: 'Redirect chain (2+ hops)', fix: 'Collapse to a single hop to the final URL.',
|
|
267
|
-
// Count hops in SQL (json_array_length, guarded by json_valid) and filter to >=2 there — so we
|
|
268
|
-
// never pull every redirect-bearing page into JS just to count + drop most of them.
|
|
269
219
|
run: (c) => rows(c, `SELECT url_key urlKey, json_array_length(redirects) hops FROM pages
|
|
270
220
|
WHERE redirects IS NOT NULL AND json_valid(redirects) AND json_array_length(redirects) >= 2`)
|
|
271
221
|
.map(r => ({ urlKey: r.urlKey, evidence: { hops: r.hops } })),
|
|
@@ -273,13 +223,6 @@ export const CHECKS = [
|
|
|
273
223
|
{
|
|
274
224
|
id: 'internal-links-to-redirects', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
|
|
275
225
|
title: 'Internal links pointing through redirects', fix: 'Repoint internal links to the final URL (saves crawl + equity).',
|
|
276
|
-
// Flag a link ONLY if the raw href it uses is itself a redirect source — i.e. that exact URL
|
|
277
|
-
// appears as a `from` hop in some redirect chain. The old query flagged any link whose target
|
|
278
|
-
// page merely HAD a `redirects` entry, which fired site-wide on www-canonical properties: every
|
|
279
|
-
// page records the seed apex→www hop, yet the actual hrefs already use the final (www) URL and
|
|
280
|
-
// never redirect. Matching the href (fragment-stripped) against the real redirect-source set is
|
|
281
|
-
// robust to that artefact. Kept entirely in SQL (json_each over pages.redirects) so we never
|
|
282
|
-
// materialise the whole links table in JS — the links table is the largest in the DB.
|
|
283
226
|
run: (c) => {
|
|
284
227
|
return rows(c, `
|
|
285
228
|
WITH redir_src AS (
|
|
@@ -302,11 +245,8 @@ export const CHECKS = [
|
|
|
302
245
|
{
|
|
303
246
|
id: 'missing-structured-data', category: 'schema', severity: 'low', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
304
247
|
title: 'No structured data', fix: 'Add relevant JSON-LD (Article, Product, Organization…).',
|
|
305
|
-
// Only "no structured data" if there's no JSON-LD AND no Microdata/RDFa either — else a
|
|
306
|
-
// page using valid Microdata (common on older themes) is falsely flagged.
|
|
307
248
|
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND (json_ld IS NULL OR json_ld='') AND COALESCE(has_microdata,0)=0 AND COALESCE(has_rdfa,0)=0`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
|
|
308
249
|
},
|
|
309
|
-
// ── Schema validation (validate captured json_ld vs maintained Rich-Results map) ──
|
|
310
250
|
{
|
|
311
251
|
id: 'invalid-schema', category: 'schema', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
312
252
|
title: 'Invalid structured data', fix: 'Fix the JSON-LD so each block parses and carries @context (https://schema.org) + a valid @type.',
|
|
@@ -335,15 +275,6 @@ export const CHECKS = [
|
|
|
335
275
|
{
|
|
336
276
|
id: 'keyword-cannibalisation', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
337
277
|
title: 'Keyword cannibalisation', fix: 'Consolidate or differentiate — multiple URLs compete for one query.',
|
|
338
|
-
// A URL only "competes" if it ranks for the query (impression-weighted pos < 20) AND holds a
|
|
339
|
-
// non-trivial share of the leader's impressions — ≥10% of the leader OR ≥500 impressions in its
|
|
340
|
-
// own right, and ≥10 impressions minimum. Without that floor, incidental long-tail appearances
|
|
341
|
-
// (a page picking up 1–7 impressions for the query) counted as competitors: e.g. "best vr headset"
|
|
342
|
-
// reported 10 URLs when ONE page held 59,570 impressions at pos 1.1 and the other nine had 1–7
|
|
343
|
-
// each — not cannibalisation, Google had decided. The absolute ≥500 backstop keeps a genuine
|
|
344
|
-
// mid-volume rival under a dominant leader (e.g. 60k leader + a real 4k second page = 6.7%, below
|
|
345
|
-
// the 10% bar) from being silently dropped. We also exclude "dominance" where the best pages both
|
|
346
|
-
// sit at pos 1–2 (indented/double results are good). Branded queries are dropped via brandExcl.
|
|
347
278
|
run: (c) => {
|
|
348
279
|
if (!c.gscMaxDate)
|
|
349
280
|
return [];
|
|
@@ -368,9 +299,6 @@ export const CHECKS = [
|
|
|
368
299
|
const keys = String(r.pk || '').split('\x1f'), imps = String(r.im || '').split('\x1f'), poss = String(r.ps || '').split('\x1f');
|
|
369
300
|
const items = keys.map((u, i) => ({ url: u, impressions: Number(imps[i]) || 0, position: Number(poss[i]) || 0, title: titleOf.get(u)?.title ?? null }))
|
|
370
301
|
.sort((a, b) => b.impressions - a.impressions);
|
|
371
|
-
// Differentiation signal: do the top-2 competing pages share significant title terms beyond the
|
|
372
|
-
// query itself? If not (and both are titled), they're likely intentionally distinct pages (e.g.
|
|
373
|
-
// pairwise comparisons), not true duplicates competing for one intent — annotate, don't suppress.
|
|
374
302
|
const q = new Set(terms(r.query));
|
|
375
303
|
const sig = items.slice(0, 2).map(it => terms(it.title ?? '').filter((w) => !q.has(w)));
|
|
376
304
|
const differentiated = sig.length === 2 && items[0].title != null && items[1].title != null
|
|
@@ -389,22 +317,16 @@ export const CHECKS = [
|
|
|
389
317
|
{
|
|
390
318
|
id: 'ctr-below-expected', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
391
319
|
title: 'CTR far below position-expected', fix: 'Rewrite title/meta — ranking well but under-clicked (snippet opportunity).',
|
|
392
|
-
// High-confidence floor: ≥500 impressions/28d. A title/meta rewrite (HIGH, ~3h) is only
|
|
393
|
-
// worth flagging where the snippet earns enough visibility for a CTR lift to pay back — a
|
|
394
|
-
// 100-impression page at 1% vs 3% expected is a 2-clicks gap, not a HIGH issue.
|
|
395
320
|
run: (c) => c.gscMaxDate ? rows(c, `SELECT page_key urlKey, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) position, SUM(clicks) clicks, SUM(impressions) impressions FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key HAVING SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) <= 10 AND SUM(impressions) >= 500`)
|
|
396
321
|
.map(r => { const ctr = r.clicks / r.impressions; const exp = expectedCtr(r.position); return { urlKey: r.urlKey, ctr, exp, position: r.position, impressions: r.impressions }; })
|
|
397
322
|
.filter(x => x.ctr < x.exp * 0.5)
|
|
398
323
|
.map(x => {
|
|
399
324
|
const ev = { position: Math.round(x.position * 10) / 10, ctr: Math.round(x.ctr * 1000) / 10 + '%', expectedCtr: Math.round(x.exp * 1000) / 10 + '%', impressions: x.impressions };
|
|
400
|
-
// Extreme case: near-zero CTR at a strong position isn't a title problem — a SERP feature
|
|
401
|
-
// (image/video/AI overview) or navigational intent is taking the clicks. Different fix.
|
|
402
325
|
if (x.position <= 5 && x.ctr < x.exp * 0.15)
|
|
403
326
|
ev.note = 'near-zero CTR for the position — likely a SERP feature or navigational intent taking the clicks; check the live SERP before rewriting the title/meta';
|
|
404
327
|
return { urlKey: x.urlKey, evidence: ev };
|
|
405
328
|
}) : [],
|
|
406
329
|
},
|
|
407
|
-
// ── Period-over-period (GSC history by date) — the trend questions SEOs live in ──
|
|
408
330
|
{
|
|
409
331
|
id: 'traffic-decay', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
410
332
|
title: 'Page losing clicks (period-over-period)', fix: 'Refresh and expand the content, and check for lost rankings — this page’s Search Console clicks fell sharply against the previous 28 days.',
|
|
@@ -416,7 +338,6 @@ export const CHECKS = [
|
|
|
416
338
|
FROM prev LEFT JOIN cur ON cur.page_key=prev.page_key
|
|
417
339
|
WHERE prev.c >= 30 AND COALESCE(cur.c,0) < prev.c * 0.6
|
|
418
340
|
ORDER BY (prev.c - COALESCE(cur.c,0)) DESC LIMIT 40`)
|
|
419
|
-
// page_key IS a url_key — carry it in the joinable column, not just evidence.
|
|
420
341
|
.map(r => ({ urlKey: r.url, evidence: { url: r.url, previousClicks: r.prevC, currentClicks: r.curC, clicksLost: r.prevC - r.curC, dropPercent: Math.round((1 - r.curC / r.prevC) * 100) + '%', clicks: r.prevC - r.curC, impressions: r.curI, position: Math.round(r.pos * 10) / 10 } })),
|
|
421
342
|
},
|
|
422
343
|
{
|
|
@@ -447,7 +368,6 @@ export const CHECKS = [
|
|
|
447
368
|
title: 'Indexable page with no search traffic', fix: 'No impressions in 90 days despite being indexable — consolidate, improve, or noindex/prune to concentrate crawl budget and internal authority (confirm it isn’t seasonal or brand-new first).',
|
|
448
369
|
run: (c) => (!c.gscMaxDate || spanDays(c) < 90) ? [] : rows(c, `SELECT url_key urlKey, ipr FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND COALESCE(click_depth, 999) >= 1 AND url_key NOT IN (SELECT DISTINCT page_key FROM search_analytics WHERE page_key IS NOT NULL AND date <= '${c.gscMaxDate}' AND date > date('${c.gscMaxDate}','-90 days') AND impressions > 0) ORDER BY ipr DESC LIMIT 100`).map(r => ({ urlKey: r.urlKey, evidence: { note: 'indexable but zero impressions in 90 days', ipr: Math.round(r.ipr) } })),
|
|
449
370
|
},
|
|
450
|
-
// ── On-page parity (crawl-only, deterministic) ──
|
|
451
371
|
{
|
|
452
372
|
id: 'duplicate-meta-description', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
453
373
|
title: 'Duplicate meta description', fix: 'Give each indexable page a unique meta description.',
|
|
@@ -506,7 +426,6 @@ export const CHECKS = [
|
|
|
506
426
|
.map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, h1: x.h1 } }));
|
|
507
427
|
},
|
|
508
428
|
},
|
|
509
|
-
// ── Content cluster (AI-era / RAG layer): exploit the chunked body text captured at crawl ──
|
|
510
429
|
{
|
|
511
430
|
id: 'body-missing-top-query', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
512
431
|
title: 'Top query missing from the page body', fix: 'The page ranks for this query yet its terms appear nowhere — not the title, H1, or body copy. Add a section that actually covers the topic; if it can’t, the page is too thin to hold the ranking and a stronger page should target it.',
|
|
@@ -527,8 +446,8 @@ export const CHECKS = [
|
|
|
527
446
|
for (const ch of JSON.parse(x.bc))
|
|
528
447
|
body += ` ${ch.heading || ''} ${ch.text || ''}`;
|
|
529
448
|
}
|
|
530
|
-
catch {
|
|
531
|
-
return !q.some((w) => titleHasTerm(body, w));
|
|
449
|
+
catch { }
|
|
450
|
+
return !q.some((w) => titleHasTerm(body, w));
|
|
532
451
|
}).map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, note: 'query terms absent from title, H1 and body' } }));
|
|
533
452
|
},
|
|
534
453
|
},
|
|
@@ -565,7 +484,7 @@ export const CHECKS = [
|
|
|
565
484
|
return r.filter(x => {
|
|
566
485
|
const q = terms(x.query);
|
|
567
486
|
if (q.length < 2)
|
|
568
|
-
return false;
|
|
487
|
+
return false;
|
|
569
488
|
let chunks;
|
|
570
489
|
try {
|
|
571
490
|
chunks = JSON.parse(x.bc);
|
|
@@ -575,8 +494,8 @@ export const CHECKS = [
|
|
|
575
494
|
}
|
|
576
495
|
const whole = `${x.title || ''} ${chunks.map(ch => `${ch.heading || ''} ${ch.text || ''}`).join(' ')}`;
|
|
577
496
|
if (!q.every((w) => titleHasTerm(whole, w)))
|
|
578
|
-
return false;
|
|
579
|
-
return !chunks.some(ch => { const h = `${ch.heading || ''} ${ch.text || ''}`; return q.every((w) => titleHasTerm(h, w)); });
|
|
497
|
+
return false;
|
|
498
|
+
return !chunks.some(ch => { const h = `${ch.heading || ''} ${ch.text || ''}`; return q.every((w) => titleHasTerm(h, w)); });
|
|
580
499
|
}).map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, note: 'terms present but never together in one passage' } }));
|
|
581
500
|
},
|
|
582
501
|
},
|
|
@@ -605,10 +524,6 @@ export const CHECKS = [
|
|
|
605
524
|
},
|
|
606
525
|
},
|
|
607
526
|
{
|
|
608
|
-
// Hobo "Signal Coherence" / Goldmine; leak: anchor_mismatch. Google leans on internal anchors to
|
|
609
|
-
// understand a page's topic — if the IN-CONTENT inbound anchors never mention the query the page
|
|
610
|
-
// actually ranks for, that's an incoherent internal signal. Guard against boilerplate FPs by using
|
|
611
|
-
// ONLY placement='body' anchors (nav/footer/aside excluded) and requiring ≥3 of them.
|
|
612
527
|
id: 'anchor-text-incoherent', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
613
528
|
title: 'Internal anchors don’t mention the page’s top query', fix: 'The in-content internal links pointing at this page never use its top-ranking query in their anchor text — and Google leans on internal anchors to understand what a page is about. Re-anchor the key internal links with descriptive, query-relevant text instead of generic “read more” / brand-only labels.',
|
|
614
529
|
run: (c) => {
|
|
@@ -619,12 +534,6 @@ export const CHECKS = [
|
|
|
619
534
|
FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} ${brandExcl(c)} GROUP BY page_key, query) s
|
|
620
535
|
JOIN pages p ON p.url_key = s.page_key
|
|
621
536
|
WHERE s.rn = 1 AND p.indexable = 1 AND s.impr >= 100`);
|
|
622
|
-
// Pool only GENUINE editorial anchors. "Chrome" (nav/footer/breadcrumb/CTA) is detected
|
|
623
|
-
// STRUCTURALLY, not via an English word-list: an anchor text reused across a large share of
|
|
624
|
-
// the site's pages is templated boilerplate — e.g. the EHI homepage's 2,940 inbound "Home"
|
|
625
|
-
// breadcrumb links — and that holds in any language ("Startseite", "Accueil"…). We also drop
|
|
626
|
-
// self-links and anchors with no letters ("(0)", page numbers, arrows). Editorial in-content
|
|
627
|
-
// anchors recur on only a handful of pages, so they survive.
|
|
628
537
|
const anchors = new Map();
|
|
629
538
|
for (const a of rows(c, `
|
|
630
539
|
WITH body_anchors AS (
|
|
@@ -645,18 +554,13 @@ export const CHECKS = [
|
|
|
645
554
|
return top.filter(x => {
|
|
646
555
|
const a = anchors.get(x.urlKey);
|
|
647
556
|
if (!a || a.n < 3)
|
|
648
|
-
return false;
|
|
557
|
+
return false;
|
|
649
558
|
const q = terms(x.query);
|
|
650
559
|
return q.length > 0 && !q.some((w) => titleHasTerm(a.pool, w));
|
|
651
560
|
}).map(x => { const a = anchors.get(x.urlKey); return { urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, inboundInContentLinks: a.n } }; });
|
|
652
561
|
},
|
|
653
562
|
},
|
|
654
563
|
{
|
|
655
|
-
// The "RAG snippetability" test. A local cross-encoder (the kind AI search uses to re-rank) scored
|
|
656
|
-
// every chunk against the page's top query; we persisted the single best-passage score. A low max
|
|
657
|
-
// means no dense, extractable answer anywhere on the page — it will lose in AI/passage search even
|
|
658
|
-
// if it keyword-matches. Model-derived (not deterministic truth) → N label, includeJudgement-gated.
|
|
659
|
-
// Requires `score_passages` to have run (like CWV needs page_lighthouse).
|
|
660
564
|
id: 'weak-passage-answer', category: 'merged', severity: 'high', labels: ['G', 'N'], certainty: 0.8, effortBase: 5, fixType: 'per-page',
|
|
661
565
|
title: 'No passage strongly answers the ranking query (AI-search risk)', fix: 'A local neural reranker found no single passage on this page that confidently answers its top query — the page covers the topic loosely but offers no dense, extractable answer, so AI/passage search will prefer a clearer source. Add a focused, self-contained passage: a heading that states the question + a direct ~50-word answer up top. Run `score_passages` to (re)populate.',
|
|
662
566
|
run: (c) => rows(c, `SELECT url_key urlKey, max_passage_score mps, max_passage_query q, max_passage_impr impr FROM pages
|
|
@@ -664,10 +568,6 @@ export const CHECKS = [
|
|
|
664
568
|
.map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.q, maxPassageScore: x.mps, impressions: x.impr ?? 0, note: 'best passage scores below the reranker confidence threshold' } })),
|
|
665
569
|
},
|
|
666
570
|
{
|
|
667
|
-
// Dejan: search weights the opening heavily and AI answers front-load. If the ranking query's
|
|
668
|
-
// terms are present LATER in the page but absent from the opening (~first 2 chunks / ~200 words),
|
|
669
|
-
// the answer is buried. (body-missing-top-query handles total absence; this is the buried case.)
|
|
670
|
-
// Informational intent only — front-loading matters less for navigational/transactional queries.
|
|
671
571
|
id: 'answer-not-front-loaded', category: 'merged', severity: 'med', labels: ['G', 'N'], certainty: 0.6, effortBase: 3, fixType: 'per-page',
|
|
672
572
|
title: 'Answer to the ranking query is buried, not front-loaded', fix: 'The page covers its top query but the terms don’t appear up top (the intro / first section). Google weights the opening heavily and AI answers front-load — move a direct ~50–100-word answer to the first section. (Heuristic — informational queries.)',
|
|
673
573
|
run: (c) => {
|
|
@@ -699,14 +599,10 @@ export const CHECKS = [
|
|
|
699
599
|
},
|
|
700
600
|
},
|
|
701
601
|
{
|
|
702
|
-
// Dejan "density beats length": AI grounds ~370 words/page, diminishing past ~1,500. A very long
|
|
703
|
-
// page with an over-long unbroken section grounds poorly — split it into focused, headed passages.
|
|
704
602
|
id: 'content-bloat', category: 'content', severity: 'low', labels: ['D', 'N'], certainty: 0.6, effortBase: 5, fixType: 'per-page',
|
|
705
603
|
title: 'Over-long section dilutes AI-grounding (density beats length)', fix: 'This page has a very long unbroken section. AI search grounds only ~370 words per page with sharp diminishing returns past ~1,500 — break the long section into focused, headed passages (or tighten it) so each answers one thing cleanly.',
|
|
706
604
|
run: (c) => {
|
|
707
605
|
const out = [];
|
|
708
|
-
// Use the real (uncapped) word_count ÷ number of headed sections — chunk TEXT is capped at
|
|
709
|
-
// extraction, so we infer over-long sections from words-per-heading, not from chunk length.
|
|
710
606
|
for (const x of iterRows(c, `SELECT url_key urlKey, word_count wc, body_chunks bc FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND word_count >= 2500 AND body_chunks IS NOT NULL`)) {
|
|
711
607
|
let chunks;
|
|
712
608
|
try {
|
|
@@ -724,15 +620,11 @@ export const CHECKS = [
|
|
|
724
620
|
},
|
|
725
621
|
},
|
|
726
622
|
{
|
|
727
|
-
// Hobo Level 3 freshness / lastSignificantUpdate. Gemini guard: YoY windows (negate seasonality
|
|
728
|
-
// + zero-click-SERP CTR loss). Flag when the page hasn't been meaningfully re-dated in >12 months
|
|
729
|
-
// AND clicks are down >25% YoY AND impressions down >15% YoY (impressions confirm ranking decay,
|
|
730
|
-
// not just CTR). Needs ~13 months of GSC — guarded by spanDays so it stays silent on shallow syncs.
|
|
731
623
|
id: 'stale-content', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
732
624
|
title: 'Stale page declining year-on-year', fix: 'This page hasn’t been meaningfully updated in over a year and its Search Console clicks are down sharply versus the same period last year — refresh and expand the content (and honestly re-date it) to rebuild the freshness signal Google rewards.',
|
|
733
625
|
run: (c) => {
|
|
734
626
|
if (!c.gscMaxDate || spanDays(c) < 455)
|
|
735
|
-
return [];
|
|
627
|
+
return [];
|
|
736
628
|
const d = c.gscMaxDate;
|
|
737
629
|
const out = [];
|
|
738
630
|
const rs = rows(c, `
|
|
@@ -748,7 +640,7 @@ export const CHECKS = [
|
|
|
748
640
|
continue;
|
|
749
641
|
const ageDays = (Date.parse(d) - Date.parse(dm)) / 86400000;
|
|
750
642
|
if (!(ageDays >= 365))
|
|
751
|
-
continue;
|
|
643
|
+
continue;
|
|
752
644
|
out.push({ urlKey: x.url, evidence: { url: x.url, dateModified: dm.slice(0, 10), clicksYoY: `${x.pyC}→${x.curC} (-${Math.round((1 - x.curC / x.pyC) * 100)}%)`, impressionsYoY: `${x.pyI}→${x.curI}`, clicks: x.pyC - x.curC, impressions: x.pyI } });
|
|
753
645
|
}
|
|
754
646
|
return out;
|
|
@@ -790,23 +682,17 @@ export const CHECKS = [
|
|
|
790
682
|
without.push(p.urlKey);
|
|
791
683
|
}
|
|
792
684
|
if (pages.length === 0 || withBc / pages.length < 0.4)
|
|
793
|
-
return [];
|
|
685
|
+
return [];
|
|
794
686
|
return without.map(u => ({ urlKey: u, evidence: { note: 'site uses BreadcrumbList elsewhere; missing here' } }));
|
|
795
687
|
},
|
|
796
688
|
},
|
|
797
|
-
// ── Merged crawl × Search Console — the "expert questions" that need both datasets ──
|
|
798
689
|
{
|
|
799
690
|
id: 'ghost-pages', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
800
691
|
title: 'Ranking page the crawl can’t reach', fix: 'Google sends impressions/clicks to this URL but the site crawl never reached it — add internal links so it’s discoverable (or confirm it should exist and isn’t blocked).',
|
|
801
692
|
run: (c) => {
|
|
802
693
|
if (!c.gscMaxDate)
|
|
803
694
|
return [];
|
|
804
|
-
// Only meaningful on a COMPLETE crawl — if the crawl hit its maxPages cap, "absent
|
|
805
|
-
// from crawl" is unreliable. Tie the guard to the crawl that produced the CURRENT
|
|
806
|
-
// pages (not the latest crawl_metadata row, which may be a later failed crawl).
|
|
807
695
|
const m = c.db.prepare('SELECT urls_crawled c, max_pages m FROM crawl_metadata WHERE crawl_id = (SELECT crawl_id FROM pages LIMIT 1)').get();
|
|
808
|
-
// max_pages is nullable (NULL = no cap = complete crawl) — `c >= null` coerces to
|
|
809
|
-
// `c >= 0`, which would silently disable the check forever on capless crawls.
|
|
810
696
|
if (!m || m.c === 0 || (m.m != null && m.c >= m.m))
|
|
811
697
|
return [];
|
|
812
698
|
return rows(c, `SELECT page_key urlKey, SUM(clicks) clicks, SUM(impressions) impressions FROM search_analytics
|
|
@@ -831,7 +717,6 @@ export const CHECKS = [
|
|
|
831
717
|
.map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, title: x.title } }));
|
|
832
718
|
},
|
|
833
719
|
},
|
|
834
|
-
// ── Internal link graph (iPR + click-depth + anchor text, from the crawl `links` table) ──
|
|
835
720
|
{
|
|
836
721
|
id: 'deep-pages', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
837
722
|
title: 'Page buried deep in the structure', fix: 'Add in-content (body) links from higher-level pages — this page is 4+ clicks from the homepage via body links.',
|
|
@@ -843,10 +728,6 @@ export const CHECKS = [
|
|
|
843
728
|
run: (c) => c.gscMaxDate ? rows(c, `SELECT p.url_key urlKey, p.ipr ipr, p.inlink_count inl, SUM(sa.impressions) impressions FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key WHERE p.indexable=1 AND p.ipr < 30 AND p.inlink_count BETWEEN 1 AND 3 AND sa.${win(c.gscMaxDate)} GROUP BY p.url_key HAVING SUM(sa.impressions) >= 300 ORDER BY impressions DESC`).map(r => ({ urlKey: r.urlKey, evidence: { impressions: r.impressions, ipr: Math.round(r.ipr), inlinks: r.inl } })) : [],
|
|
844
729
|
},
|
|
845
730
|
{
|
|
846
|
-
// The tier BETWEEN "orphan" (0 inlinks) and "fine": pages reached almost only via nav/footer.
|
|
847
|
-
// inlink_count counts ALL internal links, so a page sitting in the global nav looks well-linked
|
|
848
|
-
// even with zero EDITORIAL links — yet Google leans on in-content links for topic + equity. We
|
|
849
|
-
// count distinct in-content (placement='body') inbound sources; ≤2 + real demand = under-linked.
|
|
850
731
|
id: 'underlinked-editorial', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
851
732
|
title: 'High demand, almost no in-content internal links', fix: 'This page earns real impressions but is reached mainly via nav/footer — add descriptive in-content links to it from related articles. Editorial body links pass more topical context and equity than templated nav links.',
|
|
852
733
|
run: (c) => c.gscMaxDate ? rows(c, `
|
|
@@ -858,18 +739,6 @@ export const CHECKS = [
|
|
|
858
739
|
HAVING SUM(sa.impressions) >= 300 AND bodyLinks <= 2
|
|
859
740
|
ORDER BY impressions DESC LIMIT 40`).map(r => ({ urlKey: r.urlKey, evidence: { impressions: r.impressions, inContentLinks: r.bodyLinks, note: 'reached mainly via nav/footer — thin on editorial (in-content) links' } })) : [],
|
|
860
741
|
},
|
|
861
|
-
// NOTE: internal anchor-text checks (over-optimisation + generic/empty anchors) prototyped
|
|
862
|
-
// and PULLED twice. Re-evaluated 2026-06-22 against the AgricIDaniel/claude-seo and
|
|
863
|
-
// Bhanunamikaze/Agentic-SEO-Skill repos, this time using the links.placement='body' filter
|
|
864
|
-
// plus excluding anchors that match the target's own title/H1. On real data (ehi.com.au) the
|
|
865
|
-
// dominant survivors are still false positives: sitewide template CTAs ("home" 864/865,
|
|
866
|
-
// "contact us", "apply today") and category links whose anchor IS the page title. The
|
|
867
|
-
// page-level "mostly generic-anchored" variant returned 0 signal; raw empty anchors are
|
|
868
|
-
// image/thumbnail-link noise (3,764/23,435 body links). Reliable detection needs an
|
|
869
|
-
// editorial-vs-template link classifier we don't store (placement='body' still includes
|
|
870
|
-
// in-template CTAs and product grids). Keep pulled — a wrong finding is worse than none.
|
|
871
|
-
// ── Backlinks (need page_backlinks populated via pull_backlinks; gate so we never assert
|
|
872
|
-
// "no external links" without data) ──
|
|
873
742
|
{
|
|
874
743
|
id: 'backlinks-to-404', category: 'crawlability', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
875
744
|
title: 'External backlinks pointing to a dead page', fix: '301-redirect this URL to the best live equivalent — external link equity is hitting a 4xx/5xx page and being wasted.',
|
|
@@ -882,17 +751,14 @@ export const CHECKS = [
|
|
|
882
751
|
run: (c) => {
|
|
883
752
|
const has = c.db.prepare('SELECT COUNT(*) n FROM page_backlinks').get().n;
|
|
884
753
|
if (!has)
|
|
885
|
-
return [];
|
|
754
|
+
return [];
|
|
886
755
|
return rows(c, `SELECT p.url_key urlKey FROM pages p LEFT JOIN page_backlinks b ON b.url_key=p.url_key WHERE p.status_code=200 AND p.indexable=1 AND p.inlink_count=0 AND COALESCE(b.backlinks,0)=0`)
|
|
887
756
|
.map(r => ({ urlKey: r.urlKey, evidence: { inlinks: 0, backlinks: 0 } }));
|
|
888
757
|
},
|
|
889
758
|
},
|
|
890
|
-
// ── Phase 6a — expert questions where crawl (intent) and reality diverge ──
|
|
891
759
|
{
|
|
892
760
|
id: 'ipr-bleed-by-status', category: 'crawlability', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
|
|
893
761
|
title: 'Internal link equity flowing into dead URLs', fix: 'Repoint or 301 these internal links — they target non-200 URLs and waste the internal PageRank of the (often high-authority) pages linking to them.',
|
|
894
|
-
// Sum the iPR of the SOURCE pages linking to each non-200 internal target. "Found a 404" is
|
|
895
|
-
// junior; "this 404 drains the equity of N high-iPR pages" is the director-level find.
|
|
896
762
|
run: (c) => rows(c, `SELECT l.target_key urlKey, COUNT(DISTINCT l.source_key) linkingPages,
|
|
897
763
|
ROUND(SUM(src.ipr), 1) wastedIpr, t.status_code st
|
|
898
764
|
FROM links l
|
|
@@ -905,9 +771,6 @@ export const CHECKS = [
|
|
|
905
771
|
{
|
|
906
772
|
id: 'broken-canonical-target', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
907
773
|
title: 'Canonical points to a broken or unhealthy URL', fix: 'Point the canonical at a live, indexable, self-canonical HTTPS URL — Google ignores a canonical whose target is a 4xx/5xx/redirect, noindex, itself canonicalised elsewhere (a chain/loop), or an HTTPS→HTTP downgrade.',
|
|
908
|
-
// Join the declared canonical_key back to the crawl. Only flag when we crawled the target.
|
|
909
|
-
// Skip self-canonicals. Covers: non-200 target, noindex target, canonical chain/loop (target
|
|
910
|
-
// canonicalises onward), and HTTPS→HTTP downgrade (research: Sitebulb indexability hints).
|
|
911
774
|
run: (c) => rows(c, `SELECT p.url_key urlKey, p.canonical_url canon, p.canonical_key ck, t.status_code st, t.noindex ni, t.canonical_key tck
|
|
912
775
|
FROM pages p JOIN pages t ON t.url_key = p.canonical_key
|
|
913
776
|
WHERE p.canonical_key IS NOT NULL AND p.canonical_key != p.url_key AND p.status_code = 200
|
|
@@ -925,8 +788,6 @@ export const CHECKS = [
|
|
|
925
788
|
{
|
|
926
789
|
id: 'faceted-spider-trap', category: 'crawlability', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'global',
|
|
927
790
|
title: 'Indexable faceted URLs burning crawl budget', fix: 'noindex (or robots-disallow / canonicalise) multi-parameter filter URLs — they are indexable but earn zero search traffic, so they only waste crawl budget and risk index bloat.',
|
|
928
|
-
// Multi-parameter (>=2 params), indexable, zero GSC impressions in-window = classic facet trap.
|
|
929
|
-
// Gate on GSC so "zero search value" is a real claim, not just "no data".
|
|
930
791
|
run: (c) => {
|
|
931
792
|
if (!c.gscMaxDate)
|
|
932
793
|
return [];
|
|
@@ -939,8 +800,6 @@ export const CHECKS = [
|
|
|
939
800
|
{
|
|
940
801
|
id: 'soft-404-shell', category: 'indexation', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
941
802
|
title: 'Soft 404 — 200 OK but Google treats it as not-found', fix: 'Either populate the page with real content, or return a true 404/410 (or noindex) — it serves 200 but Google has flagged it as a soft 404.',
|
|
942
|
-
// URL Inspection page-fetch-state = soft 404 while the crawler sees a 200. The crawler alone
|
|
943
|
-
// would call this page fine; Google disagrees. Needs URL Inspection populated.
|
|
944
803
|
run: (c) => rows(c, `SELECT i.url_key urlKey, p.word_count wc, p.bytes bytes
|
|
945
804
|
FROM url_inspection i JOIN pages p ON p.url_key = i.url_key
|
|
946
805
|
WHERE LOWER(i.page_fetch_state) LIKE '%soft%' AND p.status_code = 200`).map(r => ({ urlKey: r.urlKey, evidence: { pageFetchState: 'soft 404', wordCount: r.wc, bytes: r.bytes } })),
|
|
@@ -948,9 +807,6 @@ export const CHECKS = [
|
|
|
948
807
|
{
|
|
949
808
|
id: 'rich-result-issues', category: 'schema', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
950
809
|
title: 'Google-verified rich result issues (URL Inspection)', fix: 'Fix the structured-data issues Google itself reports for this page — these come from the URL Inspection API (Google’s own validation), not our local validator, so they are authoritative. Address the listed issue messages per rich-result type.',
|
|
951
|
-
// Parses the stored richResultsResult JSON (url_inspection.rich_results) that inspect_urls
|
|
952
|
-
// already captures: detectedItems[].items[].issues[] carries Google's severity + message.
|
|
953
|
-
// Needs URL Inspection populated (run inspect_urls). Zero API cost — data is already in the DB.
|
|
954
810
|
run: (c) => {
|
|
955
811
|
const out = [];
|
|
956
812
|
for (const r of iterRows(c, `SELECT url_key urlKey, rich_results rr FROM url_inspection WHERE rich_results IS NOT NULL AND rich_results != ''`)) {
|
|
@@ -961,8 +817,6 @@ export const CHECKS = [
|
|
|
961
817
|
catch {
|
|
962
818
|
continue;
|
|
963
819
|
}
|
|
964
|
-
// Dedup per (type, severity, message) with a count — a listicle repeats the same
|
|
965
|
-
// "Missing field review" warning per product item; keep one sample item per group.
|
|
966
820
|
const groups = new Map();
|
|
967
821
|
for (const det of parsed?.detectedItems ?? []) {
|
|
968
822
|
for (const item of det?.items ?? []) {
|
|
@@ -994,17 +848,14 @@ export const CHECKS = [
|
|
|
994
848
|
title: 'hreflang missing return tag (not reciprocated)', fix: 'Add the reciprocal hreflang on the target page — Google ignores one-way hreflang annotations that don’t link back.',
|
|
995
849
|
run: (c) => hreflangFindings(c).noReturn,
|
|
996
850
|
},
|
|
997
|
-
// ── 6a finishers — consume persisted DataForSEO enrichments (gated; need the data pulled) ──
|
|
998
851
|
{
|
|
999
852
|
id: 'intent-vs-pagetype-mismatch', category: 'merged', severity: 'med', labels: ['G', 'N'], certainty: 0.7, effortBase: 8, fixType: 'per-page',
|
|
1000
853
|
title: 'Page type mismatches its query intent', fix: 'Reformat or retarget the page — its template doesn’t match the SERP intent for its top query (e.g. a product page ranking for an informational query, or an article for a transactional one). Run search_intent siteUrl:<property> to populate intents.',
|
|
1001
|
-
// Join each page's top GSC query → persisted keyword_intent → page schema flavour (from json_ld).
|
|
1002
|
-
// Judgement (N, 0.7): intent + schema-type inference is heuristic, so it only runs with includeJudgement.
|
|
1003
854
|
run: (c) => {
|
|
1004
855
|
if (!c.gscMaxDate)
|
|
1005
856
|
return [];
|
|
1006
857
|
if ((c.db.prepare('SELECT COUNT(*) n FROM keyword_intent').get().n) === 0)
|
|
1007
|
-
return [];
|
|
858
|
+
return [];
|
|
1008
859
|
const r = rows(c, `SELECT s.page_key urlKey, s.query query, ki.intent intent, p.json_ld jsonLd FROM
|
|
1009
860
|
(SELECT page_key, query, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
|
|
1010
861
|
FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
|
|
@@ -1035,7 +886,7 @@ export const CHECKS = [
|
|
|
1035
886
|
if (!c.gscMaxDate)
|
|
1036
887
|
return [];
|
|
1037
888
|
if ((c.db.prepare('SELECT COUNT(*) n FROM page_cwv').get().n) === 0)
|
|
1038
|
-
return [];
|
|
889
|
+
return [];
|
|
1039
890
|
return rows(c, `SELECT cw.url_key urlKey, cw.lcp_ms lcp, cw.cls cls, cw.performance perf,
|
|
1040
891
|
(SELECT COALESCE(SUM(clicks),0) FROM search_analytics sa WHERE sa.page_key = cw.url_key AND ${win(c.gscMaxDate)}) clicks
|
|
1041
892
|
FROM page_cwv cw
|
|
@@ -1044,14 +895,13 @@ export const CHECKS = [
|
|
|
1044
895
|
.map(r => ({ urlKey: r.urlKey, evidence: { clicks: r.clicks, lcpMs: r.lcp != null ? Math.round(r.lcp) : null, cls: r.cls != null ? Math.round(r.cls * 1000) / 1000 : null, performance: r.perf != null ? Math.round(r.perf * 100) : null } }));
|
|
1045
896
|
},
|
|
1046
897
|
},
|
|
1047
|
-
// ── 6b — per-template systemic issues (template-typed, deterministic) ──
|
|
1048
898
|
{
|
|
1049
899
|
id: 'pagination-canonical-to-page-1', category: 'crawlability', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
|
|
1050
900
|
title: 'Paginated pages canonicalising away from themselves', fix: 'Make each paginated page (page 2, 3, …) self-canonical. Canonicalising page 2+ back to page 1 tells Google the deeper pages are duplicates, so products/articles linked only from page 2+ drop out of the crawl.',
|
|
1051
901
|
run: (c) => rows(c, `SELECT url_key urlKey, url, canonical_url canon FROM pages
|
|
1052
902
|
WHERE status_code=200 AND canonical_count>0 AND canonical_key IS NOT NULL AND canonical_key != url_key
|
|
1053
903
|
AND (rel_prev=1 OR url LIKE '%/page/%' OR url GLOB '*[?&]page=[0-9]*' OR url GLOB '*[?&]paged=[0-9]*' OR url GLOB '*[?&]p=[0-9]*')`)
|
|
1054
|
-
.filter(r => !/[?&](page|p)=1(\b|&|$)/.test(r.url) && !/\/page\/1(\/?$|\?)/.test(r.url))
|
|
904
|
+
.filter(r => !/[?&](page|p)=1(\b|&|$)/.test(r.url) && !/\/page\/1(\/?$|\?)/.test(r.url))
|
|
1055
905
|
.map(r => ({ urlKey: r.urlKey, evidence: { url: r.url, canonical: r.canon } })),
|
|
1056
906
|
},
|
|
1057
907
|
{
|
|
@@ -1074,13 +924,12 @@ export const CHECKS = [
|
|
|
1074
924
|
return out;
|
|
1075
925
|
},
|
|
1076
926
|
},
|
|
1077
|
-
// ── 6d — Wikidata entity layer (heuristic H1→QID; N/judgement, gated on resolve_entities) ──
|
|
1078
927
|
{
|
|
1079
928
|
id: 'entity-internal-link-gap', category: 'crawlability', severity: 'low', labels: ['N'], certainty: 0.5, effortBase: 3, fixType: 'per-page',
|
|
1080
929
|
title: 'Topically related pages not internally linked', fix: 'Add an internal link from the broader page to the more specific one — Wikidata says their entities are related (subclass-of / part-of) but no internal link connects them, leaving a gap in the topical mesh. Run resolve_entities first; verify the entity match before acting (heuristic).',
|
|
1081
930
|
run: (c) => {
|
|
1082
931
|
if ((c.db.prepare('SELECT COUNT(*) n FROM page_entity').get().n) === 0)
|
|
1083
|
-
return [];
|
|
932
|
+
return [];
|
|
1084
933
|
return rows(c, `SELECT parent.url_key urlKey, child.url_key target, parent.label pl, child.label cl, ee.relation rel
|
|
1085
934
|
FROM entity_edge ee
|
|
1086
935
|
JOIN page_entity child ON child.qid = ee.qid
|
|
@@ -1090,7 +939,6 @@ export const CHECKS = [
|
|
|
1090
939
|
.map(r => ({ urlKey: r.urlKey, evidence: { suggestLinkTo: r.target, parentEntity: r.pl, childEntity: r.cl, relation: r.rel } }));
|
|
1091
940
|
},
|
|
1092
941
|
},
|
|
1093
|
-
// ── Sitemap ↔ crawl reconciliation (gated on a sitemap having been fetched) ──
|
|
1094
942
|
{
|
|
1095
943
|
id: 'sitemap-non-indexable', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
|
|
1096
944
|
title: 'Sitemap lists non-indexable URLs', fix: 'Remove URLs from the XML sitemap that are 4xx/5xx, redirected, noindex, canonicalised or robots-blocked — the sitemap should list only canonical, indexable pages, or Google loses trust in it.',
|
|
@@ -1121,7 +969,6 @@ export const CHECKS = [
|
|
|
1121
969
|
.map(r => ({ urlKey: r.urlKey, evidence: { inlinks: 0, note: 'in sitemap, no internal links' } }));
|
|
1122
970
|
},
|
|
1123
971
|
},
|
|
1124
|
-
// ── Cheap performance proxies (from data captured at crawl time — no extra fetch) ──
|
|
1125
972
|
{
|
|
1126
973
|
id: 'slow-response', category: 'performance', severity: 'med', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'global',
|
|
1127
974
|
title: 'Slow server response (TTFB proxy)', fix: 'Investigate slow server/TTFB — caching, CDN, or backend. Response time over ~1.5s hurts Core Web Vitals and crawl rate.',
|
|
@@ -1131,11 +978,6 @@ export const CHECKS = [
|
|
|
1131
978
|
{
|
|
1132
979
|
id: 'large-html', category: 'performance', severity: 'low', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
1133
980
|
title: 'Large HTML document (transferred)', fix: 'Trim the HTML payload — bloated markup slows render and First Contentful Paint (often huge inline SVG/CSS/JSON or unminified output). Judged on TRANSFERRED bytes, not raw: behind a compressing CDN (brotli/gzip) raw size matters far less.',
|
|
1134
|
-
// Severity is on what the browser actually downloads, not raw bytes: a 200KB page served brotli
|
|
1135
|
-
// is ~45KB over the wire and is NOT a real perf problem. We estimate transfer size (raw × ~0.22
|
|
1136
|
-
// for br/gzip, else raw) and only flag pages whose ESTIMATED transferred HTML exceeds ~60KB.
|
|
1137
|
-
// Prefilter at the 60KB threshold itself, not higher — an UNCOMPRESSED 60–150KB page
|
|
1138
|
-
// (est = raw) is exactly the case that matters most and must reach the est filter.
|
|
1139
981
|
run: (c) => rows(c, `SELECT url_key urlKey, bytes, content_encoding enc FROM pages WHERE status_code=200 AND ${HTML_CT} AND bytes > 60000 ORDER BY bytes DESC`)
|
|
1140
982
|
.map(r => { const compressed = /br|gzip|deflate|zstd/i.test(r.enc || ''); const est = compressed ? Math.round(r.bytes * 0.22) : r.bytes; return { urlKey: r.urlKey, est, compressed, raw: r.bytes }; })
|
|
1141
983
|
.filter(x => x.est > 60000)
|
|
@@ -1147,7 +989,6 @@ export const CHECKS = [
|
|
|
1147
989
|
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (content_encoding IS NULL OR content_encoding='')`)
|
|
1148
990
|
.map(r => ({ urlKey: r.urlKey, evidence: {} })),
|
|
1149
991
|
},
|
|
1150
|
-
// ── Checklist-coverage additions (2026-07-20 — see plan/checklist-coverage.md) ──
|
|
1151
992
|
{
|
|
1152
993
|
id: 'robots-blocked-with-traffic', category: 'crawlability', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
1153
994
|
title: 'Robots-blocked page still earning search traffic', fix: 'This URL is disallowed in robots.txt yet Google still shows it (usually as a bare "no information" result) and users still land on it. Either unblock it so it can be crawled and ranked properly, or — if it genuinely shouldn\'t be found — unblock it AND add noindex (a robots-blocked page can never see the noindex).',
|
|
@@ -1184,8 +1025,6 @@ export const CHECKS = [
|
|
|
1184
1025
|
{
|
|
1185
1026
|
id: 'favicon-missing', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
1186
1027
|
title: 'No favicon declared', fix: 'Add a favicon (<link rel="icon" …>) — Google shows it next to your result on mobile, and a missing one costs a little trust/recognition on every SERP appearance. One line in the template.',
|
|
1187
|
-
// Gated on the column being populated (has_favicon is NULL on crawls from before this
|
|
1188
|
-
// was captured) — never flag from a pre-feature crawl.
|
|
1189
1028
|
run: (c) => {
|
|
1190
1029
|
const hp = c.db.prepare(`SELECT url_key, has_favicon hf FROM pages WHERE status_code=200 AND has_favicon IS NOT NULL ORDER BY (click_depth=0) DESC, inlink_count DESC LIMIT 1`).get();
|
|
1191
1030
|
return hp && hp.hf === 0 ? [{ urlKey: hp.url_key, evidence: { note: 'no <link rel="icon"> on the homepage' } }] : [];
|
|
@@ -1197,10 +1036,9 @@ export const CHECKS = [
|
|
|
1197
1036
|
run: (c) => {
|
|
1198
1037
|
const all = c.db.prepare(`SELECT url_key, lastmod FROM sitemap_urls WHERE lastmod IS NOT NULL AND lastmod <> ''`).all();
|
|
1199
1038
|
if (all.length < 20)
|
|
1200
|
-
return [];
|
|
1039
|
+
return [];
|
|
1201
1040
|
const ev = { urlsWithLastmod: all.length };
|
|
1202
1041
|
let tripped = false;
|
|
1203
|
-
// Tell 1: a generator stamping every URL with "now" — >90% share one date, and that date is recent.
|
|
1204
1042
|
const byDay = new Map();
|
|
1205
1043
|
for (const r of all)
|
|
1206
1044
|
byDay.set(r.lastmod.slice(0, 10), (byDay.get(r.lastmod.slice(0, 10)) ?? 0) + 1);
|
|
@@ -1210,19 +1048,13 @@ export const CHECKS = [
|
|
|
1210
1048
|
tripped = true;
|
|
1211
1049
|
ev.sharedStamp = `${Math.round(topN / all.length * 100)}% of URLs claim ${topDay} — a generation timestamp, not a change date`;
|
|
1212
1050
|
}
|
|
1213
|
-
// Tell 2: dates in the future.
|
|
1214
1051
|
const future = all.filter(r => Date.parse(r.lastmod) > Date.now() + 86400000).length;
|
|
1215
1052
|
if (future > 0) {
|
|
1216
1053
|
tripped = true;
|
|
1217
1054
|
ev.futureDates = future;
|
|
1218
1055
|
}
|
|
1219
|
-
// Tell 3: lastmod claims a change between our two most recent crawls, yet none of the
|
|
1220
|
-
// tracked page fields (status, title, meta, H1, word count, schema types) changed.
|
|
1221
1056
|
const crawls = latestTwoCrawls(c.db);
|
|
1222
1057
|
if (crawls.length === 2) {
|
|
1223
|
-
// Compare DATE prefixes on both sides — lastmod is stored verbatim and often a full
|
|
1224
|
-
// timestamp; compared raw against a 10-char date, same-day stamps sort "after" the
|
|
1225
|
-
// boundary and are silently excluded (exactly the stamp-everything-today pattern).
|
|
1226
1058
|
const phantom = c.db.prepare(`SELECT s.url_key FROM sitemap_urls s
|
|
1227
1059
|
JOIN page_snapshots n ON n.url_key = s.url_key AND n.crawl_id = ?
|
|
1228
1060
|
JOIN page_snapshots o ON o.url_key = s.url_key AND o.crawl_id = ?
|
|
@@ -1242,8 +1074,6 @@ export const CHECKS = [
|
|
|
1242
1074
|
{
|
|
1243
1075
|
id: 'no-304-revalidation', category: 'performance', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
1244
1076
|
title: 'Server ignores conditional requests (no 304)', fix: 'Pages advertise Last-Modified/ETag but the server re-serves a full 200 when asked "has this changed?" (If-Modified-Since / If-None-Match). A properly configured server answers 304 Not Modified — it saves bandwidth on every revalidating crawler and cache, and signals stability to Googlebot. Usually a server/CDN setting.',
|
|
1245
|
-
// Populated by the crawler's post-crawl probe (pages.conditional_304); NULL-gated so
|
|
1246
|
-
// pre-feature crawls never flag. Fires only when NO probed page honoured the request.
|
|
1247
1077
|
run: (c) => {
|
|
1248
1078
|
const r = c.db.prepare(`SELECT COUNT(*) probed, COALESCE(SUM(conditional_304),0) ok FROM pages WHERE conditional_304 IS NOT NULL`).get();
|
|
1249
1079
|
if (r.probed < 5 || r.ok > 0)
|
|
@@ -1255,45 +1085,12 @@ export const CHECKS = [
|
|
|
1255
1085
|
{
|
|
1256
1086
|
id: 'analytics-missing', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
1257
1087
|
title: 'No client-side analytics detected', fix: 'No analytics or tag-manager snippet was found on any crawled page (GA4, GTM, Plausible, Matomo, Fathom, Clarity…). If you measure server-side, ignore this; otherwise you\'re flying blind — install an analytics package before making SEO decisions.',
|
|
1258
|
-
// Fires only when EVERY populated page lacks a snippet — one page with analytics = installed.
|
|
1259
1088
|
run: (c) => {
|
|
1260
1089
|
const r = c.db.prepare(`SELECT COUNT(*) total, COALESCE(SUM(has_analytics),0) withA FROM pages WHERE status_code=200 AND ${HTML_CT} AND has_analytics IS NOT NULL`).get();
|
|
1261
1090
|
return r.total >= 3 && r.withA === 0 ? [{ urlKey: null, evidence: { pagesChecked: r.total, note: 'no known analytics snippet on any crawled page (server-side measurement is invisible to a crawl)' } }] : [];
|
|
1262
1091
|
},
|
|
1263
1092
|
},
|
|
1264
|
-
|
|
1265
|
-
// fundamentals/ai-optimization-guide, read 2026-08-02) — principle → check mapping:
|
|
1266
|
-
// • Unique/compelling/non-commodity, people-first content .... thin-content, content-bloat,
|
|
1267
|
-
// ai-slop-signals (NEW below — mechanical/templated prose is the anti-signal of "unique take")
|
|
1268
|
-
// • Unique point of view / first-hand experience .............. article-no-author (NEW below —
|
|
1269
|
-
// unattributed articles are the deterministically checkable slice), stale-content
|
|
1270
|
-
// • Clear organisation (paragraphs/sections/headings) ......... poor-chunkability, content-bloat,
|
|
1271
|
-
// heading-hierarchy; answer coverage: body-missing-top-query, rag-answer-gap,
|
|
1272
|
-
// answer-not-front-loaded, weak-passage-answer, low-extractability
|
|
1273
|
-
// • Indexed + snippet-eligible / technical requirements ....... indexation family (noindex,
|
|
1274
|
-
// canonical, robots-blocked-with-traffic), soft-404-shell, sitemap reconciliation
|
|
1275
|
-
// • Crawlable public content .................................. crawlability family, broken links,
|
|
1276
|
-
// redirect chains; freshness-honesty (sitemap-lastmod-untrustworthy, no-304-revalidation)
|
|
1277
|
-
// • Semantic HTML / parseable ................................. heading-hierarchy, missing-h1,
|
|
1278
|
-
// multiple-h1, missing-lang
|
|
1279
|
-
// • Reduce duplicate content .................................. duplicate-title/meta, canonical
|
|
1280
|
-
// family, faceted-spider-trap, keyword-cannibalisation
|
|
1281
|
-
// • Images/video supporting text .............................. image-alt, images-missing-dimensions
|
|
1282
|
-
// • Page experience / latency ................................. performance proxies, high-yield-cwv-fail
|
|
1283
|
-
// • Structured data honesty (not required, but keep it valid) . schema-validate family,
|
|
1284
|
-
// article-date-illogical
|
|
1285
|
-
// • Don't chunk artificially / rewrite for AI ................. covered by NOT having such checks;
|
|
1286
|
-
// our chunk checks reward structure for humans, not tiny AI fragments
|
|
1287
|
-
// • "Don't create llms.txt" ................................... tension with agent-readiness probes,
|
|
1288
|
-
// which score llms.txt for *agent* (not Google AI-search) consumption — left as-is, different audience
|
|
1289
|
-
// • Merchant Center / Business Profile / GenAI report ......... out of scope (not crawl/GSC data)
|
|
1290
|
-
{
|
|
1291
|
-
// Anti-signal of the guide's "unique, non-commodity, people-first content": prose that reads
|
|
1292
|
-
// machine-generated. Three heuristics over body_chunks — slop-lexicon density, sentence-length
|
|
1293
|
-
// uniformity (low variance = mechanical), repeated chunk openers (page-internal boilerplate).
|
|
1294
|
-
// Precision over recall: absolute floors on every signal, ≥2 signals required, AND the composite
|
|
1295
|
-
// score must sit in the top decile of pages showing any signal. Judgement-gated — a human wrote
|
|
1296
|
-
// "delve" long before LLMs did.
|
|
1093
|
+
{
|
|
1297
1094
|
id: 'ai-slop-signals', category: 'content', severity: 'low', labels: ['N'], certainty: 0.5, effortBase: 5, fixType: 'per-page',
|
|
1298
1095
|
title: 'Prose shows machine-generated (slop) signals', fix: 'This page’s copy trips several statistical tells of generic AI-generated text: stock filler phrases, unusually uniform sentence lengths, and/or sections that all open the same way. Google’s AI-search guidance rewards unique, people-first content with a first-hand point of view — rewrite the flagged sections with specifics only you can supply (real experience, real numbers, real opinions) and cut the filler. (Heuristic — verify by reading the page; competent human writing can trip these tells.)',
|
|
1299
1096
|
run: (c) => {
|
|
@@ -1322,7 +1119,6 @@ export const CHECKS = [
|
|
|
1322
1119
|
const bodyWords = body.split(/\s+/).filter(Boolean).length;
|
|
1323
1120
|
if (bodyWords < 250)
|
|
1324
1121
|
continue;
|
|
1325
|
-
// (i) slop-lexicon density (hits per 1000 words, ≥3 distinct terms required)
|
|
1326
1122
|
const hits = new Map();
|
|
1327
1123
|
for (const p of PHRASES) {
|
|
1328
1124
|
let n = 0, i = -1;
|
|
@@ -1334,7 +1130,6 @@ export const CHECKS = [
|
|
|
1334
1130
|
const totalHits = [...hits.values()].reduce((a, b) => a + b, 0);
|
|
1335
1131
|
const density = totalHits / bodyWords * 1000;
|
|
1336
1132
|
const lexSignal = hits.size >= 3 && density >= 2.5;
|
|
1337
|
-
// (ii) sentence-length uniformity — coefficient of variation over sentence word-counts
|
|
1338
1133
|
const sentences = body.split(/[.!?]+\s/).map(s => s.split(/\s+/).filter(Boolean).length).filter(n => n >= 5 && n <= 60);
|
|
1339
1134
|
let cv = null;
|
|
1340
1135
|
if (sentences.length >= 12) {
|
|
@@ -1342,11 +1137,9 @@ export const CHECKS = [
|
|
|
1342
1137
|
cv = Math.sqrt(sentences.reduce((a, b) => a + (b - mean) ** 2, 0) / sentences.length) / mean;
|
|
1343
1138
|
}
|
|
1344
1139
|
const uniformSignal = cv !== null && cv < 0.28;
|
|
1345
|
-
// (iii) repeated openers across the page's own chunks (first 3 words, ≥3 chunks sharing one)
|
|
1346
1140
|
const openers = new Map();
|
|
1347
1141
|
if (texts.length >= 5)
|
|
1348
1142
|
for (const t of texts) {
|
|
1349
|
-
// Letters only — numeric/UI-chrome openers ("Show 10 20…") are widget text, not prose.
|
|
1350
1143
|
const o = t.toLowerCase().replace(/[^a-z\s]/g, ' ').split(/\s+/).filter(w => w.length >= 2).slice(0, 3).join(' ');
|
|
1351
1144
|
if (o.split(' ').length === 3)
|
|
1352
1145
|
openers.set(o, (openers.get(o) ?? 0) + 1);
|
|
@@ -1361,10 +1154,6 @@ export const CHECKS = [
|
|
|
1361
1154
|
}
|
|
1362
1155
|
if (!metrics.length)
|
|
1363
1156
|
return [];
|
|
1364
|
-
// Outliers only: the LEXICON signal is mandatory (uniform sentences + repeated openers
|
|
1365
|
-
// without a single slop phrase is template chrome, not slop — live-verified on simracing),
|
|
1366
|
-
// plus ≥1 corroborating signal, AND composite score in the top decile of pages that showed
|
|
1367
|
-
// any signal at all.
|
|
1368
1157
|
const sorted = metrics.map(m => m.score).sort((a, b) => a - b);
|
|
1369
1158
|
const p90 = sorted[Math.min(sorted.length - 1, Math.floor(sorted.length * 0.9))];
|
|
1370
1159
|
return metrics
|
|
@@ -1381,11 +1170,6 @@ export const CHECKS = [
|
|
|
1381
1170
|
},
|
|
1382
1171
|
},
|
|
1383
1172
|
{
|
|
1384
|
-
// The guide's "unique point of view based on personal experience or expertise", cut down to its
|
|
1385
|
-
// deterministically checkable slice: an Article/BlogPosting that carries schema yet names no
|
|
1386
|
-
// author. Attribution is the machine-readable experience/expertise signal; a byline-less article
|
|
1387
|
-
// is the commodity-content default. Only fires where Article schema EXISTS (no schema at all is
|
|
1388
|
-
// schema-opportunity territory, not this check).
|
|
1389
1173
|
id: 'article-no-author', category: 'schema', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
1390
1174
|
title: 'Article schema with no author attribution', fix: 'This page marks itself up as an Article/BlogPosting but declares no author. Google’s AI-search guidance rewards content with a demonstrable first-hand point of view — add `author` (a Person with a real name, ideally linking to an author page) to the Article schema and a visible byline to match.',
|
|
1391
1175
|
run: (c) => {
|
|
@@ -1407,7 +1191,117 @@ export const CHECKS = [
|
|
|
1407
1191
|
return out;
|
|
1408
1192
|
},
|
|
1409
1193
|
},
|
|
1194
|
+
{
|
|
1195
|
+
id: 'discover-max-image-preview', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global', yieldCoef: 0.15,
|
|
1196
|
+
title: 'Article missing max-image-preview:large (no large Discover card)',
|
|
1197
|
+
fix: 'Add `<meta name="robots" content="max-image-preview:large">` (or the same directive in an X-Robots-Tag header). Without it Google cannot feature the article with the large, high-CTR image card in Google Discover and image search — you get a thumbnail or nothing. This is usually one site-wide template/config change (WordPress core already emits it; a plugin or theme has overridden it here).',
|
|
1198
|
+
run: (c) => rows(c, `SELECT url_key urlKey, json_ld jsonLd, og_tags ogTags, robots, x_robots_tag xrt FROM pages WHERE ${DISCOVER_PREFILTER}`)
|
|
1199
|
+
.filter(r => isDiscoverEligible(r.jsonLd, r.ogTags) && !hasMaxImagePreviewLarge(r.robots, r.xrt) && !hasPreviewRestriction(r.robots, r.xrt))
|
|
1200
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { robots: r.robots ?? null, xRobotsTag: r.xrt ?? null, note: 'article-type page without max-image-preview:large' } })),
|
|
1201
|
+
},
|
|
1202
|
+
{
|
|
1203
|
+
id: 'discover-missing-og', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page', yieldCoef: 0.12,
|
|
1204
|
+
title: 'Article missing og:title or og:image (Discover cards use these)',
|
|
1205
|
+
fix: 'Add both `og:title` and `og:image` — Google frequently builds the Discover headline and card image straight from Open Graph. Make og:image ≥1200px wide at 16:9, and write og:title for click-through, not just for social.',
|
|
1206
|
+
run: (c) => rows(c, `SELECT url_key urlKey, json_ld jsonLd, og_tags ogTags FROM pages WHERE ${DISCOVER_PREFILTER}`)
|
|
1207
|
+
.filter(r => isDiscoverEligible(r.jsonLd, r.ogTags))
|
|
1208
|
+
.map(r => ({ urlKey: r.urlKey, missing: ['og:title', 'og:image'].filter(k => { const v = parseOg(r.ogTags)[k]; return !(v && String(v).trim()); }) }))
|
|
1209
|
+
.filter(x => x.missing.length)
|
|
1210
|
+
.map(x => ({ urlKey: x.urlKey, evidence: { missing: x.missing } })),
|
|
1211
|
+
},
|
|
1212
|
+
{
|
|
1213
|
+
id: 'discover-slow-response', category: 'performance', severity: 'med', labels: ['D', 'N'], certainty: 0.5, effortBase: 5, fixType: 'global', yieldCoef: 0.1,
|
|
1214
|
+
title: 'Article response looks slow for Discover (>600ms wall-time)',
|
|
1215
|
+
fix: 'Aim for server response (TTFB) under 600ms — under 200ms is optimal — since Discover favours freshly-crawlable content. NOTE: the figure here is whole-fetch WALL TIME (it includes the HTML download and any retry/back-off), not a clean server TTFB, so confirm against a real TTFB or field measurement before prioritising. Caching, a CDN, or backend work are the usual levers.',
|
|
1216
|
+
run: (c) => rows(c, `SELECT url_key urlKey, json_ld jsonLd, og_tags ogTags, response_time_ms ms FROM pages WHERE ${DISCOVER_PREFILTER} AND response_time_ms > 600 AND (redirects IS NULL OR redirects='') ORDER BY response_time_ms DESC`)
|
|
1217
|
+
.filter(r => isDiscoverEligible(r.jsonLd, r.ogTags))
|
|
1218
|
+
.map(r => ({ urlKey: r.urlKey, evidence: { wallTimeMs: r.ms, target: '<600ms (ideal <200ms)', note: 'whole-fetch wall time incl. HTML download / retries — a proxy, not a measured TTFB' } })),
|
|
1219
|
+
},
|
|
1220
|
+
{
|
|
1221
|
+
id: 'discover-crawl-waste', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'global', yieldCoef: 0.1,
|
|
1222
|
+
title: 'High crawl waste (budget spent on redirects / non-indexable URLs)',
|
|
1223
|
+
fix: 'Google allocates a finite crawl budget per host; spending it on redirects, error pages and non-indexable URLs slows discovery of your real content (and Discover leans on fast discovery). Aim for almost everything Google crawls to be an indexable 200 — repoint internal links off redirects/404s, prune faceted/parameter URLs, and keep the sitemap to canonical indexable pages.',
|
|
1224
|
+
run: (c) => {
|
|
1225
|
+
const clause = `status_code IS NOT NULL AND status_code NOT IN (429,503)`;
|
|
1226
|
+
const tot = (rows(c, `SELECT COUNT(*) n FROM pages WHERE ${clause}`)[0]?.n) || 0;
|
|
1227
|
+
if (tot < 50)
|
|
1228
|
+
return [];
|
|
1229
|
+
const byReason = rows(c, `SELECT COALESCE(indexable_reason,'(other)') reason, COUNT(*) n FROM pages WHERE ${clause} AND indexable=0 AND COALESCE(indexable_reason,'') != 'non-html' GROUP BY 1 ORDER BY 2 DESC`);
|
|
1230
|
+
const nonIndexable = byReason.reduce((s, r) => s + r.n, 0);
|
|
1231
|
+
const redirected = (rows(c, `SELECT COUNT(*) n FROM pages WHERE ${clause} AND redirects IS NOT NULL AND redirects != ''`)[0]?.n) || 0;
|
|
1232
|
+
const waste = nonIndexable + redirected;
|
|
1233
|
+
const ratio = waste / tot;
|
|
1234
|
+
if (ratio < 0.3)
|
|
1235
|
+
return [];
|
|
1236
|
+
return [{ urlKey: null, evidence: { crawledUrls: tot, wasted: waste, wasteRatio: Math.round(ratio * 1000) / 10 + '%', redirected, nonIndexableByReason: Object.fromEntries(byReason.map(r => [r.reason, r.n])), note: 'crawled URLs that redirect or cannot be indexed (429/503 throttling and non-HTML assets excluded)' } }];
|
|
1237
|
+
},
|
|
1238
|
+
},
|
|
1239
|
+
{
|
|
1240
|
+
id: 'discover-generic-article-type', category: 'schema', severity: 'low', labels: ['D', 'N'], certainty: 0.5, effortBase: 1, fixType: 'global',
|
|
1241
|
+
title: 'Generic Article type where a more specific one fits',
|
|
1242
|
+
fix: 'The page’s Article markup uses only the generic `Article` type. Google recommends the most specific applicable type — `NewsArticle` for news, `LiveBlogPosting` for live coverage, `ProfilePage` for a person/creator profile — which unlocks richer Discover/News treatment. Usually one template/plugin setting. Only change it if the more specific type genuinely fits (judgement).',
|
|
1243
|
+
run: (c) => rows(c, `SELECT url_key urlKey, json_ld jsonLd FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND json_ld LIKE '%Article%'`)
|
|
1244
|
+
.map(r => { const present = new Set(); for (const n of discoverArticleNodes(r.jsonLd))
|
|
1245
|
+
for (const t of nodeTypeList(n))
|
|
1246
|
+
if (DISCOVER_TYPES.includes(t))
|
|
1247
|
+
present.add(t); return { urlKey: r.urlKey, present }; })
|
|
1248
|
+
.filter(x => x.present.has('article') && x.present.size === 1)
|
|
1249
|
+
.map(x => ({ urlKey: x.urlKey, evidence: { articleTypesDeclared: ['Article'], note: 'the Article schema declares only the generic Article type — no more specific subtype' } })),
|
|
1250
|
+
},
|
|
1251
|
+
{
|
|
1252
|
+
id: 'discover-image-schema', category: 'schema', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page', yieldCoef: 0.12,
|
|
1253
|
+
title: 'Article schema declares no image (Discover needs a large image)',
|
|
1254
|
+
fix: 'Add an `image` to the Article structured data — Discover cards need a big image and cannot build one from the page alone. Google wants a high-res image (≥1200px wide), ideally in three crops: 16:9 (1200×675), 4:3 (1200×900) and 1:1 (1200×1200). The main visible image at the top of the article should match the 16:9 schema image.',
|
|
1255
|
+
run: (c) => rows(c, `SELECT url_key urlKey, json_ld jsonLd, og_tags ogTags FROM pages WHERE ${DISCOVER_PREFILTER}`)
|
|
1256
|
+
.map(r => ({ urlKey: r.urlKey, nodes: discoverArticleNodes(r.jsonLd) }))
|
|
1257
|
+
.filter(x => x.nodes.length > 0 && !x.nodes.some(hasSchemaImage))
|
|
1258
|
+
.map(x => ({ urlKey: x.urlKey, evidence: { note: 'no `image` on the Article schema — Discover cannot build a large image card from schema alone' } })),
|
|
1259
|
+
},
|
|
1410
1260
|
];
|
|
1261
|
+
const DISCOVER_TYPES = ['article', 'newsarticle', 'blogposting', 'liveblogposting', 'profilepage', 'reportagenewsarticle', 'opinionnewsarticle', 'reviewnewsarticle', 'techarticle', 'scholarlyarticle'];
|
|
1262
|
+
const DISCOVER_PREFILTER = `status_code=200 AND indexable=1 AND ${HTML_CT} AND (json_ld LIKE '%Article%' OR json_ld LIKE '%BlogPosting%' OR json_ld LIKE '%ProfilePage%' OR og_tags LIKE '%article%')`;
|
|
1263
|
+
function nodeTypeList(n) {
|
|
1264
|
+
const t = n?.['@type'];
|
|
1265
|
+
if (typeof t === 'string')
|
|
1266
|
+
return [t.toLowerCase()];
|
|
1267
|
+
if (Array.isArray(t))
|
|
1268
|
+
return t.filter((x) => typeof x === 'string').map(x => x.toLowerCase());
|
|
1269
|
+
return [];
|
|
1270
|
+
}
|
|
1271
|
+
function discoverArticleNodes(jsonLd) {
|
|
1272
|
+
return parseJsonLdNodes(jsonLd).filter(n => nodeTypeList(n).some(t => DISCOVER_TYPES.includes(t)));
|
|
1273
|
+
}
|
|
1274
|
+
function parseOg(ogTags) {
|
|
1275
|
+
if (!ogTags)
|
|
1276
|
+
return {};
|
|
1277
|
+
try {
|
|
1278
|
+
const o = JSON.parse(ogTags);
|
|
1279
|
+
return o && typeof o === 'object' && !Array.isArray(o) ? o : {};
|
|
1280
|
+
}
|
|
1281
|
+
catch {
|
|
1282
|
+
return {};
|
|
1283
|
+
}
|
|
1284
|
+
}
|
|
1285
|
+
function isDiscoverEligible(jsonLd, ogTags) {
|
|
1286
|
+
if (discoverArticleNodes(jsonLd).length)
|
|
1287
|
+
return true;
|
|
1288
|
+
if (parseJsonLdNodes(jsonLd).some(n => nodeTypeList(n).length))
|
|
1289
|
+
return false;
|
|
1290
|
+
return (parseOg(ogTags)['og:type'] ?? '').toLowerCase() === 'article';
|
|
1291
|
+
}
|
|
1292
|
+
function hasMaxImagePreviewLarge(robots, xrt) {
|
|
1293
|
+
return `${robots ?? ''} ${xrt ?? ''}`.toLowerCase().replace(/\s+/g, '').includes('max-image-preview:large');
|
|
1294
|
+
}
|
|
1295
|
+
function hasPreviewRestriction(robots, xrt) {
|
|
1296
|
+
const s = `${robots ?? ''} ${xrt ?? ''}`.toLowerCase().replace(/\s+/g, '');
|
|
1297
|
+
return s.includes('max-image-preview:none') || s.includes('max-image-preview:standard') || s.includes('nosnippet');
|
|
1298
|
+
}
|
|
1299
|
+
function hasSchemaImage(node) {
|
|
1300
|
+
const img = node?.image;
|
|
1301
|
+
const one = (v) => typeof v === 'string' ? v.trim().length > 0
|
|
1302
|
+
: !!v && typeof v === 'object' && !Array.isArray(v) && !!(v.url || v.contentUrl || v['@id']);
|
|
1303
|
+
return Array.isArray(img) ? img.some(one) : one(img);
|
|
1304
|
+
}
|
|
1411
1305
|
const sitemapHasRows = (c) => (c.db.prepare('SELECT COUNT(*) n FROM sitemap_urls').get().n) > 0;
|
|
1412
1306
|
let _hreflangCache = null;
|
|
1413
1307
|
function hreflangFindings(ctx) {
|
|
@@ -1419,7 +1313,6 @@ function hreflangFindings(ctx) {
|
|
|
1419
1313
|
const status = new Map();
|
|
1420
1314
|
for (const p of ctx.db.prepare(`SELECT url_key, status_code, noindex FROM pages`).all())
|
|
1421
1315
|
status.set(p.url_key, { status: p.status_code, noindex: p.noindex });
|
|
1422
|
-
// declared[sourceKey] = set of internal alternate targetKeys (excluding self)
|
|
1423
1316
|
const declared = new Map();
|
|
1424
1317
|
const parsed = [];
|
|
1425
1318
|
for (const p of pageRows) {
|
|
@@ -1442,13 +1335,10 @@ function hreflangFindings(ctx) {
|
|
|
1442
1335
|
continue;
|
|
1443
1336
|
}
|
|
1444
1337
|
if (key === p.url_key)
|
|
1445
|
-
continue;
|
|
1338
|
+
continue;
|
|
1446
1339
|
targets.push({ key, lang: (lang || '').toLowerCase() });
|
|
1447
1340
|
}
|
|
1448
1341
|
parsed.push({ srcKey: p.url_key, targets });
|
|
1449
|
-
// A return link via x-default is a VALID return tag (common when x-default is the
|
|
1450
|
-
// homepage) — include all targets here; x-default is only excluded as a reciprocation
|
|
1451
|
-
// *requirement* in the loop below, never as a way of satisfying one.
|
|
1452
1342
|
declared.set(p.url_key, new Set(targets.map(t => t.key)));
|
|
1453
1343
|
}
|
|
1454
1344
|
const broken = [];
|
|
@@ -1460,8 +1350,6 @@ function hreflangFindings(ctx) {
|
|
|
1460
1350
|
const st = status.get(t.key);
|
|
1461
1351
|
if (st && (st.status !== 200 || st.noindex === 1))
|
|
1462
1352
|
brokenTargets.push(t.key);
|
|
1463
|
-
// reciprocation only for internal, live (200) targets we crawled, excluding x-default
|
|
1464
|
-
// (a broken target is already reported by broken-hreflang-target — don't double-flag)
|
|
1465
1353
|
if (t.lang !== 'x-default' && st && st.status === 200 && st.noindex !== 1 && !(declared.get(t.key)?.has(srcKey)))
|
|
1466
1354
|
missingReturn.push(t.key);
|
|
1467
1355
|
}
|