@houtini/seo-audit-console 0.9.0 → 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -13
- package/dist/audit/checks.d.ts +0 -7
- package/dist/audit/checks.d.ts.map +1 -1
- package/dist/audit/checks.js +22 -282
- package/dist/audit/checks.js.map +1 -1
- package/dist/audit/drift.d.ts +0 -12
- package/dist/audit/drift.d.ts.map +1 -1
- package/dist/audit/drift.js +0 -15
- package/dist/audit/drift.js.map +1 -1
- package/dist/audit/engine.d.ts +0 -3
- package/dist/audit/engine.d.ts.map +1 -1
- package/dist/audit/engine.js +7 -42
- package/dist/audit/engine.js.map +1 -1
- package/dist/audit/keywordList.d.ts +0 -14
- package/dist/audit/keywordList.d.ts.map +1 -1
- package/dist/audit/keywordList.js +0 -5
- package/dist/audit/keywordList.js.map +1 -1
- package/dist/audit/opportunities.js +3 -26
- package/dist/audit/opportunities.js.map +1 -1
- package/dist/audit/recon.d.ts +0 -31
- package/dist/audit/recon.d.ts.map +1 -1
- package/dist/audit/recon.js +0 -27
- package/dist/audit/recon.js.map +1 -1
- package/dist/audit/report.d.ts +0 -4
- package/dist/audit/report.d.ts.map +1 -1
- package/dist/audit/report.js +0 -7
- package/dist/audit/report.js.map +1 -1
- package/dist/audit/schema-validate.d.ts +0 -26
- package/dist/audit/schema-validate.d.ts.map +1 -1
- package/dist/audit/schema-validate.js +8 -64
- package/dist/audit/schema-validate.js.map +1 -1
- package/dist/audit/templates.d.ts +0 -24
- package/dist/audit/templates.d.ts.map +1 -1
- package/dist/audit/templates.js +1 -21
- package/dist/audit/templates.js.map +1 -1
- package/dist/audit/topicGaps.d.ts +0 -1
- package/dist/audit/topicGaps.d.ts.map +1 -1
- package/dist/audit/topicGaps.js +3 -24
- package/dist/audit/topicGaps.js.map +1 -1
- package/dist/core/AuditDatabase.d.ts +0 -14
- package/dist/core/AuditDatabase.d.ts.map +1 -1
- package/dist/core/AuditDatabase.js +8 -61
- package/dist/core/AuditDatabase.js.map +1 -1
- package/dist/core/Backlinks.d.ts +0 -7
- package/dist/core/Backlinks.d.ts.map +1 -1
- package/dist/core/Backlinks.js +1 -23
- package/dist/core/Backlinks.js.map +1 -1
- package/dist/core/Crawler.d.ts +0 -1
- package/dist/core/Crawler.d.ts.map +1 -1
- package/dist/core/Crawler.js +24 -119
- package/dist/core/Crawler.js.map +1 -1
- package/dist/core/DataForSeoClient.d.ts +0 -74
- package/dist/core/DataForSeoClient.d.ts.map +1 -1
- package/dist/core/DataForSeoClient.js +1 -72
- package/dist/core/DataForSeoClient.js.map +1 -1
- package/dist/core/Entities.d.ts +0 -6
- package/dist/core/Entities.d.ts.map +1 -1
- package/dist/core/Entities.js +0 -8
- package/dist/core/Entities.js.map +1 -1
- package/dist/core/FirecrawlClient.d.ts +0 -18
- package/dist/core/FirecrawlClient.d.ts.map +1 -1
- package/dist/core/FirecrawlClient.js +1 -13
- package/dist/core/FirecrawlClient.js.map +1 -1
- package/dist/core/GscClient.d.ts +0 -7
- package/dist/core/GscClient.d.ts.map +1 -1
- package/dist/core/GscClient.js +1 -10
- package/dist/core/GscClient.js.map +1 -1
- package/dist/core/GscSync.d.ts +0 -4
- package/dist/core/GscSync.d.ts.map +1 -1
- package/dist/core/GscSync.js +2 -33
- package/dist/core/GscSync.js.map +1 -1
- package/dist/core/JobManager.d.ts +0 -5
- package/dist/core/JobManager.d.ts.map +1 -1
- package/dist/core/JobManager.js +0 -8
- package/dist/core/JobManager.js.map +1 -1
- package/dist/core/LinkIntersect.d.ts +0 -1
- package/dist/core/LinkIntersect.d.ts.map +1 -1
- package/dist/core/LinkIntersect.js +1 -37
- package/dist/core/LinkIntersect.js.map +1 -1
- package/dist/core/MajesticClient.d.ts +0 -26
- package/dist/core/MajesticClient.d.ts.map +1 -1
- package/dist/core/MajesticClient.js +0 -15
- package/dist/core/MajesticClient.js.map +1 -1
- package/dist/core/RankTracker.d.ts +0 -4
- package/dist/core/RankTracker.d.ts.map +1 -1
- package/dist/core/RankTracker.js +0 -9
- package/dist/core/RankTracker.js.map +1 -1
- package/dist/core/Refresh.d.ts +0 -6
- package/dist/core/Refresh.d.ts.map +1 -1
- package/dist/core/Refresh.js +0 -6
- package/dist/core/Refresh.js.map +1 -1
- package/dist/core/SupadataClient.d.ts +0 -11
- package/dist/core/SupadataClient.d.ts.map +1 -1
- package/dist/core/SupadataClient.js +0 -5
- package/dist/core/SupadataClient.js.map +1 -1
- package/dist/core/UrlInspector.d.ts +0 -5
- package/dist/core/UrlInspector.d.ts.map +1 -1
- package/dist/core/UrlInspector.js +0 -6
- package/dist/core/UrlInspector.js.map +1 -1
- package/dist/core/WikidataClient.d.ts +0 -2
- package/dist/core/WikidataClient.d.ts.map +1 -1
- package/dist/core/WikidataClient.js +0 -7
- package/dist/core/WikidataClient.js.map +1 -1
- package/dist/core/agentReadiness.d.ts +0 -9
- package/dist/core/agentReadiness.d.ts.map +1 -1
- package/dist/core/agentReadiness.js +0 -13
- package/dist/core/agentReadiness.js.map +1 -1
- package/dist/core/ctrModel.js +0 -3
- package/dist/core/ctrModel.js.map +1 -1
- package/dist/core/dashboardData.d.ts +0 -1
- package/dist/core/dashboardData.d.ts.map +1 -1
- package/dist/core/dashboardData.js +13 -95
- package/dist/core/dashboardData.js.map +1 -1
- package/dist/core/dataStorage.d.ts +0 -2
- package/dist/core/dataStorage.d.ts.map +1 -1
- package/dist/core/dataStorage.js +6 -19
- package/dist/core/dataStorage.js.map +1 -1
- package/dist/core/draftBrief.d.ts +0 -7
- package/dist/core/draftBrief.d.ts.map +1 -1
- package/dist/core/draftBrief.js +0 -10
- package/dist/core/draftBrief.js.map +1 -1
- package/dist/core/extract.d.ts +0 -11
- package/dist/core/extract.d.ts.map +1 -1
- package/dist/core/extract.js +1 -38
- package/dist/core/extract.js.map +1 -1
- package/dist/core/googleNews.d.ts +0 -11
- package/dist/core/googleNews.d.ts.map +1 -1
- package/dist/core/googleNews.js +0 -7
- package/dist/core/googleNews.js.map +1 -1
- package/dist/core/gscFreshness.d.ts +0 -10
- package/dist/core/gscFreshness.d.ts.map +1 -1
- package/dist/core/gscFreshness.js +0 -11
- package/dist/core/gscFreshness.js.map +1 -1
- package/dist/core/linkGraph.d.ts +0 -15
- package/dist/core/linkGraph.d.ts.map +1 -1
- package/dist/core/linkGraph.js +1 -23
- package/dist/core/linkGraph.js.map +1 -1
- package/dist/core/marketSizing.d.ts +0 -13
- package/dist/core/marketSizing.d.ts.map +1 -1
- package/dist/core/marketSizing.js +1 -1
- package/dist/core/marketSizing.js.map +1 -1
- package/dist/core/passageScore.d.ts +0 -13
- package/dist/core/passageScore.d.ts.map +1 -1
- package/dist/core/passageScore.js +1 -16
- package/dist/core/passageScore.js.map +1 -1
- package/dist/core/paths.d.ts +0 -3
- package/dist/core/paths.d.ts.map +1 -1
- package/dist/core/paths.js +0 -0
- package/dist/core/paths.js.map +1 -1
- package/dist/core/queryData.d.ts +0 -1
- package/dist/core/queryData.d.ts.map +1 -1
- package/dist/core/queryData.js +0 -18
- package/dist/core/queryData.js.map +1 -1
- package/dist/core/reconFetch.d.ts +0 -6
- package/dist/core/reconFetch.d.ts.map +1 -1
- package/dist/core/reconFetch.js +1 -4
- package/dist/core/reconFetch.js.map +1 -1
- package/dist/core/reconResearch.d.ts +0 -20
- package/dist/core/reconResearch.d.ts.map +1 -1
- package/dist/core/reconResearch.js +1 -16
- package/dist/core/reconResearch.js.map +1 -1
- package/dist/core/reranker.d.ts +0 -2
- package/dist/core/reranker.d.ts.map +1 -1
- package/dist/core/reranker.js +1 -11
- package/dist/core/reranker.js.map +1 -1
- package/dist/core/robots.d.ts +0 -6
- package/dist/core/robots.d.ts.map +1 -1
- package/dist/core/robots.js +1 -7
- package/dist/core/robots.js.map +1 -1
- package/dist/core/serpFootprint.d.ts +0 -12
- package/dist/core/serpFootprint.d.ts.map +1 -1
- package/dist/core/serpFootprint.js +0 -1
- package/dist/core/serpFootprint.js.map +1 -1
- package/dist/core/serpRecon.d.ts +0 -14
- package/dist/core/serpRecon.d.ts.map +1 -1
- package/dist/core/serpRecon.js +1 -13
- package/dist/core/serpRecon.js.map +1 -1
- package/dist/core/sitemap.js +3 -16
- package/dist/core/sitemap.js.map +1 -1
- package/dist/core/sql.d.ts +0 -4
- package/dist/core/sql.d.ts.map +1 -1
- package/dist/core/sql.js +0 -4
- package/dist/core/sql.js.map +1 -1
- package/dist/core/url-key.d.ts +0 -31
- package/dist/core/url-key.d.ts.map +1 -1
- package/dist/core/url-key.js +1 -38
- package/dist/core/url-key.js.map +1 -1
- package/dist/core/webHandlers.d.ts +0 -6
- package/dist/core/webHandlers.d.ts.map +1 -1
- package/dist/core/webHandlers.js +2 -5
- package/dist/core/webHandlers.js.map +1 -1
- package/dist/core/webServer.d.ts +0 -16
- package/dist/core/webServer.d.ts.map +1 -1
- package/dist/core/webServer.js +3 -17
- package/dist/core/webServer.js.map +1 -1
- package/dist/dashboard.js +0 -23
- package/dist/dashboard.js.map +1 -1
- package/dist/generators/index.d.ts +0 -7
- package/dist/generators/index.d.ts.map +1 -1
- package/dist/generators/index.js +1 -26
- package/dist/generators/index.js.map +1 -1
- package/dist/index.js +0 -2
- package/dist/index.js.map +1 -1
- package/dist/server.js +16 -166
- package/dist/server.js.map +1 -1
- package/package.json +1 -1
- package/server.json +2 -2
package/dist/audit/checks.js
CHANGED
|
@@ -5,33 +5,17 @@ import { expectedCtr } from '../core/ctrModel.js';
|
|
|
5
5
|
import { latestTwoCrawls } from './drift.js';
|
|
6
6
|
import { HTML_CT } from '../core/sql.js';
|
|
7
7
|
const rows = (ctx, sql, ...args) => ctx.db.prepare(sql).all(...args);
|
|
8
|
-
// Streaming variant — yields one row at a time instead of materialising the whole result set. Use for
|
|
9
|
-
// checks that scan a large/fat column (e.g. body_chunks) and only need independent per-row work, so a
|
|
10
|
-
// big content site doesn't load every page's body text into one array.
|
|
11
8
|
const iterRows = (ctx, sql, ...args) => ctx.db.prepare(sql).iterate(...args);
|
|
12
|
-
// d is the finalised GSC date (partial trailing days already trimmed upstream); cap the window at
|
|
13
|
-
// it so checks never count unfinalised days. winPrev already caps below d, so it's unaffected.
|
|
14
9
|
const win = (d) => `date > date('${d}', '-28 days') AND date <= '${d}'`;
|
|
15
|
-
// Prior 28-day window (the 28 days BEFORE the current window) — for period-over-period checks.
|
|
16
10
|
const winPrev = (d) => `date <= date('${d}', '-28 days') AND date > date('${d}', '-56 days')`;
|
|
17
|
-
// SQL clause to drop branded queries (whole-token match). Branded multi-URL ranking is sitelinks,
|
|
18
|
-
// not cannibalisation; branded top-queries aren't anchor targets. The brand is re-sanitised to
|
|
19
|
-
// alnum HERE (not trusting the caller) so inlining it into SQL is unconditionally injection-safe.
|
|
20
11
|
const brandExcl = (c) => {
|
|
21
12
|
const b = (c.brand ?? '').replace(/[^a-z0-9]/g, '');
|
|
22
13
|
return b ? `AND (' ' || LOWER(query) || ' ') NOT LIKE '% ${b} %'` : '';
|
|
23
14
|
};
|
|
24
|
-
// Days of GSC history actually held — period-over-period checks need enough span to be meaningful.
|
|
25
15
|
const spanDays = (c) => {
|
|
26
|
-
// Span must be measured up to the FINALISED max date the windows actually key off
|
|
27
|
-
// (gscMaxDate is trimmed ~3 days below raw MAX(date)) — measuring the raw span lets the
|
|
28
|
-
// ≥56-day guard pass while the previous-28d window still reaches before MIN(date),
|
|
29
|
-
// undercounting the prior period (rising-pages FPs, traffic-decay FNs).
|
|
30
16
|
const r = c.db.prepare(`SELECT julianday(?) - julianday(MIN(date)) d FROM search_analytics`).get(c.gscMaxDate ?? null);
|
|
31
17
|
return r?.d ?? 0;
|
|
32
18
|
};
|
|
33
|
-
// Newest dateModified/datePublished anywhere in a page's JSON-LD (recurses @graph/arrays) — the
|
|
34
|
-
// effective "last meaningfully updated" date. json_ld is a JSON array of raw block strings.
|
|
35
19
|
const newestSchemaDate = (jl) => {
|
|
36
20
|
if (!jl)
|
|
37
21
|
return null;
|
|
@@ -58,27 +42,15 @@ const newestSchemaDate = (jl) => {
|
|
|
58
42
|
try {
|
|
59
43
|
scan(JSON.parse(block));
|
|
60
44
|
}
|
|
61
|
-
catch {
|
|
45
|
+
catch { }
|
|
62
46
|
}
|
|
63
47
|
}
|
|
64
|
-
catch {
|
|
48
|
+
catch { }
|
|
65
49
|
return bestStr;
|
|
66
50
|
};
|
|
67
|
-
|
|
68
|
-
// page 1 and are intentionally absent from sitemaps, so they must NOT generate duplicate-title /
|
|
69
|
-
// duplicate-meta / missing-meta / not-in-sitemap false positives. Pass the column reference
|
|
70
|
-
// (e.g. 'url_key' or 'p.url_key') so it composes with table aliases.
|
|
71
|
-
const notPagination = (col = 'url_key') =>
|
|
72
|
-
// Anchor the param name to ?/& — a bare LIKE '%page=%' also matches per_page=/on_page=/
|
|
73
|
-
// homepage=, wrongly exempting those URLs from duplicate-title/meta/sitemap checks.
|
|
74
|
-
`${col} NOT GLOB '*/page/[0-9]*' AND ${col} NOT GLOB '*[?&]page=[0-9]*' AND ${col} NOT GLOB '*[?&]paged=[0-9]*'`;
|
|
75
|
-
// Significant query terms (drop stopwords; keep ≥2 chars so "vr"/"pc"/"ai" count).
|
|
51
|
+
const notPagination = (col = 'url_key') => `${col} NOT GLOB '*/page/[0-9]*' AND ${col} NOT GLOB '*[?&]page=[0-9]*' AND ${col} NOT GLOB '*[?&]paged=[0-9]*'`;
|
|
76
52
|
const STOP = new Set(['the', 'a', 'an', 'and', 'or', 'of', 'for', 'to', 'in', 'on', 'with', 'your', 'you', 'is', 'are', 'best', 'how', 'what', 'vs', 'why', 'can']);
|
|
77
53
|
const terms = (s) => (s || '').toLowerCase().replace(/[^a-z0-9 ]+/g, ' ').split(/\s+/).filter(t => t.length >= 2 && !STOP.has(t));
|
|
78
|
-
// A query term is "present" in the title if a TITLE WORD matches it (exact, or a
|
|
79
|
-
// plural/stem prefix either way for ≥4-char tokens), or the space-collapsed title
|
|
80
|
-
// contains it (for multi-word brands like "sync mesh" ≈ "syncmesh", ≥5 chars).
|
|
81
|
-
// Word-level avoids substring false-matches (e.g. "art" inside "smart").
|
|
82
54
|
const titleHasTerm = (title, term) => {
|
|
83
55
|
const t = title.toLowerCase();
|
|
84
56
|
const words = t.split(/[^a-z0-9]+/).filter(Boolean);
|
|
@@ -86,7 +58,6 @@ const titleHasTerm = (title, term) => {
|
|
|
86
58
|
return true;
|
|
87
59
|
return term.length >= 5 && t.replace(/[^a-z0-9]+/g, '').includes(term);
|
|
88
60
|
};
|
|
89
|
-
// Validate captured JSON-LD per page, keeping findings whose issue-kinds match `kinds`.
|
|
90
61
|
function schemaFindings(ctx, kinds) {
|
|
91
62
|
const set = new Set(kinds);
|
|
92
63
|
const out = [];
|
|
@@ -97,11 +68,8 @@ function schemaFindings(ctx, kinds) {
|
|
|
97
68
|
}
|
|
98
69
|
return out;
|
|
99
70
|
}
|
|
100
|
-
// Position→expected-CTR curve lives in core/ctrModel (shared with the dashboard). Re-export it so
|
|
101
|
-
// existing `import { expectedCtr } from './checks.js'` call sites keep working.
|
|
102
71
|
export { expectedCtr };
|
|
103
72
|
export const CHECKS = [
|
|
104
|
-
// ── On-page (crawl, deterministic) ──────────────────────────────────────
|
|
105
73
|
{
|
|
106
74
|
id: 'missing-title', category: 'onpage', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
107
75
|
title: 'Missing title tag', fix: 'Add a unique, descriptive <title> (~50–60 chars).',
|
|
@@ -132,19 +100,11 @@ export const CHECKS = [
|
|
|
132
100
|
title: 'Thin content', fix: 'Expand or consolidate — under ~200 words of body text.',
|
|
133
101
|
run: (c) => rows(c, `SELECT url_key urlKey, word_count FROM pages WHERE status_code=200 AND indexable=1 AND word_count < 200`).map(r => ({ urlKey: r.urlKey, evidence: { wordCount: r.word_count } })),
|
|
134
102
|
},
|
|
135
|
-
// ── Indexation / crawlability ───────────────────────────────────────────
|
|
136
103
|
{
|
|
137
|
-
// RETIRED: 'canonical-mismatch' (was HIGH). A 200 page whose canonical points to a *healthy*
|
|
138
|
-
// 200 indexable URL is intentional consolidation (slug variants, category merges) — normal SEO,
|
|
139
|
-
// not an issue, yet it fired HIGH on every such page (pure noise). Every actionable case is
|
|
140
|
-
// already covered by a higher-signal check: broken-canonical-target (unhealthy target),
|
|
141
|
-
// canonical-ignored (Google ranks the non-canonical page), canonical-conflict (GSC disagrees).
|
|
142
|
-
// So plain "canonical points elsewhere" has no high-confidence residual — removed.
|
|
143
104
|
id: 'broken-internal-links', category: 'crawlability', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
|
|
144
105
|
title: 'Internal links to 4xx/5xx', fix: 'Repoint internal links to a live, canonical URL.',
|
|
145
106
|
run: (c) => rows(c, `SELECT l.target_key urlKey, p.status_code status, COUNT(DISTINCT l.source_key) sources FROM links l JOIN pages p ON p.url_key=l.target_key WHERE l.is_internal=1 AND p.status_code >= 400 AND p.status_code NOT IN (429,503) GROUP BY l.target_key`).map(r => ({ urlKey: r.urlKey, evidence: { status: r.status, linkingPages: r.sources } })),
|
|
146
107
|
},
|
|
147
|
-
// ── Extractor-dependent (images + canonical shape) ──────────────────────
|
|
148
108
|
{
|
|
149
109
|
id: 'image-alt', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
150
110
|
title: 'Images missing alt text', fix: 'Add descriptive alt text to content images (alt="" only for decorative).',
|
|
@@ -160,13 +120,8 @@ export const CHECKS = [
|
|
|
160
120
|
title: 'Multiple canonical tags', fix: 'Keep exactly one rel=canonical — conflicting canonicals let Google pick (or ignore) one.',
|
|
161
121
|
run: (c) => rows(c, `SELECT url_key urlKey, canonical_count cnt FROM pages WHERE status_code=200 AND canonical_count > 1`).map(r => ({ urlKey: r.urlKey, evidence: { canonicalCount: r.cnt } })),
|
|
162
122
|
},
|
|
163
|
-
// ── Extractor additions (CLS, headings, mixed content, directives, social) ───
|
|
164
123
|
{
|
|
165
124
|
id: 'images-missing-dimensions', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
166
|
-
// Lint, NOT a measured Core Web Vital. Missing width/height attributes only cause CLS if the
|
|
167
|
-
// CSS doesn't already reserve space — modern themes using aspect-ratio / fixed boxes have ~0
|
|
168
|
-
// measured CLS despite missing attributes. So we report it as a hygiene lint and explicitly say
|
|
169
|
-
// it's not a confirmed CWV issue; escalate only against field CLS (high-yield-cwv-fail does that).
|
|
170
125
|
title: 'Images missing width/height attributes', fix: 'Add width & height (or rely on CSS aspect-ratio) so the browser reserves space. NOTE: this is a lint — if your CSS already reserves space (aspect-ratio / fixed box) measured CLS is likely ~0 and there is nothing to fix. Confirm with field CLS before prioritising.',
|
|
171
126
|
run: (c) => rows(c, `SELECT url_key urlKey, images_missing_dimensions n, image_count total FROM pages WHERE status_code=200 AND indexable=1 AND images_missing_dimensions > 0`).map(r => ({ urlKey: r.urlKey, evidence: { missingDimensions: r.n, total: r.total, note: 'lint only — no measured CLS impact unless field data shows layout shift' } })),
|
|
172
127
|
},
|
|
@@ -190,13 +145,11 @@ export const CHECKS = [
|
|
|
190
145
|
title: 'No social share tags', fix: 'Add Open Graph (og:title/og:image) and/or Twitter Card tags so shared links render a rich preview.',
|
|
191
146
|
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND (og_tags IS NULL OR og_tags='') AND (twitter_tags IS NULL OR twitter_tags='')`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
|
|
192
147
|
},
|
|
193
|
-
// ── Security / war-stories (headers now captured) ───────────────────────
|
|
194
148
|
{
|
|
195
149
|
id: 'missing-hsts', category: 'security', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
196
150
|
title: 'Missing HSTS header', fix: 'Add Strict-Transport-Security with a sensible max-age.',
|
|
197
151
|
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (security_headers IS NULL OR security_headers NOT LIKE '%hsts%') LIMIT 1`).map(r => ({ urlKey: r.urlKey, evidence: { note: 'representative page; HSTS is site-wide' } })),
|
|
198
152
|
},
|
|
199
|
-
// ── Merged GSC × crawl (the differentiator) ─────────────────────────────
|
|
200
153
|
{
|
|
201
154
|
id: 'noindex-with-traffic', category: 'indexation', severity: 'crit', labels: ['D', 'G'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
202
155
|
title: 'Noindex page still getting clicks', fix: 'Remove noindex if the page should rank — it earns clicks.',
|
|
@@ -245,7 +198,6 @@ export const CHECKS = [
|
|
|
245
198
|
title: 'Striking-distance query (page 2)', fix: 'Small on-page + internal-link push could reach page 1.',
|
|
246
199
|
run: (c) => c.gscMaxDate ? rows(c, `SELECT page_key urlKey, query, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) position, SUM(impressions) impressions FROM search_analytics WHERE query IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY query, page_key HAVING SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0)>10 AND SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0)<=20 AND SUM(impressions)>=20 ORDER BY impressions DESC LIMIT 50`).map(r => ({ urlKey: r.urlKey, evidence: { query: r.query, position: Math.round(r.position * 10) / 10, impressions: r.impressions } })) : [],
|
|
247
200
|
},
|
|
248
|
-
// ── Additions from industry checklist review (buildable on current data) ──
|
|
249
201
|
{
|
|
250
202
|
id: 'title-too-long', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
251
203
|
title: 'Title over ~60 chars', fix: 'Trim the title so the primary keyword sits within ~60 chars.',
|
|
@@ -264,8 +216,6 @@ export const CHECKS = [
|
|
|
264
216
|
{
|
|
265
217
|
id: 'redirect-chain', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
|
|
266
218
|
title: 'Redirect chain (2+ hops)', fix: 'Collapse to a single hop to the final URL.',
|
|
267
|
-
// Count hops in SQL (json_array_length, guarded by json_valid) and filter to >=2 there — so we
|
|
268
|
-
// never pull every redirect-bearing page into JS just to count + drop most of them.
|
|
269
219
|
run: (c) => rows(c, `SELECT url_key urlKey, json_array_length(redirects) hops FROM pages
|
|
270
220
|
WHERE redirects IS NOT NULL AND json_valid(redirects) AND json_array_length(redirects) >= 2`)
|
|
271
221
|
.map(r => ({ urlKey: r.urlKey, evidence: { hops: r.hops } })),
|
|
@@ -273,13 +223,6 @@ export const CHECKS = [
|
|
|
273
223
|
{
|
|
274
224
|
id: 'internal-links-to-redirects', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
|
|
275
225
|
title: 'Internal links pointing through redirects', fix: 'Repoint internal links to the final URL (saves crawl + equity).',
|
|
276
|
-
// Flag a link ONLY if the raw href it uses is itself a redirect source — i.e. that exact URL
|
|
277
|
-
// appears as a `from` hop in some redirect chain. The old query flagged any link whose target
|
|
278
|
-
// page merely HAD a `redirects` entry, which fired site-wide on www-canonical properties: every
|
|
279
|
-
// page records the seed apex→www hop, yet the actual hrefs already use the final (www) URL and
|
|
280
|
-
// never redirect. Matching the href (fragment-stripped) against the real redirect-source set is
|
|
281
|
-
// robust to that artefact. Kept entirely in SQL (json_each over pages.redirects) so we never
|
|
282
|
-
// materialise the whole links table in JS — the links table is the largest in the DB.
|
|
283
226
|
run: (c) => {
|
|
284
227
|
return rows(c, `
|
|
285
228
|
WITH redir_src AS (
|
|
@@ -302,11 +245,8 @@ export const CHECKS = [
|
|
|
302
245
|
{
|
|
303
246
|
id: 'missing-structured-data', category: 'schema', severity: 'low', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
304
247
|
title: 'No structured data', fix: 'Add relevant JSON-LD (Article, Product, Organization…).',
|
|
305
|
-
// Only "no structured data" if there's no JSON-LD AND no Microdata/RDFa either — else a
|
|
306
|
-
// page using valid Microdata (common on older themes) is falsely flagged.
|
|
307
248
|
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND indexable=1 AND (json_ld IS NULL OR json_ld='') AND COALESCE(has_microdata,0)=0 AND COALESCE(has_rdfa,0)=0`).map(r => ({ urlKey: r.urlKey, evidence: {} })),
|
|
308
249
|
},
|
|
309
|
-
// ── Schema validation (validate captured json_ld vs maintained Rich-Results map) ──
|
|
310
250
|
{
|
|
311
251
|
id: 'invalid-schema', category: 'schema', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
312
252
|
title: 'Invalid structured data', fix: 'Fix the JSON-LD so each block parses and carries @context (https://schema.org) + a valid @type.',
|
|
@@ -335,15 +275,6 @@ export const CHECKS = [
|
|
|
335
275
|
{
|
|
336
276
|
id: 'keyword-cannibalisation', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
337
277
|
title: 'Keyword cannibalisation', fix: 'Consolidate or differentiate — multiple URLs compete for one query.',
|
|
338
|
-
// A URL only "competes" if it ranks for the query (impression-weighted pos < 20) AND holds a
|
|
339
|
-
// non-trivial share of the leader's impressions — ≥10% of the leader OR ≥500 impressions in its
|
|
340
|
-
// own right, and ≥10 impressions minimum. Without that floor, incidental long-tail appearances
|
|
341
|
-
// (a page picking up 1–7 impressions for the query) counted as competitors: e.g. "best vr headset"
|
|
342
|
-
// reported 10 URLs when ONE page held 59,570 impressions at pos 1.1 and the other nine had 1–7
|
|
343
|
-
// each — not cannibalisation, Google had decided. The absolute ≥500 backstop keeps a genuine
|
|
344
|
-
// mid-volume rival under a dominant leader (e.g. 60k leader + a real 4k second page = 6.7%, below
|
|
345
|
-
// the 10% bar) from being silently dropped. We also exclude "dominance" where the best pages both
|
|
346
|
-
// sit at pos 1–2 (indented/double results are good). Branded queries are dropped via brandExcl.
|
|
347
278
|
run: (c) => {
|
|
348
279
|
if (!c.gscMaxDate)
|
|
349
280
|
return [];
|
|
@@ -368,9 +299,6 @@ export const CHECKS = [
|
|
|
368
299
|
const keys = String(r.pk || '').split('\x1f'), imps = String(r.im || '').split('\x1f'), poss = String(r.ps || '').split('\x1f');
|
|
369
300
|
const items = keys.map((u, i) => ({ url: u, impressions: Number(imps[i]) || 0, position: Number(poss[i]) || 0, title: titleOf.get(u)?.title ?? null }))
|
|
370
301
|
.sort((a, b) => b.impressions - a.impressions);
|
|
371
|
-
// Differentiation signal: do the top-2 competing pages share significant title terms beyond the
|
|
372
|
-
// query itself? If not (and both are titled), they're likely intentionally distinct pages (e.g.
|
|
373
|
-
// pairwise comparisons), not true duplicates competing for one intent — annotate, don't suppress.
|
|
374
302
|
const q = new Set(terms(r.query));
|
|
375
303
|
const sig = items.slice(0, 2).map(it => terms(it.title ?? '').filter((w) => !q.has(w)));
|
|
376
304
|
const differentiated = sig.length === 2 && items[0].title != null && items[1].title != null
|
|
@@ -389,22 +317,16 @@ export const CHECKS = [
|
|
|
389
317
|
{
|
|
390
318
|
id: 'ctr-below-expected', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
391
319
|
title: 'CTR far below position-expected', fix: 'Rewrite title/meta — ranking well but under-clicked (snippet opportunity).',
|
|
392
|
-
// High-confidence floor: ≥500 impressions/28d. A title/meta rewrite (HIGH, ~3h) is only
|
|
393
|
-
// worth flagging where the snippet earns enough visibility for a CTR lift to pay back — a
|
|
394
|
-
// 100-impression page at 1% vs 3% expected is a 2-clicks gap, not a HIGH issue.
|
|
395
320
|
run: (c) => c.gscMaxDate ? rows(c, `SELECT page_key urlKey, SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) position, SUM(clicks) clicks, SUM(impressions) impressions FROM search_analytics WHERE page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key HAVING SUM(position*impressions)*1.0/NULLIF(SUM(impressions),0) <= 10 AND SUM(impressions) >= 500`)
|
|
396
321
|
.map(r => { const ctr = r.clicks / r.impressions; const exp = expectedCtr(r.position); return { urlKey: r.urlKey, ctr, exp, position: r.position, impressions: r.impressions }; })
|
|
397
322
|
.filter(x => x.ctr < x.exp * 0.5)
|
|
398
323
|
.map(x => {
|
|
399
324
|
const ev = { position: Math.round(x.position * 10) / 10, ctr: Math.round(x.ctr * 1000) / 10 + '%', expectedCtr: Math.round(x.exp * 1000) / 10 + '%', impressions: x.impressions };
|
|
400
|
-
// Extreme case: near-zero CTR at a strong position isn't a title problem — a SERP feature
|
|
401
|
-
// (image/video/AI overview) or navigational intent is taking the clicks. Different fix.
|
|
402
325
|
if (x.position <= 5 && x.ctr < x.exp * 0.15)
|
|
403
326
|
ev.note = 'near-zero CTR for the position — likely a SERP feature or navigational intent taking the clicks; check the live SERP before rewriting the title/meta';
|
|
404
327
|
return { urlKey: x.urlKey, evidence: ev };
|
|
405
328
|
}) : [],
|
|
406
329
|
},
|
|
407
|
-
// ── Period-over-period (GSC history by date) — the trend questions SEOs live in ──
|
|
408
330
|
{
|
|
409
331
|
id: 'traffic-decay', category: 'merged', severity: 'high', labels: ['G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
410
332
|
title: 'Page losing clicks (period-over-period)', fix: 'Refresh and expand the content, and check for lost rankings — this page’s Search Console clicks fell sharply against the previous 28 days.',
|
|
@@ -416,7 +338,6 @@ export const CHECKS = [
|
|
|
416
338
|
FROM prev LEFT JOIN cur ON cur.page_key=prev.page_key
|
|
417
339
|
WHERE prev.c >= 30 AND COALESCE(cur.c,0) < prev.c * 0.6
|
|
418
340
|
ORDER BY (prev.c - COALESCE(cur.c,0)) DESC LIMIT 40`)
|
|
419
|
-
// page_key IS a url_key — carry it in the joinable column, not just evidence.
|
|
420
341
|
.map(r => ({ urlKey: r.url, evidence: { url: r.url, previousClicks: r.prevC, currentClicks: r.curC, clicksLost: r.prevC - r.curC, dropPercent: Math.round((1 - r.curC / r.prevC) * 100) + '%', clicks: r.prevC - r.curC, impressions: r.curI, position: Math.round(r.pos * 10) / 10 } })),
|
|
421
342
|
},
|
|
422
343
|
{
|
|
@@ -447,7 +368,6 @@ export const CHECKS = [
|
|
|
447
368
|
title: 'Indexable page with no search traffic', fix: 'No impressions in 90 days despite being indexable — consolidate, improve, or noindex/prune to concentrate crawl budget and internal authority (confirm it isn’t seasonal or brand-new first).',
|
|
448
369
|
run: (c) => (!c.gscMaxDate || spanDays(c) < 90) ? [] : rows(c, `SELECT url_key urlKey, ipr FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND COALESCE(click_depth, 999) >= 1 AND url_key NOT IN (SELECT DISTINCT page_key FROM search_analytics WHERE page_key IS NOT NULL AND date <= '${c.gscMaxDate}' AND date > date('${c.gscMaxDate}','-90 days') AND impressions > 0) ORDER BY ipr DESC LIMIT 100`).map(r => ({ urlKey: r.urlKey, evidence: { note: 'indexable but zero impressions in 90 days', ipr: Math.round(r.ipr) } })),
|
|
449
370
|
},
|
|
450
|
-
// ── On-page parity (crawl-only, deterministic) ──
|
|
451
371
|
{
|
|
452
372
|
id: 'duplicate-meta-description', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
453
373
|
title: 'Duplicate meta description', fix: 'Give each indexable page a unique meta description.',
|
|
@@ -506,7 +426,6 @@ export const CHECKS = [
|
|
|
506
426
|
.map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, h1: x.h1 } }));
|
|
507
427
|
},
|
|
508
428
|
},
|
|
509
|
-
// ── Content cluster (AI-era / RAG layer): exploit the chunked body text captured at crawl ──
|
|
510
429
|
{
|
|
511
430
|
id: 'body-missing-top-query', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
512
431
|
title: 'Top query missing from the page body', fix: 'The page ranks for this query yet its terms appear nowhere — not the title, H1, or body copy. Add a section that actually covers the topic; if it can’t, the page is too thin to hold the ranking and a stronger page should target it.',
|
|
@@ -527,8 +446,8 @@ export const CHECKS = [
|
|
|
527
446
|
for (const ch of JSON.parse(x.bc))
|
|
528
447
|
body += ` ${ch.heading || ''} ${ch.text || ''}`;
|
|
529
448
|
}
|
|
530
|
-
catch {
|
|
531
|
-
return !q.some((w) => titleHasTerm(body, w));
|
|
449
|
+
catch { }
|
|
450
|
+
return !q.some((w) => titleHasTerm(body, w));
|
|
532
451
|
}).map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, note: 'query terms absent from title, H1 and body' } }));
|
|
533
452
|
},
|
|
534
453
|
},
|
|
@@ -565,7 +484,7 @@ export const CHECKS = [
|
|
|
565
484
|
return r.filter(x => {
|
|
566
485
|
const q = terms(x.query);
|
|
567
486
|
if (q.length < 2)
|
|
568
|
-
return false;
|
|
487
|
+
return false;
|
|
569
488
|
let chunks;
|
|
570
489
|
try {
|
|
571
490
|
chunks = JSON.parse(x.bc);
|
|
@@ -575,8 +494,8 @@ export const CHECKS = [
|
|
|
575
494
|
}
|
|
576
495
|
const whole = `${x.title || ''} ${chunks.map(ch => `${ch.heading || ''} ${ch.text || ''}`).join(' ')}`;
|
|
577
496
|
if (!q.every((w) => titleHasTerm(whole, w)))
|
|
578
|
-
return false;
|
|
579
|
-
return !chunks.some(ch => { const h = `${ch.heading || ''} ${ch.text || ''}`; return q.every((w) => titleHasTerm(h, w)); });
|
|
497
|
+
return false;
|
|
498
|
+
return !chunks.some(ch => { const h = `${ch.heading || ''} ${ch.text || ''}`; return q.every((w) => titleHasTerm(h, w)); });
|
|
580
499
|
}).map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, note: 'terms present but never together in one passage' } }));
|
|
581
500
|
},
|
|
582
501
|
},
|
|
@@ -605,10 +524,6 @@ export const CHECKS = [
|
|
|
605
524
|
},
|
|
606
525
|
},
|
|
607
526
|
{
|
|
608
|
-
// Hobo "Signal Coherence" / Goldmine; leak: anchor_mismatch. Google leans on internal anchors to
|
|
609
|
-
// understand a page's topic — if the IN-CONTENT inbound anchors never mention the query the page
|
|
610
|
-
// actually ranks for, that's an incoherent internal signal. Guard against boilerplate FPs by using
|
|
611
|
-
// ONLY placement='body' anchors (nav/footer/aside excluded) and requiring ≥3 of them.
|
|
612
527
|
id: 'anchor-text-incoherent', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
613
528
|
title: 'Internal anchors don’t mention the page’s top query', fix: 'The in-content internal links pointing at this page never use its top-ranking query in their anchor text — and Google leans on internal anchors to understand what a page is about. Re-anchor the key internal links with descriptive, query-relevant text instead of generic “read more” / brand-only labels.',
|
|
614
529
|
run: (c) => {
|
|
@@ -619,12 +534,6 @@ export const CHECKS = [
|
|
|
619
534
|
FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} ${brandExcl(c)} GROUP BY page_key, query) s
|
|
620
535
|
JOIN pages p ON p.url_key = s.page_key
|
|
621
536
|
WHERE s.rn = 1 AND p.indexable = 1 AND s.impr >= 100`);
|
|
622
|
-
// Pool only GENUINE editorial anchors. "Chrome" (nav/footer/breadcrumb/CTA) is detected
|
|
623
|
-
// STRUCTURALLY, not via an English word-list: an anchor text reused across a large share of
|
|
624
|
-
// the site's pages is templated boilerplate — e.g. the EHI homepage's 2,940 inbound "Home"
|
|
625
|
-
// breadcrumb links — and that holds in any language ("Startseite", "Accueil"…). We also drop
|
|
626
|
-
// self-links and anchors with no letters ("(0)", page numbers, arrows). Editorial in-content
|
|
627
|
-
// anchors recur on only a handful of pages, so they survive.
|
|
628
537
|
const anchors = new Map();
|
|
629
538
|
for (const a of rows(c, `
|
|
630
539
|
WITH body_anchors AS (
|
|
@@ -645,18 +554,13 @@ export const CHECKS = [
|
|
|
645
554
|
return top.filter(x => {
|
|
646
555
|
const a = anchors.get(x.urlKey);
|
|
647
556
|
if (!a || a.n < 3)
|
|
648
|
-
return false;
|
|
557
|
+
return false;
|
|
649
558
|
const q = terms(x.query);
|
|
650
559
|
return q.length > 0 && !q.some((w) => titleHasTerm(a.pool, w));
|
|
651
560
|
}).map(x => { const a = anchors.get(x.urlKey); return { urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, inboundInContentLinks: a.n } }; });
|
|
652
561
|
},
|
|
653
562
|
},
|
|
654
563
|
{
|
|
655
|
-
// The "RAG snippetability" test. A local cross-encoder (the kind AI search uses to re-rank) scored
|
|
656
|
-
// every chunk against the page's top query; we persisted the single best-passage score. A low max
|
|
657
|
-
// means no dense, extractable answer anywhere on the page — it will lose in AI/passage search even
|
|
658
|
-
// if it keyword-matches. Model-derived (not deterministic truth) → N label, includeJudgement-gated.
|
|
659
|
-
// Requires `score_passages` to have run (like CWV needs page_lighthouse).
|
|
660
564
|
id: 'weak-passage-answer', category: 'merged', severity: 'high', labels: ['G', 'N'], certainty: 0.8, effortBase: 5, fixType: 'per-page',
|
|
661
565
|
title: 'No passage strongly answers the ranking query (AI-search risk)', fix: 'A local neural reranker found no single passage on this page that confidently answers its top query — the page covers the topic loosely but offers no dense, extractable answer, so AI/passage search will prefer a clearer source. Add a focused, self-contained passage: a heading that states the question + a direct ~50-word answer up top. Run `score_passages` to (re)populate.',
|
|
662
566
|
run: (c) => rows(c, `SELECT url_key urlKey, max_passage_score mps, max_passage_query q, max_passage_impr impr FROM pages
|
|
@@ -664,10 +568,6 @@ export const CHECKS = [
|
|
|
664
568
|
.map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.q, maxPassageScore: x.mps, impressions: x.impr ?? 0, note: 'best passage scores below the reranker confidence threshold' } })),
|
|
665
569
|
},
|
|
666
570
|
{
|
|
667
|
-
// Dejan: search weights the opening heavily and AI answers front-load. If the ranking query's
|
|
668
|
-
// terms are present LATER in the page but absent from the opening (~first 2 chunks / ~200 words),
|
|
669
|
-
// the answer is buried. (body-missing-top-query handles total absence; this is the buried case.)
|
|
670
|
-
// Informational intent only — front-loading matters less for navigational/transactional queries.
|
|
671
571
|
id: 'answer-not-front-loaded', category: 'merged', severity: 'med', labels: ['G', 'N'], certainty: 0.6, effortBase: 3, fixType: 'per-page',
|
|
672
572
|
title: 'Answer to the ranking query is buried, not front-loaded', fix: 'The page covers its top query but the terms don’t appear up top (the intro / first section). Google weights the opening heavily and AI answers front-load — move a direct ~50–100-word answer to the first section. (Heuristic — informational queries.)',
|
|
673
573
|
run: (c) => {
|
|
@@ -699,14 +599,10 @@ export const CHECKS = [
|
|
|
699
599
|
},
|
|
700
600
|
},
|
|
701
601
|
{
|
|
702
|
-
// Dejan "density beats length": AI grounds ~370 words/page, diminishing past ~1,500. A very long
|
|
703
|
-
// page with an over-long unbroken section grounds poorly — split it into focused, headed passages.
|
|
704
602
|
id: 'content-bloat', category: 'content', severity: 'low', labels: ['D', 'N'], certainty: 0.6, effortBase: 5, fixType: 'per-page',
|
|
705
603
|
title: 'Over-long section dilutes AI-grounding (density beats length)', fix: 'This page has a very long unbroken section. AI search grounds only ~370 words per page with sharp diminishing returns past ~1,500 — break the long section into focused, headed passages (or tighten it) so each answers one thing cleanly.',
|
|
706
604
|
run: (c) => {
|
|
707
605
|
const out = [];
|
|
708
|
-
// Use the real (uncapped) word_count ÷ number of headed sections — chunk TEXT is capped at
|
|
709
|
-
// extraction, so we infer over-long sections from words-per-heading, not from chunk length.
|
|
710
606
|
for (const x of iterRows(c, `SELECT url_key urlKey, word_count wc, body_chunks bc FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND word_count >= 2500 AND body_chunks IS NOT NULL`)) {
|
|
711
607
|
let chunks;
|
|
712
608
|
try {
|
|
@@ -724,15 +620,11 @@ export const CHECKS = [
|
|
|
724
620
|
},
|
|
725
621
|
},
|
|
726
622
|
{
|
|
727
|
-
// Hobo Level 3 freshness / lastSignificantUpdate. Gemini guard: YoY windows (negate seasonality
|
|
728
|
-
// + zero-click-SERP CTR loss). Flag when the page hasn't been meaningfully re-dated in >12 months
|
|
729
|
-
// AND clicks are down >25% YoY AND impressions down >15% YoY (impressions confirm ranking decay,
|
|
730
|
-
// not just CTR). Needs ~13 months of GSC — guarded by spanDays so it stays silent on shallow syncs.
|
|
731
623
|
id: 'stale-content', category: 'merged', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
732
624
|
title: 'Stale page declining year-on-year', fix: 'This page hasn’t been meaningfully updated in over a year and its Search Console clicks are down sharply versus the same period last year — refresh and expand the content (and honestly re-date it) to rebuild the freshness signal Google rewards.',
|
|
733
625
|
run: (c) => {
|
|
734
626
|
if (!c.gscMaxDate || spanDays(c) < 455)
|
|
735
|
-
return [];
|
|
627
|
+
return [];
|
|
736
628
|
const d = c.gscMaxDate;
|
|
737
629
|
const out = [];
|
|
738
630
|
const rs = rows(c, `
|
|
@@ -748,7 +640,7 @@ export const CHECKS = [
|
|
|
748
640
|
continue;
|
|
749
641
|
const ageDays = (Date.parse(d) - Date.parse(dm)) / 86400000;
|
|
750
642
|
if (!(ageDays >= 365))
|
|
751
|
-
continue;
|
|
643
|
+
continue;
|
|
752
644
|
out.push({ urlKey: x.url, evidence: { url: x.url, dateModified: dm.slice(0, 10), clicksYoY: `${x.pyC}→${x.curC} (-${Math.round((1 - x.curC / x.pyC) * 100)}%)`, impressionsYoY: `${x.pyI}→${x.curI}`, clicks: x.pyC - x.curC, impressions: x.pyI } });
|
|
753
645
|
}
|
|
754
646
|
return out;
|
|
@@ -790,23 +682,17 @@ export const CHECKS = [
|
|
|
790
682
|
without.push(p.urlKey);
|
|
791
683
|
}
|
|
792
684
|
if (pages.length === 0 || withBc / pages.length < 0.4)
|
|
793
|
-
return [];
|
|
685
|
+
return [];
|
|
794
686
|
return without.map(u => ({ urlKey: u, evidence: { note: 'site uses BreadcrumbList elsewhere; missing here' } }));
|
|
795
687
|
},
|
|
796
688
|
},
|
|
797
|
-
// ── Merged crawl × Search Console — the "expert questions" that need both datasets ──
|
|
798
689
|
{
|
|
799
690
|
id: 'ghost-pages', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
800
691
|
title: 'Ranking page the crawl can’t reach', fix: 'Google sends impressions/clicks to this URL but the site crawl never reached it — add internal links so it’s discoverable (or confirm it should exist and isn’t blocked).',
|
|
801
692
|
run: (c) => {
|
|
802
693
|
if (!c.gscMaxDate)
|
|
803
694
|
return [];
|
|
804
|
-
// Only meaningful on a COMPLETE crawl — if the crawl hit its maxPages cap, "absent
|
|
805
|
-
// from crawl" is unreliable. Tie the guard to the crawl that produced the CURRENT
|
|
806
|
-
// pages (not the latest crawl_metadata row, which may be a later failed crawl).
|
|
807
695
|
const m = c.db.prepare('SELECT urls_crawled c, max_pages m FROM crawl_metadata WHERE crawl_id = (SELECT crawl_id FROM pages LIMIT 1)').get();
|
|
808
|
-
// max_pages is nullable (NULL = no cap = complete crawl) — `c >= null` coerces to
|
|
809
|
-
// `c >= 0`, which would silently disable the check forever on capless crawls.
|
|
810
696
|
if (!m || m.c === 0 || (m.m != null && m.c >= m.m))
|
|
811
697
|
return [];
|
|
812
698
|
return rows(c, `SELECT page_key urlKey, SUM(clicks) clicks, SUM(impressions) impressions FROM search_analytics
|
|
@@ -831,7 +717,6 @@ export const CHECKS = [
|
|
|
831
717
|
.map(x => ({ urlKey: x.urlKey, evidence: { topQuery: x.query, impressions: x.impressions, title: x.title } }));
|
|
832
718
|
},
|
|
833
719
|
},
|
|
834
|
-
// ── Internal link graph (iPR + click-depth + anchor text, from the crawl `links` table) ──
|
|
835
720
|
{
|
|
836
721
|
id: 'deep-pages', category: 'crawlability', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
837
722
|
title: 'Page buried deep in the structure', fix: 'Add in-content (body) links from higher-level pages — this page is 4+ clicks from the homepage via body links.',
|
|
@@ -843,10 +728,6 @@ export const CHECKS = [
|
|
|
843
728
|
run: (c) => c.gscMaxDate ? rows(c, `SELECT p.url_key urlKey, p.ipr ipr, p.inlink_count inl, SUM(sa.impressions) impressions FROM pages p JOIN search_analytics sa ON sa.page_key=p.url_key WHERE p.indexable=1 AND p.ipr < 30 AND p.inlink_count BETWEEN 1 AND 3 AND sa.${win(c.gscMaxDate)} GROUP BY p.url_key HAVING SUM(sa.impressions) >= 300 ORDER BY impressions DESC`).map(r => ({ urlKey: r.urlKey, evidence: { impressions: r.impressions, ipr: Math.round(r.ipr), inlinks: r.inl } })) : [],
|
|
844
729
|
},
|
|
845
730
|
{
|
|
846
|
-
// The tier BETWEEN "orphan" (0 inlinks) and "fine": pages reached almost only via nav/footer.
|
|
847
|
-
// inlink_count counts ALL internal links, so a page sitting in the global nav looks well-linked
|
|
848
|
-
// even with zero EDITORIAL links — yet Google leans on in-content links for topic + equity. We
|
|
849
|
-
// count distinct in-content (placement='body') inbound sources; ≤2 + real demand = under-linked.
|
|
850
731
|
id: 'underlinked-editorial', category: 'merged', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
851
732
|
title: 'High demand, almost no in-content internal links', fix: 'This page earns real impressions but is reached mainly via nav/footer — add descriptive in-content links to it from related articles. Editorial body links pass more topical context and equity than templated nav links.',
|
|
852
733
|
run: (c) => c.gscMaxDate ? rows(c, `
|
|
@@ -858,18 +739,6 @@ export const CHECKS = [
|
|
|
858
739
|
HAVING SUM(sa.impressions) >= 300 AND bodyLinks <= 2
|
|
859
740
|
ORDER BY impressions DESC LIMIT 40`).map(r => ({ urlKey: r.urlKey, evidence: { impressions: r.impressions, inContentLinks: r.bodyLinks, note: 'reached mainly via nav/footer — thin on editorial (in-content) links' } })) : [],
|
|
860
741
|
},
|
|
861
|
-
// NOTE: internal anchor-text checks (over-optimisation + generic/empty anchors) prototyped
|
|
862
|
-
// and PULLED twice. Re-evaluated 2026-06-22 against the AgricIDaniel/claude-seo and
|
|
863
|
-
// Bhanunamikaze/Agentic-SEO-Skill repos, this time using the links.placement='body' filter
|
|
864
|
-
// plus excluding anchors that match the target's own title/H1. On real data (ehi.com.au) the
|
|
865
|
-
// dominant survivors are still false positives: sitewide template CTAs ("home" 864/865,
|
|
866
|
-
// "contact us", "apply today") and category links whose anchor IS the page title. The
|
|
867
|
-
// page-level "mostly generic-anchored" variant returned 0 signal; raw empty anchors are
|
|
868
|
-
// image/thumbnail-link noise (3,764/23,435 body links). Reliable detection needs an
|
|
869
|
-
// editorial-vs-template link classifier we don't store (placement='body' still includes
|
|
870
|
-
// in-template CTAs and product grids). Keep pulled — a wrong finding is worse than none.
|
|
871
|
-
// ── Backlinks (need page_backlinks populated via pull_backlinks; gate so we never assert
|
|
872
|
-
// "no external links" without data) ──
|
|
873
742
|
{
|
|
874
743
|
id: 'backlinks-to-404', category: 'crawlability', severity: 'crit', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
875
744
|
title: 'External backlinks pointing to a dead page', fix: '301-redirect this URL to the best live equivalent — external link equity is hitting a 4xx/5xx page and being wasted.',
|
|
@@ -882,17 +751,14 @@ export const CHECKS = [
|
|
|
882
751
|
run: (c) => {
|
|
883
752
|
const has = c.db.prepare('SELECT COUNT(*) n FROM page_backlinks').get().n;
|
|
884
753
|
if (!has)
|
|
885
|
-
return [];
|
|
754
|
+
return [];
|
|
886
755
|
return rows(c, `SELECT p.url_key urlKey FROM pages p LEFT JOIN page_backlinks b ON b.url_key=p.url_key WHERE p.status_code=200 AND p.indexable=1 AND p.inlink_count=0 AND COALESCE(b.backlinks,0)=0`)
|
|
887
756
|
.map(r => ({ urlKey: r.urlKey, evidence: { inlinks: 0, backlinks: 0 } }));
|
|
888
757
|
},
|
|
889
758
|
},
|
|
890
|
-
// ── Phase 6a — expert questions where crawl (intent) and reality diverge ──
|
|
891
759
|
{
|
|
892
760
|
id: 'ipr-bleed-by-status', category: 'crawlability', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'automated',
|
|
893
761
|
title: 'Internal link equity flowing into dead URLs', fix: 'Repoint or 301 these internal links — they target non-200 URLs and waste the internal PageRank of the (often high-authority) pages linking to them.',
|
|
894
|
-
// Sum the iPR of the SOURCE pages linking to each non-200 internal target. "Found a 404" is
|
|
895
|
-
// junior; "this 404 drains the equity of N high-iPR pages" is the director-level find.
|
|
896
762
|
run: (c) => rows(c, `SELECT l.target_key urlKey, COUNT(DISTINCT l.source_key) linkingPages,
|
|
897
763
|
ROUND(SUM(src.ipr), 1) wastedIpr, t.status_code st
|
|
898
764
|
FROM links l
|
|
@@ -905,9 +771,6 @@ export const CHECKS = [
|
|
|
905
771
|
{
|
|
906
772
|
id: 'broken-canonical-target', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
907
773
|
title: 'Canonical points to a broken or unhealthy URL', fix: 'Point the canonical at a live, indexable, self-canonical HTTPS URL — Google ignores a canonical whose target is a 4xx/5xx/redirect, noindex, itself canonicalised elsewhere (a chain/loop), or an HTTPS→HTTP downgrade.',
|
|
908
|
-
// Join the declared canonical_key back to the crawl. Only flag when we crawled the target.
|
|
909
|
-
// Skip self-canonicals. Covers: non-200 target, noindex target, canonical chain/loop (target
|
|
910
|
-
// canonicalises onward), and HTTPS→HTTP downgrade (research: Sitebulb indexability hints).
|
|
911
774
|
run: (c) => rows(c, `SELECT p.url_key urlKey, p.canonical_url canon, p.canonical_key ck, t.status_code st, t.noindex ni, t.canonical_key tck
|
|
912
775
|
FROM pages p JOIN pages t ON t.url_key = p.canonical_key
|
|
913
776
|
WHERE p.canonical_key IS NOT NULL AND p.canonical_key != p.url_key AND p.status_code = 200
|
|
@@ -925,8 +788,6 @@ export const CHECKS = [
|
|
|
925
788
|
{
|
|
926
789
|
id: 'faceted-spider-trap', category: 'crawlability', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 5, fixType: 'global',
|
|
927
790
|
title: 'Indexable faceted URLs burning crawl budget', fix: 'noindex (or robots-disallow / canonicalise) multi-parameter filter URLs — they are indexable but earn zero search traffic, so they only waste crawl budget and risk index bloat.',
|
|
928
|
-
// Multi-parameter (>=2 params), indexable, zero GSC impressions in-window = classic facet trap.
|
|
929
|
-
// Gate on GSC so "zero search value" is a real claim, not just "no data".
|
|
930
791
|
run: (c) => {
|
|
931
792
|
if (!c.gscMaxDate)
|
|
932
793
|
return [];
|
|
@@ -939,8 +800,6 @@ export const CHECKS = [
|
|
|
939
800
|
{
|
|
940
801
|
id: 'soft-404-shell', category: 'indexation', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
941
802
|
title: 'Soft 404 — 200 OK but Google treats it as not-found', fix: 'Either populate the page with real content, or return a true 404/410 (or noindex) — it serves 200 but Google has flagged it as a soft 404.',
|
|
942
|
-
// URL Inspection page-fetch-state = soft 404 while the crawler sees a 200. The crawler alone
|
|
943
|
-
// would call this page fine; Google disagrees. Needs URL Inspection populated.
|
|
944
803
|
run: (c) => rows(c, `SELECT i.url_key urlKey, p.word_count wc, p.bytes bytes
|
|
945
804
|
FROM url_inspection i JOIN pages p ON p.url_key = i.url_key
|
|
946
805
|
WHERE LOWER(i.page_fetch_state) LIKE '%soft%' AND p.status_code = 200`).map(r => ({ urlKey: r.urlKey, evidence: { pageFetchState: 'soft 404', wordCount: r.wc, bytes: r.bytes } })),
|
|
@@ -948,9 +807,6 @@ export const CHECKS = [
|
|
|
948
807
|
{
|
|
949
808
|
id: 'rich-result-issues', category: 'schema', severity: 'med', labels: ['D', 'G'], certainty: 1, effortBase: 3, fixType: 'per-page',
|
|
950
809
|
title: 'Google-verified rich result issues (URL Inspection)', fix: 'Fix the structured-data issues Google itself reports for this page — these come from the URL Inspection API (Google’s own validation), not our local validator, so they are authoritative. Address the listed issue messages per rich-result type.',
|
|
951
|
-
// Parses the stored richResultsResult JSON (url_inspection.rich_results) that inspect_urls
|
|
952
|
-
// already captures: detectedItems[].items[].issues[] carries Google's severity + message.
|
|
953
|
-
// Needs URL Inspection populated (run inspect_urls). Zero API cost — data is already in the DB.
|
|
954
810
|
run: (c) => {
|
|
955
811
|
const out = [];
|
|
956
812
|
for (const r of iterRows(c, `SELECT url_key urlKey, rich_results rr FROM url_inspection WHERE rich_results IS NOT NULL AND rich_results != ''`)) {
|
|
@@ -961,8 +817,6 @@ export const CHECKS = [
|
|
|
961
817
|
catch {
|
|
962
818
|
continue;
|
|
963
819
|
}
|
|
964
|
-
// Dedup per (type, severity, message) with a count — a listicle repeats the same
|
|
965
|
-
// "Missing field review" warning per product item; keep one sample item per group.
|
|
966
820
|
const groups = new Map();
|
|
967
821
|
for (const det of parsed?.detectedItems ?? []) {
|
|
968
822
|
for (const item of det?.items ?? []) {
|
|
@@ -994,17 +848,14 @@ export const CHECKS = [
|
|
|
994
848
|
title: 'hreflang missing return tag (not reciprocated)', fix: 'Add the reciprocal hreflang on the target page — Google ignores one-way hreflang annotations that don’t link back.',
|
|
995
849
|
run: (c) => hreflangFindings(c).noReturn,
|
|
996
850
|
},
|
|
997
|
-
// ── 6a finishers — consume persisted DataForSEO enrichments (gated; need the data pulled) ──
|
|
998
851
|
{
|
|
999
852
|
id: 'intent-vs-pagetype-mismatch', category: 'merged', severity: 'med', labels: ['G', 'N'], certainty: 0.7, effortBase: 8, fixType: 'per-page',
|
|
1000
853
|
title: 'Page type mismatches its query intent', fix: 'Reformat or retarget the page — its template doesn’t match the SERP intent for its top query (e.g. a product page ranking for an informational query, or an article for a transactional one). Run search_intent siteUrl:<property> to populate intents.',
|
|
1001
|
-
// Join each page's top GSC query → persisted keyword_intent → page schema flavour (from json_ld).
|
|
1002
|
-
// Judgement (N, 0.7): intent + schema-type inference is heuristic, so it only runs with includeJudgement.
|
|
1003
854
|
run: (c) => {
|
|
1004
855
|
if (!c.gscMaxDate)
|
|
1005
856
|
return [];
|
|
1006
857
|
if ((c.db.prepare('SELECT COUNT(*) n FROM keyword_intent').get().n) === 0)
|
|
1007
|
-
return [];
|
|
858
|
+
return [];
|
|
1008
859
|
const r = rows(c, `SELECT s.page_key urlKey, s.query query, ki.intent intent, p.json_ld jsonLd FROM
|
|
1009
860
|
(SELECT page_key, query, ROW_NUMBER() OVER (PARTITION BY page_key ORDER BY SUM(impressions) DESC) rn
|
|
1010
861
|
FROM search_analytics WHERE query IS NOT NULL AND page_key IS NOT NULL AND ${win(c.gscMaxDate)} GROUP BY page_key, query) s
|
|
@@ -1035,7 +886,7 @@ export const CHECKS = [
|
|
|
1035
886
|
if (!c.gscMaxDate)
|
|
1036
887
|
return [];
|
|
1037
888
|
if ((c.db.prepare('SELECT COUNT(*) n FROM page_cwv').get().n) === 0)
|
|
1038
|
-
return [];
|
|
889
|
+
return [];
|
|
1039
890
|
return rows(c, `SELECT cw.url_key urlKey, cw.lcp_ms lcp, cw.cls cls, cw.performance perf,
|
|
1040
891
|
(SELECT COALESCE(SUM(clicks),0) FROM search_analytics sa WHERE sa.page_key = cw.url_key AND ${win(c.gscMaxDate)}) clicks
|
|
1041
892
|
FROM page_cwv cw
|
|
@@ -1044,14 +895,13 @@ export const CHECKS = [
|
|
|
1044
895
|
.map(r => ({ urlKey: r.urlKey, evidence: { clicks: r.clicks, lcpMs: r.lcp != null ? Math.round(r.lcp) : null, cls: r.cls != null ? Math.round(r.cls * 1000) / 1000 : null, performance: r.perf != null ? Math.round(r.perf * 100) : null } }));
|
|
1045
896
|
},
|
|
1046
897
|
},
|
|
1047
|
-
// ── 6b — per-template systemic issues (template-typed, deterministic) ──
|
|
1048
898
|
{
|
|
1049
899
|
id: 'pagination-canonical-to-page-1', category: 'crawlability', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
|
|
1050
900
|
title: 'Paginated pages canonicalising away from themselves', fix: 'Make each paginated page (page 2, 3, …) self-canonical. Canonicalising page 2+ back to page 1 tells Google the deeper pages are duplicates, so products/articles linked only from page 2+ drop out of the crawl.',
|
|
1051
901
|
run: (c) => rows(c, `SELECT url_key urlKey, url, canonical_url canon FROM pages
|
|
1052
902
|
WHERE status_code=200 AND canonical_count>0 AND canonical_key IS NOT NULL AND canonical_key != url_key
|
|
1053
903
|
AND (rel_prev=1 OR url LIKE '%/page/%' OR url GLOB '*[?&]page=[0-9]*' OR url GLOB '*[?&]paged=[0-9]*' OR url GLOB '*[?&]p=[0-9]*')`)
|
|
1054
|
-
.filter(r => !/[?&](page|p)=1(\b|&|$)/.test(r.url) && !/\/page\/1(\/?$|\?)/.test(r.url))
|
|
904
|
+
.filter(r => !/[?&](page|p)=1(\b|&|$)/.test(r.url) && !/\/page\/1(\/?$|\?)/.test(r.url))
|
|
1055
905
|
.map(r => ({ urlKey: r.urlKey, evidence: { url: r.url, canonical: r.canon } })),
|
|
1056
906
|
},
|
|
1057
907
|
{
|
|
@@ -1074,13 +924,12 @@ export const CHECKS = [
|
|
|
1074
924
|
return out;
|
|
1075
925
|
},
|
|
1076
926
|
},
|
|
1077
|
-
// ── 6d — Wikidata entity layer (heuristic H1→QID; N/judgement, gated on resolve_entities) ──
|
|
1078
927
|
{
|
|
1079
928
|
id: 'entity-internal-link-gap', category: 'crawlability', severity: 'low', labels: ['N'], certainty: 0.5, effortBase: 3, fixType: 'per-page',
|
|
1080
929
|
title: 'Topically related pages not internally linked', fix: 'Add an internal link from the broader page to the more specific one — Wikidata says their entities are related (subclass-of / part-of) but no internal link connects them, leaving a gap in the topical mesh. Run resolve_entities first; verify the entity match before acting (heuristic).',
|
|
1081
930
|
run: (c) => {
|
|
1082
931
|
if ((c.db.prepare('SELECT COUNT(*) n FROM page_entity').get().n) === 0)
|
|
1083
|
-
return [];
|
|
932
|
+
return [];
|
|
1084
933
|
return rows(c, `SELECT parent.url_key urlKey, child.url_key target, parent.label pl, child.label cl, ee.relation rel
|
|
1085
934
|
FROM entity_edge ee
|
|
1086
935
|
JOIN page_entity child ON child.qid = ee.qid
|
|
@@ -1090,7 +939,6 @@ export const CHECKS = [
|
|
|
1090
939
|
.map(r => ({ urlKey: r.urlKey, evidence: { suggestLinkTo: r.target, parentEntity: r.pl, childEntity: r.cl, relation: r.rel } }));
|
|
1091
940
|
},
|
|
1092
941
|
},
|
|
1093
|
-
// ── Sitemap ↔ crawl reconciliation (gated on a sitemap having been fetched) ──
|
|
1094
942
|
{
|
|
1095
943
|
id: 'sitemap-non-indexable', category: 'indexation', severity: 'high', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'global',
|
|
1096
944
|
title: 'Sitemap lists non-indexable URLs', fix: 'Remove URLs from the XML sitemap that are 4xx/5xx, redirected, noindex, canonicalised or robots-blocked — the sitemap should list only canonical, indexable pages, or Google loses trust in it.',
|
|
@@ -1121,7 +969,6 @@ export const CHECKS = [
|
|
|
1121
969
|
.map(r => ({ urlKey: r.urlKey, evidence: { inlinks: 0, note: 'in sitemap, no internal links' } }));
|
|
1122
970
|
},
|
|
1123
971
|
},
|
|
1124
|
-
// ── Cheap performance proxies (from data captured at crawl time — no extra fetch) ──
|
|
1125
972
|
{
|
|
1126
973
|
id: 'slow-response', category: 'performance', severity: 'med', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'global',
|
|
1127
974
|
title: 'Slow server response (TTFB proxy)', fix: 'Investigate slow server/TTFB — caching, CDN, or backend. Response time over ~1.5s hurts Core Web Vitals and crawl rate.',
|
|
@@ -1131,11 +978,6 @@ export const CHECKS = [
|
|
|
1131
978
|
{
|
|
1132
979
|
id: 'large-html', category: 'performance', severity: 'low', labels: ['D'], certainty: 1, effortBase: 5, fixType: 'per-page',
|
|
1133
980
|
title: 'Large HTML document (transferred)', fix: 'Trim the HTML payload — bloated markup slows render and First Contentful Paint (often huge inline SVG/CSS/JSON or unminified output). Judged on TRANSFERRED bytes, not raw: behind a compressing CDN (brotli/gzip) raw size matters far less.',
|
|
1134
|
-
// Severity is on what the browser actually downloads, not raw bytes: a 200KB page served brotli
|
|
1135
|
-
// is ~45KB over the wire and is NOT a real perf problem. We estimate transfer size (raw × ~0.22
|
|
1136
|
-
// for br/gzip, else raw) and only flag pages whose ESTIMATED transferred HTML exceeds ~60KB.
|
|
1137
|
-
// Prefilter at the 60KB threshold itself, not higher — an UNCOMPRESSED 60–150KB page
|
|
1138
|
-
// (est = raw) is exactly the case that matters most and must reach the est filter.
|
|
1139
981
|
run: (c) => rows(c, `SELECT url_key urlKey, bytes, content_encoding enc FROM pages WHERE status_code=200 AND ${HTML_CT} AND bytes > 60000 ORDER BY bytes DESC`)
|
|
1140
982
|
.map(r => { const compressed = /br|gzip|deflate|zstd/i.test(r.enc || ''); const est = compressed ? Math.round(r.bytes * 0.22) : r.bytes; return { urlKey: r.urlKey, est, compressed, raw: r.bytes }; })
|
|
1141
983
|
.filter(x => x.est > 60000)
|
|
@@ -1147,7 +989,6 @@ export const CHECKS = [
|
|
|
1147
989
|
run: (c) => rows(c, `SELECT url_key urlKey FROM pages WHERE status_code=200 AND ${HTML_CT} AND (content_encoding IS NULL OR content_encoding='')`)
|
|
1148
990
|
.map(r => ({ urlKey: r.urlKey, evidence: {} })),
|
|
1149
991
|
},
|
|
1150
|
-
// ── Checklist-coverage additions (2026-07-20 — see plan/checklist-coverage.md) ──
|
|
1151
992
|
{
|
|
1152
993
|
id: 'robots-blocked-with-traffic', category: 'crawlability', severity: 'high', labels: ['D', 'G'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
1153
994
|
title: 'Robots-blocked page still earning search traffic', fix: 'This URL is disallowed in robots.txt yet Google still shows it (usually as a bare "no information" result) and users still land on it. Either unblock it so it can be crawled and ranked properly, or — if it genuinely shouldn\'t be found — unblock it AND add noindex (a robots-blocked page can never see the noindex).',
|
|
@@ -1184,8 +1025,6 @@ export const CHECKS = [
|
|
|
1184
1025
|
{
|
|
1185
1026
|
id: 'favicon-missing', category: 'onpage', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
1186
1027
|
title: 'No favicon declared', fix: 'Add a favicon (<link rel="icon" …>) — Google shows it next to your result on mobile, and a missing one costs a little trust/recognition on every SERP appearance. One line in the template.',
|
|
1187
|
-
// Gated on the column being populated (has_favicon is NULL on crawls from before this
|
|
1188
|
-
// was captured) — never flag from a pre-feature crawl.
|
|
1189
1028
|
run: (c) => {
|
|
1190
1029
|
const hp = c.db.prepare(`SELECT url_key, has_favicon hf FROM pages WHERE status_code=200 AND has_favicon IS NOT NULL ORDER BY (click_depth=0) DESC, inlink_count DESC LIMIT 1`).get();
|
|
1191
1030
|
return hp && hp.hf === 0 ? [{ urlKey: hp.url_key, evidence: { note: 'no <link rel="icon"> on the homepage' } }] : [];
|
|
@@ -1197,10 +1036,9 @@ export const CHECKS = [
|
|
|
1197
1036
|
run: (c) => {
|
|
1198
1037
|
const all = c.db.prepare(`SELECT url_key, lastmod FROM sitemap_urls WHERE lastmod IS NOT NULL AND lastmod <> ''`).all();
|
|
1199
1038
|
if (all.length < 20)
|
|
1200
|
-
return [];
|
|
1039
|
+
return [];
|
|
1201
1040
|
const ev = { urlsWithLastmod: all.length };
|
|
1202
1041
|
let tripped = false;
|
|
1203
|
-
// Tell 1: a generator stamping every URL with "now" — >90% share one date, and that date is recent.
|
|
1204
1042
|
const byDay = new Map();
|
|
1205
1043
|
for (const r of all)
|
|
1206
1044
|
byDay.set(r.lastmod.slice(0, 10), (byDay.get(r.lastmod.slice(0, 10)) ?? 0) + 1);
|
|
@@ -1210,19 +1048,13 @@ export const CHECKS = [
|
|
|
1210
1048
|
tripped = true;
|
|
1211
1049
|
ev.sharedStamp = `${Math.round(topN / all.length * 100)}% of URLs claim ${topDay} — a generation timestamp, not a change date`;
|
|
1212
1050
|
}
|
|
1213
|
-
// Tell 2: dates in the future.
|
|
1214
1051
|
const future = all.filter(r => Date.parse(r.lastmod) > Date.now() + 86400000).length;
|
|
1215
1052
|
if (future > 0) {
|
|
1216
1053
|
tripped = true;
|
|
1217
1054
|
ev.futureDates = future;
|
|
1218
1055
|
}
|
|
1219
|
-
// Tell 3: lastmod claims a change between our two most recent crawls, yet none of the
|
|
1220
|
-
// tracked page fields (status, title, meta, H1, word count, schema types) changed.
|
|
1221
1056
|
const crawls = latestTwoCrawls(c.db);
|
|
1222
1057
|
if (crawls.length === 2) {
|
|
1223
|
-
// Compare DATE prefixes on both sides — lastmod is stored verbatim and often a full
|
|
1224
|
-
// timestamp; compared raw against a 10-char date, same-day stamps sort "after" the
|
|
1225
|
-
// boundary and are silently excluded (exactly the stamp-everything-today pattern).
|
|
1226
1058
|
const phantom = c.db.prepare(`SELECT s.url_key FROM sitemap_urls s
|
|
1227
1059
|
JOIN page_snapshots n ON n.url_key = s.url_key AND n.crawl_id = ?
|
|
1228
1060
|
JOIN page_snapshots o ON o.url_key = s.url_key AND o.crawl_id = ?
|
|
@@ -1242,8 +1074,6 @@ export const CHECKS = [
|
|
|
1242
1074
|
{
|
|
1243
1075
|
id: 'no-304-revalidation', category: 'performance', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
1244
1076
|
title: 'Server ignores conditional requests (no 304)', fix: 'Pages advertise Last-Modified/ETag but the server re-serves a full 200 when asked "has this changed?" (If-Modified-Since / If-None-Match). A properly configured server answers 304 Not Modified — it saves bandwidth on every revalidating crawler and cache, and signals stability to Googlebot. Usually a server/CDN setting.',
|
|
1245
|
-
// Populated by the crawler's post-crawl probe (pages.conditional_304); NULL-gated so
|
|
1246
|
-
// pre-feature crawls never flag. Fires only when NO probed page honoured the request.
|
|
1247
1077
|
run: (c) => {
|
|
1248
1078
|
const r = c.db.prepare(`SELECT COUNT(*) probed, COALESCE(SUM(conditional_304),0) ok FROM pages WHERE conditional_304 IS NOT NULL`).get();
|
|
1249
1079
|
if (r.probed < 5 || r.ok > 0)
|
|
@@ -1255,45 +1085,12 @@ export const CHECKS = [
|
|
|
1255
1085
|
{
|
|
1256
1086
|
id: 'analytics-missing', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global',
|
|
1257
1087
|
title: 'No client-side analytics detected', fix: 'No analytics or tag-manager snippet was found on any crawled page (GA4, GTM, Plausible, Matomo, Fathom, Clarity…). If you measure server-side, ignore this; otherwise you\'re flying blind — install an analytics package before making SEO decisions.',
|
|
1258
|
-
// Fires only when EVERY populated page lacks a snippet — one page with analytics = installed.
|
|
1259
1088
|
run: (c) => {
|
|
1260
1089
|
const r = c.db.prepare(`SELECT COUNT(*) total, COALESCE(SUM(has_analytics),0) withA FROM pages WHERE status_code=200 AND ${HTML_CT} AND has_analytics IS NOT NULL`).get();
|
|
1261
1090
|
return r.total >= 3 && r.withA === 0 ? [{ urlKey: null, evidence: { pagesChecked: r.total, note: 'no known analytics snippet on any crawled page (server-side measurement is invisible to a crawl)' } }] : [];
|
|
1262
1091
|
},
|
|
1263
1092
|
},
|
|
1264
|
-
|
|
1265
|
-
// fundamentals/ai-optimization-guide, read 2026-08-02) — principle → check mapping:
|
|
1266
|
-
// • Unique/compelling/non-commodity, people-first content .... thin-content, content-bloat,
|
|
1267
|
-
// ai-slop-signals (NEW below — mechanical/templated prose is the anti-signal of "unique take")
|
|
1268
|
-
// • Unique point of view / first-hand experience .............. article-no-author (NEW below —
|
|
1269
|
-
// unattributed articles are the deterministically checkable slice), stale-content
|
|
1270
|
-
// • Clear organisation (paragraphs/sections/headings) ......... poor-chunkability, content-bloat,
|
|
1271
|
-
// heading-hierarchy; answer coverage: body-missing-top-query, rag-answer-gap,
|
|
1272
|
-
// answer-not-front-loaded, weak-passage-answer, low-extractability
|
|
1273
|
-
// • Indexed + snippet-eligible / technical requirements ....... indexation family (noindex,
|
|
1274
|
-
// canonical, robots-blocked-with-traffic), soft-404-shell, sitemap reconciliation
|
|
1275
|
-
// • Crawlable public content .................................. crawlability family, broken links,
|
|
1276
|
-
// redirect chains; freshness-honesty (sitemap-lastmod-untrustworthy, no-304-revalidation)
|
|
1277
|
-
// • Semantic HTML / parseable ................................. heading-hierarchy, missing-h1,
|
|
1278
|
-
// multiple-h1, missing-lang
|
|
1279
|
-
// • Reduce duplicate content .................................. duplicate-title/meta, canonical
|
|
1280
|
-
// family, faceted-spider-trap, keyword-cannibalisation
|
|
1281
|
-
// • Images/video supporting text .............................. image-alt, images-missing-dimensions
|
|
1282
|
-
// • Page experience / latency ................................. performance proxies, high-yield-cwv-fail
|
|
1283
|
-
// • Structured data honesty (not required, but keep it valid) . schema-validate family,
|
|
1284
|
-
// article-date-illogical
|
|
1285
|
-
// • Don't chunk artificially / rewrite for AI ................. covered by NOT having such checks;
|
|
1286
|
-
// our chunk checks reward structure for humans, not tiny AI fragments
|
|
1287
|
-
// • "Don't create llms.txt" ................................... tension with agent-readiness probes,
|
|
1288
|
-
// which score llms.txt for *agent* (not Google AI-search) consumption — left as-is, different audience
|
|
1289
|
-
// • Merchant Center / Business Profile / GenAI report ......... out of scope (not crawl/GSC data)
|
|
1290
|
-
{
|
|
1291
|
-
// Anti-signal of the guide's "unique, non-commodity, people-first content": prose that reads
|
|
1292
|
-
// machine-generated. Three heuristics over body_chunks — slop-lexicon density, sentence-length
|
|
1293
|
-
// uniformity (low variance = mechanical), repeated chunk openers (page-internal boilerplate).
|
|
1294
|
-
// Precision over recall: absolute floors on every signal, ≥2 signals required, AND the composite
|
|
1295
|
-
// score must sit in the top decile of pages showing any signal. Judgement-gated — a human wrote
|
|
1296
|
-
// "delve" long before LLMs did.
|
|
1093
|
+
{
|
|
1297
1094
|
id: 'ai-slop-signals', category: 'content', severity: 'low', labels: ['N'], certainty: 0.5, effortBase: 5, fixType: 'per-page',
|
|
1298
1095
|
title: 'Prose shows machine-generated (slop) signals', fix: 'This page’s copy trips several statistical tells of generic AI-generated text: stock filler phrases, unusually uniform sentence lengths, and/or sections that all open the same way. Google’s AI-search guidance rewards unique, people-first content with a first-hand point of view — rewrite the flagged sections with specifics only you can supply (real experience, real numbers, real opinions) and cut the filler. (Heuristic — verify by reading the page; competent human writing can trip these tells.)',
|
|
1299
1096
|
run: (c) => {
|
|
@@ -1322,7 +1119,6 @@ export const CHECKS = [
|
|
|
1322
1119
|
const bodyWords = body.split(/\s+/).filter(Boolean).length;
|
|
1323
1120
|
if (bodyWords < 250)
|
|
1324
1121
|
continue;
|
|
1325
|
-
// (i) slop-lexicon density (hits per 1000 words, ≥3 distinct terms required)
|
|
1326
1122
|
const hits = new Map();
|
|
1327
1123
|
for (const p of PHRASES) {
|
|
1328
1124
|
let n = 0, i = -1;
|
|
@@ -1334,7 +1130,6 @@ export const CHECKS = [
|
|
|
1334
1130
|
const totalHits = [...hits.values()].reduce((a, b) => a + b, 0);
|
|
1335
1131
|
const density = totalHits / bodyWords * 1000;
|
|
1336
1132
|
const lexSignal = hits.size >= 3 && density >= 2.5;
|
|
1337
|
-
// (ii) sentence-length uniformity — coefficient of variation over sentence word-counts
|
|
1338
1133
|
const sentences = body.split(/[.!?]+\s/).map(s => s.split(/\s+/).filter(Boolean).length).filter(n => n >= 5 && n <= 60);
|
|
1339
1134
|
let cv = null;
|
|
1340
1135
|
if (sentences.length >= 12) {
|
|
@@ -1342,11 +1137,9 @@ export const CHECKS = [
|
|
|
1342
1137
|
cv = Math.sqrt(sentences.reduce((a, b) => a + (b - mean) ** 2, 0) / sentences.length) / mean;
|
|
1343
1138
|
}
|
|
1344
1139
|
const uniformSignal = cv !== null && cv < 0.28;
|
|
1345
|
-
// (iii) repeated openers across the page's own chunks (first 3 words, ≥3 chunks sharing one)
|
|
1346
1140
|
const openers = new Map();
|
|
1347
1141
|
if (texts.length >= 5)
|
|
1348
1142
|
for (const t of texts) {
|
|
1349
|
-
// Letters only — numeric/UI-chrome openers ("Show 10 20…") are widget text, not prose.
|
|
1350
1143
|
const o = t.toLowerCase().replace(/[^a-z\s]/g, ' ').split(/\s+/).filter(w => w.length >= 2).slice(0, 3).join(' ');
|
|
1351
1144
|
if (o.split(' ').length === 3)
|
|
1352
1145
|
openers.set(o, (openers.get(o) ?? 0) + 1);
|
|
@@ -1361,10 +1154,6 @@ export const CHECKS = [
|
|
|
1361
1154
|
}
|
|
1362
1155
|
if (!metrics.length)
|
|
1363
1156
|
return [];
|
|
1364
|
-
// Outliers only: the LEXICON signal is mandatory (uniform sentences + repeated openers
|
|
1365
|
-
// without a single slop phrase is template chrome, not slop — live-verified on simracing),
|
|
1366
|
-
// plus ≥1 corroborating signal, AND composite score in the top decile of pages that showed
|
|
1367
|
-
// any signal at all.
|
|
1368
1157
|
const sorted = metrics.map(m => m.score).sort((a, b) => a - b);
|
|
1369
1158
|
const p90 = sorted[Math.min(sorted.length - 1, Math.floor(sorted.length * 0.9))];
|
|
1370
1159
|
return metrics
|
|
@@ -1381,11 +1170,6 @@ export const CHECKS = [
|
|
|
1381
1170
|
},
|
|
1382
1171
|
},
|
|
1383
1172
|
{
|
|
1384
|
-
// The guide's "unique point of view based on personal experience or expertise", cut down to its
|
|
1385
|
-
// deterministically checkable slice: an Article/BlogPosting that carries schema yet names no
|
|
1386
|
-
// author. Attribution is the machine-readable experience/expertise signal; a byline-less article
|
|
1387
|
-
// is the commodity-content default. Only fires where Article schema EXISTS (no schema at all is
|
|
1388
|
-
// schema-opportunity territory, not this check).
|
|
1389
1173
|
id: 'article-no-author', category: 'schema', severity: 'low', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'per-page',
|
|
1390
1174
|
title: 'Article schema with no author attribution', fix: 'This page marks itself up as an Article/BlogPosting but declares no author. Google’s AI-search guidance rewards content with a demonstrable first-hand point of view — add `author` (a Person with a real name, ideally linking to an author page) to the Article schema and a visible byline to match.',
|
|
1391
1175
|
run: (c) => {
|
|
@@ -1407,13 +1191,6 @@ export const CHECKS = [
|
|
|
1407
1191
|
return out;
|
|
1408
1192
|
},
|
|
1409
1193
|
},
|
|
1410
|
-
// ── Google Discover readiness (article-template surface) ────────────────
|
|
1411
|
-
// Discover ranks article/news pages on technical hygiene + visual/structured signals + entity-rich
|
|
1412
|
-
// headlines. All of these read data we ALREADY crawl (robots/x_robots_tag, json_ld, og_tags,
|
|
1413
|
-
// response_time_ms, body_chunks) and are gated to Discover-eligible templates via isDiscoverEligible
|
|
1414
|
-
// so Product/collection pages are never flagged. Honesty guard: pixel-level image-ratio checks are
|
|
1415
|
-
// deliberately NOT here — we're HEAD-only, so we verify only what the schema DECLARES, never a
|
|
1416
|
-
// "16:9 confirmed" from an image we didn't fetch. See plan/google-discover.md.
|
|
1417
1194
|
{
|
|
1418
1195
|
id: 'discover-max-image-preview', category: 'onpage', severity: 'med', labels: ['D'], certainty: 1, effortBase: 1, fixType: 'global', yieldCoef: 0.15,
|
|
1419
1196
|
title: 'Article missing max-image-preview:large (no large Discover card)',
|
|
@@ -1436,9 +1213,6 @@ export const CHECKS = [
|
|
|
1436
1213
|
id: 'discover-slow-response', category: 'performance', severity: 'med', labels: ['D', 'N'], certainty: 0.5, effortBase: 5, fixType: 'global', yieldCoef: 0.1,
|
|
1437
1214
|
title: 'Article response looks slow for Discover (>600ms wall-time)',
|
|
1438
1215
|
fix: 'Aim for server response (TTFB) under 600ms — under 200ms is optimal — since Discover favours freshly-crawlable content. NOTE: the figure here is whole-fetch WALL TIME (it includes the HTML download and any retry/back-off), not a clean server TTFB, so confirm against a real TTFB or field measurement before prioritising. Caching, a CDN, or backend work are the usual levers.',
|
|
1439
|
-
// Wall time only — response_time_ms is Date.now() across the whole fetch. We exclude pages that
|
|
1440
|
-
// resolved through a redirect chain (extra hops inflate the number) and label this JUDGEMENT (N),
|
|
1441
|
-
// because it is a proxy, not a measured TTFB. Do not present it as deterministic.
|
|
1442
1216
|
run: (c) => rows(c, `SELECT url_key urlKey, json_ld jsonLd, og_tags ogTags, response_time_ms ms FROM pages WHERE ${DISCOVER_PREFILTER} AND response_time_ms > 600 AND (redirects IS NULL OR redirects='') ORDER BY response_time_ms DESC`)
|
|
1443
1217
|
.filter(r => isDiscoverEligible(r.jsonLd, r.ogTags))
|
|
1444
1218
|
.map(r => ({ urlKey: r.urlKey, evidence: { wallTimeMs: r.ms, target: '<600ms (ideal <200ms)', note: 'whole-fetch wall time incl. HTML download / retries — a proxy, not a measured TTFB' } })),
|
|
@@ -1448,13 +1222,10 @@ export const CHECKS = [
|
|
|
1448
1222
|
title: 'High crawl waste (budget spent on redirects / non-indexable URLs)',
|
|
1449
1223
|
fix: 'Google allocates a finite crawl budget per host; spending it on redirects, error pages and non-indexable URLs slows discovery of your real content (and Discover leans on fast discovery). Aim for almost everything Google crawls to be an indexable 200 — repoint internal links off redirects/404s, prune faceted/parameter URLs, and keep the sitemap to canonical indexable pages.',
|
|
1450
1224
|
run: (c) => {
|
|
1451
|
-
const clause = `status_code IS NOT NULL AND status_code NOT IN (429,503)`;
|
|
1225
|
+
const clause = `status_code IS NOT NULL AND status_code NOT IN (429,503)`;
|
|
1452
1226
|
const tot = (rows(c, `SELECT COUNT(*) n FROM pages WHERE ${clause}`)[0]?.n) || 0;
|
|
1453
1227
|
if (tot < 50)
|
|
1454
|
-
return [];
|
|
1455
|
-
// Non-indexable HTML by reason — EXCLUDING non-html assets (images/PDFs the crawler followed are
|
|
1456
|
-
// not article crawl-waste). Internal redirects are stored as the final page + hops in
|
|
1457
|
-
// pages.redirects (not as 3xx rows), so count redirect-chain resolutions separately.
|
|
1228
|
+
return [];
|
|
1458
1229
|
const byReason = rows(c, `SELECT COALESCE(indexable_reason,'(other)') reason, COUNT(*) n FROM pages WHERE ${clause} AND indexable=0 AND COALESCE(indexable_reason,'') != 'non-html' GROUP BY 1 ORDER BY 2 DESC`);
|
|
1459
1230
|
const nonIndexable = byReason.reduce((s, r) => s + r.n, 0);
|
|
1460
1231
|
const redirected = (rows(c, `SELECT COUNT(*) n FROM pages WHERE ${clause} AND redirects IS NOT NULL AND redirects != ''`)[0]?.n) || 0;
|
|
@@ -1469,9 +1240,6 @@ export const CHECKS = [
|
|
|
1469
1240
|
id: 'discover-generic-article-type', category: 'schema', severity: 'low', labels: ['D', 'N'], certainty: 0.5, effortBase: 1, fixType: 'global',
|
|
1470
1241
|
title: 'Generic Article type where a more specific one fits',
|
|
1471
1242
|
fix: 'The page’s Article markup uses only the generic `Article` type. Google recommends the most specific applicable type — `NewsArticle` for news, `LiveBlogPosting` for live coverage, `ProfilePage` for a person/creator profile — which unlocks richer Discover/News treatment. Usually one template/plugin setting. Only change it if the more specific type genuinely fits (judgement).',
|
|
1472
|
-
// Reads the FULL @type set of each article-family node (nodeTypeList — array-@type aware). Fires
|
|
1473
|
-
// only when the article family present is exactly {article}, i.e. no more-specific subtype is
|
|
1474
|
-
// declared anywhere. A page typed ["Article","BlogPosting"] is NOT flagged (BlogPosting IS specific).
|
|
1475
1243
|
run: (c) => rows(c, `SELECT url_key urlKey, json_ld jsonLd FROM pages WHERE status_code=200 AND indexable=1 AND ${HTML_CT} AND json_ld LIKE '%Article%'`)
|
|
1476
1244
|
.map(r => { const present = new Set(); for (const n of discoverArticleNodes(r.jsonLd))
|
|
1477
1245
|
for (const t of nodeTypeList(n))
|
|
@@ -1484,27 +1252,14 @@ export const CHECKS = [
|
|
|
1484
1252
|
id: 'discover-image-schema', category: 'schema', severity: 'med', labels: ['D'], certainty: 1, effortBase: 3, fixType: 'per-page', yieldCoef: 0.12,
|
|
1485
1253
|
title: 'Article schema declares no image (Discover needs a large image)',
|
|
1486
1254
|
fix: 'Add an `image` to the Article structured data — Discover cards need a big image and cannot build one from the page alone. Google wants a high-res image (≥1200px wide), ideally in three crops: 16:9 (1200×675), 4:3 (1200×900) and 1:1 (1200×1200). The main visible image at the top of the article should match the 16:9 schema image.',
|
|
1487
|
-
// MUST have ≥1 article node whose `image` is absent — this check is about that node. Without the
|
|
1488
|
-
// length guard, `![].some(...)` is true and pages with NO article schema at all get flagged for a
|
|
1489
|
-
// schema that doesn't exist (the .some-on-empty bug that produced 100% false positives).
|
|
1490
1255
|
run: (c) => rows(c, `SELECT url_key urlKey, json_ld jsonLd, og_tags ogTags FROM pages WHERE ${DISCOVER_PREFILTER}`)
|
|
1491
1256
|
.map(r => ({ urlKey: r.urlKey, nodes: discoverArticleNodes(r.jsonLd) }))
|
|
1492
1257
|
.filter(x => x.nodes.length > 0 && !x.nodes.some(hasSchemaImage))
|
|
1493
1258
|
.map(x => ({ urlKey: x.urlKey, evidence: { note: 'no `image` on the Article schema — Discover cannot build a large image card from schema alone' } })),
|
|
1494
1259
|
},
|
|
1495
|
-
// NOTE: the Discover "front-loaded intro" angle is covered by the existing `answer-not-front-loaded`
|
|
1496
|
-
// check (title + first two body chunks, capped) — a dedicated discover-intro check duplicated it,
|
|
1497
|
-
// tested chunks[0] (which is the byline/author block on most themes) and triple-flagged one symptom.
|
|
1498
|
-
// Deliberately not added here; see plan/google-discover.md.
|
|
1499
1260
|
];
|
|
1500
|
-
// ── Google Discover helpers (article-eligibility + directive/schema parsing) ──
|
|
1501
|
-
// The article-family @types Discover surfaces. isDiscoverEligible gates every discover-* check.
|
|
1502
1261
|
const DISCOVER_TYPES = ['article', 'newsarticle', 'blogposting', 'liveblogposting', 'profilepage', 'reportagenewsarticle', 'opinionnewsarticle', 'reviewnewsarticle', 'techarticle', 'scholarlyarticle'];
|
|
1503
|
-
// Coarse SQL prefilter — narrows rows before the precise isDiscoverEligible() JSON parse (a slightly
|
|
1504
|
-
// wide net on the raw JSON, always confirmed in JS). Keeps every discover-* check off a full scan.
|
|
1505
1262
|
const DISCOVER_PREFILTER = `status_code=200 AND indexable=1 AND ${HTML_CT} AND (json_ld LIKE '%Article%' OR json_ld LIKE '%BlogPosting%' OR json_ld LIKE '%ProfilePage%' OR og_tags LIKE '%article%')`;
|
|
1506
|
-
// Every @type a node declares. A node's @type may be a STRING or an ARRAY (e.g. ["Article","BlogPosting"]);
|
|
1507
|
-
// templates.nodeType() returns only the first entry, which misreads array types — so read them all here.
|
|
1508
1263
|
function nodeTypeList(n) {
|
|
1509
1264
|
const t = n?.['@type'];
|
|
1510
1265
|
if (typeof t === 'string')
|
|
@@ -1513,7 +1268,6 @@ function nodeTypeList(n) {
|
|
|
1513
1268
|
return t.filter((x) => typeof x === 'string').map(x => x.toLowerCase());
|
|
1514
1269
|
return [];
|
|
1515
1270
|
}
|
|
1516
|
-
// JSON-LD nodes whose @type set intersects the article family (array-@type aware).
|
|
1517
1271
|
function discoverArticleNodes(jsonLd) {
|
|
1518
1272
|
return parseJsonLdNodes(jsonLd).filter(n => nodeTypeList(n).some(t => DISCOVER_TYPES.includes(t)));
|
|
1519
1273
|
}
|
|
@@ -1528,10 +1282,6 @@ function parseOg(ogTags) {
|
|
|
1528
1282
|
return {};
|
|
1529
1283
|
}
|
|
1530
1284
|
}
|
|
1531
|
-
// A page is Discover-eligible if its JSON-LD declares an article-family node. If it has typed JSON-LD
|
|
1532
|
-
// but NO article node, JSON-LD is AUTHORITATIVE → not eligible, even when og:type=article — Yoast/WP
|
|
1533
|
-
// stamps og:type=article on every non-front page (so /blog, /about, /contact would otherwise qualify).
|
|
1534
|
-
// og:type is the fallback ONLY when there is no typed JSON-LD to judge from.
|
|
1535
1285
|
function isDiscoverEligible(jsonLd, ogTags) {
|
|
1536
1286
|
if (discoverArticleNodes(jsonLd).length)
|
|
1537
1287
|
return true;
|
|
@@ -1539,17 +1289,13 @@ function isDiscoverEligible(jsonLd, ogTags) {
|
|
|
1539
1289
|
return false;
|
|
1540
1290
|
return (parseOg(ogTags)['og:type'] ?? '').toLowerCase() === 'article';
|
|
1541
1291
|
}
|
|
1542
|
-
// max-image-preview:large can be set via the robots meta OR the X-Robots-Tag header; either counts.
|
|
1543
1292
|
function hasMaxImagePreviewLarge(robots, xrt) {
|
|
1544
1293
|
return `${robots ?? ''} ${xrt ?? ''}`.toLowerCase().replace(/\s+/g, '').includes('max-image-preview:large');
|
|
1545
1294
|
}
|
|
1546
|
-
// A DELIBERATE preview restriction (none / standard / nosnippet) — the site chose to limit previews, so
|
|
1547
|
-
// "missing max-image-preview:large" is not an issue to raise (image-preview-restricted covers these).
|
|
1548
1295
|
function hasPreviewRestriction(robots, xrt) {
|
|
1549
1296
|
const s = `${robots ?? ''} ${xrt ?? ''}`.toLowerCase().replace(/\s+/g, '');
|
|
1550
1297
|
return s.includes('max-image-preview:none') || s.includes('max-image-preview:standard') || s.includes('nosnippet');
|
|
1551
1298
|
}
|
|
1552
|
-
// Does a schema node declare a usable image? (string URL, {url|contentUrl|@id} object, or array of either.)
|
|
1553
1299
|
function hasSchemaImage(node) {
|
|
1554
1300
|
const img = node?.image;
|
|
1555
1301
|
const one = (v) => typeof v === 'string' ? v.trim().length > 0
|
|
@@ -1567,7 +1313,6 @@ function hreflangFindings(ctx) {
|
|
|
1567
1313
|
const status = new Map();
|
|
1568
1314
|
for (const p of ctx.db.prepare(`SELECT url_key, status_code, noindex FROM pages`).all())
|
|
1569
1315
|
status.set(p.url_key, { status: p.status_code, noindex: p.noindex });
|
|
1570
|
-
// declared[sourceKey] = set of internal alternate targetKeys (excluding self)
|
|
1571
1316
|
const declared = new Map();
|
|
1572
1317
|
const parsed = [];
|
|
1573
1318
|
for (const p of pageRows) {
|
|
@@ -1590,13 +1335,10 @@ function hreflangFindings(ctx) {
|
|
|
1590
1335
|
continue;
|
|
1591
1336
|
}
|
|
1592
1337
|
if (key === p.url_key)
|
|
1593
|
-
continue;
|
|
1338
|
+
continue;
|
|
1594
1339
|
targets.push({ key, lang: (lang || '').toLowerCase() });
|
|
1595
1340
|
}
|
|
1596
1341
|
parsed.push({ srcKey: p.url_key, targets });
|
|
1597
|
-
// A return link via x-default is a VALID return tag (common when x-default is the
|
|
1598
|
-
// homepage) — include all targets here; x-default is only excluded as a reciprocation
|
|
1599
|
-
// *requirement* in the loop below, never as a way of satisfying one.
|
|
1600
1342
|
declared.set(p.url_key, new Set(targets.map(t => t.key)));
|
|
1601
1343
|
}
|
|
1602
1344
|
const broken = [];
|
|
@@ -1608,8 +1350,6 @@ function hreflangFindings(ctx) {
|
|
|
1608
1350
|
const st = status.get(t.key);
|
|
1609
1351
|
if (st && (st.status !== 200 || st.noindex === 1))
|
|
1610
1352
|
brokenTargets.push(t.key);
|
|
1611
|
-
// reciprocation only for internal, live (200) targets we crawled, excluding x-default
|
|
1612
|
-
// (a broken target is already reported by broken-hreflang-target — don't double-flag)
|
|
1613
1353
|
if (t.lang !== 'x-default' && st && st.status === 200 && st.noindex !== 1 && !(declared.get(t.key)?.has(srcKey)))
|
|
1614
1354
|
missingReturn.push(t.key);
|
|
1615
1355
|
}
|