blume 1.0.3 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +94 -0
- package/dist/cli/index.js +13784 -10579
- package/dist/cli/index.js.map +93 -61
- package/dist/types/core/config-input.d.ts +87 -8
- package/dist/types/core/data.d.ts +21 -0
- package/dist/types/core/deployment-env.d.ts +6 -0
- package/dist/types/core/diagnostics.d.ts +23 -0
- package/dist/types/core/i18n-ui.d.ts +140 -140
- package/dist/types/core/schema.d.ts +549 -370
- package/dist/types/core/sources/types.d.ts +3 -1
- package/dist/types/core/standard-schema.d.ts +41 -0
- package/dist/types/core/types.d.ts +23 -0
- package/dist/types/og/card.d.ts +63 -0
- package/dist/types/og/dimensions.d.ts +12 -0
- package/dist/types/openapi/references.d.ts +12 -7
- package/docs/01-quickstart.mdx +1 -1
- package/docs/02-deployment.mdx +9 -1
- package/docs/advanced/api-reference.mdx +22 -3
- package/docs/advanced/changelog.mdx +1 -1
- package/docs/advanced/skills.mdx +1 -1
- package/docs/configuration/ai.mdx +1 -1
- package/docs/configuration/customization.mdx +1 -1
- package/docs/configuration/export.mdx +1 -1
- package/docs/configuration/index.mdx +21 -1
- package/docs/configuration/search.mdx +28 -1
- package/docs/configuration/seo.mdx +40 -2
- package/docs/configuration/theming.mdx +1 -1
- package/docs/content/components.mdx +15 -2
- package/docs/content/index.mdx +1 -1
- package/docs/content/meta.mdx +1 -1
- package/docs/content/navigation.mdx +11 -1
- package/docs/content/sources.mdx +1 -1
- package/docs/content/syntax.mdx +116 -4
- package/docs/reference/cli.mdx +79 -1
- package/docs/reference/frontmatter.mdx +29 -1
- package/package.json +3 -3
- package/skills/blume-migrate/SKILL.md +170 -0
- package/skills/blume-migrate/assets/oxfmt@0.55.0.patch +20 -0
- package/skills/blume-migrate/references/docusaurus.md +95 -0
- package/skills/blume-migrate/references/fumadocs.md +95 -0
- package/skills/blume-migrate/references/mintlify.md +156 -0
- package/skills/blume-migrate/references/monorepo.md +224 -0
- package/skills/blume-migrate/references/nextra.md +76 -0
- package/skills/blume-migrate/references/starlight.md +116 -0
- package/skills/blume-migrate/scripts/mintlify-codemod.mjs +478 -0
- package/src/ai/llms.ts +15 -0
- package/src/astro/adapter-root.ts +70 -0
- package/src/astro/component-slots.ts +3 -2
- package/src/astro/generate.ts +132 -42
- package/src/astro/index.ts +1 -0
- package/src/astro/pages.ts +18 -3
- package/src/astro/templates.ts +158 -56
- package/src/audit/agent.ts +114 -0
- package/src/audit/catalog.ts +826 -0
- package/src/audit/checks/assets.ts +177 -0
- package/src/audit/checks/content.ts +231 -0
- package/src/audit/checks/duplicates.ts +131 -0
- package/src/audit/checks/i18n.ts +246 -0
- package/src/audit/checks/indexability.ts +213 -0
- package/src/audit/checks/links.ts +223 -0
- package/src/audit/checks/llms.ts +135 -0
- package/src/audit/checks/network.ts +272 -0
- package/src/audit/checks/og-image.ts +113 -0
- package/src/audit/checks/redirects.ts +87 -0
- package/src/audit/checks/robots.ts +114 -0
- package/src/audit/checks/sitemap.ts +229 -0
- package/src/audit/checks/social.ts +238 -0
- package/src/audit/crawl.ts +259 -0
- package/src/audit/graph.ts +74 -0
- package/src/audit/html.ts +54 -0
- package/src/audit/image-size.ts +63 -0
- package/src/audit/locate.ts +33 -0
- package/src/audit/redirects.ts +74 -0
- package/src/audit/report.ts +278 -0
- package/src/audit/run.ts +198 -0
- package/src/audit/snapshot.ts +189 -0
- package/src/audit/types.ts +214 -0
- package/src/audit/url.ts +103 -0
- package/src/cli/commands/audit.ts +205 -0
- package/src/cli/commands/build.ts +51 -12
- package/src/cli/index.ts +2 -0
- package/src/components/content/Callout.astro +8 -2
- package/src/components/content/Prompt.astro +25 -13
- package/src/components/content/Tabs.astro +98 -15
- package/src/components/layout/Breadcrumbs.astro +1 -1
- package/src/components/layout/Header.astro +5 -8
- package/src/components/layout/Logo.astro +13 -1
- package/src/components/layout/PageFeedback.astro +2 -2
- package/src/components/layout/PageLayout.astro +9 -9
- package/src/components/layout/Pagination.astro +7 -7
- package/src/components/layout/RootLayout.astro +9 -11
- package/src/components/layout/Search.astro +36 -7
- package/src/components/layout/TableOfContents.astro +1 -1
- package/src/components/layout/nav-utils.ts +9 -7
- package/src/components/openapi/Authorization.astro +80 -0
- package/src/components/openapi/Operation.astro +19 -1
- package/src/components/openapi/ParametersTable.astro +1 -1
- package/src/components/openapi/security.ts +201 -0
- package/src/components/openapi/snippets.ts +42 -13
- package/src/core/config-input.ts +94 -8
- package/src/core/data.ts +18 -2
- package/src/core/deployment-env.ts +9 -0
- package/src/core/diagnostics.ts +59 -12
- package/src/core/links.ts +2 -91
- package/src/core/nav-diagnostics.ts +48 -4
- package/src/core/navigation.ts +55 -13
- package/src/core/probe.ts +136 -0
- package/src/core/project-graph.ts +8 -0
- package/src/core/schema.ts +100 -1
- package/src/core/sources/normalize.ts +198 -25
- package/src/core/sources/types.ts +3 -1
- package/src/core/sources/watch.ts +5 -0
- package/src/core/standard-schema.ts +54 -0
- package/src/core/types.ts +23 -0
- package/src/deploy/adapter-output.ts +27 -15
- package/src/deploy/headers.ts +66 -0
- package/src/deploy/redirects.ts +49 -9
- package/src/markdown/index.ts +2 -0
- package/src/markdown/language-icon.ts +2 -1
- package/src/markdown/table-wrap.ts +43 -0
- package/src/og/card.ts +128 -36
- package/src/og/index.ts +1 -1
- package/src/og/logo.ts +21 -0
- package/src/openapi/references.ts +19 -16
- package/src/search/popular.ts +33 -0
- package/src/theme/entry.ts +56 -6
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
import type { Diagnostic } from "../../core/types.ts";
|
|
2
|
+
import { finding } from "../catalog.ts";
|
|
3
|
+
import type { CheckModule } from "../types.ts";
|
|
4
|
+
import { normalizePath } from "../url.ts";
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* Whether a robots.txt `Disallow` value covers a path. robots.txt matching is
|
|
8
|
+
* prefix-based, with `*` as a wildcard and `$` anchoring the end.
|
|
9
|
+
*/
|
|
10
|
+
export const disallowMatches = (rule: string, path: string): boolean => {
|
|
11
|
+
const anchored = rule.endsWith("$");
|
|
12
|
+
const pattern = anchored ? rule.slice(0, -1) : rule;
|
|
13
|
+
const parts = pattern.split("*");
|
|
14
|
+
|
|
15
|
+
let cursor = 0;
|
|
16
|
+
for (const [index, part] of parts.entries()) {
|
|
17
|
+
if (part === "") {
|
|
18
|
+
continue;
|
|
19
|
+
}
|
|
20
|
+
// The first segment is anchored to the start of the path (robots.txt rules
|
|
21
|
+
// are prefix matches); every later segment may appear anywhere after the
|
|
22
|
+
// previous one, which is what makes `*` a wildcard.
|
|
23
|
+
let at: number;
|
|
24
|
+
if (index === 0) {
|
|
25
|
+
at = path.startsWith(part) ? 0 : -1;
|
|
26
|
+
} else {
|
|
27
|
+
at = path.indexOf(part, cursor);
|
|
28
|
+
}
|
|
29
|
+
if (at === -1) {
|
|
30
|
+
return false;
|
|
31
|
+
}
|
|
32
|
+
cursor = at + part.length;
|
|
33
|
+
}
|
|
34
|
+
// A wildcard just before `$` (`/docs*$`) absorbs the rest of the path, so
|
|
35
|
+
// the anchor is already satisfied by any prefix match.
|
|
36
|
+
return anchored && !pattern.endsWith("*") ? cursor === path.length : true;
|
|
37
|
+
};
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* robots.txt: is it there, is it well-formed, does it point at the sitemap, and
|
|
41
|
+
* — the one that matters — does it block a page the sitemap is advertising?
|
|
42
|
+
*
|
|
43
|
+
* Ahrefs also tracks "robots.txt has too many redirects". A static host serves
|
|
44
|
+
* the file directly, so that is effectively unreachable here and isn't checked.
|
|
45
|
+
*/
|
|
46
|
+
export const robotsChecks: CheckModule = {
|
|
47
|
+
category: "robots",
|
|
48
|
+
run(context) {
|
|
49
|
+
const { robots } = context;
|
|
50
|
+
const { site } = context.project.config.deployment;
|
|
51
|
+
|
|
52
|
+
if (!context.project.config.seo.robots) {
|
|
53
|
+
return [];
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
if (!robots) {
|
|
57
|
+
return [
|
|
58
|
+
finding(
|
|
59
|
+
"BLUME_AUDIT_ROBOTS_MISSING",
|
|
60
|
+
{ url: "/robots.txt" },
|
|
61
|
+
"The build has no robots.txt."
|
|
62
|
+
),
|
|
63
|
+
];
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const found: Diagnostic[] = robots.invalid.map((line) =>
|
|
67
|
+
finding(
|
|
68
|
+
"BLUME_AUDIT_ROBOTS_INVALID",
|
|
69
|
+
{ file: robots.file, line: line.line, url: "/robots.txt" },
|
|
70
|
+
`robots.txt line ${line.line} is not a directive: "${line.text}"`
|
|
71
|
+
)
|
|
72
|
+
);
|
|
73
|
+
|
|
74
|
+
if (site && robots.sitemaps.length === 0) {
|
|
75
|
+
found.push(
|
|
76
|
+
finding(
|
|
77
|
+
"BLUME_AUDIT_ROBOTS_SITEMAP_MISSING",
|
|
78
|
+
{ file: robots.file, url: "/robots.txt" },
|
|
79
|
+
"robots.txt does not declare a Sitemap."
|
|
80
|
+
)
|
|
81
|
+
);
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// A page can't be both blocked from crawling and advertised for indexing.
|
|
85
|
+
// Checking the disallow rules against the sitemap (rather than against every
|
|
86
|
+
// built file) keeps this to the pages the site actually wants indexed.
|
|
87
|
+
for (const loc of context.sitemap?.urls ?? []) {
|
|
88
|
+
let pathname: string;
|
|
89
|
+
try {
|
|
90
|
+
({ pathname } = new URL(loc));
|
|
91
|
+
} catch {
|
|
92
|
+
continue;
|
|
93
|
+
}
|
|
94
|
+
// Match the pathname as served: robots.txt rules are literal prefixes,
|
|
95
|
+
// so `Disallow: /page/` must see the trailing slash to match.
|
|
96
|
+
const path = normalizePath(pathname);
|
|
97
|
+
const rule = robots.disallow.find((entry) =>
|
|
98
|
+
disallowMatches(entry, pathname)
|
|
99
|
+
);
|
|
100
|
+
if (rule) {
|
|
101
|
+
found.push(
|
|
102
|
+
finding(
|
|
103
|
+
"BLUME_AUDIT_ROBOTS_DISALLOWS_INDEXABLE",
|
|
104
|
+
{ file: robots.file, url: path },
|
|
105
|
+
`robots.txt "Disallow: ${rule}" blocks ${path}, which sitemap.xml advertises.`
|
|
106
|
+
)
|
|
107
|
+
);
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
return found;
|
|
112
|
+
},
|
|
113
|
+
tier: "static",
|
|
114
|
+
};
|
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
import { normalizeBasePath, stripBasePath } from "../../core/base-path.ts";
|
|
2
|
+
import type { Diagnostic } from "../../core/types.ts";
|
|
3
|
+
import { finding } from "../catalog.ts";
|
|
4
|
+
import { pageSite } from "../locate.ts";
|
|
5
|
+
import type { AuditContext, CheckModule } from "../types.ts";
|
|
6
|
+
import { normalizePath, siteOrigin } from "../url.ts";
|
|
7
|
+
|
|
8
|
+
const MAX_SITEMAP_BYTES = 50 * 1024 * 1024;
|
|
9
|
+
const MAX_SITEMAP_URLS = 50_000;
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Slack for future `<lastmod>` values. A date-only stamp written in a timezone
|
|
13
|
+
* ahead of UTC parses as "tomorrow" from behind it; a day of grace keeps that
|
|
14
|
+
* from being reported as a lie.
|
|
15
|
+
*/
|
|
16
|
+
const LASTMOD_SLACK_MS = 24 * 60 * 60 * 1000;
|
|
17
|
+
|
|
18
|
+
/** Error routes are never crawlable destinations, so they belong out of the sitemap. */
|
|
19
|
+
const ERROR_ROUTES = new Set(["/404", "/500"]);
|
|
20
|
+
|
|
21
|
+
/** Site paths listed in the sitemap, normalized for comparison against page URLs. */
|
|
22
|
+
const sitemapPaths = (context: AuditContext): Map<string, string> => {
|
|
23
|
+
const paths = new Map<string, string>();
|
|
24
|
+
for (const loc of context.sitemap?.urls ?? []) {
|
|
25
|
+
try {
|
|
26
|
+
paths.set(normalizePath(new URL(loc).pathname), loc);
|
|
27
|
+
} catch {
|
|
28
|
+
// A malformed <loc> is reported by SITEMAP_INVALID, not here.
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
return paths;
|
|
32
|
+
};
|
|
33
|
+
|
|
34
|
+
/** The path of a canonical URL, or null when it isn't parseable (CANONICAL_BAD_TARGET reports that). */
|
|
35
|
+
const canonicalPath = (canonical: string): string | null => {
|
|
36
|
+
try {
|
|
37
|
+
return normalizePath(new URL(canonical).pathname);
|
|
38
|
+
} catch {
|
|
39
|
+
return null;
|
|
40
|
+
}
|
|
41
|
+
};
|
|
42
|
+
|
|
43
|
+
/** Validate one `<loc>` against the build: does it exist, and is it indexable? */
|
|
44
|
+
const checkListedUrl = (
|
|
45
|
+
context: AuditContext,
|
|
46
|
+
loc: string,
|
|
47
|
+
origin: string | null,
|
|
48
|
+
file: string
|
|
49
|
+
): Diagnostic[] => {
|
|
50
|
+
let parsed: URL;
|
|
51
|
+
try {
|
|
52
|
+
parsed = new URL(loc);
|
|
53
|
+
} catch {
|
|
54
|
+
return [
|
|
55
|
+
finding(
|
|
56
|
+
"BLUME_AUDIT_SITEMAP_INVALID",
|
|
57
|
+
{ file, url: "/sitemap.xml" },
|
|
58
|
+
`sitemap.xml lists "${loc}", which is not a valid absolute URL.`
|
|
59
|
+
),
|
|
60
|
+
];
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
if (origin && parsed.origin !== origin) {
|
|
64
|
+
return [
|
|
65
|
+
finding(
|
|
66
|
+
"BLUME_AUDIT_SITEMAP_OUT_OF_SCOPE",
|
|
67
|
+
{ file, url: loc },
|
|
68
|
+
`sitemap.xml lists ${loc}, which is on another origin.`
|
|
69
|
+
),
|
|
70
|
+
];
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
// `<loc>`s carry the deployment base; page URLs (from the file tree) don't.
|
|
74
|
+
const path = normalizePath(
|
|
75
|
+
stripBasePath(
|
|
76
|
+
normalizeBasePath(context.project.config.deployment.base),
|
|
77
|
+
parsed.pathname
|
|
78
|
+
)
|
|
79
|
+
);
|
|
80
|
+
const page = context.byUrl.get(path);
|
|
81
|
+
if (!page) {
|
|
82
|
+
const redirect = context.redirects.find(
|
|
83
|
+
(entry) => normalizePath(entry.from) === path
|
|
84
|
+
);
|
|
85
|
+
return [
|
|
86
|
+
finding(
|
|
87
|
+
"BLUME_AUDIT_SITEMAP_BAD_URL",
|
|
88
|
+
{ file, url: loc },
|
|
89
|
+
redirect
|
|
90
|
+
? `sitemap.xml lists ${path}, which redirects to ${redirect.to}.`
|
|
91
|
+
: `sitemap.xml lists ${path}, which the build does not serve.`
|
|
92
|
+
),
|
|
93
|
+
];
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
const found: Diagnostic[] = [];
|
|
97
|
+
if (!page.indexable) {
|
|
98
|
+
found.push(
|
|
99
|
+
finding(
|
|
100
|
+
"BLUME_AUDIT_NOINDEX_IN_SITEMAP",
|
|
101
|
+
pageSite(context, page, ["noindex"]),
|
|
102
|
+
`${path} is in the sitemap but declares robots "${page.robots}".`
|
|
103
|
+
)
|
|
104
|
+
);
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
const canonical = page.canonical && canonicalPath(page.canonical);
|
|
108
|
+
if (canonical && canonical !== path) {
|
|
109
|
+
found.push(
|
|
110
|
+
finding(
|
|
111
|
+
"BLUME_AUDIT_NON_CANONICAL_IN_SITEMAP",
|
|
112
|
+
pageSite(context, page, ["seo", "canonical"]),
|
|
113
|
+
`${path} is in the sitemap but canonicalizes to ${canonical}.`
|
|
114
|
+
)
|
|
115
|
+
);
|
|
116
|
+
}
|
|
117
|
+
return found;
|
|
118
|
+
};
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* The sitemap, cross-checked against what was actually built.
|
|
122
|
+
*
|
|
123
|
+
* The highest-value check here is the one Ahrefs buries at info severity:
|
|
124
|
+
* `INDEXABLE_PAGE_NOT_IN_SITEMAP`. A page that a stray `draft: true` or
|
|
125
|
+
* `sidebar.hidden` quietly kept out of the sitemap is invisible to search, and
|
|
126
|
+
* nothing else in the toolchain tells you.
|
|
127
|
+
*/
|
|
128
|
+
export const sitemapChecks: CheckModule = {
|
|
129
|
+
category: "sitemap",
|
|
130
|
+
run(context) {
|
|
131
|
+
const { sitemap } = context;
|
|
132
|
+
const { site } = context.project.config.deployment;
|
|
133
|
+
|
|
134
|
+
// Without `deployment.site` Blume can't emit a sitemap at all (absolute URLs
|
|
135
|
+
// are required), and that's a config choice, not a defect. Stay quiet.
|
|
136
|
+
if (!(site && context.project.config.seo.sitemap)) {
|
|
137
|
+
return [];
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
if (!sitemap) {
|
|
141
|
+
return [
|
|
142
|
+
finding(
|
|
143
|
+
"BLUME_AUDIT_SITEMAP_INVALID",
|
|
144
|
+
{ url: "/sitemap.xml" },
|
|
145
|
+
"The build has no sitemap.xml.",
|
|
146
|
+
"Set `seo.sitemap: true` and `deployment.site` in blume.config.ts."
|
|
147
|
+
),
|
|
148
|
+
];
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
const found: Diagnostic[] = [];
|
|
152
|
+
if (sitemap.error) {
|
|
153
|
+
found.push(
|
|
154
|
+
finding(
|
|
155
|
+
"BLUME_AUDIT_SITEMAP_INVALID",
|
|
156
|
+
{ file: sitemap.file, url: "/sitemap.xml" },
|
|
157
|
+
`sitemap.xml is not a valid urlset: ${sitemap.error}`
|
|
158
|
+
)
|
|
159
|
+
);
|
|
160
|
+
return found;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
if (
|
|
164
|
+
sitemap.bytes > MAX_SITEMAP_BYTES ||
|
|
165
|
+
sitemap.urls.length > MAX_SITEMAP_URLS
|
|
166
|
+
) {
|
|
167
|
+
found.push(
|
|
168
|
+
finding(
|
|
169
|
+
"BLUME_AUDIT_SITEMAP_TOO_LARGE",
|
|
170
|
+
{ file: sitemap.file, url: "/sitemap.xml" },
|
|
171
|
+
`sitemap.xml holds ${sitemap.urls.length} URLs in ${Math.round(sitemap.bytes / 1024 / 1024)} MB.`
|
|
172
|
+
)
|
|
173
|
+
);
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
// A `<lastmod>` that lies — malformed, or claiming the future — teaches
|
|
177
|
+
// search engines to distrust every lastmod in the file, which throws away
|
|
178
|
+
// the recrawl-priority signal the field exists to provide.
|
|
179
|
+
for (const [loc, lastmod] of sitemap.lastmod ?? []) {
|
|
180
|
+
const time = Date.parse(lastmod);
|
|
181
|
+
if (Number.isNaN(time)) {
|
|
182
|
+
found.push(
|
|
183
|
+
finding(
|
|
184
|
+
"BLUME_AUDIT_SITEMAP_LASTMOD_INVALID",
|
|
185
|
+
{ file: sitemap.file, url: loc },
|
|
186
|
+
`sitemap.xml gives ${loc} a lastmod of "${lastmod}", which is not a valid W3C date.`
|
|
187
|
+
)
|
|
188
|
+
);
|
|
189
|
+
} else if (time > Date.now() + LASTMOD_SLACK_MS) {
|
|
190
|
+
found.push(
|
|
191
|
+
finding(
|
|
192
|
+
"BLUME_AUDIT_SITEMAP_LASTMOD_INVALID",
|
|
193
|
+
{ file: sitemap.file, url: loc },
|
|
194
|
+
`sitemap.xml gives ${loc} a lastmod of ${lastmod}, which is in the future.`
|
|
195
|
+
)
|
|
196
|
+
);
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
const origin = siteOrigin(site);
|
|
201
|
+
const listed = sitemapPaths(context);
|
|
202
|
+
|
|
203
|
+
for (const loc of sitemap.urls) {
|
|
204
|
+
found.push(...checkListedUrl(context, loc, origin, sitemap.file));
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
// The other direction: a page that was built, is indexable, and should be
|
|
208
|
+
// findable — but never made it into the sitemap.
|
|
209
|
+
for (const page of context.pages) {
|
|
210
|
+
if (
|
|
211
|
+
!page.indexable ||
|
|
212
|
+
ERROR_ROUTES.has(page.url) ||
|
|
213
|
+
listed.has(normalizePath(page.url))
|
|
214
|
+
) {
|
|
215
|
+
continue;
|
|
216
|
+
}
|
|
217
|
+
found.push(
|
|
218
|
+
finding(
|
|
219
|
+
"BLUME_AUDIT_INDEXABLE_PAGE_NOT_IN_SITEMAP",
|
|
220
|
+
pageSite(context, page),
|
|
221
|
+
`${page.url} is built and indexable but is not listed in sitemap.xml.`
|
|
222
|
+
)
|
|
223
|
+
);
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
return found;
|
|
227
|
+
},
|
|
228
|
+
tier: "static",
|
|
229
|
+
};
|
|
@@ -0,0 +1,238 @@
|
|
|
1
|
+
import type { Diagnostic } from "../../core/types.ts";
|
|
2
|
+
import { finding } from "../catalog.ts";
|
|
3
|
+
import { pageSite } from "../locate.ts";
|
|
4
|
+
import type { CheckModule, PageSnapshot } from "../types.ts";
|
|
5
|
+
import { normalizePath } from "../url.ts";
|
|
6
|
+
|
|
7
|
+
/** The Open Graph properties a share card is unusable without. */
|
|
8
|
+
const OG_REQUIRED = ["og:title", "og:type", "og:description"];
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* `og:url` must be an absolute URL, so Blume can only emit it once
|
|
12
|
+
* `deployment.site` is known. Requiring it on a site that hasn't set one would
|
|
13
|
+
* report the same missing config on every page; SITE_NOT_SET says it once.
|
|
14
|
+
*/
|
|
15
|
+
const ogRequired = (hasSite: boolean): string[] =>
|
|
16
|
+
hasSite ? [...OG_REQUIRED, "og:url"] : OG_REQUIRED;
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Open Graph and X (Twitter) card tags.
|
|
20
|
+
*
|
|
21
|
+
* Ahrefs splits each of these into "missing" and "incomplete". That's a
|
|
22
|
+
* distinction without a difference — a card with no tags and a card with half
|
|
23
|
+
* its tags both fail to render — so each is one check that names exactly which
|
|
24
|
+
* properties are absent.
|
|
25
|
+
*/
|
|
26
|
+
export const socialChecks: CheckModule = {
|
|
27
|
+
category: "social",
|
|
28
|
+
run(context) {
|
|
29
|
+
const found: Diagnostic[] = [];
|
|
30
|
+
const hasSite = Boolean(context.project.config.deployment.site);
|
|
31
|
+
const required = ogRequired(hasSite);
|
|
32
|
+
|
|
33
|
+
for (const page of context.pages) {
|
|
34
|
+
// Error pages are never shared, so their cards don't matter.
|
|
35
|
+
if (!page.indexable) {
|
|
36
|
+
continue;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
const missing = required.filter((property) => !page.og[property]);
|
|
40
|
+
if (missing.length > 0) {
|
|
41
|
+
found.push(
|
|
42
|
+
finding(
|
|
43
|
+
"BLUME_AUDIT_OG_INCOMPLETE",
|
|
44
|
+
pageSite(context, page, ["description"]),
|
|
45
|
+
missing.length === required.length
|
|
46
|
+
? "Page has no Open Graph tags."
|
|
47
|
+
: `Open Graph is missing ${missing.join(", ")}.`
|
|
48
|
+
)
|
|
49
|
+
);
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
// The generated OG card is served from an absolute URL, so it too depends
|
|
53
|
+
// on `deployment.site`; SITE_NOT_SET already covers that case.
|
|
54
|
+
if (hasSite && !page.og["og:image"]) {
|
|
55
|
+
found.push(
|
|
56
|
+
finding(
|
|
57
|
+
"BLUME_AUDIT_OG_IMAGE_MISSING",
|
|
58
|
+
pageSite(context, page, ["seo", "image"]),
|
|
59
|
+
"Page has no og:image — shares will render without a preview card."
|
|
60
|
+
)
|
|
61
|
+
);
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
const ogUrl = page.og["og:url"];
|
|
65
|
+
if (ogUrl && page.canonical && ogUrl !== page.canonical) {
|
|
66
|
+
found.push(
|
|
67
|
+
finding(
|
|
68
|
+
"BLUME_AUDIT_OG_URL_MISMATCH",
|
|
69
|
+
pageSite(context, page),
|
|
70
|
+
`og:url is ${ogUrl} but the canonical is ${page.canonical}.`
|
|
71
|
+
)
|
|
72
|
+
);
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
// X reads title/description/image from the Open Graph tags, so the only
|
|
76
|
+
// thing it can't infer is the card type and the account attribution.
|
|
77
|
+
if (!page.twitter["twitter:card"]) {
|
|
78
|
+
found.push(
|
|
79
|
+
finding(
|
|
80
|
+
"BLUME_AUDIT_TWITTER_CARD_INCOMPLETE",
|
|
81
|
+
pageSite(context, page),
|
|
82
|
+
"Page has no twitter:card — X will render a plain link, not a card."
|
|
83
|
+
)
|
|
84
|
+
);
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
return found;
|
|
88
|
+
},
|
|
89
|
+
tier: "static",
|
|
90
|
+
};
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* What's missing from one JSON-LD block.
|
|
94
|
+
*
|
|
95
|
+
* A block may either be a single node (`{@context, @type, …}`) or a `@graph`
|
|
96
|
+
* container (`{@context, @graph: [{@type, …}, …]}`), which is the shape Blume
|
|
97
|
+
* itself emits. In the container form `@context` is declared once at the root
|
|
98
|
+
* and each entry carries its own `@type` — so demanding `@type` on the root, or
|
|
99
|
+
* `@context` on each entry, would flag perfectly valid structured data.
|
|
100
|
+
*/
|
|
101
|
+
const jsonLdProblems = (node: unknown): string[] => {
|
|
102
|
+
if (typeof node !== "object" || node === null) {
|
|
103
|
+
return ["it is not an object"];
|
|
104
|
+
}
|
|
105
|
+
const record = node as Record<string, unknown>;
|
|
106
|
+
const problems: string[] = [];
|
|
107
|
+
if (!record["@context"]) {
|
|
108
|
+
problems.push("@context");
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
const graph = record["@graph"];
|
|
112
|
+
if (Array.isArray(graph)) {
|
|
113
|
+
const untyped = graph.filter(
|
|
114
|
+
(entry) =>
|
|
115
|
+
typeof entry !== "object" ||
|
|
116
|
+
entry === null ||
|
|
117
|
+
!(entry as Record<string, unknown>)["@type"]
|
|
118
|
+
).length;
|
|
119
|
+
if (untyped > 0) {
|
|
120
|
+
problems.push(`@type on ${untyped} of its ${graph.length} @graph nodes`);
|
|
121
|
+
}
|
|
122
|
+
} else if (!record["@type"]) {
|
|
123
|
+
problems.push("@type");
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
return problems;
|
|
127
|
+
};
|
|
128
|
+
|
|
129
|
+
/** Structured data. We validate what Blume emits, and don't pretend to do more. */
|
|
130
|
+
export const structuredDataChecks: CheckModule = {
|
|
131
|
+
category: "structured-data",
|
|
132
|
+
run(context) {
|
|
133
|
+
const found: Diagnostic[] = [];
|
|
134
|
+
for (const page of context.pages) {
|
|
135
|
+
for (const error of page.jsonldErrors) {
|
|
136
|
+
found.push(
|
|
137
|
+
finding(
|
|
138
|
+
"BLUME_AUDIT_JSONLD_INVALID",
|
|
139
|
+
pageSite(context, page),
|
|
140
|
+
`A JSON-LD block failed to parse: ${error}`
|
|
141
|
+
)
|
|
142
|
+
);
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
for (const node of page.jsonld) {
|
|
146
|
+
const problems = jsonLdProblems(node);
|
|
147
|
+
if (problems.length > 0) {
|
|
148
|
+
found.push(
|
|
149
|
+
finding(
|
|
150
|
+
"BLUME_AUDIT_JSONLD_INCOMPLETE",
|
|
151
|
+
pageSite(context, page),
|
|
152
|
+
`A JSON-LD block is missing ${problems.join(" and ")}.`
|
|
153
|
+
)
|
|
154
|
+
);
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
}
|
|
158
|
+
return found;
|
|
159
|
+
},
|
|
160
|
+
tier: "static",
|
|
161
|
+
};
|
|
162
|
+
|
|
163
|
+
/** What makes a slug untidy, with the human name for each offense. */
|
|
164
|
+
const URL_STYLE: { name: string; test: RegExp }[] = [
|
|
165
|
+
{ name: "uppercase letters", test: /[A-Z]/u },
|
|
166
|
+
{ name: "underscores", test: /_/u },
|
|
167
|
+
{ name: "spaces", test: /%20| /u },
|
|
168
|
+
];
|
|
169
|
+
|
|
170
|
+
/** A protocol-relative URL names a dotted host (`//cdn.example.com/x`). */
|
|
171
|
+
const DOTTED_HOST = /^\/\/[^/]*\./u;
|
|
172
|
+
|
|
173
|
+
/**
|
|
174
|
+
* Whether an href carries a doubled slash from a trailing-slash `basePath` /
|
|
175
|
+
* `deployment.base`. `//docs/x` is the telltale: a browser reads it as a
|
|
176
|
+
* protocol-relative URL with host `docs`, so the link silently leaves the site
|
|
177
|
+
* — and every link checker skips it as external. An interior `//` (`/docs//x`)
|
|
178
|
+
* is the same mistake composed mid-path.
|
|
179
|
+
*/
|
|
180
|
+
const doubledSlash = (href: string): boolean => {
|
|
181
|
+
if (href.startsWith("//")) {
|
|
182
|
+
return !DOTTED_HOST.test(href);
|
|
183
|
+
}
|
|
184
|
+
if (!href.startsWith("/")) {
|
|
185
|
+
return false;
|
|
186
|
+
}
|
|
187
|
+
// Only the path — a query param may legitimately carry a URL.
|
|
188
|
+
const [path] = href.split(/[?#]/u);
|
|
189
|
+
return (path ?? "").includes("//");
|
|
190
|
+
};
|
|
191
|
+
|
|
192
|
+
export const urlChecks: CheckModule = {
|
|
193
|
+
category: "links",
|
|
194
|
+
run(context) {
|
|
195
|
+
const found: Diagnostic[] = [];
|
|
196
|
+
/** Doubled hrefs are site-wide (a bad base) — one finding per target. */
|
|
197
|
+
const doubled = new Map<string, PageSnapshot>();
|
|
198
|
+
for (const page of context.pages) {
|
|
199
|
+
if (page.url.includes("//")) {
|
|
200
|
+
found.push(
|
|
201
|
+
finding(
|
|
202
|
+
"BLUME_AUDIT_DOUBLE_SLASH_URL",
|
|
203
|
+
pageSite(context, page),
|
|
204
|
+
`URL ${normalizePath(page.url)} contains a double slash.`
|
|
205
|
+
)
|
|
206
|
+
);
|
|
207
|
+
}
|
|
208
|
+
for (const link of page.links) {
|
|
209
|
+
const href = link.href.trim();
|
|
210
|
+
if (doubledSlash(href) && !doubled.has(href)) {
|
|
211
|
+
doubled.set(href, page);
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
const untidy = URL_STYLE.filter((style) => style.test.test(page.url));
|
|
216
|
+
if (untidy.length > 0) {
|
|
217
|
+
found.push(
|
|
218
|
+
finding(
|
|
219
|
+
"BLUME_AUDIT_URL_STYLE",
|
|
220
|
+
pageSite(context, page),
|
|
221
|
+
`URL ${page.url} contains ${untidy.map((style) => style.name).join(" and ")}.`
|
|
222
|
+
)
|
|
223
|
+
);
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
for (const [href, page] of doubled) {
|
|
227
|
+
found.push(
|
|
228
|
+
finding(
|
|
229
|
+
"BLUME_AUDIT_DOUBLE_SLASH_URL",
|
|
230
|
+
pageSite(context, page),
|
|
231
|
+
`Link to ${href} contains a double slash.`
|
|
232
|
+
)
|
|
233
|
+
);
|
|
234
|
+
}
|
|
235
|
+
return found;
|
|
236
|
+
},
|
|
237
|
+
tier: "static",
|
|
238
|
+
};
|