blume 1.0.4 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/CHANGELOG.md +65 -0
  2. package/dist/cli/index.js +13255 -10232
  3. package/dist/cli/index.js.map +91 -60
  4. package/dist/types/core/config-input.d.ts +61 -1
  5. package/dist/types/core/data.d.ts +9 -0
  6. package/dist/types/core/deployment-env.d.ts +6 -0
  7. package/dist/types/core/diagnostics.d.ts +23 -0
  8. package/dist/types/core/i18n-ui.d.ts +8 -8
  9. package/dist/types/core/schema.d.ts +131 -22
  10. package/dist/types/core/sources/types.d.ts +3 -1
  11. package/dist/types/core/standard-schema.d.ts +41 -0
  12. package/dist/types/core/types.d.ts +13 -0
  13. package/dist/types/og/card.d.ts +63 -0
  14. package/dist/types/og/dimensions.d.ts +12 -0
  15. package/docs/01-quickstart.mdx +1 -1
  16. package/docs/02-deployment.mdx +9 -1
  17. package/docs/advanced/api-reference.mdx +11 -0
  18. package/docs/advanced/changelog.mdx +1 -1
  19. package/docs/advanced/skills.mdx +1 -1
  20. package/docs/configuration/ai.mdx +1 -1
  21. package/docs/configuration/customization.mdx +1 -1
  22. package/docs/configuration/export.mdx +1 -1
  23. package/docs/configuration/index.mdx +21 -1
  24. package/docs/configuration/search.mdx +28 -1
  25. package/docs/configuration/seo.mdx +21 -2
  26. package/docs/configuration/theming.mdx +1 -1
  27. package/docs/content/components.mdx +14 -0
  28. package/docs/content/index.mdx +1 -1
  29. package/docs/content/meta.mdx +1 -1
  30. package/docs/content/navigation.mdx +1 -1
  31. package/docs/content/sources.mdx +1 -1
  32. package/docs/reference/cli.mdx +79 -1
  33. package/docs/reference/frontmatter.mdx +29 -1
  34. package/package.json +3 -3
  35. package/skills/blume-migrate/SKILL.md +1 -1
  36. package/skills/blume-migrate/references/mintlify.md +3 -2
  37. package/skills/blume-migrate/scripts/mintlify-codemod.mjs +16 -4
  38. package/src/ai/llms.ts +15 -0
  39. package/src/astro/adapter-root.ts +70 -0
  40. package/src/astro/generate.ts +50 -19
  41. package/src/astro/index.ts +1 -0
  42. package/src/astro/pages.ts +18 -3
  43. package/src/astro/templates.ts +65 -22
  44. package/src/audit/agent.ts +114 -0
  45. package/src/audit/catalog.ts +826 -0
  46. package/src/audit/checks/assets.ts +177 -0
  47. package/src/audit/checks/content.ts +231 -0
  48. package/src/audit/checks/duplicates.ts +131 -0
  49. package/src/audit/checks/i18n.ts +246 -0
  50. package/src/audit/checks/indexability.ts +213 -0
  51. package/src/audit/checks/links.ts +223 -0
  52. package/src/audit/checks/llms.ts +135 -0
  53. package/src/audit/checks/network.ts +272 -0
  54. package/src/audit/checks/og-image.ts +113 -0
  55. package/src/audit/checks/redirects.ts +87 -0
  56. package/src/audit/checks/robots.ts +114 -0
  57. package/src/audit/checks/sitemap.ts +229 -0
  58. package/src/audit/checks/social.ts +238 -0
  59. package/src/audit/crawl.ts +259 -0
  60. package/src/audit/graph.ts +74 -0
  61. package/src/audit/html.ts +54 -0
  62. package/src/audit/image-size.ts +63 -0
  63. package/src/audit/locate.ts +33 -0
  64. package/src/audit/redirects.ts +74 -0
  65. package/src/audit/report.ts +278 -0
  66. package/src/audit/run.ts +198 -0
  67. package/src/audit/snapshot.ts +189 -0
  68. package/src/audit/types.ts +214 -0
  69. package/src/audit/url.ts +103 -0
  70. package/src/cli/commands/audit.ts +205 -0
  71. package/src/cli/commands/build.ts +51 -12
  72. package/src/cli/index.ts +2 -0
  73. package/src/components/content/Tabs.astro +98 -15
  74. package/src/components/layout/Breadcrumbs.astro +1 -1
  75. package/src/components/layout/Header.astro +1 -0
  76. package/src/components/layout/PageFeedback.astro +1 -1
  77. package/src/components/layout/PageLayout.astro +5 -1
  78. package/src/components/layout/Pagination.astro +1 -1
  79. package/src/components/layout/RootLayout.astro +5 -3
  80. package/src/components/layout/Search.astro +35 -6
  81. package/src/components/layout/TableOfContents.astro +1 -1
  82. package/src/components/openapi/Authorization.astro +80 -0
  83. package/src/components/openapi/Operation.astro +19 -1
  84. package/src/components/openapi/ParametersTable.astro +1 -1
  85. package/src/components/openapi/security.ts +201 -0
  86. package/src/components/openapi/snippets.ts +42 -13
  87. package/src/core/config-input.ts +66 -1
  88. package/src/core/data.ts +9 -1
  89. package/src/core/deployment-env.ts +9 -0
  90. package/src/core/diagnostics.ts +59 -12
  91. package/src/core/links.ts +2 -91
  92. package/src/core/nav-diagnostics.ts +48 -4
  93. package/src/core/probe.ts +136 -0
  94. package/src/core/project-graph.ts +8 -0
  95. package/src/core/schema.ts +86 -3
  96. package/src/core/sources/normalize.ts +198 -25
  97. package/src/core/sources/types.ts +3 -1
  98. package/src/core/standard-schema.ts +54 -0
  99. package/src/core/types.ts +13 -0
  100. package/src/deploy/adapter-output.ts +27 -15
  101. package/src/deploy/headers.ts +66 -0
  102. package/src/deploy/redirects.ts +49 -9
  103. package/src/og/card.ts +98 -33
  104. package/src/og/index.ts +1 -1
  105. package/src/search/popular.ts +33 -0
  106. package/src/theme/entry.ts +6 -1
@@ -0,0 +1,246 @@
1
+ import { normalizeBasePath, stripBasePath } from "../../core/base-path.ts";
2
+ import type { Diagnostic } from "../../core/types.ts";
3
+ import { finding } from "../catalog.ts";
4
+ import { pageSite } from "../locate.ts";
5
+ import type { AuditContext, CheckModule, PageSnapshot } from "../types.ts";
6
+ import { normalizePath } from "../url.ts";
7
+
8
+ /** The normalized `deployment.base` — hreflang hrefs carry it, page URLs don't. */
9
+ const deployBaseOf = (context: AuditContext): string =>
10
+ normalizeBasePath(context.project.config.deployment.base);
11
+
12
+ /** The hreflang value meaning "use this when nothing else matches". Not a locale. */
13
+ const X_DEFAULT = "x-default";
14
+
15
+ const isValidBcp47 = (tag: string): boolean => {
16
+ try {
17
+ Intl.getCanonicalLocales(tag);
18
+ return true;
19
+ } catch {
20
+ return false;
21
+ }
22
+ };
23
+
24
+ /** The site path an hreflang `href` points at, or null when it isn't parseable. */
25
+ const alternatePath = (href: string, deployBase: string): string | null => {
26
+ try {
27
+ return normalizePath(stripBasePath(deployBase, new URL(href).pathname));
28
+ } catch {
29
+ return null;
30
+ }
31
+ };
32
+
33
+ const langChecks = (
34
+ context: AuditContext,
35
+ page: PageSnapshot
36
+ ): Diagnostic[] => {
37
+ if (!page.lang) {
38
+ return [
39
+ finding(
40
+ "BLUME_AUDIT_HTML_LANG_MISSING",
41
+ pageSite(context, page),
42
+ "The <html> element has no lang attribute."
43
+ ),
44
+ ];
45
+ }
46
+ if (!isValidBcp47(page.lang)) {
47
+ return [
48
+ finding(
49
+ "BLUME_AUDIT_HTML_LANG_INVALID",
50
+ pageSite(context, page),
51
+ `"${page.lang}" is not a valid BCP 47 language tag.`
52
+ ),
53
+ ];
54
+ }
55
+ return [];
56
+ };
57
+
58
+ /**
59
+ * Every hreflang finding for one page: validity, self-reference, x-default, and
60
+ * whether each alternate points at a real, canonical page.
61
+ */
62
+ const hreflangChecks = (
63
+ context: AuditContext,
64
+ page: PageSnapshot
65
+ ): Diagnostic[] => {
66
+ const found: Diagnostic[] = [];
67
+ const seen = new Map<string, string[]>();
68
+ const deployBase = deployBaseOf(context);
69
+
70
+ for (const alternate of page.hreflang) {
71
+ if (alternate.lang !== X_DEFAULT && !isValidBcp47(alternate.lang)) {
72
+ found.push(
73
+ finding(
74
+ "BLUME_AUDIT_HREFLANG_INVALID",
75
+ pageSite(context, page),
76
+ `hreflang="${alternate.lang}" is not a valid BCP 47 language tag.`
77
+ )
78
+ );
79
+ continue;
80
+ }
81
+
82
+ const hrefs = seen.get(alternate.lang) ?? [];
83
+ hrefs.push(alternate.href);
84
+ seen.set(alternate.lang, hrefs);
85
+
86
+ const path = alternatePath(alternate.href, deployBase);
87
+ if (path === null) {
88
+ found.push(
89
+ finding(
90
+ "BLUME_AUDIT_HREFLANG_BAD_TARGET",
91
+ pageSite(context, page),
92
+ `hreflang="${alternate.lang}" points at "${alternate.href}", which is not an absolute URL.`
93
+ )
94
+ );
95
+ continue;
96
+ }
97
+
98
+ const target = context.byUrl.get(path);
99
+ const redirect = context.redirects.find(
100
+ (entry) => normalizePath(entry.from) === path
101
+ );
102
+ if (redirect) {
103
+ found.push(
104
+ finding(
105
+ "BLUME_AUDIT_HREFLANG_BAD_TARGET",
106
+ pageSite(context, page),
107
+ `hreflang="${alternate.lang}" points at ${path}, which redirects to ${redirect.to}.`
108
+ )
109
+ );
110
+ } else if (!target) {
111
+ found.push(
112
+ finding(
113
+ "BLUME_AUDIT_HREFLANG_BAD_TARGET",
114
+ pageSite(context, page),
115
+ `hreflang="${alternate.lang}" points at ${path}, which the build does not serve.`
116
+ )
117
+ );
118
+ } else if (
119
+ target.canonical &&
120
+ alternatePath(target.canonical, deployBase) !== normalizePath(target.url)
121
+ ) {
122
+ found.push(
123
+ finding(
124
+ "BLUME_AUDIT_HREFLANG_BAD_TARGET",
125
+ pageSite(context, page),
126
+ `hreflang="${alternate.lang}" points at ${path}, which canonicalizes elsewhere.`
127
+ )
128
+ );
129
+ }
130
+ }
131
+
132
+ // One language may name exactly one page.
133
+ for (const [lang, hrefs] of seen) {
134
+ if (hrefs.length > 1) {
135
+ found.push(
136
+ finding(
137
+ "BLUME_AUDIT_HREFLANG_CONFLICT",
138
+ pageSite(context, page),
139
+ `hreflang="${lang}" is declared ${hrefs.length} times, pointing at ${hrefs.join(", ")}.`
140
+ )
141
+ );
142
+ }
143
+ }
144
+
145
+ const self = page.hreflang.find(
146
+ (alternate) =>
147
+ alternatePath(alternate.href, deployBase) === normalizePath(page.url)
148
+ );
149
+ if (self) {
150
+ if (page.lang && self.lang !== X_DEFAULT && self.lang !== page.lang) {
151
+ found.push(
152
+ finding(
153
+ "BLUME_AUDIT_HREFLANG_LANG_MISMATCH",
154
+ pageSite(context, page),
155
+ `Page declares <html lang="${page.lang}"> but its own hreflang is "${self.lang}".`
156
+ )
157
+ );
158
+ }
159
+ } else {
160
+ found.push(
161
+ finding(
162
+ "BLUME_AUDIT_HREFLANG_SELF_MISSING",
163
+ pageSite(context, page),
164
+ "Page's hreflang set does not include a self-reference."
165
+ )
166
+ );
167
+ }
168
+
169
+ if (!page.hreflang.some((alternate) => alternate.lang === X_DEFAULT)) {
170
+ found.push(
171
+ finding(
172
+ "BLUME_AUDIT_HREFLANG_XDEFAULT_MISSING",
173
+ pageSite(context, page),
174
+ "Page's hreflang set has no x-default alternate."
175
+ )
176
+ );
177
+ }
178
+
179
+ return found;
180
+ };
181
+
182
+ /**
183
+ * Reciprocity: if A names B as its alternate, B must name A back. Google ignores
184
+ * a whole hreflang group when the return tags don't line up.
185
+ *
186
+ * This is the hardest check for a real crawler — it has to hold the entire site
187
+ * in memory and cross-reference it. We already do.
188
+ */
189
+ const returnTagChecks = (context: AuditContext): Diagnostic[] => {
190
+ const found: Diagnostic[] = [];
191
+ const deployBase = deployBaseOf(context);
192
+ for (const page of context.pages) {
193
+ for (const alternate of page.hreflang) {
194
+ if (alternate.lang === X_DEFAULT) {
195
+ continue;
196
+ }
197
+ const path = alternatePath(alternate.href, deployBase);
198
+ if (path === null || path === normalizePath(page.url)) {
199
+ continue;
200
+ }
201
+ const target = context.byUrl.get(path);
202
+ if (!target || target.hreflang.length === 0) {
203
+ // A missing or hreflang-less target is already reported as a bad target.
204
+ continue;
205
+ }
206
+ const returns = target.hreflang.some(
207
+ (back) =>
208
+ alternatePath(back.href, deployBase) === normalizePath(page.url)
209
+ );
210
+ if (!returns) {
211
+ found.push(
212
+ finding(
213
+ "BLUME_AUDIT_HREFLANG_NO_RETURN_TAG",
214
+ pageSite(context, page),
215
+ `Page names ${path} as its "${alternate.lang}" alternate, but ${path} does not name it back.`
216
+ )
217
+ );
218
+ }
219
+ }
220
+ }
221
+ return found;
222
+ };
223
+
224
+ /**
225
+ * Language declarations: `<html lang>` on every page, plus the full hreflang
226
+ * cluster on pages that have translations.
227
+ *
228
+ * The hreflang checks are gated on the page actually carrying hreflang tags
229
+ * rather than on `config.i18n`, so a monolingual site sees exactly zero of them
230
+ * — but still gets its `<html lang>` validated, which is a real bug on any site.
231
+ */
232
+ export const i18nChecks: CheckModule = {
233
+ category: "i18n",
234
+ run(context) {
235
+ const found: Diagnostic[] = [];
236
+ for (const page of context.pages) {
237
+ found.push(...langChecks(context, page));
238
+ if (page.hreflang.length > 0) {
239
+ found.push(...hreflangChecks(context, page));
240
+ }
241
+ }
242
+ found.push(...returnTagChecks(context));
243
+ return found;
244
+ },
245
+ tier: "static",
246
+ };
@@ -0,0 +1,213 @@
1
+ import { SITE_INFERRING_ADAPTERS } from "../../core/deployment-env.ts";
2
+ import type { Diagnostic } from "../../core/types.ts";
3
+ import { finding } from "../catalog.ts";
4
+ import { pageSite } from "../locate.ts";
5
+ import { ERROR_ROUTES } from "../types.ts";
6
+ import type { AuditContext, CheckModule, PageSnapshot } from "../types.ts";
7
+ import { normalizePath, siteOrigin } from "../url.ts";
8
+
9
+ /** The canonical URL parsed, or null when it isn't a usable absolute URL. */
10
+ const parseCanonical = (page: PageSnapshot): URL | null => {
11
+ if (!page.canonical) {
12
+ return null;
13
+ }
14
+ try {
15
+ return new URL(page.canonical);
16
+ } catch {
17
+ return null;
18
+ }
19
+ };
20
+
21
+ const canonicalChecks = (
22
+ context: AuditContext,
23
+ page: PageSnapshot
24
+ ): Diagnostic[] => {
25
+ const { site } = context.project.config.deployment;
26
+ const origin = siteOrigin(site);
27
+
28
+ // Error routes are noindex by design, so a canonical on them is meaningless
29
+ // and its absence isn't a defect — flagging /404 would be a guaranteed
30
+ // finding on every site that has an error page.
31
+ if (ERROR_ROUTES.has(page.url)) {
32
+ return [];
33
+ }
34
+
35
+ if (!page.canonical) {
36
+ // Without `deployment.site` Blume has no absolute URL to canonicalize to, so
37
+ // report the root cause once (in the sitemap/robots checks) rather than
38
+ // flagging every page for a config field they can't fix individually.
39
+ return site
40
+ ? [
41
+ finding(
42
+ "BLUME_AUDIT_CANONICAL_MISSING",
43
+ pageSite(context, page),
44
+ "Page has no canonical URL."
45
+ ),
46
+ ]
47
+ : [];
48
+ }
49
+
50
+ const canonical = parseCanonical(page);
51
+ if (!canonical) {
52
+ return [
53
+ finding(
54
+ "BLUME_AUDIT_CANONICAL_BAD_TARGET",
55
+ pageSite(context, page, ["seo", "canonical"]),
56
+ `Canonical "${page.canonical}" is not a valid absolute URL.`
57
+ ),
58
+ ];
59
+ }
60
+
61
+ const found: Diagnostic[] = [];
62
+ if (origin && canonical.origin !== origin) {
63
+ const sameHost = canonical.host === new URL(origin).host;
64
+ // A different *protocol* on the same host is a misconfiguration worth
65
+ // calling out; a different host entirely is a deliberate cross-site
66
+ // canonical, which is legitimate.
67
+ if (sameHost) {
68
+ found.push(
69
+ finding(
70
+ "BLUME_AUDIT_CANONICAL_PROTOCOL_MISMATCH",
71
+ pageSite(context, page, ["seo", "canonical"]),
72
+ `Canonical uses ${canonical.protocol}// but the site is ${new URL(origin).protocol}//.`
73
+ )
74
+ );
75
+ }
76
+ return found;
77
+ }
78
+
79
+ const target = normalizePath(canonical.pathname);
80
+ if (target === normalizePath(page.url)) {
81
+ return found;
82
+ }
83
+
84
+ // The canonical points at another page on this site. It must exist, and it
85
+ // must not itself redirect — a canonical to a redirect is a dead end.
86
+ const redirect = context.redirects.find(
87
+ (entry) => normalizePath(entry.from) === target
88
+ );
89
+ if (redirect) {
90
+ found.push(
91
+ finding(
92
+ "BLUME_AUDIT_CANONICAL_BAD_TARGET",
93
+ pageSite(context, page, ["seo", "canonical"]),
94
+ `Canonical points at ${target}, which is a redirect to ${redirect.to}.`
95
+ )
96
+ );
97
+ } else if (context.byUrl.has(target)) {
98
+ found.push(
99
+ finding(
100
+ "BLUME_AUDIT_CANONICAL_NOT_SELF",
101
+ pageSite(context, page, ["seo", "canonical"]),
102
+ `Page declares ${target} as its canonical, so it will not be indexed itself.`
103
+ )
104
+ );
105
+ } else {
106
+ found.push(
107
+ finding(
108
+ "BLUME_AUDIT_CANONICAL_BAD_TARGET",
109
+ pageSite(context, page, ["seo", "canonical"]),
110
+ `Canonical points at ${target}, which is not a page on this site.`
111
+ )
112
+ );
113
+ }
114
+ return found;
115
+ };
116
+
117
+ /**
118
+ * Whether and how a page can be indexed: its canonical, its robots meta, and
119
+ * whether Googlebot will read all of it.
120
+ */
121
+ export const indexabilityChecks: CheckModule = {
122
+ category: "indexability",
123
+ run(context) {
124
+ const found: Diagnostic[] = [];
125
+
126
+ // Without `deployment.site` Blume has no absolute URL to build from, so it
127
+ // cannot emit a canonical, an Open Graph image, or a sitemap on *any* page.
128
+ // That's one fact about the config, not a defect on each of 200 pages —
129
+ // report it once, and let the checks that depend on it stay quiet.
130
+ if (!context.project.config.deployment.site) {
131
+ const { adapter } = context.project.config.deployment;
132
+ const site = {
133
+ file: context.project.context.configFile ?? undefined,
134
+ url: "/",
135
+ };
136
+ // On a platform adapter the value arrives from the platform's env vars
137
+ // at deploy time (`applyDeploymentEnv`), so only this local artifact is
138
+ // missing it — the deployed site won't be. Hardcoding `deployment.site`
139
+ // would duplicate state the platform owns, so the finding (which agents
140
+ // apply verbatim via `--claude`/`--codex`) must not suggest it.
141
+ found.push(
142
+ adapter && SITE_INFERRING_ADAPTERS.has(adapter)
143
+ ? finding(
144
+ "BLUME_AUDIT_SITE_INFERRED_AT_DEPLOY",
145
+ site,
146
+ `deployment.site is not set in this build — on ${adapter} it is inferred from the platform env at deploy time, so canonical URLs, Open Graph images, and the sitemap are only missing from this local artifact.`
147
+ )
148
+ : finding(
149
+ "BLUME_AUDIT_SITE_NOT_SET",
150
+ site,
151
+ "deployment.site is not set, so no canonical URLs, Open Graph images, or sitemap can be generated."
152
+ )
153
+ );
154
+ }
155
+
156
+ for (const page of context.pages) {
157
+ found.push(...canonicalChecks(context, page));
158
+
159
+ // Error routes are *meant* to carry noindex, so reporting them would be a
160
+ // guaranteed finding on every site that has a 404 page. A robots meta
161
+ // that doesn't noindex (an ejected layout emitting `index, follow`)
162
+ // leaves the page indexable and isn't this finding.
163
+ if (page.robots && !page.indexable && !ERROR_ROUTES.has(page.url)) {
164
+ found.push(
165
+ finding(
166
+ "BLUME_AUDIT_ROBOTS_META_UNEXPECTED",
167
+ pageSite(context, page, ["noindex"]),
168
+ `Page declares robots "${page.robots}" and will not be indexed.`
169
+ )
170
+ );
171
+
172
+ // Google warns against pairing the two: the canonical says "index that
173
+ // URL", the robots meta says "don't trust this page's signals" — and
174
+ // which one wins is undefined.
175
+ if (!page.indexable && page.canonical) {
176
+ found.push(
177
+ finding(
178
+ "BLUME_AUDIT_CANONICAL_ON_NOINDEX",
179
+ pageSite(context, page, ["noindex"]),
180
+ `Page is noindex but declares ${page.canonical} as its canonical.`
181
+ )
182
+ );
183
+ }
184
+ }
185
+
186
+ // Only the manifest knows a page was a draft — the built HTML looks like
187
+ // any other page, which is exactly why deploying a `--preview` build is
188
+ // so easy to miss.
189
+ if (page.route?.draft) {
190
+ found.push(
191
+ finding(
192
+ "BLUME_AUDIT_DRAFT_PAGE_PUBLISHED",
193
+ pageSite(context, page, ["draft"]),
194
+ `${page.url} is marked draft in its front matter but is in the build.`
195
+ )
196
+ );
197
+ }
198
+
199
+ if (page.bytes > context.thresholds.maxHtmlBytes) {
200
+ const mb = (page.bytes / 1024 / 1024).toFixed(1);
201
+ found.push(
202
+ finding(
203
+ "BLUME_AUDIT_HTML_TOO_LARGE",
204
+ pageSite(context, page),
205
+ `Page HTML is ${mb} MB — past Googlebot's 2 MB limit, the rest is not crawled.`
206
+ )
207
+ );
208
+ }
209
+ }
210
+ return found;
211
+ },
212
+ tier: "static",
213
+ };
@@ -0,0 +1,223 @@
1
+ import { normalizeBasePath } from "../../core/base-path.ts";
2
+ import type { Diagnostic } from "../../core/types.ts";
3
+ import { finding } from "../catalog.ts";
4
+ import { orphanPages } from "../graph.ts";
5
+ import { pageSite } from "../locate.ts";
6
+ import type {
7
+ AuditContext,
8
+ CheckModule,
9
+ PageSnapshot,
10
+ SnapshotLink,
11
+ } from "../types.ts";
12
+ import { normalizePath, resolveHref, siteOrigin } from "../url.ts";
13
+
14
+ /** Whether a path is served by the build — as a page, or as a static file. */
15
+ const isServed = (context: AuditContext, path: string): boolean =>
16
+ context.byUrl.has(path) ||
17
+ context.files.has(path) ||
18
+ // Astro's directory format serves `/docs/api` from `/docs/api/index.html`.
19
+ context.files.has(`${path}/index.html`);
20
+
21
+ /** Browser-magic fragments that scroll without needing a matching id. */
22
+ const MAGIC_FRAGMENTS = new Set(["", "top"]);
23
+
24
+ /** Whether a fragment lands on an id of the target page. */
25
+ const anchorResolves = (target: PageSnapshot, fragment: string): boolean => {
26
+ if (MAGIC_FRAGMENTS.has(fragment)) {
27
+ return true;
28
+ }
29
+ // Hrefs usually carry the fragment percent-encoded while the HTML id is not;
30
+ // check both spellings so an encoded match isn't reported as broken.
31
+ try {
32
+ return (
33
+ target.ids.has(fragment) || target.ids.has(decodeURIComponent(fragment))
34
+ );
35
+ } catch {
36
+ return target.ids.has(fragment);
37
+ }
38
+ };
39
+
40
+ const redirectFrom = (context: AuditContext, path: string) =>
41
+ context.redirects.find((entry) => normalizePath(entry.from) === path);
42
+
43
+ /**
44
+ * Internal links, and what they land on.
45
+ *
46
+ * Broken *chrome* links are deduplicated by target: Blume renders the sidebar on
47
+ * every page, so one bad nav entry would otherwise be reported once per page —
48
+ * hundreds of findings for a single typo. Body links are reported per
49
+ * occurrence, because each one lives in a different `.mdx` a reader can go fix.
50
+ */
51
+ export const linkChecks: CheckModule = {
52
+ category: "links",
53
+ run(context) {
54
+ const found: Diagnostic[] = [];
55
+ const origin = siteOrigin(context.project.config.deployment.site);
56
+ const deployBase = normalizeBasePath(
57
+ context.project.config.deployment.base
58
+ );
59
+ /** Broken chrome targets, and the first page each was seen on. */
60
+ const brokenChrome = new Map<string, PageSnapshot>();
61
+ const redirectedChrome = new Map<string, PageSnapshot>();
62
+
63
+ /** Broken anchor targets in chrome (`url#frag`), first page each was seen on. */
64
+ const brokenChromeAnchors = new Map<string, PageSnapshot>();
65
+
66
+ /**
67
+ * A fragment must land on an id of the page it targets — a miss loads the
68
+ * page but silently dumps the reader at the top, which no crawler reports
69
+ * because the HTTP response is a healthy 200.
70
+ */
71
+ const checkAnchor = (
72
+ page: PageSnapshot,
73
+ link: SnapshotLink,
74
+ target: PageSnapshot,
75
+ fragment: string
76
+ ): void => {
77
+ if (anchorResolves(target, fragment)) {
78
+ return;
79
+ }
80
+ const where =
81
+ target === page ? "this page" : `${target.url}, which has no such id`;
82
+ if (link.content) {
83
+ found.push(
84
+ finding(
85
+ "BLUME_AUDIT_ANCHOR_BROKEN",
86
+ pageSite(context, page),
87
+ `Link to ${link.href} points at #${fragment} on ${where}.`
88
+ )
89
+ );
90
+ } else {
91
+ const key = `${target.url}#${fragment}`;
92
+ if (!brokenChromeAnchors.has(key)) {
93
+ brokenChromeAnchors.set(key, page);
94
+ }
95
+ }
96
+ };
97
+
98
+ const classify = (page: PageSnapshot, link: SnapshotLink): void => {
99
+ // Same-page anchors (`#setup`) never leave the page, so resolveHref
100
+ // files them under "ignored" — but their fragment still has to exist.
101
+ if (link.href.startsWith("#")) {
102
+ checkAnchor(page, link, page, link.href.slice(1));
103
+ return;
104
+ }
105
+
106
+ const resolved = resolveHref(page.url, link.href, origin, deployBase);
107
+ if (resolved.kind === "external" || resolved.kind === "ignored") {
108
+ return;
109
+ }
110
+
111
+ const anchorTarget = resolved.hash
112
+ ? context.byUrl.get(resolved.path)
113
+ : undefined;
114
+ if (anchorTarget) {
115
+ checkAnchor(page, link, anchorTarget, resolved.hash);
116
+ }
117
+
118
+ if (resolved.kind === "self-origin") {
119
+ found.push(
120
+ finding(
121
+ "BLUME_AUDIT_INTERNAL_LINK_ABSOLUTE",
122
+ pageSite(context, page),
123
+ `Link to ${link.href} hardcodes the site's origin; use ${resolved.path} instead.`
124
+ )
125
+ );
126
+ }
127
+
128
+ if (link.rel?.includes("nofollow")) {
129
+ found.push(
130
+ finding(
131
+ "BLUME_AUDIT_INTERNAL_LINK_NOFOLLOW",
132
+ pageSite(context, page),
133
+ `Internal link to ${resolved.path} is rel="nofollow".`
134
+ )
135
+ );
136
+ }
137
+
138
+ const { path } = resolved;
139
+ const redirect = redirectFrom(context, path);
140
+ if (redirect) {
141
+ if (link.content) {
142
+ found.push(
143
+ finding(
144
+ "BLUME_AUDIT_LINK_TO_REDIRECT",
145
+ pageSite(context, page),
146
+ `Link to ${path} goes through a redirect to ${redirect.to}.`
147
+ )
148
+ );
149
+ } else if (!redirectedChrome.has(path)) {
150
+ redirectedChrome.set(path, page);
151
+ }
152
+ return;
153
+ }
154
+
155
+ if (isServed(context, path)) {
156
+ return;
157
+ }
158
+
159
+ if (link.content) {
160
+ found.push(
161
+ finding(
162
+ "BLUME_AUDIT_LINK_TO_BROKEN",
163
+ pageSite(context, page),
164
+ `Link to ${link.href} resolves to ${path}, which the build does not serve.`
165
+ )
166
+ );
167
+ } else if (!brokenChrome.has(path)) {
168
+ brokenChrome.set(path, page);
169
+ }
170
+ };
171
+
172
+ for (const page of context.pages) {
173
+ for (const link of page.links) {
174
+ classify(page, link);
175
+ }
176
+ }
177
+
178
+ for (const [path, page] of brokenChrome) {
179
+ found.push(
180
+ finding(
181
+ "BLUME_AUDIT_LINK_TO_BROKEN",
182
+ { url: page.url },
183
+ `Navigation links to ${path}, which the build does not serve.`,
184
+ "Fix the entry in your navigation config or meta file."
185
+ )
186
+ );
187
+ }
188
+ for (const [path, page] of redirectedChrome) {
189
+ const redirect = redirectFrom(context, path);
190
+ found.push(
191
+ finding(
192
+ "BLUME_AUDIT_LINK_TO_REDIRECT",
193
+ { url: page.url },
194
+ `Navigation links to ${path}, which redirects to ${redirect?.to}.`,
195
+ "Point the navigation entry straight at the destination."
196
+ )
197
+ );
198
+ }
199
+ for (const [key, page] of brokenChromeAnchors) {
200
+ found.push(
201
+ finding(
202
+ "BLUME_AUDIT_ANCHOR_BROKEN",
203
+ { url: page.url },
204
+ `Navigation links to ${key}, but the target page has no such id.`
205
+ )
206
+ );
207
+ }
208
+
209
+ const homeUrl = normalizeBasePath(context.project.config.basePath) || "/";
210
+ for (const page of orphanPages(context.pages, context.graph, homeUrl)) {
211
+ found.push(
212
+ finding(
213
+ "BLUME_AUDIT_ORPHAN_PAGE",
214
+ pageSite(context, page),
215
+ "No other page's body links here — it is reachable only from the sidebar."
216
+ )
217
+ );
218
+ }
219
+
220
+ return found;
221
+ },
222
+ tier: "static",
223
+ };