blume 1.0.3 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (126) hide show
  1. package/CHANGELOG.md +94 -0
  2. package/dist/cli/index.js +13784 -10579
  3. package/dist/cli/index.js.map +93 -61
  4. package/dist/types/core/config-input.d.ts +87 -8
  5. package/dist/types/core/data.d.ts +21 -0
  6. package/dist/types/core/deployment-env.d.ts +6 -0
  7. package/dist/types/core/diagnostics.d.ts +23 -0
  8. package/dist/types/core/i18n-ui.d.ts +140 -140
  9. package/dist/types/core/schema.d.ts +549 -370
  10. package/dist/types/core/sources/types.d.ts +3 -1
  11. package/dist/types/core/standard-schema.d.ts +41 -0
  12. package/dist/types/core/types.d.ts +23 -0
  13. package/dist/types/og/card.d.ts +63 -0
  14. package/dist/types/og/dimensions.d.ts +12 -0
  15. package/dist/types/openapi/references.d.ts +12 -7
  16. package/docs/01-quickstart.mdx +1 -1
  17. package/docs/02-deployment.mdx +9 -1
  18. package/docs/advanced/api-reference.mdx +22 -3
  19. package/docs/advanced/changelog.mdx +1 -1
  20. package/docs/advanced/skills.mdx +1 -1
  21. package/docs/configuration/ai.mdx +1 -1
  22. package/docs/configuration/customization.mdx +1 -1
  23. package/docs/configuration/export.mdx +1 -1
  24. package/docs/configuration/index.mdx +21 -1
  25. package/docs/configuration/search.mdx +28 -1
  26. package/docs/configuration/seo.mdx +40 -2
  27. package/docs/configuration/theming.mdx +1 -1
  28. package/docs/content/components.mdx +15 -2
  29. package/docs/content/index.mdx +1 -1
  30. package/docs/content/meta.mdx +1 -1
  31. package/docs/content/navigation.mdx +11 -1
  32. package/docs/content/sources.mdx +1 -1
  33. package/docs/content/syntax.mdx +116 -4
  34. package/docs/reference/cli.mdx +79 -1
  35. package/docs/reference/frontmatter.mdx +29 -1
  36. package/package.json +3 -3
  37. package/skills/blume-migrate/SKILL.md +170 -0
  38. package/skills/blume-migrate/assets/oxfmt@0.55.0.patch +20 -0
  39. package/skills/blume-migrate/references/docusaurus.md +95 -0
  40. package/skills/blume-migrate/references/fumadocs.md +95 -0
  41. package/skills/blume-migrate/references/mintlify.md +156 -0
  42. package/skills/blume-migrate/references/monorepo.md +224 -0
  43. package/skills/blume-migrate/references/nextra.md +76 -0
  44. package/skills/blume-migrate/references/starlight.md +116 -0
  45. package/skills/blume-migrate/scripts/mintlify-codemod.mjs +478 -0
  46. package/src/ai/llms.ts +15 -0
  47. package/src/astro/adapter-root.ts +70 -0
  48. package/src/astro/component-slots.ts +3 -2
  49. package/src/astro/generate.ts +132 -42
  50. package/src/astro/index.ts +1 -0
  51. package/src/astro/pages.ts +18 -3
  52. package/src/astro/templates.ts +158 -56
  53. package/src/audit/agent.ts +114 -0
  54. package/src/audit/catalog.ts +826 -0
  55. package/src/audit/checks/assets.ts +177 -0
  56. package/src/audit/checks/content.ts +231 -0
  57. package/src/audit/checks/duplicates.ts +131 -0
  58. package/src/audit/checks/i18n.ts +246 -0
  59. package/src/audit/checks/indexability.ts +213 -0
  60. package/src/audit/checks/links.ts +223 -0
  61. package/src/audit/checks/llms.ts +135 -0
  62. package/src/audit/checks/network.ts +272 -0
  63. package/src/audit/checks/og-image.ts +113 -0
  64. package/src/audit/checks/redirects.ts +87 -0
  65. package/src/audit/checks/robots.ts +114 -0
  66. package/src/audit/checks/sitemap.ts +229 -0
  67. package/src/audit/checks/social.ts +238 -0
  68. package/src/audit/crawl.ts +259 -0
  69. package/src/audit/graph.ts +74 -0
  70. package/src/audit/html.ts +54 -0
  71. package/src/audit/image-size.ts +63 -0
  72. package/src/audit/locate.ts +33 -0
  73. package/src/audit/redirects.ts +74 -0
  74. package/src/audit/report.ts +278 -0
  75. package/src/audit/run.ts +198 -0
  76. package/src/audit/snapshot.ts +189 -0
  77. package/src/audit/types.ts +214 -0
  78. package/src/audit/url.ts +103 -0
  79. package/src/cli/commands/audit.ts +205 -0
  80. package/src/cli/commands/build.ts +51 -12
  81. package/src/cli/index.ts +2 -0
  82. package/src/components/content/Callout.astro +8 -2
  83. package/src/components/content/Prompt.astro +25 -13
  84. package/src/components/content/Tabs.astro +98 -15
  85. package/src/components/layout/Breadcrumbs.astro +1 -1
  86. package/src/components/layout/Header.astro +5 -8
  87. package/src/components/layout/Logo.astro +13 -1
  88. package/src/components/layout/PageFeedback.astro +2 -2
  89. package/src/components/layout/PageLayout.astro +9 -9
  90. package/src/components/layout/Pagination.astro +7 -7
  91. package/src/components/layout/RootLayout.astro +9 -11
  92. package/src/components/layout/Search.astro +36 -7
  93. package/src/components/layout/TableOfContents.astro +1 -1
  94. package/src/components/layout/nav-utils.ts +9 -7
  95. package/src/components/openapi/Authorization.astro +80 -0
  96. package/src/components/openapi/Operation.astro +19 -1
  97. package/src/components/openapi/ParametersTable.astro +1 -1
  98. package/src/components/openapi/security.ts +201 -0
  99. package/src/components/openapi/snippets.ts +42 -13
  100. package/src/core/config-input.ts +94 -8
  101. package/src/core/data.ts +18 -2
  102. package/src/core/deployment-env.ts +9 -0
  103. package/src/core/diagnostics.ts +59 -12
  104. package/src/core/links.ts +2 -91
  105. package/src/core/nav-diagnostics.ts +48 -4
  106. package/src/core/navigation.ts +55 -13
  107. package/src/core/probe.ts +136 -0
  108. package/src/core/project-graph.ts +8 -0
  109. package/src/core/schema.ts +100 -1
  110. package/src/core/sources/normalize.ts +198 -25
  111. package/src/core/sources/types.ts +3 -1
  112. package/src/core/sources/watch.ts +5 -0
  113. package/src/core/standard-schema.ts +54 -0
  114. package/src/core/types.ts +23 -0
  115. package/src/deploy/adapter-output.ts +27 -15
  116. package/src/deploy/headers.ts +66 -0
  117. package/src/deploy/redirects.ts +49 -9
  118. package/src/markdown/index.ts +2 -0
  119. package/src/markdown/language-icon.ts +2 -1
  120. package/src/markdown/table-wrap.ts +43 -0
  121. package/src/og/card.ts +128 -36
  122. package/src/og/index.ts +1 -1
  123. package/src/og/logo.ts +21 -0
  124. package/src/openapi/references.ts +19 -16
  125. package/src/search/popular.ts +33 -0
  126. package/src/theme/entry.ts +56 -6
@@ -0,0 +1,114 @@
1
+ import type { Diagnostic } from "../../core/types.ts";
2
+ import { finding } from "../catalog.ts";
3
+ import type { CheckModule } from "../types.ts";
4
+ import { normalizePath } from "../url.ts";
5
+
6
+ /**
7
+ * Whether a robots.txt `Disallow` value covers a path. robots.txt matching is
8
+ * prefix-based, with `*` as a wildcard and `$` anchoring the end.
9
+ */
10
+ export const disallowMatches = (rule: string, path: string): boolean => {
11
+ const anchored = rule.endsWith("$");
12
+ const pattern = anchored ? rule.slice(0, -1) : rule;
13
+ const parts = pattern.split("*");
14
+
15
+ let cursor = 0;
16
+ for (const [index, part] of parts.entries()) {
17
+ if (part === "") {
18
+ continue;
19
+ }
20
+ // The first segment is anchored to the start of the path (robots.txt rules
21
+ // are prefix matches); every later segment may appear anywhere after the
22
+ // previous one, which is what makes `*` a wildcard.
23
+ let at: number;
24
+ if (index === 0) {
25
+ at = path.startsWith(part) ? 0 : -1;
26
+ } else {
27
+ at = path.indexOf(part, cursor);
28
+ }
29
+ if (at === -1) {
30
+ return false;
31
+ }
32
+ cursor = at + part.length;
33
+ }
34
+ // A wildcard just before `$` (`/docs*$`) absorbs the rest of the path, so
35
+ // the anchor is already satisfied by any prefix match.
36
+ return anchored && !pattern.endsWith("*") ? cursor === path.length : true;
37
+ };
38
+
39
+ /**
40
+ * robots.txt: is it there, is it well-formed, does it point at the sitemap, and
41
+ * — the one that matters — does it block a page the sitemap is advertising?
42
+ *
43
+ * Ahrefs also tracks "robots.txt has too many redirects". A static host serves
44
+ * the file directly, so that is effectively unreachable here and isn't checked.
45
+ */
46
+ export const robotsChecks: CheckModule = {
47
+ category: "robots",
48
+ run(context) {
49
+ const { robots } = context;
50
+ const { site } = context.project.config.deployment;
51
+
52
+ if (!context.project.config.seo.robots) {
53
+ return [];
54
+ }
55
+
56
+ if (!robots) {
57
+ return [
58
+ finding(
59
+ "BLUME_AUDIT_ROBOTS_MISSING",
60
+ { url: "/robots.txt" },
61
+ "The build has no robots.txt."
62
+ ),
63
+ ];
64
+ }
65
+
66
+ const found: Diagnostic[] = robots.invalid.map((line) =>
67
+ finding(
68
+ "BLUME_AUDIT_ROBOTS_INVALID",
69
+ { file: robots.file, line: line.line, url: "/robots.txt" },
70
+ `robots.txt line ${line.line} is not a directive: "${line.text}"`
71
+ )
72
+ );
73
+
74
+ if (site && robots.sitemaps.length === 0) {
75
+ found.push(
76
+ finding(
77
+ "BLUME_AUDIT_ROBOTS_SITEMAP_MISSING",
78
+ { file: robots.file, url: "/robots.txt" },
79
+ "robots.txt does not declare a Sitemap."
80
+ )
81
+ );
82
+ }
83
+
84
+ // A page can't be both blocked from crawling and advertised for indexing.
85
+ // Checking the disallow rules against the sitemap (rather than against every
86
+ // built file) keeps this to the pages the site actually wants indexed.
87
+ for (const loc of context.sitemap?.urls ?? []) {
88
+ let pathname: string;
89
+ try {
90
+ ({ pathname } = new URL(loc));
91
+ } catch {
92
+ continue;
93
+ }
94
+ // Match the pathname as served: robots.txt rules are literal prefixes,
95
+ // so `Disallow: /page/` must see the trailing slash to match.
96
+ const path = normalizePath(pathname);
97
+ const rule = robots.disallow.find((entry) =>
98
+ disallowMatches(entry, pathname)
99
+ );
100
+ if (rule) {
101
+ found.push(
102
+ finding(
103
+ "BLUME_AUDIT_ROBOTS_DISALLOWS_INDEXABLE",
104
+ { file: robots.file, url: path },
105
+ `robots.txt "Disallow: ${rule}" blocks ${path}, which sitemap.xml advertises.`
106
+ )
107
+ );
108
+ }
109
+ }
110
+
111
+ return found;
112
+ },
113
+ tier: "static",
114
+ };
@@ -0,0 +1,229 @@
1
+ import { normalizeBasePath, stripBasePath } from "../../core/base-path.ts";
2
+ import type { Diagnostic } from "../../core/types.ts";
3
+ import { finding } from "../catalog.ts";
4
+ import { pageSite } from "../locate.ts";
5
+ import type { AuditContext, CheckModule } from "../types.ts";
6
+ import { normalizePath, siteOrigin } from "../url.ts";
7
+
8
+ const MAX_SITEMAP_BYTES = 50 * 1024 * 1024;
9
+ const MAX_SITEMAP_URLS = 50_000;
10
+
11
+ /**
12
+ * Slack for future `<lastmod>` values. A date-only stamp written in a timezone
13
+ * ahead of UTC parses as "tomorrow" from behind it; a day of grace keeps that
14
+ * from being reported as a lie.
15
+ */
16
+ const LASTMOD_SLACK_MS = 24 * 60 * 60 * 1000;
17
+
18
+ /** Error routes are never crawlable destinations, so they belong out of the sitemap. */
19
+ const ERROR_ROUTES = new Set(["/404", "/500"]);
20
+
21
+ /** Site paths listed in the sitemap, normalized for comparison against page URLs. */
22
+ const sitemapPaths = (context: AuditContext): Map<string, string> => {
23
+ const paths = new Map<string, string>();
24
+ for (const loc of context.sitemap?.urls ?? []) {
25
+ try {
26
+ paths.set(normalizePath(new URL(loc).pathname), loc);
27
+ } catch {
28
+ // A malformed <loc> is reported by SITEMAP_INVALID, not here.
29
+ }
30
+ }
31
+ return paths;
32
+ };
33
+
34
+ /** The path of a canonical URL, or null when it isn't parseable (CANONICAL_BAD_TARGET reports that). */
35
+ const canonicalPath = (canonical: string): string | null => {
36
+ try {
37
+ return normalizePath(new URL(canonical).pathname);
38
+ } catch {
39
+ return null;
40
+ }
41
+ };
42
+
43
+ /** Validate one `<loc>` against the build: does it exist, and is it indexable? */
44
+ const checkListedUrl = (
45
+ context: AuditContext,
46
+ loc: string,
47
+ origin: string | null,
48
+ file: string
49
+ ): Diagnostic[] => {
50
+ let parsed: URL;
51
+ try {
52
+ parsed = new URL(loc);
53
+ } catch {
54
+ return [
55
+ finding(
56
+ "BLUME_AUDIT_SITEMAP_INVALID",
57
+ { file, url: "/sitemap.xml" },
58
+ `sitemap.xml lists "${loc}", which is not a valid absolute URL.`
59
+ ),
60
+ ];
61
+ }
62
+
63
+ if (origin && parsed.origin !== origin) {
64
+ return [
65
+ finding(
66
+ "BLUME_AUDIT_SITEMAP_OUT_OF_SCOPE",
67
+ { file, url: loc },
68
+ `sitemap.xml lists ${loc}, which is on another origin.`
69
+ ),
70
+ ];
71
+ }
72
+
73
+ // `<loc>`s carry the deployment base; page URLs (from the file tree) don't.
74
+ const path = normalizePath(
75
+ stripBasePath(
76
+ normalizeBasePath(context.project.config.deployment.base),
77
+ parsed.pathname
78
+ )
79
+ );
80
+ const page = context.byUrl.get(path);
81
+ if (!page) {
82
+ const redirect = context.redirects.find(
83
+ (entry) => normalizePath(entry.from) === path
84
+ );
85
+ return [
86
+ finding(
87
+ "BLUME_AUDIT_SITEMAP_BAD_URL",
88
+ { file, url: loc },
89
+ redirect
90
+ ? `sitemap.xml lists ${path}, which redirects to ${redirect.to}.`
91
+ : `sitemap.xml lists ${path}, which the build does not serve.`
92
+ ),
93
+ ];
94
+ }
95
+
96
+ const found: Diagnostic[] = [];
97
+ if (!page.indexable) {
98
+ found.push(
99
+ finding(
100
+ "BLUME_AUDIT_NOINDEX_IN_SITEMAP",
101
+ pageSite(context, page, ["noindex"]),
102
+ `${path} is in the sitemap but declares robots "${page.robots}".`
103
+ )
104
+ );
105
+ }
106
+
107
+ const canonical = page.canonical && canonicalPath(page.canonical);
108
+ if (canonical && canonical !== path) {
109
+ found.push(
110
+ finding(
111
+ "BLUME_AUDIT_NON_CANONICAL_IN_SITEMAP",
112
+ pageSite(context, page, ["seo", "canonical"]),
113
+ `${path} is in the sitemap but canonicalizes to ${canonical}.`
114
+ )
115
+ );
116
+ }
117
+ return found;
118
+ };
119
+
120
+ /**
121
+ * The sitemap, cross-checked against what was actually built.
122
+ *
123
+ * The highest-value check here is the one Ahrefs buries at info severity:
124
+ * `INDEXABLE_PAGE_NOT_IN_SITEMAP`. A page that a stray `draft: true` or
125
+ * `sidebar.hidden` quietly kept out of the sitemap is invisible to search, and
126
+ * nothing else in the toolchain tells you.
127
+ */
128
+ export const sitemapChecks: CheckModule = {
129
+ category: "sitemap",
130
+ run(context) {
131
+ const { sitemap } = context;
132
+ const { site } = context.project.config.deployment;
133
+
134
+ // Without `deployment.site` Blume can't emit a sitemap at all (absolute URLs
135
+ // are required), and that's a config choice, not a defect. Stay quiet.
136
+ if (!(site && context.project.config.seo.sitemap)) {
137
+ return [];
138
+ }
139
+
140
+ if (!sitemap) {
141
+ return [
142
+ finding(
143
+ "BLUME_AUDIT_SITEMAP_INVALID",
144
+ { url: "/sitemap.xml" },
145
+ "The build has no sitemap.xml.",
146
+ "Set `seo.sitemap: true` and `deployment.site` in blume.config.ts."
147
+ ),
148
+ ];
149
+ }
150
+
151
+ const found: Diagnostic[] = [];
152
+ if (sitemap.error) {
153
+ found.push(
154
+ finding(
155
+ "BLUME_AUDIT_SITEMAP_INVALID",
156
+ { file: sitemap.file, url: "/sitemap.xml" },
157
+ `sitemap.xml is not a valid urlset: ${sitemap.error}`
158
+ )
159
+ );
160
+ return found;
161
+ }
162
+
163
+ if (
164
+ sitemap.bytes > MAX_SITEMAP_BYTES ||
165
+ sitemap.urls.length > MAX_SITEMAP_URLS
166
+ ) {
167
+ found.push(
168
+ finding(
169
+ "BLUME_AUDIT_SITEMAP_TOO_LARGE",
170
+ { file: sitemap.file, url: "/sitemap.xml" },
171
+ `sitemap.xml holds ${sitemap.urls.length} URLs in ${Math.round(sitemap.bytes / 1024 / 1024)} MB.`
172
+ )
173
+ );
174
+ }
175
+
176
+ // A `<lastmod>` that lies — malformed, or claiming the future — teaches
177
+ // search engines to distrust every lastmod in the file, which throws away
178
+ // the recrawl-priority signal the field exists to provide.
179
+ for (const [loc, lastmod] of sitemap.lastmod ?? []) {
180
+ const time = Date.parse(lastmod);
181
+ if (Number.isNaN(time)) {
182
+ found.push(
183
+ finding(
184
+ "BLUME_AUDIT_SITEMAP_LASTMOD_INVALID",
185
+ { file: sitemap.file, url: loc },
186
+ `sitemap.xml gives ${loc} a lastmod of "${lastmod}", which is not a valid W3C date.`
187
+ )
188
+ );
189
+ } else if (time > Date.now() + LASTMOD_SLACK_MS) {
190
+ found.push(
191
+ finding(
192
+ "BLUME_AUDIT_SITEMAP_LASTMOD_INVALID",
193
+ { file: sitemap.file, url: loc },
194
+ `sitemap.xml gives ${loc} a lastmod of ${lastmod}, which is in the future.`
195
+ )
196
+ );
197
+ }
198
+ }
199
+
200
+ const origin = siteOrigin(site);
201
+ const listed = sitemapPaths(context);
202
+
203
+ for (const loc of sitemap.urls) {
204
+ found.push(...checkListedUrl(context, loc, origin, sitemap.file));
205
+ }
206
+
207
+ // The other direction: a page that was built, is indexable, and should be
208
+ // findable — but never made it into the sitemap.
209
+ for (const page of context.pages) {
210
+ if (
211
+ !page.indexable ||
212
+ ERROR_ROUTES.has(page.url) ||
213
+ listed.has(normalizePath(page.url))
214
+ ) {
215
+ continue;
216
+ }
217
+ found.push(
218
+ finding(
219
+ "BLUME_AUDIT_INDEXABLE_PAGE_NOT_IN_SITEMAP",
220
+ pageSite(context, page),
221
+ `${page.url} is built and indexable but is not listed in sitemap.xml.`
222
+ )
223
+ );
224
+ }
225
+
226
+ return found;
227
+ },
228
+ tier: "static",
229
+ };
@@ -0,0 +1,238 @@
1
+ import type { Diagnostic } from "../../core/types.ts";
2
+ import { finding } from "../catalog.ts";
3
+ import { pageSite } from "../locate.ts";
4
+ import type { CheckModule, PageSnapshot } from "../types.ts";
5
+ import { normalizePath } from "../url.ts";
6
+
7
+ /** The Open Graph properties a share card is unusable without. */
8
+ const OG_REQUIRED = ["og:title", "og:type", "og:description"];
9
+
10
+ /**
11
+ * `og:url` must be an absolute URL, so Blume can only emit it once
12
+ * `deployment.site` is known. Requiring it on a site that hasn't set one would
13
+ * report the same missing config on every page; SITE_NOT_SET says it once.
14
+ */
15
+ const ogRequired = (hasSite: boolean): string[] =>
16
+ hasSite ? [...OG_REQUIRED, "og:url"] : OG_REQUIRED;
17
+
18
+ /**
19
+ * Open Graph and X (Twitter) card tags.
20
+ *
21
+ * Ahrefs splits each of these into "missing" and "incomplete". That's a
22
+ * distinction without a difference — a card with no tags and a card with half
23
+ * its tags both fail to render — so each is one check that names exactly which
24
+ * properties are absent.
25
+ */
26
+ export const socialChecks: CheckModule = {
27
+ category: "social",
28
+ run(context) {
29
+ const found: Diagnostic[] = [];
30
+ const hasSite = Boolean(context.project.config.deployment.site);
31
+ const required = ogRequired(hasSite);
32
+
33
+ for (const page of context.pages) {
34
+ // Error pages are never shared, so their cards don't matter.
35
+ if (!page.indexable) {
36
+ continue;
37
+ }
38
+
39
+ const missing = required.filter((property) => !page.og[property]);
40
+ if (missing.length > 0) {
41
+ found.push(
42
+ finding(
43
+ "BLUME_AUDIT_OG_INCOMPLETE",
44
+ pageSite(context, page, ["description"]),
45
+ missing.length === required.length
46
+ ? "Page has no Open Graph tags."
47
+ : `Open Graph is missing ${missing.join(", ")}.`
48
+ )
49
+ );
50
+ }
51
+
52
+ // The generated OG card is served from an absolute URL, so it too depends
53
+ // on `deployment.site`; SITE_NOT_SET already covers that case.
54
+ if (hasSite && !page.og["og:image"]) {
55
+ found.push(
56
+ finding(
57
+ "BLUME_AUDIT_OG_IMAGE_MISSING",
58
+ pageSite(context, page, ["seo", "image"]),
59
+ "Page has no og:image — shares will render without a preview card."
60
+ )
61
+ );
62
+ }
63
+
64
+ const ogUrl = page.og["og:url"];
65
+ if (ogUrl && page.canonical && ogUrl !== page.canonical) {
66
+ found.push(
67
+ finding(
68
+ "BLUME_AUDIT_OG_URL_MISMATCH",
69
+ pageSite(context, page),
70
+ `og:url is ${ogUrl} but the canonical is ${page.canonical}.`
71
+ )
72
+ );
73
+ }
74
+
75
+ // X reads title/description/image from the Open Graph tags, so the only
76
+ // thing it can't infer is the card type and the account attribution.
77
+ if (!page.twitter["twitter:card"]) {
78
+ found.push(
79
+ finding(
80
+ "BLUME_AUDIT_TWITTER_CARD_INCOMPLETE",
81
+ pageSite(context, page),
82
+ "Page has no twitter:card — X will render a plain link, not a card."
83
+ )
84
+ );
85
+ }
86
+ }
87
+ return found;
88
+ },
89
+ tier: "static",
90
+ };
91
+
92
+ /**
93
+ * What's missing from one JSON-LD block.
94
+ *
95
+ * A block may either be a single node (`{@context, @type, …}`) or a `@graph`
96
+ * container (`{@context, @graph: [{@type, …}, …]}`), which is the shape Blume
97
+ * itself emits. In the container form `@context` is declared once at the root
98
+ * and each entry carries its own `@type` — so demanding `@type` on the root, or
99
+ * `@context` on each entry, would flag perfectly valid structured data.
100
+ */
101
+ const jsonLdProblems = (node: unknown): string[] => {
102
+ if (typeof node !== "object" || node === null) {
103
+ return ["it is not an object"];
104
+ }
105
+ const record = node as Record<string, unknown>;
106
+ const problems: string[] = [];
107
+ if (!record["@context"]) {
108
+ problems.push("@context");
109
+ }
110
+
111
+ const graph = record["@graph"];
112
+ if (Array.isArray(graph)) {
113
+ const untyped = graph.filter(
114
+ (entry) =>
115
+ typeof entry !== "object" ||
116
+ entry === null ||
117
+ !(entry as Record<string, unknown>)["@type"]
118
+ ).length;
119
+ if (untyped > 0) {
120
+ problems.push(`@type on ${untyped} of its ${graph.length} @graph nodes`);
121
+ }
122
+ } else if (!record["@type"]) {
123
+ problems.push("@type");
124
+ }
125
+
126
+ return problems;
127
+ };
128
+
129
+ /** Structured data. We validate what Blume emits, and don't pretend to do more. */
130
+ export const structuredDataChecks: CheckModule = {
131
+ category: "structured-data",
132
+ run(context) {
133
+ const found: Diagnostic[] = [];
134
+ for (const page of context.pages) {
135
+ for (const error of page.jsonldErrors) {
136
+ found.push(
137
+ finding(
138
+ "BLUME_AUDIT_JSONLD_INVALID",
139
+ pageSite(context, page),
140
+ `A JSON-LD block failed to parse: ${error}`
141
+ )
142
+ );
143
+ }
144
+
145
+ for (const node of page.jsonld) {
146
+ const problems = jsonLdProblems(node);
147
+ if (problems.length > 0) {
148
+ found.push(
149
+ finding(
150
+ "BLUME_AUDIT_JSONLD_INCOMPLETE",
151
+ pageSite(context, page),
152
+ `A JSON-LD block is missing ${problems.join(" and ")}.`
153
+ )
154
+ );
155
+ }
156
+ }
157
+ }
158
+ return found;
159
+ },
160
+ tier: "static",
161
+ };
162
+
163
+ /** What makes a slug untidy, with the human name for each offense. */
164
+ const URL_STYLE: { name: string; test: RegExp }[] = [
165
+ { name: "uppercase letters", test: /[A-Z]/u },
166
+ { name: "underscores", test: /_/u },
167
+ { name: "spaces", test: /%20| /u },
168
+ ];
169
+
170
+ /** A protocol-relative URL names a dotted host (`//cdn.example.com/x`). */
171
+ const DOTTED_HOST = /^\/\/[^/]*\./u;
172
+
173
+ /**
174
+ * Whether an href carries a doubled slash from a trailing-slash `basePath` /
175
+ * `deployment.base`. `//docs/x` is the telltale: a browser reads it as a
176
+ * protocol-relative URL with host `docs`, so the link silently leaves the site
177
+ * — and every link checker skips it as external. An interior `//` (`/docs//x`)
178
+ * is the same mistake composed mid-path.
179
+ */
180
+ const doubledSlash = (href: string): boolean => {
181
+ if (href.startsWith("//")) {
182
+ return !DOTTED_HOST.test(href);
183
+ }
184
+ if (!href.startsWith("/")) {
185
+ return false;
186
+ }
187
+ // Only the path — a query param may legitimately carry a URL.
188
+ const [path] = href.split(/[?#]/u);
189
+ return (path ?? "").includes("//");
190
+ };
191
+
192
+ export const urlChecks: CheckModule = {
193
+ category: "links",
194
+ run(context) {
195
+ const found: Diagnostic[] = [];
196
+ /** Doubled hrefs are site-wide (a bad base) — one finding per target. */
197
+ const doubled = new Map<string, PageSnapshot>();
198
+ for (const page of context.pages) {
199
+ if (page.url.includes("//")) {
200
+ found.push(
201
+ finding(
202
+ "BLUME_AUDIT_DOUBLE_SLASH_URL",
203
+ pageSite(context, page),
204
+ `URL ${normalizePath(page.url)} contains a double slash.`
205
+ )
206
+ );
207
+ }
208
+ for (const link of page.links) {
209
+ const href = link.href.trim();
210
+ if (doubledSlash(href) && !doubled.has(href)) {
211
+ doubled.set(href, page);
212
+ }
213
+ }
214
+
215
+ const untidy = URL_STYLE.filter((style) => style.test.test(page.url));
216
+ if (untidy.length > 0) {
217
+ found.push(
218
+ finding(
219
+ "BLUME_AUDIT_URL_STYLE",
220
+ pageSite(context, page),
221
+ `URL ${page.url} contains ${untidy.map((style) => style.name).join(" and ")}.`
222
+ )
223
+ );
224
+ }
225
+ }
226
+ for (const [href, page] of doubled) {
227
+ found.push(
228
+ finding(
229
+ "BLUME_AUDIT_DOUBLE_SLASH_URL",
230
+ pageSite(context, page),
231
+ `Link to ${href} contains a double slash.`
232
+ )
233
+ );
234
+ }
235
+ return found;
236
+ },
237
+ tier: "static",
238
+ };