blume 1.0.4 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +65 -0
- package/dist/cli/index.js +13255 -10232
- package/dist/cli/index.js.map +91 -60
- package/dist/types/core/config-input.d.ts +61 -1
- package/dist/types/core/data.d.ts +9 -0
- package/dist/types/core/deployment-env.d.ts +6 -0
- package/dist/types/core/diagnostics.d.ts +23 -0
- package/dist/types/core/i18n-ui.d.ts +8 -8
- package/dist/types/core/schema.d.ts +131 -22
- package/dist/types/core/sources/types.d.ts +3 -1
- package/dist/types/core/standard-schema.d.ts +41 -0
- package/dist/types/core/types.d.ts +13 -0
- package/dist/types/og/card.d.ts +63 -0
- package/dist/types/og/dimensions.d.ts +12 -0
- package/docs/01-quickstart.mdx +1 -1
- package/docs/02-deployment.mdx +9 -1
- package/docs/advanced/api-reference.mdx +11 -0
- package/docs/advanced/changelog.mdx +1 -1
- package/docs/advanced/skills.mdx +1 -1
- package/docs/configuration/ai.mdx +1 -1
- package/docs/configuration/customization.mdx +1 -1
- package/docs/configuration/export.mdx +1 -1
- package/docs/configuration/index.mdx +21 -1
- package/docs/configuration/search.mdx +28 -1
- package/docs/configuration/seo.mdx +21 -2
- package/docs/configuration/theming.mdx +1 -1
- package/docs/content/components.mdx +14 -0
- package/docs/content/index.mdx +1 -1
- package/docs/content/meta.mdx +1 -1
- package/docs/content/navigation.mdx +1 -1
- package/docs/content/sources.mdx +1 -1
- package/docs/reference/cli.mdx +79 -1
- package/docs/reference/frontmatter.mdx +29 -1
- package/package.json +3 -3
- package/skills/blume-migrate/SKILL.md +1 -1
- package/skills/blume-migrate/references/mintlify.md +3 -2
- package/skills/blume-migrate/scripts/mintlify-codemod.mjs +16 -4
- package/src/ai/llms.ts +15 -0
- package/src/astro/adapter-root.ts +70 -0
- package/src/astro/generate.ts +50 -19
- package/src/astro/index.ts +1 -0
- package/src/astro/pages.ts +18 -3
- package/src/astro/templates.ts +65 -22
- package/src/audit/agent.ts +114 -0
- package/src/audit/catalog.ts +826 -0
- package/src/audit/checks/assets.ts +177 -0
- package/src/audit/checks/content.ts +231 -0
- package/src/audit/checks/duplicates.ts +131 -0
- package/src/audit/checks/i18n.ts +246 -0
- package/src/audit/checks/indexability.ts +213 -0
- package/src/audit/checks/links.ts +223 -0
- package/src/audit/checks/llms.ts +135 -0
- package/src/audit/checks/network.ts +272 -0
- package/src/audit/checks/og-image.ts +113 -0
- package/src/audit/checks/redirects.ts +87 -0
- package/src/audit/checks/robots.ts +114 -0
- package/src/audit/checks/sitemap.ts +229 -0
- package/src/audit/checks/social.ts +238 -0
- package/src/audit/crawl.ts +259 -0
- package/src/audit/graph.ts +74 -0
- package/src/audit/html.ts +54 -0
- package/src/audit/image-size.ts +63 -0
- package/src/audit/locate.ts +33 -0
- package/src/audit/redirects.ts +74 -0
- package/src/audit/report.ts +278 -0
- package/src/audit/run.ts +198 -0
- package/src/audit/snapshot.ts +189 -0
- package/src/audit/types.ts +214 -0
- package/src/audit/url.ts +103 -0
- package/src/cli/commands/audit.ts +205 -0
- package/src/cli/commands/build.ts +51 -12
- package/src/cli/index.ts +2 -0
- package/src/components/content/Tabs.astro +98 -15
- package/src/components/layout/Breadcrumbs.astro +1 -1
- package/src/components/layout/Header.astro +1 -0
- package/src/components/layout/PageFeedback.astro +1 -1
- package/src/components/layout/PageLayout.astro +5 -1
- package/src/components/layout/Pagination.astro +1 -1
- package/src/components/layout/RootLayout.astro +5 -3
- package/src/components/layout/Search.astro +35 -6
- package/src/components/layout/TableOfContents.astro +1 -1
- package/src/components/openapi/Authorization.astro +80 -0
- package/src/components/openapi/Operation.astro +19 -1
- package/src/components/openapi/ParametersTable.astro +1 -1
- package/src/components/openapi/security.ts +201 -0
- package/src/components/openapi/snippets.ts +42 -13
- package/src/core/config-input.ts +66 -1
- package/src/core/data.ts +9 -1
- package/src/core/deployment-env.ts +9 -0
- package/src/core/diagnostics.ts +59 -12
- package/src/core/links.ts +2 -91
- package/src/core/nav-diagnostics.ts +48 -4
- package/src/core/probe.ts +136 -0
- package/src/core/project-graph.ts +8 -0
- package/src/core/schema.ts +86 -3
- package/src/core/sources/normalize.ts +198 -25
- package/src/core/sources/types.ts +3 -1
- package/src/core/standard-schema.ts +54 -0
- package/src/core/types.ts +13 -0
- package/src/deploy/adapter-output.ts +27 -15
- package/src/deploy/headers.ts +66 -0
- package/src/deploy/redirects.ts +49 -9
- package/src/og/card.ts +98 -33
- package/src/og/index.ts +1 -1
- package/src/search/popular.ts +33 -0
- package/src/theme/entry.ts +6 -1
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
import { normalizeBasePath, stripBasePath } from "../../core/base-path.ts";
|
|
2
|
+
import type { Diagnostic } from "../../core/types.ts";
|
|
3
|
+
import { finding } from "../catalog.ts";
|
|
4
|
+
import { pageSite } from "../locate.ts";
|
|
5
|
+
import { ERROR_ROUTES } from "../types.ts";
|
|
6
|
+
import type { AuditContext, CheckModule } from "../types.ts";
|
|
7
|
+
import { normalizePath, siteOrigin } from "../url.ts";
|
|
8
|
+
|
|
9
|
+
/** The `ai.llmsTxt` config normalized to what the checks need. */
|
|
10
|
+
const llmsConfig = (
|
|
11
|
+
context: AuditContext
|
|
12
|
+
): { enabled: boolean; openapi: boolean } => {
|
|
13
|
+
const value = context.project.config.ai?.llmsTxt;
|
|
14
|
+
if (typeof value === "object" && value !== null) {
|
|
15
|
+
return { enabled: value.enabled, openapi: value.openapi };
|
|
16
|
+
}
|
|
17
|
+
return { enabled: value !== false, openapi: true };
|
|
18
|
+
};
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* An llms.txt link target reduced to a site path, or null when it's off-site.
|
|
22
|
+
* Entry URLs carry the deployment base; page URLs (from the file tree) don't.
|
|
23
|
+
*/
|
|
24
|
+
const entryPath = (
|
|
25
|
+
url: string,
|
|
26
|
+
origin: string | null,
|
|
27
|
+
deployBase: string
|
|
28
|
+
): string | null => {
|
|
29
|
+
if (/^https?:\/\//iu.test(url)) {
|
|
30
|
+
try {
|
|
31
|
+
const parsed = new URL(url);
|
|
32
|
+
if (origin && parsed.origin !== origin) {
|
|
33
|
+
return null;
|
|
34
|
+
}
|
|
35
|
+
return normalizePath(
|
|
36
|
+
stripBasePath(deployBase, decodeURI(parsed.pathname))
|
|
37
|
+
);
|
|
38
|
+
} catch {
|
|
39
|
+
return null;
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
if (!url.startsWith("/")) {
|
|
43
|
+
return null;
|
|
44
|
+
}
|
|
45
|
+
try {
|
|
46
|
+
return normalizePath(stripBasePath(deployBase, decodeURI(url)));
|
|
47
|
+
} catch {
|
|
48
|
+
return normalizePath(stripBasePath(deployBase, url));
|
|
49
|
+
}
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* The `llms.txt` index, held to the sitemap's standard.
|
|
54
|
+
*
|
|
55
|
+
* SEO crawlers audit for Google and stop there. But Blume's promise is
|
|
56
|
+
* AI-ready docs, and `llms.txt` is the sitemap of that surface: a stale entry
|
|
57
|
+
* sends an agent to a page that is not there, and an unlisted page is
|
|
58
|
+
* invisible to every tool that starts from the index.
|
|
59
|
+
*/
|
|
60
|
+
export const llmsChecks: CheckModule = {
|
|
61
|
+
category: "ai",
|
|
62
|
+
run(context) {
|
|
63
|
+
const { enabled, openapi } = llmsConfig(context);
|
|
64
|
+
if (!enabled) {
|
|
65
|
+
return [];
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const { llms } = context;
|
|
69
|
+
if (!llms) {
|
|
70
|
+
return [
|
|
71
|
+
finding(
|
|
72
|
+
"BLUME_AUDIT_LLMS_TXT_MISSING",
|
|
73
|
+
{ url: "/llms.txt" },
|
|
74
|
+
"The build has no llms.txt."
|
|
75
|
+
),
|
|
76
|
+
];
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
const found: Diagnostic[] = [];
|
|
80
|
+
const origin = siteOrigin(context.project.config.deployment.site);
|
|
81
|
+
const deployBase = normalizeBasePath(
|
|
82
|
+
context.project.config.deployment.base
|
|
83
|
+
);
|
|
84
|
+
|
|
85
|
+
const listed = new Set<string>();
|
|
86
|
+
for (const entry of llms.entries) {
|
|
87
|
+
const path = entryPath(entry.url, origin, deployBase);
|
|
88
|
+
if (path === null) {
|
|
89
|
+
// Off-site or unparseable targets aren't pages this build can vouch
|
|
90
|
+
// for; external link health is the `--external` tier's job.
|
|
91
|
+
continue;
|
|
92
|
+
}
|
|
93
|
+
listed.add(path);
|
|
94
|
+
if (!context.byUrl.has(path)) {
|
|
95
|
+
found.push(
|
|
96
|
+
finding(
|
|
97
|
+
"BLUME_AUDIT_LLMS_TXT_STALE_ENTRY",
|
|
98
|
+
{ file: llms.file, line: entry.line, url: entry.url },
|
|
99
|
+
`llms.txt lists ${path}, which the build does not serve.`
|
|
100
|
+
)
|
|
101
|
+
);
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// The reverse direction, mirroring INDEXABLE_PAGE_NOT_IN_SITEMAP: a page
|
|
106
|
+
// that is built, indexable, and in the nav belongs in the index. Pages the
|
|
107
|
+
// generator deliberately skips — hidden, drafts, error routes, API
|
|
108
|
+
// reference when `ai.llmsTxt.openapi` is off, custom pages with no
|
|
109
|
+
// manifest route — are skipped here for the same reasons.
|
|
110
|
+
for (const page of context.pages) {
|
|
111
|
+
const { route } = page;
|
|
112
|
+
if (
|
|
113
|
+
!route ||
|
|
114
|
+
route.hidden ||
|
|
115
|
+
route.draft ||
|
|
116
|
+
!page.indexable ||
|
|
117
|
+
ERROR_ROUTES.has(page.url) ||
|
|
118
|
+
(!openapi && route.source.name === "openapi") ||
|
|
119
|
+
listed.has(normalizePath(page.url))
|
|
120
|
+
) {
|
|
121
|
+
continue;
|
|
122
|
+
}
|
|
123
|
+
found.push(
|
|
124
|
+
finding(
|
|
125
|
+
"BLUME_AUDIT_LLMS_TXT_PAGE_MISSING",
|
|
126
|
+
pageSite(context, page),
|
|
127
|
+
`${page.url} is built and indexable but is not listed in llms.txt.`
|
|
128
|
+
)
|
|
129
|
+
);
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
return found;
|
|
133
|
+
},
|
|
134
|
+
tier: "static",
|
|
135
|
+
};
|
|
@@ -0,0 +1,272 @@
|
|
|
1
|
+
import { normalizeBasePath } from "../../core/base-path.ts";
|
|
2
|
+
import { gradeExternal, probeAll } from "../../core/probe.ts";
|
|
3
|
+
import type { ProbeResult } from "../../core/probe.ts";
|
|
4
|
+
import type { Diagnostic } from "../../core/types.ts";
|
|
5
|
+
import { finding } from "../catalog.ts";
|
|
6
|
+
import { pageSite } from "../locate.ts";
|
|
7
|
+
import type { AuditContext, CheckModule, PageSnapshot } from "../types.ts";
|
|
8
|
+
import { resolveHref, siteOrigin } from "../url.ts";
|
|
9
|
+
|
|
10
|
+
const CLIENT_ERROR = 400;
|
|
11
|
+
const SERVER_ERROR = 500;
|
|
12
|
+
/** Past this, a page is slow enough that it costs you crawl budget and readers. */
|
|
13
|
+
const SLOW_MS = 1500;
|
|
14
|
+
|
|
15
|
+
/** The live URL a built page is served at, under the `--url` origin. */
|
|
16
|
+
const liveUrl = (origin: string, page: PageSnapshot): string =>
|
|
17
|
+
new URL(page.url, origin).toString();
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Whether the response failed outright, and how. Null when the page is served.
|
|
21
|
+
*
|
|
22
|
+
* Exported so the grading can be unit-tested against synthetic responses: a
|
|
23
|
+
* timeout, an HTTPS-to-HTTP downgrade, and a slow first byte are all awkward to
|
|
24
|
+
* provoke from a local test server, and faking the network to test them would
|
|
25
|
+
* only be testing the fake.
|
|
26
|
+
*/
|
|
27
|
+
export const badResponse = (
|
|
28
|
+
context: AuditContext,
|
|
29
|
+
page: PageSnapshot,
|
|
30
|
+
result: ProbeResult
|
|
31
|
+
): Diagnostic | null => {
|
|
32
|
+
const site = pageSite(context, page);
|
|
33
|
+
|
|
34
|
+
if (result.timedOut) {
|
|
35
|
+
return finding(
|
|
36
|
+
"BLUME_AUDIT_HTTP_TIMEOUT",
|
|
37
|
+
site,
|
|
38
|
+
`${page.url} did not respond within the timeout.`
|
|
39
|
+
);
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
const { status } = result;
|
|
43
|
+
if (status !== undefined && status >= SERVER_ERROR) {
|
|
44
|
+
return finding(
|
|
45
|
+
"BLUME_AUDIT_HTTP_5XX",
|
|
46
|
+
site,
|
|
47
|
+
`${page.url} returned HTTP ${status}.`
|
|
48
|
+
);
|
|
49
|
+
}
|
|
50
|
+
if (status !== undefined && status >= CLIENT_ERROR) {
|
|
51
|
+
return finding(
|
|
52
|
+
"BLUME_AUDIT_HTTP_4XX",
|
|
53
|
+
site,
|
|
54
|
+
`${page.url} is in the build but returned HTTP ${status}.`
|
|
55
|
+
);
|
|
56
|
+
}
|
|
57
|
+
if (!result.ok) {
|
|
58
|
+
return finding(
|
|
59
|
+
"BLUME_AUDIT_HTTP_5XX",
|
|
60
|
+
site,
|
|
61
|
+
`${page.url} is unreachable: ${result.error ?? "no response"}.`
|
|
62
|
+
);
|
|
63
|
+
}
|
|
64
|
+
return null;
|
|
65
|
+
};
|
|
66
|
+
|
|
67
|
+
/** What a *successful* response still gets wrong: headers, timing, protocol. */
|
|
68
|
+
export const servedPageChecks = (
|
|
69
|
+
context: AuditContext,
|
|
70
|
+
page: PageSnapshot,
|
|
71
|
+
result: ProbeResult,
|
|
72
|
+
origin: string
|
|
73
|
+
): Diagnostic[] => {
|
|
74
|
+
const site = pageSite(context, page);
|
|
75
|
+
const found: Diagnostic[] = [];
|
|
76
|
+
|
|
77
|
+
// A page served over HTTPS that redirects down to HTTP hands the reader to an
|
|
78
|
+
// insecure connection — the opposite of the usual upgrade.
|
|
79
|
+
if (
|
|
80
|
+
result.redirected &&
|
|
81
|
+
result.finalUrl?.startsWith("http://") &&
|
|
82
|
+
origin.startsWith("https://")
|
|
83
|
+
) {
|
|
84
|
+
found.push(
|
|
85
|
+
finding(
|
|
86
|
+
"BLUME_AUDIT_REDIRECT_TO_HTTP",
|
|
87
|
+
site,
|
|
88
|
+
`${page.url} redirects to ${result.finalUrl}, downgrading to HTTP.`
|
|
89
|
+
)
|
|
90
|
+
);
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
if (!result.encoding) {
|
|
94
|
+
found.push(
|
|
95
|
+
finding(
|
|
96
|
+
"BLUME_AUDIT_NOT_COMPRESSED",
|
|
97
|
+
site,
|
|
98
|
+
`${page.url} is served without gzip or brotli compression.`
|
|
99
|
+
)
|
|
100
|
+
);
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
if (result.ms !== undefined && result.ms > SLOW_MS) {
|
|
104
|
+
found.push(
|
|
105
|
+
finding(
|
|
106
|
+
"BLUME_AUDIT_SLOW_RESPONSE",
|
|
107
|
+
site,
|
|
108
|
+
`${page.url} took ${result.ms}ms to respond.`
|
|
109
|
+
)
|
|
110
|
+
);
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
// The header wins over the meta tag, so a stray `X-Robots-Tag: noindex` —
|
|
114
|
+
// Vercel sets one on password-protected and preview deploys — silently
|
|
115
|
+
// deindexes a page whose HTML looks perfectly indexable.
|
|
116
|
+
const tag = result.robotsTag;
|
|
117
|
+
if (tag?.includes("noindex") && page.indexable) {
|
|
118
|
+
found.push(
|
|
119
|
+
finding(
|
|
120
|
+
"BLUME_AUDIT_ROBOTS_HEADER_CONFLICT",
|
|
121
|
+
site,
|
|
122
|
+
`${page.url} sends "X-Robots-Tag: ${tag}" but its HTML has no noindex — the header wins.`
|
|
123
|
+
)
|
|
124
|
+
);
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
return found;
|
|
128
|
+
};
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* The built site, checked against a live deployment.
|
|
132
|
+
*
|
|
133
|
+
* Everything here needs the network, which is why it only runs with `--url`: a
|
|
134
|
+
* page that exists in `dist/` can still 404 in production behind a bad rewrite,
|
|
135
|
+
* and only the real response carries the headers (`Content-Encoding`,
|
|
136
|
+
* `X-Robots-Tag`) that decide whether the page is compressed and indexable.
|
|
137
|
+
*/
|
|
138
|
+
export const networkChecks: CheckModule = {
|
|
139
|
+
category: "network",
|
|
140
|
+
async run(context) {
|
|
141
|
+
const { origin } = context;
|
|
142
|
+
if (!origin) {
|
|
143
|
+
return [];
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
const found: Diagnostic[] = [];
|
|
147
|
+
const targets = context.pages.map((page) => liveUrl(origin, page));
|
|
148
|
+
// robots.txt and sitemap.xml are fetched alongside the pages: they're the
|
|
149
|
+
// two files a crawler asks for first, and a deploy that hides them silently
|
|
150
|
+
// undoes everything else the audit checks.
|
|
151
|
+
const robotsUrl = new URL("/robots.txt", origin).toString();
|
|
152
|
+
const sitemapUrl = new URL("/sitemap.xml", origin).toString();
|
|
153
|
+
|
|
154
|
+
const results = await probeAll([...targets, robotsUrl, sitemapUrl]);
|
|
155
|
+
|
|
156
|
+
for (const page of context.pages) {
|
|
157
|
+
const result = results.get(liveUrl(origin, page));
|
|
158
|
+
if (!result) {
|
|
159
|
+
continue;
|
|
160
|
+
}
|
|
161
|
+
const failure = badResponse(context, page, result);
|
|
162
|
+
if (failure) {
|
|
163
|
+
found.push(failure);
|
|
164
|
+
continue;
|
|
165
|
+
}
|
|
166
|
+
found.push(...servedPageChecks(context, page, result, origin));
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
const robots = results.get(robotsUrl);
|
|
170
|
+
if (robots && !robots.ok) {
|
|
171
|
+
found.push(
|
|
172
|
+
finding(
|
|
173
|
+
"BLUME_AUDIT_ROBOTS_NOT_ACCESSIBLE",
|
|
174
|
+
{ url: "/robots.txt" },
|
|
175
|
+
`robots.txt is not reachable at ${robotsUrl}.`
|
|
176
|
+
)
|
|
177
|
+
);
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
const sitemap = results.get(sitemapUrl);
|
|
181
|
+
if (context.sitemap && sitemap && !sitemap.ok) {
|
|
182
|
+
found.push(
|
|
183
|
+
finding(
|
|
184
|
+
"BLUME_AUDIT_SITEMAP_NOT_ACCESSIBLE",
|
|
185
|
+
{ url: "/sitemap.xml" },
|
|
186
|
+
`sitemap.xml is in the build but is not reachable at ${sitemapUrl}.`
|
|
187
|
+
)
|
|
188
|
+
);
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
return found;
|
|
192
|
+
},
|
|
193
|
+
tier: "network",
|
|
194
|
+
};
|
|
195
|
+
|
|
196
|
+
/**
|
|
197
|
+
* Outbound links, probed over the network (`--external`).
|
|
198
|
+
*
|
|
199
|
+
* Severity is graded rather than flat: a 404 is the author's bug, but a 403 or a
|
|
200
|
+
* 5xx is usually rate limiting or someone else's outage, and failing a build on
|
|
201
|
+
* that would make the check useless.
|
|
202
|
+
*/
|
|
203
|
+
export const externalChecks: CheckModule = {
|
|
204
|
+
category: "network",
|
|
205
|
+
async run(context) {
|
|
206
|
+
const origin = siteOrigin(context.project.config.deployment.site);
|
|
207
|
+
const deployBase = normalizeBasePath(
|
|
208
|
+
context.project.config.deployment.base
|
|
209
|
+
);
|
|
210
|
+
|
|
211
|
+
/** Every outbound URL, and the pages that link to it. */
|
|
212
|
+
const linkers = new Map<string, PageSnapshot[]>();
|
|
213
|
+
for (const page of context.pages) {
|
|
214
|
+
for (const link of page.links) {
|
|
215
|
+
const resolved = resolveHref(page.url, link.href, origin, deployBase);
|
|
216
|
+
if (resolved.kind !== "external") {
|
|
217
|
+
continue;
|
|
218
|
+
}
|
|
219
|
+
const pages = linkers.get(resolved.url);
|
|
220
|
+
if (pages) {
|
|
221
|
+
if (!pages.includes(page)) {
|
|
222
|
+
pages.push(page);
|
|
223
|
+
}
|
|
224
|
+
} else {
|
|
225
|
+
linkers.set(resolved.url, [page]);
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
if (linkers.size === 0) {
|
|
231
|
+
return [];
|
|
232
|
+
}
|
|
233
|
+
const results = await probeAll([...linkers.keys()]);
|
|
234
|
+
|
|
235
|
+
const found: Diagnostic[] = [];
|
|
236
|
+
for (const [url, pages] of linkers) {
|
|
237
|
+
const result = results.get(url);
|
|
238
|
+
if (!result) {
|
|
239
|
+
continue;
|
|
240
|
+
}
|
|
241
|
+
const site = pageSite(context, pages[0] as PageSnapshot);
|
|
242
|
+
|
|
243
|
+
const grade = gradeExternal(result);
|
|
244
|
+
if (grade) {
|
|
245
|
+
// Severity is graded, not flat. A 404 is a bug the author can fix; a 403
|
|
246
|
+
// or 5xx is usually rate limiting or someone else's outage, and failing
|
|
247
|
+
// a build on that would get `--external` switched off for good.
|
|
248
|
+
found.push({
|
|
249
|
+
...finding(
|
|
250
|
+
"BLUME_AUDIT_EXTERNAL_LINK_BROKEN",
|
|
251
|
+
site,
|
|
252
|
+
`${url} is unreachable (${grade.detail}), linked from ${pages.length} page(s).`
|
|
253
|
+
),
|
|
254
|
+
severity: grade.severity,
|
|
255
|
+
});
|
|
256
|
+
continue;
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
if (result.redirected && result.finalUrl && result.finalUrl !== url) {
|
|
260
|
+
found.push(
|
|
261
|
+
finding(
|
|
262
|
+
"BLUME_AUDIT_EXTERNAL_LINK_REDIRECT",
|
|
263
|
+
site,
|
|
264
|
+
`${url} redirects to ${result.finalUrl}.`
|
|
265
|
+
)
|
|
266
|
+
);
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
return found;
|
|
270
|
+
},
|
|
271
|
+
tier: "external",
|
|
272
|
+
};
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
import { readFile } from "node:fs/promises";
|
|
2
|
+
|
|
3
|
+
import { join } from "pathe";
|
|
4
|
+
|
|
5
|
+
import { normalizeBasePath } from "../../core/base-path.ts";
|
|
6
|
+
import type { Diagnostic } from "../../core/types.ts";
|
|
7
|
+
import { finding } from "../catalog.ts";
|
|
8
|
+
import { imageSize } from "../image-size.ts";
|
|
9
|
+
import type { ImageSize } from "../image-size.ts";
|
|
10
|
+
import { pageSite } from "../locate.ts";
|
|
11
|
+
import type { CheckModule, PageSnapshot } from "../types.ts";
|
|
12
|
+
import { resolveHref, siteOrigin } from "../url.ts";
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* Below this, large social cards render the image blurry or fall back to a
|
|
16
|
+
* small-card layout: Facebook's floor for a full-width card is 600×315
|
|
17
|
+
* (1200×630 recommended), X's for `summary_large_image` is 300×157.
|
|
18
|
+
*/
|
|
19
|
+
const MIN_WIDTH = 600;
|
|
20
|
+
const MIN_HEIGHT = 315;
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* The Open Graph image as bytes, not just a tag. `OG_IMAGE_MISSING` proves the
|
|
24
|
+
* meta tag exists; only the build can prove the tag points at a real file of a
|
|
25
|
+
* shareable size — a crawler needs the live site for either.
|
|
26
|
+
*/
|
|
27
|
+
export const ogImageChecks: CheckModule = {
|
|
28
|
+
category: "social",
|
|
29
|
+
async run(context) {
|
|
30
|
+
const found: Diagnostic[] = [];
|
|
31
|
+
/** Pending reads per file path — the site-wide default OG image is read once. */
|
|
32
|
+
const sizes = new Map<string, Promise<ImageSize | null>>();
|
|
33
|
+
|
|
34
|
+
const read = async (path: string): Promise<ImageSize | null> => {
|
|
35
|
+
try {
|
|
36
|
+
return imageSize(await readFile(join(context.staticDir, path)));
|
|
37
|
+
} catch {
|
|
38
|
+
// Unreadable bytes: existence was already established via the file
|
|
39
|
+
// index, and an unknown format is not a finding.
|
|
40
|
+
return null;
|
|
41
|
+
}
|
|
42
|
+
};
|
|
43
|
+
|
|
44
|
+
// Cache the promise, not the result, so concurrent pages sharing one
|
|
45
|
+
// image (the site-wide default) share a single read.
|
|
46
|
+
const measure = (path: string): Promise<ImageSize | null> => {
|
|
47
|
+
const cached = sizes.get(path);
|
|
48
|
+
if (cached) {
|
|
49
|
+
return cached;
|
|
50
|
+
}
|
|
51
|
+
const pending = read(path);
|
|
52
|
+
sizes.set(path, pending);
|
|
53
|
+
return pending;
|
|
54
|
+
};
|
|
55
|
+
|
|
56
|
+
const origin = siteOrigin(context.project.config.deployment.site);
|
|
57
|
+
const deployBase = normalizeBasePath(
|
|
58
|
+
context.project.config.deployment.base
|
|
59
|
+
);
|
|
60
|
+
const candidates: { page: PageSnapshot; path: string }[] = [];
|
|
61
|
+
|
|
62
|
+
for (const page of context.pages) {
|
|
63
|
+
const src = page.og["og:image"];
|
|
64
|
+
if (!src) {
|
|
65
|
+
continue;
|
|
66
|
+
}
|
|
67
|
+
// An og:image on another origin (a CDN, an external host) is outside the
|
|
68
|
+
// build — its existence can't be proven without the network, so it's the
|
|
69
|
+
// network tier's business, not this check's.
|
|
70
|
+
const resolved = resolveHref(page.url, src, origin, deployBase);
|
|
71
|
+
if (resolved.kind === "external" || resolved.kind === "ignored") {
|
|
72
|
+
continue;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
const { path } = resolved;
|
|
76
|
+
if (
|
|
77
|
+
!(context.files.has(path) || context.files.has(`${path}/index.html`))
|
|
78
|
+
) {
|
|
79
|
+
found.push(
|
|
80
|
+
finding(
|
|
81
|
+
"BLUME_AUDIT_OG_IMAGE_BROKEN",
|
|
82
|
+
pageSite(context, page, ["seo", "image"]),
|
|
83
|
+
`og:image points at ${path}, which is not in the build.`
|
|
84
|
+
)
|
|
85
|
+
);
|
|
86
|
+
continue;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
candidates.push({ page, path });
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
const measured = await Promise.all(
|
|
93
|
+
candidates.map(async (candidate) => ({
|
|
94
|
+
...candidate,
|
|
95
|
+
size: await measure(candidate.path),
|
|
96
|
+
}))
|
|
97
|
+
);
|
|
98
|
+
for (const { page, path, size } of measured) {
|
|
99
|
+
if (size && (size.width < MIN_WIDTH || size.height < MIN_HEIGHT)) {
|
|
100
|
+
found.push(
|
|
101
|
+
finding(
|
|
102
|
+
"BLUME_AUDIT_OG_IMAGE_SMALL",
|
|
103
|
+
pageSite(context, page, ["seo", "image"]),
|
|
104
|
+
`og:image ${path} is ${size.width}×${size.height} — below the ${MIN_WIDTH}×${MIN_HEIGHT} floor for large social cards.`
|
|
105
|
+
)
|
|
106
|
+
);
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
return found;
|
|
111
|
+
},
|
|
112
|
+
tier: "static",
|
|
113
|
+
};
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
import type { Diagnostic } from "../../core/types.ts";
|
|
2
|
+
import { finding } from "../catalog.ts";
|
|
3
|
+
import { pageSite } from "../locate.ts";
|
|
4
|
+
import type { CheckModule } from "../types.ts";
|
|
5
|
+
import { normalizePath } from "../url.ts";
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Configured redirects, walked through to their destinations.
|
|
9
|
+
*
|
|
10
|
+
* Ahrefs also lists "3XX redirect" and "302 redirect" as issues. Having a
|
|
11
|
+
* redirect is inventory, not a defect, and "HTTP to HTTPS redirect" — which it
|
|
12
|
+
* also flags — is correct behavior. None of those are reported. What is: a
|
|
13
|
+
* redirect that loops, dead-ends, takes the long way round, or is shadowed by a
|
|
14
|
+
* real page and therefore never fires at all.
|
|
15
|
+
*/
|
|
16
|
+
export const redirectChecks: CheckModule = {
|
|
17
|
+
category: "redirects",
|
|
18
|
+
run(context) {
|
|
19
|
+
const found: Diagnostic[] = [];
|
|
20
|
+
const configFile = context.project.context.configFile ?? undefined;
|
|
21
|
+
const site = { file: configFile, url: "/" };
|
|
22
|
+
|
|
23
|
+
for (const redirect of context.redirects) {
|
|
24
|
+
const from = normalizePath(redirect.from);
|
|
25
|
+
|
|
26
|
+
// A redirect whose source is also a built page never fires — the page
|
|
27
|
+
// wins. Ahrefs can't see this: it only observes the served response, which
|
|
28
|
+
// looks perfectly healthy.
|
|
29
|
+
if (context.byUrl.has(from)) {
|
|
30
|
+
found.push(
|
|
31
|
+
finding(
|
|
32
|
+
"BLUME_AUDIT_REDIRECT_SOURCE_IS_PAGE",
|
|
33
|
+
{ ...site, url: from },
|
|
34
|
+
`A redirect is configured from ${from}, but ${from} is also a real page — the redirect never fires.`
|
|
35
|
+
)
|
|
36
|
+
);
|
|
37
|
+
continue;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
if (redirect.outcome === "loop") {
|
|
41
|
+
found.push(
|
|
42
|
+
finding(
|
|
43
|
+
"BLUME_AUDIT_REDIRECT_LOOP",
|
|
44
|
+
{ ...site, url: from },
|
|
45
|
+
`Redirect loop: ${redirect.chain.join(" → ")}`
|
|
46
|
+
)
|
|
47
|
+
);
|
|
48
|
+
} else if (redirect.outcome === "broken") {
|
|
49
|
+
found.push(
|
|
50
|
+
finding(
|
|
51
|
+
"BLUME_AUDIT_REDIRECT_BROKEN",
|
|
52
|
+
{ ...site, url: from },
|
|
53
|
+
`Redirect from ${from} lands on ${redirect.chain.at(-1)}, which the build does not serve.`
|
|
54
|
+
)
|
|
55
|
+
);
|
|
56
|
+
} else if (redirect.outcome === "chain") {
|
|
57
|
+
const hops = redirect.chain.length - 1;
|
|
58
|
+
const severity =
|
|
59
|
+
hops > context.thresholds.maxRedirectHops ? " (too long)" : "";
|
|
60
|
+
found.push(
|
|
61
|
+
finding(
|
|
62
|
+
"BLUME_AUDIT_REDIRECT_CHAIN",
|
|
63
|
+
{ ...site, url: from },
|
|
64
|
+
`Redirect from ${from} passes through ${hops} hops${severity}: ${redirect.chain.join(" → ")}`
|
|
65
|
+
)
|
|
66
|
+
);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
// A meta refresh is a redirect the framework doesn't know about, so it can't
|
|
71
|
+
// be validated, cached, or followed reliably by crawlers.
|
|
72
|
+
for (const page of context.pages) {
|
|
73
|
+
if (page.metaRefresh) {
|
|
74
|
+
found.push(
|
|
75
|
+
finding(
|
|
76
|
+
"BLUME_AUDIT_META_REFRESH",
|
|
77
|
+
pageSite(context, page),
|
|
78
|
+
`Page uses a meta refresh redirect ("${page.metaRefresh}").`
|
|
79
|
+
)
|
|
80
|
+
);
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
return found;
|
|
85
|
+
},
|
|
86
|
+
tier: "static",
|
|
87
|
+
};
|