blume 1.0.4 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +65 -0
- package/dist/cli/index.js +13255 -10232
- package/dist/cli/index.js.map +91 -60
- package/dist/types/core/config-input.d.ts +61 -1
- package/dist/types/core/data.d.ts +9 -0
- package/dist/types/core/deployment-env.d.ts +6 -0
- package/dist/types/core/diagnostics.d.ts +23 -0
- package/dist/types/core/i18n-ui.d.ts +8 -8
- package/dist/types/core/schema.d.ts +131 -22
- package/dist/types/core/sources/types.d.ts +3 -1
- package/dist/types/core/standard-schema.d.ts +41 -0
- package/dist/types/core/types.d.ts +13 -0
- package/dist/types/og/card.d.ts +63 -0
- package/dist/types/og/dimensions.d.ts +12 -0
- package/docs/01-quickstart.mdx +1 -1
- package/docs/02-deployment.mdx +9 -1
- package/docs/advanced/api-reference.mdx +11 -0
- package/docs/advanced/changelog.mdx +1 -1
- package/docs/advanced/skills.mdx +1 -1
- package/docs/configuration/ai.mdx +1 -1
- package/docs/configuration/customization.mdx +1 -1
- package/docs/configuration/export.mdx +1 -1
- package/docs/configuration/index.mdx +21 -1
- package/docs/configuration/search.mdx +28 -1
- package/docs/configuration/seo.mdx +21 -2
- package/docs/configuration/theming.mdx +1 -1
- package/docs/content/components.mdx +14 -0
- package/docs/content/index.mdx +1 -1
- package/docs/content/meta.mdx +1 -1
- package/docs/content/navigation.mdx +1 -1
- package/docs/content/sources.mdx +1 -1
- package/docs/reference/cli.mdx +79 -1
- package/docs/reference/frontmatter.mdx +29 -1
- package/package.json +3 -3
- package/skills/blume-migrate/SKILL.md +1 -1
- package/skills/blume-migrate/references/mintlify.md +3 -2
- package/skills/blume-migrate/scripts/mintlify-codemod.mjs +16 -4
- package/src/ai/llms.ts +15 -0
- package/src/astro/adapter-root.ts +70 -0
- package/src/astro/generate.ts +50 -19
- package/src/astro/index.ts +1 -0
- package/src/astro/pages.ts +18 -3
- package/src/astro/templates.ts +65 -22
- package/src/audit/agent.ts +114 -0
- package/src/audit/catalog.ts +826 -0
- package/src/audit/checks/assets.ts +177 -0
- package/src/audit/checks/content.ts +231 -0
- package/src/audit/checks/duplicates.ts +131 -0
- package/src/audit/checks/i18n.ts +246 -0
- package/src/audit/checks/indexability.ts +213 -0
- package/src/audit/checks/links.ts +223 -0
- package/src/audit/checks/llms.ts +135 -0
- package/src/audit/checks/network.ts +272 -0
- package/src/audit/checks/og-image.ts +113 -0
- package/src/audit/checks/redirects.ts +87 -0
- package/src/audit/checks/robots.ts +114 -0
- package/src/audit/checks/sitemap.ts +229 -0
- package/src/audit/checks/social.ts +238 -0
- package/src/audit/crawl.ts +259 -0
- package/src/audit/graph.ts +74 -0
- package/src/audit/html.ts +54 -0
- package/src/audit/image-size.ts +63 -0
- package/src/audit/locate.ts +33 -0
- package/src/audit/redirects.ts +74 -0
- package/src/audit/report.ts +278 -0
- package/src/audit/run.ts +198 -0
- package/src/audit/snapshot.ts +189 -0
- package/src/audit/types.ts +214 -0
- package/src/audit/url.ts +103 -0
- package/src/cli/commands/audit.ts +205 -0
- package/src/cli/commands/build.ts +51 -12
- package/src/cli/index.ts +2 -0
- package/src/components/content/Tabs.astro +98 -15
- package/src/components/layout/Breadcrumbs.astro +1 -1
- package/src/components/layout/Header.astro +1 -0
- package/src/components/layout/PageFeedback.astro +1 -1
- package/src/components/layout/PageLayout.astro +5 -1
- package/src/components/layout/Pagination.astro +1 -1
- package/src/components/layout/RootLayout.astro +5 -3
- package/src/components/layout/Search.astro +35 -6
- package/src/components/layout/TableOfContents.astro +1 -1
- package/src/components/openapi/Authorization.astro +80 -0
- package/src/components/openapi/Operation.astro +19 -1
- package/src/components/openapi/ParametersTable.astro +1 -1
- package/src/components/openapi/security.ts +201 -0
- package/src/components/openapi/snippets.ts +42 -13
- package/src/core/config-input.ts +66 -1
- package/src/core/data.ts +9 -1
- package/src/core/deployment-env.ts +9 -0
- package/src/core/diagnostics.ts +59 -12
- package/src/core/links.ts +2 -91
- package/src/core/nav-diagnostics.ts +48 -4
- package/src/core/probe.ts +136 -0
- package/src/core/project-graph.ts +8 -0
- package/src/core/schema.ts +86 -3
- package/src/core/sources/normalize.ts +198 -25
- package/src/core/sources/types.ts +3 -1
- package/src/core/standard-schema.ts +54 -0
- package/src/core/types.ts +13 -0
- package/src/deploy/adapter-output.ts +27 -15
- package/src/deploy/headers.ts +66 -0
- package/src/deploy/redirects.ts +49 -9
- package/src/og/card.ts +98 -33
- package/src/og/index.ts +1 -1
- package/src/search/popular.ts +33 -0
- package/src/theme/entry.ts +6 -1
|
@@ -0,0 +1,259 @@
|
|
|
1
|
+
import { readFile, stat } from "node:fs/promises";
|
|
2
|
+
|
|
3
|
+
import { join, relative } from "pathe";
|
|
4
|
+
import { glob } from "tinyglobby";
|
|
5
|
+
|
|
6
|
+
import { examplesRouteBase } from "../astro/templates.ts";
|
|
7
|
+
import { stripBasePath } from "../core/base-path.ts";
|
|
8
|
+
import type { BlumeManifest, RouteManifestEntry } from "../core/types.ts";
|
|
9
|
+
import { buildSnapshot } from "./snapshot.ts";
|
|
10
|
+
import type { LlmsDoc, PageSnapshot, RobotsDoc, SitemapDoc } from "./types.ts";
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* The route prefix `<Component />` preview frames live under. They are bare
|
|
14
|
+
* documents rendered for iframes — deliberately noindex, no title worth
|
|
15
|
+
* grading, no front matter to fix — so auditing them as pages only produces
|
|
16
|
+
* findings nobody can act on.
|
|
17
|
+
*/
|
|
18
|
+
const EXAMPLES_PREFIX = `${examplesRouteBase("")}/`;
|
|
19
|
+
|
|
20
|
+
/** Everything read off disk in one pass over the built site. */
|
|
21
|
+
export interface CrawlResult {
|
|
22
|
+
pages: PageSnapshot[];
|
|
23
|
+
/** Every file in the static dir: URL path -> size in bytes. */
|
|
24
|
+
files: Map<string, number>;
|
|
25
|
+
sitemap: SitemapDoc | null;
|
|
26
|
+
robots: RobotsDoc | null;
|
|
27
|
+
llms: LlmsDoc | null;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* The URL a built HTML file is served at. Astro's directory build format emits
|
|
32
|
+
* `docs/api/index.html` for `/docs/api`, so the `index.html` leaf collapses;
|
|
33
|
+
* a flat `404.html` keeps its name.
|
|
34
|
+
*/
|
|
35
|
+
export const fileToUrl = (staticDir: string, file: string): string => {
|
|
36
|
+
const rel = relative(staticDir, file).replaceAll("\\", "/");
|
|
37
|
+
const path = rel.replace(/(?:^|\/)index\.html$/u, "").replace(/\.html$/u, "");
|
|
38
|
+
return path === "" ? "/" : `/${path}`;
|
|
39
|
+
};
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Every file in the static dir, keyed the way an HTML `src`/`href` would name
|
|
43
|
+
* it, with its size — so a reference can be resolved and weighed in one pass.
|
|
44
|
+
*/
|
|
45
|
+
const indexFiles = async (staticDir: string): Promise<Map<string, number>> => {
|
|
46
|
+
const found = await glob("**/*", { cwd: staticDir, dot: true });
|
|
47
|
+
const sized = await Promise.all(
|
|
48
|
+
found.map(async (file) => {
|
|
49
|
+
const path = `/${file.replaceAll("\\", "/")}`;
|
|
50
|
+
const info = await stat(join(staticDir, file));
|
|
51
|
+
return [path, info.size] as const;
|
|
52
|
+
})
|
|
53
|
+
);
|
|
54
|
+
return new Map(sized);
|
|
55
|
+
};
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Look a built URL up in the route manifest. Routes carry `basePath` while the
|
|
59
|
+
* built file tree may not, so both spellings are tried before giving up — a page
|
|
60
|
+
* that fails to join is still audited, it just can't name a source file to fix.
|
|
61
|
+
*/
|
|
62
|
+
const routeIndex = (
|
|
63
|
+
manifest: BlumeManifest,
|
|
64
|
+
basePath: string
|
|
65
|
+
): Map<string, RouteManifestEntry> => {
|
|
66
|
+
const index = new Map<string, RouteManifestEntry>();
|
|
67
|
+
for (const route of manifest.routes) {
|
|
68
|
+
index.set(route.path, route);
|
|
69
|
+
const stripped = stripBasePath(basePath, route.path);
|
|
70
|
+
if (!index.has(stripped)) {
|
|
71
|
+
index.set(stripped, route);
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
return index;
|
|
75
|
+
};
|
|
76
|
+
|
|
77
|
+
const SITEMAP_URL = /<url>(?<block>[\s\S]*?)<\/url>/gu;
|
|
78
|
+
const SITEMAP_LOC = /<loc>(?<loc>[\s\S]*?)<\/loc>/gu;
|
|
79
|
+
const SITEMAP_LASTMOD = /<lastmod>(?<date>[\s\S]*?)<\/lastmod>/u;
|
|
80
|
+
const XML_ENTITIES: Record<string, string> = {
|
|
81
|
+
"&": "&",
|
|
82
|
+
"'": "'",
|
|
83
|
+
">": ">",
|
|
84
|
+
"<": "<",
|
|
85
|
+
""": '"',
|
|
86
|
+
};
|
|
87
|
+
|
|
88
|
+
const unescapeXml = (value: string): string =>
|
|
89
|
+
value.replaceAll(
|
|
90
|
+
/&(?:amp|apos|gt|lt|quot);/gu,
|
|
91
|
+
(entity) => XML_ENTITIES[entity] ?? entity
|
|
92
|
+
);
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Parse `sitemap.xml`. Deliberately shallow: we only need the `<loc>` list and
|
|
96
|
+
* whether the document is a well-formed urlset, and pulling in an XML parser to
|
|
97
|
+
* learn that would be a dependency for one regex.
|
|
98
|
+
*/
|
|
99
|
+
export const parseSitemap = (
|
|
100
|
+
file: string,
|
|
101
|
+
xml: string,
|
|
102
|
+
bytes: number
|
|
103
|
+
): SitemapDoc => {
|
|
104
|
+
const doc: SitemapDoc = { bytes, file, lastmod: new Map(), urls: [] };
|
|
105
|
+
if (!xml.includes("<urlset")) {
|
|
106
|
+
doc.error = xml.includes("<sitemapindex")
|
|
107
|
+
? "sitemap is an index, not a urlset"
|
|
108
|
+
: "no <urlset> element";
|
|
109
|
+
return doc;
|
|
110
|
+
}
|
|
111
|
+
for (const match of xml.matchAll(SITEMAP_LOC)) {
|
|
112
|
+
const loc = unescapeXml((match.groups?.loc ?? "").trim());
|
|
113
|
+
if (loc) {
|
|
114
|
+
doc.urls.push(loc);
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
// `<lastmod>` is scoped per `<url>` block so it stays attached to its `<loc>`
|
|
118
|
+
// — the flat loc scan above deliberately isn't, so a sitemap with stray text
|
|
119
|
+
// between blocks still yields its URL list.
|
|
120
|
+
for (const match of xml.matchAll(SITEMAP_URL)) {
|
|
121
|
+
const block = match.groups?.block ?? "";
|
|
122
|
+
const loc = unescapeXml(
|
|
123
|
+
(
|
|
124
|
+
new RegExp(SITEMAP_LOC.source, "u").exec(block)?.groups?.loc ?? ""
|
|
125
|
+
).trim()
|
|
126
|
+
);
|
|
127
|
+
const lastmod = SITEMAP_LASTMOD.exec(block)?.groups?.date?.trim();
|
|
128
|
+
if (loc && lastmod) {
|
|
129
|
+
doc.lastmod?.set(loc, lastmod);
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
return doc;
|
|
133
|
+
};
|
|
134
|
+
|
|
135
|
+
/**
|
|
136
|
+
* Parse the `llms.txt` index into its Markdown link targets. Deliberately
|
|
137
|
+
* shallow, like {@link parseSitemap}: the checks only need "which pages does
|
|
138
|
+
* this file claim exist", not a Markdown AST.
|
|
139
|
+
*/
|
|
140
|
+
export const parseLlms = (file: string, text: string): LlmsDoc => {
|
|
141
|
+
const entries: LlmsDoc["entries"] = [];
|
|
142
|
+
const link = /\]\((?<url>[^)\s]+)\)/gu;
|
|
143
|
+
for (const [index, line] of text.split(/\r?\n/u).entries()) {
|
|
144
|
+
for (const match of line.matchAll(link)) {
|
|
145
|
+
const url = match.groups?.url;
|
|
146
|
+
if (url) {
|
|
147
|
+
entries.push({ line: index + 1, url });
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
return { entries, file };
|
|
152
|
+
};
|
|
153
|
+
|
|
154
|
+
const ROBOTS_DIRECTIVE = /^(?<field>[a-z-]+)\s*:\s*(?<value>.*)$/iu;
|
|
155
|
+
|
|
156
|
+
/** Parse `robots.txt` into the directives the audit cares about. */
|
|
157
|
+
export const parseRobots = (file: string, text: string): RobotsDoc => {
|
|
158
|
+
const doc: RobotsDoc = { disallow: [], file, invalid: [], sitemaps: [] };
|
|
159
|
+
// Only `User-agent: *` rules bind the crawlers we're auditing for; a block
|
|
160
|
+
// scoped to some other agent isn't a finding about our indexable pages.
|
|
161
|
+
let appliesToAll = false;
|
|
162
|
+
for (const [index, raw] of text.split(/\r?\n/u).entries()) {
|
|
163
|
+
const line = raw.trim();
|
|
164
|
+
if (line === "" || line.startsWith("#")) {
|
|
165
|
+
continue;
|
|
166
|
+
}
|
|
167
|
+
const match = ROBOTS_DIRECTIVE.exec(line);
|
|
168
|
+
if (!match) {
|
|
169
|
+
doc.invalid.push({ line: index + 1, text: line });
|
|
170
|
+
continue;
|
|
171
|
+
}
|
|
172
|
+
const field = (match.groups?.field ?? "").toLowerCase();
|
|
173
|
+
const value = (match.groups?.value ?? "").trim();
|
|
174
|
+
if (field === "user-agent") {
|
|
175
|
+
appliesToAll = value === "*";
|
|
176
|
+
} else if (field === "disallow" && appliesToAll && value) {
|
|
177
|
+
doc.disallow.push(value);
|
|
178
|
+
} else if (field === "sitemap" && value) {
|
|
179
|
+
doc.sitemaps.push(value);
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
return doc;
|
|
183
|
+
};
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* Whether an emitted `.html` file is a real page rather than a fragment.
|
|
187
|
+
*
|
|
188
|
+
* Astro writes standalone HTML for some components (`_home/Footer/index.html`
|
|
189
|
+
* and friends) — markup with no `<html>` or `<head>`, never served as a route.
|
|
190
|
+
* Auditing those as pages reports every one of them as missing a title, a
|
|
191
|
+
* viewport, and a `lang` attribute, which is noise about markup nobody visits.
|
|
192
|
+
* An SEO audit is about documents, so that's what we keep.
|
|
193
|
+
*/
|
|
194
|
+
const isDocument = (html: string): boolean => /<head[\s>]/iu.test(html);
|
|
195
|
+
|
|
196
|
+
const readIfPresent = async (file: string): Promise<string | null> => {
|
|
197
|
+
try {
|
|
198
|
+
return await readFile(file, "utf-8");
|
|
199
|
+
} catch {
|
|
200
|
+
return null;
|
|
201
|
+
}
|
|
202
|
+
};
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* Read the built site: every HTML page reduced to a snapshot, the full file
|
|
206
|
+
* index (for resolving subresource references), plus sitemap.xml and robots.txt.
|
|
207
|
+
*/
|
|
208
|
+
export const crawlStaticDir = async (options: {
|
|
209
|
+
staticDir: string;
|
|
210
|
+
manifest: BlumeManifest;
|
|
211
|
+
basePath: string;
|
|
212
|
+
}): Promise<CrawlResult> => {
|
|
213
|
+
const { staticDir, manifest, basePath } = options;
|
|
214
|
+
const routes = routeIndex(manifest, basePath);
|
|
215
|
+
|
|
216
|
+
const htmlFiles = await glob("**/*.html", { absolute: true, cwd: staticDir });
|
|
217
|
+
const snapshots = await Promise.all(
|
|
218
|
+
htmlFiles.toSorted().map(async (file) => {
|
|
219
|
+
const url = fileToUrl(staticDir, file);
|
|
220
|
+
if (stripBasePath(basePath, url).startsWith(EXAMPLES_PREFIX)) {
|
|
221
|
+
return null;
|
|
222
|
+
}
|
|
223
|
+
const html = await readFile(file, "utf-8");
|
|
224
|
+
if (!isDocument(html)) {
|
|
225
|
+
return null;
|
|
226
|
+
}
|
|
227
|
+
return buildSnapshot({
|
|
228
|
+
file,
|
|
229
|
+
html,
|
|
230
|
+
route: routes.get(url) ?? routes.get(stripBasePath(basePath, url)),
|
|
231
|
+
url,
|
|
232
|
+
});
|
|
233
|
+
})
|
|
234
|
+
);
|
|
235
|
+
const pages = snapshots.filter((page) => page !== null);
|
|
236
|
+
|
|
237
|
+
const sitemapFile = join(staticDir, "sitemap.xml");
|
|
238
|
+
const sitemapXml = await readIfPresent(sitemapFile);
|
|
239
|
+
const robotsFile = join(staticDir, "robots.txt");
|
|
240
|
+
const robotsTxt = await readIfPresent(robotsFile);
|
|
241
|
+
const llmsFile = join(staticDir, "llms.txt");
|
|
242
|
+
const llmsText = await readIfPresent(llmsFile);
|
|
243
|
+
const files = await indexFiles(staticDir);
|
|
244
|
+
|
|
245
|
+
return {
|
|
246
|
+
files,
|
|
247
|
+
llms: llmsText === null ? null : parseLlms(llmsFile, llmsText),
|
|
248
|
+
pages,
|
|
249
|
+
robots: robotsTxt === null ? null : parseRobots(robotsFile, robotsTxt),
|
|
250
|
+
sitemap:
|
|
251
|
+
sitemapXml === null
|
|
252
|
+
? null
|
|
253
|
+
: parseSitemap(
|
|
254
|
+
sitemapFile,
|
|
255
|
+
sitemapXml,
|
|
256
|
+
files.get("/sitemap.xml") ?? Buffer.byteLength(sitemapXml, "utf-8")
|
|
257
|
+
),
|
|
258
|
+
};
|
|
259
|
+
};
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
import type { LinkGraph, PageSnapshot } from "./types.ts";
|
|
2
|
+
import { resolveHref } from "./url.ts";
|
|
3
|
+
|
|
4
|
+
const add = (
|
|
5
|
+
map: Map<string, Set<string>>,
|
|
6
|
+
key: string,
|
|
7
|
+
value: string
|
|
8
|
+
): void => {
|
|
9
|
+
const existing = map.get(key);
|
|
10
|
+
if (existing) {
|
|
11
|
+
existing.add(value);
|
|
12
|
+
} else {
|
|
13
|
+
map.set(key, new Set([value]));
|
|
14
|
+
}
|
|
15
|
+
};
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Build the internal link graph, keeping prose links and chrome links apart.
|
|
19
|
+
*
|
|
20
|
+
* The split is the whole point. Blume's sidebar links every navigable page from
|
|
21
|
+
* every page, so a graph that lumps the two together concludes that every page
|
|
22
|
+
* has hundreds of inbound links — which makes "orphan page" unfireable and makes
|
|
23
|
+
* a single broken sidebar link look like N broken links, one per page.
|
|
24
|
+
*/
|
|
25
|
+
export const buildGraph = (
|
|
26
|
+
pages: PageSnapshot[],
|
|
27
|
+
origin: string | null,
|
|
28
|
+
deployBase = ""
|
|
29
|
+
): LinkGraph => {
|
|
30
|
+
const graph: LinkGraph = {
|
|
31
|
+
chromeIn: new Map(),
|
|
32
|
+
chromeOut: new Map(),
|
|
33
|
+
contentIn: new Map(),
|
|
34
|
+
contentOut: new Map(),
|
|
35
|
+
};
|
|
36
|
+
|
|
37
|
+
for (const page of pages) {
|
|
38
|
+
for (const link of page.links) {
|
|
39
|
+
const resolved = resolveHref(page.url, link.href, origin, deployBase);
|
|
40
|
+
if (resolved.kind !== "internal" && resolved.kind !== "self-origin") {
|
|
41
|
+
continue;
|
|
42
|
+
}
|
|
43
|
+
const out = link.content ? graph.contentOut : graph.chromeOut;
|
|
44
|
+
const incoming = link.content ? graph.contentIn : graph.chromeIn;
|
|
45
|
+
add(out, page.url, resolved.path);
|
|
46
|
+
add(incoming, resolved.path, page.url);
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
return graph;
|
|
51
|
+
};
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* Pages nothing links to from prose — reachable only through the sidebar.
|
|
55
|
+
*
|
|
56
|
+
* This is what an SEO means by "orphan": the page ships, but no other page's
|
|
57
|
+
* body ever points a reader (or a crawler following editorial links) at it.
|
|
58
|
+
* Non-indexable pages are excluded — a `noindex` page is *meant* to be
|
|
59
|
+
* unreachable — as is the home page, which is nobody's job to link to. Under a
|
|
60
|
+
* `basePath` the home page's built URL is the base itself, so the caller names
|
|
61
|
+
* it via `homeUrl`.
|
|
62
|
+
*/
|
|
63
|
+
export const orphanPages = (
|
|
64
|
+
pages: PageSnapshot[],
|
|
65
|
+
graph: LinkGraph,
|
|
66
|
+
homeUrl = "/"
|
|
67
|
+
): PageSnapshot[] =>
|
|
68
|
+
pages.filter(
|
|
69
|
+
(page) =>
|
|
70
|
+
page.indexable &&
|
|
71
|
+
page.url !== "/" &&
|
|
72
|
+
page.url !== homeUrl &&
|
|
73
|
+
(graph.contentIn.get(page.url)?.size ?? 0) === 0
|
|
74
|
+
);
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
import { parse } from "node-html-parser";
|
|
2
|
+
import type { HTMLElement } from "node-html-parser";
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* A parsed HTML document. Re-exported so the rest of the audit types against
|
|
6
|
+
* this module rather than the parser: `node-html-parser` is deliberately
|
|
7
|
+
* quarantined here, so swapping it stays a one-file change.
|
|
8
|
+
*/
|
|
9
|
+
export type HtmlDocument = HTMLElement;
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Parse built HTML into a queryable tree.
|
|
13
|
+
*
|
|
14
|
+
* `blockTextElements` keeps the raw text of `<script>`/`<style>`/`<pre>` intact
|
|
15
|
+
* instead of parsing it as markup — the audit reads `<script
|
|
16
|
+
* type="application/ld+json">` bodies verbatim to validate structured data.
|
|
17
|
+
*/
|
|
18
|
+
export const parseHtml = (html: string): HtmlDocument =>
|
|
19
|
+
parse(html, {
|
|
20
|
+
blockTextElements: { noscript: true, pre: true, script: true, style: true },
|
|
21
|
+
comment: false,
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
/** Trimmed attribute value, or null when absent or empty. */
|
|
25
|
+
export const attr = (element: HTMLElement, name: string): string | null => {
|
|
26
|
+
const value = element.getAttribute(name)?.trim();
|
|
27
|
+
return value || null;
|
|
28
|
+
};
|
|
29
|
+
|
|
30
|
+
/** `content` of every `<meta>` matching a selector, in document order. */
|
|
31
|
+
export const metaContents = (
|
|
32
|
+
document: HtmlDocument,
|
|
33
|
+
selector: string
|
|
34
|
+
): string[] =>
|
|
35
|
+
document
|
|
36
|
+
.querySelectorAll(selector)
|
|
37
|
+
.map((element) => element.getAttribute("content")?.trim() ?? "")
|
|
38
|
+
.filter((value) => value.length > 0);
|
|
39
|
+
|
|
40
|
+
/**
|
|
41
|
+
* Collapse an element's visible text. Script/style bodies and code blocks are
|
|
42
|
+
* dropped: a fenced code sample isn't prose, and counting it would let a page
|
|
43
|
+
* that is 90% code pass the word-count check on the strength of its snippets.
|
|
44
|
+
*/
|
|
45
|
+
export const visibleText = (element: HTMLElement): string => {
|
|
46
|
+
const clone = parse(element.outerHTML, {
|
|
47
|
+
blockTextElements: { noscript: true, pre: true, script: true, style: true },
|
|
48
|
+
comment: false,
|
|
49
|
+
});
|
|
50
|
+
for (const node of clone.querySelectorAll("script, style, pre, code")) {
|
|
51
|
+
node.remove();
|
|
52
|
+
}
|
|
53
|
+
return clone.structuredText.replaceAll(/\s+/gu, " ").trim();
|
|
54
|
+
};
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pixel dimensions read straight from a PNG, JPEG, or GIF header. A dedicated
|
|
3
|
+
* image library would be a dependency for three well-documented byte layouts;
|
|
4
|
+
* anything else (SVG, WebP, AVIF) yields null and its checks simply don't run.
|
|
5
|
+
*/
|
|
6
|
+
export interface ImageSize {
|
|
7
|
+
width: number;
|
|
8
|
+
height: number;
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
const PNG_SIGNATURE = Buffer.from([0x89, 0x50, 0x4e, 0x47]);
|
|
12
|
+
|
|
13
|
+
const pngSize = (bytes: Buffer): ImageSize | null => {
|
|
14
|
+
// Signature, then the IHDR chunk is required to come first: width and height
|
|
15
|
+
// are big-endian u32s at fixed offsets 16 and 20.
|
|
16
|
+
if (bytes.length < 24 || !bytes.subarray(0, 4).equals(PNG_SIGNATURE)) {
|
|
17
|
+
return null;
|
|
18
|
+
}
|
|
19
|
+
return { height: bytes.readUInt32BE(20), width: bytes.readUInt32BE(16) };
|
|
20
|
+
};
|
|
21
|
+
|
|
22
|
+
/** JPEG start-of-frame markers (C0–CF minus DHT C4, JPG C8, DAC CC). */
|
|
23
|
+
const isSof = (marker: number): boolean =>
|
|
24
|
+
marker >= 0xc0 &&
|
|
25
|
+
marker <= 0xcf &&
|
|
26
|
+
marker !== 0xc4 &&
|
|
27
|
+
marker !== 0xc8 &&
|
|
28
|
+
marker !== 0xcc;
|
|
29
|
+
|
|
30
|
+
const jpegSize = (bytes: Buffer): ImageSize | null => {
|
|
31
|
+
if (bytes.length < 4 || bytes[0] !== 0xff || bytes[1] !== 0xd8) {
|
|
32
|
+
return null;
|
|
33
|
+
}
|
|
34
|
+
// Walk the segment list: each is FF <marker> <u16 length> <payload>. The
|
|
35
|
+
// dimensions live in the first start-of-frame segment's payload, as
|
|
36
|
+
// big-endian u16s after a one-byte precision field.
|
|
37
|
+
let offset = 2;
|
|
38
|
+
while (offset + 9 < bytes.length) {
|
|
39
|
+
if (bytes[offset] !== 0xff) {
|
|
40
|
+
return null;
|
|
41
|
+
}
|
|
42
|
+
const marker = bytes[offset + 1] ?? 0;
|
|
43
|
+
if (isSof(marker)) {
|
|
44
|
+
return {
|
|
45
|
+
height: bytes.readUInt16BE(offset + 5),
|
|
46
|
+
width: bytes.readUInt16BE(offset + 7),
|
|
47
|
+
};
|
|
48
|
+
}
|
|
49
|
+
offset += 2 + bytes.readUInt16BE(offset + 2);
|
|
50
|
+
}
|
|
51
|
+
return null;
|
|
52
|
+
};
|
|
53
|
+
|
|
54
|
+
const gifSize = (bytes: Buffer): ImageSize | null => {
|
|
55
|
+
if (bytes.length < 10 || bytes.subarray(0, 4).toString("latin1") !== "GIF8") {
|
|
56
|
+
return null;
|
|
57
|
+
}
|
|
58
|
+
return { height: bytes.readUInt16LE(8), width: bytes.readUInt16LE(6) };
|
|
59
|
+
};
|
|
60
|
+
|
|
61
|
+
/** The image's pixel dimensions, or null when the format isn't recognized. */
|
|
62
|
+
export const imageSize = (bytes: Buffer): ImageSize | null =>
|
|
63
|
+
pngSize(bytes) ?? jpegSize(bytes) ?? gifSize(bytes);
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import { locateFrontmatterKey } from "../core/diagnostics.ts";
|
|
2
|
+
import type { FindingSite } from "./catalog.ts";
|
|
3
|
+
import type { AuditContext, PageSnapshot } from "./types.ts";
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Where to report a finding about `page`: the built URL, plus — when the page
|
|
7
|
+
* came from authored content — the source file and the exact front matter line
|
|
8
|
+
* to edit.
|
|
9
|
+
*
|
|
10
|
+
* This is the feature. A crawler can only tell you that `/docs/api` has no
|
|
11
|
+
* description; Blume knows the page was built from `docs/api.mdx` and can put
|
|
12
|
+
* the cursor on the line. Pass the front matter key path a fix would touch
|
|
13
|
+
* (`["description"]`, `["seo", "canonical"]`); omit it when the key doesn't
|
|
14
|
+
* exist yet, and the finding anchors to the file instead.
|
|
15
|
+
*/
|
|
16
|
+
export const pageSite = (
|
|
17
|
+
context: AuditContext,
|
|
18
|
+
page: PageSnapshot,
|
|
19
|
+
key?: readonly (string | number)[]
|
|
20
|
+
): FindingSite => {
|
|
21
|
+
const file = page.source;
|
|
22
|
+
if (!file) {
|
|
23
|
+
return { url: page.url };
|
|
24
|
+
}
|
|
25
|
+
const source = key ? context.sources.get(file) : undefined;
|
|
26
|
+
const position = source ? locateFrontmatterKey(source, key ?? []) : undefined;
|
|
27
|
+
return {
|
|
28
|
+
column: position?.column,
|
|
29
|
+
file,
|
|
30
|
+
line: position?.line,
|
|
31
|
+
url: page.url,
|
|
32
|
+
};
|
|
33
|
+
};
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
import type { RedirectResolution } from "./types.ts";
|
|
2
|
+
import { normalizePath } from "./url.ts";
|
|
3
|
+
|
|
4
|
+
interface ConfiguredRedirect {
|
|
5
|
+
from: string;
|
|
6
|
+
to: string;
|
|
7
|
+
status: number;
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Follow every configured redirect through to its destination, classifying what
|
|
12
|
+
* it lands on.
|
|
13
|
+
*
|
|
14
|
+
* - `loop` — the chain revisits a hop it has already been to. Never resolves.
|
|
15
|
+
* - `broken` — the chain ends somewhere that isn't a built page.
|
|
16
|
+
* - `chain` — it resolves, but through at least one intermediate redirect.
|
|
17
|
+
* - `ok` — one hop, straight to a real page.
|
|
18
|
+
*
|
|
19
|
+
* An external destination (`https://…`) is always `ok`: it's outside the site,
|
|
20
|
+
* so there's no local page to check it against.
|
|
21
|
+
*/
|
|
22
|
+
export const resolveRedirects = (
|
|
23
|
+
redirects: readonly ConfiguredRedirect[],
|
|
24
|
+
pageUrls: ReadonlySet<string>
|
|
25
|
+
): RedirectResolution[] => {
|
|
26
|
+
const byFrom = new Map<string, ConfiguredRedirect>();
|
|
27
|
+
for (const redirect of redirects) {
|
|
28
|
+
byFrom.set(normalizePath(redirect.from), redirect);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
return redirects.map((redirect) => {
|
|
32
|
+
const from = normalizePath(redirect.from);
|
|
33
|
+
const chain: string[] = [from];
|
|
34
|
+
const seen = new Set<string>([from]);
|
|
35
|
+
let current = redirect.to;
|
|
36
|
+
|
|
37
|
+
for (;;) {
|
|
38
|
+
// An external hop ends the walk — we can't follow it locally.
|
|
39
|
+
if (/^https?:\/\//iu.test(current)) {
|
|
40
|
+
chain.push(current);
|
|
41
|
+
break;
|
|
42
|
+
}
|
|
43
|
+
const next = normalizePath(current);
|
|
44
|
+
if (seen.has(next)) {
|
|
45
|
+
chain.push(next);
|
|
46
|
+
return {
|
|
47
|
+
...redirect,
|
|
48
|
+
chain,
|
|
49
|
+
outcome: "loop" as const,
|
|
50
|
+
};
|
|
51
|
+
}
|
|
52
|
+
chain.push(next);
|
|
53
|
+
seen.add(next);
|
|
54
|
+
const hop = byFrom.get(next);
|
|
55
|
+
if (!hop) {
|
|
56
|
+
break;
|
|
57
|
+
}
|
|
58
|
+
current = hop.to;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
const destination = chain.at(-1) ?? from;
|
|
62
|
+
const external = /^https?:\/\//iu.test(destination);
|
|
63
|
+
if (!(external || pageUrls.has(destination))) {
|
|
64
|
+
return { ...redirect, chain, outcome: "broken" as const };
|
|
65
|
+
}
|
|
66
|
+
// `chain` is [from, …hops, destination]; more than two entries means at
|
|
67
|
+
// least one intermediate redirect.
|
|
68
|
+
return {
|
|
69
|
+
...redirect,
|
|
70
|
+
chain,
|
|
71
|
+
outcome: chain.length > 2 ? ("chain" as const) : ("ok" as const),
|
|
72
|
+
};
|
|
73
|
+
});
|
|
74
|
+
};
|