@nialto-services/astro-site-checks 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Nialto Services
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,98 @@
1
+ # @nialto-services/astro-site-checks
2
+
3
+ Post-build checks for Nialto's Astro sites. Each one reads the built site in `dist/client` and fails when
4
+ something is wrong that the build itself never complains about: a link to a page that does not exist, a
5
+ redirect that has stopped working, a scoped style rule that can never apply, or a page loading from an
6
+ origin the site's Content-Security-Policy does not allow.
7
+
8
+ The checks used to live as scripts copied into each site, and the copies drifted. This package is the one
9
+ copy every site runs.
10
+
11
+ ## Install
12
+
13
+ ```sh
14
+ pnpm add -D @nialto-services/astro-site-checks
15
+ ```
16
+
17
+ Then add the four scripts every site uses to its `package.json`:
18
+
19
+ ```json
20
+ "lint:links": "astro-site-checks links",
21
+ "lint:redirects": "astro-site-checks redirects",
22
+ "lint:css-scope": "astro-site-checks css-scope",
23
+ "lint:headers": "astro-site-checks headers"
24
+ ```
25
+
26
+ Every check reads the built site, so run it after `astro build`. Each one checks `dist/client` by default;
27
+ pass `--dir <path>` to check another directory.
28
+
29
+ ## The checks
30
+
31
+ ### `links`
32
+
33
+ Fails when a root-relative link in any built page points at a path the build does not serve. Each broken
34
+ target is listed once, with up to five of the pages that link to it.
35
+
36
+ A link to the source of a redirect is a warning rather than a failure: it resolves, but through a 301, so it
37
+ is worth linking straight to the destination.
38
+
39
+ Links to other hosts, including protocol-relative ones (`//cdn.example.com/…`), `mailto:` and `tel:` links,
40
+ in-page anchors and absolute links back to the site's own host are not checked.
41
+
42
+ ### `redirects`
43
+
44
+ Checks every rule in the built `_redirects` for four faults:
45
+
46
+ - a target the build does not serve;
47
+ - a source the build serves as a real page, so the rule can never fire;
48
+ - a target that is itself the source of another rule, which makes a chain;
49
+ - a page rule without its other trailing-slash spelling pointing at the same target. Cloudflare matches a
50
+ rule's source exactly, so without both spellings one of them lands on a 404.
51
+
52
+ A placeholder (`:name`, `*`) or a target on another host is skipped only by the faults that would have to
53
+ resolve it: a placeholder rule's concrete target is still checked, and so is the source of a rule pointing at
54
+ another host. A site with no `_redirects` passes.
55
+
56
+ ### `css-scope`
57
+
58
+ Fails when a scoped style rule can never match the element it targets. Astro scopes a component's styles to
59
+ the elements that component renders, so styling a child component's root element by class compiles to a rule
60
+ that silently does nothing. A class that never appears in the built pages is left alone, since it is either
61
+ added at runtime or belongs to a component nothing renders.
62
+
63
+ ### `headers`
64
+
65
+ Fails when a built page loads a script, image, frame, stylesheet or preloaded file from an origin the site's
66
+ Content-Security-Policy does not allow. The enforced `Content-Security-Policy` header is read if the site ships
67
+ one, otherwise `Content-Security-Policy-Report-Only`.
68
+
69
+ Every site ships a policy, so a site whose built `_headers` has none fails this check. Origins a tag manager
70
+ injects while the page runs cannot be seen by a check of the built files.
71
+
72
+ ## Configuration
73
+
74
+ A site sets options under `astroSiteChecks` in its `package.json`. There is one:
75
+
76
+ ```json
77
+ "astroSiteChecks": {
78
+ "headers": {
79
+ "ungovernedPaths": ["mockups/"]
80
+ }
81
+ }
82
+ ```
83
+
84
+ `headers.ungovernedPaths` lists path prefixes, relative to the built site, whose pages the policy does not
85
+ govern: pages a site serves but did not author, such as design mock-ups reproduced as they were delivered.
86
+ The `headers` check skips them.
87
+
88
+ ## Exit codes
89
+
90
+ | Code | Meaning |
91
+ | ---- | -------------------------------------- |
92
+ | 0 | The check passed. |
93
+ | 1 | The check failed; the output says why. |
94
+ | 2 | Unknown or missing check name. |
95
+
96
+ ## Licence
97
+
98
+ MIT
@@ -0,0 +1,40 @@
1
+ #!/usr/bin/env node
2
+ import { resolve } from 'node:path'
3
+ import { DEFAULT_CLIENT_DIRECTORY } from '../src/built-site.mjs'
4
+ import { checkHeaders, checkLinks, checkRedirects, checkScopedCSS } from '../src/index.mjs'
5
+
6
+ const CHECKS = { links: checkLinks, redirects: checkRedirects, 'css-scope': checkScopedCSS, headers: checkHeaders }
7
+
8
+ const USAGE = `Usage: astro-site-checks <${Object.keys(CHECKS).join('|')}> [--dir <built client directory>]
9
+
10
+ Runs one post-build check against the built site (default: ${DEFAULT_CLIENT_DIRECTORY}). Run after astro build.`
11
+
12
+ const [name, ...options] = process.argv.slice(2)
13
+
14
+ if (name === '--help' || name === '-h') {
15
+ console.log(USAGE)
16
+ process.exit(0)
17
+ }
18
+
19
+ const check = CHECKS[name]
20
+ if (!check) {
21
+ console.error(USAGE)
22
+ process.exit(2)
23
+ }
24
+
25
+ const directoryIndex = options.indexOf('--dir')
26
+ const directory = directoryIndex === -1 ? DEFAULT_CLIENT_DIRECTORY : options[directoryIndex + 1]
27
+ if (!directory || directory.startsWith('--')) {
28
+ console.error(USAGE)
29
+ process.exit(2)
30
+ }
31
+
32
+ // A site's own mistake, such as an unreadable package.json, is reported as its message: a stack trace
33
+ // would read as a fault in this package.
34
+ try {
35
+ const passed = await check({ clientDirectory: resolve(directory), siteDirectory: process.cwd(), report: console })
36
+ process.exit(passed ? 0 : 1)
37
+ } catch (error) {
38
+ console.error(error instanceof Error ? error.message : String(error))
39
+ process.exit(1)
40
+ }
package/package.json ADDED
@@ -0,0 +1,40 @@
1
+ {
2
+ "name": "@nialto-services/astro-site-checks",
3
+ "version": "0.1.1",
4
+ "description": "Post-build checks for Nialto's Astro sites: internal links, redirects, scoped CSS and the Content-Security-Policy.",
5
+ "license": "MIT",
6
+ "type": "module",
7
+ "packageManager": "pnpm@10.30.2",
8
+ "engines": {
9
+ "node": ">=22.12"
10
+ },
11
+ "publishConfig": {
12
+ "access": "public"
13
+ },
14
+ "repository": {
15
+ "type": "git",
16
+ "url": "git+https://github.com/NialtoServices/astro-site-checks.git"
17
+ },
18
+ "bin": {
19
+ "astro-site-checks": "./bin/astro-site-checks.mjs"
20
+ },
21
+ "exports": {
22
+ ".": "./src/index.mjs"
23
+ },
24
+ "files": [
25
+ "bin",
26
+ "src",
27
+ "LICENSE",
28
+ "README.md"
29
+ ],
30
+ "scripts": {
31
+ "test": "vitest run",
32
+ "format": "prettier --write .",
33
+ "format:check": "prettier --check ."
34
+ },
35
+ "devDependencies": {
36
+ "@ianvs/prettier-plugin-sort-imports": "^4.7.1",
37
+ "prettier": "^3.9.9",
38
+ "vitest": "^5.0.2"
39
+ }
40
+ }
@@ -0,0 +1,75 @@
1
+ /**
2
+ * A built Astro site as the checks need to see it: which files the client directory holds, which
3
+ * paths those serve, and which internal links a page carries.
4
+ *
5
+ * Every check reads the site through these, so the link check and the redirect check cannot drift
6
+ * apart over what "this path resolves" means.
7
+ */
8
+ import { readdir } from 'node:fs/promises'
9
+ import { join, relative } from 'node:path'
10
+
11
+ /** Where `astro build` writes the files a site serves, relative to the site's root. */
12
+ export const DEFAULT_CLIENT_DIRECTORY = 'dist/client'
13
+
14
+ const HREF_PATTERN = /href="([^"]+)"/g
15
+
16
+ /** Every file under a directory, at any depth. */
17
+ export async function collectBuiltFiles(directory) {
18
+ const entries = await readdir(directory, { withFileTypes: true })
19
+ const files = await Promise.all(
20
+ entries.map((entry) => {
21
+ const path = join(directory, entry.name)
22
+ return entry.isDirectory() ? collectBuiltFiles(path) : [path]
23
+ })
24
+ )
25
+
26
+ return files.flat()
27
+ }
28
+
29
+ /** A built file's path as it would appear in an href, so `…/index.html` reads as its directory. */
30
+ export function servedPathOf(directory, file) {
31
+ return `/${relative(directory, file)}`.replace(/index\.html$/, '')
32
+ }
33
+
34
+ /** Every path the built site can serve, as it would appear in an href. */
35
+ export async function collectServablePaths(directory) {
36
+ const servablePaths = new Set()
37
+ for (const file of await collectBuiltFiles(directory)) {
38
+ const path = `/${relative(directory, file)}`
39
+ servablePaths.add(path)
40
+
41
+ if (path.endsWith('/index.html')) servablePaths.add(path.replace(/index\.html$/, ''))
42
+ }
43
+
44
+ return servablePaths
45
+ }
46
+
47
+ /**
48
+ * Whether a path resolves against the built site, allowing for the trailing slash
49
+ * `trailingSlash: 'always'` adds and for a directory's `index.html`.
50
+ */
51
+ export function isServable(servablePaths, path) {
52
+ if (servablePaths.has(path)) return true
53
+
54
+ return servablePaths.has(`${path}/`) || servablePaths.has(`${path}/index.html`)
55
+ }
56
+
57
+ /**
58
+ * Every root-relative href in a page, reduced to a path, with fragments and query strings dropped.
59
+ *
60
+ * `mailto:`, `tel:`, in-page fragments and links to other hosts are out of scope: their targets are not
61
+ * ours to guarantee. A protocol-relative href (`//host/path`) names another host despite its leading
62
+ * slash. Absolute links back to the site's own host are skipped too, because the head's canonical and
63
+ * pagination tags are absolute on purpose; an absolute link in body copy therefore goes unchecked, which
64
+ * is why internal links are authored root-relative.
65
+ */
66
+ export function extractInternalPaths(html) {
67
+ const paths = []
68
+ for (const [, href] of html.matchAll(HREF_PATTERN)) {
69
+ if (!href.startsWith('/') || href.startsWith('//')) continue
70
+
71
+ paths.push(href.replace(/[#?].*$/, ''))
72
+ }
73
+
74
+ return paths
75
+ }
@@ -0,0 +1,18 @@
1
+ import { readFile, stat } from 'node:fs/promises'
2
+
3
+ /** Whether the site has been built, reporting how to fix it when it has not. */
4
+ export async function requireBuiltSite(clientDirectory, report) {
5
+ const found = await stat(clientDirectory).then(
6
+ (entry) => entry.isDirectory(),
7
+ () => false
8
+ )
9
+ if (found) return true
10
+
11
+ report.error(`No built site at ${clientDirectory}. Run \`astro build\` first.`)
12
+ return false
13
+ }
14
+
15
+ /** A built file's contents, or `undefined` when the build did not produce it. */
16
+ export function readOptionalFile(path) {
17
+ return readFile(path, 'utf-8').catch(() => undefined)
18
+ }
@@ -0,0 +1,47 @@
1
+ /** Fails when a scoped CSS rule in the built site can never match the element it targets. */
2
+ import { readFile } from 'node:fs/promises'
3
+ import { relative } from 'node:path'
4
+ import { collectBuiltFiles, servedPathOf } from '../built-site.mjs'
5
+ import { extractStyleBlocks, findUnmatchedScopedRules } from '../scoped-css.mjs'
6
+ import { requireBuiltSite } from './built.mjs'
7
+
8
+ const SOURCES_LISTED_PER_RULE = 3
9
+
10
+ export async function checkScopedCSS({ clientDirectory, report }) {
11
+ if (!(await requireBuiltSite(clientDirectory, report))) return false
12
+
13
+ const files = await collectBuiltFiles(clientDirectory)
14
+ const pages = []
15
+ const styleSheets = []
16
+ for (const file of files.filter((path) => path.endsWith('.html'))) {
17
+ const markup = await readFile(file, 'utf-8')
18
+ const source = servedPathOf(clientDirectory, file)
19
+
20
+ pages.push({ source, markup })
21
+ for (const css of extractStyleBlocks(markup)) styleSheets.push({ source, css })
22
+ }
23
+
24
+ for (const file of files.filter((path) => path.endsWith('.css'))) {
25
+ styleSheets.push({ source: relative(clientDirectory, file), css: await readFile(file, 'utf-8') })
26
+ }
27
+
28
+ const unmatched = findUnmatchedScopedRules(pages, styleSheets)
29
+ if (unmatched.size > 0) {
30
+ report.error(`\nScoped rules that can never match the element they target (${unmatched.size}):\n`)
31
+ for (const [selector, rule] of unmatched) {
32
+ report.error(` ${selector}`)
33
+ report.error(` .${rule.className} only ever renders with: ${Array.from(rule.rendered).join(', ')}`)
34
+ report.error(` declared in ${Array.from(rule.sources).slice(0, SOURCES_LISTED_PER_RULE).join(', ')}`)
35
+ }
36
+ report.error(
37
+ '\nA component cannot style an element it does not render. Pass a prop, or declare a custom\n' +
38
+ 'property on an element it does own and let it inherit down.\n'
39
+ )
40
+ return false
41
+ }
42
+
43
+ report.log(
44
+ `Scoped CSS OK: checked ${styleSheets.length} stylesheets against ${pages.length} pages, every rule can match.`
45
+ )
46
+ return true
47
+ }
@@ -0,0 +1,55 @@
1
+ /**
2
+ * Fails when a built page loads an origin the site's Content-Security-Policy does not allow, or when the
3
+ * site ships no policy at all.
4
+ *
5
+ * A report-only policy lets the browser load a disallowed resource anyway, so a new third-party include
6
+ * goes unnoticed until the policy is enforced and it breaks. This closes that gap for everything the markup
7
+ * itself fetches. Origins a tag manager injects at runtime remain invisible to a static check.
8
+ */
9
+ import { readFile } from 'node:fs/promises'
10
+ import { join, relative } from 'node:path'
11
+ import { collectBuiltFiles, servedPathOf } from '../built-site.mjs'
12
+ import { readConfig } from '../config.mjs'
13
+ import { findPolicyViolations, parsePolicy } from '../csp.mjs'
14
+ import { readOptionalFile, requireBuiltSite } from './built.mjs'
15
+
16
+ export async function checkHeaders({ clientDirectory, siteDirectory, report }) {
17
+ if (!(await requireBuiltSite(clientDirectory, report))) return false
18
+
19
+ const { ungovernedPaths } = (await readConfig(siteDirectory)).headers
20
+ const policy = parsePolicy((await readOptionalFile(join(clientDirectory, '_headers'))) ?? '')
21
+ if (!policy) {
22
+ report.error(
23
+ 'No Content-Security-Policy in the built _headers. Every site ships one: add a ' +
24
+ 'Content-Security-Policy-Report-Only header for /* to public/_headers.'
25
+ )
26
+ return false
27
+ }
28
+
29
+ const pages = (await collectBuiltFiles(clientDirectory)).filter((file) => {
30
+ if (!file.endsWith('.html')) return false
31
+
32
+ const route = relative(clientDirectory, file)
33
+ return !ungovernedPaths.some((prefix) => route.startsWith(prefix))
34
+ })
35
+
36
+ const problems = []
37
+ for (const page of pages) {
38
+ for (const violation of findPolicyViolations(await readFile(page, 'utf-8'), policy)) {
39
+ problems.push({ page: servedPathOf(clientDirectory, page), ...violation })
40
+ }
41
+ }
42
+
43
+ if (problems.length > 0) {
44
+ report.error(`\nResources loaded from an origin the Content-Security-Policy does not allow (${problems.length}):\n`)
45
+ for (const problem of problems) {
46
+ report.error(` ${problem.origin} (<${problem.kind}>, needs ${problem.directive})`)
47
+ report.error(` loaded by ${problem.page}`)
48
+ }
49
+ report.error('')
50
+ return false
51
+ }
52
+
53
+ report.log(`Content-Security-Policy OK: checked ${pages.length} pages, every loaded origin is allowed.`)
54
+ return true
55
+ }
@@ -0,0 +1,105 @@
1
+ /**
2
+ * Fails when an internal link in the built site points at a page that does not exist.
3
+ *
4
+ * Content work is exactly when dangling links creep in: pages cross-reference each other, slugs get
5
+ * renamed, and a link to a page that never gets written is invisible until a visitor hits it.
6
+ *
7
+ * A link to a redirect *source* is a warning rather than a failure. It resolves, so nobody lands on a 404,
8
+ * but the visitor pays a needless 301, so it is worth linking straight to the destination.
9
+ */
10
+ import { readFile } from 'node:fs/promises'
11
+ import { join } from 'node:path'
12
+ import {
13
+ collectBuiltFiles,
14
+ collectServablePaths,
15
+ extractInternalPaths,
16
+ isServable,
17
+ servedPathOf
18
+ } from '../built-site.mjs'
19
+ import { parseRedirects } from '../redirects.mjs'
20
+ import { readOptionalFile, requireBuiltSite } from './built.mjs'
21
+
22
+ const SOURCES_LISTED_PER_LINK = 5
23
+
24
+ /** Pages that link to each path, so a repeated mistake reads as one entry with its sources. */
25
+ function groupByPath(findings) {
26
+ const grouped = new Map()
27
+ for (const { from, path } of findings) {
28
+ const sources = grouped.get(path) ?? new Set()
29
+ sources.add(from)
30
+ grouped.set(path, sources)
31
+ }
32
+
33
+ return grouped
34
+ }
35
+
36
+ /**
37
+ * A link in shared chrome appears on every page, and a list that long buries the other findings, so each
38
+ * one names a handful of pages and counts the rest.
39
+ */
40
+ function listFindings(grouped, describeTarget, write) {
41
+ for (const [path, sources] of grouped) {
42
+ write(` ${path}${describeTarget(path)}`)
43
+
44
+ const named = Array.from(sources).slice(0, SOURCES_LISTED_PER_LINK)
45
+ for (const source of named) write(` linked from ${source}`)
46
+
47
+ const remaining = sources.size - named.length
48
+ if (remaining > 0) write(` …and ${remaining} more page${remaining === 1 ? '' : 's'}`)
49
+ }
50
+ }
51
+
52
+ export async function checkLinks({ clientDirectory, report }) {
53
+ if (!(await requireBuiltSite(clientDirectory, report))) return false
54
+
55
+ const redirects = parseRedirects((await readOptionalFile(join(clientDirectory, '_redirects'))) ?? '')
56
+ const servablePaths = await collectServablePaths(clientDirectory)
57
+ const pages = (await collectBuiltFiles(clientDirectory)).filter((file) => file.endsWith('.html'))
58
+
59
+ const broken = []
60
+ const redirected = []
61
+ for (const page of pages) {
62
+ const html = await readFile(page, 'utf-8')
63
+ const from = servedPathOf(clientDirectory, page)
64
+
65
+ for (const path of extractInternalPaths(html)) {
66
+ if (redirects.has(path)) {
67
+ redirected.push({ from, path })
68
+ continue
69
+ }
70
+
71
+ if (isServable(servablePaths, path)) continue
72
+
73
+ broken.push({ from, path })
74
+ }
75
+ }
76
+
77
+ if (redirected.length > 0) {
78
+ const grouped = groupByPath(redirected)
79
+ report.warn(`\nInternal links pointing at a redirect source (${grouped.size}):\n`)
80
+ listFindings(
81
+ grouped,
82
+ (path) => ` -> ${redirects.get(path)}`,
83
+ (line) => report.warn(line)
84
+ )
85
+ report.warn('\nThese resolve, but through a 301. Link straight to the destination.\n')
86
+ }
87
+
88
+ if (broken.length > 0) {
89
+ const grouped = groupByPath(broken)
90
+ report.error(`\nBroken internal links (${grouped.size}):\n`)
91
+ listFindings(
92
+ grouped,
93
+ () => '',
94
+ (line) => report.error(line)
95
+ )
96
+ report.error('')
97
+ return false
98
+ }
99
+
100
+ report.log(
101
+ `Internal links OK: checked ${pages.length} pages, every internal link resolves` +
102
+ `${redirected.length > 0 ? `, ${redirected.length} through a redirect` : ''}.`
103
+ )
104
+ return true
105
+ }
@@ -0,0 +1,75 @@
1
+ /**
2
+ * Fails when a rule in the built `_redirects` can no longer do its job: a target the build does not serve,
3
+ * a source the build serves as a real page (so the rule can never fire), a target that is itself another
4
+ * rule's source (a chain), or a page rule carried in only one trailing-slash spelling.
5
+ *
6
+ * A site with nothing to redirect has no `_redirects`, which passes.
7
+ */
8
+ import { join } from 'node:path'
9
+ import { collectServablePaths } from '../built-site.mjs'
10
+ import {
11
+ findBrokenTargets,
12
+ findMissingSpellings,
13
+ findRedirectChains,
14
+ findShadowedSources,
15
+ parseRedirects
16
+ } from '../redirects.mjs'
17
+ import { readOptionalFile, requireBuiltSite } from './built.mjs'
18
+
19
+ function reportFaults(report, heading, faults, describe, remedy) {
20
+ if (faults.length === 0) return false
21
+
22
+ report.error(`\n${heading} (${faults.length}):\n`)
23
+ for (const fault of faults) report.error(` ${describe(fault)}`)
24
+ report.error(`\n${remedy}\n`)
25
+
26
+ return true
27
+ }
28
+
29
+ export async function checkRedirects({ clientDirectory, report }) {
30
+ if (!(await requireBuiltSite(clientDirectory, report))) return false
31
+
32
+ const text = await readOptionalFile(join(clientDirectory, '_redirects'))
33
+ if (text === undefined) {
34
+ report.log('No _redirects in the built site: nothing to check.')
35
+ return true
36
+ }
37
+
38
+ const redirects = parseRedirects(text)
39
+ const servablePaths = await collectServablePaths(clientDirectory)
40
+
41
+ const failures = [
42
+ reportFaults(
43
+ report,
44
+ 'Redirects pointing at a path the build does not serve',
45
+ findBrokenTargets(redirects, servablePaths),
46
+ ({ source, target }) => `${source} -> ${target}`,
47
+ 'Point the rule at a page that exists, or drop it if the destination has gone for good.'
48
+ ),
49
+ reportFaults(
50
+ report,
51
+ 'Redirects whose source the build serves as a page, so the rule can never fire',
52
+ findShadowedSources(redirects, servablePaths),
53
+ ({ source, target }) => `${source} -> ${target} (a page is built at ${source})`,
54
+ 'Either remove the rule or stop building the page.'
55
+ ),
56
+ reportFaults(
57
+ report,
58
+ 'Redirect chains',
59
+ findRedirectChains(redirects),
60
+ ({ source, target, eventualTarget }) => `${source} -> ${target} -> ${eventualTarget}`,
61
+ 'Flatten the chain: point the first rule straight at the page the second one reaches.'
62
+ ),
63
+ reportFaults(
64
+ report,
65
+ 'Redirects missing their other trailing-slash spelling',
66
+ findMissingSpellings(redirects),
67
+ ({ source, missing, target }) => `${source} -> ${target} (add ${missing} -> ${target})`,
68
+ 'Cloudflare matches a source exactly, so give every page rule both spellings, pointing at the same page.'
69
+ )
70
+ ]
71
+ if (failures.some(Boolean)) return false
72
+
73
+ report.log(`Redirects OK: ${redirects.size} rules checked, every target resolves in one hop.`)
74
+ return true
75
+ }
package/src/config.mjs ADDED
@@ -0,0 +1,26 @@
1
+ import { readFile } from 'node:fs/promises'
2
+ import { join } from 'node:path'
3
+
4
+ /**
5
+ * A site's options for the checks, read from `astroSiteChecks` in its package.json so there is no extra
6
+ * config file to keep. A site that sets nothing, or has no package.json, gets the defaults.
7
+ */
8
+ export async function readConfig(siteDirectory) {
9
+ const text = await readFile(join(siteDirectory, 'package.json'), 'utf-8').catch(() => '{}')
10
+
11
+ let manifest
12
+ try {
13
+ manifest = JSON.parse(text)
14
+ } catch (error) {
15
+ throw new Error(`The site's package.json is not valid JSON: ${error.message}`, { cause: error })
16
+ }
17
+
18
+ const ungovernedPaths = manifest.astroSiteChecks?.headers?.ungovernedPaths ?? []
19
+
20
+ const isPathList = Array.isArray(ungovernedPaths) && ungovernedPaths.every((path) => typeof path === 'string')
21
+ if (!isPathList) {
22
+ throw new Error('astroSiteChecks.headers.ungovernedPaths in package.json must be a list of path prefixes')
23
+ }
24
+
25
+ return { headers: { ungovernedPaths } }
26
+ }
package/src/csp.mjs ADDED
@@ -0,0 +1,112 @@
1
+ /**
2
+ * Checks the built pages against the shipped Content-Security-Policy.
3
+ *
4
+ * A report-only policy tells nobody anything until real traffic hits it, and the origins a page loads
5
+ * change as components are added. This closes the gap for the ones the build can see: anything the
6
+ * markup itself fetches is compared to the policy, so a new external script fails a check here rather
7
+ * than a browser console after launch. Origins the GTM container injects at runtime are invisible to
8
+ * this and remain the reason the policy stays report-only.
9
+ */
10
+
11
+ const POLICY_HEADERS = ['Content-Security-Policy', 'Content-Security-Policy-Report-Only']
12
+
13
+ /**
14
+ * The directives of the site's policy in a `_headers` file: the enforced header when it ships, since that
15
+ * is the one a browser applies, else the report-only one. A header name inside a comment is not a header.
16
+ */
17
+ export function parsePolicy(headers) {
18
+ for (const headerName of POLICY_HEADERS) {
19
+ const match = headers.match(new RegExp(`^\\s*${headerName}: (.*)$`, 'mi'))
20
+ if (!match) continue
21
+
22
+ const directives = new Map()
23
+ for (const part of match[1].split(';')) {
24
+ const [name, ...values] = part.trim().split(/\s+/)
25
+ if (name) directives.set(name, values)
26
+ }
27
+
28
+ return directives
29
+ }
30
+
31
+ return undefined
32
+ }
33
+
34
+ /** An attribute's value on one element, or `undefined` when it is absent. */
35
+ function attributeOf(element, name) {
36
+ return element.match(new RegExp(`\\s${name}="([^"]*)"`, 'i'))?.[1]
37
+ }
38
+
39
+ /** What a `<link>` loads, judged by its `rel` tokens and, for a preload, what it preloads. */
40
+ const PRELOAD_KIND_FOR_AS = { font: 'font', style: 'style', script: 'script', image: 'img' }
41
+
42
+ /** Every externally loaded resource in a page, as `{ kind, origin }`. Links are not loads. */
43
+ export function extractLoadedOrigins(html) {
44
+ const loaded = []
45
+ for (const match of html.matchAll(/<(script|img|iframe)\b[^>]*?\ssrc="(https?:\/\/[^"]+)"/g)) {
46
+ loaded.push({ kind: match[1], origin: new URL(match[2]).origin })
47
+ }
48
+
49
+ // Attributes are read from the whole element, since a `<link>` carries `rel`, `href` and `as` in
50
+ // whatever order its author wrote them.
51
+ for (const [element] of html.matchAll(/<link\b[^>]*>/gi)) {
52
+ const href = attributeOf(element, 'href')
53
+ if (!href || !/^https?:\/\//i.test(href)) continue
54
+
55
+ const relations = (attributeOf(element, 'rel') ?? '').toLowerCase().split(/\s+/)
56
+ if (relations.includes('stylesheet')) {
57
+ loaded.push({ kind: 'style', origin: new URL(href).origin })
58
+ } else if (relations.includes('preload')) {
59
+ const preloaded = (attributeOf(element, 'as') ?? '').toLowerCase()
60
+ loaded.push({ kind: PRELOAD_KIND_FOR_AS[preloaded] ?? 'preload', origin: new URL(href).origin })
61
+ }
62
+ }
63
+
64
+ return loaded
65
+ }
66
+
67
+ const DIRECTIVE_FOR_KIND = {
68
+ script: 'script-src',
69
+ img: 'img-src',
70
+ iframe: 'frame-src',
71
+ style: 'style-src',
72
+ font: 'font-src',
73
+ preload: 'default-src'
74
+ }
75
+
76
+ /**
77
+ * Whether a policy source covers an origin. A source may be a scheme (`https:`), a host with or without a
78
+ * scheme, a host wildcard (`*.a.com`, which covers subdomains and not the domain itself), and may carry a
79
+ * path, which does not narrow the origin it allows.
80
+ */
81
+ export function originAllowedBy(source, origin) {
82
+ if (source === '*' || source === origin) return true
83
+
84
+ const { protocol, host } = new URL(origin)
85
+ if (/^[a-z][a-z0-9+.-]*:$/i.test(source)) return source.toLowerCase() === protocol
86
+
87
+ const parts = source.match(/^(?:([a-z][a-z0-9+.-]*):\/\/)?([^/]+)/i)
88
+ if (!parts) return false
89
+
90
+ const [, scheme, sourceHost] = parts
91
+ if (scheme && `${scheme.toLowerCase()}:` !== protocol) return false
92
+ if (!sourceHost.startsWith('*.')) return sourceHost.toLowerCase() === host
93
+
94
+ return host.endsWith(sourceHost.slice(1).toLowerCase())
95
+ }
96
+
97
+ /**
98
+ * Origins a page loads that its policy does not allow, falling back to `default-src` the way a
99
+ * browser does when a directive is absent.
100
+ */
101
+ export function findPolicyViolations(html, policy) {
102
+ const violations = []
103
+ for (const { kind, origin } of extractLoadedOrigins(html)) {
104
+ const directive = DIRECTIVE_FOR_KIND[kind] ?? 'default-src'
105
+ const allowed = policy.get(directive) ?? policy.get('default-src') ?? []
106
+ if (allowed.some((source) => originAllowedBy(source, origin))) continue
107
+
108
+ violations.push({ kind, origin, directive })
109
+ }
110
+
111
+ return violations
112
+ }
package/src/index.mjs ADDED
@@ -0,0 +1,22 @@
1
+ export {
2
+ collectBuiltFiles,
3
+ collectServablePaths,
4
+ DEFAULT_CLIENT_DIRECTORY,
5
+ extractInternalPaths,
6
+ isServable,
7
+ servedPathOf
8
+ } from './built-site.mjs'
9
+ export { checkScopedCSS } from './checks/css-scope.mjs'
10
+ export { checkHeaders } from './checks/headers.mjs'
11
+ export { checkLinks } from './checks/links.mjs'
12
+ export { checkRedirects } from './checks/redirects.mjs'
13
+ export { readConfig } from './config.mjs'
14
+ export { extractLoadedOrigins, findPolicyViolations, originAllowedBy, parsePolicy } from './csp.mjs'
15
+ export {
16
+ findBrokenTargets,
17
+ findMissingSpellings,
18
+ findRedirectChains,
19
+ findShadowedSources,
20
+ parseRedirects
21
+ } from './redirects.mjs'
22
+ export { extractStyleBlocks, findUnmatchedScopedRules } from './scoped-css.mjs'
@@ -0,0 +1,113 @@
1
+ /**
2
+ * The rules in a Cloudflare `_redirects` file, and the four ways a rule stops doing its job, as pure
3
+ * functions over the parsed rules and the paths the build serves.
4
+ *
5
+ * A redirect usually carries an inbound link a previous site earned, so a rule that fails loses that link
6
+ * silently: nothing in the build errors, and the only symptom is a 404 a crawler finds months later.
7
+ */
8
+ import { isServable } from './built-site.mjs'
9
+
10
+ /** `source target status` lines, comments and blanks skipped, as a source → target map. */
11
+ export function parseRedirects(text) {
12
+ const redirects = new Map()
13
+ for (const line of text.split('\n')) {
14
+ const trimmed = line.trim()
15
+ if (trimmed === '' || trimmed.startsWith('#')) continue
16
+
17
+ const [source, target] = trimmed.split(/\s+/)
18
+ redirects.set(source, target)
19
+ }
20
+
21
+ return redirects
22
+ }
23
+
24
+ /**
25
+ * Whether one side of a rule is a concrete path on this site. A placeholder (`:name`, `*`) matches many
26
+ * URLs and a target on another host is not ours to resolve, so each fault is checked only on the side it
27
+ * reads: a placeholder source can still have a broken target, and an external target can still have a
28
+ * shadowed source.
29
+ */
30
+ function isConcretePath(path) {
31
+ return Boolean(path?.startsWith('/')) && !/[:*]/.test(path)
32
+ }
33
+
34
+ /**
35
+ * Rules whose target the build does not serve.
36
+ *
37
+ * A 301 to a missing page is worse than no rule at all: a crawler follows it, finds nothing, and drops the
38
+ * old URL without gaining a replacement.
39
+ */
40
+ export function findBrokenTargets(redirects, servablePaths) {
41
+ const broken = []
42
+ for (const [source, target] of redirects) {
43
+ if (!isConcretePath(target)) continue
44
+ if (isServable(servablePaths, target)) continue
45
+
46
+ broken.push({ source, target })
47
+ }
48
+
49
+ return broken
50
+ }
51
+
52
+ /**
53
+ * Rules whose source the build serves as a real page, so the rule can never fire.
54
+ *
55
+ * A static asset wins over a redirect rule, which makes the rule invisible rather than broken: it usually
56
+ * means a page was rebuilt under a URL that had been retired, and the two intentions now contradict each
57
+ * other.
58
+ */
59
+ export function findShadowedSources(redirects, servablePaths) {
60
+ const shadowed = []
61
+ for (const [source, target] of redirects) {
62
+ if (!isConcretePath(source)) continue
63
+ if (!isServable(servablePaths, source)) continue
64
+
65
+ shadowed.push({ source, target })
66
+ }
67
+
68
+ return shadowed
69
+ }
70
+
71
+ /**
72
+ * Rules whose target is itself the source of another rule.
73
+ *
74
+ * Cloudflare applies one rule per request rather than resolving a chain itself, so a chain costs the
75
+ * visitor a second round trip and hands a search engine a hop's worth of dilution on a link that used to
76
+ * arrive in one.
77
+ */
78
+ export function findRedirectChains(redirects) {
79
+ const chains = []
80
+ for (const [source, target] of redirects) {
81
+ if (!isConcretePath(target)) continue
82
+ if (!redirects.has(target)) continue
83
+
84
+ chains.push({ source, target, eventualTarget: redirects.get(target) })
85
+ }
86
+
87
+ return chains
88
+ }
89
+
90
+ /**
91
+ * Page rules whose other trailing-slash spelling does not lead to the same page.
92
+ *
93
+ * Cloudflare matches a rule's source exactly, and only redirects between the two spellings when a page is
94
+ * built at one of them, which a retired URL never is. A previous site usually answered both, so a backlink
95
+ * or bookmark in the spelling without a rule lands on a 404. A file's path is left alone: nothing requests
96
+ * `/sitemap.xml/`.
97
+ */
98
+ export function findMissingSpellings(redirects) {
99
+ const missing = []
100
+ for (const [source, target] of redirects) {
101
+ if (!isConcretePath(source)) continue
102
+
103
+ const lastSegment = source.replace(/\/$/, '').split('/').pop() ?? ''
104
+ if (source === '/' || lastSegment.includes('.')) continue
105
+
106
+ const otherSpelling = source.endsWith('/') ? source.slice(0, -1) : `${source}/`
107
+ if (redirects.get(otherSpelling) === target) continue
108
+
109
+ missing.push({ source, missing: otherSpelling, target })
110
+ }
111
+
112
+ return missing
113
+ }
@@ -0,0 +1,69 @@
1
+ /**
2
+ * Finds scoped CSS rules that can never match the element they target.
3
+ *
4
+ * Astro scopes a component's styles by appending `[data-astro-cid-…]` to each compound selector, and that
5
+ * attribute only lands on elements the component renders itself. So a component that styles a child
6
+ * component's root element by class (passing `<Child class='x'>` and styling `.x` from the parent) compiles
7
+ * to a selector requiring a scope the element does not carry, and the rule silently does nothing. There is
8
+ * no build error and no console warning, and the un-styled default usually looks deliberate, so one of
9
+ * these can ship and sit unnoticed for months.
10
+ *
11
+ * A rule is reported when its class appears in the built HTML but never alongside the scope the rule
12
+ * demands. A class that never appears at all is left alone: it is either applied at runtime (`.is-active`,
13
+ * `.is-open`) or belongs to a component nothing renders, and neither is this check's business.
14
+ */
15
+
16
+ const SCOPED_SELECTOR_PATTERN = /\.([A-Za-z0-9_-]+)\[data-astro-cid-([a-z0-9]+)\]/g
17
+ const ELEMENT_PATTERN = /<[a-z][a-z0-9-]*\s[^>]*>/gi
18
+ const CLASS_ATTRIBUTE_PATTERN = /\sclass="([^"]*)"/i
19
+ const SCOPE_ATTRIBUTE_PATTERN = /data-astro-cid-([a-z0-9]+)/g
20
+ const STYLE_BLOCK_PATTERN = /<style[^>]*>([\s\S]*?)<\/style>/gi
21
+
22
+ /** Every scope each class is rendered with, so a rule's demand can be checked against reality. */
23
+ function indexRenderedClasses(markup, scopesByClass) {
24
+ for (const element of markup.match(ELEMENT_PATTERN) ?? []) {
25
+ const classMatch = element.match(CLASS_ATTRIBUTE_PATTERN)
26
+ if (!classMatch) continue
27
+
28
+ const scopes = Array.from(element.matchAll(SCOPE_ATTRIBUTE_PATTERN), (match) => match[1])
29
+ for (const className of classMatch[1].split(/\s+/).filter(Boolean)) {
30
+ const known = scopesByClass.get(className) ?? new Set()
31
+ for (const scope of scopes) known.add(scope)
32
+
33
+ scopesByClass.set(className, known)
34
+ }
35
+ }
36
+ }
37
+
38
+ /** The contents of every inline `<style>` block in a page. */
39
+ export function extractStyleBlocks(markup) {
40
+ return Array.from(markup.matchAll(STYLE_BLOCK_PATTERN), (match) => match[1])
41
+ }
42
+
43
+ /**
44
+ * Scoped rules whose class renders somewhere in the built pages but never with the scope the rule
45
+ * demands, keyed by selector, with the scopes the class does render with and every source the rule was
46
+ * declared in.
47
+ */
48
+ export function findUnmatchedScopedRules(pages, styleSheets) {
49
+ const scopesByClass = new Map()
50
+ for (const { markup } of pages) indexRenderedClasses(markup, scopesByClass)
51
+
52
+ const unmatched = new Map()
53
+ for (const { source, css } of styleSheets) {
54
+ for (const [selector, className, scope] of css.matchAll(SCOPED_SELECTOR_PATTERN)) {
55
+ const rendered = scopesByClass.get(className)
56
+ if (!rendered || rendered.has(scope)) continue
57
+
58
+ const existing = unmatched.get(selector)
59
+ if (existing) {
60
+ existing.sources.add(source)
61
+ continue
62
+ }
63
+
64
+ unmatched.set(selector, { className, scope, rendered, sources: new Set([source]) })
65
+ }
66
+ }
67
+
68
+ return unmatched
69
+ }