astro-better-link-checker 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE.md ADDED
@@ -0,0 +1,7 @@
1
+ Copyright (c) 2026 Better Static Sites
2
+
3
+ Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the "Software"), to deal in the Software without restriction, including without limitation the rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit persons to whom the Software is furnished to do so, subject to the following conditions:
4
+
5
+ The above copyright notice and this permission notice shall be included in all copies or substantial portions of the Software.
6
+
7
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,87 @@
1
+ # astro-better-link-checker
2
+
3
+ Fast intra-site broken link checker for Astro. Runs after `astro build` and reports any `href` values that don't resolve to a real file in the build output.
4
+
5
+ External links are not checked. For those, point a tool like [lychee](https://github.com/lycheeverse/lychee) at your deployed site.
6
+
7
+ ## Why faster
8
+
9
+ The common approach checks links per-file: if `/docs/get-started` is linked from 400 pages, it gets stat'd 400 times. This plugin deduplicates first, checking every unique destination exactly once regardless of how many pages reference it. File reads and existence checks all run in parallel.
10
+
11
+ ## Installation
12
+
13
+ ```shell-session
14
+ npm install astro-better-link-checker
15
+ ```
16
+
17
+ ## Usage
18
+
19
+ ```ts
20
+ // astro.config.ts
21
+ import { defineConfig } from 'astro/config';
22
+ import linkChecker from 'astro-better-link-checker';
23
+
24
+ export default defineConfig({
25
+ integrations: [linkChecker()],
26
+ });
27
+ ```
28
+
29
+ ## Options
30
+
31
+ ```ts
32
+ linkChecker({
33
+ // Pages to skip crawling entirely (string = prefix match, RegExp = tested against full path)
34
+ excludeSourcePages: ['/landing/', /^\/preview\//],
35
+
36
+ // Destinations to skip checking (string = prefix match, RegExp = tested against full path)
37
+ excludeDestinations: ['/login', '/logout', /^\/external-tool\//],
38
+
39
+ // Throw and fail the build on broken links (default: true)
40
+ failOnBrokenLinks: true,
41
+
42
+ // Log every checked href, not just broken ones (default: false)
43
+ verbose: false,
44
+ })
45
+ ```
46
+
47
+ Both `excludeSourcePages` and `excludeDestinations` accept an array of strings or `RegExp` objects:
48
+
49
+ - Strings match by prefix: e.g. `'/login'` excludes any path that _starts with_ `/login`, such as `/login`, `/login/callback`, or `/login?next=/docs`. This is equivalent to the regex `/^\/login/`.
50
+ - RegExps match against the full path: use regex for patterns that don't anchor at the start, or for precise end-to-end matches.
51
+
52
+ ```ts
53
+ excludeDestinations: [
54
+ '/login', // prefix: skips /login, /login/callback, /login?next=…
55
+ '/external-tool/', // prefix: skips /external-tool/ and everything beneath it
56
+ /\/beta\//, // regex: skips any path containing /beta/ (not anchored)
57
+ /\.(pdf|zip)$/, // regex: skips paths ending in .pdf or .zip
58
+ /^\/exact-path$/, // regex: skips only the exact path /exact-path
59
+ ]
60
+ ```
61
+
62
+ ## Output
63
+
64
+ ```
65
+ [link-checker] 4231 HTML files
66
+ [link-checker] 8847 unique destinations
67
+ [link-checker] done in 2.1s
68
+ [link-checker] all links ok
69
+ ```
70
+
71
+ On failure:
72
+
73
+ ```
74
+ [link-checker] 2 broken links:
75
+ /docs/get-started/missing-page
76
+ /docs/overview
77
+ /docs/quickstarts/react
78
+ … and 12 more
79
+ /docs/api/old-endpoint
80
+ /docs/api/index
81
+ ```
82
+
83
+ ## Notes
84
+
85
+ - Only checks `href` attributes, skips `src` (images, scripts).
86
+ - Strips anchor fragments (`#section`) before checking; does not verify whether the anchor actually exists on the target page.
87
+ - Resolves relative hrefs against their source file's location and normalized to root-relative paths before deduplication and matching.
package/index.mjs ADDED
@@ -0,0 +1,477 @@
1
+ /**
2
+ * astro-better-link-checker
3
+ *
4
+ * Fast intra-site broken link, anchor, and image checker for Astro. Runs after build.
5
+ *
6
+ * Algorithm:
7
+ * 1. Walk the build output directory in parallel to collect all .html files.
8
+ * 2. Read all files concurrently; extract href, img src/srcset attributes, and id attributes.
9
+ * 3. Build linkMap<normalizedPath, Set<sourceUrl>>,
10
+ * imageMap<normalizedPath, Set<sourceUrl>>,
11
+ * fragmentMap<targetUrlPath, Map<fragment, Set<sourceUrl>>>,
12
+ * and idCache<urlPath, Set<id>>.
13
+ * A destination linked or referenced from 500 pages is stat'd exactly once.
14
+ * 4. Check all unique path destinations in parallel with fs.access.
15
+ * 5. Check all anchor fragments against idCache.
16
+ * 6. Report broken links, images, and anchors grouped by destination.
17
+ *
18
+ * External links are not checked. Use a dedicated HTTP checker for those.
19
+ *
20
+ * Usage (astro.config.ts):
21
+ *
22
+ * import linkChecker from 'astro-better-link-checker';
23
+ * export default defineConfig({
24
+ * integrations: [linkChecker()],
25
+ * });
26
+ *
27
+ * Options:
28
+ * excludeSourcePages {(string|RegExp)[]} skip pages whose URL path matches
29
+ * excludeDestinations {(string|RegExp)[]} skip destinations whose path matches
30
+ * failOnBrokenLinks {boolean} throw on any broken link (default: true)
31
+ * checkAnchors {boolean} validate #fragment targets (default: true)
32
+ * checkImages {boolean} validate img src/srcset paths (default: true)
33
+ * verbose {boolean} log every checked href/src (default: false)
34
+ */
35
+
36
+ import { readdir, readFile, access } from 'node:fs/promises';
37
+ import { join, resolve, relative, dirname } from 'node:path';
38
+ import { fileURLToPath } from 'node:url';
39
+ import process from 'node:process';
40
+
41
+ // Concurrency pool: runs at most `limit` tasks simultaneously.
42
+ function makePool(limit) {
43
+ let active = 0;
44
+ const queue = [];
45
+ const flush = () => {
46
+ while (active < limit && queue.length) {
47
+ active++;
48
+ const { fn, res, rej } = queue.shift();
49
+ fn().then(res, rej).finally(() => { active--; flush(); });
50
+ }
51
+ };
52
+ return fn => new Promise((res, rej) => { queue.push({ fn, res, rej }); flush(); });
53
+ }
54
+
55
+ // Test a value against an array of strings (prefix match) or RegExps.
56
+ function matches(value, patterns) {
57
+ return patterns.some(p => p instanceof RegExp ? p.test(value) : value.startsWith(String(p)));
58
+ }
59
+
60
+ // Walk a directory tree in parallel, collecting .html file paths.
61
+ async function walkHtml(dir) {
62
+ let entries;
63
+ try { entries = await readdir(dir, { withFileTypes: true }); }
64
+ catch { return []; }
65
+ const parts = await Promise.all(
66
+ entries.map(e => {
67
+ const p = join(dir, e.name);
68
+ if (e.isDirectory()) return walkHtml(p);
69
+ return e.name.endsWith('.html') ? [p] : [];
70
+ })
71
+ );
72
+ return parts.flat();
73
+ }
74
+
75
+ // Derive the URL path for a built HTML file.
76
+ // dist/docs/get-started/intro.html → /docs/get-started/intro
77
+ // dist/docs/index.html → /docs
78
+ // dist/index.html → /
79
+ function fileToUrlPath(htmlFile, buildDir) {
80
+ const rel = relative(buildDir, htmlFile).replace(/\\/g, '/');
81
+ if (rel === 'index.html') return '/';
82
+ if (rel.endsWith('/index.html')) return '/' + rel.slice(0, -'/index.html'.length);
83
+ if (rel.endsWith('.html')) return '/' + rel.slice(0, -5);
84
+ return '/' + rel;
85
+ }
86
+
87
+ // Extract all href attribute values from HTML.
88
+ // Strips comments, <script>, and <style> first to avoid false matches.
89
+ function extractHrefs(html) {
90
+ const stripped = html
91
+ .replace(/<!--[\s\S]*?-->/g, '')
92
+ .replace(/<script\b[^>]*>[\s\S]*?<\/script>/gi, '')
93
+ .replace(/<style\b[^>]*>[\s\S]*?<\/style>/gi, '');
94
+ const re = /\bhref=(["'])([^"']{1,2048}?)\1/gi;
95
+ const hrefs = [];
96
+ let m;
97
+ while ((m = re.exec(stripped)) !== null) hrefs.push(m[2]);
98
+ return hrefs;
99
+ }
100
+
101
+ // Extract image src and srcset URLs from HTML.
102
+ function extractSrcs(html) {
103
+ const stripped = html
104
+ .replace(/<!--[\s\S]*?-->/g, '')
105
+ .replace(/<script\b[^>]*>[\s\S]*?<\/script>/gi, '')
106
+ .replace(/<style\b[^>]*>[\s\S]*?<\/style>/gi, '');
107
+ const srcs = [];
108
+ let m;
109
+ // img, source, video, audio src=
110
+ const reSrc = /<(?:img|source|video|audio)\b[^>]*?\bsrc=(["'])([^"']{1,2048}?)\1/gi;
111
+ while ((m = reSrc.exec(stripped)) !== null) srcs.push(m[2]);
112
+ // srcset= on any element (comma-separated "url descriptor" pairs)
113
+ const reSrcset = /\bsrcset=(["'])([^"']{1,4096}?)\1/gi;
114
+ while ((m = reSrcset.exec(stripped)) !== null) {
115
+ for (const part of m[2].split(',')) {
116
+ const url = part.trim().split(/\s+/)[0];
117
+ if (url) srcs.push(url);
118
+ }
119
+ }
120
+ return srcs;
121
+ }
122
+
123
+ // Extract all id attribute values and legacy <a name="..."> anchors from HTML.
124
+ function extractIds(html) {
125
+ const stripped = html
126
+ .replace(/<!--[\s\S]*?-->/g, '')
127
+ .replace(/<script\b[^>]*>[\s\S]*?<\/script>/gi, '')
128
+ .replace(/<style\b[^>]*>[\s\S]*?<\/style>/gi, '');
129
+ const ids = new Set();
130
+ const reId = /\bid=(["'])([^"']{1,512}?)\1/gi;
131
+ let m;
132
+ while ((m = reId.exec(stripped)) !== null) ids.add(m[2]);
133
+ // legacy <a name="..."> anchors
134
+ const reName = /<a\b[^>]*\bname=(["'])([^"']{1,512}?)\1/gi;
135
+ while ((m = reName.exec(stripped)) !== null) ids.add(m[2]);
136
+ return ids;
137
+ }
138
+
139
+ // Normalize an href to {path, fragment} or null if it should be skipped.
140
+ // path is the root-relative page path, or null for fragment-only links (current page).
141
+ // fragment is the #anchor string without the leading '#', or null if no fragment.
142
+ function normalizeHref(href, htmlFile, buildDir) {
143
+ if (!href) return null;
144
+ // Skip external and special-scheme links entirely.
145
+ if (/^(?:https?:\/\/|\/\/|mailto:|tel:|javascript:|data:)/i.test(href)) return null;
146
+
147
+ // URL-decode and re-check — catches malformed links where an external URL was
148
+ // percent-encoded into a path, e.g. /page/%5Bhttps:/example.com
149
+ let decoded = href;
150
+ try { decoded = decodeURIComponent(href); } catch { return null; }
151
+ if (/(?:https?:|\/\/)/i.test(decoded.split('/').pop())) return null;
152
+ // Decoded brackets or pipes indicate a malformed link, not a real path.
153
+ if (/[[\]|\\]/.test(decoded)) return null;
154
+
155
+ const hashIdx = href.indexOf('#');
156
+ const fragment = hashIdx >= 0 ? href.slice(hashIdx + 1).split('?')[0] || null : null;
157
+ const bare = href.split('#')[0].split('?')[0];
158
+
159
+ // Fragment-only link — target path is resolved by caller to the current page.
160
+ if (!bare) return fragment ? { path: null, fragment } : null;
161
+
162
+ let path;
163
+ if (bare.startsWith('/')) {
164
+ path = bare;
165
+ } else {
166
+ // Relative href — resolve against the file's location, then make root-relative.
167
+ const abs = resolve(dirname(htmlFile), bare);
168
+ const rel = relative(buildDir, abs).replace(/\\/g, '/');
169
+ if (rel.startsWith('..')) return null; // outside build dir
170
+ path = '/' + rel;
171
+ }
172
+
173
+ return { path, fragment };
174
+ }
175
+
176
+ // Check whether a root-relative path resolves to any real file in the build output.
177
+ async function checkExists(norm, buildDir) {
178
+ if (!norm || norm === '/') return true;
179
+ const base = join(buildDir, norm.slice(1));
180
+ const ok = p => access(p).then(() => true, () => false);
181
+ const [plain, html, idx] = await Promise.all([
182
+ ok(base),
183
+ ok(base + '.html'),
184
+ ok(join(base, 'index.html')),
185
+ ]);
186
+ return plain || html || idx;
187
+ }
188
+
189
+ // Walk a directory for .md/.mdx files.
190
+ async function walkMarkdown(dir, extensions) {
191
+ let entries;
192
+ try { entries = await readdir(dir, { withFileTypes: true }); }
193
+ catch { return []; }
194
+ const parts = await Promise.all(
195
+ entries.map(e => {
196
+ const p = join(dir, e.name);
197
+ if (e.isDirectory()) return walkMarkdown(p, extensions);
198
+ return extensions.some(ext => e.name.endsWith(ext)) ? [p] : [];
199
+ })
200
+ );
201
+ return parts.flat();
202
+ }
203
+
204
+ // Replace code and non-prose regions with whitespace so they can't produce false positives.
205
+ function stripNonContentRegions(text) {
206
+ const blank = m => m.replace(/[^\n]/g, ' ');
207
+ // Frontmatter (--- block at start of file)
208
+ text = text.replace(/^\s*---\n[\s\S]*?\n---\n?/, blank);
209
+ // Fenced code blocks (backtick and tilde fences; \1 matches the opening fence length)
210
+ text = text.replace(/^(`{3,})[^\n]*\n[\s\S]*?\n\1[ \t]*$/gm, blank);
211
+ text = text.replace(/^(~{3,})[^\n]*\n[\s\S]*?\n\1[ \t]*$/gm, blank);
212
+ // Inline code spans (single or double backticks; no newlines inside)
213
+ text = text.replace(/`+[^`\n]+`+/g, blank);
214
+ // HTML/MDX comments (preserve newlines for correct line numbers)
215
+ text = text.replace(/<!--[\s\S]*?-->/g, m => '\n'.repeat((m.match(/\n/g) ?? []).length));
216
+ // MDX import/export lines
217
+ text = text.replace(/^[ \t]*(import|export)\b[^\n]*/gm, blank);
218
+ // JSX/HTML string attribute values to avoid false positives in prop strings
219
+ text = text.replace(/=["'][^"'\n]*["']/g, blank);
220
+ return text;
221
+ }
222
+
223
+ // Detect malformed markdown link syntax. Returns { file, line, col, text, kind }[].
224
+ //
225
+ // Pattern 1 — [TEXT(URL): missing ] before (.
226
+ // [^\[\]()\n]+ excludes brackets and parens from the "link text" portion, so a
227
+ // normal [text](url) never matches: the ] terminates the char class before ( is reached.
228
+ // The (?!\s*]) negative lookahead rejects the valid [text (/foo)](real-url) form where
229
+ // a URL-like string appears inside the link text itself.
230
+ //
231
+ // Pattern 2 — [TEXT] (URL): whitespace between ] and (.
232
+ // The first char of the bracket content excludes ^ (footnote refs) so [^id] is safe.
233
+ // Requires the paren content to start with a URL-like pattern; this prevents task list
234
+ // items [x] (description) and ordinary parentheticals from triggering.
235
+ function findMalformedLinks(text, filePath) {
236
+ const issues = [];
237
+ const content = stripNonContentRegions(text);
238
+ const lines = content.split('\n');
239
+
240
+ for (let i = 0; i < lines.length; i++) {
241
+ const line = lines[i];
242
+ const lineNo = i + 1;
243
+ let m;
244
+
245
+ const reMissingClose = /\[[^\[\]()\n]+\((?:https?:\/\/|[./])[^)\n]*\)(?!\s*])/g;
246
+ while ((m = reMissingClose.exec(line)) !== null)
247
+ issues.push({ file: filePath, line: lineNo, col: m.index + 1, text: m[0], kind: 'missing-close-bracket' });
248
+
249
+ const reSpaceParen = /\[[^\[\]^][^\[\]]*\][ \t]+\((?:https?:\/\/|[./])[^)\n]*\)/g;
250
+ while ((m = reSpaceParen.exec(line)) !== null)
251
+ issues.push({ file: filePath, line: lineNo, col: m.index + 1, text: m[0], kind: 'space-before-href' });
252
+ }
253
+
254
+ return issues;
255
+ }
256
+
257
+ export function markdownLinkSyntaxChecker(opts = {}) {
258
+ const {
259
+ excludePaths = [],
260
+ failOnIssues = true,
261
+ extensions = ['.md', '.mdx'],
262
+ _exit = process.exit,
263
+ } = opts;
264
+
265
+ let srcDir = '';
266
+
267
+ return {
268
+ name: 'astro-markdown-link-syntax-checker',
269
+ hooks: {
270
+ 'astro:config:done': ({ config }) => {
271
+ srcDir = config.srcDir instanceof URL
272
+ ? fileURLToPath(config.srcDir)
273
+ : String(config.srcDir);
274
+ },
275
+ 'astro:build:done': async ({ logger }) => {
276
+ if (!srcDir) return;
277
+ const log = msg => (logger ? logger.info(msg) : console.log(msg));
278
+ const pool = makePool(50);
279
+
280
+ const files = await walkMarkdown(srcDir, extensions);
281
+ log(`[md-link-syntax] checking ${files.length} markdown files`);
282
+
283
+ const allIssues = [];
284
+ await Promise.all(
285
+ files.map(file =>
286
+ pool(async () => {
287
+ const relPath = relative(srcDir, file);
288
+ if (matches(relPath, excludePaths)) return;
289
+ const text = await readFile(file, 'utf-8');
290
+ allIssues.push(...findMalformedLinks(text, relPath));
291
+ })
292
+ )
293
+ );
294
+
295
+ if (allIssues.length === 0) {
296
+ log('[md-link-syntax] no issues found');
297
+ return;
298
+ }
299
+
300
+ allIssues.sort((a, b) => a.file.localeCompare(b.file) || a.line - b.line);
301
+ const report = allIssues
302
+ .map(({ file, line, col, text, kind }) => ` ${file}:${line}:${col} [${kind}] ${text}`)
303
+ .join('\n');
304
+ log(`[md-link-syntax] ${allIssues.length} issue${allIssues.length === 1 ? '' : 's'}:\n${report}`);
305
+ if (failOnIssues) _exit(1);
306
+ },
307
+ },
308
+ };
309
+ }
310
+
311
+ export default function linkChecker(opts = {}) {
312
+ const {
313
+ excludeSourcePages = [],
314
+ excludeDestinations = [],
315
+ failOnBrokenLinks = true,
316
+ checkAnchors = true,
317
+ checkImages = true,
318
+ verbose = false,
319
+ _exit = process.exit,
320
+ } = opts;
321
+
322
+ return {
323
+ name: 'astro-better-link-checker',
324
+ hooks: {
325
+ 'astro:build:done': async ({ dir, logger }) => {
326
+ const t0 = Date.now();
327
+ const buildDir = dir instanceof URL ? fileURLToPath(dir) : String(dir);
328
+ const log = msg => (logger ? logger.info(msg) : console.log(msg));
329
+
330
+ // Phase 1: discover all HTML files.
331
+ const htmlFiles = await walkHtml(buildDir);
332
+ log(`[link-checker] ${htmlFiles.length} HTML files`);
333
+
334
+ // Phase 2: extract hrefs, img srcs, fragments, and ids across all pages.
335
+ // linkMap: normPath → Set<sourceUrlPath>
336
+ // imageMap: normPath → Set<sourceUrlPath>
337
+ // fragmentMap: targetPath → Map<fragment, Set<sourceUrlPath>>
338
+ // idCache: urlPath → Set<id>
339
+ const linkMap = new Map();
340
+ const imageMap = new Map();
341
+ const fragmentMap = new Map();
342
+ const idCache = new Map();
343
+ const pool = makePool(200);
344
+
345
+ await Promise.all(
346
+ htmlFiles.map(file =>
347
+ pool(async () => {
348
+ const urlPath = fileToUrlPath(file, buildDir);
349
+ if (matches(urlPath, excludeSourcePages)) return;
350
+
351
+ const html = await readFile(file, 'utf-8');
352
+ if (checkAnchors) idCache.set(urlPath, extractIds(html));
353
+
354
+ for (const raw of extractHrefs(html)) {
355
+ const result = normalizeHref(raw, file, buildDir);
356
+ if (!result) continue;
357
+
358
+ const { path: normPath, fragment } = result;
359
+ // Fragment-only links resolve to the current page.
360
+ const resolvedPath = normPath ?? urlPath;
361
+
362
+ if (matches(resolvedPath, excludeDestinations)) continue;
363
+
364
+ if (normPath !== null) {
365
+ let sources = linkMap.get(normPath);
366
+ if (!sources) { sources = new Set(); linkMap.set(normPath, sources); }
367
+ sources.add(urlPath);
368
+ }
369
+
370
+ if (fragment !== null && checkAnchors) {
371
+ let fragsByPath = fragmentMap.get(resolvedPath);
372
+ if (!fragsByPath) { fragsByPath = new Map(); fragmentMap.set(resolvedPath, fragsByPath); }
373
+ let fragSources = fragsByPath.get(fragment);
374
+ if (!fragSources) { fragSources = new Set(); fragsByPath.set(fragment, fragSources); }
375
+ fragSources.add(urlPath);
376
+ }
377
+ }
378
+
379
+ if (checkImages) {
380
+ for (const raw of extractSrcs(html)) {
381
+ const result = normalizeHref(raw, file, buildDir);
382
+ if (!result) continue;
383
+ const { path: normPath } = result;
384
+ if (!normPath) continue;
385
+ if (matches(normPath, excludeDestinations)) continue;
386
+ let sources = imageMap.get(normPath);
387
+ if (!sources) { sources = new Set(); imageMap.set(normPath, sources); }
388
+ sources.add(urlPath);
389
+ }
390
+ }
391
+ })
392
+ )
393
+ );
394
+
395
+ log(`[link-checker] ${linkMap.size} unique link destinations, ${imageMap.size} unique image paths`);
396
+
397
+ // Phase 3: check every unique path destination exactly once.
398
+ const broken = [];
399
+ await Promise.all([
400
+ ...[...linkMap.entries()].map(async ([norm, sources]) => {
401
+ const ok = await checkExists(norm, buildDir);
402
+ if (verbose) log(`[link-checker] ${ok ? '✓' : '✗'} ${norm}`);
403
+ if (!ok) broken.push({ href: norm, sources: [...sources].sort() });
404
+ }),
405
+ ...[...imageMap.entries()].map(async ([norm, sources]) => {
406
+ const ok = await checkExists(norm, buildDir);
407
+ if (verbose) log(`[link-checker] ${ok ? '✓' : '✗'} img ${norm}`);
408
+ if (!ok) broken.push({ href: norm, sources: [...sources].sort(), image: true });
409
+ }),
410
+ ]);
411
+
412
+ // Phase 4: check anchor fragments against the id sets of their target pages.
413
+ if (checkAnchors && fragmentMap.size > 0) {
414
+ await Promise.all(
415
+ [...fragmentMap.entries()].map(async ([targetPath, fragMap]) => {
416
+ // Skip anchor checking for pages that don't exist (already reported as broken).
417
+ if (!(await checkExists(targetPath, buildDir))) return;
418
+
419
+ // idCache keys have no trailing slash; normalize the lookup.
420
+ const idKey = targetPath !== '/' && targetPath.endsWith('/')
421
+ ? targetPath.slice(0, -1)
422
+ : targetPath;
423
+ let ids = idCache.get(idKey);
424
+
425
+ if (!ids) {
426
+ // Page exists but wasn't crawled (e.g. excluded source page) — read it now.
427
+ const base = join(buildDir, targetPath.replace(/^\//, ''));
428
+ for (const p of [base + '.html', join(base, 'index.html'), base]) {
429
+ try { ids = extractIds(await readFile(p, 'utf-8')); break; }
430
+ catch { /* try next */ }
431
+ }
432
+ if (!ids) return;
433
+ }
434
+
435
+ for (const [fragment, sources] of fragMap.entries()) {
436
+ if (ids.has(fragment)) {
437
+ if (verbose) log(`[link-checker] ✓ ${targetPath}#${fragment}`);
438
+ } else {
439
+ if (verbose) log(`[link-checker] ✗ ${targetPath}#${fragment}`);
440
+ broken.push({ href: `${targetPath}#${fragment}`, sources: [...sources].sort() });
441
+ }
442
+ }
443
+ })
444
+ );
445
+ }
446
+
447
+ const elapsed = ((Date.now() - t0) / 1000).toFixed(1);
448
+ log(`[link-checker] done in ${elapsed}s`);
449
+
450
+ if (broken.length === 0) {
451
+ log(`[link-checker] all links ok`);
452
+ return;
453
+ }
454
+
455
+ broken.sort((a, b) => a.href.localeCompare(b.href));
456
+ const report = broken
457
+ .map(({ href, sources, image }) => {
458
+ const label = image ? `img ${href}` : href;
459
+ const shown = sources.slice(0, 5).join('\n ');
460
+ const more = sources.length > 5 ? `\n … and ${sources.length - 5} more` : '';
461
+ return ` ${label}\n ${shown}${more}`;
462
+ })
463
+ .join('\n');
464
+
465
+ const brokenLinks = broken.filter(b => !b.image).length;
466
+ const brokenImages = broken.filter(b => b.image).length;
467
+ const summary = [
468
+ brokenLinks && `${brokenLinks} broken link${brokenLinks === 1 ? '' : 's'}`,
469
+ brokenImages && `${brokenImages} broken image${brokenImages === 1 ? '' : 's'}`,
470
+ ].filter(Boolean).join(', ');
471
+ const msg = `[link-checker] ${summary}:\n${report}`;
472
+ log(msg);
473
+ if (failOnBrokenLinks) _exit(1);
474
+ },
475
+ },
476
+ };
477
+ }
package/package.json ADDED
@@ -0,0 +1,35 @@
1
+ {
2
+ "name": "astro-better-link-checker",
3
+ "version": "1.0.0",
4
+ "description": "Fast intra-site broken link and image checker for Astro — deduplicates destinations and checks each one exactly once",
5
+ "keywords": [
6
+ "astro",
7
+ "astro-integration",
8
+ "link-checker",
9
+ "broken-links"
10
+ ],
11
+ "license": "MIT",
12
+ "author": "Nathan Contino <ncontino@lambdalatitudinarians.org>",
13
+ "repository": {
14
+ "type": "git",
15
+ "url": "git+https://github.com/better-static-sites/better-static-sites.github.io.git",
16
+ "directory": "astro-better-link-checker"
17
+ },
18
+ "homepage": "https://better-static-sites.github.io",
19
+ "bugs": {
20
+ "url": "https://github.com/better-static-sites/better-static-sites.github.io/issues"
21
+ },
22
+ "type": "module",
23
+ "exports": {
24
+ ".": "./index.mjs"
25
+ },
26
+ "files": [
27
+ "index.mjs"
28
+ ],
29
+ "scripts": {
30
+ "test": "node --test test/test.mjs"
31
+ },
32
+ "peerDependencies": {
33
+ "astro": ">=4.0.0"
34
+ }
35
+ }