@writedocs/generator 0.4.12 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bin/writedocs.js CHANGED
@@ -8,7 +8,7 @@ import { runDev } from '../src/cli/dev.js';
8
8
  import { runBuild } from '../src/cli/build.js';
9
9
  import { runInit } from '../src/cli/init.js';
10
10
  import { requireBuildKey } from '../src/cli/build-auth.js';
11
- import { log, step, plural, color, CliExit, errorText } from '../src/cli/output.js';
11
+ import { log, step, plural, color, CliExit, errorText, stopActiveStep } from '../src/cli/output.js';
12
12
  // O MESMO modulo que o build usa (via loadDocsConfig, que reexporta daqui) e
13
13
  // que a plataforma importa por `@writedocs/generator/config-schema` - e o que
14
14
  // faz os tres reportarem os mesmos problemas com as mesmas palavras, em vez de
@@ -170,6 +170,41 @@ program
170
170
  }
171
171
  });
172
172
 
173
+ program
174
+ // Like `validate`: no key, no build, no network - reads the project and
175
+ // exits 0 (no broken links) or 1, so it works as a CI step.
176
+ .command('broken-links')
177
+ .description('Check every internal link - pages, anchors and files - in the pages and writedocs.json')
178
+ .argument('[dir]', 'content directory (contains writedocs.json)', '.')
179
+ .action(async (dir) => {
180
+ const contentDir = path.resolve(process.cwd(), dir);
181
+ const { preflightCheck } = await import('../src/cli/preflight.js');
182
+ preflightCheck(contentDir);
183
+ const configText = fs.readFileSync(path.join(contentDir, 'writedocs.json'), 'utf-8');
184
+ const checking = step('Checking links');
185
+ // Generated OpenAPI pages are link targets too - same step dev/build run.
186
+ const { generateApiPages } = await import('../src/cli/generate-api-pages.js');
187
+ await generateApiPages({ contentDir });
188
+ const { checkLinks } = await import('../src/lib/link-check.js');
189
+ const { formatContentIssues } = await import('../src/lib/content-check.js');
190
+ const result = await checkLinks(contentDir, configText);
191
+ checking.stop();
192
+
193
+ const summary = `${plural(result.links, 'link')} in ${plural(result.pages, 'page')}${
194
+ result.external ? color.dim(` (${plural(result.external, 'external link')} not checked)`) : ''
195
+ }`;
196
+ if (result.broken.length === 0) {
197
+ log.success(`No broken links - checked ${summary}`);
198
+ return;
199
+ }
200
+ log.error(plural(result.broken.length, 'broken link'));
201
+ log.line();
202
+ log.line(formatContentIssues(result.broken));
203
+ log.line();
204
+ log.error(`${plural(result.broken.length, 'broken link')} - checked ${summary}`);
205
+ throw new CliExit(1);
206
+ });
207
+
173
208
  program
174
209
  .command('convert')
175
210
  .description("Convert another docs tool's config into writedocs.json")
@@ -203,6 +238,7 @@ program
203
238
 
204
239
  program.parseAsync(process.argv).catch((err) => {
205
240
  // A command that already printed its own error throws CliExit.
241
+ stopActiveStep();
206
242
  if (err instanceof CliExit) process.exit(err.code);
207
243
  log.error(errorText(err));
208
244
  process.exit(1);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@writedocs/generator",
3
- "version": "0.4.12",
3
+ "version": "0.5.0",
4
4
  "description": "Static site generator for docs — a writedocs.json + MDX folder in, a static site out.",
5
5
  "type": "module",
6
6
  "bin": {
@@ -3,6 +3,7 @@
3
3
  // author's file and line, and the reason in one sentence - instead of a
4
4
  // bundler error with a stack trace; a log line is either something the
5
5
  // author should see or noise to hide.
6
+ import fs from 'node:fs';
6
7
  import path from 'node:path';
7
8
 
8
9
  const ANSI = /\x1b\[[0-9;]*m/g;
@@ -12,10 +13,14 @@ export const stripAnsi = (text) => String(text ?? '').replace(ANSI, '');
12
13
  // append these sections after the message itself.
13
14
  const SECTION = /\n\s*(Stack trace|Hint|Error reference|Location|Caused by):/;
14
15
 
15
- function relativeTo(contentDir, file) {
16
- if (!file) return null;
16
+ function absolutePath(file, root) {
17
17
  const clean = file.replace(/^file:\/\/\/?/, '').replace(/[?#].*$/, '');
18
- const abs = path.resolve(clean);
18
+ return path.resolve(root ?? process.cwd(), clean);
19
+ }
20
+
21
+ function relativeTo(contentDir, file, root) {
22
+ if (!file) return null;
23
+ const abs = absolutePath(file, root);
19
24
  const rel = path.relative(contentDir, abs);
20
25
  if (rel.startsWith('..') || path.isAbsolute(rel)) return null;
21
26
  return rel.split(path.sep).join('/');
@@ -28,8 +33,9 @@ function position(line, column) {
28
33
 
29
34
  /** Everything an Astro/Vite error can say, as { where, message, hint }.
30
35
  * `input` is a serialized error (astro-worker.js serializeError) or the
31
- * text of an error log line. */
32
- export function describeError(input, contentDir) {
36
+ * text of an error log line. `root`: the folder relative paths in it are
37
+ * relative to - Astro's working folder, the package root. */
38
+ export function describeError(input, contentDir, { root } = {}) {
33
39
  const error = typeof input === 'string' ? { message: input } : input ?? {};
34
40
  let text = stripAnsi(error.message).replace(/\r/g, '');
35
41
  let file = error.loc?.file ?? null;
@@ -58,6 +64,27 @@ export function describeError(input, contentDir) {
58
64
  }
59
65
  }
60
66
 
67
+ // A rolldown diagnostic, with a code frame of the *compiled* module
68
+ // (whose line numbers aren't the page's):
69
+ // [UNRESOLVED_IMPORT] Could not resolve './pic.png' in ../docs/page.mdx
70
+ // ╭─[ ../docs/page.mdx:9:29 ]
71
+ const diagnostic = text.match(/^\[([A-Z][A-Z_]+)\]\s+([^\n]*)/);
72
+ if (diagnostic) {
73
+ const unresolved = diagnostic[2].match(/^Could not resolve '([^']+)' in (.+)$/);
74
+ if (unresolved) {
75
+ const [, spec, from] = unresolved;
76
+ file = file ?? from.trim();
77
+ text = `Can't find ${spec}, which this page uses.`;
78
+ // The line it's on in the page itself.
79
+ try {
80
+ const index = fs.readFileSync(absolutePath(file, root), 'utf8').split('\n').findIndex((l) => l.includes(spec));
81
+ if (index !== -1) line = index + 1;
82
+ } catch {}
83
+ } else {
84
+ text = diagnostic[2];
85
+ }
86
+ }
87
+
61
88
  // Astro's own sections: keep the hint, and take the file from the first
62
89
  // stack frame when it's one of the author's files.
63
90
  const sectionAt = text.search(SECTION);
@@ -93,7 +120,7 @@ export function describeError(input, contentDir) {
93
120
 
94
121
  if (name === 'YAMLException') text = `Frontmatter isn't valid YAML: ${text}`;
95
122
 
96
- const rel = relativeTo(contentDir, file);
123
+ const rel = relativeTo(contentDir, file, root);
97
124
  return {
98
125
  where: rel ? `${rel}${position(line, column)}` : null,
99
126
  message: text || 'Unknown error.',
package/src/cli/build.js CHANGED
@@ -72,7 +72,7 @@ export async function runBuild({ contentDir, packageRoot, verbose = false }) {
72
72
  if (fatal || code !== 0) {
73
73
  building.fail('Build failed');
74
74
  if (fatal) {
75
- const problem = describeError(fatal, contentDir);
75
+ const problem = describeError(fatal, contentDir, { root: packageRoot });
76
76
  log.line();
77
77
  log.line(formatProblems([{ where: problem.where, message: problem.message }]));
78
78
  if (problem.hint) log.line(`\n${color.dim(` Hint: ${problem.hint}`)}`);
package/src/cli/dev.js CHANGED
@@ -112,7 +112,7 @@ export async function runDev({ contentDir, packageRoot, port, verbose = false })
112
112
  return;
113
113
  case 'astro-log': {
114
114
  if (event.level === 'error') {
115
- showError(describeError(stripAnsi(event.message), contentDir));
115
+ showError(describeError(stripAnsi(event.message), contentDir, { root: packageRoot }));
116
116
  return;
117
117
  }
118
118
  const request = requestLog(event);
@@ -151,7 +151,7 @@ export async function runDev({ contentDir, packageRoot, port, verbose = false })
151
151
  if (!ready) starting.fail('Could not start the local preview');
152
152
  else log.error('The local preview stopped unexpectedly.');
153
153
  if (fatal) {
154
- const problem = describeError(fatal, contentDir);
154
+ const problem = describeError(fatal, contentDir, { root: packageRoot });
155
155
  log.line();
156
156
  log.line(formatProblems([{ where: problem.where, message: problem.message }]));
157
157
  if (problem.hint) log.line(`\n${color.dim(` Hint: ${problem.hint}`)}`);
package/src/cli/output.js CHANGED
@@ -67,6 +67,11 @@ export function indent(text, spaces) {
67
67
  .join('\n');
68
68
  }
69
69
 
70
+ /** Ends whatever step is still spinning - for an error thrown mid-step. */
71
+ export function stopActiveStep() {
72
+ activeSpinner?.stop();
73
+ }
74
+
70
75
  const FRAMES = ['⠋', '⠙', '⠹', '⠸', '⠼', '⠴', '⠦', '⠧', '⠇', '⠏'];
71
76
 
72
77
  /** A step in progress: a spinner with `text` on a terminal, nothing
@@ -76,6 +81,7 @@ export function step(text) {
76
81
  let frame = 0;
77
82
  let timer = null;
78
83
  const spinner = {
84
+ stop: () => end(),
79
85
  clear() {
80
86
  stream.write('\r\x1b[2K');
81
87
  },
@@ -9,6 +9,7 @@
9
9
  // - an .mdx file doesn't parse as MDX
10
10
  // - writedocs.json's navigation lists a page that doesn't exist
11
11
  // - a redirect is a pattern (`/old/:slug`) rather than one exact path
12
+ // - a Markdown image with a relative path names a file that doesn't exist
12
13
  // Warnings - the build succeeds, but not as written:
13
14
  // - an unknown component (the build shows only its content - see
14
15
  // lib/mdx-unknown-components.js)
@@ -118,6 +119,28 @@ async function checkPage(contentDir, rel, errors, warnings) {
118
119
  );
119
120
  return;
120
121
  }
122
+ // A Markdown image with a relative path is imported by the build (Astro's
123
+ // image pipeline), so a missing file fails the whole build.
124
+ visit(tree, 'image', (node) => {
125
+ const url = String(node.url ?? '');
126
+ if (!url || url.startsWith('/') || url.startsWith('#') || /^[a-z][a-z0-9+.-]*:/i.test(url) || url.startsWith('//')) return;
127
+ let target;
128
+ try {
129
+ target = path.resolve(path.dirname(path.join(contentDir, rel)), decodeURI(url.split(/[?#]/)[0]));
130
+ } catch {
131
+ return;
132
+ }
133
+ if (fs.existsSync(target)) return;
134
+ const line = node.position?.start?.line;
135
+ errors.push(
136
+ issue(
137
+ rel,
138
+ line ? line + lineOffset : undefined,
139
+ `Image ${url} doesn't exist - the build fails on it.`,
140
+ 'The path is relative to this page\'s folder. Fix it, or use a path from the project root, like "/images/example.png".'
141
+ )
142
+ );
143
+ });
121
144
  for (const { name, line } of findUnknownComponents(tree)) {
122
145
  warnings.push(
123
146
  issue(
@@ -0,0 +1,473 @@
1
+ // `writedocs broken-links`: every internal link in the content - page
2
+ // bodies, the snippets they import, and writedocs.json - checked against the
3
+ // site the build would produce, without building it. Plain JavaScript,
4
+ // loaded by plain Node from an installed package (see lib/icons.js's
5
+ // comment on why not TypeScript).
6
+ //
7
+ // A link is broken when, on the built site:
8
+ // - no page, redirect or file is at its address;
9
+ // - its #anchor isn't on the target page (headings, titled components -
10
+ // Callout/Accordion/Update and friends - and elements with an `id`);
11
+ // - it names a file the build doesn't publish (anything outside public/
12
+ // other than images and fonts - see lib/styles-asset-integration.js).
13
+ //
14
+ // Addresses are worked out the way the build works them out:
15
+ // - a page's URL is its frontmatter `slug`, or else its file path with
16
+ // every segment put through github-slugger - Astro's own glob-loader
17
+ // id (getContentEntryIdAndSlug in astro/dist/content/utils.js), which
18
+ // is why `docs/API Guide/v1.2.mdx` is served at /docs/api-guide/v12/;
19
+ // - heading ids come from github-slugger too, deduped per page the way
20
+ // Astro's rehypeHeadingIds does; titled components' ids from
21
+ // lib/mdx-title-anchor-ids.js's own rule;
22
+ // - a relative link resolves the way a browser resolves it, from the
23
+ // page's URL - which ends in a slash (/docs/guides/setup/), so
24
+ // `install` on that page means /docs/guides/setup/install/, not a
25
+ // sibling page. Mintlify's URLs have no trailing slash, so migrated
26
+ // relative links are the usual casualty; the message says so.
27
+ //
28
+ // External links (http:, mailto:, ...) aren't checked.
29
+ import fs from 'node:fs';
30
+ import path from 'node:path';
31
+ import { createRequire } from 'node:module';
32
+ import { pathToFileURL } from 'node:url';
33
+ import matter from 'gray-matter';
34
+ import { visit } from 'unist-util-visit';
35
+ import { findAllPages } from './pages.js';
36
+ import { createJsonLocator } from './config-schema.js';
37
+ import { writedocsTempDir } from './writedocs-temp-dir.js';
38
+
39
+ // The same github-slugger Astro builds page URLs and heading ids with -
40
+ // resolved through Astro itself, so the two can't drift apart.
41
+ let sluggerModule = null;
42
+ async function loadSlugger() {
43
+ if (!sluggerModule) {
44
+ const require = createRequire(import.meta.url);
45
+ const fromAstro = createRequire(require.resolve('astro/package.json'));
46
+ sluggerModule = await import(pathToFileURL(fromAstro.resolve('github-slugger')).href);
47
+ }
48
+ return sluggerModule;
49
+ }
50
+
51
+ const processors = {};
52
+ async function parse(text, format) {
53
+ if (!processors[format]) {
54
+ const { createProcessor } = await import('@mdx-js/mdx');
55
+ processors[format] = createProcessor({ format });
56
+ }
57
+ return processors[format].parse(text);
58
+ }
59
+
60
+ // Images and fonts are published from anywhere in the project; every other
61
+ // file only from public/ (lib/styles-asset-integration.js).
62
+ const ASSET_EXTENSIONS = new Set(['.svg', '.png', '.jpg', '.jpeg', '.gif', '.webp', '.avif', '.ico', '.woff', '.woff2', '.ttf', '.otf']);
63
+ // lib/mdx-title-anchor-ids.js
64
+ const TITLED_COMPONENTS = ['Callout', 'Note', 'Info', 'Tip', 'Warning', 'Danger', 'Check', 'Accordion', 'Update'];
65
+ const titleSlug = (value) =>
66
+ value
67
+ .toLowerCase()
68
+ .replace(/[^a-z0-9]+/g, '-')
69
+ .replace(/^-+|-+$/g, '');
70
+
71
+ const BASE = 'http://writedocs.invalid';
72
+
73
+ function normalizeEntryId(id) {
74
+ const trimmed = String(id).replace(/^\/+/, '').replace(/\/+$/, '');
75
+ return trimmed === '' ? 'index' : trimmed;
76
+ }
77
+ const urlForId = (id) => (id === 'index' ? '/' : `/${id}/`);
78
+
79
+ function textOf(node) {
80
+ if (typeof node.value === 'string' && (node.type === 'text' || node.type === 'inlineCode')) return node.value;
81
+ return (node.children ?? []).map(textOf).join('');
82
+ }
83
+
84
+ function attribute(node, name) {
85
+ const attr = node.attributes?.find((a) => a.type === 'mdxJsxAttribute' && a.name === name);
86
+ return attr && typeof attr.value === 'string' ? attr.value : null;
87
+ }
88
+
89
+ /** Everything a page or snippet file links to and can be linked to. */
90
+ async function scanFile(absPath, { slug, Slugger }) {
91
+ let raw;
92
+ try {
93
+ raw = fs.readFileSync(absPath, 'utf-8').replace(/\r\n/g, '\n');
94
+ } catch {
95
+ return null;
96
+ }
97
+ let parsed;
98
+ try {
99
+ parsed = matter(raw, {});
100
+ } catch {
101
+ return null; // invalid frontmatter - `writedocs validate` reports it
102
+ }
103
+ const bodyStart = raw.lastIndexOf(parsed.content);
104
+ const lineOffset = bodyStart > 0 ? raw.slice(0, bodyStart).split('\n').length - 1 : 0;
105
+ let tree;
106
+ try {
107
+ tree = await parse(parsed.content, /\.mdx$/i.test(absPath) ? 'mdx' : 'md');
108
+ } catch {
109
+ return null; // doesn't parse - `writedocs validate` reports it
110
+ }
111
+
112
+ const links = []; // { href, line, kind: 'link' | 'image' | 'src' }
113
+ const anchors = new Set();
114
+ const imports = [];
115
+ const lineOf = (node) => (node.position?.start?.line ? node.position.start.line + lineOffset : undefined);
116
+ const headingSlugger = new Slugger();
117
+ const titleSeen = new Map();
118
+
119
+ visit(tree, (node) => {
120
+ switch (node.type) {
121
+ case 'link':
122
+ case 'definition':
123
+ links.push({ href: node.url, line: lineOf(node), kind: 'link' });
124
+ break;
125
+ case 'image':
126
+ links.push({ href: node.url, line: lineOf(node), kind: 'image' });
127
+ break;
128
+ case 'heading': {
129
+ const text = textOf(node);
130
+ if (text.trim()) anchors.add(headingSlugger.slug(text));
131
+ break;
132
+ }
133
+ case 'html': {
134
+ // Raw HTML in a .md page.
135
+ for (const m of node.value.matchAll(/\bhref\s*=\s*["']([^"']*)["']/gi)) links.push({ href: m[1], line: lineOf(node), kind: 'link' });
136
+ for (const m of node.value.matchAll(/\bsrc\s*=\s*["']([^"']*)["']/gi)) links.push({ href: m[1], line: lineOf(node), kind: 'src' });
137
+ for (const m of node.value.matchAll(/\b(?:id|name)\s*=\s*["']([^"']+)["']/gi)) anchors.add(m[1]);
138
+ break;
139
+ }
140
+ case 'mdxjsEsm':
141
+ for (const m of node.value.matchAll(/\bfrom\s*['"]([^'"]+\.mdx?)['"]/g)) imports.push(m[1]);
142
+ break;
143
+ case 'mdxJsxFlowElement':
144
+ case 'mdxJsxTextElement': {
145
+ const href = attribute(node, 'href');
146
+ if (href !== null) links.push({ href, line: lineOf(node), kind: 'link' });
147
+ if (!/^(iframe|script|embed)$/i.test(node.name ?? '')) {
148
+ for (const name of ['src', 'img']) {
149
+ const value = attribute(node, name);
150
+ if (value !== null) links.push({ href: value, line: lineOf(node), kind: 'src' });
151
+ }
152
+ }
153
+ const id = attribute(node, 'id');
154
+ if (id) anchors.add(id);
155
+ if (TITLED_COMPONENTS.includes(node.name)) {
156
+ const title = attribute(node, node.name === 'Update' ? 'label' : 'title');
157
+ const base = title ? titleSlug(title) : '';
158
+ if (base) {
159
+ const prior = titleSeen.get(base) ?? 0;
160
+ titleSeen.set(base, prior + 1);
161
+ anchors.add(prior === 0 ? base : `${base}-${prior + 1}`);
162
+ }
163
+ }
164
+ break;
165
+ }
166
+ }
167
+ });
168
+ return { data: parsed.data ?? {}, links, anchors, imports };
169
+ }
170
+
171
+ function editDistance(a, b) {
172
+ const row = Array.from({ length: b.length + 1 }, (_, i) => i);
173
+ for (let i = 1; i <= a.length; i++) {
174
+ let prev = row[0];
175
+ row[0] = i;
176
+ for (let j = 1; j <= b.length; j++) {
177
+ const tmp = row[j];
178
+ row[j] = Math.min(row[j] + 1, row[j - 1] + 1, prev + (a[i - 1] === b[j - 1] ? 0 : 1));
179
+ prev = tmp;
180
+ }
181
+ }
182
+ return row[b.length];
183
+ }
184
+
185
+ function closest(value, candidates) {
186
+ let best = null;
187
+ let bestDistance = Infinity;
188
+ for (const c of candidates) {
189
+ const d = editDistance(value.toLowerCase(), c.toLowerCase());
190
+ if (d < bestDistance) {
191
+ best = c;
192
+ bestDistance = d;
193
+ }
194
+ }
195
+ return best !== null && bestDistance <= Math.max(2, Math.floor(value.length * 0.3)) ? best : null;
196
+ }
197
+
198
+ /** Generated OpenAPI pages' URLs, from the manifests generateApiPages()
199
+ * (src/cli/generate-api-pages.js) wrote - it runs first. */
200
+ function generatedPageUrls(contentDir) {
201
+ const urls = new Set();
202
+ const root = path.join(writedocsTempDir(contentDir), 'openapi');
203
+ (function walk(dir) {
204
+ let entries;
205
+ try {
206
+ entries = fs.readdirSync(dir, { withFileTypes: true });
207
+ } catch {
208
+ return;
209
+ }
210
+ for (const entry of entries) {
211
+ const abs = path.join(dir, entry.name);
212
+ if (entry.isDirectory()) walk(abs);
213
+ else if (entry.name === 'manifest.json') {
214
+ try {
215
+ for (const op of JSON.parse(fs.readFileSync(abs, 'utf-8'))) if (op.generated) urls.add(urlForId(normalizeEntryId(op.slug)));
216
+ } catch {}
217
+ }
218
+ }
219
+ })(root);
220
+ return urls;
221
+ }
222
+
223
+ /** Every `href` in writedocs.json, and redirect destinations. */
224
+ function configLinks(config) {
225
+ const found = [];
226
+ (function walk(node, trail) {
227
+ if (Array.isArray(node)) node.forEach((item, i) => walk(item, [...trail, i]));
228
+ else if (node && typeof node === 'object') {
229
+ for (const [key, value] of Object.entries(node)) {
230
+ if (key === 'href' && typeof value === 'string') found.push({ href: value, path: [...trail, key] });
231
+ else walk(value, [...trail, key]);
232
+ }
233
+ }
234
+ })(config, []);
235
+ (Array.isArray(config?.redirects) ? config.redirects : []).forEach((r, i) => {
236
+ if (typeof r?.destination === 'string') found.push({ href: r.destination, path: ['redirects', i, 'destination'] });
237
+ });
238
+ return found;
239
+ }
240
+
241
+ /**
242
+ * Checks every internal link in `contentDir`. `configText` is writedocs.json's
243
+ * raw text (valid JSON - the CLI checks that first). Returns
244
+ * { pages, links, external, broken } - `broken` is a list of issues shaped
245
+ * like lib/content-check.js's: { file, line?, message, suggestion? }.
246
+ */
247
+ export async function checkLinks(contentDir, configText) {
248
+ const { slug, default: Slugger } = await loadSlugger();
249
+ const config = JSON.parse(configText);
250
+
251
+ // --- What the site has -------------------------------------------------
252
+ const pageFiles = findAllPages(contentDir);
253
+ const pages = new Map(); // url -> { rel, abs, scan }
254
+ const urlOfFile = new Map(); // rel -> url
255
+ const scans = new Map(); // abs -> scanFile() result (or null)
256
+ const scanOnce = async (abs) => {
257
+ if (!scans.has(abs)) scans.set(abs, await scanFile(abs, { slug, Slugger }));
258
+ return scans.get(abs);
259
+ };
260
+ for (const rel of pageFiles) {
261
+ const abs = path.join(contentDir, rel);
262
+ const scan = await scanOnce(abs);
263
+ const id = scan?.data?.slug
264
+ ? normalizeEntryId(scan.data.slug)
265
+ : normalizeEntryId(
266
+ rel
267
+ .replace(/\.mdx?$/i, '')
268
+ .split('/')
269
+ .map((segment) => slug(segment))
270
+ .join('/')
271
+ .replace(/\/index$/, '')
272
+ );
273
+ const url = urlForId(id);
274
+ pages.set(url, { rel, abs, scan });
275
+ urlOfFile.set(rel, url);
276
+ }
277
+ const generated = generatedPageUrls(contentDir);
278
+ const redirects = new Set(
279
+ (Array.isArray(config.redirects) ? config.redirects : [])
280
+ .map((r) => (typeof r?.source === 'string' ? `/${r.source.replace(/^\/+|\/+$/g, '')}/`.replace(/^\/\/$/, '/') : null))
281
+ .filter(Boolean)
282
+ );
283
+ const special = new Set(['/llms.txt', '/llms-full.txt', '/404.html', '/404/']);
284
+ if (config.domain) ['/sitemap.xml', '/sitemap-index.xml'].forEach((p) => special.add(p));
285
+ const knownUrls = [...pages.keys(), ...generated, ...redirects];
286
+
287
+ // Anchors a page can be linked to: its own, plus those of every snippet
288
+ // it imports (they render inside it).
289
+ const snippetPath = (spec, fromAbs) =>
290
+ spec.startsWith('/') ? path.join(contentDir, spec) : path.resolve(path.dirname(fromAbs), spec);
291
+ const anchorCache = new Map();
292
+ async function anchorsOf(abs, seen = new Set()) {
293
+ if (anchorCache.has(abs)) return anchorCache.get(abs);
294
+ if (seen.has(abs)) return new Set();
295
+ seen.add(abs);
296
+ const scan = await scanOnce(abs);
297
+ const all = new Set(scan?.anchors ?? []);
298
+ for (const spec of scan?.imports ?? []) for (const a of await anchorsOf(snippetPath(spec, abs), seen)) all.add(a);
299
+ anchorCache.set(abs, all);
300
+ return all;
301
+ }
302
+
303
+ // Root-relative image/font paths something displays - see the "Any other
304
+ // file" case below. Snippets are scanned here too, so the set is complete
305
+ // before the first link is checked.
306
+ const displayedAssets = new Set();
307
+ async function collectDisplayed(abs, seen = new Set()) {
308
+ if (seen.has(abs)) return;
309
+ seen.add(abs);
310
+ const scan = await scanOnce(abs);
311
+ for (const link of scan?.links ?? []) if (link.kind !== 'link' && link.href.startsWith('/')) displayedAssets.add(link.href.split(/[?#]/)[0]);
312
+ for (const spec of scan?.imports ?? []) await collectDisplayed(snippetPath(spec, abs), seen);
313
+ }
314
+ for (const page of pages.values()) await collectDisplayed(page.abs);
315
+ (function walk(node) {
316
+ if (typeof node === 'string') {
317
+ if (node.startsWith('/') && ASSET_EXTENSIONS.has(path.posix.extname(node).toLowerCase())) displayedAssets.add(node);
318
+ } else if (Array.isArray(node)) node.forEach(walk);
319
+ else if (node && typeof node === 'object') Object.values(node).forEach(walk);
320
+ })(config);
321
+
322
+ // --- Checking one link -------------------------------------------------
323
+ const broken = [];
324
+ const reported = new Set();
325
+ let checked = 0;
326
+ let external = 0;
327
+
328
+ /** `pageUrl`: the URL of the page the link renders on (relative links
329
+ * resolve from it). `fileAbs`: the file the link is written in. */
330
+ async function check(href, { file, line, kind, pageUrl, fileAbs }) {
331
+ const value = String(href ?? '').trim();
332
+ if (!value || value === '#' || value.startsWith('{') || value.includes('{{')) return;
333
+ if (/^[a-z][a-z0-9+.-]*:/i.test(value) || value.startsWith('//')) {
334
+ external += 1;
335
+ return;
336
+ }
337
+ checked += 1;
338
+ const report = (message, suggestion) => {
339
+ const key = `${file}\n${line}\n${value}`;
340
+ if (reported.has(key)) return;
341
+ reported.add(key);
342
+ broken.push({ file, line, message, suggestion });
343
+ };
344
+
345
+ // A Markdown image with a relative path is resolved from the file's own
346
+ // folder at build time (Astro's image pipeline), not from the URL.
347
+ if (kind === 'image' && fileAbs && !value.startsWith('/') && !value.startsWith('#')) {
348
+ const target = path.resolve(path.dirname(fileAbs), decodeURI(value.split(/[?#]/)[0]));
349
+ if (!fs.existsSync(target)) report(`Image ${value} - there's no file at ${path.relative(contentDir, target).split(path.sep).join('/')}.`);
350
+ return;
351
+ }
352
+
353
+ let url;
354
+ try {
355
+ url = new URL(value, BASE + (pageUrl ?? '/'));
356
+ } catch {
357
+ report(`${value} - not a valid link.`);
358
+ return;
359
+ }
360
+ let pathname;
361
+ try {
362
+ pathname = decodeURI(url.pathname);
363
+ } catch {
364
+ pathname = url.pathname;
365
+ }
366
+ const hash = url.hash ? decodeURIComponent(url.hash.slice(1)) : '';
367
+ const extension = path.posix.extname(pathname.replace(/\/+$/, '')).toLowerCase();
368
+ const isRelative = !value.startsWith('/') && !value.startsWith('#');
369
+
370
+ // A page (or /x/index.html).
371
+ let pageKey = null;
372
+ if (!extension || pathname.endsWith('/index.html')) {
373
+ pageKey = pathname.replace(/\/index\.html$/, '/').replace(/\/?$/, '/');
374
+ }
375
+ if (pageKey !== null) {
376
+ if (pages.has(pageKey)) {
377
+ if (!hash || hash === 'top') return;
378
+ const anchors = await anchorsOf(pages.get(pageKey).abs);
379
+ if (anchors.has(hash)) return;
380
+ const suggestion = closest(hash, [...anchors]);
381
+ const onPage = pageKey === pageUrl && value.startsWith('#') ? 'this page' : pageKey;
382
+ report(
383
+ `${value} - ${onPage} has no heading or anchor "#${hash}".`,
384
+ suggestion ? `Did you mean #${suggestion}?` : undefined
385
+ );
386
+ return;
387
+ }
388
+ // "/" always exists: the home page, or a redirect to the first page.
389
+ if (pageKey === '/' || generated.has(pageKey) || redirects.has(pageKey) || special.has(pageKey.replace(/\/$/, '') || '/')) return;
390
+ if (special.has(pathname)) return;
391
+
392
+ // Mintlify-style relative link: resolved from the file's folder
393
+ // instead of the page's URL, would it land on a page?
394
+ if (isRelative && pageUrl && pageUrl !== '/') {
395
+ const siblingBase = pageUrl.replace(/[^/]+\/$/, '');
396
+ const alt = new URL(value, BASE + siblingBase);
397
+ const altKey = decodeURI(alt.pathname).replace(/\/?$/, '/');
398
+ if (pages.has(altKey) || generated.has(altKey)) {
399
+ report(
400
+ `${value} - relative links resolve from the page's own URL (${pageUrl}), so this one points to ${pageKey}, which doesn't exist.`,
401
+ `Write ${altKey}${alt.hash} instead.`
402
+ );
403
+ return;
404
+ }
405
+ }
406
+ const suggestion = closest(pageKey, knownUrls);
407
+ report(`${value} - there's no page at ${pageKey}.`, suggestion ? `Did you mean ${suggestion}?` : undefined);
408
+ return;
409
+ }
410
+
411
+ // A link to a page's source file (docs/setup.mdx) instead of its URL.
412
+ if (extension === '.mdx' || extension === '.md') {
413
+ const bare = pathname.replace(/\.mdx?$/i, '').replace(/\/?$/, '/');
414
+ if (extension === '.md' && config.contextMenu && pages.has(bare)) return; // the page's Markdown copy
415
+ // Written as a file path, so look it up as one: from the file's own
416
+ // folder when relative.
417
+ const filePath = isRelative && fileAbs ? path.resolve(path.dirname(fileAbs), decodeURI(value.split(/[?#]/)[0])) : path.join(contentDir, pathname);
418
+ const rel = path.relative(contentDir, filePath).split(path.sep).join('/');
419
+ const byFile = urlOfFile.get(rel) ?? (pages.has(bare) ? bare : null);
420
+ report(
421
+ `${value} - links to a source file, which isn't published.`,
422
+ byFile ? `Link to the page's URL instead: ${byFile}` : undefined
423
+ );
424
+ return;
425
+ }
426
+
427
+ // Any other file.
428
+ if (special.has(pathname)) return;
429
+ const segments = pathname.replace(/^\/+/, '').split('/');
430
+ if (fs.existsSync(path.join(contentDir, 'public', ...segments))) return;
431
+ const inProject = path.join(contentDir, ...segments);
432
+ if (fs.existsSync(inProject)) {
433
+ // Outside public/, an image or font is published only when something
434
+ // displays it - an <img src>, a Markdown image, a writedocs.json
435
+ // field - not for a plain link to it (lib/styles-asset-integration.js
436
+ // copies what the built pages' `src` attributes name).
437
+ if (ASSET_EXTENSIONS.has(extension) && (kind !== 'link' || displayedAssets.has(pathname))) return;
438
+ report(
439
+ `${value} - ${pathname} is in the project, but only files in public/ are published${
440
+ ASSET_EXTENSIONS.has(extension) ? ' for a link (an image outside it is published only where a page displays it)' : ''
441
+ }.`,
442
+ `Move it to public${pathname}.`
443
+ );
444
+ return;
445
+ }
446
+ report(`${value} - there's no file at ${pathname}.`);
447
+ }
448
+
449
+ // --- The pages, their snippets, and writedocs.json ---------------------
450
+ const relOf = (abs) => path.relative(contentDir, abs).split(path.sep).join('/');
451
+ const snippetsChecked = new Set();
452
+ async function checkFile(abs, pageUrl, seen) {
453
+ const scan = await scanOnce(abs);
454
+ if (!scan) return;
455
+ const file = relOf(abs);
456
+ for (const link of scan.links) await check(link.href, { file, line: link.line, kind: link.kind, pageUrl, fileAbs: abs });
457
+ for (const spec of scan.imports) {
458
+ const snippet = snippetPath(spec, abs);
459
+ if (seen.has(snippet) || snippetsChecked.has(`${snippet}\n${pageUrl}`)) continue;
460
+ snippetsChecked.add(`${snippet}\n${pageUrl}`);
461
+ await checkFile(snippet, pageUrl, new Set([...seen, snippet]));
462
+ }
463
+ }
464
+ for (const [url, page] of pages) await checkFile(page.abs, url, new Set([page.abs]));
465
+
466
+ const locate = createJsonLocator(configText);
467
+ for (const { href, path: jsonPath } of configLinks(config)) {
468
+ await check(href, { file: 'writedocs.json', line: locate(jsonPath)?.line, kind: 'link', pageUrl: '/' });
469
+ }
470
+
471
+ broken.sort((a, b) => a.file.localeCompare(b.file) || (a.line ?? 0) - (b.line ?? 0));
472
+ return { pages: pages.size, links: checked, external, broken };
473
+ }