@cudoment/cudoc 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/README.md +4 -3
  2. package/dist/document.d.ts +36 -1
  3. package/dist/document.d.ts.map +1 -1
  4. package/dist/document.js +54 -3
  5. package/dist/document.js.map +1 -1
  6. package/dist/internal/core/mdx/imports.d.ts +21 -0
  7. package/dist/internal/core/mdx/imports.d.ts.map +1 -0
  8. package/dist/internal/core/mdx/imports.js +80 -0
  9. package/dist/internal/core/mdx/imports.js.map +1 -0
  10. package/dist/internal/core/mdx/index.d.ts +1 -0
  11. package/dist/internal/core/mdx/index.d.ts.map +1 -1
  12. package/dist/internal/core/mdx/index.js +1 -0
  13. package/dist/internal/core/mdx/index.js.map +1 -1
  14. package/dist/internal/core/query/sections.d.ts +6 -0
  15. package/dist/internal/core/query/sections.d.ts.map +1 -1
  16. package/dist/internal/core/query/sections.js +8 -2
  17. package/dist/internal/core/query/sections.js.map +1 -1
  18. package/dist/internal/transforms/table-column-layout/create-table.d.ts.map +1 -1
  19. package/dist/internal/transforms/table-column-layout/create-table.js +40 -17
  20. package/dist/internal/transforms/table-column-layout/create-table.js.map +1 -1
  21. package/dist/internal/transforms/table-column-layout/index.d.ts +32 -0
  22. package/dist/internal/transforms/table-column-layout/index.d.ts.map +1 -1
  23. package/dist/internal/transforms/table-column-layout/index.js +66 -0
  24. package/dist/internal/transforms/table-column-layout/index.js.map +1 -1
  25. package/dist/markdown.d.ts +8 -0
  26. package/dist/markdown.d.ts.map +1 -1
  27. package/dist/markdown.js +9 -1
  28. package/dist/markdown.js.map +1 -1
  29. package/dist/node/check.d.ts +12 -6
  30. package/dist/node/check.d.ts.map +1 -1
  31. package/dist/node/check.js +460 -79
  32. package/dist/node/check.js.map +1 -1
  33. package/dist/node/cli.js +2 -56
  34. package/dist/node/cli.js.map +1 -1
  35. package/dist/node/collect.d.ts +41 -0
  36. package/dist/node/collect.d.ts.map +1 -0
  37. package/dist/node/collect.js +351 -0
  38. package/dist/node/collect.js.map +1 -0
  39. package/dist/node/command.d.ts +23 -0
  40. package/dist/node/command.d.ts.map +1 -0
  41. package/dist/node/command.js +116 -0
  42. package/dist/node/command.js.map +1 -0
  43. package/dist/node/dataset.d.ts +14 -1
  44. package/dist/node/dataset.d.ts.map +1 -1
  45. package/dist/node/dataset.js +44 -5
  46. package/dist/node/dataset.js.map +1 -1
  47. package/dist/node/glob.d.ts +21 -0
  48. package/dist/node/glob.d.ts.map +1 -0
  49. package/dist/node/glob.js +61 -0
  50. package/dist/node/glob.js.map +1 -0
  51. package/dist/node/library.d.ts +76 -6
  52. package/dist/node/library.d.ts.map +1 -1
  53. package/dist/node/library.js +63 -137
  54. package/dist/node/library.js.map +1 -1
  55. package/dist/node/load-ast.d.ts.map +1 -1
  56. package/dist/node/load-ast.js +4 -1
  57. package/dist/node/load-ast.js.map +1 -1
  58. package/dist/node/local-target.d.ts +39 -8
  59. package/dist/node/local-target.d.ts.map +1 -1
  60. package/dist/node/local-target.js +94 -13
  61. package/dist/node/local-target.js.map +1 -1
  62. package/dist/node/prepare-embeds.d.ts +52 -3
  63. package/dist/node/prepare-embeds.d.ts.map +1 -1
  64. package/dist/node/prepare-embeds.js +134 -8
  65. package/dist/node/prepare-embeds.js.map +1 -1
  66. package/dist/node/references.d.ts +39 -0
  67. package/dist/node/references.d.ts.map +1 -0
  68. package/dist/node/references.js +101 -0
  69. package/dist/node/references.js.map +1 -0
  70. package/dist/node/replace.d.ts +65 -0
  71. package/dist/node/replace.d.ts.map +1 -0
  72. package/dist/node/replace.js +285 -0
  73. package/dist/node/replace.js.map +1 -0
  74. package/dist/node/resolve-embed.d.ts +79 -1
  75. package/dist/node/resolve-embed.d.ts.map +1 -1
  76. package/dist/node/resolve-embed.js +480 -143
  77. package/dist/node/resolve-embed.js.map +1 -1
  78. package/dist/node/roots.d.ts +78 -0
  79. package/dist/node/roots.d.ts.map +1 -0
  80. package/dist/node/roots.js +170 -0
  81. package/dist/node/roots.js.map +1 -0
  82. package/dist/node/storage.d.ts +21 -3
  83. package/dist/node/storage.d.ts.map +1 -1
  84. package/dist/node/storage.js +242 -20
  85. package/dist/node/storage.js.map +1 -1
  86. package/dist/node/watch.d.ts +86 -0
  87. package/dist/node/watch.d.ts.map +1 -0
  88. package/dist/node/watch.js +174 -0
  89. package/dist/node/watch.js.map +1 -0
  90. package/dist/paged.d.ts +34 -0
  91. package/dist/paged.d.ts.map +1 -0
  92. package/dist/paged.js +44 -0
  93. package/dist/paged.js.map +1 -0
  94. package/dist/render.d.ts +1 -1
  95. package/dist/render.d.ts.map +1 -1
  96. package/dist/render.js +14 -0
  97. package/dist/render.js.map +1 -1
  98. package/dist/sections.d.ts.map +1 -1
  99. package/dist/sections.js +7 -3
  100. package/dist/sections.js.map +1 -1
  101. package/package.json +29 -16
  102. package/styles.css +5 -2
@@ -9,10 +9,20 @@
9
9
  * It resolves through the same code the build uses, so a reference this reports
10
10
  * as fine is one the build can resolve.
11
11
  */
12
- import { parseEmbedSpec } from "./resolve-embed.js";
12
+ import { unified } from "unified";
13
+ import remarkParse from "remark-parse";
14
+ import remarkGfm from "remark-gfm";
15
+ import remarkFrontmatter from "remark-frontmatter";
16
+ import remarkMdx from "remark-mdx";
17
+ import { visit } from "unist-util-visit";
18
+ import { capturedImage } from "../document.js";
19
+ import { DEFAULT_TABLE_COLUMNS, buildEmbedRow, extractCell, parseEmbedSpec, resolveDocumentReference, } from "./resolve-embed.js";
20
+ import { compileReplacedSection, replacedSectionSource, sectionText, unreplaceableSection, } from "./replace.js";
13
21
  import { collectSections } from "../sections.js";
14
- import { resolveLocalTarget } from "./local-target.js";
15
- /** A heading id that a slugger disambiguated, such as `overview-1`. */
22
+ import { isExternalPath, resolveLocalTarget, } from "./local-target.js";
23
+ import { resolveRoots } from "./roots.js";
24
+ import { EXTERNAL_URL, decodeComponent, idsInNode } from "./references.js";
25
+ /** The shape of a heading id a slugger disambiguated, such as `overview-1`. */
16
26
  const SUFFIXED = /-\d+$/;
17
27
  const walkNodes = (node, visit) => {
18
28
  visit(node);
@@ -23,19 +33,46 @@ const walkNodes = (node, visit) => {
23
33
  *
24
34
  * Both halves matter. The ids answer whether a link resolves; the origin
25
35
  * answers whether it will keep resolving, because a generated id depends on how
26
- * many same-named headings precede it.
36
+ * many same-named headings precede it. Besides headings, an id an element in
37
+ * raw HTML declares is an anchor too, and one the author wrote.
27
38
  */
28
39
  export function collectAnchors(tree) {
29
40
  const anchors = [];
30
41
  walkNodes(tree, (node) => {
31
- if (node.type !== "heading")
42
+ if (node.type !== "heading") {
43
+ for (const id of idsInNode(node))
44
+ if (id)
45
+ anchors.push({ id, explicit: true });
32
46
  return;
47
+ }
33
48
  const id = node.data?.hProperties?.id;
34
49
  if (typeof id === "string" && id)
35
50
  anchors.push({ id, explicit: node.data?.cudoc?.explicitId === true });
36
51
  });
37
52
  return anchors;
38
53
  }
54
+ /**
55
+ * The ids a section can be embedded from: headings only, since an id raw HTML
56
+ * declares starts no section.
57
+ */
58
+ const sectionIds = (anchors, tree) => {
59
+ const headings = new Set();
60
+ walkNodes(tree, (node) => {
61
+ const id = node.type === "heading" ? node.data?.hProperties?.id : undefined;
62
+ if (typeof id === "string" && id)
63
+ headings.add(id);
64
+ });
65
+ return anchors.map((entry) => entry.id).filter((id) => headings.has(id));
66
+ };
67
+ /**
68
+ * The anchor a fragment names, if the document has it. Hosts percent-encode a
69
+ * fragment that is not ASCII, so `#개요` may arrive as `#%EA%B0%9C%EC%9A%94`;
70
+ * either spelling names the same heading.
71
+ */
72
+ const findAnchor = (anchors, fragment) => {
73
+ const decoded = decodeComponent(fragment);
74
+ return anchors.find((entry) => entry.id === fragment || entry.id === decoded);
75
+ };
39
76
  /**
40
77
  * Where a reference sits in the original Markdown.
41
78
  *
@@ -119,6 +156,9 @@ const unportableComponents = (tree) => {
119
156
  return;
120
157
  if (!node.type.startsWith("mdx") && !node.type.endsWith("Directive"))
121
158
  return;
159
+ // A host's component for a Markdown image renders as that image.
160
+ if (capturedImage(node))
161
+ return;
122
162
  names.add(node.name ? `<${node.name}>` : node.type);
123
163
  });
124
164
  return [...names];
@@ -132,14 +172,120 @@ const headingText = (node) => {
132
172
  });
133
173
  return text;
134
174
  };
135
- /** The line each `cudoc-embed` fence opens on, in source order. */
136
- const embedFenceLines = (text) => {
137
- const lines = [];
138
- text.split("\n").forEach((line, index) => {
139
- if (/^\s*(?:`{3,}|~{3,})cudoc-embed\s*$/.test(line))
140
- lines.push(index + 1);
175
+ /**
176
+ * The `cudoc-embed` fences of a source, in order, as Markdown parses them: a
177
+ * fence shown inside a longer fence or an indented code block is example
178
+ * text, and one inside a quote or a list item is a block like any other. The
179
+ * source is parsed as Markdown with GFM and front matter, as MDX for an `.mdx`
180
+ * file.
181
+ */
182
+ const parsedFences = (text, mdx) => {
183
+ try {
184
+ const processor = unified()
185
+ .use(remarkParse)
186
+ .use(remarkGfm)
187
+ .use(remarkFrontmatter);
188
+ if (mdx)
189
+ processor.use(remarkMdx);
190
+ const fences = [];
191
+ visit(processor.parse(text), "code", (node) => {
192
+ if (node.lang === "cudoc-embed" && node.position)
193
+ fences.push({
194
+ line: node.position.start.line,
195
+ column: node.position.start.column,
196
+ value: node.value,
197
+ });
198
+ });
199
+ return fences;
200
+ }
201
+ catch {
202
+ // A host's own syntax may not parse here; the line scan below answers.
203
+ return undefined;
204
+ }
205
+ };
206
+ /**
207
+ * The same fences found line by line: a fence is closed the way Markdown
208
+ * closes it, by the same character, at least as long, and nothing after it,
209
+ * and its text is the lines between, less the opening fence's indentation.
210
+ */
211
+ const scannedFences = (text) => {
212
+ const found = [];
213
+ let open;
214
+ text.split(/\r?\n/).forEach((line, index) => {
215
+ const match = line.match(/^(\s*)(`{3,}|~{3,})(.*)$/);
216
+ if (open) {
217
+ const [, , marker = "", rest = ""] = match ?? [];
218
+ if (marker[0] === open.char &&
219
+ marker.length >= open.length &&
220
+ !rest.trim())
221
+ open = undefined;
222
+ else
223
+ open.content?.push(line.replace(open.indent, ""));
224
+ return;
225
+ }
226
+ if (!match)
227
+ return;
228
+ const [, indent = "", marker = "", rest = ""] = match;
229
+ open = {
230
+ char: marker[0],
231
+ length: marker.length,
232
+ indent: new RegExp(`^ {0,${indent.length}}`),
233
+ };
234
+ if (rest.trim().split(/\s+/)[0] !== "cudoc-embed")
235
+ return;
236
+ open.content = [];
237
+ found.push({
238
+ line: index + 1,
239
+ column: indent.length + 1,
240
+ content: open.content,
241
+ });
141
242
  });
142
- return lines;
243
+ return found.map(({ content, ...fence }) => ({
244
+ ...fence,
245
+ value: content.join("\n"),
246
+ }));
247
+ };
248
+ /** Block text as hosts may differ in keeping it: without `\r` or trailing whitespace. */
249
+ const comparable = (value) => value
250
+ .replace(/\r/g, "")
251
+ .split("\n")
252
+ .map((line) => line.trimEnd())
253
+ .join("\n")
254
+ .trimEnd();
255
+ /**
256
+ * The fence of each of a document's embed blocks, given their text in order,
257
+ * or nothing when neither reading of the source finds those blocks in that
258
+ * order: a host whose syntax reads a block differently from Markdown would
259
+ * otherwise put a block's error on another block's fence.
260
+ */
261
+ const embedFences = (text, mdx, values) => {
262
+ const wanted = values.map(comparable);
263
+ const found = (fences) => fences !== undefined &&
264
+ fences.length === wanted.length &&
265
+ fences.every((fence, index) => comparable(fence.value) === wanted[index]);
266
+ const parsed = parsedFences(text, mdx);
267
+ if (found(parsed))
268
+ return parsed;
269
+ const scanned = scannedFences(text);
270
+ return found(scanned) ? scanned : [];
271
+ };
272
+ /**
273
+ * Where an error in an embed block sits in the file. The YAML parser counts
274
+ * from the block's first line and column: the file's line adds the fence's,
275
+ * and the file's column adds whatever stands before the block's text on that
276
+ * line, indentation, a quote's `>` or a list item's offset alike. An error
277
+ * without a coordinate sits on the fence.
278
+ */
279
+ const blockPosition = (lines, fence, at) => {
280
+ if (!at)
281
+ return { line: fence.line, column: fence.column };
282
+ const line = fence.line + at.line;
283
+ const source = (lines[line - 1] ?? "").trimEnd();
284
+ const text = (comparable(fence.value).split("\n")[at.line - 1] ?? "").trimEnd();
285
+ const before = source.endsWith(text)
286
+ ? source.length - text.length
287
+ : fence.column - 1;
288
+ return { line, column: before + at.col };
143
289
  };
144
290
  /**
145
291
  * Whether a rule finds anything, asked the way the resolver asks it.
@@ -171,11 +317,25 @@ const matchedRules = (slice, rules) => {
171
317
  return matched;
172
318
  };
173
319
  export function checkReferences(library, options = {}) {
174
- const sourceRoot = options.sourceRoot ?? library.sourceRoot;
320
+ const roots = options.roots !== undefined || options.sourceRoot !== undefined
321
+ ? resolveRoots(options)
322
+ : library.roots;
323
+ const targetRoots = {
324
+ assetDirs: options.assetDirs,
325
+ withoutBase: options.withoutBase,
326
+ externalPaths: options.externalPaths,
327
+ };
175
328
  const ignore = new Set(options.ignore ?? []);
176
329
  const issues = [];
177
330
  let checkedReferences = 0;
331
+ // Where a copied component can render at all: an MDX host compiles the
332
+ // spliced nodes with its own component mapping, a Markdown host cannot.
333
+ const mdxHost = ["next", "docusaurus", "nextra"].includes(library.options.host ?? "");
178
334
  const anchorsById = new Map(library.documents.map((doc) => [doc.id, collectAnchors(doc.tree)]));
335
+ const sectionsById = new Map(library.documents.map((doc) => [
336
+ doc.id,
337
+ sectionIds(anchorsById.get(doc.id), doc.tree),
338
+ ]));
179
339
  const byId = new Map(library.documents.map((doc) => [doc.id, doc]));
180
340
  const report = (doc,
181
341
  /** A `position` given here wins; otherwise it is recovered from the source. */
@@ -194,11 +354,12 @@ export function checkReferences(library, options = {}) {
194
354
  };
195
355
  /** Resolves `other.md#anchor` or a bare `#anchor` against the library. */
196
356
  const checkDocumentLink = (doc, url) => {
197
- const [pathname, anchor] = url.split("#");
357
+ const [address = "", anchor] = url.split("#");
358
+ // A query names no other document: `guide/?tab=1` is still `guide/`.
359
+ const pathname = address.replace(/\?.*$/, "");
198
360
  let target = doc;
199
361
  if (pathname) {
200
- const id = resolveDocumentId(doc, pathname);
201
- const found = id === undefined ? undefined : byId.get(id);
362
+ const found = resolveDocument(doc, pathname);
202
363
  if (!found)
203
364
  return false; // not a document link; the asset pass handles it
204
365
  target = found;
@@ -206,7 +367,7 @@ export function checkReferences(library, options = {}) {
206
367
  if (!anchor)
207
368
  return true;
208
369
  const anchors = anchorsById.get(target.id) ?? [];
209
- const match = anchors.find((entry) => entry.id === anchor);
370
+ const match = findAnchor(anchors, anchor);
210
371
  if (!match) {
211
372
  report(doc, {
212
373
  code: "missing-anchor",
@@ -217,7 +378,12 @@ export function checkReferences(library, options = {}) {
217
378
  });
218
379
  return true;
219
380
  }
220
- if (!match.explicit && SUFFIXED.test(match.id))
381
+ // Only a suffix a slugger added is fragile: `## Version 2` is `version-2`
382
+ // because of its own text, and stays so. A repeat shows as the bare id
383
+ // being there too.
384
+ if (!match.explicit &&
385
+ SUFFIXED.test(match.id) &&
386
+ anchors.some((entry) => entry.id === match.id.replace(SUFFIXED, "")))
221
387
  report(doc, {
222
388
  code: "unstable-anchor-link",
223
389
  severity: "warning",
@@ -227,21 +393,135 @@ export function checkReferences(library, options = {}) {
227
393
  return true;
228
394
  };
229
395
  /**
230
- * A link's document id, if it names a document in this library.
396
+ * The document a link names, if it names one in this library.
231
397
  *
232
398
  * Hosts do not agree on how a link looks once compiled. Docusaurus and
233
399
  * Next.js leave `reference.md`, Nextra drops the extension, and VitePress
234
400
  * rewrites it to the deployed `./reference.html`. All three mean the same
235
- * document, so any of those endings resolves to the same id.
401
+ * document, so any of those endings resolves to the same id. A relative
402
+ * path resolves in the library's coordinates, so it can reach another root
403
+ * exactly when the bases mirror the directories; a root-relative one names
404
+ * a library path directly, and then once more with the deployment base
405
+ * removed, the way the exporter looks a route up.
236
406
  */
237
- const resolveDocumentId = (doc, pathname) => {
238
- if (/^(?:[a-z][\w+.-]*:|\/\/)/i.test(pathname))
407
+ const resolveDocument = (doc, pathname) => {
408
+ if (EXTERNAL_URL.test(pathname))
239
409
  return undefined;
240
- const decoded = decodeURIComponent(pathname);
241
- const joined = decoded.startsWith("/")
242
- ? decoded.slice(1)
243
- : posixJoin(dirname(doc.sourcePath), decoded);
244
- return normalize(joined).replace(/\.(?:mdx?|html?)$/i, "");
410
+ const decoded = decodeComponent(pathname);
411
+ // A directory, `./` or `guide/` as VitePress writes a link to an
412
+ // `index.md`, names that directory's index document; spelled with its
413
+ // trailing slash, it does even beside a `guide.md`.
414
+ const lookup = (value) => {
415
+ const normalized = normalize(value);
416
+ if (normalized === undefined)
417
+ return undefined;
418
+ const id = normalized.replace(/\.(?:mdx?|html?)$/i, "");
419
+ const index = byId.get(normalized ? `${normalized}/index` : "index");
420
+ if (value === "" || /(?:^|\/)(?:\.\.?)?$/.test(value))
421
+ return index ?? byId.get(id);
422
+ return byId.get(id) ?? index;
423
+ };
424
+ if (!decoded.startsWith("/"))
425
+ return lookup(posixJoin(dirname(doc.sourcePath), decoded));
426
+ const direct = lookup(decoded.slice(1));
427
+ if (direct || !options.withoutBase)
428
+ return direct;
429
+ return lookup(options.withoutBase(decoded).replace(/^\//, ""));
430
+ };
431
+ /**
432
+ * What an embed copies from one selected section: the section as
433
+ * collected, or, under replacement rules, its rewritten Markdown compiled
434
+ * again, which is the copy the resolver builds and expands. Without a
435
+ * synchronous host compiler, which `cudoc check` never has, the standalone
436
+ * compiler reads the rewritten Markdown with the library's options.
437
+ *
438
+ * `undefined` when the copy cannot be built here: the resolver refuses the
439
+ * rules, the snapshot or a repeated section id, which fails the build on
440
+ * its own, or the standalone compiler cannot read what only the host's
441
+ * parser accepts, such as Docusaurus's `{#id}` or an HTML comment in MDX.
442
+ * The section as collected is not a stand-in, because the rules may have
443
+ * changed exactly what an inspection looks for, so the caller skips it.
444
+ * Each copy is compiled once however many inspections and chains read it.
445
+ */
446
+ const rewritten = new Map();
447
+ const copiedTree = (document, section, spec) => {
448
+ const rules = spec.replace ?? [];
449
+ if (!rules.length)
450
+ return section.tree;
451
+ const includeChildren = spec.select?.includeChildren;
452
+ const key = JSON.stringify([
453
+ document.id,
454
+ section.anchorId ?? null,
455
+ includeChildren ?? true,
456
+ rules,
457
+ ]);
458
+ if (rewritten.has(key))
459
+ return rewritten.get(key);
460
+ let copy;
461
+ try {
462
+ copy = compileReplacedSection(library, document, replacedSectionSource(document, section.anchorId, rules, includeChildren), { standalone: true });
463
+ }
464
+ catch {
465
+ copy = undefined;
466
+ }
467
+ rewritten.set(key, copy);
468
+ return copy;
469
+ };
470
+ /**
471
+ * The chain of sections that brings an embed back to itself, as the
472
+ * resolver would report it, or `undefined`. It follows the embeds inside
473
+ * each copied section — only those are expanded, after the embed's own
474
+ * replacement rules have rewritten it — and stops at the depth the
475
+ * resolver gives up at.
476
+ */
477
+ const embedCycle = (spec, from, active) => {
478
+ for (const source of spec.sources ?? []) {
479
+ let resolved;
480
+ try {
481
+ resolved = resolveDocumentReference(library, String(source), from);
482
+ }
483
+ catch {
484
+ continue; // reported as a missing source
485
+ }
486
+ const { document, anchor } = resolved;
487
+ const key = `${document.id}#${anchor ?? "*"}`;
488
+ if (active.includes(key) || active.length >= 64)
489
+ return [...active, key];
490
+ let sections;
491
+ try {
492
+ sections =
493
+ anchor || spec.select
494
+ ? collectSections(document.tree, {
495
+ ...spec.select,
496
+ ...(anchor ? { anchors: [anchor] } : {}),
497
+ })
498
+ : [{ tree: document.tree }];
499
+ }
500
+ catch {
501
+ continue; // reported as a missing section
502
+ }
503
+ for (const section of sections) {
504
+ let found;
505
+ const copy = copiedTree(document, section, spec);
506
+ if (!copy)
507
+ continue; // what it would expand cannot be told here
508
+ walkNodes(copy, (node) => {
509
+ if (found || node.type !== "code" || node.lang !== "cudoc-embed")
510
+ return;
511
+ let nested;
512
+ try {
513
+ nested = parseEmbedSpec(node.value ?? "");
514
+ }
515
+ catch {
516
+ return; // reported where it is written
517
+ }
518
+ found = embedCycle(nested, document.id, [...active, key]);
519
+ });
520
+ if (found)
521
+ return found;
522
+ }
523
+ }
524
+ return undefined;
245
525
  };
246
526
  for (const doc of library.documents) {
247
527
  // Anchors the document declares, before anything references them.
@@ -276,7 +556,9 @@ export function checkReferences(library, options = {}) {
276
556
  });
277
557
  });
278
558
  walkNodes(doc.tree, (node) => {
279
- const url = node.type === "link" || node.type === "image" ? node.url : undefined;
559
+ const captured = capturedImage(node);
560
+ const image = node.type === "image" || captured !== undefined;
561
+ const url = node.type === "link" || node.type === "image" ? node.url : captured?.url;
280
562
  if (typeof url !== "string" || !url)
281
563
  return;
282
564
  // A heading permalink is machinery the host inserted, not something an
@@ -285,31 +567,43 @@ export function checkReferences(library, options = {}) {
285
567
  if (node.data?.cudoc?.kind === "permalink")
286
568
  return;
287
569
  checkedReferences++;
288
- if (/^(?:[a-z][\w+.-]*:|\/\/)/i.test(url))
570
+ if (EXTERNAL_URL.test(url))
289
571
  return; // external
572
+ // Another application's path on the same host: nothing here to check.
573
+ if (isExternalPath(url, options.externalPaths))
574
+ return;
290
575
  if (node.type === "link" && checkDocumentLink(doc, url))
291
576
  return;
292
577
  if (url.startsWith("#"))
293
578
  return; // handled above as a same-document anchor
294
- if (!sourceRoot)
579
+ if (!roots)
295
580
  return;
296
581
  const target = resolveLocalTarget(url, doc.sourcePath, {
297
- ...options,
298
- sourceRoot,
582
+ ...targetRoots,
583
+ roots,
299
584
  });
300
585
  if (target.kind === "missing")
301
586
  report(doc, {
302
- code: node.type === "image" ? "missing-asset" : "missing-document",
587
+ code: image ? "missing-asset" : "missing-document",
303
588
  severity: "error",
304
- message: node.type === "image"
305
- ? `no file for image ${url}; check sourceRoot and assetDirs`
589
+ message: image
590
+ ? `no file for image ${url}; check the collection roots and assetDirs`
306
591
  : `no document or file for ${url}`,
307
592
  reference: url,
308
593
  });
309
594
  });
310
595
  // Fence lines are read from the source because collection strips positions,
311
596
  // and they are what turns a YAML error's own coordinate into a file one.
312
- const fences = embedFenceLines(doc.source?.text ?? "");
597
+ const values = [];
598
+ walkNodes(doc.tree, (node) => {
599
+ if (node.type === "code" && node.lang === "cudoc-embed")
600
+ values.push(node.value ?? "");
601
+ });
602
+ const sourceText = doc.source?.text ?? "";
603
+ const fences = values.length
604
+ ? embedFences(sourceText, doc.source?.format === "mdx", values)
605
+ : [];
606
+ const sourceLines = sourceText.split(/\r?\n/);
313
607
  let blockNumber = 0;
314
608
  walkNodes(doc.tree, (node) => {
315
609
  if (node.type !== "code" || node.lang !== "cudoc-embed")
@@ -320,10 +614,8 @@ export function checkReferences(library, options = {}) {
320
614
  spec = parseEmbedSpec(node.value ?? "");
321
615
  }
322
616
  catch (error) {
323
- // The YAML parser counts from the start of the block; an author counts
324
- // from the start of the file. Add the fence so both agree.
325
- const at = error.linePos?.[0];
326
- const line = fence === undefined ? undefined : fence + (at?.line ?? 0);
617
+ const position = fence &&
618
+ blockPosition(sourceLines, fence, error.linePos?.[0]);
327
619
  report(doc, {
328
620
  code: "invalid-embed-spec",
329
621
  severity: "error",
@@ -336,50 +628,67 @@ export function checkReferences(library, options = {}) {
336
628
  .replace(/\s*at line \d+, column \d+:?\s*$/, "")
337
629
  .trim(),
338
630
  reference: `embed block ${blockNumber}`,
339
- position: line === undefined
340
- ? undefined
341
- : {
342
- start: { line, column: at?.col ?? 1 },
343
- end: { line, column: at?.col ?? 1 },
344
- },
631
+ position: position && { start: position, end: position },
345
632
  });
346
633
  return;
347
634
  }
635
+ const cycle = embedCycle(spec, doc.id, []);
636
+ if (cycle)
637
+ report(doc, {
638
+ code: "cyclic-embed",
639
+ severity: "error",
640
+ message: `embed block ${blockNumber} never finishes: ${cycle.join(" -> ")}. Each embed copies content that embeds the next, back to the first.`,
641
+ reference: `embed block ${blockNumber}`,
642
+ position: fence && {
643
+ start: { line: fence.line, column: fence.column },
644
+ end: { line: fence.line, column: fence.column },
645
+ },
646
+ });
348
647
  for (const reference of spec.sources ?? []) {
349
648
  checkedReferences++;
350
- const [pathname, anchor] = String(reference).split("#");
351
- const id = resolveDocumentId(doc, pathname);
352
- const target = id === undefined ? undefined : byId.get(id);
353
- if (!target) {
649
+ // The resolver's own reading of the source, so an embed this passes
650
+ // is one the build can find.
651
+ let resolved;
652
+ try {
653
+ resolved = resolveDocumentReference(library, String(reference), doc.id);
654
+ }
655
+ catch (error) {
354
656
  report(doc, {
355
657
  code: "missing-embed-source",
356
658
  severity: "error",
357
- message: `embed source ${reference} does not name a collected document`,
659
+ message: error.message.replace(/^cudoc: /, ""),
358
660
  reference: String(reference),
359
661
  });
360
662
  continue;
361
663
  }
362
- const anchors = anchorsById.get(target.id) ?? [];
363
- if (anchor && !anchors.some((entry) => entry.id === anchor)) {
664
+ const { document: target, anchor } = resolved;
665
+ const sections = sectionsById.get(target.id) ?? [];
666
+ if (anchor && !sections.includes(anchor)) {
364
667
  report(doc, {
365
668
  code: "missing-embed-anchor",
366
669
  severity: "error",
367
670
  message: `${target.id} has no section #${anchor} to embed`,
368
671
  reference: String(reference),
369
- available: anchors.map((entry) => entry.id),
672
+ available: sections,
370
673
  });
371
674
  continue;
372
675
  }
373
676
  // The same selection the resolver will apply, so what is inspected is
374
- // what would actually be copied. `select.anchors` can name a section
375
- // that does not exist, which `collectSections` rejects; the build hits
376
- // the same error, so it is reported rather than swallowed.
377
- const selection = anchor ? { anchors: [anchor] } : spec.select;
677
+ // what would actually be copied: a section named in the source is
678
+ // combined with `select`, as the resolver combines them. `select`
679
+ // can name a section that does not exist, which `collectSections`
680
+ // rejects; the build hits the same error, so it is reported rather
681
+ // than swallowed.
682
+ const selection = anchor || spec.select
683
+ ? { ...spec.select, ...(anchor ? { anchors: [anchor] } : {}) }
684
+ : undefined;
378
685
  let copied;
379
686
  try {
380
687
  copied = selection
381
688
  ? collectSections(target.tree, selection)
382
689
  : [{ anchorId: undefined, tree: target.tree }];
690
+ if (!copied.length)
691
+ throw new Error("no sections matched");
383
692
  }
384
693
  catch (error) {
385
694
  report(doc, {
@@ -387,26 +696,32 @@ export function checkReferences(library, options = {}) {
387
696
  severity: "error",
388
697
  message: `${target.id}: ${error.message.replace(/^cudoc: /, "")}`,
389
698
  reference: String(reference),
390
- available: anchors.map((entry) => entry.id),
699
+ available: sections,
391
700
  });
392
701
  continue;
393
702
  }
394
703
  // Replacement runs on the original Markdown, not the tree, so the
395
- // slices are recomputed the way `transformedSection` cuts them. A rule
704
+ // text is read through `sectionText`, as the resolver reads it. A rule
396
705
  // that matches no slice changed nothing, and the embed silently shows
397
706
  // the source's own wording in a place written to expect otherwise.
398
707
  const rules = spec.replace ?? [];
708
+ // The resolver refuses rules on a section inside another block whose
709
+ // text read on its own is not that section, and the build stops there.
710
+ if (rules.length)
711
+ for (const section of copied) {
712
+ const problem = unreplaceableSection(target, section.anchorId, spec.select?.includeChildren);
713
+ if (problem)
714
+ report(doc, {
715
+ code: "unreplaceable-embed-section",
716
+ severity: "error",
717
+ message: problem,
718
+ reference: String(reference),
719
+ });
720
+ }
399
721
  if (rules.length && target.source) {
400
722
  const matched = rules.map(() => false);
401
723
  for (const section of copied) {
402
- const range = section.anchorId
403
- ? target.source.sections[section.anchorId]
404
- : undefined;
405
- const slice = range
406
- ? target.source.text.slice(range.start, spec.select?.includeChildren === false
407
- ? (range.ownEnd ?? range.end)
408
- : range.end)
409
- : target.source.text;
724
+ const slice = sectionText(target, section.anchorId, spec.select?.includeChildren);
410
725
  matchedRules(slice, rules).forEach((hit, index) => {
411
726
  if (hit)
412
727
  matched[index] = true;
@@ -423,19 +738,83 @@ export function checkReferences(library, options = {}) {
423
738
  });
424
739
  });
425
740
  }
426
- // A summary table carries heading text and nothing else, so whatever
427
- // else the body holds never travels. Replacement above still applies,
428
- // because it reaches the title and summary columns.
429
- if (typeof spec.render === "object" && spec.render?.type === "table")
741
+ // What the rest inspects is the copy itself, after the rules above
742
+ // have rewritten it, as the resolver builds it. A section whose copy
743
+ // cannot be built here is left out rather than guessed at.
744
+ const copies = copied.flatMap((section) => {
745
+ const tree = copiedTree(target, section, spec);
746
+ return tree ? [{ anchorId: section.anchorId, tree }] : [];
747
+ });
748
+ // A table carries extracted text and nothing else, so whatever else
749
+ // the body holds never travels. What can go wrong is a column that
750
+ // finds nothing in a row: the resolver renders an empty cell, and the
751
+ // author would only notice by reading the page.
752
+ if (typeof spec.render === "object" && spec.render?.type === "table") {
753
+ const columns = spec.render.columns ?? DEFAULT_TABLE_COLUMNS;
754
+ for (const section of copies) {
755
+ const row = buildEmbedRow(target, section.anchorId, section.tree);
756
+ columns.forEach((column, index) => {
757
+ // A shorthand column is best effort: `summary` of a section
758
+ // that opens with a table is legitimately blank. A column
759
+ // written as a mapping states what every row must have.
760
+ if (typeof column === "string")
761
+ return;
762
+ let problem;
763
+ try {
764
+ problem = extractCell(library, column, row, {
765
+ documentId: doc.id,
766
+ }).problem;
767
+ }
768
+ catch (error) {
769
+ // An extractor the configuration does not register is a
770
+ // setting problem, reported once per column like a bad key.
771
+ report(doc, {
772
+ code: "invalid-embed-spec",
773
+ severity: "error",
774
+ message: error.message.replace(/^cudoc: /, ""),
775
+ reference: String(reference),
776
+ });
777
+ return;
778
+ }
779
+ if (!problem)
780
+ return;
781
+ const header = typeof column === "string"
782
+ ? column
783
+ : (column.header ??
784
+ (typeof column.value === "string" ? column.value : ""));
785
+ report(doc, {
786
+ code: "empty-embed-cell",
787
+ severity: "warning",
788
+ message: `column ${index + 1}${header ? ` "${header}"` : ""} is empty for the row from ${row.document.id}${row.section.anchorId ? `#${row.section.anchorId}` : ""}: ${problem}`,
789
+ reference: String(reference),
790
+ });
791
+ });
792
+ }
430
793
  continue;
794
+ }
431
795
  const names = [
432
- ...new Set(copied.flatMap((section) => unportableComponents(section.tree))),
796
+ ...new Set(copies.flatMap((section) => unportableComponents(section.tree))),
433
797
  ];
434
798
  if (names.length)
435
799
  report(doc, {
436
800
  code: "unportable-embed-component",
437
801
  severity: "warning",
438
- message: `this embed copies ${names.join(", ")} out of ${target.id}. A host that owns those components renders them, but standalone HTML export has no renderer for them and fails. Keep embedded sections to Markdown, or pass a renderer for each name.`,
802
+ message: mdxHost
803
+ ? `this embed copies ${names.join(", ")} out of ${target.id}. The host renders them where the copy is spliced in, but standalone HTML export has no renderer for them and fails. Keep embedded sections to Markdown, or pass a renderer for each name.`
804
+ : `this embed copies ${names.join(", ")} out of ${target.id}. Neither this host nor standalone HTML export can render them. Keep embedded sections to Markdown, or pass a renderer for each name.`,
805
+ reference: String(reference),
806
+ });
807
+ // A component the source file imports for itself does not travel
808
+ // with the copy: the export drops the import, and the embedding
809
+ // document has no binding for the name unless it imports it too.
810
+ const orphaned = names
811
+ .map((name) => name.replace(/^<|>$/g, "").split(".")[0])
812
+ .filter((name) => target.imports?.includes(name) && !doc.imports?.includes(name));
813
+ if (orphaned.length)
814
+ report(doc, {
815
+ code: "imported-embed-component",
816
+ severity: "error",
817
+ message: `this embed copies ${orphaned.map((name) => `<${name}>`).join(", ")} out of ${target.id}, which imports ${orphaned.length === 1 ? "it" : "them"} in its own file. ${doc.id} has no such import, so the spliced copy cannot render ${orphaned.length === 1 ? "it" : "them"}. Provide the component through the host's shared components, import it in ${doc.id} as well, or move it out of the embedded section.`,
439
818
  reference: String(reference),
440
819
  });
441
820
  }
@@ -443,22 +822,24 @@ export function checkReferences(library, options = {}) {
443
822
  }
444
823
  return { issues, documentCount: library.documents.length, checkedReferences };
445
824
  }
446
- // Posix path helpers, kept local so this module stays usable without importing
447
- // the whole node path surface into a browser-safe boundary by accident.
825
+ // Posix path helpers for library paths, which are always `/`-separated
826
+ // whatever the platform, so `node:path` with its native separators would be
827
+ // the wrong tool even on Node.
448
828
  const dirname = (value) => {
449
829
  const at = value.lastIndexOf("/");
450
830
  return at <= 0 ? "." : value.slice(0, at);
451
831
  };
452
832
  const posixJoin = (base, value) => base === "." ? value : `${base}/${value}`;
833
+ /** The normalized path, or `undefined` when `..` climbs out of the library. */
453
834
  const normalize = (value) => {
454
835
  const out = [];
455
836
  for (const part of value.split("/")) {
456
837
  if (!part || part === ".")
457
838
  continue;
458
- if (part === "..")
459
- out.pop();
460
- else
839
+ if (part !== "..")
461
840
  out.push(part);
841
+ else if (!out.pop())
842
+ return undefined;
462
843
  }
463
844
  return out.join("/");
464
845
  };