@cudoment/cudoc 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +3 -2
  2. package/dist/document.d.ts +25 -0
  3. package/dist/document.d.ts.map +1 -1
  4. package/dist/document.js +21 -1
  5. package/dist/document.js.map +1 -1
  6. package/dist/internal/core/query/sections.d.ts +6 -0
  7. package/dist/internal/core/query/sections.d.ts.map +1 -1
  8. package/dist/internal/core/query/sections.js +8 -2
  9. package/dist/internal/core/query/sections.js.map +1 -1
  10. package/dist/node/check.d.ts +7 -4
  11. package/dist/node/check.d.ts.map +1 -1
  12. package/dist/node/check.js +368 -65
  13. package/dist/node/check.js.map +1 -1
  14. package/dist/node/cli.js +2 -81
  15. package/dist/node/cli.js.map +1 -1
  16. package/dist/node/collect.d.ts +41 -0
  17. package/dist/node/collect.d.ts.map +1 -0
  18. package/dist/node/collect.js +351 -0
  19. package/dist/node/collect.js.map +1 -0
  20. package/dist/node/command.d.ts +23 -0
  21. package/dist/node/command.d.ts.map +1 -0
  22. package/dist/node/command.js +116 -0
  23. package/dist/node/command.js.map +1 -0
  24. package/dist/node/dataset.d.ts.map +1 -1
  25. package/dist/node/dataset.js +4 -1
  26. package/dist/node/dataset.js.map +1 -1
  27. package/dist/node/library.d.ts +9 -5
  28. package/dist/node/library.d.ts.map +1 -1
  29. package/dist/node/library.js +15 -253
  30. package/dist/node/library.js.map +1 -1
  31. package/dist/node/load-ast.d.ts.map +1 -1
  32. package/dist/node/load-ast.js +4 -1
  33. package/dist/node/load-ast.js.map +1 -1
  34. package/dist/node/local-target.d.ts +11 -0
  35. package/dist/node/local-target.d.ts.map +1 -1
  36. package/dist/node/local-target.js +54 -2
  37. package/dist/node/local-target.js.map +1 -1
  38. package/dist/node/prepare-embeds.d.ts +36 -1
  39. package/dist/node/prepare-embeds.d.ts.map +1 -1
  40. package/dist/node/prepare-embeds.js +94 -4
  41. package/dist/node/prepare-embeds.js.map +1 -1
  42. package/dist/node/references.d.ts +39 -0
  43. package/dist/node/references.d.ts.map +1 -0
  44. package/dist/node/references.js +101 -0
  45. package/dist/node/references.js.map +1 -0
  46. package/dist/node/replace.d.ts +65 -0
  47. package/dist/node/replace.d.ts.map +1 -0
  48. package/dist/node/replace.js +285 -0
  49. package/dist/node/replace.js.map +1 -0
  50. package/dist/node/resolve-embed.d.ts.map +1 -1
  51. package/dist/node/resolve-embed.js +232 -103
  52. package/dist/node/resolve-embed.js.map +1 -1
  53. package/dist/node/roots.d.ts +12 -1
  54. package/dist/node/roots.d.ts.map +1 -1
  55. package/dist/node/roots.js +47 -11
  56. package/dist/node/roots.js.map +1 -1
  57. package/dist/node/storage.d.ts.map +1 -1
  58. package/dist/node/storage.js +202 -9
  59. package/dist/node/storage.js.map +1 -1
  60. package/dist/node/watch.d.ts +9 -4
  61. package/dist/node/watch.d.ts.map +1 -1
  62. package/dist/node/watch.js +51 -17
  63. package/dist/node/watch.js.map +1 -1
  64. package/dist/render.d.ts +1 -1
  65. package/dist/render.d.ts.map +1 -1
  66. package/dist/render.js +14 -0
  67. package/dist/render.js.map +1 -1
  68. package/dist/sections.d.ts.map +1 -1
  69. package/dist/sections.js +7 -3
  70. package/dist/sections.js.map +1 -1
  71. package/package.json +15 -15
@@ -9,11 +9,20 @@
9
9
  * It resolves through the same code the build uses, so a reference this reports
10
10
  * as fine is one the build can resolve.
11
11
  */
12
- import { DEFAULT_TABLE_COLUMNS, buildEmbedRow, extractCell, parseEmbedSpec, } from "./resolve-embed.js";
12
+ import { unified } from "unified";
13
+ import remarkParse from "remark-parse";
14
+ import remarkGfm from "remark-gfm";
15
+ import remarkFrontmatter from "remark-frontmatter";
16
+ import remarkMdx from "remark-mdx";
17
+ import { visit } from "unist-util-visit";
18
+ import { capturedImage } from "../document.js";
19
+ import { DEFAULT_TABLE_COLUMNS, buildEmbedRow, extractCell, parseEmbedSpec, resolveDocumentReference, } from "./resolve-embed.js";
20
+ import { compileReplacedSection, replacedSectionSource, sectionText, unreplaceableSection, } from "./replace.js";
13
21
  import { collectSections } from "../sections.js";
14
22
  import { isExternalPath, resolveLocalTarget, } from "./local-target.js";
15
23
  import { resolveRoots } from "./roots.js";
16
- /** A heading id that a slugger disambiguated, such as `overview-1`. */
24
+ import { EXTERNAL_URL, decodeComponent, idsInNode } from "./references.js";
25
+ /** The shape of a heading id a slugger disambiguated, such as `overview-1`. */
17
26
  const SUFFIXED = /-\d+$/;
18
27
  const walkNodes = (node, visit) => {
19
28
  visit(node);
@@ -24,19 +33,46 @@ const walkNodes = (node, visit) => {
24
33
  *
25
34
  * Both halves matter. The ids answer whether a link resolves; the origin
26
35
  * answers whether it will keep resolving, because a generated id depends on how
27
- * many same-named headings precede it.
36
+ * many same-named headings precede it. Besides headings, an id an element in
37
+ * raw HTML declares is an anchor too, and one the author wrote.
28
38
  */
29
39
  export function collectAnchors(tree) {
30
40
  const anchors = [];
31
41
  walkNodes(tree, (node) => {
32
- if (node.type !== "heading")
42
+ if (node.type !== "heading") {
43
+ for (const id of idsInNode(node))
44
+ if (id)
45
+ anchors.push({ id, explicit: true });
33
46
  return;
47
+ }
34
48
  const id = node.data?.hProperties?.id;
35
49
  if (typeof id === "string" && id)
36
50
  anchors.push({ id, explicit: node.data?.cudoc?.explicitId === true });
37
51
  });
38
52
  return anchors;
39
53
  }
54
+ /**
55
+ * The ids a section can be embedded from: headings only, since an id raw HTML
56
+ * declares starts no section.
57
+ */
58
+ const sectionIds = (anchors, tree) => {
59
+ const headings = new Set();
60
+ walkNodes(tree, (node) => {
61
+ const id = node.type === "heading" ? node.data?.hProperties?.id : undefined;
62
+ if (typeof id === "string" && id)
63
+ headings.add(id);
64
+ });
65
+ return anchors.map((entry) => entry.id).filter((id) => headings.has(id));
66
+ };
67
+ /**
68
+ * The anchor a fragment names, if the document has it. Hosts percent-encode a
69
+ * fragment that is not ASCII, so `#개요` may arrive as `#%EA%B0%9C%EC%9A%94`;
70
+ * either spelling names the same heading.
71
+ */
72
+ const findAnchor = (anchors, fragment) => {
73
+ const decoded = decodeComponent(fragment);
74
+ return anchors.find((entry) => entry.id === fragment || entry.id === decoded);
75
+ };
40
76
  /**
41
77
  * Where a reference sits in the original Markdown.
42
78
  *
@@ -120,6 +156,9 @@ const unportableComponents = (tree) => {
120
156
  return;
121
157
  if (!node.type.startsWith("mdx") && !node.type.endsWith("Directive"))
122
158
  return;
159
+ // A host's component for a Markdown image renders as that image.
160
+ if (capturedImage(node))
161
+ return;
123
162
  names.add(node.name ? `<${node.name}>` : node.type);
124
163
  });
125
164
  return [...names];
@@ -133,14 +172,120 @@ const headingText = (node) => {
133
172
  });
134
173
  return text;
135
174
  };
136
- /** The line each `cudoc-embed` fence opens on, in source order. */
137
- const embedFenceLines = (text) => {
138
- const lines = [];
139
- text.split("\n").forEach((line, index) => {
140
- if (/^\s*(?:`{3,}|~{3,})cudoc-embed\s*$/.test(line))
141
- lines.push(index + 1);
175
+ /**
176
+ * The `cudoc-embed` fences of a source, in order, as Markdown parses them: a
177
+ * fence shown inside a longer fence or an indented code block is example
178
+ * text, and one inside a quote or a list item is a block like any other. The
179
+ * source is parsed as Markdown with GFM and front matter, as MDX for an `.mdx`
180
+ * file.
181
+ */
182
+ const parsedFences = (text, mdx) => {
183
+ try {
184
+ const processor = unified()
185
+ .use(remarkParse)
186
+ .use(remarkGfm)
187
+ .use(remarkFrontmatter);
188
+ if (mdx)
189
+ processor.use(remarkMdx);
190
+ const fences = [];
191
+ visit(processor.parse(text), "code", (node) => {
192
+ if (node.lang === "cudoc-embed" && node.position)
193
+ fences.push({
194
+ line: node.position.start.line,
195
+ column: node.position.start.column,
196
+ value: node.value,
197
+ });
198
+ });
199
+ return fences;
200
+ }
201
+ catch {
202
+ // A host's own syntax may not parse here; the line scan below answers.
203
+ return undefined;
204
+ }
205
+ };
206
+ /**
207
+ * The same fences found line by line: a fence is closed the way Markdown
208
+ * closes it, by the same character, at least as long, and nothing after it,
209
+ * and its text is the lines between, less the opening fence's indentation.
210
+ */
211
+ const scannedFences = (text) => {
212
+ const found = [];
213
+ let open;
214
+ text.split(/\r?\n/).forEach((line, index) => {
215
+ const match = line.match(/^(\s*)(`{3,}|~{3,})(.*)$/);
216
+ if (open) {
217
+ const [, , marker = "", rest = ""] = match ?? [];
218
+ if (marker[0] === open.char &&
219
+ marker.length >= open.length &&
220
+ !rest.trim())
221
+ open = undefined;
222
+ else
223
+ open.content?.push(line.replace(open.indent, ""));
224
+ return;
225
+ }
226
+ if (!match)
227
+ return;
228
+ const [, indent = "", marker = "", rest = ""] = match;
229
+ open = {
230
+ char: marker[0],
231
+ length: marker.length,
232
+ indent: new RegExp(`^ {0,${indent.length}}`),
233
+ };
234
+ if (rest.trim().split(/\s+/)[0] !== "cudoc-embed")
235
+ return;
236
+ open.content = [];
237
+ found.push({
238
+ line: index + 1,
239
+ column: indent.length + 1,
240
+ content: open.content,
241
+ });
142
242
  });
143
- return lines;
243
+ return found.map(({ content, ...fence }) => ({
244
+ ...fence,
245
+ value: content.join("\n"),
246
+ }));
247
+ };
248
+ /** Block text as hosts may differ in keeping it: without `\r` or trailing whitespace. */
249
+ const comparable = (value) => value
250
+ .replace(/\r/g, "")
251
+ .split("\n")
252
+ .map((line) => line.trimEnd())
253
+ .join("\n")
254
+ .trimEnd();
255
+ /**
256
+ * The fence of each of a document's embed blocks, given their text in order,
257
+ * or nothing when neither reading of the source finds those blocks in that
258
+ * order: a host whose syntax reads a block differently from Markdown would
259
+ * otherwise put a block's error on another block's fence.
260
+ */
261
+ const embedFences = (text, mdx, values) => {
262
+ const wanted = values.map(comparable);
263
+ const found = (fences) => fences !== undefined &&
264
+ fences.length === wanted.length &&
265
+ fences.every((fence, index) => comparable(fence.value) === wanted[index]);
266
+ const parsed = parsedFences(text, mdx);
267
+ if (found(parsed))
268
+ return parsed;
269
+ const scanned = scannedFences(text);
270
+ return found(scanned) ? scanned : [];
271
+ };
272
+ /**
273
+ * Where an error in an embed block sits in the file. The YAML parser counts
274
+ * from the block's first line and column: the file's line adds the fence's,
275
+ * and the file's column adds whatever stands before the block's text on that
276
+ * line, indentation, a quote's `>` or a list item's offset alike. An error
277
+ * without a coordinate sits on the fence.
278
+ */
279
+ const blockPosition = (lines, fence, at) => {
280
+ if (!at)
281
+ return { line: fence.line, column: fence.column };
282
+ const line = fence.line + at.line;
283
+ const source = (lines[line - 1] ?? "").trimEnd();
284
+ const text = (comparable(fence.value).split("\n")[at.line - 1] ?? "").trimEnd();
285
+ const before = source.endsWith(text)
286
+ ? source.length - text.length
287
+ : fence.column - 1;
288
+ return { line, column: before + at.col };
144
289
  };
145
290
  /**
146
291
  * Whether a rule finds anything, asked the way the resolver asks it.
@@ -187,6 +332,10 @@ export function checkReferences(library, options = {}) {
187
332
  // spliced nodes with its own component mapping, a Markdown host cannot.
188
333
  const mdxHost = ["next", "docusaurus", "nextra"].includes(library.options.host ?? "");
189
334
  const anchorsById = new Map(library.documents.map((doc) => [doc.id, collectAnchors(doc.tree)]));
335
+ const sectionsById = new Map(library.documents.map((doc) => [
336
+ doc.id,
337
+ sectionIds(anchorsById.get(doc.id), doc.tree),
338
+ ]));
190
339
  const byId = new Map(library.documents.map((doc) => [doc.id, doc]));
191
340
  const report = (doc,
192
341
  /** A `position` given here wins; otherwise it is recovered from the source. */
@@ -205,7 +354,9 @@ export function checkReferences(library, options = {}) {
205
354
  };
206
355
  /** Resolves `other.md#anchor` or a bare `#anchor` against the library. */
207
356
  const checkDocumentLink = (doc, url) => {
208
- const [pathname, anchor] = url.split("#");
357
+ const [address = "", anchor] = url.split("#");
358
+ // A query names no other document: `guide/?tab=1` is still `guide/`.
359
+ const pathname = address.replace(/\?.*$/, "");
209
360
  let target = doc;
210
361
  if (pathname) {
211
362
  const found = resolveDocument(doc, pathname);
@@ -216,7 +367,7 @@ export function checkReferences(library, options = {}) {
216
367
  if (!anchor)
217
368
  return true;
218
369
  const anchors = anchorsById.get(target.id) ?? [];
219
- const match = anchors.find((entry) => entry.id === anchor);
370
+ const match = findAnchor(anchors, anchor);
220
371
  if (!match) {
221
372
  report(doc, {
222
373
  code: "missing-anchor",
@@ -227,7 +378,12 @@ export function checkReferences(library, options = {}) {
227
378
  });
228
379
  return true;
229
380
  }
230
- if (!match.explicit && SUFFIXED.test(match.id))
381
+ // Only a suffix a slugger added is fragile: `## Version 2` is `version-2`
382
+ // because of its own text, and stays so. A repeat shows as the bare id
383
+ // being there too.
384
+ if (!match.explicit &&
385
+ SUFFIXED.test(match.id) &&
386
+ anchors.some((entry) => entry.id === match.id.replace(SUFFIXED, "")))
231
387
  report(doc, {
232
388
  code: "unstable-anchor-link",
233
389
  severity: "warning",
@@ -249,16 +405,123 @@ export function checkReferences(library, options = {}) {
249
405
  * removed, the way the exporter looks a route up.
250
406
  */
251
407
  const resolveDocument = (doc, pathname) => {
252
- if (/^(?:[a-z][\w+.-]*:|\/\/)/i.test(pathname))
408
+ if (EXTERNAL_URL.test(pathname))
253
409
  return undefined;
254
- const decoded = decodeURIComponent(pathname);
255
- const toId = (value) => normalize(value).replace(/\.(?:mdx?|html?)$/i, "");
410
+ const decoded = decodeComponent(pathname);
411
+ // A directory, `./` or `guide/` as VitePress writes a link to an
412
+ // `index.md`, names that directory's index document; spelled with its
413
+ // trailing slash, it does even beside a `guide.md`.
414
+ const lookup = (value) => {
415
+ const normalized = normalize(value);
416
+ if (normalized === undefined)
417
+ return undefined;
418
+ const id = normalized.replace(/\.(?:mdx?|html?)$/i, "");
419
+ const index = byId.get(normalized ? `${normalized}/index` : "index");
420
+ if (value === "" || /(?:^|\/)(?:\.\.?)?$/.test(value))
421
+ return index ?? byId.get(id);
422
+ return byId.get(id) ?? index;
423
+ };
256
424
  if (!decoded.startsWith("/"))
257
- return byId.get(toId(posixJoin(dirname(doc.sourcePath), decoded)));
258
- const direct = byId.get(toId(decoded.slice(1)));
425
+ return lookup(posixJoin(dirname(doc.sourcePath), decoded));
426
+ const direct = lookup(decoded.slice(1));
259
427
  if (direct || !options.withoutBase)
260
428
  return direct;
261
- return byId.get(toId(options.withoutBase(decoded).replace(/^\//, "")));
429
+ return lookup(options.withoutBase(decoded).replace(/^\//, ""));
430
+ };
431
+ /**
432
+ * What an embed copies from one selected section: the section as
433
+ * collected, or, under replacement rules, its rewritten Markdown compiled
434
+ * again, which is the copy the resolver builds and expands. Without a
435
+ * synchronous host compiler, which `cudoc check` never has, the standalone
436
+ * compiler reads the rewritten Markdown with the library's options.
437
+ *
438
+ * `undefined` when the copy cannot be built here: the resolver refuses the
439
+ * rules, the snapshot or a repeated section id, which fails the build on
440
+ * its own, or the standalone compiler cannot read what only the host's
441
+ * parser accepts, such as Docusaurus's `{#id}` or an HTML comment in MDX.
442
+ * The section as collected is not a stand-in, because the rules may have
443
+ * changed exactly what an inspection looks for, so the caller skips it.
444
+ * Each copy is compiled once however many inspections and chains read it.
445
+ */
446
+ const rewritten = new Map();
447
+ const copiedTree = (document, section, spec) => {
448
+ const rules = spec.replace ?? [];
449
+ if (!rules.length)
450
+ return section.tree;
451
+ const includeChildren = spec.select?.includeChildren;
452
+ const key = JSON.stringify([
453
+ document.id,
454
+ section.anchorId ?? null,
455
+ includeChildren ?? true,
456
+ rules,
457
+ ]);
458
+ if (rewritten.has(key))
459
+ return rewritten.get(key);
460
+ let copy;
461
+ try {
462
+ copy = compileReplacedSection(library, document, replacedSectionSource(document, section.anchorId, rules, includeChildren), { standalone: true });
463
+ }
464
+ catch {
465
+ copy = undefined;
466
+ }
467
+ rewritten.set(key, copy);
468
+ return copy;
469
+ };
470
+ /**
471
+ * The chain of sections that brings an embed back to itself, as the
472
+ * resolver would report it, or `undefined`. It follows the embeds inside
473
+ * each copied section — only those are expanded, after the embed's own
474
+ * replacement rules have rewritten it — and stops at the depth the
475
+ * resolver gives up at.
476
+ */
477
+ const embedCycle = (spec, from, active) => {
478
+ for (const source of spec.sources ?? []) {
479
+ let resolved;
480
+ try {
481
+ resolved = resolveDocumentReference(library, String(source), from);
482
+ }
483
+ catch {
484
+ continue; // reported as a missing source
485
+ }
486
+ const { document, anchor } = resolved;
487
+ const key = `${document.id}#${anchor ?? "*"}`;
488
+ if (active.includes(key) || active.length >= 64)
489
+ return [...active, key];
490
+ let sections;
491
+ try {
492
+ sections =
493
+ anchor || spec.select
494
+ ? collectSections(document.tree, {
495
+ ...spec.select,
496
+ ...(anchor ? { anchors: [anchor] } : {}),
497
+ })
498
+ : [{ tree: document.tree }];
499
+ }
500
+ catch {
501
+ continue; // reported as a missing section
502
+ }
503
+ for (const section of sections) {
504
+ let found;
505
+ const copy = copiedTree(document, section, spec);
506
+ if (!copy)
507
+ continue; // what it would expand cannot be told here
508
+ walkNodes(copy, (node) => {
509
+ if (found || node.type !== "code" || node.lang !== "cudoc-embed")
510
+ return;
511
+ let nested;
512
+ try {
513
+ nested = parseEmbedSpec(node.value ?? "");
514
+ }
515
+ catch {
516
+ return; // reported where it is written
517
+ }
518
+ found = embedCycle(nested, document.id, [...active, key]);
519
+ });
520
+ if (found)
521
+ return found;
522
+ }
523
+ }
524
+ return undefined;
262
525
  };
263
526
  for (const doc of library.documents) {
264
527
  // Anchors the document declares, before anything references them.
@@ -293,7 +556,9 @@ export function checkReferences(library, options = {}) {
293
556
  });
294
557
  });
295
558
  walkNodes(doc.tree, (node) => {
296
- const url = node.type === "link" || node.type === "image" ? node.url : undefined;
559
+ const captured = capturedImage(node);
560
+ const image = node.type === "image" || captured !== undefined;
561
+ const url = node.type === "link" || node.type === "image" ? node.url : captured?.url;
297
562
  if (typeof url !== "string" || !url)
298
563
  return;
299
564
  // A heading permalink is machinery the host inserted, not something an
@@ -302,7 +567,7 @@ export function checkReferences(library, options = {}) {
302
567
  if (node.data?.cudoc?.kind === "permalink")
303
568
  return;
304
569
  checkedReferences++;
305
- if (/^(?:[a-z][\w+.-]*:|\/\/)/i.test(url))
570
+ if (EXTERNAL_URL.test(url))
306
571
  return; // external
307
572
  // Another application's path on the same host: nothing here to check.
308
573
  if (isExternalPath(url, options.externalPaths))
@@ -319,9 +584,9 @@ export function checkReferences(library, options = {}) {
319
584
  });
320
585
  if (target.kind === "missing")
321
586
  report(doc, {
322
- code: node.type === "image" ? "missing-asset" : "missing-document",
587
+ code: image ? "missing-asset" : "missing-document",
323
588
  severity: "error",
324
- message: node.type === "image"
589
+ message: image
325
590
  ? `no file for image ${url}; check the collection roots and assetDirs`
326
591
  : `no document or file for ${url}`,
327
592
  reference: url,
@@ -329,7 +594,16 @@ export function checkReferences(library, options = {}) {
329
594
  });
330
595
  // Fence lines are read from the source because collection strips positions,
331
596
  // and they are what turns a YAML error's own coordinate into a file one.
332
- const fences = embedFenceLines(doc.source?.text ?? "");
597
+ const values = [];
598
+ walkNodes(doc.tree, (node) => {
599
+ if (node.type === "code" && node.lang === "cudoc-embed")
600
+ values.push(node.value ?? "");
601
+ });
602
+ const sourceText = doc.source?.text ?? "";
603
+ const fences = values.length
604
+ ? embedFences(sourceText, doc.source?.format === "mdx", values)
605
+ : [];
606
+ const sourceLines = sourceText.split(/\r?\n/);
333
607
  let blockNumber = 0;
334
608
  walkNodes(doc.tree, (node) => {
335
609
  if (node.type !== "code" || node.lang !== "cudoc-embed")
@@ -340,10 +614,8 @@ export function checkReferences(library, options = {}) {
340
614
  spec = parseEmbedSpec(node.value ?? "");
341
615
  }
342
616
  catch (error) {
343
- // The YAML parser counts from the start of the block; an author counts
344
- // from the start of the file. Add the fence so both agree.
345
- const at = error.linePos?.[0];
346
- const line = fence === undefined ? undefined : fence + (at?.line ?? 0);
617
+ const position = fence &&
618
+ blockPosition(sourceLines, fence, error.linePos?.[0]);
347
619
  report(doc, {
348
620
  code: "invalid-embed-spec",
349
621
  severity: "error",
@@ -356,51 +628,67 @@ export function checkReferences(library, options = {}) {
356
628
  .replace(/\s*at line \d+, column \d+:?\s*$/, "")
357
629
  .trim(),
358
630
  reference: `embed block ${blockNumber}`,
359
- position: line === undefined
360
- ? undefined
361
- : {
362
- start: { line, column: at?.col ?? 1 },
363
- end: { line, column: at?.col ?? 1 },
364
- },
631
+ position: position && { start: position, end: position },
365
632
  });
366
633
  return;
367
634
  }
635
+ const cycle = embedCycle(spec, doc.id, []);
636
+ if (cycle)
637
+ report(doc, {
638
+ code: "cyclic-embed",
639
+ severity: "error",
640
+ message: `embed block ${blockNumber} never finishes: ${cycle.join(" -> ")}. Each embed copies content that embeds the next, back to the first.`,
641
+ reference: `embed block ${blockNumber}`,
642
+ position: fence && {
643
+ start: { line: fence.line, column: fence.column },
644
+ end: { line: fence.line, column: fence.column },
645
+ },
646
+ });
368
647
  for (const reference of spec.sources ?? []) {
369
648
  checkedReferences++;
370
- const [pathname, anchor] = String(reference).split("#");
371
- const target = pathname
372
- ? resolveDocument(doc, pathname)
373
- : byId.get(doc.id);
374
- if (!target) {
649
+ // The resolver's own reading of the source, so an embed this passes
650
+ // is one the build can find.
651
+ let resolved;
652
+ try {
653
+ resolved = resolveDocumentReference(library, String(reference), doc.id);
654
+ }
655
+ catch (error) {
375
656
  report(doc, {
376
657
  code: "missing-embed-source",
377
658
  severity: "error",
378
- message: `embed source ${reference} does not name a collected document`,
659
+ message: error.message.replace(/^cudoc: /, ""),
379
660
  reference: String(reference),
380
661
  });
381
662
  continue;
382
663
  }
383
- const anchors = anchorsById.get(target.id) ?? [];
384
- if (anchor && !anchors.some((entry) => entry.id === anchor)) {
664
+ const { document: target, anchor } = resolved;
665
+ const sections = sectionsById.get(target.id) ?? [];
666
+ if (anchor && !sections.includes(anchor)) {
385
667
  report(doc, {
386
668
  code: "missing-embed-anchor",
387
669
  severity: "error",
388
670
  message: `${target.id} has no section #${anchor} to embed`,
389
671
  reference: String(reference),
390
- available: anchors.map((entry) => entry.id),
672
+ available: sections,
391
673
  });
392
674
  continue;
393
675
  }
394
676
  // The same selection the resolver will apply, so what is inspected is
395
- // what would actually be copied. `select.anchors` can name a section
396
- // that does not exist, which `collectSections` rejects; the build hits
397
- // the same error, so it is reported rather than swallowed.
398
- const selection = anchor ? { anchors: [anchor] } : spec.select;
677
+ // what would actually be copied: a section named in the source is
678
+ // combined with `select`, as the resolver combines them. `select`
679
+ // can name a section that does not exist, which `collectSections`
680
+ // rejects; the build hits the same error, so it is reported rather
681
+ // than swallowed.
682
+ const selection = anchor || spec.select
683
+ ? { ...spec.select, ...(anchor ? { anchors: [anchor] } : {}) }
684
+ : undefined;
399
685
  let copied;
400
686
  try {
401
687
  copied = selection
402
688
  ? collectSections(target.tree, selection)
403
689
  : [{ anchorId: undefined, tree: target.tree }];
690
+ if (!copied.length)
691
+ throw new Error("no sections matched");
404
692
  }
405
693
  catch (error) {
406
694
  report(doc, {
@@ -408,26 +696,32 @@ export function checkReferences(library, options = {}) {
408
696
  severity: "error",
409
697
  message: `${target.id}: ${error.message.replace(/^cudoc: /, "")}`,
410
698
  reference: String(reference),
411
- available: anchors.map((entry) => entry.id),
699
+ available: sections,
412
700
  });
413
701
  continue;
414
702
  }
415
703
  // Replacement runs on the original Markdown, not the tree, so the
416
- // slices are recomputed the way `transformedSection` cuts them. A rule
704
+ // text is read through `sectionText`, as the resolver reads it. A rule
417
705
  // that matches no slice changed nothing, and the embed silently shows
418
706
  // the source's own wording in a place written to expect otherwise.
419
707
  const rules = spec.replace ?? [];
708
+ // The resolver refuses rules on a section inside another block whose
709
+ // text read on its own is not that section, and the build stops there.
710
+ if (rules.length)
711
+ for (const section of copied) {
712
+ const problem = unreplaceableSection(target, section.anchorId, spec.select?.includeChildren);
713
+ if (problem)
714
+ report(doc, {
715
+ code: "unreplaceable-embed-section",
716
+ severity: "error",
717
+ message: problem,
718
+ reference: String(reference),
719
+ });
720
+ }
420
721
  if (rules.length && target.source) {
421
722
  const matched = rules.map(() => false);
422
723
  for (const section of copied) {
423
- const range = section.anchorId
424
- ? target.source.sections[section.anchorId]
425
- : undefined;
426
- const slice = range
427
- ? target.source.text.slice(range.start, spec.select?.includeChildren === false
428
- ? (range.ownEnd ?? range.end)
429
- : range.end)
430
- : target.source.text;
724
+ const slice = sectionText(target, section.anchorId, spec.select?.includeChildren);
431
725
  matchedRules(slice, rules).forEach((hit, index) => {
432
726
  if (hit)
433
727
  matched[index] = true;
@@ -444,13 +738,20 @@ export function checkReferences(library, options = {}) {
444
738
  });
445
739
  });
446
740
  }
741
+ // What the rest inspects is the copy itself, after the rules above
742
+ // have rewritten it, as the resolver builds it. A section whose copy
743
+ // cannot be built here is left out rather than guessed at.
744
+ const copies = copied.flatMap((section) => {
745
+ const tree = copiedTree(target, section, spec);
746
+ return tree ? [{ anchorId: section.anchorId, tree }] : [];
747
+ });
447
748
  // A table carries extracted text and nothing else, so whatever else
448
749
  // the body holds never travels. What can go wrong is a column that
449
750
  // finds nothing in a row: the resolver renders an empty cell, and the
450
751
  // author would only notice by reading the page.
451
752
  if (typeof spec.render === "object" && spec.render?.type === "table") {
452
753
  const columns = spec.render.columns ?? DEFAULT_TABLE_COLUMNS;
453
- for (const section of copied) {
754
+ for (const section of copies) {
454
755
  const row = buildEmbedRow(target, section.anchorId, section.tree);
455
756
  columns.forEach((column, index) => {
456
757
  // A shorthand column is best effort: `summary` of a section
@@ -492,7 +793,7 @@ export function checkReferences(library, options = {}) {
492
793
  continue;
493
794
  }
494
795
  const names = [
495
- ...new Set(copied.flatMap((section) => unportableComponents(section.tree))),
796
+ ...new Set(copies.flatMap((section) => unportableComponents(section.tree))),
496
797
  ];
497
798
  if (names.length)
498
799
  report(doc, {
@@ -521,22 +822,24 @@ export function checkReferences(library, options = {}) {
521
822
  }
522
823
  return { issues, documentCount: library.documents.length, checkedReferences };
523
824
  }
524
- // Posix path helpers, kept local so this module stays usable without importing
525
- // the whole node path surface into a browser-safe boundary by accident.
825
+ // Posix path helpers for library paths, which are always `/`-separated
826
+ // whatever the platform, so `node:path` with its native separators would be
827
+ // the wrong tool even on Node.
526
828
  const dirname = (value) => {
527
829
  const at = value.lastIndexOf("/");
528
830
  return at <= 0 ? "." : value.slice(0, at);
529
831
  };
530
832
  const posixJoin = (base, value) => base === "." ? value : `${base}/${value}`;
833
+ /** The normalized path, or `undefined` when `..` climbs out of the library. */
531
834
  const normalize = (value) => {
532
835
  const out = [];
533
836
  for (const part of value.split("/")) {
534
837
  if (!part || part === ".")
535
838
  continue;
536
- if (part === "..")
537
- out.pop();
538
- else
839
+ if (part !== "..")
539
840
  out.push(part);
841
+ else if (!out.pop())
842
+ return undefined;
540
843
  }
541
844
  return out.join("/");
542
845
  };