@cudoment/cudoc 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/README.md +3 -2
  2. package/dist/document.d.ts +25 -0
  3. package/dist/document.d.ts.map +1 -1
  4. package/dist/document.js +21 -1
  5. package/dist/document.js.map +1 -1
  6. package/dist/internal/core/query/sections.d.ts +6 -0
  7. package/dist/internal/core/query/sections.d.ts.map +1 -1
  8. package/dist/internal/core/query/sections.js +8 -2
  9. package/dist/internal/core/query/sections.js.map +1 -1
  10. package/dist/node/check.d.ts +8 -4
  11. package/dist/node/check.d.ts.map +1 -1
  12. package/dist/node/check.js +475 -65
  13. package/dist/node/check.js.map +1 -1
  14. package/dist/node/cli.js +2 -81
  15. package/dist/node/cli.js.map +1 -1
  16. package/dist/node/collect.d.ts +41 -0
  17. package/dist/node/collect.d.ts.map +1 -0
  18. package/dist/node/collect.js +351 -0
  19. package/dist/node/collect.js.map +1 -0
  20. package/dist/node/command.d.ts +23 -0
  21. package/dist/node/command.d.ts.map +1 -0
  22. package/dist/node/command.js +116 -0
  23. package/dist/node/command.js.map +1 -0
  24. package/dist/node/dataset.d.ts.map +1 -1
  25. package/dist/node/dataset.js +4 -1
  26. package/dist/node/dataset.js.map +1 -1
  27. package/dist/node/library.d.ts +9 -5
  28. package/dist/node/library.d.ts.map +1 -1
  29. package/dist/node/library.js +15 -253
  30. package/dist/node/library.js.map +1 -1
  31. package/dist/node/load-ast.d.ts.map +1 -1
  32. package/dist/node/load-ast.js +4 -1
  33. package/dist/node/load-ast.js.map +1 -1
  34. package/dist/node/local-target.d.ts +11 -0
  35. package/dist/node/local-target.d.ts.map +1 -1
  36. package/dist/node/local-target.js +54 -2
  37. package/dist/node/local-target.js.map +1 -1
  38. package/dist/node/prepare-embeds.d.ts +36 -1
  39. package/dist/node/prepare-embeds.d.ts.map +1 -1
  40. package/dist/node/prepare-embeds.js +94 -4
  41. package/dist/node/prepare-embeds.js.map +1 -1
  42. package/dist/node/references.d.ts +39 -0
  43. package/dist/node/references.d.ts.map +1 -0
  44. package/dist/node/references.js +101 -0
  45. package/dist/node/references.js.map +1 -0
  46. package/dist/node/replace.d.ts +65 -0
  47. package/dist/node/replace.d.ts.map +1 -0
  48. package/dist/node/replace.js +285 -0
  49. package/dist/node/replace.js.map +1 -0
  50. package/dist/node/report.d.ts.map +1 -1
  51. package/dist/node/report.js +5 -1
  52. package/dist/node/report.js.map +1 -1
  53. package/dist/node/resolve-embed.d.ts +99 -2
  54. package/dist/node/resolve-embed.d.ts.map +1 -1
  55. package/dist/node/resolve-embed.js +654 -116
  56. package/dist/node/resolve-embed.js.map +1 -1
  57. package/dist/node/roots.d.ts +12 -1
  58. package/dist/node/roots.d.ts.map +1 -1
  59. package/dist/node/roots.js +47 -11
  60. package/dist/node/roots.js.map +1 -1
  61. package/dist/node/storage.d.ts.map +1 -1
  62. package/dist/node/storage.js +202 -9
  63. package/dist/node/storage.js.map +1 -1
  64. package/dist/node/tree.d.ts +58 -0
  65. package/dist/node/tree.d.ts.map +1 -0
  66. package/dist/node/tree.js +151 -0
  67. package/dist/node/tree.js.map +1 -0
  68. package/dist/node/watch.d.ts +9 -4
  69. package/dist/node/watch.d.ts.map +1 -1
  70. package/dist/node/watch.js +51 -17
  71. package/dist/node/watch.js.map +1 -1
  72. package/dist/paged.d.ts +18 -0
  73. package/dist/paged.d.ts.map +1 -1
  74. package/dist/paged.js +21 -0
  75. package/dist/paged.js.map +1 -1
  76. package/dist/render.d.ts +1 -1
  77. package/dist/render.d.ts.map +1 -1
  78. package/dist/render.js +14 -0
  79. package/dist/render.js.map +1 -1
  80. package/dist/sections.d.ts.map +1 -1
  81. package/dist/sections.js +7 -3
  82. package/dist/sections.js.map +1 -1
  83. package/package.json +15 -15
  84. package/styles.css +49 -0
@@ -9,11 +9,21 @@
9
9
  * It resolves through the same code the build uses, so a reference this reports
10
10
  * as fine is one the build can resolve.
11
11
  */
12
- import { DEFAULT_TABLE_COLUMNS, buildEmbedRow, extractCell, parseEmbedSpec, } from "./resolve-embed.js";
12
+ import { unified } from "unified";
13
+ import remarkParse from "remark-parse";
14
+ import remarkGfm from "remark-gfm";
15
+ import remarkFrontmatter from "remark-frontmatter";
16
+ import remarkMdx from "remark-mdx";
17
+ import { visit } from "unist-util-visit";
18
+ import { isScalar, parseDocument } from "yaml";
19
+ import { capturedImage } from "../document.js";
20
+ import { DEFAULT_TABLE_COLUMNS, DEFAULT_TREE_COLUMNS, buildEmbedRow, extractCell, namesTreeNode, parseEmbedSpec, resolveDocumentReference, resolveTree, resolveTreeSource, } from "./resolve-embed.js";
21
+ import { compileReplacedSection, replacedSectionSource, sectionText, unreplaceableSection, } from "./replace.js";
13
22
  import { collectSections } from "../sections.js";
14
23
  import { isExternalPath, resolveLocalTarget, } from "./local-target.js";
15
24
  import { resolveRoots } from "./roots.js";
16
- /** A heading id that a slugger disambiguated, such as `overview-1`. */
25
+ import { EXTERNAL_URL, decodeComponent, idsInNode } from "./references.js";
26
+ /** The shape of a heading id a slugger disambiguated, such as `overview-1`. */
17
27
  const SUFFIXED = /-\d+$/;
18
28
  const walkNodes = (node, visit) => {
19
29
  visit(node);
@@ -24,19 +34,46 @@ const walkNodes = (node, visit) => {
24
34
  *
25
35
  * Both halves matter. The ids answer whether a link resolves; the origin
26
36
  * answers whether it will keep resolving, because a generated id depends on how
27
- * many same-named headings precede it.
37
+ * many same-named headings precede it. Besides headings, an id an element in
38
+ * raw HTML declares is an anchor too, and one the author wrote.
28
39
  */
29
40
  export function collectAnchors(tree) {
30
41
  const anchors = [];
31
42
  walkNodes(tree, (node) => {
32
- if (node.type !== "heading")
43
+ if (node.type !== "heading") {
44
+ for (const id of idsInNode(node))
45
+ if (id)
46
+ anchors.push({ id, explicit: true });
33
47
  return;
48
+ }
34
49
  const id = node.data?.hProperties?.id;
35
50
  if (typeof id === "string" && id)
36
51
  anchors.push({ id, explicit: node.data?.cudoc?.explicitId === true });
37
52
  });
38
53
  return anchors;
39
54
  }
55
+ /**
56
+ * The ids a section can be embedded from: headings only, since an id raw HTML
57
+ * declares starts no section.
58
+ */
59
+ const sectionIds = (anchors, tree) => {
60
+ const headings = new Set();
61
+ walkNodes(tree, (node) => {
62
+ const id = node.type === "heading" ? node.data?.hProperties?.id : undefined;
63
+ if (typeof id === "string" && id)
64
+ headings.add(id);
65
+ });
66
+ return anchors.map((entry) => entry.id).filter((id) => headings.has(id));
67
+ };
68
+ /**
69
+ * The anchor a fragment names, if the document has it. Hosts percent-encode a
70
+ * fragment that is not ASCII, so `#개요` may arrive as `#%EA%B0%9C%EC%9A%94`;
71
+ * either spelling names the same heading.
72
+ */
73
+ const findAnchor = (anchors, fragment) => {
74
+ const decoded = decodeComponent(fragment);
75
+ return anchors.find((entry) => entry.id === fragment || entry.id === decoded);
76
+ };
40
77
  /**
41
78
  * Where a reference sits in the original Markdown.
42
79
  *
@@ -120,6 +157,9 @@ const unportableComponents = (tree) => {
120
157
  return;
121
158
  if (!node.type.startsWith("mdx") && !node.type.endsWith("Directive"))
122
159
  return;
160
+ // A host's component for a Markdown image renders as that image.
161
+ if (capturedImage(node))
162
+ return;
123
163
  names.add(node.name ? `<${node.name}>` : node.type);
124
164
  });
125
165
  return [...names];
@@ -133,14 +173,120 @@ const headingText = (node) => {
133
173
  });
134
174
  return text;
135
175
  };
136
- /** The line each `cudoc-embed` fence opens on, in source order. */
137
- const embedFenceLines = (text) => {
138
- const lines = [];
139
- text.split("\n").forEach((line, index) => {
140
- if (/^\s*(?:`{3,}|~{3,})cudoc-embed\s*$/.test(line))
141
- lines.push(index + 1);
176
+ /**
177
+ * The `cudoc-embed` fences of a source, in order, as Markdown parses them: a
178
+ * fence shown inside a longer fence or an indented code block is example
179
+ * text, and one inside a quote or a list item is a block like any other. The
180
+ * source is parsed as Markdown with GFM and front matter, as MDX for an `.mdx`
181
+ * file.
182
+ */
183
+ const parsedFences = (text, mdx) => {
184
+ try {
185
+ const processor = unified()
186
+ .use(remarkParse)
187
+ .use(remarkGfm)
188
+ .use(remarkFrontmatter);
189
+ if (mdx)
190
+ processor.use(remarkMdx);
191
+ const fences = [];
192
+ visit(processor.parse(text), "code", (node) => {
193
+ if (node.lang === "cudoc-embed" && node.position)
194
+ fences.push({
195
+ line: node.position.start.line,
196
+ column: node.position.start.column,
197
+ value: node.value,
198
+ });
199
+ });
200
+ return fences;
201
+ }
202
+ catch {
203
+ // A host's own syntax may not parse here; the line scan below answers.
204
+ return undefined;
205
+ }
206
+ };
207
+ /**
208
+ * The same fences found line by line: a fence is closed the way Markdown
209
+ * closes it, by the same character, at least as long, and nothing after it,
210
+ * and its text is the lines between, less the opening fence's indentation.
211
+ */
212
+ const scannedFences = (text) => {
213
+ const found = [];
214
+ let open;
215
+ text.split(/\r?\n/).forEach((line, index) => {
216
+ const match = line.match(/^(\s*)(`{3,}|~{3,})(.*)$/);
217
+ if (open) {
218
+ const [, , marker = "", rest = ""] = match ?? [];
219
+ if (marker[0] === open.char &&
220
+ marker.length >= open.length &&
221
+ !rest.trim())
222
+ open = undefined;
223
+ else
224
+ open.content?.push(line.replace(open.indent, ""));
225
+ return;
226
+ }
227
+ if (!match)
228
+ return;
229
+ const [, indent = "", marker = "", rest = ""] = match;
230
+ open = {
231
+ char: marker[0],
232
+ length: marker.length,
233
+ indent: new RegExp(`^ {0,${indent.length}}`),
234
+ };
235
+ if (rest.trim().split(/\s+/)[0] !== "cudoc-embed")
236
+ return;
237
+ open.content = [];
238
+ found.push({
239
+ line: index + 1,
240
+ column: indent.length + 1,
241
+ content: open.content,
242
+ });
142
243
  });
143
- return lines;
244
+ return found.map(({ content, ...fence }) => ({
245
+ ...fence,
246
+ value: content.join("\n"),
247
+ }));
248
+ };
249
+ /** Block text as hosts may differ in keeping it: without `\r` or trailing whitespace. */
250
+ const comparable = (value) => value
251
+ .replace(/\r/g, "")
252
+ .split("\n")
253
+ .map((line) => line.trimEnd())
254
+ .join("\n")
255
+ .trimEnd();
256
+ /**
257
+ * The fence of each of a document's embed blocks, given their text in order,
258
+ * or nothing when neither reading of the source finds those blocks in that
259
+ * order: a host whose syntax reads a block differently from Markdown would
260
+ * otherwise put a block's error on another block's fence.
261
+ */
262
+ const embedFences = (text, mdx, values) => {
263
+ const wanted = values.map(comparable);
264
+ const found = (fences) => fences !== undefined &&
265
+ fences.length === wanted.length &&
266
+ fences.every((fence, index) => comparable(fence.value) === wanted[index]);
267
+ const parsed = parsedFences(text, mdx);
268
+ if (found(parsed))
269
+ return parsed;
270
+ const scanned = scannedFences(text);
271
+ return found(scanned) ? scanned : [];
272
+ };
273
+ /**
274
+ * Where an error in an embed block sits in the file. The YAML parser counts
275
+ * from the block's first line and column: the file's line adds the fence's,
276
+ * and the file's column adds whatever stands before the block's text on that
277
+ * line, indentation, a quote's `>` or a list item's offset alike. An error
278
+ * without a coordinate sits on the fence.
279
+ */
280
+ const blockPosition = (lines, fence, at) => {
281
+ if (!at)
282
+ return { line: fence.line, column: fence.column };
283
+ const line = fence.line + at.line;
284
+ const source = (lines[line - 1] ?? "").trimEnd();
285
+ const text = (comparable(fence.value).split("\n")[at.line - 1] ?? "").trimEnd();
286
+ const before = source.endsWith(text)
287
+ ? source.length - text.length
288
+ : fence.column - 1;
289
+ return { line, column: before + at.col };
144
290
  };
145
291
  /**
146
292
  * Whether a rule finds anything, asked the way the resolver asks it.
@@ -187,6 +333,10 @@ export function checkReferences(library, options = {}) {
187
333
  // spliced nodes with its own component mapping, a Markdown host cannot.
188
334
  const mdxHost = ["next", "docusaurus", "nextra"].includes(library.options.host ?? "");
189
335
  const anchorsById = new Map(library.documents.map((doc) => [doc.id, collectAnchors(doc.tree)]));
336
+ const sectionsById = new Map(library.documents.map((doc) => [
337
+ doc.id,
338
+ sectionIds(anchorsById.get(doc.id), doc.tree),
339
+ ]));
190
340
  const byId = new Map(library.documents.map((doc) => [doc.id, doc]));
191
341
  const report = (doc,
192
342
  /** A `position` given here wins; otherwise it is recovered from the source. */
@@ -205,7 +355,9 @@ export function checkReferences(library, options = {}) {
205
355
  };
206
356
  /** Resolves `other.md#anchor` or a bare `#anchor` against the library. */
207
357
  const checkDocumentLink = (doc, url) => {
208
- const [pathname, anchor] = url.split("#");
358
+ const [address = "", anchor] = url.split("#");
359
+ // A query names no other document: `guide/?tab=1` is still `guide/`.
360
+ const pathname = address.replace(/\?.*$/, "");
209
361
  let target = doc;
210
362
  if (pathname) {
211
363
  const found = resolveDocument(doc, pathname);
@@ -216,7 +368,7 @@ export function checkReferences(library, options = {}) {
216
368
  if (!anchor)
217
369
  return true;
218
370
  const anchors = anchorsById.get(target.id) ?? [];
219
- const match = anchors.find((entry) => entry.id === anchor);
371
+ const match = findAnchor(anchors, anchor);
220
372
  if (!match) {
221
373
  report(doc, {
222
374
  code: "missing-anchor",
@@ -227,7 +379,12 @@ export function checkReferences(library, options = {}) {
227
379
  });
228
380
  return true;
229
381
  }
230
- if (!match.explicit && SUFFIXED.test(match.id))
382
+ // Only a suffix a slugger added is fragile: `## Version 2` is `version-2`
383
+ // because of its own text, and stays so. A repeat shows as the bare id
384
+ // being there too.
385
+ if (!match.explicit &&
386
+ SUFFIXED.test(match.id) &&
387
+ anchors.some((entry) => entry.id === match.id.replace(SUFFIXED, "")))
231
388
  report(doc, {
232
389
  code: "unstable-anchor-link",
233
390
  severity: "warning",
@@ -249,16 +406,225 @@ export function checkReferences(library, options = {}) {
249
406
  * removed, the way the exporter looks a route up.
250
407
  */
251
408
  const resolveDocument = (doc, pathname) => {
252
- if (/^(?:[a-z][\w+.-]*:|\/\/)/i.test(pathname))
409
+ if (EXTERNAL_URL.test(pathname))
253
410
  return undefined;
254
- const decoded = decodeURIComponent(pathname);
255
- const toId = (value) => normalize(value).replace(/\.(?:mdx?|html?)$/i, "");
411
+ const decoded = decodeComponent(pathname);
412
+ // A directory, `./` or `guide/` as VitePress writes a link to an
413
+ // `index.md`, names that directory's index document; spelled with its
414
+ // trailing slash, it does even beside a `guide.md`.
415
+ const lookup = (value) => {
416
+ const normalized = normalize(value);
417
+ if (normalized === undefined)
418
+ return undefined;
419
+ const id = normalized.replace(/\.(?:mdx?|html?)$/i, "");
420
+ const index = byId.get(normalized ? `${normalized}/index` : "index");
421
+ if (value === "" || /(?:^|\/)(?:\.\.?)?$/.test(value))
422
+ return index ?? byId.get(id);
423
+ return byId.get(id) ?? index;
424
+ };
256
425
  if (!decoded.startsWith("/"))
257
- return byId.get(toId(posixJoin(dirname(doc.sourcePath), decoded)));
258
- const direct = byId.get(toId(decoded.slice(1)));
426
+ return lookup(posixJoin(dirname(doc.sourcePath), decoded));
427
+ const direct = lookup(decoded.slice(1));
259
428
  if (direct || !options.withoutBase)
260
429
  return direct;
261
- return byId.get(toId(options.withoutBase(decoded).replace(/^\//, "")));
430
+ return lookup(options.withoutBase(decoded).replace(/^\//, ""));
431
+ };
432
+ /**
433
+ * What an embed copies from one selected section: the section as
434
+ * collected, or, under replacement rules, its rewritten Markdown compiled
435
+ * again, which is the copy the resolver builds and expands. Without a
436
+ * synchronous host compiler, which `cudoc check` never has, the standalone
437
+ * compiler reads the rewritten Markdown with the library's options.
438
+ *
439
+ * `undefined` when the copy cannot be built here: the resolver refuses the
440
+ * rules, the snapshot or a repeated section id, which fails the build on
441
+ * its own, or the standalone compiler cannot read what only the host's
442
+ * parser accepts, such as Docusaurus's `{#id}` or an HTML comment in MDX.
443
+ * The section as collected is not a stand-in, because the rules may have
444
+ * changed exactly what an inspection looks for, so the caller skips it.
445
+ * Each copy is compiled once however many inspections and chains read it.
446
+ */
447
+ const rewritten = new Map();
448
+ const copiedTree = (document, section, spec) => {
449
+ const rules = spec.replace ?? [];
450
+ if (!rules.length)
451
+ return section.tree;
452
+ const includeChildren = spec.select?.includeChildren;
453
+ const key = JSON.stringify([
454
+ document.id,
455
+ section.anchorId ?? null,
456
+ includeChildren ?? true,
457
+ rules,
458
+ ]);
459
+ if (rewritten.has(key))
460
+ return rewritten.get(key);
461
+ let copy;
462
+ try {
463
+ copy = compileReplacedSection(library, document, replacedSectionSource(document, section.anchorId, rules, includeChildren), { standalone: true });
464
+ }
465
+ catch {
466
+ copy = undefined;
467
+ }
468
+ rewritten.set(key, copy);
469
+ return copy;
470
+ };
471
+ /**
472
+ * The chain of sections that brings an embed back to itself, as the
473
+ * resolver would report it, or `undefined`. It follows the embeds inside
474
+ * each copied section — only those are expanded, after the embed's own
475
+ * replacement rules have rewritten it — and stops at the depth the
476
+ * resolver gives up at.
477
+ */
478
+ const embedCycle = (spec, from, active) => {
479
+ // A tree copies no section, so nothing it names is expanded into it.
480
+ if (typeof spec.render === "object" && spec.render.type === "tree")
481
+ return undefined;
482
+ for (const source of spec.sources ?? []) {
483
+ let resolved;
484
+ try {
485
+ resolved = resolveDocumentReference(library, String(source), from);
486
+ }
487
+ catch {
488
+ continue; // reported as a missing source
489
+ }
490
+ const { document, anchor } = resolved;
491
+ const key = `${document.id}#${anchor ?? "*"}`;
492
+ if (active.includes(key) || active.length >= 64)
493
+ return [...active, key];
494
+ let sections;
495
+ try {
496
+ sections =
497
+ anchor || spec.select
498
+ ? collectSections(document.tree, {
499
+ ...spec.select,
500
+ ...(anchor ? { anchors: [anchor] } : {}),
501
+ })
502
+ : [{ tree: document.tree }];
503
+ }
504
+ catch {
505
+ continue; // reported as a missing section
506
+ }
507
+ for (const section of sections) {
508
+ let found;
509
+ const copy = copiedTree(document, section, spec);
510
+ if (!copy)
511
+ continue; // what it would expand cannot be told here
512
+ walkNodes(copy, (node) => {
513
+ if (found || node.type !== "code" || node.lang !== "cudoc-embed")
514
+ return;
515
+ let nested;
516
+ try {
517
+ nested = parseEmbedSpec(node.value ?? "");
518
+ }
519
+ catch {
520
+ return; // reported where it is written
521
+ }
522
+ found = embedCycle(nested, document.id, [...active, key]);
523
+ });
524
+ if (found)
525
+ return found;
526
+ }
527
+ }
528
+ return undefined;
529
+ };
530
+ /**
531
+ * A tree reads titles and summaries and copies nothing else, so neither a
532
+ * cycle nor a component can come of it. What can go wrong is a source that
533
+ * names nothing, a column naming an extractor the configuration does not
534
+ * register, and an `order` entry that names no line of the first level,
535
+ * which the build passes over in silence.
536
+ */
537
+ const checkTree = (doc, sources, render, fence, sourceLines) => {
538
+ let resolvable = true;
539
+ for (const reference of sources) {
540
+ checkedReferences++;
541
+ let source;
542
+ try {
543
+ source = resolveTreeSource(library, String(reference), doc.id, doc.id);
544
+ }
545
+ catch (error) {
546
+ resolvable = false;
547
+ report(doc, {
548
+ code: "missing-embed-source",
549
+ severity: "error",
550
+ message: error.message.replace(/^cudoc: /, ""),
551
+ reference: String(reference),
552
+ });
553
+ continue;
554
+ }
555
+ if ("document" in source && source.anchor !== undefined) {
556
+ const sections = sectionsById.get(source.document.id) ?? [];
557
+ if (!sections.includes(source.anchor)) {
558
+ resolvable = false;
559
+ report(doc, {
560
+ code: "missing-embed-anchor",
561
+ severity: "error",
562
+ message: `${source.document.id} has no section #${source.anchor} to embed`,
563
+ reference: String(reference),
564
+ available: sections,
565
+ });
566
+ }
567
+ }
568
+ }
569
+ // A value's own place in the block, as the YAML parser read it, rather
570
+ // than the first place its text occurs in the document.
571
+ const text = fence ? comparable(fence.value) : "";
572
+ let parsed;
573
+ const place = (path) => {
574
+ if (!fence)
575
+ return undefined;
576
+ parsed ??= parseDocument(text);
577
+ const node = parsed.getIn(path, true);
578
+ const range = isScalar(node) ? node.range : undefined;
579
+ if (!range) {
580
+ const position = blockPosition(sourceLines, fence);
581
+ return { start: position, end: position };
582
+ }
583
+ const before = text.slice(0, range[0]);
584
+ const start = blockPosition(sourceLines, fence, {
585
+ line: before.split("\n").length,
586
+ col: range[0] - (before.lastIndexOf("\n") + 1) + 1,
587
+ });
588
+ return {
589
+ start,
590
+ end: { line: start.line, column: start.column + range[1] - range[0] },
591
+ };
592
+ };
593
+ (render.columns ?? DEFAULT_TREE_COLUMNS).forEach((column, index) => {
594
+ if (typeof column !== "object" ||
595
+ typeof column.value !== "object" ||
596
+ !("extractor" in column.value) ||
597
+ library.extractors?.[column.value.extractor])
598
+ return;
599
+ report(doc, {
600
+ code: "invalid-embed-spec",
601
+ severity: "error",
602
+ message: `extractor "${column.value.extractor}" is not registered; add it to extractors in the collection options`,
603
+ reference: column.value.extractor,
604
+ position: place(["render", "columns", index, "value", "extractor"]),
605
+ });
606
+ });
607
+ if (!resolvable || !render.order?.length)
608
+ return;
609
+ let first;
610
+ try {
611
+ first = resolveTree(library, { sources, render: { type: "tree", depth: 1, columns: ["title"] } }, { documentId: doc.id });
612
+ }
613
+ catch {
614
+ return; // what fails here fails the build, reported as it stands
615
+ }
616
+ render.order.forEach((entry, index) => {
617
+ if (entry === "..." || first.some((node) => namesTreeNode(entry, node)))
618
+ return;
619
+ report(doc, {
620
+ code: "unmatched-tree-order",
621
+ severity: "warning",
622
+ message: `order names "${entry}", which is not on the tree's first level, so it moves nothing. An entry matches a document's file name or a line's title.`,
623
+ reference: entry,
624
+ position: place(["render", "order", index]),
625
+ available: first.map((node) => node.title),
626
+ });
627
+ });
262
628
  };
263
629
  for (const doc of library.documents) {
264
630
  // Anchors the document declares, before anything references them.
@@ -293,7 +659,9 @@ export function checkReferences(library, options = {}) {
293
659
  });
294
660
  });
295
661
  walkNodes(doc.tree, (node) => {
296
- const url = node.type === "link" || node.type === "image" ? node.url : undefined;
662
+ const captured = capturedImage(node);
663
+ const image = node.type === "image" || captured !== undefined;
664
+ const url = node.type === "link" || node.type === "image" ? node.url : captured?.url;
297
665
  if (typeof url !== "string" || !url)
298
666
  return;
299
667
  // A heading permalink is machinery the host inserted, not something an
@@ -302,7 +670,7 @@ export function checkReferences(library, options = {}) {
302
670
  if (node.data?.cudoc?.kind === "permalink")
303
671
  return;
304
672
  checkedReferences++;
305
- if (/^(?:[a-z][\w+.-]*:|\/\/)/i.test(url))
673
+ if (EXTERNAL_URL.test(url))
306
674
  return; // external
307
675
  // Another application's path on the same host: nothing here to check.
308
676
  if (isExternalPath(url, options.externalPaths))
@@ -319,9 +687,9 @@ export function checkReferences(library, options = {}) {
319
687
  });
320
688
  if (target.kind === "missing")
321
689
  report(doc, {
322
- code: node.type === "image" ? "missing-asset" : "missing-document",
690
+ code: image ? "missing-asset" : "missing-document",
323
691
  severity: "error",
324
- message: node.type === "image"
692
+ message: image
325
693
  ? `no file for image ${url}; check the collection roots and assetDirs`
326
694
  : `no document or file for ${url}`,
327
695
  reference: url,
@@ -329,7 +697,16 @@ export function checkReferences(library, options = {}) {
329
697
  });
330
698
  // Fence lines are read from the source because collection strips positions,
331
699
  // and they are what turns a YAML error's own coordinate into a file one.
332
- const fences = embedFenceLines(doc.source?.text ?? "");
700
+ const values = [];
701
+ walkNodes(doc.tree, (node) => {
702
+ if (node.type === "code" && node.lang === "cudoc-embed")
703
+ values.push(node.value ?? "");
704
+ });
705
+ const sourceText = doc.source?.text ?? "";
706
+ const fences = values.length
707
+ ? embedFences(sourceText, doc.source?.format === "mdx", values)
708
+ : [];
709
+ const sourceLines = sourceText.split(/\r?\n/);
333
710
  let blockNumber = 0;
334
711
  walkNodes(doc.tree, (node) => {
335
712
  if (node.type !== "code" || node.lang !== "cudoc-embed")
@@ -340,10 +717,8 @@ export function checkReferences(library, options = {}) {
340
717
  spec = parseEmbedSpec(node.value ?? "");
341
718
  }
342
719
  catch (error) {
343
- // The YAML parser counts from the start of the block; an author counts
344
- // from the start of the file. Add the fence so both agree.
345
- const at = error.linePos?.[0];
346
- const line = fence === undefined ? undefined : fence + (at?.line ?? 0);
720
+ const position = fence &&
721
+ blockPosition(sourceLines, fence, error.linePos?.[0]);
347
722
  report(doc, {
348
723
  code: "invalid-embed-spec",
349
724
  severity: "error",
@@ -356,51 +731,71 @@ export function checkReferences(library, options = {}) {
356
731
  .replace(/\s*at line \d+, column \d+:?\s*$/, "")
357
732
  .trim(),
358
733
  reference: `embed block ${blockNumber}`,
359
- position: line === undefined
360
- ? undefined
361
- : {
362
- start: { line, column: at?.col ?? 1 },
363
- end: { line, column: at?.col ?? 1 },
364
- },
734
+ position: position && { start: position, end: position },
365
735
  });
366
736
  return;
367
737
  }
738
+ if (typeof spec.render === "object" && spec.render.type === "tree") {
739
+ checkTree(doc, spec.sources, spec.render, fence, sourceLines);
740
+ return;
741
+ }
742
+ const cycle = embedCycle(spec, doc.id, []);
743
+ if (cycle)
744
+ report(doc, {
745
+ code: "cyclic-embed",
746
+ severity: "error",
747
+ message: `embed block ${blockNumber} never finishes: ${cycle.join(" -> ")}. Each embed copies content that embeds the next, back to the first.`,
748
+ reference: `embed block ${blockNumber}`,
749
+ position: fence && {
750
+ start: { line: fence.line, column: fence.column },
751
+ end: { line: fence.line, column: fence.column },
752
+ },
753
+ });
368
754
  for (const reference of spec.sources ?? []) {
369
755
  checkedReferences++;
370
- const [pathname, anchor] = String(reference).split("#");
371
- const target = pathname
372
- ? resolveDocument(doc, pathname)
373
- : byId.get(doc.id);
374
- if (!target) {
756
+ // The resolver's own reading of the source, so an embed this passes
757
+ // is one the build can find.
758
+ let resolved;
759
+ try {
760
+ resolved = resolveDocumentReference(library, String(reference), doc.id);
761
+ }
762
+ catch (error) {
375
763
  report(doc, {
376
764
  code: "missing-embed-source",
377
765
  severity: "error",
378
- message: `embed source ${reference} does not name a collected document`,
766
+ message: error.message.replace(/^cudoc: /, ""),
379
767
  reference: String(reference),
380
768
  });
381
769
  continue;
382
770
  }
383
- const anchors = anchorsById.get(target.id) ?? [];
384
- if (anchor && !anchors.some((entry) => entry.id === anchor)) {
771
+ const { document: target, anchor } = resolved;
772
+ const sections = sectionsById.get(target.id) ?? [];
773
+ if (anchor && !sections.includes(anchor)) {
385
774
  report(doc, {
386
775
  code: "missing-embed-anchor",
387
776
  severity: "error",
388
777
  message: `${target.id} has no section #${anchor} to embed`,
389
778
  reference: String(reference),
390
- available: anchors.map((entry) => entry.id),
779
+ available: sections,
391
780
  });
392
781
  continue;
393
782
  }
394
783
  // The same selection the resolver will apply, so what is inspected is
395
- // what would actually be copied. `select.anchors` can name a section
396
- // that does not exist, which `collectSections` rejects; the build hits
397
- // the same error, so it is reported rather than swallowed.
398
- const selection = anchor ? { anchors: [anchor] } : spec.select;
784
+ // what would actually be copied: a section named in the source is
785
+ // combined with `select`, as the resolver combines them. `select`
786
+ // can name a section that does not exist, which `collectSections`
787
+ // rejects; the build hits the same error, so it is reported rather
788
+ // than swallowed.
789
+ const selection = anchor || spec.select
790
+ ? { ...spec.select, ...(anchor ? { anchors: [anchor] } : {}) }
791
+ : undefined;
399
792
  let copied;
400
793
  try {
401
794
  copied = selection
402
795
  ? collectSections(target.tree, selection)
403
796
  : [{ anchorId: undefined, tree: target.tree }];
797
+ if (!copied.length)
798
+ throw new Error("no sections matched");
404
799
  }
405
800
  catch (error) {
406
801
  report(doc, {
@@ -408,26 +803,32 @@ export function checkReferences(library, options = {}) {
408
803
  severity: "error",
409
804
  message: `${target.id}: ${error.message.replace(/^cudoc: /, "")}`,
410
805
  reference: String(reference),
411
- available: anchors.map((entry) => entry.id),
806
+ available: sections,
412
807
  });
413
808
  continue;
414
809
  }
415
810
  // Replacement runs on the original Markdown, not the tree, so the
416
- // slices are recomputed the way `transformedSection` cuts them. A rule
811
+ // text is read through `sectionText`, as the resolver reads it. A rule
417
812
  // that matches no slice changed nothing, and the embed silently shows
418
813
  // the source's own wording in a place written to expect otherwise.
419
814
  const rules = spec.replace ?? [];
815
+ // The resolver refuses rules on a section inside another block whose
816
+ // text read on its own is not that section, and the build stops there.
817
+ if (rules.length)
818
+ for (const section of copied) {
819
+ const problem = unreplaceableSection(target, section.anchorId, spec.select?.includeChildren);
820
+ if (problem)
821
+ report(doc, {
822
+ code: "unreplaceable-embed-section",
823
+ severity: "error",
824
+ message: problem,
825
+ reference: String(reference),
826
+ });
827
+ }
420
828
  if (rules.length && target.source) {
421
829
  const matched = rules.map(() => false);
422
830
  for (const section of copied) {
423
- const range = section.anchorId
424
- ? target.source.sections[section.anchorId]
425
- : undefined;
426
- const slice = range
427
- ? target.source.text.slice(range.start, spec.select?.includeChildren === false
428
- ? (range.ownEnd ?? range.end)
429
- : range.end)
430
- : target.source.text;
831
+ const slice = sectionText(target, section.anchorId, spec.select?.includeChildren);
431
832
  matchedRules(slice, rules).forEach((hit, index) => {
432
833
  if (hit)
433
834
  matched[index] = true;
@@ -444,13 +845,20 @@ export function checkReferences(library, options = {}) {
444
845
  });
445
846
  });
446
847
  }
848
+ // What the rest inspects is the copy itself, after the rules above
849
+ // have rewritten it, as the resolver builds it. A section whose copy
850
+ // cannot be built here is left out rather than guessed at.
851
+ const copies = copied.flatMap((section) => {
852
+ const tree = copiedTree(target, section, spec);
853
+ return tree ? [{ anchorId: section.anchorId, tree }] : [];
854
+ });
447
855
  // A table carries extracted text and nothing else, so whatever else
448
856
  // the body holds never travels. What can go wrong is a column that
449
857
  // finds nothing in a row: the resolver renders an empty cell, and the
450
858
  // author would only notice by reading the page.
451
859
  if (typeof spec.render === "object" && spec.render?.type === "table") {
452
860
  const columns = spec.render.columns ?? DEFAULT_TABLE_COLUMNS;
453
- for (const section of copied) {
861
+ for (const section of copies) {
454
862
  const row = buildEmbedRow(target, section.anchorId, section.tree);
455
863
  columns.forEach((column, index) => {
456
864
  // A shorthand column is best effort: `summary` of a section
@@ -492,7 +900,7 @@ export function checkReferences(library, options = {}) {
492
900
  continue;
493
901
  }
494
902
  const names = [
495
- ...new Set(copied.flatMap((section) => unportableComponents(section.tree))),
903
+ ...new Set(copies.flatMap((section) => unportableComponents(section.tree))),
496
904
  ];
497
905
  if (names.length)
498
906
  report(doc, {
@@ -521,22 +929,24 @@ export function checkReferences(library, options = {}) {
521
929
  }
522
930
  return { issues, documentCount: library.documents.length, checkedReferences };
523
931
  }
524
- // Posix path helpers, kept local so this module stays usable without importing
525
- // the whole node path surface into a browser-safe boundary by accident.
932
+ // Posix path helpers for library paths, which are always `/`-separated
933
+ // whatever the platform, so `node:path` with its native separators would be
934
+ // the wrong tool even on Node.
526
935
  const dirname = (value) => {
527
936
  const at = value.lastIndexOf("/");
528
937
  return at <= 0 ? "." : value.slice(0, at);
529
938
  };
530
939
  const posixJoin = (base, value) => base === "." ? value : `${base}/${value}`;
940
+ /** The normalized path, or `undefined` when `..` climbs out of the library. */
531
941
  const normalize = (value) => {
532
942
  const out = [];
533
943
  for (const part of value.split("/")) {
534
944
  if (!part || part === ".")
535
945
  continue;
536
- if (part === "..")
537
- out.pop();
538
- else
946
+ if (part !== "..")
539
947
  out.push(part);
948
+ else if (!out.pop())
949
+ return undefined;
540
950
  }
541
951
  return out.join("/");
542
952
  };