@bevel-software/platform-shared 0.26.0 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,757 @@
1
+ /**
2
+ * The markdown link grammar a move rewrites with: where the links in a page
3
+ * are, what each one points at, and how to point it somewhere else without
4
+ * changing anything else about it.
5
+ *
6
+ * Pure and dependency-free, so the backend's move and the web app can share
7
+ * it. The resolver half is the web app's (`resolveKbHref` in the frontend's
8
+ * `routing/kb-routes.ts`) restated here; a parity test there holds the two to
9
+ * one answer on a shared fixture set.
10
+ *
11
+ * What it deliberately is NOT: a markdown parser. It finds link
12
+ * DESTINATIONS — inline links, images, reference definitions, and the inline
13
+ * links in a node's frontmatter (the `nodeType` link) — as character spans, so
14
+ * a rewrite splices the new destination into the old one's place and every
15
+ * other byte of the file stays as it was. Code — fenced blocks, indented
16
+ * blocks and inline code — is skipped: a path written there is an example,
17
+ * not a link.
18
+ */
19
+
20
+ import { extractFrontmatter } from './frontmatter.js';
21
+
22
+ /**
23
+ * An id-link destination: a node's frontmatter id (`project-hexis`), optionally
24
+ * with a heading anchor. The web app's `NODE_ID_LINK_RE`, restated: an id-link
25
+ * points at a node wherever it lives, so a move never changes one.
26
+ */
27
+ export const MD_ID_LINK_RE = /^[a-z0-9][a-z0-9_-]*(#[^/]+)?$/;
28
+
29
+ /** The app route a copied link names a file by: `/workspace/<branch>/<path>`. */
30
+ const WORKSPACE_ROUTE_PREFIX = '/workspace/';
31
+
32
+ /** One link destination found in a page. */
33
+ export interface MdLinkSpan {
34
+ /** `link` and `image` are inline (`[x](d)`, `![x](d)`); `definition` is `[x]: d`. */
35
+ kind: 'link' | 'image' | 'definition';
36
+ /** Offsets of the destination as written, angle brackets excluded. */
37
+ start: number;
38
+ end: number;
39
+ /** The destination as written (`text.slice(start, end)`). */
40
+ destination: string;
41
+ /** Written as `<dest>`. */
42
+ angle: boolean;
43
+ /** Inside the `---` frontmatter block. */
44
+ inFrontmatter: boolean;
45
+ }
46
+
47
+ /**
48
+ * Every link destination in a markdown page, in document order. Code is
49
+ * skipped: fenced blocks (``` and ~~~, in lists and quotes too) and inline code
50
+ * spans. A footnote (`[^1]: …`) is not a reference definition.
51
+ */
52
+ export function scanMarkdownLinks(text: string): MdLinkSpan[] {
53
+ const out: MdLinkSpan[] = [];
54
+ let bodyStart = 0;
55
+ const fm = extractFrontmatter(text);
56
+ if (fm) {
57
+ bodyStart = text.length - fm.body.length;
58
+ const fmStart = text.indexOf('\n') + 1;
59
+ const fmEnd = fmStart + fm.frontmatter.length;
60
+ // Line by line: a frontmatter value is one line, and a link never spans
61
+ // two. Only a value that IS one link counts — what the web app's
62
+ // frontmatter panel renders as a link (`nodeType` the usual one); a link
63
+ // written inside a prose value shows as text there, so it is left alone.
64
+ let lineStart = fmStart;
65
+ while (lineStart < fmEnd) {
66
+ const nl = text.indexOf('\n', lineStart);
67
+ const lineEnd = nl === -1 || nl > fmEnd ? fmEnd : nl;
68
+ const value = FRONTMATTER_LINK_VALUE_RE.exec(text.slice(lineStart, lineEnd).replace(/\r$/, ''));
69
+ // Only the link itself: a trailing YAML comment is not part of the value.
70
+ if (value) {
71
+ const linkStart = lineStart + value[1].length;
72
+ scanInline(text, linkStart, linkStart + value[3].length, true, out);
73
+ }
74
+ lineStart = lineEnd + 1;
75
+ }
76
+ }
77
+ scanBody(text, bodyStart, {
78
+ prose: (from, to) => scanInline(text, from, to, false, out),
79
+ definition: (span) => out.push(span),
80
+ });
81
+ return out;
82
+ }
83
+
84
+ /**
85
+ * `text` with everything that cannot start a live HTML tag blanked to spaces
86
+ * — fenced and indented code blocks, inline code spans, the frontmatter, and
87
+ * an escaped `\<`, which is a literal `<` — every kept character at its
88
+ * offset. It locates tag spans and nothing else: inside a raw tag a backslash
89
+ * is not an escape, so every other escape is kept as the two characters it
90
+ * is, and the attribute values are read from the original text.
91
+ */
92
+ function maskMarkdownCode(text: string): string {
93
+ const chars: string[] = Array.from({ length: text.length }, (_, i) => (text[i] === '\n' ? '\n' : ' '));
94
+ const keep = (from: number, to: number) => {
95
+ let i = from;
96
+ while (i < to) {
97
+ const c = text[i];
98
+ if (c === '\\') {
99
+ if (i + 1 < to && text[i + 1] === '<') {
100
+ i += 2;
101
+ continue;
102
+ }
103
+ // The escaped character is literal text: it can neither open a
104
+ // backtick run nor start a tag, so it is carried along unprocessed,
105
+ // as `scanInline` steps over every escape.
106
+ chars[i] = c;
107
+ if (i + 1 < to) chars[i + 1] = text[i + 1];
108
+ i += 2;
109
+ continue;
110
+ }
111
+ if (c === '`') {
112
+ let n = 1;
113
+ while (i + n < to && text[i + n] === '`') n += 1;
114
+ const close = findBacktickRun(text, i + n, to, n);
115
+ if (close < 0) {
116
+ for (let k = i; k < i + n; k += 1) chars[k] = text[k];
117
+ i += n;
118
+ } else {
119
+ i = close + n;
120
+ }
121
+ continue;
122
+ }
123
+ chars[i] = c;
124
+ i += 1;
125
+ }
126
+ };
127
+ const fm = extractFrontmatter(text);
128
+ scanBody(text, fm ? text.length - fm.body.length : 0, {
129
+ prose: keep,
130
+ definition: (span) => keep(span.start, span.end),
131
+ });
132
+ // An HTML comment is not rendered: a tag inside one is blanked with it.
133
+ return chars.join('').replace(/<!--[\s\S]*?-->/g, (c) => c.replace(/[^\n]/g, ' '));
134
+ }
135
+
136
+ /**
137
+ * The `href`/`src` values of the raw HTML tags a markdown page carries —
138
+ * what the app renders as a link or a picture that the markdown grammar does
139
+ * not rewrite. Only an unescaped `<tag …>` outside code counts: a tag in a
140
+ * fence, a code span or behind a `\<` is an example, and a bare `href=` in
141
+ * prose is prose.
142
+ */
143
+ export function scanMarkdownHtmlLinks(text: string): HtmlLinkTarget[] {
144
+ const out: HtmlLinkTarget[] = [];
145
+ const masked = maskMarkdownCode(text);
146
+ const tag = new RegExp(HTML_TAG_RE.source, 'g');
147
+ for (let m = tag.exec(masked); m !== null; m = tag.exec(masked)) {
148
+ // The span located on the mask, the attributes read from the page itself.
149
+ out.push(...attributeLinks(text.slice(m.index, m.index + m[0].length)));
150
+ }
151
+ return out;
152
+ }
153
+
154
+ /** Where `scanBody` hands what it finds: prose runs (code left out) and reference definitions. */
155
+ interface BodySink {
156
+ prose: (from: number, to: number) => void;
157
+ definition: (span: MdLinkSpan) => void;
158
+ }
159
+
160
+ /**
161
+ * A frontmatter line whose whole value is one markdown link, quoted or not
162
+ * (the panel's `FRONTMATTER_LINK_RE`, on the value YAML hands it — which is
163
+ * why a trailing ` # comment`, stripped by YAML, may follow). An unquoted
164
+ * one is matched too: YAML reads it as a flow sequence, so the panel shows no
165
+ * link, but the graph tooling reads `nodeType` lines with a line regex, not
166
+ * YAML, and still follows it. Group 1 is everything before the link; group 3
167
+ * is the link.
168
+ */
169
+ const FRONTMATTER_LINK_VALUE_RE =
170
+ /^([ \t]*[^\s:#][^:]*:[ \t]+(["']?))(\[[^\]]+\]\(<?[^)>]+>?\))\2(?:[ \t]+#.*)?[ \t]*$/;
171
+
172
+ /**
173
+ * An opening code fence: its character and length. Group 1 is what precedes
174
+ * the fence on the line — quote and list markers and the fence's own indent —
175
+ * which may be at most three columns past its container.
176
+ */
177
+ const FENCE_OPEN_RE = /^((?:[ \t]*>[ \t]?)*[ \t]*(?:(?:[-*+]|\d{1,9}[.)])[ \t]+)?)(`{3,}|~{3,})(.*)$/;
178
+
179
+ /** The fence's own indent: the columns after the last quote or list marker. */
180
+ function fenceIndent(prefix: string, listIndent: number | null): number {
181
+ const quoted = prefix.lastIndexOf('>');
182
+ if (quoted >= 0) return indentOf(prefix.slice(quoted + 1).replace(/^[ \t]/, ''));
183
+ const item = LIST_ITEM_RE.exec(prefix);
184
+ if (item) return 0;
185
+ return indentOf(prefix) - (listIndent ?? 0);
186
+ }
187
+
188
+ /** A list item's marker, with the spaces after it. */
189
+ const LIST_ITEM_RE = /^( *)([-*+]|\d{1,9}[.)])( +|$)/;
190
+
191
+ /** The column a line's text starts at, tabs stopping every 4 columns. */
192
+ function indentOf(line: string): number {
193
+ let col = 0;
194
+ for (const c of line) {
195
+ if (c === ' ') col += 1;
196
+ else if (c === '\t') col += 4 - (col % 4);
197
+ else break;
198
+ }
199
+ return col;
200
+ }
201
+
202
+ function scanBody(text: string, from: number, sink: BodySink): void {
203
+ let fence: { char: string; len: number } | null = null;
204
+ // An indented code block: four columns past where the enclosing list item's
205
+ // text starts (column 0 outside a list), opened where a paragraph cannot be
206
+ // continued — after a blank line, a heading, a fence, or at the top.
207
+ let indentedCode = false;
208
+ /** The column the current list item's text starts at, or null outside a list. */
209
+ let listIndent: number | null = null;
210
+ let mayOpenCode = true;
211
+ let prevBlank = true;
212
+ // The run of prose lines since the last blank line or fence: inline
213
+ // constructs (a link split over two lines) live within one such run.
214
+ let chunkStart = -1;
215
+ let chunkEnd = -1;
216
+ const flush = () => {
217
+ if (chunkStart >= 0) sink.prose(chunkStart, chunkEnd);
218
+ chunkStart = -1;
219
+ };
220
+ let lineStart = from;
221
+ while (lineStart <= text.length) {
222
+ const nl = text.indexOf('\n', lineStart);
223
+ const lineEnd = nl === -1 ? text.length : nl;
224
+ const line = text.slice(lineStart, lineEnd).replace(/\r$/, '');
225
+ // Indentation is counted past the blockquote markers, so `> code`
226
+ // is an indented block inside the quote, as CommonMark reads it — and a
227
+ // bare `>` is the quote's blank line.
228
+ const quoted = /^(?:[ \t]*>[ \t]?)+/.exec(line)?.[0] ?? '';
229
+ const inner = line.slice(quoted.length);
230
+ const blank = inner.trim() === '';
231
+ const indent = indentOf(inner);
232
+ if (fence) {
233
+ const prefix = /^(?:[ \t]*>[ \t]?)*[ \t]*/.exec(line)![0];
234
+ const run = line.slice(prefix.length).match(/^(`+|~+)[ \t]*$/);
235
+ // A closing fence, like an opening one, sits at most three columns in.
236
+ if (run && run[1][0] === fence.char && run[1].length >= fence.len && fenceIndent(prefix, listIndent) <= 3) {
237
+ fence = null;
238
+ mayOpenCode = true;
239
+ }
240
+ } else if (blank) {
241
+ flush();
242
+ mayOpenCode = true;
243
+ } else if (indentedCode && indent >= (listIndent ?? 0) + 4) {
244
+ // Still inside the indented code block.
245
+ } else {
246
+ indentedCode = false;
247
+ // A line back at the margin after a blank line has left the list.
248
+ if (listIndent !== null && prevBlank && indent < listIndent) listIndent = null;
249
+ const item: RegExpExecArray | null = indent < (listIndent ?? 0) + 4 ? LIST_ITEM_RE.exec(inner) : null;
250
+ const fenceOpen = line.match(FENCE_OPEN_RE);
251
+ // Four columns in, a fence line is code or a paragraph's continuation.
252
+ const open = fenceOpen && fenceIndent(fenceOpen[1], listIndent) <= 3 ? fenceOpen : null;
253
+ if (!item && mayOpenCode && indent >= (listIndent ?? 0) + 4) {
254
+ flush();
255
+ indentedCode = true;
256
+ } else if (open && !(open[2][0] === '`' && open[3].includes('`'))) {
257
+ // A backtick fence's info string may hold no backtick (that is inline code).
258
+ flush();
259
+ fence = { char: open[2][0], len: open[2].length };
260
+ } else {
261
+ if (item) {
262
+ const gap: number = item[3].length;
263
+ listIndent = item[1].length + item[2].length + (gap === 0 || gap > 4 ? 1 : gap);
264
+ }
265
+ const def = matchDefinition(line, lineStart);
266
+ if (def) {
267
+ flush();
268
+ sink.definition(def);
269
+ mayOpenCode = false;
270
+ } else {
271
+ if (chunkStart < 0) chunkStart = lineStart;
272
+ chunkEnd = lineEnd;
273
+ // A heading ends its block; a paragraph line can be continued.
274
+ mayOpenCode = /^ {0,3}#{1,6}(?:[ \t]|$)/.test(line);
275
+ if (mayOpenCode) flush();
276
+ }
277
+ }
278
+ }
279
+ prevBlank = blank;
280
+ if (nl === -1) break;
281
+ lineStart = nl + 1;
282
+ }
283
+ flush();
284
+ }
285
+
286
+ /**
287
+ * `[label]: destination "title"` on one line, in a blockquote too; never a
288
+ * footnote (`[^1]: …`).
289
+ */
290
+ const DEFINITION_RE = /^((?:[ \t]{0,3}>[ \t]?)* {0,3}\[(?!\^)(?:[^\]\\]|\\.)+\]:[ \t]*)(<[^<>\n]*>|[^\s<][^\s]*)(?=[ \t]|$)/;
291
+
292
+ function matchDefinition(line: string, lineStart: number): MdLinkSpan | null {
293
+ const m = DEFINITION_RE.exec(line);
294
+ if (!m) return null;
295
+ const angle = m[2].startsWith('<');
296
+ const start = lineStart + m[1].length + (angle ? 1 : 0);
297
+ const destination = angle ? m[2].slice(1, -1) : m[2];
298
+ if (destination === '') return null;
299
+ return { kind: 'definition', start, end: start + destination.length, destination, angle, inFrontmatter: false };
300
+ }
301
+
302
+ /**
303
+ * The inline links and images in `text[from, to)`. Brackets are matched as a
304
+ * stack, so a link wrapping an image (`[![a](x.png)](y.md)`) yields both;
305
+ * backslash escapes and code spans are stepped over.
306
+ */
307
+ function scanInline(text: string, from: number, to: number, inFrontmatter: boolean, out: MdLinkSpan[]): void {
308
+ const openers: { image: boolean }[] = [];
309
+ let i = from;
310
+ while (i < to) {
311
+ const c = text[i];
312
+ if (c === '\\') {
313
+ i += 2;
314
+ continue;
315
+ }
316
+ if (c === '`') {
317
+ let n = 1;
318
+ while (i + n < to && text[i + n] === '`') n += 1;
319
+ const close = findBacktickRun(text, i + n, to, n);
320
+ i = close < 0 ? i + n : close + n;
321
+ continue;
322
+ }
323
+ if (c === '!' && text[i + 1] === '[') {
324
+ openers.push({ image: true });
325
+ i += 2;
326
+ continue;
327
+ }
328
+ if (c === '[') {
329
+ openers.push({ image: false });
330
+ i += 1;
331
+ continue;
332
+ }
333
+ if (c === ']') {
334
+ const opener = openers.pop();
335
+ if (opener && text[i + 1] === '(') {
336
+ const dest = parseInlineDestination(text, i + 2, to);
337
+ if (dest) {
338
+ if (dest.end > dest.start) {
339
+ out.push({
340
+ kind: opener.image ? 'image' : 'link',
341
+ start: dest.start,
342
+ end: dest.end,
343
+ destination: text.slice(dest.start, dest.end),
344
+ angle: dest.angle,
345
+ inFrontmatter,
346
+ });
347
+ }
348
+ i = dest.close + 1;
349
+ continue;
350
+ }
351
+ }
352
+ }
353
+ i += 1;
354
+ }
355
+ }
356
+
357
+ /** Where a run of exactly `n` backticks starts in `text[from, to)`, or -1. */
358
+ function findBacktickRun(text: string, from: number, to: number, n: number): number {
359
+ let i = from;
360
+ while (i < to) {
361
+ if (text[i] !== '`') {
362
+ i += 1;
363
+ continue;
364
+ }
365
+ let m = 1;
366
+ while (i + m < to && text[i + m] === '`') m += 1;
367
+ if (m === n) return i;
368
+ i += m;
369
+ }
370
+ return -1;
371
+ }
372
+
373
+ /** Skip spaces and tabs, and at most one line break. */
374
+ function skipSpace(text: string, p: number, to: number): number {
375
+ let newline = false;
376
+ while (p < to) {
377
+ const c = text[p];
378
+ if (c === ' ' || c === '\t' || c === '\r') p += 1;
379
+ else if (c === '\n' && !newline) {
380
+ newline = true;
381
+ p += 1;
382
+ } else break;
383
+ }
384
+ return p;
385
+ }
386
+
387
+ /**
388
+ * The destination of an inline link whose `(` ends just before `p`, with the
389
+ * offset of its closing `)`; null when what follows is not a link destination
390
+ * (then the brackets were just text).
391
+ */
392
+ function parseInlineDestination(
393
+ text: string,
394
+ p: number,
395
+ to: number,
396
+ ): { start: number; end: number; angle: boolean; close: number } | null {
397
+ p = skipSpace(text, p, to);
398
+ let start: number;
399
+ let end: number;
400
+ let angle = false;
401
+ if (text[p] === '<') {
402
+ angle = true;
403
+ start = p + 1;
404
+ let q = start;
405
+ while (q < to && text[q] !== '>') {
406
+ if (text[q] === '\n' || text[q] === '<') return null;
407
+ q += text[q] === '\\' ? 2 : 1;
408
+ }
409
+ if (q >= to) return null;
410
+ end = q;
411
+ p = q + 1;
412
+ } else {
413
+ start = p;
414
+ let depth = 0;
415
+ while (p < to) {
416
+ const c = text[p];
417
+ if (c === '\\' && p + 1 < to) {
418
+ p += 2;
419
+ continue;
420
+ }
421
+ if (c.charCodeAt(0) <= 0x20) break;
422
+ if (c === '(') depth += 1;
423
+ else if (c === ')') {
424
+ if (depth === 0) break;
425
+ depth -= 1;
426
+ }
427
+ p += 1;
428
+ }
429
+ if (depth !== 0) return null;
430
+ end = p;
431
+ }
432
+ const afterDest = p;
433
+ p = skipSpace(text, p, to);
434
+ if (text[p] === ')') return { start, end, angle, close: p };
435
+ // A title must be separated from the destination by whitespace.
436
+ if (p === afterDest) return null;
437
+ const opener = text[p];
438
+ const closer = opener === '"' ? '"' : opener === "'" ? "'" : opener === '(' ? ')' : null;
439
+ if (closer === null) return null;
440
+ p += 1;
441
+ while (p < to && text[p] !== closer) p += text[p] === '\\' ? 2 : 1;
442
+ if (p >= to) return null;
443
+ p = skipSpace(text, p + 1, to);
444
+ return text[p] === ')' ? { start, end, angle, close: p } : null;
445
+ }
446
+
447
+ /**
448
+ * The `href`/`src` values in an HTML page. A move never rewrites an HTML page
449
+ * (most of its links are built in scripts nobody can see), so these are only
450
+ * read, to report the page.
451
+ */
452
+ export function scanHtmlLinks(text: string): string[] {
453
+ return scanHtmlLinkTargets(text).map((t) => t.destination);
454
+ }
455
+
456
+ /** One `href` or `src` value in an HTML tag; `image` says which — a `src` is a picture, and a picture resolves as one. */
457
+ export interface HtmlLinkTarget {
458
+ destination: string;
459
+ image: boolean;
460
+ }
461
+
462
+ /** {@link scanHtmlLinks}, with each value's attribute kept: `src` is an image, `href` a link. */
463
+ export function scanHtmlLinkTargets(text: string): HtmlLinkTarget[] {
464
+ const out: HtmlLinkTarget[] = [];
465
+ const tag = new RegExp(HTML_TAG_RE.source, 'g');
466
+ for (let m = tag.exec(text); m !== null; m = tag.exec(text)) out.push(...attributeLinks(m[0]));
467
+ return out;
468
+ }
469
+
470
+ /** An opening tag: its end is the first `>` outside a quoted attribute value. */
471
+ const HTML_TAG_RE = /<[a-zA-Z][a-zA-Z0-9-]*(?:\s+(?:"[^"]*"|'[^']*'|[^<>"'])*)?>/g;
472
+
473
+ /**
474
+ * The `href` and `src` values of ONE tag, read attribute by attribute: the
475
+ * name has to be exactly that (a `data-href` is not a link), and a value that
476
+ * happens to contain `href=` is a value, not an attribute.
477
+ */
478
+ function attributeLinks(tag: string): HtmlLinkTarget[] {
479
+ const out: HtmlLinkTarget[] = [];
480
+ const nameEnd = tag.search(/[\s/>]/);
481
+ const attrs = nameEnd < 0 ? '' : tag.slice(nameEnd, -1);
482
+ const re = /([^\s"'=<>/]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/g;
483
+ for (let m = re.exec(attrs); m !== null; m = re.exec(attrs)) {
484
+ const name = m[1].toLowerCase();
485
+ if (name !== 'href' && name !== 'src') continue;
486
+ const destination = m[2] ?? m[3] ?? m[4];
487
+ if (destination !== undefined) out.push({ destination, image: name === 'src' });
488
+ }
489
+ return out;
490
+ }
491
+
492
+ // ── Resolution ──────────────────────────────────────────────────────────────
493
+
494
+ /** How a link names its target, which a rewrite keeps. */
495
+ export type MdLinkForm = 'relative' | 'root' | 'workspace';
496
+
497
+ /** A link destination that names a file or folder in the workspace. */
498
+ export interface ResolvedMdLink {
499
+ form: MdLinkForm;
500
+ /** Workspace-relative (`knowledge-base/…`), decoded. */
501
+ path: string;
502
+ /** `#anchor` as written, or ''. */
503
+ hash: string;
504
+ /** The branch a `/workspace/<branch>/…` link names; null otherwise. */
505
+ branch: string | null;
506
+ }
507
+
508
+ export interface ResolveMdLinkOptions {
509
+ /** The workspace-relative file the link sits in. */
510
+ basePath: string;
511
+ /** The clone folder (`knowledge-base`), for the mangled-path repair. */
512
+ kbDirName: string | null;
513
+ /** An image source gets no mangled-path repair (see the web app's resolver). */
514
+ image?: boolean;
515
+ }
516
+
517
+ /** An href as a browser reads it: outer controls and spaces off, inner tabs and line breaks out. */
518
+ export function normalizeMdHref(href: string): string {
519
+ const inner = href.replace(/[\t\n\r]/g, '');
520
+ let start = 0;
521
+ let end = inner.length;
522
+ while (start < end && inner.charCodeAt(start) <= 0x20) start += 1;
523
+ while (end > start && inner.charCodeAt(end - 1) <= 0x20) end -= 1;
524
+ return inner.slice(start, end);
525
+ }
526
+
527
+ /** Leaves the workspace: a scheme, or protocol-relative. */
528
+ function isExternal(href: string): boolean {
529
+ return /^[a-z][a-z0-9+.-]*:/i.test(href) || href.startsWith('//');
530
+ }
531
+
532
+ function safeDecode(s: string): string {
533
+ try {
534
+ return decodeURIComponent(s);
535
+ } catch {
536
+ return s;
537
+ }
538
+ }
539
+
540
+ /** The markdown backslash escapes of ASCII punctuation, undone — what a renderer hands the resolver. */
541
+ function unescapeMarkdown(s: string): string {
542
+ return s.replace(/\\([!-/:-@[-`{-~])/g, '$1');
543
+ }
544
+
545
+ /** A path whose later segment is the clone folder, cut back to start there. */
546
+ function stripJunkBeforeKbDir(path: string, kbDirName: string | null): string {
547
+ if (!kbDirName) return path;
548
+ const segs = path.split('/');
549
+ const idx = segs.indexOf(kbDirName);
550
+ return idx > 0 ? segs.slice(idx).join('/') : path;
551
+ }
552
+
553
+ /** `relative` against the file `basePath`, as the web app resolves it. */
554
+ export function resolveMdRelativePath(basePath: string, relative: string): string {
555
+ const baseDir = relative.startsWith('/')
556
+ ? ''
557
+ : basePath.includes('/')
558
+ ? basePath.slice(0, basePath.lastIndexOf('/'))
559
+ : '';
560
+ const parts = baseDir ? baseDir.split('/') : [];
561
+ for (const segment of relative.split('/')) {
562
+ if (segment === '..') parts.pop();
563
+ else if (segment !== '.' && segment !== '') parts.push(segment);
564
+ }
565
+ return parts.join('/');
566
+ }
567
+
568
+ /**
569
+ * What a destination written in `basePath` points at, or null when it names
570
+ * no workspace path a move could affect: an external URL, an id-link, a
571
+ * same-page `#anchor`, or a bare `/workspace/<branch>`.
572
+ */
573
+ export function resolveMdLink(destination: string, opts: ResolveMdLinkOptions): ResolvedMdLink | null {
574
+ const href = normalizeMdHref(unescapeMarkdown(destination));
575
+ if (!href || isExternal(href) || MD_ID_LINK_RE.test(href)) return null;
576
+ const hashIdx = href.indexOf('#');
577
+ const hash = hashIdx >= 0 ? href.slice(hashIdx) : '';
578
+ const location = hashIdx >= 0 ? href.slice(0, hashIdx) : href;
579
+ if (!location) return null;
580
+ const repair = (p: string) => (opts.image ? p : stripJunkBeforeKbDir(p, opts.kbDirName));
581
+ if (location.startsWith(WORKSPACE_ROUTE_PREFIX)) {
582
+ const rest = location.slice(WORKSPACE_ROUTE_PREFIX.length);
583
+ const slash = rest.indexOf('/');
584
+ if (slash < 0) return null;
585
+ return {
586
+ form: 'workspace',
587
+ branch: safeDecode(rest.slice(0, slash)),
588
+ path: repair(safeDecode(rest.slice(slash + 1))),
589
+ hash,
590
+ };
591
+ }
592
+ return {
593
+ form: location.startsWith('/') ? 'root' : 'relative',
594
+ branch: null,
595
+ path: repair(resolveMdRelativePath(opts.basePath, safeDecode(location))),
596
+ hash,
597
+ };
598
+ }
599
+
600
+ // ── Rewriting ───────────────────────────────────────────────────────────────
601
+
602
+ /** `target` relative to the folder `dir` (both workspace-relative). */
603
+ function relativeFrom(dir: string, target: string): string {
604
+ const from = dir ? dir.split('/') : [];
605
+ const to = target.split('/');
606
+ let i = 0;
607
+ while (i < from.length && i < to.length && from[i] === to[i]) i += 1;
608
+ const rel = [...Array<string>(from.length - i).fill('..'), ...to.slice(i)].join('/');
609
+ return rel === '' ? '.' : rel;
610
+ }
611
+
612
+ /** Characters that would end or break a destination written without angle brackets. */
613
+ function escapeBare(segment: string): string {
614
+ return segment.replace(/[\s<>()\\]/g, (c) => `%${c.charCodeAt(0).toString(16).toUpperCase().padStart(2, '0')}`);
615
+ }
616
+
617
+ function encodePath(path: string, how: { encoded: boolean; angle: boolean; inFrontmatter: boolean }): string {
618
+ let out: string;
619
+ if (how.encoded) {
620
+ out = path
621
+ .split('/')
622
+ .map((s) => encodeURIComponent(s).replace(/[()]/g, (c) => (c === '(' ? '%28' : '%29')))
623
+ .join('/');
624
+ } else if (how.angle) {
625
+ out = path.replace(/[<>\n]/g, (c) => encodeURIComponent(c));
626
+ } else {
627
+ // Only what would break the link: a balanced bare path stays as it is.
628
+ out = /[\s<>\\]/.test(path) || !balanced(path) ? escapeBare(path) : path;
629
+ }
630
+ // A frontmatter link sits inside a YAML quoted string, double or single:
631
+ // either quote in the path would end the string early.
632
+ return how.inFrontmatter ? out.replace(/["']/g, (c) => (c === '"' ? '%22' : '%27')) : out;
633
+ }
634
+
635
+ function balanced(path: string): boolean {
636
+ let depth = 0;
637
+ for (const c of path) {
638
+ if (c === '(') depth += 1;
639
+ else if (c === ')' && --depth < 0) return false;
640
+ }
641
+ return depth === 0;
642
+ }
643
+
644
+ /** The parent folder of a workspace-relative file path. */
645
+ function dirOf(path: string): string {
646
+ return path.includes('/') ? path.slice(0, path.lastIndexOf('/')) : '';
647
+ }
648
+
649
+ /**
650
+ * The destination `span` must carry once its file sits at `newBase` and its
651
+ * target at `newTarget`, in the form it was written in: relative stays
652
+ * relative (`./` and a trailing `/` kept), root-anchored stays root-anchored,
653
+ * an app URL keeps its branch segment; encoding, anchor — and, because only
654
+ * the destination is spliced, the title — stay as they were.
655
+ */
656
+ export function retargetMdDestination(
657
+ span: Pick<MdLinkSpan, 'destination' | 'angle' | 'inFrontmatter'>,
658
+ resolved: ResolvedMdLink,
659
+ newBase: string,
660
+ newTarget: string,
661
+ ): string {
662
+ const href = normalizeMdHref(unescapeMarkdown(span.destination));
663
+ const hashIdx = href.indexOf('#');
664
+ const location = hashIdx >= 0 ? href.slice(0, hashIdx) : href;
665
+ const how = { encoded: /%[0-9a-f]{2}/i.test(location), angle: span.angle, inFrontmatter: span.inFrontmatter };
666
+ const trailingSlash = location.length > 1 && location.endsWith('/') ? '/' : '';
667
+ let next: string;
668
+ if (resolved.form === 'workspace') {
669
+ const rest = location.slice(WORKSPACE_ROUTE_PREFIX.length);
670
+ const branchSegment = rest.slice(0, rest.indexOf('/'));
671
+ next = `${WORKSPACE_ROUTE_PREFIX}${branchSegment}/${encodePath(newTarget, how)}`;
672
+ } else if (resolved.form === 'root') {
673
+ next = `/${encodePath(newTarget, how)}`;
674
+ } else {
675
+ let rel = relativeFrom(dirOf(newBase), newTarget);
676
+ if (location.startsWith('./') && !rel.startsWith('../') && rel !== '.') rel = `./${rel}`;
677
+ next = encodePath(rel, how);
678
+ }
679
+ return `${next}${trailingSlash && !next.endsWith('/') ? '/' : ''}${resolved.hash}`;
680
+ }
681
+
682
+ /** One destination a rewrite changed. */
683
+ export interface MdLinkEdit {
684
+ from: string;
685
+ to: string;
686
+ }
687
+
688
+ export interface RewriteMdLinksOptions {
689
+ /** Where the file is now (workspace-relative). */
690
+ oldPath: string;
691
+ /** Where the file will be after the move (`oldPath` when it does not move). */
692
+ newPath: string;
693
+ /** A workspace path's place after the move, or null when it does not move. */
694
+ mapPath: (path: string) => string | null;
695
+ kbDirName: string | null;
696
+ /** The branch the move happens on: a link (not an image) to another branch's app URL is left alone. */
697
+ branch: string;
698
+ }
699
+
700
+ /**
701
+ * Rewrite the links in one markdown page so that, once it sits at `newPath`
702
+ * and every moved path is at its new place, each points at the same target it
703
+ * did before. A link that already does is left alone, as is every byte that is
704
+ * not a changed destination.
705
+ */
706
+ export function rewriteMdLinks(text: string, opts: RewriteMdLinksOptions): { text: string; edits: MdLinkEdit[] } {
707
+ const edits: (MdLinkEdit & { start: number; end: number })[] = [];
708
+ for (const span of scanMarkdownLinks(text)) {
709
+ const image = span.kind === 'image';
710
+ const before = resolveMdLink(span.destination, { basePath: opts.oldPath, kbDirName: opts.kbDirName, image });
711
+ // A link naming another branch opens that branch, which this move does
712
+ // not touch. An image does not: the web app serves every image from the
713
+ // branch the page is read on, whatever branch its URL names.
714
+ if (!before || (!image && before.branch !== null && before.branch !== opts.branch)) continue;
715
+ const target = opts.mapPath(before.path) ?? before.path;
716
+ const after = resolveMdLink(span.destination, { basePath: opts.newPath, kbDirName: opts.kbDirName, image });
717
+ if (after && after.path === target) continue;
718
+ const to = retargetMdDestination(span, before, opts.newPath, target);
719
+ if (to === span.destination) continue;
720
+ edits.push({ start: span.start, end: span.end, from: span.destination, to });
721
+ }
722
+ if (edits.length === 0) return { text, edits: [] };
723
+ let out = '';
724
+ let at = 0;
725
+ for (const e of edits) {
726
+ out += text.slice(at, e.start) + e.to;
727
+ at = e.end;
728
+ }
729
+ out += text.slice(at);
730
+ return { text: out, edits: edits.map(({ from, to }) => ({ from, to })) };
731
+ }
732
+
733
+ /**
734
+ * The destinations in an HTML page that point at a moved path, or (for a page
735
+ * that moves itself) whose relative target the move would break. Reported,
736
+ * never rewritten.
737
+ */
738
+ export function htmlLinksAffectedByMove(
739
+ text: string,
740
+ opts: RewriteMdLinksOptions,
741
+ /** The targets to judge: an HTML page's by default, a markdown page's from {@link scanMarkdownHtmlLinks}. */
742
+ targets: HtmlLinkTarget[] = scanHtmlLinkTargets(text),
743
+ ): string[] {
744
+ const out: string[] = [];
745
+ for (const { destination, image } of targets) {
746
+ // A picture resolves as the markdown image rule resolves one: no
747
+ // mangled-path repair, and served from the branch the page is read on
748
+ // whatever branch its address names — so another branch's app URL on an
749
+ // image is a reference to THIS branch and is judged, where a link's is not.
750
+ const before = resolveMdLink(destination, { basePath: opts.oldPath, kbDirName: opts.kbDirName, image });
751
+ if (!before || (!image && before.branch !== null && before.branch !== opts.branch)) continue;
752
+ const target = opts.mapPath(before.path) ?? before.path;
753
+ const after = resolveMdLink(destination, { basePath: opts.newPath, kbDirName: opts.kbDirName, image });
754
+ if (!after || after.path !== target) out.push(destination);
755
+ }
756
+ return out;
757
+ }