@markup-carve/carve-grammars 0.1.5 → 0.1.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -162,17 +162,21 @@ Unsupported handling:
162
162
 
163
163
  - `unsupported: 'throw'` is the default. The loader throws `UnsupportedNodeError`
164
164
  instead of silently dropping content.
165
- - `unsupported: 'preserve'` first builds the richest available document and
166
- verifies that serializing it preserves the parsed AST. Unsupported subtrees
167
- use opaque `carveUnsupported` blocks; if a mapped document is still lossy,
168
- the loader falls back to one whole-document opaque block. `serializeToCarve`
169
- writes its source back byte-for-byte, including edge whitespace.
170
-
171
- All corpus documents are therefore load/save lossless in preservation mode.
172
- Some constructs remain opaque rather than directly editable, including parts
173
- of figures, advanced tables, comments, raw passthrough, and source-layout edge
174
- cases. `tiptap/schema-map.json` is the public rich-mapping authority;
175
- `tests/lib/coverage.js` records why structured conversion falls back.
165
+ - `unsupported: 'preserve'` builds the richest available document and verifies
166
+ its canonical serialization against the parsed AST. When authored columns,
167
+ delimiter choices, blank ownership, or other source layout cannot be held in
168
+ ProseMirror attributes, the document carries both the authored source and its
169
+ canonical projection. `serializeToCarve` performs a three-way merge after an
170
+ edit, preserving untouched authored layout while giving changed content the
171
+ canonical Carve spelling.
172
+
173
+ All 1,538 documents and 440 categories in the pinned corpus are load/save
174
+ lossless in preservation mode, with no whole-document fallback. Abbreviation
175
+ definitions and uses, figures and captions, advanced tables, comments, raw
176
+ passthrough, references, and footnotes all have structured editor mappings.
177
+ `carveToProseMirrorWithReport()` identifies any future construct that still has
178
+ to use a local opaque atom; `tiptap/schema-map.json` is the public mapping
179
+ authority.
176
180
 
177
181
  ## Tab sets and code groups in the editor
178
182
 
@@ -152,7 +152,7 @@
152
152
  * @returns {object} the mode's `begin`/`end` pair, so the closer used by
153
153
  * the guard and the closer used by the mode cannot drift apart.
154
154
  */
155
- const paired = (opener, closer) => {
155
+ const paired = (opener, closer, { escapeAware = false, flanked = false } = {}) => {
156
156
  // The guard used to be written `(?:[^\n]|\n(?!\s*\n))*?` - unbounded,
157
157
  // lazy, and free to cross newlines. Proving there is NO closer therefore
158
158
  // cost a whole paragraph from every position, and a document made of
@@ -183,11 +183,108 @@
183
183
  const literal = lead.length === 2 ? lead[1] : lead;
184
184
  const inClass = /[\\\]^-]/.test(literal) ? '\\' + literal : literal;
185
185
  const run = `(?:[^${inClass}\\n]|\\n(?!\\s*\\n)){0,4096}`;
186
- const guard = `${run}(?:${lead}(?!${rest})${run}){0,32}${closer.source}`;
187
- return {
186
+ /*
187
+ * `escapeAware` opts a mode's GUARD into "an escaped delimiter is not a
188
+ * delimiter" (carve-grammars#385). Only the bare highlight asks for it,
189
+ * because it is the only closer here that a document can put a
190
+ * backslash in front of and mean it: `x =\= y` renders `x == y` with no
191
+ * mark, and this mode opened a run the engine does not.
192
+ *
193
+ * THE RUN CONSUMES AN ESCAPE AS A PAIR and refuses a bare backslash,
194
+ * rather than counting the backslash run in a lookbehind. Counting is
195
+ * what #385 shipped and what carve-grammars#390 replaces: a lookbehind
196
+ * that repeats a group has to be BOUNDED or it backtracks superlinearly
197
+ * (carve-grammars#380), and every bound is reachable - at `{0,32}` the
198
+ * guard stopped seeing the run at 66 backslashes and this mode refused a
199
+ * highlight the engine marks. A pair has no bound to reach.
200
+ *
201
+ * ONLY HALF THE FIX IS HERE. The guard proves a closer EXISTS; the span
202
+ * comes from `end`, which highlight.js applies on its own and which
203
+ * would still stop at an escaped `=`. What stops it is HIGHLIGHT listing
204
+ * ESCAPE in `contains` - that submode begins at the backslash, one
205
+ * column before the delimiter, so it wins by position and eats the pair.
206
+ * Neither half works alone: without this one the mode never opens,
207
+ * without the other it closes on the escaped delimiter.
208
+ *
209
+ * THE BOUND COUNTS ATOMS, and an escape pair is two characters, so the
210
+ * worst case is 8192 rather than 4096. Still a constant per position,
211
+ * which is all the bound is for; halving it would halve every ordinary
212
+ * body too.
213
+ *
214
+ * The default path is byte-identical to before, so the twelve other
215
+ * modes are untouched.
216
+ */
217
+ const escapedRun = `(?:\\\\[^\\n]|[^${inClass}\\n\\\\]|\\n(?!\\s*\\n)){0,4096}`;
218
+ const body = escapeAware ? escapedRun : run;
219
+ /*
220
+ * `flanked` says a closer may not FOLLOW WHITESPACE, which is the
221
+ * mirror of the `(?=\S)` every bare opener already carries
222
+ * (carve-grammars#392). Without it `a =b = d` scoped a highlight the
223
+ * engine renders literally, and so did the four other bare delimiters -
224
+ * one gap with five spellings. The nine BRACED modes do not ask for it
225
+ * and must not: the engine builds `a {*b *} d`, so a guard there would
226
+ * turn an over-scope into an under-scope.
227
+ *
228
+ * `end` IS LEFT ALONE, and that is forced by the engine twice over.
229
+ *
230
+ * A lookbehind ANYWHERE in a mode's `end` stops highlight.js closing
231
+ * the mode at all - `(?<=\S)=(?!\w)` and `=(?!\w)(?<=\S=)` both leave
232
+ * `x =b= y` coloured to end of line, with no error. And a CONSUMED
233
+ * flank, `[^\s\n]=(?!\w)`, closes correctly but starts one column
234
+ * earlier, which is exactly where the escape submode starts: it wins
235
+ * the tie and closes on the `\=` that carve-grammars#390 exists to
236
+ * refuse. Both measured on a two-line grammar before this was written.
237
+ *
238
+ * SO THE FALSE CLOSER IS EATEN INSTEAD OF REFUSED, the same shape #390
239
+ * uses for the escape. `falseCloser` begins at the WHITESPACE, one
240
+ * column before `end` could fire, so it wins by position and takes the
241
+ * delimiter as content; `end` then only ever sees a closer that stands
242
+ * against a non-space. Nothing about `end` changes, so this composes
243
+ * with the escape submode rather than fighting it.
244
+ *
245
+ * THE GUARD IS THE OTHER HALF, and it may consume, because it is inside
246
+ * a lookahead where nothing is really taken. Without it a run with no
247
+ * real closer would open, eat its false closers and colour to the end
248
+ * of its parent - worse than the over-scope this fixes. `notCloser`
249
+ * widens for the same reason: `a *b * c* d` is a bold run in the
250
+ * engine, and the guard has to be able to step over the middle star to
251
+ * see the real one.
252
+ */
253
+ /*
254
+ * THE FLANK CHARACTER SHARES THE BODY'S ALPHABET. A bare `[^\s\n]`
255
+ * would match a lone backslash, so on `x =\= y` the guard ends its body
256
+ * before the backslash, eats it as the flank, and proves a closer that
257
+ * `end` - correctly held off by the escape submode - then never fires
258
+ * on: the mode opens and colours to the end of its parent. An
259
+ * escape-aware flank is a PAIR or a non-backslash, exactly like the run.
260
+ */
261
+ // AN ESCAPED SPACE IS STILL A SPACE to the flanking rule: the engine
262
+ // renders `x =\\ = y` with no mark, so the pair may not stand as the
263
+ // non-space character the closer needs behind it.
264
+ const flankUnit = escapeAware ? '(?:\\\\[^\\s\\n]|[^\\s\\n\\\\])' : '[^\\s\\n]';
265
+ const endSource = flanked ? `${flankUnit}${closer.source}` : closer.source;
266
+ const notCloser = flanked
267
+ ? `(?:\\s${lead}|${lead}(?!${rest}))`
268
+ : `${lead}(?!${rest})`;
269
+ const guard = `${body}(?:${notCloser}${body}){0,32}${endSource}`;
270
+ const mode = {
188
271
  begin: new RegExp(`${opener.source}(?=${guard})`),
189
272
  end: closer,
190
273
  };
274
+ if (flanked) mode.contains = [{ begin: new RegExp(`\\s${closer.source}`) }];
275
+
276
+ return mode;
277
+ };
278
+
279
+ // Escaped characters: \* \[ etc
280
+ //
281
+ // DEFINED HERE, above the modes rather than beside HARD_BREAK where it used
282
+ // to sit, because HIGHLIGHT now `contains` it (carve-grammars#390) and a
283
+ // `const` cannot be read before its own declaration.
284
+ const ESCAPE = {
285
+ className: 'symbol',
286
+ begin: /\\[!"#$%&'()*+,.\/:;<=>?@\[\\\]^_`{|}~-]/,
287
+ relevance: 0,
191
288
  };
192
289
 
193
290
  // Forced intraword family (PART 9 S22). Content may contain the delimiter
@@ -277,14 +374,14 @@
277
374
  // (a/b, ://); the end is a closing slash not followed by word char/slash.
278
375
  const EMPHASIS = {
279
376
  className: 'emphasis',
280
- ...paired(/(?<![\w:/])\/(?=\S)/, /\/(?![\w/])/),
377
+ ...paired(/(?<![\w:/])\/(?=\S)/, /\/(?![\w/])/, { flanked: true }),
281
378
  relevance: 0,
282
379
  };
283
380
 
284
381
  // Underline (Carve): _text_ - not in the middle of words
285
382
  const UNDERLINE = {
286
383
  className: 'emphasis',
287
- ...paired(/(?<!\w)_(?!\s)/, /_(?!\w)/),
384
+ ...paired(/(?<!\w)_(?!\s)/, /_(?!\w)/, { flanked: true }),
288
385
  relevance: 0,
289
386
  };
290
387
 
@@ -309,17 +406,27 @@
309
406
  */
310
407
  const BOLD_ITALIC = {
311
408
  className: 'strong',
312
- begin: /\/\*(?=\S)(?:[^*\n]|\*(?!\/)|\n(?!\s*\n)){1,4096}\*\//,
409
+ // The body is non-space at BOTH ends. It was guarded only at the
410
+ // opener, so `a /*b */ c` came back one combined run where the engine
411
+ // renders `<em>*b *</em>` - an italic run holding literal asterisks
412
+ // (carve-grammars#375). A fixed-width lookbehind, so the bounded
413
+ // repetition that keeps this rule linear is untouched.
414
+ begin: /\/\*(?=\S)(?:[^*\n]|\*(?!\/)|\n(?!\s*\n)){1,4096}(?<=\S)\*\//,
313
415
  relevance: 5,
314
416
  };
315
417
 
316
418
  // Strong: *text* - not in the middle of words, can contain emphasis.
317
419
  // Excludes *[ which is abbreviation-definition syntax.
420
+ const STRONG_PAIR = paired(/(?<!\w)\*(?![\s\[])/, /\*(?!\w)/, { flanked: true });
318
421
  const STRONG = {
319
422
  className: 'strong',
320
- ...paired(/(?<!\w)\*(?![\s\[])/, /\*(?!\w)/),
423
+ ...STRONG_PAIR,
321
424
  relevance: 0,
322
- contains: [EMPHASIS, UNDERLINE],
425
+ // SPREAD FIRST, then merge: `paired()` supplies a false-closer submode
426
+ // (carve-grammars#392) and a bare re-declaration would drop it, which
427
+ // is silent - the mode still opens, it just closes at the first
428
+ // delimiter standing after a space.
429
+ contains: [EMPHASIS, UNDERLINE, ...STRONG_PAIR.contains],
323
430
  };
324
431
 
325
432
  /*
@@ -351,9 +458,15 @@
351
458
  * opened and closed on the two `=` of a doubled run and scoped an empty
352
459
  * highlight over a line the engine renders literally.
353
460
  *
354
- * THE CLOSER IS DELIBERATELY NOT GUARDED, and the asymmetry is the
355
- * engine's: once a highlight is open the closer wins over the pattern, so
356
- * `x =y z<= w` marks `y z<` and `x =y z=> w` marks `y z`.
461
+ * THE CLOSER IS DELIBERATELY NOT GUARDED AGAINST SMART TYPOGRAPHY, and the
462
+ * asymmetry is the engine's: once a highlight is open the closer wins over
463
+ * the pattern, so `x =y z<= w` marks `y z<` and `x =y z=> w` marks `y z`.
464
+ *
465
+ * AN ESCAPE IS A DIFFERENT QUESTION (carve-grammars#385). An escaped `=` is
466
+ * not a delimiter at all - `x =\= y` renders `x == y` with no mark - and
467
+ * this mode closed on one. `paired`'s third argument is what says so; it is
468
+ * the only mode here that asks for it, because it is the only closer a
469
+ * document can put a backslash in front of and mean it.
357
470
  *
358
471
  * ONE SHAPE IT COSTS: `<https://e.example>=hi=`, where the `>` closes an
359
472
  * autolink rather than opening a comparison. A fixed-width lookbehind
@@ -361,9 +474,16 @@
361
474
  * that trade - the ticket's own reasoning, since a false highlight claims
362
475
  * the document holds a construct it does not.
363
476
  */
477
+ const HIGHLIGHT_PAIR = paired(/(?<![=\w])=(?=\S)(?![>=])/, /=(?![=\w])/, { escapeAware: true, flanked: true });
364
478
  const HIGHLIGHT = {
365
479
  className: 'addition',
366
- ...paired(/(?<![=\w])=(?=\S)(?![>=])/, /=(?![=\w])/),
480
+ ...HIGHLIGHT_PAIR,
481
+ // The other half of carve-grammars#390's escape fix - see `paired`.
482
+ // `end` would close on an escaped `=`; this submode begins one column
483
+ // earlier, at the backslash, and highlight.js resolves by position.
484
+ // The false-closer submode `paired()` supplies (carve-grammars#392) is
485
+ // kept alongside it: a bare `contains: [ESCAPE]` here would drop it.
486
+ contains: [ESCAPE, ...HIGHLIGHT_PAIR.contains],
367
487
  relevance: 3,
368
488
  };
369
489
 
@@ -375,16 +495,21 @@
375
495
  };
376
496
 
377
497
  // Delete: {-text-}
498
+ //
499
+ // THE BODY IS NOT EMPTY. `{--}` is a braced EN DASH, not an empty deletion -
500
+ // the engine renders `a {--} b` as `a \u2013 b` - and this mode read it as
501
+ // `{-` plus nothing plus `-}` (carve-grammars#378). One character is enough:
502
+ // `{- -}`, `{---}` and `{----}` are all deletions.
378
503
  const DELETE = {
379
504
  className: 'deletion',
380
- ...paired(/\{-/, /-\}/),
505
+ ...paired(/\{-(?!-\})/, /-\}/),
381
506
  relevance: 5,
382
507
  };
383
508
 
384
509
  // Strikethrough (Carve): ~text~ (Djot uses ~ for subscript instead)
385
510
  const STRIKETHROUGH = {
386
511
  className: 'deletion',
387
- ...paired(/(?<!\w)~(?=\S)/, /~(?!\w)/),
512
+ ...paired(/(?<!\w)~(?=\S)/, /~(?!\w)/, { flanked: true }),
388
513
  relevance: 2,
389
514
  };
390
515
 
@@ -642,7 +767,7 @@
642
767
  // dash alternatives: preceded by a non-space, or not followed by a word.
643
768
  const TYPOGRAPHY = {
644
769
  className: 'literal',
645
- begin: /\.\.\.|<-->|<--|-->|<=>|<==|==>|<->|<-|->|(?:(?<=\S)(?:---|--)|(?:---|--)(?!\w))|!=|<=|>=|\+-|\(c\)|\(r\)|\(tm\)/,
770
+ begin: /\{--\}|\.\.\.|<-->|<--|-->|<=>|<==|==>|<->|<-|->|(?:(?<=\S)(?:---|--)|(?:---|--)(?!\w))|!=|<=|>=|\+-|\(c\)|\(r\)|\(tm\)/,
646
771
  relevance: 0,
647
772
  };
648
773
 
@@ -1467,13 +1592,6 @@
1467
1592
  relevance: 5,
1468
1593
  };
1469
1594
 
1470
- // Escaped characters: \* \[ etc
1471
- const ESCAPE = {
1472
- className: 'symbol',
1473
- begin: /\\[!"#$%&'()*+,.\/:;<=>?@\[\\\]^_`{|}~-]/,
1474
- relevance: 0,
1475
- };
1476
-
1477
1595
  // Hard line break: \ at end of line
1478
1596
  const HARD_BREAK = {
1479
1597
  className: 'meta',
package/package.json CHANGED
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "@markup-carve/carve-grammars",
3
- "version": "0.1.5",
3
+ "version": "0.1.7",
4
4
  "description": "Grammars for the Carve markup language: Tiptap editor kit + serializer, plus Prism, highlight.js and TextMate syntax-highlighting grammars",
5
5
  "type": "module",
6
6
  "main": "tiptap/index.js",
7
7
  "scripts": {
8
- "test": "npm run test:types && node tests/carve-editor-test.js && node tests/block-battery-test.js && node tests/smart-typography-test.js && node tests/highlight-opener-test.js && node tests/fenced-quote-opener-test.js && node tests/coverage-test.js && node tests/construct-ledger-test.js && node tests/opaque-payload-test.js && node tests/latest-syntax-test.js && node tests/schema-map-test.js && node tests/scans-are-bounded-test.js && node tests/braced-scan-equivalence-test.js && node tests/line-ambiguity-test.js && node tests/snapshot-test.js && node tests/roundtrip-test.js && node tests/node-producers-test.js && node tests/wire-fixtures-test.js && node tests/loss-report-test.js && node tests/authored-spelling-test.js && node tests/optional-corpus-test.js && node tests/mounted-roundtrip-test.js && node tests/serializer-test.js && node tests/tabs-roundtrip-test.js && node tests/panel-bar-test.js && node tests/parse-test.js && node tests/language-attribute-test.js && node tests/blockquote-caption-test.js && node tests/composite-figure-test.js && node tests/composite-figure-tiptap-test.js && node tests/grammar-test.js && node tests/shiki-test.js && node tests/alias-parity-test.js && node tests/engine-sweep-test.js && node tests/abbreviation-term-test.js && node tests/unclosed-delimiter-test.js && node tests/textmate-sweep-test.js && node tests/kroki-test.js && node tests/client-diagram-test.js && node tests/packaging-test.js && node tests/no-git-dependencies-test.js && node tests/engine-drift-test.js && node tests/surface-drift-test.js",
9
- "test:types": "tsc -p tsconfig.types.json",
8
+ "test": "npm run test:types && node tests/carve-editor-test.js && node tests/block-battery-test.js && node tests/smart-typography-test.js && node tests/highlight-opener-test.js && node tests/highlight-escape-test.js && node tests/right-flank-test.js && node tests/bold-unclosed-test.js && node tests/textmate-harness-test.js && node tests/fenced-quote-opener-test.js && node tests/coverage-test.js && node tests/construct-ledger-test.js && node tests/opaque-payload-test.js && node tests/latest-syntax-test.js && node tests/schema-map-test.js && node tests/scans-are-bounded-test.js && node tests/braced-scan-equivalence-test.js && node tests/line-ambiguity-test.js && node tests/snapshot-test.js && node tests/roundtrip-test.js && node tests/node-producers-test.js && node tests/wire-fixtures-test.js && node tests/loss-report-test.js && node tests/authored-spelling-test.js && node tests/optional-corpus-test.js && node tests/mounted-roundtrip-test.js && node tests/serializer-test.js && node tests/task-state-roundtrip-test.js && node tests/tabs-roundtrip-test.js && node tests/visual-tabs-fuzz-test.js && node tests/panel-bar-test.js && node tests/parse-test.js && node tests/language-attribute-test.js && node tests/blockquote-caption-test.js && node tests/composite-figure-test.js && node tests/composite-figure-tiptap-test.js && node tests/grammar-test.js && node tests/shiki-test.js && node tests/alias-parity-test.js && node tests/engine-sweep-test.js && node tests/abbreviation-term-test.js && node tests/unclosed-delimiter-test.js && node tests/textmate-sweep-test.js && node tests/kroki-test.js && node tests/client-diagram-test.js && node tests/packaging-test.js && node tests/no-git-dependencies-test.js && node tests/engine-drift-test.js && node tests/surface-drift-test.js",
9
+ "test:types": "tsc -p tsconfig.types.json && node tests/source-merge-test.js",
10
10
  "test:citations": "node tests/spec-citations-test.js",
11
11
  "test:coverage": "node tests/coverage-test.js",
12
12
  "test:snapshot": "node tests/snapshot-test.js",
@@ -83,7 +83,8 @@
83
83
  "./package.json": "./package.json"
84
84
  },
85
85
  "dependencies": {
86
- "@markup-carve/carve": "^0.1.4"
86
+ "@markup-carve/carve": "^0.1.6",
87
+ "node-diff3": "^3.2.1"
87
88
  },
88
89
  "peerDependencies": {
89
90
  "@shikijs/themes": "^2 || ^3",
package/prism/carve.js CHANGED
@@ -198,8 +198,56 @@
198
198
  // Shared inline emphasis/markup, referenced from block tokens that contain
199
199
  // running text (headings, list items, table cells, quotes).
200
200
  var inline = {
201
+ // BOTH NESTING ORDERS. The engine renders `/*both*/` and `*/both/*`
202
+ // identically - each is <strong><em>both</em></strong> - and only the
203
+ // canonical order had a branch, so the mirrored one fell through to
204
+ // 'bold' and came back a bold run holding two literal slashes
205
+ // (carve-grammars#375).
206
+ //
207
+ // The mirrored branch carries guards the canonical one must not have,
208
+ // and each is the engine's own asymmetry:
209
+ //
210
+ // - `x/*b*/y` IS bold-italic and `x*/b/*y` is `x*<em>b</em>*y`,
211
+ // because the mirrored opener leads with `*` and a `*` glued to a
212
+ // word opens nothing.
213
+ // - a `/` before the opener makes the pair ambiguous with a canonical
214
+ // opener: `a /*/a/* b` is `a <em>*/a</em>* b`.
215
+ // - the closer may not follow a `*`: `a */b*/* c` is
216
+ // `a <strong>/b</strong>/* c`, while `a */*b/* c` IS combined - so
217
+ // it is the character before the closer that decides.
218
+ //
219
+ // BOTH BODIES ARE TEMPERED, BY DIFFERENT AMOUNTS (carve-grammars#382),
220
+ // and the asymmetry was measured rather than chosen. The canonical body
221
+ // admits any `*` that does not start the closer. The MIRRORED body may
222
+ // not do the same: a `/` that does not start the closer is still
223
+ // ambiguous with a canonical opener, so its `/` is admitted only
224
+ // BETWEEN TWO WORD CHARACTERS, which is what lets `a */b/c/* d` read as
225
+ // the combined run the engine renders while `x *///* y` stays bold.
226
+ //
227
+ // The opener's `(?!\*[\s*])` is what pays for it. Over all 21,844
228
+ // documents `x */ body /* y` for every body of up to seven characters
229
+ // drawn from `* / space a`, judged by asking whether the engine renders
230
+ // the WHOLE body as one combined run:
231
+ //
232
+ // strict (before) over 440 missed 603 wrong 1043
233
+ // any non-closing slash (rejected) over 5091 missed 161 wrong 5252
234
+ // word-flanked slash alone over 510 missed 461 wrong 971
235
+ // word-flanked + opener guard (this) over 226 missed 461 wrong 687
236
+ //
237
+ // The temper alone RAISES over-colouring, and every shape it adds is
238
+ // one the opener guard refuses - `a */**a/a/* d` and `a */* a/a/* d`
239
+ // are bold runs in the engine. The guard costs nothing: `missed` is
240
+ // unchanged with and without it. Repeated over `* / space a .` to
241
+ // separate a word-character flank from a non-space one, the word-
242
+ // character spelling wins there too (311 wrong against 454).
243
+ //
244
+ // Both bodies are non-space at each end, and both may cross a line
245
+ // without crossing a blank one - corpus 208 is a combined run spanning
246
+ // two lines, and the highlight.js rule spells the same body. `a /*b */ c`
247
+ // is `<em>*b *</em>` and not a combined run; the old `[^*]+` body,
248
+ // guarded only at the opener, read it as one.
201
249
  'bold-italic': {
202
- pattern: /\/\*(?=\S)[^*]+\*\//,
250
+ pattern: /\/\*(?=\S)(?:[^*\n]|\*(?!\/)|\n(?!\s*\n)){1,4096}(?<=\S)\*\/|(?<![\w*/])\*\/(?=\S)(?!\*[\s*])(?:[^/\n]|(?<=\w)\/(?=\w)|\n(?!\s*\n)){1,4096}(?<![\s*])\/\*(?!\w)/,
203
251
  alias: 'important',
204
252
  },
205
253
  // The "no leading/trailing space" rule is expressed without JS
@@ -285,13 +333,70 @@
285
333
  // the characters that changed a measurement when reverted, which
286
334
  // is why the three spellings are not identical.
287
335
  //
288
- // The closer is deliberately unguarded - once a highlight is open
289
- // the closer wins over the pattern in the engine too, so
290
- // `x =y z<= w` marks `y z<`.
336
+ // AND AN ESCAPED FLANK IS NOT A FLANK (carve-grammars#380). A
337
+ // lookbehind sees a character, not whether it was escaped, so
338
+ // `a \!=b c= d` - where the `!` is literal and the `=` after it is
339
+ // an ordinary opener the engine marks - was refused, and
340
+ // `a \=b c= d`, whose `=` cannot open at all, was taken.
341
+ //
342
+ // BACKSLASHES NEST, so both halves count them rather than looking
343
+ // at one character: the opener needs an EVEN run behind it (`\\=`
344
+ // is a literal backslash and an ordinary opener) and the admission
345
+ // needs an ODD one in front of the flank (`\\!=` is a literal
346
+ // backslash and a comparison, which the engine leaves alone). A
347
+ // first draft looked one character back and got both of those
348
+ // backwards.
349
+ //
350
+ // THE PAIR REPETITION IS BOUNDED, like every other scan in this
351
+ // file. Written as an unbounded repetition inside a lookbehind it
352
+ // backtracks superlinearly over a long backslash run - measured at
353
+ // 7 / 30 / 91 / 346 ms for 4K / 8K / 16K / 32K backslashes, four
354
+ // times per doubling, and 4 / 3 / 8 / 7 ms bounded. 64 backslashes
355
+ // in front of a highlight opener is not a document; past the bound
356
+ // the run does not open, which is the safe direction.
357
+ //
358
+ // A LOOKBEHIND IS WHAT THIS RULE HAS TO USE. Moving the `escape`
359
+ // token in front of this one does not help: the rule is `greedy`,
360
+ // so Prism runs it against the ORIGINAL text from the current
361
+ // offset and the lookbehind still sees the flank an earlier token
362
+ // consumed. Measured - with `escape` moved ahead of the inline
363
+ // family, `a \!=b c= d` still scoped nothing. The other two
364
+ // grammars need none of this, because their escape rule really
365
+ // does consume the flank before the highlight rule is asked.
366
+ // THE CLOSER IS UNGUARDED AGAINST SMART TYPOGRAPHY, deliberately -
367
+ // once a highlight is open the closer wins over the pattern in the
368
+ // engine too, so `x =y z<= w` marks `y z<`. AN ESCAPE IS A
369
+ // DIFFERENT QUESTION (carve-grammars#385): an escaped `=` is not a
370
+ // delimiter at all, so `x =\= y` renders `x == y` with no mark and
371
+ // this rule closed on it. THE BODY CONSUMES AN ESCAPE AS A PAIR and
372
+ // refuses a bare backslash, which settles both directions at once -
373
+ // the escaped `=` is eaten before the closer can see it, and a body
374
+ // that could not hold one would stop there and never reach the real
375
+ // closer past it, so `a =b c\= d= e`, which the engine marks, would
376
+ // have gone from wrongly-coloured to uncoloured. The closer needs no
377
+ // guard of its own as a CONSEQUENCE, not a simplification: the body
378
+ // can only get past a backslash by taking the pair, so the position
379
+ // immediately before an escaped `=` is never a body end.
380
+ // #385 COUNTED THE RUN IN A LOOKBEHIND on the closer and its odd
381
+ // twin on the body, and carve-grammars#390 replaced both. Every
382
+ // bound is reachable: at `{0,32}` the guard stopped seeing the run
383
+ // at 66 backslashes and this rule refused a highlight the engine
384
+ // marks. A pair has no bound - and the same spelling is what the
385
+ // TextMate family had to use, since one TextMate engine refuses a
386
+ // variable-length lookbehind outright. The OPENER's two lookbehinds
387
+ // stay: they are the escaped-FLANK guard from carve-grammars#380,
388
+ // which is a different question and which this rule cannot avoid
389
+ // for the `greedy` reason above.
291
390
  // ONE SHAPE IT COSTS: `<https://e.example>=hi=`, where the `>`
292
391
  // closes an autolink rather than opening a comparison; a
293
392
  // fixed-width lookbehind cannot tell the two apart, and this takes
294
393
  // the under-colouring side of that trade.
394
+ // THE BOUND COUNTS ATOMS, and an escape pair is two characters, so
395
+ // the worst case is 8192 characters rather than 4096. That is still
396
+ // a constant per position, which is all the bound is for; measured,
397
+ // an all-escape body costs what a plain body of the same LENGTH
398
+ // costs. Halving it to keep the character count would halve every
399
+ // ordinary body too.
295
400
  // ONE TOKEN, TWO FORMS, so the bare `=x=` alternative shares a
296
401
  // line with the braced one. Its `[^=\n]+?` was never the defect
297
402
  // this rule was fixed for - the class already excludes its own
@@ -299,7 +404,7 @@
299
404
  // the last unbounded quantifier on a line that spells a braced
300
405
  // construct, and the derived family check below reads lines. Given
301
406
  // a bound, at the same 4096 the rest of the file uses.
302
- pattern: /\{=(?=\S)[^=\n]{0,4096}(?:=(?!\})[^=\n]{0,4096}){0,32}=\}|(?<![\w=<>!])=(?=\S)(?!>)[^=\n]{1,4096}?(?<=\S)=(?![\w=])/,
407
+ pattern: /\{=(?=\S)[^=\n]{0,4096}(?:=(?!\})[^=\n]{0,4096}){0,32}=\}|(?:(?<=(?:^|[^\\])(?:\\\\){0,32})(?<![\w=<>!])|(?<=(?:^|[^\\])(?:\\\\){0,32}\\[<>!]))=(?=\S)(?!>)(?:\\.|[^=\n\\]){1,4096}?(?<=\S)=(?![\w=])/,
303
408
  alias: 'important',
304
409
  },
305
410
  // Braced-only: a bare `^` / `,` is literal text (no bare sup/sub).
@@ -1366,8 +1471,12 @@
1366
1471
  pattern: /\{\+[^+}]{0,4096}(?:\+(?!\})[^+}]{0,4096}){0,32}\+\}/,
1367
1472
  alias: 'inserted',
1368
1473
  },
1474
+ // THE BODY IS NOT EMPTY. `{--}` is a braced EN DASH, not an empty
1475
+ // deletion - the engine renders `a {--} b` as `a \u2013 b` - and this rule
1476
+ // read it as `{-` plus nothing plus `-}` (carve-grammars#378). One
1477
+ // character is enough: `{- -}`, `{---}` and `{----}` are all deletions.
1369
1478
  'deleted': {
1370
- pattern: /\{-[^\-}]{0,4096}(?:-(?!\})[^\-}]{0,4096}){0,32}-\}/,
1479
+ pattern: /\{-(?!-\})[^\-}]{0,4096}(?:-(?!\})[^\-}]{0,4096}){0,32}-\}/,
1371
1480
  alias: 'deleted',
1372
1481
  },
1373
1482
  // The one `{~ ... ~}` rule the sweep did NOT flag - a substitution's two
@@ -1452,7 +1561,7 @@
1452
1561
  // around the dash alternatives: preceded by a non-space, or not
1453
1562
  // followed by a word character.
1454
1563
  'typography': {
1455
- pattern: /\.\.\.|<-->|<--|-->|<=>|<==|==>|<->|<-|->|(?:(?<=\S)(?:---|--)|(?:---|--)(?!\w))|!=|<=|>=|\+-|\(c\)|\(r\)|\(tm\)/,
1564
+ pattern: /\{--\}|\.\.\.|<-->|<--|-->|<=>|<==|==>|<->|<-|->|(?:(?<=\S)(?:---|--)|(?:---|--)(?!\w))|!=|<=|>=|\+-|\(c\)|\(r\)|\(tm\)/,
1456
1565
  alias: 'constant',
1457
1566
  },
1458
1567
  };
@@ -1173,7 +1173,8 @@
1173
1173
  "name": "markup.inserted.carve"
1174
1174
  },
1175
1175
  "critic_del": {
1176
- "match": "\\{-[^}]*-\\}",
1176
+ "comment": "THE BODY IS NOT EMPTY. `{--}` is a braced EN DASH, not an empty deletion - grammar.ebnf gives `braced_en_dash = \"{--}\"` and the clause is normative (markup-carve/carve#1447); carve 0.1.5 renders `a {--} b` as `a \u2013 b`. This rule read it as `{-` plus nothing plus `-}` and coloured it a deletion, which is a colour that says the document holds something it does not, and no sweep could see it because the run WAS coloured (carve-grammars#378). One character of body is enough: `{- -}` deletes a space, `{---}` deletes a hyphen and `{----}` deletes an en dash, all measured against the engine.",
1177
+ "match": "\\{-(?!-\\})[^}]*-\\}",
1177
1178
  "name": "markup.deleted.carve"
1178
1179
  },
1179
1180
  "critic_sub": {
@@ -1300,22 +1301,41 @@
1300
1301
  }
1301
1302
  },
1302
1303
  "bold_italic": {
1303
- "match": "(/\\*)([^*]+)(\\*/)",
1304
- "captures": {
1305
- "1": {
1306
- "name": "punctuation.definition.bold-italic.carve"
1307
- },
1308
- "2": {
1309
- "name": "markup.bold.italic.carve"
1304
+ "comment": "BOTH NESTING ORDERS, because the engine renders them identically: `some /*both*/ text` and `some */both/* text` are each `<strong><em>both</em></strong>` (carve 0.1.5). Only the canonical order had a rule, so the mirrored one fell through to #strong and came back a bold run containing two literal slashes - a reading that says the document holds something it does not, and one no sweep could see because the row sampled the order this rule got right (carve-grammars#375). THE MIRRORED OPENER NEEDS A FLANKING GUARD AND THE CANONICAL ONE DOES NOT, and that asymmetry is the engine's: `x/*b*/y` IS bold-italic and `x*/b/*y` is `x*<em>b</em>*y`, because the mirrored opener leads with `*` and a `*` glued to a word opens nothing. Same at the other end: `x/*b*/y` closes into a word and `a */b/*y c` does not. CONTENT IS NON-SPACE AT BOTH ENDS in both orders. `a /* b */ c` is `<em>* b *</em>` and `a /*b */ c` is `<em>*b *</em>` - italic runs holding literal asterisks, not combined runs - and this rule read both as one combined run until the two guards arrived with the mirrored order. WHAT NEITHER ORDER COVERS: a body holding the OTHER order's delimiter. The engine reads `a /*b*c*/ d` and `a */b/c/* d` as combined runs and both branches decline them, which is the under-colouring side of the trade on the mirrored side - carve-grammars#382. BOTH BODIES ARE TEMPERED, BY DIFFERENT AMOUNTS, and the asymmetry was measured rather than chosen (carve-grammars#382). The canonical body admits any `*` that does not start the closer, which is what lets `a /*b*c*/ d` read as the combined run the engine renders. The MIRRORED body may not do the same: a `/` that does not start the closer is still ambiguous with a canonical opener, so its `/` is admitted only BETWEEN TWO WORD CHARACTERS - `a */b/c/* d` is claimed and `x *///* y` stays bold. The opener's `(?!\\*[\\s*])` is what pays for it. Over all 21,844 documents `x */ body /* y` for every body of up to seven characters drawn from `* / space a`, judged by asking whether the engine renders the WHOLE body as one combined run: strict was 440 over-coloured and 603 missed; admitting any non-closing slash is 5,091 and 161; the word-flanked slash alone is 510 and 461; the word-flanked slash with the opener guard is 226 and 461. The temper alone RAISES over-colouring and every shape it adds is one the opener guard refuses - `a */**a/a/* d` and `a */* a/a/* d` are bold runs in the engine - while the guard itself costs nothing, since `missed` is unchanged with and without it. TWO MORE GUARDS THE MIRRORED BRANCH NEEDS AND THE CANONICAL ONE DOES NOT, both found by probing the shapes where one order's delimiters sit inside the other's. A `/` before the opener makes the pair ambiguous with a canonical opener - `a /*/a/* b` is `a <em>*/a</em>* b`, an italic run - so the left guard refuses it as well as a word character and a `*`. And the closer may not follow a `*`: `a */b*/* c` is `a <strong>/b</strong>/* c` and `a */*a*/* b` is `a <strong>/*a</strong>/* b`, both bold runs, while `a */*b/* c` IS combined - so it is the character before the closer that decides, not the one after the opener.",
1305
+ "patterns": [
1306
+ {
1307
+ "match": "(/\\*)(?=\\S)((?:[^*\\n]|\\*(?!/)){1,4096})(?<=\\S)(\\*/)",
1308
+ "captures": {
1309
+ "1": {
1310
+ "name": "punctuation.definition.bold-italic.carve"
1311
+ },
1312
+ "2": {
1313
+ "name": "markup.bold.italic.carve"
1314
+ },
1315
+ "3": {
1316
+ "name": "punctuation.definition.bold-italic.carve"
1317
+ }
1318
+ }
1310
1319
  },
1311
- "3": {
1312
- "name": "punctuation.definition.bold-italic.carve"
1320
+ {
1321
+ "match": "(?<![\\w*/])(\\*/)(?=\\S)(?!\\*[\\s*])((?:[^/\\n]|(?<=\\w)/(?=\\w))+)(?<![\\s*])(/\\*)(?!\\w)",
1322
+ "captures": {
1323
+ "1": {
1324
+ "name": "punctuation.definition.bold-italic.carve"
1325
+ },
1326
+ "2": {
1327
+ "name": "markup.bold.italic.carve"
1328
+ },
1329
+ "3": {
1330
+ "name": "punctuation.definition.bold-italic.carve"
1331
+ }
1313
1332
  }
1314
- }
1333
+ }
1334
+ ]
1315
1335
  },
1316
1336
  "strong": {
1317
1337
  "begin": "(?<![\\w*])(\\*)(?=\\S)",
1318
- "end": "(?<=\\S)(\\*)(?![\\w*])",
1338
+ "end": "(?<=\\S)(\\*)(?![\\w*])|^[ \\t]*$",
1319
1339
  "beginCaptures": {
1320
1340
  "1": {
1321
1341
  "name": "punctuation.definition.bold.carve"
@@ -1326,7 +1346,7 @@
1326
1346
  "name": "punctuation.definition.bold.carve"
1327
1347
  }
1328
1348
  },
1329
- "contentName": "markup.bold.carve",
1349
+ "comment": "A BLANK LINE ENDS THE RUN, because an opener with no closer must not colour the rest of the file (carve-grammars#393). This is the ONLY bare inline rule spelled begin/end rather than as one `match`, and it has to be: a TextMate `match` cannot cross a line break at all - measured on a two-rule grammar - while the engine does read `a *b` over `c* d` as one bold run, which this rule gets right and its four `match` siblings get wrong. So the two properties trade off, and the trade is what the blank line bounds: without it `a *b c` scoped the next paragraph and the heading after it, and now it stops at the paragraph. WHAT IT DOES NOT FIX: a genuinely unclosed opener still colours to the end of its own paragraph, because `begin` cannot see whether a closer exists on a later line and TextMate offers no way to ask. Pinned in tests/bold-unclosed-test.js rather than left silent. MEASURE THIS RULE WITH THE RAW TOKENIZER: Shiki reports the fix as having no effect at all (carve-grammars#395), so a test written against it would call a correct grammar broken.", "contentName": "markup.bold.carve",
1330
1350
  "patterns": [
1331
1351
  {
1332
1352
  "include": "#braced_comment"
@@ -1334,6 +1354,7 @@
1334
1354
  ]
1335
1355
  },
1336
1356
  "italic": {
1357
+ "comment": "ONE `match` AND NOT `begin`/`end`, DELIBERATELY, and the trade was measured rather than assumed (carve-grammars#397). A TextMate `match` cannot cross a line break, so this rule does not see `a /b` over `c/ d` as one run the way the engine does, and it cannot hold a delimiter standing after a space either. `begin`/`end` buys both - and opens eagerly, because `begin` cannot see whether a closer exists on a later line, so an opener with no closer colours to the end of its paragraph the way `strong` still does. Converted and swept over the 1,545-document corpus, all four bare rules at once: documents holding a run this grammar MISSES fell from 13 to 9, and documents where it claims a run the document does NOT hold rose from 3 to 35. The whole corpus contains two runs that span a line break. A false run is the worse direction - carve-grammars#324 and #325 both turned on that - so the `match` spelling stays and the missed multi-line run is the price. Narrowing the openers first (`(?<![\\\\w:/])` for the URL, `(?!_)` for the dunder, `(?!>)` for the arrow) took the false count from 40 to 35 and no further.",
1337
1358
  "match": "(?<![\\w/])(/)(?=\\S)([^/\\n]+?)(?<=\\S)(/)(?![\\w/])",
1338
1359
  "captures": {
1339
1360
  "1": {
@@ -1348,6 +1369,7 @@
1348
1369
  }
1349
1370
  },
1350
1371
  "underline": {
1372
+ "comment": "ONE `match` AND NOT `begin`/`end`, deliberately: converting all four bare rules took corpus documents carrying a FALSE run from 3 to 35 to recover 4 missed ones. The measurement and the reasoning are on `italic` (carve-grammars#397).",
1351
1373
  "match": "(?<![\\w_])(_)(?=\\S)([^_\\n]+?)(?<=\\S)(_)(?![\\w_])",
1352
1374
  "captures": {
1353
1375
  "1": {
@@ -1362,6 +1384,7 @@
1362
1384
  }
1363
1385
  },
1364
1386
  "strike": {
1387
+ "comment": "ONE `match` AND NOT `begin`/`end`, deliberately: converting all four bare rules took corpus documents carrying a FALSE run from 3 to 35 to recover 4 missed ones. The measurement and the reasoning are on `italic` (carve-grammars#397).",
1365
1388
  "match": "(?<![\\w~])(~)(?=\\S)([^~\\n]+?)(?<=\\S)(~)(?![\\w~])",
1366
1389
  "captures": {
1367
1390
  "1": {
@@ -1376,7 +1399,7 @@
1376
1399
  }
1377
1400
  },
1378
1401
  "highlight": {
1379
- "comment": "A BARE `=` THAT BEGINS OR ENDS A SMART-TYPOGRAPHY PATTERN IS NOT A HIGHLIGHT OPENER (grammar.ebnf, Inline parsing precedence: \"a delimiter that begins a multi-char smart-typography pattern: `=>` is the arrow, never a highlight opener - the pattern is consumed first\"). Corpus 386's third paragraph is the shape it costs: `Not an arrow: key => value stays literal, and p <= q is a comparison.` renders with NO mark, and the opener `=` of `=>` reaching the closer `=` of `<=` scoped the whole 68-character sentence as a highlight (carve-grammars#325). That is worse than leaving a run uncoloured, because the output then claims the document holds a construct it does not - the same reason carve-grammars#324 gave for the arrows. THE GUARD IS ONLY WHAT THIS GRAMMAR MEASURABLY NEEDS. The opener may not be FOLLOWED by `>`, which is `=>` and the tail of `==>`; and it may not FOLLOW `!`, which is the `!=` comparison. It needs no guard against `<` or `>` behind it, and carried one until each character was reverted and measured: vscode-textmate resolves by POSITION, so on `a <=b c= d` the typography rule's `<=` starts one column before the highlight rule's `=` and takes it. `!=` is the comparison that rule does NOT carry (its alternation is arrows, dashes and `<=`/`>=`), which is why `!` is the one character here that changes a measurement. A leading `=` needs no guard either - the content class already refuses one. THE CLOSER IS DELIBERATELY NOT GUARDED, and the asymmetry is the engine's: once a highlight is OPEN the closer wins over the pattern, so `x =y z<= w` marks `y z<` and `x =y z=> w` marks `y z`. Measured against the pinned engine, both. ONE SHAPE IT COSTS: an autolink or a cross-reference glued straight to a highlight (`<https://e.example>=hi=`) really does mark, because there the `>` closes the link rather than opening a comparison - a flat token map cannot tell the two apart within a fixed-width lookbehind, and this takes the under-colouring side of that trade on the ticket's own reasoning.",
1402
+ "comment": "A BARE `=` THAT BEGINS OR ENDS A SMART-TYPOGRAPHY PATTERN IS NOT A HIGHLIGHT OPENER (grammar.ebnf, Inline parsing precedence: \"a delimiter that begins a multi-char smart-typography pattern: `=>` is the arrow, never a highlight opener - the pattern is consumed first\"). Corpus 386's third paragraph is the shape it costs: `Not an arrow: key => value stays literal, and p <= q is a comparison.` renders with NO mark, and the opener `=` of `=>` reaching the closer `=` of `<=` scoped the whole 68-character sentence as a highlight (carve-grammars#325). That is worse than leaving a run uncoloured, because the output then claims the document holds a construct it does not - the same reason carve-grammars#324 gave for the arrows. THE GUARD IS ONLY WHAT THIS GRAMMAR MEASURABLY NEEDS. The opener may not be FOLLOWED by `>`, which is `=>` and the tail of `==>`. It needs no guard against `<`, `>` or `!` behind it, and carried one for each until it was reverted and measured: vscode-textmate resolves by POSITION, so on `a <=b c= d` the typography rule's `<=` starts one column before the highlight rule's `=` and takes it. `!` was the last of the three to go, and it went in carve-grammars#374 rather than with the other two: while the typography alternation carried no `!=`, nothing reached `a !=b c= d` before this rule and removing the guard scoped `b c` as a highlight. The alternation carries `!=` now, so the comparison starts one column early like the other two and the guard stops changing any answer. What holds the shape is the pair that measures it - the `!=` row in tests/lib/constructs.js and `a !=b c= d` in tests/highlight-opener-test.js - rather than a lookbehind nobody can see working. Removing it also stopped an under-colouring a lookbehind cannot avoid: `a \\!=b c= d` escapes the `!`, so the `=` after it is an ordinary opener and the engine marks `b c`; this grammar refused it while the guard was here. The Prism grammar still does, and still needs its own guard because Prism applies tokens IN ORDER rather than by position - carve-grammars#380. A leading `=` needs no guard either - the content class already refuses one. THE CLOSER IS DELIBERATELY NOT GUARDED AGAINST SMART TYPOGRAPHY, and the asymmetry is the engine's: once a highlight is OPEN the closer wins over the pattern, so `x =y z<= w` marks `y z<` and `x =y z=> w` marks `y z`. Measured against the pinned engine, both. AN ESCAPE IS A DIFFERENT QUESTION (carve-grammars#385): an escaped `=` is not a delimiter at all, so `x =\\= y` renders `x == y` with no mark and this rule closed on it. THE BODY CONSUMES AN ESCAPE AS A PAIR (`\\\\.`) and refuses a bare backslash, which settles both directions at once - the escaped `=` is eaten before the closer can see it, and a body that could not hold one would stop there and never reach the real closer past it, so `a =b c\\= d= e`, which the engine marks, would have gone from wrongly-coloured to uncoloured. The closer needs no guard of its own, and that is a consequence rather than a simplification: the body can only get past a backslash by taking the pair, so the position immediately before an escaped `=` is never a body end. #385 COUNTED THE BACKSLASH RUN IN A LOOKBEHIND INSTEAD, and carve-grammars#390 replaced it for two measured reasons. The IDE's TextMate engine refuses a variable-length lookbehind by not matching ANYTHING - no error, the rule simply goes inert and `x =b= y` loses its mark too (markup-carve/intellij-carve#117), which made this file unusable in a TextMate engine Carve ships a plugin for. And every bound is reachable: at `{0,32}` the guard stopped seeing the run at 66 backslashes and refused a highlight the engine marks. A pair has no bound. THREE SHAPES STAY WRONG, pinned in tests/highlight-escape-test.js - `x =<== y`, `x =!== y` and `x =a== y`, a closer followed by another `=`, which the engine marks and `(?![=\\w])` refuses; widening that guard means letting the body hold its own delimiter. ONE SHAPE IT COSTS: an autolink or a cross-reference glued straight to a highlight (`<https://e.example>=hi=`) really does mark, because there the `>` closes the link rather than opening a comparison - a flat token map cannot tell the two apart within a fixed-width lookbehind, and this takes the under-colouring side of that trade on the ticket's own reasoning. ONE `match` AND NOT `begin`/`end`, deliberately: converting all four bare rules took corpus documents carrying a FALSE run from 3 to 35 to recover 4 missed ones. The measurement and the reasoning are on `italic` (carve-grammars#397).",
1380
1403
  "patterns": [
1381
1404
  {
1382
1405
  "match": "(\\{=)(?=\\S)([^\\n]*?)(=\\})",
@@ -1393,7 +1416,7 @@
1393
1416
  }
1394
1417
  },
1395
1418
  {
1396
- "match": "(?<![=!\\w])(=)(?=\\S)(?!>)([^=\\n]+?)(?<=\\S)(=)(?![=\\w])",
1419
+ "match": "(?<![=\\w])(=)(?=\\S)(?!>)((?:\\\\.|[^=\\n\\\\])+?)(?<=\\S)(=)(?![=\\w])",
1397
1420
  "captures": {
1398
1421
  "1": {
1399
1422
  "name": "punctuation.definition.highlight.carve"
@@ -1510,7 +1533,7 @@
1510
1533
  }
1511
1534
  },
1512
1535
  "smart_typography": {
1513
- "match": "(<-->|<--|-->|<=>|<==|==>|---|--|\\.\\.\\.|<->|<-|->|<=|>=)",
1536
+ "match": "(\\{--\\}|<-->|<--|-->|<=>|<==|==>|---|--|\\.\\.\\.|<->|<-|->|<=|>=|!=|\\+-|\\(c\\)|\\(r\\)|\\(tm\\))",
1514
1537
  "name": "constant.other.smart_typography.carve"
1515
1538
  },
1516
1539
  "math_inline": {
@@ -32,6 +32,7 @@ import { CarveMath } from './extensions/carve-math.js';
32
32
  import { CarveFootnoteDefinition } from './extensions/carve-footnote-definition.js';
33
33
  import { CarveEmbed } from './extensions/carve-embed.js';
34
34
  import { CarveAbbreviation } from './extensions/carve-abbreviation.js';
35
+ import { CarveAbbreviationDefinition } from './extensions/carve-abbreviation-definition.js';
35
36
  import { CarveDefinitionList, CarveDefinitionTerm, CarveDefinitionDescription } from './extensions/carve-definition-list.js';
36
37
  import { CarveUnsupported } from './extensions/carve-unsupported.js';
37
38
  import { CarveUnsupportedInline } from './extensions/carve-unsupported-inline.js';
@@ -586,6 +587,19 @@ export const CarveKit = Extension.create({
586
587
  'data-checked': attributes.checked,
587
588
  }),
588
589
  },
590
+ // The four non-space task states (`-` `_` `>` `?`)
591
+ // the engine emits as `data-task-state`. Null for a
592
+ // plain `[ ]`/`[x]` item, whose state `checked` holds.
593
+ // Without this, a load/save cycle collapsed every such
594
+ // item back to `[ ]` (markup-carve/carve-grammars#371).
595
+ carveTaskState: {
596
+ default: null,
597
+ keepOnSplit: false,
598
+ parseHTML: element => element.getAttribute('data-task-state') || null,
599
+ renderHTML: attributes => (
600
+ attributes.carveTaskState ? { 'data-task-state': attributes.carveTaskState } : {}
601
+ ),
602
+ },
589
603
  // A task item takes a marker attribute the same way a
590
604
  // plain item does: `-{.c} [ ] text`.
591
605
  id: { default: null },
@@ -678,6 +692,9 @@ export const CarveKit = Extension.create({
678
692
  if (this.options.carveAbbreviation !== false) {
679
693
  extensions.push(CarveAbbreviation.configure(this.options.carveAbbreviation ?? {}));
680
694
  }
695
+ if (this.options.carveAbbreviationDefinition !== false) {
696
+ extensions.push(CarveAbbreviationDefinition.configure(this.options.carveAbbreviationDefinition ?? {}));
697
+ }
681
698
 
682
699
  // Definition list nodes (maps to : term with definition)
683
700
  if (this.options.definitionList !== false) {
@@ -129,20 +129,22 @@ function opaqueDocument(source, ctx = null) {
129
129
 
130
130
  function sourceEnvelope(doc, sourceLayout, ctx = null) {
131
131
  // The rich projection is kept, but writing it back would not reproduce the
132
- // document, so the source rides along and the serializer replays it while
133
- // the document is untouched. The FIRST EDIT invalidates the fingerprint and
134
- // the projection becomes what is written - which is why this is reported.
132
+ // document, so the source and canonical projection ride along as the two
133
+ // merge bases. Edits are applied to the authored branch, preserving layout
134
+ // outside the changed region.
135
135
  record(ctx, 'preserved', 'document',
136
- 'the rich projection is not write-identical; the source envelope carries the document until it is edited');
136
+ 'the rich projection is not write-identical; edits are merged into its authored source envelope');
137
137
 
138
138
  const clean = { ...doc };
139
139
  delete clean.attrs;
140
+ const projectedSource = serializeToCarve(clean);
140
141
  return {
141
142
  ...clean,
142
143
  attrs: {
143
144
  carveSource: sourceLayout.source,
144
145
  carveFingerprint: pmFingerprint(clean),
145
146
  carveSourceLayout: JSON.stringify(sourceLayout),
147
+ carveProjectedSource: projectedSource,
146
148
  },
147
149
  };
148
150
  }
@@ -450,7 +452,14 @@ function convertBlock(node, ctx) {
450
452
  ...(typeof node.tight === 'boolean' ? { attrs: { carveTight: node.tight } } : {}),
451
453
  content: (node.items || []).map((it) => ({
452
454
  type: 'taskItem',
453
- attrs: { checked: !!it.checked, ...(convertAttrs(it.attrs) || {}) },
455
+ attrs: {
456
+ checked: !!it.checked,
457
+ // The engine carries the four non-space task states
458
+ // (`-` `_` `>` `?`) in `taskState`; absent for the
459
+ // plain `[ ]`/`[x]`, whose state `checked` already holds.
460
+ ...(it.taskState != null ? { carveTaskState: it.taskState } : {}),
461
+ ...(convertAttrs(it.attrs) || {}),
462
+ },
454
463
  content: convertBlocks(it.children || [], ctx),
455
464
  })),
456
465
  };
@@ -545,7 +554,10 @@ function convertBlock(node, ctx) {
545
554
  // does not faithfully represent. Throwing routes the category to SKIP.
546
555
  case 'abbreviation-def':
547
556
  case 'abbreviation_def':
548
- return unsupported('abbreviation-def', node, ctx);
557
+ return {
558
+ type: 'carveAbbreviationDefinition',
559
+ attrs: { abbr: node.abbr || '', expansion: node.expansion || '' },
560
+ };
549
561
  case 'raw-block':
550
562
  case 'raw_block':
551
563
  return {
@@ -684,7 +696,9 @@ function convertAdmonition(node, ctx) {
684
696
  type: 'carveTab',
685
697
  attrs: {
686
698
  label: node.label ?? inlinePlainText(node.title || []),
687
- selected: !!node.attrs?.keyValues?.selected,
699
+ // Bare attributes are represented with an empty-string value,
700
+ // so truthiness would turn `{selected}` back off.
701
+ selected: Object.hasOwn(node.attrs?.keyValues ?? {}, 'selected'),
688
702
  },
689
703
  content: convertBlocks(node.children || [], ctx),
690
704
  };
@@ -708,7 +722,23 @@ function convertDefinitionList(node, ctx) {
708
722
  // The list's OWN attribute run. Nothing read it, so `{loose}` above a
709
723
  // definition list was gone before the projection was even mounted - the
710
724
  // node arrived with no attrs at all (markup-carve/carve-grammars#344).
711
- const attrs = convertAttrs(node.attrs);
725
+ // Looseness is CONTENT here - a loose description renders its body in a
726
+ // `<p>`, a tight one bare - and PART 9 section 17 L7 leaves `{loose}` as
727
+ // its only spelling, since no blank line can carry it.
728
+ //
729
+ // The engine used to leave it in the attribute run and now reads it into
730
+ // the node's own `loose` flag, so reading `attrs` alone dropped it. Folding
731
+ // it back into the run rather than inventing a second slot means the
732
+ // existing value-less-attribute path writes it, and a mount keeps it in
733
+ // `carveKeyValues`, which is already declared.
734
+ const runAttrs = node.loose === true
735
+ ? {
736
+ ...(node.attrs || {}),
737
+ keyValues: { loose: '', ...(node.attrs?.keyValues || {}) },
738
+ order: ['loose', ...((node.attrs?.order || []).filter((k) => k !== 'loose'))],
739
+ }
740
+ : node.attrs;
741
+ const attrs = convertAttrs(runAttrs);
712
742
  for (const item of node.items || []) {
713
743
  for (const term of item.terms || []) {
714
744
  content.push({ type: 'definitionTerm', content: convertInline(term, ctx) });
@@ -908,6 +938,24 @@ function convertInlineNode(node, marks, ctx) {
908
938
  ? [{ type: 'text', text: node.value, ...(marks.length ? { marks } : {}) }]
909
939
  : [];
910
940
 
941
+ // Smart punctuation is a TYPOGRAPHIC reading of characters the author
942
+ // typed, so the characters are what the editor carries: `node.value`
943
+ // is the authored spelling (`--`, `...`, `"`) and `node.glyph` the
944
+ // rendered one. Writing the glyph would bake the engine's reading into
945
+ // the source and change the document; writing the value re-parses to
946
+ // the same node, so the round trip is byte-identical.
947
+ //
948
+ // Left unmapped, the converter THREW on ordinary prose - an em dash, an
949
+ // ellipsis or a typographic quote was enough - so any such document
950
+ // took the whole-document fallback instead of being editable.
951
+ case 'smart_punctuation':
952
+ record(ctx, 'degraded', 'smart_punctuation',
953
+ 'the typographic reading is re-derived on parse; the authored characters survive as text');
954
+
955
+ return node.value
956
+ ? [{ type: 'text', text: node.value, ...(marks.length ? { marks } : {}) }]
957
+ : [];
958
+
911
959
  case 'soft-break':
912
960
  case 'soft_break':
913
961
  // A NEWLINE, not a space. A soft break is a line break the author
@@ -1120,6 +1168,24 @@ function convertInlineNode(node, marks, ctx) {
1120
1168
  attrs: { content: node.content || '', delimited: Boolean(node.delimited) },
1121
1169
  }];
1122
1170
 
1171
+ case 'abbreviation':
1172
+ // A definition-resolved abbreviation is still editable text. The
1173
+ // mark carries its expansion for `<abbr title>`, while `resolved`
1174
+ // tells the writer the author used the bare term rather than an
1175
+ // explicit semantic span.
1176
+ if (marks.some((mark) => mark.type === 'carveAbbreviation')) {
1177
+ return [{ type: 'text', text: node.abbr || '', ...(marks.length ? { marks } : {}) }];
1178
+ }
1179
+ return [{ type: 'text', text: node.abbr || '', marks: [...marks, {
1180
+ type: 'carveAbbreviation',
1181
+ attrs: { title: node.expansion || '', resolved: true },
1182
+ }] }];
1183
+
1184
+ case 'caption_number':
1185
+ // `#` is the authored placeholder. Its rendered number is a
1186
+ // resolution artifact and therefore does not belong on the wire.
1187
+ return [{ type: 'text', text: '#', ...(marks.length ? { marks } : {}) }];
1188
+
1123
1189
  default: {
1124
1190
  const markType = INLINE_MARKS[node.type];
1125
1191
  if (markType) {
@@ -1164,8 +1230,15 @@ function convertSpan(node, marks, ctx) {
1164
1230
  // Only a lone abbr attribute round-trips through carveAbbreviation; any
1165
1231
  // companion id/class/other key would be dropped.
1166
1232
  const extraKeys = Object.keys(a.keyValues).filter((k) => k !== 'abbr');
1167
- if (extraKeys.length || a.id || (a.classes && a.classes.length)) throw new UnsupportedNodeError('span-abbr-plus-attrs', node);
1168
- return descend(node, [...marks, { type: 'carveAbbreviation', attrs: { title: a.keyValues.abbr } }], ctx);
1233
+ const hasCompanions = extraKeys.length || a.id || (a.classes && a.classes.length);
1234
+ const abbrMark = {
1235
+ type: 'carveAbbreviation',
1236
+ attrs: {
1237
+ title: a.keyValues.abbr,
1238
+ ...(hasCompanions ? { resolved: false, ...(convertAttrs(a) || {}) } : {}),
1239
+ },
1240
+ };
1241
+ return descend(node, [...marks, abbrMark], ctx);
1169
1242
  }
1170
1243
  const attrs = convertAttrs(a) || {};
1171
1244
  return descend(node, [...marks, { type: 'carveSpan', attrs }], ctx);
@@ -0,0 +1,22 @@
1
+ import { Node, mergeAttributes } from '@tiptap/core';
2
+
3
+ /** An authored document-level abbreviation definition: `*[HTML]: expansion`. */
4
+ export const CarveAbbreviationDefinition = Node.create({
5
+ name: 'carveAbbreviationDefinition',
6
+ group: 'block',
7
+ atom: true,
8
+ addAttributes() {
9
+ return {
10
+ abbr: { default: '' },
11
+ expansion: { default: '' },
12
+ };
13
+ },
14
+ parseHTML() { return [{ tag: 'div[data-carve-abbreviation-definition]' }]; },
15
+ renderHTML({ HTMLAttributes, node }) {
16
+ return ['div', mergeAttributes(HTMLAttributes, {
17
+ 'data-carve-abbreviation-definition': 'true',
18
+ }), `${node.attrs.abbr}: ${node.attrs.expansion}`];
19
+ },
20
+ });
21
+
22
+ export default CarveAbbreviationDefinition;
@@ -1,4 +1,5 @@
1
1
  import { Mark, mergeAttributes } from '@tiptap/core';
2
+ import { attributeSlots } from './carve-attribute-slots.js';
2
3
 
3
4
  /**
4
5
  * Carve Abbreviation extension for Tiptap
@@ -32,6 +33,8 @@ export const CarveAbbreviation = Mark.create({
32
33
  return { title: attributes.title };
33
34
  },
34
35
  },
36
+ resolved: { default: false, rendered: false },
37
+ ...attributeSlots(['title', 'data-carve-abbreviation']),
35
38
  };
36
39
  },
37
40
 
@@ -1,6 +1,6 @@
1
1
  import { Extension } from '@tiptap/core';
2
2
 
3
- /** Lossless source envelope for a structured document that has not been edited. */
3
+ /** Merge base for preserving authored source layout around structured edits. */
4
4
  export const CarveSourcePreservation = Extension.create({
5
5
  name: 'carveSourcePreservation',
6
6
  addGlobalAttributes() {
@@ -8,9 +8,10 @@ export const CarveSourcePreservation = Extension.create({
8
8
  {
9
9
  types: ['doc'],
10
10
  attributes: {
11
- carveSource: { default: null, rendered: false },
12
- carveFingerprint: { default: null, rendered: false },
13
- carveSourceLayout: { default: null, rendered: false },
11
+ carveSource: { default: null, rendered: false },
12
+ carveFingerprint: { default: null, rendered: false },
13
+ carveSourceLayout: { default: null, rendered: false },
14
+ carveProjectedSource: { default: null, rendered: false },
14
15
  },
15
16
  },
16
17
  {
@@ -13,6 +13,7 @@ export { CarveFootnoteDefinition } from './carve-footnote-definition.js';
13
13
  export { CarveMath } from './carve-math.js';
14
14
  export { CarveEmbed } from './carve-embed.js';
15
15
  export { CarveAbbreviation } from './carve-abbreviation.js';
16
+ export { CarveAbbreviationDefinition } from './carve-abbreviation-definition.js';
16
17
  export { CarveDefinitionList, CarveDefinitionTerm, CarveDefinitionDescription } from './carve-definition-list.js';
17
18
  export { CarveUnsupported } from './carve-unsupported.js';
18
19
  export { CarveUnsupportedInline } from './carve-unsupported-inline.js';
package/tiptap/index.d.ts CHANGED
@@ -25,6 +25,8 @@ export const CarveCriticComment: Mark;
25
25
  export const CarveDiv: Node;
26
26
  export const CarveMath: Node;
27
27
  export const CarveFootnoteDefinition: Node;
28
+ export const CarveAbbreviation: Mark;
29
+ export const CarveAbbreviationDefinition: Node;
28
30
  export const CarveMention: Node;
29
31
  export const CarveTag: Node;
30
32
  export const CarveUnsupported: Node;
package/tiptap/index.js CHANGED
@@ -47,6 +47,8 @@ export { CarveCriticComment } from './extensions/carve-critic-comment.js';
47
47
  export { CarveDiv } from './extensions/carve-div.js';
48
48
  export { CarveMath } from './extensions/carve-math.js';
49
49
  export { CarveFootnoteDefinition } from './extensions/carve-footnote-definition.js';
50
+ export { CarveAbbreviation } from './extensions/carve-abbreviation.js';
51
+ export { CarveAbbreviationDefinition } from './extensions/carve-abbreviation-definition.js';
50
52
  export { CarveKeymap } from './extensions/carve-keymap.js';
51
53
  export { CarveMention, CarveTag } from './extensions/carve-mention.js';
52
54
  export { CarveInlineExtension } from './extensions/carve-inline-extension.js';
@@ -92,6 +92,7 @@
92
92
  "carveAttrOrder": "the order the run's slots were WRITTEN in, as the AST records it: `#id`, `.class`, and each key by name",
93
93
  "carveKeyValues": "authored marker key=values",
94
94
  "checked": "task item state, taskItem only",
95
+ "carveTaskState": "non-space task marker (`-` `_` `>` `?`), taskItem only; null for a plain `[ ]`/`[x]` item",
95
96
  "class": "authored marker classes",
96
97
  "id": "authored marker id"
97
98
  }
@@ -500,10 +501,17 @@
500
501
  "id": "authored id",
501
502
  "key": "the citation key as written, without the `@`; the same string `citation.key` carries at the use site"
502
503
  }
504
+ },
505
+ "abbreviation_def": {
506
+ "kind": "node",
507
+ "pm": "carveAbbreviationDefinition",
508
+ "attrs": {
509
+ "abbr": "the defined term",
510
+ "expansion": "the authored expansion"
511
+ }
503
512
  }
504
513
  },
505
514
  "unmapped": {
506
- "abbreviation_def": "abbreviation definitions ride on the doc node's attrs",
507
515
  "caption_number": "numbered captions are a resolution artifact, not editor content",
508
516
  "citation": "a citation is one item in citation_group and rides in carveCitation's items attribute rather than becoming its own ProseMirror node",
509
517
  "raw_text": "raw text is the payload of a raw block, not a node an editor holds",
@@ -23,6 +23,7 @@
23
23
  * space (e.g. literal `**` immediately followed by bold text) - the run
24
24
  * merges into a longer literal delimiter run on reparse.
25
25
  */
26
+ import { diff3Merge } from 'node-diff3';
26
27
 
27
28
  /**
28
29
  * Serialize a Tiptap/ProseMirror JSON document to Carve markup
@@ -167,6 +168,17 @@ export function serializeToCarve(doc) {
167
168
  const clean = { ...doc };
168
169
  delete clean.attrs;
169
170
  if (pmFingerprint(clean) === preservedFingerprint) return preservedSource;
171
+ const projectedSource = doc?.attrs?.carveProjectedSource;
172
+ if (typeof projectedSource === 'string') {
173
+ // The authored source and the editable projection are two branches
174
+ // from the same canonical baseline. Merge the editor's changes
175
+ // into the authored branch so untouched columns, blank ownership,
176
+ // delimiter choices and marker placement survive. Where both sides
177
+ // changed the same characters, the editor wins: the user changed
178
+ // that construct and canonical Carve is safer than stale source.
179
+ const currentProjection = serializeToCarve(clean);
180
+ return mergeAuthoredSource(preservedSource, projectedSource, currentProjection);
181
+ }
170
182
  }
171
183
  // A whole-document fallback is already exact Carve source. Sending it
172
184
  // through the normal block joiner and edge trimmer would corrupt precisely
@@ -241,12 +253,26 @@ export function serializeToCarve(doc) {
241
253
  // is present: a list whose items hold one paragraph each is
242
254
  // loose or tight purely by the blank lines between them, which
243
255
  // no amount of looking at the items can recover.
244
- const isLoose = node.attrs?.carveTight === false || (node.content || []).some((item) => {
256
+ const multiBlockItem = (node.content || []).some((item) => {
245
257
  const blocks = (item.content || []).filter(
246
258
  (b) => !['bulletList', 'orderedList', 'taskList'].includes(b.type),
247
259
  );
248
260
  return blocks.length > 1;
249
261
  });
262
+ const isLoose = node.attrs?.carveTight === false || multiBlockItem;
263
+ // Blank lines between items are how looseness is normally
264
+ // spelled, and they need two items to sit between. A loose list
265
+ // of ONE item whose item holds one block has nowhere to put
266
+ // them, so it reads back tight and the round trip changes the
267
+ // document - which is why the attribute is written instead.
268
+ //
269
+ // This only became reachable when the engine consumed `{loose}`
270
+ // into the list's own `tight`. While it stayed an ordinary
271
+ // attribute, serializeAttributes wrote it and nothing was lost.
272
+ if (node.attrs?.carveTight === false && !multiBlockItem
273
+ && (node.content || []).length < 2) {
274
+ output += indent + '{loose}\n';
275
+ }
250
276
  let num = node.attrs?.start || 1;
251
277
  // `carveOlType` carries the style from the Carve AST; `type` is what Tiptap's own
252
278
  // OrderedList records when the editor is seeded from rendered
@@ -267,7 +293,13 @@ export function serializeToCarve(doc) {
267
293
  num++;
268
294
  } else if (node.type === 'taskList') {
269
295
  marker = '-';
270
- taskBox = '[' + (item.attrs?.checked ? 'x' : ' ') + '] ';
296
+ // `checked` wins: toggling the stock TaskItem checkbox
297
+ // updates only `checked` and leaves `carveTaskState`
298
+ // intact, so a user who checks a `[-]` item means `[x]`.
299
+ // An unchecked item keeps its non-space state, or a
300
+ // plain space when it has none.
301
+ const taskState = item.attrs?.checked ? 'x' : (item.attrs?.carveTaskState || ' ');
302
+ taskBox = '[' + taskState + '] ';
271
303
  } else {
272
304
  marker = '-';
273
305
  }
@@ -275,7 +307,7 @@ export function serializeToCarve(doc) {
275
307
  // (`- {.c} item`) is a different document: the brace is then
276
308
  // content, either literal text or a block-attribute line for
277
309
  // what follows (corpus 90-list-item-attributes-7, 172).
278
- const markerAttrs = serializeAttributes(item.attrs, ['checked']);
310
+ const markerAttrs = serializeAttributes(item.attrs, ['checked', 'carveTaskState']);
279
311
  const prefix = marker + markerAttrs + ' ';
280
312
  output += indent + prefix + taskBox;
281
313
  // The content column is measured from the prefix ACTUALLY
@@ -558,6 +590,10 @@ export function serializeToCarve(doc) {
558
590
  break;
559
591
  }
560
592
 
593
+ case 'carveAbbreviationDefinition':
594
+ output += `*[${node.attrs?.abbr || ''}]: ${node.attrs?.expansion || ''}\n`;
595
+ break;
596
+
561
597
  case 'carveCitationDefinition': {
562
598
  // `[@key]: {metadata} entry`. The metadata block LEADS the
563
599
  // entry text here, unlike the link reference definition above
@@ -664,6 +700,10 @@ export function serializeToCarve(doc) {
664
700
  }
665
701
  }
666
702
 
703
+ // The column a description's continuation blocks sit at, set by the width
704
+ // of the canonical `: ` separator this serializer writes.
705
+ const DEFINITION_CONTENT_INDENT = ' ';
706
+
667
707
  function serializeDefinitionList(dl) {
668
708
  const children = dl.content || [];
669
709
  // The list's own attribute run, on its own line above the first term.
@@ -684,16 +724,45 @@ export function serializeToCarve(doc) {
684
724
  output += ':: ' + serializeInline(child.content) + '\n';
685
725
  afterDescription = false;
686
726
  } else if (child.type === 'definitionDescription') {
687
- (child.content || []).forEach(block => {
688
- if (block.type === 'paragraph') {
689
- output += ': ' + serializeInline(block.content) + '\n';
690
- } else {
691
- // For other block types, serialize with indentation.
692
- const blockText = serializeNodeToString(block);
693
- blockText.split('\n').filter(l => l).forEach(line => {
694
- output += ': ' + line + '\n';
727
+ // ONE description, however many blocks it holds. Every block
728
+ // used to get its own `: ` marker, which spells a NEW
729
+ // description each time - a two-paragraph definition came back
730
+ // as two definitions of the same term. A continuation belongs
731
+ // at the description's content column, which the canonical
732
+ // `: ` separator puts at 2 (PART 9 section 17: the separator's
733
+ // width sets the column).
734
+ const blocks = child.content || [];
735
+ // A bare `:` is not an empty description: it is prose, and
736
+ // omitting the line removes the description altogether. Carve
737
+ // gives empty bodies an explicit canonical spelling so their
738
+ // boundary survives a rich-editor round trip.
739
+ if (blocks.length === 0) {
740
+ output += ': {empty}\n';
741
+ // Unlike a body-bearing pair, the canonical empty form is
742
+ // glued to the next term; a blank would end the list.
743
+ afterDescription = false;
744
+ return;
745
+ }
746
+ blocks.forEach((block, i) => {
747
+ const text = block.type === 'paragraph'
748
+ ? serializeInline(block.content)
749
+ : serializeNodeToString(block).replace(/\n+$/, '');
750
+ const lines = text.split('\n');
751
+ if (i === 0) {
752
+ // The first line rides the marker; the rest are already
753
+ // below it and only need the column.
754
+ output += ': ' + lines[0] + '\n';
755
+ lines.slice(1).forEach(line => {
756
+ output += (line ? DEFINITION_CONTENT_INDENT + line : '') + '\n';
695
757
  });
758
+
759
+ return;
696
760
  }
761
+ // A blank line, or the block would join the one above it.
762
+ output += '\n';
763
+ lines.forEach(line => {
764
+ output += (line ? DEFINITION_CONTENT_INDENT + line : '') + '\n';
765
+ });
697
766
  });
698
767
  afterDescription = true;
699
768
  }
@@ -904,7 +973,16 @@ export function serializeToCarve(doc) {
904
973
  //
905
974
  // Widening this set is a measurement, not a judgement: add a case that
906
975
  // loses its paragraph without the escape, then add the character.
907
- const CONTINUATION_BLOCK_OPENER = /\n([>#])/g;
976
+ // Every shape that OPENS a block at column 0, so a soft-break line holding
977
+ // one stops being text when it is written back there.
978
+ //
979
+ // Was `>` and `#` only, which left seven others leaking. `1. outer` with a
980
+ // lazy ` 1. inner` under it came back as `1. outer` / `1. inner` - two
981
+ // items where the source had one, measured against the engine rather than
982
+ // reasoned about. A lookahead rather than a capture, because the fix is to
983
+ // insert a space and never to rewrite the opener.
984
+ const CONTINUATION_BLOCK_OPENER =
985
+ /\n(?=[>#|]|[-*][ \t]|-{3,}|:{2,}|(?:[0-9]{1,9}|[A-Za-z])[.)][ \t])/g;
908
986
 
909
987
  function escapeContinuationOpeners(text) {
910
988
  // A single SPACE, not a backslash. Both keep the line as text - the
@@ -918,7 +996,7 @@ export function serializeToCarve(doc) {
918
996
  // One space is below every item's content column (the shallowest is 2,
919
997
  // for `- `), so the line stays a lazy continuation rather than becoming
920
998
  // the block it would be at that column.
921
- return text.replace(CONTINUATION_BLOCK_OPENER, (match, opener) => '\n ' + opener);
999
+ return text.replace(CONTINUATION_BLOCK_OPENER, '\n ');
922
1000
  }
923
1001
 
924
1002
  function serializeParagraphText(content) {
@@ -941,7 +1019,7 @@ export function serializeToCarve(doc) {
941
1019
  // each such atom on its own (no marks, so this recursion terminates) and
942
1020
  // hand the result to the text path as a verbatim run that still carries
943
1021
  // the marks.
944
- const content = (rawContent || []).map((node) => (
1022
+ const normalized = (rawContent || []).map((node) => (
945
1023
  node && node.type !== 'text' && (node.marks || []).length
946
1024
  ? {
947
1025
  type: 'text',
@@ -951,6 +1029,48 @@ export function serializeToCarve(doc) {
951
1029
  }
952
1030
  : node
953
1031
  ));
1032
+ // ProseMirror splits one marked range whenever a nested mark begins or
1033
+ // ends. Serializing each resulting text node independently repeats the
1034
+ // outer delimiter (`*a *` + `*/b/*` + `* c*`) instead of keeping it open
1035
+ // across the inner span. Collapse runs that share their outermost mark
1036
+ // into one verbatim text node, then recurse over their contents after
1037
+ // removing that mark. Recursion handles arbitrary nesting depth while
1038
+ // leaving the escaping and bare/autolink choices in the existing text
1039
+ // path below.
1040
+ const sameOuterMark = (left, right) => Boolean(left && right
1041
+ && left.type === right.type
1042
+ && pmFingerprint(left.attrs || {}) === pmFingerprint(right.attrs || {}));
1043
+ const groupableDelimitedMarks = new Set([
1044
+ 'bold', 'italic', 'underline', 'strike', 'highlight',
1045
+ 'superscript', 'subscript', 'carveInsert', 'carveDelete',
1046
+ ]);
1047
+ const content = [];
1048
+ for (let index = 0; index < normalized.length;) {
1049
+ const node = normalized[index];
1050
+ const candidateOuter = node?.type === 'text' ? node.marks?.[0] : null;
1051
+ const outer = groupableDelimitedMarks.has(candidateOuter?.type) ? candidateOuter : null;
1052
+ let end = index + 1;
1053
+ while (outer && end < normalized.length) {
1054
+ const candidate = normalized[end];
1055
+ if (candidate?.type !== 'text' || !sameOuterMark(outer, candidate.marks?.[0])) break;
1056
+ end++;
1057
+ }
1058
+ if (outer && end - index > 1) {
1059
+ const inner = normalized.slice(index, end).map((part) => ({
1060
+ ...part,
1061
+ marks: (part.marks || []).slice(1),
1062
+ }));
1063
+ content.push({
1064
+ type: 'text',
1065
+ text: serializeInline(inner),
1066
+ marks: [outer],
1067
+ carveVerbatim: true,
1068
+ });
1069
+ } else {
1070
+ content.push(node);
1071
+ }
1072
+ index = end;
1073
+ }
954
1074
  if (!content) return '';
955
1075
  let result = '';
956
1076
  let resumeDelimitedBold = false;
@@ -1296,7 +1416,12 @@ export function serializeToCarve(doc) {
1296
1416
  // syntax: with that extension enabled the `abbr` attribute is
1297
1417
  // promoted to a real `<abbr title="…">`; without it, it stays a
1298
1418
  // `<span abbr="…">`. (Title escaped like a link title.)
1299
- if (abbr) t = '[' + t + ']{abbr="' + escapeTitle(abbr.attrs?.title || '') + '"}';
1419
+ if (abbr && !abbr.attrs?.resolved) {
1420
+ const attrs = { ...abbr.attrs };
1421
+ attrs.carveKeyValues = { ...(attrs.carveKeyValues || {}), abbr: attrs.title || '' };
1422
+ attrs.carveAttrOrder ||= ['abbr'];
1423
+ t = '[' + t + ']' + serializeAttributes(attrs, ['title', 'resolved'], true);
1424
+ }
1300
1425
 
1301
1426
  result += t;
1302
1427
  } else if (node.type === 'hardBreak') {
@@ -1365,6 +1490,33 @@ export function serializeToCarve(doc) {
1365
1490
  return result;
1366
1491
  }
1367
1492
 
1493
+ function mergeAuthoredSource(authored, baseline, edited) {
1494
+ // Appending and prepending blocks are the most common editor operations.
1495
+ // Handle them without diff alignment: repeated punctuation can otherwise
1496
+ // make a character diff align an untouched delimiter with the new text and
1497
+ // needlessly replace the author's spelling in the original document.
1498
+ const appended = edited.startsWith(baseline) ? edited.slice(baseline.length) : null;
1499
+ const baselineClose = baseline.match(/(?:^|\n)([ \t]*(?:`{3,}|~{3,}|:{3,}))$/)?.[1];
1500
+ const authoredHasClose = !baselineClose || authored.trimEnd().endsWith(baselineClose);
1501
+ if (appended?.startsWith('\n\n') && authoredHasClose) return authored.trimEnd() + appended;
1502
+ if (edited.endsWith(baseline)) return edited.slice(0, -baseline.length) + authored;
1503
+
1504
+ const characters = Math.max(authored.length, baseline.length, edited.length) <= 20_000;
1505
+ const tokens = characters
1506
+ ? (source) => [...source]
1507
+ : (source) => source.match(/[^\n]*\n|[^\n]+$/g) || [];
1508
+ const regions = diff3Merge(tokens(authored), tokens(baseline), tokens(edited), {
1509
+ excludeFalseConflicts: true,
1510
+ });
1511
+ let merged = regions.flatMap((region) => region.ok || region.conflict?.b || []).join('');
1512
+ // Canonical projections omit structural edge whitespace. It is outside the
1513
+ // editable tree, so a content edit must not silently remove the author's
1514
+ // terminal line ending.
1515
+ const ending = authored.match(/\r\n$|[\r\n]$/)?.[0];
1516
+ if (ending && !/[\r\n]$/.test(merged)) merged += ending;
1517
+ return merged;
1518
+ }
1519
+
1368
1520
  /**
1369
1521
  * Escape the "structural" Carve constructs in a text run - the ones whose
1370
1522
  * delimiters are unambiguous regardless of flanking. Used for both plain and
@@ -1384,7 +1536,16 @@ function escapeStructural(text, trailingSafe = false) {
1384
1536
  .replace(/`/g, '\\`')
1385
1537
  .replace(/\[(?=\^)/g, '\\[')
1386
1538
  .replace(/\[(?=[^\]\n]*\][([{:])/g, '\\[')
1387
- .replace(/\{(?=[+\-~#=%])/g, '\\{')
1539
+ // An EMPTY doubled pair is text since carve-js 0.1.5, so escaping the
1540
+ // brace there does not protect a construct - it creates a difference.
1541
+ // `{--}` reaches smart typography and renders an en dash; `\{--}` is
1542
+ // the literal characters, so the escape changed the document. Skip it
1543
+ // for the five markers whose empty pair is text, the same way the
1544
+ // `:name:` rule below only escapes where a symbol would form.
1545
+ //
1546
+ // `%` is NOT among them: `{%%}` is an empty COMMENT and renders to
1547
+ // nothing, so its brace still has a construct to protect.
1548
+ .replace(/\{(?!([+\-~#=])\1\})(?=[+\-~#=%])/g, '\\{')
1388
1549
  .replace(/(^|[^\w.])@(?=[A-Za-z0-9_])/g, '$1\\@')
1389
1550
  .replace(/(^|[^\w])#(?=[A-Za-z0-9_])/g, '$1\\#')
1390
1551
  // A `:name:` symbol only opens at a word boundary, and its name starts
@@ -505,7 +505,7 @@
505
505
  },
506
506
  {
507
507
  "name": "table-with-spans",
508
- "carve": "| a || b |\n|=h |= i |\n",
508
+ "carve": "| a || b |\n|= h |= i |\n",
509
509
  "pm": {
510
510
  "type": "doc",
511
511
  "content": [