@ai-matrx/content-ir 0.12.0 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +119 -0
- package/dist/index.cjs +546 -98
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.js +539 -99
- package/dist/index.js.map +1 -1
- package/dist/source.cjs +546 -98
- package/dist/source.cjs.map +1 -1
- package/dist/source.d.cts +118 -14
- package/dist/source.d.ts +118 -14
- package/dist/source.js +539 -99
- package/dist/source.js.map +1 -1
- package/package.json +1 -1
package/dist/source.cjs
CHANGED
|
@@ -202,20 +202,84 @@ function normalizeFenceLanguage(language) {
|
|
|
202
202
|
return CODE_LANGUAGE_ALIASES[lower] ?? lower;
|
|
203
203
|
}
|
|
204
204
|
|
|
205
|
+
// source/fence-nesting.ts
|
|
206
|
+
var NESTING_FENCE_LANGUAGES = /* @__PURE__ */ new Set([
|
|
207
|
+
"markdown",
|
|
208
|
+
"md",
|
|
209
|
+
"mdx"
|
|
210
|
+
]);
|
|
211
|
+
function fenceNestsInnerFences(language) {
|
|
212
|
+
return !!language && NESTING_FENCE_LANGUAGES.has(language.toLowerCase());
|
|
213
|
+
}
|
|
214
|
+
var FENCE_WHITESPACE = " \n\r\f\v";
|
|
215
|
+
function isFenceWhitespace(ch) {
|
|
216
|
+
return ch !== void 0 && ch !== "" && FENCE_WHITESPACE.includes(ch);
|
|
217
|
+
}
|
|
218
|
+
function trimFenceLine(line) {
|
|
219
|
+
let start = 0;
|
|
220
|
+
let end = line.length;
|
|
221
|
+
while (start < end && isFenceWhitespace(line[start])) start++;
|
|
222
|
+
while (end > start && isFenceWhitespace(line[end - 1])) end--;
|
|
223
|
+
return start === 0 && end === line.length ? line : line.slice(start, end);
|
|
224
|
+
}
|
|
225
|
+
function classifyInnerFenceLine(trimmed, openTicks, nests, nestedDepth) {
|
|
226
|
+
let ticks = 0;
|
|
227
|
+
while (ticks < trimmed.length && trimmed[ticks] === "`") ticks++;
|
|
228
|
+
if (ticks < 3) return "content";
|
|
229
|
+
const info = trimFenceLine(trimmed.slice(ticks));
|
|
230
|
+
if (info === "") {
|
|
231
|
+
if (ticks < openTicks) return "content";
|
|
232
|
+
if (nests && nestedDepth > 0) return "close-nested";
|
|
233
|
+
return "close-outer";
|
|
234
|
+
}
|
|
235
|
+
if (nests && ticks >= openTicks && !info.includes("`")) return "open-nested";
|
|
236
|
+
return "content";
|
|
237
|
+
}
|
|
238
|
+
|
|
205
239
|
// source/math.ts
|
|
206
240
|
var TEX_SIGNAL = /\\[A-Za-z]+|[\\^_{}=<>+]/;
|
|
207
241
|
var LONE_VARIABLE = /^[A-Za-z](?:'|[0-9])?$/;
|
|
242
|
+
var MATH_SHAPE_CHARS = /^[A-Za-z0-9'\s(),.*\/-]+$/;
|
|
243
|
+
var FUNCTION_NAMES = /* @__PURE__ */ new Set([
|
|
244
|
+
"sin",
|
|
245
|
+
"cos",
|
|
246
|
+
"tan",
|
|
247
|
+
"sec",
|
|
248
|
+
"csc",
|
|
249
|
+
"cot",
|
|
250
|
+
"log",
|
|
251
|
+
"ln",
|
|
252
|
+
"exp",
|
|
253
|
+
"lim",
|
|
254
|
+
"max",
|
|
255
|
+
"min",
|
|
256
|
+
"det",
|
|
257
|
+
"gcd",
|
|
258
|
+
"mod",
|
|
259
|
+
"arg",
|
|
260
|
+
"sinh",
|
|
261
|
+
"cosh",
|
|
262
|
+
"tanh"
|
|
263
|
+
]);
|
|
264
|
+
function isMathShape(content) {
|
|
265
|
+
if (!MATH_SHAPE_CHARS.test(content)) return false;
|
|
266
|
+
const words = content.match(/[A-Za-z]+/g);
|
|
267
|
+
if (!words) return false;
|
|
268
|
+
if (!words.every((w) => w.length === 1 || FUNCTION_NAMES.has(w))) return false;
|
|
269
|
+
return /[A-Za-z]'*\(|[A-Za-z]'|\b[A-Za-z]\s+[A-Za-z]\b|\([^)]*[A-Za-z][^)]*\)/.test(content);
|
|
270
|
+
}
|
|
208
271
|
function isSingleDollarMath(content, after) {
|
|
209
272
|
if (!content || content.length > 400) return false;
|
|
210
273
|
if (/^\s/.test(content) || /\s$/.test(content)) return false;
|
|
211
274
|
if (after !== void 0 && /[0-9]/.test(after)) return false;
|
|
212
275
|
if (LONE_VARIABLE.test(content)) return true;
|
|
213
276
|
if (/^[0-9]/.test(content)) return /\\[A-Za-z]+|[\\^_{}]/.test(content);
|
|
214
|
-
return TEX_SIGNAL.test(content);
|
|
277
|
+
return TEX_SIGNAL.test(content) || isMathShape(content);
|
|
215
278
|
}
|
|
216
279
|
function singleDollarMathEnd(text, open, limit = text.length) {
|
|
280
|
+
const bound = Math.min(limit, open + 403);
|
|
217
281
|
let close = -1;
|
|
218
|
-
for (let k = open + 1; k <
|
|
282
|
+
for (let k = open + 1; k < bound; k += 1) {
|
|
219
283
|
const c = text[k];
|
|
220
284
|
if (c === "\n") break;
|
|
221
285
|
if (c === "\\") {
|
|
@@ -232,11 +296,69 @@ function singleDollarMathEnd(text, open, limit = text.length) {
|
|
|
232
296
|
return isSingleDollarMath(content, text[close + 1]) ? close + 1 : -1;
|
|
233
297
|
}
|
|
234
298
|
|
|
299
|
+
// source/math-pairs.ts
|
|
300
|
+
var MAX_MATH_SPAN = 600;
|
|
301
|
+
var STRUCTURAL_MARKDOWN = /\]\(|https?:\/\/|\*\*|(?:^|\n)[ \t]{0,3}#{1,6}[ \t]|(?:^|\n)[ \t]*[-*+][ \t]+|(?:^|\n)[ \t]*\d+[.)][ \t]/;
|
|
302
|
+
var LATEX_COMMAND = /\\[a-zA-Z]/;
|
|
303
|
+
var PROSE_WORD = /[A-Za-z]{3,}/g;
|
|
304
|
+
var PROSE_WORD_LIMIT = 6;
|
|
305
|
+
function looksLikeDisplayMath(inner) {
|
|
306
|
+
const s = inner.trim();
|
|
307
|
+
if (!s) return false;
|
|
308
|
+
if (STRUCTURAL_MARKDOWN.test(s)) return false;
|
|
309
|
+
if (s.length > MAX_MATH_SPAN) return false;
|
|
310
|
+
if (LATEX_COMMAND.test(s)) {
|
|
311
|
+
return (s.match(PROSE_WORD) ?? []).length < PROSE_WORD_LIMIT * 3;
|
|
312
|
+
}
|
|
313
|
+
if (/\n[ \t]*\n/.test(s)) return false;
|
|
314
|
+
return (s.match(PROSE_WORD) ?? []).length < PROSE_WORD_LIMIT;
|
|
315
|
+
}
|
|
316
|
+
function codeRanges(text) {
|
|
317
|
+
const ranges = [];
|
|
318
|
+
for (const re of [/```[\s\S]*?(?:```|$)/g, /~~~[\s\S]*?(?:~~~|$)/g, /`[^`\n]*`/g]) {
|
|
319
|
+
let m;
|
|
320
|
+
while ((m = re.exec(text)) !== null) {
|
|
321
|
+
ranges.push([m.index, m.index + m[0].length]);
|
|
322
|
+
if (m[0].length === 0) re.lastIndex++;
|
|
323
|
+
}
|
|
324
|
+
}
|
|
325
|
+
return ranges.sort((a, b) => a[0] - b[0]);
|
|
326
|
+
}
|
|
327
|
+
function pairDisplayMath(text) {
|
|
328
|
+
const pairs = /* @__PURE__ */ new Map();
|
|
329
|
+
if (!text.includes("$$")) return pairs;
|
|
330
|
+
const ranges = codeRanges(text);
|
|
331
|
+
const tokens = [];
|
|
332
|
+
let r = 0;
|
|
333
|
+
let maxEnd = -1;
|
|
334
|
+
for (let i = text.indexOf("$$"); i !== -1 && i < text.length - 1; i = text.indexOf("$$", i)) {
|
|
335
|
+
while (r < ranges.length && ranges[r][0] <= i) {
|
|
336
|
+
maxEnd = Math.max(maxEnd, ranges[r][1]);
|
|
337
|
+
r++;
|
|
338
|
+
}
|
|
339
|
+
if (maxEnd <= i) tokens.push(i);
|
|
340
|
+
i += 2;
|
|
341
|
+
}
|
|
342
|
+
let j = 0;
|
|
343
|
+
while (j < tokens.length) {
|
|
344
|
+
const open = tokens[j];
|
|
345
|
+
const close = tokens[j + 1];
|
|
346
|
+
if (close === void 0) break;
|
|
347
|
+
if (looksLikeDisplayMath(text.slice(open + 2, close))) {
|
|
348
|
+
pairs.set(open, close);
|
|
349
|
+
j += 2;
|
|
350
|
+
continue;
|
|
351
|
+
}
|
|
352
|
+
j += 1;
|
|
353
|
+
}
|
|
354
|
+
return pairs;
|
|
355
|
+
}
|
|
356
|
+
|
|
235
357
|
// source/xml-tag.ts
|
|
236
358
|
var XML_NAME_START = /[A-Za-z_]/;
|
|
237
359
|
var XML_NAME_CHARACTER = /[\w.:-]/;
|
|
238
360
|
var WHITESPACE = /\s/;
|
|
239
|
-
function readXmlTag(content, start) {
|
|
361
|
+
function readXmlTag(content, start, find = (needle, from) => content.indexOf(needle, from)) {
|
|
240
362
|
let cursor = start + 1;
|
|
241
363
|
const isClosing = content[cursor] === "/";
|
|
242
364
|
if (isClosing) cursor++;
|
|
@@ -289,8 +411,9 @@ function readXmlTag(content, start) {
|
|
|
289
411
|
const quote = content[cursor];
|
|
290
412
|
if (quote !== '"' && quote !== "'") return null;
|
|
291
413
|
const valueStart = ++cursor;
|
|
292
|
-
|
|
293
|
-
if (
|
|
414
|
+
const valueEnd = find(quote, valueStart);
|
|
415
|
+
if (valueEnd === -1) return null;
|
|
416
|
+
cursor = valueEnd;
|
|
294
417
|
attributes.push({ name, value: content.slice(valueStart, cursor) });
|
|
295
418
|
cursor++;
|
|
296
419
|
}
|
|
@@ -387,6 +510,10 @@ var XmlContainerTracker = class {
|
|
|
387
510
|
|
|
388
511
|
// source/tokenize.ts
|
|
389
512
|
var NOT_WHITESPACE = /\S/;
|
|
513
|
+
var DIRECTIVE_OPEN = /^(:{2,})([A-Za-z][\w-]*)/;
|
|
514
|
+
var MKDOCS_OPEN = /^(!!!|\?\?\?\+?)[ \t]+([A-Za-z][\w-]*)/;
|
|
515
|
+
var CALLOUT_OPEN = /^>[ \t]*\[!([A-Za-z][\w-]*)\]([+-])?/;
|
|
516
|
+
var FOOTNOTE_DEF = /^\[\^([^\]\s]+)\]:/;
|
|
390
517
|
function countStructuralBraces(source) {
|
|
391
518
|
let opens = 0;
|
|
392
519
|
let closes = 0;
|
|
@@ -509,11 +636,16 @@ function backtickRunLength(str, pos) {
|
|
|
509
636
|
while (pos + n < str.length && str[pos + n] === "`") n++;
|
|
510
637
|
return n;
|
|
511
638
|
}
|
|
639
|
+
function isSplitterTreeLine(line) {
|
|
640
|
+
return TREE_GLYPHS.test(line) || /^[\s│|]*[├└+|][\s─\-]+/.test(line);
|
|
641
|
+
}
|
|
512
642
|
var MEDIA_REF_PREFIXES = ["[Image URL:", "[Video URL:", "[Audio URL:"];
|
|
513
643
|
var Tokenizer = class {
|
|
514
|
-
constructor(text) {
|
|
644
|
+
constructor(text, streaming = false) {
|
|
515
645
|
this.text = text;
|
|
646
|
+
this.streaming = streaming;
|
|
516
647
|
this.n = text.length;
|
|
648
|
+
this.mathPairs = pairDisplayMath(text);
|
|
517
649
|
let start = 0;
|
|
518
650
|
for (let i = 0; i < text.length; i++) {
|
|
519
651
|
if (text.charCodeAt(i) === 10) {
|
|
@@ -530,6 +662,7 @@ var Tokenizer = class {
|
|
|
530
662
|
}
|
|
531
663
|
}
|
|
532
664
|
text;
|
|
665
|
+
streaming;
|
|
533
666
|
n;
|
|
534
667
|
lineStarts = [];
|
|
535
668
|
contentEnds = [];
|
|
@@ -539,6 +672,38 @@ var Tokenizer = class {
|
|
|
539
672
|
/** Lazily built: for line i, first line j >= i where brace net from i drops to <= 0. */
|
|
540
673
|
braceStop = null;
|
|
541
674
|
braceNet = null;
|
|
675
|
+
/** Every occurrence offset of each needle searched so far (overlapping, like indexOf). */
|
|
676
|
+
occurrences = /* @__PURE__ */ new Map();
|
|
677
|
+
/** `$$` opener → closer for the pairs the core renders as math. */
|
|
678
|
+
mathPairs;
|
|
679
|
+
finder = (needle, from) => this.next(needle, from);
|
|
680
|
+
/**
|
|
681
|
+
* `text.indexOf(needle, from)` in O(log k): the first query for a needle
|
|
682
|
+
* indexes all of its occurrences once, so repeated searches for a closer
|
|
683
|
+
* that never comes stay linear over the whole text.
|
|
684
|
+
*/
|
|
685
|
+
next(needle, from) {
|
|
686
|
+
let list = this.occurrences.get(needle);
|
|
687
|
+
if (!list) {
|
|
688
|
+
list = [];
|
|
689
|
+
for (let at = this.text.indexOf(needle); at !== -1; at = this.text.indexOf(needle, at + 1)) {
|
|
690
|
+
list.push(at);
|
|
691
|
+
}
|
|
692
|
+
this.occurrences.set(needle, list);
|
|
693
|
+
}
|
|
694
|
+
let lo = 0;
|
|
695
|
+
let hi = list.length;
|
|
696
|
+
while (lo < hi) {
|
|
697
|
+
const mid = lo + hi >>> 1;
|
|
698
|
+
if (list[mid] < from) lo = mid + 1;
|
|
699
|
+
else hi = mid;
|
|
700
|
+
}
|
|
701
|
+
return lo < list.length ? list[lo] : -1;
|
|
702
|
+
}
|
|
703
|
+
/** Streaming only: the final line, still waiting for its terminator. */
|
|
704
|
+
isOpenTail(k) {
|
|
705
|
+
return this.streaming && k === this.lineStarts.length - 1 && this.lineEnds[k] === this.contentEnds[k];
|
|
706
|
+
}
|
|
542
707
|
get lineCount() {
|
|
543
708
|
return this.lineStarts.length;
|
|
544
709
|
}
|
|
@@ -620,6 +785,22 @@ var Tokenizer = class {
|
|
|
620
785
|
const fm = this.frontMatter(li, t);
|
|
621
786
|
if (fm) return fm;
|
|
622
787
|
}
|
|
788
|
+
if (indent <= 3 && t.startsWith("::")) {
|
|
789
|
+
const directive = this.directive(p, li, t);
|
|
790
|
+
if (directive) return directive;
|
|
791
|
+
}
|
|
792
|
+
if (indent === 0 && (t.startsWith("!!!") || t.startsWith("???"))) {
|
|
793
|
+
const admonition = this.mkdocsAdmonition(p, li, t);
|
|
794
|
+
if (admonition) return admonition;
|
|
795
|
+
}
|
|
796
|
+
if (indent <= 3 && t.startsWith(">")) {
|
|
797
|
+
const callout = this.callout(p, li, t);
|
|
798
|
+
if (callout) return callout;
|
|
799
|
+
}
|
|
800
|
+
if (atParaStart && indent <= 3 && t.startsWith("[^")) {
|
|
801
|
+
const footnote = this.footnoteDefinition(p, li, t);
|
|
802
|
+
if (footnote) return footnote;
|
|
803
|
+
}
|
|
623
804
|
if (t.startsWith("```")) return this.backtickFence(p, li, t);
|
|
624
805
|
if (indent <= 3 && t.startsWith("~~~")) {
|
|
625
806
|
const tilde = this.tildeFence(p, li, t);
|
|
@@ -657,20 +838,116 @@ var Tokenizer = class {
|
|
|
657
838
|
return { type, start, end, complete, meta };
|
|
658
839
|
}
|
|
659
840
|
frontMatter(li, t) {
|
|
660
|
-
|
|
841
|
+
const fence = this.lineText(li);
|
|
842
|
+
if (fence !== "---" && fence !== "+++" || t !== fence) return null;
|
|
843
|
+
const toml = fence === "+++";
|
|
661
844
|
let sawContent = false;
|
|
662
845
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
663
846
|
const line = this.lineText(k);
|
|
664
|
-
if (line ===
|
|
665
|
-
return sawContent ? this.island("front_matter", 0, this.contentEnds[k], true) : null;
|
|
847
|
+
if (line === fence || !toml && line === "...") {
|
|
848
|
+
return sawContent ? this.island("front_matter", 0, this.contentEnds[k], true, { format: toml ? "toml" : "yaml" }) : null;
|
|
666
849
|
}
|
|
667
850
|
if (!NOT_WHITESPACE.test(line)) continue;
|
|
668
|
-
|
|
851
|
+
const shaped = toml ? /^\s*(?:[A-Za-z0-9_."-]+\s*=|\[|#)/.test(line) : /^(?:[A-Za-z0-9_"'-][^:]*:(?:\s|$)|\s+\S|-\s|#)/.test(line);
|
|
852
|
+
if (!shaped) return null;
|
|
669
853
|
sawContent = true;
|
|
670
854
|
}
|
|
671
855
|
return null;
|
|
672
856
|
}
|
|
673
|
-
/**
|
|
857
|
+
/**
|
|
858
|
+
* `:::name[label]{attrs}` … closing `:::` (a colon run at least as long as
|
|
859
|
+
* the opener's, nesting counted), or a leaf `::name[…]{…}` line. The whole
|
|
860
|
+
* container is ONE island: its grammar (fence lengths, labels, attributes,
|
|
861
|
+
* nested tabs) is exactly what a visual editor would re-serialize wrongly.
|
|
862
|
+
*/
|
|
863
|
+
directive(p, li, t) {
|
|
864
|
+
const open = DIRECTIVE_OPEN.exec(t);
|
|
865
|
+
if (!open) return null;
|
|
866
|
+
const colons = open[1].length;
|
|
867
|
+
const name = open[2];
|
|
868
|
+
if (colons === 2) {
|
|
869
|
+
return this.island("directive", p, this.contentEnds[li], true, { name, leaf: true });
|
|
870
|
+
}
|
|
871
|
+
const stack = [colons];
|
|
872
|
+
for (let k = li + 1; k < this.lineCount; k++) {
|
|
873
|
+
const line = this.lineText(k).trim();
|
|
874
|
+
const inner = DIRECTIVE_OPEN.exec(line);
|
|
875
|
+
if (inner && inner[1].length >= 3) {
|
|
876
|
+
stack.push(inner[1].length);
|
|
877
|
+
continue;
|
|
878
|
+
}
|
|
879
|
+
const close = /^(:{3,})\s*$/.exec(line);
|
|
880
|
+
if (close && close[1].length >= stack[stack.length - 1]) {
|
|
881
|
+
stack.pop();
|
|
882
|
+
if (stack.length === 0) return this.island("directive", p, this.contentEnds[k], true, { name });
|
|
883
|
+
}
|
|
884
|
+
}
|
|
885
|
+
return this.island("directive", p, this.n, false, { name });
|
|
886
|
+
}
|
|
887
|
+
/** MkDocs `!!! type "Title"` / `??? type` / `???+ type` and its indented (or blank-separated indented) body. */
|
|
888
|
+
mkdocsAdmonition(p, li, t) {
|
|
889
|
+
const open = MKDOCS_OPEN.exec(t);
|
|
890
|
+
if (!open) return null;
|
|
891
|
+
let last = li;
|
|
892
|
+
for (let k = li + 1; k < this.lineCount; k++) {
|
|
893
|
+
const line = this.lineText(k);
|
|
894
|
+
if (/^(?: {4}|\t)/.test(line) && NOT_WHITESPACE.test(line)) {
|
|
895
|
+
last = k;
|
|
896
|
+
continue;
|
|
897
|
+
}
|
|
898
|
+
if (!NOT_WHITESPACE.test(line)) continue;
|
|
899
|
+
break;
|
|
900
|
+
}
|
|
901
|
+
return this.island("directive", p, this.contentEnds[last], true, {
|
|
902
|
+
name: open[2].toLowerCase(),
|
|
903
|
+
syntax: "mkdocs"
|
|
904
|
+
});
|
|
905
|
+
}
|
|
906
|
+
/** A blockquote whose first line is `> [!TYPE]…` — every following `>` line belongs to it. */
|
|
907
|
+
callout(p, li, t) {
|
|
908
|
+
const open = CALLOUT_OPEN.exec(t);
|
|
909
|
+
if (!open) return null;
|
|
910
|
+
let last = li;
|
|
911
|
+
for (let k = li + 1; k < this.lineCount; k++) {
|
|
912
|
+
if (!/^ {0,3}>/.test(this.lineText(k))) break;
|
|
913
|
+
last = k;
|
|
914
|
+
}
|
|
915
|
+
return this.island("callout", p, this.contentEnds[last], true, {
|
|
916
|
+
name: open[1].toLowerCase(),
|
|
917
|
+
...open[2] ? { fold: open[2] === "-" ? "closed" : "open" } : {}
|
|
918
|
+
});
|
|
919
|
+
}
|
|
920
|
+
/** `[^id]: text` plus its continuation lines (indented, or lazy text before a blank line). */
|
|
921
|
+
footnoteDefinition(p, li, t) {
|
|
922
|
+
const open = FOOTNOTE_DEF.exec(t);
|
|
923
|
+
if (!open) return null;
|
|
924
|
+
let last = li;
|
|
925
|
+
for (let k = li + 1; k < this.lineCount; k++) {
|
|
926
|
+
const line = this.lineText(k);
|
|
927
|
+
if (!NOT_WHITESPACE.test(line)) {
|
|
928
|
+
let j = k + 1;
|
|
929
|
+
while (j < this.lineCount && !NOT_WHITESPACE.test(this.lineText(j))) j++;
|
|
930
|
+
if (j < this.lineCount && /^(?: {4}|\t)/.test(this.lineText(j))) {
|
|
931
|
+
k = j - 1;
|
|
932
|
+
continue;
|
|
933
|
+
}
|
|
934
|
+
break;
|
|
935
|
+
}
|
|
936
|
+
if (/^(?: {4}|\t)/.test(line)) {
|
|
937
|
+
last = k;
|
|
938
|
+
continue;
|
|
939
|
+
}
|
|
940
|
+
if (FOOTNOTE_DEF.test(line.trimStart()) || this.detectBlock(this.lineStarts[k], k, false)) break;
|
|
941
|
+
last = k;
|
|
942
|
+
}
|
|
943
|
+
return this.island("footnote_def", p, this.contentEnds[last], true, { id: open[1] });
|
|
944
|
+
}
|
|
945
|
+
/**
|
|
946
|
+
* Backtick fence — the renderer splitter's close rules: JSON-string aware for
|
|
947
|
+
* ```json, and THE nested-fence rule (source/fence-nesting.ts) for markdown
|
|
948
|
+
* fences, with the splitter's strict-CommonMark retry when a nested fence is
|
|
949
|
+
* still open at the end of the text.
|
|
950
|
+
*/
|
|
674
951
|
backtickFence(p, li, t) {
|
|
675
952
|
const openTicks = backtickRunLength(t, 0);
|
|
676
953
|
const info = t.slice(openTicks).trim();
|
|
@@ -679,10 +956,23 @@ var Tokenizer = class {
|
|
|
679
956
|
const normalized = normalizeFenceLanguage(lang);
|
|
680
957
|
const meta = { fence: "`", ticks: openTicks, lang };
|
|
681
958
|
if (normalized && SPECIAL_CODE_LANGUAGES.has(normalized)) meta.special = normalized;
|
|
959
|
+
const nesting = fenceNestsInnerFences(lang);
|
|
960
|
+
let close = this.backtickFenceClose(li, openTicks, isJson, nesting);
|
|
961
|
+
if (close === "retry-strict") close = this.backtickFenceClose(li, openTicks, isJson, false);
|
|
962
|
+
if (typeof close === "number") {
|
|
963
|
+
return this.fenceIsland(p, this.contentEnds[close], true, meta, li, close, isJson);
|
|
964
|
+
}
|
|
965
|
+
return this.fenceIsland(p, this.n, false, meta, li, this.lineCount, isJson);
|
|
966
|
+
}
|
|
967
|
+
/** The closing line of a backtick fence opened on line `li`, or null when it never closes. */
|
|
968
|
+
backtickFenceClose(li, openTicks, isJson, nesting) {
|
|
682
969
|
let state = { inString: false, escaped: false };
|
|
970
|
+
let nestedDepth = 0;
|
|
971
|
+
let sawNested = false;
|
|
683
972
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
973
|
+
if (this.isOpenTail(k)) continue;
|
|
684
974
|
const line = this.lineText(k);
|
|
685
|
-
const trimmedLine = line
|
|
975
|
+
const trimmedLine = trimFenceLine(line);
|
|
686
976
|
if (trimmedLine.startsWith("```")) {
|
|
687
977
|
const at2 = line.indexOf("```");
|
|
688
978
|
if (isJson) {
|
|
@@ -692,11 +982,13 @@ var Tokenizer = class {
|
|
|
692
982
|
continue;
|
|
693
983
|
}
|
|
694
984
|
}
|
|
695
|
-
const
|
|
696
|
-
|
|
697
|
-
if (
|
|
698
|
-
|
|
985
|
+
const kind = classifyInnerFenceLine(trimmedLine, openTicks, nesting, nestedDepth);
|
|
986
|
+
if (kind === "close-outer") return k;
|
|
987
|
+
if (kind === "open-nested") {
|
|
988
|
+
nestedDepth++;
|
|
989
|
+
sawNested = true;
|
|
699
990
|
}
|
|
991
|
+
if (kind === "close-nested") nestedDepth--;
|
|
700
992
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
701
993
|
continue;
|
|
702
994
|
}
|
|
@@ -710,16 +1002,14 @@ var Tokenizer = class {
|
|
|
710
1002
|
}
|
|
711
1003
|
}
|
|
712
1004
|
const closeTicks = backtickRunLength(line, at);
|
|
713
|
-
const after = line.slice(at + closeTicks)
|
|
714
|
-
if (closeTicks >= openTicks && after === "")
|
|
715
|
-
return this.fenceIsland(p, this.contentEnds[k], true, meta, li, k, isJson);
|
|
716
|
-
}
|
|
1005
|
+
const after = trimFenceLine(line.slice(at + closeTicks));
|
|
1006
|
+
if (closeTicks >= openTicks && after === "" && nestedDepth === 0) return k;
|
|
717
1007
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
718
1008
|
continue;
|
|
719
1009
|
}
|
|
720
1010
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
721
1011
|
}
|
|
722
|
-
return
|
|
1012
|
+
return sawNested && nestedDepth > 0 ? "retry-strict" : null;
|
|
723
1013
|
}
|
|
724
1014
|
fenceIsland(p, end, complete, meta, openLine, closeLine, isJson) {
|
|
725
1015
|
if (isJson && openLine + 1 < this.lineCount) {
|
|
@@ -739,24 +1029,29 @@ var Tokenizer = class {
|
|
|
739
1029
|
if (normalized && SPECIAL_CODE_LANGUAGES.has(normalized)) meta.special = normalized;
|
|
740
1030
|
const close = new RegExp(`^ {0,3}~{${ticks},}[ \\t]*$`);
|
|
741
1031
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
1032
|
+
if (this.isOpenTail(k)) continue;
|
|
742
1033
|
if (close.test(this.lineText(k))) {
|
|
743
1034
|
return this.island("fence", p, this.contentEnds[k], true, meta);
|
|
744
1035
|
}
|
|
745
1036
|
}
|
|
746
1037
|
return this.island("fence", p, this.n, false, meta);
|
|
747
1038
|
}
|
|
1039
|
+
/**
|
|
1040
|
+
* A line-leading `$$` that OPENS a math pair (source/math-pairs.ts — the
|
|
1041
|
+
* core's rule). A closer, an unpaired `$$`, or a prose pair is never a block:
|
|
1042
|
+
* a lone `$$` renders as literal text and locks nothing.
|
|
1043
|
+
*/
|
|
748
1044
|
mathBlock(p, tp) {
|
|
1045
|
+
const close = this.mathPairs.get(tp);
|
|
1046
|
+
if (close === void 0) return null;
|
|
1047
|
+
const end = close + 2;
|
|
749
1048
|
const li = this.lineOf(tp);
|
|
750
1049
|
const ce = this.contentEnds[li];
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
}
|
|
757
|
-
const close = this.text.indexOf("$$", tp + 2);
|
|
758
|
-
if (close === -1) return this.island("math_block", p, this.n, false, { delimiter: "$$" });
|
|
759
|
-
return this.island("math_block", p, close + 2, true, { delimiter: "$$" });
|
|
1050
|
+
if (end <= ce && !this.isBlank(end, ce)) return null;
|
|
1051
|
+
const endLine = this.lineOf(close);
|
|
1052
|
+
const endCe = this.contentEnds[endLine];
|
|
1053
|
+
if (endLine !== li && !this.isBlank(end, endCe)) return null;
|
|
1054
|
+
return this.island("math_block", p, end, true, { delimiter: "$$" });
|
|
760
1055
|
}
|
|
761
1056
|
/** `<artifact …>`, `<decision …>`, … at the start of a line. */
|
|
762
1057
|
attributeXml(p, tp, t) {
|
|
@@ -773,7 +1068,7 @@ var Tokenizer = class {
|
|
|
773
1068
|
attributeElement(name, p, tp, type) {
|
|
774
1069
|
const li = this.lineOf(tp);
|
|
775
1070
|
const ce = this.contentEnds[li];
|
|
776
|
-
const tag = readXmlTag(this.text, tp);
|
|
1071
|
+
const tag = readXmlTag(this.text, tp, this.finder);
|
|
777
1072
|
let openEnd;
|
|
778
1073
|
if (tag && !tag.isClosing) {
|
|
779
1074
|
if (tag.isSelfClosing) {
|
|
@@ -781,13 +1076,13 @@ var Tokenizer = class {
|
|
|
781
1076
|
}
|
|
782
1077
|
openEnd = tp + tag.raw.length;
|
|
783
1078
|
} else {
|
|
784
|
-
const gt = this.
|
|
1079
|
+
const gt = this.next(">", tp);
|
|
785
1080
|
if (gt === -1) return this.island(type, p, this.n, false, { tag: name });
|
|
786
1081
|
if (gt >= ce) return null;
|
|
787
1082
|
openEnd = gt + 1;
|
|
788
1083
|
}
|
|
789
1084
|
const closer = `</${name}>`;
|
|
790
|
-
const close = this.
|
|
1085
|
+
const close = this.next(closer, openEnd);
|
|
791
1086
|
if (close === -1) return this.island(type, p, this.n, false, { tag: name });
|
|
792
1087
|
return this.island(type, p, close + closer.length, true, { tag: name });
|
|
793
1088
|
}
|
|
@@ -797,7 +1092,7 @@ var Tokenizer = class {
|
|
|
797
1092
|
const opener = `<${name}>`;
|
|
798
1093
|
if (!t.startsWith(opener)) continue;
|
|
799
1094
|
const closer = `</${name}>`;
|
|
800
|
-
const close = this.
|
|
1095
|
+
const close = this.next(closer, tp + opener.length);
|
|
801
1096
|
if (close === -1) return this.island("xml_region", p, this.n, false, { tag: name });
|
|
802
1097
|
return this.island("xml_region", p, close + closer.length, true, { tag: name });
|
|
803
1098
|
}
|
|
@@ -833,7 +1128,7 @@ var Tokenizer = class {
|
|
|
833
1128
|
return null;
|
|
834
1129
|
}
|
|
835
1130
|
blockComment(p, tp, li) {
|
|
836
|
-
const close = this.
|
|
1131
|
+
const close = this.next("-->", tp + 4);
|
|
837
1132
|
if (close === -1) return this.island("html_comment", p, this.n, false);
|
|
838
1133
|
const end = close + 3;
|
|
839
1134
|
const raw = this.text.slice(tp, end);
|
|
@@ -911,9 +1206,8 @@ var Tokenizer = class {
|
|
|
911
1206
|
}
|
|
912
1207
|
}
|
|
913
1208
|
if (lastLine === null) {
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
return this.island("json", p, this.n, false, jsonMeta(partial));
|
|
1209
|
+
if (isRecognizedJsonHead(this.text.slice(tp, tp + 512))) {
|
|
1210
|
+
return this.island("json", p, this.n, false, jsonMeta(this.text.slice(tp)));
|
|
917
1211
|
}
|
|
918
1212
|
return null;
|
|
919
1213
|
}
|
|
@@ -948,11 +1242,43 @@ var Tokenizer = class {
|
|
|
948
1242
|
// ── prose ──────────────────────────────────────────────────────────────
|
|
949
1243
|
paragraph(pos, li) {
|
|
950
1244
|
let end = this.contentEnds[li];
|
|
1245
|
+
let breakLine = -1;
|
|
951
1246
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
952
1247
|
if (this.lineIsBlank(k)) break;
|
|
953
|
-
if (this.detectBlock(this.lineStarts[k], k, false))
|
|
1248
|
+
if (this.detectBlock(this.lineStarts[k], k, false)) {
|
|
1249
|
+
breakLine = k;
|
|
1250
|
+
break;
|
|
1251
|
+
}
|
|
954
1252
|
end = this.contentEnds[k];
|
|
955
1253
|
}
|
|
1254
|
+
if (breakLine !== -1) {
|
|
1255
|
+
const breakIsland = this.detectBlock(this.lineStarts[breakLine], breakLine, false);
|
|
1256
|
+
if (breakIsland?.type === "tree") {
|
|
1257
|
+
let root = breakLine;
|
|
1258
|
+
for (let j = breakLine - 1; j >= li; j--) {
|
|
1259
|
+
const trimmed = this.lineText(j).trim();
|
|
1260
|
+
if (!trimmed || isSplitterTreeLine(trimmed) || /^#{1,6}\s/.test(trimmed)) break;
|
|
1261
|
+
root = j;
|
|
1262
|
+
}
|
|
1263
|
+
if (root < breakLine) {
|
|
1264
|
+
const treeStart = root === li ? pos : this.lineStarts[root];
|
|
1265
|
+
const tree = { ...breakIsland, start: treeStart };
|
|
1266
|
+
if (treeStart > pos) {
|
|
1267
|
+
const proseEnd = this.contentEnds[root - 1];
|
|
1268
|
+
const scan2 = this.scanInline(pos, proseEnd);
|
|
1269
|
+
if (!scan2.breakIsland) {
|
|
1270
|
+
this.push({ kind: "prose", start: pos, end: proseEnd, island: null, inlines: scan2.inlines });
|
|
1271
|
+
this.push({ kind: "gap", start: proseEnd, end: treeStart, island: null, inlines: [] });
|
|
1272
|
+
this.push({ kind: "island", start: treeStart, end: tree.end, island: tree, inlines: [] });
|
|
1273
|
+
return tree.end;
|
|
1274
|
+
}
|
|
1275
|
+
} else {
|
|
1276
|
+
this.push({ kind: "island", start: treeStart, end: tree.end, island: tree, inlines: [] });
|
|
1277
|
+
return tree.end;
|
|
1278
|
+
}
|
|
1279
|
+
}
|
|
1280
|
+
}
|
|
1281
|
+
}
|
|
956
1282
|
const scan = this.scanInline(pos, end);
|
|
957
1283
|
if (!scan.breakIsland) {
|
|
958
1284
|
this.push({ kind: "prose", start: pos, end, island: null, inlines: scan.inlines });
|
|
@@ -985,17 +1311,49 @@ var Tokenizer = class {
|
|
|
985
1311
|
const text = this.text;
|
|
986
1312
|
const inlines = [];
|
|
987
1313
|
const hasKind = text.slice(ps, pe).includes("__kind");
|
|
1314
|
+
let kindBudget = 4 * (pe - ps) + 65536;
|
|
988
1315
|
const stop = (island, orphan = false) => ({ inlines, breakIsland: island, orphan });
|
|
1316
|
+
const within = (needle, from) => {
|
|
1317
|
+
const at = this.next(needle, from);
|
|
1318
|
+
return at !== -1 && at + needle.length <= pe ? at : -1;
|
|
1319
|
+
};
|
|
1320
|
+
let runs = null;
|
|
1321
|
+
const codeSpanClose = (n, from) => {
|
|
1322
|
+
if (!runs) {
|
|
1323
|
+
runs = /* @__PURE__ */ new Map();
|
|
1324
|
+
for (let k = ps; k < pe; ) {
|
|
1325
|
+
if (text.charCodeAt(k) !== 96) {
|
|
1326
|
+
k++;
|
|
1327
|
+
continue;
|
|
1328
|
+
}
|
|
1329
|
+
const len = backtickRunLength(text, k);
|
|
1330
|
+
const list2 = runs.get(len);
|
|
1331
|
+
if (list2) list2.push(k);
|
|
1332
|
+
else runs.set(len, [k]);
|
|
1333
|
+
k += len;
|
|
1334
|
+
}
|
|
1335
|
+
}
|
|
1336
|
+
const list = runs.get(n);
|
|
1337
|
+
if (!list) return -1;
|
|
1338
|
+
let lo = 0;
|
|
1339
|
+
let hi = list.length;
|
|
1340
|
+
while (lo < hi) {
|
|
1341
|
+
const mid = lo + hi >>> 1;
|
|
1342
|
+
if (list[mid] < from) lo = mid + 1;
|
|
1343
|
+
else hi = mid;
|
|
1344
|
+
}
|
|
1345
|
+
const at = list[lo];
|
|
1346
|
+
return at !== void 0 && at + n <= pe ? at : -1;
|
|
1347
|
+
};
|
|
989
1348
|
let i = ps;
|
|
990
1349
|
while (i < pe) {
|
|
991
1350
|
const ch = text[i];
|
|
992
1351
|
if (ch === "\\") {
|
|
993
|
-
const
|
|
994
|
-
if (
|
|
995
|
-
const
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: next === "(" ? "\\(" : "\\[" }));
|
|
1352
|
+
const nextCh = text[i + 1];
|
|
1353
|
+
if (nextCh === "(" || nextCh === "[") {
|
|
1354
|
+
const close = within(nextCh === "(" ? "\\)" : "\\]", i + 2);
|
|
1355
|
+
if (close !== -1) {
|
|
1356
|
+
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: nextCh === "(" ? "\\(" : "\\[" }));
|
|
999
1357
|
i = close + 2;
|
|
1000
1358
|
continue;
|
|
1001
1359
|
}
|
|
@@ -1005,33 +1363,20 @@ var Tokenizer = class {
|
|
|
1005
1363
|
}
|
|
1006
1364
|
if (ch === "`") {
|
|
1007
1365
|
const n = backtickRunLength(text, i);
|
|
1008
|
-
const
|
|
1009
|
-
let k = i + n;
|
|
1010
|
-
let close = -1;
|
|
1011
|
-
while (k < pe) {
|
|
1012
|
-
const idx = text.indexOf(run, k);
|
|
1013
|
-
if (idx === -1 || idx >= pe) break;
|
|
1014
|
-
if (text[idx + n] === "`") {
|
|
1015
|
-
let m = idx;
|
|
1016
|
-
while (text[m] === "`") m++;
|
|
1017
|
-
k = m;
|
|
1018
|
-
continue;
|
|
1019
|
-
}
|
|
1020
|
-
close = idx;
|
|
1021
|
-
break;
|
|
1022
|
-
}
|
|
1366
|
+
const close = codeSpanClose(n, i + n);
|
|
1023
1367
|
i = close === -1 ? i + n : close + n;
|
|
1024
1368
|
continue;
|
|
1025
1369
|
}
|
|
1026
1370
|
if (ch === "$") {
|
|
1027
1371
|
if (text[i + 1] === "$") {
|
|
1028
|
-
const close =
|
|
1029
|
-
if (close
|
|
1030
|
-
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: "$$" }));
|
|
1031
|
-
i = close + 2;
|
|
1032
|
-
} else {
|
|
1372
|
+
const close = this.mathPairs.get(i);
|
|
1373
|
+
if (close === void 0) {
|
|
1033
1374
|
i += 2;
|
|
1375
|
+
continue;
|
|
1034
1376
|
}
|
|
1377
|
+
if (close + 2 > pe) return stop(this.island("math_block", i, close + 2, true, { delimiter: "$$" }));
|
|
1378
|
+
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: "$$" }));
|
|
1379
|
+
i = close + 2;
|
|
1035
1380
|
continue;
|
|
1036
1381
|
}
|
|
1037
1382
|
const end = singleDollarMathEnd(text, i, pe);
|
|
@@ -1045,7 +1390,7 @@ var Tokenizer = class {
|
|
|
1045
1390
|
}
|
|
1046
1391
|
if (ch === "{") {
|
|
1047
1392
|
if (text[i + 1] === "{") {
|
|
1048
|
-
const close =
|
|
1393
|
+
const close = this.next("}}", i + 2);
|
|
1049
1394
|
if (close !== -1 && close + 2 <= pe && close - i <= 200 && !text.slice(i, close).includes("\n")) {
|
|
1050
1395
|
const raw = text.slice(i, close + 2);
|
|
1051
1396
|
const strict = STRICT_VARIABLE_RE.exec(raw);
|
|
@@ -1058,8 +1403,10 @@ var Tokenizer = class {
|
|
|
1058
1403
|
i += 2;
|
|
1059
1404
|
continue;
|
|
1060
1405
|
}
|
|
1061
|
-
if (hasKind) {
|
|
1062
|
-
const
|
|
1406
|
+
if (hasKind && kindBudget > 0 && /^\{\s*"/.test(text.slice(i, i + 64))) {
|
|
1407
|
+
const limit = Math.min(pe, i + 65536);
|
|
1408
|
+
const end = matchingJsonObjectEnd(text, i, limit);
|
|
1409
|
+
kindBudget -= (end ?? limit) - i;
|
|
1063
1410
|
if (end !== null) {
|
|
1064
1411
|
const kind = declaredKind(text.slice(i, end));
|
|
1065
1412
|
if (kind) {
|
|
@@ -1072,11 +1419,27 @@ var Tokenizer = class {
|
|
|
1072
1419
|
i++;
|
|
1073
1420
|
continue;
|
|
1074
1421
|
}
|
|
1422
|
+
if (ch === "[" && text[i + 1] === "[" || ch === "!" && text[i + 1] === "[" && text[i + 2] === "[") {
|
|
1423
|
+
const open = ch === "!" ? i + 3 : i + 2;
|
|
1424
|
+
const close = within("]]", open);
|
|
1425
|
+
const newline = this.next("\n", open);
|
|
1426
|
+
const inner = close === -1 ? "" : text.slice(open, close);
|
|
1427
|
+
if (close !== -1 && (newline === -1 || newline > close) && inner.trim() && !/[[\]]/.test(inner)) {
|
|
1428
|
+
const cut = inner.search(/\\?\|/);
|
|
1429
|
+
const target = (cut < 0 ? inner : inner.slice(0, cut)).trim();
|
|
1430
|
+
inlines.push(
|
|
1431
|
+
this.island("wikilink", i, close + 2, true, ch === "!" ? { target, embed: true } : { target })
|
|
1432
|
+
);
|
|
1433
|
+
i = close + 2;
|
|
1434
|
+
continue;
|
|
1435
|
+
}
|
|
1436
|
+
}
|
|
1075
1437
|
if (ch === "[") {
|
|
1076
1438
|
const media = MEDIA_REF_PREFIXES.find((prefix) => text.startsWith(prefix, i));
|
|
1077
1439
|
if (media) {
|
|
1078
|
-
const close =
|
|
1079
|
-
|
|
1440
|
+
const close = within("]", i + media.length);
|
|
1441
|
+
const newline = this.next("\n", i);
|
|
1442
|
+
if (close !== -1 && (newline === -1 || newline > close)) {
|
|
1080
1443
|
inlines.push(this.island("media_ref", i, close + 1, true, { media: media.slice(1, 6).toLowerCase() }));
|
|
1081
1444
|
i = close + 1;
|
|
1082
1445
|
continue;
|
|
@@ -1087,10 +1450,10 @@ var Tokenizer = class {
|
|
|
1087
1450
|
}
|
|
1088
1451
|
if (ch === "<") {
|
|
1089
1452
|
if (text.startsWith("<!--", i)) {
|
|
1090
|
-
const close =
|
|
1453
|
+
const close = this.next("-->", i + 4);
|
|
1091
1454
|
if (close === -1) return stop(this.island("html_comment", i, this.n, false));
|
|
1092
1455
|
const end = close + 3;
|
|
1093
|
-
const anchor = PINNED_ANCHOR_RE.exec(text.slice(i, end));
|
|
1456
|
+
const anchor = end - i <= 64 ? PINNED_ANCHOR_RE.exec(text.slice(i, end)) : null;
|
|
1094
1457
|
const type = anchor ? "anchor" : "html_comment";
|
|
1095
1458
|
const meta = anchor ? { id: anchor[1] } : {};
|
|
1096
1459
|
if (end > pe) return stop(this.island(type, i, end, true, meta));
|
|
@@ -1104,7 +1467,7 @@ var Tokenizer = class {
|
|
|
1104
1467
|
i += cite[0].length;
|
|
1105
1468
|
continue;
|
|
1106
1469
|
}
|
|
1107
|
-
const tag = readXmlTag(text, i);
|
|
1470
|
+
const tag = readXmlTag(text, i, this.finder);
|
|
1108
1471
|
if (tag && i + tag.raw.length <= pe) {
|
|
1109
1472
|
const name = tag.tagName;
|
|
1110
1473
|
const tagEnd = i + tag.raw.length;
|
|
@@ -1117,7 +1480,7 @@ var Tokenizer = class {
|
|
|
1117
1480
|
continue;
|
|
1118
1481
|
}
|
|
1119
1482
|
const closer = `</${name}>`;
|
|
1120
|
-
const close =
|
|
1483
|
+
const close = this.next(closer, tagEnd);
|
|
1121
1484
|
const blockType = isAttr ? "xml_attr" : "xml_region";
|
|
1122
1485
|
if (close === -1) return stop(this.island(blockType, i, this.n, false, { tag: name }));
|
|
1123
1486
|
const end = close + closer.length;
|
|
@@ -1141,7 +1504,7 @@ var Tokenizer = class {
|
|
|
1141
1504
|
const pending = ATTRIBUTE_XML_NAMES.find(
|
|
1142
1505
|
(name) => text.startsWith(`<${name}`, i) && /\s/.test(text[i + name.length + 1] ?? "")
|
|
1143
1506
|
);
|
|
1144
|
-
if (pending &&
|
|
1507
|
+
if (pending && this.next(">", i) === -1) {
|
|
1145
1508
|
return stop(this.island("xml_attr", i, this.n, false, { tag: pending }));
|
|
1146
1509
|
}
|
|
1147
1510
|
i++;
|
|
@@ -1152,8 +1515,8 @@ var Tokenizer = class {
|
|
|
1152
1515
|
return { inlines, breakIsland: null, orphan: false };
|
|
1153
1516
|
}
|
|
1154
1517
|
};
|
|
1155
|
-
function tokenizeSource(text) {
|
|
1156
|
-
const raw = new Tokenizer(text).run();
|
|
1518
|
+
function tokenizeSource(text, options = {}) {
|
|
1519
|
+
const raw = new Tokenizer(text, options.streaming ?? false).run();
|
|
1157
1520
|
const cp = buildCodePointIndex(text);
|
|
1158
1521
|
const inline = (island) => ({
|
|
1159
1522
|
islandType: island.type,
|
|
@@ -1202,7 +1565,7 @@ function listIslands(blocks) {
|
|
|
1202
1565
|
raw: block.raw
|
|
1203
1566
|
});
|
|
1204
1567
|
} else {
|
|
1205
|
-
|
|
1568
|
+
for (const inline of block.inlines) out.push(inline);
|
|
1206
1569
|
}
|
|
1207
1570
|
}
|
|
1208
1571
|
return out;
|
|
@@ -1232,9 +1595,44 @@ var SourceSpliceError = class extends Error {
|
|
|
1232
1595
|
function blockEdit(block, text) {
|
|
1233
1596
|
return { start: block.start, end: block.end, text };
|
|
1234
1597
|
}
|
|
1598
|
+
function islandEdit(island, text) {
|
|
1599
|
+
return { start: island.start, end: island.end, text, island: true };
|
|
1600
|
+
}
|
|
1235
1601
|
function blockRangeEdit(first, last, text) {
|
|
1236
1602
|
return { start: first.start, end: last.end, text };
|
|
1237
1603
|
}
|
|
1604
|
+
var CONTEXT_WINDOW = 64;
|
|
1605
|
+
function locateIsland(text, raw, from, expected, expectedFromEnd, oldText, oldStart, oldEnd) {
|
|
1606
|
+
const candidates = /* @__PURE__ */ new Set();
|
|
1607
|
+
if (expected >= from && text.startsWith(raw, expected)) candidates.add(expected);
|
|
1608
|
+
if (expectedFromEnd >= from && text.startsWith(raw, expectedFromEnd)) candidates.add(expectedFromEnd);
|
|
1609
|
+
const after = text.indexOf(raw, Math.max(from, expected));
|
|
1610
|
+
if (after !== -1) candidates.add(after);
|
|
1611
|
+
const before = expected > from ? text.lastIndexOf(raw, expected) : -1;
|
|
1612
|
+
if (before >= from) candidates.add(before);
|
|
1613
|
+
if (candidates.size === 0) return text.indexOf(raw, from);
|
|
1614
|
+
if (candidates.size === 1) return candidates.values().next().value;
|
|
1615
|
+
let best = -1;
|
|
1616
|
+
let bestScore = -1;
|
|
1617
|
+
for (const at of candidates) {
|
|
1618
|
+
let score = 0;
|
|
1619
|
+
for (let k = 0; k < CONTEXT_WINDOW; k++) {
|
|
1620
|
+
const n = text[at + raw.length + k];
|
|
1621
|
+
if (n === void 0 || n !== oldText[oldEnd + k]) break;
|
|
1622
|
+
score++;
|
|
1623
|
+
}
|
|
1624
|
+
for (let k = 1; k <= CONTEXT_WINDOW; k++) {
|
|
1625
|
+
const n = text[at - k];
|
|
1626
|
+
if (n === void 0 || n !== oldText[oldStart - k]) break;
|
|
1627
|
+
score++;
|
|
1628
|
+
}
|
|
1629
|
+
if (score > bestScore || score === bestScore && at === expected) {
|
|
1630
|
+
best = at;
|
|
1631
|
+
bestScore = score;
|
|
1632
|
+
}
|
|
1633
|
+
}
|
|
1634
|
+
return best;
|
|
1635
|
+
}
|
|
1238
1636
|
function commonPrefix(a, b) {
|
|
1239
1637
|
const max = Math.min(a.length, b.length);
|
|
1240
1638
|
let i = 0;
|
|
@@ -1259,13 +1657,23 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1259
1657
|
boundaries.add(block.start);
|
|
1260
1658
|
boundaries.add(block.end);
|
|
1261
1659
|
}
|
|
1660
|
+
const islands = listIslands(blocks);
|
|
1661
|
+
const islandAt = /* @__PURE__ */ new Map();
|
|
1662
|
+
for (const island of islands) islandAt.set(`${island.start}:${island.end}`, island);
|
|
1262
1663
|
const sorted = [...edits].sort((a, b) => a.start - b.start || a.end - b.end);
|
|
1263
1664
|
let previousEnd = -1;
|
|
1264
1665
|
for (const edit of sorted) {
|
|
1265
1666
|
if (edit.start < 0 || edit.end > original.length || edit.start > edit.end) {
|
|
1266
1667
|
throw new SourceSpliceError("out_of_range", `edit [${edit.start}, ${edit.end}) is outside the text`);
|
|
1267
1668
|
}
|
|
1268
|
-
if (
|
|
1669
|
+
if (edit.island) {
|
|
1670
|
+
if (!islandAt.has(`${edit.start}:${edit.end}`)) {
|
|
1671
|
+
throw new SourceSpliceError(
|
|
1672
|
+
"misaligned",
|
|
1673
|
+
`islandEdit [${edit.start}, ${edit.end}) does not name an island's exact span`
|
|
1674
|
+
);
|
|
1675
|
+
}
|
|
1676
|
+
} else if (!boundaries.has(edit.start) || !boundaries.has(edit.end)) {
|
|
1269
1677
|
throw new SourceSpliceError(
|
|
1270
1678
|
"misaligned",
|
|
1271
1679
|
`edit [${edit.start}, ${edit.end}) does not start and end on block boundaries`
|
|
@@ -1277,18 +1685,51 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1277
1685
|
previousEnd = edit.end;
|
|
1278
1686
|
}
|
|
1279
1687
|
const effective = [];
|
|
1280
|
-
|
|
1281
|
-
|
|
1282
|
-
|
|
1283
|
-
|
|
1284
|
-
const
|
|
1688
|
+
const replacedIslands = /* @__PURE__ */ new Set();
|
|
1689
|
+
const pushSegment = (oldStart, oldEnd, replacement) => {
|
|
1690
|
+
const old = original.slice(oldStart, oldEnd);
|
|
1691
|
+
if (old === replacement) return;
|
|
1692
|
+
const prefix = commonPrefix(old, replacement);
|
|
1693
|
+
const suffix = commonSuffix(old, replacement, prefix);
|
|
1285
1694
|
effective.push({
|
|
1286
|
-
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
oldEnd: edit.end - suffix,
|
|
1290
|
-
replacement: edit.text.slice(prefix, edit.text.length - suffix)
|
|
1695
|
+
oldStart: oldStart + prefix,
|
|
1696
|
+
oldEnd: oldEnd - suffix,
|
|
1697
|
+
replacement: replacement.slice(prefix, replacement.length - suffix)
|
|
1291
1698
|
});
|
|
1699
|
+
};
|
|
1700
|
+
for (const edit of sorted) {
|
|
1701
|
+
if (edit.island) {
|
|
1702
|
+
replacedIslands.add(`${edit.start}:${edit.end}`);
|
|
1703
|
+
pushSegment(edit.start, edit.end, edit.text);
|
|
1704
|
+
continue;
|
|
1705
|
+
}
|
|
1706
|
+
if (edit.text === original.slice(edit.start, edit.end)) continue;
|
|
1707
|
+
const inside = islands.filter((island) => island.start >= edit.start && island.end <= edit.end);
|
|
1708
|
+
let oldCursor2 = edit.start;
|
|
1709
|
+
let newCursor2 = 0;
|
|
1710
|
+
for (const island of inside) {
|
|
1711
|
+
const at = locateIsland(
|
|
1712
|
+
edit.text,
|
|
1713
|
+
island.raw,
|
|
1714
|
+
newCursor2,
|
|
1715
|
+
newCursor2 + (island.start - oldCursor2),
|
|
1716
|
+
edit.text.length - (edit.end - island.start),
|
|
1717
|
+
original.slice(edit.start, edit.end),
|
|
1718
|
+
island.start - edit.start,
|
|
1719
|
+
island.end - edit.start
|
|
1720
|
+
);
|
|
1721
|
+
if (at === -1) {
|
|
1722
|
+
const name = island.meta.tag ?? island.meta.name ?? island.meta.lang ?? "";
|
|
1723
|
+
throw new SourceSpliceError(
|
|
1724
|
+
"island_edit",
|
|
1725
|
+
`edit [${edit.start}, ${edit.end}) changes or removes the protected ${island.islandType}${name ? ` (${String(name)})` : ""} island at [${island.start}, ${island.end}); islands change only through islandEdit()`
|
|
1726
|
+
);
|
|
1727
|
+
}
|
|
1728
|
+
pushSegment(oldCursor2, island.start, edit.text.slice(newCursor2, at));
|
|
1729
|
+
oldCursor2 = island.end;
|
|
1730
|
+
newCursor2 = at + island.raw.length;
|
|
1731
|
+
}
|
|
1732
|
+
pushSegment(oldCursor2, edit.end, edit.text.slice(newCursor2));
|
|
1292
1733
|
}
|
|
1293
1734
|
if (effective.length === 0) {
|
|
1294
1735
|
return {
|
|
@@ -1308,9 +1749,10 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1308
1749
|
const newStart = edit.oldStart + delta;
|
|
1309
1750
|
text += edit.replacement;
|
|
1310
1751
|
const newEnd = newStart + edit.replacement.length;
|
|
1752
|
+
const owner = sorted.find((e) => e.start <= edit.oldStart && e.end >= edit.oldEnd);
|
|
1311
1753
|
pending.push({
|
|
1312
|
-
blockStart: edit.
|
|
1313
|
-
blockEnd: edit.
|
|
1754
|
+
blockStart: owner?.start ?? edit.oldStart,
|
|
1755
|
+
blockEnd: owner?.end ?? edit.oldEnd,
|
|
1314
1756
|
oldStart: edit.oldStart,
|
|
1315
1757
|
oldEnd: edit.oldEnd,
|
|
1316
1758
|
newStart,
|
|
@@ -1348,11 +1790,8 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1348
1790
|
else newIslandsByStart.set(island.start, [island.raw]);
|
|
1349
1791
|
}
|
|
1350
1792
|
const disturbed = [];
|
|
1351
|
-
for (const island of
|
|
1352
|
-
|
|
1353
|
-
(edit) => island.start < edit.blockEnd && island.end > edit.blockStart
|
|
1354
|
-
);
|
|
1355
|
-
if (touched) continue;
|
|
1793
|
+
for (const island of islands) {
|
|
1794
|
+
if (replacedIslands.has(`${island.start}:${island.end}`)) continue;
|
|
1356
1795
|
const mappedStart = mapPosition(changes, island.start, { assoc: 1 }).pos;
|
|
1357
1796
|
const candidates = newIslandsByStart.get(mappedStart);
|
|
1358
1797
|
if (!candidates || !candidates.includes(island.raw)) {
|
|
@@ -1370,10 +1809,11 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1370
1809
|
bytesOutsideEditsIdentical,
|
|
1371
1810
|
disturbed
|
|
1372
1811
|
};
|
|
1373
|
-
if (options.requireIntegrity && !integrity.ok) {
|
|
1812
|
+
if ((options.requireIntegrity ?? true) && !integrity.ok) {
|
|
1813
|
+
const named = disturbed.slice(0, 5).map((island) => `${island.islandType}@${island.start}`).join(", ");
|
|
1374
1814
|
throw new SourceSpliceError(
|
|
1375
1815
|
"integrity",
|
|
1376
|
-
`the
|
|
1816
|
+
`integrity: the save disturbed ${disturbed.length} protected island(s) it did not replace${named ? ` (${named}${disturbed.length > 5 ? ", \u2026" : ""})` : ""}${bytesOutsideEditsIdentical ? "" : "; bytes outside the edits changed"}`
|
|
1377
1817
|
);
|
|
1378
1818
|
}
|
|
1379
1819
|
return { text, changed: true, changes, integrity, blocks: newBlocks };
|
|
@@ -1415,17 +1855,24 @@ function mapRange(changes, start, end, unit = "utf16") {
|
|
|
1415
1855
|
return { start: newStart, end: newEnd, touched, collapsed: newEnd === newStart && end > start };
|
|
1416
1856
|
}
|
|
1417
1857
|
|
|
1858
|
+
exports.FENCE_WHITESPACE = FENCE_WHITESPACE;
|
|
1859
|
+
exports.NESTING_FENCE_LANGUAGES = NESTING_FENCE_LANGUAGES;
|
|
1418
1860
|
exports.SourceSpliceError = SourceSpliceError;
|
|
1419
1861
|
exports.XmlContainerTracker = XmlContainerTracker;
|
|
1420
1862
|
exports.blockAt = blockAt;
|
|
1421
1863
|
exports.blockEdit = blockEdit;
|
|
1422
1864
|
exports.blockRangeEdit = blockRangeEdit;
|
|
1423
1865
|
exports.buildCodePointIndex = buildCodePointIndex;
|
|
1866
|
+
exports.classifyInnerFenceLine = classifyInnerFenceLine;
|
|
1867
|
+
exports.fenceNestsInnerFences = fenceNestsInnerFences;
|
|
1424
1868
|
exports.isSingleDollarMath = isSingleDollarMath;
|
|
1869
|
+
exports.islandEdit = islandEdit;
|
|
1425
1870
|
exports.joinSource = joinSource;
|
|
1426
1871
|
exports.listIslands = listIslands;
|
|
1872
|
+
exports.looksLikeDisplayMath = looksLikeDisplayMath;
|
|
1427
1873
|
exports.mapPosition = mapPosition;
|
|
1428
1874
|
exports.mapRange = mapRange;
|
|
1875
|
+
exports.pairDisplayMath = pairDisplayMath;
|
|
1429
1876
|
exports.readXmlTag = readXmlTag;
|
|
1430
1877
|
exports.singleDollarMathEnd = singleDollarMathEnd;
|
|
1431
1878
|
exports.spliceSave = spliceSave;
|
|
@@ -1433,5 +1880,6 @@ exports.splitsSurrogatePair = splitsSurrogatePair;
|
|
|
1433
1880
|
exports.toCodePointOffset = toCodePointOffset;
|
|
1434
1881
|
exports.toUtf16Offset = toUtf16Offset;
|
|
1435
1882
|
exports.tokenizeSource = tokenizeSource;
|
|
1883
|
+
exports.trimFenceLine = trimFenceLine;
|
|
1436
1884
|
//# sourceMappingURL=source.cjs.map
|
|
1437
1885
|
//# sourceMappingURL=source.cjs.map
|