@ai-matrx/content-ir 0.12.0 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +119 -0
- package/dist/index.cjs +546 -98
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.js +539 -99
- package/dist/index.js.map +1 -1
- package/dist/source.cjs +546 -98
- package/dist/source.cjs.map +1 -1
- package/dist/source.d.cts +118 -14
- package/dist/source.d.ts +118 -14
- package/dist/source.js +539 -99
- package/dist/source.js.map +1 -1
- package/package.json +1 -1
package/dist/source.js
CHANGED
|
@@ -200,20 +200,84 @@ function normalizeFenceLanguage(language) {
|
|
|
200
200
|
return CODE_LANGUAGE_ALIASES[lower] ?? lower;
|
|
201
201
|
}
|
|
202
202
|
|
|
203
|
+
// source/fence-nesting.ts
|
|
204
|
+
var NESTING_FENCE_LANGUAGES = /* @__PURE__ */ new Set([
|
|
205
|
+
"markdown",
|
|
206
|
+
"md",
|
|
207
|
+
"mdx"
|
|
208
|
+
]);
|
|
209
|
+
function fenceNestsInnerFences(language) {
|
|
210
|
+
return !!language && NESTING_FENCE_LANGUAGES.has(language.toLowerCase());
|
|
211
|
+
}
|
|
212
|
+
var FENCE_WHITESPACE = " \n\r\f\v";
|
|
213
|
+
function isFenceWhitespace(ch) {
|
|
214
|
+
return ch !== void 0 && ch !== "" && FENCE_WHITESPACE.includes(ch);
|
|
215
|
+
}
|
|
216
|
+
function trimFenceLine(line) {
|
|
217
|
+
let start = 0;
|
|
218
|
+
let end = line.length;
|
|
219
|
+
while (start < end && isFenceWhitespace(line[start])) start++;
|
|
220
|
+
while (end > start && isFenceWhitespace(line[end - 1])) end--;
|
|
221
|
+
return start === 0 && end === line.length ? line : line.slice(start, end);
|
|
222
|
+
}
|
|
223
|
+
function classifyInnerFenceLine(trimmed, openTicks, nests, nestedDepth) {
|
|
224
|
+
let ticks = 0;
|
|
225
|
+
while (ticks < trimmed.length && trimmed[ticks] === "`") ticks++;
|
|
226
|
+
if (ticks < 3) return "content";
|
|
227
|
+
const info = trimFenceLine(trimmed.slice(ticks));
|
|
228
|
+
if (info === "") {
|
|
229
|
+
if (ticks < openTicks) return "content";
|
|
230
|
+
if (nests && nestedDepth > 0) return "close-nested";
|
|
231
|
+
return "close-outer";
|
|
232
|
+
}
|
|
233
|
+
if (nests && ticks >= openTicks && !info.includes("`")) return "open-nested";
|
|
234
|
+
return "content";
|
|
235
|
+
}
|
|
236
|
+
|
|
203
237
|
// source/math.ts
|
|
204
238
|
var TEX_SIGNAL = /\\[A-Za-z]+|[\\^_{}=<>+]/;
|
|
205
239
|
var LONE_VARIABLE = /^[A-Za-z](?:'|[0-9])?$/;
|
|
240
|
+
var MATH_SHAPE_CHARS = /^[A-Za-z0-9'\s(),.*\/-]+$/;
|
|
241
|
+
var FUNCTION_NAMES = /* @__PURE__ */ new Set([
|
|
242
|
+
"sin",
|
|
243
|
+
"cos",
|
|
244
|
+
"tan",
|
|
245
|
+
"sec",
|
|
246
|
+
"csc",
|
|
247
|
+
"cot",
|
|
248
|
+
"log",
|
|
249
|
+
"ln",
|
|
250
|
+
"exp",
|
|
251
|
+
"lim",
|
|
252
|
+
"max",
|
|
253
|
+
"min",
|
|
254
|
+
"det",
|
|
255
|
+
"gcd",
|
|
256
|
+
"mod",
|
|
257
|
+
"arg",
|
|
258
|
+
"sinh",
|
|
259
|
+
"cosh",
|
|
260
|
+
"tanh"
|
|
261
|
+
]);
|
|
262
|
+
function isMathShape(content) {
|
|
263
|
+
if (!MATH_SHAPE_CHARS.test(content)) return false;
|
|
264
|
+
const words = content.match(/[A-Za-z]+/g);
|
|
265
|
+
if (!words) return false;
|
|
266
|
+
if (!words.every((w) => w.length === 1 || FUNCTION_NAMES.has(w))) return false;
|
|
267
|
+
return /[A-Za-z]'*\(|[A-Za-z]'|\b[A-Za-z]\s+[A-Za-z]\b|\([^)]*[A-Za-z][^)]*\)/.test(content);
|
|
268
|
+
}
|
|
206
269
|
function isSingleDollarMath(content, after) {
|
|
207
270
|
if (!content || content.length > 400) return false;
|
|
208
271
|
if (/^\s/.test(content) || /\s$/.test(content)) return false;
|
|
209
272
|
if (after !== void 0 && /[0-9]/.test(after)) return false;
|
|
210
273
|
if (LONE_VARIABLE.test(content)) return true;
|
|
211
274
|
if (/^[0-9]/.test(content)) return /\\[A-Za-z]+|[\\^_{}]/.test(content);
|
|
212
|
-
return TEX_SIGNAL.test(content);
|
|
275
|
+
return TEX_SIGNAL.test(content) || isMathShape(content);
|
|
213
276
|
}
|
|
214
277
|
function singleDollarMathEnd(text, open, limit = text.length) {
|
|
278
|
+
const bound = Math.min(limit, open + 403);
|
|
215
279
|
let close = -1;
|
|
216
|
-
for (let k = open + 1; k <
|
|
280
|
+
for (let k = open + 1; k < bound; k += 1) {
|
|
217
281
|
const c = text[k];
|
|
218
282
|
if (c === "\n") break;
|
|
219
283
|
if (c === "\\") {
|
|
@@ -230,11 +294,69 @@ function singleDollarMathEnd(text, open, limit = text.length) {
|
|
|
230
294
|
return isSingleDollarMath(content, text[close + 1]) ? close + 1 : -1;
|
|
231
295
|
}
|
|
232
296
|
|
|
297
|
+
// source/math-pairs.ts
|
|
298
|
+
var MAX_MATH_SPAN = 600;
|
|
299
|
+
var STRUCTURAL_MARKDOWN = /\]\(|https?:\/\/|\*\*|(?:^|\n)[ \t]{0,3}#{1,6}[ \t]|(?:^|\n)[ \t]*[-*+][ \t]+|(?:^|\n)[ \t]*\d+[.)][ \t]/;
|
|
300
|
+
var LATEX_COMMAND = /\\[a-zA-Z]/;
|
|
301
|
+
var PROSE_WORD = /[A-Za-z]{3,}/g;
|
|
302
|
+
var PROSE_WORD_LIMIT = 6;
|
|
303
|
+
function looksLikeDisplayMath(inner) {
|
|
304
|
+
const s = inner.trim();
|
|
305
|
+
if (!s) return false;
|
|
306
|
+
if (STRUCTURAL_MARKDOWN.test(s)) return false;
|
|
307
|
+
if (s.length > MAX_MATH_SPAN) return false;
|
|
308
|
+
if (LATEX_COMMAND.test(s)) {
|
|
309
|
+
return (s.match(PROSE_WORD) ?? []).length < PROSE_WORD_LIMIT * 3;
|
|
310
|
+
}
|
|
311
|
+
if (/\n[ \t]*\n/.test(s)) return false;
|
|
312
|
+
return (s.match(PROSE_WORD) ?? []).length < PROSE_WORD_LIMIT;
|
|
313
|
+
}
|
|
314
|
+
function codeRanges(text) {
|
|
315
|
+
const ranges = [];
|
|
316
|
+
for (const re of [/```[\s\S]*?(?:```|$)/g, /~~~[\s\S]*?(?:~~~|$)/g, /`[^`\n]*`/g]) {
|
|
317
|
+
let m;
|
|
318
|
+
while ((m = re.exec(text)) !== null) {
|
|
319
|
+
ranges.push([m.index, m.index + m[0].length]);
|
|
320
|
+
if (m[0].length === 0) re.lastIndex++;
|
|
321
|
+
}
|
|
322
|
+
}
|
|
323
|
+
return ranges.sort((a, b) => a[0] - b[0]);
|
|
324
|
+
}
|
|
325
|
+
function pairDisplayMath(text) {
|
|
326
|
+
const pairs = /* @__PURE__ */ new Map();
|
|
327
|
+
if (!text.includes("$$")) return pairs;
|
|
328
|
+
const ranges = codeRanges(text);
|
|
329
|
+
const tokens = [];
|
|
330
|
+
let r = 0;
|
|
331
|
+
let maxEnd = -1;
|
|
332
|
+
for (let i = text.indexOf("$$"); i !== -1 && i < text.length - 1; i = text.indexOf("$$", i)) {
|
|
333
|
+
while (r < ranges.length && ranges[r][0] <= i) {
|
|
334
|
+
maxEnd = Math.max(maxEnd, ranges[r][1]);
|
|
335
|
+
r++;
|
|
336
|
+
}
|
|
337
|
+
if (maxEnd <= i) tokens.push(i);
|
|
338
|
+
i += 2;
|
|
339
|
+
}
|
|
340
|
+
let j = 0;
|
|
341
|
+
while (j < tokens.length) {
|
|
342
|
+
const open = tokens[j];
|
|
343
|
+
const close = tokens[j + 1];
|
|
344
|
+
if (close === void 0) break;
|
|
345
|
+
if (looksLikeDisplayMath(text.slice(open + 2, close))) {
|
|
346
|
+
pairs.set(open, close);
|
|
347
|
+
j += 2;
|
|
348
|
+
continue;
|
|
349
|
+
}
|
|
350
|
+
j += 1;
|
|
351
|
+
}
|
|
352
|
+
return pairs;
|
|
353
|
+
}
|
|
354
|
+
|
|
233
355
|
// source/xml-tag.ts
|
|
234
356
|
var XML_NAME_START = /[A-Za-z_]/;
|
|
235
357
|
var XML_NAME_CHARACTER = /[\w.:-]/;
|
|
236
358
|
var WHITESPACE = /\s/;
|
|
237
|
-
function readXmlTag(content, start) {
|
|
359
|
+
function readXmlTag(content, start, find = (needle, from) => content.indexOf(needle, from)) {
|
|
238
360
|
let cursor = start + 1;
|
|
239
361
|
const isClosing = content[cursor] === "/";
|
|
240
362
|
if (isClosing) cursor++;
|
|
@@ -287,8 +409,9 @@ function readXmlTag(content, start) {
|
|
|
287
409
|
const quote = content[cursor];
|
|
288
410
|
if (quote !== '"' && quote !== "'") return null;
|
|
289
411
|
const valueStart = ++cursor;
|
|
290
|
-
|
|
291
|
-
if (
|
|
412
|
+
const valueEnd = find(quote, valueStart);
|
|
413
|
+
if (valueEnd === -1) return null;
|
|
414
|
+
cursor = valueEnd;
|
|
292
415
|
attributes.push({ name, value: content.slice(valueStart, cursor) });
|
|
293
416
|
cursor++;
|
|
294
417
|
}
|
|
@@ -385,6 +508,10 @@ var XmlContainerTracker = class {
|
|
|
385
508
|
|
|
386
509
|
// source/tokenize.ts
|
|
387
510
|
var NOT_WHITESPACE = /\S/;
|
|
511
|
+
var DIRECTIVE_OPEN = /^(:{2,})([A-Za-z][\w-]*)/;
|
|
512
|
+
var MKDOCS_OPEN = /^(!!!|\?\?\?\+?)[ \t]+([A-Za-z][\w-]*)/;
|
|
513
|
+
var CALLOUT_OPEN = /^>[ \t]*\[!([A-Za-z][\w-]*)\]([+-])?/;
|
|
514
|
+
var FOOTNOTE_DEF = /^\[\^([^\]\s]+)\]:/;
|
|
388
515
|
function countStructuralBraces(source) {
|
|
389
516
|
let opens = 0;
|
|
390
517
|
let closes = 0;
|
|
@@ -507,11 +634,16 @@ function backtickRunLength(str, pos) {
|
|
|
507
634
|
while (pos + n < str.length && str[pos + n] === "`") n++;
|
|
508
635
|
return n;
|
|
509
636
|
}
|
|
637
|
+
function isSplitterTreeLine(line) {
|
|
638
|
+
return TREE_GLYPHS.test(line) || /^[\s│|]*[├└+|][\s─\-]+/.test(line);
|
|
639
|
+
}
|
|
510
640
|
var MEDIA_REF_PREFIXES = ["[Image URL:", "[Video URL:", "[Audio URL:"];
|
|
511
641
|
var Tokenizer = class {
|
|
512
|
-
constructor(text) {
|
|
642
|
+
constructor(text, streaming = false) {
|
|
513
643
|
this.text = text;
|
|
644
|
+
this.streaming = streaming;
|
|
514
645
|
this.n = text.length;
|
|
646
|
+
this.mathPairs = pairDisplayMath(text);
|
|
515
647
|
let start = 0;
|
|
516
648
|
for (let i = 0; i < text.length; i++) {
|
|
517
649
|
if (text.charCodeAt(i) === 10) {
|
|
@@ -528,6 +660,7 @@ var Tokenizer = class {
|
|
|
528
660
|
}
|
|
529
661
|
}
|
|
530
662
|
text;
|
|
663
|
+
streaming;
|
|
531
664
|
n;
|
|
532
665
|
lineStarts = [];
|
|
533
666
|
contentEnds = [];
|
|
@@ -537,6 +670,38 @@ var Tokenizer = class {
|
|
|
537
670
|
/** Lazily built: for line i, first line j >= i where brace net from i drops to <= 0. */
|
|
538
671
|
braceStop = null;
|
|
539
672
|
braceNet = null;
|
|
673
|
+
/** Every occurrence offset of each needle searched so far (overlapping, like indexOf). */
|
|
674
|
+
occurrences = /* @__PURE__ */ new Map();
|
|
675
|
+
/** `$$` opener → closer for the pairs the core renders as math. */
|
|
676
|
+
mathPairs;
|
|
677
|
+
finder = (needle, from) => this.next(needle, from);
|
|
678
|
+
/**
|
|
679
|
+
* `text.indexOf(needle, from)` in O(log k): the first query for a needle
|
|
680
|
+
* indexes all of its occurrences once, so repeated searches for a closer
|
|
681
|
+
* that never comes stay linear over the whole text.
|
|
682
|
+
*/
|
|
683
|
+
next(needle, from) {
|
|
684
|
+
let list = this.occurrences.get(needle);
|
|
685
|
+
if (!list) {
|
|
686
|
+
list = [];
|
|
687
|
+
for (let at = this.text.indexOf(needle); at !== -1; at = this.text.indexOf(needle, at + 1)) {
|
|
688
|
+
list.push(at);
|
|
689
|
+
}
|
|
690
|
+
this.occurrences.set(needle, list);
|
|
691
|
+
}
|
|
692
|
+
let lo = 0;
|
|
693
|
+
let hi = list.length;
|
|
694
|
+
while (lo < hi) {
|
|
695
|
+
const mid = lo + hi >>> 1;
|
|
696
|
+
if (list[mid] < from) lo = mid + 1;
|
|
697
|
+
else hi = mid;
|
|
698
|
+
}
|
|
699
|
+
return lo < list.length ? list[lo] : -1;
|
|
700
|
+
}
|
|
701
|
+
/** Streaming only: the final line, still waiting for its terminator. */
|
|
702
|
+
isOpenTail(k) {
|
|
703
|
+
return this.streaming && k === this.lineStarts.length - 1 && this.lineEnds[k] === this.contentEnds[k];
|
|
704
|
+
}
|
|
540
705
|
get lineCount() {
|
|
541
706
|
return this.lineStarts.length;
|
|
542
707
|
}
|
|
@@ -618,6 +783,22 @@ var Tokenizer = class {
|
|
|
618
783
|
const fm = this.frontMatter(li, t);
|
|
619
784
|
if (fm) return fm;
|
|
620
785
|
}
|
|
786
|
+
if (indent <= 3 && t.startsWith("::")) {
|
|
787
|
+
const directive = this.directive(p, li, t);
|
|
788
|
+
if (directive) return directive;
|
|
789
|
+
}
|
|
790
|
+
if (indent === 0 && (t.startsWith("!!!") || t.startsWith("???"))) {
|
|
791
|
+
const admonition = this.mkdocsAdmonition(p, li, t);
|
|
792
|
+
if (admonition) return admonition;
|
|
793
|
+
}
|
|
794
|
+
if (indent <= 3 && t.startsWith(">")) {
|
|
795
|
+
const callout = this.callout(p, li, t);
|
|
796
|
+
if (callout) return callout;
|
|
797
|
+
}
|
|
798
|
+
if (atParaStart && indent <= 3 && t.startsWith("[^")) {
|
|
799
|
+
const footnote = this.footnoteDefinition(p, li, t);
|
|
800
|
+
if (footnote) return footnote;
|
|
801
|
+
}
|
|
621
802
|
if (t.startsWith("```")) return this.backtickFence(p, li, t);
|
|
622
803
|
if (indent <= 3 && t.startsWith("~~~")) {
|
|
623
804
|
const tilde = this.tildeFence(p, li, t);
|
|
@@ -655,20 +836,116 @@ var Tokenizer = class {
|
|
|
655
836
|
return { type, start, end, complete, meta };
|
|
656
837
|
}
|
|
657
838
|
frontMatter(li, t) {
|
|
658
|
-
|
|
839
|
+
const fence = this.lineText(li);
|
|
840
|
+
if (fence !== "---" && fence !== "+++" || t !== fence) return null;
|
|
841
|
+
const toml = fence === "+++";
|
|
659
842
|
let sawContent = false;
|
|
660
843
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
661
844
|
const line = this.lineText(k);
|
|
662
|
-
if (line ===
|
|
663
|
-
return sawContent ? this.island("front_matter", 0, this.contentEnds[k], true) : null;
|
|
845
|
+
if (line === fence || !toml && line === "...") {
|
|
846
|
+
return sawContent ? this.island("front_matter", 0, this.contentEnds[k], true, { format: toml ? "toml" : "yaml" }) : null;
|
|
664
847
|
}
|
|
665
848
|
if (!NOT_WHITESPACE.test(line)) continue;
|
|
666
|
-
|
|
849
|
+
const shaped = toml ? /^\s*(?:[A-Za-z0-9_."-]+\s*=|\[|#)/.test(line) : /^(?:[A-Za-z0-9_"'-][^:]*:(?:\s|$)|\s+\S|-\s|#)/.test(line);
|
|
850
|
+
if (!shaped) return null;
|
|
667
851
|
sawContent = true;
|
|
668
852
|
}
|
|
669
853
|
return null;
|
|
670
854
|
}
|
|
671
|
-
/**
|
|
855
|
+
/**
|
|
856
|
+
* `:::name[label]{attrs}` … closing `:::` (a colon run at least as long as
|
|
857
|
+
* the opener's, nesting counted), or a leaf `::name[…]{…}` line. The whole
|
|
858
|
+
* container is ONE island: its grammar (fence lengths, labels, attributes,
|
|
859
|
+
* nested tabs) is exactly what a visual editor would re-serialize wrongly.
|
|
860
|
+
*/
|
|
861
|
+
directive(p, li, t) {
|
|
862
|
+
const open = DIRECTIVE_OPEN.exec(t);
|
|
863
|
+
if (!open) return null;
|
|
864
|
+
const colons = open[1].length;
|
|
865
|
+
const name = open[2];
|
|
866
|
+
if (colons === 2) {
|
|
867
|
+
return this.island("directive", p, this.contentEnds[li], true, { name, leaf: true });
|
|
868
|
+
}
|
|
869
|
+
const stack = [colons];
|
|
870
|
+
for (let k = li + 1; k < this.lineCount; k++) {
|
|
871
|
+
const line = this.lineText(k).trim();
|
|
872
|
+
const inner = DIRECTIVE_OPEN.exec(line);
|
|
873
|
+
if (inner && inner[1].length >= 3) {
|
|
874
|
+
stack.push(inner[1].length);
|
|
875
|
+
continue;
|
|
876
|
+
}
|
|
877
|
+
const close = /^(:{3,})\s*$/.exec(line);
|
|
878
|
+
if (close && close[1].length >= stack[stack.length - 1]) {
|
|
879
|
+
stack.pop();
|
|
880
|
+
if (stack.length === 0) return this.island("directive", p, this.contentEnds[k], true, { name });
|
|
881
|
+
}
|
|
882
|
+
}
|
|
883
|
+
return this.island("directive", p, this.n, false, { name });
|
|
884
|
+
}
|
|
885
|
+
/** MkDocs `!!! type "Title"` / `??? type` / `???+ type` and its indented (or blank-separated indented) body. */
|
|
886
|
+
mkdocsAdmonition(p, li, t) {
|
|
887
|
+
const open = MKDOCS_OPEN.exec(t);
|
|
888
|
+
if (!open) return null;
|
|
889
|
+
let last = li;
|
|
890
|
+
for (let k = li + 1; k < this.lineCount; k++) {
|
|
891
|
+
const line = this.lineText(k);
|
|
892
|
+
if (/^(?: {4}|\t)/.test(line) && NOT_WHITESPACE.test(line)) {
|
|
893
|
+
last = k;
|
|
894
|
+
continue;
|
|
895
|
+
}
|
|
896
|
+
if (!NOT_WHITESPACE.test(line)) continue;
|
|
897
|
+
break;
|
|
898
|
+
}
|
|
899
|
+
return this.island("directive", p, this.contentEnds[last], true, {
|
|
900
|
+
name: open[2].toLowerCase(),
|
|
901
|
+
syntax: "mkdocs"
|
|
902
|
+
});
|
|
903
|
+
}
|
|
904
|
+
/** A blockquote whose first line is `> [!TYPE]…` — every following `>` line belongs to it. */
|
|
905
|
+
callout(p, li, t) {
|
|
906
|
+
const open = CALLOUT_OPEN.exec(t);
|
|
907
|
+
if (!open) return null;
|
|
908
|
+
let last = li;
|
|
909
|
+
for (let k = li + 1; k < this.lineCount; k++) {
|
|
910
|
+
if (!/^ {0,3}>/.test(this.lineText(k))) break;
|
|
911
|
+
last = k;
|
|
912
|
+
}
|
|
913
|
+
return this.island("callout", p, this.contentEnds[last], true, {
|
|
914
|
+
name: open[1].toLowerCase(),
|
|
915
|
+
...open[2] ? { fold: open[2] === "-" ? "closed" : "open" } : {}
|
|
916
|
+
});
|
|
917
|
+
}
|
|
918
|
+
/** `[^id]: text` plus its continuation lines (indented, or lazy text before a blank line). */
|
|
919
|
+
footnoteDefinition(p, li, t) {
|
|
920
|
+
const open = FOOTNOTE_DEF.exec(t);
|
|
921
|
+
if (!open) return null;
|
|
922
|
+
let last = li;
|
|
923
|
+
for (let k = li + 1; k < this.lineCount; k++) {
|
|
924
|
+
const line = this.lineText(k);
|
|
925
|
+
if (!NOT_WHITESPACE.test(line)) {
|
|
926
|
+
let j = k + 1;
|
|
927
|
+
while (j < this.lineCount && !NOT_WHITESPACE.test(this.lineText(j))) j++;
|
|
928
|
+
if (j < this.lineCount && /^(?: {4}|\t)/.test(this.lineText(j))) {
|
|
929
|
+
k = j - 1;
|
|
930
|
+
continue;
|
|
931
|
+
}
|
|
932
|
+
break;
|
|
933
|
+
}
|
|
934
|
+
if (/^(?: {4}|\t)/.test(line)) {
|
|
935
|
+
last = k;
|
|
936
|
+
continue;
|
|
937
|
+
}
|
|
938
|
+
if (FOOTNOTE_DEF.test(line.trimStart()) || this.detectBlock(this.lineStarts[k], k, false)) break;
|
|
939
|
+
last = k;
|
|
940
|
+
}
|
|
941
|
+
return this.island("footnote_def", p, this.contentEnds[last], true, { id: open[1] });
|
|
942
|
+
}
|
|
943
|
+
/**
|
|
944
|
+
* Backtick fence — the renderer splitter's close rules: JSON-string aware for
|
|
945
|
+
* ```json, and THE nested-fence rule (source/fence-nesting.ts) for markdown
|
|
946
|
+
* fences, with the splitter's strict-CommonMark retry when a nested fence is
|
|
947
|
+
* still open at the end of the text.
|
|
948
|
+
*/
|
|
672
949
|
backtickFence(p, li, t) {
|
|
673
950
|
const openTicks = backtickRunLength(t, 0);
|
|
674
951
|
const info = t.slice(openTicks).trim();
|
|
@@ -677,10 +954,23 @@ var Tokenizer = class {
|
|
|
677
954
|
const normalized = normalizeFenceLanguage(lang);
|
|
678
955
|
const meta = { fence: "`", ticks: openTicks, lang };
|
|
679
956
|
if (normalized && SPECIAL_CODE_LANGUAGES.has(normalized)) meta.special = normalized;
|
|
957
|
+
const nesting = fenceNestsInnerFences(lang);
|
|
958
|
+
let close = this.backtickFenceClose(li, openTicks, isJson, nesting);
|
|
959
|
+
if (close === "retry-strict") close = this.backtickFenceClose(li, openTicks, isJson, false);
|
|
960
|
+
if (typeof close === "number") {
|
|
961
|
+
return this.fenceIsland(p, this.contentEnds[close], true, meta, li, close, isJson);
|
|
962
|
+
}
|
|
963
|
+
return this.fenceIsland(p, this.n, false, meta, li, this.lineCount, isJson);
|
|
964
|
+
}
|
|
965
|
+
/** The closing line of a backtick fence opened on line `li`, or null when it never closes. */
|
|
966
|
+
backtickFenceClose(li, openTicks, isJson, nesting) {
|
|
680
967
|
let state = { inString: false, escaped: false };
|
|
968
|
+
let nestedDepth = 0;
|
|
969
|
+
let sawNested = false;
|
|
681
970
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
971
|
+
if (this.isOpenTail(k)) continue;
|
|
682
972
|
const line = this.lineText(k);
|
|
683
|
-
const trimmedLine = line
|
|
973
|
+
const trimmedLine = trimFenceLine(line);
|
|
684
974
|
if (trimmedLine.startsWith("```")) {
|
|
685
975
|
const at2 = line.indexOf("```");
|
|
686
976
|
if (isJson) {
|
|
@@ -690,11 +980,13 @@ var Tokenizer = class {
|
|
|
690
980
|
continue;
|
|
691
981
|
}
|
|
692
982
|
}
|
|
693
|
-
const
|
|
694
|
-
|
|
695
|
-
if (
|
|
696
|
-
|
|
983
|
+
const kind = classifyInnerFenceLine(trimmedLine, openTicks, nesting, nestedDepth);
|
|
984
|
+
if (kind === "close-outer") return k;
|
|
985
|
+
if (kind === "open-nested") {
|
|
986
|
+
nestedDepth++;
|
|
987
|
+
sawNested = true;
|
|
697
988
|
}
|
|
989
|
+
if (kind === "close-nested") nestedDepth--;
|
|
698
990
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
699
991
|
continue;
|
|
700
992
|
}
|
|
@@ -708,16 +1000,14 @@ var Tokenizer = class {
|
|
|
708
1000
|
}
|
|
709
1001
|
}
|
|
710
1002
|
const closeTicks = backtickRunLength(line, at);
|
|
711
|
-
const after = line.slice(at + closeTicks)
|
|
712
|
-
if (closeTicks >= openTicks && after === "")
|
|
713
|
-
return this.fenceIsland(p, this.contentEnds[k], true, meta, li, k, isJson);
|
|
714
|
-
}
|
|
1003
|
+
const after = trimFenceLine(line.slice(at + closeTicks));
|
|
1004
|
+
if (closeTicks >= openTicks && after === "" && nestedDepth === 0) return k;
|
|
715
1005
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
716
1006
|
continue;
|
|
717
1007
|
}
|
|
718
1008
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
719
1009
|
}
|
|
720
|
-
return
|
|
1010
|
+
return sawNested && nestedDepth > 0 ? "retry-strict" : null;
|
|
721
1011
|
}
|
|
722
1012
|
fenceIsland(p, end, complete, meta, openLine, closeLine, isJson) {
|
|
723
1013
|
if (isJson && openLine + 1 < this.lineCount) {
|
|
@@ -737,24 +1027,29 @@ var Tokenizer = class {
|
|
|
737
1027
|
if (normalized && SPECIAL_CODE_LANGUAGES.has(normalized)) meta.special = normalized;
|
|
738
1028
|
const close = new RegExp(`^ {0,3}~{${ticks},}[ \\t]*$`);
|
|
739
1029
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
1030
|
+
if (this.isOpenTail(k)) continue;
|
|
740
1031
|
if (close.test(this.lineText(k))) {
|
|
741
1032
|
return this.island("fence", p, this.contentEnds[k], true, meta);
|
|
742
1033
|
}
|
|
743
1034
|
}
|
|
744
1035
|
return this.island("fence", p, this.n, false, meta);
|
|
745
1036
|
}
|
|
1037
|
+
/**
|
|
1038
|
+
* A line-leading `$$` that OPENS a math pair (source/math-pairs.ts — the
|
|
1039
|
+
* core's rule). A closer, an unpaired `$$`, or a prose pair is never a block:
|
|
1040
|
+
* a lone `$$` renders as literal text and locks nothing.
|
|
1041
|
+
*/
|
|
746
1042
|
mathBlock(p, tp) {
|
|
1043
|
+
const close = this.mathPairs.get(tp);
|
|
1044
|
+
if (close === void 0) return null;
|
|
1045
|
+
const end = close + 2;
|
|
747
1046
|
const li = this.lineOf(tp);
|
|
748
1047
|
const ce = this.contentEnds[li];
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
}
|
|
755
|
-
const close = this.text.indexOf("$$", tp + 2);
|
|
756
|
-
if (close === -1) return this.island("math_block", p, this.n, false, { delimiter: "$$" });
|
|
757
|
-
return this.island("math_block", p, close + 2, true, { delimiter: "$$" });
|
|
1048
|
+
if (end <= ce && !this.isBlank(end, ce)) return null;
|
|
1049
|
+
const endLine = this.lineOf(close);
|
|
1050
|
+
const endCe = this.contentEnds[endLine];
|
|
1051
|
+
if (endLine !== li && !this.isBlank(end, endCe)) return null;
|
|
1052
|
+
return this.island("math_block", p, end, true, { delimiter: "$$" });
|
|
758
1053
|
}
|
|
759
1054
|
/** `<artifact …>`, `<decision …>`, … at the start of a line. */
|
|
760
1055
|
attributeXml(p, tp, t) {
|
|
@@ -771,7 +1066,7 @@ var Tokenizer = class {
|
|
|
771
1066
|
attributeElement(name, p, tp, type) {
|
|
772
1067
|
const li = this.lineOf(tp);
|
|
773
1068
|
const ce = this.contentEnds[li];
|
|
774
|
-
const tag = readXmlTag(this.text, tp);
|
|
1069
|
+
const tag = readXmlTag(this.text, tp, this.finder);
|
|
775
1070
|
let openEnd;
|
|
776
1071
|
if (tag && !tag.isClosing) {
|
|
777
1072
|
if (tag.isSelfClosing) {
|
|
@@ -779,13 +1074,13 @@ var Tokenizer = class {
|
|
|
779
1074
|
}
|
|
780
1075
|
openEnd = tp + tag.raw.length;
|
|
781
1076
|
} else {
|
|
782
|
-
const gt = this.
|
|
1077
|
+
const gt = this.next(">", tp);
|
|
783
1078
|
if (gt === -1) return this.island(type, p, this.n, false, { tag: name });
|
|
784
1079
|
if (gt >= ce) return null;
|
|
785
1080
|
openEnd = gt + 1;
|
|
786
1081
|
}
|
|
787
1082
|
const closer = `</${name}>`;
|
|
788
|
-
const close = this.
|
|
1083
|
+
const close = this.next(closer, openEnd);
|
|
789
1084
|
if (close === -1) return this.island(type, p, this.n, false, { tag: name });
|
|
790
1085
|
return this.island(type, p, close + closer.length, true, { tag: name });
|
|
791
1086
|
}
|
|
@@ -795,7 +1090,7 @@ var Tokenizer = class {
|
|
|
795
1090
|
const opener = `<${name}>`;
|
|
796
1091
|
if (!t.startsWith(opener)) continue;
|
|
797
1092
|
const closer = `</${name}>`;
|
|
798
|
-
const close = this.
|
|
1093
|
+
const close = this.next(closer, tp + opener.length);
|
|
799
1094
|
if (close === -1) return this.island("xml_region", p, this.n, false, { tag: name });
|
|
800
1095
|
return this.island("xml_region", p, close + closer.length, true, { tag: name });
|
|
801
1096
|
}
|
|
@@ -831,7 +1126,7 @@ var Tokenizer = class {
|
|
|
831
1126
|
return null;
|
|
832
1127
|
}
|
|
833
1128
|
blockComment(p, tp, li) {
|
|
834
|
-
const close = this.
|
|
1129
|
+
const close = this.next("-->", tp + 4);
|
|
835
1130
|
if (close === -1) return this.island("html_comment", p, this.n, false);
|
|
836
1131
|
const end = close + 3;
|
|
837
1132
|
const raw = this.text.slice(tp, end);
|
|
@@ -909,9 +1204,8 @@ var Tokenizer = class {
|
|
|
909
1204
|
}
|
|
910
1205
|
}
|
|
911
1206
|
if (lastLine === null) {
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
return this.island("json", p, this.n, false, jsonMeta(partial));
|
|
1207
|
+
if (isRecognizedJsonHead(this.text.slice(tp, tp + 512))) {
|
|
1208
|
+
return this.island("json", p, this.n, false, jsonMeta(this.text.slice(tp)));
|
|
915
1209
|
}
|
|
916
1210
|
return null;
|
|
917
1211
|
}
|
|
@@ -946,11 +1240,43 @@ var Tokenizer = class {
|
|
|
946
1240
|
// ── prose ──────────────────────────────────────────────────────────────
|
|
947
1241
|
paragraph(pos, li) {
|
|
948
1242
|
let end = this.contentEnds[li];
|
|
1243
|
+
let breakLine = -1;
|
|
949
1244
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
950
1245
|
if (this.lineIsBlank(k)) break;
|
|
951
|
-
if (this.detectBlock(this.lineStarts[k], k, false))
|
|
1246
|
+
if (this.detectBlock(this.lineStarts[k], k, false)) {
|
|
1247
|
+
breakLine = k;
|
|
1248
|
+
break;
|
|
1249
|
+
}
|
|
952
1250
|
end = this.contentEnds[k];
|
|
953
1251
|
}
|
|
1252
|
+
if (breakLine !== -1) {
|
|
1253
|
+
const breakIsland = this.detectBlock(this.lineStarts[breakLine], breakLine, false);
|
|
1254
|
+
if (breakIsland?.type === "tree") {
|
|
1255
|
+
let root = breakLine;
|
|
1256
|
+
for (let j = breakLine - 1; j >= li; j--) {
|
|
1257
|
+
const trimmed = this.lineText(j).trim();
|
|
1258
|
+
if (!trimmed || isSplitterTreeLine(trimmed) || /^#{1,6}\s/.test(trimmed)) break;
|
|
1259
|
+
root = j;
|
|
1260
|
+
}
|
|
1261
|
+
if (root < breakLine) {
|
|
1262
|
+
const treeStart = root === li ? pos : this.lineStarts[root];
|
|
1263
|
+
const tree = { ...breakIsland, start: treeStart };
|
|
1264
|
+
if (treeStart > pos) {
|
|
1265
|
+
const proseEnd = this.contentEnds[root - 1];
|
|
1266
|
+
const scan2 = this.scanInline(pos, proseEnd);
|
|
1267
|
+
if (!scan2.breakIsland) {
|
|
1268
|
+
this.push({ kind: "prose", start: pos, end: proseEnd, island: null, inlines: scan2.inlines });
|
|
1269
|
+
this.push({ kind: "gap", start: proseEnd, end: treeStart, island: null, inlines: [] });
|
|
1270
|
+
this.push({ kind: "island", start: treeStart, end: tree.end, island: tree, inlines: [] });
|
|
1271
|
+
return tree.end;
|
|
1272
|
+
}
|
|
1273
|
+
} else {
|
|
1274
|
+
this.push({ kind: "island", start: treeStart, end: tree.end, island: tree, inlines: [] });
|
|
1275
|
+
return tree.end;
|
|
1276
|
+
}
|
|
1277
|
+
}
|
|
1278
|
+
}
|
|
1279
|
+
}
|
|
954
1280
|
const scan = this.scanInline(pos, end);
|
|
955
1281
|
if (!scan.breakIsland) {
|
|
956
1282
|
this.push({ kind: "prose", start: pos, end, island: null, inlines: scan.inlines });
|
|
@@ -983,17 +1309,49 @@ var Tokenizer = class {
|
|
|
983
1309
|
const text = this.text;
|
|
984
1310
|
const inlines = [];
|
|
985
1311
|
const hasKind = text.slice(ps, pe).includes("__kind");
|
|
1312
|
+
let kindBudget = 4 * (pe - ps) + 65536;
|
|
986
1313
|
const stop = (island, orphan = false) => ({ inlines, breakIsland: island, orphan });
|
|
1314
|
+
const within = (needle, from) => {
|
|
1315
|
+
const at = this.next(needle, from);
|
|
1316
|
+
return at !== -1 && at + needle.length <= pe ? at : -1;
|
|
1317
|
+
};
|
|
1318
|
+
let runs = null;
|
|
1319
|
+
const codeSpanClose = (n, from) => {
|
|
1320
|
+
if (!runs) {
|
|
1321
|
+
runs = /* @__PURE__ */ new Map();
|
|
1322
|
+
for (let k = ps; k < pe; ) {
|
|
1323
|
+
if (text.charCodeAt(k) !== 96) {
|
|
1324
|
+
k++;
|
|
1325
|
+
continue;
|
|
1326
|
+
}
|
|
1327
|
+
const len = backtickRunLength(text, k);
|
|
1328
|
+
const list2 = runs.get(len);
|
|
1329
|
+
if (list2) list2.push(k);
|
|
1330
|
+
else runs.set(len, [k]);
|
|
1331
|
+
k += len;
|
|
1332
|
+
}
|
|
1333
|
+
}
|
|
1334
|
+
const list = runs.get(n);
|
|
1335
|
+
if (!list) return -1;
|
|
1336
|
+
let lo = 0;
|
|
1337
|
+
let hi = list.length;
|
|
1338
|
+
while (lo < hi) {
|
|
1339
|
+
const mid = lo + hi >>> 1;
|
|
1340
|
+
if (list[mid] < from) lo = mid + 1;
|
|
1341
|
+
else hi = mid;
|
|
1342
|
+
}
|
|
1343
|
+
const at = list[lo];
|
|
1344
|
+
return at !== void 0 && at + n <= pe ? at : -1;
|
|
1345
|
+
};
|
|
987
1346
|
let i = ps;
|
|
988
1347
|
while (i < pe) {
|
|
989
1348
|
const ch = text[i];
|
|
990
1349
|
if (ch === "\\") {
|
|
991
|
-
const
|
|
992
|
-
if (
|
|
993
|
-
const
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: next === "(" ? "\\(" : "\\[" }));
|
|
1350
|
+
const nextCh = text[i + 1];
|
|
1351
|
+
if (nextCh === "(" || nextCh === "[") {
|
|
1352
|
+
const close = within(nextCh === "(" ? "\\)" : "\\]", i + 2);
|
|
1353
|
+
if (close !== -1) {
|
|
1354
|
+
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: nextCh === "(" ? "\\(" : "\\[" }));
|
|
997
1355
|
i = close + 2;
|
|
998
1356
|
continue;
|
|
999
1357
|
}
|
|
@@ -1003,33 +1361,20 @@ var Tokenizer = class {
|
|
|
1003
1361
|
}
|
|
1004
1362
|
if (ch === "`") {
|
|
1005
1363
|
const n = backtickRunLength(text, i);
|
|
1006
|
-
const
|
|
1007
|
-
let k = i + n;
|
|
1008
|
-
let close = -1;
|
|
1009
|
-
while (k < pe) {
|
|
1010
|
-
const idx = text.indexOf(run, k);
|
|
1011
|
-
if (idx === -1 || idx >= pe) break;
|
|
1012
|
-
if (text[idx + n] === "`") {
|
|
1013
|
-
let m = idx;
|
|
1014
|
-
while (text[m] === "`") m++;
|
|
1015
|
-
k = m;
|
|
1016
|
-
continue;
|
|
1017
|
-
}
|
|
1018
|
-
close = idx;
|
|
1019
|
-
break;
|
|
1020
|
-
}
|
|
1364
|
+
const close = codeSpanClose(n, i + n);
|
|
1021
1365
|
i = close === -1 ? i + n : close + n;
|
|
1022
1366
|
continue;
|
|
1023
1367
|
}
|
|
1024
1368
|
if (ch === "$") {
|
|
1025
1369
|
if (text[i + 1] === "$") {
|
|
1026
|
-
const close =
|
|
1027
|
-
if (close
|
|
1028
|
-
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: "$$" }));
|
|
1029
|
-
i = close + 2;
|
|
1030
|
-
} else {
|
|
1370
|
+
const close = this.mathPairs.get(i);
|
|
1371
|
+
if (close === void 0) {
|
|
1031
1372
|
i += 2;
|
|
1373
|
+
continue;
|
|
1032
1374
|
}
|
|
1375
|
+
if (close + 2 > pe) return stop(this.island("math_block", i, close + 2, true, { delimiter: "$$" }));
|
|
1376
|
+
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: "$$" }));
|
|
1377
|
+
i = close + 2;
|
|
1033
1378
|
continue;
|
|
1034
1379
|
}
|
|
1035
1380
|
const end = singleDollarMathEnd(text, i, pe);
|
|
@@ -1043,7 +1388,7 @@ var Tokenizer = class {
|
|
|
1043
1388
|
}
|
|
1044
1389
|
if (ch === "{") {
|
|
1045
1390
|
if (text[i + 1] === "{") {
|
|
1046
|
-
const close =
|
|
1391
|
+
const close = this.next("}}", i + 2);
|
|
1047
1392
|
if (close !== -1 && close + 2 <= pe && close - i <= 200 && !text.slice(i, close).includes("\n")) {
|
|
1048
1393
|
const raw = text.slice(i, close + 2);
|
|
1049
1394
|
const strict = STRICT_VARIABLE_RE.exec(raw);
|
|
@@ -1056,8 +1401,10 @@ var Tokenizer = class {
|
|
|
1056
1401
|
i += 2;
|
|
1057
1402
|
continue;
|
|
1058
1403
|
}
|
|
1059
|
-
if (hasKind) {
|
|
1060
|
-
const
|
|
1404
|
+
if (hasKind && kindBudget > 0 && /^\{\s*"/.test(text.slice(i, i + 64))) {
|
|
1405
|
+
const limit = Math.min(pe, i + 65536);
|
|
1406
|
+
const end = matchingJsonObjectEnd(text, i, limit);
|
|
1407
|
+
kindBudget -= (end ?? limit) - i;
|
|
1061
1408
|
if (end !== null) {
|
|
1062
1409
|
const kind = declaredKind(text.slice(i, end));
|
|
1063
1410
|
if (kind) {
|
|
@@ -1070,11 +1417,27 @@ var Tokenizer = class {
|
|
|
1070
1417
|
i++;
|
|
1071
1418
|
continue;
|
|
1072
1419
|
}
|
|
1420
|
+
if (ch === "[" && text[i + 1] === "[" || ch === "!" && text[i + 1] === "[" && text[i + 2] === "[") {
|
|
1421
|
+
const open = ch === "!" ? i + 3 : i + 2;
|
|
1422
|
+
const close = within("]]", open);
|
|
1423
|
+
const newline = this.next("\n", open);
|
|
1424
|
+
const inner = close === -1 ? "" : text.slice(open, close);
|
|
1425
|
+
if (close !== -1 && (newline === -1 || newline > close) && inner.trim() && !/[[\]]/.test(inner)) {
|
|
1426
|
+
const cut = inner.search(/\\?\|/);
|
|
1427
|
+
const target = (cut < 0 ? inner : inner.slice(0, cut)).trim();
|
|
1428
|
+
inlines.push(
|
|
1429
|
+
this.island("wikilink", i, close + 2, true, ch === "!" ? { target, embed: true } : { target })
|
|
1430
|
+
);
|
|
1431
|
+
i = close + 2;
|
|
1432
|
+
continue;
|
|
1433
|
+
}
|
|
1434
|
+
}
|
|
1073
1435
|
if (ch === "[") {
|
|
1074
1436
|
const media = MEDIA_REF_PREFIXES.find((prefix) => text.startsWith(prefix, i));
|
|
1075
1437
|
if (media) {
|
|
1076
|
-
const close =
|
|
1077
|
-
|
|
1438
|
+
const close = within("]", i + media.length);
|
|
1439
|
+
const newline = this.next("\n", i);
|
|
1440
|
+
if (close !== -1 && (newline === -1 || newline > close)) {
|
|
1078
1441
|
inlines.push(this.island("media_ref", i, close + 1, true, { media: media.slice(1, 6).toLowerCase() }));
|
|
1079
1442
|
i = close + 1;
|
|
1080
1443
|
continue;
|
|
@@ -1085,10 +1448,10 @@ var Tokenizer = class {
|
|
|
1085
1448
|
}
|
|
1086
1449
|
if (ch === "<") {
|
|
1087
1450
|
if (text.startsWith("<!--", i)) {
|
|
1088
|
-
const close =
|
|
1451
|
+
const close = this.next("-->", i + 4);
|
|
1089
1452
|
if (close === -1) return stop(this.island("html_comment", i, this.n, false));
|
|
1090
1453
|
const end = close + 3;
|
|
1091
|
-
const anchor = PINNED_ANCHOR_RE.exec(text.slice(i, end));
|
|
1454
|
+
const anchor = end - i <= 64 ? PINNED_ANCHOR_RE.exec(text.slice(i, end)) : null;
|
|
1092
1455
|
const type = anchor ? "anchor" : "html_comment";
|
|
1093
1456
|
const meta = anchor ? { id: anchor[1] } : {};
|
|
1094
1457
|
if (end > pe) return stop(this.island(type, i, end, true, meta));
|
|
@@ -1102,7 +1465,7 @@ var Tokenizer = class {
|
|
|
1102
1465
|
i += cite[0].length;
|
|
1103
1466
|
continue;
|
|
1104
1467
|
}
|
|
1105
|
-
const tag = readXmlTag(text, i);
|
|
1468
|
+
const tag = readXmlTag(text, i, this.finder);
|
|
1106
1469
|
if (tag && i + tag.raw.length <= pe) {
|
|
1107
1470
|
const name = tag.tagName;
|
|
1108
1471
|
const tagEnd = i + tag.raw.length;
|
|
@@ -1115,7 +1478,7 @@ var Tokenizer = class {
|
|
|
1115
1478
|
continue;
|
|
1116
1479
|
}
|
|
1117
1480
|
const closer = `</${name}>`;
|
|
1118
|
-
const close =
|
|
1481
|
+
const close = this.next(closer, tagEnd);
|
|
1119
1482
|
const blockType = isAttr ? "xml_attr" : "xml_region";
|
|
1120
1483
|
if (close === -1) return stop(this.island(blockType, i, this.n, false, { tag: name }));
|
|
1121
1484
|
const end = close + closer.length;
|
|
@@ -1139,7 +1502,7 @@ var Tokenizer = class {
|
|
|
1139
1502
|
const pending = ATTRIBUTE_XML_NAMES.find(
|
|
1140
1503
|
(name) => text.startsWith(`<${name}`, i) && /\s/.test(text[i + name.length + 1] ?? "")
|
|
1141
1504
|
);
|
|
1142
|
-
if (pending &&
|
|
1505
|
+
if (pending && this.next(">", i) === -1) {
|
|
1143
1506
|
return stop(this.island("xml_attr", i, this.n, false, { tag: pending }));
|
|
1144
1507
|
}
|
|
1145
1508
|
i++;
|
|
@@ -1150,8 +1513,8 @@ var Tokenizer = class {
|
|
|
1150
1513
|
return { inlines, breakIsland: null, orphan: false };
|
|
1151
1514
|
}
|
|
1152
1515
|
};
|
|
1153
|
-
function tokenizeSource(text) {
|
|
1154
|
-
const raw = new Tokenizer(text).run();
|
|
1516
|
+
function tokenizeSource(text, options = {}) {
|
|
1517
|
+
const raw = new Tokenizer(text, options.streaming ?? false).run();
|
|
1155
1518
|
const cp = buildCodePointIndex(text);
|
|
1156
1519
|
const inline = (island) => ({
|
|
1157
1520
|
islandType: island.type,
|
|
@@ -1200,7 +1563,7 @@ function listIslands(blocks) {
|
|
|
1200
1563
|
raw: block.raw
|
|
1201
1564
|
});
|
|
1202
1565
|
} else {
|
|
1203
|
-
|
|
1566
|
+
for (const inline of block.inlines) out.push(inline);
|
|
1204
1567
|
}
|
|
1205
1568
|
}
|
|
1206
1569
|
return out;
|
|
@@ -1230,9 +1593,44 @@ var SourceSpliceError = class extends Error {
|
|
|
1230
1593
|
function blockEdit(block, text) {
|
|
1231
1594
|
return { start: block.start, end: block.end, text };
|
|
1232
1595
|
}
|
|
1596
|
+
function islandEdit(island, text) {
|
|
1597
|
+
return { start: island.start, end: island.end, text, island: true };
|
|
1598
|
+
}
|
|
1233
1599
|
function blockRangeEdit(first, last, text) {
|
|
1234
1600
|
return { start: first.start, end: last.end, text };
|
|
1235
1601
|
}
|
|
1602
|
+
var CONTEXT_WINDOW = 64;
|
|
1603
|
+
function locateIsland(text, raw, from, expected, expectedFromEnd, oldText, oldStart, oldEnd) {
|
|
1604
|
+
const candidates = /* @__PURE__ */ new Set();
|
|
1605
|
+
if (expected >= from && text.startsWith(raw, expected)) candidates.add(expected);
|
|
1606
|
+
if (expectedFromEnd >= from && text.startsWith(raw, expectedFromEnd)) candidates.add(expectedFromEnd);
|
|
1607
|
+
const after = text.indexOf(raw, Math.max(from, expected));
|
|
1608
|
+
if (after !== -1) candidates.add(after);
|
|
1609
|
+
const before = expected > from ? text.lastIndexOf(raw, expected) : -1;
|
|
1610
|
+
if (before >= from) candidates.add(before);
|
|
1611
|
+
if (candidates.size === 0) return text.indexOf(raw, from);
|
|
1612
|
+
if (candidates.size === 1) return candidates.values().next().value;
|
|
1613
|
+
let best = -1;
|
|
1614
|
+
let bestScore = -1;
|
|
1615
|
+
for (const at of candidates) {
|
|
1616
|
+
let score = 0;
|
|
1617
|
+
for (let k = 0; k < CONTEXT_WINDOW; k++) {
|
|
1618
|
+
const n = text[at + raw.length + k];
|
|
1619
|
+
if (n === void 0 || n !== oldText[oldEnd + k]) break;
|
|
1620
|
+
score++;
|
|
1621
|
+
}
|
|
1622
|
+
for (let k = 1; k <= CONTEXT_WINDOW; k++) {
|
|
1623
|
+
const n = text[at - k];
|
|
1624
|
+
if (n === void 0 || n !== oldText[oldStart - k]) break;
|
|
1625
|
+
score++;
|
|
1626
|
+
}
|
|
1627
|
+
if (score > bestScore || score === bestScore && at === expected) {
|
|
1628
|
+
best = at;
|
|
1629
|
+
bestScore = score;
|
|
1630
|
+
}
|
|
1631
|
+
}
|
|
1632
|
+
return best;
|
|
1633
|
+
}
|
|
1236
1634
|
function commonPrefix(a, b) {
|
|
1237
1635
|
const max = Math.min(a.length, b.length);
|
|
1238
1636
|
let i = 0;
|
|
@@ -1257,13 +1655,23 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1257
1655
|
boundaries.add(block.start);
|
|
1258
1656
|
boundaries.add(block.end);
|
|
1259
1657
|
}
|
|
1658
|
+
const islands = listIslands(blocks);
|
|
1659
|
+
const islandAt = /* @__PURE__ */ new Map();
|
|
1660
|
+
for (const island of islands) islandAt.set(`${island.start}:${island.end}`, island);
|
|
1260
1661
|
const sorted = [...edits].sort((a, b) => a.start - b.start || a.end - b.end);
|
|
1261
1662
|
let previousEnd = -1;
|
|
1262
1663
|
for (const edit of sorted) {
|
|
1263
1664
|
if (edit.start < 0 || edit.end > original.length || edit.start > edit.end) {
|
|
1264
1665
|
throw new SourceSpliceError("out_of_range", `edit [${edit.start}, ${edit.end}) is outside the text`);
|
|
1265
1666
|
}
|
|
1266
|
-
if (
|
|
1667
|
+
if (edit.island) {
|
|
1668
|
+
if (!islandAt.has(`${edit.start}:${edit.end}`)) {
|
|
1669
|
+
throw new SourceSpliceError(
|
|
1670
|
+
"misaligned",
|
|
1671
|
+
`islandEdit [${edit.start}, ${edit.end}) does not name an island's exact span`
|
|
1672
|
+
);
|
|
1673
|
+
}
|
|
1674
|
+
} else if (!boundaries.has(edit.start) || !boundaries.has(edit.end)) {
|
|
1267
1675
|
throw new SourceSpliceError(
|
|
1268
1676
|
"misaligned",
|
|
1269
1677
|
`edit [${edit.start}, ${edit.end}) does not start and end on block boundaries`
|
|
@@ -1275,18 +1683,51 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1275
1683
|
previousEnd = edit.end;
|
|
1276
1684
|
}
|
|
1277
1685
|
const effective = [];
|
|
1278
|
-
|
|
1279
|
-
|
|
1280
|
-
|
|
1281
|
-
|
|
1282
|
-
const
|
|
1686
|
+
const replacedIslands = /* @__PURE__ */ new Set();
|
|
1687
|
+
const pushSegment = (oldStart, oldEnd, replacement) => {
|
|
1688
|
+
const old = original.slice(oldStart, oldEnd);
|
|
1689
|
+
if (old === replacement) return;
|
|
1690
|
+
const prefix = commonPrefix(old, replacement);
|
|
1691
|
+
const suffix = commonSuffix(old, replacement, prefix);
|
|
1283
1692
|
effective.push({
|
|
1284
|
-
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
oldEnd: edit.end - suffix,
|
|
1288
|
-
replacement: edit.text.slice(prefix, edit.text.length - suffix)
|
|
1693
|
+
oldStart: oldStart + prefix,
|
|
1694
|
+
oldEnd: oldEnd - suffix,
|
|
1695
|
+
replacement: replacement.slice(prefix, replacement.length - suffix)
|
|
1289
1696
|
});
|
|
1697
|
+
};
|
|
1698
|
+
for (const edit of sorted) {
|
|
1699
|
+
if (edit.island) {
|
|
1700
|
+
replacedIslands.add(`${edit.start}:${edit.end}`);
|
|
1701
|
+
pushSegment(edit.start, edit.end, edit.text);
|
|
1702
|
+
continue;
|
|
1703
|
+
}
|
|
1704
|
+
if (edit.text === original.slice(edit.start, edit.end)) continue;
|
|
1705
|
+
const inside = islands.filter((island) => island.start >= edit.start && island.end <= edit.end);
|
|
1706
|
+
let oldCursor2 = edit.start;
|
|
1707
|
+
let newCursor2 = 0;
|
|
1708
|
+
for (const island of inside) {
|
|
1709
|
+
const at = locateIsland(
|
|
1710
|
+
edit.text,
|
|
1711
|
+
island.raw,
|
|
1712
|
+
newCursor2,
|
|
1713
|
+
newCursor2 + (island.start - oldCursor2),
|
|
1714
|
+
edit.text.length - (edit.end - island.start),
|
|
1715
|
+
original.slice(edit.start, edit.end),
|
|
1716
|
+
island.start - edit.start,
|
|
1717
|
+
island.end - edit.start
|
|
1718
|
+
);
|
|
1719
|
+
if (at === -1) {
|
|
1720
|
+
const name = island.meta.tag ?? island.meta.name ?? island.meta.lang ?? "";
|
|
1721
|
+
throw new SourceSpliceError(
|
|
1722
|
+
"island_edit",
|
|
1723
|
+
`edit [${edit.start}, ${edit.end}) changes or removes the protected ${island.islandType}${name ? ` (${String(name)})` : ""} island at [${island.start}, ${island.end}); islands change only through islandEdit()`
|
|
1724
|
+
);
|
|
1725
|
+
}
|
|
1726
|
+
pushSegment(oldCursor2, island.start, edit.text.slice(newCursor2, at));
|
|
1727
|
+
oldCursor2 = island.end;
|
|
1728
|
+
newCursor2 = at + island.raw.length;
|
|
1729
|
+
}
|
|
1730
|
+
pushSegment(oldCursor2, edit.end, edit.text.slice(newCursor2));
|
|
1290
1731
|
}
|
|
1291
1732
|
if (effective.length === 0) {
|
|
1292
1733
|
return {
|
|
@@ -1306,9 +1747,10 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1306
1747
|
const newStart = edit.oldStart + delta;
|
|
1307
1748
|
text += edit.replacement;
|
|
1308
1749
|
const newEnd = newStart + edit.replacement.length;
|
|
1750
|
+
const owner = sorted.find((e) => e.start <= edit.oldStart && e.end >= edit.oldEnd);
|
|
1309
1751
|
pending.push({
|
|
1310
|
-
blockStart: edit.
|
|
1311
|
-
blockEnd: edit.
|
|
1752
|
+
blockStart: owner?.start ?? edit.oldStart,
|
|
1753
|
+
blockEnd: owner?.end ?? edit.oldEnd,
|
|
1312
1754
|
oldStart: edit.oldStart,
|
|
1313
1755
|
oldEnd: edit.oldEnd,
|
|
1314
1756
|
newStart,
|
|
@@ -1346,11 +1788,8 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1346
1788
|
else newIslandsByStart.set(island.start, [island.raw]);
|
|
1347
1789
|
}
|
|
1348
1790
|
const disturbed = [];
|
|
1349
|
-
for (const island of
|
|
1350
|
-
|
|
1351
|
-
(edit) => island.start < edit.blockEnd && island.end > edit.blockStart
|
|
1352
|
-
);
|
|
1353
|
-
if (touched) continue;
|
|
1791
|
+
for (const island of islands) {
|
|
1792
|
+
if (replacedIslands.has(`${island.start}:${island.end}`)) continue;
|
|
1354
1793
|
const mappedStart = mapPosition(changes, island.start, { assoc: 1 }).pos;
|
|
1355
1794
|
const candidates = newIslandsByStart.get(mappedStart);
|
|
1356
1795
|
if (!candidates || !candidates.includes(island.raw)) {
|
|
@@ -1368,10 +1807,11 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1368
1807
|
bytesOutsideEditsIdentical,
|
|
1369
1808
|
disturbed
|
|
1370
1809
|
};
|
|
1371
|
-
if (options.requireIntegrity && !integrity.ok) {
|
|
1810
|
+
if ((options.requireIntegrity ?? true) && !integrity.ok) {
|
|
1811
|
+
const named = disturbed.slice(0, 5).map((island) => `${island.islandType}@${island.start}`).join(", ");
|
|
1372
1812
|
throw new SourceSpliceError(
|
|
1373
1813
|
"integrity",
|
|
1374
|
-
`the
|
|
1814
|
+
`integrity: the save disturbed ${disturbed.length} protected island(s) it did not replace${named ? ` (${named}${disturbed.length > 5 ? ", \u2026" : ""})` : ""}${bytesOutsideEditsIdentical ? "" : "; bytes outside the edits changed"}`
|
|
1375
1815
|
);
|
|
1376
1816
|
}
|
|
1377
1817
|
return { text, changed: true, changes, integrity, blocks: newBlocks };
|
|
@@ -1413,6 +1853,6 @@ function mapRange(changes, start, end, unit = "utf16") {
|
|
|
1413
1853
|
return { start: newStart, end: newEnd, touched, collapsed: newEnd === newStart && end > start };
|
|
1414
1854
|
}
|
|
1415
1855
|
|
|
1416
|
-
export { SourceSpliceError, XmlContainerTracker, blockAt, blockEdit, blockRangeEdit, buildCodePointIndex, isSingleDollarMath, joinSource, listIslands, mapPosition, mapRange, readXmlTag, singleDollarMathEnd, spliceSave, splitsSurrogatePair, toCodePointOffset, toUtf16Offset, tokenizeSource };
|
|
1856
|
+
export { FENCE_WHITESPACE, NESTING_FENCE_LANGUAGES, SourceSpliceError, XmlContainerTracker, blockAt, blockEdit, blockRangeEdit, buildCodePointIndex, classifyInnerFenceLine, fenceNestsInnerFences, isSingleDollarMath, islandEdit, joinSource, listIslands, looksLikeDisplayMath, mapPosition, mapRange, pairDisplayMath, readXmlTag, singleDollarMathEnd, spliceSave, splitsSurrogatePair, toCodePointOffset, toUtf16Offset, tokenizeSource, trimFenceLine };
|
|
1417
1857
|
//# sourceMappingURL=source.js.map
|
|
1418
1858
|
//# sourceMappingURL=source.js.map
|