@ai-matrx/content-ir 0.12.0 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +48 -0
- package/dist/index.cjs +328 -87
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.js +323 -88
- package/dist/index.js.map +1 -1
- package/dist/source.cjs +328 -87
- package/dist/source.cjs.map +1 -1
- package/dist/source.d.cts +76 -8
- package/dist/source.d.ts +76 -8
- package/dist/source.js +323 -88
- package/dist/source.js.map +1 -1
- package/package.json +1 -1
package/dist/source.cjs
CHANGED
|
@@ -202,6 +202,29 @@ function normalizeFenceLanguage(language) {
|
|
|
202
202
|
return CODE_LANGUAGE_ALIASES[lower] ?? lower;
|
|
203
203
|
}
|
|
204
204
|
|
|
205
|
+
// source/fence-nesting.ts
|
|
206
|
+
var NESTING_FENCE_LANGUAGES = /* @__PURE__ */ new Set([
|
|
207
|
+
"markdown",
|
|
208
|
+
"md",
|
|
209
|
+
"mdx"
|
|
210
|
+
]);
|
|
211
|
+
function fenceNestsInnerFences(language) {
|
|
212
|
+
return !!language && NESTING_FENCE_LANGUAGES.has(language.toLowerCase());
|
|
213
|
+
}
|
|
214
|
+
function classifyInnerFenceLine(trimmed, openTicks, nests, nestedDepth) {
|
|
215
|
+
let ticks = 0;
|
|
216
|
+
while (ticks < trimmed.length && trimmed[ticks] === "`") ticks++;
|
|
217
|
+
if (ticks < 3) return "content";
|
|
218
|
+
const info = trimmed.slice(ticks).trim();
|
|
219
|
+
if (info === "") {
|
|
220
|
+
if (ticks < openTicks) return "content";
|
|
221
|
+
if (nests && nestedDepth > 0) return "close-nested";
|
|
222
|
+
return "close-outer";
|
|
223
|
+
}
|
|
224
|
+
if (nests && ticks >= openTicks && !info.includes("`")) return "open-nested";
|
|
225
|
+
return "content";
|
|
226
|
+
}
|
|
227
|
+
|
|
205
228
|
// source/math.ts
|
|
206
229
|
var TEX_SIGNAL = /\\[A-Za-z]+|[\\^_{}=<>+]/;
|
|
207
230
|
var LONE_VARIABLE = /^[A-Za-z](?:'|[0-9])?$/;
|
|
@@ -214,8 +237,9 @@ function isSingleDollarMath(content, after) {
|
|
|
214
237
|
return TEX_SIGNAL.test(content);
|
|
215
238
|
}
|
|
216
239
|
function singleDollarMathEnd(text, open, limit = text.length) {
|
|
240
|
+
const bound = Math.min(limit, open + 403);
|
|
217
241
|
let close = -1;
|
|
218
|
-
for (let k = open + 1; k <
|
|
242
|
+
for (let k = open + 1; k < bound; k += 1) {
|
|
219
243
|
const c = text[k];
|
|
220
244
|
if (c === "\n") break;
|
|
221
245
|
if (c === "\\") {
|
|
@@ -232,11 +256,69 @@ function singleDollarMathEnd(text, open, limit = text.length) {
|
|
|
232
256
|
return isSingleDollarMath(content, text[close + 1]) ? close + 1 : -1;
|
|
233
257
|
}
|
|
234
258
|
|
|
259
|
+
// source/math-pairs.ts
|
|
260
|
+
var MAX_MATH_SPAN = 600;
|
|
261
|
+
var STRUCTURAL_MARKDOWN = /\]\(|https?:\/\/|\*\*|(?:^|\n)[ \t]{0,3}#{1,6}[ \t]|(?:^|\n)[ \t]*[-*+][ \t]+|(?:^|\n)[ \t]*\d+[.)][ \t]/;
|
|
262
|
+
var LATEX_COMMAND = /\\[a-zA-Z]/;
|
|
263
|
+
var PROSE_WORD = /[A-Za-z]{3,}/g;
|
|
264
|
+
var PROSE_WORD_LIMIT = 6;
|
|
265
|
+
function looksLikeDisplayMath(inner) {
|
|
266
|
+
const s = inner.trim();
|
|
267
|
+
if (!s) return false;
|
|
268
|
+
if (STRUCTURAL_MARKDOWN.test(s)) return false;
|
|
269
|
+
if (s.length > MAX_MATH_SPAN) return false;
|
|
270
|
+
if (LATEX_COMMAND.test(s)) {
|
|
271
|
+
return (s.match(PROSE_WORD) ?? []).length < PROSE_WORD_LIMIT * 3;
|
|
272
|
+
}
|
|
273
|
+
if (/\n[ \t]*\n/.test(s)) return false;
|
|
274
|
+
return (s.match(PROSE_WORD) ?? []).length < PROSE_WORD_LIMIT;
|
|
275
|
+
}
|
|
276
|
+
function codeRanges(text) {
|
|
277
|
+
const ranges = [];
|
|
278
|
+
for (const re of [/```[\s\S]*?(?:```|$)/g, /~~~[\s\S]*?(?:~~~|$)/g, /`[^`\n]*`/g]) {
|
|
279
|
+
let m;
|
|
280
|
+
while ((m = re.exec(text)) !== null) {
|
|
281
|
+
ranges.push([m.index, m.index + m[0].length]);
|
|
282
|
+
if (m[0].length === 0) re.lastIndex++;
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
return ranges.sort((a, b) => a[0] - b[0]);
|
|
286
|
+
}
|
|
287
|
+
function pairDisplayMath(text) {
|
|
288
|
+
const pairs = /* @__PURE__ */ new Map();
|
|
289
|
+
if (!text.includes("$$")) return pairs;
|
|
290
|
+
const ranges = codeRanges(text);
|
|
291
|
+
const tokens = [];
|
|
292
|
+
let r = 0;
|
|
293
|
+
let maxEnd = -1;
|
|
294
|
+
for (let i = text.indexOf("$$"); i !== -1 && i < text.length - 1; i = text.indexOf("$$", i)) {
|
|
295
|
+
while (r < ranges.length && ranges[r][0] <= i) {
|
|
296
|
+
maxEnd = Math.max(maxEnd, ranges[r][1]);
|
|
297
|
+
r++;
|
|
298
|
+
}
|
|
299
|
+
if (maxEnd <= i) tokens.push(i);
|
|
300
|
+
i += 2;
|
|
301
|
+
}
|
|
302
|
+
let j = 0;
|
|
303
|
+
while (j < tokens.length) {
|
|
304
|
+
const open = tokens[j];
|
|
305
|
+
const close = tokens[j + 1];
|
|
306
|
+
if (close === void 0) break;
|
|
307
|
+
if (looksLikeDisplayMath(text.slice(open + 2, close))) {
|
|
308
|
+
pairs.set(open, close);
|
|
309
|
+
j += 2;
|
|
310
|
+
continue;
|
|
311
|
+
}
|
|
312
|
+
j += 1;
|
|
313
|
+
}
|
|
314
|
+
return pairs;
|
|
315
|
+
}
|
|
316
|
+
|
|
235
317
|
// source/xml-tag.ts
|
|
236
318
|
var XML_NAME_START = /[A-Za-z_]/;
|
|
237
319
|
var XML_NAME_CHARACTER = /[\w.:-]/;
|
|
238
320
|
var WHITESPACE = /\s/;
|
|
239
|
-
function readXmlTag(content, start) {
|
|
321
|
+
function readXmlTag(content, start, find = (needle, from) => content.indexOf(needle, from)) {
|
|
240
322
|
let cursor = start + 1;
|
|
241
323
|
const isClosing = content[cursor] === "/";
|
|
242
324
|
if (isClosing) cursor++;
|
|
@@ -289,8 +371,9 @@ function readXmlTag(content, start) {
|
|
|
289
371
|
const quote = content[cursor];
|
|
290
372
|
if (quote !== '"' && quote !== "'") return null;
|
|
291
373
|
const valueStart = ++cursor;
|
|
292
|
-
|
|
293
|
-
if (
|
|
374
|
+
const valueEnd = find(quote, valueStart);
|
|
375
|
+
if (valueEnd === -1) return null;
|
|
376
|
+
cursor = valueEnd;
|
|
294
377
|
attributes.push({ name, value: content.slice(valueStart, cursor) });
|
|
295
378
|
cursor++;
|
|
296
379
|
}
|
|
@@ -509,11 +592,15 @@ function backtickRunLength(str, pos) {
|
|
|
509
592
|
while (pos + n < str.length && str[pos + n] === "`") n++;
|
|
510
593
|
return n;
|
|
511
594
|
}
|
|
595
|
+
function isSplitterTreeLine(line) {
|
|
596
|
+
return TREE_GLYPHS.test(line) || /^[\s│|]*[├└+|][\s─\-]+/.test(line);
|
|
597
|
+
}
|
|
512
598
|
var MEDIA_REF_PREFIXES = ["[Image URL:", "[Video URL:", "[Audio URL:"];
|
|
513
599
|
var Tokenizer = class {
|
|
514
600
|
constructor(text) {
|
|
515
601
|
this.text = text;
|
|
516
602
|
this.n = text.length;
|
|
603
|
+
this.mathPairs = pairDisplayMath(text);
|
|
517
604
|
let start = 0;
|
|
518
605
|
for (let i = 0; i < text.length; i++) {
|
|
519
606
|
if (text.charCodeAt(i) === 10) {
|
|
@@ -539,6 +626,34 @@ var Tokenizer = class {
|
|
|
539
626
|
/** Lazily built: for line i, first line j >= i where brace net from i drops to <= 0. */
|
|
540
627
|
braceStop = null;
|
|
541
628
|
braceNet = null;
|
|
629
|
+
/** Every occurrence offset of each needle searched so far (overlapping, like indexOf). */
|
|
630
|
+
occurrences = /* @__PURE__ */ new Map();
|
|
631
|
+
/** `$$` opener → closer for the pairs the core renders as math. */
|
|
632
|
+
mathPairs;
|
|
633
|
+
finder = (needle, from) => this.next(needle, from);
|
|
634
|
+
/**
|
|
635
|
+
* `text.indexOf(needle, from)` in O(log k): the first query for a needle
|
|
636
|
+
* indexes all of its occurrences once, so repeated searches for a closer
|
|
637
|
+
* that never comes stay linear over the whole text.
|
|
638
|
+
*/
|
|
639
|
+
next(needle, from) {
|
|
640
|
+
let list = this.occurrences.get(needle);
|
|
641
|
+
if (!list) {
|
|
642
|
+
list = [];
|
|
643
|
+
for (let at = this.text.indexOf(needle); at !== -1; at = this.text.indexOf(needle, at + 1)) {
|
|
644
|
+
list.push(at);
|
|
645
|
+
}
|
|
646
|
+
this.occurrences.set(needle, list);
|
|
647
|
+
}
|
|
648
|
+
let lo = 0;
|
|
649
|
+
let hi = list.length;
|
|
650
|
+
while (lo < hi) {
|
|
651
|
+
const mid = lo + hi >>> 1;
|
|
652
|
+
if (list[mid] < from) lo = mid + 1;
|
|
653
|
+
else hi = mid;
|
|
654
|
+
}
|
|
655
|
+
return lo < list.length ? list[lo] : -1;
|
|
656
|
+
}
|
|
542
657
|
get lineCount() {
|
|
543
658
|
return this.lineStarts.length;
|
|
544
659
|
}
|
|
@@ -670,7 +785,12 @@ var Tokenizer = class {
|
|
|
670
785
|
}
|
|
671
786
|
return null;
|
|
672
787
|
}
|
|
673
|
-
/**
|
|
788
|
+
/**
|
|
789
|
+
* Backtick fence — the renderer splitter's close rules: JSON-string aware for
|
|
790
|
+
* ```json, and THE nested-fence rule (source/fence-nesting.ts) for markdown
|
|
791
|
+
* fences, with the splitter's strict-CommonMark retry when a nested fence is
|
|
792
|
+
* still open at the end of the text.
|
|
793
|
+
*/
|
|
674
794
|
backtickFence(p, li, t) {
|
|
675
795
|
const openTicks = backtickRunLength(t, 0);
|
|
676
796
|
const info = t.slice(openTicks).trim();
|
|
@@ -679,7 +799,19 @@ var Tokenizer = class {
|
|
|
679
799
|
const normalized = normalizeFenceLanguage(lang);
|
|
680
800
|
const meta = { fence: "`", ticks: openTicks, lang };
|
|
681
801
|
if (normalized && SPECIAL_CODE_LANGUAGES.has(normalized)) meta.special = normalized;
|
|
802
|
+
const nesting = fenceNestsInnerFences(lang);
|
|
803
|
+
let close = this.backtickFenceClose(li, openTicks, isJson, nesting);
|
|
804
|
+
if (close === "retry-strict") close = this.backtickFenceClose(li, openTicks, isJson, false);
|
|
805
|
+
if (typeof close === "number") {
|
|
806
|
+
return this.fenceIsland(p, this.contentEnds[close], true, meta, li, close, isJson);
|
|
807
|
+
}
|
|
808
|
+
return this.fenceIsland(p, this.n, false, meta, li, this.lineCount, isJson);
|
|
809
|
+
}
|
|
810
|
+
/** The closing line of a backtick fence opened on line `li`, or null when it never closes. */
|
|
811
|
+
backtickFenceClose(li, openTicks, isJson, nesting) {
|
|
682
812
|
let state = { inString: false, escaped: false };
|
|
813
|
+
let nestedDepth = 0;
|
|
814
|
+
let sawNested = false;
|
|
683
815
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
684
816
|
const line = this.lineText(k);
|
|
685
817
|
const trimmedLine = line.trim();
|
|
@@ -692,11 +824,13 @@ var Tokenizer = class {
|
|
|
692
824
|
continue;
|
|
693
825
|
}
|
|
694
826
|
}
|
|
695
|
-
const
|
|
696
|
-
|
|
697
|
-
if (
|
|
698
|
-
|
|
827
|
+
const kind = classifyInnerFenceLine(trimmedLine, openTicks, nesting, nestedDepth);
|
|
828
|
+
if (kind === "close-outer") return k;
|
|
829
|
+
if (kind === "open-nested") {
|
|
830
|
+
nestedDepth++;
|
|
831
|
+
sawNested = true;
|
|
699
832
|
}
|
|
833
|
+
if (kind === "close-nested") nestedDepth--;
|
|
700
834
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
701
835
|
continue;
|
|
702
836
|
}
|
|
@@ -711,15 +845,13 @@ var Tokenizer = class {
|
|
|
711
845
|
}
|
|
712
846
|
const closeTicks = backtickRunLength(line, at);
|
|
713
847
|
const after = line.slice(at + closeTicks).trim();
|
|
714
|
-
if (closeTicks >= openTicks && after === "")
|
|
715
|
-
return this.fenceIsland(p, this.contentEnds[k], true, meta, li, k, isJson);
|
|
716
|
-
}
|
|
848
|
+
if (closeTicks >= openTicks && after === "" && nestedDepth === 0) return k;
|
|
717
849
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
718
850
|
continue;
|
|
719
851
|
}
|
|
720
852
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
721
853
|
}
|
|
722
|
-
return
|
|
854
|
+
return sawNested && nestedDepth > 0 ? "retry-strict" : null;
|
|
723
855
|
}
|
|
724
856
|
fenceIsland(p, end, complete, meta, openLine, closeLine, isJson) {
|
|
725
857
|
if (isJson && openLine + 1 < this.lineCount) {
|
|
@@ -745,18 +877,22 @@ var Tokenizer = class {
|
|
|
745
877
|
}
|
|
746
878
|
return this.island("fence", p, this.n, false, meta);
|
|
747
879
|
}
|
|
880
|
+
/**
|
|
881
|
+
* A line-leading `$$` that OPENS a math pair (source/math-pairs.ts — the
|
|
882
|
+
* core's rule). A closer, an unpaired `$$`, or a prose pair is never a block:
|
|
883
|
+
* a lone `$$` renders as literal text and locks nothing.
|
|
884
|
+
*/
|
|
748
885
|
mathBlock(p, tp) {
|
|
886
|
+
const close = this.mathPairs.get(tp);
|
|
887
|
+
if (close === void 0) return null;
|
|
888
|
+
const end = close + 2;
|
|
749
889
|
const li = this.lineOf(tp);
|
|
750
890
|
const ce = this.contentEnds[li];
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
}
|
|
757
|
-
const close = this.text.indexOf("$$", tp + 2);
|
|
758
|
-
if (close === -1) return this.island("math_block", p, this.n, false, { delimiter: "$$" });
|
|
759
|
-
return this.island("math_block", p, close + 2, true, { delimiter: "$$" });
|
|
891
|
+
if (end <= ce && !this.isBlank(end, ce)) return null;
|
|
892
|
+
const endLine = this.lineOf(close);
|
|
893
|
+
const endCe = this.contentEnds[endLine];
|
|
894
|
+
if (endLine !== li && !this.isBlank(end, endCe)) return null;
|
|
895
|
+
return this.island("math_block", p, end, true, { delimiter: "$$" });
|
|
760
896
|
}
|
|
761
897
|
/** `<artifact …>`, `<decision …>`, … at the start of a line. */
|
|
762
898
|
attributeXml(p, tp, t) {
|
|
@@ -773,7 +909,7 @@ var Tokenizer = class {
|
|
|
773
909
|
attributeElement(name, p, tp, type) {
|
|
774
910
|
const li = this.lineOf(tp);
|
|
775
911
|
const ce = this.contentEnds[li];
|
|
776
|
-
const tag = readXmlTag(this.text, tp);
|
|
912
|
+
const tag = readXmlTag(this.text, tp, this.finder);
|
|
777
913
|
let openEnd;
|
|
778
914
|
if (tag && !tag.isClosing) {
|
|
779
915
|
if (tag.isSelfClosing) {
|
|
@@ -781,13 +917,13 @@ var Tokenizer = class {
|
|
|
781
917
|
}
|
|
782
918
|
openEnd = tp + tag.raw.length;
|
|
783
919
|
} else {
|
|
784
|
-
const gt = this.
|
|
920
|
+
const gt = this.next(">", tp);
|
|
785
921
|
if (gt === -1) return this.island(type, p, this.n, false, { tag: name });
|
|
786
922
|
if (gt >= ce) return null;
|
|
787
923
|
openEnd = gt + 1;
|
|
788
924
|
}
|
|
789
925
|
const closer = `</${name}>`;
|
|
790
|
-
const close = this.
|
|
926
|
+
const close = this.next(closer, openEnd);
|
|
791
927
|
if (close === -1) return this.island(type, p, this.n, false, { tag: name });
|
|
792
928
|
return this.island(type, p, close + closer.length, true, { tag: name });
|
|
793
929
|
}
|
|
@@ -797,7 +933,7 @@ var Tokenizer = class {
|
|
|
797
933
|
const opener = `<${name}>`;
|
|
798
934
|
if (!t.startsWith(opener)) continue;
|
|
799
935
|
const closer = `</${name}>`;
|
|
800
|
-
const close = this.
|
|
936
|
+
const close = this.next(closer, tp + opener.length);
|
|
801
937
|
if (close === -1) return this.island("xml_region", p, this.n, false, { tag: name });
|
|
802
938
|
return this.island("xml_region", p, close + closer.length, true, { tag: name });
|
|
803
939
|
}
|
|
@@ -833,7 +969,7 @@ var Tokenizer = class {
|
|
|
833
969
|
return null;
|
|
834
970
|
}
|
|
835
971
|
blockComment(p, tp, li) {
|
|
836
|
-
const close = this.
|
|
972
|
+
const close = this.next("-->", tp + 4);
|
|
837
973
|
if (close === -1) return this.island("html_comment", p, this.n, false);
|
|
838
974
|
const end = close + 3;
|
|
839
975
|
const raw = this.text.slice(tp, end);
|
|
@@ -911,9 +1047,8 @@ var Tokenizer = class {
|
|
|
911
1047
|
}
|
|
912
1048
|
}
|
|
913
1049
|
if (lastLine === null) {
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
return this.island("json", p, this.n, false, jsonMeta(partial));
|
|
1050
|
+
if (isRecognizedJsonHead(this.text.slice(tp, tp + 512))) {
|
|
1051
|
+
return this.island("json", p, this.n, false, jsonMeta(this.text.slice(tp)));
|
|
917
1052
|
}
|
|
918
1053
|
return null;
|
|
919
1054
|
}
|
|
@@ -948,11 +1083,43 @@ var Tokenizer = class {
|
|
|
948
1083
|
// ── prose ──────────────────────────────────────────────────────────────
|
|
949
1084
|
paragraph(pos, li) {
|
|
950
1085
|
let end = this.contentEnds[li];
|
|
1086
|
+
let breakLine = -1;
|
|
951
1087
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
952
1088
|
if (this.lineIsBlank(k)) break;
|
|
953
|
-
if (this.detectBlock(this.lineStarts[k], k, false))
|
|
1089
|
+
if (this.detectBlock(this.lineStarts[k], k, false)) {
|
|
1090
|
+
breakLine = k;
|
|
1091
|
+
break;
|
|
1092
|
+
}
|
|
954
1093
|
end = this.contentEnds[k];
|
|
955
1094
|
}
|
|
1095
|
+
if (breakLine !== -1) {
|
|
1096
|
+
const breakIsland = this.detectBlock(this.lineStarts[breakLine], breakLine, false);
|
|
1097
|
+
if (breakIsland?.type === "tree") {
|
|
1098
|
+
let root = breakLine;
|
|
1099
|
+
for (let j = breakLine - 1; j >= li; j--) {
|
|
1100
|
+
const trimmed = this.lineText(j).trim();
|
|
1101
|
+
if (!trimmed || isSplitterTreeLine(trimmed) || /^#{1,6}\s/.test(trimmed)) break;
|
|
1102
|
+
root = j;
|
|
1103
|
+
}
|
|
1104
|
+
if (root < breakLine) {
|
|
1105
|
+
const treeStart = root === li ? pos : this.lineStarts[root];
|
|
1106
|
+
const tree = { ...breakIsland, start: treeStart };
|
|
1107
|
+
if (treeStart > pos) {
|
|
1108
|
+
const proseEnd = this.contentEnds[root - 1];
|
|
1109
|
+
const scan2 = this.scanInline(pos, proseEnd);
|
|
1110
|
+
if (!scan2.breakIsland) {
|
|
1111
|
+
this.push({ kind: "prose", start: pos, end: proseEnd, island: null, inlines: scan2.inlines });
|
|
1112
|
+
this.push({ kind: "gap", start: proseEnd, end: treeStart, island: null, inlines: [] });
|
|
1113
|
+
this.push({ kind: "island", start: treeStart, end: tree.end, island: tree, inlines: [] });
|
|
1114
|
+
return tree.end;
|
|
1115
|
+
}
|
|
1116
|
+
} else {
|
|
1117
|
+
this.push({ kind: "island", start: treeStart, end: tree.end, island: tree, inlines: [] });
|
|
1118
|
+
return tree.end;
|
|
1119
|
+
}
|
|
1120
|
+
}
|
|
1121
|
+
}
|
|
1122
|
+
}
|
|
956
1123
|
const scan = this.scanInline(pos, end);
|
|
957
1124
|
if (!scan.breakIsland) {
|
|
958
1125
|
this.push({ kind: "prose", start: pos, end, island: null, inlines: scan.inlines });
|
|
@@ -985,17 +1152,49 @@ var Tokenizer = class {
|
|
|
985
1152
|
const text = this.text;
|
|
986
1153
|
const inlines = [];
|
|
987
1154
|
const hasKind = text.slice(ps, pe).includes("__kind");
|
|
1155
|
+
let kindBudget = 4 * (pe - ps) + 65536;
|
|
988
1156
|
const stop = (island, orphan = false) => ({ inlines, breakIsland: island, orphan });
|
|
1157
|
+
const within = (needle, from) => {
|
|
1158
|
+
const at = this.next(needle, from);
|
|
1159
|
+
return at !== -1 && at + needle.length <= pe ? at : -1;
|
|
1160
|
+
};
|
|
1161
|
+
let runs = null;
|
|
1162
|
+
const codeSpanClose = (n, from) => {
|
|
1163
|
+
if (!runs) {
|
|
1164
|
+
runs = /* @__PURE__ */ new Map();
|
|
1165
|
+
for (let k = ps; k < pe; ) {
|
|
1166
|
+
if (text.charCodeAt(k) !== 96) {
|
|
1167
|
+
k++;
|
|
1168
|
+
continue;
|
|
1169
|
+
}
|
|
1170
|
+
const len = backtickRunLength(text, k);
|
|
1171
|
+
const list2 = runs.get(len);
|
|
1172
|
+
if (list2) list2.push(k);
|
|
1173
|
+
else runs.set(len, [k]);
|
|
1174
|
+
k += len;
|
|
1175
|
+
}
|
|
1176
|
+
}
|
|
1177
|
+
const list = runs.get(n);
|
|
1178
|
+
if (!list) return -1;
|
|
1179
|
+
let lo = 0;
|
|
1180
|
+
let hi = list.length;
|
|
1181
|
+
while (lo < hi) {
|
|
1182
|
+
const mid = lo + hi >>> 1;
|
|
1183
|
+
if (list[mid] < from) lo = mid + 1;
|
|
1184
|
+
else hi = mid;
|
|
1185
|
+
}
|
|
1186
|
+
const at = list[lo];
|
|
1187
|
+
return at !== void 0 && at + n <= pe ? at : -1;
|
|
1188
|
+
};
|
|
989
1189
|
let i = ps;
|
|
990
1190
|
while (i < pe) {
|
|
991
1191
|
const ch = text[i];
|
|
992
1192
|
if (ch === "\\") {
|
|
993
|
-
const
|
|
994
|
-
if (
|
|
995
|
-
const
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: next === "(" ? "\\(" : "\\[" }));
|
|
1193
|
+
const nextCh = text[i + 1];
|
|
1194
|
+
if (nextCh === "(" || nextCh === "[") {
|
|
1195
|
+
const close = within(nextCh === "(" ? "\\)" : "\\]", i + 2);
|
|
1196
|
+
if (close !== -1) {
|
|
1197
|
+
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: nextCh === "(" ? "\\(" : "\\[" }));
|
|
999
1198
|
i = close + 2;
|
|
1000
1199
|
continue;
|
|
1001
1200
|
}
|
|
@@ -1005,33 +1204,20 @@ var Tokenizer = class {
|
|
|
1005
1204
|
}
|
|
1006
1205
|
if (ch === "`") {
|
|
1007
1206
|
const n = backtickRunLength(text, i);
|
|
1008
|
-
const
|
|
1009
|
-
let k = i + n;
|
|
1010
|
-
let close = -1;
|
|
1011
|
-
while (k < pe) {
|
|
1012
|
-
const idx = text.indexOf(run, k);
|
|
1013
|
-
if (idx === -1 || idx >= pe) break;
|
|
1014
|
-
if (text[idx + n] === "`") {
|
|
1015
|
-
let m = idx;
|
|
1016
|
-
while (text[m] === "`") m++;
|
|
1017
|
-
k = m;
|
|
1018
|
-
continue;
|
|
1019
|
-
}
|
|
1020
|
-
close = idx;
|
|
1021
|
-
break;
|
|
1022
|
-
}
|
|
1207
|
+
const close = codeSpanClose(n, i + n);
|
|
1023
1208
|
i = close === -1 ? i + n : close + n;
|
|
1024
1209
|
continue;
|
|
1025
1210
|
}
|
|
1026
1211
|
if (ch === "$") {
|
|
1027
1212
|
if (text[i + 1] === "$") {
|
|
1028
|
-
const close =
|
|
1029
|
-
if (close
|
|
1030
|
-
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: "$$" }));
|
|
1031
|
-
i = close + 2;
|
|
1032
|
-
} else {
|
|
1213
|
+
const close = this.mathPairs.get(i);
|
|
1214
|
+
if (close === void 0) {
|
|
1033
1215
|
i += 2;
|
|
1216
|
+
continue;
|
|
1034
1217
|
}
|
|
1218
|
+
if (close + 2 > pe) return stop(this.island("math_block", i, close + 2, true, { delimiter: "$$" }));
|
|
1219
|
+
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: "$$" }));
|
|
1220
|
+
i = close + 2;
|
|
1035
1221
|
continue;
|
|
1036
1222
|
}
|
|
1037
1223
|
const end = singleDollarMathEnd(text, i, pe);
|
|
@@ -1045,7 +1231,7 @@ var Tokenizer = class {
|
|
|
1045
1231
|
}
|
|
1046
1232
|
if (ch === "{") {
|
|
1047
1233
|
if (text[i + 1] === "{") {
|
|
1048
|
-
const close =
|
|
1234
|
+
const close = this.next("}}", i + 2);
|
|
1049
1235
|
if (close !== -1 && close + 2 <= pe && close - i <= 200 && !text.slice(i, close).includes("\n")) {
|
|
1050
1236
|
const raw = text.slice(i, close + 2);
|
|
1051
1237
|
const strict = STRICT_VARIABLE_RE.exec(raw);
|
|
@@ -1058,8 +1244,10 @@ var Tokenizer = class {
|
|
|
1058
1244
|
i += 2;
|
|
1059
1245
|
continue;
|
|
1060
1246
|
}
|
|
1061
|
-
if (hasKind) {
|
|
1062
|
-
const
|
|
1247
|
+
if (hasKind && kindBudget > 0 && /^\{\s*"/.test(text.slice(i, i + 64))) {
|
|
1248
|
+
const limit = Math.min(pe, i + 65536);
|
|
1249
|
+
const end = matchingJsonObjectEnd(text, i, limit);
|
|
1250
|
+
kindBudget -= (end ?? limit) - i;
|
|
1063
1251
|
if (end !== null) {
|
|
1064
1252
|
const kind = declaredKind(text.slice(i, end));
|
|
1065
1253
|
if (kind) {
|
|
@@ -1075,8 +1263,9 @@ var Tokenizer = class {
|
|
|
1075
1263
|
if (ch === "[") {
|
|
1076
1264
|
const media = MEDIA_REF_PREFIXES.find((prefix) => text.startsWith(prefix, i));
|
|
1077
1265
|
if (media) {
|
|
1078
|
-
const close =
|
|
1079
|
-
|
|
1266
|
+
const close = within("]", i + media.length);
|
|
1267
|
+
const newline = this.next("\n", i);
|
|
1268
|
+
if (close !== -1 && (newline === -1 || newline > close)) {
|
|
1080
1269
|
inlines.push(this.island("media_ref", i, close + 1, true, { media: media.slice(1, 6).toLowerCase() }));
|
|
1081
1270
|
i = close + 1;
|
|
1082
1271
|
continue;
|
|
@@ -1087,10 +1276,10 @@ var Tokenizer = class {
|
|
|
1087
1276
|
}
|
|
1088
1277
|
if (ch === "<") {
|
|
1089
1278
|
if (text.startsWith("<!--", i)) {
|
|
1090
|
-
const close =
|
|
1279
|
+
const close = this.next("-->", i + 4);
|
|
1091
1280
|
if (close === -1) return stop(this.island("html_comment", i, this.n, false));
|
|
1092
1281
|
const end = close + 3;
|
|
1093
|
-
const anchor = PINNED_ANCHOR_RE.exec(text.slice(i, end));
|
|
1282
|
+
const anchor = end - i <= 64 ? PINNED_ANCHOR_RE.exec(text.slice(i, end)) : null;
|
|
1094
1283
|
const type = anchor ? "anchor" : "html_comment";
|
|
1095
1284
|
const meta = anchor ? { id: anchor[1] } : {};
|
|
1096
1285
|
if (end > pe) return stop(this.island(type, i, end, true, meta));
|
|
@@ -1104,7 +1293,7 @@ var Tokenizer = class {
|
|
|
1104
1293
|
i += cite[0].length;
|
|
1105
1294
|
continue;
|
|
1106
1295
|
}
|
|
1107
|
-
const tag = readXmlTag(text, i);
|
|
1296
|
+
const tag = readXmlTag(text, i, this.finder);
|
|
1108
1297
|
if (tag && i + tag.raw.length <= pe) {
|
|
1109
1298
|
const name = tag.tagName;
|
|
1110
1299
|
const tagEnd = i + tag.raw.length;
|
|
@@ -1117,7 +1306,7 @@ var Tokenizer = class {
|
|
|
1117
1306
|
continue;
|
|
1118
1307
|
}
|
|
1119
1308
|
const closer = `</${name}>`;
|
|
1120
|
-
const close =
|
|
1309
|
+
const close = this.next(closer, tagEnd);
|
|
1121
1310
|
const blockType = isAttr ? "xml_attr" : "xml_region";
|
|
1122
1311
|
if (close === -1) return stop(this.island(blockType, i, this.n, false, { tag: name }));
|
|
1123
1312
|
const end = close + closer.length;
|
|
@@ -1141,7 +1330,7 @@ var Tokenizer = class {
|
|
|
1141
1330
|
const pending = ATTRIBUTE_XML_NAMES.find(
|
|
1142
1331
|
(name) => text.startsWith(`<${name}`, i) && /\s/.test(text[i + name.length + 1] ?? "")
|
|
1143
1332
|
);
|
|
1144
|
-
if (pending &&
|
|
1333
|
+
if (pending && this.next(">", i) === -1) {
|
|
1145
1334
|
return stop(this.island("xml_attr", i, this.n, false, { tag: pending }));
|
|
1146
1335
|
}
|
|
1147
1336
|
i++;
|
|
@@ -1232,9 +1421,22 @@ var SourceSpliceError = class extends Error {
|
|
|
1232
1421
|
function blockEdit(block, text) {
|
|
1233
1422
|
return { start: block.start, end: block.end, text };
|
|
1234
1423
|
}
|
|
1424
|
+
function islandEdit(island, text) {
|
|
1425
|
+
return { start: island.start, end: island.end, text, island: true };
|
|
1426
|
+
}
|
|
1235
1427
|
function blockRangeEdit(first, last, text) {
|
|
1236
1428
|
return { start: first.start, end: last.end, text };
|
|
1237
1429
|
}
|
|
1430
|
+
function locateIsland(text, raw, from, expected, expectedFromEnd) {
|
|
1431
|
+
if (expected >= from && text.startsWith(raw, expected)) return expected;
|
|
1432
|
+
if (expectedFromEnd >= from && text.startsWith(raw, expectedFromEnd)) return expectedFromEnd;
|
|
1433
|
+
const after = text.indexOf(raw, Math.max(from, expected));
|
|
1434
|
+
const before = expected > from ? text.lastIndexOf(raw, expected) : -1;
|
|
1435
|
+
const beforeOk = before >= from ? before : -1;
|
|
1436
|
+
if (after === -1) return beforeOk;
|
|
1437
|
+
if (beforeOk === -1) return after;
|
|
1438
|
+
return expected - beforeOk <= after - expected ? beforeOk : after;
|
|
1439
|
+
}
|
|
1238
1440
|
function commonPrefix(a, b) {
|
|
1239
1441
|
const max = Math.min(a.length, b.length);
|
|
1240
1442
|
let i = 0;
|
|
@@ -1259,13 +1461,23 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1259
1461
|
boundaries.add(block.start);
|
|
1260
1462
|
boundaries.add(block.end);
|
|
1261
1463
|
}
|
|
1464
|
+
const islands = listIslands(blocks);
|
|
1465
|
+
const islandAt = /* @__PURE__ */ new Map();
|
|
1466
|
+
for (const island of islands) islandAt.set(`${island.start}:${island.end}`, island);
|
|
1262
1467
|
const sorted = [...edits].sort((a, b) => a.start - b.start || a.end - b.end);
|
|
1263
1468
|
let previousEnd = -1;
|
|
1264
1469
|
for (const edit of sorted) {
|
|
1265
1470
|
if (edit.start < 0 || edit.end > original.length || edit.start > edit.end) {
|
|
1266
1471
|
throw new SourceSpliceError("out_of_range", `edit [${edit.start}, ${edit.end}) is outside the text`);
|
|
1267
1472
|
}
|
|
1268
|
-
if (
|
|
1473
|
+
if (edit.island) {
|
|
1474
|
+
if (!islandAt.has(`${edit.start}:${edit.end}`)) {
|
|
1475
|
+
throw new SourceSpliceError(
|
|
1476
|
+
"misaligned",
|
|
1477
|
+
`islandEdit [${edit.start}, ${edit.end}) does not name an island's exact span`
|
|
1478
|
+
);
|
|
1479
|
+
}
|
|
1480
|
+
} else if (!boundaries.has(edit.start) || !boundaries.has(edit.end)) {
|
|
1269
1481
|
throw new SourceSpliceError(
|
|
1270
1482
|
"misaligned",
|
|
1271
1483
|
`edit [${edit.start}, ${edit.end}) does not start and end on block boundaries`
|
|
@@ -1277,18 +1489,42 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1277
1489
|
previousEnd = edit.end;
|
|
1278
1490
|
}
|
|
1279
1491
|
const effective = [];
|
|
1280
|
-
|
|
1281
|
-
|
|
1282
|
-
|
|
1283
|
-
|
|
1284
|
-
const
|
|
1492
|
+
const replacedIslands = /* @__PURE__ */ new Set();
|
|
1493
|
+
const pushSegment = (oldStart, oldEnd, replacement) => {
|
|
1494
|
+
const old = original.slice(oldStart, oldEnd);
|
|
1495
|
+
if (old === replacement) return;
|
|
1496
|
+
const prefix = commonPrefix(old, replacement);
|
|
1497
|
+
const suffix = commonSuffix(old, replacement, prefix);
|
|
1285
1498
|
effective.push({
|
|
1286
|
-
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
oldEnd: edit.end - suffix,
|
|
1290
|
-
replacement: edit.text.slice(prefix, edit.text.length - suffix)
|
|
1499
|
+
oldStart: oldStart + prefix,
|
|
1500
|
+
oldEnd: oldEnd - suffix,
|
|
1501
|
+
replacement: replacement.slice(prefix, replacement.length - suffix)
|
|
1291
1502
|
});
|
|
1503
|
+
};
|
|
1504
|
+
for (const edit of sorted) {
|
|
1505
|
+
if (edit.island) {
|
|
1506
|
+
replacedIslands.add(`${edit.start}:${edit.end}`);
|
|
1507
|
+
pushSegment(edit.start, edit.end, edit.text);
|
|
1508
|
+
continue;
|
|
1509
|
+
}
|
|
1510
|
+
if (edit.text === original.slice(edit.start, edit.end)) continue;
|
|
1511
|
+
const inside = islands.filter((island) => island.start >= edit.start && island.end <= edit.end);
|
|
1512
|
+
let oldCursor2 = edit.start;
|
|
1513
|
+
let newCursor2 = 0;
|
|
1514
|
+
for (const island of inside) {
|
|
1515
|
+
const at = locateIsland(edit.text, island.raw, newCursor2, newCursor2 + (island.start - oldCursor2), edit.text.length - (edit.end - island.start));
|
|
1516
|
+
if (at === -1) {
|
|
1517
|
+
const name = island.meta.tag ?? island.meta.name ?? island.meta.lang ?? "";
|
|
1518
|
+
throw new SourceSpliceError(
|
|
1519
|
+
"island_edit",
|
|
1520
|
+
`edit [${edit.start}, ${edit.end}) changes or removes the protected ${island.islandType}${name ? ` (${String(name)})` : ""} island at [${island.start}, ${island.end}); islands change only through islandEdit()`
|
|
1521
|
+
);
|
|
1522
|
+
}
|
|
1523
|
+
pushSegment(oldCursor2, island.start, edit.text.slice(newCursor2, at));
|
|
1524
|
+
oldCursor2 = island.end;
|
|
1525
|
+
newCursor2 = at + island.raw.length;
|
|
1526
|
+
}
|
|
1527
|
+
pushSegment(oldCursor2, edit.end, edit.text.slice(newCursor2));
|
|
1292
1528
|
}
|
|
1293
1529
|
if (effective.length === 0) {
|
|
1294
1530
|
return {
|
|
@@ -1308,9 +1544,10 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1308
1544
|
const newStart = edit.oldStart + delta;
|
|
1309
1545
|
text += edit.replacement;
|
|
1310
1546
|
const newEnd = newStart + edit.replacement.length;
|
|
1547
|
+
const owner = sorted.find((e) => e.start <= edit.oldStart && e.end >= edit.oldEnd);
|
|
1311
1548
|
pending.push({
|
|
1312
|
-
blockStart: edit.
|
|
1313
|
-
blockEnd: edit.
|
|
1549
|
+
blockStart: owner?.start ?? edit.oldStart,
|
|
1550
|
+
blockEnd: owner?.end ?? edit.oldEnd,
|
|
1314
1551
|
oldStart: edit.oldStart,
|
|
1315
1552
|
oldEnd: edit.oldEnd,
|
|
1316
1553
|
newStart,
|
|
@@ -1348,11 +1585,8 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1348
1585
|
else newIslandsByStart.set(island.start, [island.raw]);
|
|
1349
1586
|
}
|
|
1350
1587
|
const disturbed = [];
|
|
1351
|
-
for (const island of
|
|
1352
|
-
|
|
1353
|
-
(edit) => island.start < edit.blockEnd && island.end > edit.blockStart
|
|
1354
|
-
);
|
|
1355
|
-
if (touched) continue;
|
|
1588
|
+
for (const island of islands) {
|
|
1589
|
+
if (replacedIslands.has(`${island.start}:${island.end}`)) continue;
|
|
1356
1590
|
const mappedStart = mapPosition(changes, island.start, { assoc: 1 }).pos;
|
|
1357
1591
|
const candidates = newIslandsByStart.get(mappedStart);
|
|
1358
1592
|
if (!candidates || !candidates.includes(island.raw)) {
|
|
@@ -1370,10 +1604,11 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1370
1604
|
bytesOutsideEditsIdentical,
|
|
1371
1605
|
disturbed
|
|
1372
1606
|
};
|
|
1373
|
-
if (options.requireIntegrity && !integrity.ok) {
|
|
1607
|
+
if ((options.requireIntegrity ?? true) && !integrity.ok) {
|
|
1608
|
+
const named = disturbed.slice(0, 5).map((island) => `${island.islandType}@${island.start}`).join(", ");
|
|
1374
1609
|
throw new SourceSpliceError(
|
|
1375
1610
|
"integrity",
|
|
1376
|
-
`the
|
|
1611
|
+
`integrity: the save disturbed ${disturbed.length} protected island(s) it did not replace${named ? ` (${named}${disturbed.length > 5 ? ", \u2026" : ""})` : ""}${bytesOutsideEditsIdentical ? "" : "; bytes outside the edits changed"}`
|
|
1377
1612
|
);
|
|
1378
1613
|
}
|
|
1379
1614
|
return { text, changed: true, changes, integrity, blocks: newBlocks };
|
|
@@ -1415,17 +1650,23 @@ function mapRange(changes, start, end, unit = "utf16") {
|
|
|
1415
1650
|
return { start: newStart, end: newEnd, touched, collapsed: newEnd === newStart && end > start };
|
|
1416
1651
|
}
|
|
1417
1652
|
|
|
1653
|
+
exports.NESTING_FENCE_LANGUAGES = NESTING_FENCE_LANGUAGES;
|
|
1418
1654
|
exports.SourceSpliceError = SourceSpliceError;
|
|
1419
1655
|
exports.XmlContainerTracker = XmlContainerTracker;
|
|
1420
1656
|
exports.blockAt = blockAt;
|
|
1421
1657
|
exports.blockEdit = blockEdit;
|
|
1422
1658
|
exports.blockRangeEdit = blockRangeEdit;
|
|
1423
1659
|
exports.buildCodePointIndex = buildCodePointIndex;
|
|
1660
|
+
exports.classifyInnerFenceLine = classifyInnerFenceLine;
|
|
1661
|
+
exports.fenceNestsInnerFences = fenceNestsInnerFences;
|
|
1424
1662
|
exports.isSingleDollarMath = isSingleDollarMath;
|
|
1663
|
+
exports.islandEdit = islandEdit;
|
|
1425
1664
|
exports.joinSource = joinSource;
|
|
1426
1665
|
exports.listIslands = listIslands;
|
|
1666
|
+
exports.looksLikeDisplayMath = looksLikeDisplayMath;
|
|
1427
1667
|
exports.mapPosition = mapPosition;
|
|
1428
1668
|
exports.mapRange = mapRange;
|
|
1669
|
+
exports.pairDisplayMath = pairDisplayMath;
|
|
1429
1670
|
exports.readXmlTag = readXmlTag;
|
|
1430
1671
|
exports.singleDollarMathEnd = singleDollarMathEnd;
|
|
1431
1672
|
exports.spliceSave = spliceSave;
|