@ai-matrx/content-ir 0.12.0 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +48 -0
- package/dist/index.cjs +328 -87
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.js +323 -88
- package/dist/index.js.map +1 -1
- package/dist/source.cjs +328 -87
- package/dist/source.cjs.map +1 -1
- package/dist/source.d.cts +76 -8
- package/dist/source.d.ts +76 -8
- package/dist/source.js +323 -88
- package/dist/source.js.map +1 -1
- package/package.json +1 -1
package/dist/source.js
CHANGED
|
@@ -200,6 +200,29 @@ function normalizeFenceLanguage(language) {
|
|
|
200
200
|
return CODE_LANGUAGE_ALIASES[lower] ?? lower;
|
|
201
201
|
}
|
|
202
202
|
|
|
203
|
+
// source/fence-nesting.ts
|
|
204
|
+
var NESTING_FENCE_LANGUAGES = /* @__PURE__ */ new Set([
|
|
205
|
+
"markdown",
|
|
206
|
+
"md",
|
|
207
|
+
"mdx"
|
|
208
|
+
]);
|
|
209
|
+
function fenceNestsInnerFences(language) {
|
|
210
|
+
return !!language && NESTING_FENCE_LANGUAGES.has(language.toLowerCase());
|
|
211
|
+
}
|
|
212
|
+
function classifyInnerFenceLine(trimmed, openTicks, nests, nestedDepth) {
|
|
213
|
+
let ticks = 0;
|
|
214
|
+
while (ticks < trimmed.length && trimmed[ticks] === "`") ticks++;
|
|
215
|
+
if (ticks < 3) return "content";
|
|
216
|
+
const info = trimmed.slice(ticks).trim();
|
|
217
|
+
if (info === "") {
|
|
218
|
+
if (ticks < openTicks) return "content";
|
|
219
|
+
if (nests && nestedDepth > 0) return "close-nested";
|
|
220
|
+
return "close-outer";
|
|
221
|
+
}
|
|
222
|
+
if (nests && ticks >= openTicks && !info.includes("`")) return "open-nested";
|
|
223
|
+
return "content";
|
|
224
|
+
}
|
|
225
|
+
|
|
203
226
|
// source/math.ts
|
|
204
227
|
var TEX_SIGNAL = /\\[A-Za-z]+|[\\^_{}=<>+]/;
|
|
205
228
|
var LONE_VARIABLE = /^[A-Za-z](?:'|[0-9])?$/;
|
|
@@ -212,8 +235,9 @@ function isSingleDollarMath(content, after) {
|
|
|
212
235
|
return TEX_SIGNAL.test(content);
|
|
213
236
|
}
|
|
214
237
|
function singleDollarMathEnd(text, open, limit = text.length) {
|
|
238
|
+
const bound = Math.min(limit, open + 403);
|
|
215
239
|
let close = -1;
|
|
216
|
-
for (let k = open + 1; k <
|
|
240
|
+
for (let k = open + 1; k < bound; k += 1) {
|
|
217
241
|
const c = text[k];
|
|
218
242
|
if (c === "\n") break;
|
|
219
243
|
if (c === "\\") {
|
|
@@ -230,11 +254,69 @@ function singleDollarMathEnd(text, open, limit = text.length) {
|
|
|
230
254
|
return isSingleDollarMath(content, text[close + 1]) ? close + 1 : -1;
|
|
231
255
|
}
|
|
232
256
|
|
|
257
|
+
// source/math-pairs.ts
|
|
258
|
+
var MAX_MATH_SPAN = 600;
|
|
259
|
+
var STRUCTURAL_MARKDOWN = /\]\(|https?:\/\/|\*\*|(?:^|\n)[ \t]{0,3}#{1,6}[ \t]|(?:^|\n)[ \t]*[-*+][ \t]+|(?:^|\n)[ \t]*\d+[.)][ \t]/;
|
|
260
|
+
var LATEX_COMMAND = /\\[a-zA-Z]/;
|
|
261
|
+
var PROSE_WORD = /[A-Za-z]{3,}/g;
|
|
262
|
+
var PROSE_WORD_LIMIT = 6;
|
|
263
|
+
function looksLikeDisplayMath(inner) {
|
|
264
|
+
const s = inner.trim();
|
|
265
|
+
if (!s) return false;
|
|
266
|
+
if (STRUCTURAL_MARKDOWN.test(s)) return false;
|
|
267
|
+
if (s.length > MAX_MATH_SPAN) return false;
|
|
268
|
+
if (LATEX_COMMAND.test(s)) {
|
|
269
|
+
return (s.match(PROSE_WORD) ?? []).length < PROSE_WORD_LIMIT * 3;
|
|
270
|
+
}
|
|
271
|
+
if (/\n[ \t]*\n/.test(s)) return false;
|
|
272
|
+
return (s.match(PROSE_WORD) ?? []).length < PROSE_WORD_LIMIT;
|
|
273
|
+
}
|
|
274
|
+
function codeRanges(text) {
|
|
275
|
+
const ranges = [];
|
|
276
|
+
for (const re of [/```[\s\S]*?(?:```|$)/g, /~~~[\s\S]*?(?:~~~|$)/g, /`[^`\n]*`/g]) {
|
|
277
|
+
let m;
|
|
278
|
+
while ((m = re.exec(text)) !== null) {
|
|
279
|
+
ranges.push([m.index, m.index + m[0].length]);
|
|
280
|
+
if (m[0].length === 0) re.lastIndex++;
|
|
281
|
+
}
|
|
282
|
+
}
|
|
283
|
+
return ranges.sort((a, b) => a[0] - b[0]);
|
|
284
|
+
}
|
|
285
|
+
function pairDisplayMath(text) {
|
|
286
|
+
const pairs = /* @__PURE__ */ new Map();
|
|
287
|
+
if (!text.includes("$$")) return pairs;
|
|
288
|
+
const ranges = codeRanges(text);
|
|
289
|
+
const tokens = [];
|
|
290
|
+
let r = 0;
|
|
291
|
+
let maxEnd = -1;
|
|
292
|
+
for (let i = text.indexOf("$$"); i !== -1 && i < text.length - 1; i = text.indexOf("$$", i)) {
|
|
293
|
+
while (r < ranges.length && ranges[r][0] <= i) {
|
|
294
|
+
maxEnd = Math.max(maxEnd, ranges[r][1]);
|
|
295
|
+
r++;
|
|
296
|
+
}
|
|
297
|
+
if (maxEnd <= i) tokens.push(i);
|
|
298
|
+
i += 2;
|
|
299
|
+
}
|
|
300
|
+
let j = 0;
|
|
301
|
+
while (j < tokens.length) {
|
|
302
|
+
const open = tokens[j];
|
|
303
|
+
const close = tokens[j + 1];
|
|
304
|
+
if (close === void 0) break;
|
|
305
|
+
if (looksLikeDisplayMath(text.slice(open + 2, close))) {
|
|
306
|
+
pairs.set(open, close);
|
|
307
|
+
j += 2;
|
|
308
|
+
continue;
|
|
309
|
+
}
|
|
310
|
+
j += 1;
|
|
311
|
+
}
|
|
312
|
+
return pairs;
|
|
313
|
+
}
|
|
314
|
+
|
|
233
315
|
// source/xml-tag.ts
|
|
234
316
|
var XML_NAME_START = /[A-Za-z_]/;
|
|
235
317
|
var XML_NAME_CHARACTER = /[\w.:-]/;
|
|
236
318
|
var WHITESPACE = /\s/;
|
|
237
|
-
function readXmlTag(content, start) {
|
|
319
|
+
function readXmlTag(content, start, find = (needle, from) => content.indexOf(needle, from)) {
|
|
238
320
|
let cursor = start + 1;
|
|
239
321
|
const isClosing = content[cursor] === "/";
|
|
240
322
|
if (isClosing) cursor++;
|
|
@@ -287,8 +369,9 @@ function readXmlTag(content, start) {
|
|
|
287
369
|
const quote = content[cursor];
|
|
288
370
|
if (quote !== '"' && quote !== "'") return null;
|
|
289
371
|
const valueStart = ++cursor;
|
|
290
|
-
|
|
291
|
-
if (
|
|
372
|
+
const valueEnd = find(quote, valueStart);
|
|
373
|
+
if (valueEnd === -1) return null;
|
|
374
|
+
cursor = valueEnd;
|
|
292
375
|
attributes.push({ name, value: content.slice(valueStart, cursor) });
|
|
293
376
|
cursor++;
|
|
294
377
|
}
|
|
@@ -507,11 +590,15 @@ function backtickRunLength(str, pos) {
|
|
|
507
590
|
while (pos + n < str.length && str[pos + n] === "`") n++;
|
|
508
591
|
return n;
|
|
509
592
|
}
|
|
593
|
+
function isSplitterTreeLine(line) {
|
|
594
|
+
return TREE_GLYPHS.test(line) || /^[\s│|]*[├└+|][\s─\-]+/.test(line);
|
|
595
|
+
}
|
|
510
596
|
var MEDIA_REF_PREFIXES = ["[Image URL:", "[Video URL:", "[Audio URL:"];
|
|
511
597
|
var Tokenizer = class {
|
|
512
598
|
constructor(text) {
|
|
513
599
|
this.text = text;
|
|
514
600
|
this.n = text.length;
|
|
601
|
+
this.mathPairs = pairDisplayMath(text);
|
|
515
602
|
let start = 0;
|
|
516
603
|
for (let i = 0; i < text.length; i++) {
|
|
517
604
|
if (text.charCodeAt(i) === 10) {
|
|
@@ -537,6 +624,34 @@ var Tokenizer = class {
|
|
|
537
624
|
/** Lazily built: for line i, first line j >= i where brace net from i drops to <= 0. */
|
|
538
625
|
braceStop = null;
|
|
539
626
|
braceNet = null;
|
|
627
|
+
/** Every occurrence offset of each needle searched so far (overlapping, like indexOf). */
|
|
628
|
+
occurrences = /* @__PURE__ */ new Map();
|
|
629
|
+
/** `$$` opener → closer for the pairs the core renders as math. */
|
|
630
|
+
mathPairs;
|
|
631
|
+
finder = (needle, from) => this.next(needle, from);
|
|
632
|
+
/**
|
|
633
|
+
* `text.indexOf(needle, from)` in O(log k): the first query for a needle
|
|
634
|
+
* indexes all of its occurrences once, so repeated searches for a closer
|
|
635
|
+
* that never comes stay linear over the whole text.
|
|
636
|
+
*/
|
|
637
|
+
next(needle, from) {
|
|
638
|
+
let list = this.occurrences.get(needle);
|
|
639
|
+
if (!list) {
|
|
640
|
+
list = [];
|
|
641
|
+
for (let at = this.text.indexOf(needle); at !== -1; at = this.text.indexOf(needle, at + 1)) {
|
|
642
|
+
list.push(at);
|
|
643
|
+
}
|
|
644
|
+
this.occurrences.set(needle, list);
|
|
645
|
+
}
|
|
646
|
+
let lo = 0;
|
|
647
|
+
let hi = list.length;
|
|
648
|
+
while (lo < hi) {
|
|
649
|
+
const mid = lo + hi >>> 1;
|
|
650
|
+
if (list[mid] < from) lo = mid + 1;
|
|
651
|
+
else hi = mid;
|
|
652
|
+
}
|
|
653
|
+
return lo < list.length ? list[lo] : -1;
|
|
654
|
+
}
|
|
540
655
|
get lineCount() {
|
|
541
656
|
return this.lineStarts.length;
|
|
542
657
|
}
|
|
@@ -668,7 +783,12 @@ var Tokenizer = class {
|
|
|
668
783
|
}
|
|
669
784
|
return null;
|
|
670
785
|
}
|
|
671
|
-
/**
|
|
786
|
+
/**
|
|
787
|
+
* Backtick fence — the renderer splitter's close rules: JSON-string aware for
|
|
788
|
+
* ```json, and THE nested-fence rule (source/fence-nesting.ts) for markdown
|
|
789
|
+
* fences, with the splitter's strict-CommonMark retry when a nested fence is
|
|
790
|
+
* still open at the end of the text.
|
|
791
|
+
*/
|
|
672
792
|
backtickFence(p, li, t) {
|
|
673
793
|
const openTicks = backtickRunLength(t, 0);
|
|
674
794
|
const info = t.slice(openTicks).trim();
|
|
@@ -677,7 +797,19 @@ var Tokenizer = class {
|
|
|
677
797
|
const normalized = normalizeFenceLanguage(lang);
|
|
678
798
|
const meta = { fence: "`", ticks: openTicks, lang };
|
|
679
799
|
if (normalized && SPECIAL_CODE_LANGUAGES.has(normalized)) meta.special = normalized;
|
|
800
|
+
const nesting = fenceNestsInnerFences(lang);
|
|
801
|
+
let close = this.backtickFenceClose(li, openTicks, isJson, nesting);
|
|
802
|
+
if (close === "retry-strict") close = this.backtickFenceClose(li, openTicks, isJson, false);
|
|
803
|
+
if (typeof close === "number") {
|
|
804
|
+
return this.fenceIsland(p, this.contentEnds[close], true, meta, li, close, isJson);
|
|
805
|
+
}
|
|
806
|
+
return this.fenceIsland(p, this.n, false, meta, li, this.lineCount, isJson);
|
|
807
|
+
}
|
|
808
|
+
/** The closing line of a backtick fence opened on line `li`, or null when it never closes. */
|
|
809
|
+
backtickFenceClose(li, openTicks, isJson, nesting) {
|
|
680
810
|
let state = { inString: false, escaped: false };
|
|
811
|
+
let nestedDepth = 0;
|
|
812
|
+
let sawNested = false;
|
|
681
813
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
682
814
|
const line = this.lineText(k);
|
|
683
815
|
const trimmedLine = line.trim();
|
|
@@ -690,11 +822,13 @@ var Tokenizer = class {
|
|
|
690
822
|
continue;
|
|
691
823
|
}
|
|
692
824
|
}
|
|
693
|
-
const
|
|
694
|
-
|
|
695
|
-
if (
|
|
696
|
-
|
|
825
|
+
const kind = classifyInnerFenceLine(trimmedLine, openTicks, nesting, nestedDepth);
|
|
826
|
+
if (kind === "close-outer") return k;
|
|
827
|
+
if (kind === "open-nested") {
|
|
828
|
+
nestedDepth++;
|
|
829
|
+
sawNested = true;
|
|
697
830
|
}
|
|
831
|
+
if (kind === "close-nested") nestedDepth--;
|
|
698
832
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
699
833
|
continue;
|
|
700
834
|
}
|
|
@@ -709,15 +843,13 @@ var Tokenizer = class {
|
|
|
709
843
|
}
|
|
710
844
|
const closeTicks = backtickRunLength(line, at);
|
|
711
845
|
const after = line.slice(at + closeTicks).trim();
|
|
712
|
-
if (closeTicks >= openTicks && after === "")
|
|
713
|
-
return this.fenceIsland(p, this.contentEnds[k], true, meta, li, k, isJson);
|
|
714
|
-
}
|
|
846
|
+
if (closeTicks >= openTicks && after === "" && nestedDepth === 0) return k;
|
|
715
847
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
716
848
|
continue;
|
|
717
849
|
}
|
|
718
850
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
719
851
|
}
|
|
720
|
-
return
|
|
852
|
+
return sawNested && nestedDepth > 0 ? "retry-strict" : null;
|
|
721
853
|
}
|
|
722
854
|
fenceIsland(p, end, complete, meta, openLine, closeLine, isJson) {
|
|
723
855
|
if (isJson && openLine + 1 < this.lineCount) {
|
|
@@ -743,18 +875,22 @@ var Tokenizer = class {
|
|
|
743
875
|
}
|
|
744
876
|
return this.island("fence", p, this.n, false, meta);
|
|
745
877
|
}
|
|
878
|
+
/**
|
|
879
|
+
* A line-leading `$$` that OPENS a math pair (source/math-pairs.ts — the
|
|
880
|
+
* core's rule). A closer, an unpaired `$$`, or a prose pair is never a block:
|
|
881
|
+
* a lone `$$` renders as literal text and locks nothing.
|
|
882
|
+
*/
|
|
746
883
|
mathBlock(p, tp) {
|
|
884
|
+
const close = this.mathPairs.get(tp);
|
|
885
|
+
if (close === void 0) return null;
|
|
886
|
+
const end = close + 2;
|
|
747
887
|
const li = this.lineOf(tp);
|
|
748
888
|
const ce = this.contentEnds[li];
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
}
|
|
755
|
-
const close = this.text.indexOf("$$", tp + 2);
|
|
756
|
-
if (close === -1) return this.island("math_block", p, this.n, false, { delimiter: "$$" });
|
|
757
|
-
return this.island("math_block", p, close + 2, true, { delimiter: "$$" });
|
|
889
|
+
if (end <= ce && !this.isBlank(end, ce)) return null;
|
|
890
|
+
const endLine = this.lineOf(close);
|
|
891
|
+
const endCe = this.contentEnds[endLine];
|
|
892
|
+
if (endLine !== li && !this.isBlank(end, endCe)) return null;
|
|
893
|
+
return this.island("math_block", p, end, true, { delimiter: "$$" });
|
|
758
894
|
}
|
|
759
895
|
/** `<artifact …>`, `<decision …>`, … at the start of a line. */
|
|
760
896
|
attributeXml(p, tp, t) {
|
|
@@ -771,7 +907,7 @@ var Tokenizer = class {
|
|
|
771
907
|
attributeElement(name, p, tp, type) {
|
|
772
908
|
const li = this.lineOf(tp);
|
|
773
909
|
const ce = this.contentEnds[li];
|
|
774
|
-
const tag = readXmlTag(this.text, tp);
|
|
910
|
+
const tag = readXmlTag(this.text, tp, this.finder);
|
|
775
911
|
let openEnd;
|
|
776
912
|
if (tag && !tag.isClosing) {
|
|
777
913
|
if (tag.isSelfClosing) {
|
|
@@ -779,13 +915,13 @@ var Tokenizer = class {
|
|
|
779
915
|
}
|
|
780
916
|
openEnd = tp + tag.raw.length;
|
|
781
917
|
} else {
|
|
782
|
-
const gt = this.
|
|
918
|
+
const gt = this.next(">", tp);
|
|
783
919
|
if (gt === -1) return this.island(type, p, this.n, false, { tag: name });
|
|
784
920
|
if (gt >= ce) return null;
|
|
785
921
|
openEnd = gt + 1;
|
|
786
922
|
}
|
|
787
923
|
const closer = `</${name}>`;
|
|
788
|
-
const close = this.
|
|
924
|
+
const close = this.next(closer, openEnd);
|
|
789
925
|
if (close === -1) return this.island(type, p, this.n, false, { tag: name });
|
|
790
926
|
return this.island(type, p, close + closer.length, true, { tag: name });
|
|
791
927
|
}
|
|
@@ -795,7 +931,7 @@ var Tokenizer = class {
|
|
|
795
931
|
const opener = `<${name}>`;
|
|
796
932
|
if (!t.startsWith(opener)) continue;
|
|
797
933
|
const closer = `</${name}>`;
|
|
798
|
-
const close = this.
|
|
934
|
+
const close = this.next(closer, tp + opener.length);
|
|
799
935
|
if (close === -1) return this.island("xml_region", p, this.n, false, { tag: name });
|
|
800
936
|
return this.island("xml_region", p, close + closer.length, true, { tag: name });
|
|
801
937
|
}
|
|
@@ -831,7 +967,7 @@ var Tokenizer = class {
|
|
|
831
967
|
return null;
|
|
832
968
|
}
|
|
833
969
|
blockComment(p, tp, li) {
|
|
834
|
-
const close = this.
|
|
970
|
+
const close = this.next("-->", tp + 4);
|
|
835
971
|
if (close === -1) return this.island("html_comment", p, this.n, false);
|
|
836
972
|
const end = close + 3;
|
|
837
973
|
const raw = this.text.slice(tp, end);
|
|
@@ -909,9 +1045,8 @@ var Tokenizer = class {
|
|
|
909
1045
|
}
|
|
910
1046
|
}
|
|
911
1047
|
if (lastLine === null) {
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
return this.island("json", p, this.n, false, jsonMeta(partial));
|
|
1048
|
+
if (isRecognizedJsonHead(this.text.slice(tp, tp + 512))) {
|
|
1049
|
+
return this.island("json", p, this.n, false, jsonMeta(this.text.slice(tp)));
|
|
915
1050
|
}
|
|
916
1051
|
return null;
|
|
917
1052
|
}
|
|
@@ -946,11 +1081,43 @@ var Tokenizer = class {
|
|
|
946
1081
|
// ── prose ──────────────────────────────────────────────────────────────
|
|
947
1082
|
paragraph(pos, li) {
|
|
948
1083
|
let end = this.contentEnds[li];
|
|
1084
|
+
let breakLine = -1;
|
|
949
1085
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
950
1086
|
if (this.lineIsBlank(k)) break;
|
|
951
|
-
if (this.detectBlock(this.lineStarts[k], k, false))
|
|
1087
|
+
if (this.detectBlock(this.lineStarts[k], k, false)) {
|
|
1088
|
+
breakLine = k;
|
|
1089
|
+
break;
|
|
1090
|
+
}
|
|
952
1091
|
end = this.contentEnds[k];
|
|
953
1092
|
}
|
|
1093
|
+
if (breakLine !== -1) {
|
|
1094
|
+
const breakIsland = this.detectBlock(this.lineStarts[breakLine], breakLine, false);
|
|
1095
|
+
if (breakIsland?.type === "tree") {
|
|
1096
|
+
let root = breakLine;
|
|
1097
|
+
for (let j = breakLine - 1; j >= li; j--) {
|
|
1098
|
+
const trimmed = this.lineText(j).trim();
|
|
1099
|
+
if (!trimmed || isSplitterTreeLine(trimmed) || /^#{1,6}\s/.test(trimmed)) break;
|
|
1100
|
+
root = j;
|
|
1101
|
+
}
|
|
1102
|
+
if (root < breakLine) {
|
|
1103
|
+
const treeStart = root === li ? pos : this.lineStarts[root];
|
|
1104
|
+
const tree = { ...breakIsland, start: treeStart };
|
|
1105
|
+
if (treeStart > pos) {
|
|
1106
|
+
const proseEnd = this.contentEnds[root - 1];
|
|
1107
|
+
const scan2 = this.scanInline(pos, proseEnd);
|
|
1108
|
+
if (!scan2.breakIsland) {
|
|
1109
|
+
this.push({ kind: "prose", start: pos, end: proseEnd, island: null, inlines: scan2.inlines });
|
|
1110
|
+
this.push({ kind: "gap", start: proseEnd, end: treeStart, island: null, inlines: [] });
|
|
1111
|
+
this.push({ kind: "island", start: treeStart, end: tree.end, island: tree, inlines: [] });
|
|
1112
|
+
return tree.end;
|
|
1113
|
+
}
|
|
1114
|
+
} else {
|
|
1115
|
+
this.push({ kind: "island", start: treeStart, end: tree.end, island: tree, inlines: [] });
|
|
1116
|
+
return tree.end;
|
|
1117
|
+
}
|
|
1118
|
+
}
|
|
1119
|
+
}
|
|
1120
|
+
}
|
|
954
1121
|
const scan = this.scanInline(pos, end);
|
|
955
1122
|
if (!scan.breakIsland) {
|
|
956
1123
|
this.push({ kind: "prose", start: pos, end, island: null, inlines: scan.inlines });
|
|
@@ -983,17 +1150,49 @@ var Tokenizer = class {
|
|
|
983
1150
|
const text = this.text;
|
|
984
1151
|
const inlines = [];
|
|
985
1152
|
const hasKind = text.slice(ps, pe).includes("__kind");
|
|
1153
|
+
let kindBudget = 4 * (pe - ps) + 65536;
|
|
986
1154
|
const stop = (island, orphan = false) => ({ inlines, breakIsland: island, orphan });
|
|
1155
|
+
const within = (needle, from) => {
|
|
1156
|
+
const at = this.next(needle, from);
|
|
1157
|
+
return at !== -1 && at + needle.length <= pe ? at : -1;
|
|
1158
|
+
};
|
|
1159
|
+
let runs = null;
|
|
1160
|
+
const codeSpanClose = (n, from) => {
|
|
1161
|
+
if (!runs) {
|
|
1162
|
+
runs = /* @__PURE__ */ new Map();
|
|
1163
|
+
for (let k = ps; k < pe; ) {
|
|
1164
|
+
if (text.charCodeAt(k) !== 96) {
|
|
1165
|
+
k++;
|
|
1166
|
+
continue;
|
|
1167
|
+
}
|
|
1168
|
+
const len = backtickRunLength(text, k);
|
|
1169
|
+
const list2 = runs.get(len);
|
|
1170
|
+
if (list2) list2.push(k);
|
|
1171
|
+
else runs.set(len, [k]);
|
|
1172
|
+
k += len;
|
|
1173
|
+
}
|
|
1174
|
+
}
|
|
1175
|
+
const list = runs.get(n);
|
|
1176
|
+
if (!list) return -1;
|
|
1177
|
+
let lo = 0;
|
|
1178
|
+
let hi = list.length;
|
|
1179
|
+
while (lo < hi) {
|
|
1180
|
+
const mid = lo + hi >>> 1;
|
|
1181
|
+
if (list[mid] < from) lo = mid + 1;
|
|
1182
|
+
else hi = mid;
|
|
1183
|
+
}
|
|
1184
|
+
const at = list[lo];
|
|
1185
|
+
return at !== void 0 && at + n <= pe ? at : -1;
|
|
1186
|
+
};
|
|
987
1187
|
let i = ps;
|
|
988
1188
|
while (i < pe) {
|
|
989
1189
|
const ch = text[i];
|
|
990
1190
|
if (ch === "\\") {
|
|
991
|
-
const
|
|
992
|
-
if (
|
|
993
|
-
const
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: next === "(" ? "\\(" : "\\[" }));
|
|
1191
|
+
const nextCh = text[i + 1];
|
|
1192
|
+
if (nextCh === "(" || nextCh === "[") {
|
|
1193
|
+
const close = within(nextCh === "(" ? "\\)" : "\\]", i + 2);
|
|
1194
|
+
if (close !== -1) {
|
|
1195
|
+
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: nextCh === "(" ? "\\(" : "\\[" }));
|
|
997
1196
|
i = close + 2;
|
|
998
1197
|
continue;
|
|
999
1198
|
}
|
|
@@ -1003,33 +1202,20 @@ var Tokenizer = class {
|
|
|
1003
1202
|
}
|
|
1004
1203
|
if (ch === "`") {
|
|
1005
1204
|
const n = backtickRunLength(text, i);
|
|
1006
|
-
const
|
|
1007
|
-
let k = i + n;
|
|
1008
|
-
let close = -1;
|
|
1009
|
-
while (k < pe) {
|
|
1010
|
-
const idx = text.indexOf(run, k);
|
|
1011
|
-
if (idx === -1 || idx >= pe) break;
|
|
1012
|
-
if (text[idx + n] === "`") {
|
|
1013
|
-
let m = idx;
|
|
1014
|
-
while (text[m] === "`") m++;
|
|
1015
|
-
k = m;
|
|
1016
|
-
continue;
|
|
1017
|
-
}
|
|
1018
|
-
close = idx;
|
|
1019
|
-
break;
|
|
1020
|
-
}
|
|
1205
|
+
const close = codeSpanClose(n, i + n);
|
|
1021
1206
|
i = close === -1 ? i + n : close + n;
|
|
1022
1207
|
continue;
|
|
1023
1208
|
}
|
|
1024
1209
|
if (ch === "$") {
|
|
1025
1210
|
if (text[i + 1] === "$") {
|
|
1026
|
-
const close =
|
|
1027
|
-
if (close
|
|
1028
|
-
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: "$$" }));
|
|
1029
|
-
i = close + 2;
|
|
1030
|
-
} else {
|
|
1211
|
+
const close = this.mathPairs.get(i);
|
|
1212
|
+
if (close === void 0) {
|
|
1031
1213
|
i += 2;
|
|
1214
|
+
continue;
|
|
1032
1215
|
}
|
|
1216
|
+
if (close + 2 > pe) return stop(this.island("math_block", i, close + 2, true, { delimiter: "$$" }));
|
|
1217
|
+
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: "$$" }));
|
|
1218
|
+
i = close + 2;
|
|
1033
1219
|
continue;
|
|
1034
1220
|
}
|
|
1035
1221
|
const end = singleDollarMathEnd(text, i, pe);
|
|
@@ -1043,7 +1229,7 @@ var Tokenizer = class {
|
|
|
1043
1229
|
}
|
|
1044
1230
|
if (ch === "{") {
|
|
1045
1231
|
if (text[i + 1] === "{") {
|
|
1046
|
-
const close =
|
|
1232
|
+
const close = this.next("}}", i + 2);
|
|
1047
1233
|
if (close !== -1 && close + 2 <= pe && close - i <= 200 && !text.slice(i, close).includes("\n")) {
|
|
1048
1234
|
const raw = text.slice(i, close + 2);
|
|
1049
1235
|
const strict = STRICT_VARIABLE_RE.exec(raw);
|
|
@@ -1056,8 +1242,10 @@ var Tokenizer = class {
|
|
|
1056
1242
|
i += 2;
|
|
1057
1243
|
continue;
|
|
1058
1244
|
}
|
|
1059
|
-
if (hasKind) {
|
|
1060
|
-
const
|
|
1245
|
+
if (hasKind && kindBudget > 0 && /^\{\s*"/.test(text.slice(i, i + 64))) {
|
|
1246
|
+
const limit = Math.min(pe, i + 65536);
|
|
1247
|
+
const end = matchingJsonObjectEnd(text, i, limit);
|
|
1248
|
+
kindBudget -= (end ?? limit) - i;
|
|
1061
1249
|
if (end !== null) {
|
|
1062
1250
|
const kind = declaredKind(text.slice(i, end));
|
|
1063
1251
|
if (kind) {
|
|
@@ -1073,8 +1261,9 @@ var Tokenizer = class {
|
|
|
1073
1261
|
if (ch === "[") {
|
|
1074
1262
|
const media = MEDIA_REF_PREFIXES.find((prefix) => text.startsWith(prefix, i));
|
|
1075
1263
|
if (media) {
|
|
1076
|
-
const close =
|
|
1077
|
-
|
|
1264
|
+
const close = within("]", i + media.length);
|
|
1265
|
+
const newline = this.next("\n", i);
|
|
1266
|
+
if (close !== -1 && (newline === -1 || newline > close)) {
|
|
1078
1267
|
inlines.push(this.island("media_ref", i, close + 1, true, { media: media.slice(1, 6).toLowerCase() }));
|
|
1079
1268
|
i = close + 1;
|
|
1080
1269
|
continue;
|
|
@@ -1085,10 +1274,10 @@ var Tokenizer = class {
|
|
|
1085
1274
|
}
|
|
1086
1275
|
if (ch === "<") {
|
|
1087
1276
|
if (text.startsWith("<!--", i)) {
|
|
1088
|
-
const close =
|
|
1277
|
+
const close = this.next("-->", i + 4);
|
|
1089
1278
|
if (close === -1) return stop(this.island("html_comment", i, this.n, false));
|
|
1090
1279
|
const end = close + 3;
|
|
1091
|
-
const anchor = PINNED_ANCHOR_RE.exec(text.slice(i, end));
|
|
1280
|
+
const anchor = end - i <= 64 ? PINNED_ANCHOR_RE.exec(text.slice(i, end)) : null;
|
|
1092
1281
|
const type = anchor ? "anchor" : "html_comment";
|
|
1093
1282
|
const meta = anchor ? { id: anchor[1] } : {};
|
|
1094
1283
|
if (end > pe) return stop(this.island(type, i, end, true, meta));
|
|
@@ -1102,7 +1291,7 @@ var Tokenizer = class {
|
|
|
1102
1291
|
i += cite[0].length;
|
|
1103
1292
|
continue;
|
|
1104
1293
|
}
|
|
1105
|
-
const tag = readXmlTag(text, i);
|
|
1294
|
+
const tag = readXmlTag(text, i, this.finder);
|
|
1106
1295
|
if (tag && i + tag.raw.length <= pe) {
|
|
1107
1296
|
const name = tag.tagName;
|
|
1108
1297
|
const tagEnd = i + tag.raw.length;
|
|
@@ -1115,7 +1304,7 @@ var Tokenizer = class {
|
|
|
1115
1304
|
continue;
|
|
1116
1305
|
}
|
|
1117
1306
|
const closer = `</${name}>`;
|
|
1118
|
-
const close =
|
|
1307
|
+
const close = this.next(closer, tagEnd);
|
|
1119
1308
|
const blockType = isAttr ? "xml_attr" : "xml_region";
|
|
1120
1309
|
if (close === -1) return stop(this.island(blockType, i, this.n, false, { tag: name }));
|
|
1121
1310
|
const end = close + closer.length;
|
|
@@ -1139,7 +1328,7 @@ var Tokenizer = class {
|
|
|
1139
1328
|
const pending = ATTRIBUTE_XML_NAMES.find(
|
|
1140
1329
|
(name) => text.startsWith(`<${name}`, i) && /\s/.test(text[i + name.length + 1] ?? "")
|
|
1141
1330
|
);
|
|
1142
|
-
if (pending &&
|
|
1331
|
+
if (pending && this.next(">", i) === -1) {
|
|
1143
1332
|
return stop(this.island("xml_attr", i, this.n, false, { tag: pending }));
|
|
1144
1333
|
}
|
|
1145
1334
|
i++;
|
|
@@ -1230,9 +1419,22 @@ var SourceSpliceError = class extends Error {
|
|
|
1230
1419
|
function blockEdit(block, text) {
|
|
1231
1420
|
return { start: block.start, end: block.end, text };
|
|
1232
1421
|
}
|
|
1422
|
+
function islandEdit(island, text) {
|
|
1423
|
+
return { start: island.start, end: island.end, text, island: true };
|
|
1424
|
+
}
|
|
1233
1425
|
function blockRangeEdit(first, last, text) {
|
|
1234
1426
|
return { start: first.start, end: last.end, text };
|
|
1235
1427
|
}
|
|
1428
|
+
function locateIsland(text, raw, from, expected, expectedFromEnd) {
|
|
1429
|
+
if (expected >= from && text.startsWith(raw, expected)) return expected;
|
|
1430
|
+
if (expectedFromEnd >= from && text.startsWith(raw, expectedFromEnd)) return expectedFromEnd;
|
|
1431
|
+
const after = text.indexOf(raw, Math.max(from, expected));
|
|
1432
|
+
const before = expected > from ? text.lastIndexOf(raw, expected) : -1;
|
|
1433
|
+
const beforeOk = before >= from ? before : -1;
|
|
1434
|
+
if (after === -1) return beforeOk;
|
|
1435
|
+
if (beforeOk === -1) return after;
|
|
1436
|
+
return expected - beforeOk <= after - expected ? beforeOk : after;
|
|
1437
|
+
}
|
|
1236
1438
|
function commonPrefix(a, b) {
|
|
1237
1439
|
const max = Math.min(a.length, b.length);
|
|
1238
1440
|
let i = 0;
|
|
@@ -1257,13 +1459,23 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1257
1459
|
boundaries.add(block.start);
|
|
1258
1460
|
boundaries.add(block.end);
|
|
1259
1461
|
}
|
|
1462
|
+
const islands = listIslands(blocks);
|
|
1463
|
+
const islandAt = /* @__PURE__ */ new Map();
|
|
1464
|
+
for (const island of islands) islandAt.set(`${island.start}:${island.end}`, island);
|
|
1260
1465
|
const sorted = [...edits].sort((a, b) => a.start - b.start || a.end - b.end);
|
|
1261
1466
|
let previousEnd = -1;
|
|
1262
1467
|
for (const edit of sorted) {
|
|
1263
1468
|
if (edit.start < 0 || edit.end > original.length || edit.start > edit.end) {
|
|
1264
1469
|
throw new SourceSpliceError("out_of_range", `edit [${edit.start}, ${edit.end}) is outside the text`);
|
|
1265
1470
|
}
|
|
1266
|
-
if (
|
|
1471
|
+
if (edit.island) {
|
|
1472
|
+
if (!islandAt.has(`${edit.start}:${edit.end}`)) {
|
|
1473
|
+
throw new SourceSpliceError(
|
|
1474
|
+
"misaligned",
|
|
1475
|
+
`islandEdit [${edit.start}, ${edit.end}) does not name an island's exact span`
|
|
1476
|
+
);
|
|
1477
|
+
}
|
|
1478
|
+
} else if (!boundaries.has(edit.start) || !boundaries.has(edit.end)) {
|
|
1267
1479
|
throw new SourceSpliceError(
|
|
1268
1480
|
"misaligned",
|
|
1269
1481
|
`edit [${edit.start}, ${edit.end}) does not start and end on block boundaries`
|
|
@@ -1275,18 +1487,42 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1275
1487
|
previousEnd = edit.end;
|
|
1276
1488
|
}
|
|
1277
1489
|
const effective = [];
|
|
1278
|
-
|
|
1279
|
-
|
|
1280
|
-
|
|
1281
|
-
|
|
1282
|
-
const
|
|
1490
|
+
const replacedIslands = /* @__PURE__ */ new Set();
|
|
1491
|
+
const pushSegment = (oldStart, oldEnd, replacement) => {
|
|
1492
|
+
const old = original.slice(oldStart, oldEnd);
|
|
1493
|
+
if (old === replacement) return;
|
|
1494
|
+
const prefix = commonPrefix(old, replacement);
|
|
1495
|
+
const suffix = commonSuffix(old, replacement, prefix);
|
|
1283
1496
|
effective.push({
|
|
1284
|
-
|
|
1285
|
-
|
|
1286
|
-
|
|
1287
|
-
oldEnd: edit.end - suffix,
|
|
1288
|
-
replacement: edit.text.slice(prefix, edit.text.length - suffix)
|
|
1497
|
+
oldStart: oldStart + prefix,
|
|
1498
|
+
oldEnd: oldEnd - suffix,
|
|
1499
|
+
replacement: replacement.slice(prefix, replacement.length - suffix)
|
|
1289
1500
|
});
|
|
1501
|
+
};
|
|
1502
|
+
for (const edit of sorted) {
|
|
1503
|
+
if (edit.island) {
|
|
1504
|
+
replacedIslands.add(`${edit.start}:${edit.end}`);
|
|
1505
|
+
pushSegment(edit.start, edit.end, edit.text);
|
|
1506
|
+
continue;
|
|
1507
|
+
}
|
|
1508
|
+
if (edit.text === original.slice(edit.start, edit.end)) continue;
|
|
1509
|
+
const inside = islands.filter((island) => island.start >= edit.start && island.end <= edit.end);
|
|
1510
|
+
let oldCursor2 = edit.start;
|
|
1511
|
+
let newCursor2 = 0;
|
|
1512
|
+
for (const island of inside) {
|
|
1513
|
+
const at = locateIsland(edit.text, island.raw, newCursor2, newCursor2 + (island.start - oldCursor2), edit.text.length - (edit.end - island.start));
|
|
1514
|
+
if (at === -1) {
|
|
1515
|
+
const name = island.meta.tag ?? island.meta.name ?? island.meta.lang ?? "";
|
|
1516
|
+
throw new SourceSpliceError(
|
|
1517
|
+
"island_edit",
|
|
1518
|
+
`edit [${edit.start}, ${edit.end}) changes or removes the protected ${island.islandType}${name ? ` (${String(name)})` : ""} island at [${island.start}, ${island.end}); islands change only through islandEdit()`
|
|
1519
|
+
);
|
|
1520
|
+
}
|
|
1521
|
+
pushSegment(oldCursor2, island.start, edit.text.slice(newCursor2, at));
|
|
1522
|
+
oldCursor2 = island.end;
|
|
1523
|
+
newCursor2 = at + island.raw.length;
|
|
1524
|
+
}
|
|
1525
|
+
pushSegment(oldCursor2, edit.end, edit.text.slice(newCursor2));
|
|
1290
1526
|
}
|
|
1291
1527
|
if (effective.length === 0) {
|
|
1292
1528
|
return {
|
|
@@ -1306,9 +1542,10 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1306
1542
|
const newStart = edit.oldStart + delta;
|
|
1307
1543
|
text += edit.replacement;
|
|
1308
1544
|
const newEnd = newStart + edit.replacement.length;
|
|
1545
|
+
const owner = sorted.find((e) => e.start <= edit.oldStart && e.end >= edit.oldEnd);
|
|
1309
1546
|
pending.push({
|
|
1310
|
-
blockStart: edit.
|
|
1311
|
-
blockEnd: edit.
|
|
1547
|
+
blockStart: owner?.start ?? edit.oldStart,
|
|
1548
|
+
blockEnd: owner?.end ?? edit.oldEnd,
|
|
1312
1549
|
oldStart: edit.oldStart,
|
|
1313
1550
|
oldEnd: edit.oldEnd,
|
|
1314
1551
|
newStart,
|
|
@@ -1346,11 +1583,8 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1346
1583
|
else newIslandsByStart.set(island.start, [island.raw]);
|
|
1347
1584
|
}
|
|
1348
1585
|
const disturbed = [];
|
|
1349
|
-
for (const island of
|
|
1350
|
-
|
|
1351
|
-
(edit) => island.start < edit.blockEnd && island.end > edit.blockStart
|
|
1352
|
-
);
|
|
1353
|
-
if (touched) continue;
|
|
1586
|
+
for (const island of islands) {
|
|
1587
|
+
if (replacedIslands.has(`${island.start}:${island.end}`)) continue;
|
|
1354
1588
|
const mappedStart = mapPosition(changes, island.start, { assoc: 1 }).pos;
|
|
1355
1589
|
const candidates = newIslandsByStart.get(mappedStart);
|
|
1356
1590
|
if (!candidates || !candidates.includes(island.raw)) {
|
|
@@ -1368,10 +1602,11 @@ function spliceSave(original, edits, options = {}) {
|
|
|
1368
1602
|
bytesOutsideEditsIdentical,
|
|
1369
1603
|
disturbed
|
|
1370
1604
|
};
|
|
1371
|
-
if (options.requireIntegrity && !integrity.ok) {
|
|
1605
|
+
if ((options.requireIntegrity ?? true) && !integrity.ok) {
|
|
1606
|
+
const named = disturbed.slice(0, 5).map((island) => `${island.islandType}@${island.start}`).join(", ");
|
|
1372
1607
|
throw new SourceSpliceError(
|
|
1373
1608
|
"integrity",
|
|
1374
|
-
`the
|
|
1609
|
+
`integrity: the save disturbed ${disturbed.length} protected island(s) it did not replace${named ? ` (${named}${disturbed.length > 5 ? ", \u2026" : ""})` : ""}${bytesOutsideEditsIdentical ? "" : "; bytes outside the edits changed"}`
|
|
1375
1610
|
);
|
|
1376
1611
|
}
|
|
1377
1612
|
return { text, changed: true, changes, integrity, blocks: newBlocks };
|
|
@@ -1413,6 +1648,6 @@ function mapRange(changes, start, end, unit = "utf16") {
|
|
|
1413
1648
|
return { start: newStart, end: newEnd, touched, collapsed: newEnd === newStart && end > start };
|
|
1414
1649
|
}
|
|
1415
1650
|
|
|
1416
|
-
export { SourceSpliceError, XmlContainerTracker, blockAt, blockEdit, blockRangeEdit, buildCodePointIndex, isSingleDollarMath, joinSource, listIslands, mapPosition, mapRange, readXmlTag, singleDollarMathEnd, spliceSave, splitsSurrogatePair, toCodePointOffset, toUtf16Offset, tokenizeSource };
|
|
1651
|
+
export { NESTING_FENCE_LANGUAGES, SourceSpliceError, XmlContainerTracker, blockAt, blockEdit, blockRangeEdit, buildCodePointIndex, classifyInnerFenceLine, fenceNestsInnerFences, isSingleDollarMath, islandEdit, joinSource, listIslands, looksLikeDisplayMath, mapPosition, mapRange, pairDisplayMath, readXmlTag, singleDollarMathEnd, spliceSave, splitsSurrogatePair, toCodePointOffset, toUtf16Offset, tokenizeSource };
|
|
1417
1652
|
//# sourceMappingURL=source.js.map
|
|
1418
1653
|
//# sourceMappingURL=source.js.map
|