@ai-matrx/content-ir 0.12.0 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +119 -0
- package/dist/index.cjs +546 -98
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.js +539 -99
- package/dist/index.js.map +1 -1
- package/dist/source.cjs +546 -98
- package/dist/source.cjs.map +1 -1
- package/dist/source.d.cts +118 -14
- package/dist/source.d.ts +118 -14
- package/dist/source.js +539 -99
- package/dist/source.js.map +1 -1
- package/package.json +1 -1
package/dist/index.cjs
CHANGED
|
@@ -5175,20 +5175,84 @@ function normalizeFenceLanguage(language) {
|
|
|
5175
5175
|
return CODE_LANGUAGE_ALIASES[lower] ?? lower;
|
|
5176
5176
|
}
|
|
5177
5177
|
|
|
5178
|
+
// source/fence-nesting.ts
|
|
5179
|
+
var NESTING_FENCE_LANGUAGES = /* @__PURE__ */ new Set([
|
|
5180
|
+
"markdown",
|
|
5181
|
+
"md",
|
|
5182
|
+
"mdx"
|
|
5183
|
+
]);
|
|
5184
|
+
function fenceNestsInnerFences(language) {
|
|
5185
|
+
return !!language && NESTING_FENCE_LANGUAGES.has(language.toLowerCase());
|
|
5186
|
+
}
|
|
5187
|
+
var FENCE_WHITESPACE = " \n\r\f\v";
|
|
5188
|
+
function isFenceWhitespace(ch) {
|
|
5189
|
+
return ch !== void 0 && ch !== "" && FENCE_WHITESPACE.includes(ch);
|
|
5190
|
+
}
|
|
5191
|
+
function trimFenceLine(line) {
|
|
5192
|
+
let start = 0;
|
|
5193
|
+
let end = line.length;
|
|
5194
|
+
while (start < end && isFenceWhitespace(line[start])) start++;
|
|
5195
|
+
while (end > start && isFenceWhitespace(line[end - 1])) end--;
|
|
5196
|
+
return start === 0 && end === line.length ? line : line.slice(start, end);
|
|
5197
|
+
}
|
|
5198
|
+
function classifyInnerFenceLine(trimmed, openTicks, nests, nestedDepth) {
|
|
5199
|
+
let ticks = 0;
|
|
5200
|
+
while (ticks < trimmed.length && trimmed[ticks] === "`") ticks++;
|
|
5201
|
+
if (ticks < 3) return "content";
|
|
5202
|
+
const info = trimFenceLine(trimmed.slice(ticks));
|
|
5203
|
+
if (info === "") {
|
|
5204
|
+
if (ticks < openTicks) return "content";
|
|
5205
|
+
if (nests && nestedDepth > 0) return "close-nested";
|
|
5206
|
+
return "close-outer";
|
|
5207
|
+
}
|
|
5208
|
+
if (nests && ticks >= openTicks && !info.includes("`")) return "open-nested";
|
|
5209
|
+
return "content";
|
|
5210
|
+
}
|
|
5211
|
+
|
|
5178
5212
|
// source/math.ts
|
|
5179
5213
|
var TEX_SIGNAL = /\\[A-Za-z]+|[\\^_{}=<>+]/;
|
|
5180
5214
|
var LONE_VARIABLE = /^[A-Za-z](?:'|[0-9])?$/;
|
|
5215
|
+
var MATH_SHAPE_CHARS = /^[A-Za-z0-9'\s(),.*\/-]+$/;
|
|
5216
|
+
var FUNCTION_NAMES = /* @__PURE__ */ new Set([
|
|
5217
|
+
"sin",
|
|
5218
|
+
"cos",
|
|
5219
|
+
"tan",
|
|
5220
|
+
"sec",
|
|
5221
|
+
"csc",
|
|
5222
|
+
"cot",
|
|
5223
|
+
"log",
|
|
5224
|
+
"ln",
|
|
5225
|
+
"exp",
|
|
5226
|
+
"lim",
|
|
5227
|
+
"max",
|
|
5228
|
+
"min",
|
|
5229
|
+
"det",
|
|
5230
|
+
"gcd",
|
|
5231
|
+
"mod",
|
|
5232
|
+
"arg",
|
|
5233
|
+
"sinh",
|
|
5234
|
+
"cosh",
|
|
5235
|
+
"tanh"
|
|
5236
|
+
]);
|
|
5237
|
+
function isMathShape(content) {
|
|
5238
|
+
if (!MATH_SHAPE_CHARS.test(content)) return false;
|
|
5239
|
+
const words = content.match(/[A-Za-z]+/g);
|
|
5240
|
+
if (!words) return false;
|
|
5241
|
+
if (!words.every((w) => w.length === 1 || FUNCTION_NAMES.has(w))) return false;
|
|
5242
|
+
return /[A-Za-z]'*\(|[A-Za-z]'|\b[A-Za-z]\s+[A-Za-z]\b|\([^)]*[A-Za-z][^)]*\)/.test(content);
|
|
5243
|
+
}
|
|
5181
5244
|
function isSingleDollarMath(content, after) {
|
|
5182
5245
|
if (!content || content.length > 400) return false;
|
|
5183
5246
|
if (/^\s/.test(content) || /\s$/.test(content)) return false;
|
|
5184
5247
|
if (after !== void 0 && /[0-9]/.test(after)) return false;
|
|
5185
5248
|
if (LONE_VARIABLE.test(content)) return true;
|
|
5186
5249
|
if (/^[0-9]/.test(content)) return /\\[A-Za-z]+|[\\^_{}]/.test(content);
|
|
5187
|
-
return TEX_SIGNAL.test(content);
|
|
5250
|
+
return TEX_SIGNAL.test(content) || isMathShape(content);
|
|
5188
5251
|
}
|
|
5189
5252
|
function singleDollarMathEnd(text, open, limit = text.length) {
|
|
5253
|
+
const bound = Math.min(limit, open + 403);
|
|
5190
5254
|
let close = -1;
|
|
5191
|
-
for (let k = open + 1; k <
|
|
5255
|
+
for (let k = open + 1; k < bound; k += 1) {
|
|
5192
5256
|
const c = text[k];
|
|
5193
5257
|
if (c === "\n") break;
|
|
5194
5258
|
if (c === "\\") {
|
|
@@ -5205,11 +5269,69 @@ function singleDollarMathEnd(text, open, limit = text.length) {
|
|
|
5205
5269
|
return isSingleDollarMath(content, text[close + 1]) ? close + 1 : -1;
|
|
5206
5270
|
}
|
|
5207
5271
|
|
|
5272
|
+
// source/math-pairs.ts
|
|
5273
|
+
var MAX_MATH_SPAN = 600;
|
|
5274
|
+
var STRUCTURAL_MARKDOWN = /\]\(|https?:\/\/|\*\*|(?:^|\n)[ \t]{0,3}#{1,6}[ \t]|(?:^|\n)[ \t]*[-*+][ \t]+|(?:^|\n)[ \t]*\d+[.)][ \t]/;
|
|
5275
|
+
var LATEX_COMMAND = /\\[a-zA-Z]/;
|
|
5276
|
+
var PROSE_WORD = /[A-Za-z]{3,}/g;
|
|
5277
|
+
var PROSE_WORD_LIMIT = 6;
|
|
5278
|
+
function looksLikeDisplayMath(inner) {
|
|
5279
|
+
const s = inner.trim();
|
|
5280
|
+
if (!s) return false;
|
|
5281
|
+
if (STRUCTURAL_MARKDOWN.test(s)) return false;
|
|
5282
|
+
if (s.length > MAX_MATH_SPAN) return false;
|
|
5283
|
+
if (LATEX_COMMAND.test(s)) {
|
|
5284
|
+
return (s.match(PROSE_WORD) ?? []).length < PROSE_WORD_LIMIT * 3;
|
|
5285
|
+
}
|
|
5286
|
+
if (/\n[ \t]*\n/.test(s)) return false;
|
|
5287
|
+
return (s.match(PROSE_WORD) ?? []).length < PROSE_WORD_LIMIT;
|
|
5288
|
+
}
|
|
5289
|
+
function codeRanges(text) {
|
|
5290
|
+
const ranges = [];
|
|
5291
|
+
for (const re of [/```[\s\S]*?(?:```|$)/g, /~~~[\s\S]*?(?:~~~|$)/g, /`[^`\n]*`/g]) {
|
|
5292
|
+
let m;
|
|
5293
|
+
while ((m = re.exec(text)) !== null) {
|
|
5294
|
+
ranges.push([m.index, m.index + m[0].length]);
|
|
5295
|
+
if (m[0].length === 0) re.lastIndex++;
|
|
5296
|
+
}
|
|
5297
|
+
}
|
|
5298
|
+
return ranges.sort((a, b) => a[0] - b[0]);
|
|
5299
|
+
}
|
|
5300
|
+
function pairDisplayMath(text) {
|
|
5301
|
+
const pairs = /* @__PURE__ */ new Map();
|
|
5302
|
+
if (!text.includes("$$")) return pairs;
|
|
5303
|
+
const ranges = codeRanges(text);
|
|
5304
|
+
const tokens = [];
|
|
5305
|
+
let r = 0;
|
|
5306
|
+
let maxEnd = -1;
|
|
5307
|
+
for (let i = text.indexOf("$$"); i !== -1 && i < text.length - 1; i = text.indexOf("$$", i)) {
|
|
5308
|
+
while (r < ranges.length && ranges[r][0] <= i) {
|
|
5309
|
+
maxEnd = Math.max(maxEnd, ranges[r][1]);
|
|
5310
|
+
r++;
|
|
5311
|
+
}
|
|
5312
|
+
if (maxEnd <= i) tokens.push(i);
|
|
5313
|
+
i += 2;
|
|
5314
|
+
}
|
|
5315
|
+
let j = 0;
|
|
5316
|
+
while (j < tokens.length) {
|
|
5317
|
+
const open = tokens[j];
|
|
5318
|
+
const close = tokens[j + 1];
|
|
5319
|
+
if (close === void 0) break;
|
|
5320
|
+
if (looksLikeDisplayMath(text.slice(open + 2, close))) {
|
|
5321
|
+
pairs.set(open, close);
|
|
5322
|
+
j += 2;
|
|
5323
|
+
continue;
|
|
5324
|
+
}
|
|
5325
|
+
j += 1;
|
|
5326
|
+
}
|
|
5327
|
+
return pairs;
|
|
5328
|
+
}
|
|
5329
|
+
|
|
5208
5330
|
// source/xml-tag.ts
|
|
5209
5331
|
var XML_NAME_START = /[A-Za-z_]/;
|
|
5210
5332
|
var XML_NAME_CHARACTER = /[\w.:-]/;
|
|
5211
5333
|
var WHITESPACE = /\s/;
|
|
5212
|
-
function readXmlTag(content, start) {
|
|
5334
|
+
function readXmlTag(content, start, find = (needle, from) => content.indexOf(needle, from)) {
|
|
5213
5335
|
let cursor = start + 1;
|
|
5214
5336
|
const isClosing = content[cursor] === "/";
|
|
5215
5337
|
if (isClosing) cursor++;
|
|
@@ -5262,8 +5384,9 @@ function readXmlTag(content, start) {
|
|
|
5262
5384
|
const quote = content[cursor];
|
|
5263
5385
|
if (quote !== '"' && quote !== "'") return null;
|
|
5264
5386
|
const valueStart = ++cursor;
|
|
5265
|
-
|
|
5266
|
-
if (
|
|
5387
|
+
const valueEnd = find(quote, valueStart);
|
|
5388
|
+
if (valueEnd === -1) return null;
|
|
5389
|
+
cursor = valueEnd;
|
|
5267
5390
|
attributes.push({ name, value: content.slice(valueStart, cursor) });
|
|
5268
5391
|
cursor++;
|
|
5269
5392
|
}
|
|
@@ -5360,6 +5483,10 @@ var XmlContainerTracker = class {
|
|
|
5360
5483
|
|
|
5361
5484
|
// source/tokenize.ts
|
|
5362
5485
|
var NOT_WHITESPACE = /\S/;
|
|
5486
|
+
var DIRECTIVE_OPEN = /^(:{2,})([A-Za-z][\w-]*)/;
|
|
5487
|
+
var MKDOCS_OPEN = /^(!!!|\?\?\?\+?)[ \t]+([A-Za-z][\w-]*)/;
|
|
5488
|
+
var CALLOUT_OPEN = /^>[ \t]*\[!([A-Za-z][\w-]*)\]([+-])?/;
|
|
5489
|
+
var FOOTNOTE_DEF = /^\[\^([^\]\s]+)\]:/;
|
|
5363
5490
|
function countStructuralBraces(source) {
|
|
5364
5491
|
let opens = 0;
|
|
5365
5492
|
let closes = 0;
|
|
@@ -5482,11 +5609,16 @@ function backtickRunLength(str2, pos) {
|
|
|
5482
5609
|
while (pos + n < str2.length && str2[pos + n] === "`") n++;
|
|
5483
5610
|
return n;
|
|
5484
5611
|
}
|
|
5612
|
+
function isSplitterTreeLine(line) {
|
|
5613
|
+
return TREE_GLYPHS.test(line) || /^[\s│|]*[├└+|][\s─\-]+/.test(line);
|
|
5614
|
+
}
|
|
5485
5615
|
var MEDIA_REF_PREFIXES = ["[Image URL:", "[Video URL:", "[Audio URL:"];
|
|
5486
5616
|
var Tokenizer = class {
|
|
5487
|
-
constructor(text) {
|
|
5617
|
+
constructor(text, streaming = false) {
|
|
5488
5618
|
this.text = text;
|
|
5619
|
+
this.streaming = streaming;
|
|
5489
5620
|
this.n = text.length;
|
|
5621
|
+
this.mathPairs = pairDisplayMath(text);
|
|
5490
5622
|
let start = 0;
|
|
5491
5623
|
for (let i = 0; i < text.length; i++) {
|
|
5492
5624
|
if (text.charCodeAt(i) === 10) {
|
|
@@ -5503,6 +5635,7 @@ var Tokenizer = class {
|
|
|
5503
5635
|
}
|
|
5504
5636
|
}
|
|
5505
5637
|
text;
|
|
5638
|
+
streaming;
|
|
5506
5639
|
n;
|
|
5507
5640
|
lineStarts = [];
|
|
5508
5641
|
contentEnds = [];
|
|
@@ -5512,6 +5645,38 @@ var Tokenizer = class {
|
|
|
5512
5645
|
/** Lazily built: for line i, first line j >= i where brace net from i drops to <= 0. */
|
|
5513
5646
|
braceStop = null;
|
|
5514
5647
|
braceNet = null;
|
|
5648
|
+
/** Every occurrence offset of each needle searched so far (overlapping, like indexOf). */
|
|
5649
|
+
occurrences = /* @__PURE__ */ new Map();
|
|
5650
|
+
/** `$$` opener → closer for the pairs the core renders as math. */
|
|
5651
|
+
mathPairs;
|
|
5652
|
+
finder = (needle, from) => this.next(needle, from);
|
|
5653
|
+
/**
|
|
5654
|
+
* `text.indexOf(needle, from)` in O(log k): the first query for a needle
|
|
5655
|
+
* indexes all of its occurrences once, so repeated searches for a closer
|
|
5656
|
+
* that never comes stay linear over the whole text.
|
|
5657
|
+
*/
|
|
5658
|
+
next(needle, from) {
|
|
5659
|
+
let list = this.occurrences.get(needle);
|
|
5660
|
+
if (!list) {
|
|
5661
|
+
list = [];
|
|
5662
|
+
for (let at = this.text.indexOf(needle); at !== -1; at = this.text.indexOf(needle, at + 1)) {
|
|
5663
|
+
list.push(at);
|
|
5664
|
+
}
|
|
5665
|
+
this.occurrences.set(needle, list);
|
|
5666
|
+
}
|
|
5667
|
+
let lo = 0;
|
|
5668
|
+
let hi = list.length;
|
|
5669
|
+
while (lo < hi) {
|
|
5670
|
+
const mid = lo + hi >>> 1;
|
|
5671
|
+
if (list[mid] < from) lo = mid + 1;
|
|
5672
|
+
else hi = mid;
|
|
5673
|
+
}
|
|
5674
|
+
return lo < list.length ? list[lo] : -1;
|
|
5675
|
+
}
|
|
5676
|
+
/** Streaming only: the final line, still waiting for its terminator. */
|
|
5677
|
+
isOpenTail(k) {
|
|
5678
|
+
return this.streaming && k === this.lineStarts.length - 1 && this.lineEnds[k] === this.contentEnds[k];
|
|
5679
|
+
}
|
|
5515
5680
|
get lineCount() {
|
|
5516
5681
|
return this.lineStarts.length;
|
|
5517
5682
|
}
|
|
@@ -5593,6 +5758,22 @@ var Tokenizer = class {
|
|
|
5593
5758
|
const fm = this.frontMatter(li, t);
|
|
5594
5759
|
if (fm) return fm;
|
|
5595
5760
|
}
|
|
5761
|
+
if (indent <= 3 && t.startsWith("::")) {
|
|
5762
|
+
const directive = this.directive(p, li, t);
|
|
5763
|
+
if (directive) return directive;
|
|
5764
|
+
}
|
|
5765
|
+
if (indent === 0 && (t.startsWith("!!!") || t.startsWith("???"))) {
|
|
5766
|
+
const admonition = this.mkdocsAdmonition(p, li, t);
|
|
5767
|
+
if (admonition) return admonition;
|
|
5768
|
+
}
|
|
5769
|
+
if (indent <= 3 && t.startsWith(">")) {
|
|
5770
|
+
const callout = this.callout(p, li, t);
|
|
5771
|
+
if (callout) return callout;
|
|
5772
|
+
}
|
|
5773
|
+
if (atParaStart && indent <= 3 && t.startsWith("[^")) {
|
|
5774
|
+
const footnote = this.footnoteDefinition(p, li, t);
|
|
5775
|
+
if (footnote) return footnote;
|
|
5776
|
+
}
|
|
5596
5777
|
if (t.startsWith("```")) return this.backtickFence(p, li, t);
|
|
5597
5778
|
if (indent <= 3 && t.startsWith("~~~")) {
|
|
5598
5779
|
const tilde = this.tildeFence(p, li, t);
|
|
@@ -5630,20 +5811,116 @@ var Tokenizer = class {
|
|
|
5630
5811
|
return { type, start, end, complete, meta };
|
|
5631
5812
|
}
|
|
5632
5813
|
frontMatter(li, t) {
|
|
5633
|
-
|
|
5814
|
+
const fence = this.lineText(li);
|
|
5815
|
+
if (fence !== "---" && fence !== "+++" || t !== fence) return null;
|
|
5816
|
+
const toml = fence === "+++";
|
|
5634
5817
|
let sawContent = false;
|
|
5635
5818
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
5636
5819
|
const line = this.lineText(k);
|
|
5637
|
-
if (line ===
|
|
5638
|
-
return sawContent ? this.island("front_matter", 0, this.contentEnds[k], true) : null;
|
|
5820
|
+
if (line === fence || !toml && line === "...") {
|
|
5821
|
+
return sawContent ? this.island("front_matter", 0, this.contentEnds[k], true, { format: toml ? "toml" : "yaml" }) : null;
|
|
5639
5822
|
}
|
|
5640
5823
|
if (!NOT_WHITESPACE.test(line)) continue;
|
|
5641
|
-
|
|
5824
|
+
const shaped = toml ? /^\s*(?:[A-Za-z0-9_."-]+\s*=|\[|#)/.test(line) : /^(?:[A-Za-z0-9_"'-][^:]*:(?:\s|$)|\s+\S|-\s|#)/.test(line);
|
|
5825
|
+
if (!shaped) return null;
|
|
5642
5826
|
sawContent = true;
|
|
5643
5827
|
}
|
|
5644
5828
|
return null;
|
|
5645
5829
|
}
|
|
5646
|
-
/**
|
|
5830
|
+
/**
|
|
5831
|
+
* `:::name[label]{attrs}` … closing `:::` (a colon run at least as long as
|
|
5832
|
+
* the opener's, nesting counted), or a leaf `::name[…]{…}` line. The whole
|
|
5833
|
+
* container is ONE island: its grammar (fence lengths, labels, attributes,
|
|
5834
|
+
* nested tabs) is exactly what a visual editor would re-serialize wrongly.
|
|
5835
|
+
*/
|
|
5836
|
+
directive(p, li, t) {
|
|
5837
|
+
const open = DIRECTIVE_OPEN.exec(t);
|
|
5838
|
+
if (!open) return null;
|
|
5839
|
+
const colons = open[1].length;
|
|
5840
|
+
const name = open[2];
|
|
5841
|
+
if (colons === 2) {
|
|
5842
|
+
return this.island("directive", p, this.contentEnds[li], true, { name, leaf: true });
|
|
5843
|
+
}
|
|
5844
|
+
const stack = [colons];
|
|
5845
|
+
for (let k = li + 1; k < this.lineCount; k++) {
|
|
5846
|
+
const line = this.lineText(k).trim();
|
|
5847
|
+
const inner = DIRECTIVE_OPEN.exec(line);
|
|
5848
|
+
if (inner && inner[1].length >= 3) {
|
|
5849
|
+
stack.push(inner[1].length);
|
|
5850
|
+
continue;
|
|
5851
|
+
}
|
|
5852
|
+
const close = /^(:{3,})\s*$/.exec(line);
|
|
5853
|
+
if (close && close[1].length >= stack[stack.length - 1]) {
|
|
5854
|
+
stack.pop();
|
|
5855
|
+
if (stack.length === 0) return this.island("directive", p, this.contentEnds[k], true, { name });
|
|
5856
|
+
}
|
|
5857
|
+
}
|
|
5858
|
+
return this.island("directive", p, this.n, false, { name });
|
|
5859
|
+
}
|
|
5860
|
+
/** MkDocs `!!! type "Title"` / `??? type` / `???+ type` and its indented (or blank-separated indented) body. */
|
|
5861
|
+
mkdocsAdmonition(p, li, t) {
|
|
5862
|
+
const open = MKDOCS_OPEN.exec(t);
|
|
5863
|
+
if (!open) return null;
|
|
5864
|
+
let last = li;
|
|
5865
|
+
for (let k = li + 1; k < this.lineCount; k++) {
|
|
5866
|
+
const line = this.lineText(k);
|
|
5867
|
+
if (/^(?: {4}|\t)/.test(line) && NOT_WHITESPACE.test(line)) {
|
|
5868
|
+
last = k;
|
|
5869
|
+
continue;
|
|
5870
|
+
}
|
|
5871
|
+
if (!NOT_WHITESPACE.test(line)) continue;
|
|
5872
|
+
break;
|
|
5873
|
+
}
|
|
5874
|
+
return this.island("directive", p, this.contentEnds[last], true, {
|
|
5875
|
+
name: open[2].toLowerCase(),
|
|
5876
|
+
syntax: "mkdocs"
|
|
5877
|
+
});
|
|
5878
|
+
}
|
|
5879
|
+
/** A blockquote whose first line is `> [!TYPE]…` — every following `>` line belongs to it. */
|
|
5880
|
+
callout(p, li, t) {
|
|
5881
|
+
const open = CALLOUT_OPEN.exec(t);
|
|
5882
|
+
if (!open) return null;
|
|
5883
|
+
let last = li;
|
|
5884
|
+
for (let k = li + 1; k < this.lineCount; k++) {
|
|
5885
|
+
if (!/^ {0,3}>/.test(this.lineText(k))) break;
|
|
5886
|
+
last = k;
|
|
5887
|
+
}
|
|
5888
|
+
return this.island("callout", p, this.contentEnds[last], true, {
|
|
5889
|
+
name: open[1].toLowerCase(),
|
|
5890
|
+
...open[2] ? { fold: open[2] === "-" ? "closed" : "open" } : {}
|
|
5891
|
+
});
|
|
5892
|
+
}
|
|
5893
|
+
/** `[^id]: text` plus its continuation lines (indented, or lazy text before a blank line). */
|
|
5894
|
+
footnoteDefinition(p, li, t) {
|
|
5895
|
+
const open = FOOTNOTE_DEF.exec(t);
|
|
5896
|
+
if (!open) return null;
|
|
5897
|
+
let last = li;
|
|
5898
|
+
for (let k = li + 1; k < this.lineCount; k++) {
|
|
5899
|
+
const line = this.lineText(k);
|
|
5900
|
+
if (!NOT_WHITESPACE.test(line)) {
|
|
5901
|
+
let j = k + 1;
|
|
5902
|
+
while (j < this.lineCount && !NOT_WHITESPACE.test(this.lineText(j))) j++;
|
|
5903
|
+
if (j < this.lineCount && /^(?: {4}|\t)/.test(this.lineText(j))) {
|
|
5904
|
+
k = j - 1;
|
|
5905
|
+
continue;
|
|
5906
|
+
}
|
|
5907
|
+
break;
|
|
5908
|
+
}
|
|
5909
|
+
if (/^(?: {4}|\t)/.test(line)) {
|
|
5910
|
+
last = k;
|
|
5911
|
+
continue;
|
|
5912
|
+
}
|
|
5913
|
+
if (FOOTNOTE_DEF.test(line.trimStart()) || this.detectBlock(this.lineStarts[k], k, false)) break;
|
|
5914
|
+
last = k;
|
|
5915
|
+
}
|
|
5916
|
+
return this.island("footnote_def", p, this.contentEnds[last], true, { id: open[1] });
|
|
5917
|
+
}
|
|
5918
|
+
/**
|
|
5919
|
+
* Backtick fence — the renderer splitter's close rules: JSON-string aware for
|
|
5920
|
+
* ```json, and THE nested-fence rule (source/fence-nesting.ts) for markdown
|
|
5921
|
+
* fences, with the splitter's strict-CommonMark retry when a nested fence is
|
|
5922
|
+
* still open at the end of the text.
|
|
5923
|
+
*/
|
|
5647
5924
|
backtickFence(p, li, t) {
|
|
5648
5925
|
const openTicks = backtickRunLength(t, 0);
|
|
5649
5926
|
const info = t.slice(openTicks).trim();
|
|
@@ -5652,10 +5929,23 @@ var Tokenizer = class {
|
|
|
5652
5929
|
const normalized = normalizeFenceLanguage(lang);
|
|
5653
5930
|
const meta = { fence: "`", ticks: openTicks, lang };
|
|
5654
5931
|
if (normalized && SPECIAL_CODE_LANGUAGES.has(normalized)) meta.special = normalized;
|
|
5932
|
+
const nesting = fenceNestsInnerFences(lang);
|
|
5933
|
+
let close = this.backtickFenceClose(li, openTicks, isJson, nesting);
|
|
5934
|
+
if (close === "retry-strict") close = this.backtickFenceClose(li, openTicks, isJson, false);
|
|
5935
|
+
if (typeof close === "number") {
|
|
5936
|
+
return this.fenceIsland(p, this.contentEnds[close], true, meta, li, close, isJson);
|
|
5937
|
+
}
|
|
5938
|
+
return this.fenceIsland(p, this.n, false, meta, li, this.lineCount, isJson);
|
|
5939
|
+
}
|
|
5940
|
+
/** The closing line of a backtick fence opened on line `li`, or null when it never closes. */
|
|
5941
|
+
backtickFenceClose(li, openTicks, isJson, nesting) {
|
|
5655
5942
|
let state = { inString: false, escaped: false };
|
|
5943
|
+
let nestedDepth = 0;
|
|
5944
|
+
let sawNested = false;
|
|
5656
5945
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
5946
|
+
if (this.isOpenTail(k)) continue;
|
|
5657
5947
|
const line = this.lineText(k);
|
|
5658
|
-
const trimmedLine = line
|
|
5948
|
+
const trimmedLine = trimFenceLine(line);
|
|
5659
5949
|
if (trimmedLine.startsWith("```")) {
|
|
5660
5950
|
const at2 = line.indexOf("```");
|
|
5661
5951
|
if (isJson) {
|
|
@@ -5665,11 +5955,13 @@ var Tokenizer = class {
|
|
|
5665
5955
|
continue;
|
|
5666
5956
|
}
|
|
5667
5957
|
}
|
|
5668
|
-
const
|
|
5669
|
-
|
|
5670
|
-
if (
|
|
5671
|
-
|
|
5958
|
+
const kind = classifyInnerFenceLine(trimmedLine, openTicks, nesting, nestedDepth);
|
|
5959
|
+
if (kind === "close-outer") return k;
|
|
5960
|
+
if (kind === "open-nested") {
|
|
5961
|
+
nestedDepth++;
|
|
5962
|
+
sawNested = true;
|
|
5672
5963
|
}
|
|
5964
|
+
if (kind === "close-nested") nestedDepth--;
|
|
5673
5965
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
5674
5966
|
continue;
|
|
5675
5967
|
}
|
|
@@ -5683,16 +5975,14 @@ var Tokenizer = class {
|
|
|
5683
5975
|
}
|
|
5684
5976
|
}
|
|
5685
5977
|
const closeTicks = backtickRunLength(line, at);
|
|
5686
|
-
const after = line.slice(at + closeTicks)
|
|
5687
|
-
if (closeTicks >= openTicks && after === "")
|
|
5688
|
-
return this.fenceIsland(p, this.contentEnds[k], true, meta, li, k, isJson);
|
|
5689
|
-
}
|
|
5978
|
+
const after = trimFenceLine(line.slice(at + closeTicks));
|
|
5979
|
+
if (closeTicks >= openTicks && after === "" && nestedDepth === 0) return k;
|
|
5690
5980
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
5691
5981
|
continue;
|
|
5692
5982
|
}
|
|
5693
5983
|
if (isJson) state = jsonStringState(line, 0, line.length, state);
|
|
5694
5984
|
}
|
|
5695
|
-
return
|
|
5985
|
+
return sawNested && nestedDepth > 0 ? "retry-strict" : null;
|
|
5696
5986
|
}
|
|
5697
5987
|
fenceIsland(p, end, complete, meta, openLine, closeLine, isJson) {
|
|
5698
5988
|
if (isJson && openLine + 1 < this.lineCount) {
|
|
@@ -5712,24 +6002,29 @@ var Tokenizer = class {
|
|
|
5712
6002
|
if (normalized && SPECIAL_CODE_LANGUAGES.has(normalized)) meta.special = normalized;
|
|
5713
6003
|
const close = new RegExp(`^ {0,3}~{${ticks},}[ \\t]*$`);
|
|
5714
6004
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
6005
|
+
if (this.isOpenTail(k)) continue;
|
|
5715
6006
|
if (close.test(this.lineText(k))) {
|
|
5716
6007
|
return this.island("fence", p, this.contentEnds[k], true, meta);
|
|
5717
6008
|
}
|
|
5718
6009
|
}
|
|
5719
6010
|
return this.island("fence", p, this.n, false, meta);
|
|
5720
6011
|
}
|
|
6012
|
+
/**
|
|
6013
|
+
* A line-leading `$$` that OPENS a math pair (source/math-pairs.ts — the
|
|
6014
|
+
* core's rule). A closer, an unpaired `$$`, or a prose pair is never a block:
|
|
6015
|
+
* a lone `$$` renders as literal text and locks nothing.
|
|
6016
|
+
*/
|
|
5721
6017
|
mathBlock(p, tp) {
|
|
6018
|
+
const close = this.mathPairs.get(tp);
|
|
6019
|
+
if (close === void 0) return null;
|
|
6020
|
+
const end = close + 2;
|
|
5722
6021
|
const li = this.lineOf(tp);
|
|
5723
6022
|
const ce = this.contentEnds[li];
|
|
5724
|
-
|
|
5725
|
-
|
|
5726
|
-
|
|
5727
|
-
|
|
5728
|
-
|
|
5729
|
-
}
|
|
5730
|
-
const close = this.text.indexOf("$$", tp + 2);
|
|
5731
|
-
if (close === -1) return this.island("math_block", p, this.n, false, { delimiter: "$$" });
|
|
5732
|
-
return this.island("math_block", p, close + 2, true, { delimiter: "$$" });
|
|
6023
|
+
if (end <= ce && !this.isBlank(end, ce)) return null;
|
|
6024
|
+
const endLine = this.lineOf(close);
|
|
6025
|
+
const endCe = this.contentEnds[endLine];
|
|
6026
|
+
if (endLine !== li && !this.isBlank(end, endCe)) return null;
|
|
6027
|
+
return this.island("math_block", p, end, true, { delimiter: "$$" });
|
|
5733
6028
|
}
|
|
5734
6029
|
/** `<artifact …>`, `<decision …>`, … at the start of a line. */
|
|
5735
6030
|
attributeXml(p, tp, t) {
|
|
@@ -5746,7 +6041,7 @@ var Tokenizer = class {
|
|
|
5746
6041
|
attributeElement(name, p, tp, type) {
|
|
5747
6042
|
const li = this.lineOf(tp);
|
|
5748
6043
|
const ce = this.contentEnds[li];
|
|
5749
|
-
const tag = readXmlTag(this.text, tp);
|
|
6044
|
+
const tag = readXmlTag(this.text, tp, this.finder);
|
|
5750
6045
|
let openEnd;
|
|
5751
6046
|
if (tag && !tag.isClosing) {
|
|
5752
6047
|
if (tag.isSelfClosing) {
|
|
@@ -5754,13 +6049,13 @@ var Tokenizer = class {
|
|
|
5754
6049
|
}
|
|
5755
6050
|
openEnd = tp + tag.raw.length;
|
|
5756
6051
|
} else {
|
|
5757
|
-
const gt = this.
|
|
6052
|
+
const gt = this.next(">", tp);
|
|
5758
6053
|
if (gt === -1) return this.island(type, p, this.n, false, { tag: name });
|
|
5759
6054
|
if (gt >= ce) return null;
|
|
5760
6055
|
openEnd = gt + 1;
|
|
5761
6056
|
}
|
|
5762
6057
|
const closer = `</${name}>`;
|
|
5763
|
-
const close = this.
|
|
6058
|
+
const close = this.next(closer, openEnd);
|
|
5764
6059
|
if (close === -1) return this.island(type, p, this.n, false, { tag: name });
|
|
5765
6060
|
return this.island(type, p, close + closer.length, true, { tag: name });
|
|
5766
6061
|
}
|
|
@@ -5770,7 +6065,7 @@ var Tokenizer = class {
|
|
|
5770
6065
|
const opener = `<${name}>`;
|
|
5771
6066
|
if (!t.startsWith(opener)) continue;
|
|
5772
6067
|
const closer = `</${name}>`;
|
|
5773
|
-
const close = this.
|
|
6068
|
+
const close = this.next(closer, tp + opener.length);
|
|
5774
6069
|
if (close === -1) return this.island("xml_region", p, this.n, false, { tag: name });
|
|
5775
6070
|
return this.island("xml_region", p, close + closer.length, true, { tag: name });
|
|
5776
6071
|
}
|
|
@@ -5806,7 +6101,7 @@ var Tokenizer = class {
|
|
|
5806
6101
|
return null;
|
|
5807
6102
|
}
|
|
5808
6103
|
blockComment(p, tp, li) {
|
|
5809
|
-
const close = this.
|
|
6104
|
+
const close = this.next("-->", tp + 4);
|
|
5810
6105
|
if (close === -1) return this.island("html_comment", p, this.n, false);
|
|
5811
6106
|
const end = close + 3;
|
|
5812
6107
|
const raw = this.text.slice(tp, end);
|
|
@@ -5884,9 +6179,8 @@ var Tokenizer = class {
|
|
|
5884
6179
|
}
|
|
5885
6180
|
}
|
|
5886
6181
|
if (lastLine === null) {
|
|
5887
|
-
|
|
5888
|
-
|
|
5889
|
-
return this.island("json", p, this.n, false, jsonMeta(partial));
|
|
6182
|
+
if (isRecognizedJsonHead(this.text.slice(tp, tp + 512))) {
|
|
6183
|
+
return this.island("json", p, this.n, false, jsonMeta(this.text.slice(tp)));
|
|
5890
6184
|
}
|
|
5891
6185
|
return null;
|
|
5892
6186
|
}
|
|
@@ -5921,11 +6215,43 @@ var Tokenizer = class {
|
|
|
5921
6215
|
// ── prose ──────────────────────────────────────────────────────────────
|
|
5922
6216
|
paragraph(pos, li) {
|
|
5923
6217
|
let end = this.contentEnds[li];
|
|
6218
|
+
let breakLine = -1;
|
|
5924
6219
|
for (let k = li + 1; k < this.lineCount; k++) {
|
|
5925
6220
|
if (this.lineIsBlank(k)) break;
|
|
5926
|
-
if (this.detectBlock(this.lineStarts[k], k, false))
|
|
6221
|
+
if (this.detectBlock(this.lineStarts[k], k, false)) {
|
|
6222
|
+
breakLine = k;
|
|
6223
|
+
break;
|
|
6224
|
+
}
|
|
5927
6225
|
end = this.contentEnds[k];
|
|
5928
6226
|
}
|
|
6227
|
+
if (breakLine !== -1) {
|
|
6228
|
+
const breakIsland = this.detectBlock(this.lineStarts[breakLine], breakLine, false);
|
|
6229
|
+
if (breakIsland?.type === "tree") {
|
|
6230
|
+
let root = breakLine;
|
|
6231
|
+
for (let j = breakLine - 1; j >= li; j--) {
|
|
6232
|
+
const trimmed = this.lineText(j).trim();
|
|
6233
|
+
if (!trimmed || isSplitterTreeLine(trimmed) || /^#{1,6}\s/.test(trimmed)) break;
|
|
6234
|
+
root = j;
|
|
6235
|
+
}
|
|
6236
|
+
if (root < breakLine) {
|
|
6237
|
+
const treeStart = root === li ? pos : this.lineStarts[root];
|
|
6238
|
+
const tree = { ...breakIsland, start: treeStart };
|
|
6239
|
+
if (treeStart > pos) {
|
|
6240
|
+
const proseEnd = this.contentEnds[root - 1];
|
|
6241
|
+
const scan2 = this.scanInline(pos, proseEnd);
|
|
6242
|
+
if (!scan2.breakIsland) {
|
|
6243
|
+
this.push({ kind: "prose", start: pos, end: proseEnd, island: null, inlines: scan2.inlines });
|
|
6244
|
+
this.push({ kind: "gap", start: proseEnd, end: treeStart, island: null, inlines: [] });
|
|
6245
|
+
this.push({ kind: "island", start: treeStart, end: tree.end, island: tree, inlines: [] });
|
|
6246
|
+
return tree.end;
|
|
6247
|
+
}
|
|
6248
|
+
} else {
|
|
6249
|
+
this.push({ kind: "island", start: treeStart, end: tree.end, island: tree, inlines: [] });
|
|
6250
|
+
return tree.end;
|
|
6251
|
+
}
|
|
6252
|
+
}
|
|
6253
|
+
}
|
|
6254
|
+
}
|
|
5929
6255
|
const scan = this.scanInline(pos, end);
|
|
5930
6256
|
if (!scan.breakIsland) {
|
|
5931
6257
|
this.push({ kind: "prose", start: pos, end, island: null, inlines: scan.inlines });
|
|
@@ -5958,17 +6284,49 @@ var Tokenizer = class {
|
|
|
5958
6284
|
const text = this.text;
|
|
5959
6285
|
const inlines = [];
|
|
5960
6286
|
const hasKind = text.slice(ps, pe).includes("__kind");
|
|
6287
|
+
let kindBudget = 4 * (pe - ps) + 65536;
|
|
5961
6288
|
const stop = (island, orphan = false) => ({ inlines, breakIsland: island, orphan });
|
|
6289
|
+
const within = (needle, from) => {
|
|
6290
|
+
const at = this.next(needle, from);
|
|
6291
|
+
return at !== -1 && at + needle.length <= pe ? at : -1;
|
|
6292
|
+
};
|
|
6293
|
+
let runs = null;
|
|
6294
|
+
const codeSpanClose = (n, from) => {
|
|
6295
|
+
if (!runs) {
|
|
6296
|
+
runs = /* @__PURE__ */ new Map();
|
|
6297
|
+
for (let k = ps; k < pe; ) {
|
|
6298
|
+
if (text.charCodeAt(k) !== 96) {
|
|
6299
|
+
k++;
|
|
6300
|
+
continue;
|
|
6301
|
+
}
|
|
6302
|
+
const len = backtickRunLength(text, k);
|
|
6303
|
+
const list2 = runs.get(len);
|
|
6304
|
+
if (list2) list2.push(k);
|
|
6305
|
+
else runs.set(len, [k]);
|
|
6306
|
+
k += len;
|
|
6307
|
+
}
|
|
6308
|
+
}
|
|
6309
|
+
const list = runs.get(n);
|
|
6310
|
+
if (!list) return -1;
|
|
6311
|
+
let lo = 0;
|
|
6312
|
+
let hi = list.length;
|
|
6313
|
+
while (lo < hi) {
|
|
6314
|
+
const mid = lo + hi >>> 1;
|
|
6315
|
+
if (list[mid] < from) lo = mid + 1;
|
|
6316
|
+
else hi = mid;
|
|
6317
|
+
}
|
|
6318
|
+
const at = list[lo];
|
|
6319
|
+
return at !== void 0 && at + n <= pe ? at : -1;
|
|
6320
|
+
};
|
|
5962
6321
|
let i = ps;
|
|
5963
6322
|
while (i < pe) {
|
|
5964
6323
|
const ch = text[i];
|
|
5965
6324
|
if (ch === "\\") {
|
|
5966
|
-
const
|
|
5967
|
-
if (
|
|
5968
|
-
const
|
|
5969
|
-
|
|
5970
|
-
|
|
5971
|
-
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: next === "(" ? "\\(" : "\\[" }));
|
|
6325
|
+
const nextCh = text[i + 1];
|
|
6326
|
+
if (nextCh === "(" || nextCh === "[") {
|
|
6327
|
+
const close = within(nextCh === "(" ? "\\)" : "\\]", i + 2);
|
|
6328
|
+
if (close !== -1) {
|
|
6329
|
+
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: nextCh === "(" ? "\\(" : "\\[" }));
|
|
5972
6330
|
i = close + 2;
|
|
5973
6331
|
continue;
|
|
5974
6332
|
}
|
|
@@ -5978,33 +6336,20 @@ var Tokenizer = class {
|
|
|
5978
6336
|
}
|
|
5979
6337
|
if (ch === "`") {
|
|
5980
6338
|
const n = backtickRunLength(text, i);
|
|
5981
|
-
const
|
|
5982
|
-
let k = i + n;
|
|
5983
|
-
let close = -1;
|
|
5984
|
-
while (k < pe) {
|
|
5985
|
-
const idx = text.indexOf(run, k);
|
|
5986
|
-
if (idx === -1 || idx >= pe) break;
|
|
5987
|
-
if (text[idx + n] === "`") {
|
|
5988
|
-
let m = idx;
|
|
5989
|
-
while (text[m] === "`") m++;
|
|
5990
|
-
k = m;
|
|
5991
|
-
continue;
|
|
5992
|
-
}
|
|
5993
|
-
close = idx;
|
|
5994
|
-
break;
|
|
5995
|
-
}
|
|
6339
|
+
const close = codeSpanClose(n, i + n);
|
|
5996
6340
|
i = close === -1 ? i + n : close + n;
|
|
5997
6341
|
continue;
|
|
5998
6342
|
}
|
|
5999
6343
|
if (ch === "$") {
|
|
6000
6344
|
if (text[i + 1] === "$") {
|
|
6001
|
-
const close =
|
|
6002
|
-
if (close
|
|
6003
|
-
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: "$$" }));
|
|
6004
|
-
i = close + 2;
|
|
6005
|
-
} else {
|
|
6345
|
+
const close = this.mathPairs.get(i);
|
|
6346
|
+
if (close === void 0) {
|
|
6006
6347
|
i += 2;
|
|
6348
|
+
continue;
|
|
6007
6349
|
}
|
|
6350
|
+
if (close + 2 > pe) return stop(this.island("math_block", i, close + 2, true, { delimiter: "$$" }));
|
|
6351
|
+
inlines.push(this.island("math_inline", i, close + 2, true, { delimiter: "$$" }));
|
|
6352
|
+
i = close + 2;
|
|
6008
6353
|
continue;
|
|
6009
6354
|
}
|
|
6010
6355
|
const end = singleDollarMathEnd(text, i, pe);
|
|
@@ -6018,7 +6363,7 @@ var Tokenizer = class {
|
|
|
6018
6363
|
}
|
|
6019
6364
|
if (ch === "{") {
|
|
6020
6365
|
if (text[i + 1] === "{") {
|
|
6021
|
-
const close =
|
|
6366
|
+
const close = this.next("}}", i + 2);
|
|
6022
6367
|
if (close !== -1 && close + 2 <= pe && close - i <= 200 && !text.slice(i, close).includes("\n")) {
|
|
6023
6368
|
const raw = text.slice(i, close + 2);
|
|
6024
6369
|
const strict = STRICT_VARIABLE_RE.exec(raw);
|
|
@@ -6031,8 +6376,10 @@ var Tokenizer = class {
|
|
|
6031
6376
|
i += 2;
|
|
6032
6377
|
continue;
|
|
6033
6378
|
}
|
|
6034
|
-
if (hasKind) {
|
|
6035
|
-
const
|
|
6379
|
+
if (hasKind && kindBudget > 0 && /^\{\s*"/.test(text.slice(i, i + 64))) {
|
|
6380
|
+
const limit = Math.min(pe, i + 65536);
|
|
6381
|
+
const end = matchingJsonObjectEnd(text, i, limit);
|
|
6382
|
+
kindBudget -= (end ?? limit) - i;
|
|
6036
6383
|
if (end !== null) {
|
|
6037
6384
|
const kind = declaredKind(text.slice(i, end));
|
|
6038
6385
|
if (kind) {
|
|
@@ -6045,11 +6392,27 @@ var Tokenizer = class {
|
|
|
6045
6392
|
i++;
|
|
6046
6393
|
continue;
|
|
6047
6394
|
}
|
|
6395
|
+
if (ch === "[" && text[i + 1] === "[" || ch === "!" && text[i + 1] === "[" && text[i + 2] === "[") {
|
|
6396
|
+
const open = ch === "!" ? i + 3 : i + 2;
|
|
6397
|
+
const close = within("]]", open);
|
|
6398
|
+
const newline = this.next("\n", open);
|
|
6399
|
+
const inner = close === -1 ? "" : text.slice(open, close);
|
|
6400
|
+
if (close !== -1 && (newline === -1 || newline > close) && inner.trim() && !/[[\]]/.test(inner)) {
|
|
6401
|
+
const cut = inner.search(/\\?\|/);
|
|
6402
|
+
const target = (cut < 0 ? inner : inner.slice(0, cut)).trim();
|
|
6403
|
+
inlines.push(
|
|
6404
|
+
this.island("wikilink", i, close + 2, true, ch === "!" ? { target, embed: true } : { target })
|
|
6405
|
+
);
|
|
6406
|
+
i = close + 2;
|
|
6407
|
+
continue;
|
|
6408
|
+
}
|
|
6409
|
+
}
|
|
6048
6410
|
if (ch === "[") {
|
|
6049
6411
|
const media = MEDIA_REF_PREFIXES.find((prefix) => text.startsWith(prefix, i));
|
|
6050
6412
|
if (media) {
|
|
6051
|
-
const close =
|
|
6052
|
-
|
|
6413
|
+
const close = within("]", i + media.length);
|
|
6414
|
+
const newline = this.next("\n", i);
|
|
6415
|
+
if (close !== -1 && (newline === -1 || newline > close)) {
|
|
6053
6416
|
inlines.push(this.island("media_ref", i, close + 1, true, { media: media.slice(1, 6).toLowerCase() }));
|
|
6054
6417
|
i = close + 1;
|
|
6055
6418
|
continue;
|
|
@@ -6060,10 +6423,10 @@ var Tokenizer = class {
|
|
|
6060
6423
|
}
|
|
6061
6424
|
if (ch === "<") {
|
|
6062
6425
|
if (text.startsWith("<!--", i)) {
|
|
6063
|
-
const close =
|
|
6426
|
+
const close = this.next("-->", i + 4);
|
|
6064
6427
|
if (close === -1) return stop(this.island("html_comment", i, this.n, false));
|
|
6065
6428
|
const end = close + 3;
|
|
6066
|
-
const anchor = PINNED_ANCHOR_RE.exec(text.slice(i, end));
|
|
6429
|
+
const anchor = end - i <= 64 ? PINNED_ANCHOR_RE.exec(text.slice(i, end)) : null;
|
|
6067
6430
|
const type = anchor ? "anchor" : "html_comment";
|
|
6068
6431
|
const meta = anchor ? { id: anchor[1] } : {};
|
|
6069
6432
|
if (end > pe) return stop(this.island(type, i, end, true, meta));
|
|
@@ -6077,7 +6440,7 @@ var Tokenizer = class {
|
|
|
6077
6440
|
i += cite[0].length;
|
|
6078
6441
|
continue;
|
|
6079
6442
|
}
|
|
6080
|
-
const tag = readXmlTag(text, i);
|
|
6443
|
+
const tag = readXmlTag(text, i, this.finder);
|
|
6081
6444
|
if (tag && i + tag.raw.length <= pe) {
|
|
6082
6445
|
const name = tag.tagName;
|
|
6083
6446
|
const tagEnd = i + tag.raw.length;
|
|
@@ -6090,7 +6453,7 @@ var Tokenizer = class {
|
|
|
6090
6453
|
continue;
|
|
6091
6454
|
}
|
|
6092
6455
|
const closer = `</${name}>`;
|
|
6093
|
-
const close =
|
|
6456
|
+
const close = this.next(closer, tagEnd);
|
|
6094
6457
|
const blockType = isAttr ? "xml_attr" : "xml_region";
|
|
6095
6458
|
if (close === -1) return stop(this.island(blockType, i, this.n, false, { tag: name }));
|
|
6096
6459
|
const end = close + closer.length;
|
|
@@ -6114,7 +6477,7 @@ var Tokenizer = class {
|
|
|
6114
6477
|
const pending = ATTRIBUTE_XML_NAMES.find(
|
|
6115
6478
|
(name) => text.startsWith(`<${name}`, i) && /\s/.test(text[i + name.length + 1] ?? "")
|
|
6116
6479
|
);
|
|
6117
|
-
if (pending &&
|
|
6480
|
+
if (pending && this.next(">", i) === -1) {
|
|
6118
6481
|
return stop(this.island("xml_attr", i, this.n, false, { tag: pending }));
|
|
6119
6482
|
}
|
|
6120
6483
|
i++;
|
|
@@ -6125,8 +6488,8 @@ var Tokenizer = class {
|
|
|
6125
6488
|
return { inlines, breakIsland: null, orphan: false };
|
|
6126
6489
|
}
|
|
6127
6490
|
};
|
|
6128
|
-
function tokenizeSource(text) {
|
|
6129
|
-
const raw = new Tokenizer(text).run();
|
|
6491
|
+
function tokenizeSource(text, options = {}) {
|
|
6492
|
+
const raw = new Tokenizer(text, options.streaming ?? false).run();
|
|
6130
6493
|
const cp = buildCodePointIndex(text);
|
|
6131
6494
|
const inline = (island) => ({
|
|
6132
6495
|
islandType: island.type,
|
|
@@ -6175,7 +6538,7 @@ function listIslands(blocks) {
|
|
|
6175
6538
|
raw: block.raw
|
|
6176
6539
|
});
|
|
6177
6540
|
} else {
|
|
6178
|
-
|
|
6541
|
+
for (const inline of block.inlines) out.push(inline);
|
|
6179
6542
|
}
|
|
6180
6543
|
}
|
|
6181
6544
|
return out;
|
|
@@ -6205,9 +6568,44 @@ var SourceSpliceError = class extends Error {
|
|
|
6205
6568
|
function blockEdit(block, text) {
|
|
6206
6569
|
return { start: block.start, end: block.end, text };
|
|
6207
6570
|
}
|
|
6571
|
+
function islandEdit(island, text) {
|
|
6572
|
+
return { start: island.start, end: island.end, text, island: true };
|
|
6573
|
+
}
|
|
6208
6574
|
function blockRangeEdit(first, last, text) {
|
|
6209
6575
|
return { start: first.start, end: last.end, text };
|
|
6210
6576
|
}
|
|
6577
|
+
var CONTEXT_WINDOW = 64;
|
|
6578
|
+
function locateIsland(text, raw, from, expected, expectedFromEnd, oldText, oldStart, oldEnd) {
|
|
6579
|
+
const candidates = /* @__PURE__ */ new Set();
|
|
6580
|
+
if (expected >= from && text.startsWith(raw, expected)) candidates.add(expected);
|
|
6581
|
+
if (expectedFromEnd >= from && text.startsWith(raw, expectedFromEnd)) candidates.add(expectedFromEnd);
|
|
6582
|
+
const after = text.indexOf(raw, Math.max(from, expected));
|
|
6583
|
+
if (after !== -1) candidates.add(after);
|
|
6584
|
+
const before = expected > from ? text.lastIndexOf(raw, expected) : -1;
|
|
6585
|
+
if (before >= from) candidates.add(before);
|
|
6586
|
+
if (candidates.size === 0) return text.indexOf(raw, from);
|
|
6587
|
+
if (candidates.size === 1) return candidates.values().next().value;
|
|
6588
|
+
let best = -1;
|
|
6589
|
+
let bestScore = -1;
|
|
6590
|
+
for (const at of candidates) {
|
|
6591
|
+
let score = 0;
|
|
6592
|
+
for (let k = 0; k < CONTEXT_WINDOW; k++) {
|
|
6593
|
+
const n = text[at + raw.length + k];
|
|
6594
|
+
if (n === void 0 || n !== oldText[oldEnd + k]) break;
|
|
6595
|
+
score++;
|
|
6596
|
+
}
|
|
6597
|
+
for (let k = 1; k <= CONTEXT_WINDOW; k++) {
|
|
6598
|
+
const n = text[at - k];
|
|
6599
|
+
if (n === void 0 || n !== oldText[oldStart - k]) break;
|
|
6600
|
+
score++;
|
|
6601
|
+
}
|
|
6602
|
+
if (score > bestScore || score === bestScore && at === expected) {
|
|
6603
|
+
best = at;
|
|
6604
|
+
bestScore = score;
|
|
6605
|
+
}
|
|
6606
|
+
}
|
|
6607
|
+
return best;
|
|
6608
|
+
}
|
|
6211
6609
|
function commonPrefix(a, b) {
|
|
6212
6610
|
const max = Math.min(a.length, b.length);
|
|
6213
6611
|
let i = 0;
|
|
@@ -6232,13 +6630,23 @@ function spliceSave(original, edits, options = {}) {
|
|
|
6232
6630
|
boundaries.add(block.start);
|
|
6233
6631
|
boundaries.add(block.end);
|
|
6234
6632
|
}
|
|
6633
|
+
const islands = listIslands(blocks);
|
|
6634
|
+
const islandAt = /* @__PURE__ */ new Map();
|
|
6635
|
+
for (const island of islands) islandAt.set(`${island.start}:${island.end}`, island);
|
|
6235
6636
|
const sorted = [...edits].sort((a, b) => a.start - b.start || a.end - b.end);
|
|
6236
6637
|
let previousEnd = -1;
|
|
6237
6638
|
for (const edit of sorted) {
|
|
6238
6639
|
if (edit.start < 0 || edit.end > original.length || edit.start > edit.end) {
|
|
6239
6640
|
throw new SourceSpliceError("out_of_range", `edit [${edit.start}, ${edit.end}) is outside the text`);
|
|
6240
6641
|
}
|
|
6241
|
-
if (
|
|
6642
|
+
if (edit.island) {
|
|
6643
|
+
if (!islandAt.has(`${edit.start}:${edit.end}`)) {
|
|
6644
|
+
throw new SourceSpliceError(
|
|
6645
|
+
"misaligned",
|
|
6646
|
+
`islandEdit [${edit.start}, ${edit.end}) does not name an island's exact span`
|
|
6647
|
+
);
|
|
6648
|
+
}
|
|
6649
|
+
} else if (!boundaries.has(edit.start) || !boundaries.has(edit.end)) {
|
|
6242
6650
|
throw new SourceSpliceError(
|
|
6243
6651
|
"misaligned",
|
|
6244
6652
|
`edit [${edit.start}, ${edit.end}) does not start and end on block boundaries`
|
|
@@ -6250,18 +6658,51 @@ function spliceSave(original, edits, options = {}) {
|
|
|
6250
6658
|
previousEnd = edit.end;
|
|
6251
6659
|
}
|
|
6252
6660
|
const effective = [];
|
|
6253
|
-
|
|
6254
|
-
|
|
6255
|
-
|
|
6256
|
-
|
|
6257
|
-
const
|
|
6661
|
+
const replacedIslands = /* @__PURE__ */ new Set();
|
|
6662
|
+
const pushSegment = (oldStart, oldEnd, replacement) => {
|
|
6663
|
+
const old = original.slice(oldStart, oldEnd);
|
|
6664
|
+
if (old === replacement) return;
|
|
6665
|
+
const prefix = commonPrefix(old, replacement);
|
|
6666
|
+
const suffix = commonSuffix(old, replacement, prefix);
|
|
6258
6667
|
effective.push({
|
|
6259
|
-
|
|
6260
|
-
|
|
6261
|
-
|
|
6262
|
-
oldEnd: edit.end - suffix,
|
|
6263
|
-
replacement: edit.text.slice(prefix, edit.text.length - suffix)
|
|
6668
|
+
oldStart: oldStart + prefix,
|
|
6669
|
+
oldEnd: oldEnd - suffix,
|
|
6670
|
+
replacement: replacement.slice(prefix, replacement.length - suffix)
|
|
6264
6671
|
});
|
|
6672
|
+
};
|
|
6673
|
+
for (const edit of sorted) {
|
|
6674
|
+
if (edit.island) {
|
|
6675
|
+
replacedIslands.add(`${edit.start}:${edit.end}`);
|
|
6676
|
+
pushSegment(edit.start, edit.end, edit.text);
|
|
6677
|
+
continue;
|
|
6678
|
+
}
|
|
6679
|
+
if (edit.text === original.slice(edit.start, edit.end)) continue;
|
|
6680
|
+
const inside = islands.filter((island) => island.start >= edit.start && island.end <= edit.end);
|
|
6681
|
+
let oldCursor2 = edit.start;
|
|
6682
|
+
let newCursor2 = 0;
|
|
6683
|
+
for (const island of inside) {
|
|
6684
|
+
const at = locateIsland(
|
|
6685
|
+
edit.text,
|
|
6686
|
+
island.raw,
|
|
6687
|
+
newCursor2,
|
|
6688
|
+
newCursor2 + (island.start - oldCursor2),
|
|
6689
|
+
edit.text.length - (edit.end - island.start),
|
|
6690
|
+
original.slice(edit.start, edit.end),
|
|
6691
|
+
island.start - edit.start,
|
|
6692
|
+
island.end - edit.start
|
|
6693
|
+
);
|
|
6694
|
+
if (at === -1) {
|
|
6695
|
+
const name = island.meta.tag ?? island.meta.name ?? island.meta.lang ?? "";
|
|
6696
|
+
throw new SourceSpliceError(
|
|
6697
|
+
"island_edit",
|
|
6698
|
+
`edit [${edit.start}, ${edit.end}) changes or removes the protected ${island.islandType}${name ? ` (${String(name)})` : ""} island at [${island.start}, ${island.end}); islands change only through islandEdit()`
|
|
6699
|
+
);
|
|
6700
|
+
}
|
|
6701
|
+
pushSegment(oldCursor2, island.start, edit.text.slice(newCursor2, at));
|
|
6702
|
+
oldCursor2 = island.end;
|
|
6703
|
+
newCursor2 = at + island.raw.length;
|
|
6704
|
+
}
|
|
6705
|
+
pushSegment(oldCursor2, edit.end, edit.text.slice(newCursor2));
|
|
6265
6706
|
}
|
|
6266
6707
|
if (effective.length === 0) {
|
|
6267
6708
|
return {
|
|
@@ -6281,9 +6722,10 @@ function spliceSave(original, edits, options = {}) {
|
|
|
6281
6722
|
const newStart = edit.oldStart + delta;
|
|
6282
6723
|
text += edit.replacement;
|
|
6283
6724
|
const newEnd = newStart + edit.replacement.length;
|
|
6725
|
+
const owner = sorted.find((e) => e.start <= edit.oldStart && e.end >= edit.oldEnd);
|
|
6284
6726
|
pending.push({
|
|
6285
|
-
blockStart: edit.
|
|
6286
|
-
blockEnd: edit.
|
|
6727
|
+
blockStart: owner?.start ?? edit.oldStart,
|
|
6728
|
+
blockEnd: owner?.end ?? edit.oldEnd,
|
|
6287
6729
|
oldStart: edit.oldStart,
|
|
6288
6730
|
oldEnd: edit.oldEnd,
|
|
6289
6731
|
newStart,
|
|
@@ -6321,11 +6763,8 @@ function spliceSave(original, edits, options = {}) {
|
|
|
6321
6763
|
else newIslandsByStart.set(island.start, [island.raw]);
|
|
6322
6764
|
}
|
|
6323
6765
|
const disturbed = [];
|
|
6324
|
-
for (const island of
|
|
6325
|
-
|
|
6326
|
-
(edit) => island.start < edit.blockEnd && island.end > edit.blockStart
|
|
6327
|
-
);
|
|
6328
|
-
if (touched) continue;
|
|
6766
|
+
for (const island of islands) {
|
|
6767
|
+
if (replacedIslands.has(`${island.start}:${island.end}`)) continue;
|
|
6329
6768
|
const mappedStart = mapPosition(changes, island.start, { assoc: 1 }).pos;
|
|
6330
6769
|
const candidates = newIslandsByStart.get(mappedStart);
|
|
6331
6770
|
if (!candidates || !candidates.includes(island.raw)) {
|
|
@@ -6343,10 +6782,11 @@ function spliceSave(original, edits, options = {}) {
|
|
|
6343
6782
|
bytesOutsideEditsIdentical,
|
|
6344
6783
|
disturbed
|
|
6345
6784
|
};
|
|
6346
|
-
if (options.requireIntegrity && !integrity.ok) {
|
|
6785
|
+
if ((options.requireIntegrity ?? true) && !integrity.ok) {
|
|
6786
|
+
const named = disturbed.slice(0, 5).map((island) => `${island.islandType}@${island.start}`).join(", ");
|
|
6347
6787
|
throw new SourceSpliceError(
|
|
6348
6788
|
"integrity",
|
|
6349
|
-
`the
|
|
6789
|
+
`integrity: the save disturbed ${disturbed.length} protected island(s) it did not replace${named ? ` (${named}${disturbed.length > 5 ? ", \u2026" : ""})` : ""}${bytesOutsideEditsIdentical ? "" : "; bytes outside the edits changed"}`
|
|
6350
6790
|
);
|
|
6351
6791
|
}
|
|
6352
6792
|
return { text, changed: true, changes, integrity, blocks: newBlocks };
|
|
@@ -6393,6 +6833,7 @@ exports.CAPABILITY_BY_CLASS = CAPABILITY_BY_CLASS;
|
|
|
6393
6833
|
exports.CLASSES = CLASSES;
|
|
6394
6834
|
exports.DIRECTIVE_VERSION = DIRECTIVE_VERSION;
|
|
6395
6835
|
exports.DirectiveDecodeError = DirectiveDecodeError;
|
|
6836
|
+
exports.FENCE_WHITESPACE = FENCE_WHITESPACE;
|
|
6396
6837
|
exports.IN_CONTENT_CLASSES = IN_CONTENT_CLASSES;
|
|
6397
6838
|
exports.IR_ENVELOPE_CACHE_VERSION = IR_ENVELOPE_CACHE_VERSION;
|
|
6398
6839
|
exports.IR_ENVELOPE_KEY = IR_ENVELOPE_KEY;
|
|
@@ -6404,6 +6845,7 @@ exports.JsonStreamTokenizer = JsonStreamTokenizer;
|
|
|
6404
6845
|
exports.KIND_KEY = KIND_KEY;
|
|
6405
6846
|
exports.KindStorageError = KindStorageError;
|
|
6406
6847
|
exports.KindStreamParser = KindStreamParser;
|
|
6848
|
+
exports.NESTING_FENCE_LANGUAGES = NESTING_FENCE_LANGUAGES;
|
|
6407
6849
|
exports.NODE_OUTCOME_KIND = NODE_OUTCOME_KIND;
|
|
6408
6850
|
exports.ParseSession = ParseSession;
|
|
6409
6851
|
exports.RESERVED_PREFIX = RESERVED_PREFIX;
|
|
@@ -6425,6 +6867,7 @@ exports.buildDirectiveSlug = buildDirectiveSlug;
|
|
|
6425
6867
|
exports.buildKindDirective = buildKindDirective;
|
|
6426
6868
|
exports.capabilityOf = capabilityOf;
|
|
6427
6869
|
exports.classifyInboundEnvelopeMetadata = classifyInboundEnvelopeMetadata;
|
|
6870
|
+
exports.classifyInnerFenceLine = classifyInnerFenceLine;
|
|
6428
6871
|
exports.collectReferencedKinds = collectReferencedKinds;
|
|
6429
6872
|
exports.collectSchemaReferencedKinds = collectSchemaReferencedKinds;
|
|
6430
6873
|
exports.compareWithExistingKindSchema = compareWithExistingKindSchema;
|
|
@@ -6444,6 +6887,7 @@ exports.envelopeCacheFromEnvelopes = envelopeCacheFromEnvelopes;
|
|
|
6444
6887
|
exports.envelopeFromCompleteValue = envelopeFromCompleteValue;
|
|
6445
6888
|
exports.executesAtOutputRoot = executesAtOutputRoot;
|
|
6446
6889
|
exports.fenceDiscriminator = fenceDiscriminator;
|
|
6890
|
+
exports.fenceNestsInnerFences = fenceNestsInnerFences;
|
|
6447
6891
|
exports.fieldsToDbPayload = fieldsToDbPayload;
|
|
6448
6892
|
exports.fingerprintText = fingerprintText;
|
|
6449
6893
|
exports.formatBlockLabel = formatBlockLabel;
|
|
@@ -6466,6 +6910,7 @@ exports.isReservedDirectiveSlug = isReservedDirectiveSlug;
|
|
|
6466
6910
|
exports.isScalarArrayType = isScalarArrayType;
|
|
6467
6911
|
exports.isSingleDollarMath = isSingleDollarMath;
|
|
6468
6912
|
exports.isTerminalKindEvent = isTerminalKindEvent;
|
|
6913
|
+
exports.islandEdit = islandEdit;
|
|
6469
6914
|
exports.itemFacts = itemFacts;
|
|
6470
6915
|
exports.itemSubtitle = itemSubtitle;
|
|
6471
6916
|
exports.itemTitle = itemTitle;
|
|
@@ -6477,6 +6922,7 @@ exports.kindVerdictOf = kindVerdictOf;
|
|
|
6477
6922
|
exports.legacyShellUses = legacyShellUses;
|
|
6478
6923
|
exports.listIslands = listIslands;
|
|
6479
6924
|
exports.looksLikeDirectiveHead = looksLikeDirectiveHead;
|
|
6925
|
+
exports.looksLikeDisplayMath = looksLikeDisplayMath;
|
|
6480
6926
|
exports.makePartialKindStalenessGate = makePartialKindStalenessGate;
|
|
6481
6927
|
exports.mapPosition = mapPosition;
|
|
6482
6928
|
exports.mapRange = mapRange;
|
|
@@ -6487,6 +6933,7 @@ exports.nounFamily = nounFamily;
|
|
|
6487
6933
|
exports.nounLabel = nounLabel;
|
|
6488
6934
|
exports.nounTitleColumn = nounTitleColumn;
|
|
6489
6935
|
exports.openParseSession = openParseSession;
|
|
6936
|
+
exports.pairDisplayMath = pairDisplayMath;
|
|
6490
6937
|
exports.parseDirectiveSlug = parseDirectiveSlug;
|
|
6491
6938
|
exports.readEnvelope = readEnvelope;
|
|
6492
6939
|
exports.readNodeOutcomeValue = readNodeOutcomeValue;
|
|
@@ -6518,6 +6965,7 @@ exports.titleCaseToken = titleCaseToken;
|
|
|
6518
6965
|
exports.toCodePointOffset = toCodePointOffset;
|
|
6519
6966
|
exports.toUtf16Offset = toUtf16Offset;
|
|
6520
6967
|
exports.tokenizeSource = tokenizeSource;
|
|
6968
|
+
exports.trimFenceLine = trimFenceLine;
|
|
6521
6969
|
exports.tryDecodeDirective = tryDecodeDirective;
|
|
6522
6970
|
exports.tryDecodeDirectiveContent = tryDecodeDirectiveContent;
|
|
6523
6971
|
exports.validateBlockSchemaSavePlan = validateBlockSchemaSavePlan;
|