solve-engine 2.15.0 → 2.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/dist/{BytecodeBuilder-CRYfrFfq.d.cts → BytecodeBuilder-aqVa7Plx.d.cts} +10 -0
  2. package/dist/{BytecodeBuilder-CRYfrFfq.d.ts → BytecodeBuilder-aqVa7Plx.d.ts} +10 -0
  3. package/dist/{EngineError-Cv5q4Rbv.d.cts → EngineError-DTk7I7hZ.d.cts} +24 -0
  4. package/dist/{EngineError-Cv5q4Rbv.d.ts → EngineError-DTk7I7hZ.d.ts} +24 -0
  5. package/dist/{PackageCompatibility-Cl7GF_Iu.d.ts → PackageCompatibility-BGXSFuL9.d.ts} +1 -1
  6. package/dist/{PackageCompatibility-C2IQZ_w3.d.cts → PackageCompatibility-BZCqaTOO.d.cts} +1 -1
  7. package/dist/{PackageRegistry-CK_JoX50.d.ts → PackageRegistry-BRoVOYzg.d.ts} +107 -5
  8. package/dist/{PackageRegistry-CZO1IbS-.d.cts → PackageRegistry-rMGnTh7W.d.cts} +107 -5
  9. package/dist/{Parselet-CK0bNO1l.d.ts → Parselet-B-WyUtX4.d.ts} +5 -1
  10. package/dist/{Parselet-DOGmj6N6.d.cts → Parselet-Bkbp9CKD.d.cts} +5 -1
  11. package/dist/{ScopeManager-BtqiTVjG.d.cts → ScopeManager-bCYewWVt.d.cts} +2 -2
  12. package/dist/{ScopeManager-C6c1WmJD.d.ts → ScopeManager-gB9UengK.d.ts} +2 -2
  13. package/dist/{TokenNormalizer-BfTHg1qt.d.cts → TokenNormalizer-MXaKLJ_m.d.cts} +182 -0
  14. package/dist/{TokenNormalizer-d7F1KQFs.d.ts → TokenNormalizer-OTPS0Otq.d.ts} +182 -0
  15. package/dist/{VMCheckpoints-DJdvpliw.d.ts → VMCheckpoints-BDRY1Kx8.d.ts} +2 -2
  16. package/dist/{VMCheckpoints-BFDlaGse.d.cts → VMCheckpoints-DGar9Yg8.d.cts} +2 -2
  17. package/dist/{WorkerError-FH7KXX4_.d.cts → WorkerError-DGBNM3gA.d.cts} +1 -1
  18. package/dist/{WorkerError-J3v3ix_z.d.ts → WorkerError-DpZgKRks.d.ts} +1 -1
  19. package/dist/{chunk-UZSPBFZN.js → chunk-2DPKJ2SM.js} +2 -2
  20. package/dist/{chunk-UZSPBFZN.js.map → chunk-2DPKJ2SM.js.map} +1 -1
  21. package/dist/{chunk-O3ALJLYC.js → chunk-3GR46UOX.js} +2 -2
  22. package/dist/{chunk-O3ALJLYC.js.map → chunk-3GR46UOX.js.map} +1 -1
  23. package/dist/{chunk-PREIL3IB.cjs → chunk-3SSMOVWB.cjs} +2 -2
  24. package/dist/{chunk-PREIL3IB.cjs.map → chunk-3SSMOVWB.cjs.map} +1 -1
  25. package/dist/{chunk-XK23K3EJ.js → chunk-4ETS224G.js} +2 -2
  26. package/dist/{chunk-XK23K3EJ.js.map → chunk-4ETS224G.js.map} +1 -1
  27. package/dist/{chunk-Q7HPPLMG.js → chunk-5FMYW44V.js} +2 -2
  28. package/dist/{chunk-Q7HPPLMG.js.map → chunk-5FMYW44V.js.map} +1 -1
  29. package/dist/{chunk-GWCBPITD.cjs → chunk-6YM67DWH.cjs} +3 -3
  30. package/dist/{chunk-GWCBPITD.cjs.map → chunk-6YM67DWH.cjs.map} +1 -1
  31. package/dist/{chunk-6U66HQKQ.cjs → chunk-75JLWK4L.cjs} +2 -2
  32. package/dist/{chunk-6U66HQKQ.cjs.map → chunk-75JLWK4L.cjs.map} +1 -1
  33. package/dist/chunk-A2RQUV5H.cjs +5 -0
  34. package/dist/chunk-A2RQUV5H.cjs.map +1 -0
  35. package/dist/{chunk-VJP4GGCW.cjs → chunk-BNCBB4H5.cjs} +2 -2
  36. package/dist/{chunk-VJP4GGCW.cjs.map → chunk-BNCBB4H5.cjs.map} +1 -1
  37. package/dist/{chunk-EEUA3OJ6.cjs → chunk-CEURQXSN.cjs} +2 -2
  38. package/dist/{chunk-EEUA3OJ6.cjs.map → chunk-CEURQXSN.cjs.map} +1 -1
  39. package/dist/chunk-CO2BI6WL.cjs +3 -0
  40. package/dist/chunk-CO2BI6WL.cjs.map +1 -0
  41. package/dist/chunk-EN3CDYOS.cjs +2 -0
  42. package/dist/chunk-EN3CDYOS.cjs.map +1 -0
  43. package/dist/{chunk-GWJJRH32.js → chunk-EQDT42JG.js} +3 -3
  44. package/dist/{chunk-GWJJRH32.js.map → chunk-EQDT42JG.js.map} +1 -1
  45. package/dist/chunk-G2MFTERZ.cjs +2 -0
  46. package/dist/chunk-G2MFTERZ.cjs.map +1 -0
  47. package/dist/chunk-G63TMUL2.js +2 -0
  48. package/dist/chunk-G63TMUL2.js.map +1 -0
  49. package/dist/{chunk-UUAFDZK6.cjs → chunk-H3JXNH7X.cjs} +3 -3
  50. package/dist/{chunk-UUAFDZK6.cjs.map → chunk-H3JXNH7X.cjs.map} +1 -1
  51. package/dist/{chunk-3Y56PSQD.cjs → chunk-HQ2PE6HY.cjs} +3 -3
  52. package/dist/{chunk-3Y56PSQD.cjs.map → chunk-HQ2PE6HY.cjs.map} +1 -1
  53. package/dist/chunk-IES365YJ.cjs +3 -0
  54. package/dist/chunk-IES365YJ.cjs.map +1 -0
  55. package/dist/{chunk-2XZCLDHJ.cjs → chunk-J7ABVGZH.cjs} +2 -2
  56. package/dist/{chunk-2XZCLDHJ.cjs.map → chunk-J7ABVGZH.cjs.map} +1 -1
  57. package/dist/chunk-JVMINMAB.js +2 -0
  58. package/dist/chunk-JVMINMAB.js.map +1 -0
  59. package/dist/chunk-JYLNQOPU.cjs +2 -0
  60. package/dist/chunk-JYLNQOPU.cjs.map +1 -0
  61. package/dist/{chunk-JOIQFDZQ.js → chunk-KEA5HRL3.js} +2 -2
  62. package/dist/{chunk-JOIQFDZQ.js.map → chunk-KEA5HRL3.js.map} +1 -1
  63. package/dist/{chunk-TVE2DNPU.js → chunk-LJFS3XHW.js} +2 -2
  64. package/dist/{chunk-TVE2DNPU.js.map → chunk-LJFS3XHW.js.map} +1 -1
  65. package/dist/{chunk-C5MP74ER.js → chunk-MRRMIBHE.js} +2 -2
  66. package/dist/{chunk-C5MP74ER.js.map → chunk-MRRMIBHE.js.map} +1 -1
  67. package/dist/chunk-N64ZK6CL.cjs +2 -0
  68. package/dist/{chunk-EEVUKZ5B.cjs.map → chunk-N64ZK6CL.cjs.map} +1 -1
  69. package/dist/chunk-O3BDXOQJ.js +2 -0
  70. package/dist/chunk-O3BDXOQJ.js.map +1 -0
  71. package/dist/chunk-PIZQIQVM.js +3 -0
  72. package/dist/chunk-PIZQIQVM.js.map +1 -0
  73. package/dist/chunk-PS5OA7QU.js +5 -0
  74. package/dist/chunk-PS5OA7QU.js.map +1 -0
  75. package/dist/{chunk-IPOU3JTA.js → chunk-QTSDIDAS.js} +3 -3
  76. package/dist/{chunk-IPOU3JTA.js.map → chunk-QTSDIDAS.js.map} +1 -1
  77. package/dist/chunk-RHE3LC4Q.js +2 -0
  78. package/dist/chunk-RHE3LC4Q.js.map +1 -0
  79. package/dist/{chunk-P4ETODMP.js → chunk-RLTH2H3U.js} +2 -2
  80. package/dist/{chunk-P4ETODMP.js.map → chunk-RLTH2H3U.js.map} +1 -1
  81. package/dist/{chunk-WOIJPV7Z.js → chunk-SEOIKHYK.js} +2 -2
  82. package/dist/{chunk-WOIJPV7Z.js.map → chunk-SEOIKHYK.js.map} +1 -1
  83. package/dist/{chunk-W5ELYJ4Z.cjs → chunk-TL545PXW.cjs} +2 -2
  84. package/dist/{chunk-W5ELYJ4Z.cjs.map → chunk-TL545PXW.cjs.map} +1 -1
  85. package/dist/{chunk-FAGFPVYK.cjs → chunk-TRDOYAKJ.cjs} +2 -2
  86. package/dist/{chunk-FAGFPVYK.cjs.map → chunk-TRDOYAKJ.cjs.map} +1 -1
  87. package/dist/chunk-WMISTHB2.cjs +2 -0
  88. package/dist/chunk-WMISTHB2.cjs.map +1 -0
  89. package/dist/{chunk-3PDDKWTJ.js → chunk-XOEZ7UQI.js} +3 -3
  90. package/dist/{chunk-3PDDKWTJ.js.map → chunk-XOEZ7UQI.js.map} +1 -1
  91. package/dist/{chunk-EXCFCHAT.cjs → chunk-XPJGDGJF.cjs} +2 -2
  92. package/dist/{chunk-EXCFCHAT.cjs.map → chunk-XPJGDGJF.cjs.map} +1 -1
  93. package/dist/chunk-YQV4WGXH.js +3 -0
  94. package/dist/chunk-YQV4WGXH.js.map +1 -0
  95. package/dist/constants.cjs +1 -1
  96. package/dist/constants.js +1 -1
  97. package/dist/engine.cjs +1 -1
  98. package/dist/engine.d.cts +8 -8
  99. package/dist/engine.d.ts +8 -8
  100. package/dist/engine.js +1 -1
  101. package/dist/errors.cjs +1 -1
  102. package/dist/errors.d.cts +3 -3
  103. package/dist/errors.d.ts +3 -3
  104. package/dist/errors.js +1 -1
  105. package/dist/format.cjs +1 -1
  106. package/dist/format.js +1 -1
  107. package/dist/index.cjs +1 -1
  108. package/dist/index.d.cts +8 -8
  109. package/dist/index.d.ts +8 -8
  110. package/dist/index.js +1 -1
  111. package/dist/language.d.cts +7 -7
  112. package/dist/language.d.ts +7 -7
  113. package/dist/lexer.cjs +1 -1
  114. package/dist/lexer.js +1 -1
  115. package/dist/normalizer.cjs +1 -1
  116. package/dist/normalizer.d.cts +2 -2
  117. package/dist/normalizer.d.ts +2 -2
  118. package/dist/normalizer.js +1 -1
  119. package/dist/packages.cjs +1 -1
  120. package/dist/packages.d.cts +6 -6
  121. package/dist/packages.d.ts +6 -6
  122. package/dist/packages.js +1 -1
  123. package/dist/parser.cjs +1 -1
  124. package/dist/parser.d.cts +2 -2
  125. package/dist/parser.d.ts +2 -2
  126. package/dist/parser.js +1 -1
  127. package/dist/resolvers.d.cts +1 -1
  128. package/dist/resolvers.d.ts +1 -1
  129. package/dist/testing.cjs +2 -2
  130. package/dist/testing.d.cts +7 -7
  131. package/dist/testing.d.ts +7 -7
  132. package/dist/testing.js +1 -1
  133. package/dist/uom.cjs +1 -1
  134. package/dist/uom.d.cts +1 -1
  135. package/dist/uom.d.ts +1 -1
  136. package/dist/uom.js +1 -1
  137. package/dist/vm.cjs +1 -1
  138. package/dist/vm.d.cts +5 -5
  139. package/dist/vm.d.ts +5 -5
  140. package/dist/vm.js +1 -1
  141. package/dist/worker.cjs +2 -2
  142. package/dist/worker.d.cts +7 -7
  143. package/dist/worker.d.ts +7 -7
  144. package/dist/worker.js +1 -1
  145. package/package.json +1 -1
  146. package/dist/chunk-4IIFFJTJ.js +0 -5
  147. package/dist/chunk-4IIFFJTJ.js.map +0 -1
  148. package/dist/chunk-5SDSIRPL.js +0 -2
  149. package/dist/chunk-5SDSIRPL.js.map +0 -1
  150. package/dist/chunk-AYTEPHRO.js +0 -3
  151. package/dist/chunk-AYTEPHRO.js.map +0 -1
  152. package/dist/chunk-CUDPOWXA.js +0 -2
  153. package/dist/chunk-CUDPOWXA.js.map +0 -1
  154. package/dist/chunk-EEVUKZ5B.cjs +0 -2
  155. package/dist/chunk-FPGYRFJY.cjs +0 -5
  156. package/dist/chunk-FPGYRFJY.cjs.map +0 -1
  157. package/dist/chunk-FVQWN5HZ.cjs +0 -2
  158. package/dist/chunk-FVQWN5HZ.cjs.map +0 -1
  159. package/dist/chunk-HDGARCI2.cjs +0 -2
  160. package/dist/chunk-HDGARCI2.cjs.map +0 -1
  161. package/dist/chunk-IMXSVHQK.cjs +0 -2
  162. package/dist/chunk-IMXSVHQK.cjs.map +0 -1
  163. package/dist/chunk-KKMLFYVW.cjs +0 -3
  164. package/dist/chunk-KKMLFYVW.cjs.map +0 -1
  165. package/dist/chunk-NAZ6PGLS.cjs +0 -3
  166. package/dist/chunk-NAZ6PGLS.cjs.map +0 -1
  167. package/dist/chunk-Q3PSTNSY.js +0 -2
  168. package/dist/chunk-Q3PSTNSY.js.map +0 -1
  169. package/dist/chunk-RC4M6HBX.cjs +0 -2
  170. package/dist/chunk-RC4M6HBX.cjs.map +0 -1
  171. package/dist/chunk-YAZ4DUJJ.js +0 -3
  172. package/dist/chunk-YAZ4DUJJ.js.map +0 -1
  173. package/dist/chunk-YFX53MZF.js +0 -2
  174. package/dist/chunk-YFX53MZF.js.map +0 -1
@@ -69,6 +69,43 @@ interface NormalizerMatch {
69
69
  */
70
70
  ruleName?: string;
71
71
  }
72
+ /**
73
+ * One position of a rule's leading shape, as a declarative constraint.
74
+ *
75
+ * A slot says what the token at that offset from the match position may be.
76
+ * Both fields are optional and an omitted one constrains nothing, so `{}` is a
77
+ * wildcard slot, useful for reaching past a position a rule does not care about
78
+ * to one it does.
79
+ *
80
+ * The exactness contract runs in one direction only, and it is the whole reason
81
+ * this is safe to adopt gradually. Declaring MORE than the rule can match costs
82
+ * a `match()` call that returns null, which is what happens today anyway.
83
+ * Declaring LESS makes the rule unreachable at the positions left out, which is
84
+ * a silent bug. So an incomplete shape (or none at all) is always correct, and
85
+ * only an over-narrow one is wrong. `NormalizerIndexFidelity.spec` checks every
86
+ * declaration against its rule's real behaviour.
87
+ */
88
+ interface RuleSlot {
89
+ /**
90
+ * Token types admitted at this slot. Omit when the type is unconstrained,
91
+ * or when the rule accepts so many that naming them filters nothing.
92
+ */
93
+ readonly types?: readonly string[];
94
+ /**
95
+ * Token values admitted at this slot, compared case-insensitively.
96
+ *
97
+ * This is the axis that separates rules sharing a start type. The
98
+ * call-fusion rules all begin at an `IDENT`, the commonest token in prose,
99
+ * so type alone leaves every one of them a candidate at every word; the word
100
+ * itself is what tells them apart, and each already owns that set as an
101
+ * exported constant.
102
+ *
103
+ * A rule whose own check is case-SENSITIVE (the stock ticker rule matches
104
+ * upper case only) may still declare its lower-cased words here: the index
105
+ * only ever over-approximates, and the rule's own check still runs.
106
+ */
107
+ readonly values?: readonly string[];
108
+ }
72
109
  /**
73
110
  * A pluggable normalization rule registered with the TokenNormalizer.
74
111
  *
@@ -139,6 +176,60 @@ interface NormalizerRule {
139
176
  * the rule silently unreachable there, which is a bug.
140
177
  */
141
178
  readonly startTokenTypes?: readonly string[];
179
+ /**
180
+ * The rule's leading shape: what the tokens from the match position onward
181
+ * may be, one {@link RuleSlot} per position.
182
+ *
183
+ * This generalises {@link startTokenTypes}, which constrains only the first
184
+ * token. Constraining the first token alone is not enough to separate the
185
+ * rules that matter: every rule firing on a bare `NUMBER` declares the same
186
+ * start type, so they all remain candidates at every number in the document.
187
+ * What distinguishes them is the token after it, `NUMBER COLON` being a clock
188
+ * time and `NUMBER SLASH` a network address, and that fact is only usable by
189
+ * an index if the rule states it rather than hiding it inside `match()`.
190
+ *
191
+ * Depth is the rule's choice, not the interface's. The normalizer builds one
192
+ * lookup plane per declared slot and intersects them, so a rule that declares
193
+ * three positions is filtered on three. It may also index fewer planes than
194
+ * were declared, which stays correct for the reason given on {@link RuleSlot}:
195
+ * a shallower filter admits more candidates, and each surviving rule still
196
+ * runs its own `match()`.
197
+ *
198
+ * Prefer this to {@link startTokenTypes} in new rules. When both are given,
199
+ * this wins; `startTokenTypes: ["IDENT"]` means exactly `shape: [{ types:
200
+ * ["IDENT"] }]`.
201
+ *
202
+ * @example
203
+ * ```ts
204
+ * // 9:00am, 16:00, a clock time is a number followed by a colon
205
+ * shape: [{ types: ["NUMBER"] }, { types: ["COLON"] }]
206
+ *
207
+ * // sha256("hi"), a known word followed by an opening parenthesis
208
+ * shape: [{ types: ["IDENT"], values: HASH_NAMES }, { types: ["LPAREN"] }]
209
+ * ```
210
+ */
211
+ readonly shape?: readonly RuleSlot[];
212
+ /**
213
+ * Why this rule cannot declare a {@link shape}, for the few that genuinely
214
+ * cannot.
215
+ *
216
+ * A rule with neither a shape nor a `startTokenTypes` hint is tried at every
217
+ * position of every line, so it raises the cost of the whole document rather
218
+ * than only its own feature. Registering one logs a warning naming the rule,
219
+ * which is how a package author finds out before their users do.
220
+ *
221
+ * Some rules really cannot be described by a leading shape: an unbounded
222
+ * forward scan, a greedy match against a table the host mutates at runtime.
223
+ * Setting this states that case, silences the warning, and leaves the reason
224
+ * in the code where the next person will read it. It is deliberately a
225
+ * sentence and not a boolean, because "why" is the part worth keeping.
226
+ *
227
+ * @example
228
+ * ```ts
229
+ * unshapedReason: "Scans forward an unbounded number of NUMBER UNIT pairs, so no fixed leading shape describes it.",
230
+ * ```
231
+ */
232
+ readonly unshapedReason?: string;
142
233
  /**
143
234
  * Attempt to match a pattern starting at position `pos` in the token stream.
144
235
  *
@@ -228,6 +319,19 @@ interface NormalizerOptions {
228
319
  * When `undefined`, fusions are still tracked internally but no callbacks fire.
229
320
  */
230
321
  onFusion?: (fusion: TokenFusion) => void;
322
+ /**
323
+ * Try every registered rule at every position, ignoring the shape index.
324
+ *
325
+ * Diagnostic only, and much slower. It exists so the indexed walk can be
326
+ * compared against the unindexed one over a corpus: the index is a pure
327
+ * filter, so the two must agree token for token, and a rule whose declared
328
+ * {@link NormalizerRule.shape} is too narrow shows up as a difference rather
329
+ * than as a feature that quietly stopped working.
330
+ * `NormalizerIndexFidelity.spec` is the consumer.
331
+ *
332
+ * @default false
333
+ */
334
+ ignoreRuleIndex?: boolean;
231
335
  }
232
336
  /**
233
337
  * Creates a new normalized token from fused source tokens.
@@ -298,6 +402,28 @@ declare class TokenNormalizer {
298
402
  * identifier. Invalidated alongside {@link sortedRulesCache}.
299
403
  */
300
404
  private rulesByTokenType;
405
+ /**
406
+ * Shape index over the priority-sorted rules, rebuilt alongside
407
+ * {@link sortedRulesCache}. `null` means "stale, rebuild on next use".
408
+ *
409
+ * This is what turns the per-position scan from "try every rule that could
410
+ * fire on this token type" into "AND a few lookup planes and try what
411
+ * survives", which is usually nothing. See {@link RuleIndex}.
412
+ */
413
+ private ruleIndexCache;
414
+ /** Reused by {@link rulesAt} so a position's candidate list costs no allocation. */
415
+ private candidateBuffer;
416
+ /**
417
+ * How many times {@link normalize} has run, capped once the index is in use.
418
+ *
419
+ * The index costs a few tens of microseconds to build and pays that back
420
+ * within a line or two, but an engine that normalises exactly once, which is
421
+ * what evaluating a single expression on a fresh engine does, would never
422
+ * reach the payback. Skipping the build on the first call keeps that case at
423
+ * the cost it had before the index existed, and a document reaches line two
424
+ * immediately.
425
+ */
426
+ private normalizeCalls;
301
427
  /**
302
428
  * Phrase trie for single-pass multi-word phrase fusion.
303
429
  * Tried at each token position BEFORE other rules, the trie walk
@@ -348,11 +474,67 @@ declare class TokenNormalizer {
348
474
  * and cached, which is what turns the per-position rule scan from "try all R
349
475
  * rules" into "try only the ones that could fire on this token".
350
476
  */
477
+ private getRuleIndex;
478
+ /**
479
+ * The rules to try at a position, in priority order.
480
+ *
481
+ * With a candidate mask, walks its set bits low to high, which is descending
482
+ * priority because bit `i` is the rule at index `i` of the priority-sorted
483
+ * list.
484
+ *
485
+ * The walk shifts a bit at a time rather than jumping to the next set bit
486
+ * with `Math.clz32(bits & -bits)`, which is the idiomatic form and was 36x
487
+ * SLOWER here: 45us per line against 1.2us, measured over the built-in rule
488
+ * set. `clz32` is specified on uint32 and the masks come out of a
489
+ * `Uint32Array` as doubles above 2^31, so every call pays a conversion that
490
+ * swamps the handful of iterations it saves. A de Bruijn table matched the
491
+ * shift scan to within noise and needs a magic constant, so the plain shift
492
+ * wins on both counts. Worst case is 32 iterations per word of pure integer
493
+ * work.
494
+ *
495
+ * With `null` (the {@link NormalizerOptions.ignoreRuleIndex} path) it falls
496
+ * back to the type-bucketed list, which is the behaviour this replaced.
497
+ *
498
+ * Returns a buffer reused across positions, so the caller must finish with it
499
+ * before calling again. Copying the surviving rules out here rather than
500
+ * yielding them lazily is deliberate twice over: it keeps the mask's own
501
+ * scratch buffer from being read after the next position overwrites it, and
502
+ * it avoids a generator on the hottest loop in the pass.
503
+ */
504
+ private rulesAt;
351
505
  private rulesForTokenType;
352
506
  /**
353
507
  * Get the number of currently registered rules (excludes phrase trie entries).
354
508
  */
355
509
  get ruleCount(): number;
510
+ /**
511
+ * Every registered rule with the shape it declared, for diagnostic display.
512
+ *
513
+ * Exposes what the index is actually working with: a rule that declares a
514
+ * shape is only tried where that shape can match, and one that declares none
515
+ * is tried at every position of every line. Which is which is invisible from
516
+ * the outside otherwise, and it is the difference between a package that
517
+ * costs the documents that use it and one that costs all of them, so the
518
+ * playground draws it.
519
+ *
520
+ * Returned in priority order, highest first, which is the order the
521
+ * normalizer tries them in.
522
+ */
523
+ private ruleShapesCache;
524
+ getRuleShapes(): Array<{
525
+ name: string;
526
+ priority: number;
527
+ shape: readonly RuleSlot[];
528
+ unshapedReason?: string;
529
+ indexedSlots: number;
530
+ }>;
531
+ /**
532
+ * How many rules could fire at each position of `tokens`, against the total.
533
+ *
534
+ * The point of the index is that most positions admit no rule at all, and
535
+ * this is what makes that visible rather than asserted.
536
+ */
537
+ getCandidateCounts(tokens: Token[]): number[];
356
538
  /**
357
539
  * Register a multi-word phrase for fusion into a single compound token.
358
540
  *
@@ -69,6 +69,43 @@ interface NormalizerMatch {
69
69
  */
70
70
  ruleName?: string;
71
71
  }
72
+ /**
73
+ * One position of a rule's leading shape, as a declarative constraint.
74
+ *
75
+ * A slot says what the token at that offset from the match position may be.
76
+ * Both fields are optional and an omitted one constrains nothing, so `{}` is a
77
+ * wildcard slot, useful for reaching past a position a rule does not care about
78
+ * to one it does.
79
+ *
80
+ * The exactness contract runs in one direction only, and it is the whole reason
81
+ * this is safe to adopt gradually. Declaring MORE than the rule can match costs
82
+ * a `match()` call that returns null, which is what happens today anyway.
83
+ * Declaring LESS makes the rule unreachable at the positions left out, which is
84
+ * a silent bug. So an incomplete shape (or none at all) is always correct, and
85
+ * only an over-narrow one is wrong. `NormalizerIndexFidelity.spec` checks every
86
+ * declaration against its rule's real behaviour.
87
+ */
88
+ interface RuleSlot {
89
+ /**
90
+ * Token types admitted at this slot. Omit when the type is unconstrained,
91
+ * or when the rule accepts so many that naming them filters nothing.
92
+ */
93
+ readonly types?: readonly string[];
94
+ /**
95
+ * Token values admitted at this slot, compared case-insensitively.
96
+ *
97
+ * This is the axis that separates rules sharing a start type. The
98
+ * call-fusion rules all begin at an `IDENT`, the commonest token in prose,
99
+ * so type alone leaves every one of them a candidate at every word; the word
100
+ * itself is what tells them apart, and each already owns that set as an
101
+ * exported constant.
102
+ *
103
+ * A rule whose own check is case-SENSITIVE (the stock ticker rule matches
104
+ * upper case only) may still declare its lower-cased words here: the index
105
+ * only ever over-approximates, and the rule's own check still runs.
106
+ */
107
+ readonly values?: readonly string[];
108
+ }
72
109
  /**
73
110
  * A pluggable normalization rule registered with the TokenNormalizer.
74
111
  *
@@ -139,6 +176,60 @@ interface NormalizerRule {
139
176
  * the rule silently unreachable there, which is a bug.
140
177
  */
141
178
  readonly startTokenTypes?: readonly string[];
179
+ /**
180
+ * The rule's leading shape: what the tokens from the match position onward
181
+ * may be, one {@link RuleSlot} per position.
182
+ *
183
+ * This generalises {@link startTokenTypes}, which constrains only the first
184
+ * token. Constraining the first token alone is not enough to separate the
185
+ * rules that matter: every rule firing on a bare `NUMBER` declares the same
186
+ * start type, so they all remain candidates at every number in the document.
187
+ * What distinguishes them is the token after it, `NUMBER COLON` being a clock
188
+ * time and `NUMBER SLASH` a network address, and that fact is only usable by
189
+ * an index if the rule states it rather than hiding it inside `match()`.
190
+ *
191
+ * Depth is the rule's choice, not the interface's. The normalizer builds one
192
+ * lookup plane per declared slot and intersects them, so a rule that declares
193
+ * three positions is filtered on three. It may also index fewer planes than
194
+ * were declared, which stays correct for the reason given on {@link RuleSlot}:
195
+ * a shallower filter admits more candidates, and each surviving rule still
196
+ * runs its own `match()`.
197
+ *
198
+ * Prefer this to {@link startTokenTypes} in new rules. When both are given,
199
+ * this wins; `startTokenTypes: ["IDENT"]` means exactly `shape: [{ types:
200
+ * ["IDENT"] }]`.
201
+ *
202
+ * @example
203
+ * ```ts
204
+ * // 9:00am, 16:00, a clock time is a number followed by a colon
205
+ * shape: [{ types: ["NUMBER"] }, { types: ["COLON"] }]
206
+ *
207
+ * // sha256("hi"), a known word followed by an opening parenthesis
208
+ * shape: [{ types: ["IDENT"], values: HASH_NAMES }, { types: ["LPAREN"] }]
209
+ * ```
210
+ */
211
+ readonly shape?: readonly RuleSlot[];
212
+ /**
213
+ * Why this rule cannot declare a {@link shape}, for the few that genuinely
214
+ * cannot.
215
+ *
216
+ * A rule with neither a shape nor a `startTokenTypes` hint is tried at every
217
+ * position of every line, so it raises the cost of the whole document rather
218
+ * than only its own feature. Registering one logs a warning naming the rule,
219
+ * which is how a package author finds out before their users do.
220
+ *
221
+ * Some rules really cannot be described by a leading shape: an unbounded
222
+ * forward scan, a greedy match against a table the host mutates at runtime.
223
+ * Setting this states that case, silences the warning, and leaves the reason
224
+ * in the code where the next person will read it. It is deliberately a
225
+ * sentence and not a boolean, because "why" is the part worth keeping.
226
+ *
227
+ * @example
228
+ * ```ts
229
+ * unshapedReason: "Scans forward an unbounded number of NUMBER UNIT pairs, so no fixed leading shape describes it.",
230
+ * ```
231
+ */
232
+ readonly unshapedReason?: string;
142
233
  /**
143
234
  * Attempt to match a pattern starting at position `pos` in the token stream.
144
235
  *
@@ -228,6 +319,19 @@ interface NormalizerOptions {
228
319
  * When `undefined`, fusions are still tracked internally but no callbacks fire.
229
320
  */
230
321
  onFusion?: (fusion: TokenFusion) => void;
322
+ /**
323
+ * Try every registered rule at every position, ignoring the shape index.
324
+ *
325
+ * Diagnostic only, and much slower. It exists so the indexed walk can be
326
+ * compared against the unindexed one over a corpus: the index is a pure
327
+ * filter, so the two must agree token for token, and a rule whose declared
328
+ * {@link NormalizerRule.shape} is too narrow shows up as a difference rather
329
+ * than as a feature that quietly stopped working.
330
+ * `NormalizerIndexFidelity.spec` is the consumer.
331
+ *
332
+ * @default false
333
+ */
334
+ ignoreRuleIndex?: boolean;
231
335
  }
232
336
  /**
233
337
  * Creates a new normalized token from fused source tokens.
@@ -298,6 +402,28 @@ declare class TokenNormalizer {
298
402
  * identifier. Invalidated alongside {@link sortedRulesCache}.
299
403
  */
300
404
  private rulesByTokenType;
405
+ /**
406
+ * Shape index over the priority-sorted rules, rebuilt alongside
407
+ * {@link sortedRulesCache}. `null` means "stale, rebuild on next use".
408
+ *
409
+ * This is what turns the per-position scan from "try every rule that could
410
+ * fire on this token type" into "AND a few lookup planes and try what
411
+ * survives", which is usually nothing. See {@link RuleIndex}.
412
+ */
413
+ private ruleIndexCache;
414
+ /** Reused by {@link rulesAt} so a position's candidate list costs no allocation. */
415
+ private candidateBuffer;
416
+ /**
417
+ * How many times {@link normalize} has run, capped once the index is in use.
418
+ *
419
+ * The index costs a few tens of microseconds to build and pays that back
420
+ * within a line or two, but an engine that normalises exactly once, which is
421
+ * what evaluating a single expression on a fresh engine does, would never
422
+ * reach the payback. Skipping the build on the first call keeps that case at
423
+ * the cost it had before the index existed, and a document reaches line two
424
+ * immediately.
425
+ */
426
+ private normalizeCalls;
301
427
  /**
302
428
  * Phrase trie for single-pass multi-word phrase fusion.
303
429
  * Tried at each token position BEFORE other rules, the trie walk
@@ -348,11 +474,67 @@ declare class TokenNormalizer {
348
474
  * and cached, which is what turns the per-position rule scan from "try all R
349
475
  * rules" into "try only the ones that could fire on this token".
350
476
  */
477
+ private getRuleIndex;
478
+ /**
479
+ * The rules to try at a position, in priority order.
480
+ *
481
+ * With a candidate mask, walks its set bits low to high, which is descending
482
+ * priority because bit `i` is the rule at index `i` of the priority-sorted
483
+ * list.
484
+ *
485
+ * The walk shifts a bit at a time rather than jumping to the next set bit
486
+ * with `Math.clz32(bits & -bits)`, which is the idiomatic form and was 36x
487
+ * SLOWER here: 45us per line against 1.2us, measured over the built-in rule
488
+ * set. `clz32` is specified on uint32 and the masks come out of a
489
+ * `Uint32Array` as doubles above 2^31, so every call pays a conversion that
490
+ * swamps the handful of iterations it saves. A de Bruijn table matched the
491
+ * shift scan to within noise and needs a magic constant, so the plain shift
492
+ * wins on both counts. Worst case is 32 iterations per word of pure integer
493
+ * work.
494
+ *
495
+ * With `null` (the {@link NormalizerOptions.ignoreRuleIndex} path) it falls
496
+ * back to the type-bucketed list, which is the behaviour this replaced.
497
+ *
498
+ * Returns a buffer reused across positions, so the caller must finish with it
499
+ * before calling again. Copying the surviving rules out here rather than
500
+ * yielding them lazily is deliberate twice over: it keeps the mask's own
501
+ * scratch buffer from being read after the next position overwrites it, and
502
+ * it avoids a generator on the hottest loop in the pass.
503
+ */
504
+ private rulesAt;
351
505
  private rulesForTokenType;
352
506
  /**
353
507
  * Get the number of currently registered rules (excludes phrase trie entries).
354
508
  */
355
509
  get ruleCount(): number;
510
+ /**
511
+ * Every registered rule with the shape it declared, for diagnostic display.
512
+ *
513
+ * Exposes what the index is actually working with: a rule that declares a
514
+ * shape is only tried where that shape can match, and one that declares none
515
+ * is tried at every position of every line. Which is which is invisible from
516
+ * the outside otherwise, and it is the difference between a package that
517
+ * costs the documents that use it and one that costs all of them, so the
518
+ * playground draws it.
519
+ *
520
+ * Returned in priority order, highest first, which is the order the
521
+ * normalizer tries them in.
522
+ */
523
+ private ruleShapesCache;
524
+ getRuleShapes(): Array<{
525
+ name: string;
526
+ priority: number;
527
+ shape: readonly RuleSlot[];
528
+ unshapedReason?: string;
529
+ indexedSlots: number;
530
+ }>;
531
+ /**
532
+ * How many rules could fire at each position of `tokens`, against the total.
533
+ *
534
+ * The point of the index is that most positions admit no rule at all, and
535
+ * this is what makes that visible rather than asserted.
536
+ */
537
+ getCandidateCounts(tokens: Token[]): number[];
356
538
  /**
357
539
  * Register a multi-word phrase for fusion into a single compound token.
358
540
  *
@@ -1,6 +1,6 @@
1
1
  import { V as Value } from './Value-DCTqTSeP.js';
2
- import { V as VM } from './ScopeManager-C6c1WmJD.js';
3
- import { U as UserFunctionDef } from './BytecodeBuilder-CRYfrFfq.js';
2
+ import { V as VM } from './ScopeManager-gB9UengK.js';
3
+ import { U as UserFunctionDef } from './BytecodeBuilder-aqVa7Plx.js';
4
4
 
5
5
  /**
6
6
  * A point-in-time snapshot of VM variable state.
@@ -1,6 +1,6 @@
1
1
  import { V as Value } from './Value-DCTqTSeP.cjs';
2
- import { V as VM } from './ScopeManager-BtqiTVjG.cjs';
3
- import { U as UserFunctionDef } from './BytecodeBuilder-CRYfrFfq.cjs';
2
+ import { V as VM } from './ScopeManager-bCYewWVt.cjs';
3
+ import { U as UserFunctionDef } from './BytecodeBuilder-aqVa7Plx.cjs';
4
4
 
5
5
  /**
6
6
  * A point-in-time snapshot of VM variable state.
@@ -1,4 +1,4 @@
1
- import { c as ErrorCategory, S as SourceSpan, E as EngineError } from './EngineError-Cv5q4Rbv.cjs';
1
+ import { c as ErrorCategory, S as SourceSpan, E as EngineError } from './EngineError-DTk7I7hZ.cjs';
2
2
 
3
3
  /**
4
4
  * Structured errors for the off-main-thread worker harness, and the pair of
@@ -1,4 +1,4 @@
1
- import { c as ErrorCategory, S as SourceSpan, E as EngineError } from './EngineError-Cv5q4Rbv.js';
1
+ import { c as ErrorCategory, S as SourceSpan, E as EngineError } from './EngineError-DTk7I7hZ.js';
2
2
 
3
3
  /**
4
4
  * Structured errors for the off-main-thread worker harness, and the pair of
@@ -1,2 +1,2 @@
1
- import {d}from'./chunk-JPFCXLJV.js';import {b}from'./chunk-TVE2DNPU.js';import {a}from'./chunk-LHLW7JLS.js';import {a as a$1}from'./chunk-UWAXJX5G.js';var u=(s=>(s.Main="main",s.Inline="inline",s.String="string",s))(u||{});var l=class{constructor(e="en",t){this.currentState="main";this.hasPeeked=false;this.tokens=[];this.tokenIdx=0;this.expressionLexer=new b(e,t);}reset(e,t){let s=t??"main";if(this.currentState=s,this.hasPeeked=false,this.peekedToken=void 0,s==="main"){if(this.expressionLexer.classifyLine(e).skip){this.tokens=[],this.tokenIdx=0;return}this.expressionLexer.reset(e),this.tokens=this.expressionLexer.tokenizeAll(),this.tokenIdx=0;}else this.expressionLexer.reset(e),this.tokens=this.expressionLexer.tokenizeAll(),this.tokenIdx=0;}classifyLine(e){return this.expressionLexer.classifyLine(e)}findInlineSolves(e){return this.expressionLexer.findInlineSolves(e)}getKeywords(){return this.expressionLexer.getKeywords()}next(){if(this.hasPeeked)return this.hasPeeked=false,this.peekedToken;if(this.tokenIdx<this.tokens.length)return this.tokens[this.tokenIdx++]}peek(){return this.hasPeeked?this.peekedToken:(this.peekedToken=this.next(),this.hasPeeked=true,this.peekedToken)}[Symbol.iterator](){return this.tokens[Symbol.iterator]()}registerVocabulary(e){this.expressionLexer.registerVocabulary(e);}unregisterVocabulary(e){this.expressionLexer.unregisterVocabulary(e);}resetExpression(e){this.currentState="main",this.hasPeeked=false,this.peekedToken=void 0,this.expressionLexer.reset(e),this.tokens=this.expressionLexer.tokenizeAll(),this.tokenIdx=0;}scanDocument(e){return this.expressionLexer.scanDocument(e)}getState(){return this.currentState}setState(e){this.currentState=e;}getHighlightTokens(e){let t=this.expressionLexer.classifyLine(e);return t.skip&&e.startsWith("> ")?this.collectHighlightTokens(e.slice(2)):t.skip?[]:this.collectHighlightTokens(e)}getHighlightTokenObjects(e){let t=this.expressionLexer.classifyLine(e);return t.skip&&e.startsWith("> ")?this.collectTokenObjects(e.slice(2)):t.skip?[]:this.collectTokenObjects(e)}collectTokenObjects(e){this.resetExpression(e);let t=[];for(let s of this)s.type==="WS"||s.type==="NEWLINE"||s.type.startsWith("MD_")||s.type==="INLINE_SOLVE_START"||s.type==="BACKTICK_CLOSE"||t.push(s);return t}collectHighlightTokens(e){return this.collectTokenObjects(e).map(t=>({type:t.type,value:t.value,offset:t.offset,col:t.col,length:t.text.length,category:d(t.type)}))}},T=new l("en",void 0);var n=class{constructor(){this.classes=[];this.localeKeywordMap=null;this.localePhraseMap=null;this.unitNames=null;}register(e){this.classes.push(e);}unregister(e){this.classes=this.classes.filter(t=>t.tokenType!==e);}setLocale(e,t){this.localeKeywordMap=e,this.localePhraseMap=t??null;}setUnits(e){this.unitNames=e;}build(){let e=new Map;if(this.localeKeywordMap)for(let[r,i]of Object.entries(this.localeKeywordMap))e.set(r.toLowerCase(),i);let t=[...this.classes].sort((r,i)=>(r.priority??0)-(i.priority??0));for(let r of t)for(let i of Object.keys(r.keywords))e.set(i.toLowerCase(),r.tokenType);let s=this.buildPhraseTrie(),o=this.buildPhraseStartWords();return {keywordToType:e,phraseTrie:s,phraseStartWords:o,unitNames:this.unitNames??new Set}}buildPhraseTrie(){let e={children:new Map};if(this.localePhraseMap)for(let[t,s]of Object.entries(this.localePhraseMap))this.insertPhrase(e,t.toLowerCase(),s);for(let t of this.classes)if(t.phrases)for(let s of Object.keys(t.phrases))this.insertPhrase(e,s.toLowerCase(),t.tokenType);return e.children.size>0?e:null}insertPhrase(e,t,s){let o=t.split(" "),r=e;for(let i of o)r.children.has(i)||r.children.set(i,{children:new Map}),r=r.children.get(i);r.type||(r.type=s);}buildPhraseStartWords(){let e=new Set;if(this.localePhraseMap)for(let t of Object.keys(this.localePhraseMap)){let s=t.split(" ")[0].toLowerCase();e.add(s);}for(let t of this.classes)if(t.phrases)for(let s of Object.keys(t.phrases)){let o=s.split(" ")[0].toLowerCase();e.add(o);}return e}};var k={"to the power of":"CARET","power of":"CARET","increase by":"INCREASE_BY","decrease by":"DECREASE_BY","times by":"TIMES_BY","multiply by":"MULTIPLY_BY","multiplied by":"MULTIPLY_BY","divide by":"DIVIDE_BY"};function P(a$2="en"){let e=a(a$2),t=new n;return t.setLocale(e.keywordMap,k),t.setUnits(a$1),t.build()}export{u as a,l as b,T as c,P as d};//# sourceMappingURL=chunk-UZSPBFZN.js.map
2
- //# sourceMappingURL=chunk-UZSPBFZN.js.map
1
+ import {d}from'./chunk-JPFCXLJV.js';import {b}from'./chunk-LJFS3XHW.js';import {a}from'./chunk-LHLW7JLS.js';import {a as a$1}from'./chunk-UWAXJX5G.js';var u=(s=>(s.Main="main",s.Inline="inline",s.String="string",s))(u||{});var l=class{constructor(e="en",t){this.currentState="main";this.hasPeeked=false;this.tokens=[];this.tokenIdx=0;this.expressionLexer=new b(e,t);}reset(e,t){let s=t??"main";if(this.currentState=s,this.hasPeeked=false,this.peekedToken=void 0,s==="main"){if(this.expressionLexer.classifyLine(e).skip){this.tokens=[],this.tokenIdx=0;return}this.expressionLexer.reset(e),this.tokens=this.expressionLexer.tokenizeAll(),this.tokenIdx=0;}else this.expressionLexer.reset(e),this.tokens=this.expressionLexer.tokenizeAll(),this.tokenIdx=0;}classifyLine(e){return this.expressionLexer.classifyLine(e)}findInlineSolves(e){return this.expressionLexer.findInlineSolves(e)}getKeywords(){return this.expressionLexer.getKeywords()}next(){if(this.hasPeeked)return this.hasPeeked=false,this.peekedToken;if(this.tokenIdx<this.tokens.length)return this.tokens[this.tokenIdx++]}peek(){return this.hasPeeked?this.peekedToken:(this.peekedToken=this.next(),this.hasPeeked=true,this.peekedToken)}[Symbol.iterator](){return this.tokens[Symbol.iterator]()}registerVocabulary(e){this.expressionLexer.registerVocabulary(e);}unregisterVocabulary(e){this.expressionLexer.unregisterVocabulary(e);}resetExpression(e){this.currentState="main",this.hasPeeked=false,this.peekedToken=void 0,this.expressionLexer.reset(e),this.tokens=this.expressionLexer.tokenizeAll(),this.tokenIdx=0;}scanDocument(e){return this.expressionLexer.scanDocument(e)}getState(){return this.currentState}setState(e){this.currentState=e;}getHighlightTokens(e){let t=this.expressionLexer.classifyLine(e);return t.skip&&e.startsWith("> ")?this.collectHighlightTokens(e.slice(2)):t.skip?[]:this.collectHighlightTokens(e)}getHighlightTokenObjects(e){let t=this.expressionLexer.classifyLine(e);return t.skip&&e.startsWith("> ")?this.collectTokenObjects(e.slice(2)):t.skip?[]:this.collectTokenObjects(e)}collectTokenObjects(e){this.resetExpression(e);let t=[];for(let s of this)s.type==="WS"||s.type==="NEWLINE"||s.type.startsWith("MD_")||s.type==="INLINE_SOLVE_START"||s.type==="BACKTICK_CLOSE"||t.push(s);return t}collectHighlightTokens(e){return this.collectTokenObjects(e).map(t=>({type:t.type,value:t.value,offset:t.offset,col:t.col,length:t.text.length,category:d(t.type)}))}},T=new l("en",void 0);var n=class{constructor(){this.classes=[];this.localeKeywordMap=null;this.localePhraseMap=null;this.unitNames=null;}register(e){this.classes.push(e);}unregister(e){this.classes=this.classes.filter(t=>t.tokenType!==e);}setLocale(e,t){this.localeKeywordMap=e,this.localePhraseMap=t??null;}setUnits(e){this.unitNames=e;}build(){let e=new Map;if(this.localeKeywordMap)for(let[r,i]of Object.entries(this.localeKeywordMap))e.set(r.toLowerCase(),i);let t=[...this.classes].sort((r,i)=>(r.priority??0)-(i.priority??0));for(let r of t)for(let i of Object.keys(r.keywords))e.set(i.toLowerCase(),r.tokenType);let s=this.buildPhraseTrie(),o=this.buildPhraseStartWords();return {keywordToType:e,phraseTrie:s,phraseStartWords:o,unitNames:this.unitNames??new Set}}buildPhraseTrie(){let e={children:new Map};if(this.localePhraseMap)for(let[t,s]of Object.entries(this.localePhraseMap))this.insertPhrase(e,t.toLowerCase(),s);for(let t of this.classes)if(t.phrases)for(let s of Object.keys(t.phrases))this.insertPhrase(e,s.toLowerCase(),t.tokenType);return e.children.size>0?e:null}insertPhrase(e,t,s){let o=t.split(" "),r=e;for(let i of o)r.children.has(i)||r.children.set(i,{children:new Map}),r=r.children.get(i);r.type||(r.type=s);}buildPhraseStartWords(){let e=new Set;if(this.localePhraseMap)for(let t of Object.keys(this.localePhraseMap)){let s=t.split(" ")[0].toLowerCase();e.add(s);}for(let t of this.classes)if(t.phrases)for(let s of Object.keys(t.phrases)){let o=s.split(" ")[0].toLowerCase();e.add(o);}return e}};var k={"to the power of":"CARET","power of":"CARET","increase by":"INCREASE_BY","decrease by":"DECREASE_BY","times by":"TIMES_BY","multiply by":"MULTIPLY_BY","multiplied by":"MULTIPLY_BY","divide by":"DIVIDE_BY"};function P(a$2="en"){let e=a(a$2),t=new n;return t.setLocale(e.keywordMap,k),t.setUnits(a$1),t.build()}export{u as a,l as b,T as c,P as d};//# sourceMappingURL=chunk-2DPKJ2SM.js.map
2
+ //# sourceMappingURL=chunk-2DPKJ2SM.js.map
@@ -1 +1 @@
1
- {"version":3,"sources":["../src/lexer/LexerState.ts","../src/lexer/Lexer.ts","../src/lexer/TokenClassRegistry.ts","../src/lexer/tokenRegistration.ts"],"names":["LexerState","Lexer","localeCode","tokenLookup","ExpressionLexer","input","state","newState","lineText","plugin","text","classification","result","token","getTokenCategory","sharedLexer","TokenClassRegistry","tokenClass","tokenType","c","keywordMap","phraseMap","unitNames","keywordToType","keyword","sorted","a","b","tc","phraseTrie","phraseStartWords","root","phrase","words","node","word","startWords","first","BUILTIN_PHRASES","buildTokenLookup","locale","getLocale","registry","knownUnits"],"mappings":"uJAMO,IAAKA,CAAAA,CAAAA,CAAAA,CAAAA,GACXA,EAAA,IAAA,CAAO,MAAA,CACPA,EAAA,MAAA,CAAS,QAAA,CACTA,EAAA,MAAA,CAAS,QAAA,CAHEA,OAAA,EAAA,ECaL,IAAMC,EAAN,KAAY,CAkBjB,YAAYC,CAAAA,CAAa,IAAA,CAAMC,EAA2B,CAf1D,IAAA,CAAQ,aAA2B,MAAA,CAEnC,IAAA,CAAQ,UAAY,KAAA,CAIpB,IAAA,CAAQ,OAAkB,EAAC,CAC3B,KAAQ,QAAA,CAAmB,CAAA,CAYzB,KAAK,eAAA,CAAkB,IAAIC,EAAgBF,CAAAA,CAAYC,CAAW,EACpE,CAEA,KAAA,CAAME,EAAeC,CAAAA,CAA0B,CAC7C,IAAMC,CAAAA,CAAWD,CAAAA,EAAS,OAQ1B,GAPA,IAAA,CAAK,aAAeC,CAAAA,CACpB,IAAA,CAAK,UAAY,KAAA,CACjB,IAAA,CAAK,YAAc,MAAA,CAKfA,CAAAA,GAAa,OAAiB,CAEhC,GADuB,KAAK,eAAA,CAAgB,YAAA,CAAaF,CAAK,CAAA,CAC3C,IAAA,CAAM,CACvB,IAAA,CAAK,MAAA,CAAS,EAAC,CACf,IAAA,CAAK,SAAW,CAAA,CAChB,MACF,CAEA,IAAA,CAAK,eAAA,CAAgB,MAAMA,CAAK,CAAA,CAChC,KAAK,MAAA,CAAS,IAAA,CAAK,gBAAgB,WAAA,EAAY,CAC/C,KAAK,QAAA,CAAW,EAClB,MAEE,IAAA,CAAK,eAAA,CAAgB,MAAMA,CAAK,CAAA,CAChC,KAAK,MAAA,CAAS,IAAA,CAAK,gBAAgB,WAAA,EAAY,CAC/C,KAAK,QAAA,CAAW,EAEpB,CAMA,YAAA,CAAaG,CAAAA,CAAsC,CACjD,OAAO,IAAA,CAAK,gBAAgB,YAAA,CAAaA,CAAQ,CACnD,CAMA,gBAAA,CAAiBA,EAAkB,CACjC,OAAO,KAAK,eAAA,CAAgB,gBAAA,CAAiBA,CAAQ,CACvD,CAMA,aAAsC,CACpC,OAAO,KAAK,eAAA,CAAgB,WAAA,EAC9B,CAEA,IAAA,EAA0B,CACxB,GAAI,IAAA,CAAK,UACP,OAAA,IAAA,CAAK,SAAA,CAAY,MACV,IAAA,CAAK,WAAA,CAGd,GAAI,IAAA,CAAK,QAAA,CAAW,KAAK,MAAA,CAAO,MAAA,CAC9B,OAAO,IAAA,CAAK,MAAA,CAAO,KAAK,QAAA,EAAU,CAGtC,CAEA,IAAA,EAA0B,CACxB,OAAI,IAAA,CAAK,SAAA,CAAkB,KAAK,WAAA,EAChC,IAAA,CAAK,YAAc,IAAA,CAAK,IAAA,GACxB,IAAA,CAAK,SAAA,CAAY,KACV,IAAA,CAAK,WAAA,CACd,CAEA,CAAC,MAAA,CAAO,QAAQ,CAAA,EAAqB,CACnC,OAAO,IAAA,CAAK,MAAA,CAAO,MAAA,CAAO,QAAQ,CAAA,EACpC,CAQA,kBAAA,CAAmBC,CAAAA,CAA+B,CAChD,IAAA,CAAK,eAAA,CAAgB,mBAAmBA,CAAM,EAChD,CAMA,oBAAA,CAAqBA,CAAAA,CAA+B,CAClD,IAAA,CAAK,eAAA,CAAgB,qBAAqBA,CAAM,EAClD,CAOA,eAAA,CAAgBJ,CAAAA,CAAqB,CACnC,IAAA,CAAK,YAAA,CAAe,OACpB,IAAA,CAAK,SAAA,CAAY,MACjB,IAAA,CAAK,WAAA,CAAc,OACnB,IAAA,CAAK,eAAA,CAAgB,MAAMA,CAAK,CAAA,CAChC,KAAK,MAAA,CAAS,IAAA,CAAK,gBAAgB,WAAA,EAAY,CAC/C,KAAK,QAAA,CAAW,EAClB,CAQA,YAAA,CAAaK,CAAAA,CAAgC,CAC3C,OAAO,IAAA,CAAK,gBAAgB,YAAA,CAAaA,CAAI,CAC/C,CAEA,QAAA,EAAuB,CACrB,OAAO,IAAA,CAAK,YACd,CAEA,QAAA,CAASJ,EAAyB,CAChC,IAAA,CAAK,aAAeA,EACtB,CAEA,mBAAmBE,CAAAA,CAAqI,CACtJ,IAAMG,CAAAA,CAAiB,IAAA,CAAK,gBAAgB,YAAA,CAAaH,CAAQ,EAKjE,OAAIG,CAAAA,CAAe,MAAQH,CAAAA,CAAS,UAAA,CAAW,IAAI,CAAA,CAC1C,IAAA,CAAK,uBAAuBA,CAAAA,CAAS,KAAA,CAAM,CAAC,CAAC,CAAA,CAGlDG,EAAe,IAAA,CACV,GAGF,IAAA,CAAK,sBAAA,CAAuBH,CAAQ,CAC7C,CAYA,yBAAyBA,CAAAA,CAA2B,CAClD,IAAMG,CAAAA,CAAiB,IAAA,CAAK,gBAAgB,YAAA,CAAaH,CAAQ,EACjE,OAAIG,CAAAA,CAAe,MAAQH,CAAAA,CAAS,UAAA,CAAW,IAAI,CAAA,CAC1C,IAAA,CAAK,oBAAoBA,CAAAA,CAAS,KAAA,CAAM,CAAC,CAAC,CAAA,CAE/CG,EAAe,IAAA,CAAa,GACzB,IAAA,CAAK,mBAAA,CAAoBH,CAAQ,CAC1C,CAEQ,oBAAoBA,CAAAA,CAA2B,CACrD,KAAK,eAAA,CAAgBA,CAAQ,EAC7B,IAAMI,CAAAA,CAAkB,EAAC,CACzB,IAAA,IAAWC,KAAS,IAAA,CACdA,CAAAA,CAAM,OAAS,IAAA,EAAQA,CAAAA,CAAM,OAAS,SAAA,EACtCA,CAAAA,CAAM,KAAK,UAAA,CAAW,KAAK,GAC3BA,CAAAA,CAAM,IAAA,GAAS,sBAAwBA,CAAAA,CAAM,IAAA,GAAS,kBAC1DD,CAAAA,CAAO,IAAA,CAAKC,CAAK,CAAA,CAEnB,OAAOD,CACT,CAEQ,sBAAA,CAAuBJ,EAAqI,CAClK,OAAO,KAAK,mBAAA,CAAoBA,CAAQ,EAAE,GAAA,CAAIK,CAAAA,GAAU,CACtD,IAAA,CAAMA,CAAAA,CAAM,KACZ,KAAA,CAAOA,CAAAA,CAAM,KAAA,CACb,MAAA,CAAQA,CAAAA,CAAM,MAAA,CACd,IAAKA,CAAAA,CAAM,GAAA,CAIX,OAAQA,CAAAA,CAAM,IAAA,CAAK,OACnB,QAAA,CAAUC,CAAAA,CAAiBD,EAAM,IAAI,CACvC,EAAE,CACJ,CACF,EAcaE,CAAAA,CAAc,IAAId,EAAM,IAAA,CAAM,MAAS,EClJ7C,IAAMe,CAAAA,CAAN,KAAyB,CAAzB,WAAA,EAAA,CACL,KAAQ,OAAA,CAAwB,GAChC,IAAA,CAAQ,gBAAA,CAAkD,KAC1D,IAAA,CAAQ,eAAA,CAAiD,KACzD,IAAA,CAAQ,SAAA,CAAwC,MAUhD,QAAA,CAASC,CAAAA,CAA8B,CACrC,IAAA,CAAK,OAAA,CAAQ,KAAKA,CAAU,EAC9B,CAMA,UAAA,CAAWC,CAAAA,CAAyB,CAClC,IAAA,CAAK,OAAA,CAAU,KAAK,OAAA,CAAQ,MAAA,CAAOC,GAAKA,CAAAA,CAAE,SAAA,GAAcD,CAAS,EACnE,CAMA,UAAUE,CAAAA,CAAoCC,CAAAA,CAA0C,CACtF,IAAA,CAAK,gBAAA,CAAmBD,EACxB,IAAA,CAAK,eAAA,CAAkBC,GAAa,KACtC,CASA,SAASC,CAAAA,CAAsC,CAC7C,KAAK,SAAA,CAAYA,EACnB,CAYA,KAAA,EAAqB,CACnB,IAAMC,CAAAA,CAAgB,IAAI,IAG1B,GAAI,IAAA,CAAK,iBACP,IAAA,GAAW,CAACC,EAASN,CAAS,CAAA,GAAK,OAAO,OAAA,CAAQ,IAAA,CAAK,gBAAgB,CAAA,CACrEK,CAAAA,CAAc,IAAIC,CAAAA,CAAQ,WAAA,GAAeN,CAAS,CAAA,CAKtD,IAAMO,CAAAA,CAAS,CAAC,GAAG,IAAA,CAAK,OAAO,EAAE,IAAA,CAAK,CAACC,EAAGC,CAAAA,GAAAA,CAAOD,CAAAA,CAAE,UAAY,CAAA,GAAMC,CAAAA,CAAE,UAAY,CAAA,CAAE,CAAA,CACrF,QAAWC,CAAAA,IAAMH,CAAAA,CACf,QAAWD,CAAAA,IAAW,MAAA,CAAO,KAAKI,CAAAA,CAAG,QAAQ,EAC3CL,CAAAA,CAAc,GAAA,CAAIC,EAAQ,WAAA,EAAY,CAAGI,EAAG,SAAS,CAAA,CAKzD,IAAMC,CAAAA,CAAa,IAAA,CAAK,iBAAgB,CAGlCC,CAAAA,CAAmB,KAAK,qBAAA,EAAsB,CAEpD,OAAO,CACL,aAAA,CAAAP,EACA,UAAA,CAAAM,CAAAA,CACA,iBAAAC,CAAAA,CACA,SAAA,CAAW,KAAK,SAAA,EAAa,IAAI,GACnC,CACF,CAIQ,iBAAqC,CAC3C,IAAMC,EAAmB,CAAE,QAAA,CAAU,IAAI,GAAM,CAAA,CAG/C,GAAI,IAAA,CAAK,eAAA,CACP,OAAW,CAACC,CAAAA,CAAQd,CAAS,CAAA,GAAK,MAAA,CAAO,QAAQ,IAAA,CAAK,eAAe,CAAA,CACnE,IAAA,CAAK,YAAA,CAAaa,CAAAA,CAAMC,EAAO,WAAA,EAAY,CAAGd,CAAS,CAAA,CAK3D,IAAA,IAAWU,KAAM,IAAA,CAAK,OAAA,CACpB,GAAKA,CAAAA,CAAG,OAAA,CACR,QAAWI,CAAAA,IAAU,MAAA,CAAO,KAAKJ,CAAAA,CAAG,OAAO,EACzC,IAAA,CAAK,YAAA,CAAaG,EAAMC,CAAAA,CAAO,WAAA,GAAeJ,CAAAA,CAAG,SAAS,EAI9D,OAAOG,CAAAA,CAAK,SAAS,IAAA,CAAO,CAAA,CAAIA,EAAO,IACzC,CAEQ,aAAaA,CAAAA,CAAkBC,CAAAA,CAAgBd,EAAyB,CAC9E,IAAMe,EAAQD,CAAAA,CAAO,KAAA,CAAM,GAAG,CAAA,CAC1BE,CAAAA,CAAOH,EACX,IAAA,IAAWI,CAAAA,IAAQF,EACZC,CAAAA,CAAK,QAAA,CAAS,IAAIC,CAAI,CAAA,EACzBD,EAAK,QAAA,CAAS,GAAA,CAAIC,EAAM,CAAE,QAAA,CAAU,IAAI,GAAM,CAAC,EAEjDD,CAAAA,CAAOA,CAAAA,CAAK,SAAS,GAAA,CAAIC,CAAI,EAG1BD,CAAAA,CAAK,IAAA,GACRA,EAAK,IAAA,CAAOhB,CAAAA,EAEhB,CAEQ,qBAAA,EAAqC,CAC3C,IAAMkB,CAAAA,CAAa,IAAI,IAGvB,GAAI,IAAA,CAAK,gBACP,IAAA,IAAWJ,CAAAA,IAAU,OAAO,IAAA,CAAK,IAAA,CAAK,eAAe,CAAA,CAAG,CACtD,IAAMK,CAAAA,CAAQL,CAAAA,CAAO,MAAM,GAAG,CAAA,CAAE,CAAC,CAAA,CAAE,WAAA,GACnCI,CAAAA,CAAW,GAAA,CAAIC,CAAK,EACtB,CAIF,QAAWT,CAAAA,IAAM,IAAA,CAAK,QACpB,GAAKA,CAAAA,CAAG,QACR,IAAA,IAAWI,CAAAA,IAAU,OAAO,IAAA,CAAKJ,CAAAA,CAAG,OAAO,CAAA,CAAG,CAC5C,IAAMS,CAAAA,CAAQL,CAAAA,CAAO,MAAM,GAAG,CAAA,CAAE,CAAC,CAAA,CAAE,WAAA,GACnCI,CAAAA,CAAW,GAAA,CAAIC,CAAK,EACtB,CAGF,OAAOD,CACT,CACF,EClOA,IAAME,CAAAA,CAA0C,CAC9C,iBAAA,CAAmB,OAAA,CACnB,WAAY,OAAA,CACZ,aAAA,CAAe,cACf,aAAA,CAAe,aAAA,CACf,WAAY,UAAA,CACZ,aAAA,CAAe,cACf,eAAA,CAAiB,aAAA,CACjB,YAAa,WACf,CAAA,CAiBO,SAASC,CAAAA,CAAiBrC,GAAAA,CAAa,KAAmB,CAC/D,IAAMsC,EAAkBC,CAAAA,CAAUvC,GAAU,EACtCwC,CAAAA,CAAW,IAAI1B,EAGrB,OAAA0B,CAAAA,CAAS,UAAUF,CAAAA,CAAO,UAAA,CAAYF,CAAe,CAAA,CAGrDI,CAAAA,CAAS,SAASC,GAAU,CAAA,CAGrBD,CAAAA,CAAS,KAAA,EAClB","file":"chunk-UZSPBFZN.js","sourcesContent":["/**\n * Lexer state machine modes.\n * - Main: document-level scanning with markdown classification\n * - Inline: expression embedded in markdown inline solve (`s\\`...\\``)\n * - String: inside a double-quoted string literal\n */\nexport enum LexerState {\n\tMain = \"main\",\n\tInline = \"inline\",\n\tString = \"string\",\n}\n","import { ExpressionLexer, LineClassification, LexerVocabulary, type ScanLineResult } from \"./ExpressionLexer\";\nimport { Token } from \"@solve-js/lexer/Token\";\nimport { LexerState } from \"@solve-js/lexer/LexerState\";\nimport { getTokenCategory } from \"@solve-js/language/TokenCategoryMap\";\nimport type { TokenCategory } from \"@solve-js/language/TokenCategory\";\nimport type { TokenLookup } from \"@solve-js/lexer/TokenClassRegistry\";\n\n/**\n * Public tokenizer wrapper around {@link ExpressionLexer}.\n *\n * `ExpressionLexer` does the actual character-by-character scanning;\n * `Lexer` adds a materialized-token-array streaming interface\n * (`next()`/`peek()`) plus line-classification state (`reset()`) so\n * callers can iterate a line's tokens without re-scanning on each peek.\n *\n * Each `ExpressionEngine` instance owns its own `Lexer`, and packages\n * extend it via {@link registerVocabulary} (keywords, operators, units)\n * see `IEnginePackage.lexerVocabulary`.\n */\nexport class Lexer {\n /** Expression-mode lexer (Phase A: V8-optimized, replaces moo) */\n private expressionLexer: ExpressionLexer;\n private currentState: LexerState = LexerState.Main;\n private peekedToken: Token | undefined;\n private hasPeeked = false;\n\n // Materialized token array from the last reset() call, used for\n // next()/peek() streaming access.\n private tokens: Token[] = [];\n private tokenIdx: number = 0;\n\n /**\n * @param localeCode - Locale code (e.g., \"en\", \"de\"). Defaults to \"en\".\n * @param tokenLookup - Optional TokenLookup from TokenClassRegistry.\n * When provided, configures ExpressionLexer to use registry-built\n * keyword/unit/phrase lookups instead of internal instance maps.\n */\n constructor(localeCode = \"en\", tokenLookup?: TokenLookup) {\n // Pass the lookup directly to ExpressionLexer's constructor, it's an\n // instance field now, not a static. Each Lexer instance gets its own\n // isolated lookup, preventing cross-instance corruption.\n this.expressionLexer = new ExpressionLexer(localeCode, tokenLookup);\n }\n\n reset(input: string, state?: LexerState): void {\n const newState = state ?? LexerState.Main;\n this.currentState = newState;\n this.hasPeeked = false;\n this.peekedToken = undefined;\n\n // Phase B: Main state classifies the line with the markdown scanner.\n // Skip lines (headings, fences, HRs, etc.) produce empty token arrays.\n // Expression lines and lines with inline solves are tokenized normally.\n if (newState === LexerState.Main) {\n const classification = this.expressionLexer.classifyLine(input);\n if (classification.skip) {\n this.tokens = [];\n this.tokenIdx = 0;\n return;\n }\n // Expression line or markdown line with inline solves, tokenize.\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n } else {\n // Non-main states (Inline, String), expression tokenization.\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n }\n }\n\n /**\n * Classify a single line of markdown text (Phase B).\n * Delegates to the ExpressionLexer's character-by-character scanner.\n */\n classifyLine(lineText: string): LineClassification {\n return this.expressionLexer.classifyLine(lineText);\n }\n\n /**\n * Find all inline solve markers in a line (Phase B).\n * Delegates to the ExpressionLexer's character-by-character scanner.\n */\n findInlineSolves(lineText: string) {\n return this.expressionLexer.findInlineSolves(lineText);\n }\n\n /**\n * Every keyword this lexer currently recognizes (locale + plugin-contributed),\n * mapped to the token type it lexes to. Delegates to the ExpressionLexer.\n */\n getKeywords(): Record<string, string> {\n return this.expressionLexer.getKeywords();\n }\n\n next(): Token | undefined {\n if (this.hasPeeked) {\n this.hasPeeked = false;\n return this.peekedToken;\n }\n // Materialized token array (ExpressionLexer path).\n if (this.tokenIdx < this.tokens.length) {\n return this.tokens[this.tokenIdx++];\n }\n return undefined;\n }\n\n peek(): Token | undefined {\n if (this.hasPeeked) return this.peekedToken;\n this.peekedToken = this.next();\n this.hasPeeked = true;\n return this.peekedToken;\n }\n\n [Symbol.iterator](): Iterator<Token> {\n return this.tokens[Symbol.iterator]();\n }\n\n /**\n * Register a plugin to extend the lexer with custom tokens.\n * Delegates to the underlying ExpressionLexer.\n *\n * @see LexerVocabulary for the supported extension points.\n */\n registerVocabulary(plugin: LexerVocabulary): void {\n this.expressionLexer.registerVocabulary(plugin);\n }\n\n /**\n * Unregister a plugin, removing its custom tokens from the lexer.\n * Delegates to the underlying ExpressionLexer.\n */\n unregisterVocabulary(plugin: LexerVocabulary): void {\n this.expressionLexer.unregisterVocabulary(plugin);\n }\n\n /**\n * Reset the lexer for expression-only text, skips the classifyLine()\n * overhead in reset() for callers that already know the input is an\n * evaluable expression (e.g., after isEmptyLine() confirmed non-skip).\n */\n resetExpression(input: string): void {\n this.currentState = LexerState.Main;\n this.hasPeeked = false;\n this.peekedToken = undefined;\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n }\n\n /**\n * Scan a full document in one pass, classifying each line and\n * tokenizing non-skipped lines. Delegates to ExpressionLexer.\n *\n * @returns ScanLineResult[], one per line, with classification + tokens.\n */\n scanDocument(text: string): ScanLineResult[] {\n return this.expressionLexer.scanDocument(text);\n }\n\n getState(): LexerState {\n return this.currentState;\n }\n\n setState(state: LexerState): void {\n this.currentState = state;\n }\n\n getHighlightTokens(lineText: string): {type: string; value: string; offset: number; col: number; length: number; category: TokenCategory | undefined}[] {\n const classification = this.expressionLexer.classifyLine(lineText);\n\n // For blockquote lines, strip the \"> \" prefix and tokenize the expression content.\n // This lets expressions inside blockquotes (e.g., \"> 1 + 2\") get syntax highlighted\n // while pure structural lines (headings, code fences) remain unhighlighted.\n if (classification.skip && lineText.startsWith(\"> \")) {\n return this.collectHighlightTokens(lineText.slice(2));\n }\n\n if (classification.skip) {\n return [];\n }\n\n return this.collectHighlightTokens(lineText);\n }\n\n /**\n * The same tokens {@link getHighlightTokens} reduces, before reduction.\n *\n * Exists because normalization operates on tokens, not on the flattened\n * shape, and a consumer that wants phrase-fused highlighting has to run the\n * normalizer between the two. See `LanguageService.getSemanticTokens`.\n *\n * @param lineText - One line of source.\n * @returns Every token on the line that is worth painting, unreduced.\n */\n getHighlightTokenObjects(lineText: string): Token[] {\n const classification = this.expressionLexer.classifyLine(lineText);\n if (classification.skip && lineText.startsWith(\"> \")) {\n return this.collectTokenObjects(lineText.slice(2));\n }\n if (classification.skip) return [];\n return this.collectTokenObjects(lineText);\n }\n\n private collectTokenObjects(lineText: string): Token[] {\n this.resetExpression(lineText);\n const result: Token[] = [];\n for (const token of this) {\n if (token.type === \"WS\" || token.type === \"NEWLINE\") continue;\n if (token.type.startsWith(\"MD_\")) continue;\n if (token.type === \"INLINE_SOLVE_START\" || token.type === \"BACKTICK_CLOSE\") continue;\n result.push(token);\n }\n return result;\n }\n\n private collectHighlightTokens(lineText: string): {type: string; value: string; offset: number; col: number; length: number; category: TokenCategory | undefined}[] {\n return this.collectTokenObjects(lineText).map(token => ({\n type: token.type,\n value: token.value,\n offset: token.offset,\n col: token.col,\n // `text`, not `value`: this is a span into the source, and the two\n // differ for a string literal, whose value is the payload while its\n // text still carries the quote characters the reader typed.\n length: token.text.length,\n category: getTokenCategory(token.type),\n }));\n }\n}\n\n/**\n * A lexer for operations that do not depend on registered vocabulary.\n *\n * Line classification and inline-solve detection read characters looking for\n * headings, comment markers, fences and backtick spans, and never consult the\n * keyword, unit or operator tables. Every lexer therefore returns the same\n * answer, so the callers that have no engine to ask can use this one. Checked\n * by `__tests__/lexer/LineClassificationIsVocabularyIndependent.spec.ts`.\n *\n * Do not tokenize with this. An engine's own lexer carries the vocabulary its\n * packages registered; this one carries none.\n */\nexport const sharedLexer = new Lexer(\"en\", undefined);","/**\n * TokenClass, Plugin-extensible keyword registration for the Lexer.\n *\n * Providers call `registry.register(tokenClass)` to teach the lexer about\n * their keywords. The registry merges locale keywords, provider keywords,\n * phrase mappings, and unit names into an optimized TokenLookup structure\n * consumed by the Lexer.\n *\n * @example\n * registry.register({\n * tokenType: 'CARET',\n * keywords: {},\n * phrases: { 'to the power of': true, 'power of': true },\n * priority: 10,\n * description: 'Exponentiation operators (x^y)',\n * });\n */\nexport interface TokenClass {\n /** The token type string produced by the lexer (e.g., \"FUNC\", \"PI\", \"CARET\").\n * Must match a token type that a ParseletRegistry has a parselet for. */\n tokenType: string;\n\n /** Single-word keywords (case-insensitive). The lexer lowercases input\n * before lookup, so these should be lowercase. Example:\n * { sqrt: true, abs: true, sin: true, cos: true } for tokenType \"FUNC\" */\n keywords: Record<string, boolean>;\n\n /** Multi-word phrases (case-insensitive). Matched by the built-in PhraseMatcher\n * via the phrase trie. Example:\n * { \"to the power of\": true, \"power of\": true } for tokenType \"CARET\" */\n phrases?: Record<string, boolean>;\n\n /** Priority for conflict resolution. When two TokenClasses register\n * the same keyword, the higher-priority class wins. Locale keywords\n * have priority 0 (set via setLocale). Providers should use\n * priority >= 10 to override locale defaults. Default: 0 */\n priority?: number;\n\n /** Human-readable description for debugging and introspection */\n description?: string;\n}\n\n// ── Phrase Trie ──────────────────────────────────────────────────────────────\n\n/** Trie node for multi-word phrase matching. */\nexport interface PhraseNode {\n /** Complete phrase token type (null = intermediate node) */\n type?: string;\n children: Map<string, PhraseNode>;\n}\n\n// ── TokenLookup, Optimized lookup structure for the Lexer ──────────────────\n\n/**\n * The optimized lookup structure built by TokenClassRegistry.build().\n * Consumed by the Lexer for O(1) keyword → token type lookups and\n * O(word-count) phrase matching.\n */\nexport interface TokenLookup {\n /** Lowercase keyword → token type. O(1) Map lookup. */\n keywordToType: Map<string, string>;\n\n /** Phrase trie for multi-word matching. Root node with children maps.\n * Null if no phrases registered. */\n phraseTrie: PhraseNode | null;\n\n /** Set of lowercase first-words of all registered phrases.\n * Used by the lexer to emit IDENT (not a phrase keyword) for words\n * that start multi-word phrases, deferring to the PhraseMatcher.\n *\n * Example: \"to\" is in phraseStartWords because \"to the power of\" is a phrase.\n * When the lexer sees \"to\", it emits IDENT and lets the phrase matcher\n * combine \"to the power of\" into a single CARET token.\n *\n * This prevents plugins from accidentally overriding phrase-start words.\n */\n phraseStartWords: Set<string>;\n\n /** Case-sensitive unit names for UNIT fallback after keyword lookup fails. */\n unitNames: ReadonlySet<string>;\n}\n\n// ── TokenClassRegistry ───────────────────────────────────────────────────────\n\n/**\n * Central registry for keyword→token-type mappings.\n *\n * Providers register TokenClasses; locales provide keyword maps;\n * units provide a name set. `build()` merges all sources into an\n * optimized TokenLookup consumed by the Lexer.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0)\n * 2. Provider keywords (sorted by priority ascending, higher priority wins)\n *\n * Unit names are stored separately (checked AFTER keyword lookup fails).\n * Phrases are stored in a trie for O(phrase-length) matching.\n */\nexport class TokenClassRegistry {\n private classes: TokenClass[] = [];\n private localeKeywordMap: Record<string, string> | null = null;\n private localePhraseMap: Record<string, string> | null = null;\n private unitNames: ReadonlySet<string> | null = null;\n\n /**\n * Register a provider's TokenClass. Must be called BEFORE build().\n * Can be called multiple times to add more entries.\n *\n * Built-in token types CANNOT be overridden, throws a EngineError\n * if the TokenClass attempts to register a keyword that conflicts\n * with an already-registered token type.\n */\n register(tokenClass: TokenClass): void {\n this.classes.push(tokenClass);\n }\n\n /**\n * Unregister all TokenClasses for a given token type.\n * Useful for plugin unload. Requires rebuild() to take effect.\n */\n unregister(tokenType: string): void {\n this.classes = this.classes.filter(c => c.tokenType !== tokenType);\n }\n\n /**\n * Set the locale's keyword→type map and optional phrase map.\n * Called on locale change. Priority 0 (cannot override providers with higher priority).\n */\n setLocale(keywordMap: Record<string, string>, phraseMap?: Record<string, string>): void {\n this.localeKeywordMap = keywordMap;\n this.localePhraseMap = phraseMap ?? null;\n }\n\n /**\n * Set the unit name set. Called when unit list changes.\n * Units are stored separately (checked AFTER keyword lookup fails).\n *\n * Takes a ReadonlySet because the caller's set is derived from the\n * conversion tables and must not be mutated; this class only ever reads it.\n */\n setUnits(unitNames: ReadonlySet<string>): void {\n this.unitNames = unitNames;\n }\n\n /**\n * Build the optimized TokenLookup from all registered sources.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0)\n * 2. Provider classes (sorted by priority ascending)\n *\n * Returns a frozen TokenLookup that the Lexer consumes.\n * Call build() again after register()/setLocale()/setUnits() changes.\n */\n build(): TokenLookup {\n const keywordToType = new Map<string, string>();\n\n // Layer 1: Locale keywords (priority 0, lowest)\n if (this.localeKeywordMap) {\n for (const [keyword, tokenType] of Object.entries(this.localeKeywordMap)) {\n keywordToType.set(keyword.toLowerCase(), tokenType);\n }\n }\n\n // Layer 2: Provider keywords (sorted by priority ascending, higher wins)\n const sorted = [...this.classes].sort((a, b) => (a.priority ?? 0) - (b.priority ?? 0));\n for (const tc of sorted) {\n for (const keyword of Object.keys(tc.keywords)) {\n keywordToType.set(keyword.toLowerCase(), tc.tokenType);\n }\n }\n\n // Build phrase trie from locale + provider phrases\n const phraseTrie = this.buildPhraseTrie();\n\n // Build phraseStartWords set (first word of every phrase)\n const phraseStartWords = this.buildPhraseStartWords();\n\n return {\n keywordToType,\n phraseTrie,\n phraseStartWords,\n unitNames: this.unitNames ?? new Set(),\n };\n }\n\n // ── Private helpers ─────────────────────────────────────────────────────\n\n private buildPhraseTrie(): PhraseNode | null {\n const root: PhraseNode = { children: new Map() };\n\n // Layer 1: Locale phrases\n if (this.localePhraseMap) {\n for (const [phrase, tokenType] of Object.entries(this.localePhraseMap)) {\n this.insertPhrase(root, phrase.toLowerCase(), tokenType);\n }\n }\n\n // Layer 2: Provider phrases\n for (const tc of this.classes) {\n if (!tc.phrases) continue;\n for (const phrase of Object.keys(tc.phrases)) {\n this.insertPhrase(root, phrase.toLowerCase(), tc.tokenType);\n }\n }\n\n return root.children.size > 0 ? root : null;\n }\n\n private insertPhrase(root: PhraseNode, phrase: string, tokenType: string): void {\n const words = phrase.split(' ');\n let node = root;\n for (const word of words) {\n if (!node.children.has(word)) {\n node.children.set(word, { children: new Map() });\n }\n node = node.children.get(word)!;\n }\n // Only set type if not already set, first-registered (locale) wins\n if (!node.type) {\n node.type = tokenType;\n }\n }\n\n private buildPhraseStartWords(): Set<string> {\n const startWords = new Set<string>();\n\n // Locale phrase first words\n if (this.localePhraseMap) {\n for (const phrase of Object.keys(this.localePhraseMap)) {\n const first = phrase.split(' ')[0].toLowerCase();\n startWords.add(first);\n }\n }\n\n // Provider phrase first words\n for (const tc of this.classes) {\n if (!tc.phrases) continue;\n for (const phrase of Object.keys(tc.phrases)) {\n const first = phrase.split(' ')[0].toLowerCase();\n startWords.add(first);\n }\n }\n\n return startWords;\n }\n}","/**\n * tokenRegistration.ts, Bootstrap for building TokenLookup from locale, units, and phrases.\n *\n * This module centralizes the assembly of the TokenLookup consumed by ExpressionLexer.\n * It merges:\n * 1. Locale keywords (from ILocale.keywordMap)\n * 2. Built-in phrase patterns (to the power of, increase by, etc.)\n * 3. Known unit names (from units.ts)\n *\n * The resulting TokenLookup is frozen and passed to ExpressionLexer.configuredLookup\n * at construction time, enabling data-driven keyword/unit/phrase resolution.\n */\nimport { TokenClassRegistry } from '@solve-js/lexer/TokenClassRegistry';\nimport type { TokenLookup } from '@solve-js/lexer/TokenClassRegistry';\nimport { getLocale, type ILocale } from '@solve-js/constants/locales';\nimport { knownUnits } from '@solve-js/lexer/units';\n\n// ── Built-in phrase map ───────────────────────────────────────────────────\n// These are the multi-word expressions handled as compound tokens.\n// Matched by the PhraseMatcher (trie-based) in ExpressionLexer.tryMatchPhrase().\nconst BUILTIN_PHRASES: Record<string, string> = {\n 'to the power of': 'CARET',\n 'power of': 'CARET',\n 'increase by': 'INCREASE_BY',\n 'decrease by': 'DECREASE_BY',\n 'times by': 'TIMES_BY',\n 'multiply by': 'MULTIPLY_BY',\n 'multiplied by': 'MULTIPLY_BY',\n 'divide by': 'DIVIDE_BY',\n};\n\n/**\n * Build a TokenLookup from locale keywords, known units, and built-in phrases.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0, lowest, can be overridden by providers)\n * 2. Built-in phrases (via locale phraseMap)\n * 3. Known units (checked AFTER keyword lookup fails, via unitNames set)\n *\n * The resulting TokenLookup replaces the internal keyword map, unit set,\n * phrase trie, and phraseStartWords in ExpressionLexer when set via\n * ExpressionLexer.configuredLookup.\n *\n * @param localeCode - The locale code (e.g., \"en\", \"de\"). Defaults to \"en\".\n * @returns A frozen TokenLookup ready for consumption by the lexer.\n */\nexport function buildTokenLookup(localeCode = 'en'): TokenLookup {\n const locale: ILocale = getLocale(localeCode);\n const registry = new TokenClassRegistry();\n\n // Layer 1: Locale keywords (priority 0, lowest)\n registry.setLocale(locale.keywordMap, BUILTIN_PHRASES);\n\n // Layer 2: Known units (checked after keyword lookup)\n registry.setUnits(knownUnits);\n\n // Build and return the lookup\n return registry.build();\n}\n"]}
1
+ {"version":3,"sources":["../src/lexer/LexerState.ts","../src/lexer/Lexer.ts","../src/lexer/TokenClassRegistry.ts","../src/lexer/tokenRegistration.ts"],"names":["LexerState","Lexer","localeCode","tokenLookup","ExpressionLexer","input","state","newState","lineText","plugin","text","classification","result","token","getTokenCategory","sharedLexer","TokenClassRegistry","tokenClass","tokenType","c","keywordMap","phraseMap","unitNames","keywordToType","keyword","sorted","a","b","tc","phraseTrie","phraseStartWords","root","phrase","words","node","word","startWords","first","BUILTIN_PHRASES","buildTokenLookup","locale","getLocale","registry","knownUnits"],"mappings":"uJAMO,IAAKA,CAAAA,CAAAA,CAAAA,CAAAA,GACXA,EAAA,IAAA,CAAO,MAAA,CACPA,EAAA,MAAA,CAAS,QAAA,CACTA,EAAA,MAAA,CAAS,QAAA,CAHEA,OAAA,EAAA,ECaL,IAAMC,EAAN,KAAY,CAkBjB,YAAYC,CAAAA,CAAa,IAAA,CAAMC,EAA2B,CAf1D,IAAA,CAAQ,aAA2B,MAAA,CAEnC,IAAA,CAAQ,UAAY,KAAA,CAIpB,IAAA,CAAQ,OAAkB,EAAC,CAC3B,KAAQ,QAAA,CAAmB,CAAA,CAYzB,KAAK,eAAA,CAAkB,IAAIC,EAAgBF,CAAAA,CAAYC,CAAW,EACpE,CAEA,KAAA,CAAME,EAAeC,CAAAA,CAA0B,CAC7C,IAAMC,CAAAA,CAAWD,CAAAA,EAAS,OAQ1B,GAPA,IAAA,CAAK,aAAeC,CAAAA,CACpB,IAAA,CAAK,UAAY,KAAA,CACjB,IAAA,CAAK,YAAc,MAAA,CAKfA,CAAAA,GAAa,OAAiB,CAEhC,GADuB,KAAK,eAAA,CAAgB,YAAA,CAAaF,CAAK,CAAA,CAC3C,IAAA,CAAM,CACvB,IAAA,CAAK,MAAA,CAAS,EAAC,CACf,IAAA,CAAK,SAAW,CAAA,CAChB,MACF,CAEA,IAAA,CAAK,eAAA,CAAgB,MAAMA,CAAK,CAAA,CAChC,KAAK,MAAA,CAAS,IAAA,CAAK,gBAAgB,WAAA,EAAY,CAC/C,KAAK,QAAA,CAAW,EAClB,MAEE,IAAA,CAAK,eAAA,CAAgB,MAAMA,CAAK,CAAA,CAChC,KAAK,MAAA,CAAS,IAAA,CAAK,gBAAgB,WAAA,EAAY,CAC/C,KAAK,QAAA,CAAW,EAEpB,CAMA,YAAA,CAAaG,CAAAA,CAAsC,CACjD,OAAO,IAAA,CAAK,gBAAgB,YAAA,CAAaA,CAAQ,CACnD,CAMA,gBAAA,CAAiBA,EAAkB,CACjC,OAAO,KAAK,eAAA,CAAgB,gBAAA,CAAiBA,CAAQ,CACvD,CAMA,aAAsC,CACpC,OAAO,KAAK,eAAA,CAAgB,WAAA,EAC9B,CAEA,IAAA,EAA0B,CACxB,GAAI,IAAA,CAAK,UACP,OAAA,IAAA,CAAK,SAAA,CAAY,MACV,IAAA,CAAK,WAAA,CAGd,GAAI,IAAA,CAAK,QAAA,CAAW,KAAK,MAAA,CAAO,MAAA,CAC9B,OAAO,IAAA,CAAK,MAAA,CAAO,KAAK,QAAA,EAAU,CAGtC,CAEA,IAAA,EAA0B,CACxB,OAAI,IAAA,CAAK,SAAA,CAAkB,KAAK,WAAA,EAChC,IAAA,CAAK,YAAc,IAAA,CAAK,IAAA,GACxB,IAAA,CAAK,SAAA,CAAY,KACV,IAAA,CAAK,WAAA,CACd,CAEA,CAAC,MAAA,CAAO,QAAQ,CAAA,EAAqB,CACnC,OAAO,IAAA,CAAK,MAAA,CAAO,MAAA,CAAO,QAAQ,CAAA,EACpC,CAQA,kBAAA,CAAmBC,CAAAA,CAA+B,CAChD,IAAA,CAAK,eAAA,CAAgB,mBAAmBA,CAAM,EAChD,CAMA,oBAAA,CAAqBA,CAAAA,CAA+B,CAClD,IAAA,CAAK,eAAA,CAAgB,qBAAqBA,CAAM,EAClD,CAOA,eAAA,CAAgBJ,CAAAA,CAAqB,CACnC,IAAA,CAAK,YAAA,CAAe,OACpB,IAAA,CAAK,SAAA,CAAY,MACjB,IAAA,CAAK,WAAA,CAAc,OACnB,IAAA,CAAK,eAAA,CAAgB,MAAMA,CAAK,CAAA,CAChC,KAAK,MAAA,CAAS,IAAA,CAAK,gBAAgB,WAAA,EAAY,CAC/C,KAAK,QAAA,CAAW,EAClB,CAQA,YAAA,CAAaK,CAAAA,CAAgC,CAC3C,OAAO,IAAA,CAAK,gBAAgB,YAAA,CAAaA,CAAI,CAC/C,CAEA,QAAA,EAAuB,CACrB,OAAO,IAAA,CAAK,YACd,CAEA,QAAA,CAASJ,EAAyB,CAChC,IAAA,CAAK,aAAeA,EACtB,CAEA,mBAAmBE,CAAAA,CAAqI,CACtJ,IAAMG,CAAAA,CAAiB,IAAA,CAAK,gBAAgB,YAAA,CAAaH,CAAQ,EAKjE,OAAIG,CAAAA,CAAe,MAAQH,CAAAA,CAAS,UAAA,CAAW,IAAI,CAAA,CAC1C,IAAA,CAAK,uBAAuBA,CAAAA,CAAS,KAAA,CAAM,CAAC,CAAC,CAAA,CAGlDG,EAAe,IAAA,CACV,GAGF,IAAA,CAAK,sBAAA,CAAuBH,CAAQ,CAC7C,CAYA,yBAAyBA,CAAAA,CAA2B,CAClD,IAAMG,CAAAA,CAAiB,IAAA,CAAK,gBAAgB,YAAA,CAAaH,CAAQ,EACjE,OAAIG,CAAAA,CAAe,MAAQH,CAAAA,CAAS,UAAA,CAAW,IAAI,CAAA,CAC1C,IAAA,CAAK,oBAAoBA,CAAAA,CAAS,KAAA,CAAM,CAAC,CAAC,CAAA,CAE/CG,EAAe,IAAA,CAAa,GACzB,IAAA,CAAK,mBAAA,CAAoBH,CAAQ,CAC1C,CAEQ,oBAAoBA,CAAAA,CAA2B,CACrD,KAAK,eAAA,CAAgBA,CAAQ,EAC7B,IAAMI,CAAAA,CAAkB,EAAC,CACzB,IAAA,IAAWC,KAAS,IAAA,CACdA,CAAAA,CAAM,OAAS,IAAA,EAAQA,CAAAA,CAAM,OAAS,SAAA,EACtCA,CAAAA,CAAM,KAAK,UAAA,CAAW,KAAK,GAC3BA,CAAAA,CAAM,IAAA,GAAS,sBAAwBA,CAAAA,CAAM,IAAA,GAAS,kBAC1DD,CAAAA,CAAO,IAAA,CAAKC,CAAK,CAAA,CAEnB,OAAOD,CACT,CAEQ,sBAAA,CAAuBJ,EAAqI,CAClK,OAAO,KAAK,mBAAA,CAAoBA,CAAQ,EAAE,GAAA,CAAIK,CAAAA,GAAU,CACtD,IAAA,CAAMA,CAAAA,CAAM,KACZ,KAAA,CAAOA,CAAAA,CAAM,KAAA,CACb,MAAA,CAAQA,CAAAA,CAAM,MAAA,CACd,IAAKA,CAAAA,CAAM,GAAA,CAIX,OAAQA,CAAAA,CAAM,IAAA,CAAK,OACnB,QAAA,CAAUC,CAAAA,CAAiBD,EAAM,IAAI,CACvC,EAAE,CACJ,CACF,EAcaE,CAAAA,CAAc,IAAId,EAAM,IAAA,CAAM,MAAS,EClJ7C,IAAMe,CAAAA,CAAN,KAAyB,CAAzB,WAAA,EAAA,CACL,KAAQ,OAAA,CAAwB,GAChC,IAAA,CAAQ,gBAAA,CAAkD,KAC1D,IAAA,CAAQ,eAAA,CAAiD,KACzD,IAAA,CAAQ,SAAA,CAAwC,MAUhD,QAAA,CAASC,CAAAA,CAA8B,CACrC,IAAA,CAAK,OAAA,CAAQ,KAAKA,CAAU,EAC9B,CAMA,UAAA,CAAWC,CAAAA,CAAyB,CAClC,IAAA,CAAK,OAAA,CAAU,KAAK,OAAA,CAAQ,MAAA,CAAOC,GAAKA,CAAAA,CAAE,SAAA,GAAcD,CAAS,EACnE,CAMA,UAAUE,CAAAA,CAAoCC,CAAAA,CAA0C,CACtF,IAAA,CAAK,gBAAA,CAAmBD,EACxB,IAAA,CAAK,eAAA,CAAkBC,GAAa,KACtC,CASA,SAASC,CAAAA,CAAsC,CAC7C,KAAK,SAAA,CAAYA,EACnB,CAYA,KAAA,EAAqB,CACnB,IAAMC,CAAAA,CAAgB,IAAI,IAG1B,GAAI,IAAA,CAAK,iBACP,IAAA,GAAW,CAACC,EAASN,CAAS,CAAA,GAAK,OAAO,OAAA,CAAQ,IAAA,CAAK,gBAAgB,CAAA,CACrEK,CAAAA,CAAc,IAAIC,CAAAA,CAAQ,WAAA,GAAeN,CAAS,CAAA,CAKtD,IAAMO,CAAAA,CAAS,CAAC,GAAG,IAAA,CAAK,OAAO,EAAE,IAAA,CAAK,CAACC,EAAGC,CAAAA,GAAAA,CAAOD,CAAAA,CAAE,UAAY,CAAA,GAAMC,CAAAA,CAAE,UAAY,CAAA,CAAE,CAAA,CACrF,QAAWC,CAAAA,IAAMH,CAAAA,CACf,QAAWD,CAAAA,IAAW,MAAA,CAAO,KAAKI,CAAAA,CAAG,QAAQ,EAC3CL,CAAAA,CAAc,GAAA,CAAIC,EAAQ,WAAA,EAAY,CAAGI,EAAG,SAAS,CAAA,CAKzD,IAAMC,CAAAA,CAAa,IAAA,CAAK,iBAAgB,CAGlCC,CAAAA,CAAmB,KAAK,qBAAA,EAAsB,CAEpD,OAAO,CACL,aAAA,CAAAP,EACA,UAAA,CAAAM,CAAAA,CACA,iBAAAC,CAAAA,CACA,SAAA,CAAW,KAAK,SAAA,EAAa,IAAI,GACnC,CACF,CAIQ,iBAAqC,CAC3C,IAAMC,EAAmB,CAAE,QAAA,CAAU,IAAI,GAAM,CAAA,CAG/C,GAAI,IAAA,CAAK,eAAA,CACP,OAAW,CAACC,CAAAA,CAAQd,CAAS,CAAA,GAAK,MAAA,CAAO,QAAQ,IAAA,CAAK,eAAe,CAAA,CACnE,IAAA,CAAK,YAAA,CAAaa,CAAAA,CAAMC,EAAO,WAAA,EAAY,CAAGd,CAAS,CAAA,CAK3D,IAAA,IAAWU,KAAM,IAAA,CAAK,OAAA,CACpB,GAAKA,CAAAA,CAAG,OAAA,CACR,QAAWI,CAAAA,IAAU,MAAA,CAAO,KAAKJ,CAAAA,CAAG,OAAO,EACzC,IAAA,CAAK,YAAA,CAAaG,EAAMC,CAAAA,CAAO,WAAA,GAAeJ,CAAAA,CAAG,SAAS,EAI9D,OAAOG,CAAAA,CAAK,SAAS,IAAA,CAAO,CAAA,CAAIA,EAAO,IACzC,CAEQ,aAAaA,CAAAA,CAAkBC,CAAAA,CAAgBd,EAAyB,CAC9E,IAAMe,EAAQD,CAAAA,CAAO,KAAA,CAAM,GAAG,CAAA,CAC1BE,CAAAA,CAAOH,EACX,IAAA,IAAWI,CAAAA,IAAQF,EACZC,CAAAA,CAAK,QAAA,CAAS,IAAIC,CAAI,CAAA,EACzBD,EAAK,QAAA,CAAS,GAAA,CAAIC,EAAM,CAAE,QAAA,CAAU,IAAI,GAAM,CAAC,EAEjDD,CAAAA,CAAOA,CAAAA,CAAK,SAAS,GAAA,CAAIC,CAAI,EAG1BD,CAAAA,CAAK,IAAA,GACRA,EAAK,IAAA,CAAOhB,CAAAA,EAEhB,CAEQ,qBAAA,EAAqC,CAC3C,IAAMkB,CAAAA,CAAa,IAAI,IAGvB,GAAI,IAAA,CAAK,gBACP,IAAA,IAAWJ,CAAAA,IAAU,OAAO,IAAA,CAAK,IAAA,CAAK,eAAe,CAAA,CAAG,CACtD,IAAMK,CAAAA,CAAQL,CAAAA,CAAO,MAAM,GAAG,CAAA,CAAE,CAAC,CAAA,CAAE,WAAA,GACnCI,CAAAA,CAAW,GAAA,CAAIC,CAAK,EACtB,CAIF,QAAWT,CAAAA,IAAM,IAAA,CAAK,QACpB,GAAKA,CAAAA,CAAG,QACR,IAAA,IAAWI,CAAAA,IAAU,OAAO,IAAA,CAAKJ,CAAAA,CAAG,OAAO,CAAA,CAAG,CAC5C,IAAMS,CAAAA,CAAQL,CAAAA,CAAO,MAAM,GAAG,CAAA,CAAE,CAAC,CAAA,CAAE,WAAA,GACnCI,CAAAA,CAAW,GAAA,CAAIC,CAAK,EACtB,CAGF,OAAOD,CACT,CACF,EClOA,IAAME,CAAAA,CAA0C,CAC9C,iBAAA,CAAmB,OAAA,CACnB,WAAY,OAAA,CACZ,aAAA,CAAe,cACf,aAAA,CAAe,aAAA,CACf,WAAY,UAAA,CACZ,aAAA,CAAe,cACf,eAAA,CAAiB,aAAA,CACjB,YAAa,WACf,CAAA,CAiBO,SAASC,CAAAA,CAAiBrC,GAAAA,CAAa,KAAmB,CAC/D,IAAMsC,EAAkBC,CAAAA,CAAUvC,GAAU,EACtCwC,CAAAA,CAAW,IAAI1B,EAGrB,OAAA0B,CAAAA,CAAS,UAAUF,CAAAA,CAAO,UAAA,CAAYF,CAAe,CAAA,CAGrDI,CAAAA,CAAS,SAASC,GAAU,CAAA,CAGrBD,CAAAA,CAAS,KAAA,EAClB","file":"chunk-2DPKJ2SM.js","sourcesContent":["/**\n * Lexer state machine modes.\n * - Main: document-level scanning with markdown classification\n * - Inline: expression embedded in markdown inline solve (`s\\`...\\``)\n * - String: inside a double-quoted string literal\n */\nexport enum LexerState {\n\tMain = \"main\",\n\tInline = \"inline\",\n\tString = \"string\",\n}\n","import { ExpressionLexer, LineClassification, LexerVocabulary, type ScanLineResult } from \"./ExpressionLexer\";\nimport { Token } from \"@solve-js/lexer/Token\";\nimport { LexerState } from \"@solve-js/lexer/LexerState\";\nimport { getTokenCategory } from \"@solve-js/language/TokenCategoryMap\";\nimport type { TokenCategory } from \"@solve-js/language/TokenCategory\";\nimport type { TokenLookup } from \"@solve-js/lexer/TokenClassRegistry\";\n\n/**\n * Public tokenizer wrapper around {@link ExpressionLexer}.\n *\n * `ExpressionLexer` does the actual character-by-character scanning;\n * `Lexer` adds a materialized-token-array streaming interface\n * (`next()`/`peek()`) plus line-classification state (`reset()`) so\n * callers can iterate a line's tokens without re-scanning on each peek.\n *\n * Each `ExpressionEngine` instance owns its own `Lexer`, and packages\n * extend it via {@link registerVocabulary} (keywords, operators, units)\n * see `IEnginePackage.lexerVocabulary`.\n */\nexport class Lexer {\n /** Expression-mode lexer (Phase A: V8-optimized, replaces moo) */\n private expressionLexer: ExpressionLexer;\n private currentState: LexerState = LexerState.Main;\n private peekedToken: Token | undefined;\n private hasPeeked = false;\n\n // Materialized token array from the last reset() call, used for\n // next()/peek() streaming access.\n private tokens: Token[] = [];\n private tokenIdx: number = 0;\n\n /**\n * @param localeCode - Locale code (e.g., \"en\", \"de\"). Defaults to \"en\".\n * @param tokenLookup - Optional TokenLookup from TokenClassRegistry.\n * When provided, configures ExpressionLexer to use registry-built\n * keyword/unit/phrase lookups instead of internal instance maps.\n */\n constructor(localeCode = \"en\", tokenLookup?: TokenLookup) {\n // Pass the lookup directly to ExpressionLexer's constructor, it's an\n // instance field now, not a static. Each Lexer instance gets its own\n // isolated lookup, preventing cross-instance corruption.\n this.expressionLexer = new ExpressionLexer(localeCode, tokenLookup);\n }\n\n reset(input: string, state?: LexerState): void {\n const newState = state ?? LexerState.Main;\n this.currentState = newState;\n this.hasPeeked = false;\n this.peekedToken = undefined;\n\n // Phase B: Main state classifies the line with the markdown scanner.\n // Skip lines (headings, fences, HRs, etc.) produce empty token arrays.\n // Expression lines and lines with inline solves are tokenized normally.\n if (newState === LexerState.Main) {\n const classification = this.expressionLexer.classifyLine(input);\n if (classification.skip) {\n this.tokens = [];\n this.tokenIdx = 0;\n return;\n }\n // Expression line or markdown line with inline solves, tokenize.\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n } else {\n // Non-main states (Inline, String), expression tokenization.\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n }\n }\n\n /**\n * Classify a single line of markdown text (Phase B).\n * Delegates to the ExpressionLexer's character-by-character scanner.\n */\n classifyLine(lineText: string): LineClassification {\n return this.expressionLexer.classifyLine(lineText);\n }\n\n /**\n * Find all inline solve markers in a line (Phase B).\n * Delegates to the ExpressionLexer's character-by-character scanner.\n */\n findInlineSolves(lineText: string) {\n return this.expressionLexer.findInlineSolves(lineText);\n }\n\n /**\n * Every keyword this lexer currently recognizes (locale + plugin-contributed),\n * mapped to the token type it lexes to. Delegates to the ExpressionLexer.\n */\n getKeywords(): Record<string, string> {\n return this.expressionLexer.getKeywords();\n }\n\n next(): Token | undefined {\n if (this.hasPeeked) {\n this.hasPeeked = false;\n return this.peekedToken;\n }\n // Materialized token array (ExpressionLexer path).\n if (this.tokenIdx < this.tokens.length) {\n return this.tokens[this.tokenIdx++];\n }\n return undefined;\n }\n\n peek(): Token | undefined {\n if (this.hasPeeked) return this.peekedToken;\n this.peekedToken = this.next();\n this.hasPeeked = true;\n return this.peekedToken;\n }\n\n [Symbol.iterator](): Iterator<Token> {\n return this.tokens[Symbol.iterator]();\n }\n\n /**\n * Register a plugin to extend the lexer with custom tokens.\n * Delegates to the underlying ExpressionLexer.\n *\n * @see LexerVocabulary for the supported extension points.\n */\n registerVocabulary(plugin: LexerVocabulary): void {\n this.expressionLexer.registerVocabulary(plugin);\n }\n\n /**\n * Unregister a plugin, removing its custom tokens from the lexer.\n * Delegates to the underlying ExpressionLexer.\n */\n unregisterVocabulary(plugin: LexerVocabulary): void {\n this.expressionLexer.unregisterVocabulary(plugin);\n }\n\n /**\n * Reset the lexer for expression-only text, skips the classifyLine()\n * overhead in reset() for callers that already know the input is an\n * evaluable expression (e.g., after isEmptyLine() confirmed non-skip).\n */\n resetExpression(input: string): void {\n this.currentState = LexerState.Main;\n this.hasPeeked = false;\n this.peekedToken = undefined;\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n }\n\n /**\n * Scan a full document in one pass, classifying each line and\n * tokenizing non-skipped lines. Delegates to ExpressionLexer.\n *\n * @returns ScanLineResult[], one per line, with classification + tokens.\n */\n scanDocument(text: string): ScanLineResult[] {\n return this.expressionLexer.scanDocument(text);\n }\n\n getState(): LexerState {\n return this.currentState;\n }\n\n setState(state: LexerState): void {\n this.currentState = state;\n }\n\n getHighlightTokens(lineText: string): {type: string; value: string; offset: number; col: number; length: number; category: TokenCategory | undefined}[] {\n const classification = this.expressionLexer.classifyLine(lineText);\n\n // For blockquote lines, strip the \"> \" prefix and tokenize the expression content.\n // This lets expressions inside blockquotes (e.g., \"> 1 + 2\") get syntax highlighted\n // while pure structural lines (headings, code fences) remain unhighlighted.\n if (classification.skip && lineText.startsWith(\"> \")) {\n return this.collectHighlightTokens(lineText.slice(2));\n }\n\n if (classification.skip) {\n return [];\n }\n\n return this.collectHighlightTokens(lineText);\n }\n\n /**\n * The same tokens {@link getHighlightTokens} reduces, before reduction.\n *\n * Exists because normalization operates on tokens, not on the flattened\n * shape, and a consumer that wants phrase-fused highlighting has to run the\n * normalizer between the two. See `LanguageService.getSemanticTokens`.\n *\n * @param lineText - One line of source.\n * @returns Every token on the line that is worth painting, unreduced.\n */\n getHighlightTokenObjects(lineText: string): Token[] {\n const classification = this.expressionLexer.classifyLine(lineText);\n if (classification.skip && lineText.startsWith(\"> \")) {\n return this.collectTokenObjects(lineText.slice(2));\n }\n if (classification.skip) return [];\n return this.collectTokenObjects(lineText);\n }\n\n private collectTokenObjects(lineText: string): Token[] {\n this.resetExpression(lineText);\n const result: Token[] = [];\n for (const token of this) {\n if (token.type === \"WS\" || token.type === \"NEWLINE\") continue;\n if (token.type.startsWith(\"MD_\")) continue;\n if (token.type === \"INLINE_SOLVE_START\" || token.type === \"BACKTICK_CLOSE\") continue;\n result.push(token);\n }\n return result;\n }\n\n private collectHighlightTokens(lineText: string): {type: string; value: string; offset: number; col: number; length: number; category: TokenCategory | undefined}[] {\n return this.collectTokenObjects(lineText).map(token => ({\n type: token.type,\n value: token.value,\n offset: token.offset,\n col: token.col,\n // `text`, not `value`: this is a span into the source, and the two\n // differ for a string literal, whose value is the payload while its\n // text still carries the quote characters the reader typed.\n length: token.text.length,\n category: getTokenCategory(token.type),\n }));\n }\n}\n\n/**\n * A lexer for operations that do not depend on registered vocabulary.\n *\n * Line classification and inline-solve detection read characters looking for\n * headings, comment markers, fences and backtick spans, and never consult the\n * keyword, unit or operator tables. Every lexer therefore returns the same\n * answer, so the callers that have no engine to ask can use this one. Checked\n * by `__tests__/lexer/LineClassificationIsVocabularyIndependent.spec.ts`.\n *\n * Do not tokenize with this. An engine's own lexer carries the vocabulary its\n * packages registered; this one carries none.\n */\nexport const sharedLexer = new Lexer(\"en\", undefined);","/**\n * TokenClass, Plugin-extensible keyword registration for the Lexer.\n *\n * Providers call `registry.register(tokenClass)` to teach the lexer about\n * their keywords. The registry merges locale keywords, provider keywords,\n * phrase mappings, and unit names into an optimized TokenLookup structure\n * consumed by the Lexer.\n *\n * @example\n * registry.register({\n * tokenType: 'CARET',\n * keywords: {},\n * phrases: { 'to the power of': true, 'power of': true },\n * priority: 10,\n * description: 'Exponentiation operators (x^y)',\n * });\n */\nexport interface TokenClass {\n /** The token type string produced by the lexer (e.g., \"FUNC\", \"PI\", \"CARET\").\n * Must match a token type that a ParseletRegistry has a parselet for. */\n tokenType: string;\n\n /** Single-word keywords (case-insensitive). The lexer lowercases input\n * before lookup, so these should be lowercase. Example:\n * { sqrt: true, abs: true, sin: true, cos: true } for tokenType \"FUNC\" */\n keywords: Record<string, boolean>;\n\n /** Multi-word phrases (case-insensitive). Matched by the built-in PhraseMatcher\n * via the phrase trie. Example:\n * { \"to the power of\": true, \"power of\": true } for tokenType \"CARET\" */\n phrases?: Record<string, boolean>;\n\n /** Priority for conflict resolution. When two TokenClasses register\n * the same keyword, the higher-priority class wins. Locale keywords\n * have priority 0 (set via setLocale). Providers should use\n * priority >= 10 to override locale defaults. Default: 0 */\n priority?: number;\n\n /** Human-readable description for debugging and introspection */\n description?: string;\n}\n\n// ── Phrase Trie ──────────────────────────────────────────────────────────────\n\n/** Trie node for multi-word phrase matching. */\nexport interface PhraseNode {\n /** Complete phrase token type (null = intermediate node) */\n type?: string;\n children: Map<string, PhraseNode>;\n}\n\n// ── TokenLookup, Optimized lookup structure for the Lexer ──────────────────\n\n/**\n * The optimized lookup structure built by TokenClassRegistry.build().\n * Consumed by the Lexer for O(1) keyword → token type lookups and\n * O(word-count) phrase matching.\n */\nexport interface TokenLookup {\n /** Lowercase keyword → token type. O(1) Map lookup. */\n keywordToType: Map<string, string>;\n\n /** Phrase trie for multi-word matching. Root node with children maps.\n * Null if no phrases registered. */\n phraseTrie: PhraseNode | null;\n\n /** Set of lowercase first-words of all registered phrases.\n * Used by the lexer to emit IDENT (not a phrase keyword) for words\n * that start multi-word phrases, deferring to the PhraseMatcher.\n *\n * Example: \"to\" is in phraseStartWords because \"to the power of\" is a phrase.\n * When the lexer sees \"to\", it emits IDENT and lets the phrase matcher\n * combine \"to the power of\" into a single CARET token.\n *\n * This prevents plugins from accidentally overriding phrase-start words.\n */\n phraseStartWords: Set<string>;\n\n /** Case-sensitive unit names for UNIT fallback after keyword lookup fails. */\n unitNames: ReadonlySet<string>;\n}\n\n// ── TokenClassRegistry ───────────────────────────────────────────────────────\n\n/**\n * Central registry for keyword→token-type mappings.\n *\n * Providers register TokenClasses; locales provide keyword maps;\n * units provide a name set. `build()` merges all sources into an\n * optimized TokenLookup consumed by the Lexer.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0)\n * 2. Provider keywords (sorted by priority ascending, higher priority wins)\n *\n * Unit names are stored separately (checked AFTER keyword lookup fails).\n * Phrases are stored in a trie for O(phrase-length) matching.\n */\nexport class TokenClassRegistry {\n private classes: TokenClass[] = [];\n private localeKeywordMap: Record<string, string> | null = null;\n private localePhraseMap: Record<string, string> | null = null;\n private unitNames: ReadonlySet<string> | null = null;\n\n /**\n * Register a provider's TokenClass. Must be called BEFORE build().\n * Can be called multiple times to add more entries.\n *\n * Built-in token types CANNOT be overridden, throws a EngineError\n * if the TokenClass attempts to register a keyword that conflicts\n * with an already-registered token type.\n */\n register(tokenClass: TokenClass): void {\n this.classes.push(tokenClass);\n }\n\n /**\n * Unregister all TokenClasses for a given token type.\n * Useful for plugin unload. Requires rebuild() to take effect.\n */\n unregister(tokenType: string): void {\n this.classes = this.classes.filter(c => c.tokenType !== tokenType);\n }\n\n /**\n * Set the locale's keyword→type map and optional phrase map.\n * Called on locale change. Priority 0 (cannot override providers with higher priority).\n */\n setLocale(keywordMap: Record<string, string>, phraseMap?: Record<string, string>): void {\n this.localeKeywordMap = keywordMap;\n this.localePhraseMap = phraseMap ?? null;\n }\n\n /**\n * Set the unit name set. Called when unit list changes.\n * Units are stored separately (checked AFTER keyword lookup fails).\n *\n * Takes a ReadonlySet because the caller's set is derived from the\n * conversion tables and must not be mutated; this class only ever reads it.\n */\n setUnits(unitNames: ReadonlySet<string>): void {\n this.unitNames = unitNames;\n }\n\n /**\n * Build the optimized TokenLookup from all registered sources.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0)\n * 2. Provider classes (sorted by priority ascending)\n *\n * Returns a frozen TokenLookup that the Lexer consumes.\n * Call build() again after register()/setLocale()/setUnits() changes.\n */\n build(): TokenLookup {\n const keywordToType = new Map<string, string>();\n\n // Layer 1: Locale keywords (priority 0, lowest)\n if (this.localeKeywordMap) {\n for (const [keyword, tokenType] of Object.entries(this.localeKeywordMap)) {\n keywordToType.set(keyword.toLowerCase(), tokenType);\n }\n }\n\n // Layer 2: Provider keywords (sorted by priority ascending, higher wins)\n const sorted = [...this.classes].sort((a, b) => (a.priority ?? 0) - (b.priority ?? 0));\n for (const tc of sorted) {\n for (const keyword of Object.keys(tc.keywords)) {\n keywordToType.set(keyword.toLowerCase(), tc.tokenType);\n }\n }\n\n // Build phrase trie from locale + provider phrases\n const phraseTrie = this.buildPhraseTrie();\n\n // Build phraseStartWords set (first word of every phrase)\n const phraseStartWords = this.buildPhraseStartWords();\n\n return {\n keywordToType,\n phraseTrie,\n phraseStartWords,\n unitNames: this.unitNames ?? new Set(),\n };\n }\n\n // ── Private helpers ─────────────────────────────────────────────────────\n\n private buildPhraseTrie(): PhraseNode | null {\n const root: PhraseNode = { children: new Map() };\n\n // Layer 1: Locale phrases\n if (this.localePhraseMap) {\n for (const [phrase, tokenType] of Object.entries(this.localePhraseMap)) {\n this.insertPhrase(root, phrase.toLowerCase(), tokenType);\n }\n }\n\n // Layer 2: Provider phrases\n for (const tc of this.classes) {\n if (!tc.phrases) continue;\n for (const phrase of Object.keys(tc.phrases)) {\n this.insertPhrase(root, phrase.toLowerCase(), tc.tokenType);\n }\n }\n\n return root.children.size > 0 ? root : null;\n }\n\n private insertPhrase(root: PhraseNode, phrase: string, tokenType: string): void {\n const words = phrase.split(' ');\n let node = root;\n for (const word of words) {\n if (!node.children.has(word)) {\n node.children.set(word, { children: new Map() });\n }\n node = node.children.get(word)!;\n }\n // Only set type if not already set, first-registered (locale) wins\n if (!node.type) {\n node.type = tokenType;\n }\n }\n\n private buildPhraseStartWords(): Set<string> {\n const startWords = new Set<string>();\n\n // Locale phrase first words\n if (this.localePhraseMap) {\n for (const phrase of Object.keys(this.localePhraseMap)) {\n const first = phrase.split(' ')[0].toLowerCase();\n startWords.add(first);\n }\n }\n\n // Provider phrase first words\n for (const tc of this.classes) {\n if (!tc.phrases) continue;\n for (const phrase of Object.keys(tc.phrases)) {\n const first = phrase.split(' ')[0].toLowerCase();\n startWords.add(first);\n }\n }\n\n return startWords;\n }\n}","/**\n * tokenRegistration.ts, Bootstrap for building TokenLookup from locale, units, and phrases.\n *\n * This module centralizes the assembly of the TokenLookup consumed by ExpressionLexer.\n * It merges:\n * 1. Locale keywords (from ILocale.keywordMap)\n * 2. Built-in phrase patterns (to the power of, increase by, etc.)\n * 3. Known unit names (from units.ts)\n *\n * The resulting TokenLookup is frozen and passed to ExpressionLexer.configuredLookup\n * at construction time, enabling data-driven keyword/unit/phrase resolution.\n */\nimport { TokenClassRegistry } from '@solve-js/lexer/TokenClassRegistry';\nimport type { TokenLookup } from '@solve-js/lexer/TokenClassRegistry';\nimport { getLocale, type ILocale } from '@solve-js/constants/locales';\nimport { knownUnits } from '@solve-js/lexer/units';\n\n// ── Built-in phrase map ───────────────────────────────────────────────────\n// These are the multi-word expressions handled as compound tokens.\n// Matched by the PhraseMatcher (trie-based) in ExpressionLexer.tryMatchPhrase().\nconst BUILTIN_PHRASES: Record<string, string> = {\n 'to the power of': 'CARET',\n 'power of': 'CARET',\n 'increase by': 'INCREASE_BY',\n 'decrease by': 'DECREASE_BY',\n 'times by': 'TIMES_BY',\n 'multiply by': 'MULTIPLY_BY',\n 'multiplied by': 'MULTIPLY_BY',\n 'divide by': 'DIVIDE_BY',\n};\n\n/**\n * Build a TokenLookup from locale keywords, known units, and built-in phrases.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0, lowest, can be overridden by providers)\n * 2. Built-in phrases (via locale phraseMap)\n * 3. Known units (checked AFTER keyword lookup fails, via unitNames set)\n *\n * The resulting TokenLookup replaces the internal keyword map, unit set,\n * phrase trie, and phraseStartWords in ExpressionLexer when set via\n * ExpressionLexer.configuredLookup.\n *\n * @param localeCode - The locale code (e.g., \"en\", \"de\"). Defaults to \"en\".\n * @returns A frozen TokenLookup ready for consumption by the lexer.\n */\nexport function buildTokenLookup(localeCode = 'en'): TokenLookup {\n const locale: ILocale = getLocale(localeCode);\n const registry = new TokenClassRegistry();\n\n // Layer 1: Locale keywords (priority 0, lowest)\n registry.setLocale(locale.keywordMap, BUILTIN_PHRASES);\n\n // Layer 2: Known units (checked after keyword lookup)\n registry.setUnits(knownUnits);\n\n // Build and return the lookup\n return registry.build();\n}\n"]}