solve-engine 1.0.0-beta.2 → 1.0.0-beta.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{Lexer-D9l4Yrk2.d.ts → Lexer-Cfm79Dak.d.cts} +18 -2
- package/dist/{Lexer-BJdhlnej.d.cts → Lexer-W9MBOP0V.d.ts} +18 -2
- package/dist/{PackageRegistry-DaH4aIKP.d.ts → PackageRegistry-CjDt-Jy_.d.ts} +60 -4
- package/dist/{PackageRegistry-D-Tv_7ca.d.cts → PackageRegistry-pHtKythi.d.cts} +60 -4
- package/dist/{Parselet-CK8cQu2a.d.cts → Parselet-Cu0bLSis.d.cts} +1 -1
- package/dist/{Parselet-ConOIdRO.d.ts → Parselet-DEdF9I7n.d.ts} +1 -1
- package/dist/{Token-0jpvUdPY.d.cts → Token-BzG5G4ja.d.cts} +44 -0
- package/dist/{Token-0jpvUdPY.d.ts → Token-BzG5G4ja.d.ts} +44 -0
- package/dist/TokenNormalizer-DGVa24Q-.d.cts +377 -0
- package/dist/TokenNormalizer-t_GotBxr.d.ts +377 -0
- package/dist/{chunk-HWSZJQCI.js → chunk-2CS6OMZK.js} +37 -15
- package/dist/chunk-2CS6OMZK.js.map +1 -0
- package/dist/{chunk-XJCMXB2E.cjs → chunk-2MV4HBKC.cjs} +187 -61
- package/dist/chunk-2MV4HBKC.cjs.map +1 -0
- package/dist/{chunk-J73SJHR3.cjs → chunk-2NPS5DQ3.cjs} +558 -341
- package/dist/chunk-2NPS5DQ3.cjs.map +1 -0
- package/dist/{chunk-QWY3VEZN.js → chunk-3MNKQZ77.js} +65 -12
- package/dist/chunk-3MNKQZ77.js.map +1 -0
- package/dist/{chunk-RZCWSXTA.cjs → chunk-43G3JT2E.cjs} +1491 -218
- package/dist/chunk-43G3JT2E.cjs.map +1 -0
- package/dist/{chunk-WQTTOGXC.cjs → chunk-4AVD7NZW.cjs} +4 -4
- package/dist/{chunk-WQTTOGXC.cjs.map → chunk-4AVD7NZW.cjs.map} +1 -1
- package/dist/{chunk-HTXVVJRA.cjs → chunk-4CVLFLOB.cjs} +112 -2
- package/dist/chunk-4CVLFLOB.cjs.map +1 -0
- package/dist/{chunk-5YEMOYSE.js → chunk-6GCKCWLB.js} +9 -3
- package/dist/chunk-6GCKCWLB.js.map +1 -0
- package/dist/{chunk-EBSPLUW4.cjs → chunk-AHCLWAM5.cjs} +28 -10
- package/dist/chunk-AHCLWAM5.cjs.map +1 -0
- package/dist/{chunk-KVILKGMS.js → chunk-AJA6LUI7.js} +36 -2
- package/dist/chunk-AJA6LUI7.js.map +1 -0
- package/dist/{chunk-EHAHVROS.cjs → chunk-CLVQBF5C.cjs} +5 -5
- package/dist/{chunk-EHAHVROS.cjs.map → chunk-CLVQBF5C.cjs.map} +1 -1
- package/dist/{chunk-34RRD7PC.js → chunk-CYFK5SY2.js} +111 -3
- package/dist/chunk-CYFK5SY2.js.map +1 -0
- package/dist/{chunk-M5LX5AOO.js → chunk-DDQVZGLO.js} +25 -12
- package/dist/chunk-DDQVZGLO.js.map +1 -0
- package/dist/{chunk-LR7YASZF.cjs → chunk-EAAHVJ4P.cjs} +3 -3
- package/dist/chunk-EAAHVJ4P.cjs.map +1 -0
- package/dist/{chunk-3PPFLFH4.js → chunk-FIFQPKBA.js} +1417 -144
- package/dist/chunk-FIFQPKBA.js.map +1 -0
- package/dist/{chunk-RIN643A3.js → chunk-GPPLSM2Z.js} +61 -2
- package/dist/chunk-GPPLSM2Z.js.map +1 -0
- package/dist/{chunk-4QADQTWS.js → chunk-KHJUSHFU.js} +231 -14
- package/dist/chunk-KHJUSHFU.js.map +1 -0
- package/dist/{chunk-NMD5VRN4.cjs → chunk-M4F66R4O.cjs} +74 -73
- package/dist/chunk-M4F66R4O.cjs.map +1 -0
- package/dist/{chunk-6NTVRDQV.cjs → chunk-M4Q47KHF.cjs} +126 -73
- package/dist/chunk-M4Q47KHF.cjs.map +1 -0
- package/dist/{chunk-LIPPNDBE.js → chunk-M5E34VG5.js} +3 -3
- package/dist/{chunk-LIPPNDBE.js.map → chunk-M5E34VG5.js.map} +1 -1
- package/dist/{chunk-53B6KDDJ.cjs → chunk-MBNQVDVC.cjs} +38 -3
- package/dist/{chunk-53B6KDDJ.cjs.map → chunk-MBNQVDVC.cjs.map} +1 -1
- package/dist/{chunk-SDGRK7EP.js → chunk-MEOHSQEH.js} +6 -5
- package/dist/chunk-MEOHSQEH.js.map +1 -0
- package/dist/{chunk-NMCRQP3Z.cjs → chunk-MTX2KVU7.cjs} +71 -70
- package/dist/chunk-MTX2KVU7.cjs.map +1 -0
- package/dist/{chunk-QNJ4ACRT.cjs → chunk-NJTXJ5AG.cjs} +37 -15
- package/dist/chunk-NJTXJ5AG.cjs.map +1 -0
- package/dist/{chunk-C4XZV6E7.cjs → chunk-NZFKROS7.cjs} +26 -20
- package/dist/chunk-NZFKROS7.cjs.map +1 -0
- package/dist/{chunk-64W6GLLZ.js → chunk-OFXOTECC.js} +24 -6
- package/dist/chunk-OFXOTECC.js.map +1 -0
- package/dist/{chunk-3YNVWKR2.cjs → chunk-OUHT5Z36.cjs} +28 -5
- package/dist/chunk-OUHT5Z36.cjs.map +1 -0
- package/dist/{chunk-EIGTWK5N.js → chunk-OWRHDUJD.js} +3 -3
- package/dist/chunk-OWRHDUJD.js.map +1 -0
- package/dist/{chunk-4MG4XKO2.js → chunk-PCPX42KL.js} +38 -3
- package/dist/{chunk-4MG4XKO2.js.map → chunk-PCPX42KL.js.map} +1 -1
- package/dist/{chunk-GW32KPCU.cjs → chunk-R24DI24X.cjs} +61 -2
- package/dist/chunk-R24DI24X.cjs.map +1 -0
- package/dist/{chunk-XVWCOTR6.js → chunk-SFQWJMKT.js} +7 -6
- package/dist/chunk-SFQWJMKT.js.map +1 -0
- package/dist/{chunk-6BKTCEUP.cjs → chunk-TY3TLZAW.cjs} +36 -2
- package/dist/chunk-TY3TLZAW.cjs.map +1 -0
- package/dist/{chunk-DM3LMRBC.js → chunk-UKPGSAZW.js} +187 -61
- package/dist/chunk-UKPGSAZW.js.map +1 -0
- package/dist/{chunk-NH2O2AUR.js → chunk-WC5FFSHB.js} +26 -4
- package/dist/chunk-WC5FFSHB.js.map +1 -0
- package/dist/{chunk-CLL7RUQV.cjs → chunk-WWOLFXTX.cjs} +40 -18
- package/dist/chunk-WWOLFXTX.cjs.map +1 -0
- package/dist/{chunk-JBSYC7BB.cjs → chunk-X2BWAZAU.cjs} +66 -53
- package/dist/chunk-X2BWAZAU.cjs.map +1 -0
- package/dist/{chunk-GOLDJNMZ.js → chunk-XBTEO4OB.js} +28 -5
- package/dist/chunk-XBTEO4OB.js.map +1 -0
- package/dist/{chunk-EPOXXJBK.js → chunk-YU2CUNFO.js} +3 -3
- package/dist/{chunk-EPOXXJBK.js.map → chunk-YU2CUNFO.js.map} +1 -1
- package/dist/constants.cjs +4 -4
- package/dist/constants.js +1 -1
- package/dist/engine.cjs +29 -30
- package/dist/engine.d.cts +6 -6
- package/dist/engine.d.ts +6 -6
- package/dist/engine.js +19 -20
- package/dist/format.cjs +7 -8
- package/dist/format.cjs.map +1 -1
- package/dist/format.js +2 -3
- package/dist/format.js.map +1 -1
- package/dist/index.cjs +28 -29
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +6 -6
- package/dist/index.d.ts +6 -6
- package/dist/index.js +20 -21
- package/dist/index.js.map +1 -1
- package/dist/language.cjs +74 -12
- package/dist/language.cjs.map +1 -1
- package/dist/language.d.cts +6 -6
- package/dist/language.d.ts +6 -6
- package/dist/language.js +68 -6
- package/dist/language.js.map +1 -1
- package/dist/lexer.cjs +19 -19
- package/dist/lexer.d.cts +3 -3
- package/dist/lexer.d.ts +3 -3
- package/dist/lexer.js +7 -7
- package/dist/normalizer.cjs +12 -12
- package/dist/normalizer.d.cts +4 -210
- package/dist/normalizer.d.ts +4 -210
- package/dist/normalizer.js +6 -6
- package/dist/packages.cjs +36 -37
- package/dist/packages.d.cts +5 -47
- package/dist/packages.d.ts +5 -47
- package/dist/packages.js +13 -14
- package/dist/parser.cjs +12 -12
- package/dist/parser.d.cts +3 -2
- package/dist/parser.d.ts +3 -2
- package/dist/parser.js +4 -4
- package/dist/resolvers.d.cts +1 -1
- package/dist/resolvers.d.ts +1 -1
- package/dist/uom.cjs +14 -14
- package/dist/uom.d.cts +12 -4
- package/dist/uom.d.ts +12 -4
- package/dist/uom.js +4 -4
- package/dist/vm.cjs +17 -17
- package/dist/vm.js +6 -6
- package/package.json +2 -1
- package/dist/NormalizerRule-BrVoVjmP.d.cts +0 -163
- package/dist/NormalizerRule-CEjf1FyD.d.ts +0 -163
- package/dist/chunk-34RRD7PC.js.map +0 -1
- package/dist/chunk-3PPFLFH4.js.map +0 -1
- package/dist/chunk-3YNVWKR2.cjs.map +0 -1
- package/dist/chunk-4QADQTWS.js.map +0 -1
- package/dist/chunk-5YEMOYSE.js.map +0 -1
- package/dist/chunk-64W6GLLZ.js.map +0 -1
- package/dist/chunk-6BKTCEUP.cjs.map +0 -1
- package/dist/chunk-6NTVRDQV.cjs.map +0 -1
- package/dist/chunk-C4XZV6E7.cjs.map +0 -1
- package/dist/chunk-CLL7RUQV.cjs.map +0 -1
- package/dist/chunk-DM3LMRBC.js.map +0 -1
- package/dist/chunk-EBSPLUW4.cjs.map +0 -1
- package/dist/chunk-EIGTWK5N.js.map +0 -1
- package/dist/chunk-GOLDJNMZ.js.map +0 -1
- package/dist/chunk-GW32KPCU.cjs.map +0 -1
- package/dist/chunk-HTXVVJRA.cjs.map +0 -1
- package/dist/chunk-HWSZJQCI.js.map +0 -1
- package/dist/chunk-J73SJHR3.cjs.map +0 -1
- package/dist/chunk-JBSYC7BB.cjs.map +0 -1
- package/dist/chunk-KVILKGMS.js.map +0 -1
- package/dist/chunk-LR7YASZF.cjs.map +0 -1
- package/dist/chunk-M5LX5AOO.js.map +0 -1
- package/dist/chunk-NH2O2AUR.js.map +0 -1
- package/dist/chunk-NMCRQP3Z.cjs.map +0 -1
- package/dist/chunk-NMD5VRN4.cjs.map +0 -1
- package/dist/chunk-OT6OJY7C.cjs +0 -114
- package/dist/chunk-OT6OJY7C.cjs.map +0 -1
- package/dist/chunk-QNJ4ACRT.cjs.map +0 -1
- package/dist/chunk-QWY3VEZN.js.map +0 -1
- package/dist/chunk-RFYD5TJE.js +0 -111
- package/dist/chunk-RFYD5TJE.js.map +0 -1
- package/dist/chunk-RIN643A3.js.map +0 -1
- package/dist/chunk-RZCWSXTA.cjs.map +0 -1
- package/dist/chunk-SDGRK7EP.js.map +0 -1
- package/dist/chunk-XJCMXB2E.cjs.map +0 -1
- package/dist/chunk-XVWCOTR6.js.map +0 -1
|
@@ -0,0 +1,377 @@
|
|
|
1
|
+
import { T as Token } from './Token-BzG5G4ja.js';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* NormalizerRule, pluggable token normalization rule for the
|
|
5
|
+
* TokenNormalizer post-lexer pass.
|
|
6
|
+
*
|
|
7
|
+
* ## Purpose
|
|
8
|
+
* After the ExpressionLexer produces raw tokens (numbers, identifiers,
|
|
9
|
+
* operators, etc.), the TokenNormalizer applies domain-specific rules
|
|
10
|
+
* to transform the token stream before parsing. This keeps the lexer
|
|
11
|
+
* focused on single-token production and moves multi-token pattern
|
|
12
|
+
* matching into a dedicated normalization layer.
|
|
13
|
+
*
|
|
14
|
+
* ## How rules work
|
|
15
|
+
* Rules are applied in priority order (highest first). At each token
|
|
16
|
+
* position, the normalizer tries every rule in priority order until one
|
|
17
|
+
* matches. Matched tokens are consumed and replaced; unmatched tokens
|
|
18
|
+
* pass through unchanged.
|
|
19
|
+
*
|
|
20
|
+
* ## What rules can do
|
|
21
|
+
* - **Phrase fusion**: Merge consecutive words into compound tokens
|
|
22
|
+
* (e.g., `"to" "the" "power" "of"` → `CARET`)
|
|
23
|
+
* - **Implicit operators**: Insert missing operators between tokens
|
|
24
|
+
* (e.g., `NUMBER IDENT` → `NUMBER STAR IDENT`)
|
|
25
|
+
* - **Domain transformations**: Coalesce item names, currency pairs, etc.
|
|
26
|
+
*
|
|
27
|
+
* ## Why a separate file?
|
|
28
|
+
* This is a duplicate-free copy of the interface defined in
|
|
29
|
+
* TokenNormalizer.ts. Storing it in a separate file avoids circular
|
|
30
|
+
* imports, TokenNormalizer imports NormalizerRule, and rule factories
|
|
31
|
+
* import TokenNormalizer's `createFusedToken`.
|
|
32
|
+
*
|
|
33
|
+
* @module NormalizerRule
|
|
34
|
+
*/
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Result of a successful rule match attempt against the token stream.
|
|
38
|
+
*
|
|
39
|
+
* When a {@link NormalizerRule.match} function finds a pattern at the
|
|
40
|
+
* current position, it returns a NormalizerMatch describing how many
|
|
41
|
+
* tokens to consume and what to replace them with.
|
|
42
|
+
*
|
|
43
|
+
* @example
|
|
44
|
+
* ```ts
|
|
45
|
+
* // The phrase "to the power of" (5 tokens) becomes a single CARET token
|
|
46
|
+
* const match: NormalizerMatch = {
|
|
47
|
+
* consumed: 5,
|
|
48
|
+
* replacement: [caretToken],
|
|
49
|
+
* };
|
|
50
|
+
* ```
|
|
51
|
+
*/
|
|
52
|
+
interface NormalizerMatch {
|
|
53
|
+
/**
|
|
54
|
+
* Number of tokens consumed from the stream at the match position.
|
|
55
|
+
* Must be ≥ 1, a match always advances the cursor.
|
|
56
|
+
*/
|
|
57
|
+
consumed: number;
|
|
58
|
+
/**
|
|
59
|
+
* Replacement tokens to insert at the match position.
|
|
60
|
+
* May be empty (deletion), a single token (fusion), or multiple
|
|
61
|
+
* tokens (expansion/splitting).
|
|
62
|
+
*/
|
|
63
|
+
replacement: Token[];
|
|
64
|
+
/**
|
|
65
|
+
* Human-readable rule name for diagnostic fusion tracking.
|
|
66
|
+
* When set, the normalizer uses this instead of the rule's `name`
|
|
67
|
+
* in {@link TokenFusion} records. Used by {@link PhraseTrie} to
|
|
68
|
+
* report which specific phrase matched (e.g., "phrase:to the power of").
|
|
69
|
+
*/
|
|
70
|
+
ruleName?: string;
|
|
71
|
+
}
|
|
72
|
+
/**
|
|
73
|
+
* A pluggable normalization rule registered with the TokenNormalizer.
|
|
74
|
+
*
|
|
75
|
+
* Each rule has a {@link name}, {@link priority}, and {@link match} function.
|
|
76
|
+
* The match function receives the current token stream and a position,
|
|
77
|
+
* and returns a {@link NormalizerMatch} on success or `null` on failure.
|
|
78
|
+
*
|
|
79
|
+
* ## Priority ordering
|
|
80
|
+
* Higher priority rules are tried first at each position. This allows
|
|
81
|
+
* long phrases (priority 100, e.g. "to the power of") to match before
|
|
82
|
+
* shorter fragments (priority 80, e.g. "power of").
|
|
83
|
+
*
|
|
84
|
+
* ## Match contract
|
|
85
|
+
* - Must be pure (no side effects, no mutation of input tokens)
|
|
86
|
+
* - Must return `null` for any position that doesn't match
|
|
87
|
+
* - Consumed tokens must be consecutive starting at `pos`
|
|
88
|
+
* - Replacement tokens must be valid for downstream parsing
|
|
89
|
+
*
|
|
90
|
+
* @example
|
|
91
|
+
* ```ts
|
|
92
|
+
* // A phrase fusion rule that converts "to the power of" into CARET
|
|
93
|
+
* const phraseRule: NormalizerRule = {
|
|
94
|
+
* name: 'phrase:to the power of',
|
|
95
|
+
* priority: 100,
|
|
96
|
+
* match: (tokens, pos) => {
|
|
97
|
+
* if (pos + 4 > tokens.length) return null;
|
|
98
|
+
* const phrase = tokens.slice(pos, pos + 5)
|
|
99
|
+
* .map(t => t.value.toLowerCase()).join(' ');
|
|
100
|
+
* if (phrase === 'to the power of') {
|
|
101
|
+
* return {
|
|
102
|
+
* consumed: 5,
|
|
103
|
+
* replacement: [createFusedToken('CARET', 'to the power of', tokens.slice(pos, pos + 5))],
|
|
104
|
+
* };
|
|
105
|
+
* }
|
|
106
|
+
* return null;
|
|
107
|
+
* },
|
|
108
|
+
* };
|
|
109
|
+
* ```
|
|
110
|
+
*/
|
|
111
|
+
interface NormalizerRule {
|
|
112
|
+
/**
|
|
113
|
+
* Human-readable name for debugging and diagnostic display.
|
|
114
|
+
* Convention: `"category:description"`, e.g. `"phrase:to the power of"`.
|
|
115
|
+
*/
|
|
116
|
+
readonly name: string;
|
|
117
|
+
/**
|
|
118
|
+
* Priority for ordering rules. Higher values are tried first.
|
|
119
|
+
* Recommended ranges:
|
|
120
|
+
* - 100: Long multi-word phrase fusion (e.g., "to the power of")
|
|
121
|
+
* - 80: Short phrase fusion (e.g., "power of", "times by")
|
|
122
|
+
* - 50: Implicit operator insertion (e.g., implicit multiply)
|
|
123
|
+
* - 20: Domain-specific transformations
|
|
124
|
+
*/
|
|
125
|
+
readonly priority: number;
|
|
126
|
+
/**
|
|
127
|
+
* Attempt to match a pattern starting at position `pos` in the token stream.
|
|
128
|
+
*
|
|
129
|
+
* @param tokens - The current token stream (may be partially normalized from prior passes)
|
|
130
|
+
* @param pos - The current position to attempt matching from
|
|
131
|
+
* @returns A {@link NormalizerMatch} if the pattern is found, or `null` if no match
|
|
132
|
+
*/
|
|
133
|
+
match(tokens: Token[], pos: number): NormalizerMatch | null;
|
|
134
|
+
}
|
|
135
|
+
/**
|
|
136
|
+
* Record of a token fusion event performed by the normalizer.
|
|
137
|
+
*
|
|
138
|
+
* When a rule merges multiple source tokens into fewer replacement tokens,
|
|
139
|
+
* the normalizer fires a {@link NormalizerOptions.onFusion | fusion callback}
|
|
140
|
+
* with this record. The playground uses these records to render the
|
|
141
|
+
* fusion detail table showing exactly which tokens were merged and by
|
|
142
|
+
* which rule.
|
|
143
|
+
*
|
|
144
|
+
* @example
|
|
145
|
+
* ```ts
|
|
146
|
+
* // "to" "the" "power" "of" fused into CARET "^"
|
|
147
|
+
* const fusion: TokenFusion = {
|
|
148
|
+
* rule: "phrase:to the power of",
|
|
149
|
+
* sourceTokens: [toToken, theToken, powerToken, ofToken],
|
|
150
|
+
* fusedToken: caretToken,
|
|
151
|
+
* };
|
|
152
|
+
* ```
|
|
153
|
+
*/
|
|
154
|
+
interface TokenFusion {
|
|
155
|
+
/** The name of the rule that triggered this fusion (e.g., "phrase:to the power of") */
|
|
156
|
+
rule: string;
|
|
157
|
+
/** The original tokens before fusion, always ≥ 2 tokens */
|
|
158
|
+
sourceTokens: Token[];
|
|
159
|
+
/** The resulting fused token with its new type and combined value */
|
|
160
|
+
fusedToken: Token;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* TokenNormalizer, post-lexer token normalization pass.
|
|
165
|
+
*
|
|
166
|
+
* ## Purpose
|
|
167
|
+
* Applies domain-specific {@link NormalizerRule | NormalizerRules} to the raw
|
|
168
|
+
* token stream produced by the {@link ExpressionLexer}. This keeps the lexer
|
|
169
|
+
* slim and focused on single-token production, while multi-token pattern
|
|
170
|
+
* matching (phrases, implicit operators, domain merges) lives here.
|
|
171
|
+
*
|
|
172
|
+
* ## What rules can do
|
|
173
|
+
* - **Phrase fusion**: Merge consecutive words into compound tokens
|
|
174
|
+
* (e.g., `IDENT + ... + IDENT` → `CARET`)
|
|
175
|
+
* - **Implicit operator insertion**: Insert missing operators between tokens
|
|
176
|
+
* (e.g., `NUMBER IDENT` → `NUMBER STAR IDENT`)
|
|
177
|
+
* - **Domain-specific transformations**: Coalesce item names, currency pairs,
|
|
178
|
+
* percentage syntax, etc.
|
|
179
|
+
*
|
|
180
|
+
* ## Architecture
|
|
181
|
+
* Providers register NormalizerRules alongside Parselets and OpCode handlers
|
|
182
|
+
* via {@link IEnginePackage.normalizerRules}. The normalizer applies them
|
|
183
|
+
* greedily left-to-right in multiple passes with safety limits.
|
|
184
|
+
*
|
|
185
|
+
* @module TokenNormalizer
|
|
186
|
+
*/
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Configuration options for the normalization pass.
|
|
190
|
+
*
|
|
191
|
+
* These control safety limits and diagnostic callbacks. The defaults
|
|
192
|
+
* are chosen to be generous enough for any realistic expression while
|
|
193
|
+
* preventing runaway token expansion from recursive rules.
|
|
194
|
+
*/
|
|
195
|
+
interface NormalizerOptions {
|
|
196
|
+
/**
|
|
197
|
+
* Maximum number of full passes over the token stream before bailing out.
|
|
198
|
+
* Prevents infinite loops from recursive rule chains.
|
|
199
|
+
* @default 100
|
|
200
|
+
*/
|
|
201
|
+
maxPasses?: number;
|
|
202
|
+
/**
|
|
203
|
+
* Maximum number of tokens allowed after normalization.
|
|
204
|
+
* If exceeded, an Error is thrown rather than passing a bloated stream
|
|
205
|
+
* to the parser.
|
|
206
|
+
* @default 10000
|
|
207
|
+
*/
|
|
208
|
+
maxTokens?: number;
|
|
209
|
+
/**
|
|
210
|
+
* Callback invoked for each fusion event during normalization.
|
|
211
|
+
* Used by diagnostic mode to populate {@link NormalizerOutput.fusions}.
|
|
212
|
+
* When `undefined`, fusions are still tracked internally but no callbacks fire.
|
|
213
|
+
*/
|
|
214
|
+
onFusion?: (fusion: TokenFusion) => void;
|
|
215
|
+
}
|
|
216
|
+
/**
|
|
217
|
+
* Creates a new normalized token from fused source tokens.
|
|
218
|
+
*
|
|
219
|
+
* The fused token inherits position information (offset, line, column)
|
|
220
|
+
* from the first source token, which preserves source-map accuracy
|
|
221
|
+
* for error messages and diagnostic highlighting.
|
|
222
|
+
*
|
|
223
|
+
* It also records where the source text ENDS, on `sourceEnd`. The start alone
|
|
224
|
+
* is not enough to describe the span a fusion covers, because `text` is the
|
|
225
|
+
* replacement rather than the original: `10 frames` fuses into a FRAME_COUNT
|
|
226
|
+
* whose text is `10`, and a timecode fuses into a token whose text is a
|
|
227
|
+
* comma-separated tuple that appears nowhere in the line. Anything painting the
|
|
228
|
+
* line needs both ends, and only this function is in a position to know them.
|
|
229
|
+
*
|
|
230
|
+
* @param type - The new token type (e.g., "CARET", "TIMES_BY")
|
|
231
|
+
* @param text - The combined text representation (e.g., "to the power of")
|
|
232
|
+
* @param sourceTokens - The original tokens being fused (at least 2)
|
|
233
|
+
* @returns A new {@link LexerToken} with the fused type and combined text
|
|
234
|
+
*/
|
|
235
|
+
declare function createFusedToken(type: string, text: string, sourceTokens: Token[]): Token;
|
|
236
|
+
/**
|
|
237
|
+
* Token normalizer: applies {@link NormalizerRule | NormalizerRules} to a token stream.
|
|
238
|
+
*
|
|
239
|
+
* ## Lifecycle
|
|
240
|
+
* 1. **Registration**: Rules are added via {@link register} and sorted by priority
|
|
241
|
+
* 2. **Normalization**: {@link normalize} applies rules greedily left-to-right
|
|
242
|
+
* 3. **Cleanup**: {@link clear} or {@link unregister} removes rules
|
|
243
|
+
*
|
|
244
|
+
* ## Normalization algorithm
|
|
245
|
+
* The normalizer uses a greedy left-to-right multi-pass algorithm:
|
|
246
|
+
* - At each token position, rules are tried in priority order (highest first)
|
|
247
|
+
* - When a rule matches, matched tokens are consumed and replaced
|
|
248
|
+
* - Processing continues from the replacement position
|
|
249
|
+
* - Multiple passes handle cascading matches (one rule's output triggers another)
|
|
250
|
+
* - Safety limits ({@link NormalizerOptions.maxPasses}) prevent infinite loops
|
|
251
|
+
*
|
|
252
|
+
* @example
|
|
253
|
+
* ```ts
|
|
254
|
+
* const normalizer = new TokenNormalizer();
|
|
255
|
+
* normalizer.register(phraseRule); // "to the power of" → CARET
|
|
256
|
+
* normalizer.register(implicitMultRule); // "2 x" → "2 * x"
|
|
257
|
+
* const normalized = normalizer.normalize(rawTokens);
|
|
258
|
+
* ```
|
|
259
|
+
*/
|
|
260
|
+
declare class TokenNormalizer {
|
|
261
|
+
/** Registered rules, unsorted, the source of truth. */
|
|
262
|
+
private rules;
|
|
263
|
+
/**
|
|
264
|
+
* Priority-sorted copy of {@link rules}, rebuilt lazily on the next
|
|
265
|
+
* {@link normalize} call after a mutation. Rules are registered once at
|
|
266
|
+
* engine/package-registration time and essentially never change during a
|
|
267
|
+
* session, but normalize() runs on every keystroke-driven evaluation, an
|
|
268
|
+
* earlier version re-sorted a fresh copy of `rules` on every single call,
|
|
269
|
+
* which meant every keystroke paid for an allocation + sort of a list that
|
|
270
|
+
* had usually not changed since the last one. `null` means "stale, rebuild
|
|
271
|
+
* on next use"; {@link register}/{@link unregister}/{@link clear} all
|
|
272
|
+
* invalidate it.
|
|
273
|
+
*/
|
|
274
|
+
private sortedRulesCache;
|
|
275
|
+
/**
|
|
276
|
+
* Phrase trie for single-pass multi-word phrase fusion.
|
|
277
|
+
* Tried at each token position BEFORE other rules, the trie walk
|
|
278
|
+
* is O(depth) vs O(R × W) for separate rule matching.
|
|
279
|
+
*/
|
|
280
|
+
private phraseTrie;
|
|
281
|
+
/** Merged options with defaults applied. */
|
|
282
|
+
private options;
|
|
283
|
+
/**
|
|
284
|
+
* @param options - Configuration overrides for safety limits and diagnostic callbacks
|
|
285
|
+
*/
|
|
286
|
+
constructor(options?: NormalizerOptions);
|
|
287
|
+
/**
|
|
288
|
+
* Register a normalization rule.
|
|
289
|
+
*
|
|
290
|
+
* Rules are sorted by priority (descending) on each {@link normalize} call.
|
|
291
|
+
* Multiple rules can share the same priority, they are tried in registration
|
|
292
|
+
* order when priorities are equal.
|
|
293
|
+
*
|
|
294
|
+
* @param rule - The rule to register
|
|
295
|
+
*/
|
|
296
|
+
register(rule: NormalizerRule): void;
|
|
297
|
+
/**
|
|
298
|
+
* Unregister a normalization rule by its {@link NormalizerRule.name | name}.
|
|
299
|
+
*
|
|
300
|
+
* If multiple rules share the same name, all are removed. This is safe to
|
|
301
|
+
* call with a name that doesn't match any rule, it simply has no effect.
|
|
302
|
+
*
|
|
303
|
+
* @param ruleName - The name of the rule to remove
|
|
304
|
+
*/
|
|
305
|
+
unregister(ruleName: string): void;
|
|
306
|
+
/**
|
|
307
|
+
* Remove all registered rules, resetting the normalizer to its initial state.
|
|
308
|
+
* Also clears the phrase trie.
|
|
309
|
+
*/
|
|
310
|
+
clear(): void;
|
|
311
|
+
/**
|
|
312
|
+
* Priority-sorted view of {@link rules} (descending priority; registration
|
|
313
|
+
* order preserved for ties, since {@link Array.prototype.sort} is stable).
|
|
314
|
+
* Cached until the next mutation. See {@link sortedRulesCache}.
|
|
315
|
+
*/
|
|
316
|
+
private getSortedRules;
|
|
317
|
+
/**
|
|
318
|
+
* Get the number of currently registered rules (excludes phrase trie entries).
|
|
319
|
+
*/
|
|
320
|
+
get ruleCount(): number;
|
|
321
|
+
/**
|
|
322
|
+
* Register a multi-word phrase for fusion into a single compound token.
|
|
323
|
+
*
|
|
324
|
+
* This is the preferred way to add phrase patterns. It inserts into the
|
|
325
|
+
* internal {@link PhraseTrie}, which collapses all phrase rules into a
|
|
326
|
+
* single O(depth) trie walk per position, no separate rule scanning.
|
|
327
|
+
*
|
|
328
|
+
* @param phrase - Multi-word phrase (e.g., "to the power of", "abyssal whip")
|
|
329
|
+
* @param tokenType - Target token type after fusion (e.g., "CARET", "ITEM")
|
|
330
|
+
*/
|
|
331
|
+
addPhrase(phrase: string, tokenType: string): void;
|
|
332
|
+
/**
|
|
333
|
+
* Check whether a word can start any registered phrase.
|
|
334
|
+
*
|
|
335
|
+
* Used by {@link implicitMultiplyRule} to suppress `*` insertion
|
|
336
|
+
* before phrase-starting identifiers (e.g., "2 power of 3" → `2 ^ 3`,
|
|
337
|
+
* not `2 * power of 3`). Delegates to {@link PhraseTrie.canStart}.
|
|
338
|
+
*/
|
|
339
|
+
/**
|
|
340
|
+
* Get all registered phrases and their target token types.
|
|
341
|
+
*
|
|
342
|
+
* Exposes the full phrase trie structure for diagnostic rendering
|
|
343
|
+
* in the playground's NormalizerTab. Returns ALL registered phrases,
|
|
344
|
+
* not just the ones that matched in the last evaluation.
|
|
345
|
+
*/
|
|
346
|
+
getPhrases(): Record<string, string>;
|
|
347
|
+
canStartPhrase(word: string): boolean;
|
|
348
|
+
/**
|
|
349
|
+
* Normalize a token stream by applying all registered rules.
|
|
350
|
+
*
|
|
351
|
+
* ## Algorithm
|
|
352
|
+
* Applies rules greedily left-to-right in multiple passes:
|
|
353
|
+
* 1. Sort rules by priority (descending)
|
|
354
|
+
* 2. Walk the token stream left to right
|
|
355
|
+
* 3. At each position, try rules in priority order
|
|
356
|
+
* 4. On match: consume matched tokens, insert replacements, restart from insert point
|
|
357
|
+
* 5. On no match: pass token through unchanged
|
|
358
|
+
* 6. Repeat until a full pass produces no changes, or maxPasses is reached
|
|
359
|
+
*
|
|
360
|
+
* ## Fusion tracking
|
|
361
|
+
* When a rule consumes more tokens than it produces, the normalizer calls
|
|
362
|
+
* `onFusion` with a {@link TokenFusion} record for diagnostic collection.
|
|
363
|
+
* This populates {@link NormalizerOutput.fusions} in the playground pipeline view.
|
|
364
|
+
*
|
|
365
|
+
* ## Safety
|
|
366
|
+
* If the normalized token count exceeds {@link NormalizerOptions.maxTokens},
|
|
367
|
+
* an Error is thrown to prevent memory exhaustion from runaway rule expansion.
|
|
368
|
+
*
|
|
369
|
+
* @param tokens - Raw tokens from the lexer
|
|
370
|
+
* @param onFusion - Optional fusion callback (overrides {@link NormalizerOptions.onFusion})
|
|
371
|
+
* @returns Normalized tokens ready for parsing
|
|
372
|
+
* @throws {Error} If the normalized token count exceeds maxTokens
|
|
373
|
+
*/
|
|
374
|
+
normalize(tokens: Token[], onFusion?: (fusion: TokenFusion) => void): Token[];
|
|
375
|
+
}
|
|
376
|
+
|
|
377
|
+
export { type NormalizerMatch as N, type TokenFusion as T, type NormalizerRule as a, type NormalizerOptions as b, TokenNormalizer as c, createFusedToken as d };
|
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
import { getTokenCategory } from './chunk-
|
|
2
|
-
import { ExpressionLexer } from './chunk-
|
|
3
|
-
import { getLocale } from './chunk-
|
|
4
|
-
import { knownUnits } from './chunk-
|
|
1
|
+
import { getTokenCategory } from './chunk-AJA6LUI7.js';
|
|
2
|
+
import { ExpressionLexer } from './chunk-SFQWJMKT.js';
|
|
3
|
+
import { getLocale } from './chunk-XBTEO4OB.js';
|
|
4
|
+
import { knownUnits } from './chunk-M5E34VG5.js';
|
|
5
5
|
|
|
6
6
|
// src/lexer/LexerState.ts
|
|
7
7
|
var LexerState = /* @__PURE__ */ ((LexerState2) => {
|
|
@@ -143,24 +143,45 @@ var Lexer = class {
|
|
|
143
143
|
}
|
|
144
144
|
return this.collectHighlightTokens(lineText);
|
|
145
145
|
}
|
|
146
|
-
|
|
146
|
+
/**
|
|
147
|
+
* The same tokens {@link getHighlightTokens} reduces, before reduction.
|
|
148
|
+
*
|
|
149
|
+
* Exists because normalization operates on tokens, not on the flattened
|
|
150
|
+
* shape, and a consumer that wants phrase-fused highlighting has to run the
|
|
151
|
+
* normalizer between the two. See `LanguageService.getSemanticTokens`.
|
|
152
|
+
*
|
|
153
|
+
* @param lineText - One line of source.
|
|
154
|
+
* @returns Every token on the line that is worth painting, unreduced.
|
|
155
|
+
*/
|
|
156
|
+
getHighlightTokenObjects(lineText) {
|
|
157
|
+
const classification = this.expressionLexer.classifyLine(lineText);
|
|
158
|
+
if (classification.skip && lineText.startsWith("> ")) {
|
|
159
|
+
return this.collectTokenObjects(lineText.slice(2));
|
|
160
|
+
}
|
|
161
|
+
if (classification.skip) return [];
|
|
162
|
+
return this.collectTokenObjects(lineText);
|
|
163
|
+
}
|
|
164
|
+
collectTokenObjects(lineText) {
|
|
147
165
|
this.resetExpression(lineText);
|
|
148
166
|
const result = [];
|
|
149
167
|
for (const token of this) {
|
|
150
168
|
if (token.type === "WS" || token.type === "NEWLINE") continue;
|
|
151
169
|
if (token.type.startsWith("MD_")) continue;
|
|
152
170
|
if (token.type === "INLINE_SOLVE_START" || token.type === "BACKTICK_CLOSE") continue;
|
|
153
|
-
result.push(
|
|
154
|
-
type: token.type,
|
|
155
|
-
value: token.value,
|
|
156
|
-
offset: token.offset,
|
|
157
|
-
col: token.col,
|
|
158
|
-
length: token.value.length,
|
|
159
|
-
category: getTokenCategory(token.type)
|
|
160
|
-
});
|
|
171
|
+
result.push(token);
|
|
161
172
|
}
|
|
162
173
|
return result;
|
|
163
174
|
}
|
|
175
|
+
collectHighlightTokens(lineText) {
|
|
176
|
+
return this.collectTokenObjects(lineText).map((token) => ({
|
|
177
|
+
type: token.type,
|
|
178
|
+
value: token.value,
|
|
179
|
+
offset: token.offset,
|
|
180
|
+
col: token.col,
|
|
181
|
+
length: token.value.length,
|
|
182
|
+
category: getTokenCategory(token.type)
|
|
183
|
+
}));
|
|
184
|
+
}
|
|
164
185
|
};
|
|
165
186
|
var sharedLexer = new Lexer("en", void 0);
|
|
166
187
|
|
|
@@ -296,6 +317,7 @@ var BUILTIN_PHRASES = {
|
|
|
296
317
|
"decrease by": "DECREASE_BY",
|
|
297
318
|
"times by": "TIMES_BY",
|
|
298
319
|
"multiply by": "MULTIPLY_BY",
|
|
320
|
+
"multiplied by": "MULTIPLY_BY",
|
|
299
321
|
"divide by": "DIVIDE_BY"
|
|
300
322
|
};
|
|
301
323
|
function buildTokenLookup(localeCode = "en") {
|
|
@@ -307,5 +329,5 @@ function buildTokenLookup(localeCode = "en") {
|
|
|
307
329
|
}
|
|
308
330
|
|
|
309
331
|
export { Lexer, LexerState, buildTokenLookup, sharedLexer };
|
|
310
|
-
//# sourceMappingURL=chunk-
|
|
311
|
-
//# sourceMappingURL=chunk-
|
|
332
|
+
//# sourceMappingURL=chunk-2CS6OMZK.js.map
|
|
333
|
+
//# sourceMappingURL=chunk-2CS6OMZK.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["../src/lexer/LexerState.ts","../src/lexer/Lexer.ts","../src/lexer/TokenClassRegistry.ts","../src/lexer/tokenRegistration.ts"],"names":["LexerState"],"mappings":";;;;;;AAMO,IAAK,UAAA,qBAAAA,WAAAA,KAAL;AACN,EAAAA,YAAA,MAAA,CAAA,GAAO,MAAA;AACP,EAAAA,YAAA,QAAA,CAAA,GAAS,QAAA;AACT,EAAAA,YAAA,QAAA,CAAA,GAAS,QAAA;AAHE,EAAA,OAAAA,WAAAA;AAAA,CAAA,EAAA,UAAA,IAAA,EAAA;;;ACaL,IAAM,QAAN,MAAY;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAkBjB,WAAA,CAAY,UAAA,GAAa,IAAA,EAAM,WAAA,EAA2B;AAf1D,IAAA,IAAA,CAAQ,YAAA,GAAA,MAAA;AAER,IAAA,IAAA,CAAQ,SAAA,GAAY,KAAA;AAIpB;AAAA;AAAA,IAAA,IAAA,CAAQ,SAAkB,EAAC;AAC3B,IAAA,IAAA,CAAQ,QAAA,GAAmB,CAAA;AAYzB,IAAA,IAAA,CAAK,eAAA,GAAkB,IAAI,eAAA,CAAgB,UAAA,EAAY,WAAW,CAAA;AAAA,EACpE;AAAA,EAEA,KAAA,CAAM,OAAe,KAAA,EAA0B;AAC7C,IAAA,MAAM,QAAA,GAAW,KAAA,IAAA,MAAA;AACjB,IAAA,IAAA,CAAK,YAAA,GAAe,QAAA;AACpB,IAAA,IAAA,CAAK,SAAA,GAAY,KAAA;AACjB,IAAA,IAAA,CAAK,WAAA,GAAc,MAAA;AAKnB,IAAA,IAAI,QAAA,KAAA,MAAA,aAA8B;AAChC,MAAA,MAAM,cAAA,GAAiB,IAAA,CAAK,eAAA,CAAgB,YAAA,CAAa,KAAK,CAAA;AAC9D,MAAA,IAAI,eAAe,IAAA,EAAM;AACvB,QAAA,IAAA,CAAK,SAAS,EAAC;AACf,QAAA,IAAA,CAAK,QAAA,GAAW,CAAA;AAChB,QAAA;AAAA,MACF;AAEA,MAAA,IAAA,CAAK,eAAA,CAAgB,MAAM,KAAK,CAAA;AAChC,MAAA,IAAA,CAAK,MAAA,GAAS,IAAA,CAAK,eAAA,CAAgB,WAAA,EAAY;AAC/C,MAAA,IAAA,CAAK,QAAA,GAAW,CAAA;AAAA,IAClB,CAAA,MAAO;AAEL,MAAA,IAAA,CAAK,eAAA,CAAgB,MAAM,KAAK,CAAA;AAChC,MAAA,IAAA,CAAK,MAAA,GAAS,IAAA,CAAK,eAAA,CAAgB,WAAA,EAAY;AAC/C,MAAA,IAAA,CAAK,QAAA,GAAW,CAAA;AAAA,IAClB;AAAA,EACF;AAAA;AAAA;AAAA;AAAA;AAAA,EAMA,aAAa,QAAA,EAAsC;AACjD,IAAA,OAAO,IAAA,CAAK,eAAA,CAAgB,YAAA,CAAa,QAAQ,CAAA;AAAA,EACnD;AAAA;AAAA;AAAA;AAAA;AAAA,EAMA,iBAAiB,QAAA,EAAkB;AACjC,IAAA,OAAO,IAAA,CAAK,eAAA,CAAgB,gBAAA,CAAiB,QAAQ,CAAA;AAAA,EACvD;AAAA;AAAA;AAAA;AAAA;AAAA,EAMA,WAAA,GAAsC;AACpC,IAAA,OAAO,IAAA,CAAK,gBAAgB,WAAA,EAAY;AAAA,EAC1C;AAAA,EAEA,IAAA,GAA0B;AACxB,IAAA,IAAI,KAAK,SAAA,EAAW;AAClB,MAAA,IAAA,CAAK,SAAA,GAAY,KAAA;AACjB,MAAA,OAAO,IAAA,CAAK,WAAA;AAAA,IACd;AAEA,IAAA,IAAI,IAAA,CAAK,QAAA,GAAW,IAAA,CAAK,MAAA,CAAO,MAAA,EAAQ;AACtC,MAAA,OAAO,IAAA,CAAK,MAAA,CAAO,IAAA,CAAK,QAAA,EAAU,CAAA;AAAA,IACpC;AACA,IAAA,OAAO,MAAA;AAAA,EACT;AAAA,EAEA,IAAA,GAA0B;AACxB,IAAA,IAAI,IAAA,CAAK,SAAA,EAAW,OAAO,IAAA,CAAK,WAAA;AAChC,IAAA,IAAA,CAAK,WAAA,GAAc,KAAK,IAAA,EAAK;AAC7B,IAAA,IAAA,CAAK,SAAA,GAAY,IAAA;AACjB,IAAA,OAAO,IAAA,CAAK,WAAA;AAAA,EACd;AAAA,EAEA,CAAC,MAAA,CAAO,QAAQ,CAAA,GAAqB;AACnC,IAAA,OAAO,IAAA,CAAK,MAAA,CAAO,MAAA,CAAO,QAAQ,CAAA,EAAE;AAAA,EACtC;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAQA,mBAAmB,MAAA,EAA+B;AAChD,IAAA,IAAA,CAAK,eAAA,CAAgB,mBAAmB,MAAM,CAAA;AAAA,EAChD;AAAA;AAAA;AAAA;AAAA;AAAA,EAMA,qBAAqB,MAAA,EAA+B;AAClD,IAAA,IAAA,CAAK,eAAA,CAAgB,qBAAqB,MAAM,CAAA;AAAA,EAClD;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAOA,gBAAgB,KAAA,EAAqB;AACnC,IAAA,IAAA,CAAK,YAAA,GAAA,MAAA;AACL,IAAA,IAAA,CAAK,SAAA,GAAY,KAAA;AACjB,IAAA,IAAA,CAAK,WAAA,GAAc,MAAA;AACnB,IAAA,IAAA,CAAK,eAAA,CAAgB,MAAM,KAAK,CAAA;AAChC,IAAA,IAAA,CAAK,MAAA,GAAS,IAAA,CAAK,eAAA,CAAgB,WAAA,EAAY;AAC/C,IAAA,IAAA,CAAK,QAAA,GAAW,CAAA;AAAA,EAClB;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAQA,aAAa,IAAA,EAAgC;AAC3C,IAAA,OAAO,IAAA,CAAK,eAAA,CAAgB,YAAA,CAAa,IAAI,CAAA;AAAA,EAC/C;AAAA,EAEA,QAAA,GAAuB;AACrB,IAAA,OAAO,IAAA,CAAK,YAAA;AAAA,EACd;AAAA,EAEA,SAAS,KAAA,EAAyB;AAChC,IAAA,IAAA,CAAK,YAAA,GAAe,KAAA;AAAA,EACtB;AAAA,EAEA,mBAAmB,QAAA,EAAqI;AACtJ,IAAA,MAAM,cAAA,GAAiB,IAAA,CAAK,eAAA,CAAgB,YAAA,CAAa,QAAQ,CAAA;AAKjE,IAAA,IAAI,cAAA,CAAe,IAAA,IAAQ,QAAA,CAAS,UAAA,CAAW,IAAI,CAAA,EAAG;AACpD,MAAA,OAAO,IAAA,CAAK,sBAAA,CAAuB,QAAA,CAAS,KAAA,CAAM,CAAC,CAAC,CAAA;AAAA,IACtD;AAEA,IAAA,IAAI,eAAe,IAAA,EAAM;AACvB,MAAA,OAAO,EAAC;AAAA,IACV;AAEA,IAAA,OAAO,IAAA,CAAK,uBAAuB,QAAQ,CAAA;AAAA,EAC7C;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAYA,yBAAyB,QAAA,EAA2B;AAClD,IAAA,MAAM,cAAA,GAAiB,IAAA,CAAK,eAAA,CAAgB,YAAA,CAAa,QAAQ,CAAA;AACjE,IAAA,IAAI,cAAA,CAAe,IAAA,IAAQ,QAAA,CAAS,UAAA,CAAW,IAAI,CAAA,EAAG;AACpD,MAAA,OAAO,IAAA,CAAK,mBAAA,CAAoB,QAAA,CAAS,KAAA,CAAM,CAAC,CAAC,CAAA;AAAA,IACnD;AACA,IAAA,IAAI,cAAA,CAAe,IAAA,EAAM,OAAO,EAAC;AACjC,IAAA,OAAO,IAAA,CAAK,oBAAoB,QAAQ,CAAA;AAAA,EAC1C;AAAA,EAEQ,oBAAoB,QAAA,EAA2B;AACrD,IAAA,IAAA,CAAK,gBAAgB,QAAQ,CAAA;AAC7B,IAAA,MAAM,SAAkB,EAAC;AACzB,IAAA,KAAA,MAAW,SAAS,IAAA,EAAM;AACxB,MAAA,IAAI,KAAA,CAAM,IAAA,KAAS,IAAA,IAAQ,KAAA,CAAM,SAAS,SAAA,EAAW;AACrD,MAAA,IAAI,KAAA,CAAM,IAAA,CAAK,UAAA,CAAW,KAAK,CAAA,EAAG;AAClC,MAAA,IAAI,KAAA,CAAM,IAAA,KAAS,oBAAA,IAAwB,KAAA,CAAM,SAAS,gBAAA,EAAkB;AAC5E,MAAA,MAAA,CAAO,KAAK,KAAK,CAAA;AAAA,IACnB;AACA,IAAA,OAAO,MAAA;AAAA,EACT;AAAA,EAEQ,uBAAuB,QAAA,EAAqI;AAClK,IAAA,OAAO,IAAA,CAAK,mBAAA,CAAoB,QAAQ,CAAA,CAAE,IAAI,CAAA,KAAA,MAAU;AAAA,MACtD,MAAM,KAAA,CAAM,IAAA;AAAA,MACZ,OAAO,KAAA,CAAM,KAAA;AAAA,MACb,QAAQ,KAAA,CAAM,MAAA;AAAA,MACd,KAAK,KAAA,CAAM,GAAA;AAAA,MACX,MAAA,EAAQ,MAAM,KAAA,CAAM,MAAA;AAAA,MACpB,QAAA,EAAU,gBAAA,CAAiB,KAAA,CAAM,IAAI;AAAA,KACvC,CAAE,CAAA;AAAA,EACJ;AACF;AAcO,IAAM,WAAA,GAAc,IAAI,KAAA,CAAM,IAAA,EAAM,MAAS;;;AC/I7C,IAAM,qBAAN,MAAyB;AAAA,EAAzB,WAAA,GAAA;AACL,IAAA,IAAA,CAAQ,UAAwB,EAAC;AACjC,IAAA,IAAA,CAAQ,gBAAA,GAAkD,IAAA;AAC1D,IAAA,IAAA,CAAQ,eAAA,GAAiD,IAAA;AACzD,IAAA,IAAA,CAAQ,SAAA,GAAwC,IAAA;AAAA,EAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAUhD,SAAS,UAAA,EAA8B;AACrC,IAAA,IAAA,CAAK,OAAA,CAAQ,KAAK,UAAU,CAAA;AAAA,EAC9B;AAAA;AAAA;AAAA;AAAA;AAAA,EAMA,WAAW,SAAA,EAAyB;AAClC,IAAA,IAAA,CAAK,UAAU,IAAA,CAAK,OAAA,CAAQ,OAAO,CAAA,CAAA,KAAK,CAAA,CAAE,cAAc,SAAS,CAAA;AAAA,EACnE;AAAA;AAAA;AAAA;AAAA;AAAA,EAMA,SAAA,CAAU,YAAoC,SAAA,EAA0C;AACtF,IAAA,IAAA,CAAK,gBAAA,GAAmB,UAAA;AACxB,IAAA,IAAA,CAAK,kBAAkB,SAAA,IAAa,IAAA;AAAA,EACtC;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EASA,SAAS,SAAA,EAAsC;AAC7C,IAAA,IAAA,CAAK,SAAA,GAAY,SAAA;AAAA,EACnB;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,EAYA,KAAA,GAAqB;AACnB,IAAA,MAAM,aAAA,uBAAoB,GAAA,EAAoB;AAG9C,IAAA,IAAI,KAAK,gBAAA,EAAkB;AACzB,MAAA,KAAA,MAAW,CAAC,SAAS,SAAS,CAAA,IAAK,OAAO,OAAA,CAAQ,IAAA,CAAK,gBAAgB,CAAA,EAAG;AACxE,QAAA,aAAA,CAAc,GAAA,CAAI,OAAA,CAAQ,WAAA,EAAY,EAAG,SAAS,CAAA;AAAA,MACpD;AAAA,IACF;AAGA,IAAA,MAAM,MAAA,GAAS,CAAC,GAAG,IAAA,CAAK,OAAO,CAAA,CAAE,IAAA,CAAK,CAAC,CAAA,EAAG,OAAO,CAAA,CAAE,QAAA,IAAY,CAAA,KAAM,CAAA,CAAE,YAAY,CAAA,CAAE,CAAA;AACrF,IAAA,KAAA,MAAW,MAAM,MAAA,EAAQ;AACvB,MAAA,KAAA,MAAW,OAAA,IAAW,MAAA,CAAO,IAAA,CAAK,EAAA,CAAG,QAAQ,CAAA,EAAG;AAC9C,QAAA,aAAA,CAAc,GAAA,CAAI,OAAA,CAAQ,WAAA,EAAY,EAAG,GAAG,SAAS,CAAA;AAAA,MACvD;AAAA,IACF;AAGA,IAAA,MAAM,UAAA,GAAa,KAAK,eAAA,EAAgB;AAGxC,IAAA,MAAM,gBAAA,GAAmB,KAAK,qBAAA,EAAsB;AAEpD,IAAA,OAAO;AAAA,MACL,aAAA;AAAA,MACA,UAAA;AAAA,MACA,gBAAA;AAAA,MACA,SAAA,EAAW,IAAA,CAAK,SAAA,oBAAa,IAAI,GAAA;AAAI,KACvC;AAAA,EACF;AAAA;AAAA,EAIQ,eAAA,GAAqC;AAC3C,IAAA,MAAM,IAAA,GAAmB,EAAE,QAAA,kBAAU,IAAI,KAAI,EAAE;AAG/C,IAAA,IAAI,KAAK,eAAA,EAAiB;AACxB,MAAA,KAAA,MAAW,CAAC,QAAQ,SAAS,CAAA,IAAK,OAAO,OAAA,CAAQ,IAAA,CAAK,eAAe,CAAA,EAAG;AACtE,QAAA,IAAA,CAAK,YAAA,CAAa,IAAA,EAAM,MAAA,CAAO,WAAA,IAAe,SAAS,CAAA;AAAA,MACzD;AAAA,IACF;AAGA,IAAA,KAAA,MAAW,EAAA,IAAM,KAAK,OAAA,EAAS;AAC7B,MAAA,IAAI,CAAC,GAAG,OAAA,EAAS;AACjB,MAAA,KAAA,MAAW,MAAA,IAAU,MAAA,CAAO,IAAA,CAAK,EAAA,CAAG,OAAO,CAAA,EAAG;AAC5C,QAAA,IAAA,CAAK,aAAa,IAAA,EAAM,MAAA,CAAO,WAAA,EAAY,EAAG,GAAG,SAAS,CAAA;AAAA,MAC5D;AAAA,IACF;AAEA,IAAA,OAAO,IAAA,CAAK,QAAA,CAAS,IAAA,GAAO,CAAA,GAAI,IAAA,GAAO,IAAA;AAAA,EACzC;AAAA,EAEQ,YAAA,CAAa,IAAA,EAAkB,MAAA,EAAgB,SAAA,EAAyB;AAC9E,IAAA,MAAM,KAAA,GAAQ,MAAA,CAAO,KAAA,CAAM,GAAG,CAAA;AAC9B,IAAA,IAAI,IAAA,GAAO,IAAA;AACX,IAAA,KAAA,MAAW,QAAQ,KAAA,EAAO;AACxB,MAAA,IAAI,CAAC,IAAA,CAAK,QAAA,CAAS,GAAA,CAAI,IAAI,CAAA,EAAG;AAC5B,QAAA,IAAA,CAAK,QAAA,CAAS,IAAI,IAAA,EAAM,EAAE,0BAAU,IAAI,GAAA,IAAO,CAAA;AAAA,MACjD;AACA,MAAA,IAAA,GAAO,IAAA,CAAK,QAAA,CAAS,GAAA,CAAI,IAAI,CAAA;AAAA,IAC/B;AAEA,IAAA,IAAI,CAAC,KAAK,IAAA,EAAM;AACd,MAAA,IAAA,CAAK,IAAA,GAAO,SAAA;AAAA,IACd;AAAA,EACF;AAAA,EAEQ,qBAAA,GAAqC;AAC3C,IAAA,MAAM,UAAA,uBAAiB,GAAA,EAAY;AAGnC,IAAA,IAAI,KAAK,eAAA,EAAiB;AACxB,MAAA,KAAA,MAAW,MAAA,IAAU,MAAA,CAAO,IAAA,CAAK,IAAA,CAAK,eAAe,CAAA,EAAG;AACtD,QAAA,MAAM,QAAQ,MAAA,CAAO,KAAA,CAAM,GAAG,CAAA,CAAE,CAAC,EAAE,WAAA,EAAY;AAC/C,QAAA,UAAA,CAAW,IAAI,KAAK,CAAA;AAAA,MACtB;AAAA,IACF;AAGA,IAAA,KAAA,MAAW,EAAA,IAAM,KAAK,OAAA,EAAS;AAC7B,MAAA,IAAI,CAAC,GAAG,OAAA,EAAS;AACjB,MAAA,KAAA,MAAW,MAAA,IAAU,MAAA,CAAO,IAAA,CAAK,EAAA,CAAG,OAAO,CAAA,EAAG;AAC5C,QAAA,MAAM,QAAQ,MAAA,CAAO,KAAA,CAAM,GAAG,CAAA,CAAE,CAAC,EAAE,WAAA,EAAY;AAC/C,QAAA,UAAA,CAAW,IAAI,KAAK,CAAA;AAAA,MACtB;AAAA,IACF;AAEA,IAAA,OAAO,UAAA;AAAA,EACT;AACF,CAAA;;;AClOA,IAAM,eAAA,GAA0C;AAAA,EAC9C,iBAAA,EAAmB,OAAA;AAAA,EACnB,UAAA,EAAY,OAAA;AAAA,EACZ,aAAA,EAAe,aAAA;AAAA,EACf,aAAA,EAAe,aAAA;AAAA,EACf,UAAA,EAAY,UAAA;AAAA,EACZ,aAAA,EAAe,aAAA;AAAA,EACf,eAAA,EAAiB,aAAA;AAAA,EACjB,WAAA,EAAa;AACf,CAAA;AAiBO,SAAS,gBAAA,CAAiB,aAAa,IAAA,EAAmB;AAC/D,EAAA,MAAM,MAAA,GAAkB,UAAU,UAAU,CAAA;AAC5C,EAAA,MAAM,QAAA,GAAW,IAAI,kBAAA,EAAmB;AAGxC,EAAA,QAAA,CAAS,SAAA,CAAU,MAAA,CAAO,UAAA,EAAY,eAAe,CAAA;AAGrD,EAAA,QAAA,CAAS,SAAS,UAAU,CAAA;AAG5B,EAAA,OAAO,SAAS,KAAA,EAAM;AACxB","file":"chunk-2CS6OMZK.js","sourcesContent":["/**\n * Lexer state machine modes.\n * - Main: document-level scanning with markdown classification\n * - Inline: expression embedded in markdown inline solve (`s\\`...\\``)\n * - String: inside a double-quoted string literal\n */\nexport enum LexerState {\n\tMain = \"main\",\n\tInline = \"inline\",\n\tString = \"string\",\n}\n","import { ExpressionLexer, LineClassification, LexerVocabulary, type ScanLineResult } from \"./ExpressionLexer\";\nimport { Token } from \"@solve-js/lexer/Token\";\nimport { LexerState } from \"@solve-js/lexer/LexerState\";\nimport { getTokenCategory } from \"@solve-js/language/TokenCategoryMap\";\nimport type { TokenCategory } from \"@solve-js/language/TokenCategory\";\nimport type { TokenLookup } from \"@solve-js/lexer/TokenClassRegistry\";\n\n/**\n * Public tokenizer wrapper around {@link ExpressionLexer}.\n *\n * `ExpressionLexer` does the actual character-by-character scanning;\n * `Lexer` adds a materialized-token-array streaming interface\n * (`next()`/`peek()`) plus line-classification state (`reset()`) so\n * callers can iterate a line's tokens without re-scanning on each peek.\n *\n * Each `ExpressionEngine` instance owns its own `Lexer`, and packages\n * extend it via {@link registerVocabulary} (keywords, operators, units)\n * see `IEnginePackage.lexerVocabulary`.\n */\nexport class Lexer {\n /** Expression-mode lexer (Phase A: V8-optimized, replaces moo) */\n private expressionLexer: ExpressionLexer;\n private currentState: LexerState = LexerState.Main;\n private peekedToken: Token | undefined;\n private hasPeeked = false;\n\n // Materialized token array from the last reset() call, used for\n // next()/peek() streaming access.\n private tokens: Token[] = [];\n private tokenIdx: number = 0;\n\n /**\n * @param localeCode - Locale code (e.g., \"en\", \"de\"). Defaults to \"en\".\n * @param tokenLookup - Optional TokenLookup from TokenClassRegistry.\n * When provided, configures ExpressionLexer to use registry-built\n * keyword/unit/phrase lookups instead of internal instance maps.\n */\n constructor(localeCode = \"en\", tokenLookup?: TokenLookup) {\n // Pass the lookup directly to ExpressionLexer's constructor, it's an\n // instance field now, not a static. Each Lexer instance gets its own\n // isolated lookup, preventing cross-instance corruption.\n this.expressionLexer = new ExpressionLexer(localeCode, tokenLookup);\n }\n\n reset(input: string, state?: LexerState): void {\n const newState = state ?? LexerState.Main;\n this.currentState = newState;\n this.hasPeeked = false;\n this.peekedToken = undefined;\n\n // Phase B: Main state classifies the line with the markdown scanner.\n // Skip lines (headings, fences, HRs, etc.) produce empty token arrays.\n // Expression lines and lines with inline solves are tokenized normally.\n if (newState === LexerState.Main) {\n const classification = this.expressionLexer.classifyLine(input);\n if (classification.skip) {\n this.tokens = [];\n this.tokenIdx = 0;\n return;\n }\n // Expression line or markdown line with inline solves, tokenize.\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n } else {\n // Non-main states (Inline, String), expression tokenization.\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n }\n }\n\n /**\n * Classify a single line of markdown text (Phase B).\n * Delegates to the ExpressionLexer's character-by-character scanner.\n */\n classifyLine(lineText: string): LineClassification {\n return this.expressionLexer.classifyLine(lineText);\n }\n\n /**\n * Find all inline solve markers in a line (Phase B).\n * Delegates to the ExpressionLexer's character-by-character scanner.\n */\n findInlineSolves(lineText: string) {\n return this.expressionLexer.findInlineSolves(lineText);\n }\n\n /**\n * Every keyword this lexer currently recognizes (locale + plugin-contributed),\n * mapped to the token type it lexes to. Delegates to the ExpressionLexer.\n */\n getKeywords(): Record<string, string> {\n return this.expressionLexer.getKeywords();\n }\n\n next(): Token | undefined {\n if (this.hasPeeked) {\n this.hasPeeked = false;\n return this.peekedToken;\n }\n // Materialized token array (ExpressionLexer path).\n if (this.tokenIdx < this.tokens.length) {\n return this.tokens[this.tokenIdx++];\n }\n return undefined;\n }\n\n peek(): Token | undefined {\n if (this.hasPeeked) return this.peekedToken;\n this.peekedToken = this.next();\n this.hasPeeked = true;\n return this.peekedToken;\n }\n\n [Symbol.iterator](): Iterator<Token> {\n return this.tokens[Symbol.iterator]();\n }\n\n /**\n * Register a plugin to extend the lexer with custom tokens.\n * Delegates to the underlying ExpressionLexer.\n *\n * @see LexerVocabulary for the supported extension points.\n */\n registerVocabulary(plugin: LexerVocabulary): void {\n this.expressionLexer.registerVocabulary(plugin);\n }\n\n /**\n * Unregister a plugin, removing its custom tokens from the lexer.\n * Delegates to the underlying ExpressionLexer.\n */\n unregisterVocabulary(plugin: LexerVocabulary): void {\n this.expressionLexer.unregisterVocabulary(plugin);\n }\n\n /**\n * Reset the lexer for expression-only text, skips the classifyLine()\n * overhead in reset() for callers that already know the input is an\n * evaluable expression (e.g., after isEmptyLine() confirmed non-skip).\n */\n resetExpression(input: string): void {\n this.currentState = LexerState.Main;\n this.hasPeeked = false;\n this.peekedToken = undefined;\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n }\n\n /**\n * Scan a full document in one pass, classifying each line and\n * tokenizing non-skipped lines. Delegates to ExpressionLexer.\n *\n * @returns ScanLineResult[], one per line, with classification + tokens.\n */\n scanDocument(text: string): ScanLineResult[] {\n return this.expressionLexer.scanDocument(text);\n }\n\n getState(): LexerState {\n return this.currentState;\n }\n\n setState(state: LexerState): void {\n this.currentState = state;\n }\n\n getHighlightTokens(lineText: string): {type: string; value: string; offset: number; col: number; length: number; category: TokenCategory | undefined}[] {\n const classification = this.expressionLexer.classifyLine(lineText);\n\n // For blockquote lines, strip the \"> \" prefix and tokenize the expression content.\n // This lets expressions inside blockquotes (e.g., \"> 1 + 2\") get syntax highlighted\n // while pure structural lines (headings, code fences) remain unhighlighted.\n if (classification.skip && lineText.startsWith(\"> \")) {\n return this.collectHighlightTokens(lineText.slice(2));\n }\n\n if (classification.skip) {\n return [];\n }\n\n return this.collectHighlightTokens(lineText);\n }\n\n /**\n * The same tokens {@link getHighlightTokens} reduces, before reduction.\n *\n * Exists because normalization operates on tokens, not on the flattened\n * shape, and a consumer that wants phrase-fused highlighting has to run the\n * normalizer between the two. See `LanguageService.getSemanticTokens`.\n *\n * @param lineText - One line of source.\n * @returns Every token on the line that is worth painting, unreduced.\n */\n getHighlightTokenObjects(lineText: string): Token[] {\n const classification = this.expressionLexer.classifyLine(lineText);\n if (classification.skip && lineText.startsWith(\"> \")) {\n return this.collectTokenObjects(lineText.slice(2));\n }\n if (classification.skip) return [];\n return this.collectTokenObjects(lineText);\n }\n\n private collectTokenObjects(lineText: string): Token[] {\n this.resetExpression(lineText);\n const result: Token[] = [];\n for (const token of this) {\n if (token.type === \"WS\" || token.type === \"NEWLINE\") continue;\n if (token.type.startsWith(\"MD_\")) continue;\n if (token.type === \"INLINE_SOLVE_START\" || token.type === \"BACKTICK_CLOSE\") continue;\n result.push(token);\n }\n return result;\n }\n\n private collectHighlightTokens(lineText: string): {type: string; value: string; offset: number; col: number; length: number; category: TokenCategory | undefined}[] {\n return this.collectTokenObjects(lineText).map(token => ({\n type: token.type,\n value: token.value,\n offset: token.offset,\n col: token.col,\n length: token.value.length,\n category: getTokenCategory(token.type),\n }));\n }\n}\n\n/**\n * A lexer for operations that do not depend on registered vocabulary.\n *\n * Line classification and inline-solve detection read characters looking for\n * headings, comment markers, fences and backtick spans, and never consult the\n * keyword, unit or operator tables. Every lexer therefore returns the same\n * answer, so the callers that have no engine to ask can use this one. Checked\n * by `__tests__/lexer/LineClassificationIsVocabularyIndependent.spec.ts`.\n *\n * Do not tokenize with this. An engine's own lexer carries the vocabulary its\n * packages registered; this one carries none.\n */\nexport const sharedLexer = new Lexer(\"en\", undefined);","/**\n * TokenClass, Plugin-extensible keyword registration for the Lexer.\n *\n * Providers call `registry.register(tokenClass)` to teach the lexer about\n * their keywords. The registry merges locale keywords, provider keywords,\n * phrase mappings, and unit names into an optimized TokenLookup structure\n * consumed by the Lexer.\n *\n * @example\n * registry.register({\n * tokenType: 'CARET',\n * keywords: {},\n * phrases: { 'to the power of': true, 'power of': true },\n * priority: 10,\n * description: 'Exponentiation operators (x^y)',\n * });\n */\nexport interface TokenClass {\n /** The token type string produced by the lexer (e.g., \"FUNC\", \"PI\", \"CARET\").\n * Must match a token type that a ParseletRegistry has a parselet for. */\n tokenType: string;\n\n /** Single-word keywords (case-insensitive). The lexer lowercases input\n * before lookup, so these should be lowercase. Example:\n * { sqrt: true, abs: true, sin: true, cos: true } for tokenType \"FUNC\" */\n keywords: Record<string, boolean>;\n\n /** Multi-word phrases (case-insensitive). Matched by the built-in PhraseMatcher\n * via the phrase trie. Example:\n * { \"to the power of\": true, \"power of\": true } for tokenType \"CARET\" */\n phrases?: Record<string, boolean>;\n\n /** Priority for conflict resolution. When two TokenClasses register\n * the same keyword, the higher-priority class wins. Locale keywords\n * have priority 0 (set via setLocale). Providers should use\n * priority >= 10 to override locale defaults. Default: 0 */\n priority?: number;\n\n /** Human-readable description for debugging and introspection */\n description?: string;\n}\n\n// ── Phrase Trie ──────────────────────────────────────────────────────────────\n\n/** Trie node for multi-word phrase matching. */\nexport interface PhraseNode {\n /** Complete phrase token type (null = intermediate node) */\n type?: string;\n children: Map<string, PhraseNode>;\n}\n\n// ── TokenLookup, Optimized lookup structure for the Lexer ──────────────────\n\n/**\n * The optimized lookup structure built by TokenClassRegistry.build().\n * Consumed by the Lexer for O(1) keyword → token type lookups and\n * O(word-count) phrase matching.\n */\nexport interface TokenLookup {\n /** Lowercase keyword → token type. O(1) Map lookup. */\n keywordToType: Map<string, string>;\n\n /** Phrase trie for multi-word matching. Root node with children maps.\n * Null if no phrases registered. */\n phraseTrie: PhraseNode | null;\n\n /** Set of lowercase first-words of all registered phrases.\n * Used by the lexer to emit IDENT (not a phrase keyword) for words\n * that start multi-word phrases, deferring to the PhraseMatcher.\n *\n * Example: \"to\" is in phraseStartWords because \"to the power of\" is a phrase.\n * When the lexer sees \"to\", it emits IDENT and lets the phrase matcher\n * combine \"to the power of\" into a single CARET token.\n *\n * This prevents plugins from accidentally overriding phrase-start words.\n */\n phraseStartWords: Set<string>;\n\n /** Case-sensitive unit names for UNIT fallback after keyword lookup fails. */\n unitNames: ReadonlySet<string>;\n}\n\n// ── TokenClassRegistry ───────────────────────────────────────────────────────\n\n/**\n * Central registry for keyword→token-type mappings.\n *\n * Providers register TokenClasses; locales provide keyword maps;\n * units provide a name set. `build()` merges all sources into an\n * optimized TokenLookup consumed by the Lexer.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0)\n * 2. Provider keywords (sorted by priority ascending, higher priority wins)\n *\n * Unit names are stored separately (checked AFTER keyword lookup fails).\n * Phrases are stored in a trie for O(phrase-length) matching.\n */\nexport class TokenClassRegistry {\n private classes: TokenClass[] = [];\n private localeKeywordMap: Record<string, string> | null = null;\n private localePhraseMap: Record<string, string> | null = null;\n private unitNames: ReadonlySet<string> | null = null;\n\n /**\n * Register a provider's TokenClass. Must be called BEFORE build().\n * Can be called multiple times to add more entries.\n *\n * Built-in token types CANNOT be overridden, throws a EngineError\n * if the TokenClass attempts to register a keyword that conflicts\n * with an already-registered token type.\n */\n register(tokenClass: TokenClass): void {\n this.classes.push(tokenClass);\n }\n\n /**\n * Unregister all TokenClasses for a given token type.\n * Useful for plugin unload. Requires rebuild() to take effect.\n */\n unregister(tokenType: string): void {\n this.classes = this.classes.filter(c => c.tokenType !== tokenType);\n }\n\n /**\n * Set the locale's keyword→type map and optional phrase map.\n * Called on locale change. Priority 0 (cannot override providers with higher priority).\n */\n setLocale(keywordMap: Record<string, string>, phraseMap?: Record<string, string>): void {\n this.localeKeywordMap = keywordMap;\n this.localePhraseMap = phraseMap ?? null;\n }\n\n /**\n * Set the unit name set. Called when unit list changes.\n * Units are stored separately (checked AFTER keyword lookup fails).\n *\n * Takes a ReadonlySet because the caller's set is derived from the\n * conversion tables and must not be mutated; this class only ever reads it.\n */\n setUnits(unitNames: ReadonlySet<string>): void {\n this.unitNames = unitNames;\n }\n\n /**\n * Build the optimized TokenLookup from all registered sources.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0)\n * 2. Provider classes (sorted by priority ascending)\n *\n * Returns a frozen TokenLookup that the Lexer consumes.\n * Call build() again after register()/setLocale()/setUnits() changes.\n */\n build(): TokenLookup {\n const keywordToType = new Map<string, string>();\n\n // Layer 1: Locale keywords (priority 0, lowest)\n if (this.localeKeywordMap) {\n for (const [keyword, tokenType] of Object.entries(this.localeKeywordMap)) {\n keywordToType.set(keyword.toLowerCase(), tokenType);\n }\n }\n\n // Layer 2: Provider keywords (sorted by priority ascending, higher wins)\n const sorted = [...this.classes].sort((a, b) => (a.priority ?? 0) - (b.priority ?? 0));\n for (const tc of sorted) {\n for (const keyword of Object.keys(tc.keywords)) {\n keywordToType.set(keyword.toLowerCase(), tc.tokenType);\n }\n }\n\n // Build phrase trie from locale + provider phrases\n const phraseTrie = this.buildPhraseTrie();\n\n // Build phraseStartWords set (first word of every phrase)\n const phraseStartWords = this.buildPhraseStartWords();\n\n return {\n keywordToType,\n phraseTrie,\n phraseStartWords,\n unitNames: this.unitNames ?? new Set(),\n };\n }\n\n // ── Private helpers ─────────────────────────────────────────────────────\n\n private buildPhraseTrie(): PhraseNode | null {\n const root: PhraseNode = { children: new Map() };\n\n // Layer 1: Locale phrases\n if (this.localePhraseMap) {\n for (const [phrase, tokenType] of Object.entries(this.localePhraseMap)) {\n this.insertPhrase(root, phrase.toLowerCase(), tokenType);\n }\n }\n\n // Layer 2: Provider phrases\n for (const tc of this.classes) {\n if (!tc.phrases) continue;\n for (const phrase of Object.keys(tc.phrases)) {\n this.insertPhrase(root, phrase.toLowerCase(), tc.tokenType);\n }\n }\n\n return root.children.size > 0 ? root : null;\n }\n\n private insertPhrase(root: PhraseNode, phrase: string, tokenType: string): void {\n const words = phrase.split(' ');\n let node = root;\n for (const word of words) {\n if (!node.children.has(word)) {\n node.children.set(word, { children: new Map() });\n }\n node = node.children.get(word)!;\n }\n // Only set type if not already set, first-registered (locale) wins\n if (!node.type) {\n node.type = tokenType;\n }\n }\n\n private buildPhraseStartWords(): Set<string> {\n const startWords = new Set<string>();\n\n // Locale phrase first words\n if (this.localePhraseMap) {\n for (const phrase of Object.keys(this.localePhraseMap)) {\n const first = phrase.split(' ')[0].toLowerCase();\n startWords.add(first);\n }\n }\n\n // Provider phrase first words\n for (const tc of this.classes) {\n if (!tc.phrases) continue;\n for (const phrase of Object.keys(tc.phrases)) {\n const first = phrase.split(' ')[0].toLowerCase();\n startWords.add(first);\n }\n }\n\n return startWords;\n }\n}","/**\n * tokenRegistration.ts, Bootstrap for building TokenLookup from locale, units, and phrases.\n *\n * This module centralizes the assembly of the TokenLookup consumed by ExpressionLexer.\n * It merges:\n * 1. Locale keywords (from ILocale.keywordMap)\n * 2. Built-in phrase patterns (to the power of, increase by, etc.)\n * 3. Known unit names (from units.ts)\n *\n * The resulting TokenLookup is frozen and passed to ExpressionLexer.configuredLookup\n * at construction time, enabling data-driven keyword/unit/phrase resolution.\n */\nimport { TokenClassRegistry } from '@solve-js/lexer/TokenClassRegistry';\nimport type { TokenLookup } from '@solve-js/lexer/TokenClassRegistry';\nimport { getLocale, type ILocale } from '@solve-js/constants/locales';\nimport { knownUnits } from '@solve-js/lexer/units';\n\n// ── Built-in phrase map ───────────────────────────────────────────────────\n// These are the multi-word expressions handled as compound tokens.\n// Matched by the PhraseMatcher (trie-based) in ExpressionLexer.tryMatchPhrase().\nconst BUILTIN_PHRASES: Record<string, string> = {\n 'to the power of': 'CARET',\n 'power of': 'CARET',\n 'increase by': 'INCREASE_BY',\n 'decrease by': 'DECREASE_BY',\n 'times by': 'TIMES_BY',\n 'multiply by': 'MULTIPLY_BY',\n 'multiplied by': 'MULTIPLY_BY',\n 'divide by': 'DIVIDE_BY',\n};\n\n/**\n * Build a TokenLookup from locale keywords, known units, and built-in phrases.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0, lowest, can be overridden by providers)\n * 2. Built-in phrases (via locale phraseMap)\n * 3. Known units (checked AFTER keyword lookup fails, via unitNames set)\n *\n * The resulting TokenLookup replaces the internal keyword map, unit set,\n * phrase trie, and phraseStartWords in ExpressionLexer when set via\n * ExpressionLexer.configuredLookup.\n *\n * @param localeCode - The locale code (e.g., \"en\", \"de\"). Defaults to \"en\".\n * @returns A frozen TokenLookup ready for consumption by the lexer.\n */\nexport function buildTokenLookup(localeCode = 'en'): TokenLookup {\n const locale: ILocale = getLocale(localeCode);\n const registry = new TokenClassRegistry();\n\n // Layer 1: Locale keywords (priority 0, lowest)\n registry.setLocale(locale.keywordMap, BUILTIN_PHRASES);\n\n // Layer 2: Known units (checked after keyword lookup)\n registry.setUnits(knownUnits);\n\n // Build and return the lookup\n return registry.build();\n}\n"]}
|