solve-engine 2.7.0 → 2.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{chunk-5FO4PDIG.js → chunk-3GAWAMT4.js} +2 -2
- package/dist/{chunk-5FO4PDIG.js.map → chunk-3GAWAMT4.js.map} +1 -1
- package/dist/{chunk-32GC2FHC.cjs → chunk-3HYO3QYW.cjs} +2 -2
- package/dist/{chunk-32GC2FHC.cjs.map → chunk-3HYO3QYW.cjs.map} +1 -1
- package/dist/{chunk-2JMFUQZH.js → chunk-44EOEWGP.js} +2 -2
- package/dist/chunk-44EOEWGP.js.map +1 -0
- package/dist/chunk-556Q7VZC.cjs +2 -0
- package/dist/chunk-556Q7VZC.cjs.map +1 -0
- package/dist/{chunk-B7J24Q5T.cjs → chunk-6ILXAGWX.cjs} +2 -2
- package/dist/{chunk-B7J24Q5T.cjs.map → chunk-6ILXAGWX.cjs.map} +1 -1
- package/dist/{chunk-CIQOK7Q4.cjs → chunk-6QRHN3I6.cjs} +2 -2
- package/dist/chunk-6QRHN3I6.cjs.map +1 -0
- package/dist/{chunk-IHZBKT7L.cjs → chunk-6TPXH6Y5.cjs} +3 -3
- package/dist/chunk-6TPXH6Y5.cjs.map +1 -0
- package/dist/{chunk-W7VOTR2J.cjs → chunk-7VJN65QP.cjs} +3 -3
- package/dist/{chunk-W7VOTR2J.cjs.map → chunk-7VJN65QP.cjs.map} +1 -1
- package/dist/{chunk-GYX6S6OS.cjs → chunk-AEPS7W3D.cjs} +3 -3
- package/dist/chunk-AEPS7W3D.cjs.map +1 -0
- package/dist/chunk-BKSKZA4W.cjs +2 -0
- package/dist/chunk-BKSKZA4W.cjs.map +1 -0
- package/dist/{chunk-VAGS5JS7.cjs → chunk-C65UDMWH.cjs} +2 -2
- package/dist/{chunk-VAGS5JS7.cjs.map → chunk-C65UDMWH.cjs.map} +1 -1
- package/dist/chunk-DLAKPF5T.js +2 -0
- package/dist/chunk-DLAKPF5T.js.map +1 -0
- package/dist/{chunk-AQWNEO7M.cjs → chunk-EURCQMZK.cjs} +2 -2
- package/dist/{chunk-AQWNEO7M.cjs.map → chunk-EURCQMZK.cjs.map} +1 -1
- package/dist/{chunk-G7TBEQRQ.js → chunk-GMSBBLUQ.js} +2 -2
- package/dist/{chunk-G7TBEQRQ.js.map → chunk-GMSBBLUQ.js.map} +1 -1
- package/dist/chunk-HE7E6M5O.js +3 -0
- package/dist/chunk-HE7E6M5O.js.map +1 -0
- package/dist/chunk-K4MME6ZT.js +2 -0
- package/dist/chunk-K4MME6ZT.js.map +1 -0
- package/dist/{chunk-5AMJNF7J.js → chunk-LH7URJZT.js} +3 -3
- package/dist/chunk-LH7URJZT.js.map +1 -0
- package/dist/{chunk-KXM7EY27.js → chunk-N46DFZLZ.js} +3 -3
- package/dist/{chunk-KXM7EY27.js.map → chunk-N46DFZLZ.js.map} +1 -1
- package/dist/{chunk-CVY2A3TP.js → chunk-OVNB7MI5.js} +2 -2
- package/dist/{chunk-CVY2A3TP.js.map → chunk-OVNB7MI5.js.map} +1 -1
- package/dist/{chunk-ZRFUD2LX.cjs → chunk-OWQF6ZOK.cjs} +4 -4
- package/dist/{chunk-ZRFUD2LX.cjs.map → chunk-OWQF6ZOK.cjs.map} +1 -1
- package/dist/{chunk-HF4H7LYA.js → chunk-P7XWFT45.js} +3 -3
- package/dist/{chunk-HF4H7LYA.js.map → chunk-P7XWFT45.js.map} +1 -1
- package/dist/{chunk-YAHGSBMC.js → chunk-QYO7IEPR.js} +2 -2
- package/dist/{chunk-YAHGSBMC.js.map → chunk-QYO7IEPR.js.map} +1 -1
- package/dist/{chunk-2LFLR3VZ.cjs → chunk-SNVPM27Q.cjs} +2 -2
- package/dist/{chunk-2LFLR3VZ.cjs.map → chunk-SNVPM27Q.cjs.map} +1 -1
- package/dist/{chunk-AR7L5OJ2.js → chunk-T3FBAFRR.js} +2 -2
- package/dist/{chunk-AR7L5OJ2.js.map → chunk-T3FBAFRR.js.map} +1 -1
- package/dist/chunk-UWAXJX5G.js +2 -0
- package/dist/chunk-UWAXJX5G.js.map +1 -0
- package/dist/chunk-WEAKI5V5.cjs +2 -0
- package/dist/chunk-WEAKI5V5.cjs.map +1 -0
- package/dist/{chunk-NXIMGB4S.cjs → chunk-WYIOXAFW.cjs} +2 -2
- package/dist/{chunk-NXIMGB4S.cjs.map → chunk-WYIOXAFW.cjs.map} +1 -1
- package/dist/{chunk-HTJPV42D.js → chunk-YTCTUBR2.js} +2 -2
- package/dist/{chunk-HTJPV42D.js.map → chunk-YTCTUBR2.js.map} +1 -1
- package/dist/constants.cjs +1 -1
- package/dist/constants.js +1 -1
- package/dist/engine.cjs +1 -1
- package/dist/engine.js +1 -1
- package/dist/format.cjs +1 -1
- package/dist/format.js +1 -1
- package/dist/index.cjs +1 -1
- package/dist/index.js +1 -1
- package/dist/language.cjs +1 -1
- package/dist/language.js +1 -1
- package/dist/lexer.cjs +1 -1
- package/dist/lexer.js +1 -1
- package/dist/normalizer.cjs +1 -1
- package/dist/normalizer.js +1 -1
- package/dist/packages.cjs +1 -1
- package/dist/packages.d.cts +1 -1
- package/dist/packages.d.ts +1 -1
- package/dist/packages.js +1 -1
- package/dist/testing.cjs +2 -2
- package/dist/testing.js +1 -1
- package/dist/uom.cjs +1 -1
- package/dist/uom.js +1 -1
- package/dist/vm.cjs +1 -1
- package/dist/vm.js +1 -1
- package/dist/worker.cjs +2 -2
- package/dist/worker.js +1 -1
- package/package.json +1 -1
- package/dist/chunk-2JMFUQZH.js.map +0 -1
- package/dist/chunk-5AMJNF7J.js.map +0 -1
- package/dist/chunk-CIQOK7Q4.cjs.map +0 -1
- package/dist/chunk-FWW3M36J.cjs +0 -2
- package/dist/chunk-FWW3M36J.cjs.map +0 -1
- package/dist/chunk-GYX6S6OS.cjs.map +0 -1
- package/dist/chunk-HAEOMJE3.cjs +0 -2
- package/dist/chunk-HAEOMJE3.cjs.map +0 -1
- package/dist/chunk-IANAMBJQ.cjs +0 -2
- package/dist/chunk-IANAMBJQ.cjs.map +0 -1
- package/dist/chunk-IHZBKT7L.cjs.map +0 -1
- package/dist/chunk-N4HJTHZR.js +0 -2
- package/dist/chunk-N4HJTHZR.js.map +0 -1
- package/dist/chunk-P5EMTVMJ.js +0 -3
- package/dist/chunk-P5EMTVMJ.js.map +0 -1
- package/dist/chunk-TRBXW6HW.js +0 -2
- package/dist/chunk-TRBXW6HW.js.map +0 -1
- package/dist/chunk-WGZ3FNHI.js +0 -2
- package/dist/chunk-WGZ3FNHI.js.map +0 -1
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
import {a}from'./chunk-
|
|
2
|
-
//# sourceMappingURL=chunk-
|
|
1
|
+
import {a}from'./chunk-GMSBBLUQ.js';import {c}from'./chunk-IJMNVBIS.js';import {c as c$1}from'./chunk-MPB7KUN5.js';var v={maxPasses:100,maxTokens:1e4,onFusion:()=>{}};function N(a$1,e,r){let n=r[0],o=r[r.length-1],t=new a(a$1,c(a$1),e,e,n.offset,n.lineBreaks??0,n.line,n.col);return t.sourceEnd=o.sourceEnd??o.offset+o.text.length,t}function C(a,e){let r=e[e.length-1];r&&(a.sourceEnd=r.sourceEnd??r.offset+r.text.length);}var w=["NUMBER","HEX","BIGINT","FLOAT","LSHIFT","RSHIFT","BIT_AND","BIT_OR","BIT_XOR","LPAREN","RPAREN","LBRACKET","RBRACKET","COMMA","COLON","EQUALS","THEREFORE","PIPE","AMPERSAND","AT","SEMICOLON","QUESTION","EXCLAMATION","EOF","WS","NEWLINE"],I=(()=>{let a=w.map(n=>c(n)),e=Math.max(...a)+1,r=new Uint8Array(e);for(let n of a)r[n]=1;return r})(),L=class{constructor(e={}){this.rules=[];this.sortedRulesCache=null;this.phraseTrie=new T;this.options={...v,...e};}register(e){this.rules.push(e),this.sortedRulesCache=null;}unregister(e){this.rules=this.rules.filter(r=>r.name!==e),this.sortedRulesCache=null;}clear(){this.rules=[],this.sortedRulesCache=null,this.phraseTrie=new T;}getSortedRules(){return this.sortedRulesCache===null&&(this.sortedRulesCache=[...this.rules].sort((e,r)=>r.priority-e.priority)),this.sortedRulesCache}get ruleCount(){return this.rules.length}addPhrase(e,r){this.phraseTrie.addPhrase(e,r);}getPhrases(){return this.phraseTrie.getAllPhrases()}canStartPhrase(e){return this.phraseTrie.canStart(e)}normalize(e,r){if(e.length===0)return e;let n=this.getSortedRules(),o=r??this.options.onFusion,t=this.options.maxPasses,s=this.options.maxTokens,i=e,m=true,p=0;for(;m&&p<t;){m=false,p++;let h=[],u=0;for(;u<i.length;){let E=false,y=i[u].typeId;if(y>=I.length||I[y]===0){let c=this.phraseTrie.matchAt(i,u);if(c){let l=i.slice(u,u+c.consumed);for(let d of c.replacement)h.push(d);c.consumed>1&&c.replacement.length===1&&(C(c.replacement[0],l),o({rule:c.ruleName??"phrase-trie",sourceTokens:l,fusedToken:c.replacement[0]})),u+=c.consumed,m=true;continue}}for(let c of n){let l=c.match(i,u);if(l){let d=i.slice(u,u+l.consumed);for(let g of l.replacement)h.push(g);if(l.consumed>1&&l.replacement.length===1)C(l.replacement[0],d),o({rule:c.name,sourceTokens:d,fusedToken:l.replacement[0]});else if(l.consumed>1&&l.replacement.length<l.consumed)for(let g of l.replacement)o({rule:c.name,sourceTokens:d,fusedToken:g});u+=l.consumed,m=true,E=true;break}}E||(h.push(i[u]),u++);}if(h.length>s)throw c$1.validation("NORMALIZED_TOKEN_LIMIT_EXCEEDED",`Normalized token count (${h.length}) exceeds safety limit (${s})`,{tokenCount:h.length,maxTokens:s});i=h;}return i}};var T=class{constructor(){this.startWords=new Set;this.root=new Map;}addPhrase(e,r){let n=e.toLowerCase().trim().split(/\s+/);if(n.length===0||n[0]==="")return;this.startWords.add(n[0]);let o=n[0],t;if(this.root.has(o)?t=this.root.get(o):(t={children:new Map,terminal:null},this.root.set(o,t)),n.length===1){t.terminal={tokenType:r,phrase:e,consumed:1};return}for(let s=1;s<n.length;s++){let i=n[s];t.children.has(i)||t.children.set(i,{children:new Map,terminal:null}),t=t.children.get(i);}t.terminal={tokenType:r,phrase:e,consumed:n.length};}matchAt(e,r){if(r>=e.length)return null;let n=e[r].type;if(n==="TAG"||n.startsWith("TAG_"))return null;let o=e[r].value.toLowerCase();if(!this.startWords.has(o))return null;let t=this.root.get(o);if(!t)return null;let s=null,i=1;if(t.terminal){let m=e.slice(r,r+1);s={consumed:1,replacement:[N(t.terminal.tokenType,t.terminal.phrase,m)],ruleName:t.terminal.phrase};}for(let m=r+1;m<e.length&&t?.children;m++){let p=e[m].type;if(p==="TAG"||p.startsWith("TAG_"))break;let h=e[m].value.toLowerCase();if(t=t.children.get(h),!t)break;if(i++,t.terminal){let u=e.slice(r,r+i);s={consumed:i,replacement:[N(t.terminal.tokenType,t.terminal.phrase,u)],ruleName:t.terminal.phrase};}}return s}get size(){return this.root.size}getAllPhrases(){let e={},r=(n,o)=>{n.terminal&&(e[o.join(" ")]=n.terminal.tokenType);for(let[t,s]of n.children)r(s,[...o,t]);};for(let[n,o]of this.root)r(o,[n]);return e}canStart(e){return this.startWords.has(e.toLowerCase())}};var S=new Set(["per","a","an","each","every"]),k=new Set(["to","power","increase","decrease","times","multiply","divide","by"]);function x(a$1=50,e){let r=e??(n=>k.has(n.toLowerCase()));return {name:"implicit:multiply",priority:a$1,match(n,o){if(o+1>=n.length)return null;let t=n[o],s=n[o+1],i=s.value.toLowerCase();if(r(i)||S.has(i)&&n[o+2]?.type==="UNIT"||!((t.type==="NUMBER"||t.type==="RPAREN")&&(s.type==="IDENT"||s.type==="LPAREN"||s.type==="PI"||s.type==="E")))return null;let p=new a("STAR",c("STAR"),"*","*",s.offset,0,s.line,s.col);return {consumed:1,replacement:[t,p]}}}}function W(a,e){let r=[];for(let n=0;n<e;n++){let o=a[n];if(o.type==="LBRACKET")r.push(true);else if(o.type==="LPAREN"){let t=a[n-1],s=!!t&&(t.type==="MAP"||t.type==="REDUCE"||t.type==="SUM_FN"||t.type==="PROD_FN"||t.type==="IDENT"&&(t.value.toLowerCase()==="map"||t.value.toLowerCase()==="reduce"||t.value.toLowerCase()==="sum"||t.value.toLowerCase()==="prod"));r.push(s);}else (o.type==="RBRACKET"||o.type==="RPAREN")&&r.pop();}return r.length>0&&r[r.length-1]}function U(){return [x()]}var Y={"to the power of":"CARET","power of":"CARET","increase by":"INCREASE_BY","decrease by":"DECREASE_BY","times by":"TIMES_BY","multiply by":"MULTIPLY_BY","divide by":"DIVIDE_BY","multiplied by":"MULTIPLY_BY","divided by":"DIVIDE_BY"};export{T as a,N as b,L as c,x as d,W as e,U as f,Y as g};//# sourceMappingURL=chunk-3GAWAMT4.js.map
|
|
2
|
+
//# sourceMappingURL=chunk-3GAWAMT4.js.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/normalizer/TokenNormalizer.ts","../src/normalizer/PhraseTrie.ts","../src/normalizer/BuiltinNormalizerRules.ts"],"names":["DEFAULT_OPTIONS","createFusedToken","type","text","sourceTokens","first","last","token","LexerToken","tokenTypeId","recordSourceSpan","fused","NON_WORD_NAMES","NON_WORD_TABLE","ids","len","table","id","TokenNormalizer","options","PhraseTrie","rule","ruleName","a","b","phrase","tokenType","word","tokens","onFusion","sorted","fusionHandler","maxPasses","maxTokens","current","changed","passCount","result","pos","matched","tid","trieMatch","rt","match","ErrorFactory","words","firstWord","node","i","startType","best","depth","contType","collect","path","child","RATE_DENOMINATOR_WORDS","PHRASE_START_WORDS","implicitMultiplyRule","priority","canStart","phraseGuard","next","nextValue","starToken","isInsideRangeContext","safeStack","t","prev","opensMapReduceCall","createBuiltinNormalizerRules","BUILTIN_PHRASES"],"mappings":"mHA2EA,IAAMA,CAAAA,CAA+C,CACnD,SAAA,CAAW,GAAA,CACX,UAAW,GAAA,CACX,QAAA,CAAU,IAAM,CAAC,CACnB,CAAA,CAwBO,SAASC,CAAAA,CACdC,IACAC,CAAAA,CACAC,CAAAA,CACO,CACP,IAAMC,CAAAA,CAAQD,CAAAA,CAAa,CAAC,CAAA,CACtBE,EAAOF,CAAAA,CAAaA,CAAAA,CAAa,MAAA,CAAS,CAAC,EAC3CG,CAAAA,CAAQ,IAAIC,CAAAA,CAChBN,GAAAA,CACAO,EAAYP,GAAI,CAAA,CAChBC,CAAAA,CACAA,CAAAA,CACAE,CAAAA,CAAM,MAAA,CACNA,CAAAA,CAAM,UAAA,EAAc,EACpBA,CAAAA,CAAM,IAAA,CACNA,CAAAA,CAAM,GACR,EAGA,OAAAE,CAAAA,CAAM,SAAA,CAAYD,CAAAA,CAAK,WAAaA,CAAAA,CAAK,MAAA,CAASA,CAAAA,CAAK,IAAA,CAAK,MAAA,CACrDC,CACT,CAkBA,SAASG,EAAiBC,CAAAA,CAAcP,CAAAA,CAA6B,CACnE,IAAME,EAAOF,CAAAA,CAAaA,CAAAA,CAAa,MAAA,CAAS,CAAC,EAC5CE,CAAAA,GACLK,CAAAA,CAAM,SAAA,CAAYL,CAAAA,CAAK,SAAA,EAAaA,CAAAA,CAAK,MAAA,CAASA,CAAAA,CAAK,KAAK,MAAA,EAC9D,CAiCO,IAAMM,CAAAA,CAAiB,CAC7B,QAAA,CAAU,KAAA,CAAO,QAAA,CAAU,OAAA,CAC3B,SAAU,QAAA,CAAU,SAAA,CAAW,QAAA,CAAU,SAAA,CACzC,QAAA,CAAU,QAAA,CAAU,UAAA,CAAY,UAAA,CAChC,QAAS,OAAA,CAAS,QAAA,CAAU,WAAA,CAAa,MAAA,CAAQ,YAAa,IAAA,CAC9D,WAAA,CAAa,UAAA,CAAY,aAAA,CACzB,MAAO,IAAA,CAAM,SACd,CAAA,CAQaC,CAAAA,CAAAA,CAA8B,IAAM,CAEhD,IAAMC,CAAAA,CAAMF,EAAe,GAAA,CAAI,CAAA,EAAKH,CAAAA,CAAY,CAAC,CAAC,CAAA,CAE5CM,CAAAA,CAAM,IAAA,CAAK,GAAA,CAAI,GAAGD,CAAG,CAAA,CAAI,CAAA,CACzBE,CAAAA,CAAQ,IAAI,UAAA,CAAWD,CAAG,CAAA,CAEhC,QAAWE,CAAAA,IAAMH,CAAAA,CAAKE,CAAAA,CAAMC,CAAE,EAAI,CAAA,CAClC,OAAOD,CACR,CAAA,IA6BaE,CAAAA,CAAN,KAAsB,CAgC3B,WAAA,CAAYC,CAAAA,CAA6B,EAAC,CAAG,CA9B7C,KAAQ,KAAA,CAA0B,EAAC,CAanC,IAAA,CAAQ,iBAA4C,IAAA,CAOpD,IAAA,CAAQ,UAAA,CAAa,IAAIC,EAWvB,IAAA,CAAK,OAAA,CAAU,CAAE,GAAGpB,CAAAA,CAAiB,GAAGmB,CAAQ,EAClD,CAaA,QAAA,CAASE,CAAAA,CAA4B,CACnC,IAAA,CAAK,MAAM,IAAA,CAAKA,CAAI,CAAA,CACpB,IAAA,CAAK,iBAAmB,KAC1B,CAUA,UAAA,CAAWC,CAAAA,CAAwB,CACjC,IAAA,CAAK,KAAA,CAAQ,IAAA,CAAK,MAAM,MAAA,CAAO,CAAA,EAAK,CAAA,CAAE,IAAA,GAASA,CAAQ,CAAA,CACvD,IAAA,CAAK,gBAAA,CAAmB,KAC1B,CAMA,KAAA,EAAc,CACZ,IAAA,CAAK,KAAA,CAAQ,EAAC,CACd,IAAA,CAAK,gBAAA,CAAmB,KACxB,IAAA,CAAK,UAAA,CAAa,IAAIF,EACxB,CAOQ,cAAA,EAAmC,CACzC,OAAI,IAAA,CAAK,mBAAqB,IAAA,GAC5B,IAAA,CAAK,gBAAA,CAAmB,CAAC,GAAG,IAAA,CAAK,KAAK,CAAA,CAAE,KAAK,CAACG,CAAAA,CAAGC,CAAAA,GAAMA,CAAAA,CAAE,SAAWD,CAAAA,CAAE,QAAQ,CAAA,CAAA,CAEzE,IAAA,CAAK,gBACd,CAKA,IAAI,SAAA,EAAoB,CACtB,OAAO,IAAA,CAAK,KAAA,CAAM,MACpB,CAcA,SAAA,CAAUE,CAAAA,CAAgBC,CAAAA,CAAyB,CACjD,KAAK,UAAA,CAAW,SAAA,CAAUD,CAAAA,CAAQC,CAAS,EAC7C,CAgBA,UAAA,EAAqC,CACnC,OAAO,IAAA,CAAK,UAAA,CAAW,aAAA,EACzB,CAEA,cAAA,CAAeC,CAAAA,CAAuB,CACpC,OAAO,KAAK,UAAA,CAAW,QAAA,CAASA,CAAI,CACtC,CA8BA,SAAA,CAAUC,CAAAA,CAAiBC,CAAAA,CAAmD,CAE5E,GAAID,CAAAA,CAAO,MAAA,GAAW,CAAA,CAAG,OAAOA,CAAAA,CAGhC,IAAME,CAAAA,CAAS,IAAA,CAAK,gBAAe,CAC7BC,CAAAA,CAAgBF,CAAAA,EAAY,IAAA,CAAK,QAAQ,QAAA,CACzCG,CAAAA,CAAY,IAAA,CAAK,OAAA,CAAQ,SAAA,CACzBC,CAAAA,CAAY,IAAA,CAAK,OAAA,CAAQ,UAE3BC,CAAAA,CAAUN,CAAAA,CACVO,CAAAA,CAAU,IAAA,CACVC,CAAAA,CAAY,CAAA,CAGhB,KAAOD,CAAAA,EAAWC,EAAYJ,CAAAA,EAAW,CACvCG,CAAAA,CAAU,KAAA,CACVC,CAAAA,EAAAA,CAEA,IAAMC,CAAAA,CAAkB,GACpBC,CAAAA,CAAM,CAAA,CAGV,KAAOA,CAAAA,CAAMJ,EAAQ,MAAA,EAAQ,CAC3B,IAAIK,CAAAA,CAAU,MAKRC,CAAAA,CAAMN,CAAAA,CAAQI,CAAG,CAAA,CAAE,MAAA,CACzB,GAAIE,CAAAA,EAAO3B,CAAAA,CAAe,QAAUA,CAAAA,CAAe2B,CAAG,CAAA,GAAM,CAAA,CAAG,CAC7D,IAAMC,CAAAA,CAAY,IAAA,CAAK,UAAA,CAAW,QAAQP,CAAAA,CAASI,CAAG,CAAA,CACtD,GAAIG,CAAAA,CAAW,CACb,IAAMrC,CAAAA,CAAe8B,EAAQ,KAAA,CAAMI,CAAAA,CAAKA,CAAAA,CAAMG,CAAAA,CAAU,QAAQ,CAAA,CAChE,IAAA,IAAWC,CAAAA,IAAMD,CAAAA,CAAU,YACzBJ,CAAAA,CAAO,IAAA,CAAKK,CAAE,CAAA,CAEZD,CAAAA,CAAU,QAAA,CAAW,CAAA,EAAKA,CAAAA,CAAU,YAAY,MAAA,GAAW,CAAA,GAC7D/B,CAAAA,CAAiB+B,CAAAA,CAAU,YAAY,CAAC,CAAA,CAAGrC,CAAY,CAAA,CACvD2B,EAAc,CACZ,IAAA,CAAMU,CAAAA,CAAU,QAAA,EAAY,aAAA,CAC5B,YAAA,CAAArC,CAAAA,CACA,UAAA,CAAYqC,EAAU,WAAA,CAAY,CAAC,CACrC,CAAC,GAEHH,CAAAA,EAAOG,CAAAA,CAAU,QAAA,CACjBN,CAAAA,CAAU,KACV,QACF,CACF,CAGA,IAAA,IAAWd,CAAAA,IAAQS,CAAAA,CAAQ,CACzB,IAAMa,EAAQtB,CAAAA,CAAK,KAAA,CAAMa,CAAAA,CAASI,CAAG,EACrC,GAAIK,CAAAA,CAAO,CAET,IAAMvC,EAAe8B,CAAAA,CAAQ,KAAA,CAAMI,CAAAA,CAAKA,CAAAA,CAAMK,CAAAA,CAAM,QAAQ,CAAA,CAG5D,IAAA,IAAWD,KAAMC,CAAAA,CAAM,WAAA,CACrBN,CAAAA,CAAO,IAAA,CAAKK,CAAE,CAAA,CAMhB,GAAIC,CAAAA,CAAM,QAAA,CAAW,GAAKA,CAAAA,CAAM,WAAA,CAAY,MAAA,GAAW,CAAA,CACrDjC,CAAAA,CAAiBiC,CAAAA,CAAM,WAAA,CAAY,CAAC,EAAGvC,CAAY,CAAA,CACnD2B,CAAAA,CAAc,CACZ,KAAMV,CAAAA,CAAK,IAAA,CACX,YAAA,CAAAjB,CAAAA,CACA,WAAYuC,CAAAA,CAAM,WAAA,CAAY,CAAC,CACjC,CAAC,CAAA,CAAA,KAAA,GACQA,CAAAA,CAAM,QAAA,CAAW,GAAKA,CAAAA,CAAM,WAAA,CAAY,MAAA,CAASA,CAAAA,CAAM,SAChE,IAAA,IAAWD,CAAAA,IAAMC,CAAAA,CAAM,WAAA,CACrBZ,EAAc,CACZ,IAAA,CAAMV,CAAAA,CAAK,IAAA,CACX,YAAA,CAAAjB,CAAAA,CACA,UAAA,CAAYsC,CACd,CAAC,CAAA,CAKLJ,CAAAA,EAAOK,CAAAA,CAAM,QAAA,CACbR,EAAU,IAAA,CACVI,CAAAA,CAAU,IAAA,CACV,KACF,CACF,CAEKA,CAAAA,GAEHF,CAAAA,CAAO,IAAA,CAAKH,CAAAA,CAAQI,CAAG,CAAC,CAAA,CACxBA,KAEJ,CAGA,GAAID,CAAAA,CAAO,MAAA,CAASJ,EAKlB,MAAMW,GAAAA,CAAa,UAAA,CACjB,iCAAA,CACA,2BAA2BP,CAAAA,CAAO,MAAM,CAAA,wBAAA,EAA2BJ,CAAS,CAAA,CAAA,CAAA,CAC5E,CAAE,UAAA,CAAYI,CAAAA,CAAO,OAAQ,SAAA,CAAAJ,CAAU,CACzC,CAAA,CAGFC,EAAUG,EACZ,CAEA,OAAOH,CACT,CACF,ECjaO,IAAMd,CAAAA,CAAN,KAAiB,CAAjB,WAAA,EAAA,CAMN,IAAA,CAAQ,UAAA,CAAa,IAAI,GAAA,CAMzB,IAAA,CAAQ,IAAA,CAAO,IAAI,KAUnB,SAAA,CAAUK,CAAAA,CAAgBC,CAAAA,CAAyB,CAClD,IAAMmB,CAAAA,CAAQpB,CAAAA,CAAO,WAAA,EAAY,CAAE,IAAA,EAAK,CAAE,KAAA,CAAM,KAAK,EACrD,GAAIoB,CAAAA,CAAM,MAAA,GAAW,CAAA,EAAKA,EAAM,CAAC,CAAA,GAAM,EAAA,CAAI,OAE3C,KAAK,UAAA,CAAW,GAAA,CAAIA,CAAAA,CAAM,CAAC,CAAC,CAAA,CAE5B,IAAMC,CAAAA,CAAYD,EAAM,CAAC,CAAA,CACrBE,CAAAA,CAUJ,GARI,KAAK,IAAA,CAAK,GAAA,CAAID,CAAS,CAAA,CAC1BC,EAAO,IAAA,CAAK,IAAA,CAAK,GAAA,CAAID,CAAS,CAAA,EAE9BC,CAAAA,CAAO,CAAE,QAAA,CAAU,IAAI,GAAA,CAAO,QAAA,CAAU,IAAK,CAAA,CAC7C,IAAA,CAAK,IAAA,CAAK,GAAA,CAAID,CAAAA,CAAWC,CAAI,CAAA,CAAA,CAI1BF,CAAAA,CAAM,MAAA,GAAW,CAAA,CAAG,CACvBE,CAAAA,CAAK,QAAA,CAAW,CAAE,UAAArB,CAAAA,CAAW,MAAA,CAAAD,CAAAA,CAAQ,QAAA,CAAU,CAAE,CAAA,CACjD,MACD,CAGA,IAAA,IAASuB,EAAI,CAAA,CAAGA,CAAAA,CAAIH,CAAAA,CAAM,MAAA,CAAQG,CAAAA,EAAAA,CAAK,CACtC,IAAMrB,CAAAA,CAAOkB,EAAMG,CAAC,CAAA,CACfD,CAAAA,CAAK,QAAA,CAAS,IAAIpB,CAAI,CAAA,EAC1BoB,CAAAA,CAAK,QAAA,CAAS,IAAIpB,CAAAA,CAAM,CAAE,QAAA,CAAU,IAAI,GAAA,CAAO,QAAA,CAAU,IAAK,CAAC,EAEhEoB,CAAAA,CAAOA,CAAAA,CAAK,QAAA,CAAS,GAAA,CAAIpB,CAAI,EAC9B,CAEAoB,CAAAA,CAAK,QAAA,CAAW,CAAE,SAAA,CAAArB,CAAAA,CAAW,MAAA,CAAAD,CAAAA,CAAQ,QAAA,CAAUoB,CAAAA,CAAM,MAAO,EAC7D,CAeA,OAAA,CAAQjB,CAAAA,CAAiBU,CAAAA,CAAqC,CAE7D,GAAIA,CAAAA,EAAOV,CAAAA,CAAO,MAAA,CAAQ,OAAO,KAejC,IAAMqB,CAAAA,CAAYrB,CAAAA,CAAOU,CAAG,CAAA,CAAE,IAAA,CAC9B,GAAIW,CAAAA,GAAc,OAASA,CAAAA,CAAU,UAAA,CAAW,MAAM,CAAA,CAAG,OAAO,IAAA,CAGhE,IAAMH,CAAAA,CAAYlB,CAAAA,CAAOU,CAAG,CAAA,CAAE,KAAA,CAAM,WAAA,EAAY,CAChD,GAAI,CAAC,IAAA,CAAK,UAAA,CAAW,IAAIQ,CAAS,CAAA,CAAG,OAAO,IAAA,CAG5C,IAAIC,CAAAA,CAA6B,IAAA,CAAK,IAAA,CAAK,GAAA,CAAID,CAAS,CAAA,CACxD,GAAI,CAACC,CAAAA,CAAM,OAAO,IAAA,CAElB,IAAIG,CAAAA,CAA+B,KAC/BC,CAAAA,CAAQ,CAAA,CAGZ,GAAIJ,CAAAA,CAAK,SAAU,CAClB,IAAM3C,CAAAA,CAAewB,CAAAA,CAAO,MAAMU,CAAAA,CAAKA,CAAAA,CAAM,CAAC,CAAA,CAC9CY,CAAAA,CAAO,CACN,QAAA,CAAU,CAAA,CACV,YAAa,CAACjD,CAAAA,CAAiB8C,CAAAA,CAAK,QAAA,CAAS,UAAWA,CAAAA,CAAK,QAAA,CAAS,MAAA,CAAQ3C,CAAY,CAAC,CAAA,CAC3F,QAAA,CAAU2C,CAAAA,CAAK,QAAA,CAAS,MACzB,EACD,CAGA,IAAA,IAASC,EAAIV,CAAAA,CAAM,CAAA,CAAGU,CAAAA,CAAIpB,CAAAA,CAAO,QAAUmB,CAAAA,EAAM,QAAA,CAAUC,CAAAA,EAAAA,CAAK,CAC/D,IAAMI,CAAAA,CAAWxB,CAAAA,CAAOoB,CAAC,CAAA,CAAE,IAAA,CAC3B,GAAII,CAAAA,GAAa,KAAA,EAASA,EAAS,UAAA,CAAW,MAAM,CAAA,CAAG,MACvD,IAAMzB,CAAAA,CAAOC,CAAAA,CAAOoB,CAAC,CAAA,CAAE,MAAM,WAAA,EAAY,CAEzC,GADAD,CAAAA,CAAOA,CAAAA,CAAK,QAAA,CAAS,GAAA,CAAIpB,CAAI,EACzB,CAACoB,CAAAA,CAAM,MAGX,GAFAI,IAEIJ,CAAAA,CAAK,QAAA,CAAU,CAElB,IAAM3C,EAAewB,CAAAA,CAAO,KAAA,CAAMU,CAAAA,CAAKA,CAAAA,CAAMa,CAAK,CAAA,CAClDD,CAAAA,CAAO,CACN,SAAUC,CAAAA,CACV,WAAA,CAAa,CACZlD,CAAAA,CAAiB8C,EAAK,QAAA,CAAS,SAAA,CAAWA,CAAAA,CAAK,QAAA,CAAS,OAAQ3C,CAAY,CAC7E,CAAA,CACA,QAAA,CAAU2C,CAAAA,CAAK,QAAA,CAAS,MACzB,EACD,CACD,CAEA,OAAOG,CACR,CAKA,IAAI,IAAA,EAAe,CAClB,OAAO,IAAA,CAAK,KAAK,IAClB,CASA,aAAA,EAAwC,CACvC,IAAMb,CAAAA,CAAiC,EAAC,CAElCgB,EAAU,CAACN,CAAAA,CAAgBO,CAAAA,GAAyB,CACrDP,EAAK,QAAA,GACRV,CAAAA,CAAOiB,CAAAA,CAAK,IAAA,CAAK,GAAG,CAAC,CAAA,CAAIP,CAAAA,CAAK,QAAA,CAAS,SAAA,CAAA,CAExC,IAAA,GAAW,CAACpB,CAAAA,CAAM4B,CAAK,CAAA,GAAKR,CAAAA,CAAK,QAAA,CAChCM,CAAAA,CAAQE,EAAO,CAAC,GAAGD,CAAAA,CAAM3B,CAAI,CAAC,EAEhC,CAAA,CAEA,IAAA,GAAW,CAACmB,CAAAA,CAAWC,CAAI,CAAA,GAAK,IAAA,CAAK,KACpCM,CAAAA,CAAQN,CAAAA,CAAM,CAACD,CAAS,CAAC,CAAA,CAG1B,OAAOT,CACR,CAGA,QAAA,CAASV,CAAAA,CAAuB,CAC/B,OAAO,IAAA,CAAK,UAAA,CAAW,GAAA,CAAIA,CAAAA,CAAK,aAAa,CAC9C,CACD,EC/NA,IAAM6B,CAAAA,CAAyB,IAAI,GAAA,CAAI,CAAC,MAAO,GAAA,CAAK,IAAA,CAAM,MAAA,CAAQ,OAAO,CAAC,CAAA,CAEpEC,CAAAA,CAAqB,IAAI,IAAI,CACjC,IAAA,CAAM,OAAA,CAAS,UAAA,CAAY,WAAY,OAAA,CAAS,UAAA,CAAY,QAAA,CAAU,IACxE,CAAC,CAAA,CA2BM,SAASC,CAAAA,CACfC,GAAAA,CAAmB,EAAA,CACnBC,CAAAA,CACiB,CACjB,IAAMC,EAAcD,CAAAA,GAAcjC,CAAAA,EAAiB8B,CAAAA,CAAmB,GAAA,CAAI9B,EAAK,WAAA,EAAa,CAAA,CAAA,CAE5F,OAAO,CACJ,IAAA,CAAM,mBAAA,CACN,QAAA,CAAAgC,GAAAA,CACA,KAAA,CAAM/B,CAAAA,CAAiBU,CAAAA,CAAqC,CAE1D,GAAIA,CAAAA,CAAM,CAAA,EAAKV,CAAAA,CAAO,MAAA,CAAQ,OAAO,IAAA,CAErC,IAAM,CAAA,CAAIA,CAAAA,CAAOU,CAAG,CAAA,CACdwB,CAAAA,CAAOlC,CAAAA,CAAOU,CAAAA,CAAM,CAAC,CAAA,CAKrByB,CAAAA,CAAYD,CAAAA,CAAK,MAAM,WAAA,EAAY,CAiBzC,GAhBID,CAAAA,CAAYE,CAAS,CAAA,EASrBP,CAAAA,CAAuB,GAAA,CAAIO,CAAS,GAAKnC,CAAAA,CAAOU,CAAAA,CAAM,CAAC,CAAA,EAAG,IAAA,GAAS,MAAA,EAOnE,EAAA,CAHD,CAAA,CAAE,OAAS,QAAA,EAAY,CAAA,CAAE,IAAA,GAAS,QAAA,IAClCwB,EAAK,IAAA,GAAS,OAAA,EAAWA,CAAAA,CAAK,IAAA,GAAS,UAAYA,CAAAA,CAAK,IAAA,GAAS,IAAA,EAAQA,CAAAA,CAAK,IAAA,GAAS,GAAA,CAAA,CAAA,CAE3E,OAAO,IAAA,CAGtB,IAAME,CAAAA,CAAY,IAAIxD,CAAAA,CACpB,MAAA,CAAQC,EAAY,MAAM,CAAA,CAAG,GAAA,CAAK,GAAA,CAClCqD,EAAK,MAAA,CAAQ,CAAA,CAAGA,CAAAA,CAAK,IAAA,CAAMA,CAAAA,CAAK,GAClC,CAAA,CAIA,OAAO,CAAE,QAAA,CAAU,CAAA,CAAG,WAAA,CAAa,CAAC,EAAGE,CAAS,CAAE,CACpD,CACF,CACF,CA2CO,SAASC,CAAAA,CAAqBrC,CAAAA,CAAiBU,CAAAA,CAAsB,CAC1E,IAAM4B,CAAAA,CAAuB,EAAC,CAC9B,IAAA,IAASlB,CAAAA,CAAI,CAAA,CAAGA,EAAIV,CAAAA,CAAKU,CAAAA,EAAAA,CAAK,CAC5B,IAAMmB,EAAIvC,CAAAA,CAAOoB,CAAC,CAAA,CAClB,GAAImB,CAAAA,CAAE,IAAA,GAAS,UAAA,CACbD,CAAAA,CAAU,KAAK,IAAI,CAAA,CAAA,KAAA,GACVC,CAAAA,CAAE,IAAA,GAAS,SAAU,CAC9B,IAAMC,CAAAA,CAAOxC,CAAAA,CAAOoB,EAAI,CAAC,CAAA,CACnBqB,CAAAA,CAAqB,CAAC,CAACD,CAAAA,GAC3BA,CAAAA,CAAK,IAAA,GAAS,OAASA,CAAAA,CAAK,IAAA,GAAS,QAAA,EAAYA,CAAAA,CAAK,OAAS,QAAA,EAAYA,CAAAA,CAAK,IAAA,GAAS,SAAA,EACxFA,EAAK,IAAA,GAAS,OAAA,GAAYA,CAAAA,CAAK,KAAA,CAAM,WAAA,EAAY,GAAM,KAAA,EAASA,CAAAA,CAAK,MAAM,WAAA,EAAY,GAAM,QAAA,EAAYA,CAAAA,CAAK,MAAM,WAAA,EAAY,GAAM,KAAA,EAASA,CAAAA,CAAK,MAAM,WAAA,EAAY,GAAM,MAAA,CAAA,CAAA,CAE/KF,CAAAA,CAAU,IAAA,CAAKG,CAAkB,EACnC,CAAA,KAAA,CAAWF,EAAE,IAAA,GAAS,UAAA,EAAcA,CAAAA,CAAE,IAAA,GAAS,WAC7CD,CAAAA,CAAU,GAAA,GAEd,CACA,OAAOA,CAAAA,CAAU,MAAA,CAAS,CAAA,EAAKA,CAAAA,CAAUA,CAAAA,CAAU,MAAA,CAAS,CAAC,CAC/D,CAoBO,SAASI,CAAAA,EAAiD,CAC/D,OAAO,CAELZ,CAAAA,EACF,CACF,KASaa,CAAAA,CAA0C,CACtD,iBAAA,CAAmB,OAAA,CACnB,UAAA,CAAY,OAAA,CACZ,aAAA,CAAe,aAAA,CACf,cAAe,aAAA,CACf,UAAA,CAAY,UAAA,CACZ,aAAA,CAAe,cACf,WAAA,CAAa,WAAA,CAKb,eAAA,CAAiB,aAAA,CACjB,aAAc,WACf","file":"chunk-5FO4PDIG.js","sourcesContent":["//#region ─── Module Overview ───────────────────────────────────────────────────\n\n/**\n * TokenNormalizer, post-lexer token normalization pass.\n *\n * ## Purpose\n * Applies domain-specific {@link NormalizerRule | NormalizerRules} to the raw\n * token stream produced by the {@link ExpressionLexer}. This keeps the lexer\n * slim and focused on single-token production, while multi-token pattern\n * matching (phrases, implicit operators, domain merges) lives here.\n *\n * ## What rules can do\n * - **Phrase fusion**: Merge consecutive words into compound tokens\n * (e.g., `IDENT + ... + IDENT` → `CARET`)\n * - **Implicit operator insertion**: Insert missing operators between tokens\n * (e.g., `NUMBER IDENT` → `NUMBER STAR IDENT`)\n * - **Domain-specific transformations**: Coalesce item names, currency pairs,\n * percentage syntax, etc.\n *\n * ## Architecture\n * Providers register NormalizerRules alongside Parselets and OpCode handlers\n * via {@link IEnginePackage.normalizerRules}. The normalizer applies them\n * greedily left-to-right in multiple passes with safety limits.\n *\n * @module TokenNormalizer\n */\n\n//#endregion\n//#region ─── Imports ──────────────────────────────────────────────────────────\n\nimport type { Token } from \"@solve-js/lexer/Token\";\nimport { tokenTypeId } from \"@solve-js/lexer/Token\";\nimport { LexerToken } from \"@solve-js/lexer/ExpressionLexer\";\nimport type { NormalizerRule, TokenFusion } from \"./NormalizerRule\";\nimport { PhraseTrie } from \"./PhraseTrie\";\nimport { ErrorFactory } from \"@solve-js/errors/UnifiedErrorFramework\";\n\n//#endregion\n//#region ─── NormalizerOptions, Configuration ────────────────────────────────\n\n/**\n * Configuration options for the normalization pass.\n *\n * These control safety limits and diagnostic callbacks. The defaults\n * are chosen to be generous enough for any realistic expression while\n * preventing runaway token expansion from recursive rules.\n */\nexport interface NormalizerOptions {\n /**\n * Maximum number of full passes over the token stream before bailing out.\n * Prevents infinite loops from recursive rule chains.\n * @default 100\n */\n maxPasses?: number;\n\n /**\n * Maximum number of tokens allowed after normalization.\n * If exceeded, an Error is thrown rather than passing a bloated stream\n * to the parser.\n * @default 10000\n */\n maxTokens?: number;\n\n /**\n * Callback invoked for each fusion event during normalization.\n * Used by diagnostic mode to populate {@link NormalizerOutput.fusions}.\n * When `undefined`, fusions are still tracked internally but no callbacks fire.\n */\n onFusion?: (fusion: TokenFusion) => void;\n}\n\n//#endregion\n//#region ─── Default Options ──────────────────────────────────────────────────\n\n/** Sensible defaults that catch infinite loops without limiting real expressions. */\nconst DEFAULT_OPTIONS: Required<NormalizerOptions> = {\n maxPasses: 100,\n maxTokens: 10000,\n onFusion: () => {},\n};\n\n//#endregion\n//#region ─── createFusedToken, Token Factory ──────────────────────────────────\n\n/**\n * Creates a new normalized token from fused source tokens.\n *\n * The fused token inherits position information (offset, line, column)\n * from the first source token, which preserves source-map accuracy\n * for error messages and diagnostic highlighting.\n *\n * It also records where the source text ENDS, on `sourceEnd`. The start alone\n * is not enough to describe the span a fusion covers, because `text` is the\n * replacement rather than the original: `10 frames` fuses into a FRAME_COUNT\n * whose text is `10`, and a timecode fuses into a token whose text is a\n * comma-separated tuple that appears nowhere in the line. Anything painting the\n * line needs both ends, and only this function is in a position to know them.\n *\n * @param type - The new token type (e.g., \"CARET\", \"TIMES_BY\")\n * @param text - The combined text representation (e.g., \"to the power of\")\n * @param sourceTokens - The original tokens being fused (at least 2)\n * @returns A new {@link LexerToken} with the fused type and combined text\n */\nexport function createFusedToken(\n type: string,\n text: string,\n sourceTokens: Token[]\n): Token {\n const first = sourceTokens[0];\n const last = sourceTokens[sourceTokens.length - 1];\n const token = new LexerToken(\n type,\n tokenTypeId(type),\n text,\n text,\n first.offset,\n first.lineBreaks ?? 0,\n first.line,\n first.col,\n );\n // `text`, not `value`: this is where the SOURCE ended, and the two differ\n // for a string literal, whose value is the payload without its quotes.\n token.sourceEnd = last.sourceEnd ?? last.offset + last.text.length;\n return token;\n}\n\n/**\n * Record, on a token that replaced several, where its source text ended.\n *\n * Called centrally rather than left to each rule. A rule that builds its\n * replacement by hand rather than through {@link createFusedToken} is doing\n * nothing wrong, and several do: the date-literal rule needs a token whose\n * value is an epoch and whose text is the source, which that factory cannot\n * express. Stamping here means every fusion carries its span, including ones\n * written after this was.\n *\n * Reads `sourceEnd` off the last source token when it has one, so a fusion of\n * a fusion still describes the original text rather than the intermediate.\n *\n * @param fused - The single token the rule produced.\n * @param sourceTokens - The tokens it consumed.\n */\nfunction recordSourceSpan(fused: Token, sourceTokens: Token[]): void {\n const last = sourceTokens[sourceTokens.length - 1];\n if (!last) return;\n fused.sourceEnd = last.sourceEnd ?? last.offset + last.text.length;\n}\n\n//#endregion\n//#region ─── NON_WORD_TOKEN_TYPES, Type-guard skip set ────────────────────────\n\n/**\n * Token types that can NEVER start a multi-word phrase.\n *\n * Used by {@link normalize} to skip the PhraseTrie walk entirely at\n * positions where the token type makes phrase matching impossible.\n * This avoids even the O(1) {@link PhraseTrie.canStart} check.\n *\n * Types NOT in this set (IDENT, KEYWORD, FUNC, UNIT, and any custom\n * types registered by packages) still pass through to the trie for\n * a full match attempt.\n */\n// ── Non-word type ID lookup table (flat Uint8Array, true O(1) array index) ──\n//\n// Index = token typeId, value = 1 if non-word (skip trie), 0 otherwise.\n// Array indexing avoids ALL hashing: no Set.has(), no Map.get(), no string ops.\n//\n// Custom types from packages get IDs beyond the table length, so the bounds\n// check `tid < TABLE.length` safely passes them through to the trie.\n//\n// Arithmetic operators (PLUS, MINUS, STAR, SLASH, CARET, MOD, PERCENT) are\n// intentionally excluded: in keyword locales the lexer maps \"times\"→STAR,\n// \"divide\"→SLASH etc., and those tokens CAN start phrases like \"times by\".\n/**\n * Token type names that can NEVER start a multi-word phrase.\n *\n * Exported for testing only, consumers should use the type-guard behavior\n * of {@link TokenNormalizer.normalize} rather than this list directly.\n */\nexport const NON_WORD_NAMES = [\n\t\"NUMBER\", \"HEX\", \"BIGINT\", \"FLOAT\",\n\t\"LSHIFT\", \"RSHIFT\", \"BIT_AND\", \"BIT_OR\", \"BIT_XOR\",\n\t\"LPAREN\", \"RPAREN\", \"LBRACKET\", \"RBRACKET\",\n\t\"COMMA\", \"COLON\", \"EQUALS\", \"THEREFORE\", \"PIPE\", \"AMPERSAND\", \"AT\",\n\t\"SEMICOLON\", \"QUESTION\", \"EXCLAMATION\",\n\t\"EOF\", \"WS\", \"NEWLINE\",\n] as const;\n\n/**\n * Flat Uint8Array lookup table: index = token typeId, value = 1 if non-word.\n *\n * Exported for testing only, consumers should not depend on the internal\n * table layout, as the set of non-word types may change.\n */\nexport const NON_WORD_TABLE: Uint8Array = (() => {\n\t// Resolve all non-word type names to their numeric IDs\n\tconst ids = NON_WORD_NAMES.map(n => tokenTypeId(n));\n\t// Size the table to cover the largest ID + 1\n\tconst len = Math.max(...ids) + 1;\n\tconst table = new Uint8Array(len);\n\t// Mark non-word type positions\n\tfor (const id of ids) table[id] = 1;\n\treturn table;\n})();\n\n//#endregion\n//#region ─── TokenNormalizer Class ─────────────────────────────────────────────\n\n/**\n * Token normalizer: applies {@link NormalizerRule | NormalizerRules} to a token stream.\n *\n * ## Lifecycle\n * 1. **Registration**: Rules are added via {@link register} and sorted by priority\n * 2. **Normalization**: {@link normalize} applies rules greedily left-to-right\n * 3. **Cleanup**: {@link clear} or {@link unregister} removes rules\n *\n * ## Normalization algorithm\n * The normalizer uses a greedy left-to-right multi-pass algorithm:\n * - At each token position, rules are tried in priority order (highest first)\n * - When a rule matches, matched tokens are consumed and replaced\n * - Processing continues from the replacement position\n * - Multiple passes handle cascading matches (one rule's output triggers another)\n * - Safety limits ({@link NormalizerOptions.maxPasses}) prevent infinite loops\n *\n * @example\n * ```ts\n * const normalizer = new TokenNormalizer();\n * normalizer.register(phraseRule); // \"to the power of\" → CARET\n * normalizer.register(implicitMultRule); // \"2 x\" → \"2 * x\"\n * const normalized = normalizer.normalize(rawTokens);\n * ```\n */\nexport class TokenNormalizer {\n /** Registered rules, unsorted, the source of truth. */\n private rules: NormalizerRule[] = [];\n\n /**\n * Priority-sorted copy of {@link rules}, rebuilt lazily on the next\n * {@link normalize} call after a mutation. Rules are registered once at\n * engine/package-registration time and essentially never change during a\n * session, but normalize() runs on every keystroke-driven evaluation, an\n * earlier version re-sorted a fresh copy of `rules` on every single call,\n * which meant every keystroke paid for an allocation + sort of a list that\n * had usually not changed since the last one. `null` means \"stale, rebuild\n * on next use\"; {@link register}/{@link unregister}/{@link clear} all\n * invalidate it.\n */\n private sortedRulesCache: NormalizerRule[] | null = null;\n\n /**\n * Phrase trie for single-pass multi-word phrase fusion.\n * Tried at each token position BEFORE other rules, the trie walk\n * is O(depth) vs O(R × W) for separate rule matching.\n */\n private phraseTrie = new PhraseTrie();\n\n /** Merged options with defaults applied. */\n private options: Required<NormalizerOptions>;\n\n // ── Constructor ──────────────────────────────────────────────────────────\n\n /**\n * @param options - Configuration overrides for safety limits and diagnostic callbacks\n */\n constructor(options: NormalizerOptions = {}) {\n this.options = { ...DEFAULT_OPTIONS, ...options };\n }\n\n // ── Rule Management ──────────────────────────────────────────────────────\n\n /**\n * Register a normalization rule.\n *\n * Rules are sorted by priority (descending) on each {@link normalize} call.\n * Multiple rules can share the same priority, they are tried in registration\n * order when priorities are equal.\n *\n * @param rule - The rule to register\n */\n register(rule: NormalizerRule): void {\n this.rules.push(rule);\n this.sortedRulesCache = null;\n }\n\n /**\n * Unregister a normalization rule by its {@link NormalizerRule.name | name}.\n *\n * If multiple rules share the same name, all are removed. This is safe to\n * call with a name that doesn't match any rule, it simply has no effect.\n *\n * @param ruleName - The name of the rule to remove\n */\n unregister(ruleName: string): void {\n this.rules = this.rules.filter(r => r.name !== ruleName);\n this.sortedRulesCache = null;\n }\n\n /**\n * Remove all registered rules, resetting the normalizer to its initial state.\n * Also clears the phrase trie.\n */\n clear(): void {\n this.rules = [];\n this.sortedRulesCache = null;\n this.phraseTrie = new PhraseTrie();\n }\n\n /**\n * Priority-sorted view of {@link rules} (descending priority; registration\n * order preserved for ties, since {@link Array.prototype.sort} is stable).\n * Cached until the next mutation. See {@link sortedRulesCache}.\n */\n private getSortedRules(): NormalizerRule[] {\n if (this.sortedRulesCache === null) {\n this.sortedRulesCache = [...this.rules].sort((a, b) => b.priority - a.priority);\n }\n return this.sortedRulesCache;\n }\n\n /**\n * Get the number of currently registered rules (excludes phrase trie entries).\n */\n get ruleCount(): number {\n return this.rules.length;\n }\n\n // ── Phrase Registration ────────────────────────────────────────────────\n\n /**\n * Register a multi-word phrase for fusion into a single compound token.\n *\n * This is the preferred way to add phrase patterns. It inserts into the\n * internal {@link PhraseTrie}, which collapses all phrase rules into a\n * single O(depth) trie walk per position, no separate rule scanning.\n *\n * @param phrase - Multi-word phrase (e.g., \"to the power of\", \"abyssal whip\")\n * @param tokenType - Target token type after fusion (e.g., \"CARET\", \"ITEM\")\n */\n addPhrase(phrase: string, tokenType: string): void {\n this.phraseTrie.addPhrase(phrase, tokenType);\n }\n\n /**\n * Check whether a word can start any registered phrase.\n *\n * Used by {@link implicitMultiplyRule} to suppress `*` insertion\n * before phrase-starting identifiers (e.g., \"2 power of 3\" → `2 ^ 3`,\n * not `2 * power of 3`). Delegates to {@link PhraseTrie.canStart}.\n */\n /**\n * Get all registered phrases and their target token types.\n *\n * Exposes the full phrase trie structure for diagnostic rendering\n * in the playground's NormalizerTab. Returns ALL registered phrases,\n * not just the ones that matched in the last evaluation.\n */\n getPhrases(): Record<string, string> {\n return this.phraseTrie.getAllPhrases();\n }\n\n canStartPhrase(word: string): boolean {\n return this.phraseTrie.canStart(word);\n }\n\n // ── Normalization ────────────────────────────────────────────────────────\n\n /**\n * Normalize a token stream by applying all registered rules.\n *\n * ## Algorithm\n * Applies rules greedily left-to-right in multiple passes:\n * 1. Sort rules by priority (descending)\n * 2. Walk the token stream left to right\n * 3. At each position, try rules in priority order\n * 4. On match: consume matched tokens, insert replacements, restart from insert point\n * 5. On no match: pass token through unchanged\n * 6. Repeat until a full pass produces no changes, or maxPasses is reached\n *\n * ## Fusion tracking\n * When a rule consumes more tokens than it produces, the normalizer calls\n * `onFusion` with a {@link TokenFusion} record for diagnostic collection.\n * This populates {@link NormalizerOutput.fusions} in the playground pipeline view.\n *\n * ## Safety\n * If the normalized token count exceeds {@link NormalizerOptions.maxTokens},\n * an Error is thrown to prevent memory exhaustion from runaway rule expansion.\n *\n * @param tokens - Raw tokens from the lexer\n * @param onFusion - Optional fusion callback (overrides {@link NormalizerOptions.onFusion})\n * @returns Normalized tokens ready for parsing\n * @throws {Error} If the normalized token count exceeds maxTokens\n */\n normalize(tokens: Token[], onFusion?: (fusion: TokenFusion) => void): Token[] {\n // ── Early exit: nothing to normalize ──\n if (tokens.length === 0) return tokens;\n\n // ── Priority-sorted rules, cached across calls, see getSortedRules() ──\n const sorted = this.getSortedRules();\n const fusionHandler = onFusion ?? this.options.onFusion;\n const maxPasses = this.options.maxPasses;\n const maxTokens = this.options.maxTokens;\n\n let current = tokens;\n let changed = true;\n let passCount = 0;\n\n // Multi-pass loop: rules may trigger cascading matches across passes\n while (changed && passCount < maxPasses) {\n changed = false;\n passCount++;\n\n const result: Token[] = [];\n let pos = 0;\n\n // Single-pass left-to-right greedy walk\n while (pos < current.length) {\n let matched = false;\n\n // ── Fast path: phrase trie (O(depth) single walk vs O(R × W) per rule) ──\n // O(1) type-guard: skip trie entirely for tokens that can't start phrases.\n // Flat Uint8Array indexed by typeId, no hashing, no Set lookup, true O(1).\n const tid = current[pos].typeId;\n if (tid >= NON_WORD_TABLE.length || NON_WORD_TABLE[tid] === 0) {\n const trieMatch = this.phraseTrie.matchAt(current, pos);\n if (trieMatch) {\n const sourceTokens = current.slice(pos, pos + trieMatch.consumed);\n for (const rt of trieMatch.replacement) {\n result.push(rt);\n }\n if (trieMatch.consumed > 1 && trieMatch.replacement.length === 1) {\n recordSourceSpan(trieMatch.replacement[0], sourceTokens);\n fusionHandler({\n rule: trieMatch.ruleName ?? \"phrase-trie\",\n sourceTokens,\n fusedToken: trieMatch.replacement[0],\n });\n }\n pos += trieMatch.consumed;\n changed = true;\n continue; // trie matched — skip other rules at this position (no need to set matched)\n }\n }\n\n // Try every rule in priority order at this position\n for (const rule of sorted) {\n const match = rule.match(current, pos);\n if (match) {\n // Collect source tokens for fusion tracking\n const sourceTokens = current.slice(pos, pos + match.consumed);\n\n // Insert replacement tokens into result\n for (const rt of match.replacement) {\n result.push(rt);\n }\n\n // Track fusion events for diagnostics:\n // - Multiple tokens → single token: classic fusion\n // - Multiple tokens → fewer tokens: partial fusion\n if (match.consumed > 1 && match.replacement.length === 1) {\n recordSourceSpan(match.replacement[0], sourceTokens);\n fusionHandler({\n rule: rule.name,\n sourceTokens,\n fusedToken: match.replacement[0],\n });\n } else if (match.consumed > 1 && match.replacement.length < match.consumed) {\n for (const rt of match.replacement) {\n fusionHandler({\n rule: rule.name,\n sourceTokens,\n fusedToken: rt,\n });\n }\n }\n\n // Advance position past consumed tokens\n pos += match.consumed;\n changed = true;\n matched = true;\n break; // Rule matched — restart at new position with highest-priority rules\n }\n }\n\n if (!matched) {\n // No rule matched at this position, pass token through unchanged\n result.push(current[pos]);\n pos++;\n }\n }\n\n // Safety: bail if token count explodes (runaway rule expansion)\n if (result.length > maxTokens) {\n // A raw Error here would bypass ThreeTierEvaluator's DAG-preservation\n // enrichment on compile failure (it specifically checks for\n // EngineError). Same reasoning as ExpressionEngineSafety.ts's\n // complexity/length checks, which this mirrors.\n throw ErrorFactory.validation(\n \"NORMALIZED_TOKEN_LIMIT_EXCEEDED\",\n `Normalized token count (${result.length}) exceeds safety limit (${maxTokens})`,\n { tokenCount: result.length, maxTokens }\n );\n }\n\n current = result;\n }\n\n return current;\n }\n}\n\n//#endregion\n","//#region ─── Module Overview ───────────────────────────────────────────────────\n\n/**\n * PhraseTrie, optimized word-level trie for multi-word phrase fusion.\n *\n * ## Problem\n * The normalizer previously applied N separate `phraseFusionRule` instances,\n * each scanning forward from the current token position. For R phrase rules\n * and W average phrase length, this cost O(N × R × W) per pass.\n *\n * ## Solution\n * A single trie walk per position collapses all phrase rules into O(D)\n * where D ≤ longest phrase depth (typically ≤ 5 words). The trie tracks\n * the deepest terminal node reached, implementing longest-match-wins\n * without priority sorting.\n *\n * ## Optimizations\n * 1. **Set<string> quick-reject**, the `startWords` set contains the first\n * word of every registered phrase. At each position, if the token's\n * lowercase value isn't in the set, we bail in O(1) without touching\n * the trie. ~80% of tokens (numbers, operators) hit this fast path.\n * 2. **Longest-match-wins**, the `matchAt()` walk continues past terminal\n * nodes, tracking the deepest one. Shorter overlapping phrases (e.g.,\n * \"power of\") don't need lower priority, the trie naturally prefers\n * the longer match.\n * 3. **Map-based children**, `Map<string, TrieNode>` gives O(1) amortized\n * child lookup per word, faster than array scanning for sparse branches.\n *\n * ## Package integration\n * Packages add phrases via {@link TokenNormalizer.addPhrase} (public API)\n * or the {@link IEnginePackage.phrases} declarative field. Each call to\n * `addPhrase()` inserts the phrase into this trie and updates `startWords`.\n *\n * @module PhraseTrie\n */\n\n//#endregion\n//#region ─── Imports ──────────────────────────────────────────────────────────\n\nimport type { Token } from \"@solve-js/lexer/Token\";\nimport type { NormalizerMatch } from \"./NormalizerRule\";\nimport { createFusedToken } from \"./TokenNormalizer\";\n\n//#endregion\n//#region ─── TrieNode ─────────────────────────────────────────────────────────\n\n/**\n * Internal trie node.\n *\n * Each node represents a matched word in a phrase. The path from root\n * to a node spells out a partial or complete phrase.\n */\ninterface TrieNode {\n\t/** Child nodes keyed by next word (all lowercase). */\n\tchildren: Map<string, TrieNode>;\n\t/**\n\t * If this node completes a phrase, the terminal metadata.\n\t * A node can be both terminal AND have children. This handles\n\t * overlapping phrases like \"power of\" and \"to the power of\".\n\t */\n\tterminal: TrieTerminal | null;\n}\n\n/** Metadata stored at terminal nodes for deferred token creation. */\ninterface TrieTerminal {\n\t/** The target token type after fusion (e.g., \"CARET\", \"TIMES_BY\"). */\n\ttokenType: string;\n\t/** The original phrase string (e.g., \"to the power of\"). */\n\tphrase: string;\n\t/** Number of tokens consumed by this phrase. */\n\tconsumed: number;\n}\n\n//#endregion\n//#region ─── PhraseTrie ─────────────────────────────────────────────────────\n\n/**\n * Word-level trie for single-pass multi-word phrase matching.\n *\n * @example\n * ```ts\n * const trie = new PhraseTrie();\n * trie.addPhrase(\"to the power of\", \"CARET\");\n * trie.addPhrase(\"power of\", \"CARET\");\n * trie.addPhrase(\"abyssal whip\", \"ITEM\");\n *\n * // At position 0 with tokens [\"to\",\"the\",\"power\",\"of\",\"3\"]\n * const match = trie.matchAt(tokens, 0);\n * // → { consumed: 4, replacement: [CARET(\"to the power of\")] }\n * ```\n */\nexport class PhraseTrie {\n\t/**\n\t * First words that can start any registered phrase (all lowercase).\n\t * O(1) quick-reject: if `tokens[pos].value.toLowerCase()` isn't here,\n\t * no phrase can match at this position.\n\t */\n\tprivate startWords = new Set<string>();\n\n\t/**\n\t * Root maps first word → child node.\n\t * Two-level root avoids an unnecessary intermediate TrieNode.\n\t */\n\tprivate root = new Map<string, TrieNode>();\n\n\t// ── Registration ──────────────────────────────────────────────────────\n\n\t/**\n\t * Register a phrase for fusion into a single compound token.\n\t *\n\t * @param phrase - Multi-word phrase (e.g., \"to the power of\")\n\t * @param tokenType - Target token type after fusion (e.g., \"CARET\")\n\t */\n\taddPhrase(phrase: string, tokenType: string): void {\n\t\tconst words = phrase.toLowerCase().trim().split(/\\s+/);\n\t\tif (words.length === 0 || words[0] === \"\") return;\n\n\t\tthis.startWords.add(words[0]);\n\n\t\tconst firstWord = words[0];\n\t\tlet node: TrieNode;\n\n\t\tif (this.root.has(firstWord)) {\n\t\t\tnode = this.root.get(firstWord)!;\n\t\t} else {\n\t\t\tnode = { children: new Map(), terminal: null };\n\t\t\tthis.root.set(firstWord, node);\n\t\t}\n\n\t\t// Single-word phrase: terminal at root's child\n\t\tif (words.length === 1) {\n\t\t\tnode.terminal = { tokenType, phrase, consumed: 1 };\n\t\t\treturn;\n\t\t}\n\n\t\t// Walk/insert remaining words\n\t\tfor (let i = 1; i < words.length; i++) {\n\t\t\tconst word = words[i];\n\t\t\tif (!node.children.has(word)) {\n\t\t\t\tnode.children.set(word, { children: new Map(), terminal: null });\n\t\t\t}\n\t\t\tnode = node.children.get(word)!;\n\t\t}\n\n\t\tnode.terminal = { tokenType, phrase, consumed: words.length };\n\t}\n\n\t// ── Matching ──────────────────────────────────────────────────────────\n\n\t/**\n\t * Attempt to match a phrase starting at `pos` in the token stream.\n\t *\n\t * Walks the trie one token at a time, tracking the deepest terminal\n\t * node reached. Returns the longest match found, or `null` if no\n\t * phrase starts at this position.\n\t *\n\t * @param tokens - The current token stream\n\t * @param pos - Position to attempt matching from\n\t * @returns The longest {@link NormalizerMatch}, or `null` on no match\n\t */\n\tmatchAt(tokens: Token[], pos: number): NormalizerMatch | null {\n\t\t// ── Bounds guard ──\n\t\tif (pos >= tokens.length) return null;\n\n\t\t// ── A tag token never takes part in phrase fusion (#197, #213) ──\n\t\t// A `#tag` is a typed token, not a bare word, so it must not start or\n\t\t// complete a phrase. The trie matches on written value alone, so without\n\t\t// this guard a tag whose name equals a phrase or a phrase-continuation\n\t\t// word (`1200 #assuming`, `total of #column`) is fused into that grammar\n\t\t// before the category-tag rules ever run, and the tag is lost.\n\t\t//\n\t\t// The guard also covers the fused aggregate tokens the tags package emits\n\t\t// (`TAG_SUM` / `TAG_COUNT` / `TAG_AVERAGE`), whose value is the tag NAME:\n\t\t// otherwise `total of #assuming` fuses correctly to TAG_SUM(\"assuming\"),\n\t\t// then the trie re-reads that value on the next pass and turns it back\n\t\t// into the finance ASSUMING keyword (#213). The whole `TAG` / `TAG_*`\n\t\t// namespace belongs to the tags package.\n\t\tconst startType = tokens[pos].type;\n\t\tif (startType === \"TAG\" || startType.startsWith(\"TAG_\")) return null;\n\n\t\t// ── O(1) quick-reject: first word not a phrase starter ──\n\t\tconst firstWord = tokens[pos].value.toLowerCase();\n\t\tif (!this.startWords.has(firstWord)) return null;\n\n\t\t// ── Walk trie, tracking deepest terminal ──\n\t\tlet node: TrieNode | undefined = this.root.get(firstWord);\n\t\tif (!node) return null;\n\n\t\tlet best: NormalizerMatch | null = null;\n\t\tlet depth = 1;\n\n\t\t// Check single-word phrase at first node\n\t\tif (node.terminal) {\n\t\t\tconst sourceTokens = tokens.slice(pos, pos + 1);\n\t\t\tbest = {\n\t\t\t\tconsumed: 1,\n\t\t\t\treplacement: [createFusedToken(node.terminal.tokenType, node.terminal.phrase, sourceTokens)],\n\t\t\t\truleName: node.terminal.phrase,\n\t\t\t};\n\t\t}\n\n\t\t// Walk deeper for multi-word phrases\n\t\tfor (let i = pos + 1; i < tokens.length && node?.children; i++) {\n\t\t\tconst contType = tokens[i].type;\n\t\t\tif (contType === \"TAG\" || contType.startsWith(\"TAG_\")) break; // a tag token can't continue a phrase (#197, #213)\n\t\t\tconst word = tokens[i].value.toLowerCase();\n\t\t\tnode = node.children.get(word);\n\t\t\tif (!node) break; // dead end\n\t\t\tdepth++;\n\n\t\t\tif (node.terminal) {\n\t\t\t\t// Deferred fused-token creation, only on deepest match found\n\t\t\t\tconst sourceTokens = tokens.slice(pos, pos + depth);\n\t\t\t\tbest = {\n\t\t\t\t\tconsumed: depth,\n\t\t\t\t\treplacement: [\n\t\t\t\t\t\tcreateFusedToken(node.terminal.tokenType, node.terminal.phrase, sourceTokens),\n\t\t\t\t\t],\n\t\t\t\t\truleName: node.terminal.phrase,\n\t\t\t\t};\n\t\t\t}\n\t\t}\n\n\t\treturn best;\n\t}\n\n\t// ── Introspection ─────────────────────────────────────────────────────\n\n\t/** Number of unique first words (not total phrases). */\n\tget size(): number {\n\t\treturn this.root.size;\n\t}\n\n\t/**\n\t * Return all registered phrases and their target token types.\n\t *\n\t * Used by diagnostic mode to expose the complete trie structure to the\n\t * playground's NormalizerTab for rendering ALL registered phrases\n\t * (not just the ones that matched in this evaluation).\n\t */\n\tgetAllPhrases(): Record<string, string> {\n\t\tconst result: Record<string, string> = {};\n\n\t\tconst collect = (node: TrieNode, path: string[]): void => {\n\t\t\tif (node.terminal) {\n\t\t\t\tresult[path.join(' ')] = node.terminal.tokenType;\n\t\t\t}\n\t\t\tfor (const [word, child] of node.children) {\n\t\t\t\tcollect(child, [...path, word]);\n\t\t\t}\n\t\t};\n\n\t\tfor (const [firstWord, node] of this.root) {\n\t\t\tcollect(node, [firstWord]);\n\t\t}\n\n\t\treturn result;\n\t}\n\n\t/** Check if any phrase starts with this word (case-insensitive). */\n\tcanStart(word: string): boolean {\n\t\treturn this.startWords.has(word.toLowerCase());\n\t}\n}\n\n//#endregion\n","//#region ─── Module Overview ───────────────────────────────────────────────────\n\n/**\n * Built-in normalization rules for the {@link TokenNormalizer}.\n *\n * These rules handle common expression patterns that span multiple tokens.\n * Phrase fusion (e.g., \"to the power of\" → CARET) is handled by the\n * internal {@link PhraseTrie}. See {@link TokenNormalizer.addPhrase}.\n * This module only exports non-phrase rules like {@link implicitMultiplyRule}.\n *\n * @module BuiltinNormalizerRules\n */\n\n//#endregion\n//#region ─── Imports ──────────────────────────────────────────────────────────\n\nimport type { Token } from \"@solve-js/lexer/Token\";\nimport { tokenTypeId } from \"@solve-js/lexer/Token\";\nimport { LexerToken } from \"@solve-js/lexer/ExpressionLexer\";\nimport type { NormalizerRule, NormalizerMatch } from \"./NormalizerRule\";\n\n//#endregion\n//#region ─── PHRASE_START_WORDS, Implicit Multiply Guard ─────────────────────\n\n/**\n * Words that can start multi-word phrases, hardcoded fallback.\n *\n * This is ONLY used as the default fallback in {@link implicitMultiplyRule}\n * when no `canStart` predicate is provided. In the recommended pattern,\n * the {@link PhraseTrie}'s live `canStart()` set is passed instead, keeping\n * the guard in sync with package-registered phrases.\n *\n * @see {@link TokenNormalizer.canStartPhrase}\n */\n/**\n * Words that introduce a rate denominator when a unit follows.\n *\n * Kept next to the implicit-multiplication rule because that is the only\n * thing they change here; the rate itself is built in the uom package.\n */\nconst RATE_DENOMINATOR_WORDS = new Set([\"per\", \"a\", \"an\", \"each\", \"every\"]);\n\nconst PHRASE_START_WORDS = new Set([\n \"to\", \"power\", \"increase\", \"decrease\", \"times\", \"multiply\", \"divide\", \"by\",\n]);\n\n//#endregion\n//#region ─── implicitMultiplyRule, Implicit Operator Insertion ────────────────\n\n/**\n * Creates a normalization rule that inserts an implicit multiplication operator\n * between adjacent tokens where multiplication is implied.\n *\n * ## When it fires\n * Inserts a STAR token between:\n * - `NUMBER IDENT` (e.g., \"2 x\" → \"2 * x\")\n * - `RPAREN IDENT` (e.g., \"(x+1)y\" → \"(x+1) * y\")\n * - `NUMBER LPAREN` (e.g., \"2(x+1)\" → \"2 * (x+1)\")\n * - `NUMBER PI` / `NUMBER E` (e.g., \"2π\" → \"2 * π\")\n *\n * ## When it doesn't fire\n * - When the following identifier starts a multi-word phrase\n * (checked against {@link PHRASE_START_WORDS})\n * - When the following token is not an identifier or parenthesized expression\n *\n * ## Priority\n * Default priority is 50, below phrase fusion so phrases match first.\n *\n * @param priority - Rule priority (default 50)\n * @returns A {@link NormalizerRule} that inserts implicit multiply operators\n */\nexport function implicitMultiplyRule(\n\tpriority: number = 50,\n\tcanStart?: (word: string) => boolean\n): NormalizerRule {\n\tconst phraseGuard = canStart ?? ((word: string) => PHRASE_START_WORDS.has(word.toLowerCase()));\n\n\treturn {\n name: \"implicit:multiply\",\n priority,\n match(tokens: Token[], pos: number): NormalizerMatch | null {\n // ── Need at least one token after the current position ──\n if (pos + 1 >= tokens.length) return null;\n\n const t = tokens[pos];\n const next = tokens[pos + 1];\n\n // ── Guard: suppress if the next identifier starts a phrase ──\n // Uses the trie's canStart when available, falls back to hardcoded set.\n // This prevents \"2 power of 3\" from becoming \"2 * power of 3\".\n const nextValue = next.value.toLowerCase();\n if (phraseGuard(nextValue)) return null;\n\n // ── Guard: a rate denominator, not a multiplication ──\n // \"99 per week\" is ninety-nine a week, not ninety-nine times something\n // called per. The trie guard above cannot cover this, because these are\n // not registered phrases; they are recognised by the uom package's\n // bare-denominator rule, which runs after this one and then finds its\n // fused token stranded in operand position. The slash spelling never had\n // the problem, which is what made it hard to see.\n if (RATE_DENOMINATOR_WORDS.has(nextValue) && tokens[pos + 2]?.type === \"UNIT\") return null;\n\n // ── Check trigger conditions ──\n const triggers =\n (t.type === \"NUMBER\" || t.type === \"RPAREN\") &&\n (next.type === \"IDENT\" || next.type === \"LPAREN\" || next.type === \"PI\" || next.type === \"E\");\n\n if (!triggers) return null;\n\n // ── Insert a STAR token at the next token's position ──\n const starToken = new LexerToken(\n \"STAR\", tokenTypeId(\"STAR\"), \"*\", \"*\",\n next.offset, 0, next.line, next.col,\n );\n\n // consumed = 1: only the current token is replaced with [current, STAR]\n // The next token is NOT consumed, it stays for the next iteration\n return { consumed: 1, replacement: [t, starToken] };\n },\n };\n}\n\n//#endregion\n//#region ─── isInsideRangeContext, Bracket/Call-Paren Context Guard ───────────\n\n/**\n * Whether token `pos` sits inside a context where a bare `NUMBER:NUMBER`\n * sequence means a Range, not a clock-time/laptime/video-timecode literal:\n * a matrix literal/index/slice (`[1,2,3]`, `a[0:3]`), OR a `map`/`reduce`/\n * `sum`/`prod` call's own argument-list parens (`map(f, 0:3)`, these\n * accept a bare Range argument directly, per the Calca spec's own\n * example). Scans backward from `pos` over the CURRENT pass's token array,\n * tracking `[`/`(` nesting depth, the same \"positional guard via\n * backward scan\" idiom already used elsewhere in this normalizer layer\n * (e.g. `LineRefNormalizerRule`'s previous-token check), generalized to\n * depth-tracking. An `LBRACKET` is unconditionally a range-safe opener; an\n * `LPAREN` is range-safe ONLY when immediately preceded by MAP/REDUCE/\n * SUM_FN/PROD_FN, an ordinary grouping/function-call paren is NOT, so\n * `(9:00) + 5` still means a clock time, not a range.\n *\n * Needed because the clock-time/laptime/video-timecode rules each match a\n * bare `NUMBER COLON NUMBER...` shape with ZERO context-awareness, a real\n * collision discovered when adding Range support, since e.g. \"0:3\" is\n * valid input to BOTH features. These contexts have no legitimate use for\n * a clock-time/laptime/timecode literal, so the carve-out costs the time\n * features nothing.\n *\n * IMPORTANT cross-pass-timing gotcha (a real bug found and fixed here,\n * not a hypothetical): the `LPAREN`-opener check deliberately tests the\n * RAW word text (`prev.value.toLowerCase()`), not `prev.type ===\n * \"MAP\"/\"REDUCE\"/...`. The normalizer's multi-pass loop hands every rule\n * the SAME frozen `tokens` snapshot for an entire pass, a rule scanning\n * a LATER position in that pass cannot see a fusion `mapReduceCallNormalizerRule`\n * performs at an EARLIER position in that SAME pass (fusion results only\n * become visible to other rules starting the NEXT pass). Checking the\n * fused token type here would miss exactly the case that matters most\n * `map(f, 0:3)` on its very first normalization pass, silently letting\n * `0:3` fuse into a clock time before `map(`'s own fusion ever lands.\n * Testing the raw word is immune to this: it's true from the very first\n * pass, regardless of whether `mapReduceCallNormalizerRule` has run yet.\n * (`LBRACKET` above has no equivalent issue, it's a genuine lexer token\n * from the start, never itself the product of a fusion.)\n */\nexport function isInsideRangeContext(tokens: Token[], pos: number): boolean {\n const safeStack: boolean[] = [];\n for (let i = 0; i < pos; i++) {\n const t = tokens[i];\n if (t.type === \"LBRACKET\") {\n safeStack.push(true);\n } else if (t.type === \"LPAREN\") {\n const prev = tokens[i - 1];\n const opensMapReduceCall = !!prev && (\n prev.type === \"MAP\" || prev.type === \"REDUCE\" || prev.type === \"SUM_FN\" || prev.type === \"PROD_FN\" ||\n (prev.type === \"IDENT\" && (prev.value.toLowerCase() === \"map\" || prev.value.toLowerCase() === \"reduce\" || prev.value.toLowerCase() === \"sum\" || prev.value.toLowerCase() === \"prod\"))\n );\n safeStack.push(opensMapReduceCall);\n } else if (t.type === \"RBRACKET\" || t.type === \"RPAREN\") {\n safeStack.pop();\n }\n }\n return safeStack.length > 0 && safeStack[safeStack.length - 1];\n}\n\n//#endregion\n//#region ─── createBuiltinNormalizerRules, All Built-in Rules ─────────────────\n\n/**\n * Creates built-in non-phrase normalization rules.\n *\n * Returns implicit multiply insertion (priority 50).\n *\n * **Prefer** calling {@link implicitMultiplyRule} directly with a\n * `canStart` predicate wired to the normalizer's phrase trie:\n * ```ts\n * normalizer.register(implicitMultiplyRule(50, (w) => normalizer.canStartPhrase(w)));\n * ```\n * Without the predicate, this function falls back to a hardcoded\n * {@link PHRASE_START_WORDS} set that won't reflect package-registered phrases.\n *\n * @returns An array of {@link NormalizerRule} instances ready for registration\n */\nexport function createBuiltinNormalizerRules(): NormalizerRule[] {\n return [\n // ── Implicit operator insertion (priority 50) ──\n implicitMultiplyRule(),\n ];\n}\n\n/**\n * Built-in phrase → tokenType mappings.\n *\n * These are registered into the engine's {@link PhraseTrie} during\n * construction. Tests that create a standalone {@link TokenNormalizer}\n * should register these via {@link TokenNormalizer.addPhrase}.\n */\nexport const BUILTIN_PHRASES: Record<string, string> = {\n\t\"to the power of\": \"CARET\",\n\t\"power of\": \"CARET\",\n\t\"increase by\": \"INCREASE_BY\",\n\t\"decrease by\": \"DECREASE_BY\",\n\t\"times by\": \"TIMES_BY\",\n\t\"multiply by\": \"MULTIPLY_BY\",\n\t\"divide by\": \"DIVIDE_BY\",\n\t// The past-tense spellings, which is how the operation is usually written\n\t// out: \"3 multiplied by 4\". Same tokens, so no new parselets. Both this\n\t// branch and the parity branch added \"multiplied by\" independently; the\n\t// merge kept both copies and the locale table rejected the duplicate key.\n\t\"multiplied by\": \"MULTIPLY_BY\",\n\t\"divided by\": \"DIVIDE_BY\",\n};\n\n//#endregion\n"]}
|
|
1
|
+
{"version":3,"sources":["../src/normalizer/TokenNormalizer.ts","../src/normalizer/PhraseTrie.ts","../src/normalizer/BuiltinNormalizerRules.ts"],"names":["DEFAULT_OPTIONS","createFusedToken","type","text","sourceTokens","first","last","token","LexerToken","tokenTypeId","recordSourceSpan","fused","NON_WORD_NAMES","NON_WORD_TABLE","ids","len","table","id","TokenNormalizer","options","PhraseTrie","rule","ruleName","a","b","phrase","tokenType","word","tokens","onFusion","sorted","fusionHandler","maxPasses","maxTokens","current","changed","passCount","result","pos","matched","tid","trieMatch","rt","match","ErrorFactory","words","firstWord","node","i","startType","best","depth","contType","collect","path","child","RATE_DENOMINATOR_WORDS","PHRASE_START_WORDS","implicitMultiplyRule","priority","canStart","phraseGuard","next","nextValue","starToken","isInsideRangeContext","safeStack","t","prev","opensMapReduceCall","createBuiltinNormalizerRules","BUILTIN_PHRASES"],"mappings":"mHA2EA,IAAMA,CAAAA,CAA+C,CACnD,SAAA,CAAW,GAAA,CACX,UAAW,GAAA,CACX,QAAA,CAAU,IAAM,CAAC,CACnB,CAAA,CAwBO,SAASC,CAAAA,CACdC,IACAC,CAAAA,CACAC,CAAAA,CACO,CACP,IAAMC,CAAAA,CAAQD,CAAAA,CAAa,CAAC,CAAA,CACtBE,EAAOF,CAAAA,CAAaA,CAAAA,CAAa,MAAA,CAAS,CAAC,EAC3CG,CAAAA,CAAQ,IAAIC,CAAAA,CAChBN,GAAAA,CACAO,EAAYP,GAAI,CAAA,CAChBC,CAAAA,CACAA,CAAAA,CACAE,CAAAA,CAAM,MAAA,CACNA,CAAAA,CAAM,UAAA,EAAc,EACpBA,CAAAA,CAAM,IAAA,CACNA,CAAAA,CAAM,GACR,EAGA,OAAAE,CAAAA,CAAM,SAAA,CAAYD,CAAAA,CAAK,WAAaA,CAAAA,CAAK,MAAA,CAASA,CAAAA,CAAK,IAAA,CAAK,MAAA,CACrDC,CACT,CAkBA,SAASG,EAAiBC,CAAAA,CAAcP,CAAAA,CAA6B,CACnE,IAAME,EAAOF,CAAAA,CAAaA,CAAAA,CAAa,MAAA,CAAS,CAAC,EAC5CE,CAAAA,GACLK,CAAAA,CAAM,SAAA,CAAYL,CAAAA,CAAK,SAAA,EAAaA,CAAAA,CAAK,MAAA,CAASA,CAAAA,CAAK,KAAK,MAAA,EAC9D,CAiCO,IAAMM,CAAAA,CAAiB,CAC7B,QAAA,CAAU,KAAA,CAAO,QAAA,CAAU,OAAA,CAC3B,SAAU,QAAA,CAAU,SAAA,CAAW,QAAA,CAAU,SAAA,CACzC,QAAA,CAAU,QAAA,CAAU,UAAA,CAAY,UAAA,CAChC,QAAS,OAAA,CAAS,QAAA,CAAU,WAAA,CAAa,MAAA,CAAQ,YAAa,IAAA,CAC9D,WAAA,CAAa,UAAA,CAAY,aAAA,CACzB,MAAO,IAAA,CAAM,SACd,CAAA,CAQaC,CAAAA,CAAAA,CAA8B,IAAM,CAEhD,IAAMC,CAAAA,CAAMF,EAAe,GAAA,CAAI,CAAA,EAAKH,CAAAA,CAAY,CAAC,CAAC,CAAA,CAE5CM,CAAAA,CAAM,IAAA,CAAK,GAAA,CAAI,GAAGD,CAAG,CAAA,CAAI,CAAA,CACzBE,CAAAA,CAAQ,IAAI,UAAA,CAAWD,CAAG,CAAA,CAEhC,QAAWE,CAAAA,IAAMH,CAAAA,CAAKE,CAAAA,CAAMC,CAAE,EAAI,CAAA,CAClC,OAAOD,CACR,CAAA,IA6BaE,CAAAA,CAAN,KAAsB,CAgC3B,WAAA,CAAYC,CAAAA,CAA6B,EAAC,CAAG,CA9B7C,KAAQ,KAAA,CAA0B,EAAC,CAanC,IAAA,CAAQ,iBAA4C,IAAA,CAOpD,IAAA,CAAQ,UAAA,CAAa,IAAIC,EAWvB,IAAA,CAAK,OAAA,CAAU,CAAE,GAAGpB,CAAAA,CAAiB,GAAGmB,CAAQ,EAClD,CAaA,QAAA,CAASE,CAAAA,CAA4B,CACnC,IAAA,CAAK,MAAM,IAAA,CAAKA,CAAI,CAAA,CACpB,IAAA,CAAK,iBAAmB,KAC1B,CAUA,UAAA,CAAWC,CAAAA,CAAwB,CACjC,IAAA,CAAK,KAAA,CAAQ,IAAA,CAAK,MAAM,MAAA,CAAO,CAAA,EAAK,CAAA,CAAE,IAAA,GAASA,CAAQ,CAAA,CACvD,IAAA,CAAK,gBAAA,CAAmB,KAC1B,CAMA,KAAA,EAAc,CACZ,IAAA,CAAK,KAAA,CAAQ,EAAC,CACd,IAAA,CAAK,gBAAA,CAAmB,KACxB,IAAA,CAAK,UAAA,CAAa,IAAIF,EACxB,CAOQ,cAAA,EAAmC,CACzC,OAAI,IAAA,CAAK,mBAAqB,IAAA,GAC5B,IAAA,CAAK,gBAAA,CAAmB,CAAC,GAAG,IAAA,CAAK,KAAK,CAAA,CAAE,KAAK,CAACG,CAAAA,CAAGC,CAAAA,GAAMA,CAAAA,CAAE,SAAWD,CAAAA,CAAE,QAAQ,CAAA,CAAA,CAEzE,IAAA,CAAK,gBACd,CAKA,IAAI,SAAA,EAAoB,CACtB,OAAO,IAAA,CAAK,KAAA,CAAM,MACpB,CAcA,SAAA,CAAUE,CAAAA,CAAgBC,CAAAA,CAAyB,CACjD,KAAK,UAAA,CAAW,SAAA,CAAUD,CAAAA,CAAQC,CAAS,EAC7C,CAgBA,UAAA,EAAqC,CACnC,OAAO,IAAA,CAAK,UAAA,CAAW,aAAA,EACzB,CAEA,cAAA,CAAeC,CAAAA,CAAuB,CACpC,OAAO,KAAK,UAAA,CAAW,QAAA,CAASA,CAAI,CACtC,CA8BA,SAAA,CAAUC,CAAAA,CAAiBC,CAAAA,CAAmD,CAE5E,GAAID,CAAAA,CAAO,MAAA,GAAW,CAAA,CAAG,OAAOA,CAAAA,CAGhC,IAAME,CAAAA,CAAS,IAAA,CAAK,gBAAe,CAC7BC,CAAAA,CAAgBF,CAAAA,EAAY,IAAA,CAAK,QAAQ,QAAA,CACzCG,CAAAA,CAAY,IAAA,CAAK,OAAA,CAAQ,SAAA,CACzBC,CAAAA,CAAY,IAAA,CAAK,OAAA,CAAQ,UAE3BC,CAAAA,CAAUN,CAAAA,CACVO,CAAAA,CAAU,IAAA,CACVC,CAAAA,CAAY,CAAA,CAGhB,KAAOD,CAAAA,EAAWC,EAAYJ,CAAAA,EAAW,CACvCG,CAAAA,CAAU,KAAA,CACVC,CAAAA,EAAAA,CAEA,IAAMC,CAAAA,CAAkB,GACpBC,CAAAA,CAAM,CAAA,CAGV,KAAOA,CAAAA,CAAMJ,EAAQ,MAAA,EAAQ,CAC3B,IAAIK,CAAAA,CAAU,MAKRC,CAAAA,CAAMN,CAAAA,CAAQI,CAAG,CAAA,CAAE,MAAA,CACzB,GAAIE,CAAAA,EAAO3B,CAAAA,CAAe,QAAUA,CAAAA,CAAe2B,CAAG,CAAA,GAAM,CAAA,CAAG,CAC7D,IAAMC,CAAAA,CAAY,IAAA,CAAK,UAAA,CAAW,QAAQP,CAAAA,CAASI,CAAG,CAAA,CACtD,GAAIG,CAAAA,CAAW,CACb,IAAMrC,CAAAA,CAAe8B,EAAQ,KAAA,CAAMI,CAAAA,CAAKA,CAAAA,CAAMG,CAAAA,CAAU,QAAQ,CAAA,CAChE,IAAA,IAAWC,CAAAA,IAAMD,CAAAA,CAAU,YACzBJ,CAAAA,CAAO,IAAA,CAAKK,CAAE,CAAA,CAEZD,CAAAA,CAAU,QAAA,CAAW,CAAA,EAAKA,CAAAA,CAAU,YAAY,MAAA,GAAW,CAAA,GAC7D/B,CAAAA,CAAiB+B,CAAAA,CAAU,YAAY,CAAC,CAAA,CAAGrC,CAAY,CAAA,CACvD2B,EAAc,CACZ,IAAA,CAAMU,CAAAA,CAAU,QAAA,EAAY,aAAA,CAC5B,YAAA,CAAArC,CAAAA,CACA,UAAA,CAAYqC,EAAU,WAAA,CAAY,CAAC,CACrC,CAAC,GAEHH,CAAAA,EAAOG,CAAAA,CAAU,QAAA,CACjBN,CAAAA,CAAU,KACV,QACF,CACF,CAGA,IAAA,IAAWd,CAAAA,IAAQS,CAAAA,CAAQ,CACzB,IAAMa,EAAQtB,CAAAA,CAAK,KAAA,CAAMa,CAAAA,CAASI,CAAG,EACrC,GAAIK,CAAAA,CAAO,CAET,IAAMvC,EAAe8B,CAAAA,CAAQ,KAAA,CAAMI,CAAAA,CAAKA,CAAAA,CAAMK,CAAAA,CAAM,QAAQ,CAAA,CAG5D,IAAA,IAAWD,KAAMC,CAAAA,CAAM,WAAA,CACrBN,CAAAA,CAAO,IAAA,CAAKK,CAAE,CAAA,CAMhB,GAAIC,CAAAA,CAAM,QAAA,CAAW,GAAKA,CAAAA,CAAM,WAAA,CAAY,MAAA,GAAW,CAAA,CACrDjC,CAAAA,CAAiBiC,CAAAA,CAAM,WAAA,CAAY,CAAC,EAAGvC,CAAY,CAAA,CACnD2B,CAAAA,CAAc,CACZ,KAAMV,CAAAA,CAAK,IAAA,CACX,YAAA,CAAAjB,CAAAA,CACA,WAAYuC,CAAAA,CAAM,WAAA,CAAY,CAAC,CACjC,CAAC,CAAA,CAAA,KAAA,GACQA,CAAAA,CAAM,QAAA,CAAW,GAAKA,CAAAA,CAAM,WAAA,CAAY,MAAA,CAASA,CAAAA,CAAM,SAChE,IAAA,IAAWD,CAAAA,IAAMC,CAAAA,CAAM,WAAA,CACrBZ,EAAc,CACZ,IAAA,CAAMV,CAAAA,CAAK,IAAA,CACX,YAAA,CAAAjB,CAAAA,CACA,UAAA,CAAYsC,CACd,CAAC,CAAA,CAKLJ,CAAAA,EAAOK,CAAAA,CAAM,QAAA,CACbR,EAAU,IAAA,CACVI,CAAAA,CAAU,IAAA,CACV,KACF,CACF,CAEKA,CAAAA,GAEHF,CAAAA,CAAO,IAAA,CAAKH,CAAAA,CAAQI,CAAG,CAAC,CAAA,CACxBA,KAEJ,CAGA,GAAID,CAAAA,CAAO,MAAA,CAASJ,EAKlB,MAAMW,GAAAA,CAAa,UAAA,CACjB,iCAAA,CACA,2BAA2BP,CAAAA,CAAO,MAAM,CAAA,wBAAA,EAA2BJ,CAAS,CAAA,CAAA,CAAA,CAC5E,CAAE,UAAA,CAAYI,CAAAA,CAAO,OAAQ,SAAA,CAAAJ,CAAU,CACzC,CAAA,CAGFC,EAAUG,EACZ,CAEA,OAAOH,CACT,CACF,ECjaO,IAAMd,CAAAA,CAAN,KAAiB,CAAjB,WAAA,EAAA,CAMN,IAAA,CAAQ,UAAA,CAAa,IAAI,GAAA,CAMzB,IAAA,CAAQ,IAAA,CAAO,IAAI,KAUnB,SAAA,CAAUK,CAAAA,CAAgBC,CAAAA,CAAyB,CAClD,IAAMmB,CAAAA,CAAQpB,CAAAA,CAAO,WAAA,EAAY,CAAE,IAAA,EAAK,CAAE,KAAA,CAAM,KAAK,EACrD,GAAIoB,CAAAA,CAAM,MAAA,GAAW,CAAA,EAAKA,EAAM,CAAC,CAAA,GAAM,EAAA,CAAI,OAE3C,KAAK,UAAA,CAAW,GAAA,CAAIA,CAAAA,CAAM,CAAC,CAAC,CAAA,CAE5B,IAAMC,CAAAA,CAAYD,EAAM,CAAC,CAAA,CACrBE,CAAAA,CAUJ,GARI,KAAK,IAAA,CAAK,GAAA,CAAID,CAAS,CAAA,CAC1BC,EAAO,IAAA,CAAK,IAAA,CAAK,GAAA,CAAID,CAAS,CAAA,EAE9BC,CAAAA,CAAO,CAAE,QAAA,CAAU,IAAI,GAAA,CAAO,QAAA,CAAU,IAAK,CAAA,CAC7C,IAAA,CAAK,IAAA,CAAK,GAAA,CAAID,CAAAA,CAAWC,CAAI,CAAA,CAAA,CAI1BF,CAAAA,CAAM,MAAA,GAAW,CAAA,CAAG,CACvBE,CAAAA,CAAK,QAAA,CAAW,CAAE,UAAArB,CAAAA,CAAW,MAAA,CAAAD,CAAAA,CAAQ,QAAA,CAAU,CAAE,CAAA,CACjD,MACD,CAGA,IAAA,IAASuB,EAAI,CAAA,CAAGA,CAAAA,CAAIH,CAAAA,CAAM,MAAA,CAAQG,CAAAA,EAAAA,CAAK,CACtC,IAAMrB,CAAAA,CAAOkB,EAAMG,CAAC,CAAA,CACfD,CAAAA,CAAK,QAAA,CAAS,IAAIpB,CAAI,CAAA,EAC1BoB,CAAAA,CAAK,QAAA,CAAS,IAAIpB,CAAAA,CAAM,CAAE,QAAA,CAAU,IAAI,GAAA,CAAO,QAAA,CAAU,IAAK,CAAC,EAEhEoB,CAAAA,CAAOA,CAAAA,CAAK,QAAA,CAAS,GAAA,CAAIpB,CAAI,EAC9B,CAEAoB,CAAAA,CAAK,QAAA,CAAW,CAAE,SAAA,CAAArB,CAAAA,CAAW,MAAA,CAAAD,CAAAA,CAAQ,QAAA,CAAUoB,CAAAA,CAAM,MAAO,EAC7D,CAeA,OAAA,CAAQjB,CAAAA,CAAiBU,CAAAA,CAAqC,CAE7D,GAAIA,CAAAA,EAAOV,CAAAA,CAAO,MAAA,CAAQ,OAAO,KAejC,IAAMqB,CAAAA,CAAYrB,CAAAA,CAAOU,CAAG,CAAA,CAAE,IAAA,CAC9B,GAAIW,CAAAA,GAAc,OAASA,CAAAA,CAAU,UAAA,CAAW,MAAM,CAAA,CAAG,OAAO,IAAA,CAGhE,IAAMH,CAAAA,CAAYlB,CAAAA,CAAOU,CAAG,CAAA,CAAE,KAAA,CAAM,WAAA,EAAY,CAChD,GAAI,CAAC,IAAA,CAAK,UAAA,CAAW,IAAIQ,CAAS,CAAA,CAAG,OAAO,IAAA,CAG5C,IAAIC,CAAAA,CAA6B,IAAA,CAAK,IAAA,CAAK,GAAA,CAAID,CAAS,CAAA,CACxD,GAAI,CAACC,CAAAA,CAAM,OAAO,IAAA,CAElB,IAAIG,CAAAA,CAA+B,KAC/BC,CAAAA,CAAQ,CAAA,CAGZ,GAAIJ,CAAAA,CAAK,SAAU,CAClB,IAAM3C,CAAAA,CAAewB,CAAAA,CAAO,MAAMU,CAAAA,CAAKA,CAAAA,CAAM,CAAC,CAAA,CAC9CY,CAAAA,CAAO,CACN,QAAA,CAAU,CAAA,CACV,YAAa,CAACjD,CAAAA,CAAiB8C,CAAAA,CAAK,QAAA,CAAS,UAAWA,CAAAA,CAAK,QAAA,CAAS,MAAA,CAAQ3C,CAAY,CAAC,CAAA,CAC3F,QAAA,CAAU2C,CAAAA,CAAK,QAAA,CAAS,MACzB,EACD,CAGA,IAAA,IAASC,EAAIV,CAAAA,CAAM,CAAA,CAAGU,CAAAA,CAAIpB,CAAAA,CAAO,QAAUmB,CAAAA,EAAM,QAAA,CAAUC,CAAAA,EAAAA,CAAK,CAC/D,IAAMI,CAAAA,CAAWxB,CAAAA,CAAOoB,CAAC,CAAA,CAAE,IAAA,CAC3B,GAAII,CAAAA,GAAa,KAAA,EAASA,EAAS,UAAA,CAAW,MAAM,CAAA,CAAG,MACvD,IAAMzB,CAAAA,CAAOC,CAAAA,CAAOoB,CAAC,CAAA,CAAE,MAAM,WAAA,EAAY,CAEzC,GADAD,CAAAA,CAAOA,CAAAA,CAAK,QAAA,CAAS,GAAA,CAAIpB,CAAI,EACzB,CAACoB,CAAAA,CAAM,MAGX,GAFAI,IAEIJ,CAAAA,CAAK,QAAA,CAAU,CAElB,IAAM3C,EAAewB,CAAAA,CAAO,KAAA,CAAMU,CAAAA,CAAKA,CAAAA,CAAMa,CAAK,CAAA,CAClDD,CAAAA,CAAO,CACN,SAAUC,CAAAA,CACV,WAAA,CAAa,CACZlD,CAAAA,CAAiB8C,EAAK,QAAA,CAAS,SAAA,CAAWA,CAAAA,CAAK,QAAA,CAAS,OAAQ3C,CAAY,CAC7E,CAAA,CACA,QAAA,CAAU2C,CAAAA,CAAK,QAAA,CAAS,MACzB,EACD,CACD,CAEA,OAAOG,CACR,CAKA,IAAI,IAAA,EAAe,CAClB,OAAO,IAAA,CAAK,KAAK,IAClB,CASA,aAAA,EAAwC,CACvC,IAAMb,CAAAA,CAAiC,EAAC,CAElCgB,EAAU,CAACN,CAAAA,CAAgBO,CAAAA,GAAyB,CACrDP,EAAK,QAAA,GACRV,CAAAA,CAAOiB,CAAAA,CAAK,IAAA,CAAK,GAAG,CAAC,CAAA,CAAIP,CAAAA,CAAK,QAAA,CAAS,SAAA,CAAA,CAExC,IAAA,GAAW,CAACpB,CAAAA,CAAM4B,CAAK,CAAA,GAAKR,CAAAA,CAAK,QAAA,CAChCM,CAAAA,CAAQE,EAAO,CAAC,GAAGD,CAAAA,CAAM3B,CAAI,CAAC,EAEhC,CAAA,CAEA,IAAA,GAAW,CAACmB,CAAAA,CAAWC,CAAI,CAAA,GAAK,IAAA,CAAK,KACpCM,CAAAA,CAAQN,CAAAA,CAAM,CAACD,CAAS,CAAC,CAAA,CAG1B,OAAOT,CACR,CAGA,QAAA,CAASV,CAAAA,CAAuB,CAC/B,OAAO,IAAA,CAAK,UAAA,CAAW,GAAA,CAAIA,CAAAA,CAAK,aAAa,CAC9C,CACD,EC/NA,IAAM6B,CAAAA,CAAyB,IAAI,GAAA,CAAI,CAAC,MAAO,GAAA,CAAK,IAAA,CAAM,MAAA,CAAQ,OAAO,CAAC,CAAA,CAEpEC,CAAAA,CAAqB,IAAI,IAAI,CACjC,IAAA,CAAM,OAAA,CAAS,UAAA,CAAY,WAAY,OAAA,CAAS,UAAA,CAAY,QAAA,CAAU,IACxE,CAAC,CAAA,CA2BM,SAASC,CAAAA,CACfC,GAAAA,CAAmB,EAAA,CACnBC,CAAAA,CACiB,CACjB,IAAMC,EAAcD,CAAAA,GAAcjC,CAAAA,EAAiB8B,CAAAA,CAAmB,GAAA,CAAI9B,EAAK,WAAA,EAAa,CAAA,CAAA,CAE5F,OAAO,CACJ,IAAA,CAAM,mBAAA,CACN,QAAA,CAAAgC,GAAAA,CACA,KAAA,CAAM/B,CAAAA,CAAiBU,CAAAA,CAAqC,CAE1D,GAAIA,CAAAA,CAAM,CAAA,EAAKV,CAAAA,CAAO,MAAA,CAAQ,OAAO,IAAA,CAErC,IAAM,CAAA,CAAIA,CAAAA,CAAOU,CAAG,CAAA,CACdwB,CAAAA,CAAOlC,CAAAA,CAAOU,CAAAA,CAAM,CAAC,CAAA,CAKrByB,CAAAA,CAAYD,CAAAA,CAAK,MAAM,WAAA,EAAY,CAiBzC,GAhBID,CAAAA,CAAYE,CAAS,CAAA,EASrBP,CAAAA,CAAuB,GAAA,CAAIO,CAAS,GAAKnC,CAAAA,CAAOU,CAAAA,CAAM,CAAC,CAAA,EAAG,IAAA,GAAS,MAAA,EAOnE,EAAA,CAHD,CAAA,CAAE,OAAS,QAAA,EAAY,CAAA,CAAE,IAAA,GAAS,QAAA,IAClCwB,EAAK,IAAA,GAAS,OAAA,EAAWA,CAAAA,CAAK,IAAA,GAAS,UAAYA,CAAAA,CAAK,IAAA,GAAS,IAAA,EAAQA,CAAAA,CAAK,IAAA,GAAS,GAAA,CAAA,CAAA,CAE3E,OAAO,IAAA,CAGtB,IAAME,CAAAA,CAAY,IAAIxD,CAAAA,CACpB,MAAA,CAAQC,EAAY,MAAM,CAAA,CAAG,GAAA,CAAK,GAAA,CAClCqD,EAAK,MAAA,CAAQ,CAAA,CAAGA,CAAAA,CAAK,IAAA,CAAMA,CAAAA,CAAK,GAClC,CAAA,CAIA,OAAO,CAAE,QAAA,CAAU,CAAA,CAAG,WAAA,CAAa,CAAC,EAAGE,CAAS,CAAE,CACpD,CACF,CACF,CA2CO,SAASC,CAAAA,CAAqBrC,CAAAA,CAAiBU,CAAAA,CAAsB,CAC1E,IAAM4B,CAAAA,CAAuB,EAAC,CAC9B,IAAA,IAASlB,CAAAA,CAAI,CAAA,CAAGA,EAAIV,CAAAA,CAAKU,CAAAA,EAAAA,CAAK,CAC5B,IAAMmB,EAAIvC,CAAAA,CAAOoB,CAAC,CAAA,CAClB,GAAImB,CAAAA,CAAE,IAAA,GAAS,UAAA,CACbD,CAAAA,CAAU,KAAK,IAAI,CAAA,CAAA,KAAA,GACVC,CAAAA,CAAE,IAAA,GAAS,SAAU,CAC9B,IAAMC,CAAAA,CAAOxC,CAAAA,CAAOoB,EAAI,CAAC,CAAA,CACnBqB,CAAAA,CAAqB,CAAC,CAACD,CAAAA,GAC3BA,CAAAA,CAAK,IAAA,GAAS,OAASA,CAAAA,CAAK,IAAA,GAAS,QAAA,EAAYA,CAAAA,CAAK,OAAS,QAAA,EAAYA,CAAAA,CAAK,IAAA,GAAS,SAAA,EACxFA,EAAK,IAAA,GAAS,OAAA,GAAYA,CAAAA,CAAK,KAAA,CAAM,WAAA,EAAY,GAAM,KAAA,EAASA,CAAAA,CAAK,MAAM,WAAA,EAAY,GAAM,QAAA,EAAYA,CAAAA,CAAK,MAAM,WAAA,EAAY,GAAM,KAAA,EAASA,CAAAA,CAAK,MAAM,WAAA,EAAY,GAAM,MAAA,CAAA,CAAA,CAE/KF,CAAAA,CAAU,IAAA,CAAKG,CAAkB,EACnC,CAAA,KAAA,CAAWF,EAAE,IAAA,GAAS,UAAA,EAAcA,CAAAA,CAAE,IAAA,GAAS,WAC7CD,CAAAA,CAAU,GAAA,GAEd,CACA,OAAOA,CAAAA,CAAU,MAAA,CAAS,CAAA,EAAKA,CAAAA,CAAUA,CAAAA,CAAU,MAAA,CAAS,CAAC,CAC/D,CAoBO,SAASI,CAAAA,EAAiD,CAC/D,OAAO,CAELZ,CAAAA,EACF,CACF,KASaa,CAAAA,CAA0C,CACtD,iBAAA,CAAmB,OAAA,CACnB,UAAA,CAAY,OAAA,CACZ,aAAA,CAAe,aAAA,CACf,cAAe,aAAA,CACf,UAAA,CAAY,UAAA,CACZ,aAAA,CAAe,cACf,WAAA,CAAa,WAAA,CAKb,eAAA,CAAiB,aAAA,CACjB,aAAc,WACf","file":"chunk-3GAWAMT4.js","sourcesContent":["//#region ─── Module Overview ───────────────────────────────────────────────────\n\n/**\n * TokenNormalizer, post-lexer token normalization pass.\n *\n * ## Purpose\n * Applies domain-specific {@link NormalizerRule | NormalizerRules} to the raw\n * token stream produced by the {@link ExpressionLexer}. This keeps the lexer\n * slim and focused on single-token production, while multi-token pattern\n * matching (phrases, implicit operators, domain merges) lives here.\n *\n * ## What rules can do\n * - **Phrase fusion**: Merge consecutive words into compound tokens\n * (e.g., `IDENT + ... + IDENT` → `CARET`)\n * - **Implicit operator insertion**: Insert missing operators between tokens\n * (e.g., `NUMBER IDENT` → `NUMBER STAR IDENT`)\n * - **Domain-specific transformations**: Coalesce item names, currency pairs,\n * percentage syntax, etc.\n *\n * ## Architecture\n * Providers register NormalizerRules alongside Parselets and OpCode handlers\n * via {@link IEnginePackage.normalizerRules}. The normalizer applies them\n * greedily left-to-right in multiple passes with safety limits.\n *\n * @module TokenNormalizer\n */\n\n//#endregion\n//#region ─── Imports ──────────────────────────────────────────────────────────\n\nimport type { Token } from \"@solve-js/lexer/Token\";\nimport { tokenTypeId } from \"@solve-js/lexer/Token\";\nimport { LexerToken } from \"@solve-js/lexer/ExpressionLexer\";\nimport type { NormalizerRule, TokenFusion } from \"./NormalizerRule\";\nimport { PhraseTrie } from \"./PhraseTrie\";\nimport { ErrorFactory } from \"@solve-js/errors/UnifiedErrorFramework\";\n\n//#endregion\n//#region ─── NormalizerOptions, Configuration ────────────────────────────────\n\n/**\n * Configuration options for the normalization pass.\n *\n * These control safety limits and diagnostic callbacks. The defaults\n * are chosen to be generous enough for any realistic expression while\n * preventing runaway token expansion from recursive rules.\n */\nexport interface NormalizerOptions {\n /**\n * Maximum number of full passes over the token stream before bailing out.\n * Prevents infinite loops from recursive rule chains.\n * @default 100\n */\n maxPasses?: number;\n\n /**\n * Maximum number of tokens allowed after normalization.\n * If exceeded, an Error is thrown rather than passing a bloated stream\n * to the parser.\n * @default 10000\n */\n maxTokens?: number;\n\n /**\n * Callback invoked for each fusion event during normalization.\n * Used by diagnostic mode to populate {@link NormalizerOutput.fusions}.\n * When `undefined`, fusions are still tracked internally but no callbacks fire.\n */\n onFusion?: (fusion: TokenFusion) => void;\n}\n\n//#endregion\n//#region ─── Default Options ──────────────────────────────────────────────────\n\n/** Sensible defaults that catch infinite loops without limiting real expressions. */\nconst DEFAULT_OPTIONS: Required<NormalizerOptions> = {\n maxPasses: 100,\n maxTokens: 10000,\n onFusion: () => {},\n};\n\n//#endregion\n//#region ─── createFusedToken, Token Factory ──────────────────────────────────\n\n/**\n * Creates a new normalized token from fused source tokens.\n *\n * The fused token inherits position information (offset, line, column)\n * from the first source token, which preserves source-map accuracy\n * for error messages and diagnostic highlighting.\n *\n * It also records where the source text ENDS, on `sourceEnd`. The start alone\n * is not enough to describe the span a fusion covers, because `text` is the\n * replacement rather than the original: `10 frames` fuses into a FRAME_COUNT\n * whose text is `10`, and a timecode fuses into a token whose text is a\n * comma-separated tuple that appears nowhere in the line. Anything painting the\n * line needs both ends, and only this function is in a position to know them.\n *\n * @param type - The new token type (e.g., \"CARET\", \"TIMES_BY\")\n * @param text - The combined text representation (e.g., \"to the power of\")\n * @param sourceTokens - The original tokens being fused (at least 2)\n * @returns A new {@link LexerToken} with the fused type and combined text\n */\nexport function createFusedToken(\n type: string,\n text: string,\n sourceTokens: Token[]\n): Token {\n const first = sourceTokens[0];\n const last = sourceTokens[sourceTokens.length - 1];\n const token = new LexerToken(\n type,\n tokenTypeId(type),\n text,\n text,\n first.offset,\n first.lineBreaks ?? 0,\n first.line,\n first.col,\n );\n // `text`, not `value`: this is where the SOURCE ended, and the two differ\n // for a string literal, whose value is the payload without its quotes.\n token.sourceEnd = last.sourceEnd ?? last.offset + last.text.length;\n return token;\n}\n\n/**\n * Record, on a token that replaced several, where its source text ended.\n *\n * Called centrally rather than left to each rule. A rule that builds its\n * replacement by hand rather than through {@link createFusedToken} is doing\n * nothing wrong, and several do: the date-literal rule needs a token whose\n * value is an epoch and whose text is the source, which that factory cannot\n * express. Stamping here means every fusion carries its span, including ones\n * written after this was.\n *\n * Reads `sourceEnd` off the last source token when it has one, so a fusion of\n * a fusion still describes the original text rather than the intermediate.\n *\n * @param fused - The single token the rule produced.\n * @param sourceTokens - The tokens it consumed.\n */\nfunction recordSourceSpan(fused: Token, sourceTokens: Token[]): void {\n const last = sourceTokens[sourceTokens.length - 1];\n if (!last) return;\n fused.sourceEnd = last.sourceEnd ?? last.offset + last.text.length;\n}\n\n//#endregion\n//#region ─── NON_WORD_TOKEN_TYPES, Type-guard skip set ────────────────────────\n\n/**\n * Token types that can NEVER start a multi-word phrase.\n *\n * Used by {@link normalize} to skip the PhraseTrie walk entirely at\n * positions where the token type makes phrase matching impossible.\n * This avoids even the O(1) {@link PhraseTrie.canStart} check.\n *\n * Types NOT in this set (IDENT, KEYWORD, FUNC, UNIT, and any custom\n * types registered by packages) still pass through to the trie for\n * a full match attempt.\n */\n// ── Non-word type ID lookup table (flat Uint8Array, true O(1) array index) ──\n//\n// Index = token typeId, value = 1 if non-word (skip trie), 0 otherwise.\n// Array indexing avoids ALL hashing: no Set.has(), no Map.get(), no string ops.\n//\n// Custom types from packages get IDs beyond the table length, so the bounds\n// check `tid < TABLE.length` safely passes them through to the trie.\n//\n// Arithmetic operators (PLUS, MINUS, STAR, SLASH, CARET, MOD, PERCENT) are\n// intentionally excluded: in keyword locales the lexer maps \"times\"→STAR,\n// \"divide\"→SLASH etc., and those tokens CAN start phrases like \"times by\".\n/**\n * Token type names that can NEVER start a multi-word phrase.\n *\n * Exported for testing only, consumers should use the type-guard behavior\n * of {@link TokenNormalizer.normalize} rather than this list directly.\n */\nexport const NON_WORD_NAMES = [\n\t\"NUMBER\", \"HEX\", \"BIGINT\", \"FLOAT\",\n\t\"LSHIFT\", \"RSHIFT\", \"BIT_AND\", \"BIT_OR\", \"BIT_XOR\",\n\t\"LPAREN\", \"RPAREN\", \"LBRACKET\", \"RBRACKET\",\n\t\"COMMA\", \"COLON\", \"EQUALS\", \"THEREFORE\", \"PIPE\", \"AMPERSAND\", \"AT\",\n\t\"SEMICOLON\", \"QUESTION\", \"EXCLAMATION\",\n\t\"EOF\", \"WS\", \"NEWLINE\",\n] as const;\n\n/**\n * Flat Uint8Array lookup table: index = token typeId, value = 1 if non-word.\n *\n * Exported for testing only, consumers should not depend on the internal\n * table layout, as the set of non-word types may change.\n */\nexport const NON_WORD_TABLE: Uint8Array = (() => {\n\t// Resolve all non-word type names to their numeric IDs\n\tconst ids = NON_WORD_NAMES.map(n => tokenTypeId(n));\n\t// Size the table to cover the largest ID + 1\n\tconst len = Math.max(...ids) + 1;\n\tconst table = new Uint8Array(len);\n\t// Mark non-word type positions\n\tfor (const id of ids) table[id] = 1;\n\treturn table;\n})();\n\n//#endregion\n//#region ─── TokenNormalizer Class ─────────────────────────────────────────────\n\n/**\n * Token normalizer: applies {@link NormalizerRule | NormalizerRules} to a token stream.\n *\n * ## Lifecycle\n * 1. **Registration**: Rules are added via {@link register} and sorted by priority\n * 2. **Normalization**: {@link normalize} applies rules greedily left-to-right\n * 3. **Cleanup**: {@link clear} or {@link unregister} removes rules\n *\n * ## Normalization algorithm\n * The normalizer uses a greedy left-to-right multi-pass algorithm:\n * - At each token position, rules are tried in priority order (highest first)\n * - When a rule matches, matched tokens are consumed and replaced\n * - Processing continues from the replacement position\n * - Multiple passes handle cascading matches (one rule's output triggers another)\n * - Safety limits ({@link NormalizerOptions.maxPasses}) prevent infinite loops\n *\n * @example\n * ```ts\n * const normalizer = new TokenNormalizer();\n * normalizer.register(phraseRule); // \"to the power of\" → CARET\n * normalizer.register(implicitMultRule); // \"2 x\" → \"2 * x\"\n * const normalized = normalizer.normalize(rawTokens);\n * ```\n */\nexport class TokenNormalizer {\n /** Registered rules, unsorted, the source of truth. */\n private rules: NormalizerRule[] = [];\n\n /**\n * Priority-sorted copy of {@link rules}, rebuilt lazily on the next\n * {@link normalize} call after a mutation. Rules are registered once at\n * engine/package-registration time and essentially never change during a\n * session, but normalize() runs on every keystroke-driven evaluation, an\n * earlier version re-sorted a fresh copy of `rules` on every single call,\n * which meant every keystroke paid for an allocation + sort of a list that\n * had usually not changed since the last one. `null` means \"stale, rebuild\n * on next use\"; {@link register}/{@link unregister}/{@link clear} all\n * invalidate it.\n */\n private sortedRulesCache: NormalizerRule[] | null = null;\n\n /**\n * Phrase trie for single-pass multi-word phrase fusion.\n * Tried at each token position BEFORE other rules, the trie walk\n * is O(depth) vs O(R × W) for separate rule matching.\n */\n private phraseTrie = new PhraseTrie();\n\n /** Merged options with defaults applied. */\n private options: Required<NormalizerOptions>;\n\n // ── Constructor ──────────────────────────────────────────────────────────\n\n /**\n * @param options - Configuration overrides for safety limits and diagnostic callbacks\n */\n constructor(options: NormalizerOptions = {}) {\n this.options = { ...DEFAULT_OPTIONS, ...options };\n }\n\n // ── Rule Management ──────────────────────────────────────────────────────\n\n /**\n * Register a normalization rule.\n *\n * Rules are sorted by priority (descending) on each {@link normalize} call.\n * Multiple rules can share the same priority, they are tried in registration\n * order when priorities are equal.\n *\n * @param rule - The rule to register\n */\n register(rule: NormalizerRule): void {\n this.rules.push(rule);\n this.sortedRulesCache = null;\n }\n\n /**\n * Unregister a normalization rule by its {@link NormalizerRule.name | name}.\n *\n * If multiple rules share the same name, all are removed. This is safe to\n * call with a name that doesn't match any rule, it simply has no effect.\n *\n * @param ruleName - The name of the rule to remove\n */\n unregister(ruleName: string): void {\n this.rules = this.rules.filter(r => r.name !== ruleName);\n this.sortedRulesCache = null;\n }\n\n /**\n * Remove all registered rules, resetting the normalizer to its initial state.\n * Also clears the phrase trie.\n */\n clear(): void {\n this.rules = [];\n this.sortedRulesCache = null;\n this.phraseTrie = new PhraseTrie();\n }\n\n /**\n * Priority-sorted view of {@link rules} (descending priority; registration\n * order preserved for ties, since {@link Array.prototype.sort} is stable).\n * Cached until the next mutation. See {@link sortedRulesCache}.\n */\n private getSortedRules(): NormalizerRule[] {\n if (this.sortedRulesCache === null) {\n this.sortedRulesCache = [...this.rules].sort((a, b) => b.priority - a.priority);\n }\n return this.sortedRulesCache;\n }\n\n /**\n * Get the number of currently registered rules (excludes phrase trie entries).\n */\n get ruleCount(): number {\n return this.rules.length;\n }\n\n // ── Phrase Registration ────────────────────────────────────────────────\n\n /**\n * Register a multi-word phrase for fusion into a single compound token.\n *\n * This is the preferred way to add phrase patterns. It inserts into the\n * internal {@link PhraseTrie}, which collapses all phrase rules into a\n * single O(depth) trie walk per position, no separate rule scanning.\n *\n * @param phrase - Multi-word phrase (e.g., \"to the power of\", \"abyssal whip\")\n * @param tokenType - Target token type after fusion (e.g., \"CARET\", \"ITEM\")\n */\n addPhrase(phrase: string, tokenType: string): void {\n this.phraseTrie.addPhrase(phrase, tokenType);\n }\n\n /**\n * Check whether a word can start any registered phrase.\n *\n * Used by {@link implicitMultiplyRule} to suppress `*` insertion\n * before phrase-starting identifiers (e.g., \"2 power of 3\" → `2 ^ 3`,\n * not `2 * power of 3`). Delegates to {@link PhraseTrie.canStart}.\n */\n /**\n * Get all registered phrases and their target token types.\n *\n * Exposes the full phrase trie structure for diagnostic rendering\n * in the playground's NormalizerTab. Returns ALL registered phrases,\n * not just the ones that matched in the last evaluation.\n */\n getPhrases(): Record<string, string> {\n return this.phraseTrie.getAllPhrases();\n }\n\n canStartPhrase(word: string): boolean {\n return this.phraseTrie.canStart(word);\n }\n\n // ── Normalization ────────────────────────────────────────────────────────\n\n /**\n * Normalize a token stream by applying all registered rules.\n *\n * ## Algorithm\n * Applies rules greedily left-to-right in multiple passes:\n * 1. Sort rules by priority (descending)\n * 2. Walk the token stream left to right\n * 3. At each position, try rules in priority order\n * 4. On match: consume matched tokens, insert replacements, restart from insert point\n * 5. On no match: pass token through unchanged\n * 6. Repeat until a full pass produces no changes, or maxPasses is reached\n *\n * ## Fusion tracking\n * When a rule consumes more tokens than it produces, the normalizer calls\n * `onFusion` with a {@link TokenFusion} record for diagnostic collection.\n * This populates {@link NormalizerOutput.fusions} in the playground pipeline view.\n *\n * ## Safety\n * If the normalized token count exceeds {@link NormalizerOptions.maxTokens},\n * an Error is thrown to prevent memory exhaustion from runaway rule expansion.\n *\n * @param tokens - Raw tokens from the lexer\n * @param onFusion - Optional fusion callback (overrides {@link NormalizerOptions.onFusion})\n * @returns Normalized tokens ready for parsing\n * @throws {Error} If the normalized token count exceeds maxTokens\n */\n normalize(tokens: Token[], onFusion?: (fusion: TokenFusion) => void): Token[] {\n // ── Early exit: nothing to normalize ──\n if (tokens.length === 0) return tokens;\n\n // ── Priority-sorted rules, cached across calls, see getSortedRules() ──\n const sorted = this.getSortedRules();\n const fusionHandler = onFusion ?? this.options.onFusion;\n const maxPasses = this.options.maxPasses;\n const maxTokens = this.options.maxTokens;\n\n let current = tokens;\n let changed = true;\n let passCount = 0;\n\n // Multi-pass loop: rules may trigger cascading matches across passes\n while (changed && passCount < maxPasses) {\n changed = false;\n passCount++;\n\n const result: Token[] = [];\n let pos = 0;\n\n // Single-pass left-to-right greedy walk\n while (pos < current.length) {\n let matched = false;\n\n // ── Fast path: phrase trie (O(depth) single walk vs O(R × W) per rule) ──\n // O(1) type-guard: skip trie entirely for tokens that can't start phrases.\n // Flat Uint8Array indexed by typeId, no hashing, no Set lookup, true O(1).\n const tid = current[pos].typeId;\n if (tid >= NON_WORD_TABLE.length || NON_WORD_TABLE[tid] === 0) {\n const trieMatch = this.phraseTrie.matchAt(current, pos);\n if (trieMatch) {\n const sourceTokens = current.slice(pos, pos + trieMatch.consumed);\n for (const rt of trieMatch.replacement) {\n result.push(rt);\n }\n if (trieMatch.consumed > 1 && trieMatch.replacement.length === 1) {\n recordSourceSpan(trieMatch.replacement[0], sourceTokens);\n fusionHandler({\n rule: trieMatch.ruleName ?? \"phrase-trie\",\n sourceTokens,\n fusedToken: trieMatch.replacement[0],\n });\n }\n pos += trieMatch.consumed;\n changed = true;\n continue; // trie matched — skip other rules at this position (no need to set matched)\n }\n }\n\n // Try every rule in priority order at this position\n for (const rule of sorted) {\n const match = rule.match(current, pos);\n if (match) {\n // Collect source tokens for fusion tracking\n const sourceTokens = current.slice(pos, pos + match.consumed);\n\n // Insert replacement tokens into result\n for (const rt of match.replacement) {\n result.push(rt);\n }\n\n // Track fusion events for diagnostics:\n // - Multiple tokens → single token: classic fusion\n // - Multiple tokens → fewer tokens: partial fusion\n if (match.consumed > 1 && match.replacement.length === 1) {\n recordSourceSpan(match.replacement[0], sourceTokens);\n fusionHandler({\n rule: rule.name,\n sourceTokens,\n fusedToken: match.replacement[0],\n });\n } else if (match.consumed > 1 && match.replacement.length < match.consumed) {\n for (const rt of match.replacement) {\n fusionHandler({\n rule: rule.name,\n sourceTokens,\n fusedToken: rt,\n });\n }\n }\n\n // Advance position past consumed tokens\n pos += match.consumed;\n changed = true;\n matched = true;\n break; // Rule matched — restart at new position with highest-priority rules\n }\n }\n\n if (!matched) {\n // No rule matched at this position, pass token through unchanged\n result.push(current[pos]);\n pos++;\n }\n }\n\n // Safety: bail if token count explodes (runaway rule expansion)\n if (result.length > maxTokens) {\n // A raw Error here would bypass ThreeTierEvaluator's DAG-preservation\n // enrichment on compile failure (it specifically checks for\n // EngineError). Same reasoning as ExpressionEngineSafety.ts's\n // complexity/length checks, which this mirrors.\n throw ErrorFactory.validation(\n \"NORMALIZED_TOKEN_LIMIT_EXCEEDED\",\n `Normalized token count (${result.length}) exceeds safety limit (${maxTokens})`,\n { tokenCount: result.length, maxTokens }\n );\n }\n\n current = result;\n }\n\n return current;\n }\n}\n\n//#endregion\n","//#region ─── Module Overview ───────────────────────────────────────────────────\n\n/**\n * PhraseTrie, optimized word-level trie for multi-word phrase fusion.\n *\n * ## Problem\n * The normalizer previously applied N separate `phraseFusionRule` instances,\n * each scanning forward from the current token position. For R phrase rules\n * and W average phrase length, this cost O(N × R × W) per pass.\n *\n * ## Solution\n * A single trie walk per position collapses all phrase rules into O(D)\n * where D ≤ longest phrase depth (typically ≤ 5 words). The trie tracks\n * the deepest terminal node reached, implementing longest-match-wins\n * without priority sorting.\n *\n * ## Optimizations\n * 1. **Set<string> quick-reject**, the `startWords` set contains the first\n * word of every registered phrase. At each position, if the token's\n * lowercase value isn't in the set, we bail in O(1) without touching\n * the trie. ~80% of tokens (numbers, operators) hit this fast path.\n * 2. **Longest-match-wins**, the `matchAt()` walk continues past terminal\n * nodes, tracking the deepest one. Shorter overlapping phrases (e.g.,\n * \"power of\") don't need lower priority, the trie naturally prefers\n * the longer match.\n * 3. **Map-based children**, `Map<string, TrieNode>` gives O(1) amortized\n * child lookup per word, faster than array scanning for sparse branches.\n *\n * ## Package integration\n * Packages add phrases via {@link TokenNormalizer.addPhrase} (public API)\n * or the {@link IEnginePackage.phrases} declarative field. Each call to\n * `addPhrase()` inserts the phrase into this trie and updates `startWords`.\n *\n * @module PhraseTrie\n */\n\n//#endregion\n//#region ─── Imports ──────────────────────────────────────────────────────────\n\nimport type { Token } from \"@solve-js/lexer/Token\";\nimport type { NormalizerMatch } from \"./NormalizerRule\";\nimport { createFusedToken } from \"./TokenNormalizer\";\n\n//#endregion\n//#region ─── TrieNode ─────────────────────────────────────────────────────────\n\n/**\n * Internal trie node.\n *\n * Each node represents a matched word in a phrase. The path from root\n * to a node spells out a partial or complete phrase.\n */\ninterface TrieNode {\n\t/** Child nodes keyed by next word (all lowercase). */\n\tchildren: Map<string, TrieNode>;\n\t/**\n\t * If this node completes a phrase, the terminal metadata.\n\t * A node can be both terminal AND have children. This handles\n\t * overlapping phrases like \"power of\" and \"to the power of\".\n\t */\n\tterminal: TrieTerminal | null;\n}\n\n/** Metadata stored at terminal nodes for deferred token creation. */\ninterface TrieTerminal {\n\t/** The target token type after fusion (e.g., \"CARET\", \"TIMES_BY\"). */\n\ttokenType: string;\n\t/** The original phrase string (e.g., \"to the power of\"). */\n\tphrase: string;\n\t/** Number of tokens consumed by this phrase. */\n\tconsumed: number;\n}\n\n//#endregion\n//#region ─── PhraseTrie ─────────────────────────────────────────────────────\n\n/**\n * Word-level trie for single-pass multi-word phrase matching.\n *\n * @example\n * ```ts\n * const trie = new PhraseTrie();\n * trie.addPhrase(\"to the power of\", \"CARET\");\n * trie.addPhrase(\"power of\", \"CARET\");\n * trie.addPhrase(\"abyssal whip\", \"ITEM\");\n *\n * // At position 0 with tokens [\"to\",\"the\",\"power\",\"of\",\"3\"]\n * const match = trie.matchAt(tokens, 0);\n * // → { consumed: 4, replacement: [CARET(\"to the power of\")] }\n * ```\n */\nexport class PhraseTrie {\n\t/**\n\t * First words that can start any registered phrase (all lowercase).\n\t * O(1) quick-reject: if `tokens[pos].value.toLowerCase()` isn't here,\n\t * no phrase can match at this position.\n\t */\n\tprivate startWords = new Set<string>();\n\n\t/**\n\t * Root maps first word → child node.\n\t * Two-level root avoids an unnecessary intermediate TrieNode.\n\t */\n\tprivate root = new Map<string, TrieNode>();\n\n\t// ── Registration ──────────────────────────────────────────────────────\n\n\t/**\n\t * Register a phrase for fusion into a single compound token.\n\t *\n\t * @param phrase - Multi-word phrase (e.g., \"to the power of\")\n\t * @param tokenType - Target token type after fusion (e.g., \"CARET\")\n\t */\n\taddPhrase(phrase: string, tokenType: string): void {\n\t\tconst words = phrase.toLowerCase().trim().split(/\\s+/);\n\t\tif (words.length === 0 || words[0] === \"\") return;\n\n\t\tthis.startWords.add(words[0]);\n\n\t\tconst firstWord = words[0];\n\t\tlet node: TrieNode;\n\n\t\tif (this.root.has(firstWord)) {\n\t\t\tnode = this.root.get(firstWord)!;\n\t\t} else {\n\t\t\tnode = { children: new Map(), terminal: null };\n\t\t\tthis.root.set(firstWord, node);\n\t\t}\n\n\t\t// Single-word phrase: terminal at root's child\n\t\tif (words.length === 1) {\n\t\t\tnode.terminal = { tokenType, phrase, consumed: 1 };\n\t\t\treturn;\n\t\t}\n\n\t\t// Walk/insert remaining words\n\t\tfor (let i = 1; i < words.length; i++) {\n\t\t\tconst word = words[i];\n\t\t\tif (!node.children.has(word)) {\n\t\t\t\tnode.children.set(word, { children: new Map(), terminal: null });\n\t\t\t}\n\t\t\tnode = node.children.get(word)!;\n\t\t}\n\n\t\tnode.terminal = { tokenType, phrase, consumed: words.length };\n\t}\n\n\t// ── Matching ──────────────────────────────────────────────────────────\n\n\t/**\n\t * Attempt to match a phrase starting at `pos` in the token stream.\n\t *\n\t * Walks the trie one token at a time, tracking the deepest terminal\n\t * node reached. Returns the longest match found, or `null` if no\n\t * phrase starts at this position.\n\t *\n\t * @param tokens - The current token stream\n\t * @param pos - Position to attempt matching from\n\t * @returns The longest {@link NormalizerMatch}, or `null` on no match\n\t */\n\tmatchAt(tokens: Token[], pos: number): NormalizerMatch | null {\n\t\t// ── Bounds guard ──\n\t\tif (pos >= tokens.length) return null;\n\n\t\t// ── A tag token never takes part in phrase fusion (#197, #213) ──\n\t\t// A `#tag` is a typed token, not a bare word, so it must not start or\n\t\t// complete a phrase. The trie matches on written value alone, so without\n\t\t// this guard a tag whose name equals a phrase or a phrase-continuation\n\t\t// word (`1200 #assuming`, `total of #column`) is fused into that grammar\n\t\t// before the category-tag rules ever run, and the tag is lost.\n\t\t//\n\t\t// The guard also covers the fused aggregate tokens the tags package emits\n\t\t// (`TAG_SUM` / `TAG_COUNT` / `TAG_AVERAGE`), whose value is the tag NAME:\n\t\t// otherwise `total of #assuming` fuses correctly to TAG_SUM(\"assuming\"),\n\t\t// then the trie re-reads that value on the next pass and turns it back\n\t\t// into the finance ASSUMING keyword (#213). The whole `TAG` / `TAG_*`\n\t\t// namespace belongs to the tags package.\n\t\tconst startType = tokens[pos].type;\n\t\tif (startType === \"TAG\" || startType.startsWith(\"TAG_\")) return null;\n\n\t\t// ── O(1) quick-reject: first word not a phrase starter ──\n\t\tconst firstWord = tokens[pos].value.toLowerCase();\n\t\tif (!this.startWords.has(firstWord)) return null;\n\n\t\t// ── Walk trie, tracking deepest terminal ──\n\t\tlet node: TrieNode | undefined = this.root.get(firstWord);\n\t\tif (!node) return null;\n\n\t\tlet best: NormalizerMatch | null = null;\n\t\tlet depth = 1;\n\n\t\t// Check single-word phrase at first node\n\t\tif (node.terminal) {\n\t\t\tconst sourceTokens = tokens.slice(pos, pos + 1);\n\t\t\tbest = {\n\t\t\t\tconsumed: 1,\n\t\t\t\treplacement: [createFusedToken(node.terminal.tokenType, node.terminal.phrase, sourceTokens)],\n\t\t\t\truleName: node.terminal.phrase,\n\t\t\t};\n\t\t}\n\n\t\t// Walk deeper for multi-word phrases\n\t\tfor (let i = pos + 1; i < tokens.length && node?.children; i++) {\n\t\t\tconst contType = tokens[i].type;\n\t\t\tif (contType === \"TAG\" || contType.startsWith(\"TAG_\")) break; // a tag token can't continue a phrase (#197, #213)\n\t\t\tconst word = tokens[i].value.toLowerCase();\n\t\t\tnode = node.children.get(word);\n\t\t\tif (!node) break; // dead end\n\t\t\tdepth++;\n\n\t\t\tif (node.terminal) {\n\t\t\t\t// Deferred fused-token creation, only on deepest match found\n\t\t\t\tconst sourceTokens = tokens.slice(pos, pos + depth);\n\t\t\t\tbest = {\n\t\t\t\t\tconsumed: depth,\n\t\t\t\t\treplacement: [\n\t\t\t\t\t\tcreateFusedToken(node.terminal.tokenType, node.terminal.phrase, sourceTokens),\n\t\t\t\t\t],\n\t\t\t\t\truleName: node.terminal.phrase,\n\t\t\t\t};\n\t\t\t}\n\t\t}\n\n\t\treturn best;\n\t}\n\n\t// ── Introspection ─────────────────────────────────────────────────────\n\n\t/** Number of unique first words (not total phrases). */\n\tget size(): number {\n\t\treturn this.root.size;\n\t}\n\n\t/**\n\t * Return all registered phrases and their target token types.\n\t *\n\t * Used by diagnostic mode to expose the complete trie structure to the\n\t * playground's NormalizerTab for rendering ALL registered phrases\n\t * (not just the ones that matched in this evaluation).\n\t */\n\tgetAllPhrases(): Record<string, string> {\n\t\tconst result: Record<string, string> = {};\n\n\t\tconst collect = (node: TrieNode, path: string[]): void => {\n\t\t\tif (node.terminal) {\n\t\t\t\tresult[path.join(' ')] = node.terminal.tokenType;\n\t\t\t}\n\t\t\tfor (const [word, child] of node.children) {\n\t\t\t\tcollect(child, [...path, word]);\n\t\t\t}\n\t\t};\n\n\t\tfor (const [firstWord, node] of this.root) {\n\t\t\tcollect(node, [firstWord]);\n\t\t}\n\n\t\treturn result;\n\t}\n\n\t/** Check if any phrase starts with this word (case-insensitive). */\n\tcanStart(word: string): boolean {\n\t\treturn this.startWords.has(word.toLowerCase());\n\t}\n}\n\n//#endregion\n","//#region ─── Module Overview ───────────────────────────────────────────────────\n\n/**\n * Built-in normalization rules for the {@link TokenNormalizer}.\n *\n * These rules handle common expression patterns that span multiple tokens.\n * Phrase fusion (e.g., \"to the power of\" → CARET) is handled by the\n * internal {@link PhraseTrie}. See {@link TokenNormalizer.addPhrase}.\n * This module only exports non-phrase rules like {@link implicitMultiplyRule}.\n *\n * @module BuiltinNormalizerRules\n */\n\n//#endregion\n//#region ─── Imports ──────────────────────────────────────────────────────────\n\nimport type { Token } from \"@solve-js/lexer/Token\";\nimport { tokenTypeId } from \"@solve-js/lexer/Token\";\nimport { LexerToken } from \"@solve-js/lexer/ExpressionLexer\";\nimport type { NormalizerRule, NormalizerMatch } from \"./NormalizerRule\";\n\n//#endregion\n//#region ─── PHRASE_START_WORDS, Implicit Multiply Guard ─────────────────────\n\n/**\n * Words that can start multi-word phrases, hardcoded fallback.\n *\n * This is ONLY used as the default fallback in {@link implicitMultiplyRule}\n * when no `canStart` predicate is provided. In the recommended pattern,\n * the {@link PhraseTrie}'s live `canStart()` set is passed instead, keeping\n * the guard in sync with package-registered phrases.\n *\n * @see {@link TokenNormalizer.canStartPhrase}\n */\n/**\n * Words that introduce a rate denominator when a unit follows.\n *\n * Kept next to the implicit-multiplication rule because that is the only\n * thing they change here; the rate itself is built in the uom package.\n */\nconst RATE_DENOMINATOR_WORDS = new Set([\"per\", \"a\", \"an\", \"each\", \"every\"]);\n\nconst PHRASE_START_WORDS = new Set([\n \"to\", \"power\", \"increase\", \"decrease\", \"times\", \"multiply\", \"divide\", \"by\",\n]);\n\n//#endregion\n//#region ─── implicitMultiplyRule, Implicit Operator Insertion ────────────────\n\n/**\n * Creates a normalization rule that inserts an implicit multiplication operator\n * between adjacent tokens where multiplication is implied.\n *\n * ## When it fires\n * Inserts a STAR token between:\n * - `NUMBER IDENT` (e.g., \"2 x\" → \"2 * x\")\n * - `RPAREN IDENT` (e.g., \"(x+1)y\" → \"(x+1) * y\")\n * - `NUMBER LPAREN` (e.g., \"2(x+1)\" → \"2 * (x+1)\")\n * - `NUMBER PI` / `NUMBER E` (e.g., \"2π\" → \"2 * π\")\n *\n * ## When it doesn't fire\n * - When the following identifier starts a multi-word phrase\n * (checked against {@link PHRASE_START_WORDS})\n * - When the following token is not an identifier or parenthesized expression\n *\n * ## Priority\n * Default priority is 50, below phrase fusion so phrases match first.\n *\n * @param priority - Rule priority (default 50)\n * @returns A {@link NormalizerRule} that inserts implicit multiply operators\n */\nexport function implicitMultiplyRule(\n\tpriority: number = 50,\n\tcanStart?: (word: string) => boolean\n): NormalizerRule {\n\tconst phraseGuard = canStart ?? ((word: string) => PHRASE_START_WORDS.has(word.toLowerCase()));\n\n\treturn {\n name: \"implicit:multiply\",\n priority,\n match(tokens: Token[], pos: number): NormalizerMatch | null {\n // ── Need at least one token after the current position ──\n if (pos + 1 >= tokens.length) return null;\n\n const t = tokens[pos];\n const next = tokens[pos + 1];\n\n // ── Guard: suppress if the next identifier starts a phrase ──\n // Uses the trie's canStart when available, falls back to hardcoded set.\n // This prevents \"2 power of 3\" from becoming \"2 * power of 3\".\n const nextValue = next.value.toLowerCase();\n if (phraseGuard(nextValue)) return null;\n\n // ── Guard: a rate denominator, not a multiplication ──\n // \"99 per week\" is ninety-nine a week, not ninety-nine times something\n // called per. The trie guard above cannot cover this, because these are\n // not registered phrases; they are recognised by the uom package's\n // bare-denominator rule, which runs after this one and then finds its\n // fused token stranded in operand position. The slash spelling never had\n // the problem, which is what made it hard to see.\n if (RATE_DENOMINATOR_WORDS.has(nextValue) && tokens[pos + 2]?.type === \"UNIT\") return null;\n\n // ── Check trigger conditions ──\n const triggers =\n (t.type === \"NUMBER\" || t.type === \"RPAREN\") &&\n (next.type === \"IDENT\" || next.type === \"LPAREN\" || next.type === \"PI\" || next.type === \"E\");\n\n if (!triggers) return null;\n\n // ── Insert a STAR token at the next token's position ──\n const starToken = new LexerToken(\n \"STAR\", tokenTypeId(\"STAR\"), \"*\", \"*\",\n next.offset, 0, next.line, next.col,\n );\n\n // consumed = 1: only the current token is replaced with [current, STAR]\n // The next token is NOT consumed, it stays for the next iteration\n return { consumed: 1, replacement: [t, starToken] };\n },\n };\n}\n\n//#endregion\n//#region ─── isInsideRangeContext, Bracket/Call-Paren Context Guard ───────────\n\n/**\n * Whether token `pos` sits inside a context where a bare `NUMBER:NUMBER`\n * sequence means a Range, not a clock-time/laptime/video-timecode literal:\n * a matrix literal/index/slice (`[1,2,3]`, `a[0:3]`), OR a `map`/`reduce`/\n * `sum`/`prod` call's own argument-list parens (`map(f, 0:3)`, these\n * accept a bare Range argument directly, per the Calca spec's own\n * example). Scans backward from `pos` over the CURRENT pass's token array,\n * tracking `[`/`(` nesting depth, the same \"positional guard via\n * backward scan\" idiom already used elsewhere in this normalizer layer\n * (e.g. `LineRefNormalizerRule`'s previous-token check), generalized to\n * depth-tracking. An `LBRACKET` is unconditionally a range-safe opener; an\n * `LPAREN` is range-safe ONLY when immediately preceded by MAP/REDUCE/\n * SUM_FN/PROD_FN, an ordinary grouping/function-call paren is NOT, so\n * `(9:00) + 5` still means a clock time, not a range.\n *\n * Needed because the clock-time/laptime/video-timecode rules each match a\n * bare `NUMBER COLON NUMBER...` shape with ZERO context-awareness, a real\n * collision discovered when adding Range support, since e.g. \"0:3\" is\n * valid input to BOTH features. These contexts have no legitimate use for\n * a clock-time/laptime/timecode literal, so the carve-out costs the time\n * features nothing.\n *\n * IMPORTANT cross-pass-timing gotcha (a real bug found and fixed here,\n * not a hypothetical): the `LPAREN`-opener check deliberately tests the\n * RAW word text (`prev.value.toLowerCase()`), not `prev.type ===\n * \"MAP\"/\"REDUCE\"/...`. The normalizer's multi-pass loop hands every rule\n * the SAME frozen `tokens` snapshot for an entire pass, a rule scanning\n * a LATER position in that pass cannot see a fusion `mapReduceCallNormalizerRule`\n * performs at an EARLIER position in that SAME pass (fusion results only\n * become visible to other rules starting the NEXT pass). Checking the\n * fused token type here would miss exactly the case that matters most\n * `map(f, 0:3)` on its very first normalization pass, silently letting\n * `0:3` fuse into a clock time before `map(`'s own fusion ever lands.\n * Testing the raw word is immune to this: it's true from the very first\n * pass, regardless of whether `mapReduceCallNormalizerRule` has run yet.\n * (`LBRACKET` above has no equivalent issue, it's a genuine lexer token\n * from the start, never itself the product of a fusion.)\n */\nexport function isInsideRangeContext(tokens: Token[], pos: number): boolean {\n const safeStack: boolean[] = [];\n for (let i = 0; i < pos; i++) {\n const t = tokens[i];\n if (t.type === \"LBRACKET\") {\n safeStack.push(true);\n } else if (t.type === \"LPAREN\") {\n const prev = tokens[i - 1];\n const opensMapReduceCall = !!prev && (\n prev.type === \"MAP\" || prev.type === \"REDUCE\" || prev.type === \"SUM_FN\" || prev.type === \"PROD_FN\" ||\n (prev.type === \"IDENT\" && (prev.value.toLowerCase() === \"map\" || prev.value.toLowerCase() === \"reduce\" || prev.value.toLowerCase() === \"sum\" || prev.value.toLowerCase() === \"prod\"))\n );\n safeStack.push(opensMapReduceCall);\n } else if (t.type === \"RBRACKET\" || t.type === \"RPAREN\") {\n safeStack.pop();\n }\n }\n return safeStack.length > 0 && safeStack[safeStack.length - 1];\n}\n\n//#endregion\n//#region ─── createBuiltinNormalizerRules, All Built-in Rules ─────────────────\n\n/**\n * Creates built-in non-phrase normalization rules.\n *\n * Returns implicit multiply insertion (priority 50).\n *\n * **Prefer** calling {@link implicitMultiplyRule} directly with a\n * `canStart` predicate wired to the normalizer's phrase trie:\n * ```ts\n * normalizer.register(implicitMultiplyRule(50, (w) => normalizer.canStartPhrase(w)));\n * ```\n * Without the predicate, this function falls back to a hardcoded\n * {@link PHRASE_START_WORDS} set that won't reflect package-registered phrases.\n *\n * @returns An array of {@link NormalizerRule} instances ready for registration\n */\nexport function createBuiltinNormalizerRules(): NormalizerRule[] {\n return [\n // ── Implicit operator insertion (priority 50) ──\n implicitMultiplyRule(),\n ];\n}\n\n/**\n * Built-in phrase → tokenType mappings.\n *\n * These are registered into the engine's {@link PhraseTrie} during\n * construction. Tests that create a standalone {@link TokenNormalizer}\n * should register these via {@link TokenNormalizer.addPhrase}.\n */\nexport const BUILTIN_PHRASES: Record<string, string> = {\n\t\"to the power of\": \"CARET\",\n\t\"power of\": \"CARET\",\n\t\"increase by\": \"INCREASE_BY\",\n\t\"decrease by\": \"DECREASE_BY\",\n\t\"times by\": \"TIMES_BY\",\n\t\"multiply by\": \"MULTIPLY_BY\",\n\t\"divide by\": \"DIVIDE_BY\",\n\t// The past-tense spellings, which is how the operation is usually written\n\t// out: \"3 multiplied by 4\". Same tokens, so no new parselets. Both this\n\t// branch and the parity branch added \"multiplied by\" independently; the\n\t// merge kept both copies and the locale table rejected the duplicate key.\n\t\"multiplied by\": \"MULTIPLY_BY\",\n\t\"divided by\": \"DIVIDE_BY\",\n};\n\n//#endregion\n"]}
|
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
'use strict';var chunkY7O4D3BJ_cjs=require('./chunk-Y7O4D3BJ.cjs'),
|
|
2
|
-
//# sourceMappingURL=chunk-
|
|
1
|
+
'use strict';var chunkY7O4D3BJ_cjs=require('./chunk-Y7O4D3BJ.cjs'),chunkSNVPM27Q_cjs=require('./chunk-SNVPM27Q.cjs'),chunkMSYX32PF_cjs=require('./chunk-MSYX32PF.cjs'),chunkWEAKI5V5_cjs=require('./chunk-WEAKI5V5.cjs');var u=(s=>(s.Main="main",s.Inline="inline",s.String="string",s))(u||{});var l=class{constructor(e="en",t){this.currentState="main";this.hasPeeked=false;this.tokens=[];this.tokenIdx=0;this.expressionLexer=new chunkSNVPM27Q_cjs.b(e,t);}reset(e,t){let s=t??"main";if(this.currentState=s,this.hasPeeked=false,this.peekedToken=void 0,s==="main"){if(this.expressionLexer.classifyLine(e).skip){this.tokens=[],this.tokenIdx=0;return}this.expressionLexer.reset(e),this.tokens=this.expressionLexer.tokenizeAll(),this.tokenIdx=0;}else this.expressionLexer.reset(e),this.tokens=this.expressionLexer.tokenizeAll(),this.tokenIdx=0;}classifyLine(e){return this.expressionLexer.classifyLine(e)}findInlineSolves(e){return this.expressionLexer.findInlineSolves(e)}getKeywords(){return this.expressionLexer.getKeywords()}next(){if(this.hasPeeked)return this.hasPeeked=false,this.peekedToken;if(this.tokenIdx<this.tokens.length)return this.tokens[this.tokenIdx++]}peek(){return this.hasPeeked?this.peekedToken:(this.peekedToken=this.next(),this.hasPeeked=true,this.peekedToken)}[Symbol.iterator](){return this.tokens[Symbol.iterator]()}registerVocabulary(e){this.expressionLexer.registerVocabulary(e);}unregisterVocabulary(e){this.expressionLexer.unregisterVocabulary(e);}resetExpression(e){this.currentState="main",this.hasPeeked=false,this.peekedToken=void 0,this.expressionLexer.reset(e),this.tokens=this.expressionLexer.tokenizeAll(),this.tokenIdx=0;}scanDocument(e){return this.expressionLexer.scanDocument(e)}getState(){return this.currentState}setState(e){this.currentState=e;}getHighlightTokens(e){let t=this.expressionLexer.classifyLine(e);return t.skip&&e.startsWith("> ")?this.collectHighlightTokens(e.slice(2)):t.skip?[]:this.collectHighlightTokens(e)}getHighlightTokenObjects(e){let t=this.expressionLexer.classifyLine(e);return t.skip&&e.startsWith("> ")?this.collectTokenObjects(e.slice(2)):t.skip?[]:this.collectTokenObjects(e)}collectTokenObjects(e){this.resetExpression(e);let t=[];for(let s of this)s.type==="WS"||s.type==="NEWLINE"||s.type.startsWith("MD_")||s.type==="INLINE_SOLVE_START"||s.type==="BACKTICK_CLOSE"||t.push(s);return t}collectHighlightTokens(e){return this.collectTokenObjects(e).map(t=>({type:t.type,value:t.value,offset:t.offset,col:t.col,length:t.text.length,category:chunkY7O4D3BJ_cjs.d(t.type)}))}},T=new l("en",void 0);var n=class{constructor(){this.classes=[];this.localeKeywordMap=null;this.localePhraseMap=null;this.unitNames=null;}register(e){this.classes.push(e);}unregister(e){this.classes=this.classes.filter(t=>t.tokenType!==e);}setLocale(e,t){this.localeKeywordMap=e,this.localePhraseMap=t??null;}setUnits(e){this.unitNames=e;}build(){let e=new Map;if(this.localeKeywordMap)for(let[r,i]of Object.entries(this.localeKeywordMap))e.set(r.toLowerCase(),i);let t=[...this.classes].sort((r,i)=>(r.priority??0)-(i.priority??0));for(let r of t)for(let i of Object.keys(r.keywords))e.set(i.toLowerCase(),r.tokenType);let s=this.buildPhraseTrie(),o=this.buildPhraseStartWords();return {keywordToType:e,phraseTrie:s,phraseStartWords:o,unitNames:this.unitNames??new Set}}buildPhraseTrie(){let e={children:new Map};if(this.localePhraseMap)for(let[t,s]of Object.entries(this.localePhraseMap))this.insertPhrase(e,t.toLowerCase(),s);for(let t of this.classes)if(t.phrases)for(let s of Object.keys(t.phrases))this.insertPhrase(e,s.toLowerCase(),t.tokenType);return e.children.size>0?e:null}insertPhrase(e,t,s){let o=t.split(" "),r=e;for(let i of o)r.children.has(i)||r.children.set(i,{children:new Map}),r=r.children.get(i);r.type||(r.type=s);}buildPhraseStartWords(){let e=new Set;if(this.localePhraseMap)for(let t of Object.keys(this.localePhraseMap)){let s=t.split(" ")[0].toLowerCase();e.add(s);}for(let t of this.classes)if(t.phrases)for(let s of Object.keys(t.phrases)){let o=s.split(" ")[0].toLowerCase();e.add(o);}return e}};var k={"to the power of":"CARET","power of":"CARET","increase by":"INCREASE_BY","decrease by":"DECREASE_BY","times by":"TIMES_BY","multiply by":"MULTIPLY_BY","multiplied by":"MULTIPLY_BY","divide by":"DIVIDE_BY"};function P(a="en"){let e=chunkMSYX32PF_cjs.a(a),t=new n;return t.setLocale(e.keywordMap,k),t.setUnits(chunkWEAKI5V5_cjs.a),t.build()}exports.a=u;exports.b=l;exports.c=T;exports.d=P;//# sourceMappingURL=chunk-3HYO3QYW.cjs.map
|
|
2
|
+
//# sourceMappingURL=chunk-3HYO3QYW.cjs.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/lexer/LexerState.ts","../src/lexer/Lexer.ts","../src/lexer/TokenClassRegistry.ts","../src/lexer/tokenRegistration.ts"],"names":["LexerState","Lexer","localeCode","tokenLookup","ExpressionLexer","input","state","newState","lineText","plugin","text","classification","result","token","getTokenCategory","sharedLexer","TokenClassRegistry","tokenClass","tokenType","c","keywordMap","phraseMap","unitNames","keywordToType","keyword","sorted","a","b","tc","phraseTrie","phraseStartWords","root","phrase","words","node","word","startWords","first","BUILTIN_PHRASES","buildTokenLookup","locale","getLocale","registry","knownUnits"],"mappings":"yNAMO,IAAKA,CAAAA,CAAAA,CAAAA,CAAAA,GACXA,EAAA,IAAA,CAAO,MAAA,CACPA,EAAA,MAAA,CAAS,QAAA,CACTA,EAAA,MAAA,CAAS,QAAA,CAHEA,OAAA,EAAA,ECaL,IAAMC,EAAN,KAAY,CAkBjB,YAAYC,CAAAA,CAAa,IAAA,CAAMC,EAA2B,CAf1D,IAAA,CAAQ,aAA2B,MAAA,CAEnC,IAAA,CAAQ,UAAY,KAAA,CAIpB,IAAA,CAAQ,OAAkB,EAAC,CAC3B,KAAQ,QAAA,CAAmB,CAAA,CAYzB,KAAK,eAAA,CAAkB,IAAIC,oBAAgBF,CAAAA,CAAYC,CAAW,EACpE,CAEA,KAAA,CAAME,EAAeC,CAAAA,CAA0B,CAC7C,IAAMC,CAAAA,CAAWD,CAAAA,EAAS,OAQ1B,GAPA,IAAA,CAAK,aAAeC,CAAAA,CACpB,IAAA,CAAK,UAAY,KAAA,CACjB,IAAA,CAAK,YAAc,MAAA,CAKfA,CAAAA,GAAa,OAAiB,CAEhC,GADuB,KAAK,eAAA,CAAgB,YAAA,CAAaF,CAAK,CAAA,CAC3C,IAAA,CAAM,CACvB,IAAA,CAAK,MAAA,CAAS,EAAC,CACf,IAAA,CAAK,SAAW,CAAA,CAChB,MACF,CAEA,IAAA,CAAK,eAAA,CAAgB,MAAMA,CAAK,CAAA,CAChC,KAAK,MAAA,CAAS,IAAA,CAAK,gBAAgB,WAAA,EAAY,CAC/C,KAAK,QAAA,CAAW,EAClB,MAEE,IAAA,CAAK,eAAA,CAAgB,MAAMA,CAAK,CAAA,CAChC,KAAK,MAAA,CAAS,IAAA,CAAK,gBAAgB,WAAA,EAAY,CAC/C,KAAK,QAAA,CAAW,EAEpB,CAMA,YAAA,CAAaG,CAAAA,CAAsC,CACjD,OAAO,IAAA,CAAK,gBAAgB,YAAA,CAAaA,CAAQ,CACnD,CAMA,gBAAA,CAAiBA,EAAkB,CACjC,OAAO,KAAK,eAAA,CAAgB,gBAAA,CAAiBA,CAAQ,CACvD,CAMA,aAAsC,CACpC,OAAO,KAAK,eAAA,CAAgB,WAAA,EAC9B,CAEA,IAAA,EAA0B,CACxB,GAAI,IAAA,CAAK,UACP,OAAA,IAAA,CAAK,SAAA,CAAY,MACV,IAAA,CAAK,WAAA,CAGd,GAAI,IAAA,CAAK,QAAA,CAAW,KAAK,MAAA,CAAO,MAAA,CAC9B,OAAO,IAAA,CAAK,MAAA,CAAO,KAAK,QAAA,EAAU,CAGtC,CAEA,IAAA,EAA0B,CACxB,OAAI,IAAA,CAAK,SAAA,CAAkB,KAAK,WAAA,EAChC,IAAA,CAAK,YAAc,IAAA,CAAK,IAAA,GACxB,IAAA,CAAK,SAAA,CAAY,KACV,IAAA,CAAK,WAAA,CACd,CAEA,CAAC,MAAA,CAAO,QAAQ,CAAA,EAAqB,CACnC,OAAO,IAAA,CAAK,MAAA,CAAO,MAAA,CAAO,QAAQ,CAAA,EACpC,CAQA,kBAAA,CAAmBC,CAAAA,CAA+B,CAChD,IAAA,CAAK,eAAA,CAAgB,mBAAmBA,CAAM,EAChD,CAMA,oBAAA,CAAqBA,CAAAA,CAA+B,CAClD,IAAA,CAAK,eAAA,CAAgB,qBAAqBA,CAAM,EAClD,CAOA,eAAA,CAAgBJ,CAAAA,CAAqB,CACnC,IAAA,CAAK,YAAA,CAAe,OACpB,IAAA,CAAK,SAAA,CAAY,MACjB,IAAA,CAAK,WAAA,CAAc,OACnB,IAAA,CAAK,eAAA,CAAgB,MAAMA,CAAK,CAAA,CAChC,KAAK,MAAA,CAAS,IAAA,CAAK,gBAAgB,WAAA,EAAY,CAC/C,KAAK,QAAA,CAAW,EAClB,CAQA,YAAA,CAAaK,CAAAA,CAAgC,CAC3C,OAAO,IAAA,CAAK,gBAAgB,YAAA,CAAaA,CAAI,CAC/C,CAEA,QAAA,EAAuB,CACrB,OAAO,IAAA,CAAK,YACd,CAEA,QAAA,CAASJ,EAAyB,CAChC,IAAA,CAAK,aAAeA,EACtB,CAEA,mBAAmBE,CAAAA,CAAqI,CACtJ,IAAMG,CAAAA,CAAiB,IAAA,CAAK,gBAAgB,YAAA,CAAaH,CAAQ,EAKjE,OAAIG,CAAAA,CAAe,MAAQH,CAAAA,CAAS,UAAA,CAAW,IAAI,CAAA,CAC1C,IAAA,CAAK,uBAAuBA,CAAAA,CAAS,KAAA,CAAM,CAAC,CAAC,CAAA,CAGlDG,EAAe,IAAA,CACV,GAGF,IAAA,CAAK,sBAAA,CAAuBH,CAAQ,CAC7C,CAYA,yBAAyBA,CAAAA,CAA2B,CAClD,IAAMG,CAAAA,CAAiB,IAAA,CAAK,gBAAgB,YAAA,CAAaH,CAAQ,EACjE,OAAIG,CAAAA,CAAe,MAAQH,CAAAA,CAAS,UAAA,CAAW,IAAI,CAAA,CAC1C,IAAA,CAAK,oBAAoBA,CAAAA,CAAS,KAAA,CAAM,CAAC,CAAC,CAAA,CAE/CG,EAAe,IAAA,CAAa,GACzB,IAAA,CAAK,mBAAA,CAAoBH,CAAQ,CAC1C,CAEQ,oBAAoBA,CAAAA,CAA2B,CACrD,KAAK,eAAA,CAAgBA,CAAQ,EAC7B,IAAMI,CAAAA,CAAkB,EAAC,CACzB,IAAA,IAAWC,KAAS,IAAA,CACdA,CAAAA,CAAM,OAAS,IAAA,EAAQA,CAAAA,CAAM,OAAS,SAAA,EACtCA,CAAAA,CAAM,KAAK,UAAA,CAAW,KAAK,GAC3BA,CAAAA,CAAM,IAAA,GAAS,sBAAwBA,CAAAA,CAAM,IAAA,GAAS,kBAC1DD,CAAAA,CAAO,IAAA,CAAKC,CAAK,CAAA,CAEnB,OAAOD,CACT,CAEQ,sBAAA,CAAuBJ,EAAqI,CAClK,OAAO,KAAK,mBAAA,CAAoBA,CAAQ,EAAE,GAAA,CAAIK,CAAAA,GAAU,CACtD,IAAA,CAAMA,CAAAA,CAAM,KACZ,KAAA,CAAOA,CAAAA,CAAM,KAAA,CACb,MAAA,CAAQA,CAAAA,CAAM,MAAA,CACd,IAAKA,CAAAA,CAAM,GAAA,CAIX,OAAQA,CAAAA,CAAM,IAAA,CAAK,OACnB,QAAA,CAAUC,mBAAAA,CAAiBD,EAAM,IAAI,CACvC,EAAE,CACJ,CACF,EAcaE,CAAAA,CAAc,IAAId,EAAM,IAAA,CAAM,MAAS,EClJ7C,IAAMe,CAAAA,CAAN,KAAyB,CAAzB,WAAA,EAAA,CACL,KAAQ,OAAA,CAAwB,GAChC,IAAA,CAAQ,gBAAA,CAAkD,KAC1D,IAAA,CAAQ,eAAA,CAAiD,KACzD,IAAA,CAAQ,SAAA,CAAwC,MAUhD,QAAA,CAASC,CAAAA,CAA8B,CACrC,IAAA,CAAK,OAAA,CAAQ,KAAKA,CAAU,EAC9B,CAMA,UAAA,CAAWC,CAAAA,CAAyB,CAClC,IAAA,CAAK,OAAA,CAAU,KAAK,OAAA,CAAQ,MAAA,CAAOC,GAAKA,CAAAA,CAAE,SAAA,GAAcD,CAAS,EACnE,CAMA,UAAUE,CAAAA,CAAoCC,CAAAA,CAA0C,CACtF,IAAA,CAAK,gBAAA,CAAmBD,EACxB,IAAA,CAAK,eAAA,CAAkBC,GAAa,KACtC,CASA,SAASC,CAAAA,CAAsC,CAC7C,KAAK,SAAA,CAAYA,EACnB,CAYA,KAAA,EAAqB,CACnB,IAAMC,CAAAA,CAAgB,IAAI,IAG1B,GAAI,IAAA,CAAK,iBACP,IAAA,GAAW,CAACC,EAASN,CAAS,CAAA,GAAK,OAAO,OAAA,CAAQ,IAAA,CAAK,gBAAgB,CAAA,CACrEK,CAAAA,CAAc,IAAIC,CAAAA,CAAQ,WAAA,GAAeN,CAAS,CAAA,CAKtD,IAAMO,CAAAA,CAAS,CAAC,GAAG,IAAA,CAAK,OAAO,EAAE,IAAA,CAAK,CAACC,EAAGC,CAAAA,GAAAA,CAAOD,CAAAA,CAAE,UAAY,CAAA,GAAMC,CAAAA,CAAE,UAAY,CAAA,CAAE,CAAA,CACrF,QAAWC,CAAAA,IAAMH,CAAAA,CACf,QAAWD,CAAAA,IAAW,MAAA,CAAO,KAAKI,CAAAA,CAAG,QAAQ,EAC3CL,CAAAA,CAAc,GAAA,CAAIC,EAAQ,WAAA,EAAY,CAAGI,EAAG,SAAS,CAAA,CAKzD,IAAMC,CAAAA,CAAa,IAAA,CAAK,iBAAgB,CAGlCC,CAAAA,CAAmB,KAAK,qBAAA,EAAsB,CAEpD,OAAO,CACL,aAAA,CAAAP,EACA,UAAA,CAAAM,CAAAA,CACA,iBAAAC,CAAAA,CACA,SAAA,CAAW,KAAK,SAAA,EAAa,IAAI,GACnC,CACF,CAIQ,iBAAqC,CAC3C,IAAMC,EAAmB,CAAE,QAAA,CAAU,IAAI,GAAM,CAAA,CAG/C,GAAI,IAAA,CAAK,eAAA,CACP,OAAW,CAACC,CAAAA,CAAQd,CAAS,CAAA,GAAK,MAAA,CAAO,QAAQ,IAAA,CAAK,eAAe,CAAA,CACnE,IAAA,CAAK,YAAA,CAAaa,CAAAA,CAAMC,EAAO,WAAA,EAAY,CAAGd,CAAS,CAAA,CAK3D,IAAA,IAAWU,KAAM,IAAA,CAAK,OAAA,CACpB,GAAKA,CAAAA,CAAG,OAAA,CACR,QAAWI,CAAAA,IAAU,MAAA,CAAO,KAAKJ,CAAAA,CAAG,OAAO,EACzC,IAAA,CAAK,YAAA,CAAaG,EAAMC,CAAAA,CAAO,WAAA,GAAeJ,CAAAA,CAAG,SAAS,EAI9D,OAAOG,CAAAA,CAAK,SAAS,IAAA,CAAO,CAAA,CAAIA,EAAO,IACzC,CAEQ,aAAaA,CAAAA,CAAkBC,CAAAA,CAAgBd,EAAyB,CAC9E,IAAMe,EAAQD,CAAAA,CAAO,KAAA,CAAM,GAAG,CAAA,CAC1BE,CAAAA,CAAOH,EACX,IAAA,IAAWI,CAAAA,IAAQF,EACZC,CAAAA,CAAK,QAAA,CAAS,IAAIC,CAAI,CAAA,EACzBD,EAAK,QAAA,CAAS,GAAA,CAAIC,EAAM,CAAE,QAAA,CAAU,IAAI,GAAM,CAAC,EAEjDD,CAAAA,CAAOA,CAAAA,CAAK,SAAS,GAAA,CAAIC,CAAI,EAG1BD,CAAAA,CAAK,IAAA,GACRA,EAAK,IAAA,CAAOhB,CAAAA,EAEhB,CAEQ,qBAAA,EAAqC,CAC3C,IAAMkB,CAAAA,CAAa,IAAI,IAGvB,GAAI,IAAA,CAAK,gBACP,IAAA,IAAWJ,CAAAA,IAAU,OAAO,IAAA,CAAK,IAAA,CAAK,eAAe,CAAA,CAAG,CACtD,IAAMK,CAAAA,CAAQL,CAAAA,CAAO,MAAM,GAAG,CAAA,CAAE,CAAC,CAAA,CAAE,WAAA,GACnCI,CAAAA,CAAW,GAAA,CAAIC,CAAK,EACtB,CAIF,QAAWT,CAAAA,IAAM,IAAA,CAAK,QACpB,GAAKA,CAAAA,CAAG,QACR,IAAA,IAAWI,CAAAA,IAAU,OAAO,IAAA,CAAKJ,CAAAA,CAAG,OAAO,CAAA,CAAG,CAC5C,IAAMS,CAAAA,CAAQL,CAAAA,CAAO,MAAM,GAAG,CAAA,CAAE,CAAC,CAAA,CAAE,WAAA,GACnCI,CAAAA,CAAW,GAAA,CAAIC,CAAK,EACtB,CAGF,OAAOD,CACT,CACF,EClOA,IAAME,CAAAA,CAA0C,CAC9C,iBAAA,CAAmB,OAAA,CACnB,WAAY,OAAA,CACZ,aAAA,CAAe,cACf,aAAA,CAAe,aAAA,CACf,WAAY,UAAA,CACZ,aAAA,CAAe,cACf,eAAA,CAAiB,aAAA,CACjB,YAAa,WACf,CAAA,CAiBO,SAASC,CAAAA,CAAiBrC,CAAAA,CAAa,KAAmB,CAC/D,IAAMsC,EAAkBC,mBAAAA,CAAUvC,CAAU,EACtCwC,CAAAA,CAAW,IAAI1B,EAGrB,OAAA0B,CAAAA,CAAS,UAAUF,CAAAA,CAAO,UAAA,CAAYF,CAAe,CAAA,CAGrDI,CAAAA,CAAS,SAASC,mBAAU,CAAA,CAGrBD,CAAAA,CAAS,KAAA,EAClB","file":"chunk-32GC2FHC.cjs","sourcesContent":["/**\n * Lexer state machine modes.\n * - Main: document-level scanning with markdown classification\n * - Inline: expression embedded in markdown inline solve (`s\\`...\\``)\n * - String: inside a double-quoted string literal\n */\nexport enum LexerState {\n\tMain = \"main\",\n\tInline = \"inline\",\n\tString = \"string\",\n}\n","import { ExpressionLexer, LineClassification, LexerVocabulary, type ScanLineResult } from \"./ExpressionLexer\";\nimport { Token } from \"@solve-js/lexer/Token\";\nimport { LexerState } from \"@solve-js/lexer/LexerState\";\nimport { getTokenCategory } from \"@solve-js/language/TokenCategoryMap\";\nimport type { TokenCategory } from \"@solve-js/language/TokenCategory\";\nimport type { TokenLookup } from \"@solve-js/lexer/TokenClassRegistry\";\n\n/**\n * Public tokenizer wrapper around {@link ExpressionLexer}.\n *\n * `ExpressionLexer` does the actual character-by-character scanning;\n * `Lexer` adds a materialized-token-array streaming interface\n * (`next()`/`peek()`) plus line-classification state (`reset()`) so\n * callers can iterate a line's tokens without re-scanning on each peek.\n *\n * Each `ExpressionEngine` instance owns its own `Lexer`, and packages\n * extend it via {@link registerVocabulary} (keywords, operators, units)\n * see `IEnginePackage.lexerVocabulary`.\n */\nexport class Lexer {\n /** Expression-mode lexer (Phase A: V8-optimized, replaces moo) */\n private expressionLexer: ExpressionLexer;\n private currentState: LexerState = LexerState.Main;\n private peekedToken: Token | undefined;\n private hasPeeked = false;\n\n // Materialized token array from the last reset() call, used for\n // next()/peek() streaming access.\n private tokens: Token[] = [];\n private tokenIdx: number = 0;\n\n /**\n * @param localeCode - Locale code (e.g., \"en\", \"de\"). Defaults to \"en\".\n * @param tokenLookup - Optional TokenLookup from TokenClassRegistry.\n * When provided, configures ExpressionLexer to use registry-built\n * keyword/unit/phrase lookups instead of internal instance maps.\n */\n constructor(localeCode = \"en\", tokenLookup?: TokenLookup) {\n // Pass the lookup directly to ExpressionLexer's constructor, it's an\n // instance field now, not a static. Each Lexer instance gets its own\n // isolated lookup, preventing cross-instance corruption.\n this.expressionLexer = new ExpressionLexer(localeCode, tokenLookup);\n }\n\n reset(input: string, state?: LexerState): void {\n const newState = state ?? LexerState.Main;\n this.currentState = newState;\n this.hasPeeked = false;\n this.peekedToken = undefined;\n\n // Phase B: Main state classifies the line with the markdown scanner.\n // Skip lines (headings, fences, HRs, etc.) produce empty token arrays.\n // Expression lines and lines with inline solves are tokenized normally.\n if (newState === LexerState.Main) {\n const classification = this.expressionLexer.classifyLine(input);\n if (classification.skip) {\n this.tokens = [];\n this.tokenIdx = 0;\n return;\n }\n // Expression line or markdown line with inline solves, tokenize.\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n } else {\n // Non-main states (Inline, String), expression tokenization.\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n }\n }\n\n /**\n * Classify a single line of markdown text (Phase B).\n * Delegates to the ExpressionLexer's character-by-character scanner.\n */\n classifyLine(lineText: string): LineClassification {\n return this.expressionLexer.classifyLine(lineText);\n }\n\n /**\n * Find all inline solve markers in a line (Phase B).\n * Delegates to the ExpressionLexer's character-by-character scanner.\n */\n findInlineSolves(lineText: string) {\n return this.expressionLexer.findInlineSolves(lineText);\n }\n\n /**\n * Every keyword this lexer currently recognizes (locale + plugin-contributed),\n * mapped to the token type it lexes to. Delegates to the ExpressionLexer.\n */\n getKeywords(): Record<string, string> {\n return this.expressionLexer.getKeywords();\n }\n\n next(): Token | undefined {\n if (this.hasPeeked) {\n this.hasPeeked = false;\n return this.peekedToken;\n }\n // Materialized token array (ExpressionLexer path).\n if (this.tokenIdx < this.tokens.length) {\n return this.tokens[this.tokenIdx++];\n }\n return undefined;\n }\n\n peek(): Token | undefined {\n if (this.hasPeeked) return this.peekedToken;\n this.peekedToken = this.next();\n this.hasPeeked = true;\n return this.peekedToken;\n }\n\n [Symbol.iterator](): Iterator<Token> {\n return this.tokens[Symbol.iterator]();\n }\n\n /**\n * Register a plugin to extend the lexer with custom tokens.\n * Delegates to the underlying ExpressionLexer.\n *\n * @see LexerVocabulary for the supported extension points.\n */\n registerVocabulary(plugin: LexerVocabulary): void {\n this.expressionLexer.registerVocabulary(plugin);\n }\n\n /**\n * Unregister a plugin, removing its custom tokens from the lexer.\n * Delegates to the underlying ExpressionLexer.\n */\n unregisterVocabulary(plugin: LexerVocabulary): void {\n this.expressionLexer.unregisterVocabulary(plugin);\n }\n\n /**\n * Reset the lexer for expression-only text, skips the classifyLine()\n * overhead in reset() for callers that already know the input is an\n * evaluable expression (e.g., after isEmptyLine() confirmed non-skip).\n */\n resetExpression(input: string): void {\n this.currentState = LexerState.Main;\n this.hasPeeked = false;\n this.peekedToken = undefined;\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n }\n\n /**\n * Scan a full document in one pass, classifying each line and\n * tokenizing non-skipped lines. Delegates to ExpressionLexer.\n *\n * @returns ScanLineResult[], one per line, with classification + tokens.\n */\n scanDocument(text: string): ScanLineResult[] {\n return this.expressionLexer.scanDocument(text);\n }\n\n getState(): LexerState {\n return this.currentState;\n }\n\n setState(state: LexerState): void {\n this.currentState = state;\n }\n\n getHighlightTokens(lineText: string): {type: string; value: string; offset: number; col: number; length: number; category: TokenCategory | undefined}[] {\n const classification = this.expressionLexer.classifyLine(lineText);\n\n // For blockquote lines, strip the \"> \" prefix and tokenize the expression content.\n // This lets expressions inside blockquotes (e.g., \"> 1 + 2\") get syntax highlighted\n // while pure structural lines (headings, code fences) remain unhighlighted.\n if (classification.skip && lineText.startsWith(\"> \")) {\n return this.collectHighlightTokens(lineText.slice(2));\n }\n\n if (classification.skip) {\n return [];\n }\n\n return this.collectHighlightTokens(lineText);\n }\n\n /**\n * The same tokens {@link getHighlightTokens} reduces, before reduction.\n *\n * Exists because normalization operates on tokens, not on the flattened\n * shape, and a consumer that wants phrase-fused highlighting has to run the\n * normalizer between the two. See `LanguageService.getSemanticTokens`.\n *\n * @param lineText - One line of source.\n * @returns Every token on the line that is worth painting, unreduced.\n */\n getHighlightTokenObjects(lineText: string): Token[] {\n const classification = this.expressionLexer.classifyLine(lineText);\n if (classification.skip && lineText.startsWith(\"> \")) {\n return this.collectTokenObjects(lineText.slice(2));\n }\n if (classification.skip) return [];\n return this.collectTokenObjects(lineText);\n }\n\n private collectTokenObjects(lineText: string): Token[] {\n this.resetExpression(lineText);\n const result: Token[] = [];\n for (const token of this) {\n if (token.type === \"WS\" || token.type === \"NEWLINE\") continue;\n if (token.type.startsWith(\"MD_\")) continue;\n if (token.type === \"INLINE_SOLVE_START\" || token.type === \"BACKTICK_CLOSE\") continue;\n result.push(token);\n }\n return result;\n }\n\n private collectHighlightTokens(lineText: string): {type: string; value: string; offset: number; col: number; length: number; category: TokenCategory | undefined}[] {\n return this.collectTokenObjects(lineText).map(token => ({\n type: token.type,\n value: token.value,\n offset: token.offset,\n col: token.col,\n // `text`, not `value`: this is a span into the source, and the two\n // differ for a string literal, whose value is the payload while its\n // text still carries the quote characters the reader typed.\n length: token.text.length,\n category: getTokenCategory(token.type),\n }));\n }\n}\n\n/**\n * A lexer for operations that do not depend on registered vocabulary.\n *\n * Line classification and inline-solve detection read characters looking for\n * headings, comment markers, fences and backtick spans, and never consult the\n * keyword, unit or operator tables. Every lexer therefore returns the same\n * answer, so the callers that have no engine to ask can use this one. Checked\n * by `__tests__/lexer/LineClassificationIsVocabularyIndependent.spec.ts`.\n *\n * Do not tokenize with this. An engine's own lexer carries the vocabulary its\n * packages registered; this one carries none.\n */\nexport const sharedLexer = new Lexer(\"en\", undefined);","/**\n * TokenClass, Plugin-extensible keyword registration for the Lexer.\n *\n * Providers call `registry.register(tokenClass)` to teach the lexer about\n * their keywords. The registry merges locale keywords, provider keywords,\n * phrase mappings, and unit names into an optimized TokenLookup structure\n * consumed by the Lexer.\n *\n * @example\n * registry.register({\n * tokenType: 'CARET',\n * keywords: {},\n * phrases: { 'to the power of': true, 'power of': true },\n * priority: 10,\n * description: 'Exponentiation operators (x^y)',\n * });\n */\nexport interface TokenClass {\n /** The token type string produced by the lexer (e.g., \"FUNC\", \"PI\", \"CARET\").\n * Must match a token type that a ParseletRegistry has a parselet for. */\n tokenType: string;\n\n /** Single-word keywords (case-insensitive). The lexer lowercases input\n * before lookup, so these should be lowercase. Example:\n * { sqrt: true, abs: true, sin: true, cos: true } for tokenType \"FUNC\" */\n keywords: Record<string, boolean>;\n\n /** Multi-word phrases (case-insensitive). Matched by the built-in PhraseMatcher\n * via the phrase trie. Example:\n * { \"to the power of\": true, \"power of\": true } for tokenType \"CARET\" */\n phrases?: Record<string, boolean>;\n\n /** Priority for conflict resolution. When two TokenClasses register\n * the same keyword, the higher-priority class wins. Locale keywords\n * have priority 0 (set via setLocale). Providers should use\n * priority >= 10 to override locale defaults. Default: 0 */\n priority?: number;\n\n /** Human-readable description for debugging and introspection */\n description?: string;\n}\n\n// ── Phrase Trie ──────────────────────────────────────────────────────────────\n\n/** Trie node for multi-word phrase matching. */\nexport interface PhraseNode {\n /** Complete phrase token type (null = intermediate node) */\n type?: string;\n children: Map<string, PhraseNode>;\n}\n\n// ── TokenLookup, Optimized lookup structure for the Lexer ──────────────────\n\n/**\n * The optimized lookup structure built by TokenClassRegistry.build().\n * Consumed by the Lexer for O(1) keyword → token type lookups and\n * O(word-count) phrase matching.\n */\nexport interface TokenLookup {\n /** Lowercase keyword → token type. O(1) Map lookup. */\n keywordToType: Map<string, string>;\n\n /** Phrase trie for multi-word matching. Root node with children maps.\n * Null if no phrases registered. */\n phraseTrie: PhraseNode | null;\n\n /** Set of lowercase first-words of all registered phrases.\n * Used by the lexer to emit IDENT (not a phrase keyword) for words\n * that start multi-word phrases, deferring to the PhraseMatcher.\n *\n * Example: \"to\" is in phraseStartWords because \"to the power of\" is a phrase.\n * When the lexer sees \"to\", it emits IDENT and lets the phrase matcher\n * combine \"to the power of\" into a single CARET token.\n *\n * This prevents plugins from accidentally overriding phrase-start words.\n */\n phraseStartWords: Set<string>;\n\n /** Case-sensitive unit names for UNIT fallback after keyword lookup fails. */\n unitNames: ReadonlySet<string>;\n}\n\n// ── TokenClassRegistry ───────────────────────────────────────────────────────\n\n/**\n * Central registry for keyword→token-type mappings.\n *\n * Providers register TokenClasses; locales provide keyword maps;\n * units provide a name set. `build()` merges all sources into an\n * optimized TokenLookup consumed by the Lexer.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0)\n * 2. Provider keywords (sorted by priority ascending, higher priority wins)\n *\n * Unit names are stored separately (checked AFTER keyword lookup fails).\n * Phrases are stored in a trie for O(phrase-length) matching.\n */\nexport class TokenClassRegistry {\n private classes: TokenClass[] = [];\n private localeKeywordMap: Record<string, string> | null = null;\n private localePhraseMap: Record<string, string> | null = null;\n private unitNames: ReadonlySet<string> | null = null;\n\n /**\n * Register a provider's TokenClass. Must be called BEFORE build().\n * Can be called multiple times to add more entries.\n *\n * Built-in token types CANNOT be overridden, throws a EngineError\n * if the TokenClass attempts to register a keyword that conflicts\n * with an already-registered token type.\n */\n register(tokenClass: TokenClass): void {\n this.classes.push(tokenClass);\n }\n\n /**\n * Unregister all TokenClasses for a given token type.\n * Useful for plugin unload. Requires rebuild() to take effect.\n */\n unregister(tokenType: string): void {\n this.classes = this.classes.filter(c => c.tokenType !== tokenType);\n }\n\n /**\n * Set the locale's keyword→type map and optional phrase map.\n * Called on locale change. Priority 0 (cannot override providers with higher priority).\n */\n setLocale(keywordMap: Record<string, string>, phraseMap?: Record<string, string>): void {\n this.localeKeywordMap = keywordMap;\n this.localePhraseMap = phraseMap ?? null;\n }\n\n /**\n * Set the unit name set. Called when unit list changes.\n * Units are stored separately (checked AFTER keyword lookup fails).\n *\n * Takes a ReadonlySet because the caller's set is derived from the\n * conversion tables and must not be mutated; this class only ever reads it.\n */\n setUnits(unitNames: ReadonlySet<string>): void {\n this.unitNames = unitNames;\n }\n\n /**\n * Build the optimized TokenLookup from all registered sources.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0)\n * 2. Provider classes (sorted by priority ascending)\n *\n * Returns a frozen TokenLookup that the Lexer consumes.\n * Call build() again after register()/setLocale()/setUnits() changes.\n */\n build(): TokenLookup {\n const keywordToType = new Map<string, string>();\n\n // Layer 1: Locale keywords (priority 0, lowest)\n if (this.localeKeywordMap) {\n for (const [keyword, tokenType] of Object.entries(this.localeKeywordMap)) {\n keywordToType.set(keyword.toLowerCase(), tokenType);\n }\n }\n\n // Layer 2: Provider keywords (sorted by priority ascending, higher wins)\n const sorted = [...this.classes].sort((a, b) => (a.priority ?? 0) - (b.priority ?? 0));\n for (const tc of sorted) {\n for (const keyword of Object.keys(tc.keywords)) {\n keywordToType.set(keyword.toLowerCase(), tc.tokenType);\n }\n }\n\n // Build phrase trie from locale + provider phrases\n const phraseTrie = this.buildPhraseTrie();\n\n // Build phraseStartWords set (first word of every phrase)\n const phraseStartWords = this.buildPhraseStartWords();\n\n return {\n keywordToType,\n phraseTrie,\n phraseStartWords,\n unitNames: this.unitNames ?? new Set(),\n };\n }\n\n // ── Private helpers ─────────────────────────────────────────────────────\n\n private buildPhraseTrie(): PhraseNode | null {\n const root: PhraseNode = { children: new Map() };\n\n // Layer 1: Locale phrases\n if (this.localePhraseMap) {\n for (const [phrase, tokenType] of Object.entries(this.localePhraseMap)) {\n this.insertPhrase(root, phrase.toLowerCase(), tokenType);\n }\n }\n\n // Layer 2: Provider phrases\n for (const tc of this.classes) {\n if (!tc.phrases) continue;\n for (const phrase of Object.keys(tc.phrases)) {\n this.insertPhrase(root, phrase.toLowerCase(), tc.tokenType);\n }\n }\n\n return root.children.size > 0 ? root : null;\n }\n\n private insertPhrase(root: PhraseNode, phrase: string, tokenType: string): void {\n const words = phrase.split(' ');\n let node = root;\n for (const word of words) {\n if (!node.children.has(word)) {\n node.children.set(word, { children: new Map() });\n }\n node = node.children.get(word)!;\n }\n // Only set type if not already set, first-registered (locale) wins\n if (!node.type) {\n node.type = tokenType;\n }\n }\n\n private buildPhraseStartWords(): Set<string> {\n const startWords = new Set<string>();\n\n // Locale phrase first words\n if (this.localePhraseMap) {\n for (const phrase of Object.keys(this.localePhraseMap)) {\n const first = phrase.split(' ')[0].toLowerCase();\n startWords.add(first);\n }\n }\n\n // Provider phrase first words\n for (const tc of this.classes) {\n if (!tc.phrases) continue;\n for (const phrase of Object.keys(tc.phrases)) {\n const first = phrase.split(' ')[0].toLowerCase();\n startWords.add(first);\n }\n }\n\n return startWords;\n }\n}","/**\n * tokenRegistration.ts, Bootstrap for building TokenLookup from locale, units, and phrases.\n *\n * This module centralizes the assembly of the TokenLookup consumed by ExpressionLexer.\n * It merges:\n * 1. Locale keywords (from ILocale.keywordMap)\n * 2. Built-in phrase patterns (to the power of, increase by, etc.)\n * 3. Known unit names (from units.ts)\n *\n * The resulting TokenLookup is frozen and passed to ExpressionLexer.configuredLookup\n * at construction time, enabling data-driven keyword/unit/phrase resolution.\n */\nimport { TokenClassRegistry } from '@solve-js/lexer/TokenClassRegistry';\nimport type { TokenLookup } from '@solve-js/lexer/TokenClassRegistry';\nimport { getLocale, type ILocale } from '@solve-js/constants/locales';\nimport { knownUnits } from '@solve-js/lexer/units';\n\n// ── Built-in phrase map ───────────────────────────────────────────────────\n// These are the multi-word expressions handled as compound tokens.\n// Matched by the PhraseMatcher (trie-based) in ExpressionLexer.tryMatchPhrase().\nconst BUILTIN_PHRASES: Record<string, string> = {\n 'to the power of': 'CARET',\n 'power of': 'CARET',\n 'increase by': 'INCREASE_BY',\n 'decrease by': 'DECREASE_BY',\n 'times by': 'TIMES_BY',\n 'multiply by': 'MULTIPLY_BY',\n 'multiplied by': 'MULTIPLY_BY',\n 'divide by': 'DIVIDE_BY',\n};\n\n/**\n * Build a TokenLookup from locale keywords, known units, and built-in phrases.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0, lowest, can be overridden by providers)\n * 2. Built-in phrases (via locale phraseMap)\n * 3. Known units (checked AFTER keyword lookup fails, via unitNames set)\n *\n * The resulting TokenLookup replaces the internal keyword map, unit set,\n * phrase trie, and phraseStartWords in ExpressionLexer when set via\n * ExpressionLexer.configuredLookup.\n *\n * @param localeCode - The locale code (e.g., \"en\", \"de\"). Defaults to \"en\".\n * @returns A frozen TokenLookup ready for consumption by the lexer.\n */\nexport function buildTokenLookup(localeCode = 'en'): TokenLookup {\n const locale: ILocale = getLocale(localeCode);\n const registry = new TokenClassRegistry();\n\n // Layer 1: Locale keywords (priority 0, lowest)\n registry.setLocale(locale.keywordMap, BUILTIN_PHRASES);\n\n // Layer 2: Known units (checked after keyword lookup)\n registry.setUnits(knownUnits);\n\n // Build and return the lookup\n return registry.build();\n}\n"]}
|
|
1
|
+
{"version":3,"sources":["../src/lexer/LexerState.ts","../src/lexer/Lexer.ts","../src/lexer/TokenClassRegistry.ts","../src/lexer/tokenRegistration.ts"],"names":["LexerState","Lexer","localeCode","tokenLookup","ExpressionLexer","input","state","newState","lineText","plugin","text","classification","result","token","getTokenCategory","sharedLexer","TokenClassRegistry","tokenClass","tokenType","c","keywordMap","phraseMap","unitNames","keywordToType","keyword","sorted","a","b","tc","phraseTrie","phraseStartWords","root","phrase","words","node","word","startWords","first","BUILTIN_PHRASES","buildTokenLookup","locale","getLocale","registry","knownUnits"],"mappings":"yNAMO,IAAKA,CAAAA,CAAAA,CAAAA,CAAAA,GACXA,EAAA,IAAA,CAAO,MAAA,CACPA,EAAA,MAAA,CAAS,QAAA,CACTA,EAAA,MAAA,CAAS,QAAA,CAHEA,OAAA,EAAA,ECaL,IAAMC,EAAN,KAAY,CAkBjB,YAAYC,CAAAA,CAAa,IAAA,CAAMC,EAA2B,CAf1D,IAAA,CAAQ,aAA2B,MAAA,CAEnC,IAAA,CAAQ,UAAY,KAAA,CAIpB,IAAA,CAAQ,OAAkB,EAAC,CAC3B,KAAQ,QAAA,CAAmB,CAAA,CAYzB,KAAK,eAAA,CAAkB,IAAIC,oBAAgBF,CAAAA,CAAYC,CAAW,EACpE,CAEA,KAAA,CAAME,EAAeC,CAAAA,CAA0B,CAC7C,IAAMC,CAAAA,CAAWD,CAAAA,EAAS,OAQ1B,GAPA,IAAA,CAAK,aAAeC,CAAAA,CACpB,IAAA,CAAK,UAAY,KAAA,CACjB,IAAA,CAAK,YAAc,MAAA,CAKfA,CAAAA,GAAa,OAAiB,CAEhC,GADuB,KAAK,eAAA,CAAgB,YAAA,CAAaF,CAAK,CAAA,CAC3C,IAAA,CAAM,CACvB,IAAA,CAAK,MAAA,CAAS,EAAC,CACf,IAAA,CAAK,SAAW,CAAA,CAChB,MACF,CAEA,IAAA,CAAK,eAAA,CAAgB,MAAMA,CAAK,CAAA,CAChC,KAAK,MAAA,CAAS,IAAA,CAAK,gBAAgB,WAAA,EAAY,CAC/C,KAAK,QAAA,CAAW,EAClB,MAEE,IAAA,CAAK,eAAA,CAAgB,MAAMA,CAAK,CAAA,CAChC,KAAK,MAAA,CAAS,IAAA,CAAK,gBAAgB,WAAA,EAAY,CAC/C,KAAK,QAAA,CAAW,EAEpB,CAMA,YAAA,CAAaG,CAAAA,CAAsC,CACjD,OAAO,IAAA,CAAK,gBAAgB,YAAA,CAAaA,CAAQ,CACnD,CAMA,gBAAA,CAAiBA,EAAkB,CACjC,OAAO,KAAK,eAAA,CAAgB,gBAAA,CAAiBA,CAAQ,CACvD,CAMA,aAAsC,CACpC,OAAO,KAAK,eAAA,CAAgB,WAAA,EAC9B,CAEA,IAAA,EAA0B,CACxB,GAAI,IAAA,CAAK,UACP,OAAA,IAAA,CAAK,SAAA,CAAY,MACV,IAAA,CAAK,WAAA,CAGd,GAAI,IAAA,CAAK,QAAA,CAAW,KAAK,MAAA,CAAO,MAAA,CAC9B,OAAO,IAAA,CAAK,MAAA,CAAO,KAAK,QAAA,EAAU,CAGtC,CAEA,IAAA,EAA0B,CACxB,OAAI,IAAA,CAAK,SAAA,CAAkB,KAAK,WAAA,EAChC,IAAA,CAAK,YAAc,IAAA,CAAK,IAAA,GACxB,IAAA,CAAK,SAAA,CAAY,KACV,IAAA,CAAK,WAAA,CACd,CAEA,CAAC,MAAA,CAAO,QAAQ,CAAA,EAAqB,CACnC,OAAO,IAAA,CAAK,MAAA,CAAO,MAAA,CAAO,QAAQ,CAAA,EACpC,CAQA,kBAAA,CAAmBC,CAAAA,CAA+B,CAChD,IAAA,CAAK,eAAA,CAAgB,mBAAmBA,CAAM,EAChD,CAMA,oBAAA,CAAqBA,CAAAA,CAA+B,CAClD,IAAA,CAAK,eAAA,CAAgB,qBAAqBA,CAAM,EAClD,CAOA,eAAA,CAAgBJ,CAAAA,CAAqB,CACnC,IAAA,CAAK,YAAA,CAAe,OACpB,IAAA,CAAK,SAAA,CAAY,MACjB,IAAA,CAAK,WAAA,CAAc,OACnB,IAAA,CAAK,eAAA,CAAgB,MAAMA,CAAK,CAAA,CAChC,KAAK,MAAA,CAAS,IAAA,CAAK,gBAAgB,WAAA,EAAY,CAC/C,KAAK,QAAA,CAAW,EAClB,CAQA,YAAA,CAAaK,CAAAA,CAAgC,CAC3C,OAAO,IAAA,CAAK,gBAAgB,YAAA,CAAaA,CAAI,CAC/C,CAEA,QAAA,EAAuB,CACrB,OAAO,IAAA,CAAK,YACd,CAEA,QAAA,CAASJ,EAAyB,CAChC,IAAA,CAAK,aAAeA,EACtB,CAEA,mBAAmBE,CAAAA,CAAqI,CACtJ,IAAMG,CAAAA,CAAiB,IAAA,CAAK,gBAAgB,YAAA,CAAaH,CAAQ,EAKjE,OAAIG,CAAAA,CAAe,MAAQH,CAAAA,CAAS,UAAA,CAAW,IAAI,CAAA,CAC1C,IAAA,CAAK,uBAAuBA,CAAAA,CAAS,KAAA,CAAM,CAAC,CAAC,CAAA,CAGlDG,EAAe,IAAA,CACV,GAGF,IAAA,CAAK,sBAAA,CAAuBH,CAAQ,CAC7C,CAYA,yBAAyBA,CAAAA,CAA2B,CAClD,IAAMG,CAAAA,CAAiB,IAAA,CAAK,gBAAgB,YAAA,CAAaH,CAAQ,EACjE,OAAIG,CAAAA,CAAe,MAAQH,CAAAA,CAAS,UAAA,CAAW,IAAI,CAAA,CAC1C,IAAA,CAAK,oBAAoBA,CAAAA,CAAS,KAAA,CAAM,CAAC,CAAC,CAAA,CAE/CG,EAAe,IAAA,CAAa,GACzB,IAAA,CAAK,mBAAA,CAAoBH,CAAQ,CAC1C,CAEQ,oBAAoBA,CAAAA,CAA2B,CACrD,KAAK,eAAA,CAAgBA,CAAQ,EAC7B,IAAMI,CAAAA,CAAkB,EAAC,CACzB,IAAA,IAAWC,KAAS,IAAA,CACdA,CAAAA,CAAM,OAAS,IAAA,EAAQA,CAAAA,CAAM,OAAS,SAAA,EACtCA,CAAAA,CAAM,KAAK,UAAA,CAAW,KAAK,GAC3BA,CAAAA,CAAM,IAAA,GAAS,sBAAwBA,CAAAA,CAAM,IAAA,GAAS,kBAC1DD,CAAAA,CAAO,IAAA,CAAKC,CAAK,CAAA,CAEnB,OAAOD,CACT,CAEQ,sBAAA,CAAuBJ,EAAqI,CAClK,OAAO,KAAK,mBAAA,CAAoBA,CAAQ,EAAE,GAAA,CAAIK,CAAAA,GAAU,CACtD,IAAA,CAAMA,CAAAA,CAAM,KACZ,KAAA,CAAOA,CAAAA,CAAM,KAAA,CACb,MAAA,CAAQA,CAAAA,CAAM,MAAA,CACd,IAAKA,CAAAA,CAAM,GAAA,CAIX,OAAQA,CAAAA,CAAM,IAAA,CAAK,OACnB,QAAA,CAAUC,mBAAAA,CAAiBD,EAAM,IAAI,CACvC,EAAE,CACJ,CACF,EAcaE,CAAAA,CAAc,IAAId,EAAM,IAAA,CAAM,MAAS,EClJ7C,IAAMe,CAAAA,CAAN,KAAyB,CAAzB,WAAA,EAAA,CACL,KAAQ,OAAA,CAAwB,GAChC,IAAA,CAAQ,gBAAA,CAAkD,KAC1D,IAAA,CAAQ,eAAA,CAAiD,KACzD,IAAA,CAAQ,SAAA,CAAwC,MAUhD,QAAA,CAASC,CAAAA,CAA8B,CACrC,IAAA,CAAK,OAAA,CAAQ,KAAKA,CAAU,EAC9B,CAMA,UAAA,CAAWC,CAAAA,CAAyB,CAClC,IAAA,CAAK,OAAA,CAAU,KAAK,OAAA,CAAQ,MAAA,CAAOC,GAAKA,CAAAA,CAAE,SAAA,GAAcD,CAAS,EACnE,CAMA,UAAUE,CAAAA,CAAoCC,CAAAA,CAA0C,CACtF,IAAA,CAAK,gBAAA,CAAmBD,EACxB,IAAA,CAAK,eAAA,CAAkBC,GAAa,KACtC,CASA,SAASC,CAAAA,CAAsC,CAC7C,KAAK,SAAA,CAAYA,EACnB,CAYA,KAAA,EAAqB,CACnB,IAAMC,CAAAA,CAAgB,IAAI,IAG1B,GAAI,IAAA,CAAK,iBACP,IAAA,GAAW,CAACC,EAASN,CAAS,CAAA,GAAK,OAAO,OAAA,CAAQ,IAAA,CAAK,gBAAgB,CAAA,CACrEK,CAAAA,CAAc,IAAIC,CAAAA,CAAQ,WAAA,GAAeN,CAAS,CAAA,CAKtD,IAAMO,CAAAA,CAAS,CAAC,GAAG,IAAA,CAAK,OAAO,EAAE,IAAA,CAAK,CAACC,EAAGC,CAAAA,GAAAA,CAAOD,CAAAA,CAAE,UAAY,CAAA,GAAMC,CAAAA,CAAE,UAAY,CAAA,CAAE,CAAA,CACrF,QAAWC,CAAAA,IAAMH,CAAAA,CACf,QAAWD,CAAAA,IAAW,MAAA,CAAO,KAAKI,CAAAA,CAAG,QAAQ,EAC3CL,CAAAA,CAAc,GAAA,CAAIC,EAAQ,WAAA,EAAY,CAAGI,EAAG,SAAS,CAAA,CAKzD,IAAMC,CAAAA,CAAa,IAAA,CAAK,iBAAgB,CAGlCC,CAAAA,CAAmB,KAAK,qBAAA,EAAsB,CAEpD,OAAO,CACL,aAAA,CAAAP,EACA,UAAA,CAAAM,CAAAA,CACA,iBAAAC,CAAAA,CACA,SAAA,CAAW,KAAK,SAAA,EAAa,IAAI,GACnC,CACF,CAIQ,iBAAqC,CAC3C,IAAMC,EAAmB,CAAE,QAAA,CAAU,IAAI,GAAM,CAAA,CAG/C,GAAI,IAAA,CAAK,eAAA,CACP,OAAW,CAACC,CAAAA,CAAQd,CAAS,CAAA,GAAK,MAAA,CAAO,QAAQ,IAAA,CAAK,eAAe,CAAA,CACnE,IAAA,CAAK,YAAA,CAAaa,CAAAA,CAAMC,EAAO,WAAA,EAAY,CAAGd,CAAS,CAAA,CAK3D,IAAA,IAAWU,KAAM,IAAA,CAAK,OAAA,CACpB,GAAKA,CAAAA,CAAG,OAAA,CACR,QAAWI,CAAAA,IAAU,MAAA,CAAO,KAAKJ,CAAAA,CAAG,OAAO,EACzC,IAAA,CAAK,YAAA,CAAaG,EAAMC,CAAAA,CAAO,WAAA,GAAeJ,CAAAA,CAAG,SAAS,EAI9D,OAAOG,CAAAA,CAAK,SAAS,IAAA,CAAO,CAAA,CAAIA,EAAO,IACzC,CAEQ,aAAaA,CAAAA,CAAkBC,CAAAA,CAAgBd,EAAyB,CAC9E,IAAMe,EAAQD,CAAAA,CAAO,KAAA,CAAM,GAAG,CAAA,CAC1BE,CAAAA,CAAOH,EACX,IAAA,IAAWI,CAAAA,IAAQF,EACZC,CAAAA,CAAK,QAAA,CAAS,IAAIC,CAAI,CAAA,EACzBD,EAAK,QAAA,CAAS,GAAA,CAAIC,EAAM,CAAE,QAAA,CAAU,IAAI,GAAM,CAAC,EAEjDD,CAAAA,CAAOA,CAAAA,CAAK,SAAS,GAAA,CAAIC,CAAI,EAG1BD,CAAAA,CAAK,IAAA,GACRA,EAAK,IAAA,CAAOhB,CAAAA,EAEhB,CAEQ,qBAAA,EAAqC,CAC3C,IAAMkB,CAAAA,CAAa,IAAI,IAGvB,GAAI,IAAA,CAAK,gBACP,IAAA,IAAWJ,CAAAA,IAAU,OAAO,IAAA,CAAK,IAAA,CAAK,eAAe,CAAA,CAAG,CACtD,IAAMK,CAAAA,CAAQL,CAAAA,CAAO,MAAM,GAAG,CAAA,CAAE,CAAC,CAAA,CAAE,WAAA,GACnCI,CAAAA,CAAW,GAAA,CAAIC,CAAK,EACtB,CAIF,QAAWT,CAAAA,IAAM,IAAA,CAAK,QACpB,GAAKA,CAAAA,CAAG,QACR,IAAA,IAAWI,CAAAA,IAAU,OAAO,IAAA,CAAKJ,CAAAA,CAAG,OAAO,CAAA,CAAG,CAC5C,IAAMS,CAAAA,CAAQL,CAAAA,CAAO,MAAM,GAAG,CAAA,CAAE,CAAC,CAAA,CAAE,WAAA,GACnCI,CAAAA,CAAW,GAAA,CAAIC,CAAK,EACtB,CAGF,OAAOD,CACT,CACF,EClOA,IAAME,CAAAA,CAA0C,CAC9C,iBAAA,CAAmB,OAAA,CACnB,WAAY,OAAA,CACZ,aAAA,CAAe,cACf,aAAA,CAAe,aAAA,CACf,WAAY,UAAA,CACZ,aAAA,CAAe,cACf,eAAA,CAAiB,aAAA,CACjB,YAAa,WACf,CAAA,CAiBO,SAASC,CAAAA,CAAiBrC,CAAAA,CAAa,KAAmB,CAC/D,IAAMsC,EAAkBC,mBAAAA,CAAUvC,CAAU,EACtCwC,CAAAA,CAAW,IAAI1B,EAGrB,OAAA0B,CAAAA,CAAS,UAAUF,CAAAA,CAAO,UAAA,CAAYF,CAAe,CAAA,CAGrDI,CAAAA,CAAS,SAASC,mBAAU,CAAA,CAGrBD,CAAAA,CAAS,KAAA,EAClB","file":"chunk-3HYO3QYW.cjs","sourcesContent":["/**\n * Lexer state machine modes.\n * - Main: document-level scanning with markdown classification\n * - Inline: expression embedded in markdown inline solve (`s\\`...\\``)\n * - String: inside a double-quoted string literal\n */\nexport enum LexerState {\n\tMain = \"main\",\n\tInline = \"inline\",\n\tString = \"string\",\n}\n","import { ExpressionLexer, LineClassification, LexerVocabulary, type ScanLineResult } from \"./ExpressionLexer\";\nimport { Token } from \"@solve-js/lexer/Token\";\nimport { LexerState } from \"@solve-js/lexer/LexerState\";\nimport { getTokenCategory } from \"@solve-js/language/TokenCategoryMap\";\nimport type { TokenCategory } from \"@solve-js/language/TokenCategory\";\nimport type { TokenLookup } from \"@solve-js/lexer/TokenClassRegistry\";\n\n/**\n * Public tokenizer wrapper around {@link ExpressionLexer}.\n *\n * `ExpressionLexer` does the actual character-by-character scanning;\n * `Lexer` adds a materialized-token-array streaming interface\n * (`next()`/`peek()`) plus line-classification state (`reset()`) so\n * callers can iterate a line's tokens without re-scanning on each peek.\n *\n * Each `ExpressionEngine` instance owns its own `Lexer`, and packages\n * extend it via {@link registerVocabulary} (keywords, operators, units)\n * see `IEnginePackage.lexerVocabulary`.\n */\nexport class Lexer {\n /** Expression-mode lexer (Phase A: V8-optimized, replaces moo) */\n private expressionLexer: ExpressionLexer;\n private currentState: LexerState = LexerState.Main;\n private peekedToken: Token | undefined;\n private hasPeeked = false;\n\n // Materialized token array from the last reset() call, used for\n // next()/peek() streaming access.\n private tokens: Token[] = [];\n private tokenIdx: number = 0;\n\n /**\n * @param localeCode - Locale code (e.g., \"en\", \"de\"). Defaults to \"en\".\n * @param tokenLookup - Optional TokenLookup from TokenClassRegistry.\n * When provided, configures ExpressionLexer to use registry-built\n * keyword/unit/phrase lookups instead of internal instance maps.\n */\n constructor(localeCode = \"en\", tokenLookup?: TokenLookup) {\n // Pass the lookup directly to ExpressionLexer's constructor, it's an\n // instance field now, not a static. Each Lexer instance gets its own\n // isolated lookup, preventing cross-instance corruption.\n this.expressionLexer = new ExpressionLexer(localeCode, tokenLookup);\n }\n\n reset(input: string, state?: LexerState): void {\n const newState = state ?? LexerState.Main;\n this.currentState = newState;\n this.hasPeeked = false;\n this.peekedToken = undefined;\n\n // Phase B: Main state classifies the line with the markdown scanner.\n // Skip lines (headings, fences, HRs, etc.) produce empty token arrays.\n // Expression lines and lines with inline solves are tokenized normally.\n if (newState === LexerState.Main) {\n const classification = this.expressionLexer.classifyLine(input);\n if (classification.skip) {\n this.tokens = [];\n this.tokenIdx = 0;\n return;\n }\n // Expression line or markdown line with inline solves, tokenize.\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n } else {\n // Non-main states (Inline, String), expression tokenization.\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n }\n }\n\n /**\n * Classify a single line of markdown text (Phase B).\n * Delegates to the ExpressionLexer's character-by-character scanner.\n */\n classifyLine(lineText: string): LineClassification {\n return this.expressionLexer.classifyLine(lineText);\n }\n\n /**\n * Find all inline solve markers in a line (Phase B).\n * Delegates to the ExpressionLexer's character-by-character scanner.\n */\n findInlineSolves(lineText: string) {\n return this.expressionLexer.findInlineSolves(lineText);\n }\n\n /**\n * Every keyword this lexer currently recognizes (locale + plugin-contributed),\n * mapped to the token type it lexes to. Delegates to the ExpressionLexer.\n */\n getKeywords(): Record<string, string> {\n return this.expressionLexer.getKeywords();\n }\n\n next(): Token | undefined {\n if (this.hasPeeked) {\n this.hasPeeked = false;\n return this.peekedToken;\n }\n // Materialized token array (ExpressionLexer path).\n if (this.tokenIdx < this.tokens.length) {\n return this.tokens[this.tokenIdx++];\n }\n return undefined;\n }\n\n peek(): Token | undefined {\n if (this.hasPeeked) return this.peekedToken;\n this.peekedToken = this.next();\n this.hasPeeked = true;\n return this.peekedToken;\n }\n\n [Symbol.iterator](): Iterator<Token> {\n return this.tokens[Symbol.iterator]();\n }\n\n /**\n * Register a plugin to extend the lexer with custom tokens.\n * Delegates to the underlying ExpressionLexer.\n *\n * @see LexerVocabulary for the supported extension points.\n */\n registerVocabulary(plugin: LexerVocabulary): void {\n this.expressionLexer.registerVocabulary(plugin);\n }\n\n /**\n * Unregister a plugin, removing its custom tokens from the lexer.\n * Delegates to the underlying ExpressionLexer.\n */\n unregisterVocabulary(plugin: LexerVocabulary): void {\n this.expressionLexer.unregisterVocabulary(plugin);\n }\n\n /**\n * Reset the lexer for expression-only text, skips the classifyLine()\n * overhead in reset() for callers that already know the input is an\n * evaluable expression (e.g., after isEmptyLine() confirmed non-skip).\n */\n resetExpression(input: string): void {\n this.currentState = LexerState.Main;\n this.hasPeeked = false;\n this.peekedToken = undefined;\n this.expressionLexer.reset(input);\n this.tokens = this.expressionLexer.tokenizeAll();\n this.tokenIdx = 0;\n }\n\n /**\n * Scan a full document in one pass, classifying each line and\n * tokenizing non-skipped lines. Delegates to ExpressionLexer.\n *\n * @returns ScanLineResult[], one per line, with classification + tokens.\n */\n scanDocument(text: string): ScanLineResult[] {\n return this.expressionLexer.scanDocument(text);\n }\n\n getState(): LexerState {\n return this.currentState;\n }\n\n setState(state: LexerState): void {\n this.currentState = state;\n }\n\n getHighlightTokens(lineText: string): {type: string; value: string; offset: number; col: number; length: number; category: TokenCategory | undefined}[] {\n const classification = this.expressionLexer.classifyLine(lineText);\n\n // For blockquote lines, strip the \"> \" prefix and tokenize the expression content.\n // This lets expressions inside blockquotes (e.g., \"> 1 + 2\") get syntax highlighted\n // while pure structural lines (headings, code fences) remain unhighlighted.\n if (classification.skip && lineText.startsWith(\"> \")) {\n return this.collectHighlightTokens(lineText.slice(2));\n }\n\n if (classification.skip) {\n return [];\n }\n\n return this.collectHighlightTokens(lineText);\n }\n\n /**\n * The same tokens {@link getHighlightTokens} reduces, before reduction.\n *\n * Exists because normalization operates on tokens, not on the flattened\n * shape, and a consumer that wants phrase-fused highlighting has to run the\n * normalizer between the two. See `LanguageService.getSemanticTokens`.\n *\n * @param lineText - One line of source.\n * @returns Every token on the line that is worth painting, unreduced.\n */\n getHighlightTokenObjects(lineText: string): Token[] {\n const classification = this.expressionLexer.classifyLine(lineText);\n if (classification.skip && lineText.startsWith(\"> \")) {\n return this.collectTokenObjects(lineText.slice(2));\n }\n if (classification.skip) return [];\n return this.collectTokenObjects(lineText);\n }\n\n private collectTokenObjects(lineText: string): Token[] {\n this.resetExpression(lineText);\n const result: Token[] = [];\n for (const token of this) {\n if (token.type === \"WS\" || token.type === \"NEWLINE\") continue;\n if (token.type.startsWith(\"MD_\")) continue;\n if (token.type === \"INLINE_SOLVE_START\" || token.type === \"BACKTICK_CLOSE\") continue;\n result.push(token);\n }\n return result;\n }\n\n private collectHighlightTokens(lineText: string): {type: string; value: string; offset: number; col: number; length: number; category: TokenCategory | undefined}[] {\n return this.collectTokenObjects(lineText).map(token => ({\n type: token.type,\n value: token.value,\n offset: token.offset,\n col: token.col,\n // `text`, not `value`: this is a span into the source, and the two\n // differ for a string literal, whose value is the payload while its\n // text still carries the quote characters the reader typed.\n length: token.text.length,\n category: getTokenCategory(token.type),\n }));\n }\n}\n\n/**\n * A lexer for operations that do not depend on registered vocabulary.\n *\n * Line classification and inline-solve detection read characters looking for\n * headings, comment markers, fences and backtick spans, and never consult the\n * keyword, unit or operator tables. Every lexer therefore returns the same\n * answer, so the callers that have no engine to ask can use this one. Checked\n * by `__tests__/lexer/LineClassificationIsVocabularyIndependent.spec.ts`.\n *\n * Do not tokenize with this. An engine's own lexer carries the vocabulary its\n * packages registered; this one carries none.\n */\nexport const sharedLexer = new Lexer(\"en\", undefined);","/**\n * TokenClass, Plugin-extensible keyword registration for the Lexer.\n *\n * Providers call `registry.register(tokenClass)` to teach the lexer about\n * their keywords. The registry merges locale keywords, provider keywords,\n * phrase mappings, and unit names into an optimized TokenLookup structure\n * consumed by the Lexer.\n *\n * @example\n * registry.register({\n * tokenType: 'CARET',\n * keywords: {},\n * phrases: { 'to the power of': true, 'power of': true },\n * priority: 10,\n * description: 'Exponentiation operators (x^y)',\n * });\n */\nexport interface TokenClass {\n /** The token type string produced by the lexer (e.g., \"FUNC\", \"PI\", \"CARET\").\n * Must match a token type that a ParseletRegistry has a parselet for. */\n tokenType: string;\n\n /** Single-word keywords (case-insensitive). The lexer lowercases input\n * before lookup, so these should be lowercase. Example:\n * { sqrt: true, abs: true, sin: true, cos: true } for tokenType \"FUNC\" */\n keywords: Record<string, boolean>;\n\n /** Multi-word phrases (case-insensitive). Matched by the built-in PhraseMatcher\n * via the phrase trie. Example:\n * { \"to the power of\": true, \"power of\": true } for tokenType \"CARET\" */\n phrases?: Record<string, boolean>;\n\n /** Priority for conflict resolution. When two TokenClasses register\n * the same keyword, the higher-priority class wins. Locale keywords\n * have priority 0 (set via setLocale). Providers should use\n * priority >= 10 to override locale defaults. Default: 0 */\n priority?: number;\n\n /** Human-readable description for debugging and introspection */\n description?: string;\n}\n\n// ── Phrase Trie ──────────────────────────────────────────────────────────────\n\n/** Trie node for multi-word phrase matching. */\nexport interface PhraseNode {\n /** Complete phrase token type (null = intermediate node) */\n type?: string;\n children: Map<string, PhraseNode>;\n}\n\n// ── TokenLookup, Optimized lookup structure for the Lexer ──────────────────\n\n/**\n * The optimized lookup structure built by TokenClassRegistry.build().\n * Consumed by the Lexer for O(1) keyword → token type lookups and\n * O(word-count) phrase matching.\n */\nexport interface TokenLookup {\n /** Lowercase keyword → token type. O(1) Map lookup. */\n keywordToType: Map<string, string>;\n\n /** Phrase trie for multi-word matching. Root node with children maps.\n * Null if no phrases registered. */\n phraseTrie: PhraseNode | null;\n\n /** Set of lowercase first-words of all registered phrases.\n * Used by the lexer to emit IDENT (not a phrase keyword) for words\n * that start multi-word phrases, deferring to the PhraseMatcher.\n *\n * Example: \"to\" is in phraseStartWords because \"to the power of\" is a phrase.\n * When the lexer sees \"to\", it emits IDENT and lets the phrase matcher\n * combine \"to the power of\" into a single CARET token.\n *\n * This prevents plugins from accidentally overriding phrase-start words.\n */\n phraseStartWords: Set<string>;\n\n /** Case-sensitive unit names for UNIT fallback after keyword lookup fails. */\n unitNames: ReadonlySet<string>;\n}\n\n// ── TokenClassRegistry ───────────────────────────────────────────────────────\n\n/**\n * Central registry for keyword→token-type mappings.\n *\n * Providers register TokenClasses; locales provide keyword maps;\n * units provide a name set. `build()` merges all sources into an\n * optimized TokenLookup consumed by the Lexer.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0)\n * 2. Provider keywords (sorted by priority ascending, higher priority wins)\n *\n * Unit names are stored separately (checked AFTER keyword lookup fails).\n * Phrases are stored in a trie for O(phrase-length) matching.\n */\nexport class TokenClassRegistry {\n private classes: TokenClass[] = [];\n private localeKeywordMap: Record<string, string> | null = null;\n private localePhraseMap: Record<string, string> | null = null;\n private unitNames: ReadonlySet<string> | null = null;\n\n /**\n * Register a provider's TokenClass. Must be called BEFORE build().\n * Can be called multiple times to add more entries.\n *\n * Built-in token types CANNOT be overridden, throws a EngineError\n * if the TokenClass attempts to register a keyword that conflicts\n * with an already-registered token type.\n */\n register(tokenClass: TokenClass): void {\n this.classes.push(tokenClass);\n }\n\n /**\n * Unregister all TokenClasses for a given token type.\n * Useful for plugin unload. Requires rebuild() to take effect.\n */\n unregister(tokenType: string): void {\n this.classes = this.classes.filter(c => c.tokenType !== tokenType);\n }\n\n /**\n * Set the locale's keyword→type map and optional phrase map.\n * Called on locale change. Priority 0 (cannot override providers with higher priority).\n */\n setLocale(keywordMap: Record<string, string>, phraseMap?: Record<string, string>): void {\n this.localeKeywordMap = keywordMap;\n this.localePhraseMap = phraseMap ?? null;\n }\n\n /**\n * Set the unit name set. Called when unit list changes.\n * Units are stored separately (checked AFTER keyword lookup fails).\n *\n * Takes a ReadonlySet because the caller's set is derived from the\n * conversion tables and must not be mutated; this class only ever reads it.\n */\n setUnits(unitNames: ReadonlySet<string>): void {\n this.unitNames = unitNames;\n }\n\n /**\n * Build the optimized TokenLookup from all registered sources.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0)\n * 2. Provider classes (sorted by priority ascending)\n *\n * Returns a frozen TokenLookup that the Lexer consumes.\n * Call build() again after register()/setLocale()/setUnits() changes.\n */\n build(): TokenLookup {\n const keywordToType = new Map<string, string>();\n\n // Layer 1: Locale keywords (priority 0, lowest)\n if (this.localeKeywordMap) {\n for (const [keyword, tokenType] of Object.entries(this.localeKeywordMap)) {\n keywordToType.set(keyword.toLowerCase(), tokenType);\n }\n }\n\n // Layer 2: Provider keywords (sorted by priority ascending, higher wins)\n const sorted = [...this.classes].sort((a, b) => (a.priority ?? 0) - (b.priority ?? 0));\n for (const tc of sorted) {\n for (const keyword of Object.keys(tc.keywords)) {\n keywordToType.set(keyword.toLowerCase(), tc.tokenType);\n }\n }\n\n // Build phrase trie from locale + provider phrases\n const phraseTrie = this.buildPhraseTrie();\n\n // Build phraseStartWords set (first word of every phrase)\n const phraseStartWords = this.buildPhraseStartWords();\n\n return {\n keywordToType,\n phraseTrie,\n phraseStartWords,\n unitNames: this.unitNames ?? new Set(),\n };\n }\n\n // ── Private helpers ─────────────────────────────────────────────────────\n\n private buildPhraseTrie(): PhraseNode | null {\n const root: PhraseNode = { children: new Map() };\n\n // Layer 1: Locale phrases\n if (this.localePhraseMap) {\n for (const [phrase, tokenType] of Object.entries(this.localePhraseMap)) {\n this.insertPhrase(root, phrase.toLowerCase(), tokenType);\n }\n }\n\n // Layer 2: Provider phrases\n for (const tc of this.classes) {\n if (!tc.phrases) continue;\n for (const phrase of Object.keys(tc.phrases)) {\n this.insertPhrase(root, phrase.toLowerCase(), tc.tokenType);\n }\n }\n\n return root.children.size > 0 ? root : null;\n }\n\n private insertPhrase(root: PhraseNode, phrase: string, tokenType: string): void {\n const words = phrase.split(' ');\n let node = root;\n for (const word of words) {\n if (!node.children.has(word)) {\n node.children.set(word, { children: new Map() });\n }\n node = node.children.get(word)!;\n }\n // Only set type if not already set, first-registered (locale) wins\n if (!node.type) {\n node.type = tokenType;\n }\n }\n\n private buildPhraseStartWords(): Set<string> {\n const startWords = new Set<string>();\n\n // Locale phrase first words\n if (this.localePhraseMap) {\n for (const phrase of Object.keys(this.localePhraseMap)) {\n const first = phrase.split(' ')[0].toLowerCase();\n startWords.add(first);\n }\n }\n\n // Provider phrase first words\n for (const tc of this.classes) {\n if (!tc.phrases) continue;\n for (const phrase of Object.keys(tc.phrases)) {\n const first = phrase.split(' ')[0].toLowerCase();\n startWords.add(first);\n }\n }\n\n return startWords;\n }\n}","/**\n * tokenRegistration.ts, Bootstrap for building TokenLookup from locale, units, and phrases.\n *\n * This module centralizes the assembly of the TokenLookup consumed by ExpressionLexer.\n * It merges:\n * 1. Locale keywords (from ILocale.keywordMap)\n * 2. Built-in phrase patterns (to the power of, increase by, etc.)\n * 3. Known unit names (from units.ts)\n *\n * The resulting TokenLookup is frozen and passed to ExpressionLexer.configuredLookup\n * at construction time, enabling data-driven keyword/unit/phrase resolution.\n */\nimport { TokenClassRegistry } from '@solve-js/lexer/TokenClassRegistry';\nimport type { TokenLookup } from '@solve-js/lexer/TokenClassRegistry';\nimport { getLocale, type ILocale } from '@solve-js/constants/locales';\nimport { knownUnits } from '@solve-js/lexer/units';\n\n// ── Built-in phrase map ───────────────────────────────────────────────────\n// These are the multi-word expressions handled as compound tokens.\n// Matched by the PhraseMatcher (trie-based) in ExpressionLexer.tryMatchPhrase().\nconst BUILTIN_PHRASES: Record<string, string> = {\n 'to the power of': 'CARET',\n 'power of': 'CARET',\n 'increase by': 'INCREASE_BY',\n 'decrease by': 'DECREASE_BY',\n 'times by': 'TIMES_BY',\n 'multiply by': 'MULTIPLY_BY',\n 'multiplied by': 'MULTIPLY_BY',\n 'divide by': 'DIVIDE_BY',\n};\n\n/**\n * Build a TokenLookup from locale keywords, known units, and built-in phrases.\n *\n * Merge order (later overrides earlier):\n * 1. Locale keywords (priority 0, lowest, can be overridden by providers)\n * 2. Built-in phrases (via locale phraseMap)\n * 3. Known units (checked AFTER keyword lookup fails, via unitNames set)\n *\n * The resulting TokenLookup replaces the internal keyword map, unit set,\n * phrase trie, and phraseStartWords in ExpressionLexer when set via\n * ExpressionLexer.configuredLookup.\n *\n * @param localeCode - The locale code (e.g., \"en\", \"de\"). Defaults to \"en\".\n * @returns A frozen TokenLookup ready for consumption by the lexer.\n */\nexport function buildTokenLookup(localeCode = 'en'): TokenLookup {\n const locale: ILocale = getLocale(localeCode);\n const registry = new TokenClassRegistry();\n\n // Layer 1: Locale keywords (priority 0, lowest)\n registry.setLocale(locale.keywordMap, BUILTIN_PHRASES);\n\n // Layer 2: Known units (checked after keyword lookup)\n registry.setUnits(knownUnits);\n\n // Build and return the lookup\n return registry.build();\n}\n"]}
|