solve-engine 2.16.0 → 2.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{BytecodeBuilder-CRYfrFfq.d.cts → BytecodeBuilder-aqVa7Plx.d.cts} +10 -0
- package/dist/{BytecodeBuilder-CRYfrFfq.d.ts → BytecodeBuilder-aqVa7Plx.d.ts} +10 -0
- package/dist/{EngineError-Cv5q4Rbv.d.cts → EngineError-DTk7I7hZ.d.cts} +24 -0
- package/dist/{EngineError-Cv5q4Rbv.d.ts → EngineError-DTk7I7hZ.d.ts} +24 -0
- package/dist/{PackageCompatibility-CHQ4kDT8.d.ts → PackageCompatibility-BGXSFuL9.d.ts} +1 -1
- package/dist/{PackageCompatibility-BET0iFb9.d.cts → PackageCompatibility-BZCqaTOO.d.cts} +1 -1
- package/dist/{PackageRegistry-YmfowYr1.d.ts → PackageRegistry-BRoVOYzg.d.ts} +78 -5
- package/dist/{PackageRegistry-DzmG-Rvu.d.cts → PackageRegistry-rMGnTh7W.d.cts} +78 -5
- package/dist/{Parselet-CK0bNO1l.d.ts → Parselet-B-WyUtX4.d.ts} +5 -1
- package/dist/{Parselet-DOGmj6N6.d.cts → Parselet-Bkbp9CKD.d.cts} +5 -1
- package/dist/{ScopeManager-BtqiTVjG.d.cts → ScopeManager-bCYewWVt.d.cts} +2 -2
- package/dist/{ScopeManager-C6c1WmJD.d.ts → ScopeManager-gB9UengK.d.ts} +2 -2
- package/dist/{TokenNormalizer-BfTHg1qt.d.cts → TokenNormalizer-MXaKLJ_m.d.cts} +182 -0
- package/dist/{TokenNormalizer-d7F1KQFs.d.ts → TokenNormalizer-OTPS0Otq.d.ts} +182 -0
- package/dist/{VMCheckpoints-DJdvpliw.d.ts → VMCheckpoints-BDRY1Kx8.d.ts} +2 -2
- package/dist/{VMCheckpoints-BFDlaGse.d.cts → VMCheckpoints-DGar9Yg8.d.cts} +2 -2
- package/dist/{WorkerError-FH7KXX4_.d.cts → WorkerError-DGBNM3gA.d.cts} +1 -1
- package/dist/{WorkerError-J3v3ix_z.d.ts → WorkerError-DpZgKRks.d.ts} +1 -1
- package/dist/{chunk-UZSPBFZN.js → chunk-2DPKJ2SM.js} +2 -2
- package/dist/{chunk-UZSPBFZN.js.map → chunk-2DPKJ2SM.js.map} +1 -1
- package/dist/{chunk-O3ALJLYC.js → chunk-3GR46UOX.js} +2 -2
- package/dist/{chunk-O3ALJLYC.js.map → chunk-3GR46UOX.js.map} +1 -1
- package/dist/{chunk-PREIL3IB.cjs → chunk-3SSMOVWB.cjs} +2 -2
- package/dist/{chunk-PREIL3IB.cjs.map → chunk-3SSMOVWB.cjs.map} +1 -1
- package/dist/{chunk-VOTTSM4L.cjs → chunk-43FUDMIP.cjs} +3 -3
- package/dist/{chunk-VOTTSM4L.cjs.map → chunk-43FUDMIP.cjs.map} +1 -1
- package/dist/{chunk-XK23K3EJ.js → chunk-4ETS224G.js} +2 -2
- package/dist/{chunk-XK23K3EJ.js.map → chunk-4ETS224G.js.map} +1 -1
- package/dist/chunk-4XXUXRST.js +2 -0
- package/dist/chunk-4XXUXRST.js.map +1 -0
- package/dist/{chunk-55B4PSPM.js → chunk-6ERKCASV.js} +2 -2
- package/dist/{chunk-55B4PSPM.js.map → chunk-6ERKCASV.js.map} +1 -1
- package/dist/{chunk-GWCBPITD.cjs → chunk-6YM67DWH.cjs} +3 -3
- package/dist/{chunk-GWCBPITD.cjs.map → chunk-6YM67DWH.cjs.map} +1 -1
- package/dist/{chunk-6U66HQKQ.cjs → chunk-75JLWK4L.cjs} +2 -2
- package/dist/{chunk-6U66HQKQ.cjs.map → chunk-75JLWK4L.cjs.map} +1 -1
- package/dist/chunk-AYRVYGQO.js +3 -0
- package/dist/chunk-AYRVYGQO.js.map +1 -0
- package/dist/{chunk-VJP4GGCW.cjs → chunk-BNCBB4H5.cjs} +2 -2
- package/dist/{chunk-VJP4GGCW.cjs.map → chunk-BNCBB4H5.cjs.map} +1 -1
- package/dist/chunk-CCTO74QZ.js +5 -0
- package/dist/chunk-CCTO74QZ.js.map +1 -0
- package/dist/chunk-CO2BI6WL.cjs +3 -0
- package/dist/chunk-CO2BI6WL.cjs.map +1 -0
- package/dist/chunk-EN3CDYOS.cjs +2 -0
- package/dist/chunk-EN3CDYOS.cjs.map +1 -0
- package/dist/{chunk-GWJJRH32.js → chunk-EQDT42JG.js} +3 -3
- package/dist/{chunk-GWJJRH32.js.map → chunk-EQDT42JG.js.map} +1 -1
- package/dist/chunk-G2MFTERZ.cjs +2 -0
- package/dist/chunk-G2MFTERZ.cjs.map +1 -0
- package/dist/chunk-G63TMUL2.js +2 -0
- package/dist/chunk-G63TMUL2.js.map +1 -0
- package/dist/{chunk-5EX2FCM3.js → chunk-GE5VBDFX.js} +3 -3
- package/dist/{chunk-5EX2FCM3.js.map → chunk-GE5VBDFX.js.map} +1 -1
- package/dist/{chunk-UUAFDZK6.cjs → chunk-H3JXNH7X.cjs} +3 -3
- package/dist/{chunk-UUAFDZK6.cjs.map → chunk-H3JXNH7X.cjs.map} +1 -1
- package/dist/{chunk-2XZCLDHJ.cjs → chunk-J7ABVGZH.cjs} +2 -2
- package/dist/{chunk-2XZCLDHJ.cjs.map → chunk-J7ABVGZH.cjs.map} +1 -1
- package/dist/chunk-JVMINMAB.js +2 -0
- package/dist/chunk-JVMINMAB.js.map +1 -0
- package/dist/chunk-JYLNQOPU.cjs +2 -0
- package/dist/chunk-JYLNQOPU.cjs.map +1 -0
- package/dist/{chunk-JOIQFDZQ.js → chunk-KEA5HRL3.js} +2 -2
- package/dist/{chunk-JOIQFDZQ.js.map → chunk-KEA5HRL3.js.map} +1 -1
- package/dist/chunk-KWA257PC.cjs +3 -0
- package/dist/chunk-KWA257PC.cjs.map +1 -0
- package/dist/{chunk-TVE2DNPU.js → chunk-LJFS3XHW.js} +2 -2
- package/dist/{chunk-TVE2DNPU.js.map → chunk-LJFS3XHW.js.map} +1 -1
- package/dist/{chunk-C5MP74ER.js → chunk-MRRMIBHE.js} +2 -2
- package/dist/{chunk-C5MP74ER.js.map → chunk-MRRMIBHE.js.map} +1 -1
- package/dist/chunk-N64ZK6CL.cjs +2 -0
- package/dist/{chunk-EEVUKZ5B.cjs.map → chunk-N64ZK6CL.cjs.map} +1 -1
- package/dist/chunk-O3BDXOQJ.js +2 -0
- package/dist/chunk-O3BDXOQJ.js.map +1 -0
- package/dist/chunk-PHNSXQ4L.cjs +2 -0
- package/dist/chunk-PHNSXQ4L.cjs.map +1 -0
- package/dist/chunk-PIZQIQVM.js +3 -0
- package/dist/chunk-PIZQIQVM.js.map +1 -0
- package/dist/{chunk-IPOU3JTA.js → chunk-QTSDIDAS.js} +3 -3
- package/dist/{chunk-IPOU3JTA.js.map → chunk-QTSDIDAS.js.map} +1 -1
- package/dist/{chunk-P4ETODMP.js → chunk-RLTH2H3U.js} +2 -2
- package/dist/{chunk-P4ETODMP.js.map → chunk-RLTH2H3U.js.map} +1 -1
- package/dist/{chunk-WOIJPV7Z.js → chunk-SEOIKHYK.js} +2 -2
- package/dist/{chunk-WOIJPV7Z.js.map → chunk-SEOIKHYK.js.map} +1 -1
- package/dist/{chunk-XZYZS4BP.cjs → chunk-TCHKFSZD.cjs} +2 -2
- package/dist/{chunk-XZYZS4BP.cjs.map → chunk-TCHKFSZD.cjs.map} +1 -1
- package/dist/{chunk-W5ELYJ4Z.cjs → chunk-TL545PXW.cjs} +2 -2
- package/dist/{chunk-W5ELYJ4Z.cjs.map → chunk-TL545PXW.cjs.map} +1 -1
- package/dist/{chunk-FAGFPVYK.cjs → chunk-TRDOYAKJ.cjs} +2 -2
- package/dist/{chunk-FAGFPVYK.cjs.map → chunk-TRDOYAKJ.cjs.map} +1 -1
- package/dist/chunk-X7DFJS7R.cjs +5 -0
- package/dist/chunk-X7DFJS7R.cjs.map +1 -0
- package/dist/{chunk-EXCFCHAT.cjs → chunk-XPJGDGJF.cjs} +2 -2
- package/dist/{chunk-EXCFCHAT.cjs.map → chunk-XPJGDGJF.cjs.map} +1 -1
- package/dist/constants.cjs +1 -1
- package/dist/constants.js +1 -1
- package/dist/engine.cjs +1 -1
- package/dist/engine.d.cts +8 -8
- package/dist/engine.d.ts +8 -8
- package/dist/engine.js +1 -1
- package/dist/errors.cjs +1 -1
- package/dist/errors.d.cts +3 -3
- package/dist/errors.d.ts +3 -3
- package/dist/errors.js +1 -1
- package/dist/format.cjs +1 -1
- package/dist/format.js +1 -1
- package/dist/index.cjs +1 -1
- package/dist/index.d.cts +8 -8
- package/dist/index.d.ts +8 -8
- package/dist/index.js +1 -1
- package/dist/language.d.cts +7 -7
- package/dist/language.d.ts +7 -7
- package/dist/lexer.cjs +1 -1
- package/dist/lexer.js +1 -1
- package/dist/normalizer.cjs +1 -1
- package/dist/normalizer.d.cts +2 -2
- package/dist/normalizer.d.ts +2 -2
- package/dist/normalizer.js +1 -1
- package/dist/packages.cjs +1 -1
- package/dist/packages.d.cts +6 -6
- package/dist/packages.d.ts +6 -6
- package/dist/packages.js +1 -1
- package/dist/parser.cjs +1 -1
- package/dist/parser.d.cts +2 -2
- package/dist/parser.d.ts +2 -2
- package/dist/parser.js +1 -1
- package/dist/resolvers.d.cts +1 -1
- package/dist/resolvers.d.ts +1 -1
- package/dist/testing.cjs +2 -2
- package/dist/testing.d.cts +7 -7
- package/dist/testing.d.ts +7 -7
- package/dist/testing.js +1 -1
- package/dist/uom.cjs +1 -1
- package/dist/uom.d.cts +1 -1
- package/dist/uom.d.ts +1 -1
- package/dist/uom.js +1 -1
- package/dist/vm.cjs +1 -1
- package/dist/vm.d.cts +5 -5
- package/dist/vm.d.ts +5 -5
- package/dist/vm.js +1 -1
- package/dist/worker.cjs +2 -2
- package/dist/worker.d.cts +7 -7
- package/dist/worker.d.ts +7 -7
- package/dist/worker.js +1 -1
- package/package.json +1 -1
- package/dist/chunk-35RNQ2OQ.js +0 -5
- package/dist/chunk-35RNQ2OQ.js.map +0 -1
- package/dist/chunk-5CWVWMIY.js +0 -2
- package/dist/chunk-5CWVWMIY.js.map +0 -1
- package/dist/chunk-7EMZOBGW.cjs +0 -3
- package/dist/chunk-7EMZOBGW.cjs.map +0 -1
- package/dist/chunk-CUDPOWXA.js +0 -2
- package/dist/chunk-CUDPOWXA.js.map +0 -1
- package/dist/chunk-EEVUKZ5B.cjs +0 -2
- package/dist/chunk-ERMNQ5QJ.js +0 -2
- package/dist/chunk-ERMNQ5QJ.js.map +0 -1
- package/dist/chunk-FVQWN5HZ.cjs +0 -2
- package/dist/chunk-FVQWN5HZ.cjs.map +0 -1
- package/dist/chunk-IMXSVHQK.cjs +0 -2
- package/dist/chunk-IMXSVHQK.cjs.map +0 -1
- package/dist/chunk-LZSSAH5D.cjs +0 -2
- package/dist/chunk-LZSSAH5D.cjs.map +0 -1
- package/dist/chunk-NAZ6PGLS.cjs +0 -3
- package/dist/chunk-NAZ6PGLS.cjs.map +0 -1
- package/dist/chunk-Q3PSTNSY.js +0 -2
- package/dist/chunk-Q3PSTNSY.js.map +0 -1
- package/dist/chunk-SPOECHDG.cjs +0 -5
- package/dist/chunk-SPOECHDG.cjs.map +0 -1
- package/dist/chunk-W64AWQCX.js +0 -3
- package/dist/chunk-W64AWQCX.js.map +0 -1
- package/dist/chunk-YAZ4DUJJ.js +0 -3
- package/dist/chunk-YAZ4DUJJ.js.map +0 -1
- package/dist/chunk-Z4BXU65Y.cjs +0 -2
- package/dist/chunk-Z4BXU65Y.cjs.map +0 -1
|
@@ -111,6 +111,16 @@ interface BytecodeProgram {
|
|
|
111
111
|
opcodes: Uint8Array;
|
|
112
112
|
numbers: Float64Array;
|
|
113
113
|
strings: string[];
|
|
114
|
+
/**
|
|
115
|
+
* Numeric constants by opcode position, restored from a snapshot.
|
|
116
|
+
*
|
|
117
|
+
* Nothing in the compile path writes this: numbers are emitted inline into
|
|
118
|
+
* {@link numbers} instead. `build()` used to attach an empty Map to every
|
|
119
|
+
* program anyway, which was one allocation per compiled expression, around a
|
|
120
|
+
* tenth of parse-and-compile time, for a collection that was never read.
|
|
121
|
+
* It is left off now, and only {@link EngineSnapshot} sets it when restoring
|
|
122
|
+
* a program that carried one; every reader already guards on its absence.
|
|
123
|
+
*/
|
|
114
124
|
constants?: Map<number, number>;
|
|
115
125
|
/**
|
|
116
126
|
* Whether the program contains any async opcodes (CALL_PLUGIN, etc.).
|
|
@@ -111,6 +111,16 @@ interface BytecodeProgram {
|
|
|
111
111
|
opcodes: Uint8Array;
|
|
112
112
|
numbers: Float64Array;
|
|
113
113
|
strings: string[];
|
|
114
|
+
/**
|
|
115
|
+
* Numeric constants by opcode position, restored from a snapshot.
|
|
116
|
+
*
|
|
117
|
+
* Nothing in the compile path writes this: numbers are emitted inline into
|
|
118
|
+
* {@link numbers} instead. `build()` used to attach an empty Map to every
|
|
119
|
+
* program anyway, which was one allocation per compiled expression, around a
|
|
120
|
+
* tenth of parse-and-compile time, for a collection that was never read.
|
|
121
|
+
* It is left off now, and only {@link EngineSnapshot} sets it when restoring
|
|
122
|
+
* a program that carried one; every reader already guards on its absence.
|
|
123
|
+
*/
|
|
114
124
|
constants?: Map<number, number>;
|
|
115
125
|
/**
|
|
116
126
|
* Whether the program contains any async opcodes (CALL_PLUGIN, etc.).
|
|
@@ -313,6 +313,30 @@ declare class EngineError extends Error {
|
|
|
313
313
|
* interop with no build-config change needed.
|
|
314
314
|
*/
|
|
315
315
|
readonly cause?: unknown;
|
|
316
|
+
/**
|
|
317
|
+
* Whether a recoverable error captures a JavaScript stack trace.
|
|
318
|
+
*
|
|
319
|
+
* Off, because a recoverable EngineError is a value rather than a fault. A
|
|
320
|
+
* line of prose in a notepad is not an expression, so parsing it fails, and
|
|
321
|
+
* that failure is the answer for that line rather than a bug to debug. The
|
|
322
|
+
* engine builds one such error per non-expression line.
|
|
323
|
+
*
|
|
324
|
+
* Capturing a stack is not cheap, and its cost grows with how deep the stack
|
|
325
|
+
* is when it happens. Measured through `parseDocument`, where the throw site
|
|
326
|
+
* sits about a dozen frames down, each capture cost around 62 microseconds
|
|
327
|
+
* and a 250-line document built 74 of them: a CPU profile put the
|
|
328
|
+
* constructor at 46% of the whole pipeline, more than lexing, normalising,
|
|
329
|
+
* parsing and executing put together.
|
|
330
|
+
*
|
|
331
|
+
* Turn it on to debug where a recoverable error is raised from. Errors that
|
|
332
|
+
* are NOT recoverable always capture, since those are the genuine faults.
|
|
333
|
+
*
|
|
334
|
+
* @example
|
|
335
|
+
* ```ts
|
|
336
|
+
* EngineError.captureRecoverableStacks = true;
|
|
337
|
+
* ```
|
|
338
|
+
*/
|
|
339
|
+
static captureRecoverableStacks: boolean;
|
|
316
340
|
constructor(category: ErrorCategory, init: EngineErrorInit);
|
|
317
341
|
/** `!recoverable`. See `EngineErrorInit.recoverable`'s doc comment for what this actually gates (message framing/telemetry, not whether evaluation continues). */
|
|
318
342
|
isFatal(): boolean;
|
|
@@ -313,6 +313,30 @@ declare class EngineError extends Error {
|
|
|
313
313
|
* interop with no build-config change needed.
|
|
314
314
|
*/
|
|
315
315
|
readonly cause?: unknown;
|
|
316
|
+
/**
|
|
317
|
+
* Whether a recoverable error captures a JavaScript stack trace.
|
|
318
|
+
*
|
|
319
|
+
* Off, because a recoverable EngineError is a value rather than a fault. A
|
|
320
|
+
* line of prose in a notepad is not an expression, so parsing it fails, and
|
|
321
|
+
* that failure is the answer for that line rather than a bug to debug. The
|
|
322
|
+
* engine builds one such error per non-expression line.
|
|
323
|
+
*
|
|
324
|
+
* Capturing a stack is not cheap, and its cost grows with how deep the stack
|
|
325
|
+
* is when it happens. Measured through `parseDocument`, where the throw site
|
|
326
|
+
* sits about a dozen frames down, each capture cost around 62 microseconds
|
|
327
|
+
* and a 250-line document built 74 of them: a CPU profile put the
|
|
328
|
+
* constructor at 46% of the whole pipeline, more than lexing, normalising,
|
|
329
|
+
* parsing and executing put together.
|
|
330
|
+
*
|
|
331
|
+
* Turn it on to debug where a recoverable error is raised from. Errors that
|
|
332
|
+
* are NOT recoverable always capture, since those are the genuine faults.
|
|
333
|
+
*
|
|
334
|
+
* @example
|
|
335
|
+
* ```ts
|
|
336
|
+
* EngineError.captureRecoverableStacks = true;
|
|
337
|
+
* ```
|
|
338
|
+
*/
|
|
339
|
+
static captureRecoverableStacks: boolean;
|
|
316
340
|
constructor(category: ErrorCategory, init: EngineErrorInit);
|
|
317
341
|
/** `!recoverable`. See `EngineErrorInit.recoverable`'s doc comment for what this actually gates (message framing/telemetry, not whether evaluation continues). */
|
|
318
342
|
isFatal(): boolean;
|
|
@@ -1,12 +1,12 @@
|
|
|
1
|
-
import { a as PrecedenceParser, b as PrefixParselet, I as InfixParselet } from './Parselet-
|
|
1
|
+
import { a as PrecedenceParser, b as PrefixParselet, I as InfixParselet } from './Parselet-B-WyUtX4.js';
|
|
2
2
|
import { V as Value, b as ValueType } from './Value-DCTqTSeP.js';
|
|
3
3
|
import { M as MarkdownLineType, L as Lexer, e as TokenCategory, c as LexerVocabulary } from './Lexer-CqagTewQ.js';
|
|
4
4
|
import { IAsyncResolver } from './resolvers.js';
|
|
5
|
-
import { T as TokenFusion, c as TokenNormalizer, a as NormalizerRule } from './TokenNormalizer-
|
|
6
|
-
import { D as DependencyGraph, V as VM, a as DagSnapshot, b as EngineContext, S as ScopeManager, P as PluginFunctionHandler } from './ScopeManager-
|
|
7
|
-
import { a as BytecodeProgram } from './BytecodeBuilder-
|
|
5
|
+
import { T as TokenFusion, c as TokenNormalizer, a as NormalizerRule } from './TokenNormalizer-OTPS0Otq.js';
|
|
6
|
+
import { D as DependencyGraph, V as VM, a as DagSnapshot, b as EngineContext, S as ScopeManager, P as PluginFunctionHandler } from './ScopeManager-gB9UengK.js';
|
|
7
|
+
import { a as BytecodeProgram } from './BytecodeBuilder-aqVa7Plx.js';
|
|
8
8
|
import { QueryClient } from '@tanstack/query-core';
|
|
9
|
-
import { E as EngineError } from './EngineError-
|
|
9
|
+
import { E as EngineError } from './EngineError-DTk7I7hZ.js';
|
|
10
10
|
import { a as DiagnosticReportJSON, D as DiagnosticPipeline } from './pipeline-QIT4iD8f.js';
|
|
11
11
|
import { T as Token } from './Token-CbP_OutD.js';
|
|
12
12
|
import { E as EngineConfigOverride, a as EngineConfig } from './Configuration-BGQn-cJ8.js';
|
|
@@ -1191,6 +1191,32 @@ interface NormalizerOutput {
|
|
|
1191
1191
|
}[];
|
|
1192
1192
|
/** Post-normalization tokens ready for parsing */
|
|
1193
1193
|
tokens: Token[];
|
|
1194
|
+
/**
|
|
1195
|
+
* Every registered rule with the shape it declared, in priority order.
|
|
1196
|
+
*
|
|
1197
|
+
* A rule that declares a shape is tried only where that shape can match; one
|
|
1198
|
+
* that declares none is tried at every position of every line. That
|
|
1199
|
+
* distinction is invisible from outside the engine and is the difference
|
|
1200
|
+
* between a package costing the documents that use it and costing all of
|
|
1201
|
+
* them, so the playground draws it.
|
|
1202
|
+
*/
|
|
1203
|
+
ruleShapes?: {
|
|
1204
|
+
name: string;
|
|
1205
|
+
priority: number;
|
|
1206
|
+
shape: readonly {
|
|
1207
|
+
types?: readonly string[];
|
|
1208
|
+
values?: readonly string[];
|
|
1209
|
+
}[];
|
|
1210
|
+
unshapedReason?: string;
|
|
1211
|
+
indexedSlots: number;
|
|
1212
|
+
}[];
|
|
1213
|
+
/**
|
|
1214
|
+
* How many rules could fire at each position of the normalised stream.
|
|
1215
|
+
*
|
|
1216
|
+
* One entry per token. Mostly zeroes, which is the point of the index: a
|
|
1217
|
+
* position where nothing can match is rejected without calling a rule.
|
|
1218
|
+
*/
|
|
1219
|
+
candidatesPerPosition?: number[];
|
|
1194
1220
|
/**
|
|
1195
1221
|
* All registered phrase → tokenType mappings from the PhraseTrie.
|
|
1196
1222
|
* Populated by the engine at diagnostic stage build time so the
|
|
@@ -1478,6 +1504,18 @@ interface EngineOptions {
|
|
|
1478
1504
|
config?: EngineConfigOverride;
|
|
1479
1505
|
/** Turn on the diagnostic pipeline (per-stage timing and detail). Defaults to `false`. */
|
|
1480
1506
|
diagnostics?: boolean;
|
|
1507
|
+
/**
|
|
1508
|
+
* Run a few throwaway expressions through the pipeline at construction, so
|
|
1509
|
+
* the first real one is not the one that pays for JIT warmup.
|
|
1510
|
+
*
|
|
1511
|
+
* Off by default, because it moves cost rather than removing it: a process
|
|
1512
|
+
* that evaluates one expression and exits pays for warming paths it never
|
|
1513
|
+
* reuses. Turn it on for anything interactive, where the first keystroke is
|
|
1514
|
+
* the one a person notices. See {@link ExpressionEngine.warmUp}.
|
|
1515
|
+
*
|
|
1516
|
+
* @default false
|
|
1517
|
+
*/
|
|
1518
|
+
warmup?: boolean;
|
|
1481
1519
|
}
|
|
1482
1520
|
/**
|
|
1483
1521
|
* Core expression evaluation engine, the top-level orchestrator.
|
|
@@ -1650,6 +1688,41 @@ declare class ExpressionEngine {
|
|
|
1650
1688
|
* had set, rather than assuming it was `null`.
|
|
1651
1689
|
*/
|
|
1652
1690
|
getDocumentModel(): DocumentModel | null;
|
|
1691
|
+
/**
|
|
1692
|
+
* A handful of expressions run through the pipeline to warm it, chosen to
|
|
1693
|
+
* cover the shapes the hot paths specialise on rather than to be
|
|
1694
|
+
* interesting.
|
|
1695
|
+
*
|
|
1696
|
+
* V8 runs a function interpreted until it has been called enough times to
|
|
1697
|
+
* be worth optimising, and it specialises on the types it has actually
|
|
1698
|
+
* seen. So the point is breadth, not volume: a number, a unit, a phrase, a
|
|
1699
|
+
* function call, a comparison and a string each drive a different branch of
|
|
1700
|
+
* the lexer, the normalizer and the VM. Warming only with `1 + 1` would
|
|
1701
|
+
* optimise those functions for integers and then deoptimise the moment a
|
|
1702
|
+
* real document mentioned kilograms, which is worse than not warming at all.
|
|
1703
|
+
*/
|
|
1704
|
+
private static readonly WARMUP_EXPRESSIONS;
|
|
1705
|
+
/**
|
|
1706
|
+
* Run the pipeline over a few throwaway expressions so the first real one
|
|
1707
|
+
* is not the one that pays for JIT warmup.
|
|
1708
|
+
*
|
|
1709
|
+
* Nothing here reaches the engine's state: no variable is defined, no line
|
|
1710
|
+
* is registered, no bytecode is cached and no async resolver is consulted.
|
|
1711
|
+
* It lexes, normalises, parses, compiles and executes into a scratch
|
|
1712
|
+
* builder and then drops the result, which is enough to move the hot
|
|
1713
|
+
* functions past the interpreter and to build the normalizer's rule index.
|
|
1714
|
+
*
|
|
1715
|
+
* The cost lands on construction instead. That is the right trade for an
|
|
1716
|
+
* editor, where the first keystroke is the one a person notices, and the
|
|
1717
|
+
* wrong one for a process that evaluates a single expression and exits,
|
|
1718
|
+
* which is why {@link EngineOptions.warmup} exists rather than this being
|
|
1719
|
+
* unconditional.
|
|
1720
|
+
*
|
|
1721
|
+
* Failures are swallowed on purpose: a warmup expression that stops parsing
|
|
1722
|
+
* because a package changed is a warmup that did less good, not a reason to
|
|
1723
|
+
* refuse to construct an engine.
|
|
1724
|
+
*/
|
|
1725
|
+
warmUp(): void;
|
|
1653
1726
|
/**
|
|
1654
1727
|
* Build the {@link LineExecutionContext} passed to `executeBytecode()`
|
|
1655
1728
|
* for a given line. `lineNumber = -1` (the existing sentinel
|
|
@@ -1,12 +1,12 @@
|
|
|
1
|
-
import { a as PrecedenceParser, b as PrefixParselet, I as InfixParselet } from './Parselet-
|
|
1
|
+
import { a as PrecedenceParser, b as PrefixParselet, I as InfixParselet } from './Parselet-Bkbp9CKD.cjs';
|
|
2
2
|
import { V as Value, b as ValueType } from './Value-DCTqTSeP.cjs';
|
|
3
3
|
import { M as MarkdownLineType, L as Lexer, e as TokenCategory, c as LexerVocabulary } from './Lexer-CCHFDcgP.cjs';
|
|
4
4
|
import { IAsyncResolver } from './resolvers.cjs';
|
|
5
|
-
import { T as TokenFusion, c as TokenNormalizer, a as NormalizerRule } from './TokenNormalizer-
|
|
6
|
-
import { D as DependencyGraph, V as VM, a as DagSnapshot, b as EngineContext, S as ScopeManager, P as PluginFunctionHandler } from './ScopeManager-
|
|
7
|
-
import { a as BytecodeProgram } from './BytecodeBuilder-
|
|
5
|
+
import { T as TokenFusion, c as TokenNormalizer, a as NormalizerRule } from './TokenNormalizer-MXaKLJ_m.cjs';
|
|
6
|
+
import { D as DependencyGraph, V as VM, a as DagSnapshot, b as EngineContext, S as ScopeManager, P as PluginFunctionHandler } from './ScopeManager-bCYewWVt.cjs';
|
|
7
|
+
import { a as BytecodeProgram } from './BytecodeBuilder-aqVa7Plx.cjs';
|
|
8
8
|
import { QueryClient } from '@tanstack/query-core';
|
|
9
|
-
import { E as EngineError } from './EngineError-
|
|
9
|
+
import { E as EngineError } from './EngineError-DTk7I7hZ.cjs';
|
|
10
10
|
import { a as DiagnosticReportJSON, D as DiagnosticPipeline } from './pipeline-B4wf3M1h.cjs';
|
|
11
11
|
import { T as Token } from './Token-CbP_OutD.cjs';
|
|
12
12
|
import { E as EngineConfigOverride, a as EngineConfig } from './Configuration-BGQn-cJ8.cjs';
|
|
@@ -1191,6 +1191,32 @@ interface NormalizerOutput {
|
|
|
1191
1191
|
}[];
|
|
1192
1192
|
/** Post-normalization tokens ready for parsing */
|
|
1193
1193
|
tokens: Token[];
|
|
1194
|
+
/**
|
|
1195
|
+
* Every registered rule with the shape it declared, in priority order.
|
|
1196
|
+
*
|
|
1197
|
+
* A rule that declares a shape is tried only where that shape can match; one
|
|
1198
|
+
* that declares none is tried at every position of every line. That
|
|
1199
|
+
* distinction is invisible from outside the engine and is the difference
|
|
1200
|
+
* between a package costing the documents that use it and costing all of
|
|
1201
|
+
* them, so the playground draws it.
|
|
1202
|
+
*/
|
|
1203
|
+
ruleShapes?: {
|
|
1204
|
+
name: string;
|
|
1205
|
+
priority: number;
|
|
1206
|
+
shape: readonly {
|
|
1207
|
+
types?: readonly string[];
|
|
1208
|
+
values?: readonly string[];
|
|
1209
|
+
}[];
|
|
1210
|
+
unshapedReason?: string;
|
|
1211
|
+
indexedSlots: number;
|
|
1212
|
+
}[];
|
|
1213
|
+
/**
|
|
1214
|
+
* How many rules could fire at each position of the normalised stream.
|
|
1215
|
+
*
|
|
1216
|
+
* One entry per token. Mostly zeroes, which is the point of the index: a
|
|
1217
|
+
* position where nothing can match is rejected without calling a rule.
|
|
1218
|
+
*/
|
|
1219
|
+
candidatesPerPosition?: number[];
|
|
1194
1220
|
/**
|
|
1195
1221
|
* All registered phrase → tokenType mappings from the PhraseTrie.
|
|
1196
1222
|
* Populated by the engine at diagnostic stage build time so the
|
|
@@ -1478,6 +1504,18 @@ interface EngineOptions {
|
|
|
1478
1504
|
config?: EngineConfigOverride;
|
|
1479
1505
|
/** Turn on the diagnostic pipeline (per-stage timing and detail). Defaults to `false`. */
|
|
1480
1506
|
diagnostics?: boolean;
|
|
1507
|
+
/**
|
|
1508
|
+
* Run a few throwaway expressions through the pipeline at construction, so
|
|
1509
|
+
* the first real one is not the one that pays for JIT warmup.
|
|
1510
|
+
*
|
|
1511
|
+
* Off by default, because it moves cost rather than removing it: a process
|
|
1512
|
+
* that evaluates one expression and exits pays for warming paths it never
|
|
1513
|
+
* reuses. Turn it on for anything interactive, where the first keystroke is
|
|
1514
|
+
* the one a person notices. See {@link ExpressionEngine.warmUp}.
|
|
1515
|
+
*
|
|
1516
|
+
* @default false
|
|
1517
|
+
*/
|
|
1518
|
+
warmup?: boolean;
|
|
1481
1519
|
}
|
|
1482
1520
|
/**
|
|
1483
1521
|
* Core expression evaluation engine, the top-level orchestrator.
|
|
@@ -1650,6 +1688,41 @@ declare class ExpressionEngine {
|
|
|
1650
1688
|
* had set, rather than assuming it was `null`.
|
|
1651
1689
|
*/
|
|
1652
1690
|
getDocumentModel(): DocumentModel | null;
|
|
1691
|
+
/**
|
|
1692
|
+
* A handful of expressions run through the pipeline to warm it, chosen to
|
|
1693
|
+
* cover the shapes the hot paths specialise on rather than to be
|
|
1694
|
+
* interesting.
|
|
1695
|
+
*
|
|
1696
|
+
* V8 runs a function interpreted until it has been called enough times to
|
|
1697
|
+
* be worth optimising, and it specialises on the types it has actually
|
|
1698
|
+
* seen. So the point is breadth, not volume: a number, a unit, a phrase, a
|
|
1699
|
+
* function call, a comparison and a string each drive a different branch of
|
|
1700
|
+
* the lexer, the normalizer and the VM. Warming only with `1 + 1` would
|
|
1701
|
+
* optimise those functions for integers and then deoptimise the moment a
|
|
1702
|
+
* real document mentioned kilograms, which is worse than not warming at all.
|
|
1703
|
+
*/
|
|
1704
|
+
private static readonly WARMUP_EXPRESSIONS;
|
|
1705
|
+
/**
|
|
1706
|
+
* Run the pipeline over a few throwaway expressions so the first real one
|
|
1707
|
+
* is not the one that pays for JIT warmup.
|
|
1708
|
+
*
|
|
1709
|
+
* Nothing here reaches the engine's state: no variable is defined, no line
|
|
1710
|
+
* is registered, no bytecode is cached and no async resolver is consulted.
|
|
1711
|
+
* It lexes, normalises, parses, compiles and executes into a scratch
|
|
1712
|
+
* builder and then drops the result, which is enough to move the hot
|
|
1713
|
+
* functions past the interpreter and to build the normalizer's rule index.
|
|
1714
|
+
*
|
|
1715
|
+
* The cost lands on construction instead. That is the right trade for an
|
|
1716
|
+
* editor, where the first keystroke is the one a person notices, and the
|
|
1717
|
+
* wrong one for a process that evaluates a single expression and exits,
|
|
1718
|
+
* which is why {@link EngineOptions.warmup} exists rather than this being
|
|
1719
|
+
* unconditional.
|
|
1720
|
+
*
|
|
1721
|
+
* Failures are swallowed on purpose: a warmup expression that stops parsing
|
|
1722
|
+
* because a package changed is a warmup that did less good, not a reason to
|
|
1723
|
+
* refuse to construct an engine.
|
|
1724
|
+
*/
|
|
1725
|
+
warmUp(): void;
|
|
1653
1726
|
/**
|
|
1654
1727
|
* Build the {@link LineExecutionContext} passed to `executeBytecode()`
|
|
1655
1728
|
* for a given line. `lineNumber = -1` (the existing sentinel
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { B as BytecodeBuilder } from './BytecodeBuilder-
|
|
1
|
+
import { B as BytecodeBuilder } from './BytecodeBuilder-aqVa7Plx.js';
|
|
2
2
|
import { T as Token } from './Token-CbP_OutD.js';
|
|
3
3
|
import { D as DiagnosticPipeline } from './pipeline-QIT4iD8f.js';
|
|
4
4
|
|
|
@@ -149,6 +149,10 @@ declare class PrecedenceParser {
|
|
|
149
149
|
private diagnosticPipeline;
|
|
150
150
|
private currentExpression;
|
|
151
151
|
private localeCode;
|
|
152
|
+
/** The locale's decimal separator, cached from {@link localeCode}. */
|
|
153
|
+
private readonly decimalSeparator;
|
|
154
|
+
/** The locale's thousands separator, cached from {@link localeCode}. */
|
|
155
|
+
private readonly thousandsSeparator;
|
|
152
156
|
/**
|
|
153
157
|
* The binding power the current infix parselet is being invoked at, i.e. the
|
|
154
158
|
* `minBp` of the expression it sits inside. Set immediately before each Tier-2
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { B as BytecodeBuilder } from './BytecodeBuilder-
|
|
1
|
+
import { B as BytecodeBuilder } from './BytecodeBuilder-aqVa7Plx.cjs';
|
|
2
2
|
import { T as Token } from './Token-CbP_OutD.cjs';
|
|
3
3
|
import { D as DiagnosticPipeline } from './pipeline-B4wf3M1h.cjs';
|
|
4
4
|
|
|
@@ -149,6 +149,10 @@ declare class PrecedenceParser {
|
|
|
149
149
|
private diagnosticPipeline;
|
|
150
150
|
private currentExpression;
|
|
151
151
|
private localeCode;
|
|
152
|
+
/** The locale's decimal separator, cached from {@link localeCode}. */
|
|
153
|
+
private readonly decimalSeparator;
|
|
154
|
+
/** The locale's thousands separator, cached from {@link localeCode}. */
|
|
155
|
+
private readonly thousandsSeparator;
|
|
152
156
|
/**
|
|
153
157
|
* The binding power the current infix parselet is being invoked at, i.e. the
|
|
154
158
|
* `minBp` of the expression it sits inside. Set immediately before each Tier-2
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { V as Value } from './Value-DCTqTSeP.cjs';
|
|
2
|
-
import { U as UserFunctionDef, A as AnonymousBodyDef, a as BytecodeProgram } from './BytecodeBuilder-
|
|
3
|
-
import { E as EngineError } from './EngineError-
|
|
2
|
+
import { U as UserFunctionDef, A as AnonymousBodyDef, a as BytecodeProgram } from './BytecodeBuilder-aqVa7Plx.cjs';
|
|
3
|
+
import { E as EngineError } from './EngineError-DTk7I7hZ.cjs';
|
|
4
4
|
import { D as DiagnosticPipeline } from './pipeline-B4wf3M1h.cjs';
|
|
5
5
|
|
|
6
6
|
/**
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import { V as Value } from './Value-DCTqTSeP.js';
|
|
2
|
-
import { U as UserFunctionDef, A as AnonymousBodyDef, a as BytecodeProgram } from './BytecodeBuilder-
|
|
3
|
-
import { E as EngineError } from './EngineError-
|
|
2
|
+
import { U as UserFunctionDef, A as AnonymousBodyDef, a as BytecodeProgram } from './BytecodeBuilder-aqVa7Plx.js';
|
|
3
|
+
import { E as EngineError } from './EngineError-DTk7I7hZ.js';
|
|
4
4
|
import { D as DiagnosticPipeline } from './pipeline-QIT4iD8f.js';
|
|
5
5
|
|
|
6
6
|
/**
|
|
@@ -69,6 +69,43 @@ interface NormalizerMatch {
|
|
|
69
69
|
*/
|
|
70
70
|
ruleName?: string;
|
|
71
71
|
}
|
|
72
|
+
/**
|
|
73
|
+
* One position of a rule's leading shape, as a declarative constraint.
|
|
74
|
+
*
|
|
75
|
+
* A slot says what the token at that offset from the match position may be.
|
|
76
|
+
* Both fields are optional and an omitted one constrains nothing, so `{}` is a
|
|
77
|
+
* wildcard slot, useful for reaching past a position a rule does not care about
|
|
78
|
+
* to one it does.
|
|
79
|
+
*
|
|
80
|
+
* The exactness contract runs in one direction only, and it is the whole reason
|
|
81
|
+
* this is safe to adopt gradually. Declaring MORE than the rule can match costs
|
|
82
|
+
* a `match()` call that returns null, which is what happens today anyway.
|
|
83
|
+
* Declaring LESS makes the rule unreachable at the positions left out, which is
|
|
84
|
+
* a silent bug. So an incomplete shape (or none at all) is always correct, and
|
|
85
|
+
* only an over-narrow one is wrong. `NormalizerIndexFidelity.spec` checks every
|
|
86
|
+
* declaration against its rule's real behaviour.
|
|
87
|
+
*/
|
|
88
|
+
interface RuleSlot {
|
|
89
|
+
/**
|
|
90
|
+
* Token types admitted at this slot. Omit when the type is unconstrained,
|
|
91
|
+
* or when the rule accepts so many that naming them filters nothing.
|
|
92
|
+
*/
|
|
93
|
+
readonly types?: readonly string[];
|
|
94
|
+
/**
|
|
95
|
+
* Token values admitted at this slot, compared case-insensitively.
|
|
96
|
+
*
|
|
97
|
+
* This is the axis that separates rules sharing a start type. The
|
|
98
|
+
* call-fusion rules all begin at an `IDENT`, the commonest token in prose,
|
|
99
|
+
* so type alone leaves every one of them a candidate at every word; the word
|
|
100
|
+
* itself is what tells them apart, and each already owns that set as an
|
|
101
|
+
* exported constant.
|
|
102
|
+
*
|
|
103
|
+
* A rule whose own check is case-SENSITIVE (the stock ticker rule matches
|
|
104
|
+
* upper case only) may still declare its lower-cased words here: the index
|
|
105
|
+
* only ever over-approximates, and the rule's own check still runs.
|
|
106
|
+
*/
|
|
107
|
+
readonly values?: readonly string[];
|
|
108
|
+
}
|
|
72
109
|
/**
|
|
73
110
|
* A pluggable normalization rule registered with the TokenNormalizer.
|
|
74
111
|
*
|
|
@@ -139,6 +176,60 @@ interface NormalizerRule {
|
|
|
139
176
|
* the rule silently unreachable there, which is a bug.
|
|
140
177
|
*/
|
|
141
178
|
readonly startTokenTypes?: readonly string[];
|
|
179
|
+
/**
|
|
180
|
+
* The rule's leading shape: what the tokens from the match position onward
|
|
181
|
+
* may be, one {@link RuleSlot} per position.
|
|
182
|
+
*
|
|
183
|
+
* This generalises {@link startTokenTypes}, which constrains only the first
|
|
184
|
+
* token. Constraining the first token alone is not enough to separate the
|
|
185
|
+
* rules that matter: every rule firing on a bare `NUMBER` declares the same
|
|
186
|
+
* start type, so they all remain candidates at every number in the document.
|
|
187
|
+
* What distinguishes them is the token after it, `NUMBER COLON` being a clock
|
|
188
|
+
* time and `NUMBER SLASH` a network address, and that fact is only usable by
|
|
189
|
+
* an index if the rule states it rather than hiding it inside `match()`.
|
|
190
|
+
*
|
|
191
|
+
* Depth is the rule's choice, not the interface's. The normalizer builds one
|
|
192
|
+
* lookup plane per declared slot and intersects them, so a rule that declares
|
|
193
|
+
* three positions is filtered on three. It may also index fewer planes than
|
|
194
|
+
* were declared, which stays correct for the reason given on {@link RuleSlot}:
|
|
195
|
+
* a shallower filter admits more candidates, and each surviving rule still
|
|
196
|
+
* runs its own `match()`.
|
|
197
|
+
*
|
|
198
|
+
* Prefer this to {@link startTokenTypes} in new rules. When both are given,
|
|
199
|
+
* this wins; `startTokenTypes: ["IDENT"]` means exactly `shape: [{ types:
|
|
200
|
+
* ["IDENT"] }]`.
|
|
201
|
+
*
|
|
202
|
+
* @example
|
|
203
|
+
* ```ts
|
|
204
|
+
* // 9:00am, 16:00, a clock time is a number followed by a colon
|
|
205
|
+
* shape: [{ types: ["NUMBER"] }, { types: ["COLON"] }]
|
|
206
|
+
*
|
|
207
|
+
* // sha256("hi"), a known word followed by an opening parenthesis
|
|
208
|
+
* shape: [{ types: ["IDENT"], values: HASH_NAMES }, { types: ["LPAREN"] }]
|
|
209
|
+
* ```
|
|
210
|
+
*/
|
|
211
|
+
readonly shape?: readonly RuleSlot[];
|
|
212
|
+
/**
|
|
213
|
+
* Why this rule cannot declare a {@link shape}, for the few that genuinely
|
|
214
|
+
* cannot.
|
|
215
|
+
*
|
|
216
|
+
* A rule with neither a shape nor a `startTokenTypes` hint is tried at every
|
|
217
|
+
* position of every line, so it raises the cost of the whole document rather
|
|
218
|
+
* than only its own feature. Registering one logs a warning naming the rule,
|
|
219
|
+
* which is how a package author finds out before their users do.
|
|
220
|
+
*
|
|
221
|
+
* Some rules really cannot be described by a leading shape: an unbounded
|
|
222
|
+
* forward scan, a greedy match against a table the host mutates at runtime.
|
|
223
|
+
* Setting this states that case, silences the warning, and leaves the reason
|
|
224
|
+
* in the code where the next person will read it. It is deliberately a
|
|
225
|
+
* sentence and not a boolean, because "why" is the part worth keeping.
|
|
226
|
+
*
|
|
227
|
+
* @example
|
|
228
|
+
* ```ts
|
|
229
|
+
* unshapedReason: "Scans forward an unbounded number of NUMBER UNIT pairs, so no fixed leading shape describes it.",
|
|
230
|
+
* ```
|
|
231
|
+
*/
|
|
232
|
+
readonly unshapedReason?: string;
|
|
142
233
|
/**
|
|
143
234
|
* Attempt to match a pattern starting at position `pos` in the token stream.
|
|
144
235
|
*
|
|
@@ -228,6 +319,19 @@ interface NormalizerOptions {
|
|
|
228
319
|
* When `undefined`, fusions are still tracked internally but no callbacks fire.
|
|
229
320
|
*/
|
|
230
321
|
onFusion?: (fusion: TokenFusion) => void;
|
|
322
|
+
/**
|
|
323
|
+
* Try every registered rule at every position, ignoring the shape index.
|
|
324
|
+
*
|
|
325
|
+
* Diagnostic only, and much slower. It exists so the indexed walk can be
|
|
326
|
+
* compared against the unindexed one over a corpus: the index is a pure
|
|
327
|
+
* filter, so the two must agree token for token, and a rule whose declared
|
|
328
|
+
* {@link NormalizerRule.shape} is too narrow shows up as a difference rather
|
|
329
|
+
* than as a feature that quietly stopped working.
|
|
330
|
+
* `NormalizerIndexFidelity.spec` is the consumer.
|
|
331
|
+
*
|
|
332
|
+
* @default false
|
|
333
|
+
*/
|
|
334
|
+
ignoreRuleIndex?: boolean;
|
|
231
335
|
}
|
|
232
336
|
/**
|
|
233
337
|
* Creates a new normalized token from fused source tokens.
|
|
@@ -298,6 +402,28 @@ declare class TokenNormalizer {
|
|
|
298
402
|
* identifier. Invalidated alongside {@link sortedRulesCache}.
|
|
299
403
|
*/
|
|
300
404
|
private rulesByTokenType;
|
|
405
|
+
/**
|
|
406
|
+
* Shape index over the priority-sorted rules, rebuilt alongside
|
|
407
|
+
* {@link sortedRulesCache}. `null` means "stale, rebuild on next use".
|
|
408
|
+
*
|
|
409
|
+
* This is what turns the per-position scan from "try every rule that could
|
|
410
|
+
* fire on this token type" into "AND a few lookup planes and try what
|
|
411
|
+
* survives", which is usually nothing. See {@link RuleIndex}.
|
|
412
|
+
*/
|
|
413
|
+
private ruleIndexCache;
|
|
414
|
+
/** Reused by {@link rulesAt} so a position's candidate list costs no allocation. */
|
|
415
|
+
private candidateBuffer;
|
|
416
|
+
/**
|
|
417
|
+
* How many times {@link normalize} has run, capped once the index is in use.
|
|
418
|
+
*
|
|
419
|
+
* The index costs a few tens of microseconds to build and pays that back
|
|
420
|
+
* within a line or two, but an engine that normalises exactly once, which is
|
|
421
|
+
* what evaluating a single expression on a fresh engine does, would never
|
|
422
|
+
* reach the payback. Skipping the build on the first call keeps that case at
|
|
423
|
+
* the cost it had before the index existed, and a document reaches line two
|
|
424
|
+
* immediately.
|
|
425
|
+
*/
|
|
426
|
+
private normalizeCalls;
|
|
301
427
|
/**
|
|
302
428
|
* Phrase trie for single-pass multi-word phrase fusion.
|
|
303
429
|
* Tried at each token position BEFORE other rules, the trie walk
|
|
@@ -348,11 +474,67 @@ declare class TokenNormalizer {
|
|
|
348
474
|
* and cached, which is what turns the per-position rule scan from "try all R
|
|
349
475
|
* rules" into "try only the ones that could fire on this token".
|
|
350
476
|
*/
|
|
477
|
+
private getRuleIndex;
|
|
478
|
+
/**
|
|
479
|
+
* The rules to try at a position, in priority order.
|
|
480
|
+
*
|
|
481
|
+
* With a candidate mask, walks its set bits low to high, which is descending
|
|
482
|
+
* priority because bit `i` is the rule at index `i` of the priority-sorted
|
|
483
|
+
* list.
|
|
484
|
+
*
|
|
485
|
+
* The walk shifts a bit at a time rather than jumping to the next set bit
|
|
486
|
+
* with `Math.clz32(bits & -bits)`, which is the idiomatic form and was 36x
|
|
487
|
+
* SLOWER here: 45us per line against 1.2us, measured over the built-in rule
|
|
488
|
+
* set. `clz32` is specified on uint32 and the masks come out of a
|
|
489
|
+
* `Uint32Array` as doubles above 2^31, so every call pays a conversion that
|
|
490
|
+
* swamps the handful of iterations it saves. A de Bruijn table matched the
|
|
491
|
+
* shift scan to within noise and needs a magic constant, so the plain shift
|
|
492
|
+
* wins on both counts. Worst case is 32 iterations per word of pure integer
|
|
493
|
+
* work.
|
|
494
|
+
*
|
|
495
|
+
* With `null` (the {@link NormalizerOptions.ignoreRuleIndex} path) it falls
|
|
496
|
+
* back to the type-bucketed list, which is the behaviour this replaced.
|
|
497
|
+
*
|
|
498
|
+
* Returns a buffer reused across positions, so the caller must finish with it
|
|
499
|
+
* before calling again. Copying the surviving rules out here rather than
|
|
500
|
+
* yielding them lazily is deliberate twice over: it keeps the mask's own
|
|
501
|
+
* scratch buffer from being read after the next position overwrites it, and
|
|
502
|
+
* it avoids a generator on the hottest loop in the pass.
|
|
503
|
+
*/
|
|
504
|
+
private rulesAt;
|
|
351
505
|
private rulesForTokenType;
|
|
352
506
|
/**
|
|
353
507
|
* Get the number of currently registered rules (excludes phrase trie entries).
|
|
354
508
|
*/
|
|
355
509
|
get ruleCount(): number;
|
|
510
|
+
/**
|
|
511
|
+
* Every registered rule with the shape it declared, for diagnostic display.
|
|
512
|
+
*
|
|
513
|
+
* Exposes what the index is actually working with: a rule that declares a
|
|
514
|
+
* shape is only tried where that shape can match, and one that declares none
|
|
515
|
+
* is tried at every position of every line. Which is which is invisible from
|
|
516
|
+
* the outside otherwise, and it is the difference between a package that
|
|
517
|
+
* costs the documents that use it and one that costs all of them, so the
|
|
518
|
+
* playground draws it.
|
|
519
|
+
*
|
|
520
|
+
* Returned in priority order, highest first, which is the order the
|
|
521
|
+
* normalizer tries them in.
|
|
522
|
+
*/
|
|
523
|
+
private ruleShapesCache;
|
|
524
|
+
getRuleShapes(): Array<{
|
|
525
|
+
name: string;
|
|
526
|
+
priority: number;
|
|
527
|
+
shape: readonly RuleSlot[];
|
|
528
|
+
unshapedReason?: string;
|
|
529
|
+
indexedSlots: number;
|
|
530
|
+
}>;
|
|
531
|
+
/**
|
|
532
|
+
* How many rules could fire at each position of `tokens`, against the total.
|
|
533
|
+
*
|
|
534
|
+
* The point of the index is that most positions admit no rule at all, and
|
|
535
|
+
* this is what makes that visible rather than asserted.
|
|
536
|
+
*/
|
|
537
|
+
getCandidateCounts(tokens: Token[]): number[];
|
|
356
538
|
/**
|
|
357
539
|
* Register a multi-word phrase for fusion into a single compound token.
|
|
358
540
|
*
|