@lokascript/framework 2.5.0 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/index.js +148 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +52 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +148 -1
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/index.cjs +148 -1
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +148 -1
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts +11 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/testing/index.js +12 -4
- package/dist/testing/index.js.map +1 -1
- package/package.json +3 -3
- package/src/core/tokenization/base-tokenizer.ts +219 -0
- package/src/core/tokenization/keyword-boundary.test.ts +73 -0
- package/src/interfaces/value-extractor.ts +23 -0
- package/src/core/pattern-matching/pattern-matcher.ts.backup +0 -1267
|
@@ -45,8 +45,27 @@ export declare abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
45
45
|
protected normalizer?: MorphologicalNormalizer;
|
|
46
46
|
/** Keywords derived from profile, sorted longest-first for greedy matching */
|
|
47
47
|
protected profileKeywords: KeywordEntry[];
|
|
48
|
+
/**
|
|
49
|
+
* Space-containing profile keywords (multi-word phrases), longest-first.
|
|
50
|
+
* Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
|
|
51
|
+
* vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
|
|
52
|
+
* profile-driven replacement for the per-language hardcoded compound lists.
|
|
53
|
+
* Empty for no-space (CJK) languages, so they are unaffected.
|
|
54
|
+
*/
|
|
55
|
+
protected multiWordKeywords: KeywordEntry[];
|
|
48
56
|
/** Map for O(1) keyword lookups by lowercase native word */
|
|
49
57
|
protected profileKeywordMap: Map<string, KeywordEntry>;
|
|
58
|
+
/**
|
|
59
|
+
* The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
|
|
60
|
+
* The keyword map is keyed by native word with last-wins insertion, so a
|
|
61
|
+
* duplicate native word inside the extras silently shadows the earlier entry
|
|
62
|
+
* (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
|
|
63
|
+
* positional expressions). Exposed so consistency tests can detect such
|
|
64
|
+
* intra-extras collisions, which are invisible in the deduplicated map.
|
|
65
|
+
*/
|
|
66
|
+
private rawExtraEntries;
|
|
67
|
+
/** Raw extras as passed in, pre-dedup — for consistency tests. */
|
|
68
|
+
getExtraKeywordEntries(): readonly KeywordEntry[];
|
|
50
69
|
/**
|
|
51
70
|
* Pluggable value extractors for domain-specific syntax.
|
|
52
71
|
* When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
|
|
@@ -143,6 +162,20 @@ export declare abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
143
162
|
* @returns Token if matched, null otherwise
|
|
144
163
|
*/
|
|
145
164
|
protected tryProfileKeyword(input: string, pos: number): LanguageToken | null;
|
|
165
|
+
/**
|
|
166
|
+
* Match the longest multi-word (space-containing) profile keyword at `pos`,
|
|
167
|
+
* requiring the match to end at a word boundary. The profile-driven
|
|
168
|
+
* counterpart of the per-language hardcoded compound lists (the hindi and
|
|
169
|
+
* vietnamese keyword extractors). Returns a keyword token (with the normalized
|
|
170
|
+
* form) or null. Case-sensitive against the stored native form, mirroring
|
|
171
|
+
* `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
|
|
172
|
+
* case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
|
|
173
|
+
*
|
|
174
|
+
* @param input - Input string
|
|
175
|
+
* @param pos - Current position (must be a token-start boundary)
|
|
176
|
+
* @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
|
|
177
|
+
*/
|
|
178
|
+
protected tryMultiWordKeyword(input: string, pos: number, isWordChar?: (char: string) => boolean): LanguageToken | null;
|
|
146
179
|
/**
|
|
147
180
|
* Check if the remaining input starts with any known keyword.
|
|
148
181
|
* Useful for non-space languages to detect word boundaries.
|
|
@@ -152,6 +185,25 @@ export declare abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
152
185
|
* @returns true if a keyword starts at this position
|
|
153
186
|
*/
|
|
154
187
|
protected isKeywordStart(input: string, pos: number): boolean;
|
|
188
|
+
/**
|
|
189
|
+
* Check if a known keyword starts at the given position AND ends at a word
|
|
190
|
+
* boundary (end of input or a non-word character).
|
|
191
|
+
*
|
|
192
|
+
* Space-delimited languages must use this (not `isKeywordStart`) for
|
|
193
|
+
* word-walk break checks: the keyword table includes English canonical
|
|
194
|
+
* fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
|
|
195
|
+
* word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
|
|
196
|
+
* "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
|
|
197
|
+
* keep using `isKeywordStart`.
|
|
198
|
+
*
|
|
199
|
+
* @param input - Input string
|
|
200
|
+
* @param pos - Current position
|
|
201
|
+
* @param isWordChar - Language-specific word-character predicate; pass the
|
|
202
|
+
* tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
|
|
203
|
+
* counts as part of a word. Defaults to Unicode letters/digits/underscore.
|
|
204
|
+
* @returns true if a keyword starts here and is not followed by a word char
|
|
205
|
+
*/
|
|
206
|
+
protected isKeywordStartAtBoundary(input: string, pos: number, isWordChar?: (char: string) => boolean): boolean;
|
|
155
207
|
/**
|
|
156
208
|
* Look up a keyword by native word (case-insensitive).
|
|
157
209
|
* O(1) lookup using the keyword map.
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"base-tokenizer.d.ts","sourceRoot":"","sources":["../../../src/core/tokenization/base-tokenizer.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AAEH,OAAO,KAAK,EAAE,aAAa,EAAE,SAAS,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,UAAU,CAAC;AACzF,OAAO,KAAK,EAAE,uBAAuB,EAAE,mBAAmB,EAAE,MAAM,oBAAoB,CAAC;AACvF,OAAO,EACL,KAAK,cAAc,EACnB,KAAK,YAAY,EAGlB,MAAM,kCAAkC,CAAC;AAC1C,OAAO,EAOL,KAAK,eAAe,EAErB,MAAM,eAAe,CAAC;
|
|
1
|
+
{"version":3,"file":"base-tokenizer.d.ts","sourceRoot":"","sources":["../../../src/core/tokenization/base-tokenizer.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AAEH,OAAO,KAAK,EAAE,aAAa,EAAE,SAAS,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,UAAU,CAAC;AACzF,OAAO,KAAK,EAAE,uBAAuB,EAAE,mBAAmB,EAAE,MAAM,oBAAoB,CAAC;AACvF,OAAO,EACL,KAAK,cAAc,EACnB,KAAK,YAAY,EAGlB,MAAM,kCAAkC,CAAC;AAC1C,OAAO,EAOL,KAAK,eAAe,EAErB,MAAM,eAAe,CAAC;AAsEvB,YAAY,EAAE,YAAY,EAAE,CAAC;AAmC7B;;;GAGG;AACH,MAAM,WAAW,gBAAgB;IAC/B,QAAQ,CAAC,QAAQ,CAAC,EAAE,MAAM,CACxB,MAAM,EACN;QAAE,OAAO,EAAE,MAAM,CAAC;QAAC,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;QAAC,UAAU,CAAC,EAAE,MAAM,CAAA;KAAE,CAClE,CAAC;IACF,QAAQ,CAAC,UAAU,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC7C,QAAQ,CAAC,WAAW,CAAC,EAAE,MAAM,CAC3B,MAAM,EACN;QAAE,OAAO,EAAE,MAAM,CAAC;QAAC,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;QAAC,QAAQ,CAAC,EAAE,MAAM,CAAA;KAAE,CAChE,CAAC;IACF,QAAQ,CAAC,UAAU,CAAC,EAAE;QACpB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;QACxB,QAAQ,CAAC,cAAc,EAAE,cAAc,GAAG,SAAS,GAAG,iBAAiB,CAAC;QACxE,QAAQ,CAAC,YAAY,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;QAC/C,QAAQ,CAAC,uBAAuB,CAAC,EAAE,OAAO,CAAC;QAC3C,QAAQ,CAAC,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;KAC5C,CAAC;CACH;AAMD;;;GAGG;AACH,8BAAsB,aAAc,YAAW,iBAAiB;IAC9D,QAAQ,CAAC,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IACnC,QAAQ,CAAC,QAAQ,CAAC,SAAS,EAAE,KAAK,GAAG,KAAK,CAAC;IAE3C,0DAA0D;IAC1D,SAAS,CAAC,UAAU,CAAC,EAAE,uBAAuB,CAAC;IAE/C,8EAA8E;IAC9E,SAAS,CAAC,eAAe,EAAE,YAAY,EAAE,CAAM;IAE/C;;;;;;OAMG;IACH,SAAS,CAAC,iBAAiB,EAAE,YAAY,EAAE,CAAM;IAEjD,4DAA4D;IAC5D,SAAS,CAAC,iBAAiB,EAAE,GAAG,CAAC,MAAM,EAAE,YAAY,CAAC,CAAa;IAEnE;;;;;;;OAOG;IACH,OAAO,CAAC,eAAe,CAAsB;IAE7C,kEAAkE;IAClE,sBAAsB,IAAI,SAAS,YAAY,EAAE;IAIjD;;;OAGG;IACH,SAAS,CAAC,UAAU,EAAE,cAAc,EAAE,CAAM;IAE5C;;;;;;;OAOG;IACH,QAAQ,CAAC,KAAK,EAAE,MAAM,GAAG,WAAW;IAYpC,QAAQ,CAAC,aAAa,CAAC,KAAK,EAAE,MAAM,GAAG,SAAS;IAEhD;;;;;;OAMG;IACH,iBAAiB,CAAC,SAAS,EAAE,cAAc,GAAG,IAAI;IAOlD;;;;OAIG;IACH,kBAAkB,CAAC,UAAU,EAAE,cAAc,EAAE,GAAG,IAAI;IAMtD;;;OAGG;IACH,eAAe,IAAI,IAAI;IAIvB;;;OAGG;IACH,SAAS,CAAC,iBAAiB,IAAI,OAAO;IAItC;;;;;;OAMG;IACH,SAAS,CAAC,sBAAsB,CAAC,KAAK,EAAE,MAAM,GAAG,WAAW;IA8E5D;;;;;;OAMG;IACH,SAAS,CAAC,mBAAmB,CAAC,IAAI,EAAE,MAAM,GAAG,SAAS;IAMtD;;;;;;;OAOG;IACH,SAAS,CAAC,iBAAiB,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,EAAE,MAAM,EAAE,aAAa,EAAE,GAAG,OAAO;IAgCzF;;;;;;;;;;;;OAYG;IACH,SAAS,CAAC,6BAA6B,CACrC,OAAO,EAAE,gBAAgB,EACzB,MAAM,GAAE,YAAY,EAAO,GAC1B,IAAI;IAoHP;;;;;;;OAOG;IACH,SAAS,CAAC,gBAAgB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM;IAMhD;;;;;;;OAOG;IACH,SAAS,CAAC,iBAAiB,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAc7E;;;;;;;;;;;;OAYG;IACH,SAAS,CAAC,mBAAmB,CAC3B,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,UAAU,GAAE,CAAC,IAAI,EAAE,MAAM,KAAK,OAAyC,GACtE,aAAa,GAAG,IAAI;IAiBvB;;;;;;;OAOG;IACH,SAAS,CAAC,cAAc,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,OAAO;IAK7D;;;;;;;;;;;;;;;;;OAiBG;IACH,SAAS,CAAC,wBAAwB,CAChC,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,UAAU,GAAE,CAAC,IAAI,EAAE,MAAM,KAAK,OAAyC,GACtE,OAAO;IASV;;;;;;OAMG;IACH,SAAS,CAAC,aAAa,CAAC,MAAM,EAAE,MAAM,GAAG,YAAY,GAAG,SAAS;IAIjE;;;;;;OAMG;IACH,SAAS,CAAC,SAAS,CAAC,MAAM,EAAE,MAAM,GAAG,OAAO;IAI5C;;OAEG;IACH,aAAa,CAAC,UAAU,EAAE,uBAAuB,GAAG,IAAI;IAIxD;;;;;;;OAOG;IACH,SAAS,CAAC,YAAY,CAAC,IAAI,EAAE,MAAM,GAAG,mBAAmB,GAAG,IAAI;IAahE;;;;;;;;;;;;;;;OAeG;IACH,SAAS,CAAC,oBAAoB,CAC5B,IAAI,EAAE,MAAM,EACZ,QAAQ,EAAE,MAAM,EAChB,MAAM,EAAE,MAAM,GACb,aAAa,GAAG,IAAI;IAgBvB;;OAEG;IACH,SAAS,CAAC,WAAW,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQvE;;;OAGG;IACH,SAAS,CAAC,gBAAgB,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAmC5E;;OAEG;IACH,SAAS,CAAC,SAAS,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQrE;;OAEG;IACH,SAAS,CAAC,SAAS,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQrE;;;OAGG;IACH,SAAS,CAAC,MAAM,CAAC,QAAQ,CAAC,mBAAmB,EAAE,SAAS,eAAe,EAAE,CAKvE;IAEF;;;;;;;;OAQG;IACH,SAAS,CAAC,gBAAgB,CACxB,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,SAAS,EAAE,SAAS,eAAe,EAAE,EACrC,cAAc,UAAQ,GACrB;QAAE,MAAM,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,CAAA;KAAE,GAAG,IAAI;IAuC5C;;;;;;;;OAQG;IACH,SAAS,CAAC,eAAe,CACvB,KAAK,EAAE,MAAM,EACb,QAAQ,EAAE,MAAM,EAChB,SAAS,UAAO,GACf;QAAE,MAAM,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,CAAA;KAAE,GAAG,IAAI;IAgC5C;;;;;;;;;;;;;OAaG;IACH,SAAS,CAAC,sBAAsB,CAC9B,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,eAAe,EAAE,SAAS,eAAe,EAAE,EAC3C,OAAO,GAAE;QAAE,SAAS,CAAC,EAAE,OAAO,CAAC;QAAC,cAAc,CAAC,EAAE,OAAO,CAAA;KAAO,GAC9D,aAAa,GAAG,IAAI;IAqBvB;;;OAGG;IACH,SAAS,CAAC,MAAM,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQlE;;;OAGG;IACH,SAAS,CAAC,cAAc,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAc1E;;;OAGG;IACH,SAAS,CAAC,WAAW,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAqBvE;;;;;;;;;;OAUG;IACH,SAAS,CAAC,oBAAoB,CAC5B,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,SAAS,EAAE,SAAS,MAAM,EAAE,GAC3B,aAAa,GAAG,IAAI;CAQxB;AAMD;;;;;;;;;;;;;GAaG;AACH,MAAM,WAAW,qBAAqB;IACpC,8BAA8B;IAC9B,QAAQ,EAAE,MAAM,CAAC;IACjB,sCAAsC;IACtC,SAAS,CAAC,EAAE,KAAK,GAAG,KAAK,CAAC;IAC1B,uEAAuE;IACvE,QAAQ,EAAE,MAAM,EAAE,CAAC;IACnB,wDAAwD;IACxD,aAAa,CAAC,EAAE,YAAY,EAAE,CAAC;IAC/B,wEAAwE;IACxE,cAAc,CAAC,EAAE,gBAAgB,CAAC;IAClC,uGAAuG;IACvG,gBAAgB,CAAC,EAAE,OAAO,CAAC;IAC3B,wDAAwD;IACxD,eAAe,CAAC,EAAE,OAAO,CAAC;IAC1B,6DAA6D;IAC7D,gBAAgB,CAAC,EAAE,cAAc,EAAE,CAAC;CACrC;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,qBAAqB,CAAC,MAAM,EAAE,qBAAqB,GAAG,iBAAiB,CA2CtF"}
|
|
@@ -641,7 +641,8 @@ function createTokenizerContext(tokenizer) {
|
|
|
641
641
|
direction: tokenizer.direction,
|
|
642
642
|
lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
|
|
643
643
|
isKeyword: tokenizer.isKeyword.bind(tokenizer),
|
|
644
|
-
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
|
|
644
|
+
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
|
|
645
|
+
...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
|
|
645
646
|
};
|
|
646
647
|
if (tokenizer.normalizer) {
|
|
647
648
|
return { ...ctx, normalizer: tokenizer.normalizer };
|
|
@@ -689,18 +690,94 @@ function createLatinCharClassifiers(letterPattern) {
|
|
|
689
690
|
|
|
690
691
|
// src/core/tokenization/base-tokenizer.ts
|
|
691
692
|
var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
|
|
693
|
+
var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
|
|
694
|
+
// Role-marker role names (profile.roleMarkers normalizeds)
|
|
695
|
+
"patient",
|
|
696
|
+
"destination",
|
|
697
|
+
"source",
|
|
698
|
+
"style",
|
|
699
|
+
"event",
|
|
700
|
+
"eventMarker",
|
|
701
|
+
"agent",
|
|
702
|
+
"goal",
|
|
703
|
+
"manner",
|
|
704
|
+
// Prepositional / positional modifier concepts matched via the role mechanism
|
|
705
|
+
// (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
|
|
706
|
+
// NOT here — they are pattern literals (see the note above).
|
|
707
|
+
"into",
|
|
708
|
+
"from",
|
|
709
|
+
"to",
|
|
710
|
+
"with",
|
|
711
|
+
"at",
|
|
712
|
+
"of",
|
|
713
|
+
"as",
|
|
714
|
+
"by",
|
|
715
|
+
"in",
|
|
716
|
+
"on",
|
|
717
|
+
"over",
|
|
718
|
+
"under",
|
|
719
|
+
"between",
|
|
720
|
+
"through",
|
|
721
|
+
"without"
|
|
722
|
+
]);
|
|
723
|
+
var ENGLISH_DOM_EVENT_NAMES = [
|
|
724
|
+
"click",
|
|
725
|
+
"dblclick",
|
|
726
|
+
"input",
|
|
727
|
+
"change",
|
|
728
|
+
"submit",
|
|
729
|
+
"keydown",
|
|
730
|
+
"keyup",
|
|
731
|
+
"keypress",
|
|
732
|
+
"mousedown",
|
|
733
|
+
"mouseup",
|
|
734
|
+
"mouseover",
|
|
735
|
+
"mouseout",
|
|
736
|
+
"mouseenter",
|
|
737
|
+
"mouseleave",
|
|
738
|
+
"mousemove",
|
|
739
|
+
"pointerdown",
|
|
740
|
+
"pointerup",
|
|
741
|
+
"pointermove",
|
|
742
|
+
"focus",
|
|
743
|
+
"blur",
|
|
744
|
+
"load",
|
|
745
|
+
"resize",
|
|
746
|
+
"scroll"
|
|
747
|
+
];
|
|
692
748
|
var _BaseTokenizer = class _BaseTokenizer {
|
|
693
749
|
constructor() {
|
|
694
750
|
/** Keywords derived from profile, sorted longest-first for greedy matching */
|
|
695
751
|
this.profileKeywords = [];
|
|
752
|
+
/**
|
|
753
|
+
* Space-containing profile keywords (multi-word phrases), longest-first.
|
|
754
|
+
* Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
|
|
755
|
+
* vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
|
|
756
|
+
* profile-driven replacement for the per-language hardcoded compound lists.
|
|
757
|
+
* Empty for no-space (CJK) languages, so they are unaffected.
|
|
758
|
+
*/
|
|
759
|
+
this.multiWordKeywords = [];
|
|
696
760
|
/** Map for O(1) keyword lookups by lowercase native word */
|
|
697
761
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
762
|
+
/**
|
|
763
|
+
* The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
|
|
764
|
+
* The keyword map is keyed by native word with last-wins insertion, so a
|
|
765
|
+
* duplicate native word inside the extras silently shadows the earlier entry
|
|
766
|
+
* (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
|
|
767
|
+
* positional expressions). Exposed so consistency tests can detect such
|
|
768
|
+
* intra-extras collisions, which are invisible in the deduplicated map.
|
|
769
|
+
*/
|
|
770
|
+
this.rawExtraEntries = [];
|
|
698
771
|
/**
|
|
699
772
|
* Pluggable value extractors for domain-specific syntax.
|
|
700
773
|
* When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
|
|
701
774
|
*/
|
|
702
775
|
this.extractors = [];
|
|
703
776
|
}
|
|
777
|
+
/** Raw extras as passed in, pre-dedup — for consistency tests. */
|
|
778
|
+
getExtraKeywordEntries() {
|
|
779
|
+
return this.rawExtraEntries;
|
|
780
|
+
}
|
|
704
781
|
/**
|
|
705
782
|
* Tokenize input string to token stream.
|
|
706
783
|
* Delegates to extractor-based tokenization if extractors are registered,
|
|
@@ -769,6 +846,12 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
769
846
|
pos++;
|
|
770
847
|
}
|
|
771
848
|
if (pos >= input.length) break;
|
|
849
|
+
const multiWord = this.tryMultiWordKeyword(input, pos);
|
|
850
|
+
if (multiWord) {
|
|
851
|
+
tokens.push(multiWord);
|
|
852
|
+
pos = multiWord.position.end;
|
|
853
|
+
continue;
|
|
854
|
+
}
|
|
772
855
|
let extracted = false;
|
|
773
856
|
for (const extractor of this.extractors) {
|
|
774
857
|
if (extractor.canExtract(input, pos)) {
|
|
@@ -868,6 +951,7 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
868
951
|
*/
|
|
869
952
|
initializeKeywordsFromProfile(profile, extras = []) {
|
|
870
953
|
const keywordMap = /* @__PURE__ */ new Map();
|
|
954
|
+
this.rawExtraEntries = extras;
|
|
871
955
|
if (profile.keywords) {
|
|
872
956
|
for (const [normalized2, translation] of Object.entries(profile.keywords)) {
|
|
873
957
|
keywordMap.set(translation.primary, {
|
|
@@ -911,12 +995,20 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
911
995
|
keywordMap.set(native, { native, normalized: normalized2 });
|
|
912
996
|
}
|
|
913
997
|
}
|
|
998
|
+
for (const evt of ENGLISH_DOM_EVENT_NAMES) {
|
|
999
|
+
if (!keywordMap.has(evt)) {
|
|
1000
|
+
keywordMap.set(evt, { native: evt, normalized: evt });
|
|
1001
|
+
}
|
|
1002
|
+
}
|
|
914
1003
|
for (const extra of extras) {
|
|
915
1004
|
keywordMap.set(extra.native, extra);
|
|
916
1005
|
}
|
|
917
1006
|
this.profileKeywords = Array.from(keywordMap.values()).sort(
|
|
918
1007
|
(a, b) => b.native.length - a.native.length
|
|
919
1008
|
);
|
|
1009
|
+
this.multiWordKeywords = this.profileKeywords.filter(
|
|
1010
|
+
(k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
|
|
1011
|
+
);
|
|
920
1012
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
921
1013
|
for (const keyword of this.profileKeywords) {
|
|
922
1014
|
this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
|
|
@@ -958,6 +1050,35 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
958
1050
|
}
|
|
959
1051
|
return null;
|
|
960
1052
|
}
|
|
1053
|
+
/**
|
|
1054
|
+
* Match the longest multi-word (space-containing) profile keyword at `pos`,
|
|
1055
|
+
* requiring the match to end at a word boundary. The profile-driven
|
|
1056
|
+
* counterpart of the per-language hardcoded compound lists (the hindi and
|
|
1057
|
+
* vietnamese keyword extractors). Returns a keyword token (with the normalized
|
|
1058
|
+
* form) or null. Case-sensitive against the stored native form, mirroring
|
|
1059
|
+
* `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
|
|
1060
|
+
* case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
|
|
1061
|
+
*
|
|
1062
|
+
* @param input - Input string
|
|
1063
|
+
* @param pos - Current position (must be a token-start boundary)
|
|
1064
|
+
* @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
|
|
1065
|
+
*/
|
|
1066
|
+
tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
1067
|
+
if (this.multiWordKeywords.length === 0) return null;
|
|
1068
|
+
const rest = input.slice(pos);
|
|
1069
|
+
for (const entry of this.multiWordKeywords) {
|
|
1070
|
+
if (!rest.startsWith(entry.native)) continue;
|
|
1071
|
+
const after = input[pos + entry.native.length];
|
|
1072
|
+
if (after !== void 0 && isWordChar(after)) continue;
|
|
1073
|
+
return createToken(
|
|
1074
|
+
entry.native,
|
|
1075
|
+
"keyword",
|
|
1076
|
+
createPosition(pos, pos + entry.native.length),
|
|
1077
|
+
entry.normalized
|
|
1078
|
+
);
|
|
1079
|
+
}
|
|
1080
|
+
return null;
|
|
1081
|
+
}
|
|
961
1082
|
/**
|
|
962
1083
|
* Check if the remaining input starts with any known keyword.
|
|
963
1084
|
* Useful for non-space languages to detect word boundaries.
|
|
@@ -970,6 +1091,32 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
970
1091
|
const remaining = input.slice(pos);
|
|
971
1092
|
return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
|
|
972
1093
|
}
|
|
1094
|
+
/**
|
|
1095
|
+
* Check if a known keyword starts at the given position AND ends at a word
|
|
1096
|
+
* boundary (end of input or a non-word character).
|
|
1097
|
+
*
|
|
1098
|
+
* Space-delimited languages must use this (not `isKeywordStart`) for
|
|
1099
|
+
* word-walk break checks: the keyword table includes English canonical
|
|
1100
|
+
* fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
|
|
1101
|
+
* word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
|
|
1102
|
+
* "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
|
|
1103
|
+
* keep using `isKeywordStart`.
|
|
1104
|
+
*
|
|
1105
|
+
* @param input - Input string
|
|
1106
|
+
* @param pos - Current position
|
|
1107
|
+
* @param isWordChar - Language-specific word-character predicate; pass the
|
|
1108
|
+
* tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
|
|
1109
|
+
* counts as part of a word. Defaults to Unicode letters/digits/underscore.
|
|
1110
|
+
* @returns true if a keyword starts here and is not followed by a word char
|
|
1111
|
+
*/
|
|
1112
|
+
isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
1113
|
+
const remaining = input.slice(pos);
|
|
1114
|
+
return this.profileKeywords.some((entry) => {
|
|
1115
|
+
if (!remaining.startsWith(entry.native)) return false;
|
|
1116
|
+
const after = input[pos + entry.native.length];
|
|
1117
|
+
return after === void 0 || !isWordChar(after);
|
|
1118
|
+
});
|
|
1119
|
+
}
|
|
973
1120
|
/**
|
|
974
1121
|
* Look up a keyword by native word (case-insensitive).
|
|
975
1122
|
* O(1) lookup using the keyword map.
|