@lokascript/framework 2.5.1 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -45,8 +45,27 @@ export declare abstract class BaseTokenizer implements LanguageTokenizer {
45
45
  protected normalizer?: MorphologicalNormalizer;
46
46
  /** Keywords derived from profile, sorted longest-first for greedy matching */
47
47
  protected profileKeywords: KeywordEntry[];
48
+ /**
49
+ * Space-containing profile keywords (multi-word phrases), longest-first.
50
+ * Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
51
+ * vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
52
+ * profile-driven replacement for the per-language hardcoded compound lists.
53
+ * Empty for no-space (CJK) languages, so they are unaffected.
54
+ */
55
+ protected multiWordKeywords: KeywordEntry[];
48
56
  /** Map for O(1) keyword lookups by lowercase native word */
49
57
  protected profileKeywordMap: Map<string, KeywordEntry>;
58
+ /**
59
+ * The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
60
+ * The keyword map is keyed by native word with last-wins insertion, so a
61
+ * duplicate native word inside the extras silently shadows the earlier entry
62
+ * (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
63
+ * positional expressions). Exposed so consistency tests can detect such
64
+ * intra-extras collisions, which are invisible in the deduplicated map.
65
+ */
66
+ private rawExtraEntries;
67
+ /** Raw extras as passed in, pre-dedup — for consistency tests. */
68
+ getExtraKeywordEntries(): readonly KeywordEntry[];
50
69
  /**
51
70
  * Pluggable value extractors for domain-specific syntax.
52
71
  * When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
@@ -143,6 +162,20 @@ export declare abstract class BaseTokenizer implements LanguageTokenizer {
143
162
  * @returns Token if matched, null otherwise
144
163
  */
145
164
  protected tryProfileKeyword(input: string, pos: number): LanguageToken | null;
165
+ /**
166
+ * Match the longest multi-word (space-containing) profile keyword at `pos`,
167
+ * requiring the match to end at a word boundary. The profile-driven
168
+ * counterpart of the per-language hardcoded compound lists (the hindi and
169
+ * vietnamese keyword extractors). Returns a keyword token (with the normalized
170
+ * form) or null. Case-sensitive against the stored native form, mirroring
171
+ * `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
172
+ * case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
173
+ *
174
+ * @param input - Input string
175
+ * @param pos - Current position (must be a token-start boundary)
176
+ * @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
177
+ */
178
+ protected tryMultiWordKeyword(input: string, pos: number, isWordChar?: (char: string) => boolean): LanguageToken | null;
146
179
  /**
147
180
  * Check if the remaining input starts with any known keyword.
148
181
  * Useful for non-space languages to detect word boundaries.
@@ -152,6 +185,25 @@ export declare abstract class BaseTokenizer implements LanguageTokenizer {
152
185
  * @returns true if a keyword starts at this position
153
186
  */
154
187
  protected isKeywordStart(input: string, pos: number): boolean;
188
+ /**
189
+ * Check if a known keyword starts at the given position AND ends at a word
190
+ * boundary (end of input or a non-word character).
191
+ *
192
+ * Space-delimited languages must use this (not `isKeywordStart`) for
193
+ * word-walk break checks: the keyword table includes English canonical
194
+ * fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
195
+ * word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
196
+ * "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
197
+ * keep using `isKeywordStart`.
198
+ *
199
+ * @param input - Input string
200
+ * @param pos - Current position
201
+ * @param isWordChar - Language-specific word-character predicate; pass the
202
+ * tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
203
+ * counts as part of a word. Defaults to Unicode letters/digits/underscore.
204
+ * @returns true if a keyword starts here and is not followed by a word char
205
+ */
206
+ protected isKeywordStartAtBoundary(input: string, pos: number, isWordChar?: (char: string) => boolean): boolean;
155
207
  /**
156
208
  * Look up a keyword by native word (case-insensitive).
157
209
  * O(1) lookup using the keyword map.
@@ -1 +1 @@
1
- {"version":3,"file":"base-tokenizer.d.ts","sourceRoot":"","sources":["../../../src/core/tokenization/base-tokenizer.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AAEH,OAAO,KAAK,EAAE,aAAa,EAAE,SAAS,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,UAAU,CAAC;AACzF,OAAO,KAAK,EAAE,uBAAuB,EAAE,mBAAmB,EAAE,MAAM,oBAAoB,CAAC;AACvF,OAAO,EACL,KAAK,cAAc,EACnB,KAAK,YAAY,EAGlB,MAAM,kCAAkC,CAAC;AAC1C,OAAO,EAOL,KAAK,eAAe,EAErB,MAAM,eAAe,CAAC;AAevB,YAAY,EAAE,YAAY,EAAE,CAAC;AAE7B;;;GAGG;AACH,MAAM,WAAW,gBAAgB;IAC/B,QAAQ,CAAC,QAAQ,CAAC,EAAE,MAAM,CACxB,MAAM,EACN;QAAE,OAAO,EAAE,MAAM,CAAC;QAAC,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;QAAC,UAAU,CAAC,EAAE,MAAM,CAAA;KAAE,CAClE,CAAC;IACF,QAAQ,CAAC,UAAU,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC7C,QAAQ,CAAC,WAAW,CAAC,EAAE,MAAM,CAC3B,MAAM,EACN;QAAE,OAAO,EAAE,MAAM,CAAC;QAAC,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;QAAC,QAAQ,CAAC,EAAE,MAAM,CAAA;KAAE,CAChE,CAAC;IACF,QAAQ,CAAC,UAAU,CAAC,EAAE;QACpB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;QACxB,QAAQ,CAAC,cAAc,EAAE,cAAc,GAAG,SAAS,GAAG,iBAAiB,CAAC;QACxE,QAAQ,CAAC,YAAY,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;QAC/C,QAAQ,CAAC,uBAAuB,CAAC,EAAE,OAAO,CAAC;QAC3C,QAAQ,CAAC,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;KAC5C,CAAC;CACH;AAMD;;;GAGG;AACH,8BAAsB,aAAc,YAAW,iBAAiB;IAC9D,QAAQ,CAAC,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IACnC,QAAQ,CAAC,QAAQ,CAAC,SAAS,EAAE,KAAK,GAAG,KAAK,CAAC;IAE3C,0DAA0D;IAC1D,SAAS,CAAC,UAAU,CAAC,EAAE,uBAAuB,CAAC;IAE/C,8EAA8E;IAC9E,SAAS,CAAC,eAAe,EAAE,YAAY,EAAE,CAAM;IAE/C,4DAA4D;IAC5D,SAAS,CAAC,iBAAiB,EAAE,GAAG,CAAC,MAAM,EAAE,YAAY,CAAC,CAAa;IAEnE;;;OAGG;IACH,SAAS,CAAC,UAAU,EAAE,cAAc,EAAE,CAAM;IAE5C;;;;;;;OAOG;IACH,QAAQ,CAAC,KAAK,EAAE,MAAM,GAAG,WAAW;IAYpC,QAAQ,CAAC,aAAa,CAAC,KAAK,EAAE,MAAM,GAAG,SAAS;IAEhD;;;;;;OAMG;IACH,iBAAiB,CAAC,SAAS,EAAE,cAAc,GAAG,IAAI;IAOlD;;;;OAIG;IACH,kBAAkB,CAAC,UAAU,EAAE,cAAc,EAAE,GAAG,IAAI;IAMtD;;;OAGG;IACH,eAAe,IAAI,IAAI;IAIvB;;;OAGG;IACH,SAAS,CAAC,iBAAiB,IAAI,OAAO;IAItC;;;;;;OAMG;IACH,SAAS,CAAC,sBAAsB,CAAC,KAAK,EAAE,MAAM,GAAG,WAAW;IAiE5D;;;;;;OAMG;IACH,SAAS,CAAC,mBAAmB,CAAC,IAAI,EAAE,MAAM,GAAG,SAAS;IAMtD;;;;;;;OAOG;IACH,SAAS,CAAC,iBAAiB,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,EAAE,MAAM,EAAE,aAAa,EAAE,GAAG,OAAO;IAgCzF;;;;;;;;;;;;OAYG;IACH,SAAS,CAAC,6BAA6B,CACrC,OAAO,EAAE,gBAAgB,EACzB,MAAM,GAAE,YAAY,EAAO,GAC1B,IAAI;IAuFP;;;;;;;OAOG;IACH,SAAS,CAAC,gBAAgB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM;IAMhD;;;;;;;OAOG;IACH,SAAS,CAAC,iBAAiB,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAc7E;;;;;;;OAOG;IACH,SAAS,CAAC,cAAc,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,OAAO;IAK7D;;;;;;OAMG;IACH,SAAS,CAAC,aAAa,CAAC,MAAM,EAAE,MAAM,GAAG,YAAY,GAAG,SAAS;IAIjE;;;;;;OAMG;IACH,SAAS,CAAC,SAAS,CAAC,MAAM,EAAE,MAAM,GAAG,OAAO;IAI5C;;OAEG;IACH,aAAa,CAAC,UAAU,EAAE,uBAAuB,GAAG,IAAI;IAIxD;;;;;;;OAOG;IACH,SAAS,CAAC,YAAY,CAAC,IAAI,EAAE,MAAM,GAAG,mBAAmB,GAAG,IAAI;IAahE;;;;;;;;;;;;;;;OAeG;IACH,SAAS,CAAC,oBAAoB,CAC5B,IAAI,EAAE,MAAM,EACZ,QAAQ,EAAE,MAAM,EAChB,MAAM,EAAE,MAAM,GACb,aAAa,GAAG,IAAI;IAgBvB;;OAEG;IACH,SAAS,CAAC,WAAW,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQvE;;;OAGG;IACH,SAAS,CAAC,gBAAgB,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAmC5E;;OAEG;IACH,SAAS,CAAC,SAAS,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQrE;;OAEG;IACH,SAAS,CAAC,SAAS,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQrE;;;OAGG;IACH,SAAS,CAAC,MAAM,CAAC,QAAQ,CAAC,mBAAmB,EAAE,SAAS,eAAe,EAAE,CAKvE;IAEF;;;;;;;;OAQG;IACH,SAAS,CAAC,gBAAgB,CACxB,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,SAAS,EAAE,SAAS,eAAe,EAAE,EACrC,cAAc,UAAQ,GACrB;QAAE,MAAM,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,CAAA;KAAE,GAAG,IAAI;IAuC5C;;;;;;;;OAQG;IACH,SAAS,CAAC,eAAe,CACvB,KAAK,EAAE,MAAM,EACb,QAAQ,EAAE,MAAM,EAChB,SAAS,UAAO,GACf;QAAE,MAAM,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,CAAA;KAAE,GAAG,IAAI;IAgC5C;;;;;;;;;;;;;OAaG;IACH,SAAS,CAAC,sBAAsB,CAC9B,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,eAAe,EAAE,SAAS,eAAe,EAAE,EAC3C,OAAO,GAAE;QAAE,SAAS,CAAC,EAAE,OAAO,CAAC;QAAC,cAAc,CAAC,EAAE,OAAO,CAAA;KAAO,GAC9D,aAAa,GAAG,IAAI;IAqBvB;;;OAGG;IACH,SAAS,CAAC,MAAM,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQlE;;;OAGG;IACH,SAAS,CAAC,cAAc,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAc1E;;;OAGG;IACH,SAAS,CAAC,WAAW,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAqBvE;;;;;;;;;;OAUG;IACH,SAAS,CAAC,oBAAoB,CAC5B,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,SAAS,EAAE,SAAS,MAAM,EAAE,GAC3B,aAAa,GAAG,IAAI;CAQxB;AAMD;;;;;;;;;;;;;GAaG;AACH,MAAM,WAAW,qBAAqB;IACpC,8BAA8B;IAC9B,QAAQ,EAAE,MAAM,CAAC;IACjB,sCAAsC;IACtC,SAAS,CAAC,EAAE,KAAK,GAAG,KAAK,CAAC;IAC1B,uEAAuE;IACvE,QAAQ,EAAE,MAAM,EAAE,CAAC;IACnB,wDAAwD;IACxD,aAAa,CAAC,EAAE,YAAY,EAAE,CAAC;IAC/B,wEAAwE;IACxE,cAAc,CAAC,EAAE,gBAAgB,CAAC;IAClC,uGAAuG;IACvG,gBAAgB,CAAC,EAAE,OAAO,CAAC;IAC3B,wDAAwD;IACxD,eAAe,CAAC,EAAE,OAAO,CAAC;IAC1B,6DAA6D;IAC7D,gBAAgB,CAAC,EAAE,cAAc,EAAE,CAAC;CACrC;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,qBAAqB,CAAC,MAAM,EAAE,qBAAqB,GAAG,iBAAiB,CA2CtF"}
1
+ {"version":3,"file":"base-tokenizer.d.ts","sourceRoot":"","sources":["../../../src/core/tokenization/base-tokenizer.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AAEH,OAAO,KAAK,EAAE,aAAa,EAAE,SAAS,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,UAAU,CAAC;AACzF,OAAO,KAAK,EAAE,uBAAuB,EAAE,mBAAmB,EAAE,MAAM,oBAAoB,CAAC;AACvF,OAAO,EACL,KAAK,cAAc,EACnB,KAAK,YAAY,EAGlB,MAAM,kCAAkC,CAAC;AAC1C,OAAO,EAOL,KAAK,eAAe,EAErB,MAAM,eAAe,CAAC;AAsEvB,YAAY,EAAE,YAAY,EAAE,CAAC;AAmC7B;;;GAGG;AACH,MAAM,WAAW,gBAAgB;IAC/B,QAAQ,CAAC,QAAQ,CAAC,EAAE,MAAM,CACxB,MAAM,EACN;QAAE,OAAO,EAAE,MAAM,CAAC;QAAC,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;QAAC,UAAU,CAAC,EAAE,MAAM,CAAA;KAAE,CAClE,CAAC;IACF,QAAQ,CAAC,UAAU,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC7C,QAAQ,CAAC,WAAW,CAAC,EAAE,MAAM,CAC3B,MAAM,EACN;QAAE,OAAO,EAAE,MAAM,CAAC;QAAC,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;QAAC,QAAQ,CAAC,EAAE,MAAM,CAAA;KAAE,CAChE,CAAC;IACF,QAAQ,CAAC,UAAU,CAAC,EAAE;QACpB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;QACxB,QAAQ,CAAC,cAAc,EAAE,cAAc,GAAG,SAAS,GAAG,iBAAiB,CAAC;QACxE,QAAQ,CAAC,YAAY,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;QAC/C,QAAQ,CAAC,uBAAuB,CAAC,EAAE,OAAO,CAAC;QAC3C,QAAQ,CAAC,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;KAC5C,CAAC;CACH;AAMD;;;GAGG;AACH,8BAAsB,aAAc,YAAW,iBAAiB;IAC9D,QAAQ,CAAC,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IACnC,QAAQ,CAAC,QAAQ,CAAC,SAAS,EAAE,KAAK,GAAG,KAAK,CAAC;IAE3C,0DAA0D;IAC1D,SAAS,CAAC,UAAU,CAAC,EAAE,uBAAuB,CAAC;IAE/C,8EAA8E;IAC9E,SAAS,CAAC,eAAe,EAAE,YAAY,EAAE,CAAM;IAE/C;;;;;;OAMG;IACH,SAAS,CAAC,iBAAiB,EAAE,YAAY,EAAE,CAAM;IAEjD,4DAA4D;IAC5D,SAAS,CAAC,iBAAiB,EAAE,GAAG,CAAC,MAAM,EAAE,YAAY,CAAC,CAAa;IAEnE;;;;;;;OAOG;IACH,OAAO,CAAC,eAAe,CAAsB;IAE7C,kEAAkE;IAClE,sBAAsB,IAAI,SAAS,YAAY,EAAE;IAIjD;;;OAGG;IACH,SAAS,CAAC,UAAU,EAAE,cAAc,EAAE,CAAM;IAE5C;;;;;;;OAOG;IACH,QAAQ,CAAC,KAAK,EAAE,MAAM,GAAG,WAAW;IAYpC,QAAQ,CAAC,aAAa,CAAC,KAAK,EAAE,MAAM,GAAG,SAAS;IAEhD;;;;;;OAMG;IACH,iBAAiB,CAAC,SAAS,EAAE,cAAc,GAAG,IAAI;IAOlD;;;;OAIG;IACH,kBAAkB,CAAC,UAAU,EAAE,cAAc,EAAE,GAAG,IAAI;IAMtD;;;OAGG;IACH,eAAe,IAAI,IAAI;IAIvB;;;OAGG;IACH,SAAS,CAAC,iBAAiB,IAAI,OAAO;IAItC;;;;;;OAMG;IACH,SAAS,CAAC,sBAAsB,CAAC,KAAK,EAAE,MAAM,GAAG,WAAW;IA8E5D;;;;;;OAMG;IACH,SAAS,CAAC,mBAAmB,CAAC,IAAI,EAAE,MAAM,GAAG,SAAS;IAMtD;;;;;;;OAOG;IACH,SAAS,CAAC,iBAAiB,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,EAAE,MAAM,EAAE,aAAa,EAAE,GAAG,OAAO;IAgCzF;;;;;;;;;;;;OAYG;IACH,SAAS,CAAC,6BAA6B,CACrC,OAAO,EAAE,gBAAgB,EACzB,MAAM,GAAE,YAAY,EAAO,GAC1B,IAAI;IAoHP;;;;;;;OAOG;IACH,SAAS,CAAC,gBAAgB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM;IAMhD;;;;;;;OAOG;IACH,SAAS,CAAC,iBAAiB,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAc7E;;;;;;;;;;;;OAYG;IACH,SAAS,CAAC,mBAAmB,CAC3B,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,UAAU,GAAE,CAAC,IAAI,EAAE,MAAM,KAAK,OAAyC,GACtE,aAAa,GAAG,IAAI;IAiBvB;;;;;;;OAOG;IACH,SAAS,CAAC,cAAc,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,OAAO;IAK7D;;;;;;;;;;;;;;;;;OAiBG;IACH,SAAS,CAAC,wBAAwB,CAChC,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,UAAU,GAAE,CAAC,IAAI,EAAE,MAAM,KAAK,OAAyC,GACtE,OAAO;IASV;;;;;;OAMG;IACH,SAAS,CAAC,aAAa,CAAC,MAAM,EAAE,MAAM,GAAG,YAAY,GAAG,SAAS;IAIjE;;;;;;OAMG;IACH,SAAS,CAAC,SAAS,CAAC,MAAM,EAAE,MAAM,GAAG,OAAO;IAI5C;;OAEG;IACH,aAAa,CAAC,UAAU,EAAE,uBAAuB,GAAG,IAAI;IAIxD;;;;;;;OAOG;IACH,SAAS,CAAC,YAAY,CAAC,IAAI,EAAE,MAAM,GAAG,mBAAmB,GAAG,IAAI;IAahE;;;;;;;;;;;;;;;OAeG;IACH,SAAS,CAAC,oBAAoB,CAC5B,IAAI,EAAE,MAAM,EACZ,QAAQ,EAAE,MAAM,EAChB,MAAM,EAAE,MAAM,GACb,aAAa,GAAG,IAAI;IAgBvB;;OAEG;IACH,SAAS,CAAC,WAAW,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQvE;;;OAGG;IACH,SAAS,CAAC,gBAAgB,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAmC5E;;OAEG;IACH,SAAS,CAAC,SAAS,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQrE;;OAEG;IACH,SAAS,CAAC,SAAS,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQrE;;;OAGG;IACH,SAAS,CAAC,MAAM,CAAC,QAAQ,CAAC,mBAAmB,EAAE,SAAS,eAAe,EAAE,CAKvE;IAEF;;;;;;;;OAQG;IACH,SAAS,CAAC,gBAAgB,CACxB,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,SAAS,EAAE,SAAS,eAAe,EAAE,EACrC,cAAc,UAAQ,GACrB;QAAE,MAAM,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,CAAA;KAAE,GAAG,IAAI;IAuC5C;;;;;;;;OAQG;IACH,SAAS,CAAC,eAAe,CACvB,KAAK,EAAE,MAAM,EACb,QAAQ,EAAE,MAAM,EAChB,SAAS,UAAO,GACf;QAAE,MAAM,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,CAAA;KAAE,GAAG,IAAI;IAgC5C;;;;;;;;;;;;;OAaG;IACH,SAAS,CAAC,sBAAsB,CAC9B,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,eAAe,EAAE,SAAS,eAAe,EAAE,EAC3C,OAAO,GAAE;QAAE,SAAS,CAAC,EAAE,OAAO,CAAC;QAAC,cAAc,CAAC,EAAE,OAAO,CAAA;KAAO,GAC9D,aAAa,GAAG,IAAI;IAqBvB;;;OAGG;IACH,SAAS,CAAC,MAAM,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQlE;;;OAGG;IACH,SAAS,CAAC,cAAc,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAc1E;;;OAGG;IACH,SAAS,CAAC,WAAW,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAqBvE;;;;;;;;;;OAUG;IACH,SAAS,CAAC,oBAAoB,CAC5B,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,SAAS,EAAE,SAAS,MAAM,EAAE,GAC3B,aAAa,GAAG,IAAI;CAQxB;AAMD;;;;;;;;;;;;;GAaG;AACH,MAAM,WAAW,qBAAqB;IACpC,8BAA8B;IAC9B,QAAQ,EAAE,MAAM,CAAC;IACjB,sCAAsC;IACtC,SAAS,CAAC,EAAE,KAAK,GAAG,KAAK,CAAC;IAC1B,uEAAuE;IACvE,QAAQ,EAAE,MAAM,EAAE,CAAC;IACnB,wDAAwD;IACxD,aAAa,CAAC,EAAE,YAAY,EAAE,CAAC;IAC/B,wEAAwE;IACxE,cAAc,CAAC,EAAE,gBAAgB,CAAC;IAClC,uGAAuG;IACvG,gBAAgB,CAAC,EAAE,OAAO,CAAC;IAC3B,wDAAwD;IACxD,eAAe,CAAC,EAAE,OAAO,CAAC;IAC1B,6DAA6D;IAC7D,gBAAgB,CAAC,EAAE,cAAc,EAAE,CAAC;CACrC;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,qBAAqB,CAAC,MAAM,EAAE,qBAAqB,GAAG,iBAAiB,CA2CtF"}
@@ -641,7 +641,8 @@ function createTokenizerContext(tokenizer) {
641
641
  direction: tokenizer.direction,
642
642
  lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
643
643
  isKeyword: tokenizer.isKeyword.bind(tokenizer),
644
- isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
644
+ isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
645
+ ...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
645
646
  };
646
647
  if (tokenizer.normalizer) {
647
648
  return { ...ctx, normalizer: tokenizer.normalizer };
@@ -689,18 +690,94 @@ function createLatinCharClassifiers(letterPattern) {
689
690
 
690
691
  // src/core/tokenization/base-tokenizer.ts
691
692
  var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
693
+ var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
694
+ // Role-marker role names (profile.roleMarkers normalizeds)
695
+ "patient",
696
+ "destination",
697
+ "source",
698
+ "style",
699
+ "event",
700
+ "eventMarker",
701
+ "agent",
702
+ "goal",
703
+ "manner",
704
+ // Prepositional / positional modifier concepts matched via the role mechanism
705
+ // (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
706
+ // NOT here — they are pattern literals (see the note above).
707
+ "into",
708
+ "from",
709
+ "to",
710
+ "with",
711
+ "at",
712
+ "of",
713
+ "as",
714
+ "by",
715
+ "in",
716
+ "on",
717
+ "over",
718
+ "under",
719
+ "between",
720
+ "through",
721
+ "without"
722
+ ]);
723
+ var ENGLISH_DOM_EVENT_NAMES = [
724
+ "click",
725
+ "dblclick",
726
+ "input",
727
+ "change",
728
+ "submit",
729
+ "keydown",
730
+ "keyup",
731
+ "keypress",
732
+ "mousedown",
733
+ "mouseup",
734
+ "mouseover",
735
+ "mouseout",
736
+ "mouseenter",
737
+ "mouseleave",
738
+ "mousemove",
739
+ "pointerdown",
740
+ "pointerup",
741
+ "pointermove",
742
+ "focus",
743
+ "blur",
744
+ "load",
745
+ "resize",
746
+ "scroll"
747
+ ];
692
748
  var _BaseTokenizer = class _BaseTokenizer {
693
749
  constructor() {
694
750
  /** Keywords derived from profile, sorted longest-first for greedy matching */
695
751
  this.profileKeywords = [];
752
+ /**
753
+ * Space-containing profile keywords (multi-word phrases), longest-first.
754
+ * Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
755
+ * vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
756
+ * profile-driven replacement for the per-language hardcoded compound lists.
757
+ * Empty for no-space (CJK) languages, so they are unaffected.
758
+ */
759
+ this.multiWordKeywords = [];
696
760
  /** Map for O(1) keyword lookups by lowercase native word */
697
761
  this.profileKeywordMap = /* @__PURE__ */ new Map();
762
+ /**
763
+ * The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
764
+ * The keyword map is keyed by native word with last-wins insertion, so a
765
+ * duplicate native word inside the extras silently shadows the earlier entry
766
+ * (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
767
+ * positional expressions). Exposed so consistency tests can detect such
768
+ * intra-extras collisions, which are invisible in the deduplicated map.
769
+ */
770
+ this.rawExtraEntries = [];
698
771
  /**
699
772
  * Pluggable value extractors for domain-specific syntax.
700
773
  * When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
701
774
  */
702
775
  this.extractors = [];
703
776
  }
777
+ /** Raw extras as passed in, pre-dedup — for consistency tests. */
778
+ getExtraKeywordEntries() {
779
+ return this.rawExtraEntries;
780
+ }
704
781
  /**
705
782
  * Tokenize input string to token stream.
706
783
  * Delegates to extractor-based tokenization if extractors are registered,
@@ -769,6 +846,12 @@ var _BaseTokenizer = class _BaseTokenizer {
769
846
  pos++;
770
847
  }
771
848
  if (pos >= input.length) break;
849
+ const multiWord = this.tryMultiWordKeyword(input, pos);
850
+ if (multiWord) {
851
+ tokens.push(multiWord);
852
+ pos = multiWord.position.end;
853
+ continue;
854
+ }
772
855
  let extracted = false;
773
856
  for (const extractor of this.extractors) {
774
857
  if (extractor.canExtract(input, pos)) {
@@ -868,6 +951,7 @@ var _BaseTokenizer = class _BaseTokenizer {
868
951
  */
869
952
  initializeKeywordsFromProfile(profile, extras = []) {
870
953
  const keywordMap = /* @__PURE__ */ new Map();
954
+ this.rawExtraEntries = extras;
871
955
  if (profile.keywords) {
872
956
  for (const [normalized2, translation] of Object.entries(profile.keywords)) {
873
957
  keywordMap.set(translation.primary, {
@@ -911,12 +995,20 @@ var _BaseTokenizer = class _BaseTokenizer {
911
995
  keywordMap.set(native, { native, normalized: normalized2 });
912
996
  }
913
997
  }
998
+ for (const evt of ENGLISH_DOM_EVENT_NAMES) {
999
+ if (!keywordMap.has(evt)) {
1000
+ keywordMap.set(evt, { native: evt, normalized: evt });
1001
+ }
1002
+ }
914
1003
  for (const extra of extras) {
915
1004
  keywordMap.set(extra.native, extra);
916
1005
  }
917
1006
  this.profileKeywords = Array.from(keywordMap.values()).sort(
918
1007
  (a, b) => b.native.length - a.native.length
919
1008
  );
1009
+ this.multiWordKeywords = this.profileKeywords.filter(
1010
+ (k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
1011
+ );
920
1012
  this.profileKeywordMap = /* @__PURE__ */ new Map();
921
1013
  for (const keyword of this.profileKeywords) {
922
1014
  this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
@@ -958,6 +1050,35 @@ var _BaseTokenizer = class _BaseTokenizer {
958
1050
  }
959
1051
  return null;
960
1052
  }
1053
+ /**
1054
+ * Match the longest multi-word (space-containing) profile keyword at `pos`,
1055
+ * requiring the match to end at a word boundary. The profile-driven
1056
+ * counterpart of the per-language hardcoded compound lists (the hindi and
1057
+ * vietnamese keyword extractors). Returns a keyword token (with the normalized
1058
+ * form) or null. Case-sensitive against the stored native form, mirroring
1059
+ * `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
1060
+ * case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
1061
+ *
1062
+ * @param input - Input string
1063
+ * @param pos - Current position (must be a token-start boundary)
1064
+ * @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
1065
+ */
1066
+ tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
1067
+ if (this.multiWordKeywords.length === 0) return null;
1068
+ const rest = input.slice(pos);
1069
+ for (const entry of this.multiWordKeywords) {
1070
+ if (!rest.startsWith(entry.native)) continue;
1071
+ const after = input[pos + entry.native.length];
1072
+ if (after !== void 0 && isWordChar(after)) continue;
1073
+ return createToken(
1074
+ entry.native,
1075
+ "keyword",
1076
+ createPosition(pos, pos + entry.native.length),
1077
+ entry.normalized
1078
+ );
1079
+ }
1080
+ return null;
1081
+ }
961
1082
  /**
962
1083
  * Check if the remaining input starts with any known keyword.
963
1084
  * Useful for non-space languages to detect word boundaries.
@@ -970,6 +1091,32 @@ var _BaseTokenizer = class _BaseTokenizer {
970
1091
  const remaining = input.slice(pos);
971
1092
  return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
972
1093
  }
1094
+ /**
1095
+ * Check if a known keyword starts at the given position AND ends at a word
1096
+ * boundary (end of input or a non-word character).
1097
+ *
1098
+ * Space-delimited languages must use this (not `isKeywordStart`) for
1099
+ * word-walk break checks: the keyword table includes English canonical
1100
+ * fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
1101
+ * word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
1102
+ * "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
1103
+ * keep using `isKeywordStart`.
1104
+ *
1105
+ * @param input - Input string
1106
+ * @param pos - Current position
1107
+ * @param isWordChar - Language-specific word-character predicate; pass the
1108
+ * tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
1109
+ * counts as part of a word. Defaults to Unicode letters/digits/underscore.
1110
+ * @returns true if a keyword starts here and is not followed by a word char
1111
+ */
1112
+ isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
1113
+ const remaining = input.slice(pos);
1114
+ return this.profileKeywords.some((entry) => {
1115
+ if (!remaining.startsWith(entry.native)) return false;
1116
+ const after = input[pos + entry.native.length];
1117
+ return after === void 0 || !isWordChar(after);
1118
+ });
1119
+ }
973
1120
  /**
974
1121
  * Look up a keyword by native word (case-insensitive).
975
1122
  * O(1) lookup using the keyword map.