@lokascript/framework 2.5.1 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1908,7 +1908,8 @@ function createTokenizerContext(tokenizer) {
1908
1908
  direction: tokenizer.direction,
1909
1909
  lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
1910
1910
  isKeyword: tokenizer.isKeyword.bind(tokenizer),
1911
- isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
1911
+ isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
1912
+ ...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
1912
1913
  };
1913
1914
  if (tokenizer.normalizer) {
1914
1915
  return { ...ctx, normalizer: tokenizer.normalizer };
@@ -4897,18 +4898,94 @@ function createLatinCharClassifiers(letterPattern) {
4897
4898
 
4898
4899
  // src/core/tokenization/base-tokenizer.ts
4899
4900
  var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
4901
+ var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
4902
+ // Role-marker role names (profile.roleMarkers normalizeds)
4903
+ "patient",
4904
+ "destination",
4905
+ "source",
4906
+ "style",
4907
+ "event",
4908
+ "eventMarker",
4909
+ "agent",
4910
+ "goal",
4911
+ "manner",
4912
+ // Prepositional / positional modifier concepts matched via the role mechanism
4913
+ // (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
4914
+ // NOT here — they are pattern literals (see the note above).
4915
+ "into",
4916
+ "from",
4917
+ "to",
4918
+ "with",
4919
+ "at",
4920
+ "of",
4921
+ "as",
4922
+ "by",
4923
+ "in",
4924
+ "on",
4925
+ "over",
4926
+ "under",
4927
+ "between",
4928
+ "through",
4929
+ "without"
4930
+ ]);
4931
+ var ENGLISH_DOM_EVENT_NAMES = [
4932
+ "click",
4933
+ "dblclick",
4934
+ "input",
4935
+ "change",
4936
+ "submit",
4937
+ "keydown",
4938
+ "keyup",
4939
+ "keypress",
4940
+ "mousedown",
4941
+ "mouseup",
4942
+ "mouseover",
4943
+ "mouseout",
4944
+ "mouseenter",
4945
+ "mouseleave",
4946
+ "mousemove",
4947
+ "pointerdown",
4948
+ "pointerup",
4949
+ "pointermove",
4950
+ "focus",
4951
+ "blur",
4952
+ "load",
4953
+ "resize",
4954
+ "scroll"
4955
+ ];
4900
4956
  var _BaseTokenizer = class _BaseTokenizer {
4901
4957
  constructor() {
4902
4958
  /** Keywords derived from profile, sorted longest-first for greedy matching */
4903
4959
  this.profileKeywords = [];
4960
+ /**
4961
+ * Space-containing profile keywords (multi-word phrases), longest-first.
4962
+ * Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
4963
+ * vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
4964
+ * profile-driven replacement for the per-language hardcoded compound lists.
4965
+ * Empty for no-space (CJK) languages, so they are unaffected.
4966
+ */
4967
+ this.multiWordKeywords = [];
4904
4968
  /** Map for O(1) keyword lookups by lowercase native word */
4905
4969
  this.profileKeywordMap = /* @__PURE__ */ new Map();
4970
+ /**
4971
+ * The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
4972
+ * The keyword map is keyed by native word with last-wins insertion, so a
4973
+ * duplicate native word inside the extras silently shadows the earlier entry
4974
+ * (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
4975
+ * positional expressions). Exposed so consistency tests can detect such
4976
+ * intra-extras collisions, which are invisible in the deduplicated map.
4977
+ */
4978
+ this.rawExtraEntries = [];
4906
4979
  /**
4907
4980
  * Pluggable value extractors for domain-specific syntax.
4908
4981
  * When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
4909
4982
  */
4910
4983
  this.extractors = [];
4911
4984
  }
4985
+ /** Raw extras as passed in, pre-dedup — for consistency tests. */
4986
+ getExtraKeywordEntries() {
4987
+ return this.rawExtraEntries;
4988
+ }
4912
4989
  /**
4913
4990
  * Tokenize input string to token stream.
4914
4991
  * Delegates to extractor-based tokenization if extractors are registered,
@@ -4977,6 +5054,12 @@ var _BaseTokenizer = class _BaseTokenizer {
4977
5054
  pos++;
4978
5055
  }
4979
5056
  if (pos >= input.length) break;
5057
+ const multiWord = this.tryMultiWordKeyword(input, pos);
5058
+ if (multiWord) {
5059
+ tokens.push(multiWord);
5060
+ pos = multiWord.position.end;
5061
+ continue;
5062
+ }
4980
5063
  let extracted = false;
4981
5064
  for (const extractor of this.extractors) {
4982
5065
  if (extractor.canExtract(input, pos)) {
@@ -5076,6 +5159,7 @@ var _BaseTokenizer = class _BaseTokenizer {
5076
5159
  */
5077
5160
  initializeKeywordsFromProfile(profile, extras = []) {
5078
5161
  const keywordMap = /* @__PURE__ */ new Map();
5162
+ this.rawExtraEntries = extras;
5079
5163
  if (profile.keywords) {
5080
5164
  for (const [normalized2, translation] of Object.entries(profile.keywords)) {
5081
5165
  keywordMap.set(translation.primary, {
@@ -5119,12 +5203,20 @@ var _BaseTokenizer = class _BaseTokenizer {
5119
5203
  keywordMap.set(native, { native, normalized: normalized2 });
5120
5204
  }
5121
5205
  }
5206
+ for (const evt of ENGLISH_DOM_EVENT_NAMES) {
5207
+ if (!keywordMap.has(evt)) {
5208
+ keywordMap.set(evt, { native: evt, normalized: evt });
5209
+ }
5210
+ }
5122
5211
  for (const extra of extras) {
5123
5212
  keywordMap.set(extra.native, extra);
5124
5213
  }
5125
5214
  this.profileKeywords = Array.from(keywordMap.values()).sort(
5126
5215
  (a, b) => b.native.length - a.native.length
5127
5216
  );
5217
+ this.multiWordKeywords = this.profileKeywords.filter(
5218
+ (k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
5219
+ );
5128
5220
  this.profileKeywordMap = /* @__PURE__ */ new Map();
5129
5221
  for (const keyword of this.profileKeywords) {
5130
5222
  this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
@@ -5166,6 +5258,35 @@ var _BaseTokenizer = class _BaseTokenizer {
5166
5258
  }
5167
5259
  return null;
5168
5260
  }
5261
+ /**
5262
+ * Match the longest multi-word (space-containing) profile keyword at `pos`,
5263
+ * requiring the match to end at a word boundary. The profile-driven
5264
+ * counterpart of the per-language hardcoded compound lists (the hindi and
5265
+ * vietnamese keyword extractors). Returns a keyword token (with the normalized
5266
+ * form) or null. Case-sensitive against the stored native form, mirroring
5267
+ * `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
5268
+ * case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
5269
+ *
5270
+ * @param input - Input string
5271
+ * @param pos - Current position (must be a token-start boundary)
5272
+ * @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
5273
+ */
5274
+ tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
5275
+ if (this.multiWordKeywords.length === 0) return null;
5276
+ const rest = input.slice(pos);
5277
+ for (const entry of this.multiWordKeywords) {
5278
+ if (!rest.startsWith(entry.native)) continue;
5279
+ const after = input[pos + entry.native.length];
5280
+ if (after !== void 0 && isWordChar(after)) continue;
5281
+ return createToken(
5282
+ entry.native,
5283
+ "keyword",
5284
+ createPosition(pos, pos + entry.native.length),
5285
+ entry.normalized
5286
+ );
5287
+ }
5288
+ return null;
5289
+ }
5169
5290
  /**
5170
5291
  * Check if the remaining input starts with any known keyword.
5171
5292
  * Useful for non-space languages to detect word boundaries.
@@ -5178,6 +5299,32 @@ var _BaseTokenizer = class _BaseTokenizer {
5178
5299
  const remaining = input.slice(pos);
5179
5300
  return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
5180
5301
  }
5302
+ /**
5303
+ * Check if a known keyword starts at the given position AND ends at a word
5304
+ * boundary (end of input or a non-word character).
5305
+ *
5306
+ * Space-delimited languages must use this (not `isKeywordStart`) for
5307
+ * word-walk break checks: the keyword table includes English canonical
5308
+ * fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
5309
+ * word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
5310
+ * "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
5311
+ * keep using `isKeywordStart`.
5312
+ *
5313
+ * @param input - Input string
5314
+ * @param pos - Current position
5315
+ * @param isWordChar - Language-specific word-character predicate; pass the
5316
+ * tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
5317
+ * counts as part of a word. Defaults to Unicode letters/digits/underscore.
5318
+ * @returns true if a keyword starts here and is not followed by a word char
5319
+ */
5320
+ isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
5321
+ const remaining = input.slice(pos);
5322
+ return this.profileKeywords.some((entry) => {
5323
+ if (!remaining.startsWith(entry.native)) return false;
5324
+ const after = input[pos + entry.native.length];
5325
+ return after === void 0 || !isWordChar(after);
5326
+ });
5327
+ }
5181
5328
  /**
5182
5329
  * Look up a keyword by native word (case-insensitive).
5183
5330
  * O(1) lookup using the keyword map.