@lokascript/framework 2.5.0 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -2051,7 +2051,8 @@ function createTokenizerContext(tokenizer) {
2051
2051
  direction: tokenizer.direction,
2052
2052
  lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
2053
2053
  isKeyword: tokenizer.isKeyword.bind(tokenizer),
2054
- isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
2054
+ isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
2055
+ ...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
2055
2056
  };
2056
2057
  if (tokenizer.normalizer) {
2057
2058
  return { ...ctx, normalizer: tokenizer.normalizer };
@@ -5025,18 +5026,94 @@ function createLatinCharClassifiers(letterPattern) {
5025
5026
 
5026
5027
  // src/core/tokenization/base-tokenizer.ts
5027
5028
  var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
5029
+ var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
5030
+ // Role-marker role names (profile.roleMarkers normalizeds)
5031
+ "patient",
5032
+ "destination",
5033
+ "source",
5034
+ "style",
5035
+ "event",
5036
+ "eventMarker",
5037
+ "agent",
5038
+ "goal",
5039
+ "manner",
5040
+ // Prepositional / positional modifier concepts matched via the role mechanism
5041
+ // (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
5042
+ // NOT here — they are pattern literals (see the note above).
5043
+ "into",
5044
+ "from",
5045
+ "to",
5046
+ "with",
5047
+ "at",
5048
+ "of",
5049
+ "as",
5050
+ "by",
5051
+ "in",
5052
+ "on",
5053
+ "over",
5054
+ "under",
5055
+ "between",
5056
+ "through",
5057
+ "without"
5058
+ ]);
5059
+ var ENGLISH_DOM_EVENT_NAMES = [
5060
+ "click",
5061
+ "dblclick",
5062
+ "input",
5063
+ "change",
5064
+ "submit",
5065
+ "keydown",
5066
+ "keyup",
5067
+ "keypress",
5068
+ "mousedown",
5069
+ "mouseup",
5070
+ "mouseover",
5071
+ "mouseout",
5072
+ "mouseenter",
5073
+ "mouseleave",
5074
+ "mousemove",
5075
+ "pointerdown",
5076
+ "pointerup",
5077
+ "pointermove",
5078
+ "focus",
5079
+ "blur",
5080
+ "load",
5081
+ "resize",
5082
+ "scroll"
5083
+ ];
5028
5084
  var _BaseTokenizer = class _BaseTokenizer {
5029
5085
  constructor() {
5030
5086
  /** Keywords derived from profile, sorted longest-first for greedy matching */
5031
5087
  this.profileKeywords = [];
5088
+ /**
5089
+ * Space-containing profile keywords (multi-word phrases), longest-first.
5090
+ * Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
5091
+ * vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
5092
+ * profile-driven replacement for the per-language hardcoded compound lists.
5093
+ * Empty for no-space (CJK) languages, so they are unaffected.
5094
+ */
5095
+ this.multiWordKeywords = [];
5032
5096
  /** Map for O(1) keyword lookups by lowercase native word */
5033
5097
  this.profileKeywordMap = /* @__PURE__ */ new Map();
5098
+ /**
5099
+ * The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
5100
+ * The keyword map is keyed by native word with last-wins insertion, so a
5101
+ * duplicate native word inside the extras silently shadows the earlier entry
5102
+ * (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
5103
+ * positional expressions). Exposed so consistency tests can detect such
5104
+ * intra-extras collisions, which are invisible in the deduplicated map.
5105
+ */
5106
+ this.rawExtraEntries = [];
5034
5107
  /**
5035
5108
  * Pluggable value extractors for domain-specific syntax.
5036
5109
  * When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
5037
5110
  */
5038
5111
  this.extractors = [];
5039
5112
  }
5113
+ /** Raw extras as passed in, pre-dedup — for consistency tests. */
5114
+ getExtraKeywordEntries() {
5115
+ return this.rawExtraEntries;
5116
+ }
5040
5117
  /**
5041
5118
  * Tokenize input string to token stream.
5042
5119
  * Delegates to extractor-based tokenization if extractors are registered,
@@ -5105,6 +5182,12 @@ var _BaseTokenizer = class _BaseTokenizer {
5105
5182
  pos++;
5106
5183
  }
5107
5184
  if (pos >= input.length) break;
5185
+ const multiWord = this.tryMultiWordKeyword(input, pos);
5186
+ if (multiWord) {
5187
+ tokens.push(multiWord);
5188
+ pos = multiWord.position.end;
5189
+ continue;
5190
+ }
5108
5191
  let extracted = false;
5109
5192
  for (const extractor of this.extractors) {
5110
5193
  if (extractor.canExtract(input, pos)) {
@@ -5204,6 +5287,7 @@ var _BaseTokenizer = class _BaseTokenizer {
5204
5287
  */
5205
5288
  initializeKeywordsFromProfile(profile, extras = []) {
5206
5289
  const keywordMap = /* @__PURE__ */ new Map();
5290
+ this.rawExtraEntries = extras;
5207
5291
  if (profile.keywords) {
5208
5292
  for (const [normalized2, translation] of Object.entries(profile.keywords)) {
5209
5293
  keywordMap.set(translation.primary, {
@@ -5247,12 +5331,20 @@ var _BaseTokenizer = class _BaseTokenizer {
5247
5331
  keywordMap.set(native, { native, normalized: normalized2 });
5248
5332
  }
5249
5333
  }
5334
+ for (const evt of ENGLISH_DOM_EVENT_NAMES) {
5335
+ if (!keywordMap.has(evt)) {
5336
+ keywordMap.set(evt, { native: evt, normalized: evt });
5337
+ }
5338
+ }
5250
5339
  for (const extra of extras) {
5251
5340
  keywordMap.set(extra.native, extra);
5252
5341
  }
5253
5342
  this.profileKeywords = Array.from(keywordMap.values()).sort(
5254
5343
  (a, b) => b.native.length - a.native.length
5255
5344
  );
5345
+ this.multiWordKeywords = this.profileKeywords.filter(
5346
+ (k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
5347
+ );
5256
5348
  this.profileKeywordMap = /* @__PURE__ */ new Map();
5257
5349
  for (const keyword of this.profileKeywords) {
5258
5350
  this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
@@ -5294,6 +5386,35 @@ var _BaseTokenizer = class _BaseTokenizer {
5294
5386
  }
5295
5387
  return null;
5296
5388
  }
5389
+ /**
5390
+ * Match the longest multi-word (space-containing) profile keyword at `pos`,
5391
+ * requiring the match to end at a word boundary. The profile-driven
5392
+ * counterpart of the per-language hardcoded compound lists (the hindi and
5393
+ * vietnamese keyword extractors). Returns a keyword token (with the normalized
5394
+ * form) or null. Case-sensitive against the stored native form, mirroring
5395
+ * `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
5396
+ * case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
5397
+ *
5398
+ * @param input - Input string
5399
+ * @param pos - Current position (must be a token-start boundary)
5400
+ * @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
5401
+ */
5402
+ tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
5403
+ if (this.multiWordKeywords.length === 0) return null;
5404
+ const rest = input.slice(pos);
5405
+ for (const entry of this.multiWordKeywords) {
5406
+ if (!rest.startsWith(entry.native)) continue;
5407
+ const after = input[pos + entry.native.length];
5408
+ if (after !== void 0 && isWordChar(after)) continue;
5409
+ return createToken(
5410
+ entry.native,
5411
+ "keyword",
5412
+ createPosition(pos, pos + entry.native.length),
5413
+ entry.normalized
5414
+ );
5415
+ }
5416
+ return null;
5417
+ }
5297
5418
  /**
5298
5419
  * Check if the remaining input starts with any known keyword.
5299
5420
  * Useful for non-space languages to detect word boundaries.
@@ -5306,6 +5427,32 @@ var _BaseTokenizer = class _BaseTokenizer {
5306
5427
  const remaining = input.slice(pos);
5307
5428
  return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
5308
5429
  }
5430
+ /**
5431
+ * Check if a known keyword starts at the given position AND ends at a word
5432
+ * boundary (end of input or a non-word character).
5433
+ *
5434
+ * Space-delimited languages must use this (not `isKeywordStart`) for
5435
+ * word-walk break checks: the keyword table includes English canonical
5436
+ * fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
5437
+ * word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
5438
+ * "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
5439
+ * keep using `isKeywordStart`.
5440
+ *
5441
+ * @param input - Input string
5442
+ * @param pos - Current position
5443
+ * @param isWordChar - Language-specific word-character predicate; pass the
5444
+ * tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
5445
+ * counts as part of a word. Defaults to Unicode letters/digits/underscore.
5446
+ * @returns true if a keyword starts here and is not followed by a word char
5447
+ */
5448
+ isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
5449
+ const remaining = input.slice(pos);
5450
+ return this.profileKeywords.some((entry) => {
5451
+ if (!remaining.startsWith(entry.native)) return false;
5452
+ const after = input[pos + entry.native.length];
5453
+ return after === void 0 || !isWordChar(after);
5454
+ });
5455
+ }
5309
5456
  /**
5310
5457
  * Look up a keyword by native word (case-insensitive).
5311
5458
  * O(1) lookup using the keyword map.