@lokascript/framework 2.5.1 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -662,7 +662,8 @@ function createTokenizerContext(tokenizer) {
662
662
  direction: tokenizer.direction,
663
663
  lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
664
664
  isKeyword: tokenizer.isKeyword.bind(tokenizer),
665
- isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
665
+ isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
666
+ ...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
666
667
  };
667
668
  if (tokenizer.normalizer) {
668
669
  return { ...ctx, normalizer: tokenizer.normalizer };
@@ -710,18 +711,94 @@ function createLatinCharClassifiers(letterPattern) {
710
711
 
711
712
  // src/core/tokenization/base-tokenizer.ts
712
713
  var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
714
+ var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
715
+ // Role-marker role names (profile.roleMarkers normalizeds)
716
+ "patient",
717
+ "destination",
718
+ "source",
719
+ "style",
720
+ "event",
721
+ "eventMarker",
722
+ "agent",
723
+ "goal",
724
+ "manner",
725
+ // Prepositional / positional modifier concepts matched via the role mechanism
726
+ // (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
727
+ // NOT here — they are pattern literals (see the note above).
728
+ "into",
729
+ "from",
730
+ "to",
731
+ "with",
732
+ "at",
733
+ "of",
734
+ "as",
735
+ "by",
736
+ "in",
737
+ "on",
738
+ "over",
739
+ "under",
740
+ "between",
741
+ "through",
742
+ "without"
743
+ ]);
744
+ var ENGLISH_DOM_EVENT_NAMES = [
745
+ "click",
746
+ "dblclick",
747
+ "input",
748
+ "change",
749
+ "submit",
750
+ "keydown",
751
+ "keyup",
752
+ "keypress",
753
+ "mousedown",
754
+ "mouseup",
755
+ "mouseover",
756
+ "mouseout",
757
+ "mouseenter",
758
+ "mouseleave",
759
+ "mousemove",
760
+ "pointerdown",
761
+ "pointerup",
762
+ "pointermove",
763
+ "focus",
764
+ "blur",
765
+ "load",
766
+ "resize",
767
+ "scroll"
768
+ ];
713
769
  var _BaseTokenizer = class _BaseTokenizer {
714
770
  constructor() {
715
771
  /** Keywords derived from profile, sorted longest-first for greedy matching */
716
772
  this.profileKeywords = [];
773
+ /**
774
+ * Space-containing profile keywords (multi-word phrases), longest-first.
775
+ * Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
776
+ * vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
777
+ * profile-driven replacement for the per-language hardcoded compound lists.
778
+ * Empty for no-space (CJK) languages, so they are unaffected.
779
+ */
780
+ this.multiWordKeywords = [];
717
781
  /** Map for O(1) keyword lookups by lowercase native word */
718
782
  this.profileKeywordMap = /* @__PURE__ */ new Map();
783
+ /**
784
+ * The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
785
+ * The keyword map is keyed by native word with last-wins insertion, so a
786
+ * duplicate native word inside the extras silently shadows the earlier entry
787
+ * (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
788
+ * positional expressions). Exposed so consistency tests can detect such
789
+ * intra-extras collisions, which are invisible in the deduplicated map.
790
+ */
791
+ this.rawExtraEntries = [];
719
792
  /**
720
793
  * Pluggable value extractors for domain-specific syntax.
721
794
  * When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
722
795
  */
723
796
  this.extractors = [];
724
797
  }
798
+ /** Raw extras as passed in, pre-dedup — for consistency tests. */
799
+ getExtraKeywordEntries() {
800
+ return this.rawExtraEntries;
801
+ }
725
802
  /**
726
803
  * Tokenize input string to token stream.
727
804
  * Delegates to extractor-based tokenization if extractors are registered,
@@ -790,6 +867,12 @@ var _BaseTokenizer = class _BaseTokenizer {
790
867
  pos++;
791
868
  }
792
869
  if (pos >= input.length) break;
870
+ const multiWord = this.tryMultiWordKeyword(input, pos);
871
+ if (multiWord) {
872
+ tokens.push(multiWord);
873
+ pos = multiWord.position.end;
874
+ continue;
875
+ }
793
876
  let extracted = false;
794
877
  for (const extractor of this.extractors) {
795
878
  if (extractor.canExtract(input, pos)) {
@@ -889,6 +972,7 @@ var _BaseTokenizer = class _BaseTokenizer {
889
972
  */
890
973
  initializeKeywordsFromProfile(profile, extras = []) {
891
974
  const keywordMap = /* @__PURE__ */ new Map();
975
+ this.rawExtraEntries = extras;
892
976
  if (profile.keywords) {
893
977
  for (const [normalized2, translation] of Object.entries(profile.keywords)) {
894
978
  keywordMap.set(translation.primary, {
@@ -932,12 +1016,20 @@ var _BaseTokenizer = class _BaseTokenizer {
932
1016
  keywordMap.set(native, { native, normalized: normalized2 });
933
1017
  }
934
1018
  }
1019
+ for (const evt of ENGLISH_DOM_EVENT_NAMES) {
1020
+ if (!keywordMap.has(evt)) {
1021
+ keywordMap.set(evt, { native: evt, normalized: evt });
1022
+ }
1023
+ }
935
1024
  for (const extra of extras) {
936
1025
  keywordMap.set(extra.native, extra);
937
1026
  }
938
1027
  this.profileKeywords = Array.from(keywordMap.values()).sort(
939
1028
  (a, b) => b.native.length - a.native.length
940
1029
  );
1030
+ this.multiWordKeywords = this.profileKeywords.filter(
1031
+ (k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
1032
+ );
941
1033
  this.profileKeywordMap = /* @__PURE__ */ new Map();
942
1034
  for (const keyword of this.profileKeywords) {
943
1035
  this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
@@ -979,6 +1071,35 @@ var _BaseTokenizer = class _BaseTokenizer {
979
1071
  }
980
1072
  return null;
981
1073
  }
1074
+ /**
1075
+ * Match the longest multi-word (space-containing) profile keyword at `pos`,
1076
+ * requiring the match to end at a word boundary. The profile-driven
1077
+ * counterpart of the per-language hardcoded compound lists (the hindi and
1078
+ * vietnamese keyword extractors). Returns a keyword token (with the normalized
1079
+ * form) or null. Case-sensitive against the stored native form, mirroring
1080
+ * `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
1081
+ * case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
1082
+ *
1083
+ * @param input - Input string
1084
+ * @param pos - Current position (must be a token-start boundary)
1085
+ * @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
1086
+ */
1087
+ tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
1088
+ if (this.multiWordKeywords.length === 0) return null;
1089
+ const rest = input.slice(pos);
1090
+ for (const entry of this.multiWordKeywords) {
1091
+ if (!rest.startsWith(entry.native)) continue;
1092
+ const after = input[pos + entry.native.length];
1093
+ if (after !== void 0 && isWordChar(after)) continue;
1094
+ return createToken(
1095
+ entry.native,
1096
+ "keyword",
1097
+ createPosition(pos, pos + entry.native.length),
1098
+ entry.normalized
1099
+ );
1100
+ }
1101
+ return null;
1102
+ }
982
1103
  /**
983
1104
  * Check if the remaining input starts with any known keyword.
984
1105
  * Useful for non-space languages to detect word boundaries.
@@ -991,6 +1112,32 @@ var _BaseTokenizer = class _BaseTokenizer {
991
1112
  const remaining = input.slice(pos);
992
1113
  return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
993
1114
  }
1115
+ /**
1116
+ * Check if a known keyword starts at the given position AND ends at a word
1117
+ * boundary (end of input or a non-word character).
1118
+ *
1119
+ * Space-delimited languages must use this (not `isKeywordStart`) for
1120
+ * word-walk break checks: the keyword table includes English canonical
1121
+ * fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
1122
+ * word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
1123
+ * "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
1124
+ * keep using `isKeywordStart`.
1125
+ *
1126
+ * @param input - Input string
1127
+ * @param pos - Current position
1128
+ * @param isWordChar - Language-specific word-character predicate; pass the
1129
+ * tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
1130
+ * counts as part of a word. Defaults to Unicode letters/digits/underscore.
1131
+ * @returns true if a keyword starts here and is not followed by a word char
1132
+ */
1133
+ isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
1134
+ const remaining = input.slice(pos);
1135
+ return this.profileKeywords.some((entry) => {
1136
+ if (!remaining.startsWith(entry.native)) return false;
1137
+ const after = input[pos + entry.native.length];
1138
+ return after === void 0 || !isWordChar(after);
1139
+ });
1140
+ }
994
1141
  /**
995
1142
  * Look up a keyword by native word (case-insensitive).
996
1143
  * O(1) lookup using the keyword map.