@lokascript/framework 2.5.1 → 2.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/dist/core/index.js +148 -1
  2. package/dist/core/index.js.map +1 -1
  3. package/dist/core/tokenization/base-tokenizer.d.ts +52 -0
  4. package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
  5. package/dist/core/tokenization/index.js +148 -1
  6. package/dist/core/tokenization/index.js.map +1 -1
  7. package/dist/index.cjs +269 -17
  8. package/dist/index.cjs.map +1 -1
  9. package/dist/index.d.ts +2 -0
  10. package/dist/index.d.ts.map +1 -1
  11. package/dist/index.js +265 -17
  12. package/dist/index.js.map +1 -1
  13. package/dist/interfaces/value-extractor.d.ts +11 -0
  14. package/dist/interfaces/value-extractor.d.ts.map +1 -1
  15. package/dist/multilingual/builders.d.ts +88 -0
  16. package/dist/multilingual/builders.d.ts.map +1 -0
  17. package/dist/multilingual/index.d.ts +7 -3
  18. package/dist/multilingual/index.d.ts.map +1 -1
  19. package/dist/multilingual/index.js +1549 -0
  20. package/dist/multilingual/index.js.map +1 -1
  21. package/dist/multilingual/types.d.ts +117 -0
  22. package/dist/multilingual/types.d.ts.map +1 -0
  23. package/dist/testing/index.js +12 -4
  24. package/dist/testing/index.js.map +1 -1
  25. package/package.json +4 -4
  26. package/src/core/tokenization/base-tokenizer.ts +219 -0
  27. package/src/core/tokenization/keyword-boundary.test.ts +73 -0
  28. package/src/index.ts +20 -0
  29. package/src/interfaces/value-extractor.ts +23 -0
  30. package/src/multilingual/bridge.test.ts +441 -0
  31. package/src/multilingual/builders.ts +224 -0
  32. package/src/multilingual/index.ts +23 -4
  33. package/src/multilingual/types.ts +121 -0
  34. package/src/core/pattern-matching/pattern-matcher.ts.backup +0 -1267
package/dist/index.cjs CHANGED
@@ -47,7 +47,10 @@ __export(index_exports, {
47
47
  WhitespaceExtractor: () => WhitespaceExtractor,
48
48
  accumulateBlocks: () => accumulateBlocks,
49
49
  buildDisambiguation: () => buildDisambiguation,
50
+ buildDomainTokenizer: () => buildDomainTokenizer,
50
51
  buildFeedback: () => buildFeedback,
52
+ buildLanguageConfig: () => buildLanguageConfig,
53
+ buildPatternProfile: () => buildPatternProfile,
51
54
  buildPhrase: () => buildPhrase,
52
55
  buildTablesFromProfiles: () => buildTablesFromProfiles,
53
56
  combineClassifiers: () => combineClassifiers,
@@ -78,6 +81,7 @@ __export(index_exports, {
78
81
  createUnicodeRangeClassifier: () => createUnicodeRangeClassifier,
79
82
  defineCommand: () => import_intent6.defineCommand,
80
83
  defineRole: () => import_intent6.defineRole,
84
+ deriveRoleMarkers: () => deriveRoleMarkers,
81
85
  detectWordOrders: () => detectWordOrders,
82
86
  extractCssSelector: () => extractCssSelector,
83
87
  extractNumber: () => extractNumber,
@@ -2051,7 +2055,8 @@ function createTokenizerContext(tokenizer) {
2051
2055
  direction: tokenizer.direction,
2052
2056
  lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
2053
2057
  isKeyword: tokenizer.isKeyword.bind(tokenizer),
2054
- isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
2058
+ isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
2059
+ ...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
2055
2060
  };
2056
2061
  if (tokenizer.normalizer) {
2057
2062
  return { ...ctx, normalizer: tokenizer.normalizer };
@@ -5007,36 +5012,96 @@ function withDefaultExtractors(tokenizer) {
5007
5012
  return tokenizer;
5008
5013
  }
5009
5014
 
5010
- // src/core/tokenization/char-classifiers.ts
5011
- function createUnicodeRangeClassifier(ranges) {
5012
- return (char) => {
5013
- const code = char.charCodeAt(0);
5014
- return ranges.some(([start, end]) => code >= start && code <= end);
5015
- };
5016
- }
5017
- function combineClassifiers(...classifiers) {
5018
- return (char) => classifiers.some((fn) => fn(char));
5019
- }
5020
- function createLatinCharClassifiers(letterPattern) {
5021
- const isLetter = (char) => letterPattern.test(char);
5022
- const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
5023
- return { isLetter, isIdentifierChar };
5024
- }
5025
-
5026
5015
  // src/core/tokenization/base-tokenizer.ts
5027
5016
  var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
5017
+ var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
5018
+ // Role-marker role names (profile.roleMarkers normalizeds)
5019
+ "patient",
5020
+ "destination",
5021
+ "source",
5022
+ "style",
5023
+ "event",
5024
+ "eventMarker",
5025
+ "agent",
5026
+ "goal",
5027
+ "manner",
5028
+ // Prepositional / positional modifier concepts matched via the role mechanism
5029
+ // (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
5030
+ // NOT here — they are pattern literals (see the note above).
5031
+ "into",
5032
+ "from",
5033
+ "to",
5034
+ "with",
5035
+ "at",
5036
+ "of",
5037
+ "as",
5038
+ "by",
5039
+ "in",
5040
+ "on",
5041
+ "over",
5042
+ "under",
5043
+ "between",
5044
+ "through",
5045
+ "without"
5046
+ ]);
5047
+ var ENGLISH_DOM_EVENT_NAMES = [
5048
+ "click",
5049
+ "dblclick",
5050
+ "input",
5051
+ "change",
5052
+ "submit",
5053
+ "keydown",
5054
+ "keyup",
5055
+ "keypress",
5056
+ "mousedown",
5057
+ "mouseup",
5058
+ "mouseover",
5059
+ "mouseout",
5060
+ "mouseenter",
5061
+ "mouseleave",
5062
+ "mousemove",
5063
+ "pointerdown",
5064
+ "pointerup",
5065
+ "pointermove",
5066
+ "focus",
5067
+ "blur",
5068
+ "load",
5069
+ "resize",
5070
+ "scroll"
5071
+ ];
5028
5072
  var _BaseTokenizer = class _BaseTokenizer {
5029
5073
  constructor() {
5030
5074
  /** Keywords derived from profile, sorted longest-first for greedy matching */
5031
5075
  this.profileKeywords = [];
5076
+ /**
5077
+ * Space-containing profile keywords (multi-word phrases), longest-first.
5078
+ * Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
5079
+ * vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
5080
+ * profile-driven replacement for the per-language hardcoded compound lists.
5081
+ * Empty for no-space (CJK) languages, so they are unaffected.
5082
+ */
5083
+ this.multiWordKeywords = [];
5032
5084
  /** Map for O(1) keyword lookups by lowercase native word */
5033
5085
  this.profileKeywordMap = /* @__PURE__ */ new Map();
5086
+ /**
5087
+ * The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
5088
+ * The keyword map is keyed by native word with last-wins insertion, so a
5089
+ * duplicate native word inside the extras silently shadows the earlier entry
5090
+ * (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
5091
+ * positional expressions). Exposed so consistency tests can detect such
5092
+ * intra-extras collisions, which are invisible in the deduplicated map.
5093
+ */
5094
+ this.rawExtraEntries = [];
5034
5095
  /**
5035
5096
  * Pluggable value extractors for domain-specific syntax.
5036
5097
  * When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
5037
5098
  */
5038
5099
  this.extractors = [];
5039
5100
  }
5101
+ /** Raw extras as passed in, pre-dedup — for consistency tests. */
5102
+ getExtraKeywordEntries() {
5103
+ return this.rawExtraEntries;
5104
+ }
5040
5105
  /**
5041
5106
  * Tokenize input string to token stream.
5042
5107
  * Delegates to extractor-based tokenization if extractors are registered,
@@ -5105,6 +5170,12 @@ var _BaseTokenizer = class _BaseTokenizer {
5105
5170
  pos++;
5106
5171
  }
5107
5172
  if (pos >= input.length) break;
5173
+ const multiWord = this.tryMultiWordKeyword(input, pos);
5174
+ if (multiWord) {
5175
+ tokens.push(multiWord);
5176
+ pos = multiWord.position.end;
5177
+ continue;
5178
+ }
5108
5179
  let extracted = false;
5109
5180
  for (const extractor of this.extractors) {
5110
5181
  if (extractor.canExtract(input, pos)) {
@@ -5204,6 +5275,7 @@ var _BaseTokenizer = class _BaseTokenizer {
5204
5275
  */
5205
5276
  initializeKeywordsFromProfile(profile, extras = []) {
5206
5277
  const keywordMap = /* @__PURE__ */ new Map();
5278
+ this.rawExtraEntries = extras;
5207
5279
  if (profile.keywords) {
5208
5280
  for (const [normalized2, translation] of Object.entries(profile.keywords)) {
5209
5281
  keywordMap.set(translation.primary, {
@@ -5247,12 +5319,20 @@ var _BaseTokenizer = class _BaseTokenizer {
5247
5319
  keywordMap.set(native, { native, normalized: normalized2 });
5248
5320
  }
5249
5321
  }
5322
+ for (const evt of ENGLISH_DOM_EVENT_NAMES) {
5323
+ if (!keywordMap.has(evt)) {
5324
+ keywordMap.set(evt, { native: evt, normalized: evt });
5325
+ }
5326
+ }
5250
5327
  for (const extra of extras) {
5251
5328
  keywordMap.set(extra.native, extra);
5252
5329
  }
5253
5330
  this.profileKeywords = Array.from(keywordMap.values()).sort(
5254
5331
  (a, b) => b.native.length - a.native.length
5255
5332
  );
5333
+ this.multiWordKeywords = this.profileKeywords.filter(
5334
+ (k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
5335
+ );
5256
5336
  this.profileKeywordMap = /* @__PURE__ */ new Map();
5257
5337
  for (const keyword of this.profileKeywords) {
5258
5338
  this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
@@ -5294,6 +5374,35 @@ var _BaseTokenizer = class _BaseTokenizer {
5294
5374
  }
5295
5375
  return null;
5296
5376
  }
5377
+ /**
5378
+ * Match the longest multi-word (space-containing) profile keyword at `pos`,
5379
+ * requiring the match to end at a word boundary. The profile-driven
5380
+ * counterpart of the per-language hardcoded compound lists (the hindi and
5381
+ * vietnamese keyword extractors). Returns a keyword token (with the normalized
5382
+ * form) or null. Case-sensitive against the stored native form, mirroring
5383
+ * `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
5384
+ * case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
5385
+ *
5386
+ * @param input - Input string
5387
+ * @param pos - Current position (must be a token-start boundary)
5388
+ * @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
5389
+ */
5390
+ tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
5391
+ if (this.multiWordKeywords.length === 0) return null;
5392
+ const rest = input.slice(pos);
5393
+ for (const entry of this.multiWordKeywords) {
5394
+ if (!rest.startsWith(entry.native)) continue;
5395
+ const after = input[pos + entry.native.length];
5396
+ if (after !== void 0 && isWordChar(after)) continue;
5397
+ return createToken(
5398
+ entry.native,
5399
+ "keyword",
5400
+ createPosition(pos, pos + entry.native.length),
5401
+ entry.normalized
5402
+ );
5403
+ }
5404
+ return null;
5405
+ }
5297
5406
  /**
5298
5407
  * Check if the remaining input starts with any known keyword.
5299
5408
  * Useful for non-space languages to detect word boundaries.
@@ -5306,6 +5415,32 @@ var _BaseTokenizer = class _BaseTokenizer {
5306
5415
  const remaining = input.slice(pos);
5307
5416
  return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
5308
5417
  }
5418
+ /**
5419
+ * Check if a known keyword starts at the given position AND ends at a word
5420
+ * boundary (end of input or a non-word character).
5421
+ *
5422
+ * Space-delimited languages must use this (not `isKeywordStart`) for
5423
+ * word-walk break checks: the keyword table includes English canonical
5424
+ * fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
5425
+ * word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
5426
+ * "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
5427
+ * keep using `isKeywordStart`.
5428
+ *
5429
+ * @param input - Input string
5430
+ * @param pos - Current position
5431
+ * @param isWordChar - Language-specific word-character predicate; pass the
5432
+ * tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
5433
+ * counts as part of a word. Defaults to Unicode letters/digits/underscore.
5434
+ * @returns true if a keyword starts here and is not followed by a word char
5435
+ */
5436
+ isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
5437
+ const remaining = input.slice(pos);
5438
+ return this.profileKeywords.some((entry) => {
5439
+ if (!remaining.startsWith(entry.native)) return false;
5440
+ const after = input[pos + entry.native.length];
5441
+ return after === void 0 || !isWordChar(after);
5442
+ });
5443
+ }
5309
5444
  /**
5310
5445
  * Look up a keyword by native word (case-insensitive).
5311
5446
  * O(1) lookup using the keyword map.
@@ -5638,6 +5773,119 @@ function createSimpleTokenizer(config) {
5638
5773
  return new SimpleTokenizer();
5639
5774
  }
5640
5775
 
5776
+ // src/multilingual/builders.ts
5777
+ function mergeRoleMarkers(slice, vocab) {
5778
+ const merged = {};
5779
+ const add = (role, marker) => {
5780
+ if (!marker?.primary) {
5781
+ delete merged[role];
5782
+ return;
5783
+ }
5784
+ merged[role] = {
5785
+ primary: marker.primary,
5786
+ ...marker.alternatives?.length && { alternatives: [...marker.alternatives] },
5787
+ ...marker.position && { position: marker.position }
5788
+ };
5789
+ };
5790
+ for (const [role, marker] of Object.entries(slice.roleMarkers ?? {})) add(role, marker);
5791
+ for (const [role, marker] of Object.entries(vocab.roleMarkerOverrides ?? {})) add(role, marker);
5792
+ return merged;
5793
+ }
5794
+ function buildPatternProfile(slice, vocab) {
5795
+ const keywords = {};
5796
+ for (const [action, translation] of Object.entries(vocab.keywords)) {
5797
+ keywords[action] = {
5798
+ primary: translation.primary,
5799
+ ...translation.alternatives?.length && { alternatives: [...translation.alternatives] }
5800
+ };
5801
+ }
5802
+ const roleMarkers = mergeRoleMarkers(slice, vocab);
5803
+ return {
5804
+ code: slice.code,
5805
+ wordOrder: slice.wordOrder,
5806
+ keywords,
5807
+ ...Object.keys(roleMarkers).length > 0 && { roleMarkers }
5808
+ };
5809
+ }
5810
+ function defaultCaseInsensitive(script) {
5811
+ return script === void 0 || script === "latin" || script === "cyrillic";
5812
+ }
5813
+ function buildDomainTokenizer(slice, vocab, options = {}) {
5814
+ const roleMarkers = mergeRoleMarkers(slice, vocab);
5815
+ const keywords = /* @__PURE__ */ new Set();
5816
+ for (const translation of Object.values(vocab.keywords)) {
5817
+ keywords.add(translation.primary);
5818
+ for (const alt of translation.alternatives ?? []) keywords.add(alt);
5819
+ }
5820
+ for (const marker of Object.values(roleMarkers)) {
5821
+ keywords.add(marker.primary);
5822
+ for (const alt of marker.alternatives ?? []) keywords.add(alt);
5823
+ }
5824
+ for (const particle of slice.tokenization?.particles ?? []) keywords.add(particle);
5825
+ for (const extra of vocab.tokenizerKeywords ?? []) keywords.add(extra);
5826
+ const profileKeywords = {};
5827
+ for (const [action, translation] of Object.entries(vocab.keywords)) {
5828
+ profileKeywords[action] = {
5829
+ primary: translation.primary,
5830
+ ...translation.alternatives?.length && { alternatives: [...translation.alternatives] },
5831
+ normalized: translation.normalized ?? action
5832
+ };
5833
+ }
5834
+ const keywordProfile = {
5835
+ keywords: profileKeywords,
5836
+ ...Object.keys(roleMarkers).length > 0 && { roleMarkers }
5837
+ };
5838
+ const customExtractors = [
5839
+ ...options.customExtractors ?? [],
5840
+ ...slice.script === "latin" ? [new LatinExtendedIdentifierExtractor()] : []
5841
+ ];
5842
+ return createSimpleTokenizer({
5843
+ language: slice.code,
5844
+ direction: slice.direction ?? "ltr",
5845
+ keywords: [...keywords],
5846
+ ...vocab.keywordExtras?.length && { keywordExtras: vocab.keywordExtras.map((e) => ({ ...e })) },
5847
+ keywordProfile,
5848
+ includeOperators: options.includeOperators ?? false,
5849
+ caseInsensitive: options.caseInsensitive ?? defaultCaseInsensitive(slice.script),
5850
+ ...customExtractors.length > 0 && { customExtractors }
5851
+ });
5852
+ }
5853
+ function buildLanguageConfig(slice, vocab, meta = {}) {
5854
+ const name = meta.name ?? slice.name ?? slice.code;
5855
+ return {
5856
+ code: slice.code,
5857
+ name,
5858
+ nativeName: meta.nativeName ?? slice.nativeName ?? name,
5859
+ tokenizer: meta.tokenizer ?? buildDomainTokenizer(slice, vocab, meta.tokenizerOptions),
5860
+ patternProfile: buildPatternProfile(slice, vocab),
5861
+ ...meta.grammarProfile && { grammarProfile: meta.grammarProfile }
5862
+ };
5863
+ }
5864
+ function deriveRoleMarkers(slice, roleMapping) {
5865
+ const derived = {};
5866
+ for (const [domainRole, semanticRole] of Object.entries(roleMapping)) {
5867
+ const marker = slice.roleMarkers?.[semanticRole];
5868
+ if (marker?.primary) derived[domainRole] = marker.primary;
5869
+ }
5870
+ return derived;
5871
+ }
5872
+
5873
+ // src/core/tokenization/char-classifiers.ts
5874
+ function createUnicodeRangeClassifier(ranges) {
5875
+ return (char) => {
5876
+ const code = char.charCodeAt(0);
5877
+ return ranges.some(([start, end]) => code >= start && code <= end);
5878
+ };
5879
+ }
5880
+ function combineClassifiers(...classifiers) {
5881
+ return (char) => classifiers.some((fn) => fn(char));
5882
+ }
5883
+ function createLatinCharClassifiers(letterPattern) {
5884
+ const isLetter = (char) => letterPattern.test(char);
5885
+ const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
5886
+ return { isLetter, isIdentifierChar };
5887
+ }
5888
+
5641
5889
  // src/core/tokenization/morphology/types.ts
5642
5890
  function noChange(word) {
5643
5891
  return { stem: word, confidence: 1 };
@@ -6204,7 +6452,10 @@ function accumulateIndented(statements, config) {
6204
6452
  WhitespaceExtractor,
6205
6453
  accumulateBlocks,
6206
6454
  buildDisambiguation,
6455
+ buildDomainTokenizer,
6207
6456
  buildFeedback,
6457
+ buildLanguageConfig,
6458
+ buildPatternProfile,
6208
6459
  buildPhrase,
6209
6460
  buildTablesFromProfiles,
6210
6461
  combineClassifiers,
@@ -6235,6 +6486,7 @@ function accumulateIndented(statements, config) {
6235
6486
  createUnicodeRangeClassifier,
6236
6487
  defineCommand,
6237
6488
  defineRole,
6489
+ deriveRoleMarkers,
6238
6490
  detectWordOrders,
6239
6491
  extractCssSelector,
6240
6492
  extractNumber,