@lokascript/framework 2.5.1 → 2.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/dist/core/index.js +148 -1
  2. package/dist/core/index.js.map +1 -1
  3. package/dist/core/tokenization/base-tokenizer.d.ts +52 -0
  4. package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
  5. package/dist/core/tokenization/index.js +148 -1
  6. package/dist/core/tokenization/index.js.map +1 -1
  7. package/dist/index.cjs +269 -17
  8. package/dist/index.cjs.map +1 -1
  9. package/dist/index.d.ts +2 -0
  10. package/dist/index.d.ts.map +1 -1
  11. package/dist/index.js +265 -17
  12. package/dist/index.js.map +1 -1
  13. package/dist/interfaces/value-extractor.d.ts +11 -0
  14. package/dist/interfaces/value-extractor.d.ts.map +1 -1
  15. package/dist/multilingual/builders.d.ts +88 -0
  16. package/dist/multilingual/builders.d.ts.map +1 -0
  17. package/dist/multilingual/index.d.ts +7 -3
  18. package/dist/multilingual/index.d.ts.map +1 -1
  19. package/dist/multilingual/index.js +1549 -0
  20. package/dist/multilingual/index.js.map +1 -1
  21. package/dist/multilingual/types.d.ts +117 -0
  22. package/dist/multilingual/types.d.ts.map +1 -0
  23. package/dist/testing/index.js +12 -4
  24. package/dist/testing/index.js.map +1 -1
  25. package/package.json +4 -4
  26. package/src/core/tokenization/base-tokenizer.ts +219 -0
  27. package/src/core/tokenization/keyword-boundary.test.ts +73 -0
  28. package/src/index.ts +20 -0
  29. package/src/interfaces/value-extractor.ts +23 -0
  30. package/src/multilingual/bridge.test.ts +441 -0
  31. package/src/multilingual/builders.ts +224 -0
  32. package/src/multilingual/index.ts +23 -4
  33. package/src/multilingual/types.ts +121 -0
  34. package/src/core/pattern-matching/pattern-matcher.ts.backup +0 -1267
package/dist/index.d.ts CHANGED
@@ -28,6 +28,8 @@ export { GrammarTransformer } from './grammar/transformer';
28
28
  export type { TransformerConfig } from './grammar/transformer';
29
29
  export { reorderRoles, insertMarkers, joinTokens } from './grammar/types';
30
30
  export type { LanguageProfile, PatternTransform, ParsedElement, WordOrder, GrammaticalMarker, AdpositionType, } from './grammar/types';
31
+ export { buildPatternProfile, buildDomainTokenizer, buildLanguageConfig, deriveRoleMarkers, } from './multilingual';
32
+ export type { GrammarProfileSlice, DomainVocabulary, DomainKeywordTranslation, DomainKeywordEntry, RoleMarkerSlice, TokenizationSlice, VerbSlice, DomainTokenizerOptions, LanguageConfigMeta, } from './multilingual';
31
33
  export * from './core';
32
34
  export type { ActionType, SemanticRole as SemanticRoleType, SemanticValue, SemanticNode, CommandSemanticNode, EventHandlerSemanticNode, ConditionalSemanticNode, CompoundSemanticNode, LoopSemanticNode, LiteralValue, SelectorValue, ReferenceValue, PropertyPathValue, ExpressionValue, SemanticMetadata, SourcePosition, LanguageToken, TokenStream, LanguageTokenizer, LanguagePattern, Annotation, ProtocolDiagnostic, AsyncVariant, MatchArm, LSEEnvelope, } from './core/types';
33
35
  export type { CommandSchema, RoleSpec } from './schema';
@@ -1 +1 @@
1
- {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,cAAc,OAAO,CAAC;AAGtB,cAAc,OAAO,CAAC;AAGtB,cAAc,cAAc,CAAC;AAG7B,cAAc,UAAU,CAAC;AACzB,cAAc,cAAc,CAAC;AAG7B,OAAO,EAAE,kBAAkB,EAAE,MAAM,uBAAuB,CAAC;AAC3D,YAAY,EAAE,iBAAiB,EAAE,MAAM,uBAAuB,CAAC;AAC/D,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAC1E,YAAY,EACV,eAAe,EACf,gBAAgB,EAChB,aAAa,EACb,SAAS,EACT,iBAAiB,EACjB,cAAc,GACf,MAAM,iBAAiB,CAAC;AAGzB,cAAc,QAAQ,CAAC;AAGvB,YAAY,EACV,UAAU,EACV,YAAY,IAAI,gBAAgB,EAChC,aAAa,EACb,YAAY,EACZ,mBAAmB,EACnB,wBAAwB,EACxB,uBAAuB,EACvB,oBAAoB,EACpB,gBAAgB,EAChB,YAAY,EACZ,aAAa,EACb,cAAc,EACd,iBAAiB,EACjB,eAAe,EACf,gBAAgB,EAChB,cAAc,EACd,aAAa,EACb,WAAW,EACX,iBAAiB,EACjB,eAAe,EAEf,UAAU,EACV,kBAAkB,EAClB,YAAY,EACZ,QAAQ,EACR,WAAW,GACZ,MAAM,cAAc,CAAC;AAEtB,YAAY,EAAE,aAAa,EAAE,QAAQ,EAAE,MAAM,UAAU,CAAC;AAGxD,YAAY,EACV,uBAAuB,EACvB,YAAY,EACZ,WAAW,EACX,cAAc,GACf,MAAM,uBAAuB,CAAC;AAC/B,OAAO,EACL,aAAa,EACb,YAAY,EACZ,WAAW,EACX,uBAAuB,EACvB,gBAAgB,EAChB,oBAAoB,GACrB,MAAM,uBAAuB,CAAC;AAG/B,YAAY,EACV,kBAAkB,EAClB,UAAU,EACV,gBAAgB,EAChB,iBAAiB,EACjB,mBAAmB,EACnB,iBAAiB,GAClB,MAAM,0BAA0B,CAAC;AAClC,OAAO,EAAE,yBAAyB,EAAE,SAAS,EAAE,gBAAgB,EAAE,MAAM,0BAA0B,CAAC;AAElG,YAAY,EACV,cAAc,EACd,SAAS,EACT,eAAe,EACf,aAAa,EACb,gBAAgB,EAChB,aAAa,EACb,gBAAgB,EAChB,iBAAiB,EACjB,eAAe,EACf,cAAc,EACd,oBAAoB,EACpB,kBAAkB,EAClB,cAAc,EACd,iBAAiB,GAClB,MAAM,OAAO,CAAC;AAEf,OAAO,EAAE,cAAc,EAAE,qBAAqB,EAAE,MAAM,OAAO,CAAC;AAG9D,OAAO,EACL,aAAa,EACb,cAAc,EACd,eAAe,EACf,kBAAkB,EAClB,gBAAgB,EAChB,iBAAiB,EACjB,sBAAsB,EACtB,qBAAqB,EACrB,kBAAkB,EAClB,cAAc,EACd,YAAY,EACZ,gBAAgB,EAEhB,aAAa,EACb,eAAe,EACf,eAAe,GAChB,MAAM,cAAc,CAAC;AAEtB,OAAO,EAAE,aAAa,EAAE,UAAU,EAAE,MAAM,UAAU,CAAC;AACrD,OAAO,EAAE,qBAAqB,EAAE,MAAM,oCAAoC,CAAC;AAC3E,YAAY,EAAE,qBAAqB,EAAE,MAAM,oCAAoC,CAAC;AAGhF,OAAO,EAAE,2BAA2B,EAAE,MAAM,gDAAgD,CAAC;AAC7F,YAAY,EACV,iBAAiB,EACjB,gBAAgB,GACjB,MAAM,gDAAgD,CAAC;AAGxD,cAAc,MAAM,CAAC;AAGrB,cAAc,WAAW,CAAC;AAG1B,cAAc,YAAY,CAAC;AAG3B,cAAc,YAAY,CAAC;AAG3B,OAAO,EAAE,0BAA0B,EAAE,gBAAgB,EAAE,MAAM,2BAA2B,CAAC;AACzF,YAAY,EACV,oBAAoB,EACpB,oBAAoB,EACpB,oBAAoB,EACpB,eAAe,EACf,cAAc,EACd,cAAc,EACd,WAAW,EACX,WAAW,EACX,UAAU,EACV,aAAa,GACd,MAAM,2BAA2B,CAAC"}
1
+ {"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,cAAc,OAAO,CAAC;AAGtB,cAAc,OAAO,CAAC;AAGtB,cAAc,cAAc,CAAC;AAG7B,cAAc,UAAU,CAAC;AACzB,cAAc,cAAc,CAAC;AAG7B,OAAO,EAAE,kBAAkB,EAAE,MAAM,uBAAuB,CAAC;AAC3D,YAAY,EAAE,iBAAiB,EAAE,MAAM,uBAAuB,CAAC;AAC/D,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAC1E,YAAY,EACV,eAAe,EACf,gBAAgB,EAChB,aAAa,EACb,SAAS,EACT,iBAAiB,EACjB,cAAc,GACf,MAAM,iBAAiB,CAAC;AAIzB,OAAO,EACL,mBAAmB,EACnB,oBAAoB,EACpB,mBAAmB,EACnB,iBAAiB,GAClB,MAAM,gBAAgB,CAAC;AACxB,YAAY,EACV,mBAAmB,EACnB,gBAAgB,EAChB,wBAAwB,EACxB,kBAAkB,EAClB,eAAe,EACf,iBAAiB,EACjB,SAAS,EACT,sBAAsB,EACtB,kBAAkB,GACnB,MAAM,gBAAgB,CAAC;AAGxB,cAAc,QAAQ,CAAC;AAGvB,YAAY,EACV,UAAU,EACV,YAAY,IAAI,gBAAgB,EAChC,aAAa,EACb,YAAY,EACZ,mBAAmB,EACnB,wBAAwB,EACxB,uBAAuB,EACvB,oBAAoB,EACpB,gBAAgB,EAChB,YAAY,EACZ,aAAa,EACb,cAAc,EACd,iBAAiB,EACjB,eAAe,EACf,gBAAgB,EAChB,cAAc,EACd,aAAa,EACb,WAAW,EACX,iBAAiB,EACjB,eAAe,EAEf,UAAU,EACV,kBAAkB,EAClB,YAAY,EACZ,QAAQ,EACR,WAAW,GACZ,MAAM,cAAc,CAAC;AAEtB,YAAY,EAAE,aAAa,EAAE,QAAQ,EAAE,MAAM,UAAU,CAAC;AAGxD,YAAY,EACV,uBAAuB,EACvB,YAAY,EACZ,WAAW,EACX,cAAc,GACf,MAAM,uBAAuB,CAAC;AAC/B,OAAO,EACL,aAAa,EACb,YAAY,EACZ,WAAW,EACX,uBAAuB,EACvB,gBAAgB,EAChB,oBAAoB,GACrB,MAAM,uBAAuB,CAAC;AAG/B,YAAY,EACV,kBAAkB,EAClB,UAAU,EACV,gBAAgB,EAChB,iBAAiB,EACjB,mBAAmB,EACnB,iBAAiB,GAClB,MAAM,0BAA0B,CAAC;AAClC,OAAO,EAAE,yBAAyB,EAAE,SAAS,EAAE,gBAAgB,EAAE,MAAM,0BAA0B,CAAC;AAElG,YAAY,EACV,cAAc,EACd,SAAS,EACT,eAAe,EACf,aAAa,EACb,gBAAgB,EAChB,aAAa,EACb,gBAAgB,EAChB,iBAAiB,EACjB,eAAe,EACf,cAAc,EACd,oBAAoB,EACpB,kBAAkB,EAClB,cAAc,EACd,iBAAiB,GAClB,MAAM,OAAO,CAAC;AAEf,OAAO,EAAE,cAAc,EAAE,qBAAqB,EAAE,MAAM,OAAO,CAAC;AAG9D,OAAO,EACL,aAAa,EACb,cAAc,EACd,eAAe,EACf,kBAAkB,EAClB,gBAAgB,EAChB,iBAAiB,EACjB,sBAAsB,EACtB,qBAAqB,EACrB,kBAAkB,EAClB,cAAc,EACd,YAAY,EACZ,gBAAgB,EAEhB,aAAa,EACb,eAAe,EACf,eAAe,GAChB,MAAM,cAAc,CAAC;AAEtB,OAAO,EAAE,aAAa,EAAE,UAAU,EAAE,MAAM,UAAU,CAAC;AACrD,OAAO,EAAE,qBAAqB,EAAE,MAAM,oCAAoC,CAAC;AAC3E,YAAY,EAAE,qBAAqB,EAAE,MAAM,oCAAoC,CAAC;AAGhF,OAAO,EAAE,2BAA2B,EAAE,MAAM,gDAAgD,CAAC;AAC7F,YAAY,EACV,iBAAiB,EACjB,gBAAgB,GACjB,MAAM,gDAAgD,CAAC;AAGxD,cAAc,MAAM,CAAC;AAGrB,cAAc,WAAW,CAAC;AAG1B,cAAc,YAAY,CAAC;AAG3B,cAAc,YAAY,CAAC;AAG3B,OAAO,EAAE,0BAA0B,EAAE,gBAAgB,EAAE,MAAM,2BAA2B,CAAC;AACzF,YAAY,EACV,oBAAoB,EACpB,oBAAoB,EACpB,oBAAoB,EACpB,eAAe,EACf,cAAc,EACd,cAAc,EACd,WAAW,EACX,WAAW,EACX,UAAU,EACV,aAAa,GACd,MAAM,2BAA2B,CAAC"}
package/dist/index.js CHANGED
@@ -1908,7 +1908,8 @@ function createTokenizerContext(tokenizer) {
1908
1908
  direction: tokenizer.direction,
1909
1909
  lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
1910
1910
  isKeyword: tokenizer.isKeyword.bind(tokenizer),
1911
- isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
1911
+ isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
1912
+ ...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
1912
1913
  };
1913
1914
  if (tokenizer.normalizer) {
1914
1915
  return { ...ctx, normalizer: tokenizer.normalizer };
@@ -4879,36 +4880,96 @@ function withDefaultExtractors(tokenizer) {
4879
4880
  return tokenizer;
4880
4881
  }
4881
4882
 
4882
- // src/core/tokenization/char-classifiers.ts
4883
- function createUnicodeRangeClassifier(ranges) {
4884
- return (char) => {
4885
- const code = char.charCodeAt(0);
4886
- return ranges.some(([start, end]) => code >= start && code <= end);
4887
- };
4888
- }
4889
- function combineClassifiers(...classifiers) {
4890
- return (char) => classifiers.some((fn) => fn(char));
4891
- }
4892
- function createLatinCharClassifiers(letterPattern) {
4893
- const isLetter = (char) => letterPattern.test(char);
4894
- const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
4895
- return { isLetter, isIdentifierChar };
4896
- }
4897
-
4898
4883
  // src/core/tokenization/base-tokenizer.ts
4899
4884
  var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
4885
+ var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
4886
+ // Role-marker role names (profile.roleMarkers normalizeds)
4887
+ "patient",
4888
+ "destination",
4889
+ "source",
4890
+ "style",
4891
+ "event",
4892
+ "eventMarker",
4893
+ "agent",
4894
+ "goal",
4895
+ "manner",
4896
+ // Prepositional / positional modifier concepts matched via the role mechanism
4897
+ // (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
4898
+ // NOT here — they are pattern literals (see the note above).
4899
+ "into",
4900
+ "from",
4901
+ "to",
4902
+ "with",
4903
+ "at",
4904
+ "of",
4905
+ "as",
4906
+ "by",
4907
+ "in",
4908
+ "on",
4909
+ "over",
4910
+ "under",
4911
+ "between",
4912
+ "through",
4913
+ "without"
4914
+ ]);
4915
+ var ENGLISH_DOM_EVENT_NAMES = [
4916
+ "click",
4917
+ "dblclick",
4918
+ "input",
4919
+ "change",
4920
+ "submit",
4921
+ "keydown",
4922
+ "keyup",
4923
+ "keypress",
4924
+ "mousedown",
4925
+ "mouseup",
4926
+ "mouseover",
4927
+ "mouseout",
4928
+ "mouseenter",
4929
+ "mouseleave",
4930
+ "mousemove",
4931
+ "pointerdown",
4932
+ "pointerup",
4933
+ "pointermove",
4934
+ "focus",
4935
+ "blur",
4936
+ "load",
4937
+ "resize",
4938
+ "scroll"
4939
+ ];
4900
4940
  var _BaseTokenizer = class _BaseTokenizer {
4901
4941
  constructor() {
4902
4942
  /** Keywords derived from profile, sorted longest-first for greedy matching */
4903
4943
  this.profileKeywords = [];
4944
+ /**
4945
+ * Space-containing profile keywords (multi-word phrases), longest-first.
4946
+ * Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
4947
+ * vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
4948
+ * profile-driven replacement for the per-language hardcoded compound lists.
4949
+ * Empty for no-space (CJK) languages, so they are unaffected.
4950
+ */
4951
+ this.multiWordKeywords = [];
4904
4952
  /** Map for O(1) keyword lookups by lowercase native word */
4905
4953
  this.profileKeywordMap = /* @__PURE__ */ new Map();
4954
+ /**
4955
+ * The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
4956
+ * The keyword map is keyed by native word with last-wins insertion, so a
4957
+ * duplicate native word inside the extras silently shadows the earlier entry
4958
+ * (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
4959
+ * positional expressions). Exposed so consistency tests can detect such
4960
+ * intra-extras collisions, which are invisible in the deduplicated map.
4961
+ */
4962
+ this.rawExtraEntries = [];
4906
4963
  /**
4907
4964
  * Pluggable value extractors for domain-specific syntax.
4908
4965
  * When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
4909
4966
  */
4910
4967
  this.extractors = [];
4911
4968
  }
4969
+ /** Raw extras as passed in, pre-dedup — for consistency tests. */
4970
+ getExtraKeywordEntries() {
4971
+ return this.rawExtraEntries;
4972
+ }
4912
4973
  /**
4913
4974
  * Tokenize input string to token stream.
4914
4975
  * Delegates to extractor-based tokenization if extractors are registered,
@@ -4977,6 +5038,12 @@ var _BaseTokenizer = class _BaseTokenizer {
4977
5038
  pos++;
4978
5039
  }
4979
5040
  if (pos >= input.length) break;
5041
+ const multiWord = this.tryMultiWordKeyword(input, pos);
5042
+ if (multiWord) {
5043
+ tokens.push(multiWord);
5044
+ pos = multiWord.position.end;
5045
+ continue;
5046
+ }
4980
5047
  let extracted = false;
4981
5048
  for (const extractor of this.extractors) {
4982
5049
  if (extractor.canExtract(input, pos)) {
@@ -5076,6 +5143,7 @@ var _BaseTokenizer = class _BaseTokenizer {
5076
5143
  */
5077
5144
  initializeKeywordsFromProfile(profile, extras = []) {
5078
5145
  const keywordMap = /* @__PURE__ */ new Map();
5146
+ this.rawExtraEntries = extras;
5079
5147
  if (profile.keywords) {
5080
5148
  for (const [normalized2, translation] of Object.entries(profile.keywords)) {
5081
5149
  keywordMap.set(translation.primary, {
@@ -5119,12 +5187,20 @@ var _BaseTokenizer = class _BaseTokenizer {
5119
5187
  keywordMap.set(native, { native, normalized: normalized2 });
5120
5188
  }
5121
5189
  }
5190
+ for (const evt of ENGLISH_DOM_EVENT_NAMES) {
5191
+ if (!keywordMap.has(evt)) {
5192
+ keywordMap.set(evt, { native: evt, normalized: evt });
5193
+ }
5194
+ }
5122
5195
  for (const extra of extras) {
5123
5196
  keywordMap.set(extra.native, extra);
5124
5197
  }
5125
5198
  this.profileKeywords = Array.from(keywordMap.values()).sort(
5126
5199
  (a, b) => b.native.length - a.native.length
5127
5200
  );
5201
+ this.multiWordKeywords = this.profileKeywords.filter(
5202
+ (k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
5203
+ );
5128
5204
  this.profileKeywordMap = /* @__PURE__ */ new Map();
5129
5205
  for (const keyword of this.profileKeywords) {
5130
5206
  this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
@@ -5166,6 +5242,35 @@ var _BaseTokenizer = class _BaseTokenizer {
5166
5242
  }
5167
5243
  return null;
5168
5244
  }
5245
+ /**
5246
+ * Match the longest multi-word (space-containing) profile keyword at `pos`,
5247
+ * requiring the match to end at a word boundary. The profile-driven
5248
+ * counterpart of the per-language hardcoded compound lists (the hindi and
5249
+ * vietnamese keyword extractors). Returns a keyword token (with the normalized
5250
+ * form) or null. Case-sensitive against the stored native form, mirroring
5251
+ * `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
5252
+ * case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
5253
+ *
5254
+ * @param input - Input string
5255
+ * @param pos - Current position (must be a token-start boundary)
5256
+ * @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
5257
+ */
5258
+ tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
5259
+ if (this.multiWordKeywords.length === 0) return null;
5260
+ const rest = input.slice(pos);
5261
+ for (const entry of this.multiWordKeywords) {
5262
+ if (!rest.startsWith(entry.native)) continue;
5263
+ const after = input[pos + entry.native.length];
5264
+ if (after !== void 0 && isWordChar(after)) continue;
5265
+ return createToken(
5266
+ entry.native,
5267
+ "keyword",
5268
+ createPosition(pos, pos + entry.native.length),
5269
+ entry.normalized
5270
+ );
5271
+ }
5272
+ return null;
5273
+ }
5169
5274
  /**
5170
5275
  * Check if the remaining input starts with any known keyword.
5171
5276
  * Useful for non-space languages to detect word boundaries.
@@ -5178,6 +5283,32 @@ var _BaseTokenizer = class _BaseTokenizer {
5178
5283
  const remaining = input.slice(pos);
5179
5284
  return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
5180
5285
  }
5286
+ /**
5287
+ * Check if a known keyword starts at the given position AND ends at a word
5288
+ * boundary (end of input or a non-word character).
5289
+ *
5290
+ * Space-delimited languages must use this (not `isKeywordStart`) for
5291
+ * word-walk break checks: the keyword table includes English canonical
5292
+ * fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
5293
+ * word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
5294
+ * "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
5295
+ * keep using `isKeywordStart`.
5296
+ *
5297
+ * @param input - Input string
5298
+ * @param pos - Current position
5299
+ * @param isWordChar - Language-specific word-character predicate; pass the
5300
+ * tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
5301
+ * counts as part of a word. Defaults to Unicode letters/digits/underscore.
5302
+ * @returns true if a keyword starts here and is not followed by a word char
5303
+ */
5304
+ isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
5305
+ const remaining = input.slice(pos);
5306
+ return this.profileKeywords.some((entry) => {
5307
+ if (!remaining.startsWith(entry.native)) return false;
5308
+ const after = input[pos + entry.native.length];
5309
+ return after === void 0 || !isWordChar(after);
5310
+ });
5311
+ }
5181
5312
  /**
5182
5313
  * Look up a keyword by native word (case-insensitive).
5183
5314
  * O(1) lookup using the keyword map.
@@ -5510,6 +5641,119 @@ function createSimpleTokenizer(config) {
5510
5641
  return new SimpleTokenizer();
5511
5642
  }
5512
5643
 
5644
+ // src/multilingual/builders.ts
5645
+ function mergeRoleMarkers(slice, vocab) {
5646
+ const merged = {};
5647
+ const add = (role, marker) => {
5648
+ if (!marker?.primary) {
5649
+ delete merged[role];
5650
+ return;
5651
+ }
5652
+ merged[role] = {
5653
+ primary: marker.primary,
5654
+ ...marker.alternatives?.length && { alternatives: [...marker.alternatives] },
5655
+ ...marker.position && { position: marker.position }
5656
+ };
5657
+ };
5658
+ for (const [role, marker] of Object.entries(slice.roleMarkers ?? {})) add(role, marker);
5659
+ for (const [role, marker] of Object.entries(vocab.roleMarkerOverrides ?? {})) add(role, marker);
5660
+ return merged;
5661
+ }
5662
+ function buildPatternProfile(slice, vocab) {
5663
+ const keywords = {};
5664
+ for (const [action, translation] of Object.entries(vocab.keywords)) {
5665
+ keywords[action] = {
5666
+ primary: translation.primary,
5667
+ ...translation.alternatives?.length && { alternatives: [...translation.alternatives] }
5668
+ };
5669
+ }
5670
+ const roleMarkers = mergeRoleMarkers(slice, vocab);
5671
+ return {
5672
+ code: slice.code,
5673
+ wordOrder: slice.wordOrder,
5674
+ keywords,
5675
+ ...Object.keys(roleMarkers).length > 0 && { roleMarkers }
5676
+ };
5677
+ }
5678
+ function defaultCaseInsensitive(script) {
5679
+ return script === void 0 || script === "latin" || script === "cyrillic";
5680
+ }
5681
+ function buildDomainTokenizer(slice, vocab, options = {}) {
5682
+ const roleMarkers = mergeRoleMarkers(slice, vocab);
5683
+ const keywords = /* @__PURE__ */ new Set();
5684
+ for (const translation of Object.values(vocab.keywords)) {
5685
+ keywords.add(translation.primary);
5686
+ for (const alt of translation.alternatives ?? []) keywords.add(alt);
5687
+ }
5688
+ for (const marker of Object.values(roleMarkers)) {
5689
+ keywords.add(marker.primary);
5690
+ for (const alt of marker.alternatives ?? []) keywords.add(alt);
5691
+ }
5692
+ for (const particle of slice.tokenization?.particles ?? []) keywords.add(particle);
5693
+ for (const extra of vocab.tokenizerKeywords ?? []) keywords.add(extra);
5694
+ const profileKeywords = {};
5695
+ for (const [action, translation] of Object.entries(vocab.keywords)) {
5696
+ profileKeywords[action] = {
5697
+ primary: translation.primary,
5698
+ ...translation.alternatives?.length && { alternatives: [...translation.alternatives] },
5699
+ normalized: translation.normalized ?? action
5700
+ };
5701
+ }
5702
+ const keywordProfile = {
5703
+ keywords: profileKeywords,
5704
+ ...Object.keys(roleMarkers).length > 0 && { roleMarkers }
5705
+ };
5706
+ const customExtractors = [
5707
+ ...options.customExtractors ?? [],
5708
+ ...slice.script === "latin" ? [new LatinExtendedIdentifierExtractor()] : []
5709
+ ];
5710
+ return createSimpleTokenizer({
5711
+ language: slice.code,
5712
+ direction: slice.direction ?? "ltr",
5713
+ keywords: [...keywords],
5714
+ ...vocab.keywordExtras?.length && { keywordExtras: vocab.keywordExtras.map((e) => ({ ...e })) },
5715
+ keywordProfile,
5716
+ includeOperators: options.includeOperators ?? false,
5717
+ caseInsensitive: options.caseInsensitive ?? defaultCaseInsensitive(slice.script),
5718
+ ...customExtractors.length > 0 && { customExtractors }
5719
+ });
5720
+ }
5721
+ function buildLanguageConfig(slice, vocab, meta = {}) {
5722
+ const name = meta.name ?? slice.name ?? slice.code;
5723
+ return {
5724
+ code: slice.code,
5725
+ name,
5726
+ nativeName: meta.nativeName ?? slice.nativeName ?? name,
5727
+ tokenizer: meta.tokenizer ?? buildDomainTokenizer(slice, vocab, meta.tokenizerOptions),
5728
+ patternProfile: buildPatternProfile(slice, vocab),
5729
+ ...meta.grammarProfile && { grammarProfile: meta.grammarProfile }
5730
+ };
5731
+ }
5732
+ function deriveRoleMarkers(slice, roleMapping) {
5733
+ const derived = {};
5734
+ for (const [domainRole, semanticRole] of Object.entries(roleMapping)) {
5735
+ const marker = slice.roleMarkers?.[semanticRole];
5736
+ if (marker?.primary) derived[domainRole] = marker.primary;
5737
+ }
5738
+ return derived;
5739
+ }
5740
+
5741
+ // src/core/tokenization/char-classifiers.ts
5742
+ function createUnicodeRangeClassifier(ranges) {
5743
+ return (char) => {
5744
+ const code = char.charCodeAt(0);
5745
+ return ranges.some(([start, end]) => code >= start && code <= end);
5746
+ };
5747
+ }
5748
+ function combineClassifiers(...classifiers) {
5749
+ return (char) => classifiers.some((fn) => fn(char));
5750
+ }
5751
+ function createLatinCharClassifiers(letterPattern) {
5752
+ const isLetter = (char) => letterPattern.test(char);
5753
+ const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
5754
+ return { isLetter, isIdentifierChar };
5755
+ }
5756
+
5513
5757
  // src/core/tokenization/morphology/types.ts
5514
5758
  function noChange(word) {
5515
5759
  return { stem: word, confidence: 1 };
@@ -6075,7 +6319,10 @@ export {
6075
6319
  WhitespaceExtractor,
6076
6320
  accumulateBlocks,
6077
6321
  buildDisambiguation,
6322
+ buildDomainTokenizer,
6078
6323
  buildFeedback,
6324
+ buildLanguageConfig,
6325
+ buildPatternProfile,
6079
6326
  buildPhrase,
6080
6327
  buildTablesFromProfiles,
6081
6328
  combineClassifiers,
@@ -6106,6 +6353,7 @@ export {
6106
6353
  createUnicodeRangeClassifier,
6107
6354
  defineCommand,
6108
6355
  defineRole,
6356
+ deriveRoleMarkers,
6109
6357
  detectWordOrders,
6110
6358
  extractCssSelector,
6111
6359
  extractNumber,