@lokascript/framework 2.6.0 → 2.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -47,7 +47,10 @@ __export(index_exports, {
47
47
  WhitespaceExtractor: () => WhitespaceExtractor,
48
48
  accumulateBlocks: () => accumulateBlocks,
49
49
  buildDisambiguation: () => buildDisambiguation,
50
+ buildDomainTokenizer: () => buildDomainTokenizer,
50
51
  buildFeedback: () => buildFeedback,
52
+ buildLanguageConfig: () => buildLanguageConfig,
53
+ buildPatternProfile: () => buildPatternProfile,
51
54
  buildPhrase: () => buildPhrase,
52
55
  buildTablesFromProfiles: () => buildTablesFromProfiles,
53
56
  combineClassifiers: () => combineClassifiers,
@@ -78,6 +81,7 @@ __export(index_exports, {
78
81
  createUnicodeRangeClassifier: () => createUnicodeRangeClassifier,
79
82
  defineCommand: () => import_intent6.defineCommand,
80
83
  defineRole: () => import_intent6.defineRole,
84
+ deriveRoleMarkers: () => deriveRoleMarkers,
81
85
  detectWordOrders: () => detectWordOrders,
82
86
  extractCssSelector: () => extractCssSelector,
83
87
  extractNumber: () => extractNumber,
@@ -5008,22 +5012,6 @@ function withDefaultExtractors(tokenizer) {
5008
5012
  return tokenizer;
5009
5013
  }
5010
5014
 
5011
- // src/core/tokenization/char-classifiers.ts
5012
- function createUnicodeRangeClassifier(ranges) {
5013
- return (char) => {
5014
- const code = char.charCodeAt(0);
5015
- return ranges.some(([start, end]) => code >= start && code <= end);
5016
- };
5017
- }
5018
- function combineClassifiers(...classifiers) {
5019
- return (char) => classifiers.some((fn) => fn(char));
5020
- }
5021
- function createLatinCharClassifiers(letterPattern) {
5022
- const isLetter = (char) => letterPattern.test(char);
5023
- const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
5024
- return { isLetter, isIdentifierChar };
5025
- }
5026
-
5027
5015
  // src/core/tokenization/base-tokenizer.ts
5028
5016
  var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
5029
5017
  var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
@@ -5785,6 +5773,119 @@ function createSimpleTokenizer(config) {
5785
5773
  return new SimpleTokenizer();
5786
5774
  }
5787
5775
 
5776
+ // src/multilingual/builders.ts
5777
+ function mergeRoleMarkers(slice, vocab) {
5778
+ const merged = {};
5779
+ const add = (role, marker) => {
5780
+ if (!marker?.primary) {
5781
+ delete merged[role];
5782
+ return;
5783
+ }
5784
+ merged[role] = {
5785
+ primary: marker.primary,
5786
+ ...marker.alternatives?.length && { alternatives: [...marker.alternatives] },
5787
+ ...marker.position && { position: marker.position }
5788
+ };
5789
+ };
5790
+ for (const [role, marker] of Object.entries(slice.roleMarkers ?? {})) add(role, marker);
5791
+ for (const [role, marker] of Object.entries(vocab.roleMarkerOverrides ?? {})) add(role, marker);
5792
+ return merged;
5793
+ }
5794
+ function buildPatternProfile(slice, vocab) {
5795
+ const keywords = {};
5796
+ for (const [action, translation] of Object.entries(vocab.keywords)) {
5797
+ keywords[action] = {
5798
+ primary: translation.primary,
5799
+ ...translation.alternatives?.length && { alternatives: [...translation.alternatives] }
5800
+ };
5801
+ }
5802
+ const roleMarkers = mergeRoleMarkers(slice, vocab);
5803
+ return {
5804
+ code: slice.code,
5805
+ wordOrder: slice.wordOrder,
5806
+ keywords,
5807
+ ...Object.keys(roleMarkers).length > 0 && { roleMarkers }
5808
+ };
5809
+ }
5810
+ function defaultCaseInsensitive(script) {
5811
+ return script === void 0 || script === "latin" || script === "cyrillic";
5812
+ }
5813
+ function buildDomainTokenizer(slice, vocab, options = {}) {
5814
+ const roleMarkers = mergeRoleMarkers(slice, vocab);
5815
+ const keywords = /* @__PURE__ */ new Set();
5816
+ for (const translation of Object.values(vocab.keywords)) {
5817
+ keywords.add(translation.primary);
5818
+ for (const alt of translation.alternatives ?? []) keywords.add(alt);
5819
+ }
5820
+ for (const marker of Object.values(roleMarkers)) {
5821
+ keywords.add(marker.primary);
5822
+ for (const alt of marker.alternatives ?? []) keywords.add(alt);
5823
+ }
5824
+ for (const particle of slice.tokenization?.particles ?? []) keywords.add(particle);
5825
+ for (const extra of vocab.tokenizerKeywords ?? []) keywords.add(extra);
5826
+ const profileKeywords = {};
5827
+ for (const [action, translation] of Object.entries(vocab.keywords)) {
5828
+ profileKeywords[action] = {
5829
+ primary: translation.primary,
5830
+ ...translation.alternatives?.length && { alternatives: [...translation.alternatives] },
5831
+ normalized: translation.normalized ?? action
5832
+ };
5833
+ }
5834
+ const keywordProfile = {
5835
+ keywords: profileKeywords,
5836
+ ...Object.keys(roleMarkers).length > 0 && { roleMarkers }
5837
+ };
5838
+ const customExtractors = [
5839
+ ...options.customExtractors ?? [],
5840
+ ...slice.script === "latin" ? [new LatinExtendedIdentifierExtractor()] : []
5841
+ ];
5842
+ return createSimpleTokenizer({
5843
+ language: slice.code,
5844
+ direction: slice.direction ?? "ltr",
5845
+ keywords: [...keywords],
5846
+ ...vocab.keywordExtras?.length && { keywordExtras: vocab.keywordExtras.map((e) => ({ ...e })) },
5847
+ keywordProfile,
5848
+ includeOperators: options.includeOperators ?? false,
5849
+ caseInsensitive: options.caseInsensitive ?? defaultCaseInsensitive(slice.script),
5850
+ ...customExtractors.length > 0 && { customExtractors }
5851
+ });
5852
+ }
5853
+ function buildLanguageConfig(slice, vocab, meta = {}) {
5854
+ const name = meta.name ?? slice.name ?? slice.code;
5855
+ return {
5856
+ code: slice.code,
5857
+ name,
5858
+ nativeName: meta.nativeName ?? slice.nativeName ?? name,
5859
+ tokenizer: meta.tokenizer ?? buildDomainTokenizer(slice, vocab, meta.tokenizerOptions),
5860
+ patternProfile: buildPatternProfile(slice, vocab),
5861
+ ...meta.grammarProfile && { grammarProfile: meta.grammarProfile }
5862
+ };
5863
+ }
5864
+ function deriveRoleMarkers(slice, roleMapping) {
5865
+ const derived = {};
5866
+ for (const [domainRole, semanticRole] of Object.entries(roleMapping)) {
5867
+ const marker = slice.roleMarkers?.[semanticRole];
5868
+ if (marker?.primary) derived[domainRole] = marker.primary;
5869
+ }
5870
+ return derived;
5871
+ }
5872
+
5873
+ // src/core/tokenization/char-classifiers.ts
5874
+ function createUnicodeRangeClassifier(ranges) {
5875
+ return (char) => {
5876
+ const code = char.charCodeAt(0);
5877
+ return ranges.some(([start, end]) => code >= start && code <= end);
5878
+ };
5879
+ }
5880
+ function combineClassifiers(...classifiers) {
5881
+ return (char) => classifiers.some((fn) => fn(char));
5882
+ }
5883
+ function createLatinCharClassifiers(letterPattern) {
5884
+ const isLetter = (char) => letterPattern.test(char);
5885
+ const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
5886
+ return { isLetter, isIdentifierChar };
5887
+ }
5888
+
5788
5889
  // src/core/tokenization/morphology/types.ts
5789
5890
  function noChange(word) {
5790
5891
  return { stem: word, confidence: 1 };
@@ -6351,7 +6452,10 @@ function accumulateIndented(statements, config) {
6351
6452
  WhitespaceExtractor,
6352
6453
  accumulateBlocks,
6353
6454
  buildDisambiguation,
6455
+ buildDomainTokenizer,
6354
6456
  buildFeedback,
6457
+ buildLanguageConfig,
6458
+ buildPatternProfile,
6355
6459
  buildPhrase,
6356
6460
  buildTablesFromProfiles,
6357
6461
  combineClassifiers,
@@ -6382,6 +6486,7 @@ function accumulateIndented(statements, config) {
6382
6486
  createUnicodeRangeClassifier,
6383
6487
  defineCommand,
6384
6488
  defineRole,
6489
+ deriveRoleMarkers,
6385
6490
  detectWordOrders,
6386
6491
  extractCssSelector,
6387
6492
  extractNumber,