@lokascript/framework 2.6.0 → 2.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.cjs +121 -16
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +117 -16
- package/dist/index.js.map +1 -1
- package/dist/multilingual/builders.d.ts +88 -0
- package/dist/multilingual/builders.d.ts.map +1 -0
- package/dist/multilingual/index.d.ts +7 -3
- package/dist/multilingual/index.d.ts.map +1 -1
- package/dist/multilingual/index.js +1549 -0
- package/dist/multilingual/index.js.map +1 -1
- package/dist/multilingual/types.d.ts +117 -0
- package/dist/multilingual/types.d.ts.map +1 -0
- package/package.json +3 -3
- package/src/index.ts +20 -0
- package/src/multilingual/bridge.test.ts +441 -0
- package/src/multilingual/builders.ts +224 -0
- package/src/multilingual/index.ts +23 -4
- package/src/multilingual/types.ts +121 -0
package/dist/index.d.ts
CHANGED
|
@@ -28,6 +28,8 @@ export { GrammarTransformer } from './grammar/transformer';
|
|
|
28
28
|
export type { TransformerConfig } from './grammar/transformer';
|
|
29
29
|
export { reorderRoles, insertMarkers, joinTokens } from './grammar/types';
|
|
30
30
|
export type { LanguageProfile, PatternTransform, ParsedElement, WordOrder, GrammaticalMarker, AdpositionType, } from './grammar/types';
|
|
31
|
+
export { buildPatternProfile, buildDomainTokenizer, buildLanguageConfig, deriveRoleMarkers, } from './multilingual';
|
|
32
|
+
export type { GrammarProfileSlice, DomainVocabulary, DomainKeywordTranslation, DomainKeywordEntry, RoleMarkerSlice, TokenizationSlice, VerbSlice, DomainTokenizerOptions, LanguageConfigMeta, } from './multilingual';
|
|
31
33
|
export * from './core';
|
|
32
34
|
export type { ActionType, SemanticRole as SemanticRoleType, SemanticValue, SemanticNode, CommandSemanticNode, EventHandlerSemanticNode, ConditionalSemanticNode, CompoundSemanticNode, LoopSemanticNode, LiteralValue, SelectorValue, ReferenceValue, PropertyPathValue, ExpressionValue, SemanticMetadata, SourcePosition, LanguageToken, TokenStream, LanguageTokenizer, LanguagePattern, Annotation, ProtocolDiagnostic, AsyncVariant, MatchArm, LSEEnvelope, } from './core/types';
|
|
33
35
|
export type { CommandSchema, RoleSpec } from './schema';
|
package/dist/index.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,cAAc,OAAO,CAAC;AAGtB,cAAc,OAAO,CAAC;AAGtB,cAAc,cAAc,CAAC;AAG7B,cAAc,UAAU,CAAC;AACzB,cAAc,cAAc,CAAC;AAG7B,OAAO,EAAE,kBAAkB,EAAE,MAAM,uBAAuB,CAAC;AAC3D,YAAY,EAAE,iBAAiB,EAAE,MAAM,uBAAuB,CAAC;AAC/D,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAC1E,YAAY,EACV,eAAe,EACf,gBAAgB,EAChB,aAAa,EACb,SAAS,EACT,iBAAiB,EACjB,cAAc,GACf,MAAM,iBAAiB,CAAC;
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,cAAc,OAAO,CAAC;AAGtB,cAAc,OAAO,CAAC;AAGtB,cAAc,cAAc,CAAC;AAG7B,cAAc,UAAU,CAAC;AACzB,cAAc,cAAc,CAAC;AAG7B,OAAO,EAAE,kBAAkB,EAAE,MAAM,uBAAuB,CAAC;AAC3D,YAAY,EAAE,iBAAiB,EAAE,MAAM,uBAAuB,CAAC;AAC/D,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAC1E,YAAY,EACV,eAAe,EACf,gBAAgB,EAChB,aAAa,EACb,SAAS,EACT,iBAAiB,EACjB,cAAc,GACf,MAAM,iBAAiB,CAAC;AAIzB,OAAO,EACL,mBAAmB,EACnB,oBAAoB,EACpB,mBAAmB,EACnB,iBAAiB,GAClB,MAAM,gBAAgB,CAAC;AACxB,YAAY,EACV,mBAAmB,EACnB,gBAAgB,EAChB,wBAAwB,EACxB,kBAAkB,EAClB,eAAe,EACf,iBAAiB,EACjB,SAAS,EACT,sBAAsB,EACtB,kBAAkB,GACnB,MAAM,gBAAgB,CAAC;AAGxB,cAAc,QAAQ,CAAC;AAGvB,YAAY,EACV,UAAU,EACV,YAAY,IAAI,gBAAgB,EAChC,aAAa,EACb,YAAY,EACZ,mBAAmB,EACnB,wBAAwB,EACxB,uBAAuB,EACvB,oBAAoB,EACpB,gBAAgB,EAChB,YAAY,EACZ,aAAa,EACb,cAAc,EACd,iBAAiB,EACjB,eAAe,EACf,gBAAgB,EAChB,cAAc,EACd,aAAa,EACb,WAAW,EACX,iBAAiB,EACjB,eAAe,EAEf,UAAU,EACV,kBAAkB,EAClB,YAAY,EACZ,QAAQ,EACR,WAAW,GACZ,MAAM,cAAc,CAAC;AAEtB,YAAY,EAAE,aAAa,EAAE,QAAQ,EAAE,MAAM,UAAU,CAAC;AAGxD,YAAY,EACV,uBAAuB,EACvB,YAAY,EACZ,WAAW,EACX,cAAc,GACf,MAAM,uBAAuB,CAAC;AAC/B,OAAO,EACL,aAAa,EACb,YAAY,EACZ,WAAW,EACX,uBAAuB,EACvB,gBAAgB,EAChB,oBAAoB,GACrB,MAAM,uBAAuB,CAAC;AAG/B,YAAY,EACV,kBAAkB,EAClB,UAAU,EACV,gBAAgB,EAChB,iBAAiB,EACjB,mBAAmB,EACnB,iBAAiB,GAClB,MAAM,0BAA0B,CAAC;AAClC,OAAO,EAAE,yBAAyB,EAAE,SAAS,EAAE,gBAAgB,EAAE,MAAM,0BAA0B,CAAC;AAElG,YAAY,EACV,cAAc,EACd,SAAS,EACT,eAAe,EACf,aAAa,EACb,gBAAgB,EAChB,aAAa,EACb,gBAAgB,EAChB,iBAAiB,EACjB,eAAe,EACf,cAAc,EACd,oBAAoB,EACpB,kBAAkB,EAClB,cAAc,EACd,iBAAiB,GAClB,MAAM,OAAO,CAAC;AAEf,OAAO,EAAE,cAAc,EAAE,qBAAqB,EAAE,MAAM,OAAO,CAAC;AAG9D,OAAO,EACL,aAAa,EACb,cAAc,EACd,eAAe,EACf,kBAAkB,EAClB,gBAAgB,EAChB,iBAAiB,EACjB,sBAAsB,EACtB,qBAAqB,EACrB,kBAAkB,EAClB,cAAc,EACd,YAAY,EACZ,gBAAgB,EAEhB,aAAa,EACb,eAAe,EACf,eAAe,GAChB,MAAM,cAAc,CAAC;AAEtB,OAAO,EAAE,aAAa,EAAE,UAAU,EAAE,MAAM,UAAU,CAAC;AACrD,OAAO,EAAE,qBAAqB,EAAE,MAAM,oCAAoC,CAAC;AAC3E,YAAY,EAAE,qBAAqB,EAAE,MAAM,oCAAoC,CAAC;AAGhF,OAAO,EAAE,2BAA2B,EAAE,MAAM,gDAAgD,CAAC;AAC7F,YAAY,EACV,iBAAiB,EACjB,gBAAgB,GACjB,MAAM,gDAAgD,CAAC;AAGxD,cAAc,MAAM,CAAC;AAGrB,cAAc,WAAW,CAAC;AAG1B,cAAc,YAAY,CAAC;AAG3B,cAAc,YAAY,CAAC;AAG3B,OAAO,EAAE,0BAA0B,EAAE,gBAAgB,EAAE,MAAM,2BAA2B,CAAC;AACzF,YAAY,EACV,oBAAoB,EACpB,oBAAoB,EACpB,oBAAoB,EACpB,eAAe,EACf,cAAc,EACd,cAAc,EACd,WAAW,EACX,WAAW,EACX,UAAU,EACV,aAAa,GACd,MAAM,2BAA2B,CAAC"}
|
package/dist/index.js
CHANGED
|
@@ -4880,22 +4880,6 @@ function withDefaultExtractors(tokenizer) {
|
|
|
4880
4880
|
return tokenizer;
|
|
4881
4881
|
}
|
|
4882
4882
|
|
|
4883
|
-
// src/core/tokenization/char-classifiers.ts
|
|
4884
|
-
function createUnicodeRangeClassifier(ranges) {
|
|
4885
|
-
return (char) => {
|
|
4886
|
-
const code = char.charCodeAt(0);
|
|
4887
|
-
return ranges.some(([start, end]) => code >= start && code <= end);
|
|
4888
|
-
};
|
|
4889
|
-
}
|
|
4890
|
-
function combineClassifiers(...classifiers) {
|
|
4891
|
-
return (char) => classifiers.some((fn) => fn(char));
|
|
4892
|
-
}
|
|
4893
|
-
function createLatinCharClassifiers(letterPattern) {
|
|
4894
|
-
const isLetter = (char) => letterPattern.test(char);
|
|
4895
|
-
const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
|
|
4896
|
-
return { isLetter, isIdentifierChar };
|
|
4897
|
-
}
|
|
4898
|
-
|
|
4899
4883
|
// src/core/tokenization/base-tokenizer.ts
|
|
4900
4884
|
var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
|
|
4901
4885
|
var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
|
|
@@ -5657,6 +5641,119 @@ function createSimpleTokenizer(config) {
|
|
|
5657
5641
|
return new SimpleTokenizer();
|
|
5658
5642
|
}
|
|
5659
5643
|
|
|
5644
|
+
// src/multilingual/builders.ts
|
|
5645
|
+
function mergeRoleMarkers(slice, vocab) {
|
|
5646
|
+
const merged = {};
|
|
5647
|
+
const add = (role, marker) => {
|
|
5648
|
+
if (!marker?.primary) {
|
|
5649
|
+
delete merged[role];
|
|
5650
|
+
return;
|
|
5651
|
+
}
|
|
5652
|
+
merged[role] = {
|
|
5653
|
+
primary: marker.primary,
|
|
5654
|
+
...marker.alternatives?.length && { alternatives: [...marker.alternatives] },
|
|
5655
|
+
...marker.position && { position: marker.position }
|
|
5656
|
+
};
|
|
5657
|
+
};
|
|
5658
|
+
for (const [role, marker] of Object.entries(slice.roleMarkers ?? {})) add(role, marker);
|
|
5659
|
+
for (const [role, marker] of Object.entries(vocab.roleMarkerOverrides ?? {})) add(role, marker);
|
|
5660
|
+
return merged;
|
|
5661
|
+
}
|
|
5662
|
+
function buildPatternProfile(slice, vocab) {
|
|
5663
|
+
const keywords = {};
|
|
5664
|
+
for (const [action, translation] of Object.entries(vocab.keywords)) {
|
|
5665
|
+
keywords[action] = {
|
|
5666
|
+
primary: translation.primary,
|
|
5667
|
+
...translation.alternatives?.length && { alternatives: [...translation.alternatives] }
|
|
5668
|
+
};
|
|
5669
|
+
}
|
|
5670
|
+
const roleMarkers = mergeRoleMarkers(slice, vocab);
|
|
5671
|
+
return {
|
|
5672
|
+
code: slice.code,
|
|
5673
|
+
wordOrder: slice.wordOrder,
|
|
5674
|
+
keywords,
|
|
5675
|
+
...Object.keys(roleMarkers).length > 0 && { roleMarkers }
|
|
5676
|
+
};
|
|
5677
|
+
}
|
|
5678
|
+
function defaultCaseInsensitive(script) {
|
|
5679
|
+
return script === void 0 || script === "latin" || script === "cyrillic";
|
|
5680
|
+
}
|
|
5681
|
+
function buildDomainTokenizer(slice, vocab, options = {}) {
|
|
5682
|
+
const roleMarkers = mergeRoleMarkers(slice, vocab);
|
|
5683
|
+
const keywords = /* @__PURE__ */ new Set();
|
|
5684
|
+
for (const translation of Object.values(vocab.keywords)) {
|
|
5685
|
+
keywords.add(translation.primary);
|
|
5686
|
+
for (const alt of translation.alternatives ?? []) keywords.add(alt);
|
|
5687
|
+
}
|
|
5688
|
+
for (const marker of Object.values(roleMarkers)) {
|
|
5689
|
+
keywords.add(marker.primary);
|
|
5690
|
+
for (const alt of marker.alternatives ?? []) keywords.add(alt);
|
|
5691
|
+
}
|
|
5692
|
+
for (const particle of slice.tokenization?.particles ?? []) keywords.add(particle);
|
|
5693
|
+
for (const extra of vocab.tokenizerKeywords ?? []) keywords.add(extra);
|
|
5694
|
+
const profileKeywords = {};
|
|
5695
|
+
for (const [action, translation] of Object.entries(vocab.keywords)) {
|
|
5696
|
+
profileKeywords[action] = {
|
|
5697
|
+
primary: translation.primary,
|
|
5698
|
+
...translation.alternatives?.length && { alternatives: [...translation.alternatives] },
|
|
5699
|
+
normalized: translation.normalized ?? action
|
|
5700
|
+
};
|
|
5701
|
+
}
|
|
5702
|
+
const keywordProfile = {
|
|
5703
|
+
keywords: profileKeywords,
|
|
5704
|
+
...Object.keys(roleMarkers).length > 0 && { roleMarkers }
|
|
5705
|
+
};
|
|
5706
|
+
const customExtractors = [
|
|
5707
|
+
...options.customExtractors ?? [],
|
|
5708
|
+
...slice.script === "latin" ? [new LatinExtendedIdentifierExtractor()] : []
|
|
5709
|
+
];
|
|
5710
|
+
return createSimpleTokenizer({
|
|
5711
|
+
language: slice.code,
|
|
5712
|
+
direction: slice.direction ?? "ltr",
|
|
5713
|
+
keywords: [...keywords],
|
|
5714
|
+
...vocab.keywordExtras?.length && { keywordExtras: vocab.keywordExtras.map((e) => ({ ...e })) },
|
|
5715
|
+
keywordProfile,
|
|
5716
|
+
includeOperators: options.includeOperators ?? false,
|
|
5717
|
+
caseInsensitive: options.caseInsensitive ?? defaultCaseInsensitive(slice.script),
|
|
5718
|
+
...customExtractors.length > 0 && { customExtractors }
|
|
5719
|
+
});
|
|
5720
|
+
}
|
|
5721
|
+
function buildLanguageConfig(slice, vocab, meta = {}) {
|
|
5722
|
+
const name = meta.name ?? slice.name ?? slice.code;
|
|
5723
|
+
return {
|
|
5724
|
+
code: slice.code,
|
|
5725
|
+
name,
|
|
5726
|
+
nativeName: meta.nativeName ?? slice.nativeName ?? name,
|
|
5727
|
+
tokenizer: meta.tokenizer ?? buildDomainTokenizer(slice, vocab, meta.tokenizerOptions),
|
|
5728
|
+
patternProfile: buildPatternProfile(slice, vocab),
|
|
5729
|
+
...meta.grammarProfile && { grammarProfile: meta.grammarProfile }
|
|
5730
|
+
};
|
|
5731
|
+
}
|
|
5732
|
+
function deriveRoleMarkers(slice, roleMapping) {
|
|
5733
|
+
const derived = {};
|
|
5734
|
+
for (const [domainRole, semanticRole] of Object.entries(roleMapping)) {
|
|
5735
|
+
const marker = slice.roleMarkers?.[semanticRole];
|
|
5736
|
+
if (marker?.primary) derived[domainRole] = marker.primary;
|
|
5737
|
+
}
|
|
5738
|
+
return derived;
|
|
5739
|
+
}
|
|
5740
|
+
|
|
5741
|
+
// src/core/tokenization/char-classifiers.ts
|
|
5742
|
+
function createUnicodeRangeClassifier(ranges) {
|
|
5743
|
+
return (char) => {
|
|
5744
|
+
const code = char.charCodeAt(0);
|
|
5745
|
+
return ranges.some(([start, end]) => code >= start && code <= end);
|
|
5746
|
+
};
|
|
5747
|
+
}
|
|
5748
|
+
function combineClassifiers(...classifiers) {
|
|
5749
|
+
return (char) => classifiers.some((fn) => fn(char));
|
|
5750
|
+
}
|
|
5751
|
+
function createLatinCharClassifiers(letterPattern) {
|
|
5752
|
+
const isLetter = (char) => letterPattern.test(char);
|
|
5753
|
+
const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
|
|
5754
|
+
return { isLetter, isIdentifierChar };
|
|
5755
|
+
}
|
|
5756
|
+
|
|
5660
5757
|
// src/core/tokenization/morphology/types.ts
|
|
5661
5758
|
function noChange(word) {
|
|
5662
5759
|
return { stem: word, confidence: 1 };
|
|
@@ -6222,7 +6319,10 @@ export {
|
|
|
6222
6319
|
WhitespaceExtractor,
|
|
6223
6320
|
accumulateBlocks,
|
|
6224
6321
|
buildDisambiguation,
|
|
6322
|
+
buildDomainTokenizer,
|
|
6225
6323
|
buildFeedback,
|
|
6324
|
+
buildLanguageConfig,
|
|
6325
|
+
buildPatternProfile,
|
|
6226
6326
|
buildPhrase,
|
|
6227
6327
|
buildTablesFromProfiles,
|
|
6228
6328
|
combineClassifiers,
|
|
@@ -6253,6 +6353,7 @@ export {
|
|
|
6253
6353
|
createUnicodeRangeClassifier,
|
|
6254
6354
|
defineCommand,
|
|
6255
6355
|
defineRole,
|
|
6356
|
+
deriveRoleMarkers,
|
|
6256
6357
|
detectWordOrders,
|
|
6257
6358
|
extractCssSelector,
|
|
6258
6359
|
extractNumber,
|