@lokascript/framework 2.5.1 → 2.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/index.js +148 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +52 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +148 -1
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/index.cjs +269 -17
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +265 -17
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts +11 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/multilingual/builders.d.ts +88 -0
- package/dist/multilingual/builders.d.ts.map +1 -0
- package/dist/multilingual/index.d.ts +7 -3
- package/dist/multilingual/index.d.ts.map +1 -1
- package/dist/multilingual/index.js +1549 -0
- package/dist/multilingual/index.js.map +1 -1
- package/dist/multilingual/types.d.ts +117 -0
- package/dist/multilingual/types.d.ts.map +1 -0
- package/dist/testing/index.js +12 -4
- package/dist/testing/index.js.map +1 -1
- package/package.json +4 -4
- package/src/core/tokenization/base-tokenizer.ts +219 -0
- package/src/core/tokenization/keyword-boundary.test.ts +73 -0
- package/src/index.ts +20 -0
- package/src/interfaces/value-extractor.ts +23 -0
- package/src/multilingual/bridge.test.ts +441 -0
- package/src/multilingual/builders.ts +224 -0
- package/src/multilingual/index.ts +23 -4
- package/src/multilingual/types.ts +121 -0
- package/src/core/pattern-matching/pattern-matcher.ts.backup +0 -1267
package/dist/index.d.ts
CHANGED
|
@@ -28,6 +28,8 @@ export { GrammarTransformer } from './grammar/transformer';
|
|
|
28
28
|
export type { TransformerConfig } from './grammar/transformer';
|
|
29
29
|
export { reorderRoles, insertMarkers, joinTokens } from './grammar/types';
|
|
30
30
|
export type { LanguageProfile, PatternTransform, ParsedElement, WordOrder, GrammaticalMarker, AdpositionType, } from './grammar/types';
|
|
31
|
+
export { buildPatternProfile, buildDomainTokenizer, buildLanguageConfig, deriveRoleMarkers, } from './multilingual';
|
|
32
|
+
export type { GrammarProfileSlice, DomainVocabulary, DomainKeywordTranslation, DomainKeywordEntry, RoleMarkerSlice, TokenizationSlice, VerbSlice, DomainTokenizerOptions, LanguageConfigMeta, } from './multilingual';
|
|
31
33
|
export * from './core';
|
|
32
34
|
export type { ActionType, SemanticRole as SemanticRoleType, SemanticValue, SemanticNode, CommandSemanticNode, EventHandlerSemanticNode, ConditionalSemanticNode, CompoundSemanticNode, LoopSemanticNode, LiteralValue, SelectorValue, ReferenceValue, PropertyPathValue, ExpressionValue, SemanticMetadata, SourcePosition, LanguageToken, TokenStream, LanguageTokenizer, LanguagePattern, Annotation, ProtocolDiagnostic, AsyncVariant, MatchArm, LSEEnvelope, } from './core/types';
|
|
33
35
|
export type { CommandSchema, RoleSpec } from './schema';
|
package/dist/index.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,cAAc,OAAO,CAAC;AAGtB,cAAc,OAAO,CAAC;AAGtB,cAAc,cAAc,CAAC;AAG7B,cAAc,UAAU,CAAC;AACzB,cAAc,cAAc,CAAC;AAG7B,OAAO,EAAE,kBAAkB,EAAE,MAAM,uBAAuB,CAAC;AAC3D,YAAY,EAAE,iBAAiB,EAAE,MAAM,uBAAuB,CAAC;AAC/D,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAC1E,YAAY,EACV,eAAe,EACf,gBAAgB,EAChB,aAAa,EACb,SAAS,EACT,iBAAiB,EACjB,cAAc,GACf,MAAM,iBAAiB,CAAC;
|
|
1
|
+
{"version":3,"file":"index.d.ts","sourceRoot":"","sources":["../src/index.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;;GAoBG;AAGH,cAAc,OAAO,CAAC;AAGtB,cAAc,OAAO,CAAC;AAGtB,cAAc,cAAc,CAAC;AAG7B,cAAc,UAAU,CAAC;AACzB,cAAc,cAAc,CAAC;AAG7B,OAAO,EAAE,kBAAkB,EAAE,MAAM,uBAAuB,CAAC;AAC3D,YAAY,EAAE,iBAAiB,EAAE,MAAM,uBAAuB,CAAC;AAC/D,OAAO,EAAE,YAAY,EAAE,aAAa,EAAE,UAAU,EAAE,MAAM,iBAAiB,CAAC;AAC1E,YAAY,EACV,eAAe,EACf,gBAAgB,EAChB,aAAa,EACb,SAAS,EACT,iBAAiB,EACjB,cAAc,GACf,MAAM,iBAAiB,CAAC;AAIzB,OAAO,EACL,mBAAmB,EACnB,oBAAoB,EACpB,mBAAmB,EACnB,iBAAiB,GAClB,MAAM,gBAAgB,CAAC;AACxB,YAAY,EACV,mBAAmB,EACnB,gBAAgB,EAChB,wBAAwB,EACxB,kBAAkB,EAClB,eAAe,EACf,iBAAiB,EACjB,SAAS,EACT,sBAAsB,EACtB,kBAAkB,GACnB,MAAM,gBAAgB,CAAC;AAGxB,cAAc,QAAQ,CAAC;AAGvB,YAAY,EACV,UAAU,EACV,YAAY,IAAI,gBAAgB,EAChC,aAAa,EACb,YAAY,EACZ,mBAAmB,EACnB,wBAAwB,EACxB,uBAAuB,EACvB,oBAAoB,EACpB,gBAAgB,EAChB,YAAY,EACZ,aAAa,EACb,cAAc,EACd,iBAAiB,EACjB,eAAe,EACf,gBAAgB,EAChB,cAAc,EACd,aAAa,EACb,WAAW,EACX,iBAAiB,EACjB,eAAe,EAEf,UAAU,EACV,kBAAkB,EAClB,YAAY,EACZ,QAAQ,EACR,WAAW,GACZ,MAAM,cAAc,CAAC;AAEtB,YAAY,EAAE,aAAa,EAAE,QAAQ,EAAE,MAAM,UAAU,CAAC;AAGxD,YAAY,EACV,uBAAuB,EACvB,YAAY,EACZ,WAAW,EACX,cAAc,GACf,MAAM,uBAAuB,CAAC;AAC/B,OAAO,EACL,aAAa,EACb,YAAY,EACZ,WAAW,EACX,uBAAuB,EACvB,gBAAgB,EAChB,oBAAoB,GACrB,MAAM,uBAAuB,CAAC;AAG/B,YAAY,EACV,kBAAkB,EAClB,UAAU,EACV,gBAAgB,EAChB,iBAAiB,EACjB,mBAAmB,EACnB,iBAAiB,GAClB,MAAM,0BAA0B,CAAC;AAClC,OAAO,EAAE,yBAAyB,EAAE,SAAS,EAAE,gBAAgB,EAAE,MAAM,0BAA0B,CAAC;AAElG,YAAY,EACV,cAAc,EACd,SAAS,EACT,eAAe,EACf,aAAa,EACb,gBAAgB,EAChB,aAAa,EACb,gBAAgB,EAChB,iBAAiB,EACjB,eAAe,EACf,cAAc,EACd,oBAAoB,EACpB,kBAAkB,EAClB,cAAc,EACd,iBAAiB,GAClB,MAAM,OAAO,CAAC;AAEf,OAAO,EAAE,cAAc,EAAE,qBAAqB,EAAE,MAAM,OAAO,CAAC;AAG9D,OAAO,EACL,aAAa,EACb,cAAc,EACd,eAAe,EACf,kBAAkB,EAClB,gBAAgB,EAChB,iBAAiB,EACjB,sBAAsB,EACtB,qBAAqB,EACrB,kBAAkB,EAClB,cAAc,EACd,YAAY,EACZ,gBAAgB,EAEhB,aAAa,EACb,eAAe,EACf,eAAe,GAChB,MAAM,cAAc,CAAC;AAEtB,OAAO,EAAE,aAAa,EAAE,UAAU,EAAE,MAAM,UAAU,CAAC;AACrD,OAAO,EAAE,qBAAqB,EAAE,MAAM,oCAAoC,CAAC;AAC3E,YAAY,EAAE,qBAAqB,EAAE,MAAM,oCAAoC,CAAC;AAGhF,OAAO,EAAE,2BAA2B,EAAE,MAAM,gDAAgD,CAAC;AAC7F,YAAY,EACV,iBAAiB,EACjB,gBAAgB,GACjB,MAAM,gDAAgD,CAAC;AAGxD,cAAc,MAAM,CAAC;AAGrB,cAAc,WAAW,CAAC;AAG1B,cAAc,YAAY,CAAC;AAG3B,cAAc,YAAY,CAAC;AAG3B,OAAO,EAAE,0BAA0B,EAAE,gBAAgB,EAAE,MAAM,2BAA2B,CAAC;AACzF,YAAY,EACV,oBAAoB,EACpB,oBAAoB,EACpB,oBAAoB,EACpB,eAAe,EACf,cAAc,EACd,cAAc,EACd,WAAW,EACX,WAAW,EACX,UAAU,EACV,aAAa,GACd,MAAM,2BAA2B,CAAC"}
|
package/dist/index.js
CHANGED
|
@@ -1908,7 +1908,8 @@ function createTokenizerContext(tokenizer) {
|
|
|
1908
1908
|
direction: tokenizer.direction,
|
|
1909
1909
|
lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
|
|
1910
1910
|
isKeyword: tokenizer.isKeyword.bind(tokenizer),
|
|
1911
|
-
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
|
|
1911
|
+
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
|
|
1912
|
+
...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
|
|
1912
1913
|
};
|
|
1913
1914
|
if (tokenizer.normalizer) {
|
|
1914
1915
|
return { ...ctx, normalizer: tokenizer.normalizer };
|
|
@@ -4879,36 +4880,96 @@ function withDefaultExtractors(tokenizer) {
|
|
|
4879
4880
|
return tokenizer;
|
|
4880
4881
|
}
|
|
4881
4882
|
|
|
4882
|
-
// src/core/tokenization/char-classifiers.ts
|
|
4883
|
-
function createUnicodeRangeClassifier(ranges) {
|
|
4884
|
-
return (char) => {
|
|
4885
|
-
const code = char.charCodeAt(0);
|
|
4886
|
-
return ranges.some(([start, end]) => code >= start && code <= end);
|
|
4887
|
-
};
|
|
4888
|
-
}
|
|
4889
|
-
function combineClassifiers(...classifiers) {
|
|
4890
|
-
return (char) => classifiers.some((fn) => fn(char));
|
|
4891
|
-
}
|
|
4892
|
-
function createLatinCharClassifiers(letterPattern) {
|
|
4893
|
-
const isLetter = (char) => letterPattern.test(char);
|
|
4894
|
-
const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
|
|
4895
|
-
return { isLetter, isIdentifierChar };
|
|
4896
|
-
}
|
|
4897
|
-
|
|
4898
4883
|
// src/core/tokenization/base-tokenizer.ts
|
|
4899
4884
|
var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
|
|
4885
|
+
var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
|
|
4886
|
+
// Role-marker role names (profile.roleMarkers normalizeds)
|
|
4887
|
+
"patient",
|
|
4888
|
+
"destination",
|
|
4889
|
+
"source",
|
|
4890
|
+
"style",
|
|
4891
|
+
"event",
|
|
4892
|
+
"eventMarker",
|
|
4893
|
+
"agent",
|
|
4894
|
+
"goal",
|
|
4895
|
+
"manner",
|
|
4896
|
+
// Prepositional / positional modifier concepts matched via the role mechanism
|
|
4897
|
+
// (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
|
|
4898
|
+
// NOT here — they are pattern literals (see the note above).
|
|
4899
|
+
"into",
|
|
4900
|
+
"from",
|
|
4901
|
+
"to",
|
|
4902
|
+
"with",
|
|
4903
|
+
"at",
|
|
4904
|
+
"of",
|
|
4905
|
+
"as",
|
|
4906
|
+
"by",
|
|
4907
|
+
"in",
|
|
4908
|
+
"on",
|
|
4909
|
+
"over",
|
|
4910
|
+
"under",
|
|
4911
|
+
"between",
|
|
4912
|
+
"through",
|
|
4913
|
+
"without"
|
|
4914
|
+
]);
|
|
4915
|
+
var ENGLISH_DOM_EVENT_NAMES = [
|
|
4916
|
+
"click",
|
|
4917
|
+
"dblclick",
|
|
4918
|
+
"input",
|
|
4919
|
+
"change",
|
|
4920
|
+
"submit",
|
|
4921
|
+
"keydown",
|
|
4922
|
+
"keyup",
|
|
4923
|
+
"keypress",
|
|
4924
|
+
"mousedown",
|
|
4925
|
+
"mouseup",
|
|
4926
|
+
"mouseover",
|
|
4927
|
+
"mouseout",
|
|
4928
|
+
"mouseenter",
|
|
4929
|
+
"mouseleave",
|
|
4930
|
+
"mousemove",
|
|
4931
|
+
"pointerdown",
|
|
4932
|
+
"pointerup",
|
|
4933
|
+
"pointermove",
|
|
4934
|
+
"focus",
|
|
4935
|
+
"blur",
|
|
4936
|
+
"load",
|
|
4937
|
+
"resize",
|
|
4938
|
+
"scroll"
|
|
4939
|
+
];
|
|
4900
4940
|
var _BaseTokenizer = class _BaseTokenizer {
|
|
4901
4941
|
constructor() {
|
|
4902
4942
|
/** Keywords derived from profile, sorted longest-first for greedy matching */
|
|
4903
4943
|
this.profileKeywords = [];
|
|
4944
|
+
/**
|
|
4945
|
+
* Space-containing profile keywords (multi-word phrases), longest-first.
|
|
4946
|
+
* Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
|
|
4947
|
+
* vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
|
|
4948
|
+
* profile-driven replacement for the per-language hardcoded compound lists.
|
|
4949
|
+
* Empty for no-space (CJK) languages, so they are unaffected.
|
|
4950
|
+
*/
|
|
4951
|
+
this.multiWordKeywords = [];
|
|
4904
4952
|
/** Map for O(1) keyword lookups by lowercase native word */
|
|
4905
4953
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
4954
|
+
/**
|
|
4955
|
+
* The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
|
|
4956
|
+
* The keyword map is keyed by native word with last-wins insertion, so a
|
|
4957
|
+
* duplicate native word inside the extras silently shadows the earlier entry
|
|
4958
|
+
* (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
|
|
4959
|
+
* positional expressions). Exposed so consistency tests can detect such
|
|
4960
|
+
* intra-extras collisions, which are invisible in the deduplicated map.
|
|
4961
|
+
*/
|
|
4962
|
+
this.rawExtraEntries = [];
|
|
4906
4963
|
/**
|
|
4907
4964
|
* Pluggable value extractors for domain-specific syntax.
|
|
4908
4965
|
* When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
|
|
4909
4966
|
*/
|
|
4910
4967
|
this.extractors = [];
|
|
4911
4968
|
}
|
|
4969
|
+
/** Raw extras as passed in, pre-dedup — for consistency tests. */
|
|
4970
|
+
getExtraKeywordEntries() {
|
|
4971
|
+
return this.rawExtraEntries;
|
|
4972
|
+
}
|
|
4912
4973
|
/**
|
|
4913
4974
|
* Tokenize input string to token stream.
|
|
4914
4975
|
* Delegates to extractor-based tokenization if extractors are registered,
|
|
@@ -4977,6 +5038,12 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
4977
5038
|
pos++;
|
|
4978
5039
|
}
|
|
4979
5040
|
if (pos >= input.length) break;
|
|
5041
|
+
const multiWord = this.tryMultiWordKeyword(input, pos);
|
|
5042
|
+
if (multiWord) {
|
|
5043
|
+
tokens.push(multiWord);
|
|
5044
|
+
pos = multiWord.position.end;
|
|
5045
|
+
continue;
|
|
5046
|
+
}
|
|
4980
5047
|
let extracted = false;
|
|
4981
5048
|
for (const extractor of this.extractors) {
|
|
4982
5049
|
if (extractor.canExtract(input, pos)) {
|
|
@@ -5076,6 +5143,7 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5076
5143
|
*/
|
|
5077
5144
|
initializeKeywordsFromProfile(profile, extras = []) {
|
|
5078
5145
|
const keywordMap = /* @__PURE__ */ new Map();
|
|
5146
|
+
this.rawExtraEntries = extras;
|
|
5079
5147
|
if (profile.keywords) {
|
|
5080
5148
|
for (const [normalized2, translation] of Object.entries(profile.keywords)) {
|
|
5081
5149
|
keywordMap.set(translation.primary, {
|
|
@@ -5119,12 +5187,20 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5119
5187
|
keywordMap.set(native, { native, normalized: normalized2 });
|
|
5120
5188
|
}
|
|
5121
5189
|
}
|
|
5190
|
+
for (const evt of ENGLISH_DOM_EVENT_NAMES) {
|
|
5191
|
+
if (!keywordMap.has(evt)) {
|
|
5192
|
+
keywordMap.set(evt, { native: evt, normalized: evt });
|
|
5193
|
+
}
|
|
5194
|
+
}
|
|
5122
5195
|
for (const extra of extras) {
|
|
5123
5196
|
keywordMap.set(extra.native, extra);
|
|
5124
5197
|
}
|
|
5125
5198
|
this.profileKeywords = Array.from(keywordMap.values()).sort(
|
|
5126
5199
|
(a, b) => b.native.length - a.native.length
|
|
5127
5200
|
);
|
|
5201
|
+
this.multiWordKeywords = this.profileKeywords.filter(
|
|
5202
|
+
(k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
|
|
5203
|
+
);
|
|
5128
5204
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
5129
5205
|
for (const keyword of this.profileKeywords) {
|
|
5130
5206
|
this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
|
|
@@ -5166,6 +5242,35 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5166
5242
|
}
|
|
5167
5243
|
return null;
|
|
5168
5244
|
}
|
|
5245
|
+
/**
|
|
5246
|
+
* Match the longest multi-word (space-containing) profile keyword at `pos`,
|
|
5247
|
+
* requiring the match to end at a word boundary. The profile-driven
|
|
5248
|
+
* counterpart of the per-language hardcoded compound lists (the hindi and
|
|
5249
|
+
* vietnamese keyword extractors). Returns a keyword token (with the normalized
|
|
5250
|
+
* form) or null. Case-sensitive against the stored native form, mirroring
|
|
5251
|
+
* `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
|
|
5252
|
+
* case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
|
|
5253
|
+
*
|
|
5254
|
+
* @param input - Input string
|
|
5255
|
+
* @param pos - Current position (must be a token-start boundary)
|
|
5256
|
+
* @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
|
|
5257
|
+
*/
|
|
5258
|
+
tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
5259
|
+
if (this.multiWordKeywords.length === 0) return null;
|
|
5260
|
+
const rest = input.slice(pos);
|
|
5261
|
+
for (const entry of this.multiWordKeywords) {
|
|
5262
|
+
if (!rest.startsWith(entry.native)) continue;
|
|
5263
|
+
const after = input[pos + entry.native.length];
|
|
5264
|
+
if (after !== void 0 && isWordChar(after)) continue;
|
|
5265
|
+
return createToken(
|
|
5266
|
+
entry.native,
|
|
5267
|
+
"keyword",
|
|
5268
|
+
createPosition(pos, pos + entry.native.length),
|
|
5269
|
+
entry.normalized
|
|
5270
|
+
);
|
|
5271
|
+
}
|
|
5272
|
+
return null;
|
|
5273
|
+
}
|
|
5169
5274
|
/**
|
|
5170
5275
|
* Check if the remaining input starts with any known keyword.
|
|
5171
5276
|
* Useful for non-space languages to detect word boundaries.
|
|
@@ -5178,6 +5283,32 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5178
5283
|
const remaining = input.slice(pos);
|
|
5179
5284
|
return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
|
|
5180
5285
|
}
|
|
5286
|
+
/**
|
|
5287
|
+
* Check if a known keyword starts at the given position AND ends at a word
|
|
5288
|
+
* boundary (end of input or a non-word character).
|
|
5289
|
+
*
|
|
5290
|
+
* Space-delimited languages must use this (not `isKeywordStart`) for
|
|
5291
|
+
* word-walk break checks: the keyword table includes English canonical
|
|
5292
|
+
* fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
|
|
5293
|
+
* word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
|
|
5294
|
+
* "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
|
|
5295
|
+
* keep using `isKeywordStart`.
|
|
5296
|
+
*
|
|
5297
|
+
* @param input - Input string
|
|
5298
|
+
* @param pos - Current position
|
|
5299
|
+
* @param isWordChar - Language-specific word-character predicate; pass the
|
|
5300
|
+
* tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
|
|
5301
|
+
* counts as part of a word. Defaults to Unicode letters/digits/underscore.
|
|
5302
|
+
* @returns true if a keyword starts here and is not followed by a word char
|
|
5303
|
+
*/
|
|
5304
|
+
isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
5305
|
+
const remaining = input.slice(pos);
|
|
5306
|
+
return this.profileKeywords.some((entry) => {
|
|
5307
|
+
if (!remaining.startsWith(entry.native)) return false;
|
|
5308
|
+
const after = input[pos + entry.native.length];
|
|
5309
|
+
return after === void 0 || !isWordChar(after);
|
|
5310
|
+
});
|
|
5311
|
+
}
|
|
5181
5312
|
/**
|
|
5182
5313
|
* Look up a keyword by native word (case-insensitive).
|
|
5183
5314
|
* O(1) lookup using the keyword map.
|
|
@@ -5510,6 +5641,119 @@ function createSimpleTokenizer(config) {
|
|
|
5510
5641
|
return new SimpleTokenizer();
|
|
5511
5642
|
}
|
|
5512
5643
|
|
|
5644
|
+
// src/multilingual/builders.ts
|
|
5645
|
+
function mergeRoleMarkers(slice, vocab) {
|
|
5646
|
+
const merged = {};
|
|
5647
|
+
const add = (role, marker) => {
|
|
5648
|
+
if (!marker?.primary) {
|
|
5649
|
+
delete merged[role];
|
|
5650
|
+
return;
|
|
5651
|
+
}
|
|
5652
|
+
merged[role] = {
|
|
5653
|
+
primary: marker.primary,
|
|
5654
|
+
...marker.alternatives?.length && { alternatives: [...marker.alternatives] },
|
|
5655
|
+
...marker.position && { position: marker.position }
|
|
5656
|
+
};
|
|
5657
|
+
};
|
|
5658
|
+
for (const [role, marker] of Object.entries(slice.roleMarkers ?? {})) add(role, marker);
|
|
5659
|
+
for (const [role, marker] of Object.entries(vocab.roleMarkerOverrides ?? {})) add(role, marker);
|
|
5660
|
+
return merged;
|
|
5661
|
+
}
|
|
5662
|
+
function buildPatternProfile(slice, vocab) {
|
|
5663
|
+
const keywords = {};
|
|
5664
|
+
for (const [action, translation] of Object.entries(vocab.keywords)) {
|
|
5665
|
+
keywords[action] = {
|
|
5666
|
+
primary: translation.primary,
|
|
5667
|
+
...translation.alternatives?.length && { alternatives: [...translation.alternatives] }
|
|
5668
|
+
};
|
|
5669
|
+
}
|
|
5670
|
+
const roleMarkers = mergeRoleMarkers(slice, vocab);
|
|
5671
|
+
return {
|
|
5672
|
+
code: slice.code,
|
|
5673
|
+
wordOrder: slice.wordOrder,
|
|
5674
|
+
keywords,
|
|
5675
|
+
...Object.keys(roleMarkers).length > 0 && { roleMarkers }
|
|
5676
|
+
};
|
|
5677
|
+
}
|
|
5678
|
+
function defaultCaseInsensitive(script) {
|
|
5679
|
+
return script === void 0 || script === "latin" || script === "cyrillic";
|
|
5680
|
+
}
|
|
5681
|
+
function buildDomainTokenizer(slice, vocab, options = {}) {
|
|
5682
|
+
const roleMarkers = mergeRoleMarkers(slice, vocab);
|
|
5683
|
+
const keywords = /* @__PURE__ */ new Set();
|
|
5684
|
+
for (const translation of Object.values(vocab.keywords)) {
|
|
5685
|
+
keywords.add(translation.primary);
|
|
5686
|
+
for (const alt of translation.alternatives ?? []) keywords.add(alt);
|
|
5687
|
+
}
|
|
5688
|
+
for (const marker of Object.values(roleMarkers)) {
|
|
5689
|
+
keywords.add(marker.primary);
|
|
5690
|
+
for (const alt of marker.alternatives ?? []) keywords.add(alt);
|
|
5691
|
+
}
|
|
5692
|
+
for (const particle of slice.tokenization?.particles ?? []) keywords.add(particle);
|
|
5693
|
+
for (const extra of vocab.tokenizerKeywords ?? []) keywords.add(extra);
|
|
5694
|
+
const profileKeywords = {};
|
|
5695
|
+
for (const [action, translation] of Object.entries(vocab.keywords)) {
|
|
5696
|
+
profileKeywords[action] = {
|
|
5697
|
+
primary: translation.primary,
|
|
5698
|
+
...translation.alternatives?.length && { alternatives: [...translation.alternatives] },
|
|
5699
|
+
normalized: translation.normalized ?? action
|
|
5700
|
+
};
|
|
5701
|
+
}
|
|
5702
|
+
const keywordProfile = {
|
|
5703
|
+
keywords: profileKeywords,
|
|
5704
|
+
...Object.keys(roleMarkers).length > 0 && { roleMarkers }
|
|
5705
|
+
};
|
|
5706
|
+
const customExtractors = [
|
|
5707
|
+
...options.customExtractors ?? [],
|
|
5708
|
+
...slice.script === "latin" ? [new LatinExtendedIdentifierExtractor()] : []
|
|
5709
|
+
];
|
|
5710
|
+
return createSimpleTokenizer({
|
|
5711
|
+
language: slice.code,
|
|
5712
|
+
direction: slice.direction ?? "ltr",
|
|
5713
|
+
keywords: [...keywords],
|
|
5714
|
+
...vocab.keywordExtras?.length && { keywordExtras: vocab.keywordExtras.map((e) => ({ ...e })) },
|
|
5715
|
+
keywordProfile,
|
|
5716
|
+
includeOperators: options.includeOperators ?? false,
|
|
5717
|
+
caseInsensitive: options.caseInsensitive ?? defaultCaseInsensitive(slice.script),
|
|
5718
|
+
...customExtractors.length > 0 && { customExtractors }
|
|
5719
|
+
});
|
|
5720
|
+
}
|
|
5721
|
+
function buildLanguageConfig(slice, vocab, meta = {}) {
|
|
5722
|
+
const name = meta.name ?? slice.name ?? slice.code;
|
|
5723
|
+
return {
|
|
5724
|
+
code: slice.code,
|
|
5725
|
+
name,
|
|
5726
|
+
nativeName: meta.nativeName ?? slice.nativeName ?? name,
|
|
5727
|
+
tokenizer: meta.tokenizer ?? buildDomainTokenizer(slice, vocab, meta.tokenizerOptions),
|
|
5728
|
+
patternProfile: buildPatternProfile(slice, vocab),
|
|
5729
|
+
...meta.grammarProfile && { grammarProfile: meta.grammarProfile }
|
|
5730
|
+
};
|
|
5731
|
+
}
|
|
5732
|
+
function deriveRoleMarkers(slice, roleMapping) {
|
|
5733
|
+
const derived = {};
|
|
5734
|
+
for (const [domainRole, semanticRole] of Object.entries(roleMapping)) {
|
|
5735
|
+
const marker = slice.roleMarkers?.[semanticRole];
|
|
5736
|
+
if (marker?.primary) derived[domainRole] = marker.primary;
|
|
5737
|
+
}
|
|
5738
|
+
return derived;
|
|
5739
|
+
}
|
|
5740
|
+
|
|
5741
|
+
// src/core/tokenization/char-classifiers.ts
|
|
5742
|
+
function createUnicodeRangeClassifier(ranges) {
|
|
5743
|
+
return (char) => {
|
|
5744
|
+
const code = char.charCodeAt(0);
|
|
5745
|
+
return ranges.some(([start, end]) => code >= start && code <= end);
|
|
5746
|
+
};
|
|
5747
|
+
}
|
|
5748
|
+
function combineClassifiers(...classifiers) {
|
|
5749
|
+
return (char) => classifiers.some((fn) => fn(char));
|
|
5750
|
+
}
|
|
5751
|
+
function createLatinCharClassifiers(letterPattern) {
|
|
5752
|
+
const isLetter = (char) => letterPattern.test(char);
|
|
5753
|
+
const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
|
|
5754
|
+
return { isLetter, isIdentifierChar };
|
|
5755
|
+
}
|
|
5756
|
+
|
|
5513
5757
|
// src/core/tokenization/morphology/types.ts
|
|
5514
5758
|
function noChange(word) {
|
|
5515
5759
|
return { stem: word, confidence: 1 };
|
|
@@ -6075,7 +6319,10 @@ export {
|
|
|
6075
6319
|
WhitespaceExtractor,
|
|
6076
6320
|
accumulateBlocks,
|
|
6077
6321
|
buildDisambiguation,
|
|
6322
|
+
buildDomainTokenizer,
|
|
6078
6323
|
buildFeedback,
|
|
6324
|
+
buildLanguageConfig,
|
|
6325
|
+
buildPatternProfile,
|
|
6079
6326
|
buildPhrase,
|
|
6080
6327
|
buildTablesFromProfiles,
|
|
6081
6328
|
combineClassifiers,
|
|
@@ -6106,6 +6353,7 @@ export {
|
|
|
6106
6353
|
createUnicodeRangeClassifier,
|
|
6107
6354
|
defineCommand,
|
|
6108
6355
|
defineRole,
|
|
6356
|
+
deriveRoleMarkers,
|
|
6109
6357
|
detectWordOrders,
|
|
6110
6358
|
extractCssSelector,
|
|
6111
6359
|
extractNumber,
|