@lokascript/framework 2.5.1 → 2.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/index.js +148 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +52 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +148 -1
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/index.cjs +269 -17
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +265 -17
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts +11 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/multilingual/builders.d.ts +88 -0
- package/dist/multilingual/builders.d.ts.map +1 -0
- package/dist/multilingual/index.d.ts +7 -3
- package/dist/multilingual/index.d.ts.map +1 -1
- package/dist/multilingual/index.js +1549 -0
- package/dist/multilingual/index.js.map +1 -1
- package/dist/multilingual/types.d.ts +117 -0
- package/dist/multilingual/types.d.ts.map +1 -0
- package/dist/testing/index.js +12 -4
- package/dist/testing/index.js.map +1 -1
- package/package.json +4 -4
- package/src/core/tokenization/base-tokenizer.ts +219 -0
- package/src/core/tokenization/keyword-boundary.test.ts +73 -0
- package/src/index.ts +20 -0
- package/src/interfaces/value-extractor.ts +23 -0
- package/src/multilingual/bridge.test.ts +441 -0
- package/src/multilingual/builders.ts +224 -0
- package/src/multilingual/index.ts +23 -4
- package/src/multilingual/types.ts +121 -0
- package/src/core/pattern-matching/pattern-matcher.ts.backup +0 -1267
package/dist/index.cjs
CHANGED
|
@@ -47,7 +47,10 @@ __export(index_exports, {
|
|
|
47
47
|
WhitespaceExtractor: () => WhitespaceExtractor,
|
|
48
48
|
accumulateBlocks: () => accumulateBlocks,
|
|
49
49
|
buildDisambiguation: () => buildDisambiguation,
|
|
50
|
+
buildDomainTokenizer: () => buildDomainTokenizer,
|
|
50
51
|
buildFeedback: () => buildFeedback,
|
|
52
|
+
buildLanguageConfig: () => buildLanguageConfig,
|
|
53
|
+
buildPatternProfile: () => buildPatternProfile,
|
|
51
54
|
buildPhrase: () => buildPhrase,
|
|
52
55
|
buildTablesFromProfiles: () => buildTablesFromProfiles,
|
|
53
56
|
combineClassifiers: () => combineClassifiers,
|
|
@@ -78,6 +81,7 @@ __export(index_exports, {
|
|
|
78
81
|
createUnicodeRangeClassifier: () => createUnicodeRangeClassifier,
|
|
79
82
|
defineCommand: () => import_intent6.defineCommand,
|
|
80
83
|
defineRole: () => import_intent6.defineRole,
|
|
84
|
+
deriveRoleMarkers: () => deriveRoleMarkers,
|
|
81
85
|
detectWordOrders: () => detectWordOrders,
|
|
82
86
|
extractCssSelector: () => extractCssSelector,
|
|
83
87
|
extractNumber: () => extractNumber,
|
|
@@ -2051,7 +2055,8 @@ function createTokenizerContext(tokenizer) {
|
|
|
2051
2055
|
direction: tokenizer.direction,
|
|
2052
2056
|
lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
|
|
2053
2057
|
isKeyword: tokenizer.isKeyword.bind(tokenizer),
|
|
2054
|
-
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
|
|
2058
|
+
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
|
|
2059
|
+
...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
|
|
2055
2060
|
};
|
|
2056
2061
|
if (tokenizer.normalizer) {
|
|
2057
2062
|
return { ...ctx, normalizer: tokenizer.normalizer };
|
|
@@ -5007,36 +5012,96 @@ function withDefaultExtractors(tokenizer) {
|
|
|
5007
5012
|
return tokenizer;
|
|
5008
5013
|
}
|
|
5009
5014
|
|
|
5010
|
-
// src/core/tokenization/char-classifiers.ts
|
|
5011
|
-
function createUnicodeRangeClassifier(ranges) {
|
|
5012
|
-
return (char) => {
|
|
5013
|
-
const code = char.charCodeAt(0);
|
|
5014
|
-
return ranges.some(([start, end]) => code >= start && code <= end);
|
|
5015
|
-
};
|
|
5016
|
-
}
|
|
5017
|
-
function combineClassifiers(...classifiers) {
|
|
5018
|
-
return (char) => classifiers.some((fn) => fn(char));
|
|
5019
|
-
}
|
|
5020
|
-
function createLatinCharClassifiers(letterPattern) {
|
|
5021
|
-
const isLetter = (char) => letterPattern.test(char);
|
|
5022
|
-
const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
|
|
5023
|
-
return { isLetter, isIdentifierChar };
|
|
5024
|
-
}
|
|
5025
|
-
|
|
5026
5015
|
// src/core/tokenization/base-tokenizer.ts
|
|
5027
5016
|
var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
|
|
5017
|
+
var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
|
|
5018
|
+
// Role-marker role names (profile.roleMarkers normalizeds)
|
|
5019
|
+
"patient",
|
|
5020
|
+
"destination",
|
|
5021
|
+
"source",
|
|
5022
|
+
"style",
|
|
5023
|
+
"event",
|
|
5024
|
+
"eventMarker",
|
|
5025
|
+
"agent",
|
|
5026
|
+
"goal",
|
|
5027
|
+
"manner",
|
|
5028
|
+
// Prepositional / positional modifier concepts matched via the role mechanism
|
|
5029
|
+
// (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
|
|
5030
|
+
// NOT here — they are pattern literals (see the note above).
|
|
5031
|
+
"into",
|
|
5032
|
+
"from",
|
|
5033
|
+
"to",
|
|
5034
|
+
"with",
|
|
5035
|
+
"at",
|
|
5036
|
+
"of",
|
|
5037
|
+
"as",
|
|
5038
|
+
"by",
|
|
5039
|
+
"in",
|
|
5040
|
+
"on",
|
|
5041
|
+
"over",
|
|
5042
|
+
"under",
|
|
5043
|
+
"between",
|
|
5044
|
+
"through",
|
|
5045
|
+
"without"
|
|
5046
|
+
]);
|
|
5047
|
+
var ENGLISH_DOM_EVENT_NAMES = [
|
|
5048
|
+
"click",
|
|
5049
|
+
"dblclick",
|
|
5050
|
+
"input",
|
|
5051
|
+
"change",
|
|
5052
|
+
"submit",
|
|
5053
|
+
"keydown",
|
|
5054
|
+
"keyup",
|
|
5055
|
+
"keypress",
|
|
5056
|
+
"mousedown",
|
|
5057
|
+
"mouseup",
|
|
5058
|
+
"mouseover",
|
|
5059
|
+
"mouseout",
|
|
5060
|
+
"mouseenter",
|
|
5061
|
+
"mouseleave",
|
|
5062
|
+
"mousemove",
|
|
5063
|
+
"pointerdown",
|
|
5064
|
+
"pointerup",
|
|
5065
|
+
"pointermove",
|
|
5066
|
+
"focus",
|
|
5067
|
+
"blur",
|
|
5068
|
+
"load",
|
|
5069
|
+
"resize",
|
|
5070
|
+
"scroll"
|
|
5071
|
+
];
|
|
5028
5072
|
var _BaseTokenizer = class _BaseTokenizer {
|
|
5029
5073
|
constructor() {
|
|
5030
5074
|
/** Keywords derived from profile, sorted longest-first for greedy matching */
|
|
5031
5075
|
this.profileKeywords = [];
|
|
5076
|
+
/**
|
|
5077
|
+
* Space-containing profile keywords (multi-word phrases), longest-first.
|
|
5078
|
+
* Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
|
|
5079
|
+
* vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
|
|
5080
|
+
* profile-driven replacement for the per-language hardcoded compound lists.
|
|
5081
|
+
* Empty for no-space (CJK) languages, so they are unaffected.
|
|
5082
|
+
*/
|
|
5083
|
+
this.multiWordKeywords = [];
|
|
5032
5084
|
/** Map for O(1) keyword lookups by lowercase native word */
|
|
5033
5085
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
5086
|
+
/**
|
|
5087
|
+
* The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
|
|
5088
|
+
* The keyword map is keyed by native word with last-wins insertion, so a
|
|
5089
|
+
* duplicate native word inside the extras silently shadows the earlier entry
|
|
5090
|
+
* (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
|
|
5091
|
+
* positional expressions). Exposed so consistency tests can detect such
|
|
5092
|
+
* intra-extras collisions, which are invisible in the deduplicated map.
|
|
5093
|
+
*/
|
|
5094
|
+
this.rawExtraEntries = [];
|
|
5034
5095
|
/**
|
|
5035
5096
|
* Pluggable value extractors for domain-specific syntax.
|
|
5036
5097
|
* When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
|
|
5037
5098
|
*/
|
|
5038
5099
|
this.extractors = [];
|
|
5039
5100
|
}
|
|
5101
|
+
/** Raw extras as passed in, pre-dedup — for consistency tests. */
|
|
5102
|
+
getExtraKeywordEntries() {
|
|
5103
|
+
return this.rawExtraEntries;
|
|
5104
|
+
}
|
|
5040
5105
|
/**
|
|
5041
5106
|
* Tokenize input string to token stream.
|
|
5042
5107
|
* Delegates to extractor-based tokenization if extractors are registered,
|
|
@@ -5105,6 +5170,12 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5105
5170
|
pos++;
|
|
5106
5171
|
}
|
|
5107
5172
|
if (pos >= input.length) break;
|
|
5173
|
+
const multiWord = this.tryMultiWordKeyword(input, pos);
|
|
5174
|
+
if (multiWord) {
|
|
5175
|
+
tokens.push(multiWord);
|
|
5176
|
+
pos = multiWord.position.end;
|
|
5177
|
+
continue;
|
|
5178
|
+
}
|
|
5108
5179
|
let extracted = false;
|
|
5109
5180
|
for (const extractor of this.extractors) {
|
|
5110
5181
|
if (extractor.canExtract(input, pos)) {
|
|
@@ -5204,6 +5275,7 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5204
5275
|
*/
|
|
5205
5276
|
initializeKeywordsFromProfile(profile, extras = []) {
|
|
5206
5277
|
const keywordMap = /* @__PURE__ */ new Map();
|
|
5278
|
+
this.rawExtraEntries = extras;
|
|
5207
5279
|
if (profile.keywords) {
|
|
5208
5280
|
for (const [normalized2, translation] of Object.entries(profile.keywords)) {
|
|
5209
5281
|
keywordMap.set(translation.primary, {
|
|
@@ -5247,12 +5319,20 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5247
5319
|
keywordMap.set(native, { native, normalized: normalized2 });
|
|
5248
5320
|
}
|
|
5249
5321
|
}
|
|
5322
|
+
for (const evt of ENGLISH_DOM_EVENT_NAMES) {
|
|
5323
|
+
if (!keywordMap.has(evt)) {
|
|
5324
|
+
keywordMap.set(evt, { native: evt, normalized: evt });
|
|
5325
|
+
}
|
|
5326
|
+
}
|
|
5250
5327
|
for (const extra of extras) {
|
|
5251
5328
|
keywordMap.set(extra.native, extra);
|
|
5252
5329
|
}
|
|
5253
5330
|
this.profileKeywords = Array.from(keywordMap.values()).sort(
|
|
5254
5331
|
(a, b) => b.native.length - a.native.length
|
|
5255
5332
|
);
|
|
5333
|
+
this.multiWordKeywords = this.profileKeywords.filter(
|
|
5334
|
+
(k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
|
|
5335
|
+
);
|
|
5256
5336
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
5257
5337
|
for (const keyword of this.profileKeywords) {
|
|
5258
5338
|
this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
|
|
@@ -5294,6 +5374,35 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5294
5374
|
}
|
|
5295
5375
|
return null;
|
|
5296
5376
|
}
|
|
5377
|
+
/**
|
|
5378
|
+
* Match the longest multi-word (space-containing) profile keyword at `pos`,
|
|
5379
|
+
* requiring the match to end at a word boundary. The profile-driven
|
|
5380
|
+
* counterpart of the per-language hardcoded compound lists (the hindi and
|
|
5381
|
+
* vietnamese keyword extractors). Returns a keyword token (with the normalized
|
|
5382
|
+
* form) or null. Case-sensitive against the stored native form, mirroring
|
|
5383
|
+
* `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
|
|
5384
|
+
* case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
|
|
5385
|
+
*
|
|
5386
|
+
* @param input - Input string
|
|
5387
|
+
* @param pos - Current position (must be a token-start boundary)
|
|
5388
|
+
* @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
|
|
5389
|
+
*/
|
|
5390
|
+
tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
5391
|
+
if (this.multiWordKeywords.length === 0) return null;
|
|
5392
|
+
const rest = input.slice(pos);
|
|
5393
|
+
for (const entry of this.multiWordKeywords) {
|
|
5394
|
+
if (!rest.startsWith(entry.native)) continue;
|
|
5395
|
+
const after = input[pos + entry.native.length];
|
|
5396
|
+
if (after !== void 0 && isWordChar(after)) continue;
|
|
5397
|
+
return createToken(
|
|
5398
|
+
entry.native,
|
|
5399
|
+
"keyword",
|
|
5400
|
+
createPosition(pos, pos + entry.native.length),
|
|
5401
|
+
entry.normalized
|
|
5402
|
+
);
|
|
5403
|
+
}
|
|
5404
|
+
return null;
|
|
5405
|
+
}
|
|
5297
5406
|
/**
|
|
5298
5407
|
* Check if the remaining input starts with any known keyword.
|
|
5299
5408
|
* Useful for non-space languages to detect word boundaries.
|
|
@@ -5306,6 +5415,32 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5306
5415
|
const remaining = input.slice(pos);
|
|
5307
5416
|
return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
|
|
5308
5417
|
}
|
|
5418
|
+
/**
|
|
5419
|
+
* Check if a known keyword starts at the given position AND ends at a word
|
|
5420
|
+
* boundary (end of input or a non-word character).
|
|
5421
|
+
*
|
|
5422
|
+
* Space-delimited languages must use this (not `isKeywordStart`) for
|
|
5423
|
+
* word-walk break checks: the keyword table includes English canonical
|
|
5424
|
+
* fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
|
|
5425
|
+
* word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
|
|
5426
|
+
* "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
|
|
5427
|
+
* keep using `isKeywordStart`.
|
|
5428
|
+
*
|
|
5429
|
+
* @param input - Input string
|
|
5430
|
+
* @param pos - Current position
|
|
5431
|
+
* @param isWordChar - Language-specific word-character predicate; pass the
|
|
5432
|
+
* tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
|
|
5433
|
+
* counts as part of a word. Defaults to Unicode letters/digits/underscore.
|
|
5434
|
+
* @returns true if a keyword starts here and is not followed by a word char
|
|
5435
|
+
*/
|
|
5436
|
+
isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
5437
|
+
const remaining = input.slice(pos);
|
|
5438
|
+
return this.profileKeywords.some((entry) => {
|
|
5439
|
+
if (!remaining.startsWith(entry.native)) return false;
|
|
5440
|
+
const after = input[pos + entry.native.length];
|
|
5441
|
+
return after === void 0 || !isWordChar(after);
|
|
5442
|
+
});
|
|
5443
|
+
}
|
|
5309
5444
|
/**
|
|
5310
5445
|
* Look up a keyword by native word (case-insensitive).
|
|
5311
5446
|
* O(1) lookup using the keyword map.
|
|
@@ -5638,6 +5773,119 @@ function createSimpleTokenizer(config) {
|
|
|
5638
5773
|
return new SimpleTokenizer();
|
|
5639
5774
|
}
|
|
5640
5775
|
|
|
5776
|
+
// src/multilingual/builders.ts
|
|
5777
|
+
function mergeRoleMarkers(slice, vocab) {
|
|
5778
|
+
const merged = {};
|
|
5779
|
+
const add = (role, marker) => {
|
|
5780
|
+
if (!marker?.primary) {
|
|
5781
|
+
delete merged[role];
|
|
5782
|
+
return;
|
|
5783
|
+
}
|
|
5784
|
+
merged[role] = {
|
|
5785
|
+
primary: marker.primary,
|
|
5786
|
+
...marker.alternatives?.length && { alternatives: [...marker.alternatives] },
|
|
5787
|
+
...marker.position && { position: marker.position }
|
|
5788
|
+
};
|
|
5789
|
+
};
|
|
5790
|
+
for (const [role, marker] of Object.entries(slice.roleMarkers ?? {})) add(role, marker);
|
|
5791
|
+
for (const [role, marker] of Object.entries(vocab.roleMarkerOverrides ?? {})) add(role, marker);
|
|
5792
|
+
return merged;
|
|
5793
|
+
}
|
|
5794
|
+
function buildPatternProfile(slice, vocab) {
|
|
5795
|
+
const keywords = {};
|
|
5796
|
+
for (const [action, translation] of Object.entries(vocab.keywords)) {
|
|
5797
|
+
keywords[action] = {
|
|
5798
|
+
primary: translation.primary,
|
|
5799
|
+
...translation.alternatives?.length && { alternatives: [...translation.alternatives] }
|
|
5800
|
+
};
|
|
5801
|
+
}
|
|
5802
|
+
const roleMarkers = mergeRoleMarkers(slice, vocab);
|
|
5803
|
+
return {
|
|
5804
|
+
code: slice.code,
|
|
5805
|
+
wordOrder: slice.wordOrder,
|
|
5806
|
+
keywords,
|
|
5807
|
+
...Object.keys(roleMarkers).length > 0 && { roleMarkers }
|
|
5808
|
+
};
|
|
5809
|
+
}
|
|
5810
|
+
function defaultCaseInsensitive(script) {
|
|
5811
|
+
return script === void 0 || script === "latin" || script === "cyrillic";
|
|
5812
|
+
}
|
|
5813
|
+
function buildDomainTokenizer(slice, vocab, options = {}) {
|
|
5814
|
+
const roleMarkers = mergeRoleMarkers(slice, vocab);
|
|
5815
|
+
const keywords = /* @__PURE__ */ new Set();
|
|
5816
|
+
for (const translation of Object.values(vocab.keywords)) {
|
|
5817
|
+
keywords.add(translation.primary);
|
|
5818
|
+
for (const alt of translation.alternatives ?? []) keywords.add(alt);
|
|
5819
|
+
}
|
|
5820
|
+
for (const marker of Object.values(roleMarkers)) {
|
|
5821
|
+
keywords.add(marker.primary);
|
|
5822
|
+
for (const alt of marker.alternatives ?? []) keywords.add(alt);
|
|
5823
|
+
}
|
|
5824
|
+
for (const particle of slice.tokenization?.particles ?? []) keywords.add(particle);
|
|
5825
|
+
for (const extra of vocab.tokenizerKeywords ?? []) keywords.add(extra);
|
|
5826
|
+
const profileKeywords = {};
|
|
5827
|
+
for (const [action, translation] of Object.entries(vocab.keywords)) {
|
|
5828
|
+
profileKeywords[action] = {
|
|
5829
|
+
primary: translation.primary,
|
|
5830
|
+
...translation.alternatives?.length && { alternatives: [...translation.alternatives] },
|
|
5831
|
+
normalized: translation.normalized ?? action
|
|
5832
|
+
};
|
|
5833
|
+
}
|
|
5834
|
+
const keywordProfile = {
|
|
5835
|
+
keywords: profileKeywords,
|
|
5836
|
+
...Object.keys(roleMarkers).length > 0 && { roleMarkers }
|
|
5837
|
+
};
|
|
5838
|
+
const customExtractors = [
|
|
5839
|
+
...options.customExtractors ?? [],
|
|
5840
|
+
...slice.script === "latin" ? [new LatinExtendedIdentifierExtractor()] : []
|
|
5841
|
+
];
|
|
5842
|
+
return createSimpleTokenizer({
|
|
5843
|
+
language: slice.code,
|
|
5844
|
+
direction: slice.direction ?? "ltr",
|
|
5845
|
+
keywords: [...keywords],
|
|
5846
|
+
...vocab.keywordExtras?.length && { keywordExtras: vocab.keywordExtras.map((e) => ({ ...e })) },
|
|
5847
|
+
keywordProfile,
|
|
5848
|
+
includeOperators: options.includeOperators ?? false,
|
|
5849
|
+
caseInsensitive: options.caseInsensitive ?? defaultCaseInsensitive(slice.script),
|
|
5850
|
+
...customExtractors.length > 0 && { customExtractors }
|
|
5851
|
+
});
|
|
5852
|
+
}
|
|
5853
|
+
function buildLanguageConfig(slice, vocab, meta = {}) {
|
|
5854
|
+
const name = meta.name ?? slice.name ?? slice.code;
|
|
5855
|
+
return {
|
|
5856
|
+
code: slice.code,
|
|
5857
|
+
name,
|
|
5858
|
+
nativeName: meta.nativeName ?? slice.nativeName ?? name,
|
|
5859
|
+
tokenizer: meta.tokenizer ?? buildDomainTokenizer(slice, vocab, meta.tokenizerOptions),
|
|
5860
|
+
patternProfile: buildPatternProfile(slice, vocab),
|
|
5861
|
+
...meta.grammarProfile && { grammarProfile: meta.grammarProfile }
|
|
5862
|
+
};
|
|
5863
|
+
}
|
|
5864
|
+
function deriveRoleMarkers(slice, roleMapping) {
|
|
5865
|
+
const derived = {};
|
|
5866
|
+
for (const [domainRole, semanticRole] of Object.entries(roleMapping)) {
|
|
5867
|
+
const marker = slice.roleMarkers?.[semanticRole];
|
|
5868
|
+
if (marker?.primary) derived[domainRole] = marker.primary;
|
|
5869
|
+
}
|
|
5870
|
+
return derived;
|
|
5871
|
+
}
|
|
5872
|
+
|
|
5873
|
+
// src/core/tokenization/char-classifiers.ts
|
|
5874
|
+
function createUnicodeRangeClassifier(ranges) {
|
|
5875
|
+
return (char) => {
|
|
5876
|
+
const code = char.charCodeAt(0);
|
|
5877
|
+
return ranges.some(([start, end]) => code >= start && code <= end);
|
|
5878
|
+
};
|
|
5879
|
+
}
|
|
5880
|
+
function combineClassifiers(...classifiers) {
|
|
5881
|
+
return (char) => classifiers.some((fn) => fn(char));
|
|
5882
|
+
}
|
|
5883
|
+
function createLatinCharClassifiers(letterPattern) {
|
|
5884
|
+
const isLetter = (char) => letterPattern.test(char);
|
|
5885
|
+
const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
|
|
5886
|
+
return { isLetter, isIdentifierChar };
|
|
5887
|
+
}
|
|
5888
|
+
|
|
5641
5889
|
// src/core/tokenization/morphology/types.ts
|
|
5642
5890
|
function noChange(word) {
|
|
5643
5891
|
return { stem: word, confidence: 1 };
|
|
@@ -6204,7 +6452,10 @@ function accumulateIndented(statements, config) {
|
|
|
6204
6452
|
WhitespaceExtractor,
|
|
6205
6453
|
accumulateBlocks,
|
|
6206
6454
|
buildDisambiguation,
|
|
6455
|
+
buildDomainTokenizer,
|
|
6207
6456
|
buildFeedback,
|
|
6457
|
+
buildLanguageConfig,
|
|
6458
|
+
buildPatternProfile,
|
|
6208
6459
|
buildPhrase,
|
|
6209
6460
|
buildTablesFromProfiles,
|
|
6210
6461
|
combineClassifiers,
|
|
@@ -6235,6 +6486,7 @@ function accumulateIndented(statements, config) {
|
|
|
6235
6486
|
createUnicodeRangeClassifier,
|
|
6236
6487
|
defineCommand,
|
|
6237
6488
|
defineRole,
|
|
6489
|
+
deriveRoleMarkers,
|
|
6238
6490
|
detectWordOrders,
|
|
6239
6491
|
extractCssSelector,
|
|
6240
6492
|
extractNumber,
|