@lokascript/framework 2.5.0 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/index.js +148 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +52 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +148 -1
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/index.cjs +148 -1
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +148 -1
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts +11 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/testing/index.js +12 -4
- package/dist/testing/index.js.map +1 -1
- package/package.json +3 -3
- package/src/core/tokenization/base-tokenizer.ts +219 -0
- package/src/core/tokenization/keyword-boundary.test.ts +73 -0
- package/src/interfaces/value-extractor.ts +23 -0
- package/src/core/pattern-matching/pattern-matcher.ts.backup +0 -1267
package/dist/index.js
CHANGED
|
@@ -1908,7 +1908,8 @@ function createTokenizerContext(tokenizer) {
|
|
|
1908
1908
|
direction: tokenizer.direction,
|
|
1909
1909
|
lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
|
|
1910
1910
|
isKeyword: tokenizer.isKeyword.bind(tokenizer),
|
|
1911
|
-
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
|
|
1911
|
+
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
|
|
1912
|
+
...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
|
|
1912
1913
|
};
|
|
1913
1914
|
if (tokenizer.normalizer) {
|
|
1914
1915
|
return { ...ctx, normalizer: tokenizer.normalizer };
|
|
@@ -4897,18 +4898,94 @@ function createLatinCharClassifiers(letterPattern) {
|
|
|
4897
4898
|
|
|
4898
4899
|
// src/core/tokenization/base-tokenizer.ts
|
|
4899
4900
|
var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
|
|
4901
|
+
var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
|
|
4902
|
+
// Role-marker role names (profile.roleMarkers normalizeds)
|
|
4903
|
+
"patient",
|
|
4904
|
+
"destination",
|
|
4905
|
+
"source",
|
|
4906
|
+
"style",
|
|
4907
|
+
"event",
|
|
4908
|
+
"eventMarker",
|
|
4909
|
+
"agent",
|
|
4910
|
+
"goal",
|
|
4911
|
+
"manner",
|
|
4912
|
+
// Prepositional / positional modifier concepts matched via the role mechanism
|
|
4913
|
+
// (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
|
|
4914
|
+
// NOT here — they are pattern literals (see the note above).
|
|
4915
|
+
"into",
|
|
4916
|
+
"from",
|
|
4917
|
+
"to",
|
|
4918
|
+
"with",
|
|
4919
|
+
"at",
|
|
4920
|
+
"of",
|
|
4921
|
+
"as",
|
|
4922
|
+
"by",
|
|
4923
|
+
"in",
|
|
4924
|
+
"on",
|
|
4925
|
+
"over",
|
|
4926
|
+
"under",
|
|
4927
|
+
"between",
|
|
4928
|
+
"through",
|
|
4929
|
+
"without"
|
|
4930
|
+
]);
|
|
4931
|
+
var ENGLISH_DOM_EVENT_NAMES = [
|
|
4932
|
+
"click",
|
|
4933
|
+
"dblclick",
|
|
4934
|
+
"input",
|
|
4935
|
+
"change",
|
|
4936
|
+
"submit",
|
|
4937
|
+
"keydown",
|
|
4938
|
+
"keyup",
|
|
4939
|
+
"keypress",
|
|
4940
|
+
"mousedown",
|
|
4941
|
+
"mouseup",
|
|
4942
|
+
"mouseover",
|
|
4943
|
+
"mouseout",
|
|
4944
|
+
"mouseenter",
|
|
4945
|
+
"mouseleave",
|
|
4946
|
+
"mousemove",
|
|
4947
|
+
"pointerdown",
|
|
4948
|
+
"pointerup",
|
|
4949
|
+
"pointermove",
|
|
4950
|
+
"focus",
|
|
4951
|
+
"blur",
|
|
4952
|
+
"load",
|
|
4953
|
+
"resize",
|
|
4954
|
+
"scroll"
|
|
4955
|
+
];
|
|
4900
4956
|
var _BaseTokenizer = class _BaseTokenizer {
|
|
4901
4957
|
constructor() {
|
|
4902
4958
|
/** Keywords derived from profile, sorted longest-first for greedy matching */
|
|
4903
4959
|
this.profileKeywords = [];
|
|
4960
|
+
/**
|
|
4961
|
+
* Space-containing profile keywords (multi-word phrases), longest-first.
|
|
4962
|
+
* Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
|
|
4963
|
+
* vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
|
|
4964
|
+
* profile-driven replacement for the per-language hardcoded compound lists.
|
|
4965
|
+
* Empty for no-space (CJK) languages, so they are unaffected.
|
|
4966
|
+
*/
|
|
4967
|
+
this.multiWordKeywords = [];
|
|
4904
4968
|
/** Map for O(1) keyword lookups by lowercase native word */
|
|
4905
4969
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
4970
|
+
/**
|
|
4971
|
+
* The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
|
|
4972
|
+
* The keyword map is keyed by native word with last-wins insertion, so a
|
|
4973
|
+
* duplicate native word inside the extras silently shadows the earlier entry
|
|
4974
|
+
* (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
|
|
4975
|
+
* positional expressions). Exposed so consistency tests can detect such
|
|
4976
|
+
* intra-extras collisions, which are invisible in the deduplicated map.
|
|
4977
|
+
*/
|
|
4978
|
+
this.rawExtraEntries = [];
|
|
4906
4979
|
/**
|
|
4907
4980
|
* Pluggable value extractors for domain-specific syntax.
|
|
4908
4981
|
* When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
|
|
4909
4982
|
*/
|
|
4910
4983
|
this.extractors = [];
|
|
4911
4984
|
}
|
|
4985
|
+
/** Raw extras as passed in, pre-dedup — for consistency tests. */
|
|
4986
|
+
getExtraKeywordEntries() {
|
|
4987
|
+
return this.rawExtraEntries;
|
|
4988
|
+
}
|
|
4912
4989
|
/**
|
|
4913
4990
|
* Tokenize input string to token stream.
|
|
4914
4991
|
* Delegates to extractor-based tokenization if extractors are registered,
|
|
@@ -4977,6 +5054,12 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
4977
5054
|
pos++;
|
|
4978
5055
|
}
|
|
4979
5056
|
if (pos >= input.length) break;
|
|
5057
|
+
const multiWord = this.tryMultiWordKeyword(input, pos);
|
|
5058
|
+
if (multiWord) {
|
|
5059
|
+
tokens.push(multiWord);
|
|
5060
|
+
pos = multiWord.position.end;
|
|
5061
|
+
continue;
|
|
5062
|
+
}
|
|
4980
5063
|
let extracted = false;
|
|
4981
5064
|
for (const extractor of this.extractors) {
|
|
4982
5065
|
if (extractor.canExtract(input, pos)) {
|
|
@@ -5076,6 +5159,7 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5076
5159
|
*/
|
|
5077
5160
|
initializeKeywordsFromProfile(profile, extras = []) {
|
|
5078
5161
|
const keywordMap = /* @__PURE__ */ new Map();
|
|
5162
|
+
this.rawExtraEntries = extras;
|
|
5079
5163
|
if (profile.keywords) {
|
|
5080
5164
|
for (const [normalized2, translation] of Object.entries(profile.keywords)) {
|
|
5081
5165
|
keywordMap.set(translation.primary, {
|
|
@@ -5119,12 +5203,20 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5119
5203
|
keywordMap.set(native, { native, normalized: normalized2 });
|
|
5120
5204
|
}
|
|
5121
5205
|
}
|
|
5206
|
+
for (const evt of ENGLISH_DOM_EVENT_NAMES) {
|
|
5207
|
+
if (!keywordMap.has(evt)) {
|
|
5208
|
+
keywordMap.set(evt, { native: evt, normalized: evt });
|
|
5209
|
+
}
|
|
5210
|
+
}
|
|
5122
5211
|
for (const extra of extras) {
|
|
5123
5212
|
keywordMap.set(extra.native, extra);
|
|
5124
5213
|
}
|
|
5125
5214
|
this.profileKeywords = Array.from(keywordMap.values()).sort(
|
|
5126
5215
|
(a, b) => b.native.length - a.native.length
|
|
5127
5216
|
);
|
|
5217
|
+
this.multiWordKeywords = this.profileKeywords.filter(
|
|
5218
|
+
(k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
|
|
5219
|
+
);
|
|
5128
5220
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
5129
5221
|
for (const keyword of this.profileKeywords) {
|
|
5130
5222
|
this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
|
|
@@ -5166,6 +5258,35 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5166
5258
|
}
|
|
5167
5259
|
return null;
|
|
5168
5260
|
}
|
|
5261
|
+
/**
|
|
5262
|
+
* Match the longest multi-word (space-containing) profile keyword at `pos`,
|
|
5263
|
+
* requiring the match to end at a word boundary. The profile-driven
|
|
5264
|
+
* counterpart of the per-language hardcoded compound lists (the hindi and
|
|
5265
|
+
* vietnamese keyword extractors). Returns a keyword token (with the normalized
|
|
5266
|
+
* form) or null. Case-sensitive against the stored native form, mirroring
|
|
5267
|
+
* `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
|
|
5268
|
+
* case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
|
|
5269
|
+
*
|
|
5270
|
+
* @param input - Input string
|
|
5271
|
+
* @param pos - Current position (must be a token-start boundary)
|
|
5272
|
+
* @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
|
|
5273
|
+
*/
|
|
5274
|
+
tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
5275
|
+
if (this.multiWordKeywords.length === 0) return null;
|
|
5276
|
+
const rest = input.slice(pos);
|
|
5277
|
+
for (const entry of this.multiWordKeywords) {
|
|
5278
|
+
if (!rest.startsWith(entry.native)) continue;
|
|
5279
|
+
const after = input[pos + entry.native.length];
|
|
5280
|
+
if (after !== void 0 && isWordChar(after)) continue;
|
|
5281
|
+
return createToken(
|
|
5282
|
+
entry.native,
|
|
5283
|
+
"keyword",
|
|
5284
|
+
createPosition(pos, pos + entry.native.length),
|
|
5285
|
+
entry.normalized
|
|
5286
|
+
);
|
|
5287
|
+
}
|
|
5288
|
+
return null;
|
|
5289
|
+
}
|
|
5169
5290
|
/**
|
|
5170
5291
|
* Check if the remaining input starts with any known keyword.
|
|
5171
5292
|
* Useful for non-space languages to detect word boundaries.
|
|
@@ -5178,6 +5299,32 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5178
5299
|
const remaining = input.slice(pos);
|
|
5179
5300
|
return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
|
|
5180
5301
|
}
|
|
5302
|
+
/**
|
|
5303
|
+
* Check if a known keyword starts at the given position AND ends at a word
|
|
5304
|
+
* boundary (end of input or a non-word character).
|
|
5305
|
+
*
|
|
5306
|
+
* Space-delimited languages must use this (not `isKeywordStart`) for
|
|
5307
|
+
* word-walk break checks: the keyword table includes English canonical
|
|
5308
|
+
* fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
|
|
5309
|
+
* word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
|
|
5310
|
+
* "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
|
|
5311
|
+
* keep using `isKeywordStart`.
|
|
5312
|
+
*
|
|
5313
|
+
* @param input - Input string
|
|
5314
|
+
* @param pos - Current position
|
|
5315
|
+
* @param isWordChar - Language-specific word-character predicate; pass the
|
|
5316
|
+
* tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
|
|
5317
|
+
* counts as part of a word. Defaults to Unicode letters/digits/underscore.
|
|
5318
|
+
* @returns true if a keyword starts here and is not followed by a word char
|
|
5319
|
+
*/
|
|
5320
|
+
isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
5321
|
+
const remaining = input.slice(pos);
|
|
5322
|
+
return this.profileKeywords.some((entry) => {
|
|
5323
|
+
if (!remaining.startsWith(entry.native)) return false;
|
|
5324
|
+
const after = input[pos + entry.native.length];
|
|
5325
|
+
return after === void 0 || !isWordChar(after);
|
|
5326
|
+
});
|
|
5327
|
+
}
|
|
5181
5328
|
/**
|
|
5182
5329
|
* Look up a keyword by native word (case-insensitive).
|
|
5183
5330
|
* O(1) lookup using the keyword map.
|