@lokascript/framework 2.5.1 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/index.js +148 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +52 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +148 -1
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/index.cjs +148 -1
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +148 -1
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts +11 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/testing/index.js +12 -4
- package/dist/testing/index.js.map +1 -1
- package/package.json +3 -3
- package/src/core/tokenization/base-tokenizer.ts +219 -0
- package/src/core/tokenization/keyword-boundary.test.ts +73 -0
- package/src/interfaces/value-extractor.ts +23 -0
- package/src/core/pattern-matching/pattern-matcher.ts.backup +0 -1267
package/dist/index.cjs
CHANGED
|
@@ -2051,7 +2051,8 @@ function createTokenizerContext(tokenizer) {
|
|
|
2051
2051
|
direction: tokenizer.direction,
|
|
2052
2052
|
lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
|
|
2053
2053
|
isKeyword: tokenizer.isKeyword.bind(tokenizer),
|
|
2054
|
-
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
|
|
2054
|
+
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
|
|
2055
|
+
...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
|
|
2055
2056
|
};
|
|
2056
2057
|
if (tokenizer.normalizer) {
|
|
2057
2058
|
return { ...ctx, normalizer: tokenizer.normalizer };
|
|
@@ -5025,18 +5026,94 @@ function createLatinCharClassifiers(letterPattern) {
|
|
|
5025
5026
|
|
|
5026
5027
|
// src/core/tokenization/base-tokenizer.ts
|
|
5027
5028
|
var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
|
|
5029
|
+
var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
|
|
5030
|
+
// Role-marker role names (profile.roleMarkers normalizeds)
|
|
5031
|
+
"patient",
|
|
5032
|
+
"destination",
|
|
5033
|
+
"source",
|
|
5034
|
+
"style",
|
|
5035
|
+
"event",
|
|
5036
|
+
"eventMarker",
|
|
5037
|
+
"agent",
|
|
5038
|
+
"goal",
|
|
5039
|
+
"manner",
|
|
5040
|
+
// Prepositional / positional modifier concepts matched via the role mechanism
|
|
5041
|
+
// (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
|
|
5042
|
+
// NOT here — they are pattern literals (see the note above).
|
|
5043
|
+
"into",
|
|
5044
|
+
"from",
|
|
5045
|
+
"to",
|
|
5046
|
+
"with",
|
|
5047
|
+
"at",
|
|
5048
|
+
"of",
|
|
5049
|
+
"as",
|
|
5050
|
+
"by",
|
|
5051
|
+
"in",
|
|
5052
|
+
"on",
|
|
5053
|
+
"over",
|
|
5054
|
+
"under",
|
|
5055
|
+
"between",
|
|
5056
|
+
"through",
|
|
5057
|
+
"without"
|
|
5058
|
+
]);
|
|
5059
|
+
var ENGLISH_DOM_EVENT_NAMES = [
|
|
5060
|
+
"click",
|
|
5061
|
+
"dblclick",
|
|
5062
|
+
"input",
|
|
5063
|
+
"change",
|
|
5064
|
+
"submit",
|
|
5065
|
+
"keydown",
|
|
5066
|
+
"keyup",
|
|
5067
|
+
"keypress",
|
|
5068
|
+
"mousedown",
|
|
5069
|
+
"mouseup",
|
|
5070
|
+
"mouseover",
|
|
5071
|
+
"mouseout",
|
|
5072
|
+
"mouseenter",
|
|
5073
|
+
"mouseleave",
|
|
5074
|
+
"mousemove",
|
|
5075
|
+
"pointerdown",
|
|
5076
|
+
"pointerup",
|
|
5077
|
+
"pointermove",
|
|
5078
|
+
"focus",
|
|
5079
|
+
"blur",
|
|
5080
|
+
"load",
|
|
5081
|
+
"resize",
|
|
5082
|
+
"scroll"
|
|
5083
|
+
];
|
|
5028
5084
|
var _BaseTokenizer = class _BaseTokenizer {
|
|
5029
5085
|
constructor() {
|
|
5030
5086
|
/** Keywords derived from profile, sorted longest-first for greedy matching */
|
|
5031
5087
|
this.profileKeywords = [];
|
|
5088
|
+
/**
|
|
5089
|
+
* Space-containing profile keywords (multi-word phrases), longest-first.
|
|
5090
|
+
* Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
|
|
5091
|
+
* vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
|
|
5092
|
+
* profile-driven replacement for the per-language hardcoded compound lists.
|
|
5093
|
+
* Empty for no-space (CJK) languages, so they are unaffected.
|
|
5094
|
+
*/
|
|
5095
|
+
this.multiWordKeywords = [];
|
|
5032
5096
|
/** Map for O(1) keyword lookups by lowercase native word */
|
|
5033
5097
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
5098
|
+
/**
|
|
5099
|
+
* The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
|
|
5100
|
+
* The keyword map is keyed by native word with last-wins insertion, so a
|
|
5101
|
+
* duplicate native word inside the extras silently shadows the earlier entry
|
|
5102
|
+
* (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
|
|
5103
|
+
* positional expressions). Exposed so consistency tests can detect such
|
|
5104
|
+
* intra-extras collisions, which are invisible in the deduplicated map.
|
|
5105
|
+
*/
|
|
5106
|
+
this.rawExtraEntries = [];
|
|
5034
5107
|
/**
|
|
5035
5108
|
* Pluggable value extractors for domain-specific syntax.
|
|
5036
5109
|
* When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
|
|
5037
5110
|
*/
|
|
5038
5111
|
this.extractors = [];
|
|
5039
5112
|
}
|
|
5113
|
+
/** Raw extras as passed in, pre-dedup — for consistency tests. */
|
|
5114
|
+
getExtraKeywordEntries() {
|
|
5115
|
+
return this.rawExtraEntries;
|
|
5116
|
+
}
|
|
5040
5117
|
/**
|
|
5041
5118
|
* Tokenize input string to token stream.
|
|
5042
5119
|
* Delegates to extractor-based tokenization if extractors are registered,
|
|
@@ -5105,6 +5182,12 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5105
5182
|
pos++;
|
|
5106
5183
|
}
|
|
5107
5184
|
if (pos >= input.length) break;
|
|
5185
|
+
const multiWord = this.tryMultiWordKeyword(input, pos);
|
|
5186
|
+
if (multiWord) {
|
|
5187
|
+
tokens.push(multiWord);
|
|
5188
|
+
pos = multiWord.position.end;
|
|
5189
|
+
continue;
|
|
5190
|
+
}
|
|
5108
5191
|
let extracted = false;
|
|
5109
5192
|
for (const extractor of this.extractors) {
|
|
5110
5193
|
if (extractor.canExtract(input, pos)) {
|
|
@@ -5204,6 +5287,7 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5204
5287
|
*/
|
|
5205
5288
|
initializeKeywordsFromProfile(profile, extras = []) {
|
|
5206
5289
|
const keywordMap = /* @__PURE__ */ new Map();
|
|
5290
|
+
this.rawExtraEntries = extras;
|
|
5207
5291
|
if (profile.keywords) {
|
|
5208
5292
|
for (const [normalized2, translation] of Object.entries(profile.keywords)) {
|
|
5209
5293
|
keywordMap.set(translation.primary, {
|
|
@@ -5247,12 +5331,20 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5247
5331
|
keywordMap.set(native, { native, normalized: normalized2 });
|
|
5248
5332
|
}
|
|
5249
5333
|
}
|
|
5334
|
+
for (const evt of ENGLISH_DOM_EVENT_NAMES) {
|
|
5335
|
+
if (!keywordMap.has(evt)) {
|
|
5336
|
+
keywordMap.set(evt, { native: evt, normalized: evt });
|
|
5337
|
+
}
|
|
5338
|
+
}
|
|
5250
5339
|
for (const extra of extras) {
|
|
5251
5340
|
keywordMap.set(extra.native, extra);
|
|
5252
5341
|
}
|
|
5253
5342
|
this.profileKeywords = Array.from(keywordMap.values()).sort(
|
|
5254
5343
|
(a, b) => b.native.length - a.native.length
|
|
5255
5344
|
);
|
|
5345
|
+
this.multiWordKeywords = this.profileKeywords.filter(
|
|
5346
|
+
(k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
|
|
5347
|
+
);
|
|
5256
5348
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
5257
5349
|
for (const keyword of this.profileKeywords) {
|
|
5258
5350
|
this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
|
|
@@ -5294,6 +5386,35 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5294
5386
|
}
|
|
5295
5387
|
return null;
|
|
5296
5388
|
}
|
|
5389
|
+
/**
|
|
5390
|
+
* Match the longest multi-word (space-containing) profile keyword at `pos`,
|
|
5391
|
+
* requiring the match to end at a word boundary. The profile-driven
|
|
5392
|
+
* counterpart of the per-language hardcoded compound lists (the hindi and
|
|
5393
|
+
* vietnamese keyword extractors). Returns a keyword token (with the normalized
|
|
5394
|
+
* form) or null. Case-sensitive against the stored native form, mirroring
|
|
5395
|
+
* `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
|
|
5396
|
+
* case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
|
|
5397
|
+
*
|
|
5398
|
+
* @param input - Input string
|
|
5399
|
+
* @param pos - Current position (must be a token-start boundary)
|
|
5400
|
+
* @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
|
|
5401
|
+
*/
|
|
5402
|
+
tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
5403
|
+
if (this.multiWordKeywords.length === 0) return null;
|
|
5404
|
+
const rest = input.slice(pos);
|
|
5405
|
+
for (const entry of this.multiWordKeywords) {
|
|
5406
|
+
if (!rest.startsWith(entry.native)) continue;
|
|
5407
|
+
const after = input[pos + entry.native.length];
|
|
5408
|
+
if (after !== void 0 && isWordChar(after)) continue;
|
|
5409
|
+
return createToken(
|
|
5410
|
+
entry.native,
|
|
5411
|
+
"keyword",
|
|
5412
|
+
createPosition(pos, pos + entry.native.length),
|
|
5413
|
+
entry.normalized
|
|
5414
|
+
);
|
|
5415
|
+
}
|
|
5416
|
+
return null;
|
|
5417
|
+
}
|
|
5297
5418
|
/**
|
|
5298
5419
|
* Check if the remaining input starts with any known keyword.
|
|
5299
5420
|
* Useful for non-space languages to detect word boundaries.
|
|
@@ -5306,6 +5427,32 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
5306
5427
|
const remaining = input.slice(pos);
|
|
5307
5428
|
return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
|
|
5308
5429
|
}
|
|
5430
|
+
/**
|
|
5431
|
+
* Check if a known keyword starts at the given position AND ends at a word
|
|
5432
|
+
* boundary (end of input or a non-word character).
|
|
5433
|
+
*
|
|
5434
|
+
* Space-delimited languages must use this (not `isKeywordStart`) for
|
|
5435
|
+
* word-walk break checks: the keyword table includes English canonical
|
|
5436
|
+
* fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
|
|
5437
|
+
* word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
|
|
5438
|
+
* "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
|
|
5439
|
+
* keep using `isKeywordStart`.
|
|
5440
|
+
*
|
|
5441
|
+
* @param input - Input string
|
|
5442
|
+
* @param pos - Current position
|
|
5443
|
+
* @param isWordChar - Language-specific word-character predicate; pass the
|
|
5444
|
+
* tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
|
|
5445
|
+
* counts as part of a word. Defaults to Unicode letters/digits/underscore.
|
|
5446
|
+
* @returns true if a keyword starts here and is not followed by a word char
|
|
5447
|
+
*/
|
|
5448
|
+
isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
5449
|
+
const remaining = input.slice(pos);
|
|
5450
|
+
return this.profileKeywords.some((entry) => {
|
|
5451
|
+
if (!remaining.startsWith(entry.native)) return false;
|
|
5452
|
+
const after = input[pos + entry.native.length];
|
|
5453
|
+
return after === void 0 || !isWordChar(after);
|
|
5454
|
+
});
|
|
5455
|
+
}
|
|
5309
5456
|
/**
|
|
5310
5457
|
* Look up a keyword by native word (case-insensitive).
|
|
5311
5458
|
* O(1) lookup using the keyword map.
|