@lokascript/framework 2.5.0 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/index.js +148 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +52 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +148 -1
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/index.cjs +148 -1
- package/dist/index.cjs.map +1 -1
- package/dist/index.js +148 -1
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts +11 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/testing/index.js +12 -4
- package/dist/testing/index.js.map +1 -1
- package/package.json +3 -3
- package/src/core/tokenization/base-tokenizer.ts +219 -0
- package/src/core/tokenization/keyword-boundary.test.ts +73 -0
- package/src/interfaces/value-extractor.ts +23 -0
- package/src/core/pattern-matching/pattern-matcher.ts.backup +0 -1267
package/dist/core/index.js
CHANGED
|
@@ -662,7 +662,8 @@ function createTokenizerContext(tokenizer) {
|
|
|
662
662
|
direction: tokenizer.direction,
|
|
663
663
|
lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
|
|
664
664
|
isKeyword: tokenizer.isKeyword.bind(tokenizer),
|
|
665
|
-
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
|
|
665
|
+
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
|
|
666
|
+
...tokenizer.isKeywordStartAtBoundary ? { isKeywordStartAtBoundary: tokenizer.isKeywordStartAtBoundary.bind(tokenizer) } : {}
|
|
666
667
|
};
|
|
667
668
|
if (tokenizer.normalizer) {
|
|
668
669
|
return { ...ctx, normalizer: tokenizer.normalizer };
|
|
@@ -710,18 +711,94 @@ function createLatinCharClassifiers(letterPattern) {
|
|
|
710
711
|
|
|
711
712
|
// src/core/tokenization/base-tokenizer.ts
|
|
712
713
|
var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
|
|
714
|
+
var MARKER_CONCEPT_NORMALIZEDS = /* @__PURE__ */ new Set([
|
|
715
|
+
// Role-marker role names (profile.roleMarkers normalizeds)
|
|
716
|
+
"patient",
|
|
717
|
+
"destination",
|
|
718
|
+
"source",
|
|
719
|
+
"style",
|
|
720
|
+
"event",
|
|
721
|
+
"eventMarker",
|
|
722
|
+
"agent",
|
|
723
|
+
"goal",
|
|
724
|
+
"manner",
|
|
725
|
+
// Prepositional / positional modifier concepts matched via the role mechanism
|
|
726
|
+
// (profile.keywords "Modifiers"). `before`/`after`/`until` are intentionally
|
|
727
|
+
// NOT here — they are pattern literals (see the note above).
|
|
728
|
+
"into",
|
|
729
|
+
"from",
|
|
730
|
+
"to",
|
|
731
|
+
"with",
|
|
732
|
+
"at",
|
|
733
|
+
"of",
|
|
734
|
+
"as",
|
|
735
|
+
"by",
|
|
736
|
+
"in",
|
|
737
|
+
"on",
|
|
738
|
+
"over",
|
|
739
|
+
"under",
|
|
740
|
+
"between",
|
|
741
|
+
"through",
|
|
742
|
+
"without"
|
|
743
|
+
]);
|
|
744
|
+
var ENGLISH_DOM_EVENT_NAMES = [
|
|
745
|
+
"click",
|
|
746
|
+
"dblclick",
|
|
747
|
+
"input",
|
|
748
|
+
"change",
|
|
749
|
+
"submit",
|
|
750
|
+
"keydown",
|
|
751
|
+
"keyup",
|
|
752
|
+
"keypress",
|
|
753
|
+
"mousedown",
|
|
754
|
+
"mouseup",
|
|
755
|
+
"mouseover",
|
|
756
|
+
"mouseout",
|
|
757
|
+
"mouseenter",
|
|
758
|
+
"mouseleave",
|
|
759
|
+
"mousemove",
|
|
760
|
+
"pointerdown",
|
|
761
|
+
"pointerup",
|
|
762
|
+
"pointermove",
|
|
763
|
+
"focus",
|
|
764
|
+
"blur",
|
|
765
|
+
"load",
|
|
766
|
+
"resize",
|
|
767
|
+
"scroll"
|
|
768
|
+
];
|
|
713
769
|
var _BaseTokenizer = class _BaseTokenizer {
|
|
714
770
|
constructor() {
|
|
715
771
|
/** Keywords derived from profile, sorted longest-first for greedy matching */
|
|
716
772
|
this.profileKeywords = [];
|
|
773
|
+
/**
|
|
774
|
+
* Space-containing profile keywords (multi-word phrases), longest-first.
|
|
775
|
+
* Used by `tryMultiWordKeyword` so natural spaced forms (hi `मेल खाता`,
|
|
776
|
+
* vi `chuyển đổi`, es `tecla abajo`, …) tokenize as ONE keyword — the
|
|
777
|
+
* profile-driven replacement for the per-language hardcoded compound lists.
|
|
778
|
+
* Empty for no-space (CJK) languages, so they are unaffected.
|
|
779
|
+
*/
|
|
780
|
+
this.multiWordKeywords = [];
|
|
717
781
|
/** Map for O(1) keyword lookups by lowercase native word */
|
|
718
782
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
783
|
+
/**
|
|
784
|
+
* The raw EXTRAS list passed to initializeKeywordsFromProfile, kept pre-dedup.
|
|
785
|
+
* The keyword map is keyed by native word with last-wins insertion, so a
|
|
786
|
+
* duplicate native word inside the extras silently shadows the earlier entry
|
|
787
|
+
* (e.g. a `nächste→closest` entry shadowing `nächste→next` broke German
|
|
788
|
+
* positional expressions). Exposed so consistency tests can detect such
|
|
789
|
+
* intra-extras collisions, which are invisible in the deduplicated map.
|
|
790
|
+
*/
|
|
791
|
+
this.rawExtraEntries = [];
|
|
719
792
|
/**
|
|
720
793
|
* Pluggable value extractors for domain-specific syntax.
|
|
721
794
|
* When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
|
|
722
795
|
*/
|
|
723
796
|
this.extractors = [];
|
|
724
797
|
}
|
|
798
|
+
/** Raw extras as passed in, pre-dedup — for consistency tests. */
|
|
799
|
+
getExtraKeywordEntries() {
|
|
800
|
+
return this.rawExtraEntries;
|
|
801
|
+
}
|
|
725
802
|
/**
|
|
726
803
|
* Tokenize input string to token stream.
|
|
727
804
|
* Delegates to extractor-based tokenization if extractors are registered,
|
|
@@ -790,6 +867,12 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
790
867
|
pos++;
|
|
791
868
|
}
|
|
792
869
|
if (pos >= input.length) break;
|
|
870
|
+
const multiWord = this.tryMultiWordKeyword(input, pos);
|
|
871
|
+
if (multiWord) {
|
|
872
|
+
tokens.push(multiWord);
|
|
873
|
+
pos = multiWord.position.end;
|
|
874
|
+
continue;
|
|
875
|
+
}
|
|
793
876
|
let extracted = false;
|
|
794
877
|
for (const extractor of this.extractors) {
|
|
795
878
|
if (extractor.canExtract(input, pos)) {
|
|
@@ -889,6 +972,7 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
889
972
|
*/
|
|
890
973
|
initializeKeywordsFromProfile(profile, extras = []) {
|
|
891
974
|
const keywordMap = /* @__PURE__ */ new Map();
|
|
975
|
+
this.rawExtraEntries = extras;
|
|
892
976
|
if (profile.keywords) {
|
|
893
977
|
for (const [normalized2, translation] of Object.entries(profile.keywords)) {
|
|
894
978
|
keywordMap.set(translation.primary, {
|
|
@@ -932,12 +1016,20 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
932
1016
|
keywordMap.set(native, { native, normalized: normalized2 });
|
|
933
1017
|
}
|
|
934
1018
|
}
|
|
1019
|
+
for (const evt of ENGLISH_DOM_EVENT_NAMES) {
|
|
1020
|
+
if (!keywordMap.has(evt)) {
|
|
1021
|
+
keywordMap.set(evt, { native: evt, normalized: evt });
|
|
1022
|
+
}
|
|
1023
|
+
}
|
|
935
1024
|
for (const extra of extras) {
|
|
936
1025
|
keywordMap.set(extra.native, extra);
|
|
937
1026
|
}
|
|
938
1027
|
this.profileKeywords = Array.from(keywordMap.values()).sort(
|
|
939
1028
|
(a, b) => b.native.length - a.native.length
|
|
940
1029
|
);
|
|
1030
|
+
this.multiWordKeywords = this.profileKeywords.filter(
|
|
1031
|
+
(k) => k.native.includes(" ") && !MARKER_CONCEPT_NORMALIZEDS.has(k.normalized)
|
|
1032
|
+
);
|
|
941
1033
|
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
942
1034
|
for (const keyword of this.profileKeywords) {
|
|
943
1035
|
this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
|
|
@@ -979,6 +1071,35 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
979
1071
|
}
|
|
980
1072
|
return null;
|
|
981
1073
|
}
|
|
1074
|
+
/**
|
|
1075
|
+
* Match the longest multi-word (space-containing) profile keyword at `pos`,
|
|
1076
|
+
* requiring the match to end at a word boundary. The profile-driven
|
|
1077
|
+
* counterpart of the per-language hardcoded compound lists (the hindi and
|
|
1078
|
+
* vietnamese keyword extractors). Returns a keyword token (with the normalized
|
|
1079
|
+
* form) or null. Case-sensitive against the stored native form, mirroring
|
|
1080
|
+
* `tryProfileKeyword`/`isKeywordStart` (the i18n dicts emit a fixed surface
|
|
1081
|
+
* case). No-op when `multiWordKeywords` is empty (no-space/CJK languages).
|
|
1082
|
+
*
|
|
1083
|
+
* @param input - Input string
|
|
1084
|
+
* @param pos - Current position (must be a token-start boundary)
|
|
1085
|
+
* @param isWordChar - End-boundary predicate (defaults to Unicode letter/digit/_)
|
|
1086
|
+
*/
|
|
1087
|
+
tryMultiWordKeyword(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
1088
|
+
if (this.multiWordKeywords.length === 0) return null;
|
|
1089
|
+
const rest = input.slice(pos);
|
|
1090
|
+
for (const entry of this.multiWordKeywords) {
|
|
1091
|
+
if (!rest.startsWith(entry.native)) continue;
|
|
1092
|
+
const after = input[pos + entry.native.length];
|
|
1093
|
+
if (after !== void 0 && isWordChar(after)) continue;
|
|
1094
|
+
return createToken(
|
|
1095
|
+
entry.native,
|
|
1096
|
+
"keyword",
|
|
1097
|
+
createPosition(pos, pos + entry.native.length),
|
|
1098
|
+
entry.normalized
|
|
1099
|
+
);
|
|
1100
|
+
}
|
|
1101
|
+
return null;
|
|
1102
|
+
}
|
|
982
1103
|
/**
|
|
983
1104
|
* Check if the remaining input starts with any known keyword.
|
|
984
1105
|
* Useful for non-space languages to detect word boundaries.
|
|
@@ -991,6 +1112,32 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
991
1112
|
const remaining = input.slice(pos);
|
|
992
1113
|
return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
|
|
993
1114
|
}
|
|
1115
|
+
/**
|
|
1116
|
+
* Check if a known keyword starts at the given position AND ends at a word
|
|
1117
|
+
* boundary (end of input or a non-word character).
|
|
1118
|
+
*
|
|
1119
|
+
* Space-delimited languages must use this (not `isKeywordStart`) for
|
|
1120
|
+
* word-walk break checks: the keyword table includes English canonical
|
|
1121
|
+
* fallbacks (me, it, you, …), so a raw `startsWith` check splits any native
|
|
1122
|
+
* word with an embedded fallback mid-word (e.g. Quechua ñit'iy contains
|
|
1123
|
+
* "it"). CJK/no-space tokenizers rely on mid-text keyword starts and must
|
|
1124
|
+
* keep using `isKeywordStart`.
|
|
1125
|
+
*
|
|
1126
|
+
* @param input - Input string
|
|
1127
|
+
* @param pos - Current position
|
|
1128
|
+
* @param isWordChar - Language-specific word-character predicate; pass the
|
|
1129
|
+
* tokenizer's letter classifier so e.g. the Quechua glottal apostrophe
|
|
1130
|
+
* counts as part of a word. Defaults to Unicode letters/digits/underscore.
|
|
1131
|
+
* @returns true if a keyword starts here and is not followed by a word char
|
|
1132
|
+
*/
|
|
1133
|
+
isKeywordStartAtBoundary(input, pos, isWordChar = (ch) => /[\p{L}\p{N}_]/u.test(ch)) {
|
|
1134
|
+
const remaining = input.slice(pos);
|
|
1135
|
+
return this.profileKeywords.some((entry) => {
|
|
1136
|
+
if (!remaining.startsWith(entry.native)) return false;
|
|
1137
|
+
const after = input[pos + entry.native.length];
|
|
1138
|
+
return after === void 0 || !isWordChar(after);
|
|
1139
|
+
});
|
|
1140
|
+
}
|
|
994
1141
|
/**
|
|
995
1142
|
* Look up a keyword by native word (case-insensitive).
|
|
996
1143
|
* O(1) lookup using the keyword map.
|