@lokascript/framework 2.8.0 → 2.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +393 -0
- package/dist/api/create-dsl.d.ts +93 -1
- package/dist/api/create-dsl.d.ts.map +1 -1
- package/dist/api/domain-registry.d.ts +5 -3
- package/dist/api/domain-registry.d.ts.map +1 -1
- package/dist/api/index.js +232 -18
- package/dist/api/index.js.map +1 -1
- package/dist/core/index.js +26 -6
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +15 -3
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +26 -6
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/core/tokenization/token-utils.d.ts +15 -0
- package/dist/core/tokenization/token-utils.d.ts.map +1 -1
- package/dist/generation/index.js +73 -47
- package/dist/generation/index.js.map +1 -1
- package/dist/generation/pattern-generator.d.ts +8 -1
- package/dist/generation/pattern-generator.d.ts.map +1 -1
- package/dist/generation/renderer.d.ts +53 -1
- package/dist/generation/renderer.d.ts.map +1 -1
- package/dist/index.cjs +302 -115
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +298 -115
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts +5 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/multilingual/index.js +25 -6
- package/dist/multilingual/index.js.map +1 -1
- package/package.json +4 -3
- package/src/api/create-dsl.test.ts +11 -0
- package/src/api/create-dsl.ts +278 -9
- package/src/api/domain-registry.ts +15 -10
- package/src/api/extensions.test.ts +322 -0
- package/src/core/tokenization/base-tokenizer.ts +23 -8
- package/src/core/tokenization/css-selector-extractor.test.ts +67 -0
- package/src/core/tokenization/token-utils.ts +18 -0
- package/src/generation/domain-renderer.test.ts +172 -0
- package/src/generation/pattern-generator.test.ts +102 -0
- package/src/generation/pattern-generator.ts +27 -19
- package/src/generation/renderer.test.ts +243 -4
- package/src/generation/renderer.ts +188 -45
- package/src/index.ts +9 -1
- package/src/interfaces/value-extractor.ts +50 -0
package/dist/core/index.js
CHANGED
|
@@ -125,6 +125,9 @@ function isQuote(char) {
|
|
|
125
125
|
function isDigit(char) {
|
|
126
126
|
return /\d/.test(char);
|
|
127
127
|
}
|
|
128
|
+
function stripOptionalDiacritics(word) {
|
|
129
|
+
return word.replace(/[ً-ْٰ]/g, "");
|
|
130
|
+
}
|
|
128
131
|
function isAsciiLetter(char) {
|
|
129
132
|
return /[a-zA-Z]/.test(char);
|
|
130
133
|
}
|
|
@@ -1080,7 +1083,7 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
1080
1083
|
* @returns Word without diacritics
|
|
1081
1084
|
*/
|
|
1082
1085
|
removeDiacritics(word) {
|
|
1083
|
-
return word
|
|
1086
|
+
return stripOptionalDiacritics(word);
|
|
1084
1087
|
}
|
|
1085
1088
|
/**
|
|
1086
1089
|
* Try to match a keyword from profile at the current position.
|
|
@@ -1171,24 +1174,40 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
1171
1174
|
});
|
|
1172
1175
|
}
|
|
1173
1176
|
/**
|
|
1174
|
-
* Look up a keyword by native word (case-insensitive).
|
|
1177
|
+
* Look up a keyword by native word (case-insensitive, diacritic-insensitive).
|
|
1175
1178
|
* O(1) lookup using the keyword map.
|
|
1176
1179
|
*
|
|
1180
|
+
* The map is INDEXED both with and without diacritics (see
|
|
1181
|
+
* `initializeKeywordsFromProfile`), so a stripped QUERY is the other half of
|
|
1182
|
+
* that: it lets a surface form carrying harakat the profile does not happen to
|
|
1183
|
+
* spell still find its entry. Only consulted after the exact lookup misses, so
|
|
1184
|
+
* every previously-matching word resolves byte-identically.
|
|
1185
|
+
*
|
|
1186
|
+
* Half-implementing this — indexing stripped but querying exact — is what made
|
|
1187
|
+
* diacritized `بَدِّل` (toggle) tokenize as `kind=particle normalized=with`:
|
|
1188
|
+
* `isKeyword` returned false, so the guard in `ArabicProcliticExtractor` that
|
|
1189
|
+
* exists to prevent exactly that handed the word on, and the single-char `ب`
|
|
1190
|
+
* bi- proclitic claimed it. A wrong CONCEPT, not a failed parse.
|
|
1191
|
+
*
|
|
1177
1192
|
* @param native - Native word to look up
|
|
1178
1193
|
* @returns KeywordEntry if found, undefined otherwise
|
|
1179
1194
|
*/
|
|
1180
1195
|
lookupKeyword(native) {
|
|
1181
|
-
|
|
1196
|
+
const exact = this.profileKeywordMap.get(native.toLowerCase());
|
|
1197
|
+
if (exact) return exact;
|
|
1198
|
+
const stripped = this.removeDiacritics(native);
|
|
1199
|
+
if (stripped === native) return void 0;
|
|
1200
|
+
return this.profileKeywordMap.get(stripped.toLowerCase());
|
|
1182
1201
|
}
|
|
1183
1202
|
/**
|
|
1184
|
-
* Check if a word is a known keyword (case-insensitive).
|
|
1185
|
-
* O(1) lookup using the keyword map.
|
|
1203
|
+
* Check if a word is a known keyword (case-insensitive, diacritic-insensitive).
|
|
1204
|
+
* O(1) lookup using the keyword map. See {@link lookupKeyword}.
|
|
1186
1205
|
*
|
|
1187
1206
|
* @param native - Native word to check
|
|
1188
1207
|
* @returns true if the word is a keyword
|
|
1189
1208
|
*/
|
|
1190
1209
|
isKeyword(native) {
|
|
1191
|
-
return this.
|
|
1210
|
+
return this.lookupKeyword(native) !== void 0;
|
|
1192
1211
|
}
|
|
1193
1212
|
/**
|
|
1194
1213
|
* Set the morphological normalizer for this tokenizer.
|
|
@@ -2984,6 +3003,7 @@ export {
|
|
|
2984
3003
|
noChange,
|
|
2985
3004
|
normalized,
|
|
2986
3005
|
patternMatcher,
|
|
3006
|
+
stripOptionalDiacritics,
|
|
2987
3007
|
validateValueType,
|
|
2988
3008
|
withDefaultExtractors
|
|
2989
3009
|
};
|