@lokascript/framework 2.8.0 → 2.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/CHANGELOG.md +393 -0
  2. package/dist/api/create-dsl.d.ts +93 -1
  3. package/dist/api/create-dsl.d.ts.map +1 -1
  4. package/dist/api/domain-registry.d.ts +5 -3
  5. package/dist/api/domain-registry.d.ts.map +1 -1
  6. package/dist/api/index.js +232 -18
  7. package/dist/api/index.js.map +1 -1
  8. package/dist/core/index.js +26 -6
  9. package/dist/core/index.js.map +1 -1
  10. package/dist/core/tokenization/base-tokenizer.d.ts +15 -3
  11. package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
  12. package/dist/core/tokenization/index.js +26 -6
  13. package/dist/core/tokenization/index.js.map +1 -1
  14. package/dist/core/tokenization/token-utils.d.ts +15 -0
  15. package/dist/core/tokenization/token-utils.d.ts.map +1 -1
  16. package/dist/generation/index.js +73 -47
  17. package/dist/generation/index.js.map +1 -1
  18. package/dist/generation/pattern-generator.d.ts +8 -1
  19. package/dist/generation/pattern-generator.d.ts.map +1 -1
  20. package/dist/generation/renderer.d.ts +53 -1
  21. package/dist/generation/renderer.d.ts.map +1 -1
  22. package/dist/index.cjs +302 -115
  23. package/dist/index.cjs.map +1 -1
  24. package/dist/index.d.ts +4 -4
  25. package/dist/index.d.ts.map +1 -1
  26. package/dist/index.js +298 -115
  27. package/dist/index.js.map +1 -1
  28. package/dist/interfaces/value-extractor.d.ts +5 -0
  29. package/dist/interfaces/value-extractor.d.ts.map +1 -1
  30. package/dist/multilingual/index.js +25 -6
  31. package/dist/multilingual/index.js.map +1 -1
  32. package/package.json +4 -3
  33. package/src/api/create-dsl.test.ts +11 -0
  34. package/src/api/create-dsl.ts +278 -9
  35. package/src/api/domain-registry.ts +15 -10
  36. package/src/api/extensions.test.ts +322 -0
  37. package/src/core/tokenization/base-tokenizer.ts +23 -8
  38. package/src/core/tokenization/css-selector-extractor.test.ts +67 -0
  39. package/src/core/tokenization/token-utils.ts +18 -0
  40. package/src/generation/domain-renderer.test.ts +172 -0
  41. package/src/generation/pattern-generator.test.ts +102 -0
  42. package/src/generation/pattern-generator.ts +27 -19
  43. package/src/generation/renderer.test.ts +243 -4
  44. package/src/generation/renderer.ts +188 -45
  45. package/src/index.ts +9 -1
  46. package/src/interfaces/value-extractor.ts +50 -0
@@ -125,6 +125,9 @@ function isQuote(char) {
125
125
  function isDigit(char) {
126
126
  return /\d/.test(char);
127
127
  }
128
+ function stripOptionalDiacritics(word) {
129
+ return word.replace(/[ً-ْٰ]/g, "");
130
+ }
128
131
  function isAsciiLetter(char) {
129
132
  return /[a-zA-Z]/.test(char);
130
133
  }
@@ -1080,7 +1083,7 @@ var _BaseTokenizer = class _BaseTokenizer {
1080
1083
  * @returns Word without diacritics
1081
1084
  */
1082
1085
  removeDiacritics(word) {
1083
- return word.replace(/[\u064B-\u0652\u0670]/g, "");
1086
+ return stripOptionalDiacritics(word);
1084
1087
  }
1085
1088
  /**
1086
1089
  * Try to match a keyword from profile at the current position.
@@ -1171,24 +1174,40 @@ var _BaseTokenizer = class _BaseTokenizer {
1171
1174
  });
1172
1175
  }
1173
1176
  /**
1174
- * Look up a keyword by native word (case-insensitive).
1177
+ * Look up a keyword by native word (case-insensitive, diacritic-insensitive).
1175
1178
  * O(1) lookup using the keyword map.
1176
1179
  *
1180
+ * The map is INDEXED both with and without diacritics (see
1181
+ * `initializeKeywordsFromProfile`), so a stripped QUERY is the other half of
1182
+ * that: it lets a surface form carrying harakat the profile does not happen to
1183
+ * spell still find its entry. Only consulted after the exact lookup misses, so
1184
+ * every previously-matching word resolves byte-identically.
1185
+ *
1186
+ * Half-implementing this — indexing stripped but querying exact — is what made
1187
+ * diacritized `بَدِّل` (toggle) tokenize as `kind=particle normalized=with`:
1188
+ * `isKeyword` returned false, so the guard in `ArabicProcliticExtractor` that
1189
+ * exists to prevent exactly that handed the word on, and the single-char `ب`
1190
+ * bi- proclitic claimed it. A wrong CONCEPT, not a failed parse.
1191
+ *
1177
1192
  * @param native - Native word to look up
1178
1193
  * @returns KeywordEntry if found, undefined otherwise
1179
1194
  */
1180
1195
  lookupKeyword(native) {
1181
- return this.profileKeywordMap.get(native.toLowerCase());
1196
+ const exact = this.profileKeywordMap.get(native.toLowerCase());
1197
+ if (exact) return exact;
1198
+ const stripped = this.removeDiacritics(native);
1199
+ if (stripped === native) return void 0;
1200
+ return this.profileKeywordMap.get(stripped.toLowerCase());
1182
1201
  }
1183
1202
  /**
1184
- * Check if a word is a known keyword (case-insensitive).
1185
- * O(1) lookup using the keyword map.
1203
+ * Check if a word is a known keyword (case-insensitive, diacritic-insensitive).
1204
+ * O(1) lookup using the keyword map. See {@link lookupKeyword}.
1186
1205
  *
1187
1206
  * @param native - Native word to check
1188
1207
  * @returns true if the word is a keyword
1189
1208
  */
1190
1209
  isKeyword(native) {
1191
- return this.profileKeywordMap.has(native.toLowerCase());
1210
+ return this.lookupKeyword(native) !== void 0;
1192
1211
  }
1193
1212
  /**
1194
1213
  * Set the morphological normalizer for this tokenizer.
@@ -2984,6 +3003,7 @@ export {
2984
3003
  noChange,
2985
3004
  normalized,
2986
3005
  patternMatcher,
3006
+ stripOptionalDiacritics,
2987
3007
  validateValueType,
2988
3008
  withDefaultExtractors
2989
3009
  };