@lokascript/framework 2.7.2 → 2.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/CHANGELOG.md +393 -0
  2. package/dist/api/create-dsl.d.ts +93 -1
  3. package/dist/api/create-dsl.d.ts.map +1 -1
  4. package/dist/api/domain-registry.d.ts +5 -3
  5. package/dist/api/domain-registry.d.ts.map +1 -1
  6. package/dist/api/index.js +238 -20
  7. package/dist/api/index.js.map +1 -1
  8. package/dist/core/index.js +67 -7
  9. package/dist/core/index.js.map +1 -1
  10. package/dist/core/tokenization/base-tokenizer.d.ts +39 -3
  11. package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
  12. package/dist/core/tokenization/extractors.d.ts +6 -0
  13. package/dist/core/tokenization/extractors.d.ts.map +1 -1
  14. package/dist/core/tokenization/index.js +67 -7
  15. package/dist/core/tokenization/index.js.map +1 -1
  16. package/dist/core/tokenization/token-utils.d.ts +15 -0
  17. package/dist/core/tokenization/token-utils.d.ts.map +1 -1
  18. package/dist/generation/index.js +75 -48
  19. package/dist/generation/index.js.map +1 -1
  20. package/dist/generation/pattern-generator.d.ts +8 -1
  21. package/dist/generation/pattern-generator.d.ts.map +1 -1
  22. package/dist/generation/renderer.d.ts +53 -1
  23. package/dist/generation/renderer.d.ts.map +1 -1
  24. package/dist/index.cjs +349 -118
  25. package/dist/index.cjs.map +1 -1
  26. package/dist/index.d.ts +4 -4
  27. package/dist/index.d.ts.map +1 -1
  28. package/dist/index.js +345 -118
  29. package/dist/index.js.map +1 -1
  30. package/dist/interfaces/value-extractor.d.ts +5 -0
  31. package/dist/interfaces/value-extractor.d.ts.map +1 -1
  32. package/dist/multilingual/index.js +66 -7
  33. package/dist/multilingual/index.js.map +1 -1
  34. package/dist/testing/index.js +4 -12
  35. package/dist/testing/index.js.map +1 -1
  36. package/package.json +4 -3
  37. package/src/api/create-dsl.test.ts +11 -0
  38. package/src/api/create-dsl.ts +278 -9
  39. package/src/api/domain-registry.ts +15 -10
  40. package/src/api/extensions.test.ts +322 -0
  41. package/src/core/tokenization/base-tokenizer.ts +78 -9
  42. package/src/core/tokenization/colon-qualifier.test.ts +129 -0
  43. package/src/core/tokenization/css-selector-extractor.test.ts +67 -0
  44. package/src/core/tokenization/extractors.ts +6 -0
  45. package/src/core/tokenization/token-utils.ts +18 -0
  46. package/src/generation/domain-renderer.test.ts +172 -0
  47. package/src/generation/pattern-generator.test.ts +102 -0
  48. package/src/generation/pattern-generator.ts +32 -20
  49. package/src/generation/renderer.test.ts +243 -4
  50. package/src/generation/renderer.ts +188 -45
  51. package/src/index.ts +9 -1
  52. package/src/interfaces/value-extractor.ts +50 -0
  53. package/src/ir/protocol-json.test.ts +21 -0
  54. package/src/ir/references.test.ts +5 -2
  55. package/src/prompts/prompt-generator.ts +4 -1
@@ -125,6 +125,9 @@ function isQuote(char) {
125
125
  function isDigit(char) {
126
126
  return /\d/.test(char);
127
127
  }
128
+ function stripOptionalDiacritics(word) {
129
+ return word.replace(/[ً-ْٰ]/g, "");
130
+ }
128
131
  function isAsciiLetter(char) {
129
132
  return /[a-zA-Z]/.test(char);
130
133
  }
@@ -915,7 +918,39 @@ var _BaseTokenizer = class _BaseTokenizer {
915
918
  pos++;
916
919
  }
917
920
  }
918
- return new TokenStreamImpl(tokens, this.language);
921
+ return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
922
+ }
923
+ /**
924
+ * Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
925
+ *
926
+ * `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
927
+ * preceded by an identifier is a qualifier (custom event namespace), not a
928
+ * sigil. The English tokenizer already merges these inside
929
+ * EnglishKeywordExtractor; this post-pass gives the other 23 languages the
930
+ * same stream. Strict position adjacency is the discriminator: whitespace
931
+ * between the tokens (`trigger :start`) breaks `end === start`, so a spaced
932
+ * local-variable reference survives untouched.
933
+ *
934
+ * Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
935
+ * sets tokenize `:` as bare punctuation (length 1), which never matches
936
+ * COLON_QUALIFIER, so this pass is a no-op for them.
937
+ */
938
+ mergeColonQualifiedNames(tokens) {
939
+ const out = [];
940
+ for (const tok of tokens) {
941
+ const prev = out[out.length - 1];
942
+ if (prev && _BaseTokenizer.ASCII_WORD.test(prev.value) && _BaseTokenizer.COLON_QUALIFIER.test(tok.value) && prev.position.end === tok.position.start) {
943
+ const merged = prev.value + tok.value;
944
+ out[out.length - 1] = createToken(
945
+ merged,
946
+ this.classifyToken(merged),
947
+ createPosition(prev.position.start, tok.position.end)
948
+ );
949
+ continue;
950
+ }
951
+ out.push(tok);
952
+ }
953
+ return out;
919
954
  }
920
955
  /**
921
956
  * Classify an unknown character when no extractor matches.
@@ -1048,7 +1083,7 @@ var _BaseTokenizer = class _BaseTokenizer {
1048
1083
  * @returns Word without diacritics
1049
1084
  */
1050
1085
  removeDiacritics(word) {
1051
- return word.replace(/[\u064B-\u0652\u0670]/g, "");
1086
+ return stripOptionalDiacritics(word);
1052
1087
  }
1053
1088
  /**
1054
1089
  * Try to match a keyword from profile at the current position.
@@ -1139,24 +1174,40 @@ var _BaseTokenizer = class _BaseTokenizer {
1139
1174
  });
1140
1175
  }
1141
1176
  /**
1142
- * Look up a keyword by native word (case-insensitive).
1177
+ * Look up a keyword by native word (case-insensitive, diacritic-insensitive).
1143
1178
  * O(1) lookup using the keyword map.
1144
1179
  *
1180
+ * The map is INDEXED both with and without diacritics (see
1181
+ * `initializeKeywordsFromProfile`), so a stripped QUERY is the other half of
1182
+ * that: it lets a surface form carrying harakat the profile does not happen to
1183
+ * spell still find its entry. Only consulted after the exact lookup misses, so
1184
+ * every previously-matching word resolves byte-identically.
1185
+ *
1186
+ * Half-implementing this — indexing stripped but querying exact — is what made
1187
+ * diacritized `بَدِّل` (toggle) tokenize as `kind=particle normalized=with`:
1188
+ * `isKeyword` returned false, so the guard in `ArabicProcliticExtractor` that
1189
+ * exists to prevent exactly that handed the word on, and the single-char `ب`
1190
+ * bi- proclitic claimed it. A wrong CONCEPT, not a failed parse.
1191
+ *
1145
1192
  * @param native - Native word to look up
1146
1193
  * @returns KeywordEntry if found, undefined otherwise
1147
1194
  */
1148
1195
  lookupKeyword(native) {
1149
- return this.profileKeywordMap.get(native.toLowerCase());
1196
+ const exact = this.profileKeywordMap.get(native.toLowerCase());
1197
+ if (exact) return exact;
1198
+ const stripped = this.removeDiacritics(native);
1199
+ if (stripped === native) return void 0;
1200
+ return this.profileKeywordMap.get(stripped.toLowerCase());
1150
1201
  }
1151
1202
  /**
1152
- * Check if a word is a known keyword (case-insensitive).
1153
- * O(1) lookup using the keyword map.
1203
+ * Check if a word is a known keyword (case-insensitive, diacritic-insensitive).
1204
+ * O(1) lookup using the keyword map. See {@link lookupKeyword}.
1154
1205
  *
1155
1206
  * @param native - Native word to check
1156
1207
  * @returns true if the word is a keyword
1157
1208
  */
1158
1209
  isKeyword(native) {
1159
- return this.profileKeywordMap.has(native.toLowerCase());
1210
+ return this.lookupKeyword(native) !== void 0;
1160
1211
  }
1161
1212
  /**
1162
1213
  * Set the morphological normalizer for this tokenizer.
@@ -1421,6 +1472,14 @@ var _BaseTokenizer = class _BaseTokenizer {
1421
1472
  return null;
1422
1473
  }
1423
1474
  };
1475
+ /**
1476
+ * ASCII word of the shape the English word-walker produces. Excludes `:`, so a
1477
+ * token that already carries a qualifier never merges again — `a:b:c` yields
1478
+ * `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
1479
+ */
1480
+ _BaseTokenizer.ASCII_WORD = /^[A-Za-z_][A-Za-z0-9_]*$/;
1481
+ /** `:name` — only a variable-ref-style extractor ever emits this token shape. */
1482
+ _BaseTokenizer.COLON_QUALIFIER = /^:[A-Za-z_][A-Za-z0-9_]*$/;
1424
1483
  /**
1425
1484
  * Configuration for native language time units.
1426
1485
  * Maps patterns to their standard suffix (ms, s, m, h).
@@ -2944,6 +3003,7 @@ export {
2944
3003
  noChange,
2945
3004
  normalized,
2946
3005
  patternMatcher,
3006
+ stripOptionalDiacritics,
2947
3007
  validateValueType,
2948
3008
  withDefaultExtractors
2949
3009
  };