@lokascript/framework 2.7.2 → 2.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +393 -0
- package/dist/api/create-dsl.d.ts +93 -1
- package/dist/api/create-dsl.d.ts.map +1 -1
- package/dist/api/domain-registry.d.ts +5 -3
- package/dist/api/domain-registry.d.ts.map +1 -1
- package/dist/api/index.js +238 -20
- package/dist/api/index.js.map +1 -1
- package/dist/core/index.js +67 -7
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +39 -3
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/extractors.d.ts +6 -0
- package/dist/core/tokenization/extractors.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +67 -7
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/core/tokenization/token-utils.d.ts +15 -0
- package/dist/core/tokenization/token-utils.d.ts.map +1 -1
- package/dist/generation/index.js +75 -48
- package/dist/generation/index.js.map +1 -1
- package/dist/generation/pattern-generator.d.ts +8 -1
- package/dist/generation/pattern-generator.d.ts.map +1 -1
- package/dist/generation/renderer.d.ts +53 -1
- package/dist/generation/renderer.d.ts.map +1 -1
- package/dist/index.cjs +349 -118
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +345 -118
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts +5 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/multilingual/index.js +66 -7
- package/dist/multilingual/index.js.map +1 -1
- package/dist/testing/index.js +4 -12
- package/dist/testing/index.js.map +1 -1
- package/package.json +4 -3
- package/src/api/create-dsl.test.ts +11 -0
- package/src/api/create-dsl.ts +278 -9
- package/src/api/domain-registry.ts +15 -10
- package/src/api/extensions.test.ts +322 -0
- package/src/core/tokenization/base-tokenizer.ts +78 -9
- package/src/core/tokenization/colon-qualifier.test.ts +129 -0
- package/src/core/tokenization/css-selector-extractor.test.ts +67 -0
- package/src/core/tokenization/extractors.ts +6 -0
- package/src/core/tokenization/token-utils.ts +18 -0
- package/src/generation/domain-renderer.test.ts +172 -0
- package/src/generation/pattern-generator.test.ts +102 -0
- package/src/generation/pattern-generator.ts +32 -20
- package/src/generation/renderer.test.ts +243 -4
- package/src/generation/renderer.ts +188 -45
- package/src/index.ts +9 -1
- package/src/interfaces/value-extractor.ts +50 -0
- package/src/ir/protocol-json.test.ts +21 -0
- package/src/ir/references.test.ts +5 -2
- package/src/prompts/prompt-generator.ts +4 -1
package/dist/core/index.js
CHANGED
|
@@ -125,6 +125,9 @@ function isQuote(char) {
|
|
|
125
125
|
function isDigit(char) {
|
|
126
126
|
return /\d/.test(char);
|
|
127
127
|
}
|
|
128
|
+
function stripOptionalDiacritics(word) {
|
|
129
|
+
return word.replace(/[ً-ْٰ]/g, "");
|
|
130
|
+
}
|
|
128
131
|
function isAsciiLetter(char) {
|
|
129
132
|
return /[a-zA-Z]/.test(char);
|
|
130
133
|
}
|
|
@@ -915,7 +918,39 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
915
918
|
pos++;
|
|
916
919
|
}
|
|
917
920
|
}
|
|
918
|
-
return new TokenStreamImpl(tokens, this.language);
|
|
921
|
+
return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
|
|
922
|
+
}
|
|
923
|
+
/**
|
|
924
|
+
* Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
|
|
925
|
+
*
|
|
926
|
+
* `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
|
|
927
|
+
* preceded by an identifier is a qualifier (custom event namespace), not a
|
|
928
|
+
* sigil. The English tokenizer already merges these inside
|
|
929
|
+
* EnglishKeywordExtractor; this post-pass gives the other 23 languages the
|
|
930
|
+
* same stream. Strict position adjacency is the discriminator: whitespace
|
|
931
|
+
* between the tokens (`trigger :start`) breaks `end === start`, so a spaced
|
|
932
|
+
* local-variable reference survives untouched.
|
|
933
|
+
*
|
|
934
|
+
* Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
|
|
935
|
+
* sets tokenize `:` as bare punctuation (length 1), which never matches
|
|
936
|
+
* COLON_QUALIFIER, so this pass is a no-op for them.
|
|
937
|
+
*/
|
|
938
|
+
mergeColonQualifiedNames(tokens) {
|
|
939
|
+
const out = [];
|
|
940
|
+
for (const tok of tokens) {
|
|
941
|
+
const prev = out[out.length - 1];
|
|
942
|
+
if (prev && _BaseTokenizer.ASCII_WORD.test(prev.value) && _BaseTokenizer.COLON_QUALIFIER.test(tok.value) && prev.position.end === tok.position.start) {
|
|
943
|
+
const merged = prev.value + tok.value;
|
|
944
|
+
out[out.length - 1] = createToken(
|
|
945
|
+
merged,
|
|
946
|
+
this.classifyToken(merged),
|
|
947
|
+
createPosition(prev.position.start, tok.position.end)
|
|
948
|
+
);
|
|
949
|
+
continue;
|
|
950
|
+
}
|
|
951
|
+
out.push(tok);
|
|
952
|
+
}
|
|
953
|
+
return out;
|
|
919
954
|
}
|
|
920
955
|
/**
|
|
921
956
|
* Classify an unknown character when no extractor matches.
|
|
@@ -1048,7 +1083,7 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
1048
1083
|
* @returns Word without diacritics
|
|
1049
1084
|
*/
|
|
1050
1085
|
removeDiacritics(word) {
|
|
1051
|
-
return word
|
|
1086
|
+
return stripOptionalDiacritics(word);
|
|
1052
1087
|
}
|
|
1053
1088
|
/**
|
|
1054
1089
|
* Try to match a keyword from profile at the current position.
|
|
@@ -1139,24 +1174,40 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
1139
1174
|
});
|
|
1140
1175
|
}
|
|
1141
1176
|
/**
|
|
1142
|
-
* Look up a keyword by native word (case-insensitive).
|
|
1177
|
+
* Look up a keyword by native word (case-insensitive, diacritic-insensitive).
|
|
1143
1178
|
* O(1) lookup using the keyword map.
|
|
1144
1179
|
*
|
|
1180
|
+
* The map is INDEXED both with and without diacritics (see
|
|
1181
|
+
* `initializeKeywordsFromProfile`), so a stripped QUERY is the other half of
|
|
1182
|
+
* that: it lets a surface form carrying harakat the profile does not happen to
|
|
1183
|
+
* spell still find its entry. Only consulted after the exact lookup misses, so
|
|
1184
|
+
* every previously-matching word resolves byte-identically.
|
|
1185
|
+
*
|
|
1186
|
+
* Half-implementing this — indexing stripped but querying exact — is what made
|
|
1187
|
+
* diacritized `بَدِّل` (toggle) tokenize as `kind=particle normalized=with`:
|
|
1188
|
+
* `isKeyword` returned false, so the guard in `ArabicProcliticExtractor` that
|
|
1189
|
+
* exists to prevent exactly that handed the word on, and the single-char `ب`
|
|
1190
|
+
* bi- proclitic claimed it. A wrong CONCEPT, not a failed parse.
|
|
1191
|
+
*
|
|
1145
1192
|
* @param native - Native word to look up
|
|
1146
1193
|
* @returns KeywordEntry if found, undefined otherwise
|
|
1147
1194
|
*/
|
|
1148
1195
|
lookupKeyword(native) {
|
|
1149
|
-
|
|
1196
|
+
const exact = this.profileKeywordMap.get(native.toLowerCase());
|
|
1197
|
+
if (exact) return exact;
|
|
1198
|
+
const stripped = this.removeDiacritics(native);
|
|
1199
|
+
if (stripped === native) return void 0;
|
|
1200
|
+
return this.profileKeywordMap.get(stripped.toLowerCase());
|
|
1150
1201
|
}
|
|
1151
1202
|
/**
|
|
1152
|
-
* Check if a word is a known keyword (case-insensitive).
|
|
1153
|
-
* O(1) lookup using the keyword map.
|
|
1203
|
+
* Check if a word is a known keyword (case-insensitive, diacritic-insensitive).
|
|
1204
|
+
* O(1) lookup using the keyword map. See {@link lookupKeyword}.
|
|
1154
1205
|
*
|
|
1155
1206
|
* @param native - Native word to check
|
|
1156
1207
|
* @returns true if the word is a keyword
|
|
1157
1208
|
*/
|
|
1158
1209
|
isKeyword(native) {
|
|
1159
|
-
return this.
|
|
1210
|
+
return this.lookupKeyword(native) !== void 0;
|
|
1160
1211
|
}
|
|
1161
1212
|
/**
|
|
1162
1213
|
* Set the morphological normalizer for this tokenizer.
|
|
@@ -1421,6 +1472,14 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
1421
1472
|
return null;
|
|
1422
1473
|
}
|
|
1423
1474
|
};
|
|
1475
|
+
/**
|
|
1476
|
+
* ASCII word of the shape the English word-walker produces. Excludes `:`, so a
|
|
1477
|
+
* token that already carries a qualifier never merges again — `a:b:c` yields
|
|
1478
|
+
* `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
|
|
1479
|
+
*/
|
|
1480
|
+
_BaseTokenizer.ASCII_WORD = /^[A-Za-z_][A-Za-z0-9_]*$/;
|
|
1481
|
+
/** `:name` — only a variable-ref-style extractor ever emits this token shape. */
|
|
1482
|
+
_BaseTokenizer.COLON_QUALIFIER = /^:[A-Za-z_][A-Za-z0-9_]*$/;
|
|
1424
1483
|
/**
|
|
1425
1484
|
* Configuration for native language time units.
|
|
1426
1485
|
* Maps patterns to their standard suffix (ms, s, m, h).
|
|
@@ -2944,6 +3003,7 @@ export {
|
|
|
2944
3003
|
noChange,
|
|
2945
3004
|
normalized,
|
|
2946
3005
|
patternMatcher,
|
|
3006
|
+
stripOptionalDiacritics,
|
|
2947
3007
|
validateValueType,
|
|
2948
3008
|
withDefaultExtractors
|
|
2949
3009
|
};
|