@lokascript/framework 2.7.2 → 2.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +393 -0
- package/dist/api/create-dsl.d.ts +93 -1
- package/dist/api/create-dsl.d.ts.map +1 -1
- package/dist/api/domain-registry.d.ts +5 -3
- package/dist/api/domain-registry.d.ts.map +1 -1
- package/dist/api/index.js +238 -20
- package/dist/api/index.js.map +1 -1
- package/dist/core/index.js +67 -7
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +39 -3
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/extractors.d.ts +6 -0
- package/dist/core/tokenization/extractors.d.ts.map +1 -1
- package/dist/core/tokenization/index.js +67 -7
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/core/tokenization/token-utils.d.ts +15 -0
- package/dist/core/tokenization/token-utils.d.ts.map +1 -1
- package/dist/generation/index.js +75 -48
- package/dist/generation/index.js.map +1 -1
- package/dist/generation/pattern-generator.d.ts +8 -1
- package/dist/generation/pattern-generator.d.ts.map +1 -1
- package/dist/generation/renderer.d.ts +53 -1
- package/dist/generation/renderer.d.ts.map +1 -1
- package/dist/index.cjs +349 -118
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +345 -118
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts +5 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/multilingual/index.js +66 -7
- package/dist/multilingual/index.js.map +1 -1
- package/dist/testing/index.js +4 -12
- package/dist/testing/index.js.map +1 -1
- package/package.json +4 -3
- package/src/api/create-dsl.test.ts +11 -0
- package/src/api/create-dsl.ts +278 -9
- package/src/api/domain-registry.ts +15 -10
- package/src/api/extensions.test.ts +322 -0
- package/src/core/tokenization/base-tokenizer.ts +78 -9
- package/src/core/tokenization/colon-qualifier.test.ts +129 -0
- package/src/core/tokenization/css-selector-extractor.test.ts +67 -0
- package/src/core/tokenization/extractors.ts +6 -0
- package/src/core/tokenization/token-utils.ts +18 -0
- package/src/generation/domain-renderer.test.ts +172 -0
- package/src/generation/pattern-generator.test.ts +102 -0
- package/src/generation/pattern-generator.ts +32 -20
- package/src/generation/renderer.test.ts +243 -4
- package/src/generation/renderer.ts +188 -45
- package/src/index.ts +9 -1
- package/src/interfaces/value-extractor.ts +50 -0
- package/src/ir/protocol-json.test.ts +21 -0
- package/src/ir/references.test.ts +5 -2
- package/src/prompts/prompt-generator.ts +4 -1
|
@@ -113,6 +113,30 @@ export declare abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
113
113
|
* @returns Token stream
|
|
114
114
|
*/
|
|
115
115
|
protected tokenizeWithExtractors(input: string): TokenStream;
|
|
116
|
+
/**
|
|
117
|
+
* ASCII word of the shape the English word-walker produces. Excludes `:`, so a
|
|
118
|
+
* token that already carries a qualifier never merges again — `a:b:c` yields
|
|
119
|
+
* `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
|
|
120
|
+
*/
|
|
121
|
+
private static readonly ASCII_WORD;
|
|
122
|
+
/** `:name` — only a variable-ref-style extractor ever emits this token shape. */
|
|
123
|
+
private static readonly COLON_QUALIFIER;
|
|
124
|
+
/**
|
|
125
|
+
* Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
|
|
126
|
+
*
|
|
127
|
+
* `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
|
|
128
|
+
* preceded by an identifier is a qualifier (custom event namespace), not a
|
|
129
|
+
* sigil. The English tokenizer already merges these inside
|
|
130
|
+
* EnglishKeywordExtractor; this post-pass gives the other 23 languages the
|
|
131
|
+
* same stream. Strict position adjacency is the discriminator: whitespace
|
|
132
|
+
* between the tokens (`trigger :start`) breaks `end === start`, so a spaced
|
|
133
|
+
* local-variable reference survives untouched.
|
|
134
|
+
*
|
|
135
|
+
* Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
|
|
136
|
+
* sets tokenize `:` as bare punctuation (length 1), which never matches
|
|
137
|
+
* COLON_QUALIFIER, so this pass is a no-op for them.
|
|
138
|
+
*/
|
|
139
|
+
protected mergeColonQualifiedNames(tokens: LanguageToken[]): LanguageToken[];
|
|
116
140
|
/**
|
|
117
141
|
* Classify an unknown character when no extractor matches.
|
|
118
142
|
* Provides sensible defaults for common syntax.
|
|
@@ -205,16 +229,28 @@ export declare abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
205
229
|
*/
|
|
206
230
|
protected isKeywordStartAtBoundary(input: string, pos: number, isWordChar?: (char: string) => boolean): boolean;
|
|
207
231
|
/**
|
|
208
|
-
* Look up a keyword by native word (case-insensitive).
|
|
232
|
+
* Look up a keyword by native word (case-insensitive, diacritic-insensitive).
|
|
209
233
|
* O(1) lookup using the keyword map.
|
|
210
234
|
*
|
|
235
|
+
* The map is INDEXED both with and without diacritics (see
|
|
236
|
+
* `initializeKeywordsFromProfile`), so a stripped QUERY is the other half of
|
|
237
|
+
* that: it lets a surface form carrying harakat the profile does not happen to
|
|
238
|
+
* spell still find its entry. Only consulted after the exact lookup misses, so
|
|
239
|
+
* every previously-matching word resolves byte-identically.
|
|
240
|
+
*
|
|
241
|
+
* Half-implementing this — indexing stripped but querying exact — is what made
|
|
242
|
+
* diacritized `بَدِّل` (toggle) tokenize as `kind=particle normalized=with`:
|
|
243
|
+
* `isKeyword` returned false, so the guard in `ArabicProcliticExtractor` that
|
|
244
|
+
* exists to prevent exactly that handed the word on, and the single-char `ب`
|
|
245
|
+
* bi- proclitic claimed it. A wrong CONCEPT, not a failed parse.
|
|
246
|
+
*
|
|
211
247
|
* @param native - Native word to look up
|
|
212
248
|
* @returns KeywordEntry if found, undefined otherwise
|
|
213
249
|
*/
|
|
214
250
|
protected lookupKeyword(native: string): KeywordEntry | undefined;
|
|
215
251
|
/**
|
|
216
|
-
* Check if a word is a known keyword (case-insensitive).
|
|
217
|
-
* O(1) lookup using the keyword map.
|
|
252
|
+
* Check if a word is a known keyword (case-insensitive, diacritic-insensitive).
|
|
253
|
+
* O(1) lookup using the keyword map. See {@link lookupKeyword}.
|
|
218
254
|
*
|
|
219
255
|
* @param native - Native word to check
|
|
220
256
|
* @returns true if the word is a keyword
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"base-tokenizer.d.ts","sourceRoot":"","sources":["../../../src/core/tokenization/base-tokenizer.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AAEH,OAAO,KAAK,EAAE,aAAa,EAAE,SAAS,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,UAAU,CAAC;AACzF,OAAO,KAAK,EAAE,uBAAuB,EAAE,mBAAmB,EAAE,MAAM,oBAAoB,CAAC;AACvF,OAAO,EACL,KAAK,cAAc,EACnB,KAAK,YAAY,EAGlB,MAAM,kCAAkC,CAAC;AAC1C,OAAO,
|
|
1
|
+
{"version":3,"file":"base-tokenizer.d.ts","sourceRoot":"","sources":["../../../src/core/tokenization/base-tokenizer.ts"],"names":[],"mappings":"AAAA;;;;;;GAMG;AAEH,OAAO,KAAK,EAAE,aAAa,EAAE,SAAS,EAAE,WAAW,EAAE,iBAAiB,EAAE,MAAM,UAAU,CAAC;AACzF,OAAO,KAAK,EAAE,uBAAuB,EAAE,mBAAmB,EAAE,MAAM,oBAAoB,CAAC;AACvF,OAAO,EACL,KAAK,cAAc,EACnB,KAAK,YAAY,EAGlB,MAAM,kCAAkC,CAAC;AAC1C,OAAO,EAQL,KAAK,eAAe,EAErB,MAAM,eAAe,CAAC;AAsEvB,YAAY,EAAE,YAAY,EAAE,CAAC;AAmC7B;;;GAGG;AACH,MAAM,WAAW,gBAAgB;IAC/B,QAAQ,CAAC,QAAQ,CAAC,EAAE,MAAM,CACxB,MAAM,EACN;QAAE,OAAO,EAAE,MAAM,CAAC;QAAC,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;QAAC,UAAU,CAAC,EAAE,MAAM,CAAA;KAAE,CAClE,CAAC;IACF,QAAQ,CAAC,UAAU,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IAC7C,QAAQ,CAAC,WAAW,CAAC,EAAE,MAAM,CAC3B,MAAM,EACN;QAAE,OAAO,EAAE,MAAM,CAAC;QAAC,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;QAAC,QAAQ,CAAC,EAAE,MAAM,CAAA;KAAE,CAChE,CAAC;IACF,QAAQ,CAAC,UAAU,CAAC,EAAE;QACpB,QAAQ,CAAC,MAAM,EAAE,MAAM,CAAC;QACxB,QAAQ,CAAC,cAAc,EAAE,cAAc,GAAG,SAAS,GAAG,iBAAiB,CAAC;QACxE,QAAQ,CAAC,YAAY,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;QAC/C,QAAQ,CAAC,uBAAuB,CAAC,EAAE,OAAO,CAAC;QAC3C,QAAQ,CAAC,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;KAC5C,CAAC;CACH;AAMD;;;GAGG;AACH,8BAAsB,aAAc,YAAW,iBAAiB;IAC9D,QAAQ,CAAC,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IACnC,QAAQ,CAAC,QAAQ,CAAC,SAAS,EAAE,KAAK,GAAG,KAAK,CAAC;IAE3C,0DAA0D;IAC1D,SAAS,CAAC,UAAU,CAAC,EAAE,uBAAuB,CAAC;IAE/C,8EAA8E;IAC9E,SAAS,CAAC,eAAe,EAAE,YAAY,EAAE,CAAM;IAE/C;;;;;;OAMG;IACH,SAAS,CAAC,iBAAiB,EAAE,YAAY,EAAE,CAAM;IAEjD,4DAA4D;IAC5D,SAAS,CAAC,iBAAiB,EAAE,GAAG,CAAC,MAAM,EAAE,YAAY,CAAC,CAAa;IAEnE;;;;;;;OAOG;IACH,OAAO,CAAC,eAAe,CAAsB;IAE7C,kEAAkE;IAClE,sBAAsB,IAAI,SAAS,YAAY,EAAE;IAIjD;;;OAGG;IACH,SAAS,CAAC,UAAU,EAAE,cAAc,EAAE,CAAM;IAE5C;;;;;;;OAOG;IACH,QAAQ,CAAC,KAAK,EAAE,MAAM,GAAG,WAAW;IAYpC,QAAQ,CAAC,aAAa,CAAC,KAAK,EAAE,MAAM,GAAG,SAAS;IAEhD;;;;;;OAMG;IACH,iBAAiB,CAAC,SAAS,EAAE,cAAc,GAAG,IAAI;IAOlD;;;;OAIG;IACH,kBAAkB,CAAC,UAAU,EAAE,cAAc,EAAE,GAAG,IAAI;IAMtD;;;OAGG;IACH,eAAe,IAAI,IAAI;IAIvB;;;OAGG;IACH,SAAS,CAAC,iBAAiB,IAAI,OAAO;IAItC;;;;;;OAMG;IACH,SAAS,CAAC,sBAAsB,CAAC,KAAK,EAAE,MAAM,GAAG,WAAW;IA8E5D;;;;OAIG;IACH,OAAO,CAAC,MAAM,CAAC,QAAQ,CAAC,UAAU,CAA8B;IAEhE,iFAAiF;IACjF,OAAO,CAAC,MAAM,CAAC,QAAQ,CAAC,eAAe,CAA+B;IAEtE;;;;;;;;;;;;;;OAcG;IACH,SAAS,CAAC,wBAAwB,CAAC,MAAM,EAAE,aAAa,EAAE,GAAG,aAAa,EAAE;IA6B5E;;;;;;OAMG;IACH,SAAS,CAAC,mBAAmB,CAAC,IAAI,EAAE,MAAM,GAAG,SAAS;IAMtD;;;;;;;OAOG;IACH,SAAS,CAAC,iBAAiB,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,EAAE,MAAM,EAAE,aAAa,EAAE,GAAG,OAAO;IAgCzF;;;;;;;;;;;;OAYG;IACH,SAAS,CAAC,6BAA6B,CACrC,OAAO,EAAE,gBAAgB,EACzB,MAAM,GAAE,YAAY,EAAO,GAC1B,IAAI;IAoHP;;;;;;;OAOG;IACH,SAAS,CAAC,gBAAgB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM;IAIhD;;;;;;;OAOG;IACH,SAAS,CAAC,iBAAiB,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAc7E;;;;;;;;;;;;OAYG;IACH,SAAS,CAAC,mBAAmB,CAC3B,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,UAAU,GAAE,CAAC,IAAI,EAAE,MAAM,KAAK,OAAyC,GACtE,aAAa,GAAG,IAAI;IAiBvB;;;;;;;OAOG;IACH,SAAS,CAAC,cAAc,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,OAAO;IAK7D;;;;;;;;;;;;;;;;;OAiBG;IACH,SAAS,CAAC,wBAAwB,CAChC,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,UAAU,GAAE,CAAC,IAAI,EAAE,MAAM,KAAK,OAAyC,GACtE,OAAO;IASV;;;;;;;;;;;;;;;;;;OAkBG;IACH,SAAS,CAAC,aAAa,CAAC,MAAM,EAAE,MAAM,GAAG,YAAY,GAAG,SAAS;IAQjE;;;;;;OAMG;IACH,SAAS,CAAC,SAAS,CAAC,MAAM,EAAE,MAAM,GAAG,OAAO;IAI5C;;OAEG;IACH,aAAa,CAAC,UAAU,EAAE,uBAAuB,GAAG,IAAI;IAIxD;;;;;;;OAOG;IACH,SAAS,CAAC,YAAY,CAAC,IAAI,EAAE,MAAM,GAAG,mBAAmB,GAAG,IAAI;IAahE;;;;;;;;;;;;;;;OAeG;IACH,SAAS,CAAC,oBAAoB,CAC5B,IAAI,EAAE,MAAM,EACZ,QAAQ,EAAE,MAAM,EAChB,MAAM,EAAE,MAAM,GACb,aAAa,GAAG,IAAI;IAgBvB;;OAEG;IACH,SAAS,CAAC,WAAW,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQvE;;;OAGG;IACH,SAAS,CAAC,gBAAgB,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAmC5E;;OAEG;IACH,SAAS,CAAC,SAAS,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQrE;;OAEG;IACH,SAAS,CAAC,SAAS,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQrE;;;OAGG;IACH,SAAS,CAAC,MAAM,CAAC,QAAQ,CAAC,mBAAmB,EAAE,SAAS,eAAe,EAAE,CAKvE;IAEF;;;;;;;;OAQG;IACH,SAAS,CAAC,gBAAgB,CACxB,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,SAAS,EAAE,SAAS,eAAe,EAAE,EACrC,cAAc,UAAQ,GACrB;QAAE,MAAM,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,CAAA;KAAE,GAAG,IAAI;IAuC5C;;;;;;;;OAQG;IACH,SAAS,CAAC,eAAe,CACvB,KAAK,EAAE,MAAM,EACb,QAAQ,EAAE,MAAM,EAChB,SAAS,UAAO,GACf;QAAE,MAAM,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,CAAA;KAAE,GAAG,IAAI;IAgC5C;;;;;;;;;;;;;OAaG;IACH,SAAS,CAAC,sBAAsB,CAC9B,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,eAAe,EAAE,SAAS,eAAe,EAAE,EAC3C,OAAO,GAAE;QAAE,SAAS,CAAC,EAAE,OAAO,CAAC;QAAC,cAAc,CAAC,EAAE,OAAO,CAAA;KAAO,GAC9D,aAAa,GAAG,IAAI;IAqBvB;;;OAGG;IACH,SAAS,CAAC,MAAM,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAQlE;;;OAGG;IACH,SAAS,CAAC,cAAc,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAc1E;;;OAGG;IACH,SAAS,CAAC,WAAW,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,aAAa,GAAG,IAAI;IAqBvE;;;;;;;;;;OAUG;IACH,SAAS,CAAC,oBAAoB,CAC5B,KAAK,EAAE,MAAM,EACb,GAAG,EAAE,MAAM,EACX,SAAS,EAAE,SAAS,MAAM,EAAE,GAC3B,aAAa,GAAG,IAAI;CAQxB;AAMD;;;;;;;;;;;;;GAaG;AACH,MAAM,WAAW,qBAAqB;IACpC,8BAA8B;IAC9B,QAAQ,EAAE,MAAM,CAAC;IACjB,sCAAsC;IACtC,SAAS,CAAC,EAAE,KAAK,GAAG,KAAK,CAAC;IAC1B,uEAAuE;IACvE,QAAQ,EAAE,MAAM,EAAE,CAAC;IACnB,wDAAwD;IACxD,aAAa,CAAC,EAAE,YAAY,EAAE,CAAC;IAC/B,wEAAwE;IACxE,cAAc,CAAC,EAAE,gBAAgB,CAAC;IAClC,uGAAuG;IACvG,gBAAgB,CAAC,EAAE,OAAO,CAAC;IAC3B,wDAAwD;IACxD,eAAe,CAAC,EAAE,OAAO,CAAC;IAC1B,6DAA6D;IAC7D,gBAAgB,CAAC,EAAE,cAAc,EAAE,CAAC;CACrC;AAED;;;;;;;;;;;;;;;GAeG;AACH,wBAAgB,qBAAqB,CAAC,MAAM,EAAE,qBAAqB,GAAG,iBAAiB,CA2CtF"}
|
|
@@ -20,6 +20,12 @@
|
|
|
20
20
|
* Method call handling:
|
|
21
21
|
* - #dialog.showModal() → stops after #dialog (method call, not compound selector)
|
|
22
22
|
* - #box.active → compound selector (no parens)
|
|
23
|
+
*
|
|
24
|
+
* NOTE: intentionally diverges from the semantic package's copy
|
|
25
|
+
* (packages/semantic/src/tokenizers/extractors/css-selector.ts), which also
|
|
26
|
+
* consumes pseudo-class/pseudo-element segments (#x:hover, .a:not(.b)). This
|
|
27
|
+
* legacy version is only used by BaseTokenizer.trySelector (no semantic call
|
|
28
|
+
* sites) and stays as-is.
|
|
23
29
|
*/
|
|
24
30
|
export declare function extractCssSelector(input: string, startPos: number): string | null;
|
|
25
31
|
/**
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"extractors.d.ts","sourceRoot":"","sources":["../../../src/core/tokenization/extractors.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAeH
|
|
1
|
+
{"version":3,"file":"extractors.d.ts","sourceRoot":"","sources":["../../../src/core/tokenization/extractors.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAeH;;;;;;;;;;;;;;;;;;;;;;GAsBG;AACH,wBAAgB,kBAAkB,CAAC,KAAK,EAAE,MAAM,EAAE,QAAQ,EAAE,MAAM,GAAG,MAAM,GAAG,IAAI,CA+LjF;AAMD;;;;;;;;GAQG;AACH,wBAAgB,kBAAkB,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,OAAO,CAYtE;AAED;;;;;GAKG;AACH,wBAAgB,oBAAoB,CAAC,KAAK,EAAE,MAAM,EAAE,QAAQ,EAAE,MAAM,GAAG,MAAM,GAAG,IAAI,CA2CnF;AAMD;;;GAGG;AACH,wBAAgB,UAAU,CAAC,KAAK,EAAE,MAAM,EAAE,GAAG,EAAE,MAAM,GAAG,OAAO,CA8B9D;AAED;;;;;;;GAOG;AACH,wBAAgB,UAAU,CAAC,KAAK,EAAE,MAAM,EAAE,QAAQ,EAAE,MAAM,GAAG,MAAM,GAAG,IAAI,CA0CzE;AAMD;;;GAGG;AACH,wBAAgB,aAAa,CAAC,KAAK,EAAE,MAAM,EAAE,QAAQ,EAAE,MAAM,GAAG,MAAM,GAAG,IAAI,CA2C5E"}
|
|
@@ -104,6 +104,9 @@ function isQuote(char) {
|
|
|
104
104
|
function isDigit(char) {
|
|
105
105
|
return /\d/.test(char);
|
|
106
106
|
}
|
|
107
|
+
function stripOptionalDiacritics(word) {
|
|
108
|
+
return word.replace(/[ً-ْٰ]/g, "");
|
|
109
|
+
}
|
|
107
110
|
function isAsciiLetter(char) {
|
|
108
111
|
return /[a-zA-Z]/.test(char);
|
|
109
112
|
}
|
|
@@ -894,7 +897,39 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
894
897
|
pos++;
|
|
895
898
|
}
|
|
896
899
|
}
|
|
897
|
-
return new TokenStreamImpl(tokens, this.language);
|
|
900
|
+
return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
|
|
901
|
+
}
|
|
902
|
+
/**
|
|
903
|
+
* Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
|
|
904
|
+
*
|
|
905
|
+
* `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
|
|
906
|
+
* preceded by an identifier is a qualifier (custom event namespace), not a
|
|
907
|
+
* sigil. The English tokenizer already merges these inside
|
|
908
|
+
* EnglishKeywordExtractor; this post-pass gives the other 23 languages the
|
|
909
|
+
* same stream. Strict position adjacency is the discriminator: whitespace
|
|
910
|
+
* between the tokens (`trigger :start`) breaks `end === start`, so a spaced
|
|
911
|
+
* local-variable reference survives untouched.
|
|
912
|
+
*
|
|
913
|
+
* Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
|
|
914
|
+
* sets tokenize `:` as bare punctuation (length 1), which never matches
|
|
915
|
+
* COLON_QUALIFIER, so this pass is a no-op for them.
|
|
916
|
+
*/
|
|
917
|
+
mergeColonQualifiedNames(tokens) {
|
|
918
|
+
const out = [];
|
|
919
|
+
for (const tok of tokens) {
|
|
920
|
+
const prev = out[out.length - 1];
|
|
921
|
+
if (prev && _BaseTokenizer.ASCII_WORD.test(prev.value) && _BaseTokenizer.COLON_QUALIFIER.test(tok.value) && prev.position.end === tok.position.start) {
|
|
922
|
+
const merged = prev.value + tok.value;
|
|
923
|
+
out[out.length - 1] = createToken(
|
|
924
|
+
merged,
|
|
925
|
+
this.classifyToken(merged),
|
|
926
|
+
createPosition(prev.position.start, tok.position.end)
|
|
927
|
+
);
|
|
928
|
+
continue;
|
|
929
|
+
}
|
|
930
|
+
out.push(tok);
|
|
931
|
+
}
|
|
932
|
+
return out;
|
|
898
933
|
}
|
|
899
934
|
/**
|
|
900
935
|
* Classify an unknown character when no extractor matches.
|
|
@@ -1027,7 +1062,7 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
1027
1062
|
* @returns Word without diacritics
|
|
1028
1063
|
*/
|
|
1029
1064
|
removeDiacritics(word) {
|
|
1030
|
-
return word
|
|
1065
|
+
return stripOptionalDiacritics(word);
|
|
1031
1066
|
}
|
|
1032
1067
|
/**
|
|
1033
1068
|
* Try to match a keyword from profile at the current position.
|
|
@@ -1118,24 +1153,40 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
1118
1153
|
});
|
|
1119
1154
|
}
|
|
1120
1155
|
/**
|
|
1121
|
-
* Look up a keyword by native word (case-insensitive).
|
|
1156
|
+
* Look up a keyword by native word (case-insensitive, diacritic-insensitive).
|
|
1122
1157
|
* O(1) lookup using the keyword map.
|
|
1123
1158
|
*
|
|
1159
|
+
* The map is INDEXED both with and without diacritics (see
|
|
1160
|
+
* `initializeKeywordsFromProfile`), so a stripped QUERY is the other half of
|
|
1161
|
+
* that: it lets a surface form carrying harakat the profile does not happen to
|
|
1162
|
+
* spell still find its entry. Only consulted after the exact lookup misses, so
|
|
1163
|
+
* every previously-matching word resolves byte-identically.
|
|
1164
|
+
*
|
|
1165
|
+
* Half-implementing this — indexing stripped but querying exact — is what made
|
|
1166
|
+
* diacritized `بَدِّل` (toggle) tokenize as `kind=particle normalized=with`:
|
|
1167
|
+
* `isKeyword` returned false, so the guard in `ArabicProcliticExtractor` that
|
|
1168
|
+
* exists to prevent exactly that handed the word on, and the single-char `ب`
|
|
1169
|
+
* bi- proclitic claimed it. A wrong CONCEPT, not a failed parse.
|
|
1170
|
+
*
|
|
1124
1171
|
* @param native - Native word to look up
|
|
1125
1172
|
* @returns KeywordEntry if found, undefined otherwise
|
|
1126
1173
|
*/
|
|
1127
1174
|
lookupKeyword(native) {
|
|
1128
|
-
|
|
1175
|
+
const exact = this.profileKeywordMap.get(native.toLowerCase());
|
|
1176
|
+
if (exact) return exact;
|
|
1177
|
+
const stripped = this.removeDiacritics(native);
|
|
1178
|
+
if (stripped === native) return void 0;
|
|
1179
|
+
return this.profileKeywordMap.get(stripped.toLowerCase());
|
|
1129
1180
|
}
|
|
1130
1181
|
/**
|
|
1131
|
-
* Check if a word is a known keyword (case-insensitive).
|
|
1132
|
-
* O(1) lookup using the keyword map.
|
|
1182
|
+
* Check if a word is a known keyword (case-insensitive, diacritic-insensitive).
|
|
1183
|
+
* O(1) lookup using the keyword map. See {@link lookupKeyword}.
|
|
1133
1184
|
*
|
|
1134
1185
|
* @param native - Native word to check
|
|
1135
1186
|
* @returns true if the word is a keyword
|
|
1136
1187
|
*/
|
|
1137
1188
|
isKeyword(native) {
|
|
1138
|
-
return this.
|
|
1189
|
+
return this.lookupKeyword(native) !== void 0;
|
|
1139
1190
|
}
|
|
1140
1191
|
/**
|
|
1141
1192
|
* Set the morphological normalizer for this tokenizer.
|
|
@@ -1400,6 +1451,14 @@ var _BaseTokenizer = class _BaseTokenizer {
|
|
|
1400
1451
|
return null;
|
|
1401
1452
|
}
|
|
1402
1453
|
};
|
|
1454
|
+
/**
|
|
1455
|
+
* ASCII word of the shape the English word-walker produces. Excludes `:`, so a
|
|
1456
|
+
* token that already carries a qualifier never merges again — `a:b:c` yields
|
|
1457
|
+
* `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
|
|
1458
|
+
*/
|
|
1459
|
+
_BaseTokenizer.ASCII_WORD = /^[A-Za-z_][A-Za-z0-9_]*$/;
|
|
1460
|
+
/** `:name` — only a variable-ref-style extractor ever emits this token shape. */
|
|
1461
|
+
_BaseTokenizer.COLON_QUALIFIER = /^:[A-Za-z_][A-Za-z0-9_]*$/;
|
|
1403
1462
|
/**
|
|
1404
1463
|
* Configuration for native language time units.
|
|
1405
1464
|
* Maps patterns to their standard suffix (ms, s, m, h).
|
|
@@ -1620,6 +1679,7 @@ export {
|
|
|
1620
1679
|
isWhitespace,
|
|
1621
1680
|
noChange,
|
|
1622
1681
|
normalized,
|
|
1682
|
+
stripOptionalDiacritics,
|
|
1623
1683
|
withDefaultExtractors
|
|
1624
1684
|
};
|
|
1625
1685
|
//# sourceMappingURL=index.js.map
|