@lokascript/framework 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (181) hide show
  1. package/LICENSE +20 -0
  2. package/README.md +142 -0
  3. package/dist/aot/aot-orchestrator.d.ts +75 -0
  4. package/dist/aot/aot-orchestrator.d.ts.map +1 -0
  5. package/dist/aot/domain-scanner.d.ts +27 -0
  6. package/dist/aot/domain-scanner.d.ts.map +1 -0
  7. package/dist/aot/index.d.ts +8 -0
  8. package/dist/aot/index.d.ts.map +1 -0
  9. package/dist/aot/types.d.ts +103 -0
  10. package/dist/aot/types.d.ts.map +1 -0
  11. package/dist/api/create-dsl.d.ts +91 -0
  12. package/dist/api/create-dsl.d.ts.map +1 -0
  13. package/dist/api/dispatcher.d.ts +108 -0
  14. package/dist/api/dispatcher.d.ts.map +1 -0
  15. package/dist/api/domain-registry.d.ts +152 -0
  16. package/dist/api/domain-registry.d.ts.map +1 -0
  17. package/dist/api/index.d.ts +7 -0
  18. package/dist/api/index.d.ts.map +1 -0
  19. package/dist/api/index.js +2082 -0
  20. package/dist/api/index.js.map +1 -0
  21. package/dist/core/index.d.ts +7 -0
  22. package/dist/core/index.d.ts.map +1 -0
  23. package/dist/core/index.js +2674 -0
  24. package/dist/core/index.js.map +1 -0
  25. package/dist/core/logger.d.ts +32 -0
  26. package/dist/core/logger.d.ts.map +1 -0
  27. package/dist/core/pattern-matching/index.d.ts +6 -0
  28. package/dist/core/pattern-matching/index.d.ts.map +1 -0
  29. package/dist/core/pattern-matching/index.js +1239 -0
  30. package/dist/core/pattern-matching/index.js.map +1 -0
  31. package/dist/core/pattern-matching/pattern-matcher.d.ts +239 -0
  32. package/dist/core/pattern-matching/pattern-matcher.d.ts.map +1 -0
  33. package/dist/core/pattern-matching/utils/index.d.ts +6 -0
  34. package/dist/core/pattern-matching/utils/index.d.ts.map +1 -0
  35. package/dist/core/pattern-matching/utils/possessive-keywords.d.ts +38 -0
  36. package/dist/core/pattern-matching/utils/possessive-keywords.d.ts.map +1 -0
  37. package/dist/core/pattern-matching/utils/type-validation.d.ts +63 -0
  38. package/dist/core/pattern-matching/utils/type-validation.d.ts.map +1 -0
  39. package/dist/core/tokenization/base-tokenizer.d.ts +344 -0
  40. package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -0
  41. package/dist/core/tokenization/char-classifiers.d.ts +56 -0
  42. package/dist/core/tokenization/char-classifiers.d.ts.map +1 -0
  43. package/dist/core/tokenization/default-extractors.d.ts +48 -0
  44. package/dist/core/tokenization/default-extractors.d.ts.map +1 -0
  45. package/dist/core/tokenization/extractors/index.d.ts +9 -0
  46. package/dist/core/tokenization/extractors/index.d.ts.map +1 -0
  47. package/dist/core/tokenization/extractors/operator.d.ts +23 -0
  48. package/dist/core/tokenization/extractors/operator.d.ts.map +1 -0
  49. package/dist/core/tokenization/extractors/punctuation.d.ts +22 -0
  50. package/dist/core/tokenization/extractors/punctuation.d.ts.map +1 -0
  51. package/dist/core/tokenization/extractors.d.ts +61 -0
  52. package/dist/core/tokenization/extractors.d.ts.map +1 -0
  53. package/dist/core/tokenization/index.d.ts +11 -0
  54. package/dist/core/tokenization/index.d.ts.map +1 -0
  55. package/dist/core/tokenization/index.js +1345 -0
  56. package/dist/core/tokenization/index.js.map +1 -0
  57. package/dist/core/tokenization/morphology/index.d.ts +5 -0
  58. package/dist/core/tokenization/morphology/index.d.ts.map +1 -0
  59. package/dist/core/tokenization/morphology/types.d.ts +110 -0
  60. package/dist/core/tokenization/morphology/types.d.ts.map +1 -0
  61. package/dist/core/tokenization/token-utils.d.ts +111 -0
  62. package/dist/core/tokenization/token-utils.d.ts.map +1 -0
  63. package/dist/core/types.d.ts +382 -0
  64. package/dist/core/types.d.ts.map +1 -0
  65. package/dist/core/types.js +108 -0
  66. package/dist/core/types.js.map +1 -0
  67. package/dist/generation/diagnostics.d.ts +120 -0
  68. package/dist/generation/diagnostics.d.ts.map +1 -0
  69. package/dist/generation/index.d.ts +7 -0
  70. package/dist/generation/index.d.ts.map +1 -0
  71. package/dist/generation/index.js +339 -0
  72. package/dist/generation/index.js.map +1 -0
  73. package/dist/generation/pattern-generator.d.ts +48 -0
  74. package/dist/generation/pattern-generator.d.ts.map +1 -0
  75. package/dist/generation/renderer.d.ts +115 -0
  76. package/dist/generation/renderer.d.ts.map +1 -0
  77. package/dist/grammar/index.d.ts +10 -0
  78. package/dist/grammar/index.d.ts.map +1 -0
  79. package/dist/grammar/index.js +391 -0
  80. package/dist/grammar/index.js.map +1 -0
  81. package/dist/grammar/transformer.d.ts +56 -0
  82. package/dist/grammar/transformer.d.ts.map +1 -0
  83. package/dist/grammar/types.d.ts +236 -0
  84. package/dist/grammar/types.d.ts.map +1 -0
  85. package/dist/index.cjs +4454 -0
  86. package/dist/index.cjs.map +1 -0
  87. package/dist/index.d.ts +46 -0
  88. package/dist/index.d.ts.map +1 -0
  89. package/dist/index.js +4336 -0
  90. package/dist/index.js.map +1 -0
  91. package/dist/interfaces/dictionary.d.ts +82 -0
  92. package/dist/interfaces/dictionary.d.ts.map +1 -0
  93. package/dist/interfaces/index.d.ts +10 -0
  94. package/dist/interfaces/index.d.ts.map +1 -0
  95. package/dist/interfaces/profile-provider.d.ts +67 -0
  96. package/dist/interfaces/profile-provider.d.ts.map +1 -0
  97. package/dist/interfaces/value-extractor.d.ts +168 -0
  98. package/dist/interfaces/value-extractor.d.ts.map +1 -0
  99. package/dist/multilingual/index.d.ts +8 -0
  100. package/dist/multilingual/index.d.ts.map +1 -0
  101. package/dist/multilingual/index.js +1 -0
  102. package/dist/multilingual/index.js.map +1 -0
  103. package/dist/parsing/index.d.ts +8 -0
  104. package/dist/parsing/index.d.ts.map +1 -0
  105. package/dist/parsing/index.js +1415 -0
  106. package/dist/parsing/index.js.map +1 -0
  107. package/dist/parsing/multi-statement.d.ts +265 -0
  108. package/dist/parsing/multi-statement.d.ts.map +1 -0
  109. package/dist/schema/command-schema.d.ts +78 -0
  110. package/dist/schema/command-schema.d.ts.map +1 -0
  111. package/dist/schema/index.d.ts +5 -0
  112. package/dist/schema/index.d.ts.map +1 -0
  113. package/dist/schema/index.js +25 -0
  114. package/dist/schema/index.js.map +1 -0
  115. package/dist/test-setup.d.ts +9 -0
  116. package/dist/test-setup.d.ts.map +1 -0
  117. package/dist/testing/index.d.ts +50 -0
  118. package/dist/testing/index.d.ts.map +1 -0
  119. package/dist/testing/index.js +16969 -0
  120. package/dist/testing/index.js.map +1 -0
  121. package/package.json +122 -0
  122. package/src/__test__/fixtures/sql-dsl.ts +232 -0
  123. package/src/__test__/sql-integration.test.ts +189 -0
  124. package/src/__test__/test-utils.ts +260 -0
  125. package/src/aot/aot-orchestrator.test.ts +413 -0
  126. package/src/aot/aot-orchestrator.ts +238 -0
  127. package/src/aot/domain-scanner.ts +178 -0
  128. package/src/aot/index.ts +8 -0
  129. package/src/aot/types.ts +124 -0
  130. package/src/api/create-dsl.ts +367 -0
  131. package/src/api/dispatcher.test.ts +336 -0
  132. package/src/api/dispatcher.ts +222 -0
  133. package/src/api/domain-registry.test.ts +336 -0
  134. package/src/api/domain-registry.ts +500 -0
  135. package/src/api/index.ts +7 -0
  136. package/src/core/index.ts +7 -0
  137. package/src/core/logger.ts +130 -0
  138. package/src/core/pattern-matching/index.ts +6 -0
  139. package/src/core/pattern-matching/pattern-matcher.test.ts +900 -0
  140. package/src/core/pattern-matching/pattern-matcher.ts +1548 -0
  141. package/src/core/pattern-matching/pattern-matcher.ts.backup +1267 -0
  142. package/src/core/pattern-matching/utils/index.ts +6 -0
  143. package/src/core/pattern-matching/utils/possessive-keywords.ts +55 -0
  144. package/src/core/pattern-matching/utils/type-validation.test.ts +316 -0
  145. package/src/core/pattern-matching/utils/type-validation.ts +134 -0
  146. package/src/core/tokenization/base-tokenizer.ts +916 -0
  147. package/src/core/tokenization/char-classifiers.ts +79 -0
  148. package/src/core/tokenization/create-simple-tokenizer.test.ts +260 -0
  149. package/src/core/tokenization/default-extractors.ts +69 -0
  150. package/src/core/tokenization/extractors/index.ts +9 -0
  151. package/src/core/tokenization/extractors/operator.ts +75 -0
  152. package/src/core/tokenization/extractors/punctuation.ts +39 -0
  153. package/src/core/tokenization/extractors.ts +452 -0
  154. package/src/core/tokenization/index.ts +11 -0
  155. package/src/core/tokenization/morphology/index.ts +5 -0
  156. package/src/core/tokenization/morphology/types.ts +211 -0
  157. package/src/core/tokenization/token-utils.ts +252 -0
  158. package/src/core/types.ts +589 -0
  159. package/src/generation/diagnostics.test.ts +171 -0
  160. package/src/generation/diagnostics.ts +239 -0
  161. package/src/generation/index.ts +7 -0
  162. package/src/generation/pattern-generator.test.ts +430 -0
  163. package/src/generation/pattern-generator.ts +315 -0
  164. package/src/generation/renderer.test.ts +266 -0
  165. package/src/generation/renderer.ts +244 -0
  166. package/src/grammar/index.ts +12 -0
  167. package/src/grammar/transformer.ts +159 -0
  168. package/src/grammar/types.ts +630 -0
  169. package/src/index.ts +157 -0
  170. package/src/interfaces/dictionary.ts +123 -0
  171. package/src/interfaces/index.ts +10 -0
  172. package/src/interfaces/profile-provider.ts +88 -0
  173. package/src/interfaces/value-extractor.ts +435 -0
  174. package/src/multilingual/index.ts +9 -0
  175. package/src/parsing/index.ts +27 -0
  176. package/src/parsing/multi-statement.test.ts +480 -0
  177. package/src/parsing/multi-statement.ts +648 -0
  178. package/src/schema/command-schema.ts +118 -0
  179. package/src/schema/index.ts +5 -0
  180. package/src/test-setup.ts +45 -0
  181. package/src/testing/index.ts +137 -0
@@ -0,0 +1,916 @@
1
+ /**
2
+ * Base Tokenizer Class
3
+ *
4
+ * Abstract base class for language-specific tokenizers.
5
+ * Provides keyword management, morphological normalization,
6
+ * and high-level token extraction methods.
7
+ */
8
+
9
+ import type { LanguageToken, TokenKind, TokenStream, LanguageTokenizer } from '../types';
10
+ import type { MorphologicalNormalizer, NormalizationResult } from './morphology/types';
11
+ import {
12
+ type ValueExtractor,
13
+ type KeywordEntry,
14
+ isContextAwareExtractor,
15
+ createTokenizerContext,
16
+ } from '../../interfaces/value-extractor';
17
+ import {
18
+ createToken,
19
+ createPosition,
20
+ isWhitespace,
21
+ isDigit,
22
+ isAsciiIdentifierChar,
23
+ TokenStreamImpl,
24
+ type TimeUnitMapping,
25
+ type CreateTokenOptions,
26
+ } from './token-utils';
27
+ import { extractCssSelector, extractStringLiteral, extractNumber, extractUrl } from './extractors';
28
+ import { DEFAULT_OPERATORS } from './extractors/operator';
29
+ import { getDefaultExtractors } from './default-extractors';
30
+
31
+ // Module-scope operator set for O(1) lookup in createSimpleTokenizer.
32
+ // Uses the canonical list from OperatorExtractor to avoid duplication.
33
+ const SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
34
+
35
+ // =============================================================================
36
+ // Types
37
+ // =============================================================================
38
+
39
+ // KeywordEntry is imported from interfaces/value-extractor and re-exported
40
+ // for backward compatibility with code importing from this module.
41
+ export type { KeywordEntry };
42
+
43
+ /**
44
+ * Profile interface for keyword derivation.
45
+ * Matches the structure of LanguageProfile but only includes fields needed for tokenization.
46
+ */
47
+ export interface TokenizerProfile {
48
+ readonly keywords?: Record<
49
+ string,
50
+ { primary: string; alternatives?: string[]; normalized?: string }
51
+ >;
52
+ readonly references?: Record<string, string>;
53
+ readonly roleMarkers?: Record<
54
+ string,
55
+ { primary: string; alternatives?: string[]; position?: string }
56
+ >;
57
+ readonly possessive?: {
58
+ readonly marker: string;
59
+ readonly markerPosition: 'after-object' | 'between' | 'before-property';
60
+ readonly specialForms?: Record<string, string>;
61
+ readonly usePossessiveAdjectives?: boolean;
62
+ readonly keywords?: Record<string, string>;
63
+ };
64
+ }
65
+
66
+ // =============================================================================
67
+ // Base Tokenizer Class
68
+ // =============================================================================
69
+
70
+ /**
71
+ * Abstract base class for language-specific tokenizers.
72
+ * Provides common functionality for CSS selectors, strings, and numbers.
73
+ */
74
+ export abstract class BaseTokenizer implements LanguageTokenizer {
75
+ abstract readonly language: string;
76
+ abstract readonly direction: 'ltr' | 'rtl';
77
+
78
+ /** Optional morphological normalizer for this language */
79
+ protected normalizer?: MorphologicalNormalizer;
80
+
81
+ /** Keywords derived from profile, sorted longest-first for greedy matching */
82
+ protected profileKeywords: KeywordEntry[] = [];
83
+
84
+ /** Map for O(1) keyword lookups by lowercase native word */
85
+ protected profileKeywordMap: Map<string, KeywordEntry> = new Map();
86
+
87
+ /**
88
+ * Pluggable value extractors for domain-specific syntax.
89
+ * When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
90
+ */
91
+ protected extractors: ValueExtractor[] = [];
92
+
93
+ /**
94
+ * Tokenize input string to token stream.
95
+ * Delegates to extractor-based tokenization if extractors are registered,
96
+ * otherwise subclass must override this method.
97
+ *
98
+ * @param input - Input string to tokenize
99
+ * @returns Token stream
100
+ */
101
+ tokenize(input: string): TokenStream {
102
+ if (this.isUsingExtractors()) {
103
+ return this.tokenizeWithExtractors(input);
104
+ }
105
+
106
+ // If no extractors registered, subclass must provide implementation
107
+ throw new Error(
108
+ `${this.constructor.name}: tokenize() not implemented and no extractors registered. ` +
109
+ 'Either register extractors or override tokenize() method.'
110
+ );
111
+ }
112
+
113
+ abstract classifyToken(token: string): TokenKind;
114
+
115
+ /**
116
+ * Register a value extractor for domain-specific syntax.
117
+ * Extractors are tried in registration order during tokenization.
118
+ * Context-aware extractors automatically receive the tokenizer context.
119
+ *
120
+ * @param extractor - Value extractor to register
121
+ */
122
+ registerExtractor(extractor: ValueExtractor): void {
123
+ if (isContextAwareExtractor(extractor)) {
124
+ extractor.setContext(createTokenizerContext(this as any));
125
+ }
126
+ this.extractors.push(extractor);
127
+ }
128
+
129
+ /**
130
+ * Register multiple value extractors at once.
131
+ *
132
+ * @param extractors - Array of value extractors to register
133
+ */
134
+ registerExtractors(extractors: ValueExtractor[]): void {
135
+ for (const extractor of extractors) {
136
+ this.registerExtractor(extractor);
137
+ }
138
+ }
139
+
140
+ /**
141
+ * Clear all registered extractors.
142
+ * Returns tokenizer to legacy mode.
143
+ */
144
+ clearExtractors(): void {
145
+ this.extractors = [];
146
+ }
147
+
148
+ /**
149
+ * Check if this tokenizer is using extractor-based tokenization.
150
+ * Returns true if any extractors are registered.
151
+ */
152
+ protected isUsingExtractors(): boolean {
153
+ return this.extractors.length > 0;
154
+ }
155
+
156
+ /**
157
+ * Tokenize input using registered value extractors.
158
+ * This is the new path - extractors handle all syntax detection.
159
+ *
160
+ * @param input - Input string to tokenize
161
+ * @returns Token stream
162
+ */
163
+ protected tokenizeWithExtractors(input: string): TokenStream {
164
+ const tokens: LanguageToken[] = [];
165
+ let pos = 0;
166
+
167
+ while (pos < input.length) {
168
+ // Skip whitespace
169
+ while (pos < input.length && isWhitespace(input[pos])) {
170
+ pos++;
171
+ }
172
+ if (pos >= input.length) break;
173
+
174
+ // Try registered extractors in order
175
+ let extracted = false;
176
+ for (const extractor of this.extractors) {
177
+ if (extractor.canExtract(input, pos)) {
178
+ const result = extractor.extract(input, pos);
179
+ if (result) {
180
+ // Promote normalized/stem/stemConfidence from metadata to top-level token options
181
+ const normalized = result.metadata?.normalized as string | undefined;
182
+ const stem = result.metadata?.stem as string | undefined;
183
+ const stemConfidence = result.metadata?.stemConfidence as number | undefined;
184
+
185
+ // Build clean metadata without promoted fields
186
+ const cleanMetadata: Record<string, unknown> = {};
187
+ if (result.metadata) {
188
+ for (const [key, value] of Object.entries(result.metadata)) {
189
+ if (key !== 'normalized' && key !== 'stem' && key !== 'stemConfidence') {
190
+ cleanMetadata[key] = value;
191
+ }
192
+ }
193
+ }
194
+
195
+ const options: CreateTokenOptions = {};
196
+ if (normalized) options.normalized = normalized;
197
+ if (stem) options.stem = stem;
198
+ if (stemConfidence !== undefined) options.stemConfidence = stemConfidence;
199
+ if (Object.keys(cleanMetadata).length > 0) options.metadata = cleanMetadata;
200
+
201
+ tokens.push(
202
+ createToken(
203
+ result.value,
204
+ this.classifyToken(result.value),
205
+ createPosition(pos, pos + result.length),
206
+ Object.keys(options).length > 0 ? options : undefined
207
+ )
208
+ );
209
+ pos += result.length;
210
+ extracted = true;
211
+ break;
212
+ }
213
+ }
214
+ }
215
+
216
+ // Fallback: single character as operator/punctuation
217
+ if (!extracted) {
218
+ const char = input[pos];
219
+ const kind = this.classifyUnknownChar(char);
220
+ tokens.push(createToken(char, kind, createPosition(pos, pos + 1)));
221
+ pos++;
222
+ }
223
+ }
224
+
225
+ return new TokenStreamImpl(tokens, this.language);
226
+ }
227
+
228
+ /**
229
+ * Classify an unknown character when no extractor matches.
230
+ * Provides sensible defaults for common syntax.
231
+ *
232
+ * @param char - Character to classify
233
+ * @returns Token kind
234
+ */
235
+ protected classifyUnknownChar(char: string): TokenKind {
236
+ if ('()[]{},:;'.includes(char)) return 'punctuation';
237
+ if ('+-*/<>=!&|'.includes(char)) return 'operator';
238
+ return 'identifier';
239
+ }
240
+
241
+ /**
242
+ * Check if current position is a property access (obj.prop) vs CSS selector (.active).
243
+ * Property access: no whitespace before '.', previous token is identifier/keyword/selector.
244
+ * Also detects standalone method calls: .identifier( pattern.
245
+ *
246
+ * Returns true if '.' was emitted as an operator token and pos should advance by 1.
247
+ * Returns false if this is a CSS selector and should be handled by trySelector().
248
+ */
249
+ protected tryPropertyAccess(input: string, pos: number, tokens: LanguageToken[]): boolean {
250
+ if (input[pos] !== '.') return false;
251
+
252
+ const lastToken = tokens[tokens.length - 1];
253
+ // Property access requires NO whitespace between tokens (e.g., "obj.prop")
254
+ const hasWhitespaceBefore = lastToken && lastToken.position.end < pos;
255
+ const isPropertyAccess =
256
+ lastToken &&
257
+ !hasWhitespaceBefore &&
258
+ (lastToken.kind === 'identifier' ||
259
+ lastToken.kind === 'keyword' ||
260
+ lastToken.kind === 'selector');
261
+
262
+ if (isPropertyAccess) {
263
+ tokens.push(createToken('.', 'operator', createPosition(pos, pos + 1)));
264
+ return true;
265
+ }
266
+
267
+ // Check for method call pattern at start: .identifier(
268
+ const methodStart = pos + 1;
269
+ let methodEnd = methodStart;
270
+ while (methodEnd < input.length && isAsciiIdentifierChar(input[methodEnd])) {
271
+ methodEnd++;
272
+ }
273
+ if (methodEnd < input.length && input[methodEnd] === '(') {
274
+ tokens.push(createToken('.', 'operator', createPosition(pos, pos + 1)));
275
+ return true;
276
+ }
277
+
278
+ return false;
279
+ }
280
+
281
+ /**
282
+ * Initialize keyword mappings from a language profile.
283
+ * Builds a list of native→english mappings from:
284
+ * - profile.keywords (primary + alternatives)
285
+ * - profile.references (me, it, you, etc.)
286
+ * - profile.roleMarkers (into, from, with, etc.)
287
+ *
288
+ * Results are sorted longest-first for greedy matching (important for non-space languages).
289
+ * Extras take precedence over profile entries when there are duplicates.
290
+ *
291
+ * @param profile - Language profile containing keyword translations
292
+ * @param extras - Additional keyword entries to include (literals, positional, events)
293
+ */
294
+ protected initializeKeywordsFromProfile(
295
+ profile: TokenizerProfile,
296
+ extras: KeywordEntry[] = []
297
+ ): void {
298
+ // Use a Map to deduplicate, with extras taking precedence
299
+ const keywordMap = new Map<string, KeywordEntry>();
300
+
301
+ // Extract from keywords (command translations)
302
+ if (profile.keywords) {
303
+ for (const [normalized, translation] of Object.entries(profile.keywords)) {
304
+ // Primary translation
305
+ keywordMap.set(translation.primary, {
306
+ native: translation.primary,
307
+ normalized: translation.normalized || normalized,
308
+ });
309
+
310
+ // Alternative forms
311
+ if (translation.alternatives) {
312
+ for (const alt of translation.alternatives) {
313
+ keywordMap.set(alt, {
314
+ native: alt,
315
+ normalized: translation.normalized || normalized,
316
+ });
317
+ }
318
+ }
319
+ }
320
+ }
321
+
322
+ // Extract from references (me, it, you, etc.)
323
+ if (profile.references) {
324
+ for (const [normalized, native] of Object.entries(profile.references)) {
325
+ keywordMap.set(native, { native, normalized });
326
+ }
327
+ // Also register English canonical forms as universal fallbacks.
328
+ // Users frequently mix English references (me, it, you) into non-English
329
+ // hyperscript (e.g., "alternar .active on me"). Without this, the English
330
+ // word "me" would be unrecognized in non-English token streams.
331
+ for (const canonical of Object.keys(profile.references)) {
332
+ if (!keywordMap.has(canonical)) {
333
+ keywordMap.set(canonical, { native: canonical, normalized: canonical });
334
+ }
335
+ }
336
+ }
337
+
338
+ // Extract from roleMarkers (into, from, with, etc.)
339
+ if (profile.roleMarkers) {
340
+ for (const [role, marker] of Object.entries(profile.roleMarkers)) {
341
+ if (marker.primary) {
342
+ keywordMap.set(marker.primary, { native: marker.primary, normalized: role });
343
+ }
344
+ if (marker.alternatives) {
345
+ for (const alt of marker.alternatives) {
346
+ keywordMap.set(alt, { native: alt, normalized: role });
347
+ }
348
+ }
349
+ }
350
+ }
351
+
352
+ // Extract from possessive keywords (e.g., ñuqapa, qampa for Quechua)
353
+ if (profile.possessive?.keywords) {
354
+ for (const [native, normalized] of Object.entries(profile.possessive.keywords)) {
355
+ keywordMap.set(native, { native, normalized });
356
+ }
357
+ }
358
+
359
+ // Add extra entries (literals, positional, events) - these OVERRIDE profile entries
360
+ for (const extra of extras) {
361
+ keywordMap.set(extra.native, extra);
362
+ }
363
+
364
+ // Convert to array and sort longest-first for greedy matching
365
+ this.profileKeywords = Array.from(keywordMap.values()).sort(
366
+ (a, b) => b.native.length - a.native.length
367
+ );
368
+
369
+ // Build Map for O(1) lookups (case-insensitive + diacritic-insensitive)
370
+ // This allows matching both 'بدّل' (with shadda) and 'بدل' (without) to the same entry
371
+ this.profileKeywordMap = new Map();
372
+ for (const keyword of this.profileKeywords) {
373
+ // Add original form (with diacritics if present)
374
+ this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
375
+
376
+ // Add diacritic-normalized form (for Arabic, Turkish, etc.)
377
+ const normalized = this.removeDiacritics(keyword.native);
378
+ if (normalized !== keyword.native && !this.profileKeywordMap.has(normalized.toLowerCase())) {
379
+ this.profileKeywordMap.set(normalized.toLowerCase(), keyword);
380
+ }
381
+ }
382
+ }
383
+
384
+ /**
385
+ * Remove diacritical marks from a word for normalization.
386
+ * Primarily for Arabic (shadda, fatha, kasra, damma, sukun, etc.)
387
+ * but could be extended for other languages.
388
+ *
389
+ * @param word - Word to normalize
390
+ * @returns Word without diacritics
391
+ */
392
+ protected removeDiacritics(word: string): string {
393
+ // Arabic diacritics: U+064B-U+0652 (fatha, kasra, damma, sukun, shadda, etc.)
394
+ // U+0670 (superscript alif)
395
+ return word.replace(/[\u064B-\u0652\u0670]/g, '');
396
+ }
397
+
398
+ /**
399
+ * Try to match a keyword from profile at the current position.
400
+ * Uses longest-first greedy matching (important for non-space languages).
401
+ *
402
+ * @param input - Input string
403
+ * @param pos - Current position
404
+ * @returns Token if matched, null otherwise
405
+ */
406
+ protected tryProfileKeyword(input: string, pos: number): LanguageToken | null {
407
+ for (const entry of this.profileKeywords) {
408
+ if (input.slice(pos).startsWith(entry.native)) {
409
+ return createToken(
410
+ entry.native,
411
+ 'keyword',
412
+ createPosition(pos, pos + entry.native.length),
413
+ entry.normalized
414
+ );
415
+ }
416
+ }
417
+ return null;
418
+ }
419
+
420
+ /**
421
+ * Check if the remaining input starts with any known keyword.
422
+ * Useful for non-space languages to detect word boundaries.
423
+ *
424
+ * @param input - Input string
425
+ * @param pos - Current position
426
+ * @returns true if a keyword starts at this position
427
+ */
428
+ protected isKeywordStart(input: string, pos: number): boolean {
429
+ const remaining = input.slice(pos);
430
+ return this.profileKeywords.some(entry => remaining.startsWith(entry.native));
431
+ }
432
+
433
+ /**
434
+ * Look up a keyword by native word (case-insensitive).
435
+ * O(1) lookup using the keyword map.
436
+ *
437
+ * @param native - Native word to look up
438
+ * @returns KeywordEntry if found, undefined otherwise
439
+ */
440
+ protected lookupKeyword(native: string): KeywordEntry | undefined {
441
+ return this.profileKeywordMap.get(native.toLowerCase());
442
+ }
443
+
444
+ /**
445
+ * Check if a word is a known keyword (case-insensitive).
446
+ * O(1) lookup using the keyword map.
447
+ *
448
+ * @param native - Native word to check
449
+ * @returns true if the word is a keyword
450
+ */
451
+ protected isKeyword(native: string): boolean {
452
+ return this.profileKeywordMap.has(native.toLowerCase());
453
+ }
454
+
455
+ /**
456
+ * Set the morphological normalizer for this tokenizer.
457
+ */
458
+ setNormalizer(normalizer: MorphologicalNormalizer): void {
459
+ this.normalizer = normalizer;
460
+ }
461
+
462
+ /**
463
+ * Try to normalize a word using the morphological normalizer.
464
+ * Returns null if no normalizer is set or normalization fails.
465
+ *
466
+ * Note: We don't check isNormalizable() here because the individual tokenizers
467
+ * historically called normalize() directly without that check. The normalize()
468
+ * method itself handles returning noChange() for words that can't be normalized.
469
+ */
470
+ protected tryNormalize(word: string): NormalizationResult | null {
471
+ if (!this.normalizer) return null;
472
+
473
+ const result = this.normalizer.normalize(word);
474
+
475
+ // Only return if actually normalized (stem differs from input)
476
+ if (result.stem !== word && result.confidence >= 0.7) {
477
+ return result;
478
+ }
479
+
480
+ return null;
481
+ }
482
+
483
+ /**
484
+ * Try morphological normalization and keyword lookup.
485
+ *
486
+ * If the word can be normalized to a stem that matches a known keyword,
487
+ * returns a keyword token with morphological metadata (stem, stemConfidence).
488
+ *
489
+ * This is the common pattern for handling conjugated verbs across languages:
490
+ * 1. Normalize the word (e.g., "toggled" → "toggle")
491
+ * 2. Look up the stem in the keyword map
492
+ * 3. Create a token with both the original form and stem metadata
493
+ *
494
+ * @param word - The word to normalize and look up
495
+ * @param startPos - Start position for the token
496
+ * @param endPos - End position for the token
497
+ * @returns Token if stem matches a keyword, null otherwise
498
+ */
499
+ protected tryMorphKeywordMatch(
500
+ word: string,
501
+ startPos: number,
502
+ endPos: number
503
+ ): LanguageToken | null {
504
+ const result = this.tryNormalize(word);
505
+ if (!result) return null;
506
+
507
+ // Check if the stem is a known keyword
508
+ const stemEntry = this.lookupKeyword(result.stem);
509
+ if (!stemEntry) return null;
510
+
511
+ const tokenOptions: CreateTokenOptions = {
512
+ normalized: stemEntry.normalized,
513
+ stem: result.stem,
514
+ stemConfidence: result.confidence,
515
+ };
516
+ return createToken(word, 'keyword', createPosition(startPos, endPos), tokenOptions);
517
+ }
518
+
519
+ /**
520
+ * Try to extract a CSS selector at the current position.
521
+ */
522
+ protected trySelector(input: string, pos: number): LanguageToken | null {
523
+ const selector = extractCssSelector(input, pos);
524
+ if (selector) {
525
+ return createToken(selector, 'selector', createPosition(pos, pos + selector.length));
526
+ }
527
+ return null;
528
+ }
529
+
530
+ /**
531
+ * Try to extract an event modifier at the current position.
532
+ * Event modifiers are .once, .debounce(N), .throttle(N), .queue(strategy)
533
+ */
534
+ protected tryEventModifier(input: string, pos: number): LanguageToken | null {
535
+ // Must start with a dot
536
+ if (input[pos] !== '.') {
537
+ return null;
538
+ }
539
+
540
+ // Match pattern: .(once|debounce|throttle|queue) followed by optional (value)
541
+ const match = input
542
+ .slice(pos)
543
+ .match(/^\.(?:once|debounce|throttle|queue)(?:\(([^)]+)\))?(?:\s|$|\.)/);
544
+ if (!match) {
545
+ return null;
546
+ }
547
+
548
+ const fullMatch = match[0].replace(/(\s|\.)$/, ''); // Remove trailing space or dot
549
+ const modifierName = fullMatch.slice(1).split('(')[0]; // Extract modifier name
550
+ const value = match[1]; // Extract value from parentheses if present
551
+
552
+ // Create token with metadata
553
+ const token = createToken(
554
+ fullMatch,
555
+ 'event-modifier',
556
+ createPosition(pos, pos + fullMatch.length)
557
+ );
558
+
559
+ // Add metadata for the modifier
560
+ return {
561
+ ...token,
562
+ metadata: {
563
+ modifierName,
564
+ value: value ? (modifierName === 'queue' ? value : parseInt(value, 10)) : undefined,
565
+ },
566
+ };
567
+ }
568
+
569
+ /**
570
+ * Try to extract a string literal at the current position.
571
+ */
572
+ protected tryString(input: string, pos: number): LanguageToken | null {
573
+ const literal = extractStringLiteral(input, pos);
574
+ if (literal) {
575
+ return createToken(literal, 'literal', createPosition(pos, pos + literal.length));
576
+ }
577
+ return null;
578
+ }
579
+
580
+ /**
581
+ * Try to extract a number at the current position.
582
+ */
583
+ protected tryNumber(input: string, pos: number): LanguageToken | null {
584
+ const number = extractNumber(input, pos);
585
+ if (number) {
586
+ return createToken(number, 'literal', createPosition(pos, pos + number.length));
587
+ }
588
+ return null;
589
+ }
590
+
591
+ /**
592
+ * Configuration for native language time units.
593
+ * Maps patterns to their standard suffix (ms, s, m, h).
594
+ */
595
+ protected static readonly STANDARD_TIME_UNITS: readonly TimeUnitMapping[] = [
596
+ { pattern: 'ms', suffix: 'ms', length: 2 },
597
+ { pattern: 's', suffix: 's', length: 1, checkBoundary: true },
598
+ { pattern: 'm', suffix: 'm', length: 1, checkBoundary: true, notFollowedBy: 's' },
599
+ { pattern: 'h', suffix: 'h', length: 1, checkBoundary: true },
600
+ ];
601
+
602
+ /**
603
+ * Try to match a time unit from a list of patterns.
604
+ *
605
+ * @param input - Input string
606
+ * @param pos - Position after the number
607
+ * @param timeUnits - Array of time unit mappings (native pattern → standard suffix)
608
+ * @param skipWhitespace - Whether to skip whitespace before time unit (default: false)
609
+ * @returns Object with matched suffix and new position, or null if no match
610
+ */
611
+ protected tryMatchTimeUnit(
612
+ input: string,
613
+ pos: number,
614
+ timeUnits: readonly TimeUnitMapping[],
615
+ skipWhitespace = false
616
+ ): { suffix: string; endPos: number } | null {
617
+ let unitPos = pos;
618
+
619
+ // Optionally skip whitespace before time unit
620
+ if (skipWhitespace) {
621
+ while (unitPos < input.length && isWhitespace(input[unitPos])) {
622
+ unitPos++;
623
+ }
624
+ }
625
+
626
+ const remaining = input.slice(unitPos);
627
+
628
+ // Check each time unit pattern
629
+ for (const unit of timeUnits) {
630
+ const candidate = remaining.slice(0, unit.length);
631
+ const matches = unit.caseInsensitive
632
+ ? candidate.toLowerCase() === unit.pattern.toLowerCase()
633
+ : candidate === unit.pattern;
634
+
635
+ if (matches) {
636
+ // Check notFollowedBy constraint (e.g., 'm' should not match 'ms')
637
+ if (unit.notFollowedBy) {
638
+ const nextChar = remaining[unit.length] || '';
639
+ if (nextChar === unit.notFollowedBy) continue;
640
+ }
641
+
642
+ // Check word boundary if required
643
+ if (unit.checkBoundary) {
644
+ const nextChar = remaining[unit.length] || '';
645
+ if (isAsciiIdentifierChar(nextChar)) continue;
646
+ }
647
+
648
+ return { suffix: unit.suffix, endPos: unitPos + unit.length };
649
+ }
650
+ }
651
+
652
+ return null;
653
+ }
654
+
655
+ /**
656
+ * Parse a base number (sign, integer, decimal) without time units.
657
+ * Returns the number string and end position.
658
+ *
659
+ * @param input - Input string
660
+ * @param startPos - Start position
661
+ * @param allowSign - Whether to allow +/- sign (default: true)
662
+ * @returns Object with number string and end position, or null
663
+ */
664
+ protected parseBaseNumber(
665
+ input: string,
666
+ startPos: number,
667
+ allowSign = true
668
+ ): { number: string; endPos: number } | null {
669
+ let pos = startPos;
670
+ let number = '';
671
+
672
+ // Optional sign
673
+ if (allowSign && (input[pos] === '-' || input[pos] === '+')) {
674
+ number += input[pos++];
675
+ }
676
+
677
+ // Must have at least one digit
678
+ if (pos >= input.length || !isDigit(input[pos])) {
679
+ return null;
680
+ }
681
+
682
+ // Integer part
683
+ while (pos < input.length && isDigit(input[pos])) {
684
+ number += input[pos++];
685
+ }
686
+
687
+ // Optional decimal
688
+ if (pos < input.length && input[pos] === '.') {
689
+ number += input[pos++];
690
+ while (pos < input.length && isDigit(input[pos])) {
691
+ number += input[pos++];
692
+ }
693
+ }
694
+
695
+ if (!number || number === '-' || number === '+') return null;
696
+
697
+ return { number, endPos: pos };
698
+ }
699
+
700
+ /**
701
+ * Try to extract a number with native language time units.
702
+ *
703
+ * This is a template method that handles the common pattern:
704
+ * 1. Parse the base number (sign, integer, decimal)
705
+ * 2. Try to match native language time units
706
+ * 3. Fall back to standard time units (ms, s, m, h)
707
+ *
708
+ * @param input - Input string
709
+ * @param pos - Start position
710
+ * @param nativeTimeUnits - Language-specific time unit mappings
711
+ * @param options - Configuration options
712
+ * @returns Token if number found, null otherwise
713
+ */
714
+ protected tryNumberWithTimeUnits(
715
+ input: string,
716
+ pos: number,
717
+ nativeTimeUnits: readonly TimeUnitMapping[],
718
+ options: { allowSign?: boolean; skipWhitespace?: boolean } = {}
719
+ ): LanguageToken | null {
720
+ const { allowSign = true, skipWhitespace = false } = options;
721
+
722
+ // Parse base number
723
+ const baseResult = this.parseBaseNumber(input, pos, allowSign);
724
+ if (!baseResult) return null;
725
+
726
+ let { number, endPos } = baseResult;
727
+
728
+ // Try native time units first, then standard
729
+ const allUnits = [...nativeTimeUnits, ...BaseTokenizer.STANDARD_TIME_UNITS];
730
+ const timeMatch = this.tryMatchTimeUnit(input, endPos, allUnits, skipWhitespace);
731
+
732
+ if (timeMatch) {
733
+ number += timeMatch.suffix;
734
+ endPos = timeMatch.endPos;
735
+ }
736
+
737
+ return createToken(number, 'literal', createPosition(pos, endPos));
738
+ }
739
+
740
+ /**
741
+ * Try to extract a URL at the current position.
742
+ * Handles /path, ./path, ../path, //domain.com, http://, https://
743
+ */
744
+ protected tryUrl(input: string, pos: number): LanguageToken | null {
745
+ const url = extractUrl(input, pos);
746
+ if (url) {
747
+ return createToken(url, 'url', createPosition(pos, pos + url.length));
748
+ }
749
+ return null;
750
+ }
751
+
752
+ /**
753
+ * Try to extract a variable reference (:varname) at the current position.
754
+ * In hyperscript, :x refers to a local variable named x.
755
+ */
756
+ protected tryVariableRef(input: string, pos: number): LanguageToken | null {
757
+ if (input[pos] !== ':') return null;
758
+ if (pos + 1 >= input.length) return null;
759
+ if (!isAsciiIdentifierChar(input[pos + 1])) return null;
760
+
761
+ let endPos = pos + 1;
762
+ while (endPos < input.length && isAsciiIdentifierChar(input[endPos])) {
763
+ endPos++;
764
+ }
765
+
766
+ const varRef = input.slice(pos, endPos);
767
+ return createToken(varRef, 'identifier', createPosition(pos, endPos));
768
+ }
769
+
770
+ /**
771
+ * Try to extract an operator or punctuation token at the current position.
772
+ * Handles two-character operators (==, !=, etc.) and single-character operators.
773
+ */
774
+ protected tryOperator(input: string, pos: number): LanguageToken | null {
775
+ // Two-character operators
776
+ const twoChar = input.slice(pos, pos + 2);
777
+ if (['==', '!=', '<=', '>=', '&&', '||', '->'].includes(twoChar)) {
778
+ return createToken(twoChar, 'operator', createPosition(pos, pos + 2));
779
+ }
780
+
781
+ // Single-character operators
782
+ const oneChar = input[pos];
783
+ if (['<', '>', '!', '+', '-', '*', '/', '='].includes(oneChar)) {
784
+ return createToken(oneChar, 'operator', createPosition(pos, pos + 1));
785
+ }
786
+
787
+ // Punctuation
788
+ if (['(', ')', '{', '}', ',', ';', ':'].includes(oneChar)) {
789
+ return createToken(oneChar, 'punctuation', createPosition(pos, pos + 1));
790
+ }
791
+
792
+ return null;
793
+ }
794
+
795
+ /**
796
+ * Try to match a multi-character particle from a list.
797
+ *
798
+ * Used by languages like Japanese, Korean, and Chinese that have
799
+ * multi-character particles (e.g., Japanese から, まで, より).
800
+ *
801
+ * @param input - Input string
802
+ * @param pos - Current position
803
+ * @param particles - Array of multi-character particles to match
804
+ * @returns Token if matched, null otherwise
805
+ */
806
+ protected tryMultiCharParticle(
807
+ input: string,
808
+ pos: number,
809
+ particles: readonly string[]
810
+ ): LanguageToken | null {
811
+ for (const particle of particles) {
812
+ if (input.slice(pos, pos + particle.length) === particle) {
813
+ return createToken(particle, 'particle', createPosition(pos, pos + particle.length));
814
+ }
815
+ }
816
+ return null;
817
+ }
818
+ }
819
+
820
+ // =============================================================================
821
+ // Simple Tokenizer Factory
822
+ // =============================================================================
823
+
824
+ /**
825
+ * Configuration for createSimpleTokenizer.
826
+ *
827
+ * Creates a tokenizer from declarative config instead of a class definition.
828
+ * Covers the common pattern used by domain packages (SQL, BDD, JSX).
829
+ *
830
+ * **Keyword resolution** uses two additive paths:
831
+ * 1. `keywords` — explicit list, checked first. Respects `caseInsensitive`.
832
+ * 2. `keywordProfile` — populates BaseTokenizer's profile keyword map via
833
+ * `initializeKeywordsFromProfile()`. Checked second via `isKeyword()`, which
834
+ * always lowercases (harmless for CJK/Arabic; notable for Latin scripts
835
+ * with `caseInsensitive: false`). Provides normalization metadata for
836
+ * non-Latin scripts.
837
+ */
838
+ export interface SimpleTokenizerConfig {
839
+ /** ISO 639-1 language code */
840
+ language: string;
841
+ /** Text direction (default: 'ltr') */
842
+ direction?: 'ltr' | 'rtl';
843
+ /** Keywords to recognize (lowercased for lookup if caseInsensitive) */
844
+ keywords: string[];
845
+ /** Extra keyword entries for non-Latin normalization */
846
+ keywordExtras?: KeywordEntry[];
847
+ /** Profile for initializeKeywordsFromProfile (for non-Latin scripts) */
848
+ keywordProfile?: TokenizerProfile;
849
+ /** Include operator classification (default: false). Uses DEFAULT_OPERATORS from OperatorExtractor. */
850
+ includeOperators?: boolean;
851
+ /** Case-insensitive keyword matching (default: true) */
852
+ caseInsensitive?: boolean;
853
+ /** Custom extractors registered BEFORE default extractors */
854
+ customExtractors?: ValueExtractor[];
855
+ }
856
+
857
+ /**
858
+ * Create a tokenizer from declarative configuration.
859
+ *
860
+ * Eliminates the boilerplate of extending BaseTokenizer for simple domain tokenizers.
861
+ * Handles keyword classification, optional operator support, and non-Latin keyword setup.
862
+ *
863
+ * @example
864
+ * ```typescript
865
+ * const englishSQL = createSimpleTokenizer({
866
+ * language: 'en',
867
+ * keywords: ['select', 'insert', 'update', 'delete', 'from', 'into', 'where', 'set', 'values'],
868
+ * includeOperators: true,
869
+ * caseInsensitive: true,
870
+ * });
871
+ * ```
872
+ */
873
+ export function createSimpleTokenizer(config: SimpleTokenizerConfig): LanguageTokenizer {
874
+ const {
875
+ language,
876
+ direction = 'ltr',
877
+ keywords,
878
+ keywordExtras,
879
+ keywordProfile,
880
+ includeOperators = false,
881
+ caseInsensitive = true,
882
+ customExtractors,
883
+ } = config;
884
+
885
+ const keywordSet = new Set(caseInsensitive ? keywords.map(k => k.toLowerCase()) : keywords);
886
+
887
+ class SimpleTokenizer extends BaseTokenizer {
888
+ readonly language = language;
889
+ readonly direction = direction;
890
+
891
+ constructor() {
892
+ super();
893
+ if (customExtractors) {
894
+ this.registerExtractors(customExtractors);
895
+ }
896
+ this.registerExtractors(getDefaultExtractors());
897
+ if (keywordProfile) {
898
+ this.initializeKeywordsFromProfile(keywordProfile, keywordExtras);
899
+ }
900
+ }
901
+
902
+ classifyToken(token: string): TokenKind {
903
+ // Fast path: explicit keywords from config (respects caseInsensitive)
904
+ const lookup = caseInsensitive ? token.toLowerCase() : token;
905
+ if (keywordSet.has(lookup)) return 'keyword';
906
+ // Profile path: non-Latin normalization (always lowercases via profileKeywordMap)
907
+ if (this.isKeyword(token)) return 'keyword';
908
+ if (/^\d/.test(token)) return 'literal';
909
+ if (/^['"]/.test(token)) return 'literal';
910
+ if (includeOperators && SIMPLE_TOKENIZER_OPERATOR_SET.has(token)) return 'operator';
911
+ return 'identifier';
912
+ }
913
+ }
914
+
915
+ return new SimpleTokenizer();
916
+ }