@lokascript/framework 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (181) hide show
  1. package/LICENSE +20 -0
  2. package/README.md +142 -0
  3. package/dist/aot/aot-orchestrator.d.ts +75 -0
  4. package/dist/aot/aot-orchestrator.d.ts.map +1 -0
  5. package/dist/aot/domain-scanner.d.ts +27 -0
  6. package/dist/aot/domain-scanner.d.ts.map +1 -0
  7. package/dist/aot/index.d.ts +8 -0
  8. package/dist/aot/index.d.ts.map +1 -0
  9. package/dist/aot/types.d.ts +103 -0
  10. package/dist/aot/types.d.ts.map +1 -0
  11. package/dist/api/create-dsl.d.ts +91 -0
  12. package/dist/api/create-dsl.d.ts.map +1 -0
  13. package/dist/api/dispatcher.d.ts +108 -0
  14. package/dist/api/dispatcher.d.ts.map +1 -0
  15. package/dist/api/domain-registry.d.ts +152 -0
  16. package/dist/api/domain-registry.d.ts.map +1 -0
  17. package/dist/api/index.d.ts +7 -0
  18. package/dist/api/index.d.ts.map +1 -0
  19. package/dist/api/index.js +2082 -0
  20. package/dist/api/index.js.map +1 -0
  21. package/dist/core/index.d.ts +7 -0
  22. package/dist/core/index.d.ts.map +1 -0
  23. package/dist/core/index.js +2674 -0
  24. package/dist/core/index.js.map +1 -0
  25. package/dist/core/logger.d.ts +32 -0
  26. package/dist/core/logger.d.ts.map +1 -0
  27. package/dist/core/pattern-matching/index.d.ts +6 -0
  28. package/dist/core/pattern-matching/index.d.ts.map +1 -0
  29. package/dist/core/pattern-matching/index.js +1239 -0
  30. package/dist/core/pattern-matching/index.js.map +1 -0
  31. package/dist/core/pattern-matching/pattern-matcher.d.ts +239 -0
  32. package/dist/core/pattern-matching/pattern-matcher.d.ts.map +1 -0
  33. package/dist/core/pattern-matching/utils/index.d.ts +6 -0
  34. package/dist/core/pattern-matching/utils/index.d.ts.map +1 -0
  35. package/dist/core/pattern-matching/utils/possessive-keywords.d.ts +38 -0
  36. package/dist/core/pattern-matching/utils/possessive-keywords.d.ts.map +1 -0
  37. package/dist/core/pattern-matching/utils/type-validation.d.ts +63 -0
  38. package/dist/core/pattern-matching/utils/type-validation.d.ts.map +1 -0
  39. package/dist/core/tokenization/base-tokenizer.d.ts +344 -0
  40. package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -0
  41. package/dist/core/tokenization/char-classifiers.d.ts +56 -0
  42. package/dist/core/tokenization/char-classifiers.d.ts.map +1 -0
  43. package/dist/core/tokenization/default-extractors.d.ts +48 -0
  44. package/dist/core/tokenization/default-extractors.d.ts.map +1 -0
  45. package/dist/core/tokenization/extractors/index.d.ts +9 -0
  46. package/dist/core/tokenization/extractors/index.d.ts.map +1 -0
  47. package/dist/core/tokenization/extractors/operator.d.ts +23 -0
  48. package/dist/core/tokenization/extractors/operator.d.ts.map +1 -0
  49. package/dist/core/tokenization/extractors/punctuation.d.ts +22 -0
  50. package/dist/core/tokenization/extractors/punctuation.d.ts.map +1 -0
  51. package/dist/core/tokenization/extractors.d.ts +61 -0
  52. package/dist/core/tokenization/extractors.d.ts.map +1 -0
  53. package/dist/core/tokenization/index.d.ts +11 -0
  54. package/dist/core/tokenization/index.d.ts.map +1 -0
  55. package/dist/core/tokenization/index.js +1345 -0
  56. package/dist/core/tokenization/index.js.map +1 -0
  57. package/dist/core/tokenization/morphology/index.d.ts +5 -0
  58. package/dist/core/tokenization/morphology/index.d.ts.map +1 -0
  59. package/dist/core/tokenization/morphology/types.d.ts +110 -0
  60. package/dist/core/tokenization/morphology/types.d.ts.map +1 -0
  61. package/dist/core/tokenization/token-utils.d.ts +111 -0
  62. package/dist/core/tokenization/token-utils.d.ts.map +1 -0
  63. package/dist/core/types.d.ts +382 -0
  64. package/dist/core/types.d.ts.map +1 -0
  65. package/dist/core/types.js +108 -0
  66. package/dist/core/types.js.map +1 -0
  67. package/dist/generation/diagnostics.d.ts +120 -0
  68. package/dist/generation/diagnostics.d.ts.map +1 -0
  69. package/dist/generation/index.d.ts +7 -0
  70. package/dist/generation/index.d.ts.map +1 -0
  71. package/dist/generation/index.js +339 -0
  72. package/dist/generation/index.js.map +1 -0
  73. package/dist/generation/pattern-generator.d.ts +48 -0
  74. package/dist/generation/pattern-generator.d.ts.map +1 -0
  75. package/dist/generation/renderer.d.ts +115 -0
  76. package/dist/generation/renderer.d.ts.map +1 -0
  77. package/dist/grammar/index.d.ts +10 -0
  78. package/dist/grammar/index.d.ts.map +1 -0
  79. package/dist/grammar/index.js +391 -0
  80. package/dist/grammar/index.js.map +1 -0
  81. package/dist/grammar/transformer.d.ts +56 -0
  82. package/dist/grammar/transformer.d.ts.map +1 -0
  83. package/dist/grammar/types.d.ts +236 -0
  84. package/dist/grammar/types.d.ts.map +1 -0
  85. package/dist/index.cjs +4454 -0
  86. package/dist/index.cjs.map +1 -0
  87. package/dist/index.d.ts +46 -0
  88. package/dist/index.d.ts.map +1 -0
  89. package/dist/index.js +4336 -0
  90. package/dist/index.js.map +1 -0
  91. package/dist/interfaces/dictionary.d.ts +82 -0
  92. package/dist/interfaces/dictionary.d.ts.map +1 -0
  93. package/dist/interfaces/index.d.ts +10 -0
  94. package/dist/interfaces/index.d.ts.map +1 -0
  95. package/dist/interfaces/profile-provider.d.ts +67 -0
  96. package/dist/interfaces/profile-provider.d.ts.map +1 -0
  97. package/dist/interfaces/value-extractor.d.ts +168 -0
  98. package/dist/interfaces/value-extractor.d.ts.map +1 -0
  99. package/dist/multilingual/index.d.ts +8 -0
  100. package/dist/multilingual/index.d.ts.map +1 -0
  101. package/dist/multilingual/index.js +1 -0
  102. package/dist/multilingual/index.js.map +1 -0
  103. package/dist/parsing/index.d.ts +8 -0
  104. package/dist/parsing/index.d.ts.map +1 -0
  105. package/dist/parsing/index.js +1415 -0
  106. package/dist/parsing/index.js.map +1 -0
  107. package/dist/parsing/multi-statement.d.ts +265 -0
  108. package/dist/parsing/multi-statement.d.ts.map +1 -0
  109. package/dist/schema/command-schema.d.ts +78 -0
  110. package/dist/schema/command-schema.d.ts.map +1 -0
  111. package/dist/schema/index.d.ts +5 -0
  112. package/dist/schema/index.d.ts.map +1 -0
  113. package/dist/schema/index.js +25 -0
  114. package/dist/schema/index.js.map +1 -0
  115. package/dist/test-setup.d.ts +9 -0
  116. package/dist/test-setup.d.ts.map +1 -0
  117. package/dist/testing/index.d.ts +50 -0
  118. package/dist/testing/index.d.ts.map +1 -0
  119. package/dist/testing/index.js +16969 -0
  120. package/dist/testing/index.js.map +1 -0
  121. package/package.json +122 -0
  122. package/src/__test__/fixtures/sql-dsl.ts +232 -0
  123. package/src/__test__/sql-integration.test.ts +189 -0
  124. package/src/__test__/test-utils.ts +260 -0
  125. package/src/aot/aot-orchestrator.test.ts +413 -0
  126. package/src/aot/aot-orchestrator.ts +238 -0
  127. package/src/aot/domain-scanner.ts +178 -0
  128. package/src/aot/index.ts +8 -0
  129. package/src/aot/types.ts +124 -0
  130. package/src/api/create-dsl.ts +367 -0
  131. package/src/api/dispatcher.test.ts +336 -0
  132. package/src/api/dispatcher.ts +222 -0
  133. package/src/api/domain-registry.test.ts +336 -0
  134. package/src/api/domain-registry.ts +500 -0
  135. package/src/api/index.ts +7 -0
  136. package/src/core/index.ts +7 -0
  137. package/src/core/logger.ts +130 -0
  138. package/src/core/pattern-matching/index.ts +6 -0
  139. package/src/core/pattern-matching/pattern-matcher.test.ts +900 -0
  140. package/src/core/pattern-matching/pattern-matcher.ts +1548 -0
  141. package/src/core/pattern-matching/pattern-matcher.ts.backup +1267 -0
  142. package/src/core/pattern-matching/utils/index.ts +6 -0
  143. package/src/core/pattern-matching/utils/possessive-keywords.ts +55 -0
  144. package/src/core/pattern-matching/utils/type-validation.test.ts +316 -0
  145. package/src/core/pattern-matching/utils/type-validation.ts +134 -0
  146. package/src/core/tokenization/base-tokenizer.ts +916 -0
  147. package/src/core/tokenization/char-classifiers.ts +79 -0
  148. package/src/core/tokenization/create-simple-tokenizer.test.ts +260 -0
  149. package/src/core/tokenization/default-extractors.ts +69 -0
  150. package/src/core/tokenization/extractors/index.ts +9 -0
  151. package/src/core/tokenization/extractors/operator.ts +75 -0
  152. package/src/core/tokenization/extractors/punctuation.ts +39 -0
  153. package/src/core/tokenization/extractors.ts +452 -0
  154. package/src/core/tokenization/index.ts +11 -0
  155. package/src/core/tokenization/morphology/index.ts +5 -0
  156. package/src/core/tokenization/morphology/types.ts +211 -0
  157. package/src/core/tokenization/token-utils.ts +252 -0
  158. package/src/core/types.ts +589 -0
  159. package/src/generation/diagnostics.test.ts +171 -0
  160. package/src/generation/diagnostics.ts +239 -0
  161. package/src/generation/index.ts +7 -0
  162. package/src/generation/pattern-generator.test.ts +430 -0
  163. package/src/generation/pattern-generator.ts +315 -0
  164. package/src/generation/renderer.test.ts +266 -0
  165. package/src/generation/renderer.ts +244 -0
  166. package/src/grammar/index.ts +12 -0
  167. package/src/grammar/transformer.ts +159 -0
  168. package/src/grammar/types.ts +630 -0
  169. package/src/index.ts +157 -0
  170. package/src/interfaces/dictionary.ts +123 -0
  171. package/src/interfaces/index.ts +10 -0
  172. package/src/interfaces/profile-provider.ts +88 -0
  173. package/src/interfaces/value-extractor.ts +435 -0
  174. package/src/multilingual/index.ts +9 -0
  175. package/src/parsing/index.ts +27 -0
  176. package/src/parsing/multi-statement.test.ts +480 -0
  177. package/src/parsing/multi-statement.ts +648 -0
  178. package/src/schema/command-schema.ts +118 -0
  179. package/src/schema/index.ts +5 -0
  180. package/src/test-setup.ts +45 -0
  181. package/src/testing/index.ts +137 -0
@@ -0,0 +1,1548 @@
1
+ /**
2
+ * Pattern Matcher
3
+ *
4
+ * Matches tokenized input against language patterns to extract semantic roles.
5
+ * This is the core algorithm for multilingual parsing.
6
+ */
7
+
8
+ import type {
9
+ LanguagePattern,
10
+ PatternToken,
11
+ PatternMatchResult,
12
+ ReferenceValue,
13
+ SemanticRole,
14
+ SemanticValue,
15
+ TokenStream,
16
+ LanguageToken,
17
+ } from '../types';
18
+ import { createSelector, createLiteral, createReference, createPropertyPath } from '../types';
19
+ import { createLogger } from '../logger';
20
+
21
+ /**
22
+ * Helper to check if a value is a built-in reference.
23
+ * In the generic framework, we don't assume any built-in references.
24
+ * DSLs that need references (like hyperscript's 'me', 'it', 'you') should
25
+ * handle them in their tokenizer or provide a custom reference validator.
26
+ *
27
+ * For generic DSLs, identifiers are treated as expressions by default.
28
+ */
29
+ function isValidReference(_value: string): boolean {
30
+ // In generic framework, no built-in references
31
+ // Identifiers should be expressions
32
+ return false;
33
+ }
34
+ import { isTypeCompatible } from './utils/type-validation';
35
+ import { getPossessiveReference } from './utils/possessive-keywords';
36
+
37
+ /**
38
+ * Minimal profile interface needed by PatternMatcher.
39
+ * DSLs can provide this to enable possessive handling and language-specific features.
40
+ */
41
+ export interface PatternMatcherProfile {
42
+ readonly code: string;
43
+ readonly possessive?: {
44
+ readonly keywords?: Record<string, string>;
45
+ };
46
+ }
47
+
48
+ // =============================================================================
49
+ // Pattern Matcher
50
+ // =============================================================================
51
+
52
+ export class PatternMatcher {
53
+ /** Maximum tokens to scan ahead when looking for a group's leading marker */
54
+ private static readonly MAX_MARKER_SCAN = 3;
55
+
56
+ /** Debug logger */
57
+ private logger = createLogger('pattern-matcher');
58
+
59
+ /** Current language profile for the pattern being matched */
60
+ private currentProfile: PatternMatcherProfile | undefined;
61
+
62
+ /**
63
+ * Safely convert a value to lowercase string.
64
+ * Provides protection against non-string values at runtime.
65
+ */
66
+ private safeToLowerCase(value: unknown): string {
67
+ if (typeof value === 'string') {
68
+ return value.toLowerCase();
69
+ }
70
+ if (value === null || value === undefined) {
71
+ return '';
72
+ }
73
+ return String(value).toLowerCase();
74
+ }
75
+
76
+ /**
77
+ * Try to match a single pattern against the token stream.
78
+ * Returns the match result or null if no match.
79
+ *
80
+ * @param tokens - Token stream to match against
81
+ * @param pattern - Pattern to match
82
+ * @param profile - Optional language profile for possessive handling
83
+ */
84
+ matchPattern(
85
+ tokens: TokenStream,
86
+ pattern: LanguagePattern,
87
+ profile?: PatternMatcherProfile
88
+ ): PatternMatchResult | null {
89
+ const mark = tokens.mark();
90
+ const captured = new Map<SemanticRole, SemanticValue>();
91
+
92
+ // Debug logging
93
+ this.logger.debug('========================================');
94
+ this.logger.debug('matchPattern ENTRY');
95
+ this.logger.debug('Pattern ID:', pattern.id);
96
+ this.logger.debug('Pattern command:', pattern.command);
97
+ this.logger.debug('Pattern language:', pattern.language);
98
+ this.logger.debug('Pattern template:', JSON.stringify(pattern.template, null, 2));
99
+
100
+ if (this.logger.isEnabled()) {
101
+ const firstTokens = [];
102
+ for (let i = 0; i < 10; i++) {
103
+ const t = tokens.peek(i);
104
+ if (t)
105
+ firstTokens.push({
106
+ type: (t as any).type,
107
+ value: (t as any).value,
108
+ kind: (t as any).kind,
109
+ });
110
+ else break;
111
+ }
112
+ this.logger.debug('Input tokens (first 10):', firstTokens);
113
+ this.logger.debug('Profile code:', profile?.code);
114
+ }
115
+
116
+ // Use provided profile for possessive keyword lookup
117
+ this.currentProfile = profile;
118
+
119
+ // Reset match counters for this pattern
120
+ this.stemMatchCount = 0;
121
+ this.totalKeywordMatches = 0;
122
+
123
+ this.logger.debug('--- Calling matchTokenSequence ---');
124
+ this.logger.debug('Pattern tokens to match:', JSON.stringify(pattern.template.tokens, null, 2));
125
+ const success = this.matchTokenSequence(tokens, pattern.template.tokens, captured);
126
+ this.logger.debug('matchTokenSequence returned:', success);
127
+ this.logger.debug(
128
+ 'Captured roles:',
129
+ Array.from(captured.entries()).map(([k, v]) => [k, JSON.stringify(v)])
130
+ );
131
+
132
+ if (!success) {
133
+ this.logger.debug('>>> MATCH FAILED - resetting token position');
134
+ tokens.reset(mark);
135
+ return null;
136
+ }
137
+
138
+ // Calculate confidence BEFORE applying defaults
139
+ // This ensures defaulted roles don't artificially inflate confidence
140
+ const confidence = this.calculateConfidence(pattern, captured);
141
+
142
+ // Apply extraction rules to fill in any missing roles with defaults
143
+ this.applyExtractionRules(pattern, captured);
144
+
145
+ return {
146
+ pattern,
147
+ captured,
148
+ consumedTokens: tokens.position() - mark.position,
149
+ confidence,
150
+ };
151
+ }
152
+
153
+ /**
154
+ * Try to match multiple patterns, return the best match.
155
+ *
156
+ * @param tokens - Token stream to match against
157
+ * @param patterns - Candidate patterns to try
158
+ * @param profile - Optional language profile for possessive handling
159
+ */
160
+ matchBest(
161
+ tokens: TokenStream,
162
+ patterns: LanguagePattern[],
163
+ profile?: PatternMatcherProfile
164
+ ): PatternMatchResult | null {
165
+ const matches: PatternMatchResult[] = [];
166
+
167
+ for (const pattern of patterns) {
168
+ const mark = tokens.mark();
169
+ const result = this.matchPattern(tokens, pattern, profile);
170
+
171
+ if (result) {
172
+ matches.push(result);
173
+ }
174
+
175
+ tokens.reset(mark);
176
+ }
177
+
178
+ if (matches.length === 0) {
179
+ return null;
180
+ }
181
+
182
+ // Sort by confidence and priority
183
+ matches.sort((a, b) => {
184
+ // First by priority
185
+ const priorityDiff = b.pattern.priority - a.pattern.priority;
186
+ if (priorityDiff !== 0) return priorityDiff;
187
+
188
+ // Then by confidence
189
+ const confidenceDiff = b.confidence - a.confidence;
190
+ if (Math.abs(confidenceDiff) > 0.001) return confidenceDiff;
191
+
192
+ // Then by tokens consumed (prefer more complete matches)
193
+ return b.consumedTokens - a.consumedTokens;
194
+ });
195
+
196
+ // Re-consume tokens for the best match
197
+ const best = matches[0];
198
+ this.matchPattern(tokens, best.pattern);
199
+
200
+ return best;
201
+ }
202
+
203
+ /**
204
+ * Match a sequence of pattern tokens against the token stream.
205
+ *
206
+ * Supports bounded single-step backtracking: if an optional role consumes
207
+ * a token and the immediately following pattern token fails, the matcher
208
+ * resets to before the optional role and retries the failed token.
209
+ */
210
+ private matchTokenSequence(
211
+ tokens: TokenStream,
212
+ patternTokens: PatternToken[],
213
+ captured: Map<SemanticRole, SemanticValue>
214
+ ): boolean {
215
+ // Skip leading conjunctions for Arabic (proclitics: و, ف, ول, وب, etc.)
216
+ // BUT NOT if the pattern explicitly expects a conjunction (proclitic patterns)
217
+ const firstPatternToken = patternTokens[0];
218
+ const patternExpectsConjunction =
219
+ firstPatternToken?.type === 'literal' &&
220
+ (firstPatternToken.value === 'and' ||
221
+ firstPatternToken.value === 'then' ||
222
+ firstPatternToken.alternatives?.includes('and') ||
223
+ firstPatternToken.alternatives?.includes('then'));
224
+
225
+ if (this.currentProfile?.code === 'ar' && !patternExpectsConjunction) {
226
+ while (tokens.peek()?.kind === 'conjunction') {
227
+ tokens.advance();
228
+ }
229
+ }
230
+
231
+ // Backtracking state: track the most recent optional role that consumed a token
232
+ let prevOptionalMark: ReturnType<TokenStream['mark']> | null = null;
233
+ let prevOptionalRole: SemanticRole | null = null;
234
+
235
+ for (let i = 0; i < patternTokens.length; i++) {
236
+ const patternToken = patternTokens[i];
237
+ this.logger.debug(' >> Matching pattern token:', JSON.stringify(patternToken, null, 2));
238
+ const currTok = tokens.peek();
239
+ this.logger.debug(
240
+ ' >> Current input token:',
241
+ currTok
242
+ ? JSON.stringify({
243
+ type: (currTok as any).type,
244
+ value: (currTok as any).value,
245
+ kind: (currTok as any).kind,
246
+ })
247
+ : 'EOF'
248
+ );
249
+
250
+ // Greedy role capture: consume all remaining tokens until the next
251
+ // recognized marker keyword or end of input
252
+ if (patternToken.type === 'role' && patternToken.greedy) {
253
+ const stopMarkers = this.collectStopMarkers(patternTokens, i + 1);
254
+ const values: string[] = [];
255
+ while (!tokens.isAtEnd()) {
256
+ const nextToken = tokens.peek();
257
+ if (!nextToken) break;
258
+ if (this.isStopMarker(nextToken, stopMarkers)) break;
259
+ values.push(nextToken.value);
260
+ tokens.advance();
261
+ }
262
+ if (values.length > 0) {
263
+ captured.set(patternToken.role, { type: 'expression', raw: values.join(' ') });
264
+ prevOptionalMark = null;
265
+ prevOptionalRole = null;
266
+ continue;
267
+ } else if (patternToken.optional) {
268
+ continue;
269
+ } else {
270
+ return false;
271
+ }
272
+ }
273
+
274
+ // Save stream position before attempting optional roles
275
+ const isOptionalRole = patternToken.type === 'role' && patternToken.optional === true;
276
+ const markBefore = isOptionalRole ? tokens.mark() : null;
277
+
278
+ const matched = this.matchPatternToken(tokens, patternToken, captured);
279
+ this.logger.debug(' >> Match result:', matched);
280
+
281
+ if (matched) {
282
+ if (isOptionalRole) {
283
+ // Track this consumption so we can undo it if the next token fails
284
+ prevOptionalMark = markBefore;
285
+ prevOptionalRole = patternToken.role;
286
+ } else {
287
+ // Non-optional succeeded — clear backtrack state
288
+ prevOptionalMark = null;
289
+ prevOptionalRole = null;
290
+ }
291
+ continue;
292
+ }
293
+
294
+ // Match failed
295
+ this.logger.debug(' >> Token match FAILED');
296
+
297
+ if (this.isOptional(patternToken)) {
298
+ continue;
299
+ }
300
+
301
+ // Required token failed — try backtracking over the previous optional role
302
+ if (prevOptionalMark && prevOptionalRole) {
303
+ this.logger.debug(' >> BACKTRACKING: undoing optional role', prevOptionalRole);
304
+ tokens.reset(prevOptionalMark);
305
+ captured.delete(prevOptionalRole);
306
+ prevOptionalMark = null;
307
+ prevOptionalRole = null;
308
+
309
+ // Retry the current (failed) pattern token from the restored position
310
+ const retryMatched = this.matchPatternToken(tokens, patternToken, captured);
311
+ this.logger.debug(' >> Backtrack retry result:', retryMatched);
312
+ if (retryMatched) {
313
+ continue;
314
+ }
315
+ }
316
+
317
+ return false;
318
+ }
319
+
320
+ return true;
321
+ }
322
+
323
+ /**
324
+ * Match a single pattern token against the current position in the stream.
325
+ */
326
+ private matchPatternToken(
327
+ tokens: TokenStream,
328
+ patternToken: PatternToken,
329
+ captured: Map<SemanticRole, SemanticValue>
330
+ ): boolean {
331
+ switch (patternToken.type) {
332
+ case 'literal':
333
+ return this.matchLiteralToken(tokens, patternToken);
334
+
335
+ case 'role':
336
+ return this.matchRoleToken(tokens, patternToken, captured);
337
+
338
+ case 'group':
339
+ return this.matchGroupToken(tokens, patternToken, captured);
340
+
341
+ default:
342
+ return false;
343
+ }
344
+ }
345
+
346
+ /**
347
+ * Match a literal pattern token (keyword or particle).
348
+ */
349
+ private matchLiteralToken(
350
+ tokens: TokenStream,
351
+ patternToken: PatternToken & { type: 'literal' }
352
+ ): boolean {
353
+ const token = tokens.peek();
354
+ this.logger.debug(' >>> matchLiteralToken: expecting', patternToken.value);
355
+ this.logger.debug(
356
+ ' >>> matchLiteralToken: got token',
357
+ token
358
+ ? JSON.stringify({
359
+ type: (token as any).type,
360
+ value: (token as any).value,
361
+ kind: (token as any).kind,
362
+ })
363
+ : 'null'
364
+ );
365
+ if (!token) {
366
+ this.logger.debug(' >>> matchLiteralToken: FAIL - no token');
367
+ return false;
368
+ }
369
+
370
+ // Check main value
371
+ const matchType = this.getMatchType(token, patternToken.value);
372
+ this.logger.debug(
373
+ ' >>> matchType for',
374
+ token.value,
375
+ 'vs',
376
+ patternToken.value,
377
+ ':',
378
+ matchType
379
+ );
380
+ if (matchType !== 'none') {
381
+ this.totalKeywordMatches++;
382
+ if (matchType === 'stem') {
383
+ this.stemMatchCount++;
384
+ }
385
+ tokens.advance();
386
+ return true;
387
+ }
388
+
389
+ // Check alternatives
390
+ if (patternToken.alternatives) {
391
+ for (const alt of patternToken.alternatives) {
392
+ const altMatchType = this.getMatchType(token, alt);
393
+ if (altMatchType !== 'none') {
394
+ this.totalKeywordMatches++;
395
+ if (altMatchType === 'stem') {
396
+ this.stemMatchCount++;
397
+ }
398
+ tokens.advance();
399
+ return true;
400
+ }
401
+ }
402
+ }
403
+
404
+ return false;
405
+ }
406
+
407
+ /**
408
+ * Match a role pattern token (captures a semantic value).
409
+ * Handles multi-token expressions like:
410
+ * - 'my value' (possessive keyword + property)
411
+ * - '#dialog.showModal()' (method call)
412
+ * - "#element's *opacity" (possessive selector + property)
413
+ */
414
+ private matchRoleToken(
415
+ tokens: TokenStream,
416
+ patternToken: PatternToken & { type: 'role' },
417
+ captured: Map<SemanticRole, SemanticValue>
418
+ ): boolean {
419
+ this.logger.debug(' >>> matchRoleToken ENTRY: capturing role', patternToken.role);
420
+ this.logger.debug(' >>> matchRoleToken: expected types', patternToken.expectedTypes);
421
+ this.logger.debug(' >>> matchRoleToken: optional?', patternToken.optional);
422
+ // Skip noise words like "the" before selectors (English idiom support)
423
+ this.skipNoiseWords(tokens);
424
+
425
+ const token = tokens.peek();
426
+ this.logger.debug(
427
+ ' >>> After skipNoiseWords, current token:',
428
+ token ? JSON.stringify({ value: (token as any).value, kind: (token as any).kind }) : 'null'
429
+ );
430
+ if (!token) {
431
+ return patternToken.optional || false;
432
+ }
433
+
434
+ // Check for possessive expression (e.g., 'my value', 'its innerHTML')
435
+ const possessiveValue = this.tryMatchPossessiveExpression(tokens);
436
+ if (possessiveValue) {
437
+ // Validate expected types if specified
438
+ if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
439
+ if (
440
+ !patternToken.expectedTypes.includes(possessiveValue.type) &&
441
+ !patternToken.expectedTypes.includes('expression')
442
+ ) {
443
+ return patternToken.optional || false;
444
+ }
445
+ }
446
+ captured.set(patternToken.role, possessiveValue);
447
+ return true;
448
+ }
449
+
450
+ // Check for method call expression (e.g., '#dialog.showModal()')
451
+ const methodCallValue = this.tryMatchMethodCallExpression(tokens);
452
+ if (methodCallValue) {
453
+ if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
454
+ if (
455
+ !patternToken.expectedTypes.includes(methodCallValue.type) &&
456
+ !patternToken.expectedTypes.includes('expression')
457
+ ) {
458
+ return patternToken.optional || false;
459
+ }
460
+ }
461
+ captured.set(patternToken.role, methodCallValue);
462
+ return true;
463
+ }
464
+
465
+ // Check for possessive selector expression (e.g., "#element's *opacity")
466
+ const possessiveSelectorValue = this.tryMatchPossessiveSelectorExpression(tokens);
467
+ if (possessiveSelectorValue) {
468
+ if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
469
+ // property-path is compatible with selector, reference, and expression
470
+ if (!isTypeCompatible(possessiveSelectorValue.type, patternToken.expectedTypes)) {
471
+ return patternToken.optional || false;
472
+ }
473
+ }
474
+ captured.set(patternToken.role, possessiveSelectorValue);
475
+ return true;
476
+ }
477
+
478
+ // Check for property access expression (e.g., 'userData.name', 'it.data')
479
+ const propertyAccessValue = this.tryMatchPropertyAccessExpression(tokens);
480
+ if (propertyAccessValue) {
481
+ if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
482
+ if (
483
+ !patternToken.expectedTypes.includes(propertyAccessValue.type) &&
484
+ !patternToken.expectedTypes.includes('expression')
485
+ ) {
486
+ return patternToken.optional || false;
487
+ }
488
+ }
489
+ captured.set(patternToken.role, propertyAccessValue);
490
+ return true;
491
+ }
492
+
493
+ // Check for selector + property expression (e.g., '#output.innerText')
494
+ // This handles cases where the tokenizer produces two selector tokens
495
+ const selectorPropertyValue = this.tryMatchSelectorPropertyExpression(tokens);
496
+ if (selectorPropertyValue) {
497
+ if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
498
+ if (!isTypeCompatible(selectorPropertyValue.type, patternToken.expectedTypes)) {
499
+ return patternToken.optional || false;
500
+ }
501
+ }
502
+ captured.set(patternToken.role, selectorPropertyValue);
503
+ return true;
504
+ }
505
+
506
+ // Try to extract a semantic value from the token
507
+ this.logger.debug(
508
+ ' >>> Trying tokenToSemanticValue for token:',
509
+ token ? JSON.stringify({ value: (token as any).value, kind: (token as any).kind }) : 'null'
510
+ );
511
+ const value = this.tokenToSemanticValue(token);
512
+ this.logger.debug(
513
+ ' >>> tokenToSemanticValue returned:',
514
+ value ? JSON.stringify(value) : 'null'
515
+ );
516
+ if (!value) {
517
+ return patternToken.optional || false;
518
+ }
519
+
520
+ // Validate expected types if specified
521
+ this.logger.debug(
522
+ ' >>> Validating type:',
523
+ value.type,
524
+ 'against expected:',
525
+ patternToken.expectedTypes
526
+ );
527
+ if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
528
+ if (!isTypeCompatible(value.type, patternToken.expectedTypes)) {
529
+ this.logger.debug(' >>> TYPE MISMATCH - returning', patternToken.optional || false);
530
+ return patternToken.optional || false;
531
+ }
532
+ }
533
+ this.logger.debug(' >>> Type validation PASSED');
534
+
535
+ captured.set(patternToken.role, value);
536
+ tokens.advance();
537
+ return true;
538
+ }
539
+
540
+ /**
541
+ * Try to match a possessive expression like 'my value' or 'its innerHTML'.
542
+ * Returns the PropertyPathValue if matched, or null if not.
543
+ */
544
+ private tryMatchPossessiveExpression(tokens: TokenStream): SemanticValue | null {
545
+ const token = tokens.peek();
546
+ if (!token) return null;
547
+
548
+ // Use profile-based possessive keyword lookup
549
+ if (!this.currentProfile) return null;
550
+
551
+ const tokenValue = token.normalized || token.value;
552
+ const tokenLower = this.safeToLowerCase(tokenValue);
553
+ const baseRef = getPossessiveReference(this.currentProfile, tokenLower);
554
+
555
+ if (!baseRef) return null;
556
+
557
+ // We have a possessive keyword, look ahead for property name
558
+ const mark = tokens.mark();
559
+ tokens.advance();
560
+
561
+ const propertyToken = tokens.peek();
562
+ if (!propertyToken) {
563
+ // Just the possessive keyword, no property - revert
564
+ tokens.reset(mark);
565
+ return null;
566
+ }
567
+
568
+ // Property should be an identifier, keyword (not structural), or selector (for style/dot/attr)
569
+ // Examples: "my value", "my innerHTML", "my *background", "my *opacity", "my @data-count"
570
+ // Also handles dot-property access: "my.textContent" tokenized as "my" + ".textContent"
571
+ if (
572
+ propertyToken.kind === 'identifier' ||
573
+ (propertyToken.kind === 'keyword' && !this.isStructuralKeyword(propertyToken.value)) ||
574
+ (propertyToken.kind === 'selector' && propertyToken.value.startsWith('*')) ||
575
+ (propertyToken.kind === 'selector' && propertyToken.value.startsWith('@')) ||
576
+ (propertyToken.kind === 'selector' &&
577
+ propertyToken.value.startsWith('.') &&
578
+ /^\.[a-zA-Z_]\w*/.test(propertyToken.value))
579
+ ) {
580
+ tokens.advance();
581
+
582
+ // For dot-property selectors (.textContent), strip the leading dot
583
+ let propertyName = propertyToken.value;
584
+ if (
585
+ propertyToken.kind === 'selector' &&
586
+ propertyName.startsWith('.') &&
587
+ /^\.[a-zA-Z_]\w*/.test(propertyName)
588
+ ) {
589
+ propertyName = propertyName.substring(1);
590
+ }
591
+
592
+ // Consume chained dot-property access (.parentElement.style.display)
593
+ let chainedProps = propertyName;
594
+ while (
595
+ tokens.peek()?.kind === 'selector' &&
596
+ tokens.peek()!.value.startsWith('.') &&
597
+ /^\.[a-zA-Z_]\w*/.test(tokens.peek()!.value)
598
+ ) {
599
+ chainedProps += tokens.peek()!.value; // keep the dots for chaining
600
+ tokens.advance();
601
+ }
602
+
603
+ // Check for method call — next token is '(' in the value (e.g., .getAttribute("data-id"))
604
+ const nextPeek = tokens.peek();
605
+ if (nextPeek?.kind === 'literal' && nextPeek.value.startsWith('(')) {
606
+ // Consume method args
607
+ chainedProps += nextPeek.value;
608
+ tokens.advance();
609
+ }
610
+
611
+ // Create property-path: my value -> { object: me, property: 'value' }
612
+ // baseRef from getPossessiveReference is always a valid reference ('me', 'you', 'it', etc.)
613
+ return createPropertyPath(createReference(baseRef as ReferenceValue['value']), chainedProps);
614
+ }
615
+
616
+ // Not a valid property, revert
617
+ tokens.reset(mark);
618
+ return null;
619
+ }
620
+
621
+ /**
622
+ * Check if a keyword is a structural keyword (preposition, control flow, etc.)
623
+ * that shouldn't be consumed as a property name.
624
+ */
625
+ private isStructuralKeyword(value: string): boolean {
626
+ const structural = new Set([
627
+ // Prepositions
628
+ 'into',
629
+ 'in',
630
+ 'to',
631
+ 'from',
632
+ 'at',
633
+ 'by',
634
+ 'with',
635
+ 'without',
636
+ 'before',
637
+ 'after',
638
+ 'of',
639
+ 'as',
640
+ 'on',
641
+ // Control flow
642
+ 'then',
643
+ 'end',
644
+ 'else',
645
+ 'if',
646
+ 'repeat',
647
+ 'while',
648
+ 'for',
649
+ // Commands (shouldn't be property names)
650
+ 'toggle',
651
+ 'add',
652
+ 'remove',
653
+ 'put',
654
+ 'set',
655
+ 'show',
656
+ 'hide',
657
+ 'increment',
658
+ 'decrement',
659
+ 'send',
660
+ 'trigger',
661
+ 'call',
662
+ ]);
663
+ return structural.has(value.toLowerCase());
664
+ }
665
+
666
+ /**
667
+ * Try to match a method call expression like '#dialog.showModal()'.
668
+ * Pattern: selector + '.' + identifier + '(' + [args] + ')'
669
+ * Returns an expression value if matched, or null if not.
670
+ */
671
+ private tryMatchMethodCallExpression(tokens: TokenStream): SemanticValue | null {
672
+ const token = tokens.peek();
673
+ if (!token || token.kind !== 'selector') return null;
674
+
675
+ // Look ahead for: . identifier (
676
+ const mark = tokens.mark();
677
+ tokens.advance(); // consume selector
678
+
679
+ const dotToken = tokens.peek();
680
+ if (!dotToken || dotToken.kind !== 'operator' || dotToken.value !== '.') {
681
+ tokens.reset(mark);
682
+ return null;
683
+ }
684
+ tokens.advance(); // consume .
685
+
686
+ const methodToken = tokens.peek();
687
+ if (!methodToken || methodToken.kind !== 'identifier') {
688
+ tokens.reset(mark);
689
+ return null;
690
+ }
691
+ tokens.advance(); // consume method name
692
+
693
+ const openParen = tokens.peek();
694
+ if (!openParen || openParen.kind !== 'punctuation' || openParen.value !== '(') {
695
+ tokens.reset(mark);
696
+ return null;
697
+ }
698
+ tokens.advance(); // consume (
699
+
700
+ // Consume arguments until we find ) (with depth limit for security)
701
+ const args: string[] = [];
702
+ while (!tokens.isAtEnd() && args.length < PatternMatcher.MAX_METHOD_ARGS) {
703
+ const argToken = tokens.peek();
704
+ if (!argToken) break;
705
+ if (argToken.kind === 'punctuation' && argToken.value === ')') {
706
+ tokens.advance(); // consume )
707
+ break;
708
+ }
709
+ // Skip commas
710
+ if (argToken.kind === 'punctuation' && argToken.value === ',') {
711
+ tokens.advance();
712
+ continue;
713
+ }
714
+ // Collect arg value
715
+ args.push(argToken.value);
716
+ tokens.advance();
717
+ }
718
+
719
+ // Create expression value: #dialog.showModal()
720
+ const methodCall = `${token.value}.${methodToken.value}(${args.join(', ')})`;
721
+ return {
722
+ type: 'expression',
723
+ raw: methodCall,
724
+ } as SemanticValue;
725
+ }
726
+
727
+ /**
728
+ * Try to match a property access expression like 'userData.name' or 'it.data'.
729
+ * Pattern: (identifier | keyword) + '.' + identifier [+ '.' + identifier ...]
730
+ * Returns an expression value if matched, or null if not.
731
+ */
732
+ private tryMatchPropertyAccessExpression(tokens: TokenStream): SemanticValue | null {
733
+ const token = tokens.peek();
734
+ if (!token) return null;
735
+
736
+ // Must start with an identifier or keyword reference
737
+ if (token.kind !== 'identifier' && token.kind !== 'keyword') return null;
738
+
739
+ // Look ahead for: . identifier
740
+ const mark = tokens.mark();
741
+ tokens.advance(); // consume first token
742
+
743
+ const dotToken = tokens.peek();
744
+ if (!dotToken || dotToken.kind !== 'operator' || dotToken.value !== '.') {
745
+ tokens.reset(mark);
746
+ return null;
747
+ }
748
+ tokens.advance(); // consume .
749
+
750
+ const propertyToken = tokens.peek();
751
+ if (!propertyToken || propertyToken.kind !== 'identifier') {
752
+ tokens.reset(mark);
753
+ return null;
754
+ }
755
+ tokens.advance(); // consume property name
756
+
757
+ // Build the property chain
758
+ let chain = `${token.value}.${propertyToken.value}`;
759
+ let depth = 1; // Already have one property access
760
+
761
+ // Continue for nested property access (e.g., userData.address.city)
762
+ // With depth limit for security
763
+ while (!tokens.isAtEnd() && depth < PatternMatcher.MAX_PROPERTY_DEPTH) {
764
+ const nextDot = tokens.peek();
765
+ if (!nextDot || nextDot.kind !== 'operator' || nextDot.value !== '.') {
766
+ break;
767
+ }
768
+ tokens.advance(); // consume .
769
+
770
+ const nextProp = tokens.peek();
771
+ if (!nextProp || nextProp.kind !== 'identifier') {
772
+ // Dot without property - put the dot back and stop
773
+ // Can't easily put a single token back, so we'll include it
774
+ break;
775
+ }
776
+ tokens.advance(); // consume property
777
+ chain += `.${nextProp.value}`;
778
+ depth++;
779
+ }
780
+
781
+ // Check for method call: chain + '(' + args + ')'
782
+ // e.g., me.insertBefore(draggedItem, dropTarget)
783
+ const openParen = tokens.peek();
784
+ if (openParen && openParen.kind === 'punctuation' && openParen.value === '(') {
785
+ tokens.advance(); // consume (
786
+
787
+ // Collect arguments (comma-separated values)
788
+ const args: string[] = [];
789
+ let argDepth = 0; // Track nested parentheses
790
+ while (!tokens.isAtEnd() && args.length < PatternMatcher.MAX_METHOD_ARGS) {
791
+ const argToken = tokens.peek();
792
+ if (!argToken) break;
793
+
794
+ // Handle close paren - respecting nesting
795
+ if (argToken.kind === 'punctuation' && argToken.value === ')') {
796
+ if (argDepth === 0) {
797
+ tokens.advance(); // consume )
798
+ break;
799
+ }
800
+ argDepth--;
801
+ }
802
+ // Track nested open parens
803
+ if (argToken.kind === 'punctuation' && argToken.value === '(') {
804
+ argDepth++;
805
+ }
806
+ // Skip commas between arguments
807
+ if (argToken.kind === 'punctuation' && argToken.value === ',') {
808
+ tokens.advance();
809
+ continue;
810
+ }
811
+ // Collect arg value
812
+ args.push(argToken.value);
813
+ tokens.advance();
814
+ }
815
+
816
+ // Create expression value with method call: me.insertBefore(a, b)
817
+ const methodCall = `${chain}(${args.join(', ')})`;
818
+ return {
819
+ type: 'expression',
820
+ raw: methodCall,
821
+ } as SemanticValue;
822
+ }
823
+
824
+ // Create expression value: userData.name
825
+ return {
826
+ type: 'expression',
827
+ raw: chain,
828
+ } as SemanticValue;
829
+ }
830
+
831
+ /**
832
+ * Try to match a possessive selector expression like "#element's *opacity".
833
+ * Pattern: selector + "'s" + (selector | identifier)
834
+ * Returns a property-path value if matched, or null if not.
835
+ */
836
+ private tryMatchPossessiveSelectorExpression(tokens: TokenStream): SemanticValue | null {
837
+ const token = tokens.peek();
838
+ if (!token || token.kind !== 'selector') return null;
839
+
840
+ // Look ahead for: 's (possessive marker)
841
+ const mark = tokens.mark();
842
+ tokens.advance(); // consume selector
843
+
844
+ const possessiveToken = tokens.peek();
845
+ if (
846
+ !possessiveToken ||
847
+ possessiveToken.kind !== 'punctuation' ||
848
+ possessiveToken.value !== "'s"
849
+ ) {
850
+ tokens.reset(mark);
851
+ return null;
852
+ }
853
+ tokens.advance(); // consume 's
854
+
855
+ const propertyToken = tokens.peek();
856
+ if (!propertyToken) {
857
+ tokens.reset(mark);
858
+ return null;
859
+ }
860
+
861
+ // Property can be a selector (*opacity) or identifier
862
+ if (propertyToken.kind !== 'selector' && propertyToken.kind !== 'identifier') {
863
+ tokens.reset(mark);
864
+ return null;
865
+ }
866
+ tokens.advance(); // consume property
867
+
868
+ // Create property-path: #element's *opacity
869
+ return createPropertyPath(createSelector(token.value), propertyToken.value);
870
+ }
871
+
872
+ /**
873
+ * Try to match a selector + property expression like "#output.innerText".
874
+ * This handles cases where the tokenizer produces two selector tokens:
875
+ * - #output (id selector)
876
+ * - .innerText (looks like class selector, but is actually property)
877
+ *
878
+ * Pattern: id-selector + class-selector-that-is-actually-property
879
+ * Returns a property-path value if matched, or null if not.
880
+ */
881
+ private tryMatchSelectorPropertyExpression(tokens: TokenStream): SemanticValue | null {
882
+ const token = tokens.peek();
883
+ if (!token || token.kind !== 'selector') return null;
884
+
885
+ // Must be an ID selector (starts with #)
886
+ if (typeof token.value !== 'string' || !token.value.startsWith('#')) return null;
887
+
888
+ // Look ahead for: selector that looks like a property (.something)
889
+ const mark = tokens.mark();
890
+ tokens.advance(); // consume first selector
891
+
892
+ const propertyToken = tokens.peek();
893
+ if (!propertyToken || propertyToken.kind !== 'selector') {
894
+ tokens.reset(mark);
895
+ return null;
896
+ }
897
+
898
+ // Second token must look like a class selector (starts with .)
899
+ // but we interpret it as a property access
900
+ if (!propertyToken.value.startsWith('.')) {
901
+ tokens.reset(mark);
902
+ return null;
903
+ }
904
+
905
+ // Verify the next token is not a selector (to avoid consuming too many)
906
+ // This helps distinguish "#output.innerText" from "#box .child"
907
+ const peek2 = tokens.peek(1);
908
+ if (peek2 && peek2.kind === 'selector') {
909
+ // Could be a compound selector chain - only take first two
910
+ }
911
+
912
+ tokens.advance(); // consume property selector
913
+
914
+ // Create property-path: #output.innerText
915
+ // Extract property name without the leading dot
916
+ const propertyName = propertyToken.value.slice(1);
917
+
918
+ return createPropertyPath(createSelector(token.value), propertyName);
919
+ }
920
+
921
+ /**
922
+ * Match a group pattern token (optional sequence).
923
+ * When the group's leading marker isn't at the current position, scans ahead
924
+ * up to MAX_MARKER_SCAN tokens to find it. This allows unmarked tokens
925
+ * (e.g., a value like 'hello') to sit between groups without blocking later
926
+ * marker matches.
927
+ */
928
+ private matchGroupToken(
929
+ tokens: TokenStream,
930
+ patternToken: PatternToken & { type: 'group' },
931
+ captured: Map<SemanticRole, SemanticValue>
932
+ ): boolean {
933
+ const mark = tokens.mark();
934
+ const capturedBefore = new Set(captured.keys());
935
+
936
+ const success = this.matchTokenSequence(tokens, patternToken.tokens, captured);
937
+ if (success) return true;
938
+
939
+ // Reset from failed attempt
940
+ tokens.reset(mark);
941
+ for (const role of captured.keys()) {
942
+ if (!capturedBefore.has(role)) captured.delete(role);
943
+ }
944
+
945
+ if (!patternToken.optional) return false;
946
+
947
+ // Marker scan: look ahead for the group's leading marker past intervening tokens.
948
+ // This handles cases like "types 'hello' into #search" where 'hello' blocks the
949
+ // 'into' marker from being found at the current position.
950
+ const leadingMarker = this.getGroupLeadingMarker(patternToken);
951
+ if (leadingMarker) {
952
+ for (let offset = 1; offset <= PatternMatcher.MAX_MARKER_SCAN; offset++) {
953
+ const ahead = tokens.peek(offset);
954
+ if (!ahead) break;
955
+ const aheadValue = (ahead.normalized || ahead.value).toLowerCase();
956
+ if (aheadValue === leadingMarker) {
957
+ this.logger.debug(
958
+ ' >> MARKER SCAN: found',
959
+ leadingMarker,
960
+ 'at offset',
961
+ offset,
962
+ '- skipping intervening tokens'
963
+ );
964
+ // Advance past intervening tokens to reach the marker
965
+ for (let s = 0; s < offset; s++) tokens.advance();
966
+
967
+ // Retry group match from marker position
968
+ const retrySuccess = this.matchTokenSequence(tokens, patternToken.tokens, captured);
969
+ if (retrySuccess) return true;
970
+
971
+ // Retry failed — full reset
972
+ tokens.reset(mark);
973
+ for (const role of captured.keys()) {
974
+ if (!capturedBefore.has(role)) captured.delete(role);
975
+ }
976
+ break;
977
+ }
978
+ }
979
+ }
980
+
981
+ return true; // Optional group, just skip
982
+ }
983
+
984
+ /**
985
+ * Get the leading marker literal from an optional group.
986
+ * Returns the lowercase marker value, or null if the group
987
+ * doesn't start with a literal token (e.g., SOV groups where
988
+ * the role comes before the marker).
989
+ */
990
+ private getGroupLeadingMarker(group: PatternToken & { type: 'group' }): string | null {
991
+ const first = group.tokens[0];
992
+ if (first?.type === 'literal') return first.value.toLowerCase();
993
+ return null;
994
+ }
995
+
996
+ /**
997
+ * Get the type of match for a token against a value.
998
+ * Used for confidence calculation.
999
+ */
1000
+ private getMatchType(
1001
+ token: LanguageToken,
1002
+ value: string
1003
+ ): 'exact' | 'normalized' | 'stem' | 'case-insensitive' | 'none' {
1004
+ // Exact match (highest confidence)
1005
+ if (token.value === value) return 'exact';
1006
+
1007
+ // Explicit keyword map normalized match (high confidence)
1008
+ if (token.normalized === value) return 'normalized';
1009
+
1010
+ // Morphologically normalized stem match (medium-high confidence)
1011
+ // Only accept if stem confidence is reasonable
1012
+ if (token.stem === value && token.stemConfidence !== undefined && token.stemConfidence >= 0.7) {
1013
+ return 'stem';
1014
+ }
1015
+
1016
+ // Case-insensitive match for keywords (medium confidence)
1017
+ if (token.kind === 'keyword' && this.safeToLowerCase(token.value) === value.toLowerCase()) {
1018
+ return 'case-insensitive';
1019
+ }
1020
+
1021
+ return 'none';
1022
+ }
1023
+
1024
+ /**
1025
+ * Collect literal values from upcoming pattern tokens that act as stop markers
1026
+ * for greedy role capture. Returns the set of lowercase values.
1027
+ */
1028
+ private collectStopMarkers(patternTokens: PatternToken[], startIndex: number): Set<string> {
1029
+ const markers = new Set<string>();
1030
+ for (let j = startIndex; j < patternTokens.length; j++) {
1031
+ const pt = patternTokens[j];
1032
+ if (pt.type === 'literal') {
1033
+ markers.add(pt.value.toLowerCase());
1034
+ if (pt.alternatives) {
1035
+ for (const alt of pt.alternatives) {
1036
+ markers.add(alt.toLowerCase());
1037
+ }
1038
+ }
1039
+ break;
1040
+ }
1041
+ if (pt.type === 'group') {
1042
+ for (const gt of pt.tokens) {
1043
+ if (gt.type === 'literal') {
1044
+ markers.add(gt.value.toLowerCase());
1045
+ if (gt.alternatives) {
1046
+ for (const alt of gt.alternatives) {
1047
+ markers.add(alt.toLowerCase());
1048
+ }
1049
+ }
1050
+ break;
1051
+ }
1052
+ }
1053
+ break;
1054
+ }
1055
+ }
1056
+ return markers;
1057
+ }
1058
+
1059
+ /**
1060
+ * Check if a token matches any stop marker for greedy capture.
1061
+ */
1062
+ private isStopMarker(token: LanguageToken, stopMarkers: Set<string>): boolean {
1063
+ if (stopMarkers.size === 0) return false;
1064
+ const value = (token.normalized || token.value).toLowerCase();
1065
+ return stopMarkers.has(value);
1066
+ }
1067
+
1068
+ /**
1069
+ * Track stem matches for confidence calculation.
1070
+ * This is set during matching and read during confidence calculation.
1071
+ */
1072
+ private stemMatchCount: number = 0;
1073
+ private totalKeywordMatches: number = 0;
1074
+
1075
+ // ==========================================================================
1076
+ // Depth Limits for Expression Parsing (security hardening)
1077
+ // ==========================================================================
1078
+
1079
+ /** Maximum depth for nested property access (e.g., a.b.c.d...) */
1080
+ private static readonly MAX_PROPERTY_DEPTH = 10;
1081
+
1082
+ /** Maximum number of arguments in method calls */
1083
+ private static readonly MAX_METHOD_ARGS = 20;
1084
+
1085
+ /**
1086
+ * Convert a language token to a semantic value.
1087
+ */
1088
+ private tokenToSemanticValue(token: LanguageToken): SemanticValue | null {
1089
+ switch (token.kind) {
1090
+ case 'selector':
1091
+ return createSelector(token.value);
1092
+
1093
+ case 'literal':
1094
+ return this.parseLiteralValue(token.value);
1095
+
1096
+ case 'keyword':
1097
+ // Keywords might be references or values
1098
+ const tokenValue = token.normalized || token.value;
1099
+ const lower = this.safeToLowerCase(tokenValue);
1100
+ if (isValidReference(lower)) {
1101
+ return createReference(lower);
1102
+ }
1103
+ return createLiteral(token.normalized || token.value);
1104
+
1105
+ case 'identifier':
1106
+ // Check if it's a variable reference (:varname)
1107
+ // Note: :varname doesn't match the ReferenceValue union but is used as a
1108
+ // reference token downstream — this cast preserves existing behavior
1109
+ if (typeof token.value === 'string' && token.value.startsWith(':')) {
1110
+ return createReference(token.value as ReferenceValue['value']);
1111
+ }
1112
+ // Check if it's a built-in reference
1113
+ const identLower = this.safeToLowerCase(token.value);
1114
+ if (isValidReference(identLower)) {
1115
+ return createReference(identLower);
1116
+ }
1117
+ // Regular identifiers are variable references - use 'expression' type
1118
+ // which gets converted to 'identifier' AST nodes by semantic-integration.ts
1119
+ return { type: 'expression', raw: token.value } as const;
1120
+
1121
+ case 'url':
1122
+ // URLs are treated as string literals (paths/URLs for navigation/fetch)
1123
+ return createLiteral(token.value, 'string');
1124
+
1125
+ default:
1126
+ return null;
1127
+ }
1128
+ }
1129
+
1130
+ /**
1131
+ * Parse a literal value (string, number, boolean).
1132
+ */
1133
+ private parseLiteralValue(value: string): SemanticValue {
1134
+ // String literal
1135
+ if (
1136
+ value.startsWith('"') ||
1137
+ value.startsWith("'") ||
1138
+ value.startsWith('`') ||
1139
+ value.startsWith('「')
1140
+ ) {
1141
+ const inner = value.slice(1, -1);
1142
+ return createLiteral(inner, 'string');
1143
+ }
1144
+
1145
+ // Boolean
1146
+ if (value === 'true') return createLiteral(true, 'boolean');
1147
+ if (value === 'false') return createLiteral(false, 'boolean');
1148
+
1149
+ // Duration (number with suffix)
1150
+ const durationMatch = value.match(/^(\d+(?:\.\d+)?)(ms|s|m|h)?$/);
1151
+ if (durationMatch) {
1152
+ const num = parseFloat(durationMatch[1]);
1153
+ const unit = durationMatch[2];
1154
+ if (unit) {
1155
+ return createLiteral(value, 'duration');
1156
+ }
1157
+ return createLiteral(num, 'number');
1158
+ }
1159
+
1160
+ // Plain number
1161
+ const num = parseFloat(value);
1162
+ if (!isNaN(num)) {
1163
+ return createLiteral(num, 'number');
1164
+ }
1165
+
1166
+ // Default to string
1167
+ return createLiteral(value, 'string');
1168
+ }
1169
+
1170
+ /**
1171
+ * Apply extraction rules to fill in static values and defaults for missing roles.
1172
+ */
1173
+ private applyExtractionRules(
1174
+ pattern: LanguagePattern,
1175
+ captured: Map<SemanticRole, SemanticValue>
1176
+ ): void {
1177
+ for (const [role, rule] of Object.entries(pattern.extraction)) {
1178
+ if (!captured.has(role as SemanticRole)) {
1179
+ if (rule.value !== undefined) {
1180
+ // Static value extraction (e.g., action: { value: "toggle" })
1181
+ captured.set(role as SemanticRole, { type: 'literal', value: rule.value });
1182
+ } else if (rule.default) {
1183
+ captured.set(role as SemanticRole, rule.default);
1184
+ }
1185
+ }
1186
+ }
1187
+ }
1188
+
1189
+ /**
1190
+ * Check if a pattern token is optional.
1191
+ */
1192
+ private isOptional(patternToken: PatternToken): boolean {
1193
+ return patternToken.type !== 'literal' && patternToken.optional === true;
1194
+ }
1195
+
1196
+ /**
1197
+ * Calculate confidence score for a match (0-1).
1198
+ *
1199
+ * Confidence is reduced for:
1200
+ * - Stem matches (morphological normalization has inherent uncertainty)
1201
+ * - Missing optional roles (but less penalty if role has a default value)
1202
+ *
1203
+ * Confidence is increased for:
1204
+ * - VSO languages (Arabic) when pattern starts with a verb
1205
+ */
1206
+ private calculateConfidence(
1207
+ pattern: LanguagePattern,
1208
+ captured: Map<SemanticRole, SemanticValue>
1209
+ ): number {
1210
+ let score = 0;
1211
+ let maxScore = 0;
1212
+
1213
+ // Helper to check if a role has a default value in extraction rules
1214
+ const hasDefault = (role: SemanticRole): boolean => {
1215
+ return pattern.extraction?.[role]?.default !== undefined;
1216
+ };
1217
+
1218
+ // Score based on captured roles
1219
+ for (const token of pattern.template.tokens) {
1220
+ if (token.type === 'role') {
1221
+ maxScore += 1;
1222
+ if (captured.has(token.role)) {
1223
+ score += 1;
1224
+ }
1225
+ } else if (token.type === 'group') {
1226
+ // Group tokens are optional - weight depends on whether they have defaults
1227
+ for (const subToken of token.tokens) {
1228
+ if (subToken.type === 'role') {
1229
+ const roleHasDefault = hasDefault(subToken.role);
1230
+ const weight = 0.8; // Optional roles: 80% weight
1231
+ maxScore += weight;
1232
+
1233
+ if (captured.has(subToken.role)) {
1234
+ // Role was explicitly provided by user
1235
+ score += weight;
1236
+ } else if (roleHasDefault) {
1237
+ // Role has a default - give 60% partial credit since command is semantically complete
1238
+ // This prevents penalizing common patterns like "toggle .active" (default: me)
1239
+ score += weight * 0.6;
1240
+ }
1241
+ // If no default and not captured, score += 0 (true penalty for missing info)
1242
+ }
1243
+ }
1244
+ }
1245
+ }
1246
+
1247
+ let baseConfidence = maxScore > 0 ? score / maxScore : 1;
1248
+
1249
+ // Apply penalty for stem matches
1250
+ // Each stem match reduces confidence slightly (e.g., 5% per stem match)
1251
+ // This ensures exact matches are preferred over morphological matches
1252
+ if (this.stemMatchCount > 0 && this.totalKeywordMatches > 0) {
1253
+ const stemPenalty = (this.stemMatchCount / this.totalKeywordMatches) * 0.15;
1254
+ baseConfidence = Math.max(0.5, baseConfidence - stemPenalty);
1255
+ }
1256
+
1257
+ // Apply VSO confidence boost for Arabic verb-first patterns
1258
+ const vsoBoost = this.calculateVSOConfidenceBoost(pattern);
1259
+ baseConfidence = Math.min(1.0, baseConfidence + vsoBoost);
1260
+
1261
+ // Apply preposition disambiguation adjustment for Arabic
1262
+ const prepositionAdjustment = this.arabicPrepositionDisambiguation(pattern, captured);
1263
+ baseConfidence = Math.max(0.0, Math.min(1.0, baseConfidence + prepositionAdjustment));
1264
+
1265
+ return baseConfidence;
1266
+ }
1267
+
1268
+ /**
1269
+ * Calculate confidence boost for VSO (Verb-Subject-Object) language patterns.
1270
+ * Arabic naturally uses VSO word order, so patterns that start with a verb
1271
+ * should receive a confidence boost.
1272
+ *
1273
+ * Returns +0.15 confidence boost if:
1274
+ * - Language is Arabic ('ar')
1275
+ * - Pattern's first token is a verb keyword
1276
+ *
1277
+ * @param pattern The language pattern being matched
1278
+ * @returns Confidence boost (0 or 0.15)
1279
+ */
1280
+ private calculateVSOConfidenceBoost(pattern: LanguagePattern): number {
1281
+ // Only apply to Arabic
1282
+ if (pattern.language !== 'ar') {
1283
+ return 0;
1284
+ }
1285
+
1286
+ // Check if first token in pattern is a literal (keyword)
1287
+ const firstToken = pattern.template.tokens[0];
1288
+ if (!firstToken || firstToken.type !== 'literal') {
1289
+ return 0;
1290
+ }
1291
+
1292
+ // List of Arabic verb keywords (command verbs)
1293
+ const ARABIC_VERBS = new Set([
1294
+ 'بدل',
1295
+ 'غير',
1296
+ 'أضف',
1297
+ 'أزل',
1298
+ 'ضع',
1299
+ 'اجعل',
1300
+ 'عين',
1301
+ 'زد',
1302
+ 'انقص',
1303
+ 'سجل',
1304
+ 'أظهر',
1305
+ 'أخف',
1306
+ 'شغل',
1307
+ 'أرسل',
1308
+ 'ركز',
1309
+ 'شوش',
1310
+ 'توقف',
1311
+ 'انسخ',
1312
+ 'احذف',
1313
+ 'اصنع',
1314
+ 'انتظر',
1315
+ 'انتقال',
1316
+ 'أو',
1317
+ ]);
1318
+
1319
+ // Check if first token value is a verb
1320
+ if (ARABIC_VERBS.has(firstToken.value)) {
1321
+ return 0.15;
1322
+ }
1323
+
1324
+ // Check alternatives
1325
+ if (firstToken.alternatives) {
1326
+ for (const alt of firstToken.alternatives) {
1327
+ if (ARABIC_VERBS.has(alt)) {
1328
+ return 0.15;
1329
+ }
1330
+ }
1331
+ }
1332
+
1333
+ return 0;
1334
+ }
1335
+
1336
+ /**
1337
+ * Arabic preposition disambiguation for confidence adjustment.
1338
+ *
1339
+ * Different Arabic prepositions are more or less natural for different semantic roles:
1340
+ * - على (on/upon) is preferred for patient/target roles (element selectors)
1341
+ * - إلى (to) is preferred for destination roles
1342
+ * - من (from) is preferred for source roles
1343
+ * - في (in) is preferred for location roles
1344
+ *
1345
+ * This method analyzes the prepositions used with captured semantic roles and
1346
+ * adjusts confidence based on idiomaticity:
1347
+ * - +0.10 for highly idiomatic preposition choices
1348
+ * - -0.10 for less natural preposition choices
1349
+ *
1350
+ * @param pattern The language pattern being matched
1351
+ * @param captured The captured semantic values
1352
+ * @returns Confidence adjustment (-0.10 to +0.10)
1353
+ */
1354
+ private arabicPrepositionDisambiguation(
1355
+ pattern: LanguagePattern,
1356
+ captured: Map<SemanticRole, SemanticValue>
1357
+ ): number {
1358
+ // Only apply to Arabic
1359
+ if (pattern.language !== 'ar') {
1360
+ return 0;
1361
+ }
1362
+
1363
+ let adjustment = 0;
1364
+
1365
+ // Preferred prepositions for each semantic role
1366
+ // Only including roles that commonly use prepositions in Arabic
1367
+ const PREFERRED_PREPOSITIONS: Partial<Record<SemanticRole, string[]>> = {
1368
+ patient: ['على'], // element selectors prefer على (on/upon)
1369
+ destination: ['إلى', 'الى'], // destination prefers إلى (to)
1370
+ source: ['من'], // source prefers من (from)
1371
+ agent: ['من'], // agent/by prefers من (from/by)
1372
+ manner: ['ب'], // manner prefers ب (with/by)
1373
+ style: ['ب'], // style prefers ب (with)
1374
+ goal: ['إلى', 'الى'], // target state prefers إلى (to)
1375
+ method: ['ب'], // method prefers ب (with/by)
1376
+ };
1377
+
1378
+ // Check each captured role for preposition metadata
1379
+ for (const [role, value] of captured.entries()) {
1380
+ // Skip if no preferred prepositions defined for this role
1381
+ const preferred = PREFERRED_PREPOSITIONS[role];
1382
+ if (!preferred || preferred.length === 0) {
1383
+ continue;
1384
+ }
1385
+
1386
+ // Check if the value has preposition metadata (from Arabic tokenizer)
1387
+ // This metadata is attached when a preposition particle token is consumed
1388
+ const metadata =
1389
+ 'metadata' in value ? (value as { metadata: Record<string, unknown> }).metadata : undefined;
1390
+ if (metadata && typeof metadata.prepositionValue === 'string') {
1391
+ const usedPreposition = metadata.prepositionValue;
1392
+
1393
+ // Check if the used preposition is in the preferred list
1394
+ if (preferred.includes(usedPreposition)) {
1395
+ // Idiomatic choice - boost confidence
1396
+ adjustment += 0.1;
1397
+ } else {
1398
+ // Less natural choice - reduce confidence
1399
+ adjustment -= 0.1;
1400
+ }
1401
+ }
1402
+ }
1403
+
1404
+ // Cap total adjustment at ±0.10 (even if multiple roles analyzed)
1405
+ return Math.max(-0.1, Math.min(0.1, adjustment));
1406
+ }
1407
+
1408
+ // ===========================================================================
1409
+ // English Idiom Support - Noise Word Handling
1410
+ // ===========================================================================
1411
+
1412
+ /**
1413
+ * Noise words that can be skipped in English for more natural syntax.
1414
+ * - "the" before selectors: "toggle the .active" → "toggle .active"
1415
+ * - "class" after class selectors: "add the .visible class" → "add .visible"
1416
+ */
1417
+ private static readonly ENGLISH_NOISE_WORDS = new Set(['the', 'a', 'an']);
1418
+
1419
+ /**
1420
+ * Skip noise words like "the" before selectors.
1421
+ * This enables more natural English syntax like "toggle the .active".
1422
+ */
1423
+ private skipNoiseWords(tokens: TokenStream): void {
1424
+ const token = tokens.peek();
1425
+ if (!token) return;
1426
+
1427
+ const tokenLower = this.safeToLowerCase(token.value);
1428
+
1429
+ // Check if current token is a noise word (like "the")
1430
+ if (PatternMatcher.ENGLISH_NOISE_WORDS.has(tokenLower)) {
1431
+ // Look ahead to see if the next token is a selector
1432
+ const mark = tokens.mark();
1433
+ tokens.advance();
1434
+ const nextToken = tokens.peek();
1435
+
1436
+ if (nextToken && nextToken.kind === 'selector') {
1437
+ // Keep the position after "the" - effectively skipping it
1438
+ return;
1439
+ }
1440
+
1441
+ // Not followed by a selector, revert
1442
+ tokens.reset(mark);
1443
+ }
1444
+
1445
+ // Also handle "class" after class selectors: ".visible class" → ".visible"
1446
+ // This is handled when the selector has already been consumed,
1447
+ // so we check if current token is "class" and skip it
1448
+ if (tokenLower === 'class') {
1449
+ // Skip "class" as it's just noise after a class selector
1450
+ tokens.advance();
1451
+ }
1452
+ }
1453
+
1454
+ /**
1455
+ * Extract event modifiers from the token stream.
1456
+ * Event modifiers are .once, .debounce(N), .throttle(N), .queue(strategy)
1457
+ * that can appear after event names.
1458
+ *
1459
+ * Returns EventModifiers object or undefined if no modifiers found.
1460
+ */
1461
+ extractEventModifiers(tokens: TokenStream): import('../types').EventModifiers | undefined {
1462
+ const modifiers: {
1463
+ once?: boolean;
1464
+ debounce?: number;
1465
+ throttle?: number;
1466
+ queue?: 'first' | 'last' | 'all' | 'none';
1467
+ from?: SemanticValue;
1468
+ } = {};
1469
+
1470
+ let foundModifier = false;
1471
+
1472
+ // Consume all consecutive event modifier tokens
1473
+ while (!tokens.isAtEnd()) {
1474
+ const token = tokens.peek();
1475
+ if (!token || token.kind !== 'event-modifier') {
1476
+ break;
1477
+ }
1478
+
1479
+ const metadata = token.metadata as
1480
+ | { modifierName: string; value?: number | string }
1481
+ | undefined;
1482
+ if (!metadata) {
1483
+ break;
1484
+ }
1485
+
1486
+ foundModifier = true;
1487
+
1488
+ switch (metadata.modifierName) {
1489
+ case 'once':
1490
+ modifiers.once = true;
1491
+ break;
1492
+ case 'debounce':
1493
+ if (typeof metadata.value === 'number') {
1494
+ modifiers.debounce = metadata.value;
1495
+ }
1496
+ break;
1497
+ case 'throttle':
1498
+ if (typeof metadata.value === 'number') {
1499
+ modifiers.throttle = metadata.value;
1500
+ }
1501
+ break;
1502
+ case 'queue':
1503
+ if (
1504
+ metadata.value === 'first' ||
1505
+ metadata.value === 'last' ||
1506
+ metadata.value === 'all' ||
1507
+ metadata.value === 'none'
1508
+ ) {
1509
+ modifiers.queue = metadata.value;
1510
+ }
1511
+ break;
1512
+ }
1513
+
1514
+ tokens.advance();
1515
+ }
1516
+
1517
+ return foundModifier ? modifiers : undefined;
1518
+ }
1519
+ }
1520
+
1521
+ // =============================================================================
1522
+ // Convenience Functions
1523
+ // =============================================================================
1524
+
1525
+ /**
1526
+ * Singleton pattern matcher instance.
1527
+ */
1528
+ export const patternMatcher = new PatternMatcher();
1529
+
1530
+ /**
1531
+ * Match tokens against a pattern.
1532
+ */
1533
+ export function matchPattern(
1534
+ tokens: TokenStream,
1535
+ pattern: LanguagePattern
1536
+ ): PatternMatchResult | null {
1537
+ return patternMatcher.matchPattern(tokens, pattern);
1538
+ }
1539
+
1540
+ /**
1541
+ * Match tokens against multiple patterns, return best match.
1542
+ */
1543
+ export function matchBest(
1544
+ tokens: TokenStream,
1545
+ patterns: LanguagePattern[]
1546
+ ): PatternMatchResult | null {
1547
+ return patternMatcher.matchBest(tokens, patterns);
1548
+ }