@lokascript/framework 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +20 -0
- package/README.md +142 -0
- package/dist/aot/aot-orchestrator.d.ts +75 -0
- package/dist/aot/aot-orchestrator.d.ts.map +1 -0
- package/dist/aot/domain-scanner.d.ts +27 -0
- package/dist/aot/domain-scanner.d.ts.map +1 -0
- package/dist/aot/index.d.ts +8 -0
- package/dist/aot/index.d.ts.map +1 -0
- package/dist/aot/types.d.ts +103 -0
- package/dist/aot/types.d.ts.map +1 -0
- package/dist/api/create-dsl.d.ts +91 -0
- package/dist/api/create-dsl.d.ts.map +1 -0
- package/dist/api/dispatcher.d.ts +108 -0
- package/dist/api/dispatcher.d.ts.map +1 -0
- package/dist/api/domain-registry.d.ts +152 -0
- package/dist/api/domain-registry.d.ts.map +1 -0
- package/dist/api/index.d.ts +7 -0
- package/dist/api/index.d.ts.map +1 -0
- package/dist/api/index.js +2082 -0
- package/dist/api/index.js.map +1 -0
- package/dist/core/index.d.ts +7 -0
- package/dist/core/index.d.ts.map +1 -0
- package/dist/core/index.js +2674 -0
- package/dist/core/index.js.map +1 -0
- package/dist/core/logger.d.ts +32 -0
- package/dist/core/logger.d.ts.map +1 -0
- package/dist/core/pattern-matching/index.d.ts +6 -0
- package/dist/core/pattern-matching/index.d.ts.map +1 -0
- package/dist/core/pattern-matching/index.js +1239 -0
- package/dist/core/pattern-matching/index.js.map +1 -0
- package/dist/core/pattern-matching/pattern-matcher.d.ts +239 -0
- package/dist/core/pattern-matching/pattern-matcher.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/index.d.ts +6 -0
- package/dist/core/pattern-matching/utils/index.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/possessive-keywords.d.ts +38 -0
- package/dist/core/pattern-matching/utils/possessive-keywords.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/type-validation.d.ts +63 -0
- package/dist/core/pattern-matching/utils/type-validation.d.ts.map +1 -0
- package/dist/core/tokenization/base-tokenizer.d.ts +344 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -0
- package/dist/core/tokenization/char-classifiers.d.ts +56 -0
- package/dist/core/tokenization/char-classifiers.d.ts.map +1 -0
- package/dist/core/tokenization/default-extractors.d.ts +48 -0
- package/dist/core/tokenization/default-extractors.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/index.d.ts +9 -0
- package/dist/core/tokenization/extractors/index.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/operator.d.ts +23 -0
- package/dist/core/tokenization/extractors/operator.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/punctuation.d.ts +22 -0
- package/dist/core/tokenization/extractors/punctuation.d.ts.map +1 -0
- package/dist/core/tokenization/extractors.d.ts +61 -0
- package/dist/core/tokenization/extractors.d.ts.map +1 -0
- package/dist/core/tokenization/index.d.ts +11 -0
- package/dist/core/tokenization/index.d.ts.map +1 -0
- package/dist/core/tokenization/index.js +1345 -0
- package/dist/core/tokenization/index.js.map +1 -0
- package/dist/core/tokenization/morphology/index.d.ts +5 -0
- package/dist/core/tokenization/morphology/index.d.ts.map +1 -0
- package/dist/core/tokenization/morphology/types.d.ts +110 -0
- package/dist/core/tokenization/morphology/types.d.ts.map +1 -0
- package/dist/core/tokenization/token-utils.d.ts +111 -0
- package/dist/core/tokenization/token-utils.d.ts.map +1 -0
- package/dist/core/types.d.ts +382 -0
- package/dist/core/types.d.ts.map +1 -0
- package/dist/core/types.js +108 -0
- package/dist/core/types.js.map +1 -0
- package/dist/generation/diagnostics.d.ts +120 -0
- package/dist/generation/diagnostics.d.ts.map +1 -0
- package/dist/generation/index.d.ts +7 -0
- package/dist/generation/index.d.ts.map +1 -0
- package/dist/generation/index.js +339 -0
- package/dist/generation/index.js.map +1 -0
- package/dist/generation/pattern-generator.d.ts +48 -0
- package/dist/generation/pattern-generator.d.ts.map +1 -0
- package/dist/generation/renderer.d.ts +115 -0
- package/dist/generation/renderer.d.ts.map +1 -0
- package/dist/grammar/index.d.ts +10 -0
- package/dist/grammar/index.d.ts.map +1 -0
- package/dist/grammar/index.js +391 -0
- package/dist/grammar/index.js.map +1 -0
- package/dist/grammar/transformer.d.ts +56 -0
- package/dist/grammar/transformer.d.ts.map +1 -0
- package/dist/grammar/types.d.ts +236 -0
- package/dist/grammar/types.d.ts.map +1 -0
- package/dist/index.cjs +4454 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.ts +46 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +4336 -0
- package/dist/index.js.map +1 -0
- package/dist/interfaces/dictionary.d.ts +82 -0
- package/dist/interfaces/dictionary.d.ts.map +1 -0
- package/dist/interfaces/index.d.ts +10 -0
- package/dist/interfaces/index.d.ts.map +1 -0
- package/dist/interfaces/profile-provider.d.ts +67 -0
- package/dist/interfaces/profile-provider.d.ts.map +1 -0
- package/dist/interfaces/value-extractor.d.ts +168 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -0
- package/dist/multilingual/index.d.ts +8 -0
- package/dist/multilingual/index.d.ts.map +1 -0
- package/dist/multilingual/index.js +1 -0
- package/dist/multilingual/index.js.map +1 -0
- package/dist/parsing/index.d.ts +8 -0
- package/dist/parsing/index.d.ts.map +1 -0
- package/dist/parsing/index.js +1415 -0
- package/dist/parsing/index.js.map +1 -0
- package/dist/parsing/multi-statement.d.ts +265 -0
- package/dist/parsing/multi-statement.d.ts.map +1 -0
- package/dist/schema/command-schema.d.ts +78 -0
- package/dist/schema/command-schema.d.ts.map +1 -0
- package/dist/schema/index.d.ts +5 -0
- package/dist/schema/index.d.ts.map +1 -0
- package/dist/schema/index.js +25 -0
- package/dist/schema/index.js.map +1 -0
- package/dist/test-setup.d.ts +9 -0
- package/dist/test-setup.d.ts.map +1 -0
- package/dist/testing/index.d.ts +50 -0
- package/dist/testing/index.d.ts.map +1 -0
- package/dist/testing/index.js +16969 -0
- package/dist/testing/index.js.map +1 -0
- package/package.json +122 -0
- package/src/__test__/fixtures/sql-dsl.ts +232 -0
- package/src/__test__/sql-integration.test.ts +189 -0
- package/src/__test__/test-utils.ts +260 -0
- package/src/aot/aot-orchestrator.test.ts +413 -0
- package/src/aot/aot-orchestrator.ts +238 -0
- package/src/aot/domain-scanner.ts +178 -0
- package/src/aot/index.ts +8 -0
- package/src/aot/types.ts +124 -0
- package/src/api/create-dsl.ts +367 -0
- package/src/api/dispatcher.test.ts +336 -0
- package/src/api/dispatcher.ts +222 -0
- package/src/api/domain-registry.test.ts +336 -0
- package/src/api/domain-registry.ts +500 -0
- package/src/api/index.ts +7 -0
- package/src/core/index.ts +7 -0
- package/src/core/logger.ts +130 -0
- package/src/core/pattern-matching/index.ts +6 -0
- package/src/core/pattern-matching/pattern-matcher.test.ts +900 -0
- package/src/core/pattern-matching/pattern-matcher.ts +1548 -0
- package/src/core/pattern-matching/pattern-matcher.ts.backup +1267 -0
- package/src/core/pattern-matching/utils/index.ts +6 -0
- package/src/core/pattern-matching/utils/possessive-keywords.ts +55 -0
- package/src/core/pattern-matching/utils/type-validation.test.ts +316 -0
- package/src/core/pattern-matching/utils/type-validation.ts +134 -0
- package/src/core/tokenization/base-tokenizer.ts +916 -0
- package/src/core/tokenization/char-classifiers.ts +79 -0
- package/src/core/tokenization/create-simple-tokenizer.test.ts +260 -0
- package/src/core/tokenization/default-extractors.ts +69 -0
- package/src/core/tokenization/extractors/index.ts +9 -0
- package/src/core/tokenization/extractors/operator.ts +75 -0
- package/src/core/tokenization/extractors/punctuation.ts +39 -0
- package/src/core/tokenization/extractors.ts +452 -0
- package/src/core/tokenization/index.ts +11 -0
- package/src/core/tokenization/morphology/index.ts +5 -0
- package/src/core/tokenization/morphology/types.ts +211 -0
- package/src/core/tokenization/token-utils.ts +252 -0
- package/src/core/types.ts +589 -0
- package/src/generation/diagnostics.test.ts +171 -0
- package/src/generation/diagnostics.ts +239 -0
- package/src/generation/index.ts +7 -0
- package/src/generation/pattern-generator.test.ts +430 -0
- package/src/generation/pattern-generator.ts +315 -0
- package/src/generation/renderer.test.ts +266 -0
- package/src/generation/renderer.ts +244 -0
- package/src/grammar/index.ts +12 -0
- package/src/grammar/transformer.ts +159 -0
- package/src/grammar/types.ts +630 -0
- package/src/index.ts +157 -0
- package/src/interfaces/dictionary.ts +123 -0
- package/src/interfaces/index.ts +10 -0
- package/src/interfaces/profile-provider.ts +88 -0
- package/src/interfaces/value-extractor.ts +435 -0
- package/src/multilingual/index.ts +9 -0
- package/src/parsing/index.ts +27 -0
- package/src/parsing/multi-statement.test.ts +480 -0
- package/src/parsing/multi-statement.ts +648 -0
- package/src/schema/command-schema.ts +118 -0
- package/src/schema/index.ts +5 -0
- package/src/test-setup.ts +45 -0
- package/src/testing/index.ts +137 -0
|
@@ -0,0 +1,1548 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pattern Matcher
|
|
3
|
+
*
|
|
4
|
+
* Matches tokenized input against language patterns to extract semantic roles.
|
|
5
|
+
* This is the core algorithm for multilingual parsing.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import type {
|
|
9
|
+
LanguagePattern,
|
|
10
|
+
PatternToken,
|
|
11
|
+
PatternMatchResult,
|
|
12
|
+
ReferenceValue,
|
|
13
|
+
SemanticRole,
|
|
14
|
+
SemanticValue,
|
|
15
|
+
TokenStream,
|
|
16
|
+
LanguageToken,
|
|
17
|
+
} from '../types';
|
|
18
|
+
import { createSelector, createLiteral, createReference, createPropertyPath } from '../types';
|
|
19
|
+
import { createLogger } from '../logger';
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Helper to check if a value is a built-in reference.
|
|
23
|
+
* In the generic framework, we don't assume any built-in references.
|
|
24
|
+
* DSLs that need references (like hyperscript's 'me', 'it', 'you') should
|
|
25
|
+
* handle them in their tokenizer or provide a custom reference validator.
|
|
26
|
+
*
|
|
27
|
+
* For generic DSLs, identifiers are treated as expressions by default.
|
|
28
|
+
*/
|
|
29
|
+
function isValidReference(_value: string): boolean {
|
|
30
|
+
// In generic framework, no built-in references
|
|
31
|
+
// Identifiers should be expressions
|
|
32
|
+
return false;
|
|
33
|
+
}
|
|
34
|
+
import { isTypeCompatible } from './utils/type-validation';
|
|
35
|
+
import { getPossessiveReference } from './utils/possessive-keywords';
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Minimal profile interface needed by PatternMatcher.
|
|
39
|
+
* DSLs can provide this to enable possessive handling and language-specific features.
|
|
40
|
+
*/
|
|
41
|
+
export interface PatternMatcherProfile {
|
|
42
|
+
readonly code: string;
|
|
43
|
+
readonly possessive?: {
|
|
44
|
+
readonly keywords?: Record<string, string>;
|
|
45
|
+
};
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
// =============================================================================
|
|
49
|
+
// Pattern Matcher
|
|
50
|
+
// =============================================================================
|
|
51
|
+
|
|
52
|
+
export class PatternMatcher {
|
|
53
|
+
/** Maximum tokens to scan ahead when looking for a group's leading marker */
|
|
54
|
+
private static readonly MAX_MARKER_SCAN = 3;
|
|
55
|
+
|
|
56
|
+
/** Debug logger */
|
|
57
|
+
private logger = createLogger('pattern-matcher');
|
|
58
|
+
|
|
59
|
+
/** Current language profile for the pattern being matched */
|
|
60
|
+
private currentProfile: PatternMatcherProfile | undefined;
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Safely convert a value to lowercase string.
|
|
64
|
+
* Provides protection against non-string values at runtime.
|
|
65
|
+
*/
|
|
66
|
+
private safeToLowerCase(value: unknown): string {
|
|
67
|
+
if (typeof value === 'string') {
|
|
68
|
+
return value.toLowerCase();
|
|
69
|
+
}
|
|
70
|
+
if (value === null || value === undefined) {
|
|
71
|
+
return '';
|
|
72
|
+
}
|
|
73
|
+
return String(value).toLowerCase();
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Try to match a single pattern against the token stream.
|
|
78
|
+
* Returns the match result or null if no match.
|
|
79
|
+
*
|
|
80
|
+
* @param tokens - Token stream to match against
|
|
81
|
+
* @param pattern - Pattern to match
|
|
82
|
+
* @param profile - Optional language profile for possessive handling
|
|
83
|
+
*/
|
|
84
|
+
matchPattern(
|
|
85
|
+
tokens: TokenStream,
|
|
86
|
+
pattern: LanguagePattern,
|
|
87
|
+
profile?: PatternMatcherProfile
|
|
88
|
+
): PatternMatchResult | null {
|
|
89
|
+
const mark = tokens.mark();
|
|
90
|
+
const captured = new Map<SemanticRole, SemanticValue>();
|
|
91
|
+
|
|
92
|
+
// Debug logging
|
|
93
|
+
this.logger.debug('========================================');
|
|
94
|
+
this.logger.debug('matchPattern ENTRY');
|
|
95
|
+
this.logger.debug('Pattern ID:', pattern.id);
|
|
96
|
+
this.logger.debug('Pattern command:', pattern.command);
|
|
97
|
+
this.logger.debug('Pattern language:', pattern.language);
|
|
98
|
+
this.logger.debug('Pattern template:', JSON.stringify(pattern.template, null, 2));
|
|
99
|
+
|
|
100
|
+
if (this.logger.isEnabled()) {
|
|
101
|
+
const firstTokens = [];
|
|
102
|
+
for (let i = 0; i < 10; i++) {
|
|
103
|
+
const t = tokens.peek(i);
|
|
104
|
+
if (t)
|
|
105
|
+
firstTokens.push({
|
|
106
|
+
type: (t as any).type,
|
|
107
|
+
value: (t as any).value,
|
|
108
|
+
kind: (t as any).kind,
|
|
109
|
+
});
|
|
110
|
+
else break;
|
|
111
|
+
}
|
|
112
|
+
this.logger.debug('Input tokens (first 10):', firstTokens);
|
|
113
|
+
this.logger.debug('Profile code:', profile?.code);
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
// Use provided profile for possessive keyword lookup
|
|
117
|
+
this.currentProfile = profile;
|
|
118
|
+
|
|
119
|
+
// Reset match counters for this pattern
|
|
120
|
+
this.stemMatchCount = 0;
|
|
121
|
+
this.totalKeywordMatches = 0;
|
|
122
|
+
|
|
123
|
+
this.logger.debug('--- Calling matchTokenSequence ---');
|
|
124
|
+
this.logger.debug('Pattern tokens to match:', JSON.stringify(pattern.template.tokens, null, 2));
|
|
125
|
+
const success = this.matchTokenSequence(tokens, pattern.template.tokens, captured);
|
|
126
|
+
this.logger.debug('matchTokenSequence returned:', success);
|
|
127
|
+
this.logger.debug(
|
|
128
|
+
'Captured roles:',
|
|
129
|
+
Array.from(captured.entries()).map(([k, v]) => [k, JSON.stringify(v)])
|
|
130
|
+
);
|
|
131
|
+
|
|
132
|
+
if (!success) {
|
|
133
|
+
this.logger.debug('>>> MATCH FAILED - resetting token position');
|
|
134
|
+
tokens.reset(mark);
|
|
135
|
+
return null;
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
// Calculate confidence BEFORE applying defaults
|
|
139
|
+
// This ensures defaulted roles don't artificially inflate confidence
|
|
140
|
+
const confidence = this.calculateConfidence(pattern, captured);
|
|
141
|
+
|
|
142
|
+
// Apply extraction rules to fill in any missing roles with defaults
|
|
143
|
+
this.applyExtractionRules(pattern, captured);
|
|
144
|
+
|
|
145
|
+
return {
|
|
146
|
+
pattern,
|
|
147
|
+
captured,
|
|
148
|
+
consumedTokens: tokens.position() - mark.position,
|
|
149
|
+
confidence,
|
|
150
|
+
};
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/**
|
|
154
|
+
* Try to match multiple patterns, return the best match.
|
|
155
|
+
*
|
|
156
|
+
* @param tokens - Token stream to match against
|
|
157
|
+
* @param patterns - Candidate patterns to try
|
|
158
|
+
* @param profile - Optional language profile for possessive handling
|
|
159
|
+
*/
|
|
160
|
+
matchBest(
|
|
161
|
+
tokens: TokenStream,
|
|
162
|
+
patterns: LanguagePattern[],
|
|
163
|
+
profile?: PatternMatcherProfile
|
|
164
|
+
): PatternMatchResult | null {
|
|
165
|
+
const matches: PatternMatchResult[] = [];
|
|
166
|
+
|
|
167
|
+
for (const pattern of patterns) {
|
|
168
|
+
const mark = tokens.mark();
|
|
169
|
+
const result = this.matchPattern(tokens, pattern, profile);
|
|
170
|
+
|
|
171
|
+
if (result) {
|
|
172
|
+
matches.push(result);
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
tokens.reset(mark);
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
if (matches.length === 0) {
|
|
179
|
+
return null;
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
// Sort by confidence and priority
|
|
183
|
+
matches.sort((a, b) => {
|
|
184
|
+
// First by priority
|
|
185
|
+
const priorityDiff = b.pattern.priority - a.pattern.priority;
|
|
186
|
+
if (priorityDiff !== 0) return priorityDiff;
|
|
187
|
+
|
|
188
|
+
// Then by confidence
|
|
189
|
+
const confidenceDiff = b.confidence - a.confidence;
|
|
190
|
+
if (Math.abs(confidenceDiff) > 0.001) return confidenceDiff;
|
|
191
|
+
|
|
192
|
+
// Then by tokens consumed (prefer more complete matches)
|
|
193
|
+
return b.consumedTokens - a.consumedTokens;
|
|
194
|
+
});
|
|
195
|
+
|
|
196
|
+
// Re-consume tokens for the best match
|
|
197
|
+
const best = matches[0];
|
|
198
|
+
this.matchPattern(tokens, best.pattern);
|
|
199
|
+
|
|
200
|
+
return best;
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
/**
|
|
204
|
+
* Match a sequence of pattern tokens against the token stream.
|
|
205
|
+
*
|
|
206
|
+
* Supports bounded single-step backtracking: if an optional role consumes
|
|
207
|
+
* a token and the immediately following pattern token fails, the matcher
|
|
208
|
+
* resets to before the optional role and retries the failed token.
|
|
209
|
+
*/
|
|
210
|
+
private matchTokenSequence(
|
|
211
|
+
tokens: TokenStream,
|
|
212
|
+
patternTokens: PatternToken[],
|
|
213
|
+
captured: Map<SemanticRole, SemanticValue>
|
|
214
|
+
): boolean {
|
|
215
|
+
// Skip leading conjunctions for Arabic (proclitics: و, ف, ول, وب, etc.)
|
|
216
|
+
// BUT NOT if the pattern explicitly expects a conjunction (proclitic patterns)
|
|
217
|
+
const firstPatternToken = patternTokens[0];
|
|
218
|
+
const patternExpectsConjunction =
|
|
219
|
+
firstPatternToken?.type === 'literal' &&
|
|
220
|
+
(firstPatternToken.value === 'and' ||
|
|
221
|
+
firstPatternToken.value === 'then' ||
|
|
222
|
+
firstPatternToken.alternatives?.includes('and') ||
|
|
223
|
+
firstPatternToken.alternatives?.includes('then'));
|
|
224
|
+
|
|
225
|
+
if (this.currentProfile?.code === 'ar' && !patternExpectsConjunction) {
|
|
226
|
+
while (tokens.peek()?.kind === 'conjunction') {
|
|
227
|
+
tokens.advance();
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
// Backtracking state: track the most recent optional role that consumed a token
|
|
232
|
+
let prevOptionalMark: ReturnType<TokenStream['mark']> | null = null;
|
|
233
|
+
let prevOptionalRole: SemanticRole | null = null;
|
|
234
|
+
|
|
235
|
+
for (let i = 0; i < patternTokens.length; i++) {
|
|
236
|
+
const patternToken = patternTokens[i];
|
|
237
|
+
this.logger.debug(' >> Matching pattern token:', JSON.stringify(patternToken, null, 2));
|
|
238
|
+
const currTok = tokens.peek();
|
|
239
|
+
this.logger.debug(
|
|
240
|
+
' >> Current input token:',
|
|
241
|
+
currTok
|
|
242
|
+
? JSON.stringify({
|
|
243
|
+
type: (currTok as any).type,
|
|
244
|
+
value: (currTok as any).value,
|
|
245
|
+
kind: (currTok as any).kind,
|
|
246
|
+
})
|
|
247
|
+
: 'EOF'
|
|
248
|
+
);
|
|
249
|
+
|
|
250
|
+
// Greedy role capture: consume all remaining tokens until the next
|
|
251
|
+
// recognized marker keyword or end of input
|
|
252
|
+
if (patternToken.type === 'role' && patternToken.greedy) {
|
|
253
|
+
const stopMarkers = this.collectStopMarkers(patternTokens, i + 1);
|
|
254
|
+
const values: string[] = [];
|
|
255
|
+
while (!tokens.isAtEnd()) {
|
|
256
|
+
const nextToken = tokens.peek();
|
|
257
|
+
if (!nextToken) break;
|
|
258
|
+
if (this.isStopMarker(nextToken, stopMarkers)) break;
|
|
259
|
+
values.push(nextToken.value);
|
|
260
|
+
tokens.advance();
|
|
261
|
+
}
|
|
262
|
+
if (values.length > 0) {
|
|
263
|
+
captured.set(patternToken.role, { type: 'expression', raw: values.join(' ') });
|
|
264
|
+
prevOptionalMark = null;
|
|
265
|
+
prevOptionalRole = null;
|
|
266
|
+
continue;
|
|
267
|
+
} else if (patternToken.optional) {
|
|
268
|
+
continue;
|
|
269
|
+
} else {
|
|
270
|
+
return false;
|
|
271
|
+
}
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
// Save stream position before attempting optional roles
|
|
275
|
+
const isOptionalRole = patternToken.type === 'role' && patternToken.optional === true;
|
|
276
|
+
const markBefore = isOptionalRole ? tokens.mark() : null;
|
|
277
|
+
|
|
278
|
+
const matched = this.matchPatternToken(tokens, patternToken, captured);
|
|
279
|
+
this.logger.debug(' >> Match result:', matched);
|
|
280
|
+
|
|
281
|
+
if (matched) {
|
|
282
|
+
if (isOptionalRole) {
|
|
283
|
+
// Track this consumption so we can undo it if the next token fails
|
|
284
|
+
prevOptionalMark = markBefore;
|
|
285
|
+
prevOptionalRole = patternToken.role;
|
|
286
|
+
} else {
|
|
287
|
+
// Non-optional succeeded — clear backtrack state
|
|
288
|
+
prevOptionalMark = null;
|
|
289
|
+
prevOptionalRole = null;
|
|
290
|
+
}
|
|
291
|
+
continue;
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
// Match failed
|
|
295
|
+
this.logger.debug(' >> Token match FAILED');
|
|
296
|
+
|
|
297
|
+
if (this.isOptional(patternToken)) {
|
|
298
|
+
continue;
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
// Required token failed — try backtracking over the previous optional role
|
|
302
|
+
if (prevOptionalMark && prevOptionalRole) {
|
|
303
|
+
this.logger.debug(' >> BACKTRACKING: undoing optional role', prevOptionalRole);
|
|
304
|
+
tokens.reset(prevOptionalMark);
|
|
305
|
+
captured.delete(prevOptionalRole);
|
|
306
|
+
prevOptionalMark = null;
|
|
307
|
+
prevOptionalRole = null;
|
|
308
|
+
|
|
309
|
+
// Retry the current (failed) pattern token from the restored position
|
|
310
|
+
const retryMatched = this.matchPatternToken(tokens, patternToken, captured);
|
|
311
|
+
this.logger.debug(' >> Backtrack retry result:', retryMatched);
|
|
312
|
+
if (retryMatched) {
|
|
313
|
+
continue;
|
|
314
|
+
}
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
return false;
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
return true;
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
/**
|
|
324
|
+
* Match a single pattern token against the current position in the stream.
|
|
325
|
+
*/
|
|
326
|
+
private matchPatternToken(
|
|
327
|
+
tokens: TokenStream,
|
|
328
|
+
patternToken: PatternToken,
|
|
329
|
+
captured: Map<SemanticRole, SemanticValue>
|
|
330
|
+
): boolean {
|
|
331
|
+
switch (patternToken.type) {
|
|
332
|
+
case 'literal':
|
|
333
|
+
return this.matchLiteralToken(tokens, patternToken);
|
|
334
|
+
|
|
335
|
+
case 'role':
|
|
336
|
+
return this.matchRoleToken(tokens, patternToken, captured);
|
|
337
|
+
|
|
338
|
+
case 'group':
|
|
339
|
+
return this.matchGroupToken(tokens, patternToken, captured);
|
|
340
|
+
|
|
341
|
+
default:
|
|
342
|
+
return false;
|
|
343
|
+
}
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
/**
|
|
347
|
+
* Match a literal pattern token (keyword or particle).
|
|
348
|
+
*/
|
|
349
|
+
private matchLiteralToken(
|
|
350
|
+
tokens: TokenStream,
|
|
351
|
+
patternToken: PatternToken & { type: 'literal' }
|
|
352
|
+
): boolean {
|
|
353
|
+
const token = tokens.peek();
|
|
354
|
+
this.logger.debug(' >>> matchLiteralToken: expecting', patternToken.value);
|
|
355
|
+
this.logger.debug(
|
|
356
|
+
' >>> matchLiteralToken: got token',
|
|
357
|
+
token
|
|
358
|
+
? JSON.stringify({
|
|
359
|
+
type: (token as any).type,
|
|
360
|
+
value: (token as any).value,
|
|
361
|
+
kind: (token as any).kind,
|
|
362
|
+
})
|
|
363
|
+
: 'null'
|
|
364
|
+
);
|
|
365
|
+
if (!token) {
|
|
366
|
+
this.logger.debug(' >>> matchLiteralToken: FAIL - no token');
|
|
367
|
+
return false;
|
|
368
|
+
}
|
|
369
|
+
|
|
370
|
+
// Check main value
|
|
371
|
+
const matchType = this.getMatchType(token, patternToken.value);
|
|
372
|
+
this.logger.debug(
|
|
373
|
+
' >>> matchType for',
|
|
374
|
+
token.value,
|
|
375
|
+
'vs',
|
|
376
|
+
patternToken.value,
|
|
377
|
+
':',
|
|
378
|
+
matchType
|
|
379
|
+
);
|
|
380
|
+
if (matchType !== 'none') {
|
|
381
|
+
this.totalKeywordMatches++;
|
|
382
|
+
if (matchType === 'stem') {
|
|
383
|
+
this.stemMatchCount++;
|
|
384
|
+
}
|
|
385
|
+
tokens.advance();
|
|
386
|
+
return true;
|
|
387
|
+
}
|
|
388
|
+
|
|
389
|
+
// Check alternatives
|
|
390
|
+
if (patternToken.alternatives) {
|
|
391
|
+
for (const alt of patternToken.alternatives) {
|
|
392
|
+
const altMatchType = this.getMatchType(token, alt);
|
|
393
|
+
if (altMatchType !== 'none') {
|
|
394
|
+
this.totalKeywordMatches++;
|
|
395
|
+
if (altMatchType === 'stem') {
|
|
396
|
+
this.stemMatchCount++;
|
|
397
|
+
}
|
|
398
|
+
tokens.advance();
|
|
399
|
+
return true;
|
|
400
|
+
}
|
|
401
|
+
}
|
|
402
|
+
}
|
|
403
|
+
|
|
404
|
+
return false;
|
|
405
|
+
}
|
|
406
|
+
|
|
407
|
+
/**
|
|
408
|
+
* Match a role pattern token (captures a semantic value).
|
|
409
|
+
* Handles multi-token expressions like:
|
|
410
|
+
* - 'my value' (possessive keyword + property)
|
|
411
|
+
* - '#dialog.showModal()' (method call)
|
|
412
|
+
* - "#element's *opacity" (possessive selector + property)
|
|
413
|
+
*/
|
|
414
|
+
private matchRoleToken(
|
|
415
|
+
tokens: TokenStream,
|
|
416
|
+
patternToken: PatternToken & { type: 'role' },
|
|
417
|
+
captured: Map<SemanticRole, SemanticValue>
|
|
418
|
+
): boolean {
|
|
419
|
+
this.logger.debug(' >>> matchRoleToken ENTRY: capturing role', patternToken.role);
|
|
420
|
+
this.logger.debug(' >>> matchRoleToken: expected types', patternToken.expectedTypes);
|
|
421
|
+
this.logger.debug(' >>> matchRoleToken: optional?', patternToken.optional);
|
|
422
|
+
// Skip noise words like "the" before selectors (English idiom support)
|
|
423
|
+
this.skipNoiseWords(tokens);
|
|
424
|
+
|
|
425
|
+
const token = tokens.peek();
|
|
426
|
+
this.logger.debug(
|
|
427
|
+
' >>> After skipNoiseWords, current token:',
|
|
428
|
+
token ? JSON.stringify({ value: (token as any).value, kind: (token as any).kind }) : 'null'
|
|
429
|
+
);
|
|
430
|
+
if (!token) {
|
|
431
|
+
return patternToken.optional || false;
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
// Check for possessive expression (e.g., 'my value', 'its innerHTML')
|
|
435
|
+
const possessiveValue = this.tryMatchPossessiveExpression(tokens);
|
|
436
|
+
if (possessiveValue) {
|
|
437
|
+
// Validate expected types if specified
|
|
438
|
+
if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
|
|
439
|
+
if (
|
|
440
|
+
!patternToken.expectedTypes.includes(possessiveValue.type) &&
|
|
441
|
+
!patternToken.expectedTypes.includes('expression')
|
|
442
|
+
) {
|
|
443
|
+
return patternToken.optional || false;
|
|
444
|
+
}
|
|
445
|
+
}
|
|
446
|
+
captured.set(patternToken.role, possessiveValue);
|
|
447
|
+
return true;
|
|
448
|
+
}
|
|
449
|
+
|
|
450
|
+
// Check for method call expression (e.g., '#dialog.showModal()')
|
|
451
|
+
const methodCallValue = this.tryMatchMethodCallExpression(tokens);
|
|
452
|
+
if (methodCallValue) {
|
|
453
|
+
if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
|
|
454
|
+
if (
|
|
455
|
+
!patternToken.expectedTypes.includes(methodCallValue.type) &&
|
|
456
|
+
!patternToken.expectedTypes.includes('expression')
|
|
457
|
+
) {
|
|
458
|
+
return patternToken.optional || false;
|
|
459
|
+
}
|
|
460
|
+
}
|
|
461
|
+
captured.set(patternToken.role, methodCallValue);
|
|
462
|
+
return true;
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
// Check for possessive selector expression (e.g., "#element's *opacity")
|
|
466
|
+
const possessiveSelectorValue = this.tryMatchPossessiveSelectorExpression(tokens);
|
|
467
|
+
if (possessiveSelectorValue) {
|
|
468
|
+
if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
|
|
469
|
+
// property-path is compatible with selector, reference, and expression
|
|
470
|
+
if (!isTypeCompatible(possessiveSelectorValue.type, patternToken.expectedTypes)) {
|
|
471
|
+
return patternToken.optional || false;
|
|
472
|
+
}
|
|
473
|
+
}
|
|
474
|
+
captured.set(patternToken.role, possessiveSelectorValue);
|
|
475
|
+
return true;
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
// Check for property access expression (e.g., 'userData.name', 'it.data')
|
|
479
|
+
const propertyAccessValue = this.tryMatchPropertyAccessExpression(tokens);
|
|
480
|
+
if (propertyAccessValue) {
|
|
481
|
+
if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
|
|
482
|
+
if (
|
|
483
|
+
!patternToken.expectedTypes.includes(propertyAccessValue.type) &&
|
|
484
|
+
!patternToken.expectedTypes.includes('expression')
|
|
485
|
+
) {
|
|
486
|
+
return patternToken.optional || false;
|
|
487
|
+
}
|
|
488
|
+
}
|
|
489
|
+
captured.set(patternToken.role, propertyAccessValue);
|
|
490
|
+
return true;
|
|
491
|
+
}
|
|
492
|
+
|
|
493
|
+
// Check for selector + property expression (e.g., '#output.innerText')
|
|
494
|
+
// This handles cases where the tokenizer produces two selector tokens
|
|
495
|
+
const selectorPropertyValue = this.tryMatchSelectorPropertyExpression(tokens);
|
|
496
|
+
if (selectorPropertyValue) {
|
|
497
|
+
if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
|
|
498
|
+
if (!isTypeCompatible(selectorPropertyValue.type, patternToken.expectedTypes)) {
|
|
499
|
+
return patternToken.optional || false;
|
|
500
|
+
}
|
|
501
|
+
}
|
|
502
|
+
captured.set(patternToken.role, selectorPropertyValue);
|
|
503
|
+
return true;
|
|
504
|
+
}
|
|
505
|
+
|
|
506
|
+
// Try to extract a semantic value from the token
|
|
507
|
+
this.logger.debug(
|
|
508
|
+
' >>> Trying tokenToSemanticValue for token:',
|
|
509
|
+
token ? JSON.stringify({ value: (token as any).value, kind: (token as any).kind }) : 'null'
|
|
510
|
+
);
|
|
511
|
+
const value = this.tokenToSemanticValue(token);
|
|
512
|
+
this.logger.debug(
|
|
513
|
+
' >>> tokenToSemanticValue returned:',
|
|
514
|
+
value ? JSON.stringify(value) : 'null'
|
|
515
|
+
);
|
|
516
|
+
if (!value) {
|
|
517
|
+
return patternToken.optional || false;
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
// Validate expected types if specified
|
|
521
|
+
this.logger.debug(
|
|
522
|
+
' >>> Validating type:',
|
|
523
|
+
value.type,
|
|
524
|
+
'against expected:',
|
|
525
|
+
patternToken.expectedTypes
|
|
526
|
+
);
|
|
527
|
+
if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
|
|
528
|
+
if (!isTypeCompatible(value.type, patternToken.expectedTypes)) {
|
|
529
|
+
this.logger.debug(' >>> TYPE MISMATCH - returning', patternToken.optional || false);
|
|
530
|
+
return patternToken.optional || false;
|
|
531
|
+
}
|
|
532
|
+
}
|
|
533
|
+
this.logger.debug(' >>> Type validation PASSED');
|
|
534
|
+
|
|
535
|
+
captured.set(patternToken.role, value);
|
|
536
|
+
tokens.advance();
|
|
537
|
+
return true;
|
|
538
|
+
}
|
|
539
|
+
|
|
540
|
+
/**
|
|
541
|
+
* Try to match a possessive expression like 'my value' or 'its innerHTML'.
|
|
542
|
+
* Returns the PropertyPathValue if matched, or null if not.
|
|
543
|
+
*/
|
|
544
|
+
private tryMatchPossessiveExpression(tokens: TokenStream): SemanticValue | null {
|
|
545
|
+
const token = tokens.peek();
|
|
546
|
+
if (!token) return null;
|
|
547
|
+
|
|
548
|
+
// Use profile-based possessive keyword lookup
|
|
549
|
+
if (!this.currentProfile) return null;
|
|
550
|
+
|
|
551
|
+
const tokenValue = token.normalized || token.value;
|
|
552
|
+
const tokenLower = this.safeToLowerCase(tokenValue);
|
|
553
|
+
const baseRef = getPossessiveReference(this.currentProfile, tokenLower);
|
|
554
|
+
|
|
555
|
+
if (!baseRef) return null;
|
|
556
|
+
|
|
557
|
+
// We have a possessive keyword, look ahead for property name
|
|
558
|
+
const mark = tokens.mark();
|
|
559
|
+
tokens.advance();
|
|
560
|
+
|
|
561
|
+
const propertyToken = tokens.peek();
|
|
562
|
+
if (!propertyToken) {
|
|
563
|
+
// Just the possessive keyword, no property - revert
|
|
564
|
+
tokens.reset(mark);
|
|
565
|
+
return null;
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
// Property should be an identifier, keyword (not structural), or selector (for style/dot/attr)
|
|
569
|
+
// Examples: "my value", "my innerHTML", "my *background", "my *opacity", "my @data-count"
|
|
570
|
+
// Also handles dot-property access: "my.textContent" tokenized as "my" + ".textContent"
|
|
571
|
+
if (
|
|
572
|
+
propertyToken.kind === 'identifier' ||
|
|
573
|
+
(propertyToken.kind === 'keyword' && !this.isStructuralKeyword(propertyToken.value)) ||
|
|
574
|
+
(propertyToken.kind === 'selector' && propertyToken.value.startsWith('*')) ||
|
|
575
|
+
(propertyToken.kind === 'selector' && propertyToken.value.startsWith('@')) ||
|
|
576
|
+
(propertyToken.kind === 'selector' &&
|
|
577
|
+
propertyToken.value.startsWith('.') &&
|
|
578
|
+
/^\.[a-zA-Z_]\w*/.test(propertyToken.value))
|
|
579
|
+
) {
|
|
580
|
+
tokens.advance();
|
|
581
|
+
|
|
582
|
+
// For dot-property selectors (.textContent), strip the leading dot
|
|
583
|
+
let propertyName = propertyToken.value;
|
|
584
|
+
if (
|
|
585
|
+
propertyToken.kind === 'selector' &&
|
|
586
|
+
propertyName.startsWith('.') &&
|
|
587
|
+
/^\.[a-zA-Z_]\w*/.test(propertyName)
|
|
588
|
+
) {
|
|
589
|
+
propertyName = propertyName.substring(1);
|
|
590
|
+
}
|
|
591
|
+
|
|
592
|
+
// Consume chained dot-property access (.parentElement.style.display)
|
|
593
|
+
let chainedProps = propertyName;
|
|
594
|
+
while (
|
|
595
|
+
tokens.peek()?.kind === 'selector' &&
|
|
596
|
+
tokens.peek()!.value.startsWith('.') &&
|
|
597
|
+
/^\.[a-zA-Z_]\w*/.test(tokens.peek()!.value)
|
|
598
|
+
) {
|
|
599
|
+
chainedProps += tokens.peek()!.value; // keep the dots for chaining
|
|
600
|
+
tokens.advance();
|
|
601
|
+
}
|
|
602
|
+
|
|
603
|
+
// Check for method call — next token is '(' in the value (e.g., .getAttribute("data-id"))
|
|
604
|
+
const nextPeek = tokens.peek();
|
|
605
|
+
if (nextPeek?.kind === 'literal' && nextPeek.value.startsWith('(')) {
|
|
606
|
+
// Consume method args
|
|
607
|
+
chainedProps += nextPeek.value;
|
|
608
|
+
tokens.advance();
|
|
609
|
+
}
|
|
610
|
+
|
|
611
|
+
// Create property-path: my value -> { object: me, property: 'value' }
|
|
612
|
+
// baseRef from getPossessiveReference is always a valid reference ('me', 'you', 'it', etc.)
|
|
613
|
+
return createPropertyPath(createReference(baseRef as ReferenceValue['value']), chainedProps);
|
|
614
|
+
}
|
|
615
|
+
|
|
616
|
+
// Not a valid property, revert
|
|
617
|
+
tokens.reset(mark);
|
|
618
|
+
return null;
|
|
619
|
+
}
|
|
620
|
+
|
|
621
|
+
/**
|
|
622
|
+
* Check if a keyword is a structural keyword (preposition, control flow, etc.)
|
|
623
|
+
* that shouldn't be consumed as a property name.
|
|
624
|
+
*/
|
|
625
|
+
private isStructuralKeyword(value: string): boolean {
|
|
626
|
+
const structural = new Set([
|
|
627
|
+
// Prepositions
|
|
628
|
+
'into',
|
|
629
|
+
'in',
|
|
630
|
+
'to',
|
|
631
|
+
'from',
|
|
632
|
+
'at',
|
|
633
|
+
'by',
|
|
634
|
+
'with',
|
|
635
|
+
'without',
|
|
636
|
+
'before',
|
|
637
|
+
'after',
|
|
638
|
+
'of',
|
|
639
|
+
'as',
|
|
640
|
+
'on',
|
|
641
|
+
// Control flow
|
|
642
|
+
'then',
|
|
643
|
+
'end',
|
|
644
|
+
'else',
|
|
645
|
+
'if',
|
|
646
|
+
'repeat',
|
|
647
|
+
'while',
|
|
648
|
+
'for',
|
|
649
|
+
// Commands (shouldn't be property names)
|
|
650
|
+
'toggle',
|
|
651
|
+
'add',
|
|
652
|
+
'remove',
|
|
653
|
+
'put',
|
|
654
|
+
'set',
|
|
655
|
+
'show',
|
|
656
|
+
'hide',
|
|
657
|
+
'increment',
|
|
658
|
+
'decrement',
|
|
659
|
+
'send',
|
|
660
|
+
'trigger',
|
|
661
|
+
'call',
|
|
662
|
+
]);
|
|
663
|
+
return structural.has(value.toLowerCase());
|
|
664
|
+
}
|
|
665
|
+
|
|
666
|
+
/**
|
|
667
|
+
* Try to match a method call expression like '#dialog.showModal()'.
|
|
668
|
+
* Pattern: selector + '.' + identifier + '(' + [args] + ')'
|
|
669
|
+
* Returns an expression value if matched, or null if not.
|
|
670
|
+
*/
|
|
671
|
+
private tryMatchMethodCallExpression(tokens: TokenStream): SemanticValue | null {
|
|
672
|
+
const token = tokens.peek();
|
|
673
|
+
if (!token || token.kind !== 'selector') return null;
|
|
674
|
+
|
|
675
|
+
// Look ahead for: . identifier (
|
|
676
|
+
const mark = tokens.mark();
|
|
677
|
+
tokens.advance(); // consume selector
|
|
678
|
+
|
|
679
|
+
const dotToken = tokens.peek();
|
|
680
|
+
if (!dotToken || dotToken.kind !== 'operator' || dotToken.value !== '.') {
|
|
681
|
+
tokens.reset(mark);
|
|
682
|
+
return null;
|
|
683
|
+
}
|
|
684
|
+
tokens.advance(); // consume .
|
|
685
|
+
|
|
686
|
+
const methodToken = tokens.peek();
|
|
687
|
+
if (!methodToken || methodToken.kind !== 'identifier') {
|
|
688
|
+
tokens.reset(mark);
|
|
689
|
+
return null;
|
|
690
|
+
}
|
|
691
|
+
tokens.advance(); // consume method name
|
|
692
|
+
|
|
693
|
+
const openParen = tokens.peek();
|
|
694
|
+
if (!openParen || openParen.kind !== 'punctuation' || openParen.value !== '(') {
|
|
695
|
+
tokens.reset(mark);
|
|
696
|
+
return null;
|
|
697
|
+
}
|
|
698
|
+
tokens.advance(); // consume (
|
|
699
|
+
|
|
700
|
+
// Consume arguments until we find ) (with depth limit for security)
|
|
701
|
+
const args: string[] = [];
|
|
702
|
+
while (!tokens.isAtEnd() && args.length < PatternMatcher.MAX_METHOD_ARGS) {
|
|
703
|
+
const argToken = tokens.peek();
|
|
704
|
+
if (!argToken) break;
|
|
705
|
+
if (argToken.kind === 'punctuation' && argToken.value === ')') {
|
|
706
|
+
tokens.advance(); // consume )
|
|
707
|
+
break;
|
|
708
|
+
}
|
|
709
|
+
// Skip commas
|
|
710
|
+
if (argToken.kind === 'punctuation' && argToken.value === ',') {
|
|
711
|
+
tokens.advance();
|
|
712
|
+
continue;
|
|
713
|
+
}
|
|
714
|
+
// Collect arg value
|
|
715
|
+
args.push(argToken.value);
|
|
716
|
+
tokens.advance();
|
|
717
|
+
}
|
|
718
|
+
|
|
719
|
+
// Create expression value: #dialog.showModal()
|
|
720
|
+
const methodCall = `${token.value}.${methodToken.value}(${args.join(', ')})`;
|
|
721
|
+
return {
|
|
722
|
+
type: 'expression',
|
|
723
|
+
raw: methodCall,
|
|
724
|
+
} as SemanticValue;
|
|
725
|
+
}
|
|
726
|
+
|
|
727
|
+
/**
|
|
728
|
+
* Try to match a property access expression like 'userData.name' or 'it.data'.
|
|
729
|
+
* Pattern: (identifier | keyword) + '.' + identifier [+ '.' + identifier ...]
|
|
730
|
+
* Returns an expression value if matched, or null if not.
|
|
731
|
+
*/
|
|
732
|
+
private tryMatchPropertyAccessExpression(tokens: TokenStream): SemanticValue | null {
|
|
733
|
+
const token = tokens.peek();
|
|
734
|
+
if (!token) return null;
|
|
735
|
+
|
|
736
|
+
// Must start with an identifier or keyword reference
|
|
737
|
+
if (token.kind !== 'identifier' && token.kind !== 'keyword') return null;
|
|
738
|
+
|
|
739
|
+
// Look ahead for: . identifier
|
|
740
|
+
const mark = tokens.mark();
|
|
741
|
+
tokens.advance(); // consume first token
|
|
742
|
+
|
|
743
|
+
const dotToken = tokens.peek();
|
|
744
|
+
if (!dotToken || dotToken.kind !== 'operator' || dotToken.value !== '.') {
|
|
745
|
+
tokens.reset(mark);
|
|
746
|
+
return null;
|
|
747
|
+
}
|
|
748
|
+
tokens.advance(); // consume .
|
|
749
|
+
|
|
750
|
+
const propertyToken = tokens.peek();
|
|
751
|
+
if (!propertyToken || propertyToken.kind !== 'identifier') {
|
|
752
|
+
tokens.reset(mark);
|
|
753
|
+
return null;
|
|
754
|
+
}
|
|
755
|
+
tokens.advance(); // consume property name
|
|
756
|
+
|
|
757
|
+
// Build the property chain
|
|
758
|
+
let chain = `${token.value}.${propertyToken.value}`;
|
|
759
|
+
let depth = 1; // Already have one property access
|
|
760
|
+
|
|
761
|
+
// Continue for nested property access (e.g., userData.address.city)
|
|
762
|
+
// With depth limit for security
|
|
763
|
+
while (!tokens.isAtEnd() && depth < PatternMatcher.MAX_PROPERTY_DEPTH) {
|
|
764
|
+
const nextDot = tokens.peek();
|
|
765
|
+
if (!nextDot || nextDot.kind !== 'operator' || nextDot.value !== '.') {
|
|
766
|
+
break;
|
|
767
|
+
}
|
|
768
|
+
tokens.advance(); // consume .
|
|
769
|
+
|
|
770
|
+
const nextProp = tokens.peek();
|
|
771
|
+
if (!nextProp || nextProp.kind !== 'identifier') {
|
|
772
|
+
// Dot without property - put the dot back and stop
|
|
773
|
+
// Can't easily put a single token back, so we'll include it
|
|
774
|
+
break;
|
|
775
|
+
}
|
|
776
|
+
tokens.advance(); // consume property
|
|
777
|
+
chain += `.${nextProp.value}`;
|
|
778
|
+
depth++;
|
|
779
|
+
}
|
|
780
|
+
|
|
781
|
+
// Check for method call: chain + '(' + args + ')'
|
|
782
|
+
// e.g., me.insertBefore(draggedItem, dropTarget)
|
|
783
|
+
const openParen = tokens.peek();
|
|
784
|
+
if (openParen && openParen.kind === 'punctuation' && openParen.value === '(') {
|
|
785
|
+
tokens.advance(); // consume (
|
|
786
|
+
|
|
787
|
+
// Collect arguments (comma-separated values)
|
|
788
|
+
const args: string[] = [];
|
|
789
|
+
let argDepth = 0; // Track nested parentheses
|
|
790
|
+
while (!tokens.isAtEnd() && args.length < PatternMatcher.MAX_METHOD_ARGS) {
|
|
791
|
+
const argToken = tokens.peek();
|
|
792
|
+
if (!argToken) break;
|
|
793
|
+
|
|
794
|
+
// Handle close paren - respecting nesting
|
|
795
|
+
if (argToken.kind === 'punctuation' && argToken.value === ')') {
|
|
796
|
+
if (argDepth === 0) {
|
|
797
|
+
tokens.advance(); // consume )
|
|
798
|
+
break;
|
|
799
|
+
}
|
|
800
|
+
argDepth--;
|
|
801
|
+
}
|
|
802
|
+
// Track nested open parens
|
|
803
|
+
if (argToken.kind === 'punctuation' && argToken.value === '(') {
|
|
804
|
+
argDepth++;
|
|
805
|
+
}
|
|
806
|
+
// Skip commas between arguments
|
|
807
|
+
if (argToken.kind === 'punctuation' && argToken.value === ',') {
|
|
808
|
+
tokens.advance();
|
|
809
|
+
continue;
|
|
810
|
+
}
|
|
811
|
+
// Collect arg value
|
|
812
|
+
args.push(argToken.value);
|
|
813
|
+
tokens.advance();
|
|
814
|
+
}
|
|
815
|
+
|
|
816
|
+
// Create expression value with method call: me.insertBefore(a, b)
|
|
817
|
+
const methodCall = `${chain}(${args.join(', ')})`;
|
|
818
|
+
return {
|
|
819
|
+
type: 'expression',
|
|
820
|
+
raw: methodCall,
|
|
821
|
+
} as SemanticValue;
|
|
822
|
+
}
|
|
823
|
+
|
|
824
|
+
// Create expression value: userData.name
|
|
825
|
+
return {
|
|
826
|
+
type: 'expression',
|
|
827
|
+
raw: chain,
|
|
828
|
+
} as SemanticValue;
|
|
829
|
+
}
|
|
830
|
+
|
|
831
|
+
/**
|
|
832
|
+
* Try to match a possessive selector expression like "#element's *opacity".
|
|
833
|
+
* Pattern: selector + "'s" + (selector | identifier)
|
|
834
|
+
* Returns a property-path value if matched, or null if not.
|
|
835
|
+
*/
|
|
836
|
+
private tryMatchPossessiveSelectorExpression(tokens: TokenStream): SemanticValue | null {
|
|
837
|
+
const token = tokens.peek();
|
|
838
|
+
if (!token || token.kind !== 'selector') return null;
|
|
839
|
+
|
|
840
|
+
// Look ahead for: 's (possessive marker)
|
|
841
|
+
const mark = tokens.mark();
|
|
842
|
+
tokens.advance(); // consume selector
|
|
843
|
+
|
|
844
|
+
const possessiveToken = tokens.peek();
|
|
845
|
+
if (
|
|
846
|
+
!possessiveToken ||
|
|
847
|
+
possessiveToken.kind !== 'punctuation' ||
|
|
848
|
+
possessiveToken.value !== "'s"
|
|
849
|
+
) {
|
|
850
|
+
tokens.reset(mark);
|
|
851
|
+
return null;
|
|
852
|
+
}
|
|
853
|
+
tokens.advance(); // consume 's
|
|
854
|
+
|
|
855
|
+
const propertyToken = tokens.peek();
|
|
856
|
+
if (!propertyToken) {
|
|
857
|
+
tokens.reset(mark);
|
|
858
|
+
return null;
|
|
859
|
+
}
|
|
860
|
+
|
|
861
|
+
// Property can be a selector (*opacity) or identifier
|
|
862
|
+
if (propertyToken.kind !== 'selector' && propertyToken.kind !== 'identifier') {
|
|
863
|
+
tokens.reset(mark);
|
|
864
|
+
return null;
|
|
865
|
+
}
|
|
866
|
+
tokens.advance(); // consume property
|
|
867
|
+
|
|
868
|
+
// Create property-path: #element's *opacity
|
|
869
|
+
return createPropertyPath(createSelector(token.value), propertyToken.value);
|
|
870
|
+
}
|
|
871
|
+
|
|
872
|
+
/**
|
|
873
|
+
* Try to match a selector + property expression like "#output.innerText".
|
|
874
|
+
* This handles cases where the tokenizer produces two selector tokens:
|
|
875
|
+
* - #output (id selector)
|
|
876
|
+
* - .innerText (looks like class selector, but is actually property)
|
|
877
|
+
*
|
|
878
|
+
* Pattern: id-selector + class-selector-that-is-actually-property
|
|
879
|
+
* Returns a property-path value if matched, or null if not.
|
|
880
|
+
*/
|
|
881
|
+
private tryMatchSelectorPropertyExpression(tokens: TokenStream): SemanticValue | null {
|
|
882
|
+
const token = tokens.peek();
|
|
883
|
+
if (!token || token.kind !== 'selector') return null;
|
|
884
|
+
|
|
885
|
+
// Must be an ID selector (starts with #)
|
|
886
|
+
if (typeof token.value !== 'string' || !token.value.startsWith('#')) return null;
|
|
887
|
+
|
|
888
|
+
// Look ahead for: selector that looks like a property (.something)
|
|
889
|
+
const mark = tokens.mark();
|
|
890
|
+
tokens.advance(); // consume first selector
|
|
891
|
+
|
|
892
|
+
const propertyToken = tokens.peek();
|
|
893
|
+
if (!propertyToken || propertyToken.kind !== 'selector') {
|
|
894
|
+
tokens.reset(mark);
|
|
895
|
+
return null;
|
|
896
|
+
}
|
|
897
|
+
|
|
898
|
+
// Second token must look like a class selector (starts with .)
|
|
899
|
+
// but we interpret it as a property access
|
|
900
|
+
if (!propertyToken.value.startsWith('.')) {
|
|
901
|
+
tokens.reset(mark);
|
|
902
|
+
return null;
|
|
903
|
+
}
|
|
904
|
+
|
|
905
|
+
// Verify the next token is not a selector (to avoid consuming too many)
|
|
906
|
+
// This helps distinguish "#output.innerText" from "#box .child"
|
|
907
|
+
const peek2 = tokens.peek(1);
|
|
908
|
+
if (peek2 && peek2.kind === 'selector') {
|
|
909
|
+
// Could be a compound selector chain - only take first two
|
|
910
|
+
}
|
|
911
|
+
|
|
912
|
+
tokens.advance(); // consume property selector
|
|
913
|
+
|
|
914
|
+
// Create property-path: #output.innerText
|
|
915
|
+
// Extract property name without the leading dot
|
|
916
|
+
const propertyName = propertyToken.value.slice(1);
|
|
917
|
+
|
|
918
|
+
return createPropertyPath(createSelector(token.value), propertyName);
|
|
919
|
+
}
|
|
920
|
+
|
|
921
|
+
/**
|
|
922
|
+
* Match a group pattern token (optional sequence).
|
|
923
|
+
* When the group's leading marker isn't at the current position, scans ahead
|
|
924
|
+
* up to MAX_MARKER_SCAN tokens to find it. This allows unmarked tokens
|
|
925
|
+
* (e.g., a value like 'hello') to sit between groups without blocking later
|
|
926
|
+
* marker matches.
|
|
927
|
+
*/
|
|
928
|
+
private matchGroupToken(
|
|
929
|
+
tokens: TokenStream,
|
|
930
|
+
patternToken: PatternToken & { type: 'group' },
|
|
931
|
+
captured: Map<SemanticRole, SemanticValue>
|
|
932
|
+
): boolean {
|
|
933
|
+
const mark = tokens.mark();
|
|
934
|
+
const capturedBefore = new Set(captured.keys());
|
|
935
|
+
|
|
936
|
+
const success = this.matchTokenSequence(tokens, patternToken.tokens, captured);
|
|
937
|
+
if (success) return true;
|
|
938
|
+
|
|
939
|
+
// Reset from failed attempt
|
|
940
|
+
tokens.reset(mark);
|
|
941
|
+
for (const role of captured.keys()) {
|
|
942
|
+
if (!capturedBefore.has(role)) captured.delete(role);
|
|
943
|
+
}
|
|
944
|
+
|
|
945
|
+
if (!patternToken.optional) return false;
|
|
946
|
+
|
|
947
|
+
// Marker scan: look ahead for the group's leading marker past intervening tokens.
|
|
948
|
+
// This handles cases like "types 'hello' into #search" where 'hello' blocks the
|
|
949
|
+
// 'into' marker from being found at the current position.
|
|
950
|
+
const leadingMarker = this.getGroupLeadingMarker(patternToken);
|
|
951
|
+
if (leadingMarker) {
|
|
952
|
+
for (let offset = 1; offset <= PatternMatcher.MAX_MARKER_SCAN; offset++) {
|
|
953
|
+
const ahead = tokens.peek(offset);
|
|
954
|
+
if (!ahead) break;
|
|
955
|
+
const aheadValue = (ahead.normalized || ahead.value).toLowerCase();
|
|
956
|
+
if (aheadValue === leadingMarker) {
|
|
957
|
+
this.logger.debug(
|
|
958
|
+
' >> MARKER SCAN: found',
|
|
959
|
+
leadingMarker,
|
|
960
|
+
'at offset',
|
|
961
|
+
offset,
|
|
962
|
+
'- skipping intervening tokens'
|
|
963
|
+
);
|
|
964
|
+
// Advance past intervening tokens to reach the marker
|
|
965
|
+
for (let s = 0; s < offset; s++) tokens.advance();
|
|
966
|
+
|
|
967
|
+
// Retry group match from marker position
|
|
968
|
+
const retrySuccess = this.matchTokenSequence(tokens, patternToken.tokens, captured);
|
|
969
|
+
if (retrySuccess) return true;
|
|
970
|
+
|
|
971
|
+
// Retry failed — full reset
|
|
972
|
+
tokens.reset(mark);
|
|
973
|
+
for (const role of captured.keys()) {
|
|
974
|
+
if (!capturedBefore.has(role)) captured.delete(role);
|
|
975
|
+
}
|
|
976
|
+
break;
|
|
977
|
+
}
|
|
978
|
+
}
|
|
979
|
+
}
|
|
980
|
+
|
|
981
|
+
return true; // Optional group, just skip
|
|
982
|
+
}
|
|
983
|
+
|
|
984
|
+
/**
|
|
985
|
+
* Get the leading marker literal from an optional group.
|
|
986
|
+
* Returns the lowercase marker value, or null if the group
|
|
987
|
+
* doesn't start with a literal token (e.g., SOV groups where
|
|
988
|
+
* the role comes before the marker).
|
|
989
|
+
*/
|
|
990
|
+
private getGroupLeadingMarker(group: PatternToken & { type: 'group' }): string | null {
|
|
991
|
+
const first = group.tokens[0];
|
|
992
|
+
if (first?.type === 'literal') return first.value.toLowerCase();
|
|
993
|
+
return null;
|
|
994
|
+
}
|
|
995
|
+
|
|
996
|
+
/**
|
|
997
|
+
* Get the type of match for a token against a value.
|
|
998
|
+
* Used for confidence calculation.
|
|
999
|
+
*/
|
|
1000
|
+
private getMatchType(
|
|
1001
|
+
token: LanguageToken,
|
|
1002
|
+
value: string
|
|
1003
|
+
): 'exact' | 'normalized' | 'stem' | 'case-insensitive' | 'none' {
|
|
1004
|
+
// Exact match (highest confidence)
|
|
1005
|
+
if (token.value === value) return 'exact';
|
|
1006
|
+
|
|
1007
|
+
// Explicit keyword map normalized match (high confidence)
|
|
1008
|
+
if (token.normalized === value) return 'normalized';
|
|
1009
|
+
|
|
1010
|
+
// Morphologically normalized stem match (medium-high confidence)
|
|
1011
|
+
// Only accept if stem confidence is reasonable
|
|
1012
|
+
if (token.stem === value && token.stemConfidence !== undefined && token.stemConfidence >= 0.7) {
|
|
1013
|
+
return 'stem';
|
|
1014
|
+
}
|
|
1015
|
+
|
|
1016
|
+
// Case-insensitive match for keywords (medium confidence)
|
|
1017
|
+
if (token.kind === 'keyword' && this.safeToLowerCase(token.value) === value.toLowerCase()) {
|
|
1018
|
+
return 'case-insensitive';
|
|
1019
|
+
}
|
|
1020
|
+
|
|
1021
|
+
return 'none';
|
|
1022
|
+
}
|
|
1023
|
+
|
|
1024
|
+
/**
|
|
1025
|
+
* Collect literal values from upcoming pattern tokens that act as stop markers
|
|
1026
|
+
* for greedy role capture. Returns the set of lowercase values.
|
|
1027
|
+
*/
|
|
1028
|
+
private collectStopMarkers(patternTokens: PatternToken[], startIndex: number): Set<string> {
|
|
1029
|
+
const markers = new Set<string>();
|
|
1030
|
+
for (let j = startIndex; j < patternTokens.length; j++) {
|
|
1031
|
+
const pt = patternTokens[j];
|
|
1032
|
+
if (pt.type === 'literal') {
|
|
1033
|
+
markers.add(pt.value.toLowerCase());
|
|
1034
|
+
if (pt.alternatives) {
|
|
1035
|
+
for (const alt of pt.alternatives) {
|
|
1036
|
+
markers.add(alt.toLowerCase());
|
|
1037
|
+
}
|
|
1038
|
+
}
|
|
1039
|
+
break;
|
|
1040
|
+
}
|
|
1041
|
+
if (pt.type === 'group') {
|
|
1042
|
+
for (const gt of pt.tokens) {
|
|
1043
|
+
if (gt.type === 'literal') {
|
|
1044
|
+
markers.add(gt.value.toLowerCase());
|
|
1045
|
+
if (gt.alternatives) {
|
|
1046
|
+
for (const alt of gt.alternatives) {
|
|
1047
|
+
markers.add(alt.toLowerCase());
|
|
1048
|
+
}
|
|
1049
|
+
}
|
|
1050
|
+
break;
|
|
1051
|
+
}
|
|
1052
|
+
}
|
|
1053
|
+
break;
|
|
1054
|
+
}
|
|
1055
|
+
}
|
|
1056
|
+
return markers;
|
|
1057
|
+
}
|
|
1058
|
+
|
|
1059
|
+
/**
|
|
1060
|
+
* Check if a token matches any stop marker for greedy capture.
|
|
1061
|
+
*/
|
|
1062
|
+
private isStopMarker(token: LanguageToken, stopMarkers: Set<string>): boolean {
|
|
1063
|
+
if (stopMarkers.size === 0) return false;
|
|
1064
|
+
const value = (token.normalized || token.value).toLowerCase();
|
|
1065
|
+
return stopMarkers.has(value);
|
|
1066
|
+
}
|
|
1067
|
+
|
|
1068
|
+
/**
|
|
1069
|
+
* Track stem matches for confidence calculation.
|
|
1070
|
+
* This is set during matching and read during confidence calculation.
|
|
1071
|
+
*/
|
|
1072
|
+
private stemMatchCount: number = 0;
|
|
1073
|
+
private totalKeywordMatches: number = 0;
|
|
1074
|
+
|
|
1075
|
+
// ==========================================================================
|
|
1076
|
+
// Depth Limits for Expression Parsing (security hardening)
|
|
1077
|
+
// ==========================================================================
|
|
1078
|
+
|
|
1079
|
+
/** Maximum depth for nested property access (e.g., a.b.c.d...) */
|
|
1080
|
+
private static readonly MAX_PROPERTY_DEPTH = 10;
|
|
1081
|
+
|
|
1082
|
+
/** Maximum number of arguments in method calls */
|
|
1083
|
+
private static readonly MAX_METHOD_ARGS = 20;
|
|
1084
|
+
|
|
1085
|
+
/**
|
|
1086
|
+
* Convert a language token to a semantic value.
|
|
1087
|
+
*/
|
|
1088
|
+
private tokenToSemanticValue(token: LanguageToken): SemanticValue | null {
|
|
1089
|
+
switch (token.kind) {
|
|
1090
|
+
case 'selector':
|
|
1091
|
+
return createSelector(token.value);
|
|
1092
|
+
|
|
1093
|
+
case 'literal':
|
|
1094
|
+
return this.parseLiteralValue(token.value);
|
|
1095
|
+
|
|
1096
|
+
case 'keyword':
|
|
1097
|
+
// Keywords might be references or values
|
|
1098
|
+
const tokenValue = token.normalized || token.value;
|
|
1099
|
+
const lower = this.safeToLowerCase(tokenValue);
|
|
1100
|
+
if (isValidReference(lower)) {
|
|
1101
|
+
return createReference(lower);
|
|
1102
|
+
}
|
|
1103
|
+
return createLiteral(token.normalized || token.value);
|
|
1104
|
+
|
|
1105
|
+
case 'identifier':
|
|
1106
|
+
// Check if it's a variable reference (:varname)
|
|
1107
|
+
// Note: :varname doesn't match the ReferenceValue union but is used as a
|
|
1108
|
+
// reference token downstream — this cast preserves existing behavior
|
|
1109
|
+
if (typeof token.value === 'string' && token.value.startsWith(':')) {
|
|
1110
|
+
return createReference(token.value as ReferenceValue['value']);
|
|
1111
|
+
}
|
|
1112
|
+
// Check if it's a built-in reference
|
|
1113
|
+
const identLower = this.safeToLowerCase(token.value);
|
|
1114
|
+
if (isValidReference(identLower)) {
|
|
1115
|
+
return createReference(identLower);
|
|
1116
|
+
}
|
|
1117
|
+
// Regular identifiers are variable references - use 'expression' type
|
|
1118
|
+
// which gets converted to 'identifier' AST nodes by semantic-integration.ts
|
|
1119
|
+
return { type: 'expression', raw: token.value } as const;
|
|
1120
|
+
|
|
1121
|
+
case 'url':
|
|
1122
|
+
// URLs are treated as string literals (paths/URLs for navigation/fetch)
|
|
1123
|
+
return createLiteral(token.value, 'string');
|
|
1124
|
+
|
|
1125
|
+
default:
|
|
1126
|
+
return null;
|
|
1127
|
+
}
|
|
1128
|
+
}
|
|
1129
|
+
|
|
1130
|
+
/**
|
|
1131
|
+
* Parse a literal value (string, number, boolean).
|
|
1132
|
+
*/
|
|
1133
|
+
private parseLiteralValue(value: string): SemanticValue {
|
|
1134
|
+
// String literal
|
|
1135
|
+
if (
|
|
1136
|
+
value.startsWith('"') ||
|
|
1137
|
+
value.startsWith("'") ||
|
|
1138
|
+
value.startsWith('`') ||
|
|
1139
|
+
value.startsWith('「')
|
|
1140
|
+
) {
|
|
1141
|
+
const inner = value.slice(1, -1);
|
|
1142
|
+
return createLiteral(inner, 'string');
|
|
1143
|
+
}
|
|
1144
|
+
|
|
1145
|
+
// Boolean
|
|
1146
|
+
if (value === 'true') return createLiteral(true, 'boolean');
|
|
1147
|
+
if (value === 'false') return createLiteral(false, 'boolean');
|
|
1148
|
+
|
|
1149
|
+
// Duration (number with suffix)
|
|
1150
|
+
const durationMatch = value.match(/^(\d+(?:\.\d+)?)(ms|s|m|h)?$/);
|
|
1151
|
+
if (durationMatch) {
|
|
1152
|
+
const num = parseFloat(durationMatch[1]);
|
|
1153
|
+
const unit = durationMatch[2];
|
|
1154
|
+
if (unit) {
|
|
1155
|
+
return createLiteral(value, 'duration');
|
|
1156
|
+
}
|
|
1157
|
+
return createLiteral(num, 'number');
|
|
1158
|
+
}
|
|
1159
|
+
|
|
1160
|
+
// Plain number
|
|
1161
|
+
const num = parseFloat(value);
|
|
1162
|
+
if (!isNaN(num)) {
|
|
1163
|
+
return createLiteral(num, 'number');
|
|
1164
|
+
}
|
|
1165
|
+
|
|
1166
|
+
// Default to string
|
|
1167
|
+
return createLiteral(value, 'string');
|
|
1168
|
+
}
|
|
1169
|
+
|
|
1170
|
+
/**
|
|
1171
|
+
* Apply extraction rules to fill in static values and defaults for missing roles.
|
|
1172
|
+
*/
|
|
1173
|
+
private applyExtractionRules(
|
|
1174
|
+
pattern: LanguagePattern,
|
|
1175
|
+
captured: Map<SemanticRole, SemanticValue>
|
|
1176
|
+
): void {
|
|
1177
|
+
for (const [role, rule] of Object.entries(pattern.extraction)) {
|
|
1178
|
+
if (!captured.has(role as SemanticRole)) {
|
|
1179
|
+
if (rule.value !== undefined) {
|
|
1180
|
+
// Static value extraction (e.g., action: { value: "toggle" })
|
|
1181
|
+
captured.set(role as SemanticRole, { type: 'literal', value: rule.value });
|
|
1182
|
+
} else if (rule.default) {
|
|
1183
|
+
captured.set(role as SemanticRole, rule.default);
|
|
1184
|
+
}
|
|
1185
|
+
}
|
|
1186
|
+
}
|
|
1187
|
+
}
|
|
1188
|
+
|
|
1189
|
+
/**
|
|
1190
|
+
* Check if a pattern token is optional.
|
|
1191
|
+
*/
|
|
1192
|
+
private isOptional(patternToken: PatternToken): boolean {
|
|
1193
|
+
return patternToken.type !== 'literal' && patternToken.optional === true;
|
|
1194
|
+
}
|
|
1195
|
+
|
|
1196
|
+
/**
|
|
1197
|
+
* Calculate confidence score for a match (0-1).
|
|
1198
|
+
*
|
|
1199
|
+
* Confidence is reduced for:
|
|
1200
|
+
* - Stem matches (morphological normalization has inherent uncertainty)
|
|
1201
|
+
* - Missing optional roles (but less penalty if role has a default value)
|
|
1202
|
+
*
|
|
1203
|
+
* Confidence is increased for:
|
|
1204
|
+
* - VSO languages (Arabic) when pattern starts with a verb
|
|
1205
|
+
*/
|
|
1206
|
+
private calculateConfidence(
|
|
1207
|
+
pattern: LanguagePattern,
|
|
1208
|
+
captured: Map<SemanticRole, SemanticValue>
|
|
1209
|
+
): number {
|
|
1210
|
+
let score = 0;
|
|
1211
|
+
let maxScore = 0;
|
|
1212
|
+
|
|
1213
|
+
// Helper to check if a role has a default value in extraction rules
|
|
1214
|
+
const hasDefault = (role: SemanticRole): boolean => {
|
|
1215
|
+
return pattern.extraction?.[role]?.default !== undefined;
|
|
1216
|
+
};
|
|
1217
|
+
|
|
1218
|
+
// Score based on captured roles
|
|
1219
|
+
for (const token of pattern.template.tokens) {
|
|
1220
|
+
if (token.type === 'role') {
|
|
1221
|
+
maxScore += 1;
|
|
1222
|
+
if (captured.has(token.role)) {
|
|
1223
|
+
score += 1;
|
|
1224
|
+
}
|
|
1225
|
+
} else if (token.type === 'group') {
|
|
1226
|
+
// Group tokens are optional - weight depends on whether they have defaults
|
|
1227
|
+
for (const subToken of token.tokens) {
|
|
1228
|
+
if (subToken.type === 'role') {
|
|
1229
|
+
const roleHasDefault = hasDefault(subToken.role);
|
|
1230
|
+
const weight = 0.8; // Optional roles: 80% weight
|
|
1231
|
+
maxScore += weight;
|
|
1232
|
+
|
|
1233
|
+
if (captured.has(subToken.role)) {
|
|
1234
|
+
// Role was explicitly provided by user
|
|
1235
|
+
score += weight;
|
|
1236
|
+
} else if (roleHasDefault) {
|
|
1237
|
+
// Role has a default - give 60% partial credit since command is semantically complete
|
|
1238
|
+
// This prevents penalizing common patterns like "toggle .active" (default: me)
|
|
1239
|
+
score += weight * 0.6;
|
|
1240
|
+
}
|
|
1241
|
+
// If no default and not captured, score += 0 (true penalty for missing info)
|
|
1242
|
+
}
|
|
1243
|
+
}
|
|
1244
|
+
}
|
|
1245
|
+
}
|
|
1246
|
+
|
|
1247
|
+
let baseConfidence = maxScore > 0 ? score / maxScore : 1;
|
|
1248
|
+
|
|
1249
|
+
// Apply penalty for stem matches
|
|
1250
|
+
// Each stem match reduces confidence slightly (e.g., 5% per stem match)
|
|
1251
|
+
// This ensures exact matches are preferred over morphological matches
|
|
1252
|
+
if (this.stemMatchCount > 0 && this.totalKeywordMatches > 0) {
|
|
1253
|
+
const stemPenalty = (this.stemMatchCount / this.totalKeywordMatches) * 0.15;
|
|
1254
|
+
baseConfidence = Math.max(0.5, baseConfidence - stemPenalty);
|
|
1255
|
+
}
|
|
1256
|
+
|
|
1257
|
+
// Apply VSO confidence boost for Arabic verb-first patterns
|
|
1258
|
+
const vsoBoost = this.calculateVSOConfidenceBoost(pattern);
|
|
1259
|
+
baseConfidence = Math.min(1.0, baseConfidence + vsoBoost);
|
|
1260
|
+
|
|
1261
|
+
// Apply preposition disambiguation adjustment for Arabic
|
|
1262
|
+
const prepositionAdjustment = this.arabicPrepositionDisambiguation(pattern, captured);
|
|
1263
|
+
baseConfidence = Math.max(0.0, Math.min(1.0, baseConfidence + prepositionAdjustment));
|
|
1264
|
+
|
|
1265
|
+
return baseConfidence;
|
|
1266
|
+
}
|
|
1267
|
+
|
|
1268
|
+
/**
|
|
1269
|
+
* Calculate confidence boost for VSO (Verb-Subject-Object) language patterns.
|
|
1270
|
+
* Arabic naturally uses VSO word order, so patterns that start with a verb
|
|
1271
|
+
* should receive a confidence boost.
|
|
1272
|
+
*
|
|
1273
|
+
* Returns +0.15 confidence boost if:
|
|
1274
|
+
* - Language is Arabic ('ar')
|
|
1275
|
+
* - Pattern's first token is a verb keyword
|
|
1276
|
+
*
|
|
1277
|
+
* @param pattern The language pattern being matched
|
|
1278
|
+
* @returns Confidence boost (0 or 0.15)
|
|
1279
|
+
*/
|
|
1280
|
+
private calculateVSOConfidenceBoost(pattern: LanguagePattern): number {
|
|
1281
|
+
// Only apply to Arabic
|
|
1282
|
+
if (pattern.language !== 'ar') {
|
|
1283
|
+
return 0;
|
|
1284
|
+
}
|
|
1285
|
+
|
|
1286
|
+
// Check if first token in pattern is a literal (keyword)
|
|
1287
|
+
const firstToken = pattern.template.tokens[0];
|
|
1288
|
+
if (!firstToken || firstToken.type !== 'literal') {
|
|
1289
|
+
return 0;
|
|
1290
|
+
}
|
|
1291
|
+
|
|
1292
|
+
// List of Arabic verb keywords (command verbs)
|
|
1293
|
+
const ARABIC_VERBS = new Set([
|
|
1294
|
+
'بدل',
|
|
1295
|
+
'غير',
|
|
1296
|
+
'أضف',
|
|
1297
|
+
'أزل',
|
|
1298
|
+
'ضع',
|
|
1299
|
+
'اجعل',
|
|
1300
|
+
'عين',
|
|
1301
|
+
'زد',
|
|
1302
|
+
'انقص',
|
|
1303
|
+
'سجل',
|
|
1304
|
+
'أظهر',
|
|
1305
|
+
'أخف',
|
|
1306
|
+
'شغل',
|
|
1307
|
+
'أرسل',
|
|
1308
|
+
'ركز',
|
|
1309
|
+
'شوش',
|
|
1310
|
+
'توقف',
|
|
1311
|
+
'انسخ',
|
|
1312
|
+
'احذف',
|
|
1313
|
+
'اصنع',
|
|
1314
|
+
'انتظر',
|
|
1315
|
+
'انتقال',
|
|
1316
|
+
'أو',
|
|
1317
|
+
]);
|
|
1318
|
+
|
|
1319
|
+
// Check if first token value is a verb
|
|
1320
|
+
if (ARABIC_VERBS.has(firstToken.value)) {
|
|
1321
|
+
return 0.15;
|
|
1322
|
+
}
|
|
1323
|
+
|
|
1324
|
+
// Check alternatives
|
|
1325
|
+
if (firstToken.alternatives) {
|
|
1326
|
+
for (const alt of firstToken.alternatives) {
|
|
1327
|
+
if (ARABIC_VERBS.has(alt)) {
|
|
1328
|
+
return 0.15;
|
|
1329
|
+
}
|
|
1330
|
+
}
|
|
1331
|
+
}
|
|
1332
|
+
|
|
1333
|
+
return 0;
|
|
1334
|
+
}
|
|
1335
|
+
|
|
1336
|
+
/**
|
|
1337
|
+
* Arabic preposition disambiguation for confidence adjustment.
|
|
1338
|
+
*
|
|
1339
|
+
* Different Arabic prepositions are more or less natural for different semantic roles:
|
|
1340
|
+
* - على (on/upon) is preferred for patient/target roles (element selectors)
|
|
1341
|
+
* - إلى (to) is preferred for destination roles
|
|
1342
|
+
* - من (from) is preferred for source roles
|
|
1343
|
+
* - في (in) is preferred for location roles
|
|
1344
|
+
*
|
|
1345
|
+
* This method analyzes the prepositions used with captured semantic roles and
|
|
1346
|
+
* adjusts confidence based on idiomaticity:
|
|
1347
|
+
* - +0.10 for highly idiomatic preposition choices
|
|
1348
|
+
* - -0.10 for less natural preposition choices
|
|
1349
|
+
*
|
|
1350
|
+
* @param pattern The language pattern being matched
|
|
1351
|
+
* @param captured The captured semantic values
|
|
1352
|
+
* @returns Confidence adjustment (-0.10 to +0.10)
|
|
1353
|
+
*/
|
|
1354
|
+
private arabicPrepositionDisambiguation(
|
|
1355
|
+
pattern: LanguagePattern,
|
|
1356
|
+
captured: Map<SemanticRole, SemanticValue>
|
|
1357
|
+
): number {
|
|
1358
|
+
// Only apply to Arabic
|
|
1359
|
+
if (pattern.language !== 'ar') {
|
|
1360
|
+
return 0;
|
|
1361
|
+
}
|
|
1362
|
+
|
|
1363
|
+
let adjustment = 0;
|
|
1364
|
+
|
|
1365
|
+
// Preferred prepositions for each semantic role
|
|
1366
|
+
// Only including roles that commonly use prepositions in Arabic
|
|
1367
|
+
const PREFERRED_PREPOSITIONS: Partial<Record<SemanticRole, string[]>> = {
|
|
1368
|
+
patient: ['على'], // element selectors prefer على (on/upon)
|
|
1369
|
+
destination: ['إلى', 'الى'], // destination prefers إلى (to)
|
|
1370
|
+
source: ['من'], // source prefers من (from)
|
|
1371
|
+
agent: ['من'], // agent/by prefers من (from/by)
|
|
1372
|
+
manner: ['ب'], // manner prefers ب (with/by)
|
|
1373
|
+
style: ['ب'], // style prefers ب (with)
|
|
1374
|
+
goal: ['إلى', 'الى'], // target state prefers إلى (to)
|
|
1375
|
+
method: ['ب'], // method prefers ب (with/by)
|
|
1376
|
+
};
|
|
1377
|
+
|
|
1378
|
+
// Check each captured role for preposition metadata
|
|
1379
|
+
for (const [role, value] of captured.entries()) {
|
|
1380
|
+
// Skip if no preferred prepositions defined for this role
|
|
1381
|
+
const preferred = PREFERRED_PREPOSITIONS[role];
|
|
1382
|
+
if (!preferred || preferred.length === 0) {
|
|
1383
|
+
continue;
|
|
1384
|
+
}
|
|
1385
|
+
|
|
1386
|
+
// Check if the value has preposition metadata (from Arabic tokenizer)
|
|
1387
|
+
// This metadata is attached when a preposition particle token is consumed
|
|
1388
|
+
const metadata =
|
|
1389
|
+
'metadata' in value ? (value as { metadata: Record<string, unknown> }).metadata : undefined;
|
|
1390
|
+
if (metadata && typeof metadata.prepositionValue === 'string') {
|
|
1391
|
+
const usedPreposition = metadata.prepositionValue;
|
|
1392
|
+
|
|
1393
|
+
// Check if the used preposition is in the preferred list
|
|
1394
|
+
if (preferred.includes(usedPreposition)) {
|
|
1395
|
+
// Idiomatic choice - boost confidence
|
|
1396
|
+
adjustment += 0.1;
|
|
1397
|
+
} else {
|
|
1398
|
+
// Less natural choice - reduce confidence
|
|
1399
|
+
adjustment -= 0.1;
|
|
1400
|
+
}
|
|
1401
|
+
}
|
|
1402
|
+
}
|
|
1403
|
+
|
|
1404
|
+
// Cap total adjustment at ±0.10 (even if multiple roles analyzed)
|
|
1405
|
+
return Math.max(-0.1, Math.min(0.1, adjustment));
|
|
1406
|
+
}
|
|
1407
|
+
|
|
1408
|
+
// ===========================================================================
|
|
1409
|
+
// English Idiom Support - Noise Word Handling
|
|
1410
|
+
// ===========================================================================
|
|
1411
|
+
|
|
1412
|
+
/**
|
|
1413
|
+
* Noise words that can be skipped in English for more natural syntax.
|
|
1414
|
+
* - "the" before selectors: "toggle the .active" → "toggle .active"
|
|
1415
|
+
* - "class" after class selectors: "add the .visible class" → "add .visible"
|
|
1416
|
+
*/
|
|
1417
|
+
private static readonly ENGLISH_NOISE_WORDS = new Set(['the', 'a', 'an']);
|
|
1418
|
+
|
|
1419
|
+
/**
|
|
1420
|
+
* Skip noise words like "the" before selectors.
|
|
1421
|
+
* This enables more natural English syntax like "toggle the .active".
|
|
1422
|
+
*/
|
|
1423
|
+
private skipNoiseWords(tokens: TokenStream): void {
|
|
1424
|
+
const token = tokens.peek();
|
|
1425
|
+
if (!token) return;
|
|
1426
|
+
|
|
1427
|
+
const tokenLower = this.safeToLowerCase(token.value);
|
|
1428
|
+
|
|
1429
|
+
// Check if current token is a noise word (like "the")
|
|
1430
|
+
if (PatternMatcher.ENGLISH_NOISE_WORDS.has(tokenLower)) {
|
|
1431
|
+
// Look ahead to see if the next token is a selector
|
|
1432
|
+
const mark = tokens.mark();
|
|
1433
|
+
tokens.advance();
|
|
1434
|
+
const nextToken = tokens.peek();
|
|
1435
|
+
|
|
1436
|
+
if (nextToken && nextToken.kind === 'selector') {
|
|
1437
|
+
// Keep the position after "the" - effectively skipping it
|
|
1438
|
+
return;
|
|
1439
|
+
}
|
|
1440
|
+
|
|
1441
|
+
// Not followed by a selector, revert
|
|
1442
|
+
tokens.reset(mark);
|
|
1443
|
+
}
|
|
1444
|
+
|
|
1445
|
+
// Also handle "class" after class selectors: ".visible class" → ".visible"
|
|
1446
|
+
// This is handled when the selector has already been consumed,
|
|
1447
|
+
// so we check if current token is "class" and skip it
|
|
1448
|
+
if (tokenLower === 'class') {
|
|
1449
|
+
// Skip "class" as it's just noise after a class selector
|
|
1450
|
+
tokens.advance();
|
|
1451
|
+
}
|
|
1452
|
+
}
|
|
1453
|
+
|
|
1454
|
+
/**
|
|
1455
|
+
* Extract event modifiers from the token stream.
|
|
1456
|
+
* Event modifiers are .once, .debounce(N), .throttle(N), .queue(strategy)
|
|
1457
|
+
* that can appear after event names.
|
|
1458
|
+
*
|
|
1459
|
+
* Returns EventModifiers object or undefined if no modifiers found.
|
|
1460
|
+
*/
|
|
1461
|
+
extractEventModifiers(tokens: TokenStream): import('../types').EventModifiers | undefined {
|
|
1462
|
+
const modifiers: {
|
|
1463
|
+
once?: boolean;
|
|
1464
|
+
debounce?: number;
|
|
1465
|
+
throttle?: number;
|
|
1466
|
+
queue?: 'first' | 'last' | 'all' | 'none';
|
|
1467
|
+
from?: SemanticValue;
|
|
1468
|
+
} = {};
|
|
1469
|
+
|
|
1470
|
+
let foundModifier = false;
|
|
1471
|
+
|
|
1472
|
+
// Consume all consecutive event modifier tokens
|
|
1473
|
+
while (!tokens.isAtEnd()) {
|
|
1474
|
+
const token = tokens.peek();
|
|
1475
|
+
if (!token || token.kind !== 'event-modifier') {
|
|
1476
|
+
break;
|
|
1477
|
+
}
|
|
1478
|
+
|
|
1479
|
+
const metadata = token.metadata as
|
|
1480
|
+
| { modifierName: string; value?: number | string }
|
|
1481
|
+
| undefined;
|
|
1482
|
+
if (!metadata) {
|
|
1483
|
+
break;
|
|
1484
|
+
}
|
|
1485
|
+
|
|
1486
|
+
foundModifier = true;
|
|
1487
|
+
|
|
1488
|
+
switch (metadata.modifierName) {
|
|
1489
|
+
case 'once':
|
|
1490
|
+
modifiers.once = true;
|
|
1491
|
+
break;
|
|
1492
|
+
case 'debounce':
|
|
1493
|
+
if (typeof metadata.value === 'number') {
|
|
1494
|
+
modifiers.debounce = metadata.value;
|
|
1495
|
+
}
|
|
1496
|
+
break;
|
|
1497
|
+
case 'throttle':
|
|
1498
|
+
if (typeof metadata.value === 'number') {
|
|
1499
|
+
modifiers.throttle = metadata.value;
|
|
1500
|
+
}
|
|
1501
|
+
break;
|
|
1502
|
+
case 'queue':
|
|
1503
|
+
if (
|
|
1504
|
+
metadata.value === 'first' ||
|
|
1505
|
+
metadata.value === 'last' ||
|
|
1506
|
+
metadata.value === 'all' ||
|
|
1507
|
+
metadata.value === 'none'
|
|
1508
|
+
) {
|
|
1509
|
+
modifiers.queue = metadata.value;
|
|
1510
|
+
}
|
|
1511
|
+
break;
|
|
1512
|
+
}
|
|
1513
|
+
|
|
1514
|
+
tokens.advance();
|
|
1515
|
+
}
|
|
1516
|
+
|
|
1517
|
+
return foundModifier ? modifiers : undefined;
|
|
1518
|
+
}
|
|
1519
|
+
}
|
|
1520
|
+
|
|
1521
|
+
// =============================================================================
|
|
1522
|
+
// Convenience Functions
|
|
1523
|
+
// =============================================================================
|
|
1524
|
+
|
|
1525
|
+
/**
|
|
1526
|
+
* Singleton pattern matcher instance.
|
|
1527
|
+
*/
|
|
1528
|
+
export const patternMatcher = new PatternMatcher();
|
|
1529
|
+
|
|
1530
|
+
/**
|
|
1531
|
+
* Match tokens against a pattern.
|
|
1532
|
+
*/
|
|
1533
|
+
export function matchPattern(
|
|
1534
|
+
tokens: TokenStream,
|
|
1535
|
+
pattern: LanguagePattern
|
|
1536
|
+
): PatternMatchResult | null {
|
|
1537
|
+
return patternMatcher.matchPattern(tokens, pattern);
|
|
1538
|
+
}
|
|
1539
|
+
|
|
1540
|
+
/**
|
|
1541
|
+
* Match tokens against multiple patterns, return best match.
|
|
1542
|
+
*/
|
|
1543
|
+
export function matchBest(
|
|
1544
|
+
tokens: TokenStream,
|
|
1545
|
+
patterns: LanguagePattern[]
|
|
1546
|
+
): PatternMatchResult | null {
|
|
1547
|
+
return patternMatcher.matchBest(tokens, patterns);
|
|
1548
|
+
}
|