@lokascript/framework 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +20 -0
- package/README.md +142 -0
- package/dist/aot/aot-orchestrator.d.ts +75 -0
- package/dist/aot/aot-orchestrator.d.ts.map +1 -0
- package/dist/aot/domain-scanner.d.ts +27 -0
- package/dist/aot/domain-scanner.d.ts.map +1 -0
- package/dist/aot/index.d.ts +8 -0
- package/dist/aot/index.d.ts.map +1 -0
- package/dist/aot/types.d.ts +103 -0
- package/dist/aot/types.d.ts.map +1 -0
- package/dist/api/create-dsl.d.ts +91 -0
- package/dist/api/create-dsl.d.ts.map +1 -0
- package/dist/api/dispatcher.d.ts +108 -0
- package/dist/api/dispatcher.d.ts.map +1 -0
- package/dist/api/domain-registry.d.ts +152 -0
- package/dist/api/domain-registry.d.ts.map +1 -0
- package/dist/api/index.d.ts +7 -0
- package/dist/api/index.d.ts.map +1 -0
- package/dist/api/index.js +2082 -0
- package/dist/api/index.js.map +1 -0
- package/dist/core/index.d.ts +7 -0
- package/dist/core/index.d.ts.map +1 -0
- package/dist/core/index.js +2674 -0
- package/dist/core/index.js.map +1 -0
- package/dist/core/logger.d.ts +32 -0
- package/dist/core/logger.d.ts.map +1 -0
- package/dist/core/pattern-matching/index.d.ts +6 -0
- package/dist/core/pattern-matching/index.d.ts.map +1 -0
- package/dist/core/pattern-matching/index.js +1239 -0
- package/dist/core/pattern-matching/index.js.map +1 -0
- package/dist/core/pattern-matching/pattern-matcher.d.ts +239 -0
- package/dist/core/pattern-matching/pattern-matcher.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/index.d.ts +6 -0
- package/dist/core/pattern-matching/utils/index.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/possessive-keywords.d.ts +38 -0
- package/dist/core/pattern-matching/utils/possessive-keywords.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/type-validation.d.ts +63 -0
- package/dist/core/pattern-matching/utils/type-validation.d.ts.map +1 -0
- package/dist/core/tokenization/base-tokenizer.d.ts +344 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -0
- package/dist/core/tokenization/char-classifiers.d.ts +56 -0
- package/dist/core/tokenization/char-classifiers.d.ts.map +1 -0
- package/dist/core/tokenization/default-extractors.d.ts +48 -0
- package/dist/core/tokenization/default-extractors.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/index.d.ts +9 -0
- package/dist/core/tokenization/extractors/index.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/operator.d.ts +23 -0
- package/dist/core/tokenization/extractors/operator.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/punctuation.d.ts +22 -0
- package/dist/core/tokenization/extractors/punctuation.d.ts.map +1 -0
- package/dist/core/tokenization/extractors.d.ts +61 -0
- package/dist/core/tokenization/extractors.d.ts.map +1 -0
- package/dist/core/tokenization/index.d.ts +11 -0
- package/dist/core/tokenization/index.d.ts.map +1 -0
- package/dist/core/tokenization/index.js +1345 -0
- package/dist/core/tokenization/index.js.map +1 -0
- package/dist/core/tokenization/morphology/index.d.ts +5 -0
- package/dist/core/tokenization/morphology/index.d.ts.map +1 -0
- package/dist/core/tokenization/morphology/types.d.ts +110 -0
- package/dist/core/tokenization/morphology/types.d.ts.map +1 -0
- package/dist/core/tokenization/token-utils.d.ts +111 -0
- package/dist/core/tokenization/token-utils.d.ts.map +1 -0
- package/dist/core/types.d.ts +382 -0
- package/dist/core/types.d.ts.map +1 -0
- package/dist/core/types.js +108 -0
- package/dist/core/types.js.map +1 -0
- package/dist/generation/diagnostics.d.ts +120 -0
- package/dist/generation/diagnostics.d.ts.map +1 -0
- package/dist/generation/index.d.ts +7 -0
- package/dist/generation/index.d.ts.map +1 -0
- package/dist/generation/index.js +339 -0
- package/dist/generation/index.js.map +1 -0
- package/dist/generation/pattern-generator.d.ts +48 -0
- package/dist/generation/pattern-generator.d.ts.map +1 -0
- package/dist/generation/renderer.d.ts +115 -0
- package/dist/generation/renderer.d.ts.map +1 -0
- package/dist/grammar/index.d.ts +10 -0
- package/dist/grammar/index.d.ts.map +1 -0
- package/dist/grammar/index.js +391 -0
- package/dist/grammar/index.js.map +1 -0
- package/dist/grammar/transformer.d.ts +56 -0
- package/dist/grammar/transformer.d.ts.map +1 -0
- package/dist/grammar/types.d.ts +236 -0
- package/dist/grammar/types.d.ts.map +1 -0
- package/dist/index.cjs +4454 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.ts +46 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +4336 -0
- package/dist/index.js.map +1 -0
- package/dist/interfaces/dictionary.d.ts +82 -0
- package/dist/interfaces/dictionary.d.ts.map +1 -0
- package/dist/interfaces/index.d.ts +10 -0
- package/dist/interfaces/index.d.ts.map +1 -0
- package/dist/interfaces/profile-provider.d.ts +67 -0
- package/dist/interfaces/profile-provider.d.ts.map +1 -0
- package/dist/interfaces/value-extractor.d.ts +168 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -0
- package/dist/multilingual/index.d.ts +8 -0
- package/dist/multilingual/index.d.ts.map +1 -0
- package/dist/multilingual/index.js +1 -0
- package/dist/multilingual/index.js.map +1 -0
- package/dist/parsing/index.d.ts +8 -0
- package/dist/parsing/index.d.ts.map +1 -0
- package/dist/parsing/index.js +1415 -0
- package/dist/parsing/index.js.map +1 -0
- package/dist/parsing/multi-statement.d.ts +265 -0
- package/dist/parsing/multi-statement.d.ts.map +1 -0
- package/dist/schema/command-schema.d.ts +78 -0
- package/dist/schema/command-schema.d.ts.map +1 -0
- package/dist/schema/index.d.ts +5 -0
- package/dist/schema/index.d.ts.map +1 -0
- package/dist/schema/index.js +25 -0
- package/dist/schema/index.js.map +1 -0
- package/dist/test-setup.d.ts +9 -0
- package/dist/test-setup.d.ts.map +1 -0
- package/dist/testing/index.d.ts +50 -0
- package/dist/testing/index.d.ts.map +1 -0
- package/dist/testing/index.js +16969 -0
- package/dist/testing/index.js.map +1 -0
- package/package.json +122 -0
- package/src/__test__/fixtures/sql-dsl.ts +232 -0
- package/src/__test__/sql-integration.test.ts +189 -0
- package/src/__test__/test-utils.ts +260 -0
- package/src/aot/aot-orchestrator.test.ts +413 -0
- package/src/aot/aot-orchestrator.ts +238 -0
- package/src/aot/domain-scanner.ts +178 -0
- package/src/aot/index.ts +8 -0
- package/src/aot/types.ts +124 -0
- package/src/api/create-dsl.ts +367 -0
- package/src/api/dispatcher.test.ts +336 -0
- package/src/api/dispatcher.ts +222 -0
- package/src/api/domain-registry.test.ts +336 -0
- package/src/api/domain-registry.ts +500 -0
- package/src/api/index.ts +7 -0
- package/src/core/index.ts +7 -0
- package/src/core/logger.ts +130 -0
- package/src/core/pattern-matching/index.ts +6 -0
- package/src/core/pattern-matching/pattern-matcher.test.ts +900 -0
- package/src/core/pattern-matching/pattern-matcher.ts +1548 -0
- package/src/core/pattern-matching/pattern-matcher.ts.backup +1267 -0
- package/src/core/pattern-matching/utils/index.ts +6 -0
- package/src/core/pattern-matching/utils/possessive-keywords.ts +55 -0
- package/src/core/pattern-matching/utils/type-validation.test.ts +316 -0
- package/src/core/pattern-matching/utils/type-validation.ts +134 -0
- package/src/core/tokenization/base-tokenizer.ts +916 -0
- package/src/core/tokenization/char-classifiers.ts +79 -0
- package/src/core/tokenization/create-simple-tokenizer.test.ts +260 -0
- package/src/core/tokenization/default-extractors.ts +69 -0
- package/src/core/tokenization/extractors/index.ts +9 -0
- package/src/core/tokenization/extractors/operator.ts +75 -0
- package/src/core/tokenization/extractors/punctuation.ts +39 -0
- package/src/core/tokenization/extractors.ts +452 -0
- package/src/core/tokenization/index.ts +11 -0
- package/src/core/tokenization/morphology/index.ts +5 -0
- package/src/core/tokenization/morphology/types.ts +211 -0
- package/src/core/tokenization/token-utils.ts +252 -0
- package/src/core/types.ts +589 -0
- package/src/generation/diagnostics.test.ts +171 -0
- package/src/generation/diagnostics.ts +239 -0
- package/src/generation/index.ts +7 -0
- package/src/generation/pattern-generator.test.ts +430 -0
- package/src/generation/pattern-generator.ts +315 -0
- package/src/generation/renderer.test.ts +266 -0
- package/src/generation/renderer.ts +244 -0
- package/src/grammar/index.ts +12 -0
- package/src/grammar/transformer.ts +159 -0
- package/src/grammar/types.ts +630 -0
- package/src/index.ts +157 -0
- package/src/interfaces/dictionary.ts +123 -0
- package/src/interfaces/index.ts +10 -0
- package/src/interfaces/profile-provider.ts +88 -0
- package/src/interfaces/value-extractor.ts +435 -0
- package/src/multilingual/index.ts +9 -0
- package/src/parsing/index.ts +27 -0
- package/src/parsing/multi-statement.test.ts +480 -0
- package/src/parsing/multi-statement.ts +648 -0
- package/src/schema/command-schema.ts +118 -0
- package/src/schema/index.ts +5 -0
- package/src/test-setup.ts +45 -0
- package/src/testing/index.ts +137 -0
|
@@ -0,0 +1,435 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Value Extractor Interface - Pluggable Tokenization
|
|
3
|
+
*
|
|
4
|
+
* Extracts typed values from input strings.
|
|
5
|
+
* DSLs can provide custom extractors for their domain-specific syntax.
|
|
6
|
+
*
|
|
7
|
+
* Includes the ContextAwareExtractor extension for extractors that need
|
|
8
|
+
* access to tokenizer state (keyword maps, morphological normalizers, etc.).
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import type { MorphologicalNormalizer } from '../core/tokenization/morphology/types';
|
|
12
|
+
|
|
13
|
+
// =============================================================================
|
|
14
|
+
// Keyword Entry (needed by TokenizerContext)
|
|
15
|
+
// =============================================================================
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Keyword entry for tokenizer - maps native word to normalized English form.
|
|
19
|
+
* Re-exported here so ContextAwareExtractor consumers can use it without
|
|
20
|
+
* depending on the base-tokenizer module directly.
|
|
21
|
+
*/
|
|
22
|
+
export interface KeywordEntry {
|
|
23
|
+
readonly native: string;
|
|
24
|
+
readonly normalized: string;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
// =============================================================================
|
|
28
|
+
// Core Extractor Types
|
|
29
|
+
// =============================================================================
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* Extraction result with value and consumed length.
|
|
33
|
+
*/
|
|
34
|
+
export interface ExtractionResult {
|
|
35
|
+
/** The extracted value */
|
|
36
|
+
readonly value: string;
|
|
37
|
+
|
|
38
|
+
/** Number of characters consumed */
|
|
39
|
+
readonly length: number;
|
|
40
|
+
|
|
41
|
+
/** Optional metadata about the extraction */
|
|
42
|
+
readonly metadata?: Record<string, unknown>;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Value extractor - identifies and extracts typed values from input.
|
|
47
|
+
*/
|
|
48
|
+
export interface ValueExtractor {
|
|
49
|
+
/** Name of this extractor (for debugging) */
|
|
50
|
+
readonly name: string;
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Check if this extractor can handle input at position.
|
|
54
|
+
*
|
|
55
|
+
* @param input - Full input string
|
|
56
|
+
* @param position - Current position
|
|
57
|
+
* @returns True if this extractor should try
|
|
58
|
+
*
|
|
59
|
+
* @example
|
|
60
|
+
* // CSS selector extractor
|
|
61
|
+
* canExtract('#button', 0) // → true (starts with #)
|
|
62
|
+
* canExtract('button', 0) // → false
|
|
63
|
+
*/
|
|
64
|
+
canExtract(input: string, position: number): boolean;
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Extract value from input at position.
|
|
68
|
+
*
|
|
69
|
+
* @param input - Full input string
|
|
70
|
+
* @param position - Start position
|
|
71
|
+
* @returns Extraction result or null if extraction failed
|
|
72
|
+
*
|
|
73
|
+
* @example
|
|
74
|
+
* extract('#button', 0) // → { value: '#button', length: 7 }
|
|
75
|
+
* extract('button', 0) // → null (can't extract CSS selector)
|
|
76
|
+
*/
|
|
77
|
+
extract(input: string, position: number): ExtractionResult | null;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* String literal extractor - handles quoted strings.
|
|
82
|
+
*/
|
|
83
|
+
export class StringLiteralExtractor implements ValueExtractor {
|
|
84
|
+
readonly name = 'string-literal';
|
|
85
|
+
|
|
86
|
+
canExtract(input: string, position: number): boolean {
|
|
87
|
+
const char = input[position];
|
|
88
|
+
return (
|
|
89
|
+
char === '"' ||
|
|
90
|
+
char === "'" ||
|
|
91
|
+
char === '`' ||
|
|
92
|
+
char === '\u201C' || // Chinese double quote open "
|
|
93
|
+
char === '\u2018' // Chinese single quote open '
|
|
94
|
+
);
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
extract(input: string, position: number): ExtractionResult | null {
|
|
98
|
+
const quote = input[position];
|
|
99
|
+
|
|
100
|
+
// Chinese double quotes " ... "
|
|
101
|
+
if (quote === '\u201C') {
|
|
102
|
+
let length = 1;
|
|
103
|
+
while (position + length < input.length) {
|
|
104
|
+
if (input[position + length] === '\u201D') {
|
|
105
|
+
length++;
|
|
106
|
+
return { value: input.substring(position, position + length), length };
|
|
107
|
+
}
|
|
108
|
+
length++;
|
|
109
|
+
}
|
|
110
|
+
return null;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
// Chinese single quotes ' ... '
|
|
114
|
+
if (quote === '\u2018') {
|
|
115
|
+
let length = 1;
|
|
116
|
+
while (position + length < input.length) {
|
|
117
|
+
if (input[position + length] === '\u2019') {
|
|
118
|
+
length++;
|
|
119
|
+
return { value: input.substring(position, position + length), length };
|
|
120
|
+
}
|
|
121
|
+
length++;
|
|
122
|
+
}
|
|
123
|
+
return null;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
// ASCII quotes (same open/close, support escaping)
|
|
127
|
+
let length = 1;
|
|
128
|
+
let escaped = false;
|
|
129
|
+
|
|
130
|
+
while (position + length < input.length) {
|
|
131
|
+
const char = input[position + length];
|
|
132
|
+
|
|
133
|
+
if (escaped) {
|
|
134
|
+
escaped = false;
|
|
135
|
+
length++;
|
|
136
|
+
continue;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
if (char === '\\') {
|
|
140
|
+
escaped = true;
|
|
141
|
+
length++;
|
|
142
|
+
continue;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
if (char === quote) {
|
|
146
|
+
length++; // Include closing quote
|
|
147
|
+
return {
|
|
148
|
+
value: input.substring(position, position + length),
|
|
149
|
+
length,
|
|
150
|
+
};
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
length++;
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
// Unterminated string
|
|
157
|
+
return null;
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
/**
|
|
162
|
+
* Number extractor - handles integers and floats.
|
|
163
|
+
*/
|
|
164
|
+
export class NumberExtractor implements ValueExtractor {
|
|
165
|
+
readonly name = 'number';
|
|
166
|
+
|
|
167
|
+
canExtract(input: string, position: number): boolean {
|
|
168
|
+
return /\d/.test(input[position]);
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
extract(input: string, position: number): ExtractionResult | null {
|
|
172
|
+
let length = 0;
|
|
173
|
+
let hasDecimal = false;
|
|
174
|
+
|
|
175
|
+
while (position + length < input.length) {
|
|
176
|
+
const char = input[position + length];
|
|
177
|
+
|
|
178
|
+
if (/\d/.test(char)) {
|
|
179
|
+
length++;
|
|
180
|
+
} else if (char === '.' && !hasDecimal) {
|
|
181
|
+
hasDecimal = true;
|
|
182
|
+
length++;
|
|
183
|
+
} else {
|
|
184
|
+
break;
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
if (length === 0) return null;
|
|
189
|
+
|
|
190
|
+
const numValue = input.substring(position, position + length);
|
|
191
|
+
const afterNum = position + length;
|
|
192
|
+
|
|
193
|
+
// Check for time unit suffixes
|
|
194
|
+
if (afterNum < input.length) {
|
|
195
|
+
const remaining = input.slice(afterNum);
|
|
196
|
+
|
|
197
|
+
// CJK multi-char time units (longest first)
|
|
198
|
+
const cjkMultiUnits: { pattern: string; suffix: string }[] = [
|
|
199
|
+
{ pattern: '毫秒', suffix: 'ms' }, // Chinese milliseconds
|
|
200
|
+
{ pattern: '分钟', suffix: 'm' }, // Chinese minutes
|
|
201
|
+
{ pattern: '小时', suffix: 'h' }, // Chinese hours
|
|
202
|
+
{ pattern: 'ミリ秒', suffix: 'ms' }, // Japanese milliseconds
|
|
203
|
+
{ pattern: '時間', suffix: 'h' }, // Japanese hours
|
|
204
|
+
];
|
|
205
|
+
for (const unit of cjkMultiUnits) {
|
|
206
|
+
if (remaining.startsWith(unit.pattern)) {
|
|
207
|
+
return {
|
|
208
|
+
value: numValue + unit.suffix,
|
|
209
|
+
length: length + unit.pattern.length,
|
|
210
|
+
metadata: { hasTimeUnit: true },
|
|
211
|
+
};
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
// ASCII 'ms' (2 chars, must check before single-char)
|
|
216
|
+
if (remaining.startsWith('ms')) {
|
|
217
|
+
return {
|
|
218
|
+
value: numValue + 'ms',
|
|
219
|
+
length: length + 2,
|
|
220
|
+
metadata: { hasTimeUnit: true },
|
|
221
|
+
};
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
// CJK single-char time units
|
|
225
|
+
const cjkSingleUnits: { pattern: string; suffix: string }[] = [
|
|
226
|
+
{ pattern: '秒', suffix: 's' }, // CJK seconds
|
|
227
|
+
{ pattern: '分', suffix: 'm' }, // CJK minutes
|
|
228
|
+
];
|
|
229
|
+
for (const unit of cjkSingleUnits) {
|
|
230
|
+
if (remaining.startsWith(unit.pattern)) {
|
|
231
|
+
return {
|
|
232
|
+
value: numValue + unit.suffix,
|
|
233
|
+
length: length + 1,
|
|
234
|
+
metadata: { hasTimeUnit: true },
|
|
235
|
+
};
|
|
236
|
+
}
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
// ASCII single-char units: s, m, h (with word boundary check)
|
|
240
|
+
if (/^[smh](?![a-zA-Z])/.test(remaining)) {
|
|
241
|
+
return {
|
|
242
|
+
value: numValue + remaining[0],
|
|
243
|
+
length: length + 1,
|
|
244
|
+
metadata: { hasTimeUnit: true },
|
|
245
|
+
};
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
return { value: numValue, length };
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/**
|
|
254
|
+
* Identifier extractor - handles variable/property names.
|
|
255
|
+
*/
|
|
256
|
+
export class IdentifierExtractor implements ValueExtractor {
|
|
257
|
+
readonly name = 'identifier';
|
|
258
|
+
|
|
259
|
+
canExtract(input: string, position: number): boolean {
|
|
260
|
+
return /[a-zA-Z_]/.test(input[position]);
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
extract(input: string, position: number): ExtractionResult | null {
|
|
264
|
+
let length = 0;
|
|
265
|
+
|
|
266
|
+
while (position + length < input.length) {
|
|
267
|
+
const char = input[position + length];
|
|
268
|
+
if (/[a-zA-Z0-9_]/.test(char)) {
|
|
269
|
+
length++;
|
|
270
|
+
} else {
|
|
271
|
+
break;
|
|
272
|
+
}
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
return length > 0
|
|
276
|
+
? {
|
|
277
|
+
value: input.substring(position, position + length),
|
|
278
|
+
length,
|
|
279
|
+
}
|
|
280
|
+
: null;
|
|
281
|
+
}
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
/**
|
|
285
|
+
* Unicode identifier extractor - handles non-Latin scripts.
|
|
286
|
+
*
|
|
287
|
+
* Matches contiguous runs of Unicode letters, numbers, and combining marks
|
|
288
|
+
* that aren't ASCII (ASCII identifiers are handled by IdentifierExtractor).
|
|
289
|
+
* Essential for DSLs supporting CJK, Arabic, Cyrillic, Devanagari, etc.
|
|
290
|
+
*/
|
|
291
|
+
export class UnicodeIdentifierExtractor implements ValueExtractor {
|
|
292
|
+
readonly name = 'unicode-identifier';
|
|
293
|
+
|
|
294
|
+
canExtract(input: string, position: number): boolean {
|
|
295
|
+
const code = input.charCodeAt(position);
|
|
296
|
+
// Skip ASCII range (handled by IdentifierExtractor)
|
|
297
|
+
if (code < 0x80) return false;
|
|
298
|
+
// Match any Unicode letter
|
|
299
|
+
return /\p{L}/u.test(input[position]);
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
extract(input: string, position: number): ExtractionResult | null {
|
|
303
|
+
let length = 0;
|
|
304
|
+
|
|
305
|
+
while (position + length < input.length) {
|
|
306
|
+
const char = input[position + length];
|
|
307
|
+
// Match Unicode letters, numbers, and combining marks (e.g., Arabic diacritics)
|
|
308
|
+
if (/[\p{L}\p{N}\p{M}]/u.test(char)) {
|
|
309
|
+
length++;
|
|
310
|
+
} else {
|
|
311
|
+
break;
|
|
312
|
+
}
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
return length > 0 ? { value: input.substring(position, position + length), length } : null;
|
|
316
|
+
}
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
/**
|
|
320
|
+
* Whitespace extractor - handles spaces, tabs, newlines.
|
|
321
|
+
*/
|
|
322
|
+
export class WhitespaceExtractor implements ValueExtractor {
|
|
323
|
+
readonly name = 'whitespace';
|
|
324
|
+
|
|
325
|
+
canExtract(input: string, position: number): boolean {
|
|
326
|
+
return /\s/.test(input[position]);
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
extract(input: string, position: number): ExtractionResult | null {
|
|
330
|
+
let length = 0;
|
|
331
|
+
|
|
332
|
+
while (position + length < input.length && /\s/.test(input[position + length])) {
|
|
333
|
+
length++;
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
return length > 0
|
|
337
|
+
? {
|
|
338
|
+
value: input.substring(position, position + length),
|
|
339
|
+
length,
|
|
340
|
+
}
|
|
341
|
+
: null;
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
// =============================================================================
|
|
346
|
+
// Context-Aware Extractor System
|
|
347
|
+
// =============================================================================
|
|
348
|
+
|
|
349
|
+
/**
|
|
350
|
+
* Tokenizer context provided to context-aware extractors.
|
|
351
|
+
* Gives extractors access to tokenizer state without tight coupling.
|
|
352
|
+
*/
|
|
353
|
+
export interface TokenizerContext {
|
|
354
|
+
/** ISO 639-1 language code */
|
|
355
|
+
readonly language: string;
|
|
356
|
+
|
|
357
|
+
/** Text direction */
|
|
358
|
+
readonly direction: 'ltr' | 'rtl';
|
|
359
|
+
|
|
360
|
+
/**
|
|
361
|
+
* Look up a keyword by its native form.
|
|
362
|
+
* Returns keyword entry with normalized form, or undefined if not found.
|
|
363
|
+
*/
|
|
364
|
+
lookupKeyword(native: string): KeywordEntry | undefined;
|
|
365
|
+
|
|
366
|
+
/**
|
|
367
|
+
* Check if a word is a known keyword.
|
|
368
|
+
*/
|
|
369
|
+
isKeyword(native: string): boolean;
|
|
370
|
+
|
|
371
|
+
/**
|
|
372
|
+
* Check if a known keyword starts at the given position.
|
|
373
|
+
* Useful for word boundary detection in non-space languages.
|
|
374
|
+
*/
|
|
375
|
+
isKeywordStart(input: string, position: number): boolean;
|
|
376
|
+
|
|
377
|
+
/**
|
|
378
|
+
* Optional morphological normalizer for this language.
|
|
379
|
+
*/
|
|
380
|
+
readonly normalizer?: MorphologicalNormalizer;
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/**
|
|
384
|
+
* Context-aware extractor - has access to tokenizer state.
|
|
385
|
+
*
|
|
386
|
+
* Use this for extractors that need:
|
|
387
|
+
* - Keyword lookup (for normalization)
|
|
388
|
+
* - Morphological analysis (for conjugation handling)
|
|
389
|
+
* - Language-specific rules
|
|
390
|
+
*
|
|
391
|
+
* For stateless extractors (strings, numbers, operators), use ValueExtractor.
|
|
392
|
+
*/
|
|
393
|
+
export interface ContextAwareExtractor extends ValueExtractor {
|
|
394
|
+
/**
|
|
395
|
+
* Set the tokenizer context.
|
|
396
|
+
* Called once by the tokenizer during registration.
|
|
397
|
+
*/
|
|
398
|
+
setContext(context: TokenizerContext): void;
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
/**
|
|
402
|
+
* Type guard to check if an extractor is context-aware.
|
|
403
|
+
*/
|
|
404
|
+
export function isContextAwareExtractor(
|
|
405
|
+
extractor: ValueExtractor | ContextAwareExtractor
|
|
406
|
+
): extractor is ContextAwareExtractor {
|
|
407
|
+
return 'setContext' in extractor && typeof extractor.setContext === 'function';
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
/**
|
|
411
|
+
* Create a TokenizerContext from a tokenizer instance.
|
|
412
|
+
* Works with any object that exposes the required methods.
|
|
413
|
+
*/
|
|
414
|
+
export function createTokenizerContext(tokenizer: {
|
|
415
|
+
language: string;
|
|
416
|
+
direction: 'ltr' | 'rtl';
|
|
417
|
+
lookupKeyword(native: string): KeywordEntry | undefined;
|
|
418
|
+
isKeyword(native: string): boolean;
|
|
419
|
+
isKeywordStart(input: string, position: number): boolean;
|
|
420
|
+
normalizer?: MorphologicalNormalizer;
|
|
421
|
+
}): TokenizerContext {
|
|
422
|
+
const ctx: TokenizerContext = {
|
|
423
|
+
language: tokenizer.language,
|
|
424
|
+
direction: tokenizer.direction,
|
|
425
|
+
lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
|
|
426
|
+
isKeyword: tokenizer.isKeyword.bind(tokenizer),
|
|
427
|
+
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer),
|
|
428
|
+
};
|
|
429
|
+
|
|
430
|
+
if (tokenizer.normalizer) {
|
|
431
|
+
return { ...ctx, normalizer: tokenizer.normalizer };
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
return ctx;
|
|
435
|
+
}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Multilingual support - language profiles and templates
|
|
3
|
+
*
|
|
4
|
+
* TODO: Extract language profile templates from semantic package
|
|
5
|
+
* For now, this module is a placeholder for future language profile utilities.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
// Re-export grammar types for language profiles
|
|
9
|
+
export type { LanguageProfile, WordOrder, AdpositionType, MorphologyType } from '../grammar';
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Parsing infrastructure
|
|
3
|
+
*/
|
|
4
|
+
|
|
5
|
+
// Re-export core parsing types
|
|
6
|
+
export type { SemanticNode, TokenStream, LanguageTokenizer } from '../core/types';
|
|
7
|
+
export { PatternMatcher } from '../core/pattern-matching';
|
|
8
|
+
|
|
9
|
+
// Multi-statement parser
|
|
10
|
+
export { createMultiStatementParser, accumulateBlocks } from './multi-statement';
|
|
11
|
+
export type {
|
|
12
|
+
MultiStatementParser,
|
|
13
|
+
MultiStatementConfig,
|
|
14
|
+
MultiStatementResult,
|
|
15
|
+
SplitConfig,
|
|
16
|
+
KeywordConfig,
|
|
17
|
+
KeywordMap,
|
|
18
|
+
WordOrderHint,
|
|
19
|
+
ContinuationConfig,
|
|
20
|
+
StatementPreprocessor,
|
|
21
|
+
PreprocessorContext,
|
|
22
|
+
ParsedStatement,
|
|
23
|
+
StatementError,
|
|
24
|
+
BlockConfig,
|
|
25
|
+
BlockResult,
|
|
26
|
+
StatementBlock,
|
|
27
|
+
} from './multi-statement';
|