@lokascript/framework 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +20 -0
- package/README.md +142 -0
- package/dist/aot/aot-orchestrator.d.ts +75 -0
- package/dist/aot/aot-orchestrator.d.ts.map +1 -0
- package/dist/aot/domain-scanner.d.ts +27 -0
- package/dist/aot/domain-scanner.d.ts.map +1 -0
- package/dist/aot/index.d.ts +8 -0
- package/dist/aot/index.d.ts.map +1 -0
- package/dist/aot/types.d.ts +103 -0
- package/dist/aot/types.d.ts.map +1 -0
- package/dist/api/create-dsl.d.ts +91 -0
- package/dist/api/create-dsl.d.ts.map +1 -0
- package/dist/api/dispatcher.d.ts +108 -0
- package/dist/api/dispatcher.d.ts.map +1 -0
- package/dist/api/domain-registry.d.ts +152 -0
- package/dist/api/domain-registry.d.ts.map +1 -0
- package/dist/api/index.d.ts +7 -0
- package/dist/api/index.d.ts.map +1 -0
- package/dist/api/index.js +2082 -0
- package/dist/api/index.js.map +1 -0
- package/dist/core/index.d.ts +7 -0
- package/dist/core/index.d.ts.map +1 -0
- package/dist/core/index.js +2674 -0
- package/dist/core/index.js.map +1 -0
- package/dist/core/logger.d.ts +32 -0
- package/dist/core/logger.d.ts.map +1 -0
- package/dist/core/pattern-matching/index.d.ts +6 -0
- package/dist/core/pattern-matching/index.d.ts.map +1 -0
- package/dist/core/pattern-matching/index.js +1239 -0
- package/dist/core/pattern-matching/index.js.map +1 -0
- package/dist/core/pattern-matching/pattern-matcher.d.ts +239 -0
- package/dist/core/pattern-matching/pattern-matcher.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/index.d.ts +6 -0
- package/dist/core/pattern-matching/utils/index.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/possessive-keywords.d.ts +38 -0
- package/dist/core/pattern-matching/utils/possessive-keywords.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/type-validation.d.ts +63 -0
- package/dist/core/pattern-matching/utils/type-validation.d.ts.map +1 -0
- package/dist/core/tokenization/base-tokenizer.d.ts +344 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -0
- package/dist/core/tokenization/char-classifiers.d.ts +56 -0
- package/dist/core/tokenization/char-classifiers.d.ts.map +1 -0
- package/dist/core/tokenization/default-extractors.d.ts +48 -0
- package/dist/core/tokenization/default-extractors.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/index.d.ts +9 -0
- package/dist/core/tokenization/extractors/index.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/operator.d.ts +23 -0
- package/dist/core/tokenization/extractors/operator.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/punctuation.d.ts +22 -0
- package/dist/core/tokenization/extractors/punctuation.d.ts.map +1 -0
- package/dist/core/tokenization/extractors.d.ts +61 -0
- package/dist/core/tokenization/extractors.d.ts.map +1 -0
- package/dist/core/tokenization/index.d.ts +11 -0
- package/dist/core/tokenization/index.d.ts.map +1 -0
- package/dist/core/tokenization/index.js +1345 -0
- package/dist/core/tokenization/index.js.map +1 -0
- package/dist/core/tokenization/morphology/index.d.ts +5 -0
- package/dist/core/tokenization/morphology/index.d.ts.map +1 -0
- package/dist/core/tokenization/morphology/types.d.ts +110 -0
- package/dist/core/tokenization/morphology/types.d.ts.map +1 -0
- package/dist/core/tokenization/token-utils.d.ts +111 -0
- package/dist/core/tokenization/token-utils.d.ts.map +1 -0
- package/dist/core/types.d.ts +382 -0
- package/dist/core/types.d.ts.map +1 -0
- package/dist/core/types.js +108 -0
- package/dist/core/types.js.map +1 -0
- package/dist/generation/diagnostics.d.ts +120 -0
- package/dist/generation/diagnostics.d.ts.map +1 -0
- package/dist/generation/index.d.ts +7 -0
- package/dist/generation/index.d.ts.map +1 -0
- package/dist/generation/index.js +339 -0
- package/dist/generation/index.js.map +1 -0
- package/dist/generation/pattern-generator.d.ts +48 -0
- package/dist/generation/pattern-generator.d.ts.map +1 -0
- package/dist/generation/renderer.d.ts +115 -0
- package/dist/generation/renderer.d.ts.map +1 -0
- package/dist/grammar/index.d.ts +10 -0
- package/dist/grammar/index.d.ts.map +1 -0
- package/dist/grammar/index.js +391 -0
- package/dist/grammar/index.js.map +1 -0
- package/dist/grammar/transformer.d.ts +56 -0
- package/dist/grammar/transformer.d.ts.map +1 -0
- package/dist/grammar/types.d.ts +236 -0
- package/dist/grammar/types.d.ts.map +1 -0
- package/dist/index.cjs +4454 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.ts +46 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +4336 -0
- package/dist/index.js.map +1 -0
- package/dist/interfaces/dictionary.d.ts +82 -0
- package/dist/interfaces/dictionary.d.ts.map +1 -0
- package/dist/interfaces/index.d.ts +10 -0
- package/dist/interfaces/index.d.ts.map +1 -0
- package/dist/interfaces/profile-provider.d.ts +67 -0
- package/dist/interfaces/profile-provider.d.ts.map +1 -0
- package/dist/interfaces/value-extractor.d.ts +168 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -0
- package/dist/multilingual/index.d.ts +8 -0
- package/dist/multilingual/index.d.ts.map +1 -0
- package/dist/multilingual/index.js +1 -0
- package/dist/multilingual/index.js.map +1 -0
- package/dist/parsing/index.d.ts +8 -0
- package/dist/parsing/index.d.ts.map +1 -0
- package/dist/parsing/index.js +1415 -0
- package/dist/parsing/index.js.map +1 -0
- package/dist/parsing/multi-statement.d.ts +265 -0
- package/dist/parsing/multi-statement.d.ts.map +1 -0
- package/dist/schema/command-schema.d.ts +78 -0
- package/dist/schema/command-schema.d.ts.map +1 -0
- package/dist/schema/index.d.ts +5 -0
- package/dist/schema/index.d.ts.map +1 -0
- package/dist/schema/index.js +25 -0
- package/dist/schema/index.js.map +1 -0
- package/dist/test-setup.d.ts +9 -0
- package/dist/test-setup.d.ts.map +1 -0
- package/dist/testing/index.d.ts +50 -0
- package/dist/testing/index.d.ts.map +1 -0
- package/dist/testing/index.js +16969 -0
- package/dist/testing/index.js.map +1 -0
- package/package.json +122 -0
- package/src/__test__/fixtures/sql-dsl.ts +232 -0
- package/src/__test__/sql-integration.test.ts +189 -0
- package/src/__test__/test-utils.ts +260 -0
- package/src/aot/aot-orchestrator.test.ts +413 -0
- package/src/aot/aot-orchestrator.ts +238 -0
- package/src/aot/domain-scanner.ts +178 -0
- package/src/aot/index.ts +8 -0
- package/src/aot/types.ts +124 -0
- package/src/api/create-dsl.ts +367 -0
- package/src/api/dispatcher.test.ts +336 -0
- package/src/api/dispatcher.ts +222 -0
- package/src/api/domain-registry.test.ts +336 -0
- package/src/api/domain-registry.ts +500 -0
- package/src/api/index.ts +7 -0
- package/src/core/index.ts +7 -0
- package/src/core/logger.ts +130 -0
- package/src/core/pattern-matching/index.ts +6 -0
- package/src/core/pattern-matching/pattern-matcher.test.ts +900 -0
- package/src/core/pattern-matching/pattern-matcher.ts +1548 -0
- package/src/core/pattern-matching/pattern-matcher.ts.backup +1267 -0
- package/src/core/pattern-matching/utils/index.ts +6 -0
- package/src/core/pattern-matching/utils/possessive-keywords.ts +55 -0
- package/src/core/pattern-matching/utils/type-validation.test.ts +316 -0
- package/src/core/pattern-matching/utils/type-validation.ts +134 -0
- package/src/core/tokenization/base-tokenizer.ts +916 -0
- package/src/core/tokenization/char-classifiers.ts +79 -0
- package/src/core/tokenization/create-simple-tokenizer.test.ts +260 -0
- package/src/core/tokenization/default-extractors.ts +69 -0
- package/src/core/tokenization/extractors/index.ts +9 -0
- package/src/core/tokenization/extractors/operator.ts +75 -0
- package/src/core/tokenization/extractors/punctuation.ts +39 -0
- package/src/core/tokenization/extractors.ts +452 -0
- package/src/core/tokenization/index.ts +11 -0
- package/src/core/tokenization/morphology/index.ts +5 -0
- package/src/core/tokenization/morphology/types.ts +211 -0
- package/src/core/tokenization/token-utils.ts +252 -0
- package/src/core/types.ts +589 -0
- package/src/generation/diagnostics.test.ts +171 -0
- package/src/generation/diagnostics.ts +239 -0
- package/src/generation/index.ts +7 -0
- package/src/generation/pattern-generator.test.ts +430 -0
- package/src/generation/pattern-generator.ts +315 -0
- package/src/generation/renderer.test.ts +266 -0
- package/src/generation/renderer.ts +244 -0
- package/src/grammar/index.ts +12 -0
- package/src/grammar/transformer.ts +159 -0
- package/src/grammar/types.ts +630 -0
- package/src/index.ts +157 -0
- package/src/interfaces/dictionary.ts +123 -0
- package/src/interfaces/index.ts +10 -0
- package/src/interfaces/profile-provider.ts +88 -0
- package/src/interfaces/value-extractor.ts +435 -0
- package/src/multilingual/index.ts +9 -0
- package/src/parsing/index.ts +27 -0
- package/src/parsing/multi-statement.test.ts +480 -0
- package/src/parsing/multi-statement.ts +648 -0
- package/src/schema/command-schema.ts +118 -0
- package/src/schema/index.ts +5 -0
- package/src/test-setup.ts +45 -0
- package/src/testing/index.ts +137 -0
|
@@ -0,0 +1,452 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Extraction Utilities
|
|
3
|
+
*
|
|
4
|
+
* Pure functions for extracting CSS selectors, string literals, URLs, and numbers
|
|
5
|
+
* from input strings. These are language-independent and used by all tokenizers.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import {
|
|
9
|
+
isSelectorStart,
|
|
10
|
+
isWhitespace,
|
|
11
|
+
isAsciiIdentifierChar,
|
|
12
|
+
isAsciiLetter,
|
|
13
|
+
isQuote,
|
|
14
|
+
isDigit,
|
|
15
|
+
} from './token-utils';
|
|
16
|
+
|
|
17
|
+
// =============================================================================
|
|
18
|
+
// CSS Selector Tokenization
|
|
19
|
+
// =============================================================================
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Extract a CSS selector from the input string starting at pos.
|
|
23
|
+
* CSS selectors are universal across languages.
|
|
24
|
+
*
|
|
25
|
+
* Supported formats:
|
|
26
|
+
* - #id
|
|
27
|
+
* - .class
|
|
28
|
+
* - [attribute]
|
|
29
|
+
* - [attribute=value]
|
|
30
|
+
* - @attribute (shorthand)
|
|
31
|
+
* - *property (CSS property shorthand)
|
|
32
|
+
* - Complex selectors with combinators (limited)
|
|
33
|
+
*
|
|
34
|
+
* Method call handling:
|
|
35
|
+
* - #dialog.showModal() → stops after #dialog (method call, not compound selector)
|
|
36
|
+
* - #box.active → compound selector (no parens)
|
|
37
|
+
*/
|
|
38
|
+
export function extractCssSelector(input: string, startPos: number): string | null {
|
|
39
|
+
if (startPos >= input.length) return null;
|
|
40
|
+
|
|
41
|
+
const char = input[startPos];
|
|
42
|
+
if (!isSelectorStart(char)) return null;
|
|
43
|
+
|
|
44
|
+
let pos = startPos;
|
|
45
|
+
let selector = '';
|
|
46
|
+
|
|
47
|
+
// Handle different selector types
|
|
48
|
+
if (char === '#' || char === '.') {
|
|
49
|
+
// ID or class selector: #id, .class
|
|
50
|
+
selector += input[pos++];
|
|
51
|
+
while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
|
|
52
|
+
selector += input[pos++];
|
|
53
|
+
}
|
|
54
|
+
// Must have at least one character after prefix
|
|
55
|
+
if (selector.length <= 1) return null;
|
|
56
|
+
|
|
57
|
+
// Check for method call pattern: #id.method() or .class.method()
|
|
58
|
+
// If we see .identifier followed by (, don't consume it - it's a method call
|
|
59
|
+
if (pos < input.length && input[pos] === '.' && char === '#') {
|
|
60
|
+
// Look ahead to see if this is a method call
|
|
61
|
+
const methodStart = pos + 1;
|
|
62
|
+
let methodEnd = methodStart;
|
|
63
|
+
while (methodEnd < input.length && isAsciiIdentifierChar(input[methodEnd])) {
|
|
64
|
+
methodEnd++;
|
|
65
|
+
}
|
|
66
|
+
// If followed by (, it's a method call - stop here
|
|
67
|
+
if (methodEnd < input.length && input[methodEnd] === '(') {
|
|
68
|
+
return selector;
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
} else if (char === '[') {
|
|
72
|
+
// Attribute selector: [attr] or [attr=value] or [attr="value"]
|
|
73
|
+
// Need to track quote state to avoid counting brackets inside quotes
|
|
74
|
+
let depth = 1;
|
|
75
|
+
let inQuote = false;
|
|
76
|
+
let quoteChar: string | null = null;
|
|
77
|
+
let escaped = false;
|
|
78
|
+
|
|
79
|
+
selector += input[pos++]; // [
|
|
80
|
+
|
|
81
|
+
while (pos < input.length && depth > 0) {
|
|
82
|
+
const c = input[pos];
|
|
83
|
+
selector += c;
|
|
84
|
+
|
|
85
|
+
if (escaped) {
|
|
86
|
+
// Skip escaped character
|
|
87
|
+
escaped = false;
|
|
88
|
+
} else if (c === '\\') {
|
|
89
|
+
// Next character is escaped
|
|
90
|
+
escaped = true;
|
|
91
|
+
} else if (inQuote) {
|
|
92
|
+
// Inside a quoted string
|
|
93
|
+
if (c === quoteChar) {
|
|
94
|
+
inQuote = false;
|
|
95
|
+
quoteChar = null;
|
|
96
|
+
}
|
|
97
|
+
} else {
|
|
98
|
+
// Not inside a quoted string
|
|
99
|
+
if (c === '"' || c === "'" || c === '`') {
|
|
100
|
+
inQuote = true;
|
|
101
|
+
quoteChar = c;
|
|
102
|
+
} else if (c === '[') {
|
|
103
|
+
depth++;
|
|
104
|
+
} else if (c === ']') {
|
|
105
|
+
depth--;
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
pos++;
|
|
109
|
+
}
|
|
110
|
+
if (depth !== 0) return null;
|
|
111
|
+
} else if (char === '@') {
|
|
112
|
+
// Attribute shorthand: @disabled
|
|
113
|
+
selector += input[pos++];
|
|
114
|
+
while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
|
|
115
|
+
selector += input[pos++];
|
|
116
|
+
}
|
|
117
|
+
if (selector.length <= 1) return null;
|
|
118
|
+
} else if (char === '*') {
|
|
119
|
+
// CSS property shorthand: *display
|
|
120
|
+
selector += input[pos++];
|
|
121
|
+
while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
|
|
122
|
+
selector += input[pos++];
|
|
123
|
+
}
|
|
124
|
+
if (selector.length <= 1) return null;
|
|
125
|
+
} else if (char === '<') {
|
|
126
|
+
// HTML literal selector with optional modifiers and attributes:
|
|
127
|
+
// - <div>
|
|
128
|
+
// - <div.class>
|
|
129
|
+
// - <div#id>
|
|
130
|
+
// - <div.class#id>
|
|
131
|
+
// - <button[disabled]/>
|
|
132
|
+
// - <div.card/>
|
|
133
|
+
// - <div.class#id[attr="value"]/>
|
|
134
|
+
selector += input[pos++]; // <
|
|
135
|
+
|
|
136
|
+
// Must be followed by an identifier (tag name)
|
|
137
|
+
if (pos >= input.length || !isAsciiLetter(input[pos])) return null;
|
|
138
|
+
|
|
139
|
+
// Extract tag name
|
|
140
|
+
while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
|
|
141
|
+
selector += input[pos++];
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// Process modifiers and attributes
|
|
145
|
+
// Can have multiple .class, one #id, and multiple [attr] in any order
|
|
146
|
+
while (pos < input.length) {
|
|
147
|
+
const modChar = input[pos];
|
|
148
|
+
|
|
149
|
+
if (modChar === '.') {
|
|
150
|
+
// Class modifier
|
|
151
|
+
selector += input[pos++]; // .
|
|
152
|
+
if (pos >= input.length || !isAsciiIdentifierChar(input[pos])) {
|
|
153
|
+
return null; // Invalid - class name required after .
|
|
154
|
+
}
|
|
155
|
+
while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
|
|
156
|
+
selector += input[pos++];
|
|
157
|
+
}
|
|
158
|
+
} else if (modChar === '#') {
|
|
159
|
+
// ID modifier
|
|
160
|
+
selector += input[pos++]; // #
|
|
161
|
+
if (pos >= input.length || !isAsciiIdentifierChar(input[pos])) {
|
|
162
|
+
return null; // Invalid - ID required after #
|
|
163
|
+
}
|
|
164
|
+
while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
|
|
165
|
+
selector += input[pos++];
|
|
166
|
+
}
|
|
167
|
+
} else if (modChar === '[') {
|
|
168
|
+
// Attribute modifier: [disabled] or [type="button"]
|
|
169
|
+
// Need to track quote state to avoid counting brackets inside quotes
|
|
170
|
+
let depth = 1;
|
|
171
|
+
let inQuote = false;
|
|
172
|
+
let quoteChar: string | null = null;
|
|
173
|
+
let escaped = false;
|
|
174
|
+
|
|
175
|
+
selector += input[pos++]; // [
|
|
176
|
+
|
|
177
|
+
while (pos < input.length && depth > 0) {
|
|
178
|
+
const c = input[pos];
|
|
179
|
+
selector += c;
|
|
180
|
+
|
|
181
|
+
if (escaped) {
|
|
182
|
+
escaped = false;
|
|
183
|
+
} else if (c === '\\') {
|
|
184
|
+
escaped = true;
|
|
185
|
+
} else if (inQuote) {
|
|
186
|
+
if (c === quoteChar) {
|
|
187
|
+
inQuote = false;
|
|
188
|
+
quoteChar = null;
|
|
189
|
+
}
|
|
190
|
+
} else {
|
|
191
|
+
if (c === '"' || c === "'" || c === '`') {
|
|
192
|
+
inQuote = true;
|
|
193
|
+
quoteChar = c;
|
|
194
|
+
} else if (c === '[') {
|
|
195
|
+
depth++;
|
|
196
|
+
} else if (c === ']') {
|
|
197
|
+
depth--;
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
pos++;
|
|
201
|
+
}
|
|
202
|
+
if (depth !== 0) return null; // Unclosed bracket
|
|
203
|
+
} else {
|
|
204
|
+
// No more modifiers
|
|
205
|
+
break;
|
|
206
|
+
}
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
// Skip whitespace before optional self-closing /
|
|
210
|
+
while (pos < input.length && isWhitespace(input[pos])) {
|
|
211
|
+
selector += input[pos++];
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
// Optional self-closing /
|
|
215
|
+
if (pos < input.length && input[pos] === '/') {
|
|
216
|
+
selector += input[pos++];
|
|
217
|
+
// Skip whitespace after /
|
|
218
|
+
while (pos < input.length && isWhitespace(input[pos])) {
|
|
219
|
+
selector += input[pos++];
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
// Must end with >
|
|
224
|
+
if (pos >= input.length || input[pos] !== '>') return null;
|
|
225
|
+
selector += input[pos++]; // >
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
return selector || null;
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
// =============================================================================
|
|
232
|
+
// String Literal Tokenization
|
|
233
|
+
// =============================================================================
|
|
234
|
+
|
|
235
|
+
/**
|
|
236
|
+
* Check if a single quote at pos is a possessive marker ('s).
|
|
237
|
+
* Returns true if this looks like possessive, not a string start.
|
|
238
|
+
*
|
|
239
|
+
* Examples:
|
|
240
|
+
* - #element's *opacity → possessive (returns true)
|
|
241
|
+
* - 'hello' → string (returns false)
|
|
242
|
+
* - it's value → possessive (returns true)
|
|
243
|
+
*/
|
|
244
|
+
export function isPossessiveMarker(input: string, pos: number): boolean {
|
|
245
|
+
if (pos >= input.length || input[pos] !== "'") return false;
|
|
246
|
+
|
|
247
|
+
// Check if followed by 's' or 'S'
|
|
248
|
+
if (pos + 1 >= input.length) return false;
|
|
249
|
+
const nextChar = input[pos + 1].toLowerCase();
|
|
250
|
+
if (nextChar !== 's') return false;
|
|
251
|
+
|
|
252
|
+
// After 's, should be end, whitespace, or special char (not alphanumeric)
|
|
253
|
+
if (pos + 2 >= input.length) return true; // end of input
|
|
254
|
+
const afterS = input[pos + 2];
|
|
255
|
+
return isWhitespace(afterS) || afterS === '*' || !isAsciiIdentifierChar(afterS);
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
/**
|
|
259
|
+
* Extract a string literal from the input starting at pos.
|
|
260
|
+
* Handles both ASCII quotes and Unicode quotes.
|
|
261
|
+
*
|
|
262
|
+
* Note: Single quotes that look like possessive markers ('s) are skipped.
|
|
263
|
+
*/
|
|
264
|
+
export function extractStringLiteral(input: string, startPos: number): string | null {
|
|
265
|
+
if (startPos >= input.length) return null;
|
|
266
|
+
|
|
267
|
+
const openQuote = input[startPos];
|
|
268
|
+
if (!isQuote(openQuote)) return null;
|
|
269
|
+
|
|
270
|
+
// Check for possessive marker - don't treat as string
|
|
271
|
+
if (openQuote === "'" && isPossessiveMarker(input, startPos)) {
|
|
272
|
+
return null;
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
// Map opening quotes to closing quotes
|
|
276
|
+
const closeQuoteMap: Record<string, string> = {
|
|
277
|
+
'"': '"',
|
|
278
|
+
"'": "'",
|
|
279
|
+
'`': '`',
|
|
280
|
+
'「': '」',
|
|
281
|
+
};
|
|
282
|
+
|
|
283
|
+
const closeQuote = closeQuoteMap[openQuote];
|
|
284
|
+
if (!closeQuote) return null;
|
|
285
|
+
|
|
286
|
+
let pos = startPos + 1;
|
|
287
|
+
let literal = openQuote;
|
|
288
|
+
let escaped = false;
|
|
289
|
+
|
|
290
|
+
while (pos < input.length) {
|
|
291
|
+
const char = input[pos];
|
|
292
|
+
literal += char;
|
|
293
|
+
|
|
294
|
+
if (escaped) {
|
|
295
|
+
escaped = false;
|
|
296
|
+
} else if (char === '\\') {
|
|
297
|
+
escaped = true;
|
|
298
|
+
} else if (char === closeQuote) {
|
|
299
|
+
// Found closing quote
|
|
300
|
+
return literal;
|
|
301
|
+
}
|
|
302
|
+
pos++;
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
// Unclosed string - return what we have
|
|
306
|
+
return literal;
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
// =============================================================================
|
|
310
|
+
// URL Tokenization
|
|
311
|
+
// =============================================================================
|
|
312
|
+
|
|
313
|
+
/**
|
|
314
|
+
* Check if the input at position starts a URL.
|
|
315
|
+
* Detects: /path, ./path, ../path, //domain.com, http://, https://
|
|
316
|
+
*/
|
|
317
|
+
export function isUrlStart(input: string, pos: number): boolean {
|
|
318
|
+
if (pos >= input.length) return false;
|
|
319
|
+
|
|
320
|
+
const char = input[pos];
|
|
321
|
+
const next = input[pos + 1] || '';
|
|
322
|
+
const third = input[pos + 2] || '';
|
|
323
|
+
|
|
324
|
+
// Absolute path: /something (but not just /)
|
|
325
|
+
// Must be followed by alphanumeric or path char, not another / (that's protocol-relative)
|
|
326
|
+
if (char === '/' && next !== '/' && /[a-zA-Z0-9._-]/.test(next)) {
|
|
327
|
+
return true;
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
// Protocol-relative: //domain.com
|
|
331
|
+
if (char === '/' && next === '/' && /[a-zA-Z]/.test(third)) {
|
|
332
|
+
return true;
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
// Relative path: ./ or ../
|
|
336
|
+
if (char === '.' && (next === '/' || (next === '.' && third === '/'))) {
|
|
337
|
+
return true;
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
// Full URL: http:// or https://
|
|
341
|
+
const slice = input.slice(pos, pos + 8).toLowerCase();
|
|
342
|
+
if (slice.startsWith('http://') || slice.startsWith('https://')) {
|
|
343
|
+
return true;
|
|
344
|
+
}
|
|
345
|
+
|
|
346
|
+
return false;
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
/**
|
|
350
|
+
* Extract a URL from the input starting at pos.
|
|
351
|
+
* Handles paths, query strings, and fragments.
|
|
352
|
+
*
|
|
353
|
+
* Fragment (#) handling:
|
|
354
|
+
* - /page#section → includes fragment as part of URL
|
|
355
|
+
* - #id alone → not a URL (CSS selector)
|
|
356
|
+
*/
|
|
357
|
+
export function extractUrl(input: string, startPos: number): string | null {
|
|
358
|
+
if (!isUrlStart(input, startPos)) return null;
|
|
359
|
+
|
|
360
|
+
let pos = startPos;
|
|
361
|
+
let url = '';
|
|
362
|
+
|
|
363
|
+
// Core URL characters (RFC 3986 unreserved + sub-delims + path/query chars)
|
|
364
|
+
// Includes: letters, digits, and - . _ ~ : / ? # [ ] @ ! $ & ' ( ) * + , ; = %
|
|
365
|
+
const urlChars = /[a-zA-Z0-9/:._\-?&=%@+~!$'()*,;[\]]/;
|
|
366
|
+
|
|
367
|
+
while (pos < input.length) {
|
|
368
|
+
const char = input[pos];
|
|
369
|
+
|
|
370
|
+
// Special handling for #
|
|
371
|
+
if (char === '#') {
|
|
372
|
+
// Only include # if we have path content before it (it's a fragment)
|
|
373
|
+
// If # appears at URL start or after certain chars, stop (might be CSS selector)
|
|
374
|
+
if (url.length > 0 && /[a-zA-Z0-9/.]$/.test(url)) {
|
|
375
|
+
// Include fragment
|
|
376
|
+
url += char;
|
|
377
|
+
pos++;
|
|
378
|
+
// Consume fragment identifier (letters, digits, underscore, hyphen)
|
|
379
|
+
while (pos < input.length && /[a-zA-Z0-9_-]/.test(input[pos])) {
|
|
380
|
+
url += input[pos++];
|
|
381
|
+
}
|
|
382
|
+
}
|
|
383
|
+
// Stop either way - fragment consumed or # is separate token
|
|
384
|
+
break;
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
if (urlChars.test(char)) {
|
|
388
|
+
url += char;
|
|
389
|
+
pos++;
|
|
390
|
+
} else {
|
|
391
|
+
break;
|
|
392
|
+
}
|
|
393
|
+
}
|
|
394
|
+
|
|
395
|
+
// Minimum length validation
|
|
396
|
+
if (url.length < 2) return null;
|
|
397
|
+
|
|
398
|
+
return url;
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
// =============================================================================
|
|
402
|
+
// Number Tokenization
|
|
403
|
+
// =============================================================================
|
|
404
|
+
|
|
405
|
+
/**
|
|
406
|
+
* Extract a number from the input starting at pos.
|
|
407
|
+
* Handles integers and decimals.
|
|
408
|
+
*/
|
|
409
|
+
export function extractNumber(input: string, startPos: number): string | null {
|
|
410
|
+
if (startPos >= input.length) return null;
|
|
411
|
+
|
|
412
|
+
const char = input[startPos];
|
|
413
|
+
if (!isDigit(char) && char !== '-' && char !== '+') return null;
|
|
414
|
+
|
|
415
|
+
let pos = startPos;
|
|
416
|
+
let number = '';
|
|
417
|
+
|
|
418
|
+
// Optional sign
|
|
419
|
+
if (input[pos] === '-' || input[pos] === '+') {
|
|
420
|
+
number += input[pos++];
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
// Must have at least one digit
|
|
424
|
+
if (pos >= input.length || !isDigit(input[pos])) {
|
|
425
|
+
return null;
|
|
426
|
+
}
|
|
427
|
+
|
|
428
|
+
// Integer part
|
|
429
|
+
while (pos < input.length && isDigit(input[pos])) {
|
|
430
|
+
number += input[pos++];
|
|
431
|
+
}
|
|
432
|
+
|
|
433
|
+
// Optional decimal part
|
|
434
|
+
if (pos < input.length && input[pos] === '.') {
|
|
435
|
+
number += input[pos++];
|
|
436
|
+
while (pos < input.length && isDigit(input[pos])) {
|
|
437
|
+
number += input[pos++];
|
|
438
|
+
}
|
|
439
|
+
}
|
|
440
|
+
|
|
441
|
+
// Optional duration suffix (s, ms, m, h)
|
|
442
|
+
if (pos < input.length) {
|
|
443
|
+
const suffix = input.slice(pos, pos + 2);
|
|
444
|
+
if (suffix === 'ms') {
|
|
445
|
+
number += 'ms';
|
|
446
|
+
} else if (input[pos] === 's' || input[pos] === 'm' || input[pos] === 'h') {
|
|
447
|
+
number += input[pos];
|
|
448
|
+
}
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
return number;
|
|
452
|
+
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tokenization infrastructure for multilingual DSLs
|
|
3
|
+
*/
|
|
4
|
+
|
|
5
|
+
export * from './token-utils';
|
|
6
|
+
export * from './extractors';
|
|
7
|
+
export * from './extractors/index'; // Generic extractors (operator, punctuation)
|
|
8
|
+
export * from './default-extractors'; // Default extractor sets
|
|
9
|
+
export * from './char-classifiers';
|
|
10
|
+
export * from './base-tokenizer';
|
|
11
|
+
export * from './morphology';
|
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Morphological Normalizer Types
|
|
3
|
+
*
|
|
4
|
+
* Defines interfaces for language-specific morphological analysis.
|
|
5
|
+
* Normalizers reduce conjugated/inflected forms to canonical stems
|
|
6
|
+
* that can be matched against keyword dictionaries.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Result of morphological normalization.
|
|
11
|
+
*/
|
|
12
|
+
export interface NormalizationResult {
|
|
13
|
+
/** The extracted stem/root form */
|
|
14
|
+
readonly stem: string;
|
|
15
|
+
|
|
16
|
+
/** Confidence in the normalization (0.0-1.0) */
|
|
17
|
+
readonly confidence: number;
|
|
18
|
+
|
|
19
|
+
/** Optional metadata about the transformation */
|
|
20
|
+
readonly metadata?: NormalizationMetadata;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Metadata about morphological transformations applied.
|
|
25
|
+
*/
|
|
26
|
+
export interface NormalizationMetadata {
|
|
27
|
+
/** Prefixes that were removed */
|
|
28
|
+
readonly removedPrefixes?: readonly string[];
|
|
29
|
+
|
|
30
|
+
/** Suffixes that were removed */
|
|
31
|
+
readonly removedSuffixes?: readonly string[];
|
|
32
|
+
|
|
33
|
+
/** Type of conjugation detected */
|
|
34
|
+
readonly conjugationType?: ConjugationType;
|
|
35
|
+
|
|
36
|
+
/** Original form classification */
|
|
37
|
+
readonly originalForm?: string;
|
|
38
|
+
|
|
39
|
+
/** Applied transformation rules (for debugging) */
|
|
40
|
+
readonly appliedRules?: readonly string[];
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Types of verb conjugation/inflection.
|
|
45
|
+
*/
|
|
46
|
+
export type ConjugationType =
|
|
47
|
+
// Tense
|
|
48
|
+
| 'present'
|
|
49
|
+
| 'past'
|
|
50
|
+
| 'future'
|
|
51
|
+
| 'progressive'
|
|
52
|
+
| 'perfect'
|
|
53
|
+
// Mood
|
|
54
|
+
| 'imperative'
|
|
55
|
+
| 'subjunctive'
|
|
56
|
+
| 'conditional'
|
|
57
|
+
// Voice
|
|
58
|
+
| 'passive'
|
|
59
|
+
| 'causative'
|
|
60
|
+
// Politeness (Japanese/Korean)
|
|
61
|
+
| 'polite'
|
|
62
|
+
| 'humble'
|
|
63
|
+
| 'honorific'
|
|
64
|
+
// Form
|
|
65
|
+
| 'negative'
|
|
66
|
+
| 'potential'
|
|
67
|
+
| 'volitional'
|
|
68
|
+
// Japanese conditional forms
|
|
69
|
+
| 'conditional-tara' // たら/したら - if/when (completed action)
|
|
70
|
+
| 'conditional-to' // と/すると - when (habitual/expected)
|
|
71
|
+
| 'conditional-ba' // ば/すれば - if (hypothetical)
|
|
72
|
+
// Korean-specific
|
|
73
|
+
| 'connective' // 하고, 해서 etc.
|
|
74
|
+
| 'conditional-myeon' // -(으)면 - if/when (general conditional)
|
|
75
|
+
| 'temporal-ttae' // -(으)ㄹ 때 - when (at the time of)
|
|
76
|
+
| 'causal-nikka' // -(으)니까 - because/since
|
|
77
|
+
// Korean honorific forms (-시- infix)
|
|
78
|
+
| 'honorific-conditional' // -하시면 - if (honorific)
|
|
79
|
+
| 'honorific-temporal' // -하실 때 - when (honorific)
|
|
80
|
+
| 'honorific-causal' // -하시니까 - because (honorific)
|
|
81
|
+
| 'honorific-past' // -하셨어요 - past (honorific)
|
|
82
|
+
| 'honorific-polite' // -하십니다 - polite (honorific)
|
|
83
|
+
// Korean sequential forms
|
|
84
|
+
| 'sequential-after' // -고 나서 - after doing
|
|
85
|
+
| 'sequential-before' // -기 전에 - before doing
|
|
86
|
+
| 'immediate' // -자마자 - as soon as
|
|
87
|
+
| 'obligation' // -아야/어야 해 - must do, should do
|
|
88
|
+
// Spanish-specific
|
|
89
|
+
| 'reflexive'
|
|
90
|
+
| 'reflexive-imperative'
|
|
91
|
+
| 'gerund'
|
|
92
|
+
| 'participle'
|
|
93
|
+
// Arabic-specific
|
|
94
|
+
| 'conditional-idha' // إذا - if/when (hypothetical)
|
|
95
|
+
| 'temporal-indama' // عندما - when (temporal conjunction)
|
|
96
|
+
| 'temporal-hina' // حين - at the time of
|
|
97
|
+
| 'temporal-lamma' // لمّا - when (past emphasis)
|
|
98
|
+
| 'past-verb' // فعل ماضي - past tense verb
|
|
99
|
+
// Turkish-specific
|
|
100
|
+
| 'conditional-se' // -se/-sa - if (hypothetical)
|
|
101
|
+
| 'temporal-ince' // -ince/-ınca/-unca/-ünce - when/as
|
|
102
|
+
| 'temporal-dikce' // -dikçe/-dıkça/-dukça/-dükçe - as/while
|
|
103
|
+
| 'aorist' // -ir/-ar - habitual/general
|
|
104
|
+
| 'optative' // -eyim/-ayım/-elim/-alım - let me/us
|
|
105
|
+
| 'necessitative' // -meli/-malı - must/should
|
|
106
|
+
// Japanese request/contracted forms
|
|
107
|
+
| 'request' // てください/でください - polite request
|
|
108
|
+
| 'casual-request' // てくれ/でくれ - casual request
|
|
109
|
+
| 'contracted' // ちゃう/じゃう - contracted completion (てしまう)
|
|
110
|
+
| 'contracted-past' // ちゃった/じゃった - contracted past completion
|
|
111
|
+
// Compound
|
|
112
|
+
| 'compound' // Multi-layer suffixes (ていなかった, 하고나서였어)
|
|
113
|
+
| 'te-form' // Japanese て-form
|
|
114
|
+
| 'dictionary'; // Base/infinitive form
|
|
115
|
+
|
|
116
|
+
/**
|
|
117
|
+
* Interface for language-specific morphological normalizers.
|
|
118
|
+
*
|
|
119
|
+
* Normalizers attempt to reduce inflected word forms to their
|
|
120
|
+
* canonical stems. This enables matching conjugated verbs against
|
|
121
|
+
* keyword dictionaries that only contain base forms.
|
|
122
|
+
*
|
|
123
|
+
* Example (Japanese):
|
|
124
|
+
* 切り替えた (past) → { stem: '切り替え', confidence: 0.85 }
|
|
125
|
+
* 切り替えます (polite) → { stem: '切り替え', confidence: 0.85 }
|
|
126
|
+
*
|
|
127
|
+
* Example (Spanish):
|
|
128
|
+
* mostrarse (reflexive infinitive) → { stem: 'mostrar', confidence: 0.85 }
|
|
129
|
+
* alternando (gerund) → { stem: 'alternar', confidence: 0.85 }
|
|
130
|
+
*/
|
|
131
|
+
export interface MorphologicalNormalizer {
|
|
132
|
+
/** Language code this normalizer handles */
|
|
133
|
+
readonly language: string;
|
|
134
|
+
|
|
135
|
+
/**
|
|
136
|
+
* Normalize a word to its canonical stem form.
|
|
137
|
+
*
|
|
138
|
+
* @param word - The word to normalize
|
|
139
|
+
* @returns Normalization result with stem and confidence
|
|
140
|
+
*/
|
|
141
|
+
normalize(word: string): NormalizationResult;
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* Check if a word appears to be a verb form that can be normalized.
|
|
145
|
+
* Optional optimization to skip normalization for non-verb tokens.
|
|
146
|
+
*
|
|
147
|
+
* @param word - The word to check
|
|
148
|
+
* @returns true if the word might be a normalizable verb form
|
|
149
|
+
*/
|
|
150
|
+
isNormalizable?(word: string): boolean;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/**
|
|
154
|
+
* Configuration for suffix-based normalization rules.
|
|
155
|
+
* Used by agglutinative languages (Japanese, Korean, Turkish).
|
|
156
|
+
*/
|
|
157
|
+
export interface SuffixRule {
|
|
158
|
+
/** The suffix pattern to match */
|
|
159
|
+
readonly pattern: string;
|
|
160
|
+
|
|
161
|
+
/** Confidence when this suffix is stripped */
|
|
162
|
+
readonly confidence: number;
|
|
163
|
+
|
|
164
|
+
/** What to replace the suffix with (empty string for simple removal) */
|
|
165
|
+
readonly replacement?: string;
|
|
166
|
+
|
|
167
|
+
/** Conjugation type this suffix indicates */
|
|
168
|
+
readonly conjugationType?: ConjugationType;
|
|
169
|
+
|
|
170
|
+
/** Minimum stem length after stripping (to avoid over-stripping) */
|
|
171
|
+
readonly minStemLength?: number;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* Configuration for prefix-based normalization rules.
|
|
176
|
+
* Used primarily by Arabic for article/conjunction prefixes.
|
|
177
|
+
*/
|
|
178
|
+
export interface PrefixRule {
|
|
179
|
+
/** The prefix pattern to match */
|
|
180
|
+
readonly pattern: string;
|
|
181
|
+
|
|
182
|
+
/** Confidence penalty when this prefix is stripped */
|
|
183
|
+
readonly confidencePenalty: number;
|
|
184
|
+
|
|
185
|
+
/** What the prefix indicates (for metadata) */
|
|
186
|
+
readonly prefixType?: 'article' | 'conjunction' | 'preposition' | 'verb-marker';
|
|
187
|
+
|
|
188
|
+
/** Minimum remaining characters after stripping (to avoid over-stripping) */
|
|
189
|
+
readonly minRemaining?: number;
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/**
|
|
193
|
+
* Helper to create a "no change" normalization result.
|
|
194
|
+
*/
|
|
195
|
+
export function noChange(word: string): NormalizationResult {
|
|
196
|
+
return { stem: word, confidence: 1.0 };
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/**
|
|
200
|
+
* Helper to create a normalization result with metadata.
|
|
201
|
+
*/
|
|
202
|
+
export function normalized(
|
|
203
|
+
stem: string,
|
|
204
|
+
confidence: number,
|
|
205
|
+
metadata?: NormalizationMetadata
|
|
206
|
+
): NormalizationResult {
|
|
207
|
+
if (metadata) {
|
|
208
|
+
return { stem, confidence, metadata };
|
|
209
|
+
}
|
|
210
|
+
return { stem, confidence };
|
|
211
|
+
}
|