@lokascript/framework 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +20 -0
- package/README.md +142 -0
- package/dist/aot/aot-orchestrator.d.ts +75 -0
- package/dist/aot/aot-orchestrator.d.ts.map +1 -0
- package/dist/aot/domain-scanner.d.ts +27 -0
- package/dist/aot/domain-scanner.d.ts.map +1 -0
- package/dist/aot/index.d.ts +8 -0
- package/dist/aot/index.d.ts.map +1 -0
- package/dist/aot/types.d.ts +103 -0
- package/dist/aot/types.d.ts.map +1 -0
- package/dist/api/create-dsl.d.ts +91 -0
- package/dist/api/create-dsl.d.ts.map +1 -0
- package/dist/api/dispatcher.d.ts +108 -0
- package/dist/api/dispatcher.d.ts.map +1 -0
- package/dist/api/domain-registry.d.ts +152 -0
- package/dist/api/domain-registry.d.ts.map +1 -0
- package/dist/api/index.d.ts +7 -0
- package/dist/api/index.d.ts.map +1 -0
- package/dist/api/index.js +2082 -0
- package/dist/api/index.js.map +1 -0
- package/dist/core/index.d.ts +7 -0
- package/dist/core/index.d.ts.map +1 -0
- package/dist/core/index.js +2674 -0
- package/dist/core/index.js.map +1 -0
- package/dist/core/logger.d.ts +32 -0
- package/dist/core/logger.d.ts.map +1 -0
- package/dist/core/pattern-matching/index.d.ts +6 -0
- package/dist/core/pattern-matching/index.d.ts.map +1 -0
- package/dist/core/pattern-matching/index.js +1239 -0
- package/dist/core/pattern-matching/index.js.map +1 -0
- package/dist/core/pattern-matching/pattern-matcher.d.ts +239 -0
- package/dist/core/pattern-matching/pattern-matcher.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/index.d.ts +6 -0
- package/dist/core/pattern-matching/utils/index.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/possessive-keywords.d.ts +38 -0
- package/dist/core/pattern-matching/utils/possessive-keywords.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/type-validation.d.ts +63 -0
- package/dist/core/pattern-matching/utils/type-validation.d.ts.map +1 -0
- package/dist/core/tokenization/base-tokenizer.d.ts +344 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -0
- package/dist/core/tokenization/char-classifiers.d.ts +56 -0
- package/dist/core/tokenization/char-classifiers.d.ts.map +1 -0
- package/dist/core/tokenization/default-extractors.d.ts +48 -0
- package/dist/core/tokenization/default-extractors.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/index.d.ts +9 -0
- package/dist/core/tokenization/extractors/index.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/operator.d.ts +23 -0
- package/dist/core/tokenization/extractors/operator.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/punctuation.d.ts +22 -0
- package/dist/core/tokenization/extractors/punctuation.d.ts.map +1 -0
- package/dist/core/tokenization/extractors.d.ts +61 -0
- package/dist/core/tokenization/extractors.d.ts.map +1 -0
- package/dist/core/tokenization/index.d.ts +11 -0
- package/dist/core/tokenization/index.d.ts.map +1 -0
- package/dist/core/tokenization/index.js +1345 -0
- package/dist/core/tokenization/index.js.map +1 -0
- package/dist/core/tokenization/morphology/index.d.ts +5 -0
- package/dist/core/tokenization/morphology/index.d.ts.map +1 -0
- package/dist/core/tokenization/morphology/types.d.ts +110 -0
- package/dist/core/tokenization/morphology/types.d.ts.map +1 -0
- package/dist/core/tokenization/token-utils.d.ts +111 -0
- package/dist/core/tokenization/token-utils.d.ts.map +1 -0
- package/dist/core/types.d.ts +382 -0
- package/dist/core/types.d.ts.map +1 -0
- package/dist/core/types.js +108 -0
- package/dist/core/types.js.map +1 -0
- package/dist/generation/diagnostics.d.ts +120 -0
- package/dist/generation/diagnostics.d.ts.map +1 -0
- package/dist/generation/index.d.ts +7 -0
- package/dist/generation/index.d.ts.map +1 -0
- package/dist/generation/index.js +339 -0
- package/dist/generation/index.js.map +1 -0
- package/dist/generation/pattern-generator.d.ts +48 -0
- package/dist/generation/pattern-generator.d.ts.map +1 -0
- package/dist/generation/renderer.d.ts +115 -0
- package/dist/generation/renderer.d.ts.map +1 -0
- package/dist/grammar/index.d.ts +10 -0
- package/dist/grammar/index.d.ts.map +1 -0
- package/dist/grammar/index.js +391 -0
- package/dist/grammar/index.js.map +1 -0
- package/dist/grammar/transformer.d.ts +56 -0
- package/dist/grammar/transformer.d.ts.map +1 -0
- package/dist/grammar/types.d.ts +236 -0
- package/dist/grammar/types.d.ts.map +1 -0
- package/dist/index.cjs +4454 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.ts +46 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +4336 -0
- package/dist/index.js.map +1 -0
- package/dist/interfaces/dictionary.d.ts +82 -0
- package/dist/interfaces/dictionary.d.ts.map +1 -0
- package/dist/interfaces/index.d.ts +10 -0
- package/dist/interfaces/index.d.ts.map +1 -0
- package/dist/interfaces/profile-provider.d.ts +67 -0
- package/dist/interfaces/profile-provider.d.ts.map +1 -0
- package/dist/interfaces/value-extractor.d.ts +168 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -0
- package/dist/multilingual/index.d.ts +8 -0
- package/dist/multilingual/index.d.ts.map +1 -0
- package/dist/multilingual/index.js +1 -0
- package/dist/multilingual/index.js.map +1 -0
- package/dist/parsing/index.d.ts +8 -0
- package/dist/parsing/index.d.ts.map +1 -0
- package/dist/parsing/index.js +1415 -0
- package/dist/parsing/index.js.map +1 -0
- package/dist/parsing/multi-statement.d.ts +265 -0
- package/dist/parsing/multi-statement.d.ts.map +1 -0
- package/dist/schema/command-schema.d.ts +78 -0
- package/dist/schema/command-schema.d.ts.map +1 -0
- package/dist/schema/index.d.ts +5 -0
- package/dist/schema/index.d.ts.map +1 -0
- package/dist/schema/index.js +25 -0
- package/dist/schema/index.js.map +1 -0
- package/dist/test-setup.d.ts +9 -0
- package/dist/test-setup.d.ts.map +1 -0
- package/dist/testing/index.d.ts +50 -0
- package/dist/testing/index.d.ts.map +1 -0
- package/dist/testing/index.js +16969 -0
- package/dist/testing/index.js.map +1 -0
- package/package.json +122 -0
- package/src/__test__/fixtures/sql-dsl.ts +232 -0
- package/src/__test__/sql-integration.test.ts +189 -0
- package/src/__test__/test-utils.ts +260 -0
- package/src/aot/aot-orchestrator.test.ts +413 -0
- package/src/aot/aot-orchestrator.ts +238 -0
- package/src/aot/domain-scanner.ts +178 -0
- package/src/aot/index.ts +8 -0
- package/src/aot/types.ts +124 -0
- package/src/api/create-dsl.ts +367 -0
- package/src/api/dispatcher.test.ts +336 -0
- package/src/api/dispatcher.ts +222 -0
- package/src/api/domain-registry.test.ts +336 -0
- package/src/api/domain-registry.ts +500 -0
- package/src/api/index.ts +7 -0
- package/src/core/index.ts +7 -0
- package/src/core/logger.ts +130 -0
- package/src/core/pattern-matching/index.ts +6 -0
- package/src/core/pattern-matching/pattern-matcher.test.ts +900 -0
- package/src/core/pattern-matching/pattern-matcher.ts +1548 -0
- package/src/core/pattern-matching/pattern-matcher.ts.backup +1267 -0
- package/src/core/pattern-matching/utils/index.ts +6 -0
- package/src/core/pattern-matching/utils/possessive-keywords.ts +55 -0
- package/src/core/pattern-matching/utils/type-validation.test.ts +316 -0
- package/src/core/pattern-matching/utils/type-validation.ts +134 -0
- package/src/core/tokenization/base-tokenizer.ts +916 -0
- package/src/core/tokenization/char-classifiers.ts +79 -0
- package/src/core/tokenization/create-simple-tokenizer.test.ts +260 -0
- package/src/core/tokenization/default-extractors.ts +69 -0
- package/src/core/tokenization/extractors/index.ts +9 -0
- package/src/core/tokenization/extractors/operator.ts +75 -0
- package/src/core/tokenization/extractors/punctuation.ts +39 -0
- package/src/core/tokenization/extractors.ts +452 -0
- package/src/core/tokenization/index.ts +11 -0
- package/src/core/tokenization/morphology/index.ts +5 -0
- package/src/core/tokenization/morphology/types.ts +211 -0
- package/src/core/tokenization/token-utils.ts +252 -0
- package/src/core/types.ts +589 -0
- package/src/generation/diagnostics.test.ts +171 -0
- package/src/generation/diagnostics.ts +239 -0
- package/src/generation/index.ts +7 -0
- package/src/generation/pattern-generator.test.ts +430 -0
- package/src/generation/pattern-generator.ts +315 -0
- package/src/generation/renderer.test.ts +266 -0
- package/src/generation/renderer.ts +244 -0
- package/src/grammar/index.ts +12 -0
- package/src/grammar/transformer.ts +159 -0
- package/src/grammar/types.ts +630 -0
- package/src/index.ts +157 -0
- package/src/interfaces/dictionary.ts +123 -0
- package/src/interfaces/index.ts +10 -0
- package/src/interfaces/profile-provider.ts +88 -0
- package/src/interfaces/value-extractor.ts +435 -0
- package/src/multilingual/index.ts +9 -0
- package/src/parsing/index.ts +27 -0
- package/src/parsing/multi-statement.test.ts +480 -0
- package/src/parsing/multi-statement.ts +648 -0
- package/src/schema/command-schema.ts +118 -0
- package/src/schema/index.ts +5 -0
- package/src/test-setup.ts +45 -0
- package/src/testing/index.ts +137 -0
package/dist/index.js
ADDED
|
@@ -0,0 +1,4336 @@
|
|
|
1
|
+
// src/core/types.ts
|
|
2
|
+
function createLiteral(value, dataType) {
|
|
3
|
+
return dataType ? { type: "literal", value, dataType } : { type: "literal", value };
|
|
4
|
+
}
|
|
5
|
+
function createSelector(value, selectorKind) {
|
|
6
|
+
return selectorKind ? { type: "selector", value, selectorKind } : { type: "selector", value };
|
|
7
|
+
}
|
|
8
|
+
function createReference(value) {
|
|
9
|
+
return { type: "reference", value };
|
|
10
|
+
}
|
|
11
|
+
function createPropertyPath(object, property) {
|
|
12
|
+
return { type: "property-path", object, property };
|
|
13
|
+
}
|
|
14
|
+
function createExpression(raw) {
|
|
15
|
+
return { type: "expression", raw };
|
|
16
|
+
}
|
|
17
|
+
function createCommandNode(action, roles, metadata) {
|
|
18
|
+
const rolesMap = roles instanceof Map ? roles : new Map(Object.entries(roles));
|
|
19
|
+
const node = {
|
|
20
|
+
kind: "command",
|
|
21
|
+
action,
|
|
22
|
+
roles: rolesMap
|
|
23
|
+
};
|
|
24
|
+
if (metadata) {
|
|
25
|
+
return { ...node, metadata };
|
|
26
|
+
}
|
|
27
|
+
return node;
|
|
28
|
+
}
|
|
29
|
+
function createEventHandlerNode(action, roles, body, metadata, eventModifiers) {
|
|
30
|
+
const rolesMap = roles instanceof Map ? roles : new Map(Object.entries(roles));
|
|
31
|
+
const base = {
|
|
32
|
+
kind: "event-handler",
|
|
33
|
+
action,
|
|
34
|
+
roles: rolesMap,
|
|
35
|
+
body
|
|
36
|
+
};
|
|
37
|
+
return {
|
|
38
|
+
...base,
|
|
39
|
+
...eventModifiers && { eventModifiers },
|
|
40
|
+
...metadata && { metadata }
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
function createConditionalNode(action, roles, thenBranch, elseBranch, metadata) {
|
|
44
|
+
const rolesMap = roles instanceof Map ? roles : new Map(Object.entries(roles));
|
|
45
|
+
const base = {
|
|
46
|
+
kind: "conditional",
|
|
47
|
+
action,
|
|
48
|
+
roles: rolesMap,
|
|
49
|
+
thenBranch
|
|
50
|
+
};
|
|
51
|
+
return {
|
|
52
|
+
...base,
|
|
53
|
+
...elseBranch && { elseBranch },
|
|
54
|
+
...metadata && { metadata }
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
function createCompoundNode(statements, chainType = "sequential", metadata) {
|
|
58
|
+
const base = {
|
|
59
|
+
kind: "compound",
|
|
60
|
+
action: "compound",
|
|
61
|
+
roles: /* @__PURE__ */ new Map(),
|
|
62
|
+
statements,
|
|
63
|
+
chainType
|
|
64
|
+
};
|
|
65
|
+
return metadata ? { ...base, metadata } : base;
|
|
66
|
+
}
|
|
67
|
+
function extractValue(value) {
|
|
68
|
+
if ("raw" in value && value.raw !== void 0) return String(value.raw);
|
|
69
|
+
if ("value" in value && value.value !== void 0) return String(value.value);
|
|
70
|
+
if (value.type === "property-path") return `${extractValue(value.object)}.${value.property}`;
|
|
71
|
+
return "";
|
|
72
|
+
}
|
|
73
|
+
function extractRoleValue(node, role) {
|
|
74
|
+
const value = node.roles.get(role);
|
|
75
|
+
if (!value) return "";
|
|
76
|
+
return extractValue(value);
|
|
77
|
+
}
|
|
78
|
+
function createLoopNode(action, roles, loopVariant, body, loopVariable, indexVariable, metadata) {
|
|
79
|
+
const rolesMap = roles instanceof Map ? roles : new Map(Object.entries(roles));
|
|
80
|
+
const base = {
|
|
81
|
+
kind: "loop",
|
|
82
|
+
action,
|
|
83
|
+
roles: rolesMap,
|
|
84
|
+
loopVariant,
|
|
85
|
+
body
|
|
86
|
+
};
|
|
87
|
+
return {
|
|
88
|
+
...base,
|
|
89
|
+
...loopVariable && { loopVariable },
|
|
90
|
+
...indexVariable && { indexVariable },
|
|
91
|
+
...metadata && { metadata }
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
// src/core/logger.ts
|
|
96
|
+
var DebugLogger = class {
|
|
97
|
+
constructor(namespace) {
|
|
98
|
+
this.namespace = namespace;
|
|
99
|
+
this.enabled = this.checkEnabled();
|
|
100
|
+
}
|
|
101
|
+
checkEnabled() {
|
|
102
|
+
if (typeof process !== "undefined" && process.env) {
|
|
103
|
+
const DEBUG = process.env.DEBUG || "";
|
|
104
|
+
return DEBUG === "*" || DEBUG.includes("framework:*") || DEBUG.includes(`framework:${this.namespace}`);
|
|
105
|
+
}
|
|
106
|
+
return false;
|
|
107
|
+
}
|
|
108
|
+
log(level, message, ...args) {
|
|
109
|
+
if (!this.enabled) return;
|
|
110
|
+
const prefix = `[framework:${this.namespace}]`;
|
|
111
|
+
const timestamp = (/* @__PURE__ */ new Date()).toISOString();
|
|
112
|
+
switch (level) {
|
|
113
|
+
case "debug":
|
|
114
|
+
console.debug(`${timestamp} ${prefix} DEBUG:`, message, ...args);
|
|
115
|
+
break;
|
|
116
|
+
case "info":
|
|
117
|
+
console.info(`${timestamp} ${prefix} INFO:`, message, ...args);
|
|
118
|
+
break;
|
|
119
|
+
case "warn":
|
|
120
|
+
console.warn(`${timestamp} ${prefix} WARN:`, message, ...args);
|
|
121
|
+
break;
|
|
122
|
+
case "error":
|
|
123
|
+
console.error(`${timestamp} ${prefix} ERROR:`, message, ...args);
|
|
124
|
+
break;
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
debug(message, ...args) {
|
|
128
|
+
this.log("debug", message, ...args);
|
|
129
|
+
}
|
|
130
|
+
info(message, ...args) {
|
|
131
|
+
this.log("info", message, ...args);
|
|
132
|
+
}
|
|
133
|
+
warn(message, ...args) {
|
|
134
|
+
this.log("warn", message, ...args);
|
|
135
|
+
}
|
|
136
|
+
error(message, ...args) {
|
|
137
|
+
this.log("error", message, ...args);
|
|
138
|
+
}
|
|
139
|
+
/**
|
|
140
|
+
* Check if logging is enabled for this namespace.
|
|
141
|
+
*/
|
|
142
|
+
isEnabled() {
|
|
143
|
+
return this.enabled;
|
|
144
|
+
}
|
|
145
|
+
/**
|
|
146
|
+
* Programmatically enable logging for this namespace.
|
|
147
|
+
*/
|
|
148
|
+
enable() {
|
|
149
|
+
this.enabled = true;
|
|
150
|
+
}
|
|
151
|
+
/**
|
|
152
|
+
* Programmatically disable logging for this namespace.
|
|
153
|
+
*/
|
|
154
|
+
disable() {
|
|
155
|
+
this.enabled = false;
|
|
156
|
+
}
|
|
157
|
+
};
|
|
158
|
+
function createLogger(namespace) {
|
|
159
|
+
return new DebugLogger(namespace);
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
// src/core/pattern-matching/utils/type-validation.ts
|
|
163
|
+
function isTypeCompatible(actualType, expectedTypes) {
|
|
164
|
+
if (!expectedTypes || expectedTypes.length === 0) {
|
|
165
|
+
return true;
|
|
166
|
+
}
|
|
167
|
+
if (expectedTypes.includes(actualType)) {
|
|
168
|
+
return true;
|
|
169
|
+
}
|
|
170
|
+
if (expectedTypes.includes("expression")) {
|
|
171
|
+
return true;
|
|
172
|
+
}
|
|
173
|
+
if (actualType === "property-path") {
|
|
174
|
+
return expectedTypes.some((t) => ["selector", "reference", "expression"].includes(t));
|
|
175
|
+
}
|
|
176
|
+
return false;
|
|
177
|
+
}
|
|
178
|
+
function validateValueType(value, expectedTypes) {
|
|
179
|
+
if (!expectedTypes || expectedTypes.length === 0) {
|
|
180
|
+
return true;
|
|
181
|
+
}
|
|
182
|
+
return isTypeCompatible(value.type, expectedTypes);
|
|
183
|
+
}
|
|
184
|
+
function isCSSSelector(value) {
|
|
185
|
+
return value.startsWith(".") || value.startsWith("#") || value.startsWith("<");
|
|
186
|
+
}
|
|
187
|
+
function isClassName(value) {
|
|
188
|
+
return value.startsWith(".");
|
|
189
|
+
}
|
|
190
|
+
function isIdSelector(value) {
|
|
191
|
+
return value.startsWith("#");
|
|
192
|
+
}
|
|
193
|
+
function isCSSPropertyRef(value) {
|
|
194
|
+
return value.startsWith("*");
|
|
195
|
+
}
|
|
196
|
+
function isNumericValue(value) {
|
|
197
|
+
const durationMatch = value.match(/^(\d+(?:\.\d+)?)(ms|s|m|h)?$/);
|
|
198
|
+
if (durationMatch) {
|
|
199
|
+
return true;
|
|
200
|
+
}
|
|
201
|
+
const num = parseFloat(value);
|
|
202
|
+
return !isNaN(num) && isFinite(num);
|
|
203
|
+
}
|
|
204
|
+
function isPropertyName(value) {
|
|
205
|
+
return /^[a-zA-Z_][a-zA-Z0-9_]*$/.test(value);
|
|
206
|
+
}
|
|
207
|
+
function isVariableRef(value) {
|
|
208
|
+
return value.startsWith(":");
|
|
209
|
+
}
|
|
210
|
+
function isBuiltInReference(value) {
|
|
211
|
+
const builtIns = /* @__PURE__ */ new Set(["me", "you", "it", "result", "event", "target", "body"]);
|
|
212
|
+
return builtIns.has(value.toLowerCase());
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
// src/core/pattern-matching/utils/possessive-keywords.ts
|
|
216
|
+
function getPossessiveReference(profile, keyword) {
|
|
217
|
+
return profile?.possessive?.keywords?.[keyword];
|
|
218
|
+
}
|
|
219
|
+
function isPossessiveKeyword(profile, keyword) {
|
|
220
|
+
return profile?.possessive?.keywords?.[keyword] !== void 0;
|
|
221
|
+
}
|
|
222
|
+
function getAllPossessiveKeywords(profile) {
|
|
223
|
+
return profile?.possessive?.keywords ?? {};
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
// src/core/pattern-matching/pattern-matcher.ts
|
|
227
|
+
function isValidReference(_value) {
|
|
228
|
+
return false;
|
|
229
|
+
}
|
|
230
|
+
var _PatternMatcher = class _PatternMatcher {
|
|
231
|
+
constructor() {
|
|
232
|
+
/** Debug logger */
|
|
233
|
+
this.logger = createLogger("pattern-matcher");
|
|
234
|
+
/**
|
|
235
|
+
* Track stem matches for confidence calculation.
|
|
236
|
+
* This is set during matching and read during confidence calculation.
|
|
237
|
+
*/
|
|
238
|
+
this.stemMatchCount = 0;
|
|
239
|
+
this.totalKeywordMatches = 0;
|
|
240
|
+
}
|
|
241
|
+
/**
|
|
242
|
+
* Safely convert a value to lowercase string.
|
|
243
|
+
* Provides protection against non-string values at runtime.
|
|
244
|
+
*/
|
|
245
|
+
safeToLowerCase(value) {
|
|
246
|
+
if (typeof value === "string") {
|
|
247
|
+
return value.toLowerCase();
|
|
248
|
+
}
|
|
249
|
+
if (value === null || value === void 0) {
|
|
250
|
+
return "";
|
|
251
|
+
}
|
|
252
|
+
return String(value).toLowerCase();
|
|
253
|
+
}
|
|
254
|
+
/**
|
|
255
|
+
* Try to match a single pattern against the token stream.
|
|
256
|
+
* Returns the match result or null if no match.
|
|
257
|
+
*
|
|
258
|
+
* @param tokens - Token stream to match against
|
|
259
|
+
* @param pattern - Pattern to match
|
|
260
|
+
* @param profile - Optional language profile for possessive handling
|
|
261
|
+
*/
|
|
262
|
+
matchPattern(tokens, pattern, profile) {
|
|
263
|
+
const mark = tokens.mark();
|
|
264
|
+
const captured = /* @__PURE__ */ new Map();
|
|
265
|
+
this.logger.debug("========================================");
|
|
266
|
+
this.logger.debug("matchPattern ENTRY");
|
|
267
|
+
this.logger.debug("Pattern ID:", pattern.id);
|
|
268
|
+
this.logger.debug("Pattern command:", pattern.command);
|
|
269
|
+
this.logger.debug("Pattern language:", pattern.language);
|
|
270
|
+
this.logger.debug("Pattern template:", JSON.stringify(pattern.template, null, 2));
|
|
271
|
+
if (this.logger.isEnabled()) {
|
|
272
|
+
const firstTokens = [];
|
|
273
|
+
for (let i = 0; i < 10; i++) {
|
|
274
|
+
const t = tokens.peek(i);
|
|
275
|
+
if (t)
|
|
276
|
+
firstTokens.push({
|
|
277
|
+
type: t.type,
|
|
278
|
+
value: t.value,
|
|
279
|
+
kind: t.kind
|
|
280
|
+
});
|
|
281
|
+
else break;
|
|
282
|
+
}
|
|
283
|
+
this.logger.debug("Input tokens (first 10):", firstTokens);
|
|
284
|
+
this.logger.debug("Profile code:", profile?.code);
|
|
285
|
+
}
|
|
286
|
+
this.currentProfile = profile;
|
|
287
|
+
this.stemMatchCount = 0;
|
|
288
|
+
this.totalKeywordMatches = 0;
|
|
289
|
+
this.logger.debug("--- Calling matchTokenSequence ---");
|
|
290
|
+
this.logger.debug("Pattern tokens to match:", JSON.stringify(pattern.template.tokens, null, 2));
|
|
291
|
+
const success = this.matchTokenSequence(tokens, pattern.template.tokens, captured);
|
|
292
|
+
this.logger.debug("matchTokenSequence returned:", success);
|
|
293
|
+
this.logger.debug(
|
|
294
|
+
"Captured roles:",
|
|
295
|
+
Array.from(captured.entries()).map(([k, v]) => [k, JSON.stringify(v)])
|
|
296
|
+
);
|
|
297
|
+
if (!success) {
|
|
298
|
+
this.logger.debug(">>> MATCH FAILED - resetting token position");
|
|
299
|
+
tokens.reset(mark);
|
|
300
|
+
return null;
|
|
301
|
+
}
|
|
302
|
+
const confidence = this.calculateConfidence(pattern, captured);
|
|
303
|
+
this.applyExtractionRules(pattern, captured);
|
|
304
|
+
return {
|
|
305
|
+
pattern,
|
|
306
|
+
captured,
|
|
307
|
+
consumedTokens: tokens.position() - mark.position,
|
|
308
|
+
confidence
|
|
309
|
+
};
|
|
310
|
+
}
|
|
311
|
+
/**
|
|
312
|
+
* Try to match multiple patterns, return the best match.
|
|
313
|
+
*
|
|
314
|
+
* @param tokens - Token stream to match against
|
|
315
|
+
* @param patterns - Candidate patterns to try
|
|
316
|
+
* @param profile - Optional language profile for possessive handling
|
|
317
|
+
*/
|
|
318
|
+
matchBest(tokens, patterns, profile) {
|
|
319
|
+
const matches = [];
|
|
320
|
+
for (const pattern of patterns) {
|
|
321
|
+
const mark = tokens.mark();
|
|
322
|
+
const result = this.matchPattern(tokens, pattern, profile);
|
|
323
|
+
if (result) {
|
|
324
|
+
matches.push(result);
|
|
325
|
+
}
|
|
326
|
+
tokens.reset(mark);
|
|
327
|
+
}
|
|
328
|
+
if (matches.length === 0) {
|
|
329
|
+
return null;
|
|
330
|
+
}
|
|
331
|
+
matches.sort((a, b) => {
|
|
332
|
+
const priorityDiff = b.pattern.priority - a.pattern.priority;
|
|
333
|
+
if (priorityDiff !== 0) return priorityDiff;
|
|
334
|
+
const confidenceDiff = b.confidence - a.confidence;
|
|
335
|
+
if (Math.abs(confidenceDiff) > 1e-3) return confidenceDiff;
|
|
336
|
+
return b.consumedTokens - a.consumedTokens;
|
|
337
|
+
});
|
|
338
|
+
const best = matches[0];
|
|
339
|
+
this.matchPattern(tokens, best.pattern);
|
|
340
|
+
return best;
|
|
341
|
+
}
|
|
342
|
+
/**
|
|
343
|
+
* Match a sequence of pattern tokens against the token stream.
|
|
344
|
+
*
|
|
345
|
+
* Supports bounded single-step backtracking: if an optional role consumes
|
|
346
|
+
* a token and the immediately following pattern token fails, the matcher
|
|
347
|
+
* resets to before the optional role and retries the failed token.
|
|
348
|
+
*/
|
|
349
|
+
matchTokenSequence(tokens, patternTokens, captured) {
|
|
350
|
+
const firstPatternToken = patternTokens[0];
|
|
351
|
+
const patternExpectsConjunction = firstPatternToken?.type === "literal" && (firstPatternToken.value === "and" || firstPatternToken.value === "then" || firstPatternToken.alternatives?.includes("and") || firstPatternToken.alternatives?.includes("then"));
|
|
352
|
+
if (this.currentProfile?.code === "ar" && !patternExpectsConjunction) {
|
|
353
|
+
while (tokens.peek()?.kind === "conjunction") {
|
|
354
|
+
tokens.advance();
|
|
355
|
+
}
|
|
356
|
+
}
|
|
357
|
+
let prevOptionalMark = null;
|
|
358
|
+
let prevOptionalRole = null;
|
|
359
|
+
for (let i = 0; i < patternTokens.length; i++) {
|
|
360
|
+
const patternToken = patternTokens[i];
|
|
361
|
+
this.logger.debug(" >> Matching pattern token:", JSON.stringify(patternToken, null, 2));
|
|
362
|
+
const currTok = tokens.peek();
|
|
363
|
+
this.logger.debug(
|
|
364
|
+
" >> Current input token:",
|
|
365
|
+
currTok ? JSON.stringify({
|
|
366
|
+
type: currTok.type,
|
|
367
|
+
value: currTok.value,
|
|
368
|
+
kind: currTok.kind
|
|
369
|
+
}) : "EOF"
|
|
370
|
+
);
|
|
371
|
+
if (patternToken.type === "role" && patternToken.greedy) {
|
|
372
|
+
const stopMarkers = this.collectStopMarkers(patternTokens, i + 1);
|
|
373
|
+
const values = [];
|
|
374
|
+
while (!tokens.isAtEnd()) {
|
|
375
|
+
const nextToken = tokens.peek();
|
|
376
|
+
if (!nextToken) break;
|
|
377
|
+
if (this.isStopMarker(nextToken, stopMarkers)) break;
|
|
378
|
+
values.push(nextToken.value);
|
|
379
|
+
tokens.advance();
|
|
380
|
+
}
|
|
381
|
+
if (values.length > 0) {
|
|
382
|
+
captured.set(patternToken.role, { type: "expression", raw: values.join(" ") });
|
|
383
|
+
prevOptionalMark = null;
|
|
384
|
+
prevOptionalRole = null;
|
|
385
|
+
continue;
|
|
386
|
+
} else if (patternToken.optional) {
|
|
387
|
+
continue;
|
|
388
|
+
} else {
|
|
389
|
+
return false;
|
|
390
|
+
}
|
|
391
|
+
}
|
|
392
|
+
const isOptionalRole = patternToken.type === "role" && patternToken.optional === true;
|
|
393
|
+
const markBefore = isOptionalRole ? tokens.mark() : null;
|
|
394
|
+
const matched = this.matchPatternToken(tokens, patternToken, captured);
|
|
395
|
+
this.logger.debug(" >> Match result:", matched);
|
|
396
|
+
if (matched) {
|
|
397
|
+
if (isOptionalRole) {
|
|
398
|
+
prevOptionalMark = markBefore;
|
|
399
|
+
prevOptionalRole = patternToken.role;
|
|
400
|
+
} else {
|
|
401
|
+
prevOptionalMark = null;
|
|
402
|
+
prevOptionalRole = null;
|
|
403
|
+
}
|
|
404
|
+
continue;
|
|
405
|
+
}
|
|
406
|
+
this.logger.debug(" >> Token match FAILED");
|
|
407
|
+
if (this.isOptional(patternToken)) {
|
|
408
|
+
continue;
|
|
409
|
+
}
|
|
410
|
+
if (prevOptionalMark && prevOptionalRole) {
|
|
411
|
+
this.logger.debug(" >> BACKTRACKING: undoing optional role", prevOptionalRole);
|
|
412
|
+
tokens.reset(prevOptionalMark);
|
|
413
|
+
captured.delete(prevOptionalRole);
|
|
414
|
+
prevOptionalMark = null;
|
|
415
|
+
prevOptionalRole = null;
|
|
416
|
+
const retryMatched = this.matchPatternToken(tokens, patternToken, captured);
|
|
417
|
+
this.logger.debug(" >> Backtrack retry result:", retryMatched);
|
|
418
|
+
if (retryMatched) {
|
|
419
|
+
continue;
|
|
420
|
+
}
|
|
421
|
+
}
|
|
422
|
+
return false;
|
|
423
|
+
}
|
|
424
|
+
return true;
|
|
425
|
+
}
|
|
426
|
+
/**
|
|
427
|
+
* Match a single pattern token against the current position in the stream.
|
|
428
|
+
*/
|
|
429
|
+
matchPatternToken(tokens, patternToken, captured) {
|
|
430
|
+
switch (patternToken.type) {
|
|
431
|
+
case "literal":
|
|
432
|
+
return this.matchLiteralToken(tokens, patternToken);
|
|
433
|
+
case "role":
|
|
434
|
+
return this.matchRoleToken(tokens, patternToken, captured);
|
|
435
|
+
case "group":
|
|
436
|
+
return this.matchGroupToken(tokens, patternToken, captured);
|
|
437
|
+
default:
|
|
438
|
+
return false;
|
|
439
|
+
}
|
|
440
|
+
}
|
|
441
|
+
/**
|
|
442
|
+
* Match a literal pattern token (keyword or particle).
|
|
443
|
+
*/
|
|
444
|
+
matchLiteralToken(tokens, patternToken) {
|
|
445
|
+
const token = tokens.peek();
|
|
446
|
+
this.logger.debug(" >>> matchLiteralToken: expecting", patternToken.value);
|
|
447
|
+
this.logger.debug(
|
|
448
|
+
" >>> matchLiteralToken: got token",
|
|
449
|
+
token ? JSON.stringify({
|
|
450
|
+
type: token.type,
|
|
451
|
+
value: token.value,
|
|
452
|
+
kind: token.kind
|
|
453
|
+
}) : "null"
|
|
454
|
+
);
|
|
455
|
+
if (!token) {
|
|
456
|
+
this.logger.debug(" >>> matchLiteralToken: FAIL - no token");
|
|
457
|
+
return false;
|
|
458
|
+
}
|
|
459
|
+
const matchType = this.getMatchType(token, patternToken.value);
|
|
460
|
+
this.logger.debug(
|
|
461
|
+
" >>> matchType for",
|
|
462
|
+
token.value,
|
|
463
|
+
"vs",
|
|
464
|
+
patternToken.value,
|
|
465
|
+
":",
|
|
466
|
+
matchType
|
|
467
|
+
);
|
|
468
|
+
if (matchType !== "none") {
|
|
469
|
+
this.totalKeywordMatches++;
|
|
470
|
+
if (matchType === "stem") {
|
|
471
|
+
this.stemMatchCount++;
|
|
472
|
+
}
|
|
473
|
+
tokens.advance();
|
|
474
|
+
return true;
|
|
475
|
+
}
|
|
476
|
+
if (patternToken.alternatives) {
|
|
477
|
+
for (const alt of patternToken.alternatives) {
|
|
478
|
+
const altMatchType = this.getMatchType(token, alt);
|
|
479
|
+
if (altMatchType !== "none") {
|
|
480
|
+
this.totalKeywordMatches++;
|
|
481
|
+
if (altMatchType === "stem") {
|
|
482
|
+
this.stemMatchCount++;
|
|
483
|
+
}
|
|
484
|
+
tokens.advance();
|
|
485
|
+
return true;
|
|
486
|
+
}
|
|
487
|
+
}
|
|
488
|
+
}
|
|
489
|
+
return false;
|
|
490
|
+
}
|
|
491
|
+
/**
|
|
492
|
+
* Match a role pattern token (captures a semantic value).
|
|
493
|
+
* Handles multi-token expressions like:
|
|
494
|
+
* - 'my value' (possessive keyword + property)
|
|
495
|
+
* - '#dialog.showModal()' (method call)
|
|
496
|
+
* - "#element's *opacity" (possessive selector + property)
|
|
497
|
+
*/
|
|
498
|
+
matchRoleToken(tokens, patternToken, captured) {
|
|
499
|
+
this.logger.debug(" >>> matchRoleToken ENTRY: capturing role", patternToken.role);
|
|
500
|
+
this.logger.debug(" >>> matchRoleToken: expected types", patternToken.expectedTypes);
|
|
501
|
+
this.logger.debug(" >>> matchRoleToken: optional?", patternToken.optional);
|
|
502
|
+
this.skipNoiseWords(tokens);
|
|
503
|
+
const token = tokens.peek();
|
|
504
|
+
this.logger.debug(
|
|
505
|
+
" >>> After skipNoiseWords, current token:",
|
|
506
|
+
token ? JSON.stringify({ value: token.value, kind: token.kind }) : "null"
|
|
507
|
+
);
|
|
508
|
+
if (!token) {
|
|
509
|
+
return patternToken.optional || false;
|
|
510
|
+
}
|
|
511
|
+
const possessiveValue = this.tryMatchPossessiveExpression(tokens);
|
|
512
|
+
if (possessiveValue) {
|
|
513
|
+
if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
|
|
514
|
+
if (!patternToken.expectedTypes.includes(possessiveValue.type) && !patternToken.expectedTypes.includes("expression")) {
|
|
515
|
+
return patternToken.optional || false;
|
|
516
|
+
}
|
|
517
|
+
}
|
|
518
|
+
captured.set(patternToken.role, possessiveValue);
|
|
519
|
+
return true;
|
|
520
|
+
}
|
|
521
|
+
const methodCallValue = this.tryMatchMethodCallExpression(tokens);
|
|
522
|
+
if (methodCallValue) {
|
|
523
|
+
if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
|
|
524
|
+
if (!patternToken.expectedTypes.includes(methodCallValue.type) && !patternToken.expectedTypes.includes("expression")) {
|
|
525
|
+
return patternToken.optional || false;
|
|
526
|
+
}
|
|
527
|
+
}
|
|
528
|
+
captured.set(patternToken.role, methodCallValue);
|
|
529
|
+
return true;
|
|
530
|
+
}
|
|
531
|
+
const possessiveSelectorValue = this.tryMatchPossessiveSelectorExpression(tokens);
|
|
532
|
+
if (possessiveSelectorValue) {
|
|
533
|
+
if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
|
|
534
|
+
if (!isTypeCompatible(possessiveSelectorValue.type, patternToken.expectedTypes)) {
|
|
535
|
+
return patternToken.optional || false;
|
|
536
|
+
}
|
|
537
|
+
}
|
|
538
|
+
captured.set(patternToken.role, possessiveSelectorValue);
|
|
539
|
+
return true;
|
|
540
|
+
}
|
|
541
|
+
const propertyAccessValue = this.tryMatchPropertyAccessExpression(tokens);
|
|
542
|
+
if (propertyAccessValue) {
|
|
543
|
+
if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
|
|
544
|
+
if (!patternToken.expectedTypes.includes(propertyAccessValue.type) && !patternToken.expectedTypes.includes("expression")) {
|
|
545
|
+
return patternToken.optional || false;
|
|
546
|
+
}
|
|
547
|
+
}
|
|
548
|
+
captured.set(patternToken.role, propertyAccessValue);
|
|
549
|
+
return true;
|
|
550
|
+
}
|
|
551
|
+
const selectorPropertyValue = this.tryMatchSelectorPropertyExpression(tokens);
|
|
552
|
+
if (selectorPropertyValue) {
|
|
553
|
+
if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
|
|
554
|
+
if (!isTypeCompatible(selectorPropertyValue.type, patternToken.expectedTypes)) {
|
|
555
|
+
return patternToken.optional || false;
|
|
556
|
+
}
|
|
557
|
+
}
|
|
558
|
+
captured.set(patternToken.role, selectorPropertyValue);
|
|
559
|
+
return true;
|
|
560
|
+
}
|
|
561
|
+
this.logger.debug(
|
|
562
|
+
" >>> Trying tokenToSemanticValue for token:",
|
|
563
|
+
token ? JSON.stringify({ value: token.value, kind: token.kind }) : "null"
|
|
564
|
+
);
|
|
565
|
+
const value = this.tokenToSemanticValue(token);
|
|
566
|
+
this.logger.debug(
|
|
567
|
+
" >>> tokenToSemanticValue returned:",
|
|
568
|
+
value ? JSON.stringify(value) : "null"
|
|
569
|
+
);
|
|
570
|
+
if (!value) {
|
|
571
|
+
return patternToken.optional || false;
|
|
572
|
+
}
|
|
573
|
+
this.logger.debug(
|
|
574
|
+
" >>> Validating type:",
|
|
575
|
+
value.type,
|
|
576
|
+
"against expected:",
|
|
577
|
+
patternToken.expectedTypes
|
|
578
|
+
);
|
|
579
|
+
if (patternToken.expectedTypes && patternToken.expectedTypes.length > 0) {
|
|
580
|
+
if (!isTypeCompatible(value.type, patternToken.expectedTypes)) {
|
|
581
|
+
this.logger.debug(" >>> TYPE MISMATCH - returning", patternToken.optional || false);
|
|
582
|
+
return patternToken.optional || false;
|
|
583
|
+
}
|
|
584
|
+
}
|
|
585
|
+
this.logger.debug(" >>> Type validation PASSED");
|
|
586
|
+
captured.set(patternToken.role, value);
|
|
587
|
+
tokens.advance();
|
|
588
|
+
return true;
|
|
589
|
+
}
|
|
590
|
+
/**
|
|
591
|
+
* Try to match a possessive expression like 'my value' or 'its innerHTML'.
|
|
592
|
+
* Returns the PropertyPathValue if matched, or null if not.
|
|
593
|
+
*/
|
|
594
|
+
tryMatchPossessiveExpression(tokens) {
|
|
595
|
+
const token = tokens.peek();
|
|
596
|
+
if (!token) return null;
|
|
597
|
+
if (!this.currentProfile) return null;
|
|
598
|
+
const tokenValue = token.normalized || token.value;
|
|
599
|
+
const tokenLower = this.safeToLowerCase(tokenValue);
|
|
600
|
+
const baseRef = getPossessiveReference(this.currentProfile, tokenLower);
|
|
601
|
+
if (!baseRef) return null;
|
|
602
|
+
const mark = tokens.mark();
|
|
603
|
+
tokens.advance();
|
|
604
|
+
const propertyToken = tokens.peek();
|
|
605
|
+
if (!propertyToken) {
|
|
606
|
+
tokens.reset(mark);
|
|
607
|
+
return null;
|
|
608
|
+
}
|
|
609
|
+
if (propertyToken.kind === "identifier" || propertyToken.kind === "keyword" && !this.isStructuralKeyword(propertyToken.value) || propertyToken.kind === "selector" && propertyToken.value.startsWith("*") || propertyToken.kind === "selector" && propertyToken.value.startsWith("@") || propertyToken.kind === "selector" && propertyToken.value.startsWith(".") && /^\.[a-zA-Z_]\w*/.test(propertyToken.value)) {
|
|
610
|
+
tokens.advance();
|
|
611
|
+
let propertyName = propertyToken.value;
|
|
612
|
+
if (propertyToken.kind === "selector" && propertyName.startsWith(".") && /^\.[a-zA-Z_]\w*/.test(propertyName)) {
|
|
613
|
+
propertyName = propertyName.substring(1);
|
|
614
|
+
}
|
|
615
|
+
let chainedProps = propertyName;
|
|
616
|
+
while (tokens.peek()?.kind === "selector" && tokens.peek().value.startsWith(".") && /^\.[a-zA-Z_]\w*/.test(tokens.peek().value)) {
|
|
617
|
+
chainedProps += tokens.peek().value;
|
|
618
|
+
tokens.advance();
|
|
619
|
+
}
|
|
620
|
+
const nextPeek = tokens.peek();
|
|
621
|
+
if (nextPeek?.kind === "literal" && nextPeek.value.startsWith("(")) {
|
|
622
|
+
chainedProps += nextPeek.value;
|
|
623
|
+
tokens.advance();
|
|
624
|
+
}
|
|
625
|
+
return createPropertyPath(createReference(baseRef), chainedProps);
|
|
626
|
+
}
|
|
627
|
+
tokens.reset(mark);
|
|
628
|
+
return null;
|
|
629
|
+
}
|
|
630
|
+
/**
|
|
631
|
+
* Check if a keyword is a structural keyword (preposition, control flow, etc.)
|
|
632
|
+
* that shouldn't be consumed as a property name.
|
|
633
|
+
*/
|
|
634
|
+
isStructuralKeyword(value) {
|
|
635
|
+
const structural = /* @__PURE__ */ new Set([
|
|
636
|
+
// Prepositions
|
|
637
|
+
"into",
|
|
638
|
+
"in",
|
|
639
|
+
"to",
|
|
640
|
+
"from",
|
|
641
|
+
"at",
|
|
642
|
+
"by",
|
|
643
|
+
"with",
|
|
644
|
+
"without",
|
|
645
|
+
"before",
|
|
646
|
+
"after",
|
|
647
|
+
"of",
|
|
648
|
+
"as",
|
|
649
|
+
"on",
|
|
650
|
+
// Control flow
|
|
651
|
+
"then",
|
|
652
|
+
"end",
|
|
653
|
+
"else",
|
|
654
|
+
"if",
|
|
655
|
+
"repeat",
|
|
656
|
+
"while",
|
|
657
|
+
"for",
|
|
658
|
+
// Commands (shouldn't be property names)
|
|
659
|
+
"toggle",
|
|
660
|
+
"add",
|
|
661
|
+
"remove",
|
|
662
|
+
"put",
|
|
663
|
+
"set",
|
|
664
|
+
"show",
|
|
665
|
+
"hide",
|
|
666
|
+
"increment",
|
|
667
|
+
"decrement",
|
|
668
|
+
"send",
|
|
669
|
+
"trigger",
|
|
670
|
+
"call"
|
|
671
|
+
]);
|
|
672
|
+
return structural.has(value.toLowerCase());
|
|
673
|
+
}
|
|
674
|
+
/**
|
|
675
|
+
* Try to match a method call expression like '#dialog.showModal()'.
|
|
676
|
+
* Pattern: selector + '.' + identifier + '(' + [args] + ')'
|
|
677
|
+
* Returns an expression value if matched, or null if not.
|
|
678
|
+
*/
|
|
679
|
+
tryMatchMethodCallExpression(tokens) {
|
|
680
|
+
const token = tokens.peek();
|
|
681
|
+
if (!token || token.kind !== "selector") return null;
|
|
682
|
+
const mark = tokens.mark();
|
|
683
|
+
tokens.advance();
|
|
684
|
+
const dotToken = tokens.peek();
|
|
685
|
+
if (!dotToken || dotToken.kind !== "operator" || dotToken.value !== ".") {
|
|
686
|
+
tokens.reset(mark);
|
|
687
|
+
return null;
|
|
688
|
+
}
|
|
689
|
+
tokens.advance();
|
|
690
|
+
const methodToken = tokens.peek();
|
|
691
|
+
if (!methodToken || methodToken.kind !== "identifier") {
|
|
692
|
+
tokens.reset(mark);
|
|
693
|
+
return null;
|
|
694
|
+
}
|
|
695
|
+
tokens.advance();
|
|
696
|
+
const openParen = tokens.peek();
|
|
697
|
+
if (!openParen || openParen.kind !== "punctuation" || openParen.value !== "(") {
|
|
698
|
+
tokens.reset(mark);
|
|
699
|
+
return null;
|
|
700
|
+
}
|
|
701
|
+
tokens.advance();
|
|
702
|
+
const args = [];
|
|
703
|
+
while (!tokens.isAtEnd() && args.length < _PatternMatcher.MAX_METHOD_ARGS) {
|
|
704
|
+
const argToken = tokens.peek();
|
|
705
|
+
if (!argToken) break;
|
|
706
|
+
if (argToken.kind === "punctuation" && argToken.value === ")") {
|
|
707
|
+
tokens.advance();
|
|
708
|
+
break;
|
|
709
|
+
}
|
|
710
|
+
if (argToken.kind === "punctuation" && argToken.value === ",") {
|
|
711
|
+
tokens.advance();
|
|
712
|
+
continue;
|
|
713
|
+
}
|
|
714
|
+
args.push(argToken.value);
|
|
715
|
+
tokens.advance();
|
|
716
|
+
}
|
|
717
|
+
const methodCall = `${token.value}.${methodToken.value}(${args.join(", ")})`;
|
|
718
|
+
return {
|
|
719
|
+
type: "expression",
|
|
720
|
+
raw: methodCall
|
|
721
|
+
};
|
|
722
|
+
}
|
|
723
|
+
/**
|
|
724
|
+
* Try to match a property access expression like 'userData.name' or 'it.data'.
|
|
725
|
+
* Pattern: (identifier | keyword) + '.' + identifier [+ '.' + identifier ...]
|
|
726
|
+
* Returns an expression value if matched, or null if not.
|
|
727
|
+
*/
|
|
728
|
+
tryMatchPropertyAccessExpression(tokens) {
|
|
729
|
+
const token = tokens.peek();
|
|
730
|
+
if (!token) return null;
|
|
731
|
+
if (token.kind !== "identifier" && token.kind !== "keyword") return null;
|
|
732
|
+
const mark = tokens.mark();
|
|
733
|
+
tokens.advance();
|
|
734
|
+
const dotToken = tokens.peek();
|
|
735
|
+
if (!dotToken || dotToken.kind !== "operator" || dotToken.value !== ".") {
|
|
736
|
+
tokens.reset(mark);
|
|
737
|
+
return null;
|
|
738
|
+
}
|
|
739
|
+
tokens.advance();
|
|
740
|
+
const propertyToken = tokens.peek();
|
|
741
|
+
if (!propertyToken || propertyToken.kind !== "identifier") {
|
|
742
|
+
tokens.reset(mark);
|
|
743
|
+
return null;
|
|
744
|
+
}
|
|
745
|
+
tokens.advance();
|
|
746
|
+
let chain = `${token.value}.${propertyToken.value}`;
|
|
747
|
+
let depth = 1;
|
|
748
|
+
while (!tokens.isAtEnd() && depth < _PatternMatcher.MAX_PROPERTY_DEPTH) {
|
|
749
|
+
const nextDot = tokens.peek();
|
|
750
|
+
if (!nextDot || nextDot.kind !== "operator" || nextDot.value !== ".") {
|
|
751
|
+
break;
|
|
752
|
+
}
|
|
753
|
+
tokens.advance();
|
|
754
|
+
const nextProp = tokens.peek();
|
|
755
|
+
if (!nextProp || nextProp.kind !== "identifier") {
|
|
756
|
+
break;
|
|
757
|
+
}
|
|
758
|
+
tokens.advance();
|
|
759
|
+
chain += `.${nextProp.value}`;
|
|
760
|
+
depth++;
|
|
761
|
+
}
|
|
762
|
+
const openParen = tokens.peek();
|
|
763
|
+
if (openParen && openParen.kind === "punctuation" && openParen.value === "(") {
|
|
764
|
+
tokens.advance();
|
|
765
|
+
const args = [];
|
|
766
|
+
let argDepth = 0;
|
|
767
|
+
while (!tokens.isAtEnd() && args.length < _PatternMatcher.MAX_METHOD_ARGS) {
|
|
768
|
+
const argToken = tokens.peek();
|
|
769
|
+
if (!argToken) break;
|
|
770
|
+
if (argToken.kind === "punctuation" && argToken.value === ")") {
|
|
771
|
+
if (argDepth === 0) {
|
|
772
|
+
tokens.advance();
|
|
773
|
+
break;
|
|
774
|
+
}
|
|
775
|
+
argDepth--;
|
|
776
|
+
}
|
|
777
|
+
if (argToken.kind === "punctuation" && argToken.value === "(") {
|
|
778
|
+
argDepth++;
|
|
779
|
+
}
|
|
780
|
+
if (argToken.kind === "punctuation" && argToken.value === ",") {
|
|
781
|
+
tokens.advance();
|
|
782
|
+
continue;
|
|
783
|
+
}
|
|
784
|
+
args.push(argToken.value);
|
|
785
|
+
tokens.advance();
|
|
786
|
+
}
|
|
787
|
+
const methodCall = `${chain}(${args.join(", ")})`;
|
|
788
|
+
return {
|
|
789
|
+
type: "expression",
|
|
790
|
+
raw: methodCall
|
|
791
|
+
};
|
|
792
|
+
}
|
|
793
|
+
return {
|
|
794
|
+
type: "expression",
|
|
795
|
+
raw: chain
|
|
796
|
+
};
|
|
797
|
+
}
|
|
798
|
+
/**
|
|
799
|
+
* Try to match a possessive selector expression like "#element's *opacity".
|
|
800
|
+
* Pattern: selector + "'s" + (selector | identifier)
|
|
801
|
+
* Returns a property-path value if matched, or null if not.
|
|
802
|
+
*/
|
|
803
|
+
tryMatchPossessiveSelectorExpression(tokens) {
|
|
804
|
+
const token = tokens.peek();
|
|
805
|
+
if (!token || token.kind !== "selector") return null;
|
|
806
|
+
const mark = tokens.mark();
|
|
807
|
+
tokens.advance();
|
|
808
|
+
const possessiveToken = tokens.peek();
|
|
809
|
+
if (!possessiveToken || possessiveToken.kind !== "punctuation" || possessiveToken.value !== "'s") {
|
|
810
|
+
tokens.reset(mark);
|
|
811
|
+
return null;
|
|
812
|
+
}
|
|
813
|
+
tokens.advance();
|
|
814
|
+
const propertyToken = tokens.peek();
|
|
815
|
+
if (!propertyToken) {
|
|
816
|
+
tokens.reset(mark);
|
|
817
|
+
return null;
|
|
818
|
+
}
|
|
819
|
+
if (propertyToken.kind !== "selector" && propertyToken.kind !== "identifier") {
|
|
820
|
+
tokens.reset(mark);
|
|
821
|
+
return null;
|
|
822
|
+
}
|
|
823
|
+
tokens.advance();
|
|
824
|
+
return createPropertyPath(createSelector(token.value), propertyToken.value);
|
|
825
|
+
}
|
|
826
|
+
/**
|
|
827
|
+
* Try to match a selector + property expression like "#output.innerText".
|
|
828
|
+
* This handles cases where the tokenizer produces two selector tokens:
|
|
829
|
+
* - #output (id selector)
|
|
830
|
+
* - .innerText (looks like class selector, but is actually property)
|
|
831
|
+
*
|
|
832
|
+
* Pattern: id-selector + class-selector-that-is-actually-property
|
|
833
|
+
* Returns a property-path value if matched, or null if not.
|
|
834
|
+
*/
|
|
835
|
+
tryMatchSelectorPropertyExpression(tokens) {
|
|
836
|
+
const token = tokens.peek();
|
|
837
|
+
if (!token || token.kind !== "selector") return null;
|
|
838
|
+
if (typeof token.value !== "string" || !token.value.startsWith("#")) return null;
|
|
839
|
+
const mark = tokens.mark();
|
|
840
|
+
tokens.advance();
|
|
841
|
+
const propertyToken = tokens.peek();
|
|
842
|
+
if (!propertyToken || propertyToken.kind !== "selector") {
|
|
843
|
+
tokens.reset(mark);
|
|
844
|
+
return null;
|
|
845
|
+
}
|
|
846
|
+
if (!propertyToken.value.startsWith(".")) {
|
|
847
|
+
tokens.reset(mark);
|
|
848
|
+
return null;
|
|
849
|
+
}
|
|
850
|
+
const peek2 = tokens.peek(1);
|
|
851
|
+
if (peek2 && peek2.kind === "selector") {
|
|
852
|
+
}
|
|
853
|
+
tokens.advance();
|
|
854
|
+
const propertyName = propertyToken.value.slice(1);
|
|
855
|
+
return createPropertyPath(createSelector(token.value), propertyName);
|
|
856
|
+
}
|
|
857
|
+
/**
|
|
858
|
+
* Match a group pattern token (optional sequence).
|
|
859
|
+
* When the group's leading marker isn't at the current position, scans ahead
|
|
860
|
+
* up to MAX_MARKER_SCAN tokens to find it. This allows unmarked tokens
|
|
861
|
+
* (e.g., a value like 'hello') to sit between groups without blocking later
|
|
862
|
+
* marker matches.
|
|
863
|
+
*/
|
|
864
|
+
matchGroupToken(tokens, patternToken, captured) {
|
|
865
|
+
const mark = tokens.mark();
|
|
866
|
+
const capturedBefore = new Set(captured.keys());
|
|
867
|
+
const success = this.matchTokenSequence(tokens, patternToken.tokens, captured);
|
|
868
|
+
if (success) return true;
|
|
869
|
+
tokens.reset(mark);
|
|
870
|
+
for (const role of captured.keys()) {
|
|
871
|
+
if (!capturedBefore.has(role)) captured.delete(role);
|
|
872
|
+
}
|
|
873
|
+
if (!patternToken.optional) return false;
|
|
874
|
+
const leadingMarker = this.getGroupLeadingMarker(patternToken);
|
|
875
|
+
if (leadingMarker) {
|
|
876
|
+
for (let offset = 1; offset <= _PatternMatcher.MAX_MARKER_SCAN; offset++) {
|
|
877
|
+
const ahead = tokens.peek(offset);
|
|
878
|
+
if (!ahead) break;
|
|
879
|
+
const aheadValue = (ahead.normalized || ahead.value).toLowerCase();
|
|
880
|
+
if (aheadValue === leadingMarker) {
|
|
881
|
+
this.logger.debug(
|
|
882
|
+
" >> MARKER SCAN: found",
|
|
883
|
+
leadingMarker,
|
|
884
|
+
"at offset",
|
|
885
|
+
offset,
|
|
886
|
+
"- skipping intervening tokens"
|
|
887
|
+
);
|
|
888
|
+
for (let s = 0; s < offset; s++) tokens.advance();
|
|
889
|
+
const retrySuccess = this.matchTokenSequence(tokens, patternToken.tokens, captured);
|
|
890
|
+
if (retrySuccess) return true;
|
|
891
|
+
tokens.reset(mark);
|
|
892
|
+
for (const role of captured.keys()) {
|
|
893
|
+
if (!capturedBefore.has(role)) captured.delete(role);
|
|
894
|
+
}
|
|
895
|
+
break;
|
|
896
|
+
}
|
|
897
|
+
}
|
|
898
|
+
}
|
|
899
|
+
return true;
|
|
900
|
+
}
|
|
901
|
+
/**
|
|
902
|
+
* Get the leading marker literal from an optional group.
|
|
903
|
+
* Returns the lowercase marker value, or null if the group
|
|
904
|
+
* doesn't start with a literal token (e.g., SOV groups where
|
|
905
|
+
* the role comes before the marker).
|
|
906
|
+
*/
|
|
907
|
+
getGroupLeadingMarker(group) {
|
|
908
|
+
const first = group.tokens[0];
|
|
909
|
+
if (first?.type === "literal") return first.value.toLowerCase();
|
|
910
|
+
return null;
|
|
911
|
+
}
|
|
912
|
+
/**
|
|
913
|
+
* Get the type of match for a token against a value.
|
|
914
|
+
* Used for confidence calculation.
|
|
915
|
+
*/
|
|
916
|
+
getMatchType(token, value) {
|
|
917
|
+
if (token.value === value) return "exact";
|
|
918
|
+
if (token.normalized === value) return "normalized";
|
|
919
|
+
if (token.stem === value && token.stemConfidence !== void 0 && token.stemConfidence >= 0.7) {
|
|
920
|
+
return "stem";
|
|
921
|
+
}
|
|
922
|
+
if (token.kind === "keyword" && this.safeToLowerCase(token.value) === value.toLowerCase()) {
|
|
923
|
+
return "case-insensitive";
|
|
924
|
+
}
|
|
925
|
+
return "none";
|
|
926
|
+
}
|
|
927
|
+
/**
|
|
928
|
+
* Collect literal values from upcoming pattern tokens that act as stop markers
|
|
929
|
+
* for greedy role capture. Returns the set of lowercase values.
|
|
930
|
+
*/
|
|
931
|
+
collectStopMarkers(patternTokens, startIndex) {
|
|
932
|
+
const markers = /* @__PURE__ */ new Set();
|
|
933
|
+
for (let j = startIndex; j < patternTokens.length; j++) {
|
|
934
|
+
const pt = patternTokens[j];
|
|
935
|
+
if (pt.type === "literal") {
|
|
936
|
+
markers.add(pt.value.toLowerCase());
|
|
937
|
+
if (pt.alternatives) {
|
|
938
|
+
for (const alt of pt.alternatives) {
|
|
939
|
+
markers.add(alt.toLowerCase());
|
|
940
|
+
}
|
|
941
|
+
}
|
|
942
|
+
break;
|
|
943
|
+
}
|
|
944
|
+
if (pt.type === "group") {
|
|
945
|
+
for (const gt of pt.tokens) {
|
|
946
|
+
if (gt.type === "literal") {
|
|
947
|
+
markers.add(gt.value.toLowerCase());
|
|
948
|
+
if (gt.alternatives) {
|
|
949
|
+
for (const alt of gt.alternatives) {
|
|
950
|
+
markers.add(alt.toLowerCase());
|
|
951
|
+
}
|
|
952
|
+
}
|
|
953
|
+
break;
|
|
954
|
+
}
|
|
955
|
+
}
|
|
956
|
+
break;
|
|
957
|
+
}
|
|
958
|
+
}
|
|
959
|
+
return markers;
|
|
960
|
+
}
|
|
961
|
+
/**
|
|
962
|
+
* Check if a token matches any stop marker for greedy capture.
|
|
963
|
+
*/
|
|
964
|
+
isStopMarker(token, stopMarkers) {
|
|
965
|
+
if (stopMarkers.size === 0) return false;
|
|
966
|
+
const value = (token.normalized || token.value).toLowerCase();
|
|
967
|
+
return stopMarkers.has(value);
|
|
968
|
+
}
|
|
969
|
+
/**
|
|
970
|
+
* Convert a language token to a semantic value.
|
|
971
|
+
*/
|
|
972
|
+
tokenToSemanticValue(token) {
|
|
973
|
+
switch (token.kind) {
|
|
974
|
+
case "selector":
|
|
975
|
+
return createSelector(token.value);
|
|
976
|
+
case "literal":
|
|
977
|
+
return this.parseLiteralValue(token.value);
|
|
978
|
+
case "keyword":
|
|
979
|
+
const tokenValue = token.normalized || token.value;
|
|
980
|
+
const lower = this.safeToLowerCase(tokenValue);
|
|
981
|
+
if (isValidReference(lower)) {
|
|
982
|
+
return createReference(lower);
|
|
983
|
+
}
|
|
984
|
+
return createLiteral(token.normalized || token.value);
|
|
985
|
+
case "identifier":
|
|
986
|
+
if (typeof token.value === "string" && token.value.startsWith(":")) {
|
|
987
|
+
return createReference(token.value);
|
|
988
|
+
}
|
|
989
|
+
const identLower = this.safeToLowerCase(token.value);
|
|
990
|
+
if (isValidReference(identLower)) {
|
|
991
|
+
return createReference(identLower);
|
|
992
|
+
}
|
|
993
|
+
return { type: "expression", raw: token.value };
|
|
994
|
+
case "url":
|
|
995
|
+
return createLiteral(token.value, "string");
|
|
996
|
+
default:
|
|
997
|
+
return null;
|
|
998
|
+
}
|
|
999
|
+
}
|
|
1000
|
+
/**
|
|
1001
|
+
* Parse a literal value (string, number, boolean).
|
|
1002
|
+
*/
|
|
1003
|
+
parseLiteralValue(value) {
|
|
1004
|
+
if (value.startsWith('"') || value.startsWith("'") || value.startsWith("`") || value.startsWith("\u300C")) {
|
|
1005
|
+
const inner = value.slice(1, -1);
|
|
1006
|
+
return createLiteral(inner, "string");
|
|
1007
|
+
}
|
|
1008
|
+
if (value === "true") return createLiteral(true, "boolean");
|
|
1009
|
+
if (value === "false") return createLiteral(false, "boolean");
|
|
1010
|
+
const durationMatch = value.match(/^(\d+(?:\.\d+)?)(ms|s|m|h)?$/);
|
|
1011
|
+
if (durationMatch) {
|
|
1012
|
+
const num2 = parseFloat(durationMatch[1]);
|
|
1013
|
+
const unit = durationMatch[2];
|
|
1014
|
+
if (unit) {
|
|
1015
|
+
return createLiteral(value, "duration");
|
|
1016
|
+
}
|
|
1017
|
+
return createLiteral(num2, "number");
|
|
1018
|
+
}
|
|
1019
|
+
const num = parseFloat(value);
|
|
1020
|
+
if (!isNaN(num)) {
|
|
1021
|
+
return createLiteral(num, "number");
|
|
1022
|
+
}
|
|
1023
|
+
return createLiteral(value, "string");
|
|
1024
|
+
}
|
|
1025
|
+
/**
|
|
1026
|
+
* Apply extraction rules to fill in static values and defaults for missing roles.
|
|
1027
|
+
*/
|
|
1028
|
+
applyExtractionRules(pattern, captured) {
|
|
1029
|
+
for (const [role, rule] of Object.entries(pattern.extraction)) {
|
|
1030
|
+
if (!captured.has(role)) {
|
|
1031
|
+
if (rule.value !== void 0) {
|
|
1032
|
+
captured.set(role, { type: "literal", value: rule.value });
|
|
1033
|
+
} else if (rule.default) {
|
|
1034
|
+
captured.set(role, rule.default);
|
|
1035
|
+
}
|
|
1036
|
+
}
|
|
1037
|
+
}
|
|
1038
|
+
}
|
|
1039
|
+
/**
|
|
1040
|
+
* Check if a pattern token is optional.
|
|
1041
|
+
*/
|
|
1042
|
+
isOptional(patternToken) {
|
|
1043
|
+
return patternToken.type !== "literal" && patternToken.optional === true;
|
|
1044
|
+
}
|
|
1045
|
+
/**
|
|
1046
|
+
* Calculate confidence score for a match (0-1).
|
|
1047
|
+
*
|
|
1048
|
+
* Confidence is reduced for:
|
|
1049
|
+
* - Stem matches (morphological normalization has inherent uncertainty)
|
|
1050
|
+
* - Missing optional roles (but less penalty if role has a default value)
|
|
1051
|
+
*
|
|
1052
|
+
* Confidence is increased for:
|
|
1053
|
+
* - VSO languages (Arabic) when pattern starts with a verb
|
|
1054
|
+
*/
|
|
1055
|
+
calculateConfidence(pattern, captured) {
|
|
1056
|
+
let score = 0;
|
|
1057
|
+
let maxScore = 0;
|
|
1058
|
+
const hasDefault = (role) => {
|
|
1059
|
+
return pattern.extraction?.[role]?.default !== void 0;
|
|
1060
|
+
};
|
|
1061
|
+
for (const token of pattern.template.tokens) {
|
|
1062
|
+
if (token.type === "role") {
|
|
1063
|
+
maxScore += 1;
|
|
1064
|
+
if (captured.has(token.role)) {
|
|
1065
|
+
score += 1;
|
|
1066
|
+
}
|
|
1067
|
+
} else if (token.type === "group") {
|
|
1068
|
+
for (const subToken of token.tokens) {
|
|
1069
|
+
if (subToken.type === "role") {
|
|
1070
|
+
const roleHasDefault = hasDefault(subToken.role);
|
|
1071
|
+
const weight = 0.8;
|
|
1072
|
+
maxScore += weight;
|
|
1073
|
+
if (captured.has(subToken.role)) {
|
|
1074
|
+
score += weight;
|
|
1075
|
+
} else if (roleHasDefault) {
|
|
1076
|
+
score += weight * 0.6;
|
|
1077
|
+
}
|
|
1078
|
+
}
|
|
1079
|
+
}
|
|
1080
|
+
}
|
|
1081
|
+
}
|
|
1082
|
+
let baseConfidence = maxScore > 0 ? score / maxScore : 1;
|
|
1083
|
+
if (this.stemMatchCount > 0 && this.totalKeywordMatches > 0) {
|
|
1084
|
+
const stemPenalty = this.stemMatchCount / this.totalKeywordMatches * 0.15;
|
|
1085
|
+
baseConfidence = Math.max(0.5, baseConfidence - stemPenalty);
|
|
1086
|
+
}
|
|
1087
|
+
const vsoBoost = this.calculateVSOConfidenceBoost(pattern);
|
|
1088
|
+
baseConfidence = Math.min(1, baseConfidence + vsoBoost);
|
|
1089
|
+
const prepositionAdjustment = this.arabicPrepositionDisambiguation(pattern, captured);
|
|
1090
|
+
baseConfidence = Math.max(0, Math.min(1, baseConfidence + prepositionAdjustment));
|
|
1091
|
+
return baseConfidence;
|
|
1092
|
+
}
|
|
1093
|
+
/**
|
|
1094
|
+
* Calculate confidence boost for VSO (Verb-Subject-Object) language patterns.
|
|
1095
|
+
* Arabic naturally uses VSO word order, so patterns that start with a verb
|
|
1096
|
+
* should receive a confidence boost.
|
|
1097
|
+
*
|
|
1098
|
+
* Returns +0.15 confidence boost if:
|
|
1099
|
+
* - Language is Arabic ('ar')
|
|
1100
|
+
* - Pattern's first token is a verb keyword
|
|
1101
|
+
*
|
|
1102
|
+
* @param pattern The language pattern being matched
|
|
1103
|
+
* @returns Confidence boost (0 or 0.15)
|
|
1104
|
+
*/
|
|
1105
|
+
calculateVSOConfidenceBoost(pattern) {
|
|
1106
|
+
if (pattern.language !== "ar") {
|
|
1107
|
+
return 0;
|
|
1108
|
+
}
|
|
1109
|
+
const firstToken = pattern.template.tokens[0];
|
|
1110
|
+
if (!firstToken || firstToken.type !== "literal") {
|
|
1111
|
+
return 0;
|
|
1112
|
+
}
|
|
1113
|
+
const ARABIC_VERBS = /* @__PURE__ */ new Set([
|
|
1114
|
+
"\u0628\u062F\u0644",
|
|
1115
|
+
"\u063A\u064A\u0631",
|
|
1116
|
+
"\u0623\u0636\u0641",
|
|
1117
|
+
"\u0623\u0632\u0644",
|
|
1118
|
+
"\u0636\u0639",
|
|
1119
|
+
"\u0627\u062C\u0639\u0644",
|
|
1120
|
+
"\u0639\u064A\u0646",
|
|
1121
|
+
"\u0632\u062F",
|
|
1122
|
+
"\u0627\u0646\u0642\u0635",
|
|
1123
|
+
"\u0633\u062C\u0644",
|
|
1124
|
+
"\u0623\u0638\u0647\u0631",
|
|
1125
|
+
"\u0623\u062E\u0641",
|
|
1126
|
+
"\u0634\u063A\u0644",
|
|
1127
|
+
"\u0623\u0631\u0633\u0644",
|
|
1128
|
+
"\u0631\u0643\u0632",
|
|
1129
|
+
"\u0634\u0648\u0634",
|
|
1130
|
+
"\u062A\u0648\u0642\u0641",
|
|
1131
|
+
"\u0627\u0646\u0633\u062E",
|
|
1132
|
+
"\u0627\u062D\u0630\u0641",
|
|
1133
|
+
"\u0627\u0635\u0646\u0639",
|
|
1134
|
+
"\u0627\u0646\u062A\u0638\u0631",
|
|
1135
|
+
"\u0627\u0646\u062A\u0642\u0627\u0644",
|
|
1136
|
+
"\u0623\u0648"
|
|
1137
|
+
]);
|
|
1138
|
+
if (ARABIC_VERBS.has(firstToken.value)) {
|
|
1139
|
+
return 0.15;
|
|
1140
|
+
}
|
|
1141
|
+
if (firstToken.alternatives) {
|
|
1142
|
+
for (const alt of firstToken.alternatives) {
|
|
1143
|
+
if (ARABIC_VERBS.has(alt)) {
|
|
1144
|
+
return 0.15;
|
|
1145
|
+
}
|
|
1146
|
+
}
|
|
1147
|
+
}
|
|
1148
|
+
return 0;
|
|
1149
|
+
}
|
|
1150
|
+
/**
|
|
1151
|
+
* Arabic preposition disambiguation for confidence adjustment.
|
|
1152
|
+
*
|
|
1153
|
+
* Different Arabic prepositions are more or less natural for different semantic roles:
|
|
1154
|
+
* - على (on/upon) is preferred for patient/target roles (element selectors)
|
|
1155
|
+
* - إلى (to) is preferred for destination roles
|
|
1156
|
+
* - من (from) is preferred for source roles
|
|
1157
|
+
* - في (in) is preferred for location roles
|
|
1158
|
+
*
|
|
1159
|
+
* This method analyzes the prepositions used with captured semantic roles and
|
|
1160
|
+
* adjusts confidence based on idiomaticity:
|
|
1161
|
+
* - +0.10 for highly idiomatic preposition choices
|
|
1162
|
+
* - -0.10 for less natural preposition choices
|
|
1163
|
+
*
|
|
1164
|
+
* @param pattern The language pattern being matched
|
|
1165
|
+
* @param captured The captured semantic values
|
|
1166
|
+
* @returns Confidence adjustment (-0.10 to +0.10)
|
|
1167
|
+
*/
|
|
1168
|
+
arabicPrepositionDisambiguation(pattern, captured) {
|
|
1169
|
+
if (pattern.language !== "ar") {
|
|
1170
|
+
return 0;
|
|
1171
|
+
}
|
|
1172
|
+
let adjustment = 0;
|
|
1173
|
+
const PREFERRED_PREPOSITIONS = {
|
|
1174
|
+
patient: ["\u0639\u0644\u0649"],
|
|
1175
|
+
// element selectors prefer على (on/upon)
|
|
1176
|
+
destination: ["\u0625\u0644\u0649", "\u0627\u0644\u0649"],
|
|
1177
|
+
// destination prefers إلى (to)
|
|
1178
|
+
source: ["\u0645\u0646"],
|
|
1179
|
+
// source prefers من (from)
|
|
1180
|
+
agent: ["\u0645\u0646"],
|
|
1181
|
+
// agent/by prefers من (from/by)
|
|
1182
|
+
manner: ["\u0628"],
|
|
1183
|
+
// manner prefers ب (with/by)
|
|
1184
|
+
style: ["\u0628"],
|
|
1185
|
+
// style prefers ب (with)
|
|
1186
|
+
goal: ["\u0625\u0644\u0649", "\u0627\u0644\u0649"],
|
|
1187
|
+
// target state prefers إلى (to)
|
|
1188
|
+
method: ["\u0628"]
|
|
1189
|
+
// method prefers ب (with/by)
|
|
1190
|
+
};
|
|
1191
|
+
for (const [role, value] of captured.entries()) {
|
|
1192
|
+
const preferred = PREFERRED_PREPOSITIONS[role];
|
|
1193
|
+
if (!preferred || preferred.length === 0) {
|
|
1194
|
+
continue;
|
|
1195
|
+
}
|
|
1196
|
+
const metadata = "metadata" in value ? value.metadata : void 0;
|
|
1197
|
+
if (metadata && typeof metadata.prepositionValue === "string") {
|
|
1198
|
+
const usedPreposition = metadata.prepositionValue;
|
|
1199
|
+
if (preferred.includes(usedPreposition)) {
|
|
1200
|
+
adjustment += 0.1;
|
|
1201
|
+
} else {
|
|
1202
|
+
adjustment -= 0.1;
|
|
1203
|
+
}
|
|
1204
|
+
}
|
|
1205
|
+
}
|
|
1206
|
+
return Math.max(-0.1, Math.min(0.1, adjustment));
|
|
1207
|
+
}
|
|
1208
|
+
/**
|
|
1209
|
+
* Skip noise words like "the" before selectors.
|
|
1210
|
+
* This enables more natural English syntax like "toggle the .active".
|
|
1211
|
+
*/
|
|
1212
|
+
skipNoiseWords(tokens) {
|
|
1213
|
+
const token = tokens.peek();
|
|
1214
|
+
if (!token) return;
|
|
1215
|
+
const tokenLower = this.safeToLowerCase(token.value);
|
|
1216
|
+
if (_PatternMatcher.ENGLISH_NOISE_WORDS.has(tokenLower)) {
|
|
1217
|
+
const mark = tokens.mark();
|
|
1218
|
+
tokens.advance();
|
|
1219
|
+
const nextToken = tokens.peek();
|
|
1220
|
+
if (nextToken && nextToken.kind === "selector") {
|
|
1221
|
+
return;
|
|
1222
|
+
}
|
|
1223
|
+
tokens.reset(mark);
|
|
1224
|
+
}
|
|
1225
|
+
if (tokenLower === "class") {
|
|
1226
|
+
tokens.advance();
|
|
1227
|
+
}
|
|
1228
|
+
}
|
|
1229
|
+
/**
|
|
1230
|
+
* Extract event modifiers from the token stream.
|
|
1231
|
+
* Event modifiers are .once, .debounce(N), .throttle(N), .queue(strategy)
|
|
1232
|
+
* that can appear after event names.
|
|
1233
|
+
*
|
|
1234
|
+
* Returns EventModifiers object or undefined if no modifiers found.
|
|
1235
|
+
*/
|
|
1236
|
+
extractEventModifiers(tokens) {
|
|
1237
|
+
const modifiers = {};
|
|
1238
|
+
let foundModifier = false;
|
|
1239
|
+
while (!tokens.isAtEnd()) {
|
|
1240
|
+
const token = tokens.peek();
|
|
1241
|
+
if (!token || token.kind !== "event-modifier") {
|
|
1242
|
+
break;
|
|
1243
|
+
}
|
|
1244
|
+
const metadata = token.metadata;
|
|
1245
|
+
if (!metadata) {
|
|
1246
|
+
break;
|
|
1247
|
+
}
|
|
1248
|
+
foundModifier = true;
|
|
1249
|
+
switch (metadata.modifierName) {
|
|
1250
|
+
case "once":
|
|
1251
|
+
modifiers.once = true;
|
|
1252
|
+
break;
|
|
1253
|
+
case "debounce":
|
|
1254
|
+
if (typeof metadata.value === "number") {
|
|
1255
|
+
modifiers.debounce = metadata.value;
|
|
1256
|
+
}
|
|
1257
|
+
break;
|
|
1258
|
+
case "throttle":
|
|
1259
|
+
if (typeof metadata.value === "number") {
|
|
1260
|
+
modifiers.throttle = metadata.value;
|
|
1261
|
+
}
|
|
1262
|
+
break;
|
|
1263
|
+
case "queue":
|
|
1264
|
+
if (metadata.value === "first" || metadata.value === "last" || metadata.value === "all" || metadata.value === "none") {
|
|
1265
|
+
modifiers.queue = metadata.value;
|
|
1266
|
+
}
|
|
1267
|
+
break;
|
|
1268
|
+
}
|
|
1269
|
+
tokens.advance();
|
|
1270
|
+
}
|
|
1271
|
+
return foundModifier ? modifiers : void 0;
|
|
1272
|
+
}
|
|
1273
|
+
};
|
|
1274
|
+
/** Maximum tokens to scan ahead when looking for a group's leading marker */
|
|
1275
|
+
_PatternMatcher.MAX_MARKER_SCAN = 3;
|
|
1276
|
+
// ==========================================================================
|
|
1277
|
+
// Depth Limits for Expression Parsing (security hardening)
|
|
1278
|
+
// ==========================================================================
|
|
1279
|
+
/** Maximum depth for nested property access (e.g., a.b.c.d...) */
|
|
1280
|
+
_PatternMatcher.MAX_PROPERTY_DEPTH = 10;
|
|
1281
|
+
/** Maximum number of arguments in method calls */
|
|
1282
|
+
_PatternMatcher.MAX_METHOD_ARGS = 20;
|
|
1283
|
+
// ===========================================================================
|
|
1284
|
+
// English Idiom Support - Noise Word Handling
|
|
1285
|
+
// ===========================================================================
|
|
1286
|
+
/**
|
|
1287
|
+
* Noise words that can be skipped in English for more natural syntax.
|
|
1288
|
+
* - "the" before selectors: "toggle the .active" → "toggle .active"
|
|
1289
|
+
* - "class" after class selectors: "add the .visible class" → "add .visible"
|
|
1290
|
+
*/
|
|
1291
|
+
_PatternMatcher.ENGLISH_NOISE_WORDS = /* @__PURE__ */ new Set(["the", "a", "an"]);
|
|
1292
|
+
var PatternMatcher = _PatternMatcher;
|
|
1293
|
+
var patternMatcher = new PatternMatcher();
|
|
1294
|
+
function matchPattern(tokens, pattern) {
|
|
1295
|
+
return patternMatcher.matchPattern(tokens, pattern);
|
|
1296
|
+
}
|
|
1297
|
+
function matchBest(tokens, patterns) {
|
|
1298
|
+
return patternMatcher.matchBest(tokens, patterns);
|
|
1299
|
+
}
|
|
1300
|
+
|
|
1301
|
+
// src/generation/pattern-generator.ts
|
|
1302
|
+
var defaultConfig = {
|
|
1303
|
+
basePriority: 100,
|
|
1304
|
+
includeOptionalVariants: true
|
|
1305
|
+
};
|
|
1306
|
+
function generatePattern(schema, profile, config = defaultConfig) {
|
|
1307
|
+
const id = `${schema.action}-${profile.code}-generated`;
|
|
1308
|
+
const priority = config.basePriority ?? 100;
|
|
1309
|
+
const keyword = profile.keywords[schema.action];
|
|
1310
|
+
if (!keyword) {
|
|
1311
|
+
throw new Error(`No keyword translation for '${schema.action}' in ${profile.code}`);
|
|
1312
|
+
}
|
|
1313
|
+
const tokens = buildTokens(schema, profile, keyword.primary, keyword.alternatives);
|
|
1314
|
+
const extraction = buildExtractionRules(schema, profile);
|
|
1315
|
+
const format = buildFormatString(schema, profile, keyword.primary);
|
|
1316
|
+
return {
|
|
1317
|
+
id,
|
|
1318
|
+
language: profile.code,
|
|
1319
|
+
command: schema.action,
|
|
1320
|
+
priority,
|
|
1321
|
+
template: {
|
|
1322
|
+
format,
|
|
1323
|
+
tokens
|
|
1324
|
+
},
|
|
1325
|
+
extraction
|
|
1326
|
+
};
|
|
1327
|
+
}
|
|
1328
|
+
function buildTokens(schema, profile, keyword, alternatives) {
|
|
1329
|
+
const tokens = [];
|
|
1330
|
+
const keywordToken = {
|
|
1331
|
+
type: "literal",
|
|
1332
|
+
value: keyword,
|
|
1333
|
+
...alternatives?.length && { alternatives }
|
|
1334
|
+
};
|
|
1335
|
+
const sortedRoles = sortRolesByWordOrder(schema.roles, profile.wordOrder);
|
|
1336
|
+
if (profile.wordOrder === "SVO") {
|
|
1337
|
+
tokens.push(keywordToken);
|
|
1338
|
+
for (const role of sortedRoles) {
|
|
1339
|
+
addRoleWithMarker(tokens, role, profile);
|
|
1340
|
+
}
|
|
1341
|
+
} else if (profile.wordOrder === "SOV") {
|
|
1342
|
+
for (const role of sortedRoles) {
|
|
1343
|
+
addRoleWithMarker(tokens, role, profile);
|
|
1344
|
+
}
|
|
1345
|
+
tokens.push(keywordToken);
|
|
1346
|
+
} else if (profile.wordOrder === "VSO") {
|
|
1347
|
+
tokens.push(keywordToken);
|
|
1348
|
+
for (const role of sortedRoles) {
|
|
1349
|
+
addRoleWithMarker(tokens, role, profile);
|
|
1350
|
+
}
|
|
1351
|
+
} else {
|
|
1352
|
+
tokens.push(keywordToken);
|
|
1353
|
+
for (const role of sortedRoles) {
|
|
1354
|
+
addRoleWithMarker(tokens, role, profile);
|
|
1355
|
+
}
|
|
1356
|
+
}
|
|
1357
|
+
return tokens;
|
|
1358
|
+
}
|
|
1359
|
+
function addRoleWithMarker(tokens, roleSpec, profile) {
|
|
1360
|
+
const marker = getMarkerForRole(roleSpec, profile);
|
|
1361
|
+
const isOptional = !roleSpec.required;
|
|
1362
|
+
const roleToken = {
|
|
1363
|
+
type: "role",
|
|
1364
|
+
role: roleSpec.role,
|
|
1365
|
+
optional: isOptional,
|
|
1366
|
+
expectedTypes: roleSpec.expectedTypes,
|
|
1367
|
+
...roleSpec.greedy && { greedy: true }
|
|
1368
|
+
};
|
|
1369
|
+
if (marker) {
|
|
1370
|
+
const markerInfo = profile.roleMarkers?.[roleSpec.role];
|
|
1371
|
+
const defaultPosition = profile.wordOrder === "SOV" ? "after" : "before";
|
|
1372
|
+
const position = markerInfo?.position ?? defaultPosition;
|
|
1373
|
+
if (isOptional) {
|
|
1374
|
+
if (position === "before") {
|
|
1375
|
+
tokens.push({
|
|
1376
|
+
type: "group",
|
|
1377
|
+
optional: true,
|
|
1378
|
+
tokens: [{ type: "literal", value: marker }, roleToken]
|
|
1379
|
+
});
|
|
1380
|
+
} else {
|
|
1381
|
+
tokens.push({
|
|
1382
|
+
type: "group",
|
|
1383
|
+
optional: true,
|
|
1384
|
+
tokens: [roleToken, { type: "literal", value: marker }]
|
|
1385
|
+
});
|
|
1386
|
+
}
|
|
1387
|
+
} else {
|
|
1388
|
+
if (position === "before") {
|
|
1389
|
+
tokens.push({ type: "literal", value: marker });
|
|
1390
|
+
tokens.push(roleToken);
|
|
1391
|
+
} else {
|
|
1392
|
+
tokens.push(roleToken);
|
|
1393
|
+
tokens.push({ type: "literal", value: marker });
|
|
1394
|
+
}
|
|
1395
|
+
}
|
|
1396
|
+
} else {
|
|
1397
|
+
tokens.push(roleToken);
|
|
1398
|
+
}
|
|
1399
|
+
}
|
|
1400
|
+
function getMarkerForRole(roleSpec, profile) {
|
|
1401
|
+
if (roleSpec.markerOverride?.[profile.code]) {
|
|
1402
|
+
return roleSpec.markerOverride[profile.code];
|
|
1403
|
+
}
|
|
1404
|
+
return profile.roleMarkers?.[roleSpec.role]?.primary;
|
|
1405
|
+
}
|
|
1406
|
+
function sortRolesByWordOrder(roles, wordOrder) {
|
|
1407
|
+
const sorted = [...roles];
|
|
1408
|
+
if (wordOrder === "SVO") {
|
|
1409
|
+
sorted.sort((a, b) => (b.svoPosition ?? 0) - (a.svoPosition ?? 0));
|
|
1410
|
+
} else if (wordOrder === "SOV") {
|
|
1411
|
+
sorted.sort((a, b) => (b.sovPosition ?? 0) - (a.sovPosition ?? 0));
|
|
1412
|
+
}
|
|
1413
|
+
return sorted;
|
|
1414
|
+
}
|
|
1415
|
+
function buildExtractionRules(schema, profile) {
|
|
1416
|
+
const rules = {};
|
|
1417
|
+
for (const roleSpec of schema.roles) {
|
|
1418
|
+
const marker = getMarkerForRole(roleSpec, profile);
|
|
1419
|
+
const rule = {
|
|
1420
|
+
...roleSpec.default !== void 0 && { default: roleSpec.default },
|
|
1421
|
+
...marker && { marker }
|
|
1422
|
+
};
|
|
1423
|
+
rules[roleSpec.role] = rule;
|
|
1424
|
+
}
|
|
1425
|
+
return rules;
|
|
1426
|
+
}
|
|
1427
|
+
function buildFormatString(schema, profile, keyword) {
|
|
1428
|
+
const parts = [];
|
|
1429
|
+
if (profile.wordOrder === "SVO" || profile.wordOrder === "VSO") {
|
|
1430
|
+
parts.push(keyword);
|
|
1431
|
+
}
|
|
1432
|
+
for (const role of schema.roles) {
|
|
1433
|
+
const marker = getMarkerForRole(role, profile);
|
|
1434
|
+
const roleName = `{${role.role}}`;
|
|
1435
|
+
if (marker) {
|
|
1436
|
+
const markerInfo = profile.roleMarkers?.[role.role];
|
|
1437
|
+
if (markerInfo?.position === "after") {
|
|
1438
|
+
parts.push(`${roleName} ${marker}`);
|
|
1439
|
+
} else {
|
|
1440
|
+
parts.push(`${marker} ${roleName}`);
|
|
1441
|
+
}
|
|
1442
|
+
} else {
|
|
1443
|
+
parts.push(roleName);
|
|
1444
|
+
}
|
|
1445
|
+
}
|
|
1446
|
+
if (profile.wordOrder === "SOV") {
|
|
1447
|
+
parts.push(keyword);
|
|
1448
|
+
}
|
|
1449
|
+
return parts.join(" ");
|
|
1450
|
+
}
|
|
1451
|
+
function generatePatternVariants(schema, profile, config = defaultConfig) {
|
|
1452
|
+
const patterns = [];
|
|
1453
|
+
patterns.push(generatePattern(schema, profile, config));
|
|
1454
|
+
return patterns;
|
|
1455
|
+
}
|
|
1456
|
+
|
|
1457
|
+
// src/grammar/types.ts
|
|
1458
|
+
function reorderRoles(roles, targetOrder) {
|
|
1459
|
+
const result = [];
|
|
1460
|
+
const usedRoles = /* @__PURE__ */ new Set();
|
|
1461
|
+
for (const role of targetOrder) {
|
|
1462
|
+
const element = roles.get(role);
|
|
1463
|
+
if (element) {
|
|
1464
|
+
result.push(element);
|
|
1465
|
+
usedRoles.add(role);
|
|
1466
|
+
}
|
|
1467
|
+
}
|
|
1468
|
+
for (const [role, element] of roles) {
|
|
1469
|
+
if (!usedRoles.has(role)) {
|
|
1470
|
+
result.push(element);
|
|
1471
|
+
}
|
|
1472
|
+
}
|
|
1473
|
+
return result;
|
|
1474
|
+
}
|
|
1475
|
+
function insertMarkers(elements, markers, adpositionType) {
|
|
1476
|
+
const result = [];
|
|
1477
|
+
for (const element of elements) {
|
|
1478
|
+
const marker = markers.find((m) => m.role === element.role);
|
|
1479
|
+
if (marker) {
|
|
1480
|
+
if (adpositionType === "preposition") {
|
|
1481
|
+
if (marker.form) result.push(marker.form);
|
|
1482
|
+
result.push(element.translated || element.value);
|
|
1483
|
+
} else if (adpositionType === "postposition") {
|
|
1484
|
+
result.push(element.translated || element.value);
|
|
1485
|
+
if (marker.form) result.push(marker.form);
|
|
1486
|
+
} else {
|
|
1487
|
+
result.push(element.translated || element.value);
|
|
1488
|
+
}
|
|
1489
|
+
} else {
|
|
1490
|
+
result.push(element.translated || element.value);
|
|
1491
|
+
}
|
|
1492
|
+
}
|
|
1493
|
+
return result;
|
|
1494
|
+
}
|
|
1495
|
+
function joinTokens(tokens) {
|
|
1496
|
+
if (tokens.length === 0) return "";
|
|
1497
|
+
let result = "";
|
|
1498
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
1499
|
+
const token = tokens[i];
|
|
1500
|
+
const nextToken = tokens[i + 1];
|
|
1501
|
+
const isPrefix = token.endsWith("-");
|
|
1502
|
+
const isSuffix = token.startsWith("-");
|
|
1503
|
+
let displayToken = token;
|
|
1504
|
+
if (isPrefix) displayToken = token.slice(0, -1);
|
|
1505
|
+
if (isSuffix) displayToken = token.substring(1);
|
|
1506
|
+
result += displayToken;
|
|
1507
|
+
if (nextToken) {
|
|
1508
|
+
const nextIsSuffix = nextToken.startsWith("-");
|
|
1509
|
+
if (!isPrefix && !nextIsSuffix) {
|
|
1510
|
+
result += " ";
|
|
1511
|
+
}
|
|
1512
|
+
}
|
|
1513
|
+
}
|
|
1514
|
+
return result;
|
|
1515
|
+
}
|
|
1516
|
+
|
|
1517
|
+
// src/grammar/transformer.ts
|
|
1518
|
+
var GrammarTransformer = class {
|
|
1519
|
+
constructor(config) {
|
|
1520
|
+
this.dictionary = config.dictionary;
|
|
1521
|
+
this.profileProvider = config.profileProvider;
|
|
1522
|
+
}
|
|
1523
|
+
/**
|
|
1524
|
+
* Transform a statement from one language to another.
|
|
1525
|
+
*
|
|
1526
|
+
* @param input - Statement in source language
|
|
1527
|
+
* @param fromLanguage - Source language code
|
|
1528
|
+
* @param toLanguage - Target language code
|
|
1529
|
+
* @returns Transformed statement in target language
|
|
1530
|
+
*
|
|
1531
|
+
* @throws Error if profiles not found or translation fails
|
|
1532
|
+
*/
|
|
1533
|
+
transform(input, fromLanguage, toLanguage) {
|
|
1534
|
+
const fromProfile = this.profileProvider.getProfile(fromLanguage);
|
|
1535
|
+
const toProfile = this.profileProvider.getProfile(toLanguage);
|
|
1536
|
+
if (!fromProfile) {
|
|
1537
|
+
throw new Error(`No profile found for language: ${fromLanguage}`);
|
|
1538
|
+
}
|
|
1539
|
+
if (!toProfile) {
|
|
1540
|
+
throw new Error(`No profile found for language: ${toLanguage}`);
|
|
1541
|
+
}
|
|
1542
|
+
const parsed = this.parseStatement(input, fromLanguage, fromProfile);
|
|
1543
|
+
const translated = this.translateRoles(parsed, fromLanguage, toLanguage);
|
|
1544
|
+
const reordered = reorderRoles(translated, toProfile.canonicalOrder);
|
|
1545
|
+
const withMarkers = insertMarkers(reordered, toProfile.markers, toProfile.adpositionType);
|
|
1546
|
+
return joinTokens(withMarkers);
|
|
1547
|
+
}
|
|
1548
|
+
/**
|
|
1549
|
+
* Parse statement into semantic roles.
|
|
1550
|
+
* This is a simplified generic parser - DSLs can provide custom parsers.
|
|
1551
|
+
*/
|
|
1552
|
+
parseStatement(input, language, profile) {
|
|
1553
|
+
const roles = /* @__PURE__ */ new Map();
|
|
1554
|
+
const tokens = input.split(/\s+/);
|
|
1555
|
+
let i = 0;
|
|
1556
|
+
while (i < tokens.length) {
|
|
1557
|
+
const token = tokens[i];
|
|
1558
|
+
const marker = profile.markers.find((m) => m.form === token);
|
|
1559
|
+
if (marker) {
|
|
1560
|
+
if (i + 1 < tokens.length) {
|
|
1561
|
+
roles.set(marker.role, tokens[i + 1]);
|
|
1562
|
+
i += 2;
|
|
1563
|
+
} else {
|
|
1564
|
+
i++;
|
|
1565
|
+
}
|
|
1566
|
+
} else {
|
|
1567
|
+
const canonical = this.dictionary.lookup(token, language);
|
|
1568
|
+
if (canonical) {
|
|
1569
|
+
roles.set("action", canonical);
|
|
1570
|
+
} else {
|
|
1571
|
+
if (token.startsWith("#") || token.startsWith(".")) {
|
|
1572
|
+
roles.set("patient", token);
|
|
1573
|
+
} else {
|
|
1574
|
+
roles.set("patient", token);
|
|
1575
|
+
}
|
|
1576
|
+
}
|
|
1577
|
+
i++;
|
|
1578
|
+
}
|
|
1579
|
+
}
|
|
1580
|
+
return roles;
|
|
1581
|
+
}
|
|
1582
|
+
/**
|
|
1583
|
+
* Translate roles using injected dictionary.
|
|
1584
|
+
* Returns a Map of ParsedElement objects expected by reorderRoles.
|
|
1585
|
+
*/
|
|
1586
|
+
translateRoles(roles, _fromLanguage, toLanguage) {
|
|
1587
|
+
const result = /* @__PURE__ */ new Map();
|
|
1588
|
+
for (const [role, value] of roles) {
|
|
1589
|
+
const translated = this.dictionary.translate(value, toLanguage);
|
|
1590
|
+
result.set(role, {
|
|
1591
|
+
role,
|
|
1592
|
+
value,
|
|
1593
|
+
// Original value
|
|
1594
|
+
translated: translated ?? value
|
|
1595
|
+
// Translated value or original if no translation
|
|
1596
|
+
});
|
|
1597
|
+
}
|
|
1598
|
+
return result;
|
|
1599
|
+
}
|
|
1600
|
+
};
|
|
1601
|
+
|
|
1602
|
+
// src/interfaces/dictionary.ts
|
|
1603
|
+
var InMemoryDictionary = class {
|
|
1604
|
+
/**
|
|
1605
|
+
* @param translations - Map of language code to canonical→localized mappings
|
|
1606
|
+
*
|
|
1607
|
+
* @example
|
|
1608
|
+
* new InMemoryDictionary({
|
|
1609
|
+
* en: { select: 'select', insert: 'insert' },
|
|
1610
|
+
* es: { select: 'seleccionar', insert: 'insertar' },
|
|
1611
|
+
* ja: { select: '選択', insert: '挿入' }
|
|
1612
|
+
* })
|
|
1613
|
+
*/
|
|
1614
|
+
constructor(translations) {
|
|
1615
|
+
this.translations = translations;
|
|
1616
|
+
}
|
|
1617
|
+
lookup(word, language) {
|
|
1618
|
+
const langDict = this.translations[language];
|
|
1619
|
+
if (!langDict) return void 0;
|
|
1620
|
+
const wordLower = word.toLowerCase();
|
|
1621
|
+
for (const [canonical, localized] of Object.entries(langDict)) {
|
|
1622
|
+
if (localized.toLowerCase() === wordLower) {
|
|
1623
|
+
return canonical;
|
|
1624
|
+
}
|
|
1625
|
+
}
|
|
1626
|
+
return void 0;
|
|
1627
|
+
}
|
|
1628
|
+
translate(canonical, targetLanguage) {
|
|
1629
|
+
return this.translations[targetLanguage]?.[canonical];
|
|
1630
|
+
}
|
|
1631
|
+
getAllTranslations(word, sourceLanguage) {
|
|
1632
|
+
const canonical = this.lookup(word, sourceLanguage);
|
|
1633
|
+
if (!canonical) return {};
|
|
1634
|
+
const result = {};
|
|
1635
|
+
for (const [lang, dict] of Object.entries(this.translations)) {
|
|
1636
|
+
const translation = dict[canonical];
|
|
1637
|
+
if (translation) {
|
|
1638
|
+
result[lang] = translation;
|
|
1639
|
+
}
|
|
1640
|
+
}
|
|
1641
|
+
return result;
|
|
1642
|
+
}
|
|
1643
|
+
};
|
|
1644
|
+
var NullDictionary = class {
|
|
1645
|
+
lookup(word, _language) {
|
|
1646
|
+
return word;
|
|
1647
|
+
}
|
|
1648
|
+
translate(canonical, _targetLanguage) {
|
|
1649
|
+
return canonical;
|
|
1650
|
+
}
|
|
1651
|
+
getAllTranslations(word, _sourceLanguage) {
|
|
1652
|
+
return { en: word };
|
|
1653
|
+
}
|
|
1654
|
+
};
|
|
1655
|
+
|
|
1656
|
+
// src/interfaces/profile-provider.ts
|
|
1657
|
+
var InMemoryProfileProvider = class {
|
|
1658
|
+
/**
|
|
1659
|
+
* @param profiles - Map of language code to profile
|
|
1660
|
+
*
|
|
1661
|
+
* @example
|
|
1662
|
+
* new InMemoryProfileProvider({
|
|
1663
|
+
* en: { code: 'en', wordOrder: 'SVO', ... },
|
|
1664
|
+
* ja: { code: 'ja', wordOrder: 'SOV', ... }
|
|
1665
|
+
* })
|
|
1666
|
+
*/
|
|
1667
|
+
constructor(profiles) {
|
|
1668
|
+
this.profiles = profiles;
|
|
1669
|
+
}
|
|
1670
|
+
getProfile(language) {
|
|
1671
|
+
return this.profiles[language];
|
|
1672
|
+
}
|
|
1673
|
+
getSupportedLanguages() {
|
|
1674
|
+
return Object.keys(this.profiles);
|
|
1675
|
+
}
|
|
1676
|
+
hasLanguage(language) {
|
|
1677
|
+
return language in this.profiles;
|
|
1678
|
+
}
|
|
1679
|
+
};
|
|
1680
|
+
var NullProfileProvider = class {
|
|
1681
|
+
getProfile(_language) {
|
|
1682
|
+
return void 0;
|
|
1683
|
+
}
|
|
1684
|
+
getSupportedLanguages() {
|
|
1685
|
+
return [];
|
|
1686
|
+
}
|
|
1687
|
+
hasLanguage(_language) {
|
|
1688
|
+
return false;
|
|
1689
|
+
}
|
|
1690
|
+
};
|
|
1691
|
+
|
|
1692
|
+
// src/interfaces/value-extractor.ts
|
|
1693
|
+
var StringLiteralExtractor = class {
|
|
1694
|
+
constructor() {
|
|
1695
|
+
this.name = "string-literal";
|
|
1696
|
+
}
|
|
1697
|
+
canExtract(input, position) {
|
|
1698
|
+
const char = input[position];
|
|
1699
|
+
return char === '"' || char === "'" || char === "`" || char === "\u201C" || // Chinese double quote open "
|
|
1700
|
+
char === "\u2018";
|
|
1701
|
+
}
|
|
1702
|
+
extract(input, position) {
|
|
1703
|
+
const quote = input[position];
|
|
1704
|
+
if (quote === "\u201C") {
|
|
1705
|
+
let length2 = 1;
|
|
1706
|
+
while (position + length2 < input.length) {
|
|
1707
|
+
if (input[position + length2] === "\u201D") {
|
|
1708
|
+
length2++;
|
|
1709
|
+
return { value: input.substring(position, position + length2), length: length2 };
|
|
1710
|
+
}
|
|
1711
|
+
length2++;
|
|
1712
|
+
}
|
|
1713
|
+
return null;
|
|
1714
|
+
}
|
|
1715
|
+
if (quote === "\u2018") {
|
|
1716
|
+
let length2 = 1;
|
|
1717
|
+
while (position + length2 < input.length) {
|
|
1718
|
+
if (input[position + length2] === "\u2019") {
|
|
1719
|
+
length2++;
|
|
1720
|
+
return { value: input.substring(position, position + length2), length: length2 };
|
|
1721
|
+
}
|
|
1722
|
+
length2++;
|
|
1723
|
+
}
|
|
1724
|
+
return null;
|
|
1725
|
+
}
|
|
1726
|
+
let length = 1;
|
|
1727
|
+
let escaped = false;
|
|
1728
|
+
while (position + length < input.length) {
|
|
1729
|
+
const char = input[position + length];
|
|
1730
|
+
if (escaped) {
|
|
1731
|
+
escaped = false;
|
|
1732
|
+
length++;
|
|
1733
|
+
continue;
|
|
1734
|
+
}
|
|
1735
|
+
if (char === "\\") {
|
|
1736
|
+
escaped = true;
|
|
1737
|
+
length++;
|
|
1738
|
+
continue;
|
|
1739
|
+
}
|
|
1740
|
+
if (char === quote) {
|
|
1741
|
+
length++;
|
|
1742
|
+
return {
|
|
1743
|
+
value: input.substring(position, position + length),
|
|
1744
|
+
length
|
|
1745
|
+
};
|
|
1746
|
+
}
|
|
1747
|
+
length++;
|
|
1748
|
+
}
|
|
1749
|
+
return null;
|
|
1750
|
+
}
|
|
1751
|
+
};
|
|
1752
|
+
var NumberExtractor = class {
|
|
1753
|
+
constructor() {
|
|
1754
|
+
this.name = "number";
|
|
1755
|
+
}
|
|
1756
|
+
canExtract(input, position) {
|
|
1757
|
+
return /\d/.test(input[position]);
|
|
1758
|
+
}
|
|
1759
|
+
extract(input, position) {
|
|
1760
|
+
let length = 0;
|
|
1761
|
+
let hasDecimal = false;
|
|
1762
|
+
while (position + length < input.length) {
|
|
1763
|
+
const char = input[position + length];
|
|
1764
|
+
if (/\d/.test(char)) {
|
|
1765
|
+
length++;
|
|
1766
|
+
} else if (char === "." && !hasDecimal) {
|
|
1767
|
+
hasDecimal = true;
|
|
1768
|
+
length++;
|
|
1769
|
+
} else {
|
|
1770
|
+
break;
|
|
1771
|
+
}
|
|
1772
|
+
}
|
|
1773
|
+
if (length === 0) return null;
|
|
1774
|
+
const numValue = input.substring(position, position + length);
|
|
1775
|
+
const afterNum = position + length;
|
|
1776
|
+
if (afterNum < input.length) {
|
|
1777
|
+
const remaining = input.slice(afterNum);
|
|
1778
|
+
const cjkMultiUnits = [
|
|
1779
|
+
{ pattern: "\u6BEB\u79D2", suffix: "ms" },
|
|
1780
|
+
// Chinese milliseconds
|
|
1781
|
+
{ pattern: "\u5206\u949F", suffix: "m" },
|
|
1782
|
+
// Chinese minutes
|
|
1783
|
+
{ pattern: "\u5C0F\u65F6", suffix: "h" },
|
|
1784
|
+
// Chinese hours
|
|
1785
|
+
{ pattern: "\u30DF\u30EA\u79D2", suffix: "ms" },
|
|
1786
|
+
// Japanese milliseconds
|
|
1787
|
+
{ pattern: "\u6642\u9593", suffix: "h" }
|
|
1788
|
+
// Japanese hours
|
|
1789
|
+
];
|
|
1790
|
+
for (const unit of cjkMultiUnits) {
|
|
1791
|
+
if (remaining.startsWith(unit.pattern)) {
|
|
1792
|
+
return {
|
|
1793
|
+
value: numValue + unit.suffix,
|
|
1794
|
+
length: length + unit.pattern.length,
|
|
1795
|
+
metadata: { hasTimeUnit: true }
|
|
1796
|
+
};
|
|
1797
|
+
}
|
|
1798
|
+
}
|
|
1799
|
+
if (remaining.startsWith("ms")) {
|
|
1800
|
+
return {
|
|
1801
|
+
value: numValue + "ms",
|
|
1802
|
+
length: length + 2,
|
|
1803
|
+
metadata: { hasTimeUnit: true }
|
|
1804
|
+
};
|
|
1805
|
+
}
|
|
1806
|
+
const cjkSingleUnits = [
|
|
1807
|
+
{ pattern: "\u79D2", suffix: "s" },
|
|
1808
|
+
// CJK seconds
|
|
1809
|
+
{ pattern: "\u5206", suffix: "m" }
|
|
1810
|
+
// CJK minutes
|
|
1811
|
+
];
|
|
1812
|
+
for (const unit of cjkSingleUnits) {
|
|
1813
|
+
if (remaining.startsWith(unit.pattern)) {
|
|
1814
|
+
return {
|
|
1815
|
+
value: numValue + unit.suffix,
|
|
1816
|
+
length: length + 1,
|
|
1817
|
+
metadata: { hasTimeUnit: true }
|
|
1818
|
+
};
|
|
1819
|
+
}
|
|
1820
|
+
}
|
|
1821
|
+
if (/^[smh](?![a-zA-Z])/.test(remaining)) {
|
|
1822
|
+
return {
|
|
1823
|
+
value: numValue + remaining[0],
|
|
1824
|
+
length: length + 1,
|
|
1825
|
+
metadata: { hasTimeUnit: true }
|
|
1826
|
+
};
|
|
1827
|
+
}
|
|
1828
|
+
}
|
|
1829
|
+
return { value: numValue, length };
|
|
1830
|
+
}
|
|
1831
|
+
};
|
|
1832
|
+
var IdentifierExtractor = class {
|
|
1833
|
+
constructor() {
|
|
1834
|
+
this.name = "identifier";
|
|
1835
|
+
}
|
|
1836
|
+
canExtract(input, position) {
|
|
1837
|
+
return /[a-zA-Z_]/.test(input[position]);
|
|
1838
|
+
}
|
|
1839
|
+
extract(input, position) {
|
|
1840
|
+
let length = 0;
|
|
1841
|
+
while (position + length < input.length) {
|
|
1842
|
+
const char = input[position + length];
|
|
1843
|
+
if (/[a-zA-Z0-9_]/.test(char)) {
|
|
1844
|
+
length++;
|
|
1845
|
+
} else {
|
|
1846
|
+
break;
|
|
1847
|
+
}
|
|
1848
|
+
}
|
|
1849
|
+
return length > 0 ? {
|
|
1850
|
+
value: input.substring(position, position + length),
|
|
1851
|
+
length
|
|
1852
|
+
} : null;
|
|
1853
|
+
}
|
|
1854
|
+
};
|
|
1855
|
+
var UnicodeIdentifierExtractor = class {
|
|
1856
|
+
constructor() {
|
|
1857
|
+
this.name = "unicode-identifier";
|
|
1858
|
+
}
|
|
1859
|
+
canExtract(input, position) {
|
|
1860
|
+
const code = input.charCodeAt(position);
|
|
1861
|
+
if (code < 128) return false;
|
|
1862
|
+
return /\p{L}/u.test(input[position]);
|
|
1863
|
+
}
|
|
1864
|
+
extract(input, position) {
|
|
1865
|
+
let length = 0;
|
|
1866
|
+
while (position + length < input.length) {
|
|
1867
|
+
const char = input[position + length];
|
|
1868
|
+
if (/[\p{L}\p{N}\p{M}]/u.test(char)) {
|
|
1869
|
+
length++;
|
|
1870
|
+
} else {
|
|
1871
|
+
break;
|
|
1872
|
+
}
|
|
1873
|
+
}
|
|
1874
|
+
return length > 0 ? { value: input.substring(position, position + length), length } : null;
|
|
1875
|
+
}
|
|
1876
|
+
};
|
|
1877
|
+
var WhitespaceExtractor = class {
|
|
1878
|
+
constructor() {
|
|
1879
|
+
this.name = "whitespace";
|
|
1880
|
+
}
|
|
1881
|
+
canExtract(input, position) {
|
|
1882
|
+
return /\s/.test(input[position]);
|
|
1883
|
+
}
|
|
1884
|
+
extract(input, position) {
|
|
1885
|
+
let length = 0;
|
|
1886
|
+
while (position + length < input.length && /\s/.test(input[position + length])) {
|
|
1887
|
+
length++;
|
|
1888
|
+
}
|
|
1889
|
+
return length > 0 ? {
|
|
1890
|
+
value: input.substring(position, position + length),
|
|
1891
|
+
length
|
|
1892
|
+
} : null;
|
|
1893
|
+
}
|
|
1894
|
+
};
|
|
1895
|
+
function isContextAwareExtractor(extractor) {
|
|
1896
|
+
return "setContext" in extractor && typeof extractor.setContext === "function";
|
|
1897
|
+
}
|
|
1898
|
+
function createTokenizerContext(tokenizer) {
|
|
1899
|
+
const ctx = {
|
|
1900
|
+
language: tokenizer.language,
|
|
1901
|
+
direction: tokenizer.direction,
|
|
1902
|
+
lookupKeyword: tokenizer.lookupKeyword.bind(tokenizer),
|
|
1903
|
+
isKeyword: tokenizer.isKeyword.bind(tokenizer),
|
|
1904
|
+
isKeywordStart: tokenizer.isKeywordStart.bind(tokenizer)
|
|
1905
|
+
};
|
|
1906
|
+
if (tokenizer.normalizer) {
|
|
1907
|
+
return { ...ctx, normalizer: tokenizer.normalizer };
|
|
1908
|
+
}
|
|
1909
|
+
return ctx;
|
|
1910
|
+
}
|
|
1911
|
+
|
|
1912
|
+
// src/api/create-dsl.ts
|
|
1913
|
+
var DSLRegistry = class {
|
|
1914
|
+
constructor(config) {
|
|
1915
|
+
this.patterns = /* @__PURE__ */ new Map();
|
|
1916
|
+
this.tokenizers = /* @__PURE__ */ new Map();
|
|
1917
|
+
this.schemas = config.schemas;
|
|
1918
|
+
for (const lang of config.languages) {
|
|
1919
|
+
this.registerLanguage(lang);
|
|
1920
|
+
}
|
|
1921
|
+
}
|
|
1922
|
+
registerLanguage(lang) {
|
|
1923
|
+
this.tokenizers.set(lang.code, lang.tokenizer);
|
|
1924
|
+
const patterns = [];
|
|
1925
|
+
for (const schema of this.schemas) {
|
|
1926
|
+
const pattern = generatePattern(schema, lang.patternProfile);
|
|
1927
|
+
patterns.push(pattern);
|
|
1928
|
+
}
|
|
1929
|
+
this.patterns.set(lang.code, patterns);
|
|
1930
|
+
}
|
|
1931
|
+
getPatterns(language) {
|
|
1932
|
+
return this.patterns.get(language) || [];
|
|
1933
|
+
}
|
|
1934
|
+
getTokenizer(language) {
|
|
1935
|
+
return this.tokenizers.get(language);
|
|
1936
|
+
}
|
|
1937
|
+
getSupportedLanguages() {
|
|
1938
|
+
return Array.from(this.patterns.keys());
|
|
1939
|
+
}
|
|
1940
|
+
};
|
|
1941
|
+
var MultilingualDSLImpl = class {
|
|
1942
|
+
constructor(config, registry, transformer) {
|
|
1943
|
+
this.registry = registry;
|
|
1944
|
+
this.matcher = new PatternMatcher();
|
|
1945
|
+
this.transformer = transformer;
|
|
1946
|
+
if (config.codeGenerator) {
|
|
1947
|
+
this.codeGenerator = config.codeGenerator;
|
|
1948
|
+
}
|
|
1949
|
+
}
|
|
1950
|
+
parse(input, language) {
|
|
1951
|
+
const result = this.parseWithConfidence(input, language);
|
|
1952
|
+
return result.node;
|
|
1953
|
+
}
|
|
1954
|
+
parseWithConfidence(input, language) {
|
|
1955
|
+
const tokenizer = this.registry.getTokenizer(language);
|
|
1956
|
+
if (!tokenizer) {
|
|
1957
|
+
throw new Error(`No tokenizer registered for language: ${language}`);
|
|
1958
|
+
}
|
|
1959
|
+
const tokens = tokenizer.tokenize(input);
|
|
1960
|
+
const patterns = this.registry.getPatterns(language);
|
|
1961
|
+
const profile = {
|
|
1962
|
+
code: language
|
|
1963
|
+
};
|
|
1964
|
+
for (const pattern of patterns) {
|
|
1965
|
+
const match = this.matcher.matchPattern(tokens, pattern, profile);
|
|
1966
|
+
if (match) {
|
|
1967
|
+
const node = {
|
|
1968
|
+
kind: "command",
|
|
1969
|
+
action: pattern.command,
|
|
1970
|
+
roles: match.captured,
|
|
1971
|
+
metadata: {
|
|
1972
|
+
sourceLanguage: language,
|
|
1973
|
+
sourceText: input,
|
|
1974
|
+
patternId: pattern.id,
|
|
1975
|
+
confidence: match.confidence
|
|
1976
|
+
}
|
|
1977
|
+
};
|
|
1978
|
+
return { node, confidence: match.confidence };
|
|
1979
|
+
}
|
|
1980
|
+
}
|
|
1981
|
+
throw new Error(`No pattern matched for input: ${input}`);
|
|
1982
|
+
}
|
|
1983
|
+
validate(input, language) {
|
|
1984
|
+
try {
|
|
1985
|
+
const result = this.parseWithConfidence(input, language);
|
|
1986
|
+
return {
|
|
1987
|
+
valid: true,
|
|
1988
|
+
node: result.node
|
|
1989
|
+
};
|
|
1990
|
+
} catch (error) {
|
|
1991
|
+
return {
|
|
1992
|
+
valid: false,
|
|
1993
|
+
errors: [error instanceof Error ? error.message : String(error)]
|
|
1994
|
+
};
|
|
1995
|
+
}
|
|
1996
|
+
}
|
|
1997
|
+
translate(input, fromLanguage, toLanguage) {
|
|
1998
|
+
return this.transformer.transform(input, fromLanguage, toLanguage);
|
|
1999
|
+
}
|
|
2000
|
+
compile(input, language) {
|
|
2001
|
+
if (!this.codeGenerator) {
|
|
2002
|
+
return {
|
|
2003
|
+
ok: false,
|
|
2004
|
+
errors: ["No code generator configured for this DSL"]
|
|
2005
|
+
};
|
|
2006
|
+
}
|
|
2007
|
+
try {
|
|
2008
|
+
const result = this.parseWithConfidence(input, language);
|
|
2009
|
+
const code = this.codeGenerator.generate(result.node);
|
|
2010
|
+
return {
|
|
2011
|
+
ok: true,
|
|
2012
|
+
code,
|
|
2013
|
+
node: result.node,
|
|
2014
|
+
metadata: {
|
|
2015
|
+
parser: "semantic",
|
|
2016
|
+
confidence: result.confidence
|
|
2017
|
+
}
|
|
2018
|
+
};
|
|
2019
|
+
} catch (error) {
|
|
2020
|
+
return {
|
|
2021
|
+
ok: false,
|
|
2022
|
+
errors: [error instanceof Error ? error.message : String(error)]
|
|
2023
|
+
};
|
|
2024
|
+
}
|
|
2025
|
+
}
|
|
2026
|
+
getSupportedLanguages() {
|
|
2027
|
+
return this.registry.getSupportedLanguages();
|
|
2028
|
+
}
|
|
2029
|
+
};
|
|
2030
|
+
function createDefaultDictionary(config) {
|
|
2031
|
+
const translations = {};
|
|
2032
|
+
for (const lang of config.languages) {
|
|
2033
|
+
const keywords = {};
|
|
2034
|
+
for (const [canonical, keywordDef] of Object.entries(lang.patternProfile.keywords)) {
|
|
2035
|
+
keywords[canonical] = keywordDef.primary;
|
|
2036
|
+
}
|
|
2037
|
+
translations[lang.code] = keywords;
|
|
2038
|
+
}
|
|
2039
|
+
return new InMemoryDictionary(translations);
|
|
2040
|
+
}
|
|
2041
|
+
function createDefaultProfileProvider(config) {
|
|
2042
|
+
const profiles = {};
|
|
2043
|
+
for (const lang of config.languages) {
|
|
2044
|
+
if (lang.grammarProfile) {
|
|
2045
|
+
profiles[lang.code] = lang.grammarProfile;
|
|
2046
|
+
}
|
|
2047
|
+
}
|
|
2048
|
+
return new InMemoryProfileProvider(profiles);
|
|
2049
|
+
}
|
|
2050
|
+
function createMultilingualDSL(config) {
|
|
2051
|
+
const dictionary = config.dictionary ?? createDefaultDictionary(config);
|
|
2052
|
+
const profileProvider = config.profileProvider ?? createDefaultProfileProvider(config);
|
|
2053
|
+
const transformer = new GrammarTransformer({
|
|
2054
|
+
dictionary,
|
|
2055
|
+
profileProvider
|
|
2056
|
+
});
|
|
2057
|
+
const registry = new DSLRegistry(config);
|
|
2058
|
+
return new MultilingualDSLImpl(config, registry, transformer);
|
|
2059
|
+
}
|
|
2060
|
+
|
|
2061
|
+
// src/api/domain-registry.ts
|
|
2062
|
+
var DomainRegistry = class {
|
|
2063
|
+
constructor() {
|
|
2064
|
+
this.descriptors = /* @__PURE__ */ new Map();
|
|
2065
|
+
this.dslCache = /* @__PURE__ */ new Map();
|
|
2066
|
+
this.rendererCache = /* @__PURE__ */ new Map();
|
|
2067
|
+
}
|
|
2068
|
+
/**
|
|
2069
|
+
* Register a domain.
|
|
2070
|
+
* @throws if a domain with the same name is already registered
|
|
2071
|
+
*/
|
|
2072
|
+
register(descriptor) {
|
|
2073
|
+
if (this.descriptors.has(descriptor.name)) {
|
|
2074
|
+
throw new Error(`Domain already registered: ${descriptor.name}`);
|
|
2075
|
+
}
|
|
2076
|
+
this.descriptors.set(descriptor.name, descriptor);
|
|
2077
|
+
}
|
|
2078
|
+
/**
|
|
2079
|
+
* Get all registered domain names.
|
|
2080
|
+
*/
|
|
2081
|
+
getDomainNames() {
|
|
2082
|
+
return Array.from(this.descriptors.keys());
|
|
2083
|
+
}
|
|
2084
|
+
/**
|
|
2085
|
+
* Get a specific domain descriptor.
|
|
2086
|
+
*/
|
|
2087
|
+
getDescriptor(name) {
|
|
2088
|
+
return this.descriptors.get(name);
|
|
2089
|
+
}
|
|
2090
|
+
/**
|
|
2091
|
+
* Generate MCP tool definitions for all registered domains.
|
|
2092
|
+
*/
|
|
2093
|
+
getToolDefinitions() {
|
|
2094
|
+
const tools = [];
|
|
2095
|
+
for (const desc of this.descriptors.values()) {
|
|
2096
|
+
tools.push(...generateToolDefinitions(desc));
|
|
2097
|
+
}
|
|
2098
|
+
return tools;
|
|
2099
|
+
}
|
|
2100
|
+
/**
|
|
2101
|
+
* Check if a tool name belongs to a registered domain.
|
|
2102
|
+
*/
|
|
2103
|
+
canHandle(toolName) {
|
|
2104
|
+
const parsed = parseToolName(toolName);
|
|
2105
|
+
if (!parsed) return false;
|
|
2106
|
+
return this.descriptors.has(parsed.domain);
|
|
2107
|
+
}
|
|
2108
|
+
/**
|
|
2109
|
+
* Handle a tool call by dispatching to the appropriate domain.
|
|
2110
|
+
* Returns null if the tool name doesn't match any registered domain.
|
|
2111
|
+
*/
|
|
2112
|
+
async handleToolCall(toolName, args) {
|
|
2113
|
+
const parsed = parseToolName(toolName);
|
|
2114
|
+
if (!parsed) return null;
|
|
2115
|
+
const descriptor = this.descriptors.get(parsed.domain);
|
|
2116
|
+
if (!descriptor) return null;
|
|
2117
|
+
try {
|
|
2118
|
+
const dsl = await this.getDSL(descriptor);
|
|
2119
|
+
switch (parsed.operation) {
|
|
2120
|
+
case "parse":
|
|
2121
|
+
return await this.handleParse(descriptor, dsl, args);
|
|
2122
|
+
case "compile":
|
|
2123
|
+
return await this.handleCompile(descriptor, dsl, args);
|
|
2124
|
+
case "validate":
|
|
2125
|
+
return await this.handleValidate(descriptor, dsl, args);
|
|
2126
|
+
case "translate":
|
|
2127
|
+
return await this.handleTranslate(descriptor, dsl, args);
|
|
2128
|
+
default:
|
|
2129
|
+
return jsonResponse({ error: `Unknown operation: ${parsed.operation}` }, true);
|
|
2130
|
+
}
|
|
2131
|
+
} catch (error) {
|
|
2132
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
2133
|
+
return jsonResponse({ error: `${descriptor.name} tool error: ${message}` }, true);
|
|
2134
|
+
}
|
|
2135
|
+
}
|
|
2136
|
+
/**
|
|
2137
|
+
* Get the DSL instance for a named domain.
|
|
2138
|
+
* Returns null if the domain is not registered.
|
|
2139
|
+
* The DSL is lazily created and cached.
|
|
2140
|
+
*/
|
|
2141
|
+
async getDSLForDomain(name) {
|
|
2142
|
+
const descriptor = this.descriptors.get(name);
|
|
2143
|
+
if (!descriptor) return null;
|
|
2144
|
+
return this.getDSL(descriptor);
|
|
2145
|
+
}
|
|
2146
|
+
// ---------------------------------------------------------------------------
|
|
2147
|
+
// Private: DSL/renderer lifecycle
|
|
2148
|
+
// ---------------------------------------------------------------------------
|
|
2149
|
+
async getDSL(descriptor) {
|
|
2150
|
+
const cached = this.dslCache.get(descriptor.name);
|
|
2151
|
+
if (cached) return cached;
|
|
2152
|
+
const dsl = await descriptor.getDSL();
|
|
2153
|
+
this.dslCache.set(descriptor.name, dsl);
|
|
2154
|
+
return dsl;
|
|
2155
|
+
}
|
|
2156
|
+
async getRenderer(descriptor) {
|
|
2157
|
+
if (!descriptor.getRenderer) return null;
|
|
2158
|
+
const cached = this.rendererCache.get(descriptor.name);
|
|
2159
|
+
if (cached) return cached;
|
|
2160
|
+
const renderer = await descriptor.getRenderer();
|
|
2161
|
+
this.rendererCache.set(descriptor.name, renderer);
|
|
2162
|
+
return renderer;
|
|
2163
|
+
}
|
|
2164
|
+
// ---------------------------------------------------------------------------
|
|
2165
|
+
// Private: Tool handlers
|
|
2166
|
+
// ---------------------------------------------------------------------------
|
|
2167
|
+
async handleParse(descriptor, dsl, args) {
|
|
2168
|
+
const input = getString(args, descriptor.inputLabel);
|
|
2169
|
+
if (!input) return missingParam(descriptor.inputLabel);
|
|
2170
|
+
const language = getString(args, "language", "en");
|
|
2171
|
+
const node = dsl.parse(input, language);
|
|
2172
|
+
const roles = {};
|
|
2173
|
+
for (const [key, value] of node.roles) {
|
|
2174
|
+
roles[key] = value;
|
|
2175
|
+
}
|
|
2176
|
+
return jsonResponse({
|
|
2177
|
+
action: node.action,
|
|
2178
|
+
roles,
|
|
2179
|
+
language,
|
|
2180
|
+
[descriptor.inputLabel]: input
|
|
2181
|
+
});
|
|
2182
|
+
}
|
|
2183
|
+
async handleCompile(descriptor, dsl, args) {
|
|
2184
|
+
const input = getString(args, descriptor.inputLabel);
|
|
2185
|
+
if (!input) return missingParam(descriptor.inputLabel);
|
|
2186
|
+
const language = getString(args, "language", "en");
|
|
2187
|
+
const result = dsl.compile(input, language);
|
|
2188
|
+
return jsonResponse({
|
|
2189
|
+
ok: result.ok,
|
|
2190
|
+
code: result.code,
|
|
2191
|
+
errors: result.errors,
|
|
2192
|
+
language,
|
|
2193
|
+
input
|
|
2194
|
+
});
|
|
2195
|
+
}
|
|
2196
|
+
async handleValidate(descriptor, dsl, args) {
|
|
2197
|
+
const input = getString(args, descriptor.inputLabel);
|
|
2198
|
+
if (!input) return missingParam(descriptor.inputLabel);
|
|
2199
|
+
const language = getString(args, "language", "en");
|
|
2200
|
+
const result = dsl.validate(input, language);
|
|
2201
|
+
return jsonResponse({
|
|
2202
|
+
valid: result.valid,
|
|
2203
|
+
errors: result.errors,
|
|
2204
|
+
language,
|
|
2205
|
+
[descriptor.inputLabel]: input
|
|
2206
|
+
});
|
|
2207
|
+
}
|
|
2208
|
+
async handleTranslate(descriptor, dsl, args) {
|
|
2209
|
+
const input = getString(args, descriptor.inputLabel);
|
|
2210
|
+
if (!input) return missingParam(descriptor.inputLabel);
|
|
2211
|
+
const from = getString(args, "from");
|
|
2212
|
+
if (!from) return missingParam("from");
|
|
2213
|
+
const to = getString(args, "to");
|
|
2214
|
+
if (!to) return missingParam("to");
|
|
2215
|
+
const node = dsl.parse(input, from);
|
|
2216
|
+
const compiled = dsl.compile(input, from);
|
|
2217
|
+
let rendered = null;
|
|
2218
|
+
const renderer = await this.getRenderer(descriptor);
|
|
2219
|
+
if (renderer) {
|
|
2220
|
+
try {
|
|
2221
|
+
rendered = typeof renderer === "function" ? renderer(node, to) : renderer.render(node, to);
|
|
2222
|
+
} catch {
|
|
2223
|
+
}
|
|
2224
|
+
}
|
|
2225
|
+
const roles = {};
|
|
2226
|
+
for (const [key, value] of node.roles) {
|
|
2227
|
+
roles[key] = value;
|
|
2228
|
+
}
|
|
2229
|
+
return jsonResponse({
|
|
2230
|
+
input: { [descriptor.inputLabel]: input, language: from },
|
|
2231
|
+
...rendered != null && { rendered: { text: rendered, language: to } },
|
|
2232
|
+
semantic: { action: node.action, roles },
|
|
2233
|
+
...compiled.ok && compiled.code != null && { compiled: compiled.code }
|
|
2234
|
+
});
|
|
2235
|
+
}
|
|
2236
|
+
};
|
|
2237
|
+
function generateToolDefinitions(desc) {
|
|
2238
|
+
const tools = [];
|
|
2239
|
+
const ops = desc.tools ?? ["parse", "compile", "validate", "translate"];
|
|
2240
|
+
const langList = desc.languages.join(", ");
|
|
2241
|
+
for (const op of ops) {
|
|
2242
|
+
switch (op) {
|
|
2243
|
+
case "parse":
|
|
2244
|
+
tools.push({
|
|
2245
|
+
name: `parse_${desc.name}`,
|
|
2246
|
+
description: `Parse a ${desc.description} input into a semantic representation. Supports: ${langList}.`,
|
|
2247
|
+
inputSchema: {
|
|
2248
|
+
type: "object",
|
|
2249
|
+
properties: {
|
|
2250
|
+
[desc.inputLabel]: {
|
|
2251
|
+
type: "string",
|
|
2252
|
+
description: desc.inputDescription
|
|
2253
|
+
},
|
|
2254
|
+
language: {
|
|
2255
|
+
type: "string",
|
|
2256
|
+
description: `Language code: ${langList}`,
|
|
2257
|
+
default: "en"
|
|
2258
|
+
}
|
|
2259
|
+
},
|
|
2260
|
+
required: [desc.inputLabel]
|
|
2261
|
+
}
|
|
2262
|
+
});
|
|
2263
|
+
break;
|
|
2264
|
+
case "compile":
|
|
2265
|
+
tools.push({
|
|
2266
|
+
name: `compile_${desc.name}`,
|
|
2267
|
+
description: `Compile a ${desc.description} input to target code. Supports: ${langList}.`,
|
|
2268
|
+
inputSchema: {
|
|
2269
|
+
type: "object",
|
|
2270
|
+
properties: {
|
|
2271
|
+
[desc.inputLabel]: {
|
|
2272
|
+
type: "string",
|
|
2273
|
+
description: desc.inputDescription
|
|
2274
|
+
},
|
|
2275
|
+
language: {
|
|
2276
|
+
type: "string",
|
|
2277
|
+
description: `Language code: ${langList}`,
|
|
2278
|
+
default: "en"
|
|
2279
|
+
}
|
|
2280
|
+
},
|
|
2281
|
+
required: [desc.inputLabel]
|
|
2282
|
+
}
|
|
2283
|
+
});
|
|
2284
|
+
break;
|
|
2285
|
+
case "validate":
|
|
2286
|
+
tools.push({
|
|
2287
|
+
name: `validate_${desc.name}`,
|
|
2288
|
+
description: `Validate ${desc.description} syntax. Returns whether it parses successfully and any errors. Supports: ${langList}.`,
|
|
2289
|
+
inputSchema: {
|
|
2290
|
+
type: "object",
|
|
2291
|
+
properties: {
|
|
2292
|
+
[desc.inputLabel]: {
|
|
2293
|
+
type: "string",
|
|
2294
|
+
description: `${desc.inputDescription} to validate`
|
|
2295
|
+
},
|
|
2296
|
+
language: {
|
|
2297
|
+
type: "string",
|
|
2298
|
+
description: `Language code: ${langList}`,
|
|
2299
|
+
default: "en"
|
|
2300
|
+
}
|
|
2301
|
+
},
|
|
2302
|
+
required: [desc.inputLabel]
|
|
2303
|
+
}
|
|
2304
|
+
});
|
|
2305
|
+
break;
|
|
2306
|
+
case "translate":
|
|
2307
|
+
tools.push({
|
|
2308
|
+
name: `translate_${desc.name}`,
|
|
2309
|
+
description: `Translate ${desc.description} input between natural languages. Parses in source language and renders in target language.`,
|
|
2310
|
+
inputSchema: {
|
|
2311
|
+
type: "object",
|
|
2312
|
+
properties: {
|
|
2313
|
+
[desc.inputLabel]: {
|
|
2314
|
+
type: "string",
|
|
2315
|
+
description: `${desc.inputDescription} to translate`
|
|
2316
|
+
},
|
|
2317
|
+
from: {
|
|
2318
|
+
type: "string",
|
|
2319
|
+
description: `Source language code: ${langList}`
|
|
2320
|
+
},
|
|
2321
|
+
to: {
|
|
2322
|
+
type: "string",
|
|
2323
|
+
description: `Target language code: ${langList}`
|
|
2324
|
+
}
|
|
2325
|
+
},
|
|
2326
|
+
required: [desc.inputLabel, "from", "to"]
|
|
2327
|
+
}
|
|
2328
|
+
});
|
|
2329
|
+
break;
|
|
2330
|
+
}
|
|
2331
|
+
}
|
|
2332
|
+
return tools;
|
|
2333
|
+
}
|
|
2334
|
+
function parseToolName(name) {
|
|
2335
|
+
const match = name.match(/^(parse|compile|validate|translate)_(.+)$/);
|
|
2336
|
+
if (!match) return null;
|
|
2337
|
+
return { operation: match[1], domain: match[2] };
|
|
2338
|
+
}
|
|
2339
|
+
function getString(args, name, defaultValue = "") {
|
|
2340
|
+
const value = args[name];
|
|
2341
|
+
return typeof value === "string" ? value : defaultValue;
|
|
2342
|
+
}
|
|
2343
|
+
function jsonResponse(data, isError) {
|
|
2344
|
+
return {
|
|
2345
|
+
content: [{ type: "text", text: JSON.stringify(data, null, 2) }],
|
|
2346
|
+
...isError && { isError: true }
|
|
2347
|
+
};
|
|
2348
|
+
}
|
|
2349
|
+
function missingParam(param) {
|
|
2350
|
+
return jsonResponse({ error: `Missing required parameter: ${param}` }, true);
|
|
2351
|
+
}
|
|
2352
|
+
|
|
2353
|
+
// src/api/dispatcher.ts
|
|
2354
|
+
var COMMENT_PATTERN = /^\s*(\/\/|--|#)/;
|
|
2355
|
+
function isSkippableLine(line) {
|
|
2356
|
+
const trimmed = line.trim();
|
|
2357
|
+
return trimmed === "" || COMMENT_PATTERN.test(trimmed);
|
|
2358
|
+
}
|
|
2359
|
+
var CrossDomainDispatcher = class {
|
|
2360
|
+
constructor(registry, options) {
|
|
2361
|
+
this.registry = registry;
|
|
2362
|
+
this.minConfidence = options?.minConfidence ?? 0.5;
|
|
2363
|
+
this.priority = options?.priority ?? [];
|
|
2364
|
+
}
|
|
2365
|
+
/**
|
|
2366
|
+
* Auto-detect which domain handles the input.
|
|
2367
|
+
* Tries all registered domains and returns the highest-confidence match.
|
|
2368
|
+
* Returns null if no domain matches above minConfidence.
|
|
2369
|
+
*/
|
|
2370
|
+
async detect(input, language = "en") {
|
|
2371
|
+
const domainNames = this.registry.getDomainNames();
|
|
2372
|
+
if (domainNames.length === 0) return null;
|
|
2373
|
+
const candidates = [];
|
|
2374
|
+
for (const name of domainNames) {
|
|
2375
|
+
const dsl = await this.registry.getDSLForDomain(name);
|
|
2376
|
+
if (!dsl) continue;
|
|
2377
|
+
try {
|
|
2378
|
+
const { node, confidence } = dsl.parseWithConfidence(input, language);
|
|
2379
|
+
if (confidence >= this.minConfidence) {
|
|
2380
|
+
candidates.push({ domain: name, node, confidence, dsl });
|
|
2381
|
+
}
|
|
2382
|
+
} catch {
|
|
2383
|
+
}
|
|
2384
|
+
}
|
|
2385
|
+
if (candidates.length === 0) return null;
|
|
2386
|
+
candidates.sort((a, b) => {
|
|
2387
|
+
if (b.confidence !== a.confidence) return b.confidence - a.confidence;
|
|
2388
|
+
return this.getPriorityIndex(a.domain) - this.getPriorityIndex(b.domain);
|
|
2389
|
+
});
|
|
2390
|
+
return candidates[0];
|
|
2391
|
+
}
|
|
2392
|
+
/**
|
|
2393
|
+
* Parse multi-line input where each line may belong to a different domain.
|
|
2394
|
+
* Empty lines and comment lines (starting with //, --, or #) are skipped.
|
|
2395
|
+
*/
|
|
2396
|
+
async parseComposite(input, language = "en") {
|
|
2397
|
+
const lines = input.split("\n");
|
|
2398
|
+
const statements = [];
|
|
2399
|
+
const errors = [];
|
|
2400
|
+
for (let i = 0; i < lines.length; i++) {
|
|
2401
|
+
const line = lines[i];
|
|
2402
|
+
if (isSkippableLine(line)) continue;
|
|
2403
|
+
const trimmed = line.trim();
|
|
2404
|
+
const result = await this.detect(trimmed, language);
|
|
2405
|
+
if (result) {
|
|
2406
|
+
statements.push({
|
|
2407
|
+
line: i + 1,
|
|
2408
|
+
input: trimmed,
|
|
2409
|
+
domain: result.domain,
|
|
2410
|
+
node: result.node,
|
|
2411
|
+
confidence: result.confidence
|
|
2412
|
+
});
|
|
2413
|
+
} else {
|
|
2414
|
+
errors.push({
|
|
2415
|
+
line: i + 1,
|
|
2416
|
+
input: trimmed,
|
|
2417
|
+
message: "No domain matched this input"
|
|
2418
|
+
});
|
|
2419
|
+
}
|
|
2420
|
+
}
|
|
2421
|
+
return { statements, errors };
|
|
2422
|
+
}
|
|
2423
|
+
/**
|
|
2424
|
+
* Compile input using auto-detected domain.
|
|
2425
|
+
* Returns the compilation result with the domain name.
|
|
2426
|
+
*/
|
|
2427
|
+
async compile(input, language = "en") {
|
|
2428
|
+
const detected = await this.detect(input, language);
|
|
2429
|
+
if (!detected) return null;
|
|
2430
|
+
const result = detected.dsl.compile(input, language);
|
|
2431
|
+
return { ...result, domain: detected.domain };
|
|
2432
|
+
}
|
|
2433
|
+
/**
|
|
2434
|
+
* Validate input against all domains, returning the best match.
|
|
2435
|
+
*/
|
|
2436
|
+
async validate(input, language = "en") {
|
|
2437
|
+
const detected = await this.detect(input, language);
|
|
2438
|
+
if (!detected) {
|
|
2439
|
+
return { valid: false, errors: ["No domain matched this input"] };
|
|
2440
|
+
}
|
|
2441
|
+
const result = detected.dsl.validate(input, language);
|
|
2442
|
+
return { ...result, domain: detected.domain };
|
|
2443
|
+
}
|
|
2444
|
+
// ---------------------------------------------------------------------------
|
|
2445
|
+
// Private
|
|
2446
|
+
// ---------------------------------------------------------------------------
|
|
2447
|
+
getPriorityIndex(domain) {
|
|
2448
|
+
const idx = this.priority.indexOf(domain);
|
|
2449
|
+
return idx === -1 ? this.priority.length : idx;
|
|
2450
|
+
}
|
|
2451
|
+
};
|
|
2452
|
+
|
|
2453
|
+
// src/aot/domain-scanner.ts
|
|
2454
|
+
function buildAttributePatterns(attributes) {
|
|
2455
|
+
const patterns = [];
|
|
2456
|
+
for (const attr of attributes) {
|
|
2457
|
+
const escaped = attr.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
2458
|
+
patterns.push(new RegExp(`${escaped}\\s*=\\s*"([^"]+)"`, "g"));
|
|
2459
|
+
patterns.push(new RegExp(`${escaped}\\s*=\\s*'([^']+)'`, "g"));
|
|
2460
|
+
patterns.push(new RegExp(`${escaped}\\s*=\\s*\`([^\`]+)\``, "g"));
|
|
2461
|
+
}
|
|
2462
|
+
return patterns;
|
|
2463
|
+
}
|
|
2464
|
+
function buildScriptPatterns(scriptTypes) {
|
|
2465
|
+
return scriptTypes.map((type) => {
|
|
2466
|
+
const escaped = type.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
2467
|
+
return new RegExp(`<script[^>]*type=["']?${escaped}["']?[^>]*>([\\s\\S]*?)<\\/script>`, "gi");
|
|
2468
|
+
});
|
|
2469
|
+
}
|
|
2470
|
+
function getLineNumber(source, position) {
|
|
2471
|
+
let line = 1;
|
|
2472
|
+
for (let i = 0; i < position && i < source.length; i++) {
|
|
2473
|
+
if (source[i] === "\n") line++;
|
|
2474
|
+
}
|
|
2475
|
+
return line;
|
|
2476
|
+
}
|
|
2477
|
+
function extractElementId(source, matchIndex) {
|
|
2478
|
+
const before = source.lastIndexOf("<", matchIndex);
|
|
2479
|
+
if (before === -1) return void 0;
|
|
2480
|
+
const tagChunk = source.slice(before, matchIndex + 200);
|
|
2481
|
+
const idMatch = tagChunk.match(/\bid=["']([^"']+)["']/);
|
|
2482
|
+
return idMatch?.[1];
|
|
2483
|
+
}
|
|
2484
|
+
function extractLangAttribute(source, matchIndex) {
|
|
2485
|
+
const before = source.lastIndexOf("<", matchIndex);
|
|
2486
|
+
if (before === -1) return void 0;
|
|
2487
|
+
const tagChunk = source.slice(before, matchIndex + 200);
|
|
2488
|
+
const langMatch = tagChunk.match(/\blang=["']([^"']+)["']/);
|
|
2489
|
+
return langMatch?.[1];
|
|
2490
|
+
}
|
|
2491
|
+
var DomainAwareScanner = class {
|
|
2492
|
+
constructor(configs) {
|
|
2493
|
+
this.configs = configs;
|
|
2494
|
+
}
|
|
2495
|
+
/**
|
|
2496
|
+
* Extract all domain snippets from an HTML source string.
|
|
2497
|
+
*/
|
|
2498
|
+
extract(source, filename) {
|
|
2499
|
+
const snippets = [];
|
|
2500
|
+
for (const config of this.configs) {
|
|
2501
|
+
const defaultLang = config.defaultLanguage ?? "en";
|
|
2502
|
+
const attrPatterns = buildAttributePatterns(config.attributes);
|
|
2503
|
+
for (const pattern of attrPatterns) {
|
|
2504
|
+
pattern.lastIndex = 0;
|
|
2505
|
+
let match;
|
|
2506
|
+
while (match = pattern.exec(source)) {
|
|
2507
|
+
const code = match[1].trim();
|
|
2508
|
+
if (!code) continue;
|
|
2509
|
+
const line = getLineNumber(source, match.index);
|
|
2510
|
+
const elementId = extractElementId(source, match.index);
|
|
2511
|
+
const langAttr = extractLangAttribute(source, match.index);
|
|
2512
|
+
snippets.push({
|
|
2513
|
+
domain: config.domain,
|
|
2514
|
+
code,
|
|
2515
|
+
language: langAttr ?? defaultLang,
|
|
2516
|
+
file: filename,
|
|
2517
|
+
line,
|
|
2518
|
+
column: 1,
|
|
2519
|
+
...elementId != null && { elementId }
|
|
2520
|
+
});
|
|
2521
|
+
}
|
|
2522
|
+
}
|
|
2523
|
+
if (config.scriptTypes) {
|
|
2524
|
+
const scriptPatterns = buildScriptPatterns(config.scriptTypes);
|
|
2525
|
+
for (const pattern of scriptPatterns) {
|
|
2526
|
+
pattern.lastIndex = 0;
|
|
2527
|
+
let match;
|
|
2528
|
+
while (match = pattern.exec(source)) {
|
|
2529
|
+
const code = match[1].trim();
|
|
2530
|
+
if (!code) continue;
|
|
2531
|
+
const line = getLineNumber(source, match.index);
|
|
2532
|
+
snippets.push({
|
|
2533
|
+
domain: config.domain,
|
|
2534
|
+
code,
|
|
2535
|
+
language: defaultLang,
|
|
2536
|
+
file: filename,
|
|
2537
|
+
line,
|
|
2538
|
+
column: 1
|
|
2539
|
+
});
|
|
2540
|
+
}
|
|
2541
|
+
}
|
|
2542
|
+
}
|
|
2543
|
+
}
|
|
2544
|
+
return snippets;
|
|
2545
|
+
}
|
|
2546
|
+
/**
|
|
2547
|
+
* Extract snippets from multiple files.
|
|
2548
|
+
*/
|
|
2549
|
+
async extractFromFiles(files, readFile) {
|
|
2550
|
+
const allSnippets = [];
|
|
2551
|
+
for (const file of files) {
|
|
2552
|
+
try {
|
|
2553
|
+
const source = await readFile(file);
|
|
2554
|
+
const snippets = this.extract(source, file);
|
|
2555
|
+
allSnippets.push(...snippets);
|
|
2556
|
+
} catch {
|
|
2557
|
+
}
|
|
2558
|
+
}
|
|
2559
|
+
return allSnippets;
|
|
2560
|
+
}
|
|
2561
|
+
};
|
|
2562
|
+
|
|
2563
|
+
// src/aot/aot-orchestrator.ts
|
|
2564
|
+
var AOTOrchestrator = class {
|
|
2565
|
+
constructor(options) {
|
|
2566
|
+
this.backends = /* @__PURE__ */ new Map();
|
|
2567
|
+
this.options = {
|
|
2568
|
+
confidenceThreshold: options?.confidenceThreshold ?? 0.7,
|
|
2569
|
+
continueOnError: options?.continueOnError ?? true,
|
|
2570
|
+
debug: options?.debug ?? false
|
|
2571
|
+
};
|
|
2572
|
+
}
|
|
2573
|
+
/**
|
|
2574
|
+
* Register a domain compilation backend.
|
|
2575
|
+
* @throws if a backend with the same domain name is already registered
|
|
2576
|
+
*/
|
|
2577
|
+
registerBackend(backend) {
|
|
2578
|
+
if (this.backends.has(backend.domain)) {
|
|
2579
|
+
throw new Error(`Backend already registered for domain: ${backend.domain}`);
|
|
2580
|
+
}
|
|
2581
|
+
this.backends.set(backend.domain, backend);
|
|
2582
|
+
}
|
|
2583
|
+
/**
|
|
2584
|
+
* Scan source files and compile all domain snippets.
|
|
2585
|
+
*/
|
|
2586
|
+
async compileFiles(files, readFile) {
|
|
2587
|
+
const configs = Array.from(this.backends.values()).map((b) => b.scanConfig);
|
|
2588
|
+
const scanner = new DomainAwareScanner(configs);
|
|
2589
|
+
const snippets = await scanner.extractFromFiles(files, readFile);
|
|
2590
|
+
if (this.options.debug) {
|
|
2591
|
+
console.log(`[aot] Extracted ${snippets.length} snippets from ${files.length} files`);
|
|
2592
|
+
}
|
|
2593
|
+
const compiled = [];
|
|
2594
|
+
const errors = [];
|
|
2595
|
+
const domainBreakdown = {};
|
|
2596
|
+
for (const snippet of snippets) {
|
|
2597
|
+
domainBreakdown[snippet.domain] = (domainBreakdown[snippet.domain] ?? 0) + 1;
|
|
2598
|
+
const result = this.compileSnippet(snippet);
|
|
2599
|
+
if ("error" in result) {
|
|
2600
|
+
errors.push({
|
|
2601
|
+
domain: snippet.domain,
|
|
2602
|
+
source: snippet.code,
|
|
2603
|
+
file: snippet.file,
|
|
2604
|
+
line: snippet.line,
|
|
2605
|
+
message: result.error
|
|
2606
|
+
});
|
|
2607
|
+
if (!this.options.continueOnError) break;
|
|
2608
|
+
} else {
|
|
2609
|
+
compiled.push(result);
|
|
2610
|
+
}
|
|
2611
|
+
}
|
|
2612
|
+
return {
|
|
2613
|
+
compiled,
|
|
2614
|
+
errors,
|
|
2615
|
+
stats: {
|
|
2616
|
+
totalSnippets: snippets.length,
|
|
2617
|
+
compiledCount: compiled.length,
|
|
2618
|
+
errorCount: errors.length,
|
|
2619
|
+
domainBreakdown
|
|
2620
|
+
}
|
|
2621
|
+
};
|
|
2622
|
+
}
|
|
2623
|
+
/**
|
|
2624
|
+
* Compile a single extracted snippet.
|
|
2625
|
+
* Returns the compiled result or an error object.
|
|
2626
|
+
*/
|
|
2627
|
+
compileSnippet(snippet) {
|
|
2628
|
+
const backend = this.backends.get(snippet.domain);
|
|
2629
|
+
if (!backend) {
|
|
2630
|
+
return { error: `No backend registered for domain: ${snippet.domain}` };
|
|
2631
|
+
}
|
|
2632
|
+
try {
|
|
2633
|
+
const result = backend.dsl.compile(snippet.code, snippet.language);
|
|
2634
|
+
if (!result.ok || !result.code) {
|
|
2635
|
+
return { error: result.errors?.join("; ") ?? "Compilation failed" };
|
|
2636
|
+
}
|
|
2637
|
+
return {
|
|
2638
|
+
domain: snippet.domain,
|
|
2639
|
+
source: snippet.code,
|
|
2640
|
+
compiled: result.code,
|
|
2641
|
+
language: snippet.language,
|
|
2642
|
+
file: snippet.file,
|
|
2643
|
+
line: snippet.line
|
|
2644
|
+
};
|
|
2645
|
+
} catch (err) {
|
|
2646
|
+
return { error: err instanceof Error ? err.message : String(err) };
|
|
2647
|
+
}
|
|
2648
|
+
}
|
|
2649
|
+
/**
|
|
2650
|
+
* Generate combined output from compilation results.
|
|
2651
|
+
*/
|
|
2652
|
+
generateOutput(result, options) {
|
|
2653
|
+
const format = options?.format ?? "esm";
|
|
2654
|
+
const includeComments = options?.includeComments ?? true;
|
|
2655
|
+
const groupByDomain = options?.groupByDomain ?? true;
|
|
2656
|
+
const sections = [];
|
|
2657
|
+
if (includeComments) {
|
|
2658
|
+
sections.push(`// Generated by @lokascript/framework AOT compiler`);
|
|
2659
|
+
sections.push(
|
|
2660
|
+
`// ${result.stats.compiledCount} snippets compiled, ${result.stats.errorCount} errors`
|
|
2661
|
+
);
|
|
2662
|
+
sections.push("");
|
|
2663
|
+
}
|
|
2664
|
+
if (groupByDomain) {
|
|
2665
|
+
const grouped = /* @__PURE__ */ new Map();
|
|
2666
|
+
for (const snippet of result.compiled) {
|
|
2667
|
+
const list = grouped.get(snippet.domain) ?? [];
|
|
2668
|
+
list.push(snippet);
|
|
2669
|
+
grouped.set(snippet.domain, list);
|
|
2670
|
+
}
|
|
2671
|
+
for (const [domain, snippets] of grouped) {
|
|
2672
|
+
if (includeComments) {
|
|
2673
|
+
sections.push(`// === Domain: ${domain} ===`);
|
|
2674
|
+
sections.push("");
|
|
2675
|
+
}
|
|
2676
|
+
for (const snippet of snippets) {
|
|
2677
|
+
if (includeComments) {
|
|
2678
|
+
sections.push(`// Source: ${snippet.file}:${snippet.line}`);
|
|
2679
|
+
sections.push(`// Language: ${snippet.language}`);
|
|
2680
|
+
}
|
|
2681
|
+
sections.push(snippet.compiled);
|
|
2682
|
+
sections.push("");
|
|
2683
|
+
}
|
|
2684
|
+
}
|
|
2685
|
+
} else {
|
|
2686
|
+
for (const snippet of result.compiled) {
|
|
2687
|
+
if (includeComments) {
|
|
2688
|
+
sections.push(`// [${snippet.domain}] ${snippet.file}:${snippet.line}`);
|
|
2689
|
+
}
|
|
2690
|
+
sections.push(snippet.compiled);
|
|
2691
|
+
sections.push("");
|
|
2692
|
+
}
|
|
2693
|
+
}
|
|
2694
|
+
const body = sections.join("\n").trimEnd();
|
|
2695
|
+
switch (format) {
|
|
2696
|
+
case "cjs":
|
|
2697
|
+
return `'use strict';
|
|
2698
|
+
|
|
2699
|
+
${body}
|
|
2700
|
+
`;
|
|
2701
|
+
case "iife":
|
|
2702
|
+
return `(function() {
|
|
2703
|
+
${body}
|
|
2704
|
+
})();
|
|
2705
|
+
`;
|
|
2706
|
+
case "esm":
|
|
2707
|
+
default:
|
|
2708
|
+
return `${body}
|
|
2709
|
+
`;
|
|
2710
|
+
}
|
|
2711
|
+
}
|
|
2712
|
+
};
|
|
2713
|
+
|
|
2714
|
+
// src/schema/command-schema.ts
|
|
2715
|
+
function defineCommand(schema) {
|
|
2716
|
+
return {
|
|
2717
|
+
description: schema.description || `${schema.action} command`,
|
|
2718
|
+
category: schema.category || "general",
|
|
2719
|
+
primaryRole: schema.primaryRole || schema.roles[0]?.role || "patient",
|
|
2720
|
+
...schema,
|
|
2721
|
+
action: schema.action,
|
|
2722
|
+
roles: schema.roles
|
|
2723
|
+
};
|
|
2724
|
+
}
|
|
2725
|
+
function defineRole(role) {
|
|
2726
|
+
return {
|
|
2727
|
+
description: role.description || `${role.role} role`,
|
|
2728
|
+
...role,
|
|
2729
|
+
role: role.role,
|
|
2730
|
+
required: role.required,
|
|
2731
|
+
expectedTypes: role.expectedTypes
|
|
2732
|
+
};
|
|
2733
|
+
}
|
|
2734
|
+
|
|
2735
|
+
// src/generation/renderer.ts
|
|
2736
|
+
function lookupKeyword(keywords, action, language) {
|
|
2737
|
+
return keywords[action]?.[language] ?? action;
|
|
2738
|
+
}
|
|
2739
|
+
function lookupMarker(markers, marker, language) {
|
|
2740
|
+
return markers[marker]?.[language] ?? marker;
|
|
2741
|
+
}
|
|
2742
|
+
function buildPhrase(...parts) {
|
|
2743
|
+
return parts.filter(Boolean).join(" ");
|
|
2744
|
+
}
|
|
2745
|
+
function buildTablesFromProfiles(schemas, profiles) {
|
|
2746
|
+
const keywords = {};
|
|
2747
|
+
const markers = {};
|
|
2748
|
+
for (const profile of profiles) {
|
|
2749
|
+
for (const [action, kw] of Object.entries(profile.keywords)) {
|
|
2750
|
+
if (!keywords[action]) keywords[action] = {};
|
|
2751
|
+
keywords[action][profile.code] = kw.primary;
|
|
2752
|
+
}
|
|
2753
|
+
}
|
|
2754
|
+
for (const schema of schemas) {
|
|
2755
|
+
for (const role of schema.roles) {
|
|
2756
|
+
if (role.markerOverride) {
|
|
2757
|
+
const markerKey = role.role;
|
|
2758
|
+
if (!markers[markerKey]) markers[markerKey] = {};
|
|
2759
|
+
for (const [lang, markerText] of Object.entries(role.markerOverride)) {
|
|
2760
|
+
markers[markerKey][lang] = markerText;
|
|
2761
|
+
}
|
|
2762
|
+
}
|
|
2763
|
+
}
|
|
2764
|
+
}
|
|
2765
|
+
for (const profile of profiles) {
|
|
2766
|
+
if (profile.roleMarkers) {
|
|
2767
|
+
for (const [markerKey, markerDef] of Object.entries(profile.roleMarkers)) {
|
|
2768
|
+
if (!markers[markerKey]) markers[markerKey] = {};
|
|
2769
|
+
markers[markerKey][profile.code] = markerDef.primary;
|
|
2770
|
+
}
|
|
2771
|
+
}
|
|
2772
|
+
}
|
|
2773
|
+
return { keywords, markers };
|
|
2774
|
+
}
|
|
2775
|
+
function detectWordOrders(profiles) {
|
|
2776
|
+
const sovLanguages = /* @__PURE__ */ new Set();
|
|
2777
|
+
const vsoLanguages = /* @__PURE__ */ new Set();
|
|
2778
|
+
for (const profile of profiles) {
|
|
2779
|
+
if (profile.wordOrder === "SOV") sovLanguages.add(profile.code);
|
|
2780
|
+
if (profile.wordOrder === "VSO") vsoLanguages.add(profile.code);
|
|
2781
|
+
}
|
|
2782
|
+
return { sovLanguages, vsoLanguages };
|
|
2783
|
+
}
|
|
2784
|
+
function createSchemaRenderer(schemas, profiles) {
|
|
2785
|
+
const { keywords, markers } = buildTablesFromProfiles(schemas, profiles);
|
|
2786
|
+
const { sovLanguages } = detectWordOrders(profiles);
|
|
2787
|
+
const schemaMap = /* @__PURE__ */ new Map();
|
|
2788
|
+
for (const s of schemas) schemaMap.set(s.action, s);
|
|
2789
|
+
return {
|
|
2790
|
+
render(node, language) {
|
|
2791
|
+
const schema = schemaMap.get(node.action);
|
|
2792
|
+
if (!schema) return node.action;
|
|
2793
|
+
const keyword = lookupKeyword(keywords, node.action, language);
|
|
2794
|
+
const isSOV = sovLanguages.has(language);
|
|
2795
|
+
const roleParts = [];
|
|
2796
|
+
for (const role of schema.roles) {
|
|
2797
|
+
const value = extractRoleValue(node, role.role);
|
|
2798
|
+
if (!value && !role.required) continue;
|
|
2799
|
+
const markerText = role.markerOverride?.[language] ?? markers[role.role]?.[language] ?? void 0;
|
|
2800
|
+
roleParts.push({
|
|
2801
|
+
...markerText != null && { marker: markerText },
|
|
2802
|
+
value: value || "",
|
|
2803
|
+
role
|
|
2804
|
+
});
|
|
2805
|
+
}
|
|
2806
|
+
const parts = [];
|
|
2807
|
+
if (isSOV) {
|
|
2808
|
+
for (const rp of roleParts) {
|
|
2809
|
+
if (rp.value) parts.push(rp.value);
|
|
2810
|
+
if (rp.marker) parts.push(rp.marker);
|
|
2811
|
+
}
|
|
2812
|
+
parts.push(keyword);
|
|
2813
|
+
} else {
|
|
2814
|
+
parts.push(keyword);
|
|
2815
|
+
for (const rp of roleParts) {
|
|
2816
|
+
if (rp.marker) parts.push(rp.marker);
|
|
2817
|
+
if (rp.value) parts.push(rp.value);
|
|
2818
|
+
}
|
|
2819
|
+
}
|
|
2820
|
+
return buildPhrase(...parts);
|
|
2821
|
+
}
|
|
2822
|
+
};
|
|
2823
|
+
}
|
|
2824
|
+
|
|
2825
|
+
// src/generation/diagnostics.ts
|
|
2826
|
+
function createDiagnosticCollector() {
|
|
2827
|
+
const diagnostics = [];
|
|
2828
|
+
function addWithSeverity(severity, message, options) {
|
|
2829
|
+
const diag = {
|
|
2830
|
+
message,
|
|
2831
|
+
severity,
|
|
2832
|
+
...options?.code != null && { code: options.code },
|
|
2833
|
+
...options?.line != null && { line: options.line },
|
|
2834
|
+
...options?.column != null && { column: options.column },
|
|
2835
|
+
...options?.source != null && { source: options.source },
|
|
2836
|
+
...options?.suggestions != null && options.suggestions.length > 0 && { suggestions: options.suggestions }
|
|
2837
|
+
};
|
|
2838
|
+
diagnostics.push(diag);
|
|
2839
|
+
}
|
|
2840
|
+
return {
|
|
2841
|
+
error(message, options) {
|
|
2842
|
+
addWithSeverity("error", message, options);
|
|
2843
|
+
},
|
|
2844
|
+
warning(message, options) {
|
|
2845
|
+
addWithSeverity("warning", message, options);
|
|
2846
|
+
},
|
|
2847
|
+
info(message, options) {
|
|
2848
|
+
addWithSeverity("info", message, options);
|
|
2849
|
+
},
|
|
2850
|
+
add(diagnostic) {
|
|
2851
|
+
diagnostics.push(diagnostic);
|
|
2852
|
+
},
|
|
2853
|
+
hasErrors() {
|
|
2854
|
+
return diagnostics.some((d) => d.severity === "error");
|
|
2855
|
+
},
|
|
2856
|
+
getDiagnostics() {
|
|
2857
|
+
return diagnostics;
|
|
2858
|
+
},
|
|
2859
|
+
toResult() {
|
|
2860
|
+
let errors = 0;
|
|
2861
|
+
let warnings = 0;
|
|
2862
|
+
let infos = 0;
|
|
2863
|
+
for (const d of diagnostics) {
|
|
2864
|
+
if (d.severity === "error") errors++;
|
|
2865
|
+
else if (d.severity === "warning") warnings++;
|
|
2866
|
+
else infos++;
|
|
2867
|
+
}
|
|
2868
|
+
return {
|
|
2869
|
+
ok: errors === 0,
|
|
2870
|
+
diagnostics,
|
|
2871
|
+
summary: { errors, warnings, infos }
|
|
2872
|
+
};
|
|
2873
|
+
}
|
|
2874
|
+
};
|
|
2875
|
+
}
|
|
2876
|
+
function fromError(error, options) {
|
|
2877
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
2878
|
+
return {
|
|
2879
|
+
message,
|
|
2880
|
+
severity: "error",
|
|
2881
|
+
...options?.code != null && { code: options.code },
|
|
2882
|
+
...options?.line != null && { line: options.line },
|
|
2883
|
+
...options?.column != null && { column: options.column },
|
|
2884
|
+
...options?.source != null && { source: options.source },
|
|
2885
|
+
...options?.suggestions != null && options.suggestions.length > 0 && { suggestions: options.suggestions }
|
|
2886
|
+
};
|
|
2887
|
+
}
|
|
2888
|
+
function filterBySeverity(diagnostics, severity) {
|
|
2889
|
+
return diagnostics.filter((d) => d.severity === severity);
|
|
2890
|
+
}
|
|
2891
|
+
|
|
2892
|
+
// src/core/tokenization/token-utils.ts
|
|
2893
|
+
var TokenStreamImpl = class {
|
|
2894
|
+
constructor(tokens, language) {
|
|
2895
|
+
this.pos = 0;
|
|
2896
|
+
this.tokens = tokens;
|
|
2897
|
+
this.language = language;
|
|
2898
|
+
}
|
|
2899
|
+
peek(offset = 0) {
|
|
2900
|
+
const index = this.pos + offset;
|
|
2901
|
+
if (index < 0 || index >= this.tokens.length) {
|
|
2902
|
+
return null;
|
|
2903
|
+
}
|
|
2904
|
+
return this.tokens[index];
|
|
2905
|
+
}
|
|
2906
|
+
advance() {
|
|
2907
|
+
if (this.isAtEnd()) {
|
|
2908
|
+
throw new Error("Unexpected end of token stream");
|
|
2909
|
+
}
|
|
2910
|
+
return this.tokens[this.pos++];
|
|
2911
|
+
}
|
|
2912
|
+
isAtEnd() {
|
|
2913
|
+
return this.pos >= this.tokens.length;
|
|
2914
|
+
}
|
|
2915
|
+
mark() {
|
|
2916
|
+
return { position: this.pos };
|
|
2917
|
+
}
|
|
2918
|
+
reset(mark) {
|
|
2919
|
+
this.pos = mark.position;
|
|
2920
|
+
}
|
|
2921
|
+
position() {
|
|
2922
|
+
return this.pos;
|
|
2923
|
+
}
|
|
2924
|
+
/**
|
|
2925
|
+
* Get remaining tokens as an array.
|
|
2926
|
+
*/
|
|
2927
|
+
remaining() {
|
|
2928
|
+
return this.tokens.slice(this.pos);
|
|
2929
|
+
}
|
|
2930
|
+
/**
|
|
2931
|
+
* Consume tokens while predicate is true.
|
|
2932
|
+
*/
|
|
2933
|
+
takeWhile(predicate) {
|
|
2934
|
+
const result = [];
|
|
2935
|
+
while (!this.isAtEnd() && predicate(this.peek())) {
|
|
2936
|
+
result.push(this.advance());
|
|
2937
|
+
}
|
|
2938
|
+
return result;
|
|
2939
|
+
}
|
|
2940
|
+
/**
|
|
2941
|
+
* Skip tokens while predicate is true.
|
|
2942
|
+
*/
|
|
2943
|
+
skipWhile(predicate) {
|
|
2944
|
+
while (!this.isAtEnd() && predicate(this.peek())) {
|
|
2945
|
+
this.advance();
|
|
2946
|
+
}
|
|
2947
|
+
}
|
|
2948
|
+
};
|
|
2949
|
+
function createPosition(start, end) {
|
|
2950
|
+
return { start, end };
|
|
2951
|
+
}
|
|
2952
|
+
function createToken(valueOrParams, kind, position, normalizedOrOptions) {
|
|
2953
|
+
if (typeof valueOrParams === "object") {
|
|
2954
|
+
const { value: value2, kind: kind2, position: position2, normalized: normalized2, stem, stemConfidence, metadata } = valueOrParams;
|
|
2955
|
+
return {
|
|
2956
|
+
value: value2,
|
|
2957
|
+
kind: kind2,
|
|
2958
|
+
position: position2,
|
|
2959
|
+
...normalized2 !== void 0 && { normalized: normalized2 },
|
|
2960
|
+
...stem !== void 0 && { stem },
|
|
2961
|
+
...stemConfidence !== void 0 && { stemConfidence },
|
|
2962
|
+
...metadata !== void 0 && { metadata }
|
|
2963
|
+
};
|
|
2964
|
+
}
|
|
2965
|
+
const value = valueOrParams;
|
|
2966
|
+
if (!kind || !position) {
|
|
2967
|
+
throw new Error("createToken requires kind and position parameters");
|
|
2968
|
+
}
|
|
2969
|
+
if (typeof normalizedOrOptions === "string") {
|
|
2970
|
+
return { value, kind, position, normalized: normalizedOrOptions };
|
|
2971
|
+
}
|
|
2972
|
+
if (normalizedOrOptions) {
|
|
2973
|
+
const { normalized: normalized2, stem, stemConfidence, metadata } = normalizedOrOptions;
|
|
2974
|
+
return {
|
|
2975
|
+
value,
|
|
2976
|
+
kind,
|
|
2977
|
+
position,
|
|
2978
|
+
...normalized2 !== void 0 && { normalized: normalized2 },
|
|
2979
|
+
...stem !== void 0 && { stem },
|
|
2980
|
+
...stemConfidence !== void 0 && { stemConfidence },
|
|
2981
|
+
...metadata !== void 0 && { metadata }
|
|
2982
|
+
};
|
|
2983
|
+
}
|
|
2984
|
+
return { value, kind, position };
|
|
2985
|
+
}
|
|
2986
|
+
function isWhitespace(char) {
|
|
2987
|
+
return /\s/.test(char);
|
|
2988
|
+
}
|
|
2989
|
+
function isSelectorStart(char) {
|
|
2990
|
+
return char === "#" || char === "." || char === "[" || char === "@" || char === "*" || char === "<";
|
|
2991
|
+
}
|
|
2992
|
+
function isQuote(char) {
|
|
2993
|
+
return char === '"' || char === "'" || char === "`" || char === "\u300C" || char === "\u300D";
|
|
2994
|
+
}
|
|
2995
|
+
function isDigit(char) {
|
|
2996
|
+
return /\d/.test(char);
|
|
2997
|
+
}
|
|
2998
|
+
function isAsciiLetter(char) {
|
|
2999
|
+
return /[a-zA-Z]/.test(char);
|
|
3000
|
+
}
|
|
3001
|
+
function isAsciiIdentifierChar(char) {
|
|
3002
|
+
return /[a-zA-Z0-9_-]/.test(char);
|
|
3003
|
+
}
|
|
3004
|
+
|
|
3005
|
+
// src/core/tokenization/extractors.ts
|
|
3006
|
+
function extractCssSelector(input, startPos) {
|
|
3007
|
+
if (startPos >= input.length) return null;
|
|
3008
|
+
const char = input[startPos];
|
|
3009
|
+
if (!isSelectorStart(char)) return null;
|
|
3010
|
+
let pos = startPos;
|
|
3011
|
+
let selector = "";
|
|
3012
|
+
if (char === "#" || char === ".") {
|
|
3013
|
+
selector += input[pos++];
|
|
3014
|
+
while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
|
|
3015
|
+
selector += input[pos++];
|
|
3016
|
+
}
|
|
3017
|
+
if (selector.length <= 1) return null;
|
|
3018
|
+
if (pos < input.length && input[pos] === "." && char === "#") {
|
|
3019
|
+
const methodStart = pos + 1;
|
|
3020
|
+
let methodEnd = methodStart;
|
|
3021
|
+
while (methodEnd < input.length && isAsciiIdentifierChar(input[methodEnd])) {
|
|
3022
|
+
methodEnd++;
|
|
3023
|
+
}
|
|
3024
|
+
if (methodEnd < input.length && input[methodEnd] === "(") {
|
|
3025
|
+
return selector;
|
|
3026
|
+
}
|
|
3027
|
+
}
|
|
3028
|
+
} else if (char === "[") {
|
|
3029
|
+
let depth = 1;
|
|
3030
|
+
let inQuote = false;
|
|
3031
|
+
let quoteChar = null;
|
|
3032
|
+
let escaped = false;
|
|
3033
|
+
selector += input[pos++];
|
|
3034
|
+
while (pos < input.length && depth > 0) {
|
|
3035
|
+
const c = input[pos];
|
|
3036
|
+
selector += c;
|
|
3037
|
+
if (escaped) {
|
|
3038
|
+
escaped = false;
|
|
3039
|
+
} else if (c === "\\") {
|
|
3040
|
+
escaped = true;
|
|
3041
|
+
} else if (inQuote) {
|
|
3042
|
+
if (c === quoteChar) {
|
|
3043
|
+
inQuote = false;
|
|
3044
|
+
quoteChar = null;
|
|
3045
|
+
}
|
|
3046
|
+
} else {
|
|
3047
|
+
if (c === '"' || c === "'" || c === "`") {
|
|
3048
|
+
inQuote = true;
|
|
3049
|
+
quoteChar = c;
|
|
3050
|
+
} else if (c === "[") {
|
|
3051
|
+
depth++;
|
|
3052
|
+
} else if (c === "]") {
|
|
3053
|
+
depth--;
|
|
3054
|
+
}
|
|
3055
|
+
}
|
|
3056
|
+
pos++;
|
|
3057
|
+
}
|
|
3058
|
+
if (depth !== 0) return null;
|
|
3059
|
+
} else if (char === "@") {
|
|
3060
|
+
selector += input[pos++];
|
|
3061
|
+
while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
|
|
3062
|
+
selector += input[pos++];
|
|
3063
|
+
}
|
|
3064
|
+
if (selector.length <= 1) return null;
|
|
3065
|
+
} else if (char === "*") {
|
|
3066
|
+
selector += input[pos++];
|
|
3067
|
+
while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
|
|
3068
|
+
selector += input[pos++];
|
|
3069
|
+
}
|
|
3070
|
+
if (selector.length <= 1) return null;
|
|
3071
|
+
} else if (char === "<") {
|
|
3072
|
+
selector += input[pos++];
|
|
3073
|
+
if (pos >= input.length || !isAsciiLetter(input[pos])) return null;
|
|
3074
|
+
while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
|
|
3075
|
+
selector += input[pos++];
|
|
3076
|
+
}
|
|
3077
|
+
while (pos < input.length) {
|
|
3078
|
+
const modChar = input[pos];
|
|
3079
|
+
if (modChar === ".") {
|
|
3080
|
+
selector += input[pos++];
|
|
3081
|
+
if (pos >= input.length || !isAsciiIdentifierChar(input[pos])) {
|
|
3082
|
+
return null;
|
|
3083
|
+
}
|
|
3084
|
+
while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
|
|
3085
|
+
selector += input[pos++];
|
|
3086
|
+
}
|
|
3087
|
+
} else if (modChar === "#") {
|
|
3088
|
+
selector += input[pos++];
|
|
3089
|
+
if (pos >= input.length || !isAsciiIdentifierChar(input[pos])) {
|
|
3090
|
+
return null;
|
|
3091
|
+
}
|
|
3092
|
+
while (pos < input.length && isAsciiIdentifierChar(input[pos])) {
|
|
3093
|
+
selector += input[pos++];
|
|
3094
|
+
}
|
|
3095
|
+
} else if (modChar === "[") {
|
|
3096
|
+
let depth = 1;
|
|
3097
|
+
let inQuote = false;
|
|
3098
|
+
let quoteChar = null;
|
|
3099
|
+
let escaped = false;
|
|
3100
|
+
selector += input[pos++];
|
|
3101
|
+
while (pos < input.length && depth > 0) {
|
|
3102
|
+
const c = input[pos];
|
|
3103
|
+
selector += c;
|
|
3104
|
+
if (escaped) {
|
|
3105
|
+
escaped = false;
|
|
3106
|
+
} else if (c === "\\") {
|
|
3107
|
+
escaped = true;
|
|
3108
|
+
} else if (inQuote) {
|
|
3109
|
+
if (c === quoteChar) {
|
|
3110
|
+
inQuote = false;
|
|
3111
|
+
quoteChar = null;
|
|
3112
|
+
}
|
|
3113
|
+
} else {
|
|
3114
|
+
if (c === '"' || c === "'" || c === "`") {
|
|
3115
|
+
inQuote = true;
|
|
3116
|
+
quoteChar = c;
|
|
3117
|
+
} else if (c === "[") {
|
|
3118
|
+
depth++;
|
|
3119
|
+
} else if (c === "]") {
|
|
3120
|
+
depth--;
|
|
3121
|
+
}
|
|
3122
|
+
}
|
|
3123
|
+
pos++;
|
|
3124
|
+
}
|
|
3125
|
+
if (depth !== 0) return null;
|
|
3126
|
+
} else {
|
|
3127
|
+
break;
|
|
3128
|
+
}
|
|
3129
|
+
}
|
|
3130
|
+
while (pos < input.length && isWhitespace(input[pos])) {
|
|
3131
|
+
selector += input[pos++];
|
|
3132
|
+
}
|
|
3133
|
+
if (pos < input.length && input[pos] === "/") {
|
|
3134
|
+
selector += input[pos++];
|
|
3135
|
+
while (pos < input.length && isWhitespace(input[pos])) {
|
|
3136
|
+
selector += input[pos++];
|
|
3137
|
+
}
|
|
3138
|
+
}
|
|
3139
|
+
if (pos >= input.length || input[pos] !== ">") return null;
|
|
3140
|
+
selector += input[pos++];
|
|
3141
|
+
}
|
|
3142
|
+
return selector || null;
|
|
3143
|
+
}
|
|
3144
|
+
function isPossessiveMarker(input, pos) {
|
|
3145
|
+
if (pos >= input.length || input[pos] !== "'") return false;
|
|
3146
|
+
if (pos + 1 >= input.length) return false;
|
|
3147
|
+
const nextChar = input[pos + 1].toLowerCase();
|
|
3148
|
+
if (nextChar !== "s") return false;
|
|
3149
|
+
if (pos + 2 >= input.length) return true;
|
|
3150
|
+
const afterS = input[pos + 2];
|
|
3151
|
+
return isWhitespace(afterS) || afterS === "*" || !isAsciiIdentifierChar(afterS);
|
|
3152
|
+
}
|
|
3153
|
+
function extractStringLiteral(input, startPos) {
|
|
3154
|
+
if (startPos >= input.length) return null;
|
|
3155
|
+
const openQuote = input[startPos];
|
|
3156
|
+
if (!isQuote(openQuote)) return null;
|
|
3157
|
+
if (openQuote === "'" && isPossessiveMarker(input, startPos)) {
|
|
3158
|
+
return null;
|
|
3159
|
+
}
|
|
3160
|
+
const closeQuoteMap = {
|
|
3161
|
+
'"': '"',
|
|
3162
|
+
"'": "'",
|
|
3163
|
+
"`": "`",
|
|
3164
|
+
"\u300C": "\u300D"
|
|
3165
|
+
};
|
|
3166
|
+
const closeQuote = closeQuoteMap[openQuote];
|
|
3167
|
+
if (!closeQuote) return null;
|
|
3168
|
+
let pos = startPos + 1;
|
|
3169
|
+
let literal = openQuote;
|
|
3170
|
+
let escaped = false;
|
|
3171
|
+
while (pos < input.length) {
|
|
3172
|
+
const char = input[pos];
|
|
3173
|
+
literal += char;
|
|
3174
|
+
if (escaped) {
|
|
3175
|
+
escaped = false;
|
|
3176
|
+
} else if (char === "\\") {
|
|
3177
|
+
escaped = true;
|
|
3178
|
+
} else if (char === closeQuote) {
|
|
3179
|
+
return literal;
|
|
3180
|
+
}
|
|
3181
|
+
pos++;
|
|
3182
|
+
}
|
|
3183
|
+
return literal;
|
|
3184
|
+
}
|
|
3185
|
+
function isUrlStart(input, pos) {
|
|
3186
|
+
if (pos >= input.length) return false;
|
|
3187
|
+
const char = input[pos];
|
|
3188
|
+
const next = input[pos + 1] || "";
|
|
3189
|
+
const third = input[pos + 2] || "";
|
|
3190
|
+
if (char === "/" && next !== "/" && /[a-zA-Z0-9._-]/.test(next)) {
|
|
3191
|
+
return true;
|
|
3192
|
+
}
|
|
3193
|
+
if (char === "/" && next === "/" && /[a-zA-Z]/.test(third)) {
|
|
3194
|
+
return true;
|
|
3195
|
+
}
|
|
3196
|
+
if (char === "." && (next === "/" || next === "." && third === "/")) {
|
|
3197
|
+
return true;
|
|
3198
|
+
}
|
|
3199
|
+
const slice = input.slice(pos, pos + 8).toLowerCase();
|
|
3200
|
+
if (slice.startsWith("http://") || slice.startsWith("https://")) {
|
|
3201
|
+
return true;
|
|
3202
|
+
}
|
|
3203
|
+
return false;
|
|
3204
|
+
}
|
|
3205
|
+
function extractUrl(input, startPos) {
|
|
3206
|
+
if (!isUrlStart(input, startPos)) return null;
|
|
3207
|
+
let pos = startPos;
|
|
3208
|
+
let url = "";
|
|
3209
|
+
const urlChars = /[a-zA-Z0-9/:._\-?&=%@+~!$'()*,;[\]]/;
|
|
3210
|
+
while (pos < input.length) {
|
|
3211
|
+
const char = input[pos];
|
|
3212
|
+
if (char === "#") {
|
|
3213
|
+
if (url.length > 0 && /[a-zA-Z0-9/.]$/.test(url)) {
|
|
3214
|
+
url += char;
|
|
3215
|
+
pos++;
|
|
3216
|
+
while (pos < input.length && /[a-zA-Z0-9_-]/.test(input[pos])) {
|
|
3217
|
+
url += input[pos++];
|
|
3218
|
+
}
|
|
3219
|
+
}
|
|
3220
|
+
break;
|
|
3221
|
+
}
|
|
3222
|
+
if (urlChars.test(char)) {
|
|
3223
|
+
url += char;
|
|
3224
|
+
pos++;
|
|
3225
|
+
} else {
|
|
3226
|
+
break;
|
|
3227
|
+
}
|
|
3228
|
+
}
|
|
3229
|
+
if (url.length < 2) return null;
|
|
3230
|
+
return url;
|
|
3231
|
+
}
|
|
3232
|
+
function extractNumber(input, startPos) {
|
|
3233
|
+
if (startPos >= input.length) return null;
|
|
3234
|
+
const char = input[startPos];
|
|
3235
|
+
if (!isDigit(char) && char !== "-" && char !== "+") return null;
|
|
3236
|
+
let pos = startPos;
|
|
3237
|
+
let number = "";
|
|
3238
|
+
if (input[pos] === "-" || input[pos] === "+") {
|
|
3239
|
+
number += input[pos++];
|
|
3240
|
+
}
|
|
3241
|
+
if (pos >= input.length || !isDigit(input[pos])) {
|
|
3242
|
+
return null;
|
|
3243
|
+
}
|
|
3244
|
+
while (pos < input.length && isDigit(input[pos])) {
|
|
3245
|
+
number += input[pos++];
|
|
3246
|
+
}
|
|
3247
|
+
if (pos < input.length && input[pos] === ".") {
|
|
3248
|
+
number += input[pos++];
|
|
3249
|
+
while (pos < input.length && isDigit(input[pos])) {
|
|
3250
|
+
number += input[pos++];
|
|
3251
|
+
}
|
|
3252
|
+
}
|
|
3253
|
+
if (pos < input.length) {
|
|
3254
|
+
const suffix = input.slice(pos, pos + 2);
|
|
3255
|
+
if (suffix === "ms") {
|
|
3256
|
+
number += "ms";
|
|
3257
|
+
} else if (input[pos] === "s" || input[pos] === "m" || input[pos] === "h") {
|
|
3258
|
+
number += input[pos];
|
|
3259
|
+
}
|
|
3260
|
+
}
|
|
3261
|
+
return number;
|
|
3262
|
+
}
|
|
3263
|
+
|
|
3264
|
+
// src/core/tokenization/extractors/operator.ts
|
|
3265
|
+
var DEFAULT_OPERATORS = [
|
|
3266
|
+
// Three-character operators
|
|
3267
|
+
"===",
|
|
3268
|
+
"!==",
|
|
3269
|
+
"->",
|
|
3270
|
+
// Two-character operators
|
|
3271
|
+
"==",
|
|
3272
|
+
"!=",
|
|
3273
|
+
"<=",
|
|
3274
|
+
">=",
|
|
3275
|
+
"&&",
|
|
3276
|
+
"||",
|
|
3277
|
+
"**",
|
|
3278
|
+
"+=",
|
|
3279
|
+
"-=",
|
|
3280
|
+
"*=",
|
|
3281
|
+
"/=",
|
|
3282
|
+
// Single-character operators
|
|
3283
|
+
"+",
|
|
3284
|
+
"-",
|
|
3285
|
+
"*",
|
|
3286
|
+
"/",
|
|
3287
|
+
"=",
|
|
3288
|
+
">",
|
|
3289
|
+
"<",
|
|
3290
|
+
"!",
|
|
3291
|
+
"&",
|
|
3292
|
+
"|",
|
|
3293
|
+
"%",
|
|
3294
|
+
"^",
|
|
3295
|
+
"~"
|
|
3296
|
+
];
|
|
3297
|
+
var OperatorExtractor = class {
|
|
3298
|
+
constructor(operators = DEFAULT_OPERATORS) {
|
|
3299
|
+
this.operators = operators;
|
|
3300
|
+
this.name = "operator";
|
|
3301
|
+
this.operators = [...operators].sort((a, b) => b.length - a.length);
|
|
3302
|
+
}
|
|
3303
|
+
canExtract(input, position) {
|
|
3304
|
+
return this.operators.some((op) => input.startsWith(op, position));
|
|
3305
|
+
}
|
|
3306
|
+
extract(input, position) {
|
|
3307
|
+
for (const op of this.operators) {
|
|
3308
|
+
if (input.startsWith(op, position)) {
|
|
3309
|
+
return {
|
|
3310
|
+
value: op,
|
|
3311
|
+
length: op.length
|
|
3312
|
+
};
|
|
3313
|
+
}
|
|
3314
|
+
}
|
|
3315
|
+
return null;
|
|
3316
|
+
}
|
|
3317
|
+
};
|
|
3318
|
+
|
|
3319
|
+
// src/core/tokenization/extractors/punctuation.ts
|
|
3320
|
+
var DEFAULT_PUNCTUATION = "()[]{},:;";
|
|
3321
|
+
var PunctuationExtractor = class {
|
|
3322
|
+
constructor(punctuation = DEFAULT_PUNCTUATION) {
|
|
3323
|
+
this.punctuation = punctuation;
|
|
3324
|
+
this.name = "punctuation";
|
|
3325
|
+
}
|
|
3326
|
+
canExtract(input, position) {
|
|
3327
|
+
return this.punctuation.includes(input[position]);
|
|
3328
|
+
}
|
|
3329
|
+
extract(input, position) {
|
|
3330
|
+
const char = input[position];
|
|
3331
|
+
if (this.punctuation.includes(char)) {
|
|
3332
|
+
return {
|
|
3333
|
+
value: char,
|
|
3334
|
+
length: 1
|
|
3335
|
+
};
|
|
3336
|
+
}
|
|
3337
|
+
return null;
|
|
3338
|
+
}
|
|
3339
|
+
};
|
|
3340
|
+
|
|
3341
|
+
// src/core/tokenization/default-extractors.ts
|
|
3342
|
+
function getDefaultExtractors() {
|
|
3343
|
+
return [
|
|
3344
|
+
new StringLiteralExtractor(),
|
|
3345
|
+
// "strings", 'strings', `strings`
|
|
3346
|
+
new NumberExtractor(),
|
|
3347
|
+
// 123, 45.67
|
|
3348
|
+
new OperatorExtractor(),
|
|
3349
|
+
// +, -, *, /, =, >, <, etc.
|
|
3350
|
+
new PunctuationExtractor(),
|
|
3351
|
+
// ( ) [ ] { } , : ;
|
|
3352
|
+
new IdentifierExtractor(),
|
|
3353
|
+
// variable_names, functionNames (ASCII)
|
|
3354
|
+
new UnicodeIdentifierExtractor()
|
|
3355
|
+
// CJK, Arabic, Cyrillic, etc.
|
|
3356
|
+
];
|
|
3357
|
+
}
|
|
3358
|
+
function withDefaultExtractors(tokenizer) {
|
|
3359
|
+
tokenizer.registerExtractors(getDefaultExtractors());
|
|
3360
|
+
return tokenizer;
|
|
3361
|
+
}
|
|
3362
|
+
|
|
3363
|
+
// src/core/tokenization/char-classifiers.ts
|
|
3364
|
+
function createUnicodeRangeClassifier(ranges) {
|
|
3365
|
+
return (char) => {
|
|
3366
|
+
const code = char.charCodeAt(0);
|
|
3367
|
+
return ranges.some(([start, end]) => code >= start && code <= end);
|
|
3368
|
+
};
|
|
3369
|
+
}
|
|
3370
|
+
function combineClassifiers(...classifiers) {
|
|
3371
|
+
return (char) => classifiers.some((fn) => fn(char));
|
|
3372
|
+
}
|
|
3373
|
+
function createLatinCharClassifiers(letterPattern) {
|
|
3374
|
+
const isLetter = (char) => letterPattern.test(char);
|
|
3375
|
+
const isIdentifierChar = (char) => isLetter(char) || /[0-9_-]/.test(char);
|
|
3376
|
+
return { isLetter, isIdentifierChar };
|
|
3377
|
+
}
|
|
3378
|
+
|
|
3379
|
+
// src/core/tokenization/base-tokenizer.ts
|
|
3380
|
+
var SIMPLE_TOKENIZER_OPERATOR_SET = new Set(DEFAULT_OPERATORS);
|
|
3381
|
+
var _BaseTokenizer = class _BaseTokenizer {
|
|
3382
|
+
constructor() {
|
|
3383
|
+
/** Keywords derived from profile, sorted longest-first for greedy matching */
|
|
3384
|
+
this.profileKeywords = [];
|
|
3385
|
+
/** Map for O(1) keyword lookups by lowercase native word */
|
|
3386
|
+
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
3387
|
+
/**
|
|
3388
|
+
* Pluggable value extractors for domain-specific syntax.
|
|
3389
|
+
* When registered, BaseTokenizer will use extractor-based tokenization instead of legacy methods.
|
|
3390
|
+
*/
|
|
3391
|
+
this.extractors = [];
|
|
3392
|
+
}
|
|
3393
|
+
/**
|
|
3394
|
+
* Tokenize input string to token stream.
|
|
3395
|
+
* Delegates to extractor-based tokenization if extractors are registered,
|
|
3396
|
+
* otherwise subclass must override this method.
|
|
3397
|
+
*
|
|
3398
|
+
* @param input - Input string to tokenize
|
|
3399
|
+
* @returns Token stream
|
|
3400
|
+
*/
|
|
3401
|
+
tokenize(input) {
|
|
3402
|
+
if (this.isUsingExtractors()) {
|
|
3403
|
+
return this.tokenizeWithExtractors(input);
|
|
3404
|
+
}
|
|
3405
|
+
throw new Error(
|
|
3406
|
+
`${this.constructor.name}: tokenize() not implemented and no extractors registered. Either register extractors or override tokenize() method.`
|
|
3407
|
+
);
|
|
3408
|
+
}
|
|
3409
|
+
/**
|
|
3410
|
+
* Register a value extractor for domain-specific syntax.
|
|
3411
|
+
* Extractors are tried in registration order during tokenization.
|
|
3412
|
+
* Context-aware extractors automatically receive the tokenizer context.
|
|
3413
|
+
*
|
|
3414
|
+
* @param extractor - Value extractor to register
|
|
3415
|
+
*/
|
|
3416
|
+
registerExtractor(extractor) {
|
|
3417
|
+
if (isContextAwareExtractor(extractor)) {
|
|
3418
|
+
extractor.setContext(createTokenizerContext(this));
|
|
3419
|
+
}
|
|
3420
|
+
this.extractors.push(extractor);
|
|
3421
|
+
}
|
|
3422
|
+
/**
|
|
3423
|
+
* Register multiple value extractors at once.
|
|
3424
|
+
*
|
|
3425
|
+
* @param extractors - Array of value extractors to register
|
|
3426
|
+
*/
|
|
3427
|
+
registerExtractors(extractors) {
|
|
3428
|
+
for (const extractor of extractors) {
|
|
3429
|
+
this.registerExtractor(extractor);
|
|
3430
|
+
}
|
|
3431
|
+
}
|
|
3432
|
+
/**
|
|
3433
|
+
* Clear all registered extractors.
|
|
3434
|
+
* Returns tokenizer to legacy mode.
|
|
3435
|
+
*/
|
|
3436
|
+
clearExtractors() {
|
|
3437
|
+
this.extractors = [];
|
|
3438
|
+
}
|
|
3439
|
+
/**
|
|
3440
|
+
* Check if this tokenizer is using extractor-based tokenization.
|
|
3441
|
+
* Returns true if any extractors are registered.
|
|
3442
|
+
*/
|
|
3443
|
+
isUsingExtractors() {
|
|
3444
|
+
return this.extractors.length > 0;
|
|
3445
|
+
}
|
|
3446
|
+
/**
|
|
3447
|
+
* Tokenize input using registered value extractors.
|
|
3448
|
+
* This is the new path - extractors handle all syntax detection.
|
|
3449
|
+
*
|
|
3450
|
+
* @param input - Input string to tokenize
|
|
3451
|
+
* @returns Token stream
|
|
3452
|
+
*/
|
|
3453
|
+
tokenizeWithExtractors(input) {
|
|
3454
|
+
const tokens = [];
|
|
3455
|
+
let pos = 0;
|
|
3456
|
+
while (pos < input.length) {
|
|
3457
|
+
while (pos < input.length && isWhitespace(input[pos])) {
|
|
3458
|
+
pos++;
|
|
3459
|
+
}
|
|
3460
|
+
if (pos >= input.length) break;
|
|
3461
|
+
let extracted = false;
|
|
3462
|
+
for (const extractor of this.extractors) {
|
|
3463
|
+
if (extractor.canExtract(input, pos)) {
|
|
3464
|
+
const result = extractor.extract(input, pos);
|
|
3465
|
+
if (result) {
|
|
3466
|
+
const normalized2 = result.metadata?.normalized;
|
|
3467
|
+
const stem = result.metadata?.stem;
|
|
3468
|
+
const stemConfidence = result.metadata?.stemConfidence;
|
|
3469
|
+
const cleanMetadata = {};
|
|
3470
|
+
if (result.metadata) {
|
|
3471
|
+
for (const [key, value] of Object.entries(result.metadata)) {
|
|
3472
|
+
if (key !== "normalized" && key !== "stem" && key !== "stemConfidence") {
|
|
3473
|
+
cleanMetadata[key] = value;
|
|
3474
|
+
}
|
|
3475
|
+
}
|
|
3476
|
+
}
|
|
3477
|
+
const options = {};
|
|
3478
|
+
if (normalized2) options.normalized = normalized2;
|
|
3479
|
+
if (stem) options.stem = stem;
|
|
3480
|
+
if (stemConfidence !== void 0) options.stemConfidence = stemConfidence;
|
|
3481
|
+
if (Object.keys(cleanMetadata).length > 0) options.metadata = cleanMetadata;
|
|
3482
|
+
tokens.push(
|
|
3483
|
+
createToken(
|
|
3484
|
+
result.value,
|
|
3485
|
+
this.classifyToken(result.value),
|
|
3486
|
+
createPosition(pos, pos + result.length),
|
|
3487
|
+
Object.keys(options).length > 0 ? options : void 0
|
|
3488
|
+
)
|
|
3489
|
+
);
|
|
3490
|
+
pos += result.length;
|
|
3491
|
+
extracted = true;
|
|
3492
|
+
break;
|
|
3493
|
+
}
|
|
3494
|
+
}
|
|
3495
|
+
}
|
|
3496
|
+
if (!extracted) {
|
|
3497
|
+
const char = input[pos];
|
|
3498
|
+
const kind = this.classifyUnknownChar(char);
|
|
3499
|
+
tokens.push(createToken(char, kind, createPosition(pos, pos + 1)));
|
|
3500
|
+
pos++;
|
|
3501
|
+
}
|
|
3502
|
+
}
|
|
3503
|
+
return new TokenStreamImpl(tokens, this.language);
|
|
3504
|
+
}
|
|
3505
|
+
/**
|
|
3506
|
+
* Classify an unknown character when no extractor matches.
|
|
3507
|
+
* Provides sensible defaults for common syntax.
|
|
3508
|
+
*
|
|
3509
|
+
* @param char - Character to classify
|
|
3510
|
+
* @returns Token kind
|
|
3511
|
+
*/
|
|
3512
|
+
classifyUnknownChar(char) {
|
|
3513
|
+
if ("()[]{},:;".includes(char)) return "punctuation";
|
|
3514
|
+
if ("+-*/<>=!&|".includes(char)) return "operator";
|
|
3515
|
+
return "identifier";
|
|
3516
|
+
}
|
|
3517
|
+
/**
|
|
3518
|
+
* Check if current position is a property access (obj.prop) vs CSS selector (.active).
|
|
3519
|
+
* Property access: no whitespace before '.', previous token is identifier/keyword/selector.
|
|
3520
|
+
* Also detects standalone method calls: .identifier( pattern.
|
|
3521
|
+
*
|
|
3522
|
+
* Returns true if '.' was emitted as an operator token and pos should advance by 1.
|
|
3523
|
+
* Returns false if this is a CSS selector and should be handled by trySelector().
|
|
3524
|
+
*/
|
|
3525
|
+
tryPropertyAccess(input, pos, tokens) {
|
|
3526
|
+
if (input[pos] !== ".") return false;
|
|
3527
|
+
const lastToken = tokens[tokens.length - 1];
|
|
3528
|
+
const hasWhitespaceBefore = lastToken && lastToken.position.end < pos;
|
|
3529
|
+
const isPropertyAccess = lastToken && !hasWhitespaceBefore && (lastToken.kind === "identifier" || lastToken.kind === "keyword" || lastToken.kind === "selector");
|
|
3530
|
+
if (isPropertyAccess) {
|
|
3531
|
+
tokens.push(createToken(".", "operator", createPosition(pos, pos + 1)));
|
|
3532
|
+
return true;
|
|
3533
|
+
}
|
|
3534
|
+
const methodStart = pos + 1;
|
|
3535
|
+
let methodEnd = methodStart;
|
|
3536
|
+
while (methodEnd < input.length && isAsciiIdentifierChar(input[methodEnd])) {
|
|
3537
|
+
methodEnd++;
|
|
3538
|
+
}
|
|
3539
|
+
if (methodEnd < input.length && input[methodEnd] === "(") {
|
|
3540
|
+
tokens.push(createToken(".", "operator", createPosition(pos, pos + 1)));
|
|
3541
|
+
return true;
|
|
3542
|
+
}
|
|
3543
|
+
return false;
|
|
3544
|
+
}
|
|
3545
|
+
/**
|
|
3546
|
+
* Initialize keyword mappings from a language profile.
|
|
3547
|
+
* Builds a list of native→english mappings from:
|
|
3548
|
+
* - profile.keywords (primary + alternatives)
|
|
3549
|
+
* - profile.references (me, it, you, etc.)
|
|
3550
|
+
* - profile.roleMarkers (into, from, with, etc.)
|
|
3551
|
+
*
|
|
3552
|
+
* Results are sorted longest-first for greedy matching (important for non-space languages).
|
|
3553
|
+
* Extras take precedence over profile entries when there are duplicates.
|
|
3554
|
+
*
|
|
3555
|
+
* @param profile - Language profile containing keyword translations
|
|
3556
|
+
* @param extras - Additional keyword entries to include (literals, positional, events)
|
|
3557
|
+
*/
|
|
3558
|
+
initializeKeywordsFromProfile(profile, extras = []) {
|
|
3559
|
+
const keywordMap = /* @__PURE__ */ new Map();
|
|
3560
|
+
if (profile.keywords) {
|
|
3561
|
+
for (const [normalized2, translation] of Object.entries(profile.keywords)) {
|
|
3562
|
+
keywordMap.set(translation.primary, {
|
|
3563
|
+
native: translation.primary,
|
|
3564
|
+
normalized: translation.normalized || normalized2
|
|
3565
|
+
});
|
|
3566
|
+
if (translation.alternatives) {
|
|
3567
|
+
for (const alt of translation.alternatives) {
|
|
3568
|
+
keywordMap.set(alt, {
|
|
3569
|
+
native: alt,
|
|
3570
|
+
normalized: translation.normalized || normalized2
|
|
3571
|
+
});
|
|
3572
|
+
}
|
|
3573
|
+
}
|
|
3574
|
+
}
|
|
3575
|
+
}
|
|
3576
|
+
if (profile.references) {
|
|
3577
|
+
for (const [normalized2, native] of Object.entries(profile.references)) {
|
|
3578
|
+
keywordMap.set(native, { native, normalized: normalized2 });
|
|
3579
|
+
}
|
|
3580
|
+
for (const canonical of Object.keys(profile.references)) {
|
|
3581
|
+
if (!keywordMap.has(canonical)) {
|
|
3582
|
+
keywordMap.set(canonical, { native: canonical, normalized: canonical });
|
|
3583
|
+
}
|
|
3584
|
+
}
|
|
3585
|
+
}
|
|
3586
|
+
if (profile.roleMarkers) {
|
|
3587
|
+
for (const [role, marker] of Object.entries(profile.roleMarkers)) {
|
|
3588
|
+
if (marker.primary) {
|
|
3589
|
+
keywordMap.set(marker.primary, { native: marker.primary, normalized: role });
|
|
3590
|
+
}
|
|
3591
|
+
if (marker.alternatives) {
|
|
3592
|
+
for (const alt of marker.alternatives) {
|
|
3593
|
+
keywordMap.set(alt, { native: alt, normalized: role });
|
|
3594
|
+
}
|
|
3595
|
+
}
|
|
3596
|
+
}
|
|
3597
|
+
}
|
|
3598
|
+
if (profile.possessive?.keywords) {
|
|
3599
|
+
for (const [native, normalized2] of Object.entries(profile.possessive.keywords)) {
|
|
3600
|
+
keywordMap.set(native, { native, normalized: normalized2 });
|
|
3601
|
+
}
|
|
3602
|
+
}
|
|
3603
|
+
for (const extra of extras) {
|
|
3604
|
+
keywordMap.set(extra.native, extra);
|
|
3605
|
+
}
|
|
3606
|
+
this.profileKeywords = Array.from(keywordMap.values()).sort(
|
|
3607
|
+
(a, b) => b.native.length - a.native.length
|
|
3608
|
+
);
|
|
3609
|
+
this.profileKeywordMap = /* @__PURE__ */ new Map();
|
|
3610
|
+
for (const keyword of this.profileKeywords) {
|
|
3611
|
+
this.profileKeywordMap.set(keyword.native.toLowerCase(), keyword);
|
|
3612
|
+
const normalized2 = this.removeDiacritics(keyword.native);
|
|
3613
|
+
if (normalized2 !== keyword.native && !this.profileKeywordMap.has(normalized2.toLowerCase())) {
|
|
3614
|
+
this.profileKeywordMap.set(normalized2.toLowerCase(), keyword);
|
|
3615
|
+
}
|
|
3616
|
+
}
|
|
3617
|
+
}
|
|
3618
|
+
/**
|
|
3619
|
+
* Remove diacritical marks from a word for normalization.
|
|
3620
|
+
* Primarily for Arabic (shadda, fatha, kasra, damma, sukun, etc.)
|
|
3621
|
+
* but could be extended for other languages.
|
|
3622
|
+
*
|
|
3623
|
+
* @param word - Word to normalize
|
|
3624
|
+
* @returns Word without diacritics
|
|
3625
|
+
*/
|
|
3626
|
+
removeDiacritics(word) {
|
|
3627
|
+
return word.replace(/[\u064B-\u0652\u0670]/g, "");
|
|
3628
|
+
}
|
|
3629
|
+
/**
|
|
3630
|
+
* Try to match a keyword from profile at the current position.
|
|
3631
|
+
* Uses longest-first greedy matching (important for non-space languages).
|
|
3632
|
+
*
|
|
3633
|
+
* @param input - Input string
|
|
3634
|
+
* @param pos - Current position
|
|
3635
|
+
* @returns Token if matched, null otherwise
|
|
3636
|
+
*/
|
|
3637
|
+
tryProfileKeyword(input, pos) {
|
|
3638
|
+
for (const entry of this.profileKeywords) {
|
|
3639
|
+
if (input.slice(pos).startsWith(entry.native)) {
|
|
3640
|
+
return createToken(
|
|
3641
|
+
entry.native,
|
|
3642
|
+
"keyword",
|
|
3643
|
+
createPosition(pos, pos + entry.native.length),
|
|
3644
|
+
entry.normalized
|
|
3645
|
+
);
|
|
3646
|
+
}
|
|
3647
|
+
}
|
|
3648
|
+
return null;
|
|
3649
|
+
}
|
|
3650
|
+
/**
|
|
3651
|
+
* Check if the remaining input starts with any known keyword.
|
|
3652
|
+
* Useful for non-space languages to detect word boundaries.
|
|
3653
|
+
*
|
|
3654
|
+
* @param input - Input string
|
|
3655
|
+
* @param pos - Current position
|
|
3656
|
+
* @returns true if a keyword starts at this position
|
|
3657
|
+
*/
|
|
3658
|
+
isKeywordStart(input, pos) {
|
|
3659
|
+
const remaining = input.slice(pos);
|
|
3660
|
+
return this.profileKeywords.some((entry) => remaining.startsWith(entry.native));
|
|
3661
|
+
}
|
|
3662
|
+
/**
|
|
3663
|
+
* Look up a keyword by native word (case-insensitive).
|
|
3664
|
+
* O(1) lookup using the keyword map.
|
|
3665
|
+
*
|
|
3666
|
+
* @param native - Native word to look up
|
|
3667
|
+
* @returns KeywordEntry if found, undefined otherwise
|
|
3668
|
+
*/
|
|
3669
|
+
lookupKeyword(native) {
|
|
3670
|
+
return this.profileKeywordMap.get(native.toLowerCase());
|
|
3671
|
+
}
|
|
3672
|
+
/**
|
|
3673
|
+
* Check if a word is a known keyword (case-insensitive).
|
|
3674
|
+
* O(1) lookup using the keyword map.
|
|
3675
|
+
*
|
|
3676
|
+
* @param native - Native word to check
|
|
3677
|
+
* @returns true if the word is a keyword
|
|
3678
|
+
*/
|
|
3679
|
+
isKeyword(native) {
|
|
3680
|
+
return this.profileKeywordMap.has(native.toLowerCase());
|
|
3681
|
+
}
|
|
3682
|
+
/**
|
|
3683
|
+
* Set the morphological normalizer for this tokenizer.
|
|
3684
|
+
*/
|
|
3685
|
+
setNormalizer(normalizer) {
|
|
3686
|
+
this.normalizer = normalizer;
|
|
3687
|
+
}
|
|
3688
|
+
/**
|
|
3689
|
+
* Try to normalize a word using the morphological normalizer.
|
|
3690
|
+
* Returns null if no normalizer is set or normalization fails.
|
|
3691
|
+
*
|
|
3692
|
+
* Note: We don't check isNormalizable() here because the individual tokenizers
|
|
3693
|
+
* historically called normalize() directly without that check. The normalize()
|
|
3694
|
+
* method itself handles returning noChange() for words that can't be normalized.
|
|
3695
|
+
*/
|
|
3696
|
+
tryNormalize(word) {
|
|
3697
|
+
if (!this.normalizer) return null;
|
|
3698
|
+
const result = this.normalizer.normalize(word);
|
|
3699
|
+
if (result.stem !== word && result.confidence >= 0.7) {
|
|
3700
|
+
return result;
|
|
3701
|
+
}
|
|
3702
|
+
return null;
|
|
3703
|
+
}
|
|
3704
|
+
/**
|
|
3705
|
+
* Try morphological normalization and keyword lookup.
|
|
3706
|
+
*
|
|
3707
|
+
* If the word can be normalized to a stem that matches a known keyword,
|
|
3708
|
+
* returns a keyword token with morphological metadata (stem, stemConfidence).
|
|
3709
|
+
*
|
|
3710
|
+
* This is the common pattern for handling conjugated verbs across languages:
|
|
3711
|
+
* 1. Normalize the word (e.g., "toggled" → "toggle")
|
|
3712
|
+
* 2. Look up the stem in the keyword map
|
|
3713
|
+
* 3. Create a token with both the original form and stem metadata
|
|
3714
|
+
*
|
|
3715
|
+
* @param word - The word to normalize and look up
|
|
3716
|
+
* @param startPos - Start position for the token
|
|
3717
|
+
* @param endPos - End position for the token
|
|
3718
|
+
* @returns Token if stem matches a keyword, null otherwise
|
|
3719
|
+
*/
|
|
3720
|
+
tryMorphKeywordMatch(word, startPos, endPos) {
|
|
3721
|
+
const result = this.tryNormalize(word);
|
|
3722
|
+
if (!result) return null;
|
|
3723
|
+
const stemEntry = this.lookupKeyword(result.stem);
|
|
3724
|
+
if (!stemEntry) return null;
|
|
3725
|
+
const tokenOptions = {
|
|
3726
|
+
normalized: stemEntry.normalized,
|
|
3727
|
+
stem: result.stem,
|
|
3728
|
+
stemConfidence: result.confidence
|
|
3729
|
+
};
|
|
3730
|
+
return createToken(word, "keyword", createPosition(startPos, endPos), tokenOptions);
|
|
3731
|
+
}
|
|
3732
|
+
/**
|
|
3733
|
+
* Try to extract a CSS selector at the current position.
|
|
3734
|
+
*/
|
|
3735
|
+
trySelector(input, pos) {
|
|
3736
|
+
const selector = extractCssSelector(input, pos);
|
|
3737
|
+
if (selector) {
|
|
3738
|
+
return createToken(selector, "selector", createPosition(pos, pos + selector.length));
|
|
3739
|
+
}
|
|
3740
|
+
return null;
|
|
3741
|
+
}
|
|
3742
|
+
/**
|
|
3743
|
+
* Try to extract an event modifier at the current position.
|
|
3744
|
+
* Event modifiers are .once, .debounce(N), .throttle(N), .queue(strategy)
|
|
3745
|
+
*/
|
|
3746
|
+
tryEventModifier(input, pos) {
|
|
3747
|
+
if (input[pos] !== ".") {
|
|
3748
|
+
return null;
|
|
3749
|
+
}
|
|
3750
|
+
const match = input.slice(pos).match(/^\.(?:once|debounce|throttle|queue)(?:\(([^)]+)\))?(?:\s|$|\.)/);
|
|
3751
|
+
if (!match) {
|
|
3752
|
+
return null;
|
|
3753
|
+
}
|
|
3754
|
+
const fullMatch = match[0].replace(/(\s|\.)$/, "");
|
|
3755
|
+
const modifierName = fullMatch.slice(1).split("(")[0];
|
|
3756
|
+
const value = match[1];
|
|
3757
|
+
const token = createToken(
|
|
3758
|
+
fullMatch,
|
|
3759
|
+
"event-modifier",
|
|
3760
|
+
createPosition(pos, pos + fullMatch.length)
|
|
3761
|
+
);
|
|
3762
|
+
return {
|
|
3763
|
+
...token,
|
|
3764
|
+
metadata: {
|
|
3765
|
+
modifierName,
|
|
3766
|
+
value: value ? modifierName === "queue" ? value : parseInt(value, 10) : void 0
|
|
3767
|
+
}
|
|
3768
|
+
};
|
|
3769
|
+
}
|
|
3770
|
+
/**
|
|
3771
|
+
* Try to extract a string literal at the current position.
|
|
3772
|
+
*/
|
|
3773
|
+
tryString(input, pos) {
|
|
3774
|
+
const literal = extractStringLiteral(input, pos);
|
|
3775
|
+
if (literal) {
|
|
3776
|
+
return createToken(literal, "literal", createPosition(pos, pos + literal.length));
|
|
3777
|
+
}
|
|
3778
|
+
return null;
|
|
3779
|
+
}
|
|
3780
|
+
/**
|
|
3781
|
+
* Try to extract a number at the current position.
|
|
3782
|
+
*/
|
|
3783
|
+
tryNumber(input, pos) {
|
|
3784
|
+
const number = extractNumber(input, pos);
|
|
3785
|
+
if (number) {
|
|
3786
|
+
return createToken(number, "literal", createPosition(pos, pos + number.length));
|
|
3787
|
+
}
|
|
3788
|
+
return null;
|
|
3789
|
+
}
|
|
3790
|
+
/**
|
|
3791
|
+
* Try to match a time unit from a list of patterns.
|
|
3792
|
+
*
|
|
3793
|
+
* @param input - Input string
|
|
3794
|
+
* @param pos - Position after the number
|
|
3795
|
+
* @param timeUnits - Array of time unit mappings (native pattern → standard suffix)
|
|
3796
|
+
* @param skipWhitespace - Whether to skip whitespace before time unit (default: false)
|
|
3797
|
+
* @returns Object with matched suffix and new position, or null if no match
|
|
3798
|
+
*/
|
|
3799
|
+
tryMatchTimeUnit(input, pos, timeUnits, skipWhitespace = false) {
|
|
3800
|
+
let unitPos = pos;
|
|
3801
|
+
if (skipWhitespace) {
|
|
3802
|
+
while (unitPos < input.length && isWhitespace(input[unitPos])) {
|
|
3803
|
+
unitPos++;
|
|
3804
|
+
}
|
|
3805
|
+
}
|
|
3806
|
+
const remaining = input.slice(unitPos);
|
|
3807
|
+
for (const unit of timeUnits) {
|
|
3808
|
+
const candidate = remaining.slice(0, unit.length);
|
|
3809
|
+
const matches = unit.caseInsensitive ? candidate.toLowerCase() === unit.pattern.toLowerCase() : candidate === unit.pattern;
|
|
3810
|
+
if (matches) {
|
|
3811
|
+
if (unit.notFollowedBy) {
|
|
3812
|
+
const nextChar = remaining[unit.length] || "";
|
|
3813
|
+
if (nextChar === unit.notFollowedBy) continue;
|
|
3814
|
+
}
|
|
3815
|
+
if (unit.checkBoundary) {
|
|
3816
|
+
const nextChar = remaining[unit.length] || "";
|
|
3817
|
+
if (isAsciiIdentifierChar(nextChar)) continue;
|
|
3818
|
+
}
|
|
3819
|
+
return { suffix: unit.suffix, endPos: unitPos + unit.length };
|
|
3820
|
+
}
|
|
3821
|
+
}
|
|
3822
|
+
return null;
|
|
3823
|
+
}
|
|
3824
|
+
/**
|
|
3825
|
+
* Parse a base number (sign, integer, decimal) without time units.
|
|
3826
|
+
* Returns the number string and end position.
|
|
3827
|
+
*
|
|
3828
|
+
* @param input - Input string
|
|
3829
|
+
* @param startPos - Start position
|
|
3830
|
+
* @param allowSign - Whether to allow +/- sign (default: true)
|
|
3831
|
+
* @returns Object with number string and end position, or null
|
|
3832
|
+
*/
|
|
3833
|
+
parseBaseNumber(input, startPos, allowSign = true) {
|
|
3834
|
+
let pos = startPos;
|
|
3835
|
+
let number = "";
|
|
3836
|
+
if (allowSign && (input[pos] === "-" || input[pos] === "+")) {
|
|
3837
|
+
number += input[pos++];
|
|
3838
|
+
}
|
|
3839
|
+
if (pos >= input.length || !isDigit(input[pos])) {
|
|
3840
|
+
return null;
|
|
3841
|
+
}
|
|
3842
|
+
while (pos < input.length && isDigit(input[pos])) {
|
|
3843
|
+
number += input[pos++];
|
|
3844
|
+
}
|
|
3845
|
+
if (pos < input.length && input[pos] === ".") {
|
|
3846
|
+
number += input[pos++];
|
|
3847
|
+
while (pos < input.length && isDigit(input[pos])) {
|
|
3848
|
+
number += input[pos++];
|
|
3849
|
+
}
|
|
3850
|
+
}
|
|
3851
|
+
if (!number || number === "-" || number === "+") return null;
|
|
3852
|
+
return { number, endPos: pos };
|
|
3853
|
+
}
|
|
3854
|
+
/**
|
|
3855
|
+
* Try to extract a number with native language time units.
|
|
3856
|
+
*
|
|
3857
|
+
* This is a template method that handles the common pattern:
|
|
3858
|
+
* 1. Parse the base number (sign, integer, decimal)
|
|
3859
|
+
* 2. Try to match native language time units
|
|
3860
|
+
* 3. Fall back to standard time units (ms, s, m, h)
|
|
3861
|
+
*
|
|
3862
|
+
* @param input - Input string
|
|
3863
|
+
* @param pos - Start position
|
|
3864
|
+
* @param nativeTimeUnits - Language-specific time unit mappings
|
|
3865
|
+
* @param options - Configuration options
|
|
3866
|
+
* @returns Token if number found, null otherwise
|
|
3867
|
+
*/
|
|
3868
|
+
tryNumberWithTimeUnits(input, pos, nativeTimeUnits, options = {}) {
|
|
3869
|
+
const { allowSign = true, skipWhitespace = false } = options;
|
|
3870
|
+
const baseResult = this.parseBaseNumber(input, pos, allowSign);
|
|
3871
|
+
if (!baseResult) return null;
|
|
3872
|
+
let { number, endPos } = baseResult;
|
|
3873
|
+
const allUnits = [...nativeTimeUnits, ..._BaseTokenizer.STANDARD_TIME_UNITS];
|
|
3874
|
+
const timeMatch = this.tryMatchTimeUnit(input, endPos, allUnits, skipWhitespace);
|
|
3875
|
+
if (timeMatch) {
|
|
3876
|
+
number += timeMatch.suffix;
|
|
3877
|
+
endPos = timeMatch.endPos;
|
|
3878
|
+
}
|
|
3879
|
+
return createToken(number, "literal", createPosition(pos, endPos));
|
|
3880
|
+
}
|
|
3881
|
+
/**
|
|
3882
|
+
* Try to extract a URL at the current position.
|
|
3883
|
+
* Handles /path, ./path, ../path, //domain.com, http://, https://
|
|
3884
|
+
*/
|
|
3885
|
+
tryUrl(input, pos) {
|
|
3886
|
+
const url = extractUrl(input, pos);
|
|
3887
|
+
if (url) {
|
|
3888
|
+
return createToken(url, "url", createPosition(pos, pos + url.length));
|
|
3889
|
+
}
|
|
3890
|
+
return null;
|
|
3891
|
+
}
|
|
3892
|
+
/**
|
|
3893
|
+
* Try to extract a variable reference (:varname) at the current position.
|
|
3894
|
+
* In hyperscript, :x refers to a local variable named x.
|
|
3895
|
+
*/
|
|
3896
|
+
tryVariableRef(input, pos) {
|
|
3897
|
+
if (input[pos] !== ":") return null;
|
|
3898
|
+
if (pos + 1 >= input.length) return null;
|
|
3899
|
+
if (!isAsciiIdentifierChar(input[pos + 1])) return null;
|
|
3900
|
+
let endPos = pos + 1;
|
|
3901
|
+
while (endPos < input.length && isAsciiIdentifierChar(input[endPos])) {
|
|
3902
|
+
endPos++;
|
|
3903
|
+
}
|
|
3904
|
+
const varRef = input.slice(pos, endPos);
|
|
3905
|
+
return createToken(varRef, "identifier", createPosition(pos, endPos));
|
|
3906
|
+
}
|
|
3907
|
+
/**
|
|
3908
|
+
* Try to extract an operator or punctuation token at the current position.
|
|
3909
|
+
* Handles two-character operators (==, !=, etc.) and single-character operators.
|
|
3910
|
+
*/
|
|
3911
|
+
tryOperator(input, pos) {
|
|
3912
|
+
const twoChar = input.slice(pos, pos + 2);
|
|
3913
|
+
if (["==", "!=", "<=", ">=", "&&", "||", "->"].includes(twoChar)) {
|
|
3914
|
+
return createToken(twoChar, "operator", createPosition(pos, pos + 2));
|
|
3915
|
+
}
|
|
3916
|
+
const oneChar = input[pos];
|
|
3917
|
+
if (["<", ">", "!", "+", "-", "*", "/", "="].includes(oneChar)) {
|
|
3918
|
+
return createToken(oneChar, "operator", createPosition(pos, pos + 1));
|
|
3919
|
+
}
|
|
3920
|
+
if (["(", ")", "{", "}", ",", ";", ":"].includes(oneChar)) {
|
|
3921
|
+
return createToken(oneChar, "punctuation", createPosition(pos, pos + 1));
|
|
3922
|
+
}
|
|
3923
|
+
return null;
|
|
3924
|
+
}
|
|
3925
|
+
/**
|
|
3926
|
+
* Try to match a multi-character particle from a list.
|
|
3927
|
+
*
|
|
3928
|
+
* Used by languages like Japanese, Korean, and Chinese that have
|
|
3929
|
+
* multi-character particles (e.g., Japanese から, まで, より).
|
|
3930
|
+
*
|
|
3931
|
+
* @param input - Input string
|
|
3932
|
+
* @param pos - Current position
|
|
3933
|
+
* @param particles - Array of multi-character particles to match
|
|
3934
|
+
* @returns Token if matched, null otherwise
|
|
3935
|
+
*/
|
|
3936
|
+
tryMultiCharParticle(input, pos, particles) {
|
|
3937
|
+
for (const particle of particles) {
|
|
3938
|
+
if (input.slice(pos, pos + particle.length) === particle) {
|
|
3939
|
+
return createToken(particle, "particle", createPosition(pos, pos + particle.length));
|
|
3940
|
+
}
|
|
3941
|
+
}
|
|
3942
|
+
return null;
|
|
3943
|
+
}
|
|
3944
|
+
};
|
|
3945
|
+
/**
|
|
3946
|
+
* Configuration for native language time units.
|
|
3947
|
+
* Maps patterns to their standard suffix (ms, s, m, h).
|
|
3948
|
+
*/
|
|
3949
|
+
_BaseTokenizer.STANDARD_TIME_UNITS = [
|
|
3950
|
+
{ pattern: "ms", suffix: "ms", length: 2 },
|
|
3951
|
+
{ pattern: "s", suffix: "s", length: 1, checkBoundary: true },
|
|
3952
|
+
{ pattern: "m", suffix: "m", length: 1, checkBoundary: true, notFollowedBy: "s" },
|
|
3953
|
+
{ pattern: "h", suffix: "h", length: 1, checkBoundary: true }
|
|
3954
|
+
];
|
|
3955
|
+
var BaseTokenizer = _BaseTokenizer;
|
|
3956
|
+
function createSimpleTokenizer(config) {
|
|
3957
|
+
const {
|
|
3958
|
+
language,
|
|
3959
|
+
direction = "ltr",
|
|
3960
|
+
keywords,
|
|
3961
|
+
keywordExtras,
|
|
3962
|
+
keywordProfile,
|
|
3963
|
+
includeOperators = false,
|
|
3964
|
+
caseInsensitive = true,
|
|
3965
|
+
customExtractors
|
|
3966
|
+
} = config;
|
|
3967
|
+
const keywordSet = new Set(caseInsensitive ? keywords.map((k) => k.toLowerCase()) : keywords);
|
|
3968
|
+
class SimpleTokenizer extends BaseTokenizer {
|
|
3969
|
+
constructor() {
|
|
3970
|
+
super();
|
|
3971
|
+
this.language = language;
|
|
3972
|
+
this.direction = direction;
|
|
3973
|
+
if (customExtractors) {
|
|
3974
|
+
this.registerExtractors(customExtractors);
|
|
3975
|
+
}
|
|
3976
|
+
this.registerExtractors(getDefaultExtractors());
|
|
3977
|
+
if (keywordProfile) {
|
|
3978
|
+
this.initializeKeywordsFromProfile(keywordProfile, keywordExtras);
|
|
3979
|
+
}
|
|
3980
|
+
}
|
|
3981
|
+
classifyToken(token) {
|
|
3982
|
+
const lookup = caseInsensitive ? token.toLowerCase() : token;
|
|
3983
|
+
if (keywordSet.has(lookup)) return "keyword";
|
|
3984
|
+
if (this.isKeyword(token)) return "keyword";
|
|
3985
|
+
if (/^\d/.test(token)) return "literal";
|
|
3986
|
+
if (/^['"]/.test(token)) return "literal";
|
|
3987
|
+
if (includeOperators && SIMPLE_TOKENIZER_OPERATOR_SET.has(token)) return "operator";
|
|
3988
|
+
return "identifier";
|
|
3989
|
+
}
|
|
3990
|
+
}
|
|
3991
|
+
return new SimpleTokenizer();
|
|
3992
|
+
}
|
|
3993
|
+
|
|
3994
|
+
// src/core/tokenization/morphology/types.ts
|
|
3995
|
+
function noChange(word) {
|
|
3996
|
+
return { stem: word, confidence: 1 };
|
|
3997
|
+
}
|
|
3998
|
+
function normalized(stem, confidence, metadata) {
|
|
3999
|
+
if (metadata) {
|
|
4000
|
+
return { stem, confidence, metadata };
|
|
4001
|
+
}
|
|
4002
|
+
return { stem, confidence };
|
|
4003
|
+
}
|
|
4004
|
+
|
|
4005
|
+
// src/parsing/multi-statement.ts
|
|
4006
|
+
function createMultiStatementParser(dsl, config) {
|
|
4007
|
+
return new MultiStatementParserImpl(dsl, config);
|
|
4008
|
+
}
|
|
4009
|
+
var MultiStatementParserImpl = class {
|
|
4010
|
+
constructor(dsl, config) {
|
|
4011
|
+
this.dsl = dsl;
|
|
4012
|
+
this.config = config;
|
|
4013
|
+
}
|
|
4014
|
+
parse(input, language) {
|
|
4015
|
+
const rawStatements = this.splitStatements(input, language);
|
|
4016
|
+
const statements = [];
|
|
4017
|
+
const errors = [];
|
|
4018
|
+
let previous;
|
|
4019
|
+
for (const raw of rawStatements) {
|
|
4020
|
+
const category = this.classifyLine(raw.text, language);
|
|
4021
|
+
let textToParse = raw.text;
|
|
4022
|
+
if (this.config.continuation) {
|
|
4023
|
+
const resolved = this.resolveContinuation(raw.text, language, previous?.category);
|
|
4024
|
+
if (resolved !== null) {
|
|
4025
|
+
textToParse = resolved;
|
|
4026
|
+
}
|
|
4027
|
+
}
|
|
4028
|
+
if (this.config.preprocessor) {
|
|
4029
|
+
const processed = this.config.preprocessor(textToParse, category, language, {
|
|
4030
|
+
...previous != null && { previous },
|
|
4031
|
+
lineNumber: raw.line,
|
|
4032
|
+
indent: raw.indent
|
|
4033
|
+
});
|
|
4034
|
+
if (processed === null) continue;
|
|
4035
|
+
textToParse = processed;
|
|
4036
|
+
}
|
|
4037
|
+
try {
|
|
4038
|
+
const node = this.dsl.parse(textToParse, language);
|
|
4039
|
+
const stmt = {
|
|
4040
|
+
node,
|
|
4041
|
+
source: raw.text,
|
|
4042
|
+
line: raw.line,
|
|
4043
|
+
...category != null && { category },
|
|
4044
|
+
indent: raw.indent
|
|
4045
|
+
};
|
|
4046
|
+
statements.push(stmt);
|
|
4047
|
+
previous = stmt;
|
|
4048
|
+
} catch (err) {
|
|
4049
|
+
errors.push({
|
|
4050
|
+
message: err instanceof Error ? err.message : String(err),
|
|
4051
|
+
line: raw.line,
|
|
4052
|
+
source: raw.text,
|
|
4053
|
+
code: "parse-error"
|
|
4054
|
+
});
|
|
4055
|
+
}
|
|
4056
|
+
}
|
|
4057
|
+
return { statements, errors };
|
|
4058
|
+
}
|
|
4059
|
+
/**
|
|
4060
|
+
* Split input into raw statement lines with metadata.
|
|
4061
|
+
*/
|
|
4062
|
+
splitStatements(input, language) {
|
|
4063
|
+
const { mode, trim = true, commentPrefixes = ["--", "//"] } = this.config.split;
|
|
4064
|
+
if (mode === "delimiter") {
|
|
4065
|
+
return this.splitByDelimiter(input, language, trim, commentPrefixes);
|
|
4066
|
+
}
|
|
4067
|
+
const lines = input.split("\n");
|
|
4068
|
+
const result = [];
|
|
4069
|
+
for (let i = 0; i < lines.length; i++) {
|
|
4070
|
+
const raw = lines[i];
|
|
4071
|
+
const indent = raw.length - raw.trimStart().length;
|
|
4072
|
+
const text = trim ? raw.trim() : raw;
|
|
4073
|
+
if (!text) continue;
|
|
4074
|
+
if (commentPrefixes.some((p) => text.startsWith(p))) continue;
|
|
4075
|
+
result.push({ text, line: i + 1, indent });
|
|
4076
|
+
}
|
|
4077
|
+
return result;
|
|
4078
|
+
}
|
|
4079
|
+
/**
|
|
4080
|
+
* Split by language-specific delimiters.
|
|
4081
|
+
*/
|
|
4082
|
+
splitByDelimiter(input, language, trim, commentPrefixes) {
|
|
4083
|
+
const delimiter = this.config.split.delimiters?.[language] ?? this.config.split.defaultDelimiter ?? /,\s*|\n\s*/;
|
|
4084
|
+
const parts = input.split(delimiter);
|
|
4085
|
+
const result = [];
|
|
4086
|
+
for (let i = 0; i < parts.length; i++) {
|
|
4087
|
+
const raw = parts[i];
|
|
4088
|
+
const text = trim ? raw.trim() : raw;
|
|
4089
|
+
if (!text) continue;
|
|
4090
|
+
if (commentPrefixes.some((p) => text.startsWith(p))) continue;
|
|
4091
|
+
result.push({ text, line: i + 1, indent: 0 });
|
|
4092
|
+
}
|
|
4093
|
+
return result;
|
|
4094
|
+
}
|
|
4095
|
+
/**
|
|
4096
|
+
* Classify a line by its keyword category.
|
|
4097
|
+
* Returns the category name or undefined if no match.
|
|
4098
|
+
*/
|
|
4099
|
+
classifyLine(line, language) {
|
|
4100
|
+
if (!this.config.keywords) return void 0;
|
|
4101
|
+
const lower = line.toLowerCase();
|
|
4102
|
+
const wordOrder = this.config.keywords.wordOrders?.[language] ?? "SVO";
|
|
4103
|
+
for (const [category, keywordMap] of Object.entries(this.config.keywords.categories)) {
|
|
4104
|
+
const keywords = keywordMap[language] ?? keywordMap["en"] ?? [];
|
|
4105
|
+
for (const kw of keywords) {
|
|
4106
|
+
const kwLower = kw.toLowerCase();
|
|
4107
|
+
if (wordOrder === "SOV") {
|
|
4108
|
+
if (lower.endsWith(kwLower) || lower.startsWith(kwLower)) {
|
|
4109
|
+
return category;
|
|
4110
|
+
}
|
|
4111
|
+
} else {
|
|
4112
|
+
if (lower.startsWith(kwLower)) {
|
|
4113
|
+
return category;
|
|
4114
|
+
}
|
|
4115
|
+
}
|
|
4116
|
+
}
|
|
4117
|
+
}
|
|
4118
|
+
return void 0;
|
|
4119
|
+
}
|
|
4120
|
+
/**
|
|
4121
|
+
* Resolve continuation keywords (e.g., 'and' → re-prefix with previous step type).
|
|
4122
|
+
* Returns the resolved string, or null if not a continuation.
|
|
4123
|
+
*/
|
|
4124
|
+
resolveContinuation(text, language, prevCategory) {
|
|
4125
|
+
if (!this.config.continuation || !prevCategory) return null;
|
|
4126
|
+
const contKeywords = this.config.continuation.keywords[language] ?? this.config.continuation.keywords["en"] ?? [];
|
|
4127
|
+
const lower = text.toLowerCase();
|
|
4128
|
+
for (const kw of contKeywords) {
|
|
4129
|
+
const kwLower = kw.toLowerCase();
|
|
4130
|
+
if (lower.startsWith(kwLower)) {
|
|
4131
|
+
const content = text.slice(kw.length).trim();
|
|
4132
|
+
if (!content) return null;
|
|
4133
|
+
if (this.config.continuation.resolve) {
|
|
4134
|
+
return this.config.continuation.resolve(
|
|
4135
|
+
content,
|
|
4136
|
+
prevCategory,
|
|
4137
|
+
language,
|
|
4138
|
+
this.config.keywords?.categories ?? {}
|
|
4139
|
+
);
|
|
4140
|
+
}
|
|
4141
|
+
const prevKeywords = this.config.keywords?.categories[prevCategory]?.[language] ?? this.config.keywords?.categories[prevCategory]?.["en"];
|
|
4142
|
+
if (prevKeywords?.[0]) {
|
|
4143
|
+
return prevKeywords[0] + " " + content;
|
|
4144
|
+
}
|
|
4145
|
+
return null;
|
|
4146
|
+
}
|
|
4147
|
+
}
|
|
4148
|
+
return null;
|
|
4149
|
+
}
|
|
4150
|
+
};
|
|
4151
|
+
function accumulateBlocks(statements, config) {
|
|
4152
|
+
if (config.nesting === "flat") {
|
|
4153
|
+
return accumulateFlat(statements, config);
|
|
4154
|
+
}
|
|
4155
|
+
return accumulateIndented(statements, config);
|
|
4156
|
+
}
|
|
4157
|
+
function accumulateFlat(statements, config) {
|
|
4158
|
+
const blocks = [];
|
|
4159
|
+
const orphans = [];
|
|
4160
|
+
let current = null;
|
|
4161
|
+
const flush = () => {
|
|
4162
|
+
if (current) {
|
|
4163
|
+
blocks.push({
|
|
4164
|
+
type: current.type,
|
|
4165
|
+
...current.name != null && { name: current.name },
|
|
4166
|
+
statements: current.stmts,
|
|
4167
|
+
children: [],
|
|
4168
|
+
line: current.line,
|
|
4169
|
+
indent: current.indent
|
|
4170
|
+
});
|
|
4171
|
+
current = null;
|
|
4172
|
+
}
|
|
4173
|
+
};
|
|
4174
|
+
for (const stmt of statements) {
|
|
4175
|
+
if (stmt.category && config.blockTypes.includes(stmt.category)) {
|
|
4176
|
+
flush();
|
|
4177
|
+
const name = config.extractName?.(stmt.source, stmt.category);
|
|
4178
|
+
current = {
|
|
4179
|
+
type: stmt.category,
|
|
4180
|
+
...name != null && { name },
|
|
4181
|
+
stmts: [stmt],
|
|
4182
|
+
line: stmt.line,
|
|
4183
|
+
indent: stmt.indent
|
|
4184
|
+
};
|
|
4185
|
+
} else if (current) {
|
|
4186
|
+
current.stmts.push(stmt);
|
|
4187
|
+
} else {
|
|
4188
|
+
orphans.push(stmt);
|
|
4189
|
+
}
|
|
4190
|
+
}
|
|
4191
|
+
flush();
|
|
4192
|
+
return { blocks, orphans };
|
|
4193
|
+
}
|
|
4194
|
+
function accumulateIndented(statements, config) {
|
|
4195
|
+
const rootBlocks = [];
|
|
4196
|
+
const orphans = [];
|
|
4197
|
+
const stack = [];
|
|
4198
|
+
const flushTo = (targetIndent) => {
|
|
4199
|
+
while (stack.length > 0) {
|
|
4200
|
+
const top = stack[stack.length - 1];
|
|
4201
|
+
if (top.indent >= targetIndent) {
|
|
4202
|
+
stack.pop();
|
|
4203
|
+
const block = {
|
|
4204
|
+
type: top.type,
|
|
4205
|
+
...top.name != null && { name: top.name },
|
|
4206
|
+
statements: top.stmts,
|
|
4207
|
+
children: top.children,
|
|
4208
|
+
line: top.line,
|
|
4209
|
+
indent: top.indent
|
|
4210
|
+
};
|
|
4211
|
+
if (stack.length > 0) {
|
|
4212
|
+
stack[stack.length - 1].children.push(block);
|
|
4213
|
+
} else {
|
|
4214
|
+
rootBlocks.push(block);
|
|
4215
|
+
}
|
|
4216
|
+
} else {
|
|
4217
|
+
break;
|
|
4218
|
+
}
|
|
4219
|
+
}
|
|
4220
|
+
};
|
|
4221
|
+
for (const stmt of statements) {
|
|
4222
|
+
if (stmt.category && config.blockTypes.includes(stmt.category)) {
|
|
4223
|
+
flushTo(stmt.indent);
|
|
4224
|
+
const name = config.extractName?.(stmt.source, stmt.category);
|
|
4225
|
+
stack.push({
|
|
4226
|
+
type: stmt.category,
|
|
4227
|
+
...name != null && { name },
|
|
4228
|
+
stmts: [stmt],
|
|
4229
|
+
children: [],
|
|
4230
|
+
line: stmt.line,
|
|
4231
|
+
indent: stmt.indent
|
|
4232
|
+
});
|
|
4233
|
+
} else if (stack.length > 0) {
|
|
4234
|
+
stack[stack.length - 1].stmts.push(stmt);
|
|
4235
|
+
} else {
|
|
4236
|
+
orphans.push(stmt);
|
|
4237
|
+
}
|
|
4238
|
+
}
|
|
4239
|
+
flushTo(-1);
|
|
4240
|
+
return { blocks: rootBlocks, orphans };
|
|
4241
|
+
}
|
|
4242
|
+
export {
|
|
4243
|
+
AOTOrchestrator,
|
|
4244
|
+
BaseTokenizer,
|
|
4245
|
+
CrossDomainDispatcher,
|
|
4246
|
+
DEFAULT_OPERATORS,
|
|
4247
|
+
DEFAULT_PUNCTUATION,
|
|
4248
|
+
DomainAwareScanner,
|
|
4249
|
+
DomainRegistry,
|
|
4250
|
+
GrammarTransformer,
|
|
4251
|
+
IdentifierExtractor,
|
|
4252
|
+
InMemoryDictionary,
|
|
4253
|
+
InMemoryProfileProvider,
|
|
4254
|
+
NullDictionary,
|
|
4255
|
+
NullProfileProvider,
|
|
4256
|
+
NumberExtractor,
|
|
4257
|
+
OperatorExtractor,
|
|
4258
|
+
PatternMatcher,
|
|
4259
|
+
PunctuationExtractor,
|
|
4260
|
+
StringLiteralExtractor,
|
|
4261
|
+
TokenStreamImpl,
|
|
4262
|
+
UnicodeIdentifierExtractor,
|
|
4263
|
+
WhitespaceExtractor,
|
|
4264
|
+
accumulateBlocks,
|
|
4265
|
+
buildPhrase,
|
|
4266
|
+
buildTablesFromProfiles,
|
|
4267
|
+
combineClassifiers,
|
|
4268
|
+
createCommandNode,
|
|
4269
|
+
createCompoundNode,
|
|
4270
|
+
createConditionalNode,
|
|
4271
|
+
createDiagnosticCollector,
|
|
4272
|
+
createEventHandlerNode,
|
|
4273
|
+
createExpression,
|
|
4274
|
+
createLatinCharClassifiers,
|
|
4275
|
+
createLiteral,
|
|
4276
|
+
createLoopNode,
|
|
4277
|
+
createMultiStatementParser,
|
|
4278
|
+
createMultilingualDSL,
|
|
4279
|
+
createPosition,
|
|
4280
|
+
createPropertyPath,
|
|
4281
|
+
createReference,
|
|
4282
|
+
createSchemaRenderer,
|
|
4283
|
+
createSelector,
|
|
4284
|
+
createSimpleTokenizer,
|
|
4285
|
+
createToken,
|
|
4286
|
+
createTokenizerContext,
|
|
4287
|
+
createUnicodeRangeClassifier,
|
|
4288
|
+
defineCommand,
|
|
4289
|
+
defineRole,
|
|
4290
|
+
detectWordOrders,
|
|
4291
|
+
extractCssSelector,
|
|
4292
|
+
extractNumber,
|
|
4293
|
+
extractRoleValue,
|
|
4294
|
+
extractStringLiteral,
|
|
4295
|
+
extractUrl,
|
|
4296
|
+
extractValue,
|
|
4297
|
+
filterBySeverity,
|
|
4298
|
+
fromError,
|
|
4299
|
+
generatePattern,
|
|
4300
|
+
generatePatternVariants,
|
|
4301
|
+
getAllPossessiveKeywords,
|
|
4302
|
+
getDefaultExtractors,
|
|
4303
|
+
getPossessiveReference,
|
|
4304
|
+
insertMarkers,
|
|
4305
|
+
isAsciiIdentifierChar,
|
|
4306
|
+
isAsciiLetter,
|
|
4307
|
+
isBuiltInReference,
|
|
4308
|
+
isCSSPropertyRef,
|
|
4309
|
+
isCSSSelector,
|
|
4310
|
+
isClassName,
|
|
4311
|
+
isContextAwareExtractor,
|
|
4312
|
+
isDigit,
|
|
4313
|
+
isIdSelector,
|
|
4314
|
+
isNumericValue,
|
|
4315
|
+
isPossessiveKeyword,
|
|
4316
|
+
isPossessiveMarker,
|
|
4317
|
+
isPropertyName,
|
|
4318
|
+
isQuote,
|
|
4319
|
+
isSelectorStart,
|
|
4320
|
+
isTypeCompatible,
|
|
4321
|
+
isUrlStart,
|
|
4322
|
+
isVariableRef,
|
|
4323
|
+
isWhitespace,
|
|
4324
|
+
joinTokens,
|
|
4325
|
+
lookupKeyword,
|
|
4326
|
+
lookupMarker,
|
|
4327
|
+
matchBest,
|
|
4328
|
+
matchPattern,
|
|
4329
|
+
noChange,
|
|
4330
|
+
normalized,
|
|
4331
|
+
patternMatcher,
|
|
4332
|
+
reorderRoles,
|
|
4333
|
+
validateValueType,
|
|
4334
|
+
withDefaultExtractors
|
|
4335
|
+
};
|
|
4336
|
+
//# sourceMappingURL=index.js.map
|