@lokascript/framework 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +20 -0
- package/README.md +142 -0
- package/dist/aot/aot-orchestrator.d.ts +75 -0
- package/dist/aot/aot-orchestrator.d.ts.map +1 -0
- package/dist/aot/domain-scanner.d.ts +27 -0
- package/dist/aot/domain-scanner.d.ts.map +1 -0
- package/dist/aot/index.d.ts +8 -0
- package/dist/aot/index.d.ts.map +1 -0
- package/dist/aot/types.d.ts +103 -0
- package/dist/aot/types.d.ts.map +1 -0
- package/dist/api/create-dsl.d.ts +91 -0
- package/dist/api/create-dsl.d.ts.map +1 -0
- package/dist/api/dispatcher.d.ts +108 -0
- package/dist/api/dispatcher.d.ts.map +1 -0
- package/dist/api/domain-registry.d.ts +152 -0
- package/dist/api/domain-registry.d.ts.map +1 -0
- package/dist/api/index.d.ts +7 -0
- package/dist/api/index.d.ts.map +1 -0
- package/dist/api/index.js +2082 -0
- package/dist/api/index.js.map +1 -0
- package/dist/core/index.d.ts +7 -0
- package/dist/core/index.d.ts.map +1 -0
- package/dist/core/index.js +2674 -0
- package/dist/core/index.js.map +1 -0
- package/dist/core/logger.d.ts +32 -0
- package/dist/core/logger.d.ts.map +1 -0
- package/dist/core/pattern-matching/index.d.ts +6 -0
- package/dist/core/pattern-matching/index.d.ts.map +1 -0
- package/dist/core/pattern-matching/index.js +1239 -0
- package/dist/core/pattern-matching/index.js.map +1 -0
- package/dist/core/pattern-matching/pattern-matcher.d.ts +239 -0
- package/dist/core/pattern-matching/pattern-matcher.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/index.d.ts +6 -0
- package/dist/core/pattern-matching/utils/index.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/possessive-keywords.d.ts +38 -0
- package/dist/core/pattern-matching/utils/possessive-keywords.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/type-validation.d.ts +63 -0
- package/dist/core/pattern-matching/utils/type-validation.d.ts.map +1 -0
- package/dist/core/tokenization/base-tokenizer.d.ts +344 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -0
- package/dist/core/tokenization/char-classifiers.d.ts +56 -0
- package/dist/core/tokenization/char-classifiers.d.ts.map +1 -0
- package/dist/core/tokenization/default-extractors.d.ts +48 -0
- package/dist/core/tokenization/default-extractors.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/index.d.ts +9 -0
- package/dist/core/tokenization/extractors/index.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/operator.d.ts +23 -0
- package/dist/core/tokenization/extractors/operator.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/punctuation.d.ts +22 -0
- package/dist/core/tokenization/extractors/punctuation.d.ts.map +1 -0
- package/dist/core/tokenization/extractors.d.ts +61 -0
- package/dist/core/tokenization/extractors.d.ts.map +1 -0
- package/dist/core/tokenization/index.d.ts +11 -0
- package/dist/core/tokenization/index.d.ts.map +1 -0
- package/dist/core/tokenization/index.js +1345 -0
- package/dist/core/tokenization/index.js.map +1 -0
- package/dist/core/tokenization/morphology/index.d.ts +5 -0
- package/dist/core/tokenization/morphology/index.d.ts.map +1 -0
- package/dist/core/tokenization/morphology/types.d.ts +110 -0
- package/dist/core/tokenization/morphology/types.d.ts.map +1 -0
- package/dist/core/tokenization/token-utils.d.ts +111 -0
- package/dist/core/tokenization/token-utils.d.ts.map +1 -0
- package/dist/core/types.d.ts +382 -0
- package/dist/core/types.d.ts.map +1 -0
- package/dist/core/types.js +108 -0
- package/dist/core/types.js.map +1 -0
- package/dist/generation/diagnostics.d.ts +120 -0
- package/dist/generation/diagnostics.d.ts.map +1 -0
- package/dist/generation/index.d.ts +7 -0
- package/dist/generation/index.d.ts.map +1 -0
- package/dist/generation/index.js +339 -0
- package/dist/generation/index.js.map +1 -0
- package/dist/generation/pattern-generator.d.ts +48 -0
- package/dist/generation/pattern-generator.d.ts.map +1 -0
- package/dist/generation/renderer.d.ts +115 -0
- package/dist/generation/renderer.d.ts.map +1 -0
- package/dist/grammar/index.d.ts +10 -0
- package/dist/grammar/index.d.ts.map +1 -0
- package/dist/grammar/index.js +391 -0
- package/dist/grammar/index.js.map +1 -0
- package/dist/grammar/transformer.d.ts +56 -0
- package/dist/grammar/transformer.d.ts.map +1 -0
- package/dist/grammar/types.d.ts +236 -0
- package/dist/grammar/types.d.ts.map +1 -0
- package/dist/index.cjs +4454 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.ts +46 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +4336 -0
- package/dist/index.js.map +1 -0
- package/dist/interfaces/dictionary.d.ts +82 -0
- package/dist/interfaces/dictionary.d.ts.map +1 -0
- package/dist/interfaces/index.d.ts +10 -0
- package/dist/interfaces/index.d.ts.map +1 -0
- package/dist/interfaces/profile-provider.d.ts +67 -0
- package/dist/interfaces/profile-provider.d.ts.map +1 -0
- package/dist/interfaces/value-extractor.d.ts +168 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -0
- package/dist/multilingual/index.d.ts +8 -0
- package/dist/multilingual/index.d.ts.map +1 -0
- package/dist/multilingual/index.js +1 -0
- package/dist/multilingual/index.js.map +1 -0
- package/dist/parsing/index.d.ts +8 -0
- package/dist/parsing/index.d.ts.map +1 -0
- package/dist/parsing/index.js +1415 -0
- package/dist/parsing/index.js.map +1 -0
- package/dist/parsing/multi-statement.d.ts +265 -0
- package/dist/parsing/multi-statement.d.ts.map +1 -0
- package/dist/schema/command-schema.d.ts +78 -0
- package/dist/schema/command-schema.d.ts.map +1 -0
- package/dist/schema/index.d.ts +5 -0
- package/dist/schema/index.d.ts.map +1 -0
- package/dist/schema/index.js +25 -0
- package/dist/schema/index.js.map +1 -0
- package/dist/test-setup.d.ts +9 -0
- package/dist/test-setup.d.ts.map +1 -0
- package/dist/testing/index.d.ts +50 -0
- package/dist/testing/index.d.ts.map +1 -0
- package/dist/testing/index.js +16969 -0
- package/dist/testing/index.js.map +1 -0
- package/package.json +122 -0
- package/src/__test__/fixtures/sql-dsl.ts +232 -0
- package/src/__test__/sql-integration.test.ts +189 -0
- package/src/__test__/test-utils.ts +260 -0
- package/src/aot/aot-orchestrator.test.ts +413 -0
- package/src/aot/aot-orchestrator.ts +238 -0
- package/src/aot/domain-scanner.ts +178 -0
- package/src/aot/index.ts +8 -0
- package/src/aot/types.ts +124 -0
- package/src/api/create-dsl.ts +367 -0
- package/src/api/dispatcher.test.ts +336 -0
- package/src/api/dispatcher.ts +222 -0
- package/src/api/domain-registry.test.ts +336 -0
- package/src/api/domain-registry.ts +500 -0
- package/src/api/index.ts +7 -0
- package/src/core/index.ts +7 -0
- package/src/core/logger.ts +130 -0
- package/src/core/pattern-matching/index.ts +6 -0
- package/src/core/pattern-matching/pattern-matcher.test.ts +900 -0
- package/src/core/pattern-matching/pattern-matcher.ts +1548 -0
- package/src/core/pattern-matching/pattern-matcher.ts.backup +1267 -0
- package/src/core/pattern-matching/utils/index.ts +6 -0
- package/src/core/pattern-matching/utils/possessive-keywords.ts +55 -0
- package/src/core/pattern-matching/utils/type-validation.test.ts +316 -0
- package/src/core/pattern-matching/utils/type-validation.ts +134 -0
- package/src/core/tokenization/base-tokenizer.ts +916 -0
- package/src/core/tokenization/char-classifiers.ts +79 -0
- package/src/core/tokenization/create-simple-tokenizer.test.ts +260 -0
- package/src/core/tokenization/default-extractors.ts +69 -0
- package/src/core/tokenization/extractors/index.ts +9 -0
- package/src/core/tokenization/extractors/operator.ts +75 -0
- package/src/core/tokenization/extractors/punctuation.ts +39 -0
- package/src/core/tokenization/extractors.ts +452 -0
- package/src/core/tokenization/index.ts +11 -0
- package/src/core/tokenization/morphology/index.ts +5 -0
- package/src/core/tokenization/morphology/types.ts +211 -0
- package/src/core/tokenization/token-utils.ts +252 -0
- package/src/core/types.ts +589 -0
- package/src/generation/diagnostics.test.ts +171 -0
- package/src/generation/diagnostics.ts +239 -0
- package/src/generation/index.ts +7 -0
- package/src/generation/pattern-generator.test.ts +430 -0
- package/src/generation/pattern-generator.ts +315 -0
- package/src/generation/renderer.test.ts +266 -0
- package/src/generation/renderer.ts +244 -0
- package/src/grammar/index.ts +12 -0
- package/src/grammar/transformer.ts +159 -0
- package/src/grammar/types.ts +630 -0
- package/src/index.ts +157 -0
- package/src/interfaces/dictionary.ts +123 -0
- package/src/interfaces/index.ts +10 -0
- package/src/interfaces/profile-provider.ts +88 -0
- package/src/interfaces/value-extractor.ts +435 -0
- package/src/multilingual/index.ts +9 -0
- package/src/parsing/index.ts +27 -0
- package/src/parsing/multi-statement.test.ts +480 -0
- package/src/parsing/multi-statement.ts +648 -0
- package/src/schema/command-schema.ts +118 -0
- package/src/schema/index.ts +5 -0
- package/src/test-setup.ts +45 -0
- package/src/testing/index.ts +137 -0
|
@@ -0,0 +1,648 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Multi-statement parser for domain DSLs.
|
|
3
|
+
*
|
|
4
|
+
* Provides generic mechanics for parsing multi-line/multi-statement DSL input:
|
|
5
|
+
* - Statement splitting (line-based or delimiter-based)
|
|
6
|
+
* - Keyword classification (SVO start / SOV end detection)
|
|
7
|
+
* - Block accumulation (grouping lines into blocks)
|
|
8
|
+
* - Error collection with line numbers
|
|
9
|
+
*
|
|
10
|
+
* Domains provide the semantics (what keywords mean, preprocessing, hierarchy).
|
|
11
|
+
* The framework provides the mechanics.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import type { SemanticNode } from '../core/types';
|
|
15
|
+
import type { MultilingualDSL } from '../api/create-dsl';
|
|
16
|
+
|
|
17
|
+
// =============================================================================
|
|
18
|
+
// Configuration Types
|
|
19
|
+
// =============================================================================
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Word order affects keyword detection position.
|
|
23
|
+
*/
|
|
24
|
+
export type WordOrderHint = 'SVO' | 'SOV' | 'VSO';
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* How to split input into individual statements.
|
|
28
|
+
*/
|
|
29
|
+
export interface SplitConfig {
|
|
30
|
+
/**
|
|
31
|
+
* Split mode:
|
|
32
|
+
* - 'line': split on newlines (for indentation-based DSLs)
|
|
33
|
+
* - 'delimiter': split on language-specific delimiters
|
|
34
|
+
*/
|
|
35
|
+
readonly mode: 'line' | 'delimiter';
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Language-specific delimiter patterns (used when mode='delimiter').
|
|
39
|
+
* Key is language code, value is regex pattern.
|
|
40
|
+
* Example: { en: /,\s*|\n\s*/, ja: /、|。|\n\s*/ }
|
|
41
|
+
*/
|
|
42
|
+
readonly delimiters?: Readonly<Record<string, RegExp>>;
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Default delimiter when language not in delimiters map.
|
|
46
|
+
*/
|
|
47
|
+
readonly defaultDelimiter?: RegExp;
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Whether to trim each statement after splitting.
|
|
51
|
+
* Default: true
|
|
52
|
+
*/
|
|
53
|
+
readonly trim?: boolean;
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Comment prefixes to strip. Lines starting with these are skipped.
|
|
57
|
+
* Default: ['--', '//']
|
|
58
|
+
*/
|
|
59
|
+
readonly commentPrefixes?: readonly string[];
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* A keyword category with its translations.
|
|
64
|
+
* Maps language code to array of keyword variants.
|
|
65
|
+
*/
|
|
66
|
+
export type KeywordMap = Readonly<Record<string, readonly string[]>>;
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* Configuration for keyword classification.
|
|
70
|
+
*/
|
|
71
|
+
export interface KeywordConfig {
|
|
72
|
+
/**
|
|
73
|
+
* Named keyword categories with their per-language translations.
|
|
74
|
+
* Example: { test: { en: ['test'], ja: ['テスト'], es: ['prueba'] } }
|
|
75
|
+
*/
|
|
76
|
+
readonly categories: Readonly<Record<string, KeywordMap>>;
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Word order hint per language (affects where keywords are detected).
|
|
80
|
+
* Default for unlisted languages: 'SVO'
|
|
81
|
+
*/
|
|
82
|
+
readonly wordOrders?: Readonly<Record<string, WordOrderHint>>;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* Configuration for continuation resolution (e.g., BDD's "and" keyword).
|
|
87
|
+
*/
|
|
88
|
+
export interface ContinuationConfig {
|
|
89
|
+
/**
|
|
90
|
+
* The keyword(s) that signal continuation, per language.
|
|
91
|
+
* Example: { en: ['and'], es: ['y'], ja: ['かつ'], ar: ['و'] }
|
|
92
|
+
*/
|
|
93
|
+
readonly keywords: KeywordMap;
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* How to resolve continuation: re-prefix with previous step's keyword.
|
|
97
|
+
* The function receives (content after 'and', previous step's category, language)
|
|
98
|
+
* and returns the re-prefixed string for re-parsing.
|
|
99
|
+
*
|
|
100
|
+
* Default: `(content, prevCategory, language, categoryKeywords) =>
|
|
101
|
+
* categoryKeywords[prevCategory]?.[language]?.[0] + ' ' + content`
|
|
102
|
+
*/
|
|
103
|
+
readonly resolve?: (
|
|
104
|
+
content: string,
|
|
105
|
+
prevCategory: string,
|
|
106
|
+
language: string,
|
|
107
|
+
categoryKeywords: Record<string, KeywordMap>
|
|
108
|
+
) => string;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/**
|
|
112
|
+
* A parsed statement with its line context.
|
|
113
|
+
*/
|
|
114
|
+
export interface ParsedStatement {
|
|
115
|
+
/** The parsed semantic node */
|
|
116
|
+
readonly node: SemanticNode;
|
|
117
|
+
/** Original source line/text */
|
|
118
|
+
readonly source: string;
|
|
119
|
+
/** Line number (1-based) in original input */
|
|
120
|
+
readonly line: number;
|
|
121
|
+
/** Detected keyword category (e.g., 'test', 'given', 'when') */
|
|
122
|
+
readonly category?: string;
|
|
123
|
+
/** Indentation level (number of leading spaces, 0-based) */
|
|
124
|
+
readonly indent: number;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* A parse error with location information.
|
|
129
|
+
*/
|
|
130
|
+
export interface StatementError {
|
|
131
|
+
/** Error message */
|
|
132
|
+
readonly message: string;
|
|
133
|
+
/** Line number (1-based) in original input */
|
|
134
|
+
readonly line: number;
|
|
135
|
+
/** Original source line/text */
|
|
136
|
+
readonly source: string;
|
|
137
|
+
/** Error code for programmatic handling */
|
|
138
|
+
readonly code?: 'parse-error' | 'unexpected-line' | 'continuation-error';
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* Result of multi-statement parsing.
|
|
143
|
+
*/
|
|
144
|
+
export interface MultiStatementResult {
|
|
145
|
+
/** Successfully parsed statements */
|
|
146
|
+
readonly statements: readonly ParsedStatement[];
|
|
147
|
+
/** Errors encountered during parsing */
|
|
148
|
+
readonly errors: readonly StatementError[];
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
/**
|
|
152
|
+
* Preprocessor function called before each statement is parsed.
|
|
153
|
+
* Returns the modified string to parse, or null to skip the line.
|
|
154
|
+
*/
|
|
155
|
+
export type StatementPreprocessor = (
|
|
156
|
+
line: string,
|
|
157
|
+
category: string | undefined,
|
|
158
|
+
language: string,
|
|
159
|
+
context: PreprocessorContext
|
|
160
|
+
) => string | null;
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* Context available to preprocessors.
|
|
164
|
+
*/
|
|
165
|
+
export interface PreprocessorContext {
|
|
166
|
+
/** The previous parsed statement, if any */
|
|
167
|
+
readonly previous?: ParsedStatement;
|
|
168
|
+
/** Line number (1-based) */
|
|
169
|
+
readonly lineNumber: number;
|
|
170
|
+
/** Indentation level */
|
|
171
|
+
readonly indent: number;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* Full configuration for multi-statement parser.
|
|
176
|
+
*/
|
|
177
|
+
export interface MultiStatementConfig {
|
|
178
|
+
/** How to split input into statements */
|
|
179
|
+
readonly split: SplitConfig;
|
|
180
|
+
/** Keyword classification (optional — if omitted, no category detection) */
|
|
181
|
+
readonly keywords?: KeywordConfig;
|
|
182
|
+
/** Continuation resolution (optional — e.g., BDD 'and') */
|
|
183
|
+
readonly continuation?: ContinuationConfig;
|
|
184
|
+
/** Statement preprocessor (optional — for article stripping, expect-prepending, etc.) */
|
|
185
|
+
readonly preprocessor?: StatementPreprocessor;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
// =============================================================================
|
|
189
|
+
// Multi-Statement Parser
|
|
190
|
+
// =============================================================================
|
|
191
|
+
|
|
192
|
+
/**
|
|
193
|
+
* A reusable multi-statement parser.
|
|
194
|
+
* Created via `createMultiStatementParser()`.
|
|
195
|
+
*/
|
|
196
|
+
export interface MultiStatementParser {
|
|
197
|
+
/**
|
|
198
|
+
* Parse multi-statement input into an array of parsed statements.
|
|
199
|
+
*/
|
|
200
|
+
parse(input: string, language: string): MultiStatementResult;
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
/**
|
|
204
|
+
* Create a multi-statement parser that wraps a `MultilingualDSL` instance.
|
|
205
|
+
*
|
|
206
|
+
* The parser handles splitting, keyword detection, continuation resolution,
|
|
207
|
+
* preprocessing, and error collection. It delegates single-statement parsing
|
|
208
|
+
* to the provided DSL's `parse()` method.
|
|
209
|
+
*
|
|
210
|
+
* @example
|
|
211
|
+
* ```typescript
|
|
212
|
+
* const parser = createMultiStatementParser(myDSL, {
|
|
213
|
+
* split: { mode: 'line', commentPrefixes: ['--', '//'] },
|
|
214
|
+
* keywords: {
|
|
215
|
+
* categories: {
|
|
216
|
+
* given: { en: ['given'], ja: ['前提'], es: ['dado'] },
|
|
217
|
+
* when: { en: ['when'], ja: ['もし'], es: ['cuando'] },
|
|
218
|
+
* then: { en: ['then'], ja: ['ならば'], es: ['entonces'] },
|
|
219
|
+
* },
|
|
220
|
+
* wordOrders: { ja: 'SOV', ar: 'VSO' },
|
|
221
|
+
* },
|
|
222
|
+
* continuation: {
|
|
223
|
+
* keywords: { en: ['and'], es: ['y'], ja: ['かつ'] },
|
|
224
|
+
* },
|
|
225
|
+
* });
|
|
226
|
+
*
|
|
227
|
+
* const result = parser.parse(`
|
|
228
|
+
* given #login is visible
|
|
229
|
+
* when click on #submit
|
|
230
|
+
* then #dashboard appears
|
|
231
|
+
* `, 'en');
|
|
232
|
+
* ```
|
|
233
|
+
*/
|
|
234
|
+
export function createMultiStatementParser(
|
|
235
|
+
dsl: MultilingualDSL,
|
|
236
|
+
config: MultiStatementConfig
|
|
237
|
+
): MultiStatementParser {
|
|
238
|
+
return new MultiStatementParserImpl(dsl, config);
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
// =============================================================================
|
|
242
|
+
// Implementation
|
|
243
|
+
// =============================================================================
|
|
244
|
+
|
|
245
|
+
class MultiStatementParserImpl implements MultiStatementParser {
|
|
246
|
+
constructor(
|
|
247
|
+
private readonly dsl: MultilingualDSL,
|
|
248
|
+
private readonly config: MultiStatementConfig
|
|
249
|
+
) {}
|
|
250
|
+
|
|
251
|
+
parse(input: string, language: string): MultiStatementResult {
|
|
252
|
+
const rawStatements = this.splitStatements(input, language);
|
|
253
|
+
const statements: ParsedStatement[] = [];
|
|
254
|
+
const errors: StatementError[] = [];
|
|
255
|
+
let previous: ParsedStatement | undefined;
|
|
256
|
+
|
|
257
|
+
for (const raw of rawStatements) {
|
|
258
|
+
// Detect keyword category
|
|
259
|
+
const category = this.classifyLine(raw.text, language);
|
|
260
|
+
|
|
261
|
+
// Handle continuation (e.g., 'and')
|
|
262
|
+
let textToParse = raw.text;
|
|
263
|
+
if (this.config.continuation) {
|
|
264
|
+
const resolved = this.resolveContinuation(raw.text, language, previous?.category);
|
|
265
|
+
if (resolved !== null) {
|
|
266
|
+
textToParse = resolved;
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
// Run preprocessor
|
|
271
|
+
if (this.config.preprocessor) {
|
|
272
|
+
const processed = this.config.preprocessor(textToParse, category, language, {
|
|
273
|
+
...(previous != null && { previous }),
|
|
274
|
+
lineNumber: raw.line,
|
|
275
|
+
indent: raw.indent,
|
|
276
|
+
});
|
|
277
|
+
if (processed === null) continue; // skip line
|
|
278
|
+
textToParse = processed;
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
// Parse with DSL
|
|
282
|
+
try {
|
|
283
|
+
const node = this.dsl.parse(textToParse, language);
|
|
284
|
+
const stmt: ParsedStatement = {
|
|
285
|
+
node,
|
|
286
|
+
source: raw.text,
|
|
287
|
+
line: raw.line,
|
|
288
|
+
...(category != null && { category }),
|
|
289
|
+
indent: raw.indent,
|
|
290
|
+
};
|
|
291
|
+
statements.push(stmt);
|
|
292
|
+
previous = stmt;
|
|
293
|
+
} catch (err) {
|
|
294
|
+
errors.push({
|
|
295
|
+
message: err instanceof Error ? err.message : String(err),
|
|
296
|
+
line: raw.line,
|
|
297
|
+
source: raw.text,
|
|
298
|
+
code: 'parse-error',
|
|
299
|
+
});
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
return { statements, errors };
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
/**
|
|
307
|
+
* Split input into raw statement lines with metadata.
|
|
308
|
+
*/
|
|
309
|
+
private splitStatements(
|
|
310
|
+
input: string,
|
|
311
|
+
language: string
|
|
312
|
+
): Array<{ text: string; line: number; indent: number }> {
|
|
313
|
+
const { mode, trim = true, commentPrefixes = ['--', '//'] } = this.config.split;
|
|
314
|
+
|
|
315
|
+
if (mode === 'delimiter') {
|
|
316
|
+
return this.splitByDelimiter(input, language, trim, commentPrefixes);
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
// Line-based splitting
|
|
320
|
+
const lines = input.split('\n');
|
|
321
|
+
const result: Array<{ text: string; line: number; indent: number }> = [];
|
|
322
|
+
|
|
323
|
+
for (let i = 0; i < lines.length; i++) {
|
|
324
|
+
const raw = lines[i];
|
|
325
|
+
const indent = raw.length - raw.trimStart().length;
|
|
326
|
+
const text = trim ? raw.trim() : raw;
|
|
327
|
+
|
|
328
|
+
// Skip empty lines
|
|
329
|
+
if (!text) continue;
|
|
330
|
+
|
|
331
|
+
// Skip comments
|
|
332
|
+
if (commentPrefixes.some(p => text.startsWith(p))) continue;
|
|
333
|
+
|
|
334
|
+
result.push({ text, line: i + 1, indent });
|
|
335
|
+
}
|
|
336
|
+
|
|
337
|
+
return result;
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
/**
|
|
341
|
+
* Split by language-specific delimiters.
|
|
342
|
+
*/
|
|
343
|
+
private splitByDelimiter(
|
|
344
|
+
input: string,
|
|
345
|
+
language: string,
|
|
346
|
+
trim: boolean,
|
|
347
|
+
commentPrefixes: readonly string[]
|
|
348
|
+
): Array<{ text: string; line: number; indent: number }> {
|
|
349
|
+
const delimiter =
|
|
350
|
+
this.config.split.delimiters?.[language] ??
|
|
351
|
+
this.config.split.defaultDelimiter ??
|
|
352
|
+
/,\s*|\n\s*/;
|
|
353
|
+
|
|
354
|
+
const parts = input.split(delimiter);
|
|
355
|
+
const result: Array<{ text: string; line: number; indent: number }> = [];
|
|
356
|
+
|
|
357
|
+
for (let i = 0; i < parts.length; i++) {
|
|
358
|
+
const raw = parts[i];
|
|
359
|
+
const text = trim ? raw.trim() : raw;
|
|
360
|
+
|
|
361
|
+
if (!text) continue;
|
|
362
|
+
if (commentPrefixes.some(p => text.startsWith(p))) continue;
|
|
363
|
+
|
|
364
|
+
result.push({ text, line: i + 1, indent: 0 });
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
return result;
|
|
368
|
+
}
|
|
369
|
+
|
|
370
|
+
/**
|
|
371
|
+
* Classify a line by its keyword category.
|
|
372
|
+
* Returns the category name or undefined if no match.
|
|
373
|
+
*/
|
|
374
|
+
private classifyLine(line: string, language: string): string | undefined {
|
|
375
|
+
if (!this.config.keywords) return undefined;
|
|
376
|
+
|
|
377
|
+
const lower = line.toLowerCase();
|
|
378
|
+
const wordOrder = this.config.keywords.wordOrders?.[language] ?? 'SVO';
|
|
379
|
+
|
|
380
|
+
for (const [category, keywordMap] of Object.entries(this.config.keywords.categories)) {
|
|
381
|
+
const keywords = keywordMap[language] ?? keywordMap['en'] ?? [];
|
|
382
|
+
for (const kw of keywords) {
|
|
383
|
+
const kwLower = kw.toLowerCase();
|
|
384
|
+
if (wordOrder === 'SOV') {
|
|
385
|
+
// SOV: check end of line first, then start
|
|
386
|
+
if (lower.endsWith(kwLower) || lower.startsWith(kwLower)) {
|
|
387
|
+
return category;
|
|
388
|
+
}
|
|
389
|
+
} else {
|
|
390
|
+
// SVO/VSO: check start of line
|
|
391
|
+
if (lower.startsWith(kwLower)) {
|
|
392
|
+
return category;
|
|
393
|
+
}
|
|
394
|
+
}
|
|
395
|
+
}
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
return undefined;
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
/**
|
|
402
|
+
* Resolve continuation keywords (e.g., 'and' → re-prefix with previous step type).
|
|
403
|
+
* Returns the resolved string, or null if not a continuation.
|
|
404
|
+
*/
|
|
405
|
+
private resolveContinuation(
|
|
406
|
+
text: string,
|
|
407
|
+
language: string,
|
|
408
|
+
prevCategory: string | undefined
|
|
409
|
+
): string | null {
|
|
410
|
+
if (!this.config.continuation || !prevCategory) return null;
|
|
411
|
+
|
|
412
|
+
const contKeywords =
|
|
413
|
+
this.config.continuation.keywords[language] ?? this.config.continuation.keywords['en'] ?? [];
|
|
414
|
+
|
|
415
|
+
const lower = text.toLowerCase();
|
|
416
|
+
for (const kw of contKeywords) {
|
|
417
|
+
const kwLower = kw.toLowerCase();
|
|
418
|
+
if (lower.startsWith(kwLower)) {
|
|
419
|
+
const content = text.slice(kw.length).trim();
|
|
420
|
+
if (!content) return null;
|
|
421
|
+
|
|
422
|
+
if (this.config.continuation.resolve) {
|
|
423
|
+
return this.config.continuation.resolve(
|
|
424
|
+
content,
|
|
425
|
+
prevCategory,
|
|
426
|
+
language,
|
|
427
|
+
this.config.keywords?.categories ?? {}
|
|
428
|
+
);
|
|
429
|
+
}
|
|
430
|
+
|
|
431
|
+
// Default: prepend previous category's keyword
|
|
432
|
+
const prevKeywords =
|
|
433
|
+
this.config.keywords?.categories[prevCategory]?.[language] ??
|
|
434
|
+
this.config.keywords?.categories[prevCategory]?.['en'];
|
|
435
|
+
if (prevKeywords?.[0]) {
|
|
436
|
+
return prevKeywords[0] + ' ' + content;
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
return null;
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
|
|
443
|
+
return null;
|
|
444
|
+
}
|
|
445
|
+
}
|
|
446
|
+
|
|
447
|
+
// =============================================================================
|
|
448
|
+
// Block Accumulator (for grouped/nested structures)
|
|
449
|
+
// =============================================================================
|
|
450
|
+
|
|
451
|
+
/**
|
|
452
|
+
* A block of accumulated statements, grouped by a block-opening keyword.
|
|
453
|
+
*/
|
|
454
|
+
export interface StatementBlock<TMeta = unknown> {
|
|
455
|
+
/** Block type (e.g., 'test', 'feature', 'scenario') */
|
|
456
|
+
readonly type: string;
|
|
457
|
+
/** Block name, if any (e.g., test name from quotes) */
|
|
458
|
+
readonly name?: string;
|
|
459
|
+
/** Statements within this block */
|
|
460
|
+
readonly statements: readonly ParsedStatement[];
|
|
461
|
+
/** Child blocks (for nested structures) */
|
|
462
|
+
readonly children: readonly StatementBlock<TMeta>[];
|
|
463
|
+
/** Starting line number */
|
|
464
|
+
readonly line: number;
|
|
465
|
+
/** Indentation level of the block opener */
|
|
466
|
+
readonly indent: number;
|
|
467
|
+
/** Domain-specific metadata */
|
|
468
|
+
readonly meta?: TMeta;
|
|
469
|
+
}
|
|
470
|
+
|
|
471
|
+
/**
|
|
472
|
+
* Configuration for block accumulation.
|
|
473
|
+
*/
|
|
474
|
+
export interface BlockConfig {
|
|
475
|
+
/**
|
|
476
|
+
* Block-opening keyword categories.
|
|
477
|
+
* Statements classified in these categories start new blocks.
|
|
478
|
+
*/
|
|
479
|
+
readonly blockTypes: readonly string[];
|
|
480
|
+
|
|
481
|
+
/**
|
|
482
|
+
* How to determine block nesting:
|
|
483
|
+
* - 'indent': indentation-based (deeper indent = child block)
|
|
484
|
+
* - 'flat': all blocks at same level (no nesting)
|
|
485
|
+
*/
|
|
486
|
+
readonly nesting: 'indent' | 'flat';
|
|
487
|
+
|
|
488
|
+
/**
|
|
489
|
+
* Extract block name from statement text.
|
|
490
|
+
* Default: extracts quoted string after keyword (e.g., test "Login" → "Login")
|
|
491
|
+
*/
|
|
492
|
+
readonly extractName?: (text: string, category: string) => string | undefined;
|
|
493
|
+
}
|
|
494
|
+
|
|
495
|
+
/**
|
|
496
|
+
* Result of block accumulation.
|
|
497
|
+
*/
|
|
498
|
+
export interface BlockResult<TMeta = unknown> {
|
|
499
|
+
/** Top-level blocks */
|
|
500
|
+
readonly blocks: readonly StatementBlock<TMeta>[];
|
|
501
|
+
/** Statements not inside any block */
|
|
502
|
+
readonly orphans: readonly ParsedStatement[];
|
|
503
|
+
}
|
|
504
|
+
|
|
505
|
+
/**
|
|
506
|
+
* Accumulate parsed statements into blocks based on keyword categories.
|
|
507
|
+
*
|
|
508
|
+
* This groups statements into hierarchical blocks based on block-opening keywords
|
|
509
|
+
* and indentation levels. Useful for test/feature/scenario structures.
|
|
510
|
+
*
|
|
511
|
+
* @example
|
|
512
|
+
* ```typescript
|
|
513
|
+
* const result = accumulateBlocks(statements, {
|
|
514
|
+
* blockTypes: ['test', 'feature', 'setup'],
|
|
515
|
+
* nesting: 'indent',
|
|
516
|
+
* extractName: (text, category) => {
|
|
517
|
+
* const match = text.match(/"([^"]+)"/);
|
|
518
|
+
* return match?.[1];
|
|
519
|
+
* },
|
|
520
|
+
* });
|
|
521
|
+
* ```
|
|
522
|
+
*/
|
|
523
|
+
export function accumulateBlocks<TMeta = unknown>(
|
|
524
|
+
statements: readonly ParsedStatement[],
|
|
525
|
+
config: BlockConfig
|
|
526
|
+
): BlockResult<TMeta> {
|
|
527
|
+
if (config.nesting === 'flat') {
|
|
528
|
+
return accumulateFlat(statements, config);
|
|
529
|
+
}
|
|
530
|
+
return accumulateIndented(statements, config);
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
function accumulateFlat<TMeta>(
|
|
534
|
+
statements: readonly ParsedStatement[],
|
|
535
|
+
config: BlockConfig
|
|
536
|
+
): BlockResult<TMeta> {
|
|
537
|
+
const blocks: StatementBlock<TMeta>[] = [];
|
|
538
|
+
const orphans: ParsedStatement[] = [];
|
|
539
|
+
let current: {
|
|
540
|
+
type: string;
|
|
541
|
+
name?: string;
|
|
542
|
+
stmts: ParsedStatement[];
|
|
543
|
+
line: number;
|
|
544
|
+
indent: number;
|
|
545
|
+
} | null = null;
|
|
546
|
+
|
|
547
|
+
const flush = () => {
|
|
548
|
+
if (current) {
|
|
549
|
+
blocks.push({
|
|
550
|
+
type: current.type,
|
|
551
|
+
...(current.name != null && { name: current.name }),
|
|
552
|
+
statements: current.stmts,
|
|
553
|
+
children: [],
|
|
554
|
+
line: current.line,
|
|
555
|
+
indent: current.indent,
|
|
556
|
+
});
|
|
557
|
+
current = null;
|
|
558
|
+
}
|
|
559
|
+
};
|
|
560
|
+
|
|
561
|
+
for (const stmt of statements) {
|
|
562
|
+
if (stmt.category && config.blockTypes.includes(stmt.category)) {
|
|
563
|
+
flush();
|
|
564
|
+
const name = config.extractName?.(stmt.source, stmt.category);
|
|
565
|
+
current = {
|
|
566
|
+
type: stmt.category,
|
|
567
|
+
...(name != null && { name }),
|
|
568
|
+
stmts: [stmt],
|
|
569
|
+
line: stmt.line,
|
|
570
|
+
indent: stmt.indent,
|
|
571
|
+
};
|
|
572
|
+
} else if (current) {
|
|
573
|
+
current.stmts.push(stmt);
|
|
574
|
+
} else {
|
|
575
|
+
orphans.push(stmt);
|
|
576
|
+
}
|
|
577
|
+
}
|
|
578
|
+
|
|
579
|
+
flush();
|
|
580
|
+
return { blocks, orphans };
|
|
581
|
+
}
|
|
582
|
+
|
|
583
|
+
function accumulateIndented<TMeta>(
|
|
584
|
+
statements: readonly ParsedStatement[],
|
|
585
|
+
config: BlockConfig
|
|
586
|
+
): BlockResult<TMeta> {
|
|
587
|
+
const rootBlocks: StatementBlock<TMeta>[] = [];
|
|
588
|
+
const orphans: ParsedStatement[] = [];
|
|
589
|
+
|
|
590
|
+
// Stack of open blocks with their indent level
|
|
591
|
+
const stack: Array<{
|
|
592
|
+
type: string;
|
|
593
|
+
name?: string;
|
|
594
|
+
stmts: ParsedStatement[];
|
|
595
|
+
children: StatementBlock<TMeta>[];
|
|
596
|
+
line: number;
|
|
597
|
+
indent: number;
|
|
598
|
+
}> = [];
|
|
599
|
+
|
|
600
|
+
const flushTo = (targetIndent: number) => {
|
|
601
|
+
while (stack.length > 0) {
|
|
602
|
+
const top = stack[stack.length - 1];
|
|
603
|
+
if (top.indent >= targetIndent) {
|
|
604
|
+
stack.pop();
|
|
605
|
+
const block: StatementBlock<TMeta> = {
|
|
606
|
+
type: top.type,
|
|
607
|
+
...(top.name != null && { name: top.name }),
|
|
608
|
+
statements: top.stmts,
|
|
609
|
+
children: top.children,
|
|
610
|
+
line: top.line,
|
|
611
|
+
indent: top.indent,
|
|
612
|
+
};
|
|
613
|
+
if (stack.length > 0) {
|
|
614
|
+
stack[stack.length - 1].children.push(block);
|
|
615
|
+
} else {
|
|
616
|
+
rootBlocks.push(block);
|
|
617
|
+
}
|
|
618
|
+
} else {
|
|
619
|
+
break;
|
|
620
|
+
}
|
|
621
|
+
}
|
|
622
|
+
};
|
|
623
|
+
|
|
624
|
+
for (const stmt of statements) {
|
|
625
|
+
if (stmt.category && config.blockTypes.includes(stmt.category)) {
|
|
626
|
+
// Flush blocks at same or deeper indent
|
|
627
|
+
flushTo(stmt.indent);
|
|
628
|
+
const name = config.extractName?.(stmt.source, stmt.category);
|
|
629
|
+
stack.push({
|
|
630
|
+
type: stmt.category,
|
|
631
|
+
...(name != null && { name }),
|
|
632
|
+
stmts: [stmt],
|
|
633
|
+
children: [],
|
|
634
|
+
line: stmt.line,
|
|
635
|
+
indent: stmt.indent,
|
|
636
|
+
});
|
|
637
|
+
} else if (stack.length > 0) {
|
|
638
|
+
stack[stack.length - 1].stmts.push(stmt);
|
|
639
|
+
} else {
|
|
640
|
+
orphans.push(stmt);
|
|
641
|
+
}
|
|
642
|
+
}
|
|
643
|
+
|
|
644
|
+
// Flush remaining
|
|
645
|
+
flushTo(-1);
|
|
646
|
+
|
|
647
|
+
return { blocks: rootBlocks, orphans };
|
|
648
|
+
}
|