@lokascript/framework 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +20 -0
- package/README.md +142 -0
- package/dist/aot/aot-orchestrator.d.ts +75 -0
- package/dist/aot/aot-orchestrator.d.ts.map +1 -0
- package/dist/aot/domain-scanner.d.ts +27 -0
- package/dist/aot/domain-scanner.d.ts.map +1 -0
- package/dist/aot/index.d.ts +8 -0
- package/dist/aot/index.d.ts.map +1 -0
- package/dist/aot/types.d.ts +103 -0
- package/dist/aot/types.d.ts.map +1 -0
- package/dist/api/create-dsl.d.ts +91 -0
- package/dist/api/create-dsl.d.ts.map +1 -0
- package/dist/api/dispatcher.d.ts +108 -0
- package/dist/api/dispatcher.d.ts.map +1 -0
- package/dist/api/domain-registry.d.ts +152 -0
- package/dist/api/domain-registry.d.ts.map +1 -0
- package/dist/api/index.d.ts +7 -0
- package/dist/api/index.d.ts.map +1 -0
- package/dist/api/index.js +2082 -0
- package/dist/api/index.js.map +1 -0
- package/dist/core/index.d.ts +7 -0
- package/dist/core/index.d.ts.map +1 -0
- package/dist/core/index.js +2674 -0
- package/dist/core/index.js.map +1 -0
- package/dist/core/logger.d.ts +32 -0
- package/dist/core/logger.d.ts.map +1 -0
- package/dist/core/pattern-matching/index.d.ts +6 -0
- package/dist/core/pattern-matching/index.d.ts.map +1 -0
- package/dist/core/pattern-matching/index.js +1239 -0
- package/dist/core/pattern-matching/index.js.map +1 -0
- package/dist/core/pattern-matching/pattern-matcher.d.ts +239 -0
- package/dist/core/pattern-matching/pattern-matcher.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/index.d.ts +6 -0
- package/dist/core/pattern-matching/utils/index.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/possessive-keywords.d.ts +38 -0
- package/dist/core/pattern-matching/utils/possessive-keywords.d.ts.map +1 -0
- package/dist/core/pattern-matching/utils/type-validation.d.ts +63 -0
- package/dist/core/pattern-matching/utils/type-validation.d.ts.map +1 -0
- package/dist/core/tokenization/base-tokenizer.d.ts +344 -0
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -0
- package/dist/core/tokenization/char-classifiers.d.ts +56 -0
- package/dist/core/tokenization/char-classifiers.d.ts.map +1 -0
- package/dist/core/tokenization/default-extractors.d.ts +48 -0
- package/dist/core/tokenization/default-extractors.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/index.d.ts +9 -0
- package/dist/core/tokenization/extractors/index.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/operator.d.ts +23 -0
- package/dist/core/tokenization/extractors/operator.d.ts.map +1 -0
- package/dist/core/tokenization/extractors/punctuation.d.ts +22 -0
- package/dist/core/tokenization/extractors/punctuation.d.ts.map +1 -0
- package/dist/core/tokenization/extractors.d.ts +61 -0
- package/dist/core/tokenization/extractors.d.ts.map +1 -0
- package/dist/core/tokenization/index.d.ts +11 -0
- package/dist/core/tokenization/index.d.ts.map +1 -0
- package/dist/core/tokenization/index.js +1345 -0
- package/dist/core/tokenization/index.js.map +1 -0
- package/dist/core/tokenization/morphology/index.d.ts +5 -0
- package/dist/core/tokenization/morphology/index.d.ts.map +1 -0
- package/dist/core/tokenization/morphology/types.d.ts +110 -0
- package/dist/core/tokenization/morphology/types.d.ts.map +1 -0
- package/dist/core/tokenization/token-utils.d.ts +111 -0
- package/dist/core/tokenization/token-utils.d.ts.map +1 -0
- package/dist/core/types.d.ts +382 -0
- package/dist/core/types.d.ts.map +1 -0
- package/dist/core/types.js +108 -0
- package/dist/core/types.js.map +1 -0
- package/dist/generation/diagnostics.d.ts +120 -0
- package/dist/generation/diagnostics.d.ts.map +1 -0
- package/dist/generation/index.d.ts +7 -0
- package/dist/generation/index.d.ts.map +1 -0
- package/dist/generation/index.js +339 -0
- package/dist/generation/index.js.map +1 -0
- package/dist/generation/pattern-generator.d.ts +48 -0
- package/dist/generation/pattern-generator.d.ts.map +1 -0
- package/dist/generation/renderer.d.ts +115 -0
- package/dist/generation/renderer.d.ts.map +1 -0
- package/dist/grammar/index.d.ts +10 -0
- package/dist/grammar/index.d.ts.map +1 -0
- package/dist/grammar/index.js +391 -0
- package/dist/grammar/index.js.map +1 -0
- package/dist/grammar/transformer.d.ts +56 -0
- package/dist/grammar/transformer.d.ts.map +1 -0
- package/dist/grammar/types.d.ts +236 -0
- package/dist/grammar/types.d.ts.map +1 -0
- package/dist/index.cjs +4454 -0
- package/dist/index.cjs.map +1 -0
- package/dist/index.d.ts +46 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +4336 -0
- package/dist/index.js.map +1 -0
- package/dist/interfaces/dictionary.d.ts +82 -0
- package/dist/interfaces/dictionary.d.ts.map +1 -0
- package/dist/interfaces/index.d.ts +10 -0
- package/dist/interfaces/index.d.ts.map +1 -0
- package/dist/interfaces/profile-provider.d.ts +67 -0
- package/dist/interfaces/profile-provider.d.ts.map +1 -0
- package/dist/interfaces/value-extractor.d.ts +168 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -0
- package/dist/multilingual/index.d.ts +8 -0
- package/dist/multilingual/index.d.ts.map +1 -0
- package/dist/multilingual/index.js +1 -0
- package/dist/multilingual/index.js.map +1 -0
- package/dist/parsing/index.d.ts +8 -0
- package/dist/parsing/index.d.ts.map +1 -0
- package/dist/parsing/index.js +1415 -0
- package/dist/parsing/index.js.map +1 -0
- package/dist/parsing/multi-statement.d.ts +265 -0
- package/dist/parsing/multi-statement.d.ts.map +1 -0
- package/dist/schema/command-schema.d.ts +78 -0
- package/dist/schema/command-schema.d.ts.map +1 -0
- package/dist/schema/index.d.ts +5 -0
- package/dist/schema/index.d.ts.map +1 -0
- package/dist/schema/index.js +25 -0
- package/dist/schema/index.js.map +1 -0
- package/dist/test-setup.d.ts +9 -0
- package/dist/test-setup.d.ts.map +1 -0
- package/dist/testing/index.d.ts +50 -0
- package/dist/testing/index.d.ts.map +1 -0
- package/dist/testing/index.js +16969 -0
- package/dist/testing/index.js.map +1 -0
- package/package.json +122 -0
- package/src/__test__/fixtures/sql-dsl.ts +232 -0
- package/src/__test__/sql-integration.test.ts +189 -0
- package/src/__test__/test-utils.ts +260 -0
- package/src/aot/aot-orchestrator.test.ts +413 -0
- package/src/aot/aot-orchestrator.ts +238 -0
- package/src/aot/domain-scanner.ts +178 -0
- package/src/aot/index.ts +8 -0
- package/src/aot/types.ts +124 -0
- package/src/api/create-dsl.ts +367 -0
- package/src/api/dispatcher.test.ts +336 -0
- package/src/api/dispatcher.ts +222 -0
- package/src/api/domain-registry.test.ts +336 -0
- package/src/api/domain-registry.ts +500 -0
- package/src/api/index.ts +7 -0
- package/src/core/index.ts +7 -0
- package/src/core/logger.ts +130 -0
- package/src/core/pattern-matching/index.ts +6 -0
- package/src/core/pattern-matching/pattern-matcher.test.ts +900 -0
- package/src/core/pattern-matching/pattern-matcher.ts +1548 -0
- package/src/core/pattern-matching/pattern-matcher.ts.backup +1267 -0
- package/src/core/pattern-matching/utils/index.ts +6 -0
- package/src/core/pattern-matching/utils/possessive-keywords.ts +55 -0
- package/src/core/pattern-matching/utils/type-validation.test.ts +316 -0
- package/src/core/pattern-matching/utils/type-validation.ts +134 -0
- package/src/core/tokenization/base-tokenizer.ts +916 -0
- package/src/core/tokenization/char-classifiers.ts +79 -0
- package/src/core/tokenization/create-simple-tokenizer.test.ts +260 -0
- package/src/core/tokenization/default-extractors.ts +69 -0
- package/src/core/tokenization/extractors/index.ts +9 -0
- package/src/core/tokenization/extractors/operator.ts +75 -0
- package/src/core/tokenization/extractors/punctuation.ts +39 -0
- package/src/core/tokenization/extractors.ts +452 -0
- package/src/core/tokenization/index.ts +11 -0
- package/src/core/tokenization/morphology/index.ts +5 -0
- package/src/core/tokenization/morphology/types.ts +211 -0
- package/src/core/tokenization/token-utils.ts +252 -0
- package/src/core/types.ts +589 -0
- package/src/generation/diagnostics.test.ts +171 -0
- package/src/generation/diagnostics.ts +239 -0
- package/src/generation/index.ts +7 -0
- package/src/generation/pattern-generator.test.ts +430 -0
- package/src/generation/pattern-generator.ts +315 -0
- package/src/generation/renderer.test.ts +266 -0
- package/src/generation/renderer.ts +244 -0
- package/src/grammar/index.ts +12 -0
- package/src/grammar/transformer.ts +159 -0
- package/src/grammar/types.ts +630 -0
- package/src/index.ts +157 -0
- package/src/interfaces/dictionary.ts +123 -0
- package/src/interfaces/index.ts +10 -0
- package/src/interfaces/profile-provider.ts +88 -0
- package/src/interfaces/value-extractor.ts +435 -0
- package/src/multilingual/index.ts +9 -0
- package/src/parsing/index.ts +27 -0
- package/src/parsing/multi-statement.test.ts +480 -0
- package/src/parsing/multi-statement.ts +648 -0
- package/src/schema/command-schema.ts +118 -0
- package/src/schema/index.ts +5 -0
- package/src/test-setup.ts +45 -0
- package/src/testing/index.ts +137 -0
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Character Classifiers
|
|
3
|
+
*
|
|
4
|
+
* Unicode range classification and Latin character classifier factories.
|
|
5
|
+
* Used by language-specific tokenizers to define character sets.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
// =============================================================================
|
|
9
|
+
// Unicode Range Classification
|
|
10
|
+
// =============================================================================
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* Unicode range tuple: [start, end] (inclusive).
|
|
14
|
+
*/
|
|
15
|
+
export type UnicodeRange = readonly [number, number];
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Create a character classifier for Unicode ranges.
|
|
19
|
+
* Returns a function that checks if a character's code point falls within any of the ranges.
|
|
20
|
+
*
|
|
21
|
+
* @example
|
|
22
|
+
* // Japanese Hiragana
|
|
23
|
+
* const isHiragana = createUnicodeRangeClassifier([[0x3040, 0x309f]]);
|
|
24
|
+
*
|
|
25
|
+
* // Korean (Hangul syllables + Jamo)
|
|
26
|
+
* const isKorean = createUnicodeRangeClassifier([
|
|
27
|
+
* [0xac00, 0xd7a3], // Hangul syllables
|
|
28
|
+
* [0x1100, 0x11ff], // Hangul Jamo
|
|
29
|
+
* [0x3130, 0x318f], // Hangul Compatibility Jamo
|
|
30
|
+
* ]);
|
|
31
|
+
*/
|
|
32
|
+
export function createUnicodeRangeClassifier(
|
|
33
|
+
ranges: readonly UnicodeRange[]
|
|
34
|
+
): (char: string) => boolean {
|
|
35
|
+
return (char: string): boolean => {
|
|
36
|
+
const code = char.charCodeAt(0);
|
|
37
|
+
return ranges.some(([start, end]) => code >= start && code <= end);
|
|
38
|
+
};
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Combine multiple character classifiers into one.
|
|
43
|
+
* Returns true if any of the classifiers return true.
|
|
44
|
+
*
|
|
45
|
+
* @example
|
|
46
|
+
* const isJapanese = combineClassifiers(isHiragana, isKatakana, isKanji);
|
|
47
|
+
*/
|
|
48
|
+
export function combineClassifiers(
|
|
49
|
+
...classifiers: Array<(char: string) => boolean>
|
|
50
|
+
): (char: string) => boolean {
|
|
51
|
+
return (char: string): boolean => classifiers.some(fn => fn(char));
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Character classifiers for a Latin-based language.
|
|
56
|
+
*/
|
|
57
|
+
export interface LatinCharClassifiers {
|
|
58
|
+
/** Check if character is a letter in this language (including accented chars). */
|
|
59
|
+
isLetter: (char: string) => boolean;
|
|
60
|
+
/** Check if character is part of an identifier (letter, digit, underscore, hyphen). */
|
|
61
|
+
isIdentifierChar: (char: string) => boolean;
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Create character classifiers for a Latin-based language.
|
|
66
|
+
* Returns isLetter and isIdentifierChar functions based on the provided regex.
|
|
67
|
+
*
|
|
68
|
+
* @example
|
|
69
|
+
* // Spanish letters
|
|
70
|
+
* const { isLetter, isIdentifierChar } = createLatinCharClassifiers(/[a-zA-ZáéíóúüñÁÉÍÓÚÜÑ]/);
|
|
71
|
+
*
|
|
72
|
+
* // German letters
|
|
73
|
+
* const { isLetter, isIdentifierChar } = createLatinCharClassifiers(/[a-zA-ZäöüÄÖÜß]/);
|
|
74
|
+
*/
|
|
75
|
+
export function createLatinCharClassifiers(letterPattern: RegExp): LatinCharClassifiers {
|
|
76
|
+
const isLetter = (char: string): boolean => letterPattern.test(char);
|
|
77
|
+
const isIdentifierChar = (char: string): boolean => isLetter(char) || /[0-9_-]/.test(char);
|
|
78
|
+
return { isLetter, isIdentifierChar };
|
|
79
|
+
}
|
|
@@ -0,0 +1,260 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Unit tests for createSimpleTokenizer factory function.
|
|
3
|
+
*/
|
|
4
|
+
|
|
5
|
+
import { describe, it, expect } from 'vitest';
|
|
6
|
+
import { createSimpleTokenizer } from './base-tokenizer';
|
|
7
|
+
import type { SimpleTokenizerConfig } from './base-tokenizer';
|
|
8
|
+
import type { ValueExtractor, ExtractionResult } from '../../interfaces/value-extractor';
|
|
9
|
+
|
|
10
|
+
// =============================================================================
|
|
11
|
+
// Helpers
|
|
12
|
+
// =============================================================================
|
|
13
|
+
|
|
14
|
+
function makeTokenizer(overrides: Partial<SimpleTokenizerConfig> = {}) {
|
|
15
|
+
return createSimpleTokenizer({
|
|
16
|
+
language: 'en',
|
|
17
|
+
keywords: ['select', 'from', 'where'],
|
|
18
|
+
...overrides,
|
|
19
|
+
});
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/** Collect all tokens from a tokenizer run as [value, kind] pairs */
|
|
23
|
+
function tokenize(input: string, config?: Partial<SimpleTokenizerConfig>) {
|
|
24
|
+
const t = makeTokenizer(config);
|
|
25
|
+
const stream = t.tokenize(input);
|
|
26
|
+
const tokens: Array<{ value: string; kind: string }> = [];
|
|
27
|
+
while (!stream.isAtEnd()) {
|
|
28
|
+
const tok = stream.advance();
|
|
29
|
+
tokens.push({ value: tok.value, kind: tok.kind });
|
|
30
|
+
}
|
|
31
|
+
return tokens;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
// =============================================================================
|
|
35
|
+
// Tests
|
|
36
|
+
// =============================================================================
|
|
37
|
+
|
|
38
|
+
describe('createSimpleTokenizer', () => {
|
|
39
|
+
// ---------------------------------------------------------------------------
|
|
40
|
+
// Keyword classification
|
|
41
|
+
// ---------------------------------------------------------------------------
|
|
42
|
+
describe('keyword classification', () => {
|
|
43
|
+
it('recognizes keywords case-insensitively by default', () => {
|
|
44
|
+
const t = makeTokenizer();
|
|
45
|
+
expect(t.classifyToken('select')).toBe('keyword');
|
|
46
|
+
expect(t.classifyToken('SELECT')).toBe('keyword');
|
|
47
|
+
expect(t.classifyToken('Select')).toBe('keyword');
|
|
48
|
+
});
|
|
49
|
+
|
|
50
|
+
it('recognizes keywords case-sensitively when caseInsensitive: false', () => {
|
|
51
|
+
const t = makeTokenizer({ caseInsensitive: false });
|
|
52
|
+
expect(t.classifyToken('select')).toBe('keyword');
|
|
53
|
+
expect(t.classifyToken('SELECT')).toBe('identifier');
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
it('classifies non-keywords as identifier', () => {
|
|
57
|
+
const t = makeTokenizer();
|
|
58
|
+
expect(t.classifyToken('users')).toBe('identifier');
|
|
59
|
+
expect(t.classifyToken('name')).toBe('identifier');
|
|
60
|
+
});
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
// ---------------------------------------------------------------------------
|
|
64
|
+
// Operator classification
|
|
65
|
+
// ---------------------------------------------------------------------------
|
|
66
|
+
describe('operator classification', () => {
|
|
67
|
+
it('classifies operators when includeOperators: true', () => {
|
|
68
|
+
const t = makeTokenizer({ includeOperators: true });
|
|
69
|
+
expect(t.classifyToken('=')).toBe('operator');
|
|
70
|
+
expect(t.classifyToken('+')).toBe('operator');
|
|
71
|
+
expect(t.classifyToken('>')).toBe('operator');
|
|
72
|
+
});
|
|
73
|
+
|
|
74
|
+
it('classifies operators as identifier when includeOperators: false', () => {
|
|
75
|
+
const t = makeTokenizer({ includeOperators: false });
|
|
76
|
+
expect(t.classifyToken('=')).toBe('identifier');
|
|
77
|
+
expect(t.classifyToken('+')).toBe('identifier');
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
it('classifies multi-char operators', () => {
|
|
81
|
+
const t = makeTokenizer({ includeOperators: true });
|
|
82
|
+
expect(t.classifyToken('>=')).toBe('operator');
|
|
83
|
+
expect(t.classifyToken('!=')).toBe('operator');
|
|
84
|
+
expect(t.classifyToken('==')).toBe('operator');
|
|
85
|
+
});
|
|
86
|
+
|
|
87
|
+
it('classifies broader operators from DEFAULT_OPERATORS', () => {
|
|
88
|
+
const t = makeTokenizer({ includeOperators: true });
|
|
89
|
+
expect(t.classifyToken('&&')).toBe('operator');
|
|
90
|
+
expect(t.classifyToken('||')).toBe('operator');
|
|
91
|
+
expect(t.classifyToken('%')).toBe('operator');
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
it('defaults includeOperators to false', () => {
|
|
95
|
+
const t = makeTokenizer();
|
|
96
|
+
expect(t.classifyToken('+')).toBe('identifier');
|
|
97
|
+
});
|
|
98
|
+
});
|
|
99
|
+
|
|
100
|
+
// ---------------------------------------------------------------------------
|
|
101
|
+
// Literal classification
|
|
102
|
+
// ---------------------------------------------------------------------------
|
|
103
|
+
describe('literal classification', () => {
|
|
104
|
+
it('classifies digit-starting tokens as literal', () => {
|
|
105
|
+
const t = makeTokenizer();
|
|
106
|
+
expect(t.classifyToken('42')).toBe('literal');
|
|
107
|
+
expect(t.classifyToken('3.14')).toBe('literal');
|
|
108
|
+
});
|
|
109
|
+
|
|
110
|
+
it('classifies quote-starting tokens as literal', () => {
|
|
111
|
+
const t = makeTokenizer();
|
|
112
|
+
expect(t.classifyToken("'hello'")).toBe('literal');
|
|
113
|
+
expect(t.classifyToken('"world"')).toBe('literal');
|
|
114
|
+
});
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
// ---------------------------------------------------------------------------
|
|
118
|
+
// Full tokenization
|
|
119
|
+
// ---------------------------------------------------------------------------
|
|
120
|
+
describe('tokenization', () => {
|
|
121
|
+
it('tokenizes simple input into keyword and identifier tokens', () => {
|
|
122
|
+
const tokens = tokenize('select name from users');
|
|
123
|
+
expect(tokens).toEqual([
|
|
124
|
+
{ value: 'select', kind: 'keyword' },
|
|
125
|
+
{ value: 'name', kind: 'identifier' },
|
|
126
|
+
{ value: 'from', kind: 'keyword' },
|
|
127
|
+
{ value: 'users', kind: 'identifier' },
|
|
128
|
+
]);
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
it('skips whitespace between tokens', () => {
|
|
132
|
+
const tokens = tokenize('select name');
|
|
133
|
+
expect(tokens.map(t => t.value)).toEqual(['select', 'name']);
|
|
134
|
+
});
|
|
135
|
+
|
|
136
|
+
it('tokenizes string literals', () => {
|
|
137
|
+
const tokens = tokenize("where 'hello'");
|
|
138
|
+
const kinds = tokens.map(t => t.kind);
|
|
139
|
+
expect(kinds).toContain('keyword'); // where
|
|
140
|
+
expect(kinds).toContain('literal'); // 'hello'
|
|
141
|
+
});
|
|
142
|
+
|
|
143
|
+
it('tokenizes numbers', () => {
|
|
144
|
+
const tokens = tokenize('select 42');
|
|
145
|
+
expect(tokens[1]).toMatchObject({ value: '42', kind: 'literal' });
|
|
146
|
+
});
|
|
147
|
+
|
|
148
|
+
it('produces tokens with positions', () => {
|
|
149
|
+
const t = makeTokenizer();
|
|
150
|
+
const stream = t.tokenize('select name');
|
|
151
|
+
const first = stream.advance();
|
|
152
|
+
expect(first).toBeDefined();
|
|
153
|
+
expect(first.position.start).toBe(0);
|
|
154
|
+
expect(first.position.end).toBe(6);
|
|
155
|
+
});
|
|
156
|
+
});
|
|
157
|
+
|
|
158
|
+
// ---------------------------------------------------------------------------
|
|
159
|
+
// Profile-based keywords (non-Latin)
|
|
160
|
+
// ---------------------------------------------------------------------------
|
|
161
|
+
describe('profile-based keywords', () => {
|
|
162
|
+
const jaConfig: Partial<SimpleTokenizerConfig> = {
|
|
163
|
+
language: 'ja',
|
|
164
|
+
keywords: ['選択', 'から', '条件'],
|
|
165
|
+
keywordExtras: [
|
|
166
|
+
{ native: '選択', normalized: 'select' },
|
|
167
|
+
{ native: 'から', normalized: 'from' },
|
|
168
|
+
{ native: '条件', normalized: 'where' },
|
|
169
|
+
],
|
|
170
|
+
keywordProfile: {
|
|
171
|
+
keywords: {
|
|
172
|
+
select: { primary: '選択' },
|
|
173
|
+
from: { primary: 'から' },
|
|
174
|
+
where: { primary: '条件' },
|
|
175
|
+
},
|
|
176
|
+
},
|
|
177
|
+
caseInsensitive: false,
|
|
178
|
+
};
|
|
179
|
+
|
|
180
|
+
it('recognizes non-Latin keywords from config.keywords', () => {
|
|
181
|
+
const t = makeTokenizer(jaConfig);
|
|
182
|
+
expect(t.classifyToken('選択')).toBe('keyword');
|
|
183
|
+
expect(t.classifyToken('から')).toBe('keyword');
|
|
184
|
+
});
|
|
185
|
+
|
|
186
|
+
it('recognizes keywords via profile path (isKeyword)', () => {
|
|
187
|
+
// Create a tokenizer where a keyword is ONLY in the profile, not in keywords[]
|
|
188
|
+
const t = makeTokenizer({
|
|
189
|
+
language: 'ja',
|
|
190
|
+
keywords: [], // empty — no fast-path keywords
|
|
191
|
+
keywordProfile: {
|
|
192
|
+
keywords: {
|
|
193
|
+
select: { primary: '選択' },
|
|
194
|
+
},
|
|
195
|
+
},
|
|
196
|
+
caseInsensitive: false,
|
|
197
|
+
});
|
|
198
|
+
expect(t.classifyToken('選択')).toBe('keyword');
|
|
199
|
+
});
|
|
200
|
+
|
|
201
|
+
it('classifies unknown tokens as identifier', () => {
|
|
202
|
+
const t = makeTokenizer(jaConfig);
|
|
203
|
+
expect(t.classifyToken('ユーザー')).toBe('identifier');
|
|
204
|
+
});
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
// ---------------------------------------------------------------------------
|
|
208
|
+
// Custom extractors
|
|
209
|
+
// ---------------------------------------------------------------------------
|
|
210
|
+
describe('custom extractors', () => {
|
|
211
|
+
it('custom extractors take priority over defaults', () => {
|
|
212
|
+
// Custom extractor that captures @-mentions
|
|
213
|
+
const mentionExtractor: ValueExtractor = {
|
|
214
|
+
name: 'mention',
|
|
215
|
+
canExtract(input: string, pos: number) {
|
|
216
|
+
return input[pos] === '@';
|
|
217
|
+
},
|
|
218
|
+
extract(input: string, pos: number): ExtractionResult | null {
|
|
219
|
+
let end = pos + 1;
|
|
220
|
+
while (end < input.length && /\w/.test(input[end])) end++;
|
|
221
|
+
if (end === pos + 1) return null;
|
|
222
|
+
return { value: input.slice(pos, end), length: end - pos };
|
|
223
|
+
},
|
|
224
|
+
};
|
|
225
|
+
|
|
226
|
+
const tokens = tokenize('select @user', {
|
|
227
|
+
customExtractors: [mentionExtractor],
|
|
228
|
+
});
|
|
229
|
+
|
|
230
|
+
expect(tokens[0]).toMatchObject({ value: 'select', kind: 'keyword' });
|
|
231
|
+
// @user extracted by custom extractor before default identifier extractor
|
|
232
|
+
expect(tokens[1]).toMatchObject({ value: '@user' });
|
|
233
|
+
});
|
|
234
|
+
});
|
|
235
|
+
|
|
236
|
+
// ---------------------------------------------------------------------------
|
|
237
|
+
// Direction
|
|
238
|
+
// ---------------------------------------------------------------------------
|
|
239
|
+
describe('direction', () => {
|
|
240
|
+
it('defaults to ltr', () => {
|
|
241
|
+
const t = makeTokenizer();
|
|
242
|
+
expect(t.direction).toBe('ltr');
|
|
243
|
+
});
|
|
244
|
+
|
|
245
|
+
it('can be set to rtl', () => {
|
|
246
|
+
const t = makeTokenizer({ direction: 'rtl' });
|
|
247
|
+
expect(t.direction).toBe('rtl');
|
|
248
|
+
});
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
// ---------------------------------------------------------------------------
|
|
252
|
+
// Language property
|
|
253
|
+
// ---------------------------------------------------------------------------
|
|
254
|
+
describe('language', () => {
|
|
255
|
+
it('exposes the configured language', () => {
|
|
256
|
+
const t = makeTokenizer({ language: 'es' });
|
|
257
|
+
expect(t.language).toBe('es');
|
|
258
|
+
});
|
|
259
|
+
});
|
|
260
|
+
});
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Default Extractor Sets
|
|
3
|
+
*
|
|
4
|
+
* Provides pre-configured sets of extractors for common use cases.
|
|
5
|
+
* DSLs can use these as a starting point and add domain-specific extractors.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import type { ValueExtractor } from '../../interfaces/value-extractor';
|
|
9
|
+
import {
|
|
10
|
+
StringLiteralExtractor,
|
|
11
|
+
NumberExtractor,
|
|
12
|
+
IdentifierExtractor,
|
|
13
|
+
UnicodeIdentifierExtractor,
|
|
14
|
+
} from '../../interfaces/value-extractor';
|
|
15
|
+
import { OperatorExtractor, PunctuationExtractor } from './extractors/index';
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Get default extractors for generic programming-language-style DSLs.
|
|
19
|
+
* These work for most DSLs (SQL, config files, scripts, etc.).
|
|
20
|
+
*
|
|
21
|
+
* Included extractors:
|
|
22
|
+
* - String literals: "double", 'single', `backtick`
|
|
23
|
+
* - Numbers: 123, 45.67
|
|
24
|
+
* - Operators: +, -, *, /, =, ==, !=, >=, <=, etc.
|
|
25
|
+
* - Punctuation: ( ) [ ] { } , : ;
|
|
26
|
+
* - Identifiers: variable_names, functionNames
|
|
27
|
+
* - Unicode identifiers: CJK, Arabic, Cyrillic, etc.
|
|
28
|
+
*
|
|
29
|
+
* @returns Array of default extractors
|
|
30
|
+
*
|
|
31
|
+
* @example
|
|
32
|
+
* ```typescript
|
|
33
|
+
* class MyDSLTokenizer extends BaseTokenizer {
|
|
34
|
+
* constructor() {
|
|
35
|
+
* super();
|
|
36
|
+
* this.registerExtractors(getDefaultExtractors());
|
|
37
|
+
* }
|
|
38
|
+
* }
|
|
39
|
+
* ```
|
|
40
|
+
*/
|
|
41
|
+
export function getDefaultExtractors(): ValueExtractor[] {
|
|
42
|
+
return [
|
|
43
|
+
new StringLiteralExtractor(), // "strings", 'strings', `strings`
|
|
44
|
+
new NumberExtractor(), // 123, 45.67
|
|
45
|
+
new OperatorExtractor(), // +, -, *, /, =, >, <, etc.
|
|
46
|
+
new PunctuationExtractor(), // ( ) [ ] { } , : ;
|
|
47
|
+
new IdentifierExtractor(), // variable_names, functionNames (ASCII)
|
|
48
|
+
new UnicodeIdentifierExtractor(), // CJK, Arabic, Cyrillic, etc.
|
|
49
|
+
];
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Auto-register default extractors in a tokenizer.
|
|
54
|
+
* Convenience helper for chaining.
|
|
55
|
+
*
|
|
56
|
+
* @param tokenizer - Tokenizer to configure
|
|
57
|
+
* @returns The same tokenizer (for chaining)
|
|
58
|
+
*
|
|
59
|
+
* @example
|
|
60
|
+
* ```typescript
|
|
61
|
+
* const tokenizer = withDefaultExtractors(new MyTokenizer());
|
|
62
|
+
* ```
|
|
63
|
+
*/
|
|
64
|
+
export function withDefaultExtractors<
|
|
65
|
+
T extends { registerExtractors(extractors: ValueExtractor[]): void },
|
|
66
|
+
>(tokenizer: T): T {
|
|
67
|
+
tokenizer.registerExtractors(getDefaultExtractors());
|
|
68
|
+
return tokenizer;
|
|
69
|
+
}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Generic Value Extractors
|
|
3
|
+
*
|
|
4
|
+
* These extractors work for most programming-language-style DSLs.
|
|
5
|
+
* DSLs can use these as-is or provide custom extractors for domain-specific syntax.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
export { OperatorExtractor, DEFAULT_OPERATORS } from './operator';
|
|
9
|
+
export { PunctuationExtractor, DEFAULT_PUNCTUATION } from './punctuation';
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Operator Extractor - Handles programming language operators
|
|
3
|
+
*
|
|
4
|
+
* Extracts operators like +, -, *, /, =, >, <, >=, <=, !=, ===, etc.
|
|
5
|
+
* Supports multi-character operators with longest-match priority.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import type { ValueExtractor, ExtractionResult } from '../../../interfaces/value-extractor';
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Default operators for most programming languages.
|
|
12
|
+
* Sorted longest-first for greedy matching.
|
|
13
|
+
*/
|
|
14
|
+
export const DEFAULT_OPERATORS = [
|
|
15
|
+
// Three-character operators
|
|
16
|
+
'===',
|
|
17
|
+
'!==',
|
|
18
|
+
'->',
|
|
19
|
+
// Two-character operators
|
|
20
|
+
'==',
|
|
21
|
+
'!=',
|
|
22
|
+
'<=',
|
|
23
|
+
'>=',
|
|
24
|
+
'&&',
|
|
25
|
+
'||',
|
|
26
|
+
'**',
|
|
27
|
+
'+=',
|
|
28
|
+
'-=',
|
|
29
|
+
'*=',
|
|
30
|
+
'/=',
|
|
31
|
+
// Single-character operators
|
|
32
|
+
'+',
|
|
33
|
+
'-',
|
|
34
|
+
'*',
|
|
35
|
+
'/',
|
|
36
|
+
'=',
|
|
37
|
+
'>',
|
|
38
|
+
'<',
|
|
39
|
+
'!',
|
|
40
|
+
'&',
|
|
41
|
+
'|',
|
|
42
|
+
'%',
|
|
43
|
+
'^',
|
|
44
|
+
'~',
|
|
45
|
+
];
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* OperatorExtractor - Extracts programming language operators.
|
|
49
|
+
*/
|
|
50
|
+
export class OperatorExtractor implements ValueExtractor {
|
|
51
|
+
readonly name = 'operator';
|
|
52
|
+
|
|
53
|
+
constructor(private operators: string[] = DEFAULT_OPERATORS) {
|
|
54
|
+
// Sort operators longest-first for greedy matching
|
|
55
|
+
this.operators = [...operators].sort((a, b) => b.length - a.length);
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
canExtract(input: string, position: number): boolean {
|
|
59
|
+
return this.operators.some(op => input.startsWith(op, position));
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
extract(input: string, position: number): ExtractionResult | null {
|
|
63
|
+
// Find longest matching operator
|
|
64
|
+
for (const op of this.operators) {
|
|
65
|
+
if (input.startsWith(op, position)) {
|
|
66
|
+
return {
|
|
67
|
+
value: op,
|
|
68
|
+
length: op.length,
|
|
69
|
+
};
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
return null;
|
|
74
|
+
}
|
|
75
|
+
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Punctuation Extractor - Handles punctuation characters
|
|
3
|
+
*
|
|
4
|
+
* Extracts punctuation like parentheses, brackets, braces, commas, colons, semicolons.
|
|
5
|
+
* Each character is extracted individually (no multi-character punctuation).
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import type { ValueExtractor, ExtractionResult } from '../../../interfaces/value-extractor';
|
|
9
|
+
|
|
10
|
+
/**
|
|
11
|
+
* Default punctuation characters for most programming languages.
|
|
12
|
+
*/
|
|
13
|
+
export const DEFAULT_PUNCTUATION = '()[]{},:;';
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* PunctuationExtractor - Extracts punctuation characters.
|
|
17
|
+
*/
|
|
18
|
+
export class PunctuationExtractor implements ValueExtractor {
|
|
19
|
+
readonly name = 'punctuation';
|
|
20
|
+
|
|
21
|
+
constructor(private punctuation: string = DEFAULT_PUNCTUATION) {}
|
|
22
|
+
|
|
23
|
+
canExtract(input: string, position: number): boolean {
|
|
24
|
+
return this.punctuation.includes(input[position]);
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
extract(input: string, position: number): ExtractionResult | null {
|
|
28
|
+
const char = input[position];
|
|
29
|
+
|
|
30
|
+
if (this.punctuation.includes(char)) {
|
|
31
|
+
return {
|
|
32
|
+
value: char,
|
|
33
|
+
length: 1,
|
|
34
|
+
};
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
return null;
|
|
38
|
+
}
|
|
39
|
+
}
|