@lokascript/framework 2.8.0 → 2.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +393 -0
- package/dist/api/create-dsl.d.ts +93 -1
- package/dist/api/create-dsl.d.ts.map +1 -1
- package/dist/api/domain-registry.d.ts +5 -3
- package/dist/api/domain-registry.d.ts.map +1 -1
- package/dist/api/index.js +232 -18
- package/dist/api/index.js.map +1 -1
- package/dist/core/index.js +26 -6
- package/dist/core/index.js.map +1 -1
- package/dist/core/tokenization/base-tokenizer.d.ts +15 -3
- package/dist/core/tokenization/base-tokenizer.d.ts.map +1 -1
- package/dist/core/tokenization/char-classifiers.d.ts +2 -2
- package/dist/core/tokenization/index.js +26 -6
- package/dist/core/tokenization/index.js.map +1 -1
- package/dist/core/tokenization/token-utils.d.ts +15 -0
- package/dist/core/tokenization/token-utils.d.ts.map +1 -1
- package/dist/generation/index.js +73 -47
- package/dist/generation/index.js.map +1 -1
- package/dist/generation/pattern-generator.d.ts +8 -1
- package/dist/generation/pattern-generator.d.ts.map +1 -1
- package/dist/generation/renderer.d.ts +53 -1
- package/dist/generation/renderer.d.ts.map +1 -1
- package/dist/index.cjs +302 -115
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +298 -115
- package/dist/index.js.map +1 -1
- package/dist/interfaces/value-extractor.d.ts +5 -0
- package/dist/interfaces/value-extractor.d.ts.map +1 -1
- package/dist/multilingual/index.js +25 -6
- package/dist/multilingual/index.js.map +1 -1
- package/package.json +4 -3
- package/src/api/create-dsl.test.ts +11 -0
- package/src/api/create-dsl.ts +278 -9
- package/src/api/domain-registry.ts +15 -10
- package/src/api/extensions.test.ts +322 -0
- package/src/core/tokenization/base-tokenizer.ts +23 -8
- package/src/core/tokenization/char-classifiers.ts +2 -2
- package/src/core/tokenization/css-selector-extractor.test.ts +67 -0
- package/src/core/tokenization/token-utils.ts +18 -0
- package/src/generation/domain-renderer.test.ts +172 -0
- package/src/generation/pattern-generator.test.ts +102 -0
- package/src/generation/pattern-generator.ts +27 -19
- package/src/generation/renderer.test.ts +243 -4
- package/src/generation/renderer.ts +188 -45
- package/src/index.ts +9 -1
- package/src/interfaces/value-extractor.ts +50 -0
|
@@ -0,0 +1,322 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* DomainExtension: adding a command to a DSL from outside its package.
|
|
3
|
+
*
|
|
4
|
+
* Every domain used to render through a hardcoded `switch (node.action)` and
|
|
5
|
+
* generate through another, so a downstream consumer could not add a command
|
|
6
|
+
* without editing the package. This is the supported path: a schema plus one
|
|
7
|
+
* vocabulary entry per language.
|
|
8
|
+
*
|
|
9
|
+
* The command exercised here ("research") is the one lokascript-learn built by
|
|
10
|
+
* hand against 2.8.0 to prove the underlying pieces worked.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { describe, it, expect } from 'vitest';
|
|
14
|
+
import {
|
|
15
|
+
createMultilingualDSL,
|
|
16
|
+
createSimpleTokenizer,
|
|
17
|
+
defineCommand,
|
|
18
|
+
defineRole,
|
|
19
|
+
type DomainExtension,
|
|
20
|
+
type ExtractionResult,
|
|
21
|
+
type LanguageTokenizer,
|
|
22
|
+
type SemanticNode,
|
|
23
|
+
type ValueExtractor,
|
|
24
|
+
} from '../index';
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Keeps `#id` / `.class` a single token, as every real domain tokenizer does.
|
|
28
|
+
* Without it the `#` splits off and an SOV source role followed by a particle
|
|
29
|
+
* never matches.
|
|
30
|
+
*/
|
|
31
|
+
class CSSSelectorExtractor implements ValueExtractor {
|
|
32
|
+
readonly name = 'css-selector';
|
|
33
|
+
|
|
34
|
+
canExtract(input: string, position: number): boolean {
|
|
35
|
+
const char = input[position];
|
|
36
|
+
if (char !== '#' && char !== '.') return false;
|
|
37
|
+
return /[a-zA-Z_-]/.test(input[position + 1] ?? '');
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
extract(input: string, position: number): ExtractionResult | null {
|
|
41
|
+
let end = position + 1;
|
|
42
|
+
while (end < input.length && /[a-zA-Z0-9_-]/.test(input[end])) end++;
|
|
43
|
+
if (end === position + 1) return null;
|
|
44
|
+
return { value: input.slice(position, end), length: end - position };
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
// =============================================================================
|
|
49
|
+
// A three-word-order toy DSL
|
|
50
|
+
// =============================================================================
|
|
51
|
+
|
|
52
|
+
const askSchema = defineCommand({
|
|
53
|
+
action: 'ask',
|
|
54
|
+
description: 'Ask a question',
|
|
55
|
+
category: 'llm',
|
|
56
|
+
primaryRole: 'patient',
|
|
57
|
+
roles: [
|
|
58
|
+
defineRole({
|
|
59
|
+
role: 'patient',
|
|
60
|
+
description: 'The question',
|
|
61
|
+
required: true,
|
|
62
|
+
expectedTypes: ['expression'],
|
|
63
|
+
}),
|
|
64
|
+
defineRole({
|
|
65
|
+
role: 'source',
|
|
66
|
+
description: 'Where to look',
|
|
67
|
+
required: true,
|
|
68
|
+
expectedTypes: ['expression'],
|
|
69
|
+
markerOverride: { en: 'from', ja: 'から', ar: 'من' },
|
|
70
|
+
}),
|
|
71
|
+
],
|
|
72
|
+
});
|
|
73
|
+
|
|
74
|
+
const researchSchema = defineCommand({
|
|
75
|
+
action: 'research',
|
|
76
|
+
description: 'Research a topic',
|
|
77
|
+
category: 'llm',
|
|
78
|
+
primaryRole: 'patient',
|
|
79
|
+
roles: [
|
|
80
|
+
defineRole({
|
|
81
|
+
role: 'patient',
|
|
82
|
+
description: 'The topic',
|
|
83
|
+
required: true,
|
|
84
|
+
expectedTypes: ['expression'],
|
|
85
|
+
}),
|
|
86
|
+
defineRole({
|
|
87
|
+
role: 'source',
|
|
88
|
+
description: 'Where to look',
|
|
89
|
+
required: true,
|
|
90
|
+
expectedTypes: ['expression'],
|
|
91
|
+
markerOverride: { en: 'from', ja: 'から', ar: 'من' },
|
|
92
|
+
}),
|
|
93
|
+
],
|
|
94
|
+
});
|
|
95
|
+
|
|
96
|
+
const PROFILES = {
|
|
97
|
+
en: {
|
|
98
|
+
code: 'en',
|
|
99
|
+
wordOrder: 'SVO' as const,
|
|
100
|
+
keywords: { ask: { primary: 'ask' } },
|
|
101
|
+
roleMarkers: {},
|
|
102
|
+
},
|
|
103
|
+
ja: {
|
|
104
|
+
code: 'ja',
|
|
105
|
+
wordOrder: 'SOV' as const,
|
|
106
|
+
keywords: { ask: { primary: '聞く' } },
|
|
107
|
+
roleMarkers: {},
|
|
108
|
+
},
|
|
109
|
+
ar: {
|
|
110
|
+
code: 'ar',
|
|
111
|
+
wordOrder: 'VSO' as const,
|
|
112
|
+
keywords: { ask: { primary: 'اسأل' } },
|
|
113
|
+
roleMarkers: {},
|
|
114
|
+
},
|
|
115
|
+
};
|
|
116
|
+
|
|
117
|
+
const research: DomainExtension = {
|
|
118
|
+
schema: researchSchema,
|
|
119
|
+
vocabulary: {
|
|
120
|
+
en: { keyword: { primary: 'research' } },
|
|
121
|
+
ja: { keyword: { primary: '調査' } },
|
|
122
|
+
ar: { keyword: { primary: 'ابحث' } },
|
|
123
|
+
},
|
|
124
|
+
};
|
|
125
|
+
|
|
126
|
+
function tokenizerFor(code: 'en' | 'ja' | 'ar'): LanguageTokenizer {
|
|
127
|
+
return createSimpleTokenizer({
|
|
128
|
+
language: code,
|
|
129
|
+
// Both the built-in and the extension vocabulary: a tokenizer is configured
|
|
130
|
+
// once, so it must already know the words an extension may introduce.
|
|
131
|
+
keywords: [
|
|
132
|
+
...Object.values(PROFILES[code].keywords).map(k => k.primary),
|
|
133
|
+
'research',
|
|
134
|
+
'調査',
|
|
135
|
+
'ابحث',
|
|
136
|
+
'from',
|
|
137
|
+
'から',
|
|
138
|
+
'من',
|
|
139
|
+
],
|
|
140
|
+
caseInsensitive: code === 'en',
|
|
141
|
+
customExtractors: [new CSSSelectorExtractor()],
|
|
142
|
+
});
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
function createToyDSL(options: { extensions?: readonly DomainExtension[] } = {}) {
|
|
146
|
+
return createMultilingualDSL({
|
|
147
|
+
name: 'Toy',
|
|
148
|
+
schemas: [askSchema],
|
|
149
|
+
languages: (['en', 'ja', 'ar'] as const).map(code => ({
|
|
150
|
+
code,
|
|
151
|
+
name: code,
|
|
152
|
+
nativeName: code,
|
|
153
|
+
tokenizer: tokenizerFor(code),
|
|
154
|
+
patternProfile: PROFILES[code],
|
|
155
|
+
})),
|
|
156
|
+
codeGenerator: {
|
|
157
|
+
generate: (node: SemanticNode) => `BASE:${node.action}`,
|
|
158
|
+
},
|
|
159
|
+
...(options.extensions && { extensions: options.extensions }),
|
|
160
|
+
});
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
function makeNode(action: string, roles: Record<string, string>): SemanticNode {
|
|
164
|
+
const rolesMap = new Map<string, { type: 'expression'; raw: string }>();
|
|
165
|
+
for (const [k, v] of Object.entries(roles)) rolesMap.set(k, { type: 'expression', raw: v });
|
|
166
|
+
return { kind: 'command', action, roles: rolesMap };
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
// =============================================================================
|
|
170
|
+
// Tests
|
|
171
|
+
// =============================================================================
|
|
172
|
+
|
|
173
|
+
describe('DomainExtension', () => {
|
|
174
|
+
describe('parsing', () => {
|
|
175
|
+
it('parses the extension command in an SVO language', () => {
|
|
176
|
+
const dsl = createToyDSL({ extensions: [research] });
|
|
177
|
+
const node = dsl.parse('research "climate" from #wiki', 'en');
|
|
178
|
+
expect(node.action).toBe('research');
|
|
179
|
+
expect(node.roles.get('patient')).toBeDefined();
|
|
180
|
+
expect(node.roles.get('source')).toBeDefined();
|
|
181
|
+
});
|
|
182
|
+
|
|
183
|
+
it('parses the extension command in an SOV language', () => {
|
|
184
|
+
const dsl = createToyDSL({ extensions: [research] });
|
|
185
|
+
const node = dsl.parse('"climate" #wiki から 調査', 'ja');
|
|
186
|
+
expect(node.action).toBe('research');
|
|
187
|
+
});
|
|
188
|
+
|
|
189
|
+
it('parses the extension command in a VSO language', () => {
|
|
190
|
+
const dsl = createToyDSL({ extensions: [research] });
|
|
191
|
+
const node = dsl.parse('ابحث "climate" من #wiki', 'ar');
|
|
192
|
+
expect(node.action).toBe('research');
|
|
193
|
+
});
|
|
194
|
+
|
|
195
|
+
it('does not parse the extension command without the extension', () => {
|
|
196
|
+
const dsl = createToyDSL();
|
|
197
|
+
expect(() => dsl.parse('research "climate" from #wiki', 'en')).toThrow(/No pattern matched/);
|
|
198
|
+
});
|
|
199
|
+
|
|
200
|
+
it('reports the extension action as supported by explicit syntax', () => {
|
|
201
|
+
const dsl = createToyDSL({ extensions: [research] });
|
|
202
|
+
const node = dsl.parse('[research patient:climate source:#wiki]', 'en');
|
|
203
|
+
expect(node.action).toBe('research');
|
|
204
|
+
});
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
describe('rendering', () => {
|
|
208
|
+
const dsl = createToyDSL({ extensions: [research] });
|
|
209
|
+
const node = makeNode('research', { patient: '"climate"', source: '#wiki' });
|
|
210
|
+
|
|
211
|
+
it('renders SVO from the schema alone', () => {
|
|
212
|
+
expect(dsl.render?.(node, 'en')).toBe('research "climate" from #wiki');
|
|
213
|
+
});
|
|
214
|
+
|
|
215
|
+
it('renders verb-final for SOV', () => {
|
|
216
|
+
expect(dsl.render?.(node, 'ja')).toBe('"climate" #wiki から 調査');
|
|
217
|
+
});
|
|
218
|
+
|
|
219
|
+
it('renders verb-initial for VSO', () => {
|
|
220
|
+
expect(dsl.render?.(node, 'ar')).toBe('ابحث "climate" من #wiki');
|
|
221
|
+
});
|
|
222
|
+
|
|
223
|
+
it('prefers an extension-supplied renderer', () => {
|
|
224
|
+
const custom = createToyDSL({
|
|
225
|
+
extensions: [{ ...research, render: () => 'CUSTOM' }],
|
|
226
|
+
});
|
|
227
|
+
expect(custom.render?.(node, 'en')).toBe('CUSTOM');
|
|
228
|
+
});
|
|
229
|
+
|
|
230
|
+
it('returns null for an action with neither renderer nor schema', () => {
|
|
231
|
+
expect(dsl.render?.(makeNode('nonsense', {}), 'en')).toBeNull();
|
|
232
|
+
});
|
|
233
|
+
|
|
234
|
+
it('prefers the domain renderer over the schema fallback for built-ins', () => {
|
|
235
|
+
const withDomainRenderer = createMultilingualDSL({
|
|
236
|
+
name: 'Toy',
|
|
237
|
+
schemas: [askSchema],
|
|
238
|
+
languages: [
|
|
239
|
+
{
|
|
240
|
+
code: 'en',
|
|
241
|
+
name: 'en',
|
|
242
|
+
nativeName: 'en',
|
|
243
|
+
tokenizer: tokenizerFor('en'),
|
|
244
|
+
patternProfile: PROFILES.en,
|
|
245
|
+
},
|
|
246
|
+
],
|
|
247
|
+
renderer: node => (node.action === 'ask' ? 'DOMAIN RENDERED' : null),
|
|
248
|
+
extensions: [{ ...research, vocabulary: { en: research.vocabulary.en } }],
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
expect(withDomainRenderer.render?.(makeNode('ask', { patient: 'q' }), 'en')).toBe(
|
|
252
|
+
'DOMAIN RENDERED'
|
|
253
|
+
);
|
|
254
|
+
// …while the extension still falls through to the schema renderer
|
|
255
|
+
expect(withDomainRenderer.render?.(node, 'en')).toBe('research "climate" from #wiki');
|
|
256
|
+
});
|
|
257
|
+
});
|
|
258
|
+
|
|
259
|
+
describe('compilation', () => {
|
|
260
|
+
it('uses the extension generator for the extension action', () => {
|
|
261
|
+
const dsl = createToyDSL({
|
|
262
|
+
extensions: [{ ...research, generate: node => `EXT:${node.action}` }],
|
|
263
|
+
});
|
|
264
|
+
const result = dsl.compile('research "climate" from #wiki', 'en');
|
|
265
|
+
expect(result.ok).toBe(true);
|
|
266
|
+
expect(result.code).toBe('EXT:research');
|
|
267
|
+
});
|
|
268
|
+
|
|
269
|
+
it('leaves the base generator handling built-in actions', () => {
|
|
270
|
+
const dsl = createToyDSL({
|
|
271
|
+
extensions: [{ ...research, generate: node => `EXT:${node.action}` }],
|
|
272
|
+
});
|
|
273
|
+
const result = dsl.compile('ask "q" from #wiki', 'en');
|
|
274
|
+
expect(result.ok).toBe(true);
|
|
275
|
+
expect(result.code).toBe('BASE:ask');
|
|
276
|
+
});
|
|
277
|
+
|
|
278
|
+
it('falls back to the base generator when the extension supplies none', () => {
|
|
279
|
+
const dsl = createToyDSL({ extensions: [research] });
|
|
280
|
+
const result = dsl.compile('research "climate" from #wiki', 'en');
|
|
281
|
+
expect(result.ok).toBe(true);
|
|
282
|
+
expect(result.code).toBe('BASE:research');
|
|
283
|
+
});
|
|
284
|
+
});
|
|
285
|
+
|
|
286
|
+
describe('built-in commands are unaffected', () => {
|
|
287
|
+
it('parses and compiles identically with and without extensions', () => {
|
|
288
|
+
const plain = createToyDSL();
|
|
289
|
+
const extended = createToyDSL({ extensions: [research] });
|
|
290
|
+
|
|
291
|
+
for (const [input, language] of [
|
|
292
|
+
['ask "q" from #wiki', 'en'],
|
|
293
|
+
['"q" #wiki から 聞く', 'ja'],
|
|
294
|
+
['اسأل "q" من #wiki', 'ar'],
|
|
295
|
+
] as const) {
|
|
296
|
+
expect(extended.parse(input, language).action).toBe(plain.parse(input, language).action);
|
|
297
|
+
expect(extended.compile(input, language).code).toBe(plain.compile(input, language).code);
|
|
298
|
+
}
|
|
299
|
+
});
|
|
300
|
+
});
|
|
301
|
+
|
|
302
|
+
describe('configuration errors', () => {
|
|
303
|
+
it('rejects an extension whose action collides with a built-in', () => {
|
|
304
|
+
const collision: DomainExtension = {
|
|
305
|
+
schema: defineCommand({
|
|
306
|
+
action: 'ask',
|
|
307
|
+
roles: [defineRole({ role: 'patient', required: true, expectedTypes: ['expression'] })],
|
|
308
|
+
}),
|
|
309
|
+
vocabulary: { en: { keyword: { primary: 'inquire' } } },
|
|
310
|
+
};
|
|
311
|
+
expect(() => createToyDSL({ extensions: [collision] })).toThrow(/collides/);
|
|
312
|
+
});
|
|
313
|
+
|
|
314
|
+
it('rejects vocabulary for a language the DSL does not configure', () => {
|
|
315
|
+
const unknownLanguage: DomainExtension = {
|
|
316
|
+
...research,
|
|
317
|
+
vocabulary: { ...research.vocabulary, xx: { keyword: { primary: 'x' } } },
|
|
318
|
+
};
|
|
319
|
+
expect(() => createToyDSL({ extensions: [unknownLanguage] })).toThrow(/not configured/);
|
|
320
|
+
});
|
|
321
|
+
});
|
|
322
|
+
});
|
|
@@ -20,6 +20,7 @@ import {
|
|
|
20
20
|
isWhitespace,
|
|
21
21
|
isDigit,
|
|
22
22
|
isAsciiIdentifierChar,
|
|
23
|
+
stripOptionalDiacritics,
|
|
23
24
|
TokenStreamImpl,
|
|
24
25
|
type TimeUnitMapping,
|
|
25
26
|
type CreateTokenOptions,
|
|
@@ -598,9 +599,7 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
598
599
|
* @returns Word without diacritics
|
|
599
600
|
*/
|
|
600
601
|
protected removeDiacritics(word: string): string {
|
|
601
|
-
|
|
602
|
-
// U+0670 (superscript alif)
|
|
603
|
-
return word.replace(/[\u064B-\u0652\u0670]/g, '');
|
|
602
|
+
return stripOptionalDiacritics(word);
|
|
604
603
|
}
|
|
605
604
|
|
|
606
605
|
/**
|
|
@@ -704,25 +703,41 @@ export abstract class BaseTokenizer implements LanguageTokenizer {
|
|
|
704
703
|
}
|
|
705
704
|
|
|
706
705
|
/**
|
|
707
|
-
* Look up a keyword by native word (case-insensitive).
|
|
706
|
+
* Look up a keyword by native word (case-insensitive, diacritic-insensitive).
|
|
708
707
|
* O(1) lookup using the keyword map.
|
|
709
708
|
*
|
|
709
|
+
* The map is INDEXED both with and without diacritics (see
|
|
710
|
+
* `initializeKeywordsFromProfile`), so a stripped QUERY is the other half of
|
|
711
|
+
* that: it lets a surface form carrying harakat the profile does not happen to
|
|
712
|
+
* spell still find its entry. Only consulted after the exact lookup misses, so
|
|
713
|
+
* every previously-matching word resolves byte-identically.
|
|
714
|
+
*
|
|
715
|
+
* Half-implementing this — indexing stripped but querying exact — is what made
|
|
716
|
+
* diacritized `بَدِّل` (toggle) tokenize as `kind=particle normalized=with`:
|
|
717
|
+
* `isKeyword` returned false, so the guard in `ArabicProcliticExtractor` that
|
|
718
|
+
* exists to prevent exactly that handed the word on, and the single-char `ب`
|
|
719
|
+
* bi- proclitic claimed it. A wrong CONCEPT, not a failed parse.
|
|
720
|
+
*
|
|
710
721
|
* @param native - Native word to look up
|
|
711
722
|
* @returns KeywordEntry if found, undefined otherwise
|
|
712
723
|
*/
|
|
713
724
|
protected lookupKeyword(native: string): KeywordEntry | undefined {
|
|
714
|
-
|
|
725
|
+
const exact = this.profileKeywordMap.get(native.toLowerCase());
|
|
726
|
+
if (exact) return exact;
|
|
727
|
+
const stripped = this.removeDiacritics(native);
|
|
728
|
+
if (stripped === native) return undefined;
|
|
729
|
+
return this.profileKeywordMap.get(stripped.toLowerCase());
|
|
715
730
|
}
|
|
716
731
|
|
|
717
732
|
/**
|
|
718
|
-
* Check if a word is a known keyword (case-insensitive).
|
|
719
|
-
* O(1) lookup using the keyword map.
|
|
733
|
+
* Check if a word is a known keyword (case-insensitive, diacritic-insensitive).
|
|
734
|
+
* O(1) lookup using the keyword map. See {@link lookupKeyword}.
|
|
720
735
|
*
|
|
721
736
|
* @param native - Native word to check
|
|
722
737
|
* @returns true if the word is a keyword
|
|
723
738
|
*/
|
|
724
739
|
protected isKeyword(native: string): boolean {
|
|
725
|
-
return this.
|
|
740
|
+
return this.lookupKeyword(native) !== undefined;
|
|
726
741
|
}
|
|
727
742
|
|
|
728
743
|
/**
|
|
@@ -67,10 +67,10 @@ export interface LatinCharClassifiers {
|
|
|
67
67
|
*
|
|
68
68
|
* @example
|
|
69
69
|
* // Spanish letters
|
|
70
|
-
* const { isLetter, isIdentifierChar } = createLatinCharClassifiers(/[a-zA-
|
|
70
|
+
* const { isLetter, isIdentifierChar } = createLatinCharClassifiers(/[a-zA-Z\u00e1\u00e9\u00ed\u00f3\u00fa\u00fc\u00f1\u00c1\u00c9\u00cd\u00d3\u00da\u00dc\u00d1]/);
|
|
71
71
|
*
|
|
72
72
|
* // German letters
|
|
73
|
-
* const { isLetter, isIdentifierChar } = createLatinCharClassifiers(/[a-zA-
|
|
73
|
+
* const { isLetter, isIdentifierChar } = createLatinCharClassifiers(/[a-zA-Z\u00e4\u00f6\u00fc\u00c4\u00d6\u00dc\u00df]/);
|
|
74
74
|
*/
|
|
75
75
|
export function createLatinCharClassifiers(letterPattern: RegExp): LatinCharClassifiers {
|
|
76
76
|
const isLetter = (char: string): boolean => letterPattern.test(char);
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import { describe, it, expect } from 'vitest';
|
|
2
|
+
import { CssSelectorExtractor } from '../../interfaces/value-extractor';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* `getDefaultExtractors()` has no CSS-selector extractor, so a DSL that does not
|
|
6
|
+
* register one splits the sigil off as its own token and the role capture keeps
|
|
7
|
+
* only that sigil — `add .active to #button` parsed with patient `"."` and
|
|
8
|
+
* destination `"#"` in domain-learn, -todo, -sql and -jsx, silently, in every
|
|
9
|
+
* language. Five other domains each carried a private copy of this class; this
|
|
10
|
+
* is the shared one.
|
|
11
|
+
*/
|
|
12
|
+
describe('CssSelectorExtractor', () => {
|
|
13
|
+
const ex = new CssSelectorExtractor();
|
|
14
|
+
|
|
15
|
+
const extract = (input: string, pos = 0) =>
|
|
16
|
+
ex.canExtract(input, pos) ? ex.extract(input, pos) : null;
|
|
17
|
+
|
|
18
|
+
it.each([
|
|
19
|
+
['.active', '.active'],
|
|
20
|
+
['#button', '#button'],
|
|
21
|
+
['.btn-primary', '.btn-primary'],
|
|
22
|
+
['._private', '._private'],
|
|
23
|
+
['#a1', '#a1'],
|
|
24
|
+
['.-leading-hyphen', '.-leading-hyphen'],
|
|
25
|
+
])('extracts %s whole', (input, expected) => {
|
|
26
|
+
expect(extract(input)).toEqual({ value: expected, length: expected.length });
|
|
27
|
+
});
|
|
28
|
+
|
|
29
|
+
it('stops at the first character that cannot be in a selector', () => {
|
|
30
|
+
expect(extract('.active to #button')).toEqual({ value: '.active', length: 7 });
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
it('keeps diacritics, so Latin-script class names survive', () => {
|
|
34
|
+
expect(extract('.año')).toEqual({ value: '.año', length: 4 });
|
|
35
|
+
expect(extract('#botón')).toEqual({ value: '#botón', length: 6 });
|
|
36
|
+
});
|
|
37
|
+
|
|
38
|
+
// The SOV languages write their particles flush against the value, so a
|
|
39
|
+
// Unicode-wide body turns `#buttonに` into a single token and carries the role
|
|
40
|
+
// marker inside the value — which is worse than truncating, because the marker
|
|
41
|
+
// is then missing from where the pattern expects it.
|
|
42
|
+
it.each([
|
|
43
|
+
['#buttonに', '#button'],
|
|
44
|
+
['.activeを', '.active'],
|
|
45
|
+
['#button에', '#button'],
|
|
46
|
+
['.active를', '.active'],
|
|
47
|
+
['#button的', '#button'],
|
|
48
|
+
])('stops at the particle in %s', (input, expected) => {
|
|
49
|
+
expect(extract(input)).toEqual({ value: expected, length: expected.length });
|
|
50
|
+
});
|
|
51
|
+
|
|
52
|
+
it.each([
|
|
53
|
+
['.', 'a bare sigil'],
|
|
54
|
+
['#', 'a bare sigil'],
|
|
55
|
+
['. active', 'a sigil followed by a space'],
|
|
56
|
+
['.1col', 'a digit — CSS identifiers cannot start with one'],
|
|
57
|
+
['button', 'no sigil at all'],
|
|
58
|
+
])('declines %s (%s)', input => {
|
|
59
|
+
expect(ex.canExtract(input, 0)).toBe(false);
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
it('declines a CJK-only class name rather than claiming the sigil alone', () => {
|
|
63
|
+
// `.追加` would otherwise extract as a bare `.`; leaving it unclaimed lets
|
|
64
|
+
// the keyword/identifier extractors see the verb.
|
|
65
|
+
expect(ex.canExtract('.追加', 0)).toBe(false);
|
|
66
|
+
});
|
|
67
|
+
});
|
|
@@ -237,6 +237,24 @@ export function isDigit(char: string): boolean {
|
|
|
237
237
|
return /\d/.test(char);
|
|
238
238
|
}
|
|
239
239
|
|
|
240
|
+
/**
|
|
241
|
+
* Strip diacritical marks that are OPTIONAL in their script — currently Arabic
|
|
242
|
+
* harakat (U+064B–U+0652: fatha, kasra, damma, sukun, shadda…) and the
|
|
243
|
+
* superscript alif (U+0670).
|
|
244
|
+
*
|
|
245
|
+
* `بدّل` and `بَدِّل` are the same word; Arabic prose writes either. So any place
|
|
246
|
+
* that compares an Arabic surface form against a declared keyword has to compare
|
|
247
|
+
* stripped, or the same word fails to match itself.
|
|
248
|
+
*
|
|
249
|
+
* Deliberately Arabic-only. Latin diacritics are NOT optional — `obtén`, `récupère`
|
|
250
|
+
* and `vá` differ from their unaccented spellings in meaning or validity — and
|
|
251
|
+
* Hebrew niqqud (U+05B0–U+05BC) is out of range too, so this is inert for every
|
|
252
|
+
* other script.
|
|
253
|
+
*/
|
|
254
|
+
export function stripOptionalDiacritics(word: string): string {
|
|
255
|
+
return word.replace(/[\u064b-\u0652\u0670]/g, '');
|
|
256
|
+
}
|
|
257
|
+
|
|
240
258
|
/**
|
|
241
259
|
* Check if a character is an ASCII letter.
|
|
242
260
|
*/
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* createDomainRenderer: hand-written overrides + schema fallthrough + null.
|
|
3
|
+
*
|
|
4
|
+
* Domains used to render through a hardcoded `switch (node.action)` whose
|
|
5
|
+
* default returned a sentinel string (`-- Unknown: <action>`), so a downstream
|
|
6
|
+
* consumer could neither add a command nor distinguish failure from success
|
|
7
|
+
* without string matching. This composes the two halves instead.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { describe, it, expect, vi } from 'vitest';
|
|
11
|
+
import { createDomainRenderer, createSchemaRenderer } from './renderer';
|
|
12
|
+
import type { SemanticNode } from '../core/types';
|
|
13
|
+
import type { CommandSchema } from '../schema';
|
|
14
|
+
import { defineCommand, defineRole } from '../schema';
|
|
15
|
+
import type { PatternGenLanguageProfile } from './pattern-generator';
|
|
16
|
+
|
|
17
|
+
function makeNode(action: string, roles: Record<string, string>): SemanticNode {
|
|
18
|
+
const rolesMap = new Map<string, { type: 'expression'; raw: string }>();
|
|
19
|
+
for (const [k, v] of Object.entries(roles)) {
|
|
20
|
+
rolesMap.set(k, { type: 'expression', raw: v });
|
|
21
|
+
}
|
|
22
|
+
return { kind: 'command', action, roles: rolesMap };
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
const selectSchema = defineCommand({
|
|
26
|
+
action: 'select',
|
|
27
|
+
description: 'Select data',
|
|
28
|
+
category: 'query',
|
|
29
|
+
primaryRole: 'columns',
|
|
30
|
+
roles: [
|
|
31
|
+
defineRole({ role: 'columns', required: true, expectedTypes: ['expression'] }),
|
|
32
|
+
defineRole({
|
|
33
|
+
role: 'source',
|
|
34
|
+
required: true,
|
|
35
|
+
expectedTypes: ['expression'],
|
|
36
|
+
markerOverride: { en: 'from', ja: 'から', ar: 'من' },
|
|
37
|
+
}),
|
|
38
|
+
],
|
|
39
|
+
});
|
|
40
|
+
|
|
41
|
+
/** Stands in for a command a downstream consumer added without editing the domain. */
|
|
42
|
+
const researchSchema = defineCommand({
|
|
43
|
+
action: 'research',
|
|
44
|
+
description: 'Research a topic',
|
|
45
|
+
category: 'query',
|
|
46
|
+
primaryRole: 'patient',
|
|
47
|
+
roles: [
|
|
48
|
+
defineRole({ role: 'patient', required: true, expectedTypes: ['expression'] }),
|
|
49
|
+
defineRole({
|
|
50
|
+
role: 'source',
|
|
51
|
+
required: true,
|
|
52
|
+
expectedTypes: ['expression'],
|
|
53
|
+
markerOverride: { en: 'from', ja: 'から', ar: 'من' },
|
|
54
|
+
}),
|
|
55
|
+
],
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
const profiles: PatternGenLanguageProfile[] = [
|
|
59
|
+
{
|
|
60
|
+
code: 'en',
|
|
61
|
+
wordOrder: 'SVO',
|
|
62
|
+
keywords: { select: { primary: 'select' }, research: { primary: 'research' } },
|
|
63
|
+
roleMarkers: {},
|
|
64
|
+
},
|
|
65
|
+
{
|
|
66
|
+
code: 'ja',
|
|
67
|
+
wordOrder: 'SOV',
|
|
68
|
+
keywords: { select: { primary: '選択' }, research: { primary: '調査' } },
|
|
69
|
+
roleMarkers: {},
|
|
70
|
+
},
|
|
71
|
+
{
|
|
72
|
+
code: 'ar',
|
|
73
|
+
wordOrder: 'VSO',
|
|
74
|
+
keywords: { select: { primary: 'اختر' }, research: { primary: 'ابحث' } },
|
|
75
|
+
roleMarkers: {},
|
|
76
|
+
},
|
|
77
|
+
];
|
|
78
|
+
|
|
79
|
+
const schemas: CommandSchema[] = [selectSchema, researchSchema];
|
|
80
|
+
|
|
81
|
+
describe('createDomainRenderer', () => {
|
|
82
|
+
describe('overrides take precedence', () => {
|
|
83
|
+
it('uses the hand-written renderer for an action it covers', () => {
|
|
84
|
+
const render = createDomainRenderer({
|
|
85
|
+
schemas,
|
|
86
|
+
profiles,
|
|
87
|
+
overrides: { select: () => 'HAND WRITTEN' },
|
|
88
|
+
});
|
|
89
|
+
expect(render(makeNode('select', { columns: 'name', source: 'users' }), 'en')).toBe(
|
|
90
|
+
'HAND WRITTEN'
|
|
91
|
+
);
|
|
92
|
+
});
|
|
93
|
+
|
|
94
|
+
it('passes the node and language through to the override', () => {
|
|
95
|
+
const override = vi.fn(() => 'x');
|
|
96
|
+
const render = createDomainRenderer({ schemas, profiles, overrides: { select: override } });
|
|
97
|
+
const node = makeNode('select', { columns: 'name', source: 'users' });
|
|
98
|
+
|
|
99
|
+
render(node, 'ja');
|
|
100
|
+
|
|
101
|
+
expect(override).toHaveBeenCalledWith(node, 'ja');
|
|
102
|
+
});
|
|
103
|
+
});
|
|
104
|
+
|
|
105
|
+
describe('schema fallthrough', () => {
|
|
106
|
+
// These are the outputs a downstream consumer gets for a command the domain
|
|
107
|
+
// package has never heard of — the whole point of the fallthrough.
|
|
108
|
+
const render = createDomainRenderer({
|
|
109
|
+
schemas,
|
|
110
|
+
profiles,
|
|
111
|
+
overrides: { select: () => 'HAND WRITTEN' },
|
|
112
|
+
});
|
|
113
|
+
const node = makeNode('research', { patient: '"climate"', source: '#wiki' });
|
|
114
|
+
|
|
115
|
+
it('renders SVO', () => {
|
|
116
|
+
expect(render(node, 'en')).toBe('research "climate" from #wiki');
|
|
117
|
+
});
|
|
118
|
+
|
|
119
|
+
it('renders verb-final for SOV', () => {
|
|
120
|
+
expect(render(node, 'ja')).toBe('"climate" #wiki から 調査');
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
it('renders verb-initial for VSO', () => {
|
|
124
|
+
expect(render(node, 'ar')).toBe('ابحث "climate" من #wiki');
|
|
125
|
+
});
|
|
126
|
+
|
|
127
|
+
it('matches createSchemaRenderer for the same schema', () => {
|
|
128
|
+
const schemaRenderer = createSchemaRenderer(schemas, profiles);
|
|
129
|
+
for (const language of ['en', 'ja', 'ar']) {
|
|
130
|
+
expect(render(node, language)).toBe(schemaRenderer.render(node, language));
|
|
131
|
+
}
|
|
132
|
+
});
|
|
133
|
+
});
|
|
134
|
+
|
|
135
|
+
describe('unknown actions', () => {
|
|
136
|
+
it('returns null rather than a sentinel string', () => {
|
|
137
|
+
const render = createDomainRenderer({ schemas, profiles });
|
|
138
|
+
expect(render(makeNode('nonsense', { patient: 'x' }), 'en')).toBeNull();
|
|
139
|
+
});
|
|
140
|
+
|
|
141
|
+
it('returns null even when other actions have overrides', () => {
|
|
142
|
+
const render = createDomainRenderer({
|
|
143
|
+
schemas,
|
|
144
|
+
profiles,
|
|
145
|
+
overrides: { select: () => 'HAND WRITTEN' },
|
|
146
|
+
});
|
|
147
|
+
expect(render(makeNode('nonsense', {}), 'ja')).toBeNull();
|
|
148
|
+
});
|
|
149
|
+
});
|
|
150
|
+
|
|
151
|
+
it('builds its tables lazily', () => {
|
|
152
|
+
// Constructing a renderer at module scope must not walk every schema and
|
|
153
|
+
// profile; domains create one per module whether or not it is ever called.
|
|
154
|
+
const profileAccess = vi.fn(() => 'SVO' as const);
|
|
155
|
+
const watchedProfiles = [
|
|
156
|
+
{
|
|
157
|
+
code: 'en',
|
|
158
|
+
keywords: { select: { primary: 'select' } },
|
|
159
|
+
roleMarkers: {},
|
|
160
|
+
get wordOrder() {
|
|
161
|
+
return profileAccess();
|
|
162
|
+
},
|
|
163
|
+
} as unknown as PatternGenLanguageProfile,
|
|
164
|
+
];
|
|
165
|
+
|
|
166
|
+
const render = createDomainRenderer({ schemas, profiles: watchedProfiles });
|
|
167
|
+
expect(profileAccess).not.toHaveBeenCalled();
|
|
168
|
+
|
|
169
|
+
render(makeNode('select', { columns: 'name', source: 'users' }), 'en');
|
|
170
|
+
expect(profileAccess).toHaveBeenCalled();
|
|
171
|
+
});
|
|
172
|
+
});
|