@lokascript/framework 2.6.0 → 2.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,441 @@
1
+ /**
2
+ * Tests for the framework ↔ semantic bridge builders.
3
+ *
4
+ * Uses inline fixture slices (en-like SVO, ja-like SOV particle, ar-like RTL
5
+ * VSO) — no semantic dependency: any `@lokascript/semantic` LanguageProfile
6
+ * satisfies GrammarProfileSlice structurally, and so do these fixtures.
7
+ */
8
+
9
+ import { describe, it, expect } from 'vitest';
10
+ import {
11
+ buildPatternProfile,
12
+ buildDomainTokenizer,
13
+ buildLanguageConfig,
14
+ deriveRoleMarkers,
15
+ } from './builders';
16
+ import type { GrammarProfileSlice, DomainVocabulary } from './types';
17
+ import { generatePattern } from '../generation/pattern-generator';
18
+ import { defineCommand, defineRole } from '../schema';
19
+ import { createMultilingualDSL } from '../api/create-dsl';
20
+
21
+ // =============================================================================
22
+ // Fixture slices (mirror the shape of semantic KNOWN_PROFILES entries)
23
+ // =============================================================================
24
+
25
+ const EN_SLICE: GrammarProfileSlice = {
26
+ code: 'en',
27
+ name: 'English',
28
+ nativeName: 'English',
29
+ wordOrder: 'SVO',
30
+ direction: 'ltr',
31
+ script: 'latin',
32
+ usesSpaces: true,
33
+ markingStrategy: 'preposition',
34
+ roleMarkers: {
35
+ source: { primary: 'from', position: 'before' },
36
+ destination: { primary: 'into', alternatives: ['to'], position: 'before' },
37
+ },
38
+ };
39
+
40
+ const JA_SLICE: GrammarProfileSlice = {
41
+ code: 'ja',
42
+ name: 'Japanese',
43
+ nativeName: '日本語',
44
+ wordOrder: 'SOV',
45
+ direction: 'ltr',
46
+ script: 'cjk',
47
+ usesSpaces: false,
48
+ markingStrategy: 'particle',
49
+ tokenization: { particles: ['を', 'に', 'から'], boundaryStrategy: 'particle' },
50
+ roleMarkers: {
51
+ patient: { primary: 'を', position: 'after' },
52
+ destination: { primary: 'に', alternatives: ['へ'], position: 'after' },
53
+ source: { primary: 'から', position: 'after' },
54
+ },
55
+ };
56
+
57
+ const AR_SLICE: GrammarProfileSlice = {
58
+ code: 'ar',
59
+ wordOrder: 'VSO',
60
+ direction: 'rtl',
61
+ script: 'arabic',
62
+ usesSpaces: true,
63
+ markingStrategy: 'preposition',
64
+ roleMarkers: {
65
+ source: { primary: 'من', position: 'before' },
66
+ destination: { primary: 'في', position: 'before' },
67
+ },
68
+ };
69
+
70
+ const EN_VOCAB: DomainVocabulary = {
71
+ keywords: {
72
+ select: { primary: 'select', alternatives: ['get'] },
73
+ insert: { primary: 'insert' },
74
+ },
75
+ tokenizerKeywords: ['where', 'and'],
76
+ };
77
+
78
+ const JA_VOCAB: DomainVocabulary = {
79
+ keywords: {
80
+ select: { primary: '選択' },
81
+ insert: { primary: '挿入' },
82
+ },
83
+ roleMarkerOverrides: {
84
+ condition: { primary: '条件', position: 'before' },
85
+ },
86
+ };
87
+
88
+ const AR_VOCAB: DomainVocabulary = {
89
+ keywords: {
90
+ select: { primary: 'اختر' },
91
+ insert: { primary: 'أدخل' },
92
+ },
93
+ };
94
+
95
+ // =============================================================================
96
+ // GrammarProfileSlice structural acceptance
97
+ // =============================================================================
98
+
99
+ describe('GrammarProfileSlice', () => {
100
+ it('accepts a full semantic-LanguageProfile-shaped object without a cast', () => {
101
+ // Mirrors the field set of a semantic LanguageProfile, including the
102
+ // hyperscript-specific fields the slice ignores (keywords, references,
103
+ // possessive, eventHandler). Structural typing must accept it as-is.
104
+ const semanticShaped = {
105
+ code: 'ja',
106
+ name: 'Japanese',
107
+ nativeName: '日本語',
108
+ regions: ['east-asian', 'priority'],
109
+ direction: 'ltr',
110
+ script: 'cjk',
111
+ wordOrder: 'SOV',
112
+ markingStrategy: 'particle',
113
+ usesSpaces: false,
114
+ defaultVerbForm: 'base',
115
+ verb: { position: 'end', suffixes: ['る', 'て'], subjectDrop: true },
116
+ references: { me: '自分', it: 'それ' },
117
+ possessive: { marker: 'の', markerPosition: 'between' },
118
+ roleMarkers: {
119
+ patient: { primary: 'を', position: 'after' },
120
+ destination: { primary: 'に', alternatives: ['へ', 'で'], position: 'after' },
121
+ },
122
+ keywords: {
123
+ toggle: { primary: '切り替え', alternatives: ['トグル'], normalized: 'toggle' },
124
+ },
125
+ tokenization: { particles: ['を', 'に'], boundaryStrategy: 'particle' },
126
+ } as const;
127
+
128
+ const slice: GrammarProfileSlice = semanticShaped;
129
+ expect(slice.code).toBe('ja');
130
+ expect(slice.roleMarkers?.patient?.primary).toBe('を');
131
+ });
132
+ });
133
+
134
+ // =============================================================================
135
+ // buildPatternProfile
136
+ // =============================================================================
137
+
138
+ describe('buildPatternProfile', () => {
139
+ it('combines domain keywords with slice grammar', () => {
140
+ const profile = buildPatternProfile(EN_SLICE, EN_VOCAB);
141
+
142
+ expect(profile.code).toBe('en');
143
+ expect(profile.wordOrder).toBe('SVO');
144
+ expect(profile.keywords.select).toEqual({ primary: 'select', alternatives: ['get'] });
145
+ expect(profile.keywords.insert).toEqual({ primary: 'insert' });
146
+ expect(profile.roleMarkers?.source).toEqual({ primary: 'from', position: 'before' });
147
+ expect(profile.roleMarkers?.destination).toEqual({
148
+ primary: 'into',
149
+ alternatives: ['to'],
150
+ position: 'before',
151
+ });
152
+ });
153
+
154
+ it('merges vocab roleMarkerOverrides over slice markers (vocab wins, adds domain roles)', () => {
155
+ const profile = buildPatternProfile(JA_SLICE, JA_VOCAB);
156
+
157
+ // Slice defaults preserved
158
+ expect(profile.roleMarkers?.patient).toEqual({ primary: 'を', position: 'after' });
159
+ // Domain-specific role added by the vocabulary
160
+ expect(profile.roleMarkers?.condition).toEqual({ primary: '条件', position: 'before' });
161
+ });
162
+
163
+ it('lets an override replace a slice marker, and an empty primary remove one', () => {
164
+ const vocab: DomainVocabulary = {
165
+ keywords: { select: { primary: 'seç' } },
166
+ roleMarkerOverrides: {
167
+ source: { primary: 'den', position: 'after' },
168
+ destination: { primary: '' },
169
+ },
170
+ };
171
+ const profile = buildPatternProfile(JA_SLICE, vocab);
172
+
173
+ expect(profile.roleMarkers?.source).toEqual({ primary: 'den', position: 'after' });
174
+ expect(profile.roleMarkers?.destination).toBeUndefined();
175
+ });
176
+
177
+ it('omits roleMarkers entirely when neither slice nor vocab provides any', () => {
178
+ const bare: GrammarProfileSlice = { code: 'xx', wordOrder: 'SVO' };
179
+ const profile = buildPatternProfile(bare, { keywords: { go: { primary: 'go' } } });
180
+
181
+ expect(profile.roleMarkers).toBeUndefined();
182
+ });
183
+
184
+ it('copies alternatives into fresh mutable arrays', () => {
185
+ const profile = buildPatternProfile(EN_SLICE, EN_VOCAB);
186
+ const alternatives = EN_VOCAB.keywords.select.alternatives!;
187
+
188
+ expect(profile.keywords.select.alternatives).not.toBe(alternatives);
189
+ expect(profile.keywords.select.alternatives).toEqual([...alternatives]);
190
+ });
191
+
192
+ describe('generatePattern integration', () => {
193
+ const moveSchema = defineCommand({
194
+ action: 'move',
195
+ description: 'Move a thing',
196
+ category: 'test',
197
+ primaryRole: 'patient',
198
+ roles: [
199
+ defineRole({
200
+ role: 'patient',
201
+ required: true,
202
+ expectedTypes: ['expression'],
203
+ svoPosition: 2,
204
+ sovPosition: 2,
205
+ }),
206
+ defineRole({
207
+ role: 'destination',
208
+ required: true,
209
+ expectedTypes: ['expression'],
210
+ svoPosition: 1,
211
+ sovPosition: 1,
212
+ }),
213
+ ],
214
+ });
215
+
216
+ it('SVO (en-like): keyword first, prepositional marker before its role', () => {
217
+ const vocab: DomainVocabulary = { keywords: { move: { primary: 'move' } } };
218
+ const pattern = generatePattern(moveSchema, buildPatternProfile(EN_SLICE, vocab));
219
+
220
+ expect(pattern.template.tokens).toEqual([
221
+ { type: 'literal', value: 'move' },
222
+ { type: 'role', role: 'patient', optional: false, expectedTypes: ['expression'] },
223
+ { type: 'literal', value: 'into' },
224
+ { type: 'role', role: 'destination', optional: false, expectedTypes: ['expression'] },
225
+ ]);
226
+ });
227
+
228
+ it('SOV (ja-like): particles after their roles, verb last', () => {
229
+ const vocab: DomainVocabulary = { keywords: { move: { primary: '移動' } } };
230
+ const pattern = generatePattern(moveSchema, buildPatternProfile(JA_SLICE, vocab));
231
+
232
+ expect(pattern.template.tokens).toEqual([
233
+ { type: 'role', role: 'patient', optional: false, expectedTypes: ['expression'] },
234
+ { type: 'literal', value: 'を' },
235
+ { type: 'role', role: 'destination', optional: false, expectedTypes: ['expression'] },
236
+ { type: 'literal', value: 'に' },
237
+ { type: 'literal', value: '移動' },
238
+ ]);
239
+ });
240
+
241
+ it('VSO (ar-like): verb first', () => {
242
+ const vocab: DomainVocabulary = { keywords: { move: { primary: 'انقل' } } };
243
+ const pattern = generatePattern(moveSchema, buildPatternProfile(AR_SLICE, vocab));
244
+
245
+ expect(pattern.template.tokens[0]).toEqual({ type: 'literal', value: 'انقل' });
246
+ });
247
+ });
248
+ });
249
+
250
+ // =============================================================================
251
+ // buildDomainTokenizer
252
+ // =============================================================================
253
+
254
+ describe('buildDomainTokenizer', () => {
255
+ it('sets language and direction from the slice', () => {
256
+ expect(buildDomainTokenizer(EN_SLICE, EN_VOCAB).language).toBe('en');
257
+ expect(buildDomainTokenizer(EN_SLICE, EN_VOCAB).direction).toBe('ltr');
258
+ expect(buildDomainTokenizer(AR_SLICE, AR_VOCAB).direction).toBe('rtl');
259
+ });
260
+
261
+ it('classifies vocab verbs, alternatives, slice markers, and extra keywords as keywords', () => {
262
+ const tokenizer = buildDomainTokenizer(EN_SLICE, EN_VOCAB);
263
+
264
+ expect(tokenizer.classifyToken('select')).toBe('keyword');
265
+ expect(tokenizer.classifyToken('get')).toBe('keyword'); // alternative
266
+ expect(tokenizer.classifyToken('from')).toBe('keyword'); // slice role marker
267
+ expect(tokenizer.classifyToken('to')).toBe('keyword'); // marker alternative
268
+ expect(tokenizer.classifyToken('where')).toBe('keyword'); // tokenizerKeywords
269
+ expect(tokenizer.classifyToken('users')).toBe('identifier');
270
+ expect(tokenizer.classifyToken('42')).toBe('literal');
271
+ });
272
+
273
+ it('is case-insensitive by default for Latin scripts', () => {
274
+ const tokenizer = buildDomainTokenizer(EN_SLICE, EN_VOCAB);
275
+ expect(tokenizer.classifyToken('SELECT')).toBe('keyword');
276
+ expect(tokenizer.classifyToken('From')).toBe('keyword');
277
+ });
278
+
279
+ it('classifies particles and non-Latin verbs as keywords (ja-like)', () => {
280
+ const tokenizer = buildDomainTokenizer(JA_SLICE, JA_VOCAB);
281
+
282
+ expect(tokenizer.classifyToken('選択')).toBe('keyword');
283
+ expect(tokenizer.classifyToken('を')).toBe('keyword'); // tokenization particle
284
+ expect(tokenizer.classifyToken('条件')).toBe('keyword'); // vocab roleMarkerOverride
285
+ });
286
+
287
+ it('excludes operators by default (matching createSimpleTokenizer) and honors includeOperators: true', () => {
288
+ const defaultTokenizer = buildDomainTokenizer(EN_SLICE, EN_VOCAB);
289
+ const withOps = buildDomainTokenizer(EN_SLICE, EN_VOCAB, { includeOperators: true });
290
+
291
+ expect(defaultTokenizer.classifyToken('=')).toBe('identifier');
292
+ expect(withOps.classifyToken('=')).toBe('operator');
293
+ });
294
+
295
+ it('keeps diacritic identifiers whole for latin-script slices (R8 by construction)', () => {
296
+ const esSlice: GrammarProfileSlice = { code: 'es', wordOrder: 'SVO', script: 'latin' };
297
+ const vocab: DomainVocabulary = { keywords: { insert: { primary: 'añadir' } } };
298
+ const tokenizer = buildDomainTokenizer(esSlice, vocab);
299
+
300
+ const stream = tokenizer.tokenize('añadir usuarios');
301
+ const first = stream.advance();
302
+ expect(first.value).toBe('añadir');
303
+ expect(tokenizer.classifyToken(first.value)).toBe('keyword');
304
+ });
305
+
306
+ it('tokenizes multi-word vocab keywords as a single normalized token', () => {
307
+ const frSlice: GrammarProfileSlice = { code: 'fr', wordOrder: 'SVO', script: 'latin' };
308
+ const vocab: DomainVocabulary = { keywords: { update: { primary: 'mettre à jour' } } };
309
+ const tokenizer = buildDomainTokenizer(frSlice, vocab);
310
+
311
+ const stream = tokenizer.tokenize('mettre à jour utilisateurs');
312
+ const first = stream.advance();
313
+ expect(first.value).toBe('mettre à jour');
314
+ expect(first.kind).toBe('keyword');
315
+ expect(first.normalized).toBe('update');
316
+ });
317
+
318
+ it('threads vocab keywordExtras into the keyword map', () => {
319
+ const vocab: DomainVocabulary = {
320
+ ...JA_VOCAB,
321
+ keywordExtras: [{ native: 'すべて 削除', normalized: 'truncate' }],
322
+ };
323
+ const tokenizer = buildDomainTokenizer(JA_SLICE, vocab);
324
+
325
+ const stream = tokenizer.tokenize('すべて 削除');
326
+ const first = stream.advance();
327
+ expect(first.value).toBe('すべて 削除');
328
+ expect(first.normalized).toBe('truncate');
329
+ });
330
+ });
331
+
332
+ // =============================================================================
333
+ // buildLanguageConfig
334
+ // =============================================================================
335
+
336
+ describe('buildLanguageConfig', () => {
337
+ it('assembles a complete LanguageConfig from slice + vocab', () => {
338
+ const config = buildLanguageConfig(JA_SLICE, JA_VOCAB);
339
+
340
+ expect(config.code).toBe('ja');
341
+ expect(config.name).toBe('Japanese');
342
+ expect(config.nativeName).toBe('日本語');
343
+ expect(config.tokenizer.language).toBe('ja');
344
+ expect(config.patternProfile.code).toBe('ja');
345
+ expect(config.patternProfile.wordOrder).toBe('SOV');
346
+ });
347
+
348
+ it('falls back to the language code when the slice has no names', () => {
349
+ const config = buildLanguageConfig(AR_SLICE, AR_VOCAB);
350
+
351
+ expect(config.name).toBe('ar');
352
+ expect(config.nativeName).toBe('ar');
353
+ });
354
+
355
+ it('lets meta override names and tokenizer', () => {
356
+ const customTokenizer = buildDomainTokenizer(EN_SLICE, EN_VOCAB, { includeOperators: false });
357
+ const config = buildLanguageConfig(EN_SLICE, EN_VOCAB, {
358
+ name: 'English (US)',
359
+ nativeName: 'American English',
360
+ tokenizer: customTokenizer,
361
+ });
362
+
363
+ expect(config.name).toBe('English (US)');
364
+ expect(config.nativeName).toBe('American English');
365
+ expect(config.tokenizer).toBe(customTokenizer);
366
+ });
367
+
368
+ describe('createMultilingualDSL integration', () => {
369
+ const selectSchema = defineCommand({
370
+ action: 'select',
371
+ description: 'Select data',
372
+ category: 'query',
373
+ primaryRole: 'columns',
374
+ roles: [
375
+ defineRole({
376
+ role: 'columns',
377
+ required: true,
378
+ expectedTypes: ['expression'],
379
+ svoPosition: 2,
380
+ sovPosition: 1,
381
+ }),
382
+ defineRole({
383
+ role: 'source',
384
+ required: true,
385
+ expectedTypes: ['expression'],
386
+ svoPosition: 1,
387
+ sovPosition: 2,
388
+ }),
389
+ ],
390
+ });
391
+
392
+ const dsl = createMultilingualDSL({
393
+ name: 'BridgeTestSQL',
394
+ schemas: [selectSchema],
395
+ languages: [buildLanguageConfig(EN_SLICE, EN_VOCAB), buildLanguageConfig(JA_SLICE, JA_VOCAB)],
396
+ });
397
+
398
+ it('parses English through a bridge-built config', () => {
399
+ const node = dsl.parse('select name from users', 'en');
400
+
401
+ expect(node.action).toBe('select');
402
+ expect(node.roles.has('columns')).toBe(true);
403
+ expect(node.roles.has('source')).toBe(true);
404
+ });
405
+
406
+ it('parses Japanese (SOV, particle markers) through a bridge-built config', () => {
407
+ const node = dsl.parse('users から name 選択', 'ja');
408
+
409
+ expect(node.action).toBe('select');
410
+ expect(node.roles.has('source')).toBe(true);
411
+ });
412
+ });
413
+ });
414
+
415
+ // =============================================================================
416
+ // deriveRoleMarkers
417
+ // =============================================================================
418
+
419
+ describe('deriveRoleMarkers', () => {
420
+ it('maps domain roles to the slice markers of their semantic counterparts', () => {
421
+ expect(deriveRoleMarkers(JA_SLICE, { table: 'source', target: 'destination' })).toEqual({
422
+ table: 'から',
423
+ target: 'に',
424
+ });
425
+ expect(deriveRoleMarkers(EN_SLICE, { table: 'source', target: 'destination' })).toEqual({
426
+ table: 'from',
427
+ target: 'into',
428
+ });
429
+ });
430
+
431
+ it('omits domain roles whose semantic role has no marker in the slice', () => {
432
+ expect(deriveRoleMarkers(EN_SLICE, { table: 'source', cond: 'condition' })).toEqual({
433
+ table: 'from',
434
+ });
435
+ });
436
+
437
+ it('returns an empty record for a slice without roleMarkers', () => {
438
+ const bare: GrammarProfileSlice = { code: 'xx', wordOrder: 'SVO' };
439
+ expect(deriveRoleMarkers(bare, { table: 'source' })).toEqual({});
440
+ });
441
+ });
@@ -0,0 +1,224 @@
1
+ /**
2
+ * Framework ↔ Semantic bridge builders.
3
+ *
4
+ * Pure functions that combine an injected {@link GrammarProfileSlice} (the
5
+ * per-language grammar facts, typically a `@lokascript/semantic`
6
+ * `LanguageProfile`) with a {@link DomainVocabulary} (the only thing a domain
7
+ * authors per language) into the shapes `createMultilingualDSL` consumes:
8
+ * a `PatternGenLanguageProfile`, a `LanguageTokenizer`, or a complete
9
+ * `LanguageConfig`.
10
+ */
11
+
12
+ import type { LanguageTokenizer } from '../core/types';
13
+ import type { PatternGenLanguageProfile } from '../generation/pattern-generator';
14
+ import type { LanguageConfig } from '../api/create-dsl';
15
+ import type { LanguageProfile as GrammarProfile } from '../grammar';
16
+ import { createSimpleTokenizer, type TokenizerProfile } from '../core/tokenization/base-tokenizer';
17
+ import {
18
+ LatinExtendedIdentifierExtractor,
19
+ type ValueExtractor,
20
+ } from '../interfaces/value-extractor';
21
+ import type { DomainVocabulary, GrammarProfileSlice, RoleMarkerSlice } from './types';
22
+
23
+ /** Mutable role-marker shape shared by PatternGenLanguageProfile and TokenizerProfile. */
24
+ type MergedRoleMarker = { primary: string; alternatives?: string[]; position?: 'before' | 'after' };
25
+
26
+ /**
27
+ * Merge the slice's language-default role markers with the vocabulary's
28
+ * domain-specific overrides (vocabulary wins). Empty-string primaries are
29
+ * dropped — an empty override means "bare positional arg, no marker".
30
+ */
31
+ function mergeRoleMarkers(
32
+ slice: GrammarProfileSlice,
33
+ vocab: DomainVocabulary
34
+ ): Record<string, MergedRoleMarker> {
35
+ const merged: Record<string, MergedRoleMarker> = {};
36
+ const add = (role: string, marker: RoleMarkerSlice | undefined) => {
37
+ if (!marker?.primary) {
38
+ delete merged[role];
39
+ return;
40
+ }
41
+ merged[role] = {
42
+ primary: marker.primary,
43
+ ...(marker.alternatives?.length && { alternatives: [...marker.alternatives] }),
44
+ ...(marker.position && { position: marker.position }),
45
+ };
46
+ };
47
+ for (const [role, marker] of Object.entries(slice.roleMarkers ?? {})) add(role, marker);
48
+ for (const [role, marker] of Object.entries(vocab.roleMarkerOverrides ?? {})) add(role, marker);
49
+ return merged;
50
+ }
51
+
52
+ /**
53
+ * Build a pattern-generation profile: domain keywords + slice grammar
54
+ * (word order, role markers). The result feeds `generatePattern` /
55
+ * `generatePatternVariants` and `LanguageConfig.patternProfile`.
56
+ */
57
+ export function buildPatternProfile(
58
+ slice: GrammarProfileSlice,
59
+ vocab: DomainVocabulary
60
+ ): PatternGenLanguageProfile {
61
+ const keywords: Record<string, { primary: string; alternatives?: string[] }> = {};
62
+ for (const [action, translation] of Object.entries(vocab.keywords)) {
63
+ keywords[action] = {
64
+ primary: translation.primary,
65
+ ...(translation.alternatives?.length && { alternatives: [...translation.alternatives] }),
66
+ };
67
+ }
68
+
69
+ const roleMarkers = mergeRoleMarkers(slice, vocab);
70
+
71
+ return {
72
+ code: slice.code,
73
+ wordOrder: slice.wordOrder,
74
+ keywords,
75
+ ...(Object.keys(roleMarkers).length > 0 && { roleMarkers }),
76
+ };
77
+ }
78
+
79
+ /** Options for {@link buildDomainTokenizer}, overriding the slice-derived defaults. */
80
+ export interface DomainTokenizerOptions {
81
+ /**
82
+ * Recognize operator tokens (default: false, matching `createSimpleTokenizer`).
83
+ * Domains whose grammar contains operators (e.g. SQL's `WHERE age > 18`) pass
84
+ * `true`; most natural-language DSLs leave it off.
85
+ */
86
+ readonly includeOperators?: boolean;
87
+ /**
88
+ * Case-insensitive keyword matching. Defaults to true for bicameral scripts
89
+ * (`latin`, `cyrillic`, or when the slice omits `script`), false otherwise.
90
+ */
91
+ readonly caseInsensitive?: boolean;
92
+ /** Extra extractors, registered before the slice-derived ones */
93
+ readonly customExtractors?: readonly ValueExtractor[];
94
+ }
95
+
96
+ function defaultCaseInsensitive(script: string | undefined): boolean {
97
+ return script === undefined || script === 'latin' || script === 'cyrillic';
98
+ }
99
+
100
+ /**
101
+ * Build a domain tokenizer from a grammar slice + domain vocabulary, wrapping
102
+ * `createSimpleTokenizer`. Derived from the slice:
103
+ *
104
+ * - `direction` (rtl for e.g. Arabic/Hebrew slices)
105
+ * - keyword set: vocabulary verbs (primary + alternatives), role markers
106
+ * (slice defaults merged with `vocab.roleMarkerOverrides`), tokenization
107
+ * particles, and `vocab.tokenizerKeywords`
108
+ * - keyword-normalization profile (native verb → action name, marker → role)
109
+ * - a `LatinExtendedIdentifierExtractor` when `slice.script === 'latin'`, so
110
+ * diacritics (ñ, é, ş, …) never split identifiers
111
+ * - case-insensitivity for bicameral scripts
112
+ */
113
+ export function buildDomainTokenizer(
114
+ slice: GrammarProfileSlice,
115
+ vocab: DomainVocabulary,
116
+ options: DomainTokenizerOptions = {}
117
+ ): LanguageTokenizer {
118
+ const roleMarkers = mergeRoleMarkers(slice, vocab);
119
+
120
+ // Flat keyword list for token classification.
121
+ const keywords = new Set<string>();
122
+ for (const translation of Object.values(vocab.keywords)) {
123
+ keywords.add(translation.primary);
124
+ for (const alt of translation.alternatives ?? []) keywords.add(alt);
125
+ }
126
+ for (const marker of Object.values(roleMarkers)) {
127
+ keywords.add(marker.primary);
128
+ for (const alt of marker.alternatives ?? []) keywords.add(alt);
129
+ }
130
+ for (const particle of slice.tokenization?.particles ?? []) keywords.add(particle);
131
+ for (const extra of vocab.tokenizerKeywords ?? []) keywords.add(extra);
132
+
133
+ // Normalization profile: native verb forms → action names, markers → roles.
134
+ const profileKeywords: NonNullable<TokenizerProfile['keywords']> = {};
135
+ for (const [action, translation] of Object.entries(vocab.keywords)) {
136
+ profileKeywords[action] = {
137
+ primary: translation.primary,
138
+ ...(translation.alternatives?.length && { alternatives: [...translation.alternatives] }),
139
+ normalized: translation.normalized ?? action,
140
+ };
141
+ }
142
+ const keywordProfile: TokenizerProfile = {
143
+ keywords: profileKeywords,
144
+ ...(Object.keys(roleMarkers).length > 0 && { roleMarkers }),
145
+ };
146
+
147
+ const customExtractors: ValueExtractor[] = [
148
+ ...(options.customExtractors ?? []),
149
+ ...(slice.script === 'latin' ? [new LatinExtendedIdentifierExtractor()] : []),
150
+ ];
151
+
152
+ return createSimpleTokenizer({
153
+ language: slice.code,
154
+ direction: slice.direction ?? 'ltr',
155
+ keywords: [...keywords],
156
+ ...(vocab.keywordExtras?.length && { keywordExtras: vocab.keywordExtras.map(e => ({ ...e })) }),
157
+ keywordProfile,
158
+ includeOperators: options.includeOperators ?? false,
159
+ caseInsensitive: options.caseInsensitive ?? defaultCaseInsensitive(slice.script),
160
+ ...(customExtractors.length > 0 && { customExtractors }),
161
+ });
162
+ }
163
+
164
+ /** Metadata / overrides for {@link buildLanguageConfig}. */
165
+ export interface LanguageConfigMeta {
166
+ /** English name (default: `slice.name`, falling back to `slice.code`) */
167
+ readonly name?: string;
168
+ /** Native name (default: `slice.nativeName`, falling back to the English name) */
169
+ readonly nativeName?: string;
170
+ /** Grammar-transformation profile to attach (optional, as on `LanguageConfig`) */
171
+ readonly grammarProfile?: GrammarProfile;
172
+ /** Use this tokenizer instead of the slice-derived one */
173
+ readonly tokenizer?: LanguageTokenizer;
174
+ /** Options for the slice-derived tokenizer (ignored when `tokenizer` is given) */
175
+ readonly tokenizerOptions?: DomainTokenizerOptions;
176
+ }
177
+
178
+ /**
179
+ * The one-call path into `createMultilingualDSL`: build a complete
180
+ * `LanguageConfig` (tokenizer + pattern profile + names) from a grammar slice
181
+ * and a domain vocabulary.
182
+ */
183
+ export function buildLanguageConfig(
184
+ slice: GrammarProfileSlice,
185
+ vocab: DomainVocabulary,
186
+ meta: LanguageConfigMeta = {}
187
+ ): LanguageConfig {
188
+ const name = meta.name ?? slice.name ?? slice.code;
189
+ return {
190
+ code: slice.code,
191
+ name,
192
+ nativeName: meta.nativeName ?? slice.nativeName ?? name,
193
+ tokenizer: meta.tokenizer ?? buildDomainTokenizer(slice, vocab, meta.tokenizerOptions),
194
+ patternProfile: buildPatternProfile(slice, vocab),
195
+ ...(meta.grammarProfile && { grammarProfile: meta.grammarProfile }),
196
+ };
197
+ }
198
+
199
+ /**
200
+ * Derive schema `markerOverride` defaults for ONE language from the slice's
201
+ * role markers, given a domain-role → semantic-role mapping. Returns
202
+ * `{ domainRole: primaryMarker }` with entries only where the slice has a
203
+ * marker for the mapped semantic role.
204
+ *
205
+ * ```ts
206
+ * deriveRoleMarkers(jaSlice, { table: 'source', target: 'destination' });
207
+ * // → { table: 'から', target: 'に' }
208
+ * ```
209
+ *
210
+ * Explicitly authored `markerOverride` entries stay authoritative — when
211
+ * assembling a per-language marker map for a schema role, spread the explicit
212
+ * entries after the derived defaults.
213
+ */
214
+ export function deriveRoleMarkers(
215
+ slice: GrammarProfileSlice,
216
+ roleMapping: Readonly<Record<string, string>>
217
+ ): Record<string, string> {
218
+ const derived: Record<string, string> = {};
219
+ for (const [domainRole, semanticRole] of Object.entries(roleMapping)) {
220
+ const marker = slice.roleMarkers?.[semanticRole];
221
+ if (marker?.primary) derived[domainRole] = marker.primary;
222
+ }
223
+ return derived;
224
+ }