@lokascript/i18n 2.11.1 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/CHANGELOG.md +13 -0
  2. package/dist/browser.cjs +5 -1690
  3. package/dist/browser.cjs.map +1 -1
  4. package/dist/browser.d.cts +2 -2
  5. package/dist/browser.d.ts +2 -2
  6. package/dist/browser.js +6 -1685
  7. package/dist/browser.js.map +1 -1
  8. package/dist/dictionaries/index.cjs +5 -1
  9. package/dist/dictionaries/index.cjs.map +1 -1
  10. package/dist/dictionaries/index.js +5 -1
  11. package/dist/dictionaries/index.js.map +1 -1
  12. package/dist/{transformer-CsOeqayN.d.cts → index-BykxjYST.d.cts} +1 -232
  13. package/dist/{transformer-DWCTG1DQ.d.ts → index-DuIef8O7.d.ts} +1 -232
  14. package/dist/index.cjs +16 -1878
  15. package/dist/index.cjs.map +1 -1
  16. package/dist/index.d.cts +35 -35
  17. package/dist/index.d.ts +35 -35
  18. package/dist/index.js +17 -1873
  19. package/dist/index.js.map +1 -1
  20. package/dist/lokascript-i18n.min.js +1 -1
  21. package/dist/lokascript-i18n.min.js.map +1 -1
  22. package/dist/lokascript-i18n.mjs +49 -2680
  23. package/dist/lokascript-i18n.mjs.map +1 -1
  24. package/dist/plugins/vite.cjs +5 -1
  25. package/dist/plugins/vite.cjs.map +1 -1
  26. package/dist/plugins/vite.js +5 -1
  27. package/dist/plugins/vite.js.map +1 -1
  28. package/dist/plugins/webpack.cjs +5 -1
  29. package/dist/plugins/webpack.cjs.map +1 -1
  30. package/dist/plugins/webpack.js +5 -1
  31. package/dist/plugins/webpack.js.map +1 -1
  32. package/package.json +4 -4
  33. package/src/browser.ts +0 -7
  34. package/src/compatibility/browser-tests/grammar-demo.spec.ts +22 -8
  35. package/src/constants.ts +1 -0
  36. package/src/dictionaries/bn.ts +5 -1
  37. package/src/grammar/index.ts +15 -9
  38. package/src/grammar/profiles.test.ts +440 -0
  39. package/src/index.ts +0 -7
  40. package/src/lexicon-parity.test.ts +77 -0
  41. package/src/grammar/grammar.test.ts +0 -2751
  42. package/src/grammar/transformer.ts +0 -2737
@@ -1,2751 +0,0 @@
1
- /**
2
- * Grammar Transformer Tests
3
- *
4
- * Tests for the generalized grammar transformation system
5
- * that handles multilingual hyperscript with proper word order
6
- * and grammatical markers.
7
- */
8
-
9
- import { describe, it, expect } from 'vitest';
10
- import {
11
- parseStatement,
12
- toLocale,
13
- toEnglish,
14
- translate,
15
- GrammarTransformer,
16
- examples,
17
- } from './transformer';
18
- import {
19
- getProfile,
20
- getSupportedLocales,
21
- profiles,
22
- englishProfile,
23
- japaneseProfile,
24
- chineseProfile,
25
- arabicProfile,
26
- } from './profiles';
27
- import {
28
- reorderRoles,
29
- insertMarkers,
30
- joinTokens,
31
- UNIVERSAL_PATTERNS,
32
- LANGUAGE_FAMILY_DEFAULTS,
33
- } from './types';
34
- import type { ParsedElement, SemanticRole } from './types';
35
-
36
- // =============================================================================
37
- // Profile Tests
38
- // =============================================================================
39
-
40
- describe('Language Profiles', () => {
41
- it('should have profiles for all supported locales', () => {
42
- const locales = getSupportedLocales();
43
- // Explicit expected list — when adding a profile, append it here so the
44
- // count assertion stays meaningful without becoming a magic number.
45
- const expectedLocales = [
46
- 'en',
47
- 'ja',
48
- 'ko',
49
- 'zh',
50
- 'ar',
51
- 'tr',
52
- 'es',
53
- 'de',
54
- 'fr',
55
- 'pt',
56
- 'id',
57
- 'ms',
58
- 'qu',
59
- 'sw',
60
- 'bn',
61
- 'it',
62
- 'ru',
63
- 'uk',
64
- 'vi',
65
- 'hi',
66
- 'tl',
67
- 'th',
68
- 'pl',
69
- 'he',
70
- ];
71
- for (const code of expectedLocales) {
72
- expect(locales, `missing profile: ${code}`).toContain(code);
73
- }
74
- expect(locales.length).toBe(expectedLocales.length);
75
- });
76
-
77
- it('should return undefined for unknown locales', () => {
78
- expect(getProfile('xx')).toBeUndefined();
79
- expect(getProfile('xyz')).toBeUndefined();
80
- });
81
-
82
- describe('English Profile', () => {
83
- it('should have SVO word order', () => {
84
- expect(englishProfile.wordOrder).toBe('SVO');
85
- });
86
-
87
- it('should use prepositions', () => {
88
- expect(englishProfile.adpositionType).toBe('preposition');
89
- });
90
-
91
- it('should have required markers', () => {
92
- const onMarker = englishProfile.markers.find(m => m.form === 'on');
93
- expect(onMarker).toBeDefined();
94
- expect(onMarker?.role).toBe('event');
95
- expect(onMarker?.required).toBe(true);
96
- });
97
- });
98
-
99
- describe('Japanese Profile', () => {
100
- it('should have SOV word order', () => {
101
- expect(japaneseProfile.wordOrder).toBe('SOV');
102
- });
103
-
104
- it('should use postpositions', () => {
105
- expect(japaneseProfile.adpositionType).toBe('postposition');
106
- });
107
-
108
- it('should have particle markers', () => {
109
- const woMarker = japaneseProfile.markers.find(m => m.form === 'を');
110
- expect(woMarker).toBeDefined();
111
- expect(woMarker?.role).toBe('patient');
112
- expect(woMarker?.position).toBe('postposition');
113
- });
114
-
115
- it('should place patient before action in canonical order', () => {
116
- const patientIndex = japaneseProfile.canonicalOrder.indexOf('patient');
117
- const actionIndex = japaneseProfile.canonicalOrder.indexOf('action');
118
- expect(patientIndex).toBeLessThan(actionIndex);
119
- });
120
- });
121
-
122
- describe('Arabic Profile', () => {
123
- it('should have VSO word order', () => {
124
- expect(arabicProfile.wordOrder).toBe('VSO');
125
- });
126
-
127
- it('should be RTL', () => {
128
- expect(arabicProfile.direction).toBe('rtl');
129
- });
130
-
131
- it('should place action first in canonical order', () => {
132
- expect(arabicProfile.canonicalOrder[0]).toBe('action');
133
- });
134
- });
135
-
136
- describe('Chinese Profile', () => {
137
- it('should have isolating morphology', () => {
138
- expect(chineseProfile.morphology).toBe('isolating');
139
- });
140
-
141
- it('should have circumfix markers for events', () => {
142
- const eventMarkers = chineseProfile.markers.filter(m => m.role === 'event');
143
- const hasPreposition = eventMarkers.some(m => m.position === 'preposition');
144
- const hasPostposition = eventMarkers.some(m => m.position === 'postposition');
145
- expect(hasPreposition).toBe(true);
146
- expect(hasPostposition).toBe(true);
147
- });
148
- });
149
- });
150
-
151
- // =============================================================================
152
- // Statement Parsing Tests
153
- // =============================================================================
154
-
155
- describe('Statement Parser', () => {
156
- describe('parseStatement', () => {
157
- it('should parse event handlers', () => {
158
- const parsed = parseStatement('on click increment #count');
159
- expect(parsed).not.toBeNull();
160
- expect(parsed?.type).toBe('event-handler');
161
- expect(parsed?.roles.get('event')?.value).toBe('click');
162
- expect(parsed?.roles.get('action')?.value).toBe('increment');
163
- expect(parsed?.roles.get('patient')?.value).toBe('#count');
164
- });
165
-
166
- it('should identify CSS selectors as patient', () => {
167
- const parsed = parseStatement('on click toggle .active');
168
- expect(parsed?.roles.get('patient')?.isSelector).toBe(true);
169
- });
170
-
171
- it('should parse commands', () => {
172
- const parsed = parseStatement('put my value into #output');
173
- expect(parsed).not.toBeNull();
174
- expect(parsed?.type).toBe('command');
175
- expect(parsed?.roles.get('action')?.value).toBe('put');
176
- });
177
-
178
- it('should parse conditionals', () => {
179
- const parsed = parseStatement('if count > 5 then log done');
180
- expect(parsed).not.toBeNull();
181
- expect(parsed?.type).toBe('conditional');
182
- });
183
-
184
- it('should return null for empty input', () => {
185
- const parsed = parseStatement('');
186
- expect(parsed).toBeNull();
187
- });
188
-
189
- it('should preserve original input', () => {
190
- const input = 'on click increment #count';
191
- const parsed = parseStatement(input);
192
- expect(parsed?.original).toBe(input);
193
- });
194
- });
195
-
196
- describe('Event Handler Parsing', () => {
197
- it('should handle various event types', () => {
198
- const events = ['click', 'input', 'keydown', 'mouseenter', 'submit'];
199
- for (const event of events) {
200
- const parsed = parseStatement(`on ${event} log done`);
201
- expect(parsed?.roles.get('event')?.value).toBe(event);
202
- }
203
- });
204
-
205
- it('should handle complex selectors', () => {
206
- const parsed = parseStatement('on click toggle .menu-item.active');
207
- expect(parsed?.roles.get('patient')?.value).toBe('.menu-item.active');
208
- });
209
- });
210
-
211
- describe('Command Parsing', () => {
212
- it('should identify destination with "to" keyword', () => {
213
- const parsed = parseStatement('add .highlight to #element');
214
- expect(parsed?.roles.get('destination')?.value).toBe('#element');
215
- });
216
-
217
- it('should identify destination with "into" keyword', () => {
218
- const parsed = parseStatement('put value into #output');
219
- expect(parsed?.roles.get('destination')?.value).toBe('#output');
220
- });
221
-
222
- it('should identify source with "from" keyword', () => {
223
- const parsed = parseStatement('get data from #input');
224
- expect(parsed?.roles.get('source')?.value).toBe('#input');
225
- });
226
- });
227
- });
228
-
229
- // =============================================================================
230
- // Role Transformation Tests
231
- // =============================================================================
232
-
233
- describe('Role Transformation', () => {
234
- describe('reorderRoles', () => {
235
- it('should reorder roles according to target order', () => {
236
- const roles = new Map<SemanticRole, ParsedElement>([
237
- ['action', { role: 'action', value: 'increment' }],
238
- ['patient', { role: 'patient', value: '#count' }],
239
- ['event', { role: 'event', value: 'click' }],
240
- ]);
241
-
242
- // Japanese order: patient, event, action
243
- const reordered = reorderRoles(roles, ['patient', 'event', 'action']);
244
-
245
- expect(reordered[0].role).toBe('patient');
246
- expect(reordered[1].role).toBe('event');
247
- expect(reordered[2].role).toBe('action');
248
- });
249
-
250
- it('should skip roles not present in input', () => {
251
- const roles = new Map<SemanticRole, ParsedElement>([
252
- ['action', { role: 'action', value: 'toggle' }],
253
- ['patient', { role: 'patient', value: '.active' }],
254
- ]);
255
-
256
- const reordered = reorderRoles(roles, ['patient', 'destination', 'action']);
257
-
258
- expect(reordered.length).toBe(2);
259
- expect(reordered[0].role).toBe('patient');
260
- expect(reordered[1].role).toBe('action');
261
- });
262
- });
263
-
264
- describe('insertMarkers', () => {
265
- it('should insert preposition markers before elements', () => {
266
- const elements: ParsedElement[] = [
267
- { role: 'destination', value: '#output', translated: '#output' },
268
- ];
269
- const markers = [
270
- {
271
- form: 'to',
272
- role: 'destination' as SemanticRole,
273
- position: 'preposition' as const,
274
- required: false,
275
- },
276
- ];
277
-
278
- const result = insertMarkers(elements, markers, 'preposition');
279
- expect(result).toEqual(['to', '#output']);
280
- });
281
-
282
- it('should insert postposition markers after elements', () => {
283
- const elements: ParsedElement[] = [
284
- { role: 'patient', value: '#count', translated: '#count' },
285
- ];
286
- const markers = [
287
- {
288
- form: 'を',
289
- role: 'patient' as SemanticRole,
290
- position: 'postposition' as const,
291
- required: true,
292
- },
293
- ];
294
-
295
- const result = insertMarkers(elements, markers, 'postposition');
296
- expect(result).toEqual(['#count', 'を']);
297
- });
298
-
299
- it('should use translated values when available', () => {
300
- const elements: ParsedElement[] = [
301
- { role: 'action', value: 'increment', translated: '増加' },
302
- ];
303
-
304
- const result = insertMarkers(elements, [], 'none');
305
- expect(result).toEqual(['増加']);
306
- });
307
- });
308
-
309
- describe('joinTokens', () => {
310
- it('should join regular tokens with spaces', () => {
311
- const result = joinTokens(['hello', 'world']);
312
- expect(result).toBe('hello world');
313
- });
314
-
315
- it('should handle empty array', () => {
316
- const result = joinTokens([]);
317
- expect(result).toBe('');
318
- });
319
-
320
- it('should handle single token', () => {
321
- const result = joinTokens(['hello']);
322
- expect(result).toBe('hello');
323
- });
324
-
325
- it('should attach suffix markers without space (Quechua -ta)', () => {
326
- // #count + -ta → #countta
327
- const result = joinTokens(['#count', '-ta']);
328
- expect(result).toBe('#countta');
329
- });
330
-
331
- it('should attach prefix markers without space (Arabic بـ-)', () => {
332
- // بـ- + الماوس → بـالماوس
333
- const result = joinTokens(['بـ-', 'الماوس']);
334
- expect(result).toBe('بـالماوس');
335
- });
336
-
337
- it('should handle multiple suffix markers (Turkish case suffixes)', () => {
338
- // value + -i + another → valuei another
339
- const result = joinTokens(['value', '-i', 'another']);
340
- expect(result).toBe('valuei another');
341
- });
342
-
343
- it('should handle Japanese particles with normal spacing', () => {
344
- // Japanese particles don't use hyphen notation, so they get spaces
345
- const result = joinTokens(['#count', 'を', 'クリック', 'で', '増加']);
346
- expect(result).toBe('#count を クリック で 増加');
347
- });
348
-
349
- it('should handle Quechua agglutinative chain', () => {
350
- // #count + -ta + click + -pi + increment
351
- const result = joinTokens(['#count', '-ta', 'click', '-pi', 'increment']);
352
- expect(result).toBe('#countta clickpi increment');
353
- });
354
-
355
- it('should handle mixed prefix and regular tokens', () => {
356
- const result = joinTokens(['كـ-', 'JSON', 'format']);
357
- expect(result).toBe('كـJSON format');
358
- });
359
- });
360
- });
361
-
362
- // =============================================================================
363
- // Grammar Transformer Tests
364
- // =============================================================================
365
-
366
- describe('GrammarTransformer', () => {
367
- describe('Constructor', () => {
368
- it('should create transformer with valid locales', () => {
369
- expect(() => new GrammarTransformer('en', 'ja')).not.toThrow();
370
- expect(() => new GrammarTransformer('en', 'zh')).not.toThrow();
371
- expect(() => new GrammarTransformer('en', 'ar')).not.toThrow();
372
- });
373
-
374
- it('should throw for invalid source locale', () => {
375
- expect(() => new GrammarTransformer('xx', 'ja')).toThrow('Unknown source locale');
376
- });
377
-
378
- it('should throw for invalid target locale', () => {
379
- expect(() => new GrammarTransformer('en', 'xx')).toThrow('Unknown target locale');
380
- });
381
- });
382
-
383
- describe('Japanese Transformation (SOV)', () => {
384
- const transformer = new GrammarTransformer('en', 'ja');
385
-
386
- it('should transform event handler to SOV order', () => {
387
- const result = transformer.transform('on click increment #count');
388
- // Should have patient (with を), event (with で), action pattern
389
- expect(result).toContain('#count');
390
- expect(result).toContain('を');
391
- });
392
-
393
- it('should preserve CSS selectors', () => {
394
- const result = transformer.transform('on click toggle .active');
395
- expect(result).toContain('.active');
396
- });
397
-
398
- it('should preserve ID selectors', () => {
399
- const result = transformer.transform('on input put value into #output');
400
- expect(result).toContain('#output');
401
- });
402
-
403
- it('should keep event guards intact and untranslated', () => {
404
- // `[key is 'Escape']` must stay one token with its contents verbatim —
405
- // the spaces must not split it, and `is` must not be translated as a verb.
406
- const result = transformer.transform("on keyup[key is 'Escape'] clear me");
407
- // Guard stays one verbatim token attached to the event (not split on its
408
- // internal spaces), and `is` inside it is not translated to a verb.
409
- expect(result).toContain("keyup[key is 'Escape']");
410
- });
411
-
412
- it('should keep an event-handler block body intact (not shredded)', () => {
413
- // `on <event> if … end` must keep the event clause first, then the whole
414
- // `if … end` block as a self-contained unit — never reordered into the
415
- // event handler's role soup.
416
- const result = transformer.transform(
417
- 'on keydown[key=="Enter"] if event.shiftKey call submitAndContinue() end'
418
- );
419
- // Event leads; the if-block follows with its condition preserved.
420
- expect(result).toMatch(/keydown\[key=="Enter"\].*もし.*event\.shiftKey/);
421
- expect(result).toContain('submitAndContinue()');
422
- // `end` keyword present (translated), block not dropped.
423
- expect(result).toContain('終わり');
424
- });
425
-
426
- it('should mask inline js bodies from word-order reordering', () => {
427
- // The raw JS body must stay verbatim and immediately after the (translated)
428
- // `js` keyword — never reordered ahead of the event like other roles.
429
- const result = transformer.transform('on click js console.log("from js") end');
430
- expect(result).toContain('console.log("from js")');
431
- // js keyword precedes the raw body, which precedes the translated `end`.
432
- expect(result).toMatch(/JS実行\s+console\.log\("from js"\)\s+終わり/);
433
- // The body must not be split/reordered: no marker particle injected inside it.
434
- expect(result).not.toMatch(/console\.log.*を.*from/);
435
- });
436
- });
437
-
438
- describe('Hindi Transformation (SOV) — put-into verb-final', () => {
439
- const transformer = new GrammarTransformer('en', 'hi');
440
-
441
- // The hindiProfile was missing the `put-into` word-order rule that every
442
- // other SOV profile (ja/ko/tr/bn) carries, so `put X into Y` fell through to a
443
- // verb-MID default (`X को रखें Y में`). The semantic parser then mis-read the
444
- // verb-mid form — destination/patient swapped/mistyped — the put.* R1 residue
445
- // (hi only; other SOV langs already had the rule). With the rule, hi emits the
446
- // verb-final form `X को Y में रखें` like ja's `X を Y に 置く`.
447
- it('places the put verb after its destination (verb-final), not mid-clause', () => {
448
- const result = transformer.transform('put "<p>x</p>" into #out');
449
- const putIdx = result.indexOf('रखें');
450
- const destIdx = result.indexOf('#out');
451
- expect(putIdx).toBeGreaterThan(-1);
452
- expect(destIdx).toBeGreaterThan(-1);
453
- // verb-final: the put verb follows its destination.
454
- expect(putIdx).toBeGreaterThan(destIdx);
455
- });
456
-
457
- it('keeps the destination marker in (में) on the put target', () => {
458
- const result = transformer.transform('put "hi" into #out');
459
- expect(result).toContain('#out में');
460
- });
461
- });
462
-
463
- describe('SOV Transformation — set-to verb-final (set.* R1 residue)', () => {
464
- // No profile carried a `set` word-order rule, so ja/ko/bn/tr/hi emitted set
465
- // VERB-MEDIAL (`X को सेट Y में`), which the generated SOV set pattern (verb-final,
466
- // markerOverride: को on the target / में on the value) never matched — set parsed
467
- // to no/swapped roles (the dominant set.* R1 residue; qu/SVO were already fine).
468
- // The `set-to` rule emits the verb-final `X को Y में सेट`, like ja's `X を Y に 設定`.
469
- const langs: Array<[string, string, string]> = [
470
- // [lang, set verb, destination marker]
471
- ['hi', 'सेट', 'में'],
472
- ['ja', '設定', 'に'],
473
- ['ko', '설정', '에'],
474
- ['bn', 'সেট', 'তে'],
475
- ['tr', 'ayarla', 'e'],
476
- ];
477
- for (const [lang, verb, mark] of langs) {
478
- it(`[${lang}] set X to Y → verb-final (set verb after the value)`, () => {
479
- const t = new GrammarTransformer('en', lang);
480
- const result = t.transform('set @disabled to true');
481
- const verbIdx = result.indexOf(verb);
482
- const valIdx = result.search(/true|doğru|참|真|সত্য|सच/u);
483
- expect(verbIdx).toBeGreaterThan(-1);
484
- // verb-final: the set verb follows the (translated) value.
485
- expect(verbIdx).toBeGreaterThan(valIdx);
486
- // marker present (markerOverride alignment): value carries the dest marker.
487
- expect(result).toContain(mark);
488
- });
489
- }
490
-
491
- // Inline `if … then set X to Y end`: the set's value sweeps up the trailing
492
- // `end` token (`destination="minWidth end"`), so verb-final reordering would push
493
- // the set verb PAST `end` and break the block (the bn behavior-resizable
494
- // faithful→lossy flip). The set-to predicate skips such sets — they stay
495
- // verb-MEDIAL, so the set verb remains BEFORE the block terminator `end`.
496
- // (Without the guard the set verb lands after `end`, dropping the `if`.)
497
- for (const [lang, verb] of langs.map(l => [l[0], l[1]] as const)) {
498
- it(`[${lang}] inline if/then/set/end keeps the set verb before end (set-to skipped)`, () => {
499
- const t = new GrammarTransformer('en', lang);
500
- const result = t
501
- .transform('if newWidth < minWidth then set newWidth to minWidth end')
502
- .trim();
503
- const endWord = t.transform('end').trim();
504
- const endIdx = result.lastIndexOf(endWord);
505
- expect(endIdx).toBeGreaterThan(-1);
506
- // verb-medial (guarded): the set verb precedes the block terminator.
507
- expect(result.indexOf(verb)).toBeLessThan(endIdx);
508
- });
509
- }
510
- });
511
-
512
- describe('Hindi Transformation (SOV) — bind-to verb-final (bind.* R1 residue)', () => {
513
- // The hindiProfile lacked a `bind` word-order rule (ja/ko/zh/tr/bn had one), so hi
514
- // emitted bind VERB-MEDIAL (`$greeting को bind #name-input में`), which the generated
515
- // verb-final SOV bind pattern never matched → the bare-event fallback mis-anchored
516
- // the fronted `$greeting` as a phantom `on` event (the rf=0.00 bind residue). The
517
- // `bind-to` rule emits verb-final `$greeting को #name-input में bind`, like ja's
518
- // `$greeting を #name-input に バインド`.
519
- it('[hi] bind $var to #el → verb-final (bind verb after the element)', () => {
520
- const t = new GrammarTransformer('en', 'hi');
521
- const result = t.transform('bind $greeting to #name-input');
522
- const verbIdx = result.indexOf('bind');
523
- const elIdx = result.indexOf('#name-input');
524
- expect(verbIdx).toBeGreaterThan(-1);
525
- expect(elIdx).toBeGreaterThan(-1);
526
- // verb-final: the bind verb follows the element (was verb-MEDIAL before the rule).
527
- expect(verbIdx).toBeGreaterThan(elIdx);
528
- // markerOverride alignment: the element carries the destination locative में.
529
- expect(result).toContain('में');
530
- });
531
- });
532
-
533
- describe('Arabic Transformation (VSO)', () => {
534
- const transformer = new GrammarTransformer('en', 'ar');
535
-
536
- it('should transform to VSO order with action first', () => {
537
- const result = transformer.transform('on click increment #count');
538
- // Arabic VSO: action comes first
539
- expect(result).toBeTruthy();
540
- // Verify action (زِد/increment) appears before patient (#count)
541
- const actionIndex = result.indexOf('زِد');
542
- const patientIndex = result.indexOf('#count');
543
- expect(actionIndex).toBeLessThan(patientIndex);
544
- });
545
-
546
- it('should preserve selectors in transformation', () => {
547
- const result = transformer.transform('on click toggle .active');
548
- expect(result).toContain('.active');
549
- });
550
- });
551
-
552
- describe('Chinese Transformation (Topic-Prominent)', () => {
553
- const transformer = new GrammarTransformer('en', 'zh');
554
-
555
- it('should use 当 marker for events', () => {
556
- const result = transformer.transform('on click increment #count');
557
- // Chinese uses 当...时 pattern but custom transform may omit 时
558
- expect(result).toContain('当');
559
- });
560
-
561
- it('should include translated action', () => {
562
- const result = transformer.transform('on click increment #count');
563
- // Should contain 增加 (increment in Chinese)
564
- expect(result).toContain('增加');
565
- });
566
-
567
- it('should preserve patient selector', () => {
568
- const result = transformer.transform('on click toggle .menu');
569
- expect(result).toContain('.menu');
570
- });
571
- });
572
-
573
- describe('Quechua Transformation (SOV)', () => {
574
- const transformer = new GrammarTransformer('en', 'qu');
575
-
576
- it('should emit the install-specific verb (tarpuy), not the put/set verb (churay)', () => {
577
- // Regression: the qu dictionary mapped `install` to `churay`, which is
578
- // also `put`/`set`. The semantic qu profile expects `install` = `tarpuy`
579
- // (churay = put), so `install Draggable` parsed as a malformed `put` and
580
- // failed (`install-behavior` baseline failure). Align the emitted verb to
581
- // the semantic profile's install keyword.
582
- const result = transformer.transform('install Draggable');
583
- expect(result).toContain('tarpuy');
584
- expect(result).not.toContain('churay');
585
- expect(result).toContain('Draggable');
586
- });
587
-
588
- it('should emit the repeat verb (kutipay), not the return verb (kutichiy)', () => {
589
- // Regression: the qu dictionary mapped `repeat` to `kutichiy`, which is the
590
- // semantic qu profile's `return` primary (repeat = kutipay there). So every
591
- // qu `repeat …` transformed to `kutichiy …` and the semantic parser read it
592
- // as `return`, dropping the loop — degenerate parses for the qu repeat-*
593
- // cluster (repeat-while, repeat-for-each). Align the emitted verb to the
594
- // semantic repeat keyword. See docs-internal/SOV_REPEAT_SCOPE.md.
595
- const result = transformer.transform(
596
- 'on click repeat for item in .items add .processed to item'
597
- );
598
- expect(result).toContain('kutipay');
599
- expect(result).not.toContain('kutichiy');
600
- });
601
- });
602
-
603
- describe('German Transformation (fetch/get disambiguation)', () => {
604
- const transformer = new GrammarTransformer('en', 'de');
605
-
606
- it('should emit the fetch-specific verb (abrufen), not the get verb (holen)', () => {
607
- // Regression: the de dictionary mapped `fetch` to `holen`, which is the
608
- // semantic profile's `get` primary (fetch = abrufen there). So
609
- // `fetch /api/data` transformed to `holen …` and the semantic parser read
610
- // it as `get`, dropping the `fetch` action — degenerate parses for the de
611
- // fetch cluster (fetch-do-not-throw, fetch-error-handling, fetch-json,
612
- // fetch-with-headers). Align the emitted verb to the semantic fetch keyword.
613
- const result = transformer.transform('on click fetch /api/data then put it into #result');
614
- expect(result).toContain('abrufen');
615
- expect(result).not.toContain('holen');
616
- expect(result).toContain('/api/data');
617
- });
618
- });
619
-
620
- describe('Swahili Transformation (as/if disambiguation)', () => {
621
- const transformer = new GrammarTransformer('en', 'sw');
622
-
623
- it('should emit kuwa for the as-marker, not the if-homonym kama', () => {
624
- // Regression: sw `kama` is both "as/like" and the IF keyword. The dict +
625
- // grammar profile emitted it for `as`, so every transformed `as <Type>`
626
- // tail (`kama JSON`, `kama Number`) grew a phantom `if` command at
627
- // semantic parse time (computed-value precision 0.500; event-debounce,
628
- // fetch-with-headers, fetch-formdata 0.667–0.750). The dict/profile now
629
- // emit `kuwa` ("to be/become" — the conversion sense, cf. badilisha kuwa).
630
- const result = transformer.transform('on click fetch /api/data as json');
631
- expect(result).toContain('kuwa');
632
- expect(result).not.toMatch(/\bkama\b/);
633
- });
634
-
635
- it('should still emit kama for a real if', () => {
636
- const result = transformer.transform('if true then log "Habari"');
637
- expect(result).toMatch(/\bkama\b/);
638
- expect(result).not.toContain('kuwa');
639
- });
640
- });
641
-
642
- describe('qu/tr while keyword alignment (fronted repeat-while head)', () => {
643
- // Regression: the qu dict emitted `kay_kaq` (unknown to the semantic qu
644
- // profile, whose while primary is `kaykamaqa`), and the tr dict emitted
645
- // `iken` (the tr profile's WHEN primary) — so the fronted repeat-while head
646
- // never formed a `while` node at parse time and the condition dropped
647
- // wholesale. Align both dicts to the profile while primary.
648
- const source = 'on click repeat while #counter.innerText < 10 increment #counter end';
649
-
650
- it('qu: repeat-while emits the profile while word kaykamaqa, not kay_kaq', () => {
651
- const result = new GrammarTransformer('en', 'qu').transform(source);
652
- expect(result).toContain('kaykamaqa');
653
- expect(result).not.toContain('kay_kaq');
654
- });
655
-
656
- it('tr: repeat-while emits süresince, not the when-homonym iken', () => {
657
- const result = new GrammarTransformer('en', 'tr').transform(source);
658
- expect(result).toContain('süresince');
659
- expect(result).not.toMatch(/\biken\b/);
660
- });
661
- });
662
-
663
- describe('Duration / literal-primary marking (no spurious object particle)', () => {
664
- // A command whose primary argument is a literal/measure (e.g. `wait <duration>`)
665
- // must NOT have that argument marked as a fronted object: the generic argument
666
- // parser used to default the leading arg to the `patient` role, so the target
667
- // emitted an object particle on the duration — Chinese `等待 把 1s` (ungrammatical;
668
- // a duration is never a BA-construction object), Japanese `1s を 待つ`, Korean
669
- // `1s 를 대기`. The marked forms failed the semantic parser's `等待 {duration}`
670
- // pattern and the trailing `wait` dropped. The transformer now honours the
671
- // command's true primary role (`wait` → `duration`, which carries no marker).
672
- // See docs-internal/ZH_BLOCK_BODY_SCOPE.md (#1 — transformer role model).
673
-
674
- it('zh: wait emits a grammatical duration (no 把 object marker)', () => {
675
- const result = new GrammarTransformer('en', 'zh').transform('wait 1s');
676
- expect(result).toContain('等待');
677
- expect(result).toContain('1s');
678
- expect(result).not.toContain('把');
679
- });
680
-
681
- it('ja: wait emits a duration with no を object particle', () => {
682
- const result = new GrammarTransformer('en', 'ja').transform('wait 1s');
683
- expect(result).toContain('待つ');
684
- expect(result).toContain('1s');
685
- expect(result).not.toContain('を');
686
- });
687
-
688
- it('ko: wait emits a duration with no 를/을 object particle', () => {
689
- const result = new GrammarTransformer('en', 'ko').transform('wait 1s');
690
- expect(result).toContain('대기');
691
- expect(result).toContain('1s');
692
- expect(result).not.toContain('를');
693
- expect(result).not.toContain('을');
694
- });
695
-
696
- it('does not disturb marker-bearing primaries: zh fetch keeps its 把 (out of scope)', () => {
697
- // `fetch`'s primary role is `source` (which IS marked in zh), so the fix must
698
- // leave it untouched — only markerless literal/measure primaries are re-marked.
699
- const result = new GrammarTransformer('en', 'zh').transform('fetch /api/data');
700
- expect(result).toContain('/api/data');
701
- expect(result).toContain('把');
702
- });
703
-
704
- it('does not disturb the SOV event-handler cue: ko `on click wait 2s` keeps its patient marker', () => {
705
- // In an event handler, a verb-final SOV language without an event particle
706
- // (Korean) relies on the leading argument's object marker to anchor the
707
- // handler. The fix is scoped to standalone command statements, so this stays
708
- // patient-marked and the `on` handler is still recognised downstream.
709
- const result = new GrammarTransformer('en', 'ko').transform(
710
- 'on click wait 2s then remove me'
711
- );
712
- expect(result).toContain('를');
713
- });
714
- });
715
- });
716
-
717
- describe('Inline `unless` guard in an event handler (no object marker on condition)', () => {
718
- // `on click unless I match .disabled toggle .selected`: the event-handler body
719
- // is a bare `unless <cond> <verb>` guard (no `end`). parseEventHandler sweeps the
720
- // whole tail into one `patient` blob, so an object-marking SVO target used to
721
- // front the *condition* with its marker (he את / zh 把) and strip the marker off
722
- // the real toggle — the semantic parser then dropped `unless`.
723
- // tryTransformEventWithUnlessGuard routes the guard through the standalone block
724
- // path so the marker lands on the toggle patient, not the condition.
725
- // See docs-internal/HANDOFF-lossy-tail.md (unless-condition arc).
726
- const en = 'on click unless I match .disabled toggle .selected';
727
-
728
- it('zh: 把 marks the toggle patient, not the unless condition', () => {
729
- const result = new GrammarTransformer('en', 'zh').transform(en);
730
- expect(result).toContain('除非'); // unless
731
- expect(result).toContain('切换'); // toggle
732
- expect(result).toContain('切换 把 .selected'); // 把 on the toggle patient
733
- expect(result).not.toContain('除非 把'); // never on the condition
734
- });
735
-
736
- it('he: את marks the toggle patient, not the unless condition (unchanged)', () => {
737
- const result = new GrammarTransformer('en', 'he').transform(en);
738
- expect(result).toContain('אלא'); // unless
739
- expect(result).toContain('מתג את .selected'); // את on the toggle patient
740
- expect(result).not.toContain('אלא את'); // never on the condition
741
- });
742
-
743
- it('qu: dict emits the spaced `mana sichus`, not the `_`-split form', () => {
744
- const result = new GrammarTransformer('en', 'qu').transform(en);
745
- expect(result).toContain('mana sichus');
746
- expect(result).not.toContain('mana_sichus');
747
- });
748
- });
749
-
750
- describe('vi render keyword (kết xuất, distinct from show)', () => {
751
- // `render`/`show` both mapped to `hiển thị`, which the semantic profile reads as
752
- // `show` — so vi `render …` parsed as `show`. Dict realigned render → `kết xuất`
753
- // (the profile's render primary); `show` keeps `hiển thị`.
754
- // See docs-internal/HANDOFF-lossy-tail.md (render cluster).
755
- it('emits `kết xuất` for render, leaving show as `hiển thị`', () => {
756
- const t = new GrammarTransformer('en', 'vi');
757
- const rendered = t.transform('on click render #x with y: $data then put it into #out');
758
- expect(rendered).toContain('kết xuất');
759
- expect(rendered).not.toMatch(/hiển thị #x/);
760
- expect(t.transform('on click show #x')).toContain('hiển thị');
761
- });
762
- });
763
-
764
- describe('qu append keyword (qatichiy, not the _-splitting qhipaman_yapay)', () => {
765
- // `qhipaman_yapay` `_`-splits at parse time to `qhipaman`+`yapay`(=add); the dict
766
- // now emits the profile's single-token append primary `qatichiy`.
767
- // See docs-internal/HANDOFF-lossy-tail.md (singleton tail).
768
- it('emits the single-token `qatichiy` for append', () => {
769
- const result = new GrammarTransformer('en', 'qu').transform(
770
- 'on click append "<li>Item</li>" to #list'
771
- );
772
- expect(result).toContain('qatichiy');
773
- expect(result).not.toContain('qhipaman_yapay');
774
- });
775
- });
776
-
777
- // =============================================================================
778
- // Convenience Function Tests
779
- // =============================================================================
780
-
781
- describe('Convenience Functions', () => {
782
- describe('toLocale', () => {
783
- it('should transform English to Japanese', () => {
784
- const result = toLocale('on click toggle .active', 'ja');
785
- expect(result).toContain('.active');
786
- expect(result).toContain('を');
787
- });
788
-
789
- it('should transform English to Chinese', () => {
790
- const result = toLocale('on click increment #count', 'zh');
791
- expect(result).toContain('当');
792
- });
793
- });
794
-
795
- describe('toEnglish', () => {
796
- it('should return unchanged when parsing fails', () => {
797
- // This tests fallback behavior
798
- const result = toEnglish('invalid input', 'ja');
799
- expect(result).toBeTruthy();
800
- });
801
- });
802
-
803
- describe('translate', () => {
804
- it('should return unchanged for same locale', () => {
805
- const input = 'on click toggle .active';
806
- expect(translate(input, 'en', 'en')).toBe(input);
807
- });
808
-
809
- it('should translate English to target locale', () => {
810
- const result = translate('on click increment #count', 'en', 'ja');
811
- expect(result).toContain('#count');
812
- });
813
-
814
- it('should translate to English from source locale', () => {
815
- const result = translate('test input', 'ja', 'en');
816
- expect(result).toBeTruthy();
817
- });
818
-
819
- it('should translate via English pivot', () => {
820
- const result = translate('on click log done', 'ja', 'zh');
821
- expect(result).toBeTruthy();
822
- });
823
- });
824
- });
825
-
826
- // =============================================================================
827
- // Universal Pattern Tests
828
- // =============================================================================
829
-
830
- describe('Universal Patterns', () => {
831
- it('should define event-increment pattern', () => {
832
- const pattern = UNIVERSAL_PATTERNS.eventIncrement;
833
- expect(pattern.name).toBe('event-increment');
834
- expect(pattern.roles).toContain('event');
835
- expect(pattern.roles).toContain('action');
836
- expect(pattern.roles).toContain('patient');
837
- });
838
-
839
- it('should define put-into pattern', () => {
840
- const pattern = UNIVERSAL_PATTERNS.putInto;
841
- expect(pattern.name).toBe('put-into');
842
- expect(pattern.roles).toContain('action');
843
- expect(pattern.roles).toContain('patient');
844
- expect(pattern.roles).toContain('destination');
845
- });
846
-
847
- it('should define wait-duration pattern', () => {
848
- const pattern = UNIVERSAL_PATTERNS.waitDuration;
849
- expect(pattern.roles).toContain('action');
850
- expect(pattern.roles).toContain('quantity');
851
- });
852
- });
853
-
854
- // =============================================================================
855
- // Language Family Defaults Tests
856
- // =============================================================================
857
-
858
- describe('Language Family Defaults', () => {
859
- it('should have Germanic defaults', () => {
860
- const germanic = LANGUAGE_FAMILY_DEFAULTS.germanic;
861
- expect(germanic.wordOrder).toBe('SVO');
862
- expect(germanic.adpositionType).toBe('preposition');
863
- });
864
-
865
- it('should have Japonic defaults', () => {
866
- const japonic = LANGUAGE_FAMILY_DEFAULTS.japonic;
867
- expect(japonic.wordOrder).toBe('SOV');
868
- expect(japonic.adpositionType).toBe('postposition');
869
- });
870
-
871
- it('should have Semitic defaults', () => {
872
- const semitic = LANGUAGE_FAMILY_DEFAULTS.semitic;
873
- expect(semitic.wordOrder).toBe('VSO');
874
- expect(semitic.direction).toBe('rtl');
875
- });
876
-
877
- it('should have Sinitic defaults', () => {
878
- const sinitic = LANGUAGE_FAMILY_DEFAULTS.sinitic;
879
- expect(sinitic.morphology).toBe('isolating');
880
- });
881
- });
882
-
883
- // =============================================================================
884
- // Examples Tests
885
- // =============================================================================
886
-
887
- describe('Grammar Examples', () => {
888
- it('should have English examples', () => {
889
- expect(examples.english.eventHandler).toBe('on click increment #count');
890
- expect(examples.english.putInto).toBe('put my value into #output');
891
- expect(examples.english.toggle).toBe('toggle .active');
892
- });
893
-
894
- it('should have Japanese examples', () => {
895
- expect(examples.japanese.eventHandler).toContain('#count');
896
- expect(examples.japanese.eventHandler).toContain('を');
897
- });
898
-
899
- it('should have Chinese examples', () => {
900
- expect(examples.chinese.eventHandler).toContain('当');
901
- expect(examples.chinese.eventHandler).toContain('时');
902
- });
903
-
904
- it('should have Arabic examples', () => {
905
- expect(examples.arabic.eventHandler).toContain('عند');
906
- });
907
- });
908
-
909
- // =============================================================================
910
- // Edge Cases
911
- // =============================================================================
912
-
913
- describe('Edge Cases', () => {
914
- it('should handle empty input gracefully', () => {
915
- const transformer = new GrammarTransformer('en', 'ja');
916
- const result = transformer.transform('');
917
- expect(result).toBe('');
918
- });
919
-
920
- it('should handle single-word input', () => {
921
- const transformer = new GrammarTransformer('en', 'ja');
922
- const result = transformer.transform('toggle');
923
- expect(result).toBeTruthy();
924
- });
925
-
926
- it('should preserve numbers', () => {
927
- const transformer = new GrammarTransformer('en', 'ja');
928
- const result = transformer.transform('wait 500');
929
- expect(result).toContain('500');
930
- });
931
-
932
- it('should handle complex selectors with special characters', () => {
933
- const parsed = parseStatement('on click toggle .menu-item[data-active="true"]');
934
- expect(parsed?.roles.get('patient')?.value).toContain('data-active');
935
- });
936
-
937
- it('should handle multiple spaces in input', () => {
938
- const parsed = parseStatement('on click toggle .active');
939
- expect(parsed).not.toBeNull();
940
- });
941
- });
942
-
943
- // =============================================================================
944
- // Chinese Circumfix Tokenization Tests
945
- // =============================================================================
946
-
947
- describe('Chinese Circumfix Parsing', () => {
948
- it('should split attached 时 suffix from event words', () => {
949
- // 点击时 should be parsed as two tokens: 点击 + 时
950
- const parsed = parseStatement('当 点击时 增加 #count', 'zh');
951
- expect(parsed).not.toBeNull();
952
- expect(parsed?.type).toBe('event-handler');
953
- });
954
-
955
- it('should handle 当...时 circumfix pattern', () => {
956
- const transformer = new GrammarTransformer('en', 'zh');
957
- const result = transformer.transform('on click increment #count');
958
- // Should produce 当 X 时 pattern
959
- expect(result).toContain('当');
960
- expect(result).toContain('时');
961
- });
962
-
963
- it('should preserve selectors when splitting suffixes', () => {
964
- const parsed = parseStatement('当 点击时 切换 .active', 'zh');
965
- // Patient may include the action in some parsing patterns
966
- expect(parsed?.roles.get('patient')?.value).toContain('.active');
967
- });
968
- });
969
-
970
- // =============================================================================
971
- // Round-Trip Translation Tests
972
- // =============================================================================
973
-
974
- describe('Round-Trip Translation', () => {
975
- describe('English → Japanese → English', () => {
976
- it('should preserve semantic roles in round-trip', () => {
977
- const original = 'on click increment #count';
978
- const toJapanese = translate(original, 'en', 'ja');
979
- expect(toJapanese).toContain('#count');
980
- expect(toJapanese).toContain('を');
981
-
982
- // Note: Perfect round-trip isn't expected due to translation,
983
- // but semantic structure should be preserved
984
- const backToEnglish = translate(toJapanese, 'ja', 'en');
985
- expect(backToEnglish).toBeTruthy();
986
- });
987
-
988
- it('should preserve CSS selectors through round-trip', () => {
989
- const original = 'toggle .menu-active';
990
- const toJapanese = translate(original, 'en', 'ja');
991
- expect(toJapanese).toContain('.menu-active');
992
-
993
- const backToEnglish = translate(toJapanese, 'ja', 'en');
994
- expect(backToEnglish).toContain('.menu-active');
995
- });
996
- });
997
-
998
- describe('English → Arabic → English', () => {
999
- it('should preserve semantic roles with VSO transformation', () => {
1000
- const original = 'on click increment #count';
1001
- const toArabic = translate(original, 'en', 'ar');
1002
- expect(toArabic).toContain('#count');
1003
- // Arabic VSO puts action first
1004
- expect(toArabic).toBeTruthy();
1005
-
1006
- const backToEnglish = translate(toArabic, 'ar', 'en');
1007
- expect(backToEnglish).toBeTruthy();
1008
- });
1009
- });
1010
-
1011
- describe('English → Chinese → English', () => {
1012
- it('should preserve structure through topic-prominent language', () => {
1013
- const original = 'on click toggle .active';
1014
- const toChinese = translate(original, 'en', 'zh');
1015
- expect(toChinese).toContain('.active');
1016
- expect(toChinese).toContain('当');
1017
-
1018
- const backToEnglish = translate(toChinese, 'zh', 'en');
1019
- expect(backToEnglish).toContain('.active');
1020
- });
1021
- });
1022
-
1023
- describe('Cross-Language via Pivot', () => {
1024
- it('should translate Japanese → Arabic via English pivot', () => {
1025
- // Start with a simple pattern
1026
- const result = translate('on click log done', 'ja', 'ar');
1027
- expect(result).toBeTruthy();
1028
- });
1029
-
1030
- it('should translate Chinese → Korean via English pivot', () => {
1031
- const result = translate('on click toggle .active', 'zh', 'ko');
1032
- expect(result).toBeTruthy();
1033
- expect(result).toContain('.active');
1034
- });
1035
- });
1036
- });
1037
-
1038
- // =============================================================================
1039
- // Language-Specific Word Order Integration Tests
1040
- // =============================================================================
1041
-
1042
- describe('Word Order Integration Tests', () => {
1043
- describe('SOV Languages (Japanese, Korean, Turkish, Quechua)', () => {
1044
- it('should place patient before action in Japanese', () => {
1045
- const transformer = new GrammarTransformer('en', 'ja');
1046
- const result = transformer.transform('on click increment #count');
1047
- // Japanese SOV: #count を ... 増加
1048
- const countIndex = result.indexOf('#count');
1049
- const actionIndex = result.indexOf('増加');
1050
- expect(countIndex).toBeLessThan(actionIndex);
1051
- });
1052
-
1053
- it('should place patient before action in Korean', () => {
1054
- const transformer = new GrammarTransformer('en', 'ko');
1055
- const result = transformer.transform('on click increment #count');
1056
- // Korean SOV: patient comes before action
1057
- expect(result).toContain('#count');
1058
- expect(result).toContain('를'); // Object marker
1059
- });
1060
-
1061
- it('should preserve Japanese particle spacing (regression)', () => {
1062
- const transformer = new GrammarTransformer('en', 'ja');
1063
- const result = transformer.transform('on click toggle .active');
1064
- // Japanese particles (を, で, に) should have spaces around them
1065
- // They do NOT use hyphen notation like Turkish suffixes
1066
- expect(result).toContain('.active を'); // Space before particle
1067
- });
1068
-
1069
- it('should produce spaced Turkish suffixes for tokenization', () => {
1070
- const transformer = new GrammarTransformer('en', 'tr');
1071
- const result = transformer.transform('on click toggle .active');
1072
- // Turkish uses case suffixes - now with spaces for tokenization
1073
- expect(result).toContain('.active');
1074
- // Verify suffixes have spaces before them (for tokenization)
1075
- expect(result).not.toContain('-i'); // No hyphenated suffixes in output
1076
- expect(result).not.toContain('-e');
1077
- });
1078
-
1079
- it('should produce spaced Turkish accusative suffix for tokenization', () => {
1080
- const transformer = new GrammarTransformer('en', 'tr');
1081
- const result = transformer.transform('on click toggle .active');
1082
- // Should have space between patient and accusative marker for tokenization
1083
- // Output: ".active i" (spaced) so semantic tokenizer can parse it
1084
- expect(result).toMatch(/\.active [iıuü]/);
1085
- });
1086
-
1087
- it('should produce spaced Turkish locative suffix for tokenization', () => {
1088
- const transformer = new GrammarTransformer('en', 'tr');
1089
- const result = transformer.transform('on click toggle .active');
1090
- // Event should have space before locative marker for tokenization
1091
- // Output: "tıklama de" (spaced) so semantic tokenizer can parse it
1092
- expect(result).toMatch(/tıklama [dD][aAeE]/);
1093
- });
1094
- });
1095
-
1096
- describe('VSO Languages (Arabic)', () => {
1097
- it('should place action first in Arabic', () => {
1098
- const transformer = new GrammarTransformer('en', 'ar');
1099
- const result = transformer.transform('on click increment #count');
1100
- // Arabic VSO: زِد (action) comes first
1101
- const actionIndex = result.indexOf('زِد');
1102
- const patientIndex = result.indexOf('#count');
1103
- expect(actionIndex).toBeLessThan(patientIndex);
1104
- });
1105
- });
1106
-
1107
- describe('SVO Languages with Special Features', () => {
1108
- it('should use circumfix pattern for Chinese events', () => {
1109
- const transformer = new GrammarTransformer('en', 'zh');
1110
- const result = transformer.transform('on click increment #count');
1111
- // Chinese uses 当...时 circumfix
1112
- expect(result).toContain('当');
1113
- expect(result).toContain('时');
1114
- });
1115
-
1116
- it('should use correct markers for Spanish', () => {
1117
- const transformer = new GrammarTransformer('en', 'es');
1118
- const result = transformer.transform('on click toggle .active');
1119
- // Spanish uses 'en' for events
1120
- expect(result).toContain('.active');
1121
- });
1122
-
1123
- it('should handle Indonesian SVO correctly', () => {
1124
- const transformer = new GrammarTransformer('en', 'id');
1125
- const result = transformer.transform('on click toggle .active');
1126
- expect(result).toContain('.active');
1127
- });
1128
-
1129
- it('should handle Swahili SVO correctly', () => {
1130
- const transformer = new GrammarTransformer('en', 'sw');
1131
- const result = transformer.transform('on click toggle .active');
1132
- expect(result).toContain('.active');
1133
- });
1134
- });
1135
-
1136
- // Regression guards for the multilingual parse-rate roadmap (see
1137
- // docs-internal/MULTILINGUAL_ROADMAP.md).
1138
- describe('Multi-event handlers (or-conjoined events)', () => {
1139
- it('keeps "or"-conjoined events together as a single event clause (ar)', () => {
1140
- const transformer = new GrammarTransformer('en', 'ar');
1141
- const result = transformer.transform('on click or keypress[key=="Enter"] toggle .active');
1142
- // The toggle action must lead (VSO); the event clause "<click> or keypress"
1143
- // stays together at the end, rather than "or keypress" being hoisted ahead
1144
- // of the command (the old bug, which read "or" as the action verb).
1145
- expect(result.indexOf('بدل')).toBeLessThan(result.indexOf('أو'));
1146
- expect(result).toContain('أو keypress[key=="Enter"]');
1147
- });
1148
-
1149
- it('keeps "or"-conjoined events together as a single event clause (tl)', () => {
1150
- const transformer = new GrammarTransformer('en', 'tl');
1151
- const result = transformer.transform('on click or keypress[key=="Enter"] toggle .active');
1152
- expect(result).toContain('o keypress[key=="Enter"]');
1153
- expect(result.indexOf('palitan')).toBeLessThan(result.indexOf(' o '));
1154
- });
1155
- });
1156
-
1157
- describe('Tagalog transition keyword alignment', () => {
1158
- it('emits the semantic transition verb "lumipat" (not "baguhin"=morph)', () => {
1159
- const transformer = new GrammarTransformer('en', 'tl');
1160
- const result = transformer.transform('on click transition opacity to 0 over 300ms');
1161
- expect(result).toContain('lumipat');
1162
- expect(result).not.toContain('baguhin');
1163
- });
1164
- });
1165
- });
1166
-
1167
- // =============================================================================
1168
- // Line Structure Preservation Tests
1169
- // =============================================================================
1170
-
1171
- describe('Line Structure Preservation', () => {
1172
- describe('Indentation Preservation', () => {
1173
- it('should preserve indentation in multi-line statements', () => {
1174
- const input = `on click
1175
- toggle .active on me
1176
- wait 1 second`;
1177
-
1178
- const transformer = new GrammarTransformer('en', 'es');
1179
- const result = transformer.transform(input);
1180
-
1181
- const lines = result.split('\n');
1182
- expect(lines.length).toBe(3);
1183
- // First line has no indentation
1184
- expect(lines[0]).not.toMatch(/^\s/);
1185
- // Subsequent lines should have indentation
1186
- expect(lines[1]).toMatch(/^\s{4}/);
1187
- expect(lines[2]).toMatch(/^\s{4}/);
1188
- });
1189
-
1190
- it('should normalize mixed tab/space indentation', () => {
1191
- const input = `on click
1192
- \ttoggle .active
1193
- wait 1 second`;
1194
-
1195
- const transformer = new GrammarTransformer('en', 'ja');
1196
- const result = transformer.transform(input);
1197
-
1198
- const lines = result.split('\n');
1199
- expect(lines.length).toBe(3);
1200
- // Both indented lines should use consistent 4-space indentation
1201
- const indent1 = lines[1].match(/^\s*/)?.[0] || '';
1202
- const indent2 = lines[2].match(/^\s*/)?.[0] || '';
1203
- // Tabs normalized to spaces
1204
- expect(indent1).not.toContain('\t');
1205
- expect(indent2).not.toContain('\t');
1206
- });
1207
-
1208
- it('should handle deeply nested indentation', () => {
1209
- const input = `on click
1210
- if something
1211
- toggle .active
1212
- wait 1 second`;
1213
-
1214
- const transformer = new GrammarTransformer('en', 'ko');
1215
- const result = transformer.transform(input);
1216
-
1217
- const lines = result.split('\n');
1218
- expect(lines.length).toBe(4);
1219
- // Check relative indentation is preserved
1220
- const indent1 = (lines[1].match(/^\s*/)?.[0] || '').length;
1221
- const indent2 = (lines[2].match(/^\s*/)?.[0] || '').length;
1222
- const indent3 = (lines[3].match(/^\s*/)?.[0] || '').length;
1223
- expect(indent2).toBeGreaterThan(indent1);
1224
- expect(indent3).toBe(indent2); // Same level as line above
1225
- });
1226
- });
1227
-
1228
- describe('Blank Line Preservation', () => {
1229
- it('should preserve blank lines between statements', () => {
1230
- const input = `on click
1231
- toggle .active
1232
-
1233
- wait 1 second`;
1234
-
1235
- const transformer = new GrammarTransformer('en', 'zh');
1236
- const result = transformer.transform(input);
1237
-
1238
- const lines = result.split('\n');
1239
- expect(lines.length).toBe(4);
1240
- expect(lines[2]).toBe(''); // Blank line preserved
1241
- });
1242
-
1243
- it('should preserve multiple consecutive blank lines', () => {
1244
- const input = `on click
1245
-
1246
-
1247
- toggle .active`;
1248
-
1249
- const transformer = new GrammarTransformer('en', 'ar');
1250
- const result = transformer.transform(input);
1251
-
1252
- const lines = result.split('\n');
1253
- expect(lines.length).toBe(4);
1254
- expect(lines[1]).toBe('');
1255
- expect(lines[2]).toBe('');
1256
- });
1257
-
1258
- it('should handle blank lines with only whitespace', () => {
1259
- const input = `on click
1260
- toggle .active
1261
-
1262
- wait 1 second`;
1263
-
1264
- const transformer = new GrammarTransformer('en', 'tr');
1265
- const result = transformer.transform(input);
1266
-
1267
- const lines = result.split('\n');
1268
- expect(lines.length).toBe(4);
1269
- // Line with only whitespace should become empty
1270
- expect(lines[2]).toBe('');
1271
- });
1272
- });
1273
-
1274
- describe('Single-line Backward Compatibility', () => {
1275
- it('should not change behavior for single-line input', () => {
1276
- const input = 'on click toggle .active';
1277
- const transformer = new GrammarTransformer('en', 'es');
1278
- const result = transformer.transform(input);
1279
-
1280
- // Should not contain newlines
1281
- expect(result).not.toContain('\n');
1282
- });
1283
-
1284
- it('should handle then-chains on single line', () => {
1285
- const input = 'on click toggle .active then wait 1 second';
1286
- const transformer = new GrammarTransformer('en', 'ja');
1287
- const result = transformer.transform(input);
1288
-
1289
- // Should not contain newlines
1290
- expect(result).not.toContain('\n');
1291
- // Should still have the translated "then" keyword
1292
- expect(result.split(' ').length).toBeGreaterThan(3);
1293
- });
1294
- });
1295
-
1296
- describe('Multi-language Structure Preservation', () => {
1297
- const languages = ['es', 'ja', 'ko', 'zh', 'ar', 'tr', 'id', 'qu', 'sw'];
1298
-
1299
- for (const lang of languages) {
1300
- it(`should preserve structure when translating to ${lang}`, () => {
1301
- const input = `on click
1302
- toggle .active
1303
-
1304
- wait 1 second
1305
- remove .active`;
1306
-
1307
- const transformer = new GrammarTransformer('en', lang);
1308
- const result = transformer.transform(input);
1309
-
1310
- const lines = result.split('\n');
1311
- expect(lines.length).toBe(5);
1312
- // Verify blank line is preserved
1313
- expect(lines[2]).toBe('');
1314
- // Verify non-blank lines have content
1315
- expect(lines[0].trim().length).toBeGreaterThan(0);
1316
- expect(lines[1].trim().length).toBeGreaterThan(0);
1317
- expect(lines[3].trim().length).toBeGreaterThan(0);
1318
- expect(lines[4].trim().length).toBeGreaterThan(0);
1319
- });
1320
- }
1321
- });
1322
- });
1323
-
1324
- // =============================================================================
1325
- // Cross-Language Command Boundary Tests
1326
- // =============================================================================
1327
-
1328
- describe('Cross-Language Command Boundaries', () => {
1329
- // A grammatical marker (preposition/postposition) that binds an argument to
1330
- // its verb must never be mistaken for a command boundary. The English base
1331
- // set covers `to`/`on`/etc.; these cases verify the same protection for
1332
- // localized markers sourced from each language's profile.
1333
- const hasStandaloneThen = (s: string) => /(^|\s)then(\s|$)/.test(s);
1334
-
1335
- it('does not split a Japanese object marker (を) from its verb', () => {
1336
- // Without locale-aware boundary modifiers, `を` before the command verb
1337
- // `増加` was treated as a boundary, injecting a spurious `then`.
1338
- const result = new GrammarTransformer('ja', 'en').transform('#count を 増加');
1339
-
1340
- expect(hasStandaloneThen(result)).toBe(false);
1341
- expect(result).toBe('#count increment');
1342
- });
1343
-
1344
- it('does not split a Korean object marker (을) from its verb', () => {
1345
- const result = new GrammarTransformer('ko', 'en').transform('.active 을 토글');
1346
-
1347
- expect(hasStandaloneThen(result)).toBe(false);
1348
- expect(result).toBe('.active toggle');
1349
- });
1350
-
1351
- it('still keeps English base-set prepositions attached', () => {
1352
- // `by` is in the English base set; the argument after it must stay
1353
- // attached to the command rather than starting a new one.
1354
- const result = new GrammarTransformer('en', 'en').transform('increment #count by 2');
1355
-
1356
- expect(result).not.toContain('\n');
1357
- expect(hasStandaloneThen(result)).toBe(false);
1358
- expect(result).toBe('increment #count by 2');
1359
- });
1360
- });
1361
-
1362
- // =============================================================================
1363
- // Has/Have Operator Translation Tests
1364
- // =============================================================================
1365
-
1366
- describe('Has/Have Operator Translations', () => {
1367
- describe('Dictionary Entries', () => {
1368
- // Import dictionaries to verify has/have entries exist
1369
- it('should have has/have in English dictionary', async () => {
1370
- const { en } = await import('../dictionaries/en');
1371
- expect(en.logical.has).toBe('has');
1372
- expect(en.logical.have).toBe('have');
1373
- });
1374
-
1375
- it('should have has/have in Spanish dictionary', async () => {
1376
- const { es } = await import('../dictionaries/es');
1377
- expect(es.logical.has).toBe('tiene'); // third-person
1378
- expect(es.logical.have).toBe('tengo'); // first-person
1379
- });
1380
-
1381
- it('should have has/have in Japanese dictionary', async () => {
1382
- const { ja } = await import('../dictionaries/ja');
1383
- expect(ja.logical.has).toBe('ある');
1384
- expect(ja.logical.have).toBe('ある');
1385
- });
1386
-
1387
- it('should have has/have in German dictionary', async () => {
1388
- const { de } = await import('../dictionaries/de');
1389
- expect(de.logical.has).toBe('hat'); // third-person
1390
- expect(de.logical.have).toBe('habe'); // first-person
1391
- });
1392
-
1393
- it('should have has/have in French dictionary', async () => {
1394
- const { fr } = await import('../dictionaries/fr');
1395
- expect(fr.logical.has).toBe('a'); // third-person
1396
- expect(fr.logical.have).toBe('ai'); // first-person
1397
- });
1398
-
1399
- it('should have has/have in Korean dictionary', async () => {
1400
- const { ko } = await import('../dictionaries/ko');
1401
- expect(ko.logical.has).toBe('있다');
1402
- expect(ko.logical.have).toBe('있다');
1403
- });
1404
-
1405
- it('should have has/have in Chinese dictionary', async () => {
1406
- const { zh } = await import('../dictionaries/zh');
1407
- expect(zh.logical.has).toBe('有');
1408
- expect(zh.logical.have).toBe('有');
1409
- });
1410
-
1411
- it('should have has/have in Arabic dictionary', async () => {
1412
- const { ar } = await import('../dictionaries/ar');
1413
- expect(ar.logical.has).toBe('لديه'); // third-person
1414
- expect(ar.logical.have).toBe('لدي'); // first-person
1415
- });
1416
- });
1417
-
1418
- describe('Conjugating Languages', () => {
1419
- // Languages that have different forms for has (3rd person) vs have (1st person)
1420
- const conjugatingLanguages = [
1421
- { code: 'es', has: 'tiene', have: 'tengo' },
1422
- { code: 'de', has: 'hat', have: 'habe' },
1423
- { code: 'fr', has: 'a', have: 'ai' },
1424
- { code: 'pt', has: 'tem', have: 'tenho' },
1425
- { code: 'it', has: 'ha', have: 'ho' },
1426
- { code: 'pl', has: 'ma', have: 'mam' },
1427
- ];
1428
-
1429
- for (const lang of conjugatingLanguages) {
1430
- it(`should have different has/have forms in ${lang.code}`, async () => {
1431
- const dict = await import(`../dictionaries/${lang.code}`);
1432
- const dictionary = Object.values(dict)[0] as { logical: { has: string; have: string } };
1433
- expect(dictionary.logical.has).toBe(lang.has);
1434
- expect(dictionary.logical.have).toBe(lang.have);
1435
- });
1436
- }
1437
- });
1438
-
1439
- describe('Non-Conjugating Languages', () => {
1440
- // Languages that use the same form for both has and have
1441
- const sameFormLanguages = [
1442
- { code: 'ja', form: 'ある' },
1443
- { code: 'ko', form: '있다' },
1444
- { code: 'zh', form: '有' },
1445
- { code: 'tr', form: 'var' },
1446
- { code: 'id', form: 'punya' },
1447
- { code: 'vi', form: 'có' },
1448
- { code: 'th', form: 'มี' },
1449
- { code: 'tl', form: 'may' },
1450
- { code: 'ms', form: 'ada' },
1451
- ];
1452
-
1453
- for (const lang of sameFormLanguages) {
1454
- it(`should have same has/have form in ${lang.code}`, async () => {
1455
- const dict = await import(`../dictionaries/${lang.code}`);
1456
- const dictionary = Object.values(dict)[0] as { logical: { has: string; have: string } };
1457
- expect(dictionary.logical.has).toBe(lang.form);
1458
- expect(dictionary.logical.have).toBe(lang.form);
1459
- });
1460
- }
1461
- });
1462
- });
1463
-
1464
- // =============================================================================
1465
- // Possessive Dot Notation Translation Tests
1466
- // =============================================================================
1467
-
1468
- describe('Possessive Dot Notation Translation', () => {
1469
- describe('my.property patterns across languages', () => {
1470
- it('should translate my.textContent to Spanish', () => {
1471
- const transformer = new GrammarTransformer('en', 'es');
1472
- const result = transformer.transform('set my.textContent to "Done!"');
1473
- expect(result).toContain('mi.textContent');
1474
- expect(result).not.toContain('my.textContent');
1475
- });
1476
-
1477
- it('should translate my.textContent to Japanese', () => {
1478
- const transformer = new GrammarTransformer('en', 'ja');
1479
- const result = transformer.transform('set my.textContent to "Done!"');
1480
- expect(result).toContain('私の.textContent');
1481
- });
1482
-
1483
- it('should translate my.textContent to German', () => {
1484
- const transformer = new GrammarTransformer('en', 'de');
1485
- const result = transformer.transform('set my.textContent to "Done!"');
1486
- expect(result).toContain('mein.textContent');
1487
- });
1488
-
1489
- it('should translate my.textContent to Korean', () => {
1490
- const transformer = new GrammarTransformer('en', 'ko');
1491
- const result = transformer.transform('set my.textContent to "Done!"');
1492
- expect(result).toContain('내.textContent');
1493
- });
1494
-
1495
- it('should translate my.textContent to Chinese', () => {
1496
- const transformer = new GrammarTransformer('en', 'zh');
1497
- const result = transformer.transform('set my.textContent to "Done!"');
1498
- expect(result).toContain('我的.textContent');
1499
- });
1500
-
1501
- it('should translate my.textContent to Turkish', () => {
1502
- const transformer = new GrammarTransformer('en', 'tr');
1503
- const result = transformer.transform('set my.textContent to "Done!"');
1504
- expect(result).toContain('benim.textContent');
1505
- });
1506
-
1507
- it('should translate my.textContent to Arabic', () => {
1508
- const transformer = new GrammarTransformer('en', 'ar');
1509
- const result = transformer.transform('set my.textContent to "Done!"');
1510
- expect(result).toContain('لي.textContent');
1511
- });
1512
- });
1513
-
1514
- describe('its.property and your.property patterns', () => {
1515
- it('should translate its.value to Spanish', () => {
1516
- const transformer = new GrammarTransformer('en', 'es');
1517
- const result = transformer.transform('get its.value');
1518
- expect(result).toContain('su.value');
1519
- });
1520
-
1521
- it('should translate your.name to Spanish', () => {
1522
- const transformer = new GrammarTransformer('en', 'es');
1523
- const result = transformer.transform('log your.name');
1524
- expect(result).toContain('tu.name');
1525
- });
1526
- });
1527
-
1528
- describe('pronoun dot notation (me., it., you.)', () => {
1529
- it('should translate me.textContent to Spanish', () => {
1530
- const transformer = new GrammarTransformer('en', 'es');
1531
- const result = transformer.transform('set me.textContent to "Done!"');
1532
- expect(result).toContain('mi.textContent');
1533
- });
1534
- });
1535
-
1536
- describe('optional chaining (?.)', () => {
1537
- it('should translate my?.textContent to Spanish', () => {
1538
- const transformer = new GrammarTransformer('en', 'es');
1539
- const result = transformer.transform('log my?.textContent');
1540
- expect(result).toContain('mi?.textContent');
1541
- });
1542
- });
1543
-
1544
- describe('chained access', () => {
1545
- it('should only translate the possessive prefix in my.value.toUpperCase()', () => {
1546
- const transformer = new GrammarTransformer('en', 'es');
1547
- const result = transformer.transform('put my.value.toUpperCase() into #output');
1548
- expect(result).toContain('mi.value.toUpperCase()');
1549
- });
1550
- });
1551
-
1552
- describe('backward compatibility', () => {
1553
- it('should still translate space-separated possessives', () => {
1554
- const transformer = new GrammarTransformer('en', 'es');
1555
- const result = transformer.transform('set my textContent to "Done!"');
1556
- expect(result).toContain('mi');
1557
- expect(result).not.toMatch(/\bmy\b/);
1558
- });
1559
- });
1560
-
1561
- // ──── Reactive `live` block scope ────
1562
- // Before this fix, the splitter cut between `live` and the body's
1563
- // first command keyword, then the join logic re-inserted the target
1564
- // language's `then` keyword between them. E.g.
1565
- // `live put X into me end` (en→de) became
1566
- // `live dann setzen X zu ich ende` (spurious `dann`).
1567
- // The fix tracks block depth from `live` to its matching `end` and
1568
- // suppresses splits inside that scope.
1569
- describe('live block does not spuriously inject "then"', () => {
1570
- const cases: Array<[string, string, RegExp]> = [
1571
- // [target lang, input, banned pattern]
1572
- ['de', 'live put $count into me end', /\bdann\b/i],
1573
- ['es', 'live put $count into me end', /\bentonces\b/i],
1574
- ['ja', 'live put $count into me end', /それから/],
1575
- ['tr', 'live put $count into me end', /\bsonra\b/i],
1576
- ['it', 'live put $count into me end', /\ballora\b/i],
1577
- ['ru', 'live put $count into me end', /\bзатем\b/i],
1578
- ['vi', 'live put $count into me end', /\brồi\b/i],
1579
- ];
1580
-
1581
- for (const [target, input, banned] of cases) {
1582
- it(`(${target}) does not inject "then" inside the body`, () => {
1583
- const transformer = new GrammarTransformer('en', target);
1584
- const result = transformer.transform(input);
1585
- expect(result, `unexpected ${banned} in: ${result}`).not.toMatch(banned);
1586
- });
1587
- }
1588
-
1589
- it('keeps live body as one unit — no `live then` artifact', () => {
1590
- const transformer = new GrammarTransformer('en', 'en');
1591
- const result = transformer.transform('live put $count into me end');
1592
- expect(result).not.toMatch(/live\s+then\b/i);
1593
- expect(result).toMatch(/^live\b/);
1594
- expect(result).toMatch(/\bend$/);
1595
- });
1596
-
1597
- it('still splits on explicit "then" OUTSIDE the live block', () => {
1598
- // Legitimate `then`-based chains MUST still split.
1599
- const transformer = new GrammarTransformer('en', 'de');
1600
- const result = transformer.transform('live put $x into me end then toggle .active');
1601
- // German "then" = "dann" — appears exactly once (between block + toggle).
1602
- const occurrences = (result.match(/\bdann\b/gi) || []).length;
1603
- expect(occurrences).toBe(1);
1604
- });
1605
- });
1606
-
1607
- // ──── Malay `socket` keyword is translated to native `soket` ────
1608
- // The ms dictionary was missing the `socket` command entry, so the
1609
- // transformer emitted the English literal `socket`. The semantic ms
1610
- // profile maps `socket` to its native primary `soket` (not the English
1611
- // form), so the untranslated `socket` token tokenized as a bare
1612
- // identifier and the `socket` block command was dropped — `socket-basic`
1613
- // parsed as a degenerate `put`. (es only worked by coincidence: its
1614
- // profile's socket.primary IS the English literal.) Fix: add
1615
- // `socket: 'soket'` to the ms dictionary, mirroring ja `socket: ソケット`.
1616
- describe('Malay socket command translates to native soket', () => {
1617
- it('(ms) emits soket, not the English literal socket', () => {
1618
- const result = new GrammarTransformer('en', 'ms').transform(
1619
- 'socket ChatSocket ws://localhost:8080 on message put it into #chat end'
1620
- );
1621
- expect(result, `expected native soket in: ${result}`).toMatch(/\bsoket\b/);
1622
- expect(result, `English socket leaked in: ${result}`).not.toMatch(/\bsocket\b/);
1623
- });
1624
- });
1625
-
1626
- // ──── ru/uk install keyword is the loanword, not the set homonym ────
1627
- // ru "install" and "set" are both `установить` (uk: `встановити`). The dict
1628
- // emitted plain `установить` for install, which the semantic parser resolves to
1629
- // `set` (the install action dropped → install-behavior degenerate). The install
1630
- // command now uses the single-token loanword `инсталлировать` (ru) /
1631
- // `інсталювати` (uk), distinct from the set primary.
1632
- describe('ru/uk install command uses the loanword, not the set homonym', () => {
1633
- // NB: substring (not /\b…\b/) — JS word boundaries are ASCII-only and never
1634
- // match adjacent to Cyrillic text.
1635
- const cases: Array<[string, string, string]> = [
1636
- // [lang, expected install loanword, the set homonym that must NOT appear]
1637
- ['ru', 'инсталлировать', 'установить'],
1638
- ['uk', 'інсталювати', 'встановити'],
1639
- ];
1640
- for (const [lang, want, banned] of cases) {
1641
- it(`(${lang}) emits the install loanword, not the set homonym`, () => {
1642
- const result = new GrammarTransformer('en', lang).transform('install Draggable');
1643
- expect(result, `expected install loanword in: ${result}`).toContain(want);
1644
- expect(result, `set homonym leaked in: ${result}`).not.toContain(banned);
1645
- });
1646
- }
1647
- });
1648
-
1649
- // ──── Block extraction for `when`, `unless`, and SOV `live` ────
1650
- // Block-syntactic tokens are pulled out before parseStatement so
1651
- // they don't end up as command verbs (`live` → action role) or get
1652
- // swept into role values (`end` → destination tail). The body
1653
- // recurses through the regular pipeline, which keeps SOV reorder
1654
- // working *inside* the body without disturbing the block frame.
1655
- describe('reactive blocks — when/unless/live block extraction', () => {
1656
- it('`when X changes Y` does not truncate — body is translated (en→de)', () => {
1657
- const t = new GrammarTransformer('en', 'de');
1658
- const result = t.transform('when $count changes log $count');
1659
- // Pre-fix output was just `wenn` (parseConditional returned only
1660
- // the action role). Now we should see head + body translated.
1661
- expect(result.length).toBeGreaterThan('wenn'.length + 5);
1662
- expect(result).toMatch(/wenn/i);
1663
- expect(result).toMatch(/protokolliere|log/i);
1664
- expect(result).toMatch(/ändert|changes/i);
1665
- });
1666
-
1667
- it('`unless X Y` does not truncate — body is translated (en→de)', () => {
1668
- const t = new GrammarTransformer('en', 'de');
1669
- const result = t.transform('unless $disabled add .ready to me');
1670
- expect(result.length).toBeGreaterThan('wennnicht'.length + 5);
1671
- expect(result).toMatch(/wennnicht|unless/i);
1672
- expect(result).toMatch(/hinzufüg|add/i);
1673
- });
1674
-
1675
- it('`live X end` keeps live at start and end at end (SOV: ja)', () => {
1676
- const t = new GrammarTransformer('en', 'ja');
1677
- const result = t.transform('live put $count into me end');
1678
- // Block frame: head first, tail last. Pre-fix the SOV reorder
1679
- // dragged `live` to the end as if it were the action verb.
1680
- expect(result).toMatch(/^(live|ライブ)/);
1681
- expect(result).toMatch(/(end|終わり)$/);
1682
- // SOV reorder applies inside the body — destination marker に
1683
- // should appear, indicating `me` was reordered with its postposition.
1684
- expect(result).toMatch(/に/);
1685
- });
1686
-
1687
- it('`live X end` keeps live at start and end at end (SOV: tr)', () => {
1688
- const t = new GrammarTransformer('en', 'tr');
1689
- const result = t.transform('live put $count into me end');
1690
- expect(result).toMatch(/^(live|canlı)/i);
1691
- expect(result).toMatch(/(end|son)$/i);
1692
- });
1693
-
1694
- it('regression: `if X then Y end` still works (no block extraction)', () => {
1695
- // `if` is intentionally not a BLOCK_HEAD_KEYWORD — splitOnThen +
1696
- // parseConditional already handle it. Verify we didn't regress.
1697
- const t = new GrammarTransformer('en', 'de');
1698
- const result = t.transform('if $x then increment $count end');
1699
- // de `if` emits `falls` (the profile's `if` primary). `wenn` was the old dict
1700
- // value but collides with the profile's `when` keyword, so the conditional
1701
- // never formed — aligned to `falls` (see de dict if-keyword alignment, A1).
1702
- expect(result).toMatch(/falls|wenn|if/i);
1703
- expect(result).toMatch(/dann|then/i);
1704
- expect(result).toMatch(/erhöh|increment/i);
1705
- });
1706
-
1707
- it('regression: `live X end then toggle .y` — exactly one "then" connector', () => {
1708
- // Splitter must still recognize the `then` *outside* the live
1709
- // block as a statement boundary, even now that we route blocks
1710
- // around parseStatement.
1711
- const t = new GrammarTransformer('en', 'de');
1712
- const result = t.transform('live put $x into me end then toggle .active');
1713
- const danns = (result.match(/\bdann\b/gi) || []).length;
1714
- expect(danns).toBe(1);
1715
- });
1716
- });
1717
- });
1718
-
1719
- describe('Caret-scoped variable read masking (`^name on <selector>`)', () => {
1720
- // `put ^count on #host into me` carries a second, overloaded `on` (the caret
1721
- // scope). The transformer masks ` on <selector>` so the splitter/event parser
1722
- // doesn't mistake it for an event/command boundary: the event clause survives
1723
- // and `^count on #host` stays adjacent. See caret-var-on-target in the roadmap.
1724
- it('keeps `^count on #host` together and preserves the event (ar)', () => {
1725
- const t = new GrammarTransformer('en', 'ar');
1726
- const result = t.transform('on click put ^count on #host into me');
1727
- expect(result).toContain('^count on #host'); // scope kept adjacent
1728
- expect(result).toContain('نقر'); // event (click) preserved
1729
- expect(result).not.toContain(''); // no leftover mask sentinel
1730
- });
1731
-
1732
- it('does not disturb a normal command without a caret scope (ar)', () => {
1733
- const t = new GrammarTransformer('en', 'ar');
1734
- const result = t.transform('on click toggle .active on #button');
1735
- expect(result).not.toContain('');
1736
- expect(result).toMatch(/بدل|بدّل/);
1737
- });
1738
- });
1739
-
1740
- describe('Event-block body with `from <source>` (focus-trap)', () => {
1741
- const raw =
1742
- 'on keydown[key=="Tab"] from .modal if target matches last <button/> in .modal focus first <button/> in .modal halt end';
1743
-
1744
- it('routes SOV `from`-source heads through the block-body path (tr)', () => {
1745
- // The if-block body's inner keywords get translated (Turkish `odak`=focus,
1746
- // `ilk`=first) instead of leaking English — and the event clause leads.
1747
- const t = new GrammarTransformer('en', 'tr');
1748
- const result = t.transform(raw);
1749
- expect(result).toMatch(/odak/); // focus → odak (block body transformed)
1750
- expect(result).toMatch(/keydown/); // event preserved
1751
- // Event clause leads (keydown appears before the if/eğer block head).
1752
- expect(result.indexOf('keydown')).toBeLessThan(result.search(/eğer/));
1753
- });
1754
-
1755
- it('keeps VSO `from`-source heads on the existing path (ar unchanged)', () => {
1756
- // VSO event-first emission with a `from` source reorders incorrectly, so ar
1757
- // stays on the existing path. Guard: transform still succeeds and translates
1758
- // the verb (`durdur`-style halt / `أوقف`), without throwing.
1759
- const t = new GrammarTransformer('en', 'ar');
1760
- expect(() => t.transform(raw)).not.toThrow();
1761
- expect(t.transform(raw).length).toBeGreaterThan(0);
1762
- });
1763
- });
1764
-
1765
- describe('if/else block-body — else split + translation (Track 5 Tier 1)', () => {
1766
- // The if-block body was reordered as one stream, so `else` rode along glued to a
1767
- // selector-led clause (marked a selector → left UNTRANSLATED) with a spurious
1768
- // `then` inserted around it. The body is now split at a top-level `else` into a
1769
- // then-branch and an else-branch, each transformed independently, and `else` is
1770
- // translated. See docs-internal/MULTILINGUAL_ROADMAP.md (Track 5 Tier 1).
1771
- const raw = 'on click if #modal exists show #modal else make a <div#modal/> put it into body end';
1772
-
1773
- it('[ar] translates else to وإلا (no English else leaks)', () => {
1774
- const result = new GrammarTransformer('en', 'ar').transform(raw);
1775
- expect(result).toContain('وإلا');
1776
- expect(result).not.toMatch(/\belse\b/);
1777
- });
1778
-
1779
- it('[it] translates else to altrimenti (no English else leaks)', () => {
1780
- const result = new GrammarTransformer('en', 'it').transform(raw);
1781
- expect(result).toContain('altrimenti');
1782
- expect(result).not.toMatch(/\belse\b/);
1783
- });
1784
-
1785
- it('[ja] translates else to そうでなければ (no English else leaks)', () => {
1786
- const result = new GrammarTransformer('en', 'ja').transform(raw);
1787
- expect(result).toContain('そうでなければ');
1788
- expect(result).not.toMatch(/\belse\b/);
1789
- });
1790
-
1791
- it('leaves an else-less if-block body unchanged in shape', () => {
1792
- // No `else` → single body transform path, no spurious split.
1793
- const noElse = 'on click if #modal exists show #modal end';
1794
- const result = new GrammarTransformer('en', 'ar').transform(noElse);
1795
- expect(result).not.toMatch(/\belse\b/);
1796
- expect(result).toContain('اظهر'); // show translated
1797
- });
1798
- });
1799
-
1800
- describe('SOV modifier-prefixed event body reorder (Track 5)', () => {
1801
- // A leading command-modifier (async/once/debounced) must not be parsed as the
1802
- // event handler's action. For SOV targets that mis-assignment surfaced the real
1803
- // verb first on reorder (`取得 /api/data を クリック …`), which the semantic parser
1804
- // collapsed to a bare `*-generated-verb-first` command (degenerate). The
1805
- // transformer now lifts the modifier out and re-emits it as a leading English
1806
- // literal, keeping the body in canonical patient-first SOV order so the event
1807
- // sits mid-stream and the parser's SOV event-extraction recovers the full body.
1808
- // See docs-internal/SOV_REORDER_SCOPE.md.
1809
-
1810
- for (const lang of ['ja', 'ko', 'tr'] as const) {
1811
- const t = new GrammarTransformer('en', lang);
1812
-
1813
- it(`[${lang}] async body: modifier leads, real verb is not first`, () => {
1814
- const out = t.transform('on click async fetch /api/data then put it into me');
1815
- // The English modifier literal leads (the parser strips it pre-parse).
1816
- expect(out.startsWith('async ')).toBe(true);
1817
- // The patient precedes the fetch verb — the body stays patient-first, so the
1818
- // verb is not the leading body token (which is what caused the degenerate parse).
1819
- expect(out.indexOf('/api/data')).toBeLessThan(out.length);
1820
- expect(out).toContain('/api/data');
1821
- });
1822
-
1823
- it(`[${lang}] once body: modifier leads and the patient survives`, () => {
1824
- const out = t.transform('on click once add .initialized to me call setup()');
1825
- expect(out.startsWith('once ')).toBe(true);
1826
- expect(out).toContain('.initialized');
1827
- expect(out).toContain('setup()');
1828
- });
1829
-
1830
- it(`[${lang}] debounced at N: modifier phrase leads intact`, () => {
1831
- const out = t.transform(
1832
- 'on keyup debounced at 300ms fetch /api/search then put it into #results'
1833
- );
1834
- expect(out.startsWith('debounced at 300ms ')).toBe(true);
1835
- });
1836
- }
1837
-
1838
- it('[es] SVO target is unaffected — modifier is not relocated to the front', () => {
1839
- const out = new GrammarTransformer('en', 'es').transform(
1840
- 'on click async fetch /api/data then put it into me'
1841
- );
1842
- // SVO keeps the body in an order the parser already handles, so the gate leaves
1843
- // it byte-identical: the handler still leads with the (translated) event clause,
1844
- // not a relocated bare `async` literal.
1845
- expect(out.startsWith('async ')).toBe(false);
1846
- });
1847
-
1848
- it('[ja] a simple handler without a modifier is unchanged', () => {
1849
- const t = new GrammarTransformer('en', 'ja');
1850
- expect(t.transform('on click toggle .active')).toBe(t.transform('on click toggle .active'));
1851
- const out = t.transform('on click toggle .active');
1852
- expect(out.startsWith('async ')).toBe(false);
1853
- expect(out.startsWith('once ')).toBe(false);
1854
- expect(out).toContain('.active');
1855
- });
1856
- });
1857
-
1858
- describe('SOV put-into verb-final reorder (Track 5)', () => {
1859
- // ko/tr/bn lacked ja's `put-into` rule, so `put X into Y` reordered to
1860
- // verb-middle (`X i koy Y e`) which the semantic parser can't match. The rule
1861
- // (gated to standalone put via a no-event predicate) emits verb-final order.
1862
- const verbFinal: Array<[string, string]> = [
1863
- ['tr', 'koy'],
1864
- ['ko', '넣다'],
1865
- ['bn', 'রাখুন'],
1866
- ];
1867
- for (const [lang, verb] of verbFinal) {
1868
- it(`[${lang}] standalone put is verb-final`, () => {
1869
- const out = new GrammarTransformer('en', lang).transform('put it into me');
1870
- // The verb is the last token (patient, destination, then verb).
1871
- expect(out.trim().endsWith(verb)).toBe(true);
1872
- });
1873
- }
1874
-
1875
- it('[tr] event-handler `put` keeps the event before the verb (predicate gate)', () => {
1876
- // The no-event predicate excludes event handlers, so the event clause is not
1877
- // pushed past the verb (which would strand it from the parser).
1878
- const out = new GrammarTransformer('en', 'tr').transform(
1879
- 'on success put event.detail.message into #sr-announce'
1880
- );
1881
- // `koy` (put) must not be verb-final here — the event (`success`) follows it.
1882
- expect(out.trim().endsWith('koy')).toBe(false);
1883
- expect(out).toMatch(/koy.*success|success.*koy/);
1884
- });
1885
- });
1886
-
1887
- // =============================================================================
1888
- // Destination `on` vs event `on` (bucket 1 — dual-`on`)
1889
- // =============================================================================
1890
- //
1891
- // `toggle X on Y` / `set @attr on Y` reuses the word `on` as a *locative
1892
- // target* preposition. But `on` is also the event-handler head keyword
1893
- // (`commands.on = 'on'` in the EN dictionary). Two bugs resulted:
1894
- //
1895
- // 1. SPLIT bug — `splitOnCommandBoundaries` treated the destination `on`
1896
- // as a command boundary (it's in `commandKeywords`) and split there,
1897
- // so the join re-inserted a spurious `then` (ثم / pagkatapos / entonces)
1898
- // and a dangling `on Y` clause.
1899
- // 2. ROLE bug — once kept whole, the argument parser mapped the locative
1900
- // `on` to the `event` role (EN profile marks `on → event`), overwriting
1901
- // the already-captured head event — dropping the trigger entirely.
1902
- //
1903
- // The combined effect was garbage like
1904
- // `on click toggle .open on #menu` → (ar) `بدل .open عند نقر ثم عند #menu`
1905
- // which silently dropped `#menu` on a round-trip back to English
1906
- // (`on click toggle .open`). The fix keeps the statement whole and routes
1907
- // the locative `on` to `destination`, matching how the semantic parser
1908
- // itself models `toggle .open on #menu` (patient `.open` + destination
1909
- // `#menu`).
1910
- describe('destination `on` is not confused with the event head `on`', () => {
1911
- describe('parse: role assignment', () => {
1912
- it('assigns the locative `on` target to destination, keeps the head event', () => {
1913
- const parsed = parseStatement('on click toggle @hidden on #panel', 'en');
1914
- expect(parsed).not.toBeNull();
1915
- expect(parsed!.roles.get('event')?.value).toBe('click');
1916
- expect(parsed!.roles.get('action')?.value).toBe('toggle');
1917
- expect(parsed!.roles.get('patient')?.value).toBe('@hidden');
1918
- // The destination `on #panel` must land in `destination`, NOT clobber `event`.
1919
- expect(parsed!.roles.get('destination')?.value).toBe('#panel');
1920
- });
1921
-
1922
- it('handles a bare command (no event handler): `toggle .active on me`', () => {
1923
- const parsed = parseStatement('toggle .active on me', 'en');
1924
- expect(parsed).not.toBeNull();
1925
- expect(parsed!.roles.get('action')?.value).toBe('toggle');
1926
- expect(parsed!.roles.get('patient')?.value).toBe('.active');
1927
- expect(parsed!.roles.get('destination')?.value).toBe('me');
1928
- // No bogus event role from the locative `on`.
1929
- expect(parsed!.roles.has('event')).toBe(false);
1930
- });
1931
- });
1932
-
1933
- describe('transform: no spurious "then" injected', () => {
1934
- // [target lang, banned "then" keyword]
1935
- const cases: Array<[string, RegExp]> = [
1936
- ['ar', /\bثم\b/],
1937
- ['tl', /\bpagkatapos\b/i],
1938
- ['es', /\bentonces\b/i],
1939
- ['ja', /それから/],
1940
- ['ko', /그러면/],
1941
- ['zh', /那么/],
1942
- ['he', /\bאז\b/],
1943
- ];
1944
- for (const [lang, banned] of cases) {
1945
- it(`(${lang}) keeps "toggle X on Y" as one statement`, () => {
1946
- const t = new GrammarTransformer('en', lang);
1947
- const result = t.transform('on click toggle .open on #menu');
1948
- expect(result, `unexpected then in: ${result}`).not.toMatch(banned);
1949
- // Both selectors must still be present at the surface — the
1950
- // destination is no longer split off into a dropped clause.
1951
- expect(result).toContain('.open');
1952
- expect(result).toContain('#menu');
1953
- });
1954
- }
1955
- });
1956
-
1957
- describe('round-trip: destination + event survive (semantic correctness)', () => {
1958
- // SVO / VSO / RTL languages place the destination before the verb (or
1959
- // after the verb but pre-event, for VSO), which the round-trip recovers
1960
- // losslessly. SOV destination-after-verb placement is a separate,
1961
- // pre-existing limitation (it also drops `#bar` for `add .foo to #bar`)
1962
- // and is intentionally out of scope here.
1963
- //
1964
- // `zh` is excluded from the *keyword* round-trip: its i18n→en reverse path
1965
- // has a pre-existing quirk where `切换`/`把` (BA construction) mangles the
1966
- // verb (`on click toggle .active` → `on click h .active`), independent of
1967
- // the locative `on`. zh destination preservation is still asserted by the
1968
- // "no spurious then" case above, and verified semantically via the parser's
1969
- // own round-trip (render) outside this unit suite.
1970
- const losslessLangs = ['ar', 'tl', 'es', 'he', 'fr', 'de', 'pt', 'it'];
1971
- for (const lang of losslessLangs) {
1972
- it(`(${lang}) en→${lang}→en preserves event, action, patient, destination`, () => {
1973
- const out = toLocale('on click toggle .open on #menu', lang);
1974
- const back = toEnglish(out, lang);
1975
- expect(back).toContain('click');
1976
- expect(back).toContain('toggle');
1977
- expect(back).toContain('.open');
1978
- expect(back).toContain('#menu'); // the destination must NOT be dropped
1979
- });
1980
- }
1981
-
1982
- it('regression: the OLD garbage form would have dropped the destination', () => {
1983
- // Sanity anchor — the corrected output keeps `#menu`.
1984
- const out = toLocale('on click toggle .open on #menu', 'ar');
1985
- expect(out).toContain('#menu');
1986
- expect(out).not.toMatch(/\bثم\b/); // no spurious "then"
1987
- });
1988
- });
1989
-
1990
- describe('does not disturb non-locative-`on` patterns', () => {
1991
- // The remap only touches the `on` token in argument position; other
1992
- // prepositions (to/into/from/by) and plain event handlers are unchanged.
1993
- const unchanged = [
1994
- 'on click increment #count',
1995
- 'add .foo to #bar',
1996
- 'remove .x from #y',
1997
- 'on click toggle .active',
1998
- ];
1999
- for (const input of unchanged) {
2000
- it(`(${input}) round-trips through Spanish unchanged in structure`, () => {
2001
- const back = toEnglish(toLocale(input, 'es'), 'es');
2002
- // First word (command/head) preserved
2003
- expect(back.split(/\s+/)[0]).toBe(input.split(/\s+/)[0]);
2004
- // No spurious "then"/"on" artifacts
2005
- expect(back).not.toMatch(/\bthen\b/);
2006
- });
2007
- }
2008
- });
2009
- });
2010
-
2011
- describe('ko event marker 할 때 + set `on <scope>` capture (S1 tabs-aria)', () => {
2012
- // koreanProfile gained the event-role marker 할 때 — the semantic
2013
- // *-event-ko-sov-* patterns anchor on it; before, every ko handler emitted a
2014
- // bare event name no fused pattern could match. A SELECTOR-shaped "event" (the
2015
- // dangling target of a locative `on`) must NOT receive that marker, or the
2016
- // emission grows a spurious mid-stream event anchor (`#sr-announce 할 때` / ja
2017
- // `#sr-announce で`). For `set @role to "alert" on #sr-announce` the locative
2018
- // `on` is now the set's SCOPE (S1): the transformer keeps it attached and
2019
- // positions it before the clause-final verb (`on #sr-announce 설정` / `設定`),
2020
- // so the scope is captured rather than dropped — and there is still exactly ONE
2021
- // event marker (the real `success`), never a spurious one on the selector.
2022
- it('[ko] a real event gets the marker', () => {
2023
- const t = new GrammarTransformer('en', 'ko');
2024
- expect(t.transform('on click increment #counter')).toBe('#counter 를 클릭 할 때 증가');
2025
- });
2026
-
2027
- it('[ko] set `on <scope>` is captured, with no spurious event marker', () => {
2028
- const t = new GrammarTransformer('en', 'ko');
2029
- const out = t.transform(
2030
- 'on success put event.detail.message into #sr-announce set @role to "alert" on #sr-announce'
2031
- );
2032
- // The real event keeps its marker…
2033
- expect(out).toContain('success 할 때');
2034
- // …and it is the ONLY event marker (the locative `on` is the set's scope,
2035
- // not a second event anchor).
2036
- expect(out.match(/할 때/g)?.length).toBe(1);
2037
- // The set's scope is emitted (passthrough `on`) before the clause-final verb.
2038
- expect(out).toContain('on #sr-announce 설정');
2039
- });
2040
-
2041
- it('[ja] set `on <scope>` is captured before the verb, no spurious で', () => {
2042
- const t = new GrammarTransformer('en', 'ja');
2043
- const out = t.transform(
2044
- 'on success put event.detail.message into #sr-announce set @role to "alert" on #sr-announce'
2045
- );
2046
- expect(out).toContain('on #sr-announce 設定');
2047
- });
2048
- });
2049
-
2050
- describe('command blur translates via commands.blur, not the blur EVENT word', () => {
2051
- // de/fr/pt/pl/sw dicts had blur only in the EVENTS section (unscharf/flou/
2052
- // desfoque/rozmycie/poteza_macho); the COMMAND `blur me` fell back to that
2053
- // event word, which no semantic profile reads as the blur verb — blur-element
2054
- // was lossy in all five. commands.blur now emits the profile's verb.
2055
- const cases: Array<[string, string]> = [
2056
- ['de', 'defokussieren'],
2057
- ['fr', 'défocaliser'],
2058
- ['pt', 'desfocar'],
2059
- ['pl', 'rozmyj'],
2060
- ['sw', 'blur'],
2061
- ];
2062
- for (const [lang, verb] of cases) {
2063
- it(`[${lang}] blur me emits ${verb}`, () => {
2064
- const t = new GrammarTransformer('en', lang);
2065
- const out = t.transform('on keydown[key=="Escape"] blur me');
2066
- expect(out).toContain(verb);
2067
- });
2068
- }
2069
-
2070
- it('[de] the blur EVENT now also emits the command word (shadowing, parse-safe)', () => {
2071
- // commands.blur shadows events.blur in event-name translation too. That is
2072
- // accepted: defokussieren is in the semantic eventNameTranslations for de,
2073
- // so `bei defokussieren …` still anchors the handler (gate green), and the
2074
- // command/event senses can never diverge again.
2075
- const t = new GrammarTransformer('en', 'de');
2076
- const out = t.transform('on blur add .error to me');
2077
- expect(out).toContain('defokussieren');
2078
- });
2079
- });
2080
-
2081
- describe('trigger/send `on <target>` keeps its target — no spurious then (behavior-sortable)', () => {
2082
- // `trigger X on me` / `send X on me` fire an event on a TARGET element. `on`
2083
- // was treated as a command boundary (not in ON_TARGET_COMMANDS), so the
2084
- // statement split into `trigger X` | `on me`, the line-join re-inserted the
2085
- // target language's `then` (`disparar sortable:start entonces en yo`), and the
2086
- // dangling `then` glued the FOLLOWING `repeat until event …` loop into a
2087
- // then-chain — dropping `repeat`/`wait`. This kept behavior-sortable lossy
2088
- // (fid 0.778) in every SVO language. `trigger`/`send` are now in
2089
- // ON_TARGET_COMMANDS so `on <target>` stays attached. Guards the i18n half of
2090
- // the sortable arc (semantic gate parse fidelity is guarded by the baseline).
2091
- const cases: Array<[string, string]> = [
2092
- ['es', 'yo'],
2093
- ['fr', 'moi'],
2094
- ['de', 'ich'],
2095
- ['it', 'io'],
2096
- ];
2097
- for (const [lang, pronoun] of cases) {
2098
- it(`[${lang}] trigger sortable:start on me — single statement, target preserved, no then`, () => {
2099
- const out = new GrammarTransformer('en', lang).transform('trigger sortable:start on me');
2100
- // The event name survives and the target pronoun is preserved.
2101
- expect(out).toContain('sortable:start');
2102
- expect(out.trim().endsWith(pronoun)).toBe(true);
2103
- // No split: exactly `<verb> sortable:start <marker> <pronoun>` (4 tokens).
2104
- // The old bug produced 5 (an extra `then` connective before the target).
2105
- expect(out.trim().split(/\s+/)).toHaveLength(4);
2106
- });
2107
-
2108
- it(`[${lang}] send foo:bar on me — target stays attached (no extra connective)`, () => {
2109
- const out = new GrammarTransformer('en', lang).transform('send foo:bar on me');
2110
- expect(out).toContain('foo:bar');
2111
- expect(out.trim().endsWith(pronoun)).toBe(true);
2112
- expect(out.trim().split(/\s+/)).toHaveLength(4);
2113
- });
2114
- }
2115
- });
2116
-
2117
- describe('Hebrew fronted accusative marker is repaired (he add-body att-fronting)', () => {
2118
- // An event-handler body that leads with a command-modifier (`on click once add …`)
2119
- // or is a control block (`on blur if … add … end`) could emit the accusative
2120
- // marker את AHEAD of the body command's verb — `add .error to me` rendering
2121
- // `… את הוסף .error …` instead of the canonical `… הוסף את .error …`. את before a
2122
- // verb is always ungrammatical Hebrew, and the semantic parser dropped the command
2123
- // in every parse path (fused-event, multi-clause, conditional-body), keeping he
2124
- // `if-empty` / `input-validation` / `event-once` lossy. transform() now repairs the
2125
- // `<accusative-marker> <verb>` adjacency back to `<verb> <accusative-marker>`.
2126
- const he = (src: string) => new GrammarTransformer('en', 'he').transform(src);
2127
-
2128
- it('[he] conditional-body add: את follows the verb (not fronted)', () => {
2129
- const out = he(
2130
- 'on blur if my value is empty add .error to me put "Required" into next .error-message end'
2131
- );
2132
- expect(out).toContain('הוסף את .error');
2133
- expect(out).not.toContain('את הוסף');
2134
- });
2135
-
2136
- it('[he] if/else-body add: את follows the verb', () => {
2137
- const out = he('on blur if my value is empty add .error to me else remove .error from me end');
2138
- expect(out).toContain('הוסף את .error');
2139
- expect(out).not.toContain('את הוסף');
2140
- });
2141
-
2142
- it('[he] modifier-prefixed (once) body add: את follows the verb', () => {
2143
- const out = he('on click once add .initialized to me call setup()');
2144
- expect(out).toContain('הוסף את .initialized');
2145
- expect(out).not.toContain('את הוסף');
2146
- });
2147
-
2148
- it('[he] a legitimate `<verb> את <obj>` form is left untouched', () => {
2149
- // send already emits `שלח את refresh` (verb then accusative); the repair must only
2150
- // swap the ungrammatical marker-then-verb adjacency, never a real `<verb> את`.
2151
- expect(he('send refresh to #widget')).toContain('שלח את refresh');
2152
- expect(he('add .error to me')).toContain('הוסף את .error');
2153
- });
2154
- });
2155
-
2156
- describe('Hebrew scroll/last command translations (he last-in-collection dict gap)', () => {
2157
- // `scroll` (command) and `last` (positional) were missing from the he i18n
2158
- // dictionary, so `scroll to last <.message/> in #chat` emitted English scroll/last
2159
- // that the semantic he parser dropped (last-in-collection lossy — scroll missing).
2160
- // Adding scroll→גלול and last→אחרון (both already recognized on the semantic side)
2161
- // makes it faithful. `in` is deliberately NOT translated: it would also rewrite the
2162
- // `for X in Y` loop iterator and break template-literal-list-build (the for-loop's
2163
- // English `in` already parses).
2164
- const he = (s: string) => new GrammarTransformer('en', 'he').transform(s);
2165
-
2166
- it('[he] scroll command translates to גלול', () => {
2167
- expect(he('scroll to last <.message/> in #chat')).toContain('גלול');
2168
- });
2169
-
2170
- it('[he] positional last translates to אחרון', () => {
2171
- expect(he('scroll to last <.message/> in #chat')).toContain('אחרון');
2172
- });
2173
-
2174
- it('[he] for-loop `in` is left English (not rewritten — guards template-literal-list-build)', () => {
2175
- expect(he('for item in $items log item')).toContain(' in ');
2176
- });
2177
- });
2178
-
2179
- describe('Hebrew event-handler inline `unless` guard (he unless-condition degenerate)', () => {
2180
- // `on click unless I match .disabled toggle .selected` is an inline guard with no
2181
- // `end`. parseEventHandler read `unless` as the action and swept the whole
2182
- // `<cond> <body>` tail into one patient blob; Hebrew then prefixed that blob with
2183
- // the accusative object marker את (`… אלא את I match .disabled מתג .selected`) and
2184
- // the inner toggle lost its own את. The semantic parser collapsed that (degenerate):
2185
- // את ahead of the condition blocks the `unless` pattern, AND the markerless `מתג
2186
- // .selected` fails the he toggle pattern (which requires את). Marker-less langs
2187
- // (de/it/ar/pl) parse the same role-blob faithfully — a Hebrew accusative-marker
2188
- // artifact, not a general gap. The guard now routes through the standalone block
2189
- // path (extractBlockStructure → transformBlock): condition kept marker-free, body
2190
- // command keeps its את. Needs the he dict `unless: אלא` entry too.
2191
- const he = (s: string) => new GrammarTransformer('en', 'he').transform(s);
2192
-
2193
- it('[he] unless guard: unless translates, condition is marker-free, toggle keeps its את', () => {
2194
- const out = he('on click unless I match .disabled toggle .selected');
2195
- expect(out).toContain('אלא I match .disabled'); // unless→אלא, no fronted את before the condition
2196
- expect(out).not.toContain('אלא את'); // condition is NOT object-marked
2197
- expect(out).toContain('מתג את .selected'); // body toggle keeps its accusative marker
2198
- expect(out).not.toContain('unless'); // no English leak
2199
- });
2200
-
2201
- it('[he] the event clause leads (SVO): `ב לחיצה` before the unless guard', () => {
2202
- const out = he('on click unless I match .disabled toggle .selected');
2203
- expect(out.indexOf('ב לחיצה')).toBeGreaterThanOrEqual(0);
2204
- expect(out.indexOf('ב לחיצה')).toBeLessThan(out.indexOf('אלא'));
2205
- });
2206
- });
2207
-
2208
- describe('Attached parenthesized arg list stays one token (tl behavior-resizable degenerate)', () => {
2209
- // The tokenizer tracked `<>` selectors and `[]` guards but not `()`, so an event
2210
- // destructure `pointerdown(clientX, clientY)` split at the comma-space into
2211
- // `pointerdown(clientX,` + `clientY)`. In the VSO from-first event-handler-head
2212
- // reorder the two halves were SEPARATED (event role = `pointerdown(clientX,`, the
2213
- // stray `clientY)` fronted) → `clientY) mula_sa ako kapag pointerdown(clientX,`,
2214
- // an unparseable head that dropped the whole `on pointerdown … end` handler
2215
- // (tl behavior-resizable DEGENERATE → {behavior}; ar lossy). Fix: track `(` as a
2216
- // depth scope so the group stays atomic. Since the cluster E fix, STANDALONE
2217
- // groups (`to ($count or 0)`) are atomic too — their interior keywords translate
2218
- // in place via translateMultiWordValue/translateWord's paren handling.
2219
- it('[tl] event-handler head with destructured params keeps the event atomic', () => {
2220
- const out = new GrammarTransformer('en', 'tl').transform(
2221
- 'on pointerdown(clientX, clientY) from me'
2222
- );
2223
- // The from-source fronts (VSO), but the event + its params stay one clean unit.
2224
- expect(out).toContain('pointerdown(clientX, clientY)');
2225
- // Before the fix the split half `clientY)` was fronted to the very start.
2226
- expect(out.startsWith('clientY)')).toBe(false);
2227
- });
2228
-
2229
- it('[ar] same head is not split at the comma either', () => {
2230
- const out = new GrammarTransformer('en', 'ar').transform(
2231
- 'on pointerdown(clientX, clientY) from me'
2232
- );
2233
- expect(out).toContain('pointerdown(clientX, clientY)');
2234
- });
2235
-
2236
- it('a standalone expression paren still translates its interior operators', () => {
2237
- // `($count or 0)` is NOT an attached call — the group is now atomic (cluster E),
2238
- // but its `or` must still translate in place. Tagalog renders `or` → `o`.
2239
- const out = new GrammarTransformer('en', 'tl').transform('set $x to ($count or 0)');
2240
- expect(out).toContain('($count o 0)'); // atomic AND interior-translated
2241
- });
2242
- });
2243
-
2244
- describe('Standalone parenthesized expressions are opaque units (R1 cluster E, computed-value)', () => {
2245
- // With standalone `(` untracked, `(the value of #price as Number)` tokenized
2246
- // LOOSE and its interior `of`/`as` keywords hit the argument modifier map: the
2247
- // parser split the expression across roles, so the transformer reordered INSIDE
2248
- // the parens, embedded the event phrase mid-expression (`(the valor de #price
2249
- // de .quantity como Number)`), and DROPPED the entire `* (my value as Number)`
2250
- // second operand — in every language, including the SVO controls. Fused groups
2251
- // keep the expression intact; interiors translate word-by-word in order.
2252
- const COMPUTED_VALUE =
2253
- 'on input from .quantity set #total.innerText to (the value of #price as Number) * (my value as Number)';
2254
-
2255
- const firstParenGroup = (s: string): string => {
2256
- const m = s.match(/\([^)]*\)/);
2257
- return m ? m[0] : '';
2258
- };
2259
-
2260
- it.each(['es', 'de', 'ko', 'hi', 'ja', 'qu'])(
2261
- '[%s] keeps both operands and the * operator',
2262
- lang => {
2263
- const out = new GrammarTransformer('en', lang).transform(COMPUTED_VALUE);
2264
- // The second operand was dropped in every language before the fix.
2265
- expect(out).toContain(') * (');
2266
- // Exactly two paren groups survive, in source order.
2267
- expect(out.match(/\(/g)?.length).toBe(2);
2268
- expect(out.match(/\)/g)?.length).toBe(2);
2269
- }
2270
- );
2271
-
2272
- it.each([
2273
- ['es', ['establecer', 'entrada']],
2274
- ['ko', ['설정', '입력']],
2275
- ['ja', ['設定', '入力']],
2276
- ] as Array<[string, string[]]>)(
2277
- '[%s] never embeds the verb or event phrase inside the parens',
2278
- (lang, forbidden) => {
2279
- const out = new GrammarTransformer('en', lang).transform(COMPUTED_VALUE);
2280
- const group = firstParenGroup(out);
2281
- expect(group).not.toBe('');
2282
- for (const word of forbidden) {
2283
- expect(group).not.toContain(word);
2284
- }
2285
- // The event source selector stays outside the expression too.
2286
- expect(group).not.toContain('.quantity');
2287
- }
2288
- );
2289
-
2290
- it('[es] interior keywords translate in place, in order', () => {
2291
- const out = new GrammarTransformer('en', 'es').transform(COMPUTED_VALUE);
2292
- expect(out).toContain('(the valor de #price como Number) * (mi valor como Number)');
2293
- });
2294
- });
2295
-
2296
- describe('Polish get translates to uzyskaj, not pobierz (pl get-value get/fetch homonym)', () => {
2297
- // The pl dict emitted `pobierz` for get, but `pobierz` ("download") is the semantic pl
2298
- // profile's FETCH primary — so every transformed get parsed as fetch (get-value lossy +
2299
- // a phantom fetch). Emit `uzyskaj` (the profile's get primary) so get stays get.
2300
- const pl = (s: string) => new GrammarTransformer('en', 'pl').transform(s);
2301
-
2302
- it('[pl] `get #x.value` emits uzyskaj (not the fetch word pobierz)', () => {
2303
- const out = pl('get #input.value');
2304
- expect(out).toContain('uzyskaj');
2305
- expect(out).not.toContain('pobierz');
2306
- });
2307
-
2308
- it('[pl] `fetch /api` still emits pobierz (fetch unaffected)', () => {
2309
- expect(pl('fetch /api/data')).toContain('pobierz');
2310
- });
2311
- });
2312
-
2313
- describe('Chinese take translates to 拿取, not 获取 (zh take/get homonym)', () => {
2314
- // The zh dict emitted `获取` for take, but `获取` is the semantic zh profile's GET primary
2315
- // — so `take …` parsed as get (take-class-from-siblings: phantom get + take dropped).
2316
- // Emit `拿取` (the profile's take primary) so take stays take.
2317
- const zh = (s: string) => new GrammarTransformer('en', 'zh').transform(s);
2318
-
2319
- it('[zh] `take .x from .y` emits 拿取 (not the get word 获取)', () => {
2320
- const out = zh('take .active from .tab-button');
2321
- expect(out).toContain('拿取');
2322
- expect(out).not.toContain('获取');
2323
- });
2324
-
2325
- it('[zh] `get #x.value` still emits a get word, not 拿取', () => {
2326
- expect(zh('get #input.value')).not.toContain('拿取');
2327
- });
2328
- });
2329
-
2330
- describe('tr resize single-token event keyword (window-resize NULL → faithful)', () => {
2331
- // The dict previously emitted `boyut_değiştir` for the resize event; the tr
2332
- // semantic tokenizer splits on `_` → `boyut` + `değiştir`, and `değiştir`
2333
- // normalizes to `toggle` (homonym collision) — which destroyed the resize event
2334
- // and made `window-resize` the lone tr parse hard-fail. A non-underscore keyword
2335
- // (`boyutlandırma`) keeps the event token whole (mirrors the ru/uk install
2336
- // single-token route). Pairs with the semantic event-map entry.
2337
- const transformer = new GrammarTransformer('en', 'tr');
2338
-
2339
- it('emits the single-token resize keyword (no underscore, no toggle homonym)', () => {
2340
- const result = transformer.transform('on resize from window call adjustLayout()');
2341
- expect(result).toContain('boyutlandırma');
2342
- expect(result).not.toContain('boyut_değiştir');
2343
- expect(result).not.toContain('değiştir'); // the toggle-homonym fragment is gone
2344
- });
2345
- });
2346
-
2347
- describe('ru/uk fused (no-underscore) event keywords (mousedown/mouseup/resize)', () => {
2348
- // The semantic tokenizer splits on `_`, so the old underscore forms
2349
- // (мышь_вниз / изменение_размера) broke event recognition → the event typed as
2350
- // a bare expression. The dict now emits the FUSED form (мышьвниз / изменениеразмера),
2351
- // registered in the ru/uk tokenizer EXTRAS. Mirrors the #510 tr resize route.
2352
- it('[ru] emits fused mousedown/mouseup/resize (no underscore)', () => {
2353
- const tr = new GrammarTransformer('en', 'ru');
2354
- expect(tr.transform('on mousedown toggle .x')).toContain('мышьвниз');
2355
- expect(tr.transform('on mouseup toggle .x')).toContain('мышьвверх');
2356
- expect(tr.transform('on resize call f()')).toContain('изменениеразмера');
2357
- expect(tr.transform('on mousedown toggle .x')).not.toContain('мышь_вниз');
2358
- });
2359
-
2360
- it('[uk] emits fused mousedown/mouseup/resize (no underscore)', () => {
2361
- const tr = new GrammarTransformer('en', 'uk');
2362
- expect(tr.transform('on mousedown toggle .x')).toContain('мишавниз');
2363
- expect(tr.transform('on mouseup toggle .x')).toContain('мишавгору');
2364
- expect(tr.transform('on resize call f()')).toContain('змінарозміру');
2365
- expect(tr.transform('on mousedown toggle .x')).not.toContain('миша_вниз');
2366
- });
2367
- });
2368
-
2369
- describe('tr/hi/qu fused (no-underscore) mouse events (mousedown/mouseup)', () => {
2370
- // The tokenizer splits on `_`, so the old underscore forms broke event
2371
- // recognition (tr `fare_bas`→"bas", qu `rat_ñitiy`→click homonym). The dict now
2372
- // emits the fused form, recognized in the tr/hi/qu tokenizer EXTRAS — which also
2373
- // routes repeat-until-event onto the fused-action recovery path. Mirrors #535.
2374
- it('[tr] emits fused mousedown/mouseup (no underscore)', () => {
2375
- const t = new GrammarTransformer('en', 'tr');
2376
- expect(t.transform('on mousedown toggle .x')).toContain('farebas');
2377
- expect(t.transform('on mouseup toggle .x')).toContain('farebırak');
2378
- expect(t.transform('on mousedown toggle .x')).not.toContain('fare_bas');
2379
- });
2380
- it('[hi] emits fused mousedown/mouseup (no underscore)', () => {
2381
- const t = new GrammarTransformer('en', 'hi');
2382
- expect(t.transform('on mousedown toggle .x')).toContain('माउसनीचे');
2383
- expect(t.transform('on mouseup toggle .x')).toContain('माउसऊपर');
2384
- expect(t.transform('on mousedown toggle .x')).not.toContain('माउस_नीचे');
2385
- });
2386
- it('[qu] emits fused mousedown/mouseup (no underscore, no click homonym)', () => {
2387
- const t = new GrammarTransformer('en', 'qu');
2388
- expect(t.transform('on mousedown toggle .x')).toContain('ratñitiy');
2389
- expect(t.transform('on mouseup toggle .x')).toContain('rathuqariy');
2390
- expect(t.transform('on mousedown toggle .x')).not.toContain('rat_ñitiy');
2391
- });
2392
- });
2393
-
2394
- describe('Predicate adjective stays inside the condition clause (empty ×8 bn/hi/tr)', () => {
2395
- // `empty` is ALSO a hyperscript command (v0.9.90), so the block-body scans
2396
- // (transformBlockBody, extractBlockStructure's unless path) cut the condition
2397
- // of `if my value is empty add .error to me` at `is` and displaced the
2398
- // adjective into the add's argument zone — where it anchored a spurious
2399
- // `empty-{lang}-generated` parse and stole a neighboring role (hi took the
2400
- // add's `.error` patient). A command-keyword candidate immediately after a
2401
- // copula is a predicate adjective, never the body's first verb.
2402
- const IF_EMPTY = 'on blur if my value is empty add .error to me else remove .error from me end';
2403
-
2404
- it('[hi] keeps है खाली adjacent inside the condition, verb after', () => {
2405
- const out = new GrammarTransformer('en', 'hi').transform(IF_EMPTY);
2406
- expect(out).toContain('है खाली'); // copula + predicate stay adjacent
2407
- // the adjective precedes the then-branch (no displacement past .error)
2408
- expect(out.indexOf('खाली')).toBeLessThan(out.indexOf('.error'));
2409
- });
2410
-
2411
- it('[tr] keeps dir boş adjacent inside the condition', () => {
2412
- const out = new GrammarTransformer('en', 'tr').transform(IF_EMPTY);
2413
- expect(out).toContain('dir boş');
2414
- expect(out.indexOf('boş')).toBeLessThan(out.indexOf('.error'));
2415
- });
2416
-
2417
- it('[bn] keeps হয় খালি adjacent inside the condition', () => {
2418
- const out = new GrammarTransformer('en', 'bn').transform(IF_EMPTY);
2419
- expect(out).toContain('হয় খালি');
2420
- expect(out.indexOf('খালি')).toBeLessThan(out.indexOf('.error'));
2421
- });
2422
-
2423
- it('unless-path body scan applies the same copula guard', () => {
2424
- // extractBlockStructure's unless heuristic shares the scan: the condition
2425
- // runs through the predicate adjective to the real body verb.
2426
- const out = new GrammarTransformer('en', 'hi').transform(
2427
- 'unless my value is empty add .error to me'
2428
- );
2429
- expect(out.indexOf('खाली')).toBeLessThan(out.indexOf('.error'));
2430
- });
2431
-
2432
- it('a real body verb right after the condition still starts the body', () => {
2433
- // No copula before `add` — the scan must still cut there (byte-identical
2434
- // pre/post for conditions without a trailing predicate adjective).
2435
- const out = new GrammarTransformer('en', 'hi').transform(
2436
- 'on blur if my value add .error to me end'
2437
- );
2438
- expect(out).toContain('.error');
2439
- });
2440
- });
2441
-
2442
- describe('take `for me` target stays in-clause (take-class spurious for ×6 bn/hi/ja/ko/qu/tr)', () => {
2443
- // `for` is ALSO hyperscript's loop command, so splitOnCommandBoundaries cut
2444
- // `take .active from .tab-button for me` at `for` and the join re-inserted
2445
- // `then` — a dangling `then for me` clause that six SOV languages parsed as
2446
- // a spurious `for` loop with patient "me". Hyperscript's only statement-head
2447
- // `for` is `for <var> in <iterable>`: a `for` with no following `in` is a
2448
- // role phrase and must stay attached (isLoopHeadFor).
2449
- const TAKE = 'on click take .active from .tab-button for me';
2450
-
2451
- it('[hi] no then-shatter: फिर absent, take clause intact', () => {
2452
- const out = new GrammarTransformer('en', 'hi').transform(TAKE);
2453
- expect(out).not.toContain('फिर');
2454
- expect(out).toContain('.tab-button');
2455
- });
2456
-
2457
- it('[ja] no then-shatter: それから absent', () => {
2458
- const out = new GrammarTransformer('en', 'ja').transform(TAKE);
2459
- expect(out).not.toContain('それから');
2460
- });
2461
-
2462
- it('[tr] no then-shatter: ardından absent', () => {
2463
- const out = new GrammarTransformer('en', 'tr').transform(TAKE);
2464
- expect(out).not.toContain('ardından');
2465
- });
2466
-
2467
- it("[bn] no then-shatter AND the pronoun renders bare — জন্য doubles as bn's `for` loop keyword", () => {
2468
- // insertMarkers suppresses the duration marker for a pronoun value: en maps
2469
- // `for` → duration lexically, and emitting জন্য after আমি re-minted the
2470
- // spurious `for` on the parse side (the SOV verb-anchoring fallback must
2471
- // keep finding জন্য as a verb for real loops).
2472
- const out = new GrammarTransformer('en', 'bn').transform(TAKE);
2473
- expect(out).not.toContain('তারপর');
2474
- expect(out).not.toContain('জন্য');
2475
- expect(out).toContain('আমি');
2476
- });
2477
-
2478
- it('a real `for <var> in <iterable>` loop is untouched (both-ways negative)', () => {
2479
- const out = new GrammarTransformer('en', 'bn').transform(
2480
- 'on click repeat for item in .items add .processed to item'
2481
- );
2482
- expect(out).toContain('জন্য'); // bn's real loop keyword survives
2483
- expect(out).toContain('তারপর'); // body still splits at the real command boundary
2484
- });
2485
-
2486
- it('`wait for <event>` and real durations keep their marker (pronoun-only suppression)', () => {
2487
- const wait = new GrammarTransformer('en', 'bn').transform('on click wait for transitionend');
2488
- expect(wait).toContain('transitionend জন্য');
2489
- const dur = new GrammarTransformer('en', 'bn').transform(
2490
- 'on click transition opacity to 0 over 300ms'
2491
- );
2492
- expect(dur).toContain('300ms জন্য');
2493
- });
2494
- });
2495
-
2496
- describe('qu fused (no-underscore) empty/null value word (chusaq)', () => {
2497
- // The qu semantic tokenizer splits on `_` by design, so the old `ch_usaq`
2498
- // shattered (ch / _ / usaq) and the `usaq` shard fused with the following
2499
- // selector once the predicate-adjective guard healed the render order —
2500
- // `add` captured patient=expression:"usaq.error" (input-validation/if-empty
2501
- // qu). The dict now emits the fused `chusaq`, which the tokenizer already
2502
- // recognizes (norm `null`). Mirrors the #535 ru/uk fused-event route and the
2503
- // L4 qu ñawpaq_kaq dict↔profile realignment.
2504
- it('emits chusaq (fused) for is-empty conditions', () => {
2505
- const out = new GrammarTransformer('en', 'qu').transform(
2506
- 'on blur if my value is empty add .error to me else remove .error from me end'
2507
- );
2508
- expect(out).toContain('kanqa chusaq');
2509
- expect(out).not.toContain('ch_usaq');
2510
- expect(out.indexOf('chusaq')).toBeLessThan(out.indexOf('.error'));
2511
- });
2512
- });
2513
-
2514
- // =============================================================================
2515
- // SOV reorder stranding (transformer-rendering arc)
2516
- // =============================================================================
2517
- //
2518
- // Three rendering defects, one family (docs-internal/HANDOFF_transformer-rendering.md):
2519
- // (a) `wait for X or Y from Z` rendered verb-first (tr) / verb-medial (qu) —
2520
- // reorderRoles' safety-net appended roles missing from canonicalOrder
2521
- // AFTER the verb, stranding the or-run/from-phrase outside the clause;
2522
- // (b) `repeat until event E from S` postposed the from-phrase after the verb;
2523
- // (c) a trailing block terminator (`… wait 200ms end` fragments from the
2524
- // then-splitter) was swept into the open role's VALUE and rendered inside
2525
- // the phrase instead of after the verb.
2526
- // Fixed by extending tr/qu/bn canonicalOrder to full verb-final role orders
2527
- // and stripping/re-appending the stranded terminator in transformSingle.
2528
-
2529
- describe('SOV reorder stranding (transformer-rendering arc)', () => {
2530
- it('[tr] wait + or-run + from-phrase renders verb-final, from-phrase in-clause', () => {
2531
- const out = new GrammarTransformer('en', 'tr').transform(
2532
- 'wait for pointermove(clientY) or pointerup(clientY) from document'
2533
- );
2534
- expect(out).toContain('veya');
2535
- expect(out.trim().endsWith('bekle')).toBe(true);
2536
- expect(out.indexOf('belge den')).toBeGreaterThanOrEqual(0);
2537
- expect(out.indexOf('belge den')).toBeLessThan(out.indexOf('bekle'));
2538
- });
2539
-
2540
- it('[qu] wait + or-run + from-phrase renders verb-final, from-phrase in-clause', () => {
2541
- const out = new GrammarTransformer('en', 'qu').transform(
2542
- 'wait for pointermove(clientY) or pointerup(clientY) from document'
2543
- );
2544
- expect(out).toContain('utaq');
2545
- expect(out.trim().endsWith('suyay')).toBe(true);
2546
- expect(out.indexOf('qillqa manta')).toBeGreaterThanOrEqual(0);
2547
- expect(out.indexOf('qillqa manta')).toBeLessThan(out.indexOf('pointermove'));
2548
- });
2549
-
2550
- it('[tr] repeat-until-event keeps the from-phrase before the verb', () => {
2551
- const out = new GrammarTransformer('en', 'tr').transform(
2552
- 'repeat until event pointerup from document'
2553
- );
2554
- expect(out.trim().endsWith('tekrarla')).toBe(true);
2555
- expect(out.indexOf('belge den')).toBeGreaterThanOrEqual(0);
2556
- expect(out.indexOf('belge den')).toBeLessThan(out.indexOf('tekrarla'));
2557
- });
2558
-
2559
- it('[bn] trailing-end fragment renders the terminator after the verb', () => {
2560
- // `then`-split fragment of `… then wait 200ms end` (repeat-while body).
2561
- const out = new GrammarTransformer('en', 'bn').transform('wait 200ms end');
2562
- expect(out.trim()).toBe('200ms কে অপেক্ষা শেষ');
2563
- expect(out).not.toContain('শেষ কে'); // terminator no longer inside the object phrase
2564
- });
2565
-
2566
- it('[qu] inline-if set keeps its verb-final tail before the terminator', () => {
2567
- const out = new GrammarTransformer('en', 'qu').transform(
2568
- 'if newWidth < minWidth then set newWidth to minWidth end'
2569
- );
2570
- expect(out).toMatch(/man churanay tukuy\s*$/);
2571
- });
2572
-
2573
- it('[tr] trailing-end set fragment renders verb-final with trailing terminator', () => {
2574
- // Previously the set-to rule's end-guard predicate skipped verb-final
2575
- // reorder for these fragments; with the terminator stripped first, the
2576
- // rule applies and the terminator follows the verb.
2577
- const out = new GrammarTransformer('en', 'tr').transform('set dragClass to "sorting" end');
2578
- expect(out.trim()).toBe('dragClass i "sorting" e ayarla son');
2579
- });
2580
-
2581
- const draggableWait =
2582
- 'wait for pointermove(clientX, clientY) or pointerup(clientX, clientY) from document';
2583
- it('[tr] draggable-shaped wait (clientX, clientY) is verb-final', () => {
2584
- const out = new GrammarTransformer('en', 'tr').transform(draggableWait);
2585
- expect(out.trim().endsWith('bekle')).toBe(true);
2586
- expect(out.indexOf('belge den')).toBeLessThan(out.indexOf('bekle'));
2587
- });
2588
- it('[qu] draggable-shaped wait (clientX, clientY) is verb-final', () => {
2589
- const out = new GrammarTransformer('en', 'qu').transform(draggableWait);
2590
- expect(out.trim().endsWith('suyay')).toBe(true);
2591
- expect(out.indexOf('qillqa manta')).toBeLessThan(out.indexOf('pointermove'));
2592
- });
2593
- });
2594
-
2595
- // The form-submit-prevent head fuses TWO juxtaposed commands (`halt the event
2596
- // call validateForm()`); parseEventHandler sweeps both into one patient blob
2597
- // and the SOV patient-first reorder fronted it whole — halt's operand stranded
2598
- // clause-initial, call's operand riding with it (tr bound validateForm() to
2599
- // halt and the event to call). sovHaltCallFusedRule splits the blob at the
2600
- // language's call verb and emits both commands verb-final, joined by the
2601
- // then-connective (R1 deferred-tail Family H,
2602
- // docs-internal/HANDOFF_r1-deferred-tail.md).
2603
- describe('SOV fused halt+call heads render both commands verb-final (Family H)', () => {
2604
- const src =
2605
- 'on submit halt the event call validateForm() if result is false log "Invalid form" end';
2606
- const EXPECT: Array<[string, string]> = [
2607
- ['ja', 'the イベント を 送信 で 停止 それから validateForm() を 呼び出し'],
2608
- ['ko', 'the 이벤트 를 제출 할 때 정지 그러면 validateForm() 를 호출'],
2609
- // Batch 3: tr submit event word gönder→gönderme (gönder is the send verb — captured event "send")
2610
- ['tr', 'the olay i gönderme de durdur ardından validateForm() i çağır'],
2611
- ['bn', 'the ঘটনা কে জমা এ থামুন তারপর validateForm() কে কল'],
2612
- ['hi', 'the घटना को जमा पर रोकें फिर validateForm() को कॉल'],
2613
- // Batch 3: qu submit event word kachay→apaykachay (kachay is the send verb — captured "send" in this slot)
2614
- ['qu', 'the ruway ta apaykachay pi sayay chayqa validateForm() ta qayay'],
2615
- ];
2616
- for (const [lang, head] of EXPECT) {
2617
- it(`[${lang}] halt keeps its operand adjacent + verb-final; call follows the connective`, () => {
2618
- const out = new GrammarTransformer('en', lang).transform(src);
2619
- expect(out.startsWith(head)).toBe(true);
2620
- });
2621
- }
2622
-
2623
- it('[ja] a plain `halt the event` handler keeps its default render (rule stands down)', () => {
2624
- const out = new GrammarTransformer('en', 'ja').transform('on submit halt the event');
2625
- expect(out.trim()).toBe('the イベント を 送信 で 停止');
2626
- });
2627
-
2628
- it('[tr] a call-only handler keeps its default render (no halt — rule stands down)', () => {
2629
- const out = new GrammarTransformer('en', 'tr').transform('on click call myFunction()');
2630
- expect(out.trim()).toBe('myFunction() i tıklama de çağır');
2631
- });
2632
- });
2633
-
2634
- // SOV renders emit nothing between an if-condition's tail and a POSITIONAL-
2635
- // headed branch operand (`… .modal 最初 <button/> …`), and the semantic fold's
2636
- // command-start detection can't see that shape — the condition swallowed the
2637
- // focus operand's head (focus-trap, ja/ko/qu). transformBlockBody now emits
2638
- // the target's then-connective at the clause/body seam, gated to exactly the
2639
- // blind shape (R1 deferred-tail Family G).
2640
- describe('SOV if-blocks get a then-connective before a positional branch head (Family G)', () => {
2641
- const src =
2642
- 'on keydown[key=="Tab"] from .modal if target matches last <button/> in .modal focus first <button/> in .modal halt end';
2643
- const SEAM: Array<[string, string]> = [
2644
- ['ja', '.modal それから 最初'],
2645
- ['ko', '.modal 그러면 첫번째'],
2646
- ['qu', '.modal chayqa ñawpaq'],
2647
- ['tr', '.modal ardından ilk'],
2648
- ['bn', '.modal তারপর প্রথম'],
2649
- ['hi', '.modal फिर पहला'],
2650
- ];
2651
- for (const [lang, seam] of SEAM) {
2652
- it(`[${lang}] focus-trap renders the connective at the condition/branch seam`, () => {
2653
- const out = new GrammarTransformer('en', lang).transform(src);
2654
- expect(out).toContain(seam);
2655
- });
2656
- }
2657
-
2658
- it('[qu] a non-positional branch head gains no connective (form-submit if unchanged)', () => {
2659
- const out = new GrammarTransformer('en', 'qu').transform(
2660
- 'on submit halt the event call validateForm() if result is false log "Invalid form" end'
2661
- );
2662
- expect(out).toContain('llulla "Invalid form" ta qillqakuy');
2663
- });
2664
- });
2665
-
2666
- // `swap #a with #b using view transition` — the tail the transformer had never
2667
- // seen. `using` is in no dictionary and no role table, so it swept into whatever
2668
- // role phrase was open; `transition` IS a translated command keyword in every
2669
- // dictionary AND is in ENGLISH_COMMANDS, so splitOnCommandBoundaries cut the
2670
- // clause there and the rejoin planted a phantom translated `transition` COMMAND
2671
- // after the target's then-connective (`intercambiar #a con #b using view
2672
- // entonces transición`). The phrase has no native form in any of the 24
2673
- // languages — semantic matches the literal English `using view` marker
2674
- // everywhere (USING_VIEW_MARKER_ALL_LANGS) — so it is masked before any
2675
- // splitting/translation and re-appended verbatim at the clause tail, which is
2676
- // where every word order's patterns admit it.
2677
- describe('`using view transition` is a passthrough tail, not translated content', () => {
2678
- const SRC = 'on click swap #a with #b using view transition';
2679
-
2680
- // One per word-order family: SVO Romance/Germanic/Slavic, V2, SOV
2681
- // particle/agglutinative, VSO, and the two object-marking SVO languages.
2682
- const TRANSLATED_TRANSITION: Array<[string, string]> = [
2683
- ['es', 'transición'],
2684
- ['fr', 'transition'], // fr's own verb form — the passthrough must not re-emit it as a command
2685
- ['de', 'übergang'],
2686
- ['ja', '遷移'],
2687
- ['ko', '전환'],
2688
- ['tr', 'geçiş'],
2689
- ['qu', 'pasay'],
2690
- ['bn', 'সংক্রমণ'],
2691
- ['hi', 'संक्रमण'],
2692
- ['ar', 'انتقال'],
2693
- ['zh', '过渡'],
2694
- ['he', 'מעבר'],
2695
- ['ru', 'анимировать'],
2696
- ['th', 'เปลี่ยนผ่าน'],
2697
- ['vi', 'chuyển tiếp'],
2698
- ['tl', 'transisyon'],
2699
- ];
2700
-
2701
- for (const [lang, translated] of TRANSLATED_TRANSITION) {
2702
- it(`[${lang}] emits the English tail verbatim at the clause end`, () => {
2703
- const out = new GrammarTransformer('en', lang).transform(SRC);
2704
- expect(out).toMatch(/using view transition$/);
2705
- // No phantom then-connective, and no translated `transition` command.
2706
- if (translated !== 'transition') {
2707
- expect(out).not.toContain(translated);
2708
- }
2709
- });
2710
- }
2711
-
2712
- it('the tail does not displace the clause it modifies (both operands survive)', () => {
2713
- for (const lang of ['es', 'ja', 'ar', 'tr', 'zh', 'qu']) {
2714
- const out = new GrammarTransformer('en', lang).transform(SRC);
2715
- expect(out, lang).toContain('#a');
2716
- expect(out, lang).toContain('#b');
2717
- }
2718
- });
2719
-
2720
- it('`transition` as a real command keyword still translates (both-ways negative)', () => {
2721
- // The mask must be anchored on the literal `using view` phrase, not on the
2722
- // word `transition` — the standalone transition command has to keep its
2723
- // dictionary form or every animate row would render English.
2724
- const es = new GrammarTransformer('en', 'es').transform(
2725
- 'on click transition opacity to 0 over 300ms'
2726
- );
2727
- expect(es).toContain('transición');
2728
- const ja = new GrammarTransformer('en', 'ja').transform(
2729
- 'on click transition opacity to 0 over 300ms'
2730
- );
2731
- expect(ja).toContain('遷移');
2732
- });
2733
-
2734
- it('a `then`-chained clause after the tail still splits (the tail never swallows a boundary)', () => {
2735
- const out = new GrammarTransformer('en', 'es').transform(
2736
- 'on click swap #a with #b using view transition then log "done"'
2737
- );
2738
- expect(out).toContain('using view transition');
2739
- expect(out).toContain('entonces');
2740
- expect(out).toContain('"done"');
2741
- });
2742
-
2743
- it('`process partials in it using view transition` keeps its tail too', () => {
2744
- for (const lang of ['es', 'ja', 'tr']) {
2745
- const out = new GrammarTransformer('en', lang).transform(
2746
- 'process partials in it using view transition'
2747
- );
2748
- expect(out, lang).toMatch(/using view transition$/);
2749
- }
2750
- });
2751
- });