@lokascript/i18n 2.11.1 → 3.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/CHANGELOG.md +13 -0
  2. package/dist/browser.cjs +5 -1690
  3. package/dist/browser.cjs.map +1 -1
  4. package/dist/browser.d.cts +2 -2
  5. package/dist/browser.d.ts +2 -2
  6. package/dist/browser.js +6 -1685
  7. package/dist/browser.js.map +1 -1
  8. package/dist/dictionaries/index.cjs +5 -1
  9. package/dist/dictionaries/index.cjs.map +1 -1
  10. package/dist/dictionaries/index.js +5 -1
  11. package/dist/dictionaries/index.js.map +1 -1
  12. package/dist/{transformer-CsOeqayN.d.cts → index-BykxjYST.d.cts} +1 -232
  13. package/dist/{transformer-DWCTG1DQ.d.ts → index-DuIef8O7.d.ts} +1 -232
  14. package/dist/index.cjs +16 -1878
  15. package/dist/index.cjs.map +1 -1
  16. package/dist/index.d.cts +1 -1
  17. package/dist/index.d.ts +1 -1
  18. package/dist/index.js +17 -1873
  19. package/dist/index.js.map +1 -1
  20. package/dist/lokascript-i18n.min.js +1 -1
  21. package/dist/lokascript-i18n.min.js.map +1 -1
  22. package/dist/lokascript-i18n.mjs +49 -2680
  23. package/dist/lokascript-i18n.mjs.map +1 -1
  24. package/dist/plugins/vite.cjs +5 -1
  25. package/dist/plugins/vite.cjs.map +1 -1
  26. package/dist/plugins/vite.js +5 -1
  27. package/dist/plugins/vite.js.map +1 -1
  28. package/dist/plugins/webpack.cjs +5 -1
  29. package/dist/plugins/webpack.cjs.map +1 -1
  30. package/dist/plugins/webpack.js +5 -1
  31. package/dist/plugins/webpack.js.map +1 -1
  32. package/package.json +4 -4
  33. package/src/browser.ts +0 -7
  34. package/src/compatibility/browser-tests/grammar-demo.spec.ts +22 -8
  35. package/src/constants.ts +1 -0
  36. package/src/dictionaries/bn.ts +5 -1
  37. package/src/grammar/index.ts +15 -9
  38. package/src/grammar/profiles.test.ts +440 -0
  39. package/src/index.ts +0 -7
  40. package/src/lexicon-parity.test.ts +77 -0
  41. package/src/grammar/grammar.test.ts +0 -2751
  42. package/src/grammar/transformer.ts +0 -2737
@@ -6,60 +6,6 @@
6
6
  * Maps English modifier keywords to their semantic roles.
7
7
  * Used by both the grammar transformer and keyword provider.
8
8
  */
9
- const ENGLISH_MODIFIER_ROLES = {
10
- to: 'destination',
11
- into: 'destination',
12
- from: 'source',
13
- with: 'style',
14
- by: 'quantity',
15
- as: 'method',
16
- on: 'event',
17
- over: 'duration',
18
- for: 'duration',
19
- };
20
- /**
21
- * Maps a command verb to its **primary** semantic role \u2014 the role of the
22
- * command's first/leading argument when no explicit modifier keyword marks it.
23
- *
24
- * The generic argument parser in the transformer defaults the first unmarked
25
- * argument to `patient`, which is correct for the majority of commands
26
- * (`toggle .x`, `add .x`, \u2026) but wrong for commands whose leading argument is a
27
- * non-patient \u2014 e.g. `wait <duration>`. Marking a duration as a patient emits a
28
- * spurious object-marker in the target language (Chinese `\u7b49\u5f85 \u628a 1s`, ungrammatical;
29
- * Japanese `1s \u3092 \u5f85\u3064`; Korean `1s \ub97c \ub300\uae30`), which the semantic parser then fails to
30
- * match \u2014 dropping the command.
31
- *
32
- * Only commands whose primary role is **not** `patient` are listed here (the
33
- * `patient` default already covers the rest). Mirrors the `primaryRole` field of
34
- * the semantic package's command schemas (`@lokascript/semantic`); kept in sync by
35
- * `schema-alignment.test.ts` so this local copy can't drift without the bundle
36
- * having to pull in the whole semantic graph.
37
- *
38
- * @see packages/i18n/src/schema-alignment.test.ts
39
- * @see docs-internal/ZH_BLOCK_BODY_SCOPE.md (#1 \u2014 transformer role model)
40
- */
41
- const COMMAND_PRIMARY_ROLES = {
42
- set: 'destination',
43
- on: 'event',
44
- trigger: 'event',
45
- send: 'event',
46
- wait: 'duration',
47
- fetch: 'source',
48
- get: 'source',
49
- if: 'condition',
50
- unless: 'condition',
51
- while: 'condition',
52
- repeat: 'loopType',
53
- go: 'destination',
54
- scroll: 'destination',
55
- tell: 'destination',
56
- default: 'destination',
57
- swap: 'destination',
58
- // morph deliberately absent: its schema primaryRole is `patient` (the
59
- // element being morphed \u2014 aligned with the transformer's patient marking
60
- // in the session-9 role-layout swap), and patient is the default.
61
- bind: 'destination',
62
- };
63
9
  /**
64
10
  * English modifier keywords (derived from ENGLISH_MODIFIER_ROLES)
65
11
  */
@@ -323,79 +269,6 @@ const ENGLISH_EXPRESSION_KEYWORDS = new Set([
323
269
  'starts with',
324
270
  'ends with',
325
271
  ]);
326
- // =============================================================================
327
- // Conditional Keywords
328
- // =============================================================================
329
- /**
330
- * Conditional keywords across languages (for statement type identification).
331
- * Includes 'when'/'where' conditional modifiers and their translations.
332
- */
333
- const CONDITIONAL_KEYWORDS = new Set([
334
- // English
335
- 'if',
336
- 'unless',
337
- 'when',
338
- 'where',
339
- // Japanese
340
- '\u3082\u3057',
341
- '\u6642\u306b',
342
- '\u3068\u304d\u306b',
343
- '\u3069\u3053\u3067',
344
- // Chinese
345
- '\u5982\u679c',
346
- '\u5f53',
347
- // Arabic
348
- '\u0625\u0630\u0627',
349
- '\u0639\u0646\u062f\u0645\u0627',
350
- '\u062d\u064a\u062b',
351
- // Spanish
352
- 'si',
353
- 'cuando',
354
- 'donde',
355
- // German
356
- 'wenn',
357
- 'wann',
358
- 'wo',
359
- // French
360
- 'quand',
361
- 'lorsque',
362
- 'o\u00f9',
363
- // Portuguese
364
- 'quando',
365
- 'onde',
366
- // Turkish
367
- 'e\u011fer',
368
- 'zaman',
369
- 'nerede',
370
- // Indonesian
371
- 'ketika',
372
- 'saat',
373
- 'dimana',
374
- // Korean
375
- '\ub54c',
376
- '\uc5b4\ub514\uc11c',
377
- // Quechua
378
- 'maypi',
379
- // Swahili
380
- 'wakati',
381
- 'wapi',
382
- ]);
383
- /**
384
- * "Then" keywords across languages (for conditional parsing).
385
- */
386
- const THEN_KEYWORDS = new Set([
387
- 'then',
388
- '\u305d\u308c\u304b\u3089',
389
- '\u90a3\u4e48',
390
- '\u062b\u0645',
391
- 'entonces',
392
- 'alors',
393
- 'dann',
394
- 'sonra',
395
- 'lalu',
396
- 'chayqa',
397
- 'kisha',
398
- ]);
399
272
 
400
273
  // packages/i18n/src/parser/create-provider.ts
401
274
  /**
@@ -623,7 +496,7 @@ function createEnglishProvider() {
623
496
 
624
497
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
625
498
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
626
- const es$1 = {
499
+ const es = {
627
500
  commands: {
628
501
  on: 'en',
629
502
  tell: 'decir',
@@ -847,13 +720,13 @@ const es$1 = {
847
720
  * parser.parse('en clic alternar .active');
848
721
  * ```
849
722
  */
850
- const esKeywords = createKeywordProvider(es$1, 'es', {
723
+ const esKeywords = createKeywordProvider(es, 'es', {
851
724
  allowEnglishFallback: true,
852
725
  });
853
726
 
854
727
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
855
728
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
856
- const ja$1 = {
729
+ const ja = {
857
730
  commands: {
858
731
  on: '\u3067',
859
732
  tell: '\u4f1d\u3048\u308b',
@@ -1091,13 +964,13 @@ const ja$1 = {
1091
964
  * parser.parse('\u30af\u30ea\u30c3\u30af \u3067 \u5207\u308a\u66ff\u3048 .active');
1092
965
  * ```
1093
966
  */
1094
- const jaKeywords = createKeywordProvider(ja$1, 'ja', {
967
+ const jaKeywords = createKeywordProvider(ja, 'ja', {
1095
968
  allowEnglishFallback: true,
1096
969
  });
1097
970
 
1098
971
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
1099
972
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
1100
- const fr$1 = {
973
+ const fr = {
1101
974
  commands: {
1102
975
  on: 'sur',
1103
976
  tell: 'dire',
@@ -1329,13 +1202,13 @@ const fr$1 = {
1329
1202
  * parser.parse('sur clic basculer .active');
1330
1203
  * ```
1331
1204
  */
1332
- const frKeywords = createKeywordProvider(fr$1, 'fr', {
1205
+ const frKeywords = createKeywordProvider(fr, 'fr', {
1333
1206
  allowEnglishFallback: true,
1334
1207
  });
1335
1208
 
1336
1209
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
1337
1210
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
1338
- const de$1 = {
1211
+ const de = {
1339
1212
  commands: {
1340
1213
  on: 'bei',
1341
1214
  tell: 'sagen',
@@ -1588,13 +1461,13 @@ const de$1 = {
1588
1461
  * parser.parse('bei klick umschalten .active');
1589
1462
  * ```
1590
1463
  */
1591
- const deKeywords = createKeywordProvider(de$1, 'de', {
1464
+ const deKeywords = createKeywordProvider(de, 'de', {
1592
1465
  allowEnglishFallback: true,
1593
1466
  });
1594
1467
 
1595
1468
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
1596
1469
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
1597
- const ar$1 = {
1470
+ const ar = {
1598
1471
  commands: {
1599
1472
  on: '\u0639\u0644\u0649',
1600
1473
  tell: '\u0623\u062e\u0628\u0631',
@@ -1832,13 +1705,13 @@ const ar$1 = {
1832
1705
  * parser.parse('\u0639\u0644\u0649 \u0646\u0642\u0631 \u0628\u062f\u0644 .active');
1833
1706
  * ```
1834
1707
  */
1835
- const arKeywords = createKeywordProvider(ar$1, 'ar', {
1708
+ const arKeywords = createKeywordProvider(ar, 'ar', {
1836
1709
  allowEnglishFallback: true,
1837
1710
  });
1838
1711
 
1839
1712
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
1840
1713
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
1841
- const ko$1 = {
1714
+ const ko = {
1842
1715
  commands: {
1843
1716
  on: '\uc5d0',
1844
1717
  tell: '\ub9d0\ud558\ub2e4',
@@ -2073,13 +1946,13 @@ const ko$1 = {
2073
1946
  * parser.parse('\uc5d0 \ud074\ub9ad \ud1a0\uae00 .active');
2074
1947
  * ```
2075
1948
  */
2076
- const koKeywords = createKeywordProvider(ko$1, 'ko', {
1949
+ const koKeywords = createKeywordProvider(ko, 'ko', {
2077
1950
  allowEnglishFallback: true,
2078
1951
  });
2079
1952
 
2080
1953
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
2081
1954
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
2082
- const zh$1 = {
1955
+ const zh = {
2083
1956
  commands: {
2084
1957
  on: '\u5f53',
2085
1958
  tell: '\u544a\u8bc9',
@@ -2328,13 +2201,13 @@ const zh$1 = {
2328
2201
  * parser.parse('\u5f53 \u70b9\u51fb \u5207\u6362 .active');
2329
2202
  * ```
2330
2203
  */
2331
- const zhKeywords = createKeywordProvider(zh$1, 'zh', {
2204
+ const zhKeywords = createKeywordProvider(zh, 'zh', {
2332
2205
  allowEnglishFallback: true,
2333
2206
  });
2334
2207
 
2335
2208
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
2336
2209
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
2337
- const tr$1 = {
2210
+ const tr = {
2338
2211
  commands: {
2339
2212
  on: '\u00fczerinde',
2340
2213
  tell: 's\u00f6yle',
@@ -2581,13 +2454,13 @@ const tr$1 = {
2581
2454
  * parser.parse('\u00fczerinde t\u0131klama de\u011fi\u015ftir .active');
2582
2455
  * ```
2583
2456
  */
2584
- const trKeywords = createKeywordProvider(tr$1, 'tr', {
2457
+ const trKeywords = createKeywordProvider(tr, 'tr', {
2585
2458
  allowEnglishFallback: true,
2586
2459
  });
2587
2460
 
2588
2461
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
2589
2462
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
2590
- const id$1 = {
2463
+ const id = {
2591
2464
  commands: {
2592
2465
  on: 'pada',
2593
2466
  tell: 'katakan',
@@ -2831,13 +2704,13 @@ const id$1 = {
2831
2704
  * parser.parse('pada klik ganti .active');
2832
2705
  * ```
2833
2706
  */
2834
- const idKeywords = createKeywordProvider(id$1, 'id', {
2707
+ const idKeywords = createKeywordProvider(id, 'id', {
2835
2708
  allowEnglishFallback: true,
2836
2709
  });
2837
2710
 
2838
2711
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
2839
2712
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
2840
- const qu$1 = {
2713
+ const qu = {
2841
2714
  commands: {
2842
2715
  on: 'kaqpi',
2843
2716
  tell: 'niy',
@@ -3097,13 +2970,13 @@ const qu$1 = {
3097
2970
  * parser.parse('\u00f1itiy-pi yapay #count-ta');
3098
2971
  * ```
3099
2972
  */
3100
- const quKeywords = createKeywordProvider(qu$1, 'qu', {
2973
+ const quKeywords = createKeywordProvider(qu, 'qu', {
3101
2974
  allowEnglishFallback: true,
3102
2975
  });
3103
2976
 
3104
2977
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
3105
2978
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
3106
- const sw$1 = {
2979
+ const sw = {
3107
2980
  commands: {
3108
2981
  on: 'kwenye',
3109
2982
  tell: 'ambia',
@@ -3370,13 +3243,13 @@ const sw$1 = {
3370
3243
  * parser.parse('kwenye bonyeza badilisha .active');
3371
3244
  * ```
3372
3245
  */
3373
- const swKeywords = createKeywordProvider(sw$1, 'sw', {
3246
+ const swKeywords = createKeywordProvider(sw, 'sw', {
3374
3247
  allowEnglishFallback: true,
3375
3248
  });
3376
3249
 
3377
3250
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
3378
3251
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
3379
- const pt$1 = {
3252
+ const pt = {
3380
3253
  commands: {
3381
3254
  on: 'em',
3382
3255
  tell: 'dizer',
@@ -3615,13 +3488,13 @@ const pt$1 = {
3615
3488
  * parser.parse('em clique alternar .active');
3616
3489
  * ```
3617
3490
  */
3618
- const ptKeywords = createKeywordProvider(pt$1, 'pt', {
3491
+ const ptKeywords = createKeywordProvider(pt, 'pt', {
3619
3492
  allowEnglishFallback: true,
3620
3493
  });
3621
3494
 
3622
3495
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
3623
3496
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
3624
- const it$1 = {
3497
+ const it = {
3625
3498
  commands: {
3626
3499
  on: 'su',
3627
3500
  tell: 'dire',
@@ -3830,13 +3703,13 @@ const it$1 = {
3830
3703
  },
3831
3704
  };
3832
3705
 
3833
- const itKeywords = createKeywordProvider(it$1, 'it', {
3706
+ const itKeywords = createKeywordProvider(it, 'it', {
3834
3707
  allowEnglishFallback: true,
3835
3708
  });
3836
3709
 
3837
3710
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
3838
3711
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
3839
- const vi$1 = {
3712
+ const vi = {
3840
3713
  commands: {
3841
3714
  on: 'khi',
3842
3715
  tell: 'n\u00f3i v\u1edbi',
@@ -4043,13 +3916,13 @@ const vi$1 = {
4043
3916
  },
4044
3917
  };
4045
3918
 
4046
- const viKeywords = createKeywordProvider(vi$1, 'vi', {
3919
+ const viKeywords = createKeywordProvider(vi, 'vi', {
4047
3920
  allowEnglishFallback: true,
4048
3921
  });
4049
3922
 
4050
3923
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
4051
3924
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
4052
- const pl$1 = {
3925
+ const pl = {
4053
3926
  commands: {
4054
3927
  on: 'gdy',
4055
3928
  tell: 'powiedz',
@@ -4264,7 +4137,7 @@ const pl$1 = {
4264
4137
  },
4265
4138
  };
4266
4139
 
4267
- const plKeywords = createKeywordProvider(pl$1, 'pl', {
4140
+ const plKeywords = createKeywordProvider(pl, 'pl', {
4268
4141
  allowEnglishFallback: true,
4269
4142
  });
4270
4143
 
@@ -5161,7 +5034,11 @@ const bengaliDictionary = {
5161
5034
  random: '\u098f\u09b2\u09cb\u09ae\u09c7\u09b2\u09cb',
5162
5035
  length: '\u09a6\u09c8\u09b0\u09cd\u0998\u09cd\u09af',
5163
5036
  index: '\u09b8\u09c2\u099a\u0995',
5164
- empty: '\u0996\u09be\u09b2\u09bf-\u0995\u09b0\u09c1\u09a8',
5037
+ // The EXPRESSION `empty` is the state predicate (`if my value is empty`),
5038
+ // not the command: `\u0996\u09be\u09b2\u09bf-\u0995\u09b0\u09c1\u09a8` is the imperative "empty it!" and belongs to
5039
+ // the `empty` COMMAND, which keeps it. Kept in step with the semantic
5040
+ // lexicon by `lexicon-parity.test.ts`.
5041
+ empty: '\u0996\u09be\u09b2\u09bf',
5165
5042
  'starts with': '\u09a6\u09bf\u09af\u09bc\u09c7_\u09b6\u09c1\u09b0\u09c1',
5166
5043
  'ends with': '\u09a6\u09bf\u09af\u09bc\u09c7_\u09b6\u09c7\u09b7',
5167
5044
  'ignoring case': '\u0995\u09c7\u09b8_\u0989\u09aa\u09c7\u0995\u09cd\u09b7\u09be',
@@ -5171,9 +5048,9 @@ const bengaliDictionary = {
5171
5048
  'joined by': '\u09a6\u09cd\u09ac\u09be\u09b0\u09be_\u09af\u09c1\u0995\u09cd\u09a4',
5172
5049
  },
5173
5050
  };
5174
- const bn$1 = bengaliDictionary;
5051
+ const bn = bengaliDictionary;
5175
5052
 
5176
- const bnKeywords = createKeywordProvider(bn$1, 'bn', {
5053
+ const bnKeywords = createKeywordProvider(bn, 'bn', {
5177
5054
  allowEnglishFallback: true,
5178
5055
  });
5179
5056
 
@@ -5358,9 +5235,9 @@ const thaiDictionary = {
5358
5235
  'joined by': '\u0e23\u0e27\u0e21\u0e14\u0e49\u0e27\u0e22',
5359
5236
  },
5360
5237
  };
5361
- const th$1 = thaiDictionary;
5238
+ const th = thaiDictionary;
5362
5239
 
5363
- const thKeywords = createKeywordProvider(th$1, 'th', {
5240
+ const thKeywords = createKeywordProvider(th, 'th', {
5364
5241
  allowEnglishFallback: true,
5365
5242
  });
5366
5243
 
@@ -5596,9 +5473,9 @@ const malayDictionary = {
5596
5473
  index: 'indeks',
5597
5474
  },
5598
5475
  };
5599
- const ms$1 = malayDictionary;
5476
+ const ms = malayDictionary;
5600
5477
 
5601
- const msKeywords = createKeywordProvider(ms$1, 'ms', {
5478
+ const msKeywords = createKeywordProvider(ms, 'ms', {
5602
5479
  allowEnglishFallback: true,
5603
5480
  });
5604
5481
 
@@ -5840,15 +5717,15 @@ const tagalogDictionary = {
5840
5717
  index: 'indeks',
5841
5718
  },
5842
5719
  };
5843
- const tl$1 = tagalogDictionary;
5720
+ const tl = tagalogDictionary;
5844
5721
 
5845
- const tlKeywords = createKeywordProvider(tl$1, 'tl', {
5722
+ const tlKeywords = createKeywordProvider(tl, 'tl', {
5846
5723
  allowEnglishFallback: true,
5847
5724
  });
5848
5725
 
5849
5726
  // Generated/merged from semantic profiles \u2014 hand-written entries are preserved
5850
5727
  // To add derived entries, update the semantic profile and run: npm run generate:language-assets
5851
- const he$1 = {
5728
+ const he = {
5852
5729
  commands: {
5853
5730
  add: '\u05d4\u05d5\u05e1\u05e3',
5854
5731
  append: '\u05e6\u05e8\u05e3',
@@ -5968,7 +5845,7 @@ const he$1 = {
5968
5845
  * parser.parse('\u05d1 \u05dc\u05d7\u05d9\u05e6\u05d4 \u05de\u05ea\u05d2 .active');
5969
5846
  * ```
5970
5847
  */
5971
- const heKeywords = createKeywordProvider(he$1, 'he', {
5848
+ const heKeywords = createKeywordProvider(he, 'he', {
5972
5849
  allowEnglishFallback: true,
5973
5850
  });
5974
5851
 
@@ -6150,7 +6027,7 @@ function detectBrowserLocale() {
6150
6027
  * English dictionary - identity mapping since English is the canonical hyperscript language.
6151
6028
  * This exists primarily for symmetry in translation operations (e.g., en -> es).
6152
6029
  */
6153
- const en$1 = {
6030
+ const en = {
6154
6031
  commands: {
6155
6032
  // Event handling
6156
6033
  on: 'on',
@@ -8250,7 +8127,7 @@ function buildMappingFromDictionaries(sourceDict, targetDict, sourceCode, target
8250
8127
  * These languages share many concepts through Kanji/Hanzi,
8251
8128
  * making direct translation more accurate than pivot translation.
8252
8129
  */
8253
- const jaZhMapping = buildMappingFromDictionaries(ja$1, zh$1, 'ja', 'zh');
8130
+ const jaZhMapping = buildMappingFromDictionaries(ja, zh, 'ja', 'zh');
8254
8131
  /**
8255
8132
  * Chinese to Japanese mapping (reverse of jaZh)
8256
8133
  */
@@ -8264,7 +8141,7 @@ const zhJaMapping = reverseMapping(jaZhMapping);
8264
8141
  * Both are SOV languages with postposition particles,
8265
8142
  * sharing grammatical structure that enables more natural translation.
8266
8143
  */
8267
- const koJaMapping = buildMappingFromDictionaries(ko$1, ja$1, 'ko', 'ja');
8144
+ const koJaMapping = buildMappingFromDictionaries(ko, ja, 'ko', 'ja');
8268
8145
  /**
8269
8146
  * Japanese to Korean mapping (reverse of koJa)
8270
8147
  */
@@ -8452,2513 +8329,5 @@ function getSupportedDirectPairs() {
8452
8329
  });
8453
8330
  }
8454
8331
 
8455
- /**
8456
- * Dictionary Index
8457
- *
8458
- * Exports dictionaries for all supported languages.
8459
- * Each dictionary maps English canonical keywords to locale-specific translations
8460
- * across 8 categories: commands, modifiers, events, logical, temporal, values,
8461
- * attributes, and expressions.
8462
- *
8463
- * Derivation utilities (deriveFromProfile, createEnglishDictionary) are available
8464
- * for generating dictionaries from semantic language profiles. See ./derive.ts.
8465
- */
8466
- // Import per-language dictionaries
8467
- // =============================================================================
8468
- // Dictionary Exports
8469
- // =============================================================================
8470
- /** English dictionary */
8471
- const en = en$1;
8472
- /** Spanish dictionary */
8473
- const es = es$1;
8474
- /** Japanese dictionary */
8475
- const ja = ja$1;
8476
- /** Korean dictionary */
8477
- const ko = ko$1;
8478
- /** Chinese dictionary */
8479
- const zh = zh$1;
8480
- /** French dictionary */
8481
- const fr = fr$1;
8482
- /** German dictionary */
8483
- const de = de$1;
8484
- /** Arabic dictionary */
8485
- const ar = ar$1;
8486
- /** Turkish dictionary */
8487
- const tr = tr$1;
8488
- /** Indonesian dictionary */
8489
- const id = id$1;
8490
- /** Portuguese dictionary */
8491
- const pt = pt$1;
8492
- /** Quechua dictionary */
8493
- const qu = qu$1;
8494
- /** Swahili dictionary */
8495
- const sw = sw$1;
8496
- /** Italian dictionary */
8497
- const it = it$1;
8498
- /** Vietnamese dictionary */
8499
- const vi = vi$1;
8500
- /** Polish dictionary */
8501
- const pl = pl$1;
8502
- /** Russian dictionary */
8503
- const ru = russianDictionary;
8504
- /** Ukrainian dictionary */
8505
- const uk = ukrainianDictionary;
8506
- /** Hindi dictionary */
8507
- const hi = hindiDictionary;
8508
- /** Bengali dictionary */
8509
- const bn = bengaliDictionary;
8510
- /** Thai dictionary */
8511
- const th = thaiDictionary;
8512
- /** Malay dictionary */
8513
- const ms = malayDictionary;
8514
- /** Tagalog dictionary */
8515
- const tl = tagalogDictionary;
8516
- /** Hebrew dictionary */
8517
- const he = he$1;
8518
- // =============================================================================
8519
- // Dictionary Registry
8520
- // =============================================================================
8521
- /**
8522
- * All available dictionaries indexed by locale code.
8523
- */
8524
- const dictionaries = {
8525
- en,
8526
- es,
8527
- ko,
8528
- zh,
8529
- fr,
8530
- de,
8531
- ja,
8532
- ar,
8533
- tr,
8534
- id,
8535
- qu,
8536
- sw,
8537
- pt,
8538
- it,
8539
- vi,
8540
- pl,
8541
- ru,
8542
- uk,
8543
- hi,
8544
- bn,
8545
- th,
8546
- ms,
8547
- tl,
8548
- he,
8549
- };
8550
-
8551
- // packages/i18n/src/types.ts
8552
- /**
8553
- * All valid dictionary categories.
8554
- */
8555
- const DICTIONARY_CATEGORIES = [
8556
- 'commands',
8557
- 'modifiers',
8558
- 'events',
8559
- 'logical',
8560
- 'temporal',
8561
- 'values',
8562
- 'attributes',
8563
- 'expressions',
8564
- ];
8565
- /**
8566
- * Find a translation in any category of a dictionary.
8567
- * Returns the English key if found, undefined otherwise.
8568
- */
8569
- function findInDictionary(dict, localizedWord) {
8570
- const normalized = localizedWord.toLowerCase();
8571
- for (const category of DICTIONARY_CATEGORIES) {
8572
- const entries = dict[category];
8573
- for (const [english, localized] of Object.entries(entries)) {
8574
- if (localized.toLowerCase() === normalized) {
8575
- return { category, englishKey: english };
8576
- }
8577
- }
8578
- }
8579
- return undefined;
8580
- }
8581
- /**
8582
- * Find a translation for an English word in any category.
8583
- * Returns the localized word if found, undefined otherwise.
8584
- */
8585
- function translateFromEnglish(dict, englishWord) {
8586
- const normalized = englishWord.toLowerCase();
8587
- for (const category of DICTIONARY_CATEGORIES) {
8588
- const entries = dict[category];
8589
- const translated = entries[normalized];
8590
- if (translated) {
8591
- return translated;
8592
- }
8593
- }
8594
- return undefined;
8595
- }
8596
-
8597
- /**
8598
- * Grammar-Aware Transformer
8599
- *
8600
- * Transforms hyperscript statements between languages using the
8601
- * generalized grammar system. The key insight is that semantic
8602
- * roles are universal - only their surface realization differs.
8603
- */
8604
- // =============================================================================
8605
- // Compound Statement Handling
8606
- // =============================================================================
8607
- /**
8608
- * Get all command keywords including translated ones for a locale.
8609
- */
8610
- function getCommandKeywordsForLocale(locale) {
8611
- const keywords = new Set(ENGLISH_COMMANDS);
8612
- // Add translated command keywords from dictionaries
8613
- const dict = dictionaries[locale];
8614
- if (dict?.commands) {
8615
- Object.values(dict.commands).forEach(cmd => {
8616
- if (typeof cmd === 'string') {
8617
- keywords.add(cmd.toLowerCase());
8618
- }
8619
- });
8620
- }
8621
- return keywords;
8622
- }
8623
- /**
8624
- * English copula forms. A command keyword IMMEDIATELY after one of these is a
8625
- * predicate adjective, not a command verb: in `if my value is empty add .error
8626
- * to me` the `empty` belongs to the condition (`empty` is also a hyperscript
8627
- * command, v0.9.90). Without this guard the condition/body scans cut the
8628
- * condition at `is` and displace the adjective into the body's argument zone,
8629
- * where it anchors a spurious `empty-{lang}-generated` parse AND steals a
8630
- * neighboring role (the empty \u00d78 bn/hi/tr family). Source-locale copulas are
8631
- * added via the dictionary in `isPredicateAdjectivePosition`.
8632
- */
8633
- const EN_COPULAS = ['is', 'are', 'was', 'were', 'am', 'be'];
8634
- function getCopulasForLocale(locale) {
8635
- const copulas = new Set(EN_COPULAS);
8636
- if (locale !== 'en') {
8637
- for (const form of EN_COPULAS) {
8638
- copulas.add(translateWord(form, 'en', locale).toLowerCase());
8639
- }
8640
- }
8641
- return copulas;
8642
- }
8643
- /**
8644
- * True when tokens[i] sits right after a copula \u2014 a predicate-adjective
8645
- * position that must never be read as the start of a body command.
8646
- */
8647
- function isPredicateAdjectivePosition(tokens, i, copulas) {
8648
- const prev = tokens[i - 1]?.toLowerCase();
8649
- return !!prev && copulas.has(prev);
8650
- }
8651
- /**
8652
- * Source-locale surface forms of the `for` command keyword and the loop's
8653
- * `in` preposition, for the loop-head test below.
8654
- */
8655
- function getForLoopWordsForLocale(locale) {
8656
- const forWords = new Set(['for']);
8657
- const inWords = new Set(['in']);
8658
- if (locale !== 'en') {
8659
- forWords.add(translateWord('for', 'en', locale).toLowerCase());
8660
- inWords.add(translateWord('in', 'en', locale).toLowerCase());
8661
- }
8662
- return { forWords, inWords };
8663
- }
8664
- /**
8665
- * True when a `for` at tokens[i] heads a real loop. Hyperscript's only
8666
- * statement-head `for` is `for <var> in <iterable>`, so a `for` with no `in`
8667
- * among the tokens before the next command keyword is a ROLE PHRASE \u2014 take's
8668
- * target (`take .active from .tab-button for me`) or a duration (`for 2s`) \u2014
8669
- * and must never split the statement: the split shattered the phrase into a
8670
- * dangling `then for me` clause, which six SOV languages then parsed as a
8671
- * spurious `for` loop with patient "me" (the take-class \u00d76 family).
8672
- */
8673
- function isLoopHeadFor(tokens, i, inWords, commandKeywords) {
8674
- for (let j = i + 1; j < tokens.length; j++) {
8675
- const lt = tokens[j].toLowerCase();
8676
- if (inWords.has(lt))
8677
- return true;
8678
- if (commandKeywords.has(lt))
8679
- return false;
8680
- }
8681
- return false;
8682
- }
8683
- /**
8684
- * Repair a FRONTED Hebrew accusative marker in transformed output.
8685
- *
8686
- * When an event-handler body leads with a command-modifier (`on click once add \u2026`,
8687
- * via {@link GrammarTransformer.tryTransformEventWithModifierBody}) or is a control
8688
- * block (`on blur if \u2026 add \u2026 end`, via `tryTransformEventWithBlockBody`), the body
8689
- * command's accusative marker \u05d0\u05ea can be emitted AHEAD of its verb \u2014 `add .error to me`
8690
- * renders `\u2026 \u05d0\u05ea \u05d4\u05d5\u05e1\u05e3 .error \u2026` instead of the canonical `\u2026 \u05d4\u05d5\u05e1\u05e3 \u05d0\u05ea .error \u2026`. \u05d0\u05ea before
8691
- * a verb is always ungrammatical Hebrew (it only ever marks a FOLLOWING definite
8692
- * object), and the semantic parser drops the command in every parse path (fused-event,
8693
- * multi-clause, conditional-body) when the marker is fronted but parses it when the
8694
- * marker follows the verb. So an `<accusative-marker> <command-verb>` adjacency is a
8695
- * pure transformer artifact: swap it back. Idempotent and safe \u2014 only touches `\u05d0\u05ea
8696
- * <verb>`, never the ~40 generated `<verb> \u05d0\u05ea {patient}` patterns that embed \u05d0\u05ea legitimately.
8697
- */
8698
- function repairHebrewFrontedAccusative(text) {
8699
- const ACC = '\u05d0\u05ea'; // hebrewProfile.markers patient marker
8700
- const verbs = getCommandKeywordsForLocale('he');
8701
- const tokens = text.split(/\s+/);
8702
- let changed = false;
8703
- for (let i = 0; i + 1 < tokens.length; i++) {
8704
- if (tokens[i] === ACC && verbs.has(tokens[i + 1].toLowerCase())) {
8705
- [tokens[i], tokens[i + 1]] = [tokens[i + 1], tokens[i]];
8706
- changed = true;
8707
- i++; // skip the marker we just moved
8708
- }
8709
- }
8710
- return changed ? tokens.join(' ') : text;
8711
- }
8712
- /**
8713
- * Detect a reactive block (`live ... end`, `when X changes Y [end]`,
8714
- * `unless X Y [end]`) and decompose it. Returns `null` when the input
8715
- * is not a reactive block, when there's content after the matched
8716
- * `end`, or when the heuristic can't locate a body \u2014 all of which fall
8717
- * through to the standard `parseStatement` path.
8718
- */
8719
- function extractBlockStructure(input, sourceLocale) {
8720
- const tokens = input.split(/\s+/);
8721
- const head = tokens[0]?.toLowerCase();
8722
- if (!head || !BLOCK_HEAD_KEYWORDS.has(head))
8723
- return null;
8724
- // Depth-aware match for the closing `end` so nested blocks
8725
- // (`live when X changes Y end end`) slice correctly.
8726
- let depth = 1;
8727
- let endIdx = -1;
8728
- for (let i = 1; i < tokens.length; i++) {
8729
- const t = tokens[i].toLowerCase();
8730
- if (BLOCK_HEAD_KEYWORDS.has(t))
8731
- depth++;
8732
- else if (t === 'end') {
8733
- depth--;
8734
- if (depth === 0) {
8735
- endIdx = i;
8736
- break;
8737
- }
8738
- }
8739
- }
8740
- // If there's trailing content after the matched `end`, bail out and
8741
- // let the existing splitter handle it. (`splitOnThen` normally
8742
- // separates trailing code before we get here.)
8743
- if (endIdx !== -1 && endIdx !== tokens.length - 1)
8744
- return null;
8745
- const inner = endIdx !== -1 ? tokens.slice(1, endIdx) : tokens.slice(1);
8746
- const base = { headKeyword: tokens[0], body: '' };
8747
- if (endIdx !== -1)
8748
- base.tailKeyword = tokens[endIdx];
8749
- if (head === 'live') {
8750
- return { ...base, body: inner.join(' ') };
8751
- }
8752
- if (head === 'when') {
8753
- // Reactive: `when <expr> changes <body>`. Without `changes`, fall
8754
- // through to the standard event-wait path (parseConditional).
8755
- const idx = inner.findIndex(t => t.toLowerCase() === 'changes');
8756
- if (idx >= 0) {
8757
- return {
8758
- ...base,
8759
- prefixExpr: inner.slice(0, idx).join(' '),
8760
- connector: inner[idx],
8761
- body: inner.slice(idx + 1).join(' '),
8762
- };
8763
- }
8764
- return null;
8765
- }
8766
- // `unless <cond> <body>`: condition runs up to the first command
8767
- // keyword in `inner`. Heuristic \u2014 works because hyperscript bodies
8768
- // always start with a command verb, and `unless` conditions rarely
8769
- // contain bare command keywords as values. A candidate right after a
8770
- // copula (`\u2026 is empty`) is a predicate adjective inside the condition,
8771
- // not a body verb \u2014 skip it.
8772
- const commands = getCommandKeywordsForLocale(sourceLocale);
8773
- const copulas = getCopulasForLocale(sourceLocale);
8774
- let bodyStart = -1;
8775
- for (let i = 0; i < inner.length; i++) {
8776
- if (commands.has(inner[i].toLowerCase()) && !isPredicateAdjectivePosition(inner, i, copulas)) {
8777
- bodyStart = i;
8778
- break;
8779
- }
8780
- }
8781
- if (bodyStart <= 0)
8782
- return null;
8783
- return {
8784
- ...base,
8785
- prefixExpr: inner.slice(0, bodyStart).join(' '),
8786
- body: inner.slice(bodyStart).join(' '),
8787
- };
8788
- }
8789
- function splitCompoundStatement(input, sourceLocale) {
8790
- // First, split on newlines (preserving non-empty lines)
8791
- const lines = input
8792
- .split(/\n/)
8793
- .map(line => line.trim())
8794
- .filter(line => line.length > 0);
8795
- // If we have multiple lines, treat each as a separate part
8796
- // (but still need to handle "then" within each line)
8797
- const parts = [];
8798
- for (const line of lines) {
8799
- const lineParts = splitOnThen(line, sourceLocale);
8800
- // Further split each part on command boundaries
8801
- for (const part of lineParts) {
8802
- const commandParts = splitOnCommandBoundaries(part, sourceLocale);
8803
- parts.push(...commandParts);
8804
- }
8805
- }
8806
- return parts;
8807
- }
8808
- /**
8809
- * Split a compound statement while preserving line structure metadata.
8810
- * This tracks indentation and blank lines for reconstruction.
8811
- */
8812
- function splitCompoundStatementWithMetadata(input, sourceLocale) {
8813
- const rawLines = input.split('\n');
8814
- const lineMetadata = [];
8815
- const parts = [];
8816
- const partToLineIndex = [];
8817
- for (let lineIndex = 0; lineIndex < rawLines.length; lineIndex++) {
8818
- const rawLine = rawLines[lineIndex];
8819
- // Capture leading whitespace
8820
- const indentMatch = rawLine.match(/^(\s*)/);
8821
- const originalIndent = indentMatch ? indentMatch[1] : '';
8822
- const trimmed = rawLine.trim();
8823
- lineMetadata.push({
8824
- content: trimmed,
8825
- originalIndent,
8826
- isBlank: trimmed.length === 0,
8827
- });
8828
- if (trimmed.length > 0) {
8829
- // Process non-empty lines for "then" and command boundaries
8830
- const lineParts = splitOnThen(trimmed, sourceLocale);
8831
- for (const part of lineParts) {
8832
- const commandParts = splitOnCommandBoundaries(part, sourceLocale);
8833
- for (const cmdPart of commandParts) {
8834
- parts.push(cmdPart);
8835
- partToLineIndex.push(lineIndex);
8836
- }
8837
- }
8838
- }
8839
- }
8840
- return { parts, lineMetadata, partToLineIndex };
8841
- }
8842
- /**
8843
- * Normalize indentation to consistent 4-space levels.
8844
- * Preserves relative indentation structure while standardizing spacing.
8845
- */
8846
- function normalizeIndentation(lineMetadata) {
8847
- // Find non-blank lines with indentation
8848
- const indentedLines = lineMetadata.filter(m => !m.isBlank && m.originalIndent.length > 0);
8849
- if (indentedLines.length === 0) {
8850
- // No indented lines, return empty strings
8851
- return lineMetadata.map(() => '');
8852
- }
8853
- // Find minimum non-zero indent (the base unit)
8854
- const indentLengths = indentedLines.map(m => {
8855
- // Convert tabs to 4 spaces for consistent measurement
8856
- const normalized = m.originalIndent.replace(/\t/g, ' ');
8857
- return normalized.length;
8858
- });
8859
- const minIndent = Math.min(...indentLengths);
8860
- const baseUnit = minIndent > 0 ? minIndent : 4;
8861
- // Normalize each line's indentation
8862
- return lineMetadata.map(meta => {
8863
- if (meta.isBlank) {
8864
- return ''; // Blank lines get no indentation
8865
- }
8866
- if (meta.originalIndent.length === 0) {
8867
- return ''; // No original indent
8868
- }
8869
- // Convert tabs and calculate level
8870
- const normalized = meta.originalIndent.replace(/\t/g, ' ');
8871
- const level = Math.round(normalized.length / baseUnit);
8872
- return ' '.repeat(level); // 4 spaces per level
8873
- });
8874
- }
8875
- /**
8876
- * Reconstruct output with preserved line structure.
8877
- * Maps transformed parts back to their original lines with proper indentation.
8878
- */
8879
- function reconstructWithLineStructure(transformedParts, lineMetadata, partToLineIndex, targetThen) {
8880
- // If there's only one non-blank line, simple case
8881
- const nonBlankCount = lineMetadata.filter(m => !m.isBlank).length;
8882
- if (nonBlankCount <= 1 && transformedParts.length <= 1) {
8883
- const normalizedIndents = normalizeIndentation(lineMetadata);
8884
- const result = [];
8885
- for (let i = 0; i < lineMetadata.length; i++) {
8886
- if (lineMetadata[i].isBlank) {
8887
- result.push('');
8888
- }
8889
- else if (transformedParts.length > 0) {
8890
- result.push(normalizedIndents[i] + transformedParts[0]);
8891
- }
8892
- }
8893
- return result.join('\n');
8894
- }
8895
- // Normalize indentation
8896
- const normalizedIndents = normalizeIndentation(lineMetadata);
8897
- // Group transformed parts by their original line
8898
- const partsPerLine = new Map();
8899
- for (let i = 0; i < transformedParts.length; i++) {
8900
- const lineIdx = partToLineIndex[i];
8901
- if (!partsPerLine.has(lineIdx)) {
8902
- partsPerLine.set(lineIdx, []);
8903
- }
8904
- partsPerLine.get(lineIdx).push(transformedParts[i]);
8905
- }
8906
- // Build result lines
8907
- const result = [];
8908
- for (let i = 0; i < lineMetadata.length; i++) {
8909
- const meta = lineMetadata[i];
8910
- const indent = normalizedIndents[i];
8911
- if (meta.isBlank) {
8912
- result.push('');
8913
- }
8914
- else {
8915
- const lineParts = partsPerLine.get(i) || [];
8916
- if (lineParts.length > 0) {
8917
- // Join multiple parts on same line with "then"
8918
- const lineContent = lineParts.join(` ${targetThen} `);
8919
- result.push(indent + lineContent);
8920
- }
8921
- }
8922
- }
8923
- return result.join('\n');
8924
- }
8925
- /**
8926
- * Split a statement on command keyword boundaries.
8927
- * E.g., "wait 2s toggle .highlight" \u2192 ["wait 2s", "toggle .highlight"]
8928
- *
8929
- * Special cases:
8930
- * - "on <event> <command>" stays together (event handler with first command)
8931
- * - Modifiers like "to", "from" don't trigger splits
8932
- */
8933
- /**
8934
- * English modifier keywords that should not trigger command boundary splits.
8935
- * These are the base set; localized equivalents (e.g. Japanese `\u306b`, Spanish
8936
- * `a`, Arabic `\u0625\u0644\u0649`) are layered on per source locale by
8937
- * `getBoundaryModifiersForLocale`, so a preposition in a non-English source
8938
- * is also recognized as a modifier rather than a spurious command boundary.
8939
- */
8940
- const BOUNDARY_MODIFIERS = new Set([
8941
- 'to',
8942
- 'into',
8943
- 'from',
8944
- 'with',
8945
- 'by',
8946
- 'as',
8947
- 'at',
8948
- 'in',
8949
- 'on',
8950
- 'of',
8951
- 'over',
8952
- ]);
8953
- /**
8954
- * Boundary modifiers resolved for a given source locale: the English base set
8955
- * (always kept, since input may mix English keywords) plus every grammatical
8956
- * marker form declared by the locale's profile. Profile markers are the
8957
- * surface realizations of semantic roles (destination, source, style, \u2026) \u2014
8958
- * they always bind to a following value, so none should be treated as a
8959
- * command boundary. Cached per locale because profiles are static.
8960
- */
8961
- const boundaryModifiersCache = new Map();
8962
- function getBoundaryModifiersForLocale(locale) {
8963
- const cached = boundaryModifiersCache.get(locale);
8964
- if (cached)
8965
- return cached;
8966
- const modifiers = new Set(BOUNDARY_MODIFIERS);
8967
- const profile = getProfile(locale);
8968
- profile?.markers.forEach(marker => {
8969
- const form = marker.form.replace(/^-|-$/g, '').toLowerCase();
8970
- if (form)
8971
- modifiers.add(form);
8972
- marker.alternatives?.forEach(alt => {
8973
- const altForm = alt.replace(/^-|-$/g, '').toLowerCase();
8974
- if (altForm)
8975
- modifiers.add(altForm);
8976
- });
8977
- });
8978
- boundaryModifiersCache.set(locale, modifiers);
8979
- return modifiers;
8980
- }
8981
- /**
8982
- * Commands for which a trailing `on <element>` is the TARGET the command acts
8983
- * on (a destination), not a new event-handler clause. For these, the locative
8984
- * `on` must neither split the statement (`splitOnCommandBoundaries`) nor be read
8985
- * as a fresh `event` role (`buildArgumentModifierMap`) \u2014 see those call sites.
8986
- *
8987
- * Deliberately narrow. `on` is overloaded (event-handler head vs. locative
8988
- * target), and the change has cross-language blast radius, so we only enable
8989
- * the locative reading for the DOM class/attribute mutators where it's the
8990
- * documented idiom (`toggle .open on #menu`, `toggle @hidden on #panel`).
8991
- *
8992
- * `trigger`/`send` \u2014 `trigger X on Y` / `send X on Y` fire an event on a target
8993
- * element; the trailing `on Y` is that target (destination), not a new clause.
8994
- * Previously excluded: keeping `on Y` attached produced `trigger X \u2192 #y` output
8995
- * that was thought to destabilise the semantic parser's multi-line behaviour
8996
- * fallback (behavior-sortable's `trigger sortable:start on me`). That concern is
8997
- * stale \u2014 the recent semantic-parser body increments fold the surrounding block
8998
- * cleanly, and splitting instead injected a spurious `then` (`disparar
8999
- * sortable:start entonces en yo`) that glued the following `repeat until event`
9000
- * loop into a then-chain and dropped it. Keeping `on Y` attached restores
9001
- * behavior-sortable to faithful across the SVO languages.
9002
- *
9003
- * Excluded on purpose:
9004
- * - `set` \u2014 `set @attr to V on Y` carries BOTH `to` (value) and `on` (target);
9005
- * English marks both as `destination`, so remapping `on` would collide with
9006
- * and clobber the value. Needs distinct value/target roles first (deferred).
9007
- * - `put` \u2014 `put X on Y into Z` has the same dual-destination collision.
9008
- *
9009
- * `add`/`remove` use `to`/`from` for their target in practice (not `on`), so
9010
- * their inclusion is a harmless no-op that documents intent.
9011
- */
9012
- const ON_TARGET_COMMANDS = new Set(['toggle', 'add', 'remove', 'trigger', 'send']);
9013
- /**
9014
- * Find the command verb of a partially-collected statement: the first command
9015
- * keyword that is neither the event-handler head (`on`/\u305d\u306e\u4ed6 event markers) nor
9016
- * an argument-introducing preposition. For `on click toggle .open` \u2192 `toggle`;
9017
- * for `trigger sortable:start` \u2192 `trigger`.
9018
- */
9019
- function commandVerbOf(tokens, commandKeywords) {
9020
- for (const token of tokens) {
9021
- const lt = token.toLowerCase();
9022
- if (commandKeywords.has(lt) && !BOUNDARY_MODIFIERS.has(lt) && !EVENT_KEYWORDS.has(lt)) {
9023
- return lt;
9024
- }
9025
- }
9026
- return null;
9027
- }
9028
- /**
9029
- * Block-introducing keywords whose body should not be split at command
9030
- * boundaries by `splitOnCommandBoundaries`. Inputs starting with one of
9031
- * these are also routed around `parseStatement` entirely by
9032
- * `extractBlockStructure` + `transformBlock` so block-syntactic tokens
9033
- * never reach `parseCommand`/`parseConditional` (where they'd be
9034
- * misinterpreted as command verbs or swept into role values).
9035
- *
9036
- * `if` deliberately stays out: `if X then Y end` already works via the
9037
- * `splitOnThen` + `parseConditional` path.
9038
- */
9039
- const BLOCK_HEAD_KEYWORDS = new Set(['live', 'when', 'unless']);
9040
- /**
9041
- * Source-language block-introducing command keywords whose body is a clause plus a
9042
- * branch/loop body (harness source is English, so these are English-keyed). Used to
9043
- * locate a block body inside an event handler and to track block depth when finding
9044
- * a top-level `else`.
9045
- */
9046
- const BLOCK_BODY_KEYWORDS = new Set(['if', 'repeat', 'unless', 'while', 'for']);
9047
- /**
9048
- * SVO targets that mark the object/patient with a particle (he \u05d0\u05ea, zh \u628a) and so
9049
- * mangle an inline `on <event> unless <cond> <body>` guard \u2014 the unless tail is
9050
- * swept into one patient blob and the marker lands ahead of the condition. These
9051
- * route through `tryTransformEventWithUnlessGuard`. SOV/VSO object-markers
9052
- * (ja/ko/tr/ar) are excluded: their event does not lead, so the SVO event-first
9053
- * emission there would be wrong (and they don't exhibit the artifact today).
9054
- */
9055
- const UNLESS_GUARD_OBJECT_MARKING_LOCALES = new Set(['he', 'zh']);
9056
- function splitOnCommandBoundaries(input, sourceLocale) {
9057
- const commandKeywords = getCommandKeywordsForLocale(sourceLocale);
9058
- const boundaryModifiers = getBoundaryModifiersForLocale(sourceLocale);
9059
- const { forWords, inWords } = getForLoopWordsForLocale(sourceLocale);
9060
- const tokens = input.split(/\s+/);
9061
- if (tokens.length === 0)
9062
- return [input];
9063
- const parts = [];
9064
- let currentPart = [];
9065
- // Check if this starts with an event handler pattern (on/em/en/bei/\u3067 + event)
9066
- const firstTokenLower = tokens[0]?.toLowerCase();
9067
- const isEventHandler = EVENT_KEYWORDS.has(firstTokenLower);
9068
- // If it's an event handler, the first command after the event is part of the handler
9069
- // So we need to track whether we've seen the first command yet
9070
- let seenFirstCommand = !isEventHandler; // If not event handler, we're already past the "first command" phase
9071
- // Track block-scope depth (live/when/bind/if/unless/for/while/...). While
9072
- // inside a block, do not split on command boundaries \u2014 the body belongs
9073
- // to the block head and must transform as one unit. See comments on
9074
- // BLOCK_HEAD_KEYWORDS for the failure mode this prevents.
9075
- let blockDepth = 0;
9076
- for (let i = 0; i < tokens.length; i++) {
9077
- const token = tokens[i];
9078
- const lowerToken = token.toLowerCase();
9079
- // Update block-scope depth before any split decision.
9080
- if (BLOCK_HEAD_KEYWORDS.has(lowerToken)) {
9081
- blockDepth++;
9082
- }
9083
- else if (lowerToken === 'end' && blockDepth > 0) {
9084
- blockDepth--;
9085
- }
9086
- // If this is a command keyword and we already have tokens in current part
9087
- if (commandKeywords.has(lowerToken) && currentPart.length > 0) {
9088
- // Check if the previous token looks like it could end a command
9089
- const prevToken = currentPart[currentPart.length - 1];
9090
- const prevLower = prevToken.toLowerCase();
9091
- // For event handlers: don't split before the first command
9092
- // E.g., "on click wait 1s" should stay together
9093
- if (!seenFirstCommand) {
9094
- // Mark that we've now seen the first command
9095
- seenFirstCommand = true;
9096
- currentPart.push(token);
9097
- continue;
9098
- }
9099
- // Don't split inside a block (live/when/bind/unless body, etc.).
9100
- // The block head and its body must transform as one statement.
9101
- if (blockDepth > 0) {
9102
- currentPart.push(token);
9103
- continue;
9104
- }
9105
- // A `for` that doesn't head a real loop (`for <var> in <iterable>`) is a
9106
- // role phrase of the current command \u2014 see isLoopHeadFor.
9107
- if (forWords.has(lowerToken) && !isLoopHeadFor(tokens, i, inWords, commandKeywords)) {
9108
- currentPart.push(token);
9109
- continue;
9110
- }
9111
- // Locative `on` for a DOM-target command (`toggle X on Y`) is the target
9112
- // element, NOT a new command. `on` lands in `commandKeywords` only
9113
- // incidentally \u2014 the EN dictionary registers `commands.on = 'on'` for the
9114
- // event-handler head \u2014 so without this guard the destination `on` split
9115
- // the statement, the join re-inserted a spurious `then` (\u062b\u0645/pagkatapos/\u2026),
9116
- // and the dangling `on Y` was misread as a second event handler. Keep it
9117
- // attached so the role parser can assign it to `destination`. Restricted
9118
- // to ON_TARGET_COMMANDS so `trigger X on me` etc. keep their prior split.
9119
- if (BOUNDARY_MODIFIERS.has(lowerToken)) {
9120
- const verb = commandVerbOf(currentPart, commandKeywords);
9121
- if (verb && ON_TARGET_COMMANDS.has(verb)) {
9122
- currentPart.push(token);
9123
- continue;
9124
- }
9125
- // `set @attr to V on <scope>` (S1 tabs-aria): the trailing `on <scope>`
9126
- // is the element(s) the attribute is set on, not a new command. `set` is
9127
- // deliberately NOT in ON_TARGET_COMMANDS (its role parser would clobber
9128
- // the value), so this is a dedicated guard \u2014 kept attached only when a
9129
- // selector/reference scope follows, then positioned by transformSingle's
9130
- // set-scope handler (transformSetWithScope). A `set` whose `on` is
9131
- // followed by a verb is left to split as before.
9132
- if (lowerToken === 'on' && verb === 'set') {
9133
- const nextTok = tokens[i + 1];
9134
- const nextLower = nextTok?.toLowerCase();
9135
- const scopeLike = !!nextTok &&
9136
- (/^[#.<@[]/.test(nextTok) ||
9137
- nextLower === 'me' ||
9138
- nextLower === 'it' ||
9139
- nextLower === 'you');
9140
- if (scopeLike) {
9141
- currentPart.push(token);
9142
- continue;
9143
- }
9144
- }
9145
- }
9146
- if (!boundaryModifiers.has(prevLower) && !commandKeywords.has(prevLower)) {
9147
- // This looks like a command boundary - save current part and start new one
9148
- parts.push(currentPart.join(' '));
9149
- currentPart = [token];
9150
- continue;
9151
- }
9152
- }
9153
- currentPart.push(token);
9154
- }
9155
- // Add the last part
9156
- if (currentPart.length > 0) {
9157
- parts.push(currentPart.join(' '));
9158
- }
9159
- return parts.filter(p => p.length > 0);
9160
- }
9161
- /**
9162
- * Split a single line on "then" keywords.
9163
- */
9164
- function splitOnThen(input, sourceLocale) {
9165
- // Build regex pattern from all known "then" keywords
9166
- const thenKeywords = Array.from(THEN_KEYWORDS);
9167
- // Add any dictionary-specific "then" keyword for the source locale
9168
- const sourceDict = sourceLocale === 'en' ? null : dictionaries[sourceLocale];
9169
- if (sourceDict?.modifiers?.then) {
9170
- thenKeywords.push(sourceDict.modifiers.then);
9171
- }
9172
- // Also check logical.then since some dictionaries put it there
9173
- if (sourceDict?.logical?.then) {
9174
- thenKeywords.push((sourceDict?.logical).then);
9175
- }
9176
- // Create a regex that matches any "then" keyword as a whole word
9177
- // Use word boundaries to avoid matching "then" inside other words
9178
- const escapedKeywords = thenKeywords.map(k => k.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
9179
- const pattern = new RegExp(`\\s+(${escapedKeywords.join('|')})\\s+`, 'gi');
9180
- // Split on "then" keywords
9181
- const parts = input.split(pattern).filter(part => {
9182
- // Filter out the "then" keywords themselves (captured by the group)
9183
- const lowerPart = part.toLowerCase().trim();
9184
- return lowerPart && !thenKeywords.some(k => k.toLowerCase() === lowerPart);
9185
- });
9186
- return parts.map(p => p.trim()).filter(p => p.length > 0);
9187
- }
9188
- /**
9189
- * Get the "then" keyword in the target language.
9190
- * Checks both modifiers and logical sections since dictionaries vary.
9191
- */
9192
- function getTargetThenKeyword(targetLocale) {
9193
- if (targetLocale === 'en')
9194
- return 'then';
9195
- const targetDict = dictionaries[targetLocale];
9196
- if (!targetDict)
9197
- return 'then';
9198
- // Check modifiers first, then logical (dictionaries vary)
9199
- return (targetDict.modifiers?.then || targetDict.logical?.then || 'then');
9200
- }
9201
- // =============================================================================
9202
- // Derived Constants from Profiles
9203
- // =============================================================================
9204
- /**
9205
- * Derive event keywords from all language profiles.
9206
- * This replaces the hardcoded eventKeywords array.
9207
- */
9208
- function deriveEventKeywordsFromProfiles() {
9209
- const keywords = new Set();
9210
- // Add 'on' as the English default
9211
- keywords.add('on');
9212
- // Extract event markers from all profiles
9213
- for (const profile of Object.values(profiles)) {
9214
- for (const marker of profile.markers) {
9215
- if (marker.role === 'event') {
9216
- // Strip hyphen notation and add
9217
- const form = marker.form.replace(/^-|-$/g, '').toLowerCase();
9218
- if (form)
9219
- keywords.add(form);
9220
- // Add alternatives
9221
- marker.alternatives?.forEach(alt => {
9222
- const altForm = alt.replace(/^-|-$/g, '').toLowerCase();
9223
- if (altForm)
9224
- keywords.add(altForm);
9225
- });
9226
- }
9227
- }
9228
- }
9229
- return keywords;
9230
- }
9231
- /** Event keywords derived from language profiles */
9232
- const EVENT_KEYWORDS = deriveEventKeywordsFromProfiles();
9233
- /**
9234
- * Conjunctions that join multiple events in an event handler head
9235
- * (`on click or keypress ...`). Source hyperscript is English, so the
9236
- * canonical form is `or`; localized equivalents are added defensively for
9237
- * non-English sources.
9238
- */
9239
- const EVENT_CONJUNCTIONS = new Set(['or']);
9240
- /**
9241
- * Command-modifier keywords that may lead an event handler body
9242
- * (`on click async fetch \u2026`, `on click once add \u2026`). They modify the handler /
9243
- * following command rather than acting as the verb, so the transformer must lift
9244
- * them out before role assignment instead of treating them as the action.
9245
- * Source hyperscript is English, so these are the canonical English forms \u2014 the
9246
- * semantic parser recognizes the same literals when stripping them pre-parse.
9247
- */
9248
- const BODY_MODIFIER_KEYWORDS = new Set([
9249
- 'async',
9250
- 'once',
9251
- 'debounced',
9252
- 'debounce',
9253
- 'throttled',
9254
- 'throttle',
9255
- ]);
9256
- // =============================================================================
9257
- // Helper: Dynamic Modifier Map
9258
- // =============================================================================
9259
- /**
9260
- * Generates a lookup map for semantic roles based on the language profile.
9261
- * Maps markers (e.g., 'to', '\u306b', 'into', '\u0625\u0644\u0649') to their semantic roles.
9262
- * This enables parsing non-English input by using the profile's markers.
9263
- */
9264
- function generateModifierMap(profile) {
9265
- const map = {};
9266
- // Map markers to roles from the profile
9267
- profile.markers.forEach(marker => {
9268
- // Strip hyphen notation for suffix/prefix markers
9269
- const form = marker.form.replace(/^-|-$/g, '').toLowerCase();
9270
- if (form) {
9271
- map[form] = marker.role;
9272
- }
9273
- // Map alternatives if they exist (e.g., Korean vowel harmony variants)
9274
- marker.alternatives?.forEach(alt => {
9275
- const altForm = alt.replace(/^-|-$/g, '').toLowerCase();
9276
- if (altForm) {
9277
- map[altForm] = marker.role;
9278
- }
9279
- });
9280
- });
9281
- // Add English modifiers as fallback (don't override profile-specific markers)
9282
- for (const [key, role] of Object.entries(ENGLISH_MODIFIER_ROLES)) {
9283
- if (!(key in map)) {
9284
- map[key] = role;
9285
- }
9286
- }
9287
- return map;
9288
- }
9289
- /**
9290
- * Modifier map for parsing the ARGUMENTS of a statement (everything after the
9291
- * command verb), as opposed to the statement head.
9292
- *
9293
- * Why a separate map: a few languages reuse one word for both the event-handler
9294
- * head marker and a locative/target preposition. English is the prime case \u2014
9295
- * `on` is the event head (`on click \u2026`) AND the toggle/set target preposition
9296
- * (`toggle .x on #y`). `generateModifierMap` maps `on \u2192 event` (from the EN
9297
- * profile's event marker), so when the destination `on` was read in argument
9298
- * position it overwrote the already-captured head event (the `click` got
9299
- * dropped, e.g. `toggle @hidden on #panel` \u2192 `\u2026 \u0639\u0646\u062f #panel` with \u0646\u0642\u0631/click gone).
9300
- *
9301
- * The fix: in argument position, remap the event role to `destination`. This is
9302
- * safe because a trigger event only ever appears at the statement head (which is
9303
- * consumed before argument parsing begins). Any "event marker" reached while
9304
- * scanning arguments is therefore being used as a locative \u2014 i.e. the element
9305
- * the command acts ON \u2014 which is exactly the `destination` role (the semantic
9306
- * parser also models `toggle .x on #y` as patient `.x` + destination `#y`).
9307
- *
9308
- * Restricted to (a) SVO source profiles and (b) ON_TARGET_COMMANDS:
9309
- * - SVO: in VSO/SOV languages the event marker can legitimately appear
9310
- * mid-statement (Arabic VSO renders an event handler as `\u0628\u062f\u0644 X \u0639\u0646\u062f \u0646\u0642\u0631`,
9311
- * classified as a command), so remapping there would mis-read a real event
9312
- * as a destination. English \u2014 and the en\u2192lang gate path \u2014 is SVO.
9313
- * - command: only the DOM class/attr mutators use a locative `on` target.
9314
- * For other commands (`trigger X on me`, `set X to V on Y`) the remap is
9315
- * either destabilising or collides with `to`/`into` \u2014 see ON_TARGET_COMMANDS.
9316
- *
9317
- * When neither applies the unmodified map is returned (event stays event).
9318
- */
9319
- function buildArgumentModifierMap(profile, actionVerb) {
9320
- const map = generateModifierMap(profile);
9321
- const verb = actionVerb?.toLowerCase();
9322
- if (profile.wordOrder !== 'SVO' || !verb || !ON_TARGET_COMMANDS.has(verb)) {
9323
- return map;
9324
- }
9325
- const remapped = {};
9326
- for (const [form, role] of Object.entries(map)) {
9327
- remapped[form] = role === 'event' ? 'destination' : role;
9328
- }
9329
- return remapped;
9330
- }
9331
- // =============================================================================
9332
- // Statement Parser
9333
- // =============================================================================
9334
- /**
9335
- * Parse a hyperscript statement into semantic roles
9336
- * This is the core analysis step that identifies WHAT each part means
9337
- */
9338
- function parseStatement(input, sourceLocale = 'en') {
9339
- const profile = getProfile(sourceLocale);
9340
- if (!profile)
9341
- return null;
9342
- const tokens = tokenize(input, profile);
9343
- // Identify statement type and extract roles
9344
- const statementType = identifyStatementType(tokens, profile);
9345
- switch (statementType) {
9346
- case 'event-handler':
9347
- return parseEventHandler(tokens, profile);
9348
- case 'command':
9349
- return parseCommand(tokens, profile);
9350
- case 'conditional':
9351
- return parseConditional(tokens);
9352
- default:
9353
- return null;
9354
- }
9355
- }
9356
- /**
9357
- * Known suffixes that may attach to words without spaces.
9358
- * These are split off during tokenization for proper parsing.
9359
- */
9360
- const ATTACHED_SUFFIXES = {
9361
- // Chinese: \u65f6 (time/when) often attaches to events like \u70b9\u51fb\u65f6 (when clicking)
9362
- zh: ['\u65f6', '\u7684', '\u5730', '\u5f97'],
9363
- // Japanese: Some particles may attach in casual writing
9364
- ja: [],
9365
- // Korean: Particles sometimes written without spaces
9366
- ko: [],
9367
- };
9368
- /**
9369
- * Known prefixes that may attach to words without spaces.
9370
- */
9371
- const ATTACHED_PREFIXES = {
9372
- // Chinese: \u5f53 (when) sometimes written attached
9373
- zh: ['\u5f53'],
9374
- // Arabic: Prepositions that attach
9375
- ar: ['\u0628\u0640', '\u0643\u0640', '\u0648'],
9376
- };
9377
- /**
9378
- * Post-process tokens to split attached suffixes/prefixes.
9379
- * E.g., "\u70b9\u51fb\u65f6" \u2192 ["\u70b9\u51fb", "\u65f6"]
9380
- */
9381
- function splitAttachedAffixes(tokens, locale) {
9382
- const suffixes = ATTACHED_SUFFIXES[locale] || [];
9383
- const prefixes = ATTACHED_PREFIXES[locale] || [];
9384
- if (suffixes.length === 0 && prefixes.length === 0) {
9385
- return tokens;
9386
- }
9387
- const result = [];
9388
- for (const token of tokens) {
9389
- // Skip CSS selectors and numbers
9390
- if (/^[#.<@]/.test(token) || /^\d+/.test(token)) {
9391
- result.push(token);
9392
- continue;
9393
- }
9394
- let processed = token;
9395
- let prefix = '';
9396
- let suffix = '';
9397
- // Check for attached prefixes
9398
- for (const p of prefixes) {
9399
- if (processed.startsWith(p) && processed.length > p.length) {
9400
- prefix = p;
9401
- processed = processed.slice(p.length);
9402
- break;
9403
- }
9404
- }
9405
- // Check for attached suffixes
9406
- for (const s of suffixes) {
9407
- if (processed.endsWith(s) && processed.length > s.length) {
9408
- suffix = s;
9409
- processed = processed.slice(0, -s.length);
9410
- break;
9411
- }
9412
- }
9413
- // Add tokens in order: prefix, main, suffix
9414
- if (prefix)
9415
- result.push(prefix);
9416
- if (processed)
9417
- result.push(processed);
9418
- if (suffix)
9419
- result.push(suffix);
9420
- }
9421
- return result;
9422
- }
9423
- /**
9424
- * Simple tokenizer that handles:
9425
- * - Keywords (from dictionary)
9426
- * - CSS selectors (#id, .class, <tag/>)
9427
- * - String literals
9428
- * - Numbers
9429
- * - Attached suffixes/prefixes (language-specific)
9430
- */
9431
- function tokenize(input, profile) {
9432
- // Split on whitespace, preserving selectors and strings
9433
- const tokens = [];
9434
- let current = '';
9435
- let inSelector = false;
9436
- let selectorDepth = 0;
9437
- let bracketDepth = 0;
9438
- let parenDepth = 0;
9439
- for (let i = 0; i < input.length; i++) {
9440
- const char = input[i];
9441
- // Track CSS selector context
9442
- if (char === '<') {
9443
- inSelector = true;
9444
- selectorDepth++;
9445
- }
9446
- else if (char === '>' && inSelector) {
9447
- selectorDepth--;
9448
- if (selectorDepth === 0)
9449
- inSelector = false;
9450
- }
9451
- // Track event-guard / attribute brackets so `[key is 'Escape']` (which has
9452
- // internal spaces) stays a single token instead of splitting into
9453
- // `[key` / `is` / `'Escape']` \u2014 which mis-assigns `is` as the action verb.
9454
- if (char === '[') {
9455
- bracketDepth++;
9456
- }
9457
- else if (char === ']' && bracketDepth > 0) {
9458
- bracketDepth--;
9459
- }
9460
- // Track EVERY parenthesized group as a depth scope so it stays ONE token:
9461
- //
9462
- // - An ATTACHED `(` (a call or event destructure: `pointerdown(clientX,
9463
- // clientY)`, `Resizable(a, b)`) must not split at the comma-space \u2014 the
9464
- // event-handler-head reorder would separate the halves and drop the
9465
- // whole handler (tl behavior-resizable degenerate).
9466
- // - A STANDALONE `(` opening an expression (`to (the value of #price as
9467
- // Number) * (my value as Number)`) must be OPAQUE to role segmentation:
9468
- // left loose, its interior `of`/`as`/`from` keywords hit the argument
9469
- // modifier map, so the parser split the expression across roles \u2014 the
9470
- // R1 cluster E mangle (computed-value: the transformer reordered INSIDE
9471
- // the parens, embedded the event phrase mid-expression, and dropped the
9472
- // whole second operand in every language). Interior keywords still
9473
- // translate IN PLACE: role values are re-split on whitespace by
9474
- // translateMultiWordValue, and translateWord strips paren punctuation
9475
- // before its dictionary lookup (`($count or 0)` \u2192 `($count o 0)` in tl \u2014
9476
- // the operator translates without the group being torn apart).
9477
- if (char === '(') {
9478
- parenDepth++;
9479
- }
9480
- else if (char === ')' && parenDepth > 0) {
9481
- parenDepth--;
9482
- }
9483
- // Split on whitespace unless inside a selector, a bracket guard, or an
9484
- // attached parenthesized argument list
9485
- if (/\s/.test(char) && !inSelector && bracketDepth === 0 && parenDepth === 0) {
9486
- if (current) {
9487
- tokens.push(current);
9488
- current = '';
9489
- }
9490
- }
9491
- else {
9492
- current += char;
9493
- }
9494
- }
9495
- if (current) {
9496
- tokens.push(current);
9497
- }
9498
- // Post-process to split attached affixes for languages that need it
9499
- return splitAttachedAffixes(tokens, profile.code);
9500
- }
9501
- /**
9502
- * Identify what type of statement this is
9503
- */
9504
- function identifyStatementType(tokens, profile) {
9505
- if (tokens.length === 0)
9506
- return 'unknown';
9507
- const firstToken = tokens[0].toLowerCase();
9508
- // Check for event handler
9509
- const eventMarker = profile.markers.find(m => m.role === 'event' && m.position === 'preposition');
9510
- if (eventMarker && firstToken === eventMarker.form.toLowerCase()) {
9511
- return 'event-handler';
9512
- }
9513
- // Check if first token is a known event keyword (derived from profiles)
9514
- if (EVENT_KEYWORDS.has(firstToken)) {
9515
- return 'event-handler';
9516
- }
9517
- // Check for conditional using shared constants
9518
- if (CONDITIONAL_KEYWORDS.has(firstToken)) {
9519
- return 'conditional';
9520
- }
9521
- return 'command';
9522
- }
9523
- /**
9524
- * Parse an event handler statement
9525
- * Pattern: on {event} {command} {target?} {modifiers?}
9526
- *
9527
- * Now handles modifiers like "by 3" in "on click increment #count by 3"
9528
- */
9529
- function parseEventHandler(tokens, profile) {
9530
- const roles = new Map();
9531
- // Skip the event keyword (e.g., 'on', '\u3067', '\u5f53', etc.) - derived from profiles
9532
- let startIndex = EVENT_KEYWORDS.has(tokens[0]?.toLowerCase()) ? 1 : 0;
9533
- // Next token is the event
9534
- if (tokens[startIndex]) {
9535
- const eventTokens = [tokens[startIndex]];
9536
- startIndex++;
9537
- // Fold "or"-conjoined events into the event value (e.g.
9538
- // "on click or keypress[...] toggle .active"). Without this, "or" would be
9539
- // read as the action verb and the second event swept into role values,
9540
- // hoisting "or <event2>" ahead of the command on reorder. Keeping the whole
9541
- // "<event1> or <event2>" string in the event role lets it translate and
9542
- // reorder as a single event clause. Only consumes "or" that immediately
9543
- // follows the event, before any source/action \u2014 so a later "or" in a guard
9544
- // or value is untouched.
9545
- while (tokens[startIndex] &&
9546
- EVENT_CONJUNCTIONS.has(tokens[startIndex].toLowerCase()) &&
9547
- tokens[startIndex + 1]) {
9548
- eventTokens.push(tokens[startIndex], tokens[startIndex + 1]);
9549
- startIndex += 2;
9550
- }
9551
- roles.set('event', {
9552
- role: 'event',
9553
- value: eventTokens.join(' '),
9554
- });
9555
- }
9556
- // Check for event source modifier before the action (e.g., "from #source" in "on input from #firstName set ...")
9557
- // Only handle 'from' (event source) \u2014 other modifiers like circumfix markers (Chinese \u65f6) should not be consumed.
9558
- if (tokens[startIndex] && tokens[startIndex].toLowerCase() === 'from' && tokens[startIndex + 1]) {
9559
- startIndex++; // skip 'from'
9560
- // Collect source value tokens until a command keyword is found
9561
- const sourceValue = [];
9562
- while (tokens[startIndex]) {
9563
- if (ENGLISH_COMMANDS.has(tokens[startIndex].toLowerCase()))
9564
- break;
9565
- sourceValue.push(tokens[startIndex]);
9566
- startIndex++;
9567
- }
9568
- if (sourceValue.length > 0) {
9569
- const value = sourceValue.join(' ');
9570
- roles.set('source', {
9571
- role: 'source',
9572
- value,
9573
- isSelector: /^[#.<@]/.test(value),
9574
- });
9575
- }
9576
- }
9577
- // Next token is the action (command verb)
9578
- if (tokens[startIndex]) {
9579
- roles.set('action', {
9580
- role: 'action',
9581
- value: tokens[startIndex],
9582
- });
9583
- startIndex++;
9584
- }
9585
- // Parse remaining tokens with modifier awareness (like parseCommand does)
9586
- // This handles "by 3" in "on click increment #count by 3".
9587
- // Argument map: for DOM-target commands the head event is already captured, so
9588
- // a locative `on` (`toggle @hidden on #panel`) maps to `destination` rather
9589
- // than overwriting the head `event`.
9590
- if (tokens[startIndex]) {
9591
- const modifierMap = buildArgumentModifierMap(profile, roles.get('action')?.value);
9592
- let currentRole = 'patient';
9593
- let currentValue = [];
9594
- for (let i = startIndex; i < tokens.length; i++) {
9595
- const token = tokens[i];
9596
- const mappedRole = modifierMap[token.toLowerCase()];
9597
- if (mappedRole) {
9598
- // Save previous role
9599
- if (currentValue.length > 0) {
9600
- const value = currentValue.join(' ');
9601
- roles.set(currentRole, {
9602
- role: currentRole,
9603
- value,
9604
- isSelector: /^[#.<@]/.test(value),
9605
- });
9606
- }
9607
- currentRole = mappedRole;
9608
- currentValue = [];
9609
- }
9610
- else {
9611
- currentValue.push(token);
9612
- }
9613
- }
9614
- // Save final role
9615
- if (currentValue.length > 0) {
9616
- const value = currentValue.join(' ');
9617
- roles.set(currentRole, {
9618
- role: currentRole,
9619
- value,
9620
- isSelector: /^[#.<@]/.test(value),
9621
- });
9622
- }
9623
- }
9624
- return {
9625
- type: 'event-handler',
9626
- roles,
9627
- original: tokens.join(' '),
9628
- };
9629
- }
9630
- /**
9631
- * Parse a command statement
9632
- * Pattern: {command} {args...}
9633
- */
9634
- function parseCommand(tokens, profile) {
9635
- const roles = new Map();
9636
- if (tokens.length === 0) {
9637
- return { type: 'command', roles, original: '' };
9638
- }
9639
- // First token is the command
9640
- roles.set('action', {
9641
- role: 'action',
9642
- value: tokens[0],
9643
- });
9644
- // Generate dynamic modifier map from language profile (argument position).
9645
- // This enables parsing non-English input (e.g., Japanese \u306b, Korean \uc5d0, Arabic \u0625\u0644\u0649).
9646
- // For a DOM-target command a locative `on` (`toggle .active on me`) maps to
9647
- // `destination` (SVO sources) rather than spawning a bogus `event` role.
9648
- const modifierMap = buildArgumentModifierMap(profile, tokens[0]);
9649
- let currentRole = 'patient';
9650
- let currentValue = [];
9651
- for (let i = 1; i < tokens.length; i++) {
9652
- const token = tokens[i];
9653
- const mappedRole = modifierMap[token.toLowerCase()];
9654
- if (mappedRole) {
9655
- // Save previous role
9656
- if (currentValue.length > 0) {
9657
- const value = currentValue.join(' ');
9658
- roles.set(currentRole, {
9659
- role: currentRole,
9660
- value,
9661
- isSelector: /^[#.<@]/.test(value),
9662
- });
9663
- }
9664
- currentRole = mappedRole;
9665
- currentValue = [];
9666
- }
9667
- else {
9668
- currentValue.push(token);
9669
- }
9670
- }
9671
- // Save final role
9672
- if (currentValue.length > 0) {
9673
- const value = currentValue.join(' ');
9674
- roles.set(currentRole, {
9675
- role: currentRole,
9676
- value,
9677
- isSelector: /^[#.<@]/.test(value),
9678
- });
9679
- }
9680
- return {
9681
- type: 'command',
9682
- roles,
9683
- original: tokens.join(' '),
9684
- };
9685
- }
9686
- /**
9687
- * Re-assign a command's mis-marked primary argument from the default `patient`
9688
- * role to its true primary role (e.g. `wait`'s leading argument is a `duration`,
9689
- * not a fronted object).
9690
- *
9691
- * The generic argument parser in `parseCommand` / `parseEventHandler` defaults the
9692
- * first unmarked argument to `patient`. For most commands that is correct, but for a
9693
- * command whose primary role is a non-patient *and which the target language does not
9694
- * give a marker* (a duration, a measure \u2014 never a BA/object construction), the
9695
- * `patient` assignment makes `insertMarkers` emit a spurious object-marker (Chinese
9696
- * `\u628a`, Japanese `\u3092`, Korean `\ub97c`). The marked form is ungrammatical and the semantic
9697
- * parser fails to match it, dropping the command.
9698
- *
9699
- * The fix is deliberately scoped to **literal/measure** primary roles
9700
- * ({@link LITERAL_PRIMARY_ROLES}: `duration`, `quantity`) \u2014 values that are *never*
9701
- * the object of an object-marking construction in any supported language, so moving
9702
- * them off `patient` can only ever *remove* a spurious marker. Marker-bearing
9703
- * primaries are intentionally left alone:
9704
- * - `set`\u2192destination `\u5230`, `fetch`\u2192source `\u4ece` carry a *correct* marker; touching
9705
- * them risks no benefit.
9706
- * - `send`/`trigger`\u2192event: in a language without an event marker (e.g. Korean has
9707
- * no event particle) un-marking the leading argument makes the semantic parser
9708
- * mis-read it as a bare event handler and emit a phantom `on` action \u2014 an
9709
- * over-generation that recall-based fidelity would not catch. So `event`-primary
9710
- * commands stay on the `patient` default.
9711
- *
9712
- * Roles are still reordered by the safety-net in `reorderRoles`, so no value is lost.
9713
- * A belt-and-suspenders target-marker guard keeps the change inert should a profile
9714
- * ever add a marker for one of these literal roles.
9715
- *
9716
- * Must run *before* `translateElements`, while the `action` value is still the
9717
- * source-language (English) keyword, so the schema lookup resolves.
9718
- *
9719
- * @see docs-internal/ZH_BLOCK_BODY_SCOPE.md (#1 \u2014 transformer role model)
9720
- */
9721
- const LITERAL_PRIMARY_ROLES = new Set([
9722
- 'duration',
9723
- 'quantity',
9724
- ]);
9725
- function applyPrimaryRole(parsed, targetProfile) {
9726
- // Only re-mark standalone command statements. In an event handler (`on click
9727
- // wait 2s \u2026`) a verb-final SOV language without an event particle (e.g. Korean)
9728
- // relies on the leading argument's patient marker as the structural cue that
9729
- // anchors the handler; un-marking it makes the semantic parser lose the event.
9730
- // The block-body / then-chain `wait` clauses this fix targets are each parsed as
9731
- // their own command statement, so they are still covered.
9732
- if (parsed.type !== 'command')
9733
- return;
9734
- const action = parsed.roles.get('action')?.value;
9735
- if (!action)
9736
- return;
9737
- const primaryRole = COMMAND_PRIMARY_ROLES[action.toLowerCase()];
9738
- if (!primaryRole || !LITERAL_PRIMARY_ROLES.has(primaryRole))
9739
- return;
9740
- // Only act on a leading argument the generic parser defaulted to `patient`,
9741
- // and only when the primary slot isn't already filled by an explicit marker.
9742
- const patientEl = parsed.roles.get('patient');
9743
- if (!patientEl || parsed.roles.has(primaryRole))
9744
- return;
9745
- // Guard: never introduce a marker that wasn't there. If the target language
9746
- // marks the primary role, leave the command as-is.
9747
- if (targetProfile.markers.some(m => m.role === primaryRole))
9748
- return;
9749
- parsed.roles.delete('patient');
9750
- parsed.roles.set(primaryRole, { ...patientEl, role: primaryRole });
9751
- }
9752
- /**
9753
- * Parse a conditional statement
9754
- */
9755
- function parseConditional(tokens, _profile) {
9756
- const roles = new Map();
9757
- // First token is the 'if' keyword
9758
- roles.set('action', {
9759
- role: 'action',
9760
- value: tokens[0],
9761
- });
9762
- // Find 'then' to split condition from body - using shared constants
9763
- const thenIndex = tokens.findIndex(t => THEN_KEYWORDS.has(t.toLowerCase()));
9764
- if (thenIndex > 1) {
9765
- const conditionValue = tokens.slice(1, thenIndex).join(' ');
9766
- roles.set('condition', {
9767
- role: 'condition',
9768
- value: conditionValue,
9769
- });
9770
- }
9771
- else if (thenIndex === -1 && tokens.length > 1) {
9772
- // Block-style `if <cond>` with the body on following lines (no inline
9773
- // `then`). Capture everything after `if` as the condition; otherwise the
9774
- // condition is silently dropped and the rendered block becomes a bare
9775
- // `if`/`\u05d0\u05dd`/`\u5982\u679c`, which the semantic block parser then rejects (null
9776
- // parse). This is the dominant failure for nested control-flow bodies in
9777
- // non-Latin languages (he, zh).
9778
- roles.set('condition', {
9779
- role: 'condition',
9780
- value: tokens.slice(1).join(' '),
9781
- });
9782
- }
9783
- return {
9784
- type: 'conditional',
9785
- roles,
9786
- original: tokens.join(' '),
9787
- };
9788
- }
9789
- // =============================================================================
9790
- // Translation
9791
- // =============================================================================
9792
- /**
9793
- * Translate words using dictionary with type-safe access.
9794
- */
9795
- function translateWord(word, sourceLocale, targetLocale) {
9796
- // Don't translate CSS selectors
9797
- if (/^[#.<@]/.test(word)) {
9798
- return word;
9799
- }
9800
- // Don't translate numbers
9801
- if (/^\d+/.test(word)) {
9802
- return word;
9803
- }
9804
- // A whole parenthesized group fused by the tokenizer (`(the value of #price
9805
- // as Number)`) reaching a single-token translate path: translate its
9806
- // interior word-by-word IN ORDER \u2014 never reordered, never re-segmented.
9807
- if (/\s/.test(word) && word.startsWith('(')) {
9808
- return word
9809
- .split(/\s+/)
9810
- .map(w => translateWord(w, sourceLocale, targetLocale))
9811
- .join(' ');
9812
- }
9813
- // A word carrying paren punctuation from a fused group after whitespace
9814
- // re-splitting (`(my` / `valor)` / `(($x`): strip the parens for the
9815
- // dictionary lookup and re-attach, so interior keywords still translate.
9816
- if (word.length > 1 && (word.startsWith('(') || word.endsWith(')'))) {
9817
- const m = word.match(/^(\(*)([^()]+)(\)*)$/);
9818
- if (m && (m[1] || m[3])) {
9819
- return m[1] + translateWord(m[2], sourceLocale, targetLocale) + m[3];
9820
- }
9821
- }
9822
- const sourceDict = sourceLocale === 'en' ? null : dictionaries[sourceLocale];
9823
- const targetDict = dictionaries[targetLocale];
9824
- if (!targetDict)
9825
- return word;
9826
- // If source is not English, first map to English using type-safe lookup
9827
- let englishWord = word;
9828
- if (sourceDict) {
9829
- const found = findInDictionary(sourceDict, word);
9830
- if (found) {
9831
- englishWord = found.englishKey;
9832
- }
9833
- }
9834
- // Now map English to target locale using type-safe lookup
9835
- const translated = translateFromEnglish(targetDict, englishWord);
9836
- return translated ?? word;
9837
- }
9838
- /**
9839
- * Possessive markers for each language.
9840
- * Used to transform "X's Y" patterns to target language structure.
9841
- */
9842
- const POSSESSIVE_MARKERS = {
9843
- en: { type: 'suffix', marker: "'s" },
9844
- es: { type: 'preposition', marker: 'de' },
9845
- pt: { type: 'preposition', marker: 'de' },
9846
- fr: { type: 'preposition', marker: 'de' },
9847
- de: { type: 'preposition', marker: 'von' },
9848
- ja: { type: 'suffix', marker: '\u306e' },
9849
- ko: { type: 'suffix', marker: '\uc758' },
9850
- zh: { type: 'suffix', marker: '\u7684' },
9851
- ar: { type: 'preposition', marker: '\u0644\u0640' },
9852
- // Spaced genitive particle (not the glued `'\u0131n`), so the tokenizer can split
9853
- // it off the selector \u2014 consistent with Turkish's other spaced case markers.
9854
- tr: { type: 'particle', marker: '\u0131n' },
9855
- id: { type: 'preposition', marker: 'dari' },
9856
- // Latin-script genitive: must be a *spaced* particle (`#picker pa`), since a
9857
- // glued `#pickerpa` can't be split from the selector by the tokenizer the
9858
- // way a non-Latin suffix (\u306e/\uc758/\u09b0) can.
9859
- qu: { type: 'particle', marker: 'pa' },
9860
- // Bengali SOV postposition genitive, like ja/ko \u2014 a spaced suffix the
9861
- // tokenizer splits off as a particle. Previously absent, so it fell back to
9862
- // the English `'s` marker and its possessive property paths never parsed.
9863
- // (Hindi `\u0915\u093e` is intentionally omitted: its `bind` lacks a verb-final
9864
- // grammar rule, so fixing its possessive alone yields a wrong `on` parse \u2014
9865
- // tracked as separate follow-up.)
9866
- bn: { type: 'suffix', marker: '\u09b0' },
9867
- sw: { type: 'preposition', marker: 'ya' },
9868
- };
9869
- /**
9870
- * Transform possessive 's syntax to target language.
9871
- *
9872
- * Examples:
9873
- * me's value \u2192 mi valor (Spanish - pronoun becomes possessive adjective)
9874
- * #button's textContent \u2192 textContent de #button (Spanish - prepositional)
9875
- * me's value \u2192 \u79c1\u306e\u5024 (Japanese - \u306e particle)
9876
- */
9877
- function translatePossessive(token, sourceLocale, targetLocale) {
9878
- // Check for 's possessive pattern
9879
- const possessiveMatch = token.match(/^(.+)'s$/i);
9880
- if (!possessiveMatch) {
9881
- return token;
9882
- }
9883
- const owner = possessiveMatch[1];
9884
- const targetMarker = POSSESSIVE_MARKERS[targetLocale] || POSSESSIVE_MARKERS.en;
9885
- // Check if owner is a pronoun that has a possessive form
9886
- const pronounPossessives = {
9887
- me: 'my',
9888
- it: 'its',
9889
- you: 'your',
9890
- };
9891
- const lowerOwner = owner.toLowerCase();
9892
- if (pronounPossessives[lowerOwner]) {
9893
- // Convert "me's" to "my" then translate
9894
- const possessiveForm = pronounPossessives[lowerOwner];
9895
- return translateWord(possessiveForm, 'en', targetLocale);
9896
- }
9897
- // For selectors and other owners, translate owner and apply target possessive marker
9898
- const translatedOwner = translateWord(owner, sourceLocale, targetLocale);
9899
- switch (targetMarker.type) {
9900
- case 'suffix':
9901
- // Japanese/Korean/Chinese: owner + marker (e.g., #button\u306e, #button\uc758)
9902
- return `${translatedOwner}${targetMarker.marker}`;
9903
- case 'particle':
9904
- // Latin-script spaced genitive (Quechua `pa`): owner + space + marker
9905
- // so the tokenizer can separate it from the selector.
9906
- return `${translatedOwner} ${targetMarker.marker}`;
9907
- case 'preposition':
9908
- // Will be handled by caller - return marker + owner format
9909
- // Store as special format to be processed later
9910
- return `__POSS__${targetMarker.marker}__${translatedOwner}__POSS__`;
9911
- default:
9912
- return `${translatedOwner}'s`;
9913
- }
9914
- }
9915
- // =============================================================================
9916
- // Possessive Dot Notation
9917
- // =============================================================================
9918
- /**
9919
- * Regex to match possessive dot notation patterns.
9920
- * Matches: my.prop, its.prop, your.prop, me.prop, it.prop, you.prop
9921
- * Also matches optional chaining: my?.prop, me?.prop, etc.
9922
- */
9923
- const POSSESSIVE_DOT_REGEX = /^(my|its|your|me|it|you)(\??\..+)$/i;
9924
- /**
9925
- * Map pronoun forms to possessive adjective forms for dictionary lookup.
9926
- */
9927
- const POSSESSIVE_DOT_PRONOUNS = {
9928
- me: 'my',
9929
- it: 'its',
9930
- you: 'your',
9931
- my: 'my',
9932
- its: 'its',
9933
- your: 'your',
9934
- };
9935
- /**
9936
- * Translate possessive dot notation like my.textContent \u2192 mi.textContent.
9937
- * Handles both possessive adjective forms (my, its, your) and pronoun forms (me, it, you).
9938
- * Also handles optional chaining (my?.prop).
9939
- * Returns null if the value doesn't match or no translation is available.
9940
- */
9941
- function translatePossessiveDotNotation(value, sourceLocale, targetLocale) {
9942
- const match = value.match(POSSESSIVE_DOT_REGEX);
9943
- if (!match)
9944
- return null;
9945
- const possessiveWord = match[1].toLowerCase();
9946
- const propertySuffix = match[2]; // ".textContent" or "?.textContent"
9947
- // Normalize to possessive adjective form for dictionary lookup
9948
- const possessiveKey = POSSESSIVE_DOT_PRONOUNS[possessiveWord] || possessiveWord;
9949
- const translated = translateWord(possessiveKey, sourceLocale, targetLocale);
9950
- // Skip if translation is multi-word (can't prefix dot notation)
9951
- if (translated.includes(' '))
9952
- return null;
9953
- if (translated !== possessiveKey) {
9954
- return translated + propertySuffix;
9955
- }
9956
- // Try original pronoun form if different from possessive key
9957
- if (possessiveWord !== possessiveKey) {
9958
- const alt = translateWord(possessiveWord, sourceLocale, targetLocale);
9959
- if (alt !== possessiveWord && !alt.includes(' ')) {
9960
- return alt + propertySuffix;
9961
- }
9962
- }
9963
- return null;
9964
- }
9965
- // =============================================================================
9966
- // Multi-Word Value Translation
9967
- // =============================================================================
9968
- /**
9969
- * Translate a multi-word value, translating each word individually.
9970
- * Handles possessives like "my value" \u2192 "mi valor" in Spanish.
9971
- * Also handles 's possessive syntax like "me's value" \u2192 "mi valor".
9972
- * Also handles possessive dot notation like "my.textContent" \u2192 "mi.textContent".
9973
- */
9974
- function translateMultiWordValue(value, sourceLocale, targetLocale) {
9975
- // Mask event-guard / attribute brackets (`[key is 'Escape']`): their contents
9976
- // are expression syntax, not translatable keywords \u2014 translating `is` -> `ni`
9977
- // etc. inside them breaks the guard. Restore verbatim after translation.
9978
- if (value.includes('[')) {
9979
- const guards = [];
9980
- const masked = value.replace(/\[[^\]]*\]/g, match => {
9981
- guards.push(match);
9982
- return `\ue000${guards.length - 1}\ue001`;
9983
- });
9984
- if (guards.length > 0) {
9985
- const translated = translateMultiWordValue(masked, sourceLocale, targetLocale);
9986
- return translated.replace(/\ue000(\d+)\ue001/g, (_, n) => guards[Number(n)]);
9987
- }
9988
- }
9989
- // If it's a single word, check for possessive then translate
9990
- if (!value.includes(' ')) {
9991
- // Check for possessive 's
9992
- if (value.includes("'s")) {
9993
- return translatePossessive(value, sourceLocale, targetLocale);
9994
- }
9995
- // Check for possessive dot notation (my.prop, its.prop, me.prop, etc.)
9996
- const dotResult = translatePossessiveDotNotation(value, sourceLocale, targetLocale);
9997
- if (dotResult !== null)
9998
- return dotResult;
9999
- return translateWord(value, sourceLocale, targetLocale);
10000
- }
10001
- // Split into words and translate each
10002
- const words = value.split(/\s+/);
10003
- const translated = [];
10004
- let i = 0;
10005
- while (i < words.length) {
10006
- const word = words[i];
10007
- // Check for possessive 's pattern FIRST (e.g., "me's value", "#button's textContent")
10008
- // This must come before selector check because "#button's" starts with #
10009
- if (word.includes("'s")) {
10010
- const possessiveResult = translatePossessive(word, sourceLocale, targetLocale);
10011
- // Check if it's a prepositional possessive that needs reordering
10012
- const prepMatch = possessiveResult.match(/^__POSS__(.+)__(.+)__POSS__$/);
10013
- if (prepMatch && i + 1 < words.length) {
10014
- // Prepositional: "X's Y" \u2192 "Y marker X" (e.g., "textContent de #button")
10015
- const marker = prepMatch[1];
10016
- const owner = prepMatch[2];
10017
- const property = words[i + 1];
10018
- const translatedProperty = translateWord(property, sourceLocale, targetLocale);
10019
- translated.push(`${translatedProperty} ${marker} ${owner}`);
10020
- i += 2; // Skip property since we consumed it
10021
- continue;
10022
- }
10023
- else if (prepMatch) {
10024
- // No property following - just output owner with marker prefix
10025
- const marker = prepMatch[1];
10026
- const owner = prepMatch[2];
10027
- translated.push(`${marker} ${owner}`);
10028
- i++;
10029
- continue;
10030
- }
10031
- // Suffix-style possessive (Japanese, Korean, etc.) or pronoun
10032
- translated.push(possessiveResult);
10033
- i++;
10034
- continue;
10035
- }
10036
- // Skip pure CSS selectors and numbers (but NOT possessives which were handled above)
10037
- if (/^[#.<@]/.test(word) || /^\d+/.test(word)) {
10038
- translated.push(word);
10039
- i++;
10040
- continue;
10041
- }
10042
- // Skip quoted strings
10043
- if (/^["'].*["']$/.test(word)) {
10044
- translated.push(word);
10045
- i++;
10046
- continue;
10047
- }
10048
- // Check for possessive dot notation (my.prop, its.prop, me.prop, etc.)
10049
- const dotResult = translatePossessiveDotNotation(word, sourceLocale, targetLocale);
10050
- if (dotResult !== null) {
10051
- translated.push(dotResult);
10052
- i++;
10053
- continue;
10054
- }
10055
- translated.push(translateWord(word, sourceLocale, targetLocale));
10056
- i++;
10057
- }
10058
- return translated.join(' ');
10059
- }
10060
- /**
10061
- * Translate all elements in a parsed statement
10062
- */
10063
- function translateElements(parsed, sourceLocale, targetLocale) {
10064
- for (const [_role, element] of parsed.roles) {
10065
- // Always process possessive 's syntax, even for selectors
10066
- // E.g., "#button's textContent" should translate the possessive
10067
- if (element.value.includes("'s")) {
10068
- element.translated = translateMultiWordValue(element.value, sourceLocale, targetLocale);
10069
- }
10070
- else if (!element.isSelector && !element.isLiteral) {
10071
- element.translated = translateMultiWordValue(element.value, sourceLocale, targetLocale);
10072
- }
10073
- else {
10074
- element.translated = element.value;
10075
- }
10076
- }
10077
- }
10078
- // =============================================================================
10079
- // Caret-scope masking (`^name on <selector>`)
10080
- // =============================================================================
10081
- /** Private-use sentinels bracketing a masked caret-scope index. */
10082
- const CARET_SCOPE_OPEN = '\uE000';
10083
- const CARET_SCOPE_CLOSE = '\uE001';
10084
- /**
10085
- * Match `^name on <selector>` and the scope's selector form (#id, .class,
10086
- * <tag/>, [attr]). The `^name` is kept; only the ` on <selector>` scope is masked.
10087
- */
10088
- const CARET_SCOPE_RE = /(\^[A-Za-z_][\w-]*)(\s+on\s+(?:[#.][\w-]+|<[^>]*\/>|\[[^\]]+\]))/g;
10089
- /**
10090
- * Mask the ` on <selector>` scope of every `^name on <selector>` read behind an
10091
- * opaque token attached to `^name`, so the overloaded `on` doesn't reach the
10092
- * splitter / event-handler parser. Returns null when there's nothing to mask.
10093
- */
10094
- function maskCaretScopes(input) {
10095
- const scopes = [];
10096
- const masked = input.replace(CARET_SCOPE_RE, (_m, varTok, scope) => {
10097
- const idx = scopes.length;
10098
- scopes.push(scope);
10099
- return `${varTok}${CARET_SCOPE_OPEN}${idx}${CARET_SCOPE_CLOSE}`;
10100
- });
10101
- return scopes.length > 0 ? { masked, scopes } : null;
10102
- }
10103
- /** Restore masked caret-scope tokens to their verbatim ` on <selector>` form. */
10104
- function restoreCaretScopes(input, scopes) {
10105
- return input.replace(new RegExp(`${CARET_SCOPE_OPEN}(\\d+)${CARET_SCOPE_CLOSE}`, 'g'), (_m, n) => scopes[Number(n)] ?? '');
10106
- }
10107
- // =============================================================================
10108
- // View-transition tail masking (`using view transition`)
10109
- // =============================================================================
10110
- /** Private-use sentinels bracketing a masked view-transition-tail index. */
10111
- const VIEW_TAIL_OPEN = '\uE002';
10112
- const VIEW_TAIL_CLOSE = '\uE003';
10113
- /**
10114
- * Match `swap`/`process`'s trailing `using view transition` modifier.
10115
- *
10116
- * `using` is in no dictionary and in no role table, so it sweeps into whatever
10117
- * role phrase is open; `transition` IS a translated command keyword in every
10118
- * dictionary (es `transici\u00f3n`, de `\u00fcbergang`, ja `\u9077\u79fb`) and is in
10119
- * `ENGLISH_COMMANDS`, so `splitOnCommandBoundaries` splits the clause there and
10120
- * the rejoin plants a phantom translated `transition` COMMAND after the target's
10121
- * `then`-connective (`intercambiar #a con #b using view entonces transici\u00f3n`).
10122
- *
10123
- * The phrase has no native form in any of the 24 languages \u2014 semantic's
10124
- * `USING_VIEW_MARKER_ALL_LANGS` matches the literal English `using view` marker
10125
- * everywhere \u2014 so it is a passthrough, matched on the English surface regardless
10126
- * of source locale. The value word is optional so a bare `using view` still
10127
- * masks rather than half-splitting; `then` is excluded so a clause boundary is
10128
- * never swallowed into the tail.
10129
- */
10130
- const VIEW_TAIL_RE = /\busing\s+view\b(?:\s+(?!then\b)[A-Za-z][\w-]*)?/gi;
10131
- /** Whether a token is a masked view-transition tail. */
10132
- const VIEW_TAIL_TOKEN_RE = new RegExp(`^${VIEW_TAIL_OPEN}(\\d+)${VIEW_TAIL_CLOSE}$`);
10133
- /**
10134
- * Mask every `using view <value>` tail behind an opaque token, so the phrase
10135
- * never reaches the splitter, the word translator, or the role parser. Returns
10136
- * null when there's nothing to mask.
10137
- */
10138
- function maskViewTransitionTails(input) {
10139
- const tails = [];
10140
- const masked = input.replace(VIEW_TAIL_RE, match => {
10141
- const idx = tails.length;
10142
- tails.push(match);
10143
- return `${VIEW_TAIL_OPEN}${idx}${VIEW_TAIL_CLOSE}`;
10144
- });
10145
- return tails.length > 0 ? { masked, tails } : null;
10146
- }
10147
- /** Restore masked view-transition tokens to their verbatim English phrase. */
10148
- function restoreViewTransitionTails(input, tails) {
10149
- return input.replace(new RegExp(`${VIEW_TAIL_OPEN}(\\d+)${VIEW_TAIL_CLOSE}`, 'g'), (_m, n) => tails[Number(n)] ?? '');
10150
- }
10151
- // =============================================================================
10152
- // Main Transformer
10153
- // =============================================================================
10154
- class GrammarTransformer {
10155
- constructor(sourceLocale = 'en', targetLocale) {
10156
- const source = getProfile(sourceLocale);
10157
- const target = getProfile(targetLocale);
10158
- if (!source)
10159
- throw new Error(`Unknown source locale: ${sourceLocale}`);
10160
- if (!target)
10161
- throw new Error(`Unknown target locale: ${targetLocale}`);
10162
- this.sourceProfile = source;
10163
- this.targetProfile = target;
10164
- }
10165
- /**
10166
- * Transform a hyperscript statement from source to target language.
10167
- * Handles compound statements with "then" by splitting, transforming each part,
10168
- * and rejoining with the target language's "then" keyword.
10169
- *
10170
- * For multi-line input, preserves line structure (indentation, blank lines).
10171
- */
10172
- transform(input) {
10173
- const out = this.transformInternal(input);
10174
- // Hebrew: repair a fronted accusative marker the body-split heuristics can emit
10175
- // (`\u2026 \u05d0\u05ea \u05d4\u05d5\u05e1\u05e3 .x \u2026` \u2192 `\u2026 \u05d4\u05d5\u05e1\u05e3 \u05d0\u05ea .x \u2026`). Applied to the assembled output; idempotent
10176
- // across the internal recursion. See repairHebrewFrontedAccusative.
10177
- return this.targetProfile.code === 'he' ? repairHebrewFrontedAccusative(out) : out;
10178
- }
10179
- transformInternal(input) {
10180
- // `using view transition` is a passthrough phrase, not translatable content:
10181
- // `using` is in no dictionary and `transition` is a COMMAND keyword, so left
10182
- // in place the splitter tears the clause apart there and the tail renders as
10183
- // a phantom translated command. Mask it before any splitting/translation and
10184
- // restore it verbatim; transformSingle re-appends the opaque token at the
10185
- // clause tail so it lands after the reorder, not inside a role phrase.
10186
- const viewTails = maskViewTransitionTails(input);
10187
- if (viewTails) {
10188
- return restoreViewTransitionTails(this.transformInternal(viewTails.masked), viewTails.tails);
10189
- }
10190
- // Caret-scoped variable reads (`^name on <selector>`) carry a second,
10191
- // overloaded `on` that the splitter/event-parser would mistake for an event
10192
- // or command boundary \u2014 mangling `put ^count on #host into me`. Mask the
10193
- // ` on <selector>` scope behind an opaque token attached to `^name` so the
10194
- // command reorders as if the patient were a single value, then restore it.
10195
- // `on` is kept verbatim (the semantic caret-scope matcher accepts it by raw
10196
- // value across languages \u2014 passthrough-alignment).
10197
- const caret = maskCaretScopes(input);
10198
- if (caret) {
10199
- return restoreCaretScopes(this.transform(caret.masked), caret.scopes);
10200
- }
10201
- const targetThen = getTargetThenKeyword(this.targetProfile.code);
10202
- // Inline JS blocks (`... js <raw js> end`) must be masked BEFORE any
10203
- // splitting/reordering: the body is raw JavaScript, not hyperscript, so it
10204
- // must never be tokenized, translated, or word-order reordered. (Single-line
10205
- // only here; multi-line js bodies are handled with the behavior work.)
10206
- if (!input.includes('\n')) {
10207
- const jsBlock = this.tryTransformJsBlock(input);
10208
- if (jsBlock !== null)
10209
- return jsBlock;
10210
- // Event handler whose body leads with a command-modifier
10211
- // (`on click async fetch \u2026`, `on click once add \u2026`): lift the modifier out
10212
- // so the real verb isn't mistaken for the action and the SOV reorder keeps
10213
- // the body patient-first (recoverable by the parser's SOV event extraction).
10214
- const eventModifier = this.tryTransformEventWithModifierBody(input);
10215
- if (eventModifier !== null)
10216
- return eventModifier;
10217
- // Event handler whose body is a block (`on <event> if/repeat/unless \u2026 end`):
10218
- // the block body must be reordered as a self-contained unit, not shredded
10219
- // across the event handler's role soup.
10220
- const eventBlock = this.tryTransformEventWithBlockBody(input);
10221
- if (eventBlock !== null)
10222
- return eventBlock;
10223
- // Event handler whose body is an un-terminated inline `unless` guard
10224
- // (`on click unless I match .disabled toggle .selected`): route the guard
10225
- // through the standalone block path so Hebrew's accusative marker lands on
10226
- // the body command, not the condition. Hebrew-only; null elsewhere.
10227
- const eventGuard = this.tryTransformEventWithUnlessGuard(input);
10228
- if (eventGuard !== null)
10229
- return eventGuard;
10230
- }
10231
- // Check if input has multi-line structure worth preserving
10232
- const hasMultiLineStructure = input.includes('\n');
10233
- if (hasMultiLineStructure) {
10234
- // Multi-line case - preserve structure (indentation, blank lines)
10235
- const { parts, lineMetadata, partToLineIndex } = splitCompoundStatementWithMetadata(input, this.sourceProfile.code);
10236
- const transformedParts = parts.map(part => this.transformSingle(part));
10237
- return reconstructWithLineStructure(transformedParts, lineMetadata, partToLineIndex, targetThen);
10238
- }
10239
- // Single-line case - use existing logic
10240
- const parts = splitCompoundStatement(input, this.sourceProfile.code);
10241
- if (parts.length > 1) {
10242
- const transformedParts = parts.map(part => this.transformSingle(part));
10243
- return transformedParts.join(` ${targetThen} `);
10244
- }
10245
- // Single statement (no "then" splitting needed)
10246
- return this.transformSingle(input);
10247
- }
10248
- /**
10249
- * Transform a single hyperscript statement (no compound "then" chains).
10250
- */
10251
- transformSingle(input) {
10252
- // 0. Reactive block? Route around parseStatement entirely so
10253
- // block-syntactic tokens (live/when/unless/end) aren't treated
10254
- // as command verbs or swept into role values, and so SOV/VSO
10255
- // reorder applies only inside the body.
10256
- const block = extractBlockStructure(input, this.sourceProfile.code);
10257
- if (block) {
10258
- return this.transformBlock(block);
10259
- }
10260
- // 0a. Fragment carrying a stranded trailing block terminator (`wait 200ms
10261
- // end`, `set x to y end` \u2014 what the `then`-splitter leaves from
10262
- // `if \u2026 then <cmd> end`). `end` is not a marker, so left in place it is
10263
- // swept into the open role's VALUE and rendered inside that phrase
10264
- // (bn `200ms \u09b6\u09c7\u09b7 \u0995\u09c7 \u0985\u09aa\u09c7\u0995\u09cd\u09b7\u09be`). Strip it, transform the clause alone, and
10265
- // re-append the translated terminator after the verb \u2014 the same tail
10266
- // position transformBlockBody emits for event-headed blocks.
10267
- const strippedEnd = this.transformWithTrailingEnd(input);
10268
- if (strippedEnd !== null) {
10269
- return strippedEnd;
10270
- }
10271
- // 0b. `set @attr to V on <scope>` (S1 tabs-aria): strip the trailing
10272
- // `on <scope>`, transform the scope-less set normally, then re-insert
10273
- // `on <scope>` where the semantic set patterns expect it.
10274
- const setScope = this.transformSetWithScope(input);
10275
- if (setScope !== null) {
10276
- return setScope;
10277
- }
10278
- // 0c. Masked `using view transition` tail (see maskViewTransitionTails):
10279
- // strip the opaque token, transform the clause without it, and re-append
10280
- // it at the clause tail. Left in the token stream it would be swept into
10281
- // the open role's VALUE and rendered ahead of that role's marker
10282
- // (ja `#b using view transition \u3067` instead of `#b \u3067 using view
10283
- // transition`) \u2014 the same failure mode transformWithTrailingEnd fixes
10284
- // for a stranded `end`.
10285
- const viewTail = this.transformWithViewTransitionTail(input);
10286
- if (viewTail !== null) {
10287
- return viewTail;
10288
- }
10289
- // 1. Parse into semantic roles
10290
- const parsed = parseStatement(input, this.sourceProfile.code);
10291
- if (!parsed) {
10292
- return input; // Return unchanged if parsing fails
10293
- }
10294
- // 1b. Re-assign a mis-marked primary argument (e.g. `wait`'s duration) off the
10295
- // default `patient` role so the target doesn't emit a spurious object-marker.
10296
- // Runs before translation while `action` is still the English keyword.
10297
- applyPrimaryRole(parsed, this.targetProfile);
10298
- // 2. Translate individual words
10299
- translateElements(parsed, this.sourceProfile.code, this.targetProfile.code);
10300
- // 3. Find applicable rule
10301
- const rule = this.findRule(parsed);
10302
- // 4. Apply transformation
10303
- if (rule?.transform.custom) {
10304
- return rule.transform.custom(parsed, this.targetProfile);
10305
- }
10306
- // 5. Reorder according to target language's canonical order
10307
- const roleOrder = rule?.transform.roleOrder || this.targetProfile.canonicalOrder;
10308
- const reordered = reorderRoles(parsed.roles, roleOrder);
10309
- // 6. Insert grammatical markers
10310
- const shouldInsertMarkers = rule?.transform.insertMarkers ?? true;
10311
- if (shouldInsertMarkers) {
10312
- const result = insertMarkers(reordered, this.targetProfile.markers, this.targetProfile.adpositionType);
10313
- // Use joinTokens for proper suffix/prefix attachment (Turkish -i, Quechua -ta, etc.)
10314
- return joinTokens(result);
10315
- }
10316
- // 7. Join without markers (still use joinTokens for consistency)
10317
- return joinTokens(reordered.map(e => e.translated || e.value));
10318
- }
10319
- /**
10320
- * Clause carrying a masked `using view transition` tail: strip the opaque
10321
- * token, transform the clause alone, and re-append the token at the very end.
10322
- *
10323
- * The tail is a clause-final modifier in every word order the corpus emits:
10324
- * the semantic side matches it as the literal `using view` marker plus a value
10325
- * word, and the SOV/VSO event-handler patterns admit it as an optional
10326
- * TRAILING group (after the with-marked operand). So the target position is
10327
- * "end of the transformed clause" for all 24 languages \u2014 no per-profile
10328
- * placement decision, which is what makes this a passthrough rather than a
10329
- * role.
10330
- *
10331
- * Returns null when the clause carries no masked tail, or when the token is
10332
- * not clause-final (nothing to reposition \u2014 leaving it in place still restores
10333
- * verbatim English).
10334
- */
10335
- transformWithViewTransitionTail(input) {
10336
- const trimmed = input.trim();
10337
- const tokens = trimmed.split(/\s+/);
10338
- if (tokens.length < 2) {
10339
- return null;
10340
- }
10341
- if (!VIEW_TAIL_TOKEN_RE.test(tokens[tokens.length - 1])) {
10342
- return null;
10343
- }
10344
- const tail = tokens[tokens.length - 1];
10345
- const head = tokens.slice(0, -1).join(' ');
10346
- return `${this.transformSingle(head)} ${tail}`;
10347
- }
10348
- /**
10349
- * `<command \u2026> end` fragments: transform the command without its stranded
10350
- * terminator, then re-append the translated terminator as a standalone
10351
- * trailing token. Fragments that open a block of their own (`if \u2026 end`,
10352
- * `repeat \u2026 end`, `js \u2026 end`) bail \u2014 their terminator belongs to them and
10353
- * their dedicated paths handle it.
10354
- */
10355
- transformWithTrailingEnd(input) {
10356
- const src = this.sourceProfile.code;
10357
- const tokens = input.trim().split(/\s+/);
10358
- if (tokens.length < 2) {
10359
- return null;
10360
- }
10361
- const sourceEnd = translateWord('end', 'en', src).toLowerCase();
10362
- if (tokens[tokens.length - 1].toLowerCase() !== sourceEnd) {
10363
- return null;
10364
- }
10365
- const openers = new Set(['if', 'repeat', 'unless', 'while', 'when', 'live', 'js'].map(k => translateWord(k, 'en', src).toLowerCase()));
10366
- if (tokens.slice(0, -1).some(t => openers.has(t.toLowerCase()))) {
10367
- return null;
10368
- }
10369
- const inner = this.transformSingle(tokens.slice(0, -1).join(' '));
10370
- const endT = translateWord(tokens[tokens.length - 1], src, this.targetProfile.code);
10371
- return `${inner} ${endT}`;
10372
- }
10373
- /**
10374
- * Detect and transform an inline JS block (`[on <event>] js <raw js> end`).
10375
- *
10376
- * The `js ... end` body is raw JavaScript: it must not be tokenized,
10377
- * translated, or word-order reordered. We mask the whole block with a single
10378
- * opaque placeholder, run the surrounding statement (the event-handler head,
10379
- * if any) through the normal reorder pipeline so the placeholder lands in the
10380
- * correct action position, then substitute the translated `js`/`end` keywords
10381
- * around the verbatim body.
10382
- *
10383
- * Returns `null` (fall through to the normal path) when there is no js block,
10384
- * no matching `end`, or trailing content after `end` (kept tight on purpose).
10385
- */
10386
- tryTransformJsBlock(input) {
10387
- const src = this.sourceProfile.code;
10388
- const dst = this.targetProfile.code;
10389
- // The `js` / `end` keyword forms in the SOURCE language.
10390
- const sourceJs = translateWord('js', 'en', src);
10391
- const sourceEnd = translateWord('end', 'en', src).toLowerCase();
10392
- const tokens = input.split(/\s+/).filter(t => t.length > 0);
10393
- // The js command token, optionally with a `(locals)` suffix: `js`, `js(me)`.
10394
- const escapedJs = sourceJs.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
10395
- const jsRe = new RegExp(`^${escapedJs}(\\(.*\\))?$`, 'i');
10396
- const jsIdx = tokens.findIndex(t => jsRe.test(t));
10397
- if (jsIdx === -1)
10398
- return null;
10399
- // First `end` after the js keyword closes the block (raw JS never contains a
10400
- // bare hyperscript `end` token).
10401
- let endIdx = -1;
10402
- for (let i = jsIdx + 1; i < tokens.length; i++) {
10403
- if (tokens[i].toLowerCase() === sourceEnd) {
10404
- endIdx = i;
10405
- break;
10406
- }
10407
- }
10408
- if (endIdx === -1)
10409
- return null;
10410
- // Trailing content after `end` \u2014 leave for the normal path.
10411
- if (endIdx !== tokens.length - 1)
10412
- return null;
10413
- const jsToken = tokens[jsIdx];
10414
- const jsParen = jsToken.match(jsRe)?.[1] ?? '';
10415
- const jsKeywordRaw = jsParen ? jsToken.slice(0, jsToken.length - jsParen.length) : jsToken;
10416
- const body = tokens.slice(jsIdx + 1, endIdx).join(' ');
10417
- const targetJs = translateWord(jsKeywordRaw, src, dst) + jsParen;
10418
- const targetEnd = translateWord(tokens[endIdx], src, dst);
10419
- const replacement = [targetJs, body, targetEnd].filter(s => s.length > 0).join(' ');
10420
- const before = tokens.slice(0, jsIdx);
10421
- // Bare `js ... end` with no leading event-handler head: emit directly.
10422
- if (before.length === 0)
10423
- return replacement;
10424
- // Mask the block as one opaque action token, reorder the surrounding
10425
- // statement, then restore the verbatim block.
10426
- const placeholder = 'JSBLOCKPLACEHOLDER';
10427
- const reordered = this.transformSingle([...before, placeholder].join(' '));
10428
- if (!reordered.includes(placeholder))
10429
- return null; // unexpected \u2014 fall through
10430
- return reordered.replace(placeholder, replacement);
10431
- }
10432
- /**
10433
- * Transform an event handler whose body is a block command
10434
- * (`on <event> [from <src>] {if|repeat|unless|while|for} \u2026 end`).
10435
- *
10436
- * `parseEventHandler` would treat the block keyword as the action and sweep the
10437
- * condition/body into role values, then reorder them \u2014 shredding the block
10438
- * (`if event.shiftKey call submitAndContinue() end` \u2192 scattered tokens). Instead
10439
- * we mask the whole block as an opaque action placeholder, reorder the event
10440
- * head normally, transform the block as a self-contained unit, and restitch.
10441
- *
10442
- * Returns `null` (fall through) when the input isn't an event handler, has no
10443
- * block-keyword body, or has no closing `end`.
10444
- */
10445
- tryTransformEventWithBlockBody(input) {
10446
- const tokens = tokenize(input, this.sourceProfile);
10447
- if (tokens.length === 0)
10448
- return null;
10449
- if (!EVENT_KEYWORDS.has(tokens[0]?.toLowerCase()))
10450
- return null;
10451
- let blockIdx = -1;
10452
- for (let i = 1; i < tokens.length; i++) {
10453
- if (BLOCK_BODY_KEYWORDS.has(tokens[i].toLowerCase())) {
10454
- blockIdx = i;
10455
- break;
10456
- }
10457
- }
10458
- if (blockIdx <= 0)
10459
- return null;
10460
- // Only handle the explicitly-terminated form; un-terminated bodies
10461
- // (`on click unless X toggle Y`) keep the existing path.
10462
- if (tokens[tokens.length - 1].toLowerCase() !== 'end')
10463
- return null;
10464
- const eventHead = tokens.slice(0, blockIdx);
10465
- const blockTokens = tokens.slice(blockIdx);
10466
- // Event heads carrying a `from <source>` modifier (`on keydown[...] from
10467
- // .modal if \u2026 end`): for SVO/SOV targets, route them through this block-body
10468
- // path too \u2014 masking the block and emitting the event clause (incl. the
10469
- // translated `from <source>`, which `transformSingle` \u2192 `parseEventHandler`
10470
- // already reorders) first, then the transformed block. This keeps the if-block
10471
- // body from being shredded across the event handler's roles and properly
10472
- // translates its inner keywords (`in`/`focus`/`first`/positional); the old
10473
- // exclusion left them English and mangled the order (this is what cleared
10474
- // `focus-trap` in tr). VSO targets (ar, tl) are kept on the existing path:
10475
- // there the event-first emission with a `from`-source reorders incorrectly and
10476
- // regresses `focus-trap`/`window-keydown`, while the existing path already
10477
- // parses them. So the `from`-source exclusion is scoped to VSO only.
10478
- if (this.targetProfile.wordOrder === 'VSO' && eventHead.some(t => t.toLowerCase() === 'from')) {
10479
- return null;
10480
- }
10481
- const placeholder = 'EVENTBLOCKPLACEHOLDER';
10482
- const headOut = this.transformSingle([...eventHead, placeholder].join(' '));
10483
- if (!headOut.includes(placeholder))
10484
- return null;
10485
- const blockOut = this.transformBlockBody(blockTokens);
10486
- // Always emit the event clause first, then the block. An event handler's
10487
- // event is a leading delimiter, and the semantic parser only matches a
10488
- // block body when it follows the event \u2014 even in verb-first (VSO) languages
10489
- // whose normal command order would push the event to the end. So strip the
10490
- // placeholder out of the (possibly reordered) head and append the block,
10491
- // rather than substituting in place.
10492
- const eventClause = headOut.replace(placeholder, '').replace(/\s+/g, ' ').trim();
10493
- return [eventClause, blockOut].filter(s => s.length > 0).join(' ');
10494
- }
10495
- /**
10496
- * Transform an event handler whose body is an inline `unless` guard with NO
10497
- * `end` (`on <event> unless <cond> <body>` \u2014 the `unless-condition` shape).
10498
- *
10499
- * Object-marking SVO targets (he, zh). `parseEventHandler` reads `unless` as the
10500
- * action and sweeps the whole `<cond> <body>` tail into a single `patient` blob;
10501
- * the target then prefixes that blob with its object marker \u2014 Hebrew's accusative
10502
- * \u05d0\u05ea (`\u2026 \u05d0\u05dc\u05d0 \u05d0\u05ea I match .disabled \u05de\u05ea\u05d2 .selected`) or Chinese's BA particle \u628a
10503
- * (`\u2026 \u9664\u975e \u628a I match .disabled \u5207\u6362 .selected`) \u2014 and the inner toggle loses its
10504
- * own marker. The semantic parser can't recover the guard from that: the marker
10505
- * ahead of the condition blocks the `unless` pattern AND the now-markerless body
10506
- * command fails its object-marked toggle pattern, so the body collapses (`unless`
10507
- * dropped). Marker-less languages (de/it/ar/pl) tolerate the same role-blob and
10508
- * stay faithful, so this is an object-marker artifact, not a general parse gap.
10509
- *
10510
- * The standalone `unless <cond> <body>` path already produces the correct shape
10511
- * (`extractBlockStructure` \u2192 `transformBlock`: condition kept marker-free, body
10512
- * command keeps its marker \u2014 he `\u05d0\u05dc\u05d0 I match .disabled \u05de\u05ea\u05d2 \u05d0\u05ea .selected`, zh
10513
- * `\u9664\u975e I match .disabled \u5207\u6362 \u628a .selected`). So we split the event head off,
10514
- * transform the guard through that path, and emit the event clause first (he and
10515
- * zh are both SVO \u2014 event leads). Returns `null` (fall through) when the input
10516
- * isn't an object-marking event handler with an un-terminated inline `unless`
10517
- * guard.
10518
- */
10519
- tryTransformEventWithUnlessGuard(input) {
10520
- // SVO object-marking targets only \u2014 these front the unless tail with an object
10521
- // marker (he \u05d0\u05ea / zh \u628a) that breaks the parse. Event-leads emission below
10522
- // assumes SVO, so SOV/VSO object-markers (ja/ko/tr/ar) are intentionally out.
10523
- if (!UNLESS_GUARD_OBJECT_MARKING_LOCALES.has(this.targetProfile.code))
10524
- return null;
10525
- const tokens = tokenize(input, this.sourceProfile);
10526
- if (tokens.length === 0)
10527
- return null;
10528
- if (!EVENT_KEYWORDS.has(tokens[0]?.toLowerCase()))
10529
- return null;
10530
- let guardIdx = -1;
10531
- for (let i = 1; i < tokens.length; i++) {
10532
- if (tokens[i].toLowerCase() === 'unless') {
10533
- guardIdx = i;
10534
- break;
10535
- }
10536
- }
10537
- if (guardIdx <= 0)
10538
- return null;
10539
- // The terminated form (`\u2026 unless \u2026 end`) is handled by
10540
- // tryTransformEventWithBlockBody's masking path; only take the inline guard.
10541
- if (tokens[tokens.length - 1].toLowerCase() === 'end')
10542
- return null;
10543
- const eventHead = tokens.slice(0, guardIdx);
10544
- const guard = tokens.slice(guardIdx).join(' ');
10545
- // `unless` is a BLOCK_HEAD keyword, so transform() routes the guard through
10546
- // extractBlockStructure \u2192 transformBlock (the standalone shape that parses).
10547
- const guardOut = this.transform(guard);
10548
- if (!guardOut)
10549
- return null;
10550
- const placeholder = 'EVENTGUARDPLACEHOLDER';
10551
- const headOut = this.transformSingle([...eventHead, placeholder].join(' '));
10552
- if (!headOut.includes(placeholder))
10553
- return null;
10554
- const eventClause = headOut.replace(placeholder, '').replace(/\s+/g, ' ').trim();
10555
- return [eventClause, guardOut].filter(s => s.length > 0).join(' ');
10556
- }
10557
- /**
10558
- * Transform an event handler whose body leads with a command-modifier
10559
- * (`on <event> [from <src>] {async|once|debounced [at N]|throttled [at N]} <body>`).
10560
- *
10561
- * `parseEventHandler` reads the first token after the event as the **action**, so
10562
- * a leading modifier is mistaken for the verb and the real verb (`fetch`/`add`) is
10563
- * swept into the patient. For SOV targets the reorder then surfaces that verb
10564
- * **first** (`\u53d6\u5f97 /api/data \u3092 \u30af\u30ea\u30c3\u30af \u2026`), and the semantic parser matches the
10565
- * leading `<verb> <patient>` with the low-priority `*-generated-verb-first`
10566
- * command pattern \u2014 returning a bare command and discarding the event + the rest
10567
- * of the body (degenerate parse).
10568
- *
10569
- * Instead, lift the modifier out, transform the modifier-free handler through the
10570
- * normal path (which keeps the body in canonical patient-first SOV order so the
10571
- * event sits mid-stream and the existing SOV event-extraction recovers it), then
10572
- * re-emit the modifier as a **leading English literal**. The semantic parser
10573
- * strips a leading `once`/`debounced`/`throttled` (`extractStandaloneModifiers`)
10574
- * and an `async` anywhere (`stripAsyncModifier`) before parsing, so the modifier
10575
- * is consumed as handler metadata rather than shadowing the body.
10576
- *
10577
- * Returns `null` (fall through) when the input isn't an event handler or the body
10578
- * doesn't lead with a modifier \u2014 leaving simple/Mode-B handlers byte-identical.
10579
- */
10580
- tryTransformEventWithModifierBody(input) {
10581
- // The verb-first degenerate parse this works around is specific to SOV
10582
- // reorder: only there does a leading modifier displace the patient-first
10583
- // order and surface the verb first. SVO/VSO/V2/other targets keep the body
10584
- // in an order the parser already handles, so leave them byte-identical.
10585
- if (this.targetProfile.wordOrder !== 'SOV')
10586
- return null;
10587
- const tokens = tokenize(input, this.sourceProfile);
10588
- if (tokens.length === 0)
10589
- return null;
10590
- if (!EVENT_KEYWORDS.has(tokens[0]?.toLowerCase()))
10591
- return null;
10592
- // Walk past the event clause head: event keyword, event token, any
10593
- // `or`-conjoined events, and an optional `from <source>` modifier \u2014 mirroring
10594
- // parseEventHandler's head parsing \u2014 to find where the body begins.
10595
- let i = 1; // past the event keyword
10596
- if (!tokens[i])
10597
- return null;
10598
- i++; // past the event token
10599
- while (tokens[i] && EVENT_CONJUNCTIONS.has(tokens[i].toLowerCase()) && tokens[i + 1]) {
10600
- i += 2;
10601
- }
10602
- if (tokens[i]?.toLowerCase() === 'from' && tokens[i + 1]) {
10603
- i++; // skip 'from'
10604
- // Collect source tokens until a command verb or a body modifier.
10605
- while (tokens[i] &&
10606
- !ENGLISH_COMMANDS.has(tokens[i].toLowerCase()) &&
10607
- !BODY_MODIFIER_KEYWORDS.has(tokens[i].toLowerCase())) {
10608
- i++;
10609
- }
10610
- }
10611
- const modWord = tokens[i]?.toLowerCase();
10612
- if (!modWord || !BODY_MODIFIER_KEYWORDS.has(modWord))
10613
- return null;
10614
- // Consume the modifier phrase. `debounced`/`throttled` may carry an optional
10615
- // `at <duration>` (or a bare duration). `async`/`once` are single tokens.
10616
- const modStart = i;
10617
- let modEnd = i + 1;
10618
- if (modWord !== 'async' && modWord !== 'once') {
10619
- if (tokens[modEnd]?.toLowerCase() === 'at')
10620
- modEnd++;
10621
- if (tokens[modEnd] && /^\d+(ms|s|m)?$/.test(tokens[modEnd]))
10622
- modEnd++;
10623
- }
10624
- const modifierPhrase = tokens.slice(modStart, modEnd).join(' ');
10625
- const rebuilt = [...tokens.slice(0, modStart), ...tokens.slice(modEnd)].join(' ');
10626
- // A handler with only a modifier and no body has nothing to keep patient-first.
10627
- if (tokens.length - (modEnd - modStart) <= 2)
10628
- return null;
10629
- // Re-run the full transform on the modifier-free handler so then-chains and
10630
- // juxtaposed bodies route through the existing (working) paths. The rebuilt
10631
- // input no longer leads with a modifier, so this never re-enters here.
10632
- const bodyOut = this.transform(rebuilt);
10633
- return [modifierPhrase, bodyOut].filter(s => s.length > 0).join(' ');
10634
- }
10635
- /**
10636
- * Transform a `set <stuff> on <scope>` clause (S1 tabs-aria). The trailing
10637
- * `on <scope>` is the element(s) the attribute is set on \u2014 kept attached by
10638
- * splitOnCommandBoundaries. The semantic parser captures it as a `scope` role
10639
- * via the passthrough literal `on` (setSchema's scope markerOverride is `on`
10640
- * in every language), so `on` is emitted verbatim and only the scope *value*
10641
- * is translated (selectors pass through; `me`/`it`/`you` translate to the
10642
- * native reference, which the parser also accepts).
10643
- *
10644
- * Positioning matches where the set patterns expect the scope: at the clause
10645
- * end for verb-first orders (SVO/VSO), and immediately before the clause-final
10646
- * verb for SOV (the generated SOV pattern is `{dest} {patient} on {scope}
10647
- * {verb}`). Returns null (fall through) when there is no trailing `on <scope>`.
10648
- *
10649
- * Source is English in the sync-translations pipeline, so the `set` verb and
10650
- * `on` marker are matched as English literals.
10651
- */
10652
- transformSetWithScope(input) {
10653
- const src = this.sourceProfile.code;
10654
- const dst = this.targetProfile.code;
10655
- const m = input.match(/^(.*\bset\b.*\S)\s+on\s+([#.<@[]\S*|me|it|you)\s*$/i);
10656
- if (!m)
10657
- return null;
10658
- const head = m[1];
10659
- const scopeRaw = m[2];
10660
- // Transform the scope-less clause via the normal path; `head` no longer ends
10661
- // in `on <scope>`, so this never re-enters transformSetWithScope.
10662
- const headOut = this.transformSingle(head);
10663
- const scopeT = /^[#.<@[]/.test(scopeRaw) ? scopeRaw : translateWord(scopeRaw, src, dst);
10664
- // SOV needs the scope positioned per how the semantic set patterns match:
10665
- // - Event-handler set (`on <event> set \u2026`): the dest-first SOV event-handler
10666
- // pattern is verb-MEDIAL and carries an optional trailing `[on {scope}]`,
10667
- // so append the scope at the clause end.
10668
- // - Standalone then-clause set: SOV emits verb-MEDIAL (`{dest} {verb}
10669
- // {patient}`), but the only command set pattern with scope is verb-LAST
10670
- // (`{dest} {patient} on {scope} {verb}`). Move the medial verb to the end
10671
- // and place `on {scope}` before it, so the generated command pattern matches.
10672
- if (this.targetProfile.wordOrder === 'SOV') {
10673
- const verb = translateWord('set', 'en', dst);
10674
- const firstTok = input.trim().split(/\s+/)[0]?.toLowerCase();
10675
- const isEventHandler = !!firstTok && EVENT_KEYWORDS.has(firstTok);
10676
- const toks = headOut.split(/\s+/).filter(Boolean);
10677
- if (!isEventHandler) {
10678
- // Standalone then-clause set: SOV emits verb-MEDIAL; move the verb to the
10679
- // end and place `on {scope}` before it so the verb-last command set
10680
- // pattern (scope before verb) matches.
10681
- const vIdx = toks.indexOf(verb);
10682
- if (vIdx >= 0) {
10683
- toks.splice(vIdx, 1);
10684
- toks.push('on', scopeT, verb);
10685
- return toks.join(' ');
10686
- }
10687
- }
10688
- else if (toks.length > 0 && toks[toks.length - 1] === verb) {
10689
- // Event-handler set whose verb is clause-final (e.g. qu
10690
- // `{dest} ta {patient} man {event} pi {verb}`): the parser extracts the
10691
- // event and matches the body as a verb-last command, so the scope must
10692
- // sit before the verb. Verb-MEDIAL SOV event handlers (ja/ko/tr/bn/hi)
10693
- // fall through to the append branch, where their fused event-handler set
10694
- // pattern carries the trailing `[on {scope}]` group.
10695
- toks.splice(toks.length - 1, 0, 'on', scopeT);
10696
- return toks.join(' ');
10697
- }
10698
- return `${headOut} on ${scopeT}`;
10699
- }
10700
- // Verb-first (SVO/VSO/V2): append `on <scope>` at the clause end, which the
10701
- // trailing `[on {scope}]` group on the set patterns matches.
10702
- return `${headOut} on ${scopeT}`;
10703
- }
10704
- /**
10705
- * Transform a self-contained block command (`{head} {clause?} {body} {end}`),
10706
- * where head \u2208 {if, repeat, unless, \u2026}. The clause (condition / `until event \u2026`)
10707
- * runs up to the first command verb and is translated word-by-word; the body is
10708
- * recursively transformed (so its inner commands reorder for the target); the
10709
- * head/tail keywords are translated. The block is never word-order reordered as
10710
- * a whole \u2014 delimiters stay at the edges regardless of target word order.
10711
- */
10712
- transformBlockBody(blockTokens) {
10713
- const src = this.sourceProfile.code;
10714
- const dst = this.targetProfile.code;
10715
- const head = blockTokens[0];
10716
- const hasEnd = blockTokens[blockTokens.length - 1]?.toLowerCase() === 'end';
10717
- const tail = hasEnd ? blockTokens[blockTokens.length - 1] : '';
10718
- const inner = blockTokens.slice(1, hasEnd ? -1 : undefined);
10719
- // Skip predicate-adjective positions (`\u2026 is empty`): the adjective is
10720
- // part of the condition clause, not the body's first verb \u2014 cutting there
10721
- // displaced it into the next command's argument zone (empty \u00d78 bn/hi/tr).
10722
- const commands = getCommandKeywordsForLocale(src);
10723
- const copulas = getCopulasForLocale(src);
10724
- let bodyStart = inner.findIndex((t, i) => commands.has(t.toLowerCase()) && !isPredicateAdjectivePosition(inner, i, copulas));
10725
- if (bodyStart < 0)
10726
- bodyStart = inner.length;
10727
- const clause = inner.slice(0, bodyStart).join(' ');
10728
- const bodyTokens = inner.slice(bodyStart);
10729
- const headT = translateWord(head, src, dst);
10730
- const tailT = tail ? translateWord(tail, src, dst) : '';
10731
- const clauseT = clause ? translateMultiWordValue(clause, src, dst) : '';
10732
- const bodyT = this.transformConditionalBody(bodyTokens);
10733
- // SOV condition/branch boundary (R1 deferred-tail Family G): the SOV body
10734
- // renders its first command operand-first (`\u6700\u521d <button/> \u306e\u4e2d .modal \u3092
10735
- // \u30d5\u30a9\u30fc\u30ab\u30b9`), so nothing marks where the condition ends and the branch
10736
- // operand begins \u2014 the semantic fold's command-start detection needs a
10737
- // `{value}{particle}` run followed by a verb, which a POSITIONAL-headed
10738
- // operand (`first <button/> \u2026`) never forms, and the condition scan
10739
- // swallows the operand's head (focus-trap: ja focus.patient fell to the
10740
- // `me` default, ko/qu to the `.modal` tail). Emit the target's
10741
- // then-connective at the seam \u2014 the boundary the fold already respects
10742
- // (isThenKeyword) \u2014 gated to exactly the blind shape: SOV target, a
10743
- // positional keyword right after the branch's command verb, and no `then`
10744
- // already ending the condition.
10745
- const POSITIONAL_BRANCH_HEADS = new Set(['first', 'last', 'next', 'previous', 'closest']);
10746
- const thenT = this.targetProfile.wordOrder === 'SOV' &&
10747
- clauseT &&
10748
- POSITIONAL_BRANCH_HEADS.has(bodyTokens[1]?.toLowerCase()) &&
10749
- inner[bodyStart - 1]?.toLowerCase() !== 'then'
10750
- ? translateWord('then', src, dst)
10751
- : '';
10752
- return [headT, clauseT, thenT, bodyT, tailT].filter(s => s.length > 0).join(' ');
10753
- }
10754
- /**
10755
- * Transform an `if`/`unless` block body, splitting it at a top-level `else` into
10756
- * a then-branch and an else-branch so each is reordered as a self-contained unit
10757
- * and the `else` keyword itself is translated. Without this, the body is reordered
10758
- * as one stream: `else` rides along glued to the preceding clause (and, when that
10759
- * clause begins with a selector, is marked a selector and left *untranslated*),
10760
- * and a spurious `then` is inserted around it \u2014 both of which break the target
10761
- * text and the downstream parse. The split is depth-aware so an `else` belonging
10762
- * to a nested block is not mistaken for this block's separator. Bodies without an
10763
- * `else` transform exactly as before.
10764
- */
10765
- transformConditionalBody(bodyTokens) {
10766
- const src = this.sourceProfile.code;
10767
- const dst = this.targetProfile.code;
10768
- const sourceElse = translateWord('else', 'en', src).toLowerCase();
10769
- let depth = 0;
10770
- let elseIdx = -1;
10771
- for (let i = 0; i < bodyTokens.length; i++) {
10772
- const t = bodyTokens[i].toLowerCase();
10773
- if (BLOCK_BODY_KEYWORDS.has(t))
10774
- depth++;
10775
- else if (t === 'end' && depth > 0)
10776
- depth--;
10777
- else if (t === sourceElse && depth === 0) {
10778
- elseIdx = i;
10779
- break;
10780
- }
10781
- }
10782
- if (elseIdx === -1) {
10783
- const body = bodyTokens.join(' ');
10784
- return body ? this.transform(body) : '';
10785
- }
10786
- const thenBranch = bodyTokens.slice(0, elseIdx).join(' ');
10787
- const elseBranch = bodyTokens.slice(elseIdx + 1).join(' ');
10788
- const elseT = translateWord(bodyTokens[elseIdx], src, dst);
10789
- return [
10790
- thenBranch ? this.transform(thenBranch) : '',
10791
- elseT,
10792
- elseBranch ? this.transform(elseBranch) : '',
10793
- ]
10794
- .filter(s => s.length > 0)
10795
- .join(' ');
10796
- }
10797
- /**
10798
- * Translate a reactive block by translating the head/tail/connector
10799
- * via the dictionary, recursively transforming the body through the
10800
- * regular pipeline, and rejoining in source-language position order.
10801
- * Block-syntactic tokens are never reordered: they're delimiters, not
10802
- * arguments, and authors expect them at start/end positions
10803
- * regardless of target word order.
10804
- */
10805
- transformBlock(block) {
10806
- const src = this.sourceProfile.code;
10807
- const dst = this.targetProfile.code;
10808
- const head = translateWord(block.headKeyword, src, dst);
10809
- const tail = block.tailKeyword ? translateWord(block.tailKeyword, src, dst) : '';
10810
- const connector = block.connector ? translateWord(block.connector, src, dst) : '';
10811
- const prefix = block.prefixExpr ? translateMultiWordValue(block.prefixExpr, src, dst) : '';
10812
- // Recurse through `transform()` (not `transformSingle`) so the body
10813
- // gets `then`-splitting and nested-block handling for free.
10814
- const body = this.transform(block.body);
10815
- return [head, prefix, connector, body, tail].filter(s => s.length > 0).join(' ');
10816
- }
10817
- /**
10818
- * Find the best matching rule for this statement
10819
- */
10820
- findRule(parsed) {
10821
- if (!this.targetProfile.rules)
10822
- return undefined;
10823
- const matchingRules = this.targetProfile.rules
10824
- .filter(rule => this.matchesRule(parsed, rule))
10825
- .sort((a, b) => b.priority - a.priority);
10826
- return matchingRules[0];
10827
- }
10828
- /**
10829
- * Check if a parsed statement matches a rule
10830
- */
10831
- matchesRule(parsed, rule) {
10832
- const { match } = rule;
10833
- // Check required roles
10834
- for (const role of match.requiredRoles) {
10835
- if (!parsed.roles.has(role)) {
10836
- return false;
10837
- }
10838
- }
10839
- // Check command match if specified
10840
- if (match.commands && match.commands.length > 0) {
10841
- const action = parsed.roles.get('action');
10842
- if (!action)
10843
- return false;
10844
- const actionValue = action.value.toLowerCase();
10845
- if (!match.commands.some(cmd => cmd.toLowerCase() === actionValue)) {
10846
- return false;
10847
- }
10848
- }
10849
- // Check custom predicate
10850
- if (match.predicate && !match.predicate(parsed)) {
10851
- return false;
10852
- }
10853
- return true;
10854
- }
10855
- }
10856
- // =============================================================================
10857
- // Convenience Functions
10858
- // =============================================================================
10859
- /**
10860
- * Transform hyperscript from English to target language
10861
- */
10862
- function toLocale(input, targetLocale) {
10863
- const transformer = new GrammarTransformer('en', targetLocale);
10864
- return transformer.transform(input);
10865
- }
10866
- /**
10867
- * Transform hyperscript from source language to English
10868
- */
10869
- function toEnglish(input, sourceLocale) {
10870
- const transformer = new GrammarTransformer(sourceLocale, 'en');
10871
- return transformer.transform(input);
10872
- }
10873
- /**
10874
- * Transform between any two languages.
10875
- *
10876
- * Uses direct translation for supported language pairs (ja\u2194zh, es\u2194pt, ko\u2194ja),
10877
- * falling back to English pivot for other pairs.
10878
- */
10879
- function translate(input, sourceLocale, targetLocale) {
10880
- if (sourceLocale === targetLocale)
10881
- return input;
10882
- if (sourceLocale === 'en')
10883
- return toLocale(input, targetLocale);
10884
- if (targetLocale === 'en')
10885
- return toEnglish(input, sourceLocale);
10886
- // Try direct translation for supported pairs
10887
- if (hasDirectMapping(sourceLocale, targetLocale)) {
10888
- return translateDirect(input, sourceLocale, targetLocale);
10889
- }
10890
- // Fallback: Via English pivot
10891
- const english = toEnglish(input, sourceLocale);
10892
- return toLocale(english, targetLocale);
10893
- }
10894
- /**
10895
- * Direct translation between language pairs without English pivot.
10896
- * More accurate for closely related languages (ja\u2194zh, es\u2194pt).
10897
- */
10898
- function translateDirect(input, sourceLocale, targetLocale) {
10899
- const mapping = getDirectMapping(sourceLocale, targetLocale);
10900
- if (!mapping) {
10901
- // Fallback to pivot translation
10902
- return toLocale(toEnglish(input, sourceLocale), targetLocale);
10903
- }
10904
- // Tokenize input
10905
- const tokens = input.split(/\s+/);
10906
- // Translate each token using direct mapping
10907
- const translated = tokens.map(token => {
10908
- // Preserve CSS selectors and literals
10909
- if (token.startsWith('#') || token.startsWith('.') || token.startsWith('@')) {
10910
- return token;
10911
- }
10912
- if (token.startsWith('"') || token.startsWith("'")) {
10913
- return token;
10914
- }
10915
- // Look up in direct mapping
10916
- const directTranslation = mapping.words[token];
10917
- if (directTranslation) {
10918
- return directTranslation;
10919
- }
10920
- // Check for suffix-attached tokens (e.g., "#count-ta" in Quechua)
10921
- const suffixMatch = token.match(/^(.+?)(-.+)$/);
10922
- if (suffixMatch) {
10923
- const [, base, suffix] = suffixMatch;
10924
- const translatedBase = mapping.words[base] || base;
10925
- return translatedBase + suffix;
10926
- }
10927
- // Return unchanged if no mapping found
10928
- return token;
10929
- });
10930
- return translated.join(' ');
10931
- }
10932
- // =============================================================================
10933
- // Examples (for testing)
10934
- // =============================================================================
10935
- const examples = {
10936
- english: {
10937
- eventHandler: 'on click increment #count',
10938
- putInto: 'put my value into #output',
10939
- toggle: 'toggle .active',
10940
- wait: 'wait 2 seconds',
10941
- },
10942
- // Expected outputs (approximate, for reference)
10943
- japanese: {
10944
- eventHandler: '#count \u3092 \u30af\u30ea\u30c3\u30af \u3067 \u5897\u52a0',
10945
- putInto: '\u79c1\u306e \u5024 \u3092 #output \u306b \u7f6e\u304f',
10946
- toggle: '.active \u3092 \u5207\u308a\u66ff\u3048',
10947
- wait: '2\u79d2 \u5f85\u3064',
10948
- },
10949
- chinese: {
10950
- eventHandler: '\u5f53 \u70b9\u51fb \u65f6 \u589e\u52a0 #count',
10951
- putInto: '\u628a \u6211\u7684\u503c \u653e \u5230 #output',
10952
- toggle: '\u5207\u6362 .active',
10953
- wait: '\u7b49\u5f85 2\u79d2',
10954
- },
10955
- arabic: {
10956
- eventHandler: '\u0632\u0650\u062f #count \u0639\u0646\u062f \u0627\u0644\u0646\u0642\u0631',
10957
- putInto: '\u0636\u0639 \u0642\u064a\u0645\u062a\u064a \u0641\u064a #output',
10958
- toggle: '\u0628\u062f\u0651\u0644 .active',
10959
- wait: '\u0627\u0646\u062a\u0638\u0631 \u062b\u0627\u0646\u064a\u062a\u064a\u0646',
10960
- },
10961
- };
10962
-
10963
- export { ENGLISH_COMMANDS, ENGLISH_KEYWORDS, GrammarTransformer, LANGUAGE_FAMILY_DEFAULTS, LocaleManager, UNIVERSAL_ENGLISH_KEYWORDS, UNIVERSAL_PATTERNS, ar$1 as ar, ar$1 as arDictionary, arKeywords, arabicProfile, bn$1 as bn, bn$1 as bnDictionary, bnKeywords, chineseProfile, createEnglishProvider, createKeywordProvider, de$1 as de, de$1 as deDictionary, deKeywords, detectBrowserLocale, directMappings, en$1 as en, englishProfile, es$1 as es, es$1 as esDictionary, esKeywords, fr$1 as fr, fr$1 as frDictionary, frKeywords, frenchProfile, germanProfile, getDirectMapping, getProfile, getSupportedDirectPairs, getSupportedLocales, examples as grammarExamples, hasDirectMapping, he$1 as he, he$1 as heDictionary, heKeywords, hebrewProfile, hindiDictionary as hiDictionary, hiKeywords, hindiDictionary, id$1 as id, id$1 as idDictionary, idKeywords, indonesianProfile, insertMarkers, it$1 as it, it$1 as itDictionary, itKeywords, ja$1 as ja, ja$1 as jaDictionary, jaKeywords, japaneseProfile, joinTokens, ko$1 as ko, ko$1 as koDictionary, koKeywords, koreanProfile, malayProfile, ms$1 as ms, ms$1 as msDictionary, msKeywords, parseStatement, pl$1 as pl, pl$1 as plDictionary, plKeywords, portugueseProfile, profiles, pt$1 as pt, pt$1 as ptDictionary, ptKeywords, qu$1 as qu, qu$1 as quDictionary, quKeywords, quechuaProfile, reorderRoles, russianDictionary as ruDictionary, ruKeywords, russianDictionary, spanishProfile, sw$1 as sw, sw$1 as swDictionary, swKeywords, swahiliProfile, th$1 as th, th$1 as thDictionary, thKeywords, tl$1 as tl, tl$1 as tlDictionary, tlKeywords, toEnglish, toLocale, tr$1 as tr, tr$1 as trDictionary, trKeywords, transformStatement, translate, translateWordDirect, turkishProfile, ukrainianDictionary as ukDictionary, ukKeywords, ukrainianDictionary, vi$1 as vi, vi$1 as viDictionary, viKeywords, zh$1 as zh, zh$1 as zhDictionary, zhKeywords };
8332
+ export { ENGLISH_COMMANDS, ENGLISH_KEYWORDS, LANGUAGE_FAMILY_DEFAULTS, LocaleManager, UNIVERSAL_ENGLISH_KEYWORDS, UNIVERSAL_PATTERNS, ar, ar as arDictionary, arKeywords, arabicProfile, bn, bn as bnDictionary, bnKeywords, chineseProfile, createEnglishProvider, createKeywordProvider, de, de as deDictionary, deKeywords, detectBrowserLocale, directMappings, en, englishProfile, es, es as esDictionary, esKeywords, fr, fr as frDictionary, frKeywords, frenchProfile, germanProfile, getDirectMapping, getProfile, getSupportedDirectPairs, getSupportedLocales, hasDirectMapping, he, he as heDictionary, heKeywords, hebrewProfile, hindiDictionary as hiDictionary, hiKeywords, hindiDictionary, id, id as idDictionary, idKeywords, indonesianProfile, insertMarkers, it, it as itDictionary, itKeywords, ja, ja as jaDictionary, jaKeywords, japaneseProfile, joinTokens, ko, ko as koDictionary, koKeywords, koreanProfile, malayProfile, ms, ms as msDictionary, msKeywords, pl, pl as plDictionary, plKeywords, portugueseProfile, profiles, pt, pt as ptDictionary, ptKeywords, qu, qu as quDictionary, quKeywords, quechuaProfile, reorderRoles, russianDictionary as ruDictionary, ruKeywords, russianDictionary, spanishProfile, sw, sw as swDictionary, swKeywords, swahiliProfile, th, th as thDictionary, thKeywords, tl, tl as tlDictionary, tlKeywords, tr, tr as trDictionary, trKeywords, transformStatement, translateWordDirect, turkishProfile, ukrainianDictionary as ukDictionary, ukKeywords, ukrainianDictionary, vi, vi as viDictionary, viKeywords, zh, zh as zhDictionary, zhKeywords };
10964
8333
  //# sourceMappingURL=lokascript-i18n.mjs.map