@lokascript/i18n 2.11.1 → 3.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -0
- package/dist/browser.cjs +5 -1690
- package/dist/browser.cjs.map +1 -1
- package/dist/browser.d.cts +2 -2
- package/dist/browser.d.ts +2 -2
- package/dist/browser.js +6 -1685
- package/dist/browser.js.map +1 -1
- package/dist/dictionaries/index.cjs +5 -1
- package/dist/dictionaries/index.cjs.map +1 -1
- package/dist/dictionaries/index.js +5 -1
- package/dist/dictionaries/index.js.map +1 -1
- package/dist/{transformer-CsOeqayN.d.cts → index-BykxjYST.d.cts} +1 -232
- package/dist/{transformer-DWCTG1DQ.d.ts → index-DuIef8O7.d.ts} +1 -232
- package/dist/index.cjs +16 -1878
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.js +17 -1873
- package/dist/index.js.map +1 -1
- package/dist/lokascript-i18n.min.js +1 -1
- package/dist/lokascript-i18n.min.js.map +1 -1
- package/dist/lokascript-i18n.mjs +49 -2680
- package/dist/lokascript-i18n.mjs.map +1 -1
- package/dist/plugins/vite.cjs +5 -1
- package/dist/plugins/vite.cjs.map +1 -1
- package/dist/plugins/vite.js +5 -1
- package/dist/plugins/vite.js.map +1 -1
- package/dist/plugins/webpack.cjs +5 -1
- package/dist/plugins/webpack.cjs.map +1 -1
- package/dist/plugins/webpack.js +5 -1
- package/dist/plugins/webpack.js.map +1 -1
- package/package.json +4 -4
- package/src/browser.ts +0 -7
- package/src/compatibility/browser-tests/grammar-demo.spec.ts +22 -8
- package/src/constants.ts +1 -0
- package/src/dictionaries/bn.ts +5 -1
- package/src/grammar/index.ts +15 -9
- package/src/grammar/profiles.test.ts +440 -0
- package/src/index.ts +0 -7
- package/src/lexicon-parity.test.ts +77 -0
- package/src/grammar/grammar.test.ts +0 -2751
- package/src/grammar/transformer.ts +0 -2737
package/dist/lokascript-i18n.mjs
CHANGED
|
@@ -6,60 +6,6 @@
|
|
|
6
6
|
* Maps English modifier keywords to their semantic roles.
|
|
7
7
|
* Used by both the grammar transformer and keyword provider.
|
|
8
8
|
*/
|
|
9
|
-
const ENGLISH_MODIFIER_ROLES = {
|
|
10
|
-
to: 'destination',
|
|
11
|
-
into: 'destination',
|
|
12
|
-
from: 'source',
|
|
13
|
-
with: 'style',
|
|
14
|
-
by: 'quantity',
|
|
15
|
-
as: 'method',
|
|
16
|
-
on: 'event',
|
|
17
|
-
over: 'duration',
|
|
18
|
-
for: 'duration',
|
|
19
|
-
};
|
|
20
|
-
/**
|
|
21
|
-
* Maps a command verb to its **primary** semantic role \u2014 the role of the
|
|
22
|
-
* command's first/leading argument when no explicit modifier keyword marks it.
|
|
23
|
-
*
|
|
24
|
-
* The generic argument parser in the transformer defaults the first unmarked
|
|
25
|
-
* argument to `patient`, which is correct for the majority of commands
|
|
26
|
-
* (`toggle .x`, `add .x`, \u2026) but wrong for commands whose leading argument is a
|
|
27
|
-
* non-patient \u2014 e.g. `wait <duration>`. Marking a duration as a patient emits a
|
|
28
|
-
* spurious object-marker in the target language (Chinese `\u7b49\u5f85 \u628a 1s`, ungrammatical;
|
|
29
|
-
* Japanese `1s \u3092 \u5f85\u3064`; Korean `1s \ub97c \ub300\uae30`), which the semantic parser then fails to
|
|
30
|
-
* match \u2014 dropping the command.
|
|
31
|
-
*
|
|
32
|
-
* Only commands whose primary role is **not** `patient` are listed here (the
|
|
33
|
-
* `patient` default already covers the rest). Mirrors the `primaryRole` field of
|
|
34
|
-
* the semantic package's command schemas (`@lokascript/semantic`); kept in sync by
|
|
35
|
-
* `schema-alignment.test.ts` so this local copy can't drift without the bundle
|
|
36
|
-
* having to pull in the whole semantic graph.
|
|
37
|
-
*
|
|
38
|
-
* @see packages/i18n/src/schema-alignment.test.ts
|
|
39
|
-
* @see docs-internal/ZH_BLOCK_BODY_SCOPE.md (#1 \u2014 transformer role model)
|
|
40
|
-
*/
|
|
41
|
-
const COMMAND_PRIMARY_ROLES = {
|
|
42
|
-
set: 'destination',
|
|
43
|
-
on: 'event',
|
|
44
|
-
trigger: 'event',
|
|
45
|
-
send: 'event',
|
|
46
|
-
wait: 'duration',
|
|
47
|
-
fetch: 'source',
|
|
48
|
-
get: 'source',
|
|
49
|
-
if: 'condition',
|
|
50
|
-
unless: 'condition',
|
|
51
|
-
while: 'condition',
|
|
52
|
-
repeat: 'loopType',
|
|
53
|
-
go: 'destination',
|
|
54
|
-
scroll: 'destination',
|
|
55
|
-
tell: 'destination',
|
|
56
|
-
default: 'destination',
|
|
57
|
-
swap: 'destination',
|
|
58
|
-
// morph deliberately absent: its schema primaryRole is `patient` (the
|
|
59
|
-
// element being morphed \u2014 aligned with the transformer's patient marking
|
|
60
|
-
// in the session-9 role-layout swap), and patient is the default.
|
|
61
|
-
bind: 'destination',
|
|
62
|
-
};
|
|
63
9
|
/**
|
|
64
10
|
* English modifier keywords (derived from ENGLISH_MODIFIER_ROLES)
|
|
65
11
|
*/
|
|
@@ -323,79 +269,6 @@ const ENGLISH_EXPRESSION_KEYWORDS = new Set([
|
|
|
323
269
|
'starts with',
|
|
324
270
|
'ends with',
|
|
325
271
|
]);
|
|
326
|
-
// =============================================================================
|
|
327
|
-
// Conditional Keywords
|
|
328
|
-
// =============================================================================
|
|
329
|
-
/**
|
|
330
|
-
* Conditional keywords across languages (for statement type identification).
|
|
331
|
-
* Includes 'when'/'where' conditional modifiers and their translations.
|
|
332
|
-
*/
|
|
333
|
-
const CONDITIONAL_KEYWORDS = new Set([
|
|
334
|
-
// English
|
|
335
|
-
'if',
|
|
336
|
-
'unless',
|
|
337
|
-
'when',
|
|
338
|
-
'where',
|
|
339
|
-
// Japanese
|
|
340
|
-
'\u3082\u3057',
|
|
341
|
-
'\u6642\u306b',
|
|
342
|
-
'\u3068\u304d\u306b',
|
|
343
|
-
'\u3069\u3053\u3067',
|
|
344
|
-
// Chinese
|
|
345
|
-
'\u5982\u679c',
|
|
346
|
-
'\u5f53',
|
|
347
|
-
// Arabic
|
|
348
|
-
'\u0625\u0630\u0627',
|
|
349
|
-
'\u0639\u0646\u062f\u0645\u0627',
|
|
350
|
-
'\u062d\u064a\u062b',
|
|
351
|
-
// Spanish
|
|
352
|
-
'si',
|
|
353
|
-
'cuando',
|
|
354
|
-
'donde',
|
|
355
|
-
// German
|
|
356
|
-
'wenn',
|
|
357
|
-
'wann',
|
|
358
|
-
'wo',
|
|
359
|
-
// French
|
|
360
|
-
'quand',
|
|
361
|
-
'lorsque',
|
|
362
|
-
'o\u00f9',
|
|
363
|
-
// Portuguese
|
|
364
|
-
'quando',
|
|
365
|
-
'onde',
|
|
366
|
-
// Turkish
|
|
367
|
-
'e\u011fer',
|
|
368
|
-
'zaman',
|
|
369
|
-
'nerede',
|
|
370
|
-
// Indonesian
|
|
371
|
-
'ketika',
|
|
372
|
-
'saat',
|
|
373
|
-
'dimana',
|
|
374
|
-
// Korean
|
|
375
|
-
'\ub54c',
|
|
376
|
-
'\uc5b4\ub514\uc11c',
|
|
377
|
-
// Quechua
|
|
378
|
-
'maypi',
|
|
379
|
-
// Swahili
|
|
380
|
-
'wakati',
|
|
381
|
-
'wapi',
|
|
382
|
-
]);
|
|
383
|
-
/**
|
|
384
|
-
* "Then" keywords across languages (for conditional parsing).
|
|
385
|
-
*/
|
|
386
|
-
const THEN_KEYWORDS = new Set([
|
|
387
|
-
'then',
|
|
388
|
-
'\u305d\u308c\u304b\u3089',
|
|
389
|
-
'\u90a3\u4e48',
|
|
390
|
-
'\u062b\u0645',
|
|
391
|
-
'entonces',
|
|
392
|
-
'alors',
|
|
393
|
-
'dann',
|
|
394
|
-
'sonra',
|
|
395
|
-
'lalu',
|
|
396
|
-
'chayqa',
|
|
397
|
-
'kisha',
|
|
398
|
-
]);
|
|
399
272
|
|
|
400
273
|
// packages/i18n/src/parser/create-provider.ts
|
|
401
274
|
/**
|
|
@@ -623,7 +496,7 @@ function createEnglishProvider() {
|
|
|
623
496
|
|
|
624
497
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
625
498
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
626
|
-
const es
|
|
499
|
+
const es = {
|
|
627
500
|
commands: {
|
|
628
501
|
on: 'en',
|
|
629
502
|
tell: 'decir',
|
|
@@ -847,13 +720,13 @@ const es$1 = {
|
|
|
847
720
|
* parser.parse('en clic alternar .active');
|
|
848
721
|
* ```
|
|
849
722
|
*/
|
|
850
|
-
const esKeywords = createKeywordProvider(es
|
|
723
|
+
const esKeywords = createKeywordProvider(es, 'es', {
|
|
851
724
|
allowEnglishFallback: true,
|
|
852
725
|
});
|
|
853
726
|
|
|
854
727
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
855
728
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
856
|
-
const ja
|
|
729
|
+
const ja = {
|
|
857
730
|
commands: {
|
|
858
731
|
on: '\u3067',
|
|
859
732
|
tell: '\u4f1d\u3048\u308b',
|
|
@@ -1091,13 +964,13 @@ const ja$1 = {
|
|
|
1091
964
|
* parser.parse('\u30af\u30ea\u30c3\u30af \u3067 \u5207\u308a\u66ff\u3048 .active');
|
|
1092
965
|
* ```
|
|
1093
966
|
*/
|
|
1094
|
-
const jaKeywords = createKeywordProvider(ja
|
|
967
|
+
const jaKeywords = createKeywordProvider(ja, 'ja', {
|
|
1095
968
|
allowEnglishFallback: true,
|
|
1096
969
|
});
|
|
1097
970
|
|
|
1098
971
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
1099
972
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
1100
|
-
const fr
|
|
973
|
+
const fr = {
|
|
1101
974
|
commands: {
|
|
1102
975
|
on: 'sur',
|
|
1103
976
|
tell: 'dire',
|
|
@@ -1329,13 +1202,13 @@ const fr$1 = {
|
|
|
1329
1202
|
* parser.parse('sur clic basculer .active');
|
|
1330
1203
|
* ```
|
|
1331
1204
|
*/
|
|
1332
|
-
const frKeywords = createKeywordProvider(fr
|
|
1205
|
+
const frKeywords = createKeywordProvider(fr, 'fr', {
|
|
1333
1206
|
allowEnglishFallback: true,
|
|
1334
1207
|
});
|
|
1335
1208
|
|
|
1336
1209
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
1337
1210
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
1338
|
-
const de
|
|
1211
|
+
const de = {
|
|
1339
1212
|
commands: {
|
|
1340
1213
|
on: 'bei',
|
|
1341
1214
|
tell: 'sagen',
|
|
@@ -1588,13 +1461,13 @@ const de$1 = {
|
|
|
1588
1461
|
* parser.parse('bei klick umschalten .active');
|
|
1589
1462
|
* ```
|
|
1590
1463
|
*/
|
|
1591
|
-
const deKeywords = createKeywordProvider(de
|
|
1464
|
+
const deKeywords = createKeywordProvider(de, 'de', {
|
|
1592
1465
|
allowEnglishFallback: true,
|
|
1593
1466
|
});
|
|
1594
1467
|
|
|
1595
1468
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
1596
1469
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
1597
|
-
const ar
|
|
1470
|
+
const ar = {
|
|
1598
1471
|
commands: {
|
|
1599
1472
|
on: '\u0639\u0644\u0649',
|
|
1600
1473
|
tell: '\u0623\u062e\u0628\u0631',
|
|
@@ -1832,13 +1705,13 @@ const ar$1 = {
|
|
|
1832
1705
|
* parser.parse('\u0639\u0644\u0649 \u0646\u0642\u0631 \u0628\u062f\u0644 .active');
|
|
1833
1706
|
* ```
|
|
1834
1707
|
*/
|
|
1835
|
-
const arKeywords = createKeywordProvider(ar
|
|
1708
|
+
const arKeywords = createKeywordProvider(ar, 'ar', {
|
|
1836
1709
|
allowEnglishFallback: true,
|
|
1837
1710
|
});
|
|
1838
1711
|
|
|
1839
1712
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
1840
1713
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
1841
|
-
const ko
|
|
1714
|
+
const ko = {
|
|
1842
1715
|
commands: {
|
|
1843
1716
|
on: '\uc5d0',
|
|
1844
1717
|
tell: '\ub9d0\ud558\ub2e4',
|
|
@@ -2073,13 +1946,13 @@ const ko$1 = {
|
|
|
2073
1946
|
* parser.parse('\uc5d0 \ud074\ub9ad \ud1a0\uae00 .active');
|
|
2074
1947
|
* ```
|
|
2075
1948
|
*/
|
|
2076
|
-
const koKeywords = createKeywordProvider(ko
|
|
1949
|
+
const koKeywords = createKeywordProvider(ko, 'ko', {
|
|
2077
1950
|
allowEnglishFallback: true,
|
|
2078
1951
|
});
|
|
2079
1952
|
|
|
2080
1953
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
2081
1954
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
2082
|
-
const zh
|
|
1955
|
+
const zh = {
|
|
2083
1956
|
commands: {
|
|
2084
1957
|
on: '\u5f53',
|
|
2085
1958
|
tell: '\u544a\u8bc9',
|
|
@@ -2328,13 +2201,13 @@ const zh$1 = {
|
|
|
2328
2201
|
* parser.parse('\u5f53 \u70b9\u51fb \u5207\u6362 .active');
|
|
2329
2202
|
* ```
|
|
2330
2203
|
*/
|
|
2331
|
-
const zhKeywords = createKeywordProvider(zh
|
|
2204
|
+
const zhKeywords = createKeywordProvider(zh, 'zh', {
|
|
2332
2205
|
allowEnglishFallback: true,
|
|
2333
2206
|
});
|
|
2334
2207
|
|
|
2335
2208
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
2336
2209
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
2337
|
-
const tr
|
|
2210
|
+
const tr = {
|
|
2338
2211
|
commands: {
|
|
2339
2212
|
on: '\u00fczerinde',
|
|
2340
2213
|
tell: 's\u00f6yle',
|
|
@@ -2581,13 +2454,13 @@ const tr$1 = {
|
|
|
2581
2454
|
* parser.parse('\u00fczerinde t\u0131klama de\u011fi\u015ftir .active');
|
|
2582
2455
|
* ```
|
|
2583
2456
|
*/
|
|
2584
|
-
const trKeywords = createKeywordProvider(tr
|
|
2457
|
+
const trKeywords = createKeywordProvider(tr, 'tr', {
|
|
2585
2458
|
allowEnglishFallback: true,
|
|
2586
2459
|
});
|
|
2587
2460
|
|
|
2588
2461
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
2589
2462
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
2590
|
-
const id
|
|
2463
|
+
const id = {
|
|
2591
2464
|
commands: {
|
|
2592
2465
|
on: 'pada',
|
|
2593
2466
|
tell: 'katakan',
|
|
@@ -2831,13 +2704,13 @@ const id$1 = {
|
|
|
2831
2704
|
* parser.parse('pada klik ganti .active');
|
|
2832
2705
|
* ```
|
|
2833
2706
|
*/
|
|
2834
|
-
const idKeywords = createKeywordProvider(id
|
|
2707
|
+
const idKeywords = createKeywordProvider(id, 'id', {
|
|
2835
2708
|
allowEnglishFallback: true,
|
|
2836
2709
|
});
|
|
2837
2710
|
|
|
2838
2711
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
2839
2712
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
2840
|
-
const qu
|
|
2713
|
+
const qu = {
|
|
2841
2714
|
commands: {
|
|
2842
2715
|
on: 'kaqpi',
|
|
2843
2716
|
tell: 'niy',
|
|
@@ -3097,13 +2970,13 @@ const qu$1 = {
|
|
|
3097
2970
|
* parser.parse('\u00f1itiy-pi yapay #count-ta');
|
|
3098
2971
|
* ```
|
|
3099
2972
|
*/
|
|
3100
|
-
const quKeywords = createKeywordProvider(qu
|
|
2973
|
+
const quKeywords = createKeywordProvider(qu, 'qu', {
|
|
3101
2974
|
allowEnglishFallback: true,
|
|
3102
2975
|
});
|
|
3103
2976
|
|
|
3104
2977
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
3105
2978
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
3106
|
-
const sw
|
|
2979
|
+
const sw = {
|
|
3107
2980
|
commands: {
|
|
3108
2981
|
on: 'kwenye',
|
|
3109
2982
|
tell: 'ambia',
|
|
@@ -3370,13 +3243,13 @@ const sw$1 = {
|
|
|
3370
3243
|
* parser.parse('kwenye bonyeza badilisha .active');
|
|
3371
3244
|
* ```
|
|
3372
3245
|
*/
|
|
3373
|
-
const swKeywords = createKeywordProvider(sw
|
|
3246
|
+
const swKeywords = createKeywordProvider(sw, 'sw', {
|
|
3374
3247
|
allowEnglishFallback: true,
|
|
3375
3248
|
});
|
|
3376
3249
|
|
|
3377
3250
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
3378
3251
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
3379
|
-
const pt
|
|
3252
|
+
const pt = {
|
|
3380
3253
|
commands: {
|
|
3381
3254
|
on: 'em',
|
|
3382
3255
|
tell: 'dizer',
|
|
@@ -3615,13 +3488,13 @@ const pt$1 = {
|
|
|
3615
3488
|
* parser.parse('em clique alternar .active');
|
|
3616
3489
|
* ```
|
|
3617
3490
|
*/
|
|
3618
|
-
const ptKeywords = createKeywordProvider(pt
|
|
3491
|
+
const ptKeywords = createKeywordProvider(pt, 'pt', {
|
|
3619
3492
|
allowEnglishFallback: true,
|
|
3620
3493
|
});
|
|
3621
3494
|
|
|
3622
3495
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
3623
3496
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
3624
|
-
const it
|
|
3497
|
+
const it = {
|
|
3625
3498
|
commands: {
|
|
3626
3499
|
on: 'su',
|
|
3627
3500
|
tell: 'dire',
|
|
@@ -3830,13 +3703,13 @@ const it$1 = {
|
|
|
3830
3703
|
},
|
|
3831
3704
|
};
|
|
3832
3705
|
|
|
3833
|
-
const itKeywords = createKeywordProvider(it
|
|
3706
|
+
const itKeywords = createKeywordProvider(it, 'it', {
|
|
3834
3707
|
allowEnglishFallback: true,
|
|
3835
3708
|
});
|
|
3836
3709
|
|
|
3837
3710
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
3838
3711
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
3839
|
-
const vi
|
|
3712
|
+
const vi = {
|
|
3840
3713
|
commands: {
|
|
3841
3714
|
on: 'khi',
|
|
3842
3715
|
tell: 'n\u00f3i v\u1edbi',
|
|
@@ -4043,13 +3916,13 @@ const vi$1 = {
|
|
|
4043
3916
|
},
|
|
4044
3917
|
};
|
|
4045
3918
|
|
|
4046
|
-
const viKeywords = createKeywordProvider(vi
|
|
3919
|
+
const viKeywords = createKeywordProvider(vi, 'vi', {
|
|
4047
3920
|
allowEnglishFallback: true,
|
|
4048
3921
|
});
|
|
4049
3922
|
|
|
4050
3923
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
4051
3924
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
4052
|
-
const pl
|
|
3925
|
+
const pl = {
|
|
4053
3926
|
commands: {
|
|
4054
3927
|
on: 'gdy',
|
|
4055
3928
|
tell: 'powiedz',
|
|
@@ -4264,7 +4137,7 @@ const pl$1 = {
|
|
|
4264
4137
|
},
|
|
4265
4138
|
};
|
|
4266
4139
|
|
|
4267
|
-
const plKeywords = createKeywordProvider(pl
|
|
4140
|
+
const plKeywords = createKeywordProvider(pl, 'pl', {
|
|
4268
4141
|
allowEnglishFallback: true,
|
|
4269
4142
|
});
|
|
4270
4143
|
|
|
@@ -5161,7 +5034,11 @@ const bengaliDictionary = {
|
|
|
5161
5034
|
random: '\u098f\u09b2\u09cb\u09ae\u09c7\u09b2\u09cb',
|
|
5162
5035
|
length: '\u09a6\u09c8\u09b0\u09cd\u0998\u09cd\u09af',
|
|
5163
5036
|
index: '\u09b8\u09c2\u099a\u0995',
|
|
5164
|
-
empty
|
|
5037
|
+
// The EXPRESSION `empty` is the state predicate (`if my value is empty`),
|
|
5038
|
+
// not the command: `\u0996\u09be\u09b2\u09bf-\u0995\u09b0\u09c1\u09a8` is the imperative "empty it!" and belongs to
|
|
5039
|
+
// the `empty` COMMAND, which keeps it. Kept in step with the semantic
|
|
5040
|
+
// lexicon by `lexicon-parity.test.ts`.
|
|
5041
|
+
empty: '\u0996\u09be\u09b2\u09bf',
|
|
5165
5042
|
'starts with': '\u09a6\u09bf\u09af\u09bc\u09c7_\u09b6\u09c1\u09b0\u09c1',
|
|
5166
5043
|
'ends with': '\u09a6\u09bf\u09af\u09bc\u09c7_\u09b6\u09c7\u09b7',
|
|
5167
5044
|
'ignoring case': '\u0995\u09c7\u09b8_\u0989\u09aa\u09c7\u0995\u09cd\u09b7\u09be',
|
|
@@ -5171,9 +5048,9 @@ const bengaliDictionary = {
|
|
|
5171
5048
|
'joined by': '\u09a6\u09cd\u09ac\u09be\u09b0\u09be_\u09af\u09c1\u0995\u09cd\u09a4',
|
|
5172
5049
|
},
|
|
5173
5050
|
};
|
|
5174
|
-
const bn
|
|
5051
|
+
const bn = bengaliDictionary;
|
|
5175
5052
|
|
|
5176
|
-
const bnKeywords = createKeywordProvider(bn
|
|
5053
|
+
const bnKeywords = createKeywordProvider(bn, 'bn', {
|
|
5177
5054
|
allowEnglishFallback: true,
|
|
5178
5055
|
});
|
|
5179
5056
|
|
|
@@ -5358,9 +5235,9 @@ const thaiDictionary = {
|
|
|
5358
5235
|
'joined by': '\u0e23\u0e27\u0e21\u0e14\u0e49\u0e27\u0e22',
|
|
5359
5236
|
},
|
|
5360
5237
|
};
|
|
5361
|
-
const th
|
|
5238
|
+
const th = thaiDictionary;
|
|
5362
5239
|
|
|
5363
|
-
const thKeywords = createKeywordProvider(th
|
|
5240
|
+
const thKeywords = createKeywordProvider(th, 'th', {
|
|
5364
5241
|
allowEnglishFallback: true,
|
|
5365
5242
|
});
|
|
5366
5243
|
|
|
@@ -5596,9 +5473,9 @@ const malayDictionary = {
|
|
|
5596
5473
|
index: 'indeks',
|
|
5597
5474
|
},
|
|
5598
5475
|
};
|
|
5599
|
-
const ms
|
|
5476
|
+
const ms = malayDictionary;
|
|
5600
5477
|
|
|
5601
|
-
const msKeywords = createKeywordProvider(ms
|
|
5478
|
+
const msKeywords = createKeywordProvider(ms, 'ms', {
|
|
5602
5479
|
allowEnglishFallback: true,
|
|
5603
5480
|
});
|
|
5604
5481
|
|
|
@@ -5840,15 +5717,15 @@ const tagalogDictionary = {
|
|
|
5840
5717
|
index: 'indeks',
|
|
5841
5718
|
},
|
|
5842
5719
|
};
|
|
5843
|
-
const tl
|
|
5720
|
+
const tl = tagalogDictionary;
|
|
5844
5721
|
|
|
5845
|
-
const tlKeywords = createKeywordProvider(tl
|
|
5722
|
+
const tlKeywords = createKeywordProvider(tl, 'tl', {
|
|
5846
5723
|
allowEnglishFallback: true,
|
|
5847
5724
|
});
|
|
5848
5725
|
|
|
5849
5726
|
// Generated/merged from semantic profiles \u2014 hand-written entries are preserved
|
|
5850
5727
|
// To add derived entries, update the semantic profile and run: npm run generate:language-assets
|
|
5851
|
-
const he
|
|
5728
|
+
const he = {
|
|
5852
5729
|
commands: {
|
|
5853
5730
|
add: '\u05d4\u05d5\u05e1\u05e3',
|
|
5854
5731
|
append: '\u05e6\u05e8\u05e3',
|
|
@@ -5968,7 +5845,7 @@ const he$1 = {
|
|
|
5968
5845
|
* parser.parse('\u05d1 \u05dc\u05d7\u05d9\u05e6\u05d4 \u05de\u05ea\u05d2 .active');
|
|
5969
5846
|
* ```
|
|
5970
5847
|
*/
|
|
5971
|
-
const heKeywords = createKeywordProvider(he
|
|
5848
|
+
const heKeywords = createKeywordProvider(he, 'he', {
|
|
5972
5849
|
allowEnglishFallback: true,
|
|
5973
5850
|
});
|
|
5974
5851
|
|
|
@@ -6150,7 +6027,7 @@ function detectBrowserLocale() {
|
|
|
6150
6027
|
* English dictionary - identity mapping since English is the canonical hyperscript language.
|
|
6151
6028
|
* This exists primarily for symmetry in translation operations (e.g., en -> es).
|
|
6152
6029
|
*/
|
|
6153
|
-
const en
|
|
6030
|
+
const en = {
|
|
6154
6031
|
commands: {
|
|
6155
6032
|
// Event handling
|
|
6156
6033
|
on: 'on',
|
|
@@ -8250,7 +8127,7 @@ function buildMappingFromDictionaries(sourceDict, targetDict, sourceCode, target
|
|
|
8250
8127
|
* These languages share many concepts through Kanji/Hanzi,
|
|
8251
8128
|
* making direct translation more accurate than pivot translation.
|
|
8252
8129
|
*/
|
|
8253
|
-
const jaZhMapping = buildMappingFromDictionaries(ja
|
|
8130
|
+
const jaZhMapping = buildMappingFromDictionaries(ja, zh, 'ja', 'zh');
|
|
8254
8131
|
/**
|
|
8255
8132
|
* Chinese to Japanese mapping (reverse of jaZh)
|
|
8256
8133
|
*/
|
|
@@ -8264,7 +8141,7 @@ const zhJaMapping = reverseMapping(jaZhMapping);
|
|
|
8264
8141
|
* Both are SOV languages with postposition particles,
|
|
8265
8142
|
* sharing grammatical structure that enables more natural translation.
|
|
8266
8143
|
*/
|
|
8267
|
-
const koJaMapping = buildMappingFromDictionaries(ko
|
|
8144
|
+
const koJaMapping = buildMappingFromDictionaries(ko, ja, 'ko', 'ja');
|
|
8268
8145
|
/**
|
|
8269
8146
|
* Japanese to Korean mapping (reverse of koJa)
|
|
8270
8147
|
*/
|
|
@@ -8452,2513 +8329,5 @@ function getSupportedDirectPairs() {
|
|
|
8452
8329
|
});
|
|
8453
8330
|
}
|
|
8454
8331
|
|
|
8455
|
-
|
|
8456
|
-
* Dictionary Index
|
|
8457
|
-
*
|
|
8458
|
-
* Exports dictionaries for all supported languages.
|
|
8459
|
-
* Each dictionary maps English canonical keywords to locale-specific translations
|
|
8460
|
-
* across 8 categories: commands, modifiers, events, logical, temporal, values,
|
|
8461
|
-
* attributes, and expressions.
|
|
8462
|
-
*
|
|
8463
|
-
* Derivation utilities (deriveFromProfile, createEnglishDictionary) are available
|
|
8464
|
-
* for generating dictionaries from semantic language profiles. See ./derive.ts.
|
|
8465
|
-
*/
|
|
8466
|
-
// Import per-language dictionaries
|
|
8467
|
-
// =============================================================================
|
|
8468
|
-
// Dictionary Exports
|
|
8469
|
-
// =============================================================================
|
|
8470
|
-
/** English dictionary */
|
|
8471
|
-
const en = en$1;
|
|
8472
|
-
/** Spanish dictionary */
|
|
8473
|
-
const es = es$1;
|
|
8474
|
-
/** Japanese dictionary */
|
|
8475
|
-
const ja = ja$1;
|
|
8476
|
-
/** Korean dictionary */
|
|
8477
|
-
const ko = ko$1;
|
|
8478
|
-
/** Chinese dictionary */
|
|
8479
|
-
const zh = zh$1;
|
|
8480
|
-
/** French dictionary */
|
|
8481
|
-
const fr = fr$1;
|
|
8482
|
-
/** German dictionary */
|
|
8483
|
-
const de = de$1;
|
|
8484
|
-
/** Arabic dictionary */
|
|
8485
|
-
const ar = ar$1;
|
|
8486
|
-
/** Turkish dictionary */
|
|
8487
|
-
const tr = tr$1;
|
|
8488
|
-
/** Indonesian dictionary */
|
|
8489
|
-
const id = id$1;
|
|
8490
|
-
/** Portuguese dictionary */
|
|
8491
|
-
const pt = pt$1;
|
|
8492
|
-
/** Quechua dictionary */
|
|
8493
|
-
const qu = qu$1;
|
|
8494
|
-
/** Swahili dictionary */
|
|
8495
|
-
const sw = sw$1;
|
|
8496
|
-
/** Italian dictionary */
|
|
8497
|
-
const it = it$1;
|
|
8498
|
-
/** Vietnamese dictionary */
|
|
8499
|
-
const vi = vi$1;
|
|
8500
|
-
/** Polish dictionary */
|
|
8501
|
-
const pl = pl$1;
|
|
8502
|
-
/** Russian dictionary */
|
|
8503
|
-
const ru = russianDictionary;
|
|
8504
|
-
/** Ukrainian dictionary */
|
|
8505
|
-
const uk = ukrainianDictionary;
|
|
8506
|
-
/** Hindi dictionary */
|
|
8507
|
-
const hi = hindiDictionary;
|
|
8508
|
-
/** Bengali dictionary */
|
|
8509
|
-
const bn = bengaliDictionary;
|
|
8510
|
-
/** Thai dictionary */
|
|
8511
|
-
const th = thaiDictionary;
|
|
8512
|
-
/** Malay dictionary */
|
|
8513
|
-
const ms = malayDictionary;
|
|
8514
|
-
/** Tagalog dictionary */
|
|
8515
|
-
const tl = tagalogDictionary;
|
|
8516
|
-
/** Hebrew dictionary */
|
|
8517
|
-
const he = he$1;
|
|
8518
|
-
// =============================================================================
|
|
8519
|
-
// Dictionary Registry
|
|
8520
|
-
// =============================================================================
|
|
8521
|
-
/**
|
|
8522
|
-
* All available dictionaries indexed by locale code.
|
|
8523
|
-
*/
|
|
8524
|
-
const dictionaries = {
|
|
8525
|
-
en,
|
|
8526
|
-
es,
|
|
8527
|
-
ko,
|
|
8528
|
-
zh,
|
|
8529
|
-
fr,
|
|
8530
|
-
de,
|
|
8531
|
-
ja,
|
|
8532
|
-
ar,
|
|
8533
|
-
tr,
|
|
8534
|
-
id,
|
|
8535
|
-
qu,
|
|
8536
|
-
sw,
|
|
8537
|
-
pt,
|
|
8538
|
-
it,
|
|
8539
|
-
vi,
|
|
8540
|
-
pl,
|
|
8541
|
-
ru,
|
|
8542
|
-
uk,
|
|
8543
|
-
hi,
|
|
8544
|
-
bn,
|
|
8545
|
-
th,
|
|
8546
|
-
ms,
|
|
8547
|
-
tl,
|
|
8548
|
-
he,
|
|
8549
|
-
};
|
|
8550
|
-
|
|
8551
|
-
// packages/i18n/src/types.ts
|
|
8552
|
-
/**
|
|
8553
|
-
* All valid dictionary categories.
|
|
8554
|
-
*/
|
|
8555
|
-
const DICTIONARY_CATEGORIES = [
|
|
8556
|
-
'commands',
|
|
8557
|
-
'modifiers',
|
|
8558
|
-
'events',
|
|
8559
|
-
'logical',
|
|
8560
|
-
'temporal',
|
|
8561
|
-
'values',
|
|
8562
|
-
'attributes',
|
|
8563
|
-
'expressions',
|
|
8564
|
-
];
|
|
8565
|
-
/**
|
|
8566
|
-
* Find a translation in any category of a dictionary.
|
|
8567
|
-
* Returns the English key if found, undefined otherwise.
|
|
8568
|
-
*/
|
|
8569
|
-
function findInDictionary(dict, localizedWord) {
|
|
8570
|
-
const normalized = localizedWord.toLowerCase();
|
|
8571
|
-
for (const category of DICTIONARY_CATEGORIES) {
|
|
8572
|
-
const entries = dict[category];
|
|
8573
|
-
for (const [english, localized] of Object.entries(entries)) {
|
|
8574
|
-
if (localized.toLowerCase() === normalized) {
|
|
8575
|
-
return { category, englishKey: english };
|
|
8576
|
-
}
|
|
8577
|
-
}
|
|
8578
|
-
}
|
|
8579
|
-
return undefined;
|
|
8580
|
-
}
|
|
8581
|
-
/**
|
|
8582
|
-
* Find a translation for an English word in any category.
|
|
8583
|
-
* Returns the localized word if found, undefined otherwise.
|
|
8584
|
-
*/
|
|
8585
|
-
function translateFromEnglish(dict, englishWord) {
|
|
8586
|
-
const normalized = englishWord.toLowerCase();
|
|
8587
|
-
for (const category of DICTIONARY_CATEGORIES) {
|
|
8588
|
-
const entries = dict[category];
|
|
8589
|
-
const translated = entries[normalized];
|
|
8590
|
-
if (translated) {
|
|
8591
|
-
return translated;
|
|
8592
|
-
}
|
|
8593
|
-
}
|
|
8594
|
-
return undefined;
|
|
8595
|
-
}
|
|
8596
|
-
|
|
8597
|
-
/**
|
|
8598
|
-
* Grammar-Aware Transformer
|
|
8599
|
-
*
|
|
8600
|
-
* Transforms hyperscript statements between languages using the
|
|
8601
|
-
* generalized grammar system. The key insight is that semantic
|
|
8602
|
-
* roles are universal - only their surface realization differs.
|
|
8603
|
-
*/
|
|
8604
|
-
// =============================================================================
|
|
8605
|
-
// Compound Statement Handling
|
|
8606
|
-
// =============================================================================
|
|
8607
|
-
/**
|
|
8608
|
-
* Get all command keywords including translated ones for a locale.
|
|
8609
|
-
*/
|
|
8610
|
-
function getCommandKeywordsForLocale(locale) {
|
|
8611
|
-
const keywords = new Set(ENGLISH_COMMANDS);
|
|
8612
|
-
// Add translated command keywords from dictionaries
|
|
8613
|
-
const dict = dictionaries[locale];
|
|
8614
|
-
if (dict?.commands) {
|
|
8615
|
-
Object.values(dict.commands).forEach(cmd => {
|
|
8616
|
-
if (typeof cmd === 'string') {
|
|
8617
|
-
keywords.add(cmd.toLowerCase());
|
|
8618
|
-
}
|
|
8619
|
-
});
|
|
8620
|
-
}
|
|
8621
|
-
return keywords;
|
|
8622
|
-
}
|
|
8623
|
-
/**
|
|
8624
|
-
* English copula forms. A command keyword IMMEDIATELY after one of these is a
|
|
8625
|
-
* predicate adjective, not a command verb: in `if my value is empty add .error
|
|
8626
|
-
* to me` the `empty` belongs to the condition (`empty` is also a hyperscript
|
|
8627
|
-
* command, v0.9.90). Without this guard the condition/body scans cut the
|
|
8628
|
-
* condition at `is` and displace the adjective into the body's argument zone,
|
|
8629
|
-
* where it anchors a spurious `empty-{lang}-generated` parse AND steals a
|
|
8630
|
-
* neighboring role (the empty \u00d78 bn/hi/tr family). Source-locale copulas are
|
|
8631
|
-
* added via the dictionary in `isPredicateAdjectivePosition`.
|
|
8632
|
-
*/
|
|
8633
|
-
const EN_COPULAS = ['is', 'are', 'was', 'were', 'am', 'be'];
|
|
8634
|
-
function getCopulasForLocale(locale) {
|
|
8635
|
-
const copulas = new Set(EN_COPULAS);
|
|
8636
|
-
if (locale !== 'en') {
|
|
8637
|
-
for (const form of EN_COPULAS) {
|
|
8638
|
-
copulas.add(translateWord(form, 'en', locale).toLowerCase());
|
|
8639
|
-
}
|
|
8640
|
-
}
|
|
8641
|
-
return copulas;
|
|
8642
|
-
}
|
|
8643
|
-
/**
|
|
8644
|
-
* True when tokens[i] sits right after a copula \u2014 a predicate-adjective
|
|
8645
|
-
* position that must never be read as the start of a body command.
|
|
8646
|
-
*/
|
|
8647
|
-
function isPredicateAdjectivePosition(tokens, i, copulas) {
|
|
8648
|
-
const prev = tokens[i - 1]?.toLowerCase();
|
|
8649
|
-
return !!prev && copulas.has(prev);
|
|
8650
|
-
}
|
|
8651
|
-
/**
|
|
8652
|
-
* Source-locale surface forms of the `for` command keyword and the loop's
|
|
8653
|
-
* `in` preposition, for the loop-head test below.
|
|
8654
|
-
*/
|
|
8655
|
-
function getForLoopWordsForLocale(locale) {
|
|
8656
|
-
const forWords = new Set(['for']);
|
|
8657
|
-
const inWords = new Set(['in']);
|
|
8658
|
-
if (locale !== 'en') {
|
|
8659
|
-
forWords.add(translateWord('for', 'en', locale).toLowerCase());
|
|
8660
|
-
inWords.add(translateWord('in', 'en', locale).toLowerCase());
|
|
8661
|
-
}
|
|
8662
|
-
return { forWords, inWords };
|
|
8663
|
-
}
|
|
8664
|
-
/**
|
|
8665
|
-
* True when a `for` at tokens[i] heads a real loop. Hyperscript's only
|
|
8666
|
-
* statement-head `for` is `for <var> in <iterable>`, so a `for` with no `in`
|
|
8667
|
-
* among the tokens before the next command keyword is a ROLE PHRASE \u2014 take's
|
|
8668
|
-
* target (`take .active from .tab-button for me`) or a duration (`for 2s`) \u2014
|
|
8669
|
-
* and must never split the statement: the split shattered the phrase into a
|
|
8670
|
-
* dangling `then for me` clause, which six SOV languages then parsed as a
|
|
8671
|
-
* spurious `for` loop with patient "me" (the take-class \u00d76 family).
|
|
8672
|
-
*/
|
|
8673
|
-
function isLoopHeadFor(tokens, i, inWords, commandKeywords) {
|
|
8674
|
-
for (let j = i + 1; j < tokens.length; j++) {
|
|
8675
|
-
const lt = tokens[j].toLowerCase();
|
|
8676
|
-
if (inWords.has(lt))
|
|
8677
|
-
return true;
|
|
8678
|
-
if (commandKeywords.has(lt))
|
|
8679
|
-
return false;
|
|
8680
|
-
}
|
|
8681
|
-
return false;
|
|
8682
|
-
}
|
|
8683
|
-
/**
|
|
8684
|
-
* Repair a FRONTED Hebrew accusative marker in transformed output.
|
|
8685
|
-
*
|
|
8686
|
-
* When an event-handler body leads with a command-modifier (`on click once add \u2026`,
|
|
8687
|
-
* via {@link GrammarTransformer.tryTransformEventWithModifierBody}) or is a control
|
|
8688
|
-
* block (`on blur if \u2026 add \u2026 end`, via `tryTransformEventWithBlockBody`), the body
|
|
8689
|
-
* command's accusative marker \u05d0\u05ea can be emitted AHEAD of its verb \u2014 `add .error to me`
|
|
8690
|
-
* renders `\u2026 \u05d0\u05ea \u05d4\u05d5\u05e1\u05e3 .error \u2026` instead of the canonical `\u2026 \u05d4\u05d5\u05e1\u05e3 \u05d0\u05ea .error \u2026`. \u05d0\u05ea before
|
|
8691
|
-
* a verb is always ungrammatical Hebrew (it only ever marks a FOLLOWING definite
|
|
8692
|
-
* object), and the semantic parser drops the command in every parse path (fused-event,
|
|
8693
|
-
* multi-clause, conditional-body) when the marker is fronted but parses it when the
|
|
8694
|
-
* marker follows the verb. So an `<accusative-marker> <command-verb>` adjacency is a
|
|
8695
|
-
* pure transformer artifact: swap it back. Idempotent and safe \u2014 only touches `\u05d0\u05ea
|
|
8696
|
-
* <verb>`, never the ~40 generated `<verb> \u05d0\u05ea {patient}` patterns that embed \u05d0\u05ea legitimately.
|
|
8697
|
-
*/
|
|
8698
|
-
function repairHebrewFrontedAccusative(text) {
|
|
8699
|
-
const ACC = '\u05d0\u05ea'; // hebrewProfile.markers patient marker
|
|
8700
|
-
const verbs = getCommandKeywordsForLocale('he');
|
|
8701
|
-
const tokens = text.split(/\s+/);
|
|
8702
|
-
let changed = false;
|
|
8703
|
-
for (let i = 0; i + 1 < tokens.length; i++) {
|
|
8704
|
-
if (tokens[i] === ACC && verbs.has(tokens[i + 1].toLowerCase())) {
|
|
8705
|
-
[tokens[i], tokens[i + 1]] = [tokens[i + 1], tokens[i]];
|
|
8706
|
-
changed = true;
|
|
8707
|
-
i++; // skip the marker we just moved
|
|
8708
|
-
}
|
|
8709
|
-
}
|
|
8710
|
-
return changed ? tokens.join(' ') : text;
|
|
8711
|
-
}
|
|
8712
|
-
/**
|
|
8713
|
-
* Detect a reactive block (`live ... end`, `when X changes Y [end]`,
|
|
8714
|
-
* `unless X Y [end]`) and decompose it. Returns `null` when the input
|
|
8715
|
-
* is not a reactive block, when there's content after the matched
|
|
8716
|
-
* `end`, or when the heuristic can't locate a body \u2014 all of which fall
|
|
8717
|
-
* through to the standard `parseStatement` path.
|
|
8718
|
-
*/
|
|
8719
|
-
function extractBlockStructure(input, sourceLocale) {
|
|
8720
|
-
const tokens = input.split(/\s+/);
|
|
8721
|
-
const head = tokens[0]?.toLowerCase();
|
|
8722
|
-
if (!head || !BLOCK_HEAD_KEYWORDS.has(head))
|
|
8723
|
-
return null;
|
|
8724
|
-
// Depth-aware match for the closing `end` so nested blocks
|
|
8725
|
-
// (`live when X changes Y end end`) slice correctly.
|
|
8726
|
-
let depth = 1;
|
|
8727
|
-
let endIdx = -1;
|
|
8728
|
-
for (let i = 1; i < tokens.length; i++) {
|
|
8729
|
-
const t = tokens[i].toLowerCase();
|
|
8730
|
-
if (BLOCK_HEAD_KEYWORDS.has(t))
|
|
8731
|
-
depth++;
|
|
8732
|
-
else if (t === 'end') {
|
|
8733
|
-
depth--;
|
|
8734
|
-
if (depth === 0) {
|
|
8735
|
-
endIdx = i;
|
|
8736
|
-
break;
|
|
8737
|
-
}
|
|
8738
|
-
}
|
|
8739
|
-
}
|
|
8740
|
-
// If there's trailing content after the matched `end`, bail out and
|
|
8741
|
-
// let the existing splitter handle it. (`splitOnThen` normally
|
|
8742
|
-
// separates trailing code before we get here.)
|
|
8743
|
-
if (endIdx !== -1 && endIdx !== tokens.length - 1)
|
|
8744
|
-
return null;
|
|
8745
|
-
const inner = endIdx !== -1 ? tokens.slice(1, endIdx) : tokens.slice(1);
|
|
8746
|
-
const base = { headKeyword: tokens[0], body: '' };
|
|
8747
|
-
if (endIdx !== -1)
|
|
8748
|
-
base.tailKeyword = tokens[endIdx];
|
|
8749
|
-
if (head === 'live') {
|
|
8750
|
-
return { ...base, body: inner.join(' ') };
|
|
8751
|
-
}
|
|
8752
|
-
if (head === 'when') {
|
|
8753
|
-
// Reactive: `when <expr> changes <body>`. Without `changes`, fall
|
|
8754
|
-
// through to the standard event-wait path (parseConditional).
|
|
8755
|
-
const idx = inner.findIndex(t => t.toLowerCase() === 'changes');
|
|
8756
|
-
if (idx >= 0) {
|
|
8757
|
-
return {
|
|
8758
|
-
...base,
|
|
8759
|
-
prefixExpr: inner.slice(0, idx).join(' '),
|
|
8760
|
-
connector: inner[idx],
|
|
8761
|
-
body: inner.slice(idx + 1).join(' '),
|
|
8762
|
-
};
|
|
8763
|
-
}
|
|
8764
|
-
return null;
|
|
8765
|
-
}
|
|
8766
|
-
// `unless <cond> <body>`: condition runs up to the first command
|
|
8767
|
-
// keyword in `inner`. Heuristic \u2014 works because hyperscript bodies
|
|
8768
|
-
// always start with a command verb, and `unless` conditions rarely
|
|
8769
|
-
// contain bare command keywords as values. A candidate right after a
|
|
8770
|
-
// copula (`\u2026 is empty`) is a predicate adjective inside the condition,
|
|
8771
|
-
// not a body verb \u2014 skip it.
|
|
8772
|
-
const commands = getCommandKeywordsForLocale(sourceLocale);
|
|
8773
|
-
const copulas = getCopulasForLocale(sourceLocale);
|
|
8774
|
-
let bodyStart = -1;
|
|
8775
|
-
for (let i = 0; i < inner.length; i++) {
|
|
8776
|
-
if (commands.has(inner[i].toLowerCase()) && !isPredicateAdjectivePosition(inner, i, copulas)) {
|
|
8777
|
-
bodyStart = i;
|
|
8778
|
-
break;
|
|
8779
|
-
}
|
|
8780
|
-
}
|
|
8781
|
-
if (bodyStart <= 0)
|
|
8782
|
-
return null;
|
|
8783
|
-
return {
|
|
8784
|
-
...base,
|
|
8785
|
-
prefixExpr: inner.slice(0, bodyStart).join(' '),
|
|
8786
|
-
body: inner.slice(bodyStart).join(' '),
|
|
8787
|
-
};
|
|
8788
|
-
}
|
|
8789
|
-
function splitCompoundStatement(input, sourceLocale) {
|
|
8790
|
-
// First, split on newlines (preserving non-empty lines)
|
|
8791
|
-
const lines = input
|
|
8792
|
-
.split(/\n/)
|
|
8793
|
-
.map(line => line.trim())
|
|
8794
|
-
.filter(line => line.length > 0);
|
|
8795
|
-
// If we have multiple lines, treat each as a separate part
|
|
8796
|
-
// (but still need to handle "then" within each line)
|
|
8797
|
-
const parts = [];
|
|
8798
|
-
for (const line of lines) {
|
|
8799
|
-
const lineParts = splitOnThen(line, sourceLocale);
|
|
8800
|
-
// Further split each part on command boundaries
|
|
8801
|
-
for (const part of lineParts) {
|
|
8802
|
-
const commandParts = splitOnCommandBoundaries(part, sourceLocale);
|
|
8803
|
-
parts.push(...commandParts);
|
|
8804
|
-
}
|
|
8805
|
-
}
|
|
8806
|
-
return parts;
|
|
8807
|
-
}
|
|
8808
|
-
/**
|
|
8809
|
-
* Split a compound statement while preserving line structure metadata.
|
|
8810
|
-
* This tracks indentation and blank lines for reconstruction.
|
|
8811
|
-
*/
|
|
8812
|
-
function splitCompoundStatementWithMetadata(input, sourceLocale) {
|
|
8813
|
-
const rawLines = input.split('\n');
|
|
8814
|
-
const lineMetadata = [];
|
|
8815
|
-
const parts = [];
|
|
8816
|
-
const partToLineIndex = [];
|
|
8817
|
-
for (let lineIndex = 0; lineIndex < rawLines.length; lineIndex++) {
|
|
8818
|
-
const rawLine = rawLines[lineIndex];
|
|
8819
|
-
// Capture leading whitespace
|
|
8820
|
-
const indentMatch = rawLine.match(/^(\s*)/);
|
|
8821
|
-
const originalIndent = indentMatch ? indentMatch[1] : '';
|
|
8822
|
-
const trimmed = rawLine.trim();
|
|
8823
|
-
lineMetadata.push({
|
|
8824
|
-
content: trimmed,
|
|
8825
|
-
originalIndent,
|
|
8826
|
-
isBlank: trimmed.length === 0,
|
|
8827
|
-
});
|
|
8828
|
-
if (trimmed.length > 0) {
|
|
8829
|
-
// Process non-empty lines for "then" and command boundaries
|
|
8830
|
-
const lineParts = splitOnThen(trimmed, sourceLocale);
|
|
8831
|
-
for (const part of lineParts) {
|
|
8832
|
-
const commandParts = splitOnCommandBoundaries(part, sourceLocale);
|
|
8833
|
-
for (const cmdPart of commandParts) {
|
|
8834
|
-
parts.push(cmdPart);
|
|
8835
|
-
partToLineIndex.push(lineIndex);
|
|
8836
|
-
}
|
|
8837
|
-
}
|
|
8838
|
-
}
|
|
8839
|
-
}
|
|
8840
|
-
return { parts, lineMetadata, partToLineIndex };
|
|
8841
|
-
}
|
|
8842
|
-
/**
|
|
8843
|
-
* Normalize indentation to consistent 4-space levels.
|
|
8844
|
-
* Preserves relative indentation structure while standardizing spacing.
|
|
8845
|
-
*/
|
|
8846
|
-
function normalizeIndentation(lineMetadata) {
|
|
8847
|
-
// Find non-blank lines with indentation
|
|
8848
|
-
const indentedLines = lineMetadata.filter(m => !m.isBlank && m.originalIndent.length > 0);
|
|
8849
|
-
if (indentedLines.length === 0) {
|
|
8850
|
-
// No indented lines, return empty strings
|
|
8851
|
-
return lineMetadata.map(() => '');
|
|
8852
|
-
}
|
|
8853
|
-
// Find minimum non-zero indent (the base unit)
|
|
8854
|
-
const indentLengths = indentedLines.map(m => {
|
|
8855
|
-
// Convert tabs to 4 spaces for consistent measurement
|
|
8856
|
-
const normalized = m.originalIndent.replace(/\t/g, ' ');
|
|
8857
|
-
return normalized.length;
|
|
8858
|
-
});
|
|
8859
|
-
const minIndent = Math.min(...indentLengths);
|
|
8860
|
-
const baseUnit = minIndent > 0 ? minIndent : 4;
|
|
8861
|
-
// Normalize each line's indentation
|
|
8862
|
-
return lineMetadata.map(meta => {
|
|
8863
|
-
if (meta.isBlank) {
|
|
8864
|
-
return ''; // Blank lines get no indentation
|
|
8865
|
-
}
|
|
8866
|
-
if (meta.originalIndent.length === 0) {
|
|
8867
|
-
return ''; // No original indent
|
|
8868
|
-
}
|
|
8869
|
-
// Convert tabs and calculate level
|
|
8870
|
-
const normalized = meta.originalIndent.replace(/\t/g, ' ');
|
|
8871
|
-
const level = Math.round(normalized.length / baseUnit);
|
|
8872
|
-
return ' '.repeat(level); // 4 spaces per level
|
|
8873
|
-
});
|
|
8874
|
-
}
|
|
8875
|
-
/**
|
|
8876
|
-
* Reconstruct output with preserved line structure.
|
|
8877
|
-
* Maps transformed parts back to their original lines with proper indentation.
|
|
8878
|
-
*/
|
|
8879
|
-
function reconstructWithLineStructure(transformedParts, lineMetadata, partToLineIndex, targetThen) {
|
|
8880
|
-
// If there's only one non-blank line, simple case
|
|
8881
|
-
const nonBlankCount = lineMetadata.filter(m => !m.isBlank).length;
|
|
8882
|
-
if (nonBlankCount <= 1 && transformedParts.length <= 1) {
|
|
8883
|
-
const normalizedIndents = normalizeIndentation(lineMetadata);
|
|
8884
|
-
const result = [];
|
|
8885
|
-
for (let i = 0; i < lineMetadata.length; i++) {
|
|
8886
|
-
if (lineMetadata[i].isBlank) {
|
|
8887
|
-
result.push('');
|
|
8888
|
-
}
|
|
8889
|
-
else if (transformedParts.length > 0) {
|
|
8890
|
-
result.push(normalizedIndents[i] + transformedParts[0]);
|
|
8891
|
-
}
|
|
8892
|
-
}
|
|
8893
|
-
return result.join('\n');
|
|
8894
|
-
}
|
|
8895
|
-
// Normalize indentation
|
|
8896
|
-
const normalizedIndents = normalizeIndentation(lineMetadata);
|
|
8897
|
-
// Group transformed parts by their original line
|
|
8898
|
-
const partsPerLine = new Map();
|
|
8899
|
-
for (let i = 0; i < transformedParts.length; i++) {
|
|
8900
|
-
const lineIdx = partToLineIndex[i];
|
|
8901
|
-
if (!partsPerLine.has(lineIdx)) {
|
|
8902
|
-
partsPerLine.set(lineIdx, []);
|
|
8903
|
-
}
|
|
8904
|
-
partsPerLine.get(lineIdx).push(transformedParts[i]);
|
|
8905
|
-
}
|
|
8906
|
-
// Build result lines
|
|
8907
|
-
const result = [];
|
|
8908
|
-
for (let i = 0; i < lineMetadata.length; i++) {
|
|
8909
|
-
const meta = lineMetadata[i];
|
|
8910
|
-
const indent = normalizedIndents[i];
|
|
8911
|
-
if (meta.isBlank) {
|
|
8912
|
-
result.push('');
|
|
8913
|
-
}
|
|
8914
|
-
else {
|
|
8915
|
-
const lineParts = partsPerLine.get(i) || [];
|
|
8916
|
-
if (lineParts.length > 0) {
|
|
8917
|
-
// Join multiple parts on same line with "then"
|
|
8918
|
-
const lineContent = lineParts.join(` ${targetThen} `);
|
|
8919
|
-
result.push(indent + lineContent);
|
|
8920
|
-
}
|
|
8921
|
-
}
|
|
8922
|
-
}
|
|
8923
|
-
return result.join('\n');
|
|
8924
|
-
}
|
|
8925
|
-
/**
|
|
8926
|
-
* Split a statement on command keyword boundaries.
|
|
8927
|
-
* E.g., "wait 2s toggle .highlight" \u2192 ["wait 2s", "toggle .highlight"]
|
|
8928
|
-
*
|
|
8929
|
-
* Special cases:
|
|
8930
|
-
* - "on <event> <command>" stays together (event handler with first command)
|
|
8931
|
-
* - Modifiers like "to", "from" don't trigger splits
|
|
8932
|
-
*/
|
|
8933
|
-
/**
|
|
8934
|
-
* English modifier keywords that should not trigger command boundary splits.
|
|
8935
|
-
* These are the base set; localized equivalents (e.g. Japanese `\u306b`, Spanish
|
|
8936
|
-
* `a`, Arabic `\u0625\u0644\u0649`) are layered on per source locale by
|
|
8937
|
-
* `getBoundaryModifiersForLocale`, so a preposition in a non-English source
|
|
8938
|
-
* is also recognized as a modifier rather than a spurious command boundary.
|
|
8939
|
-
*/
|
|
8940
|
-
const BOUNDARY_MODIFIERS = new Set([
|
|
8941
|
-
'to',
|
|
8942
|
-
'into',
|
|
8943
|
-
'from',
|
|
8944
|
-
'with',
|
|
8945
|
-
'by',
|
|
8946
|
-
'as',
|
|
8947
|
-
'at',
|
|
8948
|
-
'in',
|
|
8949
|
-
'on',
|
|
8950
|
-
'of',
|
|
8951
|
-
'over',
|
|
8952
|
-
]);
|
|
8953
|
-
/**
|
|
8954
|
-
* Boundary modifiers resolved for a given source locale: the English base set
|
|
8955
|
-
* (always kept, since input may mix English keywords) plus every grammatical
|
|
8956
|
-
* marker form declared by the locale's profile. Profile markers are the
|
|
8957
|
-
* surface realizations of semantic roles (destination, source, style, \u2026) \u2014
|
|
8958
|
-
* they always bind to a following value, so none should be treated as a
|
|
8959
|
-
* command boundary. Cached per locale because profiles are static.
|
|
8960
|
-
*/
|
|
8961
|
-
const boundaryModifiersCache = new Map();
|
|
8962
|
-
function getBoundaryModifiersForLocale(locale) {
|
|
8963
|
-
const cached = boundaryModifiersCache.get(locale);
|
|
8964
|
-
if (cached)
|
|
8965
|
-
return cached;
|
|
8966
|
-
const modifiers = new Set(BOUNDARY_MODIFIERS);
|
|
8967
|
-
const profile = getProfile(locale);
|
|
8968
|
-
profile?.markers.forEach(marker => {
|
|
8969
|
-
const form = marker.form.replace(/^-|-$/g, '').toLowerCase();
|
|
8970
|
-
if (form)
|
|
8971
|
-
modifiers.add(form);
|
|
8972
|
-
marker.alternatives?.forEach(alt => {
|
|
8973
|
-
const altForm = alt.replace(/^-|-$/g, '').toLowerCase();
|
|
8974
|
-
if (altForm)
|
|
8975
|
-
modifiers.add(altForm);
|
|
8976
|
-
});
|
|
8977
|
-
});
|
|
8978
|
-
boundaryModifiersCache.set(locale, modifiers);
|
|
8979
|
-
return modifiers;
|
|
8980
|
-
}
|
|
8981
|
-
/**
|
|
8982
|
-
* Commands for which a trailing `on <element>` is the TARGET the command acts
|
|
8983
|
-
* on (a destination), not a new event-handler clause. For these, the locative
|
|
8984
|
-
* `on` must neither split the statement (`splitOnCommandBoundaries`) nor be read
|
|
8985
|
-
* as a fresh `event` role (`buildArgumentModifierMap`) \u2014 see those call sites.
|
|
8986
|
-
*
|
|
8987
|
-
* Deliberately narrow. `on` is overloaded (event-handler head vs. locative
|
|
8988
|
-
* target), and the change has cross-language blast radius, so we only enable
|
|
8989
|
-
* the locative reading for the DOM class/attribute mutators where it's the
|
|
8990
|
-
* documented idiom (`toggle .open on #menu`, `toggle @hidden on #panel`).
|
|
8991
|
-
*
|
|
8992
|
-
* `trigger`/`send` \u2014 `trigger X on Y` / `send X on Y` fire an event on a target
|
|
8993
|
-
* element; the trailing `on Y` is that target (destination), not a new clause.
|
|
8994
|
-
* Previously excluded: keeping `on Y` attached produced `trigger X \u2192 #y` output
|
|
8995
|
-
* that was thought to destabilise the semantic parser's multi-line behaviour
|
|
8996
|
-
* fallback (behavior-sortable's `trigger sortable:start on me`). That concern is
|
|
8997
|
-
* stale \u2014 the recent semantic-parser body increments fold the surrounding block
|
|
8998
|
-
* cleanly, and splitting instead injected a spurious `then` (`disparar
|
|
8999
|
-
* sortable:start entonces en yo`) that glued the following `repeat until event`
|
|
9000
|
-
* loop into a then-chain and dropped it. Keeping `on Y` attached restores
|
|
9001
|
-
* behavior-sortable to faithful across the SVO languages.
|
|
9002
|
-
*
|
|
9003
|
-
* Excluded on purpose:
|
|
9004
|
-
* - `set` \u2014 `set @attr to V on Y` carries BOTH `to` (value) and `on` (target);
|
|
9005
|
-
* English marks both as `destination`, so remapping `on` would collide with
|
|
9006
|
-
* and clobber the value. Needs distinct value/target roles first (deferred).
|
|
9007
|
-
* - `put` \u2014 `put X on Y into Z` has the same dual-destination collision.
|
|
9008
|
-
*
|
|
9009
|
-
* `add`/`remove` use `to`/`from` for their target in practice (not `on`), so
|
|
9010
|
-
* their inclusion is a harmless no-op that documents intent.
|
|
9011
|
-
*/
|
|
9012
|
-
const ON_TARGET_COMMANDS = new Set(['toggle', 'add', 'remove', 'trigger', 'send']);
|
|
9013
|
-
/**
|
|
9014
|
-
* Find the command verb of a partially-collected statement: the first command
|
|
9015
|
-
* keyword that is neither the event-handler head (`on`/\u305d\u306e\u4ed6 event markers) nor
|
|
9016
|
-
* an argument-introducing preposition. For `on click toggle .open` \u2192 `toggle`;
|
|
9017
|
-
* for `trigger sortable:start` \u2192 `trigger`.
|
|
9018
|
-
*/
|
|
9019
|
-
function commandVerbOf(tokens, commandKeywords) {
|
|
9020
|
-
for (const token of tokens) {
|
|
9021
|
-
const lt = token.toLowerCase();
|
|
9022
|
-
if (commandKeywords.has(lt) && !BOUNDARY_MODIFIERS.has(lt) && !EVENT_KEYWORDS.has(lt)) {
|
|
9023
|
-
return lt;
|
|
9024
|
-
}
|
|
9025
|
-
}
|
|
9026
|
-
return null;
|
|
9027
|
-
}
|
|
9028
|
-
/**
|
|
9029
|
-
* Block-introducing keywords whose body should not be split at command
|
|
9030
|
-
* boundaries by `splitOnCommandBoundaries`. Inputs starting with one of
|
|
9031
|
-
* these are also routed around `parseStatement` entirely by
|
|
9032
|
-
* `extractBlockStructure` + `transformBlock` so block-syntactic tokens
|
|
9033
|
-
* never reach `parseCommand`/`parseConditional` (where they'd be
|
|
9034
|
-
* misinterpreted as command verbs or swept into role values).
|
|
9035
|
-
*
|
|
9036
|
-
* `if` deliberately stays out: `if X then Y end` already works via the
|
|
9037
|
-
* `splitOnThen` + `parseConditional` path.
|
|
9038
|
-
*/
|
|
9039
|
-
const BLOCK_HEAD_KEYWORDS = new Set(['live', 'when', 'unless']);
|
|
9040
|
-
/**
|
|
9041
|
-
* Source-language block-introducing command keywords whose body is a clause plus a
|
|
9042
|
-
* branch/loop body (harness source is English, so these are English-keyed). Used to
|
|
9043
|
-
* locate a block body inside an event handler and to track block depth when finding
|
|
9044
|
-
* a top-level `else`.
|
|
9045
|
-
*/
|
|
9046
|
-
const BLOCK_BODY_KEYWORDS = new Set(['if', 'repeat', 'unless', 'while', 'for']);
|
|
9047
|
-
/**
|
|
9048
|
-
* SVO targets that mark the object/patient with a particle (he \u05d0\u05ea, zh \u628a) and so
|
|
9049
|
-
* mangle an inline `on <event> unless <cond> <body>` guard \u2014 the unless tail is
|
|
9050
|
-
* swept into one patient blob and the marker lands ahead of the condition. These
|
|
9051
|
-
* route through `tryTransformEventWithUnlessGuard`. SOV/VSO object-markers
|
|
9052
|
-
* (ja/ko/tr/ar) are excluded: their event does not lead, so the SVO event-first
|
|
9053
|
-
* emission there would be wrong (and they don't exhibit the artifact today).
|
|
9054
|
-
*/
|
|
9055
|
-
const UNLESS_GUARD_OBJECT_MARKING_LOCALES = new Set(['he', 'zh']);
|
|
9056
|
-
function splitOnCommandBoundaries(input, sourceLocale) {
|
|
9057
|
-
const commandKeywords = getCommandKeywordsForLocale(sourceLocale);
|
|
9058
|
-
const boundaryModifiers = getBoundaryModifiersForLocale(sourceLocale);
|
|
9059
|
-
const { forWords, inWords } = getForLoopWordsForLocale(sourceLocale);
|
|
9060
|
-
const tokens = input.split(/\s+/);
|
|
9061
|
-
if (tokens.length === 0)
|
|
9062
|
-
return [input];
|
|
9063
|
-
const parts = [];
|
|
9064
|
-
let currentPart = [];
|
|
9065
|
-
// Check if this starts with an event handler pattern (on/em/en/bei/\u3067 + event)
|
|
9066
|
-
const firstTokenLower = tokens[0]?.toLowerCase();
|
|
9067
|
-
const isEventHandler = EVENT_KEYWORDS.has(firstTokenLower);
|
|
9068
|
-
// If it's an event handler, the first command after the event is part of the handler
|
|
9069
|
-
// So we need to track whether we've seen the first command yet
|
|
9070
|
-
let seenFirstCommand = !isEventHandler; // If not event handler, we're already past the "first command" phase
|
|
9071
|
-
// Track block-scope depth (live/when/bind/if/unless/for/while/...). While
|
|
9072
|
-
// inside a block, do not split on command boundaries \u2014 the body belongs
|
|
9073
|
-
// to the block head and must transform as one unit. See comments on
|
|
9074
|
-
// BLOCK_HEAD_KEYWORDS for the failure mode this prevents.
|
|
9075
|
-
let blockDepth = 0;
|
|
9076
|
-
for (let i = 0; i < tokens.length; i++) {
|
|
9077
|
-
const token = tokens[i];
|
|
9078
|
-
const lowerToken = token.toLowerCase();
|
|
9079
|
-
// Update block-scope depth before any split decision.
|
|
9080
|
-
if (BLOCK_HEAD_KEYWORDS.has(lowerToken)) {
|
|
9081
|
-
blockDepth++;
|
|
9082
|
-
}
|
|
9083
|
-
else if (lowerToken === 'end' && blockDepth > 0) {
|
|
9084
|
-
blockDepth--;
|
|
9085
|
-
}
|
|
9086
|
-
// If this is a command keyword and we already have tokens in current part
|
|
9087
|
-
if (commandKeywords.has(lowerToken) && currentPart.length > 0) {
|
|
9088
|
-
// Check if the previous token looks like it could end a command
|
|
9089
|
-
const prevToken = currentPart[currentPart.length - 1];
|
|
9090
|
-
const prevLower = prevToken.toLowerCase();
|
|
9091
|
-
// For event handlers: don't split before the first command
|
|
9092
|
-
// E.g., "on click wait 1s" should stay together
|
|
9093
|
-
if (!seenFirstCommand) {
|
|
9094
|
-
// Mark that we've now seen the first command
|
|
9095
|
-
seenFirstCommand = true;
|
|
9096
|
-
currentPart.push(token);
|
|
9097
|
-
continue;
|
|
9098
|
-
}
|
|
9099
|
-
// Don't split inside a block (live/when/bind/unless body, etc.).
|
|
9100
|
-
// The block head and its body must transform as one statement.
|
|
9101
|
-
if (blockDepth > 0) {
|
|
9102
|
-
currentPart.push(token);
|
|
9103
|
-
continue;
|
|
9104
|
-
}
|
|
9105
|
-
// A `for` that doesn't head a real loop (`for <var> in <iterable>`) is a
|
|
9106
|
-
// role phrase of the current command \u2014 see isLoopHeadFor.
|
|
9107
|
-
if (forWords.has(lowerToken) && !isLoopHeadFor(tokens, i, inWords, commandKeywords)) {
|
|
9108
|
-
currentPart.push(token);
|
|
9109
|
-
continue;
|
|
9110
|
-
}
|
|
9111
|
-
// Locative `on` for a DOM-target command (`toggle X on Y`) is the target
|
|
9112
|
-
// element, NOT a new command. `on` lands in `commandKeywords` only
|
|
9113
|
-
// incidentally \u2014 the EN dictionary registers `commands.on = 'on'` for the
|
|
9114
|
-
// event-handler head \u2014 so without this guard the destination `on` split
|
|
9115
|
-
// the statement, the join re-inserted a spurious `then` (\u062b\u0645/pagkatapos/\u2026),
|
|
9116
|
-
// and the dangling `on Y` was misread as a second event handler. Keep it
|
|
9117
|
-
// attached so the role parser can assign it to `destination`. Restricted
|
|
9118
|
-
// to ON_TARGET_COMMANDS so `trigger X on me` etc. keep their prior split.
|
|
9119
|
-
if (BOUNDARY_MODIFIERS.has(lowerToken)) {
|
|
9120
|
-
const verb = commandVerbOf(currentPart, commandKeywords);
|
|
9121
|
-
if (verb && ON_TARGET_COMMANDS.has(verb)) {
|
|
9122
|
-
currentPart.push(token);
|
|
9123
|
-
continue;
|
|
9124
|
-
}
|
|
9125
|
-
// `set @attr to V on <scope>` (S1 tabs-aria): the trailing `on <scope>`
|
|
9126
|
-
// is the element(s) the attribute is set on, not a new command. `set` is
|
|
9127
|
-
// deliberately NOT in ON_TARGET_COMMANDS (its role parser would clobber
|
|
9128
|
-
// the value), so this is a dedicated guard \u2014 kept attached only when a
|
|
9129
|
-
// selector/reference scope follows, then positioned by transformSingle's
|
|
9130
|
-
// set-scope handler (transformSetWithScope). A `set` whose `on` is
|
|
9131
|
-
// followed by a verb is left to split as before.
|
|
9132
|
-
if (lowerToken === 'on' && verb === 'set') {
|
|
9133
|
-
const nextTok = tokens[i + 1];
|
|
9134
|
-
const nextLower = nextTok?.toLowerCase();
|
|
9135
|
-
const scopeLike = !!nextTok &&
|
|
9136
|
-
(/^[#.<@[]/.test(nextTok) ||
|
|
9137
|
-
nextLower === 'me' ||
|
|
9138
|
-
nextLower === 'it' ||
|
|
9139
|
-
nextLower === 'you');
|
|
9140
|
-
if (scopeLike) {
|
|
9141
|
-
currentPart.push(token);
|
|
9142
|
-
continue;
|
|
9143
|
-
}
|
|
9144
|
-
}
|
|
9145
|
-
}
|
|
9146
|
-
if (!boundaryModifiers.has(prevLower) && !commandKeywords.has(prevLower)) {
|
|
9147
|
-
// This looks like a command boundary - save current part and start new one
|
|
9148
|
-
parts.push(currentPart.join(' '));
|
|
9149
|
-
currentPart = [token];
|
|
9150
|
-
continue;
|
|
9151
|
-
}
|
|
9152
|
-
}
|
|
9153
|
-
currentPart.push(token);
|
|
9154
|
-
}
|
|
9155
|
-
// Add the last part
|
|
9156
|
-
if (currentPart.length > 0) {
|
|
9157
|
-
parts.push(currentPart.join(' '));
|
|
9158
|
-
}
|
|
9159
|
-
return parts.filter(p => p.length > 0);
|
|
9160
|
-
}
|
|
9161
|
-
/**
|
|
9162
|
-
* Split a single line on "then" keywords.
|
|
9163
|
-
*/
|
|
9164
|
-
function splitOnThen(input, sourceLocale) {
|
|
9165
|
-
// Build regex pattern from all known "then" keywords
|
|
9166
|
-
const thenKeywords = Array.from(THEN_KEYWORDS);
|
|
9167
|
-
// Add any dictionary-specific "then" keyword for the source locale
|
|
9168
|
-
const sourceDict = sourceLocale === 'en' ? null : dictionaries[sourceLocale];
|
|
9169
|
-
if (sourceDict?.modifiers?.then) {
|
|
9170
|
-
thenKeywords.push(sourceDict.modifiers.then);
|
|
9171
|
-
}
|
|
9172
|
-
// Also check logical.then since some dictionaries put it there
|
|
9173
|
-
if (sourceDict?.logical?.then) {
|
|
9174
|
-
thenKeywords.push((sourceDict?.logical).then);
|
|
9175
|
-
}
|
|
9176
|
-
// Create a regex that matches any "then" keyword as a whole word
|
|
9177
|
-
// Use word boundaries to avoid matching "then" inside other words
|
|
9178
|
-
const escapedKeywords = thenKeywords.map(k => k.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'));
|
|
9179
|
-
const pattern = new RegExp(`\\s+(${escapedKeywords.join('|')})\\s+`, 'gi');
|
|
9180
|
-
// Split on "then" keywords
|
|
9181
|
-
const parts = input.split(pattern).filter(part => {
|
|
9182
|
-
// Filter out the "then" keywords themselves (captured by the group)
|
|
9183
|
-
const lowerPart = part.toLowerCase().trim();
|
|
9184
|
-
return lowerPart && !thenKeywords.some(k => k.toLowerCase() === lowerPart);
|
|
9185
|
-
});
|
|
9186
|
-
return parts.map(p => p.trim()).filter(p => p.length > 0);
|
|
9187
|
-
}
|
|
9188
|
-
/**
|
|
9189
|
-
* Get the "then" keyword in the target language.
|
|
9190
|
-
* Checks both modifiers and logical sections since dictionaries vary.
|
|
9191
|
-
*/
|
|
9192
|
-
function getTargetThenKeyword(targetLocale) {
|
|
9193
|
-
if (targetLocale === 'en')
|
|
9194
|
-
return 'then';
|
|
9195
|
-
const targetDict = dictionaries[targetLocale];
|
|
9196
|
-
if (!targetDict)
|
|
9197
|
-
return 'then';
|
|
9198
|
-
// Check modifiers first, then logical (dictionaries vary)
|
|
9199
|
-
return (targetDict.modifiers?.then || targetDict.logical?.then || 'then');
|
|
9200
|
-
}
|
|
9201
|
-
// =============================================================================
|
|
9202
|
-
// Derived Constants from Profiles
|
|
9203
|
-
// =============================================================================
|
|
9204
|
-
/**
|
|
9205
|
-
* Derive event keywords from all language profiles.
|
|
9206
|
-
* This replaces the hardcoded eventKeywords array.
|
|
9207
|
-
*/
|
|
9208
|
-
function deriveEventKeywordsFromProfiles() {
|
|
9209
|
-
const keywords = new Set();
|
|
9210
|
-
// Add 'on' as the English default
|
|
9211
|
-
keywords.add('on');
|
|
9212
|
-
// Extract event markers from all profiles
|
|
9213
|
-
for (const profile of Object.values(profiles)) {
|
|
9214
|
-
for (const marker of profile.markers) {
|
|
9215
|
-
if (marker.role === 'event') {
|
|
9216
|
-
// Strip hyphen notation and add
|
|
9217
|
-
const form = marker.form.replace(/^-|-$/g, '').toLowerCase();
|
|
9218
|
-
if (form)
|
|
9219
|
-
keywords.add(form);
|
|
9220
|
-
// Add alternatives
|
|
9221
|
-
marker.alternatives?.forEach(alt => {
|
|
9222
|
-
const altForm = alt.replace(/^-|-$/g, '').toLowerCase();
|
|
9223
|
-
if (altForm)
|
|
9224
|
-
keywords.add(altForm);
|
|
9225
|
-
});
|
|
9226
|
-
}
|
|
9227
|
-
}
|
|
9228
|
-
}
|
|
9229
|
-
return keywords;
|
|
9230
|
-
}
|
|
9231
|
-
/** Event keywords derived from language profiles */
|
|
9232
|
-
const EVENT_KEYWORDS = deriveEventKeywordsFromProfiles();
|
|
9233
|
-
/**
|
|
9234
|
-
* Conjunctions that join multiple events in an event handler head
|
|
9235
|
-
* (`on click or keypress ...`). Source hyperscript is English, so the
|
|
9236
|
-
* canonical form is `or`; localized equivalents are added defensively for
|
|
9237
|
-
* non-English sources.
|
|
9238
|
-
*/
|
|
9239
|
-
const EVENT_CONJUNCTIONS = new Set(['or']);
|
|
9240
|
-
/**
|
|
9241
|
-
* Command-modifier keywords that may lead an event handler body
|
|
9242
|
-
* (`on click async fetch \u2026`, `on click once add \u2026`). They modify the handler /
|
|
9243
|
-
* following command rather than acting as the verb, so the transformer must lift
|
|
9244
|
-
* them out before role assignment instead of treating them as the action.
|
|
9245
|
-
* Source hyperscript is English, so these are the canonical English forms \u2014 the
|
|
9246
|
-
* semantic parser recognizes the same literals when stripping them pre-parse.
|
|
9247
|
-
*/
|
|
9248
|
-
const BODY_MODIFIER_KEYWORDS = new Set([
|
|
9249
|
-
'async',
|
|
9250
|
-
'once',
|
|
9251
|
-
'debounced',
|
|
9252
|
-
'debounce',
|
|
9253
|
-
'throttled',
|
|
9254
|
-
'throttle',
|
|
9255
|
-
]);
|
|
9256
|
-
// =============================================================================
|
|
9257
|
-
// Helper: Dynamic Modifier Map
|
|
9258
|
-
// =============================================================================
|
|
9259
|
-
/**
|
|
9260
|
-
* Generates a lookup map for semantic roles based on the language profile.
|
|
9261
|
-
* Maps markers (e.g., 'to', '\u306b', 'into', '\u0625\u0644\u0649') to their semantic roles.
|
|
9262
|
-
* This enables parsing non-English input by using the profile's markers.
|
|
9263
|
-
*/
|
|
9264
|
-
function generateModifierMap(profile) {
|
|
9265
|
-
const map = {};
|
|
9266
|
-
// Map markers to roles from the profile
|
|
9267
|
-
profile.markers.forEach(marker => {
|
|
9268
|
-
// Strip hyphen notation for suffix/prefix markers
|
|
9269
|
-
const form = marker.form.replace(/^-|-$/g, '').toLowerCase();
|
|
9270
|
-
if (form) {
|
|
9271
|
-
map[form] = marker.role;
|
|
9272
|
-
}
|
|
9273
|
-
// Map alternatives if they exist (e.g., Korean vowel harmony variants)
|
|
9274
|
-
marker.alternatives?.forEach(alt => {
|
|
9275
|
-
const altForm = alt.replace(/^-|-$/g, '').toLowerCase();
|
|
9276
|
-
if (altForm) {
|
|
9277
|
-
map[altForm] = marker.role;
|
|
9278
|
-
}
|
|
9279
|
-
});
|
|
9280
|
-
});
|
|
9281
|
-
// Add English modifiers as fallback (don't override profile-specific markers)
|
|
9282
|
-
for (const [key, role] of Object.entries(ENGLISH_MODIFIER_ROLES)) {
|
|
9283
|
-
if (!(key in map)) {
|
|
9284
|
-
map[key] = role;
|
|
9285
|
-
}
|
|
9286
|
-
}
|
|
9287
|
-
return map;
|
|
9288
|
-
}
|
|
9289
|
-
/**
|
|
9290
|
-
* Modifier map for parsing the ARGUMENTS of a statement (everything after the
|
|
9291
|
-
* command verb), as opposed to the statement head.
|
|
9292
|
-
*
|
|
9293
|
-
* Why a separate map: a few languages reuse one word for both the event-handler
|
|
9294
|
-
* head marker and a locative/target preposition. English is the prime case \u2014
|
|
9295
|
-
* `on` is the event head (`on click \u2026`) AND the toggle/set target preposition
|
|
9296
|
-
* (`toggle .x on #y`). `generateModifierMap` maps `on \u2192 event` (from the EN
|
|
9297
|
-
* profile's event marker), so when the destination `on` was read in argument
|
|
9298
|
-
* position it overwrote the already-captured head event (the `click` got
|
|
9299
|
-
* dropped, e.g. `toggle @hidden on #panel` \u2192 `\u2026 \u0639\u0646\u062f #panel` with \u0646\u0642\u0631/click gone).
|
|
9300
|
-
*
|
|
9301
|
-
* The fix: in argument position, remap the event role to `destination`. This is
|
|
9302
|
-
* safe because a trigger event only ever appears at the statement head (which is
|
|
9303
|
-
* consumed before argument parsing begins). Any "event marker" reached while
|
|
9304
|
-
* scanning arguments is therefore being used as a locative \u2014 i.e. the element
|
|
9305
|
-
* the command acts ON \u2014 which is exactly the `destination` role (the semantic
|
|
9306
|
-
* parser also models `toggle .x on #y` as patient `.x` + destination `#y`).
|
|
9307
|
-
*
|
|
9308
|
-
* Restricted to (a) SVO source profiles and (b) ON_TARGET_COMMANDS:
|
|
9309
|
-
* - SVO: in VSO/SOV languages the event marker can legitimately appear
|
|
9310
|
-
* mid-statement (Arabic VSO renders an event handler as `\u0628\u062f\u0644 X \u0639\u0646\u062f \u0646\u0642\u0631`,
|
|
9311
|
-
* classified as a command), so remapping there would mis-read a real event
|
|
9312
|
-
* as a destination. English \u2014 and the en\u2192lang gate path \u2014 is SVO.
|
|
9313
|
-
* - command: only the DOM class/attr mutators use a locative `on` target.
|
|
9314
|
-
* For other commands (`trigger X on me`, `set X to V on Y`) the remap is
|
|
9315
|
-
* either destabilising or collides with `to`/`into` \u2014 see ON_TARGET_COMMANDS.
|
|
9316
|
-
*
|
|
9317
|
-
* When neither applies the unmodified map is returned (event stays event).
|
|
9318
|
-
*/
|
|
9319
|
-
function buildArgumentModifierMap(profile, actionVerb) {
|
|
9320
|
-
const map = generateModifierMap(profile);
|
|
9321
|
-
const verb = actionVerb?.toLowerCase();
|
|
9322
|
-
if (profile.wordOrder !== 'SVO' || !verb || !ON_TARGET_COMMANDS.has(verb)) {
|
|
9323
|
-
return map;
|
|
9324
|
-
}
|
|
9325
|
-
const remapped = {};
|
|
9326
|
-
for (const [form, role] of Object.entries(map)) {
|
|
9327
|
-
remapped[form] = role === 'event' ? 'destination' : role;
|
|
9328
|
-
}
|
|
9329
|
-
return remapped;
|
|
9330
|
-
}
|
|
9331
|
-
// =============================================================================
|
|
9332
|
-
// Statement Parser
|
|
9333
|
-
// =============================================================================
|
|
9334
|
-
/**
|
|
9335
|
-
* Parse a hyperscript statement into semantic roles
|
|
9336
|
-
* This is the core analysis step that identifies WHAT each part means
|
|
9337
|
-
*/
|
|
9338
|
-
function parseStatement(input, sourceLocale = 'en') {
|
|
9339
|
-
const profile = getProfile(sourceLocale);
|
|
9340
|
-
if (!profile)
|
|
9341
|
-
return null;
|
|
9342
|
-
const tokens = tokenize(input, profile);
|
|
9343
|
-
// Identify statement type and extract roles
|
|
9344
|
-
const statementType = identifyStatementType(tokens, profile);
|
|
9345
|
-
switch (statementType) {
|
|
9346
|
-
case 'event-handler':
|
|
9347
|
-
return parseEventHandler(tokens, profile);
|
|
9348
|
-
case 'command':
|
|
9349
|
-
return parseCommand(tokens, profile);
|
|
9350
|
-
case 'conditional':
|
|
9351
|
-
return parseConditional(tokens);
|
|
9352
|
-
default:
|
|
9353
|
-
return null;
|
|
9354
|
-
}
|
|
9355
|
-
}
|
|
9356
|
-
/**
|
|
9357
|
-
* Known suffixes that may attach to words without spaces.
|
|
9358
|
-
* These are split off during tokenization for proper parsing.
|
|
9359
|
-
*/
|
|
9360
|
-
const ATTACHED_SUFFIXES = {
|
|
9361
|
-
// Chinese: \u65f6 (time/when) often attaches to events like \u70b9\u51fb\u65f6 (when clicking)
|
|
9362
|
-
zh: ['\u65f6', '\u7684', '\u5730', '\u5f97'],
|
|
9363
|
-
// Japanese: Some particles may attach in casual writing
|
|
9364
|
-
ja: [],
|
|
9365
|
-
// Korean: Particles sometimes written without spaces
|
|
9366
|
-
ko: [],
|
|
9367
|
-
};
|
|
9368
|
-
/**
|
|
9369
|
-
* Known prefixes that may attach to words without spaces.
|
|
9370
|
-
*/
|
|
9371
|
-
const ATTACHED_PREFIXES = {
|
|
9372
|
-
// Chinese: \u5f53 (when) sometimes written attached
|
|
9373
|
-
zh: ['\u5f53'],
|
|
9374
|
-
// Arabic: Prepositions that attach
|
|
9375
|
-
ar: ['\u0628\u0640', '\u0643\u0640', '\u0648'],
|
|
9376
|
-
};
|
|
9377
|
-
/**
|
|
9378
|
-
* Post-process tokens to split attached suffixes/prefixes.
|
|
9379
|
-
* E.g., "\u70b9\u51fb\u65f6" \u2192 ["\u70b9\u51fb", "\u65f6"]
|
|
9380
|
-
*/
|
|
9381
|
-
function splitAttachedAffixes(tokens, locale) {
|
|
9382
|
-
const suffixes = ATTACHED_SUFFIXES[locale] || [];
|
|
9383
|
-
const prefixes = ATTACHED_PREFIXES[locale] || [];
|
|
9384
|
-
if (suffixes.length === 0 && prefixes.length === 0) {
|
|
9385
|
-
return tokens;
|
|
9386
|
-
}
|
|
9387
|
-
const result = [];
|
|
9388
|
-
for (const token of tokens) {
|
|
9389
|
-
// Skip CSS selectors and numbers
|
|
9390
|
-
if (/^[#.<@]/.test(token) || /^\d+/.test(token)) {
|
|
9391
|
-
result.push(token);
|
|
9392
|
-
continue;
|
|
9393
|
-
}
|
|
9394
|
-
let processed = token;
|
|
9395
|
-
let prefix = '';
|
|
9396
|
-
let suffix = '';
|
|
9397
|
-
// Check for attached prefixes
|
|
9398
|
-
for (const p of prefixes) {
|
|
9399
|
-
if (processed.startsWith(p) && processed.length > p.length) {
|
|
9400
|
-
prefix = p;
|
|
9401
|
-
processed = processed.slice(p.length);
|
|
9402
|
-
break;
|
|
9403
|
-
}
|
|
9404
|
-
}
|
|
9405
|
-
// Check for attached suffixes
|
|
9406
|
-
for (const s of suffixes) {
|
|
9407
|
-
if (processed.endsWith(s) && processed.length > s.length) {
|
|
9408
|
-
suffix = s;
|
|
9409
|
-
processed = processed.slice(0, -s.length);
|
|
9410
|
-
break;
|
|
9411
|
-
}
|
|
9412
|
-
}
|
|
9413
|
-
// Add tokens in order: prefix, main, suffix
|
|
9414
|
-
if (prefix)
|
|
9415
|
-
result.push(prefix);
|
|
9416
|
-
if (processed)
|
|
9417
|
-
result.push(processed);
|
|
9418
|
-
if (suffix)
|
|
9419
|
-
result.push(suffix);
|
|
9420
|
-
}
|
|
9421
|
-
return result;
|
|
9422
|
-
}
|
|
9423
|
-
/**
|
|
9424
|
-
* Simple tokenizer that handles:
|
|
9425
|
-
* - Keywords (from dictionary)
|
|
9426
|
-
* - CSS selectors (#id, .class, <tag/>)
|
|
9427
|
-
* - String literals
|
|
9428
|
-
* - Numbers
|
|
9429
|
-
* - Attached suffixes/prefixes (language-specific)
|
|
9430
|
-
*/
|
|
9431
|
-
function tokenize(input, profile) {
|
|
9432
|
-
// Split on whitespace, preserving selectors and strings
|
|
9433
|
-
const tokens = [];
|
|
9434
|
-
let current = '';
|
|
9435
|
-
let inSelector = false;
|
|
9436
|
-
let selectorDepth = 0;
|
|
9437
|
-
let bracketDepth = 0;
|
|
9438
|
-
let parenDepth = 0;
|
|
9439
|
-
for (let i = 0; i < input.length; i++) {
|
|
9440
|
-
const char = input[i];
|
|
9441
|
-
// Track CSS selector context
|
|
9442
|
-
if (char === '<') {
|
|
9443
|
-
inSelector = true;
|
|
9444
|
-
selectorDepth++;
|
|
9445
|
-
}
|
|
9446
|
-
else if (char === '>' && inSelector) {
|
|
9447
|
-
selectorDepth--;
|
|
9448
|
-
if (selectorDepth === 0)
|
|
9449
|
-
inSelector = false;
|
|
9450
|
-
}
|
|
9451
|
-
// Track event-guard / attribute brackets so `[key is 'Escape']` (which has
|
|
9452
|
-
// internal spaces) stays a single token instead of splitting into
|
|
9453
|
-
// `[key` / `is` / `'Escape']` \u2014 which mis-assigns `is` as the action verb.
|
|
9454
|
-
if (char === '[') {
|
|
9455
|
-
bracketDepth++;
|
|
9456
|
-
}
|
|
9457
|
-
else if (char === ']' && bracketDepth > 0) {
|
|
9458
|
-
bracketDepth--;
|
|
9459
|
-
}
|
|
9460
|
-
// Track EVERY parenthesized group as a depth scope so it stays ONE token:
|
|
9461
|
-
//
|
|
9462
|
-
// - An ATTACHED `(` (a call or event destructure: `pointerdown(clientX,
|
|
9463
|
-
// clientY)`, `Resizable(a, b)`) must not split at the comma-space \u2014 the
|
|
9464
|
-
// event-handler-head reorder would separate the halves and drop the
|
|
9465
|
-
// whole handler (tl behavior-resizable degenerate).
|
|
9466
|
-
// - A STANDALONE `(` opening an expression (`to (the value of #price as
|
|
9467
|
-
// Number) * (my value as Number)`) must be OPAQUE to role segmentation:
|
|
9468
|
-
// left loose, its interior `of`/`as`/`from` keywords hit the argument
|
|
9469
|
-
// modifier map, so the parser split the expression across roles \u2014 the
|
|
9470
|
-
// R1 cluster E mangle (computed-value: the transformer reordered INSIDE
|
|
9471
|
-
// the parens, embedded the event phrase mid-expression, and dropped the
|
|
9472
|
-
// whole second operand in every language). Interior keywords still
|
|
9473
|
-
// translate IN PLACE: role values are re-split on whitespace by
|
|
9474
|
-
// translateMultiWordValue, and translateWord strips paren punctuation
|
|
9475
|
-
// before its dictionary lookup (`($count or 0)` \u2192 `($count o 0)` in tl \u2014
|
|
9476
|
-
// the operator translates without the group being torn apart).
|
|
9477
|
-
if (char === '(') {
|
|
9478
|
-
parenDepth++;
|
|
9479
|
-
}
|
|
9480
|
-
else if (char === ')' && parenDepth > 0) {
|
|
9481
|
-
parenDepth--;
|
|
9482
|
-
}
|
|
9483
|
-
// Split on whitespace unless inside a selector, a bracket guard, or an
|
|
9484
|
-
// attached parenthesized argument list
|
|
9485
|
-
if (/\s/.test(char) && !inSelector && bracketDepth === 0 && parenDepth === 0) {
|
|
9486
|
-
if (current) {
|
|
9487
|
-
tokens.push(current);
|
|
9488
|
-
current = '';
|
|
9489
|
-
}
|
|
9490
|
-
}
|
|
9491
|
-
else {
|
|
9492
|
-
current += char;
|
|
9493
|
-
}
|
|
9494
|
-
}
|
|
9495
|
-
if (current) {
|
|
9496
|
-
tokens.push(current);
|
|
9497
|
-
}
|
|
9498
|
-
// Post-process to split attached affixes for languages that need it
|
|
9499
|
-
return splitAttachedAffixes(tokens, profile.code);
|
|
9500
|
-
}
|
|
9501
|
-
/**
|
|
9502
|
-
* Identify what type of statement this is
|
|
9503
|
-
*/
|
|
9504
|
-
function identifyStatementType(tokens, profile) {
|
|
9505
|
-
if (tokens.length === 0)
|
|
9506
|
-
return 'unknown';
|
|
9507
|
-
const firstToken = tokens[0].toLowerCase();
|
|
9508
|
-
// Check for event handler
|
|
9509
|
-
const eventMarker = profile.markers.find(m => m.role === 'event' && m.position === 'preposition');
|
|
9510
|
-
if (eventMarker && firstToken === eventMarker.form.toLowerCase()) {
|
|
9511
|
-
return 'event-handler';
|
|
9512
|
-
}
|
|
9513
|
-
// Check if first token is a known event keyword (derived from profiles)
|
|
9514
|
-
if (EVENT_KEYWORDS.has(firstToken)) {
|
|
9515
|
-
return 'event-handler';
|
|
9516
|
-
}
|
|
9517
|
-
// Check for conditional using shared constants
|
|
9518
|
-
if (CONDITIONAL_KEYWORDS.has(firstToken)) {
|
|
9519
|
-
return 'conditional';
|
|
9520
|
-
}
|
|
9521
|
-
return 'command';
|
|
9522
|
-
}
|
|
9523
|
-
/**
|
|
9524
|
-
* Parse an event handler statement
|
|
9525
|
-
* Pattern: on {event} {command} {target?} {modifiers?}
|
|
9526
|
-
*
|
|
9527
|
-
* Now handles modifiers like "by 3" in "on click increment #count by 3"
|
|
9528
|
-
*/
|
|
9529
|
-
function parseEventHandler(tokens, profile) {
|
|
9530
|
-
const roles = new Map();
|
|
9531
|
-
// Skip the event keyword (e.g., 'on', '\u3067', '\u5f53', etc.) - derived from profiles
|
|
9532
|
-
let startIndex = EVENT_KEYWORDS.has(tokens[0]?.toLowerCase()) ? 1 : 0;
|
|
9533
|
-
// Next token is the event
|
|
9534
|
-
if (tokens[startIndex]) {
|
|
9535
|
-
const eventTokens = [tokens[startIndex]];
|
|
9536
|
-
startIndex++;
|
|
9537
|
-
// Fold "or"-conjoined events into the event value (e.g.
|
|
9538
|
-
// "on click or keypress[...] toggle .active"). Without this, "or" would be
|
|
9539
|
-
// read as the action verb and the second event swept into role values,
|
|
9540
|
-
// hoisting "or <event2>" ahead of the command on reorder. Keeping the whole
|
|
9541
|
-
// "<event1> or <event2>" string in the event role lets it translate and
|
|
9542
|
-
// reorder as a single event clause. Only consumes "or" that immediately
|
|
9543
|
-
// follows the event, before any source/action \u2014 so a later "or" in a guard
|
|
9544
|
-
// or value is untouched.
|
|
9545
|
-
while (tokens[startIndex] &&
|
|
9546
|
-
EVENT_CONJUNCTIONS.has(tokens[startIndex].toLowerCase()) &&
|
|
9547
|
-
tokens[startIndex + 1]) {
|
|
9548
|
-
eventTokens.push(tokens[startIndex], tokens[startIndex + 1]);
|
|
9549
|
-
startIndex += 2;
|
|
9550
|
-
}
|
|
9551
|
-
roles.set('event', {
|
|
9552
|
-
role: 'event',
|
|
9553
|
-
value: eventTokens.join(' '),
|
|
9554
|
-
});
|
|
9555
|
-
}
|
|
9556
|
-
// Check for event source modifier before the action (e.g., "from #source" in "on input from #firstName set ...")
|
|
9557
|
-
// Only handle 'from' (event source) \u2014 other modifiers like circumfix markers (Chinese \u65f6) should not be consumed.
|
|
9558
|
-
if (tokens[startIndex] && tokens[startIndex].toLowerCase() === 'from' && tokens[startIndex + 1]) {
|
|
9559
|
-
startIndex++; // skip 'from'
|
|
9560
|
-
// Collect source value tokens until a command keyword is found
|
|
9561
|
-
const sourceValue = [];
|
|
9562
|
-
while (tokens[startIndex]) {
|
|
9563
|
-
if (ENGLISH_COMMANDS.has(tokens[startIndex].toLowerCase()))
|
|
9564
|
-
break;
|
|
9565
|
-
sourceValue.push(tokens[startIndex]);
|
|
9566
|
-
startIndex++;
|
|
9567
|
-
}
|
|
9568
|
-
if (sourceValue.length > 0) {
|
|
9569
|
-
const value = sourceValue.join(' ');
|
|
9570
|
-
roles.set('source', {
|
|
9571
|
-
role: 'source',
|
|
9572
|
-
value,
|
|
9573
|
-
isSelector: /^[#.<@]/.test(value),
|
|
9574
|
-
});
|
|
9575
|
-
}
|
|
9576
|
-
}
|
|
9577
|
-
// Next token is the action (command verb)
|
|
9578
|
-
if (tokens[startIndex]) {
|
|
9579
|
-
roles.set('action', {
|
|
9580
|
-
role: 'action',
|
|
9581
|
-
value: tokens[startIndex],
|
|
9582
|
-
});
|
|
9583
|
-
startIndex++;
|
|
9584
|
-
}
|
|
9585
|
-
// Parse remaining tokens with modifier awareness (like parseCommand does)
|
|
9586
|
-
// This handles "by 3" in "on click increment #count by 3".
|
|
9587
|
-
// Argument map: for DOM-target commands the head event is already captured, so
|
|
9588
|
-
// a locative `on` (`toggle @hidden on #panel`) maps to `destination` rather
|
|
9589
|
-
// than overwriting the head `event`.
|
|
9590
|
-
if (tokens[startIndex]) {
|
|
9591
|
-
const modifierMap = buildArgumentModifierMap(profile, roles.get('action')?.value);
|
|
9592
|
-
let currentRole = 'patient';
|
|
9593
|
-
let currentValue = [];
|
|
9594
|
-
for (let i = startIndex; i < tokens.length; i++) {
|
|
9595
|
-
const token = tokens[i];
|
|
9596
|
-
const mappedRole = modifierMap[token.toLowerCase()];
|
|
9597
|
-
if (mappedRole) {
|
|
9598
|
-
// Save previous role
|
|
9599
|
-
if (currentValue.length > 0) {
|
|
9600
|
-
const value = currentValue.join(' ');
|
|
9601
|
-
roles.set(currentRole, {
|
|
9602
|
-
role: currentRole,
|
|
9603
|
-
value,
|
|
9604
|
-
isSelector: /^[#.<@]/.test(value),
|
|
9605
|
-
});
|
|
9606
|
-
}
|
|
9607
|
-
currentRole = mappedRole;
|
|
9608
|
-
currentValue = [];
|
|
9609
|
-
}
|
|
9610
|
-
else {
|
|
9611
|
-
currentValue.push(token);
|
|
9612
|
-
}
|
|
9613
|
-
}
|
|
9614
|
-
// Save final role
|
|
9615
|
-
if (currentValue.length > 0) {
|
|
9616
|
-
const value = currentValue.join(' ');
|
|
9617
|
-
roles.set(currentRole, {
|
|
9618
|
-
role: currentRole,
|
|
9619
|
-
value,
|
|
9620
|
-
isSelector: /^[#.<@]/.test(value),
|
|
9621
|
-
});
|
|
9622
|
-
}
|
|
9623
|
-
}
|
|
9624
|
-
return {
|
|
9625
|
-
type: 'event-handler',
|
|
9626
|
-
roles,
|
|
9627
|
-
original: tokens.join(' '),
|
|
9628
|
-
};
|
|
9629
|
-
}
|
|
9630
|
-
/**
|
|
9631
|
-
* Parse a command statement
|
|
9632
|
-
* Pattern: {command} {args...}
|
|
9633
|
-
*/
|
|
9634
|
-
function parseCommand(tokens, profile) {
|
|
9635
|
-
const roles = new Map();
|
|
9636
|
-
if (tokens.length === 0) {
|
|
9637
|
-
return { type: 'command', roles, original: '' };
|
|
9638
|
-
}
|
|
9639
|
-
// First token is the command
|
|
9640
|
-
roles.set('action', {
|
|
9641
|
-
role: 'action',
|
|
9642
|
-
value: tokens[0],
|
|
9643
|
-
});
|
|
9644
|
-
// Generate dynamic modifier map from language profile (argument position).
|
|
9645
|
-
// This enables parsing non-English input (e.g., Japanese \u306b, Korean \uc5d0, Arabic \u0625\u0644\u0649).
|
|
9646
|
-
// For a DOM-target command a locative `on` (`toggle .active on me`) maps to
|
|
9647
|
-
// `destination` (SVO sources) rather than spawning a bogus `event` role.
|
|
9648
|
-
const modifierMap = buildArgumentModifierMap(profile, tokens[0]);
|
|
9649
|
-
let currentRole = 'patient';
|
|
9650
|
-
let currentValue = [];
|
|
9651
|
-
for (let i = 1; i < tokens.length; i++) {
|
|
9652
|
-
const token = tokens[i];
|
|
9653
|
-
const mappedRole = modifierMap[token.toLowerCase()];
|
|
9654
|
-
if (mappedRole) {
|
|
9655
|
-
// Save previous role
|
|
9656
|
-
if (currentValue.length > 0) {
|
|
9657
|
-
const value = currentValue.join(' ');
|
|
9658
|
-
roles.set(currentRole, {
|
|
9659
|
-
role: currentRole,
|
|
9660
|
-
value,
|
|
9661
|
-
isSelector: /^[#.<@]/.test(value),
|
|
9662
|
-
});
|
|
9663
|
-
}
|
|
9664
|
-
currentRole = mappedRole;
|
|
9665
|
-
currentValue = [];
|
|
9666
|
-
}
|
|
9667
|
-
else {
|
|
9668
|
-
currentValue.push(token);
|
|
9669
|
-
}
|
|
9670
|
-
}
|
|
9671
|
-
// Save final role
|
|
9672
|
-
if (currentValue.length > 0) {
|
|
9673
|
-
const value = currentValue.join(' ');
|
|
9674
|
-
roles.set(currentRole, {
|
|
9675
|
-
role: currentRole,
|
|
9676
|
-
value,
|
|
9677
|
-
isSelector: /^[#.<@]/.test(value),
|
|
9678
|
-
});
|
|
9679
|
-
}
|
|
9680
|
-
return {
|
|
9681
|
-
type: 'command',
|
|
9682
|
-
roles,
|
|
9683
|
-
original: tokens.join(' '),
|
|
9684
|
-
};
|
|
9685
|
-
}
|
|
9686
|
-
/**
|
|
9687
|
-
* Re-assign a command's mis-marked primary argument from the default `patient`
|
|
9688
|
-
* role to its true primary role (e.g. `wait`'s leading argument is a `duration`,
|
|
9689
|
-
* not a fronted object).
|
|
9690
|
-
*
|
|
9691
|
-
* The generic argument parser in `parseCommand` / `parseEventHandler` defaults the
|
|
9692
|
-
* first unmarked argument to `patient`. For most commands that is correct, but for a
|
|
9693
|
-
* command whose primary role is a non-patient *and which the target language does not
|
|
9694
|
-
* give a marker* (a duration, a measure \u2014 never a BA/object construction), the
|
|
9695
|
-
* `patient` assignment makes `insertMarkers` emit a spurious object-marker (Chinese
|
|
9696
|
-
* `\u628a`, Japanese `\u3092`, Korean `\ub97c`). The marked form is ungrammatical and the semantic
|
|
9697
|
-
* parser fails to match it, dropping the command.
|
|
9698
|
-
*
|
|
9699
|
-
* The fix is deliberately scoped to **literal/measure** primary roles
|
|
9700
|
-
* ({@link LITERAL_PRIMARY_ROLES}: `duration`, `quantity`) \u2014 values that are *never*
|
|
9701
|
-
* the object of an object-marking construction in any supported language, so moving
|
|
9702
|
-
* them off `patient` can only ever *remove* a spurious marker. Marker-bearing
|
|
9703
|
-
* primaries are intentionally left alone:
|
|
9704
|
-
* - `set`\u2192destination `\u5230`, `fetch`\u2192source `\u4ece` carry a *correct* marker; touching
|
|
9705
|
-
* them risks no benefit.
|
|
9706
|
-
* - `send`/`trigger`\u2192event: in a language without an event marker (e.g. Korean has
|
|
9707
|
-
* no event particle) un-marking the leading argument makes the semantic parser
|
|
9708
|
-
* mis-read it as a bare event handler and emit a phantom `on` action \u2014 an
|
|
9709
|
-
* over-generation that recall-based fidelity would not catch. So `event`-primary
|
|
9710
|
-
* commands stay on the `patient` default.
|
|
9711
|
-
*
|
|
9712
|
-
* Roles are still reordered by the safety-net in `reorderRoles`, so no value is lost.
|
|
9713
|
-
* A belt-and-suspenders target-marker guard keeps the change inert should a profile
|
|
9714
|
-
* ever add a marker for one of these literal roles.
|
|
9715
|
-
*
|
|
9716
|
-
* Must run *before* `translateElements`, while the `action` value is still the
|
|
9717
|
-
* source-language (English) keyword, so the schema lookup resolves.
|
|
9718
|
-
*
|
|
9719
|
-
* @see docs-internal/ZH_BLOCK_BODY_SCOPE.md (#1 \u2014 transformer role model)
|
|
9720
|
-
*/
|
|
9721
|
-
const LITERAL_PRIMARY_ROLES = new Set([
|
|
9722
|
-
'duration',
|
|
9723
|
-
'quantity',
|
|
9724
|
-
]);
|
|
9725
|
-
function applyPrimaryRole(parsed, targetProfile) {
|
|
9726
|
-
// Only re-mark standalone command statements. In an event handler (`on click
|
|
9727
|
-
// wait 2s \u2026`) a verb-final SOV language without an event particle (e.g. Korean)
|
|
9728
|
-
// relies on the leading argument's patient marker as the structural cue that
|
|
9729
|
-
// anchors the handler; un-marking it makes the semantic parser lose the event.
|
|
9730
|
-
// The block-body / then-chain `wait` clauses this fix targets are each parsed as
|
|
9731
|
-
// their own command statement, so they are still covered.
|
|
9732
|
-
if (parsed.type !== 'command')
|
|
9733
|
-
return;
|
|
9734
|
-
const action = parsed.roles.get('action')?.value;
|
|
9735
|
-
if (!action)
|
|
9736
|
-
return;
|
|
9737
|
-
const primaryRole = COMMAND_PRIMARY_ROLES[action.toLowerCase()];
|
|
9738
|
-
if (!primaryRole || !LITERAL_PRIMARY_ROLES.has(primaryRole))
|
|
9739
|
-
return;
|
|
9740
|
-
// Only act on a leading argument the generic parser defaulted to `patient`,
|
|
9741
|
-
// and only when the primary slot isn't already filled by an explicit marker.
|
|
9742
|
-
const patientEl = parsed.roles.get('patient');
|
|
9743
|
-
if (!patientEl || parsed.roles.has(primaryRole))
|
|
9744
|
-
return;
|
|
9745
|
-
// Guard: never introduce a marker that wasn't there. If the target language
|
|
9746
|
-
// marks the primary role, leave the command as-is.
|
|
9747
|
-
if (targetProfile.markers.some(m => m.role === primaryRole))
|
|
9748
|
-
return;
|
|
9749
|
-
parsed.roles.delete('patient');
|
|
9750
|
-
parsed.roles.set(primaryRole, { ...patientEl, role: primaryRole });
|
|
9751
|
-
}
|
|
9752
|
-
/**
|
|
9753
|
-
* Parse a conditional statement
|
|
9754
|
-
*/
|
|
9755
|
-
function parseConditional(tokens, _profile) {
|
|
9756
|
-
const roles = new Map();
|
|
9757
|
-
// First token is the 'if' keyword
|
|
9758
|
-
roles.set('action', {
|
|
9759
|
-
role: 'action',
|
|
9760
|
-
value: tokens[0],
|
|
9761
|
-
});
|
|
9762
|
-
// Find 'then' to split condition from body - using shared constants
|
|
9763
|
-
const thenIndex = tokens.findIndex(t => THEN_KEYWORDS.has(t.toLowerCase()));
|
|
9764
|
-
if (thenIndex > 1) {
|
|
9765
|
-
const conditionValue = tokens.slice(1, thenIndex).join(' ');
|
|
9766
|
-
roles.set('condition', {
|
|
9767
|
-
role: 'condition',
|
|
9768
|
-
value: conditionValue,
|
|
9769
|
-
});
|
|
9770
|
-
}
|
|
9771
|
-
else if (thenIndex === -1 && tokens.length > 1) {
|
|
9772
|
-
// Block-style `if <cond>` with the body on following lines (no inline
|
|
9773
|
-
// `then`). Capture everything after `if` as the condition; otherwise the
|
|
9774
|
-
// condition is silently dropped and the rendered block becomes a bare
|
|
9775
|
-
// `if`/`\u05d0\u05dd`/`\u5982\u679c`, which the semantic block parser then rejects (null
|
|
9776
|
-
// parse). This is the dominant failure for nested control-flow bodies in
|
|
9777
|
-
// non-Latin languages (he, zh).
|
|
9778
|
-
roles.set('condition', {
|
|
9779
|
-
role: 'condition',
|
|
9780
|
-
value: tokens.slice(1).join(' '),
|
|
9781
|
-
});
|
|
9782
|
-
}
|
|
9783
|
-
return {
|
|
9784
|
-
type: 'conditional',
|
|
9785
|
-
roles,
|
|
9786
|
-
original: tokens.join(' '),
|
|
9787
|
-
};
|
|
9788
|
-
}
|
|
9789
|
-
// =============================================================================
|
|
9790
|
-
// Translation
|
|
9791
|
-
// =============================================================================
|
|
9792
|
-
/**
|
|
9793
|
-
* Translate words using dictionary with type-safe access.
|
|
9794
|
-
*/
|
|
9795
|
-
function translateWord(word, sourceLocale, targetLocale) {
|
|
9796
|
-
// Don't translate CSS selectors
|
|
9797
|
-
if (/^[#.<@]/.test(word)) {
|
|
9798
|
-
return word;
|
|
9799
|
-
}
|
|
9800
|
-
// Don't translate numbers
|
|
9801
|
-
if (/^\d+/.test(word)) {
|
|
9802
|
-
return word;
|
|
9803
|
-
}
|
|
9804
|
-
// A whole parenthesized group fused by the tokenizer (`(the value of #price
|
|
9805
|
-
// as Number)`) reaching a single-token translate path: translate its
|
|
9806
|
-
// interior word-by-word IN ORDER \u2014 never reordered, never re-segmented.
|
|
9807
|
-
if (/\s/.test(word) && word.startsWith('(')) {
|
|
9808
|
-
return word
|
|
9809
|
-
.split(/\s+/)
|
|
9810
|
-
.map(w => translateWord(w, sourceLocale, targetLocale))
|
|
9811
|
-
.join(' ');
|
|
9812
|
-
}
|
|
9813
|
-
// A word carrying paren punctuation from a fused group after whitespace
|
|
9814
|
-
// re-splitting (`(my` / `valor)` / `(($x`): strip the parens for the
|
|
9815
|
-
// dictionary lookup and re-attach, so interior keywords still translate.
|
|
9816
|
-
if (word.length > 1 && (word.startsWith('(') || word.endsWith(')'))) {
|
|
9817
|
-
const m = word.match(/^(\(*)([^()]+)(\)*)$/);
|
|
9818
|
-
if (m && (m[1] || m[3])) {
|
|
9819
|
-
return m[1] + translateWord(m[2], sourceLocale, targetLocale) + m[3];
|
|
9820
|
-
}
|
|
9821
|
-
}
|
|
9822
|
-
const sourceDict = sourceLocale === 'en' ? null : dictionaries[sourceLocale];
|
|
9823
|
-
const targetDict = dictionaries[targetLocale];
|
|
9824
|
-
if (!targetDict)
|
|
9825
|
-
return word;
|
|
9826
|
-
// If source is not English, first map to English using type-safe lookup
|
|
9827
|
-
let englishWord = word;
|
|
9828
|
-
if (sourceDict) {
|
|
9829
|
-
const found = findInDictionary(sourceDict, word);
|
|
9830
|
-
if (found) {
|
|
9831
|
-
englishWord = found.englishKey;
|
|
9832
|
-
}
|
|
9833
|
-
}
|
|
9834
|
-
// Now map English to target locale using type-safe lookup
|
|
9835
|
-
const translated = translateFromEnglish(targetDict, englishWord);
|
|
9836
|
-
return translated ?? word;
|
|
9837
|
-
}
|
|
9838
|
-
/**
|
|
9839
|
-
* Possessive markers for each language.
|
|
9840
|
-
* Used to transform "X's Y" patterns to target language structure.
|
|
9841
|
-
*/
|
|
9842
|
-
const POSSESSIVE_MARKERS = {
|
|
9843
|
-
en: { type: 'suffix', marker: "'s" },
|
|
9844
|
-
es: { type: 'preposition', marker: 'de' },
|
|
9845
|
-
pt: { type: 'preposition', marker: 'de' },
|
|
9846
|
-
fr: { type: 'preposition', marker: 'de' },
|
|
9847
|
-
de: { type: 'preposition', marker: 'von' },
|
|
9848
|
-
ja: { type: 'suffix', marker: '\u306e' },
|
|
9849
|
-
ko: { type: 'suffix', marker: '\uc758' },
|
|
9850
|
-
zh: { type: 'suffix', marker: '\u7684' },
|
|
9851
|
-
ar: { type: 'preposition', marker: '\u0644\u0640' },
|
|
9852
|
-
// Spaced genitive particle (not the glued `'\u0131n`), so the tokenizer can split
|
|
9853
|
-
// it off the selector \u2014 consistent with Turkish's other spaced case markers.
|
|
9854
|
-
tr: { type: 'particle', marker: '\u0131n' },
|
|
9855
|
-
id: { type: 'preposition', marker: 'dari' },
|
|
9856
|
-
// Latin-script genitive: must be a *spaced* particle (`#picker pa`), since a
|
|
9857
|
-
// glued `#pickerpa` can't be split from the selector by the tokenizer the
|
|
9858
|
-
// way a non-Latin suffix (\u306e/\uc758/\u09b0) can.
|
|
9859
|
-
qu: { type: 'particle', marker: 'pa' },
|
|
9860
|
-
// Bengali SOV postposition genitive, like ja/ko \u2014 a spaced suffix the
|
|
9861
|
-
// tokenizer splits off as a particle. Previously absent, so it fell back to
|
|
9862
|
-
// the English `'s` marker and its possessive property paths never parsed.
|
|
9863
|
-
// (Hindi `\u0915\u093e` is intentionally omitted: its `bind` lacks a verb-final
|
|
9864
|
-
// grammar rule, so fixing its possessive alone yields a wrong `on` parse \u2014
|
|
9865
|
-
// tracked as separate follow-up.)
|
|
9866
|
-
bn: { type: 'suffix', marker: '\u09b0' },
|
|
9867
|
-
sw: { type: 'preposition', marker: 'ya' },
|
|
9868
|
-
};
|
|
9869
|
-
/**
|
|
9870
|
-
* Transform possessive 's syntax to target language.
|
|
9871
|
-
*
|
|
9872
|
-
* Examples:
|
|
9873
|
-
* me's value \u2192 mi valor (Spanish - pronoun becomes possessive adjective)
|
|
9874
|
-
* #button's textContent \u2192 textContent de #button (Spanish - prepositional)
|
|
9875
|
-
* me's value \u2192 \u79c1\u306e\u5024 (Japanese - \u306e particle)
|
|
9876
|
-
*/
|
|
9877
|
-
function translatePossessive(token, sourceLocale, targetLocale) {
|
|
9878
|
-
// Check for 's possessive pattern
|
|
9879
|
-
const possessiveMatch = token.match(/^(.+)'s$/i);
|
|
9880
|
-
if (!possessiveMatch) {
|
|
9881
|
-
return token;
|
|
9882
|
-
}
|
|
9883
|
-
const owner = possessiveMatch[1];
|
|
9884
|
-
const targetMarker = POSSESSIVE_MARKERS[targetLocale] || POSSESSIVE_MARKERS.en;
|
|
9885
|
-
// Check if owner is a pronoun that has a possessive form
|
|
9886
|
-
const pronounPossessives = {
|
|
9887
|
-
me: 'my',
|
|
9888
|
-
it: 'its',
|
|
9889
|
-
you: 'your',
|
|
9890
|
-
};
|
|
9891
|
-
const lowerOwner = owner.toLowerCase();
|
|
9892
|
-
if (pronounPossessives[lowerOwner]) {
|
|
9893
|
-
// Convert "me's" to "my" then translate
|
|
9894
|
-
const possessiveForm = pronounPossessives[lowerOwner];
|
|
9895
|
-
return translateWord(possessiveForm, 'en', targetLocale);
|
|
9896
|
-
}
|
|
9897
|
-
// For selectors and other owners, translate owner and apply target possessive marker
|
|
9898
|
-
const translatedOwner = translateWord(owner, sourceLocale, targetLocale);
|
|
9899
|
-
switch (targetMarker.type) {
|
|
9900
|
-
case 'suffix':
|
|
9901
|
-
// Japanese/Korean/Chinese: owner + marker (e.g., #button\u306e, #button\uc758)
|
|
9902
|
-
return `${translatedOwner}${targetMarker.marker}`;
|
|
9903
|
-
case 'particle':
|
|
9904
|
-
// Latin-script spaced genitive (Quechua `pa`): owner + space + marker
|
|
9905
|
-
// so the tokenizer can separate it from the selector.
|
|
9906
|
-
return `${translatedOwner} ${targetMarker.marker}`;
|
|
9907
|
-
case 'preposition':
|
|
9908
|
-
// Will be handled by caller - return marker + owner format
|
|
9909
|
-
// Store as special format to be processed later
|
|
9910
|
-
return `__POSS__${targetMarker.marker}__${translatedOwner}__POSS__`;
|
|
9911
|
-
default:
|
|
9912
|
-
return `${translatedOwner}'s`;
|
|
9913
|
-
}
|
|
9914
|
-
}
|
|
9915
|
-
// =============================================================================
|
|
9916
|
-
// Possessive Dot Notation
|
|
9917
|
-
// =============================================================================
|
|
9918
|
-
/**
|
|
9919
|
-
* Regex to match possessive dot notation patterns.
|
|
9920
|
-
* Matches: my.prop, its.prop, your.prop, me.prop, it.prop, you.prop
|
|
9921
|
-
* Also matches optional chaining: my?.prop, me?.prop, etc.
|
|
9922
|
-
*/
|
|
9923
|
-
const POSSESSIVE_DOT_REGEX = /^(my|its|your|me|it|you)(\??\..+)$/i;
|
|
9924
|
-
/**
|
|
9925
|
-
* Map pronoun forms to possessive adjective forms for dictionary lookup.
|
|
9926
|
-
*/
|
|
9927
|
-
const POSSESSIVE_DOT_PRONOUNS = {
|
|
9928
|
-
me: 'my',
|
|
9929
|
-
it: 'its',
|
|
9930
|
-
you: 'your',
|
|
9931
|
-
my: 'my',
|
|
9932
|
-
its: 'its',
|
|
9933
|
-
your: 'your',
|
|
9934
|
-
};
|
|
9935
|
-
/**
|
|
9936
|
-
* Translate possessive dot notation like my.textContent \u2192 mi.textContent.
|
|
9937
|
-
* Handles both possessive adjective forms (my, its, your) and pronoun forms (me, it, you).
|
|
9938
|
-
* Also handles optional chaining (my?.prop).
|
|
9939
|
-
* Returns null if the value doesn't match or no translation is available.
|
|
9940
|
-
*/
|
|
9941
|
-
function translatePossessiveDotNotation(value, sourceLocale, targetLocale) {
|
|
9942
|
-
const match = value.match(POSSESSIVE_DOT_REGEX);
|
|
9943
|
-
if (!match)
|
|
9944
|
-
return null;
|
|
9945
|
-
const possessiveWord = match[1].toLowerCase();
|
|
9946
|
-
const propertySuffix = match[2]; // ".textContent" or "?.textContent"
|
|
9947
|
-
// Normalize to possessive adjective form for dictionary lookup
|
|
9948
|
-
const possessiveKey = POSSESSIVE_DOT_PRONOUNS[possessiveWord] || possessiveWord;
|
|
9949
|
-
const translated = translateWord(possessiveKey, sourceLocale, targetLocale);
|
|
9950
|
-
// Skip if translation is multi-word (can't prefix dot notation)
|
|
9951
|
-
if (translated.includes(' '))
|
|
9952
|
-
return null;
|
|
9953
|
-
if (translated !== possessiveKey) {
|
|
9954
|
-
return translated + propertySuffix;
|
|
9955
|
-
}
|
|
9956
|
-
// Try original pronoun form if different from possessive key
|
|
9957
|
-
if (possessiveWord !== possessiveKey) {
|
|
9958
|
-
const alt = translateWord(possessiveWord, sourceLocale, targetLocale);
|
|
9959
|
-
if (alt !== possessiveWord && !alt.includes(' ')) {
|
|
9960
|
-
return alt + propertySuffix;
|
|
9961
|
-
}
|
|
9962
|
-
}
|
|
9963
|
-
return null;
|
|
9964
|
-
}
|
|
9965
|
-
// =============================================================================
|
|
9966
|
-
// Multi-Word Value Translation
|
|
9967
|
-
// =============================================================================
|
|
9968
|
-
/**
|
|
9969
|
-
* Translate a multi-word value, translating each word individually.
|
|
9970
|
-
* Handles possessives like "my value" \u2192 "mi valor" in Spanish.
|
|
9971
|
-
* Also handles 's possessive syntax like "me's value" \u2192 "mi valor".
|
|
9972
|
-
* Also handles possessive dot notation like "my.textContent" \u2192 "mi.textContent".
|
|
9973
|
-
*/
|
|
9974
|
-
function translateMultiWordValue(value, sourceLocale, targetLocale) {
|
|
9975
|
-
// Mask event-guard / attribute brackets (`[key is 'Escape']`): their contents
|
|
9976
|
-
// are expression syntax, not translatable keywords \u2014 translating `is` -> `ni`
|
|
9977
|
-
// etc. inside them breaks the guard. Restore verbatim after translation.
|
|
9978
|
-
if (value.includes('[')) {
|
|
9979
|
-
const guards = [];
|
|
9980
|
-
const masked = value.replace(/\[[^\]]*\]/g, match => {
|
|
9981
|
-
guards.push(match);
|
|
9982
|
-
return `\ue000${guards.length - 1}\ue001`;
|
|
9983
|
-
});
|
|
9984
|
-
if (guards.length > 0) {
|
|
9985
|
-
const translated = translateMultiWordValue(masked, sourceLocale, targetLocale);
|
|
9986
|
-
return translated.replace(/\ue000(\d+)\ue001/g, (_, n) => guards[Number(n)]);
|
|
9987
|
-
}
|
|
9988
|
-
}
|
|
9989
|
-
// If it's a single word, check for possessive then translate
|
|
9990
|
-
if (!value.includes(' ')) {
|
|
9991
|
-
// Check for possessive 's
|
|
9992
|
-
if (value.includes("'s")) {
|
|
9993
|
-
return translatePossessive(value, sourceLocale, targetLocale);
|
|
9994
|
-
}
|
|
9995
|
-
// Check for possessive dot notation (my.prop, its.prop, me.prop, etc.)
|
|
9996
|
-
const dotResult = translatePossessiveDotNotation(value, sourceLocale, targetLocale);
|
|
9997
|
-
if (dotResult !== null)
|
|
9998
|
-
return dotResult;
|
|
9999
|
-
return translateWord(value, sourceLocale, targetLocale);
|
|
10000
|
-
}
|
|
10001
|
-
// Split into words and translate each
|
|
10002
|
-
const words = value.split(/\s+/);
|
|
10003
|
-
const translated = [];
|
|
10004
|
-
let i = 0;
|
|
10005
|
-
while (i < words.length) {
|
|
10006
|
-
const word = words[i];
|
|
10007
|
-
// Check for possessive 's pattern FIRST (e.g., "me's value", "#button's textContent")
|
|
10008
|
-
// This must come before selector check because "#button's" starts with #
|
|
10009
|
-
if (word.includes("'s")) {
|
|
10010
|
-
const possessiveResult = translatePossessive(word, sourceLocale, targetLocale);
|
|
10011
|
-
// Check if it's a prepositional possessive that needs reordering
|
|
10012
|
-
const prepMatch = possessiveResult.match(/^__POSS__(.+)__(.+)__POSS__$/);
|
|
10013
|
-
if (prepMatch && i + 1 < words.length) {
|
|
10014
|
-
// Prepositional: "X's Y" \u2192 "Y marker X" (e.g., "textContent de #button")
|
|
10015
|
-
const marker = prepMatch[1];
|
|
10016
|
-
const owner = prepMatch[2];
|
|
10017
|
-
const property = words[i + 1];
|
|
10018
|
-
const translatedProperty = translateWord(property, sourceLocale, targetLocale);
|
|
10019
|
-
translated.push(`${translatedProperty} ${marker} ${owner}`);
|
|
10020
|
-
i += 2; // Skip property since we consumed it
|
|
10021
|
-
continue;
|
|
10022
|
-
}
|
|
10023
|
-
else if (prepMatch) {
|
|
10024
|
-
// No property following - just output owner with marker prefix
|
|
10025
|
-
const marker = prepMatch[1];
|
|
10026
|
-
const owner = prepMatch[2];
|
|
10027
|
-
translated.push(`${marker} ${owner}`);
|
|
10028
|
-
i++;
|
|
10029
|
-
continue;
|
|
10030
|
-
}
|
|
10031
|
-
// Suffix-style possessive (Japanese, Korean, etc.) or pronoun
|
|
10032
|
-
translated.push(possessiveResult);
|
|
10033
|
-
i++;
|
|
10034
|
-
continue;
|
|
10035
|
-
}
|
|
10036
|
-
// Skip pure CSS selectors and numbers (but NOT possessives which were handled above)
|
|
10037
|
-
if (/^[#.<@]/.test(word) || /^\d+/.test(word)) {
|
|
10038
|
-
translated.push(word);
|
|
10039
|
-
i++;
|
|
10040
|
-
continue;
|
|
10041
|
-
}
|
|
10042
|
-
// Skip quoted strings
|
|
10043
|
-
if (/^["'].*["']$/.test(word)) {
|
|
10044
|
-
translated.push(word);
|
|
10045
|
-
i++;
|
|
10046
|
-
continue;
|
|
10047
|
-
}
|
|
10048
|
-
// Check for possessive dot notation (my.prop, its.prop, me.prop, etc.)
|
|
10049
|
-
const dotResult = translatePossessiveDotNotation(word, sourceLocale, targetLocale);
|
|
10050
|
-
if (dotResult !== null) {
|
|
10051
|
-
translated.push(dotResult);
|
|
10052
|
-
i++;
|
|
10053
|
-
continue;
|
|
10054
|
-
}
|
|
10055
|
-
translated.push(translateWord(word, sourceLocale, targetLocale));
|
|
10056
|
-
i++;
|
|
10057
|
-
}
|
|
10058
|
-
return translated.join(' ');
|
|
10059
|
-
}
|
|
10060
|
-
/**
|
|
10061
|
-
* Translate all elements in a parsed statement
|
|
10062
|
-
*/
|
|
10063
|
-
function translateElements(parsed, sourceLocale, targetLocale) {
|
|
10064
|
-
for (const [_role, element] of parsed.roles) {
|
|
10065
|
-
// Always process possessive 's syntax, even for selectors
|
|
10066
|
-
// E.g., "#button's textContent" should translate the possessive
|
|
10067
|
-
if (element.value.includes("'s")) {
|
|
10068
|
-
element.translated = translateMultiWordValue(element.value, sourceLocale, targetLocale);
|
|
10069
|
-
}
|
|
10070
|
-
else if (!element.isSelector && !element.isLiteral) {
|
|
10071
|
-
element.translated = translateMultiWordValue(element.value, sourceLocale, targetLocale);
|
|
10072
|
-
}
|
|
10073
|
-
else {
|
|
10074
|
-
element.translated = element.value;
|
|
10075
|
-
}
|
|
10076
|
-
}
|
|
10077
|
-
}
|
|
10078
|
-
// =============================================================================
|
|
10079
|
-
// Caret-scope masking (`^name on <selector>`)
|
|
10080
|
-
// =============================================================================
|
|
10081
|
-
/** Private-use sentinels bracketing a masked caret-scope index. */
|
|
10082
|
-
const CARET_SCOPE_OPEN = '\uE000';
|
|
10083
|
-
const CARET_SCOPE_CLOSE = '\uE001';
|
|
10084
|
-
/**
|
|
10085
|
-
* Match `^name on <selector>` and the scope's selector form (#id, .class,
|
|
10086
|
-
* <tag/>, [attr]). The `^name` is kept; only the ` on <selector>` scope is masked.
|
|
10087
|
-
*/
|
|
10088
|
-
const CARET_SCOPE_RE = /(\^[A-Za-z_][\w-]*)(\s+on\s+(?:[#.][\w-]+|<[^>]*\/>|\[[^\]]+\]))/g;
|
|
10089
|
-
/**
|
|
10090
|
-
* Mask the ` on <selector>` scope of every `^name on <selector>` read behind an
|
|
10091
|
-
* opaque token attached to `^name`, so the overloaded `on` doesn't reach the
|
|
10092
|
-
* splitter / event-handler parser. Returns null when there's nothing to mask.
|
|
10093
|
-
*/
|
|
10094
|
-
function maskCaretScopes(input) {
|
|
10095
|
-
const scopes = [];
|
|
10096
|
-
const masked = input.replace(CARET_SCOPE_RE, (_m, varTok, scope) => {
|
|
10097
|
-
const idx = scopes.length;
|
|
10098
|
-
scopes.push(scope);
|
|
10099
|
-
return `${varTok}${CARET_SCOPE_OPEN}${idx}${CARET_SCOPE_CLOSE}`;
|
|
10100
|
-
});
|
|
10101
|
-
return scopes.length > 0 ? { masked, scopes } : null;
|
|
10102
|
-
}
|
|
10103
|
-
/** Restore masked caret-scope tokens to their verbatim ` on <selector>` form. */
|
|
10104
|
-
function restoreCaretScopes(input, scopes) {
|
|
10105
|
-
return input.replace(new RegExp(`${CARET_SCOPE_OPEN}(\\d+)${CARET_SCOPE_CLOSE}`, 'g'), (_m, n) => scopes[Number(n)] ?? '');
|
|
10106
|
-
}
|
|
10107
|
-
// =============================================================================
|
|
10108
|
-
// View-transition tail masking (`using view transition`)
|
|
10109
|
-
// =============================================================================
|
|
10110
|
-
/** Private-use sentinels bracketing a masked view-transition-tail index. */
|
|
10111
|
-
const VIEW_TAIL_OPEN = '\uE002';
|
|
10112
|
-
const VIEW_TAIL_CLOSE = '\uE003';
|
|
10113
|
-
/**
|
|
10114
|
-
* Match `swap`/`process`'s trailing `using view transition` modifier.
|
|
10115
|
-
*
|
|
10116
|
-
* `using` is in no dictionary and in no role table, so it sweeps into whatever
|
|
10117
|
-
* role phrase is open; `transition` IS a translated command keyword in every
|
|
10118
|
-
* dictionary (es `transici\u00f3n`, de `\u00fcbergang`, ja `\u9077\u79fb`) and is in
|
|
10119
|
-
* `ENGLISH_COMMANDS`, so `splitOnCommandBoundaries` splits the clause there and
|
|
10120
|
-
* the rejoin plants a phantom translated `transition` COMMAND after the target's
|
|
10121
|
-
* `then`-connective (`intercambiar #a con #b using view entonces transici\u00f3n`).
|
|
10122
|
-
*
|
|
10123
|
-
* The phrase has no native form in any of the 24 languages \u2014 semantic's
|
|
10124
|
-
* `USING_VIEW_MARKER_ALL_LANGS` matches the literal English `using view` marker
|
|
10125
|
-
* everywhere \u2014 so it is a passthrough, matched on the English surface regardless
|
|
10126
|
-
* of source locale. The value word is optional so a bare `using view` still
|
|
10127
|
-
* masks rather than half-splitting; `then` is excluded so a clause boundary is
|
|
10128
|
-
* never swallowed into the tail.
|
|
10129
|
-
*/
|
|
10130
|
-
const VIEW_TAIL_RE = /\busing\s+view\b(?:\s+(?!then\b)[A-Za-z][\w-]*)?/gi;
|
|
10131
|
-
/** Whether a token is a masked view-transition tail. */
|
|
10132
|
-
const VIEW_TAIL_TOKEN_RE = new RegExp(`^${VIEW_TAIL_OPEN}(\\d+)${VIEW_TAIL_CLOSE}$`);
|
|
10133
|
-
/**
|
|
10134
|
-
* Mask every `using view <value>` tail behind an opaque token, so the phrase
|
|
10135
|
-
* never reaches the splitter, the word translator, or the role parser. Returns
|
|
10136
|
-
* null when there's nothing to mask.
|
|
10137
|
-
*/
|
|
10138
|
-
function maskViewTransitionTails(input) {
|
|
10139
|
-
const tails = [];
|
|
10140
|
-
const masked = input.replace(VIEW_TAIL_RE, match => {
|
|
10141
|
-
const idx = tails.length;
|
|
10142
|
-
tails.push(match);
|
|
10143
|
-
return `${VIEW_TAIL_OPEN}${idx}${VIEW_TAIL_CLOSE}`;
|
|
10144
|
-
});
|
|
10145
|
-
return tails.length > 0 ? { masked, tails } : null;
|
|
10146
|
-
}
|
|
10147
|
-
/** Restore masked view-transition tokens to their verbatim English phrase. */
|
|
10148
|
-
function restoreViewTransitionTails(input, tails) {
|
|
10149
|
-
return input.replace(new RegExp(`${VIEW_TAIL_OPEN}(\\d+)${VIEW_TAIL_CLOSE}`, 'g'), (_m, n) => tails[Number(n)] ?? '');
|
|
10150
|
-
}
|
|
10151
|
-
// =============================================================================
|
|
10152
|
-
// Main Transformer
|
|
10153
|
-
// =============================================================================
|
|
10154
|
-
class GrammarTransformer {
|
|
10155
|
-
constructor(sourceLocale = 'en', targetLocale) {
|
|
10156
|
-
const source = getProfile(sourceLocale);
|
|
10157
|
-
const target = getProfile(targetLocale);
|
|
10158
|
-
if (!source)
|
|
10159
|
-
throw new Error(`Unknown source locale: ${sourceLocale}`);
|
|
10160
|
-
if (!target)
|
|
10161
|
-
throw new Error(`Unknown target locale: ${targetLocale}`);
|
|
10162
|
-
this.sourceProfile = source;
|
|
10163
|
-
this.targetProfile = target;
|
|
10164
|
-
}
|
|
10165
|
-
/**
|
|
10166
|
-
* Transform a hyperscript statement from source to target language.
|
|
10167
|
-
* Handles compound statements with "then" by splitting, transforming each part,
|
|
10168
|
-
* and rejoining with the target language's "then" keyword.
|
|
10169
|
-
*
|
|
10170
|
-
* For multi-line input, preserves line structure (indentation, blank lines).
|
|
10171
|
-
*/
|
|
10172
|
-
transform(input) {
|
|
10173
|
-
const out = this.transformInternal(input);
|
|
10174
|
-
// Hebrew: repair a fronted accusative marker the body-split heuristics can emit
|
|
10175
|
-
// (`\u2026 \u05d0\u05ea \u05d4\u05d5\u05e1\u05e3 .x \u2026` \u2192 `\u2026 \u05d4\u05d5\u05e1\u05e3 \u05d0\u05ea .x \u2026`). Applied to the assembled output; idempotent
|
|
10176
|
-
// across the internal recursion. See repairHebrewFrontedAccusative.
|
|
10177
|
-
return this.targetProfile.code === 'he' ? repairHebrewFrontedAccusative(out) : out;
|
|
10178
|
-
}
|
|
10179
|
-
transformInternal(input) {
|
|
10180
|
-
// `using view transition` is a passthrough phrase, not translatable content:
|
|
10181
|
-
// `using` is in no dictionary and `transition` is a COMMAND keyword, so left
|
|
10182
|
-
// in place the splitter tears the clause apart there and the tail renders as
|
|
10183
|
-
// a phantom translated command. Mask it before any splitting/translation and
|
|
10184
|
-
// restore it verbatim; transformSingle re-appends the opaque token at the
|
|
10185
|
-
// clause tail so it lands after the reorder, not inside a role phrase.
|
|
10186
|
-
const viewTails = maskViewTransitionTails(input);
|
|
10187
|
-
if (viewTails) {
|
|
10188
|
-
return restoreViewTransitionTails(this.transformInternal(viewTails.masked), viewTails.tails);
|
|
10189
|
-
}
|
|
10190
|
-
// Caret-scoped variable reads (`^name on <selector>`) carry a second,
|
|
10191
|
-
// overloaded `on` that the splitter/event-parser would mistake for an event
|
|
10192
|
-
// or command boundary \u2014 mangling `put ^count on #host into me`. Mask the
|
|
10193
|
-
// ` on <selector>` scope behind an opaque token attached to `^name` so the
|
|
10194
|
-
// command reorders as if the patient were a single value, then restore it.
|
|
10195
|
-
// `on` is kept verbatim (the semantic caret-scope matcher accepts it by raw
|
|
10196
|
-
// value across languages \u2014 passthrough-alignment).
|
|
10197
|
-
const caret = maskCaretScopes(input);
|
|
10198
|
-
if (caret) {
|
|
10199
|
-
return restoreCaretScopes(this.transform(caret.masked), caret.scopes);
|
|
10200
|
-
}
|
|
10201
|
-
const targetThen = getTargetThenKeyword(this.targetProfile.code);
|
|
10202
|
-
// Inline JS blocks (`... js <raw js> end`) must be masked BEFORE any
|
|
10203
|
-
// splitting/reordering: the body is raw JavaScript, not hyperscript, so it
|
|
10204
|
-
// must never be tokenized, translated, or word-order reordered. (Single-line
|
|
10205
|
-
// only here; multi-line js bodies are handled with the behavior work.)
|
|
10206
|
-
if (!input.includes('\n')) {
|
|
10207
|
-
const jsBlock = this.tryTransformJsBlock(input);
|
|
10208
|
-
if (jsBlock !== null)
|
|
10209
|
-
return jsBlock;
|
|
10210
|
-
// Event handler whose body leads with a command-modifier
|
|
10211
|
-
// (`on click async fetch \u2026`, `on click once add \u2026`): lift the modifier out
|
|
10212
|
-
// so the real verb isn't mistaken for the action and the SOV reorder keeps
|
|
10213
|
-
// the body patient-first (recoverable by the parser's SOV event extraction).
|
|
10214
|
-
const eventModifier = this.tryTransformEventWithModifierBody(input);
|
|
10215
|
-
if (eventModifier !== null)
|
|
10216
|
-
return eventModifier;
|
|
10217
|
-
// Event handler whose body is a block (`on <event> if/repeat/unless \u2026 end`):
|
|
10218
|
-
// the block body must be reordered as a self-contained unit, not shredded
|
|
10219
|
-
// across the event handler's role soup.
|
|
10220
|
-
const eventBlock = this.tryTransformEventWithBlockBody(input);
|
|
10221
|
-
if (eventBlock !== null)
|
|
10222
|
-
return eventBlock;
|
|
10223
|
-
// Event handler whose body is an un-terminated inline `unless` guard
|
|
10224
|
-
// (`on click unless I match .disabled toggle .selected`): route the guard
|
|
10225
|
-
// through the standalone block path so Hebrew's accusative marker lands on
|
|
10226
|
-
// the body command, not the condition. Hebrew-only; null elsewhere.
|
|
10227
|
-
const eventGuard = this.tryTransformEventWithUnlessGuard(input);
|
|
10228
|
-
if (eventGuard !== null)
|
|
10229
|
-
return eventGuard;
|
|
10230
|
-
}
|
|
10231
|
-
// Check if input has multi-line structure worth preserving
|
|
10232
|
-
const hasMultiLineStructure = input.includes('\n');
|
|
10233
|
-
if (hasMultiLineStructure) {
|
|
10234
|
-
// Multi-line case - preserve structure (indentation, blank lines)
|
|
10235
|
-
const { parts, lineMetadata, partToLineIndex } = splitCompoundStatementWithMetadata(input, this.sourceProfile.code);
|
|
10236
|
-
const transformedParts = parts.map(part => this.transformSingle(part));
|
|
10237
|
-
return reconstructWithLineStructure(transformedParts, lineMetadata, partToLineIndex, targetThen);
|
|
10238
|
-
}
|
|
10239
|
-
// Single-line case - use existing logic
|
|
10240
|
-
const parts = splitCompoundStatement(input, this.sourceProfile.code);
|
|
10241
|
-
if (parts.length > 1) {
|
|
10242
|
-
const transformedParts = parts.map(part => this.transformSingle(part));
|
|
10243
|
-
return transformedParts.join(` ${targetThen} `);
|
|
10244
|
-
}
|
|
10245
|
-
// Single statement (no "then" splitting needed)
|
|
10246
|
-
return this.transformSingle(input);
|
|
10247
|
-
}
|
|
10248
|
-
/**
|
|
10249
|
-
* Transform a single hyperscript statement (no compound "then" chains).
|
|
10250
|
-
*/
|
|
10251
|
-
transformSingle(input) {
|
|
10252
|
-
// 0. Reactive block? Route around parseStatement entirely so
|
|
10253
|
-
// block-syntactic tokens (live/when/unless/end) aren't treated
|
|
10254
|
-
// as command verbs or swept into role values, and so SOV/VSO
|
|
10255
|
-
// reorder applies only inside the body.
|
|
10256
|
-
const block = extractBlockStructure(input, this.sourceProfile.code);
|
|
10257
|
-
if (block) {
|
|
10258
|
-
return this.transformBlock(block);
|
|
10259
|
-
}
|
|
10260
|
-
// 0a. Fragment carrying a stranded trailing block terminator (`wait 200ms
|
|
10261
|
-
// end`, `set x to y end` \u2014 what the `then`-splitter leaves from
|
|
10262
|
-
// `if \u2026 then <cmd> end`). `end` is not a marker, so left in place it is
|
|
10263
|
-
// swept into the open role's VALUE and rendered inside that phrase
|
|
10264
|
-
// (bn `200ms \u09b6\u09c7\u09b7 \u0995\u09c7 \u0985\u09aa\u09c7\u0995\u09cd\u09b7\u09be`). Strip it, transform the clause alone, and
|
|
10265
|
-
// re-append the translated terminator after the verb \u2014 the same tail
|
|
10266
|
-
// position transformBlockBody emits for event-headed blocks.
|
|
10267
|
-
const strippedEnd = this.transformWithTrailingEnd(input);
|
|
10268
|
-
if (strippedEnd !== null) {
|
|
10269
|
-
return strippedEnd;
|
|
10270
|
-
}
|
|
10271
|
-
// 0b. `set @attr to V on <scope>` (S1 tabs-aria): strip the trailing
|
|
10272
|
-
// `on <scope>`, transform the scope-less set normally, then re-insert
|
|
10273
|
-
// `on <scope>` where the semantic set patterns expect it.
|
|
10274
|
-
const setScope = this.transformSetWithScope(input);
|
|
10275
|
-
if (setScope !== null) {
|
|
10276
|
-
return setScope;
|
|
10277
|
-
}
|
|
10278
|
-
// 0c. Masked `using view transition` tail (see maskViewTransitionTails):
|
|
10279
|
-
// strip the opaque token, transform the clause without it, and re-append
|
|
10280
|
-
// it at the clause tail. Left in the token stream it would be swept into
|
|
10281
|
-
// the open role's VALUE and rendered ahead of that role's marker
|
|
10282
|
-
// (ja `#b using view transition \u3067` instead of `#b \u3067 using view
|
|
10283
|
-
// transition`) \u2014 the same failure mode transformWithTrailingEnd fixes
|
|
10284
|
-
// for a stranded `end`.
|
|
10285
|
-
const viewTail = this.transformWithViewTransitionTail(input);
|
|
10286
|
-
if (viewTail !== null) {
|
|
10287
|
-
return viewTail;
|
|
10288
|
-
}
|
|
10289
|
-
// 1. Parse into semantic roles
|
|
10290
|
-
const parsed = parseStatement(input, this.sourceProfile.code);
|
|
10291
|
-
if (!parsed) {
|
|
10292
|
-
return input; // Return unchanged if parsing fails
|
|
10293
|
-
}
|
|
10294
|
-
// 1b. Re-assign a mis-marked primary argument (e.g. `wait`'s duration) off the
|
|
10295
|
-
// default `patient` role so the target doesn't emit a spurious object-marker.
|
|
10296
|
-
// Runs before translation while `action` is still the English keyword.
|
|
10297
|
-
applyPrimaryRole(parsed, this.targetProfile);
|
|
10298
|
-
// 2. Translate individual words
|
|
10299
|
-
translateElements(parsed, this.sourceProfile.code, this.targetProfile.code);
|
|
10300
|
-
// 3. Find applicable rule
|
|
10301
|
-
const rule = this.findRule(parsed);
|
|
10302
|
-
// 4. Apply transformation
|
|
10303
|
-
if (rule?.transform.custom) {
|
|
10304
|
-
return rule.transform.custom(parsed, this.targetProfile);
|
|
10305
|
-
}
|
|
10306
|
-
// 5. Reorder according to target language's canonical order
|
|
10307
|
-
const roleOrder = rule?.transform.roleOrder || this.targetProfile.canonicalOrder;
|
|
10308
|
-
const reordered = reorderRoles(parsed.roles, roleOrder);
|
|
10309
|
-
// 6. Insert grammatical markers
|
|
10310
|
-
const shouldInsertMarkers = rule?.transform.insertMarkers ?? true;
|
|
10311
|
-
if (shouldInsertMarkers) {
|
|
10312
|
-
const result = insertMarkers(reordered, this.targetProfile.markers, this.targetProfile.adpositionType);
|
|
10313
|
-
// Use joinTokens for proper suffix/prefix attachment (Turkish -i, Quechua -ta, etc.)
|
|
10314
|
-
return joinTokens(result);
|
|
10315
|
-
}
|
|
10316
|
-
// 7. Join without markers (still use joinTokens for consistency)
|
|
10317
|
-
return joinTokens(reordered.map(e => e.translated || e.value));
|
|
10318
|
-
}
|
|
10319
|
-
/**
|
|
10320
|
-
* Clause carrying a masked `using view transition` tail: strip the opaque
|
|
10321
|
-
* token, transform the clause alone, and re-append the token at the very end.
|
|
10322
|
-
*
|
|
10323
|
-
* The tail is a clause-final modifier in every word order the corpus emits:
|
|
10324
|
-
* the semantic side matches it as the literal `using view` marker plus a value
|
|
10325
|
-
* word, and the SOV/VSO event-handler patterns admit it as an optional
|
|
10326
|
-
* TRAILING group (after the with-marked operand). So the target position is
|
|
10327
|
-
* "end of the transformed clause" for all 24 languages \u2014 no per-profile
|
|
10328
|
-
* placement decision, which is what makes this a passthrough rather than a
|
|
10329
|
-
* role.
|
|
10330
|
-
*
|
|
10331
|
-
* Returns null when the clause carries no masked tail, or when the token is
|
|
10332
|
-
* not clause-final (nothing to reposition \u2014 leaving it in place still restores
|
|
10333
|
-
* verbatim English).
|
|
10334
|
-
*/
|
|
10335
|
-
transformWithViewTransitionTail(input) {
|
|
10336
|
-
const trimmed = input.trim();
|
|
10337
|
-
const tokens = trimmed.split(/\s+/);
|
|
10338
|
-
if (tokens.length < 2) {
|
|
10339
|
-
return null;
|
|
10340
|
-
}
|
|
10341
|
-
if (!VIEW_TAIL_TOKEN_RE.test(tokens[tokens.length - 1])) {
|
|
10342
|
-
return null;
|
|
10343
|
-
}
|
|
10344
|
-
const tail = tokens[tokens.length - 1];
|
|
10345
|
-
const head = tokens.slice(0, -1).join(' ');
|
|
10346
|
-
return `${this.transformSingle(head)} ${tail}`;
|
|
10347
|
-
}
|
|
10348
|
-
/**
|
|
10349
|
-
* `<command \u2026> end` fragments: transform the command without its stranded
|
|
10350
|
-
* terminator, then re-append the translated terminator as a standalone
|
|
10351
|
-
* trailing token. Fragments that open a block of their own (`if \u2026 end`,
|
|
10352
|
-
* `repeat \u2026 end`, `js \u2026 end`) bail \u2014 their terminator belongs to them and
|
|
10353
|
-
* their dedicated paths handle it.
|
|
10354
|
-
*/
|
|
10355
|
-
transformWithTrailingEnd(input) {
|
|
10356
|
-
const src = this.sourceProfile.code;
|
|
10357
|
-
const tokens = input.trim().split(/\s+/);
|
|
10358
|
-
if (tokens.length < 2) {
|
|
10359
|
-
return null;
|
|
10360
|
-
}
|
|
10361
|
-
const sourceEnd = translateWord('end', 'en', src).toLowerCase();
|
|
10362
|
-
if (tokens[tokens.length - 1].toLowerCase() !== sourceEnd) {
|
|
10363
|
-
return null;
|
|
10364
|
-
}
|
|
10365
|
-
const openers = new Set(['if', 'repeat', 'unless', 'while', 'when', 'live', 'js'].map(k => translateWord(k, 'en', src).toLowerCase()));
|
|
10366
|
-
if (tokens.slice(0, -1).some(t => openers.has(t.toLowerCase()))) {
|
|
10367
|
-
return null;
|
|
10368
|
-
}
|
|
10369
|
-
const inner = this.transformSingle(tokens.slice(0, -1).join(' '));
|
|
10370
|
-
const endT = translateWord(tokens[tokens.length - 1], src, this.targetProfile.code);
|
|
10371
|
-
return `${inner} ${endT}`;
|
|
10372
|
-
}
|
|
10373
|
-
/**
|
|
10374
|
-
* Detect and transform an inline JS block (`[on <event>] js <raw js> end`).
|
|
10375
|
-
*
|
|
10376
|
-
* The `js ... end` body is raw JavaScript: it must not be tokenized,
|
|
10377
|
-
* translated, or word-order reordered. We mask the whole block with a single
|
|
10378
|
-
* opaque placeholder, run the surrounding statement (the event-handler head,
|
|
10379
|
-
* if any) through the normal reorder pipeline so the placeholder lands in the
|
|
10380
|
-
* correct action position, then substitute the translated `js`/`end` keywords
|
|
10381
|
-
* around the verbatim body.
|
|
10382
|
-
*
|
|
10383
|
-
* Returns `null` (fall through to the normal path) when there is no js block,
|
|
10384
|
-
* no matching `end`, or trailing content after `end` (kept tight on purpose).
|
|
10385
|
-
*/
|
|
10386
|
-
tryTransformJsBlock(input) {
|
|
10387
|
-
const src = this.sourceProfile.code;
|
|
10388
|
-
const dst = this.targetProfile.code;
|
|
10389
|
-
// The `js` / `end` keyword forms in the SOURCE language.
|
|
10390
|
-
const sourceJs = translateWord('js', 'en', src);
|
|
10391
|
-
const sourceEnd = translateWord('end', 'en', src).toLowerCase();
|
|
10392
|
-
const tokens = input.split(/\s+/).filter(t => t.length > 0);
|
|
10393
|
-
// The js command token, optionally with a `(locals)` suffix: `js`, `js(me)`.
|
|
10394
|
-
const escapedJs = sourceJs.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
10395
|
-
const jsRe = new RegExp(`^${escapedJs}(\\(.*\\))?$`, 'i');
|
|
10396
|
-
const jsIdx = tokens.findIndex(t => jsRe.test(t));
|
|
10397
|
-
if (jsIdx === -1)
|
|
10398
|
-
return null;
|
|
10399
|
-
// First `end` after the js keyword closes the block (raw JS never contains a
|
|
10400
|
-
// bare hyperscript `end` token).
|
|
10401
|
-
let endIdx = -1;
|
|
10402
|
-
for (let i = jsIdx + 1; i < tokens.length; i++) {
|
|
10403
|
-
if (tokens[i].toLowerCase() === sourceEnd) {
|
|
10404
|
-
endIdx = i;
|
|
10405
|
-
break;
|
|
10406
|
-
}
|
|
10407
|
-
}
|
|
10408
|
-
if (endIdx === -1)
|
|
10409
|
-
return null;
|
|
10410
|
-
// Trailing content after `end` \u2014 leave for the normal path.
|
|
10411
|
-
if (endIdx !== tokens.length - 1)
|
|
10412
|
-
return null;
|
|
10413
|
-
const jsToken = tokens[jsIdx];
|
|
10414
|
-
const jsParen = jsToken.match(jsRe)?.[1] ?? '';
|
|
10415
|
-
const jsKeywordRaw = jsParen ? jsToken.slice(0, jsToken.length - jsParen.length) : jsToken;
|
|
10416
|
-
const body = tokens.slice(jsIdx + 1, endIdx).join(' ');
|
|
10417
|
-
const targetJs = translateWord(jsKeywordRaw, src, dst) + jsParen;
|
|
10418
|
-
const targetEnd = translateWord(tokens[endIdx], src, dst);
|
|
10419
|
-
const replacement = [targetJs, body, targetEnd].filter(s => s.length > 0).join(' ');
|
|
10420
|
-
const before = tokens.slice(0, jsIdx);
|
|
10421
|
-
// Bare `js ... end` with no leading event-handler head: emit directly.
|
|
10422
|
-
if (before.length === 0)
|
|
10423
|
-
return replacement;
|
|
10424
|
-
// Mask the block as one opaque action token, reorder the surrounding
|
|
10425
|
-
// statement, then restore the verbatim block.
|
|
10426
|
-
const placeholder = 'JSBLOCKPLACEHOLDER';
|
|
10427
|
-
const reordered = this.transformSingle([...before, placeholder].join(' '));
|
|
10428
|
-
if (!reordered.includes(placeholder))
|
|
10429
|
-
return null; // unexpected \u2014 fall through
|
|
10430
|
-
return reordered.replace(placeholder, replacement);
|
|
10431
|
-
}
|
|
10432
|
-
/**
|
|
10433
|
-
* Transform an event handler whose body is a block command
|
|
10434
|
-
* (`on <event> [from <src>] {if|repeat|unless|while|for} \u2026 end`).
|
|
10435
|
-
*
|
|
10436
|
-
* `parseEventHandler` would treat the block keyword as the action and sweep the
|
|
10437
|
-
* condition/body into role values, then reorder them \u2014 shredding the block
|
|
10438
|
-
* (`if event.shiftKey call submitAndContinue() end` \u2192 scattered tokens). Instead
|
|
10439
|
-
* we mask the whole block as an opaque action placeholder, reorder the event
|
|
10440
|
-
* head normally, transform the block as a self-contained unit, and restitch.
|
|
10441
|
-
*
|
|
10442
|
-
* Returns `null` (fall through) when the input isn't an event handler, has no
|
|
10443
|
-
* block-keyword body, or has no closing `end`.
|
|
10444
|
-
*/
|
|
10445
|
-
tryTransformEventWithBlockBody(input) {
|
|
10446
|
-
const tokens = tokenize(input, this.sourceProfile);
|
|
10447
|
-
if (tokens.length === 0)
|
|
10448
|
-
return null;
|
|
10449
|
-
if (!EVENT_KEYWORDS.has(tokens[0]?.toLowerCase()))
|
|
10450
|
-
return null;
|
|
10451
|
-
let blockIdx = -1;
|
|
10452
|
-
for (let i = 1; i < tokens.length; i++) {
|
|
10453
|
-
if (BLOCK_BODY_KEYWORDS.has(tokens[i].toLowerCase())) {
|
|
10454
|
-
blockIdx = i;
|
|
10455
|
-
break;
|
|
10456
|
-
}
|
|
10457
|
-
}
|
|
10458
|
-
if (blockIdx <= 0)
|
|
10459
|
-
return null;
|
|
10460
|
-
// Only handle the explicitly-terminated form; un-terminated bodies
|
|
10461
|
-
// (`on click unless X toggle Y`) keep the existing path.
|
|
10462
|
-
if (tokens[tokens.length - 1].toLowerCase() !== 'end')
|
|
10463
|
-
return null;
|
|
10464
|
-
const eventHead = tokens.slice(0, blockIdx);
|
|
10465
|
-
const blockTokens = tokens.slice(blockIdx);
|
|
10466
|
-
// Event heads carrying a `from <source>` modifier (`on keydown[...] from
|
|
10467
|
-
// .modal if \u2026 end`): for SVO/SOV targets, route them through this block-body
|
|
10468
|
-
// path too \u2014 masking the block and emitting the event clause (incl. the
|
|
10469
|
-
// translated `from <source>`, which `transformSingle` \u2192 `parseEventHandler`
|
|
10470
|
-
// already reorders) first, then the transformed block. This keeps the if-block
|
|
10471
|
-
// body from being shredded across the event handler's roles and properly
|
|
10472
|
-
// translates its inner keywords (`in`/`focus`/`first`/positional); the old
|
|
10473
|
-
// exclusion left them English and mangled the order (this is what cleared
|
|
10474
|
-
// `focus-trap` in tr). VSO targets (ar, tl) are kept on the existing path:
|
|
10475
|
-
// there the event-first emission with a `from`-source reorders incorrectly and
|
|
10476
|
-
// regresses `focus-trap`/`window-keydown`, while the existing path already
|
|
10477
|
-
// parses them. So the `from`-source exclusion is scoped to VSO only.
|
|
10478
|
-
if (this.targetProfile.wordOrder === 'VSO' && eventHead.some(t => t.toLowerCase() === 'from')) {
|
|
10479
|
-
return null;
|
|
10480
|
-
}
|
|
10481
|
-
const placeholder = 'EVENTBLOCKPLACEHOLDER';
|
|
10482
|
-
const headOut = this.transformSingle([...eventHead, placeholder].join(' '));
|
|
10483
|
-
if (!headOut.includes(placeholder))
|
|
10484
|
-
return null;
|
|
10485
|
-
const blockOut = this.transformBlockBody(blockTokens);
|
|
10486
|
-
// Always emit the event clause first, then the block. An event handler's
|
|
10487
|
-
// event is a leading delimiter, and the semantic parser only matches a
|
|
10488
|
-
// block body when it follows the event \u2014 even in verb-first (VSO) languages
|
|
10489
|
-
// whose normal command order would push the event to the end. So strip the
|
|
10490
|
-
// placeholder out of the (possibly reordered) head and append the block,
|
|
10491
|
-
// rather than substituting in place.
|
|
10492
|
-
const eventClause = headOut.replace(placeholder, '').replace(/\s+/g, ' ').trim();
|
|
10493
|
-
return [eventClause, blockOut].filter(s => s.length > 0).join(' ');
|
|
10494
|
-
}
|
|
10495
|
-
/**
|
|
10496
|
-
* Transform an event handler whose body is an inline `unless` guard with NO
|
|
10497
|
-
* `end` (`on <event> unless <cond> <body>` \u2014 the `unless-condition` shape).
|
|
10498
|
-
*
|
|
10499
|
-
* Object-marking SVO targets (he, zh). `parseEventHandler` reads `unless` as the
|
|
10500
|
-
* action and sweeps the whole `<cond> <body>` tail into a single `patient` blob;
|
|
10501
|
-
* the target then prefixes that blob with its object marker \u2014 Hebrew's accusative
|
|
10502
|
-
* \u05d0\u05ea (`\u2026 \u05d0\u05dc\u05d0 \u05d0\u05ea I match .disabled \u05de\u05ea\u05d2 .selected`) or Chinese's BA particle \u628a
|
|
10503
|
-
* (`\u2026 \u9664\u975e \u628a I match .disabled \u5207\u6362 .selected`) \u2014 and the inner toggle loses its
|
|
10504
|
-
* own marker. The semantic parser can't recover the guard from that: the marker
|
|
10505
|
-
* ahead of the condition blocks the `unless` pattern AND the now-markerless body
|
|
10506
|
-
* command fails its object-marked toggle pattern, so the body collapses (`unless`
|
|
10507
|
-
* dropped). Marker-less languages (de/it/ar/pl) tolerate the same role-blob and
|
|
10508
|
-
* stay faithful, so this is an object-marker artifact, not a general parse gap.
|
|
10509
|
-
*
|
|
10510
|
-
* The standalone `unless <cond> <body>` path already produces the correct shape
|
|
10511
|
-
* (`extractBlockStructure` \u2192 `transformBlock`: condition kept marker-free, body
|
|
10512
|
-
* command keeps its marker \u2014 he `\u05d0\u05dc\u05d0 I match .disabled \u05de\u05ea\u05d2 \u05d0\u05ea .selected`, zh
|
|
10513
|
-
* `\u9664\u975e I match .disabled \u5207\u6362 \u628a .selected`). So we split the event head off,
|
|
10514
|
-
* transform the guard through that path, and emit the event clause first (he and
|
|
10515
|
-
* zh are both SVO \u2014 event leads). Returns `null` (fall through) when the input
|
|
10516
|
-
* isn't an object-marking event handler with an un-terminated inline `unless`
|
|
10517
|
-
* guard.
|
|
10518
|
-
*/
|
|
10519
|
-
tryTransformEventWithUnlessGuard(input) {
|
|
10520
|
-
// SVO object-marking targets only \u2014 these front the unless tail with an object
|
|
10521
|
-
// marker (he \u05d0\u05ea / zh \u628a) that breaks the parse. Event-leads emission below
|
|
10522
|
-
// assumes SVO, so SOV/VSO object-markers (ja/ko/tr/ar) are intentionally out.
|
|
10523
|
-
if (!UNLESS_GUARD_OBJECT_MARKING_LOCALES.has(this.targetProfile.code))
|
|
10524
|
-
return null;
|
|
10525
|
-
const tokens = tokenize(input, this.sourceProfile);
|
|
10526
|
-
if (tokens.length === 0)
|
|
10527
|
-
return null;
|
|
10528
|
-
if (!EVENT_KEYWORDS.has(tokens[0]?.toLowerCase()))
|
|
10529
|
-
return null;
|
|
10530
|
-
let guardIdx = -1;
|
|
10531
|
-
for (let i = 1; i < tokens.length; i++) {
|
|
10532
|
-
if (tokens[i].toLowerCase() === 'unless') {
|
|
10533
|
-
guardIdx = i;
|
|
10534
|
-
break;
|
|
10535
|
-
}
|
|
10536
|
-
}
|
|
10537
|
-
if (guardIdx <= 0)
|
|
10538
|
-
return null;
|
|
10539
|
-
// The terminated form (`\u2026 unless \u2026 end`) is handled by
|
|
10540
|
-
// tryTransformEventWithBlockBody's masking path; only take the inline guard.
|
|
10541
|
-
if (tokens[tokens.length - 1].toLowerCase() === 'end')
|
|
10542
|
-
return null;
|
|
10543
|
-
const eventHead = tokens.slice(0, guardIdx);
|
|
10544
|
-
const guard = tokens.slice(guardIdx).join(' ');
|
|
10545
|
-
// `unless` is a BLOCK_HEAD keyword, so transform() routes the guard through
|
|
10546
|
-
// extractBlockStructure \u2192 transformBlock (the standalone shape that parses).
|
|
10547
|
-
const guardOut = this.transform(guard);
|
|
10548
|
-
if (!guardOut)
|
|
10549
|
-
return null;
|
|
10550
|
-
const placeholder = 'EVENTGUARDPLACEHOLDER';
|
|
10551
|
-
const headOut = this.transformSingle([...eventHead, placeholder].join(' '));
|
|
10552
|
-
if (!headOut.includes(placeholder))
|
|
10553
|
-
return null;
|
|
10554
|
-
const eventClause = headOut.replace(placeholder, '').replace(/\s+/g, ' ').trim();
|
|
10555
|
-
return [eventClause, guardOut].filter(s => s.length > 0).join(' ');
|
|
10556
|
-
}
|
|
10557
|
-
/**
|
|
10558
|
-
* Transform an event handler whose body leads with a command-modifier
|
|
10559
|
-
* (`on <event> [from <src>] {async|once|debounced [at N]|throttled [at N]} <body>`).
|
|
10560
|
-
*
|
|
10561
|
-
* `parseEventHandler` reads the first token after the event as the **action**, so
|
|
10562
|
-
* a leading modifier is mistaken for the verb and the real verb (`fetch`/`add`) is
|
|
10563
|
-
* swept into the patient. For SOV targets the reorder then surfaces that verb
|
|
10564
|
-
* **first** (`\u53d6\u5f97 /api/data \u3092 \u30af\u30ea\u30c3\u30af \u2026`), and the semantic parser matches the
|
|
10565
|
-
* leading `<verb> <patient>` with the low-priority `*-generated-verb-first`
|
|
10566
|
-
* command pattern \u2014 returning a bare command and discarding the event + the rest
|
|
10567
|
-
* of the body (degenerate parse).
|
|
10568
|
-
*
|
|
10569
|
-
* Instead, lift the modifier out, transform the modifier-free handler through the
|
|
10570
|
-
* normal path (which keeps the body in canonical patient-first SOV order so the
|
|
10571
|
-
* event sits mid-stream and the existing SOV event-extraction recovers it), then
|
|
10572
|
-
* re-emit the modifier as a **leading English literal**. The semantic parser
|
|
10573
|
-
* strips a leading `once`/`debounced`/`throttled` (`extractStandaloneModifiers`)
|
|
10574
|
-
* and an `async` anywhere (`stripAsyncModifier`) before parsing, so the modifier
|
|
10575
|
-
* is consumed as handler metadata rather than shadowing the body.
|
|
10576
|
-
*
|
|
10577
|
-
* Returns `null` (fall through) when the input isn't an event handler or the body
|
|
10578
|
-
* doesn't lead with a modifier \u2014 leaving simple/Mode-B handlers byte-identical.
|
|
10579
|
-
*/
|
|
10580
|
-
tryTransformEventWithModifierBody(input) {
|
|
10581
|
-
// The verb-first degenerate parse this works around is specific to SOV
|
|
10582
|
-
// reorder: only there does a leading modifier displace the patient-first
|
|
10583
|
-
// order and surface the verb first. SVO/VSO/V2/other targets keep the body
|
|
10584
|
-
// in an order the parser already handles, so leave them byte-identical.
|
|
10585
|
-
if (this.targetProfile.wordOrder !== 'SOV')
|
|
10586
|
-
return null;
|
|
10587
|
-
const tokens = tokenize(input, this.sourceProfile);
|
|
10588
|
-
if (tokens.length === 0)
|
|
10589
|
-
return null;
|
|
10590
|
-
if (!EVENT_KEYWORDS.has(tokens[0]?.toLowerCase()))
|
|
10591
|
-
return null;
|
|
10592
|
-
// Walk past the event clause head: event keyword, event token, any
|
|
10593
|
-
// `or`-conjoined events, and an optional `from <source>` modifier \u2014 mirroring
|
|
10594
|
-
// parseEventHandler's head parsing \u2014 to find where the body begins.
|
|
10595
|
-
let i = 1; // past the event keyword
|
|
10596
|
-
if (!tokens[i])
|
|
10597
|
-
return null;
|
|
10598
|
-
i++; // past the event token
|
|
10599
|
-
while (tokens[i] && EVENT_CONJUNCTIONS.has(tokens[i].toLowerCase()) && tokens[i + 1]) {
|
|
10600
|
-
i += 2;
|
|
10601
|
-
}
|
|
10602
|
-
if (tokens[i]?.toLowerCase() === 'from' && tokens[i + 1]) {
|
|
10603
|
-
i++; // skip 'from'
|
|
10604
|
-
// Collect source tokens until a command verb or a body modifier.
|
|
10605
|
-
while (tokens[i] &&
|
|
10606
|
-
!ENGLISH_COMMANDS.has(tokens[i].toLowerCase()) &&
|
|
10607
|
-
!BODY_MODIFIER_KEYWORDS.has(tokens[i].toLowerCase())) {
|
|
10608
|
-
i++;
|
|
10609
|
-
}
|
|
10610
|
-
}
|
|
10611
|
-
const modWord = tokens[i]?.toLowerCase();
|
|
10612
|
-
if (!modWord || !BODY_MODIFIER_KEYWORDS.has(modWord))
|
|
10613
|
-
return null;
|
|
10614
|
-
// Consume the modifier phrase. `debounced`/`throttled` may carry an optional
|
|
10615
|
-
// `at <duration>` (or a bare duration). `async`/`once` are single tokens.
|
|
10616
|
-
const modStart = i;
|
|
10617
|
-
let modEnd = i + 1;
|
|
10618
|
-
if (modWord !== 'async' && modWord !== 'once') {
|
|
10619
|
-
if (tokens[modEnd]?.toLowerCase() === 'at')
|
|
10620
|
-
modEnd++;
|
|
10621
|
-
if (tokens[modEnd] && /^\d+(ms|s|m)?$/.test(tokens[modEnd]))
|
|
10622
|
-
modEnd++;
|
|
10623
|
-
}
|
|
10624
|
-
const modifierPhrase = tokens.slice(modStart, modEnd).join(' ');
|
|
10625
|
-
const rebuilt = [...tokens.slice(0, modStart), ...tokens.slice(modEnd)].join(' ');
|
|
10626
|
-
// A handler with only a modifier and no body has nothing to keep patient-first.
|
|
10627
|
-
if (tokens.length - (modEnd - modStart) <= 2)
|
|
10628
|
-
return null;
|
|
10629
|
-
// Re-run the full transform on the modifier-free handler so then-chains and
|
|
10630
|
-
// juxtaposed bodies route through the existing (working) paths. The rebuilt
|
|
10631
|
-
// input no longer leads with a modifier, so this never re-enters here.
|
|
10632
|
-
const bodyOut = this.transform(rebuilt);
|
|
10633
|
-
return [modifierPhrase, bodyOut].filter(s => s.length > 0).join(' ');
|
|
10634
|
-
}
|
|
10635
|
-
/**
|
|
10636
|
-
* Transform a `set <stuff> on <scope>` clause (S1 tabs-aria). The trailing
|
|
10637
|
-
* `on <scope>` is the element(s) the attribute is set on \u2014 kept attached by
|
|
10638
|
-
* splitOnCommandBoundaries. The semantic parser captures it as a `scope` role
|
|
10639
|
-
* via the passthrough literal `on` (setSchema's scope markerOverride is `on`
|
|
10640
|
-
* in every language), so `on` is emitted verbatim and only the scope *value*
|
|
10641
|
-
* is translated (selectors pass through; `me`/`it`/`you` translate to the
|
|
10642
|
-
* native reference, which the parser also accepts).
|
|
10643
|
-
*
|
|
10644
|
-
* Positioning matches where the set patterns expect the scope: at the clause
|
|
10645
|
-
* end for verb-first orders (SVO/VSO), and immediately before the clause-final
|
|
10646
|
-
* verb for SOV (the generated SOV pattern is `{dest} {patient} on {scope}
|
|
10647
|
-
* {verb}`). Returns null (fall through) when there is no trailing `on <scope>`.
|
|
10648
|
-
*
|
|
10649
|
-
* Source is English in the sync-translations pipeline, so the `set` verb and
|
|
10650
|
-
* `on` marker are matched as English literals.
|
|
10651
|
-
*/
|
|
10652
|
-
transformSetWithScope(input) {
|
|
10653
|
-
const src = this.sourceProfile.code;
|
|
10654
|
-
const dst = this.targetProfile.code;
|
|
10655
|
-
const m = input.match(/^(.*\bset\b.*\S)\s+on\s+([#.<@[]\S*|me|it|you)\s*$/i);
|
|
10656
|
-
if (!m)
|
|
10657
|
-
return null;
|
|
10658
|
-
const head = m[1];
|
|
10659
|
-
const scopeRaw = m[2];
|
|
10660
|
-
// Transform the scope-less clause via the normal path; `head` no longer ends
|
|
10661
|
-
// in `on <scope>`, so this never re-enters transformSetWithScope.
|
|
10662
|
-
const headOut = this.transformSingle(head);
|
|
10663
|
-
const scopeT = /^[#.<@[]/.test(scopeRaw) ? scopeRaw : translateWord(scopeRaw, src, dst);
|
|
10664
|
-
// SOV needs the scope positioned per how the semantic set patterns match:
|
|
10665
|
-
// - Event-handler set (`on <event> set \u2026`): the dest-first SOV event-handler
|
|
10666
|
-
// pattern is verb-MEDIAL and carries an optional trailing `[on {scope}]`,
|
|
10667
|
-
// so append the scope at the clause end.
|
|
10668
|
-
// - Standalone then-clause set: SOV emits verb-MEDIAL (`{dest} {verb}
|
|
10669
|
-
// {patient}`), but the only command set pattern with scope is verb-LAST
|
|
10670
|
-
// (`{dest} {patient} on {scope} {verb}`). Move the medial verb to the end
|
|
10671
|
-
// and place `on {scope}` before it, so the generated command pattern matches.
|
|
10672
|
-
if (this.targetProfile.wordOrder === 'SOV') {
|
|
10673
|
-
const verb = translateWord('set', 'en', dst);
|
|
10674
|
-
const firstTok = input.trim().split(/\s+/)[0]?.toLowerCase();
|
|
10675
|
-
const isEventHandler = !!firstTok && EVENT_KEYWORDS.has(firstTok);
|
|
10676
|
-
const toks = headOut.split(/\s+/).filter(Boolean);
|
|
10677
|
-
if (!isEventHandler) {
|
|
10678
|
-
// Standalone then-clause set: SOV emits verb-MEDIAL; move the verb to the
|
|
10679
|
-
// end and place `on {scope}` before it so the verb-last command set
|
|
10680
|
-
// pattern (scope before verb) matches.
|
|
10681
|
-
const vIdx = toks.indexOf(verb);
|
|
10682
|
-
if (vIdx >= 0) {
|
|
10683
|
-
toks.splice(vIdx, 1);
|
|
10684
|
-
toks.push('on', scopeT, verb);
|
|
10685
|
-
return toks.join(' ');
|
|
10686
|
-
}
|
|
10687
|
-
}
|
|
10688
|
-
else if (toks.length > 0 && toks[toks.length - 1] === verb) {
|
|
10689
|
-
// Event-handler set whose verb is clause-final (e.g. qu
|
|
10690
|
-
// `{dest} ta {patient} man {event} pi {verb}`): the parser extracts the
|
|
10691
|
-
// event and matches the body as a verb-last command, so the scope must
|
|
10692
|
-
// sit before the verb. Verb-MEDIAL SOV event handlers (ja/ko/tr/bn/hi)
|
|
10693
|
-
// fall through to the append branch, where their fused event-handler set
|
|
10694
|
-
// pattern carries the trailing `[on {scope}]` group.
|
|
10695
|
-
toks.splice(toks.length - 1, 0, 'on', scopeT);
|
|
10696
|
-
return toks.join(' ');
|
|
10697
|
-
}
|
|
10698
|
-
return `${headOut} on ${scopeT}`;
|
|
10699
|
-
}
|
|
10700
|
-
// Verb-first (SVO/VSO/V2): append `on <scope>` at the clause end, which the
|
|
10701
|
-
// trailing `[on {scope}]` group on the set patterns matches.
|
|
10702
|
-
return `${headOut} on ${scopeT}`;
|
|
10703
|
-
}
|
|
10704
|
-
/**
|
|
10705
|
-
* Transform a self-contained block command (`{head} {clause?} {body} {end}`),
|
|
10706
|
-
* where head \u2208 {if, repeat, unless, \u2026}. The clause (condition / `until event \u2026`)
|
|
10707
|
-
* runs up to the first command verb and is translated word-by-word; the body is
|
|
10708
|
-
* recursively transformed (so its inner commands reorder for the target); the
|
|
10709
|
-
* head/tail keywords are translated. The block is never word-order reordered as
|
|
10710
|
-
* a whole \u2014 delimiters stay at the edges regardless of target word order.
|
|
10711
|
-
*/
|
|
10712
|
-
transformBlockBody(blockTokens) {
|
|
10713
|
-
const src = this.sourceProfile.code;
|
|
10714
|
-
const dst = this.targetProfile.code;
|
|
10715
|
-
const head = blockTokens[0];
|
|
10716
|
-
const hasEnd = blockTokens[blockTokens.length - 1]?.toLowerCase() === 'end';
|
|
10717
|
-
const tail = hasEnd ? blockTokens[blockTokens.length - 1] : '';
|
|
10718
|
-
const inner = blockTokens.slice(1, hasEnd ? -1 : undefined);
|
|
10719
|
-
// Skip predicate-adjective positions (`\u2026 is empty`): the adjective is
|
|
10720
|
-
// part of the condition clause, not the body's first verb \u2014 cutting there
|
|
10721
|
-
// displaced it into the next command's argument zone (empty \u00d78 bn/hi/tr).
|
|
10722
|
-
const commands = getCommandKeywordsForLocale(src);
|
|
10723
|
-
const copulas = getCopulasForLocale(src);
|
|
10724
|
-
let bodyStart = inner.findIndex((t, i) => commands.has(t.toLowerCase()) && !isPredicateAdjectivePosition(inner, i, copulas));
|
|
10725
|
-
if (bodyStart < 0)
|
|
10726
|
-
bodyStart = inner.length;
|
|
10727
|
-
const clause = inner.slice(0, bodyStart).join(' ');
|
|
10728
|
-
const bodyTokens = inner.slice(bodyStart);
|
|
10729
|
-
const headT = translateWord(head, src, dst);
|
|
10730
|
-
const tailT = tail ? translateWord(tail, src, dst) : '';
|
|
10731
|
-
const clauseT = clause ? translateMultiWordValue(clause, src, dst) : '';
|
|
10732
|
-
const bodyT = this.transformConditionalBody(bodyTokens);
|
|
10733
|
-
// SOV condition/branch boundary (R1 deferred-tail Family G): the SOV body
|
|
10734
|
-
// renders its first command operand-first (`\u6700\u521d <button/> \u306e\u4e2d .modal \u3092
|
|
10735
|
-
// \u30d5\u30a9\u30fc\u30ab\u30b9`), so nothing marks where the condition ends and the branch
|
|
10736
|
-
// operand begins \u2014 the semantic fold's command-start detection needs a
|
|
10737
|
-
// `{value}{particle}` run followed by a verb, which a POSITIONAL-headed
|
|
10738
|
-
// operand (`first <button/> \u2026`) never forms, and the condition scan
|
|
10739
|
-
// swallows the operand's head (focus-trap: ja focus.patient fell to the
|
|
10740
|
-
// `me` default, ko/qu to the `.modal` tail). Emit the target's
|
|
10741
|
-
// then-connective at the seam \u2014 the boundary the fold already respects
|
|
10742
|
-
// (isThenKeyword) \u2014 gated to exactly the blind shape: SOV target, a
|
|
10743
|
-
// positional keyword right after the branch's command verb, and no `then`
|
|
10744
|
-
// already ending the condition.
|
|
10745
|
-
const POSITIONAL_BRANCH_HEADS = new Set(['first', 'last', 'next', 'previous', 'closest']);
|
|
10746
|
-
const thenT = this.targetProfile.wordOrder === 'SOV' &&
|
|
10747
|
-
clauseT &&
|
|
10748
|
-
POSITIONAL_BRANCH_HEADS.has(bodyTokens[1]?.toLowerCase()) &&
|
|
10749
|
-
inner[bodyStart - 1]?.toLowerCase() !== 'then'
|
|
10750
|
-
? translateWord('then', src, dst)
|
|
10751
|
-
: '';
|
|
10752
|
-
return [headT, clauseT, thenT, bodyT, tailT].filter(s => s.length > 0).join(' ');
|
|
10753
|
-
}
|
|
10754
|
-
/**
|
|
10755
|
-
* Transform an `if`/`unless` block body, splitting it at a top-level `else` into
|
|
10756
|
-
* a then-branch and an else-branch so each is reordered as a self-contained unit
|
|
10757
|
-
* and the `else` keyword itself is translated. Without this, the body is reordered
|
|
10758
|
-
* as one stream: `else` rides along glued to the preceding clause (and, when that
|
|
10759
|
-
* clause begins with a selector, is marked a selector and left *untranslated*),
|
|
10760
|
-
* and a spurious `then` is inserted around it \u2014 both of which break the target
|
|
10761
|
-
* text and the downstream parse. The split is depth-aware so an `else` belonging
|
|
10762
|
-
* to a nested block is not mistaken for this block's separator. Bodies without an
|
|
10763
|
-
* `else` transform exactly as before.
|
|
10764
|
-
*/
|
|
10765
|
-
transformConditionalBody(bodyTokens) {
|
|
10766
|
-
const src = this.sourceProfile.code;
|
|
10767
|
-
const dst = this.targetProfile.code;
|
|
10768
|
-
const sourceElse = translateWord('else', 'en', src).toLowerCase();
|
|
10769
|
-
let depth = 0;
|
|
10770
|
-
let elseIdx = -1;
|
|
10771
|
-
for (let i = 0; i < bodyTokens.length; i++) {
|
|
10772
|
-
const t = bodyTokens[i].toLowerCase();
|
|
10773
|
-
if (BLOCK_BODY_KEYWORDS.has(t))
|
|
10774
|
-
depth++;
|
|
10775
|
-
else if (t === 'end' && depth > 0)
|
|
10776
|
-
depth--;
|
|
10777
|
-
else if (t === sourceElse && depth === 0) {
|
|
10778
|
-
elseIdx = i;
|
|
10779
|
-
break;
|
|
10780
|
-
}
|
|
10781
|
-
}
|
|
10782
|
-
if (elseIdx === -1) {
|
|
10783
|
-
const body = bodyTokens.join(' ');
|
|
10784
|
-
return body ? this.transform(body) : '';
|
|
10785
|
-
}
|
|
10786
|
-
const thenBranch = bodyTokens.slice(0, elseIdx).join(' ');
|
|
10787
|
-
const elseBranch = bodyTokens.slice(elseIdx + 1).join(' ');
|
|
10788
|
-
const elseT = translateWord(bodyTokens[elseIdx], src, dst);
|
|
10789
|
-
return [
|
|
10790
|
-
thenBranch ? this.transform(thenBranch) : '',
|
|
10791
|
-
elseT,
|
|
10792
|
-
elseBranch ? this.transform(elseBranch) : '',
|
|
10793
|
-
]
|
|
10794
|
-
.filter(s => s.length > 0)
|
|
10795
|
-
.join(' ');
|
|
10796
|
-
}
|
|
10797
|
-
/**
|
|
10798
|
-
* Translate a reactive block by translating the head/tail/connector
|
|
10799
|
-
* via the dictionary, recursively transforming the body through the
|
|
10800
|
-
* regular pipeline, and rejoining in source-language position order.
|
|
10801
|
-
* Block-syntactic tokens are never reordered: they're delimiters, not
|
|
10802
|
-
* arguments, and authors expect them at start/end positions
|
|
10803
|
-
* regardless of target word order.
|
|
10804
|
-
*/
|
|
10805
|
-
transformBlock(block) {
|
|
10806
|
-
const src = this.sourceProfile.code;
|
|
10807
|
-
const dst = this.targetProfile.code;
|
|
10808
|
-
const head = translateWord(block.headKeyword, src, dst);
|
|
10809
|
-
const tail = block.tailKeyword ? translateWord(block.tailKeyword, src, dst) : '';
|
|
10810
|
-
const connector = block.connector ? translateWord(block.connector, src, dst) : '';
|
|
10811
|
-
const prefix = block.prefixExpr ? translateMultiWordValue(block.prefixExpr, src, dst) : '';
|
|
10812
|
-
// Recurse through `transform()` (not `transformSingle`) so the body
|
|
10813
|
-
// gets `then`-splitting and nested-block handling for free.
|
|
10814
|
-
const body = this.transform(block.body);
|
|
10815
|
-
return [head, prefix, connector, body, tail].filter(s => s.length > 0).join(' ');
|
|
10816
|
-
}
|
|
10817
|
-
/**
|
|
10818
|
-
* Find the best matching rule for this statement
|
|
10819
|
-
*/
|
|
10820
|
-
findRule(parsed) {
|
|
10821
|
-
if (!this.targetProfile.rules)
|
|
10822
|
-
return undefined;
|
|
10823
|
-
const matchingRules = this.targetProfile.rules
|
|
10824
|
-
.filter(rule => this.matchesRule(parsed, rule))
|
|
10825
|
-
.sort((a, b) => b.priority - a.priority);
|
|
10826
|
-
return matchingRules[0];
|
|
10827
|
-
}
|
|
10828
|
-
/**
|
|
10829
|
-
* Check if a parsed statement matches a rule
|
|
10830
|
-
*/
|
|
10831
|
-
matchesRule(parsed, rule) {
|
|
10832
|
-
const { match } = rule;
|
|
10833
|
-
// Check required roles
|
|
10834
|
-
for (const role of match.requiredRoles) {
|
|
10835
|
-
if (!parsed.roles.has(role)) {
|
|
10836
|
-
return false;
|
|
10837
|
-
}
|
|
10838
|
-
}
|
|
10839
|
-
// Check command match if specified
|
|
10840
|
-
if (match.commands && match.commands.length > 0) {
|
|
10841
|
-
const action = parsed.roles.get('action');
|
|
10842
|
-
if (!action)
|
|
10843
|
-
return false;
|
|
10844
|
-
const actionValue = action.value.toLowerCase();
|
|
10845
|
-
if (!match.commands.some(cmd => cmd.toLowerCase() === actionValue)) {
|
|
10846
|
-
return false;
|
|
10847
|
-
}
|
|
10848
|
-
}
|
|
10849
|
-
// Check custom predicate
|
|
10850
|
-
if (match.predicate && !match.predicate(parsed)) {
|
|
10851
|
-
return false;
|
|
10852
|
-
}
|
|
10853
|
-
return true;
|
|
10854
|
-
}
|
|
10855
|
-
}
|
|
10856
|
-
// =============================================================================
|
|
10857
|
-
// Convenience Functions
|
|
10858
|
-
// =============================================================================
|
|
10859
|
-
/**
|
|
10860
|
-
* Transform hyperscript from English to target language
|
|
10861
|
-
*/
|
|
10862
|
-
function toLocale(input, targetLocale) {
|
|
10863
|
-
const transformer = new GrammarTransformer('en', targetLocale);
|
|
10864
|
-
return transformer.transform(input);
|
|
10865
|
-
}
|
|
10866
|
-
/**
|
|
10867
|
-
* Transform hyperscript from source language to English
|
|
10868
|
-
*/
|
|
10869
|
-
function toEnglish(input, sourceLocale) {
|
|
10870
|
-
const transformer = new GrammarTransformer(sourceLocale, 'en');
|
|
10871
|
-
return transformer.transform(input);
|
|
10872
|
-
}
|
|
10873
|
-
/**
|
|
10874
|
-
* Transform between any two languages.
|
|
10875
|
-
*
|
|
10876
|
-
* Uses direct translation for supported language pairs (ja\u2194zh, es\u2194pt, ko\u2194ja),
|
|
10877
|
-
* falling back to English pivot for other pairs.
|
|
10878
|
-
*/
|
|
10879
|
-
function translate(input, sourceLocale, targetLocale) {
|
|
10880
|
-
if (sourceLocale === targetLocale)
|
|
10881
|
-
return input;
|
|
10882
|
-
if (sourceLocale === 'en')
|
|
10883
|
-
return toLocale(input, targetLocale);
|
|
10884
|
-
if (targetLocale === 'en')
|
|
10885
|
-
return toEnglish(input, sourceLocale);
|
|
10886
|
-
// Try direct translation for supported pairs
|
|
10887
|
-
if (hasDirectMapping(sourceLocale, targetLocale)) {
|
|
10888
|
-
return translateDirect(input, sourceLocale, targetLocale);
|
|
10889
|
-
}
|
|
10890
|
-
// Fallback: Via English pivot
|
|
10891
|
-
const english = toEnglish(input, sourceLocale);
|
|
10892
|
-
return toLocale(english, targetLocale);
|
|
10893
|
-
}
|
|
10894
|
-
/**
|
|
10895
|
-
* Direct translation between language pairs without English pivot.
|
|
10896
|
-
* More accurate for closely related languages (ja\u2194zh, es\u2194pt).
|
|
10897
|
-
*/
|
|
10898
|
-
function translateDirect(input, sourceLocale, targetLocale) {
|
|
10899
|
-
const mapping = getDirectMapping(sourceLocale, targetLocale);
|
|
10900
|
-
if (!mapping) {
|
|
10901
|
-
// Fallback to pivot translation
|
|
10902
|
-
return toLocale(toEnglish(input, sourceLocale), targetLocale);
|
|
10903
|
-
}
|
|
10904
|
-
// Tokenize input
|
|
10905
|
-
const tokens = input.split(/\s+/);
|
|
10906
|
-
// Translate each token using direct mapping
|
|
10907
|
-
const translated = tokens.map(token => {
|
|
10908
|
-
// Preserve CSS selectors and literals
|
|
10909
|
-
if (token.startsWith('#') || token.startsWith('.') || token.startsWith('@')) {
|
|
10910
|
-
return token;
|
|
10911
|
-
}
|
|
10912
|
-
if (token.startsWith('"') || token.startsWith("'")) {
|
|
10913
|
-
return token;
|
|
10914
|
-
}
|
|
10915
|
-
// Look up in direct mapping
|
|
10916
|
-
const directTranslation = mapping.words[token];
|
|
10917
|
-
if (directTranslation) {
|
|
10918
|
-
return directTranslation;
|
|
10919
|
-
}
|
|
10920
|
-
// Check for suffix-attached tokens (e.g., "#count-ta" in Quechua)
|
|
10921
|
-
const suffixMatch = token.match(/^(.+?)(-.+)$/);
|
|
10922
|
-
if (suffixMatch) {
|
|
10923
|
-
const [, base, suffix] = suffixMatch;
|
|
10924
|
-
const translatedBase = mapping.words[base] || base;
|
|
10925
|
-
return translatedBase + suffix;
|
|
10926
|
-
}
|
|
10927
|
-
// Return unchanged if no mapping found
|
|
10928
|
-
return token;
|
|
10929
|
-
});
|
|
10930
|
-
return translated.join(' ');
|
|
10931
|
-
}
|
|
10932
|
-
// =============================================================================
|
|
10933
|
-
// Examples (for testing)
|
|
10934
|
-
// =============================================================================
|
|
10935
|
-
const examples = {
|
|
10936
|
-
english: {
|
|
10937
|
-
eventHandler: 'on click increment #count',
|
|
10938
|
-
putInto: 'put my value into #output',
|
|
10939
|
-
toggle: 'toggle .active',
|
|
10940
|
-
wait: 'wait 2 seconds',
|
|
10941
|
-
},
|
|
10942
|
-
// Expected outputs (approximate, for reference)
|
|
10943
|
-
japanese: {
|
|
10944
|
-
eventHandler: '#count \u3092 \u30af\u30ea\u30c3\u30af \u3067 \u5897\u52a0',
|
|
10945
|
-
putInto: '\u79c1\u306e \u5024 \u3092 #output \u306b \u7f6e\u304f',
|
|
10946
|
-
toggle: '.active \u3092 \u5207\u308a\u66ff\u3048',
|
|
10947
|
-
wait: '2\u79d2 \u5f85\u3064',
|
|
10948
|
-
},
|
|
10949
|
-
chinese: {
|
|
10950
|
-
eventHandler: '\u5f53 \u70b9\u51fb \u65f6 \u589e\u52a0 #count',
|
|
10951
|
-
putInto: '\u628a \u6211\u7684\u503c \u653e \u5230 #output',
|
|
10952
|
-
toggle: '\u5207\u6362 .active',
|
|
10953
|
-
wait: '\u7b49\u5f85 2\u79d2',
|
|
10954
|
-
},
|
|
10955
|
-
arabic: {
|
|
10956
|
-
eventHandler: '\u0632\u0650\u062f #count \u0639\u0646\u062f \u0627\u0644\u0646\u0642\u0631',
|
|
10957
|
-
putInto: '\u0636\u0639 \u0642\u064a\u0645\u062a\u064a \u0641\u064a #output',
|
|
10958
|
-
toggle: '\u0628\u062f\u0651\u0644 .active',
|
|
10959
|
-
wait: '\u0627\u0646\u062a\u0638\u0631 \u062b\u0627\u0646\u064a\u062a\u064a\u0646',
|
|
10960
|
-
},
|
|
10961
|
-
};
|
|
10962
|
-
|
|
10963
|
-
export { ENGLISH_COMMANDS, ENGLISH_KEYWORDS, GrammarTransformer, LANGUAGE_FAMILY_DEFAULTS, LocaleManager, UNIVERSAL_ENGLISH_KEYWORDS, UNIVERSAL_PATTERNS, ar$1 as ar, ar$1 as arDictionary, arKeywords, arabicProfile, bn$1 as bn, bn$1 as bnDictionary, bnKeywords, chineseProfile, createEnglishProvider, createKeywordProvider, de$1 as de, de$1 as deDictionary, deKeywords, detectBrowserLocale, directMappings, en$1 as en, englishProfile, es$1 as es, es$1 as esDictionary, esKeywords, fr$1 as fr, fr$1 as frDictionary, frKeywords, frenchProfile, germanProfile, getDirectMapping, getProfile, getSupportedDirectPairs, getSupportedLocales, examples as grammarExamples, hasDirectMapping, he$1 as he, he$1 as heDictionary, heKeywords, hebrewProfile, hindiDictionary as hiDictionary, hiKeywords, hindiDictionary, id$1 as id, id$1 as idDictionary, idKeywords, indonesianProfile, insertMarkers, it$1 as it, it$1 as itDictionary, itKeywords, ja$1 as ja, ja$1 as jaDictionary, jaKeywords, japaneseProfile, joinTokens, ko$1 as ko, ko$1 as koDictionary, koKeywords, koreanProfile, malayProfile, ms$1 as ms, ms$1 as msDictionary, msKeywords, parseStatement, pl$1 as pl, pl$1 as plDictionary, plKeywords, portugueseProfile, profiles, pt$1 as pt, pt$1 as ptDictionary, ptKeywords, qu$1 as qu, qu$1 as quDictionary, quKeywords, quechuaProfile, reorderRoles, russianDictionary as ruDictionary, ruKeywords, russianDictionary, spanishProfile, sw$1 as sw, sw$1 as swDictionary, swKeywords, swahiliProfile, th$1 as th, th$1 as thDictionary, thKeywords, tl$1 as tl, tl$1 as tlDictionary, tlKeywords, toEnglish, toLocale, tr$1 as tr, tr$1 as trDictionary, trKeywords, transformStatement, translate, translateWordDirect, turkishProfile, ukrainianDictionary as ukDictionary, ukKeywords, ukrainianDictionary, vi$1 as vi, vi$1 as viDictionary, viKeywords, zh$1 as zh, zh$1 as zhDictionary, zhKeywords };
|
|
8332
|
+
export { ENGLISH_COMMANDS, ENGLISH_KEYWORDS, LANGUAGE_FAMILY_DEFAULTS, LocaleManager, UNIVERSAL_ENGLISH_KEYWORDS, UNIVERSAL_PATTERNS, ar, ar as arDictionary, arKeywords, arabicProfile, bn, bn as bnDictionary, bnKeywords, chineseProfile, createEnglishProvider, createKeywordProvider, de, de as deDictionary, deKeywords, detectBrowserLocale, directMappings, en, englishProfile, es, es as esDictionary, esKeywords, fr, fr as frDictionary, frKeywords, frenchProfile, germanProfile, getDirectMapping, getProfile, getSupportedDirectPairs, getSupportedLocales, hasDirectMapping, he, he as heDictionary, heKeywords, hebrewProfile, hindiDictionary as hiDictionary, hiKeywords, hindiDictionary, id, id as idDictionary, idKeywords, indonesianProfile, insertMarkers, it, it as itDictionary, itKeywords, ja, ja as jaDictionary, jaKeywords, japaneseProfile, joinTokens, ko, ko as koDictionary, koKeywords, koreanProfile, malayProfile, ms, ms as msDictionary, msKeywords, pl, pl as plDictionary, plKeywords, portugueseProfile, profiles, pt, pt as ptDictionary, ptKeywords, qu, qu as quDictionary, quKeywords, quechuaProfile, reorderRoles, russianDictionary as ruDictionary, ruKeywords, russianDictionary, spanishProfile, sw, sw as swDictionary, swKeywords, swahiliProfile, th, th as thDictionary, thKeywords, tl, tl as tlDictionary, tlKeywords, tr, tr as trDictionary, trKeywords, transformStatement, translateWordDirect, turkishProfile, ukrainianDictionary as ukDictionary, ukKeywords, ukrainianDictionary, vi, vi as viDictionary, viKeywords, zh, zh as zhDictionary, zhKeywords };
|
|
10964
8333
|
//# sourceMappingURL=lokascript-i18n.mjs.map
|