@lokascript/i18n 2.11.1 → 3.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/CHANGELOG.md +13 -0
  2. package/dist/browser.cjs +5 -1690
  3. package/dist/browser.cjs.map +1 -1
  4. package/dist/browser.d.cts +2 -2
  5. package/dist/browser.d.ts +2 -2
  6. package/dist/browser.js +6 -1685
  7. package/dist/browser.js.map +1 -1
  8. package/dist/dictionaries/index.cjs +5 -1
  9. package/dist/dictionaries/index.cjs.map +1 -1
  10. package/dist/dictionaries/index.js +5 -1
  11. package/dist/dictionaries/index.js.map +1 -1
  12. package/dist/{transformer-CsOeqayN.d.cts → index-BykxjYST.d.cts} +1 -232
  13. package/dist/{transformer-DWCTG1DQ.d.ts → index-DuIef8O7.d.ts} +1 -232
  14. package/dist/index.cjs +16 -1878
  15. package/dist/index.cjs.map +1 -1
  16. package/dist/index.d.cts +1 -1
  17. package/dist/index.d.ts +1 -1
  18. package/dist/index.js +17 -1873
  19. package/dist/index.js.map +1 -1
  20. package/dist/lokascript-i18n.min.js +1 -1
  21. package/dist/lokascript-i18n.min.js.map +1 -1
  22. package/dist/lokascript-i18n.mjs +49 -2680
  23. package/dist/lokascript-i18n.mjs.map +1 -1
  24. package/dist/plugins/vite.cjs +5 -1
  25. package/dist/plugins/vite.cjs.map +1 -1
  26. package/dist/plugins/vite.js +5 -1
  27. package/dist/plugins/vite.js.map +1 -1
  28. package/dist/plugins/webpack.cjs +5 -1
  29. package/dist/plugins/webpack.cjs.map +1 -1
  30. package/dist/plugins/webpack.js +5 -1
  31. package/dist/plugins/webpack.js.map +1 -1
  32. package/package.json +4 -4
  33. package/src/browser.ts +0 -7
  34. package/src/compatibility/browser-tests/grammar-demo.spec.ts +22 -8
  35. package/src/constants.ts +1 -0
  36. package/src/dictionaries/bn.ts +5 -1
  37. package/src/grammar/index.ts +15 -9
  38. package/src/grammar/profiles.test.ts +440 -0
  39. package/src/index.ts +0 -7
  40. package/src/lexicon-parity.test.ts +77 -0
  41. package/src/grammar/grammar.test.ts +0 -2751
  42. package/src/grammar/transformer.ts +0 -2737
package/dist/index.cjs CHANGED
@@ -4525,7 +4525,11 @@ var bengaliDictionary = {
4525
4525
  random: "\u098F\u09B2\u09CB\u09AE\u09C7\u09B2\u09CB",
4526
4526
  length: "\u09A6\u09C8\u09B0\u09CD\u0998\u09CD\u09AF",
4527
4527
  index: "\u09B8\u09C2\u099A\u0995",
4528
- empty: "\u0996\u09BE\u09B2\u09BF-\u0995\u09B0\u09C1\u09A8",
4528
+ // The EXPRESSION `empty` is the state predicate (`if my value is empty`),
4529
+ // not the command: `খালি-করুন` is the imperative "empty it!" and belongs to
4530
+ // the `empty` COMMAND, which keeps it. Kept in step with the semantic
4531
+ // lexicon by `lexicon-parity.test.ts`.
4532
+ empty: "\u0996\u09BE\u09B2\u09BF",
4529
4533
  "starts with": "\u09A6\u09BF\u09AF\u09BC\u09C7_\u09B6\u09C1\u09B0\u09C1",
4530
4534
  "ends with": "\u09A6\u09BF\u09AF\u09BC\u09C7_\u09B6\u09C7\u09B7",
4531
4535
  "ignoring case": "\u0995\u09C7\u09B8_\u0989\u09AA\u09C7\u0995\u09CD\u09B7\u09BE",
@@ -7437,25 +7441,25 @@ var HyperscriptTranslator = class {
7437
7441
  }
7438
7442
  translateWithDetails(text, options) {
7439
7443
  const fromLocale = options.from || this.detectLanguage(text);
7440
- const toLocale2 = options.to;
7441
- if (fromLocale === toLocale2) {
7444
+ const toLocale = options.to;
7445
+ if (fromLocale === toLocale) {
7442
7446
  return {
7443
7447
  translated: text,
7444
7448
  original: text,
7445
7449
  tokens: [],
7446
- locale: { from: fromLocale, to: toLocale2 }
7450
+ locale: { from: fromLocale, to: toLocale }
7447
7451
  };
7448
7452
  }
7449
7453
  const fromDict = this.getDictionary(fromLocale);
7450
- const toDict = this.getDictionary(toLocale2);
7454
+ const toDict = this.getDictionary(toLocale);
7451
7455
  if (!fromDict || !toDict) {
7452
- throw new Error(`Missing dictionary for locale: ${!fromDict ? fromLocale : toLocale2}`);
7456
+ throw new Error(`Missing dictionary for locale: ${!fromDict ? fromLocale : toLocale}`);
7453
7457
  }
7454
7458
  const tokens = tokenize(text, fromLocale);
7455
- const translatedTokens = this.translateTokens(tokens, fromLocale, toLocale2);
7459
+ const translatedTokens = this.translateTokens(tokens, fromLocale, toLocale);
7456
7460
  const translated = this.reconstructText(translatedTokens);
7457
7461
  if (options.validate && toDict) {
7458
- const validation = validate(toDict, toLocale2);
7462
+ const validation = validate(toDict, toLocale);
7459
7463
  if (!validation.valid) {
7460
7464
  console.warn("Translation validation warnings:", validation.warnings);
7461
7465
  }
@@ -7463,7 +7467,7 @@ var HyperscriptTranslator = class {
7463
7467
  const result = {
7464
7468
  translated,
7465
7469
  tokens: translatedTokens,
7466
- locale: { from: fromLocale, to: toLocale2 },
7470
+ locale: { from: fromLocale, to: toLocale },
7467
7471
  warnings: []
7468
7472
  };
7469
7473
  if (options.preserveOriginal) {
@@ -7471,9 +7475,9 @@ var HyperscriptTranslator = class {
7471
7475
  }
7472
7476
  return result;
7473
7477
  }
7474
- translateTokens(tokens, fromLocale, toLocale2) {
7478
+ translateTokens(tokens, fromLocale, toLocale) {
7475
7479
  const fromDict = this.getDictionary(fromLocale);
7476
- const toDict = this.getDictionary(toLocale2);
7480
+ const toDict = this.getDictionary(toLocale);
7477
7481
  const reverseFromDict = this.getReverseDictionary(fromLocale);
7478
7482
  const emptyDict = {
7479
7483
  commands: {},
@@ -7488,7 +7492,7 @@ var HyperscriptTranslator = class {
7488
7492
  return tokens.map((token) => {
7489
7493
  let translated = token.value;
7490
7494
  if (this.isTranslatableToken(token)) {
7491
- if (fromLocale !== "en" && toLocale2 !== "en") {
7495
+ if (fromLocale !== "en" && toLocale !== "en") {
7492
7496
  const english = this.findTranslation(token.value, fromDict || emptyDict, reverseFromDict);
7493
7497
  if (english) {
7494
7498
  translated = this.findTranslation(english, toDict || emptyDict, /* @__PURE__ */ new Map()) || token.value;
@@ -7619,39 +7623,6 @@ var HyperscriptTranslator = class {
7619
7623
  new HyperscriptTranslator({ locale: "en" });
7620
7624
 
7621
7625
  // src/constants.ts
7622
- var ENGLISH_MODIFIER_ROLES = {
7623
- to: "destination",
7624
- into: "destination",
7625
- from: "source",
7626
- with: "style",
7627
- by: "quantity",
7628
- as: "method",
7629
- on: "event",
7630
- over: "duration",
7631
- for: "duration"
7632
- };
7633
- var COMMAND_PRIMARY_ROLES = {
7634
- set: "destination",
7635
- on: "event",
7636
- trigger: "event",
7637
- send: "event",
7638
- wait: "duration",
7639
- fetch: "source",
7640
- get: "source",
7641
- if: "condition",
7642
- unless: "condition",
7643
- while: "condition",
7644
- repeat: "loopType",
7645
- go: "destination",
7646
- scroll: "destination",
7647
- tell: "destination",
7648
- default: "destination",
7649
- swap: "destination",
7650
- // morph deliberately absent: its schema primaryRole is `patient` (the
7651
- // element being morphed — aligned with the transformer's patient marking
7652
- // in the session-9 role-layout swap), and patient is the default.
7653
- bind: "destination"
7654
- };
7655
7626
  var ENGLISH_MODIFIERS = /* @__PURE__ */ new Set([
7656
7627
  "to",
7657
7628
  "from",
@@ -7884,69 +7855,6 @@ var ENGLISH_EXPRESSION_KEYWORDS = /* @__PURE__ */ new Set([
7884
7855
  "starts with",
7885
7856
  "ends with"
7886
7857
  ]);
7887
- var CONDITIONAL_KEYWORDS = /* @__PURE__ */ new Set([
7888
- // English
7889
- "if",
7890
- "unless",
7891
- "when",
7892
- "where",
7893
- // Japanese
7894
- "\u3082\u3057",
7895
- "\u6642\u306B",
7896
- "\u3068\u304D\u306B",
7897
- "\u3069\u3053\u3067",
7898
- // Chinese
7899
- "\u5982\u679C",
7900
- "\u5F53",
7901
- // Arabic
7902
- "\u0625\u0630\u0627",
7903
- "\u0639\u0646\u062F\u0645\u0627",
7904
- "\u062D\u064A\u062B",
7905
- // Spanish
7906
- "si",
7907
- "cuando",
7908
- "donde",
7909
- // German
7910
- "wenn",
7911
- "wann",
7912
- "wo",
7913
- // French
7914
- "quand",
7915
- "lorsque",
7916
- "o\xF9",
7917
- // Portuguese
7918
- "quando",
7919
- "onde",
7920
- // Turkish
7921
- "e\u011Fer",
7922
- "zaman",
7923
- "nerede",
7924
- // Indonesian
7925
- "ketika",
7926
- "saat",
7927
- "dimana",
7928
- // Korean
7929
- "\uB54C",
7930
- "\uC5B4\uB514\uC11C",
7931
- // Quechua
7932
- "maypi",
7933
- // Swahili
7934
- "wakati",
7935
- "wapi"
7936
- ]);
7937
- var THEN_KEYWORDS = /* @__PURE__ */ new Set([
7938
- "then",
7939
- "\u305D\u308C\u304B\u3089",
7940
- "\u90A3\u4E48",
7941
- "\u062B\u0645",
7942
- "entonces",
7943
- "alors",
7944
- "dann",
7945
- "sonra",
7946
- "lalu",
7947
- "chayqa",
7948
- "kisha"
7949
- ]);
7950
7858
 
7951
7859
  // src/parser/create-provider.ts
7952
7860
  function createKeywordProvider(dictionary, locale, options = {}) {
@@ -14179,1770 +14087,6 @@ async function createEnhancedI18n(locale, options) {
14179
14087
  }
14180
14088
  var enhancedI18nImplementation = new TypedI18nContextImplementation();
14181
14089
 
14182
- // src/grammar/direct-mappings.ts
14183
- function reverseMapping(mapping) {
14184
- const reversedWords = {};
14185
- for (const [source, target] of Object.entries(mapping.words)) {
14186
- reversedWords[target] = source;
14187
- }
14188
- const reversedCategories = {};
14189
- if (mapping.categories) {
14190
- for (const [category, words] of Object.entries(mapping.categories)) {
14191
- reversedCategories[category] = {};
14192
- for (const [source, target] of Object.entries(words)) {
14193
- reversedCategories[category][target] = source;
14194
- }
14195
- }
14196
- }
14197
- const result = {
14198
- source: mapping.target,
14199
- target: mapping.source,
14200
- words: reversedWords
14201
- };
14202
- if (Object.keys(reversedCategories).length > 0) {
14203
- result.categories = reversedCategories;
14204
- }
14205
- return result;
14206
- }
14207
- function buildMappingFromDictionaries(sourceDict, targetDict, sourceCode, targetCode) {
14208
- const words = {};
14209
- const categories = {};
14210
- for (const category of [
14211
- "commands",
14212
- "modifiers",
14213
- "events",
14214
- "logical",
14215
- "temporal",
14216
- "values",
14217
- "attributes"
14218
- ]) {
14219
- const sourceCategory = sourceDict[category];
14220
- const targetCategory = targetDict[category];
14221
- if (sourceCategory && targetCategory) {
14222
- categories[category] = {};
14223
- for (const [englishKey, sourceWord] of Object.entries(sourceCategory)) {
14224
- const targetWord = targetCategory[englishKey];
14225
- if (targetWord && sourceWord !== targetWord) {
14226
- words[sourceWord] = targetWord;
14227
- categories[category][sourceWord] = targetWord;
14228
- }
14229
- }
14230
- }
14231
- }
14232
- return {
14233
- source: sourceCode,
14234
- target: targetCode,
14235
- words,
14236
- categories
14237
- };
14238
- }
14239
- var jaZhMapping = buildMappingFromDictionaries(ja, zh, "ja", "zh");
14240
- var zhJaMapping = reverseMapping(jaZhMapping);
14241
- var koJaMapping = buildMappingFromDictionaries(ko, ja, "ko", "ja");
14242
- var jaKoMapping = reverseMapping(koJaMapping);
14243
- var esPtMapping = {
14244
- source: "es",
14245
- target: "pt",
14246
- words: {
14247
- // Commands - many are cognates
14248
- en: "em",
14249
- // on (event)
14250
- decir: "dizer",
14251
- // tell
14252
- disparar: "disparar",
14253
- // trigger
14254
- enviar: "enviar",
14255
- // send
14256
- tomar: "pegar",
14257
- // take
14258
- poner: "colocar",
14259
- // put
14260
- establecer: "definir",
14261
- // set
14262
- obtener: "obter",
14263
- // get
14264
- agregar: "adicionar",
14265
- // add
14266
- quitar: "remover",
14267
- // remove
14268
- alternar: "alternar",
14269
- // toggle
14270
- ocultar: "esconder",
14271
- // hide
14272
- mostrar: "mostrar",
14273
- // show
14274
- si: "se",
14275
- // if
14276
- menos: "a menos",
14277
- // unless
14278
- repetir: "repetir",
14279
- // repeat
14280
- para: "para",
14281
- // for
14282
- mientras: "enquanto",
14283
- // while
14284
- hasta: "at\xE9",
14285
- // until
14286
- continuar: "continuar",
14287
- // continue
14288
- romper: "parar",
14289
- // break
14290
- detener: "parar",
14291
- // halt
14292
- esperar: "esperar",
14293
- // wait
14294
- buscar: "buscar",
14295
- // fetch
14296
- llamar: "chamar",
14297
- // call
14298
- retornar: "retornar",
14299
- // return
14300
- hacer: "fazer",
14301
- // make
14302
- registrar: "registrar",
14303
- // log
14304
- lanzar: "lan\xE7ar",
14305
- // throw
14306
- atrapar: "capturar",
14307
- // catch
14308
- medir: "medir",
14309
- // measure
14310
- transici\u00F3n: "transi\xE7\xE3o",
14311
- // transition
14312
- incrementar: "incrementar",
14313
- // increment
14314
- decrementar: "decrementar",
14315
- // decrement
14316
- predeterminar: "padr\xE3o",
14317
- // default
14318
- ir: "ir",
14319
- // go
14320
- copiar: "copiar",
14321
- // copy
14322
- escoger: "escolher",
14323
- // pick
14324
- intercambiar: "trocar",
14325
- // swap
14326
- transformar: "transformar",
14327
- // morph
14328
- a\u00F1adir: "anexar",
14329
- // append
14330
- salir: "sair",
14331
- // exit
14332
- // Modifiers
14333
- a: "para",
14334
- // to
14335
- de: "de",
14336
- // from
14337
- dentro: "em",
14338
- // into
14339
- con: "com",
14340
- // with
14341
- como: "como",
14342
- // as
14343
- por: "por",
14344
- // by
14345
- // Events
14346
- clic: "clique",
14347
- // click
14348
- cambio: "mudan\xE7a",
14349
- // change
14350
- entrada: "entrada",
14351
- // input
14352
- env\u00EDo: "envio",
14353
- // submit
14354
- carga: "carregar",
14355
- // load
14356
- enfoque: "foco",
14357
- // focus
14358
- desenfoque: "desfoque",
14359
- // blur
14360
- // Logical
14361
- verdadero: "verdadeiro",
14362
- // true
14363
- falso: "falso",
14364
- // false
14365
- y: "e",
14366
- // and
14367
- o: "ou",
14368
- // or
14369
- no: "n\xE3o",
14370
- // not
14371
- es: "\xE9",
14372
- // is
14373
- // Temporal
14374
- segundos: "segundos",
14375
- // seconds
14376
- milisegundos: "milissegundos",
14377
- // milliseconds
14378
- // Values
14379
- nulo: "nulo",
14380
- // null
14381
- indefinido: "indefinido",
14382
- // undefined
14383
- vac\u00EDo: "vazio"
14384
- // empty
14385
- },
14386
- categories: {
14387
- commands: {
14388
- en: "em",
14389
- decir: "dizer",
14390
- tomar: "pegar",
14391
- poner: "colocar",
14392
- establecer: "definir",
14393
- obtener: "obter",
14394
- agregar: "adicionar",
14395
- quitar: "remover",
14396
- ocultar: "esconder",
14397
- si: "se",
14398
- mientras: "enquanto",
14399
- hasta: "at\xE9",
14400
- romper: "parar",
14401
- detener: "parar",
14402
- llamar: "chamar",
14403
- hacer: "fazer",
14404
- lanzar: "lan\xE7ar",
14405
- atrapar: "capturar",
14406
- escoger: "escolher",
14407
- intercambiar: "trocar",
14408
- a\u00F1adir: "anexar",
14409
- salir: "sair"
14410
- },
14411
- events: {
14412
- clic: "clique",
14413
- cambio: "mudan\xE7a",
14414
- carga: "carregar",
14415
- enfoque: "foco",
14416
- desenfoque: "desfoque"
14417
- },
14418
- logical: {
14419
- verdadero: "verdadeiro",
14420
- y: "e",
14421
- o: "ou",
14422
- no: "n\xE3o",
14423
- es: "\xE9"
14424
- }
14425
- }
14426
- };
14427
- var ptEsMapping = reverseMapping(esPtMapping);
14428
- var directMappings = /* @__PURE__ */ new Map([
14429
- // Japanese ↔ Chinese
14430
- ["ja->zh", jaZhMapping],
14431
- ["zh->ja", zhJaMapping],
14432
- // Korean ↔ Japanese
14433
- ["ko->ja", koJaMapping],
14434
- ["ja->ko", jaKoMapping],
14435
- // Spanish ↔ Portuguese
14436
- ["es->pt", esPtMapping],
14437
- ["pt->es", ptEsMapping]
14438
- ]);
14439
- function hasDirectMapping(source, target) {
14440
- return directMappings.has(`${source}->${target}`);
14441
- }
14442
- function getDirectMapping(source, target) {
14443
- return directMappings.get(`${source}->${target}`);
14444
- }
14445
-
14446
- // src/grammar/transformer.ts
14447
- function getCommandKeywordsForLocale(locale) {
14448
- const keywords = new Set(ENGLISH_COMMANDS);
14449
- const dict = dictionaries[locale];
14450
- if (dict?.commands) {
14451
- Object.values(dict.commands).forEach((cmd) => {
14452
- if (typeof cmd === "string") {
14453
- keywords.add(cmd.toLowerCase());
14454
- }
14455
- });
14456
- }
14457
- return keywords;
14458
- }
14459
- var EN_COPULAS = ["is", "are", "was", "were", "am", "be"];
14460
- function getCopulasForLocale(locale) {
14461
- const copulas = new Set(EN_COPULAS);
14462
- if (locale !== "en") {
14463
- for (const form of EN_COPULAS) {
14464
- copulas.add(translateWord(form, "en", locale).toLowerCase());
14465
- }
14466
- }
14467
- return copulas;
14468
- }
14469
- function isPredicateAdjectivePosition(tokens, i, copulas) {
14470
- const prev = tokens[i - 1]?.toLowerCase();
14471
- return !!prev && copulas.has(prev);
14472
- }
14473
- function getForLoopWordsForLocale(locale) {
14474
- const forWords = /* @__PURE__ */ new Set(["for"]);
14475
- const inWords = /* @__PURE__ */ new Set(["in"]);
14476
- if (locale !== "en") {
14477
- forWords.add(translateWord("for", "en", locale).toLowerCase());
14478
- inWords.add(translateWord("in", "en", locale).toLowerCase());
14479
- }
14480
- return { forWords, inWords };
14481
- }
14482
- function isLoopHeadFor(tokens, i, inWords, commandKeywords) {
14483
- for (let j = i + 1; j < tokens.length; j++) {
14484
- const lt = tokens[j].toLowerCase();
14485
- if (inWords.has(lt)) return true;
14486
- if (commandKeywords.has(lt)) return false;
14487
- }
14488
- return false;
14489
- }
14490
- function repairHebrewFrontedAccusative(text) {
14491
- const ACC = "\u05D0\u05EA";
14492
- const verbs = getCommandKeywordsForLocale("he");
14493
- const tokens = text.split(/\s+/);
14494
- let changed = false;
14495
- for (let i = 0; i + 1 < tokens.length; i++) {
14496
- if (tokens[i] === ACC && verbs.has(tokens[i + 1].toLowerCase())) {
14497
- [tokens[i], tokens[i + 1]] = [tokens[i + 1], tokens[i]];
14498
- changed = true;
14499
- i++;
14500
- }
14501
- }
14502
- return changed ? tokens.join(" ") : text;
14503
- }
14504
- function extractBlockStructure(input, sourceLocale) {
14505
- const tokens = input.split(/\s+/);
14506
- const head = tokens[0]?.toLowerCase();
14507
- if (!head || !BLOCK_HEAD_KEYWORDS.has(head)) return null;
14508
- let depth = 1;
14509
- let endIdx = -1;
14510
- for (let i = 1; i < tokens.length; i++) {
14511
- const t = tokens[i].toLowerCase();
14512
- if (BLOCK_HEAD_KEYWORDS.has(t)) depth++;
14513
- else if (t === "end") {
14514
- depth--;
14515
- if (depth === 0) {
14516
- endIdx = i;
14517
- break;
14518
- }
14519
- }
14520
- }
14521
- if (endIdx !== -1 && endIdx !== tokens.length - 1) return null;
14522
- const inner = endIdx !== -1 ? tokens.slice(1, endIdx) : tokens.slice(1);
14523
- const base = { headKeyword: tokens[0], body: "" };
14524
- if (endIdx !== -1) base.tailKeyword = tokens[endIdx];
14525
- if (head === "live") {
14526
- return { ...base, body: inner.join(" ") };
14527
- }
14528
- if (head === "when") {
14529
- const idx = inner.findIndex((t) => t.toLowerCase() === "changes");
14530
- if (idx >= 0) {
14531
- return {
14532
- ...base,
14533
- prefixExpr: inner.slice(0, idx).join(" "),
14534
- connector: inner[idx],
14535
- body: inner.slice(idx + 1).join(" ")
14536
- };
14537
- }
14538
- return null;
14539
- }
14540
- const commands = getCommandKeywordsForLocale(sourceLocale);
14541
- const copulas = getCopulasForLocale(sourceLocale);
14542
- let bodyStart = -1;
14543
- for (let i = 0; i < inner.length; i++) {
14544
- if (commands.has(inner[i].toLowerCase()) && !isPredicateAdjectivePosition(inner, i, copulas)) {
14545
- bodyStart = i;
14546
- break;
14547
- }
14548
- }
14549
- if (bodyStart <= 0) return null;
14550
- return {
14551
- ...base,
14552
- prefixExpr: inner.slice(0, bodyStart).join(" "),
14553
- body: inner.slice(bodyStart).join(" ")
14554
- };
14555
- }
14556
- function splitCompoundStatement(input, sourceLocale) {
14557
- const lines = input.split(/\n/).map((line) => line.trim()).filter((line) => line.length > 0);
14558
- const parts = [];
14559
- for (const line of lines) {
14560
- const lineParts = splitOnThen(line, sourceLocale);
14561
- for (const part of lineParts) {
14562
- const commandParts = splitOnCommandBoundaries(part, sourceLocale);
14563
- parts.push(...commandParts);
14564
- }
14565
- }
14566
- return parts;
14567
- }
14568
- function splitCompoundStatementWithMetadata(input, sourceLocale) {
14569
- const rawLines = input.split("\n");
14570
- const lineMetadata = [];
14571
- const parts = [];
14572
- const partToLineIndex = [];
14573
- for (let lineIndex = 0; lineIndex < rawLines.length; lineIndex++) {
14574
- const rawLine = rawLines[lineIndex];
14575
- const indentMatch = rawLine.match(/^(\s*)/);
14576
- const originalIndent = indentMatch ? indentMatch[1] : "";
14577
- const trimmed = rawLine.trim();
14578
- lineMetadata.push({
14579
- content: trimmed,
14580
- originalIndent,
14581
- isBlank: trimmed.length === 0
14582
- });
14583
- if (trimmed.length > 0) {
14584
- const lineParts = splitOnThen(trimmed, sourceLocale);
14585
- for (const part of lineParts) {
14586
- const commandParts = splitOnCommandBoundaries(part, sourceLocale);
14587
- for (const cmdPart of commandParts) {
14588
- parts.push(cmdPart);
14589
- partToLineIndex.push(lineIndex);
14590
- }
14591
- }
14592
- }
14593
- }
14594
- return { parts, lineMetadata, partToLineIndex };
14595
- }
14596
- function normalizeIndentation(lineMetadata) {
14597
- const indentedLines = lineMetadata.filter((m) => !m.isBlank && m.originalIndent.length > 0);
14598
- if (indentedLines.length === 0) {
14599
- return lineMetadata.map(() => "");
14600
- }
14601
- const indentLengths = indentedLines.map((m) => {
14602
- const normalized = m.originalIndent.replace(/\t/g, " ");
14603
- return normalized.length;
14604
- });
14605
- const minIndent = Math.min(...indentLengths);
14606
- const baseUnit = minIndent > 0 ? minIndent : 4;
14607
- return lineMetadata.map((meta) => {
14608
- if (meta.isBlank) {
14609
- return "";
14610
- }
14611
- if (meta.originalIndent.length === 0) {
14612
- return "";
14613
- }
14614
- const normalized = meta.originalIndent.replace(/\t/g, " ");
14615
- const level = Math.round(normalized.length / baseUnit);
14616
- return " ".repeat(level);
14617
- });
14618
- }
14619
- function reconstructWithLineStructure(transformedParts, lineMetadata, partToLineIndex, targetThen) {
14620
- const nonBlankCount = lineMetadata.filter((m) => !m.isBlank).length;
14621
- if (nonBlankCount <= 1 && transformedParts.length <= 1) {
14622
- const normalizedIndents2 = normalizeIndentation(lineMetadata);
14623
- const result2 = [];
14624
- for (let i = 0; i < lineMetadata.length; i++) {
14625
- if (lineMetadata[i].isBlank) {
14626
- result2.push("");
14627
- } else if (transformedParts.length > 0) {
14628
- result2.push(normalizedIndents2[i] + transformedParts[0]);
14629
- }
14630
- }
14631
- return result2.join("\n");
14632
- }
14633
- const normalizedIndents = normalizeIndentation(lineMetadata);
14634
- const partsPerLine = /* @__PURE__ */ new Map();
14635
- for (let i = 0; i < transformedParts.length; i++) {
14636
- const lineIdx = partToLineIndex[i];
14637
- if (!partsPerLine.has(lineIdx)) {
14638
- partsPerLine.set(lineIdx, []);
14639
- }
14640
- partsPerLine.get(lineIdx).push(transformedParts[i]);
14641
- }
14642
- const result = [];
14643
- for (let i = 0; i < lineMetadata.length; i++) {
14644
- const meta = lineMetadata[i];
14645
- const indent = normalizedIndents[i];
14646
- if (meta.isBlank) {
14647
- result.push("");
14648
- } else {
14649
- const lineParts = partsPerLine.get(i) || [];
14650
- if (lineParts.length > 0) {
14651
- const lineContent = lineParts.join(` ${targetThen} `);
14652
- result.push(indent + lineContent);
14653
- }
14654
- }
14655
- }
14656
- return result.join("\n");
14657
- }
14658
- var BOUNDARY_MODIFIERS = /* @__PURE__ */ new Set([
14659
- "to",
14660
- "into",
14661
- "from",
14662
- "with",
14663
- "by",
14664
- "as",
14665
- "at",
14666
- "in",
14667
- "on",
14668
- "of",
14669
- "over"
14670
- ]);
14671
- var boundaryModifiersCache = /* @__PURE__ */ new Map();
14672
- function getBoundaryModifiersForLocale(locale) {
14673
- const cached = boundaryModifiersCache.get(locale);
14674
- if (cached) return cached;
14675
- const modifiers = new Set(BOUNDARY_MODIFIERS);
14676
- const profile = getProfile(locale);
14677
- profile?.markers.forEach((marker) => {
14678
- const form = marker.form.replace(/^-|-$/g, "").toLowerCase();
14679
- if (form) modifiers.add(form);
14680
- marker.alternatives?.forEach((alt) => {
14681
- const altForm = alt.replace(/^-|-$/g, "").toLowerCase();
14682
- if (altForm) modifiers.add(altForm);
14683
- });
14684
- });
14685
- boundaryModifiersCache.set(locale, modifiers);
14686
- return modifiers;
14687
- }
14688
- var ON_TARGET_COMMANDS = /* @__PURE__ */ new Set(["toggle", "add", "remove", "trigger", "send"]);
14689
- function commandVerbOf(tokens, commandKeywords) {
14690
- for (const token of tokens) {
14691
- const lt = token.toLowerCase();
14692
- if (commandKeywords.has(lt) && !BOUNDARY_MODIFIERS.has(lt) && !EVENT_KEYWORDS.has(lt)) {
14693
- return lt;
14694
- }
14695
- }
14696
- return null;
14697
- }
14698
- var BLOCK_HEAD_KEYWORDS = /* @__PURE__ */ new Set(["live", "when", "unless"]);
14699
- var BLOCK_BODY_KEYWORDS = /* @__PURE__ */ new Set(["if", "repeat", "unless", "while", "for"]);
14700
- var UNLESS_GUARD_OBJECT_MARKING_LOCALES = /* @__PURE__ */ new Set(["he", "zh"]);
14701
- function splitOnCommandBoundaries(input, sourceLocale) {
14702
- const commandKeywords = getCommandKeywordsForLocale(sourceLocale);
14703
- const boundaryModifiers = getBoundaryModifiersForLocale(sourceLocale);
14704
- const { forWords, inWords } = getForLoopWordsForLocale(sourceLocale);
14705
- const tokens = input.split(/\s+/);
14706
- if (tokens.length === 0) return [input];
14707
- const parts = [];
14708
- let currentPart = [];
14709
- const firstTokenLower = tokens[0]?.toLowerCase();
14710
- const isEventHandler = EVENT_KEYWORDS.has(firstTokenLower);
14711
- let seenFirstCommand = !isEventHandler;
14712
- let blockDepth = 0;
14713
- for (let i = 0; i < tokens.length; i++) {
14714
- const token = tokens[i];
14715
- const lowerToken = token.toLowerCase();
14716
- if (BLOCK_HEAD_KEYWORDS.has(lowerToken)) {
14717
- blockDepth++;
14718
- } else if (lowerToken === "end" && blockDepth > 0) {
14719
- blockDepth--;
14720
- }
14721
- if (commandKeywords.has(lowerToken) && currentPart.length > 0) {
14722
- const prevToken = currentPart[currentPart.length - 1];
14723
- const prevLower = prevToken.toLowerCase();
14724
- if (!seenFirstCommand) {
14725
- seenFirstCommand = true;
14726
- currentPart.push(token);
14727
- continue;
14728
- }
14729
- if (blockDepth > 0) {
14730
- currentPart.push(token);
14731
- continue;
14732
- }
14733
- if (forWords.has(lowerToken) && !isLoopHeadFor(tokens, i, inWords, commandKeywords)) {
14734
- currentPart.push(token);
14735
- continue;
14736
- }
14737
- if (BOUNDARY_MODIFIERS.has(lowerToken)) {
14738
- const verb = commandVerbOf(currentPart, commandKeywords);
14739
- if (verb && ON_TARGET_COMMANDS.has(verb)) {
14740
- currentPart.push(token);
14741
- continue;
14742
- }
14743
- if (lowerToken === "on" && verb === "set") {
14744
- const nextTok = tokens[i + 1];
14745
- const nextLower = nextTok?.toLowerCase();
14746
- const scopeLike = !!nextTok && (/^[#.<@[]/.test(nextTok) || nextLower === "me" || nextLower === "it" || nextLower === "you");
14747
- if (scopeLike) {
14748
- currentPart.push(token);
14749
- continue;
14750
- }
14751
- }
14752
- }
14753
- if (!boundaryModifiers.has(prevLower) && !commandKeywords.has(prevLower)) {
14754
- parts.push(currentPart.join(" "));
14755
- currentPart = [token];
14756
- continue;
14757
- }
14758
- }
14759
- currentPart.push(token);
14760
- }
14761
- if (currentPart.length > 0) {
14762
- parts.push(currentPart.join(" "));
14763
- }
14764
- return parts.filter((p) => p.length > 0);
14765
- }
14766
- function splitOnThen(input, sourceLocale) {
14767
- const thenKeywords = Array.from(THEN_KEYWORDS);
14768
- const sourceDict = sourceLocale === "en" ? null : dictionaries[sourceLocale];
14769
- if (sourceDict?.modifiers?.then) {
14770
- thenKeywords.push(sourceDict.modifiers.then);
14771
- }
14772
- if (sourceDict?.logical?.then) {
14773
- thenKeywords.push((sourceDict?.logical).then);
14774
- }
14775
- const escapedKeywords = thenKeywords.map((k) => k.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"));
14776
- const pattern = new RegExp(`\\s+(${escapedKeywords.join("|")})\\s+`, "gi");
14777
- const parts = input.split(pattern).filter((part) => {
14778
- const lowerPart = part.toLowerCase().trim();
14779
- return lowerPart && !thenKeywords.some((k) => k.toLowerCase() === lowerPart);
14780
- });
14781
- return parts.map((p) => p.trim()).filter((p) => p.length > 0);
14782
- }
14783
- function getTargetThenKeyword(targetLocale) {
14784
- if (targetLocale === "en") return "then";
14785
- const targetDict = dictionaries[targetLocale];
14786
- if (!targetDict) return "then";
14787
- return targetDict.modifiers?.then || targetDict.logical?.then || "then";
14788
- }
14789
- function deriveEventKeywordsFromProfiles() {
14790
- const keywords = /* @__PURE__ */ new Set();
14791
- keywords.add("on");
14792
- for (const profile of Object.values(profiles)) {
14793
- for (const marker of profile.markers) {
14794
- if (marker.role === "event") {
14795
- const form = marker.form.replace(/^-|-$/g, "").toLowerCase();
14796
- if (form) keywords.add(form);
14797
- marker.alternatives?.forEach((alt) => {
14798
- const altForm = alt.replace(/^-|-$/g, "").toLowerCase();
14799
- if (altForm) keywords.add(altForm);
14800
- });
14801
- }
14802
- }
14803
- }
14804
- return keywords;
14805
- }
14806
- var EVENT_KEYWORDS = deriveEventKeywordsFromProfiles();
14807
- var EVENT_CONJUNCTIONS = /* @__PURE__ */ new Set(["or"]);
14808
- var BODY_MODIFIER_KEYWORDS = /* @__PURE__ */ new Set([
14809
- "async",
14810
- "once",
14811
- "debounced",
14812
- "debounce",
14813
- "throttled",
14814
- "throttle"
14815
- ]);
14816
- function generateModifierMap(profile) {
14817
- const map = {};
14818
- profile.markers.forEach((marker) => {
14819
- const form = marker.form.replace(/^-|-$/g, "").toLowerCase();
14820
- if (form) {
14821
- map[form] = marker.role;
14822
- }
14823
- marker.alternatives?.forEach((alt) => {
14824
- const altForm = alt.replace(/^-|-$/g, "").toLowerCase();
14825
- if (altForm) {
14826
- map[altForm] = marker.role;
14827
- }
14828
- });
14829
- });
14830
- for (const [key, role] of Object.entries(ENGLISH_MODIFIER_ROLES)) {
14831
- if (!(key in map)) {
14832
- map[key] = role;
14833
- }
14834
- }
14835
- return map;
14836
- }
14837
- function buildArgumentModifierMap(profile, actionVerb) {
14838
- const map = generateModifierMap(profile);
14839
- const verb = actionVerb?.toLowerCase();
14840
- if (profile.wordOrder !== "SVO" || !verb || !ON_TARGET_COMMANDS.has(verb)) {
14841
- return map;
14842
- }
14843
- const remapped = {};
14844
- for (const [form, role] of Object.entries(map)) {
14845
- remapped[form] = role === "event" ? "destination" : role;
14846
- }
14847
- return remapped;
14848
- }
14849
- function parseStatement(input, sourceLocale = "en") {
14850
- const profile = getProfile(sourceLocale);
14851
- if (!profile) return null;
14852
- const tokens = tokenize2(input, profile);
14853
- const statementType = identifyStatementType(tokens, profile);
14854
- switch (statementType) {
14855
- case "event-handler":
14856
- return parseEventHandler(tokens, profile);
14857
- case "command":
14858
- return parseCommand(tokens, profile);
14859
- case "conditional":
14860
- return parseConditional(tokens);
14861
- default:
14862
- return null;
14863
- }
14864
- }
14865
- var ATTACHED_SUFFIXES = {
14866
- // Chinese: 时 (time/when) often attaches to events like 点击时 (when clicking)
14867
- zh: ["\u65F6", "\u7684", "\u5730", "\u5F97"],
14868
- // Japanese: Some particles may attach in casual writing
14869
- ja: [],
14870
- // Korean: Particles sometimes written without spaces
14871
- ko: []
14872
- };
14873
- var ATTACHED_PREFIXES = {
14874
- // Chinese: 当 (when) sometimes written attached
14875
- zh: ["\u5F53"],
14876
- // Arabic: Prepositions that attach
14877
- ar: ["\u0628\u0640", "\u0643\u0640", "\u0648"]
14878
- };
14879
- function splitAttachedAffixes(tokens, locale) {
14880
- const suffixes = ATTACHED_SUFFIXES[locale] || [];
14881
- const prefixes = ATTACHED_PREFIXES[locale] || [];
14882
- if (suffixes.length === 0 && prefixes.length === 0) {
14883
- return tokens;
14884
- }
14885
- const result = [];
14886
- for (const token of tokens) {
14887
- if (/^[#.<@]/.test(token) || /^\d+/.test(token)) {
14888
- result.push(token);
14889
- continue;
14890
- }
14891
- let processed = token;
14892
- let prefix = "";
14893
- let suffix = "";
14894
- for (const p of prefixes) {
14895
- if (processed.startsWith(p) && processed.length > p.length) {
14896
- prefix = p;
14897
- processed = processed.slice(p.length);
14898
- break;
14899
- }
14900
- }
14901
- for (const s of suffixes) {
14902
- if (processed.endsWith(s) && processed.length > s.length) {
14903
- suffix = s;
14904
- processed = processed.slice(0, -s.length);
14905
- break;
14906
- }
14907
- }
14908
- if (prefix) result.push(prefix);
14909
- if (processed) result.push(processed);
14910
- if (suffix) result.push(suffix);
14911
- }
14912
- return result;
14913
- }
14914
- function tokenize2(input, profile) {
14915
- const tokens = [];
14916
- let current = "";
14917
- let inSelector = false;
14918
- let selectorDepth = 0;
14919
- let bracketDepth = 0;
14920
- let parenDepth = 0;
14921
- for (let i = 0; i < input.length; i++) {
14922
- const char = input[i];
14923
- if (char === "<") {
14924
- inSelector = true;
14925
- selectorDepth++;
14926
- } else if (char === ">" && inSelector) {
14927
- selectorDepth--;
14928
- if (selectorDepth === 0) inSelector = false;
14929
- }
14930
- if (char === "[") {
14931
- bracketDepth++;
14932
- } else if (char === "]" && bracketDepth > 0) {
14933
- bracketDepth--;
14934
- }
14935
- if (char === "(") {
14936
- parenDepth++;
14937
- } else if (char === ")" && parenDepth > 0) {
14938
- parenDepth--;
14939
- }
14940
- if (/\s/.test(char) && !inSelector && bracketDepth === 0 && parenDepth === 0) {
14941
- if (current) {
14942
- tokens.push(current);
14943
- current = "";
14944
- }
14945
- } else {
14946
- current += char;
14947
- }
14948
- }
14949
- if (current) {
14950
- tokens.push(current);
14951
- }
14952
- return splitAttachedAffixes(tokens, profile.code);
14953
- }
14954
- function identifyStatementType(tokens, profile) {
14955
- if (tokens.length === 0) return "unknown";
14956
- const firstToken = tokens[0].toLowerCase();
14957
- const eventMarker = profile.markers.find((m) => m.role === "event" && m.position === "preposition");
14958
- if (eventMarker && firstToken === eventMarker.form.toLowerCase()) {
14959
- return "event-handler";
14960
- }
14961
- if (EVENT_KEYWORDS.has(firstToken)) {
14962
- return "event-handler";
14963
- }
14964
- if (CONDITIONAL_KEYWORDS.has(firstToken)) {
14965
- return "conditional";
14966
- }
14967
- return "command";
14968
- }
14969
- function parseEventHandler(tokens, profile) {
14970
- const roles = /* @__PURE__ */ new Map();
14971
- let startIndex = EVENT_KEYWORDS.has(tokens[0]?.toLowerCase()) ? 1 : 0;
14972
- if (tokens[startIndex]) {
14973
- const eventTokens = [tokens[startIndex]];
14974
- startIndex++;
14975
- while (tokens[startIndex] && EVENT_CONJUNCTIONS.has(tokens[startIndex].toLowerCase()) && tokens[startIndex + 1]) {
14976
- eventTokens.push(tokens[startIndex], tokens[startIndex + 1]);
14977
- startIndex += 2;
14978
- }
14979
- roles.set("event", {
14980
- role: "event",
14981
- value: eventTokens.join(" ")
14982
- });
14983
- }
14984
- if (tokens[startIndex] && tokens[startIndex].toLowerCase() === "from" && tokens[startIndex + 1]) {
14985
- startIndex++;
14986
- const sourceValue = [];
14987
- while (tokens[startIndex]) {
14988
- if (ENGLISH_COMMANDS.has(tokens[startIndex].toLowerCase())) break;
14989
- sourceValue.push(tokens[startIndex]);
14990
- startIndex++;
14991
- }
14992
- if (sourceValue.length > 0) {
14993
- const value = sourceValue.join(" ");
14994
- roles.set("source", {
14995
- role: "source",
14996
- value,
14997
- isSelector: /^[#.<@]/.test(value)
14998
- });
14999
- }
15000
- }
15001
- if (tokens[startIndex]) {
15002
- roles.set("action", {
15003
- role: "action",
15004
- value: tokens[startIndex]
15005
- });
15006
- startIndex++;
15007
- }
15008
- if (tokens[startIndex]) {
15009
- const modifierMap = buildArgumentModifierMap(profile, roles.get("action")?.value);
15010
- let currentRole = "patient";
15011
- let currentValue = [];
15012
- for (let i = startIndex; i < tokens.length; i++) {
15013
- const token = tokens[i];
15014
- const mappedRole = modifierMap[token.toLowerCase()];
15015
- if (mappedRole) {
15016
- if (currentValue.length > 0) {
15017
- const value = currentValue.join(" ");
15018
- roles.set(currentRole, {
15019
- role: currentRole,
15020
- value,
15021
- isSelector: /^[#.<@]/.test(value)
15022
- });
15023
- }
15024
- currentRole = mappedRole;
15025
- currentValue = [];
15026
- } else {
15027
- currentValue.push(token);
15028
- }
15029
- }
15030
- if (currentValue.length > 0) {
15031
- const value = currentValue.join(" ");
15032
- roles.set(currentRole, {
15033
- role: currentRole,
15034
- value,
15035
- isSelector: /^[#.<@]/.test(value)
15036
- });
15037
- }
15038
- }
15039
- return {
15040
- type: "event-handler",
15041
- roles,
15042
- original: tokens.join(" ")
15043
- };
15044
- }
15045
- function parseCommand(tokens, profile) {
15046
- const roles = /* @__PURE__ */ new Map();
15047
- if (tokens.length === 0) {
15048
- return { type: "command", roles, original: "" };
15049
- }
15050
- roles.set("action", {
15051
- role: "action",
15052
- value: tokens[0]
15053
- });
15054
- const modifierMap = buildArgumentModifierMap(profile, tokens[0]);
15055
- let currentRole = "patient";
15056
- let currentValue = [];
15057
- for (let i = 1; i < tokens.length; i++) {
15058
- const token = tokens[i];
15059
- const mappedRole = modifierMap[token.toLowerCase()];
15060
- if (mappedRole) {
15061
- if (currentValue.length > 0) {
15062
- const value = currentValue.join(" ");
15063
- roles.set(currentRole, {
15064
- role: currentRole,
15065
- value,
15066
- isSelector: /^[#.<@]/.test(value)
15067
- });
15068
- }
15069
- currentRole = mappedRole;
15070
- currentValue = [];
15071
- } else {
15072
- currentValue.push(token);
15073
- }
15074
- }
15075
- if (currentValue.length > 0) {
15076
- const value = currentValue.join(" ");
15077
- roles.set(currentRole, {
15078
- role: currentRole,
15079
- value,
15080
- isSelector: /^[#.<@]/.test(value)
15081
- });
15082
- }
15083
- return {
15084
- type: "command",
15085
- roles,
15086
- original: tokens.join(" ")
15087
- };
15088
- }
15089
- var LITERAL_PRIMARY_ROLES = /* @__PURE__ */ new Set([
15090
- "duration",
15091
- "quantity"
15092
- ]);
15093
- function applyPrimaryRole(parsed, targetProfile) {
15094
- if (parsed.type !== "command") return;
15095
- const action = parsed.roles.get("action")?.value;
15096
- if (!action) return;
15097
- const primaryRole = COMMAND_PRIMARY_ROLES[action.toLowerCase()];
15098
- if (!primaryRole || !LITERAL_PRIMARY_ROLES.has(primaryRole)) return;
15099
- const patientEl = parsed.roles.get("patient");
15100
- if (!patientEl || parsed.roles.has(primaryRole)) return;
15101
- if (targetProfile.markers.some((m) => m.role === primaryRole)) return;
15102
- parsed.roles.delete("patient");
15103
- parsed.roles.set(primaryRole, { ...patientEl, role: primaryRole });
15104
- }
15105
- function parseConditional(tokens, _profile) {
15106
- const roles = /* @__PURE__ */ new Map();
15107
- roles.set("action", {
15108
- role: "action",
15109
- value: tokens[0]
15110
- });
15111
- const thenIndex = tokens.findIndex((t) => THEN_KEYWORDS.has(t.toLowerCase()));
15112
- if (thenIndex > 1) {
15113
- const conditionValue = tokens.slice(1, thenIndex).join(" ");
15114
- roles.set("condition", {
15115
- role: "condition",
15116
- value: conditionValue
15117
- });
15118
- } else if (thenIndex === -1 && tokens.length > 1) {
15119
- roles.set("condition", {
15120
- role: "condition",
15121
- value: tokens.slice(1).join(" ")
15122
- });
15123
- }
15124
- return {
15125
- type: "conditional",
15126
- roles,
15127
- original: tokens.join(" ")
15128
- };
15129
- }
15130
- function translateWord(word, sourceLocale, targetLocale) {
15131
- if (/^[#.<@]/.test(word)) {
15132
- return word;
15133
- }
15134
- if (/^\d+/.test(word)) {
15135
- return word;
15136
- }
15137
- if (/\s/.test(word) && word.startsWith("(")) {
15138
- return word.split(/\s+/).map((w) => translateWord(w, sourceLocale, targetLocale)).join(" ");
15139
- }
15140
- if (word.length > 1 && (word.startsWith("(") || word.endsWith(")"))) {
15141
- const m = word.match(/^(\(*)([^()]+)(\)*)$/);
15142
- if (m && (m[1] || m[3])) {
15143
- return m[1] + translateWord(m[2], sourceLocale, targetLocale) + m[3];
15144
- }
15145
- }
15146
- const sourceDict = sourceLocale === "en" ? null : dictionaries[sourceLocale];
15147
- const targetDict = dictionaries[targetLocale];
15148
- if (!targetDict) return word;
15149
- let englishWord = word;
15150
- if (sourceDict) {
15151
- const found = findInDictionary(sourceDict, word);
15152
- if (found) {
15153
- englishWord = found.englishKey;
15154
- }
15155
- }
15156
- const translated = translateFromEnglish(targetDict, englishWord);
15157
- return translated ?? word;
15158
- }
15159
- var POSSESSIVE_MARKERS = {
15160
- en: { type: "suffix", marker: "'s" },
15161
- es: { type: "preposition", marker: "de" },
15162
- pt: { type: "preposition", marker: "de" },
15163
- fr: { type: "preposition", marker: "de" },
15164
- de: { type: "preposition", marker: "von" },
15165
- ja: { type: "suffix", marker: "\u306E" },
15166
- ko: { type: "suffix", marker: "\uC758" },
15167
- zh: { type: "suffix", marker: "\u7684" },
15168
- ar: { type: "preposition", marker: "\u0644\u0640" },
15169
- // Spaced genitive particle (not the glued `'ın`), so the tokenizer can split
15170
- // it off the selector — consistent with Turkish's other spaced case markers.
15171
- tr: { type: "particle", marker: "\u0131n" },
15172
- id: { type: "preposition", marker: "dari" },
15173
- // Latin-script genitive: must be a *spaced* particle (`#picker pa`), since a
15174
- // glued `#pickerpa` can't be split from the selector by the tokenizer the
15175
- // way a non-Latin suffix (の/의/র) can.
15176
- qu: { type: "particle", marker: "pa" },
15177
- // Bengali SOV postposition genitive, like ja/ko — a spaced suffix the
15178
- // tokenizer splits off as a particle. Previously absent, so it fell back to
15179
- // the English `'s` marker and its possessive property paths never parsed.
15180
- // (Hindi `का` is intentionally omitted: its `bind` lacks a verb-final
15181
- // grammar rule, so fixing its possessive alone yields a wrong `on` parse —
15182
- // tracked as separate follow-up.)
15183
- bn: { type: "suffix", marker: "\u09B0" },
15184
- sw: { type: "preposition", marker: "ya" }
15185
- };
15186
- function translatePossessive(token, sourceLocale, targetLocale) {
15187
- const possessiveMatch = token.match(/^(.+)'s$/i);
15188
- if (!possessiveMatch) {
15189
- return token;
15190
- }
15191
- const owner = possessiveMatch[1];
15192
- const targetMarker = POSSESSIVE_MARKERS[targetLocale] || POSSESSIVE_MARKERS.en;
15193
- const pronounPossessives = {
15194
- me: "my",
15195
- it: "its",
15196
- you: "your"
15197
- };
15198
- const lowerOwner = owner.toLowerCase();
15199
- if (pronounPossessives[lowerOwner]) {
15200
- const possessiveForm = pronounPossessives[lowerOwner];
15201
- return translateWord(possessiveForm, "en", targetLocale);
15202
- }
15203
- const translatedOwner = translateWord(owner, sourceLocale, targetLocale);
15204
- switch (targetMarker.type) {
15205
- case "suffix":
15206
- return `${translatedOwner}${targetMarker.marker}`;
15207
- case "particle":
15208
- return `${translatedOwner} ${targetMarker.marker}`;
15209
- case "preposition":
15210
- return `__POSS__${targetMarker.marker}__${translatedOwner}__POSS__`;
15211
- default:
15212
- return `${translatedOwner}'s`;
15213
- }
15214
- }
15215
- var POSSESSIVE_DOT_REGEX = /^(my|its|your|me|it|you)(\??\..+)$/i;
15216
- var POSSESSIVE_DOT_PRONOUNS = {
15217
- me: "my",
15218
- it: "its",
15219
- you: "your",
15220
- my: "my",
15221
- its: "its",
15222
- your: "your"
15223
- };
15224
- function translatePossessiveDotNotation(value, sourceLocale, targetLocale) {
15225
- const match = value.match(POSSESSIVE_DOT_REGEX);
15226
- if (!match) return null;
15227
- const possessiveWord = match[1].toLowerCase();
15228
- const propertySuffix = match[2];
15229
- const possessiveKey = POSSESSIVE_DOT_PRONOUNS[possessiveWord] || possessiveWord;
15230
- const translated = translateWord(possessiveKey, sourceLocale, targetLocale);
15231
- if (translated.includes(" ")) return null;
15232
- if (translated !== possessiveKey) {
15233
- return translated + propertySuffix;
15234
- }
15235
- if (possessiveWord !== possessiveKey) {
15236
- const alt = translateWord(possessiveWord, sourceLocale, targetLocale);
15237
- if (alt !== possessiveWord && !alt.includes(" ")) {
15238
- return alt + propertySuffix;
15239
- }
15240
- }
15241
- return null;
15242
- }
15243
- function translateMultiWordValue(value, sourceLocale, targetLocale) {
15244
- if (value.includes("[")) {
15245
- const guards = [];
15246
- const masked = value.replace(/\[[^\]]*\]/g, (match) => {
15247
- guards.push(match);
15248
- return `\uE000${guards.length - 1}\uE001`;
15249
- });
15250
- if (guards.length > 0) {
15251
- const translated2 = translateMultiWordValue(masked, sourceLocale, targetLocale);
15252
- return translated2.replace(/(\d+)/g, (_, n) => guards[Number(n)]);
15253
- }
15254
- }
15255
- if (!value.includes(" ")) {
15256
- if (value.includes("'s")) {
15257
- return translatePossessive(value, sourceLocale, targetLocale);
15258
- }
15259
- const dotResult = translatePossessiveDotNotation(value, sourceLocale, targetLocale);
15260
- if (dotResult !== null) return dotResult;
15261
- return translateWord(value, sourceLocale, targetLocale);
15262
- }
15263
- const words = value.split(/\s+/);
15264
- const translated = [];
15265
- let i = 0;
15266
- while (i < words.length) {
15267
- const word = words[i];
15268
- if (word.includes("'s")) {
15269
- const possessiveResult = translatePossessive(word, sourceLocale, targetLocale);
15270
- const prepMatch = possessiveResult.match(/^__POSS__(.+)__(.+)__POSS__$/);
15271
- if (prepMatch && i + 1 < words.length) {
15272
- const marker = prepMatch[1];
15273
- const owner = prepMatch[2];
15274
- const property = words[i + 1];
15275
- const translatedProperty = translateWord(property, sourceLocale, targetLocale);
15276
- translated.push(`${translatedProperty} ${marker} ${owner}`);
15277
- i += 2;
15278
- continue;
15279
- } else if (prepMatch) {
15280
- const marker = prepMatch[1];
15281
- const owner = prepMatch[2];
15282
- translated.push(`${marker} ${owner}`);
15283
- i++;
15284
- continue;
15285
- }
15286
- translated.push(possessiveResult);
15287
- i++;
15288
- continue;
15289
- }
15290
- if (/^[#.<@]/.test(word) || /^\d+/.test(word)) {
15291
- translated.push(word);
15292
- i++;
15293
- continue;
15294
- }
15295
- if (/^["'].*["']$/.test(word)) {
15296
- translated.push(word);
15297
- i++;
15298
- continue;
15299
- }
15300
- const dotResult = translatePossessiveDotNotation(word, sourceLocale, targetLocale);
15301
- if (dotResult !== null) {
15302
- translated.push(dotResult);
15303
- i++;
15304
- continue;
15305
- }
15306
- translated.push(translateWord(word, sourceLocale, targetLocale));
15307
- i++;
15308
- }
15309
- return translated.join(" ");
15310
- }
15311
- function translateElements(parsed, sourceLocale, targetLocale) {
15312
- for (const [_role, element] of parsed.roles) {
15313
- if (element.value.includes("'s")) {
15314
- element.translated = translateMultiWordValue(element.value, sourceLocale, targetLocale);
15315
- } else if (!element.isSelector && !element.isLiteral) {
15316
- element.translated = translateMultiWordValue(element.value, sourceLocale, targetLocale);
15317
- } else {
15318
- element.translated = element.value;
15319
- }
15320
- }
15321
- }
15322
- var CARET_SCOPE_OPEN = "\uE000";
15323
- var CARET_SCOPE_CLOSE = "\uE001";
15324
- var CARET_SCOPE_RE = /(\^[A-Za-z_][\w-]*)(\s+on\s+(?:[#.][\w-]+|<[^>]*\/>|\[[^\]]+\]))/g;
15325
- function maskCaretScopes(input) {
15326
- const scopes = [];
15327
- const masked = input.replace(CARET_SCOPE_RE, (_m, varTok, scope) => {
15328
- const idx = scopes.length;
15329
- scopes.push(scope);
15330
- return `${varTok}${CARET_SCOPE_OPEN}${idx}${CARET_SCOPE_CLOSE}`;
15331
- });
15332
- return scopes.length > 0 ? { masked, scopes } : null;
15333
- }
15334
- function restoreCaretScopes(input, scopes) {
15335
- return input.replace(
15336
- new RegExp(`${CARET_SCOPE_OPEN}(\\d+)${CARET_SCOPE_CLOSE}`, "g"),
15337
- (_m, n) => scopes[Number(n)] ?? ""
15338
- );
15339
- }
15340
- var VIEW_TAIL_OPEN = "\uE002";
15341
- var VIEW_TAIL_CLOSE = "\uE003";
15342
- var VIEW_TAIL_RE = /\busing\s+view\b(?:\s+(?!then\b)[A-Za-z][\w-]*)?/gi;
15343
- var VIEW_TAIL_TOKEN_RE = new RegExp(`^${VIEW_TAIL_OPEN}(\\d+)${VIEW_TAIL_CLOSE}$`);
15344
- function maskViewTransitionTails(input) {
15345
- const tails = [];
15346
- const masked = input.replace(VIEW_TAIL_RE, (match) => {
15347
- const idx = tails.length;
15348
- tails.push(match);
15349
- return `${VIEW_TAIL_OPEN}${idx}${VIEW_TAIL_CLOSE}`;
15350
- });
15351
- return tails.length > 0 ? { masked, tails } : null;
15352
- }
15353
- function restoreViewTransitionTails(input, tails) {
15354
- return input.replace(
15355
- new RegExp(`${VIEW_TAIL_OPEN}(\\d+)${VIEW_TAIL_CLOSE}`, "g"),
15356
- (_m, n) => tails[Number(n)] ?? ""
15357
- );
15358
- }
15359
- var GrammarTransformer = class {
15360
- constructor(sourceLocale = "en", targetLocale) {
15361
- const source = getProfile(sourceLocale);
15362
- const target = getProfile(targetLocale);
15363
- if (!source) throw new Error(`Unknown source locale: ${sourceLocale}`);
15364
- if (!target) throw new Error(`Unknown target locale: ${targetLocale}`);
15365
- this.sourceProfile = source;
15366
- this.targetProfile = target;
15367
- }
15368
- /**
15369
- * Transform a hyperscript statement from source to target language.
15370
- * Handles compound statements with "then" by splitting, transforming each part,
15371
- * and rejoining with the target language's "then" keyword.
15372
- *
15373
- * For multi-line input, preserves line structure (indentation, blank lines).
15374
- */
15375
- transform(input) {
15376
- const out = this.transformInternal(input);
15377
- return this.targetProfile.code === "he" ? repairHebrewFrontedAccusative(out) : out;
15378
- }
15379
- transformInternal(input) {
15380
- const viewTails = maskViewTransitionTails(input);
15381
- if (viewTails) {
15382
- return restoreViewTransitionTails(this.transformInternal(viewTails.masked), viewTails.tails);
15383
- }
15384
- const caret = maskCaretScopes(input);
15385
- if (caret) {
15386
- return restoreCaretScopes(this.transform(caret.masked), caret.scopes);
15387
- }
15388
- const targetThen = getTargetThenKeyword(this.targetProfile.code);
15389
- if (!input.includes("\n")) {
15390
- const jsBlock = this.tryTransformJsBlock(input);
15391
- if (jsBlock !== null) return jsBlock;
15392
- const eventModifier = this.tryTransformEventWithModifierBody(input);
15393
- if (eventModifier !== null) return eventModifier;
15394
- const eventBlock = this.tryTransformEventWithBlockBody(input);
15395
- if (eventBlock !== null) return eventBlock;
15396
- const eventGuard = this.tryTransformEventWithUnlessGuard(input);
15397
- if (eventGuard !== null) return eventGuard;
15398
- }
15399
- const hasMultiLineStructure = input.includes("\n");
15400
- if (hasMultiLineStructure) {
15401
- const { parts: parts2, lineMetadata, partToLineIndex } = splitCompoundStatementWithMetadata(
15402
- input,
15403
- this.sourceProfile.code
15404
- );
15405
- const transformedParts = parts2.map((part) => this.transformSingle(part));
15406
- return reconstructWithLineStructure(
15407
- transformedParts,
15408
- lineMetadata,
15409
- partToLineIndex,
15410
- targetThen
15411
- );
15412
- }
15413
- const parts = splitCompoundStatement(input, this.sourceProfile.code);
15414
- if (parts.length > 1) {
15415
- const transformedParts = parts.map((part) => this.transformSingle(part));
15416
- return transformedParts.join(` ${targetThen} `);
15417
- }
15418
- return this.transformSingle(input);
15419
- }
15420
- /**
15421
- * Transform a single hyperscript statement (no compound "then" chains).
15422
- */
15423
- transformSingle(input) {
15424
- const block = extractBlockStructure(input, this.sourceProfile.code);
15425
- if (block) {
15426
- return this.transformBlock(block);
15427
- }
15428
- const strippedEnd = this.transformWithTrailingEnd(input);
15429
- if (strippedEnd !== null) {
15430
- return strippedEnd;
15431
- }
15432
- const setScope = this.transformSetWithScope(input);
15433
- if (setScope !== null) {
15434
- return setScope;
15435
- }
15436
- const viewTail = this.transformWithViewTransitionTail(input);
15437
- if (viewTail !== null) {
15438
- return viewTail;
15439
- }
15440
- const parsed = parseStatement(input, this.sourceProfile.code);
15441
- if (!parsed) {
15442
- return input;
15443
- }
15444
- applyPrimaryRole(parsed, this.targetProfile);
15445
- translateElements(parsed, this.sourceProfile.code, this.targetProfile.code);
15446
- const rule = this.findRule(parsed);
15447
- if (rule?.transform.custom) {
15448
- return rule.transform.custom(parsed, this.targetProfile);
15449
- }
15450
- const roleOrder = rule?.transform.roleOrder || this.targetProfile.canonicalOrder;
15451
- const reordered = reorderRoles(parsed.roles, roleOrder);
15452
- const shouldInsertMarkers = rule?.transform.insertMarkers ?? true;
15453
- if (shouldInsertMarkers) {
15454
- const result = insertMarkers(
15455
- reordered,
15456
- this.targetProfile.markers,
15457
- this.targetProfile.adpositionType
15458
- );
15459
- return joinTokens(result);
15460
- }
15461
- return joinTokens(reordered.map((e) => e.translated || e.value));
15462
- }
15463
- /**
15464
- * Clause carrying a masked `using view transition` tail: strip the opaque
15465
- * token, transform the clause alone, and re-append the token at the very end.
15466
- *
15467
- * The tail is a clause-final modifier in every word order the corpus emits:
15468
- * the semantic side matches it as the literal `using view` marker plus a value
15469
- * word, and the SOV/VSO event-handler patterns admit it as an optional
15470
- * TRAILING group (after the with-marked operand). So the target position is
15471
- * "end of the transformed clause" for all 24 languages — no per-profile
15472
- * placement decision, which is what makes this a passthrough rather than a
15473
- * role.
15474
- *
15475
- * Returns null when the clause carries no masked tail, or when the token is
15476
- * not clause-final (nothing to reposition — leaving it in place still restores
15477
- * verbatim English).
15478
- */
15479
- transformWithViewTransitionTail(input) {
15480
- const trimmed = input.trim();
15481
- const tokens = trimmed.split(/\s+/);
15482
- if (tokens.length < 2) {
15483
- return null;
15484
- }
15485
- if (!VIEW_TAIL_TOKEN_RE.test(tokens[tokens.length - 1])) {
15486
- return null;
15487
- }
15488
- const tail = tokens[tokens.length - 1];
15489
- const head = tokens.slice(0, -1).join(" ");
15490
- return `${this.transformSingle(head)} ${tail}`;
15491
- }
15492
- /**
15493
- * `<command …> end` fragments: transform the command without its stranded
15494
- * terminator, then re-append the translated terminator as a standalone
15495
- * trailing token. Fragments that open a block of their own (`if … end`,
15496
- * `repeat … end`, `js … end`) bail — their terminator belongs to them and
15497
- * their dedicated paths handle it.
15498
- */
15499
- transformWithTrailingEnd(input) {
15500
- const src = this.sourceProfile.code;
15501
- const tokens = input.trim().split(/\s+/);
15502
- if (tokens.length < 2) {
15503
- return null;
15504
- }
15505
- const sourceEnd = translateWord("end", "en", src).toLowerCase();
15506
- if (tokens[tokens.length - 1].toLowerCase() !== sourceEnd) {
15507
- return null;
15508
- }
15509
- const openers = new Set(
15510
- ["if", "repeat", "unless", "while", "when", "live", "js"].map(
15511
- (k) => translateWord(k, "en", src).toLowerCase()
15512
- )
15513
- );
15514
- if (tokens.slice(0, -1).some((t) => openers.has(t.toLowerCase()))) {
15515
- return null;
15516
- }
15517
- const inner = this.transformSingle(tokens.slice(0, -1).join(" "));
15518
- const endT = translateWord(tokens[tokens.length - 1], src, this.targetProfile.code);
15519
- return `${inner} ${endT}`;
15520
- }
15521
- /**
15522
- * Detect and transform an inline JS block (`[on <event>] js <raw js> end`).
15523
- *
15524
- * The `js ... end` body is raw JavaScript: it must not be tokenized,
15525
- * translated, or word-order reordered. We mask the whole block with a single
15526
- * opaque placeholder, run the surrounding statement (the event-handler head,
15527
- * if any) through the normal reorder pipeline so the placeholder lands in the
15528
- * correct action position, then substitute the translated `js`/`end` keywords
15529
- * around the verbatim body.
15530
- *
15531
- * Returns `null` (fall through to the normal path) when there is no js block,
15532
- * no matching `end`, or trailing content after `end` (kept tight on purpose).
15533
- */
15534
- tryTransformJsBlock(input) {
15535
- const src = this.sourceProfile.code;
15536
- const dst = this.targetProfile.code;
15537
- const sourceJs = translateWord("js", "en", src);
15538
- const sourceEnd = translateWord("end", "en", src).toLowerCase();
15539
- const tokens = input.split(/\s+/).filter((t) => t.length > 0);
15540
- const escapedJs = sourceJs.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
15541
- const jsRe = new RegExp(`^${escapedJs}(\\(.*\\))?$`, "i");
15542
- const jsIdx = tokens.findIndex((t) => jsRe.test(t));
15543
- if (jsIdx === -1) return null;
15544
- let endIdx = -1;
15545
- for (let i = jsIdx + 1; i < tokens.length; i++) {
15546
- if (tokens[i].toLowerCase() === sourceEnd) {
15547
- endIdx = i;
15548
- break;
15549
- }
15550
- }
15551
- if (endIdx === -1) return null;
15552
- if (endIdx !== tokens.length - 1) return null;
15553
- const jsToken = tokens[jsIdx];
15554
- const jsParen = jsToken.match(jsRe)?.[1] ?? "";
15555
- const jsKeywordRaw = jsParen ? jsToken.slice(0, jsToken.length - jsParen.length) : jsToken;
15556
- const body = tokens.slice(jsIdx + 1, endIdx).join(" ");
15557
- const targetJs = translateWord(jsKeywordRaw, src, dst) + jsParen;
15558
- const targetEnd = translateWord(tokens[endIdx], src, dst);
15559
- const replacement = [targetJs, body, targetEnd].filter((s) => s.length > 0).join(" ");
15560
- const before = tokens.slice(0, jsIdx);
15561
- if (before.length === 0) return replacement;
15562
- const placeholder = "JSBLOCKPLACEHOLDER";
15563
- const reordered = this.transformSingle([...before, placeholder].join(" "));
15564
- if (!reordered.includes(placeholder)) return null;
15565
- return reordered.replace(placeholder, replacement);
15566
- }
15567
- /**
15568
- * Transform an event handler whose body is a block command
15569
- * (`on <event> [from <src>] {if|repeat|unless|while|for} … end`).
15570
- *
15571
- * `parseEventHandler` would treat the block keyword as the action and sweep the
15572
- * condition/body into role values, then reorder them — shredding the block
15573
- * (`if event.shiftKey call submitAndContinue() end` → scattered tokens). Instead
15574
- * we mask the whole block as an opaque action placeholder, reorder the event
15575
- * head normally, transform the block as a self-contained unit, and restitch.
15576
- *
15577
- * Returns `null` (fall through) when the input isn't an event handler, has no
15578
- * block-keyword body, or has no closing `end`.
15579
- */
15580
- tryTransformEventWithBlockBody(input) {
15581
- const tokens = tokenize2(input, this.sourceProfile);
15582
- if (tokens.length === 0) return null;
15583
- if (!EVENT_KEYWORDS.has(tokens[0]?.toLowerCase())) return null;
15584
- let blockIdx = -1;
15585
- for (let i = 1; i < tokens.length; i++) {
15586
- if (BLOCK_BODY_KEYWORDS.has(tokens[i].toLowerCase())) {
15587
- blockIdx = i;
15588
- break;
15589
- }
15590
- }
15591
- if (blockIdx <= 0) return null;
15592
- if (tokens[tokens.length - 1].toLowerCase() !== "end") return null;
15593
- const eventHead = tokens.slice(0, blockIdx);
15594
- const blockTokens = tokens.slice(blockIdx);
15595
- if (this.targetProfile.wordOrder === "VSO" && eventHead.some((t) => t.toLowerCase() === "from")) {
15596
- return null;
15597
- }
15598
- const placeholder = "EVENTBLOCKPLACEHOLDER";
15599
- const headOut = this.transformSingle([...eventHead, placeholder].join(" "));
15600
- if (!headOut.includes(placeholder)) return null;
15601
- const blockOut = this.transformBlockBody(blockTokens);
15602
- const eventClause = headOut.replace(placeholder, "").replace(/\s+/g, " ").trim();
15603
- return [eventClause, blockOut].filter((s) => s.length > 0).join(" ");
15604
- }
15605
- /**
15606
- * Transform an event handler whose body is an inline `unless` guard with NO
15607
- * `end` (`on <event> unless <cond> <body>` — the `unless-condition` shape).
15608
- *
15609
- * Object-marking SVO targets (he, zh). `parseEventHandler` reads `unless` as the
15610
- * action and sweeps the whole `<cond> <body>` tail into a single `patient` blob;
15611
- * the target then prefixes that blob with its object marker — Hebrew's accusative
15612
- * את (`… אלא את I match .disabled מתג .selected`) or Chinese's BA particle 把
15613
- * (`… 除非 把 I match .disabled 切换 .selected`) — and the inner toggle loses its
15614
- * own marker. The semantic parser can't recover the guard from that: the marker
15615
- * ahead of the condition blocks the `unless` pattern AND the now-markerless body
15616
- * command fails its object-marked toggle pattern, so the body collapses (`unless`
15617
- * dropped). Marker-less languages (de/it/ar/pl) tolerate the same role-blob and
15618
- * stay faithful, so this is an object-marker artifact, not a general parse gap.
15619
- *
15620
- * The standalone `unless <cond> <body>` path already produces the correct shape
15621
- * (`extractBlockStructure` → `transformBlock`: condition kept marker-free, body
15622
- * command keeps its marker — he `אלא I match .disabled מתג את .selected`, zh
15623
- * `除非 I match .disabled 切换 把 .selected`). So we split the event head off,
15624
- * transform the guard through that path, and emit the event clause first (he and
15625
- * zh are both SVO — event leads). Returns `null` (fall through) when the input
15626
- * isn't an object-marking event handler with an un-terminated inline `unless`
15627
- * guard.
15628
- */
15629
- tryTransformEventWithUnlessGuard(input) {
15630
- if (!UNLESS_GUARD_OBJECT_MARKING_LOCALES.has(this.targetProfile.code)) return null;
15631
- const tokens = tokenize2(input, this.sourceProfile);
15632
- if (tokens.length === 0) return null;
15633
- if (!EVENT_KEYWORDS.has(tokens[0]?.toLowerCase())) return null;
15634
- let guardIdx = -1;
15635
- for (let i = 1; i < tokens.length; i++) {
15636
- if (tokens[i].toLowerCase() === "unless") {
15637
- guardIdx = i;
15638
- break;
15639
- }
15640
- }
15641
- if (guardIdx <= 0) return null;
15642
- if (tokens[tokens.length - 1].toLowerCase() === "end") return null;
15643
- const eventHead = tokens.slice(0, guardIdx);
15644
- const guard = tokens.slice(guardIdx).join(" ");
15645
- const guardOut = this.transform(guard);
15646
- if (!guardOut) return null;
15647
- const placeholder = "EVENTGUARDPLACEHOLDER";
15648
- const headOut = this.transformSingle([...eventHead, placeholder].join(" "));
15649
- if (!headOut.includes(placeholder)) return null;
15650
- const eventClause = headOut.replace(placeholder, "").replace(/\s+/g, " ").trim();
15651
- return [eventClause, guardOut].filter((s) => s.length > 0).join(" ");
15652
- }
15653
- /**
15654
- * Transform an event handler whose body leads with a command-modifier
15655
- * (`on <event> [from <src>] {async|once|debounced [at N]|throttled [at N]} <body>`).
15656
- *
15657
- * `parseEventHandler` reads the first token after the event as the **action**, so
15658
- * a leading modifier is mistaken for the verb and the real verb (`fetch`/`add`) is
15659
- * swept into the patient. For SOV targets the reorder then surfaces that verb
15660
- * **first** (`取得 /api/data を クリック …`), and the semantic parser matches the
15661
- * leading `<verb> <patient>` with the low-priority `*-generated-verb-first`
15662
- * command pattern — returning a bare command and discarding the event + the rest
15663
- * of the body (degenerate parse).
15664
- *
15665
- * Instead, lift the modifier out, transform the modifier-free handler through the
15666
- * normal path (which keeps the body in canonical patient-first SOV order so the
15667
- * event sits mid-stream and the existing SOV event-extraction recovers it), then
15668
- * re-emit the modifier as a **leading English literal**. The semantic parser
15669
- * strips a leading `once`/`debounced`/`throttled` (`extractStandaloneModifiers`)
15670
- * and an `async` anywhere (`stripAsyncModifier`) before parsing, so the modifier
15671
- * is consumed as handler metadata rather than shadowing the body.
15672
- *
15673
- * Returns `null` (fall through) when the input isn't an event handler or the body
15674
- * doesn't lead with a modifier — leaving simple/Mode-B handlers byte-identical.
15675
- */
15676
- tryTransformEventWithModifierBody(input) {
15677
- if (this.targetProfile.wordOrder !== "SOV") return null;
15678
- const tokens = tokenize2(input, this.sourceProfile);
15679
- if (tokens.length === 0) return null;
15680
- if (!EVENT_KEYWORDS.has(tokens[0]?.toLowerCase())) return null;
15681
- let i = 1;
15682
- if (!tokens[i]) return null;
15683
- i++;
15684
- while (tokens[i] && EVENT_CONJUNCTIONS.has(tokens[i].toLowerCase()) && tokens[i + 1]) {
15685
- i += 2;
15686
- }
15687
- if (tokens[i]?.toLowerCase() === "from" && tokens[i + 1]) {
15688
- i++;
15689
- while (tokens[i] && !ENGLISH_COMMANDS.has(tokens[i].toLowerCase()) && !BODY_MODIFIER_KEYWORDS.has(tokens[i].toLowerCase())) {
15690
- i++;
15691
- }
15692
- }
15693
- const modWord = tokens[i]?.toLowerCase();
15694
- if (!modWord || !BODY_MODIFIER_KEYWORDS.has(modWord)) return null;
15695
- const modStart = i;
15696
- let modEnd = i + 1;
15697
- if (modWord !== "async" && modWord !== "once") {
15698
- if (tokens[modEnd]?.toLowerCase() === "at") modEnd++;
15699
- if (tokens[modEnd] && /^\d+(ms|s|m)?$/.test(tokens[modEnd])) modEnd++;
15700
- }
15701
- const modifierPhrase = tokens.slice(modStart, modEnd).join(" ");
15702
- const rebuilt = [...tokens.slice(0, modStart), ...tokens.slice(modEnd)].join(" ");
15703
- if (tokens.length - (modEnd - modStart) <= 2) return null;
15704
- const bodyOut = this.transform(rebuilt);
15705
- return [modifierPhrase, bodyOut].filter((s) => s.length > 0).join(" ");
15706
- }
15707
- /**
15708
- * Transform a `set <stuff> on <scope>` clause (S1 tabs-aria). The trailing
15709
- * `on <scope>` is the element(s) the attribute is set on — kept attached by
15710
- * splitOnCommandBoundaries. The semantic parser captures it as a `scope` role
15711
- * via the passthrough literal `on` (setSchema's scope markerOverride is `on`
15712
- * in every language), so `on` is emitted verbatim and only the scope *value*
15713
- * is translated (selectors pass through; `me`/`it`/`you` translate to the
15714
- * native reference, which the parser also accepts).
15715
- *
15716
- * Positioning matches where the set patterns expect the scope: at the clause
15717
- * end for verb-first orders (SVO/VSO), and immediately before the clause-final
15718
- * verb for SOV (the generated SOV pattern is `{dest} {patient} on {scope}
15719
- * {verb}`). Returns null (fall through) when there is no trailing `on <scope>`.
15720
- *
15721
- * Source is English in the sync-translations pipeline, so the `set` verb and
15722
- * `on` marker are matched as English literals.
15723
- */
15724
- transformSetWithScope(input) {
15725
- const src = this.sourceProfile.code;
15726
- const dst = this.targetProfile.code;
15727
- const m = input.match(/^(.*\bset\b.*\S)\s+on\s+([#.<@[]\S*|me|it|you)\s*$/i);
15728
- if (!m) return null;
15729
- const head = m[1];
15730
- const scopeRaw = m[2];
15731
- const headOut = this.transformSingle(head);
15732
- const scopeT = /^[#.<@[]/.test(scopeRaw) ? scopeRaw : translateWord(scopeRaw, src, dst);
15733
- if (this.targetProfile.wordOrder === "SOV") {
15734
- const verb = translateWord("set", "en", dst);
15735
- const firstTok = input.trim().split(/\s+/)[0]?.toLowerCase();
15736
- const isEventHandler = !!firstTok && EVENT_KEYWORDS.has(firstTok);
15737
- const toks = headOut.split(/\s+/).filter(Boolean);
15738
- if (!isEventHandler) {
15739
- const vIdx = toks.indexOf(verb);
15740
- if (vIdx >= 0) {
15741
- toks.splice(vIdx, 1);
15742
- toks.push("on", scopeT, verb);
15743
- return toks.join(" ");
15744
- }
15745
- } else if (toks.length > 0 && toks[toks.length - 1] === verb) {
15746
- toks.splice(toks.length - 1, 0, "on", scopeT);
15747
- return toks.join(" ");
15748
- }
15749
- return `${headOut} on ${scopeT}`;
15750
- }
15751
- return `${headOut} on ${scopeT}`;
15752
- }
15753
- /**
15754
- * Transform a self-contained block command (`{head} {clause?} {body} {end}`),
15755
- * where head ∈ {if, repeat, unless, …}. The clause (condition / `until event …`)
15756
- * runs up to the first command verb and is translated word-by-word; the body is
15757
- * recursively transformed (so its inner commands reorder for the target); the
15758
- * head/tail keywords are translated. The block is never word-order reordered as
15759
- * a whole — delimiters stay at the edges regardless of target word order.
15760
- */
15761
- transformBlockBody(blockTokens) {
15762
- const src = this.sourceProfile.code;
15763
- const dst = this.targetProfile.code;
15764
- const head = blockTokens[0];
15765
- const hasEnd = blockTokens[blockTokens.length - 1]?.toLowerCase() === "end";
15766
- const tail = hasEnd ? blockTokens[blockTokens.length - 1] : "";
15767
- const inner = blockTokens.slice(1, hasEnd ? -1 : void 0);
15768
- const commands = getCommandKeywordsForLocale(src);
15769
- const copulas = getCopulasForLocale(src);
15770
- let bodyStart = inner.findIndex(
15771
- (t, i) => commands.has(t.toLowerCase()) && !isPredicateAdjectivePosition(inner, i, copulas)
15772
- );
15773
- if (bodyStart < 0) bodyStart = inner.length;
15774
- const clause = inner.slice(0, bodyStart).join(" ");
15775
- const bodyTokens = inner.slice(bodyStart);
15776
- const headT = translateWord(head, src, dst);
15777
- const tailT = tail ? translateWord(tail, src, dst) : "";
15778
- const clauseT = clause ? translateMultiWordValue(clause, src, dst) : "";
15779
- const bodyT = this.transformConditionalBody(bodyTokens);
15780
- const POSITIONAL_BRANCH_HEADS = /* @__PURE__ */ new Set(["first", "last", "next", "previous", "closest"]);
15781
- const thenT = this.targetProfile.wordOrder === "SOV" && clauseT && POSITIONAL_BRANCH_HEADS.has(bodyTokens[1]?.toLowerCase()) && inner[bodyStart - 1]?.toLowerCase() !== "then" ? translateWord("then", src, dst) : "";
15782
- return [headT, clauseT, thenT, bodyT, tailT].filter((s) => s.length > 0).join(" ");
15783
- }
15784
- /**
15785
- * Transform an `if`/`unless` block body, splitting it at a top-level `else` into
15786
- * a then-branch and an else-branch so each is reordered as a self-contained unit
15787
- * and the `else` keyword itself is translated. Without this, the body is reordered
15788
- * as one stream: `else` rides along glued to the preceding clause (and, when that
15789
- * clause begins with a selector, is marked a selector and left *untranslated*),
15790
- * and a spurious `then` is inserted around it — both of which break the target
15791
- * text and the downstream parse. The split is depth-aware so an `else` belonging
15792
- * to a nested block is not mistaken for this block's separator. Bodies without an
15793
- * `else` transform exactly as before.
15794
- */
15795
- transformConditionalBody(bodyTokens) {
15796
- const src = this.sourceProfile.code;
15797
- const dst = this.targetProfile.code;
15798
- const sourceElse = translateWord("else", "en", src).toLowerCase();
15799
- let depth = 0;
15800
- let elseIdx = -1;
15801
- for (let i = 0; i < bodyTokens.length; i++) {
15802
- const t = bodyTokens[i].toLowerCase();
15803
- if (BLOCK_BODY_KEYWORDS.has(t)) depth++;
15804
- else if (t === "end" && depth > 0) depth--;
15805
- else if (t === sourceElse && depth === 0) {
15806
- elseIdx = i;
15807
- break;
15808
- }
15809
- }
15810
- if (elseIdx === -1) {
15811
- const body = bodyTokens.join(" ");
15812
- return body ? this.transform(body) : "";
15813
- }
15814
- const thenBranch = bodyTokens.slice(0, elseIdx).join(" ");
15815
- const elseBranch = bodyTokens.slice(elseIdx + 1).join(" ");
15816
- const elseT = translateWord(bodyTokens[elseIdx], src, dst);
15817
- return [
15818
- thenBranch ? this.transform(thenBranch) : "",
15819
- elseT,
15820
- elseBranch ? this.transform(elseBranch) : ""
15821
- ].filter((s) => s.length > 0).join(" ");
15822
- }
15823
- /**
15824
- * Translate a reactive block by translating the head/tail/connector
15825
- * via the dictionary, recursively transforming the body through the
15826
- * regular pipeline, and rejoining in source-language position order.
15827
- * Block-syntactic tokens are never reordered: they're delimiters, not
15828
- * arguments, and authors expect them at start/end positions
15829
- * regardless of target word order.
15830
- */
15831
- transformBlock(block) {
15832
- const src = this.sourceProfile.code;
15833
- const dst = this.targetProfile.code;
15834
- const head = translateWord(block.headKeyword, src, dst);
15835
- const tail = block.tailKeyword ? translateWord(block.tailKeyword, src, dst) : "";
15836
- const connector = block.connector ? translateWord(block.connector, src, dst) : "";
15837
- const prefix = block.prefixExpr ? translateMultiWordValue(block.prefixExpr, src, dst) : "";
15838
- const body = this.transform(block.body);
15839
- return [head, prefix, connector, body, tail].filter((s) => s.length > 0).join(" ");
15840
- }
15841
- /**
15842
- * Find the best matching rule for this statement
15843
- */
15844
- findRule(parsed) {
15845
- if (!this.targetProfile.rules) return void 0;
15846
- const matchingRules = this.targetProfile.rules.filter((rule) => this.matchesRule(parsed, rule)).sort((a, b) => b.priority - a.priority);
15847
- return matchingRules[0];
15848
- }
15849
- /**
15850
- * Check if a parsed statement matches a rule
15851
- */
15852
- matchesRule(parsed, rule) {
15853
- const { match } = rule;
15854
- for (const role of match.requiredRoles) {
15855
- if (!parsed.roles.has(role)) {
15856
- return false;
15857
- }
15858
- }
15859
- if (match.commands && match.commands.length > 0) {
15860
- const action = parsed.roles.get("action");
15861
- if (!action) return false;
15862
- const actionValue = action.value.toLowerCase();
15863
- if (!match.commands.some((cmd) => cmd.toLowerCase() === actionValue)) {
15864
- return false;
15865
- }
15866
- }
15867
- if (match.predicate && !match.predicate(parsed)) {
15868
- return false;
15869
- }
15870
- return true;
15871
- }
15872
- };
15873
- function toLocale(input, targetLocale) {
15874
- const transformer = new GrammarTransformer("en", targetLocale);
15875
- return transformer.transform(input);
15876
- }
15877
- function toEnglish(input, sourceLocale) {
15878
- const transformer = new GrammarTransformer(sourceLocale, "en");
15879
- return transformer.transform(input);
15880
- }
15881
- function translate(input, sourceLocale, targetLocale) {
15882
- if (sourceLocale === targetLocale) return input;
15883
- if (sourceLocale === "en") return toLocale(input, targetLocale);
15884
- if (targetLocale === "en") return toEnglish(input, sourceLocale);
15885
- if (hasDirectMapping(sourceLocale, targetLocale)) {
15886
- return translateDirect(input, sourceLocale, targetLocale);
15887
- }
15888
- const english = toEnglish(input, sourceLocale);
15889
- return toLocale(english, targetLocale);
15890
- }
15891
- function translateDirect(input, sourceLocale, targetLocale) {
15892
- const mapping = getDirectMapping(sourceLocale, targetLocale);
15893
- if (!mapping) {
15894
- return toLocale(toEnglish(input, sourceLocale), targetLocale);
15895
- }
15896
- const tokens = input.split(/\s+/);
15897
- const translated = tokens.map((token) => {
15898
- if (token.startsWith("#") || token.startsWith(".") || token.startsWith("@")) {
15899
- return token;
15900
- }
15901
- if (token.startsWith('"') || token.startsWith("'")) {
15902
- return token;
15903
- }
15904
- const directTranslation = mapping.words[token];
15905
- if (directTranslation) {
15906
- return directTranslation;
15907
- }
15908
- const suffixMatch = token.match(/^(.+?)(-.+)$/);
15909
- if (suffixMatch) {
15910
- const [, base, suffix] = suffixMatch;
15911
- const translatedBase = mapping.words[base] || base;
15912
- return translatedBase + suffix;
15913
- }
15914
- return token;
15915
- });
15916
- return translated.join(" ");
15917
- }
15918
- var examples = {
15919
- english: {
15920
- eventHandler: "on click increment #count",
15921
- putInto: "put my value into #output",
15922
- toggle: "toggle .active",
15923
- wait: "wait 2 seconds"
15924
- },
15925
- // Expected outputs (approximate, for reference)
15926
- japanese: {
15927
- eventHandler: "#count \u3092 \u30AF\u30EA\u30C3\u30AF \u3067 \u5897\u52A0",
15928
- putInto: "\u79C1\u306E \u5024 \u3092 #output \u306B \u7F6E\u304F",
15929
- toggle: ".active \u3092 \u5207\u308A\u66FF\u3048",
15930
- wait: "2\u79D2 \u5F85\u3064"
15931
- },
15932
- chinese: {
15933
- eventHandler: "\u5F53 \u70B9\u51FB \u65F6 \u589E\u52A0 #count",
15934
- putInto: "\u628A \u6211\u7684\u503C \u653E \u5230 #output",
15935
- toggle: "\u5207\u6362 .active",
15936
- wait: "\u7B49\u5F85 2\u79D2"
15937
- },
15938
- arabic: {
15939
- eventHandler: "\u0632\u0650\u062F #count \u0639\u0646\u062F \u0627\u0644\u0646\u0642\u0631",
15940
- putInto: "\u0636\u0639 \u0642\u064A\u0645\u062A\u064A \u0641\u064A #output",
15941
- toggle: "\u0628\u062F\u0651\u0644 .active",
15942
- wait: "\u0627\u0646\u062A\u0638\u0631 \u062B\u0627\u0646\u064A\u062A\u064A\u0646"
15943
- }
15944
- };
15945
-
15946
14090
  // src/index.ts
15947
14091
  var defaultTranslator = new HyperscriptTranslator({ locale: "en" });
15948
14092
  var defaultRuntime = new RuntimeI18nManager({ locale: "en" });
@@ -15953,7 +14097,6 @@ exports.ENGLISH_COMMANDS = ENGLISH_COMMANDS;
15953
14097
  exports.ENGLISH_KEYWORDS = ENGLISH_KEYWORDS;
15954
14098
  exports.EnhancedI18nInputSchema = EnhancedI18nInputSchema;
15955
14099
  exports.EnhancedI18nOutputSchema = EnhancedI18nOutputSchema;
15956
- exports.GrammarTransformer = GrammarTransformer;
15957
14100
  exports.HyperscriptI18nWebpackPlugin = HyperscriptI18nWebpackPlugin;
15958
14101
  exports.HyperscriptTranslator = HyperscriptTranslator;
15959
14102
  exports.LANGUAGE_FAMILY_DEFAULTS = LANGUAGE_FAMILY_DEFAULTS;
@@ -16007,7 +14150,6 @@ exports.getI18n = getI18n;
16007
14150
  exports.getPlural = getPlural;
16008
14151
  exports.getProfile = getProfile;
16009
14152
  exports.getSupportedLocales = getSupportedLocales;
16010
- exports.grammarExamples = examples;
16011
14153
  exports.hi = hi;
16012
14154
  exports.hiDictionary = hindiDictionary;
16013
14155
  exports.hiKeywords = hiKeywords;
@@ -16036,7 +14178,6 @@ exports.malayProfile = malayProfile;
16036
14178
  exports.ms = ms2;
16037
14179
  exports.msDictionary = ms;
16038
14180
  exports.msKeywords = msKeywords;
16039
- exports.parseStatement = parseStatement;
16040
14181
  exports.pl = pl2;
16041
14182
  exports.plDictionary = pl;
16042
14183
  exports.plKeywords = plKeywords;
@@ -16066,14 +14207,11 @@ exports.thKeywords = thKeywords;
16066
14207
  exports.tl = tl2;
16067
14208
  exports.tlDictionary = tl;
16068
14209
  exports.tlKeywords = tlKeywords;
16069
- exports.toEnglish = toEnglish;
16070
- exports.toLocale = toLocale;
16071
14210
  exports.tokenize = tokenize;
16072
14211
  exports.tr = tr2;
16073
14212
  exports.trDictionary = tr;
16074
14213
  exports.trKeywords = trKeywords;
16075
14214
  exports.transformStatement = transformStatement;
16076
- exports.translate = translate;
16077
14215
  exports.translateFromEnglish = translateFromEnglish;
16078
14216
  exports.turkishProfile = turkishProfile;
16079
14217
  exports.uk = uk;