@hyperfixi/core 2.7.1 → 2.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/dist/api/hyperscript-api.d.ts +1 -0
  2. package/dist/ast-utils/index.js +1910 -233
  3. package/dist/ast-utils/index.mjs +1910 -233
  4. package/dist/behaviors/index.js +10 -1
  5. package/dist/behaviors/index.mjs +10 -1
  6. package/dist/bundle-generator/index.d.ts +1 -1
  7. package/dist/bundle-generator/index.js +77 -68
  8. package/dist/bundle-generator/index.mjs +76 -69
  9. package/dist/bundle-generator/template-capabilities.d.ts +2 -0
  10. package/dist/chunks/bridge-DHj-SYm2.js +2 -0
  11. package/dist/chunks/browser-modular-D1m0Eikh.js +2 -0
  12. package/dist/chunks/{index-i_j9Z-1e.js → index-CuPeasRm.js} +2 -2
  13. package/dist/commands/index.js +127 -6
  14. package/dist/commands/index.mjs +127 -6
  15. package/dist/compatibility/browser-modular.d.ts +2 -2
  16. package/dist/expressions/index.d.ts +1 -1
  17. package/dist/htmx/hcon.d.ts +9 -0
  18. package/dist/htmx/htmx-translator.d.ts +1 -0
  19. package/dist/hyperfixi-browser-classic-i18n.js +1 -1
  20. package/dist/hyperfixi-browser-minimal.js +1 -1
  21. package/dist/hyperfixi-browser-standard.js +1 -1
  22. package/dist/hyperfixi-browser.js +1 -1
  23. package/dist/hyperfixi-classic-i18n.js +1 -1
  24. package/dist/hyperfixi-hx-v4.js +1 -1
  25. package/dist/hyperfixi-hx.js +1 -1
  26. package/dist/hyperfixi-hybrid-complete.js +1 -1
  27. package/dist/hyperfixi-hybrid-hx.js +1 -1
  28. package/dist/hyperfixi-minimal.js +1 -1
  29. package/dist/hyperfixi-multilingual.js +1 -1
  30. package/dist/hyperfixi-standard.js +1 -1
  31. package/dist/hyperfixi.js +1 -1
  32. package/dist/hyperfixi.mjs +1 -1
  33. package/dist/index.js +4527 -536
  34. package/dist/index.min.js +1 -1
  35. package/dist/index.mjs +4527 -536
  36. package/dist/lib/dom-globals-shim.d.ts +2 -0
  37. package/dist/lokascript-browser-classic-i18n.js +1 -1
  38. package/dist/lokascript-browser-minimal.js +1 -1
  39. package/dist/lokascript-browser-standard.js +1 -1
  40. package/dist/lokascript-browser.js +1 -1
  41. package/dist/lokascript-hybrid-complete.js +1 -1
  42. package/dist/lokascript-hybrid-hx.js +1 -1
  43. package/dist/lokascript-multilingual.js +1 -1
  44. package/dist/lse/index.d.ts +7 -7
  45. package/dist/metadata.d.ts +1 -1
  46. package/dist/metadata.js +31 -14
  47. package/dist/metadata.mjs +31 -14
  48. package/dist/multilingual/index.js +8 -1
  49. package/dist/multilingual/index.mjs +8 -1
  50. package/dist/parser/command-parsers/animation-commands.d.ts +2 -2
  51. package/dist/parser/command-parsers/async-commands.d.ts +2 -2
  52. package/dist/parser/command-parsers/dom-commands.d.ts +5 -5
  53. package/dist/parser/command-parsers/navigation-commands.d.ts +4 -0
  54. package/dist/parser/command-parsers/utility-commands.d.ts +2 -1
  55. package/dist/parser/command-parsers/variable-commands.d.ts +2 -2
  56. package/dist/parser/full-parser.js +117 -5
  57. package/dist/parser/full-parser.mjs +117 -5
  58. package/dist/parser/semantic-integration.d.ts +1 -0
  59. package/dist/performance/integration.d.ts +1 -1
  60. package/dist/registry/index.js +117 -5
  61. package/dist/registry/index.mjs +117 -5
  62. package/package.json +13 -20
  63. package/dist/chunks/bridge-lZbOVRDD.js +0 -2
  64. package/dist/chunks/browser-modular-DegkWQ8d.js +0 -2
  65. package/dist/compatibility/browser-bundle-animation-generated.d.ts +0 -16
  66. package/dist/compatibility/browser-bundle-forms-generated.d.ts +0 -16
  67. package/dist/compatibility/browser-bundle-minimal-generated.d.ts +0 -16
@@ -4262,7 +4262,39 @@ var _BaseTokenizer = class _BaseTokenizer {
4262
4262
  pos++;
4263
4263
  }
4264
4264
  }
4265
- return new TokenStreamImpl(tokens, this.language);
4265
+ return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
4266
+ }
4267
+ /**
4268
+ * Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
4269
+ *
4270
+ * `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
4271
+ * preceded by an identifier is a qualifier (custom event namespace), not a
4272
+ * sigil. The English tokenizer already merges these inside
4273
+ * EnglishKeywordExtractor; this post-pass gives the other 23 languages the
4274
+ * same stream. Strict position adjacency is the discriminator: whitespace
4275
+ * between the tokens (`trigger :start`) breaks `end === start`, so a spaced
4276
+ * local-variable reference survives untouched.
4277
+ *
4278
+ * Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
4279
+ * sets tokenize `:` as bare punctuation (length 1), which never matches
4280
+ * COLON_QUALIFIER, so this pass is a no-op for them.
4281
+ */
4282
+ mergeColonQualifiedNames(tokens) {
4283
+ const out = [];
4284
+ for (const tok of tokens) {
4285
+ const prev = out[out.length - 1];
4286
+ if (prev && _BaseTokenizer.ASCII_WORD.test(prev.value) && _BaseTokenizer.COLON_QUALIFIER.test(tok.value) && prev.position.end === tok.position.start) {
4287
+ const merged = prev.value + tok.value;
4288
+ out[out.length - 1] = createToken(
4289
+ merged,
4290
+ this.classifyToken(merged),
4291
+ createPosition(prev.position.start, tok.position.end)
4292
+ );
4293
+ continue;
4294
+ }
4295
+ out.push(tok);
4296
+ }
4297
+ return out;
4266
4298
  }
4267
4299
  /**
4268
4300
  * Classify an unknown character when no extractor matches.
@@ -4768,6 +4800,14 @@ var _BaseTokenizer = class _BaseTokenizer {
4768
4800
  return null;
4769
4801
  }
4770
4802
  };
4803
+ /**
4804
+ * ASCII word of the shape the English word-walker produces. Excludes `:`, so a
4805
+ * token that already carries a qualifier never merges again — `a:b:c` yields
4806
+ * `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
4807
+ */
4808
+ _BaseTokenizer.ASCII_WORD = /^[A-Za-z_][A-Za-z0-9_]*$/;
4809
+ /** `:name` — only a variable-ref-style extractor ever emits this token shape. */
4810
+ _BaseTokenizer.COLON_QUALIFIER = /^:[A-Za-z_][A-Za-z0-9_]*$/;
4771
4811
  /**
4772
4812
  * Configuration for native language time units.
4773
4813
  * Maps patterns to their standard suffix (ms, s, m, h).
@@ -4991,8 +5031,11 @@ var init_arabic = __esm({
4991
5031
  result: "\u0627\u0644\u0646\u062A\u064A\u062C\u0629",
4992
5032
  event: "\u0627\u0644\u062D\u062F\u062B",
4993
5033
  target: "\u0627\u0644\u0647\u062F\u0641",
4994
- body: "\u062C\u0633\u0645"
5034
+ body: "\u062C\u0633\u0645",
4995
5035
  // matches the i18n dict's emitted body word (corpus-canonical, parser must recognize it)
5036
+ document: "\u0648\u062B\u064A\u0642\u0629",
5037
+ window: "\u0646\u0627\u0641\u0630\u0629",
5038
+ detail: "\u062A\u0641\u0627\u0635\u064A\u0644"
4996
5039
  },
4997
5040
  possessive: {
4998
5041
  marker: "",
@@ -5101,6 +5144,30 @@ var init_arabic = __esm({
5101
5144
  return: { primary: "\u0627\u0631\u062C\u0639", alternatives: ["\u0639\u064F\u062F"], normalized: "return" },
5102
5145
  then: { primary: "\u062B\u0645", alternatives: ["\u0628\u0639\u062F\u0647\u0627", "\u062B\u0645\u0651"], normalized: "then" },
5103
5146
  and: { primary: "\u0648\u0623\u064A\u0636\u0627\u064B", alternatives: ["\u0623\u064A\u0636\u0627\u064B"], normalized: "and" },
5147
+ // Comparison operator (`target matches .x`). Deferred by the Phase 2 `matches`
5148
+ // slice because ar's operand ALSO leaked (`references.target` carried الهدف while
5149
+ // the dict emits هدف), and registering the operator without its operand is worse
5150
+ // than neither: modal-close-backdrop ar passed R2 only BY ACCIDENT — the unparsed
5151
+ // condition was dropped, so `hide` ran unconditionally and coincidentally matched
5152
+ // the en DOM effect. `matches` alone would parse the condition into a real
5153
+ // comparison whose operand هدف evaluates to undefined, stopping `hide` and
5154
+ // flipping R2 pass→fail at tolerance 0. Landing WITH the هدف EXTRAS entry
5155
+ // (arabic.ts tokenizer) renders `target matches .modal-backdrop`, byte-identical
5156
+ // to en. Not an ActionType and has no command schema, so no pattern is generated.
5157
+ matches: { primary: "\u064A\u0637\u0627\u0628\u0642", normalized: "matches" },
5158
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
5159
+ // keyword the surface stays an identifier and leaks verbatim into the
5160
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
5161
+ // schema, so no pattern is generated from it.
5162
+ exists: { primary: "\u0645\u0648\u062C\u0648\u062F", normalized: "exists" },
5163
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
5164
+ // seam as `exists`: without the keyword the surface stays an identifier and
5165
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
5166
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
5167
+ // Uses the dict's NATURAL spaced phrase `لا يوجد`, matched by the base
5168
+ // tokenizer's multi-word keyword walk (longest-phrase at a word boundary) —
5169
+ // the same mechanism hi `मेل खाता` uses. Does not collide with `not: 'ليس'`.
5170
+ no: { primary: "\u0644\u0627 \u064A\u0648\u062C\u062F", normalized: "no" },
5104
5171
  // آخر is deliberately ABSENT: it is the positional `last` keyword
5105
5172
  // (آخر <button/> في .modal — see pattern-matcher's positional handling).
5106
5173
  // Listing it as an end-alternative made parseBodyWithClauses chop every
@@ -5304,6 +5371,11 @@ var init_bengali = __esm({
5304
5371
  return: { primary: "\u09AB\u09BF\u09B0\u09C1\u09A8", alternatives: ["\u09AB\u09C7\u09B0\u09A4 \u09A6\u09BF\u09A8"], normalized: "return" },
5305
5372
  then: { primary: "\u09A4\u09BE\u09B0\u09AA\u09B0", alternatives: ["\u09A4\u0996\u09A8"], normalized: "then" },
5306
5373
  and: { primary: "\u098F\u09AC\u0982", alternatives: [], normalized: "and" },
5374
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
5375
+ // surface stays an identifier and leaks verbatim into the condition's raw
5376
+ // expression, which the core expression parser reads as English. Neither an
5377
+ // ActionType nor a command schema, so no pattern is generated from it.
5378
+ is: { primary: "\u09B9\u09AF\u09BC", normalized: "is" },
5307
5379
  end: { primary: "\u09B6\u09C7\u09B7", alternatives: ["\u09B8\u09AE\u09BE\u09AA\u09CD\u09A4"], normalized: "end" },
5308
5380
  // Advanced
5309
5381
  js: { primary: "\u099C\u09C7\u098F\u09B8", alternatives: ["js"], normalized: "js" },
@@ -5401,7 +5473,10 @@ var init_german = __esm({
5401
5473
  result: "Ergebnis",
5402
5474
  event: "Ereignis",
5403
5475
  target: "Ziel",
5404
- body: "K\xF6rper"
5476
+ body: "K\xF6rper",
5477
+ document: "dokument",
5478
+ window: "fenster",
5479
+ detail: "detail"
5405
5480
  },
5406
5481
  possessive: {
5407
5482
  marker: "",
@@ -5496,6 +5571,22 @@ var init_german = __esm({
5496
5571
  // Predicate keywords (conditionals) — mirrors the Spanish profile, the only
5497
5572
  // language that previously parsed `is empty`-style predicates.
5498
5573
  is: { primary: "ist", normalized: "is" },
5574
+ // Comparison operator (`target matches .x`). Without this keyword the surface
5575
+ // stays an identifier and leaks verbatim into the condition's raw expression,
5576
+ // which the core expression parser reads as English (modal-close-backdrop /
5577
+ // focus-trap drop their then-branch). Not an ActionType and has no command
5578
+ // schema, so no pattern is generated from it.
5579
+ matches: { primary: "passt", normalized: "matches" },
5580
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
5581
+ // keyword the surface stays an identifier and leaks verbatim into the
5582
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
5583
+ // schema, so no pattern is generated from it.
5584
+ exists: { primary: "existiert", normalized: "exists" },
5585
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
5586
+ // seam as `exists`: without the keyword the surface stays an identifier and
5587
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
5588
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
5589
+ no: { primary: "kein", normalized: "no" },
5499
5590
  end: { primary: "ende", alternatives: ["fertig"], normalized: "end" },
5500
5591
  js: { primary: "js", alternatives: ["javascript"], normalized: "js" },
5501
5592
  async: { primary: "asynchron", normalized: "async" },
@@ -5590,7 +5681,10 @@ var init_english = __esm({
5590
5681
  result: "result",
5591
5682
  event: "event",
5592
5683
  target: "target",
5593
- body: "body"
5684
+ body: "body",
5685
+ document: "document",
5686
+ window: "window",
5687
+ detail: "detail"
5594
5688
  },
5595
5689
  possessive: {
5596
5690
  marker: "'s",
@@ -5741,7 +5835,10 @@ var init_spanish = __esm({
5741
5835
  event: "evento",
5742
5836
  target: "objetivo",
5743
5837
  // destino is a synonym
5744
- body: "cuerpo"
5838
+ body: "cuerpo",
5839
+ document: "documento",
5840
+ window: "ventana",
5841
+ detail: "detalle"
5745
5842
  },
5746
5843
  possessive: {
5747
5844
  marker: "de",
@@ -5764,7 +5861,10 @@ var init_spanish = __esm({
5764
5861
  }
5765
5862
  },
5766
5863
  roleMarkers: {
5767
- destination: { primary: "en", alternatives: ["sobre", "a"], position: "before" },
5864
+ // `hacia` is the i18n grammar's optional destination render form ("towards");
5865
+ // without it here a rendered/user `hacia` clause silently dropped the
5866
+ // destination (add → default `me`, put → null parse). Vocab Batch 1 (V2+V4).
5867
+ destination: { primary: "en", alternatives: ["sobre", "a", "hacia"], position: "before" },
5768
5868
  source: { primary: "de", alternatives: ["desde"], position: "before" },
5769
5869
  patient: { primary: "", position: "before" },
5770
5870
  style: { primary: "con", position: "before" }
@@ -5874,6 +5974,19 @@ var init_spanish = __esm({
5874
5974
  is: { primary: "es", normalized: "is" },
5875
5975
  exists: { primary: "existe", normalized: "exists" },
5876
5976
  empty: { primary: "vac\xEDo", alternatives: ["vacio"], normalized: "empty" },
5977
+ // Comparison operator (`target matches .x`). Without this keyword the surface
5978
+ // stays an identifier and leaks verbatim into the condition's raw expression,
5979
+ // which the core expression parser reads as English (modal-close-backdrop /
5980
+ // focus-trap drop their then-branch). Not an ActionType and has no command
5981
+ // schema, so no pattern is generated from it.
5982
+ matches: { primary: "coincide", normalized: "matches" },
5983
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
5984
+ // seam as `exists`: without the keyword the surface stays an identifier and
5985
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
5986
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
5987
+ // Does NOT collide with `not: { primary: 'no' }`: the keyword map is keyed by
5988
+ // SURFACE, so this registers `ningún` and leaves the `no` surface untouched.
5989
+ no: { primary: "ning\xFAn", normalized: "no" },
5877
5990
  end: { primary: "fin", alternatives: ["final", "terminar"], normalized: "end" },
5878
5991
  // Advanced
5879
5992
  js: { primary: "js", normalized: "js" },
@@ -5977,7 +6090,10 @@ var init_french = __esm({
5977
6090
  result: "r\xE9sultat",
5978
6091
  event: "\xE9v\xE9nement",
5979
6092
  target: "cible",
5980
- body: "corps"
6093
+ body: "corps",
6094
+ document: "document",
6095
+ window: "fen\xEAtre",
6096
+ detail: "d\xE9tail"
5981
6097
  },
5982
6098
  possessive: {
5983
6099
  marker: "de",
@@ -6072,6 +6188,27 @@ var init_french = __esm({
6072
6188
  return: { primary: "retourner", alternatives: ["renvoyer"], normalized: "return" },
6073
6189
  then: { primary: "puis", alternatives: ["ensuite", "alors"], normalized: "then" },
6074
6190
  and: { primary: "et", alternatives: ["aussi", "\xE9galement"], normalized: "and" },
6191
+ // Comparison operator (`target matches .x`). Without this keyword the surface
6192
+ // stays an identifier and leaks verbatim into the condition's raw expression,
6193
+ // which the core expression parser reads as English (modal-close-backdrop /
6194
+ // focus-trap drop their then-branch). Not an ActionType and has no command
6195
+ // schema, so no pattern is generated from it.
6196
+ matches: { primary: "correspond", normalized: "matches" },
6197
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
6198
+ // keyword the surface stays an identifier and leaks verbatim into the
6199
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
6200
+ // schema, so no pattern is generated from it.
6201
+ exists: { primary: "existe", normalized: "exists" },
6202
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
6203
+ // surface stays an identifier and leaks verbatim into the condition's raw
6204
+ // expression, which the core expression parser reads as English. Neither an
6205
+ // ActionType nor a command schema, so no pattern is generated from it.
6206
+ is: { primary: "est", normalized: "is" },
6207
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
6208
+ // seam as `exists`: without the keyword the surface stays an identifier and
6209
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
6210
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
6211
+ no: { primary: "aucun", normalized: "no" },
6075
6212
  end: { primary: "fin", alternatives: ["terminer", "finir"], normalized: "end" },
6076
6213
  js: { primary: "js", normalized: "js" },
6077
6214
  async: { primary: "asynchrone", normalized: "async" },
@@ -6377,7 +6514,10 @@ var init_hindi = __esm({
6377
6514
  result: "\u092A\u0930\u093F\u0923\u093E\u092E",
6378
6515
  event: "\u0918\u091F\u0928\u093E",
6379
6516
  target: "\u0932\u0915\u094D\u0937\u094D\u092F",
6380
- body: "\u092C\u0949\u0921\u0940"
6517
+ body: "\u092C\u0949\u0921\u0940",
6518
+ document: "\u0926\u0938\u094D\u0924\u093E\u0935\u0947\u091C\u093C",
6519
+ window: "\u0935\u093F\u0902\u0921\u094B",
6520
+ detail: "\u0935\u093F\u0935\u0930\u0923"
6381
6521
  },
6382
6522
  possessive: {
6383
6523
  marker: "\u0915\u093E",
@@ -6527,6 +6667,11 @@ var init_hindi = __esm({
6527
6667
  // parser. (History: `मेल_खाता` underscore-split to मेल/_/खाता; the concatenated
6528
6668
  // `मेलखाता` parsed but isn't how Hindi is written.)
6529
6669
  matches: { primary: "\u092E\u0947\u0932 \u0916\u093E\u0924\u093E", normalized: "matches" },
6670
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
6671
+ // keyword the surface stays an identifier and leaks verbatim into the
6672
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
6673
+ // schema, so no pattern is generated from it.
6674
+ exists: { primary: "\u092E\u094C\u091C\u0942\u0926", normalized: "exists" },
6530
6675
  end: { primary: "\u0938\u092E\u093E\u092A\u094D\u0924", alternatives: ["\u0905\u0902\u0924"], normalized: "end" },
6531
6676
  // Advanced
6532
6677
  js: { primary: "\u091C\u0947\u090F\u0938", alternatives: ["js"], normalized: "js" },
@@ -6626,8 +6771,11 @@ var init_indonesian = __esm({
6626
6771
  result: "hasil",
6627
6772
  event: "peristiwa",
6628
6773
  target: "target",
6629
- body: "badan"
6774
+ body: "badan",
6630
6775
  // matches the i18n dict's emitted body word (corpus-canonical; tubuh = anatomical body)
6776
+ document: "dokumen",
6777
+ window: "jendela",
6778
+ detail: "detail"
6631
6779
  },
6632
6780
  possessive: {
6633
6781
  marker: "",
@@ -6748,6 +6896,12 @@ var init_indonesian = __esm({
6748
6896
  return: { primary: "kembalikan", alternatives: ["kembali"], normalized: "return" },
6749
6897
  then: { primary: "lalu", alternatives: ["kemudian", "setelah itu"], normalized: "then" },
6750
6898
  and: { primary: "dan", alternatives: ["juga", "serta"], normalized: "and" },
6899
+ // Comparison operator (`target matches .x`). Without this keyword the surface
6900
+ // stays an identifier and leaks verbatim into the condition's raw expression,
6901
+ // which the core expression parser reads as English (modal-close-backdrop /
6902
+ // focus-trap drop their then-branch). Not an ActionType and has no command
6903
+ // schema, so no pattern is generated from it.
6904
+ matches: { primary: "cocok", normalized: "matches" },
6751
6905
  end: { primary: "selesai", alternatives: ["akhir", "tamat"], normalized: "end" },
6752
6906
  js: { primary: "js", alternatives: ["javascript"], normalized: "js" },
6753
6907
  async: { primary: "asinkron", normalized: "async" },
@@ -6854,7 +7008,10 @@ var init_italian = __esm({
6854
7008
  result: "risultato",
6855
7009
  event: "evento",
6856
7010
  target: "obiettivo",
6857
- body: "corpo"
7011
+ body: "corpo",
7012
+ document: "documento",
7013
+ window: "finestra",
7014
+ detail: "dettaglio"
6858
7015
  },
6859
7016
  possessive: {
6860
7017
  marker: "di",
@@ -6961,6 +7118,17 @@ var init_italian = __esm({
6961
7118
  return: { primary: "ritornare", normalized: "return" },
6962
7119
  then: { primary: "allora", alternatives: ["poi", "quindi"], normalized: "then" },
6963
7120
  and: { primary: "e", alternatives: ["anche"], normalized: "and" },
7121
+ // Comparison operator (`target matches .x`). Without this keyword the surface
7122
+ // stays an identifier and leaks verbatim into the condition's raw expression,
7123
+ // which the core expression parser reads as English (modal-close-backdrop /
7124
+ // focus-trap drop their then-branch). Not an ActionType and has no command
7125
+ // schema, so no pattern is generated from it.
7126
+ matches: { primary: "corrisponde", normalized: "matches" },
7127
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
7128
+ // seam as `exists`: without the keyword the surface stays an identifier and
7129
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
7130
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
7131
+ no: { primary: "nessun", normalized: "no" },
6964
7132
  end: { primary: "fine", normalized: "end" },
6965
7133
  // Advanced
6966
7134
  js: { primary: "js", normalized: "js" },
@@ -7074,7 +7242,10 @@ var init_japanese = __esm({
7074
7242
  result: "\u7D50\u679C",
7075
7243
  event: "\u30A4\u30D9\u30F3\u30C8",
7076
7244
  target: "\u30BF\u30FC\u30B2\u30C3\u30C8",
7077
- body: "\u30DC\u30C7\u30A3"
7245
+ body: "\u30DC\u30C7\u30A3",
7246
+ document: "\u30C9\u30AD\u30E5\u30E1\u30F3\u30C8",
7247
+ window: "\u30A6\u30A3\u30F3\u30C9\u30A6",
7248
+ detail: "\u8A73\u7D30"
7078
7249
  },
7079
7250
  possessive: {
7080
7251
  marker: "\u306E",
@@ -7152,6 +7323,10 @@ var init_japanese = __esm({
7152
7323
  focus: { primary: "\u30D5\u30A9\u30FC\u30AB\u30B9", alternatives: ["\u96C6\u4E2D"], normalized: "focus" },
7153
7324
  blur: { primary: "\u307C\u304B\u3057", alternatives: ["\u30D5\u30A9\u30FC\u30AB\u30B9\u89E3\u9664", "\u30D6\u30E9\u30FC"], normalized: "blur" },
7154
7325
  // Phase 1 (v0.9.90): DOM / form state / debug
7326
+ // Batch 3: do NOT add bare 空 here — probed: registering it as an empty
7327
+ // keyword injects a phantom `empty` command into the corpus-hot `is empty`
7328
+ // expression rows (である 空), an R0-precision regression. The empty-COMMAND
7329
+ // render gap (dict renders 空, parses null) is waived instead.
7155
7330
  empty: { primary: "\u7A7A\u306B", alternatives: ["\u7A7A\u306B\u3059\u308B"], normalized: "empty" },
7156
7331
  open: { primary: "\u958B\u304F", alternatives: ["\u30AA\u30FC\u30D7\u30F3"], normalized: "open" },
7157
7332
  close: { primary: "\u9589\u3058\u308B", alternatives: ["\u30AF\u30ED\u30FC\u30BA"], normalized: "close" },
@@ -7195,6 +7370,32 @@ var init_japanese = __esm({
7195
7370
  return: { primary: "\u623B\u308B", alternatives: ["\u8FD4\u3059", "\u30EA\u30BF\u30FC\u30F3"], normalized: "return" },
7196
7371
  then: { primary: "\u305D\u308C\u304B\u3089", alternatives: ["\u6B21\u306B", "\u306A\u3089\u3070", "\u306A\u3089"], normalized: "then" },
7197
7372
  and: { primary: "\u307E\u305F", alternatives: ["\u3068", "\u305D\u3057\u3066"], normalized: "and" },
7373
+ // Comparison operator (`target matches .x`). Deferred by the Phase 2 `matches`
7374
+ // slice because ja's operand ALSO leaked (`references.target` carried ターゲット
7375
+ // while the dict emits 対象), and registering the operator without its operand is
7376
+ // worse than neither: modal-close-backdrop ja passed R2 only BY ACCIDENT — the
7377
+ // unparsed condition was dropped, so `hide` ran unconditionally and coincidentally
7378
+ // matched the en DOM effect. `matches` alone would parse the condition into a real
7379
+ // comparison whose operand 対象 evaluates to undefined, stopping `hide` and
7380
+ // flipping R2 pass→fail at tolerance 0. Landing WITH the 対象 EXTRAS entry
7381
+ // (japanese.ts tokenizer) renders `target matches .modal-backdrop`, byte-identical
7382
+ // to en. Not an ActionType and has no command schema, so no pattern is generated.
7383
+ matches: { primary: "\u4E00\u81F4\u3059\u308B", normalized: "matches" },
7384
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
7385
+ // keyword the surface stays an identifier and leaks verbatim into the
7386
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
7387
+ // schema, so no pattern is generated from it.
7388
+ exists: { primary: "\u5B58\u5728\u3059\u308B", normalized: "exists" },
7389
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
7390
+ // surface stays an identifier and leaks verbatim into the condition's raw
7391
+ // expression, which the core expression parser reads as English. Neither an
7392
+ // ActionType nor a command schema, so no pattern is generated from it.
7393
+ is: { primary: "\u3067\u3042\u308B", normalized: "is" },
7394
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
7395
+ // seam as `exists`: without the keyword the surface stays an identifier and
7396
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
7397
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
7398
+ no: { primary: "\u306A\u3044", normalized: "no" },
7198
7399
  // 終了 removed: it is the i18n dict's `exit` emission (ja.ts), so listing it
7199
7400
  // as an `end` alternative made an `exit` inside `if … exit … end` read as the
7200
7401
  // block terminator and collapse the handler body (behavior-sortable). 終わり is
@@ -7297,8 +7498,11 @@ var init_korean = __esm({
7297
7498
  result: "\uACB0\uACFC",
7298
7499
  event: "\uC774\uBCA4\uD2B8",
7299
7500
  target: "\uB300\uC0C1",
7300
- body: "\uBC14\uB514"
7501
+ body: "\uBC14\uB514",
7301
7502
  // matches the i18n dict's emitted body word (본문 = "main text", wrong for the DOM body element)
7503
+ document: "\uBB38\uC11C",
7504
+ window: "\uCC3D",
7505
+ detail: "\uC138\uBD80"
7302
7506
  },
7303
7507
  possessive: {
7304
7508
  marker: "\uC758",
@@ -7371,7 +7575,9 @@ var init_korean = __esm({
7371
7575
  focus: { primary: "\uD3EC\uCEE4\uC2A4", normalized: "focus" },
7372
7576
  blur: { primary: "\uBE14\uB7EC", normalized: "blur" },
7373
7577
  // Phase 1 (v0.9.90): DOM / form state / debug
7374
- empty: { primary: "\uBE44\uC6B0\uAE30", normalized: "empty" },
7578
+ // Batch 3: 비어있는 added — the i18n dict renders the empty COMMAND with its
7579
+ // `is empty` adjective (category-shadowed), which parsed null.
7580
+ empty: { primary: "\uBE44\uC6B0\uAE30", alternatives: ["\uBE44\uC5B4\uC788\uB294"], normalized: "empty" },
7375
7581
  open: { primary: "\uC5F4\uAE30", normalized: "open" },
7376
7582
  close: { primary: "\uB2EB\uAE30", normalized: "close" },
7377
7583
  select: { primary: "\uACE0\uB974\uAE30", normalized: "select" },
@@ -7428,6 +7634,16 @@ var init_korean = __esm({
7428
7634
  // matches .x`. Without this keyword `일치` stays an identifier and the
7429
7635
  // condition is unevaluable (modal-close-backdrop drops its then-branch).
7430
7636
  matches: { primary: "\uC77C\uCE58", normalized: "matches" },
7637
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
7638
+ // keyword the surface stays an identifier and leaks verbatim into the
7639
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
7640
+ // schema, so no pattern is generated from it.
7641
+ exists: { primary: "\uC874\uC7AC", normalized: "exists" },
7642
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
7643
+ // seam as `exists`: without the keyword the surface stays an identifier and
7644
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
7645
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
7646
+ no: { primary: "\uC5C6\uC74C", normalized: "no" },
7431
7647
  end: { primary: "\uB05D", alternatives: ["\uB9C8\uCE68"], normalized: "end" },
7432
7648
  // Advanced
7433
7649
  js: { primary: "JS\uC2E4\uD589", alternatives: ["js"], normalized: "js" },
@@ -7519,7 +7735,10 @@ var init_ms = __esm({
7519
7735
  result: "hasil",
7520
7736
  event: "peristiwa",
7521
7737
  target: "sasaran",
7522
- body: "badan"
7738
+ body: "badan",
7739
+ document: "dokumen",
7740
+ window: "tetingkap",
7741
+ detail: "perincian"
7523
7742
  },
7524
7743
  possessive: {
7525
7744
  marker: "",
@@ -7642,6 +7861,27 @@ var init_ms = __esm({
7642
7861
  return: { primary: "pulang", alternatives: ["kembali"], normalized: "return" },
7643
7862
  then: { primary: "kemudian", alternatives: ["lepas_itu"], normalized: "then" },
7644
7863
  and: { primary: "dan", normalized: "and" },
7864
+ // Comparison operator (`target matches .x`). Without this keyword the surface
7865
+ // stays an identifier and leaks verbatim into the condition's raw expression,
7866
+ // which the core expression parser reads as English (modal-close-backdrop /
7867
+ // focus-trap drop their then-branch). Not an ActionType and has no command
7868
+ // schema, so no pattern is generated from it.
7869
+ matches: { primary: "sepadan", normalized: "matches" },
7870
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
7871
+ // keyword the surface stays an identifier and leaks verbatim into the
7872
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
7873
+ // schema, so no pattern is generated from it.
7874
+ exists: { primary: "wujud", normalized: "exists" },
7875
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
7876
+ // surface stays an identifier and leaks verbatim into the condition's raw
7877
+ // expression, which the core expression parser reads as English. Neither an
7878
+ // ActionType nor a command schema, so no pattern is generated from it.
7879
+ is: { primary: "adalah", normalized: "is" },
7880
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
7881
+ // seam as `exists`: without the keyword the surface stays an identifier and
7882
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
7883
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
7884
+ no: { primary: "tiada", normalized: "no" },
7645
7885
  end: { primary: "tamat", alternatives: ["habis"], normalized: "end" },
7646
7886
  // Advanced
7647
7887
  js: { primary: "js", normalized: "js" },
@@ -7728,7 +7968,10 @@ var init_polish = __esm({
7728
7968
  result: "wynik",
7729
7969
  event: "zdarzenie",
7730
7970
  target: "cel",
7731
- body: "body"
7971
+ body: "body",
7972
+ document: "dokument",
7973
+ window: "okno",
7974
+ detail: "szczeg\xF3\u0142"
7732
7975
  },
7733
7976
  possessive: {
7734
7977
  marker: "",
@@ -7961,6 +8204,17 @@ var init_polish = __esm({
7961
8204
  normalized: "then"
7962
8205
  },
7963
8206
  and: { primary: "i", alternatives: ["oraz"], normalized: "and" },
8207
+ // Comparison operator (`target matches .x`). Without this keyword the surface
8208
+ // stays an identifier and leaks verbatim into the condition's raw expression,
8209
+ // which the core expression parser reads as English (modal-close-backdrop /
8210
+ // focus-trap drop their then-branch). Not an ActionType and has no command
8211
+ // schema, so no pattern is generated from it.
8212
+ matches: { primary: "pasuje", normalized: "matches" },
8213
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
8214
+ // seam as `exists`: without the keyword the surface stays an identifier and
8215
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
8216
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
8217
+ no: { primary: "brak", normalized: "no" },
7964
8218
  end: { primary: "koniec", normalized: "end" },
7965
8219
  // Advanced
7966
8220
  js: { primary: "js", normalized: "js" },
@@ -8065,7 +8319,10 @@ var init_portuguese = __esm({
8065
8319
  result: "resultado",
8066
8320
  event: "evento",
8067
8321
  target: "alvo",
8068
- body: "corpo"
8322
+ body: "corpo",
8323
+ document: "documento",
8324
+ window: "janela",
8325
+ detail: "detalhe"
8069
8326
  },
8070
8327
  possessive: {
8071
8328
  marker: "de",
@@ -8161,6 +8418,27 @@ var init_portuguese = __esm({
8161
8418
  return: { primary: "retornar", alternatives: ["devolver"], normalized: "return" },
8162
8419
  then: { primary: "ent\xE3o", alternatives: ["logo"], normalized: "then" },
8163
8420
  and: { primary: "e", alternatives: ["tamb\xE9m", "al\xE9m disso"], normalized: "and" },
8421
+ // Comparison operator (`target matches .x`). Without this keyword the surface
8422
+ // stays an identifier and leaks verbatim into the condition's raw expression,
8423
+ // which the core expression parser reads as English (modal-close-backdrop /
8424
+ // focus-trap drop their then-branch). Not an ActionType and has no command
8425
+ // schema, so no pattern is generated from it.
8426
+ matches: { primary: "corresponde", normalized: "matches" },
8427
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
8428
+ // keyword the surface stays an identifier and leaks verbatim into the
8429
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
8430
+ // schema, so no pattern is generated from it.
8431
+ exists: { primary: "existe", normalized: "exists" },
8432
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
8433
+ // surface stays an identifier and leaks verbatim into the condition's raw
8434
+ // expression, which the core expression parser reads as English. Neither an
8435
+ // ActionType nor a command schema, so no pattern is generated from it.
8436
+ is: { primary: "\xE9", normalized: "is" },
8437
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
8438
+ // seam as `exists`: without the keyword the surface stays an identifier and
8439
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
8440
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
8441
+ no: { primary: "nenhum", normalized: "no" },
8164
8442
  end: { primary: "fim", alternatives: ["final", "t\xE9rmino"], normalized: "end" },
8165
8443
  js: { primary: "js", normalized: "js" },
8166
8444
  async: { primary: "ass\xEDncrono", normalized: "async" },
@@ -8267,7 +8545,10 @@ var init_quechua = __esm({
8267
8545
  result: "rurasqa",
8268
8546
  event: "ruwakuq",
8269
8547
  target: "punta",
8270
- body: "kurku"
8548
+ body: "kurku",
8549
+ document: "qillqa",
8550
+ window: "k_iri",
8551
+ detail: "sut_iy"
8271
8552
  },
8272
8553
  possessive: {
8273
8554
  marker: "-pa",
@@ -8338,7 +8619,10 @@ var init_quechua = __esm({
8338
8619
  focus: { primary: "qhawachiy", alternatives: ["qhaway"], normalized: "focus" },
8339
8620
  blur: { primary: "paqariy", alternatives: ["mana qhawachiy"], normalized: "blur" },
8340
8621
  // Phase 1 (v0.9.90): DOM / form state / debug
8341
- empty: { primary: "ch'usaq", normalized: "empty" },
8622
+ // Batch 3: apostrophe-less chusaq added — the i18n dict renders the empty
8623
+ // COMMAND with it (its `is empty` expression word), which parsed null against
8624
+ // the ch'usaq-only command patterns.
8625
+ empty: { primary: "ch'usaq", alternatives: ["chusaq"], normalized: "empty" },
8342
8626
  open: { primary: "paskay", normalized: "open" },
8343
8627
  close: { primary: "wichqay", normalized: "close" },
8344
8628
  select: { primary: "marcay", normalized: "select" },
@@ -8376,6 +8660,22 @@ var init_quechua = __esm({
8376
8660
  return: { primary: "kutichiy", alternatives: ["kutimuy"], normalized: "return" },
8377
8661
  then: { primary: "chaymantataq", alternatives: ["hinaspa", "chaymanta"], normalized: "then" },
8378
8662
  and: { primary: "hinallataq", alternatives: ["ima", "chaymantawan"], normalized: "and" },
8663
+ // Comparison operator (`target matches .x`). Without this keyword the surface
8664
+ // stays an identifier and leaks verbatim into the condition's raw expression,
8665
+ // which the core expression parser reads as English (modal-close-backdrop /
8666
+ // focus-trap drop their then-branch). Not an ActionType and has no command
8667
+ // schema, so no pattern is generated from it.
8668
+ matches: { primary: "tupan", normalized: "matches" },
8669
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
8670
+ // keyword the surface stays an identifier and leaks verbatim into the
8671
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
8672
+ // schema, so no pattern is generated from it.
8673
+ exists: { primary: "tiyan", normalized: "exists" },
8674
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
8675
+ // surface stays an identifier and leaks verbatim into the condition's raw
8676
+ // expression, which the core expression parser reads as English. Neither an
8677
+ // ActionType nor a command schema, so no pattern is generated from it.
8678
+ is: { primary: "kanqa", normalized: "is" },
8379
8679
  end: { primary: "tukukuy", alternatives: ["tukuy", "puchukay"], normalized: "end" },
8380
8680
  js: { primary: "js", normalized: "js" },
8381
8681
  async: { primary: "mana waqtalla", normalized: "async" },
@@ -8471,8 +8771,11 @@ var init_russian = __esm({
8471
8771
  result: "\u0440\u0435\u0437\u0443\u043B\u044C\u0442\u0430\u0442",
8472
8772
  event: "\u0441\u043E\u0431\u044B\u0442\u0438\u0435",
8473
8773
  target: "\u0446\u0435\u043B\u044C",
8474
- body: "\u0442\u0435\u043B\u043E"
8774
+ body: "\u0442\u0435\u043B\u043E",
8475
8775
  // was an English placeholder; the i18n dict emits the Russian word
8776
+ document: "\u0434\u043E\u043A\u0443\u043C\u0435\u043D\u0442",
8777
+ window: "\u043E\u043A\u043D\u043E",
8778
+ detail: "\u0434\u0435\u0442\u0430\u043B\u0438"
8476
8779
  },
8477
8780
  possessive: {
8478
8781
  marker: "",
@@ -8718,6 +9021,21 @@ var init_russian = __esm({
8718
9021
  // so `target соответствует .x` must normalize to `target matches .x`; otherwise
8719
9022
  // `соответствует` stays an identifier and modal-close-backdrop drops its then-branch.
8720
9023
  matches: { primary: "\u0441\u043E\u043E\u0442\u0432\u0435\u0442\u0441\u0442\u0432\u0443\u0435\u0442", normalized: "matches" },
9024
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
9025
+ // keyword the surface stays an identifier and leaks verbatim into the
9026
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
9027
+ // schema, so no pattern is generated from it.
9028
+ exists: { primary: "\u0441\u0443\u0449\u0435\u0441\u0442\u0432\u0443\u0435\u0442", normalized: "exists" },
9029
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
9030
+ // surface stays an identifier and leaks verbatim into the condition's raw
9031
+ // expression, which the core expression parser reads as English. Neither an
9032
+ // ActionType nor a command schema, so no pattern is generated from it.
9033
+ is: { primary: "\u0435\u0441\u0442\u044C", normalized: "is" },
9034
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
9035
+ // seam as `exists`: without the keyword the surface stays an identifier and
9036
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
9037
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
9038
+ no: { primary: "\u043D\u0435\u0442", normalized: "no" },
8721
9039
  end: { primary: "\u043A\u043E\u043D\u0435\u0446", normalized: "end" },
8722
9040
  // Advanced
8723
9041
  js: { primary: "js", normalized: "js" },
@@ -8834,7 +9152,10 @@ var init_swahili = __esm({
8834
9152
  result: "matokeo",
8835
9153
  event: "tukio",
8836
9154
  target: "lengo",
8837
- body: "mwili"
9155
+ body: "mwili",
9156
+ document: "hati",
9157
+ window: "dirisha",
9158
+ detail: "maelezo"
8838
9159
  },
8839
9160
  possessive: {
8840
9161
  marker: "",
@@ -8946,6 +9267,17 @@ var init_swahili = __esm({
8946
9267
  // Swahili copula ("is"); only recognized in predicate position (after a value,
8947
9268
  // before an adjective like `tupu`), so it doesn't disturb command parsing.
8948
9269
  is: { primary: "ni", normalized: "is" },
9270
+ // Comparison operator (`target matches .x`). Without this keyword the surface
9271
+ // stays an identifier and leaks verbatim into the condition's raw expression,
9272
+ // which the core expression parser reads as English (modal-close-backdrop /
9273
+ // focus-trap drop their then-branch). Not an ActionType and has no command
9274
+ // schema, so no pattern is generated from it.
9275
+ matches: { primary: "inafanana", normalized: "matches" },
9276
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
9277
+ // seam as `exists`: without the keyword the surface stays an identifier and
9278
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
9279
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
9280
+ no: { primary: "hakuna", normalized: "no" },
8949
9281
  end: { primary: "mwisho", alternatives: ["maliza", "tamati"], normalized: "end" },
8950
9282
  js: { primary: "js", alternatives: ["javascript"], normalized: "js" },
8951
9283
  async: { primary: "isiyo sawia", normalized: "async" },
@@ -9139,6 +9471,11 @@ var init_thai = __esm({
9139
9471
  return: { primary: "\u0E04\u0E37\u0E19\u0E04\u0E48\u0E32", alternatives: ["\u0E01\u0E25\u0E31\u0E1A"], normalized: "return" },
9140
9472
  then: { primary: "\u0E41\u0E25\u0E49\u0E27", alternatives: [], normalized: "then" },
9141
9473
  and: { primary: "\u0E41\u0E25\u0E30", alternatives: [], normalized: "and" },
9474
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
9475
+ // keyword the surface stays an identifier and leaks verbatim into the
9476
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
9477
+ // schema, so no pattern is generated from it.
9478
+ exists: { primary: "\u0E21\u0E35\u0E2D\u0E22\u0E39\u0E48", normalized: "exists" },
9142
9479
  end: { primary: "\u0E08\u0E1A", alternatives: [], normalized: "end" },
9143
9480
  // Advanced
9144
9481
  js: { primary: "\u0E40\u0E08\u0E40\u0E2D\u0E2A", alternatives: ["js"], normalized: "js" },
@@ -9242,8 +9579,11 @@ var init_tl = __esm({
9242
9579
  // "event"
9243
9580
  target: "target",
9244
9581
  // "target"
9245
- body: "katawan"
9582
+ body: "katawan",
9246
9583
  // was an English placeholder; the i18n dict emits the Tagalog word
9584
+ document: "dokumento",
9585
+ window: "bintana",
9586
+ detail: "detalye"
9247
9587
  },
9248
9588
  possessive: {
9249
9589
  marker: "ng",
@@ -9351,6 +9691,17 @@ var init_tl = __esm({
9351
9691
  return: { primary: "ibalik", alternatives: ["bumalik"], normalized: "return" },
9352
9692
  then: { primary: "pagkatapos", alternatives: ["saka"], normalized: "then" },
9353
9693
  and: { primary: "at", normalized: "and" },
9694
+ // Comparison operator (`target matches .x`). Without this keyword the surface
9695
+ // stays an identifier and leaks verbatim into the condition's raw expression,
9696
+ // which the core expression parser reads as English (modal-close-backdrop /
9697
+ // focus-trap drop their then-branch). Not an ActionType and has no command
9698
+ // schema, so no pattern is generated from it.
9699
+ matches: { primary: "tumutugma", normalized: "matches" },
9700
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
9701
+ // surface stays an identifier and leaks verbatim into the condition's raw
9702
+ // expression, which the core expression parser reads as English. Neither an
9703
+ // ActionType nor a command schema, so no pattern is generated from it.
9704
+ is: { primary: "ay", normalized: "is" },
9354
9705
  end: { primary: "wakas", alternatives: ["tapos"], normalized: "end" },
9355
9706
  // Advanced
9356
9707
  js: { primary: "js", normalized: "js" },
@@ -9450,7 +9801,10 @@ var init_turkish = __esm({
9450
9801
  result: "sonu\xE7",
9451
9802
  event: "olay",
9452
9803
  target: "hedef",
9453
- body: "g\xF6vde"
9804
+ body: "g\xF6vde",
9805
+ document: "belge",
9806
+ window: "pencere",
9807
+ detail: "detay"
9454
9808
  },
9455
9809
  possessive: {
9456
9810
  // Genitive suffix, spaced for tokenization like Turkish's other case
@@ -9512,7 +9866,10 @@ var init_turkish = __esm({
9512
9866
  // Dative/Locative + Genitive (with buffer consonants)
9513
9867
  source: { primary: "den", alternatives: ["dan", "ten", "tan"], position: "after" },
9514
9868
  // Ablative
9515
- style: { primary: "le", alternatives: ["la", "yle", "yla"], position: "after" },
9869
+ // `ile` is the free-standing instrumental the transformer actually emits
9870
+ // for with-phrases (`getir method:"POST" body:form ile`); the suffix
9871
+ // forms cover hand-written agglutinated variants.
9872
+ style: { primary: "le", alternatives: ["la", "yle", "yla", "ile"], position: "after" },
9516
9873
  // Instrumental
9517
9874
  event: { primary: "i", alternatives: ["\u0131", "u", "\xFC"], position: "after" }
9518
9875
  // Event as accusative
@@ -9609,6 +9966,24 @@ var init_turkish = __esm({
9609
9966
  and: { primary: "ve", alternatives: ["ayr\u0131ca", "hem de"], normalized: "and" },
9610
9967
  or: { primary: "veya", normalized: "or" },
9611
9968
  not: { primary: "de\u011Fil", alternatives: ["degil"], normalized: "not" },
9969
+ // Comparison operator (`target matches .x`). Without this keyword the surface
9970
+ // stays an identifier and leaks verbatim into the condition's raw expression,
9971
+ // which the core expression parser reads as English (modal-close-backdrop /
9972
+ // focus-trap drop their then-branch). Not an ActionType and has no command
9973
+ // schema, so no pattern is generated from it.
9974
+ matches: { primary: "e\u015Fle\u015Fir", normalized: "matches" },
9975
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
9976
+ // surface stays an identifier and leaks verbatim into the condition's raw
9977
+ // expression, which the core expression parser reads as English. Neither an
9978
+ // ActionType nor a command schema, so no pattern is generated from it.
9979
+ is: { primary: "dir", normalized: "is" },
9980
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
9981
+ // seam as `exists`: without the keyword the surface stays an identifier and
9982
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
9983
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
9984
+ // `yok` is a prefix of `else: 'yoksa'`; the keyword walk sorts longest-first, so
9985
+ // `yoksa` still wins where it appears.
9986
+ no: { primary: "yok", normalized: "no" },
9612
9987
  end: { primary: "son", alternatives: ["biti\u015F", "bitti"], normalized: "end" },
9613
9988
  // Advanced
9614
9989
  js: { primary: "js", normalized: "js" },
@@ -9703,8 +10078,11 @@ var init_ukrainian = __esm({
9703
10078
  result: "\u0440\u0435\u0437\u0443\u043B\u044C\u0442\u0430\u0442",
9704
10079
  event: "\u043F\u043E\u0434\u0456\u044F",
9705
10080
  target: "\u0446\u0456\u043B\u044C",
9706
- body: "\u0442\u0456\u043B\u043E"
10081
+ body: "\u0442\u0456\u043B\u043E",
9707
10082
  // was an English placeholder; the i18n dict emits the Ukrainian word
10083
+ document: "\u0434\u043E\u043A\u0443\u043C\u0435\u043D\u0442",
10084
+ window: "\u0432\u0456\u043A\u043D\u043E",
10085
+ detail: "\u0434\u0435\u0442\u0430\u043B\u0456"
9708
10086
  },
9709
10087
  possessive: {
9710
10088
  marker: "",
@@ -9968,6 +10346,21 @@ var init_ukrainian = __esm({
9968
10346
  // so `target відповідає .x` must normalize to `target matches .x`; otherwise
9969
10347
  // `відповідає` stays an identifier and modal-close-backdrop drops its then-branch.
9970
10348
  matches: { primary: "\u0432\u0456\u0434\u043F\u043E\u0432\u0456\u0434\u0430\u0454", normalized: "matches" },
10349
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
10350
+ // keyword the surface stays an identifier and leaks verbatim into the
10351
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
10352
+ // schema, so no pattern is generated from it.
10353
+ exists: { primary: "\u0456\u0441\u043D\u0443\u0454", normalized: "exists" },
10354
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
10355
+ // surface stays an identifier and leaks verbatim into the condition's raw
10356
+ // expression, which the core expression parser reads as English. Neither an
10357
+ // ActionType nor a command schema, so no pattern is generated from it.
10358
+ is: { primary: "\u0454", normalized: "is" },
10359
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
10360
+ // seam as `exists`: without the keyword the surface stays an identifier and
10361
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
10362
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
10363
+ no: { primary: "\u043D\u0456", normalized: "no" },
9971
10364
  end: { primary: "\u043A\u0456\u043D\u0435\u0446\u044C", normalized: "end" },
9972
10365
  // Advanced
9973
10366
  js: { primary: "js", normalized: "js" },
@@ -10212,6 +10605,12 @@ var init_vietnamese = __esm({
10212
10605
  return: { primary: "tr\u1EA3 v\u1EC1", normalized: "return" },
10213
10606
  then: { primary: "r\u1ED3i", alternatives: ["sau \u0111\xF3", "th\xEC"], normalized: "then" },
10214
10607
  and: { primary: "v\xE0", normalized: "and" },
10608
+ // Comparison operator (`target matches .x`). Without this keyword the surface
10609
+ // stays an identifier and leaks verbatim into the condition's raw expression,
10610
+ // which the core expression parser reads as English (modal-close-backdrop /
10611
+ // focus-trap drop their then-branch). Not an ActionType and has no command
10612
+ // schema, so no pattern is generated from it.
10613
+ matches: { primary: "kh\u1EDBp", normalized: "matches" },
10215
10614
  end: { primary: "k\u1EBFt th\xFAc", normalized: "end" },
10216
10615
  // Advanced
10217
10616
  js: { primary: "js", normalized: "js" },
@@ -10306,7 +10705,10 @@ var init_chinese = __esm({
10306
10705
  result: "\u7ED3\u679C",
10307
10706
  event: "\u4E8B\u4EF6",
10308
10707
  target: "\u76EE\u6807",
10309
- body: "\u4E3B\u4F53"
10708
+ body: "\u4E3B\u4F53",
10709
+ document: "\u6587\u6863",
10710
+ window: "\u7A97\u53E3",
10711
+ detail: "\u8BE6\u60C5"
10310
10712
  },
10311
10713
  possessive: {
10312
10714
  marker: "\u7684",
@@ -10413,6 +10815,11 @@ var init_chinese = __esm({
10413
10815
  return: { primary: "\u8FD4\u56DE", normalized: "return" },
10414
10816
  then: { primary: "\u7136\u540E", alternatives: ["\u63A5\u7740", "\u90A3\u4E48"], normalized: "then" },
10415
10817
  and: { primary: "\u5E76\u4E14", alternatives: ["\u548C", "\u800C\u4E14"], normalized: "and" },
10818
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
10819
+ // keyword the surface stays an identifier and leaks verbatim into the
10820
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
10821
+ // schema, so no pattern is generated from it.
10822
+ exists: { primary: "\u5B58\u5728", normalized: "exists" },
10416
10823
  end: { primary: "\u7ED3\u675F", alternatives: ["\u7EC8\u6B62", "\u5B8C"], normalized: "end" },
10417
10824
  // Advanced
10418
10825
  js: { primary: "JS\u6267\u884C", alternatives: ["js"], normalized: "js" },
@@ -10908,8 +11315,22 @@ var init_schema_validator = __esm({
10908
11315
  "select",
10909
11316
  "clear",
10910
11317
  "reset",
10911
- "breakpoint"
11318
+ "breakpoint",
10912
11319
  // Zero-arg debug command
11320
+ // Feature blocks. Their meaning lives in the BODY, not in a head role: `live`
11321
+ // and `intercept` have no head at all, and eventsource/socket/worker's name and
11322
+ // url are structural, not semantic arguments. Giving them roles purely to make
11323
+ // `scoreRoleCoverage` return a non-vacuous number would inject new
11324
+ // `action.role:valueType` entries into the English R1 reference that all 23
11325
+ // other languages must also capture, or the role-fidelity ratchet fires. The
11326
+ // structural layer (`tryParseFeatureBlock`) parses them instead, and derives
11327
+ // confidence from the body — so the `maxScore === 0 → 1` shortcut is never the
11328
+ // thing that scores them.
11329
+ "live",
11330
+ "eventsource",
11331
+ "socket",
11332
+ "worker",
11333
+ "intercept"
10913
11334
  ]);
10914
11335
  }
10915
11336
  });
@@ -10953,7 +11374,7 @@ function getSchema(action) {
10953
11374
  function getDefinedSchemas() {
10954
11375
  return Object.values(commandSchemas).filter((s) => s.roles.length > 0 || s.bareKeyword === true);
10955
11376
  }
10956
- var toggleSchema, addSchema, removeSchema, putSchema, setSchema, bindSchema, liveSchema, eventsourceSchema, socketSchema, workerSchema, interceptSchema, showSchema, hideSchema, onSchema, triggerSchema, waitSchema, fetchSchema, incrementSchema, decrementSchema, appendSchema, prependSchema, logSchema, getCommandSchema, takeSchema, makeSchema, haltSchema, settleSchema, throwSchema, sendSchema, ifSchema, unlessSchema, elseSchema, repeatSchema, forSchema, whileSchema, continueSchema, goSchema, transitionSchema, cloneSchema, focusSchema, blurSchema, emptySchema, openSchema, closeSchema, selectSchema, clearSchema, resetSchema, breakpointSchema, callSchema, returnSchema, jsSchema, asyncSchema, tellSchema, defaultSchema, initSchema, behaviorSchema, installSchema, measureSchema, swapSchema, morphSchema, beepSchema, breakSchema, copySchema, exitSchema, pickSchema, scrollSchema, URL_MARKER_ALL_LANGS, PARTIALS_IN_MARKER_ALL_LANGS, pushSchema, replaceSchema, processSchema, renderSchema, commandSchemas;
11377
+ var toggleSchema, addSchema, removeSchema, putSchema, setSchema, bindSchema, liveSchema, eventsourceSchema, socketSchema, workerSchema, interceptSchema, showSchema, hideSchema, onSchema, triggerSchema, waitSchema, fetchSchema, incrementSchema, decrementSchema, appendSchema, prependSchema, logSchema, getCommandSchema, takeSchema, makeSchema, haltSchema, settleSchema, throwSchema, sendSchema, ifSchema, unlessSchema, elseSchema, repeatSchema, forSchema, whileSchema, continueSchema, URL_MARKER_ALL_LANGS, goSchema, transitionSchema, cloneSchema, focusSchema, blurSchema, emptySchema, openSchema, closeSchema, selectSchema, clearSchema, resetSchema, breakpointSchema, callSchema, returnSchema, jsSchema, asyncSchema, tellSchema, defaultSchema, initSchema, behaviorSchema, installSchema, measureSchema, swapSchema, morphSchema, beepSchema, breakSchema, copySchema, exitSchema, pickSchema, scrollSchema, PARTIALS_IN_MARKER_ALL_LANGS, pushSchema, replaceSchema, processSchema, renderSchema, commandSchemas;
10957
11378
  var init_command_schemas = __esm({
10958
11379
  "src/generators/command-schemas.ts"() {
10959
11380
  toggleSchema = {
@@ -11402,7 +11823,13 @@ var init_command_schemas = __esm({
11402
11823
  role: "source",
11403
11824
  description: "The element or property to bind to",
11404
11825
  required: true,
11405
- expectedTypes: ["selector", "reference", "expression"],
11826
+ // 'property-path' opts this role into the "of"-possessive matcher, so the
11827
+ // property-first render of `bind $x to #y's prop` (es `valor de #picker`,
11828
+ // ar `قيمة لـ #picker`) keeps its owner selector instead of collapsing to
11829
+ // the bare property word; see pattern-matcher tryMatchOfPossessiveExpression.
11830
+ // The selector-first languages (en `#picker's value`, ja `#pickerの 値`)
11831
+ // already reached property-path through tryMatchPossessiveSelectorExpression.
11832
+ expectedTypes: ["selector", "reference", "expression", "property-path"],
11406
11833
  svoPosition: 2,
11407
11834
  sovPosition: 2,
11408
11835
  // Element mirrors `set`/`add`/`put`'s value ("to") marking per language.
@@ -11589,7 +12016,15 @@ var init_command_schemas = __esm({
11589
12016
  expectedTypes: ["literal", "expression"],
11590
12017
  // expression for custom/namespaced event names
11591
12018
  svoPosition: 1,
11592
- sovPosition: 2
12019
+ sovPosition: 2,
12020
+ // hi/qu/bn mark trigger's event ACCUSATIVELY (`draggable:start को ट्रिगर`,
12021
+ // `draggable:start ta kichay`, `draggable:start কে ট্রিগার` — the corpus
12022
+ // renderings), but their profile-wide event marker is the on-handler one
12023
+ // (hi पर, qu locative pi, bn এ), so the generated SOV pattern never
12024
+ // matched and the whole line fell through to the on-handler reading (hi)
12025
+ // or failed outright (qu/bn). ja/ko were immune only because their event
12026
+ // marker IS the object particle (を / 을·를). #588 markerVariants machinery.
12027
+ markerVariants: { hi: ["\u0915\u094B"], qu: ["ta"], bn: ["\u0995\u09C7"] }
11593
12028
  },
11594
12029
  {
11595
12030
  role: "destination",
@@ -11635,14 +12070,26 @@ var init_command_schemas = __esm({
11635
12070
  renderOverride: { en: "" }
11636
12071
  // "fetch /api" (rendering — no preposition)
11637
12072
  },
12073
+ {
12074
+ role: "style",
12075
+ description: "Request options object (method, headers, body, credentials\u2026)",
12076
+ required: false,
12077
+ // expression-ONLY: the pattern matcher routes a `{ … }` run in an
12078
+ // expression-only slot through its object-literal fold, which preserves the
12079
+ // source text so the expression parser can build a real objectLiteral.
12080
+ // `style` is the role whose marker is `with` in every language profile.
12081
+ expectedTypes: ["expression"],
12082
+ svoPosition: 2,
12083
+ sovPosition: 2
12084
+ },
11638
12085
  {
11639
12086
  role: "responseType",
11640
12087
  description: "Response format (json, text, html, blob, etc.)",
11641
12088
  required: false,
11642
12089
  expectedTypes: ["literal", "expression"],
11643
12090
  // json/text/html are identifiers → expression type
11644
- svoPosition: 2,
11645
- sovPosition: 2,
12091
+ svoPosition: 3,
12092
+ sovPosition: 3,
11646
12093
  markerOverride: { en: "as" }
11647
12094
  // "fetch /api as json" — needed by schema-driven role inference
11648
12095
  },
@@ -11651,16 +12098,16 @@ var init_command_schemas = __esm({
11651
12098
  description: "HTTP method (GET, POST, etc.)",
11652
12099
  required: false,
11653
12100
  expectedTypes: ["literal"],
11654
- svoPosition: 3,
11655
- sovPosition: 3
12101
+ svoPosition: 4,
12102
+ sovPosition: 4
11656
12103
  },
11657
12104
  {
11658
12105
  role: "destination",
11659
12106
  description: "Where to store the result",
11660
12107
  required: false,
11661
12108
  expectedTypes: ["selector", "reference"],
11662
- svoPosition: 4,
11663
- sovPosition: 4
12109
+ svoPosition: 5,
12110
+ sovPosition: 5
11664
12111
  }
11665
12112
  ]
11666
12113
  };
@@ -12144,6 +12591,32 @@ var init_command_schemas = __esm({
12144
12591
  roles: []
12145
12592
  // No roles
12146
12593
  };
12594
+ URL_MARKER_ALL_LANGS = {
12595
+ en: "url",
12596
+ es: "url",
12597
+ pt: "url",
12598
+ fr: "url",
12599
+ de: "url",
12600
+ it: "url",
12601
+ ja: "url",
12602
+ ko: "url",
12603
+ zh: "url",
12604
+ ar: "url",
12605
+ he: "url",
12606
+ hi: "url",
12607
+ bn: "url",
12608
+ tr: "url",
12609
+ ru: "url",
12610
+ uk: "url",
12611
+ pl: "url",
12612
+ id: "url",
12613
+ vi: "url",
12614
+ th: "url",
12615
+ ms: "url",
12616
+ tl: "url",
12617
+ sw: "url",
12618
+ qu: "url"
12619
+ };
12147
12620
  goSchema = {
12148
12621
  action: "go",
12149
12622
  description: "Navigate to a URL",
@@ -12169,6 +12642,19 @@ var init_command_schemas = __esm({
12169
12642
  markerOptional: { en: true },
12170
12643
  markerVariants: { he: ["\u05D0\u05EA"], zh: ["\u628A"] }
12171
12644
  }
12645
+ ],
12646
+ // `go to url "/page"` — without this variant the destination captures the
12647
+ // bare word `url` and the actual URL is dropped as tolerated-trailing text,
12648
+ // in en and therefore in every render (the go-url corpus row). The required
12649
+ // `url` literal keeps the variant inert for `go back` / scroll forms.
12650
+ rolePrefixLiteralVariants: [
12651
+ {
12652
+ role: "destination",
12653
+ literal: URL_MARKER_ALL_LANGS,
12654
+ idSuffix: "url",
12655
+ priorityDelta: 5,
12656
+ methodCarrier: "method"
12657
+ }
12172
12658
  ]
12173
12659
  };
12174
12660
  transitionSchema = {
@@ -12799,7 +13285,27 @@ var init_command_schemas = __esm({
12799
13285
  th: "\u0E14\u0E49\u0E27\u0E22",
12800
13286
  vi: "v\u1EDBi",
12801
13287
  he: "\u05E2\u05DD",
12802
- zh: "\u7528"
13288
+ zh: "\u7528",
13289
+ // SOV/postpositional with-words. These follow the patient (`#b से`,
13290
+ // `#b দিয়ে`), matching the i18n `with` emission. Without them the SOV
13291
+ // patient-first swap pattern's trailing group (which binds the second
13292
+ // element to `destination`) had only the locative dest-marker (hi में,
13293
+ // bn তে) as its alternatives, so `#b <with-word>` never bound and #b
13294
+ // dropped — hi/bn/tr/qu rendered the invalid `swap with #a`. ja/ko
13295
+ // escaped only because their dest-marker alternatives already carry the
13296
+ // instrumental (で / 로). See generateSOVPatientFirstEventHandlerPattern.
13297
+ hi: "\u0938\u0947",
13298
+ bn: "\u09A6\u09BF\u09AF\u09BC\u09C7",
13299
+ tr: "ile",
13300
+ qu: "wan",
13301
+ // VSO with-words. The corpus puts the with-element AFTER the event
13302
+ // (`استبدل #a عند نقر بـ#b`, `palitan_pwesto #a kapag click nang #b`);
13303
+ // the vso-verb-first generator's swap-gated trailing group binds it to
13304
+ // `destination` via these words. ar's `بـ` is the bi-proclitic + tatweel
13305
+ // exactly as the ArabicProcliticExtractor emits it (glued to a selector
13306
+ // sigil). See generateVSOVerbFirstEventHandlerPattern.
13307
+ ar: "\u0628\u0640",
13308
+ tl: "nang"
12803
13309
  }
12804
13310
  }
12805
13311
  ]
@@ -12888,13 +13394,13 @@ var init_command_schemas = __esm({
12888
13394
  };
12889
13395
  pickSchema = {
12890
13396
  action: "pick",
12891
- description: "Select a random element from a collection",
13397
+ description: "Select item(s), character(s), a range, first/last/random N, or regex matches from a root",
12892
13398
  category: "variable",
12893
13399
  primaryRole: "patient",
12894
13400
  roles: [
12895
13401
  {
12896
13402
  role: "patient",
12897
- description: "The items to pick from",
13403
+ description: "The range/count/index/regex argument to pick",
12898
13404
  required: true,
12899
13405
  expectedTypes: ["literal", "expression", "reference"],
12900
13406
  svoPosition: 1,
@@ -12902,7 +13408,7 @@ var init_command_schemas = __esm({
12902
13408
  },
12903
13409
  {
12904
13410
  role: "source",
12905
- description: 'The array to pick from (with "from" keyword)',
13411
+ description: 'The root to pick from (with "of"/"from" keyword)',
12906
13412
  required: false,
12907
13413
  expectedTypes: ["reference", "expression"],
12908
13414
  svoPosition: 2,
@@ -12944,32 +13450,6 @@ var init_command_schemas = __esm({
12944
13450
  }
12945
13451
  ]
12946
13452
  };
12947
- URL_MARKER_ALL_LANGS = {
12948
- en: "url",
12949
- es: "url",
12950
- pt: "url",
12951
- fr: "url",
12952
- de: "url",
12953
- it: "url",
12954
- ja: "url",
12955
- ko: "url",
12956
- zh: "url",
12957
- ar: "url",
12958
- he: "url",
12959
- hi: "url",
12960
- bn: "url",
12961
- tr: "url",
12962
- ru: "url",
12963
- uk: "url",
12964
- pl: "url",
12965
- id: "url",
12966
- vi: "url",
12967
- th: "url",
12968
- ms: "url",
12969
- tl: "url",
12970
- sw: "url",
12971
- qu: "url"
12972
- };
12973
13453
  PARTIALS_IN_MARKER_ALL_LANGS = {
12974
13454
  en: "partials in",
12975
13455
  es: "partials in",
@@ -14643,17 +15123,48 @@ var init_generic_extractors = __esm({
14643
15123
  });
14644
15124
 
14645
15125
  // src/tokenizers/extractors/css-selector.ts
15126
+ function consumePseudoSegments(input, pos2) {
15127
+ let end = pos2;
15128
+ while (end < input.length && input[end] === ":") {
15129
+ const m = input.slice(end).match(/^::?[a-zA-Z][a-zA-Z0-9-]*/);
15130
+ if (!m) break;
15131
+ let segEnd = end + m[0].length;
15132
+ if (input[segEnd] === "(") {
15133
+ let depth = 0;
15134
+ let p = segEnd;
15135
+ while (p < input.length) {
15136
+ if (input[p] === "(") depth++;
15137
+ else if (input[p] === ")") {
15138
+ depth--;
15139
+ if (depth === 0) {
15140
+ p++;
15141
+ break;
15142
+ }
15143
+ }
15144
+ p++;
15145
+ }
15146
+ if (depth !== 0) break;
15147
+ segEnd = p;
15148
+ }
15149
+ end = segEnd;
15150
+ }
15151
+ return end;
15152
+ }
14646
15153
  function extractCssSelector(input, position) {
14647
15154
  const char = input[position];
14648
15155
  if (char === "#") {
14649
15156
  const match = input.slice(position).match(/^#[a-zA-Z_][\w-]*/);
14650
- return match ? match[0] : null;
15157
+ if (!match) return null;
15158
+ const end = consumePseudoSegments(input, position + match[0].length);
15159
+ return input.slice(position, end);
14651
15160
  }
14652
15161
  if (char === ".") {
14653
15162
  const dynamic = input.slice(position).match(/^\.\{[a-zA-Z_$][\w$]*\}/);
14654
15163
  if (dynamic) return dynamic[0];
14655
15164
  const match = input.slice(position).match(/^\.[a-zA-Z_][\w-]*/);
14656
- return match ? match[0] : null;
15165
+ if (!match) return null;
15166
+ const end = consumePseudoSegments(input, position + match[0].length);
15167
+ return input.slice(position, end);
14657
15168
  }
14658
15169
  if (char === "@") {
14659
15170
  const match = input.slice(position).match(/^@[a-zA-Z_][\w-]*/);
@@ -14671,7 +15182,8 @@ function extractCssSelector(input, position) {
14671
15182
  if (input[end] === "]") {
14672
15183
  depth--;
14673
15184
  if (depth === 0) {
14674
- return input.slice(position, end + 1);
15185
+ const pseudoEnd = consumePseudoSegments(input, end + 1);
15186
+ return input.slice(position, pseudoEnd);
14675
15187
  }
14676
15188
  }
14677
15189
  end++;
@@ -14679,7 +15191,9 @@ function extractCssSelector(input, position) {
14679
15191
  return null;
14680
15192
  }
14681
15193
  if (char === "<") {
14682
- const match = input.slice(position).match(/^<(?=[\w.#[])[\w-]*(?:[#.][\w-]+|\[[^\]]+\])*\s*\/>/);
15194
+ const match = input.slice(position).match(
15195
+ /^<(?=[\w.#[])[\w-]*(?:[#.][\w-]+|\[[^\]]+\]|::?[a-zA-Z][a-zA-Z0-9-]*(?:\([^)]*\))?)*\s*\/>/
15196
+ );
14683
15197
  return match ? match[0] : null;
14684
15198
  }
14685
15199
  return null;
@@ -14745,29 +15259,38 @@ var init_event_modifier = __esm({
14745
15259
  });
14746
15260
 
14747
15261
  // src/tokenizers/extractors/url.ts
15262
+ function findInterpolationEnd(input, start) {
15263
+ let depth = 1;
15264
+ for (let i = start; i < input.length; i++) {
15265
+ const ch = input[i];
15266
+ if (ch === "{") depth++;
15267
+ else if (ch === "}" && --depth === 0) return i + 1;
15268
+ }
15269
+ return -1;
15270
+ }
14748
15271
  function extractUrl(input, position) {
14749
15272
  const remaining = input.slice(position);
14750
- if (remaining.startsWith("http://") || remaining.startsWith("https://")) {
14751
- const match = remaining.match(/^https?:\/\/[^\s]*/);
14752
- return match ? match[0] : null;
14753
- }
14754
- if (remaining.startsWith("//")) {
14755
- const match = remaining.match(/^\/\/[^\s]*/);
14756
- return match ? match[0] : null;
14757
- }
14758
- if (remaining.startsWith("./") || remaining.startsWith("../")) {
14759
- const match = remaining.match(/^\.\.?\/[^\s]*/);
14760
- return match ? match[0] : null;
14761
- }
14762
- if (remaining.startsWith("/")) {
14763
- const match = remaining.match(/^\/[^\s]*/);
14764
- return match ? match[0] : null;
15273
+ const prefix = URL_PREFIXES.find((p) => remaining.startsWith(p));
15274
+ if (!prefix) return null;
15275
+ let i = prefix.length;
15276
+ while (i < remaining.length) {
15277
+ const ch = remaining[i];
15278
+ if (ch === "$" && remaining[i + 1] === "{") {
15279
+ const end = findInterpolationEnd(remaining, i + 2);
15280
+ if (end !== -1) {
15281
+ i = end;
15282
+ continue;
15283
+ }
15284
+ }
15285
+ if (/\s/.test(ch)) break;
15286
+ i++;
14765
15287
  }
14766
- return null;
15288
+ return remaining.slice(0, i);
14767
15289
  }
14768
- var UrlExtractor;
15290
+ var URL_PREFIXES, UrlExtractor;
14769
15291
  var init_url = __esm({
14770
15292
  "src/tokenizers/extractors/url.ts"() {
15293
+ URL_PREFIXES = ["http://", "https://", "//", "./", "../", "/"];
14771
15294
  UrlExtractor = class {
14772
15295
  constructor() {
14773
15296
  this.name = "url";
@@ -15860,6 +16383,18 @@ var init_arabic_proclitic = __esm({
15860
16383
  checkPos++;
15861
16384
  }
15862
16385
  if (remainingLength < 2) {
16386
+ const runIsTatweelOnly = remainingLength >= 1 && input.slice(nextPos, checkPos).split("").every((c) => c === "\u0640");
16387
+ const followChar = input[checkPos];
16388
+ if (entry.type === "preposition" && runIsTatweelOnly && (followChar === "#" || followChar === ".")) {
16389
+ return {
16390
+ value: input.slice(position, checkPos),
16391
+ length: checkPos - position,
16392
+ metadata: {
16393
+ procliticType: entry.type,
16394
+ normalized: entry.normalized
16395
+ }
16396
+ };
16397
+ }
15863
16398
  return null;
15864
16399
  }
15865
16400
  return {
@@ -16240,6 +16775,17 @@ var init_hindi_keyword = __esm({
16240
16775
  pos2 = extPos;
16241
16776
  }
16242
16777
  }
16778
+ if (this.context && input[pos2] === "_" && pos2 + 1 < input.length && isDevanagari(input[pos2 + 1])) {
16779
+ let extPos = pos2;
16780
+ let ext = word;
16781
+ while (extPos < input.length && (input[extPos] === "_" || isDevanagari(input[extPos]))) {
16782
+ ext += input[extPos++];
16783
+ }
16784
+ if (this.context.lookupKeyword(ext)) {
16785
+ word = ext;
16786
+ pos2 = extPos;
16787
+ }
16788
+ }
16243
16789
  if (!word) return null;
16244
16790
  const keywordEntry = this.context.lookupKeyword(word);
16245
16791
  const normalized2 = keywordEntry && keywordEntry.normalized !== keywordEntry.native ? keywordEntry.normalized : void 0;
@@ -16301,9 +16847,11 @@ var init_hindi_particle = __esm({
16301
16847
  }
16302
16848
  setContext(context) {
16303
16849
  this._context = context;
16304
- void this._context;
16305
16850
  }
16306
16851
  canExtract(input, position) {
16852
+ if (this.underscoreJoinedKeyword(input, position)) {
16853
+ return false;
16854
+ }
16307
16855
  for (const [particle] of COMPOUND_POSTPOSITIONS) {
16308
16856
  if (input.startsWith(particle, position)) {
16309
16857
  return true;
@@ -16317,7 +16865,27 @@ var init_hindi_particle = __esm({
16317
16865
  }
16318
16866
  return SINGLE_POSTPOSITIONS.has(word);
16319
16867
  }
16868
+ /**
16869
+ * True when the Devanagari run at `position` is `_`-joined into a keyword the
16870
+ * profile/EXTRAS registered (के_रूप_में). See the note in canExtract.
16871
+ */
16872
+ underscoreJoinedKeyword(input, position) {
16873
+ if (!this._context) return false;
16874
+ let pos2 = position;
16875
+ while (pos2 < input.length && this.isDevanagari(input[pos2])) pos2++;
16876
+ if (input[pos2] !== "_" || pos2 + 1 >= input.length || !this.isDevanagari(input[pos2 + 1])) {
16877
+ return false;
16878
+ }
16879
+ let ext = input.slice(position, pos2);
16880
+ while (pos2 < input.length && (input[pos2] === "_" || this.isDevanagari(input[pos2]))) {
16881
+ ext += input[pos2++];
16882
+ }
16883
+ return Boolean(this._context.lookupKeyword(ext));
16884
+ }
16320
16885
  extract(input, position) {
16886
+ if (this.underscoreJoinedKeyword(input, position)) {
16887
+ return null;
16888
+ }
16321
16889
  for (const [particle, metadata2] of COMPOUND_POSTPOSITIONS) {
16322
16890
  if (input.startsWith(particle, position)) {
16323
16891
  return {
@@ -16941,6 +17509,17 @@ var init_indonesian_keyword = __esm({
16941
17509
  while (pos2 < input.length && isIndonesianIdentifierChar(input[pos2])) {
16942
17510
  word += input[pos2++];
16943
17511
  }
17512
+ if (this.context && pos2 < input.length && input[pos2] === "_") {
17513
+ let extPos = pos2;
17514
+ let ext = word;
17515
+ while (extPos < input.length && (input[extPos] === "_" || isIndonesianIdentifierChar(input[extPos]))) {
17516
+ ext += input[extPos++];
17517
+ }
17518
+ if (this.context.lookupKeyword(ext.toLowerCase())) {
17519
+ word = ext;
17520
+ pos2 = extPos;
17521
+ }
17522
+ }
16944
17523
  if (!word) return null;
16945
17524
  const lower = word.toLowerCase();
16946
17525
  const isPreposition = PREPOSITIONS5.has(lower);
@@ -17391,14 +17970,16 @@ var init_quechua_keyword = __esm({
17391
17970
  metadata: { suffixValue: hyphenSuffix.toLowerCase() }
17392
17971
  };
17393
17972
  }
17394
- const maxKeywordLen = 12;
17973
+ const maxKeywordLen = 13;
17395
17974
  for (let len = Math.min(maxKeywordLen, input.length - startPos); len >= 2; len--) {
17396
17975
  const candidate = input.slice(startPos, startPos + len);
17397
17976
  const after = input[startPos + len];
17398
17977
  if (after !== void 0 && isQuechuaLetter(after)) continue;
17399
17978
  let allQuechua = true;
17400
17979
  for (let i = 0; i < candidate.length; i++) {
17401
- if (!isQuechuaLetter(candidate[i])) {
17980
+ const ch = candidate[i];
17981
+ if (ch === "_" && i > 0 && i < candidate.length - 1) continue;
17982
+ if (!isQuechuaLetter(ch)) {
17402
17983
  allQuechua = false;
17403
17984
  break;
17404
17985
  }
@@ -18285,6 +18866,12 @@ var init_japanese2 = __esm({
18285
18866
  { native: "\u524D", normalized: "previous" },
18286
18867
  { native: "\u6700\u3082\u8FD1\u3044", normalized: "closest" },
18287
18868
  { native: "\u89AA", normalized: "parent" },
18869
+ // Containment (`first <button/> in .modal`): the i18n dict emits の中, which
18870
+ // otherwise splits の(particle) + 中(identifier) — the stray identifier broke
18871
+ // the generated focus pattern's operand run (focus-trap Family G; tr/bn/hi
18872
+ // work because their in-word is one token). Whole-token entry mirrors en's
18873
+ // keyword `in` mid-run geometry.
18874
+ { native: "\u306E\u4E2D", normalized: "in" },
18288
18875
  // Events
18289
18876
  { native: "\u30AF\u30EA\u30C3\u30AF", normalized: "click" },
18290
18877
  { native: "\u5909\u66F4", normalized: "change" },
@@ -18313,6 +18900,14 @@ var init_japanese2 = __esm({
18313
18900
  // References (alternative forms not in profile)
18314
18901
  { native: "\u79C1", normalized: "me" },
18315
18902
  // Alternative to 自分 (jibun)
18903
+ // The i18n dict emits 対象 for `target` while the profile carries ターゲット, so the
18904
+ // word the authored corpus actually uses did not lex as a keyword and leaked into
18905
+ // the condition's raw expression (`if 対象 一致する .modal-backdrop`). Additive: the
18906
+ // profile's ターゲット stays registered. Must land WITH the `matches` keyword —
18907
+ // fixing the operand alone leaves the operator leaking and vice versa (see the
18908
+ // R2 note in japanese.ts's profile `matches` entry).
18909
+ { native: "\u5BFE\u8C61", normalized: "target" },
18910
+ // Alternative to ターゲット (the dict's word)
18316
18911
  // Note: Attached particle forms (を切り替え, を追加, etc.) are intentionally NOT included
18317
18912
  // because they would cause ambiguous parsing. The separate particle + verb pattern
18318
18913
  // (を + 切り替え) is preferred for consistent semantic analysis.
@@ -18324,7 +18919,11 @@ var init_japanese2 = __esm({
18324
18919
  { native: "\u79D2", normalized: "s" },
18325
18920
  { native: "\u30DF\u30EA\u79D2", normalized: "ms" },
18326
18921
  { native: "\u5206", normalized: "m" },
18327
- { native: "\u6642\u9593", normalized: "h" }
18922
+ { native: "\u6642\u9593", normalized: "h" },
18923
+ { native: "\u542B\u3080", normalized: "inclusive" },
18924
+ { native: "\u9664\u304F", normalized: "exclusive" },
18925
+ { native: "\u6587\u5B57", normalized: "characters" },
18926
+ { native: "\u30E9\u30F3\u30C0\u30E0", normalized: "random" }
18328
18927
  ];
18329
18928
  JapaneseTokenizer = class extends BaseTokenizer {
18330
18929
  constructor() {
@@ -18758,6 +19357,11 @@ var init_korean2 = __esm({
18758
19357
  { native: "\uAC70\uC9D3", normalized: "false" },
18759
19358
  { native: "\uB110", normalized: "null" },
18760
19359
  { native: "\uBBF8\uC815\uC758", normalized: "undefined" },
19360
+ // The corpus authors 정의안됨 ("not defined") for undefined (behavior-removable/
19361
+ // sortable `만약 X 이다 정의안됨`); without a whole-token entry it shatters into
19362
+ // 정 + 의안됨, leaking the invalid `is 정 의안됨`. Longest-first scan (cap 6)
19363
+ // matches the 4-char compound whole, like 마우스다운 above.
19364
+ { native: "\uC815\uC758\uC548\uB428", normalized: "undefined" },
18761
19365
  // Positional
18762
19366
  { native: "\uCCAB\uBC88\uC9F8", normalized: "first" },
18763
19367
  { native: "\uB9C8\uC9C0\uB9C9", normalized: "last" },
@@ -18765,6 +19369,11 @@ var init_korean2 = __esm({
18765
19369
  { native: "\uC774\uC804", normalized: "previous" },
18766
19370
  { native: "\uAC00\uC7A5\uAC00\uAE4C\uC6B4", normalized: "closest" },
18767
19371
  { native: "\uBD80\uBAA8", normalized: "parent" },
19372
+ // Containment (`first <button/> in .modal`): the i18n dict emits 안에, which
19373
+ // otherwise splits 안(identifier) + 에(particle) — the stray identifier broke
19374
+ // the generated focus pattern's operand run (focus-trap Family G). Whole-token
19375
+ // entry mirrors en's keyword `in` mid-run geometry.
19376
+ { native: "\uC548\uC5D0", normalized: "in" },
18768
19377
  // Events
18769
19378
  { native: "\uD074\uB9AD", normalized: "click" },
18770
19379
  { native: "\uB354\uBE14\uD074\uB9AD", normalized: "dblclick" },
@@ -18797,7 +19406,11 @@ var init_korean2 = __esm({
18797
19406
  { native: "\uCD08", normalized: "s" },
18798
19407
  { native: "\uBC00\uB9AC\uCD08", normalized: "ms" },
18799
19408
  { native: "\uBD84", normalized: "m" },
18800
- { native: "\uC2DC\uAC04", normalized: "h" }
19409
+ { native: "\uC2DC\uAC04", normalized: "h" },
19410
+ { native: "\uD3EC\uD568", normalized: "inclusive" },
19411
+ { native: "\uC81C\uC678", normalized: "exclusive" },
19412
+ { native: "\uBB38\uC790", normalized: "characters" },
19413
+ { native: "\uBB34\uC791\uC704", normalized: "random" }
18801
19414
  ];
18802
19415
  KoreanTokenizer = class extends BaseTokenizer {
18803
19416
  constructor() {
@@ -19066,6 +19679,17 @@ var init_arabic2 = __esm({
19066
19679
  // ka- (like, as)
19067
19680
  ]);
19068
19681
  ARABIC_EXTRAS = [
19682
+ // References (alternative forms not in profile). The i18n dict emits the BARE
19683
+ // nouns هدف/نتيجة while the profile carries the definite-article forms
19684
+ // الهدف/النتيجة, so the words the authored corpus actually uses did not lex as
19685
+ // keywords and leaked into the condition's raw expression (`if هدف يطابق …`).
19686
+ // Additive: the profile's الهدف/النتيجة stay registered. Same direction as the
19687
+ // profile's `body: 'جسم'` note — align to what the dict emits, never the reverse
19688
+ // (the dict wins on regeneration, so profile→dict is the convergent direction).
19689
+ { native: "\u0647\u062F\u0641", normalized: "target" },
19690
+ // Alternative to الهدف (the dict's word)
19691
+ { native: "\u0646\u062A\u064A\u062C\u0629", normalized: "result" },
19692
+ // Alternative to النتيجة (the dict's word)
19069
19693
  // Values/Literals
19070
19694
  { native: "\u0635\u062D\u064A\u062D", normalized: "true" },
19071
19695
  { native: "\u062E\u0637\u0623", normalized: "false" },
@@ -19130,13 +19754,17 @@ var init_arabic2 = __esm({
19130
19754
  { native: "\u062D\u064A\u0646", normalized: "on" },
19131
19755
  { native: "\u0644\u0645\u0651\u0627", normalized: "on" },
19132
19756
  { native: "\u0644\u0645\u0627", normalized: "on" },
19133
- { native: "\u0644\u062F\u0649", normalized: "on" }
19757
+ { native: "\u0644\u062F\u0649", normalized: "on" },
19134
19758
  //
19135
19759
  // Command spelling variants are now in the profile alternatives:
19136
19760
  // - toggle: بدل, غيّر, غير (in profile)
19137
19761
  // - add: اضف, زِد (in profile)
19138
19762
  // - remove: أزل, امسح (in profile)
19139
19763
  // - etc.
19764
+ { native: "\u0634\u0627\u0645\u0644", normalized: "inclusive" },
19765
+ { native: "\u062D\u0635\u0631\u064A", normalized: "exclusive" },
19766
+ { native: "\u062D\u0631\u0648\u0641", normalized: "characters" },
19767
+ { native: "\u0639\u0634\u0648\u0627\u0626\u064A", normalized: "random" }
19140
19768
  ];
19141
19769
  ArabicTokenizer = class extends BaseTokenizer {
19142
19770
  constructor() {
@@ -19220,7 +19848,7 @@ var init_arabic2 = __esm({
19220
19848
  pos2++;
19221
19849
  }
19222
19850
  }
19223
- return new TokenStreamImpl(tokens, this.language);
19851
+ return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
19224
19852
  }
19225
19853
  classifyToken(token) {
19226
19854
  if (CONJUNCTIONS2.has(token)) return "conjunction";
@@ -19664,8 +20292,12 @@ var init_spanish2 = __esm({
19664
20292
  // Reference alternatives (accent variation, synonym)
19665
20293
  { native: "m\xED", normalized: "me" },
19666
20294
  // Accented form of mi
19667
- { native: "destino", normalized: "target" }
20295
+ { native: "destino", normalized: "target" },
19668
20296
  // Synonym for objetivo
20297
+ { native: "inclusivo", normalized: "inclusive" },
20298
+ { native: "exclusivo", normalized: "exclusive" },
20299
+ { native: "caracteres", normalized: "characters" },
20300
+ { native: "aleatorio", normalized: "random" }
19669
20301
  ];
19670
20302
  SpanishTokenizer = class extends BaseTokenizer {
19671
20303
  constructor() {
@@ -20122,6 +20754,19 @@ var init_turkish2 = __esm({
20122
20754
  { native: "farebirak", normalized: "mouseup" },
20123
20755
  { native: "kayd\u0131r", normalized: "scroll" },
20124
20756
  { native: "kaydir", normalized: "scroll" },
20757
+ // resize/scroll nominal forms: listed in eventNameTranslations (which only
20758
+ // the SOV-extraction path consults) but not registered as keywords — so a
20759
+ // fused *-sov-simple match captured them RAW (`boyutlandırma de çağır` →
20760
+ // event:expression:boyutlandırma, the window-resize R1 flip once the
20761
+ // debounced-head junk no longer forced the SOV-extraction path). Keyword
20762
+ // entries normalize them at the token, the same route the healthy natives
20763
+ // (tıklama→click) take.
20764
+ { native: "boyutland\u0131rma", normalized: "resize" },
20765
+ { native: "boyutlandirma", normalized: "resize" },
20766
+ { native: "boyutland\u0131r", normalized: "resize" },
20767
+ { native: "boyutlandir", normalized: "resize" },
20768
+ { native: "kayd\u0131rma", normalized: "scroll" },
20769
+ { native: "kaydirma", normalized: "scroll" },
20125
20770
  { native: "tu\u015F_bas", normalized: "keydown" },
20126
20771
  { native: "tus_bas", normalized: "keydown" },
20127
20772
  { native: "tu\u015F_b\u0131rak", normalized: "keyup" },
@@ -20130,7 +20775,11 @@ var init_turkish2 = __esm({
20130
20775
  { native: "saniye", normalized: "s" },
20131
20776
  { native: "milisaniye", normalized: "ms" },
20132
20777
  { native: "dakika", normalized: "m" },
20133
- { native: "saat", normalized: "h" }
20778
+ { native: "saat", normalized: "h" },
20779
+ { native: "dahil", normalized: "inclusive" },
20780
+ { native: "hari\xE7", normalized: "exclusive" },
20781
+ { native: "karakterler", normalized: "characters" },
20782
+ { native: "rastgele", normalized: "random" }
20134
20783
  ];
20135
20784
  TurkishTokenizer = class extends BaseTokenizer {
20136
20785
  constructor() {
@@ -20309,7 +20958,16 @@ var init_chinese2 = __esm({
20309
20958
  { native: "\u524D", normalized: "before" },
20310
20959
  { native: "\u540E", normalized: "after" },
20311
20960
  { native: "\u90A3\u4E48", normalized: "then" },
20312
- { native: "\u5B8C", normalized: "end" }
20961
+ { native: "\u5B8C", normalized: "end" },
20962
+ // Connectives. Whole-token so the greedy longest-first walk claims the 2-char
20963
+ // 作为 (`as`) before its 1-char tail 为 can match the `for` command primary —
20964
+ // without it `作为 Number` shattered into `作` + `为`→`for` (`computed-value`).
20965
+ // The reverse render (CONNECTIVE_LEXICON.zh) already maps 作为→as.
20966
+ { native: "\u4F5C\u4E3A", normalized: "as" },
20967
+ { native: "\u5305\u542B", normalized: "inclusive" },
20968
+ { native: "\u6392\u9664", normalized: "exclusive" },
20969
+ { native: "\u5B57\u7B26", normalized: "characters" },
20970
+ { native: "\u968F\u673A", normalized: "random" }
20313
20971
  ];
20314
20972
  ChineseTokenizer = class extends BaseTokenizer {
20315
20973
  constructor() {
@@ -20811,7 +21469,11 @@ var init_portuguese2 = __esm({
20811
21469
  { native: "padrao", normalized: "default" },
20812
21470
  { native: "at\xE9 que", normalized: "until" },
20813
21471
  // Multi-word phrases
20814
- { native: "dentro de", normalized: "into" }
21472
+ { native: "dentro de", normalized: "into" },
21473
+ { native: "inclusivo", normalized: "inclusive" },
21474
+ { native: "exclusivo", normalized: "exclusive" },
21475
+ { native: "caracteres", normalized: "characters" },
21476
+ { native: "aleat\xF3rio", normalized: "random" }
20815
21477
  ];
20816
21478
  PortugueseTokenizer = class extends BaseTokenizer {
20817
21479
  constructor() {
@@ -21275,7 +21937,11 @@ var init_french2 = __esm({
21275
21937
  // Additional morph synonym
21276
21938
  { native: "transmuter", normalized: "morph" },
21277
21939
  // Multi-word phrases
21278
- { native: "tant que", normalized: "while" }
21940
+ { native: "tant que", normalized: "while" },
21941
+ { native: "inclusif", normalized: "inclusive" },
21942
+ { native: "exclusif", normalized: "exclusive" },
21943
+ { native: "caract\xE8res", normalized: "characters" },
21944
+ { native: "al\xE9atoire", normalized: "random" }
21279
21945
  ];
21280
21946
  FrenchTokenizer = class extends BaseTokenizer {
21281
21947
  constructor() {
@@ -21716,7 +22382,11 @@ var init_german2 = __esm({
21716
22382
  // Verb conjugation variants (imperatives for test cases)
21717
22383
  { native: "erh\xF6he", normalized: "increment" },
21718
22384
  { native: "erhohe", normalized: "increment" },
21719
- { native: "verringere", normalized: "decrement" }
22385
+ { native: "verringere", normalized: "decrement" },
22386
+ { native: "inklusiv", normalized: "inclusive" },
22387
+ { native: "exklusiv", normalized: "exclusive" },
22388
+ { native: "Zeichen", normalized: "characters" },
22389
+ { native: "zuf\xE4llig", normalized: "random" }
21720
22390
  ];
21721
22391
  GermanTokenizer = class extends BaseTokenizer {
21722
22392
  constructor() {
@@ -21813,12 +22483,27 @@ var init_indonesian2 = __esm({
21813
22483
  // outside
21814
22484
  ]);
21815
22485
  INDONESIAN_EXTRAS = [
22486
+ // window-resize compound: the dict emits underscore-joined ubah_ukuran
22487
+ // (resize), which the `_` split shattered into ubah(→change) + _ + ukuran —
22488
+ // the event slot normalized to `change` and `_ ukuran` dropped unconsumed
22489
+ // (Arc F). Whole-token entry mirrors qu's hatun_kay precedent (quechua.ts).
22490
+ { native: "ubah_ukuran", normalized: "resize" },
22491
+ // behavior-draggable's `no` operator: the dict emits underscore-joined
22492
+ // tidak_ada, which the `_` split shattered into tidak(→not) + _ + ada(→exists).
22493
+ // Whole-token entry mirrors ubah_ukuran above; the keyword walk sorts
22494
+ // longest-first, so `tidak_ada` (9) beats `tidak` (5).
22495
+ { native: "tidak_ada", normalized: "no" },
21816
22496
  // Values/Literals
21817
22497
  { native: "benar", normalized: "true" },
21818
22498
  { native: "salah", normalized: "false" },
21819
22499
  { native: "null", normalized: "null" },
21820
22500
  { native: "kosong", normalized: "null" },
21821
22501
  { native: "tidakdidefinisikan", normalized: "undefined" },
22502
+ // The corpus authors `tidak_terdefinisi` for undefined (behavior-removable/
22503
+ // sortable `jika X adalah tidak_terdefinisi`); without a whole-token entry the
22504
+ // `_` split shatters it into tidak(→not) + `_ terdefinisi`, leaking the
22505
+ // invalid `is not _ terdefinisi`. Same shape as tidak_ada above.
22506
+ { native: "tidak_terdefinisi", normalized: "undefined" },
21822
22507
  // Positional
21823
22508
  { native: "pertama", normalized: "first" },
21824
22509
  { native: "terakhir", normalized: "last" },
@@ -21849,7 +22534,11 @@ var init_indonesian2 = __esm({
21849
22534
  { native: "atau", normalized: "or" },
21850
22535
  { native: "tidak", normalized: "not" },
21851
22536
  { native: "adalah", normalized: "is" },
21852
- { native: "ada", normalized: "exists" }
22537
+ { native: "ada", normalized: "exists" },
22538
+ { native: "inklusif", normalized: "inclusive" },
22539
+ { native: "eksklusif", normalized: "exclusive" },
22540
+ { native: "karakter", normalized: "characters" },
22541
+ { native: "acak", normalized: "random" }
21853
22542
  ];
21854
22543
  IndonesianTokenizer = class extends BaseTokenizer {
21855
22544
  constructor() {
@@ -22064,7 +22753,7 @@ var init_quechua2 = __esm({
22064
22753
  this.name = "quechua-string-literal";
22065
22754
  }
22066
22755
  canExtract(input, position) {
22067
- return input[position] === '"' || input[position] === "'";
22756
+ return input[position] === '"' || input[position] === "'" || input[position] === "`";
22068
22757
  }
22069
22758
  extract(input, position) {
22070
22759
  const quote = input[position];
@@ -22123,6 +22812,8 @@ var init_quechua2 = __esm({
22123
22812
  // (set-attribute `@disabled ta cheqaq man …`); without it the value tokenized
22124
22813
  // as a bare identifier and `set @disabled to <undefined>` ran. arí/ari ("yes")
22125
22814
  // are the colloquial alternates, kept for input tolerance.
22815
+ // Pick unit word (arc 3) — mirrors the i18n dict's `characters: 'sanampa'`.
22816
+ { native: "sanampa", normalized: "characters" },
22126
22817
  { native: "cheqaq", normalized: "true" },
22127
22818
  { native: "ar\xED", normalized: "true" },
22128
22819
  { native: "ari", normalized: "true" },
@@ -22157,6 +22848,31 @@ var init_quechua2 = __esm({
22157
22848
  // aswan-prefixed compound splits (the suffix extractor strips -wan from
22158
22849
  // 'aswan'). The i18n dict emits bare 'kaylla' (near/close).
22159
22850
  { native: "kaylla", normalized: "closest" },
22851
+ // Containment (`first <button/> in .modal`): the i18n dict emits ukupi,
22852
+ // which otherwise splits uku(identifier) + pi — and the stranded `pi`
22853
+ // mis-reads as the EVENT marker (the ñawpaqpi/qhepapi class above; same
22854
+ // longest-first cure). Whole-token entry mirrors en's keyword `in` mid-run
22855
+ // geometry (focus-trap Family G).
22856
+ { native: "ukupi", normalized: "in" },
22857
+ // window-resize compounds: the dict emits underscore-joined k_iri (window)
22858
+ // and hatun_kay (resize), which the `_` split shattered into junk role
22859
+ // fragments (call.source:literal="k_iri" destination:literal="hatun_" —
22860
+ // the qu window-resize R1 row; hatun_kay sits in eventNameTranslations but
22861
+ // never arrived whole). The ñawpaq_kaq entry above is the precedent.
22862
+ { native: "k_iri", normalized: "window" },
22863
+ { native: "hatun_kay", normalized: "resize" },
22864
+ // behavior-draggable's `no` operator: the dict emits underscore-joined
22865
+ // mana_kanchu, which the `_` split shattered into mana(→not/without) + _ +
22866
+ // kanchu. Same whole-token shape as hatun_kay; longest-first makes
22867
+ // `mana_kanchu` (11) beat `mana` (4).
22868
+ { native: "mana_kanchu", normalized: "no" },
22869
+ // `undefined`: the dict emits underscore-joined `mana_riqsisqa` ("not known"),
22870
+ // which the `_` split shattered into mana(→false) + _ + riqsisqa — rendering
22871
+ // `is false _ riqsisqa` and breaking the canonical parse (behavior-removable/qu,
22872
+ // behavior-sortable/qu `if triggerEl is undefined`). The bare `mana riqsisqa`
22873
+ // (space) entry above never fires — the corpus authors the underscore form.
22874
+ // Same whole-token shape as mana_kanchu; longest-first makes it beat `mana`.
22875
+ { native: "mana_riqsisqa", normalized: "undefined" },
22160
22876
  { native: "qaylla", normalized: "closest" },
22161
22877
  { native: "tayta", normalized: "parent" },
22162
22878
  // Events
@@ -22218,7 +22934,8 @@ var init_quechua2 = __esm({
22218
22934
  { native: "qhawachiy", normalized: "focus" },
22219
22935
  { native: "mana qhawachiy", normalized: "blur" },
22220
22936
  // Suffix modifiers
22221
- { native: "-manta", normalized: "from" }
22937
+ { native: "-manta", normalized: "from" },
22938
+ { native: "imaymanata", normalized: "random" }
22222
22939
  ];
22223
22940
  QuechuaTokenizer = class extends BaseTokenizer {
22224
22941
  constructor() {
@@ -22246,7 +22963,7 @@ var init_quechua2 = __esm({
22246
22963
  return "event-modifier";
22247
22964
  if (token.startsWith("#") || token.startsWith(".") || token.startsWith("[") || token.startsWith("*") || token.startsWith("<"))
22248
22965
  return "selector";
22249
- if (token.startsWith('"')) return "literal";
22966
+ if (token.startsWith('"') || token.startsWith("'")) return "literal";
22250
22967
  if (/^\d/.test(token)) return "literal";
22251
22968
  if (["==", "!=", "<=", ">=", "<", ">", "&&", "||", "!"].includes(token)) return "operator";
22252
22969
  return "identifier";
@@ -22315,6 +23032,12 @@ var init_swahili2 = __esm({
22315
23032
  // between
22316
23033
  ]);
22317
23034
  SWAHILI_EXTRAS = [
23035
+ // window-resize compound: the dict emits underscore-joined badilisha_ukubwa
23036
+ // (resize), which the `_` split shattered into badilisha(→toggle!) + _ +
23037
+ // ukubwa — the event slot normalized to `toggle` and `_ ukubwa` dropped
23038
+ // unconsumed (Arc F). Whole-token entry mirrors qu's hatun_kay precedent
23039
+ // (quechua.ts).
23040
+ { native: "badilisha_ukubwa", normalized: "resize" },
22318
23041
  // Values/Literals
22319
23042
  { native: "kweli", normalized: "true" },
22320
23043
  { native: "uongo", normalized: "false" },
@@ -22388,7 +23111,9 @@ var init_swahili2 = __esm({
22388
23111
  { native: "si", normalized: "not" },
22389
23112
  { native: "ni", normalized: "is" },
22390
23113
  { native: "ipo", normalized: "exists" },
22391
- { native: "tupu", normalized: "empty" }
23114
+ { native: "tupu", normalized: "empty" },
23115
+ { native: "herufi", normalized: "characters" },
23116
+ { native: "nasibu", normalized: "random" }
22392
23117
  ];
22393
23118
  SwahiliTokenizer = class extends BaseTokenizer {
22394
23119
  constructor() {
@@ -23072,7 +23797,11 @@ var init_italian2 = __esm({
23072
23797
  { native: "vuoto", normalized: "empty" },
23073
23798
  // Synonyms not in profile
23074
23799
  { native: "toggle", normalized: "toggle" },
23075
- { native: "di", normalized: "tell" }
23800
+ { native: "di", normalized: "tell" },
23801
+ { native: "inclusivo", normalized: "inclusive" },
23802
+ { native: "esclusivo", normalized: "exclusive" },
23803
+ { native: "caratteri", normalized: "characters" },
23804
+ { native: "casuale", normalized: "random" }
23076
23805
  ];
23077
23806
  ItalianTokenizer = class extends BaseTokenizer {
23078
23807
  constructor() {
@@ -23177,7 +23906,11 @@ var init_vietnamese2 = __esm({
23177
23906
  { native: "t\u1ED3n t\u1EA1i", normalized: "exists" },
23178
23907
  { native: "r\u1ED7ng", normalized: "empty" },
23179
23908
  // English synonyms
23180
- { native: "javascript", normalized: "js" }
23909
+ { native: "javascript", normalized: "js" },
23910
+ { native: "bao g\u1ED3m", normalized: "inclusive" },
23911
+ { native: "lo\u1EA1i tr\u1EEB", normalized: "exclusive" },
23912
+ { native: "k\xFD t\u1EF1", normalized: "characters" },
23913
+ { native: "ng\u1EABu nhi\xEAn", normalized: "random" }
23181
23914
  ];
23182
23915
  VietnameseTokenizer = class extends BaseTokenizer {
23183
23916
  constructor() {
@@ -23560,7 +24293,11 @@ var init_polish2 = __esm({
23560
24293
  { native: "jest", normalized: "is" },
23561
24294
  { native: "istnieje", normalized: "exists" },
23562
24295
  { native: "pusty", normalized: "empty" },
23563
- { native: "puste", normalized: "empty" }
24296
+ { native: "puste", normalized: "empty" },
24297
+ { native: "w\u0142\u0105cznie", normalized: "inclusive" },
24298
+ { native: "wy\u0142\u0105cznie", normalized: "exclusive" },
24299
+ { native: "znaki", normalized: "characters" },
24300
+ { native: "losowy", normalized: "random" }
23564
24301
  ];
23565
24302
  PolishTokenizer = class extends BaseTokenizer {
23566
24303
  constructor() {
@@ -23990,6 +24727,12 @@ var init_russian2 = __esm({
23990
24727
  { native: "\u043B\u043E\u0436\u044C", normalized: "false" },
23991
24728
  { native: "null", normalized: "null" },
23992
24729
  { native: "\u043D\u0435\u043E\u043F\u0440\u0435\u0434\u0435\u043B\u0435\u043D\u043E", normalized: "undefined" },
24730
+ // `ничего` ("nothing") is the word the corpus author uses for a null
24731
+ // comparison (`если item есть ничего` → `if item is null`). Without it the
24732
+ // literal leaked verbatim and the canonical parser rejected the render
24733
+ // (behavior-sortable/ru). Its sibling `неопределено`→undefined was already
24734
+ // registered; this closes the null half.
24735
+ { native: "\u043D\u0438\u0447\u0435\u0433\u043E", normalized: "null" },
23993
24736
  // Time units (not in profile - handled by number parser)
23994
24737
  { native: "\u0441\u0435\u043A\u0443\u043D\u0434\u0430", normalized: "s" },
23995
24738
  { native: "\u0441\u0435\u043A\u0443\u043D\u0434\u044B", normalized: "s" },
@@ -24035,8 +24778,11 @@ var init_russian2 = __esm({
24035
24778
  // feminine
24036
24779
  { native: "\u043C\u043E\u0451", normalized: "my" },
24037
24780
  // neuter
24038
- { native: "\u043C\u043E\u0438", normalized: "my" }
24781
+ { native: "\u043C\u043E\u0438", normalized: "my" },
24039
24782
  // plural
24783
+ { native: "\u0432\u043A\u043B\u044E\u0447\u0438\u0442\u0435\u043B\u044C\u043D\u043E", normalized: "inclusive" },
24784
+ { native: "\u0438\u0441\u043A\u043B\u044E\u0447\u0438\u0442\u0435\u043B\u044C\u043D\u043E", normalized: "exclusive" },
24785
+ { native: "\u0441\u0438\u043C\u0432\u043E\u043B\u044B", normalized: "characters" }
24040
24786
  ];
24041
24787
  RussianTokenizer = class extends BaseTokenizer {
24042
24788
  constructor() {
@@ -24445,6 +25191,11 @@ var init_ukrainian2 = __esm({
24445
25191
  { native: "\u0445\u0438\u0431\u043D\u0456\u0441\u0442\u044C", normalized: "false" },
24446
25192
  { native: "null", normalized: "null" },
24447
25193
  { native: "\u043D\u0435\u0432\u0438\u0437\u043D\u0430\u0447\u0435\u043D\u043E", normalized: "undefined" },
25194
+ // `нічого` ("nothing") is the corpus author's word for a null comparison
25195
+ // (`якщо item є нічого` → `if item is null`); without it the literal leaked
25196
+ // verbatim and the canonical parser rejected the render (behavior-sortable/uk).
25197
+ // Sibling of the already-registered `невизначено`→undefined.
25198
+ { native: "\u043D\u0456\u0447\u043E\u0433\u043E", normalized: "null" },
24448
25199
  // Time units (not in profile - handled by number parser)
24449
25200
  { native: "\u0441\u0435\u043A\u0443\u043D\u0434\u0430", normalized: "s" },
24450
25201
  { native: "\u0441\u0435\u043A\u0443\u043D\u0434\u0438", normalized: "s" },
@@ -24490,8 +25241,11 @@ var init_ukrainian2 = __esm({
24490
25241
  // feminine
24491
25242
  { native: "\u043C\u043E\u0454", normalized: "my" },
24492
25243
  // neuter
24493
- { native: "\u043C\u043E\u0457", normalized: "my" }
25244
+ { native: "\u043C\u043E\u0457", normalized: "my" },
24494
25245
  // plural
25246
+ { native: "\u0432\u043A\u043B\u044E\u0447\u043D\u043E", normalized: "inclusive" },
25247
+ { native: "\u0432\u0438\u043A\u043B\u044E\u0447\u043D\u043E", normalized: "exclusive" },
25248
+ { native: "\u0441\u0438\u043C\u0432\u043E\u043B\u0438", normalized: "characters" }
24495
25249
  ];
24496
25250
  UkrainianTokenizer = class extends BaseTokenizer {
24497
25251
  constructor() {
@@ -24631,7 +25385,11 @@ var init_he2 = __esm({
24631
25385
  { native: "\u05D3\u05E7\u05D4", normalized: "m" },
24632
25386
  { native: "\u05D3\u05E7\u05D5\u05EA", normalized: "m" },
24633
25387
  { native: "\u05E9\u05E2\u05D4", normalized: "h" },
24634
- { native: "\u05E9\u05E2\u05D5\u05EA", normalized: "h" }
25388
+ { native: "\u05E9\u05E2\u05D5\u05EA", normalized: "h" },
25389
+ { native: "\u05DB\u05D5\u05DC\u05DC", normalized: "inclusive" },
25390
+ { native: "\u05D1\u05DC\u05E2\u05D3\u05D9", normalized: "exclusive" },
25391
+ { native: "\u05EA\u05D5\u05D5\u05D9\u05DD", normalized: "characters" },
25392
+ { native: "\u05D0\u05E7\u05E8\u05D0\u05D9", normalized: "random" }
24635
25393
  ];
24636
25394
  HebrewTokenizer = class extends BaseTokenizer {
24637
25395
  constructor() {
@@ -24788,6 +25546,12 @@ var init_hindi2 = __esm({
24788
25546
  // splits on it — see hi.ts events note). repeat-until-event / handler events.
24789
25547
  { native: "\u092E\u093E\u0909\u0938\u0928\u0940\u091A\u0947", normalized: "mousedown" },
24790
25548
  { native: "\u092E\u093E\u0909\u0938\u090A\u092A\u0930", normalized: "mouseup" },
25549
+ // window-resize compound: the dict emits underscore-joined आकार_बदलें
25550
+ // (resize), which the `_` split shattered into आकार + _ + बदलें — and the
25551
+ // stranded बदलें (toggle verb) anchored a PHANTOM toggle command while the
25552
+ // event slot grabbed the call target (the hi window-resize mis-parse,
25553
+ // Arc F). Whole-token entry mirrors qu's hatun_kay precedent (quechua.ts).
25554
+ { native: "\u0906\u0915\u093E\u0930_\u092C\u0926\u0932\u0947\u0902", normalized: "resize" },
24791
25555
  // Values
24792
25556
  { native: "\u0938\u091A", normalized: "true" },
24793
25557
  { native: "\u0938\u0924\u094D\u092F", normalized: "true" },
@@ -24811,7 +25575,26 @@ var init_hindi2 = __esm({
24811
25575
  { native: "\u0938\u094D\u0915\u094D\u0930\u0949\u0932", normalized: "scroll" },
24812
25576
  // Additional modifiers not in profile
24813
25577
  { native: "\u0915\u094B", normalized: "to" },
24814
- { native: "\u0915\u0947 \u0938\u093E\u0925", normalized: "with" }
25578
+ { native: "\u0915\u0947 \u0938\u093E\u0925", normalized: "with" },
25579
+ // Connectives. Whole-token underscore-joined surface, mirroring आकार_बदलें
25580
+ // above: the `_` split shattered के_रूप_में (`as`) into के + _ + रूप + _ + में
25581
+ // (`computed-value`). Registering it lets the tokenizer's underscore-recovery
25582
+ // block adopt the whole run. The reverse render (CONNECTIVE_LEXICON.hi) already
25583
+ // maps के_रूप_में→as; it was a documented dead entry awaiting exactly this.
25584
+ { native: "\u0915\u0947_\u0930\u0942\u092A_\u092E\u0947\u0902", normalized: "as" },
25585
+ // `या` (or) — dict hi.ts `or`; already matched by surface in the parser's
25586
+ // OR_KEYWORDS (event-adjacent `or` was absorbed), but every raw-expression
25587
+ // occurrence leaked verbatim (when-multiple-changes). Phantom-safe: `or` is
25588
+ // neither an ActionType nor a command schema.
25589
+ { native: "\u092F\u093E", normalized: "or" },
25590
+ // `बदलने पर` (changes / "on changing") — dict hi.ts `changes`, SPACED whole
25591
+ // phrase via the multi-word keyword walk (`के साथ` precedent above). NEVER
25592
+ // register bare `बदलने`: the stem `बदल` is a registered toggle-verb
25593
+ // alternative (patterns/toggle.ts) and the morphological normalizer strips
25594
+ // conjugations — a bare entry re-opens the आकार_बदलें phantom-toggle class.
25595
+ { native: "\u092C\u0926\u0932\u0928\u0947 \u092A\u0930", normalized: "changes" },
25596
+ { native: "\u0905\u0915\u094D\u0937\u0930", normalized: "characters" },
25597
+ { native: "\u092F\u093E\u0926\u0943\u091A\u094D\u091B\u093F\u0915", normalized: "random" }
24815
25598
  ];
24816
25599
  HindiTokenizer = class extends BaseTokenizer {
24817
25600
  constructor() {
@@ -24993,7 +25776,17 @@ var init_bengali2 = __esm({
24993
25776
  { native: "\u09B8\u09CD\u0995\u09CD\u09B0\u09CB\u09B2", normalized: "scroll" },
24994
25777
  // Additional modifiers not in profile
24995
25778
  { native: "\u0995\u09C7", normalized: "to" },
24996
- { native: "\u09B8\u09BE\u09A5\u09C7", normalized: "with" }
25779
+ { native: "\u09B8\u09BE\u09A5\u09C7", normalized: "with" },
25780
+ // Conjunctions. `অথবা` (or) — dict bn.ts `or`. Already matched by surface in the
25781
+ // parser's OR_KEYWORDS (event-adjacent `or` was absorbed); registering it lets
25782
+ // surfaceOf emit `or` inside raw expressions (the wait-for event list in
25783
+ // behavior-draggable/sortable). Phantom-safe: `or` is neither an ActionType nor
25784
+ // a command schema.
25785
+ { native: "\u0985\u09A5\u09AC\u09BE", normalized: "or" },
25786
+ { native: "\u0985\u09A8\u09CD\u09A4\u09B0\u09CD\u09AD\u09C1\u0995\u09CD\u09A4", normalized: "inclusive" },
25787
+ { native: "\u09AC\u09BE\u09A6", normalized: "exclusive" },
25788
+ { native: "\u0985\u0995\u09CD\u09B7\u09B0", normalized: "characters" },
25789
+ { native: "\u098F\u09B2\u09CB\u09AE\u09C7\u09B2\u09CB", normalized: "random" }
24997
25790
  ];
24998
25791
  BengaliTokenizer = class extends BaseTokenizer {
24999
25792
  constructor() {
@@ -25067,11 +25860,19 @@ var init_thai2 = __esm({
25067
25860
  { native: "\u0E2D\u0E34\u0E19\u0E1E\u0E38\u0E15", normalized: "input" },
25068
25861
  { native: "\u0E42\u0E2B\u0E25\u0E14", normalized: "load" },
25069
25862
  { native: "\u0E40\u0E25\u0E37\u0E48\u0E2D\u0E19", normalized: "scroll" },
25863
+ // `ปรับขนาด` (resize) — dict th.ts `resize`; without it the greedy scan
25864
+ // shattered it into ป + รับ(→take) + ขนาด (window-resize/th rendered
25865
+ // `on ป take ขนาด …`). Precedent: hi आकार_बदलें, tr boyutlandırma.
25866
+ { native: "\u0E1B\u0E23\u0E31\u0E1A\u0E02\u0E19\u0E32\u0E14", normalized: "resize" },
25070
25867
  // Additional modifiers
25071
25868
  { native: "\u0E40\u0E27\u0E25\u0E32", normalized: "when" },
25072
25869
  { native: "\u0E44\u0E1B\u0E22\u0E31\u0E07", normalized: "to" },
25073
25870
  { native: "\u0E14\u0E49\u0E27\u0E22", normalized: "with" },
25074
- { native: "\u0E41\u0E25\u0E30", normalized: "and" }
25871
+ { native: "\u0E41\u0E25\u0E30", normalized: "and" },
25872
+ { native: "\u0E23\u0E27\u0E21", normalized: "inclusive" },
25873
+ { native: "\u0E22\u0E01\u0E40\u0E27\u0E49\u0E19", normalized: "exclusive" },
25874
+ { native: "\u0E2D\u0E31\u0E01\u0E02\u0E23\u0E30", normalized: "characters" },
25875
+ { native: "\u0E2A\u0E38\u0E48\u0E21", normalized: "random" }
25075
25876
  ];
25076
25877
  ThaiTokenizer = class extends BaseTokenizer {
25077
25878
  constructor() {
@@ -25143,8 +25944,12 @@ var init_ms2 = __esm({
25143
25944
  // Alternative for input (means "enter")
25144
25945
  { native: "muat", normalized: "load" },
25145
25946
  { native: "tatal", normalized: "scroll" },
25146
- { native: "hover", normalized: "hover" }
25947
+ { native: "hover", normalized: "hover" },
25147
25948
  // English loanword commonly used
25949
+ { native: "inklusif", normalized: "inclusive" },
25950
+ { native: "eksklusif", normalized: "exclusive" },
25951
+ { native: "aksara", normalized: "characters" },
25952
+ { native: "rawak", normalized: "random" }
25148
25953
  ];
25149
25954
  MalayTokenizer = class extends BaseTokenizer {
25150
25955
  constructor() {
@@ -25403,7 +26208,11 @@ var init_tl2 = __esm({
25403
26208
  { native: "isumite", normalized: "submit" },
25404
26209
  { native: "input", normalized: "input" },
25405
26210
  { native: "karga", normalized: "load" },
25406
- { native: "mag_scroll", normalized: "scroll" }
26211
+ { native: "mag_scroll", normalized: "scroll" },
26212
+ { native: "kasama", normalized: "inclusive" },
26213
+ { native: "bukod", normalized: "exclusive" },
26214
+ { native: "karakter", normalized: "characters" },
26215
+ { native: "random", normalized: "random" }
25407
26216
  ];
25408
26217
  TagalogTokenizer = class extends BaseTokenizer {
25409
26218
  constructor() {
@@ -25983,6 +26792,28 @@ function getEventHandlerPatternsHi() {
25983
26792
  event: { marker: "\u0938\u0947", position: 2 }
25984
26793
  }
25985
26794
  },
26795
+ // Prefix reactive `when` — the hi member of the ja/tr/ar/he when-family
26796
+ // below (`जब $firstName या $lastName बदलने पर …`). Without it,
26797
+ // `event-hi-bare` captured the जब token itself as the event (render
26798
+ // `on when put …`) and dropped the subject list; en's `event-en-when`
26799
+ // captures the first subject as the event. The event role is
26800
+ // type-constrained so the `जब तक` while/until compound (repeat-while,
26801
+ // unless-condition) never matches — तक lexes as a keyword/literal and
26802
+ // declines, falling through to the repeat patterns unchanged.
26803
+ {
26804
+ id: "event-hi-when",
26805
+ language: "hi",
26806
+ command: "on",
26807
+ priority: 95,
26808
+ template: {
26809
+ format: "\u091C\u092C {event} {body}",
26810
+ tokens: [
26811
+ { type: "literal", value: "\u091C\u092C" },
26812
+ { type: "role", role: "event", expectedTypes: ["reference", "expression", "selector"] }
26813
+ ]
26814
+ },
26815
+ extraction: { event: { position: 1 } }
26816
+ },
25986
26817
  // Bare event name: क्लिक
25987
26818
  {
25988
26819
  id: "event-hi-bare",
@@ -27135,7 +27966,15 @@ var init_event_handler = __esm({
27135
27966
  \uBE14\uB7EC: "blur",
27136
27967
  \uB85C\uB4DC: "load",
27137
27968
  \uB9AC\uC0AC\uC774\uC988: "resize",
27138
- \uC2A4\uD06C\uB864: "scroll"
27969
+ \uC2A4\uD06C\uB864: "scroll",
27970
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
27971
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
27972
+ \uB9C8\uC6B0\uC2A4\uC5D4\uD130: "mouseenter",
27973
+ \uB9C8\uC6B0\uC2A4\uB9AC\uBE0C: "mouseleave",
27974
+ \uB9C8\uC6B0\uC2A4\uBB34\uBE0C: "mousemove",
27975
+ \uD0A4\uD504\uB808\uC2A4: "keypress",
27976
+ \uD130\uCE58\uC885\uB8CC: "touchend",
27977
+ \uD130\uCE58\uCDE8\uC18C: "touchcancel"
27139
27978
  },
27140
27979
  // Japanese event names → English
27141
27980
  ja: {
@@ -27155,7 +27994,12 @@ var init_event_handler = __esm({
27155
27994
  \u30ED\u30FC\u30C9: "load",
27156
27995
  \u8AAD\u307F\u8FBC\u307F: "load",
27157
27996
  \u30B5\u30A4\u30BA\u5909\u66F4: "resize",
27158
- \u30B9\u30AF\u30ED\u30FC\u30EB: "scroll"
27997
+ \u30B9\u30AF\u30ED\u30FC\u30EB: "scroll",
27998
+ // V3 Batch 2 alias: i18n dictionary form the ja tokenizer already
27999
+ // normalizes (probe-verified).
28000
+ \u307C\u304B\u3057: "blur"
28001
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28002
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
27159
28003
  },
27160
28004
  // Arabic event names → English
27161
28005
  ar: {
@@ -27172,7 +28016,19 @@ var init_event_handler = __esm({
27172
28016
  "\u062A\u0645\u0631\u064A\u0631 \u0627\u0644\u0645\u0627\u0648\u0633": "mouseover",
27173
28017
  \u0627\u0644\u062A\u0631\u0643\u064A\u0632: "focus",
27174
28018
  \u062A\u062D\u0645\u064A\u0644: "load",
27175
- \u062A\u0645\u0631\u064A\u0631: "scroll"
28019
+ \u062A\u0645\u0631\u064A\u0631: "scroll",
28020
+ // V3 Batch 2 aliases: i18n dictionary forms the ar tokenizer already
28021
+ // normalizes (probe-verified captured values). Appended so first-wins
28022
+ // localization canonicals above are unchanged.
28023
+ \u062A\u0631\u0643\u064A\u0632: "focus",
28024
+ "\u0645\u0641\u062A\u0627\u062D \u0623\u0633\u0641\u0644": "keydown",
28025
+ "\u0645\u0641\u062A\u0627\u062D \u0623\u0639\u0644\u0649": "keyup",
28026
+ "\u0641\u0623\u0631\u0629 \u0641\u0648\u0642": "mouseover",
28027
+ // Arc F: the dict renders resize as the two-word تغيير حجم; the event
28028
+ // slot captures only تغيير (→change) and حجم drops. The compound key is
28029
+ // matched by the parser's event-compound reclaim (offset-exact join of
28030
+ // the captured event word + the dangling fragment).
28031
+ "\u062A\u063A\u064A\u064A\u0631 \u062D\u062C\u0645": "resize"
27176
28032
  },
27177
28033
  // Spanish event names → English
27178
28034
  es: {
@@ -27189,7 +28045,26 @@ var init_event_handler = __esm({
27189
28045
  enfoque: "focus",
27190
28046
  desenfoque: "blur",
27191
28047
  carga: "load",
27192
- desplazamiento: "scroll"
28048
+ desplazamiento: "scroll",
28049
+ // V3 Batch 2 aliases: i18n dictionary verb forms the es tokenizer already
28050
+ // normalizes (probe-verified). Appended — localization canonicals unchanged.
28051
+ cambiar: "change",
28052
+ enfocar: "focus",
28053
+ desenfocar: "blur",
28054
+ cargar: "load",
28055
+ desplazar: "scroll",
28056
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28057
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28058
+ dobleclic: "dblclick",
28059
+ rat\u00F3nentrar: "mouseenter",
28060
+ rat\u00F3nsalir: "mouseleave",
28061
+ rat\u00F3nmover: "mousemove",
28062
+ teclapresar: "keypress",
28063
+ descargar: "unload",
28064
+ toqueempezar: "touchstart",
28065
+ toqueterminar: "touchend",
28066
+ toquemover: "touchmove",
28067
+ toquecancelar: "touchcancel"
27193
28068
  },
27194
28069
  // Turkish event names → English
27195
28070
  tr: {
@@ -27221,7 +28096,16 @@ var init_event_handler = __esm({
27221
28096
  // the `kaydır`/`kaydırma` scroll precedent) keeps the event token whole.
27222
28097
  boyutland\u0131rma: "resize",
27223
28098
  boyutland\u0131r: "resize",
27224
- kayd\u0131rma: "scroll"
28099
+ kayd\u0131rma: "scroll",
28100
+ // V3 Batch 2 aliases: i18n dictionary forms the tr tokenizer already
28101
+ // normalizes (probe-verified; farebas/farebırak are the deliberately fused
28102
+ // dict forms — the table's own fare_bas/fare_bırak `_` entries shatter).
28103
+ bulan\u0131k: "blur",
28104
+ farebas: "mousedown",
28105
+ fareb\u0131rak: "mouseup",
28106
+ kayd\u0131r: "scroll"
28107
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28108
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
27225
28109
  },
27226
28110
  // Portuguese event names → English
27227
28111
  pt: {
@@ -27248,7 +28132,19 @@ var init_event_handler = __esm({
27248
28132
  carregar: "load",
27249
28133
  carregamento: "load",
27250
28134
  rolagem: "scroll",
27251
- rolar: "scroll"
28135
+ rolar: "scroll",
28136
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28137
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28138
+ duploClique: "dblclick",
28139
+ mouseEntrar: "mouseenter",
28140
+ mouseSair: "mouseleave",
28141
+ mouseMover: "mousemove",
28142
+ teclaPressionar: "keypress",
28143
+ descarregar: "unload",
28144
+ toqueIn\u00EDcio: "touchstart",
28145
+ toqueFim: "touchend",
28146
+ toqueMover: "touchmove",
28147
+ toqueCancelar: "touchcancel"
27252
28148
  },
27253
28149
  // Chinese event names → English
27254
28150
  zh: {
@@ -27274,7 +28170,18 @@ var init_event_handler = __esm({
27274
28170
  \u6A21\u7CCA: "blur",
27275
28171
  \u52A0\u8F7D: "load",
27276
28172
  \u8F7D\u5165: "load",
27277
- \u6EDA\u52A8: "scroll"
28173
+ \u6EDA\u52A8: "scroll",
28174
+ // V3 Batch 2 alias: the i18n dictionary keydown form (captures keydown via
28175
+ // the registered 按键 prefix; probe-verified — kept over bare 按键 to avoid
28176
+ // colliding with the dict's keypress entry).
28177
+ \u6309\u952E\u6309\u4E0B: "keydown",
28178
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28179
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28180
+ \u9F20\u6807\u79FB\u52A8: "mousemove",
28181
+ \u5378\u8F7D: "unload",
28182
+ \u8C03\u6574\u5927\u5C0F: "resize",
28183
+ \u89E6\u6478\u5F00\u59CB: "touchstart",
28184
+ \u89E6\u6478\u79FB\u52A8: "touchmove"
27278
28185
  },
27279
28186
  // French event names → English
27280
28187
  fr: {
@@ -27299,7 +28206,22 @@ var init_event_handler = __esm({
27299
28206
  chargement: "load",
27300
28207
  charger: "load",
27301
28208
  d\u00E9filement: "scroll",
27302
- d\u00E9filer: "scroll"
28209
+ d\u00E9filer: "scroll",
28210
+ // V3 Batch 2 alias: i18n dictionary form the fr tokenizer already
28211
+ // normalizes (probe-verified).
28212
+ flou: "blur",
28213
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28214
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28215
+ doubleclic: "dblclick",
28216
+ sourisentrer: "mouseenter",
28217
+ sourissortir: "mouseleave",
28218
+ sourisbouger: "mousemove",
28219
+ touchepress\u00E9e: "keypress",
28220
+ d\u00E9charger: "unload",
28221
+ touchercommencer: "touchstart",
28222
+ toucherfin: "touchend",
28223
+ toucherbouger: "touchmove",
28224
+ toucherannuler: "touchcancel"
27303
28225
  },
27304
28226
  // German event names → English
27305
28227
  de: {
@@ -27323,7 +28245,26 @@ var init_event_handler = __esm({
27323
28245
  laden: "load",
27324
28246
  ladung: "load",
27325
28247
  scrollen: "scroll",
27326
- bl\u00E4ttern: "scroll"
28248
+ bl\u00E4ttern: "scroll",
28249
+ // V3 Batch 2 aliases: the de tokenizer's registered multi-word event forms
28250
+ // (probe-verified; the table's older `taste runter`/`taste hoch`/`maus
28251
+ // über`/`maus raus` entries are aspirational — they do not tokenize).
28252
+ "taste unten": "keydown",
28253
+ "taste oben": "keyup",
28254
+ "maus dr\xFCber": "mouseover",
28255
+ "maus weg": "mouseout",
28256
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28257
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28258
+ doppelklick: "dblclick",
28259
+ mauseintreten: "mouseenter",
28260
+ mausverlassen: "mouseleave",
28261
+ mausbewegen: "mousemove",
28262
+ tastedr\u00FCcken: "keypress",
28263
+ entladen: "unload",
28264
+ ber\u00FChrungstart: "touchstart",
28265
+ ber\u00FChrungend: "touchend",
28266
+ ber\u00FChrungbewegen: "touchmove",
28267
+ ber\u00FChrungabbrechen: "touchcancel"
27327
28268
  },
27328
28269
  // Indonesian event names → English
27329
28270
  id: {
@@ -27343,7 +28284,18 @@ var init_event_handler = __esm({
27343
28284
  muat: "load",
27344
28285
  memuat: "load",
27345
28286
  gulir: "scroll",
27346
- menggulir: "scroll"
28287
+ menggulir: "scroll",
28288
+ // V3 Batch 2 aliases: tekan_tombol captures keydown via the registered
28289
+ // `tekan`; arahkan/tinggalkan are the tokenizer's registered natives;
28290
+ // keyup is English passthrough (no parseable id native — `lepas` is
28291
+ // unregistered). All probe-verified.
28292
+ tekan_tombol: "keydown",
28293
+ keyup: "keyup",
28294
+ arahkan: "mouseover",
28295
+ tinggalkan: "mouseout",
28296
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28297
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28298
+ bongkar: "unload"
27347
28299
  },
27348
28300
  // Bengali event names → English
27349
28301
  bn: {
@@ -27356,6 +28308,8 @@ var init_event_handler = __esm({
27356
28308
  \u099D\u09BE\u09AA\u09B8\u09BE: "blur",
27357
28309
  \u09AB\u09CB\u0995\u09BE\u09B8: "focus",
27358
28310
  \u09AA\u09B0\u09BF\u09AC\u09B0\u09CD\u09A4\u09A8: "change"
28311
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28312
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
27359
28313
  },
27360
28314
  // Quechua event names → English (loanwords with native adaptations)
27361
28315
  qu: {
@@ -27366,8 +28320,14 @@ var init_event_handler = __esm({
27366
28320
  yaykuy: "input",
27367
28321
  tikray: "change",
27368
28322
  "t'ikray": "change",
28323
+ // Batch 3 aliases (appended so first-wins localization canonicals are
28324
+ // unchanged): the dict now renders kambiay/apaykachay — probe-verified to
28325
+ // capture the canonical event via the tokenizer keyword table, unlike
28326
+ // tikray (captures 'toggle') and kachay ('send' in one corpus slot).
28327
+ kambiay: "change",
27369
28328
  apachiy: "submit",
27370
28329
  kachay: "submit",
28330
+ apaykachay: "submit",
27371
28331
  "llave uray": "keydown",
27372
28332
  "llave hawa": "keyup",
27373
28333
  "q'away": "focus",
@@ -27380,6 +28340,8 @@ var init_event_handler = __esm({
27380
28340
  kunray: "scroll",
27381
28341
  muyuy: "scroll",
27382
28342
  hatun_kay: "resize"
28343
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28344
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
27383
28345
  },
27384
28346
  // Swahili event names → English
27385
28347
  sw: {
@@ -27401,7 +28363,31 @@ var init_event_handler = __esm({
27401
28363
  pakia: "load",
27402
28364
  kupakia: "load",
27403
28365
  sogeza: "scroll",
27404
- kusogeza: "scroll"
28366
+ kusogeza: "scroll",
28367
+ // V3 Batch 2 aliases: i18n dictionary forms the sw tokenizer already
28368
+ // normalizes (probe-verified; bonyeza is corpus-hot — 106 rows), plus the
28369
+ // tokenizer's registered `sogeza juu` for mouseover (the table's `panya
28370
+ // juu` is mouseup's dict form and maps there).
28371
+ bonyeza: "click",
28372
+ ingizo: "input",
28373
+ kitufe_shuka: "keydown",
28374
+ kitufe_juu: "keyup",
28375
+ panya_nje: "mouseout",
28376
+ wasilisha: "submit",
28377
+ "sogeza juu": "mouseover",
28378
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28379
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28380
+ shuka: "unload"
28381
+ },
28382
+ // Vietnamese event names → English. Minimal section: the dict renders
28383
+ // resize as the three-word đổi kích thước; the event slot captures only
28384
+ // đổi (tokenizer-normalized → change) and `kích thước` drops. The compound
28385
+ // key is matched by the parser's event-compound reclaim (Arc F,
28386
+ // offset-exact join of the captured event word + the dangling fragment).
28387
+ vi: {
28388
+ "\u0111\u1ED5i k\xEDch th\u01B0\u1EDBc": "resize"
28389
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28390
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
27405
28391
  }
27406
28392
  };
27407
28393
  Object.fromEntries(
@@ -27574,7 +28560,17 @@ function generateSOVPatientFirstEventHandlerPattern(commandSchema, profile, keyw
27574
28560
  const verbToken = keyword.alternatives ? { type: "literal", value: keyword.primary, alternatives: keyword.alternatives } : { type: "literal", value: keyword.primary };
27575
28561
  tokens.push(verbToken);
27576
28562
  tokens.push(...eventHandlerSourceGroup(commandSchema, profile.roleMarkers.source));
27577
- tokens.push(...eventHandlerDestinationGroup(commandSchema, profile.roleMarkers.destination));
28563
+ let trailingDestMarker = profile.roleMarkers.destination;
28564
+ if (commandSchema.action === "swap" && trailingDestMarker) {
28565
+ const withWord = commandSchema.roles.find((r) => r.role === "patient")?.markerOverride?.[profile.code];
28566
+ if (withWord && withWord !== trailingDestMarker.primary) {
28567
+ const existing = trailingDestMarker.alternatives ?? [];
28568
+ if (!existing.includes(withWord)) {
28569
+ trailingDestMarker = { ...trailingDestMarker, alternatives: [...existing, withWord] };
28570
+ }
28571
+ }
28572
+ }
28573
+ tokens.push(...eventHandlerDestinationGroup(commandSchema, trailingDestMarker));
27578
28574
  return {
27579
28575
  id: `${commandSchema.action}-event-${profile.code}-sov-patient-first`,
27580
28576
  language: profile.code,
@@ -28006,6 +29002,19 @@ function generateVSOVerbFirstEventHandlerPattern(commandSchema, profile, keyword
28006
29002
  tokens.push(markerToken);
28007
29003
  }
28008
29004
  tokens.push({ type: "role", role: "event", optional: false });
29005
+ if (commandSchema.action === "swap") {
29006
+ const withWord = commandSchema.roles.find((r) => r.role === "patient")?.markerOverride?.[profile.code];
29007
+ if (withWord) {
29008
+ tokens.push({
29009
+ type: "group",
29010
+ optional: true,
29011
+ tokens: [
29012
+ { type: "literal", value: withWord },
29013
+ { type: "role", role: "destination", optional: false }
29014
+ ]
29015
+ });
29016
+ }
29017
+ }
28009
29018
  return {
28010
29019
  id: `${commandSchema.action}-event-${profile.code}-vso-verb-first`,
28011
29020
  language: profile.code,
@@ -28284,12 +29293,16 @@ function generateVerbFirstPattern(schema, profile, config = defaultConfig) {
28284
29293
  const keyword = profile.keywords[schema.action];
28285
29294
  if (!keyword) return null;
28286
29295
  const verbToken = keyword.alternatives ? { type: "literal", value: keyword.primary, alternatives: keyword.alternatives } : { type: "literal", value: keyword.primary };
28287
- const roleTokens = requiredRoles.map((r) => ({
28288
- type: "role",
28289
- role: r.role,
28290
- optional: false,
28291
- expectedTypes: r.expectedTypes
28292
- }));
29296
+ const roleTokens = requiredRoles.flatMap((r) => {
29297
+ const prefix = r.valuePrefixLiteral?.[profile.code];
29298
+ const roleToken = {
29299
+ type: "role",
29300
+ role: r.role,
29301
+ optional: false,
29302
+ expectedTypes: r.expectedTypes
29303
+ };
29304
+ return prefix ? [{ type: "literal", value: prefix }, roleToken] : [roleToken];
29305
+ });
28293
29306
  return {
28294
29307
  id: `${schema.action}-${profile.code}-generated-verb-first`,
28295
29308
  language: profile.code,
@@ -28331,6 +29344,37 @@ function generatePatternVariants(schema, profile, config = defaultConfig) {
28331
29344
  patterns.push(verbFirst);
28332
29345
  }
28333
29346
  }
29347
+ for (const v of schema.rolePrefixLiteralVariants ?? []) {
29348
+ const literal = v.literal[profile.code];
29349
+ if (!literal) continue;
29350
+ const { rolePrefixLiteralVariants: _omitted, ...baseSchema } = schema;
29351
+ const cloneSchema2 = {
29352
+ ...baseSchema,
29353
+ roles: schema.roles.map(
29354
+ (r) => r.role === v.role ? { ...r, valuePrefixLiteral: { [profile.code]: literal } } : r
29355
+ )
29356
+ };
29357
+ const delta = v.priorityDelta ?? 5;
29358
+ const carrier = v.methodCarrier ? { [v.methodCarrier]: { value: literal } } : {};
29359
+ const main = generatePattern(cloneSchema2, profile, config);
29360
+ patterns.push({
29361
+ ...main,
29362
+ id: `${schema.action}-${profile.code}-generated-${v.idSuffix}`,
29363
+ priority: (config.basePriority ?? 100) + delta,
29364
+ extraction: { ...main.extraction, ...carrier }
29365
+ });
29366
+ if (config.generateVerbFirstVariants !== false) {
29367
+ const verbFirstUrl = generateVerbFirstPattern(cloneSchema2, profile, config);
29368
+ if (verbFirstUrl) {
29369
+ patterns.push({
29370
+ ...verbFirstUrl,
29371
+ id: `${schema.action}-${profile.code}-generated-verb-first-${v.idSuffix}`,
29372
+ priority: (config.basePriority ?? 100) - 20 + delta,
29373
+ extraction: { ...verbFirstUrl.extraction, ...carrier }
29374
+ });
29375
+ }
29376
+ }
29377
+ }
28334
29378
  return patterns;
28335
29379
  }
28336
29380
  function generatePatternsForLanguage(profile, config = defaultConfig) {
@@ -28554,25 +29598,31 @@ function buildRoleToken(roleSpec, profile) {
28554
29598
  const tokens = [];
28555
29599
  const overrideMarker = roleSpec.markerOverride?.[profile.code];
28556
29600
  const defaultMarker = profile.roleMarkers[roleSpec.role];
29601
+ const suppressMarker = roleSpec.renderOverride?.[profile.code] === "";
28557
29602
  const roleValueToken = {
28558
29603
  type: "role",
28559
29604
  role: roleSpec.role,
28560
29605
  optional: !roleSpec.required,
28561
29606
  expectedTypes: roleSpec.expectedTypes
28562
29607
  };
29608
+ const prefixLiteral = roleSpec.valuePrefixLiteral?.[profile.code];
29609
+ const pushPrefixed = () => {
29610
+ if (prefixLiteral) tokens.push({ type: "literal", value: prefixLiteral });
29611
+ tokens.push(roleValueToken);
29612
+ };
28563
29613
  if (overrideMarker !== void 0) {
28564
29614
  const markerWords = overrideMarker ? overrideMarker.split(/\s+/).filter(Boolean) : [];
28565
29615
  const position = defaultMarker?.position ?? "before";
28566
29616
  const optionalMarker = roleSpec.markerOptional?.[profile.code] === true;
28567
29617
  const pushWord = (word) => {
28568
- const literal = { type: "literal", value: word };
29618
+ const literal = suppressMarker ? { type: "literal", value: word, renderSuppress: true } : { type: "literal", value: word };
28569
29619
  tokens.push(optionalMarker ? { type: "group", optional: true, tokens: [literal] } : literal);
28570
29620
  };
28571
29621
  if (position === "before") {
28572
29622
  for (const word of markerWords) pushWord(word);
28573
- tokens.push(roleValueToken);
29623
+ pushPrefixed();
28574
29624
  } else {
28575
- tokens.push(roleValueToken);
29625
+ pushPrefixed();
28576
29626
  for (const word of markerWords) pushWord(word);
28577
29627
  }
28578
29628
  } else if (defaultMarker) {
@@ -28581,7 +29631,12 @@ function buildRoleToken(roleSpec, profile) {
28581
29631
  const alternatives = [
28582
29632
  .../* @__PURE__ */ new Set([...defaultMarker.alternatives ?? [], ...variantAlts])
28583
29633
  ].filter((a) => a !== defaultMarker.primary);
28584
- return alternatives.length ? { type: "literal", value: defaultMarker.primary, alternatives } : { type: "literal", value: defaultMarker.primary };
29634
+ return {
29635
+ type: "literal",
29636
+ value: defaultMarker.primary,
29637
+ ...alternatives.length ? { alternatives } : {},
29638
+ ...suppressMarker ? { renderSuppress: true } : {}
29639
+ };
28585
29640
  };
28586
29641
  const pushMarker = (marker) => {
28587
29642
  tokens.push(
@@ -28592,13 +29647,13 @@ function buildRoleToken(roleSpec, profile) {
28592
29647
  if (defaultMarker.primary) {
28593
29648
  pushMarker(asMarker());
28594
29649
  }
28595
- tokens.push(roleValueToken);
29650
+ pushPrefixed();
28596
29651
  } else {
28597
- tokens.push(roleValueToken);
29652
+ pushPrefixed();
28598
29653
  pushMarker(asMarker());
28599
29654
  }
28600
29655
  } else {
28601
- tokens.push(roleValueToken);
29656
+ pushPrefixed();
28602
29657
  }
28603
29658
  return tokens;
28604
29659
  }
@@ -28607,7 +29662,9 @@ function buildExtractionRules(schema, profile) {
28607
29662
  for (const roleSpec of schema.roles) {
28608
29663
  const overrideMarker = roleSpec.markerOverride?.[profile.code];
28609
29664
  const defaultMarker = profile.roleMarkers[roleSpec.role];
28610
- if (overrideMarker !== void 0) {
29665
+ if (roleSpec.valuePrefixLiteral?.[profile.code]) {
29666
+ rules[roleSpec.role] = { marker: roleSpec.valuePrefixLiteral[profile.code] };
29667
+ } else if (overrideMarker !== void 0) {
28611
29668
  rules[roleSpec.role] = overrideMarker ? { marker: overrideMarker } : {};
28612
29669
  } else if (defaultMarker && defaultMarker.primary) {
28613
29670
  const variantAlts = roleSpec.markerVariants?.[profile.code] ?? [];
@@ -28681,53 +29738,182 @@ var init_pattern_generator = __esm({
28681
29738
  }
28682
29739
  });
28683
29740
 
28684
- // src/patterns/toggle.ts
28685
- function getTogglePatternsBn() {
28686
- return [
28687
- // Full pattern: .active কে টগল করুন
28688
- {
28689
- id: "toggle-bn-full",
28690
- language: "bn",
28691
- command: "toggle",
28692
- priority: 100,
29741
+ // src/patterns/languages/en/fetch.ts
29742
+ var fetchWithResponseTypeEnglish, fetchWithOptionsAndResponseTypeEnglish, fetchWithOptionsEnglish, fetchSimpleEnglish, fetchPatternsEn;
29743
+ var init_fetch = __esm({
29744
+ "src/patterns/languages/en/fetch.ts"() {
29745
+ fetchWithResponseTypeEnglish = {
29746
+ id: "fetch-en-with-response-type",
29747
+ language: "en",
29748
+ command: "fetch",
29749
+ priority: 90,
29750
+ // Higher than simple pattern (80) to capture "as" modifier first
28693
29751
  template: {
28694
- format: "{patient} \u0995\u09C7 \u099F\u0997\u09B2 \u0995\u09B0\u09C1\u09A8",
29752
+ format: "fetch {source} as {responseType}",
28695
29753
  tokens: [
28696
- { type: "role", role: "patient" },
28697
- { type: "literal", value: "\u0995\u09C7" },
28698
- { type: "literal", value: "\u099F\u0997\u09B2", alternatives: ["\u09AA\u09B0\u09BF\u09AC\u09B0\u09CD\u09A4\u09A8"] },
28699
- { type: "literal", value: "\u0995\u09B0\u09C1\u09A8" }
29754
+ { type: "literal", value: "fetch" },
29755
+ { type: "role", role: "source", expectedTypes: ["literal", "expression"] },
29756
+ { type: "literal", value: "as" },
29757
+ // json/text/html are identifiers not keywords, so we need to accept 'expression' type
29758
+ { type: "role", role: "responseType", expectedTypes: ["literal", "expression"] }
28700
29759
  ]
28701
29760
  },
28702
29761
  extraction: {
28703
- patient: { position: 0 }
29762
+ source: { position: 1 },
29763
+ responseType: { marker: "as" }
28704
29764
  }
28705
- },
28706
- // Simple pattern: টগল .active
28707
- {
28708
- id: "toggle-bn-simple",
28709
- language: "bn",
28710
- command: "toggle",
28711
- priority: 90,
29765
+ };
29766
+ fetchWithOptionsAndResponseTypeEnglish = {
29767
+ id: "fetch-en-with-options-as",
29768
+ language: "en",
29769
+ command: "fetch",
29770
+ priority: 95,
28712
29771
  template: {
28713
- format: "\u099F\u0997\u09B2 {patient}",
29772
+ format: "fetch {source} with {style} as {responseType}",
28714
29773
  tokens: [
28715
- { type: "literal", value: "\u099F\u0997\u09B2", alternatives: ["\u09AA\u09B0\u09BF\u09AC\u09B0\u09CD\u09A4\u09A8"] },
28716
- { type: "role", role: "patient" }
29774
+ { type: "literal", value: "fetch" },
29775
+ { type: "role", role: "source", expectedTypes: ["literal", "expression"] },
29776
+ { type: "literal", value: "with", alternatives: ["by", "using"] },
29777
+ // expression-ONLY: routes `{ … }` to the object-literal fold, which keeps
29778
+ // the source text intact for the expression parser.
29779
+ { type: "role", role: "style", expectedTypes: ["expression"] },
29780
+ { type: "literal", value: "as" },
29781
+ { type: "role", role: "responseType", expectedTypes: ["literal", "expression"] }
28717
29782
  ]
28718
29783
  },
28719
29784
  extraction: {
28720
- patient: { position: 1 }
29785
+ source: { position: 1 },
29786
+ style: { marker: "with" },
29787
+ responseType: { marker: "as" }
28721
29788
  }
28722
- },
28723
- // With destination: #button এ .active কে টগল করুন
28724
- {
28725
- id: "toggle-bn-with-dest",
28726
- language: "bn",
28727
- command: "toggle",
28728
- priority: 95,
29789
+ };
29790
+ fetchWithOptionsEnglish = {
29791
+ id: "fetch-en-with-options",
29792
+ language: "en",
29793
+ command: "fetch",
29794
+ priority: 93,
29795
+ // Below the with+as pattern, above the response-type pattern (90)
28729
29796
  template: {
28730
- format: "{destination} \u098F {patient} \u0995\u09C7 \u099F\u0997\u09B2 \u0995\u09B0\u09C1\u09A8",
29797
+ format: "fetch {source} with {style}",
29798
+ tokens: [
29799
+ { type: "literal", value: "fetch" },
29800
+ { type: "role", role: "source", expectedTypes: ["literal", "expression"] },
29801
+ { type: "literal", value: "with", alternatives: ["by", "using"] },
29802
+ { type: "role", role: "style", expectedTypes: ["expression"] }
29803
+ ]
29804
+ },
29805
+ extraction: {
29806
+ source: { position: 1 },
29807
+ style: { marker: "with" }
29808
+ }
29809
+ };
29810
+ fetchSimpleEnglish = {
29811
+ id: "fetch-en-simple",
29812
+ language: "en",
29813
+ command: "fetch",
29814
+ priority: 80,
29815
+ // Lower than response type pattern (90) - fallback when "as" not present
29816
+ template: {
29817
+ format: "fetch {source}",
29818
+ tokens: [
29819
+ { type: "literal", value: "fetch" },
29820
+ { type: "role", role: "source" }
29821
+ ]
29822
+ },
29823
+ extraction: {
29824
+ source: { position: 1 }
29825
+ }
29826
+ };
29827
+ fetchPatternsEn = [
29828
+ fetchWithOptionsAndResponseTypeEnglish,
29829
+ fetchWithOptionsEnglish,
29830
+ fetchWithResponseTypeEnglish,
29831
+ fetchSimpleEnglish
29832
+ ];
29833
+ }
29834
+ });
29835
+
29836
+ // src/patterns/languages/en/pick.ts
29837
+ var pickVariantEnglish, pickPatternsEn;
29838
+ var init_pick = __esm({
29839
+ "src/patterns/languages/en/pick.ts"() {
29840
+ pickVariantEnglish = {
29841
+ id: "pick-en-variant",
29842
+ language: "en",
29843
+ command: "pick",
29844
+ priority: 110,
29845
+ template: {
29846
+ format: "pick {method} {patient} of {source}",
29847
+ tokens: [
29848
+ { type: "literal", value: "pick" },
29849
+ // Variant word: `characters`/`items`/`match` tokenize as identifiers
29850
+ // (expression), `first`/`last`/`random` as keywords.
29851
+ { type: "role", role: "method", expectedTypes: ["literal", "expression"] },
29852
+ // Range/count/index. The pick-range assembler folds `<a> to <b>
29853
+ // [inclusive|exclusive]` into one expression value here; a lone count
29854
+ // (`3`) is captured as a single literal.
29855
+ { type: "role", role: "patient", expectedTypes: ["literal", "expression"] },
29856
+ { type: "literal", value: "of", alternatives: ["from"] },
29857
+ { type: "role", role: "source", expectedTypes: ["selector", "reference", "expression"] }
29858
+ ]
29859
+ },
29860
+ extraction: {
29861
+ method: { position: 1 },
29862
+ patient: { position: 2 },
29863
+ source: { marker: "of", markerAlternatives: ["from"] }
29864
+ }
29865
+ };
29866
+ pickPatternsEn = [pickVariantEnglish];
29867
+ }
29868
+ });
29869
+
29870
+ // src/patterns/toggle.ts
29871
+ function getTogglePatternsBn() {
29872
+ return [
29873
+ // Full pattern: .active কে টগল করুন
29874
+ {
29875
+ id: "toggle-bn-full",
29876
+ language: "bn",
29877
+ command: "toggle",
29878
+ priority: 100,
29879
+ template: {
29880
+ format: "{patient} \u0995\u09C7 \u099F\u0997\u09B2 \u0995\u09B0\u09C1\u09A8",
29881
+ tokens: [
29882
+ { type: "role", role: "patient" },
29883
+ { type: "literal", value: "\u0995\u09C7" },
29884
+ { type: "literal", value: "\u099F\u0997\u09B2", alternatives: ["\u09AA\u09B0\u09BF\u09AC\u09B0\u09CD\u09A4\u09A8"] },
29885
+ { type: "literal", value: "\u0995\u09B0\u09C1\u09A8" }
29886
+ ]
29887
+ },
29888
+ extraction: {
29889
+ patient: { position: 0 }
29890
+ }
29891
+ },
29892
+ // Simple pattern: টগল .active
29893
+ {
29894
+ id: "toggle-bn-simple",
29895
+ language: "bn",
29896
+ command: "toggle",
29897
+ priority: 90,
29898
+ template: {
29899
+ format: "\u099F\u0997\u09B2 {patient}",
29900
+ tokens: [
29901
+ { type: "literal", value: "\u099F\u0997\u09B2", alternatives: ["\u09AA\u09B0\u09BF\u09AC\u09B0\u09CD\u09A4\u09A8"] },
29902
+ { type: "role", role: "patient" }
29903
+ ]
29904
+ },
29905
+ extraction: {
29906
+ patient: { position: 1 }
29907
+ }
29908
+ },
29909
+ // With destination: #button এ .active কে টগল করুন
29910
+ {
29911
+ id: "toggle-bn-with-dest",
29912
+ language: "bn",
29913
+ command: "toggle",
29914
+ priority: 95,
29915
+ template: {
29916
+ format: "{destination} \u098F {patient} \u0995\u09C7 \u099F\u0997\u09B2 \u0995\u09B0\u09C1\u09A8",
28731
29917
  tokens: [
28732
29918
  { type: "role", role: "destination" },
28733
29919
  { type: "literal", value: "\u098F", alternatives: ["\u09A4\u09C7"] },
@@ -29101,6 +30287,33 @@ function getTogglePatternsQu() {
29101
30287
  destination: { position: 0 },
29102
30288
  patient: { position: 2 }
29103
30289
  }
30290
+ },
30291
+ // Patient-first with trailing destination: .open ta qhipantin .panel man
30292
+ // t'ikray — the i18n full verb-final order (#636 qu canonicalOrder) puts
30293
+ // the destination AFTER the patient, but every dest-bearing variant above
30294
+ // is destination-first, so the shape fell to the verb-anchoring fallback,
30295
+ // which glued the positional run (destination:literal="qhipantin.panel"
30296
+ // vs en destination:expression="next .panel") — toggle-aria-expanded,
30297
+ // R1 deferred-tail qu tail.
30298
+ {
30299
+ id: "toggle-qu-patient-first-dest",
30300
+ language: "qu",
30301
+ command: "toggle",
30302
+ priority: 102,
30303
+ template: {
30304
+ format: "{patient} ta {destination} man t'ikray",
30305
+ tokens: [
30306
+ { type: "role", role: "patient" },
30307
+ { type: "literal", value: "ta" },
30308
+ { type: "role", role: "destination" },
30309
+ { type: "literal", value: "man", alternatives: ["pa"] },
30310
+ { type: "literal", value: "t'ikray", alternatives: ["tikray", "kutichiy"] }
30311
+ ]
30312
+ },
30313
+ extraction: {
30314
+ patient: { position: 0 },
30315
+ destination: { position: 2 }
30316
+ }
29104
30317
  }
29105
30318
  ];
29106
30319
  }
@@ -29468,11 +30681,15 @@ function repeatForInHead(language, spec) {
29468
30681
  // matches the verb's normalized form
29469
30682
  ];
29470
30683
  if (spec.forWords && spec.forWords.length > 0) {
29471
- tokens.push({
29472
- type: "group",
29473
- optional: true,
29474
- tokens: spec.forWords.map((w) => ({ type: "literal", value: w }))
29475
- });
30684
+ if (spec.requireForWords) {
30685
+ for (const w of spec.forWords) tokens.push({ type: "literal", value: w });
30686
+ } else {
30687
+ tokens.push({
30688
+ type: "group",
30689
+ optional: true,
30690
+ tokens: spec.forWords.map((w) => ({ type: "literal", value: w }))
30691
+ });
30692
+ }
29476
30693
  }
29477
30694
  tokens.push({ type: "role", role: "patient", expectedTypes: ["expression", "reference"] });
29478
30695
  for (const w of spec.inWords) tokens.push({ type: "literal", value: w });
@@ -29581,10 +30798,63 @@ function repeatUntilHeadSOV(language, spec) {
29581
30798
  }
29582
30799
  };
29583
30800
  }
30801
+ function repeatUntilHeadSOVVerbFinal(language, spec) {
30802
+ return {
30803
+ id: `repeat-${language}-until-head-verb-final`,
30804
+ language,
30805
+ command: "repeat",
30806
+ priority: 111,
30807
+ // above the post-verb variant so the correct shape wins
30808
+ template: {
30809
+ format: `${spec.untilWord} ${spec.eventWord} {event} ${spec.objMarker} {source} ${spec.fromWord} repeat`,
30810
+ tokens: [
30811
+ { type: "literal", value: spec.untilWord },
30812
+ { type: "literal", value: spec.eventWord },
30813
+ { type: "role", role: "event", expectedTypes: ["literal", "expression"] },
30814
+ { type: "literal", value: spec.objMarker },
30815
+ {
30816
+ type: "role",
30817
+ role: "source",
30818
+ expectedTypes: ["selector", "reference", "expression"]
30819
+ },
30820
+ { type: "literal", value: spec.fromWord },
30821
+ { type: "literal", value: "repeat" }
30822
+ ]
30823
+ },
30824
+ extraction: {
30825
+ loopType: { default: { type: "literal", value: "until-event" } }
30826
+ }
30827
+ };
30828
+ }
30829
+ function sovForBindingHead(language, spec) {
30830
+ return {
30831
+ id: `for-${language}-sov-basic`,
30832
+ language,
30833
+ command: "for",
30834
+ priority: 105,
30835
+ template: {
30836
+ format: `{patient} ${spec.inWords.join(" ")} {source} [${spec.objMarker}] ${spec.forVerb}`,
30837
+ tokens: [
30838
+ { type: "role", role: "patient", expectedTypes: ["expression", "reference"] },
30839
+ ...spec.inWords.map((w) => ({ type: "literal", value: w })),
30840
+ { type: "role", role: "source", expectedTypes: ["selector", "expression", "reference"] },
30841
+ {
30842
+ type: "group",
30843
+ optional: true,
30844
+ tokens: [{ type: "literal", value: spec.objMarker }]
30845
+ },
30846
+ { type: "literal", value: spec.forVerb }
30847
+ ]
30848
+ },
30849
+ extraction: {
30850
+ patient: { position: 0 }
30851
+ }
30852
+ };
30853
+ }
29584
30854
  function getRepeatPatternsForLanguage(language) {
29585
30855
  return BY_LANG.get(language) ?? [];
29586
30856
  }
29587
- var VERB_FIRST_REPEAT_TIMES, SOV_REPEAT_TIMES, FOR_IN_HEADS, WHILE_HEADS, VERB_FIRST_UNTIL_HEADS, repeatUntilHeadQuMidClause, SOV_UNTIL_HEADS, repeatUntilHeadQu, BY_LANG, addPattern;
30857
+ var VERB_FIRST_REPEAT_TIMES, SOV_REPEAT_TIMES, FOR_IN_HEADS, WHILE_HEADS, VERB_FIRST_UNTIL_HEADS, repeatUntilHeadQuMidClause, SOV_UNTIL_HEADS, repeatUntilHeadQu, SOV_FOR_BINDING_HEADS, BY_LANG, addPattern;
29588
30858
  var init_repeat = __esm({
29589
30859
  "src/patterns/repeat.ts"() {
29590
30860
  VERB_FIRST_REPEAT_TIMES = [
@@ -29599,7 +30869,7 @@ var init_repeat = __esm({
29599
30869
  ["ar", "\u0643\u0631\u0631", "times"],
29600
30870
  ["he", "\u05D7\u05D6\u05D5\u05E8", "times", "\u05D0\u05EA"],
29601
30871
  ["id", "ulangi", "times"],
29602
- ["ms", "ulang", "times"],
30872
+ ["ms", "ulang", "kali"],
29603
30873
  ["sw", "rudia", "times"],
29604
30874
  ["th", "\u0E17\u0E33\u0E0B\u0E49\u0E33", "\u0E04\u0E23\u0E31\u0E49\u0E07"],
29605
30875
  ["vi", "l\u1EB7p l\u1EA1i", "l\u1EA7n"],
@@ -29615,7 +30885,7 @@ var init_repeat = __esm({
29615
30885
  ["qu", "times", "ta"]
29616
30886
  ];
29617
30887
  FOR_IN_HEADS = [
29618
- ["en", { forWords: ["for"], inWords: ["in"] }],
30888
+ ["en", { forWords: ["for"], inWords: ["in"], requireForWords: true }],
29619
30889
  ["es", { forWords: ["para"], inWords: ["en"] }],
29620
30890
  ["pt", { forWords: ["para"], inWords: ["dentro"] }],
29621
30891
  ["fr", { forWords: ["pour"], inWords: ["en"] }],
@@ -29629,8 +30899,11 @@ var init_repeat = __esm({
29629
30899
  ["he", { forWords: ["\u05E2\u05D1\u05D5\u05E8", "\u05D0\u05EA"], inWords: ["in"] }],
29630
30900
  ["hi", { inWords: ["\u092E\u0947\u0902"] }],
29631
30901
  ["bn", { inWords: ["\u098F"] }],
29632
- ["ja", { inWords: ["\u306E", "\u4E2D"] }],
29633
- ["ko", { inWords: ["\uC548", "\uC5D0"] }],
30902
+ // ja/ko/qu containment words tokenize WHOLE (keyword→in entries added for
30903
+ // the focus-trap Family G operand run) — the old split forms (の+中, 안+에,
30904
+ // uku+pi) no longer appear in the stream.
30905
+ ["ja", { inWords: ["\u306E\u4E2D"] }],
30906
+ ["ko", { inWords: ["\uC548\uC5D0"] }],
29634
30907
  ["zh", { forWords: ["\u4E3A", "\u628A"], inWords: ["\u5728"] }],
29635
30908
  ["tr", { inWords: ["i\xE7inde"] }],
29636
30909
  ["id", { forWords: ["untuk"], inWords: ["dalam"] }],
@@ -29639,7 +30912,7 @@ var init_repeat = __esm({
29639
30912
  ["th", { forWords: ["\u0E2A\u0E33\u0E2B\u0E23\u0E31\u0E1A"], inWords: ["\u0E43\u0E19"] }],
29640
30913
  ["vi", { forWords: ["v\u1EDBi m\u1ED7i"], inWords: ["trong"] }],
29641
30914
  ["tl", { forWords: ["para_sa"], inWords: ["sa_loob"] }],
29642
- ["qu", { inWords: ["uku", "pi"] }]
30915
+ ["qu", { inWords: ["ukupi"] }]
29643
30916
  ];
29644
30917
  WHILE_HEADS = [
29645
30918
  ["en", { whileWord: "while" }],
@@ -29735,6 +31008,16 @@ var init_repeat = __esm({
29735
31008
  loopType: { default: { type: "literal", value: "until-event" } }
29736
31009
  }
29737
31010
  };
31011
+ SOV_FOR_BINDING_HEADS = [
31012
+ // ja/ko/qu in-words are single whole tokens now (keyword→in entries — see
31013
+ // the FOR_IN_HEADS note); the split forms are gone from the stream.
31014
+ ["ja", { inWords: ["\u306E\u4E2D"], objMarker: "\u3092", forVerb: "\u305F\u3081\u306B" }],
31015
+ ["ko", { inWords: ["\uC548\uC5D0"], objMarker: "\uB97C", forVerb: "\uAC01\uAC01" }],
31016
+ ["tr", { inWords: ["i\xE7inde"], objMarker: "i", forVerb: "i\xE7in" }],
31017
+ ["qu", { inWords: ["ukupi"], objMarker: "ta", forVerb: "sapankaq" }],
31018
+ ["bn", { inWords: ["\u098F"], objMarker: "\u0995\u09C7", forVerb: "\u099C\u09A8\u09CD\u09AF" }],
31019
+ ["hi", { inWords: ["\u092E\u0947\u0902"], objMarker: "\u0915\u094B", forVerb: "\u0939\u0947\u0924\u0941" }]
31020
+ ];
29738
31021
  BY_LANG = /* @__PURE__ */ new Map();
29739
31022
  addPattern = (lang, p) => {
29740
31023
  const list = BY_LANG.get(lang);
@@ -29750,6 +31033,9 @@ var init_repeat = __esm({
29750
31033
  for (const [lang, spec] of FOR_IN_HEADS) {
29751
31034
  addPattern(lang, repeatForInHead(lang, spec));
29752
31035
  }
31036
+ for (const [lang, spec] of SOV_FOR_BINDING_HEADS) {
31037
+ addPattern(lang, sovForBindingHead(lang, spec));
31038
+ }
29753
31039
  for (const [lang, spec] of WHILE_HEADS) {
29754
31040
  addPattern(lang, repeatWhileHead(lang, spec));
29755
31041
  }
@@ -29758,6 +31044,9 @@ var init_repeat = __esm({
29758
31044
  }
29759
31045
  for (const [lang, spec] of SOV_UNTIL_HEADS) {
29760
31046
  addPattern(lang, repeatUntilHeadSOV(lang, spec));
31047
+ if (lang === "tr") {
31048
+ addPattern(lang, repeatUntilHeadSOVVerbFinal(lang, spec));
31049
+ }
29761
31050
  }
29762
31051
  addPattern("qu", repeatUntilHeadQu);
29763
31052
  addPattern("qu", repeatUntilHeadQuMidClause);
@@ -29877,6 +31166,121 @@ function getWaitPatternsTl() {
29877
31166
  }
29878
31167
  ];
29879
31168
  }
31169
+ function verbFinalOrRunWait(id, language, verb, sourceMarker, orWord, parenArgCount, sourceMarkerAlternatives) {
31170
+ const parenGroup = () => ({
31171
+ type: "group",
31172
+ optional: true,
31173
+ tokens: [
31174
+ { type: "literal", value: "(" },
31175
+ ...Array.from({ length: parenArgCount }, (_, i) => [
31176
+ ...i > 0 ? [{ type: "literal", value: "," }] : [],
31177
+ {
31178
+ type: "role",
31179
+ role: "condition",
31180
+ expectedTypes: ["expression", "literal", "reference"]
31181
+ }
31182
+ ]).flat(),
31183
+ { type: "literal", value: ")" }
31184
+ ]
31185
+ });
31186
+ return {
31187
+ id,
31188
+ language,
31189
+ command: "wait",
31190
+ priority: 105,
31191
+ template: {
31192
+ format: `{source} ${sourceMarker} {duration} ${orWord} {patient} ${verb}`,
31193
+ tokens: [
31194
+ { type: "role", role: "source", expectedTypes: ["expression", "reference"] },
31195
+ {
31196
+ type: "literal",
31197
+ value: sourceMarker,
31198
+ ...sourceMarkerAlternatives ? { alternatives: sourceMarkerAlternatives } : {}
31199
+ },
31200
+ { type: "role", role: "duration", expectedTypes: ["expression", "literal"] },
31201
+ parenGroup(),
31202
+ { type: "literal", value: orWord },
31203
+ { type: "role", role: "patient", expectedTypes: ["expression", "literal"] },
31204
+ parenGroup(),
31205
+ { type: "literal", value: verb }
31206
+ ]
31207
+ },
31208
+ extraction: {
31209
+ source: { position: 0 },
31210
+ duration: { position: 2 }
31211
+ }
31212
+ };
31213
+ }
31214
+ function verbFirstOrRunWait(id, language, verb, orWord, forWord, sourceMarker, parenArgCount) {
31215
+ const parenGroup = () => ({
31216
+ type: "group",
31217
+ optional: true,
31218
+ tokens: [
31219
+ { type: "literal", value: "(" },
31220
+ ...Array.from({ length: parenArgCount }, (_, i) => [
31221
+ ...i > 0 ? [{ type: "literal", value: "," }] : [],
31222
+ {
31223
+ type: "role",
31224
+ role: "condition",
31225
+ expectedTypes: ["expression", "literal", "reference"]
31226
+ }
31227
+ ]).flat(),
31228
+ { type: "literal", value: ")" }
31229
+ ]
31230
+ });
31231
+ const forGroup = () => ({
31232
+ type: "group",
31233
+ optional: true,
31234
+ tokens: [{ type: "literal", value: forWord }]
31235
+ });
31236
+ return {
31237
+ id,
31238
+ language,
31239
+ command: "wait",
31240
+ priority: 105,
31241
+ template: {
31242
+ format: `${verb} {duration} ${orWord} [${forWord}] {patient} [${forWord}] {source} ${sourceMarker}`,
31243
+ tokens: [
31244
+ { type: "literal", value: verb },
31245
+ { type: "role", role: "duration", expectedTypes: ["expression", "literal"] },
31246
+ parenGroup(),
31247
+ { type: "literal", value: orWord },
31248
+ forGroup(),
31249
+ { type: "role", role: "patient", expectedTypes: ["expression", "literal"] },
31250
+ parenGroup(),
31251
+ forGroup(),
31252
+ { type: "role", role: "source", expectedTypes: ["expression", "reference"] },
31253
+ { type: "literal", value: sourceMarker }
31254
+ ]
31255
+ },
31256
+ extraction: {
31257
+ duration: { position: 1 },
31258
+ source: { position: 8 }
31259
+ }
31260
+ };
31261
+ }
31262
+ function getWaitPatternsBn() {
31263
+ return [
31264
+ verbFirstOrRunWait("wait-bn-or-run", "bn", "\u0985\u09AA\u09C7\u0995\u09CD\u09B7\u09BE", "\u0985\u09A5\u09AC\u09BE", "\u099C\u09A8\u09CD\u09AF", "\u09A5\u09C7\u0995\u09C7", 1),
31265
+ verbFirstOrRunWait("wait-bn-or-run-2arg", "bn", "\u0985\u09AA\u09C7\u0995\u09CD\u09B7\u09BE", "\u0985\u09A5\u09AC\u09BE", "\u099C\u09A8\u09CD\u09AF", "\u09A5\u09C7\u0995\u09C7", 2)
31266
+ ];
31267
+ }
31268
+ function getWaitPatternsTr() {
31269
+ return [
31270
+ verbFinalOrRunWait("wait-tr-or-run", "tr", "bekle", "den", "veya", 1, ["dan", "ten", "tan"]),
31271
+ verbFinalOrRunWait("wait-tr-or-run-2arg", "tr", "bekle", "den", "veya", 2, [
31272
+ "dan",
31273
+ "ten",
31274
+ "tan"
31275
+ ])
31276
+ ];
31277
+ }
31278
+ function getWaitPatternsQu() {
31279
+ return [
31280
+ verbFinalOrRunWait("wait-qu-or-run", "qu", "suyay", "manta", "utaq", 1),
31281
+ verbFinalOrRunWait("wait-qu-or-run-2arg", "qu", "suyay", "manta", "utaq", 2)
31282
+ ];
31283
+ }
29880
31284
  function getWaitPatternsForLanguage(language) {
29881
31285
  switch (language) {
29882
31286
  case "en":
@@ -29887,8 +31291,14 @@ function getWaitPatternsForLanguage(language) {
29887
31291
  return getWaitPatternsHe();
29888
31292
  case "ar":
29889
31293
  return getWaitPatternsAr();
31294
+ case "bn":
31295
+ return getWaitPatternsBn();
29890
31296
  case "tl":
29891
31297
  return getWaitPatternsTl();
31298
+ case "tr":
31299
+ return getWaitPatternsTr();
31300
+ case "qu":
31301
+ return getWaitPatternsQu();
29892
31302
  default:
29893
31303
  return [];
29894
31304
  }
@@ -29911,8 +31321,8 @@ function buildEnglishPatterns() {
29911
31321
  patterns.push(...getRepeatPatternsForLanguage("en"));
29912
31322
  patterns.push(...getWaitPatternsForLanguage("en"));
29913
31323
  patterns.push(
29914
- fetchWithResponseTypeEnglish,
29915
- fetchSimpleEnglish,
31324
+ ...fetchPatternsEn,
31325
+ ...pickPatternsEn,
29916
31326
  swapElementEnglish,
29917
31327
  swapSimpleEnglish,
29918
31328
  repeatUntilEventFromEnglish,
@@ -29930,51 +31340,18 @@ function buildEnglishPatterns() {
29930
31340
  patterns.push(...generatedPatterns);
29931
31341
  return patterns;
29932
31342
  }
29933
- var fetchWithResponseTypeEnglish, fetchSimpleEnglish, swapSimpleEnglish, swapElementEnglish, repeatUntilEventFromEnglish, repeatUntilEventEnglish, repeatTimesEnglish, repeatForeverEnglish, setPossessiveEnglish, forEnglish, ifEnglish, unlessEnglish, temporalInEnglish, temporalAfterEnglish;
31343
+ var swapSimpleEnglish, swapElementEnglish, repeatUntilEventFromEnglish, repeatUntilEventEnglish, repeatTimesEnglish, repeatForeverEnglish, setPossessiveEnglish, forEnglish, ifEnglish, unlessEnglish, temporalInEnglish, temporalAfterEnglish;
29934
31344
  var init_en = __esm({
29935
31345
  "src/patterns/en.ts"() {
29936
31346
  init_english();
29937
31347
  init_pattern_generator();
31348
+ init_fetch();
31349
+ init_pick();
29938
31350
  init_toggle();
29939
31351
  init_put();
29940
31352
  init_event_handler();
29941
31353
  init_repeat();
29942
31354
  init_wait();
29943
- fetchWithResponseTypeEnglish = {
29944
- id: "fetch-en-with-response-type",
29945
- language: "en",
29946
- command: "fetch",
29947
- priority: 90,
29948
- template: {
29949
- format: "fetch {source} as {responseType}",
29950
- tokens: [
29951
- { type: "literal", value: "fetch" },
29952
- { type: "role", role: "source", expectedTypes: ["literal", "expression"] },
29953
- { type: "literal", value: "as" },
29954
- { type: "role", role: "responseType", expectedTypes: ["literal", "expression"] }
29955
- ]
29956
- },
29957
- extraction: {
29958
- source: { position: 1 },
29959
- responseType: { marker: "as" }
29960
- }
29961
- };
29962
- fetchSimpleEnglish = {
29963
- id: "fetch-en-simple",
29964
- language: "en",
29965
- command: "fetch",
29966
- priority: 80,
29967
- template: {
29968
- format: "fetch {source}",
29969
- tokens: [
29970
- { type: "literal", value: "fetch" },
29971
- { type: "role", role: "source" }
29972
- ]
29973
- },
29974
- extraction: {
29975
- source: { position: 1 }
29976
- }
29977
- };
29978
31355
  swapSimpleEnglish = {
29979
31356
  id: "swap-en-handcrafted",
29980
31357
  language: "en",
@@ -30246,6 +31623,15 @@ init_chinese();
30246
31623
  // src/parser/pattern-matcher.ts
30247
31624
  init_command_schemas();
30248
31625
 
31626
+ // src/parser/utils/possessive-keywords.ts
31627
+ init_english();
31628
+
31629
+ // src/parser/utils/expression-lexicon.ts
31630
+ init_command_schemas();
31631
+ new Set(
31632
+ Object.keys(commandSchemas).map((a) => a.toLowerCase())
31633
+ );
31634
+
30249
31635
  // src/parser/pattern-matcher.ts
30250
31636
  init_registry();
30251
31637
  init_put();
@@ -30254,15 +31640,6 @@ init_put();
30254
31640
  new Set(
30255
31641
  Object.values(commandSchemas).filter((s) => s.bareKeyword === true).map((s) => s.action)
30256
31642
  );
30257
- /**
30258
- * Normalized command-action keywords (the schema registry's action names).
30259
- * Tokenizers normalize every language's command verbs to these forms, so the
30260
- * set is language-independent. Used to keep the positional source clause
30261
- * from consuming a following command's verb as a locative marker.
30262
- */
30263
- new Set(
30264
- Object.keys(commandSchemas).map((a) => a.toLowerCase())
30265
- );
30266
31643
 
30267
31644
  // src/tokenizers/index.ts
30268
31645
  init_registry();
@@ -30621,6 +31998,231 @@ init_wait();
30621
31998
  // src/patterns/builders.ts
30622
31999
  init_repeat();
30623
32000
 
32001
+ // src/patterns/languages/en/index.ts
32002
+ init_fetch();
32003
+
32004
+ // src/patterns/languages/en/swap.ts
32005
+ var swapSimpleEnglish2 = {
32006
+ id: "swap-en-handcrafted",
32007
+ language: "en",
32008
+ command: "swap",
32009
+ priority: 110,
32010
+ // Higher than generated patterns
32011
+ template: {
32012
+ format: "swap {method} {destination}",
32013
+ tokens: [
32014
+ { type: "literal", value: "swap" },
32015
+ { type: "role", role: "method" },
32016
+ { type: "role", role: "destination" }
32017
+ ]
32018
+ },
32019
+ extraction: {
32020
+ method: { position: 1 },
32021
+ destination: { position: 2 }
32022
+ }
32023
+ };
32024
+ var swapElementEnglish2 = {
32025
+ id: "swap-en-element",
32026
+ language: "en",
32027
+ command: "swap",
32028
+ priority: 120,
32029
+ template: {
32030
+ format: "swap {destination} with {patient}",
32031
+ tokens: [
32032
+ { type: "literal", value: "swap" },
32033
+ { type: "role", role: "destination" },
32034
+ { type: "literal", value: "with" },
32035
+ { type: "role", role: "patient" }
32036
+ ]
32037
+ },
32038
+ extraction: {}
32039
+ };
32040
+ var swapPatternsEn = [swapElementEnglish2, swapSimpleEnglish2];
32041
+
32042
+ // src/patterns/languages/en/repeat.ts
32043
+ var repeatUntilEventFromEnglish2 = {
32044
+ id: "repeat-en-until-event-from",
32045
+ language: "en",
32046
+ command: "repeat",
32047
+ priority: 120,
32048
+ // Highest priority - most specific pattern
32049
+ template: {
32050
+ format: "repeat until event {event} from {source}",
32051
+ tokens: [
32052
+ { type: "literal", value: "repeat" },
32053
+ { type: "literal", value: "until" },
32054
+ { type: "literal", value: "event" },
32055
+ { type: "role", role: "event", expectedTypes: ["literal", "expression"] },
32056
+ { type: "literal", value: "from" },
32057
+ { type: "role", role: "source", expectedTypes: ["selector", "reference", "expression"] }
32058
+ ]
32059
+ },
32060
+ extraction: {
32061
+ event: { marker: "event" },
32062
+ source: { marker: "from" },
32063
+ loopType: { default: { type: "literal", value: "until-event" } }
32064
+ }
32065
+ };
32066
+ var repeatUntilEventEnglish2 = {
32067
+ id: "repeat-en-until-event",
32068
+ language: "en",
32069
+ command: "repeat",
32070
+ priority: 110,
32071
+ // Lower than "from" variant, but higher than quantity-based repeat
32072
+ template: {
32073
+ format: "repeat until event {event}",
32074
+ tokens: [
32075
+ { type: "literal", value: "repeat" },
32076
+ { type: "literal", value: "until" },
32077
+ { type: "literal", value: "event" },
32078
+ { type: "role", role: "event", expectedTypes: ["literal", "expression"] }
32079
+ ]
32080
+ },
32081
+ extraction: {
32082
+ event: { marker: "event" },
32083
+ loopType: { default: { type: "literal", value: "until-event" } }
32084
+ }
32085
+ };
32086
+ var repeatPatternsEn = [
32087
+ repeatUntilEventFromEnglish2,
32088
+ repeatUntilEventEnglish2
32089
+ ];
32090
+
32091
+ // src/patterns/languages/en/set.ts
32092
+ var setPossessiveEnglish2 = {
32093
+ id: "set-en-possessive",
32094
+ language: "en",
32095
+ command: "set",
32096
+ priority: 100,
32097
+ // Higher than generated setSchema (80)
32098
+ template: {
32099
+ format: "set {destination} to {patient}",
32100
+ tokens: [
32101
+ { type: "literal", value: "set" },
32102
+ // Role token with property-path support for possessive syntax
32103
+ {
32104
+ type: "role",
32105
+ role: "destination",
32106
+ expectedTypes: ["property-path", "selector", "reference", "expression"]
32107
+ },
32108
+ { type: "literal", value: "to" },
32109
+ { type: "role", role: "patient", expectedTypes: ["literal", "expression", "reference"] }
32110
+ ]
32111
+ },
32112
+ extraction: {
32113
+ destination: { position: 1 },
32114
+ patient: { marker: "to" }
32115
+ }
32116
+ };
32117
+ var setPatternsEn = [setPossessiveEnglish2];
32118
+
32119
+ // src/patterns/languages/en/control-flow.ts
32120
+ var forEnglish2 = {
32121
+ id: "for-en-basic",
32122
+ language: "en",
32123
+ command: "for",
32124
+ priority: 100,
32125
+ template: {
32126
+ format: "for {patient} in {source}",
32127
+ tokens: [
32128
+ { type: "literal", value: "for" },
32129
+ { type: "role", role: "patient", expectedTypes: ["expression", "reference"] },
32130
+ // Loop variable
32131
+ { type: "literal", value: "in" },
32132
+ { type: "role", role: "source", expectedTypes: ["selector", "expression", "reference"] }
32133
+ // Collection
32134
+ ]
32135
+ },
32136
+ extraction: {
32137
+ patient: { position: 1 },
32138
+ source: { marker: "in" }
32139
+ // NOTE: no `loopType` default — see the rationale in patterns/en.ts
32140
+ // `forEnglish` (the `for` schema has no loopType role; a `loopType:literal="for"`
32141
+ // here only duplicates the action name and is the R1 outlier no translation
32142
+ // reproduces). R2-safe (forMapper reads only patient+source). Kept in sync.
32143
+ }
32144
+ };
32145
+ var ifEnglish2 = {
32146
+ id: "if-en-basic",
32147
+ language: "en",
32148
+ command: "if",
32149
+ priority: 100,
32150
+ template: {
32151
+ format: "if {condition}",
32152
+ tokens: [
32153
+ { type: "literal", value: "if" },
32154
+ { type: "role", role: "condition", expectedTypes: ["expression", "reference", "selector"] }
32155
+ ]
32156
+ },
32157
+ extraction: {
32158
+ condition: { position: 1 }
32159
+ }
32160
+ };
32161
+ var unlessEnglish2 = {
32162
+ id: "unless-en-basic",
32163
+ language: "en",
32164
+ command: "unless",
32165
+ priority: 100,
32166
+ template: {
32167
+ format: "unless {condition}",
32168
+ tokens: [
32169
+ { type: "literal", value: "unless" },
32170
+ { type: "role", role: "condition", expectedTypes: ["expression", "reference", "selector"] }
32171
+ ]
32172
+ },
32173
+ extraction: {
32174
+ condition: { position: 1 }
32175
+ }
32176
+ };
32177
+ var controlFlowPatternsEn = [forEnglish2, ifEnglish2, unlessEnglish2];
32178
+
32179
+ // src/patterns/languages/en/temporal.ts
32180
+ var temporalInEnglish2 = {
32181
+ id: "temporal-en-in",
32182
+ language: "en",
32183
+ command: "wait",
32184
+ priority: 95,
32185
+ // Lower than standard wait patterns
32186
+ template: {
32187
+ format: "in {duration}",
32188
+ tokens: [
32189
+ { type: "literal", value: "in" },
32190
+ { type: "role", role: "duration", expectedTypes: ["literal", "expression"] }
32191
+ ]
32192
+ },
32193
+ extraction: {
32194
+ duration: { position: 1 }
32195
+ }
32196
+ };
32197
+ var temporalAfterEnglish2 = {
32198
+ id: "temporal-en-after",
32199
+ language: "en",
32200
+ command: "wait",
32201
+ priority: 95,
32202
+ // Lower than standard wait patterns
32203
+ template: {
32204
+ format: "after {duration}",
32205
+ tokens: [
32206
+ { type: "literal", value: "after" },
32207
+ { type: "role", role: "duration", expectedTypes: ["literal", "expression"] }
32208
+ ]
32209
+ },
32210
+ extraction: {
32211
+ duration: { position: 1 }
32212
+ }
32213
+ };
32214
+ var temporalPatternsEn = [temporalInEnglish2, temporalAfterEnglish2];
32215
+
32216
+ // src/patterns/languages/en/index.ts
32217
+ [
32218
+ ...fetchPatternsEn,
32219
+ ...swapPatternsEn,
32220
+ ...repeatPatternsEn,
32221
+ ...setPatternsEn,
32222
+ ...controlFlowPatternsEn,
32223
+ ...temporalPatternsEn
32224
+ ];
32225
+
30624
32226
  // src/patterns/builders.ts
30625
32227
  init_pattern_generator();
30626
32228
  init_registry();
@@ -31025,6 +32627,81 @@ function inferRoles(name, args, modifiers, target) {
31025
32627
  }
31026
32628
  break;
31027
32629
  }
32630
+ case 'go': {
32631
+ const kw = (n) => {
32632
+ if (!n || typeof n !== 'object')
32633
+ return undefined;
32634
+ const v = n;
32635
+ if (v.type === 'identifier') {
32636
+ if (typeof v.name === 'string' && v.name !== '')
32637
+ return v.name;
32638
+ return typeof v.value === 'string' ? v.value : undefined;
32639
+ }
32640
+ if (v.type === 'literal' && typeof v.value === 'string')
32641
+ return v.value;
32642
+ return undefined;
32643
+ };
32644
+ const asNode = (x) => x && typeof x === 'object' && 'type' in x ? x : undefined;
32645
+ let destination;
32646
+ let method;
32647
+ const onMod = asNode(modifiers?.on);
32648
+ if (args.length === 0 && onMod) {
32649
+ destination = onMod;
32650
+ if (kw(asNode(modifiers?.method)) === 'url') {
32651
+ method = { type: 'literal', value: 'url' };
32652
+ }
32653
+ }
32654
+ else {
32655
+ const words = args.map(kw);
32656
+ const urlIdx = words.indexOf('url');
32657
+ if (urlIdx !== -1 && args[urlIdx + 1]) {
32658
+ destination = args[urlIdx + 1];
32659
+ method = { type: 'literal', value: 'url' };
32660
+ }
32661
+ else {
32662
+ const SKIP = new Set(['to', 'the']);
32663
+ const POSITION = new Set([
32664
+ 'top',
32665
+ 'middle',
32666
+ 'bottom',
32667
+ 'left',
32668
+ 'center',
32669
+ 'right',
32670
+ 'smoothly',
32671
+ 'instantly',
32672
+ 'in',
32673
+ 'new',
32674
+ 'window',
32675
+ ]);
32676
+ const headIdx = args.findIndex((_, i) => {
32677
+ const w = words[i];
32678
+ return w === undefined || !SKIP.has(w);
32679
+ });
32680
+ const headWord = headIdx !== -1 ? words[headIdx] : undefined;
32681
+ const ofIdx = words.indexOf('of');
32682
+ if (headWord === 'back' || headWord === 'forward') {
32683
+ destination = { type: 'identifier', value: headWord, name: headWord };
32684
+ }
32685
+ else if (ofIdx !== -1 && args[ofIdx + 1]) {
32686
+ destination = kw(args[ofIdx + 1]) === 'the' ? args[ofIdx + 2] : args[ofIdx + 1];
32687
+ }
32688
+ else if (headIdx !== -1 && !POSITION.has(headWord ?? '')) {
32689
+ destination = args[headIdx];
32690
+ }
32691
+ }
32692
+ }
32693
+ const destWord = kw(destination);
32694
+ if ((destWord === 'back' || destWord === 'forward') && destination?.type !== 'identifier') {
32695
+ destination = { type: 'identifier', value: destWord, name: destWord };
32696
+ }
32697
+ if (!destination && target)
32698
+ destination = target;
32699
+ if (destination)
32700
+ roles.destination = destination;
32701
+ if (method)
32702
+ roles.method = method;
32703
+ break;
32704
+ }
31028
32705
  default: {
31029
32706
  const schema = getSchema(name);
31030
32707
  if (!schema)