@hyperfixi/core 2.7.1 → 2.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/dist/api/hyperscript-api.d.ts +1 -0
  2. package/dist/ast-utils/index.js +1910 -233
  3. package/dist/ast-utils/index.mjs +1910 -233
  4. package/dist/behaviors/index.js +10 -1
  5. package/dist/behaviors/index.mjs +10 -1
  6. package/dist/bundle-generator/index.d.ts +1 -1
  7. package/dist/bundle-generator/index.js +77 -68
  8. package/dist/bundle-generator/index.mjs +76 -69
  9. package/dist/bundle-generator/template-capabilities.d.ts +2 -0
  10. package/dist/chunks/bridge-DHj-SYm2.js +2 -0
  11. package/dist/chunks/browser-modular-D1m0Eikh.js +2 -0
  12. package/dist/chunks/{index-i_j9Z-1e.js → index-CuPeasRm.js} +2 -2
  13. package/dist/commands/index.js +127 -6
  14. package/dist/commands/index.mjs +127 -6
  15. package/dist/compatibility/browser-modular.d.ts +2 -2
  16. package/dist/expressions/index.d.ts +1 -1
  17. package/dist/htmx/hcon.d.ts +9 -0
  18. package/dist/htmx/htmx-translator.d.ts +1 -0
  19. package/dist/hyperfixi-browser-classic-i18n.js +1 -1
  20. package/dist/hyperfixi-browser-minimal.js +1 -1
  21. package/dist/hyperfixi-browser-standard.js +1 -1
  22. package/dist/hyperfixi-browser.js +1 -1
  23. package/dist/hyperfixi-classic-i18n.js +1 -1
  24. package/dist/hyperfixi-hx-v4.js +1 -1
  25. package/dist/hyperfixi-hx.js +1 -1
  26. package/dist/hyperfixi-hybrid-complete.js +1 -1
  27. package/dist/hyperfixi-hybrid-hx.js +1 -1
  28. package/dist/hyperfixi-minimal.js +1 -1
  29. package/dist/hyperfixi-multilingual.js +1 -1
  30. package/dist/hyperfixi-standard.js +1 -1
  31. package/dist/hyperfixi.js +1 -1
  32. package/dist/hyperfixi.mjs +1 -1
  33. package/dist/index.js +4527 -536
  34. package/dist/index.min.js +1 -1
  35. package/dist/index.mjs +4527 -536
  36. package/dist/lib/dom-globals-shim.d.ts +2 -0
  37. package/dist/lokascript-browser-classic-i18n.js +1 -1
  38. package/dist/lokascript-browser-minimal.js +1 -1
  39. package/dist/lokascript-browser-standard.js +1 -1
  40. package/dist/lokascript-browser.js +1 -1
  41. package/dist/lokascript-hybrid-complete.js +1 -1
  42. package/dist/lokascript-hybrid-hx.js +1 -1
  43. package/dist/lokascript-multilingual.js +1 -1
  44. package/dist/lse/index.d.ts +7 -7
  45. package/dist/metadata.d.ts +1 -1
  46. package/dist/metadata.js +31 -14
  47. package/dist/metadata.mjs +31 -14
  48. package/dist/multilingual/index.js +8 -1
  49. package/dist/multilingual/index.mjs +8 -1
  50. package/dist/parser/command-parsers/animation-commands.d.ts +2 -2
  51. package/dist/parser/command-parsers/async-commands.d.ts +2 -2
  52. package/dist/parser/command-parsers/dom-commands.d.ts +5 -5
  53. package/dist/parser/command-parsers/navigation-commands.d.ts +4 -0
  54. package/dist/parser/command-parsers/utility-commands.d.ts +2 -1
  55. package/dist/parser/command-parsers/variable-commands.d.ts +2 -2
  56. package/dist/parser/full-parser.js +117 -5
  57. package/dist/parser/full-parser.mjs +117 -5
  58. package/dist/parser/semantic-integration.d.ts +1 -0
  59. package/dist/performance/integration.d.ts +1 -1
  60. package/dist/registry/index.js +117 -5
  61. package/dist/registry/index.mjs +117 -5
  62. package/package.json +13 -20
  63. package/dist/chunks/bridge-lZbOVRDD.js +0 -2
  64. package/dist/chunks/browser-modular-DegkWQ8d.js +0 -2
  65. package/dist/compatibility/browser-bundle-animation-generated.d.ts +0 -16
  66. package/dist/compatibility/browser-bundle-forms-generated.d.ts +0 -16
  67. package/dist/compatibility/browser-bundle-minimal-generated.d.ts +0 -16
@@ -4260,7 +4260,39 @@ var _BaseTokenizer = class _BaseTokenizer {
4260
4260
  pos++;
4261
4261
  }
4262
4262
  }
4263
- return new TokenStreamImpl(tokens, this.language);
4263
+ return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
4264
+ }
4265
+ /**
4266
+ * Fuse `name` + `:qualifier` into ONE identifier (`draggable:start`).
4267
+ *
4268
+ * `:name` is hyperscript's local-variable sigil, but a colon IMMEDIATELY
4269
+ * preceded by an identifier is a qualifier (custom event namespace), not a
4270
+ * sigil. The English tokenizer already merges these inside
4271
+ * EnglishKeywordExtractor; this post-pass gives the other 23 languages the
4272
+ * same stream. Strict position adjacency is the discriminator: whitespace
4273
+ * between the tokens (`trigger :start`) breaks `end === start`, so a spaced
4274
+ * local-variable reference survives untouched.
4275
+ *
4276
+ * Self-gating for non-hyperscript tokenizers (domain DSLs): their extractor
4277
+ * sets tokenize `:` as bare punctuation (length 1), which never matches
4278
+ * COLON_QUALIFIER, so this pass is a no-op for them.
4279
+ */
4280
+ mergeColonQualifiedNames(tokens) {
4281
+ const out = [];
4282
+ for (const tok of tokens) {
4283
+ const prev = out[out.length - 1];
4284
+ if (prev && _BaseTokenizer.ASCII_WORD.test(prev.value) && _BaseTokenizer.COLON_QUALIFIER.test(tok.value) && prev.position.end === tok.position.start) {
4285
+ const merged = prev.value + tok.value;
4286
+ out[out.length - 1] = createToken(
4287
+ merged,
4288
+ this.classifyToken(merged),
4289
+ createPosition(prev.position.start, tok.position.end)
4290
+ );
4291
+ continue;
4292
+ }
4293
+ out.push(tok);
4294
+ }
4295
+ return out;
4264
4296
  }
4265
4297
  /**
4266
4298
  * Classify an unknown character when no extractor matches.
@@ -4766,6 +4798,14 @@ var _BaseTokenizer = class _BaseTokenizer {
4766
4798
  return null;
4767
4799
  }
4768
4800
  };
4801
+ /**
4802
+ * ASCII word of the shape the English word-walker produces. Excludes `:`, so a
4803
+ * token that already carries a qualifier never merges again — `a:b:c` yields
4804
+ * `a:b` + `:c`, byte-matching the English extractor's single-segment merge.
4805
+ */
4806
+ _BaseTokenizer.ASCII_WORD = /^[A-Za-z_][A-Za-z0-9_]*$/;
4807
+ /** `:name` — only a variable-ref-style extractor ever emits this token shape. */
4808
+ _BaseTokenizer.COLON_QUALIFIER = /^:[A-Za-z_][A-Za-z0-9_]*$/;
4769
4809
  /**
4770
4810
  * Configuration for native language time units.
4771
4811
  * Maps patterns to their standard suffix (ms, s, m, h).
@@ -4989,8 +5029,11 @@ var init_arabic = __esm({
4989
5029
  result: "\u0627\u0644\u0646\u062A\u064A\u062C\u0629",
4990
5030
  event: "\u0627\u0644\u062D\u062F\u062B",
4991
5031
  target: "\u0627\u0644\u0647\u062F\u0641",
4992
- body: "\u062C\u0633\u0645"
5032
+ body: "\u062C\u0633\u0645",
4993
5033
  // matches the i18n dict's emitted body word (corpus-canonical, parser must recognize it)
5034
+ document: "\u0648\u062B\u064A\u0642\u0629",
5035
+ window: "\u0646\u0627\u0641\u0630\u0629",
5036
+ detail: "\u062A\u0641\u0627\u0635\u064A\u0644"
4994
5037
  },
4995
5038
  possessive: {
4996
5039
  marker: "",
@@ -5099,6 +5142,30 @@ var init_arabic = __esm({
5099
5142
  return: { primary: "\u0627\u0631\u062C\u0639", alternatives: ["\u0639\u064F\u062F"], normalized: "return" },
5100
5143
  then: { primary: "\u062B\u0645", alternatives: ["\u0628\u0639\u062F\u0647\u0627", "\u062B\u0645\u0651"], normalized: "then" },
5101
5144
  and: { primary: "\u0648\u0623\u064A\u0636\u0627\u064B", alternatives: ["\u0623\u064A\u0636\u0627\u064B"], normalized: "and" },
5145
+ // Comparison operator (`target matches .x`). Deferred by the Phase 2 `matches`
5146
+ // slice because ar's operand ALSO leaked (`references.target` carried الهدف while
5147
+ // the dict emits هدف), and registering the operator without its operand is worse
5148
+ // than neither: modal-close-backdrop ar passed R2 only BY ACCIDENT — the unparsed
5149
+ // condition was dropped, so `hide` ran unconditionally and coincidentally matched
5150
+ // the en DOM effect. `matches` alone would parse the condition into a real
5151
+ // comparison whose operand هدف evaluates to undefined, stopping `hide` and
5152
+ // flipping R2 pass→fail at tolerance 0. Landing WITH the هدف EXTRAS entry
5153
+ // (arabic.ts tokenizer) renders `target matches .modal-backdrop`, byte-identical
5154
+ // to en. Not an ActionType and has no command schema, so no pattern is generated.
5155
+ matches: { primary: "\u064A\u0637\u0627\u0628\u0642", normalized: "matches" },
5156
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
5157
+ // keyword the surface stays an identifier and leaks verbatim into the
5158
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
5159
+ // schema, so no pattern is generated from it.
5160
+ exists: { primary: "\u0645\u0648\u062C\u0648\u062F", normalized: "exists" },
5161
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
5162
+ // seam as `exists`: without the keyword the surface stays an identifier and
5163
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
5164
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
5165
+ // Uses the dict's NATURAL spaced phrase `لا يوجد`, matched by the base
5166
+ // tokenizer's multi-word keyword walk (longest-phrase at a word boundary) —
5167
+ // the same mechanism hi `मेل खाता` uses. Does not collide with `not: 'ليس'`.
5168
+ no: { primary: "\u0644\u0627 \u064A\u0648\u062C\u062F", normalized: "no" },
5102
5169
  // آخر is deliberately ABSENT: it is the positional `last` keyword
5103
5170
  // (آخر <button/> في .modal — see pattern-matcher's positional handling).
5104
5171
  // Listing it as an end-alternative made parseBodyWithClauses chop every
@@ -5302,6 +5369,11 @@ var init_bengali = __esm({
5302
5369
  return: { primary: "\u09AB\u09BF\u09B0\u09C1\u09A8", alternatives: ["\u09AB\u09C7\u09B0\u09A4 \u09A6\u09BF\u09A8"], normalized: "return" },
5303
5370
  then: { primary: "\u09A4\u09BE\u09B0\u09AA\u09B0", alternatives: ["\u09A4\u0996\u09A8"], normalized: "then" },
5304
5371
  and: { primary: "\u098F\u09AC\u0982", alternatives: [], normalized: "and" },
5372
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
5373
+ // surface stays an identifier and leaks verbatim into the condition's raw
5374
+ // expression, which the core expression parser reads as English. Neither an
5375
+ // ActionType nor a command schema, so no pattern is generated from it.
5376
+ is: { primary: "\u09B9\u09AF\u09BC", normalized: "is" },
5305
5377
  end: { primary: "\u09B6\u09C7\u09B7", alternatives: ["\u09B8\u09AE\u09BE\u09AA\u09CD\u09A4"], normalized: "end" },
5306
5378
  // Advanced
5307
5379
  js: { primary: "\u099C\u09C7\u098F\u09B8", alternatives: ["js"], normalized: "js" },
@@ -5399,7 +5471,10 @@ var init_german = __esm({
5399
5471
  result: "Ergebnis",
5400
5472
  event: "Ereignis",
5401
5473
  target: "Ziel",
5402
- body: "K\xF6rper"
5474
+ body: "K\xF6rper",
5475
+ document: "dokument",
5476
+ window: "fenster",
5477
+ detail: "detail"
5403
5478
  },
5404
5479
  possessive: {
5405
5480
  marker: "",
@@ -5494,6 +5569,22 @@ var init_german = __esm({
5494
5569
  // Predicate keywords (conditionals) — mirrors the Spanish profile, the only
5495
5570
  // language that previously parsed `is empty`-style predicates.
5496
5571
  is: { primary: "ist", normalized: "is" },
5572
+ // Comparison operator (`target matches .x`). Without this keyword the surface
5573
+ // stays an identifier and leaks verbatim into the condition's raw expression,
5574
+ // which the core expression parser reads as English (modal-close-backdrop /
5575
+ // focus-trap drop their then-branch). Not an ActionType and has no command
5576
+ // schema, so no pattern is generated from it.
5577
+ matches: { primary: "passt", normalized: "matches" },
5578
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
5579
+ // keyword the surface stays an identifier and leaks verbatim into the
5580
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
5581
+ // schema, so no pattern is generated from it.
5582
+ exists: { primary: "existiert", normalized: "exists" },
5583
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
5584
+ // seam as `exists`: without the keyword the surface stays an identifier and
5585
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
5586
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
5587
+ no: { primary: "kein", normalized: "no" },
5497
5588
  end: { primary: "ende", alternatives: ["fertig"], normalized: "end" },
5498
5589
  js: { primary: "js", alternatives: ["javascript"], normalized: "js" },
5499
5590
  async: { primary: "asynchron", normalized: "async" },
@@ -5588,7 +5679,10 @@ var init_english = __esm({
5588
5679
  result: "result",
5589
5680
  event: "event",
5590
5681
  target: "target",
5591
- body: "body"
5682
+ body: "body",
5683
+ document: "document",
5684
+ window: "window",
5685
+ detail: "detail"
5592
5686
  },
5593
5687
  possessive: {
5594
5688
  marker: "'s",
@@ -5739,7 +5833,10 @@ var init_spanish = __esm({
5739
5833
  event: "evento",
5740
5834
  target: "objetivo",
5741
5835
  // destino is a synonym
5742
- body: "cuerpo"
5836
+ body: "cuerpo",
5837
+ document: "documento",
5838
+ window: "ventana",
5839
+ detail: "detalle"
5743
5840
  },
5744
5841
  possessive: {
5745
5842
  marker: "de",
@@ -5762,7 +5859,10 @@ var init_spanish = __esm({
5762
5859
  }
5763
5860
  },
5764
5861
  roleMarkers: {
5765
- destination: { primary: "en", alternatives: ["sobre", "a"], position: "before" },
5862
+ // `hacia` is the i18n grammar's optional destination render form ("towards");
5863
+ // without it here a rendered/user `hacia` clause silently dropped the
5864
+ // destination (add → default `me`, put → null parse). Vocab Batch 1 (V2+V4).
5865
+ destination: { primary: "en", alternatives: ["sobre", "a", "hacia"], position: "before" },
5766
5866
  source: { primary: "de", alternatives: ["desde"], position: "before" },
5767
5867
  patient: { primary: "", position: "before" },
5768
5868
  style: { primary: "con", position: "before" }
@@ -5872,6 +5972,19 @@ var init_spanish = __esm({
5872
5972
  is: { primary: "es", normalized: "is" },
5873
5973
  exists: { primary: "existe", normalized: "exists" },
5874
5974
  empty: { primary: "vac\xEDo", alternatives: ["vacio"], normalized: "empty" },
5975
+ // Comparison operator (`target matches .x`). Without this keyword the surface
5976
+ // stays an identifier and leaks verbatim into the condition's raw expression,
5977
+ // which the core expression parser reads as English (modal-close-backdrop /
5978
+ // focus-trap drop their then-branch). Not an ActionType and has no command
5979
+ // schema, so no pattern is generated from it.
5980
+ matches: { primary: "coincide", normalized: "matches" },
5981
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
5982
+ // seam as `exists`: without the keyword the surface stays an identifier and
5983
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
5984
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
5985
+ // Does NOT collide with `not: { primary: 'no' }`: the keyword map is keyed by
5986
+ // SURFACE, so this registers `ningún` and leaves the `no` surface untouched.
5987
+ no: { primary: "ning\xFAn", normalized: "no" },
5875
5988
  end: { primary: "fin", alternatives: ["final", "terminar"], normalized: "end" },
5876
5989
  // Advanced
5877
5990
  js: { primary: "js", normalized: "js" },
@@ -5975,7 +6088,10 @@ var init_french = __esm({
5975
6088
  result: "r\xE9sultat",
5976
6089
  event: "\xE9v\xE9nement",
5977
6090
  target: "cible",
5978
- body: "corps"
6091
+ body: "corps",
6092
+ document: "document",
6093
+ window: "fen\xEAtre",
6094
+ detail: "d\xE9tail"
5979
6095
  },
5980
6096
  possessive: {
5981
6097
  marker: "de",
@@ -6070,6 +6186,27 @@ var init_french = __esm({
6070
6186
  return: { primary: "retourner", alternatives: ["renvoyer"], normalized: "return" },
6071
6187
  then: { primary: "puis", alternatives: ["ensuite", "alors"], normalized: "then" },
6072
6188
  and: { primary: "et", alternatives: ["aussi", "\xE9galement"], normalized: "and" },
6189
+ // Comparison operator (`target matches .x`). Without this keyword the surface
6190
+ // stays an identifier and leaks verbatim into the condition's raw expression,
6191
+ // which the core expression parser reads as English (modal-close-backdrop /
6192
+ // focus-trap drop their then-branch). Not an ActionType and has no command
6193
+ // schema, so no pattern is generated from it.
6194
+ matches: { primary: "correspond", normalized: "matches" },
6195
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
6196
+ // keyword the surface stays an identifier and leaks verbatim into the
6197
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
6198
+ // schema, so no pattern is generated from it.
6199
+ exists: { primary: "existe", normalized: "exists" },
6200
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
6201
+ // surface stays an identifier and leaks verbatim into the condition's raw
6202
+ // expression, which the core expression parser reads as English. Neither an
6203
+ // ActionType nor a command schema, so no pattern is generated from it.
6204
+ is: { primary: "est", normalized: "is" },
6205
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
6206
+ // seam as `exists`: without the keyword the surface stays an identifier and
6207
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
6208
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
6209
+ no: { primary: "aucun", normalized: "no" },
6073
6210
  end: { primary: "fin", alternatives: ["terminer", "finir"], normalized: "end" },
6074
6211
  js: { primary: "js", normalized: "js" },
6075
6212
  async: { primary: "asynchrone", normalized: "async" },
@@ -6375,7 +6512,10 @@ var init_hindi = __esm({
6375
6512
  result: "\u092A\u0930\u093F\u0923\u093E\u092E",
6376
6513
  event: "\u0918\u091F\u0928\u093E",
6377
6514
  target: "\u0932\u0915\u094D\u0937\u094D\u092F",
6378
- body: "\u092C\u0949\u0921\u0940"
6515
+ body: "\u092C\u0949\u0921\u0940",
6516
+ document: "\u0926\u0938\u094D\u0924\u093E\u0935\u0947\u091C\u093C",
6517
+ window: "\u0935\u093F\u0902\u0921\u094B",
6518
+ detail: "\u0935\u093F\u0935\u0930\u0923"
6379
6519
  },
6380
6520
  possessive: {
6381
6521
  marker: "\u0915\u093E",
@@ -6525,6 +6665,11 @@ var init_hindi = __esm({
6525
6665
  // parser. (History: `मेल_खाता` underscore-split to मेल/_/खाता; the concatenated
6526
6666
  // `मेलखाता` parsed but isn't how Hindi is written.)
6527
6667
  matches: { primary: "\u092E\u0947\u0932 \u0916\u093E\u0924\u093E", normalized: "matches" },
6668
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
6669
+ // keyword the surface stays an identifier and leaks verbatim into the
6670
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
6671
+ // schema, so no pattern is generated from it.
6672
+ exists: { primary: "\u092E\u094C\u091C\u0942\u0926", normalized: "exists" },
6528
6673
  end: { primary: "\u0938\u092E\u093E\u092A\u094D\u0924", alternatives: ["\u0905\u0902\u0924"], normalized: "end" },
6529
6674
  // Advanced
6530
6675
  js: { primary: "\u091C\u0947\u090F\u0938", alternatives: ["js"], normalized: "js" },
@@ -6624,8 +6769,11 @@ var init_indonesian = __esm({
6624
6769
  result: "hasil",
6625
6770
  event: "peristiwa",
6626
6771
  target: "target",
6627
- body: "badan"
6772
+ body: "badan",
6628
6773
  // matches the i18n dict's emitted body word (corpus-canonical; tubuh = anatomical body)
6774
+ document: "dokumen",
6775
+ window: "jendela",
6776
+ detail: "detail"
6629
6777
  },
6630
6778
  possessive: {
6631
6779
  marker: "",
@@ -6746,6 +6894,12 @@ var init_indonesian = __esm({
6746
6894
  return: { primary: "kembalikan", alternatives: ["kembali"], normalized: "return" },
6747
6895
  then: { primary: "lalu", alternatives: ["kemudian", "setelah itu"], normalized: "then" },
6748
6896
  and: { primary: "dan", alternatives: ["juga", "serta"], normalized: "and" },
6897
+ // Comparison operator (`target matches .x`). Without this keyword the surface
6898
+ // stays an identifier and leaks verbatim into the condition's raw expression,
6899
+ // which the core expression parser reads as English (modal-close-backdrop /
6900
+ // focus-trap drop their then-branch). Not an ActionType and has no command
6901
+ // schema, so no pattern is generated from it.
6902
+ matches: { primary: "cocok", normalized: "matches" },
6749
6903
  end: { primary: "selesai", alternatives: ["akhir", "tamat"], normalized: "end" },
6750
6904
  js: { primary: "js", alternatives: ["javascript"], normalized: "js" },
6751
6905
  async: { primary: "asinkron", normalized: "async" },
@@ -6852,7 +7006,10 @@ var init_italian = __esm({
6852
7006
  result: "risultato",
6853
7007
  event: "evento",
6854
7008
  target: "obiettivo",
6855
- body: "corpo"
7009
+ body: "corpo",
7010
+ document: "documento",
7011
+ window: "finestra",
7012
+ detail: "dettaglio"
6856
7013
  },
6857
7014
  possessive: {
6858
7015
  marker: "di",
@@ -6959,6 +7116,17 @@ var init_italian = __esm({
6959
7116
  return: { primary: "ritornare", normalized: "return" },
6960
7117
  then: { primary: "allora", alternatives: ["poi", "quindi"], normalized: "then" },
6961
7118
  and: { primary: "e", alternatives: ["anche"], normalized: "and" },
7119
+ // Comparison operator (`target matches .x`). Without this keyword the surface
7120
+ // stays an identifier and leaks verbatim into the condition's raw expression,
7121
+ // which the core expression parser reads as English (modal-close-backdrop /
7122
+ // focus-trap drop their then-branch). Not an ActionType and has no command
7123
+ // schema, so no pattern is generated from it.
7124
+ matches: { primary: "corrisponde", normalized: "matches" },
7125
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
7126
+ // seam as `exists`: without the keyword the surface stays an identifier and
7127
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
7128
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
7129
+ no: { primary: "nessun", normalized: "no" },
6962
7130
  end: { primary: "fine", normalized: "end" },
6963
7131
  // Advanced
6964
7132
  js: { primary: "js", normalized: "js" },
@@ -7072,7 +7240,10 @@ var init_japanese = __esm({
7072
7240
  result: "\u7D50\u679C",
7073
7241
  event: "\u30A4\u30D9\u30F3\u30C8",
7074
7242
  target: "\u30BF\u30FC\u30B2\u30C3\u30C8",
7075
- body: "\u30DC\u30C7\u30A3"
7243
+ body: "\u30DC\u30C7\u30A3",
7244
+ document: "\u30C9\u30AD\u30E5\u30E1\u30F3\u30C8",
7245
+ window: "\u30A6\u30A3\u30F3\u30C9\u30A6",
7246
+ detail: "\u8A73\u7D30"
7076
7247
  },
7077
7248
  possessive: {
7078
7249
  marker: "\u306E",
@@ -7150,6 +7321,10 @@ var init_japanese = __esm({
7150
7321
  focus: { primary: "\u30D5\u30A9\u30FC\u30AB\u30B9", alternatives: ["\u96C6\u4E2D"], normalized: "focus" },
7151
7322
  blur: { primary: "\u307C\u304B\u3057", alternatives: ["\u30D5\u30A9\u30FC\u30AB\u30B9\u89E3\u9664", "\u30D6\u30E9\u30FC"], normalized: "blur" },
7152
7323
  // Phase 1 (v0.9.90): DOM / form state / debug
7324
+ // Batch 3: do NOT add bare 空 here — probed: registering it as an empty
7325
+ // keyword injects a phantom `empty` command into the corpus-hot `is empty`
7326
+ // expression rows (である 空), an R0-precision regression. The empty-COMMAND
7327
+ // render gap (dict renders 空, parses null) is waived instead.
7153
7328
  empty: { primary: "\u7A7A\u306B", alternatives: ["\u7A7A\u306B\u3059\u308B"], normalized: "empty" },
7154
7329
  open: { primary: "\u958B\u304F", alternatives: ["\u30AA\u30FC\u30D7\u30F3"], normalized: "open" },
7155
7330
  close: { primary: "\u9589\u3058\u308B", alternatives: ["\u30AF\u30ED\u30FC\u30BA"], normalized: "close" },
@@ -7193,6 +7368,32 @@ var init_japanese = __esm({
7193
7368
  return: { primary: "\u623B\u308B", alternatives: ["\u8FD4\u3059", "\u30EA\u30BF\u30FC\u30F3"], normalized: "return" },
7194
7369
  then: { primary: "\u305D\u308C\u304B\u3089", alternatives: ["\u6B21\u306B", "\u306A\u3089\u3070", "\u306A\u3089"], normalized: "then" },
7195
7370
  and: { primary: "\u307E\u305F", alternatives: ["\u3068", "\u305D\u3057\u3066"], normalized: "and" },
7371
+ // Comparison operator (`target matches .x`). Deferred by the Phase 2 `matches`
7372
+ // slice because ja's operand ALSO leaked (`references.target` carried ターゲット
7373
+ // while the dict emits 対象), and registering the operator without its operand is
7374
+ // worse than neither: modal-close-backdrop ja passed R2 only BY ACCIDENT — the
7375
+ // unparsed condition was dropped, so `hide` ran unconditionally and coincidentally
7376
+ // matched the en DOM effect. `matches` alone would parse the condition into a real
7377
+ // comparison whose operand 対象 evaluates to undefined, stopping `hide` and
7378
+ // flipping R2 pass→fail at tolerance 0. Landing WITH the 対象 EXTRAS entry
7379
+ // (japanese.ts tokenizer) renders `target matches .modal-backdrop`, byte-identical
7380
+ // to en. Not an ActionType and has no command schema, so no pattern is generated.
7381
+ matches: { primary: "\u4E00\u81F4\u3059\u308B", normalized: "matches" },
7382
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
7383
+ // keyword the surface stays an identifier and leaks verbatim into the
7384
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
7385
+ // schema, so no pattern is generated from it.
7386
+ exists: { primary: "\u5B58\u5728\u3059\u308B", normalized: "exists" },
7387
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
7388
+ // surface stays an identifier and leaks verbatim into the condition's raw
7389
+ // expression, which the core expression parser reads as English. Neither an
7390
+ // ActionType nor a command schema, so no pattern is generated from it.
7391
+ is: { primary: "\u3067\u3042\u308B", normalized: "is" },
7392
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
7393
+ // seam as `exists`: without the keyword the surface stays an identifier and
7394
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
7395
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
7396
+ no: { primary: "\u306A\u3044", normalized: "no" },
7196
7397
  // 終了 removed: it is the i18n dict's `exit` emission (ja.ts), so listing it
7197
7398
  // as an `end` alternative made an `exit` inside `if … exit … end` read as the
7198
7399
  // block terminator and collapse the handler body (behavior-sortable). 終わり is
@@ -7295,8 +7496,11 @@ var init_korean = __esm({
7295
7496
  result: "\uACB0\uACFC",
7296
7497
  event: "\uC774\uBCA4\uD2B8",
7297
7498
  target: "\uB300\uC0C1",
7298
- body: "\uBC14\uB514"
7499
+ body: "\uBC14\uB514",
7299
7500
  // matches the i18n dict's emitted body word (본문 = "main text", wrong for the DOM body element)
7501
+ document: "\uBB38\uC11C",
7502
+ window: "\uCC3D",
7503
+ detail: "\uC138\uBD80"
7300
7504
  },
7301
7505
  possessive: {
7302
7506
  marker: "\uC758",
@@ -7369,7 +7573,9 @@ var init_korean = __esm({
7369
7573
  focus: { primary: "\uD3EC\uCEE4\uC2A4", normalized: "focus" },
7370
7574
  blur: { primary: "\uBE14\uB7EC", normalized: "blur" },
7371
7575
  // Phase 1 (v0.9.90): DOM / form state / debug
7372
- empty: { primary: "\uBE44\uC6B0\uAE30", normalized: "empty" },
7576
+ // Batch 3: 비어있는 added — the i18n dict renders the empty COMMAND with its
7577
+ // `is empty` adjective (category-shadowed), which parsed null.
7578
+ empty: { primary: "\uBE44\uC6B0\uAE30", alternatives: ["\uBE44\uC5B4\uC788\uB294"], normalized: "empty" },
7373
7579
  open: { primary: "\uC5F4\uAE30", normalized: "open" },
7374
7580
  close: { primary: "\uB2EB\uAE30", normalized: "close" },
7375
7581
  select: { primary: "\uACE0\uB974\uAE30", normalized: "select" },
@@ -7426,6 +7632,16 @@ var init_korean = __esm({
7426
7632
  // matches .x`. Without this keyword `일치` stays an identifier and the
7427
7633
  // condition is unevaluable (modal-close-backdrop drops its then-branch).
7428
7634
  matches: { primary: "\uC77C\uCE58", normalized: "matches" },
7635
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
7636
+ // keyword the surface stays an identifier and leaks verbatim into the
7637
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
7638
+ // schema, so no pattern is generated from it.
7639
+ exists: { primary: "\uC874\uC7AC", normalized: "exists" },
7640
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
7641
+ // seam as `exists`: without the keyword the surface stays an identifier and
7642
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
7643
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
7644
+ no: { primary: "\uC5C6\uC74C", normalized: "no" },
7429
7645
  end: { primary: "\uB05D", alternatives: ["\uB9C8\uCE68"], normalized: "end" },
7430
7646
  // Advanced
7431
7647
  js: { primary: "JS\uC2E4\uD589", alternatives: ["js"], normalized: "js" },
@@ -7517,7 +7733,10 @@ var init_ms = __esm({
7517
7733
  result: "hasil",
7518
7734
  event: "peristiwa",
7519
7735
  target: "sasaran",
7520
- body: "badan"
7736
+ body: "badan",
7737
+ document: "dokumen",
7738
+ window: "tetingkap",
7739
+ detail: "perincian"
7521
7740
  },
7522
7741
  possessive: {
7523
7742
  marker: "",
@@ -7640,6 +7859,27 @@ var init_ms = __esm({
7640
7859
  return: { primary: "pulang", alternatives: ["kembali"], normalized: "return" },
7641
7860
  then: { primary: "kemudian", alternatives: ["lepas_itu"], normalized: "then" },
7642
7861
  and: { primary: "dan", normalized: "and" },
7862
+ // Comparison operator (`target matches .x`). Without this keyword the surface
7863
+ // stays an identifier and leaks verbatim into the condition's raw expression,
7864
+ // which the core expression parser reads as English (modal-close-backdrop /
7865
+ // focus-trap drop their then-branch). Not an ActionType and has no command
7866
+ // schema, so no pattern is generated from it.
7867
+ matches: { primary: "sepadan", normalized: "matches" },
7868
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
7869
+ // keyword the surface stays an identifier and leaks verbatim into the
7870
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
7871
+ // schema, so no pattern is generated from it.
7872
+ exists: { primary: "wujud", normalized: "exists" },
7873
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
7874
+ // surface stays an identifier and leaks verbatim into the condition's raw
7875
+ // expression, which the core expression parser reads as English. Neither an
7876
+ // ActionType nor a command schema, so no pattern is generated from it.
7877
+ is: { primary: "adalah", normalized: "is" },
7878
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
7879
+ // seam as `exists`: without the keyword the surface stays an identifier and
7880
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
7881
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
7882
+ no: { primary: "tiada", normalized: "no" },
7643
7883
  end: { primary: "tamat", alternatives: ["habis"], normalized: "end" },
7644
7884
  // Advanced
7645
7885
  js: { primary: "js", normalized: "js" },
@@ -7726,7 +7966,10 @@ var init_polish = __esm({
7726
7966
  result: "wynik",
7727
7967
  event: "zdarzenie",
7728
7968
  target: "cel",
7729
- body: "body"
7969
+ body: "body",
7970
+ document: "dokument",
7971
+ window: "okno",
7972
+ detail: "szczeg\xF3\u0142"
7730
7973
  },
7731
7974
  possessive: {
7732
7975
  marker: "",
@@ -7959,6 +8202,17 @@ var init_polish = __esm({
7959
8202
  normalized: "then"
7960
8203
  },
7961
8204
  and: { primary: "i", alternatives: ["oraz"], normalized: "and" },
8205
+ // Comparison operator (`target matches .x`). Without this keyword the surface
8206
+ // stays an identifier and leaks verbatim into the condition's raw expression,
8207
+ // which the core expression parser reads as English (modal-close-backdrop /
8208
+ // focus-trap drop their then-branch). Not an ActionType and has no command
8209
+ // schema, so no pattern is generated from it.
8210
+ matches: { primary: "pasuje", normalized: "matches" },
8211
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
8212
+ // seam as `exists`: without the keyword the surface stays an identifier and
8213
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
8214
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
8215
+ no: { primary: "brak", normalized: "no" },
7962
8216
  end: { primary: "koniec", normalized: "end" },
7963
8217
  // Advanced
7964
8218
  js: { primary: "js", normalized: "js" },
@@ -8063,7 +8317,10 @@ var init_portuguese = __esm({
8063
8317
  result: "resultado",
8064
8318
  event: "evento",
8065
8319
  target: "alvo",
8066
- body: "corpo"
8320
+ body: "corpo",
8321
+ document: "documento",
8322
+ window: "janela",
8323
+ detail: "detalhe"
8067
8324
  },
8068
8325
  possessive: {
8069
8326
  marker: "de",
@@ -8159,6 +8416,27 @@ var init_portuguese = __esm({
8159
8416
  return: { primary: "retornar", alternatives: ["devolver"], normalized: "return" },
8160
8417
  then: { primary: "ent\xE3o", alternatives: ["logo"], normalized: "then" },
8161
8418
  and: { primary: "e", alternatives: ["tamb\xE9m", "al\xE9m disso"], normalized: "and" },
8419
+ // Comparison operator (`target matches .x`). Without this keyword the surface
8420
+ // stays an identifier and leaks verbatim into the condition's raw expression,
8421
+ // which the core expression parser reads as English (modal-close-backdrop /
8422
+ // focus-trap drop their then-branch). Not an ActionType and has no command
8423
+ // schema, so no pattern is generated from it.
8424
+ matches: { primary: "corresponde", normalized: "matches" },
8425
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
8426
+ // keyword the surface stays an identifier and leaks verbatim into the
8427
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
8428
+ // schema, so no pattern is generated from it.
8429
+ exists: { primary: "existe", normalized: "exists" },
8430
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
8431
+ // surface stays an identifier and leaks verbatim into the condition's raw
8432
+ // expression, which the core expression parser reads as English. Neither an
8433
+ // ActionType nor a command schema, so no pattern is generated from it.
8434
+ is: { primary: "\xE9", normalized: "is" },
8435
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
8436
+ // seam as `exists`: without the keyword the surface stays an identifier and
8437
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
8438
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
8439
+ no: { primary: "nenhum", normalized: "no" },
8162
8440
  end: { primary: "fim", alternatives: ["final", "t\xE9rmino"], normalized: "end" },
8163
8441
  js: { primary: "js", normalized: "js" },
8164
8442
  async: { primary: "ass\xEDncrono", normalized: "async" },
@@ -8265,7 +8543,10 @@ var init_quechua = __esm({
8265
8543
  result: "rurasqa",
8266
8544
  event: "ruwakuq",
8267
8545
  target: "punta",
8268
- body: "kurku"
8546
+ body: "kurku",
8547
+ document: "qillqa",
8548
+ window: "k_iri",
8549
+ detail: "sut_iy"
8269
8550
  },
8270
8551
  possessive: {
8271
8552
  marker: "-pa",
@@ -8336,7 +8617,10 @@ var init_quechua = __esm({
8336
8617
  focus: { primary: "qhawachiy", alternatives: ["qhaway"], normalized: "focus" },
8337
8618
  blur: { primary: "paqariy", alternatives: ["mana qhawachiy"], normalized: "blur" },
8338
8619
  // Phase 1 (v0.9.90): DOM / form state / debug
8339
- empty: { primary: "ch'usaq", normalized: "empty" },
8620
+ // Batch 3: apostrophe-less chusaq added — the i18n dict renders the empty
8621
+ // COMMAND with it (its `is empty` expression word), which parsed null against
8622
+ // the ch'usaq-only command patterns.
8623
+ empty: { primary: "ch'usaq", alternatives: ["chusaq"], normalized: "empty" },
8340
8624
  open: { primary: "paskay", normalized: "open" },
8341
8625
  close: { primary: "wichqay", normalized: "close" },
8342
8626
  select: { primary: "marcay", normalized: "select" },
@@ -8374,6 +8658,22 @@ var init_quechua = __esm({
8374
8658
  return: { primary: "kutichiy", alternatives: ["kutimuy"], normalized: "return" },
8375
8659
  then: { primary: "chaymantataq", alternatives: ["hinaspa", "chaymanta"], normalized: "then" },
8376
8660
  and: { primary: "hinallataq", alternatives: ["ima", "chaymantawan"], normalized: "and" },
8661
+ // Comparison operator (`target matches .x`). Without this keyword the surface
8662
+ // stays an identifier and leaks verbatim into the condition's raw expression,
8663
+ // which the core expression parser reads as English (modal-close-backdrop /
8664
+ // focus-trap drop their then-branch). Not an ActionType and has no command
8665
+ // schema, so no pattern is generated from it.
8666
+ matches: { primary: "tupan", normalized: "matches" },
8667
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
8668
+ // keyword the surface stays an identifier and leaks verbatim into the
8669
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
8670
+ // schema, so no pattern is generated from it.
8671
+ exists: { primary: "tiyan", normalized: "exists" },
8672
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
8673
+ // surface stays an identifier and leaks verbatim into the condition's raw
8674
+ // expression, which the core expression parser reads as English. Neither an
8675
+ // ActionType nor a command schema, so no pattern is generated from it.
8676
+ is: { primary: "kanqa", normalized: "is" },
8377
8677
  end: { primary: "tukukuy", alternatives: ["tukuy", "puchukay"], normalized: "end" },
8378
8678
  js: { primary: "js", normalized: "js" },
8379
8679
  async: { primary: "mana waqtalla", normalized: "async" },
@@ -8469,8 +8769,11 @@ var init_russian = __esm({
8469
8769
  result: "\u0440\u0435\u0437\u0443\u043B\u044C\u0442\u0430\u0442",
8470
8770
  event: "\u0441\u043E\u0431\u044B\u0442\u0438\u0435",
8471
8771
  target: "\u0446\u0435\u043B\u044C",
8472
- body: "\u0442\u0435\u043B\u043E"
8772
+ body: "\u0442\u0435\u043B\u043E",
8473
8773
  // was an English placeholder; the i18n dict emits the Russian word
8774
+ document: "\u0434\u043E\u043A\u0443\u043C\u0435\u043D\u0442",
8775
+ window: "\u043E\u043A\u043D\u043E",
8776
+ detail: "\u0434\u0435\u0442\u0430\u043B\u0438"
8474
8777
  },
8475
8778
  possessive: {
8476
8779
  marker: "",
@@ -8716,6 +9019,21 @@ var init_russian = __esm({
8716
9019
  // so `target соответствует .x` must normalize to `target matches .x`; otherwise
8717
9020
  // `соответствует` stays an identifier and modal-close-backdrop drops its then-branch.
8718
9021
  matches: { primary: "\u0441\u043E\u043E\u0442\u0432\u0435\u0442\u0441\u0442\u0432\u0443\u0435\u0442", normalized: "matches" },
9022
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
9023
+ // keyword the surface stays an identifier and leaks verbatim into the
9024
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
9025
+ // schema, so no pattern is generated from it.
9026
+ exists: { primary: "\u0441\u0443\u0449\u0435\u0441\u0442\u0432\u0443\u0435\u0442", normalized: "exists" },
9027
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
9028
+ // surface stays an identifier and leaks verbatim into the condition's raw
9029
+ // expression, which the core expression parser reads as English. Neither an
9030
+ // ActionType nor a command schema, so no pattern is generated from it.
9031
+ is: { primary: "\u0435\u0441\u0442\u044C", normalized: "is" },
9032
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
9033
+ // seam as `exists`: without the keyword the surface stays an identifier and
9034
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
9035
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
9036
+ no: { primary: "\u043D\u0435\u0442", normalized: "no" },
8719
9037
  end: { primary: "\u043A\u043E\u043D\u0435\u0446", normalized: "end" },
8720
9038
  // Advanced
8721
9039
  js: { primary: "js", normalized: "js" },
@@ -8832,7 +9150,10 @@ var init_swahili = __esm({
8832
9150
  result: "matokeo",
8833
9151
  event: "tukio",
8834
9152
  target: "lengo",
8835
- body: "mwili"
9153
+ body: "mwili",
9154
+ document: "hati",
9155
+ window: "dirisha",
9156
+ detail: "maelezo"
8836
9157
  },
8837
9158
  possessive: {
8838
9159
  marker: "",
@@ -8944,6 +9265,17 @@ var init_swahili = __esm({
8944
9265
  // Swahili copula ("is"); only recognized in predicate position (after a value,
8945
9266
  // before an adjective like `tupu`), so it doesn't disturb command parsing.
8946
9267
  is: { primary: "ni", normalized: "is" },
9268
+ // Comparison operator (`target matches .x`). Without this keyword the surface
9269
+ // stays an identifier and leaks verbatim into the condition's raw expression,
9270
+ // which the core expression parser reads as English (modal-close-backdrop /
9271
+ // focus-trap drop their then-branch). Not an ActionType and has no command
9272
+ // schema, so no pattern is generated from it.
9273
+ matches: { primary: "inafanana", normalized: "matches" },
9274
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
9275
+ // seam as `exists`: without the keyword the surface stays an identifier and
9276
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
9277
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
9278
+ no: { primary: "hakuna", normalized: "no" },
8947
9279
  end: { primary: "mwisho", alternatives: ["maliza", "tamati"], normalized: "end" },
8948
9280
  js: { primary: "js", alternatives: ["javascript"], normalized: "js" },
8949
9281
  async: { primary: "isiyo sawia", normalized: "async" },
@@ -9137,6 +9469,11 @@ var init_thai = __esm({
9137
9469
  return: { primary: "\u0E04\u0E37\u0E19\u0E04\u0E48\u0E32", alternatives: ["\u0E01\u0E25\u0E31\u0E1A"], normalized: "return" },
9138
9470
  then: { primary: "\u0E41\u0E25\u0E49\u0E27", alternatives: [], normalized: "then" },
9139
9471
  and: { primary: "\u0E41\u0E25\u0E30", alternatives: [], normalized: "and" },
9472
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
9473
+ // keyword the surface stays an identifier and leaks verbatim into the
9474
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
9475
+ // schema, so no pattern is generated from it.
9476
+ exists: { primary: "\u0E21\u0E35\u0E2D\u0E22\u0E39\u0E48", normalized: "exists" },
9140
9477
  end: { primary: "\u0E08\u0E1A", alternatives: [], normalized: "end" },
9141
9478
  // Advanced
9142
9479
  js: { primary: "\u0E40\u0E08\u0E40\u0E2D\u0E2A", alternatives: ["js"], normalized: "js" },
@@ -9240,8 +9577,11 @@ var init_tl = __esm({
9240
9577
  // "event"
9241
9578
  target: "target",
9242
9579
  // "target"
9243
- body: "katawan"
9580
+ body: "katawan",
9244
9581
  // was an English placeholder; the i18n dict emits the Tagalog word
9582
+ document: "dokumento",
9583
+ window: "bintana",
9584
+ detail: "detalye"
9245
9585
  },
9246
9586
  possessive: {
9247
9587
  marker: "ng",
@@ -9349,6 +9689,17 @@ var init_tl = __esm({
9349
9689
  return: { primary: "ibalik", alternatives: ["bumalik"], normalized: "return" },
9350
9690
  then: { primary: "pagkatapos", alternatives: ["saka"], normalized: "then" },
9351
9691
  and: { primary: "at", normalized: "and" },
9692
+ // Comparison operator (`target matches .x`). Without this keyword the surface
9693
+ // stays an identifier and leaks verbatim into the condition's raw expression,
9694
+ // which the core expression parser reads as English (modal-close-backdrop /
9695
+ // focus-trap drop their then-branch). Not an ActionType and has no command
9696
+ // schema, so no pattern is generated from it.
9697
+ matches: { primary: "tumutugma", normalized: "matches" },
9698
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
9699
+ // surface stays an identifier and leaks verbatim into the condition's raw
9700
+ // expression, which the core expression parser reads as English. Neither an
9701
+ // ActionType nor a command schema, so no pattern is generated from it.
9702
+ is: { primary: "ay", normalized: "is" },
9352
9703
  end: { primary: "wakas", alternatives: ["tapos"], normalized: "end" },
9353
9704
  // Advanced
9354
9705
  js: { primary: "js", normalized: "js" },
@@ -9448,7 +9799,10 @@ var init_turkish = __esm({
9448
9799
  result: "sonu\xE7",
9449
9800
  event: "olay",
9450
9801
  target: "hedef",
9451
- body: "g\xF6vde"
9802
+ body: "g\xF6vde",
9803
+ document: "belge",
9804
+ window: "pencere",
9805
+ detail: "detay"
9452
9806
  },
9453
9807
  possessive: {
9454
9808
  // Genitive suffix, spaced for tokenization like Turkish's other case
@@ -9510,7 +9864,10 @@ var init_turkish = __esm({
9510
9864
  // Dative/Locative + Genitive (with buffer consonants)
9511
9865
  source: { primary: "den", alternatives: ["dan", "ten", "tan"], position: "after" },
9512
9866
  // Ablative
9513
- style: { primary: "le", alternatives: ["la", "yle", "yla"], position: "after" },
9867
+ // `ile` is the free-standing instrumental the transformer actually emits
9868
+ // for with-phrases (`getir method:"POST" body:form ile`); the suffix
9869
+ // forms cover hand-written agglutinated variants.
9870
+ style: { primary: "le", alternatives: ["la", "yle", "yla", "ile"], position: "after" },
9514
9871
  // Instrumental
9515
9872
  event: { primary: "i", alternatives: ["\u0131", "u", "\xFC"], position: "after" }
9516
9873
  // Event as accusative
@@ -9607,6 +9964,24 @@ var init_turkish = __esm({
9607
9964
  and: { primary: "ve", alternatives: ["ayr\u0131ca", "hem de"], normalized: "and" },
9608
9965
  or: { primary: "veya", normalized: "or" },
9609
9966
  not: { primary: "de\u011Fil", alternatives: ["degil"], normalized: "not" },
9967
+ // Comparison operator (`target matches .x`). Without this keyword the surface
9968
+ // stays an identifier and leaks verbatim into the condition's raw expression,
9969
+ // which the core expression parser reads as English (modal-close-backdrop /
9970
+ // focus-trap drop their then-branch). Not an ActionType and has no command
9971
+ // schema, so no pattern is generated from it.
9972
+ matches: { primary: "e\u015Fle\u015Fir", normalized: "matches" },
9973
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
9974
+ // surface stays an identifier and leaks verbatim into the condition's raw
9975
+ // expression, which the core expression parser reads as English. Neither an
9976
+ // ActionType nor a command schema, so no pattern is generated from it.
9977
+ is: { primary: "dir", normalized: "is" },
9978
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
9979
+ // seam as `exists`: without the keyword the surface stays an identifier and
9980
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
9981
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
9982
+ // `yok` is a prefix of `else: 'yoksa'`; the keyword walk sorts longest-first, so
9983
+ // `yoksa` still wins where it appears.
9984
+ no: { primary: "yok", normalized: "no" },
9610
9985
  end: { primary: "son", alternatives: ["biti\u015F", "bitti"], normalized: "end" },
9611
9986
  // Advanced
9612
9987
  js: { primary: "js", normalized: "js" },
@@ -9701,8 +10076,11 @@ var init_ukrainian = __esm({
9701
10076
  result: "\u0440\u0435\u0437\u0443\u043B\u044C\u0442\u0430\u0442",
9702
10077
  event: "\u043F\u043E\u0434\u0456\u044F",
9703
10078
  target: "\u0446\u0456\u043B\u044C",
9704
- body: "\u0442\u0456\u043B\u043E"
10079
+ body: "\u0442\u0456\u043B\u043E",
9705
10080
  // was an English placeholder; the i18n dict emits the Ukrainian word
10081
+ document: "\u0434\u043E\u043A\u0443\u043C\u0435\u043D\u0442",
10082
+ window: "\u0432\u0456\u043A\u043D\u043E",
10083
+ detail: "\u0434\u0435\u0442\u0430\u043B\u0456"
9706
10084
  },
9707
10085
  possessive: {
9708
10086
  marker: "",
@@ -9966,6 +10344,21 @@ var init_ukrainian = __esm({
9966
10344
  // so `target відповідає .x` must normalize to `target matches .x`; otherwise
9967
10345
  // `відповідає` stays an identifier and modal-close-backdrop drops its then-branch.
9968
10346
  matches: { primary: "\u0432\u0456\u0434\u043F\u043E\u0432\u0456\u0434\u0430\u0454", normalized: "matches" },
10347
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
10348
+ // keyword the surface stays an identifier and leaks verbatim into the
10349
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
10350
+ // schema, so no pattern is generated from it.
10351
+ exists: { primary: "\u0456\u0441\u043D\u0443\u0454", normalized: "exists" },
10352
+ // Copula (`if result is false`, `if my value is empty`). Without the keyword the
10353
+ // surface stays an identifier and leaks verbatim into the condition's raw
10354
+ // expression, which the core expression parser reads as English. Neither an
10355
+ // ActionType nor a command schema, so no pattern is generated from it.
10356
+ is: { primary: "\u0454", normalized: "is" },
10357
+ // Negative-existence operator (`if no dragHandle set dragHandle to me`). Same
10358
+ // seam as `exists`: without the keyword the surface stays an identifier and
10359
+ // leaks verbatim into the condition's raw expression (behavior-draggable).
10360
+ // Neither an ActionType nor a command schema, so no pattern is generated from it.
10361
+ no: { primary: "\u043D\u0456", normalized: "no" },
9969
10362
  end: { primary: "\u043A\u0456\u043D\u0435\u0446\u044C", normalized: "end" },
9970
10363
  // Advanced
9971
10364
  js: { primary: "js", normalized: "js" },
@@ -10210,6 +10603,12 @@ var init_vietnamese = __esm({
10210
10603
  return: { primary: "tr\u1EA3 v\u1EC1", normalized: "return" },
10211
10604
  then: { primary: "r\u1ED3i", alternatives: ["sau \u0111\xF3", "th\xEC"], normalized: "then" },
10212
10605
  and: { primary: "v\xE0", normalized: "and" },
10606
+ // Comparison operator (`target matches .x`). Without this keyword the surface
10607
+ // stays an identifier and leaks verbatim into the condition's raw expression,
10608
+ // which the core expression parser reads as English (modal-close-backdrop /
10609
+ // focus-trap drop their then-branch). Not an ActionType and has no command
10610
+ // schema, so no pattern is generated from it.
10611
+ matches: { primary: "kh\u1EDBp", normalized: "matches" },
10213
10612
  end: { primary: "k\u1EBFt th\xFAc", normalized: "end" },
10214
10613
  // Advanced
10215
10614
  js: { primary: "js", normalized: "js" },
@@ -10304,7 +10703,10 @@ var init_chinese = __esm({
10304
10703
  result: "\u7ED3\u679C",
10305
10704
  event: "\u4E8B\u4EF6",
10306
10705
  target: "\u76EE\u6807",
10307
- body: "\u4E3B\u4F53"
10706
+ body: "\u4E3B\u4F53",
10707
+ document: "\u6587\u6863",
10708
+ window: "\u7A97\u53E3",
10709
+ detail: "\u8BE6\u60C5"
10308
10710
  },
10309
10711
  possessive: {
10310
10712
  marker: "\u7684",
@@ -10411,6 +10813,11 @@ var init_chinese = __esm({
10411
10813
  return: { primary: "\u8FD4\u56DE", normalized: "return" },
10412
10814
  then: { primary: "\u7136\u540E", alternatives: ["\u63A5\u7740", "\u90A3\u4E48"], normalized: "then" },
10413
10815
  and: { primary: "\u5E76\u4E14", alternatives: ["\u548C", "\u800C\u4E14"], normalized: "and" },
10816
+ // Existence operator (`if #modal exists`). Same seam as `matches`: without the
10817
+ // keyword the surface stays an identifier and leaks verbatim into the
10818
+ // condition's raw expression (if-exists). Neither an ActionType nor a command
10819
+ // schema, so no pattern is generated from it.
10820
+ exists: { primary: "\u5B58\u5728", normalized: "exists" },
10414
10821
  end: { primary: "\u7ED3\u675F", alternatives: ["\u7EC8\u6B62", "\u5B8C"], normalized: "end" },
10415
10822
  // Advanced
10416
10823
  js: { primary: "JS\u6267\u884C", alternatives: ["js"], normalized: "js" },
@@ -10906,8 +11313,22 @@ var init_schema_validator = __esm({
10906
11313
  "select",
10907
11314
  "clear",
10908
11315
  "reset",
10909
- "breakpoint"
11316
+ "breakpoint",
10910
11317
  // Zero-arg debug command
11318
+ // Feature blocks. Their meaning lives in the BODY, not in a head role: `live`
11319
+ // and `intercept` have no head at all, and eventsource/socket/worker's name and
11320
+ // url are structural, not semantic arguments. Giving them roles purely to make
11321
+ // `scoreRoleCoverage` return a non-vacuous number would inject new
11322
+ // `action.role:valueType` entries into the English R1 reference that all 23
11323
+ // other languages must also capture, or the role-fidelity ratchet fires. The
11324
+ // structural layer (`tryParseFeatureBlock`) parses them instead, and derives
11325
+ // confidence from the body — so the `maxScore === 0 → 1` shortcut is never the
11326
+ // thing that scores them.
11327
+ "live",
11328
+ "eventsource",
11329
+ "socket",
11330
+ "worker",
11331
+ "intercept"
10911
11332
  ]);
10912
11333
  }
10913
11334
  });
@@ -10951,7 +11372,7 @@ function getSchema(action) {
10951
11372
  function getDefinedSchemas() {
10952
11373
  return Object.values(commandSchemas).filter((s) => s.roles.length > 0 || s.bareKeyword === true);
10953
11374
  }
10954
- var toggleSchema, addSchema, removeSchema, putSchema, setSchema, bindSchema, liveSchema, eventsourceSchema, socketSchema, workerSchema, interceptSchema, showSchema, hideSchema, onSchema, triggerSchema, waitSchema, fetchSchema, incrementSchema, decrementSchema, appendSchema, prependSchema, logSchema, getCommandSchema, takeSchema, makeSchema, haltSchema, settleSchema, throwSchema, sendSchema, ifSchema, unlessSchema, elseSchema, repeatSchema, forSchema, whileSchema, continueSchema, goSchema, transitionSchema, cloneSchema, focusSchema, blurSchema, emptySchema, openSchema, closeSchema, selectSchema, clearSchema, resetSchema, breakpointSchema, callSchema, returnSchema, jsSchema, asyncSchema, tellSchema, defaultSchema, initSchema, behaviorSchema, installSchema, measureSchema, swapSchema, morphSchema, beepSchema, breakSchema, copySchema, exitSchema, pickSchema, scrollSchema, URL_MARKER_ALL_LANGS, PARTIALS_IN_MARKER_ALL_LANGS, pushSchema, replaceSchema, processSchema, renderSchema, commandSchemas;
11375
+ var toggleSchema, addSchema, removeSchema, putSchema, setSchema, bindSchema, liveSchema, eventsourceSchema, socketSchema, workerSchema, interceptSchema, showSchema, hideSchema, onSchema, triggerSchema, waitSchema, fetchSchema, incrementSchema, decrementSchema, appendSchema, prependSchema, logSchema, getCommandSchema, takeSchema, makeSchema, haltSchema, settleSchema, throwSchema, sendSchema, ifSchema, unlessSchema, elseSchema, repeatSchema, forSchema, whileSchema, continueSchema, URL_MARKER_ALL_LANGS, goSchema, transitionSchema, cloneSchema, focusSchema, blurSchema, emptySchema, openSchema, closeSchema, selectSchema, clearSchema, resetSchema, breakpointSchema, callSchema, returnSchema, jsSchema, asyncSchema, tellSchema, defaultSchema, initSchema, behaviorSchema, installSchema, measureSchema, swapSchema, morphSchema, beepSchema, breakSchema, copySchema, exitSchema, pickSchema, scrollSchema, PARTIALS_IN_MARKER_ALL_LANGS, pushSchema, replaceSchema, processSchema, renderSchema, commandSchemas;
10955
11376
  var init_command_schemas = __esm({
10956
11377
  "src/generators/command-schemas.ts"() {
10957
11378
  toggleSchema = {
@@ -11400,7 +11821,13 @@ var init_command_schemas = __esm({
11400
11821
  role: "source",
11401
11822
  description: "The element or property to bind to",
11402
11823
  required: true,
11403
- expectedTypes: ["selector", "reference", "expression"],
11824
+ // 'property-path' opts this role into the "of"-possessive matcher, so the
11825
+ // property-first render of `bind $x to #y's prop` (es `valor de #picker`,
11826
+ // ar `قيمة لـ #picker`) keeps its owner selector instead of collapsing to
11827
+ // the bare property word; see pattern-matcher tryMatchOfPossessiveExpression.
11828
+ // The selector-first languages (en `#picker's value`, ja `#pickerの 値`)
11829
+ // already reached property-path through tryMatchPossessiveSelectorExpression.
11830
+ expectedTypes: ["selector", "reference", "expression", "property-path"],
11404
11831
  svoPosition: 2,
11405
11832
  sovPosition: 2,
11406
11833
  // Element mirrors `set`/`add`/`put`'s value ("to") marking per language.
@@ -11587,7 +12014,15 @@ var init_command_schemas = __esm({
11587
12014
  expectedTypes: ["literal", "expression"],
11588
12015
  // expression for custom/namespaced event names
11589
12016
  svoPosition: 1,
11590
- sovPosition: 2
12017
+ sovPosition: 2,
12018
+ // hi/qu/bn mark trigger's event ACCUSATIVELY (`draggable:start को ट्रिगर`,
12019
+ // `draggable:start ta kichay`, `draggable:start কে ট্রিগার` — the corpus
12020
+ // renderings), but their profile-wide event marker is the on-handler one
12021
+ // (hi पर, qu locative pi, bn এ), so the generated SOV pattern never
12022
+ // matched and the whole line fell through to the on-handler reading (hi)
12023
+ // or failed outright (qu/bn). ja/ko were immune only because their event
12024
+ // marker IS the object particle (を / 을·를). #588 markerVariants machinery.
12025
+ markerVariants: { hi: ["\u0915\u094B"], qu: ["ta"], bn: ["\u0995\u09C7"] }
11591
12026
  },
11592
12027
  {
11593
12028
  role: "destination",
@@ -11633,14 +12068,26 @@ var init_command_schemas = __esm({
11633
12068
  renderOverride: { en: "" }
11634
12069
  // "fetch /api" (rendering — no preposition)
11635
12070
  },
12071
+ {
12072
+ role: "style",
12073
+ description: "Request options object (method, headers, body, credentials\u2026)",
12074
+ required: false,
12075
+ // expression-ONLY: the pattern matcher routes a `{ … }` run in an
12076
+ // expression-only slot through its object-literal fold, which preserves the
12077
+ // source text so the expression parser can build a real objectLiteral.
12078
+ // `style` is the role whose marker is `with` in every language profile.
12079
+ expectedTypes: ["expression"],
12080
+ svoPosition: 2,
12081
+ sovPosition: 2
12082
+ },
11636
12083
  {
11637
12084
  role: "responseType",
11638
12085
  description: "Response format (json, text, html, blob, etc.)",
11639
12086
  required: false,
11640
12087
  expectedTypes: ["literal", "expression"],
11641
12088
  // json/text/html are identifiers → expression type
11642
- svoPosition: 2,
11643
- sovPosition: 2,
12089
+ svoPosition: 3,
12090
+ sovPosition: 3,
11644
12091
  markerOverride: { en: "as" }
11645
12092
  // "fetch /api as json" — needed by schema-driven role inference
11646
12093
  },
@@ -11649,16 +12096,16 @@ var init_command_schemas = __esm({
11649
12096
  description: "HTTP method (GET, POST, etc.)",
11650
12097
  required: false,
11651
12098
  expectedTypes: ["literal"],
11652
- svoPosition: 3,
11653
- sovPosition: 3
12099
+ svoPosition: 4,
12100
+ sovPosition: 4
11654
12101
  },
11655
12102
  {
11656
12103
  role: "destination",
11657
12104
  description: "Where to store the result",
11658
12105
  required: false,
11659
12106
  expectedTypes: ["selector", "reference"],
11660
- svoPosition: 4,
11661
- sovPosition: 4
12107
+ svoPosition: 5,
12108
+ sovPosition: 5
11662
12109
  }
11663
12110
  ]
11664
12111
  };
@@ -12142,6 +12589,32 @@ var init_command_schemas = __esm({
12142
12589
  roles: []
12143
12590
  // No roles
12144
12591
  };
12592
+ URL_MARKER_ALL_LANGS = {
12593
+ en: "url",
12594
+ es: "url",
12595
+ pt: "url",
12596
+ fr: "url",
12597
+ de: "url",
12598
+ it: "url",
12599
+ ja: "url",
12600
+ ko: "url",
12601
+ zh: "url",
12602
+ ar: "url",
12603
+ he: "url",
12604
+ hi: "url",
12605
+ bn: "url",
12606
+ tr: "url",
12607
+ ru: "url",
12608
+ uk: "url",
12609
+ pl: "url",
12610
+ id: "url",
12611
+ vi: "url",
12612
+ th: "url",
12613
+ ms: "url",
12614
+ tl: "url",
12615
+ sw: "url",
12616
+ qu: "url"
12617
+ };
12145
12618
  goSchema = {
12146
12619
  action: "go",
12147
12620
  description: "Navigate to a URL",
@@ -12167,6 +12640,19 @@ var init_command_schemas = __esm({
12167
12640
  markerOptional: { en: true },
12168
12641
  markerVariants: { he: ["\u05D0\u05EA"], zh: ["\u628A"] }
12169
12642
  }
12643
+ ],
12644
+ // `go to url "/page"` — without this variant the destination captures the
12645
+ // bare word `url` and the actual URL is dropped as tolerated-trailing text,
12646
+ // in en and therefore in every render (the go-url corpus row). The required
12647
+ // `url` literal keeps the variant inert for `go back` / scroll forms.
12648
+ rolePrefixLiteralVariants: [
12649
+ {
12650
+ role: "destination",
12651
+ literal: URL_MARKER_ALL_LANGS,
12652
+ idSuffix: "url",
12653
+ priorityDelta: 5,
12654
+ methodCarrier: "method"
12655
+ }
12170
12656
  ]
12171
12657
  };
12172
12658
  transitionSchema = {
@@ -12797,7 +13283,27 @@ var init_command_schemas = __esm({
12797
13283
  th: "\u0E14\u0E49\u0E27\u0E22",
12798
13284
  vi: "v\u1EDBi",
12799
13285
  he: "\u05E2\u05DD",
12800
- zh: "\u7528"
13286
+ zh: "\u7528",
13287
+ // SOV/postpositional with-words. These follow the patient (`#b से`,
13288
+ // `#b দিয়ে`), matching the i18n `with` emission. Without them the SOV
13289
+ // patient-first swap pattern's trailing group (which binds the second
13290
+ // element to `destination`) had only the locative dest-marker (hi में,
13291
+ // bn তে) as its alternatives, so `#b <with-word>` never bound and #b
13292
+ // dropped — hi/bn/tr/qu rendered the invalid `swap with #a`. ja/ko
13293
+ // escaped only because their dest-marker alternatives already carry the
13294
+ // instrumental (で / 로). See generateSOVPatientFirstEventHandlerPattern.
13295
+ hi: "\u0938\u0947",
13296
+ bn: "\u09A6\u09BF\u09AF\u09BC\u09C7",
13297
+ tr: "ile",
13298
+ qu: "wan",
13299
+ // VSO with-words. The corpus puts the with-element AFTER the event
13300
+ // (`استبدل #a عند نقر بـ#b`, `palitan_pwesto #a kapag click nang #b`);
13301
+ // the vso-verb-first generator's swap-gated trailing group binds it to
13302
+ // `destination` via these words. ar's `بـ` is the bi-proclitic + tatweel
13303
+ // exactly as the ArabicProcliticExtractor emits it (glued to a selector
13304
+ // sigil). See generateVSOVerbFirstEventHandlerPattern.
13305
+ ar: "\u0628\u0640",
13306
+ tl: "nang"
12801
13307
  }
12802
13308
  }
12803
13309
  ]
@@ -12886,13 +13392,13 @@ var init_command_schemas = __esm({
12886
13392
  };
12887
13393
  pickSchema = {
12888
13394
  action: "pick",
12889
- description: "Select a random element from a collection",
13395
+ description: "Select item(s), character(s), a range, first/last/random N, or regex matches from a root",
12890
13396
  category: "variable",
12891
13397
  primaryRole: "patient",
12892
13398
  roles: [
12893
13399
  {
12894
13400
  role: "patient",
12895
- description: "The items to pick from",
13401
+ description: "The range/count/index/regex argument to pick",
12896
13402
  required: true,
12897
13403
  expectedTypes: ["literal", "expression", "reference"],
12898
13404
  svoPosition: 1,
@@ -12900,7 +13406,7 @@ var init_command_schemas = __esm({
12900
13406
  },
12901
13407
  {
12902
13408
  role: "source",
12903
- description: 'The array to pick from (with "from" keyword)',
13409
+ description: 'The root to pick from (with "of"/"from" keyword)',
12904
13410
  required: false,
12905
13411
  expectedTypes: ["reference", "expression"],
12906
13412
  svoPosition: 2,
@@ -12942,32 +13448,6 @@ var init_command_schemas = __esm({
12942
13448
  }
12943
13449
  ]
12944
13450
  };
12945
- URL_MARKER_ALL_LANGS = {
12946
- en: "url",
12947
- es: "url",
12948
- pt: "url",
12949
- fr: "url",
12950
- de: "url",
12951
- it: "url",
12952
- ja: "url",
12953
- ko: "url",
12954
- zh: "url",
12955
- ar: "url",
12956
- he: "url",
12957
- hi: "url",
12958
- bn: "url",
12959
- tr: "url",
12960
- ru: "url",
12961
- uk: "url",
12962
- pl: "url",
12963
- id: "url",
12964
- vi: "url",
12965
- th: "url",
12966
- ms: "url",
12967
- tl: "url",
12968
- sw: "url",
12969
- qu: "url"
12970
- };
12971
13451
  PARTIALS_IN_MARKER_ALL_LANGS = {
12972
13452
  en: "partials in",
12973
13453
  es: "partials in",
@@ -14641,17 +15121,48 @@ var init_generic_extractors = __esm({
14641
15121
  });
14642
15122
 
14643
15123
  // src/tokenizers/extractors/css-selector.ts
15124
+ function consumePseudoSegments(input, pos2) {
15125
+ let end = pos2;
15126
+ while (end < input.length && input[end] === ":") {
15127
+ const m = input.slice(end).match(/^::?[a-zA-Z][a-zA-Z0-9-]*/);
15128
+ if (!m) break;
15129
+ let segEnd = end + m[0].length;
15130
+ if (input[segEnd] === "(") {
15131
+ let depth = 0;
15132
+ let p = segEnd;
15133
+ while (p < input.length) {
15134
+ if (input[p] === "(") depth++;
15135
+ else if (input[p] === ")") {
15136
+ depth--;
15137
+ if (depth === 0) {
15138
+ p++;
15139
+ break;
15140
+ }
15141
+ }
15142
+ p++;
15143
+ }
15144
+ if (depth !== 0) break;
15145
+ segEnd = p;
15146
+ }
15147
+ end = segEnd;
15148
+ }
15149
+ return end;
15150
+ }
14644
15151
  function extractCssSelector(input, position) {
14645
15152
  const char = input[position];
14646
15153
  if (char === "#") {
14647
15154
  const match = input.slice(position).match(/^#[a-zA-Z_][\w-]*/);
14648
- return match ? match[0] : null;
15155
+ if (!match) return null;
15156
+ const end = consumePseudoSegments(input, position + match[0].length);
15157
+ return input.slice(position, end);
14649
15158
  }
14650
15159
  if (char === ".") {
14651
15160
  const dynamic = input.slice(position).match(/^\.\{[a-zA-Z_$][\w$]*\}/);
14652
15161
  if (dynamic) return dynamic[0];
14653
15162
  const match = input.slice(position).match(/^\.[a-zA-Z_][\w-]*/);
14654
- return match ? match[0] : null;
15163
+ if (!match) return null;
15164
+ const end = consumePseudoSegments(input, position + match[0].length);
15165
+ return input.slice(position, end);
14655
15166
  }
14656
15167
  if (char === "@") {
14657
15168
  const match = input.slice(position).match(/^@[a-zA-Z_][\w-]*/);
@@ -14669,7 +15180,8 @@ function extractCssSelector(input, position) {
14669
15180
  if (input[end] === "]") {
14670
15181
  depth--;
14671
15182
  if (depth === 0) {
14672
- return input.slice(position, end + 1);
15183
+ const pseudoEnd = consumePseudoSegments(input, end + 1);
15184
+ return input.slice(position, pseudoEnd);
14673
15185
  }
14674
15186
  }
14675
15187
  end++;
@@ -14677,7 +15189,9 @@ function extractCssSelector(input, position) {
14677
15189
  return null;
14678
15190
  }
14679
15191
  if (char === "<") {
14680
- const match = input.slice(position).match(/^<(?=[\w.#[])[\w-]*(?:[#.][\w-]+|\[[^\]]+\])*\s*\/>/);
15192
+ const match = input.slice(position).match(
15193
+ /^<(?=[\w.#[])[\w-]*(?:[#.][\w-]+|\[[^\]]+\]|::?[a-zA-Z][a-zA-Z0-9-]*(?:\([^)]*\))?)*\s*\/>/
15194
+ );
14681
15195
  return match ? match[0] : null;
14682
15196
  }
14683
15197
  return null;
@@ -14743,29 +15257,38 @@ var init_event_modifier = __esm({
14743
15257
  });
14744
15258
 
14745
15259
  // src/tokenizers/extractors/url.ts
15260
+ function findInterpolationEnd(input, start) {
15261
+ let depth = 1;
15262
+ for (let i = start; i < input.length; i++) {
15263
+ const ch = input[i];
15264
+ if (ch === "{") depth++;
15265
+ else if (ch === "}" && --depth === 0) return i + 1;
15266
+ }
15267
+ return -1;
15268
+ }
14746
15269
  function extractUrl(input, position) {
14747
15270
  const remaining = input.slice(position);
14748
- if (remaining.startsWith("http://") || remaining.startsWith("https://")) {
14749
- const match = remaining.match(/^https?:\/\/[^\s]*/);
14750
- return match ? match[0] : null;
14751
- }
14752
- if (remaining.startsWith("//")) {
14753
- const match = remaining.match(/^\/\/[^\s]*/);
14754
- return match ? match[0] : null;
14755
- }
14756
- if (remaining.startsWith("./") || remaining.startsWith("../")) {
14757
- const match = remaining.match(/^\.\.?\/[^\s]*/);
14758
- return match ? match[0] : null;
14759
- }
14760
- if (remaining.startsWith("/")) {
14761
- const match = remaining.match(/^\/[^\s]*/);
14762
- return match ? match[0] : null;
15271
+ const prefix = URL_PREFIXES.find((p) => remaining.startsWith(p));
15272
+ if (!prefix) return null;
15273
+ let i = prefix.length;
15274
+ while (i < remaining.length) {
15275
+ const ch = remaining[i];
15276
+ if (ch === "$" && remaining[i + 1] === "{") {
15277
+ const end = findInterpolationEnd(remaining, i + 2);
15278
+ if (end !== -1) {
15279
+ i = end;
15280
+ continue;
15281
+ }
15282
+ }
15283
+ if (/\s/.test(ch)) break;
15284
+ i++;
14763
15285
  }
14764
- return null;
15286
+ return remaining.slice(0, i);
14765
15287
  }
14766
- var UrlExtractor;
15288
+ var URL_PREFIXES, UrlExtractor;
14767
15289
  var init_url = __esm({
14768
15290
  "src/tokenizers/extractors/url.ts"() {
15291
+ URL_PREFIXES = ["http://", "https://", "//", "./", "../", "/"];
14769
15292
  UrlExtractor = class {
14770
15293
  constructor() {
14771
15294
  this.name = "url";
@@ -15858,6 +16381,18 @@ var init_arabic_proclitic = __esm({
15858
16381
  checkPos++;
15859
16382
  }
15860
16383
  if (remainingLength < 2) {
16384
+ const runIsTatweelOnly = remainingLength >= 1 && input.slice(nextPos, checkPos).split("").every((c) => c === "\u0640");
16385
+ const followChar = input[checkPos];
16386
+ if (entry.type === "preposition" && runIsTatweelOnly && (followChar === "#" || followChar === ".")) {
16387
+ return {
16388
+ value: input.slice(position, checkPos),
16389
+ length: checkPos - position,
16390
+ metadata: {
16391
+ procliticType: entry.type,
16392
+ normalized: entry.normalized
16393
+ }
16394
+ };
16395
+ }
15861
16396
  return null;
15862
16397
  }
15863
16398
  return {
@@ -16238,6 +16773,17 @@ var init_hindi_keyword = __esm({
16238
16773
  pos2 = extPos;
16239
16774
  }
16240
16775
  }
16776
+ if (this.context && input[pos2] === "_" && pos2 + 1 < input.length && isDevanagari(input[pos2 + 1])) {
16777
+ let extPos = pos2;
16778
+ let ext = word;
16779
+ while (extPos < input.length && (input[extPos] === "_" || isDevanagari(input[extPos]))) {
16780
+ ext += input[extPos++];
16781
+ }
16782
+ if (this.context.lookupKeyword(ext)) {
16783
+ word = ext;
16784
+ pos2 = extPos;
16785
+ }
16786
+ }
16241
16787
  if (!word) return null;
16242
16788
  const keywordEntry = this.context.lookupKeyword(word);
16243
16789
  const normalized2 = keywordEntry && keywordEntry.normalized !== keywordEntry.native ? keywordEntry.normalized : void 0;
@@ -16299,9 +16845,11 @@ var init_hindi_particle = __esm({
16299
16845
  }
16300
16846
  setContext(context) {
16301
16847
  this._context = context;
16302
- void this._context;
16303
16848
  }
16304
16849
  canExtract(input, position) {
16850
+ if (this.underscoreJoinedKeyword(input, position)) {
16851
+ return false;
16852
+ }
16305
16853
  for (const [particle] of COMPOUND_POSTPOSITIONS) {
16306
16854
  if (input.startsWith(particle, position)) {
16307
16855
  return true;
@@ -16315,7 +16863,27 @@ var init_hindi_particle = __esm({
16315
16863
  }
16316
16864
  return SINGLE_POSTPOSITIONS.has(word);
16317
16865
  }
16866
+ /**
16867
+ * True when the Devanagari run at `position` is `_`-joined into a keyword the
16868
+ * profile/EXTRAS registered (के_रूप_में). See the note in canExtract.
16869
+ */
16870
+ underscoreJoinedKeyword(input, position) {
16871
+ if (!this._context) return false;
16872
+ let pos2 = position;
16873
+ while (pos2 < input.length && this.isDevanagari(input[pos2])) pos2++;
16874
+ if (input[pos2] !== "_" || pos2 + 1 >= input.length || !this.isDevanagari(input[pos2 + 1])) {
16875
+ return false;
16876
+ }
16877
+ let ext = input.slice(position, pos2);
16878
+ while (pos2 < input.length && (input[pos2] === "_" || this.isDevanagari(input[pos2]))) {
16879
+ ext += input[pos2++];
16880
+ }
16881
+ return Boolean(this._context.lookupKeyword(ext));
16882
+ }
16318
16883
  extract(input, position) {
16884
+ if (this.underscoreJoinedKeyword(input, position)) {
16885
+ return null;
16886
+ }
16319
16887
  for (const [particle, metadata2] of COMPOUND_POSTPOSITIONS) {
16320
16888
  if (input.startsWith(particle, position)) {
16321
16889
  return {
@@ -16939,6 +17507,17 @@ var init_indonesian_keyword = __esm({
16939
17507
  while (pos2 < input.length && isIndonesianIdentifierChar(input[pos2])) {
16940
17508
  word += input[pos2++];
16941
17509
  }
17510
+ if (this.context && pos2 < input.length && input[pos2] === "_") {
17511
+ let extPos = pos2;
17512
+ let ext = word;
17513
+ while (extPos < input.length && (input[extPos] === "_" || isIndonesianIdentifierChar(input[extPos]))) {
17514
+ ext += input[extPos++];
17515
+ }
17516
+ if (this.context.lookupKeyword(ext.toLowerCase())) {
17517
+ word = ext;
17518
+ pos2 = extPos;
17519
+ }
17520
+ }
16942
17521
  if (!word) return null;
16943
17522
  const lower = word.toLowerCase();
16944
17523
  const isPreposition = PREPOSITIONS5.has(lower);
@@ -17389,14 +17968,16 @@ var init_quechua_keyword = __esm({
17389
17968
  metadata: { suffixValue: hyphenSuffix.toLowerCase() }
17390
17969
  };
17391
17970
  }
17392
- const maxKeywordLen = 12;
17971
+ const maxKeywordLen = 13;
17393
17972
  for (let len = Math.min(maxKeywordLen, input.length - startPos); len >= 2; len--) {
17394
17973
  const candidate = input.slice(startPos, startPos + len);
17395
17974
  const after = input[startPos + len];
17396
17975
  if (after !== void 0 && isQuechuaLetter(after)) continue;
17397
17976
  let allQuechua = true;
17398
17977
  for (let i = 0; i < candidate.length; i++) {
17399
- if (!isQuechuaLetter(candidate[i])) {
17978
+ const ch = candidate[i];
17979
+ if (ch === "_" && i > 0 && i < candidate.length - 1) continue;
17980
+ if (!isQuechuaLetter(ch)) {
17400
17981
  allQuechua = false;
17401
17982
  break;
17402
17983
  }
@@ -18283,6 +18864,12 @@ var init_japanese2 = __esm({
18283
18864
  { native: "\u524D", normalized: "previous" },
18284
18865
  { native: "\u6700\u3082\u8FD1\u3044", normalized: "closest" },
18285
18866
  { native: "\u89AA", normalized: "parent" },
18867
+ // Containment (`first <button/> in .modal`): the i18n dict emits の中, which
18868
+ // otherwise splits の(particle) + 中(identifier) — the stray identifier broke
18869
+ // the generated focus pattern's operand run (focus-trap Family G; tr/bn/hi
18870
+ // work because their in-word is one token). Whole-token entry mirrors en's
18871
+ // keyword `in` mid-run geometry.
18872
+ { native: "\u306E\u4E2D", normalized: "in" },
18286
18873
  // Events
18287
18874
  { native: "\u30AF\u30EA\u30C3\u30AF", normalized: "click" },
18288
18875
  { native: "\u5909\u66F4", normalized: "change" },
@@ -18311,6 +18898,14 @@ var init_japanese2 = __esm({
18311
18898
  // References (alternative forms not in profile)
18312
18899
  { native: "\u79C1", normalized: "me" },
18313
18900
  // Alternative to 自分 (jibun)
18901
+ // The i18n dict emits 対象 for `target` while the profile carries ターゲット, so the
18902
+ // word the authored corpus actually uses did not lex as a keyword and leaked into
18903
+ // the condition's raw expression (`if 対象 一致する .modal-backdrop`). Additive: the
18904
+ // profile's ターゲット stays registered. Must land WITH the `matches` keyword —
18905
+ // fixing the operand alone leaves the operator leaking and vice versa (see the
18906
+ // R2 note in japanese.ts's profile `matches` entry).
18907
+ { native: "\u5BFE\u8C61", normalized: "target" },
18908
+ // Alternative to ターゲット (the dict's word)
18314
18909
  // Note: Attached particle forms (を切り替え, を追加, etc.) are intentionally NOT included
18315
18910
  // because they would cause ambiguous parsing. The separate particle + verb pattern
18316
18911
  // (を + 切り替え) is preferred for consistent semantic analysis.
@@ -18322,7 +18917,11 @@ var init_japanese2 = __esm({
18322
18917
  { native: "\u79D2", normalized: "s" },
18323
18918
  { native: "\u30DF\u30EA\u79D2", normalized: "ms" },
18324
18919
  { native: "\u5206", normalized: "m" },
18325
- { native: "\u6642\u9593", normalized: "h" }
18920
+ { native: "\u6642\u9593", normalized: "h" },
18921
+ { native: "\u542B\u3080", normalized: "inclusive" },
18922
+ { native: "\u9664\u304F", normalized: "exclusive" },
18923
+ { native: "\u6587\u5B57", normalized: "characters" },
18924
+ { native: "\u30E9\u30F3\u30C0\u30E0", normalized: "random" }
18326
18925
  ];
18327
18926
  JapaneseTokenizer = class extends BaseTokenizer {
18328
18927
  constructor() {
@@ -18756,6 +19355,11 @@ var init_korean2 = __esm({
18756
19355
  { native: "\uAC70\uC9D3", normalized: "false" },
18757
19356
  { native: "\uB110", normalized: "null" },
18758
19357
  { native: "\uBBF8\uC815\uC758", normalized: "undefined" },
19358
+ // The corpus authors 정의안됨 ("not defined") for undefined (behavior-removable/
19359
+ // sortable `만약 X 이다 정의안됨`); without a whole-token entry it shatters into
19360
+ // 정 + 의안됨, leaking the invalid `is 정 의안됨`. Longest-first scan (cap 6)
19361
+ // matches the 4-char compound whole, like 마우스다운 above.
19362
+ { native: "\uC815\uC758\uC548\uB428", normalized: "undefined" },
18759
19363
  // Positional
18760
19364
  { native: "\uCCAB\uBC88\uC9F8", normalized: "first" },
18761
19365
  { native: "\uB9C8\uC9C0\uB9C9", normalized: "last" },
@@ -18763,6 +19367,11 @@ var init_korean2 = __esm({
18763
19367
  { native: "\uC774\uC804", normalized: "previous" },
18764
19368
  { native: "\uAC00\uC7A5\uAC00\uAE4C\uC6B4", normalized: "closest" },
18765
19369
  { native: "\uBD80\uBAA8", normalized: "parent" },
19370
+ // Containment (`first <button/> in .modal`): the i18n dict emits 안에, which
19371
+ // otherwise splits 안(identifier) + 에(particle) — the stray identifier broke
19372
+ // the generated focus pattern's operand run (focus-trap Family G). Whole-token
19373
+ // entry mirrors en's keyword `in` mid-run geometry.
19374
+ { native: "\uC548\uC5D0", normalized: "in" },
18766
19375
  // Events
18767
19376
  { native: "\uD074\uB9AD", normalized: "click" },
18768
19377
  { native: "\uB354\uBE14\uD074\uB9AD", normalized: "dblclick" },
@@ -18795,7 +19404,11 @@ var init_korean2 = __esm({
18795
19404
  { native: "\uCD08", normalized: "s" },
18796
19405
  { native: "\uBC00\uB9AC\uCD08", normalized: "ms" },
18797
19406
  { native: "\uBD84", normalized: "m" },
18798
- { native: "\uC2DC\uAC04", normalized: "h" }
19407
+ { native: "\uC2DC\uAC04", normalized: "h" },
19408
+ { native: "\uD3EC\uD568", normalized: "inclusive" },
19409
+ { native: "\uC81C\uC678", normalized: "exclusive" },
19410
+ { native: "\uBB38\uC790", normalized: "characters" },
19411
+ { native: "\uBB34\uC791\uC704", normalized: "random" }
18799
19412
  ];
18800
19413
  KoreanTokenizer = class extends BaseTokenizer {
18801
19414
  constructor() {
@@ -19064,6 +19677,17 @@ var init_arabic2 = __esm({
19064
19677
  // ka- (like, as)
19065
19678
  ]);
19066
19679
  ARABIC_EXTRAS = [
19680
+ // References (alternative forms not in profile). The i18n dict emits the BARE
19681
+ // nouns هدف/نتيجة while the profile carries the definite-article forms
19682
+ // الهدف/النتيجة, so the words the authored corpus actually uses did not lex as
19683
+ // keywords and leaked into the condition's raw expression (`if هدف يطابق …`).
19684
+ // Additive: the profile's الهدف/النتيجة stay registered. Same direction as the
19685
+ // profile's `body: 'جسم'` note — align to what the dict emits, never the reverse
19686
+ // (the dict wins on regeneration, so profile→dict is the convergent direction).
19687
+ { native: "\u0647\u062F\u0641", normalized: "target" },
19688
+ // Alternative to الهدف (the dict's word)
19689
+ { native: "\u0646\u062A\u064A\u062C\u0629", normalized: "result" },
19690
+ // Alternative to النتيجة (the dict's word)
19067
19691
  // Values/Literals
19068
19692
  { native: "\u0635\u062D\u064A\u062D", normalized: "true" },
19069
19693
  { native: "\u062E\u0637\u0623", normalized: "false" },
@@ -19128,13 +19752,17 @@ var init_arabic2 = __esm({
19128
19752
  { native: "\u062D\u064A\u0646", normalized: "on" },
19129
19753
  { native: "\u0644\u0645\u0651\u0627", normalized: "on" },
19130
19754
  { native: "\u0644\u0645\u0627", normalized: "on" },
19131
- { native: "\u0644\u062F\u0649", normalized: "on" }
19755
+ { native: "\u0644\u062F\u0649", normalized: "on" },
19132
19756
  //
19133
19757
  // Command spelling variants are now in the profile alternatives:
19134
19758
  // - toggle: بدل, غيّر, غير (in profile)
19135
19759
  // - add: اضف, زِد (in profile)
19136
19760
  // - remove: أزل, امسح (in profile)
19137
19761
  // - etc.
19762
+ { native: "\u0634\u0627\u0645\u0644", normalized: "inclusive" },
19763
+ { native: "\u062D\u0635\u0631\u064A", normalized: "exclusive" },
19764
+ { native: "\u062D\u0631\u0648\u0641", normalized: "characters" },
19765
+ { native: "\u0639\u0634\u0648\u0627\u0626\u064A", normalized: "random" }
19138
19766
  ];
19139
19767
  ArabicTokenizer = class extends BaseTokenizer {
19140
19768
  constructor() {
@@ -19218,7 +19846,7 @@ var init_arabic2 = __esm({
19218
19846
  pos2++;
19219
19847
  }
19220
19848
  }
19221
- return new TokenStreamImpl(tokens, this.language);
19849
+ return new TokenStreamImpl(this.mergeColonQualifiedNames(tokens), this.language);
19222
19850
  }
19223
19851
  classifyToken(token) {
19224
19852
  if (CONJUNCTIONS2.has(token)) return "conjunction";
@@ -19662,8 +20290,12 @@ var init_spanish2 = __esm({
19662
20290
  // Reference alternatives (accent variation, synonym)
19663
20291
  { native: "m\xED", normalized: "me" },
19664
20292
  // Accented form of mi
19665
- { native: "destino", normalized: "target" }
20293
+ { native: "destino", normalized: "target" },
19666
20294
  // Synonym for objetivo
20295
+ { native: "inclusivo", normalized: "inclusive" },
20296
+ { native: "exclusivo", normalized: "exclusive" },
20297
+ { native: "caracteres", normalized: "characters" },
20298
+ { native: "aleatorio", normalized: "random" }
19667
20299
  ];
19668
20300
  SpanishTokenizer = class extends BaseTokenizer {
19669
20301
  constructor() {
@@ -20120,6 +20752,19 @@ var init_turkish2 = __esm({
20120
20752
  { native: "farebirak", normalized: "mouseup" },
20121
20753
  { native: "kayd\u0131r", normalized: "scroll" },
20122
20754
  { native: "kaydir", normalized: "scroll" },
20755
+ // resize/scroll nominal forms: listed in eventNameTranslations (which only
20756
+ // the SOV-extraction path consults) but not registered as keywords — so a
20757
+ // fused *-sov-simple match captured them RAW (`boyutlandırma de çağır` →
20758
+ // event:expression:boyutlandırma, the window-resize R1 flip once the
20759
+ // debounced-head junk no longer forced the SOV-extraction path). Keyword
20760
+ // entries normalize them at the token, the same route the healthy natives
20761
+ // (tıklama→click) take.
20762
+ { native: "boyutland\u0131rma", normalized: "resize" },
20763
+ { native: "boyutlandirma", normalized: "resize" },
20764
+ { native: "boyutland\u0131r", normalized: "resize" },
20765
+ { native: "boyutlandir", normalized: "resize" },
20766
+ { native: "kayd\u0131rma", normalized: "scroll" },
20767
+ { native: "kaydirma", normalized: "scroll" },
20123
20768
  { native: "tu\u015F_bas", normalized: "keydown" },
20124
20769
  { native: "tus_bas", normalized: "keydown" },
20125
20770
  { native: "tu\u015F_b\u0131rak", normalized: "keyup" },
@@ -20128,7 +20773,11 @@ var init_turkish2 = __esm({
20128
20773
  { native: "saniye", normalized: "s" },
20129
20774
  { native: "milisaniye", normalized: "ms" },
20130
20775
  { native: "dakika", normalized: "m" },
20131
- { native: "saat", normalized: "h" }
20776
+ { native: "saat", normalized: "h" },
20777
+ { native: "dahil", normalized: "inclusive" },
20778
+ { native: "hari\xE7", normalized: "exclusive" },
20779
+ { native: "karakterler", normalized: "characters" },
20780
+ { native: "rastgele", normalized: "random" }
20132
20781
  ];
20133
20782
  TurkishTokenizer = class extends BaseTokenizer {
20134
20783
  constructor() {
@@ -20307,7 +20956,16 @@ var init_chinese2 = __esm({
20307
20956
  { native: "\u524D", normalized: "before" },
20308
20957
  { native: "\u540E", normalized: "after" },
20309
20958
  { native: "\u90A3\u4E48", normalized: "then" },
20310
- { native: "\u5B8C", normalized: "end" }
20959
+ { native: "\u5B8C", normalized: "end" },
20960
+ // Connectives. Whole-token so the greedy longest-first walk claims the 2-char
20961
+ // 作为 (`as`) before its 1-char tail 为 can match the `for` command primary —
20962
+ // without it `作为 Number` shattered into `作` + `为`→`for` (`computed-value`).
20963
+ // The reverse render (CONNECTIVE_LEXICON.zh) already maps 作为→as.
20964
+ { native: "\u4F5C\u4E3A", normalized: "as" },
20965
+ { native: "\u5305\u542B", normalized: "inclusive" },
20966
+ { native: "\u6392\u9664", normalized: "exclusive" },
20967
+ { native: "\u5B57\u7B26", normalized: "characters" },
20968
+ { native: "\u968F\u673A", normalized: "random" }
20311
20969
  ];
20312
20970
  ChineseTokenizer = class extends BaseTokenizer {
20313
20971
  constructor() {
@@ -20809,7 +21467,11 @@ var init_portuguese2 = __esm({
20809
21467
  { native: "padrao", normalized: "default" },
20810
21468
  { native: "at\xE9 que", normalized: "until" },
20811
21469
  // Multi-word phrases
20812
- { native: "dentro de", normalized: "into" }
21470
+ { native: "dentro de", normalized: "into" },
21471
+ { native: "inclusivo", normalized: "inclusive" },
21472
+ { native: "exclusivo", normalized: "exclusive" },
21473
+ { native: "caracteres", normalized: "characters" },
21474
+ { native: "aleat\xF3rio", normalized: "random" }
20813
21475
  ];
20814
21476
  PortugueseTokenizer = class extends BaseTokenizer {
20815
21477
  constructor() {
@@ -21273,7 +21935,11 @@ var init_french2 = __esm({
21273
21935
  // Additional morph synonym
21274
21936
  { native: "transmuter", normalized: "morph" },
21275
21937
  // Multi-word phrases
21276
- { native: "tant que", normalized: "while" }
21938
+ { native: "tant que", normalized: "while" },
21939
+ { native: "inclusif", normalized: "inclusive" },
21940
+ { native: "exclusif", normalized: "exclusive" },
21941
+ { native: "caract\xE8res", normalized: "characters" },
21942
+ { native: "al\xE9atoire", normalized: "random" }
21277
21943
  ];
21278
21944
  FrenchTokenizer = class extends BaseTokenizer {
21279
21945
  constructor() {
@@ -21714,7 +22380,11 @@ var init_german2 = __esm({
21714
22380
  // Verb conjugation variants (imperatives for test cases)
21715
22381
  { native: "erh\xF6he", normalized: "increment" },
21716
22382
  { native: "erhohe", normalized: "increment" },
21717
- { native: "verringere", normalized: "decrement" }
22383
+ { native: "verringere", normalized: "decrement" },
22384
+ { native: "inklusiv", normalized: "inclusive" },
22385
+ { native: "exklusiv", normalized: "exclusive" },
22386
+ { native: "Zeichen", normalized: "characters" },
22387
+ { native: "zuf\xE4llig", normalized: "random" }
21718
22388
  ];
21719
22389
  GermanTokenizer = class extends BaseTokenizer {
21720
22390
  constructor() {
@@ -21811,12 +22481,27 @@ var init_indonesian2 = __esm({
21811
22481
  // outside
21812
22482
  ]);
21813
22483
  INDONESIAN_EXTRAS = [
22484
+ // window-resize compound: the dict emits underscore-joined ubah_ukuran
22485
+ // (resize), which the `_` split shattered into ubah(→change) + _ + ukuran —
22486
+ // the event slot normalized to `change` and `_ ukuran` dropped unconsumed
22487
+ // (Arc F). Whole-token entry mirrors qu's hatun_kay precedent (quechua.ts).
22488
+ { native: "ubah_ukuran", normalized: "resize" },
22489
+ // behavior-draggable's `no` operator: the dict emits underscore-joined
22490
+ // tidak_ada, which the `_` split shattered into tidak(→not) + _ + ada(→exists).
22491
+ // Whole-token entry mirrors ubah_ukuran above; the keyword walk sorts
22492
+ // longest-first, so `tidak_ada` (9) beats `tidak` (5).
22493
+ { native: "tidak_ada", normalized: "no" },
21814
22494
  // Values/Literals
21815
22495
  { native: "benar", normalized: "true" },
21816
22496
  { native: "salah", normalized: "false" },
21817
22497
  { native: "null", normalized: "null" },
21818
22498
  { native: "kosong", normalized: "null" },
21819
22499
  { native: "tidakdidefinisikan", normalized: "undefined" },
22500
+ // The corpus authors `tidak_terdefinisi` for undefined (behavior-removable/
22501
+ // sortable `jika X adalah tidak_terdefinisi`); without a whole-token entry the
22502
+ // `_` split shatters it into tidak(→not) + `_ terdefinisi`, leaking the
22503
+ // invalid `is not _ terdefinisi`. Same shape as tidak_ada above.
22504
+ { native: "tidak_terdefinisi", normalized: "undefined" },
21820
22505
  // Positional
21821
22506
  { native: "pertama", normalized: "first" },
21822
22507
  { native: "terakhir", normalized: "last" },
@@ -21847,7 +22532,11 @@ var init_indonesian2 = __esm({
21847
22532
  { native: "atau", normalized: "or" },
21848
22533
  { native: "tidak", normalized: "not" },
21849
22534
  { native: "adalah", normalized: "is" },
21850
- { native: "ada", normalized: "exists" }
22535
+ { native: "ada", normalized: "exists" },
22536
+ { native: "inklusif", normalized: "inclusive" },
22537
+ { native: "eksklusif", normalized: "exclusive" },
22538
+ { native: "karakter", normalized: "characters" },
22539
+ { native: "acak", normalized: "random" }
21851
22540
  ];
21852
22541
  IndonesianTokenizer = class extends BaseTokenizer {
21853
22542
  constructor() {
@@ -22062,7 +22751,7 @@ var init_quechua2 = __esm({
22062
22751
  this.name = "quechua-string-literal";
22063
22752
  }
22064
22753
  canExtract(input, position) {
22065
- return input[position] === '"' || input[position] === "'";
22754
+ return input[position] === '"' || input[position] === "'" || input[position] === "`";
22066
22755
  }
22067
22756
  extract(input, position) {
22068
22757
  const quote = input[position];
@@ -22121,6 +22810,8 @@ var init_quechua2 = __esm({
22121
22810
  // (set-attribute `@disabled ta cheqaq man …`); without it the value tokenized
22122
22811
  // as a bare identifier and `set @disabled to <undefined>` ran. arí/ari ("yes")
22123
22812
  // are the colloquial alternates, kept for input tolerance.
22813
+ // Pick unit word (arc 3) — mirrors the i18n dict's `characters: 'sanampa'`.
22814
+ { native: "sanampa", normalized: "characters" },
22124
22815
  { native: "cheqaq", normalized: "true" },
22125
22816
  { native: "ar\xED", normalized: "true" },
22126
22817
  { native: "ari", normalized: "true" },
@@ -22155,6 +22846,31 @@ var init_quechua2 = __esm({
22155
22846
  // aswan-prefixed compound splits (the suffix extractor strips -wan from
22156
22847
  // 'aswan'). The i18n dict emits bare 'kaylla' (near/close).
22157
22848
  { native: "kaylla", normalized: "closest" },
22849
+ // Containment (`first <button/> in .modal`): the i18n dict emits ukupi,
22850
+ // which otherwise splits uku(identifier) + pi — and the stranded `pi`
22851
+ // mis-reads as the EVENT marker (the ñawpaqpi/qhepapi class above; same
22852
+ // longest-first cure). Whole-token entry mirrors en's keyword `in` mid-run
22853
+ // geometry (focus-trap Family G).
22854
+ { native: "ukupi", normalized: "in" },
22855
+ // window-resize compounds: the dict emits underscore-joined k_iri (window)
22856
+ // and hatun_kay (resize), which the `_` split shattered into junk role
22857
+ // fragments (call.source:literal="k_iri" destination:literal="hatun_" —
22858
+ // the qu window-resize R1 row; hatun_kay sits in eventNameTranslations but
22859
+ // never arrived whole). The ñawpaq_kaq entry above is the precedent.
22860
+ { native: "k_iri", normalized: "window" },
22861
+ { native: "hatun_kay", normalized: "resize" },
22862
+ // behavior-draggable's `no` operator: the dict emits underscore-joined
22863
+ // mana_kanchu, which the `_` split shattered into mana(→not/without) + _ +
22864
+ // kanchu. Same whole-token shape as hatun_kay; longest-first makes
22865
+ // `mana_kanchu` (11) beat `mana` (4).
22866
+ { native: "mana_kanchu", normalized: "no" },
22867
+ // `undefined`: the dict emits underscore-joined `mana_riqsisqa` ("not known"),
22868
+ // which the `_` split shattered into mana(→false) + _ + riqsisqa — rendering
22869
+ // `is false _ riqsisqa` and breaking the canonical parse (behavior-removable/qu,
22870
+ // behavior-sortable/qu `if triggerEl is undefined`). The bare `mana riqsisqa`
22871
+ // (space) entry above never fires — the corpus authors the underscore form.
22872
+ // Same whole-token shape as mana_kanchu; longest-first makes it beat `mana`.
22873
+ { native: "mana_riqsisqa", normalized: "undefined" },
22158
22874
  { native: "qaylla", normalized: "closest" },
22159
22875
  { native: "tayta", normalized: "parent" },
22160
22876
  // Events
@@ -22216,7 +22932,8 @@ var init_quechua2 = __esm({
22216
22932
  { native: "qhawachiy", normalized: "focus" },
22217
22933
  { native: "mana qhawachiy", normalized: "blur" },
22218
22934
  // Suffix modifiers
22219
- { native: "-manta", normalized: "from" }
22935
+ { native: "-manta", normalized: "from" },
22936
+ { native: "imaymanata", normalized: "random" }
22220
22937
  ];
22221
22938
  QuechuaTokenizer = class extends BaseTokenizer {
22222
22939
  constructor() {
@@ -22244,7 +22961,7 @@ var init_quechua2 = __esm({
22244
22961
  return "event-modifier";
22245
22962
  if (token.startsWith("#") || token.startsWith(".") || token.startsWith("[") || token.startsWith("*") || token.startsWith("<"))
22246
22963
  return "selector";
22247
- if (token.startsWith('"')) return "literal";
22964
+ if (token.startsWith('"') || token.startsWith("'")) return "literal";
22248
22965
  if (/^\d/.test(token)) return "literal";
22249
22966
  if (["==", "!=", "<=", ">=", "<", ">", "&&", "||", "!"].includes(token)) return "operator";
22250
22967
  return "identifier";
@@ -22313,6 +23030,12 @@ var init_swahili2 = __esm({
22313
23030
  // between
22314
23031
  ]);
22315
23032
  SWAHILI_EXTRAS = [
23033
+ // window-resize compound: the dict emits underscore-joined badilisha_ukubwa
23034
+ // (resize), which the `_` split shattered into badilisha(→toggle!) + _ +
23035
+ // ukubwa — the event slot normalized to `toggle` and `_ ukubwa` dropped
23036
+ // unconsumed (Arc F). Whole-token entry mirrors qu's hatun_kay precedent
23037
+ // (quechua.ts).
23038
+ { native: "badilisha_ukubwa", normalized: "resize" },
22316
23039
  // Values/Literals
22317
23040
  { native: "kweli", normalized: "true" },
22318
23041
  { native: "uongo", normalized: "false" },
@@ -22386,7 +23109,9 @@ var init_swahili2 = __esm({
22386
23109
  { native: "si", normalized: "not" },
22387
23110
  { native: "ni", normalized: "is" },
22388
23111
  { native: "ipo", normalized: "exists" },
22389
- { native: "tupu", normalized: "empty" }
23112
+ { native: "tupu", normalized: "empty" },
23113
+ { native: "herufi", normalized: "characters" },
23114
+ { native: "nasibu", normalized: "random" }
22390
23115
  ];
22391
23116
  SwahiliTokenizer = class extends BaseTokenizer {
22392
23117
  constructor() {
@@ -23070,7 +23795,11 @@ var init_italian2 = __esm({
23070
23795
  { native: "vuoto", normalized: "empty" },
23071
23796
  // Synonyms not in profile
23072
23797
  { native: "toggle", normalized: "toggle" },
23073
- { native: "di", normalized: "tell" }
23798
+ { native: "di", normalized: "tell" },
23799
+ { native: "inclusivo", normalized: "inclusive" },
23800
+ { native: "esclusivo", normalized: "exclusive" },
23801
+ { native: "caratteri", normalized: "characters" },
23802
+ { native: "casuale", normalized: "random" }
23074
23803
  ];
23075
23804
  ItalianTokenizer = class extends BaseTokenizer {
23076
23805
  constructor() {
@@ -23175,7 +23904,11 @@ var init_vietnamese2 = __esm({
23175
23904
  { native: "t\u1ED3n t\u1EA1i", normalized: "exists" },
23176
23905
  { native: "r\u1ED7ng", normalized: "empty" },
23177
23906
  // English synonyms
23178
- { native: "javascript", normalized: "js" }
23907
+ { native: "javascript", normalized: "js" },
23908
+ { native: "bao g\u1ED3m", normalized: "inclusive" },
23909
+ { native: "lo\u1EA1i tr\u1EEB", normalized: "exclusive" },
23910
+ { native: "k\xFD t\u1EF1", normalized: "characters" },
23911
+ { native: "ng\u1EABu nhi\xEAn", normalized: "random" }
23179
23912
  ];
23180
23913
  VietnameseTokenizer = class extends BaseTokenizer {
23181
23914
  constructor() {
@@ -23558,7 +24291,11 @@ var init_polish2 = __esm({
23558
24291
  { native: "jest", normalized: "is" },
23559
24292
  { native: "istnieje", normalized: "exists" },
23560
24293
  { native: "pusty", normalized: "empty" },
23561
- { native: "puste", normalized: "empty" }
24294
+ { native: "puste", normalized: "empty" },
24295
+ { native: "w\u0142\u0105cznie", normalized: "inclusive" },
24296
+ { native: "wy\u0142\u0105cznie", normalized: "exclusive" },
24297
+ { native: "znaki", normalized: "characters" },
24298
+ { native: "losowy", normalized: "random" }
23562
24299
  ];
23563
24300
  PolishTokenizer = class extends BaseTokenizer {
23564
24301
  constructor() {
@@ -23988,6 +24725,12 @@ var init_russian2 = __esm({
23988
24725
  { native: "\u043B\u043E\u0436\u044C", normalized: "false" },
23989
24726
  { native: "null", normalized: "null" },
23990
24727
  { native: "\u043D\u0435\u043E\u043F\u0440\u0435\u0434\u0435\u043B\u0435\u043D\u043E", normalized: "undefined" },
24728
+ // `ничего` ("nothing") is the word the corpus author uses for a null
24729
+ // comparison (`если item есть ничего` → `if item is null`). Without it the
24730
+ // literal leaked verbatim and the canonical parser rejected the render
24731
+ // (behavior-sortable/ru). Its sibling `неопределено`→undefined was already
24732
+ // registered; this closes the null half.
24733
+ { native: "\u043D\u0438\u0447\u0435\u0433\u043E", normalized: "null" },
23991
24734
  // Time units (not in profile - handled by number parser)
23992
24735
  { native: "\u0441\u0435\u043A\u0443\u043D\u0434\u0430", normalized: "s" },
23993
24736
  { native: "\u0441\u0435\u043A\u0443\u043D\u0434\u044B", normalized: "s" },
@@ -24033,8 +24776,11 @@ var init_russian2 = __esm({
24033
24776
  // feminine
24034
24777
  { native: "\u043C\u043E\u0451", normalized: "my" },
24035
24778
  // neuter
24036
- { native: "\u043C\u043E\u0438", normalized: "my" }
24779
+ { native: "\u043C\u043E\u0438", normalized: "my" },
24037
24780
  // plural
24781
+ { native: "\u0432\u043A\u043B\u044E\u0447\u0438\u0442\u0435\u043B\u044C\u043D\u043E", normalized: "inclusive" },
24782
+ { native: "\u0438\u0441\u043A\u043B\u044E\u0447\u0438\u0442\u0435\u043B\u044C\u043D\u043E", normalized: "exclusive" },
24783
+ { native: "\u0441\u0438\u043C\u0432\u043E\u043B\u044B", normalized: "characters" }
24038
24784
  ];
24039
24785
  RussianTokenizer = class extends BaseTokenizer {
24040
24786
  constructor() {
@@ -24443,6 +25189,11 @@ var init_ukrainian2 = __esm({
24443
25189
  { native: "\u0445\u0438\u0431\u043D\u0456\u0441\u0442\u044C", normalized: "false" },
24444
25190
  { native: "null", normalized: "null" },
24445
25191
  { native: "\u043D\u0435\u0432\u0438\u0437\u043D\u0430\u0447\u0435\u043D\u043E", normalized: "undefined" },
25192
+ // `нічого` ("nothing") is the corpus author's word for a null comparison
25193
+ // (`якщо item є нічого` → `if item is null`); without it the literal leaked
25194
+ // verbatim and the canonical parser rejected the render (behavior-sortable/uk).
25195
+ // Sibling of the already-registered `невизначено`→undefined.
25196
+ { native: "\u043D\u0456\u0447\u043E\u0433\u043E", normalized: "null" },
24446
25197
  // Time units (not in profile - handled by number parser)
24447
25198
  { native: "\u0441\u0435\u043A\u0443\u043D\u0434\u0430", normalized: "s" },
24448
25199
  { native: "\u0441\u0435\u043A\u0443\u043D\u0434\u0438", normalized: "s" },
@@ -24488,8 +25239,11 @@ var init_ukrainian2 = __esm({
24488
25239
  // feminine
24489
25240
  { native: "\u043C\u043E\u0454", normalized: "my" },
24490
25241
  // neuter
24491
- { native: "\u043C\u043E\u0457", normalized: "my" }
25242
+ { native: "\u043C\u043E\u0457", normalized: "my" },
24492
25243
  // plural
25244
+ { native: "\u0432\u043A\u043B\u044E\u0447\u043D\u043E", normalized: "inclusive" },
25245
+ { native: "\u0432\u0438\u043A\u043B\u044E\u0447\u043D\u043E", normalized: "exclusive" },
25246
+ { native: "\u0441\u0438\u043C\u0432\u043E\u043B\u0438", normalized: "characters" }
24493
25247
  ];
24494
25248
  UkrainianTokenizer = class extends BaseTokenizer {
24495
25249
  constructor() {
@@ -24629,7 +25383,11 @@ var init_he2 = __esm({
24629
25383
  { native: "\u05D3\u05E7\u05D4", normalized: "m" },
24630
25384
  { native: "\u05D3\u05E7\u05D5\u05EA", normalized: "m" },
24631
25385
  { native: "\u05E9\u05E2\u05D4", normalized: "h" },
24632
- { native: "\u05E9\u05E2\u05D5\u05EA", normalized: "h" }
25386
+ { native: "\u05E9\u05E2\u05D5\u05EA", normalized: "h" },
25387
+ { native: "\u05DB\u05D5\u05DC\u05DC", normalized: "inclusive" },
25388
+ { native: "\u05D1\u05DC\u05E2\u05D3\u05D9", normalized: "exclusive" },
25389
+ { native: "\u05EA\u05D5\u05D5\u05D9\u05DD", normalized: "characters" },
25390
+ { native: "\u05D0\u05E7\u05E8\u05D0\u05D9", normalized: "random" }
24633
25391
  ];
24634
25392
  HebrewTokenizer = class extends BaseTokenizer {
24635
25393
  constructor() {
@@ -24786,6 +25544,12 @@ var init_hindi2 = __esm({
24786
25544
  // splits on it — see hi.ts events note). repeat-until-event / handler events.
24787
25545
  { native: "\u092E\u093E\u0909\u0938\u0928\u0940\u091A\u0947", normalized: "mousedown" },
24788
25546
  { native: "\u092E\u093E\u0909\u0938\u090A\u092A\u0930", normalized: "mouseup" },
25547
+ // window-resize compound: the dict emits underscore-joined आकार_बदलें
25548
+ // (resize), which the `_` split shattered into आकार + _ + बदलें — and the
25549
+ // stranded बदलें (toggle verb) anchored a PHANTOM toggle command while the
25550
+ // event slot grabbed the call target (the hi window-resize mis-parse,
25551
+ // Arc F). Whole-token entry mirrors qu's hatun_kay precedent (quechua.ts).
25552
+ { native: "\u0906\u0915\u093E\u0930_\u092C\u0926\u0932\u0947\u0902", normalized: "resize" },
24789
25553
  // Values
24790
25554
  { native: "\u0938\u091A", normalized: "true" },
24791
25555
  { native: "\u0938\u0924\u094D\u092F", normalized: "true" },
@@ -24809,7 +25573,26 @@ var init_hindi2 = __esm({
24809
25573
  { native: "\u0938\u094D\u0915\u094D\u0930\u0949\u0932", normalized: "scroll" },
24810
25574
  // Additional modifiers not in profile
24811
25575
  { native: "\u0915\u094B", normalized: "to" },
24812
- { native: "\u0915\u0947 \u0938\u093E\u0925", normalized: "with" }
25576
+ { native: "\u0915\u0947 \u0938\u093E\u0925", normalized: "with" },
25577
+ // Connectives. Whole-token underscore-joined surface, mirroring आकार_बदलें
25578
+ // above: the `_` split shattered के_रूप_में (`as`) into के + _ + रूप + _ + में
25579
+ // (`computed-value`). Registering it lets the tokenizer's underscore-recovery
25580
+ // block adopt the whole run. The reverse render (CONNECTIVE_LEXICON.hi) already
25581
+ // maps के_रूप_में→as; it was a documented dead entry awaiting exactly this.
25582
+ { native: "\u0915\u0947_\u0930\u0942\u092A_\u092E\u0947\u0902", normalized: "as" },
25583
+ // `या` (or) — dict hi.ts `or`; already matched by surface in the parser's
25584
+ // OR_KEYWORDS (event-adjacent `or` was absorbed), but every raw-expression
25585
+ // occurrence leaked verbatim (when-multiple-changes). Phantom-safe: `or` is
25586
+ // neither an ActionType nor a command schema.
25587
+ { native: "\u092F\u093E", normalized: "or" },
25588
+ // `बदलने पर` (changes / "on changing") — dict hi.ts `changes`, SPACED whole
25589
+ // phrase via the multi-word keyword walk (`के साथ` precedent above). NEVER
25590
+ // register bare `बदलने`: the stem `बदल` is a registered toggle-verb
25591
+ // alternative (patterns/toggle.ts) and the morphological normalizer strips
25592
+ // conjugations — a bare entry re-opens the आकार_बदलें phantom-toggle class.
25593
+ { native: "\u092C\u0926\u0932\u0928\u0947 \u092A\u0930", normalized: "changes" },
25594
+ { native: "\u0905\u0915\u094D\u0937\u0930", normalized: "characters" },
25595
+ { native: "\u092F\u093E\u0926\u0943\u091A\u094D\u091B\u093F\u0915", normalized: "random" }
24813
25596
  ];
24814
25597
  HindiTokenizer = class extends BaseTokenizer {
24815
25598
  constructor() {
@@ -24991,7 +25774,17 @@ var init_bengali2 = __esm({
24991
25774
  { native: "\u09B8\u09CD\u0995\u09CD\u09B0\u09CB\u09B2", normalized: "scroll" },
24992
25775
  // Additional modifiers not in profile
24993
25776
  { native: "\u0995\u09C7", normalized: "to" },
24994
- { native: "\u09B8\u09BE\u09A5\u09C7", normalized: "with" }
25777
+ { native: "\u09B8\u09BE\u09A5\u09C7", normalized: "with" },
25778
+ // Conjunctions. `অথবা` (or) — dict bn.ts `or`. Already matched by surface in the
25779
+ // parser's OR_KEYWORDS (event-adjacent `or` was absorbed); registering it lets
25780
+ // surfaceOf emit `or` inside raw expressions (the wait-for event list in
25781
+ // behavior-draggable/sortable). Phantom-safe: `or` is neither an ActionType nor
25782
+ // a command schema.
25783
+ { native: "\u0985\u09A5\u09AC\u09BE", normalized: "or" },
25784
+ { native: "\u0985\u09A8\u09CD\u09A4\u09B0\u09CD\u09AD\u09C1\u0995\u09CD\u09A4", normalized: "inclusive" },
25785
+ { native: "\u09AC\u09BE\u09A6", normalized: "exclusive" },
25786
+ { native: "\u0985\u0995\u09CD\u09B7\u09B0", normalized: "characters" },
25787
+ { native: "\u098F\u09B2\u09CB\u09AE\u09C7\u09B2\u09CB", normalized: "random" }
24995
25788
  ];
24996
25789
  BengaliTokenizer = class extends BaseTokenizer {
24997
25790
  constructor() {
@@ -25065,11 +25858,19 @@ var init_thai2 = __esm({
25065
25858
  { native: "\u0E2D\u0E34\u0E19\u0E1E\u0E38\u0E15", normalized: "input" },
25066
25859
  { native: "\u0E42\u0E2B\u0E25\u0E14", normalized: "load" },
25067
25860
  { native: "\u0E40\u0E25\u0E37\u0E48\u0E2D\u0E19", normalized: "scroll" },
25861
+ // `ปรับขนาด` (resize) — dict th.ts `resize`; without it the greedy scan
25862
+ // shattered it into ป + รับ(→take) + ขนาด (window-resize/th rendered
25863
+ // `on ป take ขนาด …`). Precedent: hi आकार_बदलें, tr boyutlandırma.
25864
+ { native: "\u0E1B\u0E23\u0E31\u0E1A\u0E02\u0E19\u0E32\u0E14", normalized: "resize" },
25068
25865
  // Additional modifiers
25069
25866
  { native: "\u0E40\u0E27\u0E25\u0E32", normalized: "when" },
25070
25867
  { native: "\u0E44\u0E1B\u0E22\u0E31\u0E07", normalized: "to" },
25071
25868
  { native: "\u0E14\u0E49\u0E27\u0E22", normalized: "with" },
25072
- { native: "\u0E41\u0E25\u0E30", normalized: "and" }
25869
+ { native: "\u0E41\u0E25\u0E30", normalized: "and" },
25870
+ { native: "\u0E23\u0E27\u0E21", normalized: "inclusive" },
25871
+ { native: "\u0E22\u0E01\u0E40\u0E27\u0E49\u0E19", normalized: "exclusive" },
25872
+ { native: "\u0E2D\u0E31\u0E01\u0E02\u0E23\u0E30", normalized: "characters" },
25873
+ { native: "\u0E2A\u0E38\u0E48\u0E21", normalized: "random" }
25073
25874
  ];
25074
25875
  ThaiTokenizer = class extends BaseTokenizer {
25075
25876
  constructor() {
@@ -25141,8 +25942,12 @@ var init_ms2 = __esm({
25141
25942
  // Alternative for input (means "enter")
25142
25943
  { native: "muat", normalized: "load" },
25143
25944
  { native: "tatal", normalized: "scroll" },
25144
- { native: "hover", normalized: "hover" }
25945
+ { native: "hover", normalized: "hover" },
25145
25946
  // English loanword commonly used
25947
+ { native: "inklusif", normalized: "inclusive" },
25948
+ { native: "eksklusif", normalized: "exclusive" },
25949
+ { native: "aksara", normalized: "characters" },
25950
+ { native: "rawak", normalized: "random" }
25146
25951
  ];
25147
25952
  MalayTokenizer = class extends BaseTokenizer {
25148
25953
  constructor() {
@@ -25401,7 +26206,11 @@ var init_tl2 = __esm({
25401
26206
  { native: "isumite", normalized: "submit" },
25402
26207
  { native: "input", normalized: "input" },
25403
26208
  { native: "karga", normalized: "load" },
25404
- { native: "mag_scroll", normalized: "scroll" }
26209
+ { native: "mag_scroll", normalized: "scroll" },
26210
+ { native: "kasama", normalized: "inclusive" },
26211
+ { native: "bukod", normalized: "exclusive" },
26212
+ { native: "karakter", normalized: "characters" },
26213
+ { native: "random", normalized: "random" }
25405
26214
  ];
25406
26215
  TagalogTokenizer = class extends BaseTokenizer {
25407
26216
  constructor() {
@@ -25981,6 +26790,28 @@ function getEventHandlerPatternsHi() {
25981
26790
  event: { marker: "\u0938\u0947", position: 2 }
25982
26791
  }
25983
26792
  },
26793
+ // Prefix reactive `when` — the hi member of the ja/tr/ar/he when-family
26794
+ // below (`जब $firstName या $lastName बदलने पर …`). Without it,
26795
+ // `event-hi-bare` captured the जब token itself as the event (render
26796
+ // `on when put …`) and dropped the subject list; en's `event-en-when`
26797
+ // captures the first subject as the event. The event role is
26798
+ // type-constrained so the `जब तक` while/until compound (repeat-while,
26799
+ // unless-condition) never matches — तक lexes as a keyword/literal and
26800
+ // declines, falling through to the repeat patterns unchanged.
26801
+ {
26802
+ id: "event-hi-when",
26803
+ language: "hi",
26804
+ command: "on",
26805
+ priority: 95,
26806
+ template: {
26807
+ format: "\u091C\u092C {event} {body}",
26808
+ tokens: [
26809
+ { type: "literal", value: "\u091C\u092C" },
26810
+ { type: "role", role: "event", expectedTypes: ["reference", "expression", "selector"] }
26811
+ ]
26812
+ },
26813
+ extraction: { event: { position: 1 } }
26814
+ },
25984
26815
  // Bare event name: क्लिक
25985
26816
  {
25986
26817
  id: "event-hi-bare",
@@ -27133,7 +27964,15 @@ var init_event_handler = __esm({
27133
27964
  \uBE14\uB7EC: "blur",
27134
27965
  \uB85C\uB4DC: "load",
27135
27966
  \uB9AC\uC0AC\uC774\uC988: "resize",
27136
- \uC2A4\uD06C\uB864: "scroll"
27967
+ \uC2A4\uD06C\uB864: "scroll",
27968
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
27969
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
27970
+ \uB9C8\uC6B0\uC2A4\uC5D4\uD130: "mouseenter",
27971
+ \uB9C8\uC6B0\uC2A4\uB9AC\uBE0C: "mouseleave",
27972
+ \uB9C8\uC6B0\uC2A4\uBB34\uBE0C: "mousemove",
27973
+ \uD0A4\uD504\uB808\uC2A4: "keypress",
27974
+ \uD130\uCE58\uC885\uB8CC: "touchend",
27975
+ \uD130\uCE58\uCDE8\uC18C: "touchcancel"
27137
27976
  },
27138
27977
  // Japanese event names → English
27139
27978
  ja: {
@@ -27153,7 +27992,12 @@ var init_event_handler = __esm({
27153
27992
  \u30ED\u30FC\u30C9: "load",
27154
27993
  \u8AAD\u307F\u8FBC\u307F: "load",
27155
27994
  \u30B5\u30A4\u30BA\u5909\u66F4: "resize",
27156
- \u30B9\u30AF\u30ED\u30FC\u30EB: "scroll"
27995
+ \u30B9\u30AF\u30ED\u30FC\u30EB: "scroll",
27996
+ // V3 Batch 2 alias: i18n dictionary form the ja tokenizer already
27997
+ // normalizes (probe-verified).
27998
+ \u307C\u304B\u3057: "blur"
27999
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28000
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
27157
28001
  },
27158
28002
  // Arabic event names → English
27159
28003
  ar: {
@@ -27170,7 +28014,19 @@ var init_event_handler = __esm({
27170
28014
  "\u062A\u0645\u0631\u064A\u0631 \u0627\u0644\u0645\u0627\u0648\u0633": "mouseover",
27171
28015
  \u0627\u0644\u062A\u0631\u0643\u064A\u0632: "focus",
27172
28016
  \u062A\u062D\u0645\u064A\u0644: "load",
27173
- \u062A\u0645\u0631\u064A\u0631: "scroll"
28017
+ \u062A\u0645\u0631\u064A\u0631: "scroll",
28018
+ // V3 Batch 2 aliases: i18n dictionary forms the ar tokenizer already
28019
+ // normalizes (probe-verified captured values). Appended so first-wins
28020
+ // localization canonicals above are unchanged.
28021
+ \u062A\u0631\u0643\u064A\u0632: "focus",
28022
+ "\u0645\u0641\u062A\u0627\u062D \u0623\u0633\u0641\u0644": "keydown",
28023
+ "\u0645\u0641\u062A\u0627\u062D \u0623\u0639\u0644\u0649": "keyup",
28024
+ "\u0641\u0623\u0631\u0629 \u0641\u0648\u0642": "mouseover",
28025
+ // Arc F: the dict renders resize as the two-word تغيير حجم; the event
28026
+ // slot captures only تغيير (→change) and حجم drops. The compound key is
28027
+ // matched by the parser's event-compound reclaim (offset-exact join of
28028
+ // the captured event word + the dangling fragment).
28029
+ "\u062A\u063A\u064A\u064A\u0631 \u062D\u062C\u0645": "resize"
27174
28030
  },
27175
28031
  // Spanish event names → English
27176
28032
  es: {
@@ -27187,7 +28043,26 @@ var init_event_handler = __esm({
27187
28043
  enfoque: "focus",
27188
28044
  desenfoque: "blur",
27189
28045
  carga: "load",
27190
- desplazamiento: "scroll"
28046
+ desplazamiento: "scroll",
28047
+ // V3 Batch 2 aliases: i18n dictionary verb forms the es tokenizer already
28048
+ // normalizes (probe-verified). Appended — localization canonicals unchanged.
28049
+ cambiar: "change",
28050
+ enfocar: "focus",
28051
+ desenfocar: "blur",
28052
+ cargar: "load",
28053
+ desplazar: "scroll",
28054
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28055
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28056
+ dobleclic: "dblclick",
28057
+ rat\u00F3nentrar: "mouseenter",
28058
+ rat\u00F3nsalir: "mouseleave",
28059
+ rat\u00F3nmover: "mousemove",
28060
+ teclapresar: "keypress",
28061
+ descargar: "unload",
28062
+ toqueempezar: "touchstart",
28063
+ toqueterminar: "touchend",
28064
+ toquemover: "touchmove",
28065
+ toquecancelar: "touchcancel"
27191
28066
  },
27192
28067
  // Turkish event names → English
27193
28068
  tr: {
@@ -27219,7 +28094,16 @@ var init_event_handler = __esm({
27219
28094
  // the `kaydır`/`kaydırma` scroll precedent) keeps the event token whole.
27220
28095
  boyutland\u0131rma: "resize",
27221
28096
  boyutland\u0131r: "resize",
27222
- kayd\u0131rma: "scroll"
28097
+ kayd\u0131rma: "scroll",
28098
+ // V3 Batch 2 aliases: i18n dictionary forms the tr tokenizer already
28099
+ // normalizes (probe-verified; farebas/farebırak are the deliberately fused
28100
+ // dict forms — the table's own fare_bas/fare_bırak `_` entries shatter).
28101
+ bulan\u0131k: "blur",
28102
+ farebas: "mousedown",
28103
+ fareb\u0131rak: "mouseup",
28104
+ kayd\u0131r: "scroll"
28105
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28106
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
27223
28107
  },
27224
28108
  // Portuguese event names → English
27225
28109
  pt: {
@@ -27246,7 +28130,19 @@ var init_event_handler = __esm({
27246
28130
  carregar: "load",
27247
28131
  carregamento: "load",
27248
28132
  rolagem: "scroll",
27249
- rolar: "scroll"
28133
+ rolar: "scroll",
28134
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28135
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28136
+ duploClique: "dblclick",
28137
+ mouseEntrar: "mouseenter",
28138
+ mouseSair: "mouseleave",
28139
+ mouseMover: "mousemove",
28140
+ teclaPressionar: "keypress",
28141
+ descarregar: "unload",
28142
+ toqueIn\u00EDcio: "touchstart",
28143
+ toqueFim: "touchend",
28144
+ toqueMover: "touchmove",
28145
+ toqueCancelar: "touchcancel"
27250
28146
  },
27251
28147
  // Chinese event names → English
27252
28148
  zh: {
@@ -27272,7 +28168,18 @@ var init_event_handler = __esm({
27272
28168
  \u6A21\u7CCA: "blur",
27273
28169
  \u52A0\u8F7D: "load",
27274
28170
  \u8F7D\u5165: "load",
27275
- \u6EDA\u52A8: "scroll"
28171
+ \u6EDA\u52A8: "scroll",
28172
+ // V3 Batch 2 alias: the i18n dictionary keydown form (captures keydown via
28173
+ // the registered 按键 prefix; probe-verified — kept over bare 按键 to avoid
28174
+ // colliding with the dict's keypress entry).
28175
+ \u6309\u952E\u6309\u4E0B: "keydown",
28176
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28177
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28178
+ \u9F20\u6807\u79FB\u52A8: "mousemove",
28179
+ \u5378\u8F7D: "unload",
28180
+ \u8C03\u6574\u5927\u5C0F: "resize",
28181
+ \u89E6\u6478\u5F00\u59CB: "touchstart",
28182
+ \u89E6\u6478\u79FB\u52A8: "touchmove"
27276
28183
  },
27277
28184
  // French event names → English
27278
28185
  fr: {
@@ -27297,7 +28204,22 @@ var init_event_handler = __esm({
27297
28204
  chargement: "load",
27298
28205
  charger: "load",
27299
28206
  d\u00E9filement: "scroll",
27300
- d\u00E9filer: "scroll"
28207
+ d\u00E9filer: "scroll",
28208
+ // V3 Batch 2 alias: i18n dictionary form the fr tokenizer already
28209
+ // normalizes (probe-verified).
28210
+ flou: "blur",
28211
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28212
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28213
+ doubleclic: "dblclick",
28214
+ sourisentrer: "mouseenter",
28215
+ sourissortir: "mouseleave",
28216
+ sourisbouger: "mousemove",
28217
+ touchepress\u00E9e: "keypress",
28218
+ d\u00E9charger: "unload",
28219
+ touchercommencer: "touchstart",
28220
+ toucherfin: "touchend",
28221
+ toucherbouger: "touchmove",
28222
+ toucherannuler: "touchcancel"
27301
28223
  },
27302
28224
  // German event names → English
27303
28225
  de: {
@@ -27321,7 +28243,26 @@ var init_event_handler = __esm({
27321
28243
  laden: "load",
27322
28244
  ladung: "load",
27323
28245
  scrollen: "scroll",
27324
- bl\u00E4ttern: "scroll"
28246
+ bl\u00E4ttern: "scroll",
28247
+ // V3 Batch 2 aliases: the de tokenizer's registered multi-word event forms
28248
+ // (probe-verified; the table's older `taste runter`/`taste hoch`/`maus
28249
+ // über`/`maus raus` entries are aspirational — they do not tokenize).
28250
+ "taste unten": "keydown",
28251
+ "taste oben": "keyup",
28252
+ "maus dr\xFCber": "mouseover",
28253
+ "maus weg": "mouseout",
28254
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28255
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28256
+ doppelklick: "dblclick",
28257
+ mauseintreten: "mouseenter",
28258
+ mausverlassen: "mouseleave",
28259
+ mausbewegen: "mousemove",
28260
+ tastedr\u00FCcken: "keypress",
28261
+ entladen: "unload",
28262
+ ber\u00FChrungstart: "touchstart",
28263
+ ber\u00FChrungend: "touchend",
28264
+ ber\u00FChrungbewegen: "touchmove",
28265
+ ber\u00FChrungabbrechen: "touchcancel"
27325
28266
  },
27326
28267
  // Indonesian event names → English
27327
28268
  id: {
@@ -27341,7 +28282,18 @@ var init_event_handler = __esm({
27341
28282
  muat: "load",
27342
28283
  memuat: "load",
27343
28284
  gulir: "scroll",
27344
- menggulir: "scroll"
28285
+ menggulir: "scroll",
28286
+ // V3 Batch 2 aliases: tekan_tombol captures keydown via the registered
28287
+ // `tekan`; arahkan/tinggalkan are the tokenizer's registered natives;
28288
+ // keyup is English passthrough (no parseable id native — `lepas` is
28289
+ // unregistered). All probe-verified.
28290
+ tekan_tombol: "keydown",
28291
+ keyup: "keyup",
28292
+ arahkan: "mouseover",
28293
+ tinggalkan: "mouseout",
28294
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28295
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28296
+ bongkar: "unload"
27345
28297
  },
27346
28298
  // Bengali event names → English
27347
28299
  bn: {
@@ -27354,6 +28306,8 @@ var init_event_handler = __esm({
27354
28306
  \u099D\u09BE\u09AA\u09B8\u09BE: "blur",
27355
28307
  \u09AB\u09CB\u0995\u09BE\u09B8: "focus",
27356
28308
  \u09AA\u09B0\u09BF\u09AC\u09B0\u09CD\u09A4\u09A8: "change"
28309
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28310
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
27357
28311
  },
27358
28312
  // Quechua event names → English (loanwords with native adaptations)
27359
28313
  qu: {
@@ -27364,8 +28318,14 @@ var init_event_handler = __esm({
27364
28318
  yaykuy: "input",
27365
28319
  tikray: "change",
27366
28320
  "t'ikray": "change",
28321
+ // Batch 3 aliases (appended so first-wins localization canonicals are
28322
+ // unchanged): the dict now renders kambiay/apaykachay — probe-verified to
28323
+ // capture the canonical event via the tokenizer keyword table, unlike
28324
+ // tikray (captures 'toggle') and kachay ('send' in one corpus slot).
28325
+ kambiay: "change",
27367
28326
  apachiy: "submit",
27368
28327
  kachay: "submit",
28328
+ apaykachay: "submit",
27369
28329
  "llave uray": "keydown",
27370
28330
  "llave hawa": "keyup",
27371
28331
  "q'away": "focus",
@@ -27378,6 +28338,8 @@ var init_event_handler = __esm({
27378
28338
  kunray: "scroll",
27379
28339
  muyuy: "scroll",
27380
28340
  hatun_kay: "resize"
28341
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28342
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
27381
28343
  },
27382
28344
  // Swahili event names → English
27383
28345
  sw: {
@@ -27399,7 +28361,31 @@ var init_event_handler = __esm({
27399
28361
  pakia: "load",
27400
28362
  kupakia: "load",
27401
28363
  sogeza: "scroll",
27402
- kusogeza: "scroll"
28364
+ kusogeza: "scroll",
28365
+ // V3 Batch 2 aliases: i18n dictionary forms the sw tokenizer already
28366
+ // normalizes (probe-verified; bonyeza is corpus-hot — 106 rows), plus the
28367
+ // tokenizer's registered `sogeza juu` for mouseover (the table's `panya
28368
+ // juu` is mouseup's dict form and maps there).
28369
+ bonyeza: "click",
28370
+ ingizo: "input",
28371
+ kitufe_shuka: "keydown",
28372
+ kitufe_juu: "keyup",
28373
+ panya_nje: "mouseout",
28374
+ wasilisha: "submit",
28375
+ "sogeza juu": "mouseover",
28376
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28377
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
28378
+ shuka: "unload"
28379
+ },
28380
+ // Vietnamese event names → English. Minimal section: the dict renders
28381
+ // resize as the three-word đổi kích thước; the event slot captures only
28382
+ // đổi (tokenizer-normalized → change) and `kích thước` drops. The compound
28383
+ // key is matched by the parser's event-compound reclaim (Arc F,
28384
+ // offset-exact join of the captured event word + the dangling fragment).
28385
+ vi: {
28386
+ "\u0111\u1ED5i k\xEDch th\u01B0\u1EDBc": "resize"
28387
+ // V3c burn-down (2026-07-14): dictionary event words S5b never covered —
28388
+ // parse-side registration so the i18n dict forms resolve (round-trip-tested).
27403
28389
  }
27404
28390
  };
27405
28391
  Object.fromEntries(
@@ -27572,7 +28558,17 @@ function generateSOVPatientFirstEventHandlerPattern(commandSchema, profile, keyw
27572
28558
  const verbToken = keyword.alternatives ? { type: "literal", value: keyword.primary, alternatives: keyword.alternatives } : { type: "literal", value: keyword.primary };
27573
28559
  tokens.push(verbToken);
27574
28560
  tokens.push(...eventHandlerSourceGroup(commandSchema, profile.roleMarkers.source));
27575
- tokens.push(...eventHandlerDestinationGroup(commandSchema, profile.roleMarkers.destination));
28561
+ let trailingDestMarker = profile.roleMarkers.destination;
28562
+ if (commandSchema.action === "swap" && trailingDestMarker) {
28563
+ const withWord = commandSchema.roles.find((r) => r.role === "patient")?.markerOverride?.[profile.code];
28564
+ if (withWord && withWord !== trailingDestMarker.primary) {
28565
+ const existing = trailingDestMarker.alternatives ?? [];
28566
+ if (!existing.includes(withWord)) {
28567
+ trailingDestMarker = { ...trailingDestMarker, alternatives: [...existing, withWord] };
28568
+ }
28569
+ }
28570
+ }
28571
+ tokens.push(...eventHandlerDestinationGroup(commandSchema, trailingDestMarker));
27576
28572
  return {
27577
28573
  id: `${commandSchema.action}-event-${profile.code}-sov-patient-first`,
27578
28574
  language: profile.code,
@@ -28004,6 +29000,19 @@ function generateVSOVerbFirstEventHandlerPattern(commandSchema, profile, keyword
28004
29000
  tokens.push(markerToken);
28005
29001
  }
28006
29002
  tokens.push({ type: "role", role: "event", optional: false });
29003
+ if (commandSchema.action === "swap") {
29004
+ const withWord = commandSchema.roles.find((r) => r.role === "patient")?.markerOverride?.[profile.code];
29005
+ if (withWord) {
29006
+ tokens.push({
29007
+ type: "group",
29008
+ optional: true,
29009
+ tokens: [
29010
+ { type: "literal", value: withWord },
29011
+ { type: "role", role: "destination", optional: false }
29012
+ ]
29013
+ });
29014
+ }
29015
+ }
28007
29016
  return {
28008
29017
  id: `${commandSchema.action}-event-${profile.code}-vso-verb-first`,
28009
29018
  language: profile.code,
@@ -28282,12 +29291,16 @@ function generateVerbFirstPattern(schema, profile, config = defaultConfig) {
28282
29291
  const keyword = profile.keywords[schema.action];
28283
29292
  if (!keyword) return null;
28284
29293
  const verbToken = keyword.alternatives ? { type: "literal", value: keyword.primary, alternatives: keyword.alternatives } : { type: "literal", value: keyword.primary };
28285
- const roleTokens = requiredRoles.map((r) => ({
28286
- type: "role",
28287
- role: r.role,
28288
- optional: false,
28289
- expectedTypes: r.expectedTypes
28290
- }));
29294
+ const roleTokens = requiredRoles.flatMap((r) => {
29295
+ const prefix = r.valuePrefixLiteral?.[profile.code];
29296
+ const roleToken = {
29297
+ type: "role",
29298
+ role: r.role,
29299
+ optional: false,
29300
+ expectedTypes: r.expectedTypes
29301
+ };
29302
+ return prefix ? [{ type: "literal", value: prefix }, roleToken] : [roleToken];
29303
+ });
28291
29304
  return {
28292
29305
  id: `${schema.action}-${profile.code}-generated-verb-first`,
28293
29306
  language: profile.code,
@@ -28329,6 +29342,37 @@ function generatePatternVariants(schema, profile, config = defaultConfig) {
28329
29342
  patterns.push(verbFirst);
28330
29343
  }
28331
29344
  }
29345
+ for (const v of schema.rolePrefixLiteralVariants ?? []) {
29346
+ const literal = v.literal[profile.code];
29347
+ if (!literal) continue;
29348
+ const { rolePrefixLiteralVariants: _omitted, ...baseSchema } = schema;
29349
+ const cloneSchema2 = {
29350
+ ...baseSchema,
29351
+ roles: schema.roles.map(
29352
+ (r) => r.role === v.role ? { ...r, valuePrefixLiteral: { [profile.code]: literal } } : r
29353
+ )
29354
+ };
29355
+ const delta = v.priorityDelta ?? 5;
29356
+ const carrier = v.methodCarrier ? { [v.methodCarrier]: { value: literal } } : {};
29357
+ const main = generatePattern(cloneSchema2, profile, config);
29358
+ patterns.push({
29359
+ ...main,
29360
+ id: `${schema.action}-${profile.code}-generated-${v.idSuffix}`,
29361
+ priority: (config.basePriority ?? 100) + delta,
29362
+ extraction: { ...main.extraction, ...carrier }
29363
+ });
29364
+ if (config.generateVerbFirstVariants !== false) {
29365
+ const verbFirstUrl = generateVerbFirstPattern(cloneSchema2, profile, config);
29366
+ if (verbFirstUrl) {
29367
+ patterns.push({
29368
+ ...verbFirstUrl,
29369
+ id: `${schema.action}-${profile.code}-generated-verb-first-${v.idSuffix}`,
29370
+ priority: (config.basePriority ?? 100) - 20 + delta,
29371
+ extraction: { ...verbFirstUrl.extraction, ...carrier }
29372
+ });
29373
+ }
29374
+ }
29375
+ }
28332
29376
  return patterns;
28333
29377
  }
28334
29378
  function generatePatternsForLanguage(profile, config = defaultConfig) {
@@ -28552,25 +29596,31 @@ function buildRoleToken(roleSpec, profile) {
28552
29596
  const tokens = [];
28553
29597
  const overrideMarker = roleSpec.markerOverride?.[profile.code];
28554
29598
  const defaultMarker = profile.roleMarkers[roleSpec.role];
29599
+ const suppressMarker = roleSpec.renderOverride?.[profile.code] === "";
28555
29600
  const roleValueToken = {
28556
29601
  type: "role",
28557
29602
  role: roleSpec.role,
28558
29603
  optional: !roleSpec.required,
28559
29604
  expectedTypes: roleSpec.expectedTypes
28560
29605
  };
29606
+ const prefixLiteral = roleSpec.valuePrefixLiteral?.[profile.code];
29607
+ const pushPrefixed = () => {
29608
+ if (prefixLiteral) tokens.push({ type: "literal", value: prefixLiteral });
29609
+ tokens.push(roleValueToken);
29610
+ };
28561
29611
  if (overrideMarker !== void 0) {
28562
29612
  const markerWords = overrideMarker ? overrideMarker.split(/\s+/).filter(Boolean) : [];
28563
29613
  const position = defaultMarker?.position ?? "before";
28564
29614
  const optionalMarker = roleSpec.markerOptional?.[profile.code] === true;
28565
29615
  const pushWord = (word) => {
28566
- const literal = { type: "literal", value: word };
29616
+ const literal = suppressMarker ? { type: "literal", value: word, renderSuppress: true } : { type: "literal", value: word };
28567
29617
  tokens.push(optionalMarker ? { type: "group", optional: true, tokens: [literal] } : literal);
28568
29618
  };
28569
29619
  if (position === "before") {
28570
29620
  for (const word of markerWords) pushWord(word);
28571
- tokens.push(roleValueToken);
29621
+ pushPrefixed();
28572
29622
  } else {
28573
- tokens.push(roleValueToken);
29623
+ pushPrefixed();
28574
29624
  for (const word of markerWords) pushWord(word);
28575
29625
  }
28576
29626
  } else if (defaultMarker) {
@@ -28579,7 +29629,12 @@ function buildRoleToken(roleSpec, profile) {
28579
29629
  const alternatives = [
28580
29630
  .../* @__PURE__ */ new Set([...defaultMarker.alternatives ?? [], ...variantAlts])
28581
29631
  ].filter((a) => a !== defaultMarker.primary);
28582
- return alternatives.length ? { type: "literal", value: defaultMarker.primary, alternatives } : { type: "literal", value: defaultMarker.primary };
29632
+ return {
29633
+ type: "literal",
29634
+ value: defaultMarker.primary,
29635
+ ...alternatives.length ? { alternatives } : {},
29636
+ ...suppressMarker ? { renderSuppress: true } : {}
29637
+ };
28583
29638
  };
28584
29639
  const pushMarker = (marker) => {
28585
29640
  tokens.push(
@@ -28590,13 +29645,13 @@ function buildRoleToken(roleSpec, profile) {
28590
29645
  if (defaultMarker.primary) {
28591
29646
  pushMarker(asMarker());
28592
29647
  }
28593
- tokens.push(roleValueToken);
29648
+ pushPrefixed();
28594
29649
  } else {
28595
- tokens.push(roleValueToken);
29650
+ pushPrefixed();
28596
29651
  pushMarker(asMarker());
28597
29652
  }
28598
29653
  } else {
28599
- tokens.push(roleValueToken);
29654
+ pushPrefixed();
28600
29655
  }
28601
29656
  return tokens;
28602
29657
  }
@@ -28605,7 +29660,9 @@ function buildExtractionRules(schema, profile) {
28605
29660
  for (const roleSpec of schema.roles) {
28606
29661
  const overrideMarker = roleSpec.markerOverride?.[profile.code];
28607
29662
  const defaultMarker = profile.roleMarkers[roleSpec.role];
28608
- if (overrideMarker !== void 0) {
29663
+ if (roleSpec.valuePrefixLiteral?.[profile.code]) {
29664
+ rules[roleSpec.role] = { marker: roleSpec.valuePrefixLiteral[profile.code] };
29665
+ } else if (overrideMarker !== void 0) {
28609
29666
  rules[roleSpec.role] = overrideMarker ? { marker: overrideMarker } : {};
28610
29667
  } else if (defaultMarker && defaultMarker.primary) {
28611
29668
  const variantAlts = roleSpec.markerVariants?.[profile.code] ?? [];
@@ -28679,53 +29736,182 @@ var init_pattern_generator = __esm({
28679
29736
  }
28680
29737
  });
28681
29738
 
28682
- // src/patterns/toggle.ts
28683
- function getTogglePatternsBn() {
28684
- return [
28685
- // Full pattern: .active কে টগল করুন
28686
- {
28687
- id: "toggle-bn-full",
28688
- language: "bn",
28689
- command: "toggle",
28690
- priority: 100,
29739
+ // src/patterns/languages/en/fetch.ts
29740
+ var fetchWithResponseTypeEnglish, fetchWithOptionsAndResponseTypeEnglish, fetchWithOptionsEnglish, fetchSimpleEnglish, fetchPatternsEn;
29741
+ var init_fetch = __esm({
29742
+ "src/patterns/languages/en/fetch.ts"() {
29743
+ fetchWithResponseTypeEnglish = {
29744
+ id: "fetch-en-with-response-type",
29745
+ language: "en",
29746
+ command: "fetch",
29747
+ priority: 90,
29748
+ // Higher than simple pattern (80) to capture "as" modifier first
28691
29749
  template: {
28692
- format: "{patient} \u0995\u09C7 \u099F\u0997\u09B2 \u0995\u09B0\u09C1\u09A8",
29750
+ format: "fetch {source} as {responseType}",
28693
29751
  tokens: [
28694
- { type: "role", role: "patient" },
28695
- { type: "literal", value: "\u0995\u09C7" },
28696
- { type: "literal", value: "\u099F\u0997\u09B2", alternatives: ["\u09AA\u09B0\u09BF\u09AC\u09B0\u09CD\u09A4\u09A8"] },
28697
- { type: "literal", value: "\u0995\u09B0\u09C1\u09A8" }
29752
+ { type: "literal", value: "fetch" },
29753
+ { type: "role", role: "source", expectedTypes: ["literal", "expression"] },
29754
+ { type: "literal", value: "as" },
29755
+ // json/text/html are identifiers not keywords, so we need to accept 'expression' type
29756
+ { type: "role", role: "responseType", expectedTypes: ["literal", "expression"] }
28698
29757
  ]
28699
29758
  },
28700
29759
  extraction: {
28701
- patient: { position: 0 }
29760
+ source: { position: 1 },
29761
+ responseType: { marker: "as" }
28702
29762
  }
28703
- },
28704
- // Simple pattern: টগল .active
28705
- {
28706
- id: "toggle-bn-simple",
28707
- language: "bn",
28708
- command: "toggle",
28709
- priority: 90,
29763
+ };
29764
+ fetchWithOptionsAndResponseTypeEnglish = {
29765
+ id: "fetch-en-with-options-as",
29766
+ language: "en",
29767
+ command: "fetch",
29768
+ priority: 95,
28710
29769
  template: {
28711
- format: "\u099F\u0997\u09B2 {patient}",
29770
+ format: "fetch {source} with {style} as {responseType}",
28712
29771
  tokens: [
28713
- { type: "literal", value: "\u099F\u0997\u09B2", alternatives: ["\u09AA\u09B0\u09BF\u09AC\u09B0\u09CD\u09A4\u09A8"] },
28714
- { type: "role", role: "patient" }
29772
+ { type: "literal", value: "fetch" },
29773
+ { type: "role", role: "source", expectedTypes: ["literal", "expression"] },
29774
+ { type: "literal", value: "with", alternatives: ["by", "using"] },
29775
+ // expression-ONLY: routes `{ … }` to the object-literal fold, which keeps
29776
+ // the source text intact for the expression parser.
29777
+ { type: "role", role: "style", expectedTypes: ["expression"] },
29778
+ { type: "literal", value: "as" },
29779
+ { type: "role", role: "responseType", expectedTypes: ["literal", "expression"] }
28715
29780
  ]
28716
29781
  },
28717
29782
  extraction: {
28718
- patient: { position: 1 }
29783
+ source: { position: 1 },
29784
+ style: { marker: "with" },
29785
+ responseType: { marker: "as" }
28719
29786
  }
28720
- },
28721
- // With destination: #button এ .active কে টগল করুন
28722
- {
28723
- id: "toggle-bn-with-dest",
28724
- language: "bn",
28725
- command: "toggle",
28726
- priority: 95,
29787
+ };
29788
+ fetchWithOptionsEnglish = {
29789
+ id: "fetch-en-with-options",
29790
+ language: "en",
29791
+ command: "fetch",
29792
+ priority: 93,
29793
+ // Below the with+as pattern, above the response-type pattern (90)
28727
29794
  template: {
28728
- format: "{destination} \u098F {patient} \u0995\u09C7 \u099F\u0997\u09B2 \u0995\u09B0\u09C1\u09A8",
29795
+ format: "fetch {source} with {style}",
29796
+ tokens: [
29797
+ { type: "literal", value: "fetch" },
29798
+ { type: "role", role: "source", expectedTypes: ["literal", "expression"] },
29799
+ { type: "literal", value: "with", alternatives: ["by", "using"] },
29800
+ { type: "role", role: "style", expectedTypes: ["expression"] }
29801
+ ]
29802
+ },
29803
+ extraction: {
29804
+ source: { position: 1 },
29805
+ style: { marker: "with" }
29806
+ }
29807
+ };
29808
+ fetchSimpleEnglish = {
29809
+ id: "fetch-en-simple",
29810
+ language: "en",
29811
+ command: "fetch",
29812
+ priority: 80,
29813
+ // Lower than response type pattern (90) - fallback when "as" not present
29814
+ template: {
29815
+ format: "fetch {source}",
29816
+ tokens: [
29817
+ { type: "literal", value: "fetch" },
29818
+ { type: "role", role: "source" }
29819
+ ]
29820
+ },
29821
+ extraction: {
29822
+ source: { position: 1 }
29823
+ }
29824
+ };
29825
+ fetchPatternsEn = [
29826
+ fetchWithOptionsAndResponseTypeEnglish,
29827
+ fetchWithOptionsEnglish,
29828
+ fetchWithResponseTypeEnglish,
29829
+ fetchSimpleEnglish
29830
+ ];
29831
+ }
29832
+ });
29833
+
29834
+ // src/patterns/languages/en/pick.ts
29835
+ var pickVariantEnglish, pickPatternsEn;
29836
+ var init_pick = __esm({
29837
+ "src/patterns/languages/en/pick.ts"() {
29838
+ pickVariantEnglish = {
29839
+ id: "pick-en-variant",
29840
+ language: "en",
29841
+ command: "pick",
29842
+ priority: 110,
29843
+ template: {
29844
+ format: "pick {method} {patient} of {source}",
29845
+ tokens: [
29846
+ { type: "literal", value: "pick" },
29847
+ // Variant word: `characters`/`items`/`match` tokenize as identifiers
29848
+ // (expression), `first`/`last`/`random` as keywords.
29849
+ { type: "role", role: "method", expectedTypes: ["literal", "expression"] },
29850
+ // Range/count/index. The pick-range assembler folds `<a> to <b>
29851
+ // [inclusive|exclusive]` into one expression value here; a lone count
29852
+ // (`3`) is captured as a single literal.
29853
+ { type: "role", role: "patient", expectedTypes: ["literal", "expression"] },
29854
+ { type: "literal", value: "of", alternatives: ["from"] },
29855
+ { type: "role", role: "source", expectedTypes: ["selector", "reference", "expression"] }
29856
+ ]
29857
+ },
29858
+ extraction: {
29859
+ method: { position: 1 },
29860
+ patient: { position: 2 },
29861
+ source: { marker: "of", markerAlternatives: ["from"] }
29862
+ }
29863
+ };
29864
+ pickPatternsEn = [pickVariantEnglish];
29865
+ }
29866
+ });
29867
+
29868
+ // src/patterns/toggle.ts
29869
+ function getTogglePatternsBn() {
29870
+ return [
29871
+ // Full pattern: .active কে টগল করুন
29872
+ {
29873
+ id: "toggle-bn-full",
29874
+ language: "bn",
29875
+ command: "toggle",
29876
+ priority: 100,
29877
+ template: {
29878
+ format: "{patient} \u0995\u09C7 \u099F\u0997\u09B2 \u0995\u09B0\u09C1\u09A8",
29879
+ tokens: [
29880
+ { type: "role", role: "patient" },
29881
+ { type: "literal", value: "\u0995\u09C7" },
29882
+ { type: "literal", value: "\u099F\u0997\u09B2", alternatives: ["\u09AA\u09B0\u09BF\u09AC\u09B0\u09CD\u09A4\u09A8"] },
29883
+ { type: "literal", value: "\u0995\u09B0\u09C1\u09A8" }
29884
+ ]
29885
+ },
29886
+ extraction: {
29887
+ patient: { position: 0 }
29888
+ }
29889
+ },
29890
+ // Simple pattern: টগল .active
29891
+ {
29892
+ id: "toggle-bn-simple",
29893
+ language: "bn",
29894
+ command: "toggle",
29895
+ priority: 90,
29896
+ template: {
29897
+ format: "\u099F\u0997\u09B2 {patient}",
29898
+ tokens: [
29899
+ { type: "literal", value: "\u099F\u0997\u09B2", alternatives: ["\u09AA\u09B0\u09BF\u09AC\u09B0\u09CD\u09A4\u09A8"] },
29900
+ { type: "role", role: "patient" }
29901
+ ]
29902
+ },
29903
+ extraction: {
29904
+ patient: { position: 1 }
29905
+ }
29906
+ },
29907
+ // With destination: #button এ .active কে টগল করুন
29908
+ {
29909
+ id: "toggle-bn-with-dest",
29910
+ language: "bn",
29911
+ command: "toggle",
29912
+ priority: 95,
29913
+ template: {
29914
+ format: "{destination} \u098F {patient} \u0995\u09C7 \u099F\u0997\u09B2 \u0995\u09B0\u09C1\u09A8",
28729
29915
  tokens: [
28730
29916
  { type: "role", role: "destination" },
28731
29917
  { type: "literal", value: "\u098F", alternatives: ["\u09A4\u09C7"] },
@@ -29099,6 +30285,33 @@ function getTogglePatternsQu() {
29099
30285
  destination: { position: 0 },
29100
30286
  patient: { position: 2 }
29101
30287
  }
30288
+ },
30289
+ // Patient-first with trailing destination: .open ta qhipantin .panel man
30290
+ // t'ikray — the i18n full verb-final order (#636 qu canonicalOrder) puts
30291
+ // the destination AFTER the patient, but every dest-bearing variant above
30292
+ // is destination-first, so the shape fell to the verb-anchoring fallback,
30293
+ // which glued the positional run (destination:literal="qhipantin.panel"
30294
+ // vs en destination:expression="next .panel") — toggle-aria-expanded,
30295
+ // R1 deferred-tail qu tail.
30296
+ {
30297
+ id: "toggle-qu-patient-first-dest",
30298
+ language: "qu",
30299
+ command: "toggle",
30300
+ priority: 102,
30301
+ template: {
30302
+ format: "{patient} ta {destination} man t'ikray",
30303
+ tokens: [
30304
+ { type: "role", role: "patient" },
30305
+ { type: "literal", value: "ta" },
30306
+ { type: "role", role: "destination" },
30307
+ { type: "literal", value: "man", alternatives: ["pa"] },
30308
+ { type: "literal", value: "t'ikray", alternatives: ["tikray", "kutichiy"] }
30309
+ ]
30310
+ },
30311
+ extraction: {
30312
+ patient: { position: 0 },
30313
+ destination: { position: 2 }
30314
+ }
29102
30315
  }
29103
30316
  ];
29104
30317
  }
@@ -29466,11 +30679,15 @@ function repeatForInHead(language, spec) {
29466
30679
  // matches the verb's normalized form
29467
30680
  ];
29468
30681
  if (spec.forWords && spec.forWords.length > 0) {
29469
- tokens.push({
29470
- type: "group",
29471
- optional: true,
29472
- tokens: spec.forWords.map((w) => ({ type: "literal", value: w }))
29473
- });
30682
+ if (spec.requireForWords) {
30683
+ for (const w of spec.forWords) tokens.push({ type: "literal", value: w });
30684
+ } else {
30685
+ tokens.push({
30686
+ type: "group",
30687
+ optional: true,
30688
+ tokens: spec.forWords.map((w) => ({ type: "literal", value: w }))
30689
+ });
30690
+ }
29474
30691
  }
29475
30692
  tokens.push({ type: "role", role: "patient", expectedTypes: ["expression", "reference"] });
29476
30693
  for (const w of spec.inWords) tokens.push({ type: "literal", value: w });
@@ -29579,10 +30796,63 @@ function repeatUntilHeadSOV(language, spec) {
29579
30796
  }
29580
30797
  };
29581
30798
  }
30799
+ function repeatUntilHeadSOVVerbFinal(language, spec) {
30800
+ return {
30801
+ id: `repeat-${language}-until-head-verb-final`,
30802
+ language,
30803
+ command: "repeat",
30804
+ priority: 111,
30805
+ // above the post-verb variant so the correct shape wins
30806
+ template: {
30807
+ format: `${spec.untilWord} ${spec.eventWord} {event} ${spec.objMarker} {source} ${spec.fromWord} repeat`,
30808
+ tokens: [
30809
+ { type: "literal", value: spec.untilWord },
30810
+ { type: "literal", value: spec.eventWord },
30811
+ { type: "role", role: "event", expectedTypes: ["literal", "expression"] },
30812
+ { type: "literal", value: spec.objMarker },
30813
+ {
30814
+ type: "role",
30815
+ role: "source",
30816
+ expectedTypes: ["selector", "reference", "expression"]
30817
+ },
30818
+ { type: "literal", value: spec.fromWord },
30819
+ { type: "literal", value: "repeat" }
30820
+ ]
30821
+ },
30822
+ extraction: {
30823
+ loopType: { default: { type: "literal", value: "until-event" } }
30824
+ }
30825
+ };
30826
+ }
30827
+ function sovForBindingHead(language, spec) {
30828
+ return {
30829
+ id: `for-${language}-sov-basic`,
30830
+ language,
30831
+ command: "for",
30832
+ priority: 105,
30833
+ template: {
30834
+ format: `{patient} ${spec.inWords.join(" ")} {source} [${spec.objMarker}] ${spec.forVerb}`,
30835
+ tokens: [
30836
+ { type: "role", role: "patient", expectedTypes: ["expression", "reference"] },
30837
+ ...spec.inWords.map((w) => ({ type: "literal", value: w })),
30838
+ { type: "role", role: "source", expectedTypes: ["selector", "expression", "reference"] },
30839
+ {
30840
+ type: "group",
30841
+ optional: true,
30842
+ tokens: [{ type: "literal", value: spec.objMarker }]
30843
+ },
30844
+ { type: "literal", value: spec.forVerb }
30845
+ ]
30846
+ },
30847
+ extraction: {
30848
+ patient: { position: 0 }
30849
+ }
30850
+ };
30851
+ }
29582
30852
  function getRepeatPatternsForLanguage(language) {
29583
30853
  return BY_LANG.get(language) ?? [];
29584
30854
  }
29585
- var VERB_FIRST_REPEAT_TIMES, SOV_REPEAT_TIMES, FOR_IN_HEADS, WHILE_HEADS, VERB_FIRST_UNTIL_HEADS, repeatUntilHeadQuMidClause, SOV_UNTIL_HEADS, repeatUntilHeadQu, BY_LANG, addPattern;
30855
+ var VERB_FIRST_REPEAT_TIMES, SOV_REPEAT_TIMES, FOR_IN_HEADS, WHILE_HEADS, VERB_FIRST_UNTIL_HEADS, repeatUntilHeadQuMidClause, SOV_UNTIL_HEADS, repeatUntilHeadQu, SOV_FOR_BINDING_HEADS, BY_LANG, addPattern;
29586
30856
  var init_repeat = __esm({
29587
30857
  "src/patterns/repeat.ts"() {
29588
30858
  VERB_FIRST_REPEAT_TIMES = [
@@ -29597,7 +30867,7 @@ var init_repeat = __esm({
29597
30867
  ["ar", "\u0643\u0631\u0631", "times"],
29598
30868
  ["he", "\u05D7\u05D6\u05D5\u05E8", "times", "\u05D0\u05EA"],
29599
30869
  ["id", "ulangi", "times"],
29600
- ["ms", "ulang", "times"],
30870
+ ["ms", "ulang", "kali"],
29601
30871
  ["sw", "rudia", "times"],
29602
30872
  ["th", "\u0E17\u0E33\u0E0B\u0E49\u0E33", "\u0E04\u0E23\u0E31\u0E49\u0E07"],
29603
30873
  ["vi", "l\u1EB7p l\u1EA1i", "l\u1EA7n"],
@@ -29613,7 +30883,7 @@ var init_repeat = __esm({
29613
30883
  ["qu", "times", "ta"]
29614
30884
  ];
29615
30885
  FOR_IN_HEADS = [
29616
- ["en", { forWords: ["for"], inWords: ["in"] }],
30886
+ ["en", { forWords: ["for"], inWords: ["in"], requireForWords: true }],
29617
30887
  ["es", { forWords: ["para"], inWords: ["en"] }],
29618
30888
  ["pt", { forWords: ["para"], inWords: ["dentro"] }],
29619
30889
  ["fr", { forWords: ["pour"], inWords: ["en"] }],
@@ -29627,8 +30897,11 @@ var init_repeat = __esm({
29627
30897
  ["he", { forWords: ["\u05E2\u05D1\u05D5\u05E8", "\u05D0\u05EA"], inWords: ["in"] }],
29628
30898
  ["hi", { inWords: ["\u092E\u0947\u0902"] }],
29629
30899
  ["bn", { inWords: ["\u098F"] }],
29630
- ["ja", { inWords: ["\u306E", "\u4E2D"] }],
29631
- ["ko", { inWords: ["\uC548", "\uC5D0"] }],
30900
+ // ja/ko/qu containment words tokenize WHOLE (keyword→in entries added for
30901
+ // the focus-trap Family G operand run) — the old split forms (の+中, 안+에,
30902
+ // uku+pi) no longer appear in the stream.
30903
+ ["ja", { inWords: ["\u306E\u4E2D"] }],
30904
+ ["ko", { inWords: ["\uC548\uC5D0"] }],
29632
30905
  ["zh", { forWords: ["\u4E3A", "\u628A"], inWords: ["\u5728"] }],
29633
30906
  ["tr", { inWords: ["i\xE7inde"] }],
29634
30907
  ["id", { forWords: ["untuk"], inWords: ["dalam"] }],
@@ -29637,7 +30910,7 @@ var init_repeat = __esm({
29637
30910
  ["th", { forWords: ["\u0E2A\u0E33\u0E2B\u0E23\u0E31\u0E1A"], inWords: ["\u0E43\u0E19"] }],
29638
30911
  ["vi", { forWords: ["v\u1EDBi m\u1ED7i"], inWords: ["trong"] }],
29639
30912
  ["tl", { forWords: ["para_sa"], inWords: ["sa_loob"] }],
29640
- ["qu", { inWords: ["uku", "pi"] }]
30913
+ ["qu", { inWords: ["ukupi"] }]
29641
30914
  ];
29642
30915
  WHILE_HEADS = [
29643
30916
  ["en", { whileWord: "while" }],
@@ -29733,6 +31006,16 @@ var init_repeat = __esm({
29733
31006
  loopType: { default: { type: "literal", value: "until-event" } }
29734
31007
  }
29735
31008
  };
31009
+ SOV_FOR_BINDING_HEADS = [
31010
+ // ja/ko/qu in-words are single whole tokens now (keyword→in entries — see
31011
+ // the FOR_IN_HEADS note); the split forms are gone from the stream.
31012
+ ["ja", { inWords: ["\u306E\u4E2D"], objMarker: "\u3092", forVerb: "\u305F\u3081\u306B" }],
31013
+ ["ko", { inWords: ["\uC548\uC5D0"], objMarker: "\uB97C", forVerb: "\uAC01\uAC01" }],
31014
+ ["tr", { inWords: ["i\xE7inde"], objMarker: "i", forVerb: "i\xE7in" }],
31015
+ ["qu", { inWords: ["ukupi"], objMarker: "ta", forVerb: "sapankaq" }],
31016
+ ["bn", { inWords: ["\u098F"], objMarker: "\u0995\u09C7", forVerb: "\u099C\u09A8\u09CD\u09AF" }],
31017
+ ["hi", { inWords: ["\u092E\u0947\u0902"], objMarker: "\u0915\u094B", forVerb: "\u0939\u0947\u0924\u0941" }]
31018
+ ];
29736
31019
  BY_LANG = /* @__PURE__ */ new Map();
29737
31020
  addPattern = (lang, p) => {
29738
31021
  const list = BY_LANG.get(lang);
@@ -29748,6 +31031,9 @@ var init_repeat = __esm({
29748
31031
  for (const [lang, spec] of FOR_IN_HEADS) {
29749
31032
  addPattern(lang, repeatForInHead(lang, spec));
29750
31033
  }
31034
+ for (const [lang, spec] of SOV_FOR_BINDING_HEADS) {
31035
+ addPattern(lang, sovForBindingHead(lang, spec));
31036
+ }
29751
31037
  for (const [lang, spec] of WHILE_HEADS) {
29752
31038
  addPattern(lang, repeatWhileHead(lang, spec));
29753
31039
  }
@@ -29756,6 +31042,9 @@ var init_repeat = __esm({
29756
31042
  }
29757
31043
  for (const [lang, spec] of SOV_UNTIL_HEADS) {
29758
31044
  addPattern(lang, repeatUntilHeadSOV(lang, spec));
31045
+ if (lang === "tr") {
31046
+ addPattern(lang, repeatUntilHeadSOVVerbFinal(lang, spec));
31047
+ }
29759
31048
  }
29760
31049
  addPattern("qu", repeatUntilHeadQu);
29761
31050
  addPattern("qu", repeatUntilHeadQuMidClause);
@@ -29875,6 +31164,121 @@ function getWaitPatternsTl() {
29875
31164
  }
29876
31165
  ];
29877
31166
  }
31167
+ function verbFinalOrRunWait(id, language, verb, sourceMarker, orWord, parenArgCount, sourceMarkerAlternatives) {
31168
+ const parenGroup = () => ({
31169
+ type: "group",
31170
+ optional: true,
31171
+ tokens: [
31172
+ { type: "literal", value: "(" },
31173
+ ...Array.from({ length: parenArgCount }, (_, i) => [
31174
+ ...i > 0 ? [{ type: "literal", value: "," }] : [],
31175
+ {
31176
+ type: "role",
31177
+ role: "condition",
31178
+ expectedTypes: ["expression", "literal", "reference"]
31179
+ }
31180
+ ]).flat(),
31181
+ { type: "literal", value: ")" }
31182
+ ]
31183
+ });
31184
+ return {
31185
+ id,
31186
+ language,
31187
+ command: "wait",
31188
+ priority: 105,
31189
+ template: {
31190
+ format: `{source} ${sourceMarker} {duration} ${orWord} {patient} ${verb}`,
31191
+ tokens: [
31192
+ { type: "role", role: "source", expectedTypes: ["expression", "reference"] },
31193
+ {
31194
+ type: "literal",
31195
+ value: sourceMarker,
31196
+ ...sourceMarkerAlternatives ? { alternatives: sourceMarkerAlternatives } : {}
31197
+ },
31198
+ { type: "role", role: "duration", expectedTypes: ["expression", "literal"] },
31199
+ parenGroup(),
31200
+ { type: "literal", value: orWord },
31201
+ { type: "role", role: "patient", expectedTypes: ["expression", "literal"] },
31202
+ parenGroup(),
31203
+ { type: "literal", value: verb }
31204
+ ]
31205
+ },
31206
+ extraction: {
31207
+ source: { position: 0 },
31208
+ duration: { position: 2 }
31209
+ }
31210
+ };
31211
+ }
31212
+ function verbFirstOrRunWait(id, language, verb, orWord, forWord, sourceMarker, parenArgCount) {
31213
+ const parenGroup = () => ({
31214
+ type: "group",
31215
+ optional: true,
31216
+ tokens: [
31217
+ { type: "literal", value: "(" },
31218
+ ...Array.from({ length: parenArgCount }, (_, i) => [
31219
+ ...i > 0 ? [{ type: "literal", value: "," }] : [],
31220
+ {
31221
+ type: "role",
31222
+ role: "condition",
31223
+ expectedTypes: ["expression", "literal", "reference"]
31224
+ }
31225
+ ]).flat(),
31226
+ { type: "literal", value: ")" }
31227
+ ]
31228
+ });
31229
+ const forGroup = () => ({
31230
+ type: "group",
31231
+ optional: true,
31232
+ tokens: [{ type: "literal", value: forWord }]
31233
+ });
31234
+ return {
31235
+ id,
31236
+ language,
31237
+ command: "wait",
31238
+ priority: 105,
31239
+ template: {
31240
+ format: `${verb} {duration} ${orWord} [${forWord}] {patient} [${forWord}] {source} ${sourceMarker}`,
31241
+ tokens: [
31242
+ { type: "literal", value: verb },
31243
+ { type: "role", role: "duration", expectedTypes: ["expression", "literal"] },
31244
+ parenGroup(),
31245
+ { type: "literal", value: orWord },
31246
+ forGroup(),
31247
+ { type: "role", role: "patient", expectedTypes: ["expression", "literal"] },
31248
+ parenGroup(),
31249
+ forGroup(),
31250
+ { type: "role", role: "source", expectedTypes: ["expression", "reference"] },
31251
+ { type: "literal", value: sourceMarker }
31252
+ ]
31253
+ },
31254
+ extraction: {
31255
+ duration: { position: 1 },
31256
+ source: { position: 8 }
31257
+ }
31258
+ };
31259
+ }
31260
+ function getWaitPatternsBn() {
31261
+ return [
31262
+ verbFirstOrRunWait("wait-bn-or-run", "bn", "\u0985\u09AA\u09C7\u0995\u09CD\u09B7\u09BE", "\u0985\u09A5\u09AC\u09BE", "\u099C\u09A8\u09CD\u09AF", "\u09A5\u09C7\u0995\u09C7", 1),
31263
+ verbFirstOrRunWait("wait-bn-or-run-2arg", "bn", "\u0985\u09AA\u09C7\u0995\u09CD\u09B7\u09BE", "\u0985\u09A5\u09AC\u09BE", "\u099C\u09A8\u09CD\u09AF", "\u09A5\u09C7\u0995\u09C7", 2)
31264
+ ];
31265
+ }
31266
+ function getWaitPatternsTr() {
31267
+ return [
31268
+ verbFinalOrRunWait("wait-tr-or-run", "tr", "bekle", "den", "veya", 1, ["dan", "ten", "tan"]),
31269
+ verbFinalOrRunWait("wait-tr-or-run-2arg", "tr", "bekle", "den", "veya", 2, [
31270
+ "dan",
31271
+ "ten",
31272
+ "tan"
31273
+ ])
31274
+ ];
31275
+ }
31276
+ function getWaitPatternsQu() {
31277
+ return [
31278
+ verbFinalOrRunWait("wait-qu-or-run", "qu", "suyay", "manta", "utaq", 1),
31279
+ verbFinalOrRunWait("wait-qu-or-run-2arg", "qu", "suyay", "manta", "utaq", 2)
31280
+ ];
31281
+ }
29878
31282
  function getWaitPatternsForLanguage(language) {
29879
31283
  switch (language) {
29880
31284
  case "en":
@@ -29885,8 +31289,14 @@ function getWaitPatternsForLanguage(language) {
29885
31289
  return getWaitPatternsHe();
29886
31290
  case "ar":
29887
31291
  return getWaitPatternsAr();
31292
+ case "bn":
31293
+ return getWaitPatternsBn();
29888
31294
  case "tl":
29889
31295
  return getWaitPatternsTl();
31296
+ case "tr":
31297
+ return getWaitPatternsTr();
31298
+ case "qu":
31299
+ return getWaitPatternsQu();
29890
31300
  default:
29891
31301
  return [];
29892
31302
  }
@@ -29909,8 +31319,8 @@ function buildEnglishPatterns() {
29909
31319
  patterns.push(...getRepeatPatternsForLanguage("en"));
29910
31320
  patterns.push(...getWaitPatternsForLanguage("en"));
29911
31321
  patterns.push(
29912
- fetchWithResponseTypeEnglish,
29913
- fetchSimpleEnglish,
31322
+ ...fetchPatternsEn,
31323
+ ...pickPatternsEn,
29914
31324
  swapElementEnglish,
29915
31325
  swapSimpleEnglish,
29916
31326
  repeatUntilEventFromEnglish,
@@ -29928,51 +31338,18 @@ function buildEnglishPatterns() {
29928
31338
  patterns.push(...generatedPatterns);
29929
31339
  return patterns;
29930
31340
  }
29931
- var fetchWithResponseTypeEnglish, fetchSimpleEnglish, swapSimpleEnglish, swapElementEnglish, repeatUntilEventFromEnglish, repeatUntilEventEnglish, repeatTimesEnglish, repeatForeverEnglish, setPossessiveEnglish, forEnglish, ifEnglish, unlessEnglish, temporalInEnglish, temporalAfterEnglish;
31341
+ var swapSimpleEnglish, swapElementEnglish, repeatUntilEventFromEnglish, repeatUntilEventEnglish, repeatTimesEnglish, repeatForeverEnglish, setPossessiveEnglish, forEnglish, ifEnglish, unlessEnglish, temporalInEnglish, temporalAfterEnglish;
29932
31342
  var init_en = __esm({
29933
31343
  "src/patterns/en.ts"() {
29934
31344
  init_english();
29935
31345
  init_pattern_generator();
31346
+ init_fetch();
31347
+ init_pick();
29936
31348
  init_toggle();
29937
31349
  init_put();
29938
31350
  init_event_handler();
29939
31351
  init_repeat();
29940
31352
  init_wait();
29941
- fetchWithResponseTypeEnglish = {
29942
- id: "fetch-en-with-response-type",
29943
- language: "en",
29944
- command: "fetch",
29945
- priority: 90,
29946
- template: {
29947
- format: "fetch {source} as {responseType}",
29948
- tokens: [
29949
- { type: "literal", value: "fetch" },
29950
- { type: "role", role: "source", expectedTypes: ["literal", "expression"] },
29951
- { type: "literal", value: "as" },
29952
- { type: "role", role: "responseType", expectedTypes: ["literal", "expression"] }
29953
- ]
29954
- },
29955
- extraction: {
29956
- source: { position: 1 },
29957
- responseType: { marker: "as" }
29958
- }
29959
- };
29960
- fetchSimpleEnglish = {
29961
- id: "fetch-en-simple",
29962
- language: "en",
29963
- command: "fetch",
29964
- priority: 80,
29965
- template: {
29966
- format: "fetch {source}",
29967
- tokens: [
29968
- { type: "literal", value: "fetch" },
29969
- { type: "role", role: "source" }
29970
- ]
29971
- },
29972
- extraction: {
29973
- source: { position: 1 }
29974
- }
29975
- };
29976
31353
  swapSimpleEnglish = {
29977
31354
  id: "swap-en-handcrafted",
29978
31355
  language: "en",
@@ -30244,6 +31621,15 @@ init_chinese();
30244
31621
  // src/parser/pattern-matcher.ts
30245
31622
  init_command_schemas();
30246
31623
 
31624
+ // src/parser/utils/possessive-keywords.ts
31625
+ init_english();
31626
+
31627
+ // src/parser/utils/expression-lexicon.ts
31628
+ init_command_schemas();
31629
+ new Set(
31630
+ Object.keys(commandSchemas).map((a) => a.toLowerCase())
31631
+ );
31632
+
30247
31633
  // src/parser/pattern-matcher.ts
30248
31634
  init_registry();
30249
31635
  init_put();
@@ -30252,15 +31638,6 @@ init_put();
30252
31638
  new Set(
30253
31639
  Object.values(commandSchemas).filter((s) => s.bareKeyword === true).map((s) => s.action)
30254
31640
  );
30255
- /**
30256
- * Normalized command-action keywords (the schema registry's action names).
30257
- * Tokenizers normalize every language's command verbs to these forms, so the
30258
- * set is language-independent. Used to keep the positional source clause
30259
- * from consuming a following command's verb as a locative marker.
30260
- */
30261
- new Set(
30262
- Object.keys(commandSchemas).map((a) => a.toLowerCase())
30263
- );
30264
31641
 
30265
31642
  // src/tokenizers/index.ts
30266
31643
  init_registry();
@@ -30619,6 +31996,231 @@ init_wait();
30619
31996
  // src/patterns/builders.ts
30620
31997
  init_repeat();
30621
31998
 
31999
+ // src/patterns/languages/en/index.ts
32000
+ init_fetch();
32001
+
32002
+ // src/patterns/languages/en/swap.ts
32003
+ var swapSimpleEnglish2 = {
32004
+ id: "swap-en-handcrafted",
32005
+ language: "en",
32006
+ command: "swap",
32007
+ priority: 110,
32008
+ // Higher than generated patterns
32009
+ template: {
32010
+ format: "swap {method} {destination}",
32011
+ tokens: [
32012
+ { type: "literal", value: "swap" },
32013
+ { type: "role", role: "method" },
32014
+ { type: "role", role: "destination" }
32015
+ ]
32016
+ },
32017
+ extraction: {
32018
+ method: { position: 1 },
32019
+ destination: { position: 2 }
32020
+ }
32021
+ };
32022
+ var swapElementEnglish2 = {
32023
+ id: "swap-en-element",
32024
+ language: "en",
32025
+ command: "swap",
32026
+ priority: 120,
32027
+ template: {
32028
+ format: "swap {destination} with {patient}",
32029
+ tokens: [
32030
+ { type: "literal", value: "swap" },
32031
+ { type: "role", role: "destination" },
32032
+ { type: "literal", value: "with" },
32033
+ { type: "role", role: "patient" }
32034
+ ]
32035
+ },
32036
+ extraction: {}
32037
+ };
32038
+ var swapPatternsEn = [swapElementEnglish2, swapSimpleEnglish2];
32039
+
32040
+ // src/patterns/languages/en/repeat.ts
32041
+ var repeatUntilEventFromEnglish2 = {
32042
+ id: "repeat-en-until-event-from",
32043
+ language: "en",
32044
+ command: "repeat",
32045
+ priority: 120,
32046
+ // Highest priority - most specific pattern
32047
+ template: {
32048
+ format: "repeat until event {event} from {source}",
32049
+ tokens: [
32050
+ { type: "literal", value: "repeat" },
32051
+ { type: "literal", value: "until" },
32052
+ { type: "literal", value: "event" },
32053
+ { type: "role", role: "event", expectedTypes: ["literal", "expression"] },
32054
+ { type: "literal", value: "from" },
32055
+ { type: "role", role: "source", expectedTypes: ["selector", "reference", "expression"] }
32056
+ ]
32057
+ },
32058
+ extraction: {
32059
+ event: { marker: "event" },
32060
+ source: { marker: "from" },
32061
+ loopType: { default: { type: "literal", value: "until-event" } }
32062
+ }
32063
+ };
32064
+ var repeatUntilEventEnglish2 = {
32065
+ id: "repeat-en-until-event",
32066
+ language: "en",
32067
+ command: "repeat",
32068
+ priority: 110,
32069
+ // Lower than "from" variant, but higher than quantity-based repeat
32070
+ template: {
32071
+ format: "repeat until event {event}",
32072
+ tokens: [
32073
+ { type: "literal", value: "repeat" },
32074
+ { type: "literal", value: "until" },
32075
+ { type: "literal", value: "event" },
32076
+ { type: "role", role: "event", expectedTypes: ["literal", "expression"] }
32077
+ ]
32078
+ },
32079
+ extraction: {
32080
+ event: { marker: "event" },
32081
+ loopType: { default: { type: "literal", value: "until-event" } }
32082
+ }
32083
+ };
32084
+ var repeatPatternsEn = [
32085
+ repeatUntilEventFromEnglish2,
32086
+ repeatUntilEventEnglish2
32087
+ ];
32088
+
32089
+ // src/patterns/languages/en/set.ts
32090
+ var setPossessiveEnglish2 = {
32091
+ id: "set-en-possessive",
32092
+ language: "en",
32093
+ command: "set",
32094
+ priority: 100,
32095
+ // Higher than generated setSchema (80)
32096
+ template: {
32097
+ format: "set {destination} to {patient}",
32098
+ tokens: [
32099
+ { type: "literal", value: "set" },
32100
+ // Role token with property-path support for possessive syntax
32101
+ {
32102
+ type: "role",
32103
+ role: "destination",
32104
+ expectedTypes: ["property-path", "selector", "reference", "expression"]
32105
+ },
32106
+ { type: "literal", value: "to" },
32107
+ { type: "role", role: "patient", expectedTypes: ["literal", "expression", "reference"] }
32108
+ ]
32109
+ },
32110
+ extraction: {
32111
+ destination: { position: 1 },
32112
+ patient: { marker: "to" }
32113
+ }
32114
+ };
32115
+ var setPatternsEn = [setPossessiveEnglish2];
32116
+
32117
+ // src/patterns/languages/en/control-flow.ts
32118
+ var forEnglish2 = {
32119
+ id: "for-en-basic",
32120
+ language: "en",
32121
+ command: "for",
32122
+ priority: 100,
32123
+ template: {
32124
+ format: "for {patient} in {source}",
32125
+ tokens: [
32126
+ { type: "literal", value: "for" },
32127
+ { type: "role", role: "patient", expectedTypes: ["expression", "reference"] },
32128
+ // Loop variable
32129
+ { type: "literal", value: "in" },
32130
+ { type: "role", role: "source", expectedTypes: ["selector", "expression", "reference"] }
32131
+ // Collection
32132
+ ]
32133
+ },
32134
+ extraction: {
32135
+ patient: { position: 1 },
32136
+ source: { marker: "in" }
32137
+ // NOTE: no `loopType` default — see the rationale in patterns/en.ts
32138
+ // `forEnglish` (the `for` schema has no loopType role; a `loopType:literal="for"`
32139
+ // here only duplicates the action name and is the R1 outlier no translation
32140
+ // reproduces). R2-safe (forMapper reads only patient+source). Kept in sync.
32141
+ }
32142
+ };
32143
+ var ifEnglish2 = {
32144
+ id: "if-en-basic",
32145
+ language: "en",
32146
+ command: "if",
32147
+ priority: 100,
32148
+ template: {
32149
+ format: "if {condition}",
32150
+ tokens: [
32151
+ { type: "literal", value: "if" },
32152
+ { type: "role", role: "condition", expectedTypes: ["expression", "reference", "selector"] }
32153
+ ]
32154
+ },
32155
+ extraction: {
32156
+ condition: { position: 1 }
32157
+ }
32158
+ };
32159
+ var unlessEnglish2 = {
32160
+ id: "unless-en-basic",
32161
+ language: "en",
32162
+ command: "unless",
32163
+ priority: 100,
32164
+ template: {
32165
+ format: "unless {condition}",
32166
+ tokens: [
32167
+ { type: "literal", value: "unless" },
32168
+ { type: "role", role: "condition", expectedTypes: ["expression", "reference", "selector"] }
32169
+ ]
32170
+ },
32171
+ extraction: {
32172
+ condition: { position: 1 }
32173
+ }
32174
+ };
32175
+ var controlFlowPatternsEn = [forEnglish2, ifEnglish2, unlessEnglish2];
32176
+
32177
+ // src/patterns/languages/en/temporal.ts
32178
+ var temporalInEnglish2 = {
32179
+ id: "temporal-en-in",
32180
+ language: "en",
32181
+ command: "wait",
32182
+ priority: 95,
32183
+ // Lower than standard wait patterns
32184
+ template: {
32185
+ format: "in {duration}",
32186
+ tokens: [
32187
+ { type: "literal", value: "in" },
32188
+ { type: "role", role: "duration", expectedTypes: ["literal", "expression"] }
32189
+ ]
32190
+ },
32191
+ extraction: {
32192
+ duration: { position: 1 }
32193
+ }
32194
+ };
32195
+ var temporalAfterEnglish2 = {
32196
+ id: "temporal-en-after",
32197
+ language: "en",
32198
+ command: "wait",
32199
+ priority: 95,
32200
+ // Lower than standard wait patterns
32201
+ template: {
32202
+ format: "after {duration}",
32203
+ tokens: [
32204
+ { type: "literal", value: "after" },
32205
+ { type: "role", role: "duration", expectedTypes: ["literal", "expression"] }
32206
+ ]
32207
+ },
32208
+ extraction: {
32209
+ duration: { position: 1 }
32210
+ }
32211
+ };
32212
+ var temporalPatternsEn = [temporalInEnglish2, temporalAfterEnglish2];
32213
+
32214
+ // src/patterns/languages/en/index.ts
32215
+ [
32216
+ ...fetchPatternsEn,
32217
+ ...swapPatternsEn,
32218
+ ...repeatPatternsEn,
32219
+ ...setPatternsEn,
32220
+ ...controlFlowPatternsEn,
32221
+ ...temporalPatternsEn
32222
+ ];
32223
+
30622
32224
  // src/patterns/builders.ts
30623
32225
  init_pattern_generator();
30624
32226
  init_registry();
@@ -31023,6 +32625,81 @@ function inferRoles(name, args, modifiers, target) {
31023
32625
  }
31024
32626
  break;
31025
32627
  }
32628
+ case 'go': {
32629
+ const kw = (n) => {
32630
+ if (!n || typeof n !== 'object')
32631
+ return undefined;
32632
+ const v = n;
32633
+ if (v.type === 'identifier') {
32634
+ if (typeof v.name === 'string' && v.name !== '')
32635
+ return v.name;
32636
+ return typeof v.value === 'string' ? v.value : undefined;
32637
+ }
32638
+ if (v.type === 'literal' && typeof v.value === 'string')
32639
+ return v.value;
32640
+ return undefined;
32641
+ };
32642
+ const asNode = (x) => x && typeof x === 'object' && 'type' in x ? x : undefined;
32643
+ let destination;
32644
+ let method;
32645
+ const onMod = asNode(modifiers?.on);
32646
+ if (args.length === 0 && onMod) {
32647
+ destination = onMod;
32648
+ if (kw(asNode(modifiers?.method)) === 'url') {
32649
+ method = { type: 'literal', value: 'url' };
32650
+ }
32651
+ }
32652
+ else {
32653
+ const words = args.map(kw);
32654
+ const urlIdx = words.indexOf('url');
32655
+ if (urlIdx !== -1 && args[urlIdx + 1]) {
32656
+ destination = args[urlIdx + 1];
32657
+ method = { type: 'literal', value: 'url' };
32658
+ }
32659
+ else {
32660
+ const SKIP = new Set(['to', 'the']);
32661
+ const POSITION = new Set([
32662
+ 'top',
32663
+ 'middle',
32664
+ 'bottom',
32665
+ 'left',
32666
+ 'center',
32667
+ 'right',
32668
+ 'smoothly',
32669
+ 'instantly',
32670
+ 'in',
32671
+ 'new',
32672
+ 'window',
32673
+ ]);
32674
+ const headIdx = args.findIndex((_, i) => {
32675
+ const w = words[i];
32676
+ return w === undefined || !SKIP.has(w);
32677
+ });
32678
+ const headWord = headIdx !== -1 ? words[headIdx] : undefined;
32679
+ const ofIdx = words.indexOf('of');
32680
+ if (headWord === 'back' || headWord === 'forward') {
32681
+ destination = { type: 'identifier', value: headWord, name: headWord };
32682
+ }
32683
+ else if (ofIdx !== -1 && args[ofIdx + 1]) {
32684
+ destination = kw(args[ofIdx + 1]) === 'the' ? args[ofIdx + 2] : args[ofIdx + 1];
32685
+ }
32686
+ else if (headIdx !== -1 && !POSITION.has(headWord ?? '')) {
32687
+ destination = args[headIdx];
32688
+ }
32689
+ }
32690
+ }
32691
+ const destWord = kw(destination);
32692
+ if ((destWord === 'back' || destWord === 'forward') && destination?.type !== 'identifier') {
32693
+ destination = { type: 'identifier', value: destWord, name: destWord };
32694
+ }
32695
+ if (!destination && target)
32696
+ destination = target;
32697
+ if (destination)
32698
+ roles.destination = destination;
32699
+ if (method)
32700
+ roles.method = method;
32701
+ break;
32702
+ }
31026
32703
  default: {
31027
32704
  const schema = getSchema(name);
31028
32705
  if (!schema)