nomen-lang 0.6.1 → 0.6.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -61,6 +61,18 @@ pub struct List<T>: Viewable {
61
61
  self.items.replace(i, value)
62
62
  }
63
63
 
64
+ // Bounds-checked write that traps when i is outside [0, self.length) —
65
+ // for indices that can't be proven at compile time (cross-module calls
66
+ // can never discharge `set`'s parameter constraint) and out-of-range is
67
+ // a bug, not a case.
68
+ pub func set_or_panic = (ref self, int i, move T value) {
69
+ if i >= 0 && i < self.length {
70
+ self.set(i, value)
71
+ return
72
+ }
73
+ panic("index out of range")
74
+ }
75
+
64
76
  // A deep copy: a fresh List<T> with an independent copy of every element.
65
77
  // For owning element types (strings, owning value structs) each slot in the
66
78
  // copy owns its own heap data (Buffer deep-copies on store), so the source
@@ -16,7 +16,9 @@ pub struct StringBuilder {
16
16
  // Ensure capacity for at least `needed` bytes. Grows geometrically.
17
17
  // Plain Nomen + the memory externs — no pointer arithmetic needed, so
18
18
  // this doesn't even require an unsafe context.
19
- func ensure = (ref self, int needed) {
19
+ // `internal`: capacity plumbing for the builder's own appenders; not
20
+ // part of the user surface.
21
+ internal func ensure = (ref self, int needed) {
20
22
  if self.cap >= needed {
21
23
  return
22
24
  }
@@ -84,7 +86,10 @@ pub struct StringBuilder {
84
86
  // Append a null-terminated string. `unsafe` Nomen single-sourced for
85
87
  // both backends: the fat pair's pointer half feeds memcpy directly, and
86
88
  // the tracked length replaces the strlen the old C body paid.
87
- func append_string = (ref self, string s) {
89
+ // `internal`: a library-implementation append — user code appends with
90
+ // `append_char` / `append_string_view`; the library's own structs (Json,
91
+ // Regex) are on the trusted side of the boundary.
92
+ internal func append_string = (ref self, string s) {
88
93
  unsafe {
89
94
  var int sl = s.length
90
95
  self.ensure(self.len + sl + 1)
@@ -175,7 +180,7 @@ pub struct StringBuilder {
175
180
  // to [0, len]; the returned string is always an independent allocation.
176
181
  // (`move out string` — the raw body mallocs the result, so the caller
177
182
  // owns it by signature, like `Console.read_line`.)
178
- func take_since = (ref self, int mark, move out string) {
183
+ pub func take_since = (ref self, int mark, move out string) {
179
184
  ```
180
185
  #arch: c
181
186
  int start = mark;
@@ -237,7 +242,10 @@ pub struct StringBuilder {
237
242
 
238
243
  // Finalize: null-terminate, return the byte buffer as a string, and
239
244
  // release ownership (sets `data` to 0 so the destructor won't free it).
240
- func to_string = (ref self, out string) {
245
+ // Explicitly `pub`: struct members default to pub, but this is the
246
+ // builder's primary read-back — the contract should not depend on the
247
+ // default.
248
+ pub func to_string = (ref self, out string) {
241
249
  ```
242
250
  #arch: c
243
251
  int idx = self->len;
@@ -255,12 +255,13 @@ func fold_pattern = (string pattern, move out string) {
255
255
  while i < n; i += 1 {
256
256
  var char c = pattern.at(i)
257
257
  if (c as int) == 92 {
258
- // Escaped char: shorthands and backrefs keep their backslash;
259
- // other escaped letters become both-case classes; escaped
260
- // punctuation is copied literally.
258
+ // Escaped char: shorthands, backrefs, and control escapes
259
+ // (\n \r \t) keep their backslash; other escaped letters
260
+ // become both-case classes; escaped punctuation is copied
261
+ // literally.
261
262
  if i + 1 < n {
262
263
  var char e = pattern.at(i + 1)
263
- if is_shorthand(e) || is_backref_digit(e) {
264
+ if is_shorthand(e) || is_backref_digit(e) || is_control_escape(e) {
264
265
  sb.append_char(c)
265
266
  sb.append_char(e)
266
267
  } else {
@@ -317,9 +318,10 @@ func fold_class = (string pattern, int cs, int ce, ref StringBuilder dst) {
317
318
  if i + 1 < content_end {
318
319
  var char e = pattern.at(i + 1)
319
320
  dst.append_char(e)
320
- // A shorthand keeps its backslash; other escaped letters
321
- // gain their case counterpart.
322
- if is_alpha_byte(e) && !is_shorthand(e) {
321
+ // A shorthand or control escape (\n \r \t) keeps its
322
+ // backslash; other escaped letters gain their case
323
+ // counterpart.
324
+ if is_alpha_byte(e) && !is_shorthand(e) && !is_control_escape(e) {
323
325
  dst.append_char(swap_case_byte(e))
324
326
  }
325
327
  i += 1
@@ -478,9 +480,11 @@ func alt_first_bytes = (string pattern, int start, int end, ref StringBuilder sb
478
480
  while i < end {
479
481
  var char c = pattern.at(i)
480
482
  if (c as int) == 92 {
481
- // escaped literal: first byte is the escaped char — unless it's
482
- // a backreference (depends on a capture) or a complement
483
- // shorthand (not an inclusion set), which disable the prefilter.
483
+ // escaped literal: first byte is the byte the escape names
484
+ // (\n is LF, \r is CR, \t is TAB; any other escaped character
485
+ // is its own literal byte) — unless it's a backreference
486
+ // (depends on a capture) or a complement shorthand (not an
487
+ // inclusion set), which disable the prefilter.
484
488
  if i + 1 < end {
485
489
  var char e = pattern.at(i + 1)
486
490
  if is_backref_digit(e) {
@@ -491,7 +495,7 @@ func alt_first_bytes = (string pattern, int start, int end, ref StringBuilder sb
491
495
  return -1
492
496
  }
493
497
  } else {
494
- sb.append_char(e)
498
+ sb.append_char(escaped_byte_value(e) as char)
495
499
  }
496
500
  // A '?'/'*' quantifier makes the atom nullable — the first
497
501
  // byte can also come from whatever follows it.
@@ -597,22 +601,34 @@ func add_class_chars = (string pattern, int start, int end, ref StringBuilder sb
597
601
  }
598
602
  var int i = cs
599
603
  while i < end {
600
- if i + 2 < end {
601
- if pattern.at(i + 1) == '-' {
602
- var int lo = pattern.at(i) as int
603
- var int hi = pattern.at(i + 2) as int
604
- while lo <= hi {
605
- sb.append_char(lo as char)
606
- lo += 1
604
+ var char c = pattern.at(i)
605
+ if (c as int) == 92 {
606
+ // Escaped byte: contribute the byte the escape names (\n is
607
+ // LF, \r is CR, \t is TAB; any other escaped character is its
608
+ // own literal byte) so the prefilter stays a superset of the
609
+ // matcher's first bytes.
610
+ if i + 1 < end {
611
+ sb.append_char(escaped_byte_value(pattern.at(i + 1)) as char)
612
+ }
613
+ i += 2
614
+ } else {
615
+ if i + 2 < end {
616
+ if pattern.at(i + 1) == '-' {
617
+ var int lo = pattern.at(i) as int
618
+ var int hi = pattern.at(i + 2) as int
619
+ while lo <= hi {
620
+ sb.append_char(lo as char)
621
+ lo += 1
622
+ }
623
+ i += 3
624
+ } else {
625
+ sb.append_char(c)
626
+ i += 1
607
627
  }
608
- i += 3
609
628
  } else {
610
- sb.append_char(pattern.at(i))
629
+ sb.append_char(c)
611
630
  i += 1
612
631
  }
613
- } else {
614
- sb.append_char(pattern.at(i))
615
- i += 1
616
632
  }
617
633
  }
618
634
  }
@@ -760,7 +776,23 @@ func match_here = (
760
776
  }
761
777
  }
762
778
 
763
- // No quantifier: match the atom once, then the rest of the pattern.
779
+ // No quantifier: match the atom once, then the rest of the pattern. A
780
+ // GROUP atom goes through the continuation-aware path: its inner
781
+ // alternatives retry when the rest fails on an earlier one (JS-style
782
+ // backtracking — `(head|header)(\s|$|>)` must still match "header>").
783
+ if pc == '(' {
784
+ return match_group_then_rest(
785
+ pattern,
786
+ pat_pos,
787
+ a_end,
788
+ a_end,
789
+ pat_end,
790
+ input,
791
+ in_pos,
792
+ in_len,
793
+ ref caps,
794
+ )
795
+ }
764
796
  var int next = match_single_atom(pattern, pat_pos, a_end, input, in_pos, in_len, ref caps)
765
797
  if next < 0 {
766
798
  return -1
@@ -768,6 +800,97 @@ func match_here = (
768
800
  return match_here(pattern, a_end, pat_end, input, next, in_len, ref caps)
769
801
  }
770
802
 
803
+ // The position of the first top-level '|' in [start, end), or -1. Escapes,
804
+ // character classes, and nested groups are skipped so only depth-0
805
+ // alternation operators count (same scanning discipline as match_here).
806
+ func next_top_level_alt = (string pattern, int start, int end, out int) {
807
+ var depth = 0
808
+ var int i = start
809
+ while i < end {
810
+ var char c = pattern.at(i)
811
+ if (c as int) == 92 {
812
+ i += 2
813
+ } else {
814
+ if c == '[' {
815
+ i = atom_end_pos(pattern, i, end)
816
+ } else {
817
+ if c == '(' {
818
+ depth += 1
819
+ }
820
+ if c == ')' {
821
+ if depth > 0 {
822
+ depth -= 1
823
+ }
824
+ }
825
+ if (c as int) == 124 {
826
+ if depth == 0 {
827
+ return i
828
+ }
829
+ }
830
+ i += 1
831
+ }
832
+ }
833
+ }
834
+ return -1
835
+ }
836
+
837
+ // Match the group atom spanning [atom_start, atom_end) ('(' at atom_start)
838
+ // and then the continuation [rest, pat_end), retrying the group's LATER
839
+ // alternatives whenever the continuation fails on an earlier one. Without
840
+ // this, a group commits to the first alternative that matches the group
841
+ // ALONE and a failing continuation aborts the whole match — the engine had
842
+ // no backtracking into a group's alternatives (PORT.md: `(head|header)`
843
+ // never reached `header` when `(\s|$|>)` rejected what `head` left behind).
844
+ // A group without a top-level '|' behaves exactly like the old
845
+ // match-atom-then-continue path, and a failed trial restores the group's
846
+ // previous capture span before the next attempt.
847
+ func match_group_then_rest = (
848
+ string pattern,
849
+ int atom_start,
850
+ int atom_end,
851
+ int rest,
852
+ int pat_end,
853
+ string input,
854
+ int in_pos,
855
+ int in_len,
856
+ ref List<int> caps,
857
+ out int
858
+ ) {
859
+ var int g = group_index_at(pattern, atom_start)
860
+ var int at = last_capture_index(caps, g)
861
+ var int old_start = -1
862
+ var int old_end = -1
863
+ if at >= 0 {
864
+ old_start = caps.at(at + 1)
865
+ old_end = caps.at(at + 2)
866
+ }
867
+ var int inner_start = atom_start + 1
868
+ var int inner_end = atom_end - 1
869
+ var int s = inner_start
870
+ while s <= inner_end {
871
+ var int bar = next_top_level_alt(pattern, s, inner_end)
872
+ var int e = inner_end
873
+ if bar >= 0 {
874
+ e = bar
875
+ }
876
+ var int next = match_here(pattern, s, e, input, in_pos, in_len, ref caps)
877
+ if next >= 0 {
878
+ record_capture(ref caps, g, in_pos, next)
879
+ var int r = match_here(pattern, rest, pat_end, input, next, in_len, ref caps)
880
+ if r >= 0 {
881
+ return r
882
+ }
883
+ }
884
+ record_capture(ref caps, g, old_start, old_end)
885
+ if bar >= 0 {
886
+ s = bar + 1
887
+ } else {
888
+ s = inner_end + 1
889
+ }
890
+ }
891
+ return -1
892
+ }
893
+
771
894
  // Return the position just past the atom that starts at pat_pos. Handles
772
895
  // character classes [...], groups (...), and escaped characters (\x).
773
896
  func atom_end_pos = (string pattern, int pat_pos, int pat_end, out int) {
@@ -888,7 +1011,7 @@ func match_single_atom = (
888
1011
  return -1
889
1012
  }
890
1013
  if in_pos < in_len {
891
- if input.at(in_pos) == ec {
1014
+ if (input.at(in_pos) as int) == escaped_byte_value(ec) {
892
1015
  return in_pos + 1
893
1016
  }
894
1017
  }
@@ -936,7 +1059,7 @@ func char_matches_atom = (
936
1059
  if is_shorthand(ec) {
937
1060
  return matches_shorthand(ec, ic)
938
1061
  }
939
- return ic == ec
1062
+ return (ic as int) == escaped_byte_value(ec)
940
1063
  }
941
1064
  return false
942
1065
  }
@@ -1046,29 +1169,99 @@ func match_star_group = (
1046
1169
  ) {
1047
1170
  if lazy > 0 {
1048
1171
  // Lazy: stop repeating first (minimum met), then try one more
1049
- // iteration before giving up.
1172
+ // iteration before giving up. The iteration retries the group's
1173
+ // top-level alternatives (see match_group_iteration).
1050
1174
  if min_count <= 0 {
1051
1175
  var int stop = match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
1052
1176
  if stop >= 0 {
1053
1177
  return stop
1054
1178
  }
1055
1179
  }
1056
- var int next = match_single_atom(
1180
+ return match_group_iteration(
1057
1181
  pattern,
1058
1182
  atom_start,
1059
1183
  atom_end,
1184
+ rest,
1185
+ pat_end,
1060
1186
  input,
1061
1187
  in_pos,
1188
+ min_count,
1189
+ lazy,
1062
1190
  in_len,
1063
1191
  ref caps,
1064
1192
  )
1193
+ }
1194
+ // Try matching the group one more time (greedy) — each top-level
1195
+ // alternative in turn — before stopping.
1196
+ var int r = match_group_iteration(
1197
+ pattern,
1198
+ atom_start,
1199
+ atom_end,
1200
+ rest,
1201
+ pat_end,
1202
+ input,
1203
+ in_pos,
1204
+ min_count,
1205
+ lazy,
1206
+ in_len,
1207
+ ref caps,
1208
+ )
1209
+ if r >= 0 {
1210
+ return r
1211
+ }
1212
+ // Stop repeating and try the rest of the pattern, if the minimum is met.
1213
+ if min_count <= 0 {
1214
+ return match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
1215
+ }
1216
+ return -1
1217
+ }
1218
+
1219
+ // Try ONE further iteration of a quantified group at in_pos, retrying the
1220
+ // group's top-level alternatives while the remaining star recursion fails.
1221
+ // Returns the recursion's result, or -1 when no alternative leads to a
1222
+ // match. An alternative that matches the empty string is not recursed into
1223
+ // (an empty iteration cannot make progress; the caller's stop path covers
1224
+ // it), and every failed trial restores the group's previous capture span.
1225
+ func match_group_iteration = (
1226
+ string pattern,
1227
+ int atom_start,
1228
+ int atom_end,
1229
+ int rest,
1230
+ int pat_end,
1231
+ string input,
1232
+ int in_pos,
1233
+ int min_count,
1234
+ int lazy,
1235
+ int in_len,
1236
+ ref List<int> caps,
1237
+ out int
1238
+ ) {
1239
+ var int g = group_index_at(pattern, atom_start)
1240
+ var int at = last_capture_index(caps, g)
1241
+ var int old_start = -1
1242
+ var int old_end = -1
1243
+ if at >= 0 {
1244
+ old_start = caps.at(at + 1)
1245
+ old_end = caps.at(at + 2)
1246
+ }
1247
+ var int inner_start = atom_start + 1
1248
+ var int inner_end = atom_end - 1
1249
+ var int s = inner_start
1250
+ while s <= inner_end {
1251
+ var int bar = next_top_level_alt(pattern, s, inner_end)
1252
+ var int e = inner_end
1253
+ if bar >= 0 {
1254
+ e = bar
1255
+ }
1256
+ var int next = match_here(pattern, s, e, input, in_pos, in_len, ref caps)
1065
1257
  if next >= 0 {
1258
+ record_capture(ref caps, g, in_pos, next)
1066
1259
  if next != in_pos {
1067
1260
  var new_min = 0
1068
1261
  if min_count > 0 {
1069
1262
  new_min = min_count - 1
1070
1263
  }
1071
- return match_star_group(
1264
+ var int r = match_star_group(
1072
1265
  pattern,
1073
1266
  atom_start,
1074
1267
  atom_end,
@@ -1081,40 +1274,18 @@ func match_star_group = (
1081
1274
  in_len,
1082
1275
  ref caps,
1083
1276
  )
1277
+ if r >= 0 {
1278
+ return r
1279
+ }
1084
1280
  }
1085
1281
  }
1086
- return -1
1087
- }
1088
- // Try matching the group one more time (greedy).
1089
- var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len, ref caps)
1090
- if next >= 0 {
1091
- if next != in_pos {
1092
- var new_min = 0
1093
- if min_count > 0 {
1094
- new_min = min_count - 1
1095
- }
1096
- var int r = match_star_group(
1097
- pattern,
1098
- atom_start,
1099
- atom_end,
1100
- rest,
1101
- pat_end,
1102
- input,
1103
- next,
1104
- new_min,
1105
- lazy,
1106
- in_len,
1107
- ref caps,
1108
- )
1109
- if r >= 0 {
1110
- return r
1111
- }
1282
+ record_capture(ref caps, g, old_start, old_end)
1283
+ if bar >= 0 {
1284
+ s = bar + 1
1285
+ } else {
1286
+ s = inner_end + 1
1112
1287
  }
1113
1288
  }
1114
- // Stop repeating and try the rest of the pattern, if the minimum is met.
1115
- if min_count <= 0 {
1116
- return match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
1117
- }
1118
1289
  return -1
1119
1290
  }
1120
1291
 
@@ -1138,6 +1309,21 @@ func match_optional = (
1138
1309
  if skip >= 0 {
1139
1310
  return skip
1140
1311
  }
1312
+ // A group atom includes continuation-aware alternative retries (see
1313
+ // match_group_then_rest) in both moods.
1314
+ if pattern.at(atom_start) == '(' {
1315
+ return match_group_then_rest(
1316
+ pattern,
1317
+ atom_start,
1318
+ atom_end,
1319
+ rest,
1320
+ pat_end,
1321
+ input,
1322
+ in_pos,
1323
+ in_len,
1324
+ ref caps,
1325
+ )
1326
+ }
1141
1327
  var int next = match_single_atom(
1142
1328
  pattern,
1143
1329
  atom_start,
@@ -1152,11 +1338,28 @@ func match_optional = (
1152
1338
  }
1153
1339
  return -1
1154
1340
  }
1155
- var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len, ref caps)
1156
- if next >= 0 {
1157
- var int r = match_here(pattern, rest, pat_end, input, next, in_len, ref caps)
1158
- if r >= 0 {
1159
- return r
1341
+ if pattern.at(atom_start) == '(' {
1342
+ var int grouped = match_group_then_rest(
1343
+ pattern,
1344
+ atom_start,
1345
+ atom_end,
1346
+ rest,
1347
+ pat_end,
1348
+ input,
1349
+ in_pos,
1350
+ in_len,
1351
+ ref caps,
1352
+ )
1353
+ if grouped >= 0 {
1354
+ return grouped
1355
+ }
1356
+ } else {
1357
+ var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len, ref caps)
1358
+ if next >= 0 {
1359
+ var int r = match_here(pattern, rest, pat_end, input, next, in_len, ref caps)
1360
+ if r >= 0 {
1361
+ return r
1362
+ }
1160
1363
  }
1161
1364
  }
1162
1365
  return match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
@@ -1201,7 +1404,7 @@ func char_in_class = (
1201
1404
  found = true
1202
1405
  }
1203
1406
  } else {
1204
- if ic == e {
1407
+ if (ic as int) == escaped_byte_value(e) {
1205
1408
  found = true
1206
1409
  }
1207
1410
  }
@@ -1456,6 +1659,29 @@ func is_shorthand = (char c, out bool) {
1456
1659
  return code == 100 || code == 68 || code == 119 || code == 87 || code == 115 || code == 83
1457
1660
  }
1458
1661
 
1662
+ // Whether the escaped character c names a control-character shorthand
1663
+ // (\n \r \t) that the engine maps to its real control byte.
1664
+ func is_control_escape = (char c, out bool) {
1665
+ var int code = c as int
1666
+ return code == 110 || code == 114 || code == 116
1667
+ }
1668
+
1669
+ // The byte an escaped atom names: \n is LF (10), \r is CR (13), \t is TAB
1670
+ // (9); every other escaped character is the literal byte of that character.
1671
+ func escaped_byte_value = (char c, out int) {
1672
+ var int code = c as int
1673
+ if code == 110 {
1674
+ return 10
1675
+ }
1676
+ if code == 114 {
1677
+ return 13
1678
+ }
1679
+ if code == 116 {
1680
+ return 9
1681
+ }
1682
+ return code
1683
+ }
1684
+
1459
1685
  // Whether byte b belongs to the shorthand class named by c. Uppercase names
1460
1686
  // are the complements (S/D/W match anything the lowercase set excludes).
1461
1687
  func matches_shorthand = (char c, char b, out bool) {
package/dist/index.mjs CHANGED
@@ -21421,6 +21421,67 @@ function call_in_set(set, call) {
21421
21421
  return set.has(call.name.replace(/#/g, ""));
21422
21422
  }
21423
21423
  //#endregion
21424
+ //#region ../src/build_common/fold_string_const.ts
21425
+ /**
21426
+ * Fold a top-level `const`'s string `+` chain into a single string literal.
21427
+ *
21428
+ * Module-level `const` declarations are inlined at every use site (see
21429
+ * `top_level_consts` in both backends' BuildStatus): the build replays the
21430
+ * const's initializer expression wherever the name appears. For a composed
21431
+ * pattern like `pub const HTML_TAG_REGEX = "^(" + TAG_NAME + "|…)"` that
21432
+ * replay lowers to a malloc/copy/free `string_add` chain PER USE — the
21433
+ * allmark port's inline-token loop rebuilt its ~1 KB pattern on every
21434
+ * attempt, and the chain's calls were also the trigger for the aarch64
21435
+ * ref-argument clobber (PORT.md).
21436
+ *
21437
+ * When the initializer is a pure `+` chain of plain string literals and
21438
+ * references to other foldable consts, this returns one merged literal
21439
+ * ValueNode — semantically the string the source spells — that the use site
21440
+ * emits as rodata instead. Anything else (a runtime operand, a struct ctor,
21441
+ * an operator overload) returns null and the tree builds as before.
21442
+ *
21443
+ * Raw token merging is byte-preserving: escape sequences are self-contained
21444
+ * within a part's content (a content ending in a lone `\` would have
21445
+ * escaped that part's closing quote, so it cannot occur), and both
21446
+ * emitters' `\x`-as-octal re-encoding plus the 2/3-digit caps keep merged
21447
+ * escapes from absorbing a following part's text.
21448
+ */
21449
+ const fold_cache = /* @__PURE__ */ new WeakMap();
21450
+ function fold_string_const(decl, consts) {
21451
+ if (fold_cache.has(decl)) return fold_cache.get(decl);
21452
+ const folded = fold_value(decl.value, consts, /* @__PURE__ */ new Set());
21453
+ fold_cache.set(decl, folded);
21454
+ return folded;
21455
+ }
21456
+ function fold_value(node, consts, active) {
21457
+ if (!node) return null;
21458
+ if (node.node_type === "value") {
21459
+ const raw = node.value;
21460
+ if (typeof raw !== "string") return null;
21461
+ if (raw.startsWith("\"")) return literal(raw, node.start);
21462
+ if (active.has(raw)) return null;
21463
+ const ref = consts.get(raw);
21464
+ if (!ref) return null;
21465
+ active.add(raw);
21466
+ const folded = fold_value(ref.value, consts, active);
21467
+ active.delete(raw);
21468
+ return folded;
21469
+ }
21470
+ if (node.node_type === "op") {
21471
+ const op = node;
21472
+ if (!(op.op === "+" && (!op.operator_func || op.operator_func.struct_name === "string"))) return null;
21473
+ const left = fold_value(op.left_value, consts, active);
21474
+ if (!left) return null;
21475
+ const right = fold_value(op.right_value, consts, active);
21476
+ if (!right) return null;
21477
+ return literal(left.value.slice(0, -1) + right.value.slice(1), node.start);
21478
+ }
21479
+ return null;
21480
+ }
21481
+ function literal(raw, start) {
21482
+ return new ValueNode(start, raw, new Type("string", true));
21483
+ }
21484
+ //#endregion
21424
21485
  //#region ../src/build_common/string_escapes.ts
21425
21486
  /**
21426
21487
  * Source string-literal escape handling shared by both backends.
@@ -21455,8 +21516,8 @@ function reencode_hex_escapes(raw) {
21455
21516
  let j = i + 2;
21456
21517
  let value = 0;
21457
21518
  let digits = 0;
21458
- while (j < raw.length && digits < 2 && is_hex_digit(raw[j])) {
21459
- value = value * 16 + hex_value(raw[j]);
21519
+ while (j < raw.length && digits < 2 && is_hex_digit$1(raw[j])) {
21520
+ value = value * 16 + hex_value$1(raw[j]);
21460
21521
  j += 1;
21461
21522
  digits += 1;
21462
21523
  }
@@ -21471,10 +21532,10 @@ function reencode_hex_escapes(raw) {
21471
21532
  }
21472
21533
  return out;
21473
21534
  }
21474
- function is_hex_digit(c) {
21535
+ function is_hex_digit$1(c) {
21475
21536
  return c >= "0" && c <= "9" || c >= "a" && c <= "f" || c >= "A" && c <= "F";
21476
21537
  }
21477
- function hex_value(c) {
21538
+ function hex_value$1(c) {
21478
21539
  if (c <= "9") return c.charCodeAt(0) - 48;
21479
21540
  if (c <= "F") return c.charCodeAt(0) - 55;
21480
21541
  return c.charCodeAt(0) - 87;
@@ -21504,7 +21565,7 @@ function scan_string_escapes(raw) {
21504
21565
  if (next === "x") {
21505
21566
  let j = i + 2;
21506
21567
  let digits = 0;
21507
- while (j < end && digits < 2 && is_hex_digit(raw[j])) {
21568
+ while (j < end && digits < 2 && is_hex_digit$1(raw[j])) {
21508
21569
  j += 1;
21509
21570
  digits += 1;
21510
21571
  }
@@ -21539,6 +21600,16 @@ function scan_string_escapes(raw) {
21539
21600
  }
21540
21601
  return issues;
21541
21602
  }
21603
+ /** Escape one raw string-literal token (quotes included) for a GAS
21604
+ * `.asciz` directive: source `\xHH` hex escapes re-encode as 3-digit octal
21605
+ * (GAS consumes `\x` greedily — see reencode_hex_escapes) and raw newlines
21606
+ * (multi-line strings) become `\n` so the directive stays one line. */
21607
+ function escape_asciz(value) {
21608
+ const reencoded = reencode_hex_escapes(value);
21609
+ if (!reencoded.includes("\n")) return reencoded;
21610
+ const quote = reencoded[0];
21611
+ return quote + reencoded.slice(1, reencoded.endsWith(quote) ? -1 : void 0).replace(/\n/g, "\\n") + (reencoded.endsWith(quote) ? quote : "");
21612
+ }
21542
21613
  //#endregion
21543
21614
  //#region ../src/build_aarch64/build_range_node.ts
21544
21615
  function build_range_node$1(node, status) {
@@ -21752,12 +21823,6 @@ function is_value_struct_type_node(node, status) {
21752
21823
  const t = type_from_value_node$1(node);
21753
21824
  return !!t?.name && !!status.structs.find((s) => s.name === t.name && !s.is_simple_type && !s.is_class);
21754
21825
  }
21755
- function escape_asciz(value) {
21756
- const reencoded = reencode_hex_escapes(value);
21757
- if (!reencoded.includes("\n")) return reencoded;
21758
- const quote = reencoded[0];
21759
- return quote + reencoded.slice(1, reencoded.endsWith(quote) ? -1 : void 0).replace(/\n/g, "\\n") + (reencoded.endsWith(quote) ? quote : "");
21760
- }
21761
21826
  /** Allocate an array on the stack with an 8-byte length prefix.
21762
21827
  * Writes the length at the start and returns the offset of the first element. */
21763
21828
  function alloc_array_with_prefix(status, length, element_size) {
@@ -22227,6 +22292,93 @@ function resolve_string_op(op, status) {
22227
22292
  }
22228
22293
  return null;
22229
22294
  }
22295
+ const FLOAT_LITERAL_RE = /^(\+|-)?\d+\.\d+([eE](\+|-)?\d+)?$/;
22296
+ /**
22297
+ * Fold a module-level primitive `const`/`var` initializer to a literal string
22298
+ * when it is a compile-time constant: literal operands, references to other
22299
+ * foldable module-level consts, and arithmetic/bitwise operators over them.
22300
+ * Returns null when the value can only be computed at runtime (a call, a
22301
+ * field read, …), which a file-scope `.quad` initializer cannot express.
22302
+ *
22303
+ * Without this, `pub const N = 3 + 4` at module scope reserved
22304
+ * `N: .space 8` and emitted its store as dead code between functions, so
22305
+ * every read observed 0 (FOLLOWUP.md). Integer math is exact via BigInt so
22306
+ * 64-bit constants are not rounded through JS doubles; float arithmetic
22307
+ * falls back to JS doubles (which match Nomen's `float`).
22308
+ */
22309
+ function fold_const_scalar(node, status, seen) {
22310
+ if (!node) return null;
22311
+ if (node.node_type === "grouped") return fold_const_scalar(node.value, status, seen);
22312
+ if (node.node_type === "value") {
22313
+ const raw = get_raw_value$1(node, status);
22314
+ if (is_int_literal(raw)) return to_decimal_string(raw);
22315
+ if (FLOAT_LITERAL_RE.test(raw)) return raw;
22316
+ if (/^[A-Za-z_][A-Za-z0-9_]*$/.test(raw) && !seen.has(raw)) {
22317
+ const decl = find_global_const(raw, status);
22318
+ if (decl?.value) {
22319
+ seen.add(raw);
22320
+ const folded = fold_const_scalar(decl.value, status, seen);
22321
+ seen.delete(raw);
22322
+ return folded;
22323
+ }
22324
+ }
22325
+ return null;
22326
+ }
22327
+ if (node.node_type === "op") {
22328
+ const op = node;
22329
+ if (op.operator_func) return null;
22330
+ const left = fold_const_scalar(op.left_value, status, seen);
22331
+ if (left === null) return null;
22332
+ const right = fold_const_scalar(op.right_value, status, seen);
22333
+ if (right === null) return null;
22334
+ return apply_const_op(op.op, left, right);
22335
+ }
22336
+ return null;
22337
+ }
22338
+ /** Apply one arithmetic/bitwise operator to two folded literal strings.
22339
+ * Float operands use JS doubles; integral operands stay exact via BigInt. */
22340
+ function apply_const_op(op, left, right) {
22341
+ if (FLOAT_LITERAL_RE.test(left) || FLOAT_LITERAL_RE.test(right)) {
22342
+ const a = Number(left);
22343
+ const b = Number(right);
22344
+ if (!isFinite(a) || !isFinite(b)) return null;
22345
+ switch (op) {
22346
+ case "+": return String(a + b);
22347
+ case "-": return String(a - b);
22348
+ case "*": return String(a * b);
22349
+ case "/": return b === 0 ? null : String(a / b);
22350
+ default: return null;
22351
+ }
22352
+ }
22353
+ let a;
22354
+ let b;
22355
+ try {
22356
+ a = BigInt(left);
22357
+ b = BigInt(right);
22358
+ } catch {
22359
+ return null;
22360
+ }
22361
+ switch (op) {
22362
+ case "+": return (a + b).toString();
22363
+ case "-": return (a - b).toString();
22364
+ case "*": return (a * b).toString();
22365
+ case "/": return b === 0n ? null : (a / b).toString();
22366
+ case "%": return b === 0n ? null : (a % b).toString();
22367
+ case "<<": return (a << b).toString();
22368
+ case ">>": return (a >> b).toString();
22369
+ case "&": return (a & b).toString();
22370
+ case "|": return (a | b).toString();
22371
+ case "^": return (a ^ b).toString();
22372
+ default: return null;
22373
+ }
22374
+ }
22375
+ /** The root-scope `const` declaration named `name` (used to fold a const
22376
+ * reference through to its initializer). Mutable `var`s are deliberately
22377
+ * excluded: their value is established by module init, so baking in the
22378
+ * declaration's initializer could differ from the runtime value. */
22379
+ function find_global_const(name, status) {
22380
+ return (status.root.statements ?? []).find((s) => s.node_type === "declare" && s.declaration === "const" && s.name === name);
22381
+ }
22230
22382
  /**
22231
22383
  * Emit a declaration INITIALIZER expression. Under NIR-driven emission the
22232
22384
  * lowered `NirExpr` rides in and descends through `emit_expr_from_nir` (the
@@ -22258,7 +22410,7 @@ function build_declaration_node$1(node, status, nir_init, nir_swap) {
22258
22410
  const prev_heap = status.last_result_is_heap;
22259
22411
  if (node.value?.node_type === "value") {
22260
22412
  const inlined = status.top_level_consts?.get(node.value.value);
22261
- if (inlined?.value) node.value = inlined.value;
22413
+ if (inlined?.value) node.value = fold_string_const(inlined, status.top_level_consts) ?? inlined.value;
22262
22414
  }
22263
22415
  function check_heap() {
22264
22416
  if (status.last_result_is_heap) {
@@ -23200,7 +23352,15 @@ function build_declaration_node$1(node, status, nir_init, nir_swap) {
23200
23352
  if (status.function_return_label) {
23201
23353
  const offset = declare_slot_offset(status, node.name, size);
23202
23354
  status.stack_offsets.set(node.name, offset);
23203
- } else emit_data(status, `${node.name}: .space ${size}\n`);
23355
+ } else {
23356
+ const folded = fold_const_scalar(node.value, status, /* @__PURE__ */ new Set());
23357
+ if (folded !== null) {
23358
+ emit_data(status, `${node.name}: ${directive} ${folded}\n`);
23359
+ if (size % 4 !== 0) emit_data(status, `.p2align 2\n`);
23360
+ return;
23361
+ }
23362
+ emit_data(status, `${node.name}: .space ${size}\n`);
23363
+ }
23204
23364
  if (node.type.name === "string" && !node.type.is_view && is_view_value(node.value, status)) {
23205
23365
  emit_view_string_arg(node.value, status);
23206
23366
  emit_view_materialize_owned(status);
@@ -36796,7 +36956,7 @@ function build_value_node$1(node, status) {
36796
36956
  }
36797
36957
  const inlined = status.top_level_consts?.get(value);
36798
36958
  if (inlined?.value) {
36799
- build_node$1(inlined.value, status);
36959
+ build_node$1(fold_string_const(inlined, status.top_level_consts) ?? inlined.value, status);
36800
36960
  return;
36801
36961
  }
36802
36962
  if (node.is_enum_shorthand) {
@@ -37737,7 +37897,7 @@ function build_value_node(node, status) {
37737
37897
  }
37738
37898
  const inlined = status.top_level_consts?.get(original_value);
37739
37899
  if (inlined?.value) {
37740
- build_node(inlined.value, status);
37900
+ build_node(fold_string_const(inlined, status.top_level_consts) ?? inlined.value, status);
37741
37901
  return;
37742
37902
  }
37743
37903
  if (node.is_enum_shorthand) {
@@ -42980,8 +43140,6 @@ function build_access_method(node, access_func, status) {
42980
43140
  return true;
42981
43141
  };
42982
43142
  const overflow_count = Math.max(0, total_arg_slots - (8 - start_reg));
42983
- let overflow_base = 0;
42984
- if (overflow_count > 0) overflow_base = allocate_stack_space(status, overflow_count * 8, 16);
42985
43143
  const lambda_arg_indices = /* @__PURE__ */ new Set();
42986
43144
  for (let i = 0; i < access_func.params.length; i++) {
42987
43145
  const p = access_func.params[i];
@@ -42992,21 +43150,17 @@ function build_access_method(node, access_func, status) {
42992
43150
  lambda_arg_indices.add(i);
42993
43151
  }
42994
43152
  const lambda_arg_slots = [];
42995
- const has_pair_args = view_arg_set.size > 0 || string_arg_set.size > 0;
42996
- let view_spill_base = 0;
42997
- if (has_pair_args) view_spill_base = allocate_stack_space(status, total_arg_slots * 8, 16);
42998
- const view_half_store = (j, half) => {
42999
- const half_slot = start_reg + arg_slot[j] + half;
43000
- return half_slot >= 8 ? overflow_base + (half_slot - 8) * 8 : view_spill_base + (arg_slot[j] + half) * 8;
43153
+ const arg_builds_in_loop = (i) => {
43154
+ if (view_arg_set.has(i) || string_arg_set.has(i)) return true;
43155
+ if ((access_func.ref_param_indices ?? []).includes(i)) return true;
43156
+ const param_type = access_func.params[i].type?.name || "";
43157
+ if (is_struct_type(param_type, status) || is_enum_with_data_type(param_type, status)) return true;
43158
+ return !arg_deferrable(i);
43001
43159
  };
43002
- const param0 = access_func.params[0];
43003
- const param0_is_pair = view_arg_set.has(0) || string_arg_set.has(0);
43004
- const param0_is_ref = (access_func.ref_param_indices ?? []).includes(0);
43005
- const param0_type_name = param0?.type?.name || "";
43006
- const param0_builds_in_loop = access_func.params.length > 0 && (param0_is_ref || is_struct_type(param0_type_name, status) || is_enum_with_data_type(param0_type_name, status) || !arg_deferrable(0));
43007
- const static_arg0_needs_spill = start_reg === 0 && access_func.params.length > 0 && !param0_is_pair && param0_builds_in_loop;
43008
- let static_arg0_spill = 0;
43009
- if (static_arg0_needs_spill) static_arg0_spill = allocate_stack_space(status, 8, 8);
43160
+ const any_in_loop_arg = access_func.params.some((_, i) => arg_builds_in_loop(i));
43161
+ let arg_spill_base = 0;
43162
+ if (any_in_loop_arg) arg_spill_base = allocate_stack_space(status, total_arg_slots * 8, 16);
43163
+ const view_half_store = (j, half) => arg_spill_base + (arg_slot[j] + half) * 8;
43010
43164
  const ref_class_param_reload = [];
43011
43165
  const ref_class_sync_names = [];
43012
43166
  for (let i = access_func.params.length - 1; i >= 0; i--) {
@@ -43076,31 +43230,20 @@ function build_access_method(node, access_func, status) {
43076
43230
  lambda_arg_slots.push(dispose_slot);
43077
43231
  }
43078
43232
  }
43079
- const slot = start_reg + arg_slot[i];
43080
43233
  ensure_newline(status);
43081
- if (slot >= 8) emit_asm(status, `str x0, [x29, #${overflow_base + (slot - 8) * 8}]\n`);
43082
- else {
43083
- const reg = `x${slot}`;
43084
- if (reg !== "x0") emit_asm(status, `mov ${reg}, x0\n`);
43085
- else if (static_arg0_needs_spill) emit_asm(status, `str x0, [x29, #${static_arg0_spill}]\n`);
43086
- }
43234
+ emit_asm(status, `str x0, [x29, #${arg_spill_base + arg_slot[i] * 8}]\n`);
43087
43235
  }
43088
43236
  deferred_args.sort((a, b) => b.slot - a.slot);
43089
43237
  for (const d of deferred_args) {
43090
43238
  build_operand(d.param, `x${start_reg + d.slot}`, status);
43091
43239
  ensure_newline(status);
43092
43240
  }
43093
- if (has_pair_args) for (let j = 0; j < access_func.params.length; j++) {
43094
- if (!view_arg_set.has(j) && !string_arg_set.has(j)) continue;
43095
- for (const half of [0, 1]) {
43096
- const half_slot = start_reg + arg_slot[j] + half;
43097
- if (half_slot >= 8) continue;
43098
- emit_asm(status, `ldr x${half_slot}, [x29, #${view_spill_base + (arg_slot[j] + half) * 8}]\n`);
43099
- }
43100
- }
43101
- if (static_arg0_needs_spill) {
43102
- ensure_newline(status);
43103
- emit_asm(status, `ldr x0, [x29, #${static_arg0_spill}]\n`);
43241
+ const deferred_slots = new Set(deferred_args.map((d) => start_reg + d.slot));
43242
+ for (let s = 0; s < total_arg_slots; s++) {
43243
+ const slot = start_reg + s;
43244
+ if (slot >= 8) continue;
43245
+ if (deferred_slots.has(slot)) continue;
43246
+ emit_asm(status, `ldr x${slot}, [x29, #${arg_spill_base + s * 8}]\n`);
43104
43247
  }
43105
43248
  ensure_newline(status);
43106
43249
  if (receiver_is_string) emit_asm(status, frees_string_receiver ? `ldp x0, x1, [sp]\n` : `ldp x0, x1, [sp], #16\n`);
@@ -43110,8 +43253,9 @@ function build_access_method(node, access_func, status) {
43110
43253
  if (overflow_count > 0) {
43111
43254
  outgoing_size = Math.ceil(overflow_count * 8 / 16) * 16;
43112
43255
  emit_asm(status, `sub sp, sp, #${outgoing_size}\n`);
43256
+ const overflow_first = 8 - start_reg;
43113
43257
  for (let k = 0; k < overflow_count; k++) {
43114
- emit_asm(status, `ldr x9, [x29, #${overflow_base + k * 8}]\n`);
43258
+ emit_asm(status, `ldr x9, [x29, #${arg_spill_base + (overflow_first + k) * 8}]\n`);
43115
43259
  emit_asm(status, `str x9, [sp, #${k * 8}]\n`);
43116
43260
  }
43117
43261
  }
@@ -44190,10 +44334,7 @@ function build(root, options = {}) {
44190
44334
  }
44191
44335
  if (status.strings && status.strings.size > 0) {
44192
44336
  status.code += "\n";
44193
- for (const [label, value] of status.strings) {
44194
- const escaped = value.replace(/\n/g, "\\n");
44195
- status.code += `${label}: .asciz ${escaped}\n`;
44196
- }
44337
+ for (const [label, value] of status.strings) status.code += `${label}: .asciz ${escape_asciz(value)}\n`;
44197
44338
  }
44198
44339
  if (status.float_literals && status.float_literals.size > 0) {
44199
44340
  status.code += "\n.p2align 2\n";
@@ -50018,16 +50159,22 @@ function must_use_type_display_name(en, type) {
50018
50159
  */
50019
50160
  function gather_top_level_consts(block, status) {
50020
50161
  const seen = /* @__PURE__ */ new Set();
50162
+ const candidates = [];
50021
50163
  for (const child of block.statements) {
50022
50164
  if (child.node_type !== "declare") continue;
50023
50165
  const decl = child;
50024
50166
  if (decl.declaration !== "const") continue;
50025
- if (!decl.name || !decl.type?.name) continue;
50026
- if (decl.type.is_array) continue;
50167
+ if (!decl.name) continue;
50027
50168
  if (seen.has(decl.name)) continue;
50028
50169
  seen.add(decl.name);
50170
+ candidates.push(decl);
50171
+ }
50172
+ infer_const_decl_types(candidates, status);
50173
+ for (const decl of candidates) {
50174
+ if (!decl.type?.name) continue;
50175
+ if (decl.type.is_array) continue;
50029
50176
  status.values.push({
50030
- declaration: decl.declaration,
50177
+ declaration: "const",
50031
50178
  name: decl.name,
50032
50179
  type: decl.type,
50033
50180
  is_set: !!decl.value,
@@ -50037,6 +50184,60 @@ function gather_top_level_consts(block, status) {
50037
50184
  }
50038
50185
  }
50039
50186
  /**
50187
+ * Give every root-level `const` a declared type BEFORE the statement walk,
50188
+ * so a const declared BELOW its first use resolves like a free function
50189
+ * (which hoists across the whole file). Parse can only infer the type when
50190
+ * the initializer is a bare literal; a composed chain (`const TAG = "<" +
50191
+ * TAG_NAME + ">"`, the allmark htmlPatterns shape) stayed typeless and its
50192
+ * use-above-declaration failed with "Unknown value".
50193
+ *
50194
+ * The inference iterates until nothing new is derivable, so a chain may
50195
+ * reference consts declared BELOW itself too. Recognized initializers:
50196
+ * literals, references to other (typed) consts, and arithmetic/concat
50197
+ * operations (the result type is the left operand's, mirroring
50198
+ * check_operation_node). Anything else (constructor calls, function calls)
50199
+ * keeps today's behavior: annotate the type or declare before use.
50200
+ */
50201
+ function infer_const_decl_types(decls, status) {
50202
+ const known = /* @__PURE__ */ new Map();
50203
+ for (const decl of decls) if (decl.type?.name) known.set(decl.name, decl.type);
50204
+ let progress = true;
50205
+ while (progress) {
50206
+ progress = false;
50207
+ for (const decl of decls) {
50208
+ if (decl.type?.name) continue;
50209
+ const inferred = infer_const_expr_type(decl.value, known, status);
50210
+ if (!inferred) continue;
50211
+ decl.type = inferred;
50212
+ known.set(decl.name, inferred);
50213
+ progress = true;
50214
+ }
50215
+ }
50216
+ }
50217
+ function infer_const_expr_type(node, known, status) {
50218
+ if (!node) return void 0;
50219
+ if (node.node_type === "grouped") return infer_const_expr_type(node.value, known, status);
50220
+ if (node.node_type === "value") {
50221
+ const raw = node.value;
50222
+ if (typeof raw !== "string") return void 0;
50223
+ if (raw.startsWith("\"")) return new Type("string", true);
50224
+ if (raw === "true" || raw === "false") return new Type("bool", true);
50225
+ if (raw.startsWith("'") && raw.endsWith("'")) return new Type("char", true);
50226
+ if (is_int_literal(raw)) return new Type("int", true);
50227
+ if (/^(\+|-)*\d+.\d+([eE](\+|-)?\d+)?$/.test(raw)) return new Type("float", true);
50228
+ const declared = known.get(raw);
50229
+ if (declared) return declared;
50230
+ return;
50231
+ }
50232
+ if (node.node_type === "op") return infer_const_expr_type(node.left_value, known, status);
50233
+ if (node.node_type === "func_call") {
50234
+ const call = node;
50235
+ const target = status.structs.find((s) => s.name === call.name && !s.is_simple_type);
50236
+ if (target && target.type_params.length === 0) return new Type(target.name);
50237
+ return;
50238
+ }
50239
+ }
50240
+ /**
50040
50241
  * A bare statement that constructs (or obtains) a class instance without
50041
50242
  * binding it to a variable would leak: classes are heap-allocated and freed
50042
50243
  * by the owning variable's scope-exit cleanup, so an unbound instance is
@@ -50374,6 +50575,7 @@ function clone_status(status) {
50374
50575
  var_name_counter: status.var_name_counter,
50375
50576
  type_params: status.type_params,
50376
50577
  errors: status.errors,
50578
+ warnings: status.warnings,
50377
50579
  buffer_caps: status.buffer_caps,
50378
50580
  equal_lengths: status.equal_lengths?.slice(),
50379
50581
  mutated_local_names: status.mutated_local_names,
@@ -52474,6 +52676,143 @@ function is_visible(declaring_scope, visibility, access_scope, stack, declaring_
52474
52676
  return access_scope === declaring_scope;
52475
52677
  }
52476
52678
  //#endregion
52679
+ //#region ../src/check/utils/regex_pattern_lint.ts
52680
+ /**
52681
+ * Compile-time lint for Regex pattern literals (FOLLOWUP "Regex pattern
52682
+ * escapes").
52683
+ *
52684
+ * The Regex engine (core/System/Text/Regex.nm) recognizes only the
52685
+ * shorthands (`\d \D \w \W \s \S`), backreferences (`\1`-`\9`), the control
52686
+ * escapes (`\n \r \t`), and escaped metacharacters. Any OTHER engine escape
52687
+ — most notably a typo like `\z` — silently matches a literal letter. When a
52688
+ * Regex entry point's pattern argument is a static string literal, the check
52689
+ * can flag those as probable typos; dynamic runtime patterns can't be
52690
+ * validated this way.
52691
+ *
52692
+ * The literal arrives RAW (quotes included, Nomen escapes unresolved), so it
52693
+ * must first be decoded to the engine-visible bytes: a Nomen `"\n"` is a real
52694
+ * LF byte the engine treats as a literal, while `"\n"` reaches the engine as
52695
+ * a backslash + 'n' escape. Only the second spelling can carry a typo.
52696
+ */
52697
+ /** Decode a raw string-literal token (quotes included) to its engine-visible
52698
+ * bytes, resolving Nomen's own string escapes. Mirrors the pair-counting of
52699
+ * string_literal_length and the decode rules of decode_char_literal. */
52700
+ function decode_string_literal_bytes(raw) {
52701
+ const bytes = [];
52702
+ let i = 1;
52703
+ const end = raw.endsWith("\"") ? raw.length - 1 : raw.length;
52704
+ while (i < end) {
52705
+ if (raw[i] !== "\\") {
52706
+ const cp = raw.codePointAt(i) ?? 0;
52707
+ if (cp < 128) {
52708
+ bytes.push(cp);
52709
+ i += 1;
52710
+ } else if (cp < 2048) {
52711
+ bytes.push(192 | cp >> 6, 128 | cp & 63);
52712
+ i += 1;
52713
+ } else if (cp < 65536) {
52714
+ bytes.push(224 | cp >> 12, 128 | cp >> 6 & 63, 128 | cp & 63);
52715
+ i += 1;
52716
+ } else {
52717
+ bytes.push(240 | cp >> 18, 128 | cp >> 12 & 63, 128 | cp >> 6 & 63, 128 | cp & 63);
52718
+ i += 2;
52719
+ }
52720
+ continue;
52721
+ }
52722
+ const next = raw[i + 1];
52723
+ if (next === void 0) break;
52724
+ if (next === "x") {
52725
+ let j = i + 2;
52726
+ let value = 0;
52727
+ let digits = 0;
52728
+ while (j < end && digits < 2 && is_hex_digit(raw[j])) {
52729
+ value = value * 16 + hex_value(raw[j]);
52730
+ j += 1;
52731
+ digits += 1;
52732
+ }
52733
+ if (digits > 0) {
52734
+ bytes.push(value & 255);
52735
+ i = j;
52736
+ continue;
52737
+ }
52738
+ i += 2;
52739
+ continue;
52740
+ }
52741
+ if (next >= "0" && next <= "7") {
52742
+ let j = i + 1;
52743
+ let value = 0;
52744
+ let digits = 0;
52745
+ while (j < end && digits < 3 && raw[j] >= "0" && raw[j] <= "7") {
52746
+ value = value * 8 + (raw[j].charCodeAt(0) - 48);
52747
+ j += 1;
52748
+ digits += 1;
52749
+ }
52750
+ bytes.push(value & 255);
52751
+ i = j;
52752
+ continue;
52753
+ }
52754
+ switch (next) {
52755
+ case "\\":
52756
+ bytes.push(92);
52757
+ break;
52758
+ case "n":
52759
+ bytes.push(10);
52760
+ break;
52761
+ case "t":
52762
+ bytes.push(9);
52763
+ break;
52764
+ case "r":
52765
+ bytes.push(13);
52766
+ break;
52767
+ case "0":
52768
+ bytes.push(0);
52769
+ break;
52770
+ case "\"":
52771
+ bytes.push(34);
52772
+ break;
52773
+ case "'":
52774
+ bytes.push(39);
52775
+ break;
52776
+ default: bytes.push(next.charCodeAt(0));
52777
+ }
52778
+ i += 2;
52779
+ }
52780
+ return bytes;
52781
+ }
52782
+ function is_hex_digit(c) {
52783
+ return c >= "0" && c <= "9" || c >= "a" && c <= "f" || c >= "A" && c <= "F";
52784
+ }
52785
+ function hex_value(c) {
52786
+ if (c <= "9") return c.charCodeAt(0) - 48;
52787
+ if (c <= "F") return c.charCodeAt(0) - 55;
52788
+ return c.charCodeAt(0) - 87;
52789
+ }
52790
+ function is_ascii_letter(code) {
52791
+ return code >= 65 && code <= 90 || code >= 97 && code <= 122;
52792
+ }
52793
+ /** Scan the decoded pattern bytes for escapes outside the engine's known set
52794
+ * (`\d \D \w \W \s \S`, `\1`-`\9`, `\n \r \t`, escaped punctuation) and
52795
+ * return one warning message per probable typo. */
52796
+ function lint_regex_pattern(raw) {
52797
+ const bytes = decode_string_literal_bytes(raw);
52798
+ const messages = [];
52799
+ let i = 0;
52800
+ while (i < bytes.length) {
52801
+ if (bytes[i] !== 92) {
52802
+ i += 1;
52803
+ continue;
52804
+ }
52805
+ const escaped = bytes[i + 1];
52806
+ if (escaped === void 0) break;
52807
+ if (!(escaped === 100 || escaped === 68 || escaped === 119 || escaped === 87 || escaped === 115 || escaped === 83 || escaped === 110 || escaped === 114 || escaped === 116 || escaped >= 49 && escaped <= 57) && is_ascii_letter(escaped)) {
52808
+ const letter = String.fromCharCode(escaped);
52809
+ messages.push(`Regex pattern: unsupported escape '\\${letter}' matches a literal '${letter}' — supported escapes are \\d \\D \\w \\W \\s \\S \\1-\\9 \\n \\r \\t and escaped punctuation`);
52810
+ }
52811
+ i += 2;
52812
+ }
52813
+ return messages;
52814
+ }
52815
+ //#endregion
52477
52816
  //#region ../src/check/utils/string_mutation_scan.ts
52478
52817
  /**
52479
52818
  * Interprocedural no-mutation scan for borrow-position `to_string()` elision
@@ -52867,6 +53206,7 @@ function is_inside_core_method(status) {
52867
53206
  }
52868
53207
  function check_function_call(node, status, func, target_type, self_value, self_path) {
52869
53208
  set_resolved_function(node, func);
53209
+ lint_regex_entry_pattern(node, func, status);
52870
53210
  const access_scope = status.stack.at(-1);
52871
53211
  if (func.name === "#init") {
52872
53212
  const struct_name = target_type?.name || func.return_type.name;
@@ -53378,6 +53718,59 @@ function receiver_is_const(target_type, self_value, status) {
53378
53718
  const binding = status.values.findLast((v) => v.name === self_value);
53379
53719
  return !!binding && binding.declaration === "const" && !binding.type.is_ref;
53380
53720
  }
53721
+ /**
53722
+ * Regex entry points whose FIRST parameter is the pattern. When that argument
53723
+ * is a static string literal, flag engine escapes outside the known set as
53724
+ * probable typos (see regex_pattern_lint.ts) — a dynamic runtime pattern
53725
+ * can't be validated this way.
53726
+ */
53727
+ const REGEX_PATTERN_ENTRY_POINTS = /* @__PURE__ */ new Set([
53728
+ "test",
53729
+ "match",
53730
+ "count",
53731
+ "find",
53732
+ "replace_all",
53733
+ "captures",
53734
+ "test_ci",
53735
+ "match_ci",
53736
+ "find_ci",
53737
+ "captures_ci",
53738
+ "replace_all_ci"
53739
+ ]);
53740
+ /**
53741
+ * Lint a static string-literal pattern passed to a Regex entry point. Only
53742
+ * the library's own `Regex` struct qualifies (a user struct of the same name
53743
+ * with a same-named method is left alone), and only when the first argument
53744
+ * is a literal — anything else is a runtime value.
53745
+ */
53746
+ function lint_regex_entry_pattern(node, func, status) {
53747
+ if (!is_library_regex_func(func, status)) return;
53748
+ if (!REGEX_PATTERN_ENTRY_POINTS.has(func.name)) return;
53749
+ const pattern = node.params[0];
53750
+ if (!pattern || pattern.node_type !== "value") return;
53751
+ const raw = pattern.value;
53752
+ if (!raw.startsWith("\"")) return;
53753
+ if (!status.warnings) status.warnings = [];
53754
+ for (const message of lint_regex_pattern(raw)) status.warnings.push({
53755
+ message,
53756
+ start: pattern.start,
53757
+ line: 0,
53758
+ column: 0
53759
+ });
53760
+ }
53761
+ /**
53762
+ * Whether `func` is a method of the library's `Regex` struct. Two paths: the
53763
+ * method's enclosing struct is stamped on it for calls checked after the
53764
+ * struct gather (library bodies, later files), and for user call sites —
53765
+ * which are checked BEFORE the gather stamps `func.scope` — the callee is
53766
+ * matched against the gathered library struct's own methods.
53767
+ */
53768
+ function is_library_regex_func(func, status) {
53769
+ const scope = func.scope;
53770
+ if (scope && scope.node_type === "struct" && scope.is_library && scope.name === "Regex") return true;
53771
+ const regex_struct = status.structs.find((s) => s.is_library && s.name === "Regex");
53772
+ return !!regex_struct && regex_struct.functions.includes(func);
53773
+ }
53381
53774
  //#endregion
53382
53775
  //#region ../src/check/check_access_node.ts
53383
53776
  function check_access_node(node, status) {
@@ -59352,6 +59745,7 @@ function run_test_file(entry_path, lib_path, arch, audit = false, audit_runtime)
59352
59745
  const parsed = parse(input + "\n" + harness, library, resolved);
59353
59746
  if (parsed.errors.length) {
59354
59747
  result.ok = false;
59748
+ result.phase = "parse";
59355
59749
  result.crashed = parsed.errors.map((e) => `${e.message} (${e.line ?? "?"}:${e.column ?? "?"})`).join("\n");
59356
59750
  result.ms = performance.now() - start;
59357
59751
  return result;
@@ -59362,6 +59756,7 @@ function run_test_file(entry_path, lib_path, arch, audit = false, audit_runtime)
59362
59756
  });
59363
59757
  if (buildResult.errors && buildResult.errors.length) {
59364
59758
  result.ok = false;
59759
+ result.phase = "build";
59365
59760
  result.crashed = buildResult.errors.map((e) => `${e.message} (${e.line ?? "?"}:${e.column ?? "?"})`).join("\n");
59366
59761
  result.ms = performance.now() - start;
59367
59762
  return result;
@@ -59373,6 +59768,7 @@ function run_test_file(entry_path, lib_path, arch, audit = false, audit_runtime)
59373
59768
  const runtime_src = audit_runtime ? path.resolve(audit_runtime) : find_audit_runtime(path.dirname(resolved));
59374
59769
  if (!runtime_src || !fs.existsSync(runtime_src)) {
59375
59770
  result.ok = false;
59771
+ result.phase = "setup";
59376
59772
  result.crashed = "Audit enabled but audit_runtime.c was not found. Pass --audit-runtime <path>.";
59377
59773
  result.ms = performance.now() - start;
59378
59774
  return result;
@@ -59410,6 +59806,7 @@ function run_test_file(entry_path, lib_path, arch, audit = false, audit_runtime)
59410
59806
  } catch (err) {
59411
59807
  const stderr = err.stderr ? err.stderr.toString() : err.message ?? "";
59412
59808
  result.ok = false;
59809
+ result.phase = "link";
59413
59810
  result.crashed = `link failed: ${stderr.trim() || "clang error"}`;
59414
59811
  result.ms = performance.now() - start;
59415
59812
  return result;
@@ -59438,6 +59835,7 @@ function run_test_file(entry_path, lib_path, arch, audit = false, audit_runtime)
59438
59835
  result.benches = records.benches;
59439
59836
  result.other = records.other;
59440
59837
  result.crashed = crashed;
59838
+ if (crashed) result.phase = "run";
59441
59839
  result.ok = !(result.fails.length > 0 || result.tests.some((t) => t.failed > 0)) && crashed === void 0;
59442
59840
  result.ms = performance.now() - start;
59443
59841
  return result;
@@ -59465,7 +59863,8 @@ const C = {
59465
59863
  function report_file(result) {
59466
59864
  const rel = path.relative(process.cwd(), result.file);
59467
59865
  if (result.crashed && result.tests.length === 0 && result.fails.length === 0) {
59468
- console.log(` ${C.red("✗")} ${rel} ${C.red("(failed to build)")}`);
59866
+ const label = result.phase === "run" ? "(crashed before records)" : result.phase === "link" ? "(link failed)" : result.phase === "setup" ? "(setup failed)" : "(failed to build)";
59867
+ console.log(` ${C.red("✗")} ${rel} ${C.red(label)}`);
59469
59868
  console.log(C.red(result.crashed));
59470
59869
  return;
59471
59870
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "nomen-lang",
3
- "version": "0.6.1",
3
+ "version": "0.6.2",
4
4
  "description": "The CLI for the Nomen programming language.",
5
5
  "keywords": [],
6
6
  "license": "ISC",
package/src/test.ts CHANGED
@@ -40,6 +40,12 @@ export interface TestFileResult {
40
40
  // Set when the compiled binary crashed before finishing (e.g. a segfault
41
41
  // in a test). We still surface whatever records it managed to emit.
42
42
  crashed?: string;
43
+ /**
44
+ * The pipeline stage that produced `crashed`, so the report can name it
45
+ * accurately — a binary that BUILT fine and then segfaulted before
46
+ * emitting any record is not a build failure (see report_file).
47
+ */
48
+ phase?: "parse" | "build" | "setup" | "link" | "run";
43
49
  ms: number;
44
50
  }
45
51
 
@@ -289,6 +295,7 @@ export function run_test_file(
289
295
  const parsed = parse(source, library, resolved);
290
296
  if (parsed.errors.length) {
291
297
  result.ok = false;
298
+ result.phase = "parse";
292
299
  result.crashed = parsed.errors
293
300
  .map((e) => `${e.message} (${e.line ?? "?"}:${e.column ?? "?"})`)
294
301
  .join("\n");
@@ -302,6 +309,7 @@ export function run_test_file(
302
309
  });
303
310
  if (buildResult.errors && buildResult.errors.length) {
304
311
  result.ok = false;
312
+ result.phase = "build";
305
313
  result.crashed = buildResult.errors
306
314
  .map((e) => `${e.message} (${e.line ?? "?"}:${e.column ?? "?"})`)
307
315
  .join("\n");
@@ -321,6 +329,7 @@ export function run_test_file(
321
329
  : find_audit_runtime(path.dirname(resolved));
322
330
  if (!runtime_src || !fs.existsSync(runtime_src)) {
323
331
  result.ok = false;
332
+ result.phase = "setup";
324
333
  result.crashed =
325
334
  "Audit enabled but audit_runtime.c was not found. Pass --audit-runtime <path>.";
326
335
  result.ms = performance.now() - start;
@@ -367,6 +376,7 @@ export function run_test_file(
367
376
  } catch (err: any) {
368
377
  const stderr = err.stderr ? err.stderr.toString() : (err.message ?? "");
369
378
  result.ok = false;
379
+ result.phase = "link";
370
380
  result.crashed = `link failed: ${stderr.trim() || "clang error"}`;
371
381
  result.ms = performance.now() - start;
372
382
  return result;
@@ -399,6 +409,7 @@ export function run_test_file(
399
409
  result.benches = records.benches;
400
410
  result.other = records.other;
401
411
  result.crashed = crashed;
412
+ if (crashed) result.phase = "run";
402
413
  // A file "passes" when it reported no failed asserts and didn't crash.
403
414
  const failed = result.fails.length > 0 || result.tests.some((t) => t.failed > 0);
404
415
  result.ok = !failed && crashed === undefined;
@@ -435,7 +446,18 @@ const C = {
435
446
  export function report_file(result: TestFileResult): void {
436
447
  const rel = path.relative(process.cwd(), result.file);
437
448
  if (result.crashed && result.tests.length === 0 && result.fails.length === 0) {
438
- console.log(` ${C.red("✗")} ${rel} ${C.red("(failed to build)")}`);
449
+ // Name the stage that actually failed. A binary that linked fine and
450
+ // then died before emitting any record (e.g. SIGSEGV) is NOT a build
451
+ // failure — conflating the two sent debugging to the wrong layer.
452
+ const label =
453
+ result.phase === "run"
454
+ ? "(crashed before records)"
455
+ : result.phase === "link"
456
+ ? "(link failed)"
457
+ : result.phase === "setup"
458
+ ? "(setup failed)"
459
+ : "(failed to build)";
460
+ console.log(` ${C.red("✗")} ${rel} ${C.red(label)}`);
439
461
  console.log(C.red(result.crashed));
440
462
  return;
441
463
  }