nomen-lang 0.2.3 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -22,7 +22,8 @@ pub struct Regex {
22
22
  break
23
23
  }
24
24
  }
25
- if match_here(pattern, 0, pat_len, input, i, len) >= 0 {
25
+ var caps = List<int>()
26
+ if match_here(pattern, 0, pat_len, input, i, len, ref caps) >= 0 {
26
27
  return true
27
28
  }
28
29
  }
@@ -46,7 +47,8 @@ pub struct Regex {
46
47
  break
47
48
  }
48
49
  }
49
- var int end = match_here(pattern, 0, pat_len, input, i, len)
50
+ var caps = List<int>()
51
+ var int end = match_here(pattern, 0, pat_len, input, i, len, ref caps)
50
52
  if end >= 0 {
51
53
  var int j = i
52
54
  while j < end; j += 1 {
@@ -58,6 +60,80 @@ pub struct Regex {
58
60
  return match_result
59
61
  }
60
62
 
63
+ // Try to match pattern anywhere in input. On a match, append the whole
64
+ // match as group 0 followed by capture groups 1..N to `out` ("" for a
65
+ // group that did not participate in the match) and return true. Return
66
+ // false with `out` untouched when nothing matches. A group that repeats
67
+ // (`(a)+`) reports its LAST iteration.
68
+ pub func captures = (string pattern, string input, ref List<string> dst, out bool) { var int pat_len = pattern.length
69
+ var int len = input.length
70
+ var string charset = first_byte_set(pattern, pat_len)
71
+ var int clen = charset.length
72
+ var int groups = capture_count(pattern)
73
+ var i = 0
74
+ while i <= len; i += 1 {
75
+ if clen > 0 && i < len {
76
+ i = find_next_byte(input, i, len, charset)
77
+ if i > len {
78
+ break
79
+ }
80
+ }
81
+ var caps = List<int>()
82
+ var int end = match_here(pattern, 0, pat_len, input, i, len, ref caps)
83
+ if end >= 0 {
84
+ var string whole = input.substring(i, end)
85
+ dst.push(whole)
86
+ var int g = 1
87
+ while g <= groups; g += 1 {
88
+ var int at = last_capture_index(caps, g)
89
+ if at >= 0 {
90
+ var int s = caps.at(at + 1)
91
+ if s >= 0 {
92
+ var string cap = input.substring(s, caps.at(at + 2))
93
+ dst.push(cap)
94
+ } else {
95
+ dst.push("")
96
+ }
97
+ } else {
98
+ dst.push("")
99
+ }
100
+ }
101
+ return true
102
+ }
103
+ }
104
+ return false
105
+ }
106
+
107
+ // Find the first match of pattern in input and fill `dst` with where it
108
+ // sits: .found with byte offsets .start/.end, the matched .text, and
109
+ // .length (all zero/empty when .found is false — mvzr-style positions
110
+ // for callers doing their own index bookkeeping). `dst` is fully reset
111
+ // first, so a reused RegexMatch carries nothing over. Scans every
112
+ // position: the first-byte prefilter would skip nullable patterns'
113
+ // empty matches at skipped positions, reporting a wrong .start.
114
+ pub func find = (string pattern, string input, ref RegexMatch dst) {
115
+ dst.found = false
116
+ dst.start = 0
117
+ dst.end = 0
118
+ dst.length = 0
119
+ dst.text = ""
120
+ var int pat_len = pattern.length
121
+ var int len = input.length
122
+ var i = 0
123
+ while i <= len; i += 1 {
124
+ var caps = List<int>()
125
+ var int end = match_here(pattern, 0, pat_len, input, i, len, ref caps)
126
+ if end >= 0 {
127
+ dst.found = true
128
+ dst.start = i
129
+ dst.end = end
130
+ dst.length = end - i
131
+ dst.text = input.substring(i, end)
132
+ return
133
+ }
134
+ }
135
+ }
136
+
61
137
  // Count the number of non-overlapping occurrences of pattern in input.
62
138
  pub func count = (string pattern, string input, out int) {
63
139
  var int pat_len = pattern.length
@@ -73,7 +149,8 @@ pub struct Regex {
73
149
  break
74
150
  }
75
151
  }
76
- var int end = match_here(pattern, 0, pat_len, input, pos, len)
152
+ var caps = List<int>()
153
+ var int end = match_here(pattern, 0, pat_len, input, pos, len, ref caps)
77
154
  if end >= 0 {
78
155
  total += 1
79
156
  if end == pos {
@@ -89,8 +166,7 @@ pub struct Regex {
89
166
  }
90
167
 
91
168
  // Replace all non-overlapping occurrences of pattern in input with replacement.
92
- pub func replace_all = (string pattern, string input, string replacement, out string) {
93
- var sb = StringBuilder()
169
+ pub func replace_all = (string pattern, string input, string replacement, out string) { var sb = StringBuilder()
94
170
  var int pat_len = pattern.length
95
171
  var int len = input.length
96
172
  var string charset = first_byte_set(pattern, pat_len)
@@ -108,7 +184,8 @@ pub struct Regex {
108
184
  break
109
185
  }
110
186
  }
111
- var int end = match_here(pattern, 0, pat_len, input, pos, len)
187
+ var caps = List<int>()
188
+ var int end = match_here(pattern, 0, pat_len, input, pos, len, ref caps)
112
189
  if end >= 0 {
113
190
  sb.append_string(replacement)
114
191
  if end == pos {
@@ -124,6 +201,172 @@ pub struct Regex {
124
201
  }
125
202
  return sb.to_string()
126
203
  }
204
+
205
+ // Whether pattern matches anywhere in input, ignoring ASCII case.
206
+ pub func test_ci = (string pattern, string input, out bool) {
207
+ var string folded = fold_pattern(pattern)
208
+ return Regex.test(folded, input)
209
+ }
210
+
211
+ // First match of pattern in input, ignoring ASCII case ("" if none).
212
+ pub func match_ci = (string pattern, string input, move out string) {
213
+ var string folded = fold_pattern(pattern)
214
+ return Regex.match(folded, input)
215
+ }
216
+
217
+ // find, ignoring ASCII case.
218
+ pub func find_ci = (string pattern, string input, ref RegexMatch dst) {
219
+ var string folded = fold_pattern(pattern)
220
+ Regex.find(folded, input, ref dst)
221
+ }
222
+
223
+ // captures, ignoring ASCII case.
224
+ pub func captures_ci = (string pattern, string input, ref List<string> dst, out bool) {
225
+ var string folded = fold_pattern(pattern)
226
+ return Regex.captures(folded, input, ref dst)
227
+ }
228
+
229
+ // replace_all, ignoring ASCII case.
230
+ pub func replace_all_ci = (string pattern, string input, string replacement, out string) {
231
+ var string folded = fold_pattern(pattern)
232
+ return Regex.replace_all(folded, input, replacement)
233
+ }
234
+ }
235
+
236
+ // Rewrite pattern so the case-sensitive engine matches case-insensitively:
237
+ // literals become both-case classes ([kK]), class members and ranges gain
238
+ // their case counterparts ([a-z] -> [a-zA-Z]). Structural syntax — groups,
239
+ // alternation, quantifiers, anchors, escaped punctuation — is copied
240
+ // through unchanged.
241
+ func fold_pattern = (string pattern, move out string) {
242
+ var int n = pattern.length
243
+ var sb = StringBuilder()
244
+ var int i = 0
245
+ while i < n; i += 1 {
246
+ var char c = pattern.at(i)
247
+ if (c as int) == 92 {
248
+ // Escaped char: shorthands and backrefs keep their backslash;
249
+ // other escaped letters become both-case classes; escaped
250
+ // punctuation is copied literally.
251
+ if i + 1 < n {
252
+ var char e = pattern.at(i + 1)
253
+ if is_shorthand(e) || is_backref_digit(e) {
254
+ sb.append_char(c)
255
+ sb.append_char(e)
256
+ } else {
257
+ if is_alpha_byte(e) {
258
+ sb.append_char('[')
259
+ sb.append_char(e)
260
+ sb.append_char(swap_case_byte(e))
261
+ sb.append_char(']')
262
+ } else {
263
+ sb.append_char(c)
264
+ sb.append_char(e)
265
+ }
266
+ }
267
+ i += 1
268
+ } else {
269
+ sb.append_char(c)
270
+ }
271
+ } else {
272
+ if c == '[' {
273
+ var int ce = atom_end_pos(pattern, i, n)
274
+ fold_class(pattern, i, ce, ref sb)
275
+ i = ce - 1
276
+ } else {
277
+ if is_alpha_byte(c) {
278
+ sb.append_char('[')
279
+ sb.append_char(c)
280
+ sb.append_char(swap_case_byte(c))
281
+ sb.append_char(']')
282
+ } else {
283
+ sb.append_char(c)
284
+ }
285
+ }
286
+ }
287
+ }
288
+ return sb.to_string()
289
+ }
290
+
291
+ // fold a character class [cs..ce): copy its contents, adding the case
292
+ // counterpart of every letter member and mirroring letter ranges.
293
+ func fold_class = (string pattern, int cs, int ce, ref StringBuilder dst) {
294
+ dst.append_char('[')
295
+ var int i = cs + 1
296
+ if i < ce - 1 {
297
+ if pattern.at(i) == '^' {
298
+ dst.append_char('^')
299
+ i += 1
300
+ }
301
+ }
302
+ var int content_end = ce - 1
303
+ while i < content_end; i += 1 {
304
+ var char c = pattern.at(i)
305
+ if (c as int) == 92 {
306
+ dst.append_char(c)
307
+ if i + 1 < content_end {
308
+ var char e = pattern.at(i + 1)
309
+ dst.append_char(e)
310
+ // A shorthand keeps its backslash; other escaped letters
311
+ // gain their case counterpart.
312
+ if is_alpha_byte(e) && !is_shorthand(e) {
313
+ dst.append_char(swap_case_byte(e))
314
+ }
315
+ i += 1
316
+ }
317
+ } else {
318
+ if i + 2 < content_end {
319
+ if pattern.at(i + 1) == '-' {
320
+ var char lo = c
321
+ var char hi = pattern.at(i + 2)
322
+ dst.append_char(lo)
323
+ dst.append_char('-')
324
+ dst.append_char(hi)
325
+ if is_alpha_byte(lo) {
326
+ if is_alpha_byte(hi) {
327
+ dst.append_char(swap_case_byte(lo))
328
+ dst.append_char('-')
329
+ dst.append_char(swap_case_byte(hi))
330
+ }
331
+ }
332
+ i += 2
333
+ } else {
334
+ if is_alpha_byte(c) {
335
+ dst.append_char(c)
336
+ dst.append_char(swap_case_byte(c))
337
+ } else {
338
+ dst.append_char(c)
339
+ }
340
+ }
341
+ } else {
342
+ if is_alpha_byte(c) {
343
+ dst.append_char(c)
344
+ dst.append_char(swap_case_byte(c))
345
+ } else {
346
+ dst.append_char(c)
347
+ }
348
+ }
349
+ }
350
+ }
351
+ dst.append_char(']')
352
+ }
353
+
354
+ // ASCII letter test for pattern characters.
355
+ func is_alpha_byte = (char c, out bool) {
356
+ var int code = c as int
357
+ return (code > 64 && code < 91) || (code > 96 && code < 123)
358
+ }
359
+
360
+ // The ASCII letter with the opposite case (unchanged for non-letters).
361
+ func swap_case_byte = (char c, out char) {
362
+ var int code = c as int
363
+ if code > 64 && code < 91 {
364
+ return (code + 32) as char
365
+ }
366
+ if code > 96 && code < 123 {
367
+ return (code - 32) as char
368
+ }
369
+ return c
127
370
  }
128
371
 
129
372
  // Scan input[pos..len) for the next byte that appears in charset.
@@ -219,9 +462,21 @@ func alt_first_bytes = (string pattern, int start, int end, ref StringBuilder sb
219
462
  while i < end {
220
463
  var char c = pattern.at(i)
221
464
  if (c as int) == 92 {
222
- // escaped literal: first byte is the escaped char
465
+ // escaped literal: first byte is the escaped char — unless it's
466
+ // a backreference (depends on a capture) or a complement
467
+ // shorthand (not an inclusion set), which disable the prefilter.
223
468
  if i + 1 < end {
224
- sb.append_char(pattern.at(i + 1))
469
+ var char e = pattern.at(i + 1)
470
+ if is_backref_digit(e) {
471
+ return -1
472
+ }
473
+ if is_shorthand(e) {
474
+ if shorthand_members(e, ref sb) < 0 {
475
+ return -1
476
+ }
477
+ return 0
478
+ }
479
+ sb.append_char(e)
225
480
  }
226
481
  return 0
227
482
  }
@@ -248,7 +503,8 @@ func alt_first_bytes = (string pattern, int start, int end, ref StringBuilder sb
248
503
  return -1
249
504
  }
250
505
  // A '?'/'*' quantifier makes the group nullable — the first byte
251
- // can also come from whatever follows it.
506
+ // can also come from whatever follows it. A second '?' marks the
507
+ // quantifier lazy and is skipped either way.
252
508
  var int after = a_end
253
509
  var int nullable = 0
254
510
  if after < end {
@@ -256,6 +512,11 @@ func alt_first_bytes = (string pattern, int start, int end, ref StringBuilder sb
256
512
  if q == '?' || q == '*' {
257
513
  nullable = 1
258
514
  after += 1
515
+ if after < end {
516
+ if pattern.at(after) == '?' {
517
+ after += 1
518
+ }
519
+ }
259
520
  }
260
521
  }
261
522
  if nullable == 0 {
@@ -317,6 +578,8 @@ func add_class_chars = (string pattern, int start, int end, ref StringBuilder sb
317
578
  // Return contract: a successful match never reads past in_len, so the result
318
579
  // is always <= in_len (and -1 on failure). This lets callers prove an index
319
580
  // bounded by the result is also bounded by the input length.
581
+ // `caps` collects capture spans as (group_index, start, end) triples in
582
+ // match order; a later entry for the same group supersedes an earlier one.
320
583
  func match_here = (
321
584
  string pattern,
322
585
  int pat_pos,
@@ -324,6 +587,7 @@ func match_here = (
324
587
  string input,
325
588
  int in_pos,
326
589
  int in_len,
590
+ ref List<int> caps,
327
591
  out int: out <= in_len
328
592
  ) {
329
593
  // Handle top-level alternation: a '|' at bracket/paren depth 0 splits the
@@ -349,11 +613,11 @@ func match_here = (
349
613
  }
350
614
  if (c as int) == 124 {
351
615
  if depth == 0 {
352
- var int left = match_here(pattern, pat_pos, i, input, in_pos, in_len)
616
+ var int left = match_here(pattern, pat_pos, i, input, in_pos, in_len, ref caps)
353
617
  if left >= 0 {
354
618
  return left
355
619
  }
356
- return match_here(pattern, i + 1, pat_end, input, in_pos, in_len)
620
+ return match_here(pattern, i + 1, pat_end, input, in_pos, in_len, ref caps)
357
621
  }
358
622
  }
359
623
  i += 1
@@ -373,38 +637,48 @@ func match_here = (
373
637
  if in_pos != 0 {
374
638
  return -1
375
639
  }
376
- return match_here(pattern, pat_pos + 1, pat_end, input, in_pos, in_len)
640
+ return match_here(pattern, pat_pos + 1, pat_end, input, in_pos, in_len, ref caps)
377
641
  }
378
642
  if pc == '$' {
379
643
  if in_pos != in_len {
380
644
  return -1
381
645
  }
382
- return match_here(pattern, pat_pos + 1, pat_end, input, in_pos, in_len)
646
+ return match_here(pattern, pat_pos + 1, pat_end, input, in_pos, in_len, ref caps)
383
647
  }
384
648
 
385
649
  // Determine the bounds of the current atom.
386
650
  var int a_end = atom_end_pos(pattern, pat_pos, pat_end)
387
651
 
388
- // Check for a quantifier following the atom.
652
+ // Check for a quantifier following the atom. A trailing '?' marks the
653
+ // quantifier LAZY (shortest match first) — the rest of the pattern then
654
+ // starts after both symbols.
389
655
  if a_end < pat_end {
390
656
  var char q = pattern.at(a_end)
391
- if q == '*' {
392
- return match_star(pattern, pat_pos, a_end, pat_end, input, in_pos, 0, in_len)
393
- }
394
- if q == '+' {
395
- return match_star(pattern, pat_pos, a_end, pat_end, input, in_pos, 1, in_len)
396
- }
397
- if q == '?' {
398
- return match_optional(pattern, pat_pos, a_end, pat_end, input, in_pos, in_len)
657
+ if q == '*' || q == '+' || q == '?' {
658
+ var int lazy = 0
659
+ var int rest = a_end + 1
660
+ if rest < pat_end {
661
+ if pattern.at(rest) == '?' {
662
+ lazy = 1
663
+ rest += 1
664
+ }
665
+ }
666
+ if q == '*' {
667
+ return match_star(pattern, pat_pos, a_end, rest, pat_end, input, in_pos, 0, lazy, in_len, ref caps)
668
+ }
669
+ if q == '+' {
670
+ return match_star(pattern, pat_pos, a_end, rest, pat_end, input, in_pos, 1, lazy, in_len, ref caps)
671
+ }
672
+ return match_optional(pattern, pat_pos, a_end, rest, pat_end, input, in_pos, lazy, in_len, ref caps)
399
673
  }
400
674
  }
401
675
 
402
676
  // No quantifier: match the atom once, then the rest of the pattern.
403
- var int next = match_single_atom(pattern, pat_pos, a_end, input, in_pos, in_len)
677
+ var int next = match_single_atom(pattern, pat_pos, a_end, input, in_pos, in_len, ref caps)
404
678
  if next < 0 {
405
679
  return -1
406
680
  }
407
- return match_here(pattern, a_end, pat_end, input, next, in_len)
681
+ return match_here(pattern, a_end, pat_end, input, next, in_len, ref caps)
408
682
  }
409
683
 
410
684
  // Return the position just past the atom that starts at pat_pos. Handles
@@ -460,6 +734,7 @@ func match_single_atom = (
460
734
  string input,
461
735
  int in_pos,
462
736
  int in_len,
737
+ ref List<int> caps,
463
738
  out int
464
739
  ) {
465
740
  var char pc = pattern.at(atom_start)
@@ -481,12 +756,42 @@ func match_single_atom = (
481
756
  return -1
482
757
  }
483
758
  if pc == '(' {
484
- // Match the group's inner content.
485
- return match_here(pattern, atom_start + 1, atom_end - 1, input, in_pos, in_len)
759
+ // Match the group's inner content, recording the group's span on
760
+ // success. On failure, restore the span this path started with (a
761
+ // repeated group's earlier iterations survive a failed extra trial)
762
+ // — or mark the group unset when it had no earlier span here.
763
+ var int g = group_index_at(pattern, atom_start)
764
+ var int at = last_capture_index(caps, g)
765
+ var int old_start = -1
766
+ var int old_end = -1
767
+ if at >= 0 {
768
+ old_start = caps.at(at + 1)
769
+ old_end = caps.at(at + 2)
770
+ }
771
+ var int next = match_here(pattern, atom_start + 1, atom_end - 1, input, in_pos, in_len, ref caps)
772
+ if next >= 0 {
773
+ record_capture(ref caps, g, in_pos, next)
774
+ return next
775
+ }
776
+ record_capture(ref caps, g, old_start, old_end)
777
+ return -1
486
778
  }
487
779
  if (pc as int) == 92 {
488
780
  if atom_start + 1 < atom_end {
489
781
  var char ec = pattern.at(atom_start + 1)
782
+ if is_backref_digit(ec) {
783
+ // \1..\9: match the same text capture group N matched.
784
+ return match_backref(input, backref_group(ec), in_pos, in_len, ref caps)
785
+ }
786
+ if is_shorthand(ec) {
787
+ // \d \s \w and complements: class membership by name.
788
+ if in_pos < in_len {
789
+ if matches_shorthand(ec, input.at(in_pos)) {
790
+ return in_pos + 1
791
+ }
792
+ }
793
+ return -1
794
+ }
490
795
  if in_pos < in_len {
491
796
  if input.at(in_pos) == ec {
492
797
  return in_pos + 1
@@ -527,25 +832,38 @@ func char_matches_atom = (
527
832
  }
528
833
  if (pc as int) == 92 {
529
834
  if atom_start + 1 < atom_end {
530
- return ic == pattern.at(atom_start + 1)
835
+ var char ec = pattern.at(atom_start + 1)
836
+ if is_backref_digit(ec) {
837
+ // A backreference is a multi-byte atom; the greedy
838
+ // single-character loop never consults this for it.
839
+ return false
840
+ }
841
+ if is_shorthand(ec) {
842
+ return matches_shorthand(ec, ic)
843
+ }
844
+ return ic == ec
531
845
  }
532
846
  return false
533
847
  }
534
848
  return pc == ic
535
849
  }
536
850
 
537
- // Greedily match the atom [atom_start..atom_end) followed by the rest of the
538
- // pattern pattern[atom_end+1..pat_end). min_count is 0 for '*' and 1 for '+'.
539
- // Single-character atoms are handled iteratively; groups recurse.
851
+ // Match the atom [atom_start..atom_end) followed by the rest of the pattern
852
+ // pattern[rest..pat_end). min_count is 0 for '*' and 1 for '+'; lazy picks
853
+ // the shortest repetition first instead of the longest. Single-character
854
+ // atoms are handled iteratively; groups recurse.
540
855
  func match_star = (
541
856
  string pattern,
542
857
  int atom_start,
543
858
  int atom_end,
859
+ int rest,
544
860
  int pat_end,
545
861
  string input,
546
862
  int in_pos,
547
863
  int min_count,
864
+ int lazy,
548
865
  int in_len,
866
+ ref List<int> caps,
549
867
  out int
550
868
  ) {
551
869
  var char pc = pattern.at(atom_start)
@@ -554,13 +872,36 @@ func match_star = (
554
872
  pattern,
555
873
  atom_start,
556
874
  atom_end,
875
+ rest,
557
876
  pat_end,
558
877
  input,
559
878
  in_pos,
560
879
  min_count,
880
+ lazy,
561
881
  in_len,
882
+ ref caps,
562
883
  )
563
884
  }
885
+ if (pc as int) == 92 {
886
+ if atom_start + 1 < atom_end {
887
+ if is_backref_digit(pattern.at(atom_start + 1)) {
888
+ // A backreference matches a whole span per iteration — the
889
+ // single-character greedy loop below can't apply.
890
+ return match_star_backref(
891
+ pattern,
892
+ atom_start,
893
+ rest,
894
+ pat_end,
895
+ input,
896
+ in_pos,
897
+ min_count,
898
+ lazy,
899
+ in_len,
900
+ ref caps,
901
+ )
902
+ }
903
+ }
904
+ }
564
905
 
565
906
  // Greedily consume the longest run of matching characters.
566
907
  var int max_pos = in_pos
@@ -571,10 +912,21 @@ func match_star = (
571
912
  break
572
913
  }
573
914
  }
915
+ if lazy > 0 {
916
+ // Lazy: extend from the minimum until the rest matches.
917
+ var int try_pos = in_pos + min_count
918
+ while try_pos <= max_pos; try_pos += 1 {
919
+ var int r = match_here(pattern, rest, pat_end, input, try_pos, in_len, ref caps)
920
+ if r >= 0 {
921
+ return r
922
+ }
923
+ }
924
+ return -1
925
+ }
574
926
  // Backtrack from the longest match down to the minimum, trying the rest.
575
927
  var int try_pos = max_pos
576
928
  while try_pos >= in_pos + min_count; try_pos -= 1 {
577
- var int r = match_here(pattern, atom_end + 1, pat_end, input, try_pos, in_len)
929
+ var int r = match_here(pattern, rest, pat_end, input, try_pos, in_len, ref caps)
578
930
  if r >= 0 {
579
931
  return r
580
932
  }
@@ -582,20 +934,56 @@ func match_star = (
582
934
  return -1
583
935
  }
584
936
 
585
- // Recursive greedy backtracking for group atoms quantified with '*' or '+'.
937
+ // Recursive backtracking for group atoms quantified with '*' or '+'.
586
938
  func match_star_group = (
587
939
  string pattern,
588
940
  int atom_start,
589
941
  int atom_end,
942
+ int rest,
590
943
  int pat_end,
591
944
  string input,
592
945
  int in_pos,
593
946
  int min_count,
947
+ int lazy,
594
948
  int in_len,
949
+ ref List<int> caps,
595
950
  out int
596
951
  ) {
952
+ if lazy > 0 {
953
+ // Lazy: stop repeating first (minimum met), then try one more
954
+ // iteration before giving up.
955
+ if min_count <= 0 {
956
+ var int stop = match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
957
+ if stop >= 0 {
958
+ return stop
959
+ }
960
+ }
961
+ var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len, ref caps)
962
+ if next >= 0 {
963
+ if next != in_pos {
964
+ var new_min = 0
965
+ if min_count > 0 {
966
+ new_min = min_count - 1
967
+ }
968
+ return match_star_group(
969
+ pattern,
970
+ atom_start,
971
+ atom_end,
972
+ rest,
973
+ pat_end,
974
+ input,
975
+ next,
976
+ new_min,
977
+ lazy,
978
+ in_len,
979
+ ref caps,
980
+ )
981
+ }
982
+ }
983
+ return -1
984
+ }
597
985
  // Try matching the group one more time (greedy).
598
- var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len)
986
+ var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len, ref caps)
599
987
  if next >= 0 {
600
988
  if next != in_pos {
601
989
  var new_min = 0
@@ -606,11 +994,14 @@ func match_star_group = (
606
994
  pattern,
607
995
  atom_start,
608
996
  atom_end,
997
+ rest,
609
998
  pat_end,
610
999
  input,
611
1000
  next,
612
1001
  new_min,
1002
+ lazy,
613
1003
  in_len,
1004
+ ref caps,
614
1005
  )
615
1006
  if r >= 0 {
616
1007
  return r
@@ -619,30 +1010,45 @@ func match_star_group = (
619
1010
  }
620
1011
  // Stop repeating and try the rest of the pattern, if the minimum is met.
621
1012
  if min_count <= 0 {
622
- return match_here(pattern, atom_end + 1, pat_end, input, in_pos, in_len)
1013
+ return match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
623
1014
  }
624
1015
  return -1
625
1016
  }
626
1017
 
627
- // Match the atom optionally ('?'), preferring to include it.
1018
+ // Match the atom optionally ('?'), preferring to include it unless lazy.
628
1019
  func match_optional = (
629
1020
  string pattern,
630
1021
  int atom_start,
631
1022
  int atom_end,
1023
+ int rest,
632
1024
  int pat_end,
633
1025
  string input,
634
1026
  int in_pos,
1027
+ int lazy,
635
1028
  int in_len,
1029
+ ref List<int> caps,
636
1030
  out int
637
1031
  ) {
638
- var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len)
1032
+ if lazy > 0 {
1033
+ // Lazy: try without the atom first.
1034
+ var int skip = match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
1035
+ if skip >= 0 {
1036
+ return skip
1037
+ }
1038
+ var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len, ref caps)
1039
+ if next >= 0 {
1040
+ return match_here(pattern, rest, pat_end, input, next, in_len, ref caps)
1041
+ }
1042
+ return -1
1043
+ }
1044
+ var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len, ref caps)
639
1045
  if next >= 0 {
640
- var int r = match_here(pattern, atom_end + 1, pat_end, input, next, in_len)
1046
+ var int r = match_here(pattern, rest, pat_end, input, next, in_len, ref caps)
641
1047
  if r >= 0 {
642
1048
  return r
643
1049
  }
644
1050
  }
645
- return match_here(pattern, atom_end + 1, pat_end, input, in_pos, in_len)
1051
+ return match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
646
1052
  }
647
1053
 
648
1054
  // Report whether input.at(in_pos) belongs to the character class whose '[' is
@@ -677,8 +1083,16 @@ func char_in_class = (
677
1083
  var char c = pattern.at(i)
678
1084
  if (c as int) == 92 {
679
1085
  if i + 1 < content_end {
680
- if ic == pattern.at(i + 1) {
681
- found = true
1086
+ var char e = pattern.at(i + 1)
1087
+ if is_shorthand(e) {
1088
+ // \d \s \w (and complements) inside a class.
1089
+ if matches_shorthand(e, ic) {
1090
+ found = true
1091
+ }
1092
+ } else {
1093
+ if ic == e {
1094
+ found = true
1095
+ }
682
1096
  }
683
1097
  }
684
1098
  i = i + 2
@@ -716,3 +1130,275 @@ func char_in_class = (
716
1130
  }
717
1131
  return found
718
1132
  }
1133
+
1134
+ // The number of capture groups in the whole pattern: every '(' that is not
1135
+ // escaped and not inside a character class (POSIX ERE has no non-capturing
1136
+ // groups).
1137
+ func capture_count = (string pattern, out int) {
1138
+ var int n = pattern.length
1139
+ var int count = 0
1140
+ var int i = 0
1141
+ while i < n; i += 1 {
1142
+ var char c = pattern.at(i)
1143
+ if (c as int) == 92 {
1144
+ i += 1
1145
+ } else {
1146
+ if c == '[' {
1147
+ i = atom_end_pos(pattern, i, n) - 1
1148
+ } else {
1149
+ if c == '(' {
1150
+ count += 1
1151
+ }
1152
+ }
1153
+ }
1154
+ }
1155
+ return count
1156
+ }
1157
+
1158
+ // The capture-group index (1-based) of the group whose '(' opens at
1159
+ // open_pos: one more than the number of capture opens before it in the
1160
+ // pattern (the whole match is group 0).
1161
+ func group_index_at = (string pattern, int open_pos, out int) {
1162
+ var int count = 0
1163
+ var int i = 0
1164
+ while i < open_pos; i += 1 {
1165
+ var char c = pattern.at(i)
1166
+ if (c as int) == 92 {
1167
+ i += 1
1168
+ } else {
1169
+ if c == '[' {
1170
+ i = atom_end_pos(pattern, i, open_pos) - 1
1171
+ } else {
1172
+ if c == '(' {
1173
+ count += 1
1174
+ }
1175
+ }
1176
+ }
1177
+ }
1178
+ return count + 1
1179
+ }
1180
+
1181
+ // Record a capture span (start, end) for the group whose '(' opens at
1182
+ // group_pos. A later span for the same group supersedes the earlier one, so
1183
+ // a repeated group reports its last iteration; a failed group records
1184
+ // (-1, -1) so stale spans from an abandoned attempt never surface.
1185
+ func record_capture = (ref List<int> caps, int group, int start, int end) {
1186
+ var int at = last_capture_index(caps, group)
1187
+ if at >= 0 {
1188
+ caps.set(at + 1, start)
1189
+ caps.set(at + 2, end)
1190
+ } else {
1191
+ caps.push(group)
1192
+ caps.push(start)
1193
+ caps.push(end)
1194
+ }
1195
+ }
1196
+
1197
+ // The log index of the LAST entry for `group` in caps, or -1 when the group
1198
+ // has no entry. Entries are (group, start, end) triples.
1199
+ func last_capture_index = (List<int> caps, int group, out int) {
1200
+ var int best = -1
1201
+ var int i = 0
1202
+ while i < caps.length; i += 3 {
1203
+ if caps.at(i) == group {
1204
+ best = i
1205
+ }
1206
+ }
1207
+ return best
1208
+ }
1209
+
1210
+ /**
1211
+ * A Regex.find result: whether the pattern matched, where, and what
1212
+ **/
1213
+ pub struct RegexMatch {
1214
+ var bool found = false
1215
+ var int start = 0
1216
+ var int end = 0
1217
+ var int length = 0
1218
+ var string text = ""
1219
+ }
1220
+
1221
+ // Whether the escaped character c (the byte after a backslash) is a
1222
+ // backreference (\1..\9; \0 is the whole-match escape, not supported here).
1223
+ func is_backref_digit = (char c, out bool) {
1224
+ var int code = c as int
1225
+ return code >= 49 && code <= 57
1226
+ }
1227
+
1228
+ // The capture group a backreference refers to: '1' -> 1 .. '9' -> 9.
1229
+ func backref_group = (char c, out int) {
1230
+ return (c as int) - 48
1231
+ }
1232
+
1233
+ // Match the text capture group `group` last matched, at in_pos. Returns the
1234
+ // position after the span, or -1 when the group has no span here (it did not
1235
+ // participate) or the text differs.
1236
+ func match_backref = (string input, int group, int in_pos, int in_len, ref List<int> caps, out int) {
1237
+ var int at = last_capture_index(caps, group)
1238
+ if at < 0 {
1239
+ return -1
1240
+ }
1241
+ var int s = caps.at(at + 1)
1242
+ if s < 0 {
1243
+ return -1
1244
+ }
1245
+ var int len = caps.at(at + 2) - s
1246
+ if in_pos + len > in_len {
1247
+ return -1
1248
+ }
1249
+ var int i = 0
1250
+ while i < len; i += 1 {
1251
+ if input.at(in_pos + i) != input.at(s + i) {
1252
+ return -1
1253
+ }
1254
+ }
1255
+ return in_pos + len
1256
+ }
1257
+
1258
+ // Greedy/lazy backtracking for a backreference atom quantified with '*' or
1259
+ // '+' (each iteration consumes the referenced span).
1260
+ func match_star_backref = (
1261
+ string pattern,
1262
+ int atom_start,
1263
+ int rest,
1264
+ int pat_end,
1265
+ string input,
1266
+ int in_pos,
1267
+ int min_count,
1268
+ int lazy,
1269
+ int in_len,
1270
+ ref List<int> caps,
1271
+ out int
1272
+ ) {
1273
+ var int group = backref_group(pattern.at(atom_start + 1))
1274
+ if lazy > 0 {
1275
+ if min_count <= 0 {
1276
+ var int stop = match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
1277
+ if stop >= 0 {
1278
+ return stop
1279
+ }
1280
+ }
1281
+ var int next = match_backref(input, group, in_pos, in_len, ref caps)
1282
+ if next >= 0 {
1283
+ if next != in_pos {
1284
+ var new_min = 0
1285
+ if min_count > 0 {
1286
+ new_min = min_count - 1
1287
+ }
1288
+ return match_star_backref(
1289
+ pattern,
1290
+ atom_start,
1291
+ rest,
1292
+ pat_end,
1293
+ input,
1294
+ next,
1295
+ new_min,
1296
+ lazy,
1297
+ in_len,
1298
+ ref caps,
1299
+ )
1300
+ }
1301
+ }
1302
+ return -1
1303
+ }
1304
+ var int next = match_backref(input, group, in_pos, in_len, ref caps)
1305
+ if next >= 0 {
1306
+ if next != in_pos {
1307
+ var new_min = 0
1308
+ if min_count > 0 {
1309
+ new_min = min_count - 1
1310
+ }
1311
+ var int r = match_star_backref(
1312
+ pattern,
1313
+ atom_start,
1314
+ rest,
1315
+ pat_end,
1316
+ input,
1317
+ next,
1318
+ new_min,
1319
+ lazy,
1320
+ in_len,
1321
+ ref caps,
1322
+ )
1323
+ if r >= 0 {
1324
+ return r
1325
+ }
1326
+ }
1327
+ }
1328
+ if min_count <= 0 {
1329
+ return match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
1330
+ }
1331
+ return -1
1332
+ }
1333
+
1334
+ // Whether the escaped character c names an ASCII shorthand class:
1335
+ // \d \D (digits), \w \W (word characters), \s \S (whitespace).
1336
+ func is_shorthand = (char c, out bool) {
1337
+ var int code = c as int
1338
+ return code == 100 || code == 68 || code == 119 || code == 87 || code == 115 || code == 83
1339
+ }
1340
+
1341
+ // Whether byte b belongs to the shorthand class named by c. Uppercase names
1342
+ // are the complements (S/D/W match anything the lowercase set excludes).
1343
+ func matches_shorthand = (char c, char b, out bool) {
1344
+ var int kind = c as int
1345
+ var int code = b as int
1346
+ var found = false
1347
+ if kind == 100 || kind == 68 {
1348
+ // \d \D
1349
+ found = code >= 48 && code <= 57
1350
+ } else {
1351
+ if kind == 119 || kind == 87 {
1352
+ // \w \W
1353
+ found = (code >= 97 && code <= 122) || (code >= 65 && code <= 90) || (code >= 48 && code <= 57) || code == 95
1354
+ } else {
1355
+ // \s \S
1356
+ found = (code >= 9 && code <= 13) || code == 32
1357
+ }
1358
+ }
1359
+ if kind >= 65 && kind <= 90 {
1360
+ return !found
1361
+ }
1362
+ return found
1363
+ }
1364
+
1365
+ // Append the member bytes of the lowercase shorthand class named by c to the
1366
+ // first-byte charset. Complement shorthands (S/D/W) match "anything else" —
1367
+ // not expressible as an inclusion set — so they return -1, which disables
1368
+ // the caller's prefilter.
1369
+ func shorthand_members = (char c, ref StringBuilder sb, out int) {
1370
+ var int kind = c as int
1371
+ if kind == 83 || kind == 68 || kind == 87 {
1372
+ return -1
1373
+ }
1374
+ if kind == 100 {
1375
+ var int code = 48
1376
+ while code <= 57; code += 1 {
1377
+ sb.append_char(code as char)
1378
+ }
1379
+ return 0
1380
+ }
1381
+ if kind == 119 {
1382
+ var int code = 48
1383
+ while code <= 57; code += 1 {
1384
+ sb.append_char(code as char)
1385
+ }
1386
+ var int upper = 65
1387
+ while upper <= 90; upper += 1 {
1388
+ sb.append_char(upper as char)
1389
+ }
1390
+ var int lower = 97
1391
+ while lower <= 122; lower += 1 {
1392
+ sb.append_char(lower as char)
1393
+ }
1394
+ sb.append_char('_')
1395
+ return 0
1396
+ }
1397
+ // \s: tab, LF, VT, FF, CR, space
1398
+ var int w = 9
1399
+ while w <= 13; w += 1 {
1400
+ sb.append_char(w as char)
1401
+ }
1402
+ sb.append_char(32 as char)
1403
+ return 0
1404
+ }