nomen-lang 0.2.3 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/core/System/Arena.nm +166 -0
- package/core/System/Array.nm +30 -161
- package/core/System/BigInt.nm +31 -42
- package/core/System/Buffer.nm +67 -249
- package/core/System/ClassBuffer.nm +43 -105
- package/core/System/Controls/LayoutParams.nm +1 -1
- package/core/System/Map.nm +3 -0
- package/core/System/String.nm +256 -19
- package/core/System/StringBuilder.nm +53 -117
- package/core/System/Text/CharIndex.nm +144 -0
- package/core/System/Text/Chars.nm +32 -0
- package/core/System/Text/JsonTree.nm +33 -81
- package/core/System/Text/Regex.nm +726 -40
- package/core/System/Text/Utf8.nm +163 -0
- package/core/System/char.nm +25 -0
- package/dist/index.mjs +3477 -405
- package/package.json +1 -1
- package/src/index.ts +58 -7
|
@@ -22,7 +22,8 @@ pub struct Regex {
|
|
|
22
22
|
break
|
|
23
23
|
}
|
|
24
24
|
}
|
|
25
|
-
|
|
25
|
+
var caps = List<int>()
|
|
26
|
+
if match_here(pattern, 0, pat_len, input, i, len, ref caps) >= 0 {
|
|
26
27
|
return true
|
|
27
28
|
}
|
|
28
29
|
}
|
|
@@ -46,7 +47,8 @@ pub struct Regex {
|
|
|
46
47
|
break
|
|
47
48
|
}
|
|
48
49
|
}
|
|
49
|
-
var
|
|
50
|
+
var caps = List<int>()
|
|
51
|
+
var int end = match_here(pattern, 0, pat_len, input, i, len, ref caps)
|
|
50
52
|
if end >= 0 {
|
|
51
53
|
var int j = i
|
|
52
54
|
while j < end; j += 1 {
|
|
@@ -58,6 +60,80 @@ pub struct Regex {
|
|
|
58
60
|
return match_result
|
|
59
61
|
}
|
|
60
62
|
|
|
63
|
+
// Try to match pattern anywhere in input. On a match, append the whole
|
|
64
|
+
// match as group 0 followed by capture groups 1..N to `out` ("" for a
|
|
65
|
+
// group that did not participate in the match) and return true. Return
|
|
66
|
+
// false with `out` untouched when nothing matches. A group that repeats
|
|
67
|
+
// (`(a)+`) reports its LAST iteration.
|
|
68
|
+
pub func captures = (string pattern, string input, ref List<string> dst, out bool) { var int pat_len = pattern.length
|
|
69
|
+
var int len = input.length
|
|
70
|
+
var string charset = first_byte_set(pattern, pat_len)
|
|
71
|
+
var int clen = charset.length
|
|
72
|
+
var int groups = capture_count(pattern)
|
|
73
|
+
var i = 0
|
|
74
|
+
while i <= len; i += 1 {
|
|
75
|
+
if clen > 0 && i < len {
|
|
76
|
+
i = find_next_byte(input, i, len, charset)
|
|
77
|
+
if i > len {
|
|
78
|
+
break
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
var caps = List<int>()
|
|
82
|
+
var int end = match_here(pattern, 0, pat_len, input, i, len, ref caps)
|
|
83
|
+
if end >= 0 {
|
|
84
|
+
var string whole = input.substring(i, end)
|
|
85
|
+
dst.push(whole)
|
|
86
|
+
var int g = 1
|
|
87
|
+
while g <= groups; g += 1 {
|
|
88
|
+
var int at = last_capture_index(caps, g)
|
|
89
|
+
if at >= 0 {
|
|
90
|
+
var int s = caps.at(at + 1)
|
|
91
|
+
if s >= 0 {
|
|
92
|
+
var string cap = input.substring(s, caps.at(at + 2))
|
|
93
|
+
dst.push(cap)
|
|
94
|
+
} else {
|
|
95
|
+
dst.push("")
|
|
96
|
+
}
|
|
97
|
+
} else {
|
|
98
|
+
dst.push("")
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
return true
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
return false
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
// Find the first match of pattern in input and fill `dst` with where it
|
|
108
|
+
// sits: .found with byte offsets .start/.end, the matched .text, and
|
|
109
|
+
// .length (all zero/empty when .found is false — mvzr-style positions
|
|
110
|
+
// for callers doing their own index bookkeeping). `dst` is fully reset
|
|
111
|
+
// first, so a reused RegexMatch carries nothing over. Scans every
|
|
112
|
+
// position: the first-byte prefilter would skip nullable patterns'
|
|
113
|
+
// empty matches at skipped positions, reporting a wrong .start.
|
|
114
|
+
pub func find = (string pattern, string input, ref RegexMatch dst) {
|
|
115
|
+
dst.found = false
|
|
116
|
+
dst.start = 0
|
|
117
|
+
dst.end = 0
|
|
118
|
+
dst.length = 0
|
|
119
|
+
dst.text = ""
|
|
120
|
+
var int pat_len = pattern.length
|
|
121
|
+
var int len = input.length
|
|
122
|
+
var i = 0
|
|
123
|
+
while i <= len; i += 1 {
|
|
124
|
+
var caps = List<int>()
|
|
125
|
+
var int end = match_here(pattern, 0, pat_len, input, i, len, ref caps)
|
|
126
|
+
if end >= 0 {
|
|
127
|
+
dst.found = true
|
|
128
|
+
dst.start = i
|
|
129
|
+
dst.end = end
|
|
130
|
+
dst.length = end - i
|
|
131
|
+
dst.text = input.substring(i, end)
|
|
132
|
+
return
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
61
137
|
// Count the number of non-overlapping occurrences of pattern in input.
|
|
62
138
|
pub func count = (string pattern, string input, out int) {
|
|
63
139
|
var int pat_len = pattern.length
|
|
@@ -73,7 +149,8 @@ pub struct Regex {
|
|
|
73
149
|
break
|
|
74
150
|
}
|
|
75
151
|
}
|
|
76
|
-
var
|
|
152
|
+
var caps = List<int>()
|
|
153
|
+
var int end = match_here(pattern, 0, pat_len, input, pos, len, ref caps)
|
|
77
154
|
if end >= 0 {
|
|
78
155
|
total += 1
|
|
79
156
|
if end == pos {
|
|
@@ -89,8 +166,7 @@ pub struct Regex {
|
|
|
89
166
|
}
|
|
90
167
|
|
|
91
168
|
// Replace all non-overlapping occurrences of pattern in input with replacement.
|
|
92
|
-
pub func replace_all = (string pattern, string input, string replacement, out string) {
|
|
93
|
-
var sb = StringBuilder()
|
|
169
|
+
pub func replace_all = (string pattern, string input, string replacement, out string) { var sb = StringBuilder()
|
|
94
170
|
var int pat_len = pattern.length
|
|
95
171
|
var int len = input.length
|
|
96
172
|
var string charset = first_byte_set(pattern, pat_len)
|
|
@@ -108,7 +184,8 @@ pub struct Regex {
|
|
|
108
184
|
break
|
|
109
185
|
}
|
|
110
186
|
}
|
|
111
|
-
var
|
|
187
|
+
var caps = List<int>()
|
|
188
|
+
var int end = match_here(pattern, 0, pat_len, input, pos, len, ref caps)
|
|
112
189
|
if end >= 0 {
|
|
113
190
|
sb.append_string(replacement)
|
|
114
191
|
if end == pos {
|
|
@@ -124,6 +201,172 @@ pub struct Regex {
|
|
|
124
201
|
}
|
|
125
202
|
return sb.to_string()
|
|
126
203
|
}
|
|
204
|
+
|
|
205
|
+
// Whether pattern matches anywhere in input, ignoring ASCII case.
|
|
206
|
+
pub func test_ci = (string pattern, string input, out bool) {
|
|
207
|
+
var string folded = fold_pattern(pattern)
|
|
208
|
+
return Regex.test(folded, input)
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
// First match of pattern in input, ignoring ASCII case ("" if none).
|
|
212
|
+
pub func match_ci = (string pattern, string input, move out string) {
|
|
213
|
+
var string folded = fold_pattern(pattern)
|
|
214
|
+
return Regex.match(folded, input)
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
// find, ignoring ASCII case.
|
|
218
|
+
pub func find_ci = (string pattern, string input, ref RegexMatch dst) {
|
|
219
|
+
var string folded = fold_pattern(pattern)
|
|
220
|
+
Regex.find(folded, input, ref dst)
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
// captures, ignoring ASCII case.
|
|
224
|
+
pub func captures_ci = (string pattern, string input, ref List<string> dst, out bool) {
|
|
225
|
+
var string folded = fold_pattern(pattern)
|
|
226
|
+
return Regex.captures(folded, input, ref dst)
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
// replace_all, ignoring ASCII case.
|
|
230
|
+
pub func replace_all_ci = (string pattern, string input, string replacement, out string) {
|
|
231
|
+
var string folded = fold_pattern(pattern)
|
|
232
|
+
return Regex.replace_all(folded, input, replacement)
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
// Rewrite pattern so the case-sensitive engine matches case-insensitively:
|
|
237
|
+
// literals become both-case classes ([kK]), class members and ranges gain
|
|
238
|
+
// their case counterparts ([a-z] -> [a-zA-Z]). Structural syntax — groups,
|
|
239
|
+
// alternation, quantifiers, anchors, escaped punctuation — is copied
|
|
240
|
+
// through unchanged.
|
|
241
|
+
func fold_pattern = (string pattern, move out string) {
|
|
242
|
+
var int n = pattern.length
|
|
243
|
+
var sb = StringBuilder()
|
|
244
|
+
var int i = 0
|
|
245
|
+
while i < n; i += 1 {
|
|
246
|
+
var char c = pattern.at(i)
|
|
247
|
+
if (c as int) == 92 {
|
|
248
|
+
// Escaped char: shorthands and backrefs keep their backslash;
|
|
249
|
+
// other escaped letters become both-case classes; escaped
|
|
250
|
+
// punctuation is copied literally.
|
|
251
|
+
if i + 1 < n {
|
|
252
|
+
var char e = pattern.at(i + 1)
|
|
253
|
+
if is_shorthand(e) || is_backref_digit(e) {
|
|
254
|
+
sb.append_char(c)
|
|
255
|
+
sb.append_char(e)
|
|
256
|
+
} else {
|
|
257
|
+
if is_alpha_byte(e) {
|
|
258
|
+
sb.append_char('[')
|
|
259
|
+
sb.append_char(e)
|
|
260
|
+
sb.append_char(swap_case_byte(e))
|
|
261
|
+
sb.append_char(']')
|
|
262
|
+
} else {
|
|
263
|
+
sb.append_char(c)
|
|
264
|
+
sb.append_char(e)
|
|
265
|
+
}
|
|
266
|
+
}
|
|
267
|
+
i += 1
|
|
268
|
+
} else {
|
|
269
|
+
sb.append_char(c)
|
|
270
|
+
}
|
|
271
|
+
} else {
|
|
272
|
+
if c == '[' {
|
|
273
|
+
var int ce = atom_end_pos(pattern, i, n)
|
|
274
|
+
fold_class(pattern, i, ce, ref sb)
|
|
275
|
+
i = ce - 1
|
|
276
|
+
} else {
|
|
277
|
+
if is_alpha_byte(c) {
|
|
278
|
+
sb.append_char('[')
|
|
279
|
+
sb.append_char(c)
|
|
280
|
+
sb.append_char(swap_case_byte(c))
|
|
281
|
+
sb.append_char(']')
|
|
282
|
+
} else {
|
|
283
|
+
sb.append_char(c)
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
return sb.to_string()
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
// fold a character class [cs..ce): copy its contents, adding the case
|
|
292
|
+
// counterpart of every letter member and mirroring letter ranges.
|
|
293
|
+
func fold_class = (string pattern, int cs, int ce, ref StringBuilder dst) {
|
|
294
|
+
dst.append_char('[')
|
|
295
|
+
var int i = cs + 1
|
|
296
|
+
if i < ce - 1 {
|
|
297
|
+
if pattern.at(i) == '^' {
|
|
298
|
+
dst.append_char('^')
|
|
299
|
+
i += 1
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
var int content_end = ce - 1
|
|
303
|
+
while i < content_end; i += 1 {
|
|
304
|
+
var char c = pattern.at(i)
|
|
305
|
+
if (c as int) == 92 {
|
|
306
|
+
dst.append_char(c)
|
|
307
|
+
if i + 1 < content_end {
|
|
308
|
+
var char e = pattern.at(i + 1)
|
|
309
|
+
dst.append_char(e)
|
|
310
|
+
// A shorthand keeps its backslash; other escaped letters
|
|
311
|
+
// gain their case counterpart.
|
|
312
|
+
if is_alpha_byte(e) && !is_shorthand(e) {
|
|
313
|
+
dst.append_char(swap_case_byte(e))
|
|
314
|
+
}
|
|
315
|
+
i += 1
|
|
316
|
+
}
|
|
317
|
+
} else {
|
|
318
|
+
if i + 2 < content_end {
|
|
319
|
+
if pattern.at(i + 1) == '-' {
|
|
320
|
+
var char lo = c
|
|
321
|
+
var char hi = pattern.at(i + 2)
|
|
322
|
+
dst.append_char(lo)
|
|
323
|
+
dst.append_char('-')
|
|
324
|
+
dst.append_char(hi)
|
|
325
|
+
if is_alpha_byte(lo) {
|
|
326
|
+
if is_alpha_byte(hi) {
|
|
327
|
+
dst.append_char(swap_case_byte(lo))
|
|
328
|
+
dst.append_char('-')
|
|
329
|
+
dst.append_char(swap_case_byte(hi))
|
|
330
|
+
}
|
|
331
|
+
}
|
|
332
|
+
i += 2
|
|
333
|
+
} else {
|
|
334
|
+
if is_alpha_byte(c) {
|
|
335
|
+
dst.append_char(c)
|
|
336
|
+
dst.append_char(swap_case_byte(c))
|
|
337
|
+
} else {
|
|
338
|
+
dst.append_char(c)
|
|
339
|
+
}
|
|
340
|
+
}
|
|
341
|
+
} else {
|
|
342
|
+
if is_alpha_byte(c) {
|
|
343
|
+
dst.append_char(c)
|
|
344
|
+
dst.append_char(swap_case_byte(c))
|
|
345
|
+
} else {
|
|
346
|
+
dst.append_char(c)
|
|
347
|
+
}
|
|
348
|
+
}
|
|
349
|
+
}
|
|
350
|
+
}
|
|
351
|
+
dst.append_char(']')
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
// ASCII letter test for pattern characters.
|
|
355
|
+
func is_alpha_byte = (char c, out bool) {
|
|
356
|
+
var int code = c as int
|
|
357
|
+
return (code > 64 && code < 91) || (code > 96 && code < 123)
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
// The ASCII letter with the opposite case (unchanged for non-letters).
|
|
361
|
+
func swap_case_byte = (char c, out char) {
|
|
362
|
+
var int code = c as int
|
|
363
|
+
if code > 64 && code < 91 {
|
|
364
|
+
return (code + 32) as char
|
|
365
|
+
}
|
|
366
|
+
if code > 96 && code < 123 {
|
|
367
|
+
return (code - 32) as char
|
|
368
|
+
}
|
|
369
|
+
return c
|
|
127
370
|
}
|
|
128
371
|
|
|
129
372
|
// Scan input[pos..len) for the next byte that appears in charset.
|
|
@@ -219,9 +462,21 @@ func alt_first_bytes = (string pattern, int start, int end, ref StringBuilder sb
|
|
|
219
462
|
while i < end {
|
|
220
463
|
var char c = pattern.at(i)
|
|
221
464
|
if (c as int) == 92 {
|
|
222
|
-
// escaped literal: first byte is the escaped char
|
|
465
|
+
// escaped literal: first byte is the escaped char — unless it's
|
|
466
|
+
// a backreference (depends on a capture) or a complement
|
|
467
|
+
// shorthand (not an inclusion set), which disable the prefilter.
|
|
223
468
|
if i + 1 < end {
|
|
224
|
-
|
|
469
|
+
var char e = pattern.at(i + 1)
|
|
470
|
+
if is_backref_digit(e) {
|
|
471
|
+
return -1
|
|
472
|
+
}
|
|
473
|
+
if is_shorthand(e) {
|
|
474
|
+
if shorthand_members(e, ref sb) < 0 {
|
|
475
|
+
return -1
|
|
476
|
+
}
|
|
477
|
+
return 0
|
|
478
|
+
}
|
|
479
|
+
sb.append_char(e)
|
|
225
480
|
}
|
|
226
481
|
return 0
|
|
227
482
|
}
|
|
@@ -248,7 +503,8 @@ func alt_first_bytes = (string pattern, int start, int end, ref StringBuilder sb
|
|
|
248
503
|
return -1
|
|
249
504
|
}
|
|
250
505
|
// A '?'/'*' quantifier makes the group nullable — the first byte
|
|
251
|
-
// can also come from whatever follows it.
|
|
506
|
+
// can also come from whatever follows it. A second '?' marks the
|
|
507
|
+
// quantifier lazy and is skipped either way.
|
|
252
508
|
var int after = a_end
|
|
253
509
|
var int nullable = 0
|
|
254
510
|
if after < end {
|
|
@@ -256,6 +512,11 @@ func alt_first_bytes = (string pattern, int start, int end, ref StringBuilder sb
|
|
|
256
512
|
if q == '?' || q == '*' {
|
|
257
513
|
nullable = 1
|
|
258
514
|
after += 1
|
|
515
|
+
if after < end {
|
|
516
|
+
if pattern.at(after) == '?' {
|
|
517
|
+
after += 1
|
|
518
|
+
}
|
|
519
|
+
}
|
|
259
520
|
}
|
|
260
521
|
}
|
|
261
522
|
if nullable == 0 {
|
|
@@ -317,6 +578,8 @@ func add_class_chars = (string pattern, int start, int end, ref StringBuilder sb
|
|
|
317
578
|
// Return contract: a successful match never reads past in_len, so the result
|
|
318
579
|
// is always <= in_len (and -1 on failure). This lets callers prove an index
|
|
319
580
|
// bounded by the result is also bounded by the input length.
|
|
581
|
+
// `caps` collects capture spans as (group_index, start, end) triples in
|
|
582
|
+
// match order; a later entry for the same group supersedes an earlier one.
|
|
320
583
|
func match_here = (
|
|
321
584
|
string pattern,
|
|
322
585
|
int pat_pos,
|
|
@@ -324,6 +587,7 @@ func match_here = (
|
|
|
324
587
|
string input,
|
|
325
588
|
int in_pos,
|
|
326
589
|
int in_len,
|
|
590
|
+
ref List<int> caps,
|
|
327
591
|
out int: out <= in_len
|
|
328
592
|
) {
|
|
329
593
|
// Handle top-level alternation: a '|' at bracket/paren depth 0 splits the
|
|
@@ -349,11 +613,11 @@ func match_here = (
|
|
|
349
613
|
}
|
|
350
614
|
if (c as int) == 124 {
|
|
351
615
|
if depth == 0 {
|
|
352
|
-
var int left = match_here(pattern, pat_pos, i, input, in_pos, in_len)
|
|
616
|
+
var int left = match_here(pattern, pat_pos, i, input, in_pos, in_len, ref caps)
|
|
353
617
|
if left >= 0 {
|
|
354
618
|
return left
|
|
355
619
|
}
|
|
356
|
-
return match_here(pattern, i + 1, pat_end, input, in_pos, in_len)
|
|
620
|
+
return match_here(pattern, i + 1, pat_end, input, in_pos, in_len, ref caps)
|
|
357
621
|
}
|
|
358
622
|
}
|
|
359
623
|
i += 1
|
|
@@ -373,38 +637,48 @@ func match_here = (
|
|
|
373
637
|
if in_pos != 0 {
|
|
374
638
|
return -1
|
|
375
639
|
}
|
|
376
|
-
return match_here(pattern, pat_pos + 1, pat_end, input, in_pos, in_len)
|
|
640
|
+
return match_here(pattern, pat_pos + 1, pat_end, input, in_pos, in_len, ref caps)
|
|
377
641
|
}
|
|
378
642
|
if pc == '$' {
|
|
379
643
|
if in_pos != in_len {
|
|
380
644
|
return -1
|
|
381
645
|
}
|
|
382
|
-
return match_here(pattern, pat_pos + 1, pat_end, input, in_pos, in_len)
|
|
646
|
+
return match_here(pattern, pat_pos + 1, pat_end, input, in_pos, in_len, ref caps)
|
|
383
647
|
}
|
|
384
648
|
|
|
385
649
|
// Determine the bounds of the current atom.
|
|
386
650
|
var int a_end = atom_end_pos(pattern, pat_pos, pat_end)
|
|
387
651
|
|
|
388
|
-
// Check for a quantifier following the atom.
|
|
652
|
+
// Check for a quantifier following the atom. A trailing '?' marks the
|
|
653
|
+
// quantifier LAZY (shortest match first) — the rest of the pattern then
|
|
654
|
+
// starts after both symbols.
|
|
389
655
|
if a_end < pat_end {
|
|
390
656
|
var char q = pattern.at(a_end)
|
|
391
|
-
if q == '*' {
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
657
|
+
if q == '*' || q == '+' || q == '?' {
|
|
658
|
+
var int lazy = 0
|
|
659
|
+
var int rest = a_end + 1
|
|
660
|
+
if rest < pat_end {
|
|
661
|
+
if pattern.at(rest) == '?' {
|
|
662
|
+
lazy = 1
|
|
663
|
+
rest += 1
|
|
664
|
+
}
|
|
665
|
+
}
|
|
666
|
+
if q == '*' {
|
|
667
|
+
return match_star(pattern, pat_pos, a_end, rest, pat_end, input, in_pos, 0, lazy, in_len, ref caps)
|
|
668
|
+
}
|
|
669
|
+
if q == '+' {
|
|
670
|
+
return match_star(pattern, pat_pos, a_end, rest, pat_end, input, in_pos, 1, lazy, in_len, ref caps)
|
|
671
|
+
}
|
|
672
|
+
return match_optional(pattern, pat_pos, a_end, rest, pat_end, input, in_pos, lazy, in_len, ref caps)
|
|
399
673
|
}
|
|
400
674
|
}
|
|
401
675
|
|
|
402
676
|
// No quantifier: match the atom once, then the rest of the pattern.
|
|
403
|
-
var int next = match_single_atom(pattern, pat_pos, a_end, input, in_pos, in_len)
|
|
677
|
+
var int next = match_single_atom(pattern, pat_pos, a_end, input, in_pos, in_len, ref caps)
|
|
404
678
|
if next < 0 {
|
|
405
679
|
return -1
|
|
406
680
|
}
|
|
407
|
-
return match_here(pattern, a_end, pat_end, input, next, in_len)
|
|
681
|
+
return match_here(pattern, a_end, pat_end, input, next, in_len, ref caps)
|
|
408
682
|
}
|
|
409
683
|
|
|
410
684
|
// Return the position just past the atom that starts at pat_pos. Handles
|
|
@@ -460,6 +734,7 @@ func match_single_atom = (
|
|
|
460
734
|
string input,
|
|
461
735
|
int in_pos,
|
|
462
736
|
int in_len,
|
|
737
|
+
ref List<int> caps,
|
|
463
738
|
out int
|
|
464
739
|
) {
|
|
465
740
|
var char pc = pattern.at(atom_start)
|
|
@@ -481,12 +756,42 @@ func match_single_atom = (
|
|
|
481
756
|
return -1
|
|
482
757
|
}
|
|
483
758
|
if pc == '(' {
|
|
484
|
-
// Match the group's inner content
|
|
485
|
-
|
|
759
|
+
// Match the group's inner content, recording the group's span on
|
|
760
|
+
// success. On failure, restore the span this path started with (a
|
|
761
|
+
// repeated group's earlier iterations survive a failed extra trial)
|
|
762
|
+
// — or mark the group unset when it had no earlier span here.
|
|
763
|
+
var int g = group_index_at(pattern, atom_start)
|
|
764
|
+
var int at = last_capture_index(caps, g)
|
|
765
|
+
var int old_start = -1
|
|
766
|
+
var int old_end = -1
|
|
767
|
+
if at >= 0 {
|
|
768
|
+
old_start = caps.at(at + 1)
|
|
769
|
+
old_end = caps.at(at + 2)
|
|
770
|
+
}
|
|
771
|
+
var int next = match_here(pattern, atom_start + 1, atom_end - 1, input, in_pos, in_len, ref caps)
|
|
772
|
+
if next >= 0 {
|
|
773
|
+
record_capture(ref caps, g, in_pos, next)
|
|
774
|
+
return next
|
|
775
|
+
}
|
|
776
|
+
record_capture(ref caps, g, old_start, old_end)
|
|
777
|
+
return -1
|
|
486
778
|
}
|
|
487
779
|
if (pc as int) == 92 {
|
|
488
780
|
if atom_start + 1 < atom_end {
|
|
489
781
|
var char ec = pattern.at(atom_start + 1)
|
|
782
|
+
if is_backref_digit(ec) {
|
|
783
|
+
// \1..\9: match the same text capture group N matched.
|
|
784
|
+
return match_backref(input, backref_group(ec), in_pos, in_len, ref caps)
|
|
785
|
+
}
|
|
786
|
+
if is_shorthand(ec) {
|
|
787
|
+
// \d \s \w and complements: class membership by name.
|
|
788
|
+
if in_pos < in_len {
|
|
789
|
+
if matches_shorthand(ec, input.at(in_pos)) {
|
|
790
|
+
return in_pos + 1
|
|
791
|
+
}
|
|
792
|
+
}
|
|
793
|
+
return -1
|
|
794
|
+
}
|
|
490
795
|
if in_pos < in_len {
|
|
491
796
|
if input.at(in_pos) == ec {
|
|
492
797
|
return in_pos + 1
|
|
@@ -527,25 +832,38 @@ func char_matches_atom = (
|
|
|
527
832
|
}
|
|
528
833
|
if (pc as int) == 92 {
|
|
529
834
|
if atom_start + 1 < atom_end {
|
|
530
|
-
|
|
835
|
+
var char ec = pattern.at(atom_start + 1)
|
|
836
|
+
if is_backref_digit(ec) {
|
|
837
|
+
// A backreference is a multi-byte atom; the greedy
|
|
838
|
+
// single-character loop never consults this for it.
|
|
839
|
+
return false
|
|
840
|
+
}
|
|
841
|
+
if is_shorthand(ec) {
|
|
842
|
+
return matches_shorthand(ec, ic)
|
|
843
|
+
}
|
|
844
|
+
return ic == ec
|
|
531
845
|
}
|
|
532
846
|
return false
|
|
533
847
|
}
|
|
534
848
|
return pc == ic
|
|
535
849
|
}
|
|
536
850
|
|
|
537
|
-
//
|
|
538
|
-
// pattern
|
|
539
|
-
//
|
|
851
|
+
// Match the atom [atom_start..atom_end) followed by the rest of the pattern
|
|
852
|
+
// pattern[rest..pat_end). min_count is 0 for '*' and 1 for '+'; lazy picks
|
|
853
|
+
// the shortest repetition first instead of the longest. Single-character
|
|
854
|
+
// atoms are handled iteratively; groups recurse.
|
|
540
855
|
func match_star = (
|
|
541
856
|
string pattern,
|
|
542
857
|
int atom_start,
|
|
543
858
|
int atom_end,
|
|
859
|
+
int rest,
|
|
544
860
|
int pat_end,
|
|
545
861
|
string input,
|
|
546
862
|
int in_pos,
|
|
547
863
|
int min_count,
|
|
864
|
+
int lazy,
|
|
548
865
|
int in_len,
|
|
866
|
+
ref List<int> caps,
|
|
549
867
|
out int
|
|
550
868
|
) {
|
|
551
869
|
var char pc = pattern.at(atom_start)
|
|
@@ -554,13 +872,36 @@ func match_star = (
|
|
|
554
872
|
pattern,
|
|
555
873
|
atom_start,
|
|
556
874
|
atom_end,
|
|
875
|
+
rest,
|
|
557
876
|
pat_end,
|
|
558
877
|
input,
|
|
559
878
|
in_pos,
|
|
560
879
|
min_count,
|
|
880
|
+
lazy,
|
|
561
881
|
in_len,
|
|
882
|
+
ref caps,
|
|
562
883
|
)
|
|
563
884
|
}
|
|
885
|
+
if (pc as int) == 92 {
|
|
886
|
+
if atom_start + 1 < atom_end {
|
|
887
|
+
if is_backref_digit(pattern.at(atom_start + 1)) {
|
|
888
|
+
// A backreference matches a whole span per iteration — the
|
|
889
|
+
// single-character greedy loop below can't apply.
|
|
890
|
+
return match_star_backref(
|
|
891
|
+
pattern,
|
|
892
|
+
atom_start,
|
|
893
|
+
rest,
|
|
894
|
+
pat_end,
|
|
895
|
+
input,
|
|
896
|
+
in_pos,
|
|
897
|
+
min_count,
|
|
898
|
+
lazy,
|
|
899
|
+
in_len,
|
|
900
|
+
ref caps,
|
|
901
|
+
)
|
|
902
|
+
}
|
|
903
|
+
}
|
|
904
|
+
}
|
|
564
905
|
|
|
565
906
|
// Greedily consume the longest run of matching characters.
|
|
566
907
|
var int max_pos = in_pos
|
|
@@ -571,10 +912,21 @@ func match_star = (
|
|
|
571
912
|
break
|
|
572
913
|
}
|
|
573
914
|
}
|
|
915
|
+
if lazy > 0 {
|
|
916
|
+
// Lazy: extend from the minimum until the rest matches.
|
|
917
|
+
var int try_pos = in_pos + min_count
|
|
918
|
+
while try_pos <= max_pos; try_pos += 1 {
|
|
919
|
+
var int r = match_here(pattern, rest, pat_end, input, try_pos, in_len, ref caps)
|
|
920
|
+
if r >= 0 {
|
|
921
|
+
return r
|
|
922
|
+
}
|
|
923
|
+
}
|
|
924
|
+
return -1
|
|
925
|
+
}
|
|
574
926
|
// Backtrack from the longest match down to the minimum, trying the rest.
|
|
575
927
|
var int try_pos = max_pos
|
|
576
928
|
while try_pos >= in_pos + min_count; try_pos -= 1 {
|
|
577
|
-
var int r = match_here(pattern,
|
|
929
|
+
var int r = match_here(pattern, rest, pat_end, input, try_pos, in_len, ref caps)
|
|
578
930
|
if r >= 0 {
|
|
579
931
|
return r
|
|
580
932
|
}
|
|
@@ -582,20 +934,56 @@ func match_star = (
|
|
|
582
934
|
return -1
|
|
583
935
|
}
|
|
584
936
|
|
|
585
|
-
// Recursive
|
|
937
|
+
// Recursive backtracking for group atoms quantified with '*' or '+'.
|
|
586
938
|
func match_star_group = (
|
|
587
939
|
string pattern,
|
|
588
940
|
int atom_start,
|
|
589
941
|
int atom_end,
|
|
942
|
+
int rest,
|
|
590
943
|
int pat_end,
|
|
591
944
|
string input,
|
|
592
945
|
int in_pos,
|
|
593
946
|
int min_count,
|
|
947
|
+
int lazy,
|
|
594
948
|
int in_len,
|
|
949
|
+
ref List<int> caps,
|
|
595
950
|
out int
|
|
596
951
|
) {
|
|
952
|
+
if lazy > 0 {
|
|
953
|
+
// Lazy: stop repeating first (minimum met), then try one more
|
|
954
|
+
// iteration before giving up.
|
|
955
|
+
if min_count <= 0 {
|
|
956
|
+
var int stop = match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
|
|
957
|
+
if stop >= 0 {
|
|
958
|
+
return stop
|
|
959
|
+
}
|
|
960
|
+
}
|
|
961
|
+
var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len, ref caps)
|
|
962
|
+
if next >= 0 {
|
|
963
|
+
if next != in_pos {
|
|
964
|
+
var new_min = 0
|
|
965
|
+
if min_count > 0 {
|
|
966
|
+
new_min = min_count - 1
|
|
967
|
+
}
|
|
968
|
+
return match_star_group(
|
|
969
|
+
pattern,
|
|
970
|
+
atom_start,
|
|
971
|
+
atom_end,
|
|
972
|
+
rest,
|
|
973
|
+
pat_end,
|
|
974
|
+
input,
|
|
975
|
+
next,
|
|
976
|
+
new_min,
|
|
977
|
+
lazy,
|
|
978
|
+
in_len,
|
|
979
|
+
ref caps,
|
|
980
|
+
)
|
|
981
|
+
}
|
|
982
|
+
}
|
|
983
|
+
return -1
|
|
984
|
+
}
|
|
597
985
|
// Try matching the group one more time (greedy).
|
|
598
|
-
var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len)
|
|
986
|
+
var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len, ref caps)
|
|
599
987
|
if next >= 0 {
|
|
600
988
|
if next != in_pos {
|
|
601
989
|
var new_min = 0
|
|
@@ -606,11 +994,14 @@ func match_star_group = (
|
|
|
606
994
|
pattern,
|
|
607
995
|
atom_start,
|
|
608
996
|
atom_end,
|
|
997
|
+
rest,
|
|
609
998
|
pat_end,
|
|
610
999
|
input,
|
|
611
1000
|
next,
|
|
612
1001
|
new_min,
|
|
1002
|
+
lazy,
|
|
613
1003
|
in_len,
|
|
1004
|
+
ref caps,
|
|
614
1005
|
)
|
|
615
1006
|
if r >= 0 {
|
|
616
1007
|
return r
|
|
@@ -619,30 +1010,45 @@ func match_star_group = (
|
|
|
619
1010
|
}
|
|
620
1011
|
// Stop repeating and try the rest of the pattern, if the minimum is met.
|
|
621
1012
|
if min_count <= 0 {
|
|
622
|
-
return match_here(pattern,
|
|
1013
|
+
return match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
|
|
623
1014
|
}
|
|
624
1015
|
return -1
|
|
625
1016
|
}
|
|
626
1017
|
|
|
627
|
-
// Match the atom optionally ('?'), preferring to include it.
|
|
1018
|
+
// Match the atom optionally ('?'), preferring to include it unless lazy.
|
|
628
1019
|
func match_optional = (
|
|
629
1020
|
string pattern,
|
|
630
1021
|
int atom_start,
|
|
631
1022
|
int atom_end,
|
|
1023
|
+
int rest,
|
|
632
1024
|
int pat_end,
|
|
633
1025
|
string input,
|
|
634
1026
|
int in_pos,
|
|
1027
|
+
int lazy,
|
|
635
1028
|
int in_len,
|
|
1029
|
+
ref List<int> caps,
|
|
636
1030
|
out int
|
|
637
1031
|
) {
|
|
638
|
-
|
|
1032
|
+
if lazy > 0 {
|
|
1033
|
+
// Lazy: try without the atom first.
|
|
1034
|
+
var int skip = match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
|
|
1035
|
+
if skip >= 0 {
|
|
1036
|
+
return skip
|
|
1037
|
+
}
|
|
1038
|
+
var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len, ref caps)
|
|
1039
|
+
if next >= 0 {
|
|
1040
|
+
return match_here(pattern, rest, pat_end, input, next, in_len, ref caps)
|
|
1041
|
+
}
|
|
1042
|
+
return -1
|
|
1043
|
+
}
|
|
1044
|
+
var int next = match_single_atom(pattern, atom_start, atom_end, input, in_pos, in_len, ref caps)
|
|
639
1045
|
if next >= 0 {
|
|
640
|
-
var int r = match_here(pattern,
|
|
1046
|
+
var int r = match_here(pattern, rest, pat_end, input, next, in_len, ref caps)
|
|
641
1047
|
if r >= 0 {
|
|
642
1048
|
return r
|
|
643
1049
|
}
|
|
644
1050
|
}
|
|
645
|
-
return match_here(pattern,
|
|
1051
|
+
return match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
|
|
646
1052
|
}
|
|
647
1053
|
|
|
648
1054
|
// Report whether input.at(in_pos) belongs to the character class whose '[' is
|
|
@@ -677,8 +1083,16 @@ func char_in_class = (
|
|
|
677
1083
|
var char c = pattern.at(i)
|
|
678
1084
|
if (c as int) == 92 {
|
|
679
1085
|
if i + 1 < content_end {
|
|
680
|
-
|
|
681
|
-
|
|
1086
|
+
var char e = pattern.at(i + 1)
|
|
1087
|
+
if is_shorthand(e) {
|
|
1088
|
+
// \d \s \w (and complements) inside a class.
|
|
1089
|
+
if matches_shorthand(e, ic) {
|
|
1090
|
+
found = true
|
|
1091
|
+
}
|
|
1092
|
+
} else {
|
|
1093
|
+
if ic == e {
|
|
1094
|
+
found = true
|
|
1095
|
+
}
|
|
682
1096
|
}
|
|
683
1097
|
}
|
|
684
1098
|
i = i + 2
|
|
@@ -716,3 +1130,275 @@ func char_in_class = (
|
|
|
716
1130
|
}
|
|
717
1131
|
return found
|
|
718
1132
|
}
|
|
1133
|
+
|
|
1134
|
+
// The number of capture groups in the whole pattern: every '(' that is not
|
|
1135
|
+
// escaped and not inside a character class (POSIX ERE has no non-capturing
|
|
1136
|
+
// groups).
|
|
1137
|
+
func capture_count = (string pattern, out int) {
|
|
1138
|
+
var int n = pattern.length
|
|
1139
|
+
var int count = 0
|
|
1140
|
+
var int i = 0
|
|
1141
|
+
while i < n; i += 1 {
|
|
1142
|
+
var char c = pattern.at(i)
|
|
1143
|
+
if (c as int) == 92 {
|
|
1144
|
+
i += 1
|
|
1145
|
+
} else {
|
|
1146
|
+
if c == '[' {
|
|
1147
|
+
i = atom_end_pos(pattern, i, n) - 1
|
|
1148
|
+
} else {
|
|
1149
|
+
if c == '(' {
|
|
1150
|
+
count += 1
|
|
1151
|
+
}
|
|
1152
|
+
}
|
|
1153
|
+
}
|
|
1154
|
+
}
|
|
1155
|
+
return count
|
|
1156
|
+
}
|
|
1157
|
+
|
|
1158
|
+
// The capture-group index (1-based) of the group whose '(' opens at
|
|
1159
|
+
// open_pos: one more than the number of capture opens before it in the
|
|
1160
|
+
// pattern (the whole match is group 0).
|
|
1161
|
+
func group_index_at = (string pattern, int open_pos, out int) {
|
|
1162
|
+
var int count = 0
|
|
1163
|
+
var int i = 0
|
|
1164
|
+
while i < open_pos; i += 1 {
|
|
1165
|
+
var char c = pattern.at(i)
|
|
1166
|
+
if (c as int) == 92 {
|
|
1167
|
+
i += 1
|
|
1168
|
+
} else {
|
|
1169
|
+
if c == '[' {
|
|
1170
|
+
i = atom_end_pos(pattern, i, open_pos) - 1
|
|
1171
|
+
} else {
|
|
1172
|
+
if c == '(' {
|
|
1173
|
+
count += 1
|
|
1174
|
+
}
|
|
1175
|
+
}
|
|
1176
|
+
}
|
|
1177
|
+
}
|
|
1178
|
+
return count + 1
|
|
1179
|
+
}
|
|
1180
|
+
|
|
1181
|
+
// Record a capture span (start, end) for the group whose '(' opens at
|
|
1182
|
+
// group_pos. A later span for the same group supersedes the earlier one, so
|
|
1183
|
+
// a repeated group reports its last iteration; a failed group records
|
|
1184
|
+
// (-1, -1) so stale spans from an abandoned attempt never surface.
|
|
1185
|
+
func record_capture = (ref List<int> caps, int group, int start, int end) {
|
|
1186
|
+
var int at = last_capture_index(caps, group)
|
|
1187
|
+
if at >= 0 {
|
|
1188
|
+
caps.set(at + 1, start)
|
|
1189
|
+
caps.set(at + 2, end)
|
|
1190
|
+
} else {
|
|
1191
|
+
caps.push(group)
|
|
1192
|
+
caps.push(start)
|
|
1193
|
+
caps.push(end)
|
|
1194
|
+
}
|
|
1195
|
+
}
|
|
1196
|
+
|
|
1197
|
+
// The log index of the LAST entry for `group` in caps, or -1 when the group
|
|
1198
|
+
// has no entry. Entries are (group, start, end) triples.
|
|
1199
|
+
func last_capture_index = (List<int> caps, int group, out int) {
|
|
1200
|
+
var int best = -1
|
|
1201
|
+
var int i = 0
|
|
1202
|
+
while i < caps.length; i += 3 {
|
|
1203
|
+
if caps.at(i) == group {
|
|
1204
|
+
best = i
|
|
1205
|
+
}
|
|
1206
|
+
}
|
|
1207
|
+
return best
|
|
1208
|
+
}
|
|
1209
|
+
|
|
1210
|
+
/**
|
|
1211
|
+
* A Regex.find result: whether the pattern matched, where, and what
|
|
1212
|
+
**/
|
|
1213
|
+
pub struct RegexMatch {
|
|
1214
|
+
var bool found = false
|
|
1215
|
+
var int start = 0
|
|
1216
|
+
var int end = 0
|
|
1217
|
+
var int length = 0
|
|
1218
|
+
var string text = ""
|
|
1219
|
+
}
|
|
1220
|
+
|
|
1221
|
+
// Whether the escaped character c (the byte after a backslash) is a
|
|
1222
|
+
// backreference (\1..\9; \0 is the whole-match escape, not supported here).
|
|
1223
|
+
func is_backref_digit = (char c, out bool) {
|
|
1224
|
+
var int code = c as int
|
|
1225
|
+
return code >= 49 && code <= 57
|
|
1226
|
+
}
|
|
1227
|
+
|
|
1228
|
+
// The capture group a backreference refers to: '1' -> 1 .. '9' -> 9.
|
|
1229
|
+
func backref_group = (char c, out int) {
|
|
1230
|
+
return (c as int) - 48
|
|
1231
|
+
}
|
|
1232
|
+
|
|
1233
|
+
// Match the text capture group `group` last matched, at in_pos. Returns the
|
|
1234
|
+
// position after the span, or -1 when the group has no span here (it did not
|
|
1235
|
+
// participate) or the text differs.
|
|
1236
|
+
func match_backref = (string input, int group, int in_pos, int in_len, ref List<int> caps, out int) {
|
|
1237
|
+
var int at = last_capture_index(caps, group)
|
|
1238
|
+
if at < 0 {
|
|
1239
|
+
return -1
|
|
1240
|
+
}
|
|
1241
|
+
var int s = caps.at(at + 1)
|
|
1242
|
+
if s < 0 {
|
|
1243
|
+
return -1
|
|
1244
|
+
}
|
|
1245
|
+
var int len = caps.at(at + 2) - s
|
|
1246
|
+
if in_pos + len > in_len {
|
|
1247
|
+
return -1
|
|
1248
|
+
}
|
|
1249
|
+
var int i = 0
|
|
1250
|
+
while i < len; i += 1 {
|
|
1251
|
+
if input.at(in_pos + i) != input.at(s + i) {
|
|
1252
|
+
return -1
|
|
1253
|
+
}
|
|
1254
|
+
}
|
|
1255
|
+
return in_pos + len
|
|
1256
|
+
}
|
|
1257
|
+
|
|
1258
|
+
// Greedy/lazy backtracking for a backreference atom quantified with '*' or
|
|
1259
|
+
// '+' (each iteration consumes the referenced span).
|
|
1260
|
+
func match_star_backref = (
|
|
1261
|
+
string pattern,
|
|
1262
|
+
int atom_start,
|
|
1263
|
+
int rest,
|
|
1264
|
+
int pat_end,
|
|
1265
|
+
string input,
|
|
1266
|
+
int in_pos,
|
|
1267
|
+
int min_count,
|
|
1268
|
+
int lazy,
|
|
1269
|
+
int in_len,
|
|
1270
|
+
ref List<int> caps,
|
|
1271
|
+
out int
|
|
1272
|
+
) {
|
|
1273
|
+
var int group = backref_group(pattern.at(atom_start + 1))
|
|
1274
|
+
if lazy > 0 {
|
|
1275
|
+
if min_count <= 0 {
|
|
1276
|
+
var int stop = match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
|
|
1277
|
+
if stop >= 0 {
|
|
1278
|
+
return stop
|
|
1279
|
+
}
|
|
1280
|
+
}
|
|
1281
|
+
var int next = match_backref(input, group, in_pos, in_len, ref caps)
|
|
1282
|
+
if next >= 0 {
|
|
1283
|
+
if next != in_pos {
|
|
1284
|
+
var new_min = 0
|
|
1285
|
+
if min_count > 0 {
|
|
1286
|
+
new_min = min_count - 1
|
|
1287
|
+
}
|
|
1288
|
+
return match_star_backref(
|
|
1289
|
+
pattern,
|
|
1290
|
+
atom_start,
|
|
1291
|
+
rest,
|
|
1292
|
+
pat_end,
|
|
1293
|
+
input,
|
|
1294
|
+
next,
|
|
1295
|
+
new_min,
|
|
1296
|
+
lazy,
|
|
1297
|
+
in_len,
|
|
1298
|
+
ref caps,
|
|
1299
|
+
)
|
|
1300
|
+
}
|
|
1301
|
+
}
|
|
1302
|
+
return -1
|
|
1303
|
+
}
|
|
1304
|
+
var int next = match_backref(input, group, in_pos, in_len, ref caps)
|
|
1305
|
+
if next >= 0 {
|
|
1306
|
+
if next != in_pos {
|
|
1307
|
+
var new_min = 0
|
|
1308
|
+
if min_count > 0 {
|
|
1309
|
+
new_min = min_count - 1
|
|
1310
|
+
}
|
|
1311
|
+
var int r = match_star_backref(
|
|
1312
|
+
pattern,
|
|
1313
|
+
atom_start,
|
|
1314
|
+
rest,
|
|
1315
|
+
pat_end,
|
|
1316
|
+
input,
|
|
1317
|
+
next,
|
|
1318
|
+
new_min,
|
|
1319
|
+
lazy,
|
|
1320
|
+
in_len,
|
|
1321
|
+
ref caps,
|
|
1322
|
+
)
|
|
1323
|
+
if r >= 0 {
|
|
1324
|
+
return r
|
|
1325
|
+
}
|
|
1326
|
+
}
|
|
1327
|
+
}
|
|
1328
|
+
if min_count <= 0 {
|
|
1329
|
+
return match_here(pattern, rest, pat_end, input, in_pos, in_len, ref caps)
|
|
1330
|
+
}
|
|
1331
|
+
return -1
|
|
1332
|
+
}
|
|
1333
|
+
|
|
1334
|
+
// Whether the escaped character c names an ASCII shorthand class:
|
|
1335
|
+
// \d \D (digits), \w \W (word characters), \s \S (whitespace).
|
|
1336
|
+
func is_shorthand = (char c, out bool) {
|
|
1337
|
+
var int code = c as int
|
|
1338
|
+
return code == 100 || code == 68 || code == 119 || code == 87 || code == 115 || code == 83
|
|
1339
|
+
}
|
|
1340
|
+
|
|
1341
|
+
// Whether byte b belongs to the shorthand class named by c. Uppercase names
|
|
1342
|
+
// are the complements (S/D/W match anything the lowercase set excludes).
|
|
1343
|
+
func matches_shorthand = (char c, char b, out bool) {
|
|
1344
|
+
var int kind = c as int
|
|
1345
|
+
var int code = b as int
|
|
1346
|
+
var found = false
|
|
1347
|
+
if kind == 100 || kind == 68 {
|
|
1348
|
+
// \d \D
|
|
1349
|
+
found = code >= 48 && code <= 57
|
|
1350
|
+
} else {
|
|
1351
|
+
if kind == 119 || kind == 87 {
|
|
1352
|
+
// \w \W
|
|
1353
|
+
found = (code >= 97 && code <= 122) || (code >= 65 && code <= 90) || (code >= 48 && code <= 57) || code == 95
|
|
1354
|
+
} else {
|
|
1355
|
+
// \s \S
|
|
1356
|
+
found = (code >= 9 && code <= 13) || code == 32
|
|
1357
|
+
}
|
|
1358
|
+
}
|
|
1359
|
+
if kind >= 65 && kind <= 90 {
|
|
1360
|
+
return !found
|
|
1361
|
+
}
|
|
1362
|
+
return found
|
|
1363
|
+
}
|
|
1364
|
+
|
|
1365
|
+
// Append the member bytes of the lowercase shorthand class named by c to the
|
|
1366
|
+
// first-byte charset. Complement shorthands (S/D/W) match "anything else" —
|
|
1367
|
+
// not expressible as an inclusion set — so they return -1, which disables
|
|
1368
|
+
// the caller's prefilter.
|
|
1369
|
+
func shorthand_members = (char c, ref StringBuilder sb, out int) {
|
|
1370
|
+
var int kind = c as int
|
|
1371
|
+
if kind == 83 || kind == 68 || kind == 87 {
|
|
1372
|
+
return -1
|
|
1373
|
+
}
|
|
1374
|
+
if kind == 100 {
|
|
1375
|
+
var int code = 48
|
|
1376
|
+
while code <= 57; code += 1 {
|
|
1377
|
+
sb.append_char(code as char)
|
|
1378
|
+
}
|
|
1379
|
+
return 0
|
|
1380
|
+
}
|
|
1381
|
+
if kind == 119 {
|
|
1382
|
+
var int code = 48
|
|
1383
|
+
while code <= 57; code += 1 {
|
|
1384
|
+
sb.append_char(code as char)
|
|
1385
|
+
}
|
|
1386
|
+
var int upper = 65
|
|
1387
|
+
while upper <= 90; upper += 1 {
|
|
1388
|
+
sb.append_char(upper as char)
|
|
1389
|
+
}
|
|
1390
|
+
var int lower = 97
|
|
1391
|
+
while lower <= 122; lower += 1 {
|
|
1392
|
+
sb.append_char(lower as char)
|
|
1393
|
+
}
|
|
1394
|
+
sb.append_char('_')
|
|
1395
|
+
return 0
|
|
1396
|
+
}
|
|
1397
|
+
// \s: tab, LF, VT, FF, CR, space
|
|
1398
|
+
var int w = 9
|
|
1399
|
+
while w <= 13; w += 1 {
|
|
1400
|
+
sb.append_char(w as char)
|
|
1401
|
+
}
|
|
1402
|
+
sb.append_char(32 as char)
|
|
1403
|
+
return 0
|
|
1404
|
+
}
|