nomen-lang 0.2.3 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/core/System/Arena.nm +166 -0
- package/core/System/Array.nm +30 -161
- package/core/System/BigInt.nm +31 -42
- package/core/System/Buffer.nm +67 -249
- package/core/System/ClassBuffer.nm +43 -105
- package/core/System/Controls/LayoutParams.nm +1 -1
- package/core/System/Map.nm +3 -0
- package/core/System/String.nm +256 -19
- package/core/System/StringBuilder.nm +53 -117
- package/core/System/Text/CharIndex.nm +144 -0
- package/core/System/Text/Chars.nm +32 -0
- package/core/System/Text/JsonTree.nm +33 -81
- package/core/System/Text/Regex.nm +726 -40
- package/core/System/Text/Utf8.nm +163 -0
- package/core/System/char.nm +25 -0
- package/dist/index.mjs +3477 -405
- package/package.json +1 -1
- package/src/index.ts +58 -7
package/core/System/String.nm
CHANGED
|
@@ -3,6 +3,19 @@
|
|
|
3
3
|
// result is re-wrapped with its length.
|
|
4
4
|
extern func strdup = (string s, move out string)
|
|
5
5
|
|
|
6
|
+
// Raw memory externs — the allocation primitives every `unsafe` Nomen body
|
|
7
|
+
// across core (Buffer, StringBuilder, Array, ...) is built on. This file is
|
|
8
|
+
// part of the always-linked base types, so the declarations are visible to
|
|
9
|
+
// every core file regardless of pull order. Free externs emit under an
|
|
10
|
+
// `extern_<name>` label that wraps the identically-named C symbol, so these
|
|
11
|
+
// can never collide with it (docs/CORE_RAW.md roadmap items 1+2).
|
|
12
|
+
extern func calloc = (int count, int size, out uint64)
|
|
13
|
+
extern func malloc = (int bytes, out uint64)
|
|
14
|
+
extern func realloc = (uint64 address, int bytes, out uint64)
|
|
15
|
+
extern func free = (uint64 address)
|
|
16
|
+
extern func memcpy = (uint64 dst, uint64 src, int bytes, out uint64)
|
|
17
|
+
extern func memset = (uint64 dst, int value, int bytes, out uint64)
|
|
18
|
+
|
|
6
19
|
/**
|
|
7
20
|
* A UTF-8 text string — a heap-owned, length-tracked sequence of bytes
|
|
8
21
|
**/
|
|
@@ -28,17 +41,13 @@ pub struct string: Stringable, Hashable, Equatable {
|
|
|
28
41
|
return strdup(self)
|
|
29
42
|
}
|
|
30
43
|
|
|
44
|
+
// The indexing primitive everything string code builds on — `unsafe`
|
|
45
|
+
// Nomen, single-sourced for both backends. `self as ptr char` is the fat
|
|
46
|
+
// value's backing bytes (the thin `char*` the raw blocks used to poke).
|
|
31
47
|
pub func at = (self, int index: index >= 0 && index < self.length, out char) {
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
```
|
|
36
|
-
```
|
|
37
|
-
#arch: aarch64
|
|
38
|
-
// Fat-string ABI: x19 = self ptr, x20 = self len; index in x2
|
|
39
|
-
// (self's pair occupies slots 0-1).
|
|
40
|
-
ldrb w0, [x19, x2]
|
|
41
|
-
```
|
|
48
|
+
unsafe {
|
|
49
|
+
return (self as ptr char)[index]
|
|
50
|
+
}
|
|
42
51
|
}
|
|
43
52
|
|
|
44
53
|
// Bounds-checked access: `fallback` when index is outside
|
|
@@ -87,15 +96,9 @@ pub struct string: Stringable, Hashable, Equatable {
|
|
|
87
96
|
}
|
|
88
97
|
|
|
89
98
|
pub func set = (ref self, int index: index >= 0 && index < self.length, char value) {
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
```
|
|
94
|
-
```
|
|
95
|
-
#arch: aarch64
|
|
96
|
-
// x19 = self (pointer), x1 = index, w2 = value
|
|
97
|
-
strb w2, [x19, x1]
|
|
98
|
-
```
|
|
99
|
+
unsafe {
|
|
100
|
+
(self as ptr char)[index] = value
|
|
101
|
+
}
|
|
99
102
|
}
|
|
100
103
|
|
|
101
104
|
pub func #op_eq = (self, string other, out bool) {
|
|
@@ -209,4 +212,238 @@ pub struct string: Stringable, Hashable, Equatable {
|
|
|
209
212
|
ldp x20, x21, [sp], #16
|
|
210
213
|
```
|
|
211
214
|
}
|
|
215
|
+
|
|
216
|
+
// Replace the first non-overlapping occurrence of needle with
|
|
217
|
+
// replacement. No match — or an empty needle — returns an owned copy
|
|
218
|
+
// of self. Views pass straight in (no `to_string` copy): both sides
|
|
219
|
+
// are only ever read.
|
|
220
|
+
pub func replace_first = (self, view string needle, view string replacement, out string) {
|
|
221
|
+
var sb = StringBuilder()
|
|
222
|
+
var int i = 0
|
|
223
|
+
var int n = self.length
|
|
224
|
+
var int nlen = needle.length
|
|
225
|
+
var bool replaced = nlen == 0
|
|
226
|
+
while i < n {
|
|
227
|
+
if !replaced && nlen > 0 && i + nlen <= n && match_at(self, needle, i) {
|
|
228
|
+
sb.append_string_view(replacement)
|
|
229
|
+
i = i + nlen
|
|
230
|
+
replaced = true
|
|
231
|
+
} else {
|
|
232
|
+
sb.append_char(self.at(i))
|
|
233
|
+
i = i + 1
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
return sb.to_string()
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
// Replace all non-overlapping occurrences of needle with replacement,
|
|
240
|
+
// left to right. Same edge behavior as replace_first.
|
|
241
|
+
pub func replace_all = (self, view string needle, view string replacement, out string) {
|
|
242
|
+
var sb = StringBuilder()
|
|
243
|
+
var int i = 0
|
|
244
|
+
var int n = self.length
|
|
245
|
+
var int nlen = needle.length
|
|
246
|
+
while i < n {
|
|
247
|
+
if nlen > 0 && i + nlen <= n && match_at(self, needle, i) {
|
|
248
|
+
sb.append_string_view(replacement)
|
|
249
|
+
i = i + nlen
|
|
250
|
+
} else {
|
|
251
|
+
sb.append_char(self.at(i))
|
|
252
|
+
i = i + 1
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
return sb.to_string()
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
// First index of `needle` in self, or -1 when absent. An empty needle
|
|
259
|
+
// matches at 0 (JavaScript indexOf semantics).
|
|
260
|
+
pub func index_of = (self, view string needle, out int) {
|
|
261
|
+
return self.index_of_from(needle, 0)
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
// First index of `needle` at or after byte offset `from`, or -1. `from`
|
|
265
|
+
// is clamped into [0, self.length]; an empty needle returns the clamped
|
|
266
|
+
// `from`, so a `from` past the end yields self.length for the empty
|
|
267
|
+
// needle and -1 otherwise.
|
|
268
|
+
pub func index_of_from = (self, view string needle, int from, out int) {
|
|
269
|
+
var int f = from
|
|
270
|
+
if f < 0 {
|
|
271
|
+
f = 0
|
|
272
|
+
}
|
|
273
|
+
if f > self.length {
|
|
274
|
+
f = self.length
|
|
275
|
+
}
|
|
276
|
+
if needle.length == 0 {
|
|
277
|
+
return f
|
|
278
|
+
}
|
|
279
|
+
var int last = self.length - needle.length
|
|
280
|
+
var int i = f
|
|
281
|
+
while i <= last; i += 1 {
|
|
282
|
+
if match_at(self, needle, i) {
|
|
283
|
+
return i
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
return -1
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
// Whether `needle` occurs anywhere in self.
|
|
290
|
+
pub func contains = (self, view string needle, out bool) {
|
|
291
|
+
return self.index_of(needle) >= 0
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
// Whether self begins with `prefix`.
|
|
295
|
+
pub func starts_with = (self, view string prefix, out bool) {
|
|
296
|
+
if prefix.length > self.length {
|
|
297
|
+
return false
|
|
298
|
+
}
|
|
299
|
+
return match_at(self, prefix, 0)
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
// Whether self ends with `suffix`.
|
|
303
|
+
pub func ends_with = (self, view string suffix, out bool) {
|
|
304
|
+
if suffix.length > self.length {
|
|
305
|
+
return false
|
|
306
|
+
}
|
|
307
|
+
return match_at(self, suffix, self.length - suffix.length)
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
// The code of the byte at `index`, or -1 when out of bounds — the
|
|
311
|
+
// JavaScript charCodeAt NaN case as a plain int parsers can test.
|
|
312
|
+
// The code of the byte at `index`. Like `at`, the constraint makes
|
|
313
|
+
// out-of-bounds reads unrepresentable: a literal index is verified at
|
|
314
|
+
// compile time, and a dynamic index must be proven in-range by flow
|
|
315
|
+
// facts (a `while i < self.length` loop or an `if` guard) — an
|
|
316
|
+
// unverifiable call is a compile error. For a sentinel read, use
|
|
317
|
+
// `char_code_at_or`; for a trapping read, `char_code_at_or_panic`.
|
|
318
|
+
pub func char_code_at = (self, int index: index >= 0 && index < self.length, out int) {
|
|
319
|
+
return self.at(index) as int
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
// The code of the byte at `index`, with `fallback` when index is
|
|
323
|
+
// outside [0, self.length).
|
|
324
|
+
pub func char_code_at_or = (self, int index, int fallback, out int) {
|
|
325
|
+
if index >= 0 && index < self.length {
|
|
326
|
+
return self.at(index) as int
|
|
327
|
+
}
|
|
328
|
+
return fallback
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
// The code of the byte at `index`; traps when index is outside
|
|
332
|
+
// [0, self.length) — out-of-range is a bug, not a case.
|
|
333
|
+
pub func char_code_at_or_panic = (self, int index, out int) {
|
|
334
|
+
if index >= 0 && index < self.length {
|
|
335
|
+
return self.at(index) as int
|
|
336
|
+
}
|
|
337
|
+
panic("index out of range")
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
// Owned copy of self[start, end) with both bounds clamped into
|
|
341
|
+
// [0, self.length] and an inverted pair swapped (JavaScript substring
|
|
342
|
+
// semantics: out-of-range operands can never trap).
|
|
343
|
+
pub func substring = (self, int start, int end, out string) {
|
|
344
|
+
var int a = start
|
|
345
|
+
var int b = end
|
|
346
|
+
if a < 0 {
|
|
347
|
+
a = 0
|
|
348
|
+
}
|
|
349
|
+
if a > self.length {
|
|
350
|
+
a = self.length
|
|
351
|
+
}
|
|
352
|
+
if b > self.length {
|
|
353
|
+
b = self.length
|
|
354
|
+
}
|
|
355
|
+
if b < a {
|
|
356
|
+
b = a
|
|
357
|
+
}
|
|
358
|
+
var view v = self.slice(a, b)
|
|
359
|
+
return v.to_string()
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
// Owned copy of self with leading and trailing ASCII whitespace
|
|
363
|
+
// (tab, LF, VT, FF, CR, space) removed. All-whitespace yields "".
|
|
364
|
+
pub func trim = (self, out string) {
|
|
365
|
+
var int start = 0
|
|
366
|
+
while start < self.length; start += 1 {
|
|
367
|
+
if !self.at(start).is_ascii_space() {
|
|
368
|
+
break
|
|
369
|
+
}
|
|
370
|
+
}
|
|
371
|
+
var int end = self.length
|
|
372
|
+
while end > start; end -= 1 {
|
|
373
|
+
if !self.at(end - 1).is_ascii_space() {
|
|
374
|
+
break
|
|
375
|
+
}
|
|
376
|
+
}
|
|
377
|
+
var view v = self.slice(start, end)
|
|
378
|
+
return v.to_string()
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
// Owned copy of self with trailing ASCII whitespace removed.
|
|
382
|
+
pub func trim_end = (self, out string) {
|
|
383
|
+
var int end = self.length
|
|
384
|
+
while end > 0; end -= 1 {
|
|
385
|
+
if !self.at(end - 1).is_ascii_space() {
|
|
386
|
+
break
|
|
387
|
+
}
|
|
388
|
+
}
|
|
389
|
+
var view v = self.slice(0, end)
|
|
390
|
+
return v.to_string()
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
// Owned copy of self with leading ASCII whitespace removed.
|
|
394
|
+
pub func trim_start = (self, out string) {
|
|
395
|
+
var int start = 0
|
|
396
|
+
while start < self.length; start += 1 {
|
|
397
|
+
if !self.at(start).is_ascii_space() {
|
|
398
|
+
break
|
|
399
|
+
}
|
|
400
|
+
}
|
|
401
|
+
var view v = self.slice(start, self.length)
|
|
402
|
+
return v.to_string()
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
// Owned ASCII-lowercased copy of self (A-Z mapped to a-z; all other
|
|
406
|
+
// bytes — including UTF-8 sequences — pass through unchanged).
|
|
407
|
+
pub func to_lowercase = (self, out string) {
|
|
408
|
+
var sb = StringBuilder()
|
|
409
|
+
var int i = 0
|
|
410
|
+
while i < self.length; i += 1 {
|
|
411
|
+
var int code = self.at(i) as int
|
|
412
|
+
if code > 64 && code < 91 {
|
|
413
|
+
code += 32
|
|
414
|
+
}
|
|
415
|
+
sb.append_char(code as char)
|
|
416
|
+
}
|
|
417
|
+
return sb.to_string()
|
|
418
|
+
}
|
|
419
|
+
|
|
420
|
+
// Owned ASCII-uppercased copy of self (a-z mapped to A-Z; all other
|
|
421
|
+
// bytes — including UTF-8 sequences — pass through unchanged).
|
|
422
|
+
pub func to_uppercase = (self, out string) {
|
|
423
|
+
var sb = StringBuilder()
|
|
424
|
+
var int i = 0
|
|
425
|
+
while i < self.length; i += 1 {
|
|
426
|
+
var int code = self.at(i) as int
|
|
427
|
+
if code > 96 && code < 123 {
|
|
428
|
+
code -= 32
|
|
429
|
+
}
|
|
430
|
+
sb.append_char(code as char)
|
|
431
|
+
}
|
|
432
|
+
return sb.to_string()
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
|
|
436
|
+
// needle matches text at byte offset pos (caller guarantees pos + len fits;
|
|
437
|
+
// the trailing guard makes the bound explicit so `.at` verifies).
|
|
438
|
+
func match_at = (view string text, view string needle, int pos, out bool) {
|
|
439
|
+
var int i = 0
|
|
440
|
+
var int nlen = needle.length
|
|
441
|
+
var int n = text.length
|
|
442
|
+
while i < nlen && pos + i < n {
|
|
443
|
+
if text.at(pos + i) != needle.at(i) {
|
|
444
|
+
return false
|
|
445
|
+
}
|
|
446
|
+
i = i + 1
|
|
447
|
+
}
|
|
448
|
+
return i == nlen
|
|
212
449
|
}
|
|
@@ -14,48 +14,19 @@ pub struct StringBuilder {
|
|
|
14
14
|
var cap = 0
|
|
15
15
|
|
|
16
16
|
// Ensure capacity for at least `needed` bytes. Grows geometrically.
|
|
17
|
+
// Plain Nomen + the memory externs — no pointer arithmetic needed, so
|
|
18
|
+
// this doesn't even require an unsafe context.
|
|
17
19
|
func ensure = (ref self, int needed) {
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
int new_cap = self
|
|
22
|
-
if
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
#arch: aarch64
|
|
29
|
-
// x19 = self, x1 = needed
|
|
30
|
-
ldr x2, [x19, #24] // cap
|
|
31
|
-
cmp x2, x1
|
|
32
|
-
b.ge .Lsb_ensure_end
|
|
33
|
-
// pick new_cap
|
|
34
|
-
mov x3, #16
|
|
35
|
-
cmp x2, #0
|
|
36
|
-
csel x3, x3, x2, eq
|
|
37
|
-
.Lsb_ensure_loop:
|
|
38
|
-
cmp x3, x1
|
|
39
|
-
b.ge .Lsb_ensure_grow
|
|
40
|
-
lsl x3, x3, #1
|
|
41
|
-
b .Lsb_ensure_loop
|
|
42
|
-
.Lsb_ensure_grow:
|
|
43
|
-
// save needed (x1) and new_cap (x3) across realloc
|
|
44
|
-
sub sp, sp, #32
|
|
45
|
-
str x1, [sp, #0]
|
|
46
|
-
str x3, [sp, #8]
|
|
47
|
-
str x2, [sp, #16] // old cap (unused after, but keeps stack aligned)
|
|
48
|
-
ldr x0, [x19, #8] // data
|
|
49
|
-
mov x1, x3
|
|
50
|
-
bl _realloc
|
|
51
|
-
str x0, [sp, #24] // saved new data ptr
|
|
52
|
-
ldr x1, [sp, #8] // new_cap
|
|
53
|
-
str x1, [x19, #24] // cap = new_cap
|
|
54
|
-
ldr x0, [sp, #24]
|
|
55
|
-
str x0, [x19, #8] // data = realloc result
|
|
56
|
-
add sp, sp, #32
|
|
57
|
-
.Lsb_ensure_end:
|
|
58
|
-
```
|
|
20
|
+
if self.cap >= needed {
|
|
21
|
+
return
|
|
22
|
+
}
|
|
23
|
+
var int new_cap = self.cap
|
|
24
|
+
if new_cap == 0 {
|
|
25
|
+
new_cap = 16
|
|
26
|
+
}
|
|
27
|
+
while new_cap < needed; new_cap *= 2 {}
|
|
28
|
+
self.data = realloc(self.data, new_cap)
|
|
29
|
+
self.cap = new_cap
|
|
59
30
|
}
|
|
60
31
|
|
|
61
32
|
// Append a single byte (char).
|
|
@@ -110,65 +81,43 @@ pub struct StringBuilder {
|
|
|
110
81
|
```
|
|
111
82
|
}
|
|
112
83
|
|
|
113
|
-
// Append a null-terminated string.
|
|
84
|
+
// Append a null-terminated string. `unsafe` Nomen single-sourced for
|
|
85
|
+
// both backends: the fat pair's pointer half feeds memcpy directly, and
|
|
86
|
+
// the tracked length replaces the strlen the old C body paid.
|
|
114
87
|
func append_string = (ref self, string s) {
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
88
|
+
unsafe {
|
|
89
|
+
var int sl = s.length
|
|
90
|
+
self.ensure(self.len + sl + 1)
|
|
91
|
+
memcpy(self.data + self.len, (s as ptr char) as uint64, sl)
|
|
92
|
+
self.len = self.len + sl
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
// Append a view's bytes without materializing it: `append_string` takes
|
|
97
|
+
// an owned string, so passing a slice would need a `to_string` copy per
|
|
98
|
+
// call. Views lower to the same (ptr, len) fat pair, so the bytes are
|
|
99
|
+
// memcpy'd straight out of the borrowed buffer (used by
|
|
100
|
+
// `string.replace_first`/`replace_all`).
|
|
101
|
+
pub func append_string_view = (ref self, view string s) {
|
|
102
|
+
unsafe {
|
|
103
|
+
var int sl = s.length
|
|
104
|
+
self.ensure(self.len + sl + 1)
|
|
105
|
+
memcpy(self.data + self.len, (s as ptr char) as uint64, sl)
|
|
106
|
+
self.len = self.len + sl
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
// Replace the buffer's contents with a copy of `s`'s bytes. The only
|
|
111
|
+
// way to start a StringBuilder from a slice without a `to_string()`
|
|
112
|
+
// copy: the view's bytes are copied once into fresh storage, so the
|
|
113
|
+
// result is owned and independent of the source.
|
|
114
|
+
pub func seed = (ref self, view string s) {
|
|
115
|
+
unsafe {
|
|
116
|
+
var int sl = s.length
|
|
117
|
+
self.ensure(sl + 1)
|
|
118
|
+
memcpy(self.data, (s as ptr char) as uint64, sl)
|
|
119
|
+
self.len = sl
|
|
125
120
|
}
|
|
126
|
-
memcpy((unsigned char*)(unsigned long long)self->data + self->len, s, sl);
|
|
127
|
-
self->len = self->len + sl;
|
|
128
|
-
```
|
|
129
|
-
```
|
|
130
|
-
#arch: aarch64
|
|
131
|
-
// Fat-string ABI: x19 = self; s = (x1 ptr, x2 len). The length is
|
|
132
|
-
// already in a register — no strlen.
|
|
133
|
-
sub sp, sp, #32
|
|
134
|
-
str x1, [sp, #0] // s ptr
|
|
135
|
-
str x2, [sp, #8] // sl
|
|
136
|
-
ldr x1, [x19, #16] // len
|
|
137
|
-
add x1, x1, x2 // len + sl
|
|
138
|
-
add x1, x1, #1 // +1 for null room
|
|
139
|
-
// inline ensure(x1)
|
|
140
|
-
ldr x2, [x19, #24]
|
|
141
|
-
cmp x2, x1
|
|
142
|
-
b.ge .Lsb_as_copy
|
|
143
|
-
mov x3, #16
|
|
144
|
-
cmp x2, #0
|
|
145
|
-
csel x3, x3, x2, eq
|
|
146
|
-
.Lsb_as_loop:
|
|
147
|
-
cmp x3, x1
|
|
148
|
-
b.ge .Lsb_as_grow
|
|
149
|
-
lsl x3, x3, #1
|
|
150
|
-
b .Lsb_as_loop
|
|
151
|
-
.Lsb_as_grow:
|
|
152
|
-
str x3, [x19, #24]
|
|
153
|
-
ldr x0, [x19, #8]
|
|
154
|
-
mov x1, x3
|
|
155
|
-
bl _realloc
|
|
156
|
-
str x0, [x19, #8]
|
|
157
|
-
.Lsb_as_copy:
|
|
158
|
-
// memcpy(data + len, s.ptr, sl)
|
|
159
|
-
ldr x0, [x19, #8] // data
|
|
160
|
-
ldr x1, [x19, #16] // len
|
|
161
|
-
add x0, x0, x1 // dest = data + len
|
|
162
|
-
ldr x1, [sp, #0] // src = s ptr
|
|
163
|
-
ldr x2, [sp, #8] // n = sl
|
|
164
|
-
bl _memcpy
|
|
165
|
-
// len = len + sl
|
|
166
|
-
ldr x0, [x19, #16]
|
|
167
|
-
ldr x1, [sp, #8]
|
|
168
|
-
add x0, x0, x1
|
|
169
|
-
str x0, [x19, #16]
|
|
170
|
-
add sp, sp, #32
|
|
171
|
-
```
|
|
172
121
|
}
|
|
173
122
|
|
|
174
123
|
// Finalize: null-terminate, return the byte buffer as a string, and
|
|
@@ -188,7 +137,7 @@ pub struct StringBuilder {
|
|
|
188
137
|
((unsigned char*)(unsigned long long)self->data)[idx] = 0;
|
|
189
138
|
unsigned long long result = self->data;
|
|
190
139
|
self->data = 0;
|
|
191
|
-
return (char*)result;
|
|
140
|
+
return (nomen_string){ (char*)result, (long)idx };
|
|
192
141
|
```
|
|
193
142
|
```
|
|
194
143
|
#arch: aarch64
|
|
@@ -224,24 +173,11 @@ pub struct StringBuilder {
|
|
|
224
173
|
|
|
225
174
|
// Drop the buffer without finalizing (free the byte buffer outright).
|
|
226
175
|
func #destroy = () {
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
self
|
|
232
|
-
self->len = 0;
|
|
233
|
-
self->cap = 0;
|
|
176
|
+
if self.data != 0 {
|
|
177
|
+
free(self.data)
|
|
178
|
+
self.data = 0
|
|
179
|
+
self.len = 0
|
|
180
|
+
self.cap = 0
|
|
234
181
|
}
|
|
235
|
-
```
|
|
236
|
-
```
|
|
237
|
-
#arch: aarch64
|
|
238
|
-
ldr x0, [x19, #8]
|
|
239
|
-
cbz x0, .Lsb_destroy_end
|
|
240
|
-
bl _free
|
|
241
|
-
str xzr, [x19, #8]
|
|
242
|
-
str xzr, [x19, #16]
|
|
243
|
-
str xzr, [x19, #24]
|
|
244
|
-
.Lsb_destroy_end:
|
|
245
|
-
```
|
|
246
182
|
}
|
|
247
183
|
}
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
// A run-length index of character widths for O(log n) random access.
|
|
2
|
+
//
|
|
3
|
+
// ASCII text needs no records at all (width-1 runs are implicit); only runs
|
|
4
|
+
// of multi-byte characters are stored, each as {byte_index, char_start,
|
|
5
|
+
// width, count}. All lookups binary-search the runs, so repeated indexed
|
|
6
|
+
// access stays logarithmic no matter how often callers re-walk. The index
|
|
7
|
+
// borrows its source (no copy of the text).
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* A run of consecutive same-width multi-byte characters
|
|
11
|
+
**/
|
|
12
|
+
pub struct CharRun {
|
|
13
|
+
var int byte_index
|
|
14
|
+
var int char_start
|
|
15
|
+
var int width
|
|
16
|
+
var int count
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* A run-length character-width index (construct with `CharIndex(text)`,
|
|
21
|
+
* then `byte_offset_of` / `char_index_of` / `char_at` in O(log runs))
|
|
22
|
+
**/
|
|
23
|
+
pub struct CharIndex {
|
|
24
|
+
var view string source
|
|
25
|
+
var List<CharRun> runs = List<CharRun>()
|
|
26
|
+
var int char_count = 0
|
|
27
|
+
|
|
28
|
+
pub func #init = (self, view string source) {
|
|
29
|
+
self.source = source
|
|
30
|
+
var int pos = 0
|
|
31
|
+
var int len = source.length
|
|
32
|
+
var int chars = 0
|
|
33
|
+
var int open_at = -1
|
|
34
|
+
var int open_width = 0
|
|
35
|
+
var int open_start = 0
|
|
36
|
+
var int open_count = 0
|
|
37
|
+
while pos < len {
|
|
38
|
+
var int w = Utf8.width_at(source, pos)
|
|
39
|
+
switch {
|
|
40
|
+
case w == 1 {
|
|
41
|
+
if open_at >= 0 {
|
|
42
|
+
self.runs.push(CharRun(open_at, open_start, open_width, open_count))
|
|
43
|
+
open_at = -1
|
|
44
|
+
}
|
|
45
|
+
pos = pos + 1
|
|
46
|
+
chars = chars + 1
|
|
47
|
+
}
|
|
48
|
+
case open_at >= 0 && w == open_width {
|
|
49
|
+
open_count = open_count + 1
|
|
50
|
+
pos = pos + w
|
|
51
|
+
chars = chars + 1
|
|
52
|
+
}
|
|
53
|
+
else {
|
|
54
|
+
if open_at >= 0 {
|
|
55
|
+
self.runs.push(CharRun(open_at, open_start, open_width, open_count))
|
|
56
|
+
}
|
|
57
|
+
open_at = pos
|
|
58
|
+
open_start = chars
|
|
59
|
+
open_width = w
|
|
60
|
+
open_count = 1
|
|
61
|
+
pos = pos + w
|
|
62
|
+
chars = chars + 1
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
if open_at >= 0 {
|
|
67
|
+
self.runs.push(CharRun(open_at, open_start, open_width, open_count))
|
|
68
|
+
}
|
|
69
|
+
self.char_count = chars
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
// Index built over source in one pass. Construct directly —
|
|
73
|
+
// `CharIndex(text)` borrows the caller's text with no copy.
|
|
74
|
+
pub func byte_offset_of = (self, int char_index, out int) {
|
|
75
|
+
if char_index < 0 || char_index >= self.char_count {
|
|
76
|
+
panic("character index out of range")
|
|
77
|
+
}
|
|
78
|
+
var int found = self.rightmost_run(char_index, true)
|
|
79
|
+
if found < 0 {
|
|
80
|
+
// No run starts at or before it: pure ASCII prefix.
|
|
81
|
+
return char_index
|
|
82
|
+
}
|
|
83
|
+
var r = self.runs.at_or_panic(found)
|
|
84
|
+
if char_index < r.char_start + r.count {
|
|
85
|
+
return r.byte_index + (char_index - r.char_start) * r.width
|
|
86
|
+
}
|
|
87
|
+
var int after_chars = r.char_start + r.count
|
|
88
|
+
var int after_bytes = r.byte_index + r.count * r.width
|
|
89
|
+
return after_bytes + (char_index - after_chars)
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
// Character index containing byte_offset. A mid-character offset floors
|
|
93
|
+
// into its character; byte_offset == source.length maps to char_count.
|
|
94
|
+
// Panics if byte_offset is out of [0, source.length].
|
|
95
|
+
pub func char_index_of = (self, int byte_offset, out int) {
|
|
96
|
+
if byte_offset < 0 || byte_offset > self.source.length {
|
|
97
|
+
panic("byte offset out of range")
|
|
98
|
+
}
|
|
99
|
+
if byte_offset == self.source.length {
|
|
100
|
+
return self.char_count
|
|
101
|
+
}
|
|
102
|
+
var int found = self.rightmost_run(byte_offset, false)
|
|
103
|
+
if found < 0 {
|
|
104
|
+
return byte_offset
|
|
105
|
+
}
|
|
106
|
+
var r = self.runs.at_or_panic(found)
|
|
107
|
+
if byte_offset < r.byte_index + r.count * r.width {
|
|
108
|
+
return r.char_start + (byte_offset - r.byte_index) / r.width
|
|
109
|
+
}
|
|
110
|
+
var int after_chars = r.char_start + r.count
|
|
111
|
+
var int after_bytes = r.byte_index + r.count * r.width
|
|
112
|
+
return after_chars + (byte_offset - after_bytes)
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// Decoded code point of the char_index-th character. Panics if
|
|
116
|
+
// char_index is out of [0, char_count).
|
|
117
|
+
pub func char_at = (self, int char_index, out int) {
|
|
118
|
+
var int at = self.byte_offset_of(char_index)
|
|
119
|
+
var d = Utf8.decode_at(self.source, at)
|
|
120
|
+
return d.code_point
|
|
121
|
+
}
|
|
122
|
+
// Rightmost run whose key (char_start when by_char, else byte_index) is
|
|
123
|
+
// <= target, or -1 when all runs start after it.
|
|
124
|
+
func rightmost_run = (self, int target, bool by_char, out int) {
|
|
125
|
+
var int lo = 0
|
|
126
|
+
var int hi = self.runs.length - 1
|
|
127
|
+
var int found = -1
|
|
128
|
+
while lo <= hi {
|
|
129
|
+
var int mid = (lo + hi) / 2
|
|
130
|
+
var r = self.runs.at_or_panic(mid)
|
|
131
|
+
var int key = r.char_start
|
|
132
|
+
if !by_char {
|
|
133
|
+
key = r.byte_index
|
|
134
|
+
}
|
|
135
|
+
if key <= target {
|
|
136
|
+
found = mid
|
|
137
|
+
lo = mid + 1
|
|
138
|
+
} else {
|
|
139
|
+
hi = mid - 1
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
return found
|
|
143
|
+
}
|
|
144
|
+
}
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
// A single-pass cursor over a string's Unicode scalar values.
|
|
2
|
+
//
|
|
3
|
+
// The cursor borrows its source (no copy) and decodes lazily: each `next`
|
|
4
|
+
// costs one decode step. For repeated random access, build a CharIndex
|
|
5
|
+
// instead — re-walking from scratch on every access is O(n) per lookup.
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* A single-pass cursor over Unicode scalar values (`has_next`/`next`)
|
|
9
|
+
**/
|
|
10
|
+
pub struct Chars {
|
|
11
|
+
var view string source
|
|
12
|
+
var int byte_pos = 0
|
|
13
|
+
|
|
14
|
+
pub func #init = (self, view string source) {
|
|
15
|
+
self.source = source
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
pub func has_next = (self, out bool) {
|
|
19
|
+
return self.byte_pos < self.source.length
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
// The code point at the cursor, advancing past it. Malformed sequences
|
|
23
|
+
// yield U+FFFD. Panics when exhausted — check has_next first.
|
|
24
|
+
pub func next = (ref self, out int) {
|
|
25
|
+
if self.byte_pos >= self.source.length {
|
|
26
|
+
panic("Chars exhausted; check has_next first")
|
|
27
|
+
}
|
|
28
|
+
var d = Utf8.decode_at(self.source, self.byte_pos)
|
|
29
|
+
self.byte_pos = self.byte_pos + d.byte_length
|
|
30
|
+
return d.code_point
|
|
31
|
+
}
|
|
32
|
+
}
|