nomen-lang 0.2.3 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/NOMEN_AGENTS.md +7 -1
- package/core/System/Arena.nm +166 -0
- package/core/System/Array.nm +30 -161
- package/core/System/BigInt.nm +36 -57
- package/core/System/Buffer.nm +197 -279
- package/core/System/ClassBuffer.nm +120 -130
- package/core/System/Console.nm +4 -2
- package/core/System/Controls/Button.nm +4 -4
- package/core/System/Controls/CheckBox.nm +1 -1
- package/core/System/Controls/LayoutParams.nm +1 -1
- package/core/System/Controls/Text.nm +2 -2
- package/core/System/Controls/TextBox.nm +6 -6
- package/core/System/Controls/Window.nm +2 -2
- package/core/System/Graph.nm +4 -4
- package/core/System/LinkedList.nm +4 -4
- package/core/System/List.nm +6 -6
- package/core/System/Map.nm +33 -30
- package/core/System/Set.nm +24 -24
- package/core/System/Stream/Directory.nm +4 -4
- package/core/System/Stream/File.nm +19 -12
- package/core/System/String.nm +300 -34
- package/core/System/StringBuilder.nm +53 -117
- package/core/System/Text/CharIndex.nm +144 -0
- package/core/System/Text/Chars.nm +32 -0
- package/core/System/Text/JsonTree.nm +48 -96
- package/core/System/Text/Regex.nm +726 -40
- package/core/System/Text/Utf8.nm +163 -0
- package/core/System/bool.nm +2 -1
- package/core/System/char.nm +27 -1
- package/core/System/float.nm +2 -1
- package/core/System/float32.nm +2 -1
- package/core/System/float64.nm +2 -1
- package/core/System/int.nm +4 -2
- package/core/System/int16.nm +2 -1
- package/core/System/int32.nm +2 -1
- package/core/System/int64.nm +2 -1
- package/core/System/int8.nm +2 -1
- package/core/System/ufloat.nm +2 -1
- package/core/System/ufloat32.nm +2 -1
- package/core/System/ufloat64.nm +2 -1
- package/core/System/uint.nm +2 -1
- package/core/System/uint16.nm +2 -1
- package/core/System/uint32.nm +2 -1
- package/core/System/uint64.nm +2 -1
- package/core/System/uint8.nm +2 -1
- package/core/docs/System/BigInt.md +0 -1
- package/dist/index.mjs +5273 -1339
- package/package.json +1 -1
- package/src/index.ts +58 -7
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
// A run-length index of character widths for O(log n) random access.
|
|
2
|
+
//
|
|
3
|
+
// ASCII text needs no records at all (width-1 runs are implicit); only runs
|
|
4
|
+
// of multi-byte characters are stored, each as {byte_index, char_start,
|
|
5
|
+
// width, count}. All lookups binary-search the runs, so repeated indexed
|
|
6
|
+
// access stays logarithmic no matter how often callers re-walk. The index
|
|
7
|
+
// borrows its source (no copy of the text).
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* A run of consecutive same-width multi-byte characters
|
|
11
|
+
**/
|
|
12
|
+
pub struct CharRun {
|
|
13
|
+
var int byte_index
|
|
14
|
+
var int char_start
|
|
15
|
+
var int width
|
|
16
|
+
var int count
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* A run-length character-width index (construct with `CharIndex(text)`,
|
|
21
|
+
* then `byte_offset_of` / `char_index_of` / `char_at` in O(log runs))
|
|
22
|
+
**/
|
|
23
|
+
pub struct CharIndex {
|
|
24
|
+
var view string source
|
|
25
|
+
var List<CharRun> runs = List<CharRun>()
|
|
26
|
+
var int char_count = 0
|
|
27
|
+
|
|
28
|
+
pub func #init = (self, view string source) {
|
|
29
|
+
self.source = source
|
|
30
|
+
var int pos = 0
|
|
31
|
+
var int len = source.length
|
|
32
|
+
var int chars = 0
|
|
33
|
+
var int open_at = -1
|
|
34
|
+
var int open_width = 0
|
|
35
|
+
var int open_start = 0
|
|
36
|
+
var int open_count = 0
|
|
37
|
+
while pos < len {
|
|
38
|
+
var int w = Utf8.width_at(source, pos)
|
|
39
|
+
switch {
|
|
40
|
+
case w == 1 {
|
|
41
|
+
if open_at >= 0 {
|
|
42
|
+
self.runs.push(CharRun(open_at, open_start, open_width, open_count))
|
|
43
|
+
open_at = -1
|
|
44
|
+
}
|
|
45
|
+
pos = pos + 1
|
|
46
|
+
chars = chars + 1
|
|
47
|
+
}
|
|
48
|
+
case open_at >= 0 && w == open_width {
|
|
49
|
+
open_count = open_count + 1
|
|
50
|
+
pos = pos + w
|
|
51
|
+
chars = chars + 1
|
|
52
|
+
}
|
|
53
|
+
else {
|
|
54
|
+
if open_at >= 0 {
|
|
55
|
+
self.runs.push(CharRun(open_at, open_start, open_width, open_count))
|
|
56
|
+
}
|
|
57
|
+
open_at = pos
|
|
58
|
+
open_start = chars
|
|
59
|
+
open_width = w
|
|
60
|
+
open_count = 1
|
|
61
|
+
pos = pos + w
|
|
62
|
+
chars = chars + 1
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
if open_at >= 0 {
|
|
67
|
+
self.runs.push(CharRun(open_at, open_start, open_width, open_count))
|
|
68
|
+
}
|
|
69
|
+
self.char_count = chars
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
// Index built over source in one pass. Construct directly —
|
|
73
|
+
// `CharIndex(text)` borrows the caller's text with no copy.
|
|
74
|
+
pub func byte_offset_of = (self, int char_index, out int) {
|
|
75
|
+
if char_index < 0 || char_index >= self.char_count {
|
|
76
|
+
panic("character index out of range")
|
|
77
|
+
}
|
|
78
|
+
var int found = self.rightmost_run(char_index, true)
|
|
79
|
+
if found < 0 {
|
|
80
|
+
// No run starts at or before it: pure ASCII prefix.
|
|
81
|
+
return char_index
|
|
82
|
+
}
|
|
83
|
+
var r = self.runs.at_or_panic(found)
|
|
84
|
+
if char_index < r.char_start + r.count {
|
|
85
|
+
return r.byte_index + (char_index - r.char_start) * r.width
|
|
86
|
+
}
|
|
87
|
+
var int after_chars = r.char_start + r.count
|
|
88
|
+
var int after_bytes = r.byte_index + r.count * r.width
|
|
89
|
+
return after_bytes + (char_index - after_chars)
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
// Character index containing byte_offset. A mid-character offset floors
|
|
93
|
+
// into its character; byte_offset == source.length maps to char_count.
|
|
94
|
+
// Panics if byte_offset is out of [0, source.length].
|
|
95
|
+
pub func char_index_of = (self, int byte_offset, out int) {
|
|
96
|
+
if byte_offset < 0 || byte_offset > self.source.length {
|
|
97
|
+
panic("byte offset out of range")
|
|
98
|
+
}
|
|
99
|
+
if byte_offset == self.source.length {
|
|
100
|
+
return self.char_count
|
|
101
|
+
}
|
|
102
|
+
var int found = self.rightmost_run(byte_offset, false)
|
|
103
|
+
if found < 0 {
|
|
104
|
+
return byte_offset
|
|
105
|
+
}
|
|
106
|
+
var r = self.runs.at_or_panic(found)
|
|
107
|
+
if byte_offset < r.byte_index + r.count * r.width {
|
|
108
|
+
return r.char_start + (byte_offset - r.byte_index) / r.width
|
|
109
|
+
}
|
|
110
|
+
var int after_chars = r.char_start + r.count
|
|
111
|
+
var int after_bytes = r.byte_index + r.count * r.width
|
|
112
|
+
return after_chars + (byte_offset - after_bytes)
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// Decoded code point of the char_index-th character. Panics if
|
|
116
|
+
// char_index is out of [0, char_count).
|
|
117
|
+
pub func char_at = (self, int char_index, out int) {
|
|
118
|
+
var int at = self.byte_offset_of(char_index)
|
|
119
|
+
var d = Utf8.decode_at(self.source, at)
|
|
120
|
+
return d.code_point
|
|
121
|
+
}
|
|
122
|
+
// Rightmost run whose key (char_start when by_char, else byte_index) is
|
|
123
|
+
// <= target, or -1 when all runs start after it.
|
|
124
|
+
func rightmost_run = (self, int target, bool by_char, out int) {
|
|
125
|
+
var int lo = 0
|
|
126
|
+
var int hi = self.runs.length - 1
|
|
127
|
+
var int found = -1
|
|
128
|
+
while lo <= hi {
|
|
129
|
+
var int mid = (lo + hi) / 2
|
|
130
|
+
var r = self.runs.at_or_panic(mid)
|
|
131
|
+
var int key = r.char_start
|
|
132
|
+
if !by_char {
|
|
133
|
+
key = r.byte_index
|
|
134
|
+
}
|
|
135
|
+
if key <= target {
|
|
136
|
+
found = mid
|
|
137
|
+
lo = mid + 1
|
|
138
|
+
} else {
|
|
139
|
+
hi = mid - 1
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
return found
|
|
143
|
+
}
|
|
144
|
+
}
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
// A single-pass cursor over a string's Unicode scalar values.
|
|
2
|
+
//
|
|
3
|
+
// The cursor borrows its source (no copy) and decodes lazily: each `next`
|
|
4
|
+
// costs one decode step. For repeated random access, build a CharIndex
|
|
5
|
+
// instead — re-walking from scratch on every access is O(n) per lookup.
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* A single-pass cursor over Unicode scalar values (`has_next`/`next`)
|
|
9
|
+
**/
|
|
10
|
+
pub struct Chars {
|
|
11
|
+
var view string source
|
|
12
|
+
var int byte_pos = 0
|
|
13
|
+
|
|
14
|
+
pub func #init = (self, view string source) {
|
|
15
|
+
self.source = source
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
pub func has_next = (self, out bool) {
|
|
19
|
+
return self.byte_pos < self.source.length
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
// The code point at the cursor, advancing past it. Malformed sequences
|
|
23
|
+
// yield U+FFFD. Panics when exhausted — check has_next first.
|
|
24
|
+
pub func next = (ref self, out int) {
|
|
25
|
+
if self.byte_pos >= self.source.length {
|
|
26
|
+
panic("Chars exhausted; check has_next first")
|
|
27
|
+
}
|
|
28
|
+
var d = Utf8.decode_at(self.source, self.byte_pos)
|
|
29
|
+
self.byte_pos = self.byte_pos + d.byte_length
|
|
30
|
+
return d.code_point
|
|
31
|
+
}
|
|
32
|
+
}
|
|
@@ -28,71 +28,43 @@ pub struct JsonTree {
|
|
|
28
28
|
|
|
29
29
|
func alloc_node = (ref self, out int) {
|
|
30
30
|
var int idx = self.count
|
|
31
|
-
self.nodes.
|
|
31
|
+
self.nodes.grow(idx + 1)
|
|
32
32
|
self.count = idx + 1
|
|
33
33
|
var default = JsonNode()
|
|
34
|
-
self.nodes.
|
|
34
|
+
self.nodes.store(idx, default)
|
|
35
35
|
// store_T deep-copies the owning `text` field (strdup of the shared
|
|
36
36
|
// empty-string literal) into the slot. JsonTree manages node texts
|
|
37
37
|
// manually (set_text strdup's into the slab; free_text/release frees
|
|
38
38
|
// them; NULL means "no text"), so the slot must start NULL — free the
|
|
39
|
-
// strdup'd default and NULL it, or reset() would leak it.
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
ldr x0, [x19, #16]
|
|
53
|
-
ldr x1, [x19, #32]
|
|
54
|
-
sub x1, x1, #1
|
|
55
|
-
mov x3, #56
|
|
56
|
-
madd x1, x1, x3, xzr
|
|
57
|
-
add x0, x0, x1
|
|
58
|
-
add x0, x0, #16
|
|
59
|
-
ldr x1, [x0]
|
|
60
|
-
str xzr, [x0]
|
|
61
|
-
str xzr, [x0, #8]
|
|
62
|
-
cbz x1, .Lalloc_node_text_skip
|
|
63
|
-
mov x0, x1
|
|
64
|
-
bl _free
|
|
65
|
-
.Lalloc_node_text_skip:
|
|
66
|
-
```
|
|
39
|
+
// strdup'd default and NULL it, or reset() would leak it. The slab
|
|
40
|
+
// slot layout: 56-byte JsonNode elements — 7 words — with the text
|
|
41
|
+
// pair at words 2 and 3 of each element.
|
|
42
|
+
unsafe {
|
|
43
|
+
var ptr uint64 slots = self.nodes.data as ptr uint64
|
|
44
|
+
var int t = idx * 7 + 2
|
|
45
|
+
var uint64 old = slots[t]
|
|
46
|
+
if old != 0 {
|
|
47
|
+
free(old)
|
|
48
|
+
}
|
|
49
|
+
slots[t] = 0
|
|
50
|
+
slots[t + 1] = 0
|
|
51
|
+
}
|
|
67
52
|
return idx
|
|
68
53
|
}
|
|
69
54
|
|
|
70
55
|
// Free a node's strdup'd text pointer (set_text writes a heap pointer via
|
|
71
56
|
// raw strdup; this mirrors it). idx is x1, following set_text's layout.
|
|
72
57
|
func free_text = (ref self, int idx) {
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
madd x1, x1, x3, xzr
|
|
84
|
-
add x0, x0, x1
|
|
85
|
-
add x0, x0, #16
|
|
86
|
-
ldr x2, [x0]
|
|
87
|
-
cbz x2, .Ljsontree_free_text_skip
|
|
88
|
-
str x0, [sp, #-16]!
|
|
89
|
-
mov x0, x2
|
|
90
|
-
bl _free
|
|
91
|
-
ldr x0, [sp], #16
|
|
92
|
-
str xzr, [x0]
|
|
93
|
-
str xzr, [x0, #8]
|
|
94
|
-
.Ljsontree_free_text_skip:
|
|
95
|
-
```
|
|
58
|
+
unsafe {
|
|
59
|
+
var ptr uint64 slots = self.nodes.data as ptr uint64
|
|
60
|
+
var int t = idx * 7 + 2
|
|
61
|
+
var uint64 p = slots[t]
|
|
62
|
+
if p != 0 {
|
|
63
|
+
free(p)
|
|
64
|
+
slots[t] = 0
|
|
65
|
+
slots[t + 1] = 0
|
|
66
|
+
}
|
|
67
|
+
}
|
|
96
68
|
}
|
|
97
69
|
|
|
98
70
|
// Free every strdup'd text pointer before reusing the slab, otherwise
|
|
@@ -121,84 +93,64 @@ pub struct JsonTree {
|
|
|
121
93
|
// emitted behind the _raw_ adapter, so `text` arrives as a thin char*;
|
|
122
94
|
// aarch64 receives the pair x2 = ptr, x3 = len.)
|
|
123
95
|
func set_text = (ref self, int idx, string text) {
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
// text field is at offset 16 in JsonNode (vt[0] + kind[8]); the fat
|
|
135
|
-
// field is ptr@+16, len@+24; sizeof(JsonNode) = 56.
|
|
136
|
-
sub sp, sp, #32
|
|
137
|
-
str x1, [sp, #0]
|
|
138
|
-
str x3, [sp, #8]
|
|
139
|
-
mov x0, x2
|
|
140
|
-
bl _strdup
|
|
141
|
-
str x0, [sp, #16]
|
|
142
|
-
ldr x0, [x19, #16]
|
|
143
|
-
ldr x1, [sp, #0]
|
|
144
|
-
mov x3, #56
|
|
145
|
-
madd x1, x1, x3, xzr
|
|
146
|
-
add x0, x0, x1
|
|
147
|
-
add x0, x0, #16
|
|
148
|
-
ldr x1, [sp, #16]
|
|
149
|
-
str x1, [x0]
|
|
150
|
-
ldr x1, [sp, #8]
|
|
151
|
-
str x1, [x0, #8]
|
|
152
|
-
add sp, sp, #32
|
|
153
|
-
```
|
|
96
|
+
unsafe {
|
|
97
|
+
var int len = text.length
|
|
98
|
+
var uint64 buf = malloc(len + 1)
|
|
99
|
+
memcpy(buf, (text as ptr char) as uint64, len)
|
|
100
|
+
memset(buf + len, 0, 1)
|
|
101
|
+
var ptr uint64 slots = self.nodes.data as ptr uint64
|
|
102
|
+
var int t = idx * 7 + 2
|
|
103
|
+
slots[t] = buf
|
|
104
|
+
slots[t + 1] = len
|
|
105
|
+
}
|
|
154
106
|
}
|
|
155
107
|
|
|
156
108
|
func set_kind = (ref self, int idx, int val) {
|
|
157
|
-
var JsonNode n = self.nodes.
|
|
109
|
+
var JsonNode n = self.nodes.load(idx)
|
|
158
110
|
n.kind = val
|
|
159
|
-
self.nodes.
|
|
111
|
+
self.nodes.store(idx, n)
|
|
160
112
|
}
|
|
161
113
|
|
|
162
114
|
func set_child = (ref self, int idx, int val) {
|
|
163
|
-
var JsonNode n = self.nodes.
|
|
115
|
+
var JsonNode n = self.nodes.load(idx)
|
|
164
116
|
n.child = val
|
|
165
|
-
self.nodes.
|
|
117
|
+
self.nodes.store(idx, n)
|
|
166
118
|
}
|
|
167
119
|
|
|
168
120
|
func set_next = (ref self, int idx, int val) {
|
|
169
|
-
var JsonNode n = self.nodes.
|
|
121
|
+
var JsonNode n = self.nodes.load(idx)
|
|
170
122
|
n.next = val
|
|
171
|
-
self.nodes.
|
|
123
|
+
self.nodes.store(idx, n)
|
|
172
124
|
}
|
|
173
125
|
|
|
174
126
|
func set_val = (ref self, int idx, int val) {
|
|
175
|
-
var JsonNode n = self.nodes.
|
|
127
|
+
var JsonNode n = self.nodes.load(idx)
|
|
176
128
|
n.val = val
|
|
177
|
-
self.nodes.
|
|
129
|
+
self.nodes.store(idx, n)
|
|
178
130
|
}
|
|
179
131
|
|
|
180
132
|
func get_kind = (self, int idx, out int) {
|
|
181
|
-
var JsonNode n = self.nodes.
|
|
133
|
+
var JsonNode n = self.nodes.load(idx)
|
|
182
134
|
return n.kind
|
|
183
135
|
}
|
|
184
136
|
|
|
185
137
|
func get_text = (self, int idx, out string) {
|
|
186
|
-
var JsonNode n = self.nodes.
|
|
138
|
+
var JsonNode n = self.nodes.load(idx)
|
|
187
139
|
return n.text
|
|
188
140
|
}
|
|
189
141
|
|
|
190
142
|
func get_child = (self, int idx, out int) {
|
|
191
|
-
var JsonNode n = self.nodes.
|
|
143
|
+
var JsonNode n = self.nodes.load(idx)
|
|
192
144
|
return n.child
|
|
193
145
|
}
|
|
194
146
|
|
|
195
147
|
func get_next = (self, int idx, out int) {
|
|
196
|
-
var JsonNode n = self.nodes.
|
|
148
|
+
var JsonNode n = self.nodes.load(idx)
|
|
197
149
|
return n.next
|
|
198
150
|
}
|
|
199
151
|
|
|
200
152
|
func get_val = (self, int idx, out int) {
|
|
201
|
-
var JsonNode n = self.nodes.
|
|
153
|
+
var JsonNode n = self.nodes.load(idx)
|
|
202
154
|
return n.val
|
|
203
155
|
}
|
|
204
156
|
}
|