nomen-lang 0.2.3 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/NOMEN_AGENTS.md +7 -1
  2. package/core/System/Arena.nm +166 -0
  3. package/core/System/Array.nm +30 -161
  4. package/core/System/BigInt.nm +36 -57
  5. package/core/System/Buffer.nm +197 -279
  6. package/core/System/ClassBuffer.nm +120 -130
  7. package/core/System/Console.nm +4 -2
  8. package/core/System/Controls/Button.nm +4 -4
  9. package/core/System/Controls/CheckBox.nm +1 -1
  10. package/core/System/Controls/LayoutParams.nm +1 -1
  11. package/core/System/Controls/Text.nm +2 -2
  12. package/core/System/Controls/TextBox.nm +6 -6
  13. package/core/System/Controls/Window.nm +2 -2
  14. package/core/System/Graph.nm +4 -4
  15. package/core/System/LinkedList.nm +4 -4
  16. package/core/System/List.nm +6 -6
  17. package/core/System/Map.nm +33 -30
  18. package/core/System/Set.nm +24 -24
  19. package/core/System/Stream/Directory.nm +4 -4
  20. package/core/System/Stream/File.nm +19 -12
  21. package/core/System/String.nm +300 -34
  22. package/core/System/StringBuilder.nm +53 -117
  23. package/core/System/Text/CharIndex.nm +144 -0
  24. package/core/System/Text/Chars.nm +32 -0
  25. package/core/System/Text/JsonTree.nm +48 -96
  26. package/core/System/Text/Regex.nm +726 -40
  27. package/core/System/Text/Utf8.nm +163 -0
  28. package/core/System/bool.nm +2 -1
  29. package/core/System/char.nm +27 -1
  30. package/core/System/float.nm +2 -1
  31. package/core/System/float32.nm +2 -1
  32. package/core/System/float64.nm +2 -1
  33. package/core/System/int.nm +4 -2
  34. package/core/System/int16.nm +2 -1
  35. package/core/System/int32.nm +2 -1
  36. package/core/System/int64.nm +2 -1
  37. package/core/System/int8.nm +2 -1
  38. package/core/System/ufloat.nm +2 -1
  39. package/core/System/ufloat32.nm +2 -1
  40. package/core/System/ufloat64.nm +2 -1
  41. package/core/System/uint.nm +2 -1
  42. package/core/System/uint16.nm +2 -1
  43. package/core/System/uint32.nm +2 -1
  44. package/core/System/uint64.nm +2 -1
  45. package/core/System/uint8.nm +2 -1
  46. package/core/docs/System/BigInt.md +0 -1
  47. package/dist/index.mjs +5273 -1339
  48. package/package.json +1 -1
  49. package/src/index.ts +58 -7
@@ -0,0 +1,144 @@
1
+ // A run-length index of character widths for O(log n) random access.
2
+ //
3
+ // ASCII text needs no records at all (width-1 runs are implicit); only runs
4
+ // of multi-byte characters are stored, each as {byte_index, char_start,
5
+ // width, count}. All lookups binary-search the runs, so repeated indexed
6
+ // access stays logarithmic no matter how often callers re-walk. The index
7
+ // borrows its source (no copy of the text).
8
+
9
+ /**
10
+ * A run of consecutive same-width multi-byte characters
11
+ **/
12
+ pub struct CharRun {
13
+ var int byte_index
14
+ var int char_start
15
+ var int width
16
+ var int count
17
+ }
18
+
19
+ /**
20
+ * A run-length character-width index (construct with `CharIndex(text)`,
21
+ * then `byte_offset_of` / `char_index_of` / `char_at` in O(log runs))
22
+ **/
23
+ pub struct CharIndex {
24
+ var view string source
25
+ var List<CharRun> runs = List<CharRun>()
26
+ var int char_count = 0
27
+
28
+ pub func #init = (self, view string source) {
29
+ self.source = source
30
+ var int pos = 0
31
+ var int len = source.length
32
+ var int chars = 0
33
+ var int open_at = -1
34
+ var int open_width = 0
35
+ var int open_start = 0
36
+ var int open_count = 0
37
+ while pos < len {
38
+ var int w = Utf8.width_at(source, pos)
39
+ switch {
40
+ case w == 1 {
41
+ if open_at >= 0 {
42
+ self.runs.push(CharRun(open_at, open_start, open_width, open_count))
43
+ open_at = -1
44
+ }
45
+ pos = pos + 1
46
+ chars = chars + 1
47
+ }
48
+ case open_at >= 0 && w == open_width {
49
+ open_count = open_count + 1
50
+ pos = pos + w
51
+ chars = chars + 1
52
+ }
53
+ else {
54
+ if open_at >= 0 {
55
+ self.runs.push(CharRun(open_at, open_start, open_width, open_count))
56
+ }
57
+ open_at = pos
58
+ open_start = chars
59
+ open_width = w
60
+ open_count = 1
61
+ pos = pos + w
62
+ chars = chars + 1
63
+ }
64
+ }
65
+ }
66
+ if open_at >= 0 {
67
+ self.runs.push(CharRun(open_at, open_start, open_width, open_count))
68
+ }
69
+ self.char_count = chars
70
+ }
71
+
72
+ // Index built over source in one pass. Construct directly —
73
+ // `CharIndex(text)` borrows the caller's text with no copy.
74
+ pub func byte_offset_of = (self, int char_index, out int) {
75
+ if char_index < 0 || char_index >= self.char_count {
76
+ panic("character index out of range")
77
+ }
78
+ var int found = self.rightmost_run(char_index, true)
79
+ if found < 0 {
80
+ // No run starts at or before it: pure ASCII prefix.
81
+ return char_index
82
+ }
83
+ var r = self.runs.at_or_panic(found)
84
+ if char_index < r.char_start + r.count {
85
+ return r.byte_index + (char_index - r.char_start) * r.width
86
+ }
87
+ var int after_chars = r.char_start + r.count
88
+ var int after_bytes = r.byte_index + r.count * r.width
89
+ return after_bytes + (char_index - after_chars)
90
+ }
91
+
92
+ // Character index containing byte_offset. A mid-character offset floors
93
+ // into its character; byte_offset == source.length maps to char_count.
94
+ // Panics if byte_offset is out of [0, source.length].
95
+ pub func char_index_of = (self, int byte_offset, out int) {
96
+ if byte_offset < 0 || byte_offset > self.source.length {
97
+ panic("byte offset out of range")
98
+ }
99
+ if byte_offset == self.source.length {
100
+ return self.char_count
101
+ }
102
+ var int found = self.rightmost_run(byte_offset, false)
103
+ if found < 0 {
104
+ return byte_offset
105
+ }
106
+ var r = self.runs.at_or_panic(found)
107
+ if byte_offset < r.byte_index + r.count * r.width {
108
+ return r.char_start + (byte_offset - r.byte_index) / r.width
109
+ }
110
+ var int after_chars = r.char_start + r.count
111
+ var int after_bytes = r.byte_index + r.count * r.width
112
+ return after_chars + (byte_offset - after_bytes)
113
+ }
114
+
115
+ // Decoded code point of the char_index-th character. Panics if
116
+ // char_index is out of [0, char_count).
117
+ pub func char_at = (self, int char_index, out int) {
118
+ var int at = self.byte_offset_of(char_index)
119
+ var d = Utf8.decode_at(self.source, at)
120
+ return d.code_point
121
+ }
122
+ // Rightmost run whose key (char_start when by_char, else byte_index) is
123
+ // <= target, or -1 when all runs start after it.
124
+ func rightmost_run = (self, int target, bool by_char, out int) {
125
+ var int lo = 0
126
+ var int hi = self.runs.length - 1
127
+ var int found = -1
128
+ while lo <= hi {
129
+ var int mid = (lo + hi) / 2
130
+ var r = self.runs.at_or_panic(mid)
131
+ var int key = r.char_start
132
+ if !by_char {
133
+ key = r.byte_index
134
+ }
135
+ if key <= target {
136
+ found = mid
137
+ lo = mid + 1
138
+ } else {
139
+ hi = mid - 1
140
+ }
141
+ }
142
+ return found
143
+ }
144
+ }
@@ -0,0 +1,32 @@
1
+ // A single-pass cursor over a string's Unicode scalar values.
2
+ //
3
+ // The cursor borrows its source (no copy) and decodes lazily: each `next`
4
+ // costs one decode step. For repeated random access, build a CharIndex
5
+ // instead — re-walking from scratch on every access is O(n) per lookup.
6
+
7
+ /**
8
+ * A single-pass cursor over Unicode scalar values (`has_next`/`next`)
9
+ **/
10
+ pub struct Chars {
11
+ var view string source
12
+ var int byte_pos = 0
13
+
14
+ pub func #init = (self, view string source) {
15
+ self.source = source
16
+ }
17
+
18
+ pub func has_next = (self, out bool) {
19
+ return self.byte_pos < self.source.length
20
+ }
21
+
22
+ // The code point at the cursor, advancing past it. Malformed sequences
23
+ // yield U+FFFD. Panics when exhausted — check has_next first.
24
+ pub func next = (ref self, out int) {
25
+ if self.byte_pos >= self.source.length {
26
+ panic("Chars exhausted; check has_next first")
27
+ }
28
+ var d = Utf8.decode_at(self.source, self.byte_pos)
29
+ self.byte_pos = self.byte_pos + d.byte_length
30
+ return d.code_point
31
+ }
32
+ }
@@ -28,71 +28,43 @@ pub struct JsonTree {
28
28
 
29
29
  func alloc_node = (ref self, out int) {
30
30
  var int idx = self.count
31
- self.nodes.grow_T(idx + 1)
31
+ self.nodes.grow(idx + 1)
32
32
  self.count = idx + 1
33
33
  var default = JsonNode()
34
- self.nodes.store_T(idx, default)
34
+ self.nodes.store(idx, default)
35
35
  // store_T deep-copies the owning `text` field (strdup of the shared
36
36
  // empty-string literal) into the slot. JsonTree manages node texts
37
37
  // manually (set_text strdup's into the slab; free_text/release frees
38
38
  // them; NULL means "no text"), so the slot must start NULL — free the
39
- // strdup'd default and NULL it, or reset() would leak it. idx is x1.
40
- ```
41
- #arch: c
42
- JsonNode *nodes = (JsonNode*)(unsigned long long)self->nodes.data;
43
- free(nodes[idx].text.ptr);
44
- nodes[idx].text.ptr = 0;
45
- nodes[idx].text.len = 0;
46
- ```
47
- ```
48
- #arch: aarch64
49
- // x19 = self. count is at self+32 and already equals idx+1 here, so
50
- // idx = count - 1. nodes.data at self+16, JsonNode size 56 (fat text),
51
- // text at +16 (ptr) / +24 (len).
52
- ldr x0, [x19, #16]
53
- ldr x1, [x19, #32]
54
- sub x1, x1, #1
55
- mov x3, #56
56
- madd x1, x1, x3, xzr
57
- add x0, x0, x1
58
- add x0, x0, #16
59
- ldr x1, [x0]
60
- str xzr, [x0]
61
- str xzr, [x0, #8]
62
- cbz x1, .Lalloc_node_text_skip
63
- mov x0, x1
64
- bl _free
65
- .Lalloc_node_text_skip:
66
- ```
39
+ // strdup'd default and NULL it, or reset() would leak it. The slab
40
+ // slot layout: 56-byte JsonNode elements — 7 words — with the text
41
+ // pair at words 2 and 3 of each element.
42
+ unsafe {
43
+ var ptr uint64 slots = self.nodes.data as ptr uint64
44
+ var int t = idx * 7 + 2
45
+ var uint64 old = slots[t]
46
+ if old != 0 {
47
+ free(old)
48
+ }
49
+ slots[t] = 0
50
+ slots[t + 1] = 0
51
+ }
67
52
  return idx
68
53
  }
69
54
 
70
55
  // Free a node's strdup'd text pointer (set_text writes a heap pointer via
71
56
  // raw strdup; this mirrors it). idx is x1, following set_text's layout.
72
57
  func free_text = (ref self, int idx) {
73
- ```
74
- #arch: c
75
- JsonNode *nodes = (JsonNode*)(unsigned long long)self->nodes.data;
76
- if (nodes[idx].text.ptr) { free(nodes[idx].text.ptr); nodes[idx].text.ptr = 0; nodes[idx].text.len = 0; }
77
- ```
78
- ```
79
- #arch: aarch64
80
- // x19 = self, x1 = idx. Compute &nodes[idx].text (JsonNode size 56).
81
- ldr x0, [x19, #16]
82
- mov x3, #56
83
- madd x1, x1, x3, xzr
84
- add x0, x0, x1
85
- add x0, x0, #16
86
- ldr x2, [x0]
87
- cbz x2, .Ljsontree_free_text_skip
88
- str x0, [sp, #-16]!
89
- mov x0, x2
90
- bl _free
91
- ldr x0, [sp], #16
92
- str xzr, [x0]
93
- str xzr, [x0, #8]
94
- .Ljsontree_free_text_skip:
95
- ```
58
+ unsafe {
59
+ var ptr uint64 slots = self.nodes.data as ptr uint64
60
+ var int t = idx * 7 + 2
61
+ var uint64 p = slots[t]
62
+ if p != 0 {
63
+ free(p)
64
+ slots[t] = 0
65
+ slots[t + 1] = 0
66
+ }
67
+ }
96
68
  }
97
69
 
98
70
  // Free every strdup'd text pointer before reusing the slab, otherwise
@@ -121,84 +93,64 @@ pub struct JsonTree {
121
93
  // emitted behind the _raw_ adapter, so `text` arrives as a thin char*;
122
94
  // aarch64 receives the pair x2 = ptr, x3 = len.)
123
95
  func set_text = (ref self, int idx, string text) {
124
- ```
125
- #arch: c
126
- JsonNode *nodes = (JsonNode*)(unsigned long long)self->nodes.data;
127
- nodes[idx].text.ptr = strdup(text);
128
- nodes[idx].text.len = (long)strlen(text);
129
- ```
130
- ```
131
- #arch: aarch64
132
- // x19 = self, x1 = idx, x2 = text.ptr, x3 = text.len
133
- // nodes.data is at self+16 (JsonTree.vt[0] + Buffer.vt[8])
134
- // text field is at offset 16 in JsonNode (vt[0] + kind[8]); the fat
135
- // field is ptr@+16, len@+24; sizeof(JsonNode) = 56.
136
- sub sp, sp, #32
137
- str x1, [sp, #0]
138
- str x3, [sp, #8]
139
- mov x0, x2
140
- bl _strdup
141
- str x0, [sp, #16]
142
- ldr x0, [x19, #16]
143
- ldr x1, [sp, #0]
144
- mov x3, #56
145
- madd x1, x1, x3, xzr
146
- add x0, x0, x1
147
- add x0, x0, #16
148
- ldr x1, [sp, #16]
149
- str x1, [x0]
150
- ldr x1, [sp, #8]
151
- str x1, [x0, #8]
152
- add sp, sp, #32
153
- ```
96
+ unsafe {
97
+ var int len = text.length
98
+ var uint64 buf = malloc(len + 1)
99
+ memcpy(buf, (text as ptr char) as uint64, len)
100
+ memset(buf + len, 0, 1)
101
+ var ptr uint64 slots = self.nodes.data as ptr uint64
102
+ var int t = idx * 7 + 2
103
+ slots[t] = buf
104
+ slots[t + 1] = len
105
+ }
154
106
  }
155
107
 
156
108
  func set_kind = (ref self, int idx, int val) {
157
- var JsonNode n = self.nodes.load_T(idx)
109
+ var JsonNode n = self.nodes.load(idx)
158
110
  n.kind = val
159
- self.nodes.store_T(idx, n)
111
+ self.nodes.store(idx, n)
160
112
  }
161
113
 
162
114
  func set_child = (ref self, int idx, int val) {
163
- var JsonNode n = self.nodes.load_T(idx)
115
+ var JsonNode n = self.nodes.load(idx)
164
116
  n.child = val
165
- self.nodes.store_T(idx, n)
117
+ self.nodes.store(idx, n)
166
118
  }
167
119
 
168
120
  func set_next = (ref self, int idx, int val) {
169
- var JsonNode n = self.nodes.load_T(idx)
121
+ var JsonNode n = self.nodes.load(idx)
170
122
  n.next = val
171
- self.nodes.store_T(idx, n)
123
+ self.nodes.store(idx, n)
172
124
  }
173
125
 
174
126
  func set_val = (ref self, int idx, int val) {
175
- var JsonNode n = self.nodes.load_T(idx)
127
+ var JsonNode n = self.nodes.load(idx)
176
128
  n.val = val
177
- self.nodes.store_T(idx, n)
129
+ self.nodes.store(idx, n)
178
130
  }
179
131
 
180
132
  func get_kind = (self, int idx, out int) {
181
- var JsonNode n = self.nodes.load_T(idx)
133
+ var JsonNode n = self.nodes.load(idx)
182
134
  return n.kind
183
135
  }
184
136
 
185
137
  func get_text = (self, int idx, out string) {
186
- var JsonNode n = self.nodes.load_T(idx)
138
+ var JsonNode n = self.nodes.load(idx)
187
139
  return n.text
188
140
  }
189
141
 
190
142
  func get_child = (self, int idx, out int) {
191
- var JsonNode n = self.nodes.load_T(idx)
143
+ var JsonNode n = self.nodes.load(idx)
192
144
  return n.child
193
145
  }
194
146
 
195
147
  func get_next = (self, int idx, out int) {
196
- var JsonNode n = self.nodes.load_T(idx)
148
+ var JsonNode n = self.nodes.load(idx)
197
149
  return n.next
198
150
  }
199
151
 
200
152
  func get_val = (self, int idx, out int) {
201
- var JsonNode n = self.nodes.load_T(idx)
153
+ var JsonNode n = self.nodes.load(idx)
202
154
  return n.val
203
155
  }
204
156
  }