nomen-lang 0.2.2 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,163 @@
1
+ // UTF-8 scalar helpers, implemented in pure Nomen.
2
+ //
3
+ // Strings are byte arrays; these functions decode and encode individual
4
+ // Unicode scalar values without copying the text. Malformed sequences yield
5
+ // U+FFFD with a byte length of 1 (each invalid byte is one replacement
6
+ // character), so every byte offset either starts a character or is itself
7
+ // reported as U+FFFD — walking with width_at can never stall or skip.
8
+
9
+ /**
10
+ * A decoded character: its Unicode code point and UTF-8 byte length
11
+ **/
12
+ pub struct DecodedChar {
13
+ var int code_point
14
+ var int byte_length
15
+ }
16
+
17
+ /**
18
+ * UTF-8 scalar helpers (`decode_at`/`width_at`/`encode`/`char_count`)
19
+ **/
20
+ pub struct Utf8 {
21
+ // Decode the code point starting at byte_index.
22
+ // Malformed sequences yield U+FFFD with a byte_length of 1 (each
23
+ // invalid byte is one replacement character). Panics if byte_index is
24
+ // out of range (use Chars for end-aware iteration).
25
+ pub func decode_at = (view string text, int byte_index, out DecodedChar) {
26
+ if byte_index < 0 || byte_index >= text.length {
27
+ panic("byte index out of range")
28
+ }
29
+ var int b0 = text.at(byte_index) as int
30
+ // ASCII fast path.
31
+ if b0 < 0x80 {
32
+ return DecodedChar(b0, 1)
33
+ }
34
+ // Continuation bytes and C0/C1 never start a character.
35
+ if b0 < 0xC2 {
36
+ return DecodedChar(0xFFFD, 1)
37
+ }
38
+ if b0 < 0xE0 {
39
+ var int b1 = cont_byte(text, byte_index + 1, 0x80, 0xBF)
40
+ if b1 < 0 {
41
+ return DecodedChar(0xFFFD, 1)
42
+ }
43
+ return DecodedChar(((b0 & 0x1F) << 6) | (b1 & 0x3F), 2)
44
+ }
45
+ if b0 < 0xF0 {
46
+ // E0 forbids second bytes below A0 (overlong); ED forbids
47
+ // AA-BF (surrogate halves).
48
+ var int lo = 0x80
49
+ var int hi = 0xBF
50
+ if b0 == 0xE0 {
51
+ lo = 0xA0
52
+ }
53
+ if b0 == 0xED {
54
+ hi = 0x9F
55
+ }
56
+ var int b1 = cont_byte(text, byte_index + 1, lo, hi)
57
+ if b1 < 0 {
58
+ return DecodedChar(0xFFFD, 1)
59
+ }
60
+ var int b2 = cont_byte(text, byte_index + 2, 0x80, 0xBF)
61
+ if b2 < 0 {
62
+ return DecodedChar(0xFFFD, 1)
63
+ }
64
+ return DecodedChar(((b0 & 0x0F) << 12) | ((b1 & 0x3F) << 6) | (b2 & 0x3F), 3)
65
+ }
66
+ if b0 > 0xF4 {
67
+ return DecodedChar(0xFFFD, 1)
68
+ }
69
+ // F0 forbids second bytes below 90 (overlong); F4 forbids above 8F
70
+ // (past U+10FFFF).
71
+ var int lo = 0x80
72
+ var int hi = 0xBF
73
+ if b0 == 0xF0 {
74
+ lo = 0x90
75
+ }
76
+ if b0 == 0xF4 {
77
+ hi = 0x8F
78
+ }
79
+ var int b1 = cont_byte(text, byte_index + 1, lo, hi)
80
+ if b1 < 0 {
81
+ return DecodedChar(0xFFFD, 1)
82
+ }
83
+ var int b2 = cont_byte(text, byte_index + 2, 0x80, 0xBF)
84
+ if b2 < 0 {
85
+ return DecodedChar(0xFFFD, 1)
86
+ }
87
+ var int b3 = cont_byte(text, byte_index + 3, 0x80, 0xBF)
88
+ if b3 < 0 {
89
+ return DecodedChar(0xFFFD, 1)
90
+ }
91
+ return DecodedChar(
92
+ ((b0 & 0x07) << 18) | ((b1 & 0x3F) << 12) | ((b2 & 0x3F) << 6) | (b3 & 0x3F),
93
+ 4,
94
+ )
95
+ }
96
+
97
+ // Byte length of the character starting at byte_index (1-4; malformed
98
+ // sequences report 1). Panics if byte_index is out of range.
99
+ pub func width_at = (view string text, int byte_index, out int) {
100
+ var d = Utf8.decode_at(text, byte_index)
101
+ return d.byte_length
102
+ }
103
+
104
+ // Encode a code point as UTF-8. Out-of-range values and surrogate halves
105
+ // encode as U+FFFD.
106
+ pub func encode = (int code_point, move out string) {
107
+ var int code = code_point
108
+ if code < 0 || code > 0x10FFFF {
109
+ code = 0xFFFD
110
+ }
111
+ if code >= 0xD800 && code <= 0xDFFF {
112
+ code = 0xFFFD
113
+ }
114
+ var sb = StringBuilder()
115
+ switch {
116
+ case code < 0x80 {
117
+ sb.append_char(code as char)
118
+ }
119
+ case code < 0x800 {
120
+ sb.append_char(((0xC0 | (code >> 6)) as char))
121
+ sb.append_char(((0x80 | (code & 0x3F)) as char))
122
+ }
123
+ case code < 0x10000 {
124
+ sb.append_char(((0xE0 | (code >> 12)) as char))
125
+ sb.append_char(((0x80 | ((code >> 6) & 0x3F)) as char))
126
+ sb.append_char(((0x80 | (code & 0x3F)) as char))
127
+ }
128
+ else {
129
+ sb.append_char(((0xF0 | (code >> 18)) as char))
130
+ sb.append_char(((0x80 | ((code >> 12) & 0x3F)) as char))
131
+ sb.append_char(((0x80 | ((code >> 6) & 0x3F)) as char))
132
+ sb.append_char(((0x80 | (code & 0x3F)) as char))
133
+ }
134
+ }
135
+ return sb.to_string()
136
+ }
137
+
138
+ // Count the code points in text (one pass; malformed bytes each count
139
+ // as one U+FFFD).
140
+ pub func char_count = (view string text, out int) {
141
+ var int n = 0
142
+ var int pos = 0
143
+ var int len = text.length
144
+ while pos < len {
145
+ pos = pos + Utf8.width_at(text, pos)
146
+ n = n + 1
147
+ }
148
+ return n
149
+ }
150
+ }
151
+
152
+ // A continuation byte in [lo, hi] at position i, or -1 when i is past the
153
+ // end or the byte is not a continuation in range.
154
+ func cont_byte = (view string text, int i, int lo, int hi, out int) {
155
+ if i < 0 || i >= text.length {
156
+ return -1
157
+ }
158
+ var int b = text.at(i) as int
159
+ if b < lo || b > hi {
160
+ return -1
161
+ }
162
+ return b
163
+ }
@@ -44,4 +44,29 @@ pub struct char: Stringable, Hashable, Equatable {
44
44
  .Lend_char_to_string:
45
45
  ```
46
46
  }
47
+
48
+ // ASCII digit: '0'-'9'. Non-ASCII bytes are false.
49
+ pub func is_digit = (self, out bool) {
50
+ var int code = self as int
51
+ return code >= 48 && code <= 57
52
+ }
53
+
54
+ // ASCII letter: 'A'-'Z' or 'a'-'z'. Non-ASCII bytes are false.
55
+ pub func is_alpha = (self, out bool) {
56
+ var int code = self as int
57
+ return (code >= 65 && code <= 90) || (code >= 97 && code <= 122)
58
+ }
59
+
60
+ // ASCII letter or digit.
61
+ pub func is_alphanumeric = (self, out bool) {
62
+ return self.is_alpha() || self.is_digit()
63
+ }
64
+
65
+ // ASCII whitespace: tab, LF, VT, FF, CR, space. Unicode spaces are
66
+ // false — byte-oriented callers trim with this and pass multibyte
67
+ // sequences through untouched.
68
+ pub func is_ascii_space = (self, out bool) {
69
+ var int code = self as int
70
+ return (code >= 9 && code <= 13) || code == 32
71
+ }
47
72
  }