@amritk/nish-aarch64-linux 0.14.0 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,528 @@
1
+ /* Nish runtime, the network half: addresses, non-blocking TCP and UDP, and
2
+ * the readiness loop that waits on them (WP34 N5), the `nish:net` builtin
3
+ * module. Every socket it makes is non-blocking
4
+ * and close-on-exec, so a program owns the loop that waits on it, and a child
5
+ * `spawnSync` starts inherits none of them.
6
+ *
7
+ * A translation unit of its own for the reason runtime-host.c is one: each
8
+ * file carries its own measured `.text*` ceiling in tests/run.js, and sockets
9
+ * are a surface that grows, which would push another unit past its ceiling
10
+ * rather than into a new one. Section GC
11
+ * means a program that calls none of these pays for none of them, and
12
+ * scripts/build.sh pairs this file with runtime.c like the other halves, so a
13
+ * link line still names one runtime.
14
+ *
15
+ * The error convention is the one `signalFd` set: an `i32`, `>= 0` for
16
+ * success (a descriptor, a byte count, 0), and a negative errno otherwise.
17
+ * The codes a loop branches on are Linux's numbers on every platform (-11
18
+ * would block, -95 unsupported, -32 the peer is gone, -104 reset, -98 the
19
+ * address is in use, -22 a bad argument), so `nish_net_err` translates
20
+ * Darwin's; any other failure is the host's own `-errno`. Nothing here
21
+ * allocates: the addresses a call reads or writes are the caller's `u8[]`.
22
+ *
23
+ * An address is 18 bytes of that array: the 16 bytes of an IPv6 address, an
24
+ * IPv4 one as `::ffff:a.b.c.d`, then the port, big-endian. An IPv4 address is
25
+ * an `AF_INET` socket; any other is `AF_INET6` with `IPV6_V6ONLY` off, so
26
+ * `::` hears both families, and when the kernel has no IPv6 at all
27
+ * (`EAFNOSUPPORT`, as in many containers) `::` falls back to IPv4's
28
+ * `0.0.0.0`.
29
+ *
30
+ * Linux and Darwin differ in five places: `SOCK_NONBLOCK | SOCK_CLOEXEC` and
31
+ * `accept4` against a `socket` or `accept` followed by `fcntl`, `MSG_NOSIGNAL`
32
+ * against the `SO_NOSIGPIPE` socket option (a write to a gone peer is -32 on
33
+ * both, never SIGPIPE), the errno numbers, UDP's segmentation offload
34
+ * (`UDP_SEGMENT`, `UDP_GRO`) and ECN marks, which Linux has and for which
35
+ * Darwin answers -95, and the readiness loop, which is epoll on Linux and
36
+ * kqueue on Darwin. The Darwin branch is compiled by CI's Darwin bootstrap
37
+ * rows and run by nothing. A WASI build has none of this (the checker refuses
38
+ * every `nish:net` call under a wasm target), so there the file is empty.
39
+ */
40
+ #if defined(__wasi__) || defined(__wasm__)
41
+ /* ISO C wants at least one declaration in a translation unit. */
42
+ typedef int nish_net_unused;
43
+ #else
44
+ #if defined(__linux__)
45
+ /* glibc declares `accept4` only with this feature macro. */
46
+ #define _GNU_SOURCE
47
+ #endif
48
+ #include <arpa/inet.h>
49
+ #include <errno.h>
50
+ #include <fcntl.h>
51
+ #include <netinet/in.h>
52
+ #include <stdint.h>
53
+ #include <string.h>
54
+ #include <sys/socket.h>
55
+ #include <unistd.h>
56
+ #if defined(__linux__)
57
+ #include <netinet/udp.h>
58
+ #include <sys/epoll.h>
59
+ /* glibc before 2.29 names neither; the numbers are the kernel's ABI. */
60
+ #ifndef UDP_SEGMENT
61
+ #define UDP_SEGMENT 103
62
+ #endif
63
+ #ifndef UDP_GRO
64
+ #define UDP_GRO 104
65
+ #endif
66
+ #else
67
+ #include <sys/event.h>
68
+ #include <sys/time.h>
69
+ #endif
70
+
71
+ #include "nish.h"
72
+
73
+ /* The bytes an address takes in a caller's `u8[]`. */
74
+ #define NISH_ADDR_BYTES 18
75
+
76
+ /* One buffer big enough for either family, read as whichever it is. */
77
+ typedef union nish_sockaddr {
78
+ struct sockaddr sa;
79
+ struct sockaddr_in v4;
80
+ struct sockaddr_in6 v6;
81
+ } nish_sockaddr;
82
+
83
+ /* The first twelve bytes of an IPv4-mapped IPv6 address. */
84
+ static const unsigned char nish_mapped_prefix[12] = {0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0xff, 0xff};
85
+
86
+ /* `errno` as the language answers it: negative, and in Linux's numbering for
87
+ the codes a loop branches on. */
88
+ static int32_t nish_net_err(int e) {
89
+ #if !defined(__linux__)
90
+ if (e == EAGAIN) return -11;
91
+ if (e == EOPNOTSUPP || e == ENOTSUP) return -95;
92
+ if (e == ECONNRESET) return -104;
93
+ if (e == EADDRINUSE) return -98;
94
+ #endif
95
+ return -e;
96
+ }
97
+
98
+ /* The same for a call that has just failed. */
99
+ static int32_t nish_net_fail(void) { return nish_net_err(errno); }
100
+
101
+ /* A numeric host into the 16 address bytes: an IPv4 dotted quad as its mapped
102
+ form, or an IPv6 literal. No name resolution. 1 when it parsed. A string
103
+ with a NUL inside is not a literal, however the part before it reads. */
104
+ static int nish_net_parse(const nish_str *s, unsigned char a[16]) {
105
+ const char *host = s->data;
106
+ if (strlen(host) != s->len) return 0;
107
+ if (inet_pton(AF_INET, host, a + 12) == 1) {
108
+ memcpy(a, nish_mapped_prefix, 12);
109
+ return 1;
110
+ }
111
+ return inet_pton(AF_INET6, host, a) == 1;
112
+ }
113
+
114
+ /* `a` and `port` as a socket address of the family `v4` names. */
115
+ static socklen_t nish_net_sockaddr(nish_sockaddr *s, const unsigned char a[16], int32_t port, int v4) {
116
+ memset(s, 0, sizeof *s);
117
+ if (v4) {
118
+ s->v4.sin_family = AF_INET;
119
+ s->v4.sin_port = htons((uint16_t)port);
120
+ memcpy(&s->v4.sin_addr, a + 12, 4);
121
+ return sizeof s->v4;
122
+ }
123
+ s->v6.sin6_family = AF_INET6;
124
+ s->v6.sin6_port = htons((uint16_t)port);
125
+ memcpy(&s->v6.sin6_addr, a, 16);
126
+ return sizeof s->v6;
127
+ }
128
+
129
+ /* A socket that is non-blocking and close-on-exec from the moment it exists
130
+ on Linux, and from the `fcntl` after it on Darwin, where a child another
131
+ thread spawns in between can inherit it. Darwin's `SO_NOSIGPIPE` is set
132
+ here too, because it is a property of the socket rather than of the call. */
133
+ #if !defined(__linux__)
134
+ static int nish_net_setup(int fd) {
135
+ int one = 1;
136
+ if (fd >= 0) {
137
+ fcntl(fd, F_SETFD, FD_CLOEXEC);
138
+ fcntl(fd, F_SETFL, O_NONBLOCK);
139
+ setsockopt(fd, SOL_SOCKET, SO_NOSIGPIPE, &one, sizeof one);
140
+ }
141
+ return fd;
142
+ }
143
+ #define NISH_NOSIGNAL 0
144
+ #else
145
+ #define NISH_NOSIGNAL MSG_NOSIGNAL
146
+ #endif
147
+
148
+ static int nish_net_socket(int family, int type) {
149
+ #if defined(__linux__)
150
+ return socket(family, type | SOCK_NONBLOCK | SOCK_CLOEXEC, 0);
151
+ #else
152
+ return nish_net_setup(socket(family, type, 0));
153
+ #endif
154
+ }
155
+
156
+ /* `netAddress(out, host, port)`: the 18-byte form of a numeric host and a
157
+ port, or -22 for a host that is not a literal, a port outside 0..65535 or
158
+ an `out` shorter than 18 bytes. */
159
+ int32_t nish_net_address(nish_array *out, const nish_str *host, int32_t port) {
160
+ unsigned char a[16];
161
+ if (out->len < NISH_ADDR_BYTES || (uint32_t)port > 65535 || !nish_net_parse(host, a)) return -22;
162
+ memcpy(out->data, a, 16);
163
+ out->data[16] = (char)(port >> 8);
164
+ out->data[17] = (char)port;
165
+ return 0;
166
+ }
167
+
168
+ /* `netLocalPort(fd)`: the port the socket is bound to, which is how a
169
+ program that listened on port 0 learns the one the kernel chose. */
170
+ int32_t nish_net_local_port(int32_t fd) {
171
+ nish_sockaddr s;
172
+ socklen_t n = sizeof s;
173
+ if (getsockname(fd, &s.sa, &n) != 0) return nish_net_fail();
174
+ return ntohs(s.sa.sa_family == AF_INET ? s.v4.sin_port : s.v6.sin6_port);
175
+ }
176
+
177
+ /* A socket of `type` bound to a numeric host and port, or the failure. An
178
+ IPv4 host is an `AF_INET` socket; any other is `AF_INET6` with
179
+ `IPV6_V6ONLY` off, so `::` hears both families. A stream gets
180
+ `SO_REUSEADDR` and `arg` its backlog; a datagram socket's `arg` is
181
+ `udpBind`'s flags. Every option is set before the bind, which is where
182
+ `SO_REUSEPORT` has to be, and where `UDP_GRO` has to be for the first
183
+ datagram to be coalesced. */
184
+ static int32_t nish_net_bound(const nish_str *host, int32_t port, int type, int32_t arg) {
185
+ unsigned char a[16];
186
+ if ((uint32_t)port > 65535 || !nish_net_parse(host, a)) return -22;
187
+ int v4 = memcmp(a, nish_mapped_prefix, 12) == 0;
188
+ int fd = nish_net_socket(v4 ? AF_INET : AF_INET6, type);
189
+ /* `::` on a kernel without IPv6: the same wildcard in the family there is.
190
+ All sixteen bytes are zero, so the last four are already `0.0.0.0`. */
191
+ static const unsigned char any[16];
192
+ if (fd < 0 && errno == EAFNOSUPPORT && memcmp(a, any, 16) == 0) {
193
+ v4 = 1;
194
+ fd = nish_net_socket(AF_INET, type);
195
+ }
196
+ if (fd < 0) return nish_net_fail();
197
+ int one = 1;
198
+ int zero = 0;
199
+ /* Both families at once on an IPv6 socket; on an IPv4 one this fails with
200
+ `ENOPROTOOPT` and changes nothing, which is cheaper than asking. */
201
+ setsockopt(fd, IPPROTO_IPV6, IPV6_V6ONLY, &zero, sizeof zero);
202
+ int bad = 0;
203
+ if (type == SOCK_STREAM) {
204
+ setsockopt(fd, SOL_SOCKET, SO_REUSEADDR, &one, sizeof one);
205
+ } else {
206
+ if (arg & 1) bad = setsockopt(fd, SOL_SOCKET, SO_REUSEPORT, &one, sizeof one);
207
+ #if defined(__linux__)
208
+ if (arg & 2) bad |= setsockopt(fd, IPPROTO_UDP, UDP_GRO, &one, sizeof one);
209
+ /* ECN is always read. An IPv6 socket needs both: a datagram from an IPv4
210
+ peer reports its TOS byte, one from an IPv6 peer its traffic class. */
211
+ setsockopt(fd, IPPROTO_IP, IP_RECVTOS, &one, sizeof one);
212
+ setsockopt(fd, IPPROTO_IPV6, IPV6_RECVTCLASS, &one, sizeof one);
213
+ #endif
214
+ }
215
+ nish_sockaddr s;
216
+ socklen_t n = nish_net_sockaddr(&s, a, port, v4);
217
+ if (bad != 0 || bind(fd, &s.sa, n) != 0 || (type == SOCK_STREAM && listen(fd, arg) != 0)) {
218
+ int e = errno;
219
+ close(fd);
220
+ return nish_net_err(e);
221
+ }
222
+ return fd;
223
+ }
224
+
225
+ /* `tcpListen(host, port, backlog)`: a listening socket with `SO_REUSEADDR`,
226
+ so a restarted server rebinds a port whose old connections are still in
227
+ TIME_WAIT. */
228
+ int32_t nish_tcp_listen(const nish_str *host, int32_t port, int32_t backlog) {
229
+ return nish_net_bound(host, port, SOCK_STREAM, backlog);
230
+ }
231
+
232
+ /* A socket address into the caller's 18-byte form. An IPv4 peer is written
233
+ mapped, and the port is already in network order, which is the form's
234
+ big-endian. */
235
+ static void nish_net_put(nish_array *out, const nish_sockaddr *s) {
236
+ unsigned char *p = (unsigned char *)out->data;
237
+ uint16_t port = s->v6.sin6_port;
238
+ if (s->sa.sa_family == AF_INET) {
239
+ memcpy(p, nish_mapped_prefix, 12);
240
+ memcpy(p + 12, &s->v4.sin_addr, 4);
241
+ port = s->v4.sin_port;
242
+ } else {
243
+ memcpy(p, &s->v6.sin6_addr, 16);
244
+ }
245
+ memcpy(p + 16, &port, 2);
246
+ }
247
+
248
+ /* `tcpAccept(fd, peer)`: the next connection as a descriptor of its own,
249
+ non-blocking and close-on-exec, with the peer's address written into
250
+ `peer`; -11 when none is waiting. A `peer` shorter than 18 bytes is -22
251
+ before anything is accepted, so no connection is taken and lost. */
252
+ int32_t nish_tcp_accept(int32_t fd, nish_array *peer) {
253
+ if (peer->len < NISH_ADDR_BYTES) return -22;
254
+ nish_sockaddr s;
255
+ socklen_t n = sizeof s;
256
+ #if defined(__linux__)
257
+ int c = accept4(fd, &s.sa, &n, SOCK_NONBLOCK | SOCK_CLOEXEC);
258
+ #else
259
+ int c = nish_net_setup(accept(fd, &s.sa, &n));
260
+ #endif
261
+ if (c < 0) return nish_net_fail();
262
+ nish_net_put(peer, &s);
263
+ return c;
264
+ }
265
+
266
+ /* `netRead(fd, buf, off, len)`: at most `len` bytes into `buf` from `off`.
267
+ The compiled call has already checked `0 <= off <= off + len <=
268
+ buf.length`, unless the program was built with `--unchecked-indexing`. */
269
+ int32_t nish_net_read(int32_t fd, nish_array *buf, int64_t off, int64_t len) {
270
+ ssize_t n = recv(fd, buf->data + off, (size_t)len, 0);
271
+ return n < 0 ? nish_net_fail() : (int32_t)n;
272
+ }
273
+
274
+ /* `netWrite(fd, buf, off, len)`: at most `len` bytes of `buf` from `off`,
275
+ with a peer that has gone answering -32 rather than raising SIGPIPE. */
276
+ int32_t nish_net_write(int32_t fd, const nish_array *buf, int64_t off, int64_t len) {
277
+ ssize_t n = send(fd, buf->data + off, (size_t)len, NISH_NOSIGNAL);
278
+ return n < 0 ? nish_net_fail() : (int32_t)n;
279
+ }
280
+
281
+ /* `netShutdown(fd, how)`: 0 the read side, 1 the write side, 2 both, which
282
+ are `SHUT_RD`, `SHUT_WR` and `SHUT_RDWR` on both platforms. */
283
+ int32_t nish_net_shutdown(int32_t fd, int32_t how) {
284
+ if ((uint32_t)how > 2) return -22;
285
+ return shutdown(fd, how) != 0 ? nish_net_fail() : 0;
286
+ }
287
+
288
+ /* `netClose(fd)`. The descriptor is gone whatever this answers, as `close`
289
+ promises on Linux, so a failure is information rather than a retry. */
290
+ int32_t nish_net_close(int32_t fd) { return close(fd) != 0 ? nish_net_fail() : 0; }
291
+
292
+ /* `udpBind(host, port, flags)`: a datagram socket, non-blocking and
293
+ close-on-exec, bound like `tcpListen`'s, with `SO_REUSEPORT` for flag 1
294
+ and `UDP_GRO` for flag 2; any other bit is -22. */
295
+ int32_t nish_udp_bind(const nish_str *host, int32_t port, int32_t flags) {
296
+ if ((uint32_t)flags > 3) return -22;
297
+ #if !defined(__linux__)
298
+ if (flags & 2) return -95;
299
+ #endif
300
+ return nish_net_bound(host, port, SOCK_DGRAM, flags);
301
+ }
302
+
303
+ /* Room for the two control messages either direction carries: a segment size
304
+ (a `uint16_t` out, an `int` in) and a TOS byte or traffic class (an `int`
305
+ either way). A union, so the buffer is aligned as a `cmsghdr` must be. */
306
+ typedef union nish_cmsgs {
307
+ struct cmsghdr h;
308
+ char b[CMSG_SPACE(sizeof(int)) * 2];
309
+ } nish_cmsgs;
310
+
311
+ /* `udpSendTo(fd, buf, off, len, to, segment, ecn)`: `buf[off, off + len)` to
312
+ the address in `to`, in one `sendmsg`. A `segment` above 0 is
313
+ `UDP_SEGMENT`: the kernel cuts the payload into datagrams of that many
314
+ bytes, the last one shorter, which is one system call for many datagrams.
315
+ `ecn` is the two ECN bits of the IP header (0 not ECN-capable, 1 ECT(1),
316
+ 2 ECT(0), 3 CE). A `to` shorter than 18 bytes, a `segment` outside 0..65535
317
+ or an `ecn` outside 0..3 is -22. */
318
+ int32_t nish_udp_send_to(int32_t fd, const nish_array *buf, int64_t off, int64_t len, const nish_array *to,
319
+ int32_t segment, int32_t ecn) {
320
+ if (to->len < NISH_ADDR_BYTES || (uint32_t)segment > 65535 || (uint32_t)ecn > 3) return -22;
321
+ const unsigned char *p = (const unsigned char *)to->data;
322
+ int v4 = memcmp(p, nish_mapped_prefix, 12) == 0;
323
+ #if !defined(__linux__)
324
+ if (segment > 0 || ecn > 0) return -95;
325
+ /* Darwin's IPv6 socket takes an IPv4 peer only in its mapped form. */
326
+ nish_sockaddr self;
327
+ socklen_t k = sizeof self;
328
+ if (v4 && getsockname(fd, &self.sa, &k) == 0 && self.sa.sa_family == AF_INET6) v4 = 0;
329
+ #endif
330
+ /* Linux: an IPv4 peer as an `AF_INET` address, which an IPv4 socket needs
331
+ and a dual-stack IPv6 one accepts, sending it as IPv4. */
332
+ nish_sockaddr s;
333
+ socklen_t n = nish_net_sockaddr(&s, p, (p[16] << 8) | p[17], v4);
334
+ struct iovec iov = {buf->data + off, (size_t)len};
335
+ struct msghdr m = {.msg_name = &s, .msg_namelen = n, .msg_iov = &iov, .msg_iovlen = 1};
336
+ #if defined(__linux__)
337
+ /* Zeroed, so the padding `CMSG_SPACE` adds after each message is too. */
338
+ nish_cmsgs c;
339
+ memset(&c, 0, sizeof c);
340
+ size_t used = 0;
341
+ struct cmsghdr *h = &c.h;
342
+ if (segment > 0) {
343
+ uint16_t size = (uint16_t)segment;
344
+ h->cmsg_level = SOL_UDP;
345
+ h->cmsg_type = UDP_SEGMENT;
346
+ h->cmsg_len = CMSG_LEN(sizeof size);
347
+ memcpy(CMSG_DATA(h), &size, sizeof size);
348
+ used = CMSG_SPACE(sizeof size);
349
+ h = (struct cmsghdr *)(c.b + used);
350
+ }
351
+ if (ecn > 0) {
352
+ /* The level follows the header the datagram leaves with: IPv4's TOS for
353
+ an IPv4 peer, whatever the socket's family, IPv6's traffic class
354
+ otherwise. */
355
+ h->cmsg_level = v4 ? IPPROTO_IP : IPPROTO_IPV6;
356
+ h->cmsg_type = v4 ? IP_TOS : IPV6_TCLASS;
357
+ h->cmsg_len = CMSG_LEN(sizeof ecn);
358
+ memcpy(CMSG_DATA(h), &ecn, sizeof ecn);
359
+ used += CMSG_SPACE(sizeof ecn);
360
+ }
361
+ if (used > 0) {
362
+ m.msg_control = c.b;
363
+ m.msg_controllen = used;
364
+ }
365
+ #endif
366
+ ssize_t sent = sendmsg(fd, &m, 0);
367
+ return sent < 0 ? nish_net_fail() : (int32_t)sent;
368
+ }
369
+
370
+ /* `udpRecvFrom(fd, buf, off, len, from, meta)`: the next datagram into
371
+ `buf[off, off + len)` in one `recvmsg`, its sender's address into `from`,
372
+ and into `meta` the segment size (`meta[0]`: the size of each datagram
373
+ `UDP_GRO` coalesced into this one, 0 when it coalesced none) and the ECN
374
+ bits it arrived with (`meta[1]`). A datagram longer than `len` is cut to
375
+ it. -11 when none is waiting; a `from` shorter than 18 bytes or a `meta`
376
+ shorter than 2 is -22 before anything is taken. */
377
+ int32_t nish_udp_recv_from(int32_t fd, nish_array *buf, int64_t off, int64_t len, nish_array *from,
378
+ nish_array *meta) {
379
+ if (from->len < NISH_ADDR_BYTES || meta->len < 2) return -22;
380
+ nish_sockaddr s;
381
+ struct iovec iov = {buf->data + off, (size_t)len};
382
+ nish_cmsgs c;
383
+ struct msghdr m = {
384
+ .msg_name = &s, .msg_namelen = sizeof s, .msg_iov = &iov, .msg_iovlen = 1, .msg_control = c.b, .msg_controllen = sizeof c};
385
+ ssize_t got = recvmsg(fd, &m, 0);
386
+ if (got < 0) return nish_net_fail();
387
+ /* Darwin sets neither option, so both stay 0 there. */
388
+ int32_t info[2] = {0, 0};
389
+ #if defined(__linux__)
390
+ for (struct cmsghdr *h = CMSG_FIRSTHDR(&m); h != NULL; h = CMSG_NXTHDR(&m, h)) {
391
+ /* `UDP_GRO` and `IPV6_TCLASS` carry an `int`, IPv4's `IP_TOS` one byte. */
392
+ const unsigned char *d = CMSG_DATA(h);
393
+ int v = *d;
394
+ if (h->cmsg_len >= CMSG_LEN(sizeof v)) memcpy(&v, d, sizeof v);
395
+ if (h->cmsg_level == SOL_UDP && h->cmsg_type == UDP_GRO) info[0] = v;
396
+ if ((h->cmsg_level == IPPROTO_IP && h->cmsg_type == IP_TOS) ||
397
+ (h->cmsg_level == IPPROTO_IPV6 && h->cmsg_type == IPV6_TCLASS))
398
+ info[1] = v & 3;
399
+ }
400
+ #endif
401
+ memcpy(meta->data, info, sizeof info);
402
+ nish_net_put(from, &s);
403
+ return (int32_t)got;
404
+ }
405
+
406
+ /* The readiness loop. Level-triggered: a descriptor that is still ready is
407
+ reported again by the next wait, so a program that handles one event of
408
+ many loses none. The events a program asks for and is told of are bits: 1
409
+ readable, 2 writable, and, told only, 4 for a hang-up or an error. A token
410
+ is the program's own `i32` for the descriptor, handed back as it was
411
+ given. Nothing allocates: the kernel's records for one wait are on the
412
+ stack, at most `NISH_POLL_CHUNK` of them, and whatever did not fit is still
413
+ ready at the next wait. */
414
+ #define NISH_POLL_CHUNK 64
415
+ #if defined(__linux__)
416
+ #define NISH_POLL_ADD EPOLL_CTL_ADD
417
+ #define NISH_POLL_MODIFY EPOLL_CTL_MOD
418
+ #define NISH_POLL_REMOVE EPOLL_CTL_DEL
419
+ #else
420
+ #define NISH_POLL_ADD 0
421
+ #define NISH_POLL_MODIFY 1
422
+ #define NISH_POLL_REMOVE 2
423
+ #endif
424
+
425
+ /* `pollCreate()`: a loop descriptor, close-on-exec from the moment it exists
426
+ on Linux, where `epoll_create1(EPOLL_CLOEXEC)` sets it atomically, and from
427
+ the `fcntl` after `kqueue()` on Darwin, where a child another thread spawns
428
+ in between can inherit it, as `nish_net_setup`'s sockets can. */
429
+ int32_t nish_poll_create(void) {
430
+ #if defined(__linux__)
431
+ int fd = epoll_create1(EPOLL_CLOEXEC);
432
+ #else
433
+ int fd = kqueue();
434
+ if (fd >= 0) fcntl(fd, F_SETFD, FD_CLOEXEC);
435
+ #endif
436
+ return fd < 0 ? nish_net_fail() : fd;
437
+ }
438
+
439
+ #if !defined(__linux__)
440
+ /* One kqueue change, answered as an errno or 0. `EV_RECEIPT` returns each
441
+ change's own result rather than failing the whole call, and takes no event
442
+ off the queue. */
443
+ static int nish_poll_change(int32_t loop, int32_t fd, int filter, int flags, int32_t token) {
444
+ struct kevent k;
445
+ EV_SET(&k, fd, filter, flags | EV_RECEIPT, 0, 0, (void *)(intptr_t)token);
446
+ if (kevent(loop, &k, 1, &k, 1, NULL) < 0) return errno;
447
+ return (k.flags & EV_ERROR) ? (int)k.data : 0;
448
+ }
449
+ #endif
450
+
451
+ /* Add, modify or remove `fd` in `loop`. epoll does each in one call, with
452
+ the token in `data.u32`. kqueue has a filter per direction rather than a
453
+ set of events, so a change is the filters deleted and then the ones asked
454
+ for added, each carrying the token in `udata`. Deleting a filter that was
455
+ never added is not an error, except to a removal that finds neither, which
456
+ answers epoll's ENOENT; a modification adds what it asks for either way, so
457
+ a descriptor added watching nothing, which has no filter, can be modified. */
458
+ static int32_t nish_poll_ctl(int32_t loop, int op, int32_t fd, int32_t events, int32_t token) {
459
+ if ((uint32_t)events > 3) return -22;
460
+ #if defined(__linux__)
461
+ struct epoll_event e;
462
+ memset(&e, 0, sizeof e);
463
+ e.events = ((events & 1) ? EPOLLIN : 0) | ((events & 2) ? EPOLLOUT : 0);
464
+ e.data.u32 = (uint32_t)token;
465
+ return epoll_ctl(loop, op, fd, &e) != 0 ? nish_net_fail() : 0;
466
+ #else
467
+ int e = 0;
468
+ if (op != NISH_POLL_ADD) {
469
+ int r = nish_poll_change(loop, fd, EVFILT_READ, EV_DELETE, 0);
470
+ int w = nish_poll_change(loop, fd, EVFILT_WRITE, EV_DELETE, 0);
471
+ if (op == NISH_POLL_REMOVE) return r != 0 && w != 0 ? nish_net_err(r) : 0;
472
+ }
473
+ if (events & 1) e = nish_poll_change(loop, fd, EVFILT_READ, EV_ADD | EV_ENABLE, token);
474
+ if (e == 0 && (events & 2)) e = nish_poll_change(loop, fd, EVFILT_WRITE, EV_ADD | EV_ENABLE, token);
475
+ return e != 0 ? nish_net_err(e) : 0;
476
+ #endif
477
+ }
478
+
479
+ /* `pollAdd(loop, fd, events, token)`: watch `fd` for `events` under `token`. */
480
+ int32_t nish_poll_add(int32_t loop, int32_t fd, int32_t events, int32_t token) {
481
+ return nish_poll_ctl(loop, NISH_POLL_ADD, fd, events, token);
482
+ }
483
+
484
+ /* `pollModify(loop, fd, events, token)`: the same descriptor, new events or a new token. */
485
+ int32_t nish_poll_modify(int32_t loop, int32_t fd, int32_t events, int32_t token) {
486
+ return nish_poll_ctl(loop, NISH_POLL_MODIFY, fd, events, token);
487
+ }
488
+
489
+ /* `pollRemove(loop, fd)`: stop watching `fd`. Closing it does the same. */
490
+ int32_t nish_poll_remove(int32_t loop, int32_t fd) { return nish_poll_ctl(loop, NISH_POLL_REMOVE, fd, 0, 0); }
491
+
492
+ /* `pollWait(loop, ready, timeoutMs)`: wait until a descriptor is ready or
493
+ `timeout_ms` milliseconds pass, a negative one never, and write a token and
494
+ its events into `ready` per ready descriptor, `ready[2k]` and
495
+ `ready[2k + 1]`. The count is at most `ready.length / 2` and at most
496
+ `NISH_POLL_CHUNK`. A signal that interrupts the wait answers 0, as a
497
+ timeout does, rather than waiting again past the deadline; the program's
498
+ loop comes round and, if it watches `signalFd()`, finds the signal ready.
499
+ On Darwin a descriptor both readable and writable is two pairs with the
500
+ same token, one per filter. A `ready` shorter than 2 is -22. */
501
+ int32_t nish_poll_wait(int32_t loop, nish_array *ready, int32_t timeout_ms) {
502
+ uint64_t room = ready->len / 2;
503
+ if (room == 0) return -22;
504
+ int max = room < NISH_POLL_CHUNK ? (int)room : NISH_POLL_CHUNK;
505
+ int32_t *out = (int32_t *)ready->data;
506
+ #if defined(__linux__)
507
+ struct epoll_event e[NISH_POLL_CHUNK];
508
+ /* epoll waits forever for any negative timeout, as the language does. */
509
+ int n = epoll_wait(loop, e, max, timeout_ms);
510
+ #else
511
+ struct kevent e[NISH_POLL_CHUNK];
512
+ struct timespec t = {timeout_ms / 1000, (long)(timeout_ms % 1000) * 1000000};
513
+ int n = kevent(loop, NULL, 0, e, max, timeout_ms < 0 ? NULL : &t);
514
+ #endif
515
+ if (n < 0) return errno == EINTR ? 0 : nish_net_fail();
516
+ for (int i = 0; i < n; i++) {
517
+ #if defined(__linux__)
518
+ uint32_t got = e[i].events;
519
+ out[2 * i] = (int32_t)e[i].data.u32;
520
+ out[2 * i + 1] = ((got & EPOLLIN) ? 1 : 0) | ((got & EPOLLOUT) ? 2 : 0) | ((got & (EPOLLHUP | EPOLLERR)) ? 4 : 0);
521
+ #else
522
+ out[2 * i] = (int32_t)(intptr_t)e[i].udata;
523
+ out[2 * i + 1] = (e[i].filter == EVFILT_READ ? 1 : 2) | ((e[i].flags & (EV_EOF | EV_ERROR)) ? 4 : 0);
524
+ #endif
525
+ }
526
+ return n;
527
+ }
528
+ #endif
package/runtime/shim.mjs CHANGED
@@ -740,6 +740,21 @@ export function readSignal() {
740
740
  return signalFd();
741
741
  }
742
742
 
743
+ /**
744
+ * The `nish:net` functions (WP34 N5) have no faithful reading under Node
745
+ * either, for `signalFd`'s reason: Node's sockets are asynchronous only, and a
746
+ * socket becomes readable, writable or connected only to the event loop, which
747
+ * a program that owns its loop and spins or blocks in it never returns to. So
748
+ * each throws, naming itself, rather than answer -11 forever; `nish.mjs`
749
+ * installs one per name.
750
+ * docs/wp33-round-trip.md is the note that asks for one stated answer.
751
+ */
752
+ export function noNetReading(name) {
753
+ throw new Error(
754
+ `\`${name}\` has no synchronous reading under Node: a socket is ready only to the event loop, which a program that owns its loop never returns to (docs/wp33-round-trip.md)`
755
+ );
756
+ }
757
+
743
758
  /** `process.argv`: index 0 is the program (the script here, the executable natively), then the arguments. */
744
759
  export function argv() {
745
760
  return process.argv.slice(1);
package/scripts/build.sh CHANGED
@@ -3,12 +3,13 @@
3
3
  #
4
4
  # scripts/build.sh <module.ll> [more .ll/.c files...] -o <out> [--profile debug|speed|size|wasm]
5
5
  #
6
- # The C runtime is four translation units and is named as one: an input
6
+ # The C runtime is five translation units and is named as one: an input
7
7
  # <dir>/runtime.c also compiles <dir>/runtime-os.c, the half that wraps the
8
8
  # system calls (files, directories, subprocesses, the environment, the clock),
9
9
  # <dir>/runtime-parallel.c, the half that divides a range of work across
10
- # threads, and <dir>/runtime-host.c, the wall clock, entropy, file times and
11
- # signals. Each of those files says why they are compiled and measured apart.
10
+ # threads, <dir>/runtime-host.c, the wall clock, entropy, file times and
11
+ # signals, and <dir>/runtime-net.c, the sockets of `nish:net`. Each of those
12
+ # files says why they are compiled and measured apart.
12
13
  #
13
14
  # Profiles:
14
15
  # debug clang defaults: no optimisation, symbols kept. The "before" number.
@@ -84,9 +85,9 @@ done
84
85
  [ ${#inputs[@]} -gt 0 ] || { echo "error: no input files" >&2; exit 2; }
85
86
  [ -n "$out" ] || { echo "error: -o <out> is required" >&2; exit 2; }
86
87
 
87
- # The runtime is four translation units, and a caller names one: whoever passes
88
- # <dir>/runtime.c gets <dir>/runtime-os.c, <dir>/runtime-parallel.c and
89
- # <dir>/runtime-host.c compiled beside it. They were one file until the operating-system half was split out
88
+ # The runtime is five translation units, and a caller names one: whoever passes
89
+ # <dir>/runtime.c gets <dir>/runtime-os.c, <dir>/runtime-parallel.c,
90
+ # <dir>/runtime-host.c and <dir>/runtime-net.c compiled beside it. They were one file until the operating-system half was split out
90
91
  # for its own size budget, and the parallel half followed for the same reason
91
92
  # (each file's header comment says why), and a link line is where those splits
92
93
  # would otherwise leak: `nish --link` builds its command line in
@@ -98,7 +99,7 @@ done
98
99
  for i in ${inputs[@]+"${inputs[@]}"}; do
99
100
  case "$i" in
100
101
  */runtime.c|runtime.c)
101
- for half in runtime-os.c runtime-parallel.c runtime-host.c; do
102
+ for half in runtime-os.c runtime-parallel.c runtime-host.c runtime-net.c; do
102
103
  side="${i%runtime.c}$half"
103
104
  have=0
104
105
  for j in "${inputs[@]}"; do