kino 0.2.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,37 @@
1
+ //! How many CPUs this process may actually use: the default worker count.
2
+ //!
3
+ //! `Etc.nprocessors` honours the affinity mask but not a cgroup CPU quota,
4
+ //! so a container limited to two CPUs on a 64-core host would spawn 64
5
+ //! workers. The standard library's `available_parallelism` reads both the
6
+ //! mask and the cgroup v1/v2 quota on Linux (rounding a fractional quota
7
+ //! up), which is also what tokio sizes its own pool by.
8
+
9
+ use magnus::{Error, Ruby};
10
+
11
+ /// The usable CPU count, never below one (a quota of 0.5 CPU still needs
12
+ /// a worker; an unreadable count falls back to one rather than failing
13
+ /// boot).
14
+ pub fn available_parallelism(_ruby: &Ruby) -> Result<usize, Error> {
15
+ Ok(count())
16
+ }
17
+
18
+ fn count() -> usize {
19
+ std::thread::available_parallelism().map_or(1, |n| n.get())
20
+ }
21
+
22
+ #[cfg(test)]
23
+ mod tests {
24
+ use super::count;
25
+
26
+ #[test]
27
+ fn reports_at_least_one_cpu() {
28
+ assert!(count() >= 1);
29
+ }
30
+
31
+ #[test]
32
+ fn never_exceeds_what_the_os_reports_as_online() {
33
+ // A quota can only lower the count below the online CPUs; never raise it.
34
+ let online = std::thread::available_parallelism().map_or(1, |n| n.get());
35
+ assert!(count() <= online);
36
+ }
37
+ }
data/ext/kino/src/lib.rs CHANGED
@@ -4,9 +4,15 @@
4
4
  #[global_allocator]
5
5
  static GLOBAL: mimalloc::MiMalloc = mimalloc::MiMalloc;
6
6
 
7
+ mod access_log;
8
+ mod control;
9
+ mod cpus;
7
10
  mod env_strings;
8
11
  mod gvl;
12
+ mod listen;
13
+ mod log;
9
14
  mod logsink;
15
+ mod mono;
10
16
  mod pin;
11
17
  mod queue;
12
18
  mod registry;
@@ -40,6 +46,16 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
40
46
  native.define_singleton_method("close_queue", function!(server::close_queue, 1))?;
41
47
  native.define_singleton_method("queue_stats", function!(server::queue_stats, 1))?;
42
48
  native.define_singleton_method("server_stats", function!(server::server_stats, 1))?;
49
+ native.define_singleton_method("worker_stats", function!(server::worker_stats, 1))?;
50
+ native.define_singleton_method("queue_time", function!(server::queue_time, 1))?;
51
+ native.define_singleton_method("quarantine_slot", function!(server::quarantine_slot, 2))?;
52
+ native.define_singleton_method(
53
+ "record_quarantine_replacement",
54
+ function!(server::record_quarantine_replacement, 1),
55
+ )?;
56
+ native.define_singleton_method("control_ready", function!(server::control_ready, 1))?;
57
+ native.define_singleton_method("record_respawn", function!(server::record_respawn, 1))?;
58
+ native.define_singleton_method("control_stop", function!(control::control_stop, 1))?;
43
59
  native.define_singleton_method("abort_inflight", function!(server::abort_inflight, 2))?;
44
60
  native.define_singleton_method(
45
61
  "abort_all_inflight",
@@ -50,8 +66,12 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
50
66
  function!(server::interrupt_all_workers, 1),
51
67
  )?;
52
68
  native.define_singleton_method("shutdown_runtime", function!(server::shutdown_runtime, 2))?;
53
- native.define_singleton_method("log_error", function!(server::log_error, 1))?;
69
+ native.define_singleton_method("log_line", function!(log::log_line, 3))?;
54
70
  native.define_singleton_method("sleep_chunk", function!(timer::sleep_chunk, 1))?;
71
+ native.define_singleton_method(
72
+ "available_parallelism",
73
+ function!(cpus::available_parallelism, 0),
74
+ )?;
55
75
  native.define_singleton_method("log_device_open", function!(logsink::device_open, 1))?;
56
76
  native.define_singleton_method("log_device_write", function!(logsink::device_write, 2))?;
57
77
  native.define_singleton_method("log_device_close", function!(logsink::device_close, 1))?;
@@ -75,6 +95,7 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
75
95
  request.define_method("write_chunk", method!(Request::write_chunk, 1))?;
76
96
  request.define_method("finish", method!(Request::finish, 0))?;
77
97
  request.define_method("abort", method!(Request::abort, 0))?;
98
+ request.define_method("timing", method!(Request::set_timing, 2))?;
78
99
 
79
100
  // Force-resolve the TypedData class cache on the main ractor: magnus
80
101
  // resolves it lazily on first wrap, and a racy first resolution from two
@@ -0,0 +1,134 @@
1
+ //! Listening sockets for the main server and the control plane: a TCP
2
+ //! `host:port`, or a `unix://path` domain socket (the usual shape behind
3
+ //! nginx). Binding is synchronous so an address conflict surfaces at boot.
4
+
5
+ use std::io;
6
+ use std::os::unix::net::UnixListener;
7
+ use std::path::{Path, PathBuf};
8
+
9
+ /// The bind scheme that selects a unix domain socket; anything else is a
10
+ /// TCP host.
11
+ pub const UNIX_SCHEME: &str = "unix://";
12
+
13
+ /// The socket path of a `unix://` bind, or None for a TCP host.
14
+ pub fn unix_path(bind: &str) -> Option<&Path> {
15
+ bind.strip_prefix(UNIX_SCHEME).map(Path::new)
16
+ }
17
+
18
+ /// A bound, non-blocking listener of either kind.
19
+ pub enum Listener {
20
+ Tcp(std::net::TcpListener),
21
+ Unix(UnixListener, PathBuf),
22
+ }
23
+
24
+ impl Listener {
25
+ /// Bind `bind:port` (TCP; a hostname resolves to its addresses and the
26
+ /// first that binds wins) or `unix://path`.
27
+ pub fn bind(bind: &str, port: u16) -> io::Result<Listener> {
28
+ match unix_path(bind) {
29
+ Some(path) => Ok(Listener::Unix(bind_unix(path)?, path.to_path_buf())),
30
+ None => {
31
+ let listener = std::net::TcpListener::bind((bind, port))?;
32
+ listener.set_nonblocking(true)?;
33
+ Ok(Listener::Tcp(listener))
34
+ }
35
+ }
36
+ }
37
+
38
+ /// The bound TCP port; 0 for a unix socket, which has none.
39
+ pub fn port(&self) -> io::Result<u16> {
40
+ match self {
41
+ Listener::Tcp(listener) => Ok(listener.local_addr()?.port()),
42
+ Listener::Unix(..) => Ok(0),
43
+ }
44
+ }
45
+ }
46
+
47
+ /// Bind a unix domain socket at `path`. A path that already exists is
48
+ /// either a live listener (refuse: never steal it) or a stale file left
49
+ /// behind by a crashed process (unlink and reclaim). A connect probe tells
50
+ /// them apart: a successful connect means someone is accepting right now.
51
+ pub fn bind_unix(path: &Path) -> io::Result<UnixListener> {
52
+ match std::os::unix::net::UnixStream::connect(path) {
53
+ Ok(_) => return Err(io::Error::new(io::ErrorKind::AddrInUse, "socket is in use")),
54
+ Err(e) if e.kind() == io::ErrorKind::ConnectionRefused => {
55
+ match std::fs::remove_file(path) {
56
+ Ok(()) => {}
57
+ Err(e) if e.kind() == io::ErrorKind::NotFound => {}
58
+ Err(e) => return Err(e),
59
+ }
60
+ }
61
+ Err(e) if e.kind() == io::ErrorKind::NotFound => {}
62
+ Err(e) => return Err(e),
63
+ }
64
+ let listener = UnixListener::bind(path)?;
65
+ listener.set_nonblocking(true)?;
66
+ Ok(listener)
67
+ }
68
+
69
+ /// Remove a socket file at shutdown; one that is already gone is fine.
70
+ pub fn cleanup_unix(path: &Path) {
71
+ let _ = std::fs::remove_file(path);
72
+ }
73
+
74
+ #[cfg(test)]
75
+ mod tests {
76
+ use super::{bind_unix, unix_path, Listener};
77
+ use std::path::PathBuf;
78
+
79
+ /// A socket path unique to this process and test. macOS caps sun_path
80
+ /// at 104 bytes, so the name stays short.
81
+ fn socket_path(name: &str) -> PathBuf {
82
+ let path = std::env::temp_dir().join(format!("kino-{}-{name}.sock", std::process::id()));
83
+ let _ = std::fs::remove_file(&path);
84
+ path
85
+ }
86
+
87
+ #[test]
88
+ fn unix_path_recognises_only_the_unix_scheme() {
89
+ assert_eq!(unix_path("unix:///run/kino.sock").unwrap().to_str(), Some("/run/kino.sock"));
90
+ assert!(unix_path("127.0.0.1").is_none());
91
+ assert!(unix_path("unix.example.com").is_none());
92
+ }
93
+
94
+ #[test]
95
+ fn binds_a_unix_socket_and_reports_no_port() {
96
+ let path = socket_path("bind");
97
+ let listener = Listener::bind(&format!("unix://{}", path.display()), 9292).unwrap();
98
+ assert!(matches!(listener, Listener::Unix(..)));
99
+ assert_eq!(listener.port().unwrap(), 0);
100
+ assert!(std::fs::metadata(&path).is_ok());
101
+ drop(listener);
102
+ let _ = std::fs::remove_file(&path);
103
+ }
104
+
105
+ #[test]
106
+ fn reclaims_a_stale_socket_file() {
107
+ let path = socket_path("stale");
108
+ // Dropping a listener closes the socket but leaves its file behind,
109
+ // exactly what a crashed process leaves.
110
+ drop(bind_unix(&path).unwrap());
111
+ assert!(std::fs::metadata(&path).is_ok());
112
+ let reclaimed = bind_unix(&path).unwrap();
113
+ drop(reclaimed);
114
+ let _ = std::fs::remove_file(&path);
115
+ }
116
+
117
+ #[test]
118
+ fn refuses_a_socket_someone_is_listening_on() {
119
+ let path = socket_path("live");
120
+ let live = bind_unix(&path).unwrap();
121
+ let err = bind_unix(&path).unwrap_err();
122
+ assert_eq!(err.kind(), std::io::ErrorKind::AddrInUse);
123
+ assert!(err.to_string().contains("in use"));
124
+ drop(live);
125
+ let _ = std::fs::remove_file(&path);
126
+ }
127
+
128
+ #[test]
129
+ fn binds_tcp_on_an_ephemeral_port() {
130
+ let listener = Listener::bind("127.0.0.1", 0).unwrap();
131
+ assert!(matches!(listener, Listener::Tcp(_)));
132
+ assert_ne!(listener.port().unwrap(), 0);
133
+ }
134
+ }
@@ -0,0 +1,144 @@
1
+ //! Server log lines: lifecycle notices, crashes and respawns, hook
2
+ //! failures, the failed-request report, and whatever apps write to
3
+ //! rack.errors, all in one shape:
4
+ //!
5
+ //! ```text
6
+ //! kino[4213] worker-3: after_worker_boot hook raised RuntimeError: boom
7
+ //! ```
8
+ //!
9
+ //! The label is syslog's `ident[pid]` tag plus the source that spoke (the
10
+ //! ractor and/or thread name, `main` for neither), styled by level; the
11
+ //! message stays plain. Ruby builds the source, since only Ruby knows its
12
+ //! ractor and thread names, and hands the rest over; this side decides
13
+ //! color per stream and writes, so worker ractors never touch $stdout or
14
+ //! $stderr themselves. A multi-line message is labelled on its first line.
15
+
16
+ use std::io::Write;
17
+
18
+ use magnus::{Error, Ruby};
19
+
20
+ use crate::style::{self, Stream};
21
+
22
+ #[derive(Clone, Copy, Debug, PartialEq, Eq)]
23
+ pub enum Level {
24
+ Info,
25
+ Warn,
26
+ Error,
27
+ }
28
+
29
+ impl Level {
30
+ /// The level named by Ruby ("info", "warn", "error").
31
+ pub fn parse(name: &str) -> Option<Level> {
32
+ match name {
33
+ "info" => Some(Level::Info),
34
+ "warn" => Some(Level::Warn),
35
+ "error" => Some(Level::Error),
36
+ _ => None,
37
+ }
38
+ }
39
+
40
+ /// Notes go to stdout; warnings and errors to stderr.
41
+ fn stream(self) -> Stream {
42
+ match self {
43
+ Level::Info => Stream::Stdout,
44
+ Level::Warn | Level::Error => Stream::Stderr,
45
+ }
46
+ }
47
+
48
+ fn sgr(self) -> &'static str {
49
+ match self {
50
+ Level::Info => style::DIM,
51
+ Level::Warn => style::WARN,
52
+ Level::Error => style::ERROR,
53
+ }
54
+ }
55
+ }
56
+
57
+ /// Write one line from `source` at `level`.
58
+ pub fn emit(level: Level, source: &str, message: &str) {
59
+ let stream = level.stream();
60
+ let label = label(std::process::id(), source);
61
+ let line = format_line(level, &label, message, style::enabled(stream));
62
+ match stream {
63
+ Stream::Stdout => {
64
+ let _ = writeln!(std::io::stdout().lock(), "{line}");
65
+ }
66
+ Stream::Stderr => {
67
+ let _ = writeln!(std::io::stderr().lock(), "{line}");
68
+ }
69
+ }
70
+ }
71
+
72
+ /// The `kino[<pid>] <source>:` tag.
73
+ pub fn label(pid: u32, source: &str) -> String {
74
+ format!("kino[{pid}] {source}:")
75
+ }
76
+
77
+ /// The label styled by level, then the message as given.
78
+ pub fn format_line(level: Level, label: &str, message: &str, color: bool) -> String {
79
+ format!("{} {message}", style::sgr(level.sgr(), label, color))
80
+ }
81
+
82
+ /// The Ruby entry point (Kino::Log): Ruby knows its ractor and thread,
83
+ /// the native side knows the terminal.
84
+ pub fn log_line(ruby: &Ruby, level: String, source: String, message: String) -> Result<(), Error> {
85
+ let level = Level::parse(&level).ok_or_else(|| {
86
+ Error::new(
87
+ ruby.exception_arg_error(),
88
+ format!("unknown log level {level:?}"),
89
+ )
90
+ })?;
91
+ emit(level, &source, &message);
92
+ Ok(())
93
+ }
94
+
95
+ #[cfg(test)]
96
+ mod tests {
97
+ use super::{format_line, label, Level};
98
+
99
+ #[test]
100
+ fn label_is_a_syslog_tag_plus_the_source() {
101
+ assert_eq!(label(4213, "main"), "kino[4213] main:");
102
+ assert_eq!(label(4213, "worker-3/thread-2"), "kino[4213] worker-3/thread-2:");
103
+ }
104
+
105
+ #[test]
106
+ fn plain_line_is_label_then_message() {
107
+ assert_eq!(
108
+ format_line(Level::Info, "kino[1] main:", "hello", false),
109
+ "kino[1] main: hello"
110
+ );
111
+ }
112
+
113
+ #[test]
114
+ fn color_styles_only_the_label_by_level() {
115
+ assert_eq!(
116
+ format_line(Level::Info, "kino[1] main:", "hello", true),
117
+ "\x1b[90mkino[1] main:\x1b[0m hello"
118
+ );
119
+ assert_eq!(
120
+ format_line(Level::Warn, "kino[1] main:", "careful", true),
121
+ "\x1b[33mkino[1] main:\x1b[0m careful"
122
+ );
123
+ assert_eq!(
124
+ format_line(Level::Error, "kino[1] main:", "broke", true),
125
+ "\x1b[91mkino[1] main:\x1b[0m broke"
126
+ );
127
+ }
128
+
129
+ #[test]
130
+ fn a_report_is_labelled_on_its_first_line_only() {
131
+ assert_eq!(
132
+ format_line(Level::Error, "kino[1] main:", "500 GET / · X: y\n a.rb:1", false),
133
+ "kino[1] main: 500 GET / · X: y\n a.rb:1"
134
+ );
135
+ }
136
+
137
+ #[test]
138
+ fn levels_parse_from_their_ruby_names() {
139
+ assert_eq!(Level::parse("info"), Some(Level::Info));
140
+ assert_eq!(Level::parse("warn"), Some(Level::Warn));
141
+ assert_eq!(Level::parse("error"), Some(Level::Error));
142
+ assert_eq!(Level::parse("debug"), None);
143
+ }
144
+ }
@@ -0,0 +1,26 @@
1
+ //! Process-monotonic millisecond clock, shared by the request hot path
2
+ //! (recording) and the control thread (reading busy age). Independent of
3
+ //! wall-clock and of Ruby, so it is identical in :ractor and :threaded.
4
+
5
+ use std::sync::OnceLock;
6
+ use std::time::Instant;
7
+
8
+ static MONO_EPOCH: OnceLock<Instant> = OnceLock::new();
9
+
10
+ /// Milliseconds since the first call anywhere in the process. The u128
11
+ /// millis fit u64 for any realistic uptime (u64 ms is ~584 million years).
12
+ pub fn mono_ms() -> u64 {
13
+ MONO_EPOCH.get_or_init(Instant::now).elapsed().as_millis() as u64
14
+ }
15
+
16
+ #[cfg(test)]
17
+ mod tests {
18
+ use super::*;
19
+
20
+ #[test]
21
+ fn mono_ms_is_monotonic_nondecreasing() {
22
+ let a = mono_ms();
23
+ let b = mono_ms();
24
+ assert!(b >= a, "clock went backwards: {a} then {b}");
25
+ }
26
+ }
@@ -116,6 +116,16 @@ fn admit(
116
116
  mut ctx: BoxedCtx,
117
117
  ) -> Result<RHash, Error> {
118
118
  server.served.fetch_add(1, Ordering::Relaxed);
119
+ slot.served.fetch_add(1, Ordering::Relaxed);
120
+ slot.last_started_ms.store(crate::mono::mono_ms(), Ordering::Relaxed);
121
+ slot.in_flight.fetch_add(1, Ordering::Relaxed);
122
+ // One clock read serves the histogram and, for the access log, the
123
+ // request's queue wait and the start of its time in Ruby.
124
+ let now = std::time::Instant::now();
125
+ let wait = now.duration_since(ctx.enqueued_at);
126
+ server.queue_histogram.record(wait.as_micros() as u64);
127
+ ctx.wait = wait;
128
+ ctx.admitted_at = now;
119
129
  slot.current.lock().push(Arc::downgrade(&ctx.responder));
120
130
  // Wire the slot into the request so blocked body reads/writes are
121
131
  // interruptible the same way the queue pop is.
@@ -137,6 +147,7 @@ fn checkout(ruby: &Ruby, server_id: u64, worker_id: usize) -> Result<Option<Chec
137
147
 
138
148
  // The previous batch is fully answered once the worker comes back.
139
149
  slot.current.lock().clear();
150
+ slot.in_flight.store(0, Ordering::Relaxed);
140
151
  slot.interrupted.store(false, Ordering::SeqCst);
141
152
 
142
153
  Ok(block_take(&server, &slot)?.map(|ctx| (server, slot, ctx)))
@@ -11,6 +11,20 @@ use parking_lot::{Mutex, RwLock};
11
11
  use crate::request::RequestCtx;
12
12
  use crate::response::Responder;
13
13
 
14
+ /// Lifecycle as seen by the control plane's /ready.
15
+ pub const STATE_BOOTING: u8 = 0;
16
+ pub const STATE_READY: u8 = 1;
17
+ pub const STATE_DRAINING: u8 = 2;
18
+
19
+ /// Boot-time configuration echoed by /stats. Stored resolved: mode is
20
+ /// "ractor" or "threaded", never "auto".
21
+ pub struct Topology {
22
+ pub mode: String,
23
+ pub workers: usize,
24
+ pub threads: usize,
25
+ pub batch: usize,
26
+ }
27
+
14
28
  /// Requests travel through channels boxed: one heap allocation at accept
15
29
  /// time instead of moving ~300 bytes by value through every channel hop.
16
30
  pub type BoxedCtx = Box<RequestCtx>;
@@ -45,7 +59,18 @@ pub struct ServerInner {
45
59
  /// 413 (truthful Content-Length) or a mid-stream abort (chunked/lying).
46
60
  pub max_body_size: usize,
47
61
  pub timeouts: AtomicU64,
62
+ /// Lifecycle for /ready: booting until Ruby reports the workers up,
63
+ /// draining once stop_accepting runs. Relaxed everywhere (advisory).
64
+ pub state: std::sync::atomic::AtomicU8,
65
+ /// Worker respawns, recorded from the Ruby supervisor. Lives here so
66
+ /// the control plane reads it without touching Ruby.
67
+ pub respawns: AtomicU64,
68
+ /// Replacements spawned by the quarantine monitor (Relaxed, advisory).
69
+ pub quarantine_replacements: AtomicU64,
70
+ pub topology: Topology,
48
71
  pub https: bool,
72
+ /// The socket file of a `unix://` bind, removed at shutdown.
73
+ pub unix_path: Option<std::path::PathBuf>,
49
74
  /// Native access log sink (None unless log_requests is on).
50
75
  pub access_log: Option<crate::logsink::Sink>,
51
76
  /// Lane-dispatch mode: per-worker queues, awake-preferring dispatch.
@@ -55,6 +80,9 @@ pub struct ServerInner {
55
80
  /// GC roots for zero-copy response buffers (pin.rs). The Ruby Server
56
81
  /// object holds the marking PinKeeper for this slab.
57
82
  pub pin_slab: Arc<crate::pin::PinSlab>,
83
+ /// Queue-wait histogram: recorded at admit (queue.rs), emitted by the
84
+ /// control plane.
85
+ pub queue_histogram: QueueHistogram,
58
86
  }
59
87
 
60
88
  /// One per worker *thread* (slot count = workers × threads). The interrupt
@@ -72,12 +100,90 @@ pub struct WorkerSlot {
72
100
  pub lane_tx: Mutex<Option<flume::Sender<BoxedCtx>>>,
73
101
  pub lane_rx: Option<flume::Receiver<BoxedCtx>>,
74
102
  pub parked: std::sync::atomic::AtomicBool,
103
+ /// Per-slot sensors (Relaxed, advisory). served/in_flight mirror the
104
+ /// global counters at slot granularity; last_started_ms (stamped on
105
+ /// admit) drives busy-age (wedge) reporting.
106
+ pub served: AtomicU64,
107
+ pub in_flight: AtomicUsize,
108
+ pub last_started_ms: AtomicU64,
109
+ /// Set by the quarantine monitor when this slot is abandoned as wedged:
110
+ /// excluded from wedge detection, and its busy_ms is reported as 0.
111
+ pub quarantined: std::sync::atomic::AtomicBool,
75
112
  }
76
113
 
77
114
  /// Per-lane depth cap: small, so a slow handler can only ever delay this
78
115
  /// many queued neighbors (work stealing rescues them anyway).
79
116
  pub const LANE_DEPTH: usize = 4;
80
117
 
118
+ /// Fixed queue-wait bucket boundaries in microseconds (0.5ms .. 10s),
119
+ /// ascending. Emitted in seconds. Not a knob (YAGNI).
120
+ pub const QUEUE_BOUNDS_US: [u64; 14] = [
121
+ 500, 1_000, 2_500, 5_000, 10_000, 25_000, 50_000, 100_000,
122
+ 250_000, 500_000, 1_000_000, 2_500_000, 5_000_000, 10_000_000,
123
+ ];
124
+
125
+ /// Queue-wait histogram: per-bucket counts plus an overflow (the implicit
126
+ /// +Inf bucket), the sum of waits, and the total count. Relaxed atomics,
127
+ /// advisory like the other counters.
128
+ pub struct QueueHistogram {
129
+ pub buckets: [AtomicU64; QUEUE_BOUNDS_US.len()],
130
+ pub overflow: AtomicU64,
131
+ pub sum_us: AtomicU64,
132
+ pub count: AtomicU64,
133
+ }
134
+
135
+ /// A plain (non-atomic) snapshot for the control thread to emit.
136
+ pub struct QueueHistogramSnapshot {
137
+ pub buckets: [u64; QUEUE_BOUNDS_US.len()],
138
+ pub overflow: u64,
139
+ pub sum_us: u64,
140
+ pub count: u64,
141
+ }
142
+
143
+ impl QueueHistogram {
144
+ pub fn new() -> Self {
145
+ QueueHistogram {
146
+ buckets: std::array::from_fn(|_| AtomicU64::new(0)),
147
+ overflow: AtomicU64::new(0),
148
+ sum_us: AtomicU64::new(0),
149
+ count: AtomicU64::new(0),
150
+ }
151
+ }
152
+
153
+ /// Place one wait in its bucket (first bound >= wait, else overflow) and
154
+ /// update sum and count. A linear scan over 14 bounds is trivial.
155
+ pub fn record(&self, wait_us: u64) {
156
+ match QUEUE_BOUNDS_US.iter().position(|&bound| wait_us <= bound) {
157
+ Some(i) => self.buckets[i].fetch_add(1, Ordering::Relaxed),
158
+ None => self.overflow.fetch_add(1, Ordering::Relaxed),
159
+ };
160
+ self.sum_us.fetch_add(wait_us, Ordering::Relaxed);
161
+ self.count.fetch_add(1, Ordering::Relaxed);
162
+ }
163
+
164
+ pub fn snapshot(&self) -> QueueHistogramSnapshot {
165
+ QueueHistogramSnapshot {
166
+ buckets: std::array::from_fn(|i| self.buckets[i].load(Ordering::Relaxed)),
167
+ overflow: self.overflow.load(Ordering::Relaxed),
168
+ sum_us: self.sum_us.load(Ordering::Relaxed),
169
+ count: self.count.load(Ordering::Relaxed),
170
+ }
171
+ }
172
+ }
173
+
174
+ impl Default for QueueHistogram {
175
+ fn default() -> Self {
176
+ Self::new()
177
+ }
178
+ }
179
+
180
+ impl QueueHistogramSnapshot {
181
+ /// Summed queue wait, converted from the stored microseconds to seconds.
182
+ pub fn sum_seconds(&self) -> f64 {
183
+ self.sum_us as f64 / 1_000_000.0
184
+ }
185
+ }
186
+
81
187
  impl WorkerSlot {
82
188
  fn new(lanes: bool) -> Self {
83
189
  let (lane_tx, lane_rx) = if lanes {
@@ -92,6 +198,10 @@ impl WorkerSlot {
92
198
  lane_tx: Mutex::new(lane_tx),
93
199
  lane_rx,
94
200
  parked: std::sync::atomic::AtomicBool::new(false),
201
+ served: AtomicU64::new(0),
202
+ in_flight: AtomicUsize::new(0),
203
+ last_started_ms: AtomicU64::new(0),
204
+ quarantined: std::sync::atomic::AtomicBool::new(false),
95
205
  }
96
206
  }
97
207
  }
@@ -188,11 +298,17 @@ pub fn test_server(lanes: bool, queue_depth: usize) -> Arc<ServerInner> {
188
298
  request_timeout_ms: 0,
189
299
  max_body_size: 0,
190
300
  timeouts: AtomicU64::new(0),
301
+ state: std::sync::atomic::AtomicU8::new(STATE_BOOTING),
302
+ respawns: AtomicU64::new(0),
303
+ quarantine_replacements: AtomicU64::new(0),
304
+ topology: Topology { mode: "threaded".to_string(), workers: 0, threads: 0, batch: 1 },
191
305
  https: false,
306
+ unix_path: None,
192
307
  access_log: None,
193
308
  lanes,
194
309
  lane_cursor: AtomicUsize::new(0),
195
310
  pin_slab: Arc::new(crate::pin::PinSlab::new()),
311
+ queue_histogram: QueueHistogram::new(),
196
312
  })
197
313
  }
198
314
 
@@ -273,4 +389,46 @@ mod tests {
273
389
  let b = next_server_id();
274
390
  assert_ne!(a, b);
275
391
  }
392
+
393
+ #[test]
394
+ fn servers_boot_in_the_booting_state_with_zero_respawns() {
395
+ let server = test_server(false, 4);
396
+ assert_eq!(server.state.load(Ordering::Relaxed), STATE_BOOTING);
397
+ assert_eq!(server.respawns.load(Ordering::Relaxed), 0);
398
+ assert_eq!(server.topology.batch, 1);
399
+ }
400
+
401
+ #[test]
402
+ fn fresh_slot_has_zeroed_per_worker_sensors() {
403
+ let server = test_server(false, 4);
404
+ server.register_worker();
405
+ let slots = server.slots.read();
406
+ let slot = &slots[0];
407
+ assert_eq!(slot.served.load(Ordering::Relaxed), 0);
408
+ assert_eq!(slot.in_flight.load(Ordering::Relaxed), 0);
409
+ assert_eq!(slot.last_started_ms.load(Ordering::Relaxed), 0);
410
+ }
411
+
412
+ #[test]
413
+ fn fresh_slot_is_not_quarantined() {
414
+ let server = test_server(false, 4);
415
+ server.register_worker();
416
+ assert!(!server.slots.read()[0].quarantined.load(Ordering::Relaxed));
417
+ assert_eq!(server.quarantine_replacements.load(Ordering::Relaxed), 0);
418
+ }
419
+
420
+ #[test]
421
+ fn queue_histogram_buckets_by_wait() {
422
+ let h = QueueHistogram::new();
423
+ h.record(400); // <= 500 -> bucket 0
424
+ h.record(500); // == 500 -> bucket 0 (inclusive)
425
+ h.record(600); // (500, 1000] -> bucket 1
426
+ h.record(20_000_000); // > last bound -> overflow
427
+ let s = h.snapshot();
428
+ assert_eq!(s.buckets[0], 2);
429
+ assert_eq!(s.buckets[1], 1);
430
+ assert_eq!(s.overflow, 1);
431
+ assert_eq!(s.count, 4);
432
+ assert_eq!(s.sum_us, 400 + 500 + 600 + 20_000_000);
433
+ }
276
434
  }