kino 0.2.1 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +72 -0
- data/Cargo.lock +192 -52
- data/README.md +131 -16
- data/exe/kino +0 -4
- data/ext/kino/Cargo.toml +4 -2
- data/ext/kino/src/access_log.rs +245 -0
- data/ext/kino/src/control.rs +701 -0
- data/ext/kino/src/cpus.rs +37 -0
- data/ext/kino/src/lib.rs +22 -1
- data/ext/kino/src/listen.rs +134 -0
- data/ext/kino/src/log.rs +144 -0
- data/ext/kino/src/mono.rs +26 -0
- data/ext/kino/src/queue.rs +11 -0
- data/ext/kino/src/registry.rs +158 -0
- data/ext/kino/src/request.rs +52 -2
- data/ext/kino/src/server.rs +229 -53
- data/ext/kino/src/style.rs +42 -39
- data/lib/kino/cli.rb +23 -9
- data/lib/kino/configuration.rb +77 -5
- data/lib/kino/errors_stream.rb +4 -3
- data/lib/kino/hook_fire.rb +22 -0
- data/lib/kino/log.rb +104 -0
- data/lib/kino/quarantine_monitor.rb +73 -0
- data/lib/kino/ractor_supervisor.rb +66 -16
- data/lib/kino/server.rb +201 -34
- data/lib/kino/templates/kino.rb.tt +62 -5
- data/lib/kino/version.rb +1 -1
- data/lib/kino/worker.rb +46 -30
- data/lib/kino/worker_hooks.rb +13 -0
- data/lib/kino.rb +14 -0
- data/lib/rackup/handler/kino.rb +88 -0
- data/sig/kino.rbs +71 -1
- metadata +26 -1
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
//! How many CPUs this process may actually use: the default worker count.
|
|
2
|
+
//!
|
|
3
|
+
//! `Etc.nprocessors` honours the affinity mask but not a cgroup CPU quota,
|
|
4
|
+
//! so a container limited to two CPUs on a 64-core host would spawn 64
|
|
5
|
+
//! workers. The standard library's `available_parallelism` reads both the
|
|
6
|
+
//! mask and the cgroup v1/v2 quota on Linux (rounding a fractional quota
|
|
7
|
+
//! up), which is also what tokio sizes its own pool by.
|
|
8
|
+
|
|
9
|
+
use magnus::{Error, Ruby};
|
|
10
|
+
|
|
11
|
+
/// The usable CPU count, never below one (a quota of 0.5 CPU still needs
|
|
12
|
+
/// a worker; an unreadable count falls back to one rather than failing
|
|
13
|
+
/// boot).
|
|
14
|
+
pub fn available_parallelism(_ruby: &Ruby) -> Result<usize, Error> {
|
|
15
|
+
Ok(count())
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
fn count() -> usize {
|
|
19
|
+
std::thread::available_parallelism().map_or(1, |n| n.get())
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
#[cfg(test)]
|
|
23
|
+
mod tests {
|
|
24
|
+
use super::count;
|
|
25
|
+
|
|
26
|
+
#[test]
|
|
27
|
+
fn reports_at_least_one_cpu() {
|
|
28
|
+
assert!(count() >= 1);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
#[test]
|
|
32
|
+
fn never_exceeds_what_the_os_reports_as_online() {
|
|
33
|
+
// A quota can only lower the count below the online CPUs; never raise it.
|
|
34
|
+
let online = std::thread::available_parallelism().map_or(1, |n| n.get());
|
|
35
|
+
assert!(count() <= online);
|
|
36
|
+
}
|
|
37
|
+
}
|
data/ext/kino/src/lib.rs
CHANGED
|
@@ -4,9 +4,15 @@
|
|
|
4
4
|
#[global_allocator]
|
|
5
5
|
static GLOBAL: mimalloc::MiMalloc = mimalloc::MiMalloc;
|
|
6
6
|
|
|
7
|
+
mod access_log;
|
|
8
|
+
mod control;
|
|
9
|
+
mod cpus;
|
|
7
10
|
mod env_strings;
|
|
8
11
|
mod gvl;
|
|
12
|
+
mod listen;
|
|
13
|
+
mod log;
|
|
9
14
|
mod logsink;
|
|
15
|
+
mod mono;
|
|
10
16
|
mod pin;
|
|
11
17
|
mod queue;
|
|
12
18
|
mod registry;
|
|
@@ -40,6 +46,16 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
|
40
46
|
native.define_singleton_method("close_queue", function!(server::close_queue, 1))?;
|
|
41
47
|
native.define_singleton_method("queue_stats", function!(server::queue_stats, 1))?;
|
|
42
48
|
native.define_singleton_method("server_stats", function!(server::server_stats, 1))?;
|
|
49
|
+
native.define_singleton_method("worker_stats", function!(server::worker_stats, 1))?;
|
|
50
|
+
native.define_singleton_method("queue_time", function!(server::queue_time, 1))?;
|
|
51
|
+
native.define_singleton_method("quarantine_slot", function!(server::quarantine_slot, 2))?;
|
|
52
|
+
native.define_singleton_method(
|
|
53
|
+
"record_quarantine_replacement",
|
|
54
|
+
function!(server::record_quarantine_replacement, 1),
|
|
55
|
+
)?;
|
|
56
|
+
native.define_singleton_method("control_ready", function!(server::control_ready, 1))?;
|
|
57
|
+
native.define_singleton_method("record_respawn", function!(server::record_respawn, 1))?;
|
|
58
|
+
native.define_singleton_method("control_stop", function!(control::control_stop, 1))?;
|
|
43
59
|
native.define_singleton_method("abort_inflight", function!(server::abort_inflight, 2))?;
|
|
44
60
|
native.define_singleton_method(
|
|
45
61
|
"abort_all_inflight",
|
|
@@ -50,8 +66,12 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
|
50
66
|
function!(server::interrupt_all_workers, 1),
|
|
51
67
|
)?;
|
|
52
68
|
native.define_singleton_method("shutdown_runtime", function!(server::shutdown_runtime, 2))?;
|
|
53
|
-
native.define_singleton_method("
|
|
69
|
+
native.define_singleton_method("log_line", function!(log::log_line, 3))?;
|
|
54
70
|
native.define_singleton_method("sleep_chunk", function!(timer::sleep_chunk, 1))?;
|
|
71
|
+
native.define_singleton_method(
|
|
72
|
+
"available_parallelism",
|
|
73
|
+
function!(cpus::available_parallelism, 0),
|
|
74
|
+
)?;
|
|
55
75
|
native.define_singleton_method("log_device_open", function!(logsink::device_open, 1))?;
|
|
56
76
|
native.define_singleton_method("log_device_write", function!(logsink::device_write, 2))?;
|
|
57
77
|
native.define_singleton_method("log_device_close", function!(logsink::device_close, 1))?;
|
|
@@ -75,6 +95,7 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
|
75
95
|
request.define_method("write_chunk", method!(Request::write_chunk, 1))?;
|
|
76
96
|
request.define_method("finish", method!(Request::finish, 0))?;
|
|
77
97
|
request.define_method("abort", method!(Request::abort, 0))?;
|
|
98
|
+
request.define_method("timing", method!(Request::set_timing, 2))?;
|
|
78
99
|
|
|
79
100
|
// Force-resolve the TypedData class cache on the main ractor: magnus
|
|
80
101
|
// resolves it lazily on first wrap, and a racy first resolution from two
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
//! Listening sockets for the main server and the control plane: a TCP
|
|
2
|
+
//! `host:port`, or a `unix://path` domain socket (the usual shape behind
|
|
3
|
+
//! nginx). Binding is synchronous so an address conflict surfaces at boot.
|
|
4
|
+
|
|
5
|
+
use std::io;
|
|
6
|
+
use std::os::unix::net::UnixListener;
|
|
7
|
+
use std::path::{Path, PathBuf};
|
|
8
|
+
|
|
9
|
+
/// The bind scheme that selects a unix domain socket; anything else is a
|
|
10
|
+
/// TCP host.
|
|
11
|
+
pub const UNIX_SCHEME: &str = "unix://";
|
|
12
|
+
|
|
13
|
+
/// The socket path of a `unix://` bind, or None for a TCP host.
|
|
14
|
+
pub fn unix_path(bind: &str) -> Option<&Path> {
|
|
15
|
+
bind.strip_prefix(UNIX_SCHEME).map(Path::new)
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
/// A bound, non-blocking listener of either kind.
|
|
19
|
+
pub enum Listener {
|
|
20
|
+
Tcp(std::net::TcpListener),
|
|
21
|
+
Unix(UnixListener, PathBuf),
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
impl Listener {
|
|
25
|
+
/// Bind `bind:port` (TCP; a hostname resolves to its addresses and the
|
|
26
|
+
/// first that binds wins) or `unix://path`.
|
|
27
|
+
pub fn bind(bind: &str, port: u16) -> io::Result<Listener> {
|
|
28
|
+
match unix_path(bind) {
|
|
29
|
+
Some(path) => Ok(Listener::Unix(bind_unix(path)?, path.to_path_buf())),
|
|
30
|
+
None => {
|
|
31
|
+
let listener = std::net::TcpListener::bind((bind, port))?;
|
|
32
|
+
listener.set_nonblocking(true)?;
|
|
33
|
+
Ok(Listener::Tcp(listener))
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/// The bound TCP port; 0 for a unix socket, which has none.
|
|
39
|
+
pub fn port(&self) -> io::Result<u16> {
|
|
40
|
+
match self {
|
|
41
|
+
Listener::Tcp(listener) => Ok(listener.local_addr()?.port()),
|
|
42
|
+
Listener::Unix(..) => Ok(0),
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/// Bind a unix domain socket at `path`. A path that already exists is
|
|
48
|
+
/// either a live listener (refuse: never steal it) or a stale file left
|
|
49
|
+
/// behind by a crashed process (unlink and reclaim). A connect probe tells
|
|
50
|
+
/// them apart: a successful connect means someone is accepting right now.
|
|
51
|
+
pub fn bind_unix(path: &Path) -> io::Result<UnixListener> {
|
|
52
|
+
match std::os::unix::net::UnixStream::connect(path) {
|
|
53
|
+
Ok(_) => return Err(io::Error::new(io::ErrorKind::AddrInUse, "socket is in use")),
|
|
54
|
+
Err(e) if e.kind() == io::ErrorKind::ConnectionRefused => {
|
|
55
|
+
match std::fs::remove_file(path) {
|
|
56
|
+
Ok(()) => {}
|
|
57
|
+
Err(e) if e.kind() == io::ErrorKind::NotFound => {}
|
|
58
|
+
Err(e) => return Err(e),
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
Err(e) if e.kind() == io::ErrorKind::NotFound => {}
|
|
62
|
+
Err(e) => return Err(e),
|
|
63
|
+
}
|
|
64
|
+
let listener = UnixListener::bind(path)?;
|
|
65
|
+
listener.set_nonblocking(true)?;
|
|
66
|
+
Ok(listener)
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/// Remove a socket file at shutdown; one that is already gone is fine.
|
|
70
|
+
pub fn cleanup_unix(path: &Path) {
|
|
71
|
+
let _ = std::fs::remove_file(path);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
#[cfg(test)]
|
|
75
|
+
mod tests {
|
|
76
|
+
use super::{bind_unix, unix_path, Listener};
|
|
77
|
+
use std::path::PathBuf;
|
|
78
|
+
|
|
79
|
+
/// A socket path unique to this process and test. macOS caps sun_path
|
|
80
|
+
/// at 104 bytes, so the name stays short.
|
|
81
|
+
fn socket_path(name: &str) -> PathBuf {
|
|
82
|
+
let path = std::env::temp_dir().join(format!("kino-{}-{name}.sock", std::process::id()));
|
|
83
|
+
let _ = std::fs::remove_file(&path);
|
|
84
|
+
path
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
#[test]
|
|
88
|
+
fn unix_path_recognises_only_the_unix_scheme() {
|
|
89
|
+
assert_eq!(unix_path("unix:///run/kino.sock").unwrap().to_str(), Some("/run/kino.sock"));
|
|
90
|
+
assert!(unix_path("127.0.0.1").is_none());
|
|
91
|
+
assert!(unix_path("unix.example.com").is_none());
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
#[test]
|
|
95
|
+
fn binds_a_unix_socket_and_reports_no_port() {
|
|
96
|
+
let path = socket_path("bind");
|
|
97
|
+
let listener = Listener::bind(&format!("unix://{}", path.display()), 9292).unwrap();
|
|
98
|
+
assert!(matches!(listener, Listener::Unix(..)));
|
|
99
|
+
assert_eq!(listener.port().unwrap(), 0);
|
|
100
|
+
assert!(std::fs::metadata(&path).is_ok());
|
|
101
|
+
drop(listener);
|
|
102
|
+
let _ = std::fs::remove_file(&path);
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
#[test]
|
|
106
|
+
fn reclaims_a_stale_socket_file() {
|
|
107
|
+
let path = socket_path("stale");
|
|
108
|
+
// Dropping a listener closes the socket but leaves its file behind,
|
|
109
|
+
// exactly what a crashed process leaves.
|
|
110
|
+
drop(bind_unix(&path).unwrap());
|
|
111
|
+
assert!(std::fs::metadata(&path).is_ok());
|
|
112
|
+
let reclaimed = bind_unix(&path).unwrap();
|
|
113
|
+
drop(reclaimed);
|
|
114
|
+
let _ = std::fs::remove_file(&path);
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
#[test]
|
|
118
|
+
fn refuses_a_socket_someone_is_listening_on() {
|
|
119
|
+
let path = socket_path("live");
|
|
120
|
+
let live = bind_unix(&path).unwrap();
|
|
121
|
+
let err = bind_unix(&path).unwrap_err();
|
|
122
|
+
assert_eq!(err.kind(), std::io::ErrorKind::AddrInUse);
|
|
123
|
+
assert!(err.to_string().contains("in use"));
|
|
124
|
+
drop(live);
|
|
125
|
+
let _ = std::fs::remove_file(&path);
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
#[test]
|
|
129
|
+
fn binds_tcp_on_an_ephemeral_port() {
|
|
130
|
+
let listener = Listener::bind("127.0.0.1", 0).unwrap();
|
|
131
|
+
assert!(matches!(listener, Listener::Tcp(_)));
|
|
132
|
+
assert_ne!(listener.port().unwrap(), 0);
|
|
133
|
+
}
|
|
134
|
+
}
|
data/ext/kino/src/log.rs
ADDED
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
//! Server log lines: lifecycle notices, crashes and respawns, hook
|
|
2
|
+
//! failures, the failed-request report, and whatever apps write to
|
|
3
|
+
//! rack.errors, all in one shape:
|
|
4
|
+
//!
|
|
5
|
+
//! ```text
|
|
6
|
+
//! kino[4213] worker-3: after_worker_boot hook raised RuntimeError: boom
|
|
7
|
+
//! ```
|
|
8
|
+
//!
|
|
9
|
+
//! The label is syslog's `ident[pid]` tag plus the source that spoke (the
|
|
10
|
+
//! ractor and/or thread name, `main` for neither), styled by level; the
|
|
11
|
+
//! message stays plain. Ruby builds the source, since only Ruby knows its
|
|
12
|
+
//! ractor and thread names, and hands the rest over; this side decides
|
|
13
|
+
//! color per stream and writes, so worker ractors never touch $stdout or
|
|
14
|
+
//! $stderr themselves. A multi-line message is labelled on its first line.
|
|
15
|
+
|
|
16
|
+
use std::io::Write;
|
|
17
|
+
|
|
18
|
+
use magnus::{Error, Ruby};
|
|
19
|
+
|
|
20
|
+
use crate::style::{self, Stream};
|
|
21
|
+
|
|
22
|
+
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
|
23
|
+
pub enum Level {
|
|
24
|
+
Info,
|
|
25
|
+
Warn,
|
|
26
|
+
Error,
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
impl Level {
|
|
30
|
+
/// The level named by Ruby ("info", "warn", "error").
|
|
31
|
+
pub fn parse(name: &str) -> Option<Level> {
|
|
32
|
+
match name {
|
|
33
|
+
"info" => Some(Level::Info),
|
|
34
|
+
"warn" => Some(Level::Warn),
|
|
35
|
+
"error" => Some(Level::Error),
|
|
36
|
+
_ => None,
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/// Notes go to stdout; warnings and errors to stderr.
|
|
41
|
+
fn stream(self) -> Stream {
|
|
42
|
+
match self {
|
|
43
|
+
Level::Info => Stream::Stdout,
|
|
44
|
+
Level::Warn | Level::Error => Stream::Stderr,
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
fn sgr(self) -> &'static str {
|
|
49
|
+
match self {
|
|
50
|
+
Level::Info => style::DIM,
|
|
51
|
+
Level::Warn => style::WARN,
|
|
52
|
+
Level::Error => style::ERROR,
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/// Write one line from `source` at `level`.
|
|
58
|
+
pub fn emit(level: Level, source: &str, message: &str) {
|
|
59
|
+
let stream = level.stream();
|
|
60
|
+
let label = label(std::process::id(), source);
|
|
61
|
+
let line = format_line(level, &label, message, style::enabled(stream));
|
|
62
|
+
match stream {
|
|
63
|
+
Stream::Stdout => {
|
|
64
|
+
let _ = writeln!(std::io::stdout().lock(), "{line}");
|
|
65
|
+
}
|
|
66
|
+
Stream::Stderr => {
|
|
67
|
+
let _ = writeln!(std::io::stderr().lock(), "{line}");
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
/// The `kino[<pid>] <source>:` tag.
|
|
73
|
+
pub fn label(pid: u32, source: &str) -> String {
|
|
74
|
+
format!("kino[{pid}] {source}:")
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/// The label styled by level, then the message as given.
|
|
78
|
+
pub fn format_line(level: Level, label: &str, message: &str, color: bool) -> String {
|
|
79
|
+
format!("{} {message}", style::sgr(level.sgr(), label, color))
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/// The Ruby entry point (Kino::Log): Ruby knows its ractor and thread,
|
|
83
|
+
/// the native side knows the terminal.
|
|
84
|
+
pub fn log_line(ruby: &Ruby, level: String, source: String, message: String) -> Result<(), Error> {
|
|
85
|
+
let level = Level::parse(&level).ok_or_else(|| {
|
|
86
|
+
Error::new(
|
|
87
|
+
ruby.exception_arg_error(),
|
|
88
|
+
format!("unknown log level {level:?}"),
|
|
89
|
+
)
|
|
90
|
+
})?;
|
|
91
|
+
emit(level, &source, &message);
|
|
92
|
+
Ok(())
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
#[cfg(test)]
|
|
96
|
+
mod tests {
|
|
97
|
+
use super::{format_line, label, Level};
|
|
98
|
+
|
|
99
|
+
#[test]
|
|
100
|
+
fn label_is_a_syslog_tag_plus_the_source() {
|
|
101
|
+
assert_eq!(label(4213, "main"), "kino[4213] main:");
|
|
102
|
+
assert_eq!(label(4213, "worker-3/thread-2"), "kino[4213] worker-3/thread-2:");
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
#[test]
|
|
106
|
+
fn plain_line_is_label_then_message() {
|
|
107
|
+
assert_eq!(
|
|
108
|
+
format_line(Level::Info, "kino[1] main:", "hello", false),
|
|
109
|
+
"kino[1] main: hello"
|
|
110
|
+
);
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
#[test]
|
|
114
|
+
fn color_styles_only_the_label_by_level() {
|
|
115
|
+
assert_eq!(
|
|
116
|
+
format_line(Level::Info, "kino[1] main:", "hello", true),
|
|
117
|
+
"\x1b[90mkino[1] main:\x1b[0m hello"
|
|
118
|
+
);
|
|
119
|
+
assert_eq!(
|
|
120
|
+
format_line(Level::Warn, "kino[1] main:", "careful", true),
|
|
121
|
+
"\x1b[33mkino[1] main:\x1b[0m careful"
|
|
122
|
+
);
|
|
123
|
+
assert_eq!(
|
|
124
|
+
format_line(Level::Error, "kino[1] main:", "broke", true),
|
|
125
|
+
"\x1b[91mkino[1] main:\x1b[0m broke"
|
|
126
|
+
);
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
#[test]
|
|
130
|
+
fn a_report_is_labelled_on_its_first_line_only() {
|
|
131
|
+
assert_eq!(
|
|
132
|
+
format_line(Level::Error, "kino[1] main:", "500 GET / · X: y\n a.rb:1", false),
|
|
133
|
+
"kino[1] main: 500 GET / · X: y\n a.rb:1"
|
|
134
|
+
);
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
#[test]
|
|
138
|
+
fn levels_parse_from_their_ruby_names() {
|
|
139
|
+
assert_eq!(Level::parse("info"), Some(Level::Info));
|
|
140
|
+
assert_eq!(Level::parse("warn"), Some(Level::Warn));
|
|
141
|
+
assert_eq!(Level::parse("error"), Some(Level::Error));
|
|
142
|
+
assert_eq!(Level::parse("debug"), None);
|
|
143
|
+
}
|
|
144
|
+
}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
//! Process-monotonic millisecond clock, shared by the request hot path
|
|
2
|
+
//! (recording) and the control thread (reading busy age). Independent of
|
|
3
|
+
//! wall-clock and of Ruby, so it is identical in :ractor and :threaded.
|
|
4
|
+
|
|
5
|
+
use std::sync::OnceLock;
|
|
6
|
+
use std::time::Instant;
|
|
7
|
+
|
|
8
|
+
static MONO_EPOCH: OnceLock<Instant> = OnceLock::new();
|
|
9
|
+
|
|
10
|
+
/// Milliseconds since the first call anywhere in the process. The u128
|
|
11
|
+
/// millis fit u64 for any realistic uptime (u64 ms is ~584 million years).
|
|
12
|
+
pub fn mono_ms() -> u64 {
|
|
13
|
+
MONO_EPOCH.get_or_init(Instant::now).elapsed().as_millis() as u64
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
#[cfg(test)]
|
|
17
|
+
mod tests {
|
|
18
|
+
use super::*;
|
|
19
|
+
|
|
20
|
+
#[test]
|
|
21
|
+
fn mono_ms_is_monotonic_nondecreasing() {
|
|
22
|
+
let a = mono_ms();
|
|
23
|
+
let b = mono_ms();
|
|
24
|
+
assert!(b >= a, "clock went backwards: {a} then {b}");
|
|
25
|
+
}
|
|
26
|
+
}
|
data/ext/kino/src/queue.rs
CHANGED
|
@@ -116,6 +116,16 @@ fn admit(
|
|
|
116
116
|
mut ctx: BoxedCtx,
|
|
117
117
|
) -> Result<RHash, Error> {
|
|
118
118
|
server.served.fetch_add(1, Ordering::Relaxed);
|
|
119
|
+
slot.served.fetch_add(1, Ordering::Relaxed);
|
|
120
|
+
slot.last_started_ms.store(crate::mono::mono_ms(), Ordering::Relaxed);
|
|
121
|
+
slot.in_flight.fetch_add(1, Ordering::Relaxed);
|
|
122
|
+
// One clock read serves the histogram and, for the access log, the
|
|
123
|
+
// request's queue wait and the start of its time in Ruby.
|
|
124
|
+
let now = std::time::Instant::now();
|
|
125
|
+
let wait = now.duration_since(ctx.enqueued_at);
|
|
126
|
+
server.queue_histogram.record(wait.as_micros() as u64);
|
|
127
|
+
ctx.wait = wait;
|
|
128
|
+
ctx.admitted_at = now;
|
|
119
129
|
slot.current.lock().push(Arc::downgrade(&ctx.responder));
|
|
120
130
|
// Wire the slot into the request so blocked body reads/writes are
|
|
121
131
|
// interruptible the same way the queue pop is.
|
|
@@ -137,6 +147,7 @@ fn checkout(ruby: &Ruby, server_id: u64, worker_id: usize) -> Result<Option<Chec
|
|
|
137
147
|
|
|
138
148
|
// The previous batch is fully answered once the worker comes back.
|
|
139
149
|
slot.current.lock().clear();
|
|
150
|
+
slot.in_flight.store(0, Ordering::Relaxed);
|
|
140
151
|
slot.interrupted.store(false, Ordering::SeqCst);
|
|
141
152
|
|
|
142
153
|
Ok(block_take(&server, &slot)?.map(|ctx| (server, slot, ctx)))
|
data/ext/kino/src/registry.rs
CHANGED
|
@@ -11,6 +11,20 @@ use parking_lot::{Mutex, RwLock};
|
|
|
11
11
|
use crate::request::RequestCtx;
|
|
12
12
|
use crate::response::Responder;
|
|
13
13
|
|
|
14
|
+
/// Lifecycle as seen by the control plane's /ready.
|
|
15
|
+
pub const STATE_BOOTING: u8 = 0;
|
|
16
|
+
pub const STATE_READY: u8 = 1;
|
|
17
|
+
pub const STATE_DRAINING: u8 = 2;
|
|
18
|
+
|
|
19
|
+
/// Boot-time configuration echoed by /stats. Stored resolved: mode is
|
|
20
|
+
/// "ractor" or "threaded", never "auto".
|
|
21
|
+
pub struct Topology {
|
|
22
|
+
pub mode: String,
|
|
23
|
+
pub workers: usize,
|
|
24
|
+
pub threads: usize,
|
|
25
|
+
pub batch: usize,
|
|
26
|
+
}
|
|
27
|
+
|
|
14
28
|
/// Requests travel through channels boxed: one heap allocation at accept
|
|
15
29
|
/// time instead of moving ~300 bytes by value through every channel hop.
|
|
16
30
|
pub type BoxedCtx = Box<RequestCtx>;
|
|
@@ -45,7 +59,18 @@ pub struct ServerInner {
|
|
|
45
59
|
/// 413 (truthful Content-Length) or a mid-stream abort (chunked/lying).
|
|
46
60
|
pub max_body_size: usize,
|
|
47
61
|
pub timeouts: AtomicU64,
|
|
62
|
+
/// Lifecycle for /ready: booting until Ruby reports the workers up,
|
|
63
|
+
/// draining once stop_accepting runs. Relaxed everywhere (advisory).
|
|
64
|
+
pub state: std::sync::atomic::AtomicU8,
|
|
65
|
+
/// Worker respawns, recorded from the Ruby supervisor. Lives here so
|
|
66
|
+
/// the control plane reads it without touching Ruby.
|
|
67
|
+
pub respawns: AtomicU64,
|
|
68
|
+
/// Replacements spawned by the quarantine monitor (Relaxed, advisory).
|
|
69
|
+
pub quarantine_replacements: AtomicU64,
|
|
70
|
+
pub topology: Topology,
|
|
48
71
|
pub https: bool,
|
|
72
|
+
/// The socket file of a `unix://` bind, removed at shutdown.
|
|
73
|
+
pub unix_path: Option<std::path::PathBuf>,
|
|
49
74
|
/// Native access log sink (None unless log_requests is on).
|
|
50
75
|
pub access_log: Option<crate::logsink::Sink>,
|
|
51
76
|
/// Lane-dispatch mode: per-worker queues, awake-preferring dispatch.
|
|
@@ -55,6 +80,9 @@ pub struct ServerInner {
|
|
|
55
80
|
/// GC roots for zero-copy response buffers (pin.rs). The Ruby Server
|
|
56
81
|
/// object holds the marking PinKeeper for this slab.
|
|
57
82
|
pub pin_slab: Arc<crate::pin::PinSlab>,
|
|
83
|
+
/// Queue-wait histogram: recorded at admit (queue.rs), emitted by the
|
|
84
|
+
/// control plane.
|
|
85
|
+
pub queue_histogram: QueueHistogram,
|
|
58
86
|
}
|
|
59
87
|
|
|
60
88
|
/// One per worker *thread* (slot count = workers × threads). The interrupt
|
|
@@ -72,12 +100,90 @@ pub struct WorkerSlot {
|
|
|
72
100
|
pub lane_tx: Mutex<Option<flume::Sender<BoxedCtx>>>,
|
|
73
101
|
pub lane_rx: Option<flume::Receiver<BoxedCtx>>,
|
|
74
102
|
pub parked: std::sync::atomic::AtomicBool,
|
|
103
|
+
/// Per-slot sensors (Relaxed, advisory). served/in_flight mirror the
|
|
104
|
+
/// global counters at slot granularity; last_started_ms (stamped on
|
|
105
|
+
/// admit) drives busy-age (wedge) reporting.
|
|
106
|
+
pub served: AtomicU64,
|
|
107
|
+
pub in_flight: AtomicUsize,
|
|
108
|
+
pub last_started_ms: AtomicU64,
|
|
109
|
+
/// Set by the quarantine monitor when this slot is abandoned as wedged:
|
|
110
|
+
/// excluded from wedge detection, and its busy_ms is reported as 0.
|
|
111
|
+
pub quarantined: std::sync::atomic::AtomicBool,
|
|
75
112
|
}
|
|
76
113
|
|
|
77
114
|
/// Per-lane depth cap: small, so a slow handler can only ever delay this
|
|
78
115
|
/// many queued neighbors (work stealing rescues them anyway).
|
|
79
116
|
pub const LANE_DEPTH: usize = 4;
|
|
80
117
|
|
|
118
|
+
/// Fixed queue-wait bucket boundaries in microseconds (0.5ms .. 10s),
|
|
119
|
+
/// ascending. Emitted in seconds. Not a knob (YAGNI).
|
|
120
|
+
pub const QUEUE_BOUNDS_US: [u64; 14] = [
|
|
121
|
+
500, 1_000, 2_500, 5_000, 10_000, 25_000, 50_000, 100_000,
|
|
122
|
+
250_000, 500_000, 1_000_000, 2_500_000, 5_000_000, 10_000_000,
|
|
123
|
+
];
|
|
124
|
+
|
|
125
|
+
/// Queue-wait histogram: per-bucket counts plus an overflow (the implicit
|
|
126
|
+
/// +Inf bucket), the sum of waits, and the total count. Relaxed atomics,
|
|
127
|
+
/// advisory like the other counters.
|
|
128
|
+
pub struct QueueHistogram {
|
|
129
|
+
pub buckets: [AtomicU64; QUEUE_BOUNDS_US.len()],
|
|
130
|
+
pub overflow: AtomicU64,
|
|
131
|
+
pub sum_us: AtomicU64,
|
|
132
|
+
pub count: AtomicU64,
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/// A plain (non-atomic) snapshot for the control thread to emit.
|
|
136
|
+
pub struct QueueHistogramSnapshot {
|
|
137
|
+
pub buckets: [u64; QUEUE_BOUNDS_US.len()],
|
|
138
|
+
pub overflow: u64,
|
|
139
|
+
pub sum_us: u64,
|
|
140
|
+
pub count: u64,
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
impl QueueHistogram {
|
|
144
|
+
pub fn new() -> Self {
|
|
145
|
+
QueueHistogram {
|
|
146
|
+
buckets: std::array::from_fn(|_| AtomicU64::new(0)),
|
|
147
|
+
overflow: AtomicU64::new(0),
|
|
148
|
+
sum_us: AtomicU64::new(0),
|
|
149
|
+
count: AtomicU64::new(0),
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
/// Place one wait in its bucket (first bound >= wait, else overflow) and
|
|
154
|
+
/// update sum and count. A linear scan over 14 bounds is trivial.
|
|
155
|
+
pub fn record(&self, wait_us: u64) {
|
|
156
|
+
match QUEUE_BOUNDS_US.iter().position(|&bound| wait_us <= bound) {
|
|
157
|
+
Some(i) => self.buckets[i].fetch_add(1, Ordering::Relaxed),
|
|
158
|
+
None => self.overflow.fetch_add(1, Ordering::Relaxed),
|
|
159
|
+
};
|
|
160
|
+
self.sum_us.fetch_add(wait_us, Ordering::Relaxed);
|
|
161
|
+
self.count.fetch_add(1, Ordering::Relaxed);
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
pub fn snapshot(&self) -> QueueHistogramSnapshot {
|
|
165
|
+
QueueHistogramSnapshot {
|
|
166
|
+
buckets: std::array::from_fn(|i| self.buckets[i].load(Ordering::Relaxed)),
|
|
167
|
+
overflow: self.overflow.load(Ordering::Relaxed),
|
|
168
|
+
sum_us: self.sum_us.load(Ordering::Relaxed),
|
|
169
|
+
count: self.count.load(Ordering::Relaxed),
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
impl Default for QueueHistogram {
|
|
175
|
+
fn default() -> Self {
|
|
176
|
+
Self::new()
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
impl QueueHistogramSnapshot {
|
|
181
|
+
/// Summed queue wait, converted from the stored microseconds to seconds.
|
|
182
|
+
pub fn sum_seconds(&self) -> f64 {
|
|
183
|
+
self.sum_us as f64 / 1_000_000.0
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
|
|
81
187
|
impl WorkerSlot {
|
|
82
188
|
fn new(lanes: bool) -> Self {
|
|
83
189
|
let (lane_tx, lane_rx) = if lanes {
|
|
@@ -92,6 +198,10 @@ impl WorkerSlot {
|
|
|
92
198
|
lane_tx: Mutex::new(lane_tx),
|
|
93
199
|
lane_rx,
|
|
94
200
|
parked: std::sync::atomic::AtomicBool::new(false),
|
|
201
|
+
served: AtomicU64::new(0),
|
|
202
|
+
in_flight: AtomicUsize::new(0),
|
|
203
|
+
last_started_ms: AtomicU64::new(0),
|
|
204
|
+
quarantined: std::sync::atomic::AtomicBool::new(false),
|
|
95
205
|
}
|
|
96
206
|
}
|
|
97
207
|
}
|
|
@@ -188,11 +298,17 @@ pub fn test_server(lanes: bool, queue_depth: usize) -> Arc<ServerInner> {
|
|
|
188
298
|
request_timeout_ms: 0,
|
|
189
299
|
max_body_size: 0,
|
|
190
300
|
timeouts: AtomicU64::new(0),
|
|
301
|
+
state: std::sync::atomic::AtomicU8::new(STATE_BOOTING),
|
|
302
|
+
respawns: AtomicU64::new(0),
|
|
303
|
+
quarantine_replacements: AtomicU64::new(0),
|
|
304
|
+
topology: Topology { mode: "threaded".to_string(), workers: 0, threads: 0, batch: 1 },
|
|
191
305
|
https: false,
|
|
306
|
+
unix_path: None,
|
|
192
307
|
access_log: None,
|
|
193
308
|
lanes,
|
|
194
309
|
lane_cursor: AtomicUsize::new(0),
|
|
195
310
|
pin_slab: Arc::new(crate::pin::PinSlab::new()),
|
|
311
|
+
queue_histogram: QueueHistogram::new(),
|
|
196
312
|
})
|
|
197
313
|
}
|
|
198
314
|
|
|
@@ -273,4 +389,46 @@ mod tests {
|
|
|
273
389
|
let b = next_server_id();
|
|
274
390
|
assert_ne!(a, b);
|
|
275
391
|
}
|
|
392
|
+
|
|
393
|
+
#[test]
|
|
394
|
+
fn servers_boot_in_the_booting_state_with_zero_respawns() {
|
|
395
|
+
let server = test_server(false, 4);
|
|
396
|
+
assert_eq!(server.state.load(Ordering::Relaxed), STATE_BOOTING);
|
|
397
|
+
assert_eq!(server.respawns.load(Ordering::Relaxed), 0);
|
|
398
|
+
assert_eq!(server.topology.batch, 1);
|
|
399
|
+
}
|
|
400
|
+
|
|
401
|
+
#[test]
|
|
402
|
+
fn fresh_slot_has_zeroed_per_worker_sensors() {
|
|
403
|
+
let server = test_server(false, 4);
|
|
404
|
+
server.register_worker();
|
|
405
|
+
let slots = server.slots.read();
|
|
406
|
+
let slot = &slots[0];
|
|
407
|
+
assert_eq!(slot.served.load(Ordering::Relaxed), 0);
|
|
408
|
+
assert_eq!(slot.in_flight.load(Ordering::Relaxed), 0);
|
|
409
|
+
assert_eq!(slot.last_started_ms.load(Ordering::Relaxed), 0);
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
#[test]
|
|
413
|
+
fn fresh_slot_is_not_quarantined() {
|
|
414
|
+
let server = test_server(false, 4);
|
|
415
|
+
server.register_worker();
|
|
416
|
+
assert!(!server.slots.read()[0].quarantined.load(Ordering::Relaxed));
|
|
417
|
+
assert_eq!(server.quarantine_replacements.load(Ordering::Relaxed), 0);
|
|
418
|
+
}
|
|
419
|
+
|
|
420
|
+
#[test]
|
|
421
|
+
fn queue_histogram_buckets_by_wait() {
|
|
422
|
+
let h = QueueHistogram::new();
|
|
423
|
+
h.record(400); // <= 500 -> bucket 0
|
|
424
|
+
h.record(500); // == 500 -> bucket 0 (inclusive)
|
|
425
|
+
h.record(600); // (500, 1000] -> bucket 1
|
|
426
|
+
h.record(20_000_000); // > last bound -> overflow
|
|
427
|
+
let s = h.snapshot();
|
|
428
|
+
assert_eq!(s.buckets[0], 2);
|
|
429
|
+
assert_eq!(s.buckets[1], 1);
|
|
430
|
+
assert_eq!(s.overflow, 1);
|
|
431
|
+
assert_eq!(s.count, 4);
|
|
432
|
+
assert_eq!(s.sum_us, 400 + 500 + 600 + 20_000_000);
|
|
433
|
+
}
|
|
276
434
|
}
|