raptor 0.19.0 → 0.20.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +16 -0
- data/README.md +65 -39
- data/docs/brisrails-talk.md +10 -7
- data/docs/raptor-vs-puma.md +31 -22
- data/lib/rackup/handler/raptor.rb +12 -3
- data/lib/raptor/cli.rb +35 -2
- data/lib/raptor/cluster.rb +78 -5
- data/lib/raptor/control_server.rb +131 -0
- data/lib/raptor/http1.rb +32 -12
- data/lib/raptor/http2.rb +22 -2
- data/lib/raptor/thread_locals.rb +38 -0
- data/lib/raptor/version.rb +1 -1
- data/sig/generated/raptor/cli.rbs +8 -0
- data/sig/generated/raptor/cluster.rbs +42 -16
- data/sig/generated/raptor/control_server.rbs +62 -0
- data/sig/generated/raptor/http1.rbs +16 -6
- data/sig/generated/raptor/http2.rbs +13 -2
- data/sig/generated/raptor/thread_locals.rbs +24 -0
- metadata +5 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: e5a9aa19c5b29e7935a63f703d240ae6a132f4c7cc698481b02b9d66f1f65432
|
|
4
|
+
data.tar.gz: e443530d17e9c217982a425f5059c033ddd4eafa260f4ec60a1144bc99c0db4c
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: fcd061377838787f57e3555ba5997679ccf321ed2b8415a3eef7f9d15ccc2b221eb0335f16da285fe33054c430741b2b913663c26f6fad621ab3008e721257b1
|
|
7
|
+
data.tar.gz: 05b46a702067778709f0fb720ad908356650bbfc340cdde8809e49f536ed57dc7d2091bb96a07cda91af56a7a2daf634798e92af9a592d44002c1e84a18ad730
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,21 @@
|
|
|
1
1
|
## [Unreleased]
|
|
2
2
|
|
|
3
|
+
## [0.20.1] - 2026-09-26
|
|
4
|
+
|
|
5
|
+
- Pass worker lifecycle settings through the Rackup handler
|
|
6
|
+
|
|
7
|
+
## [0.20.0] - 2026-08-30
|
|
8
|
+
|
|
9
|
+
- Close connected control socket clients during shutdown
|
|
10
|
+
- Skip TCP corking when `writev(2)` emits a response in one call
|
|
11
|
+
- Return idle HTTP/1.1 keep-alive connections to the reactor without waiting
|
|
12
|
+
- Increase the HTTP/1.1 keep-alive request limit to 1,000 and requeue the connection when other work is waiting
|
|
13
|
+
- Add a read-only `/stats` control socket
|
|
14
|
+
- Pass the worker index to worker boot and shutdown hooks
|
|
15
|
+
- Add `RAPTOR_WORKERS`, `RAPTOR_THREADS`, and `RAPTOR_MAX_THREADS`
|
|
16
|
+
- Make worker CPU affinity opt-in
|
|
17
|
+
- Add request-local thread and Fiber cleanup options
|
|
18
|
+
|
|
3
19
|
## [0.19.0] - 2026-08-29
|
|
4
20
|
|
|
5
21
|
- Skip HTTP/2 Ractor pools without SSL bindings
|
data/README.md
CHANGED
|
@@ -36,29 +36,29 @@ run proc { |_env| [200, { "content-type" => "text/plain" }, ["Hello, World!"]] }
|
|
|
36
36
|
|
|
37
37
|
```
|
|
38
38
|
> bundle exec raptor -w 10 -t 3 hello_world.ru
|
|
39
|
-
[Raptor
|
|
40
|
-
[Raptor
|
|
41
|
-
[Raptor
|
|
42
|
-
[Raptor
|
|
43
|
-
[Raptor
|
|
44
|
-
[Raptor
|
|
45
|
-
[Raptor
|
|
46
|
-
[Raptor
|
|
47
|
-
[Raptor
|
|
48
|
-
[Raptor
|
|
49
|
-
[Raptor
|
|
50
|
-
[Raptor
|
|
51
|
-
[Raptor
|
|
52
|
-
[Raptor
|
|
53
|
-
[Raptor
|
|
54
|
-
[Raptor
|
|
55
|
-
[Raptor
|
|
56
|
-
[Raptor
|
|
57
|
-
[Raptor
|
|
58
|
-
[Raptor
|
|
59
|
-
[Raptor
|
|
60
|
-
[Raptor
|
|
61
|
-
[Raptor
|
|
39
|
+
[Raptor 72876|Main|Main] Cluster initializing:
|
|
40
|
+
[Raptor 72876|Main|Main] ├─ Version: 0.20.1
|
|
41
|
+
[Raptor 72876|Main|Main] ├─ Ruby Version: ruby 4.0.6 (2026-07-14 revision 03b6d3f889) +YJIT +PRISM [arm64-darwin23]
|
|
42
|
+
[Raptor 72876|Main|Main] ├─ Environment: development
|
|
43
|
+
[Raptor 72876|Main|Main] ├─ Master PID: 72876
|
|
44
|
+
[Raptor 72876|Main|Main] │ └─ 10 worker processes
|
|
45
|
+
[Raptor 72876|Main|Main] │ ├─ 1 server thread
|
|
46
|
+
[Raptor 72876|Main|Main] │ ├─ 1 reactor thread
|
|
47
|
+
[Raptor 72876|Main|Main] │ ├─ 1 HTTP/1.1 pipeline ractor
|
|
48
|
+
[Raptor 72876|Main|Main] │ ├─ 1 pipeline collector thread
|
|
49
|
+
[Raptor 72876|Main|Main] │ ├─ 3 worker threads (scaling, no limit)
|
|
50
|
+
[Raptor 72876|Main|Main] │ └─ 1 stats thread
|
|
51
|
+
[Raptor 72876|Main|Main] └─ Listening on 0.0.0.0:9292
|
|
52
|
+
[Raptor 72884|Main|Main] Worker 0 booted
|
|
53
|
+
[Raptor 72885|Main|Main] Worker 1 booted
|
|
54
|
+
[Raptor 72886|Main|Main] Worker 2 booted
|
|
55
|
+
[Raptor 72887|Main|Main] Worker 3 booted
|
|
56
|
+
[Raptor 72891|Main|Main] Worker 7 booted
|
|
57
|
+
[Raptor 72888|Main|Main] Worker 4 booted
|
|
58
|
+
[Raptor 72890|Main|Main] Worker 6 booted
|
|
59
|
+
[Raptor 72889|Main|Main] Worker 5 booted
|
|
60
|
+
[Raptor 72892|Main|Main] Worker 8 booted
|
|
61
|
+
[Raptor 72893|Main|Main] Worker 9 booted
|
|
62
62
|
```
|
|
63
63
|
|
|
64
64
|
```
|
|
@@ -73,6 +73,9 @@ Also works with `rackup` and `rails server`:
|
|
|
73
73
|
> bundle exec rails server -u raptor
|
|
74
74
|
```
|
|
75
75
|
|
|
76
|
+
Rails apps using `SOLID_QUEUE_IN_PUMA` run jobs through `config/puma.rb`, which Raptor does not load. Run `bin/jobs` as
|
|
77
|
+
a separately supervised process or container instead.
|
|
78
|
+
|
|
76
79
|
## Configuration
|
|
77
80
|
|
|
78
81
|
Raptor accepts configuration via command-line flags, a Ruby config file, or both (CLI flags override config file
|
|
@@ -93,6 +96,9 @@ The config file is a Ruby file that evaluates to a hash of options. By default R
|
|
|
93
96
|
workers: 4, # `Etc.nprocessors`
|
|
94
97
|
threads: 3,
|
|
95
98
|
max_threads: Float::INFINITY, # set to `threads` for a fixed pool
|
|
99
|
+
cpu_affinity: false,
|
|
100
|
+
clean_thread_locals: true,
|
|
101
|
+
clean_fiber_locals: false,
|
|
96
102
|
chdir: nil,
|
|
97
103
|
environment: nil, # falls back to `RAILS_ENV`, then `RACK_ENV`, then `"development"`
|
|
98
104
|
connection: {
|
|
@@ -105,7 +111,7 @@ The config file is a Ruby file that evaluates to a hash of options. By default R
|
|
|
105
111
|
http1: {
|
|
106
112
|
ractors: nil,
|
|
107
113
|
persistent_data_timeout: 65,
|
|
108
|
-
max_keepalive_requests:
|
|
114
|
+
max_keepalive_requests: 1000,
|
|
109
115
|
},
|
|
110
116
|
http2: {
|
|
111
117
|
ractors: nil,
|
|
@@ -121,6 +127,7 @@ The config file is a Ruby file that evaluates to a hash of options. By default R
|
|
|
121
127
|
before_worker_shutdown: [],
|
|
122
128
|
before_refork: [],
|
|
123
129
|
stats_file: "tmp/raptor.json",
|
|
130
|
+
control_url: nil,
|
|
124
131
|
pid_file: nil,
|
|
125
132
|
stdout_file: nil,
|
|
126
133
|
stderr_file: nil,
|
|
@@ -133,6 +140,19 @@ without a fixed limit when queued work is held up by blocking operations. It doe
|
|
|
133
140
|
GVL is the bottleneck, and temporary threads leave after the queue drains. Set `max_threads` to cap growth, or set it
|
|
134
141
|
to the same value as `threads` for a fixed pool.
|
|
135
142
|
|
|
143
|
+
Set `cpu_affinity` to `true` to pin each worker to a distinct CPU when the worker count fits within the process's
|
|
144
|
+
allowed CPU set. It is off by default because container runtimes commonly expose CPUs that are shared with other
|
|
145
|
+
containers.
|
|
146
|
+
|
|
147
|
+
Raptor clears application thread locals after each request by default. Set `clean_thread_locals` to `false` to disable
|
|
148
|
+
it. Set `clean_fiber_locals` to `true` to run each request in a fresh Fiber, isolating Fiber-local state as well.
|
|
149
|
+
|
|
150
|
+
`RAPTOR_WORKERS`, `RAPTOR_THREADS`, and `RAPTOR_MAX_THREADS` can set the corresponding options without a config file.
|
|
151
|
+
Config files override defaults, environment variables override config files, and command-line options override both.
|
|
152
|
+
`RAPTOR_MAX_THREADS=unlimited` leaves adaptive growth uncapped.
|
|
153
|
+
|
|
154
|
+
`before_worker_boot` and `before_worker_shutdown` hooks receive the worker index.
|
|
155
|
+
|
|
136
156
|
## Bindings
|
|
137
157
|
|
|
138
158
|
Raptor accepts multiple `binds:` URIs across three schemes.
|
|
@@ -204,9 +224,14 @@ Worker 1 (phase 0): pid=91351, requests=1199, busy=1/3, backlog=0, booted, last_
|
|
|
204
224
|
...
|
|
205
225
|
```
|
|
206
226
|
|
|
227
|
+
Set `control_url` to a Unix socket URL such as `unix:///tmp/raptor-control.sock` to expose cluster stats over `/stats`.
|
|
228
|
+
For adaptive pools, `max_threads` in each worker's status is its current thread count, so
|
|
229
|
+
`pool_capacity / max_threads` measures the capacity available at that moment rather than comparing against an
|
|
230
|
+
unbounded configured limit. The control server is read-only and currently exposes only `/stats`.
|
|
231
|
+
|
|
207
232
|
## (Micro) Benchmarks
|
|
208
233
|
|
|
209
|
-
Raptor 0.
|
|
234
|
+
Raptor 0.20.1 vs Puma 8.0.2 vs Falcon 0.57.0 across two workload profiles. **IO-bound** is a GET endpoint that
|
|
210
235
|
interleaves 5-10 short sleeps (total 2.5-15ms) with small CPU work, simulating a read path that makes several DB or
|
|
211
236
|
cache calls. **CPU-bound** is a POST endpoint that accepts a small JSON body, interleaves 3-5 chunks of JSON item
|
|
212
237
|
building (total 450-1500 items) with sub-100µs sleeps, and returns the built array, simulating a write path that does
|
|
@@ -214,27 +239,28 @@ most of its work in Ruby with a few near-zero-cost cache hits.
|
|
|
214
239
|
|
|
215
240
|
Raptor is run in two modes: **Fixed** keeps 3 application threads per worker, matching Puma, while **Scaling** starts
|
|
216
241
|
with 3 and may add threads without a fixed limit when queued work is blocked outside the GVL. Both modes are compared
|
|
217
|
-
with both Puma and Falcon in the table below.
|
|
242
|
+
with both Puma and Falcon in the table below. Raptor request-local cleanup and Puma's Fiber-per-request mode are
|
|
243
|
+
disabled, and both threaded servers allow 999 requests per HTTP/1.1 keep-alive connection.
|
|
218
244
|
|
|
219
245
|
Each cell reports the median throughput and median p95 latency independently across 3 runs, so the two numbers in a row
|
|
220
246
|
may come from different runs. Every run starts a fresh server process so the samples are independent of each other;
|
|
221
247
|
state accumulated in a previous run cannot bias the next. Across the whole table, the widest spread
|
|
222
|
-
((max - min) / 2 / median) between runs of a single cell was ±
|
|
248
|
+
((max - min) / 2 / median) between runs of a single cell was ±12.4% for throughput and ±16.5% for p95.
|
|
223
249
|
|
|
224
250
|
| Protocol | Workload | Raptor mode | Raptor req/s | Raptor p95 | Puma req/s | Puma p95 | vs Puma req/s | vs Puma p95 | Falcon req/s | Falcon p95 | vs Falcon req/s | vs Falcon p95 |
|
|
225
251
|
| --------------------- | -------- | ----------- | ------------ | ---------- | ----------- | --------- | ------------- | ------------ | ------------ | ---------- | --------------- | ------------- |
|
|
226
|
-
| HTTP/1.1 | IO | Fixed |
|
|
227
|
-
| HTTP/1.1 | IO | Scaling |
|
|
228
|
-
| HTTP/1.1 | CPU | Fixed |
|
|
229
|
-
| HTTP/1.1 | CPU | Scaling |
|
|
230
|
-
| HTTP/1.1 (keep-alive) | IO | Fixed |
|
|
231
|
-
| HTTP/1.1 (keep-alive) | IO | Scaling |
|
|
232
|
-
| HTTP/1.1 (keep-alive) | CPU | Fixed |
|
|
233
|
-
| HTTP/1.1 (keep-alive) | CPU | Scaling |
|
|
234
|
-
| HTTP/2 | IO | Fixed | 1.
|
|
235
|
-
| HTTP/2 | IO | Scaling |
|
|
236
|
-
| HTTP/2 | CPU | Fixed |
|
|
237
|
-
| HTTP/2 | CPU | Scaling | 6.
|
|
252
|
+
| HTTP/1.1 | IO | Fixed | 2.90k req/s | 81.10 ms | 1.41k req/s | 141.50 ms | 105.8% higher | 42.7% lower | 12.26k req/s | 14.00 ms | 76.4% lower | 479.3% higher |
|
|
253
|
+
| HTTP/1.1 | IO | Scaling | 6.24k req/s | 36.20 ms | 1.41k req/s | 141.50 ms | 343.2% higher | 74.4% lower | 12.26k req/s | 14.00 ms | 49.1% lower | 158.6% higher |
|
|
254
|
+
| HTTP/1.1 | CPU | Fixed | 7.21k req/s | 37.90 ms | 8.20k req/s | 23.20 ms | 12.1% lower | 63.4% higher | 5.28k req/s | 35.00 ms | 36.6% higher | 8.3% higher |
|
|
255
|
+
| HTTP/1.1 | CPU | Scaling | 6.04k req/s | 41.60 ms | 8.20k req/s | 23.20 ms | 26.4% lower | 79.3% higher | 5.28k req/s | 35.00 ms | 14.4% higher | 18.9% higher |
|
|
256
|
+
| HTTP/1.1 (keep-alive) | IO | Fixed | 2.35k req/s | 73.50 ms | 1.37k req/s | 124.50 ms | 71.2% higher | 41.0% lower | 6.35k req/s | 27.50 ms | 63.0% lower | 167.3% higher |
|
|
257
|
+
| HTTP/1.1 (keep-alive) | IO | Scaling | 7.83k req/s | 23.20 ms | 1.37k req/s | 124.50 ms | 470.1% higher | 81.4% lower | 6.35k req/s | 27.50 ms | 23.2% higher | 15.6% lower |
|
|
258
|
+
| HTTP/1.1 (keep-alive) | CPU | Fixed | 7.14k req/s | 28.60 ms | 7.85k req/s | 24.50 ms | 9.1% lower | 16.7% higher | 5.56k req/s | 41.60 ms | 28.4% higher | 31.2% lower |
|
|
259
|
+
| HTTP/1.1 (keep-alive) | CPU | Scaling | 7.48k req/s | 26.20 ms | 7.85k req/s | 24.50 ms | 4.8% lower | 6.9% higher | 5.56k req/s | 41.60 ms | 34.5% higher | 37.0% lower |
|
|
260
|
+
| HTTP/2 | IO | Fixed | 1.13k req/s | 174.70 ms | N/A | N/A | - | - | 7.18k req/s | 26.37 ms | 84.3% lower | 562.5% higher |
|
|
261
|
+
| HTTP/2 | IO | Scaling | 6.48k req/s | 29.15 ms | N/A | N/A | - | - | 7.18k req/s | 26.37 ms | 9.7% lower | 10.6% higher |
|
|
262
|
+
| HTTP/2 | CPU | Fixed | 6.31k req/s | 31.74 ms | N/A | N/A | - | - | 6.85k req/s | 51.60 ms | 7.9% lower | 38.5% lower |
|
|
263
|
+
| HTTP/2 | CPU | Scaling | 6.70k req/s | 31.91 ms | N/A | N/A | - | - | 6.85k req/s | 51.60 ms | 2.2% lower | 38.2% lower |
|
|
238
264
|
|
|
239
265
|
> ruby 4.0.6 (2026-07-14 revision 03b6d3f889) +YJIT +PRISM [aarch64-linux]
|
|
240
266
|
> 10 worker processes; fixed Raptor and Puma run 3 threads per worker; scaling Raptor starts at 3 with no fixed limit;
|
data/docs/brisrails-talk.md
CHANGED
|
@@ -1098,6 +1098,8 @@ The distinction matters. More threads help when requests are asleep in database
|
|
|
1098
1098
|
|
|
1099
1099
|
Growth has no fixed limit by default. Set `max_threads` to cap it, or set it to `threads` to keep the pool fixed. OS threads still are not as cheap as fibers.
|
|
1100
1100
|
|
|
1101
|
+
Raptor clears application thread locals when each request finishes. Running every request in a fresh Fiber is also available when an application needs Fiber-local isolation. Parser and response buffers are kept separately so the server can still reuse them without carrying application state into the next request.
|
|
1102
|
+
|
|
1101
1103
|
<br>
|
|
1102
1104
|
<br>
|
|
1103
1105
|
<br>
|
|
@@ -1318,7 +1320,7 @@ flowchart TB
|
|
|
1318
1320
|
Col -->|"complete"| ATP
|
|
1319
1321
|
Col -->|"incomplete"| Rct
|
|
1320
1322
|
ATP -->|"response bytes"| Client
|
|
1321
|
-
ATP -.->|"keep-alive:
|
|
1323
|
+
ATP -.->|"keep-alive: parse ready bytes inline"| ATP
|
|
1322
1324
|
```
|
|
1323
1325
|
|
|
1324
1326
|
<br>
|
|
@@ -1386,20 +1388,20 @@ After writing a response, if keep-alive is on:
|
|
|
1386
1388
|
|
|
1387
1389
|
```ruby
|
|
1388
1390
|
loop do
|
|
1389
|
-
unless socket.wait_readable(0
|
|
1391
|
+
unless socket.wait_readable(0)
|
|
1390
1392
|
reactor.persist(socket, id, ...)
|
|
1391
1393
|
return
|
|
1392
1394
|
end
|
|
1393
1395
|
|
|
1394
|
-
# Bytes
|
|
1396
|
+
# Bytes are ready. Parse the next request inline on this thread.
|
|
1395
1397
|
# ...
|
|
1396
1398
|
end
|
|
1397
1399
|
```
|
|
1398
1400
|
|
|
1399
|
-
- <big>
|
|
1400
|
-
- <big>If bytes
|
|
1401
|
-
- <big>Return to the reactor when no bytes
|
|
1402
|
-
- <big>
|
|
1401
|
+
- <big>Check for the next request on the same connection without waiting</big>
|
|
1402
|
+
- <big>If bytes are ready: parse and dispatch inline, on the same thread</big>
|
|
1403
|
+
- <big>Return to the reactor immediately when no bytes are ready, or when a request is incomplete</big>
|
|
1404
|
+
- <big>Back-to-back requests avoid a reactor round-trip without holding an app thread open</big>
|
|
1403
1405
|
|
|
1404
1406
|
<br>
|
|
1405
1407
|
<br>
|
|
@@ -1707,6 +1709,7 @@ flowchart TB
|
|
|
1707
1709
|
- <big>Each worker writes a 49-byte slot every second: pid, phase, requests, backlog, busy and available threads, boot time, checkin time, booted flag</big>
|
|
1708
1710
|
- <big>Master reads the whole region directly. No JSON. No pipe drain. No signal.</big>
|
|
1709
1711
|
- <big>`bundle exec raptor stats` prints the region as JSON, essentially instantly</big>
|
|
1712
|
+
- <big>An optional read-only Unix socket exposes the cluster snapshot at `GET /stats` for monitoring</big>
|
|
1710
1713
|
|
|
1711
1714
|
Wrapped in a small C extension I wrote: **`mmap-ruby`**.
|
|
1712
1715
|
|
data/docs/raptor-vs-puma.md
CHANGED
|
@@ -44,7 +44,7 @@ The rest of this doc explains why the shape looks like that.
|
|
|
44
44
|
| Cluster dispatch | Workers race on inherited listeners with a load-proportional accept delay | Two-choice load-aware BPF dispatch for TCP on Linux; shared-listener fallback |
|
|
45
45
|
| Work queue | Ruby `Queue` coordinated under the pool mutex | Lock-free Michael-Scott FIFO queue |
|
|
46
46
|
| HTTP/2 | Not implemented | Native C parser + HPACK, lock-free per-connection frame writer |
|
|
47
|
-
| Keep-alive fast path | Same-thread inline dispatch when spare threads exist | Same-thread inline
|
|
47
|
+
| Keep-alive fast path | Same-thread inline dispatch when spare threads exist | Same-thread inline dispatch for bytes that are already waiting |
|
|
48
48
|
| Native extensions | 1 (Ragel HTTP/1 parser + MiniSSL) | 3, all Ractor-safe (Ragel HTTP/1 parser; HTTP/2 parser + HPACK; `writev`, `sched_setaffinity`, `prctl` wrappers) |
|
|
49
49
|
| Shared state (worker↔master) | Pipes and signals | Anonymous shared-memory `mmap` region |
|
|
50
50
|
| Restart primitives | Phased (USR1), hot (USR2 re-exec, inherits FDs via env), refork (SIGURG) | Phased (USR1), hot (USR2 re-exec, inherits FDs via env), refork (SIGURG) |
|
|
@@ -181,8 +181,12 @@ Raptor takes a different position on nearly every axis. It is opinionated in a w
|
|
|
181
181
|
|
|
182
182
|
There is no single mode. Raptor is always a cluster. A master forks N workers, monitors them, and restarts crashed workers. The Rack app is always loaded in the master before forking, so copy-on-write is preserved by default (no user-visible `preload_app` knob).
|
|
183
183
|
|
|
184
|
+
The process and app-thread counts can be set with `RAPTOR_WORKERS`, `RAPTOR_THREADS`, and `RAPTOR_MAX_THREADS`, which lets a deployment tune them without generating a config file. Config files override the built-in defaults, environment variables override config files, and command-line options override both. `RAPTOR_MAX_THREADS=unlimited` leaves adaptive growth uncapped.
|
|
185
|
+
|
|
184
186
|
The master is a supervisor. It never handles requests. It forks workers, watches them via a shared-memory region (more on that in a moment), traps signals, restarts crashed workers, and orchestrates restarts.
|
|
185
187
|
|
|
188
|
+
Worker boot and shutdown hooks receive the worker's slot index, so setup and cleanup can identify the same slot across restarts. Hooks that do not take an argument continue to work.
|
|
189
|
+
|
|
186
190
|
Two kinds of restart are supported:
|
|
187
191
|
|
|
188
192
|
1. **Phased restart on SIGUSR1.** Same idea as Puma. Kill each worker in sequence, wait for its replacement to boot, move on. Existing workers drain their connections while their replacements come up. This is cheap and safe when the change does not require a fresh master.
|
|
@@ -193,7 +197,7 @@ Systemd socket activation is a native feature and slots straight into this model
|
|
|
193
197
|
|
|
194
198
|
Routine worker monitoring does not use pipes. Every worker writes its stats (pid, request count, backlog, busy and available threads, last checkin timestamp, booted flag) into a fixed-size slot in an anonymous shared-memory region allocated with `mmap-ruby` before the fork. The master reads the region directly. There is no serialisation, pipe drain, or signal to trigger the read; it is 49 bytes per worker of native memory. `bundle exec raptor stats` prints a JSON snapshot. Refork coordination is separate and does use a pair of pipes between the master and seed.
|
|
195
199
|
|
|
196
|
-
On Linux,
|
|
200
|
+
On Linux, `cpu_affinity: true` pins each worker to a distinct CPU via `sched_setaffinity` when the worker count fits within the process's allowed CPU set, so it stays on one core and its L1/L2 caches stay warm. It is off by default because an allowed CPU in a container is not necessarily dedicated to that container. When workers outnumber available CPUs the pin is skipped and the kernel scheduler manages placement.
|
|
197
201
|
|
|
198
202
|
### Refork
|
|
199
203
|
|
|
@@ -242,6 +246,8 @@ The pool starts at `threads` and scales without a fixed limit by default. Set `m
|
|
|
242
246
|
|
|
243
247
|
The pool still uses an `AtomicConditionVariable` under the hood to park idle threads (idle threads call `Thread.stop` and get woken with `Thread#wakeup`; there is no spinning), because idle spinning would waste CPU. The difference from Puma's pool is not "no locks anywhere" but rather "the hot path (enqueue and dequeue when the queue has items) is lock-free". Once every worker is busy the mechanics look similar; where things diverge is under contention when you have many threads all trying to push and pop.
|
|
244
248
|
|
|
249
|
+
Application thread locals are cleared when each request finishes. Running every request in a fresh Fiber is independently configurable, but opt-in. The pool keeps its own parser and response buffers in private thread variables so the server can reuse them without preserving application state between requests.
|
|
250
|
+
|
|
245
251
|
The knock-on effect is that the server thread can read `pool.queue_size + pool.active_count` on every accept-loop iteration without acquiring the queue's mutation lock. Those are still synchronised atomic reads, but they do not serialise producers and consumers behind one mutex.
|
|
246
252
|
|
|
247
253
|
### I/O model
|
|
@@ -305,7 +311,7 @@ The eager keep-alive loop is one of Raptor's more deliberate latency/occupancy t
|
|
|
305
311
|
|
|
306
312
|
```ruby
|
|
307
313
|
loop do
|
|
308
|
-
unless socket.wait_readable(
|
|
314
|
+
unless socket.wait_readable(0)
|
|
309
315
|
reactor.persist(socket, id, request_count, ...)
|
|
310
316
|
return
|
|
311
317
|
end
|
|
@@ -314,9 +320,9 @@ loop do
|
|
|
314
320
|
end
|
|
315
321
|
```
|
|
316
322
|
|
|
317
|
-
The thread
|
|
323
|
+
The thread checks for bytes without waiting. If they are already available, it parses them inline and calls the Rack app again. Otherwise the connection returns to the reactor immediately. Pipelined requests avoid a reactor round-trip without letting an idle keep-alive connection occupy an app thread.
|
|
318
324
|
|
|
319
|
-
Puma has a similar shape
|
|
325
|
+
Puma has a similar shape. It checks buffered back-to-back requests, then eagerly drains bytes already available on the socket. If a complete request is ready and the pool has a waiting thread, the current thread loops inline; otherwise Puma queues the client or returns it to the reactor.
|
|
320
326
|
|
|
321
327
|
The `reactor.persist` call re-registers the socket with the reactor using `persistent_data_timeout` (65s) as the new deadline. When the next bytes arrive, the reactor treats the socket like any other partially-read connection.
|
|
322
328
|
|
|
@@ -384,7 +390,7 @@ flowchart TB
|
|
|
384
390
|
|
|
385
391
|
CHK{"Request<br/>complete?"}
|
|
386
392
|
KA{"Keep-alive?"}
|
|
387
|
-
EAG{"wait_readable<br/>
|
|
393
|
+
EAG{"wait_readable(0)<br/>bytes ready?"}
|
|
388
394
|
|
|
389
395
|
SRV -->|"HTTP/1.1 eager_accept, parse inline, push proc"| ATP
|
|
390
396
|
SRV -->|"HTTP/2 eager_accept, parse inline, push proc"| ATP
|
|
@@ -399,7 +405,7 @@ flowchart TB
|
|
|
399
405
|
ATP -->|"app.call + write"| KA
|
|
400
406
|
KA -->|"no, close"| CLS["close socket"]
|
|
401
407
|
KA -->|"yes"| EAG
|
|
402
|
-
EAG -.->|"bytes
|
|
408
|
+
EAG -.->|"bytes ready, parse+dispatch on same thread"| ATP
|
|
403
409
|
EAG -->|"no bytes, reactor.persist"| RCT
|
|
404
410
|
RCT -->|"timeout expired"| TO["write 408, close"]
|
|
405
411
|
|
|
@@ -431,7 +437,7 @@ The critical structural difference from Puma is that Raptor has a separate proto
|
|
|
431
437
|
|
|
432
438
|
**Puma.** Parsing happens on an app thread. The C parser callbacks build the env hash. A fresh client first enters the thread pool, where eager reads may complete it immediately; partial and idle keep-alive connections wait in the reactor before returning to the pool. Parsing shares the worker's GVL with the app.
|
|
433
439
|
|
|
434
|
-
**Raptor.** Fresh and immediate keep-alive requests parse inline on the server or app thread. A connection that needs more bytes takes the longer path: reactor (I/O) → protocol Ractor pool (parse) → collector → app thread pool (Rack + write). Between keep-alive requests, the app thread
|
|
440
|
+
**Raptor.** Fresh and immediate keep-alive requests parse inline on the server or app thread. A connection that needs more bytes takes the longer path: reactor (I/O) → protocol Ractor pool (parse) → collector → app thread pool (Rack + write). Between keep-alive requests, the app thread checks for bytes without waiting before returning an idle connection to the reactor. Parsing in the Ractor pipeline has its own GVL; parsing on an eager path does not.
|
|
435
441
|
|
|
436
442
|
In practice, Puma has one process-wide GVL per worker. Every Ruby thread inside that worker takes turns holding it. Raptor has the same main-Ractor GVL plus one GVL per protocol Ractor, so a pipeline Ractor can parse one connection while an app thread executes Rack for another. That parallelism is real, but so are the costs of making state shareable and crossing the Ractor and collector boundaries. Which side wins depends on how much work the request gives the protocol pipeline; the current CPU benchmark leaves Raptor and Puma close rather than proving a universal parsing advantage.
|
|
437
443
|
|
|
@@ -457,11 +463,11 @@ Under moderate load, queue mechanics are unlikely to dominate either server. Und
|
|
|
457
463
|
|
|
458
464
|
**Puma.** After a response, if the connection is keep-alive and there are already buffered bytes for the next request (`has_back_to_back_requests?`) and there is a spare app thread, loop inline. Otherwise, if `eagerly_finish` (non-blocking reads while data is already buffered) returns true, either loop inline (if spare threads) or hand back to the thread pool (`@thread_pool << client`). Otherwise, back to the reactor with `@persistent_timeout`.
|
|
459
465
|
|
|
460
|
-
**Raptor.** After a response, the app thread
|
|
466
|
+
**Raptor.** After a response, the app thread checks `socket.wait_readable(0)`. If bytes are already available, it parses the next request inline. If other work is waiting, the parsed request goes to the back of the pool queue; otherwise the same thread dispatches it inline. If no bytes are ready, `reactor.persist` and return.
|
|
461
467
|
|
|
462
|
-
|
|
468
|
+
Both servers therefore keep a back-to-back request on an app thread when its bytes are already waiting, and return an idle connection to the reactor without deliberately holding an app thread open. The details of their parsing and pool handoff still differ, but neither widens the inline window by waiting for the client.
|
|
463
469
|
|
|
464
|
-
|
|
470
|
+
Raptor closes an HTTP/1.1 connection after 1,000 requests by default. The finite limit bounds connection-lifetime state without forcing frequent TCP teardown and reconnection under sustained keep-alive traffic.
|
|
465
471
|
|
|
466
472
|
### Backpressure
|
|
467
473
|
|
|
@@ -477,6 +483,8 @@ This fast path is one plausible contributor to Raptor's keep-alive result. Reque
|
|
|
477
483
|
|
|
478
484
|
The performance difference here is negligible because the update happens once per worker per second, outside request processing. The design mainly gives the master a fixed-size snapshot it can inspect without draining per-worker messages.
|
|
479
485
|
|
|
486
|
+
For external monitoring, `control_url` can expose a read-only `GET /stats` endpoint over a Unix socket. Its per-worker status includes backlog, busy threads, free capacity, request count, and the thread count currently available. Adaptive pools report their current size as `max_threads`, rather than their configured limit, so `pool_capacity / max_threads` remains a useful measure while the pool grows and shrinks.
|
|
487
|
+
|
|
480
488
|
### HTTP/2
|
|
481
489
|
|
|
482
490
|
**Puma.** Not implemented. Puma's [position](https://github.com/puma/puma/issues/2697) is that HTTP/2 belongs at the edge (nginx, Caddy, ALB), which terminates it and speaks HTTP/1.1 to the app server. That's a reasonable call for the deployments Puma is aimed at, and it's where most Rails production actually sits.
|
|
@@ -489,7 +497,7 @@ At the throughput numbers the benchmark shows, a small set of concurrent connect
|
|
|
489
497
|
|
|
490
498
|
### Response writing
|
|
491
499
|
|
|
492
|
-
Both servers support the same fundamental response shapes: file bodies through `IO.copy_stream`, non-blocking writes with `wait_writable(timeout)` on EAGAIN, and chunked transfer encoding for enumerable bodies without a known length. Puma uses `TCP_CORK` on Linux around HTTP/1.1 responses. Raptor corks
|
|
500
|
+
Both servers support the same fundamental response shapes: file bodies through `IO.copy_stream`, non-blocking writes with `wait_writable(timeout)` on EAGAIN, and chunked transfer encoding for enumerable bodies without a known length. Puma uses `TCP_CORK` on Linux around HTTP/1.1 responses. Raptor only corks a response that will close the connection when its body cannot already be emitted with the headers in one `writev` call.
|
|
493
501
|
|
|
494
502
|
On the HTTP/1.1 path, Raptor has a small `writev(2)` wrapper (`Raptor::VectorIO`) that can scatter-write the status line, headers, and body in one call for non-chunked responses. Puma sends the same content over multiple `write` calls batched by `TCP_CORK` at the kernel; Raptor groups the buffers in userspace and lets `writev` handle partial writes when necessary.
|
|
495
503
|
|
|
@@ -501,7 +509,7 @@ Around the response boundary, HTTP/1.1 also amortises the common per-request all
|
|
|
501
509
|
|
|
502
510
|
### Keep-alive request by request
|
|
503
511
|
|
|
504
|
-
To make the keep-alive
|
|
512
|
+
To make the keep-alive paths concrete, here is one possible timing for three requests on the same connection. The second request is already buffered when the first response completes; the third arrives after the non-blocking check. Drawn separately so the participant columns stay wide enough to read.
|
|
505
513
|
|
|
506
514
|
**Puma, three keep-alive requests:**
|
|
507
515
|
|
|
@@ -541,6 +549,7 @@ sequenceDiagram
|
|
|
541
549
|
autonumber
|
|
542
550
|
participant Client
|
|
543
551
|
participant RS as Server thread
|
|
552
|
+
participant RR as Reactor thread
|
|
544
553
|
participant RP as App thread
|
|
545
554
|
|
|
546
555
|
Note over Client,RP: Request 1, initial
|
|
@@ -549,21 +558,21 @@ sequenceDiagram
|
|
|
549
558
|
RS->>RP: push proc to thread pool
|
|
550
559
|
RP->>Client: response 1
|
|
551
560
|
|
|
552
|
-
Note over Client,RP: Request 2,
|
|
553
|
-
RP
|
|
554
|
-
|
|
561
|
+
Note over Client,RP: Request 2, already buffered
|
|
562
|
+
Client->>RP: request bytes already waiting
|
|
563
|
+
RP->>RP: wait_readable(0) returns true
|
|
555
564
|
RP->>RP: parse inline on app thread
|
|
556
565
|
RP->>Client: response 2
|
|
557
566
|
|
|
558
|
-
Note over Client,RP: Request 3,
|
|
559
|
-
RP
|
|
560
|
-
|
|
561
|
-
Client->>
|
|
562
|
-
|
|
567
|
+
Note over Client,RP: Request 3, not yet available
|
|
568
|
+
RP->>RP: wait_readable(0) returns false
|
|
569
|
+
RP->>RR: reactor.persist
|
|
570
|
+
Client->>RR: request bytes
|
|
571
|
+
RR->>RP: dispatch back to thread pool
|
|
563
572
|
RP->>Client: response 3
|
|
564
573
|
```
|
|
565
574
|
|
|
566
|
-
In this timing,
|
|
575
|
+
In this timing, both servers keep Request 2 inline and return Request 3 to the reactor. Their implementations differ, but the occupancy rule is the same: use the current app thread for bytes that are ready, not to wait for future bytes.
|
|
567
576
|
|
|
568
577
|
## Part IV: What Raptor's design buys you
|
|
569
578
|
|
|
@@ -71,8 +71,8 @@ module Rackup
|
|
|
71
71
|
cli_defaults = ::Raptor::CLI::DEFAULT_OPTIONS
|
|
72
72
|
config_path = options[:Config] || ::Raptor::CLI.default_config_path
|
|
73
73
|
config = config_path ? ::Raptor::CLI.load_config_file(config_path) : {}
|
|
74
|
-
threads = (options[:Threads] || config[:threads] || cli_defaults[:threads]).to_i
|
|
75
|
-
max_threads = options[:MaxThreads] || config.fetch(:max_threads, cli_defaults[:max_threads])
|
|
74
|
+
threads = (options[:Threads] || ENV["RAPTOR_THREADS"] || config[:threads] || cli_defaults[:threads]).to_i
|
|
75
|
+
max_threads = options[:MaxThreads] || ENV["RAPTOR_MAX_THREADS"] || config.fetch(:max_threads, cli_defaults[:max_threads])
|
|
76
76
|
|
|
77
77
|
result = {
|
|
78
78
|
binds: if options[:Host] || options[:Port]
|
|
@@ -82,9 +82,12 @@ module Rackup
|
|
|
82
82
|
end,
|
|
83
83
|
socket_backlog: (config[:socket_backlog] || cli_defaults[:socket_backlog]).to_i,
|
|
84
84
|
drain_accept_queue: config.key?(:drain_accept_queue) ? config[:drain_accept_queue] : cli_defaults[:drain_accept_queue],
|
|
85
|
-
workers: (options[:Workers] || config[:workers] || Concurrent.available_processor_count).to_i,
|
|
85
|
+
workers: (options[:Workers] || ENV["RAPTOR_WORKERS"] || config[:workers] || Concurrent.available_processor_count).to_i,
|
|
86
86
|
threads: threads,
|
|
87
87
|
max_threads: ::Raptor::CLI.parse_max_threads(max_threads, threads: threads),
|
|
88
|
+
cpu_affinity: config.key?(:cpu_affinity) ? config[:cpu_affinity] : cli_defaults[:cpu_affinity],
|
|
89
|
+
clean_thread_locals: config.key?(:clean_thread_locals) ? config[:clean_thread_locals] : cli_defaults[:clean_thread_locals],
|
|
90
|
+
clean_fiber_locals: config.key?(:clean_fiber_locals) ? config[:clean_fiber_locals] : cli_defaults[:clean_fiber_locals],
|
|
88
91
|
app: app
|
|
89
92
|
}
|
|
90
93
|
result[:rackup] = config[:rackup] if config.key?(:rackup)
|
|
@@ -97,7 +100,13 @@ module Rackup
|
|
|
97
100
|
result[:worker_timeout] = (config[:worker_timeout] || cli_defaults[:worker_timeout]).to_i
|
|
98
101
|
result[:worker_drain_timeout] = (config[:worker_drain_timeout] || cli_defaults[:worker_drain_timeout]).to_i
|
|
99
102
|
result[:worker_shutdown_timeout] = (config[:worker_shutdown_timeout] || cli_defaults[:worker_shutdown_timeout]).to_i
|
|
103
|
+
result[:refork_after] = config.fetch(:refork_after, cli_defaults[:refork_after])
|
|
104
|
+
result[:before_fork] = config.fetch(:before_fork, cli_defaults[:before_fork])
|
|
105
|
+
result[:before_worker_boot] = config.fetch(:before_worker_boot, cli_defaults[:before_worker_boot])
|
|
106
|
+
result[:before_worker_shutdown] = config.fetch(:before_worker_shutdown, cli_defaults[:before_worker_shutdown])
|
|
107
|
+
result[:before_refork] = config.fetch(:before_refork, cli_defaults[:before_refork])
|
|
100
108
|
result[:stats_file] = config.key?(:stats_file) ? config[:stats_file] : cli_defaults[:stats_file]
|
|
109
|
+
result[:control_url] = config[:control_url] if config.key?(:control_url)
|
|
101
110
|
result[:pid_file] = config[:pid_file] if config.key?(:pid_file)
|
|
102
111
|
result[:stdout_file] = config[:stdout_file] if config.key?(:stdout_file)
|
|
103
112
|
result[:stderr_file] = config[:stderr_file] if config.key?(:stderr_file)
|
data/lib/raptor/cli.rb
CHANGED
|
@@ -35,6 +35,9 @@ module Raptor
|
|
|
35
35
|
workers: DEFAULT_WORKER_COUNT,
|
|
36
36
|
threads: 3,
|
|
37
37
|
max_threads: Float::INFINITY,
|
|
38
|
+
cpu_affinity: false,
|
|
39
|
+
clean_thread_locals: true,
|
|
40
|
+
clean_fiber_locals: false,
|
|
38
41
|
rackup: "config.ru",
|
|
39
42
|
chdir: nil,
|
|
40
43
|
environment: nil,
|
|
@@ -48,7 +51,7 @@ module Raptor
|
|
|
48
51
|
http1: {
|
|
49
52
|
ractors: nil,
|
|
50
53
|
persistent_data_timeout: 65,
|
|
51
|
-
max_keepalive_requests:
|
|
54
|
+
max_keepalive_requests: 1000,
|
|
52
55
|
},
|
|
53
56
|
http2: {
|
|
54
57
|
ractors: nil,
|
|
@@ -64,6 +67,7 @@ module Raptor
|
|
|
64
67
|
before_worker_shutdown: [].freeze,
|
|
65
68
|
before_refork: [].freeze,
|
|
66
69
|
stats_file: "tmp/raptor.json",
|
|
70
|
+
control_url: nil,
|
|
67
71
|
pid_file: nil,
|
|
68
72
|
stdout_file: nil,
|
|
69
73
|
stderr_file: nil,
|
|
@@ -151,6 +155,7 @@ module Raptor
|
|
|
151
155
|
end
|
|
152
156
|
|
|
153
157
|
apply_config_file(extract_config_path(argv) || self.class.default_config_path)
|
|
158
|
+
apply_environment
|
|
154
159
|
|
|
155
160
|
@parser = create_parser
|
|
156
161
|
@parser.parse!(argv)
|
|
@@ -232,6 +237,18 @@ module Raptor
|
|
|
232
237
|
end
|
|
233
238
|
end
|
|
234
239
|
|
|
240
|
+
# Applies worker and thread settings from the environment. Explicit CLI
|
|
241
|
+
# options are parsed afterwards and take precedence.
|
|
242
|
+
#
|
|
243
|
+
# @return [void]
|
|
244
|
+
#
|
|
245
|
+
# @rbs () -> void
|
|
246
|
+
def apply_environment
|
|
247
|
+
@options[:workers] = Integer(ENV["RAPTOR_WORKERS"], 10) if ENV["RAPTOR_WORKERS"]
|
|
248
|
+
@options[:threads] = Integer(ENV["RAPTOR_THREADS"], 10) if ENV["RAPTOR_THREADS"]
|
|
249
|
+
@options[:max_threads] = ENV["RAPTOR_MAX_THREADS"] if ENV["RAPTOR_MAX_THREADS"]
|
|
250
|
+
end
|
|
251
|
+
|
|
235
252
|
# Creates the OptionParser instance with all supported command-line options.
|
|
236
253
|
#
|
|
237
254
|
# @return [OptionParser] configured option parser
|
|
@@ -273,6 +290,18 @@ module Raptor
|
|
|
273
290
|
@options[:max_threads] = num
|
|
274
291
|
end
|
|
275
292
|
|
|
293
|
+
opts.on("--[no-]cpu-affinity", "Pin each worker process to a CPU (default: off)") do |bool|
|
|
294
|
+
@options[:cpu_affinity] = bool
|
|
295
|
+
end
|
|
296
|
+
|
|
297
|
+
opts.on("--[no-]clean-thread-locals", "Clear application thread locals after each request (default: on)") do |bool|
|
|
298
|
+
@options[:clean_thread_locals] = bool
|
|
299
|
+
end
|
|
300
|
+
|
|
301
|
+
opts.on("--[no-]clean-fiber-locals", "Process each request in a fresh Fiber (default: off)") do |bool|
|
|
302
|
+
@options[:clean_fiber_locals] = bool
|
|
303
|
+
end
|
|
304
|
+
|
|
276
305
|
opts.on("-C", "--chdir PATH", String, "Change to PATH before loading the Rack application (default: none)") do |path|
|
|
277
306
|
@options[:chdir] = path
|
|
278
307
|
end
|
|
@@ -309,7 +338,7 @@ module Raptor
|
|
|
309
338
|
@options[:http1][:persistent_data_timeout] = timeout
|
|
310
339
|
end
|
|
311
340
|
|
|
312
|
-
opts.on("--http1-max-keepalive-requests NUM", Integer, "Maximum HTTP/1.1 requests per keep-alive connection (default:
|
|
341
|
+
opts.on("--http1-max-keepalive-requests NUM", Integer, "Maximum HTTP/1.1 requests per keep-alive connection (default: 1000)") do |num|
|
|
313
342
|
@options[:http1][:max_keepalive_requests] = num
|
|
314
343
|
end
|
|
315
344
|
|
|
@@ -345,6 +374,10 @@ module Raptor
|
|
|
345
374
|
@options[:stats_file] = path
|
|
346
375
|
end
|
|
347
376
|
|
|
377
|
+
opts.on("--control-url URI", String, "Serve cluster stats on a unix:// URI (default: off)") do |uri|
|
|
378
|
+
@options[:control_url] = uri
|
|
379
|
+
end
|
|
380
|
+
|
|
348
381
|
opts.on("--pid-file PATH", String, "PID file path (default: none)") do |path|
|
|
349
382
|
@options[:pid_file] = path
|
|
350
383
|
end
|