kino 0.5.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +46 -0
- data/Cargo.lock +92 -1
- data/README.md +101 -44
- data/doc/architecture.md +54 -3
- data/doc/benchmarks.md +120 -26
- data/ext/kino/Cargo.toml +8 -3
- data/ext/kino/src/env_strings.rs +162 -10
- data/ext/kino/src/lib.rs +11 -7
- data/ext/kino/src/queue.rs +59 -43
- data/ext/kino/src/registry.rs +4 -0
- data/ext/kino/src/request.rs +150 -69
- data/ext/kino/src/response.rs +39 -5
- data/ext/kino/src/server.rs +647 -31
- data/ext/kino/src/tls.rs +119 -9
- data/lib/kino/configuration.rb +6 -0
- data/lib/kino/server.rb +2 -0
- data/lib/kino/templates/kino.rb.tt +5 -0
- data/lib/kino/version.rb +1 -1
- data/lib/kino/worker.rb +14 -11
- data/sig/kino.rbs +1 -0
- metadata +15 -1
data/doc/benchmarks.md
CHANGED
|
@@ -17,9 +17,21 @@ the deployment most apps run today.
|
|
|
17
17
|
9R14 (Genoa), 16 GB RAM, Amazon Linux 2023, kernel 6.18. A realistic
|
|
18
18
|
app-server size, deliberately: nobody provisions a 32-core box per
|
|
19
19
|
app process.
|
|
20
|
-
- Toolchain built on the box via mise: Ruby 4.0.
|
|
21
|
-
`RUBY_YJIT_ENABLE=1` for every server), Rust
|
|
20
|
+
- Toolchain built on the box via mise: Ruby 4.0.6 (**YJIT enabled**,
|
|
21
|
+
`RUBY_YJIT_ENABLE=1` for every server), Rust stable, Kino compiled in
|
|
22
22
|
the release profile.
|
|
23
|
+
- **2026-09 full re-measurement** (a fresh c7a.2xlarge): every number in
|
|
24
|
+
this document and the README was re-run on Ruby 4.0.6 / kernel 6.18 /
|
|
25
|
+
Puma 8.0.2, adding the sharded-I/O and HTTP/2 studies. Kino's numbers
|
|
26
|
+
reproduced within 1-3% across the board—including the /io slot
|
|
27
|
+
ceiling and the arena balloon, both intact. The one real shift: Puma
|
|
28
|
+
8.0.2 is faster than 7.x (142k plaintext, 61k /cpu), narrowing
|
|
29
|
+
ractor mode's /cpu lead to +25% (was +34%). Reversed-boot-order
|
|
30
|
+
re-runs reproduced within ~1%. A methodology trap worth recording:
|
|
31
|
+
a 5-second single-endpoint warmup understates memory badly (81 MB
|
|
32
|
+
where the full battery shows 137 MB) and hides the arena balloon
|
|
33
|
+
entirely—memory is only comparable after the full endpoint battery,
|
|
34
|
+
and /io numbers are only comparable at equal slot counts.
|
|
23
35
|
- Load generator: wrk 4.2 on the same host, 8-second windows, 64
|
|
24
36
|
connections (`bench/run.sh 8 64`). Same-host load generation costs
|
|
25
37
|
both sides CPU equally; we verified the generator was not the
|
|
@@ -249,14 +261,17 @@ visible.
|
|
|
249
261
|
|
|
250
262
|
| config | RSS | PSS |
|
|
251
263
|
|---|---:|---:|
|
|
252
|
-
| Kino :ractor 8×1 (default) |
|
|
253
|
-
| Kino lanes 8×1 |
|
|
254
|
-
| Kino :ractor 8×3 |
|
|
264
|
+
| Kino :ractor 8×1 (default) | 137 | **135** |
|
|
265
|
+
| Kino lanes 8×1 | 128 | **126** |
|
|
266
|
+
| Kino :ractor 8×3 | 170 | **168** |
|
|
255
267
|
| Kino :threaded 8×3 (`MALLOC_ARENA_MAX=2`) | 109 | **107** |
|
|
256
|
-
| Kino :threaded 8×3 (no arena cap) |
|
|
257
|
-
| Puma cluster 8×3 | 1,
|
|
268
|
+
| Kino :threaded 8×3 (no arena cap) | 672 | **670**¹ |
|
|
269
|
+
| Puma cluster 8×3 | 1,216 | **1,072** |
|
|
258
270
|
|
|
259
|
-
|
|
271
|
+
(2026-09 re-measure; the 2026-06 numbers reproduced within a few MB on
|
|
272
|
+
every row—the arena-capped threaded row to the megabyte.)
|
|
273
|
+
|
|
274
|
+
The tiny app is ~8× lighter than the cluster in ractor mode, ~10× in
|
|
260
275
|
arena-capped threaded mode. RSS ≈ PSS for every Kino row (one process,
|
|
261
276
|
nothing to share) and within ~12% for Puma here: a trivial app has almost
|
|
262
277
|
no shared state, so Puma's footprint is ~1,051 MB of *private* per-worker
|
|
@@ -280,18 +295,19 @@ Here copy-on-write **does** matter, which is exactly why PSS is mandatory:
|
|
|
280
295
|
|
|
281
296
|
| config | RSS | PSS |
|
|
282
297
|
|---|---:|---:|
|
|
283
|
-
| Kino :threaded (one process) | 97 | **
|
|
284
|
-
| Puma cluster 8×
|
|
298
|
+
| Kino :threaded (one process) | 97 | **95** |
|
|
299
|
+
| Puma cluster 8×5 (preload) | 813 | **405** |
|
|
300
|
+
| Puma cluster 8×5 (no preload) | 824 | **646** |
|
|
285
301
|
|
|
286
302
|
Puma serves the same Rails framework from 8 forks that share it
|
|
287
|
-
copy-on-write; RSS counts that shared framework once per worker (
|
|
288
|
-
PSS counts it once (
|
|
303
|
+
copy-on-write; RSS counts that shared framework once per worker (813 MB),
|
|
304
|
+
PSS counts it once (405 MB). The fair ratio is **~4×**, not the ~8× a
|
|
289
305
|
naive RSS sum reports—this is the correction that prompted the whole
|
|
290
|
-
re-measure.
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
not the live object heap.
|
|
306
|
+
re-measure. The 2026-09 run also split out preload: it now saves a
|
|
307
|
+
third (405 vs 646 MB PSS)—worth turning on—but even preloaded, Ruby's
|
|
308
|
+
GC dirties heap pages and breaks copy-on-write, so each worker keeps a
|
|
309
|
+
large private heap. That is why "CoW should make a fork cluster nearly
|
|
310
|
+
free" is only half true—it shares the code, not the live object heap.
|
|
295
311
|
|
|
296
312
|
## Run-to-run variance (a.k.a. "is this a regression?")
|
|
297
313
|
|
|
@@ -376,27 +392,46 @@ crash semantics, stealing fairness, and drain behavior have spec
|
|
|
376
392
|
coverage but not production mileage. (On loopback-bound macOS, lanes
|
|
377
393
|
lose a few percent instead; see the secondary table below.)
|
|
378
394
|
|
|
395
|
+
## Sharded I/O (`io_shards true`)
|
|
396
|
+
|
|
397
|
+
`bench/studies.sh 8 64 shards`, ractor 8×3, 2026-09 reference box:
|
|
398
|
+
|
|
399
|
+
| case (/plaintext) | req/s |
|
|
400
|
+
|---|---:|
|
|
401
|
+
| shared tokio runtime (baseline) | 193,461 |
|
|
402
|
+
| io_shards, default shard count | 195,436 |
|
|
403
|
+
| io_shards, `io_threads 2` | 170,509 |
|
|
404
|
+
| io_shards, `io_threads 4` | 196,298 |
|
|
405
|
+
| io_shards, `io_threads 8` | 199,657 |
|
|
406
|
+
|
|
407
|
+
On 8 cores the shards buy +1-3%, best at `io_threads 8` (the
|
|
408
|
+
half-the-cores default is close behind; 2 shards choke on accept
|
|
409
|
+
handoff). The design removes work-stealing and cross-thread wakeups,
|
|
410
|
+
so the win scales with scheduler contention—expect more on boxes with
|
|
411
|
+
more cores and connections, and measure on your own core count.
|
|
412
|
+
|
|
379
413
|
## Logging costs
|
|
380
414
|
|
|
381
415
|
Measured at full plaintext saturation (one log line per request—rates
|
|
382
416
|
that no real deployment logs at; treat these as worst-case ceilings, not
|
|
383
417
|
typical costs):
|
|
384
418
|
|
|
385
|
-
| case (8×3, same session) | req/s |
|
|
419
|
+
| case (8×3, same session, 2026-09) | req/s |
|
|
386
420
|
|---|---:|
|
|
387
|
-
| threaded, no logging |
|
|
388
|
-
| threaded, `log_requests true` (native access log) |
|
|
389
|
-
| ractor, access log off / on |
|
|
390
|
-
| app logs 1 line/req via shared `::Logger` (file) | **
|
|
391
|
-
| app logs 1 line/req via `Kino::Logger` (file) | **
|
|
421
|
+
| threaded, no logging | 213,166 |
|
|
422
|
+
| threaded, `log_requests true` (native access log) | 173,501 (−19%) |
|
|
423
|
+
| ractor, access log off / on | 191,152 / 163,777 (−14%) |
|
|
424
|
+
| app logs 1 line/req via shared `::Logger` (file) | **90,175** |
|
|
425
|
+
| app logs 1 line/req via `Kino::Logger` (file) | **154,653 (1.7×)** |
|
|
392
426
|
|
|
393
427
|
The shared-`::Logger` cost is the mutex: 24 worker threads serialize
|
|
394
428
|
through one lock plus a write syscall per line. `Kino::Logger` hands the
|
|
395
429
|
formatted line to a lock-free channel and returns—the remaining cost vs
|
|
396
430
|
not logging at all is Ruby-side formatting, which no device can remove.
|
|
397
|
-
(
|
|
398
|
-
|
|
399
|
-
|
|
431
|
+
(The multiple moves with the environment: 2.4× on the 2026-06 box,
|
|
432
|
+
1.7× on the 2026-09 one, 8.5× under Docker, where overlay-fs write
|
|
433
|
+
latency punished the synchronous logger hardest. The ranking is
|
|
434
|
+
environment-independent; the multiple is not.)
|
|
400
435
|
|
|
401
436
|
One trade-off worth knowing: the sink **never blocks** request threads,
|
|
402
437
|
so at absurd rates against a slow disk it drops lines once its 8192-line
|
|
@@ -408,6 +443,65 @@ Puma comparison note: request logging is opt-in there too (`--quiet` is
|
|
|
408
443
|
the default, `-v/--log-requests` enables it)—Kino's default-off
|
|
409
444
|
`log_requests` matches the ecosystem's standard behavior.
|
|
410
445
|
|
|
446
|
+
## HTTP/2
|
|
447
|
+
|
|
448
|
+
`bench/h2.sh`, Linux only (on macOS run it under Docker; the 2026-09
|
|
449
|
+
numbers below are from the c7a.2xlarge reference box). Every lane is
|
|
450
|
+
measured with h2load so the generator is identical everywhere: h2
|
|
451
|
+
lanes run 8 connections × 8 concurrent streams, h1 lanes 64
|
|
452
|
+
connections—the same total in-flight. Servers without native h2 get
|
|
453
|
+
the standard pattern instead: nginx terminating h2 and proxying
|
|
454
|
+
HTTP/1.1 upstream over keep-alive. One labeled run (kino ractor 8×3,
|
|
455
|
+
falcon `--count 8`, puma `-w 8 -t 3:3`, 5 s/lane); re-run the whole
|
|
456
|
+
script for close calls, per the variance section.
|
|
457
|
+
|
|
458
|
+
| target (h2 unless noted) | /plaintext | /10k | /big-cookie | /upload (64 KB) |
|
|
459
|
+
|---|---:|---:|---:|---:|
|
|
460
|
+
| kino h2c | 207,461 | 150,105 | 195,093 | 26,304 |
|
|
461
|
+
| kino h1 cleartext (same boot) | 116,082 | 97,634 | 110,614 | 24,338 |
|
|
462
|
+
| kino h2 TLS | 164,044 | 116,617 | 154,301 | 19,226 |
|
|
463
|
+
| kino h1 TLS (same boot) | 80,789 | 69,942 | 76,654 | 17,210 |
|
|
464
|
+
| falcon TLS (native h2) | 55,637 | 37,541 | 49,151 | 18,838 |
|
|
465
|
+
| nginx h2 → puma h1 | 81,234 | 57,891 | 51,332 | 1,247 |
|
|
466
|
+
| nginx h2 → kino h1 | 109,219 | 66,634 | 54,996 | 1,217 |
|
|
467
|
+
|
|
468
|
+
What the numbers say:
|
|
469
|
+
|
|
470
|
+
- **Native h2 beats h1 on the same server by +79% cleartext and +103%
|
|
471
|
+
over TLS** on /plaintext: the same 64 in-flight requests ride 8
|
|
472
|
+
connections instead of 64, so frames batch into fewer, larger
|
|
473
|
+
syscalls—and TLS amplifies it, since h1's 64 connections each pay
|
|
474
|
+
crypto per record. The `/big-cookie` lane (a ~2 KB cookie per
|
|
475
|
+
request) shows HPACK on top: the cookie crosses the wire once per
|
|
476
|
+
connection, not once per request, and holds 94% of bare-plaintext
|
|
477
|
+
throughput where h1 loses 5%.
|
|
478
|
+
- **Native h2 beats proxied h2 by +50%** with the *same backend*: the
|
|
479
|
+
nginx→kino-h1 lane is the proxy-cost control, and the extra hop,
|
|
480
|
+
re-parse, and re-serialize cost ~55k req/s on /plaintext.
|
|
481
|
+
- **Uploads run at h1 parity**—but only after a fix this lane
|
|
482
|
+
caught: h2 delivers bodies as 16 KB DATA frames, and `read_body`
|
|
483
|
+
originally crossed the GVL once per chunk, halving upload
|
|
484
|
+
throughput. It now drains every queued chunk per crossing
|
|
485
|
+
(doc/architecture.md), and a knob sweep over hyper's h2 codec
|
|
486
|
+
(frame size, adaptive/bigger windows) moved nothing afterwards.
|
|
487
|
+
nginx's h2 upload collapse (~1.5k) is its default-config
|
|
488
|
+
request-body flow control; tune `http2_body_preread_size`/buffering
|
|
489
|
+
before drawing conclusions there.
|
|
490
|
+
- falcon lands at roughly a third of kino-h2-TLS on fast handlers and
|
|
491
|
+
slightly behind on uploads. Single-run caveat: the nginx lanes
|
|
492
|
+
showed ±20% swings between runs in the Docker environment; on the
|
|
493
|
+
reference box the kino-vs-kino and kino-vs-proxy ratios reproduced
|
|
494
|
+
across runs, the nginx lanes remain the noisiest.
|
|
495
|
+
- **Header-value interning** (user-agent, accept-*, sec-ch-*: one
|
|
496
|
+
frozen string instead of a fresh allocation per request) was
|
|
497
|
+
measured with a realistic 11-header browser set on /plaintext,
|
|
498
|
+
using the within-boot header cost (bare vs with-headers) as the
|
|
499
|
+
drift-resistant metric: over h2 that cost fell from ~16% to ~12-13%
|
|
500
|
+
(~+3-5% throughput on the headers lane); h1's smaller header cost
|
|
501
|
+
stayed within noise. The effect the tiny app understates: 6-8 fewer
|
|
502
|
+
string allocations per request is GC pressure a real app feels more
|
|
503
|
+
than this one does.
|
|
504
|
+
|
|
411
505
|
## Hot-path notes
|
|
412
506
|
|
|
413
507
|
For the curious, the dispatch-path work behind the numbers: a try-pop
|
data/ext/kino/Cargo.toml
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "kino"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.6.0"
|
|
4
4
|
edition = "2021"
|
|
5
5
|
authors = ["Yaroslav Markin <yaroslav@markin.net>"]
|
|
6
6
|
license = "MIT"
|
|
@@ -21,8 +21,8 @@ smallvec = "1"
|
|
|
21
21
|
lru = "0.18"
|
|
22
22
|
mimalloc = { version = "0.1", default-features = false }
|
|
23
23
|
tokio = { version = "1.45", features = ["rt-multi-thread", "net", "time", "sync", "io-util", "macros"] }
|
|
24
|
-
hyper = { version = "1.6", features = ["http1", "server"] }
|
|
25
|
-
hyper-util = { version = "0.1", features = ["server", "tokio", "http1"] }
|
|
24
|
+
hyper = { version = "1.6", features = ["http1", "http2", "server"] }
|
|
25
|
+
hyper-util = { version = "0.1", features = ["server", "server-auto", "tokio", "http1", "http2"] }
|
|
26
26
|
http = "1"
|
|
27
27
|
http-body-util = "0.1"
|
|
28
28
|
bytes = "1"
|
|
@@ -49,3 +49,8 @@ rb-sys-env = "0.2.2"
|
|
|
49
49
|
# undefined (the host process provides them at runtime); feature
|
|
50
50
|
# unification turns this on only for `cargo test`.
|
|
51
51
|
rb-sys = { version = "0.9", features = ["link-ruby"] }
|
|
52
|
+
# The protocol tests drive the server end of a duplex pipe with hyper's
|
|
53
|
+
# own client, and pause the clock to test timeouts; feature unification
|
|
54
|
+
# enables both only for `cargo test`.
|
|
55
|
+
hyper = { version = "1.6", features = ["client"] }
|
|
56
|
+
tokio = { version = "1.45", features = ["test-util"] }
|
data/ext/kino/src/env_strings.rs
CHANGED
|
@@ -21,6 +21,7 @@ use parking_lot::{Mutex, RwLock};
|
|
|
21
21
|
/// These maps are probed several times per request; ahash beats the
|
|
22
22
|
/// DoS-resistant default since every key here is our own static data.
|
|
23
23
|
type HashMap<K, V> = std::collections::HashMap<K, V, ahash::RandomState>;
|
|
24
|
+
type HashSet<K> = std::collections::HashSet<K, ahash::RandomState>;
|
|
24
25
|
|
|
25
26
|
pub struct EnvStrings {
|
|
26
27
|
// keys
|
|
@@ -44,17 +45,25 @@ pub struct EnvStrings {
|
|
|
44
45
|
pub https: Opaque<RString>,
|
|
45
46
|
pub http10: Opaque<RString>,
|
|
46
47
|
pub http11: Opaque<RString>,
|
|
48
|
+
pub http2: Opaque<RString>,
|
|
47
49
|
pub methods: HashMap<&'static str, Opaque<RString>>,
|
|
48
50
|
/// lowercase header name -> frozen "HTTP_<UPPER>" key
|
|
49
51
|
pub header_names: HashMap<&'static str, Opaque<RString>>,
|
|
50
|
-
/// Host-header bytes -> frozen
|
|
51
|
-
///
|
|
52
|
-
///
|
|
52
|
+
/// Host-header or :authority bytes -> frozen host values, and peer
|
|
53
|
+
/// IP -> frozen REMOTE_ADDR value. Real traffic has low cardinality
|
|
54
|
+
/// on both, so these kill 3 string allocations per request.
|
|
53
55
|
/// LRU-bounded: entries are BoxValue-rooted (registered with the GC on
|
|
54
56
|
/// insert, UNregistered on eviction-drop), so a rotating-host attack
|
|
55
57
|
/// recycles cache slots instead of leaking immortal strings.
|
|
56
|
-
pub hosts: Mutex<lru::LruCache<Vec<u8>,
|
|
58
|
+
pub hosts: Mutex<lru::LruCache<Vec<u8>, HostEntry, ahash::RandomState>>,
|
|
57
59
|
pub addrs: Mutex<lru::LruCache<IpAddr, CachedStr, ahash::RandomState>>,
|
|
60
|
+
/// Interned values of low-cardinality headers (see
|
|
61
|
+
/// [`INTERNABLE_VALUES`]): value bytes -> frozen RString, shared
|
|
62
|
+
/// across headers that happen to carry the same bytes. Same rooting
|
|
63
|
+
/// and locking contract as `hosts`/`addrs`.
|
|
64
|
+
pub values: Mutex<lru::LruCache<Vec<u8>, CachedStr, ahash::RandomState>>,
|
|
65
|
+
/// The names whose values go through the `values` cache.
|
|
66
|
+
pub internable: HashSet<&'static str>,
|
|
58
67
|
/// Ractor-shareable defaults provided by the Ruby layer at boot:
|
|
59
68
|
/// the frozen rack.errors writer and the frozen null rack.input.
|
|
60
69
|
pub errors_stream: RwLock<Option<Opaque<Value>>>,
|
|
@@ -63,6 +72,53 @@ pub struct EnvStrings {
|
|
|
63
72
|
|
|
64
73
|
const HOST_CACHE_CAP: usize = 256;
|
|
65
74
|
const ADDR_CACHE_CAP: usize = 1024;
|
|
75
|
+
const VALUE_CACHE_CAP: usize = 512;
|
|
76
|
+
|
|
77
|
+
/// Values longer than this are never interned: past it the memcpy into
|
|
78
|
+
/// a fresh Ruby string is cheap relative to the bytes themselves, and
|
|
79
|
+
/// unbounded keys would let one client fill the cache with garbage.
|
|
80
|
+
const VALUE_INTERN_MAX_LEN: usize = 512;
|
|
81
|
+
|
|
82
|
+
/// Headers whose values are effectively enums or per-install constants
|
|
83
|
+
/// (a browser resends the same UA, accept-*, and sec-ch-* on every
|
|
84
|
+
/// request), so caching kills an allocation + copy per header per
|
|
85
|
+
/// request: the env-side analogue of what HPACK does on the wire.
|
|
86
|
+
/// Deliberately absent: `cookie` and `authorization` (per-user
|
|
87
|
+
/// cardinality would churn the cache, and secrets should not outlive
|
|
88
|
+
/// their request in an evict-to-free cache), `referer`/`x-request-id`
|
|
89
|
+
/// and friends (unbounded cardinality).
|
|
90
|
+
const INTERNABLE_VALUES: &[&str] = &[
|
|
91
|
+
"user-agent",
|
|
92
|
+
"accept",
|
|
93
|
+
"accept-encoding",
|
|
94
|
+
"accept-language",
|
|
95
|
+
"cache-control",
|
|
96
|
+
"dnt",
|
|
97
|
+
"origin",
|
|
98
|
+
"pragma",
|
|
99
|
+
"priority",
|
|
100
|
+
"sec-ch-ua",
|
|
101
|
+
"sec-ch-ua-mobile",
|
|
102
|
+
"sec-ch-ua-platform",
|
|
103
|
+
"sec-fetch-dest",
|
|
104
|
+
"sec-fetch-mode",
|
|
105
|
+
"sec-fetch-site",
|
|
106
|
+
"sec-fetch-user",
|
|
107
|
+
"upgrade-insecure-requests",
|
|
108
|
+
"x-requested-with",
|
|
109
|
+
];
|
|
110
|
+
|
|
111
|
+
/// One hosts-cache entry: the frozen SERVER_NAME/SERVER_PORT pair, plus
|
|
112
|
+
/// the frozen full authority ("host[:port]" as sent) used as the
|
|
113
|
+
/// HTTP_HOST value for requests that carry the name in the URI (the h2
|
|
114
|
+
/// :authority pseudo-header) rather than a Host header. Lazily filled:
|
|
115
|
+
/// Host-header entries and the NUL-prefixed socket-fallback entries
|
|
116
|
+
/// never allocate it.
|
|
117
|
+
pub struct HostEntry {
|
|
118
|
+
name: CachedStr,
|
|
119
|
+
port: CachedStr,
|
|
120
|
+
host: Option<CachedStr>,
|
|
121
|
+
}
|
|
66
122
|
|
|
67
123
|
/// A frozen RString rooted via BoxValue (GC-registered address; unregisters
|
|
68
124
|
/// on Drop, so LRU eviction actually frees the string).
|
|
@@ -84,6 +140,16 @@ impl CachedStr {
|
|
|
84
140
|
CachedStr(magnus::value::BoxValue::new(string))
|
|
85
141
|
}
|
|
86
142
|
|
|
143
|
+
/// Header values are bytes on the wire (not guaranteed UTF-8), so
|
|
144
|
+
/// they cache as the same binary strings `str_from_slice` builds on
|
|
145
|
+
/// the uncached path; interning must not change the encoding an
|
|
146
|
+
/// app observes.
|
|
147
|
+
fn new_from_slice(ruby: &Ruby, bytes: &[u8]) -> Self {
|
|
148
|
+
let string = ruby.str_from_slice(bytes);
|
|
149
|
+
string.freeze();
|
|
150
|
+
CachedStr(magnus::value::BoxValue::new(string))
|
|
151
|
+
}
|
|
152
|
+
|
|
87
153
|
fn get(&self) -> RString {
|
|
88
154
|
*self.0
|
|
89
155
|
}
|
|
@@ -142,6 +208,8 @@ const COMMON_HEADERS: &[&str] = &[
|
|
|
142
208
|
"sec-ch-ua-mobile",
|
|
143
209
|
"sec-ch-ua-platform",
|
|
144
210
|
"keep-alive",
|
|
211
|
+
"priority",
|
|
212
|
+
"alt-used",
|
|
145
213
|
];
|
|
146
214
|
|
|
147
215
|
pub fn cgi_name(lower: &str) -> String {
|
|
@@ -194,6 +262,7 @@ pub fn init(ruby: &Ruby) {
|
|
|
194
262
|
https: frozen(ruby, "https"),
|
|
195
263
|
http10: frozen(ruby, "HTTP/1.0"),
|
|
196
264
|
http11: frozen(ruby, "HTTP/1.1"),
|
|
265
|
+
http2: frozen(ruby, "HTTP/2"),
|
|
197
266
|
methods,
|
|
198
267
|
header_names,
|
|
199
268
|
hosts: Mutex::new(lru::LruCache::with_hasher(
|
|
@@ -204,6 +273,11 @@ pub fn init(ruby: &Ruby) {
|
|
|
204
273
|
std::num::NonZeroUsize::new(ADDR_CACHE_CAP).unwrap(),
|
|
205
274
|
ahash::RandomState::new(),
|
|
206
275
|
)),
|
|
276
|
+
values: Mutex::new(lru::LruCache::with_hasher(
|
|
277
|
+
std::num::NonZeroUsize::new(VALUE_CACHE_CAP).unwrap(),
|
|
278
|
+
ahash::RandomState::new(),
|
|
279
|
+
)),
|
|
280
|
+
internable: INTERNABLE_VALUES.iter().copied().collect(),
|
|
207
281
|
errors_stream: RwLock::new(None),
|
|
208
282
|
null_input: RwLock::new(None),
|
|
209
283
|
};
|
|
@@ -249,20 +323,98 @@ pub fn set_host_env(
|
|
|
249
323
|
let s = get();
|
|
250
324
|
let mut hosts = s.hosts.lock();
|
|
251
325
|
let (name, port) = match hosts.get(host) {
|
|
252
|
-
Some(
|
|
326
|
+
Some(entry) => (entry.name.get(), entry.port.get()),
|
|
253
327
|
None => {
|
|
254
328
|
let (name_s, port_n) = make();
|
|
255
|
-
let entry =
|
|
256
|
-
CachedStr::new(ruby, &name_s),
|
|
257
|
-
CachedStr::new(ruby, &port_n.to_string()),
|
|
329
|
+
let entry = HostEntry {
|
|
330
|
+
name: CachedStr::new(ruby, &name_s),
|
|
331
|
+
port: CachedStr::new(ruby, &port_n.to_string()),
|
|
332
|
+
host: None,
|
|
333
|
+
};
|
|
334
|
+
let values = (entry.name.get(), entry.port.get());
|
|
335
|
+
hosts.put(host.to_vec(), entry); // may evict + free an old entry
|
|
336
|
+
values
|
|
337
|
+
}
|
|
338
|
+
};
|
|
339
|
+
env.aset(ruby.get_inner(s.server_name), name)?;
|
|
340
|
+
env.aset(ruby.get_inner(s.server_port), port)?;
|
|
341
|
+
Ok(())
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
/// Set SERVER_NAME/SERVER_PORT *and* HTTP_HOST on `env` from the URI
|
|
345
|
+
/// authority (every h2 request via :authority; also h1 absolute-form).
|
|
346
|
+
/// HTTP_HOST is set here because such requests carry no Host header for
|
|
347
|
+
/// the header loop to surface. Same cache and locking contract as
|
|
348
|
+
/// [`set_host_env`]; an entry first created by a Host header upgrades in
|
|
349
|
+
/// place, gaining the full-authority string on first use.
|
|
350
|
+
pub fn set_authority_env(
|
|
351
|
+
ruby: &Ruby,
|
|
352
|
+
env: magnus::RHash,
|
|
353
|
+
authority: &str,
|
|
354
|
+
make: impl FnOnce() -> (String, u16),
|
|
355
|
+
) -> Result<(), magnus::Error> {
|
|
356
|
+
let s = get();
|
|
357
|
+
let mut hosts = s.hosts.lock();
|
|
358
|
+
let (name, port, host) = match hosts.get_mut(authority.as_bytes()) {
|
|
359
|
+
Some(entry) => {
|
|
360
|
+
if entry.host.is_none() {
|
|
361
|
+
entry.host = Some(CachedStr::new(ruby, authority));
|
|
362
|
+
}
|
|
363
|
+
(
|
|
364
|
+
entry.name.get(),
|
|
365
|
+
entry.port.get(),
|
|
366
|
+
entry.host.as_ref().expect("just filled").get(),
|
|
367
|
+
)
|
|
368
|
+
}
|
|
369
|
+
None => {
|
|
370
|
+
let (name_s, port_n) = make();
|
|
371
|
+
let entry = HostEntry {
|
|
372
|
+
name: CachedStr::new(ruby, &name_s),
|
|
373
|
+
port: CachedStr::new(ruby, &port_n.to_string()),
|
|
374
|
+
host: Some(CachedStr::new(ruby, authority)),
|
|
375
|
+
};
|
|
376
|
+
let values = (
|
|
377
|
+
entry.name.get(),
|
|
378
|
+
entry.port.get(),
|
|
379
|
+
entry.host.as_ref().expect("just built").get(),
|
|
258
380
|
);
|
|
259
|
-
|
|
260
|
-
hosts.put(host.to_vec(), entry); // may evict + free an old pair
|
|
381
|
+
hosts.put(authority.as_bytes().to_vec(), entry);
|
|
261
382
|
values
|
|
262
383
|
}
|
|
263
384
|
};
|
|
264
385
|
env.aset(ruby.get_inner(s.server_name), name)?;
|
|
265
386
|
env.aset(ruby.get_inner(s.server_port), port)?;
|
|
387
|
+
let host_key = *s.header_names.get("host").expect("host is a common header");
|
|
388
|
+
env.aset(ruby.get_inner(host_key), host)?;
|
|
389
|
+
Ok(())
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
/// Set one header's value on `env` under `key`: through the interned
|
|
393
|
+
/// value cache when the header qualifies (low-cardinality name, bounded
|
|
394
|
+
/// length), else a fresh per-request string. The cached aset happens
|
|
395
|
+
/// under the cache lock; see CachedStr's safety contract.
|
|
396
|
+
pub fn set_value_env(
|
|
397
|
+
ruby: &Ruby,
|
|
398
|
+
env: magnus::RHash,
|
|
399
|
+
key: RString,
|
|
400
|
+
name: &str,
|
|
401
|
+
value: &[u8],
|
|
402
|
+
) -> Result<(), magnus::Error> {
|
|
403
|
+
let s = get();
|
|
404
|
+
if value.len() > VALUE_INTERN_MAX_LEN || !s.internable.contains(name) {
|
|
405
|
+
return env.aset(key, ruby.str_from_slice(value));
|
|
406
|
+
}
|
|
407
|
+
let mut values = s.values.lock();
|
|
408
|
+
let cached = match values.get(value) {
|
|
409
|
+
Some(cached) => cached.get(),
|
|
410
|
+
None => {
|
|
411
|
+
let entry = CachedStr::new_from_slice(ruby, value);
|
|
412
|
+
let string = entry.get();
|
|
413
|
+
values.put(value.to_vec(), entry); // may evict + free an old value
|
|
414
|
+
string
|
|
415
|
+
}
|
|
416
|
+
};
|
|
417
|
+
env.aset(key, cached)?;
|
|
266
418
|
Ok(())
|
|
267
419
|
}
|
|
268
420
|
|
data/ext/kino/src/lib.rs
CHANGED
|
@@ -41,8 +41,7 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
|
41
41
|
let native = module.define_module("Native")?;
|
|
42
42
|
native.define_singleton_method("server_start", function!(server::server_start, 1))?;
|
|
43
43
|
native.define_singleton_method("register_worker", function!(server::register_worker, 1))?;
|
|
44
|
-
native.define_singleton_method("
|
|
45
|
-
native.define_singleton_method("take_batch", function!(queue::take_batch, 3))?;
|
|
44
|
+
native.define_singleton_method("worker", function!(queue::worker, 2))?;
|
|
46
45
|
native.define_singleton_method("stop_accepting", function!(server::stop_accepting, 1))?;
|
|
47
46
|
native.define_singleton_method("close_queue", function!(server::close_queue, 1))?;
|
|
48
47
|
native.define_singleton_method("queue_stats", function!(server::queue_stats, 1))?;
|
|
@@ -84,11 +83,15 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
|
84
83
|
native.define_class("PinKeeper", ruby.class_object())?;
|
|
85
84
|
native.define_singleton_method("pin_keeper", function!(server::pin_keeper, 1))?;
|
|
86
85
|
|
|
86
|
+
let worker = native.define_class("Worker", ruby.class_object())?;
|
|
87
|
+
worker.define_method("take_one", method!(queue::Worker::take_one, 0))?;
|
|
88
|
+
worker.define_method("take_batch", method!(queue::Worker::take_batch, 1))?;
|
|
89
|
+
|
|
87
90
|
let request = native.define_class("Request", ruby.class_object())?;
|
|
88
|
-
request.define_method("respond_and_take", method!(queue::respond_and_take,
|
|
91
|
+
request.define_method("respond_and_take", method!(queue::respond_and_take, 5))?;
|
|
89
92
|
request.define_method(
|
|
90
93
|
"respond_and_take_one",
|
|
91
|
-
method!(queue::respond_and_take_one,
|
|
94
|
+
method!(queue::respond_and_take_one, 4),
|
|
92
95
|
)?;
|
|
93
96
|
request.define_method("read_body", method!(Request::read_body, 1))?;
|
|
94
97
|
request.define_method("send_simple", method!(crate::request::respond_simple, 3))?;
|
|
@@ -98,10 +101,11 @@ fn init(ruby: &Ruby) -> Result<(), Error> {
|
|
|
98
101
|
request.define_method("abort", method!(Request::abort, 0))?;
|
|
99
102
|
request.define_method("timing", method!(Request::set_timing, 2))?;
|
|
100
103
|
|
|
101
|
-
// Force-resolve the TypedData class
|
|
102
|
-
// resolves
|
|
103
|
-
// worker ractors is the failure mode we must rule out.
|
|
104
|
+
// Force-resolve the TypedData class caches on the main ractor: magnus
|
|
105
|
+
// resolves them lazily on first wrap, and a racy first resolution from
|
|
106
|
+
// two worker ractors is the failure mode we must rule out.
|
|
104
107
|
let _ = <Request as magnus::TypedData>::class(ruby);
|
|
108
|
+
let _ = <queue::Worker as magnus::TypedData>::class(ruby);
|
|
105
109
|
|
|
106
110
|
// Frozen env key/value cache: built once here (main ractor, GVL held),
|
|
107
111
|
// shared by every worker ractor afterwards.
|
data/ext/kino/src/queue.rs
CHANGED
|
@@ -138,54 +138,73 @@ fn admit(
|
|
|
138
138
|
Ok(env)
|
|
139
139
|
}
|
|
140
140
|
|
|
141
|
-
|
|
141
|
+
/// Per-worker native handle: the registry lookup and slot resolution are
|
|
142
|
+
/// paid once at worker boot instead of on every take. Created inside the
|
|
143
|
+
/// worker thread/ractor that uses it (like Request handles), so ownership
|
|
144
|
+
/// is correct by construction. Holds the server weakly: after teardown the
|
|
145
|
+
/// upgrade fails, which is the same clean shutdown signal the per-take
|
|
146
|
+
/// registry lookup used to give.
|
|
147
|
+
#[magnus::wrap(class = "Kino::Native::Worker", free_immediately)]
|
|
148
|
+
pub struct Worker {
|
|
149
|
+
server: std::sync::Weak<ServerInner>,
|
|
150
|
+
slot: Arc<WorkerSlot>,
|
|
151
|
+
}
|
|
142
152
|
|
|
143
|
-
|
|
153
|
+
/// Resolve a (server, worker) pair into a Worker handle; nil when the
|
|
154
|
+
/// server is already gone (the caller treats that as shutdown).
|
|
155
|
+
pub fn worker(ruby: &Ruby, server_id: u64, worker_id: usize) -> Result<Option<Worker>, Error> {
|
|
144
156
|
let Some(server) = registry::try_get(server_id) else {
|
|
145
|
-
return Ok(None);
|
|
157
|
+
return Ok(None);
|
|
146
158
|
};
|
|
147
159
|
let slot = server.slot(ruby, worker_id)?;
|
|
160
|
+
Ok(Some(Worker {
|
|
161
|
+
server: Arc::downgrade(&server),
|
|
162
|
+
slot,
|
|
163
|
+
}))
|
|
164
|
+
}
|
|
148
165
|
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
166
|
+
impl Worker {
|
|
167
|
+
fn checkout(&self) -> Result<Option<(Arc<ServerInner>, BoxedCtx)>, Error> {
|
|
168
|
+
let Some(server) = self.server.upgrade() else {
|
|
169
|
+
return Ok(None); // server torn down → clean shutdown signal
|
|
170
|
+
};
|
|
153
171
|
|
|
154
|
-
|
|
155
|
-
|
|
172
|
+
// The previous batch is fully answered once the worker comes back.
|
|
173
|
+
self.slot.current.lock().clear();
|
|
174
|
+
self.slot.in_flight.store(0, Ordering::Relaxed);
|
|
175
|
+
self.slot.interrupted.store(false, Ordering::SeqCst);
|
|
156
176
|
|
|
157
|
-
|
|
158
|
-
/// "kino.request") or nil on shutdown. The batch-of-one hot path: no
|
|
159
|
-
/// arrays allocated at all.
|
|
160
|
-
pub fn take_one(ruby: &Ruby, server_id: u64, worker_id: usize) -> Result<Option<RHash>, Error> {
|
|
161
|
-
match checkout(ruby, server_id, worker_id)? {
|
|
162
|
-
Some((server, slot, ctx)) => Ok(Some(admit(ruby, &server, &slot, ctx)?)),
|
|
163
|
-
None => Ok(None),
|
|
177
|
+
Ok(block_take(&server, &self.slot)?.map(|ctx| (server, ctx)))
|
|
164
178
|
}
|
|
165
|
-
}
|
|
166
179
|
|
|
167
|
-
/// Take
|
|
168
|
-
///
|
|
169
|
-
///
|
|
170
|
-
pub fn
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
let Some((server, slot, first)) = checkout(ruby, server_id, worker_id)? else {
|
|
177
|
-
return Ok(None);
|
|
178
|
-
};
|
|
180
|
+
/// Take one request; returns its env Hash (request handle inside under
|
|
181
|
+
/// "kino.request") or nil on shutdown. The batch-of-one hot path: no
|
|
182
|
+
/// arrays allocated at all.
|
|
183
|
+
pub fn take_one(ruby: &Ruby, rb_self: &Worker) -> Result<Option<RHash>, Error> {
|
|
184
|
+
match rb_self.checkout()? {
|
|
185
|
+
Some((server, ctx)) => Ok(Some(admit(ruby, &server, &rb_self.slot, ctx)?)),
|
|
186
|
+
None => Ok(None),
|
|
187
|
+
}
|
|
188
|
+
}
|
|
179
189
|
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
190
|
+
/// Take up to `max` requests: block for the first, drain the rest
|
|
191
|
+
/// non-blocking (they only batch when the queue is already deep).
|
|
192
|
+
/// Returns nil on shutdown; otherwise an Array of env Hashes.
|
|
193
|
+
pub fn take_batch(ruby: &Ruby, rb_self: &Worker, max: usize) -> Result<Option<RArray>, Error> {
|
|
194
|
+
let Some((server, first)) = rb_self.checkout()? else {
|
|
195
|
+
return Ok(None);
|
|
196
|
+
};
|
|
197
|
+
|
|
198
|
+
let batch = ruby.ary_new_capa(max.max(1));
|
|
199
|
+
batch.push(admit(ruby, &server, &rb_self.slot, first)?)?;
|
|
200
|
+
for _ in 1..max {
|
|
201
|
+
match server.req_rx.try_recv() {
|
|
202
|
+
Ok(ctx) => batch.push(admit(ruby, &server, &rb_self.slot, ctx)?)?,
|
|
203
|
+
Err(_) => break,
|
|
204
|
+
}
|
|
186
205
|
}
|
|
206
|
+
Ok(Some(batch))
|
|
187
207
|
}
|
|
188
|
-
Ok(Some(batch))
|
|
189
208
|
}
|
|
190
209
|
|
|
191
210
|
/// The fused hot path: answer `request` (complete response in one shot)
|
|
@@ -194,28 +213,25 @@ pub fn take_batch(
|
|
|
194
213
|
pub fn respond_and_take_one(
|
|
195
214
|
ruby: &Ruby,
|
|
196
215
|
request: &Request,
|
|
197
|
-
|
|
198
|
-
worker_id: usize,
|
|
216
|
+
worker: magnus::typed_data::Obj<Worker>,
|
|
199
217
|
status: u16,
|
|
200
218
|
headers: RHash,
|
|
201
219
|
body: RString,
|
|
202
220
|
) -> Result<Option<RHash>, Error> {
|
|
203
221
|
crate::request::respond_simple(ruby, request, status, headers, body)?;
|
|
204
|
-
take_one(ruby,
|
|
222
|
+
Worker::take_one(ruby, &worker)
|
|
205
223
|
}
|
|
206
224
|
|
|
207
225
|
/// Batch variant of the fused call.
|
|
208
|
-
#[allow(clippy::too_many_arguments)]
|
|
209
226
|
pub fn respond_and_take(
|
|
210
227
|
ruby: &Ruby,
|
|
211
228
|
request: &Request,
|
|
212
|
-
|
|
213
|
-
worker_id: usize,
|
|
229
|
+
worker: magnus::typed_data::Obj<Worker>,
|
|
214
230
|
max: usize,
|
|
215
231
|
status: u16,
|
|
216
232
|
headers: RHash,
|
|
217
233
|
body: RString,
|
|
218
234
|
) -> Result<Option<RArray>, Error> {
|
|
219
235
|
crate::request::respond_simple(ruby, request, status, headers, body)?;
|
|
220
|
-
take_batch(ruby,
|
|
236
|
+
Worker::take_batch(ruby, &worker, max)
|
|
221
237
|
}
|