peregrine-server 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- peregrine_server-0.8.0/ARCHITECTURE.md +337 -0
- peregrine_server-0.8.0/CONFIG.md +548 -0
- peregrine_server-0.8.0/INSTALLATION.md +287 -0
- peregrine_server-0.8.0/MANIFEST.in +11 -0
- peregrine_server-0.8.0/PKG-INFO +440 -0
- peregrine_server-0.8.0/Package.swift +113 -0
- peregrine_server-0.8.0/README.md +421 -0
- peregrine_server-0.8.0/Sources/CPeregrine/include/peregrine_crypto.h +160 -0
- peregrine_server-0.8.0/Sources/CPeregrine/include/peregrine_metrics.h +77 -0
- peregrine_server-0.8.0/Sources/CPeregrine/include/peregrine_py.h +244 -0
- peregrine_server-0.8.0/Sources/CPeregrine/include/peregrine_sys.h +201 -0
- peregrine_server-0.8.0/Sources/CPeregrine/include/peregrine_tls.h +67 -0
- peregrine_server-0.8.0/Sources/CPeregrine/include/peregrine_udp.h +70 -0
- peregrine_server-0.8.0/Sources/CPeregrine/peregrine_crypto.c +690 -0
- peregrine_server-0.8.0/Sources/CPeregrine/peregrine_extra.c +348 -0
- peregrine_server-0.8.0/Sources/CPeregrine/peregrine_metrics.c +101 -0
- peregrine_server-0.8.0/Sources/CPeregrine/peregrine_py.c +576 -0
- peregrine_server-0.8.0/Sources/CPeregrine/peregrine_sys.c +472 -0
- peregrine_server-0.8.0/Sources/CPeregrine/peregrine_tls.c +340 -0
- peregrine_server-0.8.0/Sources/CPeregrine/peregrine_udp.c +393 -0
- peregrine_server-0.8.0/Sources/CPython/module.modulemap +4 -0
- peregrine_server-0.8.0/Sources/CPython/shim.h +9 -0
- peregrine_server-0.8.0/Sources/PeregrineASGI/ASGIScope.swift +338 -0
- peregrine_server-0.8.0/Sources/PeregrineCore/Allocation.swift +43 -0
- peregrine_server-0.8.0/Sources/PeregrineCore/BufferPool.swift +89 -0
- peregrine_server-0.8.0/Sources/PeregrineCore/ByteBuffer.swift +279 -0
- peregrine_server-0.8.0/Sources/PeregrineCore/Bytes.swift +228 -0
- peregrine_server-0.8.0/Sources/PeregrineCore/Log.swift +293 -0
- peregrine_server-0.8.0/Sources/PeregrineCore/Poller.swift +73 -0
- peregrine_server-0.8.0/Sources/PeregrineFuzzTargets/FuzzCorpus.swift +67 -0
- peregrine_server-0.8.0/Sources/PeregrineFuzzTargets/FuzzTargets.swift +310 -0
- peregrine_server-0.8.0/Sources/PeregrineFuzzTargets/Seeds.swift +134 -0
- peregrine_server-0.8.0/Sources/PeregrineHTTP/ChunkedDecoder.swift +162 -0
- peregrine_server-0.8.0/Sources/PeregrineHTTP/ForwardedTrust.swift +318 -0
- peregrine_server-0.8.0/Sources/PeregrineHTTP/HPACK.swift +584 -0
- peregrine_server-0.8.0/Sources/PeregrineHTTP/HPACKTables.swift +147 -0
- peregrine_server-0.8.0/Sources/PeregrineHTTP/HTTP2Frame.swift +202 -0
- peregrine_server-0.8.0/Sources/PeregrineHTTP/HTTP3Frame.swift +108 -0
- peregrine_server-0.8.0/Sources/PeregrineHTTP/HTTPParser.swift +331 -0
- peregrine_server-0.8.0/Sources/PeregrineHTTP/HTTPRequest.swift +139 -0
- peregrine_server-0.8.0/Sources/PeregrineHTTP/HTTPResponse.swift +212 -0
- peregrine_server-0.8.0/Sources/PeregrineHTTP/QPACK.swift +333 -0
- peregrine_server-0.8.0/Sources/PeregrineHTTP/QPACKTables.swift +117 -0
- peregrine_server-0.8.0/Sources/PeregrineHTTP/WebSocketFrame.swift +262 -0
- peregrine_server-0.8.0/Sources/PeregrinePython/Interned.swift +187 -0
- peregrine_server-0.8.0/Sources/PeregrinePython/Interpreter.swift +526 -0
- peregrine_server-0.8.0/Sources/PeregrinePython/PyRef.swift +127 -0
- peregrine_server-0.8.0/Sources/PeregrinePython/PySeq.swift +119 -0
- peregrine_server-0.8.0/Sources/PeregrinePython/PyStringCache.swift +117 -0
- peregrine_server-0.8.0/Sources/PeregrinePython/PyTrampoline.swift +55 -0
- peregrine_server-0.8.0/Sources/PeregrinePython/PyTypeBuilder.swift +185 -0
- peregrine_server-0.8.0/Sources/PeregrineQUIC/QUICConnection.swift +1211 -0
- peregrine_server-0.8.0/Sources/PeregrineQUIC/QUICCrypto.swift +535 -0
- peregrine_server-0.8.0/Sources/PeregrineQUIC/QUICFrame.swift +276 -0
- peregrine_server-0.8.0/Sources/PeregrineQUIC/QUICLoss.swift +322 -0
- peregrine_server-0.8.0/Sources/PeregrineQUIC/QUICPacket.swift +274 -0
- peregrine_server-0.8.0/Sources/PeregrineQUIC/QUICSend.swift +819 -0
- peregrine_server-0.8.0/Sources/PeregrineQUIC/QUICStream.swift +359 -0
- peregrine_server-0.8.0/Sources/PeregrineQUIC/TLS13.swift +706 -0
- peregrine_server-0.8.0/Sources/PeregrineQUIC/TLSTransportParameters.swift +194 -0
- peregrine_server-0.8.0/Sources/PeregrineQUIC/Varint.swift +267 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/ASGIDispatch.swift +959 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/Config.swift +214 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/Connection.swift +327 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/ForwardedApply.swift +89 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/FreeThreaded.swift +621 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/HTTP2.swift +1124 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/HTTP2Response.swift +238 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/HTTP3.swift +947 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/HTTP3Response.swift +333 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/Metrics.swift +203 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/MetricsListener.swift +261 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/QUICListener.swift +297 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/QUICWorker.swift +102 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/ReloadWatcher.swift +111 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/Runtime.swift +618 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/TLS.swift +165 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/WSGIDispatch.swift +681 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/WSGIMultiplexed.swift +178 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/WSGIPool.swift +646 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/WSGIResponse.swift +442 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/WebSocket.swift +902 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/WebTransport.swift +1075 -0
- peregrine_server-0.8.0/Sources/PeregrineServer/Worker.swift +1440 -0
- peregrine_server-0.8.0/Sources/PeregrineWSGI/WSGIInputStream.swift +167 -0
- peregrine_server-0.8.0/Sources/PeregrineWSGI/WSGIRuntime.swift +325 -0
- peregrine_server-0.8.0/Sources/PeregrineWSGI/WSGIStartResponse.swift +188 -0
- peregrine_server-0.8.0/Sources/peregrine/main.swift +457 -0
- peregrine_server-0.8.0/Sources/pgfuzz/main.swift +263 -0
- peregrine_server-0.8.0/TRANSPORT.md +542 -0
- peregrine_server-0.8.0/Tests/PeregrineTests/CoreTests.swift +413 -0
- peregrine_server-0.8.0/Tests/PeregrineTests/FuzzCorpusTests.swift +61 -0
- peregrine_server-0.8.0/Tests/PeregrineTests/HPACKTests.swift +269 -0
- peregrine_server-0.8.0/Tests/PeregrineTests/HTTPParserTests.swift +326 -0
- peregrine_server-0.8.0/Tests/PeregrineTests/QUICPacketTests.swift +437 -0
- peregrine_server-0.8.0/Tests/PeregrineTests/QUICStreamTests.swift +270 -0
- peregrine_server-0.8.0/Tests/PeregrineTests/WebSocketTests.swift +462 -0
- peregrine_server-0.8.0/assets/peregrine-app-icon.png +0 -0
- peregrine_server-0.8.0/assets/peregrine-logo-dark.jpg +0 -0
- peregrine_server-0.8.0/assets/peregrine-logo-dark.png +0 -0
- peregrine_server-0.8.0/assets/peregrine-logo-light.jpg +0 -0
- peregrine_server-0.8.0/assets/peregrine-logo-light.png +0 -0
- peregrine_server-0.8.0/examples/asgi_app.py +471 -0
- peregrine_server-0.8.0/examples/django_app.py +161 -0
- peregrine_server-0.8.0/examples/fastapi_app.py +145 -0
- peregrine_server-0.8.0/examples/wsgi_app.py +238 -0
- peregrine_server-0.8.0/pyproject.toml +51 -0
- peregrine_server-0.8.0/python/peregrine/__init__.py +97 -0
- peregrine_server-0.8.0/python/peregrine/__main__.py +6 -0
- peregrine_server-0.8.0/python/peregrine/contrib/__init__.py +18 -0
- peregrine_server-0.8.0/python/peregrine/contrib/asgi.py +255 -0
- peregrine_server-0.8.0/python/peregrine/contrib/django.py +98 -0
- peregrine_server-0.8.0/python/peregrine/contrib/fastapi.py +195 -0
- peregrine_server-0.8.0/python/peregrine/webtransport.py +510 -0
- peregrine_server-0.8.0/python/peregrine_server.egg-info/PKG-INFO +440 -0
- peregrine_server-0.8.0/python/peregrine_server.egg-info/SOURCES.txt +132 -0
- peregrine_server-0.8.0/python/peregrine_server.egg-info/dependency_links.txt +1 -0
- peregrine_server-0.8.0/python/peregrine_server.egg-info/entry_points.txt +2 -0
- peregrine_server-0.8.0/python/peregrine_server.egg-info/requires.txt +3 -0
- peregrine_server-0.8.0/python/peregrine_server.egg-info/top_level.txt +1 -0
- peregrine_server-0.8.0/scripts/contrib_test.py +508 -0
- peregrine_server-0.8.0/scripts/feature-test.py +1984 -0
- peregrine_server-0.8.0/scripts/framework-test.sh +209 -0
- peregrine_server-0.8.0/scripts/gen-hpack-tables.py +65 -0
- peregrine_server-0.8.0/scripts/gen-qpack-table.py +88 -0
- peregrine_server-0.8.0/scripts/http2-test.py +739 -0
- peregrine_server-0.8.0/scripts/http3-test.py +750 -0
- peregrine_server-0.8.0/scripts/integration-test.sh +218 -0
- peregrine_server-0.8.0/scripts/quic-vectors.py +83 -0
- peregrine_server-0.8.0/scripts/serverlib.sh +120 -0
- peregrine_server-0.8.0/scripts/webtransport-test.py +787 -0
- peregrine_server-0.8.0/setup.cfg +4 -0
- peregrine_server-0.8.0/setup.py +220 -0
- peregrine_server-0.8.0/setupdist.py +149 -0
|
@@ -0,0 +1,337 @@
|
|
|
1
|
+
<p align="center">
|
|
2
|
+
<img src="assets/peregrine-logo-dark.jpg" alt="peregrine" width="360">
|
|
3
|
+
</p>
|
|
4
|
+
|
|
5
|
+
# Architecture
|
|
6
|
+
|
|
7
|
+
Peregrine embeds CPython. There is no socket between Swift and Python, no
|
|
8
|
+
serialisation step and no second process: Swift owns the accept loop, the
|
|
9
|
+
parser and the response writer, and calls the application directly. Everything
|
|
10
|
+
below follows from that one decision.
|
|
11
|
+
|
|
12
|
+
The protocols themselves are in [TRANSPORT.md](TRANSPORT.md).
|
|
13
|
+
|
|
14
|
+
```
|
|
15
|
+
Sources/
|
|
16
|
+
CPeregrine/ C shim: epoll/kqueue, sockets, signals, TLS, crypto,
|
|
17
|
+
UDP, and the CPython macros Swift cannot import
|
|
18
|
+
PeregrineCore/ buffers, buffer pool, poller, logging, date cache
|
|
19
|
+
PeregrineHTTP/ HTTP/1.1 parser, chunked decoder, response writer,
|
|
20
|
+
HPACK, QPACK, HTTP/2 and HTTP/3 framing
|
|
21
|
+
PeregrinePython/ PyRef, interned constants, custom Python types
|
|
22
|
+
PeregrineQUIC/ QUIC transport and the TLS 1.3 handshake it needs
|
|
23
|
+
PeregrineWSGI/ environ building, wsgi.input, start_response
|
|
24
|
+
PeregrineASGI/ scope and message building
|
|
25
|
+
PeregrineServer/ connection table, worker loop, both dispatchers,
|
|
26
|
+
HTTP/2, HTTP/3, WebSocket, WebTransport, supervisor
|
|
27
|
+
peregrine/ command line entry point
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
`Python.h`, `openssl/ssl.h` and `openssl/evp.h` never appear in a header Swift
|
|
31
|
+
imports. Everything they offer arrives through opaque functions in the shim,
|
|
32
|
+
which is what keeps the Swift side free of the macro soup and the C++-ish
|
|
33
|
+
declarations those headers contain.
|
|
34
|
+
|
|
35
|
+
---
|
|
36
|
+
|
|
37
|
+
## One process per worker
|
|
38
|
+
|
|
39
|
+
Each worker is a separate process with its own interpreter and its own poller.
|
|
40
|
+
How the listening socket is shared depends on the address family, because the
|
|
41
|
+
two have opposite constraints:
|
|
42
|
+
|
|
43
|
+
- **TCP:** every worker opens its own socket with `SO_REUSEPORT`, so each gets
|
|
44
|
+
an independent accept queue in the kernel. No shared accept lock, no
|
|
45
|
+
thundering herd; the kernel spreads connections by hashing the four-tuple.
|
|
46
|
+
- **UDP (HTTP/3):** the same, with the same consequence and one extra one — a
|
|
47
|
+
client that migrates to an address hashing to a different worker reaches a
|
|
48
|
+
worker that has never heard of its connection.
|
|
49
|
+
- **Unix:** a path can only be bound once, so the supervisor creates the
|
|
50
|
+
listener and the workers inherit it across `fork`. Letting each worker bind
|
|
51
|
+
for itself would have every worker unlink and replace the socket the previous
|
|
52
|
+
one had just published, leaving only the last one reachable.
|
|
53
|
+
|
|
54
|
+
The supervisor restarts workers that die, forwards signals, and owns the
|
|
55
|
+
`--reload` watcher. `SIGHUP` restarts the workers without dropping the
|
|
56
|
+
listening socket.
|
|
57
|
+
|
|
58
|
+
---
|
|
59
|
+
|
|
60
|
+
## One thread per worker, when the interpreter allows it
|
|
61
|
+
|
|
62
|
+
A worker is a process for one reason: the GIL. A second thread cannot serve a
|
|
63
|
+
second request, so the only way to a second core is a second interpreter, and
|
|
64
|
+
the only way to a second interpreter is a second process. CPython 3.13 shipped
|
|
65
|
+
a build without the GIL (PEP 703) and that reasoning stops applying.
|
|
66
|
+
`--free-threaded` is the option that says so — the workers become threads of
|
|
67
|
+
one process, and nothing else about them changes.
|
|
68
|
+
|
|
69
|
+
What makes that a small change rather than a rewrite is that a worker never
|
|
70
|
+
reaches outside itself. It owns its poller, its connection slab, its buffer
|
|
71
|
+
pool, its date cache and its event loop; the only things it reads that it does
|
|
72
|
+
not own are written once at start-up and never again — the interned constants,
|
|
73
|
+
the internal Python types, the glue functions, the application object. So the
|
|
74
|
+
work was to move the last few pieces of per-worker state off the process:
|
|
75
|
+
|
|
76
|
+
- `currentWorker` became a thread-local. It is what a `send`/`receive` callable
|
|
77
|
+
or the asyncio reader callback uses to find its worker, and those arrive from
|
|
78
|
+
Python carrying nothing but a connection token.
|
|
79
|
+
- The ASGI event loop, the lifespan handle and the scope builder moved from
|
|
80
|
+
statics onto `Worker`. The scope builder is the one that mattered: it carries
|
|
81
|
+
a mutable memoised header-name cache and a scratch buffer, so one shared
|
|
82
|
+
across threads would have been a data race on the request path.
|
|
83
|
+
|
|
84
|
+
The main thread is not a worker. It costs one mostly-idle thread and buys the
|
|
85
|
+
ordering that matters: **signals land somewhere that is not serving a request.**
|
|
86
|
+
Worker threads are started with every signal blocked; the main thread owns the
|
|
87
|
+
signal pipe and asks each worker to drain by writing down a pipe that worker
|
|
88
|
+
already polls, so `handleSignals` cannot tell the difference between that and a
|
|
89
|
+
real signal.
|
|
90
|
+
|
|
91
|
+
**The lifespan follows the event loop, not the process.** This started out the
|
|
92
|
+
other way around — one lifespan on the main thread's loop, one `startup` for one
|
|
93
|
+
application — and that was wrong. `startup` is where an application builds
|
|
94
|
+
asyncio objects, and an asyncio object binds to the loop that was running when
|
|
95
|
+
it was created; a pool built on the supervising loop and awaited from a worker's
|
|
96
|
+
loop is the "attached to a different loop" error, when it fails loudly at all.
|
|
97
|
+
So each worker thread runs the lifespan on its own loop and publishes its own
|
|
98
|
+
`state` mapping to the scopes that loop serves, and shuts it down on that loop
|
|
99
|
+
once its own requests have drained. The startups are serialised behind a mutex:
|
|
100
|
+
they run against one application object that has never had to be thread-safe.
|
|
101
|
+
|
|
102
|
+
`--lifespan-scope process` asks for the original reading — exactly one `startup`,
|
|
103
|
+
on the supervising loop — which is right for start-up that opens nothing
|
|
104
|
+
loop-bound, and only then. There the shutdown ordering is what it always was:
|
|
105
|
+
every worker drains and is joined *first*, and only then does the application get
|
|
106
|
+
`lifespan.shutdown`.
|
|
107
|
+
|
|
108
|
+
What is genuinely shared is the application, which is the point. One import,
|
|
109
|
+
one set of module-level caches, one warm JIT — instead of N copies. Four workers
|
|
110
|
+
serving a CPU-bound application on four cores reach the same throughput either
|
|
111
|
+
way, in 47 MB as threads against 143 MB as processes. Connection pools are the
|
|
112
|
+
exception, and belong to their loop for the reason above: an application that
|
|
113
|
+
wants one pool per worker thread puts it in the lifespan `state` mapping, which
|
|
114
|
+
is per loop here, rather than in a module global.
|
|
115
|
+
|
|
116
|
+
The trade is isolation: a crash takes every worker with it, where the process
|
|
117
|
+
supervisor would have restarted one. So the two compose rather than compete —
|
|
118
|
+
`--free-threaded --reload` puts the supervisor in front of a single threaded
|
|
119
|
+
child, and in production systemd plays the same part.
|
|
120
|
+
|
|
121
|
+
---
|
|
122
|
+
|
|
123
|
+
## The asyncio integration is one file descriptor
|
|
124
|
+
|
|
125
|
+
The interesting trick in the ASGI path: an epoll (or kqueue) descriptor is
|
|
126
|
+
*itself pollable*. So instead of running a Swift I/O thread and marshalling
|
|
127
|
+
work across to the Python loop, Peregrine hands its poller to asyncio:
|
|
128
|
+
|
|
129
|
+
```python
|
|
130
|
+
loop.add_reader(poller_fd, drain) # drain is a C-level Swift callback
|
|
131
|
+
loop.run_forever()
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
asyncio then treats the entire server as one more readable descriptor. The
|
|
135
|
+
result is one thread, one event loop per worker: no cross-thread queues, no
|
|
136
|
+
`call_soon_threadsafe` wakeups, no GIL handoffs — and uvloop works unchanged,
|
|
137
|
+
because `add_reader` is part of the loop contract. Under `--free-threaded` a
|
|
138
|
+
process holds several of these, one per worker thread, and each is still the
|
|
139
|
+
same self-contained arrangement — the loops never speak to each other.
|
|
140
|
+
|
|
141
|
+
HTTP/3 adds one thing to this: QUIC has timers of its own — an acknowledgement
|
|
142
|
+
owed in milliseconds, a probe that has to fire — so a worker serving QUIC
|
|
143
|
+
cannot sleep for the usual interval. The loop's periodic callback runs at 20 ms
|
|
144
|
+
instead of the default, and the poll timeout is shortened to whatever the
|
|
145
|
+
nearest QUIC deadline is.
|
|
146
|
+
|
|
147
|
+
The WSGI path uses no asyncio at all. The poller is the only thing that blocks,
|
|
148
|
+
and the GIL is released around it so application threads — the optional pool,
|
|
149
|
+
or threads the application started itself — still run.
|
|
150
|
+
|
|
151
|
+
---
|
|
152
|
+
|
|
153
|
+
## The connection table
|
|
154
|
+
|
|
155
|
+
Connections live in one contiguous slab indexed by slot, with a free list
|
|
156
|
+
threaded through the unused entries. Accepting is an index pop; closing is an
|
|
157
|
+
index push.
|
|
158
|
+
|
|
159
|
+
Poller tokens pack `(generation, slot)` into 64 bits, and the generation makes
|
|
160
|
+
a stale event — one epoll collected for a descriptor we closed earlier in the
|
|
161
|
+
same batch — a discarded compare rather than a use-after-free. The same token
|
|
162
|
+
is what a Python `send`/`receive` callable carries, so an application holding
|
|
163
|
+
one after its connection has gone finds an empty slot instead of somebody
|
|
164
|
+
else's.
|
|
165
|
+
|
|
166
|
+
A slot is a connection *or* a stream. A stream slot has `fd = -1` and a pointer
|
|
167
|
+
back to its parent, and everything above the transport treats the two
|
|
168
|
+
identically; see [one request path](TRANSPORT.md#one-request-path).
|
|
169
|
+
|
|
170
|
+
---
|
|
171
|
+
|
|
172
|
+
## Minimising ARC
|
|
173
|
+
|
|
174
|
+
The brief was to keep Swift's reference counting off the request path — not to
|
|
175
|
+
ban classes outright. Start-up configuration, the WSGI thread pool, the QUIC
|
|
176
|
+
connection objects and the `--reload` watcher use ordinary Swift classes and
|
|
177
|
+
arrays, because they run once per process, or once per connection, and clarity
|
|
178
|
+
is worth more there. On the request path:
|
|
179
|
+
|
|
180
|
+
**Python objects are never wrapped in Swift classes.** A `PyObject` already has
|
|
181
|
+
its own reference count, which under a standard CPython build is a non-atomic
|
|
182
|
+
increment protected by the GIL. Putting it behind a Swift class would mean
|
|
183
|
+
paying *two* counts, one of them atomic. Instead `PyRef` is a `~Copyable`
|
|
184
|
+
struct whose `deinit` calls `Py_DECREF`; the compiler proves single ownership
|
|
185
|
+
and inserts the decref exactly once on every path, at zero runtime cost.
|
|
186
|
+
Borrowed references are a bare `OpaquePointer`.
|
|
187
|
+
|
|
188
|
+
**No object per connection.** The slab above.
|
|
189
|
+
|
|
190
|
+
**Buffers are values, not objects.** `ByteBuffer` is a trivial struct — a
|
|
191
|
+
pointer and three integers, passed in registers — with an explicit `destroy()`
|
|
192
|
+
at the one place a buffer dies. It is not a class (that would be ARC on every
|
|
193
|
+
hand-off) and not `~Copyable` with a `deinit` (that fights the move-only
|
|
194
|
+
checker on every partial mutation of a slab entry). Ownership is a documented
|
|
195
|
+
invariant here rather than a language-enforced one; that is the trade this
|
|
196
|
+
server is built to make, and it is confined to a handful of files.
|
|
197
|
+
|
|
198
|
+
**Nothing on the request path becomes a `String`.** The parser produces
|
|
199
|
+
`(offset, length)` pairs into the read buffer. Header names, values, paths and
|
|
200
|
+
query strings stay as bytes until the moment they are handed to Python, where
|
|
201
|
+
they are copied exactly once into a `str` or `bytes`. Logging assembles bytes
|
|
202
|
+
in a stack buffer and issues one `write(2)`; there is no string interpolation
|
|
203
|
+
anywhere in the server.
|
|
204
|
+
|
|
205
|
+
**Foundation is not linked.** It would drag in ARC-heavy bridging types for no
|
|
206
|
+
benefit here.
|
|
207
|
+
|
|
208
|
+
The Python side gets the same treatment. `send`, `receive` and the awaitable
|
|
209
|
+
they return are C-level types built with `PyType_FromSpec` whose slots are
|
|
210
|
+
Swift `@convention(c)` functions, so `await send(msg)` is a `tp_call` plus a
|
|
211
|
+
`tp_iternext` and nothing else — no Python frame, and no trip through the event
|
|
212
|
+
loop, because a send that completes synchronously returns a pre-completed
|
|
213
|
+
awaitable that raises `StopIteration` on its first step.
|
|
214
|
+
|
|
215
|
+
---
|
|
216
|
+
|
|
217
|
+
## The WSGI thread pool
|
|
218
|
+
|
|
219
|
+
`--wsgi-threads N` turns on a bounded pool. A synchronous application spends
|
|
220
|
+
most of its wall clock *waiting* — on a database, a cache, another service —
|
|
221
|
+
and CPython releases the GIL around every blocking syscall, so those waits can
|
|
222
|
+
overlap. The GIL is not the reason to run one request at a time; it just means
|
|
223
|
+
the pool buys nothing for CPU-bound work, which is why the default is still one
|
|
224
|
+
thread and the inline path is unchanged.
|
|
225
|
+
|
|
226
|
+
The split is what keeps the pool safe:
|
|
227
|
+
|
|
228
|
+
- the **loop thread** owns every connection, buffer and poller, builds the
|
|
229
|
+
environ, and encodes anything that touches a per-connection compressor;
|
|
230
|
+
- a **pool thread** owns only the job, and holds the GIL while it calls the
|
|
231
|
+
application and serialises the response into the job buffer.
|
|
232
|
+
|
|
233
|
+
Bytes cross back under one mutex, and the loop is woken through a pipe it
|
|
234
|
+
already polls.
|
|
235
|
+
|
|
236
|
+
Backpressure is real rather than advisory: when a job buffer passes the high
|
|
237
|
+
water mark the producing thread releases the GIL and blocks until the loop has
|
|
238
|
+
written enough of it, so a streaming response runs at the speed of the client.
|
|
239
|
+
|
|
240
|
+
---
|
|
241
|
+
|
|
242
|
+
## Backpressure
|
|
243
|
+
|
|
244
|
+
Three producers can outrun their consumer, and each is stopped by the same
|
|
245
|
+
idea — refuse to buffer, and let the pressure reach whoever is producing.
|
|
246
|
+
|
|
247
|
+
- **An ASGI application writing a response.** `await send(...)` normally
|
|
248
|
+
completes without suspending, because the bytes go straight into the write
|
|
249
|
+
buffer. When that buffer passes the high water mark it returns a real
|
|
250
|
+
`Future` instead, resolved once the connection has drained back below the low
|
|
251
|
+
one. On a multiplexed stream "drained" means acknowledged by the peer, not
|
|
252
|
+
written to a socket — so the transport tells the layer above when an
|
|
253
|
+
acknowledgement frees send buffer, and a producer parked on a window update
|
|
254
|
+
that a peer with a large window would never send is a bug that has been
|
|
255
|
+
fixed rather than a hazard to live with.
|
|
256
|
+
- **A client uploading a body.** Body bytes are read no further ahead than the
|
|
257
|
+
application has asked for: past the high water mark the worker stops reading
|
|
258
|
+
the socket, so an upload nobody is consuming costs TCP window rather than
|
|
259
|
+
memory. On HTTP/2 and HTTP/3 the window is only given back as the application
|
|
260
|
+
actually reads.
|
|
261
|
+
- **A peer flooding a WebSocket.** Decoded messages queue, bounded by
|
|
262
|
+
`--ws-max-queue` and `--ws-max-queue-bytes`, and the read side switches off
|
|
263
|
+
at the bound.
|
|
264
|
+
- **A WSGI application streaming a response.** PEP 3333 makes this the
|
|
265
|
+
application's own problem to feel: a yielded block goes to the socket before
|
|
266
|
+
the next is requested, and a `write()` goes out before it returns, so a
|
|
267
|
+
producer faster than the client parks in the write that will not complete.
|
|
268
|
+
Inline that means draining the block to the socket entirely, waiting on
|
|
269
|
+
writability as often as it takes — stopping at the high water mark instead
|
|
270
|
+
strands up to that much of the block, and on the inline path there is nothing
|
|
271
|
+
to send it while the application runs. On a pool thread the same block is
|
|
272
|
+
handed to the loop, which writes it while the application produces the next
|
|
273
|
+
one, and the high water mark parks the thread. Buffering it all until the
|
|
274
|
+
application returns, which is what the server used to do, hides the pressure
|
|
275
|
+
and delays every byte.
|
|
276
|
+
|
|
277
|
+
The exception is a WSGI response on an HTTP/2 or HTTP/3 stream, where a block
|
|
278
|
+
can only go as far as the peer's flow-control window allows. Waiting for more
|
|
279
|
+
window inline would deadlock — the `WINDOW_UPDATE` that would release it
|
|
280
|
+
arrives on the loop that is blocked — so the bytes stay with the transport
|
|
281
|
+
and go out when the loop next runs. `--wsgi-threads` is what removes that
|
|
282
|
+
gap, because then the loop is running.
|
|
283
|
+
|
|
284
|
+
Bodies are bounded by `--max-body`, heads by `--max-header-size`, header count
|
|
285
|
+
by a fixed limit, and connections by `--max-connections`; a full table answers
|
|
286
|
+
503 and hangs up rather than queueing without bound.
|
|
287
|
+
|
|
288
|
+
---
|
|
289
|
+
|
|
290
|
+
## Other things that make it fast
|
|
291
|
+
|
|
292
|
+
- **A prototype environ/scope dict** holding every constant entry is built once
|
|
293
|
+
and shallow-copied per request. `PyDict_Copy` on a small dict is a table
|
|
294
|
+
memcpy; the alternative is ten-plus hashed insertions every request.
|
|
295
|
+
- **Interned keys.** Every environ and scope key is interned at start-up, so
|
|
296
|
+
dict insertion compares a cached hash instead of hashing key bytes again.
|
|
297
|
+
- **Memoised header keys.** `User-Agent` becomes `HTTP_USER_AGENT` (WSGI) or
|
|
298
|
+
lowercased `b"user-agent"` (ASGI) once per process, in an open-addressed
|
|
299
|
+
cache keyed by the raw bytes. The cache stops growing once half full, so a
|
|
300
|
+
flood of unique header names cannot become a memory-exhaustion vector.
|
|
301
|
+
- **Character classes are register constants.** `tchar` membership is two
|
|
302
|
+
64-bit shifts, not a table lookup.
|
|
303
|
+
- **A cached `Date` header**, reformatted at most once a second by a
|
|
304
|
+
no-allocation, no-locale civil-from-days conversion.
|
|
305
|
+
- **Vectorcall everywhere** — no intermediate argument tuples.
|
|
306
|
+
- **Pooled read buffers** recycled LIFO, so the block handed out next is the
|
|
307
|
+
one still in cache.
|
|
308
|
+
- **Framing decided with full information.** A WSGI response that is a list
|
|
309
|
+
gets an exact `Content-Length`; a generator gets chunked encoding on
|
|
310
|
+
HTTP/1.1, and on a multiplexed stream the end of the stream is the framing.
|
|
311
|
+
- **`writev`, `TCP_NODELAY`, `accept4`, `MSG`-free reads**, and one
|
|
312
|
+
`epoll_ctl` only when the interest mask actually changes.
|
|
313
|
+
- **`recvmmsg` for QUIC**, 32 datagrams per syscall.
|
|
314
|
+
|
|
315
|
+
---
|
|
316
|
+
|
|
317
|
+
## Shutdown
|
|
318
|
+
|
|
319
|
+
`SIGTERM` or `SIGINT` drains gracefully, with a deadline. The listener stops
|
|
320
|
+
accepting, idle connections close immediately, websockets are sent a `going
|
|
321
|
+
away` close, and in-flight requests get `--graceful-timeout` to finish.
|
|
322
|
+
Whatever is still running when that expires is cancelled and awaited — so
|
|
323
|
+
cancellation is actually delivered rather than merely requested — and only then
|
|
324
|
+
does the application receive `lifespan.shutdown`.
|
|
325
|
+
|
|
326
|
+
Doing it in that order is the point: cancelling every task first would cancel
|
|
327
|
+
the lifespan task too, and the application would never reach the code after its
|
|
328
|
+
`yield`, so its cleanup — closing database pools, flushing telemetry — would
|
|
329
|
+
silently not run.
|
|
330
|
+
|
|
331
|
+
Every layer of that is cooperative, and cooperation is not a guarantee: a task
|
|
332
|
+
can catch `CancelledError` and carry on, a C extension can sit in a syscall,
|
|
333
|
+
and a single worker started without `--workers` has no supervisor to escalate
|
|
334
|
+
to. So the cancellation phase has its own bound, the lifespan handler's
|
|
335
|
+
cancellation has one too, async generator cleanup has one, and behind all of it
|
|
336
|
+
a `SIGALRM` watchdog `_exit`s the process once the grace period plus a margin
|
|
337
|
+
has passed. A deadline that nothing enforces is not a deadline.
|