yamine 0.21.2 → 0.22.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +170 -0
- data/README.md +39 -12
- data/lib/ask/skills/yamine/SKILL.md +20 -1
- data/lib/yamine/cli/boot.rb +291 -26
- data/lib/yamine/cli/context.rb +1 -1
- data/lib/yamine/cli/routes.rb +96 -19
- data/lib/yamine/cli/system.rb +20 -10
- data/lib/yamine/cli.rb +3 -2
- data/lib/yamine/doctor.rb +9 -0
- data/lib/yamine/process_tree.rb +55 -0
- data/lib/yamine/proxy.rb +238 -53
- data/lib/yamine/runner.rb +10 -4
- data/lib/yamine/supervisor.rb +147 -15
- data/lib/yamine/trust.rb +183 -8
- data/lib/yamine/version.rb +1 -1
- metadata +1 -1
data/lib/yamine/process_tree.rb
CHANGED
|
@@ -57,9 +57,58 @@ module Yamine
|
|
|
57
57
|
# "can't be called from trap context" — a stop path that only works
|
|
58
58
|
# outside a trap is a stop that never happens on Ctrl-C.
|
|
59
59
|
PGROUP_LEADERS = {}
|
|
60
|
+
# Exit statuses of what we spawned, by pid, filled in by the reaper
|
|
61
|
+
# that `detach` starts. Same reasoning, same shape, and the same
|
|
62
|
+
# bounded cost as PGROUP_LEADERS above: an entry is a few dozen bytes
|
|
63
|
+
# per process this one ever spawned, dropped by `forget` whenever
|
|
64
|
+
# the pid is signalled.
|
|
65
|
+
STATUSES = {}
|
|
60
66
|
|
|
61
67
|
module_function
|
|
62
68
|
|
|
69
|
+
# Spawn a child and reap it without blocking us, keeping what it
|
|
70
|
+
# exited with. A drop-in for Process.detach — nobody waits on a live
|
|
71
|
+
# backend, a dead pid is all the boot needs — that also keeps the one
|
|
72
|
+
# thing only a reaper can know. Without it the answer dies with the
|
|
73
|
+
# child: `kill(0, pid)` says a process is gone and nothing says
|
|
74
|
+
# whether it exited 0 or was killed, so "web is down" is all a
|
|
75
|
+
# supervisor could ever report.
|
|
76
|
+
#
|
|
77
|
+
# Unlocked, like PGROUP_LEADERS, and a Hash read from a supervision
|
|
78
|
+
# thread: every operation is a single call the GVL makes atomic.
|
|
79
|
+
def detach(pid)
|
|
80
|
+
return pid unless pid.to_i.positive?
|
|
81
|
+
|
|
82
|
+
Thread.new do
|
|
83
|
+
_waited, status = ::Process.waitpid2(pid)
|
|
84
|
+
STATUSES[pid] = status
|
|
85
|
+
rescue SystemCallError
|
|
86
|
+
nil
|
|
87
|
+
end
|
|
88
|
+
pid
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
# Process::Status for a pid we spawned, or nil when there is nothing
|
|
92
|
+
# to report: not our child, or it has not been reaped yet.
|
|
93
|
+
#
|
|
94
|
+
# Never blocks, which is the whole point — the caller is a loop that
|
|
95
|
+
# has other routes to watch, and asking a live child how it is doing
|
|
96
|
+
# would hang it. So a status that has not arrived yet is nil, and the
|
|
97
|
+
# caller asks again on its next pass.
|
|
98
|
+
def status(pid)
|
|
99
|
+
recorded = STATUSES[pid]
|
|
100
|
+
return recorded if recorded
|
|
101
|
+
|
|
102
|
+
# Not one of ours to have reaped (a pid out of a route entry, a
|
|
103
|
+
# double in a test): ask the kernel. WNOHANG so a live pid is not
|
|
104
|
+
# waited on, and ECHILD — which is the answer for anything that is
|
|
105
|
+
# not a child of ours — is a nil, not a failure.
|
|
106
|
+
_waited, status = ::Process.waitpid2(pid, ::Process::WNOHANG)
|
|
107
|
+
status
|
|
108
|
+
rescue StandardError
|
|
109
|
+
nil
|
|
110
|
+
end
|
|
111
|
+
|
|
63
112
|
# Spawn a boot process as its own group leader and record that it is
|
|
64
113
|
# one. This is the only place a boot process is created, so "every
|
|
65
114
|
# process we boot leads a group" is one fact in one place rather
|
|
@@ -85,8 +134,14 @@ module Yamine
|
|
|
85
134
|
PGROUP_LEADERS.key?(pid)
|
|
86
135
|
end
|
|
87
136
|
|
|
137
|
+
# Drop every record of a pid we have finished with. Both tables are
|
|
138
|
+
# about a spawn, not about a number that has to keep meaning one:
|
|
139
|
+
# leaving them behind is how a pid that later gets recycled gets
|
|
140
|
+
# signalled as a group it never led, or reported with an exit status
|
|
141
|
+
# that belongs to somebody else.
|
|
88
142
|
def forget(pid)
|
|
89
143
|
PGROUP_LEADERS.delete(pid)
|
|
144
|
+
STATUSES.delete(pid)
|
|
90
145
|
end
|
|
91
146
|
|
|
92
147
|
# Does this pid lead its own process group? Ours, if we spawned it.
|
data/lib/yamine/proxy.rb
CHANGED
|
@@ -17,6 +17,11 @@ module Yamine
|
|
|
17
17
|
class Proxy
|
|
18
18
|
HOPS_HEADER = "x-yamine-hops"
|
|
19
19
|
HEALTH_HEADER = "x-yamine"
|
|
20
|
+
# Failure class on a 502 (backend-refused / backend-silent), so an
|
|
21
|
+
# agent can branch on what went wrong without parsing the page. Same
|
|
22
|
+
# split of labour as HEALTH_HEADER: a browser shows the human the body,
|
|
23
|
+
# an agent reads the header.
|
|
24
|
+
ERROR_HEADER = "x-yamine-error"
|
|
20
25
|
MAX_HOPS = 5
|
|
21
26
|
MAX_HEAD_BYTES = 64 * 1024
|
|
22
27
|
MAX_HOSTNAME_BYTES = 253
|
|
@@ -36,6 +41,28 @@ module Yamine
|
|
|
36
41
|
# Raised when a peer is silent past the idle bound. An IOError so the
|
|
37
42
|
# existing rescue-and-close paths treat it like any dead peer.
|
|
38
43
|
IdleTimeout = Class.new(IOError)
|
|
44
|
+
# Head bound: how long a backend gets to send the response *head*
|
|
45
|
+
# after the request is fully forwarded. Split from IDLE_TIMEOUT
|
|
46
|
+
# because the two answer different questions, and answering them with
|
|
47
|
+
# one clock is what made a dead backend and a slow one look identical.
|
|
48
|
+
#
|
|
49
|
+
# IDLE_TIMEOUT caps silence between *body* bytes, so it has to stay
|
|
50
|
+
# generous: a long-lived streaming response (an SSE chat backend)
|
|
51
|
+
# commits its head at once and then goes quiet for unbounded stretches
|
|
52
|
+
# with no heartbeat, and any clock that bounds its life is wrong. The
|
|
53
|
+
# head is the backend's first chance to say anything at all, and until
|
|
54
|
+
# it does the client has nothing to stream — a font that hangs 60s and
|
|
55
|
+
# then 502s tells neither a human nor an agent why.
|
|
56
|
+
#
|
|
57
|
+
# Defaults to the same 60s the head read already got (it used to ride
|
|
58
|
+
# IDLE_TIMEOUT's default), so splitting the clocks cannot turn a
|
|
59
|
+
# request that works today into a failure.
|
|
60
|
+
HEAD_TIMEOUT = Float(ENV.fetch("YAMINE_PROXY_HEAD_TIMEOUT", "60"))
|
|
61
|
+
# Raised when a response head runs out its own budget. A distinct class
|
|
62
|
+
# because the two timeouts mean different things: HeadTimeout is the
|
|
63
|
+
# one backend failure where the app is provably up and merely slow,
|
|
64
|
+
# which is also the only one a retry can be safe for.
|
|
65
|
+
HeadTimeout = Class.new(IdleTimeout)
|
|
39
66
|
CHUNK_BYTES = 16_384
|
|
40
67
|
# Route cache: routes.json is the source of truth, but re-reading
|
|
41
68
|
# and re-parsing it on every request is wasteful under HMR polling.
|
|
@@ -47,7 +74,8 @@ module Yamine
|
|
|
47
74
|
LoopDetected = Struct.new(:host, :hops)
|
|
48
75
|
|
|
49
76
|
def initialize(store:, port: 443, tls: true, state_dir: nil, on_error: nil,
|
|
50
|
-
supervisor: nil, max_connections: MAX_CONNECTIONS, idle_timeout: IDLE_TIMEOUT,
|
|
77
|
+
supervisor: nil, max_connections: MAX_CONNECTIONS, idle_timeout: IDLE_TIMEOUT,
|
|
78
|
+
head_timeout: HEAD_TIMEOUT, tlds: nil)
|
|
51
79
|
@store = store
|
|
52
80
|
@port = port
|
|
53
81
|
@tls = tls
|
|
@@ -56,6 +84,7 @@ module Yamine
|
|
|
56
84
|
@supervisor = supervisor
|
|
57
85
|
@max_connections = max_connections
|
|
58
86
|
@idle_timeout = idle_timeout
|
|
87
|
+
@head_timeout = head_timeout
|
|
59
88
|
@tlds = Array(tlds).flatten.compact.map(&:downcase)
|
|
60
89
|
@tlds = [Hostname::DEFAULT_TLD] if @tlds.empty?
|
|
61
90
|
@inflight = 0
|
|
@@ -78,7 +107,7 @@ module Yamine
|
|
|
78
107
|
servers.each { |s| s.listen(@port) }
|
|
79
108
|
@port = servers.first.addr[1]
|
|
80
109
|
Ownership.chown_state_dir(@store.dir)
|
|
81
|
-
|
|
110
|
+
ensure_ca_trust
|
|
82
111
|
trap("INT") { stop(servers) }
|
|
83
112
|
trap("TERM") do
|
|
84
113
|
@supervisor&.shutdown
|
|
@@ -159,14 +188,30 @@ module Yamine
|
|
|
159
188
|
|
|
160
189
|
private
|
|
161
190
|
|
|
162
|
-
#
|
|
163
|
-
#
|
|
164
|
-
#
|
|
165
|
-
|
|
166
|
-
|
|
191
|
+
# The proxy is the process the browser's TLS stack actually talks to,
|
|
192
|
+
# so it is where the CA has to be trusted — as root, into the System
|
|
193
|
+
# keychain silently, and as a normal user, into the login keychain.
|
|
194
|
+
# A non-elevated proxy used to skip this entirely, which is how a
|
|
195
|
+
# first run as an ordinary user installed a CA that could not work.
|
|
196
|
+
#
|
|
197
|
+
# `Trust.trusted?` asks the trust store, not just the state-dir
|
|
198
|
+
# marker, so a CA that is present but untrusted is repaired instead
|
|
199
|
+
# of being short-circuited. Trust.trust only writes the marker once
|
|
200
|
+
# the trust setting is confirmed, so a failure is retried (and
|
|
201
|
+
# reported) on the next boot rather than cached as a success.
|
|
202
|
+
#
|
|
203
|
+
# YAMINE_SKIP_CA_TRUST is for machines whose CA arrives some other
|
|
204
|
+
# way — an MDM profile, a hand-run `security add-trusted-cert` — and
|
|
205
|
+
# for the suite, which spawns real proxies and must never reach a real
|
|
206
|
+
# keychain.
|
|
207
|
+
def ensure_ca_trust
|
|
208
|
+
return if ENV["YAMINE_SKIP_CA_TRUST"]
|
|
209
|
+
return if Trust.trusted?(@state_dir)
|
|
167
210
|
|
|
168
211
|
result = Trust.trust(@state_dir)
|
|
169
|
-
|
|
212
|
+
return if result[:trusted]
|
|
213
|
+
|
|
214
|
+
@on_error.call("CA trust warning: #{result[:error]}")
|
|
170
215
|
end
|
|
171
216
|
|
|
172
217
|
def admit?
|
|
@@ -228,6 +273,12 @@ module Yamine
|
|
|
228
273
|
# responses with Content-Length allow the loop to continue.
|
|
229
274
|
def handle(sock)
|
|
230
275
|
tls_handshake(sock)
|
|
276
|
+
# The 502 below names the app, so it needs the route it was serving
|
|
277
|
+
# and the host that asked for it. A connection can fail before
|
|
278
|
+
# either exists (a client that connects and stalls), so both start
|
|
279
|
+
# out empty and the page degrades instead of naming a wrong app.
|
|
280
|
+
entry = nil
|
|
281
|
+
host = ""
|
|
231
282
|
buf = +""
|
|
232
283
|
loop do
|
|
233
284
|
head, buf = read_head(sock, buf)
|
|
@@ -256,42 +307,87 @@ module Yamine
|
|
|
256
307
|
# Supervised managed apps may be stopped (idle/crashed/restarted):
|
|
257
308
|
# boot on request, then serve.
|
|
258
309
|
if @supervisor && entry["kind"] == "socket" && entry["spec"]
|
|
259
|
-
|
|
260
|
-
unless
|
|
261
|
-
render_bad_gateway(sock)
|
|
310
|
+
booted = @supervisor.ensure_running(entry)
|
|
311
|
+
unless booted
|
|
312
|
+
render_bad_gateway(sock, entry, host, :refused)
|
|
262
313
|
break
|
|
263
314
|
end
|
|
315
|
+
entry = booted
|
|
264
316
|
end
|
|
265
317
|
@supervisor&.touch(entry["hostname"])
|
|
266
318
|
|
|
267
319
|
headers[HOPS_HEADER] = (headers[HOPS_HEADER].to_i + 1).to_s
|
|
268
320
|
set_forwarded(headers, sock, tls: @tls)
|
|
269
321
|
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
322
|
+
keep, reason = forward(sock, entry, method, target, headers, buf)
|
|
323
|
+
# One retry, and only where retrying cannot duplicate work. A
|
|
324
|
+
# backend that was merely slow answers the second attempt; a POST
|
|
325
|
+
# that arrived twice is worse than a slow page, so a bodiless
|
|
326
|
+
# request is the only one that gets another chance.
|
|
327
|
+
if reason && retriable?(reason, method, headers)
|
|
328
|
+
@on_error.call("retrying #{method} #{host} after #{reason}")
|
|
329
|
+
keep, reason = forward(sock, entry, method, target, headers, buf)
|
|
276
330
|
end
|
|
277
331
|
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
backend.close rescue nil
|
|
332
|
+
if reason
|
|
333
|
+
render_bad_gateway(sock, entry, host, reason)
|
|
281
334
|
break
|
|
282
335
|
end
|
|
283
|
-
|
|
336
|
+
break unless keep == :keep_alive
|
|
284
337
|
end
|
|
285
338
|
rescue SystemCallError, OpenSSL::SSL::SSLError, IOError => e
|
|
286
339
|
@on_error.call("Proxy error: #{e.message}")
|
|
287
|
-
render_bad_gateway(sock) rescue nil
|
|
340
|
+
render_bad_gateway(sock, entry, host, :gone) rescue nil
|
|
288
341
|
ensure
|
|
289
342
|
sock.close rescue nil
|
|
290
343
|
end
|
|
291
344
|
|
|
345
|
+
# One attempt at the backend: dial, forward the request, relay the
|
|
346
|
+
# response. Returns [keep_alive?, reason], where reason is nil on
|
|
347
|
+
# success and otherwise names which backend failure this was. That one
|
|
348
|
+
# value decides both whether a retry is safe and what the 502 says, so
|
|
349
|
+
# "refused" and "silent" can never drift apart again.
|
|
350
|
+
#
|
|
351
|
+
# The backend socket is always closed here: it is never reused (only
|
|
352
|
+
# the client connection is), so a failed attempt has nothing to keep
|
|
353
|
+
# alive either.
|
|
354
|
+
def forward(sock, entry, method, target, headers, buf)
|
|
355
|
+
begin
|
|
356
|
+
backend = dial(entry)
|
|
357
|
+
rescue SystemCallError => e
|
|
358
|
+
@on_error.call("dial failed for #{entry["hostname"]}: #{e.message}")
|
|
359
|
+
return [:close, :refused]
|
|
360
|
+
end
|
|
361
|
+
|
|
362
|
+
begin
|
|
363
|
+
pipe_request(sock, backend, method, target, headers, buf)
|
|
364
|
+
rescue HeadTimeout
|
|
365
|
+
[:close, :silent]
|
|
366
|
+
rescue EOFError
|
|
367
|
+
[:close, :gone]
|
|
368
|
+
ensure
|
|
369
|
+
backend.close rescue nil
|
|
370
|
+
end
|
|
371
|
+
end
|
|
372
|
+
|
|
373
|
+
# A refused dial never reached the app, so any method replays safely —
|
|
374
|
+
# the request body is still unread on the client socket. A head
|
|
375
|
+
# timeout means the head DID reach the app, so the only safe replays
|
|
376
|
+
# are the bodiless methods: the body has already been consumed off the
|
|
377
|
+
# client socket and cannot be faithfully resent, and a duplicated
|
|
378
|
+
# message is a worse outcome than a slow page.
|
|
379
|
+
def retriable?(reason, method, headers)
|
|
380
|
+
return true if reason == :refused
|
|
381
|
+
return false unless reason == :silent
|
|
382
|
+
return false unless %w[GET HEAD OPTIONS].include?(method.to_s.upcase)
|
|
383
|
+
|
|
384
|
+
request_body_length(headers) == 0
|
|
385
|
+
end
|
|
386
|
+
|
|
292
387
|
# Forward one request (body framed by Content-Length), then relay
|
|
293
|
-
# the response. Returns :keep_alive when both
|
|
294
|
-
# the connection and the response length was
|
|
388
|
+
# the response. Returns [keep_alive?, reason]: :keep_alive when both
|
|
389
|
+
# sides want to reuse the connection and the response length was
|
|
390
|
+
# known, and a reason when the backend failed to answer instead.
|
|
295
391
|
def pipe_request(sock, backend, method, target, headers, buf)
|
|
296
392
|
write_all(backend, rebuild_head(method, target, headers))
|
|
297
393
|
|
|
@@ -301,7 +397,7 @@ module Yamine
|
|
|
301
397
|
write_all(backend, buf) unless buf.empty?
|
|
302
398
|
copy_stream(sock, backend)
|
|
303
399
|
relay_response_close(backend, sock)
|
|
304
|
-
return :close
|
|
400
|
+
return [:close, nil]
|
|
305
401
|
end
|
|
306
402
|
|
|
307
403
|
remaining = body_len
|
|
@@ -313,11 +409,9 @@ module Yamine
|
|
|
313
409
|
end
|
|
314
410
|
copy_stream(sock, backend, remaining) if remaining > 0
|
|
315
411
|
|
|
316
|
-
rhead, rbuf =
|
|
317
|
-
if rhead.nil?
|
|
318
|
-
|
|
319
|
-
return :close
|
|
320
|
-
end
|
|
412
|
+
rhead, rbuf = read_response_head(backend)
|
|
413
|
+
return [:close, :gone] if rhead.nil?
|
|
414
|
+
|
|
321
415
|
_rm, _rt, rheaders = parse_head(rhead)
|
|
322
416
|
write_all(sock, rhead)
|
|
323
417
|
write_all(sock, rbuf) unless rbuf.empty?
|
|
@@ -326,7 +420,7 @@ module Yamine
|
|
|
326
420
|
if rlen.nil?
|
|
327
421
|
# No Content-Length: close-delimited response.
|
|
328
422
|
copy_stream(backend, sock)
|
|
329
|
-
return :close
|
|
423
|
+
return [:close, nil]
|
|
330
424
|
end
|
|
331
425
|
remaining = rlen - rbuf.bytesize
|
|
332
426
|
copy_stream(backend, sock, remaining) if remaining > 0
|
|
@@ -334,14 +428,14 @@ module Yamine
|
|
|
334
428
|
if keep_alive?(headers) && keep_alive?(rheaders)
|
|
335
429
|
# Any bytes beyond Content-Length on the backend are a second
|
|
336
430
|
# pipelined response on a connection we won't reuse — drop.
|
|
337
|
-
:keep_alive
|
|
431
|
+
[:keep_alive, nil]
|
|
338
432
|
else
|
|
339
|
-
:close
|
|
433
|
+
[:close, nil]
|
|
340
434
|
end
|
|
341
435
|
end
|
|
342
436
|
|
|
343
437
|
def relay_response_close(backend, sock)
|
|
344
|
-
rhead, rbuf =
|
|
438
|
+
rhead, rbuf = read_response_head(backend)
|
|
345
439
|
return if rhead.nil?
|
|
346
440
|
|
|
347
441
|
write_all(sock, rhead)
|
|
@@ -407,19 +501,41 @@ module Yamine
|
|
|
407
501
|
# (pipelined requests) across calls. Returns [head, buf]. The wait
|
|
408
502
|
# for more bytes is idle-bounded: a peer that sends nothing is
|
|
409
503
|
# dropped (nil) instead of pinning the thread.
|
|
410
|
-
|
|
504
|
+
#
|
|
505
|
+
# A caller may pass its own budget (`timeout:`), which is the only way
|
|
506
|
+
# a timeout escapes as HeadTimeout instead of being flattened into the
|
|
507
|
+
# same [nil, ...] a dead peer returns — see read_response_head for why
|
|
508
|
+
# that distinction is the whole point.
|
|
509
|
+
def read_head(sock, buf, timeout: nil)
|
|
411
510
|
loop do
|
|
412
511
|
if (idx = buf.index("\r\n\r\n"))
|
|
413
512
|
return [buf.byteslice(0, idx + 4), buf.byteslice(idx + 4..) || +""]
|
|
414
513
|
end
|
|
415
514
|
return [nil, buf] if buf.bytesize > MAX_HEAD_BYTES
|
|
416
515
|
|
|
417
|
-
buf << read_chunk(sock, CHUNK_BYTES)
|
|
516
|
+
buf << read_chunk(sock, CHUNK_BYTES, timeout: timeout)
|
|
418
517
|
end
|
|
518
|
+
rescue IdleTimeout => e
|
|
519
|
+
# A caller that brought its own budget needs to be told a read timed
|
|
520
|
+
# out; everyone else keeps the old contract, where a peer that goes
|
|
521
|
+
# quiet is just another [nil, ...].
|
|
522
|
+
raise HeadTimeout, e.message, e.backtrace if timeout
|
|
523
|
+
|
|
524
|
+
[nil, +""]
|
|
419
525
|
rescue EOFError, IOError
|
|
420
526
|
[nil, +""]
|
|
421
527
|
end
|
|
422
528
|
|
|
529
|
+
# The response head is read on its own budget, and is the one read
|
|
530
|
+
# where a timeout is allowed out of read_head: until the head arrives
|
|
531
|
+
# the client has no response at all, so "how long do we wait" is a
|
|
532
|
+
# real question with a tunable answer — and the answer is also what
|
|
533
|
+
# makes a retry safe. Once the head is in, the body is on the idle
|
|
534
|
+
# clock, which stays generous for a streaming response.
|
|
535
|
+
def read_response_head(sock)
|
|
536
|
+
read_head(sock, +"", timeout: @head_timeout)
|
|
537
|
+
end
|
|
538
|
+
|
|
423
539
|
# Finish a deferred TLS handshake under the idle bound. A no-op for
|
|
424
540
|
# plain sockets and for SSLSockets that already handshook (a second
|
|
425
541
|
# accept on an established connection returns immediately).
|
|
@@ -439,13 +555,13 @@ module Yamine
|
|
|
439
555
|
# blocks in the kernel without a select deadline, so no stalled peer
|
|
440
556
|
# pins the thread — including mid-TLS-handshake stalls, which plain
|
|
441
557
|
# readpartial would ride out forever.
|
|
442
|
-
def read_chunk(sock, size)
|
|
558
|
+
def read_chunk(sock, size, timeout: nil)
|
|
443
559
|
loop do
|
|
444
560
|
result = sock.read_nonblock(size, exception: false)
|
|
445
561
|
return result if result.is_a?(String)
|
|
446
562
|
raise EOFError, "end of file reached" if result.nil?
|
|
447
563
|
|
|
448
|
-
wait_for(sock, result)
|
|
564
|
+
wait_for(sock, result, timeout)
|
|
449
565
|
end
|
|
450
566
|
end
|
|
451
567
|
|
|
@@ -490,29 +606,34 @@ module Yamine
|
|
|
490
606
|
end
|
|
491
607
|
|
|
492
608
|
# Block until the socket is ready for the direction a nonblocking op
|
|
493
|
-
# asked for;
|
|
494
|
-
|
|
609
|
+
# asked for; a timeout when the bound passes with no progress. A
|
|
610
|
+
# caller-supplied bound only changes how long we wait — what a timeout
|
|
611
|
+
# *means* is read_head's call, because only it knows whether it was
|
|
612
|
+
# waiting on a client, on a body, or on a backend's first word.
|
|
613
|
+
def wait_for(sock, wait_kind, timeout = nil)
|
|
614
|
+
bound = timeout || @idle_timeout
|
|
495
615
|
if wait_kind == :wait_readable
|
|
496
|
-
raise IdleTimeout, "
|
|
497
|
-
elsif !writable?(sock)
|
|
498
|
-
raise IdleTimeout, "
|
|
616
|
+
raise IdleTimeout, "no bytes within #{bound}s" unless readable?(sock, bound)
|
|
617
|
+
elsif !writable?(sock, bound)
|
|
618
|
+
raise IdleTimeout, "no bytes within #{bound}s"
|
|
499
619
|
end
|
|
500
620
|
nil
|
|
501
621
|
end
|
|
502
622
|
|
|
503
|
-
# select(2) with the idle
|
|
504
|
-
# readable without a syscall. A closed socket
|
|
505
|
-
# report not-ready and let the nonblocking op
|
|
506
|
-
|
|
623
|
+
# select(2) with the caller's bound (the idle one by default). SSL-
|
|
624
|
+
# buffered bytes count as readable without a syscall. A closed socket
|
|
625
|
+
# raises in select — report not-ready and let the nonblocking op
|
|
626
|
+
# raise the real error.
|
|
627
|
+
def readable?(sock, timeout = @idle_timeout)
|
|
507
628
|
return true if sock.respond_to?(:pending) && sock.pending.positive?
|
|
508
629
|
|
|
509
|
-
!IO.select([sock], nil, nil,
|
|
630
|
+
!IO.select([sock], nil, nil, timeout).nil?
|
|
510
631
|
rescue IOError, SystemCallError
|
|
511
632
|
false
|
|
512
633
|
end
|
|
513
634
|
|
|
514
|
-
def writable?(sock)
|
|
515
|
-
!IO.select(nil, [sock], nil,
|
|
635
|
+
def writable?(sock, timeout = @idle_timeout)
|
|
636
|
+
!IO.select(nil, [sock], nil, nil, timeout).nil?
|
|
516
637
|
rescue IOError, SystemCallError
|
|
517
638
|
false
|
|
518
639
|
end
|
|
@@ -652,12 +773,74 @@ module Yamine
|
|
|
652
773
|
respond(sock, 503, body)
|
|
653
774
|
end
|
|
654
775
|
|
|
655
|
-
|
|
656
|
-
|
|
776
|
+
# The 502 used to be a fixed 60 bytes: "The target app is not
|
|
777
|
+
# responding." No hostname, no target, no owner, no directory, no log,
|
|
778
|
+
# no next command — so a dead app and a busy one produced byte-identical
|
|
779
|
+
# pages, and a 60-second font request came back with nothing to act on.
|
|
780
|
+
# The two need opposite responses (start it vs. go read why it is
|
|
781
|
+
# stuck), so the page names the app, the backend it points at, which
|
|
782
|
+
# of the two it was, who registered it, where it lives, and the exact
|
|
783
|
+
# command to run. Modeled on render_not_found, which already answers
|
|
784
|
+
# "nobody knows what this hostname is" properly.
|
|
785
|
+
def render_bad_gateway(sock, entry, host, reason = :refused)
|
|
786
|
+
kind = reason == :silent ? "backend-silent" : "backend-refused"
|
|
787
|
+
headers = { ERROR_HEADER => kind }
|
|
788
|
+
bare = Hostname.strip_port(host)
|
|
789
|
+
# Same DNS-rebinding boundary as render_not_found: a Host outside our
|
|
790
|
+
# TLDs is a website that got us to answer, not the local developer
|
|
791
|
+
# asking, so it never learns the app's directory or the agent's name.
|
|
792
|
+
return respond(sock, 502, "<h1>Bad Gateway</h1>", headers: headers) unless friendly_host?(bare)
|
|
793
|
+
|
|
794
|
+
target = entry ? entry["target"].to_s : ""
|
|
795
|
+
body = "<h1>Bad Gateway</h1><p>#{what_happened(reason, target)}</p>" \
|
|
796
|
+
"#{bad_gateway_owner(entry, bare)}#{bad_gateway_fix(entry)}"
|
|
797
|
+
respond(sock, 502, body, headers: headers)
|
|
657
798
|
rescue IOError, SystemCallError
|
|
658
799
|
nil
|
|
659
800
|
end
|
|
660
801
|
|
|
802
|
+
# Which of the two failures this was, said the way each one has to be
|
|
803
|
+
# read: not listening means start the app, silent means go find out why
|
|
804
|
+
# an app that is up is not answering. One "not responding" for both
|
|
805
|
+
# sends the reader to the wrong place either way.
|
|
806
|
+
def what_happened(reason, target)
|
|
807
|
+
at = target.empty? ? "" : " at <code>#{escape(target)}</code>"
|
|
808
|
+
case reason
|
|
809
|
+
when :silent
|
|
810
|
+
"The backend#{at} accepted the connection, then sent no response for " \
|
|
811
|
+
"#{format("%.4g", @head_timeout)} seconds — it is up, not down."
|
|
812
|
+
when :gone
|
|
813
|
+
"The backend#{at} accepted the connection, then closed it without " \
|
|
814
|
+
"sending a response."
|
|
815
|
+
else
|
|
816
|
+
"Nothing is listening#{at}."
|
|
817
|
+
end
|
|
818
|
+
end
|
|
819
|
+
|
|
820
|
+
def bad_gateway_owner(entry, host)
|
|
821
|
+
return "" if entry.nil?
|
|
822
|
+
|
|
823
|
+
# spec.dir, never Dir.pwd: the proxy runs from somewhere else entirely
|
|
824
|
+
# (a launchd daemon, a worktree of yamine itself), and inside one a
|
|
825
|
+
# Dir.pwd-derived path names the wrong checkout entirely.
|
|
826
|
+
dir = entry.dig("spec", "dir")
|
|
827
|
+
where = dir ? ", app in <code>#{escape(dir)}</code>" : ""
|
|
828
|
+
agent = entry["agent"].to_s
|
|
829
|
+
by = agent.empty? ? "" : " by agent <strong>#{escape(agent)}</strong>"
|
|
830
|
+
return "<p>Registered#{by}#{where}.</p>" if host.empty?
|
|
831
|
+
|
|
832
|
+
"<p>The backend for <strong>#{escape(host)}</strong> is registered#{by}#{where}.</p>"
|
|
833
|
+
end
|
|
834
|
+
|
|
835
|
+
def bad_gateway_fix(entry)
|
|
836
|
+
dir = entry&.dig("spec", "dir")
|
|
837
|
+
return "<p>Start it with <code>yamine start</code> in that app's directory.</p>" unless dir
|
|
838
|
+
|
|
839
|
+
log = File.join(dir, "log", "development.log")
|
|
840
|
+
"<p>Start it: <code>cd #{escape(dir)} && yamine start</code></p>" \
|
|
841
|
+
"<p>What it said before it went quiet: <code>#{escape(log)}</code></p>"
|
|
842
|
+
end
|
|
843
|
+
|
|
661
844
|
def render_loop(sock, host)
|
|
662
845
|
respond(sock, 508, "<h1>Loop Detected</h1><p>#{escape(host)} passed through " \
|
|
663
846
|
"yamine too many times. Check dev-server proxy config.</p>")
|
|
@@ -665,11 +848,13 @@ module Yamine
|
|
|
665
848
|
nil
|
|
666
849
|
end
|
|
667
850
|
|
|
668
|
-
def respond(sock, status, body)
|
|
851
|
+
def respond(sock, status, body, headers: {})
|
|
669
852
|
message = { 404 => "Not Found", 502 => "Bad Gateway", 508 => "Loop Detected" }[status]
|
|
853
|
+
extra = headers.map { |k, v| "#{k}: #{v}\r\n" }.join
|
|
670
854
|
write_all(sock, "HTTP/1.1 #{status} #{message}\r\n" \
|
|
671
855
|
"Content-Type: text/html\r\n" \
|
|
672
856
|
"#{HEALTH_HEADER}: 1\r\n" \
|
|
857
|
+
"#{extra}" \
|
|
673
858
|
"Content-Length: #{body.bytesize}\r\n" \
|
|
674
859
|
"Connection: close\r\n\r\n#{body}")
|
|
675
860
|
rescue IOError, SystemCallError
|
data/lib/yamine/runner.rb
CHANGED
|
@@ -12,6 +12,12 @@ module Yamine
|
|
|
12
12
|
# (puma-dev model). Without puma, managed degrades to rackup on TCP
|
|
13
13
|
# (run-mode shape). Run mode (everything else): subprocess with
|
|
14
14
|
# injected PORT/YAMINE_URL, readiness by TCP connect.
|
|
15
|
+
#
|
|
16
|
+
# Every spawn is reaped through ProcessTree.detach rather than
|
|
17
|
+
# Process.detach. Nothing waits on a live backend — a dead pid is all
|
|
18
|
+
# a health wait needs — and the reaper is the only thing that can
|
|
19
|
+
# still say what the child exited with once it is gone, which is what
|
|
20
|
+
# `yamine start` reports when a process dies mid-run.
|
|
15
21
|
class Runner
|
|
16
22
|
SOCKET_DIR = File.join("tmp", "sockets")
|
|
17
23
|
SOCKET_NAME = "yamine.sock"
|
|
@@ -76,7 +82,7 @@ module Yamine
|
|
|
76
82
|
config_ru = File.join(dir, "config.ru")
|
|
77
83
|
cmd = ["rackup", "-o", "127.0.0.1", "-p", port.to_s, config_ru]
|
|
78
84
|
pid = with_clean_env { spawn(env, *cmd, chdir: dir, out: log_path(dir, name), err: [:child, :out]) }
|
|
79
|
-
|
|
85
|
+
ProcessTree.detach(pid)
|
|
80
86
|
unless wait_for_tcp(port, timeout: 60)
|
|
81
87
|
stop_pid(pid)
|
|
82
88
|
raise Error, "App '#{name}' did not boot within 60s. " \
|
|
@@ -113,7 +119,7 @@ module Yamine
|
|
|
113
119
|
database_url: database_url, database_env: database_env, extra_env: extra_env)
|
|
114
120
|
path = log_path(dir, name)
|
|
115
121
|
pid = with_clean_env { ProcessTree.spawn(env, *command, chdir: dir, out: path, err: [:child, :out]) }
|
|
116
|
-
|
|
122
|
+
ProcessTree.detach(pid)
|
|
117
123
|
target = "127.0.0.1:#{port}"
|
|
118
124
|
if register
|
|
119
125
|
begin
|
|
@@ -146,7 +152,7 @@ module Yamine
|
|
|
146
152
|
database_url: database_url, database_env: database_env, extra_env: extra_env)
|
|
147
153
|
path = log_path(dir, name)
|
|
148
154
|
pid = with_clean_env { ProcessTree.spawn(env, *command, chdir: dir, out: path, err: [:child, :out]) }
|
|
149
|
-
|
|
155
|
+
ProcessTree.detach(pid)
|
|
150
156
|
App.new(name: name, hostname: hostname, url: url, pid: pid,
|
|
151
157
|
target: "127.0.0.1:#{port}", kind: "tcp", command: command)
|
|
152
158
|
end
|
|
@@ -265,7 +271,7 @@ module Yamine
|
|
|
265
271
|
env = child_env(dir, url: url, port: nil)
|
|
266
272
|
cmd = socket_command(dir, socket_path)
|
|
267
273
|
pid = with_clean_env { spawn(env, *cmd, chdir: dir, out: log_path(dir, name), err: [:child, :out]) }
|
|
268
|
-
|
|
274
|
+
ProcessTree.detach(pid)
|
|
269
275
|
pid
|
|
270
276
|
end
|
|
271
277
|
|