yamine 0.21.2 → 0.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/lib/yamine/proxy.rb CHANGED
@@ -17,6 +17,11 @@ module Yamine
17
17
  class Proxy
18
18
  HOPS_HEADER = "x-yamine-hops"
19
19
  HEALTH_HEADER = "x-yamine"
20
+ # Failure class on a 502 (backend-refused / backend-silent), so an
21
+ # agent can branch on what went wrong without parsing the page. Same
22
+ # split of labour as HEALTH_HEADER: a browser shows the human the body,
23
+ # an agent reads the header.
24
+ ERROR_HEADER = "x-yamine-error"
20
25
  MAX_HOPS = 5
21
26
  MAX_HEAD_BYTES = 64 * 1024
22
27
  MAX_HOSTNAME_BYTES = 253
@@ -36,6 +41,28 @@ module Yamine
36
41
  # Raised when a peer is silent past the idle bound. An IOError so the
37
42
  # existing rescue-and-close paths treat it like any dead peer.
38
43
  IdleTimeout = Class.new(IOError)
44
+ # Head bound: how long a backend gets to send the response *head*
45
+ # after the request is fully forwarded. Split from IDLE_TIMEOUT
46
+ # because the two answer different questions, and answering them with
47
+ # one clock is what made a dead backend and a slow one look identical.
48
+ #
49
+ # IDLE_TIMEOUT caps silence between *body* bytes, so it has to stay
50
+ # generous: a long-lived streaming response (an SSE chat backend)
51
+ # commits its head at once and then goes quiet for unbounded stretches
52
+ # with no heartbeat, and any clock that bounds its life is wrong. The
53
+ # head is the backend's first chance to say anything at all, and until
54
+ # it does the client has nothing to stream — a font that hangs 60s and
55
+ # then 502s tells neither a human nor an agent why.
56
+ #
57
+ # Defaults to the same 60s the head read already got (it used to ride
58
+ # IDLE_TIMEOUT's default), so splitting the clocks cannot turn a
59
+ # request that works today into a failure.
60
+ HEAD_TIMEOUT = Float(ENV.fetch("YAMINE_PROXY_HEAD_TIMEOUT", "60"))
61
+ # Raised when a response head runs out its own budget. A distinct class
62
+ # because the two timeouts mean different things: HeadTimeout is the
63
+ # one backend failure where the app is provably up and merely slow,
64
+ # which is also the only one a retry can be safe for.
65
+ HeadTimeout = Class.new(IdleTimeout)
39
66
  CHUNK_BYTES = 16_384
40
67
  # Route cache: routes.json is the source of truth, but re-reading
41
68
  # and re-parsing it on every request is wasteful under HMR polling.
@@ -47,7 +74,8 @@ module Yamine
47
74
  LoopDetected = Struct.new(:host, :hops)
48
75
 
49
76
  def initialize(store:, port: 443, tls: true, state_dir: nil, on_error: nil,
50
- supervisor: nil, max_connections: MAX_CONNECTIONS, idle_timeout: IDLE_TIMEOUT, tlds: nil)
77
+ supervisor: nil, max_connections: MAX_CONNECTIONS, idle_timeout: IDLE_TIMEOUT,
78
+ head_timeout: HEAD_TIMEOUT, tlds: nil)
51
79
  @store = store
52
80
  @port = port
53
81
  @tls = tls
@@ -56,6 +84,7 @@ module Yamine
56
84
  @supervisor = supervisor
57
85
  @max_connections = max_connections
58
86
  @idle_timeout = idle_timeout
87
+ @head_timeout = head_timeout
59
88
  @tlds = Array(tlds).flatten.compact.map(&:downcase)
60
89
  @tlds = [Hostname::DEFAULT_TLD] if @tlds.empty?
61
90
  @inflight = 0
@@ -228,6 +257,12 @@ module Yamine
228
257
  # responses with Content-Length allow the loop to continue.
229
258
  def handle(sock)
230
259
  tls_handshake(sock)
260
+ # The 502 below names the app, so it needs the route it was serving
261
+ # and the host that asked for it. A connection can fail before
262
+ # either exists (a client that connects and stalls), so both start
263
+ # out empty and the page degrades instead of naming a wrong app.
264
+ entry = nil
265
+ host = ""
231
266
  buf = +""
232
267
  loop do
233
268
  head, buf = read_head(sock, buf)
@@ -256,42 +291,87 @@ module Yamine
256
291
  # Supervised managed apps may be stopped (idle/crashed/restarted):
257
292
  # boot on request, then serve.
258
293
  if @supervisor && entry["kind"] == "socket" && entry["spec"]
259
- entry = @supervisor.ensure_running(entry)
260
- unless entry
261
- render_bad_gateway(sock)
294
+ booted = @supervisor.ensure_running(entry)
295
+ unless booted
296
+ render_bad_gateway(sock, entry, host, :refused)
262
297
  break
263
298
  end
299
+ entry = booted
264
300
  end
265
301
  @supervisor&.touch(entry["hostname"])
266
302
 
267
303
  headers[HOPS_HEADER] = (headers[HOPS_HEADER].to_i + 1).to_s
268
304
  set_forwarded(headers, sock, tls: @tls)
269
305
 
270
- begin
271
- backend = dial(entry)
272
- rescue SystemCallError => e
273
- @on_error.call("dial failed for #{host}: #{e.message}")
274
- render_bad_gateway(sock)
275
- break
306
+ keep, reason = forward(sock, entry, method, target, headers, buf)
307
+ # One retry, and only where retrying cannot duplicate work. A
308
+ # backend that was merely slow answers the second attempt; a POST
309
+ # that arrived twice is worse than a slow page, so a bodiless
310
+ # request is the only one that gets another chance.
311
+ if reason && retriable?(reason, method, headers)
312
+ @on_error.call("retrying #{method} #{host} after #{reason}")
313
+ keep, reason = forward(sock, entry, method, target, headers, buf)
276
314
  end
277
315
 
278
- keep = pipe_request(sock, backend, method, target, headers, buf)
279
- unless keep == :keep_alive
280
- backend.close rescue nil
316
+ if reason
317
+ render_bad_gateway(sock, entry, host, reason)
281
318
  break
282
319
  end
283
- backend.close rescue nil
320
+ break unless keep == :keep_alive
284
321
  end
285
322
  rescue SystemCallError, OpenSSL::SSL::SSLError, IOError => e
286
323
  @on_error.call("Proxy error: #{e.message}")
287
- render_bad_gateway(sock) rescue nil
324
+ render_bad_gateway(sock, entry, host, :gone) rescue nil
288
325
  ensure
289
326
  sock.close rescue nil
290
327
  end
291
328
 
329
+ # One attempt at the backend: dial, forward the request, relay the
330
+ # response. Returns [keep_alive?, reason], where reason is nil on
331
+ # success and otherwise names which backend failure this was. That one
332
+ # value decides both whether a retry is safe and what the 502 says, so
333
+ # "refused" and "silent" can never drift apart again.
334
+ #
335
+ # The backend socket is always closed here: it is never reused (only
336
+ # the client connection is), so a failed attempt has nothing to keep
337
+ # alive either.
338
+ def forward(sock, entry, method, target, headers, buf)
339
+ begin
340
+ backend = dial(entry)
341
+ rescue SystemCallError => e
342
+ @on_error.call("dial failed for #{entry["hostname"]}: #{e.message}")
343
+ return [:close, :refused]
344
+ end
345
+
346
+ begin
347
+ pipe_request(sock, backend, method, target, headers, buf)
348
+ rescue HeadTimeout
349
+ [:close, :silent]
350
+ rescue EOFError
351
+ [:close, :gone]
352
+ ensure
353
+ backend.close rescue nil
354
+ end
355
+ end
356
+
357
+ # A refused dial never reached the app, so any method replays safely —
358
+ # the request body is still unread on the client socket. A head
359
+ # timeout means the head DID reach the app, so the only safe replays
360
+ # are the bodiless methods: the body has already been consumed off the
361
+ # client socket and cannot be faithfully resent, and a duplicated
362
+ # message is a worse outcome than a slow page.
363
+ def retriable?(reason, method, headers)
364
+ return true if reason == :refused
365
+ return false unless reason == :silent
366
+ return false unless %w[GET HEAD OPTIONS].include?(method.to_s.upcase)
367
+
368
+ request_body_length(headers) == 0
369
+ end
370
+
292
371
  # Forward one request (body framed by Content-Length), then relay
293
- # the response. Returns :keep_alive when both sides want to reuse
294
- # the connection and the response length was known.
372
+ # the response. Returns [keep_alive?, reason]: :keep_alive when both
373
+ # sides want to reuse the connection and the response length was
374
+ # known, and a reason when the backend failed to answer instead.
295
375
  def pipe_request(sock, backend, method, target, headers, buf)
296
376
  write_all(backend, rebuild_head(method, target, headers))
297
377
 
@@ -301,7 +381,7 @@ module Yamine
301
381
  write_all(backend, buf) unless buf.empty?
302
382
  copy_stream(sock, backend)
303
383
  relay_response_close(backend, sock)
304
- return :close
384
+ return [:close, nil]
305
385
  end
306
386
 
307
387
  remaining = body_len
@@ -313,11 +393,9 @@ module Yamine
313
393
  end
314
394
  copy_stream(sock, backend, remaining) if remaining > 0
315
395
 
316
- rhead, rbuf = read_head(backend, +"")
317
- if rhead.nil?
318
- render_bad_gateway(sock)
319
- return :close
320
- end
396
+ rhead, rbuf = read_response_head(backend)
397
+ return [:close, :gone] if rhead.nil?
398
+
321
399
  _rm, _rt, rheaders = parse_head(rhead)
322
400
  write_all(sock, rhead)
323
401
  write_all(sock, rbuf) unless rbuf.empty?
@@ -326,7 +404,7 @@ module Yamine
326
404
  if rlen.nil?
327
405
  # No Content-Length: close-delimited response.
328
406
  copy_stream(backend, sock)
329
- return :close
407
+ return [:close, nil]
330
408
  end
331
409
  remaining = rlen - rbuf.bytesize
332
410
  copy_stream(backend, sock, remaining) if remaining > 0
@@ -334,14 +412,14 @@ module Yamine
334
412
  if keep_alive?(headers) && keep_alive?(rheaders)
335
413
  # Any bytes beyond Content-Length on the backend are a second
336
414
  # pipelined response on a connection we won't reuse — drop.
337
- :keep_alive
415
+ [:keep_alive, nil]
338
416
  else
339
- :close
417
+ [:close, nil]
340
418
  end
341
419
  end
342
420
 
343
421
  def relay_response_close(backend, sock)
344
- rhead, rbuf = read_head(backend, +"")
422
+ rhead, rbuf = read_response_head(backend)
345
423
  return if rhead.nil?
346
424
 
347
425
  write_all(sock, rhead)
@@ -407,19 +485,41 @@ module Yamine
407
485
  # (pipelined requests) across calls. Returns [head, buf]. The wait
408
486
  # for more bytes is idle-bounded: a peer that sends nothing is
409
487
  # dropped (nil) instead of pinning the thread.
410
- def read_head(sock, buf)
488
+ #
489
+ # A caller may pass its own budget (`timeout:`), which is the only way
490
+ # a timeout escapes as HeadTimeout instead of being flattened into the
491
+ # same [nil, ...] a dead peer returns — see read_response_head for why
492
+ # that distinction is the whole point.
493
+ def read_head(sock, buf, timeout: nil)
411
494
  loop do
412
495
  if (idx = buf.index("\r\n\r\n"))
413
496
  return [buf.byteslice(0, idx + 4), buf.byteslice(idx + 4..) || +""]
414
497
  end
415
498
  return [nil, buf] if buf.bytesize > MAX_HEAD_BYTES
416
499
 
417
- buf << read_chunk(sock, CHUNK_BYTES)
500
+ buf << read_chunk(sock, CHUNK_BYTES, timeout: timeout)
418
501
  end
502
+ rescue IdleTimeout => e
503
+ # A caller that brought its own budget needs to be told a read timed
504
+ # out; everyone else keeps the old contract, where a peer that goes
505
+ # quiet is just another [nil, ...].
506
+ raise HeadTimeout, e.message, e.backtrace if timeout
507
+
508
+ [nil, +""]
419
509
  rescue EOFError, IOError
420
510
  [nil, +""]
421
511
  end
422
512
 
513
+ # The response head is read on its own budget, and is the one read
514
+ # where a timeout is allowed out of read_head: until the head arrives
515
+ # the client has no response at all, so "how long do we wait" is a
516
+ # real question with a tunable answer — and the answer is also what
517
+ # makes a retry safe. Once the head is in, the body is on the idle
518
+ # clock, which stays generous for a streaming response.
519
+ def read_response_head(sock)
520
+ read_head(sock, +"", timeout: @head_timeout)
521
+ end
522
+
423
523
  # Finish a deferred TLS handshake under the idle bound. A no-op for
424
524
  # plain sockets and for SSLSockets that already handshook (a second
425
525
  # accept on an established connection returns immediately).
@@ -439,13 +539,13 @@ module Yamine
439
539
  # blocks in the kernel without a select deadline, so no stalled peer
440
540
  # pins the thread — including mid-TLS-handshake stalls, which plain
441
541
  # readpartial would ride out forever.
442
- def read_chunk(sock, size)
542
+ def read_chunk(sock, size, timeout: nil)
443
543
  loop do
444
544
  result = sock.read_nonblock(size, exception: false)
445
545
  return result if result.is_a?(String)
446
546
  raise EOFError, "end of file reached" if result.nil?
447
547
 
448
- wait_for(sock, result)
548
+ wait_for(sock, result, timeout)
449
549
  end
450
550
  end
451
551
 
@@ -490,29 +590,34 @@ module Yamine
490
590
  end
491
591
 
492
592
  # Block until the socket is ready for the direction a nonblocking op
493
- # asked for; IdleTimeout when the bound passes with no progress.
494
- def wait_for(sock, wait_kind)
593
+ # asked for; a timeout when the bound passes with no progress. A
594
+ # caller-supplied bound only changes how long we wait — what a timeout
595
+ # *means* is read_head's call, because only it knows whether it was
596
+ # waiting on a client, on a body, or on a backend's first word.
597
+ def wait_for(sock, wait_kind, timeout = nil)
598
+ bound = timeout || @idle_timeout
495
599
  if wait_kind == :wait_readable
496
- raise IdleTimeout, "idle timeout after #{@idle_timeout}s with no bytes" unless readable?(sock)
497
- elsif !writable?(sock)
498
- raise IdleTimeout, "idle timeout after #{@idle_timeout}s with no bytes"
600
+ raise IdleTimeout, "no bytes within #{bound}s" unless readable?(sock, bound)
601
+ elsif !writable?(sock, bound)
602
+ raise IdleTimeout, "no bytes within #{bound}s"
499
603
  end
500
604
  nil
501
605
  end
502
606
 
503
- # select(2) with the idle bound. SSL-buffered bytes count as
504
- # readable without a syscall. A closed socket raises in select —
505
- # report not-ready and let the nonblocking op raise the real error.
506
- def readable?(sock)
607
+ # select(2) with the caller's bound (the idle one by default). SSL-
608
+ # buffered bytes count as readable without a syscall. A closed socket
609
+ # raises in select — report not-ready and let the nonblocking op
610
+ # raise the real error.
611
+ def readable?(sock, timeout = @idle_timeout)
507
612
  return true if sock.respond_to?(:pending) && sock.pending.positive?
508
613
 
509
- !IO.select([sock], nil, nil, @idle_timeout).nil?
614
+ !IO.select([sock], nil, nil, timeout).nil?
510
615
  rescue IOError, SystemCallError
511
616
  false
512
617
  end
513
618
 
514
- def writable?(sock)
515
- !IO.select(nil, [sock], nil, @idle_timeout).nil?
619
+ def writable?(sock, timeout = @idle_timeout)
620
+ !IO.select(nil, [sock], nil, nil, timeout).nil?
516
621
  rescue IOError, SystemCallError
517
622
  false
518
623
  end
@@ -652,12 +757,74 @@ module Yamine
652
757
  respond(sock, 503, body)
653
758
  end
654
759
 
655
- def render_bad_gateway(sock)
656
- respond(sock, 502, "<h1>Bad Gateway</h1><p>The target app is not responding.</p>")
760
+ # The 502 used to be a fixed 60 bytes: "The target app is not
761
+ # responding." No hostname, no target, no owner, no directory, no log,
762
+ # no next command — so a dead app and a busy one produced byte-identical
763
+ # pages, and a 60-second font request came back with nothing to act on.
764
+ # The two need opposite responses (start it vs. go read why it is
765
+ # stuck), so the page names the app, the backend it points at, which
766
+ # of the two it was, who registered it, where it lives, and the exact
767
+ # command to run. Modeled on render_not_found, which already answers
768
+ # "nobody knows what this hostname is" properly.
769
+ def render_bad_gateway(sock, entry, host, reason = :refused)
770
+ kind = reason == :silent ? "backend-silent" : "backend-refused"
771
+ headers = { ERROR_HEADER => kind }
772
+ bare = Hostname.strip_port(host)
773
+ # Same DNS-rebinding boundary as render_not_found: a Host outside our
774
+ # TLDs is a website that got us to answer, not the local developer
775
+ # asking, so it never learns the app's directory or the agent's name.
776
+ return respond(sock, 502, "<h1>Bad Gateway</h1>", headers: headers) unless friendly_host?(bare)
777
+
778
+ target = entry ? entry["target"].to_s : ""
779
+ body = "<h1>Bad Gateway</h1><p>#{what_happened(reason, target)}</p>" \
780
+ "#{bad_gateway_owner(entry, bare)}#{bad_gateway_fix(entry)}"
781
+ respond(sock, 502, body, headers: headers)
657
782
  rescue IOError, SystemCallError
658
783
  nil
659
784
  end
660
785
 
786
+ # Which of the two failures this was, said the way each one has to be
787
+ # read: not listening means start the app, silent means go find out why
788
+ # an app that is up is not answering. One "not responding" for both
789
+ # sends the reader to the wrong place either way.
790
+ def what_happened(reason, target)
791
+ at = target.empty? ? "" : " at <code>#{escape(target)}</code>"
792
+ case reason
793
+ when :silent
794
+ "The backend#{at} accepted the connection, then sent no response for " \
795
+ "#{format("%.4g", @head_timeout)} seconds — it is up, not down."
796
+ when :gone
797
+ "The backend#{at} accepted the connection, then closed it without " \
798
+ "sending a response."
799
+ else
800
+ "Nothing is listening#{at}."
801
+ end
802
+ end
803
+
804
+ def bad_gateway_owner(entry, host)
805
+ return "" if entry.nil?
806
+
807
+ # spec.dir, never Dir.pwd: the proxy runs from somewhere else entirely
808
+ # (a launchd daemon, a worktree of yamine itself), and inside one a
809
+ # Dir.pwd-derived path names the wrong checkout entirely.
810
+ dir = entry.dig("spec", "dir")
811
+ where = dir ? ", app in <code>#{escape(dir)}</code>" : ""
812
+ agent = entry["agent"].to_s
813
+ by = agent.empty? ? "" : " by agent <strong>#{escape(agent)}</strong>"
814
+ return "<p>Registered#{by}#{where}.</p>" if host.empty?
815
+
816
+ "<p>The backend for <strong>#{escape(host)}</strong> is registered#{by}#{where}.</p>"
817
+ end
818
+
819
+ def bad_gateway_fix(entry)
820
+ dir = entry&.dig("spec", "dir")
821
+ return "<p>Start it with <code>yamine start</code> in that app's directory.</p>" unless dir
822
+
823
+ log = File.join(dir, "log", "development.log")
824
+ "<p>Start it: <code>cd #{escape(dir)} && yamine start</code></p>" \
825
+ "<p>What it said before it went quiet: <code>#{escape(log)}</code></p>"
826
+ end
827
+
661
828
  def render_loop(sock, host)
662
829
  respond(sock, 508, "<h1>Loop Detected</h1><p>#{escape(host)} passed through " \
663
830
  "yamine too many times. Check dev-server proxy config.</p>")
@@ -665,11 +832,13 @@ module Yamine
665
832
  nil
666
833
  end
667
834
 
668
- def respond(sock, status, body)
835
+ def respond(sock, status, body, headers: {})
669
836
  message = { 404 => "Not Found", 502 => "Bad Gateway", 508 => "Loop Detected" }[status]
837
+ extra = headers.map { |k, v| "#{k}: #{v}\r\n" }.join
670
838
  write_all(sock, "HTTP/1.1 #{status} #{message}\r\n" \
671
839
  "Content-Type: text/html\r\n" \
672
840
  "#{HEALTH_HEADER}: 1\r\n" \
841
+ "#{extra}" \
673
842
  "Content-Length: #{body.bytesize}\r\n" \
674
843
  "Connection: close\r\n\r\n#{body}")
675
844
  rescue IOError, SystemCallError
data/lib/yamine/runner.rb CHANGED
@@ -12,6 +12,12 @@ module Yamine
12
12
  # (puma-dev model). Without puma, managed degrades to rackup on TCP
13
13
  # (run-mode shape). Run mode (everything else): subprocess with
14
14
  # injected PORT/YAMINE_URL, readiness by TCP connect.
15
+ #
16
+ # Every spawn is reaped through ProcessTree.detach rather than
17
+ # Process.detach. Nothing waits on a live backend — a dead pid is all
18
+ # a health wait needs — and the reaper is the only thing that can
19
+ # still say what the child exited with once it is gone, which is what
20
+ # `yamine start` reports when a process dies mid-run.
15
21
  class Runner
16
22
  SOCKET_DIR = File.join("tmp", "sockets")
17
23
  SOCKET_NAME = "yamine.sock"
@@ -76,7 +82,7 @@ module Yamine
76
82
  config_ru = File.join(dir, "config.ru")
77
83
  cmd = ["rackup", "-o", "127.0.0.1", "-p", port.to_s, config_ru]
78
84
  pid = with_clean_env { spawn(env, *cmd, chdir: dir, out: log_path(dir, name), err: [:child, :out]) }
79
- Process.detach(pid)
85
+ ProcessTree.detach(pid)
80
86
  unless wait_for_tcp(port, timeout: 60)
81
87
  stop_pid(pid)
82
88
  raise Error, "App '#{name}' did not boot within 60s. " \
@@ -113,7 +119,7 @@ module Yamine
113
119
  database_url: database_url, database_env: database_env, extra_env: extra_env)
114
120
  path = log_path(dir, name)
115
121
  pid = with_clean_env { ProcessTree.spawn(env, *command, chdir: dir, out: path, err: [:child, :out]) }
116
- Process.detach(pid)
122
+ ProcessTree.detach(pid)
117
123
  target = "127.0.0.1:#{port}"
118
124
  if register
119
125
  begin
@@ -146,7 +152,7 @@ module Yamine
146
152
  database_url: database_url, database_env: database_env, extra_env: extra_env)
147
153
  path = log_path(dir, name)
148
154
  pid = with_clean_env { ProcessTree.spawn(env, *command, chdir: dir, out: path, err: [:child, :out]) }
149
- Process.detach(pid)
155
+ ProcessTree.detach(pid)
150
156
  App.new(name: name, hostname: hostname, url: url, pid: pid,
151
157
  target: "127.0.0.1:#{port}", kind: "tcp", command: command)
152
158
  end
@@ -265,7 +271,7 @@ module Yamine
265
271
  env = child_env(dir, url: url, port: nil)
266
272
  cmd = socket_command(dir, socket_path)
267
273
  pid = with_clean_env { spawn(env, *cmd, chdir: dir, out: log_path(dir, name), err: [:child, :out]) }
268
- Process.detach(pid)
274
+ ProcessTree.detach(pid)
269
275
  pid
270
276
  end
271
277