yamine 0.21.2 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +132 -0
- data/README.md +39 -12
- data/lib/ask/skills/yamine/SKILL.md +20 -1
- data/lib/yamine/cli/boot.rb +291 -26
- data/lib/yamine/cli/context.rb +1 -1
- data/lib/yamine/cli/routes.rb +96 -19
- data/lib/yamine/cli/system.rb +13 -3
- data/lib/yamine/cli.rb +3 -2
- data/lib/yamine/process_tree.rb +55 -0
- data/lib/yamine/proxy.rb +215 -46
- data/lib/yamine/runner.rb +10 -4
- data/lib/yamine/supervisor.rb +147 -15
- data/lib/yamine/version.rb +1 -1
- metadata +1 -1
data/lib/yamine/supervisor.rb
CHANGED
|
@@ -2,20 +2,31 @@
|
|
|
2
2
|
|
|
3
3
|
require "fileutils"
|
|
4
4
|
require "monitor"
|
|
5
|
+
require "socket"
|
|
5
6
|
|
|
6
7
|
module Yamine
|
|
7
|
-
# Daemon-owned supervision for managed
|
|
8
|
+
# Daemon-owned supervision for managed apps (puma-dev model).
|
|
8
9
|
#
|
|
9
10
|
# The proxy daemon is long-lived and sees every request, so it owns:
|
|
10
11
|
# last-used tracking (idle kill), tmp/restart.txt watching, backend
|
|
11
|
-
# liveness, and boot-on-request for stopped apps.
|
|
12
|
-
#
|
|
12
|
+
# liveness, and boot-on-request for stopped apps.
|
|
13
|
+
#
|
|
14
|
+
# What it watches is any route that names a directory to boot from
|
|
15
|
+
# (spec.dir) — socket and tcp alike, because a `yamine start` tree
|
|
16
|
+
# registers tcp routes and was therefore invisible to all of it. What
|
|
17
|
+
# it may DO with a route is not the same set: see rebootable?.
|
|
18
|
+
# Static aliases (pid 0, no spec) are never supervised.
|
|
13
19
|
#
|
|
14
20
|
# State is in-memory (the daemon is the only supervisor); routes.json
|
|
15
21
|
# stays declarative. Killing is graceful-first with a short KILL
|
|
16
22
|
# fallback so puma drains.
|
|
17
23
|
class Supervisor
|
|
18
24
|
DEFAULT_IDLE = 900 # 15 minutes
|
|
25
|
+
# Bound on the connect that decides whether a tcp backend is
|
|
26
|
+
# answering. A local port answers in microseconds or not at all; the
|
|
27
|
+
# case that actually blocks is a full accept backlog, and this loop
|
|
28
|
+
# supervises every route on the machine.
|
|
29
|
+
CONNECT_TIMEOUT = 2
|
|
19
30
|
|
|
20
31
|
attr_reader :interval
|
|
21
32
|
|
|
@@ -38,8 +49,32 @@ module Yamine
|
|
|
38
49
|
v.to_f
|
|
39
50
|
end
|
|
40
51
|
|
|
52
|
+
# The daemon watches a route when it knows where the app lives on
|
|
53
|
+
# disk, because that is all restart.txt and the boot path need. Both
|
|
54
|
+
# kinds qualify, and the tcp half is not a nicety: `yamine start`
|
|
55
|
+
# registers kind "tcp" (Runner#boot_run, Runner#adopt), so excluding
|
|
56
|
+
# it made `yamine restart` a no-op for every app yamine itself
|
|
57
|
+
# started — it touched tmp/restart.txt and announced "managed app
|
|
58
|
+
# restarts on next request" for a tree nothing was watching. A spec
|
|
59
|
+
# is required, so static aliases and hand-written routes stay out:
|
|
60
|
+
# there is no directory to watch and no app to restart.
|
|
41
61
|
def supervised?(route)
|
|
42
|
-
|
|
62
|
+
%w[socket tcp].include?(route["kind"]) &&
|
|
63
|
+
route["spec"].is_a?(Hash) && route["spec"]["dir"].is_a?(String)
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
# Can the daemon put this app back on its feet by itself?
|
|
67
|
+
#
|
|
68
|
+
# Only a socket route: its target is derived (dir/tmp/sockets/
|
|
69
|
+
# yamine.sock), so a fresh backend can be pointed at the same route,
|
|
70
|
+
# and the proxy asks the supervisor about socket routes by design. A
|
|
71
|
+
# tcp route's port is a free port chosen at boot and the route
|
|
72
|
+
# records no command to re-run, so there is nothing to rebuild it
|
|
73
|
+
# from — the two halves of "stopped, boots on next request" cannot
|
|
74
|
+
# both be true for one, and the half that is true is the half that
|
|
75
|
+
# matters.
|
|
76
|
+
def rebootable?(route)
|
|
77
|
+
route["kind"] == "socket"
|
|
43
78
|
end
|
|
44
79
|
|
|
45
80
|
def touch(hostname)
|
|
@@ -50,10 +85,14 @@ module Yamine
|
|
|
50
85
|
|
|
51
86
|
# Boot a stopped app on demand. Returns the route (unchanged — the
|
|
52
87
|
# target path is deterministic) or nil on boot failure.
|
|
88
|
+
#
|
|
89
|
+
# A tcp route returns unchanged even when its backend is down: the
|
|
90
|
+
# proxy only ever asks about socket routes, and there is nothing this
|
|
91
|
+
# could rebuild (see rebootable?).
|
|
53
92
|
def ensure_running(route, timeout: 60)
|
|
54
|
-
return route unless supervised?(route)
|
|
93
|
+
return route unless supervised?(route) && rebootable?(route)
|
|
55
94
|
|
|
56
|
-
|
|
95
|
+
backend_alive?(route) ? route : boot(route, timeout)
|
|
57
96
|
end
|
|
58
97
|
|
|
59
98
|
# Start the background supervision thread.
|
|
@@ -83,27 +122,44 @@ module Yamine
|
|
|
83
122
|
|
|
84
123
|
hostname = route["hostname"]
|
|
85
124
|
st = state_for(hostname, route, now)
|
|
125
|
+
alive = backend_alive?(route)
|
|
126
|
+
# The latch below only exists to stop a second kill landing on a
|
|
127
|
+
# backend that is already down, and it is cleared the moment the
|
|
128
|
+
# app is back — booted on request, or `yamine start` again in that
|
|
129
|
+
# directory. Left one-way (as it was) a route was supervised for
|
|
130
|
+
# exactly one event in the life of the daemon, so the second
|
|
131
|
+
# `yamine restart` was the no-op this method exists to remove.
|
|
132
|
+
st[:restarting] = false if alive
|
|
86
133
|
next if st[:restarting]
|
|
87
134
|
|
|
88
135
|
if restart_changed?(route, st)
|
|
89
136
|
@on_event.call("restart.txt changed for #{hostname} — stopping backend")
|
|
90
137
|
kill_backend(route)
|
|
91
138
|
st[:restarting] = true
|
|
92
|
-
elsif !
|
|
93
|
-
@on_event.call("backend for #{hostname} is down —
|
|
139
|
+
elsif !alive
|
|
140
|
+
@on_event.call("backend for #{hostname} is down — #{recovery(route)}")
|
|
94
141
|
st[:restarting] = true
|
|
95
|
-
elsif idle?(st, now)
|
|
96
|
-
@on_event.call("#{hostname} idle — stopping backend (
|
|
142
|
+
elsif idle?(st, now) && idle_kill?(route)
|
|
143
|
+
@on_event.call("#{hostname} idle — stopping backend, #{recovery(route)}")
|
|
97
144
|
kill_backend(route)
|
|
98
145
|
st[:restarting] = true
|
|
99
146
|
end
|
|
100
147
|
end
|
|
101
148
|
end
|
|
102
149
|
|
|
103
|
-
# Kill every
|
|
150
|
+
# Kill every backend the daemon booted (daemon shutdown).
|
|
151
|
+
#
|
|
152
|
+
# Only the ones it would have booted back. A socket app exists
|
|
153
|
+
# because the daemon started it, so it goes down with it — that is
|
|
154
|
+
# the puma-dev contract and the reason this method exists. A
|
|
155
|
+
# `yamine start` app has a supervising process of its own and a
|
|
156
|
+
# lifecycle of its own: the proxy is the listener in front of it, not
|
|
157
|
+
# its parent, and restarting the proxy is no reason to take every
|
|
158
|
+
# developer's app on the machine down. Supervised (watched) is not
|
|
159
|
+
# the same set as owned (booted by us and answerable to us).
|
|
104
160
|
def shutdown
|
|
105
161
|
@store.load_routes.each do |route|
|
|
106
|
-
kill_backend(route) if supervised?(route)
|
|
162
|
+
kill_backend(route) if supervised?(route) && rebootable?(route)
|
|
107
163
|
end
|
|
108
164
|
end
|
|
109
165
|
|
|
@@ -114,6 +170,60 @@ module Yamine
|
|
|
114
170
|
(now - st[:last_used]) > @idle_timeout
|
|
115
171
|
end
|
|
116
172
|
|
|
173
|
+
# Is the app behind this route serving right now? Both kinds, one
|
|
174
|
+
# question: a socket route answers a connect on its socket file, a
|
|
175
|
+
# tcp route answers a connect on its port. The daemon needs this
|
|
176
|
+
# rather than a socket-only probe because tcp routes are now watched
|
|
177
|
+
# (see supervised?) — and it is the same question
|
|
178
|
+
# CLI::Context#backend_alive? answers for `yamine list`, which is
|
|
179
|
+
# what makes the two surfaces able to disagree about one app.
|
|
180
|
+
def backend_alive?(route)
|
|
181
|
+
case route["kind"]
|
|
182
|
+
when "socket" then socket_alive?(route)
|
|
183
|
+
when "tcp"
|
|
184
|
+
host, port = route["target"].to_s.split(":", 2)
|
|
185
|
+
# Bounded: this runs in the daemon's single supervision loop, so
|
|
186
|
+
# a connect that hangs on a full accept backlog would stall every
|
|
187
|
+
# other route's supervision behind it.
|
|
188
|
+
Socket.tcp(host, port.to_i, connect_timeout: CONNECT_TIMEOUT) { |sock| sock.close }
|
|
189
|
+
true
|
|
190
|
+
else
|
|
191
|
+
false
|
|
192
|
+
end
|
|
193
|
+
rescue SystemCallError, IOError
|
|
194
|
+
false
|
|
195
|
+
end
|
|
196
|
+
|
|
197
|
+
# Stop an idle backend only when the daemon can bring it back.
|
|
198
|
+
#
|
|
199
|
+
# Idle-kill is the half of puma-dev that depends on the other half: it
|
|
200
|
+
# promises "stopped, boots on next request", and for a tcp route the
|
|
201
|
+
# daemon cannot keep that promise (see rebootable?). Killing one
|
|
202
|
+
# anyway would take a developer's app down for an afternoon and leave
|
|
203
|
+
# a 503 in place of it, so it is opt-in — YAMINE_IDLE_TCP=1 — rather
|
|
204
|
+
# than a default nobody asked for. Watched is not idle-killed:
|
|
205
|
+
# restart.txt and crash detection still work for these routes.
|
|
206
|
+
def idle_kill?(route)
|
|
207
|
+
return true if rebootable?(route)
|
|
208
|
+
|
|
209
|
+
idle_tcp_opt_in?
|
|
210
|
+
end
|
|
211
|
+
|
|
212
|
+
def idle_tcp_opt_in?
|
|
213
|
+
%w[1 true yes on].include?(ENV["YAMINE_IDLE_TCP"].to_s.strip.downcase)
|
|
214
|
+
end
|
|
215
|
+
|
|
216
|
+
# What happens next, in the one form that is true for both kinds: a
|
|
217
|
+
# socket app is rebooted on the next request, a tcp app has to be
|
|
218
|
+
# started again by hand. Saying "will boot on next request" for a
|
|
219
|
+
# route nothing will reboot is the exact lie this file stopped
|
|
220
|
+
# telling when tcp routes became supervised.
|
|
221
|
+
def recovery(route)
|
|
222
|
+
return "will boot on next request" if rebootable?(route)
|
|
223
|
+
|
|
224
|
+
"run `yamine start` in #{route.dig("spec", "dir") || "its directory"} again"
|
|
225
|
+
end
|
|
226
|
+
|
|
117
227
|
def socket_alive?(route)
|
|
118
228
|
target = route["target"]
|
|
119
229
|
return false unless target && File.socket?(target)
|
|
@@ -137,14 +247,25 @@ module Yamine
|
|
|
137
247
|
def kill_backend(route)
|
|
138
248
|
pid = backend_pid(route)
|
|
139
249
|
if pid && pid_alive?(pid)
|
|
250
|
+
# By group when the pid leads one, by pid when it does not. A tcp
|
|
251
|
+
# route's sidecar names the `sh -c` shell that `yamine start`
|
|
252
|
+
# wraps every command in, and TERM to that shell alone leaves the
|
|
253
|
+
# app behind it running (linux) — which would make a restarted
|
|
254
|
+
# tcp app keep serving while the route was re-registered, i.e. the
|
|
255
|
+
# old process under a new pid's name (ProcessTree). One syscall,
|
|
256
|
+
# no sleeping: the wait below still owns "and make sure it is
|
|
257
|
+
# gone".
|
|
140
258
|
begin
|
|
141
|
-
|
|
259
|
+
ProcessTree.term(pid)
|
|
142
260
|
wait_exit(pid, 10)
|
|
143
261
|
rescue SystemCallError
|
|
144
262
|
nil
|
|
145
263
|
end
|
|
146
264
|
end
|
|
147
|
-
|
|
265
|
+
# Only a socket target is a file we own and can leave behind; a tcp
|
|
266
|
+
# target is an address, and unlinking the string would only ever
|
|
267
|
+
# unlink a coincidence in the daemon's working directory.
|
|
268
|
+
FileUtils.rm_f(route["target"]) if route["kind"] == "socket" && route["target"]
|
|
148
269
|
FileUtils.rm_f(File.join(@store.dir, "backend-#{route["hostname"]}.pid"))
|
|
149
270
|
end
|
|
150
271
|
|
|
@@ -159,7 +280,18 @@ module Yamine
|
|
|
159
280
|
s = st(hostname)
|
|
160
281
|
# First sighting: baseline so a freshly booted app gets a full
|
|
161
282
|
# idle window and an existing restart.txt doesn't count as changed.
|
|
162
|
-
|
|
283
|
+
#
|
|
284
|
+
# Keyed on :baselined rather than on the mtime being nil, because
|
|
285
|
+
# "no restart.txt" IS a value here: a nil mtime re-baselined on
|
|
286
|
+
# every pass absorbs the first `yamine restart` an app ever gets
|
|
287
|
+
# — the file appears between two ticks, the new mtime becomes the
|
|
288
|
+
# baseline, and the change that was asked for is the change that
|
|
289
|
+
# gets missed. `yamine restart` in an app that had never been
|
|
290
|
+
# restarted did nothing at all, for either kind of route.
|
|
291
|
+
unless s[:baselined]
|
|
292
|
+
s[:restart_mtime] = restart_mtime(route)
|
|
293
|
+
s[:baselined] = true
|
|
294
|
+
end
|
|
163
295
|
s[:last_used] = now if s[:last_used].nil?
|
|
164
296
|
s
|
|
165
297
|
end
|
data/lib/yamine/version.rb
CHANGED