yamine 0.21.2 → 0.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -36,14 +36,7 @@ module Yamine
36
36
  routes = ctx.store.load_routes
37
37
  port = ctx.proxy_port
38
38
  tls = ctx.proxy_tls
39
- entries = routes.map do |r|
40
- { hostname: r["hostname"],
41
- url: Hostname.url(r["hostname"], port: port, tls: tls),
42
- target: r["target"], kind: r["kind"],
43
- pid: r["pid"], agent: r["agent"],
44
- supervised: !r["spec"].nil?,
45
- alive: alive_state(ctx, r) }
46
- end
39
+ entries = routes.map { |r| entry_for(ctx, r, port: port, tls: tls) }
47
40
  if json
48
41
  require "json"
49
42
  puts JSON.generate({ routes: entries, proxy_port: port, tls: tls })
@@ -61,6 +54,25 @@ module Yamine
61
54
  puts
62
55
  end
63
56
 
57
+ # One route as `yamine list` reports it.
58
+ #
59
+ # `pid` stays the route's recorded owner (the yamine process that
60
+ # registered it — that is what routes.json holds and what
61
+ # `yamine stop` and the ownership gate compare), and `backend_pid`
62
+ # is added beside it because `alive` is now the app's state: an
63
+ # agent reading `alive: running` next to a `pid` that has since
64
+ # exited could not tell which process the verdict was about. Both
65
+ # are in the payload, so nothing that was readable is lost.
66
+ def entry_for(ctx, route, port:, tls:)
67
+ { hostname: route["hostname"],
68
+ url: Hostname.url(route["hostname"], port: port, tls: tls),
69
+ target: route["target"], kind: route["kind"],
70
+ pid: route["pid"], backend_pid: ctx.backend_pid_for(route),
71
+ agent: route["agent"],
72
+ supervised: !route["spec"].nil?,
73
+ alive: alive_state(ctx, route) }
74
+ end
75
+
64
76
  # Shared discovery: `get --all` lists every live route (any owner)
65
77
  # with its URL and agent, so one agent can find another's services
66
78
  # without coupling. `--json` emits the same stable keys as list.
@@ -89,24 +101,64 @@ module Yamine
89
101
  end
90
102
  end
91
103
 
104
+ # Liveness of the APP, not of the yamine process that registered
105
+ # the route. `route["pid"]` is that process's own pid — Runner
106
+ # passes Process.pid at every add_route call site — and it outlives
107
+ # the app it booted, so asking it whether the app is alive answered
108
+ # a different question. Every crashed backend read as "running",
109
+ # in the one command an agent would reach for to find out. The
110
+ # app's pid is in the sidecar the boot wrote
111
+ # (state_dir/backend-<hostname>.pid), the same file `yamine stop`
112
+ # reads for exactly this reason.
113
+ #
114
+ # Five states, and the split that matters is running vs
115
+ # backend-gone:
116
+ #
117
+ # running the app's process is there
118
+ # backend-gone the CLI is still up, the app it booted is not
119
+ # owner-gone nothing is there and no app pid was recorded
120
+ # unknown no app pid recorded, and the route's own process
121
+ # is — unknowable, deliberately not "down"
122
+ # reachable / a static alias (pid 0) names no process at all,
123
+ # unreachable so it reports the probe and nothing else
124
+ #
125
+ # `unknown` is the honest answer for a route with no sidecar: one
126
+ # written by a yamine old enough not to write sidecars, or by a
127
+ # boot that died between registering the route and writing the
128
+ # file. There is no evidence of a dead app there, and calling it
129
+ # down would have every pre-existing route on the machine read as
130
+ # broken after an upgrade.
92
131
  def alive_state(ctx, route)
93
- if route["pid"] == 0
94
- ctx.backend_alive?(route) ? "reachable" : "unreachable"
95
- elsif ProxyControl.pid_alive?(route["pid"])
96
- "running"
97
- else
98
- "owner-gone"
99
- end
132
+ return ctx.backend_alive?(route) ? "reachable" : "unreachable" if route["pid"] == 0
133
+
134
+ backend = ctx.backend_pid_for(route)
135
+ return owner_state(ctx, route) if backend.nil?
136
+
137
+ ProxyControl.pid_alive?(backend) ? "running" : "backend-gone"
100
138
  rescue StandardError
101
139
  "unknown"
102
140
  end
103
141
 
142
+ # What we know with no app pid to ask: whether the process that
143
+ # registered the route is still there. `owner-gone` keeps the
144
+ # meaning it has always had — the route outlived its owner, which
145
+ # is the case load_routes leaves behind for `yamine prune`.
146
+ def owner_state(ctx, route)
147
+ ProxyControl.pid_alive?(route["pid"]) ? "unknown" : "owner-gone"
148
+ end
149
+
150
+ # The human line. The pid shown next to the state is the one the
151
+ # state is about: the app's, when there is one. Printing the route
152
+ # owner's pid beside "running" is how a dead app kept reading as a
153
+ # healthy one.
104
154
  def label_for(entry)
105
155
  owner = entry[:agent] ? " #{entry[:agent]}" : ""
106
156
  if entry[:pid] == 0
107
157
  "(alias, #{entry[:alive]}#{owner})"
158
+ elsif entry[:backend_pid]
159
+ "(backend #{entry[:backend_pid]}, #{entry[:alive]}#{owner})"
108
160
  else
109
- "(pid #{entry[:pid]}, #{entry[:alive]}#{owner})"
161
+ "(owner #{entry[:pid]}, #{entry[:alive]}#{owner})"
110
162
  end
111
163
  end
112
164
 
@@ -115,6 +167,7 @@ module Yamine
115
167
  # report the probe, not a process.
116
168
  def route_label(ctx, route)
117
169
  entry = { pid: route["pid"], alive: alive_state(ctx, route) }
170
+ entry[:backend_pid] = ctx.backend_pid_for(route)
118
171
  label_for(entry)
119
172
  end
120
173
 
@@ -247,13 +300,37 @@ module Yamine
247
300
  2
248
301
  end
249
302
 
250
- # Touch tmp/restart.txt so a supervised managed app reboots.
251
- def restart(_ctx, _args)
303
+ # Touch tmp/restart.txt so a supervised app's backend is stopped.
304
+ #
305
+ # What happens next is not the same for every route, so it is said
306
+ # rather than assumed: a managed socket app is rebooted on the next
307
+ # request, while a `yamine start` app is stopped and has to be
308
+ # started again (the daemon cannot rebuild a tcp backend — see
309
+ # Supervisor#rebootable?). The old one-liner promised a reboot for
310
+ # both, and for a tcp route nothing was even watching the file.
311
+ def restart(ctx, _args)
252
312
  path = File.join(Dir.pwd, "tmp", "restart.txt")
253
313
  require "fileutils"
254
314
  FileUtils.mkdir_p(File.dirname(path))
255
315
  FileUtils.touch(path)
256
- puts "Touched #{path} — managed app restarts on next request."
316
+ puts "Touched #{path}."
317
+ # The routes this directory registered, found by spec.dir rather
318
+ # than by resolving the app: the file was written either way, and
319
+ # a directory with no config/local.yml must not turn a no-op into
320
+ # a config error.
321
+ here = File.expand_path(Dir.pwd)
322
+ entries = ctx.store.load_routes.select { |r| r.dig("spec", "dir") == here }
323
+ if entries.empty?
324
+ puts "No yamine app is registered for this directory — nothing will be restarted."
325
+ return
326
+ end
327
+ entries.each do |entry|
328
+ if entry["kind"] == "socket"
329
+ puts "#{entry["hostname"]} reboots on the next request."
330
+ else
331
+ puts "#{entry["hostname"]} will be stopped; start it again with `yamine start`."
332
+ end
333
+ end
257
334
  end
258
335
 
259
336
  # Tail the shared app log (default 50 lines); --follow streams.
@@ -946,23 +946,33 @@ module Yamine
946
946
 
947
947
  yamine start # block until every route is healthy, then supervise
948
948
  yamine start --no-wait # fire-and-forget (register routes immediately)
949
+ yamine start --detach # boot in the background, return once the app is healthy
949
950
  yamine start --json # machine-readable result payload (--wait only)
950
951
  yamine start -- --help # pass --help to the app, not here
951
952
 
952
953
  Options are passed through to the boot path:
953
954
  --variant <v> --tld <tld> --force --app-port <port>
954
- --wait (alias, default) --no-wait --json
955
+ --wait (alias, default) --no-wait --json --detach
955
956
 
956
957
  Default waits: every process is spawned concurrently, each
957
958
  healthcheck (or TCP accept when none is declared) is polled,
958
959
  routes are registered only when all are healthy, and `ready:`
959
960
  is printed with exit 0. On failure everything spawned is
960
961
  killed, the failed process + its log tail is printed, and
961
- the CLI exits 1 — no half-booted routes. `--no-wait` keeps
962
- the old fire-and-forget path.
962
+ the CLI exits 1 — no half-booted routes. `--no-wait` keeps the
963
+ old fire-and-forget path.
964
+
965
+ --detach forks the boot: the child owns the tree, its pid is
966
+ the one recorded in routes.json, and this process waits for
967
+ the app to answer, prints the URL, that pid and the log path
968
+ under the state dir, then exits 0. It is idempotent — a tree
969
+ already running for this directory is reported, not started
970
+ again. --no-wait is ignored with it: the point of detaching is
971
+ that the command returns once the app is actually serving.
963
972
 
964
973
  Setup failures become hard errors pointing at `yamine setup`;
965
974
  non-interactive CI without a running proxy exits immediately.
975
+
966
976
  HELP
967
977
  return
968
978
  end
data/lib/yamine/cli.rb CHANGED
@@ -83,6 +83,7 @@ module Yamine
83
83
 
84
84
  Usage:
85
85
  yamine start One-setup-and-go: setup if needed, then boot -> https://<app>.localhost
86
+ yamine start --detach Same, in the background; returns once the app is healthy
86
87
  yamine setup One-shot workstation setup without booting (run once)
87
88
  yamine Bare form of `start` — boots every process in config/local.yml
88
89
  yamine get <name> Print URL for a service
@@ -103,10 +104,10 @@ module Yamine
103
104
  yamine hosts sync|clean Manage /etc/hosts entries
104
105
  yamine kamal <variant> Preview-deploy snippet for Kamal
105
106
  yamine stop Stop this app's backend + routes
106
- yamine restart Touch tmp/restart.txt
107
+ yamine restart Touch tmp/restart.txt (a supervised app is stopped)
107
108
  yamine log [-F] [n] Tail (or follow) log/development.log
108
109
 
109
- Flags: --variant, --tld, --force, --app-port, --wait (default), --no-wait, --json, --branch
110
+ Flags: --variant, --tld, --force, --app-port, --wait (default), --no-wait, --detach, --json, --branch
110
111
  Env: YAMINE_VARIANT/TLD/PORT/STATE_DIR/AGENT, YAMINE_BRANCH=1
111
112
  HELP
112
113
  end
@@ -57,9 +57,58 @@ module Yamine
57
57
  # "can't be called from trap context" — a stop path that only works
58
58
  # outside a trap is a stop that never happens on Ctrl-C.
59
59
  PGROUP_LEADERS = {}
60
+ # Exit statuses of what we spawned, by pid, filled in by the reaper
61
+ # that `detach` starts. Same reasoning, same shape, and the same
62
+ # bounded cost as PGROUP_LEADERS above: an entry is a few dozen bytes
63
+ # per process this one ever spawned, dropped by `forget` whenever
64
+ # the pid is signalled.
65
+ STATUSES = {}
60
66
 
61
67
  module_function
62
68
 
69
+ # Spawn a child and reap it without blocking us, keeping what it
70
+ # exited with. A drop-in for Process.detach — nobody waits on a live
71
+ # backend, a dead pid is all the boot needs — that also keeps the one
72
+ # thing only a reaper can know. Without it the answer dies with the
73
+ # child: `kill(0, pid)` says a process is gone and nothing says
74
+ # whether it exited 0 or was killed, so "web is down" is all a
75
+ # supervisor could ever report.
76
+ #
77
+ # Unlocked, like PGROUP_LEADERS, and a Hash read from a supervision
78
+ # thread: every operation is a single call the GVL makes atomic.
79
+ def detach(pid)
80
+ return pid unless pid.to_i.positive?
81
+
82
+ Thread.new do
83
+ _waited, status = ::Process.waitpid2(pid)
84
+ STATUSES[pid] = status
85
+ rescue SystemCallError
86
+ nil
87
+ end
88
+ pid
89
+ end
90
+
91
+ # Process::Status for a pid we spawned, or nil when there is nothing
92
+ # to report: not our child, or it has not been reaped yet.
93
+ #
94
+ # Never blocks, which is the whole point — the caller is a loop that
95
+ # has other routes to watch, and asking a live child how it is doing
96
+ # would hang it. So a status that has not arrived yet is nil, and the
97
+ # caller asks again on its next pass.
98
+ def status(pid)
99
+ recorded = STATUSES[pid]
100
+ return recorded if recorded
101
+
102
+ # Not one of ours to have reaped (a pid out of a route entry, a
103
+ # double in a test): ask the kernel. WNOHANG so a live pid is not
104
+ # waited on, and ECHILD — which is the answer for anything that is
105
+ # not a child of ours — is a nil, not a failure.
106
+ _waited, status = ::Process.waitpid2(pid, ::Process::WNOHANG)
107
+ status
108
+ rescue StandardError
109
+ nil
110
+ end
111
+
63
112
  # Spawn a boot process as its own group leader and record that it is
64
113
  # one. This is the only place a boot process is created, so "every
65
114
  # process we boot leads a group" is one fact in one place rather
@@ -85,8 +134,14 @@ module Yamine
85
134
  PGROUP_LEADERS.key?(pid)
86
135
  end
87
136
 
137
+ # Drop every record of a pid we have finished with. Both tables are
138
+ # about a spawn, not about a number that has to keep meaning one:
139
+ # leaving them behind is how a pid that later gets recycled gets
140
+ # signalled as a group it never led, or reported with an exit status
141
+ # that belongs to somebody else.
88
142
  def forget(pid)
89
143
  PGROUP_LEADERS.delete(pid)
144
+ STATUSES.delete(pid)
90
145
  end
91
146
 
92
147
  # Does this pid lead its own process group? Ours, if we spawned it.