yamine 0.21.2 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +132 -0
- data/README.md +39 -12
- data/lib/ask/skills/yamine/SKILL.md +20 -1
- data/lib/yamine/cli/boot.rb +291 -26
- data/lib/yamine/cli/context.rb +1 -1
- data/lib/yamine/cli/routes.rb +96 -19
- data/lib/yamine/cli/system.rb +13 -3
- data/lib/yamine/cli.rb +3 -2
- data/lib/yamine/process_tree.rb +55 -0
- data/lib/yamine/proxy.rb +215 -46
- data/lib/yamine/runner.rb +10 -4
- data/lib/yamine/supervisor.rb +147 -15
- data/lib/yamine/version.rb +1 -1
- metadata +1 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 69848a3d38cd9997ca18e52ea892aa57962524e0cd41df16199a651409e86b76
|
|
4
|
+
data.tar.gz: aebe3dd18c4bab95865491850bb940ca9a7e42c83f53d851b4d3ddd691f80305
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 7c1bb5ef1104c29a53c8e4274a5b4a42f7231f895f40bb7d12defc97dee72232944340c24b4bec62b9a736d29bdab01d975b32cf6e1af7aaf7f1b643ec0ac602
|
|
7
|
+
data.tar.gz: 66ef5d87789c27a94994c0d24f1567c1a762d592ecbfb00b3d95db8a7f538f3c6868be3703a9870a88d0101aabb37dcfaba8f6db85b2e4a74799b4a4381c7280
|
data/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,138 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.22.0] - 2026-09-29
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- **`yamine start --detach` — boot in the background and get your prompt
|
|
10
|
+
back.** An agent (or a script) that needs an app running had one
|
|
11
|
+
option: block, or reach for `nohup` — which leaves a process tree
|
|
12
|
+
whose owner the first Ctrl-C cannot reach. `--detach` forks the boot:
|
|
13
|
+
the child takes its own session, records its pid in
|
|
14
|
+
`~/.yamine/start-<hostname>.pid`, sends its own output to
|
|
15
|
+
`~/.yamine/start-<hostname>.log`, boots, keeps the tree supervised and
|
|
16
|
+
outlives the shell that started it. The parent waits for the app to
|
|
17
|
+
answer, prints the URL, that pid and the log path, and exits 0.
|
|
18
|
+
The boot happens in the child on purpose: `add_route` records
|
|
19
|
+
`Process.pid`, and `load_routes` prunes any route whose pid is dead, so
|
|
20
|
+
a parent that registered the routes and exited would have its own route
|
|
21
|
+
pruned on the next read and the proxy would 503 an app that is running
|
|
22
|
+
perfectly well. It is idempotent — a tree already running for the
|
|
23
|
+
directory (detached, or foreground) is reported, not started again —
|
|
24
|
+
and a boot that never becomes healthy exits 1 with the log instead of
|
|
25
|
+
a tree that is not there. `--json` gets the same three facts
|
|
26
|
+
as a payload. `--no-wait` is ignored with `--detach`: returning before
|
|
27
|
+
the app answers is the lie this flag exists to stop.
|
|
28
|
+
|
|
29
|
+
### Fixed
|
|
30
|
+
|
|
31
|
+
- **`yamine list` and `yamine get --all` report the app, not the CLI
|
|
32
|
+
that started it.** A route's `pid` is the yamine process that
|
|
33
|
+
registered it, and it outlives the app it booted, so `alive_state` was
|
|
34
|
+
asking the wrong process whether the app was alive: every crashed
|
|
35
|
+
backend read as `running`, in the one command an agent would reach for
|
|
36
|
+
to find out. Liveness now comes from the sidecar the boot already
|
|
37
|
+
writes (`backend-<hostname>.pid`, the same file `yamine stop` reads),
|
|
38
|
+
and the vocabulary says what it knows:
|
|
39
|
+
`running` (the app's process is there), `backend-gone` (the CLI is
|
|
40
|
+
still up, the app it booted is not), `owner-gone` (nothing is there
|
|
41
|
+
and no app pid was recorded), `unknown` (no app pid recorded, and the
|
|
42
|
+
route's own process is — a route written before sidecars existed, or
|
|
43
|
+
by a boot that died between registering the route and writing the
|
|
44
|
+
file: `unknown`, never "down", so no pre-existing route reads as
|
|
45
|
+
broken after an upgrade), and `reachable` / `unreachable` for a static
|
|
46
|
+
alias, which names no process and keeps reporting the probe.
|
|
47
|
+
`list --json` gained `backend_pid` beside `pid` so the state and the
|
|
48
|
+
process it is about travel together, and the human line prints the
|
|
49
|
+
backend's pid rather than the owner's.
|
|
50
|
+
- **`yamine restart` works for the apps yamine started.** Supervision
|
|
51
|
+
required `kind == "socket"`, and `yamine start` registers `kind
|
|
52
|
+
"tcp"`, so a `yamine start` tree was invisible to the daemon:
|
|
53
|
+
`yamine restart` touched `tmp/restart.txt`, printed "managed app
|
|
54
|
+
restarts on next request", and nothing was watching the file. Any
|
|
55
|
+
route carrying a spec is now watched, so restart.txt and crash
|
|
56
|
+
detection reach both kinds. Two things that were broken *inside* the
|
|
57
|
+
socket case went with it: the restart baseline was re-derived while
|
|
58
|
+
`restart.txt` did not exist, so the *first* `yamine restart` an app
|
|
59
|
+
ever got was absorbed as the new baseline and did nothing; and the
|
|
60
|
+
"restarting" latch was never cleared, so a route was supervised for
|
|
61
|
+
exactly one event in the life of the daemon. The kill now signals the
|
|
62
|
+
process group when the pid leads one, because a tcp route's sidecar
|
|
63
|
+
names the `sh -c` shell the boot wrapped the app in.
|
|
64
|
+
- **A child that dies no longer ends `yamine start` with exit 0.**
|
|
65
|
+
`supervise_tree` exited 0 with the routes already removed and the rest
|
|
66
|
+
of the tree being killed, so a zero read as "the app came up" to
|
|
67
|
+
anything scripted — which then blamed the app for the 503 it was
|
|
68
|
+
about to get. It exits 1, the same code `--wait` already uses for a
|
|
69
|
+
boot that never became healthy, and the message names the process, its
|
|
70
|
+
exit status (or the signal that took it down — a reaped child is the
|
|
71
|
+
only place that answer exists, so the reaper keeps it) and the tail of
|
|
72
|
+
the app log. `--json` gets the same as
|
|
73
|
+
`{"ok": false, "error": "child-exited", …}`. `yamine stop`'s
|
|
74
|
+
0/2/3/4 are untouched: those answer "what did the stop do", and this
|
|
75
|
+
answers "is the app up".
|
|
76
|
+
- **`yamine restart` no longer promises a reboot it cannot perform.**
|
|
77
|
+
A managed socket app is rebooted on the next request; a `yamine start`
|
|
78
|
+
app is stopped and has to be started again, because the daemon cannot
|
|
79
|
+
rebuild a tcp backend (its port is a free port chosen at boot and the
|
|
80
|
+
route records no command to re-run). The command and the daemon's own
|
|
81
|
+
events now say which one you have.
|
|
82
|
+
- **A boot that started nothing no longer reports `ready:` and hangs.**
|
|
83
|
+
A process whose `cmd` is a compound line is refused with an error and
|
|
84
|
+
leaves the spawn plan empty; the boot then printed `ready:` and sat in
|
|
85
|
+
the supervision loop for ever with nothing to watch — a hang that looks
|
|
86
|
+
exactly like a healthy start. It exits 1 instead, naming the error
|
|
87
|
+
above it.
|
|
88
|
+
|
|
89
|
+
### Changed
|
|
90
|
+
|
|
91
|
+
- **A `yamine start` app is watched by the proxy daemon, but is not
|
|
92
|
+
idle-killed.** This is the one visible consequence of the supervision
|
|
93
|
+
fix above, so it is called out rather than buried: previously *no*
|
|
94
|
+
`yamine start` app was ever idle-killed, and now restart.txt and crash
|
|
95
|
+
detection work — but the idle clock still does not touch them, because
|
|
96
|
+
idle-kill is the half of puma-dev that depends on the other half
|
|
97
|
+
("stopped, boots on next request"), and the daemon cannot keep that
|
|
98
|
+
promise for a tcp route. Killing one would take a developer's app down
|
|
99
|
+
for an afternoon and leave a 503 where it was. Set `YAMINE_IDLE_TCP=1`
|
|
100
|
+
for the puma-dev behaviour on `yamine start` apps, with the same
|
|
101
|
+
`YAMINE_IDLE_TIMEOUT` clock (15 minutes by default; that is the
|
|
102
|
+
supervisor's clock, distinct from the proxy's own
|
|
103
|
+
`YAMINE_PROXY_IDLE_TIMEOUT`). For the same reason `yamine proxy stop`
|
|
104
|
+
no longer stops `yamine start` apps along with the socket ones the
|
|
105
|
+
daemon booted: those have a supervising process of their own.
|
|
106
|
+
|
|
107
|
+
### Fixed
|
|
108
|
+
|
|
109
|
+
- **A slow app and a dead app no longer come back as the same useless
|
|
110
|
+
502.** Waiting for the backend's response *head* and waiting for body
|
|
111
|
+
bytes shared one clock, and `read_head` swallowed its own timeout into
|
|
112
|
+
the same `[nil, ...]` a dead backend returns — so `pipe_request` could
|
|
113
|
+
not tell them apart and every one of them got the same 60-byte page,
|
|
114
|
+
"The target app is not responding." A font request that took 60,068ms
|
|
115
|
+
came back with no hostname, no target, no owner, no directory and no
|
|
116
|
+
next step. The head now has its own budget
|
|
117
|
+
(`YAMINE_PROXY_HEAD_TIMEOUT`, default 60s — the same bound it already
|
|
118
|
+
got, so nothing that works today starts failing), the body keeps
|
|
119
|
+
`YAMINE_PROXY_IDLE_TIMEOUT` untouched, and a silent backend is reported
|
|
120
|
+
as what it is: *the backend at 127.0.0.1:3000 accepted the connection,
|
|
121
|
+
then sent no response for 60 seconds*, with the owning agent, the app's
|
|
122
|
+
own directory (`spec.dir`, not the proxy's cwd — inside a worktree that
|
|
123
|
+
is the wrong checkout), the path to its `log/development.log`, and the
|
|
124
|
+
`cd <dir> && yamine start` that fixes it. Machines get the same split
|
|
125
|
+
in a header: `x-yamine-error: backend-refused` or `backend-silent`.
|
|
126
|
+
A body clock that was tuned tight for one and a head clock that is not
|
|
127
|
+
are no longer the same dial, so a streaming response that goes quiet
|
|
128
|
+
between bytes still runs indefinitely.
|
|
129
|
+
- **A request that was merely slow now gets a second attempt.** One
|
|
130
|
+
retry, and only where a retry cannot duplicate work: a refused dial
|
|
131
|
+
never reached the app, so any method replays; a head timeout means the
|
|
132
|
+
head *did* arrive, so only bodiless GET, HEAD and OPTIONS replay. A
|
|
133
|
+
GET that takes 61 seconds to answer now serves instead of 502ing. A
|
|
134
|
+
POST never replays — the body is already consumed off the client socket
|
|
135
|
+
and cannot be resent faithfully, and a duplicated message is worse than
|
|
136
|
+
a slow page.
|
|
5
137
|
## [0.21.2] — 2026-09-29
|
|
6
138
|
|
|
7
139
|
### Fixed
|
data/README.md
CHANGED
|
@@ -355,11 +355,12 @@ as the app answering 404. `yamine status` reports which mode an app is in.
|
|
|
355
355
|
|
|
356
356
|
```bash
|
|
357
357
|
yamine # boot app (waits until healthy, then supervises)
|
|
358
|
+
yamine start --detach # boot in the background; returns once healthy, prints url/pid/log
|
|
358
359
|
yamine start --no-wait # fire-and-forget (register routes immediately)
|
|
359
360
|
yamine start --json # machine-readable wait result (--wait default)
|
|
360
361
|
yamine get <name> # print URL for cross-service wiring
|
|
361
362
|
yamine alias <name> <port> # static route (e.g. Docker)
|
|
362
|
-
yamine list [--json] # show active routes (+
|
|
363
|
+
yamine list [--json] # show active routes (+ the APP's liveness)
|
|
363
364
|
yamine status [--json] # show effective naming context here
|
|
364
365
|
yamine doctor [--json] # machine-readable health checks
|
|
365
366
|
yamine open [name] # open the app URL in a browser
|
|
@@ -369,7 +370,7 @@ yamine prune # remove stale routes
|
|
|
369
370
|
yamine db list|create|drop|describe # per-worktree databases (multi-database aware)
|
|
370
371
|
yamine worktree list|add|remove|clean # worktree lifecycle
|
|
371
372
|
yamine stop # stop this app's backend + routes
|
|
372
|
-
yamine restart # touch tmp/restart.txt (
|
|
373
|
+
yamine restart # touch tmp/restart.txt (a supervised app's backend is stopped)
|
|
373
374
|
yamine log [-F] [n] # tail (or follow) log/development.log
|
|
374
375
|
yamine proxy start|stop # control the proxy
|
|
375
376
|
yamine service install|status|uninstall # root-owned OS startup service
|
|
@@ -413,7 +414,19 @@ in-process). Hostnames that fall outside the configured TLDs get a bare
|
|
|
413
414
|
(healthcheck path when declared, TCP accept otherwise). On failure it
|
|
414
415
|
exits 1 with the failed process, its phase, and the tail of its own log
|
|
415
416
|
— no guessing, no polling, no half-booted routes. `--no-wait` keeps the
|
|
416
|
-
old fire-and-forget path.
|
|
417
|
+
old fire-and-forget path. A child that dies later ends the run the same
|
|
418
|
+
way: exit 1, naming the process, its exit status (or the signal that
|
|
419
|
+
took it down) and the tail of its log. Zero means the app is up.
|
|
420
|
+
|
|
421
|
+
`yamine start --detach` boots the same tree into the background and
|
|
422
|
+
returns once it is healthy, printing the URL, the pid that owns the tree
|
|
423
|
+
and its log path under the state dir (`start-<hostname>.pid` /
|
|
424
|
+
`start-<hostname>.log`). The boot runs in the child, so that pid is the
|
|
425
|
+
one recorded in `routes.json` — the route outlives the command. It is
|
|
426
|
+
idempotent (a tree already running for the directory is reported, not
|
|
427
|
+
restarted), it ignores `--no-wait` (returning before the app answers is
|
|
428
|
+
the thing it exists to prevent), and `--json` returns
|
|
429
|
+
`{ok, url, pid, log_path, started}`.
|
|
417
430
|
|
|
418
431
|
## Log rotation
|
|
419
432
|
|
|
@@ -424,15 +437,29 @@ lose the tail. `doctor` warns when the state dir passes 100MB.
|
|
|
424
437
|
|
|
425
438
|
## Supervision
|
|
426
439
|
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
440
|
+
Apps are supervised by the proxy daemon, not the CLI. A route is
|
|
441
|
+
supervised when it names a directory to boot from (`spec.dir`), which
|
|
442
|
+
is every app yamine boots — managed socket apps *and* `yamine start`
|
|
443
|
+
trees. Static aliases are not.
|
|
444
|
+
|
|
445
|
+
- touching `tmp/restart.txt` stops the backend. A managed socket app
|
|
446
|
+
then reboots on the next request; a `yamine start` app has to be
|
|
447
|
+
started again (`yamine restart` says which one you have)
|
|
448
|
+
- crashed backends are detected, and a managed app is rebooted on the
|
|
449
|
+
next request
|
|
450
|
+
- a managed app idle-kills after 15 minutes (`YAMINE_IDLE_TIMEOUT`
|
|
451
|
+
seconds; `0` disables) and boots transparently on the next request
|
|
452
|
+
- daemon shutdown stops the backends the daemon booted (no orphans)
|
|
453
|
+
|
|
454
|
+
A `yamine start` app is **watched but not idle-killed**, and is not
|
|
455
|
+
stopped when the proxy stops. Idle-kill is the half of puma-dev that
|
|
456
|
+
depends on the other half — "stopped, boots on next request" — and the
|
|
457
|
+
daemon can only keep that promise for a socket app: a tcp route's port
|
|
458
|
+
is a free port chosen at boot and the route records no command to
|
|
459
|
+
re-run, so nothing could bring it back. Set `YAMINE_IDLE_TCP=1` to get
|
|
460
|
+
the puma-dev behaviour for `yamine start` apps too (same
|
|
461
|
+
`YAMINE_IDLE_TIMEOUT` clock — the supervisor's, distinct from the
|
|
462
|
+
proxy's own `YAMINE_PROXY_IDLE_TIMEOUT`).
|
|
436
463
|
|
|
437
464
|
## Ask ecosystem integration
|
|
438
465
|
|
|
@@ -36,6 +36,7 @@ no extra gem — the proxied hostname is allowed automatically via
|
|
|
36
36
|
yamine start # setup if needed, then boot every process (waits until healthy)
|
|
37
37
|
yamine start --no-wait # fire-and-forget (register routes immediately)
|
|
38
38
|
yamine start --json # machine-readable wait result (--wait default)
|
|
39
|
+
yamine start --detach # boot in the background; returns once healthy, prints url/pid/log
|
|
39
40
|
yamine # same as start
|
|
40
41
|
yamine stop # stop this app's backend + routes
|
|
41
42
|
yamine status # show service, processes, and URLs
|
|
@@ -44,7 +45,16 @@ yamine log [-F] # tail log/development.log (every process)
|
|
|
44
45
|
|
|
45
46
|
`$PORT` and `YAMINE_URL` are injected per process; HTTP processes get
|
|
46
47
|
stable URLs, background ones are supervised without routes. A process
|
|
47
|
-
that exits cleans up the whole tree
|
|
48
|
+
that exits cleans up the whole tree — and ends the run with exit 1,
|
|
49
|
+
naming that process, its exit status and the tail of its log. Zero from
|
|
50
|
+
`yamine start` means the app is up.
|
|
51
|
+
|
|
52
|
+
`--detach` is the one to use from a tool: it forks the boot, so the
|
|
53
|
+
command returns with the URL, the pid that owns the tree (recorded in
|
|
54
|
+
`routes.json`, so the route outlives the command) and that tree's log
|
|
55
|
+
path under `~/.yamine/`. Run it again and it reports the running tree
|
|
56
|
+
instead of starting a second one; `yamine stop` stops it. Do not reach
|
|
57
|
+
for `nohup` — the tree yamine starts is the one `yamine stop` can reach.
|
|
48
58
|
|
|
49
59
|
The boot narrates every phase — deps, db, schema, and each process's
|
|
50
60
|
healthcheck — one line per phase, so a slow boot is never a black box:
|
|
@@ -193,6 +203,15 @@ yamine prune # clear stale routes from crashed sessions
|
|
|
193
203
|
yamine start --json # boot readiness payload (pass/fail + log tail)
|
|
194
204
|
```
|
|
195
205
|
|
|
206
|
+
`yamine list` reports the state of the **app**, not of the yamine
|
|
207
|
+
process that started it: `running`, `backend-gone` (the app died, the
|
|
208
|
+
process that booted it is still there), `owner-gone` (both gone),
|
|
209
|
+
`unknown` (no backend pid recorded — nothing is known, it is not a
|
|
210
|
+
failure), and `reachable` / `unreachable` for a `yamine alias`, which
|
|
211
|
+
points at a port rather than a process. `yamine restart` stops the
|
|
212
|
+
backend; a managed app reboots on the next request, a `yamine start` app
|
|
213
|
+
has to be started again.
|
|
214
|
+
|
|
196
215
|
If a hostname does not resolve — or boot/doctor report it as not in
|
|
197
216
|
`/etc/hosts` and you use clients that read only that file (CGO-disabled
|
|
198
217
|
Go binaries): `yamine hosts sync`. If the browser warns about TLS:
|
data/lib/yamine/cli/boot.rb
CHANGED
|
@@ -8,6 +8,13 @@ module Yamine
|
|
|
8
8
|
# No inference, no Procfile at boot, no single-process default.
|
|
9
9
|
module BootCommand
|
|
10
10
|
PORT_IGNORING = %w[jekyll middleman bridgetown].freeze
|
|
11
|
+
# How long the detaching process waits for the app to answer before
|
|
12
|
+
# it reports that it did not. Generous on purpose: the boot running
|
|
13
|
+
# inside the child has its own budget per phase (Readiness's, 45s a
|
|
14
|
+
# process, plus deps/db/schema), and this is the parent giving up
|
|
15
|
+
# on a child still doing legitimate work. A boot that fails ends
|
|
16
|
+
# the wait on its own, long before this.
|
|
17
|
+
DETACH_TIMEOUT = 300
|
|
11
18
|
|
|
12
19
|
module_function
|
|
13
20
|
|
|
@@ -16,12 +23,14 @@ module Yamine
|
|
|
16
23
|
# boots every process, supervises the tree, cleans up on exit.
|
|
17
24
|
def run_inferred(ctx, args)
|
|
18
25
|
variant = ENV["YAMINE_VARIANT"]
|
|
19
|
-
opts = ctx.parse_flags(args, %i[variant tld force app_port wait no_wait json branch])
|
|
26
|
+
opts = ctx.parse_flags(args, %i[variant tld force app_port wait no_wait json branch detach])
|
|
20
27
|
resolved = resolve!(ctx, variant: opts[:variant] || variant, tld: opts[:tld],
|
|
21
28
|
use_branch: opts[:branch])
|
|
22
29
|
# Ownership gate before any side effects: no proxy spawn, no
|
|
23
30
|
# port allocation when we'd refuse anyway.
|
|
24
31
|
check_worktree_ownership!(ctx, resolved, force: opts[:force])
|
|
32
|
+
return detach_boot(ctx, resolved, opts) if opts[:detach]
|
|
33
|
+
|
|
25
34
|
ensure_proxy!(ctx, json: opts[:json])
|
|
26
35
|
boot_all(ctx, resolved, opts)
|
|
27
36
|
end
|
|
@@ -163,8 +172,19 @@ module Yamine
|
|
|
163
172
|
routes_registered.each { |r| named_pids[r[:app].pid] = r[:app].name }
|
|
164
173
|
children.each { |c| named_pids[c[:pid]] = c[:name] }
|
|
165
174
|
all_hostnames = routes_registered.flat_map { |r| r[:hostnames] }
|
|
175
|
+
# Nothing to supervise is not a boot. collect_spawns refuses a
|
|
176
|
+
# compound cmd line with an error and leaves the plan empty, and
|
|
177
|
+
# the boot then printed `ready: ` and sat in supervise_tree for
|
|
178
|
+
# ever with an empty pid map — a hang indistinguishable from a
|
|
179
|
+
# healthy start, and one `--detach` would have made its caller
|
|
180
|
+
# wait out. The errors above it name the process it refused.
|
|
181
|
+
if named_pids.empty?
|
|
182
|
+
$stderr.puts "Error: no process was started — see the errors above."
|
|
183
|
+
exit 1
|
|
184
|
+
end
|
|
166
185
|
trap_cleanup(ctx, all_hostnames, named_pids.keys)
|
|
167
|
-
supervise_tree(ctx, all_hostnames, named_pids, reporter: events
|
|
186
|
+
supervise_tree(ctx, all_hostnames, named_pids, reporter: events,
|
|
187
|
+
json: opts[:json])
|
|
168
188
|
end
|
|
169
189
|
|
|
170
190
|
# TERM the old server and make sure the pidfile no longer names
|
|
@@ -332,13 +352,10 @@ module Yamine
|
|
|
332
352
|
end
|
|
333
353
|
|
|
334
354
|
def tail_for(failure, apps)
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
path =
|
|
338
|
-
|
|
339
|
-
{ path: path, tail: lines }
|
|
340
|
-
rescue SystemCallError
|
|
341
|
-
{ path: path, tail: "(unreadable log)" }
|
|
355
|
+
# Every process appends to the same file, so the app slot only
|
|
356
|
+
# decides whether there is a process to blame for it.
|
|
357
|
+
path = apps[failure[:name]] ? app_log_path : nil
|
|
358
|
+
{ path: path, tail: path ? log_tail(path) : "(no log file)" }
|
|
342
359
|
end
|
|
343
360
|
|
|
344
361
|
def stop_spawned(app)
|
|
@@ -845,30 +862,278 @@ module Yamine
|
|
|
845
862
|
File.file?(File.join(Dir.pwd, "config", "application.rb"))
|
|
846
863
|
end
|
|
847
864
|
|
|
865
|
+
# `yamine start --detach`: boot into the background and hand control
|
|
866
|
+
# back, so an agent gets its prompt (and its exit code) without
|
|
867
|
+
# reaching for nohup.
|
|
868
|
+
#
|
|
869
|
+
# The route's recorded owner pid is the thing this has to get
|
|
870
|
+
# right, and it is why the boot happens in the child and never in
|
|
871
|
+
# the parent: `add_route` records `Process.pid`, and
|
|
872
|
+
# RouteStore#load_routes prunes every route whose pid is dead. A
|
|
873
|
+
# parent that registered the routes and exited would have its own
|
|
874
|
+
# route pruned on the next read, and the proxy would 503 an app
|
|
875
|
+
# that is running perfectly well. So the child boots — its pid is
|
|
876
|
+
# what lands in routes.json — keeps the tree supervised, and
|
|
877
|
+
# outlives this process. The parent only waits and reports.
|
|
878
|
+
def detach_boot(ctx, resolved, opts)
|
|
879
|
+
hostname = Resolver.hostname_for(resolved, Resolver.primary_proc(resolved))
|
|
880
|
+
raise Error, "no HTTP process (proxy: true) in config/local.yml to detach" unless hostname
|
|
881
|
+
|
|
882
|
+
url = Hostname.url(hostname, port: ctx.proxy_port, tls: ctx.proxy_tls)
|
|
883
|
+
pidfile, log = detach_paths(ctx.store, hostname)
|
|
884
|
+
running = detached_pid(pidfile) || foreground_owner(ctx, hostname)
|
|
885
|
+
return report_detached(hostname, url, log, running, opts, started: false) if running
|
|
886
|
+
|
|
887
|
+
# Ensured here, in the process still attached to the caller: a
|
|
888
|
+
# sudo prompt, a port clash or a missing setup has to be reported
|
|
889
|
+
# by something whose exit code and output the caller can see.
|
|
890
|
+
ensure_proxy!(ctx, json: opts[:json])
|
|
891
|
+
pid = fork { detached_child(ctx, resolved, opts, hostname, pidfile, log) }
|
|
892
|
+
await_detached(ctx, hostname, url, log, pid, opts)
|
|
893
|
+
end
|
|
894
|
+
|
|
895
|
+
# Detach bookkeeping under the state dir, beside the routes the
|
|
896
|
+
# tree owns. The pidfile is the only record of WHICH process
|
|
897
|
+
# supervises a detached tree (routes.json holds the app's route and
|
|
898
|
+
# `yamine stop` reaches the backend through the sidecar), which is
|
|
899
|
+
# also what makes a second `--detach` a no-op instead of a second
|
|
900
|
+
# app fighting over the same hostname.
|
|
901
|
+
def detach_paths(store, hostname)
|
|
902
|
+
[File.join(store.dir, "start-#{hostname}.pid"),
|
|
903
|
+
File.join(store.dir, "start-#{hostname}.log")]
|
|
904
|
+
end
|
|
905
|
+
|
|
906
|
+
# The pid of a detached tree, or nil. A pidfile whose process is
|
|
907
|
+
# gone is not a tree — it is a crash or a stop that never got to
|
|
908
|
+
# clean up — and reading it as one would make `--detach` refuse to
|
|
909
|
+
# start anything on that hostname again.
|
|
910
|
+
def detached_pid(pidfile)
|
|
911
|
+
return nil unless File.file?(pidfile)
|
|
912
|
+
|
|
913
|
+
pid = File.read(pidfile).strip.to_i
|
|
914
|
+
pid.positive? && ProxyControl.pid_alive?(pid) ? pid : nil
|
|
915
|
+
rescue SystemCallError, ArgumentError
|
|
916
|
+
nil
|
|
917
|
+
end
|
|
918
|
+
|
|
919
|
+
# A foreground `yamine start` in this directory has no pidfile, so
|
|
920
|
+
# the route it registered is the other record of a tree already
|
|
921
|
+
# running here. Without this, `--detach` beside a live foreground
|
|
922
|
+
# boot would fork, lose the race for the route, and report a boot
|
|
923
|
+
# failure for an app that is up and serving.
|
|
924
|
+
def foreground_owner(ctx, hostname)
|
|
925
|
+
entry = ctx.store.find(hostname)
|
|
926
|
+
return nil unless entry && entry["pid"] != 0
|
|
927
|
+
return nil unless entry["agent"] == Agent.name
|
|
928
|
+
return nil unless entry.dig("spec", "dir") == File.expand_path(Dir.pwd)
|
|
929
|
+
|
|
930
|
+
ProxyControl.pid_alive?(entry["pid"]) ? entry["pid"] : nil
|
|
931
|
+
end
|
|
932
|
+
|
|
933
|
+
# The detached half: its own session, its own stdio, and the whole
|
|
934
|
+
# boot. `setsid` so the tree outlives the shell that started it and
|
|
935
|
+
# takes no SIGHUP from a terminal about to close; the log file so
|
|
936
|
+
# the boot's narration, and the message that ends the run, have
|
|
937
|
+
# somewhere to land once this process is gone.
|
|
938
|
+
def detached_child(ctx, resolved, opts, hostname, pidfile, log)
|
|
939
|
+
Process.setsid
|
|
940
|
+
redirect_detached_io(log)
|
|
941
|
+
File.write(pidfile, "#{Process.pid}\n")
|
|
942
|
+
# `--no-wait` has nothing left to say here: the promise of
|
|
943
|
+
# detaching is that the command returns once the app answers, so
|
|
944
|
+
# the child always takes the health-gated path.
|
|
945
|
+
boot_all(ctx, resolved, opts.merge(no_wait: nil, detach: nil))
|
|
946
|
+
rescue SystemExit => e
|
|
947
|
+
# The boot exits on purpose — a failed health wait, a child that
|
|
948
|
+
# died, a signal from `yamine stop` — and that status is the only
|
|
949
|
+
# report there will ever be. `exit!` rather than `exit` because a
|
|
950
|
+
# forked block turns a raise into a generic failure, and rather
|
|
951
|
+
# than a normal exit because the pidfile is ours to remove: a
|
|
952
|
+
# dead tree must not leave a pidfile claiming one is running.
|
|
953
|
+
detach_forget(pidfile)
|
|
954
|
+
exit!(e.status)
|
|
955
|
+
rescue StandardError => e
|
|
956
|
+
$stderr.puts "[yamine] detached boot failed: #{e.message.lines.first&.strip}"
|
|
957
|
+
detach_forget(pidfile)
|
|
958
|
+
exit!(1)
|
|
959
|
+
end
|
|
960
|
+
|
|
961
|
+
# stdin from /dev/null so a read never blocks on a terminal that
|
|
962
|
+
# has gone away; both streams into the log; sync, because a
|
|
963
|
+
# buffered stream in a process that may live for hours is a log
|
|
964
|
+
# that shows nothing until it exits.
|
|
965
|
+
def redirect_detached_io(log)
|
|
966
|
+
io = Log.open_append(log)
|
|
967
|
+
$stdin.reopen(File::NULL)
|
|
968
|
+
$stdout.reopen(io)
|
|
969
|
+
$stderr.reopen(io)
|
|
970
|
+
$stdout.sync = $stderr.sync = true
|
|
971
|
+
ensure
|
|
972
|
+
io&.close
|
|
973
|
+
end
|
|
974
|
+
|
|
975
|
+
# Remove the pidfile, but only while it is still ours: a second
|
|
976
|
+
# `--detach` that already replaced the file must not have its
|
|
977
|
+
# record deleted by the first tree's cleanup.
|
|
978
|
+
def detach_forget(pidfile)
|
|
979
|
+
return unless File.read(pidfile).strip.to_i == Process.pid
|
|
980
|
+
|
|
981
|
+
FileUtils.rm_f(pidfile)
|
|
982
|
+
rescue SystemCallError
|
|
983
|
+
nil
|
|
984
|
+
end
|
|
985
|
+
|
|
986
|
+
# Wait for the app to answer, then say what happened. Two things
|
|
987
|
+
# end the wait: the route appears — which on the health-gated path
|
|
988
|
+
# means every process passed its check, since the child registers
|
|
989
|
+
# nothing until they all do — or the child exits, which is a boot
|
|
990
|
+
# that failed and has already written why into the log.
|
|
991
|
+
#
|
|
992
|
+
# The route's recorded pid is what proves it is THIS child's: a
|
|
993
|
+
# route left over from an earlier run would otherwise pass for a
|
|
994
|
+
# successful boot of one that never happened.
|
|
995
|
+
def await_detached(ctx, hostname, url, log, pid, opts)
|
|
996
|
+
deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + DETACH_TIMEOUT
|
|
997
|
+
status = nil
|
|
998
|
+
loop do
|
|
999
|
+
entry = ctx.store.find(hostname)
|
|
1000
|
+
return report_detached(hostname, url, log, pid, opts, started: true) if
|
|
1001
|
+
entry && entry["pid"] == pid
|
|
1002
|
+
# WNOHANG, and it has to be a wait: the child is this
|
|
1003
|
+
# process's own child, so a dead one sits in the process table
|
|
1004
|
+
# as a zombie until it is reaped and `kill(0, pid)` would
|
|
1005
|
+
# report it alive for as long as we sat here waiting.
|
|
1006
|
+
_waited, status = Process.waitpid2(pid, Process::WNOHANG)
|
|
1007
|
+
break if status
|
|
1008
|
+
if Process.clock_gettime(Process::CLOCK_MONOTONIC) > deadline
|
|
1009
|
+
status = :timeout
|
|
1010
|
+
break
|
|
1011
|
+
end
|
|
1012
|
+
|
|
1013
|
+
sleep 0.25
|
|
1014
|
+
end
|
|
1015
|
+
detach_failed(hostname, url, log, status, opts)
|
|
1016
|
+
end
|
|
1017
|
+
|
|
1018
|
+
# The report on success, and the report on "was already up": the
|
|
1019
|
+
# URL, the pid that owns the tree, and where its output goes.
|
|
1020
|
+
# `--json` gets the same three as a payload, because an agent's
|
|
1021
|
+
# next move is `yamine list` and "started" with no pid to stop
|
|
1022
|
+
# again is not a usable answer.
|
|
1023
|
+
def report_detached(hostname, url, log, pid, opts, started:)
|
|
1024
|
+
payload = { ok: true, url: url, hostnames: [hostname], pid: pid,
|
|
1025
|
+
log_path: log, started: started }
|
|
1026
|
+
if opts[:json]
|
|
1027
|
+
puts JSON.generate(payload)
|
|
1028
|
+
return
|
|
1029
|
+
end
|
|
1030
|
+
if started
|
|
1031
|
+
puts "Detached: #{url}"
|
|
1032
|
+
else
|
|
1033
|
+
puts "Already running: #{url} (pid #{pid})"
|
|
1034
|
+
puts " Nothing was started; `yamine stop` stops this one."
|
|
1035
|
+
return
|
|
1036
|
+
end
|
|
1037
|
+
puts " pid #{pid}"
|
|
1038
|
+
puts " log #{log}"
|
|
1039
|
+
end
|
|
1040
|
+
|
|
1041
|
+
# Never zero, and never "started". The reason it is not serving is
|
|
1042
|
+
# in the log and nowhere else, so that is where the report points.
|
|
1043
|
+
def detach_failed(hostname, url, log, status, opts)
|
|
1044
|
+
reason =
|
|
1045
|
+
case status
|
|
1046
|
+
when :timeout then "did not become healthy within #{DETACH_TIMEOUT}s"
|
|
1047
|
+
when nil then "was killed before serving"
|
|
1048
|
+
else "exited (#{describe_status(status)})"
|
|
1049
|
+
end
|
|
1050
|
+
payload = { ok: false, url: url, hostnames: [hostname], reason: reason,
|
|
1051
|
+
log_path: log, log_tail: log_tail(log) }
|
|
1052
|
+
if opts[:json]
|
|
1053
|
+
$stderr.puts "Error: yamine start --detach: #{hostname} #{reason}."
|
|
1054
|
+
puts JSON.generate(payload)
|
|
1055
|
+
else
|
|
1056
|
+
$stderr.puts "Error: yamine start --detach: #{hostname} #{reason}."
|
|
1057
|
+
$stderr.puts log_tail(log, lines: 10)
|
|
1058
|
+
$stderr.puts " log: #{log}"
|
|
1059
|
+
end
|
|
1060
|
+
exit 1
|
|
1061
|
+
end
|
|
1062
|
+
|
|
1063
|
+
# The one log every process in a tree appends to (Runner#log_path),
|
|
1064
|
+
# so it is also the one log worth reading when a child dies.
|
|
1065
|
+
def app_log_path
|
|
1066
|
+
File.expand_path(File.join(Dir.pwd, "log", "development.log"))
|
|
1067
|
+
end
|
|
1068
|
+
|
|
1069
|
+
# Last lines of the app log, for failure payloads and for the
|
|
1070
|
+
# message that ends a run. Never raises: a missing or unreadable
|
|
1071
|
+
# log is part of the failure being reported, not a second failure.
|
|
1072
|
+
def log_tail(path = app_log_path, lines: 20)
|
|
1073
|
+
return "(no log file)" unless File.file?(path)
|
|
1074
|
+
|
|
1075
|
+
File.readlines(path).last(lines).join
|
|
1076
|
+
rescue SystemCallError
|
|
1077
|
+
"(unreadable log)"
|
|
1078
|
+
end
|
|
1079
|
+
|
|
1080
|
+
# What a pid we spawned exited with, in words an agent can read:
|
|
1081
|
+
# a code, or the signal that took it down. A pid with no status to
|
|
1082
|
+
# report (never ours, not yet reaped) says so rather than guessing
|
|
1083
|
+
# a code.
|
|
1084
|
+
def describe_status(status)
|
|
1085
|
+
return "status unknown" unless status
|
|
1086
|
+
return "signal #{status.termsig}" if status.signaled?
|
|
1087
|
+
|
|
1088
|
+
"exit #{status.exitstatus}"
|
|
1089
|
+
end
|
|
1090
|
+
|
|
848
1091
|
# Supervise the booted tree: the first child to exit ends the run,
|
|
849
1092
|
# because a half-stack is worse than no stack — a dead jobs worker
|
|
850
1093
|
# with a live web process looks healthy until someone wonders why
|
|
851
|
-
# nothing is being processed. Name the casualty and its
|
|
852
|
-
# process exited" left the user to guess which one, and the
|
|
853
|
-
# worth reading is
|
|
854
|
-
|
|
1094
|
+
# nothing is being processed. Name the casualty, its status and its
|
|
1095
|
+
# log: "a process exited" left the user to guess which one, and the
|
|
1096
|
+
# log worth reading is the one every process appends to.
|
|
1097
|
+
#
|
|
1098
|
+
# The status is 1, never 0. `yamine start` is how an agent decides
|
|
1099
|
+
# whether the app came up, and a zero here reads as "healthy": the
|
|
1100
|
+
# routes are already removed, the rest of the tree is being killed,
|
|
1101
|
+
# and the app is not serving. An agent that trusted it would go on
|
|
1102
|
+
# to curl a URL that answers 503 and blame the app. It is the code
|
|
1103
|
+
# the --wait path already uses for a boot that never became
|
|
1104
|
+
# healthy, so "the app is not up" stays one meaning — and it is a
|
|
1105
|
+
# different question from the one `yamine stop` answers with its
|
|
1106
|
+
# 0/2/3/4, which is about what a stop did.
|
|
1107
|
+
def supervise_tree(ctx, hostnames, named_pids, reporter: nil, json: false)
|
|
855
1108
|
loop do
|
|
856
1109
|
sleep 0.5
|
|
857
1110
|
dead = named_pids.find { |pid, _| !ProxyControl.pid_alive?(pid) }
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
1111
|
+
next unless dead
|
|
1112
|
+
|
|
1113
|
+
pid, name = dead
|
|
1114
|
+
status = ProcessTree.status(pid)
|
|
1115
|
+
log = app_log_path
|
|
1116
|
+
$stderr.puts "\n[#{name}] exited (pid #{pid}, #{describe_status(status)}) " \
|
|
1117
|
+
"— stopping the whole tree."
|
|
1118
|
+
reporter&.note("#{name} exited; cleaning up routes")
|
|
1119
|
+
if json
|
|
1120
|
+
# stdout stays the machine stream: the event lines above it,
|
|
1121
|
+
# this payload last, exactly as the --wait failure reads.
|
|
1122
|
+
puts JSON.generate({ ok: false, error: "child-exited", name: name,
|
|
1123
|
+
pid: pid, status: status&.exitstatus, signal: status&.termsig,
|
|
1124
|
+
log_path: File.file?(log) ? log : nil,
|
|
1125
|
+
log_tail: log_tail(log) }.compact)
|
|
1126
|
+
elsif File.file?(log)
|
|
1127
|
+
$stderr.puts log_tail(log, lines: 10)
|
|
1128
|
+
$stderr.puts " log: #{log}"
|
|
1129
|
+
end
|
|
1130
|
+
cleanup_routes(ctx, hostnames)
|
|
1131
|
+
# Kill remaining children — by group, so a `sh -c` backend's
|
|
1132
|
+
# process dies with the shell we hold a pid for.
|
|
1133
|
+
named_pids.each_key do |other|
|
|
1134
|
+
ProcessTree.term(other)
|
|
871
1135
|
end
|
|
1136
|
+
exit 1
|
|
872
1137
|
end
|
|
873
1138
|
end
|
|
874
1139
|
|
data/lib/yamine/cli/context.rb
CHANGED
|
@@ -104,7 +104,7 @@ module Yamine
|
|
|
104
104
|
elsif arg.start_with?("--")
|
|
105
105
|
key = arg.sub(/\A--/, "").tr("-", "_").to_sym
|
|
106
106
|
if known.include?(key)
|
|
107
|
-
if %i[branch force wait no_wait json].include?(key)
|
|
107
|
+
if %i[branch force wait no_wait json detach].include?(key)
|
|
108
108
|
opts[key] = true
|
|
109
109
|
i += 1
|
|
110
110
|
else
|