yamine 0.21.2 → 0.22.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +170 -0
- data/README.md +39 -12
- data/lib/ask/skills/yamine/SKILL.md +20 -1
- data/lib/yamine/cli/boot.rb +291 -26
- data/lib/yamine/cli/context.rb +1 -1
- data/lib/yamine/cli/routes.rb +96 -19
- data/lib/yamine/cli/system.rb +20 -10
- data/lib/yamine/cli.rb +3 -2
- data/lib/yamine/doctor.rb +9 -0
- data/lib/yamine/process_tree.rb +55 -0
- data/lib/yamine/proxy.rb +238 -53
- data/lib/yamine/runner.rb +10 -4
- data/lib/yamine/supervisor.rb +147 -15
- data/lib/yamine/trust.rb +183 -8
- data/lib/yamine/version.rb +1 -1
- metadata +1 -1
data/lib/yamine/cli/boot.rb
CHANGED
|
@@ -8,6 +8,13 @@ module Yamine
|
|
|
8
8
|
# No inference, no Procfile at boot, no single-process default.
|
|
9
9
|
module BootCommand
|
|
10
10
|
PORT_IGNORING = %w[jekyll middleman bridgetown].freeze
|
|
11
|
+
# How long the detaching process waits for the app to answer before
|
|
12
|
+
# it reports that it did not. Generous on purpose: the boot running
|
|
13
|
+
# inside the child has its own budget per phase (Readiness's, 45s a
|
|
14
|
+
# process, plus deps/db/schema), and this is the parent giving up
|
|
15
|
+
# on a child still doing legitimate work. A boot that fails ends
|
|
16
|
+
# the wait on its own, long before this.
|
|
17
|
+
DETACH_TIMEOUT = 300
|
|
11
18
|
|
|
12
19
|
module_function
|
|
13
20
|
|
|
@@ -16,12 +23,14 @@ module Yamine
|
|
|
16
23
|
# boots every process, supervises the tree, cleans up on exit.
|
|
17
24
|
def run_inferred(ctx, args)
|
|
18
25
|
variant = ENV["YAMINE_VARIANT"]
|
|
19
|
-
opts = ctx.parse_flags(args, %i[variant tld force app_port wait no_wait json branch])
|
|
26
|
+
opts = ctx.parse_flags(args, %i[variant tld force app_port wait no_wait json branch detach])
|
|
20
27
|
resolved = resolve!(ctx, variant: opts[:variant] || variant, tld: opts[:tld],
|
|
21
28
|
use_branch: opts[:branch])
|
|
22
29
|
# Ownership gate before any side effects: no proxy spawn, no
|
|
23
30
|
# port allocation when we'd refuse anyway.
|
|
24
31
|
check_worktree_ownership!(ctx, resolved, force: opts[:force])
|
|
32
|
+
return detach_boot(ctx, resolved, opts) if opts[:detach]
|
|
33
|
+
|
|
25
34
|
ensure_proxy!(ctx, json: opts[:json])
|
|
26
35
|
boot_all(ctx, resolved, opts)
|
|
27
36
|
end
|
|
@@ -163,8 +172,19 @@ module Yamine
|
|
|
163
172
|
routes_registered.each { |r| named_pids[r[:app].pid] = r[:app].name }
|
|
164
173
|
children.each { |c| named_pids[c[:pid]] = c[:name] }
|
|
165
174
|
all_hostnames = routes_registered.flat_map { |r| r[:hostnames] }
|
|
175
|
+
# Nothing to supervise is not a boot. collect_spawns refuses a
|
|
176
|
+
# compound cmd line with an error and leaves the plan empty, and
|
|
177
|
+
# the boot then printed `ready: ` and sat in supervise_tree for
|
|
178
|
+
# ever with an empty pid map — a hang indistinguishable from a
|
|
179
|
+
# healthy start, and one `--detach` would have made its caller
|
|
180
|
+
# wait out. The errors above it name the process it refused.
|
|
181
|
+
if named_pids.empty?
|
|
182
|
+
$stderr.puts "Error: no process was started — see the errors above."
|
|
183
|
+
exit 1
|
|
184
|
+
end
|
|
166
185
|
trap_cleanup(ctx, all_hostnames, named_pids.keys)
|
|
167
|
-
supervise_tree(ctx, all_hostnames, named_pids, reporter: events
|
|
186
|
+
supervise_tree(ctx, all_hostnames, named_pids, reporter: events,
|
|
187
|
+
json: opts[:json])
|
|
168
188
|
end
|
|
169
189
|
|
|
170
190
|
# TERM the old server and make sure the pidfile no longer names
|
|
@@ -332,13 +352,10 @@ module Yamine
|
|
|
332
352
|
end
|
|
333
353
|
|
|
334
354
|
def tail_for(failure, apps)
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
path =
|
|
338
|
-
|
|
339
|
-
{ path: path, tail: lines }
|
|
340
|
-
rescue SystemCallError
|
|
341
|
-
{ path: path, tail: "(unreadable log)" }
|
|
355
|
+
# Every process appends to the same file, so the app slot only
|
|
356
|
+
# decides whether there is a process to blame for it.
|
|
357
|
+
path = apps[failure[:name]] ? app_log_path : nil
|
|
358
|
+
{ path: path, tail: path ? log_tail(path) : "(no log file)" }
|
|
342
359
|
end
|
|
343
360
|
|
|
344
361
|
def stop_spawned(app)
|
|
@@ -845,30 +862,278 @@ module Yamine
|
|
|
845
862
|
File.file?(File.join(Dir.pwd, "config", "application.rb"))
|
|
846
863
|
end
|
|
847
864
|
|
|
865
|
+
# `yamine start --detach`: boot into the background and hand control
|
|
866
|
+
# back, so an agent gets its prompt (and its exit code) without
|
|
867
|
+
# reaching for nohup.
|
|
868
|
+
#
|
|
869
|
+
# The route's recorded owner pid is the thing this has to get
|
|
870
|
+
# right, and it is why the boot happens in the child and never in
|
|
871
|
+
# the parent: `add_route` records `Process.pid`, and
|
|
872
|
+
# RouteStore#load_routes prunes every route whose pid is dead. A
|
|
873
|
+
# parent that registered the routes and exited would have its own
|
|
874
|
+
# route pruned on the next read, and the proxy would 503 an app
|
|
875
|
+
# that is running perfectly well. So the child boots — its pid is
|
|
876
|
+
# what lands in routes.json — keeps the tree supervised, and
|
|
877
|
+
# outlives this process. The parent only waits and reports.
|
|
878
|
+
def detach_boot(ctx, resolved, opts)
|
|
879
|
+
hostname = Resolver.hostname_for(resolved, Resolver.primary_proc(resolved))
|
|
880
|
+
raise Error, "no HTTP process (proxy: true) in config/local.yml to detach" unless hostname
|
|
881
|
+
|
|
882
|
+
url = Hostname.url(hostname, port: ctx.proxy_port, tls: ctx.proxy_tls)
|
|
883
|
+
pidfile, log = detach_paths(ctx.store, hostname)
|
|
884
|
+
running = detached_pid(pidfile) || foreground_owner(ctx, hostname)
|
|
885
|
+
return report_detached(hostname, url, log, running, opts, started: false) if running
|
|
886
|
+
|
|
887
|
+
# Ensured here, in the process still attached to the caller: a
|
|
888
|
+
# sudo prompt, a port clash or a missing setup has to be reported
|
|
889
|
+
# by something whose exit code and output the caller can see.
|
|
890
|
+
ensure_proxy!(ctx, json: opts[:json])
|
|
891
|
+
pid = fork { detached_child(ctx, resolved, opts, hostname, pidfile, log) }
|
|
892
|
+
await_detached(ctx, hostname, url, log, pid, opts)
|
|
893
|
+
end
|
|
894
|
+
|
|
895
|
+
# Detach bookkeeping under the state dir, beside the routes the
|
|
896
|
+
# tree owns. The pidfile is the only record of WHICH process
|
|
897
|
+
# supervises a detached tree (routes.json holds the app's route and
|
|
898
|
+
# `yamine stop` reaches the backend through the sidecar), which is
|
|
899
|
+
# also what makes a second `--detach` a no-op instead of a second
|
|
900
|
+
# app fighting over the same hostname.
|
|
901
|
+
def detach_paths(store, hostname)
|
|
902
|
+
[File.join(store.dir, "start-#{hostname}.pid"),
|
|
903
|
+
File.join(store.dir, "start-#{hostname}.log")]
|
|
904
|
+
end
|
|
905
|
+
|
|
906
|
+
# The pid of a detached tree, or nil. A pidfile whose process is
|
|
907
|
+
# gone is not a tree — it is a crash or a stop that never got to
|
|
908
|
+
# clean up — and reading it as one would make `--detach` refuse to
|
|
909
|
+
# start anything on that hostname again.
|
|
910
|
+
def detached_pid(pidfile)
|
|
911
|
+
return nil unless File.file?(pidfile)
|
|
912
|
+
|
|
913
|
+
pid = File.read(pidfile).strip.to_i
|
|
914
|
+
pid.positive? && ProxyControl.pid_alive?(pid) ? pid : nil
|
|
915
|
+
rescue SystemCallError, ArgumentError
|
|
916
|
+
nil
|
|
917
|
+
end
|
|
918
|
+
|
|
919
|
+
# A foreground `yamine start` in this directory has no pidfile, so
|
|
920
|
+
# the route it registered is the other record of a tree already
|
|
921
|
+
# running here. Without this, `--detach` beside a live foreground
|
|
922
|
+
# boot would fork, lose the race for the route, and report a boot
|
|
923
|
+
# failure for an app that is up and serving.
|
|
924
|
+
def foreground_owner(ctx, hostname)
|
|
925
|
+
entry = ctx.store.find(hostname)
|
|
926
|
+
return nil unless entry && entry["pid"] != 0
|
|
927
|
+
return nil unless entry["agent"] == Agent.name
|
|
928
|
+
return nil unless entry.dig("spec", "dir") == File.expand_path(Dir.pwd)
|
|
929
|
+
|
|
930
|
+
ProxyControl.pid_alive?(entry["pid"]) ? entry["pid"] : nil
|
|
931
|
+
end
|
|
932
|
+
|
|
933
|
+
# The detached half: its own session, its own stdio, and the whole
|
|
934
|
+
# boot. `setsid` so the tree outlives the shell that started it and
|
|
935
|
+
# takes no SIGHUP from a terminal about to close; the log file so
|
|
936
|
+
# the boot's narration, and the message that ends the run, have
|
|
937
|
+
# somewhere to land once this process is gone.
|
|
938
|
+
def detached_child(ctx, resolved, opts, hostname, pidfile, log)
|
|
939
|
+
Process.setsid
|
|
940
|
+
redirect_detached_io(log)
|
|
941
|
+
File.write(pidfile, "#{Process.pid}\n")
|
|
942
|
+
# `--no-wait` has nothing left to say here: the promise of
|
|
943
|
+
# detaching is that the command returns once the app answers, so
|
|
944
|
+
# the child always takes the health-gated path.
|
|
945
|
+
boot_all(ctx, resolved, opts.merge(no_wait: nil, detach: nil))
|
|
946
|
+
rescue SystemExit => e
|
|
947
|
+
# The boot exits on purpose — a failed health wait, a child that
|
|
948
|
+
# died, a signal from `yamine stop` — and that status is the only
|
|
949
|
+
# report there will ever be. `exit!` rather than `exit` because a
|
|
950
|
+
# forked block turns a raise into a generic failure, and rather
|
|
951
|
+
# than a normal exit because the pidfile is ours to remove: a
|
|
952
|
+
# dead tree must not leave a pidfile claiming one is running.
|
|
953
|
+
detach_forget(pidfile)
|
|
954
|
+
exit!(e.status)
|
|
955
|
+
rescue StandardError => e
|
|
956
|
+
$stderr.puts "[yamine] detached boot failed: #{e.message.lines.first&.strip}"
|
|
957
|
+
detach_forget(pidfile)
|
|
958
|
+
exit!(1)
|
|
959
|
+
end
|
|
960
|
+
|
|
961
|
+
# stdin from /dev/null so a read never blocks on a terminal that
|
|
962
|
+
# has gone away; both streams into the log; sync, because a
|
|
963
|
+
# buffered stream in a process that may live for hours is a log
|
|
964
|
+
# that shows nothing until it exits.
|
|
965
|
+
def redirect_detached_io(log)
|
|
966
|
+
io = Log.open_append(log)
|
|
967
|
+
$stdin.reopen(File::NULL)
|
|
968
|
+
$stdout.reopen(io)
|
|
969
|
+
$stderr.reopen(io)
|
|
970
|
+
$stdout.sync = $stderr.sync = true
|
|
971
|
+
ensure
|
|
972
|
+
io&.close
|
|
973
|
+
end
|
|
974
|
+
|
|
975
|
+
# Remove the pidfile, but only while it is still ours: a second
|
|
976
|
+
# `--detach` that already replaced the file must not have its
|
|
977
|
+
# record deleted by the first tree's cleanup.
|
|
978
|
+
def detach_forget(pidfile)
|
|
979
|
+
return unless File.read(pidfile).strip.to_i == Process.pid
|
|
980
|
+
|
|
981
|
+
FileUtils.rm_f(pidfile)
|
|
982
|
+
rescue SystemCallError
|
|
983
|
+
nil
|
|
984
|
+
end
|
|
985
|
+
|
|
986
|
+
# Wait for the app to answer, then say what happened. Two things
|
|
987
|
+
# end the wait: the route appears — which on the health-gated path
|
|
988
|
+
# means every process passed its check, since the child registers
|
|
989
|
+
# nothing until they all do — or the child exits, which is a boot
|
|
990
|
+
# that failed and has already written why into the log.
|
|
991
|
+
#
|
|
992
|
+
# The route's recorded pid is what proves it is THIS child's: a
|
|
993
|
+
# route left over from an earlier run would otherwise pass for a
|
|
994
|
+
# successful boot of one that never happened.
|
|
995
|
+
def await_detached(ctx, hostname, url, log, pid, opts)
|
|
996
|
+
deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + DETACH_TIMEOUT
|
|
997
|
+
status = nil
|
|
998
|
+
loop do
|
|
999
|
+
entry = ctx.store.find(hostname)
|
|
1000
|
+
return report_detached(hostname, url, log, pid, opts, started: true) if
|
|
1001
|
+
entry && entry["pid"] == pid
|
|
1002
|
+
# WNOHANG, and it has to be a wait: the child is this
|
|
1003
|
+
# process's own child, so a dead one sits in the process table
|
|
1004
|
+
# as a zombie until it is reaped and `kill(0, pid)` would
|
|
1005
|
+
# report it alive for as long as we sat here waiting.
|
|
1006
|
+
_waited, status = Process.waitpid2(pid, Process::WNOHANG)
|
|
1007
|
+
break if status
|
|
1008
|
+
if Process.clock_gettime(Process::CLOCK_MONOTONIC) > deadline
|
|
1009
|
+
status = :timeout
|
|
1010
|
+
break
|
|
1011
|
+
end
|
|
1012
|
+
|
|
1013
|
+
sleep 0.25
|
|
1014
|
+
end
|
|
1015
|
+
detach_failed(hostname, url, log, status, opts)
|
|
1016
|
+
end
|
|
1017
|
+
|
|
1018
|
+
# The report on success, and the report on "was already up": the
|
|
1019
|
+
# URL, the pid that owns the tree, and where its output goes.
|
|
1020
|
+
# `--json` gets the same three as a payload, because an agent's
|
|
1021
|
+
# next move is `yamine list` and "started" with no pid to stop
|
|
1022
|
+
# again is not a usable answer.
|
|
1023
|
+
def report_detached(hostname, url, log, pid, opts, started:)
|
|
1024
|
+
payload = { ok: true, url: url, hostnames: [hostname], pid: pid,
|
|
1025
|
+
log_path: log, started: started }
|
|
1026
|
+
if opts[:json]
|
|
1027
|
+
puts JSON.generate(payload)
|
|
1028
|
+
return
|
|
1029
|
+
end
|
|
1030
|
+
if started
|
|
1031
|
+
puts "Detached: #{url}"
|
|
1032
|
+
else
|
|
1033
|
+
puts "Already running: #{url} (pid #{pid})"
|
|
1034
|
+
puts " Nothing was started; `yamine stop` stops this one."
|
|
1035
|
+
return
|
|
1036
|
+
end
|
|
1037
|
+
puts " pid #{pid}"
|
|
1038
|
+
puts " log #{log}"
|
|
1039
|
+
end
|
|
1040
|
+
|
|
1041
|
+
# Never zero, and never "started". The reason it is not serving is
|
|
1042
|
+
# in the log and nowhere else, so that is where the report points.
|
|
1043
|
+
def detach_failed(hostname, url, log, status, opts)
|
|
1044
|
+
reason =
|
|
1045
|
+
case status
|
|
1046
|
+
when :timeout then "did not become healthy within #{DETACH_TIMEOUT}s"
|
|
1047
|
+
when nil then "was killed before serving"
|
|
1048
|
+
else "exited (#{describe_status(status)})"
|
|
1049
|
+
end
|
|
1050
|
+
payload = { ok: false, url: url, hostnames: [hostname], reason: reason,
|
|
1051
|
+
log_path: log, log_tail: log_tail(log) }
|
|
1052
|
+
if opts[:json]
|
|
1053
|
+
$stderr.puts "Error: yamine start --detach: #{hostname} #{reason}."
|
|
1054
|
+
puts JSON.generate(payload)
|
|
1055
|
+
else
|
|
1056
|
+
$stderr.puts "Error: yamine start --detach: #{hostname} #{reason}."
|
|
1057
|
+
$stderr.puts log_tail(log, lines: 10)
|
|
1058
|
+
$stderr.puts " log: #{log}"
|
|
1059
|
+
end
|
|
1060
|
+
exit 1
|
|
1061
|
+
end
|
|
1062
|
+
|
|
1063
|
+
# The one log every process in a tree appends to (Runner#log_path),
|
|
1064
|
+
# so it is also the one log worth reading when a child dies.
|
|
1065
|
+
def app_log_path
|
|
1066
|
+
File.expand_path(File.join(Dir.pwd, "log", "development.log"))
|
|
1067
|
+
end
|
|
1068
|
+
|
|
1069
|
+
# Last lines of the app log, for failure payloads and for the
|
|
1070
|
+
# message that ends a run. Never raises: a missing or unreadable
|
|
1071
|
+
# log is part of the failure being reported, not a second failure.
|
|
1072
|
+
def log_tail(path = app_log_path, lines: 20)
|
|
1073
|
+
return "(no log file)" unless File.file?(path)
|
|
1074
|
+
|
|
1075
|
+
File.readlines(path).last(lines).join
|
|
1076
|
+
rescue SystemCallError
|
|
1077
|
+
"(unreadable log)"
|
|
1078
|
+
end
|
|
1079
|
+
|
|
1080
|
+
# What a pid we spawned exited with, in words an agent can read:
|
|
1081
|
+
# a code, or the signal that took it down. A pid with no status to
|
|
1082
|
+
# report (never ours, not yet reaped) says so rather than guessing
|
|
1083
|
+
# a code.
|
|
1084
|
+
def describe_status(status)
|
|
1085
|
+
return "status unknown" unless status
|
|
1086
|
+
return "signal #{status.termsig}" if status.signaled?
|
|
1087
|
+
|
|
1088
|
+
"exit #{status.exitstatus}"
|
|
1089
|
+
end
|
|
1090
|
+
|
|
848
1091
|
# Supervise the booted tree: the first child to exit ends the run,
|
|
849
1092
|
# because a half-stack is worse than no stack — a dead jobs worker
|
|
850
1093
|
# with a live web process looks healthy until someone wonders why
|
|
851
|
-
# nothing is being processed. Name the casualty and its
|
|
852
|
-
# process exited" left the user to guess which one, and the
|
|
853
|
-
# worth reading is
|
|
854
|
-
|
|
1094
|
+
# nothing is being processed. Name the casualty, its status and its
|
|
1095
|
+
# log: "a process exited" left the user to guess which one, and the
|
|
1096
|
+
# log worth reading is the one every process appends to.
|
|
1097
|
+
#
|
|
1098
|
+
# The status is 1, never 0. `yamine start` is how an agent decides
|
|
1099
|
+
# whether the app came up, and a zero here reads as "healthy": the
|
|
1100
|
+
# routes are already removed, the rest of the tree is being killed,
|
|
1101
|
+
# and the app is not serving. An agent that trusted it would go on
|
|
1102
|
+
# to curl a URL that answers 503 and blame the app. It is the code
|
|
1103
|
+
# the --wait path already uses for a boot that never became
|
|
1104
|
+
# healthy, so "the app is not up" stays one meaning — and it is a
|
|
1105
|
+
# different question from the one `yamine stop` answers with its
|
|
1106
|
+
# 0/2/3/4, which is about what a stop did.
|
|
1107
|
+
def supervise_tree(ctx, hostnames, named_pids, reporter: nil, json: false)
|
|
855
1108
|
loop do
|
|
856
1109
|
sleep 0.5
|
|
857
1110
|
dead = named_pids.find { |pid, _| !ProxyControl.pid_alive?(pid) }
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
|
|
861
|
-
|
|
862
|
-
|
|
863
|
-
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
1111
|
+
next unless dead
|
|
1112
|
+
|
|
1113
|
+
pid, name = dead
|
|
1114
|
+
status = ProcessTree.status(pid)
|
|
1115
|
+
log = app_log_path
|
|
1116
|
+
$stderr.puts "\n[#{name}] exited (pid #{pid}, #{describe_status(status)}) " \
|
|
1117
|
+
"— stopping the whole tree."
|
|
1118
|
+
reporter&.note("#{name} exited; cleaning up routes")
|
|
1119
|
+
if json
|
|
1120
|
+
# stdout stays the machine stream: the event lines above it,
|
|
1121
|
+
# this payload last, exactly as the --wait failure reads.
|
|
1122
|
+
puts JSON.generate({ ok: false, error: "child-exited", name: name,
|
|
1123
|
+
pid: pid, status: status&.exitstatus, signal: status&.termsig,
|
|
1124
|
+
log_path: File.file?(log) ? log : nil,
|
|
1125
|
+
log_tail: log_tail(log) }.compact)
|
|
1126
|
+
elsif File.file?(log)
|
|
1127
|
+
$stderr.puts log_tail(log, lines: 10)
|
|
1128
|
+
$stderr.puts " log: #{log}"
|
|
1129
|
+
end
|
|
1130
|
+
cleanup_routes(ctx, hostnames)
|
|
1131
|
+
# Kill remaining children — by group, so a `sh -c` backend's
|
|
1132
|
+
# process dies with the shell we hold a pid for.
|
|
1133
|
+
named_pids.each_key do |other|
|
|
1134
|
+
ProcessTree.term(other)
|
|
871
1135
|
end
|
|
1136
|
+
exit 1
|
|
872
1137
|
end
|
|
873
1138
|
end
|
|
874
1139
|
|
data/lib/yamine/cli/context.rb
CHANGED
|
@@ -104,7 +104,7 @@ module Yamine
|
|
|
104
104
|
elsif arg.start_with?("--")
|
|
105
105
|
key = arg.sub(/\A--/, "").tr("-", "_").to_sym
|
|
106
106
|
if known.include?(key)
|
|
107
|
-
if %i[branch force wait no_wait json].include?(key)
|
|
107
|
+
if %i[branch force wait no_wait json detach].include?(key)
|
|
108
108
|
opts[key] = true
|
|
109
109
|
i += 1
|
|
110
110
|
else
|
data/lib/yamine/cli/routes.rb
CHANGED
|
@@ -36,14 +36,7 @@ module Yamine
|
|
|
36
36
|
routes = ctx.store.load_routes
|
|
37
37
|
port = ctx.proxy_port
|
|
38
38
|
tls = ctx.proxy_tls
|
|
39
|
-
entries = routes.map
|
|
40
|
-
{ hostname: r["hostname"],
|
|
41
|
-
url: Hostname.url(r["hostname"], port: port, tls: tls),
|
|
42
|
-
target: r["target"], kind: r["kind"],
|
|
43
|
-
pid: r["pid"], agent: r["agent"],
|
|
44
|
-
supervised: !r["spec"].nil?,
|
|
45
|
-
alive: alive_state(ctx, r) }
|
|
46
|
-
end
|
|
39
|
+
entries = routes.map { |r| entry_for(ctx, r, port: port, tls: tls) }
|
|
47
40
|
if json
|
|
48
41
|
require "json"
|
|
49
42
|
puts JSON.generate({ routes: entries, proxy_port: port, tls: tls })
|
|
@@ -61,6 +54,25 @@ module Yamine
|
|
|
61
54
|
puts
|
|
62
55
|
end
|
|
63
56
|
|
|
57
|
+
# One route as `yamine list` reports it.
|
|
58
|
+
#
|
|
59
|
+
# `pid` stays the route's recorded owner (the yamine process that
|
|
60
|
+
# registered it — that is what routes.json holds and what
|
|
61
|
+
# `yamine stop` and the ownership gate compare), and `backend_pid`
|
|
62
|
+
# is added beside it because `alive` is now the app's state: an
|
|
63
|
+
# agent reading `alive: running` next to a `pid` that has since
|
|
64
|
+
# exited could not tell which process the verdict was about. Both
|
|
65
|
+
# are in the payload, so nothing that was readable is lost.
|
|
66
|
+
def entry_for(ctx, route, port:, tls:)
|
|
67
|
+
{ hostname: route["hostname"],
|
|
68
|
+
url: Hostname.url(route["hostname"], port: port, tls: tls),
|
|
69
|
+
target: route["target"], kind: route["kind"],
|
|
70
|
+
pid: route["pid"], backend_pid: ctx.backend_pid_for(route),
|
|
71
|
+
agent: route["agent"],
|
|
72
|
+
supervised: !route["spec"].nil?,
|
|
73
|
+
alive: alive_state(ctx, route) }
|
|
74
|
+
end
|
|
75
|
+
|
|
64
76
|
# Shared discovery: `get --all` lists every live route (any owner)
|
|
65
77
|
# with its URL and agent, so one agent can find another's services
|
|
66
78
|
# without coupling. `--json` emits the same stable keys as list.
|
|
@@ -89,24 +101,64 @@ module Yamine
|
|
|
89
101
|
end
|
|
90
102
|
end
|
|
91
103
|
|
|
104
|
+
# Liveness of the APP, not of the yamine process that registered
|
|
105
|
+
# the route. `route["pid"]` is that process's own pid — Runner
|
|
106
|
+
# passes Process.pid at every add_route call site — and it outlives
|
|
107
|
+
# the app it booted, so asking it whether the app is alive answered
|
|
108
|
+
# a different question. Every crashed backend read as "running",
|
|
109
|
+
# in the one command an agent would reach for to find out. The
|
|
110
|
+
# app's pid is in the sidecar the boot wrote
|
|
111
|
+
# (state_dir/backend-<hostname>.pid), the same file `yamine stop`
|
|
112
|
+
# reads for exactly this reason.
|
|
113
|
+
#
|
|
114
|
+
# Five states, and the split that matters is running vs
|
|
115
|
+
# backend-gone:
|
|
116
|
+
#
|
|
117
|
+
# running the app's process is there
|
|
118
|
+
# backend-gone the CLI is still up, the app it booted is not
|
|
119
|
+
# owner-gone nothing is there and no app pid was recorded
|
|
120
|
+
# unknown no app pid recorded, and the route's own process
|
|
121
|
+
# is — unknowable, deliberately not "down"
|
|
122
|
+
# reachable / a static alias (pid 0) names no process at all,
|
|
123
|
+
# unreachable so it reports the probe and nothing else
|
|
124
|
+
#
|
|
125
|
+
# `unknown` is the honest answer for a route with no sidecar: one
|
|
126
|
+
# written by a yamine old enough not to write sidecars, or by a
|
|
127
|
+
# boot that died between registering the route and writing the
|
|
128
|
+
# file. There is no evidence of a dead app there, and calling it
|
|
129
|
+
# down would have every pre-existing route on the machine read as
|
|
130
|
+
# broken after an upgrade.
|
|
92
131
|
def alive_state(ctx, route)
|
|
93
|
-
if route["pid"] == 0
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
end
|
|
132
|
+
return ctx.backend_alive?(route) ? "reachable" : "unreachable" if route["pid"] == 0
|
|
133
|
+
|
|
134
|
+
backend = ctx.backend_pid_for(route)
|
|
135
|
+
return owner_state(ctx, route) if backend.nil?
|
|
136
|
+
|
|
137
|
+
ProxyControl.pid_alive?(backend) ? "running" : "backend-gone"
|
|
100
138
|
rescue StandardError
|
|
101
139
|
"unknown"
|
|
102
140
|
end
|
|
103
141
|
|
|
142
|
+
# What we know with no app pid to ask: whether the process that
|
|
143
|
+
# registered the route is still there. `owner-gone` keeps the
|
|
144
|
+
# meaning it has always had — the route outlived its owner, which
|
|
145
|
+
# is the case load_routes leaves behind for `yamine prune`.
|
|
146
|
+
def owner_state(ctx, route)
|
|
147
|
+
ProxyControl.pid_alive?(route["pid"]) ? "unknown" : "owner-gone"
|
|
148
|
+
end
|
|
149
|
+
|
|
150
|
+
# The human line. The pid shown next to the state is the one the
|
|
151
|
+
# state is about: the app's, when there is one. Printing the route
|
|
152
|
+
# owner's pid beside "running" is how a dead app kept reading as a
|
|
153
|
+
# healthy one.
|
|
104
154
|
def label_for(entry)
|
|
105
155
|
owner = entry[:agent] ? " #{entry[:agent]}" : ""
|
|
106
156
|
if entry[:pid] == 0
|
|
107
157
|
"(alias, #{entry[:alive]}#{owner})"
|
|
158
|
+
elsif entry[:backend_pid]
|
|
159
|
+
"(backend #{entry[:backend_pid]}, #{entry[:alive]}#{owner})"
|
|
108
160
|
else
|
|
109
|
-
"(
|
|
161
|
+
"(owner #{entry[:pid]}, #{entry[:alive]}#{owner})"
|
|
110
162
|
end
|
|
111
163
|
end
|
|
112
164
|
|
|
@@ -115,6 +167,7 @@ module Yamine
|
|
|
115
167
|
# report the probe, not a process.
|
|
116
168
|
def route_label(ctx, route)
|
|
117
169
|
entry = { pid: route["pid"], alive: alive_state(ctx, route) }
|
|
170
|
+
entry[:backend_pid] = ctx.backend_pid_for(route)
|
|
118
171
|
label_for(entry)
|
|
119
172
|
end
|
|
120
173
|
|
|
@@ -247,13 +300,37 @@ module Yamine
|
|
|
247
300
|
2
|
|
248
301
|
end
|
|
249
302
|
|
|
250
|
-
# Touch tmp/restart.txt so a supervised
|
|
251
|
-
|
|
303
|
+
# Touch tmp/restart.txt so a supervised app's backend is stopped.
|
|
304
|
+
#
|
|
305
|
+
# What happens next is not the same for every route, so it is said
|
|
306
|
+
# rather than assumed: a managed socket app is rebooted on the next
|
|
307
|
+
# request, while a `yamine start` app is stopped and has to be
|
|
308
|
+
# started again (the daemon cannot rebuild a tcp backend — see
|
|
309
|
+
# Supervisor#rebootable?). The old one-liner promised a reboot for
|
|
310
|
+
# both, and for a tcp route nothing was even watching the file.
|
|
311
|
+
def restart(ctx, _args)
|
|
252
312
|
path = File.join(Dir.pwd, "tmp", "restart.txt")
|
|
253
313
|
require "fileutils"
|
|
254
314
|
FileUtils.mkdir_p(File.dirname(path))
|
|
255
315
|
FileUtils.touch(path)
|
|
256
|
-
puts "Touched #{path}
|
|
316
|
+
puts "Touched #{path}."
|
|
317
|
+
# The routes this directory registered, found by spec.dir rather
|
|
318
|
+
# than by resolving the app: the file was written either way, and
|
|
319
|
+
# a directory with no config/local.yml must not turn a no-op into
|
|
320
|
+
# a config error.
|
|
321
|
+
here = File.expand_path(Dir.pwd)
|
|
322
|
+
entries = ctx.store.load_routes.select { |r| r.dig("spec", "dir") == here }
|
|
323
|
+
if entries.empty?
|
|
324
|
+
puts "No yamine app is registered for this directory — nothing will be restarted."
|
|
325
|
+
return
|
|
326
|
+
end
|
|
327
|
+
entries.each do |entry|
|
|
328
|
+
if entry["kind"] == "socket"
|
|
329
|
+
puts "#{entry["hostname"]} reboots on the next request."
|
|
330
|
+
else
|
|
331
|
+
puts "#{entry["hostname"]} will be stopped; start it again with `yamine start`."
|
|
332
|
+
end
|
|
333
|
+
end
|
|
257
334
|
end
|
|
258
335
|
|
|
259
336
|
# Tail the shared app log (default 50 lines); --follow streams.
|
data/lib/yamine/cli/system.rb
CHANGED
|
@@ -517,15 +517,15 @@ module Yamine
|
|
|
517
517
|
# expiring, or renamed.
|
|
518
518
|
#
|
|
519
519
|
# The ordering is the whole point, so it lives in one place rather
|
|
520
|
-
# than at each call site.
|
|
521
|
-
#
|
|
522
|
-
#
|
|
523
|
-
# the CA is current
|
|
524
|
-
# lets the proxy come up serving a
|
|
525
|
-
#
|
|
520
|
+
# than at each call site. The state-dir marker is only half of it
|
|
521
|
+
# (`Trust.trusted?` also asks the OS): a marker that says "we
|
|
522
|
+
# trusted this" survives a keychain that never took the setting,
|
|
523
|
+
# so checking it before the CA is current — or on its own — reads
|
|
524
|
+
# a stale match, skips trust, and lets the proxy come up serving a
|
|
525
|
+
# CA no browser trusts.
|
|
526
526
|
def ca_current_and_trusted?(dir = Certs.state_dir)
|
|
527
527
|
Certs.ensure_ca(dir)
|
|
528
|
-
|
|
528
|
+
Trust.trusted?(dir)
|
|
529
529
|
end
|
|
530
530
|
|
|
531
531
|
# Root-only CA trust: the System keychain (all users, no prompt).
|
|
@@ -946,23 +946,33 @@ module Yamine
|
|
|
946
946
|
|
|
947
947
|
yamine start # block until every route is healthy, then supervise
|
|
948
948
|
yamine start --no-wait # fire-and-forget (register routes immediately)
|
|
949
|
+
yamine start --detach # boot in the background, return once the app is healthy
|
|
949
950
|
yamine start --json # machine-readable result payload (--wait only)
|
|
950
951
|
yamine start -- --help # pass --help to the app, not here
|
|
951
952
|
|
|
952
953
|
Options are passed through to the boot path:
|
|
953
954
|
--variant <v> --tld <tld> --force --app-port <port>
|
|
954
|
-
--wait (alias, default) --no-wait --json
|
|
955
|
+
--wait (alias, default) --no-wait --json --detach
|
|
955
956
|
|
|
956
957
|
Default waits: every process is spawned concurrently, each
|
|
957
958
|
healthcheck (or TCP accept when none is declared) is polled,
|
|
958
959
|
routes are registered only when all are healthy, and `ready:`
|
|
959
960
|
is printed with exit 0. On failure everything spawned is
|
|
960
961
|
killed, the failed process + its log tail is printed, and
|
|
961
|
-
the CLI exits 1 — no half-booted routes. `--no-wait` keeps
|
|
962
|
-
|
|
962
|
+
the CLI exits 1 — no half-booted routes. `--no-wait` keeps the
|
|
963
|
+
old fire-and-forget path.
|
|
964
|
+
|
|
965
|
+
--detach forks the boot: the child owns the tree, its pid is
|
|
966
|
+
the one recorded in routes.json, and this process waits for
|
|
967
|
+
the app to answer, prints the URL, that pid and the log path
|
|
968
|
+
under the state dir, then exits 0. It is idempotent — a tree
|
|
969
|
+
already running for this directory is reported, not started
|
|
970
|
+
again. --no-wait is ignored with it: the point of detaching is
|
|
971
|
+
that the command returns once the app is actually serving.
|
|
963
972
|
|
|
964
973
|
Setup failures become hard errors pointing at `yamine setup`;
|
|
965
974
|
non-interactive CI without a running proxy exits immediately.
|
|
975
|
+
|
|
966
976
|
HELP
|
|
967
977
|
return
|
|
968
978
|
end
|
data/lib/yamine/cli.rb
CHANGED
|
@@ -83,6 +83,7 @@ module Yamine
|
|
|
83
83
|
|
|
84
84
|
Usage:
|
|
85
85
|
yamine start One-setup-and-go: setup if needed, then boot -> https://<app>.localhost
|
|
86
|
+
yamine start --detach Same, in the background; returns once the app is healthy
|
|
86
87
|
yamine setup One-shot workstation setup without booting (run once)
|
|
87
88
|
yamine Bare form of `start` — boots every process in config/local.yml
|
|
88
89
|
yamine get <name> Print URL for a service
|
|
@@ -103,10 +104,10 @@ module Yamine
|
|
|
103
104
|
yamine hosts sync|clean Manage /etc/hosts entries
|
|
104
105
|
yamine kamal <variant> Preview-deploy snippet for Kamal
|
|
105
106
|
yamine stop Stop this app's backend + routes
|
|
106
|
-
yamine restart Touch tmp/restart.txt
|
|
107
|
+
yamine restart Touch tmp/restart.txt (a supervised app is stopped)
|
|
107
108
|
yamine log [-F] [n] Tail (or follow) log/development.log
|
|
108
109
|
|
|
109
|
-
Flags: --variant, --tld, --force, --app-port, --wait (default), --no-wait, --json, --branch
|
|
110
|
+
Flags: --variant, --tld, --force, --app-port, --wait (default), --no-wait, --detach, --json, --branch
|
|
110
111
|
Env: YAMINE_VARIANT/TLD/PORT/STATE_DIR/AGENT, YAMINE_BRANCH=1
|
|
111
112
|
HELP
|
|
112
113
|
end
|
data/lib/yamine/doctor.rb
CHANGED
|
@@ -192,6 +192,15 @@ module Yamine
|
|
|
192
192
|
unless Certs.trusted?(dir)
|
|
193
193
|
return Check.new(name: "ca", ok: false, message: "CA not trusted — run: yamine trust")
|
|
194
194
|
end
|
|
195
|
+
# The marker says WE trusted this certificate; it says nothing about
|
|
196
|
+
# whether the OS recorded a trust setting for it. A CA sitting in
|
|
197
|
+
# the keychain with no trust setting is the state that produced
|
|
198
|
+
# ERR_CERT_AUTHORITY_INVALID on every route, with the marker
|
|
199
|
+
# cheerfully reporting success.
|
|
200
|
+
unless Trust.trusted?(dir)
|
|
201
|
+
return Check.new(name: "ca", ok: false,
|
|
202
|
+
message: "CA is installed but macOS does not trust it — run: yamine trust")
|
|
203
|
+
end
|
|
195
204
|
|
|
196
205
|
# Trusted CAs that are not the one now on disk. They accumulate from
|
|
197
206
|
# CA regeneration (missing, expiring, or renamed — the ask-local
|