react_on_rails 17.1.0 → 17.2.0.rc.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. checksums.yaml +4 -4
  2. data/Gemfile +1 -1
  3. data/Gemfile.development_dependencies +2 -2
  4. data/Gemfile.lock +65 -73
  5. data/Steepfile +1 -0
  6. data/docs/agent/install-and-upgrade.md +8 -3
  7. data/docs/agent/rsc-adoption.md +3 -0
  8. data/lib/generators/react_on_rails/base_generator.rb +2 -1
  9. data/lib/generators/react_on_rails/generated_webpack_config_pair.rb +84 -0
  10. data/lib/generators/react_on_rails/generator_helper.rb +2 -4
  11. data/lib/generators/react_on_rails/generator_messages.rb +1 -1
  12. data/lib/generators/react_on_rails/install_generator.rb +2 -1
  13. data/lib/generators/react_on_rails/js_dependency_manager.rb +14 -14
  14. data/lib/generators/react_on_rails/pro/USAGE +14 -3
  15. data/lib/generators/react_on_rails/pro_generator.rb +1 -1
  16. data/lib/generators/react_on_rails/pro_setup.rb +41 -213
  17. data/lib/generators/react_on_rails/rsc_setup.rb +7 -16
  18. data/lib/generators/react_on_rails/templates/pro/base/config/initializers/react_on_rails_pro.rb.tt +1 -1
  19. data/lib/react_on_rails/agent_guardrails.rb +217 -39
  20. data/lib/react_on_rails/dev/process_manager.rb +14 -0
  21. data/lib/react_on_rails/dev/server_manager.rb +1727 -103
  22. data/lib/react_on_rails/doctor.rb +875 -44
  23. data/lib/react_on_rails/helper.rb +3 -3
  24. data/lib/react_on_rails/react_component/render_options.rb +5 -7
  25. data/lib/react_on_rails/server_rendering_pool/ruby_embedded_java_script.rb +27 -116
  26. data/lib/react_on_rails/utils.rb +65 -0
  27. data/lib/react_on_rails/version.rb +1 -1
  28. data/lib/react_on_rails/version_checker/lockfile_resolution.rb +170 -0
  29. data/lib/react_on_rails/version_checker.rb +43 -22
  30. data/rakelib/run_rspec.rake +6 -1
  31. data/sig/react_on_rails/configuration.rbs +28 -26
  32. data/sig/react_on_rails/controller.rbs +2 -1
  33. data/sig/react_on_rails/dev/process_manager.rbs +1 -0
  34. data/sig/react_on_rails/dev/server_manager.rbs +150 -10
  35. data/sig/react_on_rails/git_utils.rbs +5 -1
  36. data/sig/react_on_rails/locales.rbs +4 -4
  37. data/sig/react_on_rails/prerender_error.rbs +2 -2
  38. data/sig/react_on_rails/test_helper.rbs +15 -1
  39. data/sig/react_on_rails/utils.rbs +2 -2
  40. data/sig/react_on_rails/version_checker/lockfile_resolution.rbs +31 -0
  41. data/sig/react_on_rails/version_checker.rbs +30 -7
  42. metadata +4 -1
@@ -2,6 +2,7 @@
2
2
 
3
3
  require "English"
4
4
  require "fileutils"
5
+ require "json"
5
6
  require "net/http"
6
7
  require "open3"
7
8
  require "optparse"
@@ -9,9 +10,12 @@ require "rainbow"
9
10
  require "erb"
10
11
  require "rbconfig"
11
12
  require "socket"
13
+ require "tempfile"
14
+ require "timeout"
12
15
  require "time"
13
16
  require "uri"
14
17
  require "yaml"
18
+ require_relative "../node_renderer_procfile"
15
19
  require_relative "../packer_utils"
16
20
  require_relative "../shakapacker_config_helpers"
17
21
  require_relative "../system_checker"
@@ -31,6 +35,75 @@ module ReactOnRails
31
35
  CLEAN_SHAKAPACKER_ENVIRONMENTS = %w[development test production].freeze
32
36
  OPEN_BROWSER_WAIT_TIMEOUT = 60
33
37
  OPEN_BROWSER_POLL_INTERVAL = 0.5
38
+
39
+ # Per-app-directory run state written by `bin/dev` and consumed by
40
+ # `bin/dev kill`. See the "Scoped dev shutdown" section further down.
41
+ DEV_SESSION_RELATIVE_PATH = File.join("tmp", "react_on_rails", "dev-session.json")
42
+ DEV_SESSION_LOCK_RELATIVE_PATH = File.join("tmp", "react_on_rails", "dev-session.lock")
43
+ DEV_SESSION_SCHEMA = 1
44
+ # Grace given to a graceful stop (`overmind quit` / SIGTERM) before escalating.
45
+ SHUTDOWN_TERM_GRACE_SECS = 10
46
+ # Grace given to the forceful stop (`overmind kill` / SIGKILL) before giving up.
47
+ SHUTDOWN_KILL_GRACE_SECS = 5
48
+ SHUTDOWN_POLL_INTERVAL_SECS = 0.2
49
+ # Bounded budget for the Overmind control-socket probe. `UNIXSocket.new`
50
+ # blocks indefinitely when the accept queue is full or the server is
51
+ # paused, which would hang `bin/dev kill` inside an otherwise bounded
52
+ # shutdown loop. Mirrors FileManager::SOCKET_PROBE_TIMEOUT_SECS.
53
+ OVERMIND_PROBE_TIMEOUT_SECS = 0.15
54
+ # O_NOFOLLOW so a symlink planted at the session path fails loudly rather
55
+ # than having its target silently truncated. `overmind_endpoint_owned?`
56
+ # resolves realpath for the same reason on the read side.
57
+ # Windows also requires binary mode before Ruby honors SHARE_DELETE. Both
58
+ # the old session handle and the tempfile remain open across the atomic
59
+ # rename, and the published handle remains open when its path is deleted.
60
+ DEV_SESSION_DELETE_SHARING_FLAGS = File::BINARY | File::SHARE_DELETE
61
+ DEV_SESSION_OPEN_FLAGS = File::RDWR | File::CREAT | DEV_SESSION_DELETE_SHARING_FLAGS |
62
+ (defined?(File::NOFOLLOW) ? File::NOFOLLOW : 0)
63
+ # Opening an existing fixed lock must remain separate from creating one:
64
+ # including CREAT here would bypass the bounded CREAT|EXCL race below.
65
+ DEV_SESSION_LOCK_OPEN_FLAGS = File::RDWR |
66
+ (defined?(File::NOFOLLOW) ? File::NOFOLLOW : 0) |
67
+ (defined?(File::NONBLOCK) ? File::NONBLOCK : 0)
68
+ DEV_SESSION_LOCK_WRITE_FLAGS = File::WRONLY |
69
+ (defined?(File::NOFOLLOW) ? File::NOFOLLOW : 0) |
70
+ (defined?(File::NONBLOCK) ? File::NONBLOCK : 0)
71
+ DEV_SESSION_LOCK_CREATE_FLAGS = DEV_SESSION_LOCK_OPEN_FLAGS | File::CREAT | File::EXCL
72
+ # The read side needs the same O_NOFOLLOW guarantee as the write side: it
73
+ # is the path that leads to signalling, so a symlink here is worth more to
74
+ # an attacker than one on the write path. O_NONBLOCK additionally keeps a
75
+ # FIFO planted at this path from blocking the open forever - without it
76
+ # the "must be a regular file" check below is unreachable, because the
77
+ # open never returns.
78
+ DEV_SESSION_READ_FLAGS = File::RDONLY | DEV_SESSION_DELETE_SHARING_FLAGS |
79
+ (defined?(File::NOFOLLOW) ? File::NOFOLLOW : 0) |
80
+ (defined?(File::NONBLOCK) ? File::NONBLOCK : 0)
81
+ # Bounded budget for an Overmind control command and for each cleanup
82
+ # phase after that budget expires. The launcher execs the selected
83
+ # Overmind binary so the isolated process-group leader is the real
84
+ # control client, not a wrapper that can die before reaping its child.
85
+ OVERMIND_CONTROL_TIMEOUT_SECS = 10
86
+ # A phase controls every discovered socket under one shared deadline.
87
+ # Without this second bound, N wedged endpoints each consumed a fresh
88
+ # command timeout and made shutdown latency grow linearly with N.
89
+ OVERMIND_CONTROL_BATCH_TIMEOUT_SECS = 10
90
+ OVERMIND_CONTROL_TERMINATION_GRACE_SECS = 1
91
+ OVERMIND_CONTROL_RUNNER = <<~RUBY
92
+ require "react_on_rails/dev"
93
+
94
+ manager = ReactOnRails::Dev::ServerManager
95
+ exit(manager.send(:execute_overmind_command, ARGV) ? 0 : 1)
96
+ RUBY
97
+ DEV_SESSION_CLAIM_ATTEMPTS = 2
98
+
99
+ # Markers that identify an app root when a command is run from a
100
+ # subdirectory. Checked nearest-first while walking up from the cwd.
101
+ APP_ROOT_MARKERS = [File.join("config", "environment.rb"), "Gemfile", DEV_SESSION_RELATIVE_PATH].freeze
102
+
103
+ # Shutdown outcomes that must never be reported as success: `bin/dev kill`
104
+ # exits non-zero on these, and `bin/dev clean` refuses to delete anything.
105
+ SHUTDOWN_FAILURE_STATUSES = %i[refused unverified].freeze
106
+
34
107
  DOCS_BASE_URL = "https://reactonrails.com/docs"
35
108
  DEV_SERVER_AND_TESTING_DOCS_URL = "#{DOCS_BASE_URL}/building-features/dev-server-and-testing/".freeze
36
109
  TESTING_CONFIGURATION_DOCS_URL = "#{DOCS_BASE_URL}/building-features/testing-configuration/".freeze
@@ -61,27 +134,34 @@ module ReactOnRails
61
134
  end
62
135
  end
63
136
 
137
+ # Stops the development processes owned by THIS app directory and
138
+ # verifies they are gone before reporting success.
139
+ #
140
+ # This is deliberately not an OS-wide development-process cleanup:
141
+ # nothing is signalled unless it can be positively attributed to the
142
+ # current app root. See the "Scoped dev shutdown" section below for
143
+ # the ownership model and its known limitation.
144
+ # Returns the shutdown status symbol so callers can distinguish a
145
+ # verified stop from a refusal. See SHUTDOWN_FAILURE_STATUSES.
64
146
  def kill_processes
65
- puts "🔪 Killing all development processes..."
147
+ puts "🔪 Stopping this app's development processes..."
148
+ puts " #{current_app_root}"
66
149
  puts ""
67
150
 
68
- # Run every cleanup step unconditionally so a successful first step
69
- # (e.g. pattern-based kill) doesn't leave stale port-bound processes
70
- # or socket/pid files behind. `.any?` still gives us the
71
- # "anything actually got killed?" signal for the summary message.
72
- killed_any = [
73
- kill_running_processes,
74
- kill_port_processes(killable_ports),
75
- cleanup_socket_files
76
- ].any?
77
-
78
- print_kill_summary(killed_any)
151
+ status, blockers = shutdown_dev_session
152
+ print_kill_summary(status, blockers)
153
+ status
79
154
  end
80
155
 
156
+ # Returns true when the cleanup ran to completion without warnings.
157
+ # Deleting bundles is refused outright when the dev processes could not
158
+ # be stopped - removing output from under a still-running dev server
159
+ # leaves it serving files this command just deleted.
81
160
  def clean_generated_assets_and_caches
82
161
  puts "🧹 Cleaning generated bundles and caches..."
83
162
  puts ""
84
- kill_processes
163
+ return false if shutdown_failed?(kill_processes) && abort_clean_after_failed_shutdown
164
+
85
165
  puts ""
86
166
  print_shakapacker_config_status
87
167
  puts ""
@@ -94,14 +174,17 @@ module ReactOnRails
94
174
  else
95
175
  puts "⚠️ Cleanup completed with warnings"
96
176
  end
177
+
178
+ clean_finished_without_warnings
97
179
  end
98
180
 
99
181
  # Fallback port list for the port-scan kill path. Uses the base-port
100
182
  # derived ports when REACT_ON_RAILS_BASE_PORT / CONDUCTOR_PORT is set,
101
183
  # so `bin/dev kill` in a worktree on ports 5000/5001/5002 targets the
102
184
  # right ports instead of the 3000/3001 default. Falls back to
103
- # [3000, 3001] when no base port is configured, plus the renderer port
104
- # when Pro renderer support is active. Uses PortSelector's pure
185
+ # [3000, 3035] when no base port is configured, plus a renderer port
186
+ # when explicit configuration or an active generated Procfile identifies
187
+ # one. Uses PortSelector's pure
105
188
  # #base_port_hash so no "Base port detected" banner prints during a kill.
106
189
  #
107
190
  # In base-port mode we include base[:renderer] whenever the Pro gem is
@@ -109,12 +192,12 @@ module ReactOnRails
109
192
  # user has explicitly claimed this port range, and `bin/dev kill` is
110
193
  # usually invoked from a fresh shell where RENDERER_PORT / *_URL aren't
111
194
  # carried over from the dev session — so requiring env-var presence
112
- # would let a stale renderer survive. Pattern-based killing
113
- # (development_processes / node.*react[-_]on[-_]rails) does NOT catch
114
- # the Pro renderer because it runs as `node renderer/node-renderer.js`
195
+ # would let a stale renderer survive. Pattern-based process discovery
196
+ # (`node.*react[-_]on[-_]rails`) does NOT catch the Pro renderer because
197
+ # it runs as `node renderer/node-renderer.js`
115
198
  # with no "react_on_rails" substring in the command line. Port-based
116
- # killing is the only reliable path. The default-port branch keeps the
117
- # tighter renderer_env_signal? guard via configured_renderer_port_for_kill
199
+ # killing is the only reliable path. The default-port branch requires an
200
+ # explicit renderer signal or an active recognized generated Procfile
118
201
  # because 3800 is a shared default that could belong to an unrelated process.
119
202
  def killable_ports
120
203
  base = PortSelector.base_port_hash
@@ -135,27 +218,83 @@ module ReactOnRails
135
218
  end
136
219
 
137
220
  def default_killable_ports
138
- ports = [3000, 3001]
139
- if pro_renderer_active?
140
- renderer_port = configured_renderer_port_for_kill
141
- ports << renderer_port if renderer_port
142
- end
143
- ports
221
+ ports = [default_rails_kill_port, default_dev_server_kill_port]
222
+ ports.concat(configured_renderer_ports_for_kill)
223
+ ports.uniq
224
+ end
225
+
226
+ def default_rails_kill_port
227
+ raw = ENV.fetch("PORT", nil)
228
+ PortSelector.valid_port_string?(raw) ? raw.strip.to_i : PortSelector::DEFAULT_RAILS_PORT
229
+ end
230
+
231
+ # `Procfile.dev` runs `bin/shakapacker-dev-server`, whose port comes from
232
+ # SHAKAPACKER_DEV_SERVER_PORT or `dev_server.port` in
233
+ # config/shakapacker.yml and defaults to 3035 - it is NOT "the Rails port
234
+ # plus one". The hard-coded 3001 this replaced meant a default-mode kill
235
+ # neither stopped nor even looked at a dev server on 3035, so
236
+ # `bin/dev kill` could report a verified shutdown while it was still
237
+ # serving. Widening the scanned list is safe: the cwd-attribution filter
238
+ # still decides whether anything is actually signalled.
239
+ def default_dev_server_kill_port
240
+ raw = ENV.fetch("SHAKAPACKER_DEV_SERVER_PORT", nil)
241
+ return raw.strip.to_i if PortSelector.valid_port_string?(raw)
242
+
243
+ configured = configured_dev_server_port
244
+ return configured if configured
245
+
246
+ PortSelector::DEFAULT_WEBPACK_PORT
144
247
  end
145
248
 
146
- def configured_renderer_port_for_kill
249
+ def configured_dev_server_port
250
+ value = development_dev_server_config["port"]
251
+ PortSelector.valid_port_string?(value.to_s) ? value.to_s.strip.to_i : nil
252
+ rescue StandardError
253
+ nil
254
+ end
255
+
256
+ def configured_renderer_ports_for_kill
147
257
  raw_port = ENV.fetch("RENDERER_PORT", nil)
148
- return raw_port.strip.to_i if valid_port_string?(raw_port)
258
+ return [raw_port.strip.to_i] if valid_port_string?(raw_port)
149
259
 
150
260
  local_url_port = local_renderer_url_port_for_kill
151
- return local_url_port if local_url_port
152
- return nil if remote_renderer_url_configured?
261
+ return [local_url_port] if local_url_port
262
+ return [] if remote_renderer_url_configured?
263
+
264
+ procfile_ports = renderer_procfile_ports_for_kill
265
+ return procfile_ports if procfile_ports.any?
266
+
267
+ # Only fall back to the default renderer port when the user has set a
268
+ # renderer env var. An active generated Procfile is an independent,
269
+ # stronger signal and was handled above. Without either signal, 3800
270
+ # could belong to an unrelated process in an OSS+Pro-gem app.
271
+ renderer_env_signal? ? [3800] : []
272
+ end
153
273
 
154
- # Only fall back to the default renderer port when the user has set
155
- # at least one renderer env var. Without that signal (Pro gem loaded
156
- # but no renderer ever started), `bin/dev kill` would otherwise
157
- # target an unrelated process bound to 3800 in OSS+Pro-gem apps.
158
- renderer_env_signal? ? 3800 : nil
274
+ # Read only the known bin/dev launchers and accept only active renderer
275
+ # commands that use the generated `${RENDERER_PORT:-PORT}` form. This
276
+ # establishes the effective local fallback without treating a loaded
277
+ # Pro gem, a comment, or an unrelated Procfile service as evidence.
278
+ def renderer_procfile_ports_for_kill(root = current_app_root)
279
+ ReactOnRails::NodeRendererProcfile::DEFAULT_COMMANDS.keys.flat_map do |procfile|
280
+ renderer_procfile_default_ports(File.join(root, procfile))
281
+ end.uniq
282
+ end
283
+
284
+ def renderer_procfile_default_ports(path)
285
+ return [] unless File.file?(path)
286
+
287
+ File.foreach(path).filter_map do |line|
288
+ next if line.match?(/^\s*#/)
289
+
290
+ active_command = line.sub(/\s+#.*$/, "")
291
+ next unless active_command.match?(ReactOnRails::NodeRendererProcfile::PROCESS_WITH_RENDERER_PORT_REGEX)
292
+
293
+ raw_port = active_command[/\bRENDERER_PORT=\$\{RENDERER_PORT:-(\d+)\}/, 1]
294
+ raw_port.to_i if valid_port_string?(raw_port)
295
+ end
296
+ rescue SystemCallError, IOError
297
+ []
159
298
  end
160
299
 
161
300
  def local_renderer_url_port_for_kill
@@ -182,44 +321,11 @@ module ReactOnRails
182
321
  end
183
322
  end
184
323
 
185
- def development_processes
186
- {
187
- "rails" => "Rails server",
188
- "node.*react[-_]on[-_]rails" => "React on Rails Node processes",
189
- "overmind" => "Overmind process manager",
190
- "foreman" => "Foreman process manager",
191
- "ruby.*puma" => "Puma server",
192
- "webpack-dev-server" => "Webpack dev server",
193
- "bin/shakapacker-dev-server" => "Shakapacker dev server"
194
- }
195
- end
196
-
197
- def kill_running_processes
198
- killed_any = false
199
-
200
- development_processes.each do |pattern, description|
201
- pids = find_process_pids(pattern)
202
- next unless pids.any?
203
-
204
- puts " ☠️ Killing #{description} (PIDs: #{pids.join(', ')})"
205
- terminate_processes(pids)
206
- killed_any = true
207
- end
208
-
209
- killed_any
210
- end
211
-
212
- def find_process_pids(pattern)
213
- stdout, _status = Open3.capture2("pgrep", "-f", pattern, err: File::NULL)
214
- stdout.split("\n").map(&:to_i).reject { |pid| pid == Process.pid }
215
- rescue Errno::ENOENT
216
- # pgrep command not found
217
- []
218
- end
219
-
220
- def terminate_processes(pids)
324
+ # Signals every pid in `pids`. Callers must have already attributed
325
+ # each pid to this app directory - this helper does no filtering.
326
+ def terminate_processes(pids, signal = "TERM")
221
327
  pids.each do |pid|
222
- Process.kill("TERM", pid)
328
+ Process.kill(signal, pid)
223
329
  rescue Errno::ESRCH, ArgumentError, RangeError
224
330
  # Process already stopped, or invalid signal/PID - silently skip
225
331
  nil
@@ -230,57 +336,125 @@ module ReactOnRails
230
336
  end
231
337
  end
232
338
 
339
+ # Last-resort path used when there is no live dev-session owner to
340
+ # control. Only LISTEN sockets are considered, and every candidate pid
341
+ # must have a working directory inside this app root before it is
342
+ # signalled. Listeners that cannot be positively attributed are
343
+ # reported as diagnostics and left alone.
233
344
  def kill_port_processes(ports)
234
- killed_any = false
235
-
236
- ports.each do |port|
237
- pids = find_port_pids(port)
238
- next unless pids.any?
345
+ scan = classify_port_listeners(ports)
346
+ report_unsignalled_listeners(scan)
347
+ return false if scan[:owned].empty?
239
348
 
240
- puts " ☠️ Killing process on port #{port} (PIDs: #{pids.join(', ')})"
241
- terminate_processes(pids)
242
- killed_any = true
349
+ scan[:owned].each do |port, pids|
350
+ puts " ☠️ Stopping this app's process on port #{port} (PIDs: #{pids.join(', ')})"
243
351
  end
244
352
 
245
- killed_any
353
+ stop_attributed_pids(scan[:owned].values.flatten.uniq)
354
+ true
355
+ end
356
+
357
+ # Everything the scan turned up that this app directory will not signal.
358
+ def report_unsignalled_listeners(scan)
359
+ report_foreign_listeners(scan[:foreign])
360
+ report_unattributable_listeners(scan[:unattributable])
361
+ report_unavailable_port_probes(scan[:unavailable])
246
362
  end
247
363
 
364
+ def stop_attributed_pids(pids)
365
+ terminate_processes(pids)
366
+ wait_until(SHUTDOWN_TERM_GRACE_SECS) { pids.none? { |pid| process_alive?(pid) } }
367
+ survivors = pids.select { |pid| process_alive?(pid) }
368
+ return if survivors.empty?
369
+
370
+ puts " ☠️ Escalating to KILL for survivors (PIDs: #{survivors.join(', ')})"
371
+ terminate_processes(survivors, "KILL")
372
+ wait_until(SHUTDOWN_KILL_GRACE_SECS) { survivors.none? { |pid| process_alive?(pid) } }
373
+ end
374
+
375
+ # Returns the pids holding a LISTEN socket on `port`.
376
+ #
377
+ # `-sTCP:LISTEN` matters: the unfiltered `lsof -ti :PORT` this replaced
378
+ # also matched *client* sockets connected to the port (a browser tab, a
379
+ # curl, another app's HTTP client), and those pids were then signalled.
248
380
  def find_port_pids(port)
249
- stdout, _status = Open3.capture2("lsof", "-ti", ":#{port}", err: File::NULL)
250
- stdout.split("\n").map(&:to_i).reject { |pid| pid == Process.pid }
381
+ probe_port_listeners(port).first
382
+ end
383
+
384
+ # Returns [pids, :ok] or [[], :unavailable].
385
+ #
386
+ # `find_port_pids` keeps the plain-Array contract for the public API and
387
+ # the signalling path, but verification needs to tell "lsof ran and found
388
+ # nothing" apart from "lsof could not run".
389
+ #
390
+ # Caveat, deliberately not papered over: lsof exits 1 both when it
391
+ # matches nothing and for some soft failures, so only a failure to run at
392
+ # all (missing binary, spawn failure) or an exit status above 1 can be
393
+ # reported as :unavailable.
394
+ def probe_port_listeners(port)
395
+ stdout, status = Open3.capture2("lsof", "-nP", "-t", "-iTCP:#{port}", "-sTCP:LISTEN", err: File::NULL)
396
+ exit_status = status.respond_to?(:exitstatus) ? status.exitstatus.to_i : 0
397
+ return [[], :unavailable] if exit_status > 1
398
+
399
+ [stdout.split("\n").map(&:to_i).reject { |pid| pid <= 1 || pid == Process.pid }, :ok]
251
400
  rescue StandardError
252
401
  # lsof command not found or other error (permission denied, etc.)
402
+ [[], :unavailable]
403
+ end
404
+
405
+ # Read-only probe used by `bin/dev test-watch` to notice an existing
406
+ # Shakapacker watcher. Deliberately NOT part of the shutdown path:
407
+ # `pgrep -f` matches command lines machine-wide, so its results say
408
+ # nothing about which checkout a process belongs to and must never be
409
+ # used as signal authority.
410
+ def find_process_pids(pattern)
411
+ stdout, _status = Open3.capture2("pgrep", "-f", pattern, err: File::NULL)
412
+ stdout.split("\n").map(&:to_i).reject { |pid| pid == Process.pid }
413
+ rescue Errno::ENOENT
414
+ # pgrep command not found
253
415
  []
254
416
  end
255
417
 
418
+ # Removes this app's Overmind sockets and Rails pid file. Mirrors
419
+ # FileManager#cleanup_overmind_sockets so renamed/copied variants like
420
+ # overmind-4100.sock are removed during `bin/dev kill`, not just at
421
+ # startup. A socket that still answers is left in place - removing a
422
+ # live endpoint would strand the process manager behind it.
256
423
  def cleanup_socket_files
257
- # Mirrors FileManager#cleanup_overmind_sockets so renamed/copied
258
- # variants like overmind-4100.sock are removed during `bin/dev kill`,
259
- # not just at startup.
260
- overmind_sockets = Dir.glob("tmp/sockets/overmind*.sock")
261
- files = [".overmind.sock", *overmind_sockets, "tmp/pids/server.pid"].uniq
262
- killed_any = false
424
+ root = current_app_root
425
+ files = stale_cleanup_candidates(root)
426
+ cleaned_any = false
263
427
 
264
428
  files.each do |file|
265
429
  next unless File.exist?(file)
430
+ # Remove a socket only when it is positively dead; an unreachable
431
+ # probe must not be taken as permission to delete a live endpoint.
432
+ next if File.socket?(file) && overmind_endpoint_state(file) != :gone
266
433
 
267
- puts " 🧹 Removing #{file}"
434
+ puts " 🧹 Removing #{relative_to_app_root(file, root)}"
268
435
  File.delete(file)
269
- killed_any = true
436
+ cleaned_any = true
270
437
  rescue StandardError
271
438
  nil
272
439
  end
273
440
 
274
- killed_any
441
+ cleaned_any
275
442
  end
276
443
 
277
- def print_kill_summary(killed_any)
278
- if killed_any
444
+ def print_kill_summary(status, blockers = [])
445
+ case status
446
+ when :verified
447
+ puts ""
448
+ puts "✅ This app's development processes are stopped, and verified gone"
449
+ puts "💡 You can now run 'bin/dev' for a clean start"
450
+ when :recovered_stale
279
451
  puts ""
280
- puts "✅ All processes terminated and sockets cleaned"
452
+ puts "✅ Cleaned up stale dev session state - nothing was running"
281
453
  puts "💡 You can now run 'bin/dev' for a clean start"
454
+ when :nothing_running
455
+ puts " ℹ️ No development processes owned by this app directory are running"
282
456
  else
283
- puts " ℹ️ No development processes found running"
457
+ print_kill_failure(status, blockers)
284
458
  end
285
459
  end
286
460
 
@@ -339,9 +513,9 @@ module ReactOnRails
339
513
  open_browser: options[:open_browser],
340
514
  open_browser_once: options[:open_browser_once])
341
515
  when "kill"
342
- kill_processes
516
+ exit 1 if shutdown_failed?(kill_processes)
343
517
  when "clean"
344
- clean_generated_assets_and_caches
518
+ exit 1 unless clean_generated_assets_and_caches
345
519
  when "help"
346
520
  show_help
347
521
  when "test-watch"
@@ -361,6 +535,1456 @@ module ReactOnRails
361
535
 
362
536
  private
363
537
 
538
+ # =================================================================
539
+ # Scoped dev shutdown
540
+ #
541
+ # `bin/dev kill` must never terminate processes belonging to another
542
+ # checkout of the same app. Ownership is therefore established when the
543
+ # session STARTS rather than guessed at kill time:
544
+ #
545
+ # * `bin/dev` claims `tmp/react_on_rails/dev-session.json` with an
546
+ # exclusive `flock` that it holds for the whole session, and records
547
+ # the absolute realpath of the app root it belongs to, its own pid,
548
+ # the process group it leads (when it leads one), and the Overmind
549
+ # endpoint it would use.
550
+ # * `bin/dev kill` re-opens that file. Failing to take the lock is
551
+ # positive proof the owner is still alive; taking it is positive
552
+ # proof the owner is gone and the recorded numbers are stale. A bare
553
+ # pid or pgid is never authority on its own, because pids are
554
+ # recycled - the lock is the liveness proof, and the recorded app
555
+ # root is the identity proof.
556
+ # * Anything that cannot be positively attributed to this app root is
557
+ # reported as a diagnostic and never signalled. Missing, malformed,
558
+ # foreign or unreadable state fails closed: nothing is signalled and
559
+ # no success is printed.
560
+ #
561
+ # Known limitation: the guarantee covers foreground Procfile processes
562
+ # and their ordinary descendants. A process that deliberately escapes
563
+ # its process group with `setsid`/daemonization is out of scope for this
564
+ # mechanism. Overmind's tmux server is exactly such a process, which is
565
+ # why an Overmind session is shut down through its own per-worktree
566
+ # control socket instead of by signalling the process group.
567
+ # =================================================================
568
+
569
+ def dev_session_path(root)
570
+ File.join(root, DEV_SESSION_RELATIVE_PATH)
571
+ end
572
+
573
+ def dev_session_lock_path(root)
574
+ File.join(root, DEV_SESSION_LOCK_RELATIVE_PATH)
575
+ end
576
+
577
+ # Absolute, symlink-resolved identity of the app directory this command
578
+ # belongs to. Every ownership decision is made against this.
579
+ #
580
+ # `bin/dev kill` is routinely run from a subdirectory (`../../bin/dev
581
+ # kill` from app/models). Anchoring on the cwd alone computed a root of
582
+ # <app>/app/models, found no session state there, and reported "nothing
583
+ # running" while the session was very much alive - so walk up to the
584
+ # nearest ancestor that looks like an app root and fall back to the cwd
585
+ # only when nothing matches.
586
+ def current_app_root
587
+ start = File.realpath(Dir.pwd)
588
+ app_root_ancestor(start) || start
589
+ rescue SystemCallError
590
+ File.expand_path(Dir.pwd)
591
+ end
592
+
593
+ def app_root_ancestor(start)
594
+ dir = start
595
+ loop do
596
+ return dir if APP_ROOT_MARKERS.any? { |marker| File.exist?(File.join(dir, marker)) }
597
+
598
+ parent = File.dirname(dir)
599
+ return nil if parent == dir
600
+
601
+ dir = parent
602
+ end
603
+ end
604
+
605
+ # Distinct from #path_inside_app_root? (used by `bin/dev clean`, which
606
+ # anchors on the Shakapacker config base dir): shutdown ownership is
607
+ # decided against the symlink-resolved cwd, and the root itself counts
608
+ # as inside.
609
+ def inside_dev_app_root?(path, root = current_app_root)
610
+ return false unless path.is_a?(String) && !path.empty?
611
+
612
+ path == root || path.start_with?("#{root}#{File::SEPARATOR}")
613
+ end
614
+
615
+ def relative_to_app_root(path, root = current_app_root)
616
+ prefix = "#{root}#{File::SEPARATOR}"
617
+ path.start_with?(prefix) ? path[prefix.length..].to_s : path
618
+ end
619
+
620
+ # ---- start path: claiming the session -------------------------
621
+
622
+ # Wraps a foreground process-manager run so `bin/dev kill` has a
623
+ # trustworthy, worktree-scoped handle on it. Ordinary session-document
624
+ # bookkeeping failures do not block startup - the kill path degrades to
625
+ # port-attributed cleanup instead. Failure or contention on the fixed
626
+ # ownership lock is different: a kill may be acting on the previous
627
+ # session, so an unrecorded replacement must not start.
628
+ def with_dev_session
629
+ handle = claim_dev_session(current_app_root)
630
+ begin
631
+ yield
632
+ ensure
633
+ release_dev_session(handle)
634
+ end
635
+ end
636
+
637
+ def claim_dev_session(root)
638
+ path = dev_session_path(root)
639
+ begin
640
+ FileUtils.mkdir_p(File.dirname(path))
641
+ rescue SystemCallError, IOError => e
642
+ abort_dev_session_claim(root, "the fixed ownership lock directory could not be prepared (#{e.class})")
643
+ end
644
+
645
+ claim_lock = open_dev_session_claim_lock(root)
646
+
647
+ locked = begin
648
+ claim_lock.flock(File::LOCK_EX | File::LOCK_NB)
649
+ rescue SystemCallError, IOError => e
650
+ abort_dev_session_claim(root, "the fixed ownership lock could not be acquired (#{e.class})")
651
+ end
652
+
653
+ abort_dev_session_claim(root, "another `bin/dev` start or kill operation owns the fixed lock") unless locked
654
+
655
+ if dev_session_handle_detached?(dev_session_lock_path(root), claim_lock)
656
+ abort_dev_session_claim(root, "the fixed ownership lock was replaced during acquisition")
657
+ end
658
+
659
+ DEV_SESSION_CLAIM_ATTEMPTS.times do
660
+ file = File.open(path, DEV_SESSION_OPEN_FLAGS, 0o644)
661
+ return warn_dev_session_contended(file, path, root) unless file.flock(File::LOCK_EX | File::LOCK_NB)
662
+
663
+ # Holding the lock is not enough. The previous owner can unlink the
664
+ # path between our open and our flock, leaving us locking an inode
665
+ # that `dev-session.json` no longer names - we would then write
666
+ # state nothing can ever read back while running as though we were
667
+ # recorded. Mirror of the :replaced guard on the kill side.
668
+ return write_claimed_dev_session(file, path, root) unless dev_session_handle_detached?(path, file)
669
+
670
+ release_dev_session_lock(file)
671
+ end
672
+
673
+ warn_dev_session_unrecorded("the session file kept being replaced")
674
+ nil
675
+ rescue SystemCallError, IOError => e
676
+ warn_dev_session_unrecorded(e.class)
677
+ nil
678
+ ensure
679
+ release_dev_session_lock(claim_lock)
680
+ end
681
+
682
+ def write_claimed_dev_session(file, path, root)
683
+ published = write_dev_session(path, root)
684
+ release_dev_session_lock(file)
685
+ { path:, handle: published }
686
+ rescue Interrupt
687
+ discard_partial_dev_session(published, path)
688
+ discard_partial_dev_session(file, path)
689
+ raise
690
+ rescue StandardError => e
691
+ discard_partial_dev_session(published, path)
692
+ discard_partial_dev_session(file, path)
693
+ warn_dev_session_unrecorded(e.class)
694
+ nil
695
+ end
696
+
697
+ # True when `path` no longer names this handle's inode, whether it was
698
+ # replaced or unlinked.
699
+ #
700
+ # Deliberately NOT the same question as #dev_session_replaced?, which
701
+ # treats a vanished path as "the owner tidied up after itself" rather
702
+ # than as a replacement. On the claim side a vanished path means our
703
+ # handle is orphaned, so it has to count. Do not merge the two.
704
+ def dev_session_handle_detached?(path, file)
705
+ # lstat, not stat: both opens are O_NOFOLLOW, so a handle can never
706
+ # refer to a symlink target. Resolving through a link here would let a
707
+ # symlink planted over the path still compare equal to our handle.
708
+ File.lstat(path).ino != file.stat.ino
709
+ rescue SystemCallError, IOError
710
+ true
711
+ end
712
+
713
+ def release_dev_session_lock(file)
714
+ return if file.nil? || file.closed?
715
+
716
+ file.flock(File::LOCK_UN)
717
+ rescue SystemCallError, IOError
718
+ nil
719
+ ensure
720
+ close_dev_session_handle(file)
721
+ end
722
+
723
+ def close_dev_session_handle(file)
724
+ file.close unless file.nil? || file.closed?
725
+ rescue SystemCallError, IOError
726
+ nil
727
+ end
728
+
729
+ def write_dev_session(path, root)
730
+ file = Tempfile.create(
731
+ ["dev-session-", ".json"],
732
+ File.dirname(path),
733
+ mode: DEV_SESSION_DELETE_SHARING_FLAGS
734
+ )
735
+ file.write(JSON.pretty_generate(dev_session_payload(root)))
736
+ file.flush
737
+ raise IOError, "could not lock the published dev session" unless file.flock(File::LOCK_EX | File::LOCK_NB)
738
+
739
+ file.chmod(0o644 & ~File.umask)
740
+ File.rename(file.path, path)
741
+ file
742
+ rescue StandardError, Interrupt
743
+ unless file.nil?
744
+ # The rename can succeed immediately before an asynchronous
745
+ # Interrupt. Remove `path` only when it still names this handle;
746
+ # before the rename it names the prior session and must survive
747
+ # until the outer cleanup releases that original handle.
748
+ discard_partial_dev_session(file, path)
749
+ file.close unless file.closed?
750
+ FileUtils.rm_f(file.path)
751
+ end
752
+ raise
753
+ end
754
+
755
+ # A write that fails once the lock is held (a full filesystem, say) must
756
+ # not leave a locked handle and a truncated file behind: startup would
757
+ # announce port-scoped fallback while every later `bin/dev kill` found
758
+ # malformed *locked* state and refused, until GC happened to close the
759
+ # descriptor. Drop the file and the lock before degrading.
760
+ def discard_partial_dev_session(file, path)
761
+ return if file.nil?
762
+
763
+ File.delete(path) if File.exist?(path) && !dev_session_replaced?(path, file)
764
+ rescue SystemCallError, IOError
765
+ nil
766
+ ensure
767
+ begin
768
+ release_dev_session_lock(file)
769
+ ensure
770
+ close_dev_session_handle(file)
771
+ end
772
+ end
773
+
774
+ def warn_dev_session_unrecorded(reason)
775
+ warn " ⚠️ Could not record dev session state (#{reason}). " \
776
+ "`bin/dev kill` will fall back to port-scoped cleanup."
777
+ end
778
+
779
+ def warn_dev_session_contended(file, path, root)
780
+ file.close
781
+ puts " ℹ️ Another `bin/dev` already owns #{relative_to_app_root(path, root)}; " \
782
+ "leaving its session state untouched."
783
+ puts " ⚠️ This run is NOT recorded, so `bin/dev kill` will control that other run, " \
784
+ "not this one. Stop this one from its own terminal."
785
+ nil
786
+ end
787
+
788
+ def open_dev_session_claim_lock(root)
789
+ path = dev_session_lock_path(root)
790
+ file = open_existing_or_create_dev_session_lock(path)
791
+ return file if file.stat.file?
792
+
793
+ file.close
794
+ abort_dev_session_claim(root, "the fixed ownership lock is not a regular file")
795
+ rescue Errno::ELOOP
796
+ abort_dev_session_claim(root, "the fixed ownership lock is a symlink")
797
+ rescue Errno::EISDIR
798
+ abort_dev_session_claim(root, "the fixed ownership lock is not a regular file")
799
+ rescue SystemCallError, IOError => e
800
+ release_dev_session_lock(file)
801
+ abort_dev_session_claim(root, "the fixed ownership lock could not be opened (#{e.class})")
802
+ end
803
+
804
+ def abort_dev_session_claim(root, reason)
805
+ path = dev_session_lock_path(root)
806
+ relative_path = relative_to_app_root(path, root)
807
+ warn " ❌ Cannot start `bin/dev` because #{relative_path} cannot be trusted: #{reason}. " \
808
+ "Wait for it to finish, then retry; if no operation is active, " \
809
+ "correct the ownership-lock problem first."
810
+ exit 1
811
+ end
812
+
813
+ def dev_session_payload(root)
814
+ {
815
+ "schema" => DEV_SESSION_SCHEMA,
816
+ "app_root" => root,
817
+ "pid" => Process.pid,
818
+ "pgid" => owned_process_group,
819
+ "overmind_socket" => owned_overmind_socket_path(root),
820
+ "ports" => selected_session_ports,
821
+ "started_at" => Time.now.utc.iso8601
822
+ }
823
+ end
824
+
825
+ # The ports this run actually selected, read after `configure_ports` has
826
+ # written them to ENV. Recording them means `bin/dev kill` verifies the
827
+ # session's real ports instead of re-deriving a guess from whatever
828
+ # environment the killing shell happens to have.
829
+ def selected_session_ports
830
+ %w[PORT SHAKAPACKER_DEV_SERVER_PORT RENDERER_PORT].filter_map do |var|
831
+ raw = ENV.fetch(var, nil)
832
+ PortSelector.valid_port_string?(raw) ? raw.strip.to_i : nil
833
+ end.uniq
834
+ end
835
+
836
+ # Only report a pgid we actually lead. When the shell's job control put
837
+ # `bin/dev` at the head of its own process group, that group contains
838
+ # `bin/dev` and its descendants and nothing else, so signalling it can
839
+ # never reach the parent shell or a sibling job. If we are not the
840
+ # group leader the group is somebody else's and is not ours to signal.
841
+ #
842
+ # We deliberately do NOT call `Process.setpgid`/`setpgrp` to manufacture
843
+ # a group when we do not already lead one: that detaches `bin/dev` from
844
+ # the terminal's foreground process group and breaks Ctrl-C, which is
845
+ # how people actually stop the dev server. When no group is owned (a
846
+ # non-job-control invocation, e.g. a plain background job in a script),
847
+ # the pgid is recorded as null. Shutdown can then use a recorded Overmind
848
+ # endpoint, but a live owner with neither a pgid nor an endpoint is
849
+ # refused outright. It does not guess at cwd-attributed ports.
850
+ def owned_process_group
851
+ pgid = Process.getpgrp
852
+ # The `> 1` floor mirrors #valid_session_pgid? on the read side, and
853
+ # has to be here too or we write a document we then refuse to read.
854
+ # A minimal container with no init wrapper runs `bin/dev` as PID 1 and
855
+ # typically setsid()s it, giving pgid == pid == 1; recording that made
856
+ # valid_dev_session_document? reject the whole document, so every
857
+ # `bin/dev kill` printed "Refusing to signal anything". Group 1 is
858
+ # unusable anyway - `Process.kill(sig, -1)` broadcasts rather than
859
+ # targeting a group - so a nil pgid is the honest record. Shutdown can
860
+ # then use a recorded Overmind endpoint; without one, a live owner is
861
+ # refused and only a released owner reaches cwd-attributed port cleanup.
862
+ pgid == Process.pid && pgid > 1 ? pgid : nil
863
+ rescue NotImplementedError, SystemCallError
864
+ nil
865
+ end
866
+
867
+ # The endpoint Overmind will use for this run. Only recorded when it
868
+ # lives inside this app root - an OVERMIND_SOCKET pointing elsewhere is
869
+ # not ours to control.
870
+ def owned_overmind_socket_path(root)
871
+ configured = ENV.fetch("OVERMIND_SOCKET", nil).to_s.strip
872
+ path = configured.empty? ? default_overmind_socket_path(root) : File.expand_path(configured, root)
873
+ inside_dev_app_root?(path, root) ? path : nil
874
+ end
875
+
876
+ def default_overmind_socket_path(root)
877
+ File.join(root, ".overmind.sock")
878
+ end
879
+
880
+ def release_dev_session(claim)
881
+ return if claim.nil?
882
+
883
+ path = claim[:path]
884
+ handle = claim[:handle]
885
+ File.delete(path) if File.exist?(path) && !dev_session_replaced?(path, handle)
886
+ rescue SystemCallError, IOError
887
+ nil
888
+ ensure
889
+ begin
890
+ release_dev_session_lock(handle)
891
+ ensure
892
+ close_dev_session_handle(handle)
893
+ end
894
+ end
895
+
896
+ # ---- kill path: reading and classifying the session ------------
897
+
898
+ def shutdown_dev_session
899
+ view = dev_session_view(current_app_root)
900
+ begin
901
+ view = view.merge(ports: shutdown_ports(view))
902
+ dispatch_dev_shutdown(view)
903
+ ensure
904
+ close_session_handle(view)
905
+ end
906
+ end
907
+
908
+ # Union rather than "prefer recorded". Recorded ports capture what this
909
+ # run actually selected, but they only cover variables present in the
910
+ # parent environment - the generated Pro Procfile starts the renderer
911
+ # with `RENDERER_PORT=${RENDERER_PORT:-3800}`, so a defaulted renderer
912
+ # port is never recorded. Preferring recorded exclusively hid that port
913
+ # from both signalling and verification. A superset is strictly safer:
914
+ # signalling is cwd-gated, so a wider scan cannot signal anything this
915
+ # app root does not own, and verification only gains coverage.
916
+ #
917
+ # A refused view neither signals nor verifies, so skip the derivation
918
+ # entirely - `killable_ports` prints a base-port diagnostic that would
919
+ # otherwise land immediately before "Refusing to signal anything", for a
920
+ # value that is then discarded.
921
+ def shutdown_ports(view)
922
+ return [] if view[:kind] == :refused
923
+
924
+ recorded = view.dig(:session, :ports)
925
+ recorded = [] unless recorded.is_a?(Array)
926
+ recorded | killable_ports
927
+ end
928
+
929
+ def dispatch_dev_shutdown(view)
930
+ case view[:kind]
931
+ when :refused then [:refused, view[:blockers]]
932
+ when :owner_alive then shutdown_live_owner(view)
933
+ else shutdown_without_owner(view)
934
+ end
935
+ end
936
+
937
+ def dev_session_view(root)
938
+ path = dev_session_path(root)
939
+ return { kind: :absent, root:, path:, handle: nil, claim_handle: nil } if dev_session_path_absent?(path)
940
+
941
+ lock_outcome, claim_handle = open_dev_session_read_lock(root)
942
+ if lock_outcome == :refused
943
+ return { kind: :refused, root:, path:, handle: nil, claim_handle: nil, blockers: [claim_handle] }
944
+ end
945
+
946
+ outcome, payload = open_dev_session(path)
947
+ return { kind: :absent, root:, path:, handle: nil, claim_handle: } if outcome == :absent
948
+ if outcome == :refused
949
+ return { kind: :refused, root:, path:, handle: nil, claim_handle:, blockers: [payload] }
950
+ end
951
+
952
+ kind, result = classify_dev_session(root, path, payload)
953
+ base = { kind:, root:, path:, handle: payload, claim_handle: }
954
+ kind == :refused ? base.merge(blockers: [result]) : base.merge(session: result)
955
+ end
956
+
957
+ def dev_session_path_absent?(path)
958
+ File.lstat(path)
959
+ false
960
+ rescue Errno::ENOENT
961
+ true
962
+ rescue SystemCallError, IOError
963
+ false
964
+ end
965
+
966
+ # A claimant takes this lock exclusively before it touches
967
+ # `dev-session.json`. A kill reader also takes it exclusively and retains
968
+ # it through shutdown cleanup, so neither another reader nor a claimant
969
+ # can make a stale payload look live while that reader owns the JSON
970
+ # lock. A session absent before this lock is opened remains the separate
971
+ # fallback-scan race tracked by #4943 item 2.
972
+ def open_dev_session_read_lock(root)
973
+ path = dev_session_lock_path(root)
974
+ file = open_existing_or_create_dev_session_lock(path)
975
+ unless file.stat.file?
976
+ file.close
977
+ return [:refused, "#{path} is not a regular file, so dev session ownership cannot be trusted"]
978
+ end
979
+
980
+ unless file.flock(File::LOCK_EX | File::LOCK_NB)
981
+ file.close
982
+ return [:refused, "#{path} is locked because another dev session operation is still in progress"]
983
+ end
984
+ if dev_session_handle_detached?(path, file)
985
+ file.close
986
+ return [:refused, "#{path} was replaced while its ownership lock was being acquired"]
987
+ end
988
+
989
+ [:opened, file]
990
+ rescue Errno::ELOOP
991
+ [:refused, "#{path} is a symlink; the dev session lock must be a regular file"]
992
+ rescue Errno::EISDIR
993
+ [:refused, "#{path} is not a regular file, so dev session ownership cannot be trusted"]
994
+ rescue SystemCallError, IOError => e
995
+ release_dev_session_lock(file)
996
+ [:refused, "could not lock #{path} to determine dev session ownership (#{e.class})"]
997
+ end
998
+
999
+ # Existing ownership locks are coordination handles, not state we
1000
+ # mutate. Prefer a writable descriptor because Linux NFS emulates an
1001
+ # exclusive flock with fcntl and rejects read-only descriptors. If the
1002
+ # lock is writable but not readable, keep the NFS-compatible write-only
1003
+ # path; if it is only readable, retain the local-filesystem path that can
1004
+ # still lock it read-only. O_EXCL keeps a concurrent creator a retry
1005
+ # instead of silently creating through the existing-file path.
1006
+ def open_existing_or_create_dev_session_lock(path)
1007
+ DEV_SESSION_CLAIM_ATTEMPTS.times do
1008
+ return open_existing_dev_session_lock(path)
1009
+ rescue Errno::ENOENT
1010
+ begin
1011
+ return File.open(path, DEV_SESSION_LOCK_CREATE_FLAGS, 0o644)
1012
+ rescue Errno::EEXIST
1013
+ next
1014
+ end
1015
+ end
1016
+
1017
+ open_existing_dev_session_lock(path)
1018
+ end
1019
+
1020
+ def open_existing_dev_session_lock(path)
1021
+ File.open(path, DEV_SESSION_LOCK_OPEN_FLAGS)
1022
+ rescue Errno::EACCES, Errno::EPERM, Errno::EROFS
1023
+ begin
1024
+ File.open(path, DEV_SESSION_LOCK_WRITE_FLAGS)
1025
+ rescue Errno::EACCES, Errno::EPERM, Errno::EROFS
1026
+ File.open(path, DEV_SESSION_READ_FLAGS)
1027
+ end
1028
+ end
1029
+
1030
+ # Returns [:opened, File], [:absent, nil] or [:refused, message].
1031
+ #
1032
+ # Only a genuinely missing file means "no session here". Anything else -
1033
+ # a permissions change, a read-only filesystem, a directory in the way -
1034
+ # means we could not inspect a lock that may well be held, so it fails
1035
+ # closed instead of letting the no-owner path go on to signal
1036
+ # port-attributed processes and print success.
1037
+ #
1038
+ # Opened read-only: this handle is never written through, and `flock`
1039
+ # works on a read-only descriptor, so requiring write access would
1040
+ # refuse sessions we can perfectly well inspect.
1041
+ def open_dev_session(path)
1042
+ file = File.open(path, DEV_SESSION_READ_FLAGS)
1043
+ return [:opened, file] if file.stat.file?
1044
+
1045
+ file.close
1046
+ [:refused, "#{path} is not a regular file, so it cannot be trusted as dev session state"]
1047
+ rescue Errno::ELOOP
1048
+ # A symlink here is never legitimate, and it is the highest-value
1049
+ # target in this file: a link to a locked, valid-looking document
1050
+ # naming this app root would otherwise be classified as a live owner
1051
+ # and its pgid signalled. Refuse rather than fall through to :absent -
1052
+ # :absent means "nothing here, carry on with port cleanup", which is
1053
+ # the wrong disposition for state that is actively suspicious.
1054
+ [:refused, "#{path} is a symlink; dev session state must be a regular file"]
1055
+ rescue Errno::ENOENT
1056
+ [:absent, nil]
1057
+ rescue SystemCallError, IOError => e
1058
+ release_dev_session_lock(file)
1059
+ [:refused, "could not open #{path} to determine dev session ownership (#{e.class})"]
1060
+ end
1061
+
1062
+ def classify_dev_session(root, path, file)
1063
+ session = parse_dev_session(file.read)
1064
+ return [:refused, "#{path} is not readable React on Rails dev session state"] if session.nil?
1065
+
1066
+ unless session[:app_root] == root
1067
+ return [:refused, "#{path} records app root #{session[:app_root]}, which is not #{root}"]
1068
+ end
1069
+
1070
+ case lock_dev_session(file)
1071
+ when :error
1072
+ [:refused, "could not determine whether #{path} is still owned by a running `bin/dev`"]
1073
+ when :held
1074
+ confirm_locked_dev_session(path, file, session)
1075
+ else
1076
+ classify_released_dev_session(path, file, session)
1077
+ end
1078
+ rescue SystemCallError, IOError
1079
+ [:refused, "could not read #{path}"]
1080
+ end
1081
+
1082
+ # The caller holds the shared claim lock through this ownership
1083
+ # decision, preventing a new writer from entering its publication
1084
+ # phase. Re-read under the session-file lock verdict as an additional
1085
+ # guard against an unexpected mutation; a mismatch, truncated read, or
1086
+ # invalid document still fails closed.
1087
+ def confirm_locked_dev_session(path, file, session)
1088
+ file.rewind
1089
+ confirmed = parse_dev_session(file.read)
1090
+ return [:refused, "#{path} changed while its owner was being identified"] unless confirmed == session
1091
+ return [:refused, "#{path} was replaced while it was being read"] if dev_session_replaced?(path, file)
1092
+
1093
+ [:owner_alive, confirmed]
1094
+ end
1095
+
1096
+ # The owner released the lock: either it tidied up after itself or it
1097
+ # died. Both mean the recorded pid/pgid are stale and must not be
1098
+ # signalled. A file that was *replaced* under us is a contradiction, so
1099
+ # that fails closed instead.
1100
+ def classify_released_dev_session(path, file, session)
1101
+ return [:refused, "#{path} was replaced while it was being read"] if dev_session_replaced?(path, file)
1102
+
1103
+ [:stale, session]
1104
+ end
1105
+
1106
+ def parse_dev_session(raw)
1107
+ data = JSON.parse(raw.to_s)
1108
+ return nil unless valid_dev_session_document?(data)
1109
+
1110
+ { app_root: data["app_root"], pid: data["pid"], pgid: data["pgid"],
1111
+ overmind_socket: data["overmind_socket"], ports: data["ports"] }
1112
+ rescue JSON::ParserError, TypeError
1113
+ nil
1114
+ end
1115
+
1116
+ # Validates the parsed shape before anything reads a key. `JSON.parse`
1117
+ # happily returns a String, an Integer or nil for valid-but-wrong
1118
+ # documents (`"x"`, `7`, `null`), so the Hash check has to come first -
1119
+ # otherwise a well-formed but meaningless file would raise
1120
+ # NoMethodError instead of being rejected as untrusted state.
1121
+ def valid_dev_session_document?(data)
1122
+ return false unless data.is_a?(Hash)
1123
+ return false unless data["schema"] == DEV_SESSION_SCHEMA
1124
+ return false unless valid_session_root?(data["app_root"])
1125
+ return false unless valid_session_pid?(data["pid"])
1126
+ return false unless valid_session_pgid?(data["pgid"])
1127
+ return false unless valid_session_ports?(data["ports"])
1128
+
1129
+ valid_session_socket?(data["overmind_socket"])
1130
+ end
1131
+
1132
+ # `pid` is used only for identity and display (see
1133
+ # #unreachable_owner_message), never for signalling, so every positive
1134
+ # value is legitimate - including 1, which is exactly what `bin/dev`
1135
+ # gets in a minimal Docker dev container with no init wrapper. Rejecting
1136
+ # it there disabled the whole mechanism and made every kill refuse.
1137
+ def valid_session_pid?(value)
1138
+ value.is_a?(Integer) && value.positive?
1139
+ end
1140
+
1141
+ def valid_session_socket?(value)
1142
+ value.nil? || value.is_a?(String)
1143
+ end
1144
+
1145
+ def valid_session_ports?(value)
1146
+ return true if value.nil?
1147
+ return false unless value.is_a?(Array)
1148
+
1149
+ value.all? { |port| port.is_a?(Integer) && port.between?(1, PortSelector::TCP_PORT_MAX) }
1150
+ end
1151
+
1152
+ def valid_session_root?(value)
1153
+ value.is_a?(String) && !value.empty?
1154
+ end
1155
+
1156
+ # Deliberately stricter than #valid_session_pid? above: this value IS
1157
+ # signalled, and `Process.kill(sig, -1)` broadcasts to every process the
1158
+ # user may signal. The asymmetry is load-bearing - do not harmonise it.
1159
+ def valid_session_pgid?(value)
1160
+ value.nil? || (value.is_a?(Integer) && value > 1)
1161
+ end
1162
+
1163
+ # Returns :held when another process owns the lock (the owner is
1164
+ # alive), :acquired when we took it (the owner is gone), :error when we
1165
+ # cannot tell - which the caller treats as a refusal.
1166
+ def lock_dev_session(file)
1167
+ file.flock(File::LOCK_EX | File::LOCK_NB) ? :acquired : :held
1168
+ rescue SystemCallError, IOError
1169
+ :error
1170
+ end
1171
+
1172
+ # True when the path now resolves to a different inode than the handle
1173
+ # we read - i.e. a newer `bin/dev` replaced the state underneath us.
1174
+ # A path that simply vanished is the owner tidying up after itself, not
1175
+ # a replacement.
1176
+ def dev_session_replaced?(path, file)
1177
+ # lstat for the same reason as #dev_session_handle_detached?: a
1178
+ # symlink appearing over the session path is a replacement, never a
1179
+ # match.
1180
+ File.lstat(path).ino != file.stat.ino
1181
+ rescue Errno::ENOENT
1182
+ false
1183
+ rescue SystemCallError, IOError
1184
+ true
1185
+ end
1186
+
1187
+ def close_session_handle(view)
1188
+ return if view.nil?
1189
+
1190
+ # Keep the fixed lock until after the JSON lock is gone. Reversing
1191
+ # this order would let another kill reader authenticate the stale JSON
1192
+ # against the lock still held by this reader.
1193
+ release_dev_session_lock(view[:handle])
1194
+ release_dev_session_lock(view[:claim_handle])
1195
+ rescue SystemCallError, IOError
1196
+ nil
1197
+ end
1198
+
1199
+ # ---- kill path: shutting down a live owner ---------------------
1200
+
1201
+ def shutdown_live_owner(view)
1202
+ session = view[:session]
1203
+ endpoints = live_overmind_endpoints(view[:root], session[:overmind_socket])
1204
+ pgid = signalable_pgid(session[:pgid])
1205
+ return [:refused, [unreachable_owner_message(view, session)]] if endpoints.empty? && pgid.nil?
1206
+
1207
+ request_owner_shutdown(view, pgid, endpoints)
1208
+ verify_owner_shutdown(view, pgid, endpoints)
1209
+ end
1210
+
1211
+ def unreachable_owner_message(view, session)
1212
+ "`bin/dev` (pid #{session[:pid]}) still owns #{view[:path]}, but it recorded no process group " \
1213
+ "and has no live Overmind endpoint. Stop it in its own terminal instead."
1214
+ end
1215
+
1216
+ # Escalation ladder: each step runs only if the previous one did not
1217
+ # produce a verified shutdown inside its grace window. TERM always
1218
+ # precedes KILL, and KILL only ever reaches whatever survived TERM.
1219
+ def request_owner_shutdown(view, pgid, endpoints)
1220
+ shutdown_steps(view, pgid, endpoints).each do |label, grace, action|
1221
+ # Never escalate against state a newer `bin/dev` has taken over: both
1222
+ # the recorded pgid and the socket path can have been reused, so the
1223
+ # next signal would land on somebody else's session.
1224
+ return false if session_replaced?(view)
1225
+
1226
+ puts " #{label}"
1227
+ action.call
1228
+ return true if wait_until(grace) { shutdown_settled?(view, pgid, endpoints) }
1229
+ end
1230
+
1231
+ false
1232
+ end
1233
+
1234
+ # Stop waiting once the shutdown is complete OR the session has been
1235
+ # replaced - polling out the rest of the grace window changes nothing
1236
+ # and only delays the honest failure report.
1237
+ def shutdown_settled?(view, pgid, endpoints)
1238
+ session_replaced?(view) || owner_shutdown_complete?(view, pgid, endpoints)
1239
+ end
1240
+
1241
+ def shutdown_steps(view, pgid, endpoints)
1242
+ steps = []
1243
+ if endpoints.any?
1244
+ steps << ["🛑 Asking Overmind to quit via #{endpoints.join(', ')}", SHUTDOWN_TERM_GRACE_SECS,
1245
+ -> { control_overmind_endpoints(view, "quit", endpoints) }]
1246
+ steps << ["☠️ Overmind did not quit in time; running `overmind kill`", SHUTDOWN_KILL_GRACE_SECS,
1247
+ -> { control_overmind_endpoints(view, "kill", endpoints) }]
1248
+ end
1249
+ if pgid
1250
+ steps << ["🛑 Sending TERM to this app's process group (PGID #{pgid})", SHUTDOWN_TERM_GRACE_SECS,
1251
+ -> { signal_process_group(pgid, "TERM") }]
1252
+ steps << ["☠️ Sending KILL to the survivors of PGID #{pgid}", SHUTDOWN_KILL_GRACE_SECS,
1253
+ -> { signal_process_group(pgid, "KILL") }]
1254
+ end
1255
+ steps
1256
+ end
1257
+
1258
+ def control_overmind_endpoints(view, subcommand, endpoints)
1259
+ deadline = monotonic_now + OVERMIND_CONTROL_BATCH_TIMEOUT_SECS
1260
+ all_succeeded = true
1261
+
1262
+ endpoints.each_with_index do |endpoint, index|
1263
+ # A preceding command can outlive the session we began shutting down.
1264
+ return false if session_replaced?(view)
1265
+
1266
+ remaining = deadline - monotonic_now
1267
+ unless remaining.positive?
1268
+ skipped = endpoints.length - index
1269
+ puts " ⚠️ The shared Overmind control budget was exhausted; " \
1270
+ "skipping #{skipped} remaining endpoint(s)"
1271
+ return false
1272
+ end
1273
+
1274
+ timeout_secs = [OVERMIND_CONTROL_TIMEOUT_SECS, remaining].min
1275
+ succeeded = overmind_control(subcommand, endpoint, timeout_secs:)
1276
+ all_succeeded = succeeded && all_succeeded
1277
+ end
1278
+
1279
+ all_succeeded
1280
+ end
1281
+
1282
+ # Cleanup runs only once verification has fully succeeded. Removing the
1283
+ # session file first destroyed the retry path: it carries the ports this
1284
+ # run actually selected, so deleting it on the way to reporting
1285
+ # :unverified meant the retry fell back to a re-derived guess - exactly
1286
+ # the fidelity `selected_session_ports` exists to provide, lost on the
1287
+ # one path where it matters most. It also made `print_kill_failure` tell
1288
+ # the user to remove a file that was already gone.
1289
+ def verify_owner_shutdown(view, pgid, endpoints)
1290
+ blockers = shutdown_blockers(view, pgid, endpoints)
1291
+ return [:unverified, blockers] if blockers.any?
1292
+
1293
+ leftover = outstanding_shutdown_blockers(view)
1294
+ return [:unverified, leftover] if leftover.any?
1295
+
1296
+ # The listener scan shells out to lsof once per port, and by the time
1297
+ # it returns the path may belong to a newer `bin/dev`. That window is
1298
+ # real rather than theoretical: the previous owner deletes its state
1299
+ # on exit, so #owner_release_state returns :released without ever
1300
+ # taking the lock, leaving the path free for anyone to claim. Reporting
1301
+ # :verified here would tell `bin/dev kill && bin/dev` to start a second
1302
+ # session on top of the one that just claimed this directory - and
1303
+ # cleanup_socket_files below would delete that session's socket on the
1304
+ # way out. Re-check and refuse instead of discarding the signal
1305
+ # #remove_dev_session_file already computes.
1306
+ return [:unverified, [session_replaced_blocker(view)]] if session_replaced?(view)
1307
+
1308
+ remove_dev_session_file(view)
1309
+ cleanup_socket_files
1310
+ [:verified, []]
1311
+ end
1312
+
1313
+ # Everything that has to be positively observed as gone before a
1314
+ # shutdown may be called verified: leftover listeners, ports that could
1315
+ # not be scanned, and endpoints that could not be probed.
1316
+ def outstanding_shutdown_blockers(view)
1317
+ leftover_owned_listeners(view) +
1318
+ outstanding_overmind_endpoint_blockers(view)
1319
+ end
1320
+
1321
+ def owner_shutdown_complete?(view, pgid, endpoints)
1322
+ shutdown_blockers(view, pgid, endpoints).empty?
1323
+ end
1324
+
1325
+ def shutdown_blockers(view, pgid, endpoints)
1326
+ blockers = owner_state_blockers(view)
1327
+ blockers << "process group #{pgid} still has members" if pgid && process_group_alive?(pgid)
1328
+ blockers.concat(endpoints.flat_map { |endpoint| endpoint_blockers(endpoint) })
1329
+ blockers
1330
+ end
1331
+
1332
+ def session_replaced_blocker(view)
1333
+ "#{view[:path]} was replaced by a newer `bin/dev` while this shutdown was running"
1334
+ end
1335
+
1336
+ def owner_state_blockers(view)
1337
+ case owner_release_state(view)
1338
+ when :held
1339
+ ["the `bin/dev` that owns #{view[:path]} is still running"]
1340
+ when :replaced
1341
+ [session_replaced_blocker(view)]
1342
+ else
1343
+ []
1344
+ end
1345
+ end
1346
+
1347
+ def endpoint_blockers(endpoint)
1348
+ return [] if endpoint.nil?
1349
+
1350
+ case overmind_endpoint_state(endpoint)
1351
+ when :alive
1352
+ [live_overmind_endpoint_message(endpoint)]
1353
+ when :unknown
1354
+ [unprobeable_endpoint_message(endpoint)]
1355
+ else
1356
+ []
1357
+ end
1358
+ end
1359
+
1360
+ # :released, :held, or :replaced.
1361
+ #
1362
+ # The re-check has to use the handle we already hold. `flock` locks
1363
+ # attach to the open file description, not to the process, so a second
1364
+ # descriptor on the same path can be denied by OUR OWN lock - reporting
1365
+ # "the owner is still running" when we are the lock holder and the
1366
+ # shutdown in fact succeeded. Re-locking the same description is
1367
+ # idempotent, so it answers that question honestly.
1368
+ #
1369
+ # But an old handle can also be an UNLINKED inode: if a newer `bin/dev`
1370
+ # recreated and locked the path after the previous owner removed it, we
1371
+ # would happily re-lock the dead inode and call it released - and the
1372
+ # escalation ladder could then aim `overmind kill` at the new session
1373
+ # through the reused socket path. So check for replacement first, and
1374
+ # never treat it as success. Anything we cannot observe counts as held.
1375
+ def owner_release_state(view)
1376
+ path = view[:path]
1377
+ return :released unless File.exist?(path)
1378
+
1379
+ handle = view[:handle]
1380
+ if handle && !handle.closed?
1381
+ return :replaced if dev_session_replaced?(path, handle)
1382
+
1383
+ return lock_dev_session(handle) == :acquired ? :released : :held
1384
+ end
1385
+
1386
+ # Same flags as #open_dev_session: this is the other read of the
1387
+ # session path, and it decides whether a shutdown counts as verified.
1388
+ File.open(path, DEV_SESSION_READ_FLAGS) do |file|
1389
+ file.flock(File::LOCK_EX | File::LOCK_NB) ? :released : :held
1390
+ end
1391
+ rescue Errno::ENOENT
1392
+ :released
1393
+ rescue SystemCallError, IOError
1394
+ :held
1395
+ end
1396
+
1397
+ # True once a newer `bin/dev` has claimed the path this shutdown was
1398
+ # working against. Used to abandon the escalation ladder rather than
1399
+ # keep signalling on behalf of state we no longer own.
1400
+ def session_replaced?(view)
1401
+ handle = view[:handle]
1402
+ return false if handle.nil? || handle.closed?
1403
+
1404
+ File.exist?(view[:path]) && dev_session_replaced?(view[:path], handle)
1405
+ end
1406
+
1407
+ # Verification is stricter than signalling. For signalling, "cannot
1408
+ # attribute" correctly means "leave it alone"; for verification, a probe
1409
+ # that could not run at all must not be reported as "nothing left", or a
1410
+ # transient lsof failure silently upgrades a surviving listener to
1411
+ # :verified.
1412
+ def leftover_owned_listeners(view)
1413
+ scan = classify_port_listeners(view[:ports])
1414
+ blockers = scan[:owned].map do |port, pids|
1415
+ "port #{port} is still held by this app directory (PIDs: #{pids.join(', ')})"
1416
+ end
1417
+ scan[:unattributable].each do |port, pids|
1418
+ blockers << "port #{port} still has a listener that could not be attributed " \
1419
+ "(PIDs: #{pids.join(', ')}), so it cannot be confirmed gone"
1420
+ end
1421
+ scan[:unavailable].each { |port| blockers << "could not verify port #{port} is free (lsof unavailable)" }
1422
+ blockers
1423
+ end
1424
+
1425
+ def remove_dev_session_file(view)
1426
+ handle = view[:handle]
1427
+ path = view[:path]
1428
+ return false if handle.nil? || !File.exist?(path)
1429
+ return false if dev_session_replaced?(path, handle)
1430
+
1431
+ puts " 🧹 Removing #{relative_to_app_root(path, view[:root])}"
1432
+ File.delete(path)
1433
+ true
1434
+ rescue SystemCallError
1435
+ false
1436
+ end
1437
+
1438
+ # ---- kill path: no live owner ----------------------------------
1439
+
1440
+ def shutdown_without_owner(view)
1441
+ endpoints = live_overmind_endpoints(view[:root], view.dig(:session, :overmind_socket))
1442
+ return shutdown_orphaned_overmind(view, endpoints) if endpoints.any?
1443
+
1444
+ cleaned = cleanup_socket_files
1445
+ killed = kill_port_processes(view[:ports])
1446
+
1447
+ # Verification runs even when nothing was signalled. "We looked and
1448
+ # found nothing" and "we could not look" must not collapse into the
1449
+ # same answer: with lsof missing, every relevant port goes uninspected
1450
+ # and reporting :nothing_running would let `bin/dev kill && bin/dev`
1451
+ # start a second stack on top of a live one. The same holds for an
1452
+ # in-root Overmind endpoint we could not probe - falling through to
1453
+ # port-only cleanup would call the session gone without ever reaching
1454
+ # it.
1455
+ leftover = outstanding_shutdown_blockers(view)
1456
+ return [:unverified, leftover] if leftover.any?
1457
+
1458
+ # Same rule as verify_owner_shutdown: the recorded ports outlive an
1459
+ # unverified outcome so a retry still has them.
1460
+ remove_dev_session_file(view) if view[:kind] == :stale
1461
+ return [:nothing_running, []] unless cleaned || killed || view[:kind] == :stale
1462
+
1463
+ killed ? [:verified, []] : [:recovered_stale, []]
1464
+ end
1465
+
1466
+ # An Overmind endpoint inside this app root that still answers is a
1467
+ # live, worktree-scoped handle even though the `bin/dev` that started it
1468
+ # is gone (its terminal was closed, or it was SIGKILLed). Controlling it
1469
+ # natively is scoped; guessing at pids from the stale record is not, so
1470
+ # the recorded pgid is deliberately not used here.
1471
+ def shutdown_orphaned_overmind(view, endpoints)
1472
+ puts " ℹ️ Found orphaned Overmind endpoints for this app directory; controlling them directly."
1473
+ request_owner_shutdown(view, nil, endpoints)
1474
+ verify_owner_shutdown(view, nil, endpoints)
1475
+ end
1476
+
1477
+ def shutdown_failed?(status)
1478
+ SHUTDOWN_FAILURE_STATUSES.include?(status)
1479
+ end
1480
+
1481
+ # Always returns true so the caller can read as a single guard clause.
1482
+ def abort_clean_after_failed_shutdown
1483
+ puts ""
1484
+ puts "⛔ Not cleaning: this app directory's development processes could not be stopped."
1485
+ puts " Removing bundles and caches under a running dev server would leave it serving"
1486
+ puts " output this command just deleted. Resolve the shutdown above, then retry."
1487
+ true
1488
+ end
1489
+
1490
+ def print_kill_failure(status, blockers)
1491
+ puts ""
1492
+ if status == :refused
1493
+ puts "❌ Refusing to signal anything: this app's dev session state could not be trusted"
1494
+ else
1495
+ puts "❌ Shutdown could not be verified - some processes are still running"
1496
+ end
1497
+ Array(blockers).each { |blocker| puts " • #{blocker}" }
1498
+ puts ""
1499
+ puts "💡 Resolve the blockers above before retrying. If only stale session state remains"
1500
+ puts " and you are certain nothing is running, remove #{DEV_SESSION_RELATIVE_PATH},"
1501
+ puts " then run `bin/dev kill` again."
1502
+ end
1503
+
1504
+ # ---- process / endpoint observation ----------------------------
1505
+
1506
+ # Revalidates the recorded endpoint at control time rather than
1507
+ # trusting what startup wrote: the socket must still live inside this
1508
+ # app root, still be a socket, and still answer.
1509
+ # Candidates in decreasing order of authority: what the session
1510
+ # recorded, what OVERMIND_SOCKET names in the killing shell (only when
1511
+ # it lands inside this app root), the default path, then sockets found
1512
+ # under tmp/sockets. A candidate is not an authority - each still has
1513
+ # to clear the containment and realpath checks below.
1514
+ def live_overmind_endpoint(root, recorded)
1515
+ live_overmind_endpoints(root, recorded).first
1516
+ end
1517
+
1518
+ def live_overmind_endpoints(root, recorded)
1519
+ scan_overmind_endpoints(root, recorded)[:alive] || []
1520
+ end
1521
+
1522
+ def overmind_endpoint_candidates(root, recorded, discovered = nil)
1523
+ discovered ||= overmind_socket_discovery(root)[:paths]
1524
+ [recorded, owned_overmind_socket_path(root), default_overmind_socket_path(root), *discovered].compact.uniq
1525
+ end
1526
+
1527
+ # A missing or readable empty directory is positive evidence that no
1528
+ # renamed endpoints are present. Any failure to inspect an existing
1529
+ # directory is different: verification must retain a blocker rather
1530
+ # than silently turning "could not look" into "nothing is running".
1531
+ def overmind_socket_discovery(root)
1532
+ directory = File.join(root, "tmp", "sockets")
1533
+ begin
1534
+ File.lstat(directory)
1535
+ rescue Errno::ENOENT
1536
+ return { paths: [], blockers: [] }
1537
+ rescue SystemCallError => e
1538
+ return failed_overmind_socket_discovery(directory, e)
1539
+ end
1540
+
1541
+ resolved_directory = File.realpath(directory)
1542
+ unless inside_dev_app_root?(resolved_directory, root)
1543
+ blocker = "could not inspect #{directory} for Overmind endpoints because " \
1544
+ "it resolves outside this app root; inspect the symlink and configure " \
1545
+ "an app-local socket directory before retrying"
1546
+ return {
1547
+ paths: [],
1548
+ blockers: [blocker]
1549
+ }
1550
+ end
1551
+
1552
+ paths = Dir.children(directory)
1553
+ .select { |name| name.start_with?("overmind") && name.end_with?(".sock") }
1554
+ .sort
1555
+ .map { |name| File.join(directory, name) }
1556
+ { paths:, blockers: [] }
1557
+ rescue SystemCallError => e
1558
+ failed_overmind_socket_discovery(directory, e)
1559
+ end
1560
+
1561
+ def failed_overmind_socket_discovery(directory, error)
1562
+ {
1563
+ paths: [],
1564
+ blockers: ["could not inspect #{directory} for Overmind endpoints (#{error.class})"]
1565
+ }
1566
+ end
1567
+
1568
+ # Probes every in-root candidate once and groups them by state, so a
1569
+ # bounded connect runs at most once per candidate per pass.
1570
+ def scan_overmind_endpoints(root, recorded)
1571
+ discovery = overmind_socket_discovery(root)
1572
+ scan = { discovery_blockers: discovery[:blockers].dup }
1573
+ overmind_endpoint_candidates(root, recorded, discovery[:paths]).each do |path|
1574
+ case overmind_endpoint_ownership(path, root)
1575
+ when :owned
1576
+ state = overmind_endpoint_state(path)
1577
+ (scan[state] ||= []) << path
1578
+ when :unknown
1579
+ scan[:discovery_blockers] << "could not inspect #{path} as an Overmind endpoint"
1580
+ end
1581
+ end
1582
+ scan
1583
+ end
1584
+
1585
+ # Re-scan every candidate after shutdown. A newly discovered live socket
1586
+ # or an endpoint whose state cannot be observed must block the verified
1587
+ # claim, even when it was not part of the initial control set.
1588
+ def outstanding_overmind_endpoint_blockers(view)
1589
+ scan = scan_overmind_endpoints(view[:root], view.dig(:session, :overmind_socket))
1590
+ Array(scan[:alive]).map { |path| live_overmind_endpoint_message(path) } +
1591
+ unprobeable_endpoint_blockers(scan[:unknown]) +
1592
+ Array(scan[:discovery_blockers])
1593
+ end
1594
+
1595
+ def unprobeable_endpoint_blockers(endpoints)
1596
+ Array(endpoints).map { |path| unprobeable_endpoint_message(path) }
1597
+ end
1598
+
1599
+ def unprobeable_endpoint_message(path)
1600
+ "could not determine whether the Overmind endpoint #{path} is still live"
1601
+ end
1602
+
1603
+ def live_overmind_endpoint_message(path)
1604
+ "the Overmind endpoint #{path} is still accepting connections"
1605
+ end
1606
+
1607
+ # Containment check for a control endpoint. The path is resolved before
1608
+ # comparing: `File.socket?` follows symlinks, so comparing the raw string
1609
+ # let `<root>/.overmind.sock` be a symlink pointing at ANOTHER checkout's
1610
+ # live endpoint and still pass as "ours" - and `overmind kill -s` would
1611
+ # then tear down that other checkout's session, with no session-file
1612
+ # tampering required because the default path is always a candidate.
1613
+ # Resolving also collapses `..` escapes in a recorded path.
1614
+ def overmind_endpoint_ownership(path, root = current_app_root)
1615
+ return :foreign unless path.is_a?(String) && !path.empty?
1616
+
1617
+ resolved = File.realpath(path)
1618
+ return :foreign unless inside_dev_app_root?(resolved, root)
1619
+
1620
+ File.stat(resolved).socket? ? :owned : :gone
1621
+ rescue Errno::ENOENT, Errno::ENOTDIR
1622
+ :gone
1623
+ rescue SystemCallError, IOError
1624
+ :unknown
1625
+ end
1626
+
1627
+ def overmind_endpoint_owned?(path, root = current_app_root)
1628
+ overmind_endpoint_ownership(path, root) == :owned
1629
+ end
1630
+
1631
+ # :alive, :gone, or :unknown. Same principle as the port probes: a
1632
+ # refused connection is positive evidence the endpoint is dead, but a
1633
+ # probe that could not run is not, and verification must not read the
1634
+ # second as the first.
1635
+ def overmind_endpoint_state(path)
1636
+ return :gone unless File.stat(path).socket?
1637
+
1638
+ begin
1639
+ sockaddr = Socket.sockaddr_un(path)
1640
+ rescue ArgumentError
1641
+ # Only the "too long unix socket path" case (sun_path is capped at
1642
+ # ~104/108 bytes); nothing could be listening there anyway.
1643
+ return :gone
1644
+ end
1645
+
1646
+ probe_unix_socket(sockaddr)
1647
+ rescue Errno::ENOENT, Errno::ENOTDIR
1648
+ :gone
1649
+ rescue SystemCallError, IOError
1650
+ :unknown
1651
+ end
1652
+
1653
+ # Bounded connect. A blocking `UNIXSocket.new` hangs indefinitely when
1654
+ # the server's accept queue is full or its process is paused, which
1655
+ # could stall `bin/dev kill` outside any of its own timeouts. A probe
1656
+ # that runs out of budget is :unknown, not :gone - it blocks
1657
+ # verification rather than licensing a false "verified".
1658
+ def probe_unix_socket(sockaddr)
1659
+ socket = Socket.new(Socket::AF_UNIX, Socket::SOCK_STREAM, 0)
1660
+ begin
1661
+ socket.connect_nonblock(sockaddr)
1662
+ :alive
1663
+ rescue IO::WaitWritable
1664
+ return :unknown unless socket.wait_writable(OVERMIND_PROBE_TIMEOUT_SECS)
1665
+
1666
+ socket.getsockopt(Socket::SOL_SOCKET, Socket::SO_ERROR).int.zero? ? :alive : :gone
1667
+ rescue Errno::EISCONN
1668
+ :alive
1669
+ rescue Errno::ECONNREFUSED, Errno::ENOENT, Errno::ENOTSOCK
1670
+ :gone
1671
+ rescue SystemCallError, IOError
1672
+ :unknown
1673
+ ensure
1674
+ socket.close
1675
+ end
1676
+ end
1677
+
1678
+ def overmind_endpoint_alive?(path)
1679
+ overmind_endpoint_state(path) == :alive
1680
+ end
1681
+
1682
+ # Re-checks ownership immediately before handing control to Overmind,
1683
+ # so a socket that was replaced or removed between validation and use
1684
+ # cannot be acted on.
1685
+ def overmind_control(subcommand, endpoint, timeout_secs: OVERMIND_CONTROL_TIMEOUT_SECS)
1686
+ ownership = overmind_endpoint_ownership(endpoint)
1687
+ unless ownership == :owned
1688
+ reason = ownership == :unknown ? "could not be inspected" : "is no longer this app's endpoint"
1689
+ puts " ⚠️ Skipping `overmind #{subcommand}`: #{endpoint} #{reason}"
1690
+ return false
1691
+ end
1692
+
1693
+ return true if run_overmind_command([subcommand, "-s", endpoint], timeout_secs:)
1694
+
1695
+ puts " ⚠️ `overmind #{subcommand}` did not run successfully for #{endpoint}"
1696
+ false
1697
+ rescue Errno::ENOENT, Interrupt
1698
+ puts " ⚠️ Could not run `overmind #{subcommand}` for #{endpoint}"
1699
+ false
1700
+ end
1701
+
1702
+ # Control commands have to take the same route startup took.
1703
+ #
1704
+ # ProcessManager falls back to running the process outside the Bundler
1705
+ # context when the system-installed binary is not usable inside it - and
1706
+ # that is the documented install shape for this project: install the
1707
+ # process manager globally and deliberately keep it OUT of the Gemfile.
1708
+ # A bare `system("overmind", ...)` here just repeats the context that
1709
+ # already failed at startup, so `quit` and `kill` both return false and
1710
+ # the session becomes unkillable by the tool that started it. The
1711
+ # process-group fallback cannot rescue it either: Overmind's tmux server
1712
+ # daemonizes to PPID 1 in its own process group, out of reach of any
1713
+ # group signal.
1714
+ #
1715
+ # The child launcher uses ProcessManager's availability checks and
1716
+ # Bundler API-compat shim, then execs the selected binary. Exec keeps the
1717
+ # real control client at the pid and process group this parent owns, so
1718
+ # timeout cleanup can terminate and reap that process directly.
1719
+ def run_overmind_command(args, timeout_secs: OVERMIND_CONTROL_TIMEOUT_SECS)
1720
+ pid = spawn_overmind_command(args)
1721
+ reaped = false
1722
+ status = wait_for_overmind_command(pid, timeout_secs)
1723
+ if status.is_a?(Process::Status)
1724
+ reaped = true
1725
+ return status.success? == true
1726
+ end
1727
+ if status == :reaped_without_status
1728
+ reaped = true
1729
+ puts " ⚠️ `overmind #{args.first}` finished but its exit status was unavailable; " \
1730
+ "treating it as unsuccessful"
1731
+ return false
1732
+ end
1733
+
1734
+ terminate_overmind_command(pid)
1735
+ reaped = true
1736
+ # Treated exactly like the runner reporting failure, so the escalation
1737
+ # ladder moves on to the next step instead of hanging here forever.
1738
+ puts " ⚠️ `overmind #{args.first}` did not return within " \
1739
+ "#{timeout_secs}s; moving on"
1740
+ false
1741
+ ensure
1742
+ terminate_overmind_command(pid) if pid && !reaped
1743
+ end
1744
+
1745
+ def spawn_overmind_command(args)
1746
+ lib_dir = File.expand_path("../..", __dir__)
1747
+ Process.spawn(
1748
+ RbConfig.ruby, "-I", lib_dir, "-e", OVERMIND_CONTROL_RUNNER, "--", *args,
1749
+ pgroup: true
1750
+ )
1751
+ end
1752
+
1753
+ def execute_overmind_command(args)
1754
+ ProcessManager.exec_process_if_available("overmind", args)
1755
+ end
1756
+
1757
+ def wait_for_overmind_command(pid, timeout_secs)
1758
+ deadline = monotonic_now + timeout_secs
1759
+ loop do
1760
+ waited_pid, status = Process.wait2(pid, Process::WNOHANG)
1761
+ return status if waited_pid
1762
+
1763
+ remaining = deadline - monotonic_now
1764
+ return nil unless remaining.positive?
1765
+
1766
+ sleep([SHUTDOWN_POLL_INTERVAL_SECS, remaining].min)
1767
+ end
1768
+ rescue Errno::ECHILD, Errno::ESRCH
1769
+ :reaped_without_status
1770
+ end
1771
+
1772
+ def terminate_overmind_command(pid)
1773
+ reaped = false
1774
+ signal_process_group(pid, "TERM")
1775
+ group_gone, reaped = wait_for_overmind_group_exit(
1776
+ pid, OVERMIND_CONTROL_TERMINATION_GRACE_SECS, reaped
1777
+ )
1778
+ unless group_gone
1779
+ signal_process_group(pid, "KILL")
1780
+ _group_gone, reaped = wait_for_overmind_group_exit(
1781
+ pid, OVERMIND_CONTROL_TERMINATION_GRACE_SECS, reaped
1782
+ )
1783
+ end
1784
+ ensure
1785
+ reap_overmind_command(pid) unless reaped
1786
+ end
1787
+
1788
+ def wait_for_overmind_group_exit(pid, timeout_secs, reaped)
1789
+ deadline = monotonic_now + timeout_secs
1790
+ loop do
1791
+ reaped ||= overmind_command_reaped?(pid)
1792
+ return [true, reaped] unless process_group_alive?(pid)
1793
+
1794
+ remaining = deadline - monotonic_now
1795
+ return [false, reaped] unless remaining.positive?
1796
+
1797
+ sleep([SHUTDOWN_POLL_INTERVAL_SECS, remaining].min)
1798
+ end
1799
+ end
1800
+
1801
+ def overmind_command_reaped?(pid)
1802
+ waited_pid, = Process.wait2(pid, Process::WNOHANG)
1803
+ !waited_pid.nil?
1804
+ rescue Errno::ECHILD, Errno::ESRCH
1805
+ true
1806
+ end
1807
+
1808
+ def reap_overmind_command(pid)
1809
+ Process.detach(pid).join(OVERMIND_CONTROL_TERMINATION_GRACE_SECS)
1810
+ rescue Errno::ECHILD, Errno::ESRCH
1811
+ nil
1812
+ end
1813
+
1814
+ def signalable_pgid(pgid)
1815
+ return nil unless pgid.is_a?(Integer) && pgid > 1
1816
+ return nil if pgid == own_process_group
1817
+
1818
+ pgid
1819
+ end
1820
+
1821
+ def own_process_group
1822
+ Process.getpgrp
1823
+ rescue NotImplementedError, SystemCallError
1824
+ nil
1825
+ end
1826
+
1827
+ def signal_process_group(pgid, signal)
1828
+ Process.kill(signal, -pgid)
1829
+ true
1830
+ rescue Errno::ESRCH, ArgumentError, RangeError, NotImplementedError
1831
+ false
1832
+ rescue Errno::EPERM
1833
+ puts " ⚠️ Permission denied signalling process group #{pgid}"
1834
+ false
1835
+ end
1836
+
1837
+ # Fails closed: anything we cannot positively observe as gone counts
1838
+ # as still alive, so an unverifiable shutdown is never reported as
1839
+ # success.
1840
+ def process_group_alive?(pgid)
1841
+ Process.kill(0, -pgid)
1842
+ true
1843
+ rescue Errno::ESRCH
1844
+ false
1845
+ rescue Errno::EPERM, ArgumentError, RangeError, NotImplementedError
1846
+ true
1847
+ end
1848
+
1849
+ def process_alive?(pid)
1850
+ Process.kill(0, pid)
1851
+ true
1852
+ rescue Errno::ESRCH, ArgumentError, RangeError
1853
+ false
1854
+ rescue Errno::EPERM
1855
+ true
1856
+ end
1857
+
1858
+ # Returns a scan hash with four buckets, keeping "someone else's" and
1859
+ # "could not tell" strictly apart:
1860
+ #
1861
+ # owned Hash[port => pids] positively this app directory's
1862
+ # foreign Hash[port => pids] positively somebody else's
1863
+ # unattributable Hash[port => pids] the cwd probe failed
1864
+ # unavailable Array[port] the listener probe failed
1865
+ #
1866
+ # Signalling consumes `owned` only. Verification must block on
1867
+ # `unattributable` and `unavailable` too: folding a failed probe into
1868
+ # `foreign` is what let a surviving listener be reported as gone.
1869
+ #
1870
+ # Known blind spot, measured rather than assumed: run as a normal user,
1871
+ # `lsof` only attributes sockets it can correlate through the owning
1872
+ # process, so a port held by ANOTHER user's process is reported as
1873
+ # having no listener at all rather than as an unattributable one. Such a
1874
+ # port reads as free here. That is not something this command could act
1875
+ # on anyway - it could not signal that process either - but it does mean
1876
+ # "verified" means "free of anything this user can see".
1877
+ def classify_port_listeners(ports)
1878
+ root = current_app_root
1879
+ scan = { owned: {}, foreign: {}, unattributable: {}, unavailable: [] }
1880
+
1881
+ Array(ports).uniq.each do |port|
1882
+ pids, probe = probe_port_listeners(port)
1883
+ next scan[:unavailable] << port if probe == :unavailable
1884
+ next if pids.empty?
1885
+
1886
+ grouped = pids.group_by { |pid| attribute_pid(pid, root) }
1887
+ scan[:owned][port] = grouped[:owned] if grouped[:owned]
1888
+ scan[:foreign][port] = grouped[:foreign] if grouped[:foreign]
1889
+ scan[:unattributable][port] = grouped[:unknown] if grouped[:unknown]
1890
+ end
1891
+
1892
+ scan
1893
+ end
1894
+
1895
+ def report_unattributable_listeners(unattributable)
1896
+ unattributable.each do |port, pids|
1897
+ puts " ⚠️ Could not determine who owns port #{port} (PIDs: #{pids.join(', ')}); leaving it alone"
1898
+ end
1899
+ end
1900
+
1901
+ def report_unavailable_port_probes(ports)
1902
+ return if ports.empty?
1903
+
1904
+ puts " ⚠️ Could not check port#{'s' if ports.size > 1} #{ports.join(', ')} " \
1905
+ "for listeners (lsof unavailable)"
1906
+ end
1907
+
1908
+ def report_foreign_listeners(foreign)
1909
+ foreign.each do |port, pids|
1910
+ puts " ℹ️ Leaving port #{port} alone (PIDs: #{pids.join(', ')}): " \
1911
+ "not attributable to this app directory"
1912
+ end
1913
+ end
1914
+
1915
+ # :owned, :foreign, :gone, or :unknown when nothing could be
1916
+ # established. `working_directory_for_pid` returns nil for a missing
1917
+ # tool, a permission error and unparsable output alike - none of which
1918
+ # is evidence that the process belongs to somebody else - so an
1919
+ # unreadable cwd falls through to a second, cheaper probe.
1920
+ def attribute_pid(pid, root)
1921
+ cwd = working_directory_for_pid(pid)
1922
+ return inside_dev_app_root?(cwd, root) ? :owned : :foreign if cwd
1923
+
1924
+ signal_permission_attribution(pid)
1925
+ end
1926
+
1927
+ # Fallback when the working-directory probe told us nothing.
1928
+ # `Process.kill(0, pid)` runs the existence and permission checks
1929
+ # without sending anything:
1930
+ #
1931
+ # ESRCH the pid is gone, so it is not a leftover at all
1932
+ # EPERM it exists and we may not signal it, so its user differs from
1933
+ # ours; `bin/dev` starts every dev process as the invoking
1934
+ # user, so this is somebody else's process
1935
+ # ok a signalable process whose cwd we still could not read:
1936
+ # genuinely unknown, and verification keeps blocking on it
1937
+ #
1938
+ # Narrow exception, deliberately accepted: a Procfile line that changes
1939
+ # user (`sudo -u ...`, a setuid helper) yields one of our own processes
1940
+ # that answers EPERM. We could not signal it either way, so the only
1941
+ # choice is between reporting success while it survives and refusing
1942
+ # forever - and it is still printed as a port left alone, with its pid,
1943
+ # so it stays visible rather than silently dropped.
1944
+ def signal_permission_attribution(pid)
1945
+ Process.kill(0, pid)
1946
+ :unknown
1947
+ rescue Errno::ESRCH, ArgumentError, RangeError
1948
+ :gone
1949
+ rescue Errno::EPERM
1950
+ :foreign
1951
+ end
1952
+
1953
+ # `lsof -a -p PID -d cwd -Fn` prints the process's working directory in
1954
+ # a machine-readable form. Returning nil (tooling missing, permission
1955
+ # denied, unparsable) means "not attributable", which keeps the pid out
1956
+ # of the signal set.
1957
+ def working_directory_for_pid(pid)
1958
+ stdout, status = Open3.capture2("lsof", "-a", "-p", pid.to_s, "-d", "cwd", "-Fn", err: File::NULL)
1959
+ return nil unless status.success?
1960
+
1961
+ line = stdout.split("\n").find { |entry| entry.start_with?("n") }
1962
+ return nil if line.nil?
1963
+
1964
+ File.realpath(line[1..].to_s)
1965
+ rescue StandardError
1966
+ nil
1967
+ end
1968
+
1969
+ def stale_cleanup_candidates(root)
1970
+ [default_overmind_socket_path(root), *overmind_socket_discovery(root)[:paths],
1971
+ File.join(root, "tmp", "pids", "server.pid")].uniq
1972
+ end
1973
+
1974
+ def wait_until(timeout_secs)
1975
+ deadline = monotonic_now + timeout_secs
1976
+ loop do
1977
+ return true if yield
1978
+ return false if monotonic_now >= deadline
1979
+
1980
+ sleep(SHUTDOWN_POLL_INTERVAL_SECS)
1981
+ end
1982
+ end
1983
+
1984
+ def monotonic_now
1985
+ Process.clock_gettime(Process::CLOCK_MONOTONIC)
1986
+ end
1987
+
364
1988
  def clean_targets
365
1989
  deduplicate_clean_targets(
366
1990
  shakapacker_clean_targets +
@@ -854,8 +2478,8 @@ module ReactOnRails
854
2478
  #{Rainbow('test-watch').green.bold} #{Rainbow('Watch and rebuild test assets with smart defaults').white}
855
2479
  #{Rainbow('→ Uses:').yellow} bin/shakapacker --watch (RAILS_ENV=test)
856
2480
 
857
- #{Rainbow('kill').red.bold} #{Rainbow('Kill all development processes for a clean start').white}
858
- #{Rainbow('clean').red.bold} #{Rainbow('Kill dev processes and remove generated bundles/caches').white}
2481
+ #{Rainbow('kill').red.bold} #{Rainbow('Stop the dev processes owned by this app directory').white}
2482
+ #{Rainbow('clean').red.bold} #{Rainbow('Stop dev processes, then remove generated bundles/caches').white}
859
2483
  #{Rainbow('help').blue.bold} #{Rainbow('Show this help message').white}
860
2484
  COMMANDS
861
2485
  end
@@ -1181,7 +2805,7 @@ module ReactOnRails
1181
2805
  open_browser:,
1182
2806
  open_browser_once:)
1183
2807
  ProcessManager.ensure_procfile(procfile)
1184
- ProcessManager.run_with_process_manager(procfile)
2808
+ with_dev_session { ProcessManager.run_with_process_manager(procfile) }
1185
2809
  else
1186
2810
  puts "❌ Asset precompilation failed"
1187
2811
  puts ""
@@ -1291,7 +2915,7 @@ module ReactOnRails
1291
2915
  open_browser:,
1292
2916
  open_browser_once:)
1293
2917
  ProcessManager.ensure_procfile(procfile)
1294
- ProcessManager.run_with_process_manager(procfile)
2918
+ with_dev_session { ProcessManager.run_with_process_manager(procfile) }
1295
2919
  end
1296
2920
 
1297
2921
  def run_development(procfile, verbose: false, route: nil, skip_database_check: false,
@@ -1313,7 +2937,7 @@ module ReactOnRails
1313
2937
  open_browser:,
1314
2938
  open_browser_once:)
1315
2939
  ProcessManager.ensure_procfile(procfile)
1316
- ProcessManager.run_with_process_manager(procfile)
2940
+ with_dev_session { ProcessManager.run_with_process_manager(procfile) }
1317
2941
  end
1318
2942
 
1319
2943
  def print_server_info(title, features, port = 3000, route: nil)