wurk 1.1.0 → 1.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +25 -0
  3. data/app/controllers/concerns/wurk/same_origin_guard.rb +40 -0
  4. data/app/controllers/concerns/wurk/sse_streaming.rb +48 -0
  5. data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +53 -0
  6. data/app/controllers/wurk/api/pagination.rb +60 -11
  7. data/app/controllers/wurk/api/serializers.rb +5 -1
  8. data/app/controllers/wurk/api_controller.rb +51 -70
  9. data/app/controllers/wurk/application_controller.rb +26 -0
  10. data/app/controllers/wurk/dashboard_controller.rb +23 -1
  11. data/app/controllers/wurk/extensions_controller.rb +3 -12
  12. data/app/controllers/wurk/profiles_controller.rb +6 -1
  13. data/lib/wurk/batch/death_handler.rb +13 -0
  14. data/lib/wurk/batch/server_middleware.rb +9 -5
  15. data/lib/wurk/batch.rb +3 -0
  16. data/lib/wurk/capsule.rb +34 -21
  17. data/lib/wurk/cli.rb +16 -3
  18. data/lib/wurk/client/buffered.rb +30 -10
  19. data/lib/wurk/client.rb +45 -1
  20. data/lib/wurk/component.rb +23 -9
  21. data/lib/wurk/configuration.rb +83 -18
  22. data/lib/wurk/cron.rb +13 -1
  23. data/lib/wurk/dead_set.rb +16 -1
  24. data/lib/wurk/engine.rb +17 -1
  25. data/lib/wurk/fetcher/reliable.rb +46 -13
  26. data/lib/wurk/fetcher.rb +5 -0
  27. data/lib/wurk/health.rb +74 -22
  28. data/lib/wurk/history.rb +4 -21
  29. data/lib/wurk/job_retry.rb +3 -5
  30. data/lib/wurk/launcher.rb +26 -3
  31. data/lib/wurk/limiter/server_middleware.rb +16 -1
  32. data/lib/wurk/limiter.rb +6 -5
  33. data/lib/wurk/lua/loader.rb +11 -5
  34. data/lib/wurk/lua.rb +100 -9
  35. data/lib/wurk/manager.rb +59 -19
  36. data/lib/wurk/metrics/history.rb +24 -38
  37. data/lib/wurk/metrics/queue_rollup.rb +4 -21
  38. data/lib/wurk/metrics/rollup.rb +4 -21
  39. data/lib/wurk/processor.rb +8 -8
  40. data/lib/wurk/profile_set.rb +21 -6
  41. data/lib/wurk/profiler.rb +7 -5
  42. data/lib/wurk/rails_boot.rb +176 -0
  43. data/lib/wurk/railtie.rb +19 -47
  44. data/lib/wurk/redis_connection.rb +6 -9
  45. data/lib/wurk/redis_pool.rb +148 -28
  46. data/lib/wurk/scheduled.rb +35 -12
  47. data/lib/wurk/swarm/backoff.rb +70 -0
  48. data/lib/wurk/swarm/child_boot.rb +92 -13
  49. data/lib/wurk/swarm/orphan_guard.rb +105 -0
  50. data/lib/wurk/swarm/restart.rb +196 -0
  51. data/lib/wurk/swarm.rb +194 -78
  52. data/lib/wurk/timer_loop.rb +49 -0
  53. data/lib/wurk/version.rb +1 -1
  54. data/lib/wurk/web/extension.rb +4 -1
  55. data/lib/wurk/web/pool_scope.rb +46 -0
  56. data/lib/wurk/web/rack_app.rb +2 -1
  57. data/lib/wurk/web/search.rb +77 -18
  58. data/lib/wurk/web.rb +1 -0
  59. data/lib/wurk/worker/setter.rb +6 -1
  60. data/lib/wurk.rb +6 -2
  61. data/vendor/assets/dashboard/assets/ArgsValue-DYfBiXrJ.js +1 -0
  62. data/vendor/assets/dashboard/assets/BatchDetail-DTZ2HzcD.js +1 -0
  63. data/vendor/assets/dashboard/assets/Batches-BrnXA332.js +1 -0
  64. data/vendor/assets/dashboard/assets/Busy-C57G8Xb3.js +1 -0
  65. data/vendor/assets/dashboard/assets/Cron-DlyH88oo.js +1 -0
  66. data/vendor/assets/dashboard/assets/Dashboard-B887pxWf.js +1 -0
  67. data/vendor/assets/dashboard/assets/Dead-BPA7gs-X.js +1 -0
  68. data/vendor/assets/dashboard/assets/Extension-Bunf6XuU.js +1 -0
  69. data/vendor/assets/dashboard/assets/FilterBox-FCDi4ZCU.js +1 -0
  70. data/vendor/assets/dashboard/assets/JobDetailModal-DuMdKUMm.js +2 -0
  71. data/vendor/assets/dashboard/assets/Limiters-Br0aCPMK.js +1 -0
  72. data/vendor/assets/dashboard/assets/Metrics-DxBmuywH.js +1 -0
  73. data/vendor/assets/dashboard/assets/Modal-t4FI_LaY.js +1 -0
  74. data/vendor/assets/dashboard/assets/PageHeader-CsDvJSOA.js +1 -0
  75. data/vendor/assets/dashboard/assets/Profiles-Bkhoqjlq.js +1 -0
  76. data/vendor/assets/dashboard/assets/Queues-BvhA-vfI.js +1 -0
  77. data/vendor/assets/dashboard/assets/Retries-JEpB-1Yl.js +1 -0
  78. data/vendor/assets/dashboard/assets/Scheduled-DN_FbSwP.js +1 -0
  79. data/vendor/assets/dashboard/assets/Search-DJuK0YCJ.js +1 -0
  80. data/vendor/assets/dashboard/assets/Skeleton-DOYDkzg1.js +1 -0
  81. data/vendor/assets/dashboard/assets/charts-CVK0zAnC.js +1 -0
  82. data/vendor/assets/dashboard/assets/index-BnPX9Ptn.css +1 -0
  83. data/vendor/assets/dashboard/assets/index-CZTcs-pM.js +141 -0
  84. data/vendor/assets/dashboard/assets/useResetPageOnEmpty-CoZU4b3a.js +1 -0
  85. data/vendor/assets/dashboard/assets/useSort-wQcnbdsa.js +1 -0
  86. data/vendor/assets/dashboard/index.html +3 -3
  87. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  88. metadata +55 -26
  89. data/vendor/assets/dashboard/assets/ArgsValue-BUqJa-eG.js +0 -1
  90. data/vendor/assets/dashboard/assets/BatchDetail-C5dkqAzp.js +0 -1
  91. data/vendor/assets/dashboard/assets/Batches-bgkXn8tc.js +0 -1
  92. data/vendor/assets/dashboard/assets/Busy-QSHBFXhy.js +0 -1
  93. data/vendor/assets/dashboard/assets/Cron-CUHppvTA.js +0 -1
  94. data/vendor/assets/dashboard/assets/Dashboard-CzPudckV.js +0 -1
  95. data/vendor/assets/dashboard/assets/Dead-BIq4Nz_i.js +0 -1
  96. data/vendor/assets/dashboard/assets/Extension-CW36i9R1.js +0 -1
  97. data/vendor/assets/dashboard/assets/JobDetailModal-BhNdWSp7.js +0 -2
  98. data/vendor/assets/dashboard/assets/Limiters-CiI_DfUK.js +0 -1
  99. data/vendor/assets/dashboard/assets/Metrics-BzZ8ugms.js +0 -1
  100. data/vendor/assets/dashboard/assets/Modal-DzVfgsSF.js +0 -1
  101. data/vendor/assets/dashboard/assets/PageHeader-Dp3qhX3e.js +0 -1
  102. data/vendor/assets/dashboard/assets/Profiles-BTtIlTdR.js +0 -1
  103. data/vendor/assets/dashboard/assets/Queues-BuXoxQ4W.js +0 -1
  104. data/vendor/assets/dashboard/assets/Retries-D4HAPaOQ.js +0 -1
  105. data/vendor/assets/dashboard/assets/Scheduled-6NCZYVJh.js +0 -1
  106. data/vendor/assets/dashboard/assets/Search-JaB_-52c.js +0 -1
  107. data/vendor/assets/dashboard/assets/charts-6uvCyY0x.js +0 -1
  108. data/vendor/assets/dashboard/assets/i18n-gIeA5VLo.js +0 -1
  109. data/vendor/assets/dashboard/assets/index-BDG9tvBA.css +0 -1
  110. data/vendor/assets/dashboard/assets/index-DWfWAGBc.js +0 -141
  111. data/vendor/assets/dashboard/assets/useJobSetActions-DTDaAoWr.js +0 -1
  112. data/vendor/assets/dashboard/assets/usePageParam-BcAvRko-.js +0 -1
  113. data/vendor/assets/dashboard/assets/useSort-D5Am4bGq.js +0 -1
data/lib/wurk/railtie.rb CHANGED
@@ -1,60 +1,32 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require 'rails/railtie'
4
+ require 'active_support/ordered_options'
5
+ require_relative 'rails_boot'
4
6
 
5
7
  module Wurk
6
- # Boot the swarm after the host app has fully initialized.
7
- # Skip when: WURK_DISABLED=1, Rails console mode, or Rails test env.
8
+ # Rails integration surface only: register the host-facing `config.wurk`
9
+ # namespace and wire the two boot hooks to Wurk::RailsBoot. All boot policy
10
+ # (skip/fork/embed/refuse, prefork detection) and execution lives there — this
11
+ # class exists solely to invoke the coordinator from the Rails lifecycle.
8
12
  # See docs/idea/03-process-model.md for the exact ordering.
9
13
  class Railtie < ::Rails::Railtie
10
- # This Rails process forks the workers, so it IS the server. Enter server
11
- # mode BEFORE config/initializers load — otherwise the app's
12
- # `Sidekiq.configure_server` blocks gate on `config.server?` (still false)
13
- # and are silently dropped. Gated identically to the swarm boot: a process
14
- # that won't run workers (disabled / console / test) is not a server.
15
- initializer 'wurk.server_mode', before: :load_config_initializers do
16
- Wurk.enter_server_mode unless Wurk::Railtie.skip_boot?
17
- end
14
+ # Host-facing config namespace. Pre-created so a host can write
15
+ # `config.wurk.embed_in_web = true` in config/application.rb without a
16
+ # NoMethodError — the setter needs the OrderedOptions to already exist.
17
+ # Shared with the application config via Rails' Railtie::Configuration
18
+ # @@options, so `Rails.application.config.wurk` reads back the same object.
19
+ config.wurk = ::ActiveSupport::OrderedOptions.new
18
20
 
19
- config.after_initialize do |_app|
20
- next if Wurk::Railtie.skip_boot?
21
-
22
- swarm = Wurk::Swarm.new(topology: Wurk.configuration.topology)
23
- swarm.boot
24
- # Embedded mode keeps the Rails process serving HTTP; supervise must
25
- # run somewhere or signal_queue never drains, crashed children never
26
- # respawn, and memory pressure checks never fire. Background thread
27
- # leaves the host's main thread free for Rails.
28
- Thread.new do
29
- swarm.supervise
30
- rescue StandardError => e
31
- Wurk.configuration.logger.error { "wurk supervisor thread died: #{e.class}: #{e.message}" }
32
- end
33
- end
34
-
35
- # A process that won't run workers isn't a server: skip both server mode
36
- # and the swarm boot. Console mode is detected reliably here — the console
37
- # command file defines ::Rails::Console before initializers run.
38
- def self.skip_boot?
39
- ENV['WURK_DISABLED'] == '1' ||
40
- building? ||
41
- defined?(::Rails::Console) ||
42
- ::Rails.env.test?
21
+ # Enter server mode BEFORE config/initializers load — otherwise the app's
22
+ # `Sidekiq.configure_server` blocks gate on `config.server?` (still false)
23
+ # and are silently dropped. RailsBoot decides whether this process serves.
24
+ initializer 'wurk.server_mode', before: :load_config_initializers do |app|
25
+ Wurk::RailsBoot.enter_server_mode_if_serving(app)
43
26
  end
44
27
 
45
- # A build/precompile step must never fork the swarm (#247). The default
46
- # Rails Dockerfile runs `SECRET_KEY_BASE_DUMMY=1 ./bin/rails
47
- # assets:precompile`; that loads `:environment` → fires after_initialize,
48
- # but there's no Redis during `docker build`, so a fork would hang/fail the
49
- # build. Same for other env-loading rake tasks (db:prepare, db:migrate).
50
- # The real server path is unaffected: `rails server` / `puma` boot through
51
- # Rails::Command, not Rake, and don't set the dummy secret.
52
- def self.building?
53
- return true if ENV.key?('SECRET_KEY_BASE_DUMMY')
54
-
55
- defined?(::Rake) && ::Rake.application.top_level_tasks.any?
56
- rescue StandardError
57
- false
28
+ config.after_initialize do |app|
29
+ Wurk::RailsBoot.boot(app)
58
30
  end
59
31
  end
60
32
  end
@@ -10,21 +10,18 @@ module Wurk
10
10
  # (connection_pool-backed, redis-client adapter), which is `.with`-compatible
11
11
  # with everything that expects a Sidekiq pool.
12
12
  #
13
- # Accepts Sidekiq's option keys (`url`, `size`, `pool_timeout`, `name`),
14
- # string- or symbol-keyed. Anything omitted falls back to RedisPool's defaults
15
- # (URL = ENV["REDIS_URL"] or redis://localhost:6379/0).
13
+ # Accepts Sidekiq's option keys (`url`, `size`, `pool_timeout`, `name`) plus
14
+ # the socket knobs (`connect_timeout`/`read_timeout`/`write_timeout`/
15
+ # `reconnect_attempts`/`driver`), string- or symbol-keyed. Anything omitted
16
+ # falls back to RedisPool's defaults (URL = ENV["REDIS_URL"] or
17
+ # redis://localhost:6379/0).
16
18
  module RedisConnection
17
19
  # Housekeeping/standalone default; per-capsule pools size to concurrency.
18
20
  DEFAULT_POOL_SIZE = 10
19
21
 
20
22
  def self.create(options = {})
21
23
  opts = options.transform_keys(&:to_sym)
22
- RedisPool.new(
23
- size: opts[:size] || DEFAULT_POOL_SIZE,
24
- url: opts[:url] || RedisPool::DEFAULT_URL,
25
- timeout: opts[:pool_timeout] || RedisPool::DEFAULT_TIMEOUT,
26
- name: opts[:name] || RedisPool::DEFAULT_NAME
27
- )
24
+ RedisPool.new(size: opts.delete(:size) || DEFAULT_POOL_SIZE, **opts)
28
25
  end
29
26
  end
30
27
  end
@@ -9,41 +9,86 @@ module Wurk
9
9
  # across forks: the parent closes the pool before fork, each child opens a
10
10
  # fresh one (see docs/idea/03-process-model.md, steps 3 and 5).
11
11
  #
12
- # Retry policy on conn-level errors: close + retry once for messages
13
- # prefixed READONLY / NOREPLICAS / UNBLOCKED. Spec: docs/target/sidekiq-free.md §26.
12
+ # #with absorbs transient Redis failures (production incident #101):
13
+ # * READONLY / NOREPLICAS / UNBLOCKED — a failover happened; close and retry
14
+ # once immediately so redis-client redials the new primary (spec §26).
15
+ # * ConnectionError (incl. CannotConnect / Read- / WriteTimeout) — a blip;
16
+ # close and retry with exponential backoff up to CONN_MAX_ATTEMPTS, then
17
+ # raise. At-least-once tolerates the rare duplicate and JobRetry re-runs
18
+ # the job anyway, so retrying at the pool layer is strictly better.
19
+ # * ConnectionPool::TimeoutError — checkout starved; retry once after a
20
+ # short jittered pause, then raise (sizing is the fix, not queuing).
21
+ # Every retry and final give-up is reported through the injected `on_error`
22
+ # telemetry hook (Wurk::Configuration#on_redis_error).
14
23
  class RedisPool
15
- DEFAULT_URL = ENV.fetch('REDIS_URL', 'redis://localhost:6379/0')
16
- DEFAULT_TIMEOUT = 1.0
17
- DEFAULT_NAME = 'default'
24
+ DEFAULT_URL = ENV.fetch('REDIS_URL', 'redis://localhost:6379/0')
25
+ DEFAULT_NAME = 'default'
18
26
 
19
- # Server-side messages where Sidekiq (and therefore Wurk) closes the
20
- # connection and retries the block exactly once. Any other RedisClient::Error
21
- # propagates immediately.
27
+ # ConnectionPool checkout wait — how long #with blocks for a free slot.
28
+ DEFAULT_POOL_TIMEOUT = 1.0
29
+
30
+ # Socket-level timeouts handed to RedisClient, split apart from the checkout
31
+ # wait above. read/write are deliberately wider than connect so a briefly-
32
+ # slow-but-alive Redis (RDB fork pause, a large BLMOVE payload) doesn't
33
+ # spuriously ReadTimeout — the production incident (#101) the single
34
+ # dual-use timeout caused. reconnect_attempts re-dials a dropped socket once.
35
+ DEFAULT_CONNECT_TIMEOUT = 1.0
36
+ DEFAULT_READ_TIMEOUT = 2.5
37
+ DEFAULT_WRITE_TIMEOUT = 2.5
38
+ DEFAULT_RECONNECT_ATTEMPTS = 1
39
+
40
+ # Server-side messages where the connection is closed and the block retried
41
+ # exactly once. READONLY is itself a RedisClient::ConnectionError subclass,
42
+ # so this message match must be tested BEFORE the generic ConnectionError
43
+ # backoff below (otherwise a failover would sleep instead of redialing).
22
44
  RETRYABLE_MSG = /\A(READONLY|NOREPLICAS|UNBLOCKED)/
23
45
 
24
- attr_reader :size, :url, :timeout, :name
46
+ # ConnectionError backoff: CONN_MAX_ATTEMPTS total tries, sleeping
47
+ # (BASE * 2**attempt) + rand*JITTER before each retry. The 1.0s + 2.0s pair
48
+ # rides out a sub-4s blip; the jitter de-syncs a fleet reconnecting at once.
49
+ CONN_MAX_ATTEMPTS = 3
50
+ CONN_BACKOFF_BASE = 0.5
51
+ CONN_BACKOFF_JITTER = 0.25
52
+
53
+ # Checkout-timeout retry: one retry after a jittered pause in
54
+ # [POOL_RETRY_MIN, POOL_RETRY_MIN + POOL_RETRY_SPREAD). No loop — sustained
55
+ # checkout starvation is a sizing bug, not something to queue behind.
56
+ POOL_RETRY_MIN = 0.1
57
+ POOL_RETRY_SPREAD = 0.2
58
+
59
+ attr_reader :size, :url, :name, :pool_timeout, :client_config
25
60
 
26
- def initialize(size:, url: DEFAULT_URL, timeout: DEFAULT_TIMEOUT, name: DEFAULT_NAME)
27
- @size = size
28
- @url = url
29
- @timeout = timeout
30
- @name = name
31
- @pool = ConnectionPool.new(size: size, timeout: timeout) { build_client }
61
+ # Takes the standard Sidekiq `config.redis` hash: `pool_timeout` tunes the
62
+ # ConnectionPool checkout; `connect_timeout`/`read_timeout`/`write_timeout`/
63
+ # `reconnect_attempts` plus any other key (driver, ssl_params, …) forward
64
+ # verbatim to RedisClient.config. `on_error` is an optional callable fired
65
+ # per retry / final give-up with { error:, attempt:, retried:, pool: }.
66
+ def initialize(size:, name: DEFAULT_NAME, on_error: nil, **options)
67
+ @size = size
68
+ @name = name
69
+ @on_error = on_error
70
+ @pool_timeout = options.fetch(:pool_timeout, DEFAULT_POOL_TIMEOUT)
71
+ @client_config = build_client_config(options)
72
+ @url = @client_config[:url]
73
+ @pool = ConnectionPool.new(size: size, timeout: @pool_timeout) { build_client }
32
74
  end
33
75
 
34
- def with
35
- @pool.with do |conn|
36
- attempts = 0
37
- begin
38
- yield conn
39
- rescue RedisClient::Error => e
40
- raise unless RETRYABLE_MSG.match?(e.message.to_s)
41
- raise if attempts >= 1
42
-
43
- attempts += 1
44
- safe_close(conn)
45
- retry
76
+ # Checkout a connection and run the block. ConnectionPool::TimeoutError is
77
+ # raised by @pool.with *before* the block runs, so it is caught out here
78
+ # (the in-block #run rescue never sees it) — one retry, then raise.
79
+ def with(&block)
80
+ checkout_retried = false
81
+ begin
82
+ @pool.with { |conn| run(conn, &block) }
83
+ rescue ConnectionPool::TimeoutError => e
84
+ if checkout_retried
85
+ notify_error(e, attempt: 2, retried: false)
86
+ raise
46
87
  end
88
+ checkout_retried = true
89
+ notify_error(e, attempt: 1, retried: true)
90
+ sleep(checkout_delay)
91
+ retry
47
92
  end
48
93
  end
49
94
 
@@ -51,17 +96,40 @@ module Wurk
51
96
  @pool.shutdown { |conn| safe_close(conn) }
52
97
  end
53
98
 
99
+ # Redis INFO parsed to a Hash, with this pool's own `size` / `available`
100
+ # slot counts merged in — one call gives a heartbeat both Redis health and
101
+ # local pool saturation. (Real Redis INFO has no `size`/`available` field.)
54
102
  def info
55
103
  with { |conn| parse_info(conn.call('INFO')) }
104
+ .merge('size' => @size, 'available' => available)
105
+ end
106
+
107
+ # Free (unchecked-out) slots right now. Local and cheap — no Redis round
108
+ # trip — so a monitor can poll it without perturbing the pool.
109
+ def available
110
+ @pool.available
56
111
  end
57
112
 
58
113
  private
59
114
 
115
+ # Socket config forwarded to RedisClient.config. Host-supplied keys win over
116
+ # the defaults; `pool_timeout` is dropped (it's a pool concern, not a socket
117
+ # one) and unknown keys pass straight through.
118
+ def build_client_config(options)
119
+ {
120
+ url: DEFAULT_URL,
121
+ connect_timeout: DEFAULT_CONNECT_TIMEOUT,
122
+ read_timeout: DEFAULT_READ_TIMEOUT,
123
+ write_timeout: DEFAULT_WRITE_TIMEOUT,
124
+ reconnect_attempts: DEFAULT_RECONNECT_ATTEMPTS
125
+ }.merge(options.except(:pool_timeout)).freeze
126
+ end
127
+
60
128
  # Wrapped in the CompatClient decorator so `Sidekiq.redis { |c| c.smembers }`
61
129
  # method-style commands work like Sidekiq 7+ (#204). Wurk's own code paths
62
130
  # use #call, which the decorator forwards.
63
131
  def build_client
64
- RedisClientAdapter::CompatClient.new(RedisClient.config(url: @url, timeout: @timeout).new_client)
132
+ RedisClientAdapter::CompatClient.new(RedisClient.config(**@client_config).new_client)
65
133
  end
66
134
 
67
135
  def safe_close(conn)
@@ -70,6 +138,58 @@ module Wurk
70
138
  nil
71
139
  end
72
140
 
141
+ # Runs the block on the checked-out `conn`, retrying transient RedisClient
142
+ # errors in place: the same slot is reused across retries (redis-client
143
+ # redials a closed socket lazily), so a busy fetcher can't leak checkouts.
144
+ def run(conn)
145
+ attempts = 0
146
+ begin
147
+ attempts += 1
148
+ yield conn
149
+ rescue RedisClient::Error => e
150
+ plan = retry_plan(e, attempts)
151
+ raise if plan == :propagate
152
+
153
+ notify_error(e, attempt: attempts, retried: plan != :exhausted)
154
+ raise if plan == :exhausted
155
+
156
+ safe_close(conn)
157
+ sleep(backoff_delay(attempts)) if plan == :backoff
158
+ retry
159
+ end
160
+ end
161
+
162
+ # Pure classification of a RedisClient error against the attempt count:
163
+ # :failover → close + immediate retry (a primary swap)
164
+ # :backoff → close + sleep + retry (a connection blip)
165
+ # :exhausted → ConnectionError past the cap; report and give up
166
+ # :propagate → not transient (or a spent failover); raise as-is
167
+ def retry_plan(err, attempts)
168
+ if RETRYABLE_MSG.match?(err.message.to_s)
169
+ attempts > 1 ? :propagate : :failover
170
+ elsif err.is_a?(RedisClient::ConnectionError)
171
+ attempts >= CONN_MAX_ATTEMPTS ? :exhausted : :backoff
172
+ else
173
+ :propagate
174
+ end
175
+ end
176
+
177
+ def notify_error(error, attempt:, retried:)
178
+ return if @on_error.nil?
179
+
180
+ @on_error.call({ error:, attempt:, retried:, pool: @name })
181
+ rescue StandardError
182
+ nil
183
+ end
184
+
185
+ def backoff_delay(attempt)
186
+ (CONN_BACKOFF_BASE * (2**attempt)) + (rand * CONN_BACKOFF_JITTER)
187
+ end
188
+
189
+ def checkout_delay
190
+ POOL_RETRY_MIN + (rand * POOL_RETRY_SPREAD)
191
+ end
192
+
73
193
  def parse_info(raw)
74
194
  raw.to_s.each_line.with_object({}) do |line, h|
75
195
  line = line.strip
@@ -39,9 +39,7 @@ module Wurk
39
39
  # client. `now` is captured once per set so a slow loop on one ZSET
40
40
  # can't keep grabbing newly-scheduled jobs from a moving window.
41
41
  def enqueue_jobs(sorted_sets = SETS)
42
- @config.redis do |conn|
43
- sorted_sets.each { |sset| drain_set(conn, sset) }
44
- end
42
+ sorted_sets.each { |sset| drain_set(sset) }
45
43
  end
46
44
 
47
45
  def terminate
@@ -50,18 +48,34 @@ module Wurk
50
48
 
51
49
  private
52
50
 
53
- def drain_set(conn, sset)
51
+ # Pop under one checkout (the pooled EVALSHA loop), push outside it —
52
+ # `@client.push` checks out its own connection, so nesting it inside
53
+ # `@config.redis` would hold two checkouts from the same pool at once
54
+ # per job, halving effective pool concurrency under load.
55
+ def drain_set(sset)
54
56
  now = real_time.to_s
55
57
  loop do
56
58
  break if @done
57
59
 
58
- jobstr = Wurk::Lua::Loader.eval_cached(conn, :zpopbyscore, keys: [sset], argv: [now])
60
+ jobstr = @config.redis { |conn| Wurk::Lua::Loader.eval_cached(conn, :zpopbyscore, keys: [sset], argv: [now]) }
59
61
  break unless jobstr
60
62
 
61
- @client.push(Wurk.load_json(jobstr))
63
+ push_promoted(jobstr, sset)
62
64
  end
63
65
  end
64
66
 
67
+ # A raising `@client.push` (bad payload, transient Redis error) must not
68
+ # abort the drain and strand the remaining due jobs until the next poll —
69
+ # rescue per-job, report, continue. The already-popped job IS lost here
70
+ # (ZPOPBYSCORE removed it); that pop→push loss window is the default
71
+ # scheduler's known tradeoff — `reliable_scheduler!` (ReliableEnq) is the
72
+ # loss-free fix, so we don't re-engineer around it here.
73
+ def push_promoted(jobstr, sset)
74
+ @client.push(Wurk.load_json(jobstr))
75
+ rescue StandardError => e
76
+ handle_exception(e, { context: 'scheduler_promote', set: sset })
77
+ end
78
+
65
79
  def real_time
66
80
  ::Process.clock_gettime(::Process::CLOCK_REALTIME)
67
81
  end
@@ -78,6 +92,12 @@ module Wurk
78
92
  class ReliableEnq
79
93
  include Component
80
94
 
95
+ # Batched: each Lua call promotes at most this many members (the script
96
+ # runs atomically, so one giant sweep would stall Redis for every
97
+ # client). `promote` loops until a short batch signals the backlog is
98
+ # dry, stopping early on terminate.
99
+ PROMOTE_BATCH = 500
100
+
81
101
  def initialize(container)
82
102
  @config = container
83
103
  @done = false
@@ -96,12 +116,15 @@ module Wurk
96
116
  private
97
117
 
98
118
  def promote(conn, sset)
99
- Wurk::Lua::Loader.eval_cached(
100
- conn,
101
- :reliable_schedule_promote,
102
- keys: [sset, Keys::QUEUES_SET],
103
- argv: [real_time.to_s, Keys::QUEUE_PREFIX]
104
- )
119
+ loop do
120
+ promoted = Wurk::Lua::Loader.eval_cached(
121
+ conn,
122
+ :reliable_schedule_promote,
123
+ keys: [sset, Keys::QUEUES_SET],
124
+ argv: [real_time.to_s, Keys::QUEUE_PREFIX, real_ms.to_s, PROMOTE_BATCH.to_s]
125
+ ).to_i
126
+ break if promoted < PROMOTE_BATCH || @done
127
+ end
105
128
  end
106
129
 
107
130
  def real_time
@@ -0,0 +1,70 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Wurk
4
+ class Swarm
5
+ # Per-key exponential backoff timer with survival-based reset. Pure
6
+ # bookkeeping — no sleeping, no I/O — so the supervise thread never blocks
7
+ # on it: it records a failure, then asks `ready?` each tick and acts only
8
+ # once the delay has elapsed.
9
+ #
10
+ # Delay grows base → 2·base → 4·base … capped at `cap`. A key whose child
11
+ # survived at least `reset_after` seconds counts as a fresh first failure,
12
+ # so a slot that ran healthily for a while doesn't inherit an old crash
13
+ # streak. Keyed by slot index for crash-respawn and (in Swarm::Restart) for
14
+ # replacement-retry delays.
15
+ class Backoff
16
+ BASE = 1.0
17
+ CAP = 30.0
18
+ RESET_AFTER = 60.0
19
+
20
+ def initialize(base: BASE, cap: CAP, reset_after: RESET_AFTER, clock: nil)
21
+ @base = base
22
+ @cap = cap
23
+ @reset_after = reset_after
24
+ @clock = clock || -> { ::Process.clock_gettime(::Process::CLOCK_MONOTONIC) }
25
+ @streak = Hash.new(0)
26
+ @due_at = {}
27
+ end
28
+
29
+ # Record a failure and schedule the next attempt. `lifetime` is how long
30
+ # the failed child lived; >= reset_after resets the streak. Returns the
31
+ # delay applied.
32
+ def fail(key, lifetime: 0.0)
33
+ @streak[key] = 0 if lifetime >= @reset_after
34
+ @streak[key] += 1
35
+ delay = [@base * (2**(@streak[key] - 1)), @cap].min
36
+ @due_at[key] = now + delay
37
+ delay
38
+ end
39
+
40
+ # A retry is scheduled and not yet issued.
41
+ def pending?(key)
42
+ @due_at.key?(key)
43
+ end
44
+
45
+ # The scheduled delay has elapsed (or nothing is scheduled).
46
+ def ready?(key)
47
+ at = @due_at[key]
48
+ at.nil? || now >= at
49
+ end
50
+
51
+ # Mark the scheduled retry as issued without resetting the streak, so a
52
+ # child that crashes straight back escalates toward the cap.
53
+ def consume(key)
54
+ @due_at.delete(key)
55
+ end
56
+
57
+ # Forget the key entirely (slot retired or restart succeeded).
58
+ def clear(key)
59
+ @streak.delete(key)
60
+ @due_at.delete(key)
61
+ end
62
+
63
+ private
64
+
65
+ def now
66
+ @clock.call
67
+ end
68
+ end
69
+ end
70
+ end
@@ -3,6 +3,8 @@
3
3
  require_relative '../component'
4
4
  require_relative '../launcher'
5
5
  require_relative '../fetcher/reliable'
6
+ require_relative '../lua'
7
+ require_relative 'orphan_guard'
6
8
 
7
9
  module Wurk
8
10
  class Swarm
@@ -21,11 +23,23 @@ module Wurk
21
23
 
22
24
  CHILD_SIGNALS = { 'TERM' => :term, 'INT' => :term, 'TSTP' => :tstp, 'USR2' => :usr2 }.freeze
23
25
 
24
- def initialize(config, slot, index)
26
+ # `parent_pid` is captured by the swarm before it forks (race-free) and
27
+ # threaded through so OrphanGuard can tell "still supervised" from
28
+ # "reparented after the supervisor died". Defaults to the live parent for
29
+ # the non-swarm callers (tests) that construct a ChildBoot directly.
30
+ # `start_quiet:` — the swarm was TSTP-quieted before this child was
31
+ # forked (respawn/recycle during maintenance); boot the launcher already
32
+ # quieted so the replacement doesn't resume fetching. Delivered as a
33
+ # constructor flag, not a post-fork TSTP, because the signal would race
34
+ # the trap-reset window (default TSTP disposition suspends the child).
35
+ def initialize(config, slot, index, parent_pid: ::Process.ppid, start_quiet: false)
25
36
  @config = config
26
37
  @slot = slot
27
38
  @index = index
28
- @signal_queue = ::Thread::Queue.new
39
+ @parent_pid = parent_pid
40
+ @start_quiet = start_quiet
41
+ @signal_read = nil
42
+ @signal_write = nil
29
43
  end
30
44
 
31
45
  def run
@@ -56,28 +70,71 @@ module Wurk
56
70
 
57
71
  # Boot the launcher and block until shutdown. wait_loop joins the
58
72
  # signal-dispatch thread, so the child can't fall through to `exit 0`
59
- # mid-drain.
73
+ # mid-drain. Orphan protection is armed right AFTER launcher.run — the
74
+ # TERM handler is in place (so pdeathsig / the watchdog drain gracefully)
75
+ # and the managers are up (so a self-TERM can't race launcher.run). A
76
+ # parent that died during boot is still caught immediately: the watchdog's
77
+ # first getppid check sees the reparent and drains at once.
60
78
  def run_launcher
61
79
  launcher = Wurk::Launcher.new(@config)
62
80
  install_signal_handlers(launcher)
63
81
  launcher.run
82
+ launcher.quiet if @start_quiet
83
+ arm_orphan_guard
64
84
  wait_loop(launcher)
65
85
  end
66
86
 
67
- # Parent installed traps for TERM/INT/TSTP/CONT/USR1 — the child
68
- # needs its own behavior, not the parent's.
87
+ def arm_orphan_guard
88
+ @orphan_guard = OrphanGuard.new(@parent_pid, logger: @config.logger)
89
+ @watchdog = @orphan_guard.arm
90
+ end
91
+
92
+ # Parent installed traps for TERM/INT/TSTP — the child needs its own
93
+ # behavior, not the parent's. USR2 too: the child owns log-reopen.
94
+ # USR1 (rolling restart) is a no-op in the child — trap it with a log
95
+ # so stray USR1s don't trigger default termination.
69
96
  def reset_inherited_signals
70
- %w[TERM INT TSTP CONT USR1 USR2].each { |s| ::Signal.trap(s, 'DEFAULT') }
97
+ %w[TERM INT TSTP USR2].each { |s| ::Signal.trap(s, 'DEFAULT') }
98
+ # A genuine no-op in the child (rolling restart is the parent's job) —
99
+ # trap it empty so a stray USR1 can't fall through to default
100
+ # termination. Must NOT log: Logger synchronizes writes and would
101
+ # deadlock if the trap fired mid-write.
102
+ ::Signal.trap('USR1') { nil }
103
+ rescue ArgumentError
104
+ nil
71
105
  end
72
106
 
73
107
  def reconnect_after_fork
74
108
  @config.reset_redis_pools!
109
+ validate_redis!
110
+ reconnect_active_record
111
+ end
112
+
113
+ # Prove the child's fresh Redis socket reaches a live server before it
114
+ # starts fetching: one PING through RedisPool#with, which owns the
115
+ # retry+backoff (production incident #101), so a transient blip during
116
+ # boot rides out instead of racing straight into a dead pool. A PING that
117
+ # still fails past the wrapper's retries propagates and crashes the child
118
+ # (the swarm respawns it) rather than booting a worker that can't reach
119
+ # Redis. The same checkout eagerly primes every Lua script in one
120
+ # pipelined round-trip so the first EVALSHA hits a warm cache.
121
+ def validate_redis!
122
+ @config.redis_pool.with do |conn|
123
+ conn.call('PING')
124
+ Wurk::Lua::Loader.script_load_all(conn)
125
+ end
126
+ end
127
+
128
+ # AR reconnect is best-effort — a Redis-only worker with no database still
129
+ # runs — but the silent `rescue nil` here hid a real misconfiguration
130
+ # during the #101 audit, so warn loudly instead of swallowing.
131
+ def reconnect_active_record
75
132
  return unless defined?(::ActiveRecord::Base)
76
133
 
77
- begin
78
- ::ActiveRecord::Base.establish_connection
79
- rescue StandardError
80
- nil
134
+ ::ActiveRecord::Base.establish_connection
135
+ rescue StandardError => e
136
+ @config.logger.warn do
137
+ "swarm child ##{@index} (#{::Process.pid}) ActiveRecord reconnect failed: #{e.class}: #{e.message}"
81
138
  end
82
139
  end
83
140
 
@@ -90,11 +147,29 @@ module Wurk
90
147
  # below — for every entry point, not just the swarm.
91
148
  end
92
149
 
150
+ # Self-pipe pattern (same as Wurk::CLI / Wurk::Swarm): the trap only
151
+ # writes the signal name to a pipe — never Thread::Queue#push, whose
152
+ # mutex a trap can deadlock against.
93
153
  def install_signal_handlers(launcher)
94
- CHILD_SIGNALS.each { |sig, sym| ::Signal.trap(sig) { @signal_queue << sym } }
154
+ @signal_read, @signal_write = ::IO.pipe
155
+ CHILD_SIGNALS.each_key do |sig|
156
+ ::Signal.trap(sig) { emit_signal(sig) }
157
+ rescue ArgumentError
158
+ nil
159
+ end
95
160
  @dispatcher = Thread.new { dispatch_signals(launcher) }
96
161
  end
97
162
 
163
+ # Non-blocking self-pipe write from trap context: a blocking `puts` could
164
+ # stall the TERM drain if the pipe ever filled. `exception: false` returns
165
+ # :wait_writable instead of raising when full (the queued duplicate
166
+ # coalesces); a closed pipe during shutdown is ignored too.
167
+ def emit_signal(sig)
168
+ @signal_write.write_nonblock("#{sig}\n", exception: false)
169
+ rescue ::IOError, ::Errno::EPIPE, ::Errno::EBADF
170
+ nil
171
+ end
172
+
98
173
  # TSTP/USR2 keep looping; TERM/INT run the full launcher.stop
99
174
  # (which blocks on manager drain) and then return — wait_loop
100
175
  # joins this thread, so the main child thread can't `exit 0`
@@ -102,10 +177,14 @@ module Wurk
102
177
  # and the main thread would race past the unfinished managers.
103
178
  def dispatch_signals(launcher)
104
179
  loop do
105
- case @signal_queue.pop
180
+ @signal_read.wait_readable
181
+ sig = @signal_read.gets&.strip
182
+ break if sig.nil?
183
+
184
+ case CHILD_SIGNALS[sig]
106
185
  when :term
107
186
  launcher.stop
108
- return
187
+ break
109
188
  when :tstp then launcher.quiet
110
189
  when :usr2 then reopen_logs
111
190
  end