wurk 1.3.1 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +4 -1
  3. data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +22 -5
  4. data/lib/wurk/batch/callbacks.rb +82 -12
  5. data/lib/wurk/batch/death_handler.rb +7 -4
  6. data/lib/wurk/batch/server_middleware.rb +1 -1
  7. data/lib/wurk/batch.rb +121 -15
  8. data/lib/wurk/capsule.rb +45 -16
  9. data/lib/wurk/cli.rb +48 -14
  10. data/lib/wurk/client/buffered.rb +200 -43
  11. data/lib/wurk/client.rb +181 -38
  12. data/lib/wurk/compat.rb +1 -1
  13. data/lib/wurk/component.rb +36 -4
  14. data/lib/wurk/configuration.rb +47 -7
  15. data/lib/wurk/context.rb +1 -1
  16. data/lib/wurk/cron.rb +94 -37
  17. data/lib/wurk/deploy.rb +5 -3
  18. data/lib/wurk/embedded.rb +13 -0
  19. data/lib/wurk/engine.rb +41 -1
  20. data/lib/wurk/errors.rb +15 -0
  21. data/lib/wurk/fetcher/reaper.rb +125 -58
  22. data/lib/wurk/fetcher/reliable.rb +366 -51
  23. data/lib/wurk/fetcher.rb +6 -0
  24. data/lib/wurk/heartbeat.rb +24 -12
  25. data/lib/wurk/history.rb +13 -1
  26. data/lib/wurk/job_logger.rb +16 -7
  27. data/lib/wurk/job_set.rb +3 -2
  28. data/lib/wurk/job_util.rb +44 -24
  29. data/lib/wurk/launcher.rb +230 -119
  30. data/lib/wurk/leader.rb +63 -14
  31. data/lib/wurk/limiter/base.rb +8 -10
  32. data/lib/wurk/limiter/bucket.rb +1 -1
  33. data/lib/wurk/limiter/concurrent.rb +27 -22
  34. data/lib/wurk/limiter/window.rb +13 -11
  35. data/lib/wurk/limiter.rb +7 -4
  36. data/lib/wurk/logger.rb +1 -1
  37. data/lib/wurk/lua/loader.rb +9 -3
  38. data/lib/wurk/lua.rb +97 -14
  39. data/lib/wurk/manager.rb +29 -13
  40. data/lib/wurk/metrics/accumulator.rb +95 -0
  41. data/lib/wurk/metrics/flusher.rb +70 -0
  42. data/lib/wurk/metrics/history.rb +102 -32
  43. data/lib/wurk/metrics/queue_rollup.rb +13 -1
  44. data/lib/wurk/metrics/rollup.rb +13 -1
  45. data/lib/wurk/metrics/statsd.rb +32 -18
  46. data/lib/wurk/middleware/chain.rb +31 -14
  47. data/lib/wurk/middleware/interrupt_handler.rb +7 -6
  48. data/lib/wurk/middleware/poison_pill.rb +93 -31
  49. data/lib/wurk/middleware.rb +2 -2
  50. data/lib/wurk/pool_checkout.rb +39 -0
  51. data/lib/wurk/process_set.rb +10 -5
  52. data/lib/wurk/processor.rb +83 -9
  53. data/lib/wurk/profiler.rb +9 -4
  54. data/lib/wurk/queue.rb +18 -7
  55. data/lib/wurk/rails_boot.rb +38 -7
  56. data/lib/wurk/redis_client_adapter.rb +49 -5
  57. data/lib/wurk/redis_pool.rb +71 -25
  58. data/lib/wurk/scheduled.rb +30 -2
  59. data/lib/wurk/shutdown_gate.rb +79 -0
  60. data/lib/wurk/stats.rb +19 -10
  61. data/lib/wurk/swarm/child_boot.rb +36 -4
  62. data/lib/wurk/swarm.rb +258 -43
  63. data/lib/wurk/timer_loop.rb +14 -0
  64. data/lib/wurk/version.rb +1 -1
  65. data/lib/wurk/web/config.rb +11 -7
  66. data/lib/wurk/web/enterprise.rb +58 -6
  67. data/lib/wurk/web/extension.rb +1 -1
  68. data/lib/wurk/web/search.rb +5 -3
  69. data/lib/wurk.rb +12 -10
  70. data/vendor/assets/dashboard/assets/{ArgsValue-D74zX0MI.js → ArgsValue-CcR2ya6e.js} +1 -1
  71. data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-CUXJUQ3Q.js} +1 -1
  72. data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-Cxan6Ngw.js} +1 -1
  73. data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-DC5EGM0g.js} +1 -1
  74. data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-Dlt8tXJA.js} +1 -1
  75. data/vendor/assets/dashboard/assets/Dashboard-DNLu_WCg.js +1 -0
  76. data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-dZ7VGlKS.js} +1 -1
  77. data/vendor/assets/dashboard/assets/Extension-DaFpEIJf.js +1 -0
  78. data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-CO3aYWIq.js} +1 -1
  79. data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-DSWbT6G0.js} +1 -1
  80. data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-Cb4PKXNR.js} +1 -1
  81. data/vendor/assets/dashboard/assets/Metrics-CCGzgCsT.js +1 -0
  82. data/vendor/assets/dashboard/assets/Modal-B86q6ruL.js +1 -0
  83. data/vendor/assets/dashboard/assets/{PageHeader-C44KNMGm.js → PageHeader-fPrCcp_-.js} +1 -1
  84. data/vendor/assets/dashboard/assets/{Profiles-xEVTyS2N.js → Profiles-BnS82nR_.js} +1 -1
  85. data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-CIyPevOy.js} +1 -1
  86. data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-DopwXkXl.js} +1 -1
  87. data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-1-Z7i1zE.js} +1 -1
  88. data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-ByA6eTma.js} +1 -1
  89. data/vendor/assets/dashboard/assets/{Skeleton-DzR7XNxz.js → Skeleton-bC7HfQ9r.js} +1 -1
  90. data/vendor/assets/dashboard/assets/{charts-BVHHGof7.js → charts-CLLzJ7vK.js} +1 -1
  91. data/vendor/assets/dashboard/assets/index-B1N8hQUh.js +141 -0
  92. data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
  93. data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-DpBjkf6_.js} +1 -1
  94. data/vendor/assets/dashboard/assets/{useSort-BeYbztkN.js → useSort-DvpwuNQE.js} +1 -1
  95. data/vendor/assets/dashboard/index.html +3 -3
  96. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  97. metadata +32 -27
  98. data/vendor/assets/dashboard/assets/Dashboard-B9rOrkzk.js +0 -1
  99. data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
  100. data/vendor/assets/dashboard/assets/Metrics-BBTDxcaE.js +0 -1
  101. data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
  102. data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
  103. data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
data/lib/wurk/cli.rb CHANGED
@@ -106,16 +106,24 @@ module Wurk
106
106
  enter_server_mode
107
107
  boot_application if boot_app
108
108
  self_read, self_write = ::IO.pipe
109
- trap_signals(self_write)
110
- validate_redis!
111
- validate_pool_sizes!
112
- @config[:identity] = identity
113
- # Force lazy server-middleware chain so worker threads don't race
114
- # against each other constructing it. Spec: Sidekiq::CLI line 104.
115
- @config.server_middleware
116
- ::Process.warmup if warmup && ::Process.respond_to?(:warmup) && ENV['RUBY_DISABLE_WARMUP'] != '1'
117
- fire_event(:startup, reverse: false, reraise: true)
118
- launch(self_read)
109
+ begin
110
+ trap_signals(self_write)
111
+ validate_redis!
112
+ validate_pool_sizes!
113
+ @config[:identity] = identity
114
+ # Force lazy server-middleware chain so worker threads don't race
115
+ # against each other constructing it. Spec: Sidekiq::CLI line 104.
116
+ @config.server_middleware
117
+ warm_up_process if warmup
118
+ fire_event(:startup, reverse: false, reraise: true)
119
+ launch(self_read)
120
+ ensure
121
+ # Every exit path — clean return, a raise out of validate/startup, and
122
+ # the SystemExit that `launch` unwinds on Interrupt. Embedded and test
123
+ # callers survive `run`, so a leaked pair is a real FD leak.
124
+ self_read.close
125
+ self_write.close
126
+ end
119
127
  end
120
128
 
121
129
  # Standalone multi-process boot — the `sidekiqswarm` entry point (Ent §7).
@@ -141,11 +149,18 @@ module Wurk
141
149
  validate_redis!
142
150
  validate_pool_sizes!
143
151
  @config[:identity] = identity
144
- ::Process.warmup if warmup && ::Process.respond_to?(:warmup) && ENV['RUBY_DISABLE_WARMUP'] != '1'
152
+ warm_up_process if warmup
145
153
  @swarm = Wurk::Swarm.new(topology: @config.topology, config: @config,
146
154
  shutdown_timeout: @config[:timeout] || Swarm::DEFAULT_SHUTDOWN_TIMEOUT)
147
- @swarm.boot(install_signals: true)
148
- @swarm.supervise
155
+ begin
156
+ @swarm.boot(install_signals: true)
157
+ @swarm.supervise
158
+ ensure
159
+ # A fork failure part-way through `boot`, or anything raising out of
160
+ # `supervise`, otherwise leaves live children behind with no supervisor.
161
+ # A no-op once the loop already drained (no children, pipe closed).
162
+ @swarm.shutdown
163
+ end
149
164
  end
150
165
 
151
166
  def handle_signal(sig)
@@ -164,6 +179,15 @@ module Wurk
164
179
  Wurk.enter_server_mode(@config)
165
180
  end
166
181
 
182
+ # Pre-fault and compact the heap before workers start — in `run_swarm` that
183
+ # happens before the fork, so children share the warmed pages copy-on-write.
184
+ # RUBY_DISABLE_WARMUP=1 is the escape hatch for hosts that already warmed up.
185
+ def warm_up_process
186
+ return unless ::Process.respond_to?(:warmup) && ENV['RUBY_DISABLE_WARMUP'] != '1'
187
+
188
+ ::Process.warmup
189
+ end
190
+
167
191
  def launch(self_read)
168
192
  @launcher = Wurk::Launcher.new(@config)
169
193
  begin
@@ -187,13 +211,23 @@ module Wurk
187
211
  # over `self_read` in `launch`. Same approach as Sidekiq.
188
212
  def trap_signals(self_write)
189
213
  signal_names.each do |sig|
190
- ::Signal.trap(sig) { self_write.puts(sig) }
214
+ ::Signal.trap(sig) { emit_signal(self_write, sig) }
191
215
  rescue ArgumentError
192
216
  # JRuby and platforms without certain signals — log and move on.
193
217
  warn("Signal #{sig} not supported")
194
218
  end
195
219
  end
196
220
 
221
+ # Non-blocking write from trap context (same shape as Swarm#emit_signal): a
222
+ # blocking `puts` can stall signal delivery once the pipe fills, and the
223
+ # traps outlive `run`'s ensure — a signal landing after the pipe is closed
224
+ # would otherwise raise IOError out of a trap and into the main thread.
225
+ def emit_signal(pipe, sig)
226
+ pipe.write_nonblock("#{sig}\n", exception: false)
227
+ rescue ::IOError, ::Errno::EPIPE, ::Errno::EBADF
228
+ nil
229
+ end
230
+
197
231
  def signal_names
198
232
  %w[INT TERM TSTP TTIN INFO USR2]
199
233
  end
@@ -21,24 +21,43 @@ module Wurk
21
21
  OVERFLOW_MODES = %i[drop_oldest raise].freeze
22
22
  DEFAULT_OVERFLOW_MODE = :drop_oldest
23
23
 
24
- # Raised by `enbuffer` when the cap would be exceeded under
25
- # `overflow_mode == :raise`. Inherits from RuntimeError so callers
26
- # can rescue narrowly. The payload that triggered the overflow rides
27
- # along so the caller can persist/log/forward it.
24
+ # Raised when the cap would be exceeded under `overflow_mode == :raise`.
25
+ # Inherits from RuntimeError so callers can rescue narrowly. Carries
26
+ # EVERY payload the call failed to deliver — the tail that did not fit,
27
+ # plus, on a mixed push, the batched payloads that never buffer — so a
28
+ # caller can persist/log/forward the lot. `cause` is the connection error
29
+ # that sent the push to the buffer in the first place.
28
30
  class Overflow < RuntimeError
29
- attr_reader :payload
31
+ attr_reader :payloads
30
32
 
31
- def initialize(payload)
32
- @payload = payload
33
- super("reliable_push buffer is full (cap=#{Buffered.buffer_cap})")
33
+ def initialize(payloads)
34
+ @payloads = payloads
35
+ super("reliable_push buffer is full (cap=#{Buffered.buffer_cap}), " \
36
+ "#{payloads.size} payload(s) undelivered")
34
37
  end
35
38
  end
36
39
 
40
+ # Returned by the append helpers when everything fit.
41
+ NOTHING_UNDELIVERED = [].freeze
42
+
37
43
  # Eagerly initialized: `||=` inside an accessor is not atomic — two
38
44
  # threads racing first-touch could end up holding distinct Mutex
39
45
  # instances and lose all synchronization on the shared buffer.
40
- INSTALL_MUTEX = Mutex.new
41
- BUFFER_MUTEX = Mutex.new
46
+ #
47
+ # Module ivars rather than constants so `reset_after_fork!` can replace
48
+ # them outright. MRI abandons a mutex whose owner thread didn't survive
49
+ # the fork (rb_thread_atfork), but that's an implementation detail rather
50
+ # than a documented guarantee, and it does NOT cover a fork taken from
51
+ # inside either critical section — there the child inherits the lock
52
+ # still owned, and its first `Client#push` (which drains, so it
53
+ # synchronizes, before pushing) blocks forever. Two allocations per fork
54
+ # buys immunity from both.
55
+ @install_mutex = Mutex.new
56
+ @buffer_mutex = Mutex.new
57
+
58
+ # Process that owns the state above; a mismatch means we're running in a
59
+ # fork and the inherited copy has to go.
60
+ @owner_pid = ::Process.pid
42
61
 
43
62
  class << self
44
63
  attr_accessor :buffer_client_factory
@@ -100,40 +119,77 @@ module Wurk
100
119
  @overflow_mode = nil
101
120
  @buffer_client_factory = nil
102
121
  end
122
+ # Stop before dropping: an unstopped drainer thread survives with
123
+ # its factory nil'd out from under it and ticks forever against
124
+ # nothing, leaking the thread and everything its closure retains.
125
+ install_mutex.synchronize do
126
+ @drainer&.stop
127
+ @drainer = nil
128
+ end
129
+ end
130
+
131
+ # Fork hook, called from the `Process._fork` prepend below and from
132
+ # `Swarm::ChildBoot#reconnect_after_fork`. Whichever runs first wins
133
+ # and returns true; the pid guard makes the other a no-op returning
134
+ # false, so a caller can tell which one rebuilt the state.
135
+ #
136
+ # A child inherits a copy of every ivar here: the buffered payloads,
137
+ # the Drainer (whose thread did not survive the fork), and both mutexes
138
+ # (see their definition for why replacing them matters).
139
+ #
140
+ # The child DROPS its inherited payloads rather than replaying them:
141
+ # the parent still holds the same buffer and replays it on its own next
142
+ # push, so a child that also drained would enqueue every buffered job
143
+ # once per fork — `(children + 1) x N` duplicates. Only the parent
144
+ # replays.
145
+ #
146
+ # `@drainer` is dropped, never `stop`ped — its `@lock` carries the same
147
+ # inherited-mutex hazard. A parent-configured drainer is replaced by an
148
+ # equivalent fresh one so an opted-in child keeps flushing the buffer it
149
+ # fills itself; the captured client factory goes with it, since it
150
+ # closes over the parent's pre-fork Redis pool.
151
+ #
152
+ # Deliberately unsynchronized: the child has exactly one thread here,
153
+ # and waiting on the very mutex being replaced is what would hang it.
154
+ def reset_after_fork! # rubocop:disable Naming/PredicateMethod
155
+ return false if @owner_pid == ::Process.pid
156
+
157
+ @owner_pid = ::Process.pid
158
+ @install_mutex = Mutex.new
159
+ @buffer_mutex = Mutex.new
160
+ @buffer = []
161
+ @buffer_client_factory = nil
162
+ interval = @drainer&.interval
163
+ @drainer = nil
164
+ start_drainer!(interval: interval) if interval
165
+ true
103
166
  end
104
167
 
105
168
  # Append payloads to the buffer. Behavior on cap exhaustion depends
106
169
  # on `overflow_mode`:
107
170
  # * :drop_oldest (default, spec) — ring buffer, oldest evicted.
108
- # * :raise — Overflow raised, buffer left
109
- # unchanged for already-appended
110
- # siblings in the same call; the
111
- # offending payload is attached
112
- # to the exception.
171
+ # * :raise — fills the remaining capacity, then
172
+ # raises one Overflow carrying every
173
+ # payload that did not fit.
113
174
  # Drops batched payloads — caller is expected to re-raise for those.
114
175
  # If client is provided, captures its pool for drainer to use by default.
115
176
  def enbuffer(payloads, client: nil)
116
177
  capture_pool_from_client(client)
117
178
 
118
- cap = buffer_cap
179
+ cap = buffer_cap
119
180
  mode = overflow_mode
120
- buffer_mutex.synchronize do
121
- payloads.each do |p|
122
- if buffer.size >= cap
123
- raise Overflow, p if mode == :raise
124
-
125
- buffer.shift # :drop_oldest
126
- end
127
- buffer << p
128
- end
181
+ undelivered = buffer_mutex.synchronize do
182
+ mode == :raise ? append_within_capacity(payloads, cap) : append_dropping_oldest(payloads, cap)
129
183
  end
184
+
185
+ raise Overflow, undelivered unless undelivered.empty?
130
186
  end
131
187
 
132
188
  private
133
189
 
134
- # Capture the pool from the provided client and set it as the default
135
- # factory for the drainer. Ensures buffered jobs are drained to the
136
- # same pool they were pushed to, unless explicitly overridden.
190
+ # Remember how the buffering client reaches Redis, so the drainer
191
+ # replays into the same server it was pushed to unless explicitly
192
+ # overridden.
137
193
  def capture_pool_from_client(client)
138
194
  return unless client && !buffer_client_factory
139
195
 
@@ -141,10 +197,56 @@ module Wurk
141
197
  # A nil capture must not install a factory: it would pin the drainer
142
198
  # to the DEFAULT pool forever (the `!buffer_client_factory` guard
143
199
  # blocks any later, correct capture) — wrong Redis for jobs pushed
144
- # through an explicit-pool client.
200
+ # through an explicit-pool client. A pool-less client already resolves
201
+ # its config at push time, and so does the fallback factory.
145
202
  return unless pool
146
203
 
147
- self.buffer_client_factory = -> { Wurk::Client.new(pool: pool) }
204
+ resolver = pool_resolver(client, pool)
205
+ self.buffer_client_factory = -> { Wurk::Client.new(pool: resolver.call) }
206
+ end
207
+
208
+ # What must NOT be captured is the pool object. `reset_redis_pools!` —
209
+ # every fork, every embedded teardown — disconnects a pool and drops it
210
+ # for a lazily rebuilt one, and ConnectionPool#shutdown is terminal, so
211
+ # a pinned instance leaves the drainer replaying into dead sockets for
212
+ # the rest of the process's life. The config (a Configuration or a
213
+ # Capsule) is what survives that rebuild, so ask it again at drain time
214
+ # whenever the client's pool is the one it hands out. A pool the config
215
+ # does not own is a second Redis nothing else can produce — that one
216
+ # stays pinned, stale or not, because replaying it anywhere else writes
217
+ # to the wrong server.
218
+ def pool_resolver(client, pool)
219
+ config = client.instance_variable_get(:@config)
220
+ config_owns = config.respond_to?(:redis_pool) && config.redis_pool.equal?(pool)
221
+ config_owns ? -> { config.redis_pool } : -> { pool }
222
+ end
223
+
224
+ # Both append helpers run with buffer_mutex held and return the payloads
225
+ # they could not take.
226
+
227
+ def append_dropping_oldest(payloads, cap)
228
+ payloads.each do |p|
229
+ buffer.shift if buffer.size >= cap
230
+ buffer << p
231
+ end
232
+ NOTHING_UNDELIVERED
233
+ end
234
+
235
+ # The split has to be decided before any mutation: raising from inside
236
+ # the append loop leaves every payload after the rejected one neither
237
+ # buffered, nor enqueued, nor attached to the exception. `room` goes
238
+ # negative when the cap was lowered after the buffer filled — clamped,
239
+ # so an over-full buffer rejects the whole call instead of raising on
240
+ # `first`/`drop`.
241
+ def append_within_capacity(payloads, cap)
242
+ room = (cap - buffer.size).clamp(0, payloads.size)
243
+ if room == payloads.size
244
+ buffer.concat(payloads)
245
+ return NOTHING_UNDELIVERED
246
+ end
247
+
248
+ buffer.concat(payloads.first(room))
249
+ payloads.drop(room)
148
250
  end
149
251
 
150
252
  public
@@ -153,7 +255,10 @@ module Wurk
153
255
  # the first transient failure (ConnectionError past the pool's own
154
256
  # retries, or a starved checkout), preserving order at the head of
155
257
  # the buffer so the next push retries the same payload. Emits statsd
156
- # `jobs.recovered.push` per drained payload.
258
+ # `jobs.recovered.push` per drained payload, plus the `jobs.enqueued`
259
+ # the buffering push deliberately did not emit — the replay is where
260
+ # the job actually reaches Redis, so a buffered-then-drained job counts
261
+ # once as enqueued and once as recovered.
157
262
  def drain!(client)
158
263
  drained = 0
159
264
  while (payload = pop_head)
@@ -173,6 +278,7 @@ module Wurk
173
278
  break
174
279
  end
175
280
 
281
+ client.send(:emit_enqueued, [payload])
176
282
  Wurk::Metrics::Statsd.increment('jobs.recovered.push')
177
283
  drained += 1
178
284
  end
@@ -191,7 +297,7 @@ module Wurk
191
297
  # handles the case where push activity stops mid-outage so the
192
298
  # passive (drain-on-next-push) path never fires.
193
299
  def start_drainer!(interval: Drainer::DEFAULT_INTERVAL, client_factory: nil)
194
- INSTALL_MUTEX.synchronize do
300
+ install_mutex.synchronize do
195
301
  @drainer&.stop
196
302
  factory = client_factory || buffer_client_factory || -> { Wurk::Client.new }
197
303
  @drainer = Drainer.new(interval: interval, client_factory: factory)
@@ -200,25 +306,19 @@ module Wurk
200
306
  end
201
307
 
202
308
  def stop_drainer!
203
- INSTALL_MUTEX.synchronize do
309
+ install_mutex.synchronize do
204
310
  @drainer&.stop
205
311
  @drainer = nil
206
312
  end
207
313
  end
208
314
 
209
315
  def drainer_running?
210
- INSTALL_MUTEX.synchronize { @drainer&.running? == true }
316
+ install_mutex.synchronize { @drainer&.running? == true }
211
317
  end
212
318
 
213
319
  private
214
320
 
215
- def install_mutex
216
- INSTALL_MUTEX
217
- end
218
-
219
- def buffer_mutex
220
- BUFFER_MUTEX
221
- end
321
+ attr_reader :install_mutex, :buffer_mutex
222
322
 
223
323
  def pop_head
224
324
  buffer_mutex.synchronize { buffer.shift }
@@ -248,6 +348,10 @@ module Wurk
248
348
  DEFAULT_INTERVAL = 2.0
249
349
  STOP_JOIN_TIMEOUT = 5.0
250
350
 
351
+ # Read by `Buffered.reset_after_fork!` off the *inherited* drainer, to
352
+ # rebuild an equivalent one in the child without touching its lock.
353
+ attr_reader :interval
354
+
251
355
  def initialize(interval: DEFAULT_INTERVAL, client_factory: -> { Wurk::Client.new })
252
356
  unless interval.is_a?(Numeric) && interval.positive?
253
357
  raise ArgumentError, 'interval must be a positive Numeric'
@@ -326,16 +430,69 @@ module Wurk
326
430
 
327
431
  private
328
432
 
433
+ # Opens the delivery ledger Client writes into. It lives here rather
434
+ # than in Client because this rescue is its only reader: a Client
435
+ # without reliable_push! never allocates it.
329
436
  def raw_push(payloads)
437
+ Thread.current[DELIVERED_KEY] = []
330
438
  super
331
439
  rescue RedisClient::ConnectionError, ConnectionPool::TimeoutError
332
440
  raise if Thread.current[Buffered::DRAINING_KEY]
333
441
 
334
- bidless, batched = payloads.partition { |p| !p['bid'] }
335
- Buffered.enbuffer(bidless, client: self) if bidless.any?
442
+ bidless, batched = undelivered(payloads).partition { |p| !p['bid'] }
443
+ enbuffer_bidless(bidless, batched)
336
444
  raise unless batched.empty?
445
+
446
+ # Client#push subtracts these from the enqueued metric: they are in
447
+ # the buffer, not in Redis. Whatever the group did deliver stays out
448
+ # of the set and still counts.
449
+ bidless
450
+ ensure
451
+ Thread.current[DELIVERED_KEY] = nil
452
+ end
453
+
454
+ # The payloads this push is not known to have written. A push spanning
455
+ # several queues, or one mixing plain and batched jobs, fails after
456
+ # some of its groups already landed; buffering those would replay them
457
+ # into duplicate jobs once the outage clears. The ledger holds the very
458
+ # Hash objects Client just handed to Redis, hence the identity subtract.
459
+ def undelivered(payloads)
460
+ delivered = Thread.current[DELIVERED_KEY]
461
+ return payloads if delivered.empty?
462
+
463
+ reject_by_identity(payloads, delivered)
464
+ end
465
+
466
+ # Bare `raise` re-raises whatever `$!` holds: the connection error from
467
+ # the caller's rescue, or the Overflow once we're inside this one.
468
+ def enbuffer_bidless(bidless, batched)
469
+ Buffered.enbuffer(bidless, client: self) if bidless.any?
470
+ rescue Buffered::Overflow => e
471
+ # An overflow pre-empts the caller's connection-error re-raise, which
472
+ # would strand the batched payloads silently — they never buffer. One
473
+ # exception, every payload that failed to get through; `cause` stays
474
+ # the connection error rather than the folded-in Overflow.
475
+ raise if batched.empty?
476
+
477
+ raise Buffered::Overflow, e.payloads + batched, cause: e.cause
478
+ end
479
+ end
480
+
481
+ # Ruby >= 3.1 routes every `fork` / `Process.fork` through
482
+ # `Process._fork`, which is the only way to catch the forks Wurk never
483
+ # sees: a Puma or Unicorn parent that preloaded the app — and may already
484
+ # be holding buffered payloads — spawning its workers. Registered at
485
+ # require time because `reliable_push!` can be installed after the fork
486
+ # that copied the state. Guarded on the fork-less runtimes (JRuby), where
487
+ # there is no `super` to call.
488
+ module ForkHook
489
+ def _fork
490
+ pid = super
491
+ Buffered.reset_after_fork! if pid.zero?
492
+ pid
337
493
  end
338
494
  end
495
+ ::Process.singleton_class.prepend(ForkHook) if ::Process.respond_to?(:_fork)
339
496
  end
340
497
 
341
498
  class << self