wurk 1.3.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +1 -0
  3. data/lib/wurk/batch/callbacks.rb +82 -12
  4. data/lib/wurk/batch/death_handler.rb +7 -4
  5. data/lib/wurk/batch/server_middleware.rb +1 -1
  6. data/lib/wurk/batch.rb +121 -15
  7. data/lib/wurk/capsule.rb +5 -4
  8. data/lib/wurk/cli.rb +48 -14
  9. data/lib/wurk/client/buffered.rb +193 -43
  10. data/lib/wurk/client.rb +87 -14
  11. data/lib/wurk/compat.rb +1 -1
  12. data/lib/wurk/component.rb +2 -2
  13. data/lib/wurk/configuration.rb +10 -2
  14. data/lib/wurk/cron.rb +94 -37
  15. data/lib/wurk/deploy.rb +5 -3
  16. data/lib/wurk/embedded.rb +13 -0
  17. data/lib/wurk/fetcher/reaper.rb +113 -56
  18. data/lib/wurk/fetcher/reliable.rb +62 -9
  19. data/lib/wurk/heartbeat.rb +22 -10
  20. data/lib/wurk/history.rb +13 -1
  21. data/lib/wurk/launcher.rb +133 -66
  22. data/lib/wurk/leader.rb +29 -10
  23. data/lib/wurk/limiter/base.rb +8 -10
  24. data/lib/wurk/limiter/bucket.rb +1 -1
  25. data/lib/wurk/limiter/concurrent.rb +27 -22
  26. data/lib/wurk/limiter/window.rb +13 -11
  27. data/lib/wurk/limiter.rb +7 -4
  28. data/lib/wurk/lua.rb +97 -14
  29. data/lib/wurk/manager.rb +29 -13
  30. data/lib/wurk/metrics/history.rb +4 -3
  31. data/lib/wurk/metrics/queue_rollup.rb +13 -1
  32. data/lib/wurk/metrics/rollup.rb +13 -1
  33. data/lib/wurk/middleware/interrupt_handler.rb +7 -6
  34. data/lib/wurk/middleware/poison_pill.rb +70 -29
  35. data/lib/wurk/middleware.rb +2 -2
  36. data/lib/wurk/pool_checkout.rb +29 -0
  37. data/lib/wurk/process_set.rb +10 -5
  38. data/lib/wurk/processor.rb +6 -0
  39. data/lib/wurk/profiler.rb +3 -2
  40. data/lib/wurk/queue.rb +10 -7
  41. data/lib/wurk/rails_boot.rb +38 -7
  42. data/lib/wurk/redis_client_adapter.rb +48 -4
  43. data/lib/wurk/redis_options.rb +142 -0
  44. data/lib/wurk/redis_pool.rb +102 -39
  45. data/lib/wurk/scheduled.rb +30 -2
  46. data/lib/wurk/stats.rb +14 -9
  47. data/lib/wurk/swarm/child_boot.rb +12 -0
  48. data/lib/wurk/swarm.rb +174 -33
  49. data/lib/wurk/timer_loop.rb +14 -0
  50. data/lib/wurk/version.rb +1 -1
  51. data/lib/wurk/web/enterprise.rb +58 -6
  52. data/lib/wurk/web/extension.rb +1 -1
  53. data/lib/wurk/web/search.rb +5 -3
  54. data/lib/wurk.rb +53 -2
  55. data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-OmC5NPgw.js} +1 -1
  56. data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-CIpai7St.js} +1 -1
  57. data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-A_kwSR6Q.js} +1 -1
  58. data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-BG7HTqlp.js} +1 -1
  59. data/vendor/assets/dashboard/assets/{Dashboard-B9rOrkzk.js → Dashboard-A_ToqHoo.js} +1 -1
  60. data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-8J21jMyK.js} +1 -1
  61. data/vendor/assets/dashboard/assets/Extension-B4Q9FIQu.js +1 -0
  62. data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-Fh_Ae7UW.js} +1 -1
  63. data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-Ceng0PMB.js} +1 -1
  64. data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-CruDWvNZ.js} +1 -1
  65. data/vendor/assets/dashboard/assets/{Metrics-BBTDxcaE.js → Metrics-CIT7VCoN.js} +1 -1
  66. data/vendor/assets/dashboard/assets/Modal-CN3rdKA_.js +1 -0
  67. data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-D86FYohJ.js} +1 -1
  68. data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-Bz1O1D-i.js} +1 -1
  69. data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-B6h2akTu.js} +1 -1
  70. data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-OOu22e5s.js} +1 -1
  71. data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
  72. data/vendor/assets/dashboard/assets/index-D_lSDwKw.js +141 -0
  73. data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-dVPGEWzn.js} +1 -1
  74. data/vendor/assets/dashboard/index.html +2 -2
  75. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  76. metadata +22 -20
  77. data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
  78. data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
  79. data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
  80. data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
@@ -21,24 +21,43 @@ module Wurk
21
21
  OVERFLOW_MODES = %i[drop_oldest raise].freeze
22
22
  DEFAULT_OVERFLOW_MODE = :drop_oldest
23
23
 
24
- # Raised by `enbuffer` when the cap would be exceeded under
25
- # `overflow_mode == :raise`. Inherits from RuntimeError so callers
26
- # can rescue narrowly. The payload that triggered the overflow rides
27
- # along so the caller can persist/log/forward it.
24
+ # Raised when the cap would be exceeded under `overflow_mode == :raise`.
25
+ # Inherits from RuntimeError so callers can rescue narrowly. Carries
26
+ # EVERY payload the call failed to deliver — the tail that did not fit,
27
+ # plus, on a mixed push, the batched payloads that never buffer — so a
28
+ # caller can persist/log/forward the lot. `cause` is the connection error
29
+ # that sent the push to the buffer in the first place.
28
30
  class Overflow < RuntimeError
29
- attr_reader :payload
31
+ attr_reader :payloads
30
32
 
31
- def initialize(payload)
32
- @payload = payload
33
- super("reliable_push buffer is full (cap=#{Buffered.buffer_cap})")
33
+ def initialize(payloads)
34
+ @payloads = payloads
35
+ super("reliable_push buffer is full (cap=#{Buffered.buffer_cap}), " \
36
+ "#{payloads.size} payload(s) undelivered")
34
37
  end
35
38
  end
36
39
 
40
+ # Returned by the append helpers when everything fit.
41
+ NOTHING_UNDELIVERED = [].freeze
42
+
37
43
  # Eagerly initialized: `||=` inside an accessor is not atomic — two
38
44
  # threads racing first-touch could end up holding distinct Mutex
39
45
  # instances and lose all synchronization on the shared buffer.
40
- INSTALL_MUTEX = Mutex.new
41
- BUFFER_MUTEX = Mutex.new
46
+ #
47
+ # Module ivars rather than constants so `reset_after_fork!` can replace
48
+ # them outright. MRI abandons a mutex whose owner thread didn't survive
49
+ # the fork (rb_thread_atfork), but that's an implementation detail rather
50
+ # than a documented guarantee, and it does NOT cover a fork taken from
51
+ # inside either critical section — there the child inherits the lock
52
+ # still owned, and its first `Client#push` (which drains, so it
53
+ # synchronizes, before pushing) blocks forever. Two allocations per fork
54
+ # buys immunity from both.
55
+ @install_mutex = Mutex.new
56
+ @buffer_mutex = Mutex.new
57
+
58
+ # Process that owns the state above; a mismatch means we're running in a
59
+ # fork and the inherited copy has to go.
60
+ @owner_pid = ::Process.pid
42
61
 
43
62
  class << self
44
63
  attr_accessor :buffer_client_factory
@@ -102,38 +121,68 @@ module Wurk
102
121
  end
103
122
  end
104
123
 
124
+ # Fork hook, called from the `Process._fork` prepend below and from
125
+ # `Swarm::ChildBoot#reconnect_after_fork`. Whichever runs first wins
126
+ # and returns true; the pid guard makes the other a no-op returning
127
+ # false, so a caller can tell which one rebuilt the state.
128
+ #
129
+ # A child inherits a copy of every ivar here: the buffered payloads,
130
+ # the Drainer (whose thread did not survive the fork), and both mutexes
131
+ # (see their definition for why replacing them matters).
132
+ #
133
+ # The child DROPS its inherited payloads rather than replaying them:
134
+ # the parent still holds the same buffer and replays it on its own next
135
+ # push, so a child that also drained would enqueue every buffered job
136
+ # once per fork — `(children + 1) x N` duplicates. Only the parent
137
+ # replays.
138
+ #
139
+ # `@drainer` is dropped, never `stop`ped — its `@lock` carries the same
140
+ # inherited-mutex hazard. A parent-configured drainer is replaced by an
141
+ # equivalent fresh one so an opted-in child keeps flushing the buffer it
142
+ # fills itself; the captured client factory goes with it, since it
143
+ # closes over the parent's pre-fork Redis pool.
144
+ #
145
+ # Deliberately unsynchronized: the child has exactly one thread here,
146
+ # and waiting on the very mutex being replaced is what would hang it.
147
+ def reset_after_fork! # rubocop:disable Naming/PredicateMethod
148
+ return false if @owner_pid == ::Process.pid
149
+
150
+ @owner_pid = ::Process.pid
151
+ @install_mutex = Mutex.new
152
+ @buffer_mutex = Mutex.new
153
+ @buffer = []
154
+ @buffer_client_factory = nil
155
+ interval = @drainer&.interval
156
+ @drainer = nil
157
+ start_drainer!(interval: interval) if interval
158
+ true
159
+ end
160
+
105
161
  # Append payloads to the buffer. Behavior on cap exhaustion depends
106
162
  # on `overflow_mode`:
107
163
  # * :drop_oldest (default, spec) — ring buffer, oldest evicted.
108
- # * :raise — Overflow raised, buffer left
109
- # unchanged for already-appended
110
- # siblings in the same call; the
111
- # offending payload is attached
112
- # to the exception.
164
+ # * :raise — fills the remaining capacity, then
165
+ # raises one Overflow carrying every
166
+ # payload that did not fit.
113
167
  # Drops batched payloads — caller is expected to re-raise for those.
114
168
  # If client is provided, captures its pool for drainer to use by default.
115
169
  def enbuffer(payloads, client: nil)
116
170
  capture_pool_from_client(client)
117
171
 
118
- cap = buffer_cap
172
+ cap = buffer_cap
119
173
  mode = overflow_mode
120
- buffer_mutex.synchronize do
121
- payloads.each do |p|
122
- if buffer.size >= cap
123
- raise Overflow, p if mode == :raise
124
-
125
- buffer.shift # :drop_oldest
126
- end
127
- buffer << p
128
- end
174
+ undelivered = buffer_mutex.synchronize do
175
+ mode == :raise ? append_within_capacity(payloads, cap) : append_dropping_oldest(payloads, cap)
129
176
  end
177
+
178
+ raise Overflow, undelivered unless undelivered.empty?
130
179
  end
131
180
 
132
181
  private
133
182
 
134
- # Capture the pool from the provided client and set it as the default
135
- # factory for the drainer. Ensures buffered jobs are drained to the
136
- # same pool they were pushed to, unless explicitly overridden.
183
+ # Remember how the buffering client reaches Redis, so the drainer
184
+ # replays into the same server it was pushed to unless explicitly
185
+ # overridden.
137
186
  def capture_pool_from_client(client)
138
187
  return unless client && !buffer_client_factory
139
188
 
@@ -141,10 +190,56 @@ module Wurk
141
190
  # A nil capture must not install a factory: it would pin the drainer
142
191
  # to the DEFAULT pool forever (the `!buffer_client_factory` guard
143
192
  # blocks any later, correct capture) — wrong Redis for jobs pushed
144
- # through an explicit-pool client.
193
+ # through an explicit-pool client. A pool-less client already resolves
194
+ # its config at push time, and so does the fallback factory.
145
195
  return unless pool
146
196
 
147
- self.buffer_client_factory = -> { Wurk::Client.new(pool: pool) }
197
+ resolver = pool_resolver(client, pool)
198
+ self.buffer_client_factory = -> { Wurk::Client.new(pool: resolver.call) }
199
+ end
200
+
201
+ # What must NOT be captured is the pool object. `reset_redis_pools!` —
202
+ # every fork, every embedded teardown — disconnects a pool and drops it
203
+ # for a lazily rebuilt one, and ConnectionPool#shutdown is terminal, so
204
+ # a pinned instance leaves the drainer replaying into dead sockets for
205
+ # the rest of the process's life. The config (a Configuration or a
206
+ # Capsule) is what survives that rebuild, so ask it again at drain time
207
+ # whenever the client's pool is the one it hands out. A pool the config
208
+ # does not own is a second Redis nothing else can produce — that one
209
+ # stays pinned, stale or not, because replaying it anywhere else writes
210
+ # to the wrong server.
211
+ def pool_resolver(client, pool)
212
+ config = client.instance_variable_get(:@config)
213
+ config_owns = config.respond_to?(:redis_pool) && config.redis_pool.equal?(pool)
214
+ config_owns ? -> { config.redis_pool } : -> { pool }
215
+ end
216
+
217
+ # Both append helpers run with buffer_mutex held and return the payloads
218
+ # they could not take.
219
+
220
+ def append_dropping_oldest(payloads, cap)
221
+ payloads.each do |p|
222
+ buffer.shift if buffer.size >= cap
223
+ buffer << p
224
+ end
225
+ NOTHING_UNDELIVERED
226
+ end
227
+
228
+ # The split has to be decided before any mutation: raising from inside
229
+ # the append loop leaves every payload after the rejected one neither
230
+ # buffered, nor enqueued, nor attached to the exception. `room` goes
231
+ # negative when the cap was lowered after the buffer filled — clamped,
232
+ # so an over-full buffer rejects the whole call instead of raising on
233
+ # `first`/`drop`.
234
+ def append_within_capacity(payloads, cap)
235
+ room = (cap - buffer.size).clamp(0, payloads.size)
236
+ if room == payloads.size
237
+ buffer.concat(payloads)
238
+ return NOTHING_UNDELIVERED
239
+ end
240
+
241
+ buffer.concat(payloads.first(room))
242
+ payloads.drop(room)
148
243
  end
149
244
 
150
245
  public
@@ -153,7 +248,10 @@ module Wurk
153
248
  # the first transient failure (ConnectionError past the pool's own
154
249
  # retries, or a starved checkout), preserving order at the head of
155
250
  # the buffer so the next push retries the same payload. Emits statsd
156
- # `jobs.recovered.push` per drained payload.
251
+ # `jobs.recovered.push` per drained payload, plus the `jobs.enqueued`
252
+ # the buffering push deliberately did not emit — the replay is where
253
+ # the job actually reaches Redis, so a buffered-then-drained job counts
254
+ # once as enqueued and once as recovered.
157
255
  def drain!(client)
158
256
  drained = 0
159
257
  while (payload = pop_head)
@@ -173,6 +271,7 @@ module Wurk
173
271
  break
174
272
  end
175
273
 
274
+ client.send(:emit_enqueued, [payload])
176
275
  Wurk::Metrics::Statsd.increment('jobs.recovered.push')
177
276
  drained += 1
178
277
  end
@@ -191,7 +290,7 @@ module Wurk
191
290
  # handles the case where push activity stops mid-outage so the
192
291
  # passive (drain-on-next-push) path never fires.
193
292
  def start_drainer!(interval: Drainer::DEFAULT_INTERVAL, client_factory: nil)
194
- INSTALL_MUTEX.synchronize do
293
+ install_mutex.synchronize do
195
294
  @drainer&.stop
196
295
  factory = client_factory || buffer_client_factory || -> { Wurk::Client.new }
197
296
  @drainer = Drainer.new(interval: interval, client_factory: factory)
@@ -200,25 +299,19 @@ module Wurk
200
299
  end
201
300
 
202
301
  def stop_drainer!
203
- INSTALL_MUTEX.synchronize do
302
+ install_mutex.synchronize do
204
303
  @drainer&.stop
205
304
  @drainer = nil
206
305
  end
207
306
  end
208
307
 
209
308
  def drainer_running?
210
- INSTALL_MUTEX.synchronize { @drainer&.running? == true }
309
+ install_mutex.synchronize { @drainer&.running? == true }
211
310
  end
212
311
 
213
312
  private
214
313
 
215
- def install_mutex
216
- INSTALL_MUTEX
217
- end
218
-
219
- def buffer_mutex
220
- BUFFER_MUTEX
221
- end
314
+ attr_reader :install_mutex, :buffer_mutex
222
315
 
223
316
  def pop_head
224
317
  buffer_mutex.synchronize { buffer.shift }
@@ -248,6 +341,10 @@ module Wurk
248
341
  DEFAULT_INTERVAL = 2.0
249
342
  STOP_JOIN_TIMEOUT = 5.0
250
343
 
344
+ # Read by `Buffered.reset_after_fork!` off the *inherited* drainer, to
345
+ # rebuild an equivalent one in the child without touching its lock.
346
+ attr_reader :interval
347
+
251
348
  def initialize(interval: DEFAULT_INTERVAL, client_factory: -> { Wurk::Client.new })
252
349
  unless interval.is_a?(Numeric) && interval.positive?
253
350
  raise ArgumentError, 'interval must be a positive Numeric'
@@ -326,16 +423,69 @@ module Wurk
326
423
 
327
424
  private
328
425
 
426
+ # Opens the delivery ledger Client writes into. It lives here rather
427
+ # than in Client because this rescue is its only reader: a Client
428
+ # without reliable_push! never allocates it.
329
429
  def raw_push(payloads)
430
+ Thread.current[DELIVERED_KEY] = []
330
431
  super
331
432
  rescue RedisClient::ConnectionError, ConnectionPool::TimeoutError
332
433
  raise if Thread.current[Buffered::DRAINING_KEY]
333
434
 
334
- bidless, batched = payloads.partition { |p| !p['bid'] }
335
- Buffered.enbuffer(bidless, client: self) if bidless.any?
435
+ bidless, batched = undelivered(payloads).partition { |p| !p['bid'] }
436
+ enbuffer_bidless(bidless, batched)
336
437
  raise unless batched.empty?
438
+
439
+ # Client#push subtracts these from the enqueued metric: they are in
440
+ # the buffer, not in Redis. Whatever the group did deliver stays out
441
+ # of the set and still counts.
442
+ bidless
443
+ ensure
444
+ Thread.current[DELIVERED_KEY] = nil
445
+ end
446
+
447
+ # The payloads this push is not known to have written. A push spanning
448
+ # several queues, or one mixing plain and batched jobs, fails after
449
+ # some of its groups already landed; buffering those would replay them
450
+ # into duplicate jobs once the outage clears. The ledger holds the very
451
+ # Hash objects Client just handed to Redis, hence the identity subtract.
452
+ def undelivered(payloads)
453
+ delivered = Thread.current[DELIVERED_KEY]
454
+ return payloads if delivered.empty?
455
+
456
+ reject_by_identity(payloads, delivered)
457
+ end
458
+
459
+ # Bare `raise` re-raises whatever `$!` holds: the connection error from
460
+ # the caller's rescue, or the Overflow once we're inside this one.
461
+ def enbuffer_bidless(bidless, batched)
462
+ Buffered.enbuffer(bidless, client: self) if bidless.any?
463
+ rescue Buffered::Overflow => e
464
+ # An overflow pre-empts the caller's connection-error re-raise, which
465
+ # would strand the batched payloads silently — they never buffer. One
466
+ # exception, every payload that failed to get through; `cause` stays
467
+ # the connection error rather than the folded-in Overflow.
468
+ raise if batched.empty?
469
+
470
+ raise Buffered::Overflow, e.payloads + batched, cause: e.cause
471
+ end
472
+ end
473
+
474
+ # Ruby >= 3.1 routes every `fork` / `Process.fork` through
475
+ # `Process._fork`, which is the only way to catch the forks Wurk never
476
+ # sees: a Puma or Unicorn parent that preloaded the app — and may already
477
+ # be holding buffered payloads — spawning its workers. Registered at
478
+ # require time because `reliable_push!` can be installed after the fork
479
+ # that copied the state. Guarded on the fork-less runtimes (JRuby), where
480
+ # there is no `super` to call.
481
+ module ForkHook
482
+ def _fork
483
+ pid = super
484
+ Buffered.reset_after_fork! if pid.zero?
485
+ pid
337
486
  end
338
487
  end
488
+ ::Process.singleton_class.prepend(ForkHook) if ::Process.respond_to?(:_fork)
339
489
  end
340
490
 
341
491
  class << self
data/lib/wurk/client.rb CHANGED
@@ -26,6 +26,15 @@ module Wurk
26
26
  SCHEDULED_BATCH_SIZE = 100
27
27
  SPREAD_INTERVAL_FLOOR = 5
28
28
 
29
+ # Thread-local slot holding the payloads of the current push whose Redis
30
+ # write is confirmed applied. {Client::Buffered} subtracts them from the set
31
+ # it re-buffers when a *later* phase of the same push loses the connection,
32
+ # so an already-written job is never replayed into a second copy.
33
+ # Thread-local because one Client instance serves every producer thread;
34
+ # opened and closed by Buffered, the only reader, so an un-prepended Client
35
+ # pays a single nil check per write phase.
36
+ DELIVERED_KEY = :wurk_client_delivered
37
+
29
38
  attr_accessor :redis_pool
30
39
 
31
40
  def initialize(pool: nil, config: nil, chain: nil)
@@ -51,8 +60,8 @@ module Wurk
51
60
  return nil unless payload
52
61
 
53
62
  verify_json(payload)
54
- raw_push([payload])
55
- emit_enqueued([payload])
63
+ buffered = raw_push([payload])
64
+ emit_enqueued([payload], buffered)
56
65
  payload['jid']
57
66
  end
58
67
 
@@ -169,8 +178,8 @@ module Wurk
169
178
  payloads = build_bulk_payloads(slice, base, ats)
170
179
  compacted = payloads.compact
171
180
  if compacted.any?
172
- raw_push(compacted)
173
- emit_enqueued(compacted)
181
+ buffered = raw_push(compacted)
182
+ emit_enqueued(compacted, buffered)
174
183
  end
175
184
  jids.concat(payloads.map { |p| p && p['jid'] })
176
185
  end
@@ -222,16 +231,32 @@ module Wurk
222
231
  # Adds happen one payload at a time so an `autoflush = N` actually bounds
223
232
  # the pipeline size — a bulk push of 100 with N=2 must flush 2/2/... not
224
233
  # 100 in one shot.
234
+ #
235
+ # Returns the payloads it did NOT get to Redis: always nil here, since a
236
+ # plain Client either writes them all or raises. {Client::Buffered}
237
+ # overrides the contract — the payloads it diverted into the outage buffer
238
+ # come back so #push can keep them out of the enqueued metric.
225
239
  def raw_push(payloads)
226
240
  # Test modes short-circuit the Redis write (and the batch buffer): :fake
227
241
  # collects payloads in-memory, :inline runs them now. Client middleware
228
242
  # has already run by this point, matching Sidekiq.
229
- return ::Wurk::Testing.dispatch_push(payloads) if ::Wurk::Testing.enabled?
243
+ if ::Wurk::Testing.enabled?
244
+ ::Wurk::Testing.dispatch_push(payloads)
245
+ return nil
246
+ end
230
247
 
231
248
  buffer = Thread.current[Wurk::Batch::BUFFER_KEY]
232
249
  return buffer_add(buffer, payloads) if buffer && payloads.all? { |p| p['bid'] && !p['at'] }
233
250
 
251
+ # No apply-safety claim: every command below appends (LPUSH, ZADD, the
252
+ # batch Lua's counters), so a block replayed after a lost reply is a
253
+ # second copy of the job. A post-write timeout raises out of here instead
254
+ # — {Client::Buffered} turns that into an outage-buffer entry, and a plain
255
+ # Client hands it to whoever called `perform_async`. The pool's pre-apply
256
+ # retry only fires while this block has landed nothing, so a queue group
257
+ # that already went out is never re-pushed by a replay.
234
258
  pool.with { |conn| atomic_push(conn, payloads) }
259
+ nil
235
260
  end
236
261
 
237
262
  # Batch autoflush path: accumulate each non-scheduled batched payload into
@@ -262,7 +287,10 @@ module Wurk
262
287
  # duplicate the scheduled entry.
263
288
  def push_scheduled_split(conn, payloads)
264
289
  batched, plain = payloads.partition { |j| j['bid'] }
265
- conn.pipelined { |pipe| push_scheduled(pipe, plain) } unless plain.empty?
290
+ unless plain.empty?
291
+ conn.pipelined { |pipe| push_scheduled(pipe, plain) }
292
+ mark_delivered(plain)
293
+ end
266
294
  push_batched_scheduled_pipelined(conn, batched) unless batched.empty?
267
295
  end
268
296
 
@@ -282,7 +310,7 @@ module Wurk
282
310
  def push_immediate(conn, payloads)
283
311
  now = now_in_millis
284
312
  batched, plain = payloads.partition { |j| j['bid'] }
285
- conn.pipelined { |pipe| push_plain(pipe, plain, now) } unless plain.empty?
313
+ push_plain(conn, plain, now) unless plain.empty?
286
314
  push_batched_pipelined(conn, batched, now) unless batched.empty?
287
315
  end
288
316
 
@@ -313,15 +341,31 @@ module Wurk
313
341
  conn.pipelined { |pipe| push_batched_scheduled(pipe, batched, eval_method: :eval_with_source) }
314
342
  end
315
343
 
344
+ # One pipeline per queue, marked delivered the moment its reply is in.
345
+ # A push can die between groups — or land its whole plain phase and then
346
+ # lose the connection in the batched phase above — and whatever Redis
347
+ # already accepted must stay out of the reliable_push buffer; replaying it
348
+ # would enqueue a second copy of a job that ran fine.
349
+ #
350
+ # The group in flight when the socket drops stays unmarked and so is
351
+ # replayed: a lost reply is indistinguishable from a lost command, so that
352
+ # residual is at-least-once by construction. The split bounds it to one
353
+ # queue group per failed push instead of the entire payload set.
354
+ #
355
+ # Cost is a round trip per distinct queue. The single-queue push — every
356
+ # `perform_async`, every same-class `push_bulk` — still writes exactly the
357
+ # one SADD + LPUSH pipeline it did before.
316
358
  def push_plain(conn, payloads, now)
317
- grouped = payloads.group_by { |j| j['queue'] }
318
- conn.call('SADD', 'queues', *grouped.keys)
319
- grouped.each do |queue, jobs|
359
+ payloads.group_by { |j| j['queue'] }.each do |queue, jobs|
320
360
  serialized = jobs.map do |j|
321
361
  j['enqueued_at'] = now
322
362
  Wurk.dump_json(j)
323
363
  end
324
- conn.call('LPUSH', "queue:#{queue}", *serialized)
364
+ conn.pipelined do |pipe|
365
+ pipe.call('SADD', 'queues', queue)
366
+ pipe.call('LPUSH', "queue:#{queue}", *serialized)
367
+ end
368
+ mark_delivered(jobs)
325
369
  end
326
370
  end
327
371
 
@@ -341,7 +385,7 @@ module Wurk
341
385
  :batch_push,
342
386
  keys: ["b-#{j['bid']}", "b-#{j['bid']}-jids", "queue:#{j['queue']}", 'queues',
343
387
  "b-#{j['bid']}-died", 'dead-batches'],
344
- argv: [j['queue'], j['jid'], Wurk.dump_json(j), j['bid']]
388
+ argv: [j['queue'], j['jid'], Wurk.dump_json(j), j['bid'], Wurk::Batch::DEFAULT_EXPIRY_SECONDS]
345
389
  )
346
390
  end
347
391
  end
@@ -360,7 +404,8 @@ module Wurk
360
404
  conn,
361
405
  :batch_schedule,
362
406
  keys: ['schedule', "b-#{j['bid']}", "b-#{j['bid']}-jids"],
363
- argv: [j['at'].to_s, Wurk.dump_json(j.except('enqueued_at', 'at')), j['jid']]
407
+ argv: [j['at'].to_s, Wurk.dump_json(j.except('enqueued_at', 'at')), j['jid'],
408
+ Wurk::Batch::DEFAULT_EXPIRY_SECONDS]
364
409
  )
365
410
  end
366
411
  end
@@ -369,11 +414,30 @@ module Wurk
369
414
  @redis_pool || Thread.current[:wurk_via_pool] || @config.redis_pool
370
415
  end
371
416
 
417
+ # Record a write Redis has acknowledged, for {Client::Buffered} to subtract
418
+ # from what it re-buffers: one push spans several pipelines (plain vs
419
+ # batched, immediate vs scheduled), and the ledger is what keeps a group
420
+ # that already landed out of the buffer when a later group fails. It
421
+ # accumulates across pool attempts and is never pruned — #raw_push claims no
422
+ # apply-safety, and RedisPool refuses to replay such a block once one of its
423
+ # round trips has completed, so the only replay left starts from an empty
424
+ # ledger.
425
+ def mark_delivered(payloads)
426
+ Thread.current[DELIVERED_KEY]&.concat(payloads)
427
+ end
428
+
372
429
  # Best-effort `sidekiq.jobs.enqueued` counter — one increment per payload
373
430
  # that actually made it past middleware AND Redis. Tags follow the same
374
431
  # `worker:`/`queue:` shape as Wurk::Metrics::Statsd so dashboards built
375
432
  # for the server-side emissions work unchanged.
376
- def emit_enqueued(payloads)
433
+ #
434
+ # `buffered` is what reliable_push swallowed into its outage buffer (see
435
+ # #raw_push). Those payloads are not enqueued: the ring buffer may still
436
+ # evict them, and the drain that does land one counts it then — so booking
437
+ # them here would inflate the counter on an outage and double-count every
438
+ # payload that later replays.
439
+ def emit_enqueued(payloads, buffered = nil)
440
+ payloads = reject_by_identity(payloads, buffered) if buffered && !buffered.empty?
377
441
  payloads.each do |p|
378
442
  Wurk::Metrics::Statsd.increment(
379
443
  'jobs.enqueued',
@@ -381,5 +445,14 @@ module Wurk
381
445
  )
382
446
  end
383
447
  end
448
+
449
+ # Set difference by object identity — `==` would fold two jobs carrying the
450
+ # same fields into one. Both sides are always the very Hash objects this
451
+ # push built, so identity is both exact and cheaper than hashing them.
452
+ def reject_by_identity(payloads, excluded)
453
+ seen = {}.compare_by_identity
454
+ excluded.each { |p| seen[p] = true }
455
+ payloads.reject { |p| seen.key?(p) }
456
+ end
384
457
  end
385
458
  end
data/lib/wurk/compat.rb CHANGED
@@ -170,7 +170,7 @@ module Sidekiq
170
170
  def configure_client(&) = Wurk.configure_client(&)
171
171
  def configure_embed(&) = Wurk.configure_embed(&)
172
172
  def default_configuration = Wurk.default_configuration
173
- def redis(&) = Wurk.redis(&)
173
+ def redis(idempotent: false, &) = Wurk.redis(idempotent:, &)
174
174
  def redis_pool = Wurk.redis_pool
175
175
  def logger = Wurk.logger
176
176
 
@@ -63,8 +63,8 @@ module Wurk
63
63
  config.logger
64
64
  end
65
65
 
66
- def redis(&)
67
- config.redis(&)
66
+ def redis(idempotent: false, &)
67
+ config.redis(idempotent:, &)
68
68
  end
69
69
 
70
70
  def handle_exception(ex, ctx = {})
@@ -4,8 +4,10 @@ require 'etc'
4
4
  require 'logger'
5
5
  require_relative 'middleware/chain'
6
6
  require_relative 'capsule'
7
+ require_relative 'pool_checkout'
7
8
  require_relative 'context'
8
9
  require_relative 'topology'
10
+ require_relative 'redis_options'
9
11
 
10
12
  module Wurk
11
13
  # Owns runtime knobs (concurrency, queues, timeouts, lifecycle events,
@@ -169,8 +171,14 @@ module Wurk
169
171
 
170
172
  # --- Redis ------------------------------------------------------------
171
173
 
174
+ # Validated here, in the process running the initializer, rather than later
175
+ # in whichever process first builds a pool. The swarm's children are the ones
176
+ # that construct pools, so a bad key used to kill every child on boot while
177
+ # the parent stayed up and healthy — Running pod, passing probe, zero jobs
178
+ # processed (#283).
172
179
  def redis=(hash)
173
180
  guard_frozen!
181
+ RedisOptions.validate!(hash)
174
182
  @redis_config = @redis_config.merge(hash.transform_keys(&:to_sym))
175
183
  end
176
184
 
@@ -191,8 +199,8 @@ module Wurk
191
199
  build_redis_pool(size: size, name: name)
192
200
  end
193
201
 
194
- def redis(&)
195
- redis_pool.with(&)
202
+ def redis(idempotent: false, &)
203
+ PoolCheckout.with(redis_pool, idempotent, &)
196
204
  end
197
205
 
198
206
  # --- Web dashboard Redis pool ----------------------------------------