wurk 1.3.1 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (103) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +4 -1
  3. data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +22 -5
  4. data/lib/wurk/batch/callbacks.rb +82 -12
  5. data/lib/wurk/batch/death_handler.rb +7 -4
  6. data/lib/wurk/batch/server_middleware.rb +1 -1
  7. data/lib/wurk/batch.rb +121 -15
  8. data/lib/wurk/capsule.rb +45 -16
  9. data/lib/wurk/cli.rb +48 -14
  10. data/lib/wurk/client/buffered.rb +200 -43
  11. data/lib/wurk/client.rb +181 -38
  12. data/lib/wurk/compat.rb +1 -1
  13. data/lib/wurk/component.rb +36 -4
  14. data/lib/wurk/configuration.rb +47 -7
  15. data/lib/wurk/context.rb +1 -1
  16. data/lib/wurk/cron.rb +94 -37
  17. data/lib/wurk/deploy.rb +5 -3
  18. data/lib/wurk/embedded.rb +13 -0
  19. data/lib/wurk/engine.rb +41 -1
  20. data/lib/wurk/errors.rb +15 -0
  21. data/lib/wurk/fetcher/reaper.rb +125 -58
  22. data/lib/wurk/fetcher/reliable.rb +366 -51
  23. data/lib/wurk/fetcher.rb +6 -0
  24. data/lib/wurk/heartbeat.rb +24 -12
  25. data/lib/wurk/history.rb +13 -1
  26. data/lib/wurk/job_logger.rb +16 -7
  27. data/lib/wurk/job_set.rb +3 -2
  28. data/lib/wurk/job_util.rb +44 -24
  29. data/lib/wurk/launcher.rb +230 -119
  30. data/lib/wurk/leader.rb +63 -14
  31. data/lib/wurk/limiter/base.rb +8 -10
  32. data/lib/wurk/limiter/bucket.rb +1 -1
  33. data/lib/wurk/limiter/concurrent.rb +27 -22
  34. data/lib/wurk/limiter/window.rb +13 -11
  35. data/lib/wurk/limiter.rb +7 -4
  36. data/lib/wurk/logger.rb +1 -1
  37. data/lib/wurk/lua/loader.rb +9 -3
  38. data/lib/wurk/lua.rb +97 -14
  39. data/lib/wurk/manager.rb +29 -13
  40. data/lib/wurk/metrics/accumulator.rb +95 -0
  41. data/lib/wurk/metrics/flusher.rb +70 -0
  42. data/lib/wurk/metrics/history.rb +102 -32
  43. data/lib/wurk/metrics/queue_rollup.rb +13 -1
  44. data/lib/wurk/metrics/rollup.rb +13 -1
  45. data/lib/wurk/metrics/statsd.rb +32 -18
  46. data/lib/wurk/middleware/chain.rb +31 -14
  47. data/lib/wurk/middleware/interrupt_handler.rb +7 -6
  48. data/lib/wurk/middleware/poison_pill.rb +93 -31
  49. data/lib/wurk/middleware.rb +2 -2
  50. data/lib/wurk/pool_checkout.rb +39 -0
  51. data/lib/wurk/process_set.rb +10 -5
  52. data/lib/wurk/processor.rb +83 -9
  53. data/lib/wurk/profiler.rb +9 -4
  54. data/lib/wurk/queue.rb +18 -7
  55. data/lib/wurk/rails_boot.rb +38 -7
  56. data/lib/wurk/redis_client_adapter.rb +49 -5
  57. data/lib/wurk/redis_pool.rb +71 -25
  58. data/lib/wurk/scheduled.rb +30 -2
  59. data/lib/wurk/shutdown_gate.rb +79 -0
  60. data/lib/wurk/stats.rb +19 -10
  61. data/lib/wurk/swarm/child_boot.rb +36 -4
  62. data/lib/wurk/swarm.rb +258 -43
  63. data/lib/wurk/timer_loop.rb +14 -0
  64. data/lib/wurk/version.rb +1 -1
  65. data/lib/wurk/web/config.rb +11 -7
  66. data/lib/wurk/web/enterprise.rb +58 -6
  67. data/lib/wurk/web/extension.rb +1 -1
  68. data/lib/wurk/web/search.rb +5 -3
  69. data/lib/wurk.rb +12 -10
  70. data/vendor/assets/dashboard/assets/{ArgsValue-D74zX0MI.js → ArgsValue-CcR2ya6e.js} +1 -1
  71. data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-CUXJUQ3Q.js} +1 -1
  72. data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-Cxan6Ngw.js} +1 -1
  73. data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-DC5EGM0g.js} +1 -1
  74. data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-Dlt8tXJA.js} +1 -1
  75. data/vendor/assets/dashboard/assets/Dashboard-DNLu_WCg.js +1 -0
  76. data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-dZ7VGlKS.js} +1 -1
  77. data/vendor/assets/dashboard/assets/Extension-DaFpEIJf.js +1 -0
  78. data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-CO3aYWIq.js} +1 -1
  79. data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-DSWbT6G0.js} +1 -1
  80. data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-Cb4PKXNR.js} +1 -1
  81. data/vendor/assets/dashboard/assets/Metrics-CCGzgCsT.js +1 -0
  82. data/vendor/assets/dashboard/assets/Modal-B86q6ruL.js +1 -0
  83. data/vendor/assets/dashboard/assets/{PageHeader-C44KNMGm.js → PageHeader-fPrCcp_-.js} +1 -1
  84. data/vendor/assets/dashboard/assets/{Profiles-xEVTyS2N.js → Profiles-BnS82nR_.js} +1 -1
  85. data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-CIyPevOy.js} +1 -1
  86. data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-DopwXkXl.js} +1 -1
  87. data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-1-Z7i1zE.js} +1 -1
  88. data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-ByA6eTma.js} +1 -1
  89. data/vendor/assets/dashboard/assets/{Skeleton-DzR7XNxz.js → Skeleton-bC7HfQ9r.js} +1 -1
  90. data/vendor/assets/dashboard/assets/{charts-BVHHGof7.js → charts-CLLzJ7vK.js} +1 -1
  91. data/vendor/assets/dashboard/assets/index-B1N8hQUh.js +141 -0
  92. data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
  93. data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-DpBjkf6_.js} +1 -1
  94. data/vendor/assets/dashboard/assets/{useSort-BeYbztkN.js → useSort-DvpwuNQE.js} +1 -1
  95. data/vendor/assets/dashboard/index.html +3 -3
  96. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  97. metadata +32 -27
  98. data/vendor/assets/dashboard/assets/Dashboard-B9rOrkzk.js +0 -1
  99. data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
  100. data/vendor/assets/dashboard/assets/Metrics-BBTDxcaE.js +0 -1
  101. data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
  102. data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
  103. data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
data/lib/wurk/client.rb CHANGED
@@ -26,6 +26,29 @@ module Wurk
26
26
  SCHEDULED_BATCH_SIZE = 100
27
27
  SPREAD_INTERVAL_FLOOR = 5
28
28
 
29
+ # Batched (`bid`) payloads per EVALSHA pipeline. Ours, not Sidekiq's — it
30
+ # has no batches. Sized to DEFAULT_BATCH_SIZE so no existing caller's
31
+ # round-trip count moves: `push_bulk` already hands #raw_push at most that
32
+ # many payloads, so its batched pipeline stays exactly one round trip.
33
+ #
34
+ # The cap is for the one path that isn't pre-sliced: `autoflush = true`
35
+ # buffers a whole `Batch#jobs` block, so #flush_batched can be handed an
36
+ # unbounded payload set. Unsliced that is one pipeline holding every
37
+ # command and every reply in memory at once, and — Lua being atomic and
38
+ # single-threaded — one uninterrupted server-side sweep that blocks every
39
+ # other client for its duration. Same reasoning as the LIMIT on
40
+ # RELIABLE_SCHEDULE_PROMOTE.
41
+ BATCH_PIPELINE_SLICE = 1_000
42
+
43
+ # Thread-local slot holding the payloads of the current push whose Redis
44
+ # write is confirmed applied. {Client::Buffered} subtracts them from the set
45
+ # it re-buffers when a *later* phase of the same push loses the connection,
46
+ # so an already-written job is never replayed into a second copy.
47
+ # Thread-local because one Client instance serves every producer thread;
48
+ # opened and closed by Buffered, the only reader, so an un-prepended Client
49
+ # pays a single nil check per write phase.
50
+ DELIVERED_KEY = :wurk_client_delivered
51
+
29
52
  attr_accessor :redis_pool
30
53
 
31
54
  def initialize(pool: nil, config: nil, chain: nil)
@@ -51,8 +74,8 @@ module Wurk
51
74
  return nil unless payload
52
75
 
53
76
  verify_json(payload)
54
- raw_push([payload])
55
- emit_enqueued([payload])
77
+ buffered = raw_push([payload])
78
+ emit_enqueued([payload], buffered)
56
79
  payload['jid']
57
80
  end
58
81
 
@@ -129,7 +152,21 @@ module Wurk
129
152
 
130
153
  private
131
154
 
155
+ # #push and #push_bulk verify at different points and both match Sidekiq
156
+ # exactly: push walks the payload the chain handed back (sidekiq
157
+ # client.rb:101 — normalize → middleware → verify → raw_push), bulk walks it
158
+ # inside the innermost block (sidekiq client.rb:165). Push used to do both,
159
+ # and since `strict_args_mode` defaults to :raise the second full recursive
160
+ # args walk was never skipped.
161
+ #
162
+ # Bulk keeps its walk inside the block on purpose: a client middleware that
163
+ # halts the job there short-circuits the walk, and hoisting it out would
164
+ # raise on args that middleware was about to drop.
132
165
  def invoke_chain(normed)
166
+ @chain.invoke(normed['class'], normed, normed['queue'], pool) { normed }
167
+ end
168
+
169
+ def invoke_chain_verified(normed)
133
170
  @chain.invoke(normed['class'], normed, normed['queue'], pool) do
134
171
  verify_json(normed)
135
172
  normed
@@ -169,8 +206,8 @@ module Wurk
169
206
  payloads = build_bulk_payloads(slice, base, ats)
170
207
  compacted = payloads.compact
171
208
  if compacted.any?
172
- raw_push(compacted)
173
- emit_enqueued(compacted)
209
+ buffered = raw_push(compacted)
210
+ emit_enqueued(compacted, buffered)
174
211
  end
175
212
  jids.concat(payloads.map { |p| p && p['jid'] })
176
213
  end
@@ -182,7 +219,7 @@ module Wurk
182
219
  item = base.merge('args' => job_args)
183
220
  item['at'] = ats[idx] if ats
184
221
  normed = normalize_item(item)
185
- invoke_chain(normed)
222
+ invoke_chain_verified(normed)
186
223
  end
187
224
  end
188
225
 
@@ -222,16 +259,32 @@ module Wurk
222
259
  # Adds happen one payload at a time so an `autoflush = N` actually bounds
223
260
  # the pipeline size — a bulk push of 100 with N=2 must flush 2/2/... not
224
261
  # 100 in one shot.
262
+ #
263
+ # Returns the payloads it did NOT get to Redis: always nil here, since a
264
+ # plain Client either writes them all or raises. {Client::Buffered}
265
+ # overrides the contract — the payloads it diverted into the outage buffer
266
+ # come back so #push can keep them out of the enqueued metric.
225
267
  def raw_push(payloads)
226
268
  # Test modes short-circuit the Redis write (and the batch buffer): :fake
227
269
  # collects payloads in-memory, :inline runs them now. Client middleware
228
270
  # has already run by this point, matching Sidekiq.
229
- return ::Wurk::Testing.dispatch_push(payloads) if ::Wurk::Testing.enabled?
271
+ if ::Wurk::Testing.enabled?
272
+ ::Wurk::Testing.dispatch_push(payloads)
273
+ return nil
274
+ end
230
275
 
231
276
  buffer = Thread.current[Wurk::Batch::BUFFER_KEY]
232
277
  return buffer_add(buffer, payloads) if buffer && payloads.all? { |p| p['bid'] && !p['at'] }
233
278
 
279
+ # No apply-safety claim: every command below appends (LPUSH, ZADD, the
280
+ # batch Lua's counters), so a block replayed after a lost reply is a
281
+ # second copy of the job. A post-write timeout raises out of here instead
282
+ # — {Client::Buffered} turns that into an outage-buffer entry, and a plain
283
+ # Client hands it to whoever called `perform_async`. The pool's pre-apply
284
+ # retry only fires while this block has landed nothing, so a queue group
285
+ # that already went out is never re-pushed by a replay.
234
286
  pool.with { |conn| atomic_push(conn, payloads) }
287
+ nil
235
288
  end
236
289
 
237
290
  # Batch autoflush path: accumulate each non-scheduled batched payload into
@@ -262,7 +315,10 @@ module Wurk
262
315
  # duplicate the scheduled entry.
263
316
  def push_scheduled_split(conn, payloads)
264
317
  batched, plain = payloads.partition { |j| j['bid'] }
265
- conn.pipelined { |pipe| push_scheduled(pipe, plain) } unless plain.empty?
318
+ unless plain.empty?
319
+ conn.pipelined { |pipe| push_scheduled(pipe, plain) }
320
+ mark_delivered(plain)
321
+ end
266
322
  push_batched_scheduled_pipelined(conn, batched) unless batched.empty?
267
323
  end
268
324
 
@@ -282,10 +338,31 @@ module Wurk
282
338
  def push_immediate(conn, payloads)
283
339
  now = now_in_millis
284
340
  batched, plain = payloads.partition { |j| j['bid'] }
285
- conn.pipelined { |pipe| push_plain(pipe, plain, now) } unless plain.empty?
341
+ push_plain(conn, plain, now) unless plain.empty?
286
342
  push_batched_pipelined(conn, batched, now) unless batched.empty?
287
343
  end
288
344
 
345
+ # One pipeline per BATCH_PIPELINE_SLICE payloads, each marked delivered the
346
+ # moment its reply is in — same contract as push_plain_group, and for the
347
+ # same reason: a slice that Redis already accepted must stay out of the
348
+ # reliable_push ledger, or a later slice's failure would report it as
349
+ # undelivered.
350
+ def push_batched_pipelined(conn, batched, now)
351
+ batched.each_slice(BATCH_PIPELINE_SLICE) do |slice|
352
+ eval_batched_slice(conn) { |pipe, eval_method| push_batched(pipe, slice, now, eval_method: eval_method) }
353
+ mark_delivered(slice)
354
+ end
355
+ end
356
+
357
+ # Same slicing and NOSCRIPT recovery as push_batched_pipelined, for the
358
+ # scheduled batched path (BATCH_SCHEDULE instead of BATCH_PUSH).
359
+ def push_batched_scheduled_pipelined(conn, batched)
360
+ batched.each_slice(BATCH_PIPELINE_SLICE) do |slice|
361
+ eval_batched_slice(conn) { |pipe, eval_method| push_batched_scheduled(pipe, slice, eval_method: eval_method) }
362
+ mark_delivered(slice)
363
+ end
364
+ end
365
+
289
366
  # Outside of test boots and `SCRIPT FLUSH` the rescue branch is dead
290
367
  # code; the eager `script_load_all` after fork keeps the script cache
291
368
  # hot for the life of the connection. The retry uses EVAL (source-embedded)
@@ -293,45 +370,76 @@ module Wurk
293
370
  # NOSCRIPT a second time under heavy CI load (WorkerTest 3.4/7.2 flake).
294
371
  # `script_load_all` still primes the cache so the *next* pipeline returns
295
372
  # to the EVALSHA fast path.
296
- def push_batched_pipelined(conn, batched, now)
297
- conn.pipelined { |pipe| push_batched(pipe, batched, now) }
373
+ #
374
+ # Replaying the slice is safe precisely because every command in it is the
375
+ # same script: a flushed cache NOSCRIPTs all of them and applies none.
376
+ # Recovery is per slice, so the slices already acknowledged above are never
377
+ # re-sent. Mirrors Fetcher::Reliable#requeue_pipelined.
378
+ def eval_batched_slice(conn)
379
+ conn.pipelined { |pipe| yield(pipe, :eval_cached) }
298
380
  rescue RedisClient::CommandError => e
299
381
  raise unless e.message.to_s.start_with?('NOSCRIPT')
300
382
 
301
383
  Wurk::Lua::Loader.script_load_all(conn)
302
- conn.pipelined { |pipe| push_batched(pipe, batched, now, eval_method: :eval_with_source) }
384
+ conn.pipelined { |pipe| yield(pipe, :eval_with_source) }
303
385
  end
304
386
 
305
- # Same NOSCRIPT-recovery shape as push_batched_pipelined, for the scheduled
306
- # batched path (BATCH_SCHEDULE instead of BATCH_PUSH).
307
- def push_batched_scheduled_pipelined(conn, batched)
308
- conn.pipelined { |pipe| push_batched_scheduled(pipe, batched) }
309
- rescue RedisClient::CommandError => e
310
- raise unless e.message.to_s.start_with?('NOSCRIPT')
387
+ # One pipeline per queue, marked delivered the moment its reply is in.
388
+ # A push can die between groups — or land its whole plain phase and then
389
+ # lose the connection in the batched phase above — and whatever Redis
390
+ # already accepted must stay out of the reliable_push buffer; replaying it
391
+ # would enqueue a second copy of a job that ran fine.
392
+ #
393
+ # The group in flight when the socket drops stays unmarked and so is
394
+ # replayed: a lost reply is indistinguishable from a lost command, so that
395
+ # residual is at-least-once by construction. The split bounds it to one
396
+ # queue group per failed push instead of the entire payload set.
397
+ #
398
+ # Cost is a round trip per distinct queue. The single-queue push — every
399
+ # `perform_async`, every same-class `push_bulk` — still writes exactly the
400
+ # one SADD + LPUSH pipeline it did before.
401
+ #
402
+ # `uniform_queue` short-circuits the common case (one job, or many jobs
403
+ # all destined for the same queue) without paying for the `group_by`
404
+ # Hash + per-group Array allocations; only a genuinely mixed-queue batch
405
+ # falls through to grouping.
406
+ def push_plain(conn, payloads, now)
407
+ queue = uniform_queue(payloads)
408
+ return push_plain_group(conn, queue, payloads, now) if queue
311
409
 
312
- Wurk::Lua::Loader.script_load_all(conn)
313
- conn.pipelined { |pipe| push_batched_scheduled(pipe, batched, eval_method: :eval_with_source) }
410
+ payloads.group_by { |j| j['queue'] }.each { |q, jobs| push_plain_group(conn, q, jobs, now) }
314
411
  end
315
412
 
316
- def push_plain(conn, payloads, now)
317
- grouped = payloads.group_by { |j| j['queue'] }
318
- conn.call('SADD', 'queues', *grouped.keys)
319
- grouped.each do |queue, jobs|
320
- serialized = jobs.map do |j|
321
- j['enqueued_at'] = now
322
- Wurk.dump_json(j)
323
- end
324
- conn.call('LPUSH', "queue:#{queue}", *serialized)
413
+ def uniform_queue(payloads)
414
+ first = payloads[0]['queue']
415
+ return first if payloads.size == 1
416
+
417
+ first if payloads.all? { |j| j['queue'] == first }
418
+ end
419
+
420
+ def push_plain_group(conn, queue, jobs, now)
421
+ serialized = jobs.map do |j|
422
+ j['enqueued_at'] = now
423
+ Wurk.dump_json(j)
424
+ end
425
+ conn.pipelined do |pipe|
426
+ pipe.call('SADD', 'queues', queue)
427
+ pipe.call('LPUSH', "queue:#{queue}", *serialized)
325
428
  end
429
+ mark_delivered(jobs)
326
430
  end
327
431
 
328
432
  # Batched jobs route through BATCH_PUSH: increments b-<bid> total+pending,
329
433
  # SADDs jid into the live set, registers the queue, LPUSHes the payload —
330
- # all atomically. One Redis round-trip per job (no pipeline grouping)
331
- # because the lua needs per-job KEYS bound. Acceptable cost: batch
332
- # enqueue is not the hot path; correctness is. `eval_method` is the
434
+ # all atomically. The Lua binds per-job KEYS, so grouping N jobs into one
435
+ # EVALSHA isn't available; they ride one pipeline instead (the `conn` here
436
+ # is always the pipeline #push_batched_pipelined opened), so the cost is
437
+ # N commands and one round trip, not N round trips. `eval_method` is the
333
438
  # Wurk::Lua::Loader entry point (`:eval_cached` for the hot EVALSHA path,
334
- # `:eval_with_source` for the EVAL-source retry).
439
+ # `:eval_with_source` for the EVAL-source retry) — neither reads the reply,
440
+ # which is what makes the pipelined form legal: `eval_cached`'s inline
441
+ # NOSCRIPT rescue can't fire against a buffered call, so recovery is the
442
+ # caller's finalize-time rescue.
335
443
  def push_batched(conn, payloads, now, eval_method: :eval_cached)
336
444
  payloads.each do |j|
337
445
  j['enqueued_at'] = now
@@ -341,7 +449,7 @@ module Wurk
341
449
  :batch_push,
342
450
  keys: ["b-#{j['bid']}", "b-#{j['bid']}-jids", "queue:#{j['queue']}", 'queues',
343
451
  "b-#{j['bid']}-died", 'dead-batches'],
344
- argv: [j['queue'], j['jid'], Wurk.dump_json(j), j['bid']]
452
+ argv: [j['queue'], j['jid'], Wurk.dump_json(j), j['bid'], Wurk::Batch::DEFAULT_EXPIRY_SECONDS]
345
453
  )
346
454
  end
347
455
  end
@@ -350,9 +458,9 @@ module Wurk
350
458
  # total/pending increment registers the job in its batch at creation, and
351
459
  # the ZADD defers it onto `schedule`. Payload is stripped of `at`/
352
460
  # `enqueued_at` exactly like push_scheduled — `enqueued_at` is stamped fresh
353
- # at promotion, never while the job sits scheduled (spec §7.1). One Redis
354
- # round-trip per job because the Lua binds per-job KEYS; scheduled batch
355
- # enqueue is not the hot path.
461
+ # at promotion, never while the job sits scheduled (spec §7.1). Per-job
462
+ # KEYS again, so one EVALSHA per job — pipelined by
463
+ # #push_batched_scheduled_pipelined into one round trip per slice.
356
464
  def push_batched_scheduled(conn, payloads, eval_method: :eval_cached)
357
465
  payloads.each do |j|
358
466
  Wurk::Lua::Loader.public_send(
@@ -360,7 +468,8 @@ module Wurk
360
468
  conn,
361
469
  :batch_schedule,
362
470
  keys: ['schedule', "b-#{j['bid']}", "b-#{j['bid']}-jids"],
363
- argv: [j['at'].to_s, Wurk.dump_json(j.except('enqueued_at', 'at')), j['jid']]
471
+ argv: [j['at'].to_s, Wurk.dump_json(j.except('enqueued_at', 'at')), j['jid'],
472
+ Wurk::Batch::DEFAULT_EXPIRY_SECONDS]
364
473
  )
365
474
  end
366
475
  end
@@ -369,11 +478,36 @@ module Wurk
369
478
  @redis_pool || Thread.current[:wurk_via_pool] || @config.redis_pool
370
479
  end
371
480
 
481
+ # Record a write Redis has acknowledged, for {Client::Buffered} to subtract
482
+ # from what it re-buffers: one push spans several pipelines (plain vs
483
+ # batched, immediate vs scheduled), and the ledger is what keeps a group
484
+ # that already landed out of the buffer when a later group fails. It
485
+ # accumulates across pool attempts and is never pruned — #raw_push claims no
486
+ # apply-safety, and RedisPool refuses to replay such a block once one of its
487
+ # round trips has completed, so the only replay left starts from an empty
488
+ # ledger.
489
+ def mark_delivered(payloads)
490
+ Thread.current[DELIVERED_KEY]&.concat(payloads)
491
+ end
492
+
372
493
  # Best-effort `sidekiq.jobs.enqueued` counter — one increment per payload
373
494
  # that actually made it past middleware AND Redis. Tags follow the same
374
495
  # `worker:`/`queue:` shape as Wurk::Metrics::Statsd so dashboards built
375
496
  # for the server-side emissions work unchanged.
376
- def emit_enqueued(payloads)
497
+ #
498
+ # `buffered` is what reliable_push swallowed into its outage buffer (see
499
+ # #raw_push). Those payloads are not enqueued: the ring buffer may still
500
+ # evict them, and the drain that does land one counts it then — so booking
501
+ # them here would inflate the counter on an outage and double-count every
502
+ # payload that later replays.
503
+ #
504
+ # Resolve the client once for the whole batch and bail before touching a
505
+ # payload: unconfigured is the common case, and the tags below cost two
506
+ # Strings and an Array per job for `increment` to immediately drop.
507
+ def emit_enqueued(payloads, buffered = nil)
508
+ return if Wurk::Metrics::Statsd.safe_client.nil?
509
+
510
+ payloads = reject_by_identity(payloads, buffered) if buffered && !buffered.empty?
377
511
  payloads.each do |p|
378
512
  Wurk::Metrics::Statsd.increment(
379
513
  'jobs.enqueued',
@@ -381,5 +515,14 @@ module Wurk
381
515
  )
382
516
  end
383
517
  end
518
+
519
+ # Set difference by object identity — `==` would fold two jobs carrying the
520
+ # same fields into one. Both sides are always the very Hash objects this
521
+ # push built, so identity is both exact and cheaper than hashing them.
522
+ def reject_by_identity(payloads, excluded)
523
+ seen = {}.compare_by_identity
524
+ excluded.each { |p| seen[p] = true }
525
+ payloads.reject { |p| seen.key?(p) }
526
+ end
384
527
  end
385
528
  end
data/lib/wurk/compat.rb CHANGED
@@ -170,7 +170,7 @@ module Sidekiq
170
170
  def configure_client(&) = Wurk.configure_client(&)
171
171
  def configure_embed(&) = Wurk.configure_embed(&)
172
172
  def default_configuration = Wurk.default_configuration
173
- def redis(&) = Wurk.redis(&)
173
+ def redis(idempotent: false, &) = Wurk.redis(idempotent:, &)
174
174
  def redis_pool = Wurk.redis_pool
175
175
  def logger = Wurk.logger
176
176
 
@@ -37,8 +37,34 @@ module Wurk
37
37
 
38
38
  # --- identity -------------------------------------------------------
39
39
 
40
+ # Base36 `thread.object_id ^ pid` — the id in every log line and the key
41
+ # each Processor publishes its in-flight job under. Constant for the life
42
+ # of a thread inside one process and read several times per job, so it is
43
+ # memoized per thread (frozen: it is used as a Hash key, and an unfrozen
44
+ # String key is duped on every store).
45
+ #
46
+ # The pid is memoized alongside it because the thread that calls fork keeps
47
+ # its thread-locals in the child, where the pid — and therefore the tid —
48
+ # has changed. Without the guard a forked child would report the parent's
49
+ # tid and collide with it in `<identity>:work`.
50
+ #
51
+ # Thread-local, not `Thread#[]`: the latter is fiber-local, so a job that
52
+ # runs inside a Fiber (or any Enumerator) would miss the memo and allocate
53
+ # a fresh String on every read — the identity the memo exists to cache is
54
+ # the thread's, and it does not change when a fiber does.
55
+ def self.tid
56
+ thread = Thread.current
57
+ memo = thread.thread_variable_get(:wurk_tid)
58
+ pid = ::Process.pid
59
+ return memo[1] if memo && memo[0] == pid
60
+
61
+ id = (thread.object_id ^ pid).to_s(36).freeze
62
+ thread.thread_variable_set(:wurk_tid, [pid, id].freeze)
63
+ id
64
+ end
65
+
40
66
  def tid
41
- (Thread.current.object_id ^ ::Process.pid).to_s(36)
67
+ Component.tid
42
68
  end
43
69
 
44
70
  def hostname
@@ -63,8 +89,8 @@ module Wurk
63
89
  config.logger
64
90
  end
65
91
 
66
- def redis(&)
67
- config.redis(&)
92
+ def redis(idempotent: false, &)
93
+ config.redis(idempotent:, &)
68
94
  end
69
95
 
70
96
  def handle_exception(ex, ctx = {})
@@ -110,10 +136,16 @@ module Wurk
110
136
  # Spawns a named thread that runs `block` under `watchdog(name)`. The
111
137
  # parent must retain the returned Thread; otherwise GC may not, but
112
138
  # report_on_exception is disabled so we don't double-log on death.
139
+ #
140
+ # Priority resolution matches Sidekiq (component.rb:44-48): explicit
141
+ # argument, then `config.thread_priority`, then -1. Ruby's default of 0
142
+ # buys a 100ms timeslice; each negative step halves it, so -1 keeps a
143
+ # CPU-heavy capsule from starving its siblings for a whole tick.
113
144
  def safe_thread(name, priority: nil, &block)
145
+ resolved = priority || config.thread_priority || DEFAULT_THREAD_PRIORITY
114
146
  Thread.new do
115
147
  Thread.current.name = name
116
- Thread.current.priority = priority || DEFAULT_THREAD_PRIORITY
148
+ Thread.current.priority = resolved
117
149
  Thread.current.report_on_exception = false
118
150
  watchdog(name, &block)
119
151
  end
@@ -4,6 +4,7 @@ require 'etc'
4
4
  require 'logger'
5
5
  require_relative 'middleware/chain'
6
6
  require_relative 'capsule'
7
+ require_relative 'pool_checkout'
7
8
  require_relative 'context'
8
9
  require_relative 'topology'
9
10
  require_relative 'redis_options'
@@ -86,7 +87,16 @@ module Wurk
86
87
  # config.dogstatsd = -> { Datadog::Statsd.new('host', 8125) }
87
88
  #
88
89
  # Spec: docs/target/sidekiq-pro.md §9.1.
89
- attr_accessor :dogstatsd
90
+ attr_reader :dogstatsd
91
+
92
+ # Assignment drops Statsd's resolved-client memo. That memo caches the
93
+ # "nothing configured" answer too (so the no-client emit path allocates
94
+ # nothing), which means a builder wired up after the first emit would
95
+ # otherwise never be picked up.
96
+ def dogstatsd=(builder)
97
+ @dogstatsd = builder
98
+ Wurk::Metrics::Statsd.reset!
99
+ end
90
100
 
91
101
  def initialize(options = {})
92
102
  @options = deep_dup_defaults.merge(options)
@@ -198,8 +208,8 @@ module Wurk
198
208
  build_redis_pool(size: size, name: name)
199
209
  end
200
210
 
201
- def redis(&)
202
- redis_pool.with(&)
211
+ def redis(idempotent: false, &)
212
+ PoolCheckout.trusted(redis_pool, idempotent, &)
203
213
  end
204
214
 
205
215
  # --- Web dashboard Redis pool ----------------------------------------
@@ -242,6 +252,11 @@ module Wurk
242
252
  @directory[name] = instance
243
253
  end
244
254
 
255
+ # Memoizes on a miss, and the first miss can land long after boot (an
256
+ # extension resolved on its first tick), which is why `freeze!` leaves
257
+ # `@directory` writable: frozen, that lookup raised FrozenError instead of
258
+ # building the default. `register` — the host-facing half — still refuses
259
+ # writes past the freeze, so the closed surface is unchanged.
245
260
  def lookup(name, default_class = nil)
246
261
  @directory[name] ||= default_class&.new
247
262
  end
@@ -481,14 +496,39 @@ module Wurk
481
496
  mb&.positive? ? mb * 1024 : nil
482
497
  end
483
498
 
499
+ # The pre-fork half of `freeze!`, and the only half a forking parent can
500
+ # run: capsules stay writable until each child has applied its slot
501
+ # (ChildBoot#apply_slot_to_config) and opened its own Redis pools. What is
502
+ # left is slot-independent — the options Hash and every capsule's middleware
503
+ # chains — so the swarm parent settles it once and every child inherits the
504
+ # result copy-on-write instead of allocating and dirtying its own copy.
505
+ # Freezing the options here also makes a post-fork option write raise in the
506
+ # child that wrote it, rather than silently diverging from its siblings.
507
+ #
508
+ # `@directory` is deliberately left out — see #lookup.
509
+ def prepare_for_fork!
510
+ # The capsule every swarm child configures (ChildBoot reaches for it by
511
+ # name), and the one a client-only config builds on its first enqueue.
512
+ # Materialized here so its chains are shared rather than rebuilt N times
513
+ # — and so `freeze!` can't close `@capsules` around a name that is only
514
+ # ever resolved later, which turned that first resolution into a
515
+ # FrozenError on the frozen Hash.
516
+ default_capsule
517
+ @capsules.each_value(&:prepare_shared!)
518
+ @options.freeze
519
+ @frozen = true
520
+ self
521
+ end
522
+
523
+ # Guarded on the capsule table, not `@frozen`: a swarm child reaches this
524
+ # with `prepare_for_fork!` already run in its parent, so `@frozen` is true
525
+ # while the capsules it has just configured are still open.
484
526
  def freeze!
485
- return self if @frozen
527
+ return self if @capsules.frozen?
486
528
 
529
+ prepare_for_fork!
487
530
  @capsules.each_value(&:freeze)
488
531
  @capsules.freeze
489
- @options.freeze
490
- @directory.freeze
491
- @frozen = true
492
532
  self
493
533
  end
494
534
 
data/lib/wurk/context.rb CHANGED
@@ -14,7 +14,7 @@ module Wurk
14
14
  # context is restored on exit, even if the block raises — safe to nest.
15
15
  def self.with(hash)
16
16
  prior = Thread.current[KEY]
17
- Thread.current[KEY] = (prior || {}).merge(hash)
17
+ Thread.current[KEY] = prior ? prior.merge(hash) : hash
18
18
  yield
19
19
  ensure
20
20
  Thread.current[KEY] = prior