wurk 1.3.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +1 -0
  3. data/lib/wurk/batch/callbacks.rb +82 -12
  4. data/lib/wurk/batch/death_handler.rb +7 -4
  5. data/lib/wurk/batch/server_middleware.rb +1 -1
  6. data/lib/wurk/batch.rb +121 -15
  7. data/lib/wurk/capsule.rb +5 -4
  8. data/lib/wurk/cli.rb +48 -14
  9. data/lib/wurk/client/buffered.rb +193 -43
  10. data/lib/wurk/client.rb +87 -14
  11. data/lib/wurk/compat.rb +1 -1
  12. data/lib/wurk/component.rb +2 -2
  13. data/lib/wurk/configuration.rb +10 -2
  14. data/lib/wurk/cron.rb +94 -37
  15. data/lib/wurk/deploy.rb +5 -3
  16. data/lib/wurk/embedded.rb +13 -0
  17. data/lib/wurk/fetcher/reaper.rb +113 -56
  18. data/lib/wurk/fetcher/reliable.rb +62 -9
  19. data/lib/wurk/heartbeat.rb +22 -10
  20. data/lib/wurk/history.rb +13 -1
  21. data/lib/wurk/launcher.rb +133 -66
  22. data/lib/wurk/leader.rb +29 -10
  23. data/lib/wurk/limiter/base.rb +8 -10
  24. data/lib/wurk/limiter/bucket.rb +1 -1
  25. data/lib/wurk/limiter/concurrent.rb +27 -22
  26. data/lib/wurk/limiter/window.rb +13 -11
  27. data/lib/wurk/limiter.rb +7 -4
  28. data/lib/wurk/lua.rb +97 -14
  29. data/lib/wurk/manager.rb +29 -13
  30. data/lib/wurk/metrics/history.rb +4 -3
  31. data/lib/wurk/metrics/queue_rollup.rb +13 -1
  32. data/lib/wurk/metrics/rollup.rb +13 -1
  33. data/lib/wurk/middleware/interrupt_handler.rb +7 -6
  34. data/lib/wurk/middleware/poison_pill.rb +70 -29
  35. data/lib/wurk/middleware.rb +2 -2
  36. data/lib/wurk/pool_checkout.rb +29 -0
  37. data/lib/wurk/process_set.rb +10 -5
  38. data/lib/wurk/processor.rb +6 -0
  39. data/lib/wurk/profiler.rb +3 -2
  40. data/lib/wurk/queue.rb +10 -7
  41. data/lib/wurk/rails_boot.rb +38 -7
  42. data/lib/wurk/redis_client_adapter.rb +48 -4
  43. data/lib/wurk/redis_options.rb +142 -0
  44. data/lib/wurk/redis_pool.rb +102 -39
  45. data/lib/wurk/scheduled.rb +30 -2
  46. data/lib/wurk/stats.rb +14 -9
  47. data/lib/wurk/swarm/child_boot.rb +12 -0
  48. data/lib/wurk/swarm.rb +174 -33
  49. data/lib/wurk/timer_loop.rb +14 -0
  50. data/lib/wurk/version.rb +1 -1
  51. data/lib/wurk/web/enterprise.rb +58 -6
  52. data/lib/wurk/web/extension.rb +1 -1
  53. data/lib/wurk/web/search.rb +5 -3
  54. data/lib/wurk.rb +53 -2
  55. data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-OmC5NPgw.js} +1 -1
  56. data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-CIpai7St.js} +1 -1
  57. data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-A_kwSR6Q.js} +1 -1
  58. data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-BG7HTqlp.js} +1 -1
  59. data/vendor/assets/dashboard/assets/{Dashboard-B9rOrkzk.js → Dashboard-A_ToqHoo.js} +1 -1
  60. data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-8J21jMyK.js} +1 -1
  61. data/vendor/assets/dashboard/assets/Extension-B4Q9FIQu.js +1 -0
  62. data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-Fh_Ae7UW.js} +1 -1
  63. data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-Ceng0PMB.js} +1 -1
  64. data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-CruDWvNZ.js} +1 -1
  65. data/vendor/assets/dashboard/assets/{Metrics-BBTDxcaE.js → Metrics-CIT7VCoN.js} +1 -1
  66. data/vendor/assets/dashboard/assets/Modal-CN3rdKA_.js +1 -0
  67. data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-D86FYohJ.js} +1 -1
  68. data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-Bz1O1D-i.js} +1 -1
  69. data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-B6h2akTu.js} +1 -1
  70. data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-OOu22e5s.js} +1 -1
  71. data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
  72. data/vendor/assets/dashboard/assets/index-D_lSDwKw.js +141 -0
  73. data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-dVPGEWzn.js} +1 -1
  74. data/vendor/assets/dashboard/index.html +2 -2
  75. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  76. metadata +22 -20
  77. data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
  78. data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
  79. data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
  80. data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
data/lib/wurk/cron.rb CHANGED
@@ -6,6 +6,7 @@ require 'time'
6
6
  require_relative 'component'
7
7
  require_relative 'client'
8
8
  require_relative 'leader'
9
+ require_relative 'timer_loop'
9
10
 
10
11
  module Wurk
11
12
  # Sidekiq Enterprise periodic jobs. Pure leader-driven cron — only the
@@ -477,8 +478,8 @@ module Wurk
477
478
 
478
479
  private
479
480
 
480
- def redis(&)
481
- @config ? @config.redis(&) : Wurk.redis(&)
481
+ def redis(idempotent: false, &)
482
+ @config ? @config.redis(idempotent:, &) : Wurk.redis(idempotent:, &)
482
483
  end
483
484
  end
484
485
 
@@ -520,34 +521,35 @@ module Wurk
520
521
 
521
522
  def initialize(config)
522
523
  @config = config
523
- @done = false
524
- @mutex = ::Mutex.new
525
- @sleeper = ::ConditionVariable.new
526
524
  @client = Client.new(config: config)
527
525
  @thread = nil
528
526
  # Operators never need to touch this; integration tests shrink it so a
529
527
  # due loop fires within the test window instead of waiting a full minute.
530
528
  @tick_interval = config[:cron_tick_interval] || DEFAULT_TICK_SECONDS
529
+ @timer = TimerLoop.new(@tick_interval)
531
530
  end
532
531
 
532
+ # TimerLoop waits one interval before the first tick: don't fire a
533
+ # catch-up burst the instant we boot (the leader is barely settled), and
534
+ # let a short-lived process exit without ticking at all.
533
535
  def start
534
- @poller_thread ||= safe_thread('cron-poller') do # rubocop:disable Naming/MemoizedInstanceVariableName
535
- # Wait one interval before the first tick: don't fire a catch-up burst
536
- # the instant we boot (the leader is barely settled), and let a
537
- # short-lived process exit without ticking at all.
538
- wait
539
- until @done
540
- tick
541
- wait
542
- end
543
- end
536
+ return @thread if @thread
537
+
538
+ @timer.reset
539
+ @thread = safe_thread('cron-poller') { @timer.run { tick } }
544
540
  end
545
541
 
542
+ # Blocks until the thread is really gone: the launcher releases the
543
+ # cluster lock immediately after this returns, and a tick still in flight
544
+ # would enqueue loops the next leader is about to fire itself.
545
+ #
546
+ # Cleared only on a confirmed join (Thread#join returns nil on timeout):
547
+ # a wedged thread must stay tracked so #start's guard returns it instead
548
+ # of calling @timer.reset, which would un-terminate the loop it is still
549
+ # inside and leave two tick threads double-enqueuing the same loops.
546
550
  def terminate
547
- @mutex.synchronize do
548
- @done = true
549
- @sleeper.signal
550
- end
551
+ @timer.terminate
552
+ @thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
551
553
  end
552
554
 
553
555
  # Leader-gated by the single cluster lock (Component#leader? reads
@@ -562,18 +564,23 @@ module Wurk
562
564
  handle_exception(e, { context: 'cron-poller' })
563
565
  end
564
566
 
567
+ # The marks advance *before* the push, via CAS: the leader gate is a
568
+ # cached read, so a second poller can reach the same due loop for a few
569
+ # seconds after a handover. Losing the CAS means another tick owns this
570
+ # slot — enqueue nothing.
565
571
  def enqueue_if_due(loop_obj)
566
572
  return if loop_obj.paused?
567
573
 
568
574
  now = ::Time.now.to_i
569
- prev_fire, next_fire = read_fire_marks(loop_obj.lid)
570
- next_fire ||= loop_obj.next_fire_at(prev_fire || (now - @tick_interval))
571
- return if next_fire.nil? || next_fire > now
575
+ slot, mark = due_slot(loop_obj, now)
576
+ return if slot.nil?
572
577
 
573
- warn_missed_tick(loop_obj, next_fire, now)
574
- jid = enqueue!(loop_obj)
575
- future = loop_obj.next_fire_after(next_fire, now)
576
- record_fire(loop_obj, jid, now, future)
578
+ future = loop_obj.next_fire_after(slot, now)
579
+ return unless claim_fire?(loop_obj, mark, now, future)
580
+
581
+ warn_missed_tick(loop_obj, slot, now)
582
+ jid = enqueue_claimed!(loop_obj)
583
+ record_history(loop_obj, jid, now)
577
584
  jid
578
585
  end
579
586
 
@@ -590,10 +597,16 @@ module Wurk
590
597
 
591
598
  private
592
599
 
593
- def wait
594
- @mutex.synchronize do
595
- @sleeper.wait(@mutex, @tick_interval) unless @done
596
- end
600
+ # The slot this tick would fire and the token that claims it, or nil when
601
+ # the loop has no resolvable occurrence or none is due yet. The token is
602
+ # the stored `nf` when there is one, otherwise the slot we derived — see
603
+ # the `cron_claim_fire` script for how the two cases differ.
604
+ def due_slot(loop_obj, now)
605
+ prev_fire, next_fire, mark = read_fire_marks(loop_obj.lid)
606
+ next_fire ||= loop_obj.next_fire_at(prev_fire || (now - @tick_interval))
607
+ return if next_fire.nil? || next_fire > now
608
+
609
+ [next_fire, mark || next_fire.to_s]
597
610
  end
598
611
 
599
612
  def warn_missed_tick(loop_obj, expected, now)
@@ -605,6 +618,34 @@ module Wurk
605
618
  )
606
619
  end
607
620
 
621
+ # Compare-and-swap the fire marks: only the tick still looking at the
622
+ # mark it read wins the slot. Manual `#fire` deliberately skips this —
623
+ # an operator-requested run contends with no schedule slot.
624
+ def claim_fire?(loop_obj, mark, fired_at, future)
625
+ @config.redis do |c|
626
+ Wurk::Lua::Loader.eval_cached(
627
+ c, :cron_claim_fire,
628
+ keys: ["#{LOOP_PREFIX}#{loop_obj.lid}"],
629
+ argv: [mark, fired_at.to_s, future.nil? ? '' : future.to_s]
630
+ )
631
+ end == 1
632
+ end
633
+
634
+ # The slot is already claimed, so a failed push loses this occurrence
635
+ # instead of leaving the next tick to fire it again. Ent leader election
636
+ # is explicitly best-effort coordination (sidekiq-ent.md §6), so a missed
637
+ # occurrence is tolerable where a duplicate one is not — log the gap so it
638
+ # stays attributable.
639
+ def enqueue_claimed!(loop_obj)
640
+ enqueue!(loop_obj)
641
+ rescue StandardError => e
642
+ logger.warn(
643
+ "[cron] fire lost lid=#{loop_obj.lid} klass=#{loop_obj.klass} " \
644
+ "mark already advanced, occurrence skipped: #{e.class}: #{e.message}"
645
+ )
646
+ raise
647
+ end
648
+
608
649
  def enqueue!(loop_obj)
609
650
  aj = active_job_class(loop_obj.klass)
610
651
  return enqueue_active_job(aj, loop_obj) if aj
@@ -643,24 +684,40 @@ module Wurk
643
684
  job.provider_job_id if job.respond_to?(:provider_job_id)
644
685
  end
645
686
 
687
+ # Third element is the raw `nf` string — the CAS token for #claim_fire.
688
+ # Comparing the bytes Redis holds, not the parsed integer, keeps the
689
+ # token exact regardless of how the value was written.
646
690
  def read_fire_marks(lid)
647
691
  @config.redis do |c|
648
- vals = c.call('HMGET', "#{LOOP_PREFIX}#{lid}", 'lf', 'nf')
649
- next_fire = vals[1]
692
+ prev_fire, next_fire = c.call('HMGET', "#{LOOP_PREFIX}#{lid}", 'lf', 'nf')
650
693
  # Preserve nil: treat missing or empty 'nf' as nil, not 0.
651
- [vals[0]&.to_i, next_fire.nil? || next_fire.empty? ? nil : next_fire.to_i]
694
+ next_fire = nil if next_fire.nil? || next_fire.empty?
695
+ [prev_fire&.to_i, next_fire&.to_i, next_fire]
652
696
  end
653
697
  end
654
698
 
655
699
  def record_fire(loop_obj, jid, fired_at, future)
656
- history_entry = Wurk.dump_json([fired_at, jid])
700
+ write_fire_marks(loop_obj.lid, fired_at, future)
701
+ record_history(loop_obj, jid, fired_at)
702
+ end
703
+
704
+ # Unconditional mark write, for the manual `#fire` path only. The
705
+ # scheduled path advances the marks through #claim_fire instead.
706
+ def write_fire_marks(lid, fired_at, future)
707
+ key = "#{LOOP_PREFIX}#{lid}"
657
708
  @config.redis do |c|
658
709
  if future.nil?
659
- c.call('HSET', "#{LOOP_PREFIX}#{loop_obj.lid}", 'lf', fired_at.to_s)
660
- c.call('HDEL', "#{LOOP_PREFIX}#{loop_obj.lid}", 'nf')
710
+ c.call('HSET', key, 'lf', fired_at.to_s)
711
+ c.call('HDEL', key, 'nf')
661
712
  else
662
- c.call('HSET', "#{LOOP_PREFIX}#{loop_obj.lid}", 'lf', fired_at.to_s, 'nf', future.to_s)
713
+ c.call('HSET', key, 'lf', fired_at.to_s, 'nf', future.to_s)
663
714
  end
715
+ end
716
+ end
717
+
718
+ def record_history(loop_obj, jid, fired_at)
719
+ history_entry = Wurk.dump_json([fired_at, jid])
720
+ @config.redis do |c|
664
721
  c.call('LPUSH', "#{HISTORY_PREFIX}#{loop_obj.lid}", history_entry)
665
722
  c.call('LTRIM', "#{HISTORY_PREFIX}#{loop_obj.lid}", 0, HISTORY_CAP - 1)
666
723
  end
data/lib/wurk/deploy.rb CHANGED
@@ -1,5 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require_relative 'pool_checkout'
4
+
3
5
  module Wurk
4
6
  # Records deploy markers into Redis so the history pane can overlay
5
7
  # "deployed at" lines onto job-throughput charts. Wire-compatible with
@@ -84,11 +86,11 @@ module Wurk
84
86
  nil
85
87
  end
86
88
 
87
- def with_redis(&)
89
+ def with_redis(idempotent: false, &)
88
90
  if @pool
89
- @pool.with(&)
91
+ PoolCheckout.with(@pool, idempotent, &)
90
92
  else
91
- Wurk.redis(&)
93
+ Wurk.redis(idempotent:, &)
92
94
  end
93
95
  end
94
96
  end
data/lib/wurk/embedded.rb CHANGED
@@ -41,6 +41,12 @@ module Wurk
41
41
  sleep 0.2
42
42
  logger.info { "Wurk running embedded, total process thread count: #{Thread.list.size}" }
43
43
  logger.debug { Thread.list.map(&:name).to_s }
44
+ rescue StandardError
45
+ # The host is left believing Wurk never started, so anything the launcher
46
+ # did get up would keep fetching, beating and campaigning for the leader
47
+ # lock unowned for the life of the process.
48
+ roll_back_partial_boot
49
+ raise
44
50
  end
45
51
 
46
52
  # Stop fetching new work; in-flight jobs continue.
@@ -56,6 +62,13 @@ module Wurk
56
62
 
57
63
  private
58
64
 
65
+ # Guarded so the caller sees why the boot failed, not why the rollback did.
66
+ def roll_back_partial_boot
67
+ stop
68
+ rescue StandardError => e
69
+ handle_exception(e, { context: 'embedded-boot-rollback' })
70
+ end
71
+
59
72
  # Extracted so tests can swap in a fake launcher without monkey-patching
60
73
  # Wurk::Launcher.new globally (which races other parallel tests).
61
74
  def build_launcher
@@ -9,7 +9,7 @@ module Wurk
9
9
  # Orphan reclamation for the reliable fetcher (Pro super_fetch §3.2).
10
10
  #
11
11
  # The Reliable fetcher moves each job from a public queue into a
12
- # per-process private list (`queue:<public>|<host>|<pid>|<idx>`) and
12
+ # per-process private list (`queue:<public>|<host>|<pid>|<nonce>|<idx>`) and
13
13
  # leaves it there until the Processor ACKs. A SIGKILLed or crashed
14
14
  # worker therefore strands its in-flight jobs in private lists that
15
15
  # nobody will ever ACK. The Reaper is the recovery half: it periodically
@@ -17,15 +17,16 @@ module Wurk
17
17
  # moves their jobs back to the public queue so a live worker re-runs them.
18
18
  #
19
19
  # Liveness is decided per owner:
20
- # * same host — the OS is authoritative: `Process.kill(0, pid)`. This
21
- # is instant and ignores a stale `processes` SET entry whose 60s TTL
22
- # hasn't lapsed yet, so a `kill -9`ed sibling is reclaimed the moment
23
- # the supervisor reaps it rather than 60s later. (Pid reuse by an
24
- # unrelated local process is the one blind spot — the supervisor
25
- # respawns with a fresh pid, so it does not arise in practice.)
26
- # * other host — we cannot ping the pid, so we trust the heartbeat:
27
- # the owner is alive iff some live `processes` member (one whose
28
- # `info` hash still exists) shares its `host:pid`. Cross-host reclaim
20
+ # * our own host and nonce — the pid was minted in our PID namespace, so
21
+ # the OS is authoritative: `Process.kill(0, pid)`. This is instant and
22
+ # ignores a stale `processes` SET entry whose 60s TTL hasn't lapsed
23
+ # yet, so a `kill -9`ed sibling is reclaimed the moment the supervisor
24
+ # reaps it rather than 60s later. (Pid reuse by an unrelated process in
25
+ # the same tree is the one blind spot — the supervisor respawns with a
26
+ # fresh pid, so it does not arise in practice.)
27
+ # * any other incarnation — its pid means nothing in our namespace, so we
28
+ # trust the heartbeat: the owner is alive iff its identity is a live
29
+ # `processes` member (one whose `info` hash still exists). Such reclaim
29
30
  # therefore waits out the 60s heartbeat TTL, exactly as the spec says.
30
31
  #
31
32
  # Re-pushed jobs run through Wurk::Middleware::PoisonPill, which caps a
@@ -116,8 +117,8 @@ module Wurk
116
117
  # jobs reclaimed (re-queued or killed). Public so boot paths and tests
117
118
  # can drive a deterministic pass without the cluster lock.
118
119
  def reclaim!
119
- prefixes = live_process_prefixes
120
- served_queues.sum { |public_q| reclaim_queue(public_q, prefixes) }
120
+ owners = live_owners
121
+ served_queues.sum { |public_q| reclaim_queue(public_q, owners) }
121
122
  end
122
123
 
123
124
  # One unguarded full-keyspace sweep: every `queue:*|*` private list, even
@@ -125,10 +126,10 @@ module Wurk
125
126
  # of jobs reclaimed. Public so boot paths and tests can drive it without
126
127
  # the hourly lock.
127
128
  def reclaim_full!
128
- prefixes = live_process_prefixes
129
+ owners = live_owners
129
130
  reclaimed = 0
130
- each_full_private_list do |key, public_q, host, pid|
131
- next if owner_alive?(host, pid, prefixes)
131
+ each_full_private_list do |key, public_q, host, pid, nonce|
132
+ next if owner_alive?(host, pid, nonce, owners)
132
133
 
133
134
  reclaimed += drain(key, public_q)
134
135
  end
@@ -148,39 +149,43 @@ module Wurk
148
149
  end
149
150
 
150
151
  # SCAN for this public queue's private lists, reclaim the orphaned ones.
151
- def reclaim_queue(public_q, prefixes)
152
+ def reclaim_queue(public_q, owners)
152
153
  reclaimed = 0
153
- each_private_list(public_q) do |key, host, pid|
154
- next if owner_alive?(host, pid, prefixes)
154
+ each_private_list(public_q) do |key, host, pid, nonce|
155
+ next if owner_alive?(host, pid, nonce, owners)
155
156
 
156
157
  reclaimed += drain(key, public_q)
157
158
  end
158
159
  reclaimed
159
160
  end
160
161
 
161
- # Yields [private_list_key, host, pid] for each private list of
162
+ # Yields [private_list_key, host, pid, nonce] for each private list of
162
163
  # `public_q`. MATCH `<public_q>|*` matches only this queue's private
163
164
  # lists (public queue keys carry no `|`).
164
165
  def each_private_list(public_q)
165
166
  cursor = '0'
166
167
  loop do
167
- cursor, keys = redis { |c| c.call('SCAN', cursor, 'MATCH', "#{public_q}|*", 'COUNT', SCAN_COUNT) }
168
+ cursor, keys = redis(idempotent: true) do |c|
169
+ c.call('SCAN', cursor, 'MATCH', "#{public_q}|*", 'COUNT', SCAN_COUNT)
170
+ end
168
171
  keys.each do |key|
169
- host, pid = parse_owner(public_q, key)
170
- yield key, host, pid if pid
172
+ host, pid, nonce = parse_owner(public_q, key)
173
+ yield key, host, pid, nonce if pid
171
174
  end
172
175
  break if cursor == '0'
173
176
  end
174
177
  end
175
178
 
176
- # Yields [private_list_key, public_q, host, pid] for every private list in
179
+ # Yields [private_list_key, public_q, host, pid, nonce] for every list in
177
180
  # the keyspace. MATCH `queue:*|*` matches only private lists (public queue
178
181
  # keys carry no `|`); parse_full_key drops anything that isn't a
179
- # well-formed `queue:<public>|<host>|<pid>|<idx>`.
182
+ # well-formed `queue:<public>|<host>|<pid>|<nonce>|<idx>`.
180
183
  def each_full_private_list
181
184
  cursor = '0'
182
185
  loop do
183
- cursor, keys = redis { |c| c.call('SCAN', cursor, 'MATCH', "#{Keys::QUEUE_PREFIX}*|*", 'COUNT', SCAN_COUNT) }
186
+ cursor, keys = redis(idempotent: true) do |c|
187
+ c.call('SCAN', cursor, 'MATCH', "#{Keys::QUEUE_PREFIX}*|*", 'COUNT', SCAN_COUNT)
188
+ end
184
189
  keys.each do |key|
185
190
  parsed = parse_full_key(key)
186
191
  yield key, *parsed if parsed
@@ -189,45 +194,88 @@ module Wurk
189
194
  end
190
195
  end
191
196
 
192
- # `queue:<public>|<host>|<pid>|<idx>` → [public_q, host, pid], parsed from
193
- # the right (pid + idx are integers, host precedes them) so a `|` inside
194
- # the queue name is tolerated. nil when the key isn't a well-formed
195
- # private list.
197
+ # `queue:<public>|<host>|<pid>|<nonce>|<idx>` → [public_q, host, pid,
198
+ # nonce], parsed from the right so a `|` inside the queue name is
199
+ # tolerated. nil when the key isn't a well-formed private list.
200
+ #
201
+ # With no known prefix to split on, both owner shapes are tried in
202
+ # preference order and the first one leaving a real public queue behind
203
+ # wins. The narrow reading is what saves a pre-nonce key from an all-digit
204
+ # host (`queue:q|123456789012|<pid>|<idx>` — a bare Docker hostname is 12
205
+ # hex chars): read wide, its host segment eats the whole queue name.
196
206
  def parse_full_key(key)
197
207
  parts = key.split('|')
198
- return nil if parts.size < 4
199
-
200
- host, pid, idx = parts.last(3)
201
- return nil unless integer?(pid) && integer?(idx)
208
+ owner_tails(parts).each do |host, pid, nonce, width|
209
+ public_q = parts[0...-width].join('|')
210
+ next unless public_q.start_with?(Keys::QUEUE_PREFIX) && public_q != Keys::QUEUE_PREFIX
202
211
 
203
- public_q = parts[0...-3].join('|')
204
- return nil unless public_q.start_with?(Keys::QUEUE_PREFIX) && public_q != Keys::QUEUE_PREFIX
205
-
206
- [public_q, host, pid.to_i]
212
+ return [public_q, host, pid, nonce]
213
+ end
214
+ nil
207
215
  end
208
216
 
209
- # `<public_q>|<host>|<pid>|<idx>` → [host, pid] (pid as Integer), or
210
- # [nil, nil] when the suffix isn't a well-formed `host|pid|idx` triple.
211
- # Splitting the suffix off the known public-queue prefix tolerates a
212
- # `|` inside the queue name itself.
217
+ # `<public_q>|<host>|<pid>|<nonce>|<idx>` → [host, pid, nonce] (pid as
218
+ # Integer), or all-nil when the suffix isn't a well-formed owner tail.
219
+ # Splitting the suffix off the known public-queue prefix tolerates a `|`
220
+ # inside the queue name itself, and leaves the tail unambiguous: exactly
221
+ # 4 segments for the current shape, exactly 3 for the pre-nonce one.
213
222
  def parse_owner(public_q, key)
214
223
  suffix = key.delete_prefix("#{public_q}|")
215
- return [nil, nil] if suffix == key
224
+ return [nil, nil, nil] if suffix == key
225
+
226
+ host, pid, nonce = owner_tails(suffix.split('|')).first
227
+ [host, pid, nonce]
228
+ end
229
+
230
+ # Owner segments of a private-list key, taken from the right, as
231
+ # [host, pid, nonce, segment_count] readings in preference order (empty
232
+ # when nothing parses). The wide shape is preferred: an all-digit nonce is
233
+ # rare but reachable, and reading such a key narrow would take the pid for
234
+ # the host and the nonce for the pid — draining a live owner's list out
235
+ # from under it.
236
+ def owner_tails(parts)
237
+ return [] unless parts.size >= 3 && integer?(parts[-1])
238
+
239
+ [wide_tail(parts), narrow_tail(parts)].compact
240
+ end
216
241
 
217
- host, pid, idx = suffix.split('|')
218
- return [nil, nil] unless host && integer?(pid) && integer?(idx)
242
+ # `<host>|<pid>|<nonce>|<idx>` — the shape every current process writes.
243
+ def wide_tail(parts)
244
+ [parts[-4], parts[-3].to_i, parts[-2], 4] if parts.size >= 4 && integer?(parts[-3])
245
+ end
219
246
 
220
- [host, pid.to_i]
247
+ # `<host>|<pid>|<idx>` — written before the nonce existed. Such a list can
248
+ # still hold a pre-upgrade process's in-flight jobs across a rolling
249
+ # upgrade, so it stays reclaimable even though nothing writes it anymore.
250
+ def narrow_tail(parts)
251
+ [parts[-3], parts[-2].to_i, nil, 3] if integer?(parts[-2])
221
252
  end
222
253
 
223
254
  def integer?(str)
224
255
  str.is_a?(String) && str.match?(/\A\d+\z/)
225
256
  end
226
257
 
227
- def owner_alive?(host, pid, prefixes)
228
- return local_pid_alive?(pid) if host == hostname
229
-
230
- prefixes.include?("#{host}:#{pid}")
258
+ # `Process.kill(0, pid)` answers "does this pid exist *in my PID
259
+ # namespace*", which is the question we're actually asking only when the
260
+ # key was written from that same namespace. A shared hostname does not
261
+ # imply it: a container restarting under a fixed hostname comes back in a
262
+ # fresh namespace where the dead owner's pid is likely taken again (list
263
+ # read as live, jobs stranded forever), and two containers sharing a
264
+ # host's network namespace but not its pid namespace each hold pids the
265
+ # other lacks (live owner read as dead, list drained mid-job, duplicate
266
+ # run). So kill(0) can serve as neither a positive nor a negative signal
267
+ # off our own namespace.
268
+ #
269
+ # The nonce settles it: minted once per process image and inherited across
270
+ # fork, so a key carrying ours provably came from this very process tree.
271
+ # Every other owner goes through the namespace-blind heartbeat — alive iff
272
+ # its identity is a live `processes` member. A pre-nonce key can only be
273
+ # matched on the `<host>:<pid>` prefix of that identity.
274
+ def owner_alive?(host, pid, nonce, owners)
275
+ return local_pid_alive?(pid) if nonce == process_nonce && host == hostname
276
+ return owners.include?("#{host}:#{pid}:#{nonce}") if nonce
277
+
278
+ owners.include?("#{host}:#{pid}")
231
279
  end
232
280
 
233
281
  def local_pid_alive?(pid)
@@ -239,18 +287,21 @@ module Wurk
239
287
  true
240
288
  end
241
289
 
242
- # `host:pid` of every live process — a member of `processes` whose
243
- # `info` hash still exists. A bare SET membership isn't enough: the
244
- # member lingers after its 60s hash TTL until ProcessSet#cleanup prunes
245
- # it, and we must treat that window as dead for cross-host reclaim.
246
- def live_process_prefixes
247
- redis do |conn|
290
+ # Every live process indexed both ways: the full `<host>:<pid>:<nonce>`
291
+ # identity a nonce-bearing private list is matched on, and the
292
+ # `<host>:<pid>` prefix a pre-nonce one has to settle for. Live means a
293
+ # member of `processes` whose `info` hash still exists — a bare SET
294
+ # membership isn't enough, since the member lingers after its 60s hash
295
+ # TTL until ProcessSet#cleanup prunes it and that window must read as
296
+ # dead or the owner's jobs are never reclaimed.
297
+ def live_owners
298
+ redis(idempotent: true) do |conn|
248
299
  members = conn.call('SMEMBERS', Keys::PROCESSES)
249
300
  next ::Set.new if members.empty?
250
301
 
251
302
  infos = conn.pipelined { |pipe| members.each { |m| pipe.call('HGET', m, 'info') } }
252
303
  members.zip(infos).each_with_object(::Set.new) do |(member, info), set|
253
- set << host_pid(member) if info
304
+ set << member << host_pid(member) if info
254
305
  end
255
306
  end
256
307
  end
@@ -265,6 +316,12 @@ module Wurk
265
316
  # poison check, so a crash mid-drain leaves the job safely in the public
266
317
  # queue (at-least-once), never lost. Poison jobs are killed to the dead
267
318
  # set by PoisonPill.track! and then LREM'd out of the public queue.
319
+ #
320
+ # Unlike the fetcher's LMOVE, this one does *not* claim apply-safety: a
321
+ # replay after a lost reply moves the next job and never poison-checks the
322
+ # one already on the public tail, so a job that kills its worker every time
323
+ # would get a free recovery past the cap. A raise instead lands in the
324
+ # rescue below, and the next sweep re-drains what's left.
268
325
  def drain(private_list, public_q)
269
326
  queue_name = public_q.delete_prefix(Keys::QUEUE_PREFIX)
270
327
  count = 0
@@ -5,11 +5,12 @@ require_relative '../component'
5
5
  require_relative '../keys'
6
6
  require_relative '../lua'
7
7
  require_relative '../fetcher'
8
+ require_relative '../middleware/poison_pill'
8
9
 
9
10
  module Wurk
10
11
  class Fetcher
11
12
  # Default fetcher. Each public queue is paired with a per-process
12
- # private list (`queue:<name>|<host>|<pid>|<idx>`); a job is moved
13
+ # private list (`queue:<name>|<host>|<pid>|<nonce>|<idx>`); a job is moved
13
14
  # atomically from the public tail to the private head via LMOVE, and
14
15
  # stays there until the Processor explicitly ACKs (LREM). SIGKILL
15
16
  # between fetch and ack leaves the job in the private list, where the
@@ -30,15 +31,38 @@ module Wurk
30
31
  # Default BLMOVE block timeout; overridable via config.fetch_poll_interval.
31
32
  TIMEOUT = 2
32
33
 
34
+ # Backoff for the quieted short-circuit. Manager#quiet terminates the
35
+ # shared fetcher before it terminates the processors, and Processor#run
36
+ # loops on its *own* flag — so in that window every processor would spin
37
+ # on an instant nil. Kept below Manager::PAUSE_TIME, which #stop sleeps
38
+ # immediately after #quiet, so this pause adds no drain latency.
39
+ QUIET_PAUSE = 0.05
40
+
33
41
  # Carries the public queue key, the raw (still-JSON) job payload,
34
42
  # and the capsule we use to reach Redis. ACK removes from the private
35
43
  # list; requeue pushes back to the public queue head so the job is
36
44
  # next pulled. LREM count=1 is idempotent for our payloads since
37
45
  # each job's JSON contains a unique `jid`.
38
- UnitOfWork = Struct.new(:queue, :job, :config, keyword_init: true) do
46
+ #
47
+ # `jid` is filled in by the Processor once it has parsed the payload —
48
+ # the fetcher never parses. It is only used to retire the job's
49
+ # poison-pill recovery counter, so an ACK without one is still a
50
+ # complete ACK.
51
+ UnitOfWork = Struct.new(:queue, :job, :config, :jid, keyword_init: true) do
52
+ # The counter DEL rides this round trip rather than taking one of its
53
+ # own: the ACK is the only Redis call the success path makes, and a
54
+ # per-job call would be a fetch+execute regression for the sake of a
55
+ # key that exists for roughly no jobs. See Middleware::PoisonPill.
39
56
  def acknowledge
57
+ private_list = Reliable.private_queue_name(queue)
58
+ job_jid = jid.to_s
59
+ return config.redis { |conn| conn.call('LREM', private_list, 1, job) } if job_jid.empty?
60
+
40
61
  config.redis do |conn|
41
- conn.call('LREM', Reliable.private_queue_name(queue), 1, job)
62
+ conn.pipelined do |pipe|
63
+ pipe.call('LREM', private_list, 1, job)
64
+ Middleware::PoisonPill.clear_in(pipe, job_jid)
65
+ end
42
66
  end
43
67
  end
44
68
 
@@ -55,9 +79,16 @@ module Wurk
55
79
  # carrying a back-reference to its parent fetcher. Index defaults to
56
80
  # 0 — we run one fetcher per capsule today. Multi-processor topology
57
81
  # (one private list per processor slot) is a future Manager concern.
82
+ #
83
+ # The nonce marks the incarnation. host+pid alone is ambiguous once PID
84
+ # namespaces are in play: a restarted container reuses both, so the
85
+ # reaper's `kill(0)` liveness check would read a dead owner's list as
86
+ # live (jobs stranded) or a live owner's as dead (job run twice). Keys
87
+ # written before the nonce existed stay reclaimable — Reaper#parse_owner
88
+ # accepts both shapes.
58
89
  def self.private_queue_name(public_queue, index = 0)
59
90
  host = ENV['DYNO'] || Socket.gethostname
60
- "#{public_queue}|#{host}|#{::Process.pid}|#{index}"
91
+ "#{public_queue}|#{host}|#{::Process.pid}|#{Component::PROCESS_NONCE}|#{index}"
61
92
  end
62
93
 
63
94
  def initialize(capsule)
@@ -66,11 +97,26 @@ module Wurk
66
97
  @done = false
67
98
  end
68
99
 
100
+ # Every pass that yields no job has to cost wall-clock time: Processor#run
101
+ # drives `process_one` in a bare `until @done` loop with no pause of its
102
+ # own, so any nil returned instantly turns N processor threads into a hot
103
+ # loop. The blocking BLMOVE pays that cost on the normal empty-queue path;
104
+ # the two short-circuits below have to pay it themselves.
69
105
  def retrieve_work
70
- return nil if @done
106
+ if @done
107
+ sleep QUIET_PAUSE
108
+ return nil
109
+ end
71
110
 
72
111
  queues = queues_cmd
73
- return nil if queues.empty?
112
+ # Nothing fetchable — every queue paused, or none configured. Back off a
113
+ # full poll interval rather than re-running queues_cmd (an SMEMBERS per
114
+ # pass, on the main pool) as fast as the CPU allows. Mirrors Sidekiq's
115
+ # BasicFetch guard, upstream #4825.
116
+ if queues.empty?
117
+ sleep poll_interval
118
+ return nil
119
+ end
74
120
 
75
121
  queues.each do |public_q|
76
122
  uow = lmove(public_q)
@@ -148,12 +194,19 @@ module Wurk
148
194
  # is dominated by the BLMOVE that follows. Returns a Set for O(1)
149
195
  # lookup against the (often weighted-expanded) queue list.
150
196
  def paused_names
151
- config.redis { |conn| conn.call('SMEMBERS', Keys::PAUSED_SET) }.to_set
197
+ config.redis(idempotent: true) { |conn| conn.call('SMEMBERS', Keys::PAUSED_SET) }.to_set
152
198
  end
153
199
 
200
+ # Both LMOVE forms claim apply-safety, so fetch keeps the full
201
+ # connection-blip backoff the F5 split otherwise takes away: a move that
202
+ # applied but whose reply was lost leaves the job in *this* process's
203
+ # private list, un-ACKed — byte-for-byte the state a SIGKILL between fetch
204
+ # and ack leaves behind, which the next boot's Reaper already reclaims. The
205
+ # replay then pulls a different job; nothing duplicates and nothing is
206
+ # lost, at worst one job waits out this process's lifetime.
154
207
  def lmove(public_q)
155
208
  priv = self.class.private_queue_name(public_q)
156
- job = config.redis { |conn| conn.call('LMOVE', public_q, priv, 'RIGHT', 'LEFT') }
209
+ job = config.redis(idempotent: true) { |conn| conn.call('LMOVE', public_q, priv, 'RIGHT', 'LEFT') }
157
210
  job ? UnitOfWork.new(queue: public_q, job: job, config: config) : nil
158
211
  end
159
212
 
@@ -165,7 +218,7 @@ module Wurk
165
218
  # starving the main pool's background loops (#101). Extend the socket
166
219
  # read-timeout one second past BLMOVE's own server-side timeout so the
167
220
  # connection's read timeout can't fire while BLMOVE is legitimately blocked.
168
- job = config.fetch_redis do |conn|
221
+ job = config.fetch_redis(idempotent: true) do |conn|
169
222
  conn.blocking_call(timeout + 1, 'BLMOVE', public_q, priv, 'RIGHT', 'LEFT', timeout)
170
223
  end
171
224
  job ? UnitOfWork.new(queue: public_q, job: job, config: config) : nil