wurk 1.3.0 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +1 -0
- data/lib/wurk/batch/callbacks.rb +82 -12
- data/lib/wurk/batch/death_handler.rb +7 -4
- data/lib/wurk/batch/server_middleware.rb +1 -1
- data/lib/wurk/batch.rb +121 -15
- data/lib/wurk/capsule.rb +5 -4
- data/lib/wurk/cli.rb +48 -14
- data/lib/wurk/client/buffered.rb +193 -43
- data/lib/wurk/client.rb +87 -14
- data/lib/wurk/compat.rb +1 -1
- data/lib/wurk/component.rb +2 -2
- data/lib/wurk/configuration.rb +10 -2
- data/lib/wurk/cron.rb +94 -37
- data/lib/wurk/deploy.rb +5 -3
- data/lib/wurk/embedded.rb +13 -0
- data/lib/wurk/fetcher/reaper.rb +113 -56
- data/lib/wurk/fetcher/reliable.rb +62 -9
- data/lib/wurk/heartbeat.rb +22 -10
- data/lib/wurk/history.rb +13 -1
- data/lib/wurk/launcher.rb +133 -66
- data/lib/wurk/leader.rb +29 -10
- data/lib/wurk/limiter/base.rb +8 -10
- data/lib/wurk/limiter/bucket.rb +1 -1
- data/lib/wurk/limiter/concurrent.rb +27 -22
- data/lib/wurk/limiter/window.rb +13 -11
- data/lib/wurk/limiter.rb +7 -4
- data/lib/wurk/lua.rb +97 -14
- data/lib/wurk/manager.rb +29 -13
- data/lib/wurk/metrics/history.rb +4 -3
- data/lib/wurk/metrics/queue_rollup.rb +13 -1
- data/lib/wurk/metrics/rollup.rb +13 -1
- data/lib/wurk/middleware/interrupt_handler.rb +7 -6
- data/lib/wurk/middleware/poison_pill.rb +70 -29
- data/lib/wurk/middleware.rb +2 -2
- data/lib/wurk/pool_checkout.rb +29 -0
- data/lib/wurk/process_set.rb +10 -5
- data/lib/wurk/processor.rb +6 -0
- data/lib/wurk/profiler.rb +3 -2
- data/lib/wurk/queue.rb +10 -7
- data/lib/wurk/rails_boot.rb +38 -7
- data/lib/wurk/redis_client_adapter.rb +48 -4
- data/lib/wurk/redis_options.rb +142 -0
- data/lib/wurk/redis_pool.rb +102 -39
- data/lib/wurk/scheduled.rb +30 -2
- data/lib/wurk/stats.rb +14 -9
- data/lib/wurk/swarm/child_boot.rb +12 -0
- data/lib/wurk/swarm.rb +174 -33
- data/lib/wurk/timer_loop.rb +14 -0
- data/lib/wurk/version.rb +1 -1
- data/lib/wurk/web/enterprise.rb +58 -6
- data/lib/wurk/web/extension.rb +1 -1
- data/lib/wurk/web/search.rb +5 -3
- data/lib/wurk.rb +53 -2
- data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-OmC5NPgw.js} +1 -1
- data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-CIpai7St.js} +1 -1
- data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-A_kwSR6Q.js} +1 -1
- data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-BG7HTqlp.js} +1 -1
- data/vendor/assets/dashboard/assets/{Dashboard-B9rOrkzk.js → Dashboard-A_ToqHoo.js} +1 -1
- data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-8J21jMyK.js} +1 -1
- data/vendor/assets/dashboard/assets/Extension-B4Q9FIQu.js +1 -0
- data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-Fh_Ae7UW.js} +1 -1
- data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-Ceng0PMB.js} +1 -1
- data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-CruDWvNZ.js} +1 -1
- data/vendor/assets/dashboard/assets/{Metrics-BBTDxcaE.js → Metrics-CIT7VCoN.js} +1 -1
- data/vendor/assets/dashboard/assets/Modal-CN3rdKA_.js +1 -0
- data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-D86FYohJ.js} +1 -1
- data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-Bz1O1D-i.js} +1 -1
- data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-B6h2akTu.js} +1 -1
- data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-OOu22e5s.js} +1 -1
- data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
- data/vendor/assets/dashboard/assets/index-D_lSDwKw.js +141 -0
- data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-dVPGEWzn.js} +1 -1
- data/vendor/assets/dashboard/index.html +2 -2
- data/vendor/assets/dashboard/wurk-manifest.json +2 -2
- metadata +22 -20
- data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
- data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
- data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
- data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
data/lib/wurk/cron.rb
CHANGED
|
@@ -6,6 +6,7 @@ require 'time'
|
|
|
6
6
|
require_relative 'component'
|
|
7
7
|
require_relative 'client'
|
|
8
8
|
require_relative 'leader'
|
|
9
|
+
require_relative 'timer_loop'
|
|
9
10
|
|
|
10
11
|
module Wurk
|
|
11
12
|
# Sidekiq Enterprise periodic jobs. Pure leader-driven cron — only the
|
|
@@ -477,8 +478,8 @@ module Wurk
|
|
|
477
478
|
|
|
478
479
|
private
|
|
479
480
|
|
|
480
|
-
def redis(&)
|
|
481
|
-
@config ? @config.redis(&) : Wurk.redis(&)
|
|
481
|
+
def redis(idempotent: false, &)
|
|
482
|
+
@config ? @config.redis(idempotent:, &) : Wurk.redis(idempotent:, &)
|
|
482
483
|
end
|
|
483
484
|
end
|
|
484
485
|
|
|
@@ -520,34 +521,35 @@ module Wurk
|
|
|
520
521
|
|
|
521
522
|
def initialize(config)
|
|
522
523
|
@config = config
|
|
523
|
-
@done = false
|
|
524
|
-
@mutex = ::Mutex.new
|
|
525
|
-
@sleeper = ::ConditionVariable.new
|
|
526
524
|
@client = Client.new(config: config)
|
|
527
525
|
@thread = nil
|
|
528
526
|
# Operators never need to touch this; integration tests shrink it so a
|
|
529
527
|
# due loop fires within the test window instead of waiting a full minute.
|
|
530
528
|
@tick_interval = config[:cron_tick_interval] || DEFAULT_TICK_SECONDS
|
|
529
|
+
@timer = TimerLoop.new(@tick_interval)
|
|
531
530
|
end
|
|
532
531
|
|
|
532
|
+
# TimerLoop waits one interval before the first tick: don't fire a
|
|
533
|
+
# catch-up burst the instant we boot (the leader is barely settled), and
|
|
534
|
+
# let a short-lived process exit without ticking at all.
|
|
533
535
|
def start
|
|
534
|
-
@
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
wait
|
|
539
|
-
until @done
|
|
540
|
-
tick
|
|
541
|
-
wait
|
|
542
|
-
end
|
|
543
|
-
end
|
|
536
|
+
return @thread if @thread
|
|
537
|
+
|
|
538
|
+
@timer.reset
|
|
539
|
+
@thread = safe_thread('cron-poller') { @timer.run { tick } }
|
|
544
540
|
end
|
|
545
541
|
|
|
542
|
+
# Blocks until the thread is really gone: the launcher releases the
|
|
543
|
+
# cluster lock immediately after this returns, and a tick still in flight
|
|
544
|
+
# would enqueue loops the next leader is about to fire itself.
|
|
545
|
+
#
|
|
546
|
+
# Cleared only on a confirmed join (Thread#join returns nil on timeout):
|
|
547
|
+
# a wedged thread must stay tracked so #start's guard returns it instead
|
|
548
|
+
# of calling @timer.reset, which would un-terminate the loop it is still
|
|
549
|
+
# inside and leave two tick threads double-enqueuing the same loops.
|
|
546
550
|
def terminate
|
|
547
|
-
@
|
|
548
|
-
|
|
549
|
-
@sleeper.signal
|
|
550
|
-
end
|
|
551
|
+
@timer.terminate
|
|
552
|
+
@thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
|
|
551
553
|
end
|
|
552
554
|
|
|
553
555
|
# Leader-gated by the single cluster lock (Component#leader? reads
|
|
@@ -562,18 +564,23 @@ module Wurk
|
|
|
562
564
|
handle_exception(e, { context: 'cron-poller' })
|
|
563
565
|
end
|
|
564
566
|
|
|
567
|
+
# The marks advance *before* the push, via CAS: the leader gate is a
|
|
568
|
+
# cached read, so a second poller can reach the same due loop for a few
|
|
569
|
+
# seconds after a handover. Losing the CAS means another tick owns this
|
|
570
|
+
# slot — enqueue nothing.
|
|
565
571
|
def enqueue_if_due(loop_obj)
|
|
566
572
|
return if loop_obj.paused?
|
|
567
573
|
|
|
568
574
|
now = ::Time.now.to_i
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
return if next_fire.nil? || next_fire > now
|
|
575
|
+
slot, mark = due_slot(loop_obj, now)
|
|
576
|
+
return if slot.nil?
|
|
572
577
|
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
578
|
+
future = loop_obj.next_fire_after(slot, now)
|
|
579
|
+
return unless claim_fire?(loop_obj, mark, now, future)
|
|
580
|
+
|
|
581
|
+
warn_missed_tick(loop_obj, slot, now)
|
|
582
|
+
jid = enqueue_claimed!(loop_obj)
|
|
583
|
+
record_history(loop_obj, jid, now)
|
|
577
584
|
jid
|
|
578
585
|
end
|
|
579
586
|
|
|
@@ -590,10 +597,16 @@ module Wurk
|
|
|
590
597
|
|
|
591
598
|
private
|
|
592
599
|
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
600
|
+
# The slot this tick would fire and the token that claims it, or nil when
|
|
601
|
+
# the loop has no resolvable occurrence or none is due yet. The token is
|
|
602
|
+
# the stored `nf` when there is one, otherwise the slot we derived — see
|
|
603
|
+
# the `cron_claim_fire` script for how the two cases differ.
|
|
604
|
+
def due_slot(loop_obj, now)
|
|
605
|
+
prev_fire, next_fire, mark = read_fire_marks(loop_obj.lid)
|
|
606
|
+
next_fire ||= loop_obj.next_fire_at(prev_fire || (now - @tick_interval))
|
|
607
|
+
return if next_fire.nil? || next_fire > now
|
|
608
|
+
|
|
609
|
+
[next_fire, mark || next_fire.to_s]
|
|
597
610
|
end
|
|
598
611
|
|
|
599
612
|
def warn_missed_tick(loop_obj, expected, now)
|
|
@@ -605,6 +618,34 @@ module Wurk
|
|
|
605
618
|
)
|
|
606
619
|
end
|
|
607
620
|
|
|
621
|
+
# Compare-and-swap the fire marks: only the tick still looking at the
|
|
622
|
+
# mark it read wins the slot. Manual `#fire` deliberately skips this —
|
|
623
|
+
# an operator-requested run contends with no schedule slot.
|
|
624
|
+
def claim_fire?(loop_obj, mark, fired_at, future)
|
|
625
|
+
@config.redis do |c|
|
|
626
|
+
Wurk::Lua::Loader.eval_cached(
|
|
627
|
+
c, :cron_claim_fire,
|
|
628
|
+
keys: ["#{LOOP_PREFIX}#{loop_obj.lid}"],
|
|
629
|
+
argv: [mark, fired_at.to_s, future.nil? ? '' : future.to_s]
|
|
630
|
+
)
|
|
631
|
+
end == 1
|
|
632
|
+
end
|
|
633
|
+
|
|
634
|
+
# The slot is already claimed, so a failed push loses this occurrence
|
|
635
|
+
# instead of leaving the next tick to fire it again. Ent leader election
|
|
636
|
+
# is explicitly best-effort coordination (sidekiq-ent.md §6), so a missed
|
|
637
|
+
# occurrence is tolerable where a duplicate one is not — log the gap so it
|
|
638
|
+
# stays attributable.
|
|
639
|
+
def enqueue_claimed!(loop_obj)
|
|
640
|
+
enqueue!(loop_obj)
|
|
641
|
+
rescue StandardError => e
|
|
642
|
+
logger.warn(
|
|
643
|
+
"[cron] fire lost lid=#{loop_obj.lid} klass=#{loop_obj.klass} " \
|
|
644
|
+
"mark already advanced, occurrence skipped: #{e.class}: #{e.message}"
|
|
645
|
+
)
|
|
646
|
+
raise
|
|
647
|
+
end
|
|
648
|
+
|
|
608
649
|
def enqueue!(loop_obj)
|
|
609
650
|
aj = active_job_class(loop_obj.klass)
|
|
610
651
|
return enqueue_active_job(aj, loop_obj) if aj
|
|
@@ -643,24 +684,40 @@ module Wurk
|
|
|
643
684
|
job.provider_job_id if job.respond_to?(:provider_job_id)
|
|
644
685
|
end
|
|
645
686
|
|
|
687
|
+
# Third element is the raw `nf` string — the CAS token for #claim_fire.
|
|
688
|
+
# Comparing the bytes Redis holds, not the parsed integer, keeps the
|
|
689
|
+
# token exact regardless of how the value was written.
|
|
646
690
|
def read_fire_marks(lid)
|
|
647
691
|
@config.redis do |c|
|
|
648
|
-
|
|
649
|
-
next_fire = vals[1]
|
|
692
|
+
prev_fire, next_fire = c.call('HMGET', "#{LOOP_PREFIX}#{lid}", 'lf', 'nf')
|
|
650
693
|
# Preserve nil: treat missing or empty 'nf' as nil, not 0.
|
|
651
|
-
|
|
694
|
+
next_fire = nil if next_fire.nil? || next_fire.empty?
|
|
695
|
+
[prev_fire&.to_i, next_fire&.to_i, next_fire]
|
|
652
696
|
end
|
|
653
697
|
end
|
|
654
698
|
|
|
655
699
|
def record_fire(loop_obj, jid, fired_at, future)
|
|
656
|
-
|
|
700
|
+
write_fire_marks(loop_obj.lid, fired_at, future)
|
|
701
|
+
record_history(loop_obj, jid, fired_at)
|
|
702
|
+
end
|
|
703
|
+
|
|
704
|
+
# Unconditional mark write, for the manual `#fire` path only. The
|
|
705
|
+
# scheduled path advances the marks through #claim_fire instead.
|
|
706
|
+
def write_fire_marks(lid, fired_at, future)
|
|
707
|
+
key = "#{LOOP_PREFIX}#{lid}"
|
|
657
708
|
@config.redis do |c|
|
|
658
709
|
if future.nil?
|
|
659
|
-
c.call('HSET',
|
|
660
|
-
c.call('HDEL',
|
|
710
|
+
c.call('HSET', key, 'lf', fired_at.to_s)
|
|
711
|
+
c.call('HDEL', key, 'nf')
|
|
661
712
|
else
|
|
662
|
-
c.call('HSET',
|
|
713
|
+
c.call('HSET', key, 'lf', fired_at.to_s, 'nf', future.to_s)
|
|
663
714
|
end
|
|
715
|
+
end
|
|
716
|
+
end
|
|
717
|
+
|
|
718
|
+
def record_history(loop_obj, jid, fired_at)
|
|
719
|
+
history_entry = Wurk.dump_json([fired_at, jid])
|
|
720
|
+
@config.redis do |c|
|
|
664
721
|
c.call('LPUSH', "#{HISTORY_PREFIX}#{loop_obj.lid}", history_entry)
|
|
665
722
|
c.call('LTRIM', "#{HISTORY_PREFIX}#{loop_obj.lid}", 0, HISTORY_CAP - 1)
|
|
666
723
|
end
|
data/lib/wurk/deploy.rb
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require_relative 'pool_checkout'
|
|
4
|
+
|
|
3
5
|
module Wurk
|
|
4
6
|
# Records deploy markers into Redis so the history pane can overlay
|
|
5
7
|
# "deployed at" lines onto job-throughput charts. Wire-compatible with
|
|
@@ -84,11 +86,11 @@ module Wurk
|
|
|
84
86
|
nil
|
|
85
87
|
end
|
|
86
88
|
|
|
87
|
-
def with_redis(&)
|
|
89
|
+
def with_redis(idempotent: false, &)
|
|
88
90
|
if @pool
|
|
89
|
-
|
|
91
|
+
PoolCheckout.with(@pool, idempotent, &)
|
|
90
92
|
else
|
|
91
|
-
Wurk.redis(&)
|
|
93
|
+
Wurk.redis(idempotent:, &)
|
|
92
94
|
end
|
|
93
95
|
end
|
|
94
96
|
end
|
data/lib/wurk/embedded.rb
CHANGED
|
@@ -41,6 +41,12 @@ module Wurk
|
|
|
41
41
|
sleep 0.2
|
|
42
42
|
logger.info { "Wurk running embedded, total process thread count: #{Thread.list.size}" }
|
|
43
43
|
logger.debug { Thread.list.map(&:name).to_s }
|
|
44
|
+
rescue StandardError
|
|
45
|
+
# The host is left believing Wurk never started, so anything the launcher
|
|
46
|
+
# did get up would keep fetching, beating and campaigning for the leader
|
|
47
|
+
# lock unowned for the life of the process.
|
|
48
|
+
roll_back_partial_boot
|
|
49
|
+
raise
|
|
44
50
|
end
|
|
45
51
|
|
|
46
52
|
# Stop fetching new work; in-flight jobs continue.
|
|
@@ -56,6 +62,13 @@ module Wurk
|
|
|
56
62
|
|
|
57
63
|
private
|
|
58
64
|
|
|
65
|
+
# Guarded so the caller sees why the boot failed, not why the rollback did.
|
|
66
|
+
def roll_back_partial_boot
|
|
67
|
+
stop
|
|
68
|
+
rescue StandardError => e
|
|
69
|
+
handle_exception(e, { context: 'embedded-boot-rollback' })
|
|
70
|
+
end
|
|
71
|
+
|
|
59
72
|
# Extracted so tests can swap in a fake launcher without monkey-patching
|
|
60
73
|
# Wurk::Launcher.new globally (which races other parallel tests).
|
|
61
74
|
def build_launcher
|
data/lib/wurk/fetcher/reaper.rb
CHANGED
|
@@ -9,7 +9,7 @@ module Wurk
|
|
|
9
9
|
# Orphan reclamation for the reliable fetcher (Pro super_fetch §3.2).
|
|
10
10
|
#
|
|
11
11
|
# The Reliable fetcher moves each job from a public queue into a
|
|
12
|
-
# per-process private list (`queue:<public>|<host>|<pid>|<idx>`) and
|
|
12
|
+
# per-process private list (`queue:<public>|<host>|<pid>|<nonce>|<idx>`) and
|
|
13
13
|
# leaves it there until the Processor ACKs. A SIGKILLed or crashed
|
|
14
14
|
# worker therefore strands its in-flight jobs in private lists that
|
|
15
15
|
# nobody will ever ACK. The Reaper is the recovery half: it periodically
|
|
@@ -17,15 +17,16 @@ module Wurk
|
|
|
17
17
|
# moves their jobs back to the public queue so a live worker re-runs them.
|
|
18
18
|
#
|
|
19
19
|
# Liveness is decided per owner:
|
|
20
|
-
# *
|
|
21
|
-
#
|
|
22
|
-
#
|
|
23
|
-
#
|
|
24
|
-
#
|
|
25
|
-
#
|
|
26
|
-
#
|
|
27
|
-
#
|
|
28
|
-
#
|
|
20
|
+
# * our own host and nonce — the pid was minted in our PID namespace, so
|
|
21
|
+
# the OS is authoritative: `Process.kill(0, pid)`. This is instant and
|
|
22
|
+
# ignores a stale `processes` SET entry whose 60s TTL hasn't lapsed
|
|
23
|
+
# yet, so a `kill -9`ed sibling is reclaimed the moment the supervisor
|
|
24
|
+
# reaps it rather than 60s later. (Pid reuse by an unrelated process in
|
|
25
|
+
# the same tree is the one blind spot — the supervisor respawns with a
|
|
26
|
+
# fresh pid, so it does not arise in practice.)
|
|
27
|
+
# * any other incarnation — its pid means nothing in our namespace, so we
|
|
28
|
+
# trust the heartbeat: the owner is alive iff its identity is a live
|
|
29
|
+
# `processes` member (one whose `info` hash still exists). Such reclaim
|
|
29
30
|
# therefore waits out the 60s heartbeat TTL, exactly as the spec says.
|
|
30
31
|
#
|
|
31
32
|
# Re-pushed jobs run through Wurk::Middleware::PoisonPill, which caps a
|
|
@@ -116,8 +117,8 @@ module Wurk
|
|
|
116
117
|
# jobs reclaimed (re-queued or killed). Public so boot paths and tests
|
|
117
118
|
# can drive a deterministic pass without the cluster lock.
|
|
118
119
|
def reclaim!
|
|
119
|
-
|
|
120
|
-
served_queues.sum { |public_q| reclaim_queue(public_q,
|
|
120
|
+
owners = live_owners
|
|
121
|
+
served_queues.sum { |public_q| reclaim_queue(public_q, owners) }
|
|
121
122
|
end
|
|
122
123
|
|
|
123
124
|
# One unguarded full-keyspace sweep: every `queue:*|*` private list, even
|
|
@@ -125,10 +126,10 @@ module Wurk
|
|
|
125
126
|
# of jobs reclaimed. Public so boot paths and tests can drive it without
|
|
126
127
|
# the hourly lock.
|
|
127
128
|
def reclaim_full!
|
|
128
|
-
|
|
129
|
+
owners = live_owners
|
|
129
130
|
reclaimed = 0
|
|
130
|
-
each_full_private_list do |key, public_q, host, pid|
|
|
131
|
-
next if owner_alive?(host, pid,
|
|
131
|
+
each_full_private_list do |key, public_q, host, pid, nonce|
|
|
132
|
+
next if owner_alive?(host, pid, nonce, owners)
|
|
132
133
|
|
|
133
134
|
reclaimed += drain(key, public_q)
|
|
134
135
|
end
|
|
@@ -148,39 +149,43 @@ module Wurk
|
|
|
148
149
|
end
|
|
149
150
|
|
|
150
151
|
# SCAN for this public queue's private lists, reclaim the orphaned ones.
|
|
151
|
-
def reclaim_queue(public_q,
|
|
152
|
+
def reclaim_queue(public_q, owners)
|
|
152
153
|
reclaimed = 0
|
|
153
|
-
each_private_list(public_q) do |key, host, pid|
|
|
154
|
-
next if owner_alive?(host, pid,
|
|
154
|
+
each_private_list(public_q) do |key, host, pid, nonce|
|
|
155
|
+
next if owner_alive?(host, pid, nonce, owners)
|
|
155
156
|
|
|
156
157
|
reclaimed += drain(key, public_q)
|
|
157
158
|
end
|
|
158
159
|
reclaimed
|
|
159
160
|
end
|
|
160
161
|
|
|
161
|
-
# Yields [private_list_key, host, pid] for each private list of
|
|
162
|
+
# Yields [private_list_key, host, pid, nonce] for each private list of
|
|
162
163
|
# `public_q`. MATCH `<public_q>|*` matches only this queue's private
|
|
163
164
|
# lists (public queue keys carry no `|`).
|
|
164
165
|
def each_private_list(public_q)
|
|
165
166
|
cursor = '0'
|
|
166
167
|
loop do
|
|
167
|
-
cursor, keys = redis
|
|
168
|
+
cursor, keys = redis(idempotent: true) do |c|
|
|
169
|
+
c.call('SCAN', cursor, 'MATCH', "#{public_q}|*", 'COUNT', SCAN_COUNT)
|
|
170
|
+
end
|
|
168
171
|
keys.each do |key|
|
|
169
|
-
host, pid = parse_owner(public_q, key)
|
|
170
|
-
yield key, host, pid if pid
|
|
172
|
+
host, pid, nonce = parse_owner(public_q, key)
|
|
173
|
+
yield key, host, pid, nonce if pid
|
|
171
174
|
end
|
|
172
175
|
break if cursor == '0'
|
|
173
176
|
end
|
|
174
177
|
end
|
|
175
178
|
|
|
176
|
-
# Yields [private_list_key, public_q, host, pid] for every
|
|
179
|
+
# Yields [private_list_key, public_q, host, pid, nonce] for every list in
|
|
177
180
|
# the keyspace. MATCH `queue:*|*` matches only private lists (public queue
|
|
178
181
|
# keys carry no `|`); parse_full_key drops anything that isn't a
|
|
179
|
-
# well-formed `queue:<public>|<host>|<pid>|<idx>`.
|
|
182
|
+
# well-formed `queue:<public>|<host>|<pid>|<nonce>|<idx>`.
|
|
180
183
|
def each_full_private_list
|
|
181
184
|
cursor = '0'
|
|
182
185
|
loop do
|
|
183
|
-
cursor, keys = redis
|
|
186
|
+
cursor, keys = redis(idempotent: true) do |c|
|
|
187
|
+
c.call('SCAN', cursor, 'MATCH', "#{Keys::QUEUE_PREFIX}*|*", 'COUNT', SCAN_COUNT)
|
|
188
|
+
end
|
|
184
189
|
keys.each do |key|
|
|
185
190
|
parsed = parse_full_key(key)
|
|
186
191
|
yield key, *parsed if parsed
|
|
@@ -189,45 +194,88 @@ module Wurk
|
|
|
189
194
|
end
|
|
190
195
|
end
|
|
191
196
|
|
|
192
|
-
# `queue:<public>|<host>|<pid>|<idx>` → [public_q, host, pid
|
|
193
|
-
#
|
|
194
|
-
#
|
|
195
|
-
#
|
|
197
|
+
# `queue:<public>|<host>|<pid>|<nonce>|<idx>` → [public_q, host, pid,
|
|
198
|
+
# nonce], parsed from the right so a `|` inside the queue name is
|
|
199
|
+
# tolerated. nil when the key isn't a well-formed private list.
|
|
200
|
+
#
|
|
201
|
+
# With no known prefix to split on, both owner shapes are tried in
|
|
202
|
+
# preference order and the first one leaving a real public queue behind
|
|
203
|
+
# wins. The narrow reading is what saves a pre-nonce key from an all-digit
|
|
204
|
+
# host (`queue:q|123456789012|<pid>|<idx>` — a bare Docker hostname is 12
|
|
205
|
+
# hex chars): read wide, its host segment eats the whole queue name.
|
|
196
206
|
def parse_full_key(key)
|
|
197
207
|
parts = key.split('|')
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
return nil unless integer?(pid) && integer?(idx)
|
|
208
|
+
owner_tails(parts).each do |host, pid, nonce, width|
|
|
209
|
+
public_q = parts[0...-width].join('|')
|
|
210
|
+
next unless public_q.start_with?(Keys::QUEUE_PREFIX) && public_q != Keys::QUEUE_PREFIX
|
|
202
211
|
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
[public_q, host, pid.to_i]
|
|
212
|
+
return [public_q, host, pid, nonce]
|
|
213
|
+
end
|
|
214
|
+
nil
|
|
207
215
|
end
|
|
208
216
|
|
|
209
|
-
# `<public_q>|<host>|<pid>|<idx>` → [host, pid] (pid as
|
|
210
|
-
#
|
|
211
|
-
# Splitting the suffix off the known public-queue prefix tolerates a
|
|
212
|
-
#
|
|
217
|
+
# `<public_q>|<host>|<pid>|<nonce>|<idx>` → [host, pid, nonce] (pid as
|
|
218
|
+
# Integer), or all-nil when the suffix isn't a well-formed owner tail.
|
|
219
|
+
# Splitting the suffix off the known public-queue prefix tolerates a `|`
|
|
220
|
+
# inside the queue name itself, and leaves the tail unambiguous: exactly
|
|
221
|
+
# 4 segments for the current shape, exactly 3 for the pre-nonce one.
|
|
213
222
|
def parse_owner(public_q, key)
|
|
214
223
|
suffix = key.delete_prefix("#{public_q}|")
|
|
215
|
-
return [nil, nil] if suffix == key
|
|
224
|
+
return [nil, nil, nil] if suffix == key
|
|
225
|
+
|
|
226
|
+
host, pid, nonce = owner_tails(suffix.split('|')).first
|
|
227
|
+
[host, pid, nonce]
|
|
228
|
+
end
|
|
229
|
+
|
|
230
|
+
# Owner segments of a private-list key, taken from the right, as
|
|
231
|
+
# [host, pid, nonce, segment_count] readings in preference order (empty
|
|
232
|
+
# when nothing parses). The wide shape is preferred: an all-digit nonce is
|
|
233
|
+
# rare but reachable, and reading such a key narrow would take the pid for
|
|
234
|
+
# the host and the nonce for the pid — draining a live owner's list out
|
|
235
|
+
# from under it.
|
|
236
|
+
def owner_tails(parts)
|
|
237
|
+
return [] unless parts.size >= 3 && integer?(parts[-1])
|
|
238
|
+
|
|
239
|
+
[wide_tail(parts), narrow_tail(parts)].compact
|
|
240
|
+
end
|
|
216
241
|
|
|
217
|
-
|
|
218
|
-
|
|
242
|
+
# `<host>|<pid>|<nonce>|<idx>` — the shape every current process writes.
|
|
243
|
+
def wide_tail(parts)
|
|
244
|
+
[parts[-4], parts[-3].to_i, parts[-2], 4] if parts.size >= 4 && integer?(parts[-3])
|
|
245
|
+
end
|
|
219
246
|
|
|
220
|
-
|
|
247
|
+
# `<host>|<pid>|<idx>` — written before the nonce existed. Such a list can
|
|
248
|
+
# still hold a pre-upgrade process's in-flight jobs across a rolling
|
|
249
|
+
# upgrade, so it stays reclaimable even though nothing writes it anymore.
|
|
250
|
+
def narrow_tail(parts)
|
|
251
|
+
[parts[-3], parts[-2].to_i, nil, 3] if integer?(parts[-2])
|
|
221
252
|
end
|
|
222
253
|
|
|
223
254
|
def integer?(str)
|
|
224
255
|
str.is_a?(String) && str.match?(/\A\d+\z/)
|
|
225
256
|
end
|
|
226
257
|
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
258
|
+
# `Process.kill(0, pid)` answers "does this pid exist *in my PID
|
|
259
|
+
# namespace*", which is the question we're actually asking only when the
|
|
260
|
+
# key was written from that same namespace. A shared hostname does not
|
|
261
|
+
# imply it: a container restarting under a fixed hostname comes back in a
|
|
262
|
+
# fresh namespace where the dead owner's pid is likely taken again (list
|
|
263
|
+
# read as live, jobs stranded forever), and two containers sharing a
|
|
264
|
+
# host's network namespace but not its pid namespace each hold pids the
|
|
265
|
+
# other lacks (live owner read as dead, list drained mid-job, duplicate
|
|
266
|
+
# run). So kill(0) can serve as neither a positive nor a negative signal
|
|
267
|
+
# off our own namespace.
|
|
268
|
+
#
|
|
269
|
+
# The nonce settles it: minted once per process image and inherited across
|
|
270
|
+
# fork, so a key carrying ours provably came from this very process tree.
|
|
271
|
+
# Every other owner goes through the namespace-blind heartbeat — alive iff
|
|
272
|
+
# its identity is a live `processes` member. A pre-nonce key can only be
|
|
273
|
+
# matched on the `<host>:<pid>` prefix of that identity.
|
|
274
|
+
def owner_alive?(host, pid, nonce, owners)
|
|
275
|
+
return local_pid_alive?(pid) if nonce == process_nonce && host == hostname
|
|
276
|
+
return owners.include?("#{host}:#{pid}:#{nonce}") if nonce
|
|
277
|
+
|
|
278
|
+
owners.include?("#{host}:#{pid}")
|
|
231
279
|
end
|
|
232
280
|
|
|
233
281
|
def local_pid_alive?(pid)
|
|
@@ -239,18 +287,21 @@ module Wurk
|
|
|
239
287
|
true
|
|
240
288
|
end
|
|
241
289
|
|
|
242
|
-
#
|
|
243
|
-
#
|
|
244
|
-
#
|
|
245
|
-
#
|
|
246
|
-
|
|
247
|
-
|
|
290
|
+
# Every live process indexed both ways: the full `<host>:<pid>:<nonce>`
|
|
291
|
+
# identity a nonce-bearing private list is matched on, and the
|
|
292
|
+
# `<host>:<pid>` prefix a pre-nonce one has to settle for. Live means a
|
|
293
|
+
# member of `processes` whose `info` hash still exists — a bare SET
|
|
294
|
+
# membership isn't enough, since the member lingers after its 60s hash
|
|
295
|
+
# TTL until ProcessSet#cleanup prunes it and that window must read as
|
|
296
|
+
# dead or the owner's jobs are never reclaimed.
|
|
297
|
+
def live_owners
|
|
298
|
+
redis(idempotent: true) do |conn|
|
|
248
299
|
members = conn.call('SMEMBERS', Keys::PROCESSES)
|
|
249
300
|
next ::Set.new if members.empty?
|
|
250
301
|
|
|
251
302
|
infos = conn.pipelined { |pipe| members.each { |m| pipe.call('HGET', m, 'info') } }
|
|
252
303
|
members.zip(infos).each_with_object(::Set.new) do |(member, info), set|
|
|
253
|
-
set << host_pid(member) if info
|
|
304
|
+
set << member << host_pid(member) if info
|
|
254
305
|
end
|
|
255
306
|
end
|
|
256
307
|
end
|
|
@@ -265,6 +316,12 @@ module Wurk
|
|
|
265
316
|
# poison check, so a crash mid-drain leaves the job safely in the public
|
|
266
317
|
# queue (at-least-once), never lost. Poison jobs are killed to the dead
|
|
267
318
|
# set by PoisonPill.track! and then LREM'd out of the public queue.
|
|
319
|
+
#
|
|
320
|
+
# Unlike the fetcher's LMOVE, this one does *not* claim apply-safety: a
|
|
321
|
+
# replay after a lost reply moves the next job and never poison-checks the
|
|
322
|
+
# one already on the public tail, so a job that kills its worker every time
|
|
323
|
+
# would get a free recovery past the cap. A raise instead lands in the
|
|
324
|
+
# rescue below, and the next sweep re-drains what's left.
|
|
268
325
|
def drain(private_list, public_q)
|
|
269
326
|
queue_name = public_q.delete_prefix(Keys::QUEUE_PREFIX)
|
|
270
327
|
count = 0
|
|
@@ -5,11 +5,12 @@ require_relative '../component'
|
|
|
5
5
|
require_relative '../keys'
|
|
6
6
|
require_relative '../lua'
|
|
7
7
|
require_relative '../fetcher'
|
|
8
|
+
require_relative '../middleware/poison_pill'
|
|
8
9
|
|
|
9
10
|
module Wurk
|
|
10
11
|
class Fetcher
|
|
11
12
|
# Default fetcher. Each public queue is paired with a per-process
|
|
12
|
-
# private list (`queue:<name>|<host>|<pid>|<idx>`); a job is moved
|
|
13
|
+
# private list (`queue:<name>|<host>|<pid>|<nonce>|<idx>`); a job is moved
|
|
13
14
|
# atomically from the public tail to the private head via LMOVE, and
|
|
14
15
|
# stays there until the Processor explicitly ACKs (LREM). SIGKILL
|
|
15
16
|
# between fetch and ack leaves the job in the private list, where the
|
|
@@ -30,15 +31,38 @@ module Wurk
|
|
|
30
31
|
# Default BLMOVE block timeout; overridable via config.fetch_poll_interval.
|
|
31
32
|
TIMEOUT = 2
|
|
32
33
|
|
|
34
|
+
# Backoff for the quieted short-circuit. Manager#quiet terminates the
|
|
35
|
+
# shared fetcher before it terminates the processors, and Processor#run
|
|
36
|
+
# loops on its *own* flag — so in that window every processor would spin
|
|
37
|
+
# on an instant nil. Kept below Manager::PAUSE_TIME, which #stop sleeps
|
|
38
|
+
# immediately after #quiet, so this pause adds no drain latency.
|
|
39
|
+
QUIET_PAUSE = 0.05
|
|
40
|
+
|
|
33
41
|
# Carries the public queue key, the raw (still-JSON) job payload,
|
|
34
42
|
# and the capsule we use to reach Redis. ACK removes from the private
|
|
35
43
|
# list; requeue pushes back to the public queue head so the job is
|
|
36
44
|
# next pulled. LREM count=1 is idempotent for our payloads since
|
|
37
45
|
# each job's JSON contains a unique `jid`.
|
|
38
|
-
|
|
46
|
+
#
|
|
47
|
+
# `jid` is filled in by the Processor once it has parsed the payload —
|
|
48
|
+
# the fetcher never parses. It is only used to retire the job's
|
|
49
|
+
# poison-pill recovery counter, so an ACK without one is still a
|
|
50
|
+
# complete ACK.
|
|
51
|
+
UnitOfWork = Struct.new(:queue, :job, :config, :jid, keyword_init: true) do
|
|
52
|
+
# The counter DEL rides this round trip rather than taking one of its
|
|
53
|
+
# own: the ACK is the only Redis call the success path makes, and a
|
|
54
|
+
# per-job call would be a fetch+execute regression for the sake of a
|
|
55
|
+
# key that exists for roughly no jobs. See Middleware::PoisonPill.
|
|
39
56
|
def acknowledge
|
|
57
|
+
private_list = Reliable.private_queue_name(queue)
|
|
58
|
+
job_jid = jid.to_s
|
|
59
|
+
return config.redis { |conn| conn.call('LREM', private_list, 1, job) } if job_jid.empty?
|
|
60
|
+
|
|
40
61
|
config.redis do |conn|
|
|
41
|
-
conn.
|
|
62
|
+
conn.pipelined do |pipe|
|
|
63
|
+
pipe.call('LREM', private_list, 1, job)
|
|
64
|
+
Middleware::PoisonPill.clear_in(pipe, job_jid)
|
|
65
|
+
end
|
|
42
66
|
end
|
|
43
67
|
end
|
|
44
68
|
|
|
@@ -55,9 +79,16 @@ module Wurk
|
|
|
55
79
|
# carrying a back-reference to its parent fetcher. Index defaults to
|
|
56
80
|
# 0 — we run one fetcher per capsule today. Multi-processor topology
|
|
57
81
|
# (one private list per processor slot) is a future Manager concern.
|
|
82
|
+
#
|
|
83
|
+
# The nonce marks the incarnation. host+pid alone is ambiguous once PID
|
|
84
|
+
# namespaces are in play: a restarted container reuses both, so the
|
|
85
|
+
# reaper's `kill(0)` liveness check would read a dead owner's list as
|
|
86
|
+
# live (jobs stranded) or a live owner's as dead (job run twice). Keys
|
|
87
|
+
# written before the nonce existed stay reclaimable — Reaper#parse_owner
|
|
88
|
+
# accepts both shapes.
|
|
58
89
|
def self.private_queue_name(public_queue, index = 0)
|
|
59
90
|
host = ENV['DYNO'] || Socket.gethostname
|
|
60
|
-
"#{public_queue}|#{host}|#{::Process.pid}|#{index}"
|
|
91
|
+
"#{public_queue}|#{host}|#{::Process.pid}|#{Component::PROCESS_NONCE}|#{index}"
|
|
61
92
|
end
|
|
62
93
|
|
|
63
94
|
def initialize(capsule)
|
|
@@ -66,11 +97,26 @@ module Wurk
|
|
|
66
97
|
@done = false
|
|
67
98
|
end
|
|
68
99
|
|
|
100
|
+
# Every pass that yields no job has to cost wall-clock time: Processor#run
|
|
101
|
+
# drives `process_one` in a bare `until @done` loop with no pause of its
|
|
102
|
+
# own, so any nil returned instantly turns N processor threads into a hot
|
|
103
|
+
# loop. The blocking BLMOVE pays that cost on the normal empty-queue path;
|
|
104
|
+
# the two short-circuits below have to pay it themselves.
|
|
69
105
|
def retrieve_work
|
|
70
|
-
|
|
106
|
+
if @done
|
|
107
|
+
sleep QUIET_PAUSE
|
|
108
|
+
return nil
|
|
109
|
+
end
|
|
71
110
|
|
|
72
111
|
queues = queues_cmd
|
|
73
|
-
|
|
112
|
+
# Nothing fetchable — every queue paused, or none configured. Back off a
|
|
113
|
+
# full poll interval rather than re-running queues_cmd (an SMEMBERS per
|
|
114
|
+
# pass, on the main pool) as fast as the CPU allows. Mirrors Sidekiq's
|
|
115
|
+
# BasicFetch guard, upstream #4825.
|
|
116
|
+
if queues.empty?
|
|
117
|
+
sleep poll_interval
|
|
118
|
+
return nil
|
|
119
|
+
end
|
|
74
120
|
|
|
75
121
|
queues.each do |public_q|
|
|
76
122
|
uow = lmove(public_q)
|
|
@@ -148,12 +194,19 @@ module Wurk
|
|
|
148
194
|
# is dominated by the BLMOVE that follows. Returns a Set for O(1)
|
|
149
195
|
# lookup against the (often weighted-expanded) queue list.
|
|
150
196
|
def paused_names
|
|
151
|
-
config.redis { |conn| conn.call('SMEMBERS', Keys::PAUSED_SET) }.to_set
|
|
197
|
+
config.redis(idempotent: true) { |conn| conn.call('SMEMBERS', Keys::PAUSED_SET) }.to_set
|
|
152
198
|
end
|
|
153
199
|
|
|
200
|
+
# Both LMOVE forms claim apply-safety, so fetch keeps the full
|
|
201
|
+
# connection-blip backoff the F5 split otherwise takes away: a move that
|
|
202
|
+
# applied but whose reply was lost leaves the job in *this* process's
|
|
203
|
+
# private list, un-ACKed — byte-for-byte the state a SIGKILL between fetch
|
|
204
|
+
# and ack leaves behind, which the next boot's Reaper already reclaims. The
|
|
205
|
+
# replay then pulls a different job; nothing duplicates and nothing is
|
|
206
|
+
# lost, at worst one job waits out this process's lifetime.
|
|
154
207
|
def lmove(public_q)
|
|
155
208
|
priv = self.class.private_queue_name(public_q)
|
|
156
|
-
job = config.redis { |conn| conn.call('LMOVE', public_q, priv, 'RIGHT', 'LEFT') }
|
|
209
|
+
job = config.redis(idempotent: true) { |conn| conn.call('LMOVE', public_q, priv, 'RIGHT', 'LEFT') }
|
|
157
210
|
job ? UnitOfWork.new(queue: public_q, job: job, config: config) : nil
|
|
158
211
|
end
|
|
159
212
|
|
|
@@ -165,7 +218,7 @@ module Wurk
|
|
|
165
218
|
# starving the main pool's background loops (#101). Extend the socket
|
|
166
219
|
# read-timeout one second past BLMOVE's own server-side timeout so the
|
|
167
220
|
# connection's read timeout can't fire while BLMOVE is legitimately blocked.
|
|
168
|
-
job = config.fetch_redis do |conn|
|
|
221
|
+
job = config.fetch_redis(idempotent: true) do |conn|
|
|
169
222
|
conn.blocking_call(timeout + 1, 'BLMOVE', public_q, priv, 'RIGHT', 'LEFT', timeout)
|
|
170
223
|
end
|
|
171
224
|
job ? UnitOfWork.new(queue: public_q, job: job, config: config) : nil
|