wurk 1.3.1 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +4 -1
- data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +22 -5
- data/lib/wurk/batch/callbacks.rb +82 -12
- data/lib/wurk/batch/death_handler.rb +7 -4
- data/lib/wurk/batch/server_middleware.rb +1 -1
- data/lib/wurk/batch.rb +121 -15
- data/lib/wurk/capsule.rb +45 -16
- data/lib/wurk/cli.rb +48 -14
- data/lib/wurk/client/buffered.rb +200 -43
- data/lib/wurk/client.rb +181 -38
- data/lib/wurk/compat.rb +1 -1
- data/lib/wurk/component.rb +36 -4
- data/lib/wurk/configuration.rb +47 -7
- data/lib/wurk/context.rb +1 -1
- data/lib/wurk/cron.rb +94 -37
- data/lib/wurk/deploy.rb +5 -3
- data/lib/wurk/embedded.rb +13 -0
- data/lib/wurk/engine.rb +41 -1
- data/lib/wurk/errors.rb +15 -0
- data/lib/wurk/fetcher/reaper.rb +125 -58
- data/lib/wurk/fetcher/reliable.rb +366 -51
- data/lib/wurk/fetcher.rb +6 -0
- data/lib/wurk/heartbeat.rb +24 -12
- data/lib/wurk/history.rb +13 -1
- data/lib/wurk/job_logger.rb +16 -7
- data/lib/wurk/job_set.rb +3 -2
- data/lib/wurk/job_util.rb +44 -24
- data/lib/wurk/launcher.rb +230 -119
- data/lib/wurk/leader.rb +63 -14
- data/lib/wurk/limiter/base.rb +8 -10
- data/lib/wurk/limiter/bucket.rb +1 -1
- data/lib/wurk/limiter/concurrent.rb +27 -22
- data/lib/wurk/limiter/window.rb +13 -11
- data/lib/wurk/limiter.rb +7 -4
- data/lib/wurk/logger.rb +1 -1
- data/lib/wurk/lua/loader.rb +9 -3
- data/lib/wurk/lua.rb +97 -14
- data/lib/wurk/manager.rb +29 -13
- data/lib/wurk/metrics/accumulator.rb +95 -0
- data/lib/wurk/metrics/flusher.rb +70 -0
- data/lib/wurk/metrics/history.rb +102 -32
- data/lib/wurk/metrics/queue_rollup.rb +13 -1
- data/lib/wurk/metrics/rollup.rb +13 -1
- data/lib/wurk/metrics/statsd.rb +32 -18
- data/lib/wurk/middleware/chain.rb +31 -14
- data/lib/wurk/middleware/interrupt_handler.rb +7 -6
- data/lib/wurk/middleware/poison_pill.rb +93 -31
- data/lib/wurk/middleware.rb +2 -2
- data/lib/wurk/pool_checkout.rb +39 -0
- data/lib/wurk/process_set.rb +10 -5
- data/lib/wurk/processor.rb +83 -9
- data/lib/wurk/profiler.rb +9 -4
- data/lib/wurk/queue.rb +18 -7
- data/lib/wurk/rails_boot.rb +38 -7
- data/lib/wurk/redis_client_adapter.rb +49 -5
- data/lib/wurk/redis_pool.rb +71 -25
- data/lib/wurk/scheduled.rb +30 -2
- data/lib/wurk/shutdown_gate.rb +79 -0
- data/lib/wurk/stats.rb +19 -10
- data/lib/wurk/swarm/child_boot.rb +36 -4
- data/lib/wurk/swarm.rb +258 -43
- data/lib/wurk/timer_loop.rb +14 -0
- data/lib/wurk/version.rb +1 -1
- data/lib/wurk/web/config.rb +11 -7
- data/lib/wurk/web/enterprise.rb +58 -6
- data/lib/wurk/web/extension.rb +1 -1
- data/lib/wurk/web/search.rb +5 -3
- data/lib/wurk.rb +12 -10
- data/vendor/assets/dashboard/assets/{ArgsValue-D74zX0MI.js → ArgsValue-CcR2ya6e.js} +1 -1
- data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-CUXJUQ3Q.js} +1 -1
- data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-Cxan6Ngw.js} +1 -1
- data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-DC5EGM0g.js} +1 -1
- data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-Dlt8tXJA.js} +1 -1
- data/vendor/assets/dashboard/assets/Dashboard-DNLu_WCg.js +1 -0
- data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-dZ7VGlKS.js} +1 -1
- data/vendor/assets/dashboard/assets/Extension-DaFpEIJf.js +1 -0
- data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-CO3aYWIq.js} +1 -1
- data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-DSWbT6G0.js} +1 -1
- data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-Cb4PKXNR.js} +1 -1
- data/vendor/assets/dashboard/assets/Metrics-CCGzgCsT.js +1 -0
- data/vendor/assets/dashboard/assets/Modal-B86q6ruL.js +1 -0
- data/vendor/assets/dashboard/assets/{PageHeader-C44KNMGm.js → PageHeader-fPrCcp_-.js} +1 -1
- data/vendor/assets/dashboard/assets/{Profiles-xEVTyS2N.js → Profiles-BnS82nR_.js} +1 -1
- data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-CIyPevOy.js} +1 -1
- data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-DopwXkXl.js} +1 -1
- data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-1-Z7i1zE.js} +1 -1
- data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-ByA6eTma.js} +1 -1
- data/vendor/assets/dashboard/assets/{Skeleton-DzR7XNxz.js → Skeleton-bC7HfQ9r.js} +1 -1
- data/vendor/assets/dashboard/assets/{charts-BVHHGof7.js → charts-CLLzJ7vK.js} +1 -1
- data/vendor/assets/dashboard/assets/index-B1N8hQUh.js +141 -0
- data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
- data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-DpBjkf6_.js} +1 -1
- data/vendor/assets/dashboard/assets/{useSort-BeYbztkN.js → useSort-DvpwuNQE.js} +1 -1
- data/vendor/assets/dashboard/index.html +3 -3
- data/vendor/assets/dashboard/wurk-manifest.json +2 -2
- metadata +32 -27
- data/vendor/assets/dashboard/assets/Dashboard-B9rOrkzk.js +0 -1
- data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
- data/vendor/assets/dashboard/assets/Metrics-BBTDxcaE.js +0 -1
- data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
- data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
- data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
data/lib/wurk/cron.rb
CHANGED
|
@@ -6,6 +6,7 @@ require 'time'
|
|
|
6
6
|
require_relative 'component'
|
|
7
7
|
require_relative 'client'
|
|
8
8
|
require_relative 'leader'
|
|
9
|
+
require_relative 'timer_loop'
|
|
9
10
|
|
|
10
11
|
module Wurk
|
|
11
12
|
# Sidekiq Enterprise periodic jobs. Pure leader-driven cron — only the
|
|
@@ -477,8 +478,8 @@ module Wurk
|
|
|
477
478
|
|
|
478
479
|
private
|
|
479
480
|
|
|
480
|
-
def redis(&)
|
|
481
|
-
@config ? @config.redis(&) : Wurk.redis(&)
|
|
481
|
+
def redis(idempotent: false, &)
|
|
482
|
+
@config ? @config.redis(idempotent:, &) : Wurk.redis(idempotent:, &)
|
|
482
483
|
end
|
|
483
484
|
end
|
|
484
485
|
|
|
@@ -520,34 +521,35 @@ module Wurk
|
|
|
520
521
|
|
|
521
522
|
def initialize(config)
|
|
522
523
|
@config = config
|
|
523
|
-
@done = false
|
|
524
|
-
@mutex = ::Mutex.new
|
|
525
|
-
@sleeper = ::ConditionVariable.new
|
|
526
524
|
@client = Client.new(config: config)
|
|
527
525
|
@thread = nil
|
|
528
526
|
# Operators never need to touch this; integration tests shrink it so a
|
|
529
527
|
# due loop fires within the test window instead of waiting a full minute.
|
|
530
528
|
@tick_interval = config[:cron_tick_interval] || DEFAULT_TICK_SECONDS
|
|
529
|
+
@timer = TimerLoop.new(@tick_interval)
|
|
531
530
|
end
|
|
532
531
|
|
|
532
|
+
# TimerLoop waits one interval before the first tick: don't fire a
|
|
533
|
+
# catch-up burst the instant we boot (the leader is barely settled), and
|
|
534
|
+
# let a short-lived process exit without ticking at all.
|
|
533
535
|
def start
|
|
534
|
-
@
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
wait
|
|
539
|
-
until @done
|
|
540
|
-
tick
|
|
541
|
-
wait
|
|
542
|
-
end
|
|
543
|
-
end
|
|
536
|
+
return @thread if @thread
|
|
537
|
+
|
|
538
|
+
@timer.reset
|
|
539
|
+
@thread = safe_thread('cron-poller') { @timer.run { tick } }
|
|
544
540
|
end
|
|
545
541
|
|
|
542
|
+
# Blocks until the thread is really gone: the launcher releases the
|
|
543
|
+
# cluster lock immediately after this returns, and a tick still in flight
|
|
544
|
+
# would enqueue loops the next leader is about to fire itself.
|
|
545
|
+
#
|
|
546
|
+
# Cleared only on a confirmed join (Thread#join returns nil on timeout):
|
|
547
|
+
# a wedged thread must stay tracked so #start's guard returns it instead
|
|
548
|
+
# of calling @timer.reset, which would un-terminate the loop it is still
|
|
549
|
+
# inside and leave two tick threads double-enqueuing the same loops.
|
|
546
550
|
def terminate
|
|
547
|
-
@
|
|
548
|
-
|
|
549
|
-
@sleeper.signal
|
|
550
|
-
end
|
|
551
|
+
@timer.terminate
|
|
552
|
+
@thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
|
|
551
553
|
end
|
|
552
554
|
|
|
553
555
|
# Leader-gated by the single cluster lock (Component#leader? reads
|
|
@@ -562,18 +564,23 @@ module Wurk
|
|
|
562
564
|
handle_exception(e, { context: 'cron-poller' })
|
|
563
565
|
end
|
|
564
566
|
|
|
567
|
+
# The marks advance *before* the push, via CAS: the leader gate is a
|
|
568
|
+
# cached read, so a second poller can reach the same due loop for a few
|
|
569
|
+
# seconds after a handover. Losing the CAS means another tick owns this
|
|
570
|
+
# slot — enqueue nothing.
|
|
565
571
|
def enqueue_if_due(loop_obj)
|
|
566
572
|
return if loop_obj.paused?
|
|
567
573
|
|
|
568
574
|
now = ::Time.now.to_i
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
return if next_fire.nil? || next_fire > now
|
|
575
|
+
slot, mark = due_slot(loop_obj, now)
|
|
576
|
+
return if slot.nil?
|
|
572
577
|
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
578
|
+
future = loop_obj.next_fire_after(slot, now)
|
|
579
|
+
return unless claim_fire?(loop_obj, mark, now, future)
|
|
580
|
+
|
|
581
|
+
warn_missed_tick(loop_obj, slot, now)
|
|
582
|
+
jid = enqueue_claimed!(loop_obj)
|
|
583
|
+
record_history(loop_obj, jid, now)
|
|
577
584
|
jid
|
|
578
585
|
end
|
|
579
586
|
|
|
@@ -590,10 +597,16 @@ module Wurk
|
|
|
590
597
|
|
|
591
598
|
private
|
|
592
599
|
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
600
|
+
# The slot this tick would fire and the token that claims it, or nil when
|
|
601
|
+
# the loop has no resolvable occurrence or none is due yet. The token is
|
|
602
|
+
# the stored `nf` when there is one, otherwise the slot we derived — see
|
|
603
|
+
# the `cron_claim_fire` script for how the two cases differ.
|
|
604
|
+
def due_slot(loop_obj, now)
|
|
605
|
+
prev_fire, next_fire, mark = read_fire_marks(loop_obj.lid)
|
|
606
|
+
next_fire ||= loop_obj.next_fire_at(prev_fire || (now - @tick_interval))
|
|
607
|
+
return if next_fire.nil? || next_fire > now
|
|
608
|
+
|
|
609
|
+
[next_fire, mark || next_fire.to_s]
|
|
597
610
|
end
|
|
598
611
|
|
|
599
612
|
def warn_missed_tick(loop_obj, expected, now)
|
|
@@ -605,6 +618,34 @@ module Wurk
|
|
|
605
618
|
)
|
|
606
619
|
end
|
|
607
620
|
|
|
621
|
+
# Compare-and-swap the fire marks: only the tick still looking at the
|
|
622
|
+
# mark it read wins the slot. Manual `#fire` deliberately skips this —
|
|
623
|
+
# an operator-requested run contends with no schedule slot.
|
|
624
|
+
def claim_fire?(loop_obj, mark, fired_at, future)
|
|
625
|
+
@config.redis do |c|
|
|
626
|
+
Wurk::Lua::Loader.eval_cached(
|
|
627
|
+
c, :cron_claim_fire,
|
|
628
|
+
keys: ["#{LOOP_PREFIX}#{loop_obj.lid}"],
|
|
629
|
+
argv: [mark, fired_at.to_s, future.nil? ? '' : future.to_s]
|
|
630
|
+
)
|
|
631
|
+
end == 1
|
|
632
|
+
end
|
|
633
|
+
|
|
634
|
+
# The slot is already claimed, so a failed push loses this occurrence
|
|
635
|
+
# instead of leaving the next tick to fire it again. Ent leader election
|
|
636
|
+
# is explicitly best-effort coordination (sidekiq-ent.md §6), so a missed
|
|
637
|
+
# occurrence is tolerable where a duplicate one is not — log the gap so it
|
|
638
|
+
# stays attributable.
|
|
639
|
+
def enqueue_claimed!(loop_obj)
|
|
640
|
+
enqueue!(loop_obj)
|
|
641
|
+
rescue StandardError => e
|
|
642
|
+
logger.warn(
|
|
643
|
+
"[cron] fire lost lid=#{loop_obj.lid} klass=#{loop_obj.klass} " \
|
|
644
|
+
"mark already advanced, occurrence skipped: #{e.class}: #{e.message}"
|
|
645
|
+
)
|
|
646
|
+
raise
|
|
647
|
+
end
|
|
648
|
+
|
|
608
649
|
def enqueue!(loop_obj)
|
|
609
650
|
aj = active_job_class(loop_obj.klass)
|
|
610
651
|
return enqueue_active_job(aj, loop_obj) if aj
|
|
@@ -643,24 +684,40 @@ module Wurk
|
|
|
643
684
|
job.provider_job_id if job.respond_to?(:provider_job_id)
|
|
644
685
|
end
|
|
645
686
|
|
|
687
|
+
# Third element is the raw `nf` string — the CAS token for #claim_fire.
|
|
688
|
+
# Comparing the bytes Redis holds, not the parsed integer, keeps the
|
|
689
|
+
# token exact regardless of how the value was written.
|
|
646
690
|
def read_fire_marks(lid)
|
|
647
691
|
@config.redis do |c|
|
|
648
|
-
|
|
649
|
-
next_fire = vals[1]
|
|
692
|
+
prev_fire, next_fire = c.call('HMGET', "#{LOOP_PREFIX}#{lid}", 'lf', 'nf')
|
|
650
693
|
# Preserve nil: treat missing or empty 'nf' as nil, not 0.
|
|
651
|
-
|
|
694
|
+
next_fire = nil if next_fire.nil? || next_fire.empty?
|
|
695
|
+
[prev_fire&.to_i, next_fire&.to_i, next_fire]
|
|
652
696
|
end
|
|
653
697
|
end
|
|
654
698
|
|
|
655
699
|
def record_fire(loop_obj, jid, fired_at, future)
|
|
656
|
-
|
|
700
|
+
write_fire_marks(loop_obj.lid, fired_at, future)
|
|
701
|
+
record_history(loop_obj, jid, fired_at)
|
|
702
|
+
end
|
|
703
|
+
|
|
704
|
+
# Unconditional mark write, for the manual `#fire` path only. The
|
|
705
|
+
# scheduled path advances the marks through #claim_fire instead.
|
|
706
|
+
def write_fire_marks(lid, fired_at, future)
|
|
707
|
+
key = "#{LOOP_PREFIX}#{lid}"
|
|
657
708
|
@config.redis do |c|
|
|
658
709
|
if future.nil?
|
|
659
|
-
c.call('HSET',
|
|
660
|
-
c.call('HDEL',
|
|
710
|
+
c.call('HSET', key, 'lf', fired_at.to_s)
|
|
711
|
+
c.call('HDEL', key, 'nf')
|
|
661
712
|
else
|
|
662
|
-
c.call('HSET',
|
|
713
|
+
c.call('HSET', key, 'lf', fired_at.to_s, 'nf', future.to_s)
|
|
663
714
|
end
|
|
715
|
+
end
|
|
716
|
+
end
|
|
717
|
+
|
|
718
|
+
def record_history(loop_obj, jid, fired_at)
|
|
719
|
+
history_entry = Wurk.dump_json([fired_at, jid])
|
|
720
|
+
@config.redis do |c|
|
|
664
721
|
c.call('LPUSH', "#{HISTORY_PREFIX}#{loop_obj.lid}", history_entry)
|
|
665
722
|
c.call('LTRIM', "#{HISTORY_PREFIX}#{loop_obj.lid}", 0, HISTORY_CAP - 1)
|
|
666
723
|
end
|
data/lib/wurk/deploy.rb
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
+
require_relative 'pool_checkout'
|
|
4
|
+
|
|
3
5
|
module Wurk
|
|
4
6
|
# Records deploy markers into Redis so the history pane can overlay
|
|
5
7
|
# "deployed at" lines onto job-throughput charts. Wire-compatible with
|
|
@@ -84,11 +86,11 @@ module Wurk
|
|
|
84
86
|
nil
|
|
85
87
|
end
|
|
86
88
|
|
|
87
|
-
def with_redis(&)
|
|
89
|
+
def with_redis(idempotent: false, &)
|
|
88
90
|
if @pool
|
|
89
|
-
|
|
91
|
+
PoolCheckout.with(@pool, idempotent, &)
|
|
90
92
|
else
|
|
91
|
-
Wurk.redis(&)
|
|
93
|
+
Wurk.redis(idempotent:, &)
|
|
92
94
|
end
|
|
93
95
|
end
|
|
94
96
|
end
|
data/lib/wurk/embedded.rb
CHANGED
|
@@ -41,6 +41,12 @@ module Wurk
|
|
|
41
41
|
sleep 0.2
|
|
42
42
|
logger.info { "Wurk running embedded, total process thread count: #{Thread.list.size}" }
|
|
43
43
|
logger.debug { Thread.list.map(&:name).to_s }
|
|
44
|
+
rescue StandardError
|
|
45
|
+
# The host is left believing Wurk never started, so anything the launcher
|
|
46
|
+
# did get up would keep fetching, beating and campaigning for the leader
|
|
47
|
+
# lock unowned for the life of the process.
|
|
48
|
+
roll_back_partial_boot
|
|
49
|
+
raise
|
|
44
50
|
end
|
|
45
51
|
|
|
46
52
|
# Stop fetching new work; in-flight jobs continue.
|
|
@@ -56,6 +62,13 @@ module Wurk
|
|
|
56
62
|
|
|
57
63
|
private
|
|
58
64
|
|
|
65
|
+
# Guarded so the caller sees why the boot failed, not why the rollback did.
|
|
66
|
+
def roll_back_partial_boot
|
|
67
|
+
stop
|
|
68
|
+
rescue StandardError => e
|
|
69
|
+
handle_exception(e, { context: 'embedded-boot-rollback' })
|
|
70
|
+
end
|
|
71
|
+
|
|
59
72
|
# Extracted so tests can swap in a fake launcher without monkey-patching
|
|
60
73
|
# Wurk::Launcher.new globally (which races other parallel tests).
|
|
61
74
|
def build_launcher
|
data/lib/wurk/engine.rb
CHANGED
|
@@ -25,6 +25,25 @@ module Wurk
|
|
|
25
25
|
class AssetMount
|
|
26
26
|
PREFIX = '/wurk-assets'
|
|
27
27
|
|
|
28
|
+
# Vite fingerprints everything it emits into `assets/`
|
|
29
|
+
# (`Dashboard-cGKycyd0.js`), so the bytes behind one of those URLs can
|
|
30
|
+
# never change — cache them for a year and never revalidate. The rest of
|
|
31
|
+
# the bundle (index.html, the favicons, wurk-manifest.json,
|
|
32
|
+
# .vite/manifest.json) keeps a stable name across builds, so it must be
|
|
33
|
+
# revalidated or a client would pin a stale dashboard shell forever.
|
|
34
|
+
FINGERPRINTED_PREFIX = '/assets/'
|
|
35
|
+
IMMUTABLE_CACHE_CONTROL = 'public, max-age=31536000, immutable'
|
|
36
|
+
REVALIDATE_CACHE_CONTROL = 'public, no-cache'
|
|
37
|
+
|
|
38
|
+
# Only a served body gets a cache directive. Rack::Files answers more than
|
|
39
|
+
# file reads: OPTIONS with 200 + Allow, 405 for any other verb
|
|
40
|
+
# (`ALLOWED_VERBS = %w[GET HEAD OPTIONS]`), and 416 for an unsatisfiable
|
|
41
|
+
# Range. 405 is cacheable by default (RFC 9110 §15.5.6), so stamping it
|
|
42
|
+
# `immutable` would let a shared proxy serve "method not allowed" for a
|
|
43
|
+
# year to everyone behind it.
|
|
44
|
+
CACHEABLE_METHODS = %w[GET HEAD].freeze
|
|
45
|
+
CACHEABLE_STATUSES = [200, 206, 304].freeze
|
|
46
|
+
|
|
28
47
|
def initialize(app, root:)
|
|
29
48
|
@app = app
|
|
30
49
|
@files = ::Rack::Files.new(root)
|
|
@@ -38,7 +57,28 @@ module Wurk
|
|
|
38
57
|
stripped = path.delete_prefix(PREFIX)
|
|
39
58
|
inner[::Rack::PATH_INFO] = stripped.empty? ? '/' : stripped
|
|
40
59
|
response = @files.call(inner)
|
|
41
|
-
|
|
60
|
+
return @app.call(env) if response[0] == 404
|
|
61
|
+
|
|
62
|
+
stamp_cache_control(response, env, stripped)
|
|
63
|
+
response
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
private
|
|
67
|
+
|
|
68
|
+
# This mount is inserted at index 0 of the HOST app's stack (see the
|
|
69
|
+
# initializer below), so Rack::ETag and Rack::ConditionalGet never see
|
|
70
|
+
# these responses — without this the dashboard's fingerprinted bundle
|
|
71
|
+
# shipped with no cache directive at all and was refetched every load.
|
|
72
|
+
def stamp_cache_control(response, env, stripped)
|
|
73
|
+
return unless CACHEABLE_METHODS.include?(env[::Rack::REQUEST_METHOD])
|
|
74
|
+
return unless CACHEABLE_STATUSES.include?(response[0])
|
|
75
|
+
|
|
76
|
+
response[1]['cache-control'] ||=
|
|
77
|
+
if stripped.start_with?(FINGERPRINTED_PREFIX)
|
|
78
|
+
IMMUTABLE_CACHE_CONTROL
|
|
79
|
+
else
|
|
80
|
+
REVALIDATE_CACHE_CONTROL
|
|
81
|
+
end
|
|
42
82
|
end
|
|
43
83
|
end
|
|
44
84
|
|
data/lib/wurk/errors.rb
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Wurk
|
|
4
|
+
class Error < StandardError; end
|
|
5
|
+
|
|
6
|
+
# Raised inside a worker process to abort the run loop. User code must not
|
|
7
|
+
# rescue this — the swarm uses it to signal teardown across thread boundaries.
|
|
8
|
+
# Spec: docs/target/sidekiq-free.md §3.
|
|
9
|
+
#
|
|
10
|
+
# Lives in its own file, loaded before everything else, because Processor
|
|
11
|
+
# freezes its `Thread.handle_interrupt` masks into constants: those are
|
|
12
|
+
# evaluated the moment `wurk/processor` is required, so this class has to
|
|
13
|
+
# exist by then.
|
|
14
|
+
class Shutdown < Interrupt; end
|
|
15
|
+
end
|
data/lib/wurk/fetcher/reaper.rb
CHANGED
|
@@ -3,13 +3,14 @@
|
|
|
3
3
|
require_relative '../component'
|
|
4
4
|
require_relative '../keys'
|
|
5
5
|
require_relative '../middleware/poison_pill'
|
|
6
|
+
require_relative '../timer_loop'
|
|
6
7
|
|
|
7
8
|
module Wurk
|
|
8
9
|
class Fetcher
|
|
9
10
|
# Orphan reclamation for the reliable fetcher (Pro super_fetch §3.2).
|
|
10
11
|
#
|
|
11
12
|
# The Reliable fetcher moves each job from a public queue into a
|
|
12
|
-
# per-process private list (`queue:<public>|<host>|<pid>|<idx>`) and
|
|
13
|
+
# per-process private list (`queue:<public>|<host>|<pid>|<nonce>|<idx>`) and
|
|
13
14
|
# leaves it there until the Processor ACKs. A SIGKILLed or crashed
|
|
14
15
|
# worker therefore strands its in-flight jobs in private lists that
|
|
15
16
|
# nobody will ever ACK. The Reaper is the recovery half: it periodically
|
|
@@ -17,15 +18,16 @@ module Wurk
|
|
|
17
18
|
# moves their jobs back to the public queue so a live worker re-runs them.
|
|
18
19
|
#
|
|
19
20
|
# Liveness is decided per owner:
|
|
20
|
-
# *
|
|
21
|
-
#
|
|
22
|
-
#
|
|
23
|
-
#
|
|
24
|
-
#
|
|
25
|
-
#
|
|
26
|
-
#
|
|
27
|
-
#
|
|
28
|
-
#
|
|
21
|
+
# * our own host and nonce — the pid was minted in our PID namespace, so
|
|
22
|
+
# the OS is authoritative: `Process.kill(0, pid)`. This is instant and
|
|
23
|
+
# ignores a stale `processes` SET entry whose 60s TTL hasn't lapsed
|
|
24
|
+
# yet, so a `kill -9`ed sibling is reclaimed the moment the supervisor
|
|
25
|
+
# reaps it rather than 60s later. (Pid reuse by an unrelated process in
|
|
26
|
+
# the same tree is the one blind spot — the supervisor respawns with a
|
|
27
|
+
# fresh pid, so it does not arise in practice.)
|
|
28
|
+
# * any other incarnation — its pid means nothing in our namespace, so we
|
|
29
|
+
# trust the heartbeat: the owner is alive iff its identity is a live
|
|
30
|
+
# `processes` member (one whose `info` hash still exists). Such reclaim
|
|
29
31
|
# therefore waits out the 60s heartbeat TTL, exactly as the spec says.
|
|
30
32
|
#
|
|
31
33
|
# Re-pushed jobs run through Wurk::Middleware::PoisonPill, which caps a
|
|
@@ -90,13 +92,22 @@ module Wurk
|
|
|
90
92
|
@thread
|
|
91
93
|
end
|
|
92
94
|
|
|
95
|
+
# Bounded at TimerLoop::JOIN_TIMEOUT like every other periodic component.
|
|
96
|
+
# The full-keyspace sweep is exactly the tick that outlasts a stop: it
|
|
97
|
+
# SCANs the whole `queue:*|*` keyspace and drains what it finds, so on a
|
|
98
|
+
# large or slow Redis it can still be running when shutdown lands. Waiting
|
|
99
|
+
# it out held the entire process's teardown open past the swarm parent's
|
|
100
|
+
# SHUTDOWN_GRACE — which SIGKILLs the child mid-drain, the one outcome
|
|
101
|
+
# this component exists to recover from. A straggler is left running
|
|
102
|
+
# instead — and still referenced, like every other periodic component, so
|
|
103
|
+
# a restart after a timed-out stop can't spawn a second sweep loop
|
|
104
|
+
# alongside it. Its own tick_once rescues and reports.
|
|
93
105
|
def stop
|
|
94
106
|
@mutex.synchronize do
|
|
95
107
|
@done = true
|
|
96
108
|
@sleeper.signal
|
|
97
109
|
end
|
|
98
|
-
@thread&.join
|
|
99
|
-
@thread = nil
|
|
110
|
+
@thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
|
|
100
111
|
end
|
|
101
112
|
|
|
102
113
|
def running?
|
|
@@ -116,8 +127,8 @@ module Wurk
|
|
|
116
127
|
# jobs reclaimed (re-queued or killed). Public so boot paths and tests
|
|
117
128
|
# can drive a deterministic pass without the cluster lock.
|
|
118
129
|
def reclaim!
|
|
119
|
-
|
|
120
|
-
served_queues.sum { |public_q| reclaim_queue(public_q,
|
|
130
|
+
owners = live_owners
|
|
131
|
+
served_queues.sum { |public_q| reclaim_queue(public_q, owners) }
|
|
121
132
|
end
|
|
122
133
|
|
|
123
134
|
# One unguarded full-keyspace sweep: every `queue:*|*` private list, even
|
|
@@ -125,10 +136,10 @@ module Wurk
|
|
|
125
136
|
# of jobs reclaimed. Public so boot paths and tests can drive it without
|
|
126
137
|
# the hourly lock.
|
|
127
138
|
def reclaim_full!
|
|
128
|
-
|
|
139
|
+
owners = live_owners
|
|
129
140
|
reclaimed = 0
|
|
130
|
-
each_full_private_list do |key, public_q, host, pid|
|
|
131
|
-
next if owner_alive?(host, pid,
|
|
141
|
+
each_full_private_list do |key, public_q, host, pid, nonce|
|
|
142
|
+
next if owner_alive?(host, pid, nonce, owners)
|
|
132
143
|
|
|
133
144
|
reclaimed += drain(key, public_q)
|
|
134
145
|
end
|
|
@@ -148,39 +159,43 @@ module Wurk
|
|
|
148
159
|
end
|
|
149
160
|
|
|
150
161
|
# SCAN for this public queue's private lists, reclaim the orphaned ones.
|
|
151
|
-
def reclaim_queue(public_q,
|
|
162
|
+
def reclaim_queue(public_q, owners)
|
|
152
163
|
reclaimed = 0
|
|
153
|
-
each_private_list(public_q) do |key, host, pid|
|
|
154
|
-
next if owner_alive?(host, pid,
|
|
164
|
+
each_private_list(public_q) do |key, host, pid, nonce|
|
|
165
|
+
next if owner_alive?(host, pid, nonce, owners)
|
|
155
166
|
|
|
156
167
|
reclaimed += drain(key, public_q)
|
|
157
168
|
end
|
|
158
169
|
reclaimed
|
|
159
170
|
end
|
|
160
171
|
|
|
161
|
-
# Yields [private_list_key, host, pid] for each private list of
|
|
172
|
+
# Yields [private_list_key, host, pid, nonce] for each private list of
|
|
162
173
|
# `public_q`. MATCH `<public_q>|*` matches only this queue's private
|
|
163
174
|
# lists (public queue keys carry no `|`).
|
|
164
175
|
def each_private_list(public_q)
|
|
165
176
|
cursor = '0'
|
|
166
177
|
loop do
|
|
167
|
-
cursor, keys = redis
|
|
178
|
+
cursor, keys = redis(idempotent: true) do |c|
|
|
179
|
+
c.call('SCAN', cursor, 'MATCH', "#{public_q}|*", 'COUNT', SCAN_COUNT)
|
|
180
|
+
end
|
|
168
181
|
keys.each do |key|
|
|
169
|
-
host, pid = parse_owner(public_q, key)
|
|
170
|
-
yield key, host, pid if pid
|
|
182
|
+
host, pid, nonce = parse_owner(public_q, key)
|
|
183
|
+
yield key, host, pid, nonce if pid
|
|
171
184
|
end
|
|
172
185
|
break if cursor == '0'
|
|
173
186
|
end
|
|
174
187
|
end
|
|
175
188
|
|
|
176
|
-
# Yields [private_list_key, public_q, host, pid] for every
|
|
189
|
+
# Yields [private_list_key, public_q, host, pid, nonce] for every list in
|
|
177
190
|
# the keyspace. MATCH `queue:*|*` matches only private lists (public queue
|
|
178
191
|
# keys carry no `|`); parse_full_key drops anything that isn't a
|
|
179
|
-
# well-formed `queue:<public>|<host>|<pid>|<idx>`.
|
|
192
|
+
# well-formed `queue:<public>|<host>|<pid>|<nonce>|<idx>`.
|
|
180
193
|
def each_full_private_list
|
|
181
194
|
cursor = '0'
|
|
182
195
|
loop do
|
|
183
|
-
cursor, keys = redis
|
|
196
|
+
cursor, keys = redis(idempotent: true) do |c|
|
|
197
|
+
c.call('SCAN', cursor, 'MATCH', "#{Keys::QUEUE_PREFIX}*|*", 'COUNT', SCAN_COUNT)
|
|
198
|
+
end
|
|
184
199
|
keys.each do |key|
|
|
185
200
|
parsed = parse_full_key(key)
|
|
186
201
|
yield key, *parsed if parsed
|
|
@@ -189,45 +204,88 @@ module Wurk
|
|
|
189
204
|
end
|
|
190
205
|
end
|
|
191
206
|
|
|
192
|
-
# `queue:<public>|<host>|<pid>|<idx>` → [public_q, host, pid
|
|
193
|
-
#
|
|
194
|
-
#
|
|
195
|
-
#
|
|
207
|
+
# `queue:<public>|<host>|<pid>|<nonce>|<idx>` → [public_q, host, pid,
|
|
208
|
+
# nonce], parsed from the right so a `|` inside the queue name is
|
|
209
|
+
# tolerated. nil when the key isn't a well-formed private list.
|
|
210
|
+
#
|
|
211
|
+
# With no known prefix to split on, both owner shapes are tried in
|
|
212
|
+
# preference order and the first one leaving a real public queue behind
|
|
213
|
+
# wins. The narrow reading is what saves a pre-nonce key from an all-digit
|
|
214
|
+
# host (`queue:q|123456789012|<pid>|<idx>` — a bare Docker hostname is 12
|
|
215
|
+
# hex chars): read wide, its host segment eats the whole queue name.
|
|
196
216
|
def parse_full_key(key)
|
|
197
217
|
parts = key.split('|')
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
return nil unless integer?(pid) && integer?(idx)
|
|
218
|
+
owner_tails(parts).each do |host, pid, nonce, width|
|
|
219
|
+
public_q = parts[0...-width].join('|')
|
|
220
|
+
next unless public_q.start_with?(Keys::QUEUE_PREFIX) && public_q != Keys::QUEUE_PREFIX
|
|
202
221
|
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
[public_q, host, pid.to_i]
|
|
222
|
+
return [public_q, host, pid, nonce]
|
|
223
|
+
end
|
|
224
|
+
nil
|
|
207
225
|
end
|
|
208
226
|
|
|
209
|
-
# `<public_q>|<host>|<pid>|<idx>` → [host, pid] (pid as
|
|
210
|
-
#
|
|
211
|
-
# Splitting the suffix off the known public-queue prefix tolerates a
|
|
212
|
-
#
|
|
227
|
+
# `<public_q>|<host>|<pid>|<nonce>|<idx>` → [host, pid, nonce] (pid as
|
|
228
|
+
# Integer), or all-nil when the suffix isn't a well-formed owner tail.
|
|
229
|
+
# Splitting the suffix off the known public-queue prefix tolerates a `|`
|
|
230
|
+
# inside the queue name itself, and leaves the tail unambiguous: exactly
|
|
231
|
+
# 4 segments for the current shape, exactly 3 for the pre-nonce one.
|
|
213
232
|
def parse_owner(public_q, key)
|
|
214
233
|
suffix = key.delete_prefix("#{public_q}|")
|
|
215
|
-
return [nil, nil] if suffix == key
|
|
234
|
+
return [nil, nil, nil] if suffix == key
|
|
216
235
|
|
|
217
|
-
host, pid,
|
|
218
|
-
|
|
236
|
+
host, pid, nonce = owner_tails(suffix.split('|')).first
|
|
237
|
+
[host, pid, nonce]
|
|
238
|
+
end
|
|
219
239
|
|
|
220
|
-
|
|
240
|
+
# Owner segments of a private-list key, taken from the right, as
|
|
241
|
+
# [host, pid, nonce, segment_count] readings in preference order (empty
|
|
242
|
+
# when nothing parses). The wide shape is preferred: an all-digit nonce is
|
|
243
|
+
# rare but reachable, and reading such a key narrow would take the pid for
|
|
244
|
+
# the host and the nonce for the pid — draining a live owner's list out
|
|
245
|
+
# from under it.
|
|
246
|
+
def owner_tails(parts)
|
|
247
|
+
return [] unless parts.size >= 3 && integer?(parts[-1])
|
|
248
|
+
|
|
249
|
+
[wide_tail(parts), narrow_tail(parts)].compact
|
|
250
|
+
end
|
|
251
|
+
|
|
252
|
+
# `<host>|<pid>|<nonce>|<idx>` — the shape every current process writes.
|
|
253
|
+
def wide_tail(parts)
|
|
254
|
+
[parts[-4], parts[-3].to_i, parts[-2], 4] if parts.size >= 4 && integer?(parts[-3])
|
|
255
|
+
end
|
|
256
|
+
|
|
257
|
+
# `<host>|<pid>|<idx>` — written before the nonce existed. Such a list can
|
|
258
|
+
# still hold a pre-upgrade process's in-flight jobs across a rolling
|
|
259
|
+
# upgrade, so it stays reclaimable even though nothing writes it anymore.
|
|
260
|
+
def narrow_tail(parts)
|
|
261
|
+
[parts[-3], parts[-2].to_i, nil, 3] if integer?(parts[-2])
|
|
221
262
|
end
|
|
222
263
|
|
|
223
264
|
def integer?(str)
|
|
224
265
|
str.is_a?(String) && str.match?(/\A\d+\z/)
|
|
225
266
|
end
|
|
226
267
|
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
268
|
+
# `Process.kill(0, pid)` answers "does this pid exist *in my PID
|
|
269
|
+
# namespace*", which is the question we're actually asking only when the
|
|
270
|
+
# key was written from that same namespace. A shared hostname does not
|
|
271
|
+
# imply it: a container restarting under a fixed hostname comes back in a
|
|
272
|
+
# fresh namespace where the dead owner's pid is likely taken again (list
|
|
273
|
+
# read as live, jobs stranded forever), and two containers sharing a
|
|
274
|
+
# host's network namespace but not its pid namespace each hold pids the
|
|
275
|
+
# other lacks (live owner read as dead, list drained mid-job, duplicate
|
|
276
|
+
# run). So kill(0) can serve as neither a positive nor a negative signal
|
|
277
|
+
# off our own namespace.
|
|
278
|
+
#
|
|
279
|
+
# The nonce settles it: minted once per process image and inherited across
|
|
280
|
+
# fork, so a key carrying ours provably came from this very process tree.
|
|
281
|
+
# Every other owner goes through the namespace-blind heartbeat — alive iff
|
|
282
|
+
# its identity is a live `processes` member. A pre-nonce key can only be
|
|
283
|
+
# matched on the `<host>:<pid>` prefix of that identity.
|
|
284
|
+
def owner_alive?(host, pid, nonce, owners)
|
|
285
|
+
return local_pid_alive?(pid) if nonce == process_nonce && host == hostname
|
|
286
|
+
return owners.include?("#{host}:#{pid}:#{nonce}") if nonce
|
|
287
|
+
|
|
288
|
+
owners.include?("#{host}:#{pid}")
|
|
231
289
|
end
|
|
232
290
|
|
|
233
291
|
def local_pid_alive?(pid)
|
|
@@ -239,18 +297,21 @@ module Wurk
|
|
|
239
297
|
true
|
|
240
298
|
end
|
|
241
299
|
|
|
242
|
-
#
|
|
243
|
-
#
|
|
244
|
-
#
|
|
245
|
-
#
|
|
246
|
-
|
|
247
|
-
|
|
300
|
+
# Every live process indexed both ways: the full `<host>:<pid>:<nonce>`
|
|
301
|
+
# identity a nonce-bearing private list is matched on, and the
|
|
302
|
+
# `<host>:<pid>` prefix a pre-nonce one has to settle for. Live means a
|
|
303
|
+
# member of `processes` whose `info` hash still exists — a bare SET
|
|
304
|
+
# membership isn't enough, since the member lingers after its 60s hash
|
|
305
|
+
# TTL until ProcessSet#cleanup prunes it and that window must read as
|
|
306
|
+
# dead or the owner's jobs are never reclaimed.
|
|
307
|
+
def live_owners
|
|
308
|
+
redis(idempotent: true) do |conn|
|
|
248
309
|
members = conn.call('SMEMBERS', Keys::PROCESSES)
|
|
249
310
|
next ::Set.new if members.empty?
|
|
250
311
|
|
|
251
312
|
infos = conn.pipelined { |pipe| members.each { |m| pipe.call('HGET', m, 'info') } }
|
|
252
313
|
members.zip(infos).each_with_object(::Set.new) do |(member, info), set|
|
|
253
|
-
set << host_pid(member) if info
|
|
314
|
+
set << member << host_pid(member) if info
|
|
254
315
|
end
|
|
255
316
|
end
|
|
256
317
|
end
|
|
@@ -265,6 +326,12 @@ module Wurk
|
|
|
265
326
|
# poison check, so a crash mid-drain leaves the job safely in the public
|
|
266
327
|
# queue (at-least-once), never lost. Poison jobs are killed to the dead
|
|
267
328
|
# set by PoisonPill.track! and then LREM'd out of the public queue.
|
|
329
|
+
#
|
|
330
|
+
# Unlike the fetcher's LMOVE, this one does *not* claim apply-safety: a
|
|
331
|
+
# replay after a lost reply moves the next job and never poison-checks the
|
|
332
|
+
# one already on the public tail, so a job that kills its worker every time
|
|
333
|
+
# would get a free recovery past the cap. A raise instead lands in the
|
|
334
|
+
# rescue below, and the next sweep re-drains what's left.
|
|
268
335
|
def drain(private_list, public_q)
|
|
269
336
|
queue_name = public_q.delete_prefix(Keys::QUEUE_PREFIX)
|
|
270
337
|
count = 0
|