wurk 1.3.1 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +1 -0
- data/lib/wurk/batch/callbacks.rb +82 -12
- data/lib/wurk/batch/death_handler.rb +7 -4
- data/lib/wurk/batch/server_middleware.rb +1 -1
- data/lib/wurk/batch.rb +121 -15
- data/lib/wurk/capsule.rb +5 -4
- data/lib/wurk/cli.rb +48 -14
- data/lib/wurk/client/buffered.rb +193 -43
- data/lib/wurk/client.rb +87 -14
- data/lib/wurk/compat.rb +1 -1
- data/lib/wurk/component.rb +2 -2
- data/lib/wurk/configuration.rb +3 -2
- data/lib/wurk/cron.rb +94 -37
- data/lib/wurk/deploy.rb +5 -3
- data/lib/wurk/embedded.rb +13 -0
- data/lib/wurk/fetcher/reaper.rb +113 -56
- data/lib/wurk/fetcher/reliable.rb +62 -9
- data/lib/wurk/heartbeat.rb +22 -10
- data/lib/wurk/history.rb +13 -1
- data/lib/wurk/launcher.rb +133 -66
- data/lib/wurk/leader.rb +29 -10
- data/lib/wurk/limiter/base.rb +8 -10
- data/lib/wurk/limiter/bucket.rb +1 -1
- data/lib/wurk/limiter/concurrent.rb +27 -22
- data/lib/wurk/limiter/window.rb +13 -11
- data/lib/wurk/limiter.rb +7 -4
- data/lib/wurk/lua.rb +97 -14
- data/lib/wurk/manager.rb +29 -13
- data/lib/wurk/metrics/history.rb +4 -3
- data/lib/wurk/metrics/queue_rollup.rb +13 -1
- data/lib/wurk/metrics/rollup.rb +13 -1
- data/lib/wurk/middleware/interrupt_handler.rb +7 -6
- data/lib/wurk/middleware/poison_pill.rb +70 -29
- data/lib/wurk/middleware.rb +2 -2
- data/lib/wurk/pool_checkout.rb +29 -0
- data/lib/wurk/process_set.rb +10 -5
- data/lib/wurk/processor.rb +6 -0
- data/lib/wurk/profiler.rb +3 -2
- data/lib/wurk/queue.rb +10 -7
- data/lib/wurk/rails_boot.rb +38 -7
- data/lib/wurk/redis_client_adapter.rb +48 -4
- data/lib/wurk/redis_pool.rb +70 -25
- data/lib/wurk/scheduled.rb +30 -2
- data/lib/wurk/stats.rb +14 -9
- data/lib/wurk/swarm/child_boot.rb +12 -0
- data/lib/wurk/swarm.rb +174 -33
- data/lib/wurk/timer_loop.rb +14 -0
- data/lib/wurk/version.rb +1 -1
- data/lib/wurk/web/enterprise.rb +58 -6
- data/lib/wurk/web/extension.rb +1 -1
- data/lib/wurk/web/search.rb +5 -3
- data/lib/wurk.rb +10 -2
- data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-OmC5NPgw.js} +1 -1
- data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-CIpai7St.js} +1 -1
- data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-A_kwSR6Q.js} +1 -1
- data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-BG7HTqlp.js} +1 -1
- data/vendor/assets/dashboard/assets/{Dashboard-B9rOrkzk.js → Dashboard-A_ToqHoo.js} +1 -1
- data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-8J21jMyK.js} +1 -1
- data/vendor/assets/dashboard/assets/Extension-B4Q9FIQu.js +1 -0
- data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-Fh_Ae7UW.js} +1 -1
- data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-Ceng0PMB.js} +1 -1
- data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-CruDWvNZ.js} +1 -1
- data/vendor/assets/dashboard/assets/{Metrics-BBTDxcaE.js → Metrics-CIT7VCoN.js} +1 -1
- data/vendor/assets/dashboard/assets/Modal-CN3rdKA_.js +1 -0
- data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-D86FYohJ.js} +1 -1
- data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-Bz1O1D-i.js} +1 -1
- data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-B6h2akTu.js} +1 -1
- data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-OOu22e5s.js} +1 -1
- data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
- data/vendor/assets/dashboard/assets/index-D_lSDwKw.js +141 -0
- data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-dVPGEWzn.js} +1 -1
- data/vendor/assets/dashboard/index.html +2 -2
- data/vendor/assets/dashboard/wurk-manifest.json +2 -2
- metadata +21 -20
- data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
- data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
- data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
- data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
data/lib/wurk/process_set.rb
CHANGED
|
@@ -32,7 +32,7 @@ module Wurk
|
|
|
32
32
|
# absence (process never registered) and expiry (heartbeat lapsed,
|
|
33
33
|
# info field gone) both return nil.
|
|
34
34
|
def self.[](identity)
|
|
35
|
-
exists, fields = Wurk.redis do |conn|
|
|
35
|
+
exists, fields = Wurk.redis(idempotent: true) do |conn|
|
|
36
36
|
conn.pipelined do |pipe|
|
|
37
37
|
pipe.call('SISMEMBER', Keys::PROCESSES, identity)
|
|
38
38
|
pipe.call('HMGET', identity, *LOOKUP_FIELDS)
|
|
@@ -48,6 +48,11 @@ module Wurk
|
|
|
48
48
|
# don't dogpile the prune. Returns the number of identities removed
|
|
49
49
|
# (or 0 when the lock was held by someone else).
|
|
50
50
|
#
|
|
51
|
+
# No apply-safety claim, unlike the reads around it: a replay re-decides
|
|
52
|
+
# which identities are dead, so it could prune one that registered in
|
|
53
|
+
# between and hasn't written `info` yet. A blip here just skips one prune —
|
|
54
|
+
# the next caller a minute later does it.
|
|
55
|
+
#
|
|
51
56
|
# Spec: docs/target/sidekiq-free.md §31.17.
|
|
52
57
|
def cleanup
|
|
53
58
|
return 0 unless acquired_cleanup_lock?
|
|
@@ -84,7 +89,7 @@ module Wurk
|
|
|
84
89
|
# SCARD over `processes`. Not pruned — may include identities whose
|
|
85
90
|
# heartbeat has lapsed. Use `each` for the accurate count.
|
|
86
91
|
def size
|
|
87
|
-
Wurk.redis { |conn| conn.call('SCARD', Keys::PROCESSES) }
|
|
92
|
+
Wurk.redis(idempotent: true) { |conn| conn.call('SCARD', Keys::PROCESSES) }
|
|
88
93
|
end
|
|
89
94
|
|
|
90
95
|
# Sum of `concurrency` across live processes. Iterates `each` so dead
|
|
@@ -104,7 +109,7 @@ module Wurk
|
|
|
104
109
|
# `||=` with empty-string fallback distinguishes "leader is unset" from
|
|
105
110
|
# "memoization not yet computed".
|
|
106
111
|
def leader
|
|
107
|
-
@leader ||= Wurk.redis { |c| c.call('GET', 'dear-leader') } || ''
|
|
112
|
+
@leader ||= Wurk.redis(idempotent: true) { |c| c.call('GET', 'dear-leader') } || ''
|
|
108
113
|
end
|
|
109
114
|
|
|
110
115
|
class << self
|
|
@@ -134,7 +139,7 @@ module Wurk
|
|
|
134
139
|
private
|
|
135
140
|
|
|
136
141
|
def fetch_each_rows
|
|
137
|
-
Wurk.redis do |conn|
|
|
142
|
+
Wurk.redis(idempotent: true) do |conn|
|
|
138
143
|
procs = conn.call('SMEMBERS', Keys::PROCESSES).sort
|
|
139
144
|
next [] if procs.empty?
|
|
140
145
|
|
|
@@ -223,7 +228,7 @@ module Wurk
|
|
|
223
228
|
# Compares identity against the `dear-leader` STRING. Ent-only;
|
|
224
229
|
# always false in OSS/free.
|
|
225
230
|
def leader?
|
|
226
|
-
Wurk.redis { |c| c.call('GET', 'dear-leader') == identity }
|
|
231
|
+
Wurk.redis(idempotent: true) { |c| c.call('GET', 'dear-leader') == identity }
|
|
227
232
|
end
|
|
228
233
|
|
|
229
234
|
private
|
data/lib/wurk/processor.rb
CHANGED
|
@@ -164,6 +164,12 @@ module Wurk
|
|
|
164
164
|
job_hash = parse_or_kill(jobstr, uow)
|
|
165
165
|
return if job_hash.nil?
|
|
166
166
|
|
|
167
|
+
# The fetcher never parses, so hand it the jid we just read: the ACK
|
|
168
|
+
# retires this job's poison-pill recovery counter inside the round trip
|
|
169
|
+
# it already makes. A fetcher plugged in via `config[:fetch_class]` has
|
|
170
|
+
# no jid slot and simply ACKs — the counter then ages out on its 72h TTL.
|
|
171
|
+
uow.jid = job_hash['jid'] if uow.respond_to?(:jid=)
|
|
172
|
+
|
|
167
173
|
ack = false
|
|
168
174
|
begin
|
|
169
175
|
Thread.handle_interrupt(Wurk::Shutdown => :never) do
|
data/lib/wurk/profiler.rb
CHANGED
|
@@ -5,6 +5,7 @@ require 'zlib'
|
|
|
5
5
|
require 'stringio'
|
|
6
6
|
require 'tempfile'
|
|
7
7
|
require_relative 'keys'
|
|
8
|
+
require_relative 'pool_checkout'
|
|
8
9
|
|
|
9
10
|
module Wurk
|
|
10
11
|
# Job profiling (Sidekiq 8.0+, OSS). When a job is pushed with a `profile`
|
|
@@ -113,8 +114,8 @@ module Wurk
|
|
|
113
114
|
end
|
|
114
115
|
end
|
|
115
116
|
|
|
116
|
-
def with_pool(pool, &)
|
|
117
|
-
pool ?
|
|
117
|
+
def with_pool(pool, idempotent: false, &)
|
|
118
|
+
pool ? PoolCheckout.with(pool, idempotent, &) : Wurk.redis(idempotent:, &)
|
|
118
119
|
end
|
|
119
120
|
|
|
120
121
|
def now
|
data/lib/wurk/queue.rb
CHANGED
|
@@ -25,7 +25,7 @@ module Wurk
|
|
|
25
25
|
|
|
26
26
|
# @return [Array<Queue>] one per known queue, sorted by name.
|
|
27
27
|
def self.all
|
|
28
|
-
names = Wurk.redis { |conn| conn.call('SMEMBERS', Keys::QUEUES_SET) }
|
|
28
|
+
names = Wurk.redis(idempotent: true) { |conn| conn.call('SMEMBERS', Keys::QUEUES_SET) }
|
|
29
29
|
names.sort.map { |n| new(n) }
|
|
30
30
|
end
|
|
31
31
|
|
|
@@ -35,12 +35,12 @@ module Wurk
|
|
|
35
35
|
end
|
|
36
36
|
|
|
37
37
|
def size
|
|
38
|
-
Wurk.redis { |conn| conn.call('LLEN', @rname) }
|
|
38
|
+
Wurk.redis(idempotent: true) { |conn| conn.call('LLEN', @rname) }
|
|
39
39
|
end
|
|
40
40
|
|
|
41
41
|
# Seconds since the oldest job (tail of LIST) was enqueued. 0.0 when empty.
|
|
42
42
|
def latency
|
|
43
|
-
payload = Wurk.redis { |conn| conn.call('LRANGE', @rname, -1, -1).first }
|
|
43
|
+
payload = Wurk.redis(idempotent: true) { |conn| conn.call('LRANGE', @rname, -1, -1).first }
|
|
44
44
|
return 0.0 if payload.nil?
|
|
45
45
|
|
|
46
46
|
JobRecord.latency_from(Wurk.load_json(payload)['enqueued_at'])
|
|
@@ -51,19 +51,19 @@ module Wurk
|
|
|
51
51
|
# True iff this queue's name is a member of the `paused` SET. Wurk
|
|
52
52
|
# implements the Pro contract for free; fetchers consult the same set.
|
|
53
53
|
def paused?
|
|
54
|
-
Wurk.redis { |conn| conn.call('SISMEMBER', Keys::PAUSED_SET, @name) } == 1
|
|
54
|
+
Wurk.redis(idempotent: true) { |conn| conn.call('SISMEMBER', Keys::PAUSED_SET, @name) } == 1
|
|
55
55
|
end
|
|
56
56
|
|
|
57
57
|
# Pause new fetches against this queue. Idempotent — `SADD` returns
|
|
58
58
|
# 0 when the name was already present. In-flight jobs are untouched.
|
|
59
59
|
def pause! # rubocop:disable Naming/PredicateMethod
|
|
60
|
-
Wurk.redis { |conn| conn.call('SADD', Keys::PAUSED_SET, @name) }
|
|
60
|
+
Wurk.redis(idempotent: true) { |conn| conn.call('SADD', Keys::PAUSED_SET, @name) }
|
|
61
61
|
true
|
|
62
62
|
end
|
|
63
63
|
|
|
64
64
|
# Resume fetches. Idempotent.
|
|
65
65
|
def unpause! # rubocop:disable Naming/PredicateMethod
|
|
66
|
-
Wurk.redis { |conn| conn.call('SREM', Keys::PAUSED_SET, @name) }
|
|
66
|
+
Wurk.redis(idempotent: true) { |conn| conn.call('SREM', Keys::PAUSED_SET, @name) }
|
|
67
67
|
true
|
|
68
68
|
end
|
|
69
69
|
|
|
@@ -74,7 +74,7 @@ module Wurk
|
|
|
74
74
|
loop do
|
|
75
75
|
start = page * PAGE_SIZE
|
|
76
76
|
stop = start + PAGE_SIZE - 1
|
|
77
|
-
slice = Wurk.redis { |conn| conn.call('LRANGE', @rname, start, stop) }
|
|
77
|
+
slice = Wurk.redis(idempotent: true) { |conn| conn.call('LRANGE', @rname, start, stop) }
|
|
78
78
|
slice.each { |value| yield JobRecord.new(value, @name) }
|
|
79
79
|
break if slice.size < PAGE_SIZE
|
|
80
80
|
|
|
@@ -91,6 +91,9 @@ module Wurk
|
|
|
91
91
|
# UNLINK the list + drop the queue from the `queues` set. Pipelined
|
|
92
92
|
# so a partial failure leaves at most one of the two ops applied.
|
|
93
93
|
# Method name is Sidekiq wire-compat — `clear?` would break the alias.
|
|
94
|
+
#
|
|
95
|
+
# Unlike pause!/unpause!, this one can't claim apply-safety: a replay after
|
|
96
|
+
# a lost reply would UNLINK whatever a producer enqueued in between.
|
|
94
97
|
def clear # rubocop:disable Naming/PredicateMethod
|
|
95
98
|
Wurk.redis do |conn|
|
|
96
99
|
conn.pipelined do |pipe|
|
data/lib/wurk/rails_boot.rb
CHANGED
|
@@ -8,6 +8,11 @@ module Wurk
|
|
|
8
8
|
# boot policy stays pure and unit-testable without the Railtie DSL.
|
|
9
9
|
# See docs/idea/03-process-model.md for the exact ordering.
|
|
10
10
|
module RailsBoot
|
|
11
|
+
# What `at_exit` waits for the supervise thread on top of the swarm's own
|
|
12
|
+
# drain budget: one supervise tick to notice the request, plus slack for
|
|
13
|
+
# the final reap.
|
|
14
|
+
DRAIN_JOIN_SLACK = 1
|
|
15
|
+
|
|
11
16
|
module_function
|
|
12
17
|
|
|
13
18
|
# Invoked from the `wurk.server_mode` initializer, before config/initializers
|
|
@@ -103,36 +108,62 @@ module Wurk
|
|
|
103
108
|
end
|
|
104
109
|
|
|
105
110
|
def boot_swarm
|
|
106
|
-
|
|
107
|
-
|
|
111
|
+
timeout = Wurk.configuration[:timeout] || Swarm::DEFAULT_SHUTDOWN_TIMEOUT
|
|
112
|
+
swarm = Wurk::Swarm.new(topology: Wurk.configuration.topology, shutdown_timeout: timeout)
|
|
113
|
+
supervisor = nil
|
|
114
|
+
# Registered BEFORE boot, which forks the children one slot at a time: a
|
|
115
|
+
# fork that raises partway through would otherwise leave the children it
|
|
116
|
+
# already spawned with nothing to drain them on host exit. Children
|
|
117
|
+
# inherit the hook — stop_swarm no-ops off the process that forked them.
|
|
118
|
+
at_exit { stop_swarm(swarm, supervisor, timeout + Swarm::SHUTDOWN_GRACE + DRAIN_JOIN_SLACK) }
|
|
108
119
|
# Co-hosted in the web process (e.g. Puma single mode): the host owns the
|
|
109
120
|
# process-wide TERM/INT traps. Installing the swarm's own would hijack
|
|
110
121
|
# them — a deploy TERM would drain the swarm but never stop the HTTP
|
|
111
122
|
# server. Let the host keep signal ownership and drain the swarm on its
|
|
112
123
|
# graceful exit (same contract as boot_embedded).
|
|
113
124
|
swarm.boot(install_signals: false)
|
|
114
|
-
at_exit { swarm.shutdown }
|
|
115
125
|
# supervise must still run somewhere or crashed children never respawn and
|
|
116
126
|
# memory checks never fire. A background thread keeps the host's main
|
|
117
127
|
# thread free to serve HTTP.
|
|
118
|
-
Thread.new do
|
|
128
|
+
supervisor = Thread.new do
|
|
119
129
|
swarm.supervise
|
|
120
130
|
rescue StandardError => e
|
|
121
131
|
logger.error { "wurk supervisor thread died: #{e.class}: #{e.message}" }
|
|
122
132
|
end
|
|
123
133
|
end
|
|
124
134
|
|
|
135
|
+
# at_exit fires on the host's main thread, but the supervise thread owns the
|
|
136
|
+
# child table and two threads inside `shutdown` race on it — so request the
|
|
137
|
+
# drain and wait for the supervisor to run it. Draining here is the fallback
|
|
138
|
+
# for when no live supervisor will: it never started (boot raised) or it
|
|
139
|
+
# died early; after one that drained it finds no children and no-ops. A
|
|
140
|
+
# supervisor still alive past the join is wedged mid-drain — leave it be
|
|
141
|
+
# rather than race it; its children self-terminate (OrphanGuard) once this
|
|
142
|
+
# process is gone.
|
|
143
|
+
def stop_swarm(swarm, supervisor, drain_wait)
|
|
144
|
+
return unless swarm.owner?
|
|
145
|
+
|
|
146
|
+
swarm.request_shutdown
|
|
147
|
+
supervisor&.join(drain_wait)
|
|
148
|
+
swarm.shutdown unless supervisor&.alive?
|
|
149
|
+
end
|
|
150
|
+
|
|
125
151
|
# Sidekiq-embedded parity: a threads-only worker inside the web process, no
|
|
126
152
|
# fork. Redis validation failure keeps the host serving HTTP (log + carry
|
|
127
153
|
# on). at_exit drains on the graceful shutdown the web server runs on TERM.
|
|
128
|
-
def boot_embedded
|
|
129
|
-
instance =
|
|
130
|
-
|
|
154
|
+
def boot_embedded(embedded = Wurk::Embedded)
|
|
155
|
+
instance = embedded.new(Wurk.configuration)
|
|
156
|
+
# Registered BEFORE run, for the reason boot_swarm registers before the
|
|
157
|
+
# fork: run brings the heartbeat, pollers, managers and health listener up
|
|
158
|
+
# one at a time, and a raise partway through would otherwise leave the ones
|
|
159
|
+
# already running with nothing to drain them on host exit.
|
|
131
160
|
at_exit { instance.stop }
|
|
161
|
+
instance.run
|
|
132
162
|
logger.info { 'wurk: running embedded in the web process (config.wurk.embed_in_web) — threads only, no fork' }
|
|
133
163
|
instance
|
|
134
164
|
rescue StandardError => e
|
|
135
165
|
logger.error { "wurk: embedded boot failed: #{e.class}: #{e.message}" }
|
|
166
|
+
instance&.stop
|
|
136
167
|
nil
|
|
137
168
|
end
|
|
138
169
|
|
|
@@ -21,13 +21,18 @@ module Wurk
|
|
|
21
21
|
|
|
22
22
|
DEPRECATED_COMMANDS = %i[rpoplpush zrangebyscore zrevrange zrevrangebyscore getset hmset setex setnx].to_set
|
|
23
23
|
|
|
24
|
+
# Every method here dispatches through `call` rather than `@client.call` so
|
|
25
|
+
# the round-trip odometer below sees method-style commands too — a host
|
|
26
|
+
# block doing `conn.sadd(...)` then `conn.lpush(...)` has to be as
|
|
27
|
+
# replay-protected as one written with `conn.call`. On the pipeline
|
|
28
|
+
# decorator `call` is the same buffered forward it always was.
|
|
24
29
|
module CompatMethods
|
|
25
30
|
def info
|
|
26
|
-
|
|
31
|
+
call('INFO') { |i| i.lines(chomp: true).map { |l| l.split(':', 2) }.select { |l| l.size == 2 }.to_h }
|
|
27
32
|
end
|
|
28
33
|
|
|
29
34
|
def evalsha(sha, keys, argv)
|
|
30
|
-
|
|
35
|
+
call('EVALSHA', sha, keys.size, *keys, *argv)
|
|
31
36
|
end
|
|
32
37
|
|
|
33
38
|
# The Redis commands Sidekiq itself uses — defined eagerly so the
|
|
@@ -41,7 +46,7 @@ module Wurk
|
|
|
41
46
|
|
|
42
47
|
USED_COMMANDS.each do |name|
|
|
43
48
|
define_method(name) do |*args, **kwargs|
|
|
44
|
-
|
|
49
|
+
call(name, *args, **kwargs)
|
|
45
50
|
end
|
|
46
51
|
end
|
|
47
52
|
|
|
@@ -52,7 +57,7 @@ module Wurk
|
|
|
52
57
|
if DEPRECATED_COMMANDS.include?(args.first)
|
|
53
58
|
warn("[sidekiq#5788] Redis has deprecated the `#{args.first}` command, called at #{caller(1..1)}")
|
|
54
59
|
end
|
|
55
|
-
|
|
60
|
+
call(*args, &)
|
|
56
61
|
end
|
|
57
62
|
ruby2_keywords :method_missing if respond_to?(:ruby2_keywords, true)
|
|
58
63
|
|
|
@@ -64,9 +69,48 @@ module Wurk
|
|
|
64
69
|
CompatClient = RedisClient::Decorator.create(CompatMethods)
|
|
65
70
|
|
|
66
71
|
class CompatClient
|
|
72
|
+
# Dispatch methods that put commands on the wire and wait for the reply.
|
|
73
|
+
# SCAN and friends are left out on purpose: they are pure reads, so
|
|
74
|
+
# re-running one applies nothing and cannot make a replay unsafe.
|
|
75
|
+
DISPATCH_METHODS = %i[call call_v call_once call_once_v
|
|
76
|
+
blocking_call blocking_call_v pipelined multi].freeze
|
|
77
|
+
|
|
78
|
+
# Round trips this connection has completed, monotonic for its whole life.
|
|
79
|
+
#
|
|
80
|
+
# {Wurk::RedisPool} replays a failed block only while it can prove nothing
|
|
81
|
+
# in it applied, and a connect-phase error proves that for the command
|
|
82
|
+
# that raised — never for the ones before it. redis-client re-dials a
|
|
83
|
+
# dropped socket mid-block, so `CannotConnectError` surfaces on the second
|
|
84
|
+
# pipeline of a block whose first one already landed. The pool snapshots
|
|
85
|
+
# this counter around the block and refuses the replay once it has moved.
|
|
86
|
+
attr_reader :round_trips
|
|
87
|
+
|
|
88
|
+
def initialize(client)
|
|
89
|
+
super
|
|
90
|
+
@round_trips = 0
|
|
91
|
+
end
|
|
92
|
+
|
|
67
93
|
def config
|
|
68
94
|
@client.config
|
|
69
95
|
end
|
|
96
|
+
|
|
97
|
+
# Counted *after* the call returns: a command that never reached a server
|
|
98
|
+
# leaves the odometer where it was, which is what keeps the pre-apply
|
|
99
|
+
# backoff alive for blocks that failed on their very first round trip.
|
|
100
|
+
DISPATCH_METHODS.each do |name|
|
|
101
|
+
class_eval(<<~RUBY, __FILE__, __LINE__ + 1)
|
|
102
|
+
# def call(...)
|
|
103
|
+
# result = super
|
|
104
|
+
# @round_trips += 1
|
|
105
|
+
# result
|
|
106
|
+
# end
|
|
107
|
+
def #{name}(...)
|
|
108
|
+
result = super
|
|
109
|
+
@round_trips += 1
|
|
110
|
+
result
|
|
111
|
+
end
|
|
112
|
+
RUBY
|
|
113
|
+
end
|
|
70
114
|
end
|
|
71
115
|
end
|
|
72
116
|
end
|
data/lib/wurk/redis_pool.rb
CHANGED
|
@@ -10,17 +10,29 @@ module Wurk
|
|
|
10
10
|
# across forks: the parent closes the pool before fork, each child opens a
|
|
11
11
|
# fresh one (see docs/idea/03-process-model.md, steps 3 and 5).
|
|
12
12
|
#
|
|
13
|
-
# #with absorbs transient Redis failures (production incident #101)
|
|
13
|
+
# #with absorbs transient Redis failures (production incident #101), but only
|
|
14
|
+
# where replaying the caller's block cannot change what the server already
|
|
15
|
+
# did — the block is arbitrary Ruby, so a replay re-issues every command in it:
|
|
14
16
|
# * READONLY / NOREPLICAS / UNBLOCKED — a failover happened; close and retry
|
|
15
17
|
# once immediately so redis-client redials the new primary (spec §26).
|
|
16
|
-
# *
|
|
17
|
-
#
|
|
18
|
-
#
|
|
19
|
-
#
|
|
18
|
+
# * CannotConnect / Failover — raised while dialing, so the command that hit
|
|
19
|
+
# one never reached a server and cannot have applied; close and retry with
|
|
20
|
+
# exponential backoff up to CONN_MAX_ATTEMPTS, then raise.
|
|
21
|
+
# * Read-/WriteTimeout and bare ConnectionError — the command may already
|
|
22
|
+
# have applied server-side, so these raise. Replaying would double-push a
|
|
23
|
+
# job, double-count a stat, or drop a member ZPOPed by the lost reply.
|
|
24
|
+
# Blocks that are safe to re-run (pure reads, an LMOVE the reaper reclaims,
|
|
25
|
+
# owner-CAS scripts) opt back into the backoff with `with(idempotent: true)`.
|
|
20
26
|
# * ConnectionPool::TimeoutError — checkout starved; retry once after a
|
|
21
27
|
# short jittered pause, then raise (sizing is the fix, not queuing).
|
|
22
|
-
#
|
|
23
|
-
#
|
|
28
|
+
# Those proofs are about the command that raised, not the block around it: a
|
|
29
|
+
# block is several round trips, and redis-client re-dials mid-block, so a
|
|
30
|
+
# CannotConnect can surface on the second pipeline of a block whose first one
|
|
31
|
+
# already landed. So the pool also watches the connection's round-trip
|
|
32
|
+
# odometer (RedisClientAdapter::CompatClient#round_trips) and refuses to
|
|
33
|
+
# replay a non-idempotent block that has already completed one.
|
|
34
|
+
# Every retry, refused replay, and final give-up is reported through the
|
|
35
|
+
# injected `on_error` telemetry hook (Wurk::Configuration#on_redis_error).
|
|
24
36
|
class RedisPool
|
|
25
37
|
DEFAULT_URL = ENV.fetch('REDIS_URL', 'redis://localhost:6379/0')
|
|
26
38
|
DEFAULT_NAME = 'default'
|
|
@@ -32,7 +44,9 @@ module Wurk
|
|
|
32
44
|
# wait above. read/write are deliberately wider than connect so a briefly-
|
|
33
45
|
# slow-but-alive Redis (RDB fork pause, a large BLMOVE payload) doesn't
|
|
34
46
|
# spuriously ReadTimeout — the production incident (#101) the single
|
|
35
|
-
# dual-use timeout caused. reconnect_attempts re-dials a dropped socket once
|
|
47
|
+
# dual-use timeout caused. reconnect_attempts re-dials a dropped socket once;
|
|
48
|
+
# note that redis-client's re-dial also re-sends the one in-flight command,
|
|
49
|
+
# so the apply-safety split below bounds block replay, not command replay.
|
|
36
50
|
DEFAULT_CONNECT_TIMEOUT = 1.0
|
|
37
51
|
DEFAULT_READ_TIMEOUT = 2.5
|
|
38
52
|
DEFAULT_WRITE_TIMEOUT = 2.5
|
|
@@ -53,6 +67,17 @@ module Wurk
|
|
|
53
67
|
# backoff below (otherwise a failover would sleep instead of redialing).
|
|
54
68
|
RETRYABLE_MSG = /\A(READONLY|NOREPLICAS|UNBLOCKED)/
|
|
55
69
|
|
|
70
|
+
# ConnectionErrors that can only be raised while dialing, so the block
|
|
71
|
+
# provably never applied and replaying it is safe whatever it contains.
|
|
72
|
+
# CannotConnect covers every connect-phase failure (redis-client converts a
|
|
73
|
+
# stalled handshake into it); Failover is the Sentinel resolver rejecting a
|
|
74
|
+
# server whose role changed. Any other ConnectionError — Read-/WriteTimeout
|
|
75
|
+
# or a reset mid-command — leaves the outcome unknown.
|
|
76
|
+
PRE_APPLY_ERRORS = [RedisClient::CannotConnectError, RedisClient::FailoverError].freeze
|
|
77
|
+
|
|
78
|
+
# The two retry_plan verdicts that re-run the block; the rest raise.
|
|
79
|
+
REPLAY_PLANS = %i[failover backoff].freeze
|
|
80
|
+
|
|
56
81
|
# ConnectionError backoff: CONN_MAX_ATTEMPTS total tries, sleeping
|
|
57
82
|
# (BASE * 2**attempt) + rand*JITTER before each retry. The 1.0s + 2.0s pair
|
|
58
83
|
# rides out a sub-4s blip; the jitter de-syncs a fleet reconnecting at once.
|
|
@@ -89,10 +114,14 @@ module Wurk
|
|
|
89
114
|
# Checkout a connection and run the block. ConnectionPool::TimeoutError is
|
|
90
115
|
# raised by @pool.with *before* the block runs, so it is caught out here
|
|
91
116
|
# (the in-block #run rescue never sees it) — one retry, then raise.
|
|
92
|
-
|
|
117
|
+
#
|
|
118
|
+
# `idempotent: true` asserts the block can be re-run after a command may
|
|
119
|
+
# already have applied server-side, which buys back the full ConnectionError
|
|
120
|
+
# backoff. Only claim it for pure reads or writes whose repeat is a no-op.
|
|
121
|
+
def with(idempotent: false, &block)
|
|
93
122
|
checkout_retried = false
|
|
94
123
|
begin
|
|
95
|
-
@pool.with { |conn| run(conn, &block) }
|
|
124
|
+
@pool.with { |conn| run(conn, idempotent, &block) }
|
|
96
125
|
rescue ConnectionPool::TimeoutError => e
|
|
97
126
|
if checkout_retried
|
|
98
127
|
notify_error(e, attempt: 2, retried: false)
|
|
@@ -113,7 +142,7 @@ module Wurk
|
|
|
113
142
|
# slot counts merged in — one call gives a heartbeat both Redis health and
|
|
114
143
|
# local pool saturation. (Real Redis INFO has no `size`/`available` field.)
|
|
115
144
|
def info
|
|
116
|
-
with { |conn| parse_info(conn.call('INFO')) }
|
|
145
|
+
with(idempotent: true) { |conn| parse_info(conn.call('INFO')) }
|
|
117
146
|
.merge('size' => @size, 'available' => available)
|
|
118
147
|
end
|
|
119
148
|
|
|
@@ -159,17 +188,19 @@ module Wurk
|
|
|
159
188
|
# Runs the block on the checked-out `conn`, retrying transient RedisClient
|
|
160
189
|
# errors in place: the same slot is reused across retries (redis-client
|
|
161
190
|
# redials a closed socket lazily), so a busy fetcher can't leak checkouts.
|
|
162
|
-
def run(conn)
|
|
191
|
+
def run(conn, idempotent)
|
|
163
192
|
attempts = 0
|
|
164
193
|
begin
|
|
165
194
|
attempts += 1
|
|
195
|
+
odometer = conn.round_trips
|
|
166
196
|
yield conn
|
|
167
197
|
rescue RedisClient::Error => e
|
|
168
|
-
plan = retry_plan(e, attempts)
|
|
198
|
+
plan = retry_plan(e, attempts, idempotent, conn.round_trips != odometer)
|
|
169
199
|
raise if plan == :propagate
|
|
170
200
|
|
|
171
|
-
|
|
172
|
-
|
|
201
|
+
replaying = REPLAY_PLANS.include?(plan)
|
|
202
|
+
notify_error(e, attempt: attempts, retried: replaying)
|
|
203
|
+
raise unless replaying
|
|
173
204
|
|
|
174
205
|
safe_close(conn)
|
|
175
206
|
sleep(backoff_delay(attempts)) if plan == :backoff
|
|
@@ -177,19 +208,33 @@ module Wurk
|
|
|
177
208
|
end
|
|
178
209
|
end
|
|
179
210
|
|
|
180
|
-
# Pure classification of a RedisClient error against the attempt count
|
|
211
|
+
# Pure classification of a RedisClient error against the attempt count, the
|
|
212
|
+
# caller's apply-safety claim, and whether this attempt already completed a
|
|
213
|
+
# round trip (`dirty`):
|
|
181
214
|
# :failover → close + immediate retry (a primary swap)
|
|
182
215
|
# :backoff → close + sleep + retry (a connection blip)
|
|
183
|
-
# :
|
|
216
|
+
# :unsafe → something may have applied; report the blip, then raise
|
|
217
|
+
# :exhausted → replayable ConnectionError past the cap; report and give up
|
|
184
218
|
# :propagate → not transient (or a spent failover); raise as-is
|
|
185
|
-
def retry_plan(err, attempts)
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
219
|
+
def retry_plan(err, attempts, idempotent, dirty)
|
|
220
|
+
failover = RETRYABLE_MSG.match?(err.message.to_s)
|
|
221
|
+
return :propagate unless failover || err.is_a?(RedisClient::ConnectionError)
|
|
222
|
+
return :unsafe unless replayable?(err, idempotent, dirty, failover)
|
|
223
|
+
return attempts > 1 ? :propagate : :failover if failover
|
|
224
|
+
|
|
225
|
+
attempts >= CONN_MAX_ATTEMPTS ? :exhausted : :backoff
|
|
226
|
+
end
|
|
227
|
+
|
|
228
|
+
# A replay re-issues the whole block, so one completed round trip voids
|
|
229
|
+
# every pre-apply proof: however provably the *failing* command missed the
|
|
230
|
+
# server, the ones ahead of it in the block did not. On a still-clean block
|
|
231
|
+
# a failover reply is proof enough by itself (the command was rejected
|
|
232
|
+
# outright); a bare ConnectionError needs one of the connect-phase classes.
|
|
233
|
+
def replayable?(err, idempotent, dirty, failover)
|
|
234
|
+
return true if idempotent
|
|
235
|
+
return false if dirty
|
|
236
|
+
|
|
237
|
+
failover || PRE_APPLY_ERRORS.any? { |klass| err.is_a?(klass) }
|
|
193
238
|
end
|
|
194
239
|
|
|
195
240
|
def notify_error(error, attempt:, retried:)
|
data/lib/wurk/scheduled.rb
CHANGED
|
@@ -6,6 +6,7 @@ require_relative 'lua'
|
|
|
6
6
|
require_relative 'lua/loader'
|
|
7
7
|
require_relative 'client'
|
|
8
8
|
require_relative 'process_set'
|
|
9
|
+
require_relative 'timer_loop'
|
|
9
10
|
|
|
10
11
|
module Wurk
|
|
11
12
|
# Promotes due jobs from the `retry` and `schedule` sorted sets back onto
|
|
@@ -57,13 +58,30 @@ module Wurk
|
|
|
57
58
|
loop do
|
|
58
59
|
break if @done
|
|
59
60
|
|
|
60
|
-
jobstr =
|
|
61
|
+
jobstr = pop_due(sset, now)
|
|
61
62
|
break unless jobstr
|
|
62
63
|
|
|
63
64
|
push_promoted(jobstr, sset)
|
|
64
65
|
end
|
|
65
66
|
end
|
|
66
67
|
|
|
68
|
+
# ZPOPBYSCORE is destructive and carries its result in the reply, so this
|
|
69
|
+
# block never claims apply-safety: a replay discards whatever the lost
|
|
70
|
+
# reply already removed. The pool therefore raises on a Read-/WriteTimeout,
|
|
71
|
+
# which leaves the outcome of *this* pop unknown — a due job may or may not
|
|
72
|
+
# have come off the ZSET. Report it and end this set's drain (the nil makes
|
|
73
|
+
# #drain_set break) rather than pop again blind; a job caught in that window
|
|
74
|
+
# falls into the same pop→push loss the default scheduler already documents
|
|
75
|
+
# on #push_promoted, and `reliable_scheduler!` (ReliableEnq) is the loss-free
|
|
76
|
+
# fix. Rescuing here rather than around #drain_set keeps the sibling set
|
|
77
|
+
# draining on this tick.
|
|
78
|
+
def pop_due(sset, now)
|
|
79
|
+
@config.redis { |conn| Wurk::Lua::Loader.eval_cached(conn, :zpopbyscore, keys: [sset], argv: [now]) }
|
|
80
|
+
rescue RedisClient::ConnectionError => e
|
|
81
|
+
handle_exception(e, { context: 'scheduler_pop', set: sset })
|
|
82
|
+
nil
|
|
83
|
+
end
|
|
84
|
+
|
|
67
85
|
# A raising `@client.push` (bad payload, transient Redis error) must not
|
|
68
86
|
# abort the drain and strand the remaining due jobs until the next poll —
|
|
69
87
|
# rescue per-job, report, continue. The already-popped job IS lost here
|
|
@@ -169,13 +187,23 @@ module Wurk
|
|
|
169
187
|
|
|
170
188
|
# Idempotent. Wakes the sleeping thread so it observes @done and exits.
|
|
171
189
|
# Also propagates the stop signal to @enq so any in-flight drain loop
|
|
172
|
-
# short-circuits instead of running to completion.
|
|
190
|
+
# short-circuits instead of running to completion. Terminal, not a pause:
|
|
191
|
+
# @enq's stop flag is one-way, so this poller never polls again.
|
|
192
|
+
#
|
|
193
|
+
# Joins before returning — the caller (Launcher#quiet, then #stop) clears
|
|
194
|
+
# the heartbeat right after, and a sweep still in flight would promote
|
|
195
|
+
# jobs on behalf of a process that no longer exists.
|
|
196
|
+
#
|
|
197
|
+
# Cleared only on a confirmed join (Thread#join returns nil on timeout):
|
|
198
|
+
# a wedged sweep must stay tracked so #start's ||= guard returns it
|
|
199
|
+
# rather than spawning a second scheduler thread alongside it.
|
|
173
200
|
def terminate
|
|
174
201
|
@mutex.synchronize do
|
|
175
202
|
@done = true
|
|
176
203
|
@enq.terminate
|
|
177
204
|
@sleeper.signal
|
|
178
205
|
end
|
|
206
|
+
@thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
|
|
179
207
|
end
|
|
180
208
|
|
|
181
209
|
# Called on every wake. Any raise inside the Enq is reported and the
|
data/lib/wurk/stats.rb
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require 'date'
|
|
4
|
+
require_relative 'pool_checkout'
|
|
4
5
|
|
|
5
6
|
module Wurk
|
|
6
7
|
# Read-only inspector for cluster state in Redis. The cheap counters are
|
|
@@ -38,7 +39,7 @@ module Wurk
|
|
|
38
39
|
# Sum of the `busy` HASH field across every live process identity.
|
|
39
40
|
# Pipelined but unbounded by process count.
|
|
40
41
|
def workers_size
|
|
41
|
-
Wurk.redis do |conn|
|
|
42
|
+
Wurk.redis(idempotent: true) do |conn|
|
|
42
43
|
identities = conn.call('SMEMBERS', Keys::PROCESSES)
|
|
43
44
|
next 0 if identities.empty?
|
|
44
45
|
|
|
@@ -55,7 +56,7 @@ module Wurk
|
|
|
55
56
|
# and dashboards reading this rely on that order. Match it exactly with
|
|
56
57
|
# `sort_by { |_, size| -size }`.
|
|
57
58
|
def queues
|
|
58
|
-
Wurk.redis do |conn|
|
|
59
|
+
Wurk.redis(idempotent: true) do |conn|
|
|
59
60
|
names = conn.call('SMEMBERS', Keys::QUEUES_SET)
|
|
60
61
|
next {} if names.empty?
|
|
61
62
|
|
|
@@ -71,7 +72,7 @@ module Wurk
|
|
|
71
72
|
# (`queue_summaries.sort_by { |qd| -qd.size }`) — this feeds the
|
|
72
73
|
# dashboard's queue table (api_controller#queues).
|
|
73
74
|
def queue_summaries
|
|
74
|
-
Wurk.redis do |conn|
|
|
75
|
+
Wurk.redis(idempotent: true) do |conn|
|
|
75
76
|
names = conn.call('SMEMBERS', Keys::QUEUES_SET)
|
|
76
77
|
next [] if names.empty?
|
|
77
78
|
|
|
@@ -89,13 +90,17 @@ module Wurk
|
|
|
89
90
|
# Latency (secs) of the `default` queue — the most-asked-about gauge.
|
|
90
91
|
def default_queue_latency
|
|
91
92
|
now_ms = ::Process.clock_gettime(::Process::CLOCK_REALTIME, :millisecond)
|
|
92
|
-
payload = Wurk.redis { |c| c.call('LRANGE', Keys.queue('default'), -1, -1) }.first
|
|
93
|
+
payload = Wurk.redis(idempotent: true) { |c| c.call('LRANGE', Keys.queue('default'), -1, -1) }.first
|
|
93
94
|
compute_latency(payload, now_ms)
|
|
94
95
|
end
|
|
95
96
|
|
|
96
97
|
# Resets the named global counters. With no args, clears `processed`,
|
|
97
98
|
# `failed`, and `expired`. SET … 0 (not DEL — keeps the key around so
|
|
98
99
|
# reads stay `Integer` not `nil`).
|
|
100
|
+
#
|
|
101
|
+
# The only write here, and the only block in this class that can't claim
|
|
102
|
+
# apply-safety: a replay after a lost reply would re-zero the counters,
|
|
103
|
+
# discarding whatever the fleet counted in between.
|
|
99
104
|
def reset(*stats)
|
|
100
105
|
all = %w[failed processed expired]
|
|
101
106
|
to_clear = stats.empty? ? all : all & stats.flatten.map(&:to_s)
|
|
@@ -123,7 +128,7 @@ module Wurk
|
|
|
123
128
|
private_constant :FAST_QUERIES, :FAST_KEYS
|
|
124
129
|
|
|
125
130
|
def fetch_stats_fast!
|
|
126
|
-
raw = Wurk.redis do |conn|
|
|
131
|
+
raw = Wurk.redis(idempotent: true) do |conn|
|
|
127
132
|
conn.pipelined { |pipe| FAST_QUERIES.each { |args| pipe.call(*args) } }
|
|
128
133
|
end
|
|
129
134
|
@stats = FAST_KEYS.zip(raw.map(&:to_i)).to_h
|
|
@@ -179,17 +184,17 @@ module Wurk
|
|
|
179
184
|
|
|
180
185
|
def date_stat_hash(stat)
|
|
181
186
|
keys = (0...@days_previous).map { |i| (@start_date - i).strftime('%Y-%m-%d') }
|
|
182
|
-
values = with_redis do |conn|
|
|
187
|
+
values = with_redis(idempotent: true) do |conn|
|
|
183
188
|
conn.pipelined { |pipe| keys.each { |d| pipe.call('GET', "stat:#{stat}:#{d}") } }
|
|
184
189
|
end
|
|
185
190
|
keys.zip(values.map(&:to_i)).to_h
|
|
186
191
|
end
|
|
187
192
|
|
|
188
|
-
def with_redis(&)
|
|
193
|
+
def with_redis(idempotent: false, &)
|
|
189
194
|
if @pool
|
|
190
|
-
|
|
195
|
+
PoolCheckout.with(@pool, idempotent, &)
|
|
191
196
|
else
|
|
192
|
-
Wurk.redis(&)
|
|
197
|
+
Wurk.redis(idempotent:, &)
|
|
193
198
|
end
|
|
194
199
|
end
|
|
195
200
|
end
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
require_relative '../component'
|
|
4
4
|
require_relative '../launcher'
|
|
5
|
+
require_relative '../client/buffered'
|
|
5
6
|
require_relative '../fetcher/reliable'
|
|
6
7
|
require_relative '../lua'
|
|
7
8
|
require_relative 'orphan_guard'
|
|
@@ -106,8 +107,19 @@ module Wurk
|
|
|
106
107
|
|
|
107
108
|
def reconnect_after_fork
|
|
108
109
|
@config.reset_redis_pools!
|
|
110
|
+
# The reliable_push outage buffer, its drainer thread and its mutexes
|
|
111
|
+
# are process-global and were copied wholesale from the parent. The
|
|
112
|
+
# Process._fork hook normally beats us to it (making this a no-op) —
|
|
113
|
+
# the explicit call keeps the swarm path deterministic and ordered
|
|
114
|
+
# after the pool reset, so a re-armed drainer can only ever see the
|
|
115
|
+
# child's own pool.
|
|
116
|
+
Wurk::Client::Buffered.reset_after_fork!
|
|
109
117
|
validate_redis!
|
|
110
118
|
reconnect_active_record
|
|
119
|
+
# The dogstatsd client is memoized at the class level (Statsd.client),
|
|
120
|
+
# so without a reset every child would share the parent's UDP socket
|
|
121
|
+
# and thread-locals instead of building its own after fork.
|
|
122
|
+
Wurk::Metrics::Statsd.reset!
|
|
111
123
|
end
|
|
112
124
|
|
|
113
125
|
# Prove the child's fresh Redis socket reaches a live server before it
|