wurk 1.3.0 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +1 -0
- data/lib/wurk/batch/callbacks.rb +82 -12
- data/lib/wurk/batch/death_handler.rb +7 -4
- data/lib/wurk/batch/server_middleware.rb +1 -1
- data/lib/wurk/batch.rb +121 -15
- data/lib/wurk/capsule.rb +5 -4
- data/lib/wurk/cli.rb +48 -14
- data/lib/wurk/client/buffered.rb +193 -43
- data/lib/wurk/client.rb +87 -14
- data/lib/wurk/compat.rb +1 -1
- data/lib/wurk/component.rb +2 -2
- data/lib/wurk/configuration.rb +10 -2
- data/lib/wurk/cron.rb +94 -37
- data/lib/wurk/deploy.rb +5 -3
- data/lib/wurk/embedded.rb +13 -0
- data/lib/wurk/fetcher/reaper.rb +113 -56
- data/lib/wurk/fetcher/reliable.rb +62 -9
- data/lib/wurk/heartbeat.rb +22 -10
- data/lib/wurk/history.rb +13 -1
- data/lib/wurk/launcher.rb +133 -66
- data/lib/wurk/leader.rb +29 -10
- data/lib/wurk/limiter/base.rb +8 -10
- data/lib/wurk/limiter/bucket.rb +1 -1
- data/lib/wurk/limiter/concurrent.rb +27 -22
- data/lib/wurk/limiter/window.rb +13 -11
- data/lib/wurk/limiter.rb +7 -4
- data/lib/wurk/lua.rb +97 -14
- data/lib/wurk/manager.rb +29 -13
- data/lib/wurk/metrics/history.rb +4 -3
- data/lib/wurk/metrics/queue_rollup.rb +13 -1
- data/lib/wurk/metrics/rollup.rb +13 -1
- data/lib/wurk/middleware/interrupt_handler.rb +7 -6
- data/lib/wurk/middleware/poison_pill.rb +70 -29
- data/lib/wurk/middleware.rb +2 -2
- data/lib/wurk/pool_checkout.rb +29 -0
- data/lib/wurk/process_set.rb +10 -5
- data/lib/wurk/processor.rb +6 -0
- data/lib/wurk/profiler.rb +3 -2
- data/lib/wurk/queue.rb +10 -7
- data/lib/wurk/rails_boot.rb +38 -7
- data/lib/wurk/redis_client_adapter.rb +48 -4
- data/lib/wurk/redis_options.rb +142 -0
- data/lib/wurk/redis_pool.rb +102 -39
- data/lib/wurk/scheduled.rb +30 -2
- data/lib/wurk/stats.rb +14 -9
- data/lib/wurk/swarm/child_boot.rb +12 -0
- data/lib/wurk/swarm.rb +174 -33
- data/lib/wurk/timer_loop.rb +14 -0
- data/lib/wurk/version.rb +1 -1
- data/lib/wurk/web/enterprise.rb +58 -6
- data/lib/wurk/web/extension.rb +1 -1
- data/lib/wurk/web/search.rb +5 -3
- data/lib/wurk.rb +53 -2
- data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-OmC5NPgw.js} +1 -1
- data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-CIpai7St.js} +1 -1
- data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-A_kwSR6Q.js} +1 -1
- data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-BG7HTqlp.js} +1 -1
- data/vendor/assets/dashboard/assets/{Dashboard-B9rOrkzk.js → Dashboard-A_ToqHoo.js} +1 -1
- data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-8J21jMyK.js} +1 -1
- data/vendor/assets/dashboard/assets/Extension-B4Q9FIQu.js +1 -0
- data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-Fh_Ae7UW.js} +1 -1
- data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-Ceng0PMB.js} +1 -1
- data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-CruDWvNZ.js} +1 -1
- data/vendor/assets/dashboard/assets/{Metrics-BBTDxcaE.js → Metrics-CIT7VCoN.js} +1 -1
- data/vendor/assets/dashboard/assets/Modal-CN3rdKA_.js +1 -0
- data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-D86FYohJ.js} +1 -1
- data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-Bz1O1D-i.js} +1 -1
- data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-B6h2akTu.js} +1 -1
- data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-OOu22e5s.js} +1 -1
- data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
- data/vendor/assets/dashboard/assets/index-D_lSDwKw.js +141 -0
- data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-dVPGEWzn.js} +1 -1
- data/vendor/assets/dashboard/index.html +2 -2
- data/vendor/assets/dashboard/wurk-manifest.json +2 -2
- metadata +22 -20
- data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
- data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
- data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
- data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
data/lib/wurk/redis_pool.rb
CHANGED
|
@@ -3,23 +3,36 @@
|
|
|
3
3
|
require 'redis-client'
|
|
4
4
|
require 'connection_pool'
|
|
5
5
|
require_relative 'redis_client_adapter'
|
|
6
|
+
require_relative 'redis_options'
|
|
6
7
|
|
|
7
8
|
module Wurk
|
|
8
9
|
# Per-process pool over redis-client + connection_pool. Never share a socket
|
|
9
10
|
# across forks: the parent closes the pool before fork, each child opens a
|
|
10
11
|
# fresh one (see docs/idea/03-process-model.md, steps 3 and 5).
|
|
11
12
|
#
|
|
12
|
-
# #with absorbs transient Redis failures (production incident #101)
|
|
13
|
+
# #with absorbs transient Redis failures (production incident #101), but only
|
|
14
|
+
# where replaying the caller's block cannot change what the server already
|
|
15
|
+
# did — the block is arbitrary Ruby, so a replay re-issues every command in it:
|
|
13
16
|
# * READONLY / NOREPLICAS / UNBLOCKED — a failover happened; close and retry
|
|
14
17
|
# once immediately so redis-client redials the new primary (spec §26).
|
|
15
|
-
# *
|
|
16
|
-
#
|
|
17
|
-
#
|
|
18
|
-
#
|
|
18
|
+
# * CannotConnect / Failover — raised while dialing, so the command that hit
|
|
19
|
+
# one never reached a server and cannot have applied; close and retry with
|
|
20
|
+
# exponential backoff up to CONN_MAX_ATTEMPTS, then raise.
|
|
21
|
+
# * Read-/WriteTimeout and bare ConnectionError — the command may already
|
|
22
|
+
# have applied server-side, so these raise. Replaying would double-push a
|
|
23
|
+
# job, double-count a stat, or drop a member ZPOPed by the lost reply.
|
|
24
|
+
# Blocks that are safe to re-run (pure reads, an LMOVE the reaper reclaims,
|
|
25
|
+
# owner-CAS scripts) opt back into the backoff with `with(idempotent: true)`.
|
|
19
26
|
# * ConnectionPool::TimeoutError — checkout starved; retry once after a
|
|
20
27
|
# short jittered pause, then raise (sizing is the fix, not queuing).
|
|
21
|
-
#
|
|
22
|
-
#
|
|
28
|
+
# Those proofs are about the command that raised, not the block around it: a
|
|
29
|
+
# block is several round trips, and redis-client re-dials mid-block, so a
|
|
30
|
+
# CannotConnect can surface on the second pipeline of a block whose first one
|
|
31
|
+
# already landed. So the pool also watches the connection's round-trip
|
|
32
|
+
# odometer (RedisClientAdapter::CompatClient#round_trips) and refuses to
|
|
33
|
+
# replay a non-idempotent block that has already completed one.
|
|
34
|
+
# Every retry, refused replay, and final give-up is reported through the
|
|
35
|
+
# injected `on_error` telemetry hook (Wurk::Configuration#on_redis_error).
|
|
23
36
|
class RedisPool
|
|
24
37
|
DEFAULT_URL = ENV.fetch('REDIS_URL', 'redis://localhost:6379/0')
|
|
25
38
|
DEFAULT_NAME = 'default'
|
|
@@ -31,18 +44,40 @@ module Wurk
|
|
|
31
44
|
# wait above. read/write are deliberately wider than connect so a briefly-
|
|
32
45
|
# slow-but-alive Redis (RDB fork pause, a large BLMOVE payload) doesn't
|
|
33
46
|
# spuriously ReadTimeout — the production incident (#101) the single
|
|
34
|
-
# dual-use timeout caused. reconnect_attempts re-dials a dropped socket once
|
|
47
|
+
# dual-use timeout caused. reconnect_attempts re-dials a dropped socket once;
|
|
48
|
+
# note that redis-client's re-dial also re-sends the one in-flight command,
|
|
49
|
+
# so the apply-safety split below bounds block replay, not command replay.
|
|
35
50
|
DEFAULT_CONNECT_TIMEOUT = 1.0
|
|
36
51
|
DEFAULT_READ_TIMEOUT = 2.5
|
|
37
52
|
DEFAULT_WRITE_TIMEOUT = 2.5
|
|
38
53
|
DEFAULT_RECONNECT_ATTEMPTS = 1
|
|
39
54
|
|
|
55
|
+
# The floor every pool starts from; any key the host passed wins over it.
|
|
56
|
+
DEFAULT_CLIENT_CONFIG = {
|
|
57
|
+
url: DEFAULT_URL,
|
|
58
|
+
connect_timeout: DEFAULT_CONNECT_TIMEOUT,
|
|
59
|
+
read_timeout: DEFAULT_READ_TIMEOUT,
|
|
60
|
+
write_timeout: DEFAULT_WRITE_TIMEOUT,
|
|
61
|
+
reconnect_attempts: DEFAULT_RECONNECT_ATTEMPTS
|
|
62
|
+
}.freeze
|
|
63
|
+
|
|
40
64
|
# Server-side messages where the connection is closed and the block retried
|
|
41
65
|
# exactly once. READONLY is itself a RedisClient::ConnectionError subclass,
|
|
42
66
|
# so this message match must be tested BEFORE the generic ConnectionError
|
|
43
67
|
# backoff below (otherwise a failover would sleep instead of redialing).
|
|
44
68
|
RETRYABLE_MSG = /\A(READONLY|NOREPLICAS|UNBLOCKED)/
|
|
45
69
|
|
|
70
|
+
# ConnectionErrors that can only be raised while dialing, so the block
|
|
71
|
+
# provably never applied and replaying it is safe whatever it contains.
|
|
72
|
+
# CannotConnect covers every connect-phase failure (redis-client converts a
|
|
73
|
+
# stalled handshake into it); Failover is the Sentinel resolver rejecting a
|
|
74
|
+
# server whose role changed. Any other ConnectionError — Read-/WriteTimeout
|
|
75
|
+
# or a reset mid-command — leaves the outcome unknown.
|
|
76
|
+
PRE_APPLY_ERRORS = [RedisClient::CannotConnectError, RedisClient::FailoverError].freeze
|
|
77
|
+
|
|
78
|
+
# The two retry_plan verdicts that re-run the block; the rest raise.
|
|
79
|
+
REPLAY_PLANS = %i[failover backoff].freeze
|
|
80
|
+
|
|
46
81
|
# ConnectionError backoff: CONN_MAX_ATTEMPTS total tries, sleeping
|
|
47
82
|
# (BASE * 2**attempt) + rand*JITTER before each retry. The 1.0s + 2.0s pair
|
|
48
83
|
# rides out a sub-4s blip; the jitter de-syncs a fleet reconnecting at once.
|
|
@@ -60,9 +95,12 @@ module Wurk
|
|
|
60
95
|
|
|
61
96
|
# Takes the standard Sidekiq `config.redis` hash: `pool_timeout` tunes the
|
|
62
97
|
# ConnectionPool checkout; `connect_timeout`/`read_timeout`/`write_timeout`/
|
|
63
|
-
# `reconnect_attempts` plus any other key (driver, ssl_params,
|
|
64
|
-
#
|
|
65
|
-
#
|
|
98
|
+
# `reconnect_attempts` plus any other redis-client key (driver, ssl_params,
|
|
99
|
+
# sentinels, …) reach the client. Sidekiq-only spellings (`network_timeout`,
|
|
100
|
+
# `master_name`, `logger`, …) are translated or dropped by {RedisOptions};
|
|
101
|
+
# a key redis-client would reject raises there with the key named.
|
|
102
|
+
# `on_error` is an optional callable fired per retry / final give-up with
|
|
103
|
+
# { error:, attempt:, retried:, pool: }.
|
|
66
104
|
def initialize(size:, name: DEFAULT_NAME, on_error: nil, **options)
|
|
67
105
|
@size = size
|
|
68
106
|
@name = name
|
|
@@ -76,10 +114,14 @@ module Wurk
|
|
|
76
114
|
# Checkout a connection and run the block. ConnectionPool::TimeoutError is
|
|
77
115
|
# raised by @pool.with *before* the block runs, so it is caught out here
|
|
78
116
|
# (the in-block #run rescue never sees it) — one retry, then raise.
|
|
79
|
-
|
|
117
|
+
#
|
|
118
|
+
# `idempotent: true` asserts the block can be re-run after a command may
|
|
119
|
+
# already have applied server-side, which buys back the full ConnectionError
|
|
120
|
+
# backoff. Only claim it for pure reads or writes whose repeat is a no-op.
|
|
121
|
+
def with(idempotent: false, &block)
|
|
80
122
|
checkout_retried = false
|
|
81
123
|
begin
|
|
82
|
-
@pool.with { |conn| run(conn, &block) }
|
|
124
|
+
@pool.with { |conn| run(conn, idempotent, &block) }
|
|
83
125
|
rescue ConnectionPool::TimeoutError => e
|
|
84
126
|
if checkout_retried
|
|
85
127
|
notify_error(e, attempt: 2, retried: false)
|
|
@@ -100,7 +142,7 @@ module Wurk
|
|
|
100
142
|
# slot counts merged in — one call gives a heartbeat both Redis health and
|
|
101
143
|
# local pool saturation. (Real Redis INFO has no `size`/`available` field.)
|
|
102
144
|
def info
|
|
103
|
-
with { |conn| parse_info(conn.call('INFO')) }
|
|
145
|
+
with(idempotent: true) { |conn| parse_info(conn.call('INFO')) }
|
|
104
146
|
.merge('size' => @size, 'available' => available)
|
|
105
147
|
end
|
|
106
148
|
|
|
@@ -112,24 +154,29 @@ module Wurk
|
|
|
112
154
|
|
|
113
155
|
private
|
|
114
156
|
|
|
115
|
-
# Socket config forwarded to
|
|
116
|
-
# the
|
|
117
|
-
#
|
|
157
|
+
# Socket config forwarded to redis-client. RedisOptions owns the translation
|
|
158
|
+
# of the Sidekiq-shaped hash (network_timeout, master_name, pool-only keys,
|
|
159
|
+
# …) so this class stays about pooling; host-supplied keys win over the
|
|
160
|
+
# defaults.
|
|
118
161
|
def build_client_config(options)
|
|
119
|
-
|
|
120
|
-
url: DEFAULT_URL,
|
|
121
|
-
connect_timeout: DEFAULT_CONNECT_TIMEOUT,
|
|
122
|
-
read_timeout: DEFAULT_READ_TIMEOUT,
|
|
123
|
-
write_timeout: DEFAULT_WRITE_TIMEOUT,
|
|
124
|
-
reconnect_attempts: DEFAULT_RECONNECT_ATTEMPTS
|
|
125
|
-
}.merge(options.except(:pool_timeout)).freeze
|
|
162
|
+
RedisOptions.normalize(options, defaults: DEFAULT_CLIENT_CONFIG).freeze
|
|
126
163
|
end
|
|
127
164
|
|
|
128
165
|
# Wrapped in the CompatClient decorator so `Sidekiq.redis { |c| c.smembers }`
|
|
129
166
|
# method-style commands work like Sidekiq 7+ (#204). Wurk's own code paths
|
|
130
167
|
# use #call, which the decorator forwards.
|
|
131
168
|
def build_client
|
|
132
|
-
RedisClientAdapter::CompatClient.new(
|
|
169
|
+
RedisClientAdapter::CompatClient.new(redis_client_config.new_client)
|
|
170
|
+
end
|
|
171
|
+
|
|
172
|
+
# A Sentinel set is a different constructor, not a different keyword:
|
|
173
|
+
# `RedisClient.config(sentinels: [...])` raises. Sidekiq routes the same way.
|
|
174
|
+
def redis_client_config
|
|
175
|
+
if RedisOptions.sentinel?(@client_config)
|
|
176
|
+
RedisClient.sentinel(**@client_config)
|
|
177
|
+
else
|
|
178
|
+
RedisClient.config(**@client_config)
|
|
179
|
+
end
|
|
133
180
|
end
|
|
134
181
|
|
|
135
182
|
def safe_close(conn)
|
|
@@ -141,17 +188,19 @@ module Wurk
|
|
|
141
188
|
# Runs the block on the checked-out `conn`, retrying transient RedisClient
|
|
142
189
|
# errors in place: the same slot is reused across retries (redis-client
|
|
143
190
|
# redials a closed socket lazily), so a busy fetcher can't leak checkouts.
|
|
144
|
-
def run(conn)
|
|
191
|
+
def run(conn, idempotent)
|
|
145
192
|
attempts = 0
|
|
146
193
|
begin
|
|
147
194
|
attempts += 1
|
|
195
|
+
odometer = conn.round_trips
|
|
148
196
|
yield conn
|
|
149
197
|
rescue RedisClient::Error => e
|
|
150
|
-
plan = retry_plan(e, attempts)
|
|
198
|
+
plan = retry_plan(e, attempts, idempotent, conn.round_trips != odometer)
|
|
151
199
|
raise if plan == :propagate
|
|
152
200
|
|
|
153
|
-
|
|
154
|
-
|
|
201
|
+
replaying = REPLAY_PLANS.include?(plan)
|
|
202
|
+
notify_error(e, attempt: attempts, retried: replaying)
|
|
203
|
+
raise unless replaying
|
|
155
204
|
|
|
156
205
|
safe_close(conn)
|
|
157
206
|
sleep(backoff_delay(attempts)) if plan == :backoff
|
|
@@ -159,19 +208,33 @@ module Wurk
|
|
|
159
208
|
end
|
|
160
209
|
end
|
|
161
210
|
|
|
162
|
-
# Pure classification of a RedisClient error against the attempt count
|
|
211
|
+
# Pure classification of a RedisClient error against the attempt count, the
|
|
212
|
+
# caller's apply-safety claim, and whether this attempt already completed a
|
|
213
|
+
# round trip (`dirty`):
|
|
163
214
|
# :failover → close + immediate retry (a primary swap)
|
|
164
215
|
# :backoff → close + sleep + retry (a connection blip)
|
|
165
|
-
# :
|
|
216
|
+
# :unsafe → something may have applied; report the blip, then raise
|
|
217
|
+
# :exhausted → replayable ConnectionError past the cap; report and give up
|
|
166
218
|
# :propagate → not transient (or a spent failover); raise as-is
|
|
167
|
-
def retry_plan(err, attempts)
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
219
|
+
def retry_plan(err, attempts, idempotent, dirty)
|
|
220
|
+
failover = RETRYABLE_MSG.match?(err.message.to_s)
|
|
221
|
+
return :propagate unless failover || err.is_a?(RedisClient::ConnectionError)
|
|
222
|
+
return :unsafe unless replayable?(err, idempotent, dirty, failover)
|
|
223
|
+
return attempts > 1 ? :propagate : :failover if failover
|
|
224
|
+
|
|
225
|
+
attempts >= CONN_MAX_ATTEMPTS ? :exhausted : :backoff
|
|
226
|
+
end
|
|
227
|
+
|
|
228
|
+
# A replay re-issues the whole block, so one completed round trip voids
|
|
229
|
+
# every pre-apply proof: however provably the *failing* command missed the
|
|
230
|
+
# server, the ones ahead of it in the block did not. On a still-clean block
|
|
231
|
+
# a failover reply is proof enough by itself (the command was rejected
|
|
232
|
+
# outright); a bare ConnectionError needs one of the connect-phase classes.
|
|
233
|
+
def replayable?(err, idempotent, dirty, failover)
|
|
234
|
+
return true if idempotent
|
|
235
|
+
return false if dirty
|
|
236
|
+
|
|
237
|
+
failover || PRE_APPLY_ERRORS.any? { |klass| err.is_a?(klass) }
|
|
175
238
|
end
|
|
176
239
|
|
|
177
240
|
def notify_error(error, attempt:, retried:)
|
data/lib/wurk/scheduled.rb
CHANGED
|
@@ -6,6 +6,7 @@ require_relative 'lua'
|
|
|
6
6
|
require_relative 'lua/loader'
|
|
7
7
|
require_relative 'client'
|
|
8
8
|
require_relative 'process_set'
|
|
9
|
+
require_relative 'timer_loop'
|
|
9
10
|
|
|
10
11
|
module Wurk
|
|
11
12
|
# Promotes due jobs from the `retry` and `schedule` sorted sets back onto
|
|
@@ -57,13 +58,30 @@ module Wurk
|
|
|
57
58
|
loop do
|
|
58
59
|
break if @done
|
|
59
60
|
|
|
60
|
-
jobstr =
|
|
61
|
+
jobstr = pop_due(sset, now)
|
|
61
62
|
break unless jobstr
|
|
62
63
|
|
|
63
64
|
push_promoted(jobstr, sset)
|
|
64
65
|
end
|
|
65
66
|
end
|
|
66
67
|
|
|
68
|
+
# ZPOPBYSCORE is destructive and carries its result in the reply, so this
|
|
69
|
+
# block never claims apply-safety: a replay discards whatever the lost
|
|
70
|
+
# reply already removed. The pool therefore raises on a Read-/WriteTimeout,
|
|
71
|
+
# which leaves the outcome of *this* pop unknown — a due job may or may not
|
|
72
|
+
# have come off the ZSET. Report it and end this set's drain (the nil makes
|
|
73
|
+
# #drain_set break) rather than pop again blind; a job caught in that window
|
|
74
|
+
# falls into the same pop→push loss the default scheduler already documents
|
|
75
|
+
# on #push_promoted, and `reliable_scheduler!` (ReliableEnq) is the loss-free
|
|
76
|
+
# fix. Rescuing here rather than around #drain_set keeps the sibling set
|
|
77
|
+
# draining on this tick.
|
|
78
|
+
def pop_due(sset, now)
|
|
79
|
+
@config.redis { |conn| Wurk::Lua::Loader.eval_cached(conn, :zpopbyscore, keys: [sset], argv: [now]) }
|
|
80
|
+
rescue RedisClient::ConnectionError => e
|
|
81
|
+
handle_exception(e, { context: 'scheduler_pop', set: sset })
|
|
82
|
+
nil
|
|
83
|
+
end
|
|
84
|
+
|
|
67
85
|
# A raising `@client.push` (bad payload, transient Redis error) must not
|
|
68
86
|
# abort the drain and strand the remaining due jobs until the next poll —
|
|
69
87
|
# rescue per-job, report, continue. The already-popped job IS lost here
|
|
@@ -169,13 +187,23 @@ module Wurk
|
|
|
169
187
|
|
|
170
188
|
# Idempotent. Wakes the sleeping thread so it observes @done and exits.
|
|
171
189
|
# Also propagates the stop signal to @enq so any in-flight drain loop
|
|
172
|
-
# short-circuits instead of running to completion.
|
|
190
|
+
# short-circuits instead of running to completion. Terminal, not a pause:
|
|
191
|
+
# @enq's stop flag is one-way, so this poller never polls again.
|
|
192
|
+
#
|
|
193
|
+
# Joins before returning — the caller (Launcher#quiet, then #stop) clears
|
|
194
|
+
# the heartbeat right after, and a sweep still in flight would promote
|
|
195
|
+
# jobs on behalf of a process that no longer exists.
|
|
196
|
+
#
|
|
197
|
+
# Cleared only on a confirmed join (Thread#join returns nil on timeout):
|
|
198
|
+
# a wedged sweep must stay tracked so #start's ||= guard returns it
|
|
199
|
+
# rather than spawning a second scheduler thread alongside it.
|
|
173
200
|
def terminate
|
|
174
201
|
@mutex.synchronize do
|
|
175
202
|
@done = true
|
|
176
203
|
@enq.terminate
|
|
177
204
|
@sleeper.signal
|
|
178
205
|
end
|
|
206
|
+
@thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
|
|
179
207
|
end
|
|
180
208
|
|
|
181
209
|
# Called on every wake. Any raise inside the Enq is reported and the
|
data/lib/wurk/stats.rb
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require 'date'
|
|
4
|
+
require_relative 'pool_checkout'
|
|
4
5
|
|
|
5
6
|
module Wurk
|
|
6
7
|
# Read-only inspector for cluster state in Redis. The cheap counters are
|
|
@@ -38,7 +39,7 @@ module Wurk
|
|
|
38
39
|
# Sum of the `busy` HASH field across every live process identity.
|
|
39
40
|
# Pipelined but unbounded by process count.
|
|
40
41
|
def workers_size
|
|
41
|
-
Wurk.redis do |conn|
|
|
42
|
+
Wurk.redis(idempotent: true) do |conn|
|
|
42
43
|
identities = conn.call('SMEMBERS', Keys::PROCESSES)
|
|
43
44
|
next 0 if identities.empty?
|
|
44
45
|
|
|
@@ -55,7 +56,7 @@ module Wurk
|
|
|
55
56
|
# and dashboards reading this rely on that order. Match it exactly with
|
|
56
57
|
# `sort_by { |_, size| -size }`.
|
|
57
58
|
def queues
|
|
58
|
-
Wurk.redis do |conn|
|
|
59
|
+
Wurk.redis(idempotent: true) do |conn|
|
|
59
60
|
names = conn.call('SMEMBERS', Keys::QUEUES_SET)
|
|
60
61
|
next {} if names.empty?
|
|
61
62
|
|
|
@@ -71,7 +72,7 @@ module Wurk
|
|
|
71
72
|
# (`queue_summaries.sort_by { |qd| -qd.size }`) — this feeds the
|
|
72
73
|
# dashboard's queue table (api_controller#queues).
|
|
73
74
|
def queue_summaries
|
|
74
|
-
Wurk.redis do |conn|
|
|
75
|
+
Wurk.redis(idempotent: true) do |conn|
|
|
75
76
|
names = conn.call('SMEMBERS', Keys::QUEUES_SET)
|
|
76
77
|
next [] if names.empty?
|
|
77
78
|
|
|
@@ -89,13 +90,17 @@ module Wurk
|
|
|
89
90
|
# Latency (secs) of the `default` queue — the most-asked-about gauge.
|
|
90
91
|
def default_queue_latency
|
|
91
92
|
now_ms = ::Process.clock_gettime(::Process::CLOCK_REALTIME, :millisecond)
|
|
92
|
-
payload = Wurk.redis { |c| c.call('LRANGE', Keys.queue('default'), -1, -1) }.first
|
|
93
|
+
payload = Wurk.redis(idempotent: true) { |c| c.call('LRANGE', Keys.queue('default'), -1, -1) }.first
|
|
93
94
|
compute_latency(payload, now_ms)
|
|
94
95
|
end
|
|
95
96
|
|
|
96
97
|
# Resets the named global counters. With no args, clears `processed`,
|
|
97
98
|
# `failed`, and `expired`. SET … 0 (not DEL — keeps the key around so
|
|
98
99
|
# reads stay `Integer` not `nil`).
|
|
100
|
+
#
|
|
101
|
+
# The only write here, and the only block in this class that can't claim
|
|
102
|
+
# apply-safety: a replay after a lost reply would re-zero the counters,
|
|
103
|
+
# discarding whatever the fleet counted in between.
|
|
99
104
|
def reset(*stats)
|
|
100
105
|
all = %w[failed processed expired]
|
|
101
106
|
to_clear = stats.empty? ? all : all & stats.flatten.map(&:to_s)
|
|
@@ -123,7 +128,7 @@ module Wurk
|
|
|
123
128
|
private_constant :FAST_QUERIES, :FAST_KEYS
|
|
124
129
|
|
|
125
130
|
def fetch_stats_fast!
|
|
126
|
-
raw = Wurk.redis do |conn|
|
|
131
|
+
raw = Wurk.redis(idempotent: true) do |conn|
|
|
127
132
|
conn.pipelined { |pipe| FAST_QUERIES.each { |args| pipe.call(*args) } }
|
|
128
133
|
end
|
|
129
134
|
@stats = FAST_KEYS.zip(raw.map(&:to_i)).to_h
|
|
@@ -179,17 +184,17 @@ module Wurk
|
|
|
179
184
|
|
|
180
185
|
def date_stat_hash(stat)
|
|
181
186
|
keys = (0...@days_previous).map { |i| (@start_date - i).strftime('%Y-%m-%d') }
|
|
182
|
-
values = with_redis do |conn|
|
|
187
|
+
values = with_redis(idempotent: true) do |conn|
|
|
183
188
|
conn.pipelined { |pipe| keys.each { |d| pipe.call('GET', "stat:#{stat}:#{d}") } }
|
|
184
189
|
end
|
|
185
190
|
keys.zip(values.map(&:to_i)).to_h
|
|
186
191
|
end
|
|
187
192
|
|
|
188
|
-
def with_redis(&)
|
|
193
|
+
def with_redis(idempotent: false, &)
|
|
189
194
|
if @pool
|
|
190
|
-
|
|
195
|
+
PoolCheckout.with(@pool, idempotent, &)
|
|
191
196
|
else
|
|
192
|
-
Wurk.redis(&)
|
|
197
|
+
Wurk.redis(idempotent:, &)
|
|
193
198
|
end
|
|
194
199
|
end
|
|
195
200
|
end
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
require_relative '../component'
|
|
4
4
|
require_relative '../launcher'
|
|
5
|
+
require_relative '../client/buffered'
|
|
5
6
|
require_relative '../fetcher/reliable'
|
|
6
7
|
require_relative '../lua'
|
|
7
8
|
require_relative 'orphan_guard'
|
|
@@ -106,8 +107,19 @@ module Wurk
|
|
|
106
107
|
|
|
107
108
|
def reconnect_after_fork
|
|
108
109
|
@config.reset_redis_pools!
|
|
110
|
+
# The reliable_push outage buffer, its drainer thread and its mutexes
|
|
111
|
+
# are process-global and were copied wholesale from the parent. The
|
|
112
|
+
# Process._fork hook normally beats us to it (making this a no-op) —
|
|
113
|
+
# the explicit call keeps the swarm path deterministic and ordered
|
|
114
|
+
# after the pool reset, so a re-armed drainer can only ever see the
|
|
115
|
+
# child's own pool.
|
|
116
|
+
Wurk::Client::Buffered.reset_after_fork!
|
|
109
117
|
validate_redis!
|
|
110
118
|
reconnect_active_record
|
|
119
|
+
# The dogstatsd client is memoized at the class level (Statsd.client),
|
|
120
|
+
# so without a reset every child would share the parent's UDP socket
|
|
121
|
+
# and thread-locals instead of building its own after fork.
|
|
122
|
+
Wurk::Metrics::Statsd.reset!
|
|
111
123
|
end
|
|
112
124
|
|
|
113
125
|
# Prove the child's fresh Redis socket reaches a live server before it
|