wurk 1.3.0 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +1 -0
  3. data/lib/wurk/batch/callbacks.rb +82 -12
  4. data/lib/wurk/batch/death_handler.rb +7 -4
  5. data/lib/wurk/batch/server_middleware.rb +1 -1
  6. data/lib/wurk/batch.rb +121 -15
  7. data/lib/wurk/capsule.rb +5 -4
  8. data/lib/wurk/cli.rb +48 -14
  9. data/lib/wurk/client/buffered.rb +193 -43
  10. data/lib/wurk/client.rb +87 -14
  11. data/lib/wurk/compat.rb +1 -1
  12. data/lib/wurk/component.rb +2 -2
  13. data/lib/wurk/configuration.rb +10 -2
  14. data/lib/wurk/cron.rb +94 -37
  15. data/lib/wurk/deploy.rb +5 -3
  16. data/lib/wurk/embedded.rb +13 -0
  17. data/lib/wurk/fetcher/reaper.rb +113 -56
  18. data/lib/wurk/fetcher/reliable.rb +62 -9
  19. data/lib/wurk/heartbeat.rb +22 -10
  20. data/lib/wurk/history.rb +13 -1
  21. data/lib/wurk/launcher.rb +133 -66
  22. data/lib/wurk/leader.rb +29 -10
  23. data/lib/wurk/limiter/base.rb +8 -10
  24. data/lib/wurk/limiter/bucket.rb +1 -1
  25. data/lib/wurk/limiter/concurrent.rb +27 -22
  26. data/lib/wurk/limiter/window.rb +13 -11
  27. data/lib/wurk/limiter.rb +7 -4
  28. data/lib/wurk/lua.rb +97 -14
  29. data/lib/wurk/manager.rb +29 -13
  30. data/lib/wurk/metrics/history.rb +4 -3
  31. data/lib/wurk/metrics/queue_rollup.rb +13 -1
  32. data/lib/wurk/metrics/rollup.rb +13 -1
  33. data/lib/wurk/middleware/interrupt_handler.rb +7 -6
  34. data/lib/wurk/middleware/poison_pill.rb +70 -29
  35. data/lib/wurk/middleware.rb +2 -2
  36. data/lib/wurk/pool_checkout.rb +29 -0
  37. data/lib/wurk/process_set.rb +10 -5
  38. data/lib/wurk/processor.rb +6 -0
  39. data/lib/wurk/profiler.rb +3 -2
  40. data/lib/wurk/queue.rb +10 -7
  41. data/lib/wurk/rails_boot.rb +38 -7
  42. data/lib/wurk/redis_client_adapter.rb +48 -4
  43. data/lib/wurk/redis_options.rb +142 -0
  44. data/lib/wurk/redis_pool.rb +102 -39
  45. data/lib/wurk/scheduled.rb +30 -2
  46. data/lib/wurk/stats.rb +14 -9
  47. data/lib/wurk/swarm/child_boot.rb +12 -0
  48. data/lib/wurk/swarm.rb +174 -33
  49. data/lib/wurk/timer_loop.rb +14 -0
  50. data/lib/wurk/version.rb +1 -1
  51. data/lib/wurk/web/enterprise.rb +58 -6
  52. data/lib/wurk/web/extension.rb +1 -1
  53. data/lib/wurk/web/search.rb +5 -3
  54. data/lib/wurk.rb +53 -2
  55. data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-OmC5NPgw.js} +1 -1
  56. data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-CIpai7St.js} +1 -1
  57. data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-A_kwSR6Q.js} +1 -1
  58. data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-BG7HTqlp.js} +1 -1
  59. data/vendor/assets/dashboard/assets/{Dashboard-B9rOrkzk.js → Dashboard-A_ToqHoo.js} +1 -1
  60. data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-8J21jMyK.js} +1 -1
  61. data/vendor/assets/dashboard/assets/Extension-B4Q9FIQu.js +1 -0
  62. data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-Fh_Ae7UW.js} +1 -1
  63. data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-Ceng0PMB.js} +1 -1
  64. data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-CruDWvNZ.js} +1 -1
  65. data/vendor/assets/dashboard/assets/{Metrics-BBTDxcaE.js → Metrics-CIT7VCoN.js} +1 -1
  66. data/vendor/assets/dashboard/assets/Modal-CN3rdKA_.js +1 -0
  67. data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-D86FYohJ.js} +1 -1
  68. data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-Bz1O1D-i.js} +1 -1
  69. data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-B6h2akTu.js} +1 -1
  70. data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-OOu22e5s.js} +1 -1
  71. data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
  72. data/vendor/assets/dashboard/assets/index-D_lSDwKw.js +141 -0
  73. data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-dVPGEWzn.js} +1 -1
  74. data/vendor/assets/dashboard/index.html +2 -2
  75. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  76. metadata +22 -20
  77. data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
  78. data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
  79. data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
  80. data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
@@ -3,23 +3,36 @@
3
3
  require 'redis-client'
4
4
  require 'connection_pool'
5
5
  require_relative 'redis_client_adapter'
6
+ require_relative 'redis_options'
6
7
 
7
8
  module Wurk
8
9
  # Per-process pool over redis-client + connection_pool. Never share a socket
9
10
  # across forks: the parent closes the pool before fork, each child opens a
10
11
  # fresh one (see docs/idea/03-process-model.md, steps 3 and 5).
11
12
  #
12
- # #with absorbs transient Redis failures (production incident #101):
13
+ # #with absorbs transient Redis failures (production incident #101), but only
14
+ # where replaying the caller's block cannot change what the server already
15
+ # did — the block is arbitrary Ruby, so a replay re-issues every command in it:
13
16
  # * READONLY / NOREPLICAS / UNBLOCKED — a failover happened; close and retry
14
17
  # once immediately so redis-client redials the new primary (spec §26).
15
- # * ConnectionError (incl. CannotConnect / Read- / WriteTimeout) — a blip;
16
- # close and retry with exponential backoff up to CONN_MAX_ATTEMPTS, then
17
- # raise. At-least-once tolerates the rare duplicate and JobRetry re-runs
18
- # the job anyway, so retrying at the pool layer is strictly better.
18
+ # * CannotConnect / Failover — raised while dialing, so the command that hit
19
+ # one never reached a server and cannot have applied; close and retry with
20
+ # exponential backoff up to CONN_MAX_ATTEMPTS, then raise.
21
+ # * Read-/WriteTimeout and bare ConnectionError — the command may already
22
+ # have applied server-side, so these raise. Replaying would double-push a
23
+ # job, double-count a stat, or drop a member ZPOPed by the lost reply.
24
+ # Blocks that are safe to re-run (pure reads, an LMOVE the reaper reclaims,
25
+ # owner-CAS scripts) opt back into the backoff with `with(idempotent: true)`.
19
26
  # * ConnectionPool::TimeoutError — checkout starved; retry once after a
20
27
  # short jittered pause, then raise (sizing is the fix, not queuing).
21
- # Every retry and final give-up is reported through the injected `on_error`
22
- # telemetry hook (Wurk::Configuration#on_redis_error).
28
+ # Those proofs are about the command that raised, not the block around it: a
29
+ # block is several round trips, and redis-client re-dials mid-block, so a
30
+ # CannotConnect can surface on the second pipeline of a block whose first one
31
+ # already landed. So the pool also watches the connection's round-trip
32
+ # odometer (RedisClientAdapter::CompatClient#round_trips) and refuses to
33
+ # replay a non-idempotent block that has already completed one.
34
+ # Every retry, refused replay, and final give-up is reported through the
35
+ # injected `on_error` telemetry hook (Wurk::Configuration#on_redis_error).
23
36
  class RedisPool
24
37
  DEFAULT_URL = ENV.fetch('REDIS_URL', 'redis://localhost:6379/0')
25
38
  DEFAULT_NAME = 'default'
@@ -31,18 +44,40 @@ module Wurk
31
44
  # wait above. read/write are deliberately wider than connect so a briefly-
32
45
  # slow-but-alive Redis (RDB fork pause, a large BLMOVE payload) doesn't
33
46
  # spuriously ReadTimeout — the production incident (#101) the single
34
- # dual-use timeout caused. reconnect_attempts re-dials a dropped socket once.
47
+ # dual-use timeout caused. reconnect_attempts re-dials a dropped socket once;
48
+ # note that redis-client's re-dial also re-sends the one in-flight command,
49
+ # so the apply-safety split below bounds block replay, not command replay.
35
50
  DEFAULT_CONNECT_TIMEOUT = 1.0
36
51
  DEFAULT_READ_TIMEOUT = 2.5
37
52
  DEFAULT_WRITE_TIMEOUT = 2.5
38
53
  DEFAULT_RECONNECT_ATTEMPTS = 1
39
54
 
55
+ # The floor every pool starts from; any key the host passed wins over it.
56
+ DEFAULT_CLIENT_CONFIG = {
57
+ url: DEFAULT_URL,
58
+ connect_timeout: DEFAULT_CONNECT_TIMEOUT,
59
+ read_timeout: DEFAULT_READ_TIMEOUT,
60
+ write_timeout: DEFAULT_WRITE_TIMEOUT,
61
+ reconnect_attempts: DEFAULT_RECONNECT_ATTEMPTS
62
+ }.freeze
63
+
40
64
  # Server-side messages where the connection is closed and the block retried
41
65
  # exactly once. READONLY is itself a RedisClient::ConnectionError subclass,
42
66
  # so this message match must be tested BEFORE the generic ConnectionError
43
67
  # backoff below (otherwise a failover would sleep instead of redialing).
44
68
  RETRYABLE_MSG = /\A(READONLY|NOREPLICAS|UNBLOCKED)/
45
69
 
70
+ # ConnectionErrors that can only be raised while dialing, so the block
71
+ # provably never applied and replaying it is safe whatever it contains.
72
+ # CannotConnect covers every connect-phase failure (redis-client converts a
73
+ # stalled handshake into it); Failover is the Sentinel resolver rejecting a
74
+ # server whose role changed. Any other ConnectionError — Read-/WriteTimeout
75
+ # or a reset mid-command — leaves the outcome unknown.
76
+ PRE_APPLY_ERRORS = [RedisClient::CannotConnectError, RedisClient::FailoverError].freeze
77
+
78
+ # The two retry_plan verdicts that re-run the block; the rest raise.
79
+ REPLAY_PLANS = %i[failover backoff].freeze
80
+
46
81
  # ConnectionError backoff: CONN_MAX_ATTEMPTS total tries, sleeping
47
82
  # (BASE * 2**attempt) + rand*JITTER before each retry. The 1.0s + 2.0s pair
48
83
  # rides out a sub-4s blip; the jitter de-syncs a fleet reconnecting at once.
@@ -60,9 +95,12 @@ module Wurk
60
95
 
61
96
  # Takes the standard Sidekiq `config.redis` hash: `pool_timeout` tunes the
62
97
  # ConnectionPool checkout; `connect_timeout`/`read_timeout`/`write_timeout`/
63
- # `reconnect_attempts` plus any other key (driver, ssl_params, …) forward
64
- # verbatim to RedisClient.config. `on_error` is an optional callable fired
65
- # per retry / final give-up with { error:, attempt:, retried:, pool: }.
98
+ # `reconnect_attempts` plus any other redis-client key (driver, ssl_params,
99
+ # sentinels, …) reach the client. Sidekiq-only spellings (`network_timeout`,
100
+ # `master_name`, `logger`, …) are translated or dropped by {RedisOptions};
101
+ # a key redis-client would reject raises there with the key named.
102
+ # `on_error` is an optional callable fired per retry / final give-up with
103
+ # { error:, attempt:, retried:, pool: }.
66
104
  def initialize(size:, name: DEFAULT_NAME, on_error: nil, **options)
67
105
  @size = size
68
106
  @name = name
@@ -76,10 +114,14 @@ module Wurk
76
114
  # Checkout a connection and run the block. ConnectionPool::TimeoutError is
77
115
  # raised by @pool.with *before* the block runs, so it is caught out here
78
116
  # (the in-block #run rescue never sees it) — one retry, then raise.
79
- def with(&block)
117
+ #
118
+ # `idempotent: true` asserts the block can be re-run after a command may
119
+ # already have applied server-side, which buys back the full ConnectionError
120
+ # backoff. Only claim it for pure reads or writes whose repeat is a no-op.
121
+ def with(idempotent: false, &block)
80
122
  checkout_retried = false
81
123
  begin
82
- @pool.with { |conn| run(conn, &block) }
124
+ @pool.with { |conn| run(conn, idempotent, &block) }
83
125
  rescue ConnectionPool::TimeoutError => e
84
126
  if checkout_retried
85
127
  notify_error(e, attempt: 2, retried: false)
@@ -100,7 +142,7 @@ module Wurk
100
142
  # slot counts merged in — one call gives a heartbeat both Redis health and
101
143
  # local pool saturation. (Real Redis INFO has no `size`/`available` field.)
102
144
  def info
103
- with { |conn| parse_info(conn.call('INFO')) }
145
+ with(idempotent: true) { |conn| parse_info(conn.call('INFO')) }
104
146
  .merge('size' => @size, 'available' => available)
105
147
  end
106
148
 
@@ -112,24 +154,29 @@ module Wurk
112
154
 
113
155
  private
114
156
 
115
- # Socket config forwarded to RedisClient.config. Host-supplied keys win over
116
- # the defaults; `pool_timeout` is dropped (it's a pool concern, not a socket
117
- # one) and unknown keys pass straight through.
157
+ # Socket config forwarded to redis-client. RedisOptions owns the translation
158
+ # of the Sidekiq-shaped hash (network_timeout, master_name, pool-only keys,
159
+ # …) so this class stays about pooling; host-supplied keys win over the
160
+ # defaults.
118
161
  def build_client_config(options)
119
- {
120
- url: DEFAULT_URL,
121
- connect_timeout: DEFAULT_CONNECT_TIMEOUT,
122
- read_timeout: DEFAULT_READ_TIMEOUT,
123
- write_timeout: DEFAULT_WRITE_TIMEOUT,
124
- reconnect_attempts: DEFAULT_RECONNECT_ATTEMPTS
125
- }.merge(options.except(:pool_timeout)).freeze
162
+ RedisOptions.normalize(options, defaults: DEFAULT_CLIENT_CONFIG).freeze
126
163
  end
127
164
 
128
165
  # Wrapped in the CompatClient decorator so `Sidekiq.redis { |c| c.smembers }`
129
166
  # method-style commands work like Sidekiq 7+ (#204). Wurk's own code paths
130
167
  # use #call, which the decorator forwards.
131
168
  def build_client
132
- RedisClientAdapter::CompatClient.new(RedisClient.config(**@client_config).new_client)
169
+ RedisClientAdapter::CompatClient.new(redis_client_config.new_client)
170
+ end
171
+
172
+ # A Sentinel set is a different constructor, not a different keyword:
173
+ # `RedisClient.config(sentinels: [...])` raises. Sidekiq routes the same way.
174
+ def redis_client_config
175
+ if RedisOptions.sentinel?(@client_config)
176
+ RedisClient.sentinel(**@client_config)
177
+ else
178
+ RedisClient.config(**@client_config)
179
+ end
133
180
  end
134
181
 
135
182
  def safe_close(conn)
@@ -141,17 +188,19 @@ module Wurk
141
188
  # Runs the block on the checked-out `conn`, retrying transient RedisClient
142
189
  # errors in place: the same slot is reused across retries (redis-client
143
190
  # redials a closed socket lazily), so a busy fetcher can't leak checkouts.
144
- def run(conn)
191
+ def run(conn, idempotent)
145
192
  attempts = 0
146
193
  begin
147
194
  attempts += 1
195
+ odometer = conn.round_trips
148
196
  yield conn
149
197
  rescue RedisClient::Error => e
150
- plan = retry_plan(e, attempts)
198
+ plan = retry_plan(e, attempts, idempotent, conn.round_trips != odometer)
151
199
  raise if plan == :propagate
152
200
 
153
- notify_error(e, attempt: attempts, retried: plan != :exhausted)
154
- raise if plan == :exhausted
201
+ replaying = REPLAY_PLANS.include?(plan)
202
+ notify_error(e, attempt: attempts, retried: replaying)
203
+ raise unless replaying
155
204
 
156
205
  safe_close(conn)
157
206
  sleep(backoff_delay(attempts)) if plan == :backoff
@@ -159,19 +208,33 @@ module Wurk
159
208
  end
160
209
  end
161
210
 
162
- # Pure classification of a RedisClient error against the attempt count:
211
+ # Pure classification of a RedisClient error against the attempt count, the
212
+ # caller's apply-safety claim, and whether this attempt already completed a
213
+ # round trip (`dirty`):
163
214
  # :failover → close + immediate retry (a primary swap)
164
215
  # :backoff → close + sleep + retry (a connection blip)
165
- # :exhausted → ConnectionError past the cap; report and give up
216
+ # :unsafe → something may have applied; report the blip, then raise
217
+ # :exhausted → replayable ConnectionError past the cap; report and give up
166
218
  # :propagate → not transient (or a spent failover); raise as-is
167
- def retry_plan(err, attempts)
168
- if RETRYABLE_MSG.match?(err.message.to_s)
169
- attempts > 1 ? :propagate : :failover
170
- elsif err.is_a?(RedisClient::ConnectionError)
171
- attempts >= CONN_MAX_ATTEMPTS ? :exhausted : :backoff
172
- else
173
- :propagate
174
- end
219
+ def retry_plan(err, attempts, idempotent, dirty)
220
+ failover = RETRYABLE_MSG.match?(err.message.to_s)
221
+ return :propagate unless failover || err.is_a?(RedisClient::ConnectionError)
222
+ return :unsafe unless replayable?(err, idempotent, dirty, failover)
223
+ return attempts > 1 ? :propagate : :failover if failover
224
+
225
+ attempts >= CONN_MAX_ATTEMPTS ? :exhausted : :backoff
226
+ end
227
+
228
+ # A replay re-issues the whole block, so one completed round trip voids
229
+ # every pre-apply proof: however provably the *failing* command missed the
230
+ # server, the ones ahead of it in the block did not. On a still-clean block
231
+ # a failover reply is proof enough by itself (the command was rejected
232
+ # outright); a bare ConnectionError needs one of the connect-phase classes.
233
+ def replayable?(err, idempotent, dirty, failover)
234
+ return true if idempotent
235
+ return false if dirty
236
+
237
+ failover || PRE_APPLY_ERRORS.any? { |klass| err.is_a?(klass) }
175
238
  end
176
239
 
177
240
  def notify_error(error, attempt:, retried:)
@@ -6,6 +6,7 @@ require_relative 'lua'
6
6
  require_relative 'lua/loader'
7
7
  require_relative 'client'
8
8
  require_relative 'process_set'
9
+ require_relative 'timer_loop'
9
10
 
10
11
  module Wurk
11
12
  # Promotes due jobs from the `retry` and `schedule` sorted sets back onto
@@ -57,13 +58,30 @@ module Wurk
57
58
  loop do
58
59
  break if @done
59
60
 
60
- jobstr = @config.redis { |conn| Wurk::Lua::Loader.eval_cached(conn, :zpopbyscore, keys: [sset], argv: [now]) }
61
+ jobstr = pop_due(sset, now)
61
62
  break unless jobstr
62
63
 
63
64
  push_promoted(jobstr, sset)
64
65
  end
65
66
  end
66
67
 
68
+ # ZPOPBYSCORE is destructive and carries its result in the reply, so this
69
+ # block never claims apply-safety: a replay discards whatever the lost
70
+ # reply already removed. The pool therefore raises on a Read-/WriteTimeout,
71
+ # which leaves the outcome of *this* pop unknown — a due job may or may not
72
+ # have come off the ZSET. Report it and end this set's drain (the nil makes
73
+ # #drain_set break) rather than pop again blind; a job caught in that window
74
+ # falls into the same pop→push loss the default scheduler already documents
75
+ # on #push_promoted, and `reliable_scheduler!` (ReliableEnq) is the loss-free
76
+ # fix. Rescuing here rather than around #drain_set keeps the sibling set
77
+ # draining on this tick.
78
+ def pop_due(sset, now)
79
+ @config.redis { |conn| Wurk::Lua::Loader.eval_cached(conn, :zpopbyscore, keys: [sset], argv: [now]) }
80
+ rescue RedisClient::ConnectionError => e
81
+ handle_exception(e, { context: 'scheduler_pop', set: sset })
82
+ nil
83
+ end
84
+
67
85
  # A raising `@client.push` (bad payload, transient Redis error) must not
68
86
  # abort the drain and strand the remaining due jobs until the next poll —
69
87
  # rescue per-job, report, continue. The already-popped job IS lost here
@@ -169,13 +187,23 @@ module Wurk
169
187
 
170
188
  # Idempotent. Wakes the sleeping thread so it observes @done and exits.
171
189
  # Also propagates the stop signal to @enq so any in-flight drain loop
172
- # short-circuits instead of running to completion.
190
+ # short-circuits instead of running to completion. Terminal, not a pause:
191
+ # @enq's stop flag is one-way, so this poller never polls again.
192
+ #
193
+ # Joins before returning — the caller (Launcher#quiet, then #stop) clears
194
+ # the heartbeat right after, and a sweep still in flight would promote
195
+ # jobs on behalf of a process that no longer exists.
196
+ #
197
+ # Cleared only on a confirmed join (Thread#join returns nil on timeout):
198
+ # a wedged sweep must stay tracked so #start's ||= guard returns it
199
+ # rather than spawning a second scheduler thread alongside it.
173
200
  def terminate
174
201
  @mutex.synchronize do
175
202
  @done = true
176
203
  @enq.terminate
177
204
  @sleeper.signal
178
205
  end
206
+ @thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
179
207
  end
180
208
 
181
209
  # Called on every wake. Any raise inside the Enq is reported and the
data/lib/wurk/stats.rb CHANGED
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require 'date'
4
+ require_relative 'pool_checkout'
4
5
 
5
6
  module Wurk
6
7
  # Read-only inspector for cluster state in Redis. The cheap counters are
@@ -38,7 +39,7 @@ module Wurk
38
39
  # Sum of the `busy` HASH field across every live process identity.
39
40
  # Pipelined but unbounded by process count.
40
41
  def workers_size
41
- Wurk.redis do |conn|
42
+ Wurk.redis(idempotent: true) do |conn|
42
43
  identities = conn.call('SMEMBERS', Keys::PROCESSES)
43
44
  next 0 if identities.empty?
44
45
 
@@ -55,7 +56,7 @@ module Wurk
55
56
  # and dashboards reading this rely on that order. Match it exactly with
56
57
  # `sort_by { |_, size| -size }`.
57
58
  def queues
58
- Wurk.redis do |conn|
59
+ Wurk.redis(idempotent: true) do |conn|
59
60
  names = conn.call('SMEMBERS', Keys::QUEUES_SET)
60
61
  next {} if names.empty?
61
62
 
@@ -71,7 +72,7 @@ module Wurk
71
72
  # (`queue_summaries.sort_by { |qd| -qd.size }`) — this feeds the
72
73
  # dashboard's queue table (api_controller#queues).
73
74
  def queue_summaries
74
- Wurk.redis do |conn|
75
+ Wurk.redis(idempotent: true) do |conn|
75
76
  names = conn.call('SMEMBERS', Keys::QUEUES_SET)
76
77
  next [] if names.empty?
77
78
 
@@ -89,13 +90,17 @@ module Wurk
89
90
  # Latency (secs) of the `default` queue — the most-asked-about gauge.
90
91
  def default_queue_latency
91
92
  now_ms = ::Process.clock_gettime(::Process::CLOCK_REALTIME, :millisecond)
92
- payload = Wurk.redis { |c| c.call('LRANGE', Keys.queue('default'), -1, -1) }.first
93
+ payload = Wurk.redis(idempotent: true) { |c| c.call('LRANGE', Keys.queue('default'), -1, -1) }.first
93
94
  compute_latency(payload, now_ms)
94
95
  end
95
96
 
96
97
  # Resets the named global counters. With no args, clears `processed`,
97
98
  # `failed`, and `expired`. SET … 0 (not DEL — keeps the key around so
98
99
  # reads stay `Integer` not `nil`).
100
+ #
101
+ # The only write here, and the only block in this class that can't claim
102
+ # apply-safety: a replay after a lost reply would re-zero the counters,
103
+ # discarding whatever the fleet counted in between.
99
104
  def reset(*stats)
100
105
  all = %w[failed processed expired]
101
106
  to_clear = stats.empty? ? all : all & stats.flatten.map(&:to_s)
@@ -123,7 +128,7 @@ module Wurk
123
128
  private_constant :FAST_QUERIES, :FAST_KEYS
124
129
 
125
130
  def fetch_stats_fast!
126
- raw = Wurk.redis do |conn|
131
+ raw = Wurk.redis(idempotent: true) do |conn|
127
132
  conn.pipelined { |pipe| FAST_QUERIES.each { |args| pipe.call(*args) } }
128
133
  end
129
134
  @stats = FAST_KEYS.zip(raw.map(&:to_i)).to_h
@@ -179,17 +184,17 @@ module Wurk
179
184
 
180
185
  def date_stat_hash(stat)
181
186
  keys = (0...@days_previous).map { |i| (@start_date - i).strftime('%Y-%m-%d') }
182
- values = with_redis do |conn|
187
+ values = with_redis(idempotent: true) do |conn|
183
188
  conn.pipelined { |pipe| keys.each { |d| pipe.call('GET', "stat:#{stat}:#{d}") } }
184
189
  end
185
190
  keys.zip(values.map(&:to_i)).to_h
186
191
  end
187
192
 
188
- def with_redis(&)
193
+ def with_redis(idempotent: false, &)
189
194
  if @pool
190
- @pool.with(&)
195
+ PoolCheckout.with(@pool, idempotent, &)
191
196
  else
192
- Wurk.redis(&)
197
+ Wurk.redis(idempotent:, &)
193
198
  end
194
199
  end
195
200
  end
@@ -2,6 +2,7 @@
2
2
 
3
3
  require_relative '../component'
4
4
  require_relative '../launcher'
5
+ require_relative '../client/buffered'
5
6
  require_relative '../fetcher/reliable'
6
7
  require_relative '../lua'
7
8
  require_relative 'orphan_guard'
@@ -106,8 +107,19 @@ module Wurk
106
107
 
107
108
  def reconnect_after_fork
108
109
  @config.reset_redis_pools!
110
+ # The reliable_push outage buffer, its drainer thread and its mutexes
111
+ # are process-global and were copied wholesale from the parent. The
112
+ # Process._fork hook normally beats us to it (making this a no-op) —
113
+ # the explicit call keeps the swarm path deterministic and ordered
114
+ # after the pool reset, so a re-armed drainer can only ever see the
115
+ # child's own pool.
116
+ Wurk::Client::Buffered.reset_after_fork!
109
117
  validate_redis!
110
118
  reconnect_active_record
119
+ # The dogstatsd client is memoized at the class level (Statsd.client),
120
+ # so without a reset every child would share the parent's UDP socket
121
+ # and thread-locals instead of building its own after fork.
122
+ Wurk::Metrics::Statsd.reset!
111
123
  end
112
124
 
113
125
  # Prove the child's fresh Redis socket reaches a live server before it