wurk 1.3.1 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +1 -0
  3. data/lib/wurk/batch/callbacks.rb +82 -12
  4. data/lib/wurk/batch/death_handler.rb +7 -4
  5. data/lib/wurk/batch/server_middleware.rb +1 -1
  6. data/lib/wurk/batch.rb +121 -15
  7. data/lib/wurk/capsule.rb +5 -4
  8. data/lib/wurk/cli.rb +48 -14
  9. data/lib/wurk/client/buffered.rb +193 -43
  10. data/lib/wurk/client.rb +87 -14
  11. data/lib/wurk/compat.rb +1 -1
  12. data/lib/wurk/component.rb +2 -2
  13. data/lib/wurk/configuration.rb +3 -2
  14. data/lib/wurk/cron.rb +94 -37
  15. data/lib/wurk/deploy.rb +5 -3
  16. data/lib/wurk/embedded.rb +13 -0
  17. data/lib/wurk/fetcher/reaper.rb +113 -56
  18. data/lib/wurk/fetcher/reliable.rb +62 -9
  19. data/lib/wurk/heartbeat.rb +22 -10
  20. data/lib/wurk/history.rb +13 -1
  21. data/lib/wurk/launcher.rb +133 -66
  22. data/lib/wurk/leader.rb +29 -10
  23. data/lib/wurk/limiter/base.rb +8 -10
  24. data/lib/wurk/limiter/bucket.rb +1 -1
  25. data/lib/wurk/limiter/concurrent.rb +27 -22
  26. data/lib/wurk/limiter/window.rb +13 -11
  27. data/lib/wurk/limiter.rb +7 -4
  28. data/lib/wurk/lua.rb +97 -14
  29. data/lib/wurk/manager.rb +29 -13
  30. data/lib/wurk/metrics/history.rb +4 -3
  31. data/lib/wurk/metrics/queue_rollup.rb +13 -1
  32. data/lib/wurk/metrics/rollup.rb +13 -1
  33. data/lib/wurk/middleware/interrupt_handler.rb +7 -6
  34. data/lib/wurk/middleware/poison_pill.rb +70 -29
  35. data/lib/wurk/middleware.rb +2 -2
  36. data/lib/wurk/pool_checkout.rb +29 -0
  37. data/lib/wurk/process_set.rb +10 -5
  38. data/lib/wurk/processor.rb +6 -0
  39. data/lib/wurk/profiler.rb +3 -2
  40. data/lib/wurk/queue.rb +10 -7
  41. data/lib/wurk/rails_boot.rb +38 -7
  42. data/lib/wurk/redis_client_adapter.rb +48 -4
  43. data/lib/wurk/redis_pool.rb +70 -25
  44. data/lib/wurk/scheduled.rb +30 -2
  45. data/lib/wurk/stats.rb +14 -9
  46. data/lib/wurk/swarm/child_boot.rb +12 -0
  47. data/lib/wurk/swarm.rb +174 -33
  48. data/lib/wurk/timer_loop.rb +14 -0
  49. data/lib/wurk/version.rb +1 -1
  50. data/lib/wurk/web/enterprise.rb +58 -6
  51. data/lib/wurk/web/extension.rb +1 -1
  52. data/lib/wurk/web/search.rb +5 -3
  53. data/lib/wurk.rb +10 -2
  54. data/vendor/assets/dashboard/assets/{BatchDetail-YRymNsrB.js → BatchDetail-OmC5NPgw.js} +1 -1
  55. data/vendor/assets/dashboard/assets/{Batches-HY4hHdQU.js → Batches-CIpai7St.js} +1 -1
  56. data/vendor/assets/dashboard/assets/{Busy-FCEN1Bpx.js → Busy-A_kwSR6Q.js} +1 -1
  57. data/vendor/assets/dashboard/assets/{Cron-DO3J2zcp.js → Cron-BG7HTqlp.js} +1 -1
  58. data/vendor/assets/dashboard/assets/{Dashboard-B9rOrkzk.js → Dashboard-A_ToqHoo.js} +1 -1
  59. data/vendor/assets/dashboard/assets/{Dead-Bi4GGk9a.js → Dead-8J21jMyK.js} +1 -1
  60. data/vendor/assets/dashboard/assets/Extension-B4Q9FIQu.js +1 -0
  61. data/vendor/assets/dashboard/assets/{FilterBox-IJkHYpdm.js → FilterBox-Fh_Ae7UW.js} +1 -1
  62. data/vendor/assets/dashboard/assets/{JobDetailModal-DS1ypyoc.js → JobDetailModal-Ceng0PMB.js} +1 -1
  63. data/vendor/assets/dashboard/assets/{Limiters-Nz7UbNeJ.js → Limiters-CruDWvNZ.js} +1 -1
  64. data/vendor/assets/dashboard/assets/{Metrics-BBTDxcaE.js → Metrics-CIT7VCoN.js} +1 -1
  65. data/vendor/assets/dashboard/assets/Modal-CN3rdKA_.js +1 -0
  66. data/vendor/assets/dashboard/assets/{Queues-D9PH_THs.js → Queues-D86FYohJ.js} +1 -1
  67. data/vendor/assets/dashboard/assets/{Retries-CAKzDgYG.js → Retries-Bz1O1D-i.js} +1 -1
  68. data/vendor/assets/dashboard/assets/{Scheduled-DTYw1X8S.js → Scheduled-B6h2akTu.js} +1 -1
  69. data/vendor/assets/dashboard/assets/{Search-c4vFxDG_.js → Search-OOu22e5s.js} +1 -1
  70. data/vendor/assets/dashboard/assets/index-BdiUEDXX.css +1 -0
  71. data/vendor/assets/dashboard/assets/index-D_lSDwKw.js +141 -0
  72. data/vendor/assets/dashboard/assets/{useResetPageOnEmpty-B_FsMah6.js → useResetPageOnEmpty-dVPGEWzn.js} +1 -1
  73. data/vendor/assets/dashboard/index.html +2 -2
  74. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  75. metadata +21 -20
  76. data/vendor/assets/dashboard/assets/Extension-BSv8ddW_.js +0 -1
  77. data/vendor/assets/dashboard/assets/Modal-Crrsu64-.js +0 -1
  78. data/vendor/assets/dashboard/assets/index-BxjdeuOa.css +0 -1
  79. data/vendor/assets/dashboard/assets/index-DQu7WY9y.js +0 -141
@@ -32,7 +32,7 @@ module Wurk
32
32
  # absence (process never registered) and expiry (heartbeat lapsed,
33
33
  # info field gone) both return nil.
34
34
  def self.[](identity)
35
- exists, fields = Wurk.redis do |conn|
35
+ exists, fields = Wurk.redis(idempotent: true) do |conn|
36
36
  conn.pipelined do |pipe|
37
37
  pipe.call('SISMEMBER', Keys::PROCESSES, identity)
38
38
  pipe.call('HMGET', identity, *LOOKUP_FIELDS)
@@ -48,6 +48,11 @@ module Wurk
48
48
  # don't dogpile the prune. Returns the number of identities removed
49
49
  # (or 0 when the lock was held by someone else).
50
50
  #
51
+ # No apply-safety claim, unlike the reads around it: a replay re-decides
52
+ # which identities are dead, so it could prune one that registered in
53
+ # between and hasn't written `info` yet. A blip here just skips one prune —
54
+ # the next caller a minute later does it.
55
+ #
51
56
  # Spec: docs/target/sidekiq-free.md §31.17.
52
57
  def cleanup
53
58
  return 0 unless acquired_cleanup_lock?
@@ -84,7 +89,7 @@ module Wurk
84
89
  # SCARD over `processes`. Not pruned — may include identities whose
85
90
  # heartbeat has lapsed. Use `each` for the accurate count.
86
91
  def size
87
- Wurk.redis { |conn| conn.call('SCARD', Keys::PROCESSES) }
92
+ Wurk.redis(idempotent: true) { |conn| conn.call('SCARD', Keys::PROCESSES) }
88
93
  end
89
94
 
90
95
  # Sum of `concurrency` across live processes. Iterates `each` so dead
@@ -104,7 +109,7 @@ module Wurk
104
109
  # `||=` with empty-string fallback distinguishes "leader is unset" from
105
110
  # "memoization not yet computed".
106
111
  def leader
107
- @leader ||= Wurk.redis { |c| c.call('GET', 'dear-leader') } || ''
112
+ @leader ||= Wurk.redis(idempotent: true) { |c| c.call('GET', 'dear-leader') } || ''
108
113
  end
109
114
 
110
115
  class << self
@@ -134,7 +139,7 @@ module Wurk
134
139
  private
135
140
 
136
141
  def fetch_each_rows
137
- Wurk.redis do |conn|
142
+ Wurk.redis(idempotent: true) do |conn|
138
143
  procs = conn.call('SMEMBERS', Keys::PROCESSES).sort
139
144
  next [] if procs.empty?
140
145
 
@@ -223,7 +228,7 @@ module Wurk
223
228
  # Compares identity against the `dear-leader` STRING. Ent-only;
224
229
  # always false in OSS/free.
225
230
  def leader?
226
- Wurk.redis { |c| c.call('GET', 'dear-leader') == identity }
231
+ Wurk.redis(idempotent: true) { |c| c.call('GET', 'dear-leader') == identity }
227
232
  end
228
233
 
229
234
  private
@@ -164,6 +164,12 @@ module Wurk
164
164
  job_hash = parse_or_kill(jobstr, uow)
165
165
  return if job_hash.nil?
166
166
 
167
+ # The fetcher never parses, so hand it the jid we just read: the ACK
168
+ # retires this job's poison-pill recovery counter inside the round trip
169
+ # it already makes. A fetcher plugged in via `config[:fetch_class]` has
170
+ # no jid slot and simply ACKs — the counter then ages out on its 72h TTL.
171
+ uow.jid = job_hash['jid'] if uow.respond_to?(:jid=)
172
+
167
173
  ack = false
168
174
  begin
169
175
  Thread.handle_interrupt(Wurk::Shutdown => :never) do
data/lib/wurk/profiler.rb CHANGED
@@ -5,6 +5,7 @@ require 'zlib'
5
5
  require 'stringio'
6
6
  require 'tempfile'
7
7
  require_relative 'keys'
8
+ require_relative 'pool_checkout'
8
9
 
9
10
  module Wurk
10
11
  # Job profiling (Sidekiq 8.0+, OSS). When a job is pushed with a `profile`
@@ -113,8 +114,8 @@ module Wurk
113
114
  end
114
115
  end
115
116
 
116
- def with_pool(pool, &)
117
- pool ? pool.with(&) : Wurk.redis(&)
117
+ def with_pool(pool, idempotent: false, &)
118
+ pool ? PoolCheckout.with(pool, idempotent, &) : Wurk.redis(idempotent:, &)
118
119
  end
119
120
 
120
121
  def now
data/lib/wurk/queue.rb CHANGED
@@ -25,7 +25,7 @@ module Wurk
25
25
 
26
26
  # @return [Array<Queue>] one per known queue, sorted by name.
27
27
  def self.all
28
- names = Wurk.redis { |conn| conn.call('SMEMBERS', Keys::QUEUES_SET) }
28
+ names = Wurk.redis(idempotent: true) { |conn| conn.call('SMEMBERS', Keys::QUEUES_SET) }
29
29
  names.sort.map { |n| new(n) }
30
30
  end
31
31
 
@@ -35,12 +35,12 @@ module Wurk
35
35
  end
36
36
 
37
37
  def size
38
- Wurk.redis { |conn| conn.call('LLEN', @rname) }
38
+ Wurk.redis(idempotent: true) { |conn| conn.call('LLEN', @rname) }
39
39
  end
40
40
 
41
41
  # Seconds since the oldest job (tail of LIST) was enqueued. 0.0 when empty.
42
42
  def latency
43
- payload = Wurk.redis { |conn| conn.call('LRANGE', @rname, -1, -1).first }
43
+ payload = Wurk.redis(idempotent: true) { |conn| conn.call('LRANGE', @rname, -1, -1).first }
44
44
  return 0.0 if payload.nil?
45
45
 
46
46
  JobRecord.latency_from(Wurk.load_json(payload)['enqueued_at'])
@@ -51,19 +51,19 @@ module Wurk
51
51
  # True iff this queue's name is a member of the `paused` SET. Wurk
52
52
  # implements the Pro contract for free; fetchers consult the same set.
53
53
  def paused?
54
- Wurk.redis { |conn| conn.call('SISMEMBER', Keys::PAUSED_SET, @name) } == 1
54
+ Wurk.redis(idempotent: true) { |conn| conn.call('SISMEMBER', Keys::PAUSED_SET, @name) } == 1
55
55
  end
56
56
 
57
57
  # Pause new fetches against this queue. Idempotent — `SADD` returns
58
58
  # 0 when the name was already present. In-flight jobs are untouched.
59
59
  def pause! # rubocop:disable Naming/PredicateMethod
60
- Wurk.redis { |conn| conn.call('SADD', Keys::PAUSED_SET, @name) }
60
+ Wurk.redis(idempotent: true) { |conn| conn.call('SADD', Keys::PAUSED_SET, @name) }
61
61
  true
62
62
  end
63
63
 
64
64
  # Resume fetches. Idempotent.
65
65
  def unpause! # rubocop:disable Naming/PredicateMethod
66
- Wurk.redis { |conn| conn.call('SREM', Keys::PAUSED_SET, @name) }
66
+ Wurk.redis(idempotent: true) { |conn| conn.call('SREM', Keys::PAUSED_SET, @name) }
67
67
  true
68
68
  end
69
69
 
@@ -74,7 +74,7 @@ module Wurk
74
74
  loop do
75
75
  start = page * PAGE_SIZE
76
76
  stop = start + PAGE_SIZE - 1
77
- slice = Wurk.redis { |conn| conn.call('LRANGE', @rname, start, stop) }
77
+ slice = Wurk.redis(idempotent: true) { |conn| conn.call('LRANGE', @rname, start, stop) }
78
78
  slice.each { |value| yield JobRecord.new(value, @name) }
79
79
  break if slice.size < PAGE_SIZE
80
80
 
@@ -91,6 +91,9 @@ module Wurk
91
91
  # UNLINK the list + drop the queue from the `queues` set. Pipelined
92
92
  # so a partial failure leaves at most one of the two ops applied.
93
93
  # Method name is Sidekiq wire-compat — `clear?` would break the alias.
94
+ #
95
+ # Unlike pause!/unpause!, this one can't claim apply-safety: a replay after
96
+ # a lost reply would UNLINK whatever a producer enqueued in between.
94
97
  def clear # rubocop:disable Naming/PredicateMethod
95
98
  Wurk.redis do |conn|
96
99
  conn.pipelined do |pipe|
@@ -8,6 +8,11 @@ module Wurk
8
8
  # boot policy stays pure and unit-testable without the Railtie DSL.
9
9
  # See docs/idea/03-process-model.md for the exact ordering.
10
10
  module RailsBoot
11
+ # What `at_exit` waits for the supervise thread on top of the swarm's own
12
+ # drain budget: one supervise tick to notice the request, plus slack for
13
+ # the final reap.
14
+ DRAIN_JOIN_SLACK = 1
15
+
11
16
  module_function
12
17
 
13
18
  # Invoked from the `wurk.server_mode` initializer, before config/initializers
@@ -103,36 +108,62 @@ module Wurk
103
108
  end
104
109
 
105
110
  def boot_swarm
106
- swarm = Wurk::Swarm.new(topology: Wurk.configuration.topology,
107
- shutdown_timeout: Wurk.configuration[:timeout] || Swarm::DEFAULT_SHUTDOWN_TIMEOUT)
111
+ timeout = Wurk.configuration[:timeout] || Swarm::DEFAULT_SHUTDOWN_TIMEOUT
112
+ swarm = Wurk::Swarm.new(topology: Wurk.configuration.topology, shutdown_timeout: timeout)
113
+ supervisor = nil
114
+ # Registered BEFORE boot, which forks the children one slot at a time: a
115
+ # fork that raises partway through would otherwise leave the children it
116
+ # already spawned with nothing to drain them on host exit. Children
117
+ # inherit the hook — stop_swarm no-ops off the process that forked them.
118
+ at_exit { stop_swarm(swarm, supervisor, timeout + Swarm::SHUTDOWN_GRACE + DRAIN_JOIN_SLACK) }
108
119
  # Co-hosted in the web process (e.g. Puma single mode): the host owns the
109
120
  # process-wide TERM/INT traps. Installing the swarm's own would hijack
110
121
  # them — a deploy TERM would drain the swarm but never stop the HTTP
111
122
  # server. Let the host keep signal ownership and drain the swarm on its
112
123
  # graceful exit (same contract as boot_embedded).
113
124
  swarm.boot(install_signals: false)
114
- at_exit { swarm.shutdown }
115
125
  # supervise must still run somewhere or crashed children never respawn and
116
126
  # memory checks never fire. A background thread keeps the host's main
117
127
  # thread free to serve HTTP.
118
- Thread.new do
128
+ supervisor = Thread.new do
119
129
  swarm.supervise
120
130
  rescue StandardError => e
121
131
  logger.error { "wurk supervisor thread died: #{e.class}: #{e.message}" }
122
132
  end
123
133
  end
124
134
 
135
+ # at_exit fires on the host's main thread, but the supervise thread owns the
136
+ # child table and two threads inside `shutdown` race on it — so request the
137
+ # drain and wait for the supervisor to run it. Draining here is the fallback
138
+ # for when no live supervisor will: it never started (boot raised) or it
139
+ # died early; after one that drained it finds no children and no-ops. A
140
+ # supervisor still alive past the join is wedged mid-drain — leave it be
141
+ # rather than race it; its children self-terminate (OrphanGuard) once this
142
+ # process is gone.
143
+ def stop_swarm(swarm, supervisor, drain_wait)
144
+ return unless swarm.owner?
145
+
146
+ swarm.request_shutdown
147
+ supervisor&.join(drain_wait)
148
+ swarm.shutdown unless supervisor&.alive?
149
+ end
150
+
125
151
  # Sidekiq-embedded parity: a threads-only worker inside the web process, no
126
152
  # fork. Redis validation failure keeps the host serving HTTP (log + carry
127
153
  # on). at_exit drains on the graceful shutdown the web server runs on TERM.
128
- def boot_embedded
129
- instance = Wurk::Embedded.new(Wurk.configuration)
130
- instance.run
154
+ def boot_embedded(embedded = Wurk::Embedded)
155
+ instance = embedded.new(Wurk.configuration)
156
+ # Registered BEFORE run, for the reason boot_swarm registers before the
157
+ # fork: run brings the heartbeat, pollers, managers and health listener up
158
+ # one at a time, and a raise partway through would otherwise leave the ones
159
+ # already running with nothing to drain them on host exit.
131
160
  at_exit { instance.stop }
161
+ instance.run
132
162
  logger.info { 'wurk: running embedded in the web process (config.wurk.embed_in_web) — threads only, no fork' }
133
163
  instance
134
164
  rescue StandardError => e
135
165
  logger.error { "wurk: embedded boot failed: #{e.class}: #{e.message}" }
166
+ instance&.stop
136
167
  nil
137
168
  end
138
169
 
@@ -21,13 +21,18 @@ module Wurk
21
21
 
22
22
  DEPRECATED_COMMANDS = %i[rpoplpush zrangebyscore zrevrange zrevrangebyscore getset hmset setex setnx].to_set
23
23
 
24
+ # Every method here dispatches through `call` rather than `@client.call` so
25
+ # the round-trip odometer below sees method-style commands too — a host
26
+ # block doing `conn.sadd(...)` then `conn.lpush(...)` has to be as
27
+ # replay-protected as one written with `conn.call`. On the pipeline
28
+ # decorator `call` is the same buffered forward it always was.
24
29
  module CompatMethods
25
30
  def info
26
- @client.call('INFO') { |i| i.lines(chomp: true).map { |l| l.split(':', 2) }.select { |l| l.size == 2 }.to_h }
31
+ call('INFO') { |i| i.lines(chomp: true).map { |l| l.split(':', 2) }.select { |l| l.size == 2 }.to_h }
27
32
  end
28
33
 
29
34
  def evalsha(sha, keys, argv)
30
- @client.call('EVALSHA', sha, keys.size, *keys, *argv)
35
+ call('EVALSHA', sha, keys.size, *keys, *argv)
31
36
  end
32
37
 
33
38
  # The Redis commands Sidekiq itself uses — defined eagerly so the
@@ -41,7 +46,7 @@ module Wurk
41
46
 
42
47
  USED_COMMANDS.each do |name|
43
48
  define_method(name) do |*args, **kwargs|
44
- @client.call(name, *args, **kwargs)
49
+ call(name, *args, **kwargs)
45
50
  end
46
51
  end
47
52
 
@@ -52,7 +57,7 @@ module Wurk
52
57
  if DEPRECATED_COMMANDS.include?(args.first)
53
58
  warn("[sidekiq#5788] Redis has deprecated the `#{args.first}` command, called at #{caller(1..1)}")
54
59
  end
55
- @client.call(*args, &)
60
+ call(*args, &)
56
61
  end
57
62
  ruby2_keywords :method_missing if respond_to?(:ruby2_keywords, true)
58
63
 
@@ -64,9 +69,48 @@ module Wurk
64
69
  CompatClient = RedisClient::Decorator.create(CompatMethods)
65
70
 
66
71
  class CompatClient
72
+ # Dispatch methods that put commands on the wire and wait for the reply.
73
+ # SCAN and friends are left out on purpose: they are pure reads, so
74
+ # re-running one applies nothing and cannot make a replay unsafe.
75
+ DISPATCH_METHODS = %i[call call_v call_once call_once_v
76
+ blocking_call blocking_call_v pipelined multi].freeze
77
+
78
+ # Round trips this connection has completed, monotonic for its whole life.
79
+ #
80
+ # {Wurk::RedisPool} replays a failed block only while it can prove nothing
81
+ # in it applied, and a connect-phase error proves that for the command
82
+ # that raised — never for the ones before it. redis-client re-dials a
83
+ # dropped socket mid-block, so `CannotConnectError` surfaces on the second
84
+ # pipeline of a block whose first one already landed. The pool snapshots
85
+ # this counter around the block and refuses the replay once it has moved.
86
+ attr_reader :round_trips
87
+
88
+ def initialize(client)
89
+ super
90
+ @round_trips = 0
91
+ end
92
+
67
93
  def config
68
94
  @client.config
69
95
  end
96
+
97
+ # Counted *after* the call returns: a command that never reached a server
98
+ # leaves the odometer where it was, which is what keeps the pre-apply
99
+ # backoff alive for blocks that failed on their very first round trip.
100
+ DISPATCH_METHODS.each do |name|
101
+ class_eval(<<~RUBY, __FILE__, __LINE__ + 1)
102
+ # def call(...)
103
+ # result = super
104
+ # @round_trips += 1
105
+ # result
106
+ # end
107
+ def #{name}(...)
108
+ result = super
109
+ @round_trips += 1
110
+ result
111
+ end
112
+ RUBY
113
+ end
70
114
  end
71
115
  end
72
116
  end
@@ -10,17 +10,29 @@ module Wurk
10
10
  # across forks: the parent closes the pool before fork, each child opens a
11
11
  # fresh one (see docs/idea/03-process-model.md, steps 3 and 5).
12
12
  #
13
- # #with absorbs transient Redis failures (production incident #101):
13
+ # #with absorbs transient Redis failures (production incident #101), but only
14
+ # where replaying the caller's block cannot change what the server already
15
+ # did — the block is arbitrary Ruby, so a replay re-issues every command in it:
14
16
  # * READONLY / NOREPLICAS / UNBLOCKED — a failover happened; close and retry
15
17
  # once immediately so redis-client redials the new primary (spec §26).
16
- # * ConnectionError (incl. CannotConnect / Read- / WriteTimeout) — a blip;
17
- # close and retry with exponential backoff up to CONN_MAX_ATTEMPTS, then
18
- # raise. At-least-once tolerates the rare duplicate and JobRetry re-runs
19
- # the job anyway, so retrying at the pool layer is strictly better.
18
+ # * CannotConnect / Failover — raised while dialing, so the command that hit
19
+ # one never reached a server and cannot have applied; close and retry with
20
+ # exponential backoff up to CONN_MAX_ATTEMPTS, then raise.
21
+ # * Read-/WriteTimeout and bare ConnectionError — the command may already
22
+ # have applied server-side, so these raise. Replaying would double-push a
23
+ # job, double-count a stat, or drop a member ZPOPed by the lost reply.
24
+ # Blocks that are safe to re-run (pure reads, an LMOVE the reaper reclaims,
25
+ # owner-CAS scripts) opt back into the backoff with `with(idempotent: true)`.
20
26
  # * ConnectionPool::TimeoutError — checkout starved; retry once after a
21
27
  # short jittered pause, then raise (sizing is the fix, not queuing).
22
- # Every retry and final give-up is reported through the injected `on_error`
23
- # telemetry hook (Wurk::Configuration#on_redis_error).
28
+ # Those proofs are about the command that raised, not the block around it: a
29
+ # block is several round trips, and redis-client re-dials mid-block, so a
30
+ # CannotConnect can surface on the second pipeline of a block whose first one
31
+ # already landed. So the pool also watches the connection's round-trip
32
+ # odometer (RedisClientAdapter::CompatClient#round_trips) and refuses to
33
+ # replay a non-idempotent block that has already completed one.
34
+ # Every retry, refused replay, and final give-up is reported through the
35
+ # injected `on_error` telemetry hook (Wurk::Configuration#on_redis_error).
24
36
  class RedisPool
25
37
  DEFAULT_URL = ENV.fetch('REDIS_URL', 'redis://localhost:6379/0')
26
38
  DEFAULT_NAME = 'default'
@@ -32,7 +44,9 @@ module Wurk
32
44
  # wait above. read/write are deliberately wider than connect so a briefly-
33
45
  # slow-but-alive Redis (RDB fork pause, a large BLMOVE payload) doesn't
34
46
  # spuriously ReadTimeout — the production incident (#101) the single
35
- # dual-use timeout caused. reconnect_attempts re-dials a dropped socket once.
47
+ # dual-use timeout caused. reconnect_attempts re-dials a dropped socket once;
48
+ # note that redis-client's re-dial also re-sends the one in-flight command,
49
+ # so the apply-safety split below bounds block replay, not command replay.
36
50
  DEFAULT_CONNECT_TIMEOUT = 1.0
37
51
  DEFAULT_READ_TIMEOUT = 2.5
38
52
  DEFAULT_WRITE_TIMEOUT = 2.5
@@ -53,6 +67,17 @@ module Wurk
53
67
  # backoff below (otherwise a failover would sleep instead of redialing).
54
68
  RETRYABLE_MSG = /\A(READONLY|NOREPLICAS|UNBLOCKED)/
55
69
 
70
+ # ConnectionErrors that can only be raised while dialing, so the block
71
+ # provably never applied and replaying it is safe whatever it contains.
72
+ # CannotConnect covers every connect-phase failure (redis-client converts a
73
+ # stalled handshake into it); Failover is the Sentinel resolver rejecting a
74
+ # server whose role changed. Any other ConnectionError — Read-/WriteTimeout
75
+ # or a reset mid-command — leaves the outcome unknown.
76
+ PRE_APPLY_ERRORS = [RedisClient::CannotConnectError, RedisClient::FailoverError].freeze
77
+
78
+ # The two retry_plan verdicts that re-run the block; the rest raise.
79
+ REPLAY_PLANS = %i[failover backoff].freeze
80
+
56
81
  # ConnectionError backoff: CONN_MAX_ATTEMPTS total tries, sleeping
57
82
  # (BASE * 2**attempt) + rand*JITTER before each retry. The 1.0s + 2.0s pair
58
83
  # rides out a sub-4s blip; the jitter de-syncs a fleet reconnecting at once.
@@ -89,10 +114,14 @@ module Wurk
89
114
  # Checkout a connection and run the block. ConnectionPool::TimeoutError is
90
115
  # raised by @pool.with *before* the block runs, so it is caught out here
91
116
  # (the in-block #run rescue never sees it) — one retry, then raise.
92
- def with(&block)
117
+ #
118
+ # `idempotent: true` asserts the block can be re-run after a command may
119
+ # already have applied server-side, which buys back the full ConnectionError
120
+ # backoff. Only claim it for pure reads or writes whose repeat is a no-op.
121
+ def with(idempotent: false, &block)
93
122
  checkout_retried = false
94
123
  begin
95
- @pool.with { |conn| run(conn, &block) }
124
+ @pool.with { |conn| run(conn, idempotent, &block) }
96
125
  rescue ConnectionPool::TimeoutError => e
97
126
  if checkout_retried
98
127
  notify_error(e, attempt: 2, retried: false)
@@ -113,7 +142,7 @@ module Wurk
113
142
  # slot counts merged in — one call gives a heartbeat both Redis health and
114
143
  # local pool saturation. (Real Redis INFO has no `size`/`available` field.)
115
144
  def info
116
- with { |conn| parse_info(conn.call('INFO')) }
145
+ with(idempotent: true) { |conn| parse_info(conn.call('INFO')) }
117
146
  .merge('size' => @size, 'available' => available)
118
147
  end
119
148
 
@@ -159,17 +188,19 @@ module Wurk
159
188
  # Runs the block on the checked-out `conn`, retrying transient RedisClient
160
189
  # errors in place: the same slot is reused across retries (redis-client
161
190
  # redials a closed socket lazily), so a busy fetcher can't leak checkouts.
162
- def run(conn)
191
+ def run(conn, idempotent)
163
192
  attempts = 0
164
193
  begin
165
194
  attempts += 1
195
+ odometer = conn.round_trips
166
196
  yield conn
167
197
  rescue RedisClient::Error => e
168
- plan = retry_plan(e, attempts)
198
+ plan = retry_plan(e, attempts, idempotent, conn.round_trips != odometer)
169
199
  raise if plan == :propagate
170
200
 
171
- notify_error(e, attempt: attempts, retried: plan != :exhausted)
172
- raise if plan == :exhausted
201
+ replaying = REPLAY_PLANS.include?(plan)
202
+ notify_error(e, attempt: attempts, retried: replaying)
203
+ raise unless replaying
173
204
 
174
205
  safe_close(conn)
175
206
  sleep(backoff_delay(attempts)) if plan == :backoff
@@ -177,19 +208,33 @@ module Wurk
177
208
  end
178
209
  end
179
210
 
180
- # Pure classification of a RedisClient error against the attempt count:
211
+ # Pure classification of a RedisClient error against the attempt count, the
212
+ # caller's apply-safety claim, and whether this attempt already completed a
213
+ # round trip (`dirty`):
181
214
  # :failover → close + immediate retry (a primary swap)
182
215
  # :backoff → close + sleep + retry (a connection blip)
183
- # :exhausted → ConnectionError past the cap; report and give up
216
+ # :unsafe → something may have applied; report the blip, then raise
217
+ # :exhausted → replayable ConnectionError past the cap; report and give up
184
218
  # :propagate → not transient (or a spent failover); raise as-is
185
- def retry_plan(err, attempts)
186
- if RETRYABLE_MSG.match?(err.message.to_s)
187
- attempts > 1 ? :propagate : :failover
188
- elsif err.is_a?(RedisClient::ConnectionError)
189
- attempts >= CONN_MAX_ATTEMPTS ? :exhausted : :backoff
190
- else
191
- :propagate
192
- end
219
+ def retry_plan(err, attempts, idempotent, dirty)
220
+ failover = RETRYABLE_MSG.match?(err.message.to_s)
221
+ return :propagate unless failover || err.is_a?(RedisClient::ConnectionError)
222
+ return :unsafe unless replayable?(err, idempotent, dirty, failover)
223
+ return attempts > 1 ? :propagate : :failover if failover
224
+
225
+ attempts >= CONN_MAX_ATTEMPTS ? :exhausted : :backoff
226
+ end
227
+
228
+ # A replay re-issues the whole block, so one completed round trip voids
229
+ # every pre-apply proof: however provably the *failing* command missed the
230
+ # server, the ones ahead of it in the block did not. On a still-clean block
231
+ # a failover reply is proof enough by itself (the command was rejected
232
+ # outright); a bare ConnectionError needs one of the connect-phase classes.
233
+ def replayable?(err, idempotent, dirty, failover)
234
+ return true if idempotent
235
+ return false if dirty
236
+
237
+ failover || PRE_APPLY_ERRORS.any? { |klass| err.is_a?(klass) }
193
238
  end
194
239
 
195
240
  def notify_error(error, attempt:, retried:)
@@ -6,6 +6,7 @@ require_relative 'lua'
6
6
  require_relative 'lua/loader'
7
7
  require_relative 'client'
8
8
  require_relative 'process_set'
9
+ require_relative 'timer_loop'
9
10
 
10
11
  module Wurk
11
12
  # Promotes due jobs from the `retry` and `schedule` sorted sets back onto
@@ -57,13 +58,30 @@ module Wurk
57
58
  loop do
58
59
  break if @done
59
60
 
60
- jobstr = @config.redis { |conn| Wurk::Lua::Loader.eval_cached(conn, :zpopbyscore, keys: [sset], argv: [now]) }
61
+ jobstr = pop_due(sset, now)
61
62
  break unless jobstr
62
63
 
63
64
  push_promoted(jobstr, sset)
64
65
  end
65
66
  end
66
67
 
68
+ # ZPOPBYSCORE is destructive and carries its result in the reply, so this
69
+ # block never claims apply-safety: a replay discards whatever the lost
70
+ # reply already removed. The pool therefore raises on a Read-/WriteTimeout,
71
+ # which leaves the outcome of *this* pop unknown — a due job may or may not
72
+ # have come off the ZSET. Report it and end this set's drain (the nil makes
73
+ # #drain_set break) rather than pop again blind; a job caught in that window
74
+ # falls into the same pop→push loss the default scheduler already documents
75
+ # on #push_promoted, and `reliable_scheduler!` (ReliableEnq) is the loss-free
76
+ # fix. Rescuing here rather than around #drain_set keeps the sibling set
77
+ # draining on this tick.
78
+ def pop_due(sset, now)
79
+ @config.redis { |conn| Wurk::Lua::Loader.eval_cached(conn, :zpopbyscore, keys: [sset], argv: [now]) }
80
+ rescue RedisClient::ConnectionError => e
81
+ handle_exception(e, { context: 'scheduler_pop', set: sset })
82
+ nil
83
+ end
84
+
67
85
  # A raising `@client.push` (bad payload, transient Redis error) must not
68
86
  # abort the drain and strand the remaining due jobs until the next poll —
69
87
  # rescue per-job, report, continue. The already-popped job IS lost here
@@ -169,13 +187,23 @@ module Wurk
169
187
 
170
188
  # Idempotent. Wakes the sleeping thread so it observes @done and exits.
171
189
  # Also propagates the stop signal to @enq so any in-flight drain loop
172
- # short-circuits instead of running to completion.
190
+ # short-circuits instead of running to completion. Terminal, not a pause:
191
+ # @enq's stop flag is one-way, so this poller never polls again.
192
+ #
193
+ # Joins before returning — the caller (Launcher#quiet, then #stop) clears
194
+ # the heartbeat right after, and a sweep still in flight would promote
195
+ # jobs on behalf of a process that no longer exists.
196
+ #
197
+ # Cleared only on a confirmed join (Thread#join returns nil on timeout):
198
+ # a wedged sweep must stay tracked so #start's ||= guard returns it
199
+ # rather than spawning a second scheduler thread alongside it.
173
200
  def terminate
174
201
  @mutex.synchronize do
175
202
  @done = true
176
203
  @enq.terminate
177
204
  @sleeper.signal
178
205
  end
206
+ @thread = nil if @thread&.join(TimerLoop::JOIN_TIMEOUT)
179
207
  end
180
208
 
181
209
  # Called on every wake. Any raise inside the Enq is reported and the
data/lib/wurk/stats.rb CHANGED
@@ -1,6 +1,7 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require 'date'
4
+ require_relative 'pool_checkout'
4
5
 
5
6
  module Wurk
6
7
  # Read-only inspector for cluster state in Redis. The cheap counters are
@@ -38,7 +39,7 @@ module Wurk
38
39
  # Sum of the `busy` HASH field across every live process identity.
39
40
  # Pipelined but unbounded by process count.
40
41
  def workers_size
41
- Wurk.redis do |conn|
42
+ Wurk.redis(idempotent: true) do |conn|
42
43
  identities = conn.call('SMEMBERS', Keys::PROCESSES)
43
44
  next 0 if identities.empty?
44
45
 
@@ -55,7 +56,7 @@ module Wurk
55
56
  # and dashboards reading this rely on that order. Match it exactly with
56
57
  # `sort_by { |_, size| -size }`.
57
58
  def queues
58
- Wurk.redis do |conn|
59
+ Wurk.redis(idempotent: true) do |conn|
59
60
  names = conn.call('SMEMBERS', Keys::QUEUES_SET)
60
61
  next {} if names.empty?
61
62
 
@@ -71,7 +72,7 @@ module Wurk
71
72
  # (`queue_summaries.sort_by { |qd| -qd.size }`) — this feeds the
72
73
  # dashboard's queue table (api_controller#queues).
73
74
  def queue_summaries
74
- Wurk.redis do |conn|
75
+ Wurk.redis(idempotent: true) do |conn|
75
76
  names = conn.call('SMEMBERS', Keys::QUEUES_SET)
76
77
  next [] if names.empty?
77
78
 
@@ -89,13 +90,17 @@ module Wurk
89
90
  # Latency (secs) of the `default` queue — the most-asked-about gauge.
90
91
  def default_queue_latency
91
92
  now_ms = ::Process.clock_gettime(::Process::CLOCK_REALTIME, :millisecond)
92
- payload = Wurk.redis { |c| c.call('LRANGE', Keys.queue('default'), -1, -1) }.first
93
+ payload = Wurk.redis(idempotent: true) { |c| c.call('LRANGE', Keys.queue('default'), -1, -1) }.first
93
94
  compute_latency(payload, now_ms)
94
95
  end
95
96
 
96
97
  # Resets the named global counters. With no args, clears `processed`,
97
98
  # `failed`, and `expired`. SET … 0 (not DEL — keeps the key around so
98
99
  # reads stay `Integer` not `nil`).
100
+ #
101
+ # The only write here, and the only block in this class that can't claim
102
+ # apply-safety: a replay after a lost reply would re-zero the counters,
103
+ # discarding whatever the fleet counted in between.
99
104
  def reset(*stats)
100
105
  all = %w[failed processed expired]
101
106
  to_clear = stats.empty? ? all : all & stats.flatten.map(&:to_s)
@@ -123,7 +128,7 @@ module Wurk
123
128
  private_constant :FAST_QUERIES, :FAST_KEYS
124
129
 
125
130
  def fetch_stats_fast!
126
- raw = Wurk.redis do |conn|
131
+ raw = Wurk.redis(idempotent: true) do |conn|
127
132
  conn.pipelined { |pipe| FAST_QUERIES.each { |args| pipe.call(*args) } }
128
133
  end
129
134
  @stats = FAST_KEYS.zip(raw.map(&:to_i)).to_h
@@ -179,17 +184,17 @@ module Wurk
179
184
 
180
185
  def date_stat_hash(stat)
181
186
  keys = (0...@days_previous).map { |i| (@start_date - i).strftime('%Y-%m-%d') }
182
- values = with_redis do |conn|
187
+ values = with_redis(idempotent: true) do |conn|
183
188
  conn.pipelined { |pipe| keys.each { |d| pipe.call('GET', "stat:#{stat}:#{d}") } }
184
189
  end
185
190
  keys.zip(values.map(&:to_i)).to_h
186
191
  end
187
192
 
188
- def with_redis(&)
193
+ def with_redis(idempotent: false, &)
189
194
  if @pool
190
- @pool.with(&)
195
+ PoolCheckout.with(@pool, idempotent, &)
191
196
  else
192
- Wurk.redis(&)
197
+ Wurk.redis(idempotent:, &)
193
198
  end
194
199
  end
195
200
  end
@@ -2,6 +2,7 @@
2
2
 
3
3
  require_relative '../component'
4
4
  require_relative '../launcher'
5
+ require_relative '../client/buffered'
5
6
  require_relative '../fetcher/reliable'
6
7
  require_relative '../lua'
7
8
  require_relative 'orphan_guard'
@@ -106,8 +107,19 @@ module Wurk
106
107
 
107
108
  def reconnect_after_fork
108
109
  @config.reset_redis_pools!
110
+ # The reliable_push outage buffer, its drainer thread and its mutexes
111
+ # are process-global and were copied wholesale from the parent. The
112
+ # Process._fork hook normally beats us to it (making this a no-op) —
113
+ # the explicit call keeps the swarm path deterministic and ordered
114
+ # after the pool reset, so a re-armed drainer can only ever see the
115
+ # child's own pool.
116
+ Wurk::Client::Buffered.reset_after_fork!
109
117
  validate_redis!
110
118
  reconnect_active_record
119
+ # The dogstatsd client is memoized at the class level (Statsd.client),
120
+ # so without a reset every child would share the parent's UDP socket
121
+ # and thread-locals instead of building its own after fork.
122
+ Wurk::Metrics::Statsd.reset!
111
123
  end
112
124
 
113
125
  # Prove the child's fresh Redis socket reaches a live server before it