wurk 1.1.0 → 1.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +25 -0
  3. data/app/controllers/concerns/wurk/same_origin_guard.rb +40 -0
  4. data/app/controllers/concerns/wurk/sse_streaming.rb +48 -0
  5. data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +53 -0
  6. data/app/controllers/wurk/api/pagination.rb +60 -11
  7. data/app/controllers/wurk/api/serializers.rb +5 -1
  8. data/app/controllers/wurk/api_controller.rb +51 -70
  9. data/app/controllers/wurk/application_controller.rb +26 -0
  10. data/app/controllers/wurk/dashboard_controller.rb +23 -1
  11. data/app/controllers/wurk/extensions_controller.rb +3 -12
  12. data/app/controllers/wurk/profiles_controller.rb +6 -1
  13. data/lib/wurk/batch/death_handler.rb +13 -0
  14. data/lib/wurk/batch/server_middleware.rb +9 -5
  15. data/lib/wurk/batch.rb +3 -0
  16. data/lib/wurk/capsule.rb +34 -21
  17. data/lib/wurk/cli.rb +16 -3
  18. data/lib/wurk/client/buffered.rb +30 -10
  19. data/lib/wurk/client.rb +45 -1
  20. data/lib/wurk/component.rb +23 -9
  21. data/lib/wurk/configuration.rb +83 -18
  22. data/lib/wurk/cron.rb +13 -1
  23. data/lib/wurk/dead_set.rb +16 -1
  24. data/lib/wurk/engine.rb +17 -1
  25. data/lib/wurk/fetcher/reliable.rb +46 -13
  26. data/lib/wurk/fetcher.rb +5 -0
  27. data/lib/wurk/health.rb +74 -22
  28. data/lib/wurk/history.rb +4 -21
  29. data/lib/wurk/job_retry.rb +3 -5
  30. data/lib/wurk/launcher.rb +26 -3
  31. data/lib/wurk/limiter/server_middleware.rb +16 -1
  32. data/lib/wurk/limiter.rb +6 -5
  33. data/lib/wurk/lua/loader.rb +11 -5
  34. data/lib/wurk/lua.rb +100 -9
  35. data/lib/wurk/manager.rb +59 -19
  36. data/lib/wurk/metrics/history.rb +24 -38
  37. data/lib/wurk/metrics/queue_rollup.rb +4 -21
  38. data/lib/wurk/metrics/rollup.rb +4 -21
  39. data/lib/wurk/processor.rb +8 -8
  40. data/lib/wurk/profile_set.rb +21 -6
  41. data/lib/wurk/profiler.rb +7 -5
  42. data/lib/wurk/rails_boot.rb +176 -0
  43. data/lib/wurk/railtie.rb +19 -47
  44. data/lib/wurk/redis_connection.rb +6 -9
  45. data/lib/wurk/redis_pool.rb +148 -28
  46. data/lib/wurk/scheduled.rb +35 -12
  47. data/lib/wurk/swarm/backoff.rb +70 -0
  48. data/lib/wurk/swarm/child_boot.rb +92 -13
  49. data/lib/wurk/swarm/orphan_guard.rb +105 -0
  50. data/lib/wurk/swarm/restart.rb +196 -0
  51. data/lib/wurk/swarm.rb +194 -78
  52. data/lib/wurk/timer_loop.rb +49 -0
  53. data/lib/wurk/version.rb +1 -1
  54. data/lib/wurk/web/extension.rb +4 -1
  55. data/lib/wurk/web/pool_scope.rb +46 -0
  56. data/lib/wurk/web/rack_app.rb +2 -1
  57. data/lib/wurk/web/search.rb +77 -18
  58. data/lib/wurk/web.rb +1 -0
  59. data/lib/wurk/worker/setter.rb +6 -1
  60. data/lib/wurk.rb +6 -2
  61. data/vendor/assets/dashboard/assets/ArgsValue-DYfBiXrJ.js +1 -0
  62. data/vendor/assets/dashboard/assets/BatchDetail-DTZ2HzcD.js +1 -0
  63. data/vendor/assets/dashboard/assets/Batches-BrnXA332.js +1 -0
  64. data/vendor/assets/dashboard/assets/Busy-C57G8Xb3.js +1 -0
  65. data/vendor/assets/dashboard/assets/Cron-DlyH88oo.js +1 -0
  66. data/vendor/assets/dashboard/assets/Dashboard-B887pxWf.js +1 -0
  67. data/vendor/assets/dashboard/assets/Dead-BPA7gs-X.js +1 -0
  68. data/vendor/assets/dashboard/assets/Extension-Bunf6XuU.js +1 -0
  69. data/vendor/assets/dashboard/assets/FilterBox-FCDi4ZCU.js +1 -0
  70. data/vendor/assets/dashboard/assets/JobDetailModal-DuMdKUMm.js +2 -0
  71. data/vendor/assets/dashboard/assets/Limiters-Br0aCPMK.js +1 -0
  72. data/vendor/assets/dashboard/assets/Metrics-DxBmuywH.js +1 -0
  73. data/vendor/assets/dashboard/assets/Modal-t4FI_LaY.js +1 -0
  74. data/vendor/assets/dashboard/assets/PageHeader-CsDvJSOA.js +1 -0
  75. data/vendor/assets/dashboard/assets/Profiles-Bkhoqjlq.js +1 -0
  76. data/vendor/assets/dashboard/assets/Queues-BvhA-vfI.js +1 -0
  77. data/vendor/assets/dashboard/assets/Retries-JEpB-1Yl.js +1 -0
  78. data/vendor/assets/dashboard/assets/Scheduled-DN_FbSwP.js +1 -0
  79. data/vendor/assets/dashboard/assets/Search-DJuK0YCJ.js +1 -0
  80. data/vendor/assets/dashboard/assets/Skeleton-DOYDkzg1.js +1 -0
  81. data/vendor/assets/dashboard/assets/charts-CVK0zAnC.js +1 -0
  82. data/vendor/assets/dashboard/assets/index-BnPX9Ptn.css +1 -0
  83. data/vendor/assets/dashboard/assets/index-CZTcs-pM.js +141 -0
  84. data/vendor/assets/dashboard/assets/useResetPageOnEmpty-CoZU4b3a.js +1 -0
  85. data/vendor/assets/dashboard/assets/useSort-wQcnbdsa.js +1 -0
  86. data/vendor/assets/dashboard/index.html +3 -3
  87. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  88. metadata +55 -26
  89. data/vendor/assets/dashboard/assets/ArgsValue-BUqJa-eG.js +0 -1
  90. data/vendor/assets/dashboard/assets/BatchDetail-C5dkqAzp.js +0 -1
  91. data/vendor/assets/dashboard/assets/Batches-bgkXn8tc.js +0 -1
  92. data/vendor/assets/dashboard/assets/Busy-QSHBFXhy.js +0 -1
  93. data/vendor/assets/dashboard/assets/Cron-CUHppvTA.js +0 -1
  94. data/vendor/assets/dashboard/assets/Dashboard-CzPudckV.js +0 -1
  95. data/vendor/assets/dashboard/assets/Dead-BIq4Nz_i.js +0 -1
  96. data/vendor/assets/dashboard/assets/Extension-CW36i9R1.js +0 -1
  97. data/vendor/assets/dashboard/assets/JobDetailModal-BhNdWSp7.js +0 -2
  98. data/vendor/assets/dashboard/assets/Limiters-CiI_DfUK.js +0 -1
  99. data/vendor/assets/dashboard/assets/Metrics-BzZ8ugms.js +0 -1
  100. data/vendor/assets/dashboard/assets/Modal-DzVfgsSF.js +0 -1
  101. data/vendor/assets/dashboard/assets/PageHeader-Dp3qhX3e.js +0 -1
  102. data/vendor/assets/dashboard/assets/Profiles-BTtIlTdR.js +0 -1
  103. data/vendor/assets/dashboard/assets/Queues-BuXoxQ4W.js +0 -1
  104. data/vendor/assets/dashboard/assets/Retries-D4HAPaOQ.js +0 -1
  105. data/vendor/assets/dashboard/assets/Scheduled-6NCZYVJh.js +0 -1
  106. data/vendor/assets/dashboard/assets/Search-JaB_-52c.js +0 -1
  107. data/vendor/assets/dashboard/assets/charts-6uvCyY0x.js +0 -1
  108. data/vendor/assets/dashboard/assets/i18n-gIeA5VLo.js +0 -1
  109. data/vendor/assets/dashboard/assets/index-BDG9tvBA.css +0 -1
  110. data/vendor/assets/dashboard/assets/index-DWfWAGBc.js +0 -141
  111. data/vendor/assets/dashboard/assets/useJobSetActions-DTDaAoWr.js +0 -1
  112. data/vendor/assets/dashboard/assets/usePageParam-BcAvRko-.js +0 -1
  113. data/vendor/assets/dashboard/assets/useSort-D5Am4bGq.js +0 -1
@@ -3,6 +3,7 @@
3
3
  require 'socket'
4
4
  require_relative '../component'
5
5
  require_relative '../keys'
6
+ require_relative '../lua'
6
7
  require_relative '../fetcher'
7
8
 
8
9
  module Wurk
@@ -79,19 +80,20 @@ module Wurk
79
80
  end
80
81
 
81
82
  # Called on shutdown for jobs the Processor couldn't finish in time.
82
- # One pipelined RPUSH per public queue (head insert) so on next boot
83
- # they're picked again ahead of fresh enqueues.
83
+ # Atomically moves each still-private UoW back to its public queue via
84
+ # the RELIABLE_REQUEUE Lua (LREM-guarded RPUSH): the job leaves the
85
+ # per-process private list and reappears on the public queue in one hop,
86
+ # so it's visible immediately after a deploy instead of waiting for the
87
+ # next boot's reaper. The guard makes the move idempotent against the
88
+ # cross-thread `job`-read race in Manager#hard_shutdown — a Processor
89
+ # that ACKed in that window is a no-op (LREM misses, RPUSH skipped), so a
90
+ # finished job is never resurrected. Sidekiq Pro super_fetch §3 retains
91
+ # in-flight in the private list until the next boot; we prefer the
92
+ # immediate move so a rolling deploy recovers work without a restart.
84
93
  def bulk_requeue(in_progress)
85
94
  return if in_progress.nil? || in_progress.empty?
86
95
 
87
- grouped = in_progress.group_by(&:queue)
88
- config.redis do |conn|
89
- conn.pipelined do |pipe|
90
- grouped.each do |public_q, uows|
91
- pipe.call('RPUSH', public_q, *uows.map(&:job))
92
- end
93
- end
94
- end
96
+ config.redis { |conn| requeue_pipelined(conn, in_progress) }
95
97
  end
96
98
 
97
99
  # Prefixed queue keys (`queue:<name>`) in fetch order. Strict mode
@@ -107,12 +109,40 @@ module Wurk
107
109
  names.map { |q| "#{Keys::QUEUE_PREFIX}#{q}" }
108
110
  end
109
111
 
112
+ # Quiet hook (Manager#quiet). Flips the drain flag so retrieve_work
113
+ # short-circuits: once quieted, no processor can pull a fresh UoW, even
114
+ # one sitting in the between-jobs window (Processor#run only re-checks its
115
+ # own @done between iterations). Quiet is one-way — matches Sidekiq TSTP
116
+ # (spec §21.3), there is no un-terminate.
110
117
  def terminate
111
118
  @done = true
112
119
  end
113
120
 
114
121
  private
115
122
 
123
+ # One pipelined RELIABLE_REQUEUE EVALSHA per UoW. Mirrors
124
+ # Client#push_batched_pipelined: a pipelined EVALSHA surfaces NOSCRIPT
125
+ # only at finalize (never to eval_cached's inline rescue). Every command
126
+ # here is the same script, so a flushed cache fails all of them and
127
+ # applies none — recover by reloading once and replaying the whole
128
+ # pipeline via source-embedded EVAL.
129
+ def requeue_pipelined(conn, in_progress, eval_method: :eval_cached)
130
+ conn.pipelined do |pipe|
131
+ in_progress.each do |uow|
132
+ Wurk::Lua::Loader.public_send(
133
+ eval_method, pipe, :reliable_requeue,
134
+ keys: [self.class.private_queue_name(uow.queue), uow.queue],
135
+ argv: [uow.job]
136
+ )
137
+ end
138
+ end
139
+ rescue RedisClient::CommandError => e
140
+ raise unless e.message.to_s.start_with?('NOSCRIPT')
141
+
142
+ Wurk::Lua::Loader.script_load_all(conn)
143
+ requeue_pipelined(conn, in_progress, eval_method: :eval_with_source)
144
+ end
145
+
116
146
  # SMEMBERS of the `paused` SET. One round-trip per fetch pass; the
117
147
  # set is tiny in practice (one entry per paused queue) so the cost
118
148
  # is dominated by the BLMOVE that follows. Returns a Set for O(1)
@@ -130,9 +160,12 @@ module Wurk
130
160
  def blmove(public_q)
131
161
  priv = self.class.private_queue_name(public_q)
132
162
  timeout = poll_interval
133
- # Extend the socket read-timeout past BLMOVE's own timeout so the
134
- # default 1s pool timeout doesn't fire before BLMOVE returns.
135
- job = config.redis do |conn|
163
+ # Dedicated fetch pool, not the main one: a parked BLMOVE holds its slot
164
+ # for the whole block window, so routing it here keeps idle fetchers from
165
+ # starving the main pool's background loops (#101). Extend the socket
166
+ # read-timeout one second past BLMOVE's own server-side timeout so the
167
+ # connection's read timeout can't fire while BLMOVE is legitimately blocked.
168
+ job = config.fetch_redis do |conn|
136
169
  conn.blocking_call(timeout + 1, 'BLMOVE', public_q, priv, 'RIGHT', 'LEFT', timeout)
137
170
  end
138
171
  job ? UnitOfWork.new(queue: public_q, job: job, config: config) : nil
data/lib/wurk/fetcher.rb CHANGED
@@ -7,5 +7,10 @@ module Wurk
7
7
  class Fetcher
8
8
  def retrieve_work; end
9
9
  def bulk_requeue(in_progress); end
10
+
11
+ # Quiet hook: Manager#quiet calls this so retrieve_work can short-circuit
12
+ # and stop pulling new work the instant a process is quieted. No-op in the
13
+ # abstract base; Reliable flips its drain flag.
14
+ def terminate; end
10
15
  end
11
16
  end
data/lib/wurk/health.rb CHANGED
@@ -28,40 +28,41 @@ module Wurk
28
28
  # start/stop; safe to call from Launcher#run / Launcher#stop.
29
29
  class Server
30
30
  ACCEPT_TIMEOUT = 0.2
31
+ # How often a non-owner child re-attempts the shared port. Short enough
32
+ # that probes come back quickly after the owner dies, long enough not to
33
+ # spin. See #start_retry_loop.
34
+ RETRY_INTERVAL = 5
31
35
 
32
36
  attr_reader :port, :bind
33
37
 
34
- def initialize(launcher, port: DEFAULT_PORT, bind: DEFAULT_BIND, ready_window: DEFAULT_READY_WINDOW)
35
- @launcher = launcher
36
- @config = launcher.instance_variable_get(:@config)
37
- @port = port
38
- @bind = bind
39
- @ready_window = ready_window
40
- @server = nil
41
- @thread = nil
42
- @done = false
38
+ def initialize(launcher, port: DEFAULT_PORT, bind: DEFAULT_BIND,
39
+ ready_window: DEFAULT_READY_WINDOW, retry_interval: RETRY_INTERVAL)
40
+ @launcher = launcher
41
+ @config = launcher.instance_variable_get(:@config)
42
+ @port = port
43
+ @bind = bind
44
+ @ready_window = ready_window
45
+ @retry_interval = retry_interval
46
+ @server = nil
47
+ @thread = nil
48
+ @retry_thread = nil
49
+ @done = false
43
50
  end
44
51
 
45
52
  def start
46
- @server = ::TCPServer.new(@bind, @port)
47
- # Capture the OS-assigned port when caller passed 0 (test pattern,
48
- # also lets the kernel pick a free port at boot).
49
- @port = @server.addr[1]
53
+ # Idempotent (see class doc): a second start on a live instance would
54
+ # re-bind the same port, hit EADDRINUSE, and null out @server/@thread —
55
+ # leaking the original listener so stop could never close it.
56
+ return self if running? || retrying?
57
+
50
58
  @done = false
51
- @thread = ::Thread.new { run }
52
- @thread.name = 'wurk-health'
53
- self
54
- rescue ::Errno::EADDRINUSE => e
55
- # Swarm children all try to bind the same port — only the first wins.
56
- # Don't crash the worker; just log and skip.
57
- logger&.warn { "Wurk::Health: port #{@port} in use; health server NOT started (#{e.message})" }
58
- @server = nil
59
- @thread = nil
59
+ bind_and_serve || start_retry_loop
60
60
  self
61
61
  end
62
62
 
63
63
  def stop
64
64
  @done = true
65
+ stop_retry_loop
65
66
  srv = @server
66
67
  @server = nil
67
68
  srv&.close
@@ -73,8 +74,59 @@ module Wurk
73
74
  @thread&.alive? == true
74
75
  end
75
76
 
77
+ # True while a non-owner child is still polling to take the shared port
78
+ # over (see #start_retry_loop). Distinct from #running?, which reports the
79
+ # accept thread specifically.
80
+ def retrying?
81
+ @retry_thread&.alive? == true
82
+ end
83
+
76
84
  private
77
85
 
86
+ # One bind attempt. On success spins the accept thread and returns true.
87
+ # On EADDRINUSE — a sibling swarm child already owns the shared port —
88
+ # returns false so the caller schedules a retry.
89
+ def bind_and_serve
90
+ @server = ::TCPServer.new(@bind, @port)
91
+ # Capture the OS-assigned port when caller passed 0 (test pattern,
92
+ # also lets the kernel pick a free port at boot).
93
+ @port = @server.addr[1]
94
+ @thread = ::Thread.new { run }
95
+ @thread.name = 'wurk-health'
96
+ true
97
+ rescue ::Errno::EADDRINUSE
98
+ @server = nil
99
+ @thread = nil
100
+ false
101
+ end
102
+
103
+ # Non-owner children poll the shared port instead of giving up. A single
104
+ # bind-at-boot went dark to k8s the moment the owning child died —
105
+ # nothing rebound until the pod restarted. Now a survivor takes the port
106
+ # over within RETRY_INTERVAL of the owner's exit, so liveness/readiness
107
+ # ride out ordinary child churn (crash-respawn, rolling restart, recycle).
108
+ def start_retry_loop
109
+ logger&.warn do
110
+ "Wurk::Health: port #{@port} in use; polling every #{@retry_interval}s to take it over"
111
+ end
112
+ @retry_thread = ::Thread.new do
113
+ ::Thread.current.name = 'wurk-health-retry'
114
+ ::Thread.current.report_on_exception = false
115
+ sleep @retry_interval until @done || bind_and_serve
116
+ end
117
+ end
118
+
119
+ def stop_retry_loop
120
+ thread = @retry_thread
121
+ @retry_thread = nil
122
+ return unless thread
123
+
124
+ thread.wakeup if thread.alive?
125
+ thread.join(@retry_interval + 1)
126
+ rescue ThreadError
127
+ nil
128
+ end
129
+
78
130
  def run
79
131
  until @done
80
132
  ready = ::IO.select([@server], nil, nil, ACCEPT_TIMEOUT)
data/lib/wurk/history.rb CHANGED
@@ -4,6 +4,7 @@ require_relative 'component'
4
4
  require_relative 'keys'
5
5
  require_relative 'stats'
6
6
  require_relative 'metrics/statsd'
7
+ require_relative 'timer_loop'
7
8
 
8
9
  module Wurk
9
10
  # Sidekiq Enterprise §5 Historical Metrics snapshotter. A leader-gated
@@ -63,30 +64,18 @@ module Wurk
63
64
 
64
65
  def initialize(config)
65
66
  @config = config
66
- @interval = config.history_interval
67
67
  @collector = config.history_collector
68
68
  @stream_cap = config[:history_stream_cap] || STREAM_CAP
69
- @done = false
70
- @mutex = ::Mutex.new
71
- @sleeper = ::ConditionVariable.new
69
+ @timer = TimerLoop.new(config.history_interval)
72
70
  @thread = nil
73
71
  end
74
72
 
75
73
  def start
76
- @thread ||= safe_thread('history-snapshot') do # rubocop:disable Naming/MemoizedInstanceVariableName
77
- wait
78
- until @done
79
- tick
80
- wait
81
- end
82
- end
74
+ @thread ||= safe_thread('history-snapshot') { @timer.run { tick } } # rubocop:disable Naming/MemoizedInstanceVariableName
83
75
  end
84
76
 
85
77
  def terminate
86
- @mutex.synchronize do
87
- @done = true
88
- @sleeper.signal
89
- end
78
+ @timer.terminate
90
79
  end
91
80
 
92
81
  # Leader-gated: only the elected leader emits, so N workers don't each
@@ -164,11 +153,5 @@ module Wurk
164
153
 
165
154
  values.each { |field, value| client.gauge("sidekiq.#{field}", value) }
166
155
  end
167
-
168
- def wait
169
- @mutex.synchronize do
170
- @sleeper.wait(@mutex, @interval) unless @done
171
- end
172
- end
173
156
  end
174
157
  end
@@ -2,6 +2,7 @@
2
2
 
3
3
  require 'zlib'
4
4
  require_relative 'component'
5
+ require_relative 'dead_set'
5
6
 
6
7
  module Wurk
7
8
  # Owns the retry pipeline. When perform raises, JobRetry decides whether to
@@ -194,7 +195,7 @@ module Wurk
194
195
  if item.is_a?(::Float)
195
196
  ::Time.at(item)
196
197
  else
197
- ::Time.at(item / 1000, item % 1000)
198
+ ::Time.at(item / 1000, item % 1000, :millisecond)
198
199
  end
199
200
  end
200
201
 
@@ -262,10 +263,7 @@ module Wurk
262
263
 
263
264
  def send_to_morgue(msg)
264
265
  logger.info { "Adding dead #{msg['class']} job #{msg['jid']}" }
265
- payload = Wurk.dump_json(msg)
266
- now = ::Time.now.to_f
267
- redis { |conn| conn.call('ZADD', Keys::DEAD, now.to_s, payload) }
268
- DeadSet.new.trim
266
+ DeadSet.new.kill_raw(Wurk.dump_json(msg))
269
267
  end
270
268
 
271
269
  def run_death_handlers(job, exception)
data/lib/wurk/launcher.rb CHANGED
@@ -65,6 +65,7 @@ module Wurk
65
65
  @started_at = nil
66
66
  @heartbeat = nil
67
67
  @heartbeat_thread = nil
68
+ @boot_reclaim_thread = nil
68
69
  @health_server = build_health_server
69
70
  end
70
71
 
@@ -94,7 +95,9 @@ module Wurk
94
95
  @history&.start
95
96
  @managers.each(&:start)
96
97
  @reaper.start
97
- boot_reclaim
98
+ # Run on a background thread so /ready probe isn't delayed by a large
99
+ # orphan sweep (reaper.reclaim! is atomic, but can scan many entries).
100
+ @boot_reclaim_thread = safe_thread('boot-reclaim', &method(:boot_reclaim))
98
101
  @health_server&.start
99
102
  end
100
103
 
@@ -231,6 +234,9 @@ module Wurk
231
234
  @stopped = true
232
235
  thread = @heartbeat_thread
233
236
  return unless thread
237
+ # Embedded dashboard-TERM runs `stop` from the beat itself; a self-join
238
+ # raises ThreadError. @stopped is set, so the loop exits after this beat.
239
+ return if thread == Thread.current
234
240
 
235
241
  begin
236
242
  thread.wakeup
@@ -253,15 +259,32 @@ module Wurk
253
259
  logger.info('Heartbeat stopping...')
254
260
  end
255
261
 
262
+ # Dashboard-queued signals must behave exactly like OS signals, so a
263
+ # standalone process re-delivers to itself and lets the installed trap
264
+ # run — that wakes the main thread (CLI self-pipe / child dispatcher)
265
+ # so the process actually exits instead of stopping its managers and
266
+ # then parking forever. Embedded mode owns no traps (and self-TERM
267
+ # would kill the host app), so it calls quiet/stop directly — stop on
268
+ # its own thread because `stop` joins the heartbeat thread we're on.
256
269
  def dispatch_signal(sig)
257
270
  case sig
258
- when 'TSTP' then quiet
259
- when 'TERM' then stop
271
+ when 'TSTP', 'TERM'
272
+ if @embedded
273
+ sig == 'TSTP' ? quiet : Thread.new { stop }
274
+ else
275
+ redeliver(sig)
276
+ end
260
277
  else
261
278
  logger.warn { "Unknown signal in #{identity}-signals: #{sig.inspect}" }
262
279
  end
263
280
  end
264
281
 
282
+ # Separate method so tests can stub it — really sending TERM/TSTP would
283
+ # kill or suspend the test process.
284
+ def redeliver(sig)
285
+ ::Process.kill(sig, ::Process.pid)
286
+ end
287
+
265
288
  def build_poller
266
289
  Wurk::Scheduled::Poller.new(@config)
267
290
  end
@@ -1,14 +1,28 @@
1
1
  # frozen_string_literal: true
2
2
 
3
+ require_relative '../job_retry'
4
+
3
5
  module Wurk
4
6
  module Limiter
7
+ # Control signal raised after a job has been rescheduled: the limiter has
8
+ # already re-enqueued it (via `Client.push` at `Time.now + backoff`), so
9
+ # the run is *neither* a success nor a failure. Subclassing
10
+ # `JobRetry::Skip` routes it through the existing "middleware re-pushed the
11
+ # job — ack cleanly, book no retry" contract: the retrier re-raises it
12
+ # untouched, the outer Batch middleware skips both acks (its `rescue
13
+ # Handled` re-raises), and the Processor acks the UnitOfWork. Returning
14
+ # normally instead would let the outer Batch onion ack success for a job
15
+ # that never ran. Internal only — not part of the Sidekiq drop-in surface.
16
+ class Rescheduled < Wurk::JobRetry::Skip; end
17
+
5
18
  # Catches OverLimit (and any class registered in `Limiter.config.errors`),
6
19
  # bumps `job['overrated']`, and decides what to do next:
7
20
  #
8
21
  # * reschedule disabled (`reschedule: 0`) → re-raise so the normal
9
22
  # retry/dead pipeline handles it (spec §1.2/§1.4 behaviour).
10
23
  # * still under the cap → reschedule onto the same queue at
11
- # `Time.now + backoff` via `Client.push`.
24
+ # `Time.now + backoff` via `Client.push`, then raise `Rescheduled` so
25
+ # the outcome is neither success nor failure (batch onion skips acks).
12
26
  # * cap reached (`overrated >= reschedule`, default 20) → **poison
13
27
  # brake** (#16): a job that's still rate-limited after N reschedules
14
28
  # is saturating the limiter, so instead of dumping it into another
@@ -58,6 +72,7 @@ module Wurk
58
72
  backoff_proc = (limiter && limiter.options[:backoff]) || Wurk::Limiter.config.backoff
59
73
  delay = backoff_proc.call(limiter, job, exc).to_f
60
74
  Wurk::Client.new.push(job.merge('at' => ::Time.now.to_f + delay))
75
+ raise Rescheduled
61
76
  end
62
77
 
63
78
  # Poison brake: stamp a clear reason, drop the job in the dead set, and
data/lib/wurk/limiter.rb CHANGED
@@ -77,7 +77,7 @@ module Wurk
77
77
  @backoff = DEFAULT_BACKOFF
78
78
  @errors = [OverLimit]
79
79
  @redis = nil
80
- @redis_pool = nil
80
+ @pool = nil
81
81
  end
82
82
 
83
83
  # Accept either a Hash (the documented Sidekiq Ent shape — `{ size:,
@@ -86,7 +86,9 @@ module Wurk
86
86
  # responsibility (same contract as Wurk.redis_pool).
87
87
  def redis=(value)
88
88
  @redis = value
89
- @redis_pool = nil
89
+ # Reset the memo `pool` actually reads — resetting a different ivar
90
+ # left the old pool pinned after `config.redis = {...}`.
91
+ @pool = nil
90
92
  end
91
93
 
92
94
  def pool
@@ -97,9 +99,8 @@ module Wurk
97
99
  when Hash
98
100
  Wurk::RedisPool.new(
99
101
  size: @redis[:size] || 10,
100
- url: @redis[:url] || Wurk::RedisPool::DEFAULT_URL,
101
- timeout: @redis[:timeout] || Wurk::RedisPool::DEFAULT_TIMEOUT,
102
- name: 'limiter'
102
+ name: 'limiter',
103
+ **@redis.except(:size, :name)
103
104
  )
104
105
  else
105
106
  raise ArgumentError, "Limiter.config.redis must be Hash or RedisPool, got #{@redis.class}"
@@ -14,12 +14,18 @@ module Wurk
14
14
  NOSCRIPT_PREFIX = 'NOSCRIPT'
15
15
 
16
16
  class << self
17
- # Eagerly upload every registered script to the given connection.
18
- # Idempotent on the Redis side: `SCRIPT LOAD` of the same source
19
- # returns the same SHA regardless of how often it's called.
20
- # Manager calls this once per child after the post-fork reconnect.
17
+ # Eagerly upload every registered script to the given connection in a
18
+ # single pipelined round-trip (all SCRIPT LOADs, one RTT — not one per
19
+ # script). Idempotent on the Redis side: `SCRIPT LOAD` of the same source
20
+ # returns the same SHA no matter how often it runs. ChildBoot calls this
21
+ # once per child right after the post-fork reconnect so the first real
22
+ # EVALSHA hits a warm cache instead of paying a NOSCRIPT reload. Transient
23
+ # connection errors are the pool wrapper's job (Wurk::RedisPool#with);
24
+ # this only ships the loads.
21
25
  def script_load_all(redis)
22
- SCRIPTS.each_value { |src| redis.call('SCRIPT', 'LOAD', src) }
26
+ redis.pipelined do |pipe|
27
+ SCRIPTS.each_value { |src| pipe.call('SCRIPT', 'LOAD', src) }
28
+ end
23
29
  end
24
30
 
25
31
  # @param redis [RedisClient] a single connection (not a pool)
data/lib/wurk/lua.rb CHANGED
@@ -37,28 +37,91 @@ module Wurk
37
37
  # Pro reliable scheduler: atomically promote all due jobs in a sorted
38
38
  # set to their target queues. Pure-Ruby promotion does ZRANGE → ZREM →
39
39
  # LPUSH non-atomically and can lose jobs on a mid-step crash.
40
+ #
41
+ # Each promoted payload is restamped with a fresh `enqueued_at` (ARGV[3],
42
+ # epoch ms): that field marks arrival on an *immediate* queue, so a job
43
+ # leaving `schedule`/`retry` must get a new one at promotion rather than
44
+ # keep its stale scheduled-origin value (or none). This matches the default
45
+ # Ruby scheduler, whose push path restamps `enqueued_at` too
46
+ # (Client#push_plain / #push_batched) — so both schedulers emit
47
+ # wire-identical promoted payloads (spec §7.1).
48
+ #
49
+ # The restamp is a surgical string patch, NOT a cjson.decode -> cjson.encode
50
+ # round-trip: cjson maps every JSON number to a Lua double, so re-encoding
51
+ # would silently corrupt integer args past 2^53 (snowflake IDs, 64-bit
52
+ # counters) and reformat 15+ digit numbers into scientific notation --
53
+ # breaking wire-compat AND diverging from the loss-free Ruby scheduler,
54
+ # whose Ruby-side JSON keeps big integers exact. So the stored member is
55
+ # preserved byte-for-byte and only its top-level `enqueued_at` is rewritten:
56
+ # replaced in place when present (retry members keep theirs), inserted after
57
+ # the opening brace when absent (schedule members are stored stripped of it,
58
+ # via Client#push_scheduled). cjson.decode is still called -- but only to
59
+ # read the `queue` string and test for a top-level `enqueued_at`, never to
60
+ # re-serialize, so a lossy decode never reaches the payload. (Residual: a
61
+ # retry member whose user args embed a numeric key literally named
62
+ # `enqueued_at` ahead of the top-level one patches the arg instead --
63
+ # vanishingly rare, and still strictly safer than round-tripping every arg.)
40
64
  # KEYS = [sorted_set, queues_set]
41
- # ARGV = [now, queue_prefix]
65
+ # ARGV = [now, queue_prefix, now_ms]
42
66
  # Returns the number of jobs promoted.
43
67
  # Order matters: decode + push BEFORE zrem. Redis Lua has no rollback,
44
68
  # so a failed cjson.decode after a zrem would lose the job. Decode first;
45
- # push first; only then remove from the sorted set. Worst case is a
46
- # crash between lpush and zrem → at-least-once redelivery, never loss.
69
+ # push first; only then remove from the sorted set — and zrem the ORIGINAL
70
+ # member, not the restamped copy. Worst case is a crash between lpush and
71
+ # zrem → at-least-once redelivery, never loss.
72
+ # ARGV[4] caps members per call: Lua is atomic and single-threaded in
73
+ # Redis, so an unbatched promote of a post-outage backlog (100k+ due
74
+ # members) would block every client for the whole sweep. The Ruby caller
75
+ # loops until a short batch comes back.
47
76
  RELIABLE_SCHEDULE_PROMOTE = <<~LUA
48
- local jobs = redis.call("zrangebyscore", KEYS[1], "-inf", ARGV[1])
77
+ local jobs = redis.call("zrangebyscore", KEYS[1], "-inf", ARGV[1], "LIMIT", 0, tonumber(ARGV[4]))
49
78
  for i = 1, #jobs do
50
79
  local job = jobs[i]
51
- local q = cjson.decode(job)["queue"]
80
+ local decoded = cjson.decode(job)
81
+ local q = decoded["queue"]
82
+ local stamped
83
+ if decoded["enqueued_at"] == nil then
84
+ stamped = string.gsub(job, "^{", '{"enqueued_at":' .. ARGV[3] .. ",", 1)
85
+ else
86
+ stamped = string.gsub(job, '"enqueued_at":%-?%d[%d.eE+-]*', '"enqueued_at":' .. ARGV[3], 1)
87
+ end
52
88
  redis.call("sadd", KEYS[2], q)
53
- redis.call("lpush", ARGV[2] .. q, job)
89
+ redis.call("lpush", ARGV[2] .. q, stamped)
54
90
  redis.call("zrem", KEYS[1], job)
55
91
  end
56
92
  return #jobs
57
93
  LUA
58
94
 
95
+ # Reliable fetch (Pro super_fetch §3) shutdown requeue: atomically move
96
+ # one in-flight job from a per-process private list back to its public
97
+ # queue. The LREM guard is the whole point — RPUSH runs only when the job
98
+ # was still in the private list (LREM removed exactly 1). A job the
99
+ # Processor ACKed in the window between hard_shutdown's cross-thread `job`
100
+ # read and this move (LREM removes 0) is NOT re-pushed, so the job lands in
101
+ # exactly one place and can't double-execute. RPUSH (public tail) not LPUSH
102
+ # so the reclaimed job is fetched next — LMOVE pops the tail — ahead of
103
+ # fresh LPUSH'd enqueues.
104
+ # KEYS = [private_list, public_queue]
105
+ # ARGV = [job_json]
106
+ # Returns 1 when the job was moved, 0 when it was already acked.
107
+ RELIABLE_REQUEUE = <<~LUA
108
+ if redis.call("lrem", KEYS[1], 1, ARGV[1]) == 1 then
109
+ redis.call("rpush", KEYS[2], ARGV[1])
110
+ return 1
111
+ end
112
+ return 0
113
+ LUA
114
+
59
115
  # Pro Batch: register a job into a batch and push it to its queue
60
116
  # atomically. Keeps total/pending in sync with the jids set.
61
117
  #
118
+ # SADD into the live jids set is the registration guard: total/pending
119
+ # increment only when the jid is genuinely new (SADD == 1). A jid already
120
+ # live is a re-push — a retry or scheduled promotion re-enqueueing a job
121
+ # that never left the batch — so it re-LPUSHes the payload but must NOT
122
+ # recount, or pending inflates past the acks and `:success` never fires
123
+ # (spec §2.3/§2.5: total = distinct jobs added, pending = not-yet-succeeded).
124
+ #
62
125
  # A jid found in `b-<bid>-died` is a manual retry of a dead job (morgue
63
126
  # "retry" / "add to queue") — it rejoins the live set without recounting:
64
127
  # total and pending already include it, because a death never decrements
@@ -79,15 +142,41 @@ module Wurk
79
142
  redis.call("zrem", KEYS[6], ARGV[4])
80
143
  end
81
144
  else
82
- redis.call("hincrby", KEYS[1], "total", 1)
83
- redis.call("hincrby", KEYS[1], "pending", 1)
84
- redis.call("sadd", KEYS[2], ARGV[2])
145
+ if redis.call("sadd", KEYS[2], ARGV[2]) == 1 then
146
+ redis.call("hincrby", KEYS[1], "total", 1)
147
+ redis.call("hincrby", KEYS[1], "pending", 1)
148
+ end
85
149
  end
86
150
  redis.call("sadd", KEYS[4], ARGV[1])
87
151
  redis.call("lpush", KEYS[3], ARGV[3])
88
152
  return 1
89
153
  LUA
90
154
 
155
+ # Pro Batch: register a scheduled (`at`) job into a batch AND ZADD it onto
156
+ # the `schedule` set, atomically. This is the deferred sibling of BATCH_PUSH:
157
+ # same SADD-guarded total/pending counting, but the enqueue action is a ZADD
158
+ # (schedule for later) instead of an LPUSH (queue now). A `perform_in` inside
159
+ # `batch.jobs` must move `total`/`pending` at *creation* — otherwise the
160
+ # empty-marker check (batch.rb) sees no counter movement and fires
161
+ # `:complete`/`:success` while real jobs still sit in `schedule`.
162
+ #
163
+ # No died / dead-batches handling (unlike BATCH_PUSH): a job scheduled at
164
+ # creation time is always new to the batch. When the scheduler later promotes
165
+ # it, the re-push routes through BATCH_PUSH, whose guard finds the jid already
166
+ # live (SADD == 0) → pure LPUSH, no recount. So registration happens exactly
167
+ # once, here, at enqueue.
168
+ # KEYS = [schedule, b-<bid>, b-<bid>-jids]
169
+ # ARGV = [at_score, job_json, jid]
170
+ # Returns 1.
171
+ BATCH_SCHEDULE = <<~LUA
172
+ if redis.call("sadd", KEYS[3], ARGV[3]) == 1 then
173
+ redis.call("hincrby", KEYS[2], "total", 1)
174
+ redis.call("hincrby", KEYS[2], "pending", 1)
175
+ end
176
+ redis.call("zadd", KEYS[1], ARGV[1], ARGV[2])
177
+ return 1
178
+ LUA
179
+
91
180
  # Pro Batch: ACK a job that completed successfully. SREM from the live
92
181
  # jids set and decrement pending iff the jid was a member (idempotent
93
182
  # against double-success on a flaky retry). A success also clears any
@@ -259,7 +348,9 @@ module Wurk
259
348
  zpopbyscore: ZPOPBYSCORE,
260
349
  bulk_push: BULK_PUSH,
261
350
  reliable_schedule_promote: RELIABLE_SCHEDULE_PROMOTE,
351
+ reliable_requeue: RELIABLE_REQUEUE,
262
352
  batch_push: BATCH_PUSH,
353
+ batch_schedule: BATCH_SCHEDULE,
263
354
  batch_ack_success: BATCH_ACK_SUCCESS,
264
355
  batch_ack_failed: BATCH_ACK_FAILED,
265
356
  batch_ack_complete: BATCH_ACK_COMPLETE,