wurk 1.1.0 → 1.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +25 -0
  3. data/app/controllers/concerns/wurk/same_origin_guard.rb +40 -0
  4. data/app/controllers/concerns/wurk/sse_streaming.rb +48 -0
  5. data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +53 -0
  6. data/app/controllers/wurk/api/pagination.rb +60 -11
  7. data/app/controllers/wurk/api/serializers.rb +5 -1
  8. data/app/controllers/wurk/api_controller.rb +51 -70
  9. data/app/controllers/wurk/application_controller.rb +26 -0
  10. data/app/controllers/wurk/dashboard_controller.rb +23 -1
  11. data/app/controllers/wurk/extensions_controller.rb +3 -12
  12. data/app/controllers/wurk/profiles_controller.rb +6 -1
  13. data/lib/wurk/batch/death_handler.rb +13 -0
  14. data/lib/wurk/batch/server_middleware.rb +9 -5
  15. data/lib/wurk/batch.rb +3 -0
  16. data/lib/wurk/capsule.rb +34 -21
  17. data/lib/wurk/cli.rb +16 -3
  18. data/lib/wurk/client/buffered.rb +30 -10
  19. data/lib/wurk/client.rb +45 -1
  20. data/lib/wurk/component.rb +23 -9
  21. data/lib/wurk/configuration.rb +83 -18
  22. data/lib/wurk/cron.rb +13 -1
  23. data/lib/wurk/dead_set.rb +16 -1
  24. data/lib/wurk/engine.rb +17 -1
  25. data/lib/wurk/fetcher/reliable.rb +46 -13
  26. data/lib/wurk/fetcher.rb +5 -0
  27. data/lib/wurk/health.rb +74 -22
  28. data/lib/wurk/history.rb +4 -21
  29. data/lib/wurk/job_retry.rb +3 -5
  30. data/lib/wurk/launcher.rb +26 -3
  31. data/lib/wurk/limiter/server_middleware.rb +16 -1
  32. data/lib/wurk/limiter.rb +6 -5
  33. data/lib/wurk/lua/loader.rb +11 -5
  34. data/lib/wurk/lua.rb +100 -9
  35. data/lib/wurk/manager.rb +59 -19
  36. data/lib/wurk/metrics/history.rb +24 -38
  37. data/lib/wurk/metrics/queue_rollup.rb +4 -21
  38. data/lib/wurk/metrics/rollup.rb +4 -21
  39. data/lib/wurk/processor.rb +8 -8
  40. data/lib/wurk/profile_set.rb +21 -6
  41. data/lib/wurk/profiler.rb +7 -5
  42. data/lib/wurk/rails_boot.rb +176 -0
  43. data/lib/wurk/railtie.rb +19 -47
  44. data/lib/wurk/redis_connection.rb +6 -9
  45. data/lib/wurk/redis_pool.rb +148 -28
  46. data/lib/wurk/scheduled.rb +35 -12
  47. data/lib/wurk/swarm/backoff.rb +70 -0
  48. data/lib/wurk/swarm/child_boot.rb +92 -13
  49. data/lib/wurk/swarm/orphan_guard.rb +105 -0
  50. data/lib/wurk/swarm/restart.rb +196 -0
  51. data/lib/wurk/swarm.rb +194 -78
  52. data/lib/wurk/timer_loop.rb +49 -0
  53. data/lib/wurk/version.rb +1 -1
  54. data/lib/wurk/web/extension.rb +4 -1
  55. data/lib/wurk/web/pool_scope.rb +46 -0
  56. data/lib/wurk/web/rack_app.rb +2 -1
  57. data/lib/wurk/web/search.rb +77 -18
  58. data/lib/wurk/web.rb +1 -0
  59. data/lib/wurk/worker/setter.rb +6 -1
  60. data/lib/wurk.rb +6 -2
  61. data/vendor/assets/dashboard/assets/ArgsValue-DYfBiXrJ.js +1 -0
  62. data/vendor/assets/dashboard/assets/BatchDetail-DTZ2HzcD.js +1 -0
  63. data/vendor/assets/dashboard/assets/Batches-BrnXA332.js +1 -0
  64. data/vendor/assets/dashboard/assets/Busy-C57G8Xb3.js +1 -0
  65. data/vendor/assets/dashboard/assets/Cron-DlyH88oo.js +1 -0
  66. data/vendor/assets/dashboard/assets/Dashboard-B887pxWf.js +1 -0
  67. data/vendor/assets/dashboard/assets/Dead-BPA7gs-X.js +1 -0
  68. data/vendor/assets/dashboard/assets/Extension-Bunf6XuU.js +1 -0
  69. data/vendor/assets/dashboard/assets/FilterBox-FCDi4ZCU.js +1 -0
  70. data/vendor/assets/dashboard/assets/JobDetailModal-DuMdKUMm.js +2 -0
  71. data/vendor/assets/dashboard/assets/Limiters-Br0aCPMK.js +1 -0
  72. data/vendor/assets/dashboard/assets/Metrics-DxBmuywH.js +1 -0
  73. data/vendor/assets/dashboard/assets/Modal-t4FI_LaY.js +1 -0
  74. data/vendor/assets/dashboard/assets/PageHeader-CsDvJSOA.js +1 -0
  75. data/vendor/assets/dashboard/assets/Profiles-Bkhoqjlq.js +1 -0
  76. data/vendor/assets/dashboard/assets/Queues-BvhA-vfI.js +1 -0
  77. data/vendor/assets/dashboard/assets/Retries-JEpB-1Yl.js +1 -0
  78. data/vendor/assets/dashboard/assets/Scheduled-DN_FbSwP.js +1 -0
  79. data/vendor/assets/dashboard/assets/Search-DJuK0YCJ.js +1 -0
  80. data/vendor/assets/dashboard/assets/Skeleton-DOYDkzg1.js +1 -0
  81. data/vendor/assets/dashboard/assets/charts-CVK0zAnC.js +1 -0
  82. data/vendor/assets/dashboard/assets/index-BnPX9Ptn.css +1 -0
  83. data/vendor/assets/dashboard/assets/index-CZTcs-pM.js +141 -0
  84. data/vendor/assets/dashboard/assets/useResetPageOnEmpty-CoZU4b3a.js +1 -0
  85. data/vendor/assets/dashboard/assets/useSort-wQcnbdsa.js +1 -0
  86. data/vendor/assets/dashboard/index.html +3 -3
  87. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  88. metadata +55 -26
  89. data/vendor/assets/dashboard/assets/ArgsValue-BUqJa-eG.js +0 -1
  90. data/vendor/assets/dashboard/assets/BatchDetail-C5dkqAzp.js +0 -1
  91. data/vendor/assets/dashboard/assets/Batches-bgkXn8tc.js +0 -1
  92. data/vendor/assets/dashboard/assets/Busy-QSHBFXhy.js +0 -1
  93. data/vendor/assets/dashboard/assets/Cron-CUHppvTA.js +0 -1
  94. data/vendor/assets/dashboard/assets/Dashboard-CzPudckV.js +0 -1
  95. data/vendor/assets/dashboard/assets/Dead-BIq4Nz_i.js +0 -1
  96. data/vendor/assets/dashboard/assets/Extension-CW36i9R1.js +0 -1
  97. data/vendor/assets/dashboard/assets/JobDetailModal-BhNdWSp7.js +0 -2
  98. data/vendor/assets/dashboard/assets/Limiters-CiI_DfUK.js +0 -1
  99. data/vendor/assets/dashboard/assets/Metrics-BzZ8ugms.js +0 -1
  100. data/vendor/assets/dashboard/assets/Modal-DzVfgsSF.js +0 -1
  101. data/vendor/assets/dashboard/assets/PageHeader-Dp3qhX3e.js +0 -1
  102. data/vendor/assets/dashboard/assets/Profiles-BTtIlTdR.js +0 -1
  103. data/vendor/assets/dashboard/assets/Queues-BuXoxQ4W.js +0 -1
  104. data/vendor/assets/dashboard/assets/Retries-D4HAPaOQ.js +0 -1
  105. data/vendor/assets/dashboard/assets/Scheduled-6NCZYVJh.js +0 -1
  106. data/vendor/assets/dashboard/assets/Search-JaB_-52c.js +0 -1
  107. data/vendor/assets/dashboard/assets/charts-6uvCyY0x.js +0 -1
  108. data/vendor/assets/dashboard/assets/i18n-gIeA5VLo.js +0 -1
  109. data/vendor/assets/dashboard/assets/index-BDG9tvBA.css +0 -1
  110. data/vendor/assets/dashboard/assets/index-DWfWAGBc.js +0 -141
  111. data/vendor/assets/dashboard/assets/useJobSetActions-DTDaAoWr.js +0 -1
  112. data/vendor/assets/dashboard/assets/usePageParam-BcAvRko-.js +0 -1
  113. data/vendor/assets/dashboard/assets/useSort-D5Am4bGq.js +0 -1
@@ -17,8 +17,9 @@ module Wurk
17
17
  # failure → BATCH_ACK_FAILED → the jid joins `b-<bid>-failed` and
18
18
  # `failures` reflects the count of currently-failing jobs. A later
19
19
  # successful retry clears it; a terminal death moves it to `b-<bid>-died`.
20
- # Clean handled exits (JobRetry::Skip from expiry/interrupt, cooperative
21
- # IterableJob interruption) are re-raised without counting as failures.
20
+ # Clean handled exits (JobRetry::Skip — from the interrupt handler or a
21
+ # Limiter::Rescheduled re-enqueue — and cooperative IterableJob
22
+ # interruption) are re-raised as neither success nor failure.
22
23
  #
23
24
  # Invalidated batches short-circuit: the job is skipped without
24
25
  # raising — counts as a "success" for batch purposes per spec §12.
@@ -43,9 +44,12 @@ module Wurk
43
44
 
44
45
  private
45
46
 
46
- # A handled/skip exit or a cooperative interruption is not a failure —
47
- # re-raise it untouched. Any other exception means the job failed and
48
- # will retry (or eventually die): record it before re-raising.
47
+ # A handled/skip exit — including a Limiter::Rescheduled, where the
48
+ # limiter already re-enqueued the job (Rescheduled < JobRetry::Skip <
49
+ # Handled) — or a cooperative interruption is *neither* success nor
50
+ # failure: re-raise it untouched, acking neither. Any other exception
51
+ # means the job failed and will retry (or eventually die): record it
52
+ # before re-raising.
49
53
  def run_and_ack(bid, jid)
50
54
  yield
51
55
  ack_success(bid, jid)
data/lib/wurk/batch.rb CHANGED
@@ -185,6 +185,9 @@ module Wurk
185
185
  collect_jobs(&block)
186
186
  # By the time we check, the buffer (if any) has flushed, so `total`
187
187
  # reflects everything the block pushed — a flat count is reliable.
188
+ # Scheduled (`perform_in`) jobs count here too: BATCH_SCHEDULE moves
189
+ # `total` at creation, so a scheduled-only block does not misfire the
190
+ # marker as if it were empty.
188
191
  enqueue_empty_marker if job_count == pre_count
189
192
  @mutable = false
190
193
  self
data/lib/wurk/capsule.rb CHANGED
@@ -25,7 +25,7 @@ module Wurk
25
25
  @weights = { 'default' => 0 }
26
26
  @fetcher = nil
27
27
  @redis_pool = nil
28
- @local_redis_pool = nil
28
+ @fetch_redis_pool = nil
29
29
  @client_chain = nil
30
30
  @server_chain = nil
31
31
  end
@@ -65,7 +65,7 @@ module Wurk
65
65
  def prepare!
66
66
  @fetcher ||= build_fetcher
67
67
  redis_pool
68
- local_redis_pool
68
+ fetch_redis_pool
69
69
  client_middleware
70
70
  server_middleware
71
71
  self
@@ -86,20 +86,27 @@ module Wurk
86
86
  chain
87
87
  end
88
88
 
89
- # Pool size = concurrency + POOL_OVERHEAD. Every processor thread can
90
- # be parked in a blocking BLMOVE (reliable fetch), holding its slot
91
- # for ~3s. POOL_OVERHEAD reserves connections for the Launcher's
92
- # heartbeat thread + the Scheduled::Poller so they can never be
93
- # starved by busy fetchers. Without this, concurrency=1 deadlocks
94
- # immediately (worker holds the only slot, heartbeat times out).
95
- POOL_OVERHEAD = 2
89
+ # Headroom above `concurrency` for the main pool; the whole pool is then
90
+ # floored at MIN_POOL_SIZE. Blocking BLMOVE fetch has its own pool
91
+ # (#fetch_redis_pool), so the main pool serves only the background loops —
92
+ # heartbeat, scheduled poller, leader election, cron, the two metrics
93
+ # rollups, reaper, history, health probe — plus the host's own job-code
94
+ # checkouts. `concurrency + 5` (floor 10) gives each an unstarvable slot;
95
+ # the old `concurrency + 2` starved them once fetch also drew from here —
96
+ # the #101 `0/N` pool-exhaustion incident. Override via `config.redis[:size]`.
97
+ POOL_HEADROOM = 5
98
+ MIN_POOL_SIZE = 10
96
99
 
97
100
  def redis_pool
98
- @redis_pool ||= build_pool(size: @concurrency + POOL_OVERHEAD, name: "#{@name}-main")
101
+ @redis_pool ||= build_pool(size: main_pool_size, name: "#{@name}-main")
99
102
  end
100
103
 
101
- def local_redis_pool
102
- @local_redis_pool ||= build_pool(size: @concurrency, name: "#{@name}-local")
104
+ # Dedicated pool for the reliable fetcher's blocking BLMOVE: one slot per
105
+ # processor thread (`concurrency`), since at most that many threads park in
106
+ # fetch at once. Keeping fetch off the main pool is what lets an idle worker
107
+ # hold zero main-pool connections again.
108
+ def fetch_redis_pool
109
+ @fetch_redis_pool ||= build_pool(size: @concurrency, name: "#{@name}-fetch")
103
110
  end
104
111
 
105
112
  # Disconnect and drop cached pools. Called by Wurk::Swarm just before
@@ -109,14 +116,20 @@ module Wurk
109
116
  def reset_redis_pools!
110
117
  @redis_pool&.disconnect!
111
118
  @redis_pool = nil
112
- @local_redis_pool&.disconnect!
113
- @local_redis_pool = nil
119
+ @fetch_redis_pool&.disconnect!
120
+ @fetch_redis_pool = nil
114
121
  end
115
122
 
116
123
  def redis(&)
117
124
  redis_pool.with(&)
118
125
  end
119
126
 
127
+ # Checkout from the dedicated fetch pool. Only the reliable fetcher's
128
+ # blocking BLMOVE uses this, so a parked fetch never holds a main-pool slot.
129
+ def fetch_redis(&)
130
+ fetch_redis_pool.with(&)
131
+ end
132
+
120
133
  def lookup(name)
121
134
  @config.lookup(name)
122
135
  end
@@ -196,14 +209,14 @@ module Wurk
196
209
  parsed.flat_map { |q, w| [q] * w }
197
210
  end
198
211
 
212
+ # `config.redis = { size: N }` pins the main pool at N; otherwise
213
+ # `concurrency + POOL_HEADROOM`, floored at MIN_POOL_SIZE.
214
+ def main_pool_size
215
+ @config.redis_config[:size] || [@concurrency + POOL_HEADROOM, MIN_POOL_SIZE].max
216
+ end
217
+
199
218
  def build_pool(size:, name:)
200
- cfg = @config.redis_config
201
- RedisPool.new(
202
- size: size,
203
- url: cfg[:url] || RedisPool::DEFAULT_URL,
204
- timeout: cfg[:timeout] || RedisPool::DEFAULT_TIMEOUT,
205
- name: name
206
- )
219
+ @config.new_redis_pool(size, name)
207
220
  end
208
221
  end
209
222
  end
data/lib/wurk/cli.rb CHANGED
@@ -45,7 +45,12 @@ module Wurk
45
45
  cli.launcher.quiet
46
46
  end,
47
47
  'TTIN' => BACKTRACE_DUMPER,
48
- 'INFO' => BACKTRACE_DUMPER
48
+ 'INFO' => BACKTRACE_DUMPER,
49
+ 'USR2' => lambda do |cli|
50
+ cli.logger.info 'Received USR2, reopening logs'
51
+ # reopen_logs is private — an explicit receiver needs __send__.
52
+ cli.__send__(:reopen_logs)
53
+ end
49
54
  }.freeze
50
55
 
51
56
  # Table-driven so adding a flag doesn't grow `define_value_flags`'s ABC
@@ -137,7 +142,8 @@ module Wurk
137
142
  validate_pool_sizes!
138
143
  @config[:identity] = identity
139
144
  ::Process.warmup if warmup && ::Process.respond_to?(:warmup) && ENV['RUBY_DISABLE_WARMUP'] != '1'
140
- @swarm = Wurk::Swarm.new(topology: @config.topology, config: @config)
145
+ @swarm = Wurk::Swarm.new(topology: @config.topology, config: @config,
146
+ shutdown_timeout: @config[:timeout] || Swarm::DEFAULT_SHUTDOWN_TIMEOUT)
141
147
  @swarm.boot(install_signals: true)
142
148
  @swarm.supervise
143
149
  end
@@ -189,7 +195,14 @@ module Wurk
189
195
  end
190
196
 
191
197
  def signal_names
192
- %w[INT TERM TSTP TTIN INFO]
198
+ %w[INT TERM TSTP TTIN INFO USR2]
199
+ end
200
+
201
+ def reopen_logs
202
+ log = logger
203
+ log.reopen if log.respond_to?(:reopen)
204
+ rescue StandardError
205
+ nil
193
206
  end
194
207
 
195
208
  def validate_redis!
@@ -137,20 +137,38 @@ module Wurk
137
137
  def capture_pool_from_client(client)
138
138
  return unless client && !buffer_client_factory
139
139
 
140
- pool = client.instance_variable_get(:@pool)
140
+ pool = client.instance_variable_get(:@redis_pool)
141
+ # A nil capture must not install a factory: it would pin the drainer
142
+ # to the DEFAULT pool forever (the `!buffer_client_factory` guard
143
+ # blocks any later, correct capture) — wrong Redis for jobs pushed
144
+ # through an explicit-pool client.
145
+ return unless pool
146
+
141
147
  self.buffer_client_factory = -> { Wurk::Client.new(pool: pool) }
142
148
  end
143
149
 
144
150
  public
145
151
 
146
152
  # Drain payloads through `raw_push` on the given client. Stops on
147
- # the first ConnectionError, preserving order at the head of the
148
- # buffer so the next push retries the same payload. Emits statsd
153
+ # the first transient failure (ConnectionError past the pool's own
154
+ # retries, or a starved checkout), preserving order at the head of
155
+ # the buffer so the next push retries the same payload. Emits statsd
149
156
  # `jobs.recovered.push` per drained payload.
150
157
  def drain!(client)
151
158
  drained = 0
152
159
  while (payload = pop_head)
153
- unless attempt_replay(client, payload)
160
+ begin
161
+ replayed = attempt_replay(client, payload)
162
+ rescue StandardError
163
+ # Non-connection failures (OOM, LOADING, READONLY…) must not
164
+ # drop the popped payload — restore it before propagating, or
165
+ # a recovering-but-not-ready Redis silently eats one buffered
166
+ # job per drain tick.
167
+ buffer_mutex.synchronize { buffer.unshift(payload) }
168
+ raise
169
+ end
170
+
171
+ unless replayed
154
172
  buffer_mutex.synchronize { buffer.unshift(payload) }
155
173
  break
156
174
  end
@@ -206,14 +224,14 @@ module Wurk
206
224
  buffer_mutex.synchronize { buffer.shift }
207
225
  end
208
226
 
209
- # Drain marks the thread so our prepended raw_push re-raises
210
- # ConnectionError back here instead of swallowing it into the buffer
227
+ # Drain marks the thread so our prepended raw_push re-raises the
228
+ # transient error back here instead of swallowing it into the buffer
211
229
  # (which would spin forever).
212
230
  def attempt_replay(client, payload)
213
231
  Thread.current[DRAINING_KEY] = true
214
232
  client.send(:raw_push, [payload])
215
233
  true
216
- rescue RedisClient::ConnectionError
234
+ rescue RedisClient::ConnectionError, ConnectionPool::TimeoutError
217
235
  false
218
236
  ensure
219
237
  Thread.current[DRAINING_KEY] = false
@@ -222,7 +240,7 @@ module Wurk
222
240
 
223
241
  # Background drain thread. Wakes every `interval` seconds and tries
224
242
  # `Buffered.drain!` against a fresh Wurk::Client. drain! already
225
- # short-circuits on the first ConnectionError, so a still-down Redis
243
+ # short-circuits on the first transient failure, so a still-down Redis
226
244
  # just leaves the buffer alone for this tick — no exponential
227
245
  # backoff or explicit "reconnect detection" needed; the inner
228
246
  # connection retry already lives inside `client.raw_push`.
@@ -292,7 +310,9 @@ module Wurk
292
310
  end
293
311
 
294
312
  # Wraps Wurk::Client. push / push_bulk drain the buffer first;
295
- # raw_push catches ConnectionError and buffers non-batched payloads.
313
+ # raw_push catches transient failures — RedisClient::ConnectionError
314
+ # past RedisPool's own retries, or a starved checkout
315
+ # (ConnectionPool::TimeoutError) — and buffers non-batched payloads.
296
316
  module InstanceMethods
297
317
  def push(item)
298
318
  Buffered.drain!(self)
@@ -308,7 +328,7 @@ module Wurk
308
328
 
309
329
  def raw_push(payloads)
310
330
  super
311
- rescue RedisClient::ConnectionError
331
+ rescue RedisClient::ConnectionError, ConnectionPool::TimeoutError
312
332
  raise if Thread.current[Buffered::DRAINING_KEY]
313
333
 
314
334
  bidless, batched = payloads.partition { |p| !p['bid'] }
data/lib/wurk/client.rb CHANGED
@@ -246,12 +246,26 @@ module Wurk
246
246
 
247
247
  def atomic_push(conn, payloads)
248
248
  if payloads.first['at']
249
- conn.pipelined { |pipe| push_scheduled(pipe, payloads) }
249
+ push_scheduled_split(conn, payloads)
250
250
  else
251
251
  push_immediate(conn, payloads)
252
252
  end
253
253
  end
254
254
 
255
+ # Scheduled payloads split like push_immediate: a `bid`-carrying job (a
256
+ # `perform_in` inside `batch.jobs`) must register into its batch at creation
257
+ # via BATCH_SCHEDULE so `total`/`pending` move now — a bare ZADD would leave
258
+ # the batch counters at zero and the empty-marker check would misfire. Plain
259
+ # scheduled jobs take the bare ZADD. Separate pipelines for the same
260
+ # NOSCRIPT-replay reason as push_immediate: a Lua NOSCRIPT surfaces only at
261
+ # pipeline finalize, so a unified retry would replay the plain ZADD and
262
+ # duplicate the scheduled entry.
263
+ def push_scheduled_split(conn, payloads)
264
+ batched, plain = payloads.partition { |j| j['bid'] }
265
+ conn.pipelined { |pipe| push_scheduled(pipe, plain) } unless plain.empty?
266
+ push_batched_scheduled_pipelined(conn, batched) unless batched.empty?
267
+ end
268
+
255
269
  def push_scheduled(conn, payloads)
256
270
  args = payloads.flat_map do |hash|
257
271
  [hash['at'].to_s, Wurk.dump_json(hash.except('enqueued_at', 'at'))]
@@ -288,6 +302,17 @@ module Wurk
288
302
  conn.pipelined { |pipe| push_batched(pipe, batched, now, eval_method: :eval_with_source) }
289
303
  end
290
304
 
305
+ # Same NOSCRIPT-recovery shape as push_batched_pipelined, for the scheduled
306
+ # batched path (BATCH_SCHEDULE instead of BATCH_PUSH).
307
+ def push_batched_scheduled_pipelined(conn, batched)
308
+ conn.pipelined { |pipe| push_batched_scheduled(pipe, batched) }
309
+ rescue RedisClient::CommandError => e
310
+ raise unless e.message.to_s.start_with?('NOSCRIPT')
311
+
312
+ Wurk::Lua::Loader.script_load_all(conn)
313
+ conn.pipelined { |pipe| push_batched_scheduled(pipe, batched, eval_method: :eval_with_source) }
314
+ end
315
+
291
316
  def push_plain(conn, payloads, now)
292
317
  grouped = payloads.group_by { |j| j['queue'] }
293
318
  conn.call('SADD', 'queues', *grouped.keys)
@@ -321,6 +346,25 @@ module Wurk
321
346
  end
322
347
  end
323
348
 
349
+ # Scheduled batched jobs route through BATCH_SCHEDULE: the SADD-guarded
350
+ # total/pending increment registers the job in its batch at creation, and
351
+ # the ZADD defers it onto `schedule`. Payload is stripped of `at`/
352
+ # `enqueued_at` exactly like push_scheduled — `enqueued_at` is stamped fresh
353
+ # at promotion, never while the job sits scheduled (spec §7.1). One Redis
354
+ # round-trip per job because the Lua binds per-job KEYS; scheduled batch
355
+ # enqueue is not the hot path.
356
+ def push_batched_scheduled(conn, payloads, eval_method: :eval_cached)
357
+ payloads.each do |j|
358
+ Wurk::Lua::Loader.public_send(
359
+ eval_method,
360
+ conn,
361
+ :batch_schedule,
362
+ keys: ['schedule', "b-#{j['bid']}", "b-#{j['bid']}-jids"],
363
+ argv: [j['at'].to_s, Wurk.dump_json(j.except('enqueued_at', 'at')), j['jid']]
364
+ )
365
+ end
366
+ end
367
+
324
368
  def pool
325
369
  @redis_pool || Thread.current[:wurk_via_pool] || @config.redis_pool
326
370
  end
@@ -22,6 +22,9 @@ module Wurk
22
22
 
23
23
  attr_reader :config
24
24
 
25
+ # `leader?` cache TTL — see the method doc below.
26
+ LEADER_CACHE_TTL_MS = 5_000
27
+
25
28
  # --- clocks ---------------------------------------------------------
26
29
 
27
30
  def real_ms
@@ -71,20 +74,25 @@ module Wurk
71
74
  # --- cluster leadership --------------------------------------------
72
75
 
73
76
  # True iff this process currently holds the cluster `dear-leader` lock.
74
- # Per spec, the check is performed at call time (Wurk does not cache);
75
- # callers must not poll faster than the 60s follower cadence. Returns
76
- # false unconditionally when `WURK_LEADER=false` (or `SIDEKIQ_LEADER=false`)
77
- # is set on the process (opt-out hot-standby). Any Redis error is swallowed →
78
- # false, so a transient partition can't propagate as an exception into user
79
- # code.
77
+ # Cached per Component instance for `LEADER_CACHE_TTL_MS` (~5s): cron and
78
+ # the metrics rollups call this every tick, and an uncached GET would
79
+ # double their Redis traffic at short intervals for no benefit — the
80
+ # lock's own renewal cadence (60s+, spec §6.1) easily tolerates a
81
+ # few-second-stale read. Returns false unconditionally when
82
+ # `WURK_LEADER=false` (or `SIDEKIQ_LEADER=false`) is set on the process
83
+ # (opt-out hot-standby). Any Redis error is swallowed → false, so a
84
+ # transient partition can't propagate as an exception into user code.
80
85
  #
81
86
  # Spec: docs/target/sidekiq-ent.md §6.1.
82
87
  def leader?
83
88
  return false if Wurk::Leader.opted_out?
84
89
 
85
- redis { |c| c.call('GET', Wurk::Leader::DEFAULT_KEY) } == identity
86
- rescue StandardError
87
- false
90
+ now = mono_ms
91
+ if @leader_checked_at.nil? || (now - @leader_checked_at) >= LEADER_CACHE_TTL_MS
92
+ @leader_checked_at = now
93
+ @leader_cached = fetch_leader?
94
+ end
95
+ @leader_cached
88
96
  end
89
97
 
90
98
  # --- thread boundaries ---------------------------------------------
@@ -127,6 +135,12 @@ module Wurk
127
135
 
128
136
  private
129
137
 
138
+ def fetch_leader?
139
+ redis { |c| c.call('GET', Wurk::Leader::DEFAULT_KEY) } == identity
140
+ rescue StandardError
141
+ false
142
+ end
143
+
130
144
  def run_lifecycle_hook(hook, event, reraise)
131
145
  hook.call
132
146
  rescue StandardError => e
@@ -43,7 +43,8 @@ module Wurk
43
43
  reloader: proc { |&b| b.call },
44
44
  backtrace_cleaner: ->(bt) { bt },
45
45
  logged_job_attributes: %w[bid tags],
46
- redis_idle_timeout: nil
46
+ redis_idle_timeout: nil,
47
+ redis_error_handlers: []
47
48
  }.freeze
48
49
 
49
50
  # :fork fires only inside swarm children, after fork + internal AR/Redis
@@ -51,6 +52,13 @@ module Wurk
51
52
  LIFECYCLE_EVENTS = %i[startup fork quiet shutdown exit heartbeat beat leader].freeze
52
53
  DEFAULT_THREAD_PRIORITY = -1
53
54
 
55
+ # Redis client / pool errors that the pool wrapper already retried before
56
+ # re-raising. Logged one level up (WARN, not INFO) so a transient blip
57
+ # surfaces in ops dashboards without drowning steady-state noise (#101).
58
+ # RedisClient + ConnectionPool are always loaded before this file (capsule
59
+ # → redis_pool requires both), so referencing them here is safe.
60
+ REDIS_ERROR_CLASSES = [RedisClient::Error, ConnectionPool::TimeoutError].freeze
61
+
54
62
  # Default error handler. Wraps the report in the thread-local
55
63
  # Wurk::Context so logger formatters/JSON layouts can pick up jid/bid/tags.
56
64
  # `full_message` (with backtrace) in dev/debug, `detailed_message` in prod —
@@ -62,7 +70,8 @@ module Wurk
62
70
  Wurk::Context.with(safe_ctx) do
63
71
  dev = $DEBUG || ENV['WURK_DEBUG'] || cfg.logger.debug?
64
72
  msg = dev ? ex.full_message : ex.detailed_message
65
- cfg.logger.info { msg }
73
+ level = REDIS_ERROR_CLASSES.any? { |k| ex.is_a?(k) } ? :warn : :info
74
+ cfg.logger.public_send(level) { msg }
66
75
  end
67
76
  end
68
77
 
@@ -86,6 +95,7 @@ module Wurk
86
95
  @client_chain = Middleware::Chain.new
87
96
  @server_chain = Middleware::Chain.new
88
97
  @redis_config = { url: ENV.fetch('REDIS_URL', 'redis://localhost:6379/0') }
98
+ @web_redis_pool = nil
89
99
  @logger = nil
90
100
  @thread_priority = DEFAULT_THREAD_PRIORITY
91
101
  @frozen = false
@@ -168,17 +178,13 @@ module Wurk
168
178
  default_capsule.redis_pool
169
179
  end
170
180
 
171
- def local_redis_pool
172
- @local_redis_pool ||= build_redis_pool(size: 10, name: 'internal')
173
- end
174
-
175
- # Disconnect and drop every cached pool — the per-capsule mains plus
176
- # the config-level internal pool. Used by Wurk::Swarm so the parent
177
- # never leaks sockets into forks and each child can build fresh ones.
181
+ # Disconnect and drop every capsule's cached pools (main + fetch) plus the
182
+ # web pool. Used by Wurk::Swarm so the parent never leaks sockets into forks
183
+ # and each child can build fresh ones.
178
184
  def reset_redis_pools!
179
185
  @capsules.each_value(&:reset_redis_pools!)
180
- @local_redis_pool&.disconnect!
181
- @local_redis_pool = nil
186
+ @web_redis_pool&.disconnect!
187
+ @web_redis_pool = nil
182
188
  end
183
189
 
184
190
  def new_redis_pool(size, name = 'custom')
@@ -189,6 +195,39 @@ module Wurk
189
195
  redis_pool.with(&)
190
196
  end
191
197
 
198
+ # --- Web dashboard Redis pool ----------------------------------------
199
+
200
+ # Default connection count for the dedicated web pool (#web_redis_pool).
201
+ WEB_POOL_DEFAULT_SIZE = 5
202
+
203
+ # Deliberately short checkout wait for the web pool: a saturated dashboard
204
+ # should fail fast rather than tie up a web-server thread queuing for a slot.
205
+ WEB_POOL_TIMEOUT = 1.0
206
+
207
+ # Connections in the dedicated web pool. `config.web_pool_size = N` overrides
208
+ # the default; independent of `config.redis[:size]`, which sizes the worker
209
+ # capsules — the two pools are deliberately disjoint (#101).
210
+ def web_pool_size
211
+ @options[:web_pool_size] || WEB_POOL_DEFAULT_SIZE
212
+ end
213
+
214
+ def web_pool_size=(size)
215
+ guard_frozen!
216
+ @options[:web_pool_size] = Integer(size)
217
+ end
218
+
219
+ # Dedicated Redis pool for the dashboard / JSON API / SSE, disjoint from
220
+ # every worker capsule's pool. Dashboard load — an API burst, a long-lived
221
+ # SSE stream — can no longer drain the connections a co-located (embedded)
222
+ # worker needs to fetch and heartbeat, and vice versa: the #101 `0/N`
223
+ # pool-exhaustion incident. Lazy, so a headless worker that never serves the
224
+ # dashboard builds nothing; web entry points route `Wurk.redis` here through
225
+ # Wurk::Web::PoolScope. The Configuration instance is never frozen (only its
226
+ # @options/@capsules are), so this `||=` is safe to fire post-boot.
227
+ def web_redis_pool
228
+ @web_redis_pool ||= build_redis_pool(size: web_pool_size, name: 'web', pool_timeout: WEB_POOL_TIMEOUT)
229
+ end
230
+
192
231
  # --- Service locator (extension registry) ----------------------------
193
232
 
194
233
  def register(name, instance)
@@ -210,6 +249,19 @@ module Wurk
210
249
  @options[:death_handlers]
211
250
  end
212
251
 
252
+ # Telemetry hook fired by RedisPool on every transient-error retry and
253
+ # final give-up. The block receives one Hash: { error:, attempt:, retried:,
254
+ # pool: }. Opt-in — pools stay silent until a handler is registered.
255
+ def on_redis_error(&block)
256
+ raise ArgumentError, 'block required for on_redis_error' unless block
257
+
258
+ @options[:redis_error_handlers] << block
259
+ end
260
+
261
+ def redis_error_handlers
262
+ @options[:redis_error_handlers]
263
+ end
264
+
213
265
  def average_scheduled_poll_interval=(interval)
214
266
  @options[:average_scheduled_poll_interval] = interval
215
267
  end
@@ -495,13 +547,26 @@ module Wurk
495
547
  logger
496
548
  end
497
549
 
498
- def build_redis_pool(size:, name:)
499
- RedisPool.new(
500
- size: size,
501
- url: @redis_config[:url] || RedisPool::DEFAULT_URL,
502
- timeout: @redis_config[:timeout] || RedisPool::DEFAULT_TIMEOUT,
503
- name: name
504
- )
550
+ # `size`/`name` are pool-structural (the caller owns them); every other
551
+ # key the host set via `config.redis = {...}` — url, the split timeouts,
552
+ # reconnect_attempts, driver, … — flows through to RedisPool verbatim.
553
+ # `overrides` win over the host config (the web pool pins its own
554
+ # pool_timeout this way). Every pool built here is wired to the redis-error
555
+ # telemetry dispatcher.
556
+ def build_redis_pool(size:, name:, **overrides)
557
+ RedisPool.new(size: size, name: name, on_error: method(:dispatch_redis_error),
558
+ **@redis_config.except(:size, :name), **overrides)
559
+ end
560
+
561
+ # Fan a RedisPool retry/give-up event out to the registered handlers. A
562
+ # raising handler is logged and skipped so one bad hook can't break the
563
+ # pool's retry path (mirrors #handle_exception).
564
+ def dispatch_redis_error(info)
565
+ redis_error_handlers.each do |handler|
566
+ handler.call(info)
567
+ rescue StandardError => e
568
+ logger.error("redis_error_handler raised: #{e.class}: #{e.message}")
569
+ end
505
570
  end
506
571
  end
507
572
  end
data/lib/wurk/cron.rb CHANGED
@@ -170,11 +170,23 @@ module Wurk
170
170
  step_i = Integer(step)
171
171
  raise ArgumentError, "cron step must be >= 1 (got #{step_i})" if step_i < 1
172
172
 
173
- base_values = base == '*' ? (min..max).to_a : parse_chunk(base, min, max)
173
+ base_values = step_base_values(base, min, max)
174
174
  start = base_values.first
175
175
  base_values.select { |v| ((v - start) % step_i).zero? }
176
176
  end
177
177
 
178
+ # Vixie cron: a single-value step base means "start..max" (`5/15` in the
179
+ # minute field = 5,20,35,50), not the single value alone — fugit and
180
+ # Sidekiq Ent agree.
181
+ def step_base_values(base, min, max)
182
+ return (min..max).to_a if base == '*'
183
+ return parse_range(base, min, max) if base.include?('-')
184
+
185
+ v = Integer(base)
186
+ validate_value!(v, min, max)
187
+ (v..max).to_a
188
+ end
189
+
178
190
  def parse_range(chunk, min, max)
179
191
  a, b = chunk.split('-', 2).map { |x| Integer(x) }
180
192
  raise ArgumentError, "cron range start > end (#{a} > #{b})" if a > b
data/lib/wurk/dead_set.rb CHANGED
@@ -38,12 +38,27 @@ module Wurk
38
38
  Wurk.redis do |conn|
39
39
  conn.pipelined do |pipe|
40
40
  pipe.call('ZREMRANGEBYSCORE', @name, '-inf', "(#{cutoff}")
41
- pipe.call('ZREMRANGEBYRANK', @name, 0, -(max_jobs + 1))
41
+ pipe.call('ZREMRANGEBYRANK', @name, 0, -max_jobs)
42
42
  end
43
43
  end
44
44
  true
45
45
  end
46
46
 
47
+ # ZADD the raw JSON payload (score = now) then apply the two-axis #trim.
48
+ # Shared primitive behind both morgue entry points so the trim runs on
49
+ # EVERY kill (spec §31.8), including the malformed-JSON path:
50
+ # * JobRetry#send_to_morgue — exhausted retries
51
+ # * Processor#parse_or_kill — unparseable payloads
52
+ # No death handlers: JobRetry fires its own via #run_death_handlers and the
53
+ # malformed path has no parseable job to hand a handler. `max_jobs:` /
54
+ # `timeout:` propagate to #trim (see there for the override rationale).
55
+ def kill_raw(payload, max_jobs: nil, timeout: nil) # rubocop:disable Naming/PredicateMethod
56
+ now = ::Process.clock_gettime(::Process::CLOCK_REALTIME)
57
+ Wurk.redis { |conn| conn.call('ZADD', @name, now.to_s, payload) }
58
+ trim(max_jobs: max_jobs, timeout: timeout)
59
+ true
60
+ end
61
+
47
62
  # ZADD the raw JSON payload, trim, fire death handlers. `notify_failure:
48
63
  # true` (default) routes the kill through the death-handler chain;
49
64
  # UI-initiated kills pass false. `ex` is the originating exception (or
data/lib/wurk/engine.rb CHANGED
@@ -45,11 +45,27 @@ module Wurk
45
45
  # Precompiled SPA lives in vendor/assets/dashboard; the engine serves
46
46
  # those files as static assets under the /wurk-assets mount point via
47
47
  # AssetMount (above).
48
+ #
49
+ # Deliberately unauthenticated: this middleware is inserted straight into
50
+ # the *host app's* middleware stack (`app.middleware`), not the engine's
51
+ # own (`middleware.use` in this class, below) — so it runs before
52
+ # anything wired via `Wurk::Web.use`/`Authorization` even sees the
53
+ # request, and `/wurk-assets/*` is reachable without passing either
54
+ # check. That's intentional, not an oversight: the bundle is the
55
+ # compiled JS/CSS/font shell only (no job payloads, no Redis reads, no
56
+ # per-user state — every data-bearing byte comes from the JSON API,
57
+ # which *does* sit behind the engine's Authorization middleware). It's
58
+ # the same trust model as serving `public/assets` from any other Rails
59
+ # app. See README "Security notes" for the full reasoning.
48
60
  initializer 'wurk.assets' do |app|
49
61
  assets_path = Wurk::Engine.root.join('vendor', 'assets', 'dashboard')
50
62
  if assets_path.exist?
63
+ # Index 0, not insert_before(ActionDispatch::Static): Static is only
64
+ # in the stack when public_file_server is enabled, and a stock
65
+ # production deploy behind nginx (RAILS_SERVE_STATIC_FILES unset)
66
+ # omits it — insert_before(Static) would crash the host app's boot.
51
67
  app.middleware.insert_before(
52
- ::ActionDispatch::Static,
68
+ 0,
53
69
  ::Wurk::Engine::AssetMount,
54
70
  root: assets_path.to_s
55
71
  )