wurk 1.7.6 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (217) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +7 -4
  3. data/app/controllers/concerns/wurk/sse_streaming.rb +2 -1
  4. data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +33 -9
  5. data/app/controllers/wurk/api/pagination.rb +25 -7
  6. data/app/controllers/wurk/api_controller.rb +49 -25
  7. data/app/controllers/wurk/extensions_controller.rb +8 -2
  8. data/app/controllers/wurk/profiles_controller.rb +45 -9
  9. data/config/routes.rb +31 -10
  10. data/lib/generators/wurk/install/templates/wurk.rb +1 -1
  11. data/lib/sidekiq/capsule.rb +4 -0
  12. data/lib/sidekiq/component.rb +4 -0
  13. data/lib/sidekiq/config.rb +4 -0
  14. data/lib/sidekiq/deploy.rb +4 -0
  15. data/lib/sidekiq/embedded.rb +4 -0
  16. data/lib/sidekiq/fetch.rb +4 -0
  17. data/lib/sidekiq/iterable_job.rb +4 -0
  18. data/lib/sidekiq/job/interrupt_handler.rb +4 -0
  19. data/lib/sidekiq/job/iterable/active_record_enumerator.rb +4 -0
  20. data/lib/sidekiq/job/iterable/csv_enumerator.rb +4 -0
  21. data/lib/sidekiq/job/iterable.rb +4 -0
  22. data/lib/sidekiq/job_logger.rb +4 -0
  23. data/lib/sidekiq/job_util.rb +4 -0
  24. data/lib/sidekiq/loader.rb +4 -0
  25. data/lib/sidekiq/logger.rb +4 -0
  26. data/lib/sidekiq/metrics/query.rb +4 -0
  27. data/lib/sidekiq/middleware/current_attributes.rb +7 -0
  28. data/lib/sidekiq/middleware/i18n.rb +6 -0
  29. data/lib/sidekiq/middleware/modules.rb +4 -0
  30. data/lib/sidekiq/pro/web.rb +5 -0
  31. data/lib/sidekiq/profiler.rb +4 -0
  32. data/lib/sidekiq/redis_client_adapter.rb +4 -0
  33. data/lib/sidekiq/test_api.rb +5 -0
  34. data/lib/sidekiq/testing/inline.rb +8 -0
  35. data/lib/sidekiq/transaction_aware_client.rb +4 -0
  36. data/lib/sidekiq/worker_compatibility_alias.rb +4 -0
  37. data/lib/sidekiq-ent/periodic/testing.rb +5 -0
  38. data/lib/sidekiq-ent/web.rb +5 -0
  39. data/lib/sidekiq-ent.rb +7 -0
  40. data/lib/sidekiq-pro.rb +7 -0
  41. data/lib/wurk/api/fast.rb +8 -33
  42. data/lib/wurk/api/idempotency.rb +28 -5
  43. data/lib/wurk/api/swarm.rb +1 -1
  44. data/lib/wurk/batch/callbacks.rb +143 -92
  45. data/lib/wurk/batch/death_handler.rb +19 -8
  46. data/lib/wurk/batch/server_middleware.rb +37 -24
  47. data/lib/wurk/batch/status.rb +49 -15
  48. data/lib/wurk/batch.rb +164 -65
  49. data/lib/wurk/capsule.rb +0 -2
  50. data/lib/wurk/client/buffered.rb +227 -106
  51. data/lib/wurk/client.rb +33 -13
  52. data/lib/wurk/compat.rb +13 -0
  53. data/lib/wurk/configuration.rb +155 -24
  54. data/lib/wurk/cpu_count.rb +97 -0
  55. data/lib/wurk/cron.rb +97 -34
  56. data/lib/wurk/dead_set.rb +6 -7
  57. data/lib/wurk/encryption.rb +46 -13
  58. data/lib/wurk/fetcher/many_queues.rb +65 -0
  59. data/lib/wurk/fetcher/orphan_grace.rb +67 -0
  60. data/lib/wurk/fetcher/private_list_key.rb +76 -0
  61. data/lib/wurk/fetcher/reaper.rb +55 -74
  62. data/lib/wurk/fetcher/reliable.rb +23 -17
  63. data/lib/wurk/fetcher/unit_of_work.rb +19 -1
  64. data/lib/wurk/fetcher/unparseable_keys.rb +56 -0
  65. data/lib/wurk/flow/completion.rb +28 -3
  66. data/lib/wurk/flow/creation.rb +24 -4
  67. data/lib/wurk/flow.rb +24 -2
  68. data/lib/wurk/health.rb +182 -40
  69. data/lib/wurk/heartbeat.rb +37 -14
  70. data/lib/wurk/history.rb +7 -1
  71. data/lib/wurk/import/sidekiq_cron.rb +248 -0
  72. data/lib/wurk/job/options.rb +22 -3
  73. data/lib/wurk/job_logger.rb +13 -3
  74. data/lib/wurk/job_record.rb +68 -19
  75. data/lib/wurk/job_retry.rb +12 -6
  76. data/lib/wurk/job_set.rb +99 -51
  77. data/lib/wurk/job_util.rb +27 -8
  78. data/lib/wurk/launcher.rb +84 -40
  79. data/lib/wurk/leader.rb +53 -28
  80. data/lib/wurk/limiter/base.rb +11 -30
  81. data/lib/wurk/limiter/bucket.rb +14 -5
  82. data/lib/wurk/limiter/server_middleware.rb +2 -1
  83. data/lib/wurk/limiter.rb +1 -1
  84. data/lib/wurk/loader.rb +48 -0
  85. data/lib/wurk/lua/batch_ack_complete.lua +28 -0
  86. data/lib/wurk/lua/batch_ack_failed.lua +18 -0
  87. data/lib/wurk/lua/batch_ack_success.lua +35 -0
  88. data/lib/wurk/lua/batch_append_callback.lua +66 -0
  89. data/lib/wurk/lua/batch_delete.lua +31 -0
  90. data/lib/wurk/lua/batch_hold.lua +17 -0
  91. data/lib/wurk/lua/batch_invalidate.lua +14 -0
  92. data/lib/wurk/lua/batch_kid_done.lua +14 -0
  93. data/lib/wurk/lua/batch_push.lua +47 -0
  94. data/lib/wurk/lua/batch_remove_jobs.lua +29 -0
  95. data/lib/wurk/lua/batch_schedule.lua +27 -0
  96. data/lib/wurk/lua/fetch_first.lua +27 -0
  97. data/lib/wurk/lua/flow_abandon.lua +16 -21
  98. data/lib/wurk/lua/flow_advance.lua +41 -26
  99. data/lib/wurk/lua/flow_create.lua +24 -9
  100. data/lib/wurk/lua/flow_fail.lua +3 -3
  101. data/lib/wurk/lua/leader_campaign.lua +19 -0
  102. data/lib/wurk/lua/limiter_bucket_acquire.lua +3 -2
  103. data/lib/wurk/lua/limiter_list_sweep.lua +8 -9
  104. data/lib/wurk/lua/loader.rb +36 -23
  105. data/lib/wurk/lua/throttle_slot.lua +13 -26
  106. data/lib/wurk/lua.rb +31 -244
  107. data/lib/wurk/manager.rb +25 -18
  108. data/lib/wurk/metrics/accumulator.rb +31 -10
  109. data/lib/wurk/metrics/dashboard_series.rb +163 -0
  110. data/lib/wurk/metrics/histogram.rb +46 -0
  111. data/lib/wurk/metrics/history.rb +49 -57
  112. data/lib/wurk/metrics/prometheus.rb +232 -0
  113. data/lib/wurk/metrics/query.rb +127 -155
  114. data/lib/wurk/metrics/rollup.rb +2 -2
  115. data/lib/wurk/metrics/statsd.rb +1 -1
  116. data/lib/wurk/middleware/interrupt_handler.rb +6 -2
  117. data/lib/wurk/middleware/poison_pill.rb +5 -4
  118. data/lib/wurk/middleware/timeout.rb +8 -8
  119. data/lib/wurk/process_set.rb +5 -5
  120. data/lib/wurk/processor.rb +69 -31
  121. data/lib/wurk/profile_set.rb +19 -24
  122. data/lib/wurk/profiler.rb +48 -50
  123. data/lib/wurk/queue.rb +36 -19
  124. data/lib/wurk/rails_boot.rb +33 -1
  125. data/lib/wurk/railtie.rb +2 -0
  126. data/lib/wurk/rake_tasks.rb +18 -0
  127. data/lib/wurk/redact.rb +39 -0
  128. data/lib/wurk/redis_connection.rb +6 -3
  129. data/lib/wurk/redis_options.rb +21 -4
  130. data/lib/wurk/redis_pool.rb +119 -21
  131. data/lib/wurk/scheduled.rb +44 -11
  132. data/lib/wurk/sentry/error_handler.rb +53 -7
  133. data/lib/wurk/sentry/middleware.rb +7 -8
  134. data/lib/wurk/shutdown_gate.rb +24 -1
  135. data/lib/wurk/sorted_entry.rb +16 -5
  136. data/lib/wurk/stats.rb +9 -4
  137. data/lib/wurk/status.rb +2 -2
  138. data/lib/wurk/swarm/child_boot.rb +48 -38
  139. data/lib/wurk/swarm/liveness.rb +153 -0
  140. data/lib/wurk/swarm/restart.rb +24 -3
  141. data/lib/wurk/swarm.rb +115 -24
  142. data/lib/wurk/telemetry/server_middleware.rb +2 -5
  143. data/lib/wurk/testing.rb +15 -5
  144. data/lib/wurk/throttle.rb +16 -1
  145. data/lib/wurk/topology.rb +1 -5
  146. data/lib/wurk/unique.rb +1 -1
  147. data/lib/wurk/version.rb +1 -1
  148. data/lib/wurk/watchdog.rb +2 -2
  149. data/lib/wurk/web/config.rb +56 -20
  150. data/lib/wurk/web/enterprise.rb +21 -18
  151. data/lib/wurk/web/extension.rb +98 -22
  152. data/lib/wurk/web/rack_app.rb +12 -11
  153. data/lib/wurk/web/search.rb +18 -8
  154. data/lib/wurk/worker/setter.rb +17 -14
  155. data/lib/wurk/worker.rb +6 -67
  156. data/lib/wurk.rb +2 -3
  157. data/vendor/assets/dashboard/assets/ArgsValue-DsPb_KZ-.js +1 -0
  158. data/vendor/assets/dashboard/assets/BatchDetail-BCIXJOpZ.js +1 -0
  159. data/vendor/assets/dashboard/assets/Batches-IjaQfXW6.js +1 -0
  160. data/vendor/assets/dashboard/assets/Busy-CaQrf1Ve.js +1 -0
  161. data/vendor/assets/dashboard/assets/Cron-B18EHn_Y.js +1 -0
  162. data/vendor/assets/dashboard/assets/Dashboard-DdT5ZaUg.js +1 -0
  163. data/vendor/assets/dashboard/assets/Dead-DWIy-Mu8.js +1 -0
  164. data/vendor/assets/dashboard/assets/Extension-BjBUWhHf.js +1 -0
  165. data/vendor/assets/dashboard/assets/FilterBox-DwH3PXXt.js +1 -0
  166. data/vendor/assets/dashboard/assets/FlowDetail-DnexL72n.js +1 -0
  167. data/vendor/assets/dashboard/assets/FlowState-Cr4aX6-d.js +1 -0
  168. data/vendor/assets/dashboard/assets/Flows-DO6EimbZ.js +1 -0
  169. data/vendor/assets/dashboard/assets/JobDetailModal-aPjgq-U9.js +2 -0
  170. data/vendor/assets/dashboard/assets/Limiters-DxpwWCNe.js +1 -0
  171. data/vendor/assets/dashboard/assets/Metrics-C3JL1GIH.js +1 -0
  172. data/vendor/assets/dashboard/assets/NotFound-BfP9uAV9.js +1 -0
  173. data/vendor/assets/dashboard/assets/PageHeader-DbC97afG.js +1 -0
  174. data/vendor/assets/dashboard/assets/Profiles-BcaHreMe.js +1 -0
  175. data/vendor/assets/dashboard/assets/Queues-ZlBc5tyD.js +1 -0
  176. data/vendor/assets/dashboard/assets/Retries-Bg3nkeiL.js +1 -0
  177. data/vendor/assets/dashboard/assets/Scheduled-BrMZPZGl.js +1 -0
  178. data/vendor/assets/dashboard/assets/Search-B5X5pGht.js +1 -0
  179. data/vendor/assets/dashboard/assets/charts-uDyf4PN0.js +1 -0
  180. data/vendor/assets/dashboard/assets/i18n-CJeYfeVt.js +1 -0
  181. data/vendor/assets/dashboard/assets/index-B36QCAf-.css +1 -0
  182. data/vendor/assets/dashboard/assets/index-DQujgZgH.js +141 -0
  183. data/vendor/assets/dashboard/assets/useResetPageOnEmpty-O3-5rMk3.js +1 -0
  184. data/vendor/assets/dashboard/assets/useSort-xfnwjQAu.js +1 -0
  185. data/vendor/assets/dashboard/assets/utils-BIKHwNKB.js +1 -0
  186. data/vendor/assets/dashboard/index.html +3 -3
  187. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  188. metadata +90 -34
  189. data/config/locales/en.yml +0 -15
  190. data/vendor/assets/dashboard/assets/ArgsValue-D-x_ifLY.js +0 -1
  191. data/vendor/assets/dashboard/assets/BatchDetail-C39NJuew.js +0 -1
  192. data/vendor/assets/dashboard/assets/Batches-CSwo7Asa.js +0 -1
  193. data/vendor/assets/dashboard/assets/Busy-BOFMu-sq.js +0 -1
  194. data/vendor/assets/dashboard/assets/Cron-Dy8RQzDI.js +0 -1
  195. data/vendor/assets/dashboard/assets/Dashboard-BuTHI-O1.js +0 -1
  196. data/vendor/assets/dashboard/assets/Dead-B9KRvQ0N.js +0 -1
  197. data/vendor/assets/dashboard/assets/Extension-BnBVHfux.js +0 -1
  198. data/vendor/assets/dashboard/assets/FilterBox-DC24zite.js +0 -1
  199. data/vendor/assets/dashboard/assets/FlowDetail-DyLuzUvt.js +0 -1
  200. data/vendor/assets/dashboard/assets/FlowState-DAPKUahm.js +0 -1
  201. data/vendor/assets/dashboard/assets/Flows-Fr3rjZM_.js +0 -1
  202. data/vendor/assets/dashboard/assets/JobDetailModal-N6kiJXq3.js +0 -2
  203. data/vendor/assets/dashboard/assets/Limiters-kbFA7uS1.js +0 -1
  204. data/vendor/assets/dashboard/assets/Metrics-Dj2uoZ3o.js +0 -1
  205. data/vendor/assets/dashboard/assets/PageHeader-B_F94azl.js +0 -1
  206. data/vendor/assets/dashboard/assets/Profiles-D_DjEezN.js +0 -1
  207. data/vendor/assets/dashboard/assets/Queues-CO4V9hAz.js +0 -1
  208. data/vendor/assets/dashboard/assets/Retries-DCWnzeLa.js +0 -1
  209. data/vendor/assets/dashboard/assets/Scheduled-BebDUjLU.js +0 -1
  210. data/vendor/assets/dashboard/assets/Search-Cvr5fy4Y.js +0 -1
  211. data/vendor/assets/dashboard/assets/Skeleton-Bu3Ke6rV.js +0 -1
  212. data/vendor/assets/dashboard/assets/charts-BCs9bQKz.js +0 -1
  213. data/vendor/assets/dashboard/assets/index-BIwyOC5Q.js +0 -141
  214. data/vendor/assets/dashboard/assets/index-DBQN6Jk8.css +0 -1
  215. data/vendor/assets/dashboard/assets/useResetPageOnEmpty-Bzh-BJyL.js +0 -1
  216. data/vendor/assets/dashboard/assets/useSort-COA3fVJ5.js +0 -1
  217. data/vendor/assets/dashboard/assets/utils-BIrvZ1hi.js +0 -1
@@ -9,6 +9,11 @@ module Wurk
9
9
  # creation or batch-context pushes (`bid` on payload): BATCH_PUSH has
10
10
  # atomic counter side-effects we can't safely replay.
11
11
  #
12
+ # One buffer serves every pool, so each entry remembers the Redis it was
13
+ # headed for (its {Origin}) and is only ever replayed there: a sharded app
14
+ # pushing through `Client.via(shard_a)` must not have its outage backlog
15
+ # land on shard B because B's producer happened to push next.
16
+ #
12
17
  # Spec: docs/target/sidekiq-pro.md §5.
13
18
  module Buffered
14
19
  DEFAULT_BUFFER_CAP = 1_000
@@ -40,6 +45,31 @@ module Wurk
40
45
  # Returned by the append helpers when everything fit.
41
46
  NOTHING_UNDELIVERED = [].freeze
42
47
 
48
+ Entry = Struct.new(:payload, :origin)
49
+
50
+ # Where a buffered payload has to be replayed. What must NOT be captured
51
+ # is a pool object the config hands out: `reset_redis_pools!` — every
52
+ # fork, every embedded teardown — disconnects it and drops it for a
53
+ # lazily rebuilt one, and ConnectionPool#shutdown is terminal, so a
54
+ # pinned instance replays into dead sockets for the rest of the process's
55
+ # life. For such a pool the config (a Configuration or a Capsule, which
56
+ # survives the rebuild) is kept and asked again at replay time. A pool
57
+ # the config does not own — `Client.new(pool:)`, `Client.via(pool)` — is
58
+ # a second Redis nothing else can produce, so that one stays pinned,
59
+ # stale or not: replaying it anywhere else writes to the wrong server.
60
+ Origin = Struct.new(:config, :pool) do
61
+ def self.for(client)
62
+ pool = client.send(:pool)
63
+ config = client.instance_variable_get(:@config)
64
+ owned = config.respond_to?(:redis_pool) && config.redis_pool.equal?(pool)
65
+ owned ? new(config, nil) : new(nil, pool)
66
+ end
67
+
68
+ def resolve
69
+ config ? config.redis_pool : pool
70
+ end
71
+ end
72
+
43
73
  # Eagerly initialized: `||=` inside an accessor is not atomic — two
44
74
  # threads racing first-touch could end up holding distinct Mutex
45
75
  # instances and lose all synchronization on the shared buffer.
@@ -60,8 +90,6 @@ module Wurk
60
90
  @owner_pid = ::Process.pid
61
91
 
62
92
  class << self
63
- attr_accessor :buffer_client_factory
64
-
65
93
  # Idempotent. Prepends the wrapper module into Wurk::Client so push /
66
94
  # push_bulk drain the buffer before each call and raw_push catches
67
95
  # connection errors. Safe to call from multiple threads.
@@ -117,11 +145,10 @@ module Wurk
117
145
  @buffer = []
118
146
  @buffer_cap = nil
119
147
  @overflow_mode = nil
120
- @buffer_client_factory = nil
121
148
  end
122
- # Stop before dropping: an unstopped drainer thread survives with
123
- # its factory nil'd out from under it and ticks forever against
124
- # nothing, leaking the thread and everything its closure retains.
149
+ @drops = DropLog.new
150
+ # Stop before dropping: an unstopped drainer thread would otherwise
151
+ # tick on forever, unreachable, leaking the thread.
125
152
  install_mutex.synchronize do
126
153
  @drainer&.stop
127
154
  @drainer = nil
@@ -146,8 +173,7 @@ module Wurk
146
173
  # `@drainer` is dropped, never `stop`ped — its `@lock` carries the same
147
174
  # inherited-mutex hazard. A parent-configured drainer is replaced by an
148
175
  # equivalent fresh one so an opted-in child keeps flushing the buffer it
149
- # fills itself; the captured client factory goes with it, since it
150
- # closes over the parent's pre-fork Redis pool.
176
+ # fills itself.
151
177
  #
152
178
  # Deliberately unsynchronized: the child has exactly one thread here,
153
179
  # and waiting on the very mutex being replaced is what would hang it.
@@ -158,7 +184,7 @@ module Wurk
158
184
  @install_mutex = Mutex.new
159
185
  @buffer_mutex = Mutex.new
160
186
  @buffer = []
161
- @buffer_client_factory = nil
187
+ @drops = DropLog.new
162
188
  interval = @drainer&.interval
163
189
  @drainer = nil
164
190
  start_drainer!(interval: interval) if interval
@@ -172,62 +198,32 @@ module Wurk
172
198
  # raises one Overflow carrying every
173
199
  # payload that did not fit.
174
200
  # Drops batched payloads — caller is expected to re-raise for those.
175
- # If client is provided, captures its pool for drainer to use by default.
176
- def enbuffer(payloads, client: nil)
177
- capture_pool_from_client(client)
201
+ # `client` is the one whose push failed; every payload is tagged with
202
+ # the Redis that client was writing to (see Origin).
203
+ def enbuffer(payloads, client:)
204
+ origin = Origin.for(client)
205
+ entries = payloads.map { |payload| Entry.new(payload, origin) }
178
206
 
179
207
  cap = buffer_cap
180
208
  mode = overflow_mode
181
209
  undelivered = buffer_mutex.synchronize do
182
- mode == :raise ? append_within_capacity(payloads, cap) : append_dropping_oldest(payloads, cap)
210
+ mode == :raise ? append_within_capacity(entries, cap) : append_dropping_oldest(entries, cap)
183
211
  end
184
212
 
185
- raise Overflow, undelivered unless undelivered.empty?
213
+ raise Overflow, undelivered.map(&:payload) unless undelivered.empty?
214
+ ensure
215
+ drops.report(cap)
186
216
  end
187
217
 
188
218
  private
189
219
 
190
- # Remember how the buffering client reaches Redis, so the drainer
191
- # replays into the same server it was pushed to unless explicitly
192
- # overridden.
193
- def capture_pool_from_client(client)
194
- return unless client && !buffer_client_factory
195
-
196
- pool = client.instance_variable_get(:@redis_pool)
197
- # A nil capture must not install a factory: it would pin the drainer
198
- # to the DEFAULT pool forever (the `!buffer_client_factory` guard
199
- # blocks any later, correct capture) — wrong Redis for jobs pushed
200
- # through an explicit-pool client. A pool-less client already resolves
201
- # its config at push time, and so does the fallback factory.
202
- return unless pool
203
-
204
- resolver = pool_resolver(client, pool)
205
- self.buffer_client_factory = -> { Wurk::Client.new(pool: resolver.call) }
206
- end
207
-
208
- # What must NOT be captured is the pool object. `reset_redis_pools!` —
209
- # every fork, every embedded teardown — disconnects a pool and drops it
210
- # for a lazily rebuilt one, and ConnectionPool#shutdown is terminal, so
211
- # a pinned instance leaves the drainer replaying into dead sockets for
212
- # the rest of the process's life. The config (a Configuration or a
213
- # Capsule) is what survives that rebuild, so ask it again at drain time
214
- # whenever the client's pool is the one it hands out. A pool the config
215
- # does not own is a second Redis nothing else can produce — that one
216
- # stays pinned, stale or not, because replaying it anywhere else writes
217
- # to the wrong server.
218
- def pool_resolver(client, pool)
219
- config = client.instance_variable_get(:@config)
220
- config_owns = config.respond_to?(:redis_pool) && config.redis_pool.equal?(pool)
221
- config_owns ? -> { config.redis_pool } : -> { pool }
222
- end
223
-
224
- # Both append helpers run with buffer_mutex held and return the payloads
220
+ # Both append helpers run with buffer_mutex held and return the entries
225
221
  # they could not take.
226
222
 
227
- def append_dropping_oldest(payloads, cap)
228
- payloads.each do |p|
229
- buffer.shift if buffer.size >= cap
230
- buffer << p
223
+ def append_dropping_oldest(entries, cap)
224
+ entries.each do |entry|
225
+ drops.record(buffer.shift.payload) if buffer.size >= cap
226
+ buffer << entry
231
227
  end
232
228
  NOTHING_UNDELIVERED
233
229
  end
@@ -238,69 +234,78 @@ module Wurk
238
234
  # negative when the cap was lowered after the buffer filled — clamped,
239
235
  # so an over-full buffer rejects the whole call instead of raising on
240
236
  # `first`/`drop`.
241
- def append_within_capacity(payloads, cap)
242
- room = (cap - buffer.size).clamp(0, payloads.size)
243
- if room == payloads.size
244
- buffer.concat(payloads)
237
+ def append_within_capacity(entries, cap)
238
+ room = (cap - buffer.size).clamp(0, entries.size)
239
+ if room == entries.size
240
+ buffer.concat(entries)
245
241
  return NOTHING_UNDELIVERED
246
242
  end
247
243
 
248
- buffer.concat(payloads.first(room))
249
- payloads.drop(room)
244
+ buffer.concat(entries.first(room))
245
+ entries.drop(room)
250
246
  end
251
247
 
252
248
  public
253
249
 
254
- # Drain payloads through `raw_push` on the given client. Stops on
255
- # the first transient failure (ConnectionError past the pool's own
256
- # retries, or a starved checkout), preserving order at the head of
257
- # the buffer so the next push retries the same payload. Emits statsd
258
- # `jobs.recovered.push` per drained payload, plus the `jobs.enqueued`
259
- # the buffering push deliberately did not emit — the replay is where
260
- # the job actually reaches Redis, so a buffered-then-drained job counts
261
- # once as enqueued and once as recovered.
250
+ # Replay, through `raw_push` on `client`, the buffered payloads headed
251
+ # for the Redis that client writes to — oldest first; entries for any
252
+ # other Redis are left in place, so a push to a healthy shard neither
253
+ # misroutes nor waits on a dead one's backlog. Stops on the first
254
+ # transient failure (ConnectionError past the pool's own retries, or a
255
+ # starved checkout) and puts that payload back at the head, so the next
256
+ # push retries it first. Emits statsd `jobs.recovered.push` per drained
257
+ # payload, plus the `jobs.enqueued` the buffering push deliberately did
258
+ # not emit — the replay is where the job actually reaches Redis, so a
259
+ # buffered-then-drained job counts once as enqueued and once as
260
+ # recovered.
262
261
  def drain!(client)
263
- drained = 0
264
- while (payload = pop_head)
265
- begin
266
- replayed = attempt_replay(client, payload)
267
- rescue StandardError
268
- # Non-connection failures (OOM, LOADING, READONLY…) must not
269
- # drop the popped payload — restore it before propagating, or
270
- # a recovering-but-not-ready Redis silently eats one buffered
271
- # job per drain tick.
272
- buffer_mutex.synchronize { buffer.unshift(payload) }
273
- raise
274
- end
262
+ return 0 if buffer_mutex.synchronize { buffer.empty? }
275
263
 
276
- unless replayed
277
- buffer_mutex.synchronize { buffer.unshift(payload) }
278
- break
279
- end
264
+ target = client.send(:pool)
265
+ drained = 0
266
+ while (entry = take_next(target))
267
+ break unless replay(client, entry)
280
268
 
281
- client.send(:emit_enqueued, [payload])
282
- Wurk::Metrics::Statsd.increment('jobs.recovered.push')
283
269
  drained += 1
284
270
  end
271
+ drops.end_burst if drained.positive?
272
+ drained
273
+ end
274
+
275
+ # The background drainer's pass: every origin in the buffer, each
276
+ # through a client of its own, so one Redis still down does not hold
277
+ # back another that has recovered. Returns the total replayed; the
278
+ # first non-transient error is re-raised after every origin had its turn.
279
+ def drain_all!
280
+ failure = nil
281
+ drained = buffered_pools.sum do |pool|
282
+ drain!(Wurk::Client.new(pool: pool))
283
+ rescue StandardError => e
284
+ failure ||= e
285
+ 0
286
+ end
287
+ raise failure if failure
288
+
285
289
  drained
286
290
  end
287
291
 
288
- # Internal — visible for tests. Treat as private.
292
+ # Internal — visible for tests. Treat as private. Holds Entry structs.
289
293
  def buffer
290
294
  @buffer ||= []
291
295
  end
292
296
 
297
+ attr_reader :drops
298
+
293
299
  # Start a background drain thread that wakes every `interval`
294
300
  # seconds and tries to flush the buffer. Idempotent — replaces
295
301
  # any prior drainer with one at the new interval. Issue #19
296
302
  # requirement: "Background drain thread flushes on reconnect" —
297
303
  # handles the case where push activity stops mid-outage so the
298
304
  # passive (drain-on-next-push) path never fires.
299
- def start_drainer!(interval: Drainer::DEFAULT_INTERVAL, client_factory: nil)
305
+ def start_drainer!(interval: Drainer::DEFAULT_INTERVAL)
300
306
  install_mutex.synchronize do
301
307
  @drainer&.stop
302
- factory = client_factory || buffer_client_factory || -> { Wurk::Client.new }
303
- @drainer = Drainer.new(interval: interval, client_factory: factory)
308
+ @drainer = Drainer.new(interval: interval)
304
309
  @drainer.start
305
310
  end
306
311
  end
@@ -320,8 +325,45 @@ module Wurk
320
325
 
321
326
  attr_reader :install_mutex, :buffer_mutex
322
327
 
323
- def pop_head
324
- buffer_mutex.synchronize { buffer.shift }
328
+ # The oldest entry bound for `target`, removed from the buffer.
329
+ def take_next(target)
330
+ buffer_mutex.synchronize do
331
+ index = buffer.index { |entry| entry.origin.resolve.equal?(target) }
332
+ index && buffer.delete_at(index)
333
+ end
334
+ end
335
+
336
+ def buffered_pools
337
+ buffer_mutex.synchronize { buffer.map { |entry| entry.origin.resolve } }.uniq(&:object_id)
338
+ end
339
+
340
+ # The entry once it is in Redis, nil when it went back to the buffer.
341
+ # Same-origin order is all that matters (other entries go to another
342
+ # server), so a payload that could not be replayed goes back at the
343
+ # absolute head: ahead of every later entry for its own Redis.
344
+ def replay(client, entry)
345
+ begin
346
+ replayed = attempt_replay(client, entry.payload)
347
+ rescue StandardError
348
+ # Non-connection failures (OOM, LOADING, READONLY…) must not drop
349
+ # the taken payload — restore it before propagating, or a
350
+ # recovering-but-not-ready Redis silently eats one buffered job per
351
+ # drain tick.
352
+ restore(entry)
353
+ raise
354
+ end
355
+ unless replayed
356
+ restore(entry)
357
+ return
358
+ end
359
+
360
+ client.send(:emit_enqueued, [entry.payload])
361
+ Wurk::Metrics::Statsd.increment('jobs.recovered.push')
362
+ entry
363
+ end
364
+
365
+ def restore(entry)
366
+ buffer_mutex.synchronize { buffer.unshift(entry) }
325
367
  end
326
368
 
327
369
  # Drain marks the thread so our prepended raw_push re-raises the
@@ -338,12 +380,82 @@ module Wurk
338
380
  end
339
381
  end
340
382
 
341
- # Background drain thread. Wakes every `interval` seconds and tries
342
- # `Buffered.drain!` against a fresh Wurk::Client. drain! already
343
- # short-circuits on the first transient failure, so a still-down Redis
344
- # just leaves the buffer alone for this tick — no exponential
345
- # backoff or explicit "reconnect detection" needed; the inner
346
- # connection retry already lives inside `client.raw_push`.
383
+ # The jobs `:drop_oldest` evicts. A drop is a job the caller was told was
384
+ # enqueued and that will now never run, so it must never be silent. One
385
+ # ERROR per burst — the first drop since a replay last reached Redis —
386
+ # because an outage that outlasts the cap drops on every push, and a line
387
+ # per job would bury the one that says what happened. Every drop still
388
+ # counts, as the statsd `jobs.dropped.push` counter.
389
+ #
390
+ # #record runs under the buffer's mutex; #report runs after it is
391
+ # released, so neither the log write nor statsd holds up other pushes.
392
+ class DropLog
393
+ def initialize
394
+ @lock = Mutex.new
395
+ @pending = []
396
+ @burst = false
397
+ end
398
+
399
+ def record(payload)
400
+ @lock.synchronize { @pending << payload }
401
+ end
402
+
403
+ def report(cap)
404
+ dropped, announce = @lock.synchronize do
405
+ next [nil, false] if @pending.empty?
406
+
407
+ taken = @pending
408
+ @pending = []
409
+ first = !@burst
410
+ @burst = true
411
+ [taken, first]
412
+ end
413
+ return unless dropped
414
+
415
+ log(dropped.first, cap) if announce
416
+ dropped.size.times { count_drop }
417
+ end
418
+
419
+ # A replay reached Redis, so the outage that filled the buffer is over:
420
+ # the next overflow is a new incident and gets its own ERROR.
421
+ def end_burst
422
+ @lock.synchronize { @burst = false }
423
+ end
424
+
425
+ private
426
+
427
+ # Runs in Buffered.enbuffer's `ensure`: anything raised here would
428
+ # replace the push's own result (its return, or the Overflow it is
429
+ # raising), so the metric is strictly best-effort.
430
+ def count_drop
431
+ Wurk::Metrics::Statsd.increment('jobs.dropped.push')
432
+ rescue StandardError
433
+ nil
434
+ end
435
+
436
+ def log(first, cap)
437
+ Wurk.configuration.logger.error do
438
+ "reliable_push buffer full (cap=#{cap}): dropping the oldest buffered jobs to make room — " \
439
+ "#{first['class']} jid=#{first['jid']} is the first of this burst and will never run. " \
440
+ 'Further drops are counted in statsd jobs.dropped.push until Redis takes a replay. Raise ' \
441
+ '`Wurk::Client.reliable_push_buffer`, or set `reliable_push_overflow = :raise` to handle ' \
442
+ 'overflow yourself.'
443
+ end
444
+ rescue StandardError
445
+ nil
446
+ end
447
+ end
448
+
449
+ # Eager for the reason the mutexes are: a lazy `||=` could hand two
450
+ # racing first pushes different logs.
451
+ @drops = DropLog.new
452
+
453
+ # Background drain thread. Wakes every `interval` seconds and runs
454
+ # `Buffered.drain_all!`. drain! already short-circuits on the first
455
+ # transient failure, so a still-down Redis just leaves its entries alone
456
+ # for this tick — no exponential backoff or explicit "reconnect
457
+ # detection" needed; the inner connection retry already lives inside
458
+ # `client.raw_push`.
347
459
  class Drainer
348
460
  DEFAULT_INTERVAL = 2.0
349
461
  STOP_JOIN_TIMEOUT = 5.0
@@ -352,13 +464,12 @@ module Wurk
352
464
  # rebuild an equivalent one in the child without touching its lock.
353
465
  attr_reader :interval
354
466
 
355
- def initialize(interval: DEFAULT_INTERVAL, client_factory: -> { Wurk::Client.new })
467
+ def initialize(interval: DEFAULT_INTERVAL)
356
468
  unless interval.is_a?(Numeric) && interval.positive?
357
469
  raise ArgumentError, 'interval must be a positive Numeric'
358
470
  end
359
471
 
360
472
  @interval = interval
361
- @client_factory = client_factory
362
473
  @done = false
363
474
  @thread = nil
364
475
  @wake = ConditionVariable.new
@@ -397,15 +508,25 @@ module Wurk
397
508
  wait_interval
398
509
  break if @done
399
510
 
400
- begin
401
- Buffered.drain!(@client_factory.call)
402
- rescue StandardError
403
- # Swallow — next tick retries. Don't let a transient blow up
404
- # the daemon thread.
405
- end
511
+ tick
406
512
  end
407
513
  end
408
514
 
515
+ # A raise must not end the thread — the next tick retries — but it is
516
+ # reported: a transient outage never gets here (drain! absorbs it), so
517
+ # whatever does is something an operator needs to see.
518
+ def tick
519
+ Buffered.drain_all!
520
+ rescue StandardError => e
521
+ report(e)
522
+ end
523
+
524
+ def report(error)
525
+ Wurk.configuration.handle_exception(error, { context: 'reliable_push drainer' })
526
+ rescue StandardError
527
+ nil
528
+ end
529
+
409
530
  # Mutex+ConditionVariable lets `stop` wake the thread immediately
410
531
  # instead of waiting up to `interval` seconds for sleep to return.
411
532
  def wait_interval
@@ -499,7 +620,7 @@ module Wurk
499
620
  # Activate reliable_push! mode globally. Idempotent — call from the
500
621
  # top level of an initializer (NOT inside Wurk.configure_*). Spec:
501
622
  # docs/target/sidekiq-pro.md §5.
502
- def reliable_push! # rubocop:disable Naming/PredicateMethod
623
+ def reliable_push!
503
624
  Buffered.install!
504
625
  true
505
626
  end
data/lib/wurk/client.rb CHANGED
@@ -50,6 +50,13 @@ module Wurk
50
50
  # pays a single nil check per write phase.
51
51
  DELIVERED_KEY = :wurk_client_delivered
52
52
 
53
+ # Thread-local write-progress marker for a caller that has to know whether
54
+ # a failed push may already be in Redis — {API::Idempotency} opens it at
55
+ # :clean around a produce request. #raw_push moves it to :attempted when a
56
+ # write goes out and to :applied once one lands; nil (no caller watching)
57
+ # costs one lookup per write phase.
58
+ WRITE_STATE_KEY = :wurk_client_write_state
59
+
53
60
  attr_accessor :redis_pool
54
61
 
55
62
  def initialize(pool: nil, config: nil, chain: nil)
@@ -266,10 +273,10 @@ module Wurk
266
273
  Array.new(count) { now + (rand * window) }
267
274
  end
268
275
 
269
- # Inside an autoflush `Batch#jobs` block immediate batched pushes are
270
- # accumulated in the buffer rather than written; it flushes every N jobs
271
- # (when autoflush is an Integer) and Batch#jobs drains the remainder at
272
- # block exit. Scheduled (`at`) or non-batched payloads bypass the buffer.
276
+ # Inside a `Batch#jobs` block immediate batched pushes are accumulated in
277
+ # the buffer rather than written; it flushes every N jobs (when autoflush
278
+ # is an Integer) and Batch#jobs drains the remainder at block exit.
279
+ # Scheduled (`at`) or non-batched payloads bypass the buffer.
273
280
  #
274
281
  # Adds happen one payload at a time so an `autoflush = N` actually bounds
275
282
  # the pipeline size — a bulk push of 100 with N=2 must flush 2/2/... not
@@ -293,16 +300,28 @@ module Wurk
293
300
 
294
301
  # No apply-safety claim: every command below appends (LPUSH, ZADD, the
295
302
  # batch Lua's counters), so a block replayed after a lost reply is a
296
- # second copy of the job. A post-write timeout raises out of here instead
297
- # — {Client::Buffered} turns that into an outage-buffer entry, and a plain
298
- # Client hands it to whoever called `perform_async`. The pool's pre-apply
299
- # retry only fires while this block has landed nothing, so a queue group
300
- # that already went out is never re-pushed by a replay.
301
- pool.with { |conn| atomic_push(conn, payloads) }
303
+ # second copy of the job. The pool's pre-apply retry only fires while this
304
+ # block has landed nothing, so a queue group that already went out is
305
+ # never re-pushed by a block replay. That is the pool's guarantee, not the
306
+ # socket's: redis-client re-sends the one in-flight pipeline once on a
307
+ # dropped connection (RedisPool::DEFAULT_RECONNECT_ATTEMPTS, Sidekiq's
308
+ # setting too), so a reply lost mid-pipeline can still land that group
309
+ # twice. Only an error that outlasts that re-send raises out of here —
310
+ # {Client::Buffered} turns it into an outage-buffer entry, and a plain
311
+ # Client hands it to whoever called `perform_async`.
312
+ tracked_push(payloads)
302
313
  nil
303
314
  end
304
315
 
305
- # Batch autoflush path: accumulate each non-scheduled batched payload into
316
+ # The write itself, moving WRITE_STATE_KEY along for a caller watching it.
317
+ def tracked_push(payloads)
318
+ state = Thread.current[WRITE_STATE_KEY]
319
+ Thread.current[WRITE_STATE_KEY] = :attempted if state == :clean
320
+ pool.with { |conn| atomic_push(conn, payloads) }
321
+ Thread.current[WRITE_STATE_KEY] = :applied if state
322
+ end
323
+
324
+ # Batch#jobs path: accumulate each non-scheduled batched payload into
306
325
  # the active buffer, flushing every N adds (when `buffer.ready?`).
307
326
  def buffer_add(buffer, payloads)
308
327
  payloads.each do |payload|
@@ -379,8 +398,8 @@ module Wurk
379
398
  end
380
399
 
381
400
  # Outside of test boots and `SCRIPT FLUSH` the rescue branch is dead
382
- # code; the eager `script_load_all` after fork keeps the script cache
383
- # hot for the life of the connection. The retry uses EVAL (source-embedded)
401
+ # code; the swarm child's post-fork cache check (Loader.load_missing)
402
+ # keeps the script cache hot for the life of the connection. The retry uses EVAL (source-embedded)
384
403
  # instead of EVALSHA so a freshly-loaded script can't race the retry and
385
404
  # NOSCRIPT a second time under heavy CI load (WorkerTest 3.4/7.2 flake).
386
405
  # `script_load_all` still primes the cache so the *next* pipeline returns
@@ -527,6 +546,7 @@ module Wurk
527
546
  # ledger.
528
547
  def mark_delivered(payloads)
529
548
  Thread.current[DELIVERED_KEY]&.concat(payloads)
549
+ Thread.current[WRITE_STATE_KEY] &&= :applied
530
550
  end
531
551
 
532
552
  # Best-effort `sidekiq.jobs.enqueued` counter — one increment per payload
data/lib/wurk/compat.rb CHANGED
@@ -5,6 +5,8 @@
5
5
  #
6
6
  # Spec: docs/target/sidekiq-{free,pro,ent}.md.
7
7
 
8
+ require_relative 'loader'
9
+
8
10
  # The `Sidekiq::*` compatibility namespace. Every public Wurk class is exposed
9
11
  # here under its Sidekiq name so an existing Sidekiq/Pro/Enterprise app runs on
10
12
  # Wurk after a one-line gem swap. `Sidekiq::Job` / `Sidekiq::Worker` are
@@ -136,10 +138,15 @@ module Sidekiq
136
138
  JobLogger = Wurk::JobLogger
137
139
  JobRecord = Wurk::JobRecord
138
140
  JobRetry = Wurk::JobRetry
141
+ # sidekiq-unique-jobs reopens these with `class Sidekiq::JobSet` to prepend
142
+ # its lock release; without the alias that defines a fresh, unused class and
143
+ # deleted jobs never give their locks back.
144
+ JobSet = Wurk::JobSet
139
145
  JobUtil = Wurk::JobUtil
140
146
  Keys = Wurk::Keys
141
147
  Launcher = Wurk::Launcher
142
148
  Limiter = Wurk::Limiter
149
+ Loader = Wurk::Loader
143
150
  Logger = Wurk::Logger
144
151
  Manager = Wurk::Manager
145
152
  Metrics = Wurk::Metrics
@@ -160,6 +167,7 @@ module Sidekiq
160
167
  ScheduledSet = Wurk::ScheduledSet
161
168
  Shutdown = Wurk::Shutdown
162
169
  SortedEntry = Wurk::SortedEntry
170
+ SortedSet = Wurk::SortedSet
163
171
  Stats = Wurk::Stats
164
172
  # No `Sidekiq::Status` alias on purpose, same reason as `Sidekiq::Cron`
165
173
  # (#204): the sidekiq-status gem owns that namespace and opens it with
@@ -212,5 +220,10 @@ module Sidekiq
212
220
  def testing? = Wurk.testing?
213
221
  def load_json(str) = Wurk.load_json(str)
214
222
  def dump_json(obj) = Wurk.dump_json(obj)
223
+ def loader = Wurk.loader
215
224
  end
216
225
  end
226
+
227
+ # Upstream fires this at the end of sidekiq/api.rb; Wurk's API is fully loaded
228
+ # by the time its aliases are.
229
+ Wurk.loader.run_load_hooks(:api)