wurk 1.7.6 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (217) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +7 -4
  3. data/app/controllers/concerns/wurk/sse_streaming.rb +2 -1
  4. data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +33 -9
  5. data/app/controllers/wurk/api/pagination.rb +25 -7
  6. data/app/controllers/wurk/api_controller.rb +49 -25
  7. data/app/controllers/wurk/extensions_controller.rb +8 -2
  8. data/app/controllers/wurk/profiles_controller.rb +45 -9
  9. data/config/routes.rb +31 -10
  10. data/lib/generators/wurk/install/templates/wurk.rb +1 -1
  11. data/lib/sidekiq/capsule.rb +4 -0
  12. data/lib/sidekiq/component.rb +4 -0
  13. data/lib/sidekiq/config.rb +4 -0
  14. data/lib/sidekiq/deploy.rb +4 -0
  15. data/lib/sidekiq/embedded.rb +4 -0
  16. data/lib/sidekiq/fetch.rb +4 -0
  17. data/lib/sidekiq/iterable_job.rb +4 -0
  18. data/lib/sidekiq/job/interrupt_handler.rb +4 -0
  19. data/lib/sidekiq/job/iterable/active_record_enumerator.rb +4 -0
  20. data/lib/sidekiq/job/iterable/csv_enumerator.rb +4 -0
  21. data/lib/sidekiq/job/iterable.rb +4 -0
  22. data/lib/sidekiq/job_logger.rb +4 -0
  23. data/lib/sidekiq/job_util.rb +4 -0
  24. data/lib/sidekiq/loader.rb +4 -0
  25. data/lib/sidekiq/logger.rb +4 -0
  26. data/lib/sidekiq/metrics/query.rb +4 -0
  27. data/lib/sidekiq/middleware/current_attributes.rb +7 -0
  28. data/lib/sidekiq/middleware/i18n.rb +6 -0
  29. data/lib/sidekiq/middleware/modules.rb +4 -0
  30. data/lib/sidekiq/pro/web.rb +5 -0
  31. data/lib/sidekiq/profiler.rb +4 -0
  32. data/lib/sidekiq/redis_client_adapter.rb +4 -0
  33. data/lib/sidekiq/test_api.rb +5 -0
  34. data/lib/sidekiq/testing/inline.rb +8 -0
  35. data/lib/sidekiq/transaction_aware_client.rb +4 -0
  36. data/lib/sidekiq/worker_compatibility_alias.rb +4 -0
  37. data/lib/sidekiq-ent/periodic/testing.rb +5 -0
  38. data/lib/sidekiq-ent/web.rb +5 -0
  39. data/lib/sidekiq-ent.rb +7 -0
  40. data/lib/sidekiq-pro.rb +7 -0
  41. data/lib/wurk/api/fast.rb +8 -33
  42. data/lib/wurk/api/idempotency.rb +28 -5
  43. data/lib/wurk/api/swarm.rb +1 -1
  44. data/lib/wurk/batch/callbacks.rb +143 -92
  45. data/lib/wurk/batch/death_handler.rb +19 -8
  46. data/lib/wurk/batch/server_middleware.rb +37 -24
  47. data/lib/wurk/batch/status.rb +49 -15
  48. data/lib/wurk/batch.rb +164 -65
  49. data/lib/wurk/capsule.rb +0 -2
  50. data/lib/wurk/client/buffered.rb +227 -106
  51. data/lib/wurk/client.rb +33 -13
  52. data/lib/wurk/compat.rb +13 -0
  53. data/lib/wurk/configuration.rb +155 -24
  54. data/lib/wurk/cpu_count.rb +97 -0
  55. data/lib/wurk/cron.rb +97 -34
  56. data/lib/wurk/dead_set.rb +6 -7
  57. data/lib/wurk/encryption.rb +46 -13
  58. data/lib/wurk/fetcher/many_queues.rb +65 -0
  59. data/lib/wurk/fetcher/orphan_grace.rb +67 -0
  60. data/lib/wurk/fetcher/private_list_key.rb +76 -0
  61. data/lib/wurk/fetcher/reaper.rb +55 -74
  62. data/lib/wurk/fetcher/reliable.rb +23 -17
  63. data/lib/wurk/fetcher/unit_of_work.rb +19 -1
  64. data/lib/wurk/fetcher/unparseable_keys.rb +56 -0
  65. data/lib/wurk/flow/completion.rb +28 -3
  66. data/lib/wurk/flow/creation.rb +24 -4
  67. data/lib/wurk/flow.rb +24 -2
  68. data/lib/wurk/health.rb +182 -40
  69. data/lib/wurk/heartbeat.rb +37 -14
  70. data/lib/wurk/history.rb +7 -1
  71. data/lib/wurk/import/sidekiq_cron.rb +248 -0
  72. data/lib/wurk/job/options.rb +22 -3
  73. data/lib/wurk/job_logger.rb +13 -3
  74. data/lib/wurk/job_record.rb +68 -19
  75. data/lib/wurk/job_retry.rb +12 -6
  76. data/lib/wurk/job_set.rb +99 -51
  77. data/lib/wurk/job_util.rb +27 -8
  78. data/lib/wurk/launcher.rb +84 -40
  79. data/lib/wurk/leader.rb +53 -28
  80. data/lib/wurk/limiter/base.rb +11 -30
  81. data/lib/wurk/limiter/bucket.rb +14 -5
  82. data/lib/wurk/limiter/server_middleware.rb +2 -1
  83. data/lib/wurk/limiter.rb +1 -1
  84. data/lib/wurk/loader.rb +48 -0
  85. data/lib/wurk/lua/batch_ack_complete.lua +28 -0
  86. data/lib/wurk/lua/batch_ack_failed.lua +18 -0
  87. data/lib/wurk/lua/batch_ack_success.lua +35 -0
  88. data/lib/wurk/lua/batch_append_callback.lua +66 -0
  89. data/lib/wurk/lua/batch_delete.lua +31 -0
  90. data/lib/wurk/lua/batch_hold.lua +17 -0
  91. data/lib/wurk/lua/batch_invalidate.lua +14 -0
  92. data/lib/wurk/lua/batch_kid_done.lua +14 -0
  93. data/lib/wurk/lua/batch_push.lua +47 -0
  94. data/lib/wurk/lua/batch_remove_jobs.lua +29 -0
  95. data/lib/wurk/lua/batch_schedule.lua +27 -0
  96. data/lib/wurk/lua/fetch_first.lua +27 -0
  97. data/lib/wurk/lua/flow_abandon.lua +16 -21
  98. data/lib/wurk/lua/flow_advance.lua +41 -26
  99. data/lib/wurk/lua/flow_create.lua +24 -9
  100. data/lib/wurk/lua/flow_fail.lua +3 -3
  101. data/lib/wurk/lua/leader_campaign.lua +19 -0
  102. data/lib/wurk/lua/limiter_bucket_acquire.lua +3 -2
  103. data/lib/wurk/lua/limiter_list_sweep.lua +8 -9
  104. data/lib/wurk/lua/loader.rb +36 -23
  105. data/lib/wurk/lua/throttle_slot.lua +13 -26
  106. data/lib/wurk/lua.rb +31 -244
  107. data/lib/wurk/manager.rb +25 -18
  108. data/lib/wurk/metrics/accumulator.rb +31 -10
  109. data/lib/wurk/metrics/dashboard_series.rb +163 -0
  110. data/lib/wurk/metrics/histogram.rb +46 -0
  111. data/lib/wurk/metrics/history.rb +49 -57
  112. data/lib/wurk/metrics/prometheus.rb +232 -0
  113. data/lib/wurk/metrics/query.rb +127 -155
  114. data/lib/wurk/metrics/rollup.rb +2 -2
  115. data/lib/wurk/metrics/statsd.rb +1 -1
  116. data/lib/wurk/middleware/interrupt_handler.rb +6 -2
  117. data/lib/wurk/middleware/poison_pill.rb +5 -4
  118. data/lib/wurk/middleware/timeout.rb +8 -8
  119. data/lib/wurk/process_set.rb +5 -5
  120. data/lib/wurk/processor.rb +69 -31
  121. data/lib/wurk/profile_set.rb +19 -24
  122. data/lib/wurk/profiler.rb +48 -50
  123. data/lib/wurk/queue.rb +36 -19
  124. data/lib/wurk/rails_boot.rb +33 -1
  125. data/lib/wurk/railtie.rb +2 -0
  126. data/lib/wurk/rake_tasks.rb +18 -0
  127. data/lib/wurk/redact.rb +39 -0
  128. data/lib/wurk/redis_connection.rb +6 -3
  129. data/lib/wurk/redis_options.rb +21 -4
  130. data/lib/wurk/redis_pool.rb +119 -21
  131. data/lib/wurk/scheduled.rb +44 -11
  132. data/lib/wurk/sentry/error_handler.rb +53 -7
  133. data/lib/wurk/sentry/middleware.rb +7 -8
  134. data/lib/wurk/shutdown_gate.rb +24 -1
  135. data/lib/wurk/sorted_entry.rb +16 -5
  136. data/lib/wurk/stats.rb +9 -4
  137. data/lib/wurk/status.rb +2 -2
  138. data/lib/wurk/swarm/child_boot.rb +48 -38
  139. data/lib/wurk/swarm/liveness.rb +153 -0
  140. data/lib/wurk/swarm/restart.rb +24 -3
  141. data/lib/wurk/swarm.rb +115 -24
  142. data/lib/wurk/telemetry/server_middleware.rb +2 -5
  143. data/lib/wurk/testing.rb +15 -5
  144. data/lib/wurk/throttle.rb +16 -1
  145. data/lib/wurk/topology.rb +1 -5
  146. data/lib/wurk/unique.rb +1 -1
  147. data/lib/wurk/version.rb +1 -1
  148. data/lib/wurk/watchdog.rb +2 -2
  149. data/lib/wurk/web/config.rb +56 -20
  150. data/lib/wurk/web/enterprise.rb +21 -18
  151. data/lib/wurk/web/extension.rb +98 -22
  152. data/lib/wurk/web/rack_app.rb +12 -11
  153. data/lib/wurk/web/search.rb +18 -8
  154. data/lib/wurk/worker/setter.rb +17 -14
  155. data/lib/wurk/worker.rb +6 -67
  156. data/lib/wurk.rb +2 -3
  157. data/vendor/assets/dashboard/assets/ArgsValue-DsPb_KZ-.js +1 -0
  158. data/vendor/assets/dashboard/assets/BatchDetail-BCIXJOpZ.js +1 -0
  159. data/vendor/assets/dashboard/assets/Batches-IjaQfXW6.js +1 -0
  160. data/vendor/assets/dashboard/assets/Busy-CaQrf1Ve.js +1 -0
  161. data/vendor/assets/dashboard/assets/Cron-B18EHn_Y.js +1 -0
  162. data/vendor/assets/dashboard/assets/Dashboard-DdT5ZaUg.js +1 -0
  163. data/vendor/assets/dashboard/assets/Dead-DWIy-Mu8.js +1 -0
  164. data/vendor/assets/dashboard/assets/Extension-BjBUWhHf.js +1 -0
  165. data/vendor/assets/dashboard/assets/FilterBox-DwH3PXXt.js +1 -0
  166. data/vendor/assets/dashboard/assets/FlowDetail-DnexL72n.js +1 -0
  167. data/vendor/assets/dashboard/assets/FlowState-Cr4aX6-d.js +1 -0
  168. data/vendor/assets/dashboard/assets/Flows-DO6EimbZ.js +1 -0
  169. data/vendor/assets/dashboard/assets/JobDetailModal-aPjgq-U9.js +2 -0
  170. data/vendor/assets/dashboard/assets/Limiters-DxpwWCNe.js +1 -0
  171. data/vendor/assets/dashboard/assets/Metrics-C3JL1GIH.js +1 -0
  172. data/vendor/assets/dashboard/assets/NotFound-BfP9uAV9.js +1 -0
  173. data/vendor/assets/dashboard/assets/PageHeader-DbC97afG.js +1 -0
  174. data/vendor/assets/dashboard/assets/Profiles-BcaHreMe.js +1 -0
  175. data/vendor/assets/dashboard/assets/Queues-ZlBc5tyD.js +1 -0
  176. data/vendor/assets/dashboard/assets/Retries-Bg3nkeiL.js +1 -0
  177. data/vendor/assets/dashboard/assets/Scheduled-BrMZPZGl.js +1 -0
  178. data/vendor/assets/dashboard/assets/Search-B5X5pGht.js +1 -0
  179. data/vendor/assets/dashboard/assets/charts-uDyf4PN0.js +1 -0
  180. data/vendor/assets/dashboard/assets/i18n-CJeYfeVt.js +1 -0
  181. data/vendor/assets/dashboard/assets/index-B36QCAf-.css +1 -0
  182. data/vendor/assets/dashboard/assets/index-DQujgZgH.js +141 -0
  183. data/vendor/assets/dashboard/assets/useResetPageOnEmpty-O3-5rMk3.js +1 -0
  184. data/vendor/assets/dashboard/assets/useSort-xfnwjQAu.js +1 -0
  185. data/vendor/assets/dashboard/assets/utils-BIKHwNKB.js +1 -0
  186. data/vendor/assets/dashboard/index.html +3 -3
  187. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  188. metadata +90 -34
  189. data/config/locales/en.yml +0 -15
  190. data/vendor/assets/dashboard/assets/ArgsValue-D-x_ifLY.js +0 -1
  191. data/vendor/assets/dashboard/assets/BatchDetail-C39NJuew.js +0 -1
  192. data/vendor/assets/dashboard/assets/Batches-CSwo7Asa.js +0 -1
  193. data/vendor/assets/dashboard/assets/Busy-BOFMu-sq.js +0 -1
  194. data/vendor/assets/dashboard/assets/Cron-Dy8RQzDI.js +0 -1
  195. data/vendor/assets/dashboard/assets/Dashboard-BuTHI-O1.js +0 -1
  196. data/vendor/assets/dashboard/assets/Dead-B9KRvQ0N.js +0 -1
  197. data/vendor/assets/dashboard/assets/Extension-BnBVHfux.js +0 -1
  198. data/vendor/assets/dashboard/assets/FilterBox-DC24zite.js +0 -1
  199. data/vendor/assets/dashboard/assets/FlowDetail-DyLuzUvt.js +0 -1
  200. data/vendor/assets/dashboard/assets/FlowState-DAPKUahm.js +0 -1
  201. data/vendor/assets/dashboard/assets/Flows-Fr3rjZM_.js +0 -1
  202. data/vendor/assets/dashboard/assets/JobDetailModal-N6kiJXq3.js +0 -2
  203. data/vendor/assets/dashboard/assets/Limiters-kbFA7uS1.js +0 -1
  204. data/vendor/assets/dashboard/assets/Metrics-Dj2uoZ3o.js +0 -1
  205. data/vendor/assets/dashboard/assets/PageHeader-B_F94azl.js +0 -1
  206. data/vendor/assets/dashboard/assets/Profiles-D_DjEezN.js +0 -1
  207. data/vendor/assets/dashboard/assets/Queues-CO4V9hAz.js +0 -1
  208. data/vendor/assets/dashboard/assets/Retries-DCWnzeLa.js +0 -1
  209. data/vendor/assets/dashboard/assets/Scheduled-BebDUjLU.js +0 -1
  210. data/vendor/assets/dashboard/assets/Search-Cvr5fy4Y.js +0 -1
  211. data/vendor/assets/dashboard/assets/Skeleton-Bu3Ke6rV.js +0 -1
  212. data/vendor/assets/dashboard/assets/charts-BCs9bQKz.js +0 -1
  213. data/vendor/assets/dashboard/assets/index-BIwyOC5Q.js +0 -141
  214. data/vendor/assets/dashboard/assets/index-DBQN6Jk8.css +0 -1
  215. data/vendor/assets/dashboard/assets/useResetPageOnEmpty-Bzh-BJyL.js +0 -1
  216. data/vendor/assets/dashboard/assets/useSort-COA3fVJ5.js +0 -1
  217. data/vendor/assets/dashboard/assets/utils-BIrvZ1hi.js +0 -1
@@ -10,32 +10,43 @@ module Wurk
10
10
  # carries a `bid`, we BATCH_ACK_COMPLETE → record the death → fire
11
11
  # `:death` callback exactly once per batch (first death only).
12
12
  #
13
+ # The retry layer runs this after the job is already in the morgue and
14
+ # acked off its private list, and swallows whatever it raises — nothing
15
+ # would ever deliver this death to the batch again. So the whole pass is
16
+ # re-driven in-process (`Callbacks.retrying`): the ack script is
17
+ # SREM/SADD-guarded and the fires are marker-guarded, so a replay is safe.
18
+ # A replay can't trust `first_death` (the lost attempt may already have
19
+ # moved the jid into the died set), so it falls back to the `:death`
20
+ # claim marker to decide whether `:death` still has to fire.
21
+ #
13
22
  # Spec: docs/target/sidekiq-pro.md §2.4 (`:death`).
14
23
  class DeathHandler
15
24
  def self.call(job, _exception)
16
25
  bid = job['bid']
17
26
  return unless bid
18
27
 
19
- result = Wurk.redis do |conn|
28
+ Callbacks.retrying { |attempt| record(bid, job['jid'], replay: attempt.positive?) }
29
+ end
30
+
31
+ def self.record(bid, jid, replay:)
32
+ live, _died, first_death, kids, pending = Wurk.redis do |conn|
20
33
  Wurk::Lua::Loader.eval_cached(
21
34
  conn,
22
35
  :batch_ack_complete,
23
- keys: ["b-#{bid}", "b-#{bid}-jids", "b-#{bid}-died", "b-#{bid}-failed"],
24
- argv: [job['jid'], Batch::DEFAULT_EXPIRY_SECONDS]
36
+ keys: ["b-#{bid}", "b-#{bid}-jids", "b-#{bid}-died", "b-#{bid}-failed", "b-#{bid}-pkids"],
37
+ argv: [jid, Batch::DEFAULT_EXPIRY_SECONDS]
25
38
  )
26
- end
27
- live, _died, first_death = Array(result).map(&:to_i)
39
+ end.map(&:to_i)
28
40
 
29
41
  restamp_ttls(bid)
30
42
 
31
- Wurk::Batch::Callbacks.fire_death(bid) if first_death == 1
32
- return unless live.zero?
43
+ Callbacks.fire_death(bid) if first_death == 1 || (replay && !Callbacks.dedup_marked?(bid, 'death'))
33
44
 
34
45
  # Through the gated maybe_fire, not a direct fire_complete: this batch
35
46
  # may still have running child batches, and spec §2.4 ordering says
36
47
  # its `:complete` must wait for theirs (#209). `:success` stays
37
48
  # suppressed regardless — the death above set the durable death flag.
38
- Wurk::Batch::Callbacks.maybe_fire(bid, pending: Wurk::Batch::Callbacks.pending_for(bid), live: 0)
49
+ Callbacks.maybe_fire(bid, pending: pending, live: live, kids: kids)
39
50
  end
40
51
 
41
52
  # BATCH_ACK_COMPLETE stamps the two keys it can itself resurrect; this
@@ -1,17 +1,16 @@
1
1
  # frozen_string_literal: true
2
2
 
3
- require 'json'
4
3
  require_relative '../middleware'
5
4
  require_relative '../lua'
6
5
  require_relative '../job'
7
6
  require_relative '../job_retry'
7
+ require_relative 'callbacks'
8
8
 
9
9
  module Wurk
10
10
  class Batch
11
11
  # Server middleware. Runs around `perform` for any job carrying a `bid`.
12
- # On success → BATCH_ACK_SUCCESS → if pending hit zero and no deaths,
13
- # enqueue `:success` callback jobs; if live jids hit zero, enqueue
14
- # `:complete` callback jobs.
12
+ # On success → BATCH_ACK_SUCCESS → when that drains the batch, fire its
13
+ # callbacks (Callbacks.maybe_fire).
15
14
  #
16
15
  # On a job raising (and thus heading to retry), records a transient
17
16
  # failure → BATCH_ACK_FAILED → the jid joins `b-<bid>-failed` and
@@ -21,8 +20,18 @@ module Wurk
21
20
  # Limiter::Rescheduled re-enqueue — and cooperative IterableJob
22
21
  # interruption) are re-raised as neither success nor failure.
23
22
  #
23
+ # The success ack runs outside the job's rescue: once `perform` returned,
24
+ # nothing the batch bookkeeping raises may be mistaken for the job failing,
25
+ # or a job that already did its work is retried and runs twice. The ack
26
+ # itself is retried in-process and only raises (sending the job to retry,
27
+ # at-least-once) when Redis stays unreachable — a jid left in the live set
28
+ # would hold the batch open forever. The fire after it never raises: it is
29
+ # re-driven in-process and then reported to the error handlers.
30
+ #
24
31
  # Invalidated batches short-circuit: the job is skipped without
25
32
  # raising — counts as a "success" for batch purposes per spec §12.
33
+ # A callback job is never skipped: it rides in its *parent* batch (see
34
+ # Callbacks), and a cancelled parent must not swallow a child's callback.
26
35
  #
27
36
  # Death handling lives in Wurk::Batch::DeathHandler (registered as a
28
37
  # config death_handler) because death is signalled from the retry layer,
@@ -30,16 +39,15 @@ module Wurk
30
39
  class ServerMiddleware
31
40
  include Wurk::Middleware::ServerMiddleware
32
41
 
42
+ CALLBACK_JOB = 'Wurk::Batch::CallbackJob'
43
+
33
44
  def call(_worker, job, _queue, &)
34
45
  bid = job['bid']
35
46
  return yield unless bid
36
47
 
37
- if invalidated?(bid)
38
- ack_success(bid, job['jid'])
39
- return
40
- end
41
-
42
- run_and_ack(bid, job['jid'], &)
48
+ jid = job['jid']
49
+ run(bid, jid, &) if job['class'] == CALLBACK_JOB || !invalidated?(bid)
50
+ ack_success(bid, jid)
43
51
  end
44
52
 
45
53
  private
@@ -50,9 +58,8 @@ module Wurk
50
58
  # failure: re-raise it untouched, acking neither. Any other exception
51
59
  # means the job failed and will retry (or eventually die): record it
52
60
  # before re-raising.
53
- def run_and_ack(bid, jid)
61
+ def run(bid, jid)
54
62
  yield
55
- ack_success(bid, jid)
56
63
  rescue Wurk::JobRetry::Handled, Wurk::Job::Interrupted
57
64
  raise
58
65
  rescue StandardError
@@ -64,19 +71,27 @@ module Wurk
64
71
  redis_pool.with { |conn| conn.call('HGET', "b-#{bid}", 'invalidated') } == '1'
65
72
  end
66
73
 
74
+ # A jid that was no longer live (`removed == 0`) is a re-run — most
75
+ # often a job reclaimed after a SIGKILL that landed between this ack and
76
+ # the fire. It still goes through the gate with the batch's current
77
+ # state: if the batch is drained, the dead run's fire may never have
78
+ # happened, and the callback markers absorb it if it did.
67
79
  def ack_success(bid, jid)
68
- result = redis_pool.with do |conn|
69
- Wurk::Lua::Loader.eval_cached(
70
- conn,
71
- :batch_ack_success,
72
- keys: ["b-#{bid}", "b-#{bid}-jids", "b-#{bid}-failed"],
73
- argv: [jid]
74
- )
80
+ _removed, pending, live, kids = Callbacks.retrying do
81
+ redis_pool.with { |conn| Batch.ack_success(conn, bid, jid) }
75
82
  end
76
- pending, live = Array(result).map(&:to_i)
77
- return if pending.negative?
83
+ fire(bid, pending, live, kids)
84
+ end
78
85
 
79
- Wurk::Batch::Callbacks.maybe_fire(bid, pending: pending, live: live)
86
+ def fire(bid, pending, live, kids)
87
+ Callbacks.retrying { Callbacks.maybe_fire(bid, pending: pending, live: live, kids: kids) }
88
+ rescue StandardError => e
89
+ error_config.handle_exception(e, { context: "batch #{bid}: firing callbacks after ack", bid: bid })
90
+ end
91
+
92
+ # `config` is a Configuration or a Capsule; only the former reports.
93
+ def error_config
94
+ config.respond_to?(:handle_exception) ? config : config.config
80
95
  end
81
96
 
82
97
  def ack_failed(bid, jid)
@@ -93,6 +108,4 @@ module Wurk
93
108
  end
94
109
  end
95
110
 
96
- require_relative 'callbacks'
97
-
98
111
  Wurk.configuration.server_middleware.add(Wurk::Batch::ServerMiddleware)
@@ -1,6 +1,8 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require 'json'
4
+ require_relative '../lua'
5
+ require_relative 'callbacks'
4
6
 
5
7
  module Wurk
6
8
  class Batch
@@ -10,7 +12,7 @@ module Wurk
10
12
  #
11
13
  # `#data` returns the JSON-friendly hash served by the polling endpoint.
12
14
  # `#join` blocks the current thread until `complete?` — test/util only.
13
- # `#delete` UNLINKs every key associated with the batch.
15
+ # `#delete` removes the batch and every reference to it.
14
16
  class Status
15
17
  JOIN_POLL_INTERVAL = 0.5
16
18
 
@@ -47,7 +49,7 @@ module Wurk
47
49
  def complete?
48
50
  return true if @data['complete'] == '1'
49
51
 
50
- total.positive? && live_jids_count.zero?
52
+ complete_with?(live_jids_count)
51
53
  end
52
54
 
53
55
  def failed_jids
@@ -81,8 +83,10 @@ module Wurk
81
83
  end
82
84
 
83
85
  # JSON-serializable snapshot used by the polling middleware / web UI.
84
- # Field names are wire-compat with Sidekiq Pro's BatchStatus.
86
+ # Field names are wire-compat with Sidekiq Pro's BatchStatus. The four
87
+ # set reads it needs ride one pipeline.
85
88
  def data
89
+ failed, dead, kids, live = set_snapshot
86
90
  {
87
91
  'bid' => @bid,
88
92
  'total' => total,
@@ -92,14 +96,14 @@ module Wurk
92
96
  'complete_at' => complete_at,
93
97
  'success_at' => success_at,
94
98
  'death_at' => death_at,
95
- 'complete' => complete?,
99
+ 'complete' => complete_with?(live.to_i),
96
100
  'invalidated' => invalidated?,
97
101
  'description' => description,
98
102
  'parent_bid' => parent_bid,
99
103
  'tags' => tags,
100
- 'failed_jids' => failed_jids,
101
- 'dead_jids' => dead_jids,
102
- 'child_count' => child_count
104
+ 'failed_jids' => failed,
105
+ 'dead_jids' => dead,
106
+ 'child_count' => kids.to_i
103
107
  }
104
108
  end
105
109
 
@@ -115,15 +119,22 @@ module Wurk
115
119
  end
116
120
  end
117
121
 
118
- # Nukes every key for this batch. Dangerous if jobs are still in flight
119
- # — they'll succeed/fail without a batch to ack against, callbacks won't
120
- # fire, and counts get permanently inconsistent. Caller's problem.
122
+ # Nukes every key for this batch, its index entries and tag indexes,
123
+ # and its membership in the parent's `-kids`/`-pkids` — one atomic
124
+ # script (BATCH_DELETE). Dangerous if jobs are still in flight: they'll
125
+ # succeed/fail without a batch to ack against and this batch's callbacks
126
+ # won't fire. The *parent* is not left waiting on a child that no longer
127
+ # exists: when this was the last thing it waited on, its callbacks fire
128
+ # here.
121
129
  def delete
122
- Wurk.redis do |conn|
123
- conn.call('UNLINK', *Batch.keys_for(@bid))
124
- conn.call('ZREM', 'batches', @bid)
125
- conn.call('ZREM', 'dead-batches', @bid)
126
- end
130
+ reload!
131
+ parent = parent_bid.to_s
132
+ own = Batch.keys_for(@bid)
133
+ pending, live, kids = Wurk.redis do |conn|
134
+ Wurk::Lua::Loader.eval_cached(conn, :batch_delete, keys: delete_keys(own, parent),
135
+ argv: [@bid, own.size, parent.empty? ? '0' : '1'])
136
+ end.map(&:to_i)
137
+ Callbacks.maybe_fire(parent, pending: pending, live: live, kids: kids) unless parent.empty?
127
138
  nil
128
139
  end
129
140
 
@@ -135,6 +146,29 @@ module Wurk
135
146
 
136
147
  private
137
148
 
149
+ # [failed jids, dead jids, child count, live count] in one round trip.
150
+ def set_snapshot
151
+ Wurk.redis do |conn|
152
+ conn.pipelined do |pipe|
153
+ pipe.call('SMEMBERS', "b-#{@bid}-failed")
154
+ pipe.call('SMEMBERS', "b-#{@bid}-died")
155
+ pipe.call('SCARD', "b-#{@bid}-kids")
156
+ pipe.call('SCARD', "b-#{@bid}-jids")
157
+ end
158
+ end
159
+ end
160
+
161
+ # KEYS layout BATCH_DELETE expects.
162
+ def delete_keys(own, parent)
163
+ keys = [*own, 'batches', 'dead-batches']
164
+ keys.push("b-#{parent}", "b-#{parent}-jids", "b-#{parent}-pkids", "b-#{parent}-kids") unless parent.empty?
165
+ keys.concat(tags.map { |t| "tags:#{t}" })
166
+ end
167
+
168
+ def complete_with?(live)
169
+ @data['complete'] == '1' || (total.positive? && live.zero?)
170
+ end
171
+
138
172
  def live_jids_count
139
173
  Wurk.redis { |conn| conn.call('SCARD', "b-#{@bid}-jids") }.to_i
140
174
  end
data/lib/wurk/batch.rb CHANGED
@@ -4,6 +4,7 @@ require 'json'
4
4
  require 'securerandom'
5
5
  require_relative 'lua'
6
6
  require_relative 'batch/buffer'
7
+ require_relative 'batch/callbacks'
7
8
 
8
9
  module Wurk
9
10
  # Sidekiq Pro Batches. Group jobs, attach success/complete/death callbacks,
@@ -26,12 +27,14 @@ module Wurk
26
27
  # or callback only). `mutable?` is false because reopening implies the
27
28
  # first flush already happened.
28
29
  # 3. `#jobs { ... }` collects `Job.perform_async` calls via the client
29
- # middleware (Thread.current[:wurk_current_batch] is the signal),
30
- # atomically registering each via BATCH_PUSH.
30
+ # middleware (Thread.current[:wurk_current_batch] is the signal) and
31
+ # pushes them at block exit, each registered via BATCH_PUSH. A hold
32
+ # keeps the batch from draining while the block is open.
31
33
  # 4. Workers ack on success → BATCH_ACK_SUCCESS → pending--.
32
34
  # Death handler acks on permanent failure → BATCH_ACK_COMPLETE.
33
- # 5. When live jids hits zero → fire `:complete`. When pending also
34
- # hits zero with zero deaths → fire `:success`.
35
+ # 5. When live jids and pending child batches both hit zero → fire
36
+ # `:complete`. When pending also hits zero with zero deaths → fire
37
+ # `:success`.
35
38
  #
36
39
  # Nested batches: a job opening its OWN batch (`batch.jobs { ... }`)
37
40
  # increments live counters on the existing batch. A callback opening its
@@ -82,11 +85,22 @@ module Wurk
82
85
 
83
86
  THREAD_KEY = :wurk_current_batch
84
87
 
88
+ EMPTY_JOB = 'Sidekiq::Batch::Empty'
89
+
85
90
  # Set on the current thread (to a Buffer) only inside an autoflush
86
91
  # `#jobs` block. Client#raw_push reads it: when present, batched pushes
87
92
  # accumulate here instead of round-tripping per job.
88
93
  BUFFER_KEY = :wurk_batch_buffer
89
94
 
95
+ # Bids whose `#jobs` hold this thread already owns, so a block nested in
96
+ # another block of the same batch neither takes a second hold nor drops
97
+ # the outer one when it exits.
98
+ HOLDS_KEY = :wurk_batch_holds
99
+
100
+ # Prefix of the hold sentinel's member in `b-<bid>-jids`. Real jids are
101
+ # hex, so a sentinel can never collide with one.
102
+ HOLD_PREFIX = 'hold:'
103
+
90
104
  attr_reader :bid, :parent_bid, :linger, :callback_class
91
105
  attr_accessor :description, :callback_queue, :autoflush
92
106
 
@@ -95,6 +109,29 @@ module Wurk
95
109
  [base, *KEY_SUFFIXES.map { |s| "#{base}-#{s}" }]
96
110
  end
97
111
 
112
+ # Runs the block with `batch` as the thread's active batch (the client
113
+ # middleware stamps its bid) and `buffer` collecting batched pushes —
114
+ # either may be nil — restoring whatever an enclosing block had set.
115
+ def self.with_thread_batch(batch, buffer)
116
+ previous = Thread.current[THREAD_KEY]
117
+ prev_buffer = Thread.current[BUFFER_KEY]
118
+ Thread.current[THREAD_KEY] = batch
119
+ Thread.current[BUFFER_KEY] = buffer
120
+ yield
121
+ ensure
122
+ Thread.current[THREAD_KEY] = previous
123
+ Thread.current[BUFFER_KEY] = prev_buffer
124
+ end
125
+
126
+ # BATCH_ACK_SUCCESS for `jid`, as Integers: [removed, pending, live, kids].
127
+ # The last three are the fire-gate inputs `Callbacks.maybe_fire` takes.
128
+ def self.ack_success(conn, bid, jid)
129
+ Wurk::Lua::Loader.eval_cached(
130
+ conn, :batch_ack_success,
131
+ keys: ["b-#{bid}", "b-#{bid}-jids", "b-#{bid}-failed", "b-#{bid}-pkids"], argv: [jid]
132
+ ).map(&:to_i)
133
+ end
134
+
98
135
  # Two-axis trim of a batch index ZSET (`batches`, `dead-batches`), in the
99
136
  # shape of the morgue trim (`DeadSet#trim`): `ZREMRANGEBYSCORE` evicts
100
137
  # entries older than `timeout`, `ZREMRANGEBYRANK 0 -max` caps the member
@@ -199,22 +236,25 @@ module Wurk
199
236
  end
200
237
 
201
238
  # Remove jobs from the batch. Decrements pending/total by exactly the
202
- # count of jids actually removed (idempotent for repeated calls).
239
+ # count of jids actually removed (idempotent for repeated calls), in one
240
+ # atomic script. Removing the last live jid drains the batch like an ack
241
+ # would, so the callbacks fire here rather than never.
203
242
  def remove_jobs(*jids)
204
243
  return 0 if jids.empty?
205
244
 
206
- Wurk.redis do |conn|
207
- removed = conn.call('SREM', "b-#{@bid}-jids", *jids).to_i
208
- if removed.positive?
209
- conn.call('HINCRBY', "b-#{@bid}", 'pending', -removed)
210
- conn.call('HINCRBY', "b-#{@bid}", 'total', -removed)
211
- end
212
- removed
213
- end
245
+ removed, pending, live, kids = Wurk.redis do |conn|
246
+ Wurk::Lua::Loader.eval_cached(
247
+ conn, :batch_remove_jobs,
248
+ keys: ["b-#{@bid}", "b-#{@bid}-jids", "b-#{@bid}-failed", "b-#{@bid}-pkids"], argv: jids
249
+ )
250
+ end.map(&:to_i)
251
+ Callbacks.maybe_fire(@bid, pending: pending, live: live, kids: kids) if removed.positive?
252
+ removed
214
253
  end
215
254
 
216
255
  # Mark batch invalid. Pending jobs still exist in their queues; the
217
- # server middleware short-circuits them when it observes the flag.
256
+ # server middleware short-circuits them when it observes the flag and acks
257
+ # them as successes (spec §12), so the batch still drains and fires.
218
258
  # Cascades to descendant batches via b-<bid>-kids.
219
259
  def invalidate_all
220
260
  cascade_invalidate(@bid)
@@ -253,23 +293,35 @@ module Wurk
253
293
  self
254
294
  end
255
295
 
256
- # Atomic enqueue block. Inside the block, `Job.perform_async` finds
257
- # this batch via Thread.current[THREAD_KEY] and stamps `bid` onto the
258
- # payload — the client middleware then uses BATCH_PUSH to register and
259
- # push atomically. Empty blocks synthesise a Batch::Empty no-op so
260
- # callbacks still fire.
296
+ # Atomic enqueue block (spec §2.3). Inside the block, `Job.perform_async`
297
+ # finds this batch via Thread.current[THREAD_KEY] and stamps `bid` onto the
298
+ # payload; the pushes are collected and flushed at block exit, each
299
+ # through BATCH_PUSH. A block that raises pushes nothing it collected.
300
+ # `autoflush = N` flushes every N jobs instead, trading that atomicity for
301
+ # bounded memory. A block that creates the batch and adds nothing to it —
302
+ # no job, no child batch — synthesises a Batch::Empty no-op (spec §2.3);
303
+ # a block re-entering an existing batch needs none, the batch already has
304
+ # members (or drains when the hold is released).
305
+ #
306
+ # For the whole block the batch carries a hold (BATCH_HOLD), so a job
307
+ # pushed early — a flushed autoflush slice, a scheduled job, a nested
308
+ # child batch — cannot ack the batch empty and fire its callbacks before
309
+ # the rest is in. The hold is released at block exit through the same ack
310
+ # path as a job. When the block that *created* the batch raises, the hold
311
+ # stays: nothing the block collected was pushed, and a batch that fired
312
+ # for whatever slipped out early would be the partial batch §2.3 rules
313
+ # out. A block re-entering an existing batch releases it either way, or
314
+ # the batch's own jobs could never fire it.
261
315
  def jobs(&block)
262
316
  raise ArgumentError, 'jobs requires a block' unless block
263
317
 
318
+ threshold = autoflush_threshold
319
+ created = !@flushed_once
264
320
  ensure_first_flush!
265
- pre_count = job_count
266
- collect_jobs(&block)
267
- # By the time we check, the buffer (if any) has flushed, so `total`
268
- # reflects everything the block pushed — a flat count is reliable.
269
- # Scheduled (`perform_in`) jobs count here too: BATCH_SCHEDULE moves
270
- # `total` at creation, so a scheduled-only block does not misfire the
271
- # marker as if it were empty.
272
- enqueue_empty_marker if job_count == pre_count
321
+ with_hold(release_on_error: !created) do
322
+ collect_jobs(threshold, &block)
323
+ enqueue_empty_marker if created && untouched?
324
+ end
273
325
  @mutable = false
274
326
  self
275
327
  end
@@ -277,41 +329,70 @@ module Wurk
277
329
  private
278
330
 
279
331
  # Runs the block with this batch active so the client middleware stamps
280
- # the bid. With autoflush on, batched pushes accumulate in a Buffer and
281
- # flush once at exit; the per-N flushing happens in Client#raw_push.
282
- def collect_jobs
283
- previous = Thread.current[THREAD_KEY]
284
- prev_buffer = Thread.current[BUFFER_KEY]
285
- buffer = new_buffer
286
- Thread.current[THREAD_KEY] = self
287
- Thread.current[BUFFER_KEY] = buffer
332
+ # the bid. Batched pushes accumulate in a Buffer and flush once at exit
333
+ # (only on a normal exit); the per-N flushing happens in Client#raw_push.
334
+ def collect_jobs(threshold)
335
+ buffer = Buffer.new([], threshold)
336
+ Batch.with_thread_batch(self, buffer) do
337
+ yield
338
+ flush_buffer(buffer)
339
+ end
340
+ end
341
+
342
+ # Unset (or `true`/`false`) → buffer the whole block (nil threshold,
343
+ # drained at exit); positive Integer → flush every N. Any other value is
344
+ # a config typo (`0`, `-1`, `"5"`) — fail fast instead of silently
345
+ # degrading to "flush at block exit". Checked before anything touches
346
+ # Redis, so a typo never leaves a batch created and held.
347
+ def autoflush_threshold
348
+ return nil if @autoflush.nil? || @autoflush == true || @autoflush == false
349
+ return @autoflush if @autoflush.is_a?(Integer) && @autoflush.positive?
350
+
351
+ raise ArgumentError, "autoflush must be true or a positive Integer, got #{@autoflush.inspect}"
352
+ end
353
+
354
+ def with_hold(release_on_error:)
355
+ holds = (Thread.current[HOLDS_KEY] ||= {})
356
+ outermost = !holds.key?(@bid)
357
+ sentinel = take_hold if outermost
358
+ holds[@bid] = true if outermost
359
+ completed = false
288
360
  begin
289
361
  yield
290
- flush_buffer(buffer) if buffer
362
+ completed = true
291
363
  ensure
292
- Thread.current[THREAD_KEY] = previous
293
- Thread.current[BUFFER_KEY] = prev_buffer
364
+ holds.delete(@bid) if outermost
365
+ release_hold(sentinel) if sentinel && (completed || release_on_error)
294
366
  end
295
367
  end
296
368
 
297
- # Buffering is opt-in via `autoflush`: `true` buffers the whole block and
298
- # flushes once at exit; a positive Integer flushes every N jobs. Anything
299
- # falsy keeps the default per-job immediate push.
300
- def new_buffer
301
- return nil unless @autoflush
302
-
303
- Buffer.new([], autoflush_threshold)
369
+ # Named after the running job when there is one, so a job reclaimed after
370
+ # a SIGKILL mid-block re-takes the *same* hold (BATCH_HOLD's SADD guard
371
+ # makes that a no-op) and its release clears the one the dead run left.
372
+ def take_hold
373
+ sentinel = "#{HOLD_PREFIX}#{Wurk::Context.current[:jid] || SecureRandom.hex(12)}"
374
+ Wurk.redis do |conn|
375
+ Wurk::Lua::Loader.eval_cached(conn, :batch_hold,
376
+ keys: ["b-#{@bid}", "b-#{@bid}-jids"], argv: [sentinel, @expires_in])
377
+ end
378
+ sentinel
304
379
  end
305
380
 
306
- # `true` → buffer the whole block (nil threshold, drained at exit);
307
- # positive Integer → flush every N. Any other truthy value is a config
308
- # typo (`0`, `-1`, `"5"`) — fail fast instead of silently degrading to
309
- # "flush at block exit".
310
- def autoflush_threshold
311
- return nil if @autoflush == true
312
- return @autoflush if @autoflush.is_a?(Integer) && @autoflush.positive?
381
+ # The ack is retried and raises if Redis stays down — a hold left behind
382
+ # means the batch never fires, which the caller has to hear about. The
383
+ # fire after it is reported, not raised: every job is already pushed, and
384
+ # a caller retrying the block would push them all twice.
385
+ def release_hold(sentinel)
386
+ _removed, pending, live, kids = Callbacks.retrying do
387
+ Wurk.redis { |conn| Batch.ack_success(conn, @bid, sentinel) }
388
+ end
389
+ fire_after_release(pending, live, kids)
390
+ end
313
391
 
314
- raise ArgumentError, "autoflush must be true or a positive Integer, got #{@autoflush.inspect}"
392
+ def fire_after_release(pending, live, kids)
393
+ Callbacks.retrying { Callbacks.maybe_fire(@bid, pending: pending, live: live, kids: kids) }
394
+ rescue StandardError => e
395
+ Wurk.configuration.handle_exception(e, { context: "batch #{@bid}: firing callbacks at #jobs exit", bid: @bid })
315
396
  end
316
397
 
317
398
  def flush_buffer(buffer)
@@ -384,16 +465,25 @@ module Wurk
384
465
  # Only `b-#{@bid}` is stamped here — none of the sub-keys exist yet at first
385
466
  # flush (BATCH_PUSH/BATCH_SCHEDULE create `-jids`, the acks create
386
467
  # `-failed`/`-died`), and EXPIRE on a missing key is a no-op. Each key is
387
- # stamped `NX` where it is created instead; see BATCH_PUSH in lua.rb.
468
+ # stamped `NX` where it is created instead; see lua/batch_push.lua.
388
469
  def pipelined_first_flush(pipe, now)
389
470
  pipe.call('HSET', "b-#{@bid}", *first_flush_hash(now).flatten)
390
471
  pipe.call('EXPIRE', "b-#{@bid}", @expires_in)
391
472
  pipe.call('ZADD', 'batches', now.to_s, @bid)
392
473
  Batch.trim_index(pipe, 'batches')
393
- @tags.each { |t| pipe.call('SADD', "tags:#{t}", @bid) }
474
+ @tags.each { |t| index_tag(pipe, "tags:#{t}") }
394
475
  link_to_parent(pipe) if current_parent_bid
395
476
  end
396
477
 
478
+ # The reverse index is shared by every batch carrying the tag, so it must
479
+ # outlive the longest-lived of them: NX gives a fresh (or legacy TTL-less)
480
+ # set a clock, GT only ever extends it. `Status#delete` SREMs the bid.
481
+ def index_tag(pipe, key)
482
+ pipe.call('SADD', key, @bid)
483
+ pipe.call('EXPIRE', key, @expires_in, 'NX')
484
+ pipe.call('EXPIRE', key, @expires_in, 'GT')
485
+ end
486
+
397
487
  def first_flush_hash(now)
398
488
  {
399
489
  'created_at' => now.to_s,
@@ -432,24 +522,33 @@ module Wurk
432
522
  pipe.call('EXPIRE', "#{parent_key}-pkids", DEFAULT_EXPIRY_SECONDS, 'NX')
433
523
  end
434
524
 
435
- def job_count
436
- Wurk.redis { |conn| conn.call('HGET', "b-#{@bid}", 'total') }.to_i
525
+ # Read after the buffer flushed, so `total` reflects everything the block
526
+ # pushed. Scheduled (`perform_in`) jobs count too: BATCH_SCHEDULE moves
527
+ # `total` at creation, so a scheduled-only block is not mistaken for an
528
+ # empty one.
529
+ def untouched?
530
+ total, kids = Wurk.redis do |conn|
531
+ conn.pipelined do |pipe|
532
+ pipe.call('HGET', "b-#{@bid}", 'total')
533
+ pipe.call('SCARD', "b-#{@bid}-kids")
534
+ end
535
+ end
536
+ total.to_i.zero? && kids.to_i.zero?
437
537
  end
438
538
 
539
+ # Pushed straight through, never into an enclosing batch's buffer: it has
540
+ # to be live before this block's hold is released. The payload names the
541
+ # class `Sidekiq::Batch::Empty`, the spec's wire name (§2.3) — the alias
542
+ # resolves it here, and Sidekiq Pro can still run it after a swap back.
439
543
  def enqueue_empty_marker
440
- require_relative 'batch/empty'
441
- previous = Thread.current[THREAD_KEY]
442
- Thread.current[THREAD_KEY] = self
443
- begin
444
- Wurk::Batch::Empty.perform_async
445
- ensure
446
- Thread.current[THREAD_KEY] = previous
544
+ Batch.with_thread_batch(self, nil) do
545
+ Wurk::Client.push('class' => EMPTY_JOB, 'args' => [], 'queue' => 'default', 'retry' => false)
447
546
  end
448
547
  end
449
548
 
450
549
  def cascade_invalidate(bid)
451
550
  Wurk.redis do |conn|
452
- Wurk::Lua::Loader.eval_cached(conn, :batch_invalidate, keys: ["b-#{bid}", "b-#{bid}-jids"], argv: [])
551
+ Wurk::Lua::Loader.eval_cached(conn, :batch_invalidate, keys: ["b-#{bid}"], argv: [])
453
552
  kids = conn.call('SMEMBERS', "b-#{bid}-kids") || []
454
553
  kids.each { |child| cascade_invalidate(child) }
455
554
  end
data/lib/wurk/capsule.rb CHANGED
@@ -13,8 +13,6 @@ module Wurk
13
13
  #
14
14
  # Spec: docs/target/sidekiq-free.md §5 (Sidekiq::Capsule).
15
15
  class Capsule
16
- MODES = %i[strict weighted random].freeze
17
-
18
16
  attr_reader :name, :queues, :mode, :weights, :config, :watchdog
19
17
  attr_accessor :concurrency, :fetcher
20
18