wurk 1.7.6 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (217) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +7 -4
  3. data/app/controllers/concerns/wurk/sse_streaming.rb +2 -1
  4. data/app/controllers/concerns/wurk/stream_concurrency_guard.rb +33 -9
  5. data/app/controllers/wurk/api/pagination.rb +25 -7
  6. data/app/controllers/wurk/api_controller.rb +49 -25
  7. data/app/controllers/wurk/extensions_controller.rb +8 -2
  8. data/app/controllers/wurk/profiles_controller.rb +45 -9
  9. data/config/routes.rb +31 -10
  10. data/lib/generators/wurk/install/templates/wurk.rb +1 -1
  11. data/lib/sidekiq/capsule.rb +4 -0
  12. data/lib/sidekiq/component.rb +4 -0
  13. data/lib/sidekiq/config.rb +4 -0
  14. data/lib/sidekiq/deploy.rb +4 -0
  15. data/lib/sidekiq/embedded.rb +4 -0
  16. data/lib/sidekiq/fetch.rb +4 -0
  17. data/lib/sidekiq/iterable_job.rb +4 -0
  18. data/lib/sidekiq/job/interrupt_handler.rb +4 -0
  19. data/lib/sidekiq/job/iterable/active_record_enumerator.rb +4 -0
  20. data/lib/sidekiq/job/iterable/csv_enumerator.rb +4 -0
  21. data/lib/sidekiq/job/iterable.rb +4 -0
  22. data/lib/sidekiq/job_logger.rb +4 -0
  23. data/lib/sidekiq/job_util.rb +4 -0
  24. data/lib/sidekiq/loader.rb +4 -0
  25. data/lib/sidekiq/logger.rb +4 -0
  26. data/lib/sidekiq/metrics/query.rb +4 -0
  27. data/lib/sidekiq/middleware/current_attributes.rb +7 -0
  28. data/lib/sidekiq/middleware/i18n.rb +6 -0
  29. data/lib/sidekiq/middleware/modules.rb +4 -0
  30. data/lib/sidekiq/pro/web.rb +5 -0
  31. data/lib/sidekiq/profiler.rb +4 -0
  32. data/lib/sidekiq/redis_client_adapter.rb +4 -0
  33. data/lib/sidekiq/test_api.rb +5 -0
  34. data/lib/sidekiq/testing/inline.rb +8 -0
  35. data/lib/sidekiq/transaction_aware_client.rb +4 -0
  36. data/lib/sidekiq/worker_compatibility_alias.rb +4 -0
  37. data/lib/sidekiq-ent/periodic/testing.rb +5 -0
  38. data/lib/sidekiq-ent/web.rb +5 -0
  39. data/lib/sidekiq-ent.rb +7 -0
  40. data/lib/sidekiq-pro.rb +7 -0
  41. data/lib/wurk/api/fast.rb +8 -33
  42. data/lib/wurk/api/idempotency.rb +28 -5
  43. data/lib/wurk/api/swarm.rb +1 -1
  44. data/lib/wurk/batch/callbacks.rb +143 -92
  45. data/lib/wurk/batch/death_handler.rb +19 -8
  46. data/lib/wurk/batch/server_middleware.rb +37 -24
  47. data/lib/wurk/batch/status.rb +49 -15
  48. data/lib/wurk/batch.rb +164 -65
  49. data/lib/wurk/capsule.rb +0 -2
  50. data/lib/wurk/client/buffered.rb +227 -106
  51. data/lib/wurk/client.rb +33 -13
  52. data/lib/wurk/compat.rb +13 -0
  53. data/lib/wurk/configuration.rb +155 -24
  54. data/lib/wurk/cpu_count.rb +97 -0
  55. data/lib/wurk/cron.rb +97 -34
  56. data/lib/wurk/dead_set.rb +6 -7
  57. data/lib/wurk/encryption.rb +46 -13
  58. data/lib/wurk/fetcher/many_queues.rb +65 -0
  59. data/lib/wurk/fetcher/orphan_grace.rb +67 -0
  60. data/lib/wurk/fetcher/private_list_key.rb +76 -0
  61. data/lib/wurk/fetcher/reaper.rb +55 -74
  62. data/lib/wurk/fetcher/reliable.rb +23 -17
  63. data/lib/wurk/fetcher/unit_of_work.rb +19 -1
  64. data/lib/wurk/fetcher/unparseable_keys.rb +56 -0
  65. data/lib/wurk/flow/completion.rb +28 -3
  66. data/lib/wurk/flow/creation.rb +24 -4
  67. data/lib/wurk/flow.rb +24 -2
  68. data/lib/wurk/health.rb +182 -40
  69. data/lib/wurk/heartbeat.rb +37 -14
  70. data/lib/wurk/history.rb +7 -1
  71. data/lib/wurk/import/sidekiq_cron.rb +248 -0
  72. data/lib/wurk/job/options.rb +22 -3
  73. data/lib/wurk/job_logger.rb +13 -3
  74. data/lib/wurk/job_record.rb +68 -19
  75. data/lib/wurk/job_retry.rb +12 -6
  76. data/lib/wurk/job_set.rb +99 -51
  77. data/lib/wurk/job_util.rb +27 -8
  78. data/lib/wurk/launcher.rb +84 -40
  79. data/lib/wurk/leader.rb +53 -28
  80. data/lib/wurk/limiter/base.rb +11 -30
  81. data/lib/wurk/limiter/bucket.rb +14 -5
  82. data/lib/wurk/limiter/server_middleware.rb +2 -1
  83. data/lib/wurk/limiter.rb +1 -1
  84. data/lib/wurk/loader.rb +48 -0
  85. data/lib/wurk/lua/batch_ack_complete.lua +28 -0
  86. data/lib/wurk/lua/batch_ack_failed.lua +18 -0
  87. data/lib/wurk/lua/batch_ack_success.lua +35 -0
  88. data/lib/wurk/lua/batch_append_callback.lua +66 -0
  89. data/lib/wurk/lua/batch_delete.lua +31 -0
  90. data/lib/wurk/lua/batch_hold.lua +17 -0
  91. data/lib/wurk/lua/batch_invalidate.lua +14 -0
  92. data/lib/wurk/lua/batch_kid_done.lua +14 -0
  93. data/lib/wurk/lua/batch_push.lua +47 -0
  94. data/lib/wurk/lua/batch_remove_jobs.lua +29 -0
  95. data/lib/wurk/lua/batch_schedule.lua +27 -0
  96. data/lib/wurk/lua/fetch_first.lua +27 -0
  97. data/lib/wurk/lua/flow_abandon.lua +16 -21
  98. data/lib/wurk/lua/flow_advance.lua +41 -26
  99. data/lib/wurk/lua/flow_create.lua +24 -9
  100. data/lib/wurk/lua/flow_fail.lua +3 -3
  101. data/lib/wurk/lua/leader_campaign.lua +19 -0
  102. data/lib/wurk/lua/limiter_bucket_acquire.lua +3 -2
  103. data/lib/wurk/lua/limiter_list_sweep.lua +8 -9
  104. data/lib/wurk/lua/loader.rb +36 -23
  105. data/lib/wurk/lua/throttle_slot.lua +13 -26
  106. data/lib/wurk/lua.rb +31 -244
  107. data/lib/wurk/manager.rb +25 -18
  108. data/lib/wurk/metrics/accumulator.rb +31 -10
  109. data/lib/wurk/metrics/dashboard_series.rb +163 -0
  110. data/lib/wurk/metrics/histogram.rb +46 -0
  111. data/lib/wurk/metrics/history.rb +49 -57
  112. data/lib/wurk/metrics/prometheus.rb +232 -0
  113. data/lib/wurk/metrics/query.rb +127 -155
  114. data/lib/wurk/metrics/rollup.rb +2 -2
  115. data/lib/wurk/metrics/statsd.rb +1 -1
  116. data/lib/wurk/middleware/interrupt_handler.rb +6 -2
  117. data/lib/wurk/middleware/poison_pill.rb +5 -4
  118. data/lib/wurk/middleware/timeout.rb +8 -8
  119. data/lib/wurk/process_set.rb +5 -5
  120. data/lib/wurk/processor.rb +69 -31
  121. data/lib/wurk/profile_set.rb +19 -24
  122. data/lib/wurk/profiler.rb +48 -50
  123. data/lib/wurk/queue.rb +36 -19
  124. data/lib/wurk/rails_boot.rb +33 -1
  125. data/lib/wurk/railtie.rb +2 -0
  126. data/lib/wurk/rake_tasks.rb +18 -0
  127. data/lib/wurk/redact.rb +39 -0
  128. data/lib/wurk/redis_connection.rb +6 -3
  129. data/lib/wurk/redis_options.rb +21 -4
  130. data/lib/wurk/redis_pool.rb +119 -21
  131. data/lib/wurk/scheduled.rb +44 -11
  132. data/lib/wurk/sentry/error_handler.rb +53 -7
  133. data/lib/wurk/sentry/middleware.rb +7 -8
  134. data/lib/wurk/shutdown_gate.rb +24 -1
  135. data/lib/wurk/sorted_entry.rb +16 -5
  136. data/lib/wurk/stats.rb +9 -4
  137. data/lib/wurk/status.rb +2 -2
  138. data/lib/wurk/swarm/child_boot.rb +48 -38
  139. data/lib/wurk/swarm/liveness.rb +153 -0
  140. data/lib/wurk/swarm/restart.rb +24 -3
  141. data/lib/wurk/swarm.rb +115 -24
  142. data/lib/wurk/telemetry/server_middleware.rb +2 -5
  143. data/lib/wurk/testing.rb +15 -5
  144. data/lib/wurk/throttle.rb +16 -1
  145. data/lib/wurk/topology.rb +1 -5
  146. data/lib/wurk/unique.rb +1 -1
  147. data/lib/wurk/version.rb +1 -1
  148. data/lib/wurk/watchdog.rb +2 -2
  149. data/lib/wurk/web/config.rb +56 -20
  150. data/lib/wurk/web/enterprise.rb +21 -18
  151. data/lib/wurk/web/extension.rb +98 -22
  152. data/lib/wurk/web/rack_app.rb +12 -11
  153. data/lib/wurk/web/search.rb +18 -8
  154. data/lib/wurk/worker/setter.rb +17 -14
  155. data/lib/wurk/worker.rb +6 -67
  156. data/lib/wurk.rb +2 -3
  157. data/vendor/assets/dashboard/assets/ArgsValue-DsPb_KZ-.js +1 -0
  158. data/vendor/assets/dashboard/assets/BatchDetail-BCIXJOpZ.js +1 -0
  159. data/vendor/assets/dashboard/assets/Batches-IjaQfXW6.js +1 -0
  160. data/vendor/assets/dashboard/assets/Busy-CaQrf1Ve.js +1 -0
  161. data/vendor/assets/dashboard/assets/Cron-B18EHn_Y.js +1 -0
  162. data/vendor/assets/dashboard/assets/Dashboard-DdT5ZaUg.js +1 -0
  163. data/vendor/assets/dashboard/assets/Dead-DWIy-Mu8.js +1 -0
  164. data/vendor/assets/dashboard/assets/Extension-BjBUWhHf.js +1 -0
  165. data/vendor/assets/dashboard/assets/FilterBox-DwH3PXXt.js +1 -0
  166. data/vendor/assets/dashboard/assets/FlowDetail-DnexL72n.js +1 -0
  167. data/vendor/assets/dashboard/assets/FlowState-Cr4aX6-d.js +1 -0
  168. data/vendor/assets/dashboard/assets/Flows-DO6EimbZ.js +1 -0
  169. data/vendor/assets/dashboard/assets/JobDetailModal-aPjgq-U9.js +2 -0
  170. data/vendor/assets/dashboard/assets/Limiters-DxpwWCNe.js +1 -0
  171. data/vendor/assets/dashboard/assets/Metrics-C3JL1GIH.js +1 -0
  172. data/vendor/assets/dashboard/assets/NotFound-BfP9uAV9.js +1 -0
  173. data/vendor/assets/dashboard/assets/PageHeader-DbC97afG.js +1 -0
  174. data/vendor/assets/dashboard/assets/Profiles-BcaHreMe.js +1 -0
  175. data/vendor/assets/dashboard/assets/Queues-ZlBc5tyD.js +1 -0
  176. data/vendor/assets/dashboard/assets/Retries-Bg3nkeiL.js +1 -0
  177. data/vendor/assets/dashboard/assets/Scheduled-BrMZPZGl.js +1 -0
  178. data/vendor/assets/dashboard/assets/Search-B5X5pGht.js +1 -0
  179. data/vendor/assets/dashboard/assets/charts-uDyf4PN0.js +1 -0
  180. data/vendor/assets/dashboard/assets/i18n-CJeYfeVt.js +1 -0
  181. data/vendor/assets/dashboard/assets/index-B36QCAf-.css +1 -0
  182. data/vendor/assets/dashboard/assets/index-DQujgZgH.js +141 -0
  183. data/vendor/assets/dashboard/assets/useResetPageOnEmpty-O3-5rMk3.js +1 -0
  184. data/vendor/assets/dashboard/assets/useSort-xfnwjQAu.js +1 -0
  185. data/vendor/assets/dashboard/assets/utils-BIKHwNKB.js +1 -0
  186. data/vendor/assets/dashboard/index.html +3 -3
  187. data/vendor/assets/dashboard/wurk-manifest.json +2 -2
  188. metadata +90 -34
  189. data/config/locales/en.yml +0 -15
  190. data/vendor/assets/dashboard/assets/ArgsValue-D-x_ifLY.js +0 -1
  191. data/vendor/assets/dashboard/assets/BatchDetail-C39NJuew.js +0 -1
  192. data/vendor/assets/dashboard/assets/Batches-CSwo7Asa.js +0 -1
  193. data/vendor/assets/dashboard/assets/Busy-BOFMu-sq.js +0 -1
  194. data/vendor/assets/dashboard/assets/Cron-Dy8RQzDI.js +0 -1
  195. data/vendor/assets/dashboard/assets/Dashboard-BuTHI-O1.js +0 -1
  196. data/vendor/assets/dashboard/assets/Dead-B9KRvQ0N.js +0 -1
  197. data/vendor/assets/dashboard/assets/Extension-BnBVHfux.js +0 -1
  198. data/vendor/assets/dashboard/assets/FilterBox-DC24zite.js +0 -1
  199. data/vendor/assets/dashboard/assets/FlowDetail-DyLuzUvt.js +0 -1
  200. data/vendor/assets/dashboard/assets/FlowState-DAPKUahm.js +0 -1
  201. data/vendor/assets/dashboard/assets/Flows-Fr3rjZM_.js +0 -1
  202. data/vendor/assets/dashboard/assets/JobDetailModal-N6kiJXq3.js +0 -2
  203. data/vendor/assets/dashboard/assets/Limiters-kbFA7uS1.js +0 -1
  204. data/vendor/assets/dashboard/assets/Metrics-Dj2uoZ3o.js +0 -1
  205. data/vendor/assets/dashboard/assets/PageHeader-B_F94azl.js +0 -1
  206. data/vendor/assets/dashboard/assets/Profiles-D_DjEezN.js +0 -1
  207. data/vendor/assets/dashboard/assets/Queues-CO4V9hAz.js +0 -1
  208. data/vendor/assets/dashboard/assets/Retries-DCWnzeLa.js +0 -1
  209. data/vendor/assets/dashboard/assets/Scheduled-BebDUjLU.js +0 -1
  210. data/vendor/assets/dashboard/assets/Search-Cvr5fy4Y.js +0 -1
  211. data/vendor/assets/dashboard/assets/Skeleton-Bu3Ke6rV.js +0 -1
  212. data/vendor/assets/dashboard/assets/charts-BCs9bQKz.js +0 -1
  213. data/vendor/assets/dashboard/assets/index-BIwyOC5Q.js +0 -141
  214. data/vendor/assets/dashboard/assets/index-DBQN6Jk8.css +0 -1
  215. data/vendor/assets/dashboard/assets/useResetPageOnEmpty-Bzh-BJyL.js +0 -1
  216. data/vendor/assets/dashboard/assets/useSort-COA3fVJ5.js +0 -1
  217. data/vendor/assets/dashboard/assets/utils-BIrvZ1hi.js +0 -1
@@ -98,8 +98,26 @@ module Wurk
98
98
  config.redis(idempotent: true) { |conn| conn.call('ZREM', slot_key, slot_token) }
99
99
  end
100
100
 
101
+ # Puts a claimed job back without running it — the public half of the
102
+ # Sidekiq UnitOfWork contract that throttling/rate-limiting gems call
103
+ # (sidekiq-throttled's requeue path, anything holding a BasicFetch UoW,
104
+ # which is this one under the `Sidekiq::BasicFetch` alias). Sidekiq's
105
+ # BasicFetch has no private list, so its bare RPUSH is the whole move;
106
+ # here the private copy has to go in the same step, or the job sits in
107
+ # both lists and runs twice — once from the public queue, once when the
108
+ # reaper reclaims this process's private list.
109
+ #
110
+ # LREM then RPUSH in one MULTI, unconditionally: unlike bulk_requeue
111
+ # (which races a Processor ACKing the same UoW and so guards the push on
112
+ # the LREM), the caller here owns the unit, and a job whose ACK already
113
+ # went out must still be re-queued rather than dropped.
101
114
  def requeue
102
- config.redis { |conn| conn.call('RPUSH', queue, job) }
115
+ config.redis do |conn|
116
+ conn.multi do |tx|
117
+ tx.call('LREM', private_queue, LREM_COUNT, job)
118
+ tx.call('RPUSH', queue, job)
119
+ end
120
+ end
103
121
  end
104
122
  end
105
123
  end
@@ -0,0 +1,56 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative '../keys'
4
+
5
+ module Wurk
6
+ class Fetcher
7
+ # The Reaper's report on a `queue:*|*` key that is not a private list it
8
+ # can read the owner of — a Sidekiq Pro super_fetch list in a shape Wurk
9
+ # does not write, say, left behind by a live migration. Nothing will ever
10
+ # reclaim its jobs, so it is said once per key per process (WARN, with the
11
+ # list's length) rather than skipped in silence on every sweep.
12
+ #
13
+ # A public queue whose own name contains `|` matches the same SCAN pattern,
14
+ # so a key whose name is in the `queues` SET is not reported.
15
+ class UnparseableKeys
16
+ # Distinct keys remembered as already reported. Bounded so a keyspace
17
+ # full of odd keys cannot grow it without limit; past the bound the
18
+ # memory is cleared and keys are reported again.
19
+ MEMORY = 1000
20
+
21
+ def initialize(config)
22
+ @config = config
23
+ @reported = ::Set.new
24
+ end
25
+
26
+ def report(key)
27
+ return if @reported.include?(key)
28
+
29
+ public_queue, size = inspect_key(key)
30
+ return if public_queue
31
+
32
+ @reported.clear if @reported.size >= MEMORY
33
+ @reported << key
34
+ @config.logger.warn do
35
+ "reaper: #{key.inspect} looks like a reliable-fetch private list but does not match " \
36
+ "queue:<queue>|<host>|<pid>|<nonce>|<index>; its #{size} job(s) will not be recovered " \
37
+ 'automatically. Inspect it and LMOVE them back onto their queue by hand if they belong to a dead process.'
38
+ end
39
+ rescue StandardError => e
40
+ @config.handle_exception(e, context: 'wurk-reaper')
41
+ end
42
+
43
+ private
44
+
45
+ def inspect_key(key)
46
+ member, size = @config.redis(idempotent: true) do |conn|
47
+ conn.pipelined do |pipe|
48
+ pipe.call('SISMEMBER', Keys::QUEUES_SET, key.delete_prefix(Keys::QUEUE_PREFIX))
49
+ pipe.call('LLEN', key)
50
+ end
51
+ end
52
+ [member == 1, size]
53
+ end
54
+ end
55
+ end
56
+ end
@@ -59,19 +59,44 @@ module Wurk
59
59
  # spent: a replay after a lost reply finds the node already out of the
60
60
  # state it claims from and writes nothing, so the pool may retry a
61
61
  # connection error rather than give up on a callback that fires once.
62
+ #
63
+ # The graph around the node is read first so the script is handed every
64
+ # key it may write as a declared key (Redis Cluster and Dragonfly refuse
65
+ # one built inside Lua). Safe to read ahead: a node's jid, its
66
+ # dependents, and their bids and queues are all fixed at creation; the
67
+ # counters and states the decision turns on are read inside the script.
62
68
  def advance(fid, index)
63
69
  Wurk.redis(idempotent: true) do |conn|
70
+ node_key = Keys.flow_node(fid, index)
71
+ jid, dependents = conn.call('HMGET', node_key, 'jid', 'dependents')
72
+ deps = dependents ? Wurk.load_json(dependents) : []
64
73
  Wurk::Lua::Loader.eval_cached(
65
74
  conn, :flow_advance,
66
- keys: [Keys.flow(fid), 'queues'],
67
- argv: [index, now_seconds, now_millis, Keys::STATUS_PREFIX]
75
+ keys: [Keys.flow(fid), 'queues', Keys::SCHEDULE, Keys.flow_dead(fid), node_key,
76
+ Keys.status(jid.to_s), *dependent_keys(conn, fid, deps)],
77
+ argv: [index, now_seconds, now_millis, *deps.map(&:to_s)]
68
78
  )
69
79
  end
70
80
  end
71
81
 
82
+ # Four keys per dependent, in `deps` order: its record, its batch, the
83
+ # batch's live-jid set, and the queue it is released onto.
84
+ def dependent_keys(conn, fid, deps)
85
+ return [] if deps.empty?
86
+
87
+ rows = conn.pipelined { |pipe| deps.each { |d| pipe.call('HMGET', Keys.flow_node(fid, d), 'bid', 'queue') } }
88
+ deps.zip(rows).flat_map do |d, (bid, queue)|
89
+ [Keys.flow_node(fid, d), "b-#{bid}", "b-#{bid}-jids", Keys.queue(queue)]
90
+ end
91
+ end
92
+
72
93
  def mark_dead(fid, index)
73
94
  Wurk.redis(idempotent: true) do |conn|
74
- Wurk::Lua::Loader.eval_cached(conn, :flow_fail, keys: [Keys.flow(fid)], argv: [index, now_seconds])
95
+ Wurk::Lua::Loader.eval_cached(
96
+ conn, :flow_fail,
97
+ keys: [Keys.flow(fid), Keys.flow_node(fid, index), Keys.flow_dead(fid)],
98
+ argv: [index, now_seconds]
99
+ )
75
100
  end
76
101
  end
77
102
 
@@ -160,13 +160,24 @@ module Wurk
160
160
  # pool may retry a connection error that a non-idempotent write would have
161
161
  # to give up on.
162
162
  def write(payloads)
163
- keys = [Keys.flow(@flow.fid), Keys::FLOWS_SET, 'batches', 'queues']
163
+ keys = [Keys.flow(@flow.fid), Keys::FLOWS_SET, 'batches', 'queues', Keys::SCHEDULE, *node_keys(payloads)]
164
164
  argv = script_argv(payloads)
165
165
  Wurk.redis(idempotent: true) do |conn|
166
166
  Wurk::Lua::Loader.eval_cached(conn, :flow_create, keys: keys, argv: argv)
167
167
  end
168
168
  end
169
169
 
170
+ # Four declared keys per node, in the same topological order as the
171
+ # envelopes in {#script_argv}: its record, its batch, the batch's live-jid
172
+ # set, and its queue.
173
+ def node_keys(payloads)
174
+ @flow.nodes.flat_map do |node|
175
+ payload = payloads[node.index]
176
+ bid = payload['bid']
177
+ [Keys.flow_node(@flow.fid, node.index), "b-#{bid}", "b-#{bid}-jids", Keys.queue(payload['queue'])]
178
+ end
179
+ end
180
+
170
181
  def script_argv(payloads)
171
182
  now = ::Process.clock_gettime(::Process::CLOCK_REALTIME)
172
183
  cutoff, rank = Wurk::Batch.trim_bounds
@@ -185,19 +196,28 @@ module Wurk
185
196
  # `enqueued_at` marks arrival on an immediate queue, so only the nodes
186
197
  # this write actually queues carry one. A waiting node's stored payload
187
198
  # gets its stamp from the push that releases it.
199
+ #
200
+ # A node declared with `at:` is deferred, root or not, so its payload is
201
+ # stored as the `schedule` member Client#push would write — `at` is the
202
+ # score, and the promoter stamps `enqueued_at` — and the epoch rides
203
+ # beside it for whichever script releases the node.
188
204
  def envelope(node, payload)
189
- payload['enqueued_at'] = now_in_millis if node.root?
205
+ payload['enqueued_at'] = now_in_millis if node.root? && !payload.key?('at')
190
206
  Wurk.dump_json(node_fields(node, payload).merge(edge_fields(node)))
191
207
  end
192
208
 
193
209
  def node_fields(node, payload)
194
- json = Wurk.dump_json(payload)
210
+ json = stored_payload(payload)
195
211
  { 'i' => node.index.to_s, 'name' => node.name.to_s, 'class' => payload['class'],
196
212
  'queue' => payload['queue'], 'jid' => payload['jid'], 'bid' => payload['bid'],
197
- 'state' => node.root? ? ENQUEUED : WAITING, 'payload' => json,
213
+ 'state' => node.root? ? ENQUEUED : WAITING, 'payload' => json, 'at' => payload['at'].to_s,
198
214
  'desc' => node.label, 'cb' => callbacks_json(node), 'pipe' => pipe_field(node, json) }
199
215
  end
200
216
 
217
+ def stored_payload(payload)
218
+ payload.key?('at') ? JobUtil.scheduled_member(payload) : Wurk.dump_json(payload)
219
+ end
220
+
201
221
  # Empty for an ordinary node, and the sentinel — exactly as it appears in
202
222
  # the stored bytes — for a piped one, which is both the flag the release
203
223
  # tests and the needle it splices at.
data/lib/wurk/flow.rb CHANGED
@@ -179,14 +179,36 @@ module Wurk
179
179
  def abandon(fid) # rubocop:disable Naming/PredicateMethod
180
180
  now = ::Process.clock_gettime(::Process::CLOCK_REALTIME).to_s
181
181
  released = Wurk.redis(idempotent: true) do |conn|
182
+ bids = node_bids(conn, fid)
182
183
  Wurk::Lua::Loader.eval_cached(
183
184
  conn, :flow_abandon,
184
- keys: [Keys.flow(fid), 'batches', 'dead-batches'],
185
- argv: [now, *Wurk::Batch::KEY_SUFFIXES]
185
+ keys: [Keys.flow(fid), 'batches', 'dead-batches', *abandoned_keys(fid, bids)],
186
+ argv: [now, *bids.compact]
186
187
  )
187
188
  end
188
189
  released.to_i >= 0
189
190
  end
191
+
192
+ private
193
+
194
+ # Every node's bid, in index order, nil where a record is gone. Read ahead
195
+ # of the script so it can be handed declared keys: a node's bid is fixed at
196
+ # creation, and the script's own state claim still decides whether
197
+ # anything is released.
198
+ def node_bids(conn, fid)
199
+ total = conn.call('HGET', Keys.flow(fid), 'total').to_i
200
+ return [] if total.zero?
201
+
202
+ conn.pipelined { |pipe| total.times { |i| pipe.call('HGET', Keys.flow_node(fid, i), 'bid') } }
203
+ end
204
+
205
+ def abandoned_keys(fid, bids)
206
+ keys = bids.each_with_index.flat_map do |bid, i|
207
+ batch = bid ? ["b-#{bid}", *Wurk::Batch::KEY_SUFFIXES.map { |suffix| "b-#{bid}-#{suffix}" }] : []
208
+ [Keys.flow_node(fid, i), *batch]
209
+ end
210
+ keys << Keys.flow_dead(fid)
211
+ end
190
212
  end
191
213
 
192
214
  def initialize(&block)
data/lib/wurk/health.rb CHANGED
@@ -2,6 +2,7 @@
2
2
 
3
3
  require 'socket'
4
4
  require 'json'
5
+ require_relative 'metrics/prometheus'
5
6
 
6
7
  module Wurk
7
8
  # Thin HTTP listener for k8s liveness/readiness probes. Optional, off by
@@ -10,10 +11,16 @@ module Wurk
10
11
  # Endpoints:
11
12
  # * GET /live → 200 while the Launcher is running (not in quiet/stop).
12
13
  # * GET /ready → 200 only when Redis is reachable AND the heartbeat has
13
- # fired within `ready_window` seconds. 503 otherwise.
14
+ # fired within `ready_window` seconds AND, inside a swarm
15
+ # child, enough of the swarm's children have (see
16
+ # Health::Fleet). 503 otherwise.
17
+ # * GET /metrics → Prometheus text exposition (Wurk::Metrics::Prometheus).
18
+ # On whenever the listener is; `metrics: false` in
19
+ # `:health_check_options` turns it into a 404.
14
20
  # Anything else returns 404 JSON.
15
21
  #
16
- # The server uses a raw TCPServer and one accept thread. No Rack, no
22
+ # The server uses a raw TCPServer, one accept thread and a short-lived
23
+ # thread per connection (bounded, deadline-capped). No Rack, no
17
24
  # dependencies — it lives inside every worker process where Rails may or
18
25
  # may not exist (standalone CLI, Embedded, swarm child). Bound to a
19
26
  # dedicated port so it does not collide with the host application's HTTP.
@@ -23,6 +30,19 @@ module Wurk
23
30
  DEFAULT_PORT = 7433
24
31
  DEFAULT_BIND = '0.0.0.0'
25
32
  DEFAULT_READY_WINDOW = 30
33
+ JSON_TYPE = 'application/json'
34
+
35
+ class << self
36
+ # Children the swarm was configured with, set by Swarm::ChildBoot inside
37
+ # each child (inherited from nowhere: the parent never runs a listener).
38
+ # nil outside a swarm child — standalone and embedded processes are a
39
+ # fleet of one and /ready judges them on their own heartbeat alone.
40
+ attr_accessor :fleet_size
41
+
42
+ def json(status, **extra)
43
+ ::JSON.generate({ status: status }.merge(extra))
44
+ end
45
+ end
26
46
 
27
47
  # The HTTP listener. Owns one TCPServer + one accept thread. Idempotent
28
48
  # start/stop; safe to call from Launcher#run / Launcher#stop.
@@ -32,6 +52,13 @@ module Wurk
32
52
  # that probes come back quickly after the owner dies, long enough not to
33
53
  # spin. See #start_retry_loop.
34
54
  RETRY_INTERVAL = 5
55
+ # Overall budget for one request head. A kubelet probe sends its head in
56
+ # one segment; anything slower is broken or hostile.
57
+ REQUEST_TIMEOUT = 1.0
58
+ MAX_REQUEST_BYTES = 8192
59
+ MAX_CONNECTIONS = 16
60
+ REASONS = { 200 => 'OK', 404 => 'Not Found', 405 => 'Method Not Allowed',
61
+ 503 => 'Service Unavailable' }.freeze
35
62
 
36
63
  attr_reader :port, :bind
37
64
 
@@ -47,6 +74,9 @@ module Wurk
47
74
  @thread = nil
48
75
  @retry_thread = nil
49
76
  @done = false
77
+ @slots = ::Mutex.new
78
+ @connections = 0
79
+ @fleet = nil
50
80
  end
51
81
 
52
82
  def start
@@ -133,7 +163,7 @@ module Wurk
133
163
 
134
164
  begin
135
165
  client, _addr = @server.accept_nonblock(exception: false)
136
- handle(client) if client
166
+ dispatch(client) if client
137
167
  rescue ::IO::WaitReadable
138
168
  next
139
169
  rescue ::StandardError => e
@@ -145,41 +175,105 @@ module Wurk
145
175
  # Server was closed during shutdown — expected.
146
176
  end
147
177
 
148
- def handle(client) # rubocop:disable Metrics/AbcSize
149
- return unless client.wait_readable(1.0)
178
+ # Each connection gets its own short-lived thread so a client that
179
+ # dribbles bytes (slowloris) cannot hold the accept loop — and with it
180
+ # every kubelet probe — hostage. Past MAX_CONNECTIONS the socket is
181
+ # closed unanswered rather than queued: a probe that loses that race
182
+ # retries, an attacker gains nothing.
183
+ def dispatch(client)
184
+ return client.close unless claim_slot
185
+
186
+ spawn_handler(client)
187
+ rescue ::ThreadError
188
+ release_slot
189
+ client.close
190
+ end
150
191
 
151
- request_line = client.gets("\r\n")
152
- return if request_line.nil?
192
+ def spawn_handler(client)
193
+ ::Thread.new(client) do |sock|
194
+ ::Thread.current.name = 'wurk-health-conn'
195
+ ::Thread.current.report_on_exception = false
196
+ handle(sock)
197
+ ensure
198
+ release_slot
199
+ end
200
+ end
153
201
 
154
- method, path, = request_line.strip.split(' ', 3)
155
- # Drain remaining headers; ignore the body (probes don't send one).
156
- # Wait for readability before each gets so a stalled client can't
157
- # block the single accept thread mid-headers.
158
- loop do
159
- break unless client.wait_readable(1.0)
160
-
161
- line = client.gets("\r\n")
162
- break if line.nil? || line == "\r\n"
202
+ def claim_slot
203
+ @slots.synchronize do
204
+ return false if @connections >= MAX_CONNECTIONS
205
+
206
+ @connections += 1
207
+ true
163
208
  end
209
+ end
164
210
 
165
- body, status = response_for(method, path)
166
- write_response(client, status, body)
211
+ def release_slot
212
+ @slots.synchronize { @connections -= 1 }
213
+ end
214
+
215
+ def handle(client)
216
+ request_line = read_request_line(client)
217
+ return if request_line.nil?
218
+
219
+ method, path, = request_line.strip.split(' ', 3)
220
+ body, status, type = response_for(method, path)
221
+ write_response(client, status, body, type || JSON_TYPE)
167
222
  rescue ::StandardError => e
168
223
  logger&.error { "Wurk::Health request: #{e.class}: #{e.message}" }
169
224
  ensure
170
225
  client&.close
171
226
  end
172
227
 
228
+ # Reads the head under one overall deadline and a MAX_REQUEST_BYTES cap.
229
+ # Headers are read only to drain them (closing a socket with unread bytes
230
+ # makes the kernel answer RST, which can eat the response); the answer
231
+ # depends on the request line alone, so a client that sends that line
232
+ # and then stalls is still answered when the deadline expires.
233
+ def read_request_line(client)
234
+ buffer = +''
235
+ deadline = monotonic + REQUEST_TIMEOUT
236
+ until buffer.include?("\r\n\r\n") || buffer.bytesize >= MAX_REQUEST_BYTES
237
+ chunk = read_chunk(client, MAX_REQUEST_BYTES - buffer.bytesize, deadline)
238
+ break unless chunk
239
+
240
+ buffer << chunk
241
+ end
242
+ buffer[/\A[^\r\n]*(?=\r\n)/]
243
+ end
244
+
245
+ # nil once the deadline passes or the peer closes.
246
+ def read_chunk(client, max_bytes, deadline)
247
+ remaining = deadline - monotonic
248
+ return unless remaining.positive? && client.wait_readable(remaining)
249
+
250
+ chunk = client.read_nonblock(max_bytes, exception: false)
251
+ chunk == :wait_readable ? '' : chunk
252
+ end
253
+
254
+ def monotonic
255
+ ::Process.clock_gettime(::Process::CLOCK_MONOTONIC)
256
+ end
257
+
173
258
  def response_for(method, path)
174
259
  return [json('error', message: 'method not allowed'), 405] unless method == 'GET'
175
260
 
176
261
  case path
177
262
  when '/live' then live_response
178
263
  when '/ready' then ready_response
179
- else [json('error', message: 'not found', path: path), 404]
264
+ when '/metrics' then fleet.metrics_response || not_found(path)
265
+ else not_found(path)
180
266
  end
181
267
  end
182
268
 
269
+ def not_found(path)
270
+ [json('error', message: 'not found', path: path), 404]
271
+ end
272
+
273
+ def fleet
274
+ @fleet ||= Fleet.new(@config, @ready_window)
275
+ end
276
+
183
277
  def live_response
184
278
  if @launcher.stopping?
185
279
  [json('down', check: 'live', reason: 'stopping'), 503]
@@ -189,15 +283,10 @@ module Wurk
189
283
  end
190
284
 
191
285
  def ready_response
192
- redis_ok = ping_redis
193
- beat_fresh = heartbeat_fresh?
286
+ return [json('down', check: 'ready', reason: 'redis unreachable'), 503] unless ping_redis
287
+ return [json('down', check: 'ready', reason: 'heartbeat stale'), 503] unless heartbeat_fresh?
194
288
 
195
- if redis_ok && beat_fresh
196
- [json('ok', check: 'ready'), 200]
197
- else
198
- reason = redis_ok ? 'heartbeat stale' : 'redis unreachable'
199
- [json('down', check: 'ready', reason: reason), 503]
200
- end
289
+ fleet.ready_response
201
290
  end
202
291
 
203
292
  def ping_redis
@@ -216,22 +305,13 @@ module Wurk
216
305
  (::Time.now.to_f - last) < @ready_window
217
306
  end
218
307
 
219
- def json(status, **extra)
220
- ::JSON.generate({ status: status }.merge(extra))
221
- end
222
-
223
- def write_response(client, status, body)
224
- reason = case status
225
- when 200 then 'OK'
226
- when 404 then 'Not Found'
227
- when 405 then 'Method Not Allowed'
228
- when 503 then 'Service Unavailable'
229
- else 'Status'
230
- end
308
+ def json(...) = Health.json(...)
231
309
 
310
+ def write_response(client, status, body, type = JSON_TYPE)
311
+ reason = REASONS.fetch(status, 'Status')
232
312
  client.write(
233
313
  "HTTP/1.1 #{status} #{reason}\r\n" \
234
- "Content-Type: application/json\r\n" \
314
+ "Content-Type: #{type}\r\n" \
235
315
  "Content-Length: #{body.bytesize}\r\n" \
236
316
  "Connection: close\r\n\r\n" \
237
317
  "#{body}"
@@ -242,5 +322,67 @@ module Wurk
242
322
  @config&.logger
243
323
  end
244
324
  end
325
+
326
+ # The swarm-wide half of the listener: the /metrics body and the fleet
327
+ # check behind /ready. One Prometheus collector serves both, so a scrape and
328
+ # a probe landing in the same second cost one Redis refresh between them.
329
+ #
330
+ # A swarm pod is one probe target however many children it runs, and the
331
+ # listener lives in whichever child won the port. Judging readiness on that
332
+ # child alone said "ready" while its siblings crash-looped. Ready now also
333
+ # needs `min_ready` children (default: half the fleet, rounded up) with a
334
+ # heartbeat inside `ready_window` — a majority-ish bar rather than all, so
335
+ # one slot in respawn backoff does not pull a pod that is still doing most
336
+ # of its work out of a rollout. /live stays local on purpose: failing it
337
+ # restarts the whole pod, and replacing one wedged child is the swarm
338
+ # parent's job (Swarm::Liveness).
339
+ class Fleet
340
+ def initialize(config, ready_window)
341
+ @config = config
342
+ @ready_window = ready_window
343
+ @collector = nil
344
+ end
345
+
346
+ # [body, status, content type]; nil when `metrics: false`.
347
+ def metrics_response
348
+ return nil if option(:metrics) == false
349
+
350
+ body = collector.render(expected_children: Health.fleet_size, fresh_window: @ready_window)
351
+ [body, 200, Metrics::Prometheus::CONTENT_TYPE]
352
+ end
353
+
354
+ # The /ready answer once the local checks (Redis, own heartbeat) passed.
355
+ def ready_response
356
+ expected = Health.fleet_size
357
+ return [Health.json('ok', check: 'ready'), 200] unless expected
358
+
359
+ needed = min_ready(expected)
360
+ fresh = collector.fresh_local(collector.snapshot, @ready_window).size
361
+ if fresh >= needed
362
+ [Health.json('ok', check: 'ready', children: fresh, expected: expected), 200]
363
+ else
364
+ [Health.json('down', check: 'ready', reason: 'too few live children', children: fresh,
365
+ needed: needed, expected: expected), 503]
366
+ end
367
+ end
368
+
369
+ private
370
+
371
+ def collector
372
+ @collector ||= Metrics::Prometheus.new(@config)
373
+ end
374
+
375
+ def min_ready(expected)
376
+ configured = option(:min_ready)
377
+ return [Integer(configured), expected].min if configured
378
+
379
+ (expected / 2.0).ceil
380
+ end
381
+
382
+ def option(key)
383
+ opts = @config.respond_to?(:[]) ? @config[:health_check_options] : nil
384
+ opts.is_a?(Hash) ? opts[key] : nil
385
+ end
386
+ end
245
387
  end
246
388
  end
@@ -22,7 +22,7 @@ module Wurk
22
22
  # EXPIRE <identity>:work 60 (only if WORK_STATE non-empty)
23
23
  # EVALSHA refresh_slots <held slots> (only if QueueSlot::HELD non-empty)
24
24
  # then the signal drain:
25
- # LPOP <identity>-signals × BEAT_PAUSE
25
+ # RPOP <identity>-signals × BEAT_PAUSE
26
26
  # Two rather than one because only the first is safe to replay — see
27
27
  # #pipelined_beat and #drain_signals.
28
28
  #
@@ -44,13 +44,31 @@ module Wurk
44
44
  # Cadence in seconds. Key TTL is 60s — a process is dead after ~6 misses.
45
45
  BEAT_PAUSE = 10
46
46
  TTL_SECONDS = 60
47
+ FALLBACK_PAGE_SIZE = 4096
48
+
49
+ # statm counts pages, and a page is not 4 KB everywhere: arm64 kernels
50
+ # commonly run 16 KB or 64 KB pages, where a hard-coded ×4 under-reports
51
+ # RSS by 4–16×.
52
+ def self.page_size(etc = Etc)
53
+ size = etc.sysconf(Etc::SC_PAGESIZE)
54
+ size.is_a?(Integer) && size.positive? ? size : FALLBACK_PAGE_SIZE
55
+ # NotImplementedError (a ScriptError) is what Etc raises without sysconf.
56
+ rescue StandardError, NotImplementedError
57
+ FALLBACK_PAGE_SIZE
58
+ end
59
+
60
+ PAGE_SIZE = page_size
61
+
62
+ # Resident KB from a /proc/<pid>/statm line (field 1 is resident pages).
63
+ # Shared with Swarm, which reads its children's statm to recycle bloat.
64
+ def self.statm_rss_kb(statm)
65
+ statm.split[1].to_i * PAGE_SIZE / 1024
66
+ end
47
67
 
48
68
  attr_reader :identity, :rtt_us, :last_beat_at
49
69
 
50
70
  # `quiet:` is a callable so Launcher can keep ownership of its `@done`
51
- # flag without an awkward setter contract. `info_overrides:` lets the
52
- # caller (Launcher#embedded, tests) inject fields without forcing
53
- # Heartbeat to know about every flag the host process tracks.
71
+ # flag without an awkward setter contract.
54
72
  def initialize(identity:, config:, started_at: nil, embedded: false, quiet: nil)
55
73
  @identity = identity
56
74
  @config = config
@@ -153,19 +171,24 @@ module Wurk
153
171
  pipe.call('EXPIRE', work_key, TTL_SECONDS)
154
172
  end
155
173
 
156
- # Its own checkout, and never an idempotent one: LPOP is destructive and
174
+ # Its own checkout, and never an idempotent one: RPOP is destructive and
157
175
  # carries its result in the reply, so a replay after a lost reply discards
158
176
  # whatever the first attempt already popped — and a discarded entry is a
159
- # dashboard TERM or TSTP this process never acts on. Fused into the beat
160
- # pipeline it would have dragged those writes down to the same no-replay
161
- # default, or worse, invited a later sweep to claim the LPOPs alongside
162
- # them. Runs after the beat so a signal only leaves Redis once the write
177
+ # dashboard TERM or TSTP this process never acts on. The pool never
178
+ # replays it; redis-client's own single re-send on a dropped socket
179
+ # (RedisPool::DEFAULT_RECONNECT_ATTEMPTS) still can, and an entry popped by
180
+ # the lost first attempt is a signal this process never acts on — the same
181
+ # exposure Sidekiq has. Fused into the beat pipeline it would have dragged
182
+ # those writes down to the same no-replay default, or worse, invited a
183
+ # later sweep to claim the RPOPs alongside them. Runs after the beat so a signal only leaves Redis once the write
163
184
  # that reports us alive has landed.
164
185
  #
165
- # LPOP one entry per second of cadence so a flood of queued signals
166
- # can't stall the beat; anything older drains on the next beat.
186
+ # RPOP because ProcessSet::Process#signal LPUSHes: signals run in the
187
+ # order they were sent (Sidekiq's launcher pops the same end). One entry
188
+ # per second of cadence so a flood of queued signals can't stall the beat;
189
+ # anything newer drains on the next beat.
167
190
  def drain_signals
168
- redis { |conn| conn.pipelined { |pipe| BEAT_PAUSE.times { pipe.call('LPOP', "#{@identity}-signals") } } }
191
+ redis { |conn| conn.pipelined { |pipe| BEAT_PAUSE.times { pipe.call('RPOP', "#{@identity}-signals") } } }
169
192
  end
170
193
 
171
194
  def info_hash
@@ -270,12 +293,12 @@ module Wurk
270
293
  end
271
294
  end
272
295
 
273
- # Linux first via /proc/self/statm[1] (resident pages × 4 KB);
296
+ # Linux first via /proc/self/statm[1] (resident pages × page size);
274
297
  # `ps` fallback for macOS/BSD test runners. Zero on failure — the
275
298
  # dashboard shows "—" rather than crashing.
276
299
  def memory_usage_kb
277
300
  if ::File.exist?('/proc/self/statm')
278
- ::File.read('/proc/self/statm').split[1].to_i * 4
301
+ Heartbeat.statm_rss_kb(::File.read('/proc/self/statm'))
279
302
  else
280
303
  `ps -o rss= -p #{::Process.pid}`.to_i
281
304
  end
data/lib/wurk/history.rb CHANGED
@@ -157,13 +157,19 @@ module Wurk
157
157
  end
158
158
 
159
159
  # §5.2 default gauge set carries the `sidekiq.` prefix so a dashboard built
160
- # for Sidekiq Ent reads it unchanged. A custom collector replaces it.
160
+ # for Sidekiq Ent reads it unchanged, plus the per-queue size/latency
161
+ # gauges tagged `queue:<name>`. A custom collector replaces it all.
161
162
  def emit_statsd(values)
162
163
  client = Wurk::Metrics::Statsd.client
163
164
  return if client.nil?
164
165
  return @collector.call(client) if @collector
165
166
 
166
167
  values.each { |field, value| client.gauge("sidekiq.#{field}", value) }
168
+ Wurk::Stats.new.queue_summaries.each do |q|
169
+ tags = ["queue:#{q.name}"]
170
+ client.gauge('sidekiq.queue.size', q.size, tags: tags)
171
+ client.gauge('sidekiq.queue.latency', q.latency, tags: tags)
172
+ end
167
173
  end
168
174
  end
169
175
  end