pgbus 0.14.1 → 0.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +35 -0
  3. data/README.md +33 -0
  4. data/Rakefile +6 -1
  5. data/app/helpers/pgbus/application_helper.rb +7 -0
  6. data/app/models/pgbus/batch_entry.rb +78 -6
  7. data/app/models/pgbus/batch_execution.rb +28 -0
  8. data/app/models/pgbus/blocked_execution.rb +2 -2
  9. data/app/models/pgbus/uniqueness_key.rb +50 -4
  10. data/app/views/pgbus/batches/_batches_table.html.erb +3 -3
  11. data/app/views/pgbus/batches/show.html.erb +7 -7
  12. data/app/views/pgbus/dead_letter/show.html.erb +17 -0
  13. data/app/views/pgbus/events/_pending_table.html.erb +21 -0
  14. data/app/views/pgbus/jobs/show.html.erb +17 -0
  15. data/config/locales/da.yml +5 -2
  16. data/config/locales/de.yml +5 -2
  17. data/config/locales/en.yml +5 -2
  18. data/config/locales/es.yml +5 -2
  19. data/config/locales/fi.yml +5 -2
  20. data/config/locales/fr.yml +5 -2
  21. data/config/locales/it.yml +5 -2
  22. data/config/locales/ja.yml +5 -2
  23. data/config/locales/nb.yml +5 -2
  24. data/config/locales/nl.yml +5 -2
  25. data/config/locales/pt.yml +5 -2
  26. data/config/locales/sv.yml +5 -2
  27. data/lib/generators/pgbus/add_batch_callback_jobs_generator.rb +47 -0
  28. data/lib/generators/pgbus/add_batch_executions_generator.rb +44 -0
  29. data/lib/generators/pgbus/templates/add_batch_callback_jobs.rb.erb +13 -0
  30. data/lib/generators/pgbus/templates/add_batch_executions.rb.erb +50 -0
  31. data/lib/generators/pgbus/templates/initializer.rb.erb +2 -0
  32. data/lib/generators/pgbus/templates/migration.rb.erb +23 -2
  33. data/lib/pgbus/active_job/adapter.rb +146 -28
  34. data/lib/pgbus/active_job/batch_id.rb +48 -0
  35. data/lib/pgbus/active_job/current_attributes.rb +52 -0
  36. data/lib/pgbus/active_job/executor.rb +45 -9
  37. data/lib/pgbus/batch/sweep.rb +163 -0
  38. data/lib/pgbus/batch.rb +448 -63
  39. data/lib/pgbus/client/fair_read.rb +187 -0
  40. data/lib/pgbus/client.rb +221 -18
  41. data/lib/pgbus/concurrency/blocked_execution.rb +14 -1
  42. data/lib/pgbus/configuration.rb +81 -1
  43. data/lib/pgbus/current_attributes.rb +156 -0
  44. data/lib/pgbus/engine.rb +2 -0
  45. data/lib/pgbus/event.rb +7 -2
  46. data/lib/pgbus/event_bus/handler.rb +12 -2
  47. data/lib/pgbus/event_bus/publisher.rb +37 -2
  48. data/lib/pgbus/event_bus/subscriber.rb +4 -0
  49. data/lib/pgbus/fair_share.rb +110 -0
  50. data/lib/pgbus/generators/migration_detector.rb +30 -0
  51. data/lib/pgbus/instrumentation.rb +5 -0
  52. data/lib/pgbus/outbox/poller.rb +6 -7
  53. data/lib/pgbus/outbox.rb +8 -1
  54. data/lib/pgbus/process/consumer.rb +38 -1
  55. data/lib/pgbus/process/dispatcher.rb +41 -17
  56. data/lib/pgbus/process/worker.rb +41 -0
  57. data/lib/pgbus/recurring/schedule.rb +12 -1
  58. data/lib/pgbus/testing.rb +2 -1
  59. data/lib/pgbus/uniqueness.rb +35 -7
  60. data/lib/pgbus/version.rb +1 -1
  61. data/lib/pgbus/web/data_source.rb +47 -9
  62. data/lib/pgbus/web/job_context.rb +80 -0
  63. data/lib/pgbus.rb +2 -0
  64. metadata +13 -1
@@ -10,9 +10,13 @@ module Pgbus
10
10
  payload_hash = Serializer.serialize_job_hash(active_job)
11
11
  payload_hash = Concurrency.inject_metadata(active_job, payload_hash)
12
12
  payload_hash = Uniqueness.inject_metadata(active_job, payload_hash)
13
- payload_hash = inject_batch_metadata(payload_hash)
13
+ payload_hash = FairShare.inject_metadata(active_job, payload_hash)
14
+ payload_hash = inject_batch_metadata(payload_hash, active_job: active_job)
14
15
 
15
- return active_job if uniqueness_rejected?(active_job, payload_hash)
16
+ if uniqueness_rejected?(active_job, payload_hash, queue: queue)
17
+ uncount_batch_job(payload_hash)
18
+ return active_job
19
+ end
16
20
 
17
21
  enqueue_with_concurrency(active_job, queue, payload_hash)
18
22
  end
@@ -22,18 +26,23 @@ module Pgbus
22
26
  payload_hash = Serializer.serialize_job_hash(active_job)
23
27
  payload_hash = Concurrency.inject_metadata(active_job, payload_hash)
24
28
  payload_hash = Uniqueness.inject_metadata(active_job, payload_hash)
25
- payload_hash = inject_batch_metadata(payload_hash)
29
+ payload_hash = FairShare.inject_metadata(active_job, payload_hash)
30
+ payload_hash = inject_batch_metadata(payload_hash, active_job: active_job)
26
31
  delay = [(timestamp - Time.current.to_f).ceil, 0].max
27
32
 
28
- return active_job if uniqueness_rejected?(active_job, payload_hash)
33
+ if uniqueness_rejected?(active_job, payload_hash, queue: queue)
34
+ uncount_batch_job(payload_hash)
35
+ return active_job
36
+ end
29
37
 
30
38
  enqueue_with_concurrency(active_job, queue, payload_hash, delay: delay)
31
39
  end
32
40
 
33
41
  def enqueue_all(active_jobs)
34
- # Jobs with uniqueness must go through individual enqueue to acquire locks
35
- unique, bulk = active_jobs.partition { |j| Uniqueness.uniqueness_config(j) }
36
- unique.each do |j|
42
+ # Jobs with uniqueness or concurrency must go through individual enqueue
43
+ # to acquire locks/semaphores — the bulk path cannot (issue #413)
44
+ individual, bulk = active_jobs.partition { |j| Uniqueness.uniqueness_config(j) || concurrency_config(j) }
45
+ individual.each do |j|
37
46
  if scheduled_in_future?(j)
38
47
  enqueue_at(j, j.scheduled_at.to_f)
39
48
  else
@@ -41,9 +50,12 @@ module Pgbus
41
50
  end
42
51
  end
43
52
 
44
- bulk.group_by { |j| j.queue_name || Pgbus.configuration.default_queue }.each do |queue, jobs|
53
+ # Group by priority too: send_batch routes through the queue strategy,
54
+ # so a mixed-priority bulk send needs one produce_batch per level.
55
+ bulk.group_by { |j| [j.queue_name || Pgbus.configuration.default_queue, j.try(:priority)] }
56
+ .each do |(queue, priority), jobs|
45
57
  immediate, scheduled = jobs.partition { |j| !scheduled_in_future?(j) }
46
- enqueue_immediate(queue, immediate)
58
+ enqueue_immediate(queue, immediate, priority: priority)
47
59
  scheduled.each { |j| enqueue_at(j, j.scheduled_at.to_f) }
48
60
  end
49
61
 
@@ -56,6 +68,8 @@ module Pgbus
56
68
  key = Concurrency.extract_key(payload_hash)
57
69
  concurrency = concurrency_config(active_job)
58
70
  priority = active_job.try(:priority)
71
+ msg_id = nil
72
+ blocked = false
59
73
 
60
74
  if key && concurrency
61
75
  result = Concurrency::Semaphore.acquire(key, concurrency[:limit], concurrency[:duration])
@@ -64,34 +78,48 @@ module Pgbus
64
78
  msg_id = Pgbus.client.send_message(queue, payload_hash, delay: delay, priority: priority)
65
79
  active_job.provider_job_id = msg_id
66
80
  else
67
- handle_conflict(concurrency, active_job, key, queue, payload_hash, priority: priority)
81
+ blocked = handle_conflict(concurrency, active_job, key, queue, payload_hash, priority: priority)
68
82
  end
69
83
  else
70
84
  msg_id = Pgbus.client.send_message(queue, payload_hash, delay: delay, priority: priority)
71
85
  active_job.provider_job_id = msg_id
72
86
  end
73
87
 
88
+ # Bind before backfill so a live message is never left with an unbound
89
+ # uniqueness row if execution-row bookkeeping raises.
90
+ bind_acquired_uniqueness_lock(queue, msg_id) if msg_id
91
+ Batch.backfill_execution(payload_hash, msg_id, physical_queue(queue, priority)) if msg_id
92
+ # A retry re-enqueue that is now live (sent, or parked as a blocked
93
+ # execution) must stop the original attempt from signalling completion.
94
+ Batch.note_retry_reenqueued(payload_hash["job_id"]) if (msg_id || blocked) && retry_retagged?(payload_hash)
95
+ uniqueness_key = Thread.current[:pgbus_acquired_uniqueness_key]
96
+ UniquenessKey.clear_bind_stamp!(uniqueness_key) if uniqueness_key
74
97
  Thread.current[:pgbus_acquired_uniqueness_key] = nil
75
98
  active_job
76
99
  rescue StandardError => e
77
- # Roll back the uniqueness lock if enqueue failed
78
- rollback_key = Thread.current[:pgbus_acquired_uniqueness_key]
79
- if rollback_key
80
- begin
81
- Uniqueness.release_lock(rollback_key)
82
- rescue StandardError => rollback_error
83
- Pgbus.logger.warn { "[Pgbus] Lock rollback failed: #{rollback_error.message}" }
84
- end
100
+ if msg_id.nil?
101
+ rollback_acquired_uniqueness_lock
102
+ uncount_batch_job(payload_hash)
103
+ else
104
+ # Message is live: drop the thread-local so a later discard on this
105
+ # thread cannot release that job's uniqueness lock, but do not
106
+ # DELETE the pgbus_uniqueness_keys row.
85
107
  Thread.current[:pgbus_acquired_uniqueness_key] = nil
86
108
  end
87
109
  raise e
88
110
  end
89
111
 
112
+ def physical_queue(queue, priority)
113
+ Pgbus.client.target_queue(queue, priority)
114
+ end
115
+
90
116
  def concurrency_config(active_job)
91
117
  active_job.class.respond_to?(:pgbus_concurrency) && active_job.class.pgbus_concurrency
92
118
  end
93
119
 
94
- def handle_conflict(concurrency, active_job, key, queue, payload_hash, priority: nil)
120
+ # Returns true when the job was parked as a blocked execution (it will
121
+ # run later), false when it was dropped.
122
+ def handle_conflict(concurrency, active_job, key, queue, payload_hash, priority: nil) # rubocop:disable Naming/PredicateMethod
95
123
  case concurrency[:on_conflict]
96
124
  when :block
97
125
  Concurrency::BlockedExecution.insert(
@@ -101,14 +129,37 @@ module Pgbus
101
129
  priority: priority || Pgbus.configuration.default_priority,
102
130
  duration: concurrency[:duration]
103
131
  )
132
+ return true
104
133
  when :discard
105
134
  Pgbus.logger.info { "[Pgbus] Discarding job #{active_job.class.name}: concurrency limit for #{key}" }
135
+ # The job will never run: roll back an :until_executed uniqueness lock
136
+ # acquired earlier in this enqueue (no executor will release it), and
137
+ # uncount it from its batch so completion is not waited on forever.
138
+ rollback_acquired_uniqueness_lock
139
+ uncount_batch_job(payload_hash)
106
140
  when :raise
107
141
  raise ConcurrencyLimitExceeded, "Concurrency limit reached for key: #{key}"
108
142
  end
143
+ false
144
+ end
145
+
146
+ # Releases an :until_executed lock this enqueue acquired, if any.
147
+ # Used when the job is dropped before a message is sent (concurrency
148
+ # :discard, or send_message raising) — otherwise the lock is orphaned
149
+ # because no executor will ever release it.
150
+ def rollback_acquired_uniqueness_lock
151
+ rollback_key = Thread.current[:pgbus_acquired_uniqueness_key]
152
+ return unless rollback_key
153
+
154
+ begin
155
+ Uniqueness.release_lock(rollback_key)
156
+ rescue StandardError => e
157
+ Pgbus.logger.warn { "[Pgbus] Lock rollback failed: #{e.message}" }
158
+ end
159
+ Thread.current[:pgbus_acquired_uniqueness_key] = nil
109
160
  end
110
161
 
111
- def uniqueness_rejected?(active_job, payload_hash)
162
+ def uniqueness_rejected?(active_job, payload_hash, queue:)
112
163
  uniqueness_key = Uniqueness.extract_key(payload_hash)
113
164
  return false unless uniqueness_key
114
165
 
@@ -123,7 +174,7 @@ module Pgbus
123
174
  # See issue #333.
124
175
  return false if active_job.executions.to_i.positive?
125
176
 
126
- result = Uniqueness.acquire_enqueue_lock(uniqueness_key, active_job)
177
+ result = Uniqueness.acquire_enqueue_lock(uniqueness_key, active_job, queue_name: queue)
127
178
 
128
179
  # :no_lock means no enqueue-time lock needed (e.g. :while_executing strategy)
129
180
  return false if result == :no_lock
@@ -147,25 +198,92 @@ module Pgbus
147
198
  end
148
199
  end
149
200
 
150
- def inject_batch_metadata(payload_hash)
201
+ def bind_acquired_uniqueness_lock(queue, msg_id)
202
+ key = Thread.current[:pgbus_acquired_uniqueness_key]
203
+ return unless key
204
+
205
+ Uniqueness.bind_lock(key, queue_name: queue, msg_id: msg_id)
206
+ rescue StandardError => e
207
+ Pgbus.logger.warn { "[Pgbus] Uniqueness bind failed: #{e.message}" }
208
+ end
209
+
210
+ # Reverses inject_batch_metadata for a job discarded at enqueue time
211
+ # (uniqueness duplicate or concurrency :discard conflict): the message is
212
+ # never sent, so it can never signal completion. Only applies while the
213
+ # tagging batch's block is still active on this thread.
214
+ def uncount_batch_job(payload_hash)
215
+ batch_id = payload_hash[Batch::METADATA_KEY]
216
+ return unless batch_id
217
+
218
+ if batch_id == Thread.current[:pgbus_batch_id]
219
+ Batch.untrack_enqueue(payload_hash)
220
+ elsif retry_retagged?(payload_hash)
221
+ # The retry never became live; the original attempt's row and its
222
+ # normal completion signal stand.
223
+ Batch.forget_retry_reenqueued(payload_hash["job_id"])
224
+ end
225
+ end
226
+
227
+ # Tagged for a batch while no Batch#enqueue block is active on this
228
+ # thread — only a retry_on re-enqueue gets there (issue #424).
229
+ def retry_retagged?(payload_hash)
230
+ payload_hash[Batch::METADATA_KEY] && Thread.current[:pgbus_batch_id].nil?
231
+ end
232
+
233
+ # A retry_on re-enqueue of a job that is already a batch member: same
234
+ # job_id (executions > 0), batch_id carried by the BatchId mixin. It
235
+ # rejoins its batch without being counted again. A first-attempt job
236
+ # that merely has a batch_id outside a block is NOT tagged — membership
237
+ # stays explicit; only the callback_batch_id never re-tags.
238
+ def retry_batch_id_for(active_job)
239
+ return nil unless active_job.respond_to?(:batch_id)
240
+ return nil unless active_job.executions.to_i.positive?
241
+
242
+ active_job.batch_id
243
+ end
244
+
245
+ # Tag the payload with the active batch and count it in (guarded
246
+ # increment + execution row, before the send — Batch.track_enqueue). Pass
247
+ # track: false to only tag, when the caller counts a bulk once.
248
+ def inject_batch_metadata(payload_hash, active_job: nil, track: true)
151
249
  batch_id = Thread.current[:pgbus_batch_id]
152
- return payload_hash unless batch_id
250
+ if batch_id
251
+ tagged = payload_hash.merge(Batch::METADATA_KEY => batch_id)
252
+ Batch.track_enqueue(tagged) if track
253
+ return tagged
254
+ end
255
+
256
+ retry_batch_id = active_job && retry_batch_id_for(active_job)
257
+ return payload_hash unless retry_batch_id
153
258
 
154
- Thread.current[:pgbus_batch_job_count] = (Thread.current[:pgbus_batch_job_count] || 0) + 1
155
- payload_hash.merge(Batch::METADATA_KEY => batch_id)
259
+ tagged = payload_hash.merge(Batch::METADATA_KEY => retry_batch_id)
260
+ Batch.track_retry(tagged)
261
+ tagged
156
262
  end
157
263
 
158
- def enqueue_immediate(queue, jobs)
264
+ def enqueue_immediate(queue, jobs, priority: nil)
159
265
  return if jobs.empty?
160
266
 
161
- payloads = jobs.map { |j| Serializer.serialize_job_hash(j) }
162
- msg_ids = Pgbus.client.send_batch(queue, payloads)
267
+ payloads = jobs.map do |j|
268
+ inject_batch_metadata(FairShare.inject_metadata(j, Serializer.serialize_job_hash(j)), track: false)
269
+ end
270
+ # One guarded increment for the whole bulk, not one per job.
271
+ Batch.track_enqueue(payloads) if Thread.current[:pgbus_batch_id]
272
+ physical = physical_queue(queue, priority)
273
+ msg_ids = nil
274
+ msg_ids = Pgbus.client.send_batch(queue, payloads, priority: priority)
163
275
 
164
276
  unless msg_ids.is_a?(Array) && msg_ids.size == jobs.size
165
277
  raise Pgbus::EnqueueError, "Pgbus batch enqueue failed: expected #{jobs.size} ids, got #{msg_ids&.size || 0}"
166
278
  end
167
279
 
168
280
  jobs.zip(msg_ids).each { |job, id| job.provider_job_id = id }
281
+ payloads.zip(msg_ids).each { |payload, id| Batch.backfill_execution(payload, id, physical) }
282
+ rescue Pgbus::EnqueueError
283
+ Array(payloads).each_with_index do |payload, index|
284
+ Batch.untrack_enqueue(payload) if msg_ids.nil? || msg_ids[index].nil?
285
+ end
286
+ raise
169
287
  rescue Pgbus::SchemaNotReady => e
170
288
  Pgbus.logger.error { "[Pgbus] #{e.message}" }
171
289
  raise
@@ -0,0 +1,48 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Pgbus
4
+ module ActiveJob
5
+ # Gives every ActiveJob a handle on the batch it belongs to.
6
+ #
7
+ # Two distinct ids, mirroring solid_queue's ActiveJob::BatchId:
8
+ #
9
+ # * +batch_id+ — the batch this job is a *member* of. Assigned by the
10
+ # executor from the payload's +pgbus_batch_id+ before +perform+, so a
11
+ # running job can call +batch.enqueue+ to add siblings.
12
+ # * +callback_batch_id+ — the batch this job *reports on*. Set on
13
+ # +on_finish+/+on_success+/+on_failure+ jobs at fire time. A callback is
14
+ # never a member of the batch it reports on, so its +batch_id+ is nil.
15
+ #
16
+ # Both round-trip through +serialize+/+deserialize+, and both are omitted
17
+ # from the serialized hash when unset — an unbatched job's payload is
18
+ # byte-for-byte what it was before this mixin existed.
19
+ module BatchId
20
+ extend ActiveSupport::Concern
21
+
22
+ included do
23
+ attr_accessor :batch_id, :callback_batch_id
24
+ end
25
+
26
+ def serialize
27
+ data = super
28
+ data["batch_id"] = batch_id if batch_id
29
+ data["callback_batch_id"] = callback_batch_id if callback_batch_id
30
+ data
31
+ end
32
+
33
+ def deserialize(job_data)
34
+ super
35
+ self.batch_id = job_data["batch_id"]
36
+ self.callback_batch_id = job_data["callback_batch_id"]
37
+ end
38
+
39
+ # The batch this job reports on, or is a member of. nil when neither.
40
+ def batch
41
+ return @batch if defined?(@batch)
42
+
43
+ id = callback_batch_id || batch_id
44
+ @batch = id && Pgbus::Batch.find(id)
45
+ end
46
+ end
47
+ end
48
+ end
@@ -0,0 +1,52 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Pgbus
4
+ module ActiveJob
5
+ # Carries ActiveSupport::CurrentAttributes across enqueue → perform
6
+ # (issue #430). Included on ActiveJob::Base by the engine next to BatchId.
7
+ #
8
+ # * +serialize+ snapshots the persisted Current classes
9
+ # (Pgbus::CurrentAttributes.capture) under +pgbus_current+. A job that
10
+ # was itself deserialized re-serializes the context it was enqueued
11
+ # with — so a +retry_on+ re-enqueue keeps the original context even if
12
+ # Current changed during the attempt. Nothing is added when the feature
13
+ # is off or no attribute is assigned: the payload is byte-for-byte what
14
+ # it was before this mixin existed.
15
+ # * +perform_now+ is wrapped (not an +around_perform+) so the restored
16
+ # context also covers +rescue_from+ / +retry_on+ / +discard_on+ blocks,
17
+ # which run in +perform_now+'s rescue outside the perform callbacks —
18
+ # and it works identically under the pgbus worker, Rails' :test and
19
+ # :inline adapters, and a bare +job.perform_now+.
20
+ #
21
+ # Per-class control: +self.pgbus_persist_current_attributes = false+
22
+ # (never persist for this job class) or a spec in the same shapes as
23
+ # +config.current_attributes+ (an Array / Hash) to replace the config's
24
+ # list for this class. +nil+ (default) follows the config.
25
+ module CurrentAttributes
26
+ extend ActiveSupport::Concern
27
+
28
+ included do
29
+ attr_accessor :pgbus_current_attributes
30
+
31
+ class_attribute :pgbus_persist_current_attributes, instance_writer: false, default: nil
32
+ end
33
+
34
+ def serialize
35
+ data = super
36
+ captured = pgbus_current_attributes ||
37
+ Pgbus::CurrentAttributes.capture(override: self.class.pgbus_persist_current_attributes)
38
+ data[Pgbus::CurrentAttributes::METADATA_KEY] = captured if captured
39
+ data
40
+ end
41
+
42
+ def deserialize(job_data)
43
+ super
44
+ self.pgbus_current_attributes = job_data[Pgbus::CurrentAttributes::METADATA_KEY]
45
+ end
46
+
47
+ def perform_now
48
+ Pgbus::CurrentAttributes.restore(pgbus_current_attributes) { super }
49
+ end
50
+ end
51
+ end
52
+ end
@@ -56,9 +56,14 @@ module Pgbus
56
56
  # released on completion or DLQ.
57
57
  nil
58
58
  when :while_executing
59
- # Acquire the lock now. If another worker is already executing
60
- # this job, skip it — VT will expire and it'll be retried.
61
- unless Uniqueness.acquire_execution_lock(uniqueness_key, payload)
59
+ # Acquire the lock now, bound to this message. If another worker is
60
+ # already executing this job, skip it — VT will expire and it'll be
61
+ # retried. A row left by a crashed attempt of THIS message is
62
+ # re-acquired, not treated as a duplicate.
63
+ acquired = Uniqueness.acquire_execution_lock(
64
+ uniqueness_key, payload, msg_id: message.msg_id.to_i, queue_name: queue_name
65
+ )
66
+ unless acquired
62
67
  Pgbus.logger.info { "[Pgbus] Skipping duplicate execution for #{job_class}" }
63
68
  return :skipped
64
69
  end
@@ -67,6 +72,7 @@ module Pgbus
67
72
 
68
73
  Pgbus.logger.debug { "[Pgbus::Executor] deserialized #{tag} job_class=#{job_class}" }
69
74
  job_succeeded = false
75
+ retried = false
70
76
 
71
77
  msg_id = message.msg_id.to_i
72
78
  instrument_payload = {
@@ -85,13 +91,22 @@ module Pgbus
85
91
  # (issue #368). Pass this executor's config so an injected allowlist
86
92
  # is not silently ignored in favour of Pgbus.configuration.
87
93
  job = Serializer.deserialize_job_data(payload, configuration: config)
94
+ # Batch membership rides the pgbus metadata key, not the serialized
95
+ # job data, so hand it to the job before perform — that is what makes
96
+ # `batch` (and `batch.enqueue` for open batches) work inside a job.
97
+ assign_batch_id(job, payload)
88
98
  Pgbus.logger.debug { "[Pgbus::Executor] running #{tag} job_class=#{job_class}" }
89
99
  execute_job(job)
100
+ # retry_on re-enqueues from inside perform_now and returns normally:
101
+ # this attempt is done (archive it) but the job is not — the retry
102
+ # message carries the batch tag and signals on its own outcome.
103
+ retried = Batch.retry_reenqueued?(payload["job_id"])
90
104
  Pgbus.logger.debug { "[Pgbus::Executor] perform_returned #{tag} job_class=#{job_class}" }
91
105
  archive_from(queue_name, msg_id, source_queue: source_queue)
92
106
  Pgbus.logger.debug { "[Pgbus::Executor] archived #{tag} job_class=#{job_class}" }
93
- FailedEventRecorder.clear!(queue_name: queue_name, msg_id: msg_id)
94
107
  job_succeeded = true
108
+ release_uniqueness_lock(uniqueness_key)
109
+ FailedEventRecorder.clear!(queue_name: queue_name, msg_id: msg_id)
95
110
  end
96
111
 
97
112
  instrument("pgbus.job_completed", queue: queue_name, job_class: job_class)
@@ -108,6 +123,10 @@ module Pgbus
108
123
  # silently lost control flow — no failed event row, no job_failed
109
124
  # notification, uniqueness lock held until VT expired. See issue #126.
110
125
  handle_failure(message, queue_name, e, payload: payload)
126
+ # A failed :while_executing attempt is no longer executing — release
127
+ # so the retry (same message, after VT) can acquire. :until_executed
128
+ # keeps its lock until success/DLQ by design (#126, #333).
129
+ release_uniqueness_lock(uniqueness_key) if uniqueness_strategy == :while_executing
111
130
  instrument(
112
131
  "pgbus.job_failed",
113
132
  queue: queue_name,
@@ -129,15 +148,32 @@ module Pgbus
129
148
  # job_succeeded is set AFTER archive_message, so if archive fails the
130
149
  # semaphore slot stays held until VT expires and the job is retried.
131
150
  if job_succeeded
151
+ # Uniqueness is released once, immediately after archive. A second
152
+ # key-only DELETE here can drop a successor that acquired the same
153
+ # key if the first DELETE committed but the client raised afterward.
132
154
  signal_concurrency(payload)
133
- signal_batch_completed(payload)
134
- # Release uniqueness lock on successful completion (both strategies)
135
- Uniqueness.release_lock(uniqueness_key) if uniqueness_key
155
+ signal_batch_completed(payload) unless retried
136
156
  end
157
+ Batch.clear_retry_reenqueued
137
158
  end
138
159
 
139
160
  private
140
161
 
162
+ def assign_batch_id(job, payload)
163
+ batch_id = payload[Batch::METADATA_KEY]
164
+ return unless batch_id && job.respond_to?(:batch_id=)
165
+
166
+ job.batch_id = batch_id
167
+ end
168
+
169
+ def release_uniqueness_lock(key)
170
+ return unless key
171
+
172
+ Uniqueness.release_lock(key)
173
+ rescue StandardError => e
174
+ Pgbus.logger.warn { "[Pgbus] Uniqueness release failed: #{e.message}" }
175
+ end
176
+
141
177
  def execute_job(job)
142
178
  if defined?(Rails) && Rails.respond_to?(:application) && Rails.application
143
179
  wrapper = reloading? ? Rails.application.reloader : Rails.application.executor
@@ -286,7 +322,7 @@ module Pgbus
286
322
  batch_id = payload[Batch::METADATA_KEY]
287
323
  return unless batch_id
288
324
 
289
- Batch.job_completed(batch_id)
325
+ Batch.job_completed(batch_id, job_id: payload["job_id"])
290
326
  rescue StandardError => e
291
327
  Pgbus.logger.warn { "[Pgbus] Batch completion signal failed: #{e.message}" }
292
328
  end
@@ -295,7 +331,7 @@ module Pgbus
295
331
  batch_id = payload[Batch::METADATA_KEY]
296
332
  return unless batch_id
297
333
 
298
- Batch.job_discarded(batch_id)
334
+ Batch.job_discarded(batch_id, job_id: payload["job_id"])
299
335
  rescue StandardError => e
300
336
  Pgbus.logger.warn { "[Pgbus] Batch discard signal failed: #{e.message}" }
301
337
  end
@@ -0,0 +1,163 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Pgbus
4
+ class Batch
5
+ # Repairs batches the regular completion path cannot finish: worker crash
6
+ # between archive and row-delete, enqueue crash between row-insert and
7
+ # send, a pending batch whose enqueue block never returned, or a processing
8
+ # batch whose finish UPDATE rolled back after callbacks failed to enqueue.
9
+ module Sweep
10
+ STALL_THRESHOLD = 300 # seconds; solid_queue default stalled_for: 5.minutes
11
+
12
+ class << self
13
+ def run(stalled_for: Pgbus.configuration.batch_stall_threshold, batch_size: 500, client: Pgbus.client)
14
+ return unless Batch.executions_migrated?
15
+
16
+ payload = { stale_executions: 0, orphan_rows: 0, started_batches: 0, finished_batches: 0,
17
+ stalled_for: stalled_for }
18
+ Instrumentation.instrument("pgbus.batch_sweep", payload) do |p|
19
+ p[:stale_executions] = sweep_stale_executions(batch_size: batch_size, client: client, stalled_for: stalled_for)
20
+ p[:orphan_rows] = sweep_orphan_rows(stalled_for: stalled_for, batch_size: batch_size, client: client)
21
+ p[:started_batches] = start_stalled_pending(stalled_for: stalled_for, batch_size: batch_size)
22
+ p[:finished_batches] = finish_stalled_processing(batch_size: batch_size)
23
+ end
24
+ end
25
+
26
+ private
27
+
28
+ def sweep_stale_executions(batch_size:, client:, stalled_for:)
29
+ swept = 0
30
+ cutoff = Time.current - stalled_for
31
+ BatchExecution.where.not(msg_id: nil).where("created_at < ?", cutoff).find_each(batch_size: batch_size) do |row|
32
+ outcome = classify_stale(row, client)
33
+ next if outcome == :still_present
34
+
35
+ resolve_stale(row, outcome)
36
+ swept += 1
37
+ end
38
+ swept
39
+ end
40
+
41
+ def classify_stale(row, client)
42
+ in_queue = client.message_in_queue?(row.queue_name, msg_id: row.msg_id)
43
+ return :still_present if in_queue != false
44
+
45
+ dlq = client.dead_letter_physical_name(row.queue_name)
46
+ return :failed if client.message_with_job_id?(dlq, job_id: row.job_id) == true
47
+
48
+ return :completed if client.message_archived?(row.queue_name, msg_id: row.msg_id) == true
49
+
50
+ Pgbus.logger.warn do
51
+ "[Pgbus] Batch execution #{row.job_id} (batch #{row.batch_id}) has no queue, " \
52
+ "archive, or DLQ message — resolving as completed"
53
+ end
54
+ :completed
55
+ end
56
+
57
+ def resolve_stale(row, outcome)
58
+ column = outcome == :failed ? "failed_jobs" : "completed_jobs"
59
+ Batch.job_completed(row.batch_id, job_id: row.job_id) if column == "completed_jobs"
60
+ Batch.job_discarded(row.batch_id, job_id: row.job_id) if column == "failed_jobs"
61
+ end
62
+
63
+ # Rows with no msg_id older than the threshold. A row only stays
64
+ # msg_id-less when the enqueue died between row insert and send — OR
65
+ # when the send landed and only the backfill failed, in which case the
66
+ # message is live and must not be un-counted (issue #423). Probe the
67
+ # queue (and DLQ) by job_id before deciding; nil (unknown) keeps the row.
68
+ def sweep_orphan_rows(stalled_for:, batch_size:, client:)
69
+ swept = 0
70
+ cutoff = Time.current - stalled_for
71
+ blocked = blocked_job_ids
72
+ BatchExecution.where(msg_id: nil).where("created_at < ?", cutoff).find_each(batch_size: batch_size) do |row|
73
+ next if blocked.include?(row.job_id)
74
+
75
+ case classify_orphan(row, client)
76
+ when :live then next
77
+ when :failed
78
+ Batch.job_discarded(row.batch_id, job_id: row.job_id)
79
+ else
80
+ next unless uncount_orphan!(row)
81
+ end
82
+ swept += 1
83
+ end
84
+ swept
85
+ end
86
+
87
+ def classify_orphan(row, client)
88
+ return :live if row.queue_name.nil?
89
+
90
+ in_queue = client.message_with_job_id?(row.queue_name, job_id: row.job_id)
91
+ return :live if in_queue != false
92
+
93
+ dlq = client.dead_letter_physical_name(row.queue_name)
94
+ return :failed if client.message_with_job_id?(dlq, job_id: row.job_id) == true
95
+
96
+ :orphan
97
+ end
98
+
99
+ # Returns true when this sweep removed the row (CAS on msg_id NULL).
100
+ def uncount_orphan!(row) # rubocop:disable Naming/PredicateMethod
101
+ deleted = BatchExecution.where(id: row.id, msg_id: nil).delete_all
102
+ return false unless deleted.positive?
103
+
104
+ BatchEntry.decrement_total_jobs!(row.batch_id)
105
+ Batch.send(:finish_if_needed, Batch.try_finish!(row.batch_id))
106
+ true
107
+ end
108
+
109
+ def blocked_job_ids
110
+ return Set.new unless BlockedExecution.table_exists?
111
+
112
+ BlockedExecution.pluck(Arel.sql("payload->>'job_id'")).compact.to_set
113
+ rescue StandardError
114
+ Set.new
115
+ end
116
+
117
+ # A pending batch whose enqueue block never returned. Jobs counted
118
+ # themselves in as they were enqueued (issue #423), so total_jobs is
119
+ # already right — only the status moves. "Stalled" means no execution
120
+ # row was inserted within the threshold either: a long-running block
121
+ # that is still enqueuing is live, not stalled.
122
+ def start_stalled_pending(stalled_for:, batch_size:)
123
+ started = 0
124
+ cutoff = Time.current - stalled_for
125
+ BatchEntry.pending
126
+ .where("created_at < ?", cutoff)
127
+ .where(
128
+ "NOT EXISTS (SELECT 1 FROM pgbus_batch_executions e " \
129
+ "WHERE e.batch_id = pgbus_batches.batch_id AND e.created_at >= ?)", cutoff
130
+ )
131
+ .find_each(batch_size: batch_size) do |record|
132
+ BatchEntry.where(batch_id: record.batch_id, status: "pending").update_all(status: "processing")
133
+ Batch.send(:finish_if_needed, Batch.try_finish!(record.batch_id))
134
+ started += 1
135
+ end
136
+ started
137
+ end
138
+
139
+ def finish_stalled_processing(batch_size:)
140
+ finished = 0
141
+ BatchEntry.processing.without_executions.find_each(batch_size: batch_size) do |record|
142
+ # Legacy in-flight batches (migrated with an empty executions table)
143
+ # have zero rows but counters short of total_jobs — leave them on
144
+ # the counter path. The true stalls are counters already terminal
145
+ # (finish UPDATE rolled back after callback enqueue) and
146
+ # total_jobs = 0 (enqueue block crashed before enqueuing a job).
147
+ next unless counters_terminal?(record)
148
+
149
+ result = Batch.try_finish!(record.batch_id)
150
+ Batch.send(:finish_if_needed, result)
151
+ finished += 1 if result&.fetch(:just_finished, false)
152
+ end
153
+ finished
154
+ end
155
+
156
+ def counters_terminal?(record)
157
+ failures = record.discarded_jobs.to_i
158
+ (record.completed_jobs + failures) == record.total_jobs
159
+ end
160
+ end
161
+ end
162
+ end
163
+ end