pgbus 0.14.1 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +35 -0
- data/README.md +33 -0
- data/Rakefile +6 -1
- data/app/helpers/pgbus/application_helper.rb +7 -0
- data/app/models/pgbus/batch_entry.rb +78 -6
- data/app/models/pgbus/batch_execution.rb +28 -0
- data/app/models/pgbus/blocked_execution.rb +2 -2
- data/app/models/pgbus/uniqueness_key.rb +50 -4
- data/app/views/pgbus/batches/_batches_table.html.erb +3 -3
- data/app/views/pgbus/batches/show.html.erb +7 -7
- data/app/views/pgbus/dead_letter/show.html.erb +17 -0
- data/app/views/pgbus/events/_pending_table.html.erb +21 -0
- data/app/views/pgbus/jobs/show.html.erb +17 -0
- data/config/locales/da.yml +5 -2
- data/config/locales/de.yml +5 -2
- data/config/locales/en.yml +5 -2
- data/config/locales/es.yml +5 -2
- data/config/locales/fi.yml +5 -2
- data/config/locales/fr.yml +5 -2
- data/config/locales/it.yml +5 -2
- data/config/locales/ja.yml +5 -2
- data/config/locales/nb.yml +5 -2
- data/config/locales/nl.yml +5 -2
- data/config/locales/pt.yml +5 -2
- data/config/locales/sv.yml +5 -2
- data/lib/generators/pgbus/add_batch_callback_jobs_generator.rb +47 -0
- data/lib/generators/pgbus/add_batch_executions_generator.rb +44 -0
- data/lib/generators/pgbus/templates/add_batch_callback_jobs.rb.erb +13 -0
- data/lib/generators/pgbus/templates/add_batch_executions.rb.erb +50 -0
- data/lib/generators/pgbus/templates/initializer.rb.erb +2 -0
- data/lib/generators/pgbus/templates/migration.rb.erb +23 -2
- data/lib/pgbus/active_job/adapter.rb +146 -28
- data/lib/pgbus/active_job/batch_id.rb +48 -0
- data/lib/pgbus/active_job/current_attributes.rb +52 -0
- data/lib/pgbus/active_job/executor.rb +45 -9
- data/lib/pgbus/batch/sweep.rb +163 -0
- data/lib/pgbus/batch.rb +448 -63
- data/lib/pgbus/client/fair_read.rb +187 -0
- data/lib/pgbus/client.rb +221 -18
- data/lib/pgbus/concurrency/blocked_execution.rb +14 -1
- data/lib/pgbus/configuration.rb +81 -1
- data/lib/pgbus/current_attributes.rb +156 -0
- data/lib/pgbus/engine.rb +2 -0
- data/lib/pgbus/event.rb +7 -2
- data/lib/pgbus/event_bus/handler.rb +12 -2
- data/lib/pgbus/event_bus/publisher.rb +37 -2
- data/lib/pgbus/event_bus/subscriber.rb +4 -0
- data/lib/pgbus/fair_share.rb +110 -0
- data/lib/pgbus/generators/migration_detector.rb +30 -0
- data/lib/pgbus/instrumentation.rb +5 -0
- data/lib/pgbus/outbox/poller.rb +6 -7
- data/lib/pgbus/outbox.rb +8 -1
- data/lib/pgbus/process/consumer.rb +38 -1
- data/lib/pgbus/process/dispatcher.rb +41 -17
- data/lib/pgbus/process/worker.rb +41 -0
- data/lib/pgbus/recurring/schedule.rb +12 -1
- data/lib/pgbus/testing.rb +2 -1
- data/lib/pgbus/uniqueness.rb +35 -7
- data/lib/pgbus/version.rb +1 -1
- data/lib/pgbus/web/data_source.rb +47 -9
- data/lib/pgbus/web/job_context.rb +80 -0
- data/lib/pgbus.rb +2 -0
- metadata +13 -1
|
@@ -10,9 +10,13 @@ module Pgbus
|
|
|
10
10
|
payload_hash = Serializer.serialize_job_hash(active_job)
|
|
11
11
|
payload_hash = Concurrency.inject_metadata(active_job, payload_hash)
|
|
12
12
|
payload_hash = Uniqueness.inject_metadata(active_job, payload_hash)
|
|
13
|
-
payload_hash =
|
|
13
|
+
payload_hash = FairShare.inject_metadata(active_job, payload_hash)
|
|
14
|
+
payload_hash = inject_batch_metadata(payload_hash, active_job: active_job)
|
|
14
15
|
|
|
15
|
-
|
|
16
|
+
if uniqueness_rejected?(active_job, payload_hash, queue: queue)
|
|
17
|
+
uncount_batch_job(payload_hash)
|
|
18
|
+
return active_job
|
|
19
|
+
end
|
|
16
20
|
|
|
17
21
|
enqueue_with_concurrency(active_job, queue, payload_hash)
|
|
18
22
|
end
|
|
@@ -22,18 +26,23 @@ module Pgbus
|
|
|
22
26
|
payload_hash = Serializer.serialize_job_hash(active_job)
|
|
23
27
|
payload_hash = Concurrency.inject_metadata(active_job, payload_hash)
|
|
24
28
|
payload_hash = Uniqueness.inject_metadata(active_job, payload_hash)
|
|
25
|
-
payload_hash =
|
|
29
|
+
payload_hash = FairShare.inject_metadata(active_job, payload_hash)
|
|
30
|
+
payload_hash = inject_batch_metadata(payload_hash, active_job: active_job)
|
|
26
31
|
delay = [(timestamp - Time.current.to_f).ceil, 0].max
|
|
27
32
|
|
|
28
|
-
|
|
33
|
+
if uniqueness_rejected?(active_job, payload_hash, queue: queue)
|
|
34
|
+
uncount_batch_job(payload_hash)
|
|
35
|
+
return active_job
|
|
36
|
+
end
|
|
29
37
|
|
|
30
38
|
enqueue_with_concurrency(active_job, queue, payload_hash, delay: delay)
|
|
31
39
|
end
|
|
32
40
|
|
|
33
41
|
def enqueue_all(active_jobs)
|
|
34
|
-
# Jobs with uniqueness must go through individual enqueue
|
|
35
|
-
|
|
36
|
-
|
|
42
|
+
# Jobs with uniqueness or concurrency must go through individual enqueue
|
|
43
|
+
# to acquire locks/semaphores — the bulk path cannot (issue #413)
|
|
44
|
+
individual, bulk = active_jobs.partition { |j| Uniqueness.uniqueness_config(j) || concurrency_config(j) }
|
|
45
|
+
individual.each do |j|
|
|
37
46
|
if scheduled_in_future?(j)
|
|
38
47
|
enqueue_at(j, j.scheduled_at.to_f)
|
|
39
48
|
else
|
|
@@ -41,9 +50,12 @@ module Pgbus
|
|
|
41
50
|
end
|
|
42
51
|
end
|
|
43
52
|
|
|
44
|
-
|
|
53
|
+
# Group by priority too: send_batch routes through the queue strategy,
|
|
54
|
+
# so a mixed-priority bulk send needs one produce_batch per level.
|
|
55
|
+
bulk.group_by { |j| [j.queue_name || Pgbus.configuration.default_queue, j.try(:priority)] }
|
|
56
|
+
.each do |(queue, priority), jobs|
|
|
45
57
|
immediate, scheduled = jobs.partition { |j| !scheduled_in_future?(j) }
|
|
46
|
-
enqueue_immediate(queue, immediate)
|
|
58
|
+
enqueue_immediate(queue, immediate, priority: priority)
|
|
47
59
|
scheduled.each { |j| enqueue_at(j, j.scheduled_at.to_f) }
|
|
48
60
|
end
|
|
49
61
|
|
|
@@ -56,6 +68,8 @@ module Pgbus
|
|
|
56
68
|
key = Concurrency.extract_key(payload_hash)
|
|
57
69
|
concurrency = concurrency_config(active_job)
|
|
58
70
|
priority = active_job.try(:priority)
|
|
71
|
+
msg_id = nil
|
|
72
|
+
blocked = false
|
|
59
73
|
|
|
60
74
|
if key && concurrency
|
|
61
75
|
result = Concurrency::Semaphore.acquire(key, concurrency[:limit], concurrency[:duration])
|
|
@@ -64,34 +78,48 @@ module Pgbus
|
|
|
64
78
|
msg_id = Pgbus.client.send_message(queue, payload_hash, delay: delay, priority: priority)
|
|
65
79
|
active_job.provider_job_id = msg_id
|
|
66
80
|
else
|
|
67
|
-
handle_conflict(concurrency, active_job, key, queue, payload_hash, priority: priority)
|
|
81
|
+
blocked = handle_conflict(concurrency, active_job, key, queue, payload_hash, priority: priority)
|
|
68
82
|
end
|
|
69
83
|
else
|
|
70
84
|
msg_id = Pgbus.client.send_message(queue, payload_hash, delay: delay, priority: priority)
|
|
71
85
|
active_job.provider_job_id = msg_id
|
|
72
86
|
end
|
|
73
87
|
|
|
88
|
+
# Bind before backfill so a live message is never left with an unbound
|
|
89
|
+
# uniqueness row if execution-row bookkeeping raises.
|
|
90
|
+
bind_acquired_uniqueness_lock(queue, msg_id) if msg_id
|
|
91
|
+
Batch.backfill_execution(payload_hash, msg_id, physical_queue(queue, priority)) if msg_id
|
|
92
|
+
# A retry re-enqueue that is now live (sent, or parked as a blocked
|
|
93
|
+
# execution) must stop the original attempt from signalling completion.
|
|
94
|
+
Batch.note_retry_reenqueued(payload_hash["job_id"]) if (msg_id || blocked) && retry_retagged?(payload_hash)
|
|
95
|
+
uniqueness_key = Thread.current[:pgbus_acquired_uniqueness_key]
|
|
96
|
+
UniquenessKey.clear_bind_stamp!(uniqueness_key) if uniqueness_key
|
|
74
97
|
Thread.current[:pgbus_acquired_uniqueness_key] = nil
|
|
75
98
|
active_job
|
|
76
99
|
rescue StandardError => e
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
end
|
|
100
|
+
if msg_id.nil?
|
|
101
|
+
rollback_acquired_uniqueness_lock
|
|
102
|
+
uncount_batch_job(payload_hash)
|
|
103
|
+
else
|
|
104
|
+
# Message is live: drop the thread-local so a later discard on this
|
|
105
|
+
# thread cannot release that job's uniqueness lock, but do not
|
|
106
|
+
# DELETE the pgbus_uniqueness_keys row.
|
|
85
107
|
Thread.current[:pgbus_acquired_uniqueness_key] = nil
|
|
86
108
|
end
|
|
87
109
|
raise e
|
|
88
110
|
end
|
|
89
111
|
|
|
112
|
+
def physical_queue(queue, priority)
|
|
113
|
+
Pgbus.client.target_queue(queue, priority)
|
|
114
|
+
end
|
|
115
|
+
|
|
90
116
|
def concurrency_config(active_job)
|
|
91
117
|
active_job.class.respond_to?(:pgbus_concurrency) && active_job.class.pgbus_concurrency
|
|
92
118
|
end
|
|
93
119
|
|
|
94
|
-
|
|
120
|
+
# Returns true when the job was parked as a blocked execution (it will
|
|
121
|
+
# run later), false when it was dropped.
|
|
122
|
+
def handle_conflict(concurrency, active_job, key, queue, payload_hash, priority: nil) # rubocop:disable Naming/PredicateMethod
|
|
95
123
|
case concurrency[:on_conflict]
|
|
96
124
|
when :block
|
|
97
125
|
Concurrency::BlockedExecution.insert(
|
|
@@ -101,14 +129,37 @@ module Pgbus
|
|
|
101
129
|
priority: priority || Pgbus.configuration.default_priority,
|
|
102
130
|
duration: concurrency[:duration]
|
|
103
131
|
)
|
|
132
|
+
return true
|
|
104
133
|
when :discard
|
|
105
134
|
Pgbus.logger.info { "[Pgbus] Discarding job #{active_job.class.name}: concurrency limit for #{key}" }
|
|
135
|
+
# The job will never run: roll back an :until_executed uniqueness lock
|
|
136
|
+
# acquired earlier in this enqueue (no executor will release it), and
|
|
137
|
+
# uncount it from its batch so completion is not waited on forever.
|
|
138
|
+
rollback_acquired_uniqueness_lock
|
|
139
|
+
uncount_batch_job(payload_hash)
|
|
106
140
|
when :raise
|
|
107
141
|
raise ConcurrencyLimitExceeded, "Concurrency limit reached for key: #{key}"
|
|
108
142
|
end
|
|
143
|
+
false
|
|
144
|
+
end
|
|
145
|
+
|
|
146
|
+
# Releases an :until_executed lock this enqueue acquired, if any.
|
|
147
|
+
# Used when the job is dropped before a message is sent (concurrency
|
|
148
|
+
# :discard, or send_message raising) — otherwise the lock is orphaned
|
|
149
|
+
# because no executor will ever release it.
|
|
150
|
+
def rollback_acquired_uniqueness_lock
|
|
151
|
+
rollback_key = Thread.current[:pgbus_acquired_uniqueness_key]
|
|
152
|
+
return unless rollback_key
|
|
153
|
+
|
|
154
|
+
begin
|
|
155
|
+
Uniqueness.release_lock(rollback_key)
|
|
156
|
+
rescue StandardError => e
|
|
157
|
+
Pgbus.logger.warn { "[Pgbus] Lock rollback failed: #{e.message}" }
|
|
158
|
+
end
|
|
159
|
+
Thread.current[:pgbus_acquired_uniqueness_key] = nil
|
|
109
160
|
end
|
|
110
161
|
|
|
111
|
-
def uniqueness_rejected?(active_job, payload_hash)
|
|
162
|
+
def uniqueness_rejected?(active_job, payload_hash, queue:)
|
|
112
163
|
uniqueness_key = Uniqueness.extract_key(payload_hash)
|
|
113
164
|
return false unless uniqueness_key
|
|
114
165
|
|
|
@@ -123,7 +174,7 @@ module Pgbus
|
|
|
123
174
|
# See issue #333.
|
|
124
175
|
return false if active_job.executions.to_i.positive?
|
|
125
176
|
|
|
126
|
-
result = Uniqueness.acquire_enqueue_lock(uniqueness_key, active_job)
|
|
177
|
+
result = Uniqueness.acquire_enqueue_lock(uniqueness_key, active_job, queue_name: queue)
|
|
127
178
|
|
|
128
179
|
# :no_lock means no enqueue-time lock needed (e.g. :while_executing strategy)
|
|
129
180
|
return false if result == :no_lock
|
|
@@ -147,25 +198,92 @@ module Pgbus
|
|
|
147
198
|
end
|
|
148
199
|
end
|
|
149
200
|
|
|
150
|
-
def
|
|
201
|
+
def bind_acquired_uniqueness_lock(queue, msg_id)
|
|
202
|
+
key = Thread.current[:pgbus_acquired_uniqueness_key]
|
|
203
|
+
return unless key
|
|
204
|
+
|
|
205
|
+
Uniqueness.bind_lock(key, queue_name: queue, msg_id: msg_id)
|
|
206
|
+
rescue StandardError => e
|
|
207
|
+
Pgbus.logger.warn { "[Pgbus] Uniqueness bind failed: #{e.message}" }
|
|
208
|
+
end
|
|
209
|
+
|
|
210
|
+
# Reverses inject_batch_metadata for a job discarded at enqueue time
|
|
211
|
+
# (uniqueness duplicate or concurrency :discard conflict): the message is
|
|
212
|
+
# never sent, so it can never signal completion. Only applies while the
|
|
213
|
+
# tagging batch's block is still active on this thread.
|
|
214
|
+
def uncount_batch_job(payload_hash)
|
|
215
|
+
batch_id = payload_hash[Batch::METADATA_KEY]
|
|
216
|
+
return unless batch_id
|
|
217
|
+
|
|
218
|
+
if batch_id == Thread.current[:pgbus_batch_id]
|
|
219
|
+
Batch.untrack_enqueue(payload_hash)
|
|
220
|
+
elsif retry_retagged?(payload_hash)
|
|
221
|
+
# The retry never became live; the original attempt's row and its
|
|
222
|
+
# normal completion signal stand.
|
|
223
|
+
Batch.forget_retry_reenqueued(payload_hash["job_id"])
|
|
224
|
+
end
|
|
225
|
+
end
|
|
226
|
+
|
|
227
|
+
# Tagged for a batch while no Batch#enqueue block is active on this
|
|
228
|
+
# thread — only a retry_on re-enqueue gets there (issue #424).
|
|
229
|
+
def retry_retagged?(payload_hash)
|
|
230
|
+
payload_hash[Batch::METADATA_KEY] && Thread.current[:pgbus_batch_id].nil?
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
# A retry_on re-enqueue of a job that is already a batch member: same
|
|
234
|
+
# job_id (executions > 0), batch_id carried by the BatchId mixin. It
|
|
235
|
+
# rejoins its batch without being counted again. A first-attempt job
|
|
236
|
+
# that merely has a batch_id outside a block is NOT tagged — membership
|
|
237
|
+
# stays explicit; only the callback_batch_id never re-tags.
|
|
238
|
+
def retry_batch_id_for(active_job)
|
|
239
|
+
return nil unless active_job.respond_to?(:batch_id)
|
|
240
|
+
return nil unless active_job.executions.to_i.positive?
|
|
241
|
+
|
|
242
|
+
active_job.batch_id
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
# Tag the payload with the active batch and count it in (guarded
|
|
246
|
+
# increment + execution row, before the send — Batch.track_enqueue). Pass
|
|
247
|
+
# track: false to only tag, when the caller counts a bulk once.
|
|
248
|
+
def inject_batch_metadata(payload_hash, active_job: nil, track: true)
|
|
151
249
|
batch_id = Thread.current[:pgbus_batch_id]
|
|
152
|
-
|
|
250
|
+
if batch_id
|
|
251
|
+
tagged = payload_hash.merge(Batch::METADATA_KEY => batch_id)
|
|
252
|
+
Batch.track_enqueue(tagged) if track
|
|
253
|
+
return tagged
|
|
254
|
+
end
|
|
255
|
+
|
|
256
|
+
retry_batch_id = active_job && retry_batch_id_for(active_job)
|
|
257
|
+
return payload_hash unless retry_batch_id
|
|
153
258
|
|
|
154
|
-
|
|
155
|
-
|
|
259
|
+
tagged = payload_hash.merge(Batch::METADATA_KEY => retry_batch_id)
|
|
260
|
+
Batch.track_retry(tagged)
|
|
261
|
+
tagged
|
|
156
262
|
end
|
|
157
263
|
|
|
158
|
-
def enqueue_immediate(queue, jobs)
|
|
264
|
+
def enqueue_immediate(queue, jobs, priority: nil)
|
|
159
265
|
return if jobs.empty?
|
|
160
266
|
|
|
161
|
-
payloads = jobs.map
|
|
162
|
-
|
|
267
|
+
payloads = jobs.map do |j|
|
|
268
|
+
inject_batch_metadata(FairShare.inject_metadata(j, Serializer.serialize_job_hash(j)), track: false)
|
|
269
|
+
end
|
|
270
|
+
# One guarded increment for the whole bulk, not one per job.
|
|
271
|
+
Batch.track_enqueue(payloads) if Thread.current[:pgbus_batch_id]
|
|
272
|
+
physical = physical_queue(queue, priority)
|
|
273
|
+
msg_ids = nil
|
|
274
|
+
msg_ids = Pgbus.client.send_batch(queue, payloads, priority: priority)
|
|
163
275
|
|
|
164
276
|
unless msg_ids.is_a?(Array) && msg_ids.size == jobs.size
|
|
165
277
|
raise Pgbus::EnqueueError, "Pgbus batch enqueue failed: expected #{jobs.size} ids, got #{msg_ids&.size || 0}"
|
|
166
278
|
end
|
|
167
279
|
|
|
168
280
|
jobs.zip(msg_ids).each { |job, id| job.provider_job_id = id }
|
|
281
|
+
payloads.zip(msg_ids).each { |payload, id| Batch.backfill_execution(payload, id, physical) }
|
|
282
|
+
rescue Pgbus::EnqueueError
|
|
283
|
+
Array(payloads).each_with_index do |payload, index|
|
|
284
|
+
Batch.untrack_enqueue(payload) if msg_ids.nil? || msg_ids[index].nil?
|
|
285
|
+
end
|
|
286
|
+
raise
|
|
169
287
|
rescue Pgbus::SchemaNotReady => e
|
|
170
288
|
Pgbus.logger.error { "[Pgbus] #{e.message}" }
|
|
171
289
|
raise
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Pgbus
|
|
4
|
+
module ActiveJob
|
|
5
|
+
# Gives every ActiveJob a handle on the batch it belongs to.
|
|
6
|
+
#
|
|
7
|
+
# Two distinct ids, mirroring solid_queue's ActiveJob::BatchId:
|
|
8
|
+
#
|
|
9
|
+
# * +batch_id+ — the batch this job is a *member* of. Assigned by the
|
|
10
|
+
# executor from the payload's +pgbus_batch_id+ before +perform+, so a
|
|
11
|
+
# running job can call +batch.enqueue+ to add siblings.
|
|
12
|
+
# * +callback_batch_id+ — the batch this job *reports on*. Set on
|
|
13
|
+
# +on_finish+/+on_success+/+on_failure+ jobs at fire time. A callback is
|
|
14
|
+
# never a member of the batch it reports on, so its +batch_id+ is nil.
|
|
15
|
+
#
|
|
16
|
+
# Both round-trip through +serialize+/+deserialize+, and both are omitted
|
|
17
|
+
# from the serialized hash when unset — an unbatched job's payload is
|
|
18
|
+
# byte-for-byte what it was before this mixin existed.
|
|
19
|
+
module BatchId
|
|
20
|
+
extend ActiveSupport::Concern
|
|
21
|
+
|
|
22
|
+
included do
|
|
23
|
+
attr_accessor :batch_id, :callback_batch_id
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def serialize
|
|
27
|
+
data = super
|
|
28
|
+
data["batch_id"] = batch_id if batch_id
|
|
29
|
+
data["callback_batch_id"] = callback_batch_id if callback_batch_id
|
|
30
|
+
data
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def deserialize(job_data)
|
|
34
|
+
super
|
|
35
|
+
self.batch_id = job_data["batch_id"]
|
|
36
|
+
self.callback_batch_id = job_data["callback_batch_id"]
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# The batch this job reports on, or is a member of. nil when neither.
|
|
40
|
+
def batch
|
|
41
|
+
return @batch if defined?(@batch)
|
|
42
|
+
|
|
43
|
+
id = callback_batch_id || batch_id
|
|
44
|
+
@batch = id && Pgbus::Batch.find(id)
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
end
|
|
48
|
+
end
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Pgbus
|
|
4
|
+
module ActiveJob
|
|
5
|
+
# Carries ActiveSupport::CurrentAttributes across enqueue → perform
|
|
6
|
+
# (issue #430). Included on ActiveJob::Base by the engine next to BatchId.
|
|
7
|
+
#
|
|
8
|
+
# * +serialize+ snapshots the persisted Current classes
|
|
9
|
+
# (Pgbus::CurrentAttributes.capture) under +pgbus_current+. A job that
|
|
10
|
+
# was itself deserialized re-serializes the context it was enqueued
|
|
11
|
+
# with — so a +retry_on+ re-enqueue keeps the original context even if
|
|
12
|
+
# Current changed during the attempt. Nothing is added when the feature
|
|
13
|
+
# is off or no attribute is assigned: the payload is byte-for-byte what
|
|
14
|
+
# it was before this mixin existed.
|
|
15
|
+
# * +perform_now+ is wrapped (not an +around_perform+) so the restored
|
|
16
|
+
# context also covers +rescue_from+ / +retry_on+ / +discard_on+ blocks,
|
|
17
|
+
# which run in +perform_now+'s rescue outside the perform callbacks —
|
|
18
|
+
# and it works identically under the pgbus worker, Rails' :test and
|
|
19
|
+
# :inline adapters, and a bare +job.perform_now+.
|
|
20
|
+
#
|
|
21
|
+
# Per-class control: +self.pgbus_persist_current_attributes = false+
|
|
22
|
+
# (never persist for this job class) or a spec in the same shapes as
|
|
23
|
+
# +config.current_attributes+ (an Array / Hash) to replace the config's
|
|
24
|
+
# list for this class. +nil+ (default) follows the config.
|
|
25
|
+
module CurrentAttributes
|
|
26
|
+
extend ActiveSupport::Concern
|
|
27
|
+
|
|
28
|
+
included do
|
|
29
|
+
attr_accessor :pgbus_current_attributes
|
|
30
|
+
|
|
31
|
+
class_attribute :pgbus_persist_current_attributes, instance_writer: false, default: nil
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def serialize
|
|
35
|
+
data = super
|
|
36
|
+
captured = pgbus_current_attributes ||
|
|
37
|
+
Pgbus::CurrentAttributes.capture(override: self.class.pgbus_persist_current_attributes)
|
|
38
|
+
data[Pgbus::CurrentAttributes::METADATA_KEY] = captured if captured
|
|
39
|
+
data
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
def deserialize(job_data)
|
|
43
|
+
super
|
|
44
|
+
self.pgbus_current_attributes = job_data[Pgbus::CurrentAttributes::METADATA_KEY]
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
def perform_now
|
|
48
|
+
Pgbus::CurrentAttributes.restore(pgbus_current_attributes) { super }
|
|
49
|
+
end
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
end
|
|
@@ -56,9 +56,14 @@ module Pgbus
|
|
|
56
56
|
# released on completion or DLQ.
|
|
57
57
|
nil
|
|
58
58
|
when :while_executing
|
|
59
|
-
# Acquire the lock now. If another worker is
|
|
60
|
-
# this job, skip it — VT will expire and it'll be
|
|
61
|
-
|
|
59
|
+
# Acquire the lock now, bound to this message. If another worker is
|
|
60
|
+
# already executing this job, skip it — VT will expire and it'll be
|
|
61
|
+
# retried. A row left by a crashed attempt of THIS message is
|
|
62
|
+
# re-acquired, not treated as a duplicate.
|
|
63
|
+
acquired = Uniqueness.acquire_execution_lock(
|
|
64
|
+
uniqueness_key, payload, msg_id: message.msg_id.to_i, queue_name: queue_name
|
|
65
|
+
)
|
|
66
|
+
unless acquired
|
|
62
67
|
Pgbus.logger.info { "[Pgbus] Skipping duplicate execution for #{job_class}" }
|
|
63
68
|
return :skipped
|
|
64
69
|
end
|
|
@@ -67,6 +72,7 @@ module Pgbus
|
|
|
67
72
|
|
|
68
73
|
Pgbus.logger.debug { "[Pgbus::Executor] deserialized #{tag} job_class=#{job_class}" }
|
|
69
74
|
job_succeeded = false
|
|
75
|
+
retried = false
|
|
70
76
|
|
|
71
77
|
msg_id = message.msg_id.to_i
|
|
72
78
|
instrument_payload = {
|
|
@@ -85,13 +91,22 @@ module Pgbus
|
|
|
85
91
|
# (issue #368). Pass this executor's config so an injected allowlist
|
|
86
92
|
# is not silently ignored in favour of Pgbus.configuration.
|
|
87
93
|
job = Serializer.deserialize_job_data(payload, configuration: config)
|
|
94
|
+
# Batch membership rides the pgbus metadata key, not the serialized
|
|
95
|
+
# job data, so hand it to the job before perform — that is what makes
|
|
96
|
+
# `batch` (and `batch.enqueue` for open batches) work inside a job.
|
|
97
|
+
assign_batch_id(job, payload)
|
|
88
98
|
Pgbus.logger.debug { "[Pgbus::Executor] running #{tag} job_class=#{job_class}" }
|
|
89
99
|
execute_job(job)
|
|
100
|
+
# retry_on re-enqueues from inside perform_now and returns normally:
|
|
101
|
+
# this attempt is done (archive it) but the job is not — the retry
|
|
102
|
+
# message carries the batch tag and signals on its own outcome.
|
|
103
|
+
retried = Batch.retry_reenqueued?(payload["job_id"])
|
|
90
104
|
Pgbus.logger.debug { "[Pgbus::Executor] perform_returned #{tag} job_class=#{job_class}" }
|
|
91
105
|
archive_from(queue_name, msg_id, source_queue: source_queue)
|
|
92
106
|
Pgbus.logger.debug { "[Pgbus::Executor] archived #{tag} job_class=#{job_class}" }
|
|
93
|
-
FailedEventRecorder.clear!(queue_name: queue_name, msg_id: msg_id)
|
|
94
107
|
job_succeeded = true
|
|
108
|
+
release_uniqueness_lock(uniqueness_key)
|
|
109
|
+
FailedEventRecorder.clear!(queue_name: queue_name, msg_id: msg_id)
|
|
95
110
|
end
|
|
96
111
|
|
|
97
112
|
instrument("pgbus.job_completed", queue: queue_name, job_class: job_class)
|
|
@@ -108,6 +123,10 @@ module Pgbus
|
|
|
108
123
|
# silently lost control flow — no failed event row, no job_failed
|
|
109
124
|
# notification, uniqueness lock held until VT expired. See issue #126.
|
|
110
125
|
handle_failure(message, queue_name, e, payload: payload)
|
|
126
|
+
# A failed :while_executing attempt is no longer executing — release
|
|
127
|
+
# so the retry (same message, after VT) can acquire. :until_executed
|
|
128
|
+
# keeps its lock until success/DLQ by design (#126, #333).
|
|
129
|
+
release_uniqueness_lock(uniqueness_key) if uniqueness_strategy == :while_executing
|
|
111
130
|
instrument(
|
|
112
131
|
"pgbus.job_failed",
|
|
113
132
|
queue: queue_name,
|
|
@@ -129,15 +148,32 @@ module Pgbus
|
|
|
129
148
|
# job_succeeded is set AFTER archive_message, so if archive fails the
|
|
130
149
|
# semaphore slot stays held until VT expires and the job is retried.
|
|
131
150
|
if job_succeeded
|
|
151
|
+
# Uniqueness is released once, immediately after archive. A second
|
|
152
|
+
# key-only DELETE here can drop a successor that acquired the same
|
|
153
|
+
# key if the first DELETE committed but the client raised afterward.
|
|
132
154
|
signal_concurrency(payload)
|
|
133
|
-
signal_batch_completed(payload)
|
|
134
|
-
# Release uniqueness lock on successful completion (both strategies)
|
|
135
|
-
Uniqueness.release_lock(uniqueness_key) if uniqueness_key
|
|
155
|
+
signal_batch_completed(payload) unless retried
|
|
136
156
|
end
|
|
157
|
+
Batch.clear_retry_reenqueued
|
|
137
158
|
end
|
|
138
159
|
|
|
139
160
|
private
|
|
140
161
|
|
|
162
|
+
def assign_batch_id(job, payload)
|
|
163
|
+
batch_id = payload[Batch::METADATA_KEY]
|
|
164
|
+
return unless batch_id && job.respond_to?(:batch_id=)
|
|
165
|
+
|
|
166
|
+
job.batch_id = batch_id
|
|
167
|
+
end
|
|
168
|
+
|
|
169
|
+
def release_uniqueness_lock(key)
|
|
170
|
+
return unless key
|
|
171
|
+
|
|
172
|
+
Uniqueness.release_lock(key)
|
|
173
|
+
rescue StandardError => e
|
|
174
|
+
Pgbus.logger.warn { "[Pgbus] Uniqueness release failed: #{e.message}" }
|
|
175
|
+
end
|
|
176
|
+
|
|
141
177
|
def execute_job(job)
|
|
142
178
|
if defined?(Rails) && Rails.respond_to?(:application) && Rails.application
|
|
143
179
|
wrapper = reloading? ? Rails.application.reloader : Rails.application.executor
|
|
@@ -286,7 +322,7 @@ module Pgbus
|
|
|
286
322
|
batch_id = payload[Batch::METADATA_KEY]
|
|
287
323
|
return unless batch_id
|
|
288
324
|
|
|
289
|
-
Batch.job_completed(batch_id)
|
|
325
|
+
Batch.job_completed(batch_id, job_id: payload["job_id"])
|
|
290
326
|
rescue StandardError => e
|
|
291
327
|
Pgbus.logger.warn { "[Pgbus] Batch completion signal failed: #{e.message}" }
|
|
292
328
|
end
|
|
@@ -295,7 +331,7 @@ module Pgbus
|
|
|
295
331
|
batch_id = payload[Batch::METADATA_KEY]
|
|
296
332
|
return unless batch_id
|
|
297
333
|
|
|
298
|
-
Batch.job_discarded(batch_id)
|
|
334
|
+
Batch.job_discarded(batch_id, job_id: payload["job_id"])
|
|
299
335
|
rescue StandardError => e
|
|
300
336
|
Pgbus.logger.warn { "[Pgbus] Batch discard signal failed: #{e.message}" }
|
|
301
337
|
end
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Pgbus
|
|
4
|
+
class Batch
|
|
5
|
+
# Repairs batches the regular completion path cannot finish: worker crash
|
|
6
|
+
# between archive and row-delete, enqueue crash between row-insert and
|
|
7
|
+
# send, a pending batch whose enqueue block never returned, or a processing
|
|
8
|
+
# batch whose finish UPDATE rolled back after callbacks failed to enqueue.
|
|
9
|
+
module Sweep
|
|
10
|
+
STALL_THRESHOLD = 300 # seconds; solid_queue default stalled_for: 5.minutes
|
|
11
|
+
|
|
12
|
+
class << self
|
|
13
|
+
def run(stalled_for: Pgbus.configuration.batch_stall_threshold, batch_size: 500, client: Pgbus.client)
|
|
14
|
+
return unless Batch.executions_migrated?
|
|
15
|
+
|
|
16
|
+
payload = { stale_executions: 0, orphan_rows: 0, started_batches: 0, finished_batches: 0,
|
|
17
|
+
stalled_for: stalled_for }
|
|
18
|
+
Instrumentation.instrument("pgbus.batch_sweep", payload) do |p|
|
|
19
|
+
p[:stale_executions] = sweep_stale_executions(batch_size: batch_size, client: client, stalled_for: stalled_for)
|
|
20
|
+
p[:orphan_rows] = sweep_orphan_rows(stalled_for: stalled_for, batch_size: batch_size, client: client)
|
|
21
|
+
p[:started_batches] = start_stalled_pending(stalled_for: stalled_for, batch_size: batch_size)
|
|
22
|
+
p[:finished_batches] = finish_stalled_processing(batch_size: batch_size)
|
|
23
|
+
end
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
private
|
|
27
|
+
|
|
28
|
+
def sweep_stale_executions(batch_size:, client:, stalled_for:)
|
|
29
|
+
swept = 0
|
|
30
|
+
cutoff = Time.current - stalled_for
|
|
31
|
+
BatchExecution.where.not(msg_id: nil).where("created_at < ?", cutoff).find_each(batch_size: batch_size) do |row|
|
|
32
|
+
outcome = classify_stale(row, client)
|
|
33
|
+
next if outcome == :still_present
|
|
34
|
+
|
|
35
|
+
resolve_stale(row, outcome)
|
|
36
|
+
swept += 1
|
|
37
|
+
end
|
|
38
|
+
swept
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
def classify_stale(row, client)
|
|
42
|
+
in_queue = client.message_in_queue?(row.queue_name, msg_id: row.msg_id)
|
|
43
|
+
return :still_present if in_queue != false
|
|
44
|
+
|
|
45
|
+
dlq = client.dead_letter_physical_name(row.queue_name)
|
|
46
|
+
return :failed if client.message_with_job_id?(dlq, job_id: row.job_id) == true
|
|
47
|
+
|
|
48
|
+
return :completed if client.message_archived?(row.queue_name, msg_id: row.msg_id) == true
|
|
49
|
+
|
|
50
|
+
Pgbus.logger.warn do
|
|
51
|
+
"[Pgbus] Batch execution #{row.job_id} (batch #{row.batch_id}) has no queue, " \
|
|
52
|
+
"archive, or DLQ message — resolving as completed"
|
|
53
|
+
end
|
|
54
|
+
:completed
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
def resolve_stale(row, outcome)
|
|
58
|
+
column = outcome == :failed ? "failed_jobs" : "completed_jobs"
|
|
59
|
+
Batch.job_completed(row.batch_id, job_id: row.job_id) if column == "completed_jobs"
|
|
60
|
+
Batch.job_discarded(row.batch_id, job_id: row.job_id) if column == "failed_jobs"
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# Rows with no msg_id older than the threshold. A row only stays
|
|
64
|
+
# msg_id-less when the enqueue died between row insert and send — OR
|
|
65
|
+
# when the send landed and only the backfill failed, in which case the
|
|
66
|
+
# message is live and must not be un-counted (issue #423). Probe the
|
|
67
|
+
# queue (and DLQ) by job_id before deciding; nil (unknown) keeps the row.
|
|
68
|
+
def sweep_orphan_rows(stalled_for:, batch_size:, client:)
|
|
69
|
+
swept = 0
|
|
70
|
+
cutoff = Time.current - stalled_for
|
|
71
|
+
blocked = blocked_job_ids
|
|
72
|
+
BatchExecution.where(msg_id: nil).where("created_at < ?", cutoff).find_each(batch_size: batch_size) do |row|
|
|
73
|
+
next if blocked.include?(row.job_id)
|
|
74
|
+
|
|
75
|
+
case classify_orphan(row, client)
|
|
76
|
+
when :live then next
|
|
77
|
+
when :failed
|
|
78
|
+
Batch.job_discarded(row.batch_id, job_id: row.job_id)
|
|
79
|
+
else
|
|
80
|
+
next unless uncount_orphan!(row)
|
|
81
|
+
end
|
|
82
|
+
swept += 1
|
|
83
|
+
end
|
|
84
|
+
swept
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
def classify_orphan(row, client)
|
|
88
|
+
return :live if row.queue_name.nil?
|
|
89
|
+
|
|
90
|
+
in_queue = client.message_with_job_id?(row.queue_name, job_id: row.job_id)
|
|
91
|
+
return :live if in_queue != false
|
|
92
|
+
|
|
93
|
+
dlq = client.dead_letter_physical_name(row.queue_name)
|
|
94
|
+
return :failed if client.message_with_job_id?(dlq, job_id: row.job_id) == true
|
|
95
|
+
|
|
96
|
+
:orphan
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
# Returns true when this sweep removed the row (CAS on msg_id NULL).
|
|
100
|
+
def uncount_orphan!(row) # rubocop:disable Naming/PredicateMethod
|
|
101
|
+
deleted = BatchExecution.where(id: row.id, msg_id: nil).delete_all
|
|
102
|
+
return false unless deleted.positive?
|
|
103
|
+
|
|
104
|
+
BatchEntry.decrement_total_jobs!(row.batch_id)
|
|
105
|
+
Batch.send(:finish_if_needed, Batch.try_finish!(row.batch_id))
|
|
106
|
+
true
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def blocked_job_ids
|
|
110
|
+
return Set.new unless BlockedExecution.table_exists?
|
|
111
|
+
|
|
112
|
+
BlockedExecution.pluck(Arel.sql("payload->>'job_id'")).compact.to_set
|
|
113
|
+
rescue StandardError
|
|
114
|
+
Set.new
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
# A pending batch whose enqueue block never returned. Jobs counted
|
|
118
|
+
# themselves in as they were enqueued (issue #423), so total_jobs is
|
|
119
|
+
# already right — only the status moves. "Stalled" means no execution
|
|
120
|
+
# row was inserted within the threshold either: a long-running block
|
|
121
|
+
# that is still enqueuing is live, not stalled.
|
|
122
|
+
def start_stalled_pending(stalled_for:, batch_size:)
|
|
123
|
+
started = 0
|
|
124
|
+
cutoff = Time.current - stalled_for
|
|
125
|
+
BatchEntry.pending
|
|
126
|
+
.where("created_at < ?", cutoff)
|
|
127
|
+
.where(
|
|
128
|
+
"NOT EXISTS (SELECT 1 FROM pgbus_batch_executions e " \
|
|
129
|
+
"WHERE e.batch_id = pgbus_batches.batch_id AND e.created_at >= ?)", cutoff
|
|
130
|
+
)
|
|
131
|
+
.find_each(batch_size: batch_size) do |record|
|
|
132
|
+
BatchEntry.where(batch_id: record.batch_id, status: "pending").update_all(status: "processing")
|
|
133
|
+
Batch.send(:finish_if_needed, Batch.try_finish!(record.batch_id))
|
|
134
|
+
started += 1
|
|
135
|
+
end
|
|
136
|
+
started
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
def finish_stalled_processing(batch_size:)
|
|
140
|
+
finished = 0
|
|
141
|
+
BatchEntry.processing.without_executions.find_each(batch_size: batch_size) do |record|
|
|
142
|
+
# Legacy in-flight batches (migrated with an empty executions table)
|
|
143
|
+
# have zero rows but counters short of total_jobs — leave them on
|
|
144
|
+
# the counter path. The true stalls are counters already terminal
|
|
145
|
+
# (finish UPDATE rolled back after callback enqueue) and
|
|
146
|
+
# total_jobs = 0 (enqueue block crashed before enqueuing a job).
|
|
147
|
+
next unless counters_terminal?(record)
|
|
148
|
+
|
|
149
|
+
result = Batch.try_finish!(record.batch_id)
|
|
150
|
+
Batch.send(:finish_if_needed, result)
|
|
151
|
+
finished += 1 if result&.fetch(:just_finished, false)
|
|
152
|
+
end
|
|
153
|
+
finished
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
def counters_terminal?(record)
|
|
157
|
+
failures = record.discarded_jobs.to_i
|
|
158
|
+
(record.completed_jobs + failures) == record.total_jobs
|
|
159
|
+
end
|
|
160
|
+
end
|
|
161
|
+
end
|
|
162
|
+
end
|
|
163
|
+
end
|