workhorse 1.5.2 → 2.0.0.rc1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. checksums.yaml +4 -4
  2. data/.github/workflows/ruby.yml +137 -1
  3. data/CHANGELOG.md +150 -0
  4. data/Gemfile +16 -1
  5. data/README.md +316 -72
  6. data/Rakefile +1 -0
  7. data/VERSION +1 -1
  8. data/bin/rubocop +5 -1
  9. data/lib/generators/workhorse/install_generator.rb +10 -1
  10. data/lib/generators/workhorse/templates/config/initializers/workhorse.rb +55 -0
  11. data/lib/generators/workhorse/templates/create_table_jobs.rb +15 -2
  12. data/lib/generators/workhorse/templates/create_table_workhorse_schedules.rb +42 -0
  13. data/lib/workhorse/daemon/shell_handler.rb +4 -1
  14. data/lib/workhorse/daemon.rb +57 -7
  15. data/lib/workhorse/db_job.rb +98 -8
  16. data/lib/workhorse/enqueuer.rb +51 -8
  17. data/lib/workhorse/jobs/cleanup_succeeded_jobs.rb +26 -8
  18. data/lib/workhorse/jobs/detect_late_schedules_job.rb +59 -0
  19. data/lib/workhorse/notifiers/base.rb +55 -0
  20. data/lib/workhorse/notifiers/file_system.rb +66 -0
  21. data/lib/workhorse/notifiers/none.rb +8 -0
  22. data/lib/workhorse/notifiers/redis.rb +227 -0
  23. data/lib/workhorse/performer.rb +29 -2
  24. data/lib/workhorse/poller.rb +303 -21
  25. data/lib/workhorse/pool.rb +12 -6
  26. data/lib/workhorse/schedule.rb +288 -0
  27. data/lib/workhorse/schedules.rb +197 -0
  28. data/lib/workhorse/worker.rb +102 -31
  29. data/lib/workhorse.rb +136 -0
  30. data/test/lib/db_schema.rb +36 -3
  31. data/test/lib/jobs.rb +29 -0
  32. data/test/lib/test_helper.rb +113 -20
  33. data/test/workhorse/daemon_test.rb +33 -0
  34. data/test/workhorse/db_job_test.rb +2 -4
  35. data/test/workhorse/notifier_test.rb +487 -0
  36. data/test/workhorse/performer_test.rb +7 -9
  37. data/test/workhorse/poller_test.rb +97 -23
  38. data/test/workhorse/schedule_test.rb +967 -0
  39. data/test/workhorse/worker_test.rb +201 -76
  40. data/workhorse.gemspec +6 -5
  41. metadata +29 -3
@@ -19,25 +19,25 @@ class Workhorse::PollerTest < WorkhorseTest
19
19
  Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: :q2
20
20
  Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: :q2
21
21
 
22
- assert_equal %w[q1 q2], w.poller.send(:valid_queues)
22
+ assert_equal %w[q1 q2], valid_queues_of(w)
23
23
 
24
24
  first_job = Workhorse::DbJob.first
25
25
  first_job.mark_locked!(42)
26
26
 
27
- assert_equal %w[q2], w.poller.send(:valid_queues)
27
+ assert_equal %w[q2], valid_queues_of(w)
28
28
 
29
29
  first_job.mark_started!
30
30
 
31
- assert_equal %w[q2], w.poller.send(:valid_queues)
31
+ assert_equal %w[q2], valid_queues_of(w)
32
32
 
33
33
  first_job.mark_succeeded!
34
34
 
35
- assert_equal %w[q1 q2], w.poller.send(:valid_queues)
35
+ assert_equal %w[q1 q2], valid_queues_of(w)
36
36
 
37
37
  last_job = Workhorse::DbJob.last
38
38
  last_job.mark_locked!(42)
39
39
 
40
- assert_equal %w[q1], w.poller.send(:valid_queues)
40
+ assert_equal %w[q1], valid_queues_of(w)
41
41
 
42
42
  begin
43
43
  fail 'Some exception'
@@ -45,30 +45,30 @@ class Workhorse::PollerTest < WorkhorseTest
45
45
  last_job.mark_failed!(e)
46
46
  end
47
47
 
48
- assert_equal %w[q1 q2], w.poller.send(:valid_queues)
48
+ assert_equal %w[q1 q2], valid_queues_of(w)
49
49
  end
50
50
 
51
51
  def test_valid_queues_2
52
52
  w = Workhorse::Worker.new(polling_interval: 60)
53
53
 
54
- assert_equal [], w.poller.send(:valid_queues)
54
+ assert_equal [], valid_queues_of(w)
55
55
 
56
56
  Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: nil
57
57
 
58
- assert_equal [nil], w.poller.send(:valid_queues)
58
+ assert_equal [nil], valid_queues_of(w)
59
59
 
60
60
  a_job = Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: :a
61
61
 
62
- assert_equal [nil, 'a'], w.poller.send(:valid_queues)
62
+ assert_equal [nil, 'a'], valid_queues_of(w)
63
63
 
64
64
  a_job.update_attribute :state, :locked
65
65
 
66
- assert_equal [nil], w.poller.send(:valid_queues)
66
+ assert_equal [nil], valid_queues_of(w)
67
67
  end
68
68
 
69
69
  def test_no_queues
70
70
  w = Workhorse::Worker.new(polling_interval: 60)
71
- assert_equal [], w.poller.send(:valid_queues)
71
+ assert_equal [], valid_queues_of(w)
72
72
  end
73
73
 
74
74
  # Not every adapter returns a result set from `execute`: the Oracle enhanced
@@ -82,7 +82,7 @@ class Workhorse::PollerTest < WorkhorseTest
82
82
  Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: :a
83
83
 
84
84
  with_return_value(Workhorse::DbJob.connection, :execute, true) do
85
- assert_equal [nil, 'a'], w.poller.send(:valid_queues)
85
+ assert_equal [nil, 'a'], valid_queues_of(w)
86
86
  end
87
87
  end
88
88
 
@@ -94,11 +94,11 @@ class Workhorse::PollerTest < WorkhorseTest
94
94
  end
95
95
  jobs = Workhorse::DbJob.all
96
96
 
97
- assert_equal [nil], w.poller.send(:valid_queues)
97
+ assert_equal [nil], valid_queues_of(w)
98
98
 
99
99
  jobs[0].mark_locked!(42)
100
100
 
101
- assert_equal [nil], w.poller.send(:valid_queues)
101
+ assert_equal [nil], valid_queues_of(w)
102
102
  end
103
103
 
104
104
  def test_with_instant_repolling
@@ -135,10 +135,43 @@ class Workhorse::PollerTest < WorkhorseTest
135
135
  Workhorse.enqueue BasicJob.new(some_param: i, sleep_time: 0)
136
136
  end
137
137
 
138
- # Create 10 worker processes that work for 3s each
138
+ # Create 10 worker processes that work until all 100 jobs have succeeded,
139
+ # and for at least 3s, so that each is still polling while the second
140
+ # batch comes in. Not for a fixed time: every poll holds the global lock
141
+ # throughout, so the workers take turns, and 3s got through as few as 25
142
+ # of the 100 jobs on a loaded CI runner, and 61 on Oracle. What is tested
143
+ # is that none is taken twice and every worker gets a turn, not how fast.
139
144
  10.times do
140
145
  Process.fork do
141
- work 3, pool_size: 1, polling_interval: 0.1
146
+ started = Time.now
147
+
148
+ # A worker that cannot even be shut down would leave waitall below
149
+ # waiting for good, so it reports where it is stuck and gives up.
150
+ Thread.new do
151
+ sleep 90
152
+ report_threads 'Worker process still running after 90s'
153
+ exit!(1)
154
+ end
155
+
156
+ with_worker(pool_size: 1, polling_interval: 0.1, auto_terminate: false) do
157
+ loop do
158
+ elapsed = Time.now - started
159
+ break if elapsed >= 3 && Workhorse::DbJob.succeeded.count >= 100
160
+
161
+ if elapsed >= 60
162
+ report_threads 'Worker process gave up waiting for all jobs to succeed after 60s'
163
+ break
164
+ end
165
+
166
+ sleep 0.1
167
+ end
168
+ end
169
+ ensure
170
+ # Exit without running the at_exit handlers of the test process: one
171
+ # of them is Minitest's, which joins threads this fork did not
172
+ # inherit and hangs the child, leaving waitall below waiting forever.
173
+ # In an ensure, as a child that raised must not run them either.
174
+ exit!(0)
142
175
  end
143
176
  end
144
177
 
@@ -149,24 +182,51 @@ class Workhorse::PollerTest < WorkhorseTest
149
182
  Workhorse.enqueue BasicJob.new(sleep_time: 0)
150
183
  end
151
184
 
152
- # Wait for all forked processes to finish (should take ~3s)
185
+ # Wait for all forked processes to finish
153
186
  Process.waitall
154
187
 
155
188
  total = Workhorse::DbJob.count
156
189
  succeeded = Workhorse::DbJob.succeeded.count
157
- used_workers = Workhorse::DbJob.lock.pluck(:locked_by).uniq.size
190
+ # In a transaction, as Oracle commits a FOR UPDATE outside of one before
191
+ # its rows are fetched and then fails with "fetch out of sequence".
192
+ used_workers = Workhorse::DbJob.transaction do
193
+ Workhorse::DbJob.lock.pluck(:locked_by).uniq.size
194
+ end
158
195
 
159
196
  # Make sure there are 100 jobs, all jobs have succeeded and that all of the
160
- # workers have had their turn.
197
+ # workers have had their turn. The jobs that have not are listed, as all
198
+ # a count says is that something got stuck, not where.
199
+ unfinished = Workhorse::DbJob.where.not(state: 'succeeded').order(:id).map do |job|
200
+ "##{job.id} #{job.state}, locked_by #{job.locked_by.inspect}, " \
201
+ "locked_at #{job.locked_at.inspect}, started_at #{job.started_at.inspect}"
202
+ end
203
+
161
204
  assert_equal 100, total
162
- assert_equal 100, succeeded
205
+ assert_equal 100, succeeded, "Jobs that did not succeed:\n#{unfinished.join("\n")}"
163
206
  assert_equal 10, used_workers
164
207
  end
165
208
 
209
+ # A poll that finds the global lock taken waits at least as long as it asked
210
+ # to before giving up. Below half a second is what the poller asks for with
211
+ # a short polling interval, and what MySQL and Oracle, which take the timeout
212
+ # as whole seconds, used to round down to not waiting at all.
213
+ def test_contended_global_lock_waits_for_its_timeout
214
+ poller = Workhorse::Worker.new(polling_interval: 0.3).poller
215
+ ran = false
216
+
217
+ elapsed = with_global_lock_held do
218
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
219
+ poller.send(:with_global_lock, timeout: 0.3, count_failures: false) { ran = true }
220
+ Process.clock_gettime(Process::CLOCK_MONOTONIC) - started
221
+ end
222
+
223
+ assert_not ran
224
+ assert_operator elapsed, :>=, 0.25
225
+ end
226
+
166
227
  def test_connection_loss
167
- # rubocop: disable Style/GlobalVars
228
+ # rubocop: disable-next Style/GlobalVars
168
229
  $thread_conn = nil
169
- # rubocop: enable Style/GlobalVars
170
230
 
171
231
  Workhorse.enqueue BasicJob.new(sleep_time: 3)
172
232
 
@@ -199,7 +259,14 @@ class Workhorse::PollerTest < WorkhorseTest
199
259
  Workhorse.clean_stuck_jobs = clean
200
260
  with_daemon do
201
261
  Workhorse.enqueue BasicJob.new(sleep_time: 5)
202
- sleep 0.2
262
+
263
+ # Waited for rather than slept on: until a worker has taken the job,
264
+ # locked_by is empty and the cleanup, which matches on the host in it,
265
+ # cannot recognise the job as one of its own.
266
+ with_retries do
267
+ assert_equal 'started', Workhorse::DbJob.first.state
268
+ end
269
+
203
270
  kill_deamon_workers
204
271
 
205
272
  assert_equal 1, Workhorse::DbJob.count
@@ -265,6 +332,13 @@ class Workhorse::PollerTest < WorkhorseTest
265
332
 
266
333
  private
267
334
 
335
+ # The worker's valid queues, the queueless one first. Sorted here, as they
336
+ # come from a SELECT DISTINCT without ORDER BY: MySQL happens to return NULL
337
+ # first, Oracle does not, and the poller does not depend on the order.
338
+ def valid_queues_of(worker)
339
+ return worker.poller.send(:valid_queues).sort_by { |queue| [queue.nil? ? 0 : 1, queue.to_s] }
340
+ end
341
+
268
342
  def kill_deamon_workers
269
343
  pids = daemon.workers.map(&:pid)
270
344
  pids.each do |pid|