workhorse 2.0.0.rc0 → 2.0.0.rc1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -291,7 +291,7 @@ module Workhorse
291
291
  return unless daemon_id
292
292
 
293
293
  path = self.class.heartbeat_file_for(daemon_id)
294
- FileUtils.touch(path) if path
294
+ touch_pid_file(path) if path
295
295
  rescue StandardError => e
296
296
  Workhorse.debug_log("[Job worker #{id}] Heartbeat touch failed: #{e.class}: #{e.message}")
297
297
  end
@@ -363,7 +363,7 @@ module Workhorse
363
363
  Workhorse.debug_log("[Job worker #{id}] Memory limit exceeded: #{mem}MB > #{max}MB, initiating shutdown")
364
364
 
365
365
  if defined?(Rails)
366
- FileUtils.touch self.class.shutdown_file_for(pid)
366
+ touch_pid_file(self.class.shutdown_file_for(pid))
367
367
  end
368
368
 
369
369
  log "Worker process #{id.inspect} memory consumption (RSS) of #{mem}MB exceeds " \
@@ -444,8 +444,17 @@ module Workhorse
444
444
  # Create shutdown file for watch to detect
445
445
  shutdown_file = self.class.shutdown_file_for(pid)
446
446
  if shutdown_file
447
- FileUtils.touch(shutdown_file)
448
- Workhorse.debug_log("[Job worker #{id}] Shutdown file created: #{shutdown_file}")
447
+ begin
448
+ touch_pid_file(shutdown_file)
449
+ Workhorse.debug_log("[Job worker #{id}] Shutdown file created: #{shutdown_file}")
450
+ rescue StandardError => e
451
+ # The shutdown has to go ahead regardless: the worker stopped
452
+ # accepting jobs above, so giving up here would leave it taking none
453
+ # and never exiting. Without the file, watch sees a worker that is
454
+ # gone rather than one that asked to be restarted, and starts it all
455
+ # the same.
456
+ log "Could not create shutdown file #{shutdown_file}, shutting down regardless: #{e.message}", :warn
457
+ end
449
458
  end
450
459
 
451
460
  # Monitor in a separate thread to avoid blocking the signal handler
@@ -533,5 +542,17 @@ module Workhorse
533
542
  Kernel.sleep 0.2
534
543
  end
535
544
  end
545
+
546
+ # Touches a file in the pids directory, creating the directory first. It
547
+ # is not guaranteed to exist: `tmp` is rarely checked in, so a fresh
548
+ # checkout or deployment has none until something writes there.
549
+ #
550
+ # @param path [Pathname, String]
551
+ # @return [void]
552
+ # @private
553
+ def touch_pid_file(path)
554
+ FileUtils.mkdir_p(File.dirname(path))
555
+ FileUtils.touch(path)
556
+ end
536
557
  end
537
558
  end
@@ -1,10 +1,21 @@
1
1
  ActiveRecord::Schema.define do
2
2
  self.verbose = false
3
3
 
4
+ # Prefix lengths for the indexes on string columns. MySQL needs them, as the
5
+ # default `utf8mb4` charset puts a full `varchar(255)` past the maximum key
6
+ # length; Oracle indexes the whole column and rejects the option.
7
+ state_length = DB_ORACLE ? {} : { length: { state: 191 } }
8
+ queue_length = DB_ORACLE ? {} : { length: 191 }
9
+ key_length = DB_ORACLE ? {} : { length: 191 }
10
+
4
11
  create_table :jobs, force: true do |t|
5
12
  t.string :state, null: false, default: 'waiting'
6
13
  t.string :queue, null: true
7
- t.text :handler, null: false, limit: 4_294_967_295
14
+
15
+ # Binary rather than text, matching the generated migration: the handler is
16
+ # a `Marshal.dump`, and a text column is character data - on Oracle a CLOB,
17
+ # whose character set conversion would corrupt it.
18
+ t.binary :handler, null: false, limit: 4_294_967_295
8
19
 
9
20
  t.string :locked_by
10
21
  t.datetime :locked_at
@@ -25,10 +36,12 @@ ActiveRecord::Schema.define do
25
36
  t.timestamps null: false
26
37
  end
27
38
 
28
- add_index :jobs, :queue, length: 191
29
- add_index :jobs, %i[state perform_at], length: { state: 191 }, name: 'idx_jobs_state_perform_at'
30
- add_index :jobs, %i[state priority created_at], length: { state: 191 }, name: 'idx_jobs_state_prio_created'
31
- add_index :jobs, %i[state expires_at], length: { state: 191 }, name: 'idx_jobs_state_expires_at'
39
+ # The index names are given explicitly because the ones Rails would derive
40
+ # exceed the 30 characters Oracle allows before 12.2.
41
+ add_index :jobs, :queue, **queue_length
42
+ add_index :jobs, %i[state perform_at], name: 'idx_jobs_state_perform_at', **state_length
43
+ add_index :jobs, %i[state priority created_at], name: 'idx_jobs_state_prio_created', **state_length
44
+ add_index :jobs, %i[state expires_at], name: 'idx_jobs_state_expires_at', **state_length
32
45
  add_index :jobs, :perform_at
33
46
 
34
47
  create_table :workhorse_schedules, force: true do |t|
@@ -44,6 +57,6 @@ ActiveRecord::Schema.define do
44
57
  t.timestamps null: false
45
58
  end
46
59
 
47
- add_index :workhorse_schedules, :key, unique: true, length: 191, name: 'idx_wh_schedules_key'
60
+ add_index :workhorse_schedules, :key, unique: true, name: 'idx_wh_schedules_key', **key_length
48
61
  add_index :workhorse_schedules, %i[enabled next_at], name: 'idx_wh_schedules_due'
49
62
  end
@@ -6,8 +6,8 @@ require 'benchmark'
6
6
  require 'concurrent'
7
7
  require 'jobs'
8
8
 
9
- # The adapter to run the test suite against. Both MySQL / MariaDB adapters are
10
- # supported, as workhorse has to work with either of them.
9
+ # The adapter to run the test suite against. Workhorse has to work with each of
10
+ # them, so each is covered by CI.
11
11
  DB_ADAPTER = ENV.fetch('DB_ADAPTER', 'mysql2')
12
12
 
13
13
  case DB_ADAPTER
@@ -15,10 +15,16 @@ when 'mysql2'
15
15
  require 'mysql2'
16
16
  when 'trilogy'
17
17
  require 'trilogy'
18
+ when 'oracle_enhanced'
19
+ require 'active_record/connection_adapters/oracle_enhanced_adapter'
18
20
  else
19
- fail "Unsupported DB_ADAPTER #{DB_ADAPTER.inspect}, use 'mysql2' or 'trilogy'."
21
+ fail "Unsupported DB_ADAPTER #{DB_ADAPTER.inspect}, use 'mysql2', 'trilogy' or 'oracle_enhanced'."
20
22
  end
21
23
 
24
+ # Whether the suite is running against Oracle. Used where the schema or an
25
+ # assertion cannot be written the same way for both families.
26
+ DB_ORACLE = DB_ADAPTER == 'oracle_enhanced'
27
+
22
28
  class MockRailsEnv < String
23
29
  def production?
24
30
  self == 'production'
@@ -44,6 +50,38 @@ class Rails
44
50
  end
45
51
 
46
52
  class WorkhorseTest < ActiveSupport::TestCase
53
+ # Seconds a test may run before the backtraces of all its threads are
54
+ # printed. A hung test otherwise only shows up as a CI attempt killed at its
55
+ # time limit, with nothing saying where it hung.
56
+ HANG_REPORT_AFTER = 120
57
+
58
+ # Callbacks rather than methods, so that a test class defining its own setup
59
+ # or teardown does not skip them.
60
+ setup do
61
+ test = "#{self.class}##{name}"
62
+
63
+ @hang_watchdog = Thread.new do
64
+ sleep HANG_REPORT_AFTER
65
+ report_threads "#{test} is still running after #{HANG_REPORT_AFTER}s"
66
+ end
67
+ end
68
+
69
+ # Prints the backtraces of all threads of this process but the calling one.
70
+ def report_threads(title)
71
+ warn "#{title} (PID #{Process.pid}). Its threads:"
72
+
73
+ Thread.list.each do |thread|
74
+ next if thread == Thread.current
75
+
76
+ warn "--- #{thread.inspect}\n#{(thread.backtrace || ['(no backtrace)']).join("\n")}"
77
+ end
78
+ end
79
+
80
+ teardown do
81
+ @hang_watchdog&.kill
82
+ restore_termination_traps
83
+ end
84
+
47
85
  def setup
48
86
  remove_pids!
49
87
  clear_locks_and_db_threads!
@@ -68,6 +106,36 @@ class WorkhorseTest < ActiveSupport::TestCase
68
106
  object.singleton_class.send(:remove_method, method)
69
107
  end
70
108
 
109
+ # Holds workhorse's global lock on a connection of its own, so that the code
110
+ # under test sees it as taken by another worker.
111
+ def with_global_lock_held
112
+ connection = ActiveRecord::Base.connection_pool.checkout
113
+ connection.select_value(acquire_global_lock_sql)
114
+ yield
115
+ ensure
116
+ if connection
117
+ connection.select_value(release_global_lock_sql)
118
+ ActiveRecord::Base.connection_pool.checkin(connection)
119
+ end
120
+ end
121
+
122
+ # Mirrors what Workhorse::Poller#with_global_lock emits, so that the lock the
123
+ # code under test tries to take is the same one.
124
+ def acquire_global_lock_sql
125
+ return <<~SQL.strip if DB_ORACLE
126
+ SELECT DBMS_LOCK.REQUEST(#{Workhorse::Poller::ORACLE_LOCK_HANDLE}, #{Workhorse::Poller::ORACLE_LOCK_MODE}, 1)
127
+ FROM DUAL
128
+ SQL
129
+
130
+ return "SELECT GET_LOCK(CONCAT(DATABASE(), '_workhorse'), 1)"
131
+ end
132
+
133
+ def release_global_lock_sql
134
+ return "SELECT DBMS_LOCK.RELEASE(#{Workhorse::Poller::ORACLE_LOCK_HANDLE}) FROM DUAL" if DB_ORACLE
135
+
136
+ return "SELECT RELEASE_LOCK(CONCAT(DATABASE(), '_workhorse'))"
137
+ end
138
+
71
139
  def clear_locks_and_db_threads!
72
140
  # Releases the locks held by *this* connection, which is all a fresh run
73
141
  # needs. It used to kill the queries of every other connection as well, to
@@ -78,7 +146,16 @@ class WorkhorseTest < ActiveSupport::TestCase
78
146
  # spurious "Lost connection to server during query" failures. Should a
79
147
  # crashed run ever leave a lock behind, kill its *connection*, which does
80
148
  # release it.
81
- Workhorse::DbJob.connection.execute('SELECT RELEASE_ALL_LOCKS()')
149
+ if DB_ORACLE
150
+ # Oracle has no "release everything" call, but workhorse takes exactly
151
+ # one handle. RELEASE reports a status rather than raising when the lock
152
+ # is not held, so this needs no guard.
153
+ Workhorse::DbJob.connection.execute(
154
+ "SELECT DBMS_LOCK.RELEASE(#{Workhorse::Poller::ORACLE_LOCK_HANDLE}) FROM DUAL"
155
+ )
156
+ else
157
+ Workhorse::DbJob.connection.execute('SELECT RELEASE_ALL_LOCKS()')
158
+ end
82
159
  end
83
160
 
84
161
  def remove_pids!
@@ -144,8 +221,14 @@ class WorkhorseTest < ActiveSupport::TestCase
144
221
  w.shutdown
145
222
  end
146
223
 
224
+ # Without auto_terminate unless asked for, as work and work_until have it:
225
+ # a worker that has it traps TERM and INT for the whole process and never
226
+ # gives them back, so the test process ignored the TERM meant to stop it for
227
+ # the rest of the run - and a timed-out CI attempt kept running alongside
228
+ # the next. Every test hands them back afterwards regardless, see the
229
+ # teardown above.
147
230
  def with_worker(options = {})
148
- w = Workhorse::Worker.new(**options)
231
+ w = Workhorse::Worker.new(auto_terminate: false, **options)
149
232
  w.start
150
233
  begin
151
234
  yield(w)
@@ -171,6 +254,12 @@ class WorkhorseTest < ActiveSupport::TestCase
171
254
  daemon.stop(quiet: true)
172
255
  end
173
256
 
257
+ # Hands TERM and INT back to the handlers the process started with, which a
258
+ # worker started with auto_terminate replaced.
259
+ def restore_termination_traps
260
+ ORIGINAL_TERMINATION_TRAPS.each { |signal, handler| Signal.trap(signal, handler) }
261
+ end
262
+
174
263
  def with_retries(max = 50, interval: 0.1, &_block)
175
264
  runs = 0
176
265
 
@@ -200,15 +289,24 @@ class WorkhorseTest < ActiveSupport::TestCase
200
289
  end
201
290
  end
202
291
 
292
+ # On Oracle the "database" is a service name rather than a schema, and the
293
+ # schema is the user the suite connects as.
203
294
  ActiveRecord::Base.establish_connection(
204
295
  adapter: DB_ADAPTER,
205
- database: ENV.fetch('DB_NAME', nil) || 'workhorse',
206
- username: ENV.fetch('DB_USERNAME', nil) || 'root',
207
- password: ENV.fetch('DB_PASSWORD', nil) || '',
296
+ database: ENV.fetch('DB_NAME', nil) || (DB_ORACLE ? 'FREEPDB1' : 'workhorse'),
297
+ username: ENV.fetch('DB_USERNAME', nil) || (DB_ORACLE ? 'workhorse' : 'root'),
298
+ password: ENV.fetch('DB_PASSWORD', nil) || (DB_ORACLE ? 'workhorse' : ''),
208
299
  host: ENV.fetch('DB_HOST', nil) || '127.0.0.1',
209
- port: ENV.fetch('DB_PORT', nil) || 3306,
300
+ port: ENV.fetch('DB_PORT', nil) || (DB_ORACLE ? 1521 : 3306),
210
301
  pool: 10
211
302
  )
212
303
 
213
304
  require 'db_schema'
214
305
  require 'workhorse'
306
+
307
+ # The handlers the process started with, see #restore_termination_traps.
308
+ ORIGINAL_TERMINATION_TRAPS = Workhorse::Worker::SHUTDOWN_SIGNALS.to_h do |signal|
309
+ handler = Signal.trap(signal, 'DEFAULT')
310
+ Signal.trap(signal, handler)
311
+ [signal, handler]
312
+ end
@@ -10,9 +10,7 @@ class Workhorse::DbJobTest < WorkhorseTest
10
10
 
11
11
  def test_reset_failed
12
12
  job = Workhorse.enqueue FailingTestJob.new
13
- work 0.5
14
- job.reload
15
- assert_equal 'failed', job.state
13
+ work_until(pool_size: 5, polling_interval: 0.2) { assert_equal 'failed', job.reload.state }
16
14
 
17
15
  job.reset!
18
16
 
@@ -398,19 +398,6 @@ class Workhorse::NotifierTest < WorkhorseTest
398
398
  return File.join(Dir.tmpdir, 'workhorse_test.wake')
399
399
  end
400
400
 
401
- # Holds workhorse's global lock on a connection of its own, so that the code
402
- # under test sees it as taken by another worker.
403
- def with_global_lock_held
404
- connection = ActiveRecord::Base.connection_pool.checkout
405
- connection.select_value("SELECT GET_LOCK(CONCAT(DATABASE(), '_workhorse'), 1)")
406
- yield
407
- ensure
408
- if connection
409
- connection.select_value("SELECT RELEASE_LOCK(CONCAT(DATABASE(), '_workhorse'))")
410
- ActiveRecord::Base.connection_pool.checkin(connection)
411
- end
412
- end
413
-
414
401
  def file_notifier
415
402
  return Workhorse::Notifiers::FileSystem.new(path: wake_path)
416
403
  end
@@ -8,28 +8,26 @@ class Workhorse::PerformerTest < WorkhorseTest
8
8
  Workhorse.enqueue DbConnectionTestJob.new
9
9
  end
10
10
 
11
- work 0.2, polling_interval: 0.2
11
+ work_until(pool_size: 5, polling_interval: 0.2) do
12
+ assert_equal 2, DbConnectionTestJob.db_connections.count
13
+ end
12
14
 
13
- assert_equal 2, DbConnectionTestJob.db_connections.count
14
15
  assert_equal 2, DbConnectionTestJob.db_connections.uniq.count
15
16
  end
16
17
 
17
18
  def test_success
18
19
  Workhorse.enqueue BasicJob.new(sleep_time: 0.1)
19
- work 0.2, polling_interval: 0.2
20
- assert_equal 'succeeded', Workhorse::DbJob.first.state
20
+ work_until(pool_size: 5, polling_interval: 0.2) { assert_equal 'succeeded', Workhorse::DbJob.first.state }
21
21
  end
22
22
 
23
23
  def test_exception
24
24
  Workhorse.enqueue FailingTestJob.new
25
- work 0.2, polling_interval: 0.2
26
- assert_equal 'failed', Workhorse::DbJob.first.state
25
+ work_until(pool_size: 5, polling_interval: 0.2) { assert_equal 'failed', Workhorse::DbJob.first.state }
27
26
  end
28
27
 
29
28
  def test_syntax_exception
30
29
  Workhorse.enqueue SyntaxErrorJob
31
- work 0.2, polling_interval: 0.2
32
- assert_equal 'failed', Workhorse::DbJob.first.state
30
+ work_until(pool_size: 5, polling_interval: 0.2) { assert_equal 'failed', Workhorse::DbJob.first.state }
33
31
  end
34
32
 
35
33
  def test_on_exception
@@ -41,7 +39,7 @@ class Workhorse::PerformerTest < WorkhorseTest
41
39
  end
42
40
 
43
41
  Workhorse.enqueue FailingTestJob.new
44
- work 0.2, polling_interval: 0.2
42
+ work_until(pool_size: 5, polling_interval: 0.2) { assert exception, 'expected on_exception to be called' }
45
43
 
46
44
  assert_equal exception.message, FailingTestJob::MESSAGE
47
45
  ensure
@@ -19,25 +19,25 @@ class Workhorse::PollerTest < WorkhorseTest
19
19
  Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: :q2
20
20
  Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: :q2
21
21
 
22
- assert_equal %w[q1 q2], w.poller.send(:valid_queues)
22
+ assert_equal %w[q1 q2], valid_queues_of(w)
23
23
 
24
24
  first_job = Workhorse::DbJob.first
25
25
  first_job.mark_locked!(42)
26
26
 
27
- assert_equal %w[q2], w.poller.send(:valid_queues)
27
+ assert_equal %w[q2], valid_queues_of(w)
28
28
 
29
29
  first_job.mark_started!
30
30
 
31
- assert_equal %w[q2], w.poller.send(:valid_queues)
31
+ assert_equal %w[q2], valid_queues_of(w)
32
32
 
33
33
  first_job.mark_succeeded!
34
34
 
35
- assert_equal %w[q1 q2], w.poller.send(:valid_queues)
35
+ assert_equal %w[q1 q2], valid_queues_of(w)
36
36
 
37
37
  last_job = Workhorse::DbJob.last
38
38
  last_job.mark_locked!(42)
39
39
 
40
- assert_equal %w[q1], w.poller.send(:valid_queues)
40
+ assert_equal %w[q1], valid_queues_of(w)
41
41
 
42
42
  begin
43
43
  fail 'Some exception'
@@ -45,34 +45,36 @@ class Workhorse::PollerTest < WorkhorseTest
45
45
  last_job.mark_failed!(e)
46
46
  end
47
47
 
48
- assert_equal %w[q1 q2], w.poller.send(:valid_queues)
48
+ assert_equal %w[q1 q2], valid_queues_of(w)
49
49
  end
50
50
 
51
51
  def test_valid_queues_2
52
52
  w = Workhorse::Worker.new(polling_interval: 60)
53
53
 
54
- assert_equal [], w.poller.send(:valid_queues)
54
+ assert_equal [], valid_queues_of(w)
55
55
 
56
56
  Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: nil
57
57
 
58
- assert_equal [nil], w.poller.send(:valid_queues)
58
+ assert_equal [nil], valid_queues_of(w)
59
59
 
60
60
  a_job = Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: :a
61
61
 
62
- assert_equal [nil, 'a'], w.poller.send(:valid_queues)
62
+ assert_equal [nil, 'a'], valid_queues_of(w)
63
63
 
64
64
  a_job.update_attribute :state, :locked
65
65
 
66
- assert_equal [nil], w.poller.send(:valid_queues)
66
+ assert_equal [nil], valid_queues_of(w)
67
67
  end
68
68
 
69
69
  def test_no_queues
70
70
  w = Workhorse::Worker.new(polling_interval: 60)
71
- assert_equal [], w.poller.send(:valid_queues)
71
+ assert_equal [], valid_queues_of(w)
72
72
  end
73
73
 
74
- # Not every adapter returns a result set from `execute`, so querying the
75
- # valid queues must not rely on its return value.
74
+ # Not every adapter returns a result set from `execute`: the Oracle enhanced
75
+ # adapter, for instance, returns `true` for queries, both before and after
76
+ # version 7.0.0. Querying the valid queues must therefore not rely on the
77
+ # return value of `execute`.
76
78
  def test_valid_queues_without_usable_execute
77
79
  w = Workhorse::Worker.new(polling_interval: 60)
78
80
 
@@ -80,7 +82,7 @@ class Workhorse::PollerTest < WorkhorseTest
80
82
  Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: :a
81
83
 
82
84
  with_return_value(Workhorse::DbJob.connection, :execute, true) do
83
- assert_equal [nil, 'a'], w.poller.send(:valid_queues)
85
+ assert_equal [nil, 'a'], valid_queues_of(w)
84
86
  end
85
87
  end
86
88
 
@@ -92,11 +94,11 @@ class Workhorse::PollerTest < WorkhorseTest
92
94
  end
93
95
  jobs = Workhorse::DbJob.all
94
96
 
95
- assert_equal [nil], w.poller.send(:valid_queues)
97
+ assert_equal [nil], valid_queues_of(w)
96
98
 
97
99
  jobs[0].mark_locked!(42)
98
100
 
99
- assert_equal [nil], w.poller.send(:valid_queues)
101
+ assert_equal [nil], valid_queues_of(w)
100
102
  end
101
103
 
102
104
  def test_with_instant_repolling
@@ -133,10 +135,37 @@ class Workhorse::PollerTest < WorkhorseTest
133
135
  Workhorse.enqueue BasicJob.new(some_param: i, sleep_time: 0)
134
136
  end
135
137
 
136
- # Create 10 worker processes that work for 3s each
138
+ # Create 10 worker processes that work until all 100 jobs have succeeded,
139
+ # and for at least 3s, so that each is still polling while the second
140
+ # batch comes in. Not for a fixed time: every poll holds the global lock
141
+ # throughout, so the workers take turns, and 3s got through as few as 25
142
+ # of the 100 jobs on a loaded CI runner, and 61 on Oracle. What is tested
143
+ # is that none is taken twice and every worker gets a turn, not how fast.
137
144
  10.times do
138
145
  Process.fork do
139
- work 3, pool_size: 1, polling_interval: 0.1
146
+ started = Time.now
147
+
148
+ # A worker that cannot even be shut down would leave waitall below
149
+ # waiting for good, so it reports where it is stuck and gives up.
150
+ Thread.new do
151
+ sleep 90
152
+ report_threads 'Worker process still running after 90s'
153
+ exit!(1)
154
+ end
155
+
156
+ with_worker(pool_size: 1, polling_interval: 0.1, auto_terminate: false) do
157
+ loop do
158
+ elapsed = Time.now - started
159
+ break if elapsed >= 3 && Workhorse::DbJob.succeeded.count >= 100
160
+
161
+ if elapsed >= 60
162
+ report_threads 'Worker process gave up waiting for all jobs to succeed after 60s'
163
+ break
164
+ end
165
+
166
+ sleep 0.1
167
+ end
168
+ end
140
169
  ensure
141
170
  # Exit without running the at_exit handlers of the test process: one
142
171
  # of them is Minitest's, which joins threads this fork did not
@@ -153,24 +182,51 @@ class Workhorse::PollerTest < WorkhorseTest
153
182
  Workhorse.enqueue BasicJob.new(sleep_time: 0)
154
183
  end
155
184
 
156
- # Wait for all forked processes to finish (should take ~3s)
185
+ # Wait for all forked processes to finish
157
186
  Process.waitall
158
187
 
159
188
  total = Workhorse::DbJob.count
160
189
  succeeded = Workhorse::DbJob.succeeded.count
161
- used_workers = Workhorse::DbJob.lock.pluck(:locked_by).uniq.size
190
+ # In a transaction, as Oracle commits a FOR UPDATE outside of one before
191
+ # its rows are fetched and then fails with "fetch out of sequence".
192
+ used_workers = Workhorse::DbJob.transaction do
193
+ Workhorse::DbJob.lock.pluck(:locked_by).uniq.size
194
+ end
162
195
 
163
196
  # Make sure there are 100 jobs, all jobs have succeeded and that all of the
164
- # workers have had their turn.
197
+ # workers have had their turn. The jobs that have not are listed, as all
198
+ # a count says is that something got stuck, not where.
199
+ unfinished = Workhorse::DbJob.where.not(state: 'succeeded').order(:id).map do |job|
200
+ "##{job.id} #{job.state}, locked_by #{job.locked_by.inspect}, " \
201
+ "locked_at #{job.locked_at.inspect}, started_at #{job.started_at.inspect}"
202
+ end
203
+
165
204
  assert_equal 100, total
166
- assert_equal 100, succeeded
205
+ assert_equal 100, succeeded, "Jobs that did not succeed:\n#{unfinished.join("\n")}"
167
206
  assert_equal 10, used_workers
168
207
  end
169
208
 
209
+ # A poll that finds the global lock taken waits at least as long as it asked
210
+ # to before giving up. Below half a second is what the poller asks for with
211
+ # a short polling interval, and what MySQL and Oracle, which take the timeout
212
+ # as whole seconds, used to round down to not waiting at all.
213
+ def test_contended_global_lock_waits_for_its_timeout
214
+ poller = Workhorse::Worker.new(polling_interval: 0.3).poller
215
+ ran = false
216
+
217
+ elapsed = with_global_lock_held do
218
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
219
+ poller.send(:with_global_lock, timeout: 0.3, count_failures: false) { ran = true }
220
+ Process.clock_gettime(Process::CLOCK_MONOTONIC) - started
221
+ end
222
+
223
+ assert_not ran
224
+ assert_operator elapsed, :>=, 0.25
225
+ end
226
+
170
227
  def test_connection_loss
171
- # rubocop: disable Style/GlobalVars
228
+ # rubocop: disable-next Style/GlobalVars
172
229
  $thread_conn = nil
173
- # rubocop: enable Style/GlobalVars
174
230
 
175
231
  Workhorse.enqueue BasicJob.new(sleep_time: 3)
176
232
 
@@ -203,7 +259,14 @@ class Workhorse::PollerTest < WorkhorseTest
203
259
  Workhorse.clean_stuck_jobs = clean
204
260
  with_daemon do
205
261
  Workhorse.enqueue BasicJob.new(sleep_time: 5)
206
- sleep 0.2
262
+
263
+ # Waited for rather than slept on: until a worker has taken the job,
264
+ # locked_by is empty and the cleanup, which matches on the host in it,
265
+ # cannot recognise the job as one of its own.
266
+ with_retries do
267
+ assert_equal 'started', Workhorse::DbJob.first.state
268
+ end
269
+
207
270
  kill_deamon_workers
208
271
 
209
272
  assert_equal 1, Workhorse::DbJob.count
@@ -269,6 +332,13 @@ class Workhorse::PollerTest < WorkhorseTest
269
332
 
270
333
  private
271
334
 
335
+ # The worker's valid queues, the queueless one first. Sorted here, as they
336
+ # come from a SELECT DISTINCT without ORDER BY: MySQL happens to return NULL
337
+ # first, Oracle does not, and the poller does not depend on the order.
338
+ def valid_queues_of(worker)
339
+ return worker.poller.send(:valid_queues).sort_by { |queue| [queue.nil? ? 0 : 1, queue.to_s] }
340
+ end
341
+
272
342
  def kill_deamon_workers
273
343
  pids = daemon.workers.map(&:pid)
274
344
  pids.each do |pid|
@@ -362,7 +362,7 @@ class Workhorse::ScheduleTest < WorkhorseTest
362
362
  end
363
363
 
364
364
  assert_equal 1, Workhorse::DbJob.succeeded.count
365
- assert exceptions.any? { |e| e.is_a?(NameError) }, "expected a NameError, got #{exceptions.map(&:class)}"
365
+ assert exceptions.any?(NameError), "expected a NameError, got #{exceptions.map(&:class)}"
366
366
  end
367
367
 
368
368
  # An occurrence must not be consumed unless the job for it exists. Were the