workhorse 2.0.0.rc0 → 2.0.0.rc1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.github/workflows/ruby.yml +137 -1
- data/CHANGELOG.md +48 -0
- data/Gemfile +16 -1
- data/README.md +31 -16
- data/VERSION +1 -1
- data/lib/generators/workhorse/templates/create_table_jobs.rb +19 -4
- data/lib/generators/workhorse/templates/create_table_workhorse_schedules.rb +14 -1
- data/lib/workhorse/daemon/shell_handler.rb +4 -1
- data/lib/workhorse/daemon.rb +5 -1
- data/lib/workhorse/db_job.rb +40 -1
- data/lib/workhorse/performer.rb +1 -2
- data/lib/workhorse/poller.rb +55 -15
- data/lib/workhorse/worker.rb +25 -4
- data/test/lib/db_schema.rb +19 -6
- data/test/lib/test_helper.rb +107 -9
- data/test/workhorse/db_job_test.rb +1 -3
- data/test/workhorse/notifier_test.rb +0 -13
- data/test/workhorse/performer_test.rb +7 -9
- data/test/workhorse/poller_test.rb +95 -25
- data/test/workhorse/schedule_test.rb +1 -1
- data/test/workhorse/worker_test.rb +109 -76
- data/workhorse.gemspec +3 -3
- metadata +2 -2
data/lib/workhorse/worker.rb
CHANGED
|
@@ -291,7 +291,7 @@ module Workhorse
|
|
|
291
291
|
return unless daemon_id
|
|
292
292
|
|
|
293
293
|
path = self.class.heartbeat_file_for(daemon_id)
|
|
294
|
-
|
|
294
|
+
touch_pid_file(path) if path
|
|
295
295
|
rescue StandardError => e
|
|
296
296
|
Workhorse.debug_log("[Job worker #{id}] Heartbeat touch failed: #{e.class}: #{e.message}")
|
|
297
297
|
end
|
|
@@ -363,7 +363,7 @@ module Workhorse
|
|
|
363
363
|
Workhorse.debug_log("[Job worker #{id}] Memory limit exceeded: #{mem}MB > #{max}MB, initiating shutdown")
|
|
364
364
|
|
|
365
365
|
if defined?(Rails)
|
|
366
|
-
|
|
366
|
+
touch_pid_file(self.class.shutdown_file_for(pid))
|
|
367
367
|
end
|
|
368
368
|
|
|
369
369
|
log "Worker process #{id.inspect} memory consumption (RSS) of #{mem}MB exceeds " \
|
|
@@ -444,8 +444,17 @@ module Workhorse
|
|
|
444
444
|
# Create shutdown file for watch to detect
|
|
445
445
|
shutdown_file = self.class.shutdown_file_for(pid)
|
|
446
446
|
if shutdown_file
|
|
447
|
-
|
|
448
|
-
|
|
447
|
+
begin
|
|
448
|
+
touch_pid_file(shutdown_file)
|
|
449
|
+
Workhorse.debug_log("[Job worker #{id}] Shutdown file created: #{shutdown_file}")
|
|
450
|
+
rescue StandardError => e
|
|
451
|
+
# The shutdown has to go ahead regardless: the worker stopped
|
|
452
|
+
# accepting jobs above, so giving up here would leave it taking none
|
|
453
|
+
# and never exiting. Without the file, watch sees a worker that is
|
|
454
|
+
# gone rather than one that asked to be restarted, and starts it all
|
|
455
|
+
# the same.
|
|
456
|
+
log "Could not create shutdown file #{shutdown_file}, shutting down regardless: #{e.message}", :warn
|
|
457
|
+
end
|
|
449
458
|
end
|
|
450
459
|
|
|
451
460
|
# Monitor in a separate thread to avoid blocking the signal handler
|
|
@@ -533,5 +542,17 @@ module Workhorse
|
|
|
533
542
|
Kernel.sleep 0.2
|
|
534
543
|
end
|
|
535
544
|
end
|
|
545
|
+
|
|
546
|
+
# Touches a file in the pids directory, creating the directory first. It
|
|
547
|
+
# is not guaranteed to exist: `tmp` is rarely checked in, so a fresh
|
|
548
|
+
# checkout or deployment has none until something writes there.
|
|
549
|
+
#
|
|
550
|
+
# @param path [Pathname, String]
|
|
551
|
+
# @return [void]
|
|
552
|
+
# @private
|
|
553
|
+
def touch_pid_file(path)
|
|
554
|
+
FileUtils.mkdir_p(File.dirname(path))
|
|
555
|
+
FileUtils.touch(path)
|
|
556
|
+
end
|
|
536
557
|
end
|
|
537
558
|
end
|
data/test/lib/db_schema.rb
CHANGED
|
@@ -1,10 +1,21 @@
|
|
|
1
1
|
ActiveRecord::Schema.define do
|
|
2
2
|
self.verbose = false
|
|
3
3
|
|
|
4
|
+
# Prefix lengths for the indexes on string columns. MySQL needs them, as the
|
|
5
|
+
# default `utf8mb4` charset puts a full `varchar(255)` past the maximum key
|
|
6
|
+
# length; Oracle indexes the whole column and rejects the option.
|
|
7
|
+
state_length = DB_ORACLE ? {} : { length: { state: 191 } }
|
|
8
|
+
queue_length = DB_ORACLE ? {} : { length: 191 }
|
|
9
|
+
key_length = DB_ORACLE ? {} : { length: 191 }
|
|
10
|
+
|
|
4
11
|
create_table :jobs, force: true do |t|
|
|
5
12
|
t.string :state, null: false, default: 'waiting'
|
|
6
13
|
t.string :queue, null: true
|
|
7
|
-
|
|
14
|
+
|
|
15
|
+
# Binary rather than text, matching the generated migration: the handler is
|
|
16
|
+
# a `Marshal.dump`, and a text column is character data - on Oracle a CLOB,
|
|
17
|
+
# whose character set conversion would corrupt it.
|
|
18
|
+
t.binary :handler, null: false, limit: 4_294_967_295
|
|
8
19
|
|
|
9
20
|
t.string :locked_by
|
|
10
21
|
t.datetime :locked_at
|
|
@@ -25,10 +36,12 @@ ActiveRecord::Schema.define do
|
|
|
25
36
|
t.timestamps null: false
|
|
26
37
|
end
|
|
27
38
|
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
add_index :jobs,
|
|
31
|
-
add_index :jobs, %i[state
|
|
39
|
+
# The index names are given explicitly because the ones Rails would derive
|
|
40
|
+
# exceed the 30 characters Oracle allows before 12.2.
|
|
41
|
+
add_index :jobs, :queue, **queue_length
|
|
42
|
+
add_index :jobs, %i[state perform_at], name: 'idx_jobs_state_perform_at', **state_length
|
|
43
|
+
add_index :jobs, %i[state priority created_at], name: 'idx_jobs_state_prio_created', **state_length
|
|
44
|
+
add_index :jobs, %i[state expires_at], name: 'idx_jobs_state_expires_at', **state_length
|
|
32
45
|
add_index :jobs, :perform_at
|
|
33
46
|
|
|
34
47
|
create_table :workhorse_schedules, force: true do |t|
|
|
@@ -44,6 +57,6 @@ ActiveRecord::Schema.define do
|
|
|
44
57
|
t.timestamps null: false
|
|
45
58
|
end
|
|
46
59
|
|
|
47
|
-
add_index :workhorse_schedules, :key, unique: true,
|
|
60
|
+
add_index :workhorse_schedules, :key, unique: true, name: 'idx_wh_schedules_key', **key_length
|
|
48
61
|
add_index :workhorse_schedules, %i[enabled next_at], name: 'idx_wh_schedules_due'
|
|
49
62
|
end
|
data/test/lib/test_helper.rb
CHANGED
|
@@ -6,8 +6,8 @@ require 'benchmark'
|
|
|
6
6
|
require 'concurrent'
|
|
7
7
|
require 'jobs'
|
|
8
8
|
|
|
9
|
-
# The adapter to run the test suite against.
|
|
10
|
-
#
|
|
9
|
+
# The adapter to run the test suite against. Workhorse has to work with each of
|
|
10
|
+
# them, so each is covered by CI.
|
|
11
11
|
DB_ADAPTER = ENV.fetch('DB_ADAPTER', 'mysql2')
|
|
12
12
|
|
|
13
13
|
case DB_ADAPTER
|
|
@@ -15,10 +15,16 @@ when 'mysql2'
|
|
|
15
15
|
require 'mysql2'
|
|
16
16
|
when 'trilogy'
|
|
17
17
|
require 'trilogy'
|
|
18
|
+
when 'oracle_enhanced'
|
|
19
|
+
require 'active_record/connection_adapters/oracle_enhanced_adapter'
|
|
18
20
|
else
|
|
19
|
-
fail "Unsupported DB_ADAPTER #{DB_ADAPTER.inspect}, use 'mysql2' or '
|
|
21
|
+
fail "Unsupported DB_ADAPTER #{DB_ADAPTER.inspect}, use 'mysql2', 'trilogy' or 'oracle_enhanced'."
|
|
20
22
|
end
|
|
21
23
|
|
|
24
|
+
# Whether the suite is running against Oracle. Used where the schema or an
|
|
25
|
+
# assertion cannot be written the same way for both families.
|
|
26
|
+
DB_ORACLE = DB_ADAPTER == 'oracle_enhanced'
|
|
27
|
+
|
|
22
28
|
class MockRailsEnv < String
|
|
23
29
|
def production?
|
|
24
30
|
self == 'production'
|
|
@@ -44,6 +50,38 @@ class Rails
|
|
|
44
50
|
end
|
|
45
51
|
|
|
46
52
|
class WorkhorseTest < ActiveSupport::TestCase
|
|
53
|
+
# Seconds a test may run before the backtraces of all its threads are
|
|
54
|
+
# printed. A hung test otherwise only shows up as a CI attempt killed at its
|
|
55
|
+
# time limit, with nothing saying where it hung.
|
|
56
|
+
HANG_REPORT_AFTER = 120
|
|
57
|
+
|
|
58
|
+
# Callbacks rather than methods, so that a test class defining its own setup
|
|
59
|
+
# or teardown does not skip them.
|
|
60
|
+
setup do
|
|
61
|
+
test = "#{self.class}##{name}"
|
|
62
|
+
|
|
63
|
+
@hang_watchdog = Thread.new do
|
|
64
|
+
sleep HANG_REPORT_AFTER
|
|
65
|
+
report_threads "#{test} is still running after #{HANG_REPORT_AFTER}s"
|
|
66
|
+
end
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
# Prints the backtraces of all threads of this process but the calling one.
|
|
70
|
+
def report_threads(title)
|
|
71
|
+
warn "#{title} (PID #{Process.pid}). Its threads:"
|
|
72
|
+
|
|
73
|
+
Thread.list.each do |thread|
|
|
74
|
+
next if thread == Thread.current
|
|
75
|
+
|
|
76
|
+
warn "--- #{thread.inspect}\n#{(thread.backtrace || ['(no backtrace)']).join("\n")}"
|
|
77
|
+
end
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
teardown do
|
|
81
|
+
@hang_watchdog&.kill
|
|
82
|
+
restore_termination_traps
|
|
83
|
+
end
|
|
84
|
+
|
|
47
85
|
def setup
|
|
48
86
|
remove_pids!
|
|
49
87
|
clear_locks_and_db_threads!
|
|
@@ -68,6 +106,36 @@ class WorkhorseTest < ActiveSupport::TestCase
|
|
|
68
106
|
object.singleton_class.send(:remove_method, method)
|
|
69
107
|
end
|
|
70
108
|
|
|
109
|
+
# Holds workhorse's global lock on a connection of its own, so that the code
|
|
110
|
+
# under test sees it as taken by another worker.
|
|
111
|
+
def with_global_lock_held
|
|
112
|
+
connection = ActiveRecord::Base.connection_pool.checkout
|
|
113
|
+
connection.select_value(acquire_global_lock_sql)
|
|
114
|
+
yield
|
|
115
|
+
ensure
|
|
116
|
+
if connection
|
|
117
|
+
connection.select_value(release_global_lock_sql)
|
|
118
|
+
ActiveRecord::Base.connection_pool.checkin(connection)
|
|
119
|
+
end
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
# Mirrors what Workhorse::Poller#with_global_lock emits, so that the lock the
|
|
123
|
+
# code under test tries to take is the same one.
|
|
124
|
+
def acquire_global_lock_sql
|
|
125
|
+
return <<~SQL.strip if DB_ORACLE
|
|
126
|
+
SELECT DBMS_LOCK.REQUEST(#{Workhorse::Poller::ORACLE_LOCK_HANDLE}, #{Workhorse::Poller::ORACLE_LOCK_MODE}, 1)
|
|
127
|
+
FROM DUAL
|
|
128
|
+
SQL
|
|
129
|
+
|
|
130
|
+
return "SELECT GET_LOCK(CONCAT(DATABASE(), '_workhorse'), 1)"
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
def release_global_lock_sql
|
|
134
|
+
return "SELECT DBMS_LOCK.RELEASE(#{Workhorse::Poller::ORACLE_LOCK_HANDLE}) FROM DUAL" if DB_ORACLE
|
|
135
|
+
|
|
136
|
+
return "SELECT RELEASE_LOCK(CONCAT(DATABASE(), '_workhorse'))"
|
|
137
|
+
end
|
|
138
|
+
|
|
71
139
|
def clear_locks_and_db_threads!
|
|
72
140
|
# Releases the locks held by *this* connection, which is all a fresh run
|
|
73
141
|
# needs. It used to kill the queries of every other connection as well, to
|
|
@@ -78,7 +146,16 @@ class WorkhorseTest < ActiveSupport::TestCase
|
|
|
78
146
|
# spurious "Lost connection to server during query" failures. Should a
|
|
79
147
|
# crashed run ever leave a lock behind, kill its *connection*, which does
|
|
80
148
|
# release it.
|
|
81
|
-
|
|
149
|
+
if DB_ORACLE
|
|
150
|
+
# Oracle has no "release everything" call, but workhorse takes exactly
|
|
151
|
+
# one handle. RELEASE reports a status rather than raising when the lock
|
|
152
|
+
# is not held, so this needs no guard.
|
|
153
|
+
Workhorse::DbJob.connection.execute(
|
|
154
|
+
"SELECT DBMS_LOCK.RELEASE(#{Workhorse::Poller::ORACLE_LOCK_HANDLE}) FROM DUAL"
|
|
155
|
+
)
|
|
156
|
+
else
|
|
157
|
+
Workhorse::DbJob.connection.execute('SELECT RELEASE_ALL_LOCKS()')
|
|
158
|
+
end
|
|
82
159
|
end
|
|
83
160
|
|
|
84
161
|
def remove_pids!
|
|
@@ -144,8 +221,14 @@ class WorkhorseTest < ActiveSupport::TestCase
|
|
|
144
221
|
w.shutdown
|
|
145
222
|
end
|
|
146
223
|
|
|
224
|
+
# Without auto_terminate unless asked for, as work and work_until have it:
|
|
225
|
+
# a worker that has it traps TERM and INT for the whole process and never
|
|
226
|
+
# gives them back, so the test process ignored the TERM meant to stop it for
|
|
227
|
+
# the rest of the run - and a timed-out CI attempt kept running alongside
|
|
228
|
+
# the next. Every test hands them back afterwards regardless, see the
|
|
229
|
+
# teardown above.
|
|
147
230
|
def with_worker(options = {})
|
|
148
|
-
w = Workhorse::Worker.new(**options)
|
|
231
|
+
w = Workhorse::Worker.new(auto_terminate: false, **options)
|
|
149
232
|
w.start
|
|
150
233
|
begin
|
|
151
234
|
yield(w)
|
|
@@ -171,6 +254,12 @@ class WorkhorseTest < ActiveSupport::TestCase
|
|
|
171
254
|
daemon.stop(quiet: true)
|
|
172
255
|
end
|
|
173
256
|
|
|
257
|
+
# Hands TERM and INT back to the handlers the process started with, which a
|
|
258
|
+
# worker started with auto_terminate replaced.
|
|
259
|
+
def restore_termination_traps
|
|
260
|
+
ORIGINAL_TERMINATION_TRAPS.each { |signal, handler| Signal.trap(signal, handler) }
|
|
261
|
+
end
|
|
262
|
+
|
|
174
263
|
def with_retries(max = 50, interval: 0.1, &_block)
|
|
175
264
|
runs = 0
|
|
176
265
|
|
|
@@ -200,15 +289,24 @@ class WorkhorseTest < ActiveSupport::TestCase
|
|
|
200
289
|
end
|
|
201
290
|
end
|
|
202
291
|
|
|
292
|
+
# On Oracle the "database" is a service name rather than a schema, and the
|
|
293
|
+
# schema is the user the suite connects as.
|
|
203
294
|
ActiveRecord::Base.establish_connection(
|
|
204
295
|
adapter: DB_ADAPTER,
|
|
205
|
-
database: ENV.fetch('DB_NAME', nil) || 'workhorse',
|
|
206
|
-
username: ENV.fetch('DB_USERNAME', nil) || 'root',
|
|
207
|
-
password: ENV.fetch('DB_PASSWORD', nil) || '',
|
|
296
|
+
database: ENV.fetch('DB_NAME', nil) || (DB_ORACLE ? 'FREEPDB1' : 'workhorse'),
|
|
297
|
+
username: ENV.fetch('DB_USERNAME', nil) || (DB_ORACLE ? 'workhorse' : 'root'),
|
|
298
|
+
password: ENV.fetch('DB_PASSWORD', nil) || (DB_ORACLE ? 'workhorse' : ''),
|
|
208
299
|
host: ENV.fetch('DB_HOST', nil) || '127.0.0.1',
|
|
209
|
-
port: ENV.fetch('DB_PORT', nil) || 3306,
|
|
300
|
+
port: ENV.fetch('DB_PORT', nil) || (DB_ORACLE ? 1521 : 3306),
|
|
210
301
|
pool: 10
|
|
211
302
|
)
|
|
212
303
|
|
|
213
304
|
require 'db_schema'
|
|
214
305
|
require 'workhorse'
|
|
306
|
+
|
|
307
|
+
# The handlers the process started with, see #restore_termination_traps.
|
|
308
|
+
ORIGINAL_TERMINATION_TRAPS = Workhorse::Worker::SHUTDOWN_SIGNALS.to_h do |signal|
|
|
309
|
+
handler = Signal.trap(signal, 'DEFAULT')
|
|
310
|
+
Signal.trap(signal, handler)
|
|
311
|
+
[signal, handler]
|
|
312
|
+
end
|
|
@@ -10,9 +10,7 @@ class Workhorse::DbJobTest < WorkhorseTest
|
|
|
10
10
|
|
|
11
11
|
def test_reset_failed
|
|
12
12
|
job = Workhorse.enqueue FailingTestJob.new
|
|
13
|
-
|
|
14
|
-
job.reload
|
|
15
|
-
assert_equal 'failed', job.state
|
|
13
|
+
work_until(pool_size: 5, polling_interval: 0.2) { assert_equal 'failed', job.reload.state }
|
|
16
14
|
|
|
17
15
|
job.reset!
|
|
18
16
|
|
|
@@ -398,19 +398,6 @@ class Workhorse::NotifierTest < WorkhorseTest
|
|
|
398
398
|
return File.join(Dir.tmpdir, 'workhorse_test.wake')
|
|
399
399
|
end
|
|
400
400
|
|
|
401
|
-
# Holds workhorse's global lock on a connection of its own, so that the code
|
|
402
|
-
# under test sees it as taken by another worker.
|
|
403
|
-
def with_global_lock_held
|
|
404
|
-
connection = ActiveRecord::Base.connection_pool.checkout
|
|
405
|
-
connection.select_value("SELECT GET_LOCK(CONCAT(DATABASE(), '_workhorse'), 1)")
|
|
406
|
-
yield
|
|
407
|
-
ensure
|
|
408
|
-
if connection
|
|
409
|
-
connection.select_value("SELECT RELEASE_LOCK(CONCAT(DATABASE(), '_workhorse'))")
|
|
410
|
-
ActiveRecord::Base.connection_pool.checkin(connection)
|
|
411
|
-
end
|
|
412
|
-
end
|
|
413
|
-
|
|
414
401
|
def file_notifier
|
|
415
402
|
return Workhorse::Notifiers::FileSystem.new(path: wake_path)
|
|
416
403
|
end
|
|
@@ -8,28 +8,26 @@ class Workhorse::PerformerTest < WorkhorseTest
|
|
|
8
8
|
Workhorse.enqueue DbConnectionTestJob.new
|
|
9
9
|
end
|
|
10
10
|
|
|
11
|
-
|
|
11
|
+
work_until(pool_size: 5, polling_interval: 0.2) do
|
|
12
|
+
assert_equal 2, DbConnectionTestJob.db_connections.count
|
|
13
|
+
end
|
|
12
14
|
|
|
13
|
-
assert_equal 2, DbConnectionTestJob.db_connections.count
|
|
14
15
|
assert_equal 2, DbConnectionTestJob.db_connections.uniq.count
|
|
15
16
|
end
|
|
16
17
|
|
|
17
18
|
def test_success
|
|
18
19
|
Workhorse.enqueue BasicJob.new(sleep_time: 0.1)
|
|
19
|
-
|
|
20
|
-
assert_equal 'succeeded', Workhorse::DbJob.first.state
|
|
20
|
+
work_until(pool_size: 5, polling_interval: 0.2) { assert_equal 'succeeded', Workhorse::DbJob.first.state }
|
|
21
21
|
end
|
|
22
22
|
|
|
23
23
|
def test_exception
|
|
24
24
|
Workhorse.enqueue FailingTestJob.new
|
|
25
|
-
|
|
26
|
-
assert_equal 'failed', Workhorse::DbJob.first.state
|
|
25
|
+
work_until(pool_size: 5, polling_interval: 0.2) { assert_equal 'failed', Workhorse::DbJob.first.state }
|
|
27
26
|
end
|
|
28
27
|
|
|
29
28
|
def test_syntax_exception
|
|
30
29
|
Workhorse.enqueue SyntaxErrorJob
|
|
31
|
-
|
|
32
|
-
assert_equal 'failed', Workhorse::DbJob.first.state
|
|
30
|
+
work_until(pool_size: 5, polling_interval: 0.2) { assert_equal 'failed', Workhorse::DbJob.first.state }
|
|
33
31
|
end
|
|
34
32
|
|
|
35
33
|
def test_on_exception
|
|
@@ -41,7 +39,7 @@ class Workhorse::PerformerTest < WorkhorseTest
|
|
|
41
39
|
end
|
|
42
40
|
|
|
43
41
|
Workhorse.enqueue FailingTestJob.new
|
|
44
|
-
|
|
42
|
+
work_until(pool_size: 5, polling_interval: 0.2) { assert exception, 'expected on_exception to be called' }
|
|
45
43
|
|
|
46
44
|
assert_equal exception.message, FailingTestJob::MESSAGE
|
|
47
45
|
ensure
|
|
@@ -19,25 +19,25 @@ class Workhorse::PollerTest < WorkhorseTest
|
|
|
19
19
|
Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: :q2
|
|
20
20
|
Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: :q2
|
|
21
21
|
|
|
22
|
-
assert_equal %w[q1 q2], w
|
|
22
|
+
assert_equal %w[q1 q2], valid_queues_of(w)
|
|
23
23
|
|
|
24
24
|
first_job = Workhorse::DbJob.first
|
|
25
25
|
first_job.mark_locked!(42)
|
|
26
26
|
|
|
27
|
-
assert_equal %w[q2], w
|
|
27
|
+
assert_equal %w[q2], valid_queues_of(w)
|
|
28
28
|
|
|
29
29
|
first_job.mark_started!
|
|
30
30
|
|
|
31
|
-
assert_equal %w[q2], w
|
|
31
|
+
assert_equal %w[q2], valid_queues_of(w)
|
|
32
32
|
|
|
33
33
|
first_job.mark_succeeded!
|
|
34
34
|
|
|
35
|
-
assert_equal %w[q1 q2], w
|
|
35
|
+
assert_equal %w[q1 q2], valid_queues_of(w)
|
|
36
36
|
|
|
37
37
|
last_job = Workhorse::DbJob.last
|
|
38
38
|
last_job.mark_locked!(42)
|
|
39
39
|
|
|
40
|
-
assert_equal %w[q1], w
|
|
40
|
+
assert_equal %w[q1], valid_queues_of(w)
|
|
41
41
|
|
|
42
42
|
begin
|
|
43
43
|
fail 'Some exception'
|
|
@@ -45,34 +45,36 @@ class Workhorse::PollerTest < WorkhorseTest
|
|
|
45
45
|
last_job.mark_failed!(e)
|
|
46
46
|
end
|
|
47
47
|
|
|
48
|
-
assert_equal %w[q1 q2], w
|
|
48
|
+
assert_equal %w[q1 q2], valid_queues_of(w)
|
|
49
49
|
end
|
|
50
50
|
|
|
51
51
|
def test_valid_queues_2
|
|
52
52
|
w = Workhorse::Worker.new(polling_interval: 60)
|
|
53
53
|
|
|
54
|
-
assert_equal [], w
|
|
54
|
+
assert_equal [], valid_queues_of(w)
|
|
55
55
|
|
|
56
56
|
Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: nil
|
|
57
57
|
|
|
58
|
-
assert_equal [nil], w
|
|
58
|
+
assert_equal [nil], valid_queues_of(w)
|
|
59
59
|
|
|
60
60
|
a_job = Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: :a
|
|
61
61
|
|
|
62
|
-
assert_equal [nil, 'a'], w
|
|
62
|
+
assert_equal [nil, 'a'], valid_queues_of(w)
|
|
63
63
|
|
|
64
64
|
a_job.update_attribute :state, :locked
|
|
65
65
|
|
|
66
|
-
assert_equal [nil], w
|
|
66
|
+
assert_equal [nil], valid_queues_of(w)
|
|
67
67
|
end
|
|
68
68
|
|
|
69
69
|
def test_no_queues
|
|
70
70
|
w = Workhorse::Worker.new(polling_interval: 60)
|
|
71
|
-
assert_equal [], w
|
|
71
|
+
assert_equal [], valid_queues_of(w)
|
|
72
72
|
end
|
|
73
73
|
|
|
74
|
-
# Not every adapter returns a result set from `execute
|
|
75
|
-
#
|
|
74
|
+
# Not every adapter returns a result set from `execute`: the Oracle enhanced
|
|
75
|
+
# adapter, for instance, returns `true` for queries, both before and after
|
|
76
|
+
# version 7.0.0. Querying the valid queues must therefore not rely on the
|
|
77
|
+
# return value of `execute`.
|
|
76
78
|
def test_valid_queues_without_usable_execute
|
|
77
79
|
w = Workhorse::Worker.new(polling_interval: 60)
|
|
78
80
|
|
|
@@ -80,7 +82,7 @@ class Workhorse::PollerTest < WorkhorseTest
|
|
|
80
82
|
Workhorse.enqueue BasicJob.new(sleep_time: 2), queue: :a
|
|
81
83
|
|
|
82
84
|
with_return_value(Workhorse::DbJob.connection, :execute, true) do
|
|
83
|
-
assert_equal [nil, 'a'], w
|
|
85
|
+
assert_equal [nil, 'a'], valid_queues_of(w)
|
|
84
86
|
end
|
|
85
87
|
end
|
|
86
88
|
|
|
@@ -92,11 +94,11 @@ class Workhorse::PollerTest < WorkhorseTest
|
|
|
92
94
|
end
|
|
93
95
|
jobs = Workhorse::DbJob.all
|
|
94
96
|
|
|
95
|
-
assert_equal [nil], w
|
|
97
|
+
assert_equal [nil], valid_queues_of(w)
|
|
96
98
|
|
|
97
99
|
jobs[0].mark_locked!(42)
|
|
98
100
|
|
|
99
|
-
assert_equal [nil], w
|
|
101
|
+
assert_equal [nil], valid_queues_of(w)
|
|
100
102
|
end
|
|
101
103
|
|
|
102
104
|
def test_with_instant_repolling
|
|
@@ -133,10 +135,37 @@ class Workhorse::PollerTest < WorkhorseTest
|
|
|
133
135
|
Workhorse.enqueue BasicJob.new(some_param: i, sleep_time: 0)
|
|
134
136
|
end
|
|
135
137
|
|
|
136
|
-
# Create 10 worker processes that work
|
|
138
|
+
# Create 10 worker processes that work until all 100 jobs have succeeded,
|
|
139
|
+
# and for at least 3s, so that each is still polling while the second
|
|
140
|
+
# batch comes in. Not for a fixed time: every poll holds the global lock
|
|
141
|
+
# throughout, so the workers take turns, and 3s got through as few as 25
|
|
142
|
+
# of the 100 jobs on a loaded CI runner, and 61 on Oracle. What is tested
|
|
143
|
+
# is that none is taken twice and every worker gets a turn, not how fast.
|
|
137
144
|
10.times do
|
|
138
145
|
Process.fork do
|
|
139
|
-
|
|
146
|
+
started = Time.now
|
|
147
|
+
|
|
148
|
+
# A worker that cannot even be shut down would leave waitall below
|
|
149
|
+
# waiting for good, so it reports where it is stuck and gives up.
|
|
150
|
+
Thread.new do
|
|
151
|
+
sleep 90
|
|
152
|
+
report_threads 'Worker process still running after 90s'
|
|
153
|
+
exit!(1)
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
with_worker(pool_size: 1, polling_interval: 0.1, auto_terminate: false) do
|
|
157
|
+
loop do
|
|
158
|
+
elapsed = Time.now - started
|
|
159
|
+
break if elapsed >= 3 && Workhorse::DbJob.succeeded.count >= 100
|
|
160
|
+
|
|
161
|
+
if elapsed >= 60
|
|
162
|
+
report_threads 'Worker process gave up waiting for all jobs to succeed after 60s'
|
|
163
|
+
break
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
sleep 0.1
|
|
167
|
+
end
|
|
168
|
+
end
|
|
140
169
|
ensure
|
|
141
170
|
# Exit without running the at_exit handlers of the test process: one
|
|
142
171
|
# of them is Minitest's, which joins threads this fork did not
|
|
@@ -153,24 +182,51 @@ class Workhorse::PollerTest < WorkhorseTest
|
|
|
153
182
|
Workhorse.enqueue BasicJob.new(sleep_time: 0)
|
|
154
183
|
end
|
|
155
184
|
|
|
156
|
-
# Wait for all forked processes to finish
|
|
185
|
+
# Wait for all forked processes to finish
|
|
157
186
|
Process.waitall
|
|
158
187
|
|
|
159
188
|
total = Workhorse::DbJob.count
|
|
160
189
|
succeeded = Workhorse::DbJob.succeeded.count
|
|
161
|
-
|
|
190
|
+
# In a transaction, as Oracle commits a FOR UPDATE outside of one before
|
|
191
|
+
# its rows are fetched and then fails with "fetch out of sequence".
|
|
192
|
+
used_workers = Workhorse::DbJob.transaction do
|
|
193
|
+
Workhorse::DbJob.lock.pluck(:locked_by).uniq.size
|
|
194
|
+
end
|
|
162
195
|
|
|
163
196
|
# Make sure there are 100 jobs, all jobs have succeeded and that all of the
|
|
164
|
-
# workers have had their turn.
|
|
197
|
+
# workers have had their turn. The jobs that have not are listed, as all
|
|
198
|
+
# a count says is that something got stuck, not where.
|
|
199
|
+
unfinished = Workhorse::DbJob.where.not(state: 'succeeded').order(:id).map do |job|
|
|
200
|
+
"##{job.id} #{job.state}, locked_by #{job.locked_by.inspect}, " \
|
|
201
|
+
"locked_at #{job.locked_at.inspect}, started_at #{job.started_at.inspect}"
|
|
202
|
+
end
|
|
203
|
+
|
|
165
204
|
assert_equal 100, total
|
|
166
|
-
assert_equal 100, succeeded
|
|
205
|
+
assert_equal 100, succeeded, "Jobs that did not succeed:\n#{unfinished.join("\n")}"
|
|
167
206
|
assert_equal 10, used_workers
|
|
168
207
|
end
|
|
169
208
|
|
|
209
|
+
# A poll that finds the global lock taken waits at least as long as it asked
|
|
210
|
+
# to before giving up. Below half a second is what the poller asks for with
|
|
211
|
+
# a short polling interval, and what MySQL and Oracle, which take the timeout
|
|
212
|
+
# as whole seconds, used to round down to not waiting at all.
|
|
213
|
+
def test_contended_global_lock_waits_for_its_timeout
|
|
214
|
+
poller = Workhorse::Worker.new(polling_interval: 0.3).poller
|
|
215
|
+
ran = false
|
|
216
|
+
|
|
217
|
+
elapsed = with_global_lock_held do
|
|
218
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
219
|
+
poller.send(:with_global_lock, timeout: 0.3, count_failures: false) { ran = true }
|
|
220
|
+
Process.clock_gettime(Process::CLOCK_MONOTONIC) - started
|
|
221
|
+
end
|
|
222
|
+
|
|
223
|
+
assert_not ran
|
|
224
|
+
assert_operator elapsed, :>=, 0.25
|
|
225
|
+
end
|
|
226
|
+
|
|
170
227
|
def test_connection_loss
|
|
171
|
-
# rubocop: disable Style/GlobalVars
|
|
228
|
+
# rubocop: disable-next Style/GlobalVars
|
|
172
229
|
$thread_conn = nil
|
|
173
|
-
# rubocop: enable Style/GlobalVars
|
|
174
230
|
|
|
175
231
|
Workhorse.enqueue BasicJob.new(sleep_time: 3)
|
|
176
232
|
|
|
@@ -203,7 +259,14 @@ class Workhorse::PollerTest < WorkhorseTest
|
|
|
203
259
|
Workhorse.clean_stuck_jobs = clean
|
|
204
260
|
with_daemon do
|
|
205
261
|
Workhorse.enqueue BasicJob.new(sleep_time: 5)
|
|
206
|
-
|
|
262
|
+
|
|
263
|
+
# Waited for rather than slept on: until a worker has taken the job,
|
|
264
|
+
# locked_by is empty and the cleanup, which matches on the host in it,
|
|
265
|
+
# cannot recognise the job as one of its own.
|
|
266
|
+
with_retries do
|
|
267
|
+
assert_equal 'started', Workhorse::DbJob.first.state
|
|
268
|
+
end
|
|
269
|
+
|
|
207
270
|
kill_deamon_workers
|
|
208
271
|
|
|
209
272
|
assert_equal 1, Workhorse::DbJob.count
|
|
@@ -269,6 +332,13 @@ class Workhorse::PollerTest < WorkhorseTest
|
|
|
269
332
|
|
|
270
333
|
private
|
|
271
334
|
|
|
335
|
+
# The worker's valid queues, the queueless one first. Sorted here, as they
|
|
336
|
+
# come from a SELECT DISTINCT without ORDER BY: MySQL happens to return NULL
|
|
337
|
+
# first, Oracle does not, and the poller does not depend on the order.
|
|
338
|
+
def valid_queues_of(worker)
|
|
339
|
+
return worker.poller.send(:valid_queues).sort_by { |queue| [queue.nil? ? 0 : 1, queue.to_s] }
|
|
340
|
+
end
|
|
341
|
+
|
|
272
342
|
def kill_deamon_workers
|
|
273
343
|
pids = daemon.workers.map(&:pid)
|
|
274
344
|
pids.each do |pid|
|
|
@@ -362,7 +362,7 @@ class Workhorse::ScheduleTest < WorkhorseTest
|
|
|
362
362
|
end
|
|
363
363
|
|
|
364
364
|
assert_equal 1, Workhorse::DbJob.succeeded.count
|
|
365
|
-
assert exceptions.any?
|
|
365
|
+
assert exceptions.any?(NameError), "expected a NameError, got #{exceptions.map(&:class)}"
|
|
366
366
|
end
|
|
367
367
|
|
|
368
368
|
# An occurrence must not be consumed unless the job for it exists. Were the
|