workhorse 1.5.2 → 2.0.0.rc1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.github/workflows/ruby.yml +137 -1
- data/CHANGELOG.md +150 -0
- data/Gemfile +16 -1
- data/README.md +316 -72
- data/Rakefile +1 -0
- data/VERSION +1 -1
- data/bin/rubocop +5 -1
- data/lib/generators/workhorse/install_generator.rb +10 -1
- data/lib/generators/workhorse/templates/config/initializers/workhorse.rb +55 -0
- data/lib/generators/workhorse/templates/create_table_jobs.rb +15 -2
- data/lib/generators/workhorse/templates/create_table_workhorse_schedules.rb +42 -0
- data/lib/workhorse/daemon/shell_handler.rb +4 -1
- data/lib/workhorse/daemon.rb +57 -7
- data/lib/workhorse/db_job.rb +98 -8
- data/lib/workhorse/enqueuer.rb +51 -8
- data/lib/workhorse/jobs/cleanup_succeeded_jobs.rb +26 -8
- data/lib/workhorse/jobs/detect_late_schedules_job.rb +59 -0
- data/lib/workhorse/notifiers/base.rb +55 -0
- data/lib/workhorse/notifiers/file_system.rb +66 -0
- data/lib/workhorse/notifiers/none.rb +8 -0
- data/lib/workhorse/notifiers/redis.rb +227 -0
- data/lib/workhorse/performer.rb +29 -2
- data/lib/workhorse/poller.rb +303 -21
- data/lib/workhorse/pool.rb +12 -6
- data/lib/workhorse/schedule.rb +288 -0
- data/lib/workhorse/schedules.rb +197 -0
- data/lib/workhorse/worker.rb +102 -31
- data/lib/workhorse.rb +136 -0
- data/test/lib/db_schema.rb +36 -3
- data/test/lib/jobs.rb +29 -0
- data/test/lib/test_helper.rb +113 -20
- data/test/workhorse/daemon_test.rb +33 -0
- data/test/workhorse/db_job_test.rb +2 -4
- data/test/workhorse/notifier_test.rb +487 -0
- data/test/workhorse/performer_test.rb +7 -9
- data/test/workhorse/poller_test.rb +97 -23
- data/test/workhorse/schedule_test.rb +967 -0
- data/test/workhorse/worker_test.rb +201 -76
- data/workhorse.gemspec +6 -5
- metadata +29 -3
data/lib/workhorse.rb
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
require 'active_record'
|
|
2
2
|
require 'active_support/all'
|
|
3
3
|
require 'concurrent'
|
|
4
|
+
require 'fugit'
|
|
4
5
|
require 'socket'
|
|
5
6
|
require 'uri'
|
|
6
7
|
|
|
@@ -84,6 +85,43 @@ module Workhorse
|
|
|
84
85
|
mattr_accessor :silence_watcher
|
|
85
86
|
self.silence_watcher = false
|
|
86
87
|
|
|
88
|
+
# Callback invoked when a job passed its `expires_at` before any worker got
|
|
89
|
+
# to it. The job is in state `expired` and will not be performed.
|
|
90
|
+
#
|
|
91
|
+
# Workhorse logs an expiry at `warn` regardless of this callback. Set the
|
|
92
|
+
# callback to report it wherever failures belong, e.g.
|
|
93
|
+
#
|
|
94
|
+
# ```ruby
|
|
95
|
+
# config.on_job_expired = proc do |db_job|
|
|
96
|
+
# ExceptionNotifier.notify_exception(
|
|
97
|
+
# StandardError.new("Job #{db_job.id} (#{db_job.description}) expired")
|
|
98
|
+
# )
|
|
99
|
+
# end
|
|
100
|
+
# ```
|
|
101
|
+
#
|
|
102
|
+
# Called outside the global lock, but on the poller thread: a slow callback
|
|
103
|
+
# delays this worker's next poll. Anything it raises is passed to
|
|
104
|
+
# {.on_exception} and does not affect the worker.
|
|
105
|
+
#
|
|
106
|
+
# @return [Proc] The expiry callback
|
|
107
|
+
mattr_accessor :on_job_expired
|
|
108
|
+
self.on_job_expired = proc do |db_job|
|
|
109
|
+
# Do something with this job, i.e. notify about it
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
# Callback invoked when a job started later than its `max_lateness` allows.
|
|
113
|
+
# In contrast to {.on_job_expired} the job does run; it was simply late.
|
|
114
|
+
#
|
|
115
|
+
# Called on the worker thread that performs the job, just before it starts,
|
|
116
|
+
# and receives the job and its lateness in seconds. Anything it raises is
|
|
117
|
+
# passed to {.on_exception}.
|
|
118
|
+
#
|
|
119
|
+
# @return [Proc] The lateness callback
|
|
120
|
+
mattr_accessor :on_job_late
|
|
121
|
+
self.on_job_late = proc do |db_job, lateness|
|
|
122
|
+
# Do something with this job, i.e. notify about it
|
|
123
|
+
end
|
|
124
|
+
|
|
87
125
|
# Controls whether jobs are performed within database transactions.
|
|
88
126
|
# Individual job classes can override this with skip_tx?.
|
|
89
127
|
#
|
|
@@ -105,6 +143,55 @@ module Workhorse
|
|
|
105
143
|
mattr_accessor :max_worker_memory_mb
|
|
106
144
|
self.max_worker_memory_mb = 0
|
|
107
145
|
|
|
146
|
+
# Channel a {Workhorse::Notifiers::Redis} notifier publishes on. Defaults to
|
|
147
|
+
# {Workhorse::Notifiers::Redis::DEFAULT_CHANNEL} when nil.
|
|
148
|
+
#
|
|
149
|
+
# @return [String, nil] Channel name
|
|
150
|
+
mattr_accessor :notification_channel
|
|
151
|
+
self.notification_channel = nil
|
|
152
|
+
|
|
153
|
+
# Redis client used by a {Workhorse::Notifiers::Redis} notifier.
|
|
154
|
+
#
|
|
155
|
+
# @return [Object, nil] A Redis client
|
|
156
|
+
mattr_accessor :notification_redis
|
|
157
|
+
self.notification_redis = nil
|
|
158
|
+
|
|
159
|
+
# Path of the file a {Workhorse::Notifiers::FileSystem} notifier touches.
|
|
160
|
+
# Every process that enqueues jobs and every worker must agree on it.
|
|
161
|
+
#
|
|
162
|
+
# Defaults to `tmp/pids/workhorse.wake` below the Rails root, or below the
|
|
163
|
+
# working directory outside of Rails.
|
|
164
|
+
#
|
|
165
|
+
# @return [String] Path of the notification file
|
|
166
|
+
def self.notification_path
|
|
167
|
+
return @notification_path if @notification_path
|
|
168
|
+
|
|
169
|
+
root = defined?(Rails) ? Rails.root : Dir.pwd
|
|
170
|
+
|
|
171
|
+
return ::File.join(root.to_s, 'tmp', 'pids', 'workhorse.wake')
|
|
172
|
+
end
|
|
173
|
+
|
|
174
|
+
# Sets the path of the notification file, see {.notification_path}.
|
|
175
|
+
#
|
|
176
|
+
# @param value [String, nil] Path, or nil to restore the default
|
|
177
|
+
# @return [void]
|
|
178
|
+
def self.notification_path=(value)
|
|
179
|
+
@notification_path = value
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
# Seconds the daemon's `stop` waits for a worker to shut down gracefully
|
|
183
|
+
# before killing it.
|
|
184
|
+
#
|
|
185
|
+
# A worker normally goes away as soon as the job it is performing finishes,
|
|
186
|
+
# so this only takes effect for one that is wedged or performing something
|
|
187
|
+
# very long. Without a limit `stop` waits forever, and so does whatever is
|
|
188
|
+
# waiting on it. Set to nil to wait indefinitely, which was the behaviour
|
|
189
|
+
# before this setting existed.
|
|
190
|
+
#
|
|
191
|
+
# @return [Numeric, nil] Timeout in seconds, or nil for no limit
|
|
192
|
+
mattr_accessor :shutdown_timeout
|
|
193
|
+
self.shutdown_timeout = 300
|
|
194
|
+
|
|
108
195
|
# Path to a debug log file for diagnosing log rotation and signal handling issues.
|
|
109
196
|
# When set, Workhorse writes timestamped debug entries to this file at key points
|
|
110
197
|
# (worker startup, HUP signal handling, restart-logging command flow).
|
|
@@ -130,6 +217,48 @@ module Workhorse
|
|
|
130
217
|
rescue Exception # rubocop:disable Lint/SuppressedException
|
|
131
218
|
end
|
|
132
219
|
|
|
220
|
+
# Notifier that lets a worker start an enqueued job without waiting for its
|
|
221
|
+
# next poll. Polling stays the floor, so a notification that is not
|
|
222
|
+
# delivered costs latency and nothing else.
|
|
223
|
+
#
|
|
224
|
+
# Set to `:none` (the default), `:file`, `:redis`, or an instance of a
|
|
225
|
+
# {Workhorse::Notifiers::Base} subclass.
|
|
226
|
+
#
|
|
227
|
+
# @return [Workhorse::Notifiers::Base] The configured notifier
|
|
228
|
+
def self.notifier
|
|
229
|
+
return @notifier ||= Workhorse::Notifiers::None.new
|
|
230
|
+
end
|
|
231
|
+
|
|
232
|
+
# Sets the notifier, see {.notifier}.
|
|
233
|
+
#
|
|
234
|
+
# @param value [Symbol, Workhorse::Notifiers::Base] The notifier to use
|
|
235
|
+
# @return [void]
|
|
236
|
+
def self.notifier=(value)
|
|
237
|
+
@notifier = case value
|
|
238
|
+
when :none, nil then Workhorse::Notifiers::None.new
|
|
239
|
+
when :file then Workhorse::Notifiers::FileSystem.new
|
|
240
|
+
when :redis then Workhorse::Notifiers::Redis.new
|
|
241
|
+
when Symbol then fail(ArgumentError, "Unknown notifier #{value.inspect}, use :none, :file or :redis.")
|
|
242
|
+
else value
|
|
243
|
+
end
|
|
244
|
+
end
|
|
245
|
+
|
|
246
|
+
# Registers scheduled jobs, see {Workhorse::Schedules::Dsl#schedule}.
|
|
247
|
+
#
|
|
248
|
+
# ```ruby
|
|
249
|
+
# Workhorse.schedules do
|
|
250
|
+
# schedule 'cleanup_jobs',
|
|
251
|
+
# job: 'Workhorse::Jobs::CleanupSucceededJobs',
|
|
252
|
+
# cron: '10 0 * * *'
|
|
253
|
+
# end
|
|
254
|
+
# ```
|
|
255
|
+
#
|
|
256
|
+
# @yield Block evaluated against {Workhorse::Schedules::Dsl}
|
|
257
|
+
# @return [void]
|
|
258
|
+
def self.schedules(&block)
|
|
259
|
+
Workhorse::Schedules.define(&block)
|
|
260
|
+
end
|
|
261
|
+
|
|
133
262
|
# Configuration method for setting up Workhorse options.
|
|
134
263
|
#
|
|
135
264
|
# @yield [self] Configuration block
|
|
@@ -142,7 +271,13 @@ module Workhorse
|
|
|
142
271
|
end
|
|
143
272
|
end
|
|
144
273
|
|
|
274
|
+
require 'workhorse/notifiers/base'
|
|
275
|
+
require 'workhorse/notifiers/none'
|
|
276
|
+
require 'workhorse/notifiers/file_system'
|
|
277
|
+
require 'workhorse/notifiers/redis'
|
|
145
278
|
require 'workhorse/db_job'
|
|
279
|
+
require 'workhorse/schedules'
|
|
280
|
+
require 'workhorse/schedule'
|
|
146
281
|
require 'workhorse/performer'
|
|
147
282
|
require 'workhorse/poller'
|
|
148
283
|
require 'workhorse/pool'
|
|
@@ -151,6 +286,7 @@ require 'workhorse/jobs/run_rails_op'
|
|
|
151
286
|
require 'workhorse/jobs/run_active_job'
|
|
152
287
|
require 'workhorse/jobs/cleanup_succeeded_jobs'
|
|
153
288
|
require 'workhorse/jobs/detect_stale_jobs_job'
|
|
289
|
+
require 'workhorse/jobs/detect_late_schedules_job'
|
|
154
290
|
|
|
155
291
|
# Daemon functionality is not available on java platforms
|
|
156
292
|
if RUBY_PLATFORM != 'java'
|
data/test/lib/db_schema.rb
CHANGED
|
@@ -1,10 +1,21 @@
|
|
|
1
1
|
ActiveRecord::Schema.define do
|
|
2
2
|
self.verbose = false
|
|
3
3
|
|
|
4
|
+
# Prefix lengths for the indexes on string columns. MySQL needs them, as the
|
|
5
|
+
# default `utf8mb4` charset puts a full `varchar(255)` past the maximum key
|
|
6
|
+
# length; Oracle indexes the whole column and rejects the option.
|
|
7
|
+
state_length = DB_ORACLE ? {} : { length: { state: 191 } }
|
|
8
|
+
queue_length = DB_ORACLE ? {} : { length: 191 }
|
|
9
|
+
key_length = DB_ORACLE ? {} : { length: 191 }
|
|
10
|
+
|
|
4
11
|
create_table :jobs, force: true do |t|
|
|
5
12
|
t.string :state, null: false, default: 'waiting'
|
|
6
13
|
t.string :queue, null: true
|
|
7
|
-
|
|
14
|
+
|
|
15
|
+
# Binary rather than text, matching the generated migration: the handler is
|
|
16
|
+
# a `Marshal.dump`, and a text column is character data - on Oracle a CLOB,
|
|
17
|
+
# whose character set conversion would corrupt it.
|
|
18
|
+
t.binary :handler, null: false, limit: 4_294_967_295
|
|
8
19
|
|
|
9
20
|
t.string :locked_by
|
|
10
21
|
t.datetime :locked_at
|
|
@@ -17,13 +28,35 @@ ActiveRecord::Schema.define do
|
|
|
17
28
|
|
|
18
29
|
t.integer :priority, null: false
|
|
19
30
|
t.datetime :perform_at, null: true
|
|
31
|
+
t.datetime :expires_at, null: true
|
|
32
|
+
t.integer :max_lateness, null: true
|
|
20
33
|
|
|
21
34
|
t.string :description, null: true
|
|
22
35
|
|
|
23
36
|
t.timestamps null: false
|
|
24
37
|
end
|
|
25
38
|
|
|
26
|
-
|
|
27
|
-
|
|
39
|
+
# The index names are given explicitly because the ones Rails would derive
|
|
40
|
+
# exceed the 30 characters Oracle allows before 12.2.
|
|
41
|
+
add_index :jobs, :queue, **queue_length
|
|
42
|
+
add_index :jobs, %i[state perform_at], name: 'idx_jobs_state_perform_at', **state_length
|
|
43
|
+
add_index :jobs, %i[state priority created_at], name: 'idx_jobs_state_prio_created', **state_length
|
|
44
|
+
add_index :jobs, %i[state expires_at], name: 'idx_jobs_state_expires_at', **state_length
|
|
28
45
|
add_index :jobs, :perform_at
|
|
46
|
+
|
|
47
|
+
create_table :workhorse_schedules, force: true do |t|
|
|
48
|
+
t.string :key, null: false
|
|
49
|
+
t.string :cron, null: false
|
|
50
|
+
t.string :timezone, null: true
|
|
51
|
+
t.boolean :enabled, null: false, default: true
|
|
52
|
+
t.datetime :next_at, null: false
|
|
53
|
+
t.datetime :last_enqueued_at, null: true
|
|
54
|
+
t.datetime :last_occurrence, null: true
|
|
55
|
+
t.integer :last_job_id, null: true
|
|
56
|
+
|
|
57
|
+
t.timestamps null: false
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
add_index :workhorse_schedules, :key, unique: true, name: 'idx_wh_schedules_key', **key_length
|
|
61
|
+
add_index :workhorse_schedules, %i[enabled next_at], name: 'idx_wh_schedules_due'
|
|
29
62
|
end
|
data/test/lib/jobs.rb
CHANGED
|
@@ -67,3 +67,32 @@ class DummyRailsOpsOp
|
|
|
67
67
|
results << @params
|
|
68
68
|
end
|
|
69
69
|
end
|
|
70
|
+
|
|
71
|
+
class ScheduledActiveJob < ActiveJob::Base
|
|
72
|
+
def perform(*); end
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
# Minimal stand-in for RailsOps, which workhorse only soft-depends on. Its
|
|
76
|
+
# presence is what selects the operation branch of Workhorse::Enqueuer.
|
|
77
|
+
module RailsOps
|
|
78
|
+
class Operation
|
|
79
|
+
def self.results
|
|
80
|
+
return @results ||= Concurrent::Array.new
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
# Workhorse::Jobs::RunRailsOp calls the class method, as RailsOps does.
|
|
84
|
+
def self.run!(params = {})
|
|
85
|
+
return new(params).run!
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
def initialize(params = {})
|
|
89
|
+
@params = params
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
def run!
|
|
93
|
+
self.class.results << @params
|
|
94
|
+
end
|
|
95
|
+
end
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
class DummyScheduledOp < RailsOps::Operation; end
|
data/test/lib/test_helper.rb
CHANGED
|
@@ -6,8 +6,8 @@ require 'benchmark'
|
|
|
6
6
|
require 'concurrent'
|
|
7
7
|
require 'jobs'
|
|
8
8
|
|
|
9
|
-
# The adapter to run the test suite against.
|
|
10
|
-
#
|
|
9
|
+
# The adapter to run the test suite against. Workhorse has to work with each of
|
|
10
|
+
# them, so each is covered by CI.
|
|
11
11
|
DB_ADAPTER = ENV.fetch('DB_ADAPTER', 'mysql2')
|
|
12
12
|
|
|
13
13
|
case DB_ADAPTER
|
|
@@ -15,10 +15,16 @@ when 'mysql2'
|
|
|
15
15
|
require 'mysql2'
|
|
16
16
|
when 'trilogy'
|
|
17
17
|
require 'trilogy'
|
|
18
|
+
when 'oracle_enhanced'
|
|
19
|
+
require 'active_record/connection_adapters/oracle_enhanced_adapter'
|
|
18
20
|
else
|
|
19
|
-
fail "Unsupported DB_ADAPTER #{DB_ADAPTER.inspect}, use 'mysql2' or '
|
|
21
|
+
fail "Unsupported DB_ADAPTER #{DB_ADAPTER.inspect}, use 'mysql2', 'trilogy' or 'oracle_enhanced'."
|
|
20
22
|
end
|
|
21
23
|
|
|
24
|
+
# Whether the suite is running against Oracle. Used where the schema or an
|
|
25
|
+
# assertion cannot be written the same way for both families.
|
|
26
|
+
DB_ORACLE = DB_ADAPTER == 'oracle_enhanced'
|
|
27
|
+
|
|
22
28
|
class MockRailsEnv < String
|
|
23
29
|
def production?
|
|
24
30
|
self == 'production'
|
|
@@ -44,6 +50,38 @@ class Rails
|
|
|
44
50
|
end
|
|
45
51
|
|
|
46
52
|
class WorkhorseTest < ActiveSupport::TestCase
|
|
53
|
+
# Seconds a test may run before the backtraces of all its threads are
|
|
54
|
+
# printed. A hung test otherwise only shows up as a CI attempt killed at its
|
|
55
|
+
# time limit, with nothing saying where it hung.
|
|
56
|
+
HANG_REPORT_AFTER = 120
|
|
57
|
+
|
|
58
|
+
# Callbacks rather than methods, so that a test class defining its own setup
|
|
59
|
+
# or teardown does not skip them.
|
|
60
|
+
setup do
|
|
61
|
+
test = "#{self.class}##{name}"
|
|
62
|
+
|
|
63
|
+
@hang_watchdog = Thread.new do
|
|
64
|
+
sleep HANG_REPORT_AFTER
|
|
65
|
+
report_threads "#{test} is still running after #{HANG_REPORT_AFTER}s"
|
|
66
|
+
end
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
# Prints the backtraces of all threads of this process but the calling one.
|
|
70
|
+
def report_threads(title)
|
|
71
|
+
warn "#{title} (PID #{Process.pid}). Its threads:"
|
|
72
|
+
|
|
73
|
+
Thread.list.each do |thread|
|
|
74
|
+
next if thread == Thread.current
|
|
75
|
+
|
|
76
|
+
warn "--- #{thread.inspect}\n#{(thread.backtrace || ['(no backtrace)']).join("\n")}"
|
|
77
|
+
end
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
teardown do
|
|
81
|
+
@hang_watchdog&.kill
|
|
82
|
+
restore_termination_traps
|
|
83
|
+
end
|
|
84
|
+
|
|
47
85
|
def setup
|
|
48
86
|
remove_pids!
|
|
49
87
|
clear_locks_and_db_threads!
|
|
@@ -68,22 +106,56 @@ class WorkhorseTest < ActiveSupport::TestCase
|
|
|
68
106
|
object.singleton_class.send(:remove_method, method)
|
|
69
107
|
end
|
|
70
108
|
|
|
71
|
-
|
|
72
|
-
|
|
109
|
+
# Holds workhorse's global lock on a connection of its own, so that the code
|
|
110
|
+
# under test sees it as taken by another worker.
|
|
111
|
+
def with_global_lock_held
|
|
112
|
+
connection = ActiveRecord::Base.connection_pool.checkout
|
|
113
|
+
connection.select_value(acquire_global_lock_sql)
|
|
114
|
+
yield
|
|
115
|
+
ensure
|
|
116
|
+
if connection
|
|
117
|
+
connection.select_value(release_global_lock_sql)
|
|
118
|
+
ActiveRecord::Base.connection_pool.checkin(connection)
|
|
119
|
+
end
|
|
120
|
+
end
|
|
73
121
|
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
122
|
+
# Mirrors what Workhorse::Poller#with_global_lock emits, so that the lock the
|
|
123
|
+
# code under test tries to take is the same one.
|
|
124
|
+
def acquire_global_lock_sql
|
|
125
|
+
return <<~SQL.strip if DB_ORACLE
|
|
126
|
+
SELECT DBMS_LOCK.REQUEST(#{Workhorse::Poller::ORACLE_LOCK_HANDLE}, #{Workhorse::Poller::ORACLE_LOCK_MODE}, 1)
|
|
127
|
+
FROM DUAL
|
|
78
128
|
SQL
|
|
79
129
|
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
130
|
+
return "SELECT GET_LOCK(CONCAT(DATABASE(), '_workhorse'), 1)"
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
def release_global_lock_sql
|
|
134
|
+
return "SELECT DBMS_LOCK.RELEASE(#{Workhorse::Poller::ORACLE_LOCK_HANDLE}) FROM DUAL" if DB_ORACLE
|
|
135
|
+
|
|
136
|
+
return "SELECT RELEASE_LOCK(CONCAT(DATABASE(), '_workhorse'))"
|
|
137
|
+
end
|
|
85
138
|
|
|
86
|
-
|
|
139
|
+
def clear_locks_and_db_threads!
|
|
140
|
+
# Releases the locks held by *this* connection, which is all a fresh run
|
|
141
|
+
# needs. It used to kill the queries of every other connection as well, to
|
|
142
|
+
# clear one left behind by a crashed run - but KILL QUERY only aborts a
|
|
143
|
+
# query and does not release a named lock, which is held by the connection
|
|
144
|
+
# rather than the statement, so it never achieved that. What it did
|
|
145
|
+
# achieve was killing queries belonging to the current run, surfacing as
|
|
146
|
+
# spurious "Lost connection to server during query" failures. Should a
|
|
147
|
+
# crashed run ever leave a lock behind, kill its *connection*, which does
|
|
148
|
+
# release it.
|
|
149
|
+
if DB_ORACLE
|
|
150
|
+
# Oracle has no "release everything" call, but workhorse takes exactly
|
|
151
|
+
# one handle. RELEASE reports a status rather than raising when the lock
|
|
152
|
+
# is not held, so this needs no guard.
|
|
153
|
+
Workhorse::DbJob.connection.execute(
|
|
154
|
+
"SELECT DBMS_LOCK.RELEASE(#{Workhorse::Poller::ORACLE_LOCK_HANDLE}) FROM DUAL"
|
|
155
|
+
)
|
|
156
|
+
else
|
|
157
|
+
Workhorse::DbJob.connection.execute('SELECT RELEASE_ALL_LOCKS()')
|
|
158
|
+
end
|
|
87
159
|
end
|
|
88
160
|
|
|
89
161
|
def remove_pids!
|
|
@@ -149,8 +221,14 @@ class WorkhorseTest < ActiveSupport::TestCase
|
|
|
149
221
|
w.shutdown
|
|
150
222
|
end
|
|
151
223
|
|
|
224
|
+
# Without auto_terminate unless asked for, as work and work_until have it:
|
|
225
|
+
# a worker that has it traps TERM and INT for the whole process and never
|
|
226
|
+
# gives them back, so the test process ignored the TERM meant to stop it for
|
|
227
|
+
# the rest of the run - and a timed-out CI attempt kept running alongside
|
|
228
|
+
# the next. Every test hands them back afterwards regardless, see the
|
|
229
|
+
# teardown above.
|
|
152
230
|
def with_worker(options = {})
|
|
153
|
-
w = Workhorse::Worker.new(**options)
|
|
231
|
+
w = Workhorse::Worker.new(auto_terminate: false, **options)
|
|
154
232
|
w.start
|
|
155
233
|
begin
|
|
156
234
|
yield(w)
|
|
@@ -176,6 +254,12 @@ class WorkhorseTest < ActiveSupport::TestCase
|
|
|
176
254
|
daemon.stop(quiet: true)
|
|
177
255
|
end
|
|
178
256
|
|
|
257
|
+
# Hands TERM and INT back to the handlers the process started with, which a
|
|
258
|
+
# worker started with auto_terminate replaced.
|
|
259
|
+
def restore_termination_traps
|
|
260
|
+
ORIGINAL_TERMINATION_TRAPS.each { |signal, handler| Signal.trap(signal, handler) }
|
|
261
|
+
end
|
|
262
|
+
|
|
179
263
|
def with_retries(max = 50, interval: 0.1, &_block)
|
|
180
264
|
runs = 0
|
|
181
265
|
|
|
@@ -205,15 +289,24 @@ class WorkhorseTest < ActiveSupport::TestCase
|
|
|
205
289
|
end
|
|
206
290
|
end
|
|
207
291
|
|
|
292
|
+
# On Oracle the "database" is a service name rather than a schema, and the
|
|
293
|
+
# schema is the user the suite connects as.
|
|
208
294
|
ActiveRecord::Base.establish_connection(
|
|
209
295
|
adapter: DB_ADAPTER,
|
|
210
|
-
database: ENV.fetch('DB_NAME', nil) || 'workhorse',
|
|
211
|
-
username: ENV.fetch('DB_USERNAME', nil) || 'root',
|
|
212
|
-
password: ENV.fetch('DB_PASSWORD', nil) || '',
|
|
296
|
+
database: ENV.fetch('DB_NAME', nil) || (DB_ORACLE ? 'FREEPDB1' : 'workhorse'),
|
|
297
|
+
username: ENV.fetch('DB_USERNAME', nil) || (DB_ORACLE ? 'workhorse' : 'root'),
|
|
298
|
+
password: ENV.fetch('DB_PASSWORD', nil) || (DB_ORACLE ? 'workhorse' : ''),
|
|
213
299
|
host: ENV.fetch('DB_HOST', nil) || '127.0.0.1',
|
|
214
|
-
port: ENV.fetch('DB_PORT', nil) || 3306,
|
|
300
|
+
port: ENV.fetch('DB_PORT', nil) || (DB_ORACLE ? 1521 : 3306),
|
|
215
301
|
pool: 10
|
|
216
302
|
)
|
|
217
303
|
|
|
218
304
|
require 'db_schema'
|
|
219
305
|
require 'workhorse'
|
|
306
|
+
|
|
307
|
+
# The handlers the process started with, see #restore_termination_traps.
|
|
308
|
+
ORIGINAL_TERMINATION_TRAPS = Workhorse::Worker::SHUTDOWN_SIGNALS.to_h do |signal|
|
|
309
|
+
handler = Signal.trap(signal, 'DEFAULT')
|
|
310
|
+
Signal.trap(signal, handler)
|
|
311
|
+
[signal, handler]
|
|
312
|
+
end
|
|
@@ -1,6 +1,39 @@
|
|
|
1
1
|
require 'test_helper'
|
|
2
2
|
|
|
3
3
|
class Workhorse::DaemonTest < WorkhorseTest
|
|
4
|
+
# A worker that ignores TERM used to leave `stop` - and whatever is waiting
|
|
5
|
+
# on it, a deployment usually - looping forever.
|
|
6
|
+
def test_stop_kills_a_worker_that_ignores_term
|
|
7
|
+
previous = Workhorse.shutdown_timeout
|
|
8
|
+
Workhorse.shutdown_timeout = 2
|
|
9
|
+
|
|
10
|
+
daemon = Workhorse::Daemon.new(pidfile: 'tmp/pids/stubborn%s.pid') do |d|
|
|
11
|
+
d.worker 'Stubborn' do
|
|
12
|
+
Signal.trap('TERM') { nil }
|
|
13
|
+
Signal.trap('INT') { nil }
|
|
14
|
+
# A bare sleep returns as soon as a handler runs, so it has to be
|
|
15
|
+
# re-entered to actually ignore the signal.
|
|
16
|
+
loop { sleep 1 }
|
|
17
|
+
end
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
daemon.start(quiet: true)
|
|
21
|
+
pid = daemon.workers.first.pid
|
|
22
|
+
|
|
23
|
+
with_retries(50, interval: 0.1) { assert process?(pid) }
|
|
24
|
+
|
|
25
|
+
capture_stderr do
|
|
26
|
+
Timeout.timeout(30) { daemon.stop(quiet: true) }
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
wait_for_process_exit(pid)
|
|
30
|
+
|
|
31
|
+
refute process?(pid)
|
|
32
|
+
ensure
|
|
33
|
+
Workhorse.shutdown_timeout = previous
|
|
34
|
+
FileUtils.rm_f Dir['tmp/pids/stubborn*.pid']
|
|
35
|
+
end
|
|
36
|
+
|
|
4
37
|
def setup
|
|
5
38
|
remove_pids!
|
|
6
39
|
end
|
|
@@ -10,9 +10,7 @@ class Workhorse::DbJobTest < WorkhorseTest
|
|
|
10
10
|
|
|
11
11
|
def test_reset_failed
|
|
12
12
|
job = Workhorse.enqueue FailingTestJob.new
|
|
13
|
-
|
|
14
|
-
job.reload
|
|
15
|
-
assert_equal 'failed', job.state
|
|
13
|
+
work_until(pool_size: 5, polling_interval: 0.2) { assert_equal 'failed', job.reload.state }
|
|
16
14
|
|
|
17
15
|
job.reset!
|
|
18
16
|
|
|
@@ -26,7 +24,7 @@ class Workhorse::DbJobTest < WorkhorseTest
|
|
|
26
24
|
err = assert_raises do
|
|
27
25
|
job.reset!
|
|
28
26
|
end
|
|
29
|
-
assert_equal %(Job #{job.id} is not in state [:succeeded, :failed] but in state "locked".), err.message
|
|
27
|
+
assert_equal %(Job #{job.id} is not in state [:succeeded, :failed, :expired] but in state "locked".), err.message
|
|
30
28
|
end
|
|
31
29
|
|
|
32
30
|
def test_forced_reset
|