workhorse 1.5.2 → 2.0.0.rc1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.github/workflows/ruby.yml +137 -1
- data/CHANGELOG.md +150 -0
- data/Gemfile +16 -1
- data/README.md +316 -72
- data/Rakefile +1 -0
- data/VERSION +1 -1
- data/bin/rubocop +5 -1
- data/lib/generators/workhorse/install_generator.rb +10 -1
- data/lib/generators/workhorse/templates/config/initializers/workhorse.rb +55 -0
- data/lib/generators/workhorse/templates/create_table_jobs.rb +15 -2
- data/lib/generators/workhorse/templates/create_table_workhorse_schedules.rb +42 -0
- data/lib/workhorse/daemon/shell_handler.rb +4 -1
- data/lib/workhorse/daemon.rb +57 -7
- data/lib/workhorse/db_job.rb +98 -8
- data/lib/workhorse/enqueuer.rb +51 -8
- data/lib/workhorse/jobs/cleanup_succeeded_jobs.rb +26 -8
- data/lib/workhorse/jobs/detect_late_schedules_job.rb +59 -0
- data/lib/workhorse/notifiers/base.rb +55 -0
- data/lib/workhorse/notifiers/file_system.rb +66 -0
- data/lib/workhorse/notifiers/none.rb +8 -0
- data/lib/workhorse/notifiers/redis.rb +227 -0
- data/lib/workhorse/performer.rb +29 -2
- data/lib/workhorse/poller.rb +303 -21
- data/lib/workhorse/pool.rb +12 -6
- data/lib/workhorse/schedule.rb +288 -0
- data/lib/workhorse/schedules.rb +197 -0
- data/lib/workhorse/worker.rb +102 -31
- data/lib/workhorse.rb +136 -0
- data/test/lib/db_schema.rb +36 -3
- data/test/lib/jobs.rb +29 -0
- data/test/lib/test_helper.rb +113 -20
- data/test/workhorse/daemon_test.rb +33 -0
- data/test/workhorse/db_job_test.rb +2 -4
- data/test/workhorse/notifier_test.rb +487 -0
- data/test/workhorse/performer_test.rb +7 -9
- data/test/workhorse/poller_test.rb +97 -23
- data/test/workhorse/schedule_test.rb +967 -0
- data/test/workhorse/worker_test.rb +201 -76
- data/workhorse.gemspec +6 -5
- metadata +29 -3
data/lib/workhorse/daemon.rb
CHANGED
|
@@ -3,6 +3,9 @@ module Workhorse
|
|
|
3
3
|
# Provides functionality to start, stop, restart, and monitor worker processes
|
|
4
4
|
# through a simple Ruby DSL.
|
|
5
5
|
class Daemon
|
|
6
|
+
# Seconds spent waiting for a process to disappear after KILL, which it
|
|
7
|
+
# can only survive by being stuck in an uninterruptible syscall.
|
|
8
|
+
KILL_TIMEOUT = 10
|
|
6
9
|
# Internal representation of a worker process.
|
|
7
10
|
# Stores worker metadata and the block to execute.
|
|
8
11
|
class Worker
|
|
@@ -308,9 +311,13 @@ module Workhorse
|
|
|
308
311
|
$0 = process_name(worker)
|
|
309
312
|
# Close inherited lockfile fd to prevent holding the flock after parent exits
|
|
310
313
|
@lockfile&.close
|
|
311
|
-
# Reopen pipes to prevent #107576
|
|
314
|
+
# Reopen pipes to prevent #107576. Not the block form: the descriptor
|
|
315
|
+
# has to outlive this call, as stdout and stderr keep pointing at it
|
|
316
|
+
# for the lifetime of the worker.
|
|
317
|
+
# rubocop:disable Style/FileOpen
|
|
312
318
|
$stdin.reopen File.open(File::NULL, 'r')
|
|
313
|
-
null_out = File.open(File::NULL, 'w')
|
|
319
|
+
null_out = File.open(File::NULL, 'w')
|
|
320
|
+
# rubocop:enable Style/FileOpen
|
|
314
321
|
$stdout.reopen null_out
|
|
315
322
|
$stderr.reopen null_out
|
|
316
323
|
|
|
@@ -318,7 +325,27 @@ module Workhorse
|
|
|
318
325
|
# by slot (see Workhorse::Worker#heartbeat!). Same id as the pidfile.
|
|
319
326
|
ENV['WORKHORSE_DAEMON_WORKER_ID'] = worker.id.to_s
|
|
320
327
|
|
|
321
|
-
|
|
328
|
+
begin
|
|
329
|
+
worker.block.call
|
|
330
|
+
|
|
331
|
+
# Exit without running the at_exit handlers the parent registered:
|
|
332
|
+
# they belong to whatever started the daemon, not to this worker,
|
|
333
|
+
# and one that waits on threads the fork did not inherit hangs a
|
|
334
|
+
# worker that has finished - which then ignores TERM. ShellHandler
|
|
335
|
+
# skips them for the same reason.
|
|
336
|
+
exit!(0)
|
|
337
|
+
rescue Exception => e
|
|
338
|
+
# Reported here because exit! below denies it to every other route
|
|
339
|
+
# out: stderr is /dev/null by now, and an error reporter installed
|
|
340
|
+
# as an at_exit handler never runs. A worker that dies of an
|
|
341
|
+
# unhandled exception would otherwise do so silently.
|
|
342
|
+
begin
|
|
343
|
+
Workhorse.on_exception.call(e)
|
|
344
|
+
rescue Exception # rubocop:disable Lint/SuppressedException
|
|
345
|
+
end
|
|
346
|
+
|
|
347
|
+
exit!(1)
|
|
348
|
+
end
|
|
322
349
|
end
|
|
323
350
|
worker.pid = pid
|
|
324
351
|
File.write(pid_file_for(worker), pid)
|
|
@@ -337,18 +364,41 @@ module Workhorse
|
|
|
337
364
|
signals = kill ? %w[KILL] : %w[TERM INT]
|
|
338
365
|
|
|
339
366
|
Workhorse.debug_log("Daemon: stopping PID #{pid} with signals #{signals.join(', ')}")
|
|
367
|
+
|
|
368
|
+
unless signal_until_gone(pid, signals, Workhorse.shutdown_timeout)
|
|
369
|
+
# A worker that does not go away leaves `stop` - and the deployment
|
|
370
|
+
# waiting on it - hanging indefinitely, so it is escalated the way an
|
|
371
|
+
# init system would.
|
|
372
|
+
warn "Worker #{pid} did not stop within #{Workhorse.shutdown_timeout}s, killing it"
|
|
373
|
+
Workhorse.debug_log("Daemon: PID #{pid} did not stop in time, sending KILL")
|
|
374
|
+
signal_until_gone(pid, %w[KILL], KILL_TIMEOUT)
|
|
375
|
+
end
|
|
376
|
+
|
|
377
|
+
Workhorse.debug_log("Daemon: PID #{pid} stopped")
|
|
378
|
+
FileUtils.rm_f(pid_file)
|
|
379
|
+
end
|
|
380
|
+
|
|
381
|
+
# Sends the given signals once a second until the process is gone.
|
|
382
|
+
#
|
|
383
|
+
# @param pid [Integer] The process to signal
|
|
384
|
+
# @param signals [Array<String>] Signals to send on each attempt
|
|
385
|
+
# @param timeout [Numeric, nil] Seconds to keep trying, or nil for no limit
|
|
386
|
+
# @return [Boolean] Whether the process is gone
|
|
387
|
+
# @private
|
|
388
|
+
def signal_until_gone(pid, signals, timeout)
|
|
389
|
+
deadline = timeout ? Process.clock_gettime(Process::CLOCK_MONOTONIC) + timeout : nil
|
|
390
|
+
|
|
340
391
|
loop do
|
|
341
392
|
begin
|
|
342
393
|
signals.each { |signal| Process.kill(signal, pid) }
|
|
343
394
|
rescue Errno::ESRCH
|
|
344
|
-
|
|
395
|
+
return true
|
|
345
396
|
end
|
|
346
397
|
|
|
398
|
+
return false if deadline && Process.clock_gettime(Process::CLOCK_MONOTONIC) >= deadline
|
|
399
|
+
|
|
347
400
|
sleep 1
|
|
348
401
|
end
|
|
349
|
-
|
|
350
|
-
Workhorse.debug_log("Daemon: PID #{pid} stopped")
|
|
351
|
-
File.delete(pid_file)
|
|
352
402
|
end
|
|
353
403
|
|
|
354
404
|
# Sends HUP signal to a worker process.
|
data/lib/workhorse/db_job.rb
CHANGED
|
@@ -18,15 +18,22 @@ module Workhorse
|
|
|
18
18
|
STATE_STARTED = :started
|
|
19
19
|
STATE_SUCCEEDED = :succeeded
|
|
20
20
|
STATE_FAILED = :failed
|
|
21
|
+
STATE_EXPIRED = :expired
|
|
21
22
|
|
|
22
23
|
EXP_LOCKED_BY = /^(.*?)\.(\d+?)\.([^.]+)$/
|
|
23
24
|
|
|
24
25
|
if respond_to?(:attr_accessible)
|
|
25
|
-
attr_accessible :queue, :priority, :perform_at, :handler, :description
|
|
26
|
+
attr_accessible :queue, :priority, :perform_at, :handler, :description,
|
|
27
|
+
:expires_at, :max_lateness
|
|
26
28
|
end
|
|
27
29
|
|
|
28
30
|
self.table_name = 'jobs'
|
|
29
31
|
|
|
32
|
+
# Notifying before the commit would wake a worker that cannot see the row
|
|
33
|
+
# yet, sending it back to sleep for a whole polling interval - the very
|
|
34
|
+
# delay the notification exists to avoid.
|
|
35
|
+
after_commit :notify_workers, on: :create
|
|
36
|
+
|
|
30
37
|
# Returns jobs in waiting state.
|
|
31
38
|
#
|
|
32
39
|
# @return [ActiveRecord::Relation] Jobs waiting to be processed
|
|
@@ -62,13 +69,57 @@ module Workhorse
|
|
|
62
69
|
where(state: STATE_FAILED)
|
|
63
70
|
end
|
|
64
71
|
|
|
72
|
+
# Returns jobs that passed their deadline and were therefore not run.
|
|
73
|
+
#
|
|
74
|
+
# @return [ActiveRecord::Relation] Jobs that expired before being started
|
|
75
|
+
def self.expired
|
|
76
|
+
where(state: STATE_EXPIRED)
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
# Marks the job as expired, having passed its deadline before any worker
|
|
80
|
+
# got to it.
|
|
81
|
+
#
|
|
82
|
+
# @raise [RuntimeError] If the job is not in waiting state
|
|
83
|
+
# @private Only to be used by workhorse
|
|
84
|
+
def mark_expired!
|
|
85
|
+
assert_state! STATE_WAITING
|
|
86
|
+
|
|
87
|
+
self.state = STATE_EXPIRED
|
|
88
|
+
save!
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
# Returns how much later than intended this job started, or nil if it has
|
|
92
|
+
# not started or was not given an intended time.
|
|
93
|
+
#
|
|
94
|
+
# @return [Float, nil] Lateness in seconds
|
|
95
|
+
def lateness
|
|
96
|
+
return nil unless started_at && perform_at
|
|
97
|
+
|
|
98
|
+
return started_at - perform_at
|
|
99
|
+
end
|
|
100
|
+
|
|
65
101
|
# Returns a relation with split locked_by field for easier querying.
|
|
66
102
|
# Extracts host, PID, and random string components from locked_by.
|
|
67
103
|
#
|
|
104
|
+
# `locked_by` is "<host>.<pid>.<random>", and the host may itself contain
|
|
105
|
+
# dots, so both dialects below work backwards from the last two.
|
|
106
|
+
#
|
|
68
107
|
# @return [ActiveRecord::Relation] Relation with additional computed columns
|
|
69
108
|
# @private
|
|
70
109
|
def self.with_split_locked_by
|
|
71
|
-
select(
|
|
110
|
+
return select(oracle? ? split_locked_by_sql_oracle : split_locked_by_sql_mysql)
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
# @return [Boolean] Whether the connection speaks Oracle
|
|
114
|
+
# @private
|
|
115
|
+
def self.oracle?
|
|
116
|
+
return connection.adapter_name == 'OracleEnhanced'
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
# @return [String]
|
|
120
|
+
# @private
|
|
121
|
+
def self.split_locked_by_sql_mysql
|
|
122
|
+
return <<~SQL
|
|
72
123
|
#{table_name}.*,
|
|
73
124
|
|
|
74
125
|
-- random string
|
|
@@ -91,12 +142,37 @@ module Workhorse
|
|
|
91
142
|
SQL
|
|
92
143
|
end
|
|
93
144
|
|
|
145
|
+
# Oracle has no `substring_index`. `instr` with a negative position counts
|
|
146
|
+
# occurrences from the right, which gives the two delimiters directly.
|
|
147
|
+
#
|
|
148
|
+
# @return [String]
|
|
149
|
+
# @private
|
|
150
|
+
def self.split_locked_by_sql_oracle
|
|
151
|
+
return <<~SQL
|
|
152
|
+
#{table_name}.*,
|
|
153
|
+
|
|
154
|
+
-- random string
|
|
155
|
+
substr(locked_by, instr(locked_by, '.', -1, 1) + 1) as locked_by_rnd,
|
|
156
|
+
|
|
157
|
+
-- pid
|
|
158
|
+
substr(
|
|
159
|
+
locked_by,
|
|
160
|
+
instr(locked_by, '.', -1, 2) + 1,
|
|
161
|
+
instr(locked_by, '.', -1, 1) - instr(locked_by, '.', -1, 2) - 1
|
|
162
|
+
) as locked_by_pid,
|
|
163
|
+
|
|
164
|
+
-- get host
|
|
165
|
+
substr(locked_by, 1, instr(locked_by, '.', -1, 2) - 1) as locked_by_host
|
|
166
|
+
SQL
|
|
167
|
+
end
|
|
168
|
+
|
|
94
169
|
# Resets job to state "waiting" and clears all meta fields
|
|
95
170
|
# set by workhorse in course of processing this job.
|
|
96
171
|
#
|
|
97
|
-
# This is only allowed if the job is in a final state ("succeeded"
|
|
98
|
-
# "failed"), as only those jobs are safe to modify; workhorse
|
|
99
|
-
# these jobs. To reset a job without checking the state it
|
|
172
|
+
# This is only allowed if the job is in a final state ("succeeded",
|
|
173
|
+
# "failed" or "expired"), as only those jobs are safe to modify; workhorse
|
|
174
|
+
# will not touch these jobs. To reset a job without checking the state it
|
|
175
|
+
# is in, set
|
|
100
176
|
# "force" to true. Prior to doing so, ensure that the job is not still being
|
|
101
177
|
# processed by a worker. If possible, shut down all workers before
|
|
102
178
|
# performing a forced reset.
|
|
@@ -111,7 +187,7 @@ module Workhorse
|
|
|
111
187
|
# @raise [RuntimeError] If job is not in a final state and force is false
|
|
112
188
|
def reset!(force = false)
|
|
113
189
|
unless force
|
|
114
|
-
assert_state! STATE_SUCCEEDED, STATE_FAILED
|
|
190
|
+
assert_state! STATE_SUCCEEDED, STATE_FAILED, STATE_EXPIRED
|
|
115
191
|
end
|
|
116
192
|
|
|
117
193
|
self.state = STATE_WAITING
|
|
@@ -136,8 +212,6 @@ module Workhorse
|
|
|
136
212
|
end
|
|
137
213
|
|
|
138
214
|
if locked_at
|
|
139
|
-
# TODO: Remove this debug output
|
|
140
|
-
# puts "Already locked. Job: #{self.id} Worker: #{worker_id}"
|
|
141
215
|
fail "Job #{id} is already locked by #{locked_by.inspect}."
|
|
142
216
|
end
|
|
143
217
|
|
|
@@ -185,6 +259,22 @@ module Workhorse
|
|
|
185
259
|
save!
|
|
186
260
|
end
|
|
187
261
|
|
|
262
|
+
# Announces this job to waiting workers, see {Workhorse.notifier}.
|
|
263
|
+
#
|
|
264
|
+
# Jobs that are not due yet are not announced: a woken worker would find
|
|
265
|
+
# nothing to do. They are picked up by the regular poll once their
|
|
266
|
+
# `perform_at` has passed.
|
|
267
|
+
#
|
|
268
|
+
# @return [void]
|
|
269
|
+
# @private
|
|
270
|
+
def notify_workers
|
|
271
|
+
return if perform_at && perform_at > Time.now
|
|
272
|
+
|
|
273
|
+
Workhorse.notifier.notify(queue: queue)
|
|
274
|
+
|
|
275
|
+
return
|
|
276
|
+
end
|
|
277
|
+
|
|
188
278
|
# Asserts that the job is in one of the specified states.
|
|
189
279
|
#
|
|
190
280
|
# @param states [Array<Symbol>] Valid states for the job
|
data/lib/workhorse/enqueuer.rb
CHANGED
|
@@ -10,15 +10,28 @@ module Workhorse
|
|
|
10
10
|
# @param priority [Integer] Job priority (lower numbers = higher priority)
|
|
11
11
|
# @param perform_at [Time] When to perform the job
|
|
12
12
|
# @param description [String, nil] Optional job description
|
|
13
|
+
# @param expires_at [Time, nil] Deadline after which the job is no longer
|
|
14
|
+
# worth running. It is then set to state `expired` instead of being
|
|
15
|
+
# performed, and {Workhorse.on_job_expired} is called.
|
|
16
|
+
# @param max_lateness [Numeric, nil] Seconds the job may start after its
|
|
17
|
+
# `perform_at` before {Workhorse.on_job_late} is called.
|
|
13
18
|
# @return [Workhorse::DbJob] The created database job record
|
|
14
|
-
def enqueue(job, queue: nil, priority: 0, perform_at: Time.now, description: nil
|
|
15
|
-
|
|
19
|
+
def enqueue(job, queue: nil, priority: 0, perform_at: Time.now, description: nil,
|
|
20
|
+
expires_at: nil, max_lateness: nil)
|
|
21
|
+
attributes = {
|
|
16
22
|
queue: queue,
|
|
17
23
|
priority: priority,
|
|
18
24
|
perform_at: perform_at,
|
|
19
25
|
description: description,
|
|
20
26
|
handler: Marshal.dump(job)
|
|
21
|
-
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
# Only set when used, so that an installation that has not run the
|
|
30
|
+
# migration adding these columns keeps working as before.
|
|
31
|
+
attributes[:expires_at] = expires_at if expires_at
|
|
32
|
+
attributes[:max_lateness] = max_lateness if max_lateness
|
|
33
|
+
|
|
34
|
+
return DbJob.create!(**attributes)
|
|
22
35
|
end
|
|
23
36
|
|
|
24
37
|
# Enqueues an ActiveJob job instance.
|
|
@@ -27,16 +40,23 @@ module Workhorse
|
|
|
27
40
|
# @param perform_at [Time] When to perform the job
|
|
28
41
|
# @param queue [String, Symbol, nil] Optional queue override
|
|
29
42
|
# @param description [String, nil] Optional job description
|
|
43
|
+
# @param priority [Integer, nil] Priority override. Defaults to the job's
|
|
44
|
+
# own priority.
|
|
45
|
+
# @param expires_at [Time, nil] See {#enqueue}
|
|
46
|
+
# @param max_lateness [Numeric, nil] See {#enqueue}
|
|
30
47
|
# @return [Workhorse::DbJob] The created database job record
|
|
31
|
-
def enqueue_active_job(job, perform_at: Time.now, queue: nil, description: nil
|
|
48
|
+
def enqueue_active_job(job, perform_at: Time.now, queue: nil, description: nil,
|
|
49
|
+
priority: nil, expires_at: nil, max_lateness: nil)
|
|
32
50
|
wrapper_job = Jobs::RunActiveJob.new(job.serialize)
|
|
33
51
|
queue ||= job.queue_name if job.queue_name.present?
|
|
34
52
|
db_job = enqueue(
|
|
35
53
|
wrapper_job,
|
|
36
|
-
queue:
|
|
37
|
-
priority:
|
|
38
|
-
perform_at:
|
|
39
|
-
description:
|
|
54
|
+
queue: queue,
|
|
55
|
+
priority: priority || job.priority || 0,
|
|
56
|
+
perform_at: Time.at(perform_at),
|
|
57
|
+
description: description,
|
|
58
|
+
expires_at: expires_at,
|
|
59
|
+
max_lateness: max_lateness
|
|
40
60
|
)
|
|
41
61
|
job.provider_job_id = db_job.id
|
|
42
62
|
return db_job
|
|
@@ -65,5 +85,28 @@ module Workhorse
|
|
|
65
85
|
job = Workhorse::Jobs::RunRailsOp.new(cls, op_args)
|
|
66
86
|
enqueue job, **workhorse_args
|
|
67
87
|
end
|
|
88
|
+
|
|
89
|
+
# Enqueues a job given by class name, working out how it wants to be
|
|
90
|
+
# instantiated: as a RailsOps operation, as an ActiveJob job, or as a
|
|
91
|
+
# plain object responding to `perform`.
|
|
92
|
+
#
|
|
93
|
+
# @param job_class [String, Class] The job class
|
|
94
|
+
# @param params [Hash] Operation params for RailsOps operations, keyword
|
|
95
|
+
# arguments to the constructor otherwise
|
|
96
|
+
# @param options [Hash] Passed on to {#enqueue}
|
|
97
|
+
# @return [Workhorse::DbJob] The created database job record
|
|
98
|
+
def enqueue_job_class(job_class, params: {}, **options)
|
|
99
|
+
cls = job_class.is_a?(String) ? job_class.constantize : job_class
|
|
100
|
+
|
|
101
|
+
if defined?(RailsOps::Operation) && cls < RailsOps::Operation
|
|
102
|
+
return enqueue_op(cls, options, params)
|
|
103
|
+
elsif defined?(ActiveJob::Base) && cls < ActiveJob::Base
|
|
104
|
+
job = params.present? ? cls.new(**params) : cls.new
|
|
105
|
+
return enqueue_active_job(job, **options)
|
|
106
|
+
else
|
|
107
|
+
job = params.present? ? cls.new(**params) : cls.new
|
|
108
|
+
return enqueue(job, **options)
|
|
109
|
+
end
|
|
110
|
+
end
|
|
68
111
|
end
|
|
69
112
|
end
|
|
@@ -6,26 +6,44 @@ module Workhorse::Jobs
|
|
|
6
6
|
# @example Schedule cleanup job
|
|
7
7
|
# Workhorse.enqueue(CleanupSucceededJobs.new(max_age: 30))
|
|
8
8
|
#
|
|
9
|
-
# @example Daily cleanup
|
|
10
|
-
#
|
|
11
|
-
#
|
|
9
|
+
# @example Daily cleanup on a schedule
|
|
10
|
+
# Workhorse.schedules do
|
|
11
|
+
# schedule 'cleanup_jobs', job: 'Workhorse::Jobs::CleanupSucceededJobs', cron: '0 2 * * *'
|
|
12
|
+
# end
|
|
12
13
|
class CleanupSucceededJobs
|
|
14
|
+
# States that are cleaned up unless told otherwise. Both are terminal and
|
|
15
|
+
# will never run again; `expired` is included so that a schedule using
|
|
16
|
+
# `expires_after` and regularly missing its window - the very case the
|
|
17
|
+
# option exists for - cannot grow the table without bound.
|
|
18
|
+
DEFAULT_STATES = [
|
|
19
|
+
Workhorse::DbJob::STATE_SUCCEEDED,
|
|
20
|
+
Workhorse::DbJob::STATE_EXPIRED
|
|
21
|
+
].freeze
|
|
22
|
+
|
|
13
23
|
# Instantiates a new job.
|
|
14
24
|
#
|
|
15
25
|
# @param max_age [Integer] The maximal age of jobs to retain, in days. Will
|
|
16
26
|
# be evaluated at perform time.
|
|
17
|
-
|
|
27
|
+
# @param states [Array<Symbol>] The job states to clean up. Defaults to
|
|
28
|
+
# {DEFAULT_STATES}. Pass `[Workhorse::DbJob::STATE_SUCCEEDED]` to keep
|
|
29
|
+
# expired jobs around.
|
|
30
|
+
def initialize(max_age: 14, states: DEFAULT_STATES)
|
|
18
31
|
@max_age = max_age
|
|
32
|
+
@states = states
|
|
19
33
|
end
|
|
20
34
|
|
|
21
|
-
# Executes the cleanup by deleting old
|
|
35
|
+
# Executes the cleanup by deleting old jobs in the configured states.
|
|
22
36
|
#
|
|
23
37
|
# @return [void]
|
|
24
38
|
def perform
|
|
25
39
|
age_limit = seconds_ago(@max_age)
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
)
|
|
40
|
+
|
|
41
|
+
# An instance of this job enqueued before the upgrade unmarshals without
|
|
42
|
+
# @states, and `where(state: nil)` would quietly match nothing - leaving
|
|
43
|
+
# the table growing, which is what this job is for.
|
|
44
|
+
Workhorse::DbJob.where(state: @states || DEFAULT_STATES)
|
|
45
|
+
.where('UPDATED_AT <= ?', age_limit)
|
|
46
|
+
.delete_all
|
|
29
47
|
end
|
|
30
48
|
|
|
31
49
|
private
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
module Workhorse::Jobs
|
|
2
|
+
# Job that detects schedules nothing is materialising.
|
|
3
|
+
#
|
|
4
|
+
# This is the one check that can catch silence. {Workhorse.on_job_expired}
|
|
5
|
+
# and {Workhorse.on_job_late} are precise but can only fire for a job that
|
|
6
|
+
# exists; if no worker is polling, or the global lock is stuck, occurrences
|
|
7
|
+
# stop being materialised altogether and no callback can report it. A
|
|
8
|
+
# schedule whose `next_at` lies well in the past is the evidence of that.
|
|
9
|
+
#
|
|
10
|
+
# Schedule this regularly, the same way as
|
|
11
|
+
# {Workhorse::Jobs::DetectStaleJobsJob}. Note that it is itself performed by
|
|
12
|
+
# a worker, so it reports a stall only while at least one worker still runs
|
|
13
|
+
# - use it alongside external monitoring, not instead of it.
|
|
14
|
+
#
|
|
15
|
+
# @example Schedule with the default threshold
|
|
16
|
+
# Workhorse.schedules do
|
|
17
|
+
# schedule 'detect_late_schedules',
|
|
18
|
+
# job: 'Workhorse::Jobs::DetectLateSchedulesJob',
|
|
19
|
+
# cron: '*/30 * * * *'
|
|
20
|
+
# end
|
|
21
|
+
class DetectLateSchedulesJob
|
|
22
|
+
# Creates a new late schedule detection job.
|
|
23
|
+
#
|
|
24
|
+
# @param threshold [Integer] Number of seconds a schedule's next
|
|
25
|
+
# occurrence may lie in the past before it is reported.
|
|
26
|
+
# @param keys [Array<String>, nil] If given, only check these schedules.
|
|
27
|
+
# If `nil` (default), all of them are checked.
|
|
28
|
+
def initialize(threshold: 15 * 60, keys: nil)
|
|
29
|
+
@threshold = threshold
|
|
30
|
+
@keys = keys
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# Executes the detection.
|
|
34
|
+
#
|
|
35
|
+
# @return [void]
|
|
36
|
+
# @raise [RuntimeError] If schedules are found that are overdue
|
|
37
|
+
def perform
|
|
38
|
+
# Only schedules this process declares. A row whose declaration is gone
|
|
39
|
+
# is waiting to be cleaned up by reconciliation and nothing materializes
|
|
40
|
+
# it, so reporting it would be a false alarm.
|
|
41
|
+
keys = @keys || Workhorse::Schedules.definitions.keys
|
|
42
|
+
|
|
43
|
+
rel = Workhorse::Schedule.where(enabled: true, key: keys)
|
|
44
|
+
rel = rel.where(Workhorse::Schedule.arel_table[:next_at].lt(@threshold.seconds.ago))
|
|
45
|
+
|
|
46
|
+
overdue = rel.pluck(:key, :next_at)
|
|
47
|
+
|
|
48
|
+
return if overdue.empty?
|
|
49
|
+
|
|
50
|
+
descriptions = overdue.map do |key, next_at|
|
|
51
|
+
"#{key.inspect} (due #{next_at}, #{(Time.now - next_at).round}s ago)"
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
fail "The next occurrence of #{overdue.size} schedule(s) is more than #{@threshold}s in the past, " \
|
|
55
|
+
'which means they are not being materialized. Check that at least one worker is polling and ' \
|
|
56
|
+
"that the global lock is obtainable. Affected: #{descriptions.join(', ')}."
|
|
57
|
+
end
|
|
58
|
+
end
|
|
59
|
+
end
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
module Workhorse
|
|
2
|
+
module Notifiers
|
|
3
|
+
# Interface for notifiers, which let a worker start a job without waiting
|
|
4
|
+
# for its next poll.
|
|
5
|
+
#
|
|
6
|
+
# Polling remains the floor: a notification that is never delivered costs
|
|
7
|
+
# latency and nothing else, as the regular poll still happens after
|
|
8
|
+
# `polling_interval` and picks the job up. Implementations must therefore
|
|
9
|
+
# be best-effort and must never raise from {#notify}, which runs in the
|
|
10
|
+
# process that enqueues a job.
|
|
11
|
+
#
|
|
12
|
+
# Workers do not consume notifications. Instead, {#token} returns a value
|
|
13
|
+
# that changes whenever a notification has arrived, and each poller
|
|
14
|
+
# remembers the last one it saw. This keeps several workers in one process
|
|
15
|
+
# independent of one another, as none of them can take a notification away
|
|
16
|
+
# from the others.
|
|
17
|
+
#
|
|
18
|
+
# @abstract Subclass and override {#notify} and {#token}.
|
|
19
|
+
class Base
|
|
20
|
+
# Announces that a job has been enqueued. Called in the enqueueing
|
|
21
|
+
# process, after the surrounding transaction has committed.
|
|
22
|
+
#
|
|
23
|
+
# Must never raise: an enqueue may not fail because a notification could
|
|
24
|
+
# not be delivered.
|
|
25
|
+
#
|
|
26
|
+
# @param queue [String, Symbol, nil] Queue the job was enqueued into
|
|
27
|
+
# @return [void]
|
|
28
|
+
def notify(queue: nil); end
|
|
29
|
+
|
|
30
|
+
# Prepares this notifier for use by a worker in this process. Called
|
|
31
|
+
# when a worker starts, and may be called several times per process.
|
|
32
|
+
#
|
|
33
|
+
# @return [void]
|
|
34
|
+
def start; end
|
|
35
|
+
|
|
36
|
+
# Releases what {#start} acquired. Called when a worker shuts down.
|
|
37
|
+
#
|
|
38
|
+
# @return [void]
|
|
39
|
+
def stop; end
|
|
40
|
+
|
|
41
|
+
# Returns a value that differs from the previously returned one whenever
|
|
42
|
+
# at least one notification has arrived in between. Pollers compare it
|
|
43
|
+
# with `!=` and hold their own last-seen value, so no ordering is implied
|
|
44
|
+
# and nothing is consumed.
|
|
45
|
+
#
|
|
46
|
+
# Returns nil when nothing can be said, for instance because no
|
|
47
|
+
# notification has ever been sent.
|
|
48
|
+
#
|
|
49
|
+
# @return [Object, nil] Comparable token, or nil
|
|
50
|
+
def token
|
|
51
|
+
return nil
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
module Workhorse
|
|
2
|
+
module Notifiers
|
|
3
|
+
# Notifier that announces an enqueued job by touching a single file, which
|
|
4
|
+
# waiting workers stat once per sleep slice.
|
|
5
|
+
#
|
|
6
|
+
# This costs no database work at all and roughly a microsecond of local CPU
|
|
7
|
+
# per check, on a tick the poller performs anyway, so a worker can react
|
|
8
|
+
# within {Workhorse::Poller::SLEEP_SLICE} rather than within its polling
|
|
9
|
+
# interval.
|
|
10
|
+
#
|
|
11
|
+
# It requires that the enqueueing process and the workers share a
|
|
12
|
+
# filesystem, which in practice means the same host. Where they do not -
|
|
13
|
+
# workers on a machine of their own, a console on another host - the touch
|
|
14
|
+
# simply never reaches those workers and they fall back to polling. Use
|
|
15
|
+
# {Workhorse::Notifiers::Redis} if that case has to be fast as well.
|
|
16
|
+
#
|
|
17
|
+
# One file is used for every queue. A worker that serves a subset of queues
|
|
18
|
+
# therefore wakes for jobs it cannot run and polls once for nothing; this
|
|
19
|
+
# is cheaper than having such a worker scan a directory on every tick.
|
|
20
|
+
#
|
|
21
|
+
# Detection relies on the file's modification time, so the filesystem has
|
|
22
|
+
# to keep sub-second mtimes - every current local filesystem does. On one
|
|
23
|
+
# that does not, two touches within the same second may be seen as one, and
|
|
24
|
+
# the worker waits for its regular poll instead.
|
|
25
|
+
class FileSystem < Base
|
|
26
|
+
# @param path [String, nil] Path of the file to touch. Defaults to
|
|
27
|
+
# {Workhorse.notification_path}.
|
|
28
|
+
def initialize(path: nil)
|
|
29
|
+
super()
|
|
30
|
+
@path = path
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# Resolved when used rather than when constructed, so that
|
|
34
|
+
# {Workhorse.notification_path} can be set in any order relative to
|
|
35
|
+
# {Workhorse.notifier=}.
|
|
36
|
+
#
|
|
37
|
+
# @return [String] Path of the file that is touched
|
|
38
|
+
def path
|
|
39
|
+
return (@path || Workhorse.notification_path).to_s
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
# Touches the notification file, creating it and its directory if
|
|
43
|
+
# necessary.
|
|
44
|
+
#
|
|
45
|
+
# @param queue [String, Symbol, nil] Ignored, see the class description
|
|
46
|
+
# @return [void]
|
|
47
|
+
def notify(queue: nil) # rubocop:disable Lint/UnusedMethodArgument
|
|
48
|
+
FileUtils.mkdir_p(::File.dirname(path))
|
|
49
|
+
FileUtils.touch(path)
|
|
50
|
+
rescue StandardError => e
|
|
51
|
+
# Best-effort by contract: a job must still be enqueued when it cannot
|
|
52
|
+
# be announced. The worker finds it on its next poll.
|
|
53
|
+
Workhorse.debug_log("Notification failed: #{e.class}: #{e.message}")
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# Returns the notification file's modification time.
|
|
57
|
+
#
|
|
58
|
+
# @return [Time, nil] mtime, or nil if the file does not exist yet
|
|
59
|
+
def token
|
|
60
|
+
return ::File.mtime(path)
|
|
61
|
+
rescue SystemCallError
|
|
62
|
+
return nil
|
|
63
|
+
end
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
end
|