workhorse 1.5.2 → 2.0.0.rc1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.github/workflows/ruby.yml +137 -1
- data/CHANGELOG.md +150 -0
- data/Gemfile +16 -1
- data/README.md +316 -72
- data/Rakefile +1 -0
- data/VERSION +1 -1
- data/bin/rubocop +5 -1
- data/lib/generators/workhorse/install_generator.rb +10 -1
- data/lib/generators/workhorse/templates/config/initializers/workhorse.rb +55 -0
- data/lib/generators/workhorse/templates/create_table_jobs.rb +15 -2
- data/lib/generators/workhorse/templates/create_table_workhorse_schedules.rb +42 -0
- data/lib/workhorse/daemon/shell_handler.rb +4 -1
- data/lib/workhorse/daemon.rb +57 -7
- data/lib/workhorse/db_job.rb +98 -8
- data/lib/workhorse/enqueuer.rb +51 -8
- data/lib/workhorse/jobs/cleanup_succeeded_jobs.rb +26 -8
- data/lib/workhorse/jobs/detect_late_schedules_job.rb +59 -0
- data/lib/workhorse/notifiers/base.rb +55 -0
- data/lib/workhorse/notifiers/file_system.rb +66 -0
- data/lib/workhorse/notifiers/none.rb +8 -0
- data/lib/workhorse/notifiers/redis.rb +227 -0
- data/lib/workhorse/performer.rb +29 -2
- data/lib/workhorse/poller.rb +303 -21
- data/lib/workhorse/pool.rb +12 -6
- data/lib/workhorse/schedule.rb +288 -0
- data/lib/workhorse/schedules.rb +197 -0
- data/lib/workhorse/worker.rb +102 -31
- data/lib/workhorse.rb +136 -0
- data/test/lib/db_schema.rb +36 -3
- data/test/lib/jobs.rb +29 -0
- data/test/lib/test_helper.rb +113 -20
- data/test/workhorse/daemon_test.rb +33 -0
- data/test/workhorse/db_job_test.rb +2 -4
- data/test/workhorse/notifier_test.rb +487 -0
- data/test/workhorse/performer_test.rb +7 -9
- data/test/workhorse/poller_test.rb +97 -23
- data/test/workhorse/schedule_test.rb +967 -0
- data/test/workhorse/worker_test.rb +201 -76
- data/workhorse.gemspec +6 -5
- metadata +29 -3
|
@@ -0,0 +1,288 @@
|
|
|
1
|
+
module Workhorse
|
|
2
|
+
# A schedule's persisted state: which occurrence is next.
|
|
3
|
+
#
|
|
4
|
+
# This is what makes a scheduled job survive a process that is not running
|
|
5
|
+
# when its time comes. An in-memory scheduler computes the next occurrence
|
|
6
|
+
# from *now*, so an occurrence that passes while it is down never happens and
|
|
7
|
+
# leaves no trace. Here the next occurrence is a row, so a worker that comes
|
|
8
|
+
# back at any later point still sees that it is due and applies the
|
|
9
|
+
# schedule's catch-up policy to it.
|
|
10
|
+
#
|
|
11
|
+
# Rows are reconciled from {Workhorse::Schedules} on worker startup and are
|
|
12
|
+
# materialised into jobs during {Workhorse::Poller#poll}.
|
|
13
|
+
class Schedule < ActiveRecord::Base
|
|
14
|
+
self.table_name = 'workhorse_schedules'
|
|
15
|
+
|
|
16
|
+
# How long a schedule that nothing declares any more is kept before its
|
|
17
|
+
# row is deleted.
|
|
18
|
+
OBSOLETE_GRACE = 24 * 60 * 60
|
|
19
|
+
|
|
20
|
+
# Most candidates {#advance} steps over before giving up. An hour can
|
|
21
|
+
# repeat only once, so this is only ever reached by a schedule with a very
|
|
22
|
+
# short period, whose repeats are bounded by the length of that hour.
|
|
23
|
+
MAX_REPEATED_STEPS = 3600
|
|
24
|
+
|
|
25
|
+
# Returns the schedules whose next occurrence has come.
|
|
26
|
+
#
|
|
27
|
+
# @param now [Time]
|
|
28
|
+
# @return [ActiveRecord::Relation]
|
|
29
|
+
def self.due(now = Time.now)
|
|
30
|
+
return where(enabled: true).where(arel_table[:next_at].lteq(now))
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# Brings the table in line with the registry: inserts schedules that are
|
|
34
|
+
# new, recomputes the next occurrence of those whose cron expression or
|
|
35
|
+
# timezone changed, and deletes those that are gone.
|
|
36
|
+
#
|
|
37
|
+
# Safe to call from several workers at once, as each step is idempotent
|
|
38
|
+
# and a lost race leaves the table in the same state.
|
|
39
|
+
#
|
|
40
|
+
# @param now [Time]
|
|
41
|
+
# @return [void]
|
|
42
|
+
def self.reconcile!(now = Time.now)
|
|
43
|
+
# See Workhorse::Schedule#pending_occurrences for why fugit is never
|
|
44
|
+
# handed a local time.
|
|
45
|
+
now = now.getutc
|
|
46
|
+
definitions = Workhorse::Schedules.definitions
|
|
47
|
+
|
|
48
|
+
definitions.each_value do |definition|
|
|
49
|
+
record = find_by(key: definition.key)
|
|
50
|
+
|
|
51
|
+
if record.nil?
|
|
52
|
+
# A new schedule does not fire immediately: its first occurrence is
|
|
53
|
+
# the next one the expression produces.
|
|
54
|
+
create!(
|
|
55
|
+
key: definition.key,
|
|
56
|
+
cron: definition.cron,
|
|
57
|
+
timezone: definition.timezone,
|
|
58
|
+
next_at: definition.parsed_cron.next_time(now).to_t
|
|
59
|
+
)
|
|
60
|
+
elsif record.cron != definition.cron || record.timezone != definition.timezone
|
|
61
|
+
record.update!(
|
|
62
|
+
cron: definition.cron,
|
|
63
|
+
timezone: definition.timezone,
|
|
64
|
+
next_at: definition.parsed_cron.next_time(now).to_t
|
|
65
|
+
)
|
|
66
|
+
end
|
|
67
|
+
rescue ActiveRecord::RecordNotUnique
|
|
68
|
+
# Another worker inserted the same schedule first, which is the
|
|
69
|
+
# outcome this would have produced anyway.
|
|
70
|
+
nil
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
# Mark everything this process knows about as seen, so that a schedule is
|
|
74
|
+
# only considered gone once no process has claimed it for a while. Were
|
|
75
|
+
# rows deleted as soon as one process did not declare them, a rolling
|
|
76
|
+
# deployment would have the old and the new version delete each other's
|
|
77
|
+
# schedules in turn - and recreating a row resets its next occurrence,
|
|
78
|
+
# losing exactly the occurrence this whole mechanism exists to keep.
|
|
79
|
+
where(key: definitions.keys).update_all(updated_at: now) if definitions.any?
|
|
80
|
+
|
|
81
|
+
where.not(key: definitions.keys)
|
|
82
|
+
.where(arel_table[:updated_at].lt(now - OBSOLETE_GRACE))
|
|
83
|
+
.delete_all
|
|
84
|
+
|
|
85
|
+
return
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
# @return [Workhorse::Schedules::Definition, nil] The registry entry
|
|
89
|
+
def definition
|
|
90
|
+
return Workhorse::Schedules[key]
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
# Returns the occurrences that are due, according to the schedule's
|
|
94
|
+
# catch-up policy, together with the occurrence to wait for next.
|
|
95
|
+
#
|
|
96
|
+
# @param now [Time]
|
|
97
|
+
# @return [Array(Array<Time>, Time)] Occurrences to enqueue and the new
|
|
98
|
+
# `next_at`.
|
|
99
|
+
def pending_occurrences(now = Time.now)
|
|
100
|
+
# EtOrbi resolves a Time's zone by rebuilding it locally and comparing
|
|
101
|
+
# the abbreviation, which fails for the whole repeated hour of a
|
|
102
|
+
# fall-back transition ("Cannot determine timezone from \"CEST\"") when
|
|
103
|
+
# ENV['TZ'] is unset. UTC always resolves, and the cron's own zone still
|
|
104
|
+
# governs the computation. `getutc` rather than `utc`, which would
|
|
105
|
+
# convert the caller's own object in place.
|
|
106
|
+
now = now.getutc
|
|
107
|
+
cron = definition.parsed_cron
|
|
108
|
+
|
|
109
|
+
# Nothing is due, and the schedule must not be rewound: returning the
|
|
110
|
+
# next occurrence after now could move next_at backwards for a caller
|
|
111
|
+
# outside the poller.
|
|
112
|
+
return [[], next_at] if next_at > now
|
|
113
|
+
|
|
114
|
+
# Walked backwards from now rather than forwards from next_at, so that
|
|
115
|
+
# the work is bounded by how many occurrences a policy could use - one,
|
|
116
|
+
# or `max_catch_up` - instead of by the length of the outage. Walking
|
|
117
|
+
# forwards costs about 0.45s per 10 000 occurrences, and this runs while
|
|
118
|
+
# the poller holds the global lock, whose timeout for other workers is
|
|
119
|
+
# at most a second.
|
|
120
|
+
occurrences = []
|
|
121
|
+
cursor = now
|
|
122
|
+
|
|
123
|
+
# An occurrence falling on `now` is due as well, and walking backwards
|
|
124
|
+
# would step straight past it. Truncated to the second, as `match?`
|
|
125
|
+
# ignores the fraction and perform_at must be the occurrence itself
|
|
126
|
+
# rather than the moment this poll happened to run.
|
|
127
|
+
occurrences << Time.at(now.to_i).utc if cron.match?(now)
|
|
128
|
+
|
|
129
|
+
while occurrences.size < wanted_occurrences
|
|
130
|
+
# Normalized on every step, not just the first: fugit hands back a
|
|
131
|
+
# time in the cron's zone, and feeding an ambiguous one straight back
|
|
132
|
+
# in is what raises.
|
|
133
|
+
cursor = cron.previous_time(cursor).to_t.getutc
|
|
134
|
+
break if cursor < next_at
|
|
135
|
+
|
|
136
|
+
occurrences.unshift(cursor)
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
return [apply_catch_up(deduplicate(occurrences), now), advance(cron, now, occurrences)]
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
# Returns how many occurrences the catch-up policy could use at most.
|
|
143
|
+
#
|
|
144
|
+
# @return [Integer]
|
|
145
|
+
# @private
|
|
146
|
+
def wanted_occurrences
|
|
147
|
+
return definition.catch_up == :run ? definition.max_catch_up : 1
|
|
148
|
+
end
|
|
149
|
+
|
|
150
|
+
# Claims this schedule by advancing it to `next_at`, and reports whether
|
|
151
|
+
# the claim succeeded.
|
|
152
|
+
#
|
|
153
|
+
# Written as a compare-and-swap rather than a plain update so that exactly
|
|
154
|
+
# one worker materializes an occurrence even if the claim is ever made
|
|
155
|
+
# without the global lock the poller currently holds.
|
|
156
|
+
#
|
|
157
|
+
# @param new_next_at [Time]
|
|
158
|
+
# @return [Boolean]
|
|
159
|
+
# rubocop:disable Naming/PredicateMethod -- reports whether the mutation
|
|
160
|
+
# took effect, which a predicate name would misrepresent as a question.
|
|
161
|
+
def claim!(new_next_at)
|
|
162
|
+
claimed = self.class
|
|
163
|
+
.where(id: id, next_at: next_at)
|
|
164
|
+
.update_all(next_at: new_next_at, updated_at: Time.now)
|
|
165
|
+
|
|
166
|
+
return claimed == 1
|
|
167
|
+
end
|
|
168
|
+
# rubocop:enable Naming/PredicateMethod
|
|
169
|
+
|
|
170
|
+
# Enqueues the job for the given occurrence.
|
|
171
|
+
#
|
|
172
|
+
# `perform_at` is the occurrence's own time rather than the current one,
|
|
173
|
+
# so that the job carries the moment it was meant to run. The difference
|
|
174
|
+
# to `started_at` is then the lateness of that occurrence, which is what
|
|
175
|
+
# {Workhorse.on_job_late} and any reporting on the table can go by.
|
|
176
|
+
#
|
|
177
|
+
# @param occurrence [Time]
|
|
178
|
+
# @return [Workhorse::DbJob]
|
|
179
|
+
def enqueue!(occurrence)
|
|
180
|
+
db_job = Workhorse.enqueue_job_class(
|
|
181
|
+
definition.job,
|
|
182
|
+
queue: definition.queue,
|
|
183
|
+
priority: definition.priority,
|
|
184
|
+
perform_at: occurrence,
|
|
185
|
+
description: definition.description || "#{definition.job} (#{key})",
|
|
186
|
+
params: definition.params,
|
|
187
|
+
max_lateness: definition.max_lateness,
|
|
188
|
+
expires_at: definition.expires_after ? occurrence + definition.expires_after : nil
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
update_columns(
|
|
192
|
+
last_enqueued_at: Time.now,
|
|
193
|
+
last_occurrence: occurrence,
|
|
194
|
+
last_job_id: db_job.id,
|
|
195
|
+
updated_at: Time.now
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
return db_job
|
|
199
|
+
end
|
|
200
|
+
|
|
201
|
+
private
|
|
202
|
+
|
|
203
|
+
# Returns the occurrence to wait for after the given ones.
|
|
204
|
+
#
|
|
205
|
+
# Skips an occurrence repeating the wall-clock time just emitted: the
|
|
206
|
+
# clocks going back make a local time happen twice, and only
|
|
207
|
+
# {#deduplicate} would see those two instants when they fall in the same
|
|
208
|
+
# walk. At the default `catch_up: :run_once` they never do, as only one
|
|
209
|
+
# occurrence is taken per call.
|
|
210
|
+
#
|
|
211
|
+
# @param cron [Fugit::Cron]
|
|
212
|
+
# @param now [Time]
|
|
213
|
+
# @param occurrences [Array<Time>]
|
|
214
|
+
# @return [Time]
|
|
215
|
+
def advance(cron, now, occurrences)
|
|
216
|
+
next_at = cron.next_time(now).to_t
|
|
217
|
+
|
|
218
|
+
return next_at unless wall_clock_anchored?
|
|
219
|
+
return next_at if occurrences.empty?
|
|
220
|
+
|
|
221
|
+
# Stepped in a loop, not once: an hour holding several occurrences -
|
|
222
|
+
# `0,30 2 * * *` - replays all of them, and consecutive candidates
|
|
223
|
+
# never collide pairwise. Formatted times sort chronologically, and the
|
|
224
|
+
# wall clock advances again once the transition is over, so this
|
|
225
|
+
# terminates; bounded regardless.
|
|
226
|
+
last = wall_clock(occurrences.last)
|
|
227
|
+
|
|
228
|
+
MAX_REPEATED_STEPS.times do
|
|
229
|
+
break if wall_clock(next_at) > last
|
|
230
|
+
|
|
231
|
+
next_at = cron.next_time(next_at.getutc).to_t
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
return next_at
|
|
235
|
+
end
|
|
236
|
+
|
|
237
|
+
# Drops occurrences that fall on the same wall-clock time, see {#advance}.
|
|
238
|
+
#
|
|
239
|
+
# Only for a schedule anchored to one: "at 02:30" means that clock time,
|
|
240
|
+
# which happens once. An expression with a wildcard hour ticks on every
|
|
241
|
+
# real instant instead, so both halves of a repeated hour are genuine and
|
|
242
|
+
# dropping one would silently skip an hour of work once a year.
|
|
243
|
+
#
|
|
244
|
+
# @param occurrences [Array<Time>]
|
|
245
|
+
# @return [Array<Time>]
|
|
246
|
+
def deduplicate(occurrences)
|
|
247
|
+
return occurrences unless wall_clock_anchored?
|
|
248
|
+
|
|
249
|
+
return occurrences.uniq { |occurrence| wall_clock(occurrence) }
|
|
250
|
+
end
|
|
251
|
+
|
|
252
|
+
# @return [Boolean] Whether the expression names an hour rather than
|
|
253
|
+
# ticking within every one
|
|
254
|
+
def wall_clock_anchored?
|
|
255
|
+
return !definition.parsed_cron.hours.nil?
|
|
256
|
+
end
|
|
257
|
+
|
|
258
|
+
# @param time [Time]
|
|
259
|
+
# @return [String] The local time, in the schedule's zone
|
|
260
|
+
def wall_clock(time)
|
|
261
|
+
# The zone may be given as an option or as a trailing token of the
|
|
262
|
+
# expression itself, and fugit reports either.
|
|
263
|
+
zone = definition.parsed_cron.zone || definition.timezone
|
|
264
|
+
local = zone ? time.in_time_zone(zone) : time.getlocal
|
|
265
|
+
|
|
266
|
+
return local.strftime('%F %T')
|
|
267
|
+
end
|
|
268
|
+
|
|
269
|
+
# Applies the catch-up policy to the occurrences that are due.
|
|
270
|
+
#
|
|
271
|
+
# @param occurrences [Array<Time>]
|
|
272
|
+
# @param now [Time]
|
|
273
|
+
# @return [Array<Time>]
|
|
274
|
+
def apply_catch_up(occurrences, now)
|
|
275
|
+
return occurrences if occurrences.empty?
|
|
276
|
+
|
|
277
|
+
case definition.catch_up
|
|
278
|
+
when :run, :run_once
|
|
279
|
+
# Already bounded by #wanted_occurrences.
|
|
280
|
+
return occurrences
|
|
281
|
+
when :skip
|
|
282
|
+
return occurrences.select { |occurrence| now - occurrence <= definition.grace }
|
|
283
|
+
else
|
|
284
|
+
return []
|
|
285
|
+
end
|
|
286
|
+
end
|
|
287
|
+
end
|
|
288
|
+
end
|
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
module Workhorse
|
|
2
|
+
# Registry of scheduled jobs, populated through {Workhorse.schedules}.
|
|
3
|
+
#
|
|
4
|
+
# A schedule's job class and options live here, in code, and are addressed
|
|
5
|
+
# from the database by key. Storing the class in the database instead would
|
|
6
|
+
# turn renaming it into a data migration; keeping it here makes it a
|
|
7
|
+
# reconciliation, which {Workhorse::Schedule.reconcile!} performs on worker
|
|
8
|
+
# startup.
|
|
9
|
+
#
|
|
10
|
+
# @see Workhorse::Schedule
|
|
11
|
+
module Schedules
|
|
12
|
+
# Catch-up policies, deciding what happens to occurrences whose time
|
|
13
|
+
# passed while nothing was materialising them.
|
|
14
|
+
CATCH_UP_POLICIES = %i[run run_once skip].freeze
|
|
15
|
+
|
|
16
|
+
# Default number of missed occurrences a `:run` schedule materialises at
|
|
17
|
+
# once, so that a long outage cannot enqueue thousands of jobs.
|
|
18
|
+
DEFAULT_MAX_CATCH_UP = 10
|
|
19
|
+
|
|
20
|
+
# One entry of the registry.
|
|
21
|
+
class Definition
|
|
22
|
+
attr_reader :key
|
|
23
|
+
attr_reader :job
|
|
24
|
+
attr_reader :cron
|
|
25
|
+
attr_reader :timezone
|
|
26
|
+
attr_reader :queue
|
|
27
|
+
attr_reader :priority
|
|
28
|
+
attr_reader :description
|
|
29
|
+
attr_reader :params
|
|
30
|
+
attr_reader :catch_up
|
|
31
|
+
attr_reader :grace
|
|
32
|
+
attr_reader :max_catch_up
|
|
33
|
+
attr_reader :max_lateness
|
|
34
|
+
attr_reader :expires_after
|
|
35
|
+
|
|
36
|
+
# @see Workhorse::Schedules::Dsl#schedule
|
|
37
|
+
def initialize(key, job:, cron:, timezone: nil, queue: nil, priority: 0, description: nil,
|
|
38
|
+
params: {}, catch_up: :run_once, grace: nil, max_catch_up: DEFAULT_MAX_CATCH_UP,
|
|
39
|
+
max_lateness: nil, expires_after: nil)
|
|
40
|
+
@key = key.to_s
|
|
41
|
+
@job = job.to_s
|
|
42
|
+
@cron = cron.to_s
|
|
43
|
+
@timezone = timezone&.to_s
|
|
44
|
+
@queue = queue
|
|
45
|
+
@priority = priority
|
|
46
|
+
@description = description
|
|
47
|
+
@params = (params || {}).freeze
|
|
48
|
+
@catch_up = catch_up.to_sym
|
|
49
|
+
@grace = grace
|
|
50
|
+
@max_catch_up = max_catch_up
|
|
51
|
+
@max_lateness = max_lateness
|
|
52
|
+
@expires_after = expires_after
|
|
53
|
+
|
|
54
|
+
validate!
|
|
55
|
+
freeze
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
# Returns the parsed cron expression.
|
|
59
|
+
#
|
|
60
|
+
# @return [Fugit::Cron]
|
|
61
|
+
def parsed_cron
|
|
62
|
+
return self.class.parse_cron(cron, timezone)
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
# Parses a cron expression, applying a timezone if one is given. Fugit
|
|
66
|
+
# takes the zone as a trailing token of the expression.
|
|
67
|
+
#
|
|
68
|
+
# @param cron [String]
|
|
69
|
+
# @param timezone [String, nil]
|
|
70
|
+
# @return [Fugit::Cron, nil] nil if the expression cannot be parsed
|
|
71
|
+
def self.parse_cron(cron, timezone = nil)
|
|
72
|
+
return Fugit::Cron.parse(timezone ? "#{cron} #{timezone}" : cron)
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
private
|
|
76
|
+
|
|
77
|
+
def validate!
|
|
78
|
+
fail ArgumentError, 'Schedule key must not be blank.' if key.empty?
|
|
79
|
+
fail ArgumentError, "Schedule #{key.inspect}: job must not be blank." if job.empty?
|
|
80
|
+
|
|
81
|
+
unless CATCH_UP_POLICIES.include?(catch_up)
|
|
82
|
+
fail ArgumentError, "Schedule #{key.inspect}: catch_up must be one of " \
|
|
83
|
+
"#{CATCH_UP_POLICIES.inspect}, got #{catch_up.inspect}."
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
# Without a grace period, ":skip" has no way of telling an occurrence
|
|
87
|
+
# that is merely a moment late from one that is a day late, and would
|
|
88
|
+
# silently drop every occurrence.
|
|
89
|
+
if catch_up == :skip && grace.nil?
|
|
90
|
+
fail ArgumentError, "Schedule #{key.inspect}: catch_up :skip requires a grace period, " \
|
|
91
|
+
'which is how late an occurrence may be and still run.'
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
if catch_up == :run && !(max_catch_up.is_a?(Integer) && max_catch_up >= 1)
|
|
95
|
+
fail ArgumentError, "Schedule #{key.inspect}: max_catch_up must be an Integer >= 1, " \
|
|
96
|
+
"got #{max_catch_up.inspect}."
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
if parsed_cron.nil?
|
|
100
|
+
zone = timezone ? " for timezone #{timezone.inspect}" : nil
|
|
101
|
+
fail ArgumentError, "Schedule #{key.inspect}: #{cron.inspect} is not a valid cron " \
|
|
102
|
+
"expression#{zone}."
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
return
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
# Evaluator for the {Workhorse.schedules} block.
|
|
110
|
+
class Dsl
|
|
111
|
+
# @private
|
|
112
|
+
def initialize(definitions)
|
|
113
|
+
@definitions = definitions
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
# Registers a scheduled job.
|
|
117
|
+
#
|
|
118
|
+
# @param key [String, Symbol] Stable name of this schedule. Occurrences
|
|
119
|
+
# are tracked under it, so renaming it starts a new schedule.
|
|
120
|
+
# @param job [String, Class] Job class. May be a plain class responding
|
|
121
|
+
# to `perform`, an `ActiveJob::Base` subclass, or a
|
|
122
|
+
# `RailsOps::Operation`.
|
|
123
|
+
# @param cron [String] Cron expression in crontab format.
|
|
124
|
+
# @param timezone [String, nil] Timezone the expression is read in, e.g.
|
|
125
|
+
# `'Europe/Zurich'`. Defaults to the process's local time.
|
|
126
|
+
# @param queue [String, Symbol, nil] Queue to enqueue into.
|
|
127
|
+
# @param priority [Integer] Job priority, lower runs first.
|
|
128
|
+
# @param description [String, nil] Description of the enqueued job.
|
|
129
|
+
# @param params [Hash] Parameters for the job. Passed as operation
|
|
130
|
+
# params for RailsOps operations and splatted as keyword arguments
|
|
131
|
+
# otherwise.
|
|
132
|
+
# @param catch_up [Symbol] What to do with occurrences whose time passed
|
|
133
|
+
# while nothing was materialising them:
|
|
134
|
+
# * `:run_once` (default) collapses them into a single job,
|
|
135
|
+
# * `:run` materialises each, up to `max_catch_up`,
|
|
136
|
+
# * `:skip` drops those older than `grace`.
|
|
137
|
+
# @param grace [Numeric, nil] Seconds an occurrence may be late and
|
|
138
|
+
# still run. Required for `catch_up: :skip`.
|
|
139
|
+
# @param max_catch_up [Integer] Most occurrences `catch_up: :run`
|
|
140
|
+
# materialises at once.
|
|
141
|
+
# @param max_lateness [Numeric, nil] Seconds the job may start after its
|
|
142
|
+
# occurrence before {Workhorse.on_job_late} is called.
|
|
143
|
+
# @param expires_after [Numeric, nil] Seconds after its occurrence at
|
|
144
|
+
# which the job is no longer worth running. It is then expired instead
|
|
145
|
+
# of performed, and {Workhorse.on_job_expired} is called.
|
|
146
|
+
# @return [void]
|
|
147
|
+
def schedule(key, **options)
|
|
148
|
+
definition = Definition.new(key, **options)
|
|
149
|
+
|
|
150
|
+
if @definitions.key?(definition.key)
|
|
151
|
+
fail ArgumentError, "Schedule #{definition.key.inspect} is already defined."
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
@definitions[definition.key] = definition
|
|
155
|
+
|
|
156
|
+
return
|
|
157
|
+
end
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
class << self
|
|
161
|
+
# @return [Hash{String => Definition}] The registered schedules by key
|
|
162
|
+
def definitions
|
|
163
|
+
return @definitions ||= {}
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
# Registers schedules, see {Workhorse::Schedules::Dsl#schedule}.
|
|
167
|
+
#
|
|
168
|
+
# @yield Block evaluated against the DSL
|
|
169
|
+
# @return [void]
|
|
170
|
+
def define(&block)
|
|
171
|
+
Dsl.new(definitions).instance_eval(&block)
|
|
172
|
+
|
|
173
|
+
return
|
|
174
|
+
end
|
|
175
|
+
|
|
176
|
+
# @param key [String, Symbol]
|
|
177
|
+
# @return [Definition, nil] The schedule registered under `key`
|
|
178
|
+
def [](key)
|
|
179
|
+
return definitions[key.to_s]
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
# @return [Boolean] Whether any schedule is registered
|
|
183
|
+
def any?
|
|
184
|
+
return definitions.any?
|
|
185
|
+
end
|
|
186
|
+
|
|
187
|
+
# Empties the registry. Intended for testing.
|
|
188
|
+
#
|
|
189
|
+
# @return [void]
|
|
190
|
+
def reset!
|
|
191
|
+
@definitions = {}
|
|
192
|
+
|
|
193
|
+
return
|
|
194
|
+
end
|
|
195
|
+
end
|
|
196
|
+
end
|
|
197
|
+
end
|
data/lib/workhorse/worker.rb
CHANGED
|
@@ -23,6 +23,9 @@ module Workhorse
|
|
|
23
23
|
LOG_REOPEN_SIGNAL = 'HUP'.freeze
|
|
24
24
|
SOFT_RESTART_SIGNAL = 'USR1'.freeze
|
|
25
25
|
|
|
26
|
+
# Seconds a repeated {#shutdown} waits for the one already in progress.
|
|
27
|
+
REPEAT_SHUTDOWN_TIMEOUT = 60
|
|
28
|
+
|
|
26
29
|
# @return [Array<Symbol>] The queues this worker processes
|
|
27
30
|
attr_reader :queues
|
|
28
31
|
|
|
@@ -193,30 +196,56 @@ module Workhorse
|
|
|
193
196
|
end
|
|
194
197
|
|
|
195
198
|
# Shuts down worker and DB poller. Jobs currently being processed are
|
|
196
|
-
# properly finished before this method returns
|
|
197
|
-
#
|
|
199
|
+
# properly finished before this method returns, including for a call that
|
|
200
|
+
# finds another thread already shutting the worker down - such a call
|
|
201
|
+
# waits rather than returning early. Shutting down a worker that was
|
|
202
|
+
# never started does nothing.
|
|
198
203
|
#
|
|
199
204
|
# @return [void]
|
|
200
205
|
def shutdown
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
return if @state == :shutdown
|
|
204
|
-
|
|
205
|
-
# TODO: There is a race-condition with this shutdown:
|
|
206
|
-
# - If the poller is currently locking a job, it may call
|
|
207
|
-
# "worker.perform", which in turn tries to synchronize the same mutex.
|
|
208
|
-
mutex.synchronize do
|
|
209
|
-
assert_state! :running
|
|
206
|
+
transitioned = mutex.synchronize do
|
|
207
|
+
next false unless @state == :running
|
|
210
208
|
|
|
211
209
|
Workhorse.debug_log("[Job worker #{id}] Shutdown starting")
|
|
212
210
|
log 'Shutting down'
|
|
213
211
|
@state = :shutdown
|
|
214
212
|
|
|
213
|
+
next true
|
|
214
|
+
end
|
|
215
|
+
|
|
216
|
+
unless transitioned
|
|
217
|
+
# Another thread is doing the shutting down, or it is already done.
|
|
218
|
+
# Wait for the pool rather than returning, so that this call keeps the
|
|
219
|
+
# promise above that running jobs have finished once it returns. On
|
|
220
|
+
# a worker that was never started nothing will ever shut the pool
|
|
221
|
+
# down, and #wait would block forever.
|
|
222
|
+
#
|
|
223
|
+
# Bounded, because this runs inside the signal handler for the second
|
|
224
|
+
# of the TERM and INT the daemon's stop sends, which joins it: were
|
|
225
|
+
# the shutdown it waits on stuck, an unbounded wait would take the
|
|
226
|
+
# whole process down with it - including its ability to be killed.
|
|
227
|
+
if @state == :shutdown && !@pool.wait(timeout: REPEAT_SHUTDOWN_TIMEOUT)
|
|
228
|
+
log 'Gave up waiting for the concurrent shutdown to finish', :warn
|
|
229
|
+
end
|
|
230
|
+
|
|
231
|
+
return
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
# Deliberately outside the mutex. Stopping the poller waits for the
|
|
235
|
+
# poller thread, and that thread may be in #perform waiting for this
|
|
236
|
+
# very mutex, having just committed the lock on a job. Holding the mutex
|
|
237
|
+
# here would deadlock the two, leaving a worker that ignores TERM.
|
|
238
|
+
begin
|
|
215
239
|
@poller.shutdown
|
|
240
|
+
ensure
|
|
241
|
+
# Even if the poller was already gone - it shuts itself down after an
|
|
242
|
+
# exception - the pool has to be stopped, or #wait never returns and a
|
|
243
|
+
# concurrent shutdown waits forever.
|
|
216
244
|
@pool.shutdown
|
|
217
|
-
log 'Shut down'
|
|
218
|
-
Workhorse.debug_log("[Job worker #{id}] Shutdown complete")
|
|
219
245
|
end
|
|
246
|
+
|
|
247
|
+
log 'Shut down'
|
|
248
|
+
Workhorse.debug_log("[Job worker #{id}] Shutdown complete")
|
|
220
249
|
end
|
|
221
250
|
|
|
222
251
|
# Waits until the worker is shut down. This only happens if {#shutdown} gets
|
|
@@ -262,7 +291,7 @@ module Workhorse
|
|
|
262
291
|
return unless daemon_id
|
|
263
292
|
|
|
264
293
|
path = self.class.heartbeat_file_for(daemon_id)
|
|
265
|
-
|
|
294
|
+
touch_pid_file(path) if path
|
|
266
295
|
rescue StandardError => e
|
|
267
296
|
Workhorse.debug_log("[Job worker #{id}] Heartbeat touch failed: #{e.class}: #{e.message}")
|
|
268
297
|
end
|
|
@@ -272,27 +301,48 @@ module Workhorse
|
|
|
272
301
|
# @param db_job_id [Integer] The ID of the {Workhorse::DbJob} to perform
|
|
273
302
|
# @return [void]
|
|
274
303
|
def perform(db_job_id)
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
304
|
+
mutex.synchronize do
|
|
305
|
+
# The poller commits the lock on a job before posting it here, so the
|
|
306
|
+
# worker may have begun shutting down in between.
|
|
307
|
+
next release(db_job_id) unless @state == :running
|
|
308
|
+
|
|
309
|
+
log "Posting job #{db_job_id} to thread pool"
|
|
310
|
+
|
|
311
|
+
@pool.post do
|
|
312
|
+
begin # rubocop:disable Style/RedundantBegin
|
|
313
|
+
Workhorse::Performer.new(db_job_id, self).perform
|
|
314
|
+
rescue Exception => e
|
|
315
|
+
log %(#{e.message}\n#{e.backtrace.join("\n")}), :error
|
|
316
|
+
Workhorse.on_exception.call(e)
|
|
287
317
|
end
|
|
288
318
|
end
|
|
289
|
-
rescue Exception => e
|
|
290
|
-
Workhorse.on_exception.call(e)
|
|
291
319
|
end
|
|
320
|
+
rescue Exception => e
|
|
321
|
+
Workhorse.on_exception.call(e)
|
|
292
322
|
end
|
|
293
323
|
|
|
294
324
|
private
|
|
295
325
|
|
|
326
|
+
# Resets a job this worker locked but will not run back to `waiting`, so
|
|
327
|
+
# that another worker picks it up.
|
|
328
|
+
#
|
|
329
|
+
# Leaving it locked would be worse than the delay: a locked job in a named
|
|
330
|
+
# queue blocks that queue entirely until somebody resets it by hand.
|
|
331
|
+
#
|
|
332
|
+
# @param db_job_id [Integer] The ID of the {Workhorse::DbJob} to release
|
|
333
|
+
# @return [void]
|
|
334
|
+
# @private
|
|
335
|
+
def release(db_job_id)
|
|
336
|
+
db_job = Workhorse::DbJob.find(db_job_id)
|
|
337
|
+
|
|
338
|
+
return unless db_job.state.to_sym == Workhorse::DbJob::STATE_LOCKED
|
|
339
|
+
|
|
340
|
+
log "Not performing job #{db_job_id} as the worker is shutting down, resetting it to waiting"
|
|
341
|
+
db_job.reset!(true)
|
|
342
|
+
|
|
343
|
+
return
|
|
344
|
+
end
|
|
345
|
+
|
|
296
346
|
# Checks current memory usage and initiates shutdown if limit exceeded.
|
|
297
347
|
#
|
|
298
348
|
# @return [Boolean] True if memory is within limits, false if exceeded
|
|
@@ -313,7 +363,7 @@ module Workhorse
|
|
|
313
363
|
Workhorse.debug_log("[Job worker #{id}] Memory limit exceeded: #{mem}MB > #{max}MB, initiating shutdown")
|
|
314
364
|
|
|
315
365
|
if defined?(Rails)
|
|
316
|
-
|
|
366
|
+
touch_pid_file(self.class.shutdown_file_for(pid))
|
|
317
367
|
end
|
|
318
368
|
|
|
319
369
|
log "Worker process #{id.inspect} memory consumption (RSS) of #{mem}MB exceeds " \
|
|
@@ -394,8 +444,17 @@ module Workhorse
|
|
|
394
444
|
# Create shutdown file for watch to detect
|
|
395
445
|
shutdown_file = self.class.shutdown_file_for(pid)
|
|
396
446
|
if shutdown_file
|
|
397
|
-
|
|
398
|
-
|
|
447
|
+
begin
|
|
448
|
+
touch_pid_file(shutdown_file)
|
|
449
|
+
Workhorse.debug_log("[Job worker #{id}] Shutdown file created: #{shutdown_file}")
|
|
450
|
+
rescue StandardError => e
|
|
451
|
+
# The shutdown has to go ahead regardless: the worker stopped
|
|
452
|
+
# accepting jobs above, so giving up here would leave it taking none
|
|
453
|
+
# and never exiting. Without the file, watch sees a worker that is
|
|
454
|
+
# gone rather than one that asked to be restarted, and starts it all
|
|
455
|
+
# the same.
|
|
456
|
+
log "Could not create shutdown file #{shutdown_file}, shutting down regardless: #{e.message}", :warn
|
|
457
|
+
end
|
|
399
458
|
end
|
|
400
459
|
|
|
401
460
|
# Monitor in a separate thread to avoid blocking the signal handler
|
|
@@ -483,5 +542,17 @@ module Workhorse
|
|
|
483
542
|
Kernel.sleep 0.2
|
|
484
543
|
end
|
|
485
544
|
end
|
|
545
|
+
|
|
546
|
+
# Touches a file in the pids directory, creating the directory first. It
|
|
547
|
+
# is not guaranteed to exist: `tmp` is rarely checked in, so a fresh
|
|
548
|
+
# checkout or deployment has none until something writes there.
|
|
549
|
+
#
|
|
550
|
+
# @param path [Pathname, String]
|
|
551
|
+
# @return [void]
|
|
552
|
+
# @private
|
|
553
|
+
def touch_pid_file(path)
|
|
554
|
+
FileUtils.mkdir_p(File.dirname(path))
|
|
555
|
+
FileUtils.touch(path)
|
|
556
|
+
end
|
|
486
557
|
end
|
|
487
558
|
end
|