workhorse 1.5.2 → 2.0.0.rc1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. checksums.yaml +4 -4
  2. data/.github/workflows/ruby.yml +137 -1
  3. data/CHANGELOG.md +150 -0
  4. data/Gemfile +16 -1
  5. data/README.md +316 -72
  6. data/Rakefile +1 -0
  7. data/VERSION +1 -1
  8. data/bin/rubocop +5 -1
  9. data/lib/generators/workhorse/install_generator.rb +10 -1
  10. data/lib/generators/workhorse/templates/config/initializers/workhorse.rb +55 -0
  11. data/lib/generators/workhorse/templates/create_table_jobs.rb +15 -2
  12. data/lib/generators/workhorse/templates/create_table_workhorse_schedules.rb +42 -0
  13. data/lib/workhorse/daemon/shell_handler.rb +4 -1
  14. data/lib/workhorse/daemon.rb +57 -7
  15. data/lib/workhorse/db_job.rb +98 -8
  16. data/lib/workhorse/enqueuer.rb +51 -8
  17. data/lib/workhorse/jobs/cleanup_succeeded_jobs.rb +26 -8
  18. data/lib/workhorse/jobs/detect_late_schedules_job.rb +59 -0
  19. data/lib/workhorse/notifiers/base.rb +55 -0
  20. data/lib/workhorse/notifiers/file_system.rb +66 -0
  21. data/lib/workhorse/notifiers/none.rb +8 -0
  22. data/lib/workhorse/notifiers/redis.rb +227 -0
  23. data/lib/workhorse/performer.rb +29 -2
  24. data/lib/workhorse/poller.rb +303 -21
  25. data/lib/workhorse/pool.rb +12 -6
  26. data/lib/workhorse/schedule.rb +288 -0
  27. data/lib/workhorse/schedules.rb +197 -0
  28. data/lib/workhorse/worker.rb +102 -31
  29. data/lib/workhorse.rb +136 -0
  30. data/test/lib/db_schema.rb +36 -3
  31. data/test/lib/jobs.rb +29 -0
  32. data/test/lib/test_helper.rb +113 -20
  33. data/test/workhorse/daemon_test.rb +33 -0
  34. data/test/workhorse/db_job_test.rb +2 -4
  35. data/test/workhorse/notifier_test.rb +487 -0
  36. data/test/workhorse/performer_test.rb +7 -9
  37. data/test/workhorse/poller_test.rb +97 -23
  38. data/test/workhorse/schedule_test.rb +967 -0
  39. data/test/workhorse/worker_test.rb +201 -76
  40. data/workhorse.gemspec +6 -5
  41. metadata +29 -3
@@ -0,0 +1,288 @@
1
+ module Workhorse
2
+ # A schedule's persisted state: which occurrence is next.
3
+ #
4
+ # This is what makes a scheduled job survive a process that is not running
5
+ # when its time comes. An in-memory scheduler computes the next occurrence
6
+ # from *now*, so an occurrence that passes while it is down never happens and
7
+ # leaves no trace. Here the next occurrence is a row, so a worker that comes
8
+ # back at any later point still sees that it is due and applies the
9
+ # schedule's catch-up policy to it.
10
+ #
11
+ # Rows are reconciled from {Workhorse::Schedules} on worker startup and are
12
+ # materialised into jobs during {Workhorse::Poller#poll}.
13
+ class Schedule < ActiveRecord::Base
14
+ self.table_name = 'workhorse_schedules'
15
+
16
+ # How long a schedule that nothing declares any more is kept before its
17
+ # row is deleted.
18
+ OBSOLETE_GRACE = 24 * 60 * 60
19
+
20
+ # Most candidates {#advance} steps over before giving up. An hour can
21
+ # repeat only once, so this is only ever reached by a schedule with a very
22
+ # short period, whose repeats are bounded by the length of that hour.
23
+ MAX_REPEATED_STEPS = 3600
24
+
25
+ # Returns the schedules whose next occurrence has come.
26
+ #
27
+ # @param now [Time]
28
+ # @return [ActiveRecord::Relation]
29
+ def self.due(now = Time.now)
30
+ return where(enabled: true).where(arel_table[:next_at].lteq(now))
31
+ end
32
+
33
+ # Brings the table in line with the registry: inserts schedules that are
34
+ # new, recomputes the next occurrence of those whose cron expression or
35
+ # timezone changed, and deletes those that are gone.
36
+ #
37
+ # Safe to call from several workers at once, as each step is idempotent
38
+ # and a lost race leaves the table in the same state.
39
+ #
40
+ # @param now [Time]
41
+ # @return [void]
42
+ def self.reconcile!(now = Time.now)
43
+ # See Workhorse::Schedule#pending_occurrences for why fugit is never
44
+ # handed a local time.
45
+ now = now.getutc
46
+ definitions = Workhorse::Schedules.definitions
47
+
48
+ definitions.each_value do |definition|
49
+ record = find_by(key: definition.key)
50
+
51
+ if record.nil?
52
+ # A new schedule does not fire immediately: its first occurrence is
53
+ # the next one the expression produces.
54
+ create!(
55
+ key: definition.key,
56
+ cron: definition.cron,
57
+ timezone: definition.timezone,
58
+ next_at: definition.parsed_cron.next_time(now).to_t
59
+ )
60
+ elsif record.cron != definition.cron || record.timezone != definition.timezone
61
+ record.update!(
62
+ cron: definition.cron,
63
+ timezone: definition.timezone,
64
+ next_at: definition.parsed_cron.next_time(now).to_t
65
+ )
66
+ end
67
+ rescue ActiveRecord::RecordNotUnique
68
+ # Another worker inserted the same schedule first, which is the
69
+ # outcome this would have produced anyway.
70
+ nil
71
+ end
72
+
73
+ # Mark everything this process knows about as seen, so that a schedule is
74
+ # only considered gone once no process has claimed it for a while. Were
75
+ # rows deleted as soon as one process did not declare them, a rolling
76
+ # deployment would have the old and the new version delete each other's
77
+ # schedules in turn - and recreating a row resets its next occurrence,
78
+ # losing exactly the occurrence this whole mechanism exists to keep.
79
+ where(key: definitions.keys).update_all(updated_at: now) if definitions.any?
80
+
81
+ where.not(key: definitions.keys)
82
+ .where(arel_table[:updated_at].lt(now - OBSOLETE_GRACE))
83
+ .delete_all
84
+
85
+ return
86
+ end
87
+
88
+ # @return [Workhorse::Schedules::Definition, nil] The registry entry
89
+ def definition
90
+ return Workhorse::Schedules[key]
91
+ end
92
+
93
+ # Returns the occurrences that are due, according to the schedule's
94
+ # catch-up policy, together with the occurrence to wait for next.
95
+ #
96
+ # @param now [Time]
97
+ # @return [Array(Array<Time>, Time)] Occurrences to enqueue and the new
98
+ # `next_at`.
99
+ def pending_occurrences(now = Time.now)
100
+ # EtOrbi resolves a Time's zone by rebuilding it locally and comparing
101
+ # the abbreviation, which fails for the whole repeated hour of a
102
+ # fall-back transition ("Cannot determine timezone from \"CEST\"") when
103
+ # ENV['TZ'] is unset. UTC always resolves, and the cron's own zone still
104
+ # governs the computation. `getutc` rather than `utc`, which would
105
+ # convert the caller's own object in place.
106
+ now = now.getutc
107
+ cron = definition.parsed_cron
108
+
109
+ # Nothing is due, and the schedule must not be rewound: returning the
110
+ # next occurrence after now could move next_at backwards for a caller
111
+ # outside the poller.
112
+ return [[], next_at] if next_at > now
113
+
114
+ # Walked backwards from now rather than forwards from next_at, so that
115
+ # the work is bounded by how many occurrences a policy could use - one,
116
+ # or `max_catch_up` - instead of by the length of the outage. Walking
117
+ # forwards costs about 0.45s per 10 000 occurrences, and this runs while
118
+ # the poller holds the global lock, whose timeout for other workers is
119
+ # at most a second.
120
+ occurrences = []
121
+ cursor = now
122
+
123
+ # An occurrence falling on `now` is due as well, and walking backwards
124
+ # would step straight past it. Truncated to the second, as `match?`
125
+ # ignores the fraction and perform_at must be the occurrence itself
126
+ # rather than the moment this poll happened to run.
127
+ occurrences << Time.at(now.to_i).utc if cron.match?(now)
128
+
129
+ while occurrences.size < wanted_occurrences
130
+ # Normalized on every step, not just the first: fugit hands back a
131
+ # time in the cron's zone, and feeding an ambiguous one straight back
132
+ # in is what raises.
133
+ cursor = cron.previous_time(cursor).to_t.getutc
134
+ break if cursor < next_at
135
+
136
+ occurrences.unshift(cursor)
137
+ end
138
+
139
+ return [apply_catch_up(deduplicate(occurrences), now), advance(cron, now, occurrences)]
140
+ end
141
+
142
+ # Returns how many occurrences the catch-up policy could use at most.
143
+ #
144
+ # @return [Integer]
145
+ # @private
146
+ def wanted_occurrences
147
+ return definition.catch_up == :run ? definition.max_catch_up : 1
148
+ end
149
+
150
+ # Claims this schedule by advancing it to `next_at`, and reports whether
151
+ # the claim succeeded.
152
+ #
153
+ # Written as a compare-and-swap rather than a plain update so that exactly
154
+ # one worker materializes an occurrence even if the claim is ever made
155
+ # without the global lock the poller currently holds.
156
+ #
157
+ # @param new_next_at [Time]
158
+ # @return [Boolean]
159
+ # rubocop:disable Naming/PredicateMethod -- reports whether the mutation
160
+ # took effect, which a predicate name would misrepresent as a question.
161
+ def claim!(new_next_at)
162
+ claimed = self.class
163
+ .where(id: id, next_at: next_at)
164
+ .update_all(next_at: new_next_at, updated_at: Time.now)
165
+
166
+ return claimed == 1
167
+ end
168
+ # rubocop:enable Naming/PredicateMethod
169
+
170
+ # Enqueues the job for the given occurrence.
171
+ #
172
+ # `perform_at` is the occurrence's own time rather than the current one,
173
+ # so that the job carries the moment it was meant to run. The difference
174
+ # to `started_at` is then the lateness of that occurrence, which is what
175
+ # {Workhorse.on_job_late} and any reporting on the table can go by.
176
+ #
177
+ # @param occurrence [Time]
178
+ # @return [Workhorse::DbJob]
179
+ def enqueue!(occurrence)
180
+ db_job = Workhorse.enqueue_job_class(
181
+ definition.job,
182
+ queue: definition.queue,
183
+ priority: definition.priority,
184
+ perform_at: occurrence,
185
+ description: definition.description || "#{definition.job} (#{key})",
186
+ params: definition.params,
187
+ max_lateness: definition.max_lateness,
188
+ expires_at: definition.expires_after ? occurrence + definition.expires_after : nil
189
+ )
190
+
191
+ update_columns(
192
+ last_enqueued_at: Time.now,
193
+ last_occurrence: occurrence,
194
+ last_job_id: db_job.id,
195
+ updated_at: Time.now
196
+ )
197
+
198
+ return db_job
199
+ end
200
+
201
+ private
202
+
203
+ # Returns the occurrence to wait for after the given ones.
204
+ #
205
+ # Skips an occurrence repeating the wall-clock time just emitted: the
206
+ # clocks going back make a local time happen twice, and only
207
+ # {#deduplicate} would see those two instants when they fall in the same
208
+ # walk. At the default `catch_up: :run_once` they never do, as only one
209
+ # occurrence is taken per call.
210
+ #
211
+ # @param cron [Fugit::Cron]
212
+ # @param now [Time]
213
+ # @param occurrences [Array<Time>]
214
+ # @return [Time]
215
+ def advance(cron, now, occurrences)
216
+ next_at = cron.next_time(now).to_t
217
+
218
+ return next_at unless wall_clock_anchored?
219
+ return next_at if occurrences.empty?
220
+
221
+ # Stepped in a loop, not once: an hour holding several occurrences -
222
+ # `0,30 2 * * *` - replays all of them, and consecutive candidates
223
+ # never collide pairwise. Formatted times sort chronologically, and the
224
+ # wall clock advances again once the transition is over, so this
225
+ # terminates; bounded regardless.
226
+ last = wall_clock(occurrences.last)
227
+
228
+ MAX_REPEATED_STEPS.times do
229
+ break if wall_clock(next_at) > last
230
+
231
+ next_at = cron.next_time(next_at.getutc).to_t
232
+ end
233
+
234
+ return next_at
235
+ end
236
+
237
+ # Drops occurrences that fall on the same wall-clock time, see {#advance}.
238
+ #
239
+ # Only for a schedule anchored to one: "at 02:30" means that clock time,
240
+ # which happens once. An expression with a wildcard hour ticks on every
241
+ # real instant instead, so both halves of a repeated hour are genuine and
242
+ # dropping one would silently skip an hour of work once a year.
243
+ #
244
+ # @param occurrences [Array<Time>]
245
+ # @return [Array<Time>]
246
+ def deduplicate(occurrences)
247
+ return occurrences unless wall_clock_anchored?
248
+
249
+ return occurrences.uniq { |occurrence| wall_clock(occurrence) }
250
+ end
251
+
252
+ # @return [Boolean] Whether the expression names an hour rather than
253
+ # ticking within every one
254
+ def wall_clock_anchored?
255
+ return !definition.parsed_cron.hours.nil?
256
+ end
257
+
258
+ # @param time [Time]
259
+ # @return [String] The local time, in the schedule's zone
260
+ def wall_clock(time)
261
+ # The zone may be given as an option or as a trailing token of the
262
+ # expression itself, and fugit reports either.
263
+ zone = definition.parsed_cron.zone || definition.timezone
264
+ local = zone ? time.in_time_zone(zone) : time.getlocal
265
+
266
+ return local.strftime('%F %T')
267
+ end
268
+
269
+ # Applies the catch-up policy to the occurrences that are due.
270
+ #
271
+ # @param occurrences [Array<Time>]
272
+ # @param now [Time]
273
+ # @return [Array<Time>]
274
+ def apply_catch_up(occurrences, now)
275
+ return occurrences if occurrences.empty?
276
+
277
+ case definition.catch_up
278
+ when :run, :run_once
279
+ # Already bounded by #wanted_occurrences.
280
+ return occurrences
281
+ when :skip
282
+ return occurrences.select { |occurrence| now - occurrence <= definition.grace }
283
+ else
284
+ return []
285
+ end
286
+ end
287
+ end
288
+ end
@@ -0,0 +1,197 @@
1
+ module Workhorse
2
+ # Registry of scheduled jobs, populated through {Workhorse.schedules}.
3
+ #
4
+ # A schedule's job class and options live here, in code, and are addressed
5
+ # from the database by key. Storing the class in the database instead would
6
+ # turn renaming it into a data migration; keeping it here makes it a
7
+ # reconciliation, which {Workhorse::Schedule.reconcile!} performs on worker
8
+ # startup.
9
+ #
10
+ # @see Workhorse::Schedule
11
+ module Schedules
12
+ # Catch-up policies, deciding what happens to occurrences whose time
13
+ # passed while nothing was materialising them.
14
+ CATCH_UP_POLICIES = %i[run run_once skip].freeze
15
+
16
+ # Default number of missed occurrences a `:run` schedule materialises at
17
+ # once, so that a long outage cannot enqueue thousands of jobs.
18
+ DEFAULT_MAX_CATCH_UP = 10
19
+
20
+ # One entry of the registry.
21
+ class Definition
22
+ attr_reader :key
23
+ attr_reader :job
24
+ attr_reader :cron
25
+ attr_reader :timezone
26
+ attr_reader :queue
27
+ attr_reader :priority
28
+ attr_reader :description
29
+ attr_reader :params
30
+ attr_reader :catch_up
31
+ attr_reader :grace
32
+ attr_reader :max_catch_up
33
+ attr_reader :max_lateness
34
+ attr_reader :expires_after
35
+
36
+ # @see Workhorse::Schedules::Dsl#schedule
37
+ def initialize(key, job:, cron:, timezone: nil, queue: nil, priority: 0, description: nil,
38
+ params: {}, catch_up: :run_once, grace: nil, max_catch_up: DEFAULT_MAX_CATCH_UP,
39
+ max_lateness: nil, expires_after: nil)
40
+ @key = key.to_s
41
+ @job = job.to_s
42
+ @cron = cron.to_s
43
+ @timezone = timezone&.to_s
44
+ @queue = queue
45
+ @priority = priority
46
+ @description = description
47
+ @params = (params || {}).freeze
48
+ @catch_up = catch_up.to_sym
49
+ @grace = grace
50
+ @max_catch_up = max_catch_up
51
+ @max_lateness = max_lateness
52
+ @expires_after = expires_after
53
+
54
+ validate!
55
+ freeze
56
+ end
57
+
58
+ # Returns the parsed cron expression.
59
+ #
60
+ # @return [Fugit::Cron]
61
+ def parsed_cron
62
+ return self.class.parse_cron(cron, timezone)
63
+ end
64
+
65
+ # Parses a cron expression, applying a timezone if one is given. Fugit
66
+ # takes the zone as a trailing token of the expression.
67
+ #
68
+ # @param cron [String]
69
+ # @param timezone [String, nil]
70
+ # @return [Fugit::Cron, nil] nil if the expression cannot be parsed
71
+ def self.parse_cron(cron, timezone = nil)
72
+ return Fugit::Cron.parse(timezone ? "#{cron} #{timezone}" : cron)
73
+ end
74
+
75
+ private
76
+
77
+ def validate!
78
+ fail ArgumentError, 'Schedule key must not be blank.' if key.empty?
79
+ fail ArgumentError, "Schedule #{key.inspect}: job must not be blank." if job.empty?
80
+
81
+ unless CATCH_UP_POLICIES.include?(catch_up)
82
+ fail ArgumentError, "Schedule #{key.inspect}: catch_up must be one of " \
83
+ "#{CATCH_UP_POLICIES.inspect}, got #{catch_up.inspect}."
84
+ end
85
+
86
+ # Without a grace period, ":skip" has no way of telling an occurrence
87
+ # that is merely a moment late from one that is a day late, and would
88
+ # silently drop every occurrence.
89
+ if catch_up == :skip && grace.nil?
90
+ fail ArgumentError, "Schedule #{key.inspect}: catch_up :skip requires a grace period, " \
91
+ 'which is how late an occurrence may be and still run.'
92
+ end
93
+
94
+ if catch_up == :run && !(max_catch_up.is_a?(Integer) && max_catch_up >= 1)
95
+ fail ArgumentError, "Schedule #{key.inspect}: max_catch_up must be an Integer >= 1, " \
96
+ "got #{max_catch_up.inspect}."
97
+ end
98
+
99
+ if parsed_cron.nil?
100
+ zone = timezone ? " for timezone #{timezone.inspect}" : nil
101
+ fail ArgumentError, "Schedule #{key.inspect}: #{cron.inspect} is not a valid cron " \
102
+ "expression#{zone}."
103
+ end
104
+
105
+ return
106
+ end
107
+ end
108
+
109
+ # Evaluator for the {Workhorse.schedules} block.
110
+ class Dsl
111
+ # @private
112
+ def initialize(definitions)
113
+ @definitions = definitions
114
+ end
115
+
116
+ # Registers a scheduled job.
117
+ #
118
+ # @param key [String, Symbol] Stable name of this schedule. Occurrences
119
+ # are tracked under it, so renaming it starts a new schedule.
120
+ # @param job [String, Class] Job class. May be a plain class responding
121
+ # to `perform`, an `ActiveJob::Base` subclass, or a
122
+ # `RailsOps::Operation`.
123
+ # @param cron [String] Cron expression in crontab format.
124
+ # @param timezone [String, nil] Timezone the expression is read in, e.g.
125
+ # `'Europe/Zurich'`. Defaults to the process's local time.
126
+ # @param queue [String, Symbol, nil] Queue to enqueue into.
127
+ # @param priority [Integer] Job priority, lower runs first.
128
+ # @param description [String, nil] Description of the enqueued job.
129
+ # @param params [Hash] Parameters for the job. Passed as operation
130
+ # params for RailsOps operations and splatted as keyword arguments
131
+ # otherwise.
132
+ # @param catch_up [Symbol] What to do with occurrences whose time passed
133
+ # while nothing was materialising them:
134
+ # * `:run_once` (default) collapses them into a single job,
135
+ # * `:run` materialises each, up to `max_catch_up`,
136
+ # * `:skip` drops those older than `grace`.
137
+ # @param grace [Numeric, nil] Seconds an occurrence may be late and
138
+ # still run. Required for `catch_up: :skip`.
139
+ # @param max_catch_up [Integer] Most occurrences `catch_up: :run`
140
+ # materialises at once.
141
+ # @param max_lateness [Numeric, nil] Seconds the job may start after its
142
+ # occurrence before {Workhorse.on_job_late} is called.
143
+ # @param expires_after [Numeric, nil] Seconds after its occurrence at
144
+ # which the job is no longer worth running. It is then expired instead
145
+ # of performed, and {Workhorse.on_job_expired} is called.
146
+ # @return [void]
147
+ def schedule(key, **options)
148
+ definition = Definition.new(key, **options)
149
+
150
+ if @definitions.key?(definition.key)
151
+ fail ArgumentError, "Schedule #{definition.key.inspect} is already defined."
152
+ end
153
+
154
+ @definitions[definition.key] = definition
155
+
156
+ return
157
+ end
158
+ end
159
+
160
+ class << self
161
+ # @return [Hash{String => Definition}] The registered schedules by key
162
+ def definitions
163
+ return @definitions ||= {}
164
+ end
165
+
166
+ # Registers schedules, see {Workhorse::Schedules::Dsl#schedule}.
167
+ #
168
+ # @yield Block evaluated against the DSL
169
+ # @return [void]
170
+ def define(&block)
171
+ Dsl.new(definitions).instance_eval(&block)
172
+
173
+ return
174
+ end
175
+
176
+ # @param key [String, Symbol]
177
+ # @return [Definition, nil] The schedule registered under `key`
178
+ def [](key)
179
+ return definitions[key.to_s]
180
+ end
181
+
182
+ # @return [Boolean] Whether any schedule is registered
183
+ def any?
184
+ return definitions.any?
185
+ end
186
+
187
+ # Empties the registry. Intended for testing.
188
+ #
189
+ # @return [void]
190
+ def reset!
191
+ @definitions = {}
192
+
193
+ return
194
+ end
195
+ end
196
+ end
197
+ end
@@ -23,6 +23,9 @@ module Workhorse
23
23
  LOG_REOPEN_SIGNAL = 'HUP'.freeze
24
24
  SOFT_RESTART_SIGNAL = 'USR1'.freeze
25
25
 
26
+ # Seconds a repeated {#shutdown} waits for the one already in progress.
27
+ REPEAT_SHUTDOWN_TIMEOUT = 60
28
+
26
29
  # @return [Array<Symbol>] The queues this worker processes
27
30
  attr_reader :queues
28
31
 
@@ -193,30 +196,56 @@ module Workhorse
193
196
  end
194
197
 
195
198
  # Shuts down worker and DB poller. Jobs currently being processed are
196
- # properly finished before this method returns. Subsequent calls to this
197
- # method are ignored.
199
+ # properly finished before this method returns, including for a call that
200
+ # finds another thread already shutting the worker down - such a call
201
+ # waits rather than returning early. Shutting down a worker that was
202
+ # never started does nothing.
198
203
  #
199
204
  # @return [void]
200
205
  def shutdown
201
- # This is safe to be checked outside of the mutex as 'shutdown' is the
202
- # final state this worker can be in.
203
- return if @state == :shutdown
204
-
205
- # TODO: There is a race-condition with this shutdown:
206
- # - If the poller is currently locking a job, it may call
207
- # "worker.perform", which in turn tries to synchronize the same mutex.
208
- mutex.synchronize do
209
- assert_state! :running
206
+ transitioned = mutex.synchronize do
207
+ next false unless @state == :running
210
208
 
211
209
  Workhorse.debug_log("[Job worker #{id}] Shutdown starting")
212
210
  log 'Shutting down'
213
211
  @state = :shutdown
214
212
 
213
+ next true
214
+ end
215
+
216
+ unless transitioned
217
+ # Another thread is doing the shutting down, or it is already done.
218
+ # Wait for the pool rather than returning, so that this call keeps the
219
+ # promise above that running jobs have finished once it returns. On
220
+ # a worker that was never started nothing will ever shut the pool
221
+ # down, and #wait would block forever.
222
+ #
223
+ # Bounded, because this runs inside the signal handler for the second
224
+ # of the TERM and INT the daemon's stop sends, which joins it: were
225
+ # the shutdown it waits on stuck, an unbounded wait would take the
226
+ # whole process down with it - including its ability to be killed.
227
+ if @state == :shutdown && !@pool.wait(timeout: REPEAT_SHUTDOWN_TIMEOUT)
228
+ log 'Gave up waiting for the concurrent shutdown to finish', :warn
229
+ end
230
+
231
+ return
232
+ end
233
+
234
+ # Deliberately outside the mutex. Stopping the poller waits for the
235
+ # poller thread, and that thread may be in #perform waiting for this
236
+ # very mutex, having just committed the lock on a job. Holding the mutex
237
+ # here would deadlock the two, leaving a worker that ignores TERM.
238
+ begin
215
239
  @poller.shutdown
240
+ ensure
241
+ # Even if the poller was already gone - it shuts itself down after an
242
+ # exception - the pool has to be stopped, or #wait never returns and a
243
+ # concurrent shutdown waits forever.
216
244
  @pool.shutdown
217
- log 'Shut down'
218
- Workhorse.debug_log("[Job worker #{id}] Shutdown complete")
219
245
  end
246
+
247
+ log 'Shut down'
248
+ Workhorse.debug_log("[Job worker #{id}] Shutdown complete")
220
249
  end
221
250
 
222
251
  # Waits until the worker is shut down. This only happens if {#shutdown} gets
@@ -262,7 +291,7 @@ module Workhorse
262
291
  return unless daemon_id
263
292
 
264
293
  path = self.class.heartbeat_file_for(daemon_id)
265
- FileUtils.touch(path) if path
294
+ touch_pid_file(path) if path
266
295
  rescue StandardError => e
267
296
  Workhorse.debug_log("[Job worker #{id}] Heartbeat touch failed: #{e.class}: #{e.message}")
268
297
  end
@@ -272,27 +301,48 @@ module Workhorse
272
301
  # @param db_job_id [Integer] The ID of the {Workhorse::DbJob} to perform
273
302
  # @return [void]
274
303
  def perform(db_job_id)
275
- begin # rubocop:disable Style/RedundantBegin
276
- mutex.synchronize do
277
- assert_state! :running
278
- log "Posting job #{db_job_id} to thread pool"
279
-
280
- @pool.post do
281
- begin # rubocop:disable Style/RedundantBegin
282
- Workhorse::Performer.new(db_job_id, self).perform
283
- rescue Exception => e
284
- log %(#{e.message}\n#{e.backtrace.join("\n")}), :error
285
- Workhorse.on_exception.call(e)
286
- end
304
+ mutex.synchronize do
305
+ # The poller commits the lock on a job before posting it here, so the
306
+ # worker may have begun shutting down in between.
307
+ next release(db_job_id) unless @state == :running
308
+
309
+ log "Posting job #{db_job_id} to thread pool"
310
+
311
+ @pool.post do
312
+ begin # rubocop:disable Style/RedundantBegin
313
+ Workhorse::Performer.new(db_job_id, self).perform
314
+ rescue Exception => e
315
+ log %(#{e.message}\n#{e.backtrace.join("\n")}), :error
316
+ Workhorse.on_exception.call(e)
287
317
  end
288
318
  end
289
- rescue Exception => e
290
- Workhorse.on_exception.call(e)
291
319
  end
320
+ rescue Exception => e
321
+ Workhorse.on_exception.call(e)
292
322
  end
293
323
 
294
324
  private
295
325
 
326
+ # Resets a job this worker locked but will not run back to `waiting`, so
327
+ # that another worker picks it up.
328
+ #
329
+ # Leaving it locked would be worse than the delay: a locked job in a named
330
+ # queue blocks that queue entirely until somebody resets it by hand.
331
+ #
332
+ # @param db_job_id [Integer] The ID of the {Workhorse::DbJob} to release
333
+ # @return [void]
334
+ # @private
335
+ def release(db_job_id)
336
+ db_job = Workhorse::DbJob.find(db_job_id)
337
+
338
+ return unless db_job.state.to_sym == Workhorse::DbJob::STATE_LOCKED
339
+
340
+ log "Not performing job #{db_job_id} as the worker is shutting down, resetting it to waiting"
341
+ db_job.reset!(true)
342
+
343
+ return
344
+ end
345
+
296
346
  # Checks current memory usage and initiates shutdown if limit exceeded.
297
347
  #
298
348
  # @return [Boolean] True if memory is within limits, false if exceeded
@@ -313,7 +363,7 @@ module Workhorse
313
363
  Workhorse.debug_log("[Job worker #{id}] Memory limit exceeded: #{mem}MB > #{max}MB, initiating shutdown")
314
364
 
315
365
  if defined?(Rails)
316
- FileUtils.touch self.class.shutdown_file_for(pid)
366
+ touch_pid_file(self.class.shutdown_file_for(pid))
317
367
  end
318
368
 
319
369
  log "Worker process #{id.inspect} memory consumption (RSS) of #{mem}MB exceeds " \
@@ -394,8 +444,17 @@ module Workhorse
394
444
  # Create shutdown file for watch to detect
395
445
  shutdown_file = self.class.shutdown_file_for(pid)
396
446
  if shutdown_file
397
- FileUtils.touch(shutdown_file)
398
- Workhorse.debug_log("[Job worker #{id}] Shutdown file created: #{shutdown_file}")
447
+ begin
448
+ touch_pid_file(shutdown_file)
449
+ Workhorse.debug_log("[Job worker #{id}] Shutdown file created: #{shutdown_file}")
450
+ rescue StandardError => e
451
+ # The shutdown has to go ahead regardless: the worker stopped
452
+ # accepting jobs above, so giving up here would leave it taking none
453
+ # and never exiting. Without the file, watch sees a worker that is
454
+ # gone rather than one that asked to be restarted, and starts it all
455
+ # the same.
456
+ log "Could not create shutdown file #{shutdown_file}, shutting down regardless: #{e.message}", :warn
457
+ end
399
458
  end
400
459
 
401
460
  # Monitor in a separate thread to avoid blocking the signal handler
@@ -483,5 +542,17 @@ module Workhorse
483
542
  Kernel.sleep 0.2
484
543
  end
485
544
  end
545
+
546
+ # Touches a file in the pids directory, creating the directory first. It
547
+ # is not guaranteed to exist: `tmp` is rarely checked in, so a fresh
548
+ # checkout or deployment has none until something writes there.
549
+ #
550
+ # @param path [Pathname, String]
551
+ # @return [void]
552
+ # @private
553
+ def touch_pid_file(path)
554
+ FileUtils.mkdir_p(File.dirname(path))
555
+ FileUtils.touch(path)
556
+ end
486
557
  end
487
558
  end