workhorse 1.5.2 → 2.0.0.rc0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +102 -0
  3. data/README.md +303 -74
  4. data/Rakefile +1 -0
  5. data/VERSION +1 -1
  6. data/bin/rubocop +5 -1
  7. data/lib/generators/workhorse/install_generator.rb +10 -1
  8. data/lib/generators/workhorse/templates/config/initializers/workhorse.rb +55 -0
  9. data/lib/generators/workhorse/templates/create_table_jobs.rb +11 -13
  10. data/lib/generators/workhorse/templates/create_table_workhorse_schedules.rb +29 -0
  11. data/lib/workhorse/daemon.rb +52 -6
  12. data/lib/workhorse/db_job.rb +58 -7
  13. data/lib/workhorse/enqueuer.rb +51 -8
  14. data/lib/workhorse/jobs/cleanup_succeeded_jobs.rb +26 -8
  15. data/lib/workhorse/jobs/detect_late_schedules_job.rb +59 -0
  16. data/lib/workhorse/notifiers/base.rb +55 -0
  17. data/lib/workhorse/notifiers/file_system.rb +66 -0
  18. data/lib/workhorse/notifiers/none.rb +8 -0
  19. data/lib/workhorse/notifiers/redis.rb +227 -0
  20. data/lib/workhorse/performer.rb +28 -0
  21. data/lib/workhorse/poller.rb +289 -47
  22. data/lib/workhorse/pool.rb +12 -6
  23. data/lib/workhorse/schedule.rb +288 -0
  24. data/lib/workhorse/schedules.rb +197 -0
  25. data/lib/workhorse/worker.rb +77 -27
  26. data/lib/workhorse.rb +136 -0
  27. data/test/lib/db_schema.rb +21 -1
  28. data/test/lib/jobs.rb +29 -0
  29. data/test/lib/test_helper.rb +9 -14
  30. data/test/workhorse/daemon_test.rb +33 -0
  31. data/test/workhorse/db_job_test.rb +1 -1
  32. data/test/workhorse/notifier_test.rb +500 -0
  33. data/test/workhorse/poller_test.rb +8 -4
  34. data/test/workhorse/schedule_test.rb +967 -0
  35. data/test/workhorse/worker_test.rb +92 -0
  36. data/workhorse.gemspec +6 -5
  37. metadata +29 -3
@@ -23,4 +23,59 @@ Workhorse.setup do |config|
23
23
  # # Do something with exception, i.e.
24
24
  # # ExceptionNotifier.notify_exception(exception)
25
25
  # end
26
+
27
+ # Seconds the daemon's `stop` waits for a worker to finish what it is doing
28
+ # before killing it. Set to nil to wait indefinitely.
29
+ #
30
+ # config.shutdown_timeout = 300
31
+
32
+ # Enable this to let an enqueued job start without waiting for the next
33
+ # poll. Use :file where the workers share a filesystem with the application
34
+ # and :redis where they do not. Polling stays the floor either way, so raise
35
+ # the polling interval only as far as you are willing to wait when a
36
+ # notification is missed.
37
+ #
38
+ # config.notifier = :file
39
+ # config.notification_path = Rails.root.join('tmp', 'pids', 'workhorse.wake')
40
+ #
41
+ # config.notifier = :redis
42
+ # config.notification_redis = -> { Redis.new(url: ENV['REDIS_URL']) }
43
+ # config.notification_channel = 'workhorse:jobs'
44
+
45
+ # Enable and configure these to be told about a job that passed its
46
+ # `expires_at` before any worker got to it, and about one that started later
47
+ # than its `max_lateness` allows. Neither can affect the worker or the job.
48
+ #
49
+ # config.on_job_expired = proc do |db_job|
50
+ # # Do something with the job, i.e.
51
+ # # ExceptionNotifier.notify_exception(
52
+ # # StandardError.new("Job #{db_job.id} (#{db_job.description}) expired")
53
+ # # )
54
+ # end
55
+ #
56
+ # config.on_job_late = proc do |db_job, lateness|
57
+ # # Do something with the job, i.e.
58
+ # # ExceptionNotifier.notify_exception(
59
+ # # StandardError.new("Job #{db_job.id} started #{lateness.round}s late")
60
+ # # )
61
+ # end
26
62
  end
63
+
64
+ # Jobs that run on a schedule. Each of these owns a row in the
65
+ # `workhorse_schedules` table holding the next occurrence that has not been
66
+ # materialized yet, so an occurrence whose time passes while nothing is
67
+ # running is not lost. See the README for the catch-up policies.
68
+ #
69
+ # Workhorse.schedules do
70
+ # schedule 'cleanup_jobs',
71
+ # job: 'Workhorse::Jobs::CleanupSucceededJobs',
72
+ # cron: '10 0 * * *'
73
+ #
74
+ # schedule 'detect_stale_jobs',
75
+ # job: 'Workhorse::Jobs::DetectStaleJobsJob',
76
+ # cron: '30 * * * *'
77
+ #
78
+ # schedule 'detect_late_schedules',
79
+ # job: 'Workhorse::Jobs::DetectLateSchedulesJob',
80
+ # cron: '*/30 * * * *'
81
+ # end
@@ -17,24 +17,22 @@ class CreateTableJobs < ActiveRecord::Migration[7.1]
17
17
  t.integer :priority, null: false
18
18
  t.datetime :perform_at, null: true
19
19
 
20
+ # Deadline; the job is then set to state 'expired' rather than performed.
21
+ t.datetime :expires_at, null: true
22
+
23
+ # Seconds the job may start later than its perform_at before
24
+ # Workhorse.on_job_late is called.
25
+ t.integer :max_lateness, null: true
26
+
20
27
  t.string :description, null: true
21
28
 
22
29
  t.timestamps null: false
23
30
  end
24
31
 
25
- if oracle?
26
- add_index :jobs, :queue
27
- add_index :jobs, :state
28
- else
29
- add_index :jobs, :queue, length: 191
30
- add_index :jobs, :state, length: 191
31
- end
32
+ add_index :jobs, :queue, length: 191
33
+ add_index :jobs, %i[state perform_at], length: { state: 191 }, name: 'idx_jobs_state_perform_at'
34
+ add_index :jobs, %i[state priority created_at], length: { state: 191 }, name: 'idx_jobs_state_prio_created'
35
+ add_index :jobs, %i[state expires_at], length: { state: 191 }, name: 'idx_jobs_state_expires_at'
32
36
  add_index :jobs, :perform_at
33
37
  end
34
-
35
- private
36
-
37
- def oracle?
38
- ActiveRecord::Base.connection.adapter_name == 'OracleEnhanced'
39
- end
40
38
  end
@@ -0,0 +1,29 @@
1
+ class CreateTableWorkhorseSchedules < ActiveRecord::Migration[7.1]
2
+ def change
3
+ # Schedules are addressed by key; the job class and its options live in
4
+ # the Workhorse.schedules definition rather than here.
5
+ create_table :workhorse_schedules, force: true do |t|
6
+ t.string :key, null: false
7
+
8
+ # The cron expression and timezone are kept here so that a change to
9
+ # either is recognised on reconciliation.
10
+ t.string :cron, null: false
11
+ t.string :timezone, null: true
12
+
13
+ # Lets a schedule be switched off without a deployment.
14
+ t.boolean :enabled, null: false, default: true
15
+
16
+ # The next occurrence that has not been materialised yet.
17
+ t.datetime :next_at, null: false
18
+
19
+ t.datetime :last_enqueued_at, null: true
20
+ t.datetime :last_occurrence, null: true
21
+ t.integer :last_job_id, null: true
22
+
23
+ t.timestamps null: false
24
+ end
25
+
26
+ add_index :workhorse_schedules, :key, unique: true, length: 191, name: 'idx_wh_schedules_key'
27
+ add_index :workhorse_schedules, %i[enabled next_at], name: 'idx_wh_schedules_due'
28
+ end
29
+ end
@@ -3,6 +3,9 @@ module Workhorse
3
3
  # Provides functionality to start, stop, restart, and monitor worker processes
4
4
  # through a simple Ruby DSL.
5
5
  class Daemon
6
+ # Seconds spent waiting for a process to disappear after KILL, which it
7
+ # can only survive by being stuck in an uninterruptible syscall.
8
+ KILL_TIMEOUT = 10
6
9
  # Internal representation of a worker process.
7
10
  # Stores worker metadata and the block to execute.
8
11
  class Worker
@@ -310,7 +313,7 @@ module Workhorse
310
313
  @lockfile&.close
311
314
  # Reopen pipes to prevent #107576
312
315
  $stdin.reopen File.open(File::NULL, 'r')
313
- null_out = File.open(File::NULL, 'w') # rubocop:disable Style/FileOpen
316
+ null_out = File.open(File::NULL, 'w')
314
317
  $stdout.reopen null_out
315
318
  $stderr.reopen null_out
316
319
 
@@ -318,7 +321,27 @@ module Workhorse
318
321
  # by slot (see Workhorse::Worker#heartbeat!). Same id as the pidfile.
319
322
  ENV['WORKHORSE_DAEMON_WORKER_ID'] = worker.id.to_s
320
323
 
321
- worker.block.call
324
+ begin
325
+ worker.block.call
326
+
327
+ # Exit without running the at_exit handlers the parent registered:
328
+ # they belong to whatever started the daemon, not to this worker,
329
+ # and one that waits on threads the fork did not inherit hangs a
330
+ # worker that has finished - which then ignores TERM. ShellHandler
331
+ # skips them for the same reason.
332
+ exit!(0)
333
+ rescue Exception => e
334
+ # Reported here because exit! below denies it to every other route
335
+ # out: stderr is /dev/null by now, and an error reporter installed
336
+ # as an at_exit handler never runs. A worker that dies of an
337
+ # unhandled exception would otherwise do so silently.
338
+ begin
339
+ Workhorse.on_exception.call(e)
340
+ rescue Exception # rubocop:disable Lint/SuppressedException
341
+ end
342
+
343
+ exit!(1)
344
+ end
322
345
  end
323
346
  worker.pid = pid
324
347
  File.write(pid_file_for(worker), pid)
@@ -337,18 +360,41 @@ module Workhorse
337
360
  signals = kill ? %w[KILL] : %w[TERM INT]
338
361
 
339
362
  Workhorse.debug_log("Daemon: stopping PID #{pid} with signals #{signals.join(', ')}")
363
+
364
+ unless signal_until_gone(pid, signals, Workhorse.shutdown_timeout)
365
+ # A worker that does not go away leaves `stop` - and the deployment
366
+ # waiting on it - hanging indefinitely, so it is escalated the way an
367
+ # init system would.
368
+ warn "Worker #{pid} did not stop within #{Workhorse.shutdown_timeout}s, killing it"
369
+ Workhorse.debug_log("Daemon: PID #{pid} did not stop in time, sending KILL")
370
+ signal_until_gone(pid, %w[KILL], KILL_TIMEOUT)
371
+ end
372
+
373
+ Workhorse.debug_log("Daemon: PID #{pid} stopped")
374
+ FileUtils.rm_f(pid_file)
375
+ end
376
+
377
+ # Sends the given signals once a second until the process is gone.
378
+ #
379
+ # @param pid [Integer] The process to signal
380
+ # @param signals [Array<String>] Signals to send on each attempt
381
+ # @param timeout [Numeric, nil] Seconds to keep trying, or nil for no limit
382
+ # @return [Boolean] Whether the process is gone
383
+ # @private
384
+ def signal_until_gone(pid, signals, timeout)
385
+ deadline = timeout ? Process.clock_gettime(Process::CLOCK_MONOTONIC) + timeout : nil
386
+
340
387
  loop do
341
388
  begin
342
389
  signals.each { |signal| Process.kill(signal, pid) }
343
390
  rescue Errno::ESRCH
344
- break
391
+ return true
345
392
  end
346
393
 
394
+ return false if deadline && Process.clock_gettime(Process::CLOCK_MONOTONIC) >= deadline
395
+
347
396
  sleep 1
348
397
  end
349
-
350
- Workhorse.debug_log("Daemon: PID #{pid} stopped")
351
- File.delete(pid_file)
352
398
  end
353
399
 
354
400
  # Sends HUP signal to a worker process.
@@ -18,15 +18,22 @@ module Workhorse
18
18
  STATE_STARTED = :started
19
19
  STATE_SUCCEEDED = :succeeded
20
20
  STATE_FAILED = :failed
21
+ STATE_EXPIRED = :expired
21
22
 
22
23
  EXP_LOCKED_BY = /^(.*?)\.(\d+?)\.([^.]+)$/
23
24
 
24
25
  if respond_to?(:attr_accessible)
25
- attr_accessible :queue, :priority, :perform_at, :handler, :description
26
+ attr_accessible :queue, :priority, :perform_at, :handler, :description,
27
+ :expires_at, :max_lateness
26
28
  end
27
29
 
28
30
  self.table_name = 'jobs'
29
31
 
32
+ # Notifying before the commit would wake a worker that cannot see the row
33
+ # yet, sending it back to sleep for a whole polling interval - the very
34
+ # delay the notification exists to avoid.
35
+ after_commit :notify_workers, on: :create
36
+
30
37
  # Returns jobs in waiting state.
31
38
  #
32
39
  # @return [ActiveRecord::Relation] Jobs waiting to be processed
@@ -62,6 +69,35 @@ module Workhorse
62
69
  where(state: STATE_FAILED)
63
70
  end
64
71
 
72
+ # Returns jobs that passed their deadline and were therefore not run.
73
+ #
74
+ # @return [ActiveRecord::Relation] Jobs that expired before being started
75
+ def self.expired
76
+ where(state: STATE_EXPIRED)
77
+ end
78
+
79
+ # Marks the job as expired, having passed its deadline before any worker
80
+ # got to it.
81
+ #
82
+ # @raise [RuntimeError] If the job is not in waiting state
83
+ # @private Only to be used by workhorse
84
+ def mark_expired!
85
+ assert_state! STATE_WAITING
86
+
87
+ self.state = STATE_EXPIRED
88
+ save!
89
+ end
90
+
91
+ # Returns how much later than intended this job started, or nil if it has
92
+ # not started or was not given an intended time.
93
+ #
94
+ # @return [Float, nil] Lateness in seconds
95
+ def lateness
96
+ return nil unless started_at && perform_at
97
+
98
+ return started_at - perform_at
99
+ end
100
+
65
101
  # Returns a relation with split locked_by field for easier querying.
66
102
  # Extracts host, PID, and random string components from locked_by.
67
103
  #
@@ -94,9 +130,10 @@ module Workhorse
94
130
  # Resets job to state "waiting" and clears all meta fields
95
131
  # set by workhorse in course of processing this job.
96
132
  #
97
- # This is only allowed if the job is in a final state ("succeeded" or
98
- # "failed"), as only those jobs are safe to modify; workhorse will not touch
99
- # these jobs. To reset a job without checking the state it is in, set
133
+ # This is only allowed if the job is in a final state ("succeeded",
134
+ # "failed" or "expired"), as only those jobs are safe to modify; workhorse
135
+ # will not touch these jobs. To reset a job without checking the state it
136
+ # is in, set
100
137
  # "force" to true. Prior to doing so, ensure that the job is not still being
101
138
  # processed by a worker. If possible, shut down all workers before
102
139
  # performing a forced reset.
@@ -111,7 +148,7 @@ module Workhorse
111
148
  # @raise [RuntimeError] If job is not in a final state and force is false
112
149
  def reset!(force = false)
113
150
  unless force
114
- assert_state! STATE_SUCCEEDED, STATE_FAILED
151
+ assert_state! STATE_SUCCEEDED, STATE_FAILED, STATE_EXPIRED
115
152
  end
116
153
 
117
154
  self.state = STATE_WAITING
@@ -136,8 +173,6 @@ module Workhorse
136
173
  end
137
174
 
138
175
  if locked_at
139
- # TODO: Remove this debug output
140
- # puts "Already locked. Job: #{self.id} Worker: #{worker_id}"
141
176
  fail "Job #{id} is already locked by #{locked_by.inspect}."
142
177
  end
143
178
 
@@ -185,6 +220,22 @@ module Workhorse
185
220
  save!
186
221
  end
187
222
 
223
+ # Announces this job to waiting workers, see {Workhorse.notifier}.
224
+ #
225
+ # Jobs that are not due yet are not announced: a woken worker would find
226
+ # nothing to do. They are picked up by the regular poll once their
227
+ # `perform_at` has passed.
228
+ #
229
+ # @return [void]
230
+ # @private
231
+ def notify_workers
232
+ return if perform_at && perform_at > Time.now
233
+
234
+ Workhorse.notifier.notify(queue: queue)
235
+
236
+ return
237
+ end
238
+
188
239
  # Asserts that the job is in one of the specified states.
189
240
  #
190
241
  # @param states [Array<Symbol>] Valid states for the job
@@ -10,15 +10,28 @@ module Workhorse
10
10
  # @param priority [Integer] Job priority (lower numbers = higher priority)
11
11
  # @param perform_at [Time] When to perform the job
12
12
  # @param description [String, nil] Optional job description
13
+ # @param expires_at [Time, nil] Deadline after which the job is no longer
14
+ # worth running. It is then set to state `expired` instead of being
15
+ # performed, and {Workhorse.on_job_expired} is called.
16
+ # @param max_lateness [Numeric, nil] Seconds the job may start after its
17
+ # `perform_at` before {Workhorse.on_job_late} is called.
13
18
  # @return [Workhorse::DbJob] The created database job record
14
- def enqueue(job, queue: nil, priority: 0, perform_at: Time.now, description: nil)
15
- return DbJob.create!(
19
+ def enqueue(job, queue: nil, priority: 0, perform_at: Time.now, description: nil,
20
+ expires_at: nil, max_lateness: nil)
21
+ attributes = {
16
22
  queue: queue,
17
23
  priority: priority,
18
24
  perform_at: perform_at,
19
25
  description: description,
20
26
  handler: Marshal.dump(job)
21
- )
27
+ }
28
+
29
+ # Only set when used, so that an installation that has not run the
30
+ # migration adding these columns keeps working as before.
31
+ attributes[:expires_at] = expires_at if expires_at
32
+ attributes[:max_lateness] = max_lateness if max_lateness
33
+
34
+ return DbJob.create!(**attributes)
22
35
  end
23
36
 
24
37
  # Enqueues an ActiveJob job instance.
@@ -27,16 +40,23 @@ module Workhorse
27
40
  # @param perform_at [Time] When to perform the job
28
41
  # @param queue [String, Symbol, nil] Optional queue override
29
42
  # @param description [String, nil] Optional job description
43
+ # @param priority [Integer, nil] Priority override. Defaults to the job's
44
+ # own priority.
45
+ # @param expires_at [Time, nil] See {#enqueue}
46
+ # @param max_lateness [Numeric, nil] See {#enqueue}
30
47
  # @return [Workhorse::DbJob] The created database job record
31
- def enqueue_active_job(job, perform_at: Time.now, queue: nil, description: nil)
48
+ def enqueue_active_job(job, perform_at: Time.now, queue: nil, description: nil,
49
+ priority: nil, expires_at: nil, max_lateness: nil)
32
50
  wrapper_job = Jobs::RunActiveJob.new(job.serialize)
33
51
  queue ||= job.queue_name if job.queue_name.present?
34
52
  db_job = enqueue(
35
53
  wrapper_job,
36
- queue: queue,
37
- priority: job.priority || 0,
38
- perform_at: Time.at(perform_at),
39
- description: description
54
+ queue: queue,
55
+ priority: priority || job.priority || 0,
56
+ perform_at: Time.at(perform_at),
57
+ description: description,
58
+ expires_at: expires_at,
59
+ max_lateness: max_lateness
40
60
  )
41
61
  job.provider_job_id = db_job.id
42
62
  return db_job
@@ -65,5 +85,28 @@ module Workhorse
65
85
  job = Workhorse::Jobs::RunRailsOp.new(cls, op_args)
66
86
  enqueue job, **workhorse_args
67
87
  end
88
+
89
+ # Enqueues a job given by class name, working out how it wants to be
90
+ # instantiated: as a RailsOps operation, as an ActiveJob job, or as a
91
+ # plain object responding to `perform`.
92
+ #
93
+ # @param job_class [String, Class] The job class
94
+ # @param params [Hash] Operation params for RailsOps operations, keyword
95
+ # arguments to the constructor otherwise
96
+ # @param options [Hash] Passed on to {#enqueue}
97
+ # @return [Workhorse::DbJob] The created database job record
98
+ def enqueue_job_class(job_class, params: {}, **options)
99
+ cls = job_class.is_a?(String) ? job_class.constantize : job_class
100
+
101
+ if defined?(RailsOps::Operation) && cls < RailsOps::Operation
102
+ return enqueue_op(cls, options, params)
103
+ elsif defined?(ActiveJob::Base) && cls < ActiveJob::Base
104
+ job = params.present? ? cls.new(**params) : cls.new
105
+ return enqueue_active_job(job, **options)
106
+ else
107
+ job = params.present? ? cls.new(**params) : cls.new
108
+ return enqueue(job, **options)
109
+ end
110
+ end
68
111
  end
69
112
  end
@@ -6,26 +6,44 @@ module Workhorse::Jobs
6
6
  # @example Schedule cleanup job
7
7
  # Workhorse.enqueue(CleanupSucceededJobs.new(max_age: 30))
8
8
  #
9
- # @example Daily cleanup with cron
10
- # # Clean up jobs older than 14 days every day at 2 AM
11
- # Workhorse.enqueue(CleanupSucceededJobs.new, perform_at: 1.day.from_now.beginning_of_day + 2.hours)
9
+ # @example Daily cleanup on a schedule
10
+ # Workhorse.schedules do
11
+ # schedule 'cleanup_jobs', job: 'Workhorse::Jobs::CleanupSucceededJobs', cron: '0 2 * * *'
12
+ # end
12
13
  class CleanupSucceededJobs
14
+ # States that are cleaned up unless told otherwise. Both are terminal and
15
+ # will never run again; `expired` is included so that a schedule using
16
+ # `expires_after` and regularly missing its window - the very case the
17
+ # option exists for - cannot grow the table without bound.
18
+ DEFAULT_STATES = [
19
+ Workhorse::DbJob::STATE_SUCCEEDED,
20
+ Workhorse::DbJob::STATE_EXPIRED
21
+ ].freeze
22
+
13
23
  # Instantiates a new job.
14
24
  #
15
25
  # @param max_age [Integer] The maximal age of jobs to retain, in days. Will
16
26
  # be evaluated at perform time.
17
- def initialize(max_age: 14)
27
+ # @param states [Array<Symbol>] The job states to clean up. Defaults to
28
+ # {DEFAULT_STATES}. Pass `[Workhorse::DbJob::STATE_SUCCEEDED]` to keep
29
+ # expired jobs around.
30
+ def initialize(max_age: 14, states: DEFAULT_STATES)
18
31
  @max_age = max_age
32
+ @states = states
19
33
  end
20
34
 
21
- # Executes the cleanup by deleting old succeeded jobs.
35
+ # Executes the cleanup by deleting old jobs in the configured states.
22
36
  #
23
37
  # @return [void]
24
38
  def perform
25
39
  age_limit = seconds_ago(@max_age)
26
- Workhorse::DbJob.where(
27
- 'STATE = ? AND UPDATED_AT <= ?', Workhorse::DbJob::STATE_SUCCEEDED, age_limit
28
- ).delete_all
40
+
41
+ # An instance of this job enqueued before the upgrade unmarshals without
42
+ # @states, and `where(state: nil)` would quietly match nothing - leaving
43
+ # the table growing, which is what this job is for.
44
+ Workhorse::DbJob.where(state: @states || DEFAULT_STATES)
45
+ .where('UPDATED_AT <= ?', age_limit)
46
+ .delete_all
29
47
  end
30
48
 
31
49
  private
@@ -0,0 +1,59 @@
1
+ module Workhorse::Jobs
2
+ # Job that detects schedules nothing is materialising.
3
+ #
4
+ # This is the one check that can catch silence. {Workhorse.on_job_expired}
5
+ # and {Workhorse.on_job_late} are precise but can only fire for a job that
6
+ # exists; if no worker is polling, or the global lock is stuck, occurrences
7
+ # stop being materialised altogether and no callback can report it. A
8
+ # schedule whose `next_at` lies well in the past is the evidence of that.
9
+ #
10
+ # Schedule this regularly, the same way as
11
+ # {Workhorse::Jobs::DetectStaleJobsJob}. Note that it is itself performed by
12
+ # a worker, so it reports a stall only while at least one worker still runs
13
+ # - use it alongside external monitoring, not instead of it.
14
+ #
15
+ # @example Schedule with the default threshold
16
+ # Workhorse.schedules do
17
+ # schedule 'detect_late_schedules',
18
+ # job: 'Workhorse::Jobs::DetectLateSchedulesJob',
19
+ # cron: '*/30 * * * *'
20
+ # end
21
+ class DetectLateSchedulesJob
22
+ # Creates a new late schedule detection job.
23
+ #
24
+ # @param threshold [Integer] Number of seconds a schedule's next
25
+ # occurrence may lie in the past before it is reported.
26
+ # @param keys [Array<String>, nil] If given, only check these schedules.
27
+ # If `nil` (default), all of them are checked.
28
+ def initialize(threshold: 15 * 60, keys: nil)
29
+ @threshold = threshold
30
+ @keys = keys
31
+ end
32
+
33
+ # Executes the detection.
34
+ #
35
+ # @return [void]
36
+ # @raise [RuntimeError] If schedules are found that are overdue
37
+ def perform
38
+ # Only schedules this process declares. A row whose declaration is gone
39
+ # is waiting to be cleaned up by reconciliation and nothing materializes
40
+ # it, so reporting it would be a false alarm.
41
+ keys = @keys || Workhorse::Schedules.definitions.keys
42
+
43
+ rel = Workhorse::Schedule.where(enabled: true, key: keys)
44
+ rel = rel.where(Workhorse::Schedule.arel_table[:next_at].lt(@threshold.seconds.ago))
45
+
46
+ overdue = rel.pluck(:key, :next_at)
47
+
48
+ return if overdue.empty?
49
+
50
+ descriptions = overdue.map do |key, next_at|
51
+ "#{key.inspect} (due #{next_at}, #{(Time.now - next_at).round}s ago)"
52
+ end
53
+
54
+ fail "The next occurrence of #{overdue.size} schedule(s) is more than #{@threshold}s in the past, " \
55
+ 'which means they are not being materialized. Check that at least one worker is polling and ' \
56
+ "that the global lock is obtainable. Affected: #{descriptions.join(', ')}."
57
+ end
58
+ end
59
+ end
@@ -0,0 +1,55 @@
1
+ module Workhorse
2
+ module Notifiers
3
+ # Interface for notifiers, which let a worker start a job without waiting
4
+ # for its next poll.
5
+ #
6
+ # Polling remains the floor: a notification that is never delivered costs
7
+ # latency and nothing else, as the regular poll still happens after
8
+ # `polling_interval` and picks the job up. Implementations must therefore
9
+ # be best-effort and must never raise from {#notify}, which runs in the
10
+ # process that enqueues a job.
11
+ #
12
+ # Workers do not consume notifications. Instead, {#token} returns a value
13
+ # that changes whenever a notification has arrived, and each poller
14
+ # remembers the last one it saw. This keeps several workers in one process
15
+ # independent of one another, as none of them can take a notification away
16
+ # from the others.
17
+ #
18
+ # @abstract Subclass and override {#notify} and {#token}.
19
+ class Base
20
+ # Announces that a job has been enqueued. Called in the enqueueing
21
+ # process, after the surrounding transaction has committed.
22
+ #
23
+ # Must never raise: an enqueue may not fail because a notification could
24
+ # not be delivered.
25
+ #
26
+ # @param queue [String, Symbol, nil] Queue the job was enqueued into
27
+ # @return [void]
28
+ def notify(queue: nil); end
29
+
30
+ # Prepares this notifier for use by a worker in this process. Called
31
+ # when a worker starts, and may be called several times per process.
32
+ #
33
+ # @return [void]
34
+ def start; end
35
+
36
+ # Releases what {#start} acquired. Called when a worker shuts down.
37
+ #
38
+ # @return [void]
39
+ def stop; end
40
+
41
+ # Returns a value that differs from the previously returned one whenever
42
+ # at least one notification has arrived in between. Pollers compare it
43
+ # with `!=` and hold their own last-seen value, so no ordering is implied
44
+ # and nothing is consumed.
45
+ #
46
+ # Returns nil when nothing can be said, for instance because no
47
+ # notification has ever been sent.
48
+ #
49
+ # @return [Object, nil] Comparable token, or nil
50
+ def token
51
+ return nil
52
+ end
53
+ end
54
+ end
55
+ end
@@ -0,0 +1,66 @@
1
+ module Workhorse
2
+ module Notifiers
3
+ # Notifier that announces an enqueued job by touching a single file, which
4
+ # waiting workers stat once per sleep slice.
5
+ #
6
+ # This costs no database work at all and roughly a microsecond of local CPU
7
+ # per check, on a tick the poller performs anyway, so a worker can react
8
+ # within {Workhorse::Poller::SLEEP_SLICE} rather than within its polling
9
+ # interval.
10
+ #
11
+ # It requires that the enqueueing process and the workers share a
12
+ # filesystem, which in practice means the same host. Where they do not -
13
+ # workers on a machine of their own, a console on another host - the touch
14
+ # simply never reaches those workers and they fall back to polling. Use
15
+ # {Workhorse::Notifiers::Redis} if that case has to be fast as well.
16
+ #
17
+ # One file is used for every queue. A worker that serves a subset of queues
18
+ # therefore wakes for jobs it cannot run and polls once for nothing; this
19
+ # is cheaper than having such a worker scan a directory on every tick.
20
+ #
21
+ # Detection relies on the file's modification time, so the filesystem has
22
+ # to keep sub-second mtimes - every current local filesystem does. On one
23
+ # that does not, two touches within the same second may be seen as one, and
24
+ # the worker waits for its regular poll instead.
25
+ class FileSystem < Base
26
+ # @param path [String, nil] Path of the file to touch. Defaults to
27
+ # {Workhorse.notification_path}.
28
+ def initialize(path: nil)
29
+ super()
30
+ @path = path
31
+ end
32
+
33
+ # Resolved when used rather than when constructed, so that
34
+ # {Workhorse.notification_path} can be set in any order relative to
35
+ # {Workhorse.notifier=}.
36
+ #
37
+ # @return [String] Path of the file that is touched
38
+ def path
39
+ return (@path || Workhorse.notification_path).to_s
40
+ end
41
+
42
+ # Touches the notification file, creating it and its directory if
43
+ # necessary.
44
+ #
45
+ # @param queue [String, Symbol, nil] Ignored, see the class description
46
+ # @return [void]
47
+ def notify(queue: nil) # rubocop:disable Lint/UnusedMethodArgument
48
+ FileUtils.mkdir_p(::File.dirname(path))
49
+ FileUtils.touch(path)
50
+ rescue StandardError => e
51
+ # Best-effort by contract: a job must still be enqueued when it cannot
52
+ # be announced. The worker finds it on its next poll.
53
+ Workhorse.debug_log("Notification failed: #{e.class}: #{e.message}")
54
+ end
55
+
56
+ # Returns the notification file's modification time.
57
+ #
58
+ # @return [Time, nil] mtime, or nil if the file does not exist yet
59
+ def token
60
+ return ::File.mtime(path)
61
+ rescue SystemCallError
62
+ return nil
63
+ end
64
+ end
65
+ end
66
+ end
@@ -0,0 +1,8 @@
1
+ module Workhorse
2
+ module Notifiers
3
+ # Notifier that does nothing, leaving workers to discover jobs by polling.
4
+ # This is the default.
5
+ class None < Base
6
+ end
7
+ end
8
+ end