workhorse 1.5.2 → 2.0.0.rc1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. checksums.yaml +4 -4
  2. data/.github/workflows/ruby.yml +137 -1
  3. data/CHANGELOG.md +150 -0
  4. data/Gemfile +16 -1
  5. data/README.md +316 -72
  6. data/Rakefile +1 -0
  7. data/VERSION +1 -1
  8. data/bin/rubocop +5 -1
  9. data/lib/generators/workhorse/install_generator.rb +10 -1
  10. data/lib/generators/workhorse/templates/config/initializers/workhorse.rb +55 -0
  11. data/lib/generators/workhorse/templates/create_table_jobs.rb +15 -2
  12. data/lib/generators/workhorse/templates/create_table_workhorse_schedules.rb +42 -0
  13. data/lib/workhorse/daemon/shell_handler.rb +4 -1
  14. data/lib/workhorse/daemon.rb +57 -7
  15. data/lib/workhorse/db_job.rb +98 -8
  16. data/lib/workhorse/enqueuer.rb +51 -8
  17. data/lib/workhorse/jobs/cleanup_succeeded_jobs.rb +26 -8
  18. data/lib/workhorse/jobs/detect_late_schedules_job.rb +59 -0
  19. data/lib/workhorse/notifiers/base.rb +55 -0
  20. data/lib/workhorse/notifiers/file_system.rb +66 -0
  21. data/lib/workhorse/notifiers/none.rb +8 -0
  22. data/lib/workhorse/notifiers/redis.rb +227 -0
  23. data/lib/workhorse/performer.rb +29 -2
  24. data/lib/workhorse/poller.rb +303 -21
  25. data/lib/workhorse/pool.rb +12 -6
  26. data/lib/workhorse/schedule.rb +288 -0
  27. data/lib/workhorse/schedules.rb +197 -0
  28. data/lib/workhorse/worker.rb +102 -31
  29. data/lib/workhorse.rb +136 -0
  30. data/test/lib/db_schema.rb +36 -3
  31. data/test/lib/jobs.rb +29 -0
  32. data/test/lib/test_helper.rb +113 -20
  33. data/test/workhorse/daemon_test.rb +33 -0
  34. data/test/workhorse/db_job_test.rb +2 -4
  35. data/test/workhorse/notifier_test.rb +487 -0
  36. data/test/workhorse/performer_test.rb +7 -9
  37. data/test/workhorse/poller_test.rb +97 -23
  38. data/test/workhorse/schedule_test.rb +967 -0
  39. data/test/workhorse/worker_test.rb +201 -76
  40. data/workhorse.gemspec +6 -5
  41. metadata +29 -3
@@ -10,6 +10,14 @@ module Workhorse
10
10
  MIN_LOCK_TIMEOUT = 0.1 # In seconds
11
11
  MAX_LOCK_TIMEOUT = 1.0 # In seconds
12
12
 
13
+ # Most jobs one poll expires, see {#expire_due_jobs}.
14
+ MAX_EXPIRIES_PER_POLL = 100
15
+
16
+ # Length of one slice of the poller's sleep, in seconds. The poller sleeps
17
+ # in slices rather than for the whole polling interval so that it stays
18
+ # responsive to shutdown, to instant repolling and to notifications.
19
+ SLEEP_SLICE = 0.1
20
+
13
21
  ORACLE_LOCK_MODE = 6 # X_MODE (exclusive)
14
22
  ORACLE_LOCK_HANDLE = 478_564_848 # Randomly chosen number
15
23
 
@@ -51,6 +59,18 @@ module Workhorse
51
59
 
52
60
  Workhorse.debug_log("[Job worker #{worker.id}] Poller starting")
53
61
 
62
+ begin
63
+ Workhorse.notifier.start
64
+ rescue StandardError => e
65
+ worker.log "Starting the notifier failed, falling back to polling: #{e.class}: #{e.message}", :warn
66
+ end
67
+
68
+ # Only jobs announced from now on concern this worker; anything enqueued
69
+ # earlier is found by the poll that follows.
70
+ @last_notification = notifier_token
71
+
72
+ reconcile_schedules!
73
+
54
74
  clean_stuck_jobs! if Workhorse.clean_stuck_jobs
55
75
 
56
76
  @thread = Thread.new do
@@ -91,6 +111,13 @@ module Workhorse
91
111
  Workhorse.debug_log("[Job worker #{worker.id}] Poller shutting down")
92
112
  @running = false
93
113
  wait
114
+
115
+ begin
116
+ Workhorse.notifier.stop
117
+ rescue StandardError => e
118
+ worker.log "Stopping the notifier failed: #{e.class}: #{e.message}", :warn
119
+ end
120
+
94
121
  Workhorse.debug_log("[Job worker #{worker.id}] Poller shut down")
95
122
  end
96
123
 
@@ -112,6 +139,43 @@ module Workhorse
112
139
 
113
140
  private
114
141
 
142
+ # Brings the schedules table in line with the registry, see
143
+ # {Workhorse::Schedule.reconcile!}.
144
+ #
145
+ # A worker that cannot reconcile still works through the queue, so this
146
+ # reports rather than raises.
147
+ #
148
+ # @return [void]
149
+ # @private
150
+ def reconcile_schedules!
151
+ @schedules_available = Workhorse::Schedule.table_exists?
152
+
153
+ unless @schedules_available
154
+ if Workhorse::Schedules.any?
155
+ message = 'Schedules are declared but the workhorse_schedules table does not exist. ' \
156
+ 'Run the migration that creates it; no scheduled job will run until then.'
157
+ worker.log message, :error
158
+
159
+ begin
160
+ Workhorse.on_exception.call(StandardError.new(message))
161
+ rescue Exception => e
162
+ # Reported through the rescue below would feed the callback its
163
+ # own failure; escaping leaves the worker half-started.
164
+ Workhorse.debug_log("on_exception failed: #{e.class}: #{e.message}")
165
+ end
166
+ end
167
+
168
+ return
169
+ end
170
+
171
+ with_global_lock timeout: MAX_LOCK_TIMEOUT do
172
+ Workhorse::Schedule.reconcile!
173
+ end
174
+ rescue Exception => e
175
+ worker.log %(Could not reconcile schedules: #{e.message}), :error
176
+ Workhorse.on_exception.call(e)
177
+ end
178
+
115
179
  # Cleans up jobs stuck in locked or started states from dead processes.
116
180
  # Only cleans jobs from the current hostname.
117
181
  #
@@ -174,7 +238,9 @@ module Workhorse
174
238
  end
175
239
  end
176
240
 
177
- # Sleeps for the configured polling interval with instant repoll support.
241
+ # Sleeps for the configured polling interval, returning early for an
242
+ # instant repoll, for shutdown, or as soon as a notification announces a
243
+ # newly enqueued job.
178
244
  #
179
245
  # @return [void]
180
246
  # @private
@@ -182,9 +248,63 @@ module Workhorse
182
248
  remaining = worker.polling_interval
183
249
 
184
250
  while running? && remaining > 0 && @instant_repoll.false?
185
- Kernel.sleep 0.1
186
- remaining -= 0.1
251
+ Kernel.sleep SLEEP_SLICE
252
+ remaining -= SLEEP_SLICE
253
+
254
+ next unless notified?
255
+
256
+ worker.log 'Job was announced, polling ahead of the interval', :debug
257
+ break
187
258
  end
259
+
260
+ # Time left on the clock means the sleep was cut short, which #poll
261
+ # passes on as `count_failures: false`.
262
+ @poll_brought_forward = remaining > 0
263
+ end
264
+
265
+ # Returns whether a job has been announced since this poller last looked,
266
+ # and records what it saw.
267
+ #
268
+ # A worker with no idle thread deliberately leaves the token untouched, so
269
+ # that the notification is still pending once it has capacity again.
270
+ #
271
+ # @return [Boolean]
272
+ # @private
273
+ def notified?
274
+ return false unless worker.accepting_jobs?
275
+ return false if worker.idle.zero?
276
+
277
+ token = notifier_token
278
+
279
+ return false if token.nil? || token == @last_notification
280
+
281
+ @last_notification = token
282
+
283
+ return true
284
+ end
285
+
286
+ # Reads the notifier's token, returning nil if it cannot be read.
287
+ #
288
+ # A notifier is an accelerator, so a broken one must cost latency rather
289
+ # than take the worker down: anything raised here would reach the poller's
290
+ # own rescue, which shuts the worker down. As this runs on every sleep
291
+ # slice, the failure is reported once rather than many times a second.
292
+ #
293
+ # @return [Object, nil]
294
+ # @private
295
+ def notifier_token
296
+ return Workhorse.notifier.token
297
+ rescue StandardError => e
298
+ message = "Reading the notifier failed, falling back to polling: #{e.class}: #{e.message}"
299
+
300
+ if @notifier_failed
301
+ worker.log message, :debug
302
+ else
303
+ @notifier_failed = true
304
+ worker.log message, :warn
305
+ end
306
+
307
+ return nil
188
308
  end
189
309
 
190
310
  # Executes a block with a global database lock.
@@ -192,11 +312,19 @@ module Workhorse
192
312
  #
193
313
  # @param name [Symbol] Lock name identifier
194
314
  # @param timeout [Integer] Lock timeout in seconds
315
+ # @param count_failures [Boolean] Whether a failure to obtain the lock
316
+ # counts towards {Workhorse.max_global_lock_fails}
195
317
  # @yield Block to execute while holding the lock
196
318
  # @return [void]
197
319
  # @private
198
- def with_global_lock(name: :workhorse, timeout: 2, &_block)
199
- begin # rubocop:disable Style/RedundantBegin
320
+ def with_global_lock(name: :workhorse, timeout: 2, count_failures: true, &_block)
321
+ # Whole seconds, rounded up: MySQL and Oracle both take the timeout as an
322
+ # integer and round a fraction of a second down to not waiting at all -
323
+ # only MariaDB honours one. A poll finding the lock taken would then give
324
+ # up at once, and count towards max_global_lock_fails for mere contention.
325
+ timeout = timeout.ceil
326
+
327
+ begin
200
328
  if @is_oracle
201
329
  result = Workhorse::DbJob.connection.select_all(
202
330
  "SELECT DBMS_LOCK.REQUEST(#{ORACLE_LOCK_HANDLE}, #{ORACLE_LOCK_MODE}, #{timeout}) FROM DUAL"
@@ -213,6 +341,13 @@ module Workhorse
213
341
  if success
214
342
  @global_lock_fails = 0
215
343
  @max_global_lock_fails_reached = false
344
+ elsif !count_failures
345
+ # Losing the race for the lock is the expected outcome when several
346
+ # workers were woken by the same announcement, and says nothing about
347
+ # a crashed worker. Counting it would let the alarm below fire within
348
+ # seconds rather than after the polling intervals it is calibrated
349
+ # for.
350
+ worker.log 'Could not obtain global lock for a poll that was brought forward, skipping it.', :debug
216
351
  else
217
352
  @global_lock_fails += 1
218
353
 
@@ -265,10 +400,20 @@ module Workhorse
265
400
 
266
401
  @instant_repoll.make_false
267
402
 
403
+ # A poll the sleep cut short is not the scheduled one the lock-failure
404
+ # alarm is calibrated against, see #with_global_lock.
405
+ brought_forward = @poll_brought_forward
406
+ @poll_brought_forward = false
407
+
408
+ expired = []
409
+
268
410
  timeout = worker.polling_interval.clamp(MIN_LOCK_TIMEOUT, MAX_LOCK_TIMEOUT)
269
- with_global_lock timeout: timeout do
411
+ with_global_lock timeout: timeout, count_failures: !brought_forward do
270
412
  job_ids = []
271
413
 
414
+ materialize_schedules if Workhorse::Schedules.any? && @schedules_available
415
+ expired = expire_due_jobs
416
+
272
417
  Workhorse.tx_callback.call do
273
418
  # As we are the only thread posting into the worker pool, it is safe to
274
419
  # get the number of idle threads without mutex synchronization. The
@@ -303,6 +448,11 @@ module Workhorse
303
448
  job_ids.each { |job_id| worker.perform(job_id) } if running? && worker.accepting_jobs?
304
449
  end
305
450
 
451
+ # Deliberately outside the global lock: the callback is the application's
452
+ # and may do something slow, such as sending mail, which would otherwise
453
+ # block every other worker's poll.
454
+ notify_expired(expired)
455
+
306
456
  # Record that this worker successfully polled. Done at the very end so it
307
457
  # only advances when the poll actually completed (a poll that raises never
308
458
  # reaches here). Skipped on the early return above when the worker is no
@@ -310,6 +460,117 @@ module Workhorse
310
460
  worker.heartbeat!
311
461
  end
312
462
 
463
+ # Materialises the occurrences that have come due into jobs.
464
+ #
465
+ # A failing schedule must not take the worker down with it, nor stop the
466
+ # other schedules, so each is handled on its own.
467
+ #
468
+ # @return [void]
469
+ # @private
470
+ def materialize_schedules
471
+ Workhorse::Schedule.due.to_a.each do |schedule|
472
+ next if schedule.definition.nil?
473
+
474
+ occurrences, next_at = schedule.pending_occurrences
475
+
476
+ # The claim advances the schedule past these occurrences, so
477
+ # committing it before the jobs exist would lose them for good if
478
+ # enqueuing then failed - the schedule would move on and
479
+ # DetectLateSchedulesJob would see nothing wrong.
480
+ Workhorse.tx_callback.call do
481
+ next unless schedule.claim!(next_at)
482
+
483
+ occurrences.each do |occurrence|
484
+ db_job = schedule.enqueue!(occurrence)
485
+ worker.log "Materialized schedule #{schedule.key.inspect} for #{occurrence} as job #{db_job.id}", :debug
486
+ end
487
+ end
488
+
489
+ @failed_schedules&.delete(schedule.key)
490
+ rescue Exception => e
491
+ report_schedule_failure(schedule, e)
492
+ end
493
+
494
+ return
495
+ rescue Exception => e
496
+ # The query itself failed, so no individual schedule can be blamed. This
497
+ # must not reach the poller's own rescue, which shuts the worker down.
498
+ worker.log %(Could not query due schedules: #{e.message}), :error
499
+ Workhorse.on_exception.call(e)
500
+ end
501
+
502
+ # Reports a schedule that could not be materialized.
503
+ #
504
+ # Reported once per schedule per worker: a schedule that can never enqueue
505
+ # - a renamed job class, params its constructor rejects - fails on every
506
+ # poll, and one typo must not turn into a notification every polling
507
+ # interval for as long as the worker runs.
508
+ #
509
+ # @param schedule [Workhorse::Schedule]
510
+ # @param exception [Exception]
511
+ # @return [void]
512
+ # @private
513
+ def report_schedule_failure(schedule, exception)
514
+ message = %(Could not materialize schedule #{schedule.key.inspect}: #{exception.message})
515
+ @failed_schedules ||= Set.new
516
+
517
+ if @failed_schedules.include?(schedule.key)
518
+ worker.log message, :debug
519
+ return
520
+ end
521
+
522
+ @failed_schedules << schedule.key
523
+ worker.log message, :error
524
+ Workhorse.on_exception.call(exception)
525
+
526
+ return
527
+ end
528
+
529
+ # Marks jobs that passed their deadline before any worker got to them.
530
+ #
531
+ # @return [Array<Workhorse::DbJob>] The jobs that were expired
532
+ # @private
533
+ def expire_due_jobs
534
+ return [] unless expiry_supported?
535
+
536
+ expired = []
537
+
538
+ Workhorse.tx_callback.call do
539
+ rel = Workhorse::DbJob.waiting.where(Workhorse::DbJob.arel_table[:expires_at].lteq(Time.now))
540
+
541
+ # Bounded, as this runs while the poller holds the global lock: a
542
+ # backlog of jobs that all expired at once - workers down over a
543
+ # weekend, or a bulk enqueue - would otherwise turn one poll into
544
+ # thousands of updates that block every other worker. The rest is
545
+ # expired by the following polls, most overdue first.
546
+ rel.order(:expires_at).limit(MAX_EXPIRIES_PER_POLL).each do |db_job|
547
+ db_job.mark_expired!
548
+ expired << db_job
549
+ worker.log "Job #{db_job.id} passed its deadline of #{db_job.expires_at} and was not run", :warn
550
+ end
551
+ end
552
+
553
+ return expired
554
+ end
555
+
556
+ # Calls {Workhorse.on_job_expired} for each expired job, keeping a failing
557
+ # callback away from the poller's own error handling, which would shut the
558
+ # worker down.
559
+ #
560
+ # @param expired [Array<Workhorse::DbJob>]
561
+ # @return [void]
562
+ # @private
563
+ def notify_expired(expired)
564
+ expired.each do |db_job|
565
+ Workhorse.on_job_expired.call(db_job)
566
+ rescue Exception => e
567
+ worker.log %(on_job_expired failed for job #{db_job.id}: #{e.message}), :error
568
+ Workhorse.on_exception.call(e)
569
+ end
570
+
571
+ return
572
+ end
573
+
313
574
  # Returns an array of {Workhorse::DbJob}s that can be started.
314
575
  # Uses complex SQL with UNIONs to respect queue ordering and limits.
315
576
  #
@@ -335,7 +596,7 @@ module Workhorse
335
596
  # any presumptions on the order.
336
597
  record_number = queue.nil? ? limit : 1
337
598
 
338
- union_parts << agnostic_limit(select, record_number)
599
+ union_parts << limited_sql(select, record_number)
339
600
  end
340
601
 
341
602
  return [] if union_parts.empty?
@@ -350,9 +611,9 @@ module Workhorse
350
611
  # uses the keyword 'AS' in SQL generated for Oracle, which is invalid for
351
612
  # table aliases.
352
613
  union_query_sql = '('
353
- union_query_sql += "SELECT * FROM (#{union_parts.shift.to_sql}) union_0"
614
+ union_query_sql += "SELECT * FROM (#{union_parts.shift}) union_0"
354
615
  union_parts.each_with_index do |part, idx|
355
- union_query_sql += " UNION SELECT * FROM (#{part.to_sql}) union_#{idx + 1}"
616
+ union_query_sql += " UNION SELECT * FROM (#{part}) union_#{idx + 1}"
356
617
  end
357
618
  union_query_sql += ') subselect'
358
619
 
@@ -365,22 +626,39 @@ module Workhorse
365
626
  select = table.project(Arel.star).where(table[:id].in(select.project(:id)))
366
627
  select = order(select)
367
628
 
368
- # Limit number of records
369
- select = agnostic_limit(select, limit)
370
-
371
- return Workhorse::DbJob.find_by_sql(select.to_sql).to_a
629
+ return Workhorse::DbJob.find_by_sql(limited_sql(select, limit)).to_a
372
630
  end
373
631
 
374
632
  # Returns a fresh Arel select manager containing the id of all waiting jobs.
375
633
  #
376
634
  # @return [Arel::SelectManager] the select manager
377
635
  def valid_select_id
636
+ now = Time.now
637
+
378
638
  select = table.project(table[:id])
379
639
  select = select.where(table[:state].eq(:waiting))
380
- select = select.where(table[:perform_at].lteq(Time.now).or(table[:perform_at].eq(nil)))
640
+ select = select.where(table[:perform_at].lteq(now).or(table[:perform_at].eq(nil)))
641
+
642
+ # The deadline is enforced here and not only by #expire_due_jobs, which
643
+ # expires at most MAX_EXPIRIES_PER_POLL jobs per poll: a larger backlog
644
+ # would otherwise leave the surplus selectable and performed in the very
645
+ # same poll, past the deadline the caller set.
646
+ if expiry_supported?
647
+ select = select.where(table[:expires_at].gt(now).or(table[:expires_at].eq(nil)))
648
+ end
649
+
381
650
  return select
382
651
  end
383
652
 
653
+ # Returns whether this installation has the columns the expiry feature
654
+ # needs. They are absent until the migration adding them has been run.
655
+ #
656
+ # @return [Boolean]
657
+ # @private
658
+ def expiry_supported?
659
+ return Workhorse::DbJob.column_names.include?('expires_at')
660
+ end
661
+
384
662
  # Returns a fresh Arel select manager containing the id of all waiting jobs,
385
663
  # ordered with {#order}.
386
664
  #
@@ -398,15 +676,19 @@ module Workhorse
398
676
  select.order(Arel.sql('priority').asc).order(Arel.sql('created_at').asc)
399
677
  end
400
678
 
401
- # Limits the number of records
679
+ # Returns the SQL of a select, limited to the given number of records.
680
+ #
681
+ # On Oracle this is `FETCH FIRST`, which applies after `ORDER BY`. Filtering
682
+ # on `ROWNUM` instead - the only option before 12c - numbers the rows
683
+ # before they are sorted, so it returned an arbitrary subset and ignored
684
+ # the priority order altogether.
402
685
  #
403
- # @param select [Arel::SelectManager] the select manager on which to apply
404
- # the limit
686
+ # @param select [Arel::SelectManager] the select manager to limit
405
687
  # @param number [Integer] the maximum number of records to return
406
- # @return [Arel::SelectManager] the resultant select manager
407
- def agnostic_limit(select, number)
408
- return select.where(Arel.sql('ROWNUM').lteq(number)) if @is_oracle
409
- return select.take(number)
688
+ # @return [String] the resultant SQL
689
+ def limited_sql(select, number)
690
+ return "#{select.to_sql} FETCH FIRST #{Integer(number)} ROWS ONLY" if @is_oracle
691
+ return select.take(number).to_sql
410
692
  end
411
693
 
412
694
  # Returns an Array of queue names for which a job may be posted
@@ -72,17 +72,23 @@ module Workhorse
72
72
  @size - @active_threads.value
73
73
  end
74
74
 
75
- # Waits until the pool is shut down. This will wait forever unless you
76
- # eventually call {#shutdown} (either before calling `wait` or after it in
77
- # another thread).
75
+ # Waits until the pool is shut down. Without a timeout this waits forever
76
+ # unless you eventually call {#shutdown} (either before calling `wait` or
77
+ # after it in another thread).
78
78
  #
79
- # @return [void]
80
- def wait
79
+ # @param timeout [Numeric, nil] Seconds to wait at most, or nil for no
80
+ # limit.
81
+ # @return [Boolean] Whether the pool is shut down
82
+ def wait(timeout: nil)
83
+ deadline = timeout ? Process.clock_gettime(Process::CLOCK_MONOTONIC) + timeout : nil
84
+
81
85
  # Here we use a loop-sleep combination instead of using
82
86
  # ThreadPoolExecutor's `wait_for_termination`. See issue #21 for more
83
87
  # information.
84
88
  loop do
85
- break if @executor.shutdown?
89
+ return true if @executor.shutdown?
90
+ return false if deadline && Process.clock_gettime(Process::CLOCK_MONOTONIC) >= deadline
91
+
86
92
  sleep 0.1
87
93
  end
88
94
  end