workhorse 1.5.2 → 2.0.0.rc0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +102 -0
  3. data/README.md +303 -74
  4. data/Rakefile +1 -0
  5. data/VERSION +1 -1
  6. data/bin/rubocop +5 -1
  7. data/lib/generators/workhorse/install_generator.rb +10 -1
  8. data/lib/generators/workhorse/templates/config/initializers/workhorse.rb +55 -0
  9. data/lib/generators/workhorse/templates/create_table_jobs.rb +11 -13
  10. data/lib/generators/workhorse/templates/create_table_workhorse_schedules.rb +29 -0
  11. data/lib/workhorse/daemon.rb +52 -6
  12. data/lib/workhorse/db_job.rb +58 -7
  13. data/lib/workhorse/enqueuer.rb +51 -8
  14. data/lib/workhorse/jobs/cleanup_succeeded_jobs.rb +26 -8
  15. data/lib/workhorse/jobs/detect_late_schedules_job.rb +59 -0
  16. data/lib/workhorse/notifiers/base.rb +55 -0
  17. data/lib/workhorse/notifiers/file_system.rb +66 -0
  18. data/lib/workhorse/notifiers/none.rb +8 -0
  19. data/lib/workhorse/notifiers/redis.rb +227 -0
  20. data/lib/workhorse/performer.rb +28 -0
  21. data/lib/workhorse/poller.rb +289 -47
  22. data/lib/workhorse/pool.rb +12 -6
  23. data/lib/workhorse/schedule.rb +288 -0
  24. data/lib/workhorse/schedules.rb +197 -0
  25. data/lib/workhorse/worker.rb +77 -27
  26. data/lib/workhorse.rb +136 -0
  27. data/test/lib/db_schema.rb +21 -1
  28. data/test/lib/jobs.rb +29 -0
  29. data/test/lib/test_helper.rb +9 -14
  30. data/test/workhorse/daemon_test.rb +33 -0
  31. data/test/workhorse/db_job_test.rb +1 -1
  32. data/test/workhorse/notifier_test.rb +500 -0
  33. data/test/workhorse/poller_test.rb +8 -4
  34. data/test/workhorse/schedule_test.rb +967 -0
  35. data/test/workhorse/worker_test.rb +92 -0
  36. data/workhorse.gemspec +6 -5
  37. metadata +29 -3
@@ -1,7 +1,6 @@
1
1
  module Workhorse
2
2
  # Database poller that discovers and locks jobs for execution.
3
3
  # Handles job querying, global locking, and job distribution to workers.
4
- # Supports both MySQL and Oracle databases with database-specific optimizations.
5
4
  #
6
5
  # @example Basic usage (typically used internally)
7
6
  # poller = Workhorse::Poller.new(worker, proc { true })
@@ -10,8 +9,13 @@ module Workhorse
10
9
  MIN_LOCK_TIMEOUT = 0.1 # In seconds
11
10
  MAX_LOCK_TIMEOUT = 1.0 # In seconds
12
11
 
13
- ORACLE_LOCK_MODE = 6 # X_MODE (exclusive)
14
- ORACLE_LOCK_HANDLE = 478_564_848 # Randomly chosen number
12
+ # Most jobs one poll expires, see {#expire_due_jobs}.
13
+ MAX_EXPIRIES_PER_POLL = 100
14
+
15
+ # Length of one slice of the poller's sleep, in seconds. The poller sleeps
16
+ # in slices rather than for the whole polling interval so that it stays
17
+ # responsive to shutdown, to instant repolling and to notifications.
18
+ SLEEP_SLICE = 0.1
15
19
 
16
20
  # @return [Workhorse::Worker] The worker this poller serves
17
21
  attr_reader :worker
@@ -27,7 +31,6 @@ module Workhorse
27
31
  @worker = worker
28
32
  @running = false
29
33
  @table = Workhorse::DbJob.arel_table
30
- @is_oracle = ActiveRecord::Base.connection.adapter_name == 'OracleEnhanced'
31
34
  @instant_repoll = Concurrent::AtomicBoolean.new(false)
32
35
  @global_lock_fails = 0
33
36
  @max_global_lock_fails_reached = false
@@ -51,6 +54,18 @@ module Workhorse
51
54
 
52
55
  Workhorse.debug_log("[Job worker #{worker.id}] Poller starting")
53
56
 
57
+ begin
58
+ Workhorse.notifier.start
59
+ rescue StandardError => e
60
+ worker.log "Starting the notifier failed, falling back to polling: #{e.class}: #{e.message}", :warn
61
+ end
62
+
63
+ # Only jobs announced from now on concern this worker; anything enqueued
64
+ # earlier is found by the poll that follows.
65
+ @last_notification = notifier_token
66
+
67
+ reconcile_schedules!
68
+
54
69
  clean_stuck_jobs! if Workhorse.clean_stuck_jobs
55
70
 
56
71
  @thread = Thread.new do
@@ -91,6 +106,13 @@ module Workhorse
91
106
  Workhorse.debug_log("[Job worker #{worker.id}] Poller shutting down")
92
107
  @running = false
93
108
  wait
109
+
110
+ begin
111
+ Workhorse.notifier.stop
112
+ rescue StandardError => e
113
+ worker.log "Stopping the notifier failed: #{e.class}: #{e.message}", :warn
114
+ end
115
+
94
116
  Workhorse.debug_log("[Job worker #{worker.id}] Poller shut down")
95
117
  end
96
118
 
@@ -112,6 +134,43 @@ module Workhorse
112
134
 
113
135
  private
114
136
 
137
+ # Brings the schedules table in line with the registry, see
138
+ # {Workhorse::Schedule.reconcile!}.
139
+ #
140
+ # A worker that cannot reconcile still works through the queue, so this
141
+ # reports rather than raises.
142
+ #
143
+ # @return [void]
144
+ # @private
145
+ def reconcile_schedules!
146
+ @schedules_available = Workhorse::Schedule.table_exists?
147
+
148
+ unless @schedules_available
149
+ if Workhorse::Schedules.any?
150
+ message = 'Schedules are declared but the workhorse_schedules table does not exist. ' \
151
+ 'Run the migration that creates it; no scheduled job will run until then.'
152
+ worker.log message, :error
153
+
154
+ begin
155
+ Workhorse.on_exception.call(StandardError.new(message))
156
+ rescue Exception => e
157
+ # Reported through the rescue below would feed the callback its
158
+ # own failure; escaping leaves the worker half-started.
159
+ Workhorse.debug_log("on_exception failed: #{e.class}: #{e.message}")
160
+ end
161
+ end
162
+
163
+ return
164
+ end
165
+
166
+ with_global_lock timeout: MAX_LOCK_TIMEOUT do
167
+ Workhorse::Schedule.reconcile!
168
+ end
169
+ rescue Exception => e
170
+ worker.log %(Could not reconcile schedules: #{e.message}), :error
171
+ Workhorse.on_exception.call(e)
172
+ end
173
+
115
174
  # Cleans up jobs stuck in locked or started states from dead processes.
116
175
  # Only cleans jobs from the current hostname.
117
176
  #
@@ -174,7 +233,9 @@ module Workhorse
174
233
  end
175
234
  end
176
235
 
177
- # Sleeps for the configured polling interval with instant repoll support.
236
+ # Sleeps for the configured polling interval, returning early for an
237
+ # instant repoll, for shutdown, or as soon as a notification announces a
238
+ # newly enqueued job.
178
239
  #
179
240
  # @return [void]
180
241
  # @private
@@ -182,37 +243,91 @@ module Workhorse
182
243
  remaining = worker.polling_interval
183
244
 
184
245
  while running? && remaining > 0 && @instant_repoll.false?
185
- Kernel.sleep 0.1
186
- remaining -= 0.1
246
+ Kernel.sleep SLEEP_SLICE
247
+ remaining -= SLEEP_SLICE
248
+
249
+ next unless notified?
250
+
251
+ worker.log 'Job was announced, polling ahead of the interval', :debug
252
+ break
253
+ end
254
+
255
+ # Time left on the clock means the sleep was cut short, which #poll
256
+ # passes on as `count_failures: false`.
257
+ @poll_brought_forward = remaining > 0
258
+ end
259
+
260
+ # Returns whether a job has been announced since this poller last looked,
261
+ # and records what it saw.
262
+ #
263
+ # A worker with no idle thread deliberately leaves the token untouched, so
264
+ # that the notification is still pending once it has capacity again.
265
+ #
266
+ # @return [Boolean]
267
+ # @private
268
+ def notified?
269
+ return false unless worker.accepting_jobs?
270
+ return false if worker.idle.zero?
271
+
272
+ token = notifier_token
273
+
274
+ return false if token.nil? || token == @last_notification
275
+
276
+ @last_notification = token
277
+
278
+ return true
279
+ end
280
+
281
+ # Reads the notifier's token, returning nil if it cannot be read.
282
+ #
283
+ # A notifier is an accelerator, so a broken one must cost latency rather
284
+ # than take the worker down: anything raised here would reach the poller's
285
+ # own rescue, which shuts the worker down. As this runs on every sleep
286
+ # slice, the failure is reported once rather than many times a second.
287
+ #
288
+ # @return [Object, nil]
289
+ # @private
290
+ def notifier_token
291
+ return Workhorse.notifier.token
292
+ rescue StandardError => e
293
+ message = "Reading the notifier failed, falling back to polling: #{e.class}: #{e.message}"
294
+
295
+ if @notifier_failed
296
+ worker.log message, :debug
297
+ else
298
+ @notifier_failed = true
299
+ worker.log message, :warn
187
300
  end
301
+
302
+ return nil
188
303
  end
189
304
 
190
305
  # Executes a block with a global database lock.
191
- # Supports both MySQL GET_LOCK and Oracle DBMS_LOCK.
192
306
  #
193
307
  # @param name [Symbol] Lock name identifier
194
308
  # @param timeout [Integer] Lock timeout in seconds
309
+ # @param count_failures [Boolean] Whether a failure to obtain the lock
310
+ # counts towards {Workhorse.max_global_lock_fails}
195
311
  # @yield Block to execute while holding the lock
196
312
  # @return [void]
197
313
  # @private
198
- def with_global_lock(name: :workhorse, timeout: 2, &_block)
314
+ def with_global_lock(name: :workhorse, timeout: 2, count_failures: true, &_block)
199
315
  begin # rubocop:disable Style/RedundantBegin
200
- if @is_oracle
201
- result = Workhorse::DbJob.connection.select_all(
202
- "SELECT DBMS_LOCK.REQUEST(#{ORACLE_LOCK_HANDLE}, #{ORACLE_LOCK_MODE}, #{timeout}) FROM DUAL"
203
- ).first.values.last
204
-
205
- success = result == 0
206
- else
207
- result = Workhorse::DbJob.connection.select_all(
208
- "SELECT GET_LOCK(CONCAT(DATABASE(), '_#{name}'), #{timeout})"
209
- ).first.values.last
210
- success = result == 1
211
- end
316
+ result = Workhorse::DbJob.connection.select_all(
317
+ "SELECT GET_LOCK(CONCAT(DATABASE(), '_#{name}'), #{timeout})"
318
+ ).first.values.last
319
+ success = result == 1
212
320
 
213
321
  if success
214
322
  @global_lock_fails = 0
215
323
  @max_global_lock_fails_reached = false
324
+ elsif !count_failures
325
+ # Losing the race for the lock is the expected outcome when several
326
+ # workers were woken by the same announcement, and says nothing about
327
+ # a crashed worker. Counting it would let the alarm below fire within
328
+ # seconds rather than after the polling intervals it is calibrated
329
+ # for.
330
+ worker.log 'Could not obtain global lock for a poll that was brought forward, skipping it.', :debug
216
331
  else
217
332
  @global_lock_fails += 1
218
333
 
@@ -247,11 +362,7 @@ module Workhorse
247
362
  yield
248
363
  ensure
249
364
  if success
250
- if @is_oracle
251
- Workhorse::DbJob.connection.execute("SELECT DBMS_LOCK.RELEASE(#{ORACLE_LOCK_HANDLE}) FROM DUAL")
252
- else
253
- Workhorse::DbJob.connection.execute("SELECT RELEASE_LOCK(CONCAT(DATABASE(), '_#{name}'))")
254
- end
365
+ Workhorse::DbJob.connection.execute("SELECT RELEASE_LOCK(CONCAT(DATABASE(), '_#{name}'))")
255
366
  end
256
367
  end
257
368
  end
@@ -265,10 +376,20 @@ module Workhorse
265
376
 
266
377
  @instant_repoll.make_false
267
378
 
379
+ # A poll the sleep cut short is not the scheduled one the lock-failure
380
+ # alarm is calibrated against, see #with_global_lock.
381
+ brought_forward = @poll_brought_forward
382
+ @poll_brought_forward = false
383
+
384
+ expired = []
385
+
268
386
  timeout = worker.polling_interval.clamp(MIN_LOCK_TIMEOUT, MAX_LOCK_TIMEOUT)
269
- with_global_lock timeout: timeout do
387
+ with_global_lock timeout: timeout, count_failures: !brought_forward do
270
388
  job_ids = []
271
389
 
390
+ materialize_schedules if Workhorse::Schedules.any? && @schedules_available
391
+ expired = expire_due_jobs
392
+
272
393
  Workhorse.tx_callback.call do
273
394
  # As we are the only thread posting into the worker pool, it is safe to
274
395
  # get the number of idle threads without mutex synchronization. The
@@ -303,6 +424,11 @@ module Workhorse
303
424
  job_ids.each { |job_id| worker.perform(job_id) } if running? && worker.accepting_jobs?
304
425
  end
305
426
 
427
+ # Deliberately outside the global lock: the callback is the application's
428
+ # and may do something slow, such as sending mail, which would otherwise
429
+ # block every other worker's poll.
430
+ notify_expired(expired)
431
+
306
432
  # Record that this worker successfully polled. Done at the very end so it
307
433
  # only advances when the poll actually completed (a poll that raises never
308
434
  # reaches here). Skipped on the early return above when the worker is no
@@ -310,6 +436,117 @@ module Workhorse
310
436
  worker.heartbeat!
311
437
  end
312
438
 
439
+ # Materialises the occurrences that have come due into jobs.
440
+ #
441
+ # A failing schedule must not take the worker down with it, nor stop the
442
+ # other schedules, so each is handled on its own.
443
+ #
444
+ # @return [void]
445
+ # @private
446
+ def materialize_schedules
447
+ Workhorse::Schedule.due.to_a.each do |schedule|
448
+ next if schedule.definition.nil?
449
+
450
+ occurrences, next_at = schedule.pending_occurrences
451
+
452
+ # The claim advances the schedule past these occurrences, so
453
+ # committing it before the jobs exist would lose them for good if
454
+ # enqueuing then failed - the schedule would move on and
455
+ # DetectLateSchedulesJob would see nothing wrong.
456
+ Workhorse.tx_callback.call do
457
+ next unless schedule.claim!(next_at)
458
+
459
+ occurrences.each do |occurrence|
460
+ db_job = schedule.enqueue!(occurrence)
461
+ worker.log "Materialized schedule #{schedule.key.inspect} for #{occurrence} as job #{db_job.id}", :debug
462
+ end
463
+ end
464
+
465
+ @failed_schedules&.delete(schedule.key)
466
+ rescue Exception => e
467
+ report_schedule_failure(schedule, e)
468
+ end
469
+
470
+ return
471
+ rescue Exception => e
472
+ # The query itself failed, so no individual schedule can be blamed. This
473
+ # must not reach the poller's own rescue, which shuts the worker down.
474
+ worker.log %(Could not query due schedules: #{e.message}), :error
475
+ Workhorse.on_exception.call(e)
476
+ end
477
+
478
+ # Reports a schedule that could not be materialized.
479
+ #
480
+ # Reported once per schedule per worker: a schedule that can never enqueue
481
+ # - a renamed job class, params its constructor rejects - fails on every
482
+ # poll, and one typo must not turn into a notification every polling
483
+ # interval for as long as the worker runs.
484
+ #
485
+ # @param schedule [Workhorse::Schedule]
486
+ # @param exception [Exception]
487
+ # @return [void]
488
+ # @private
489
+ def report_schedule_failure(schedule, exception)
490
+ message = %(Could not materialize schedule #{schedule.key.inspect}: #{exception.message})
491
+ @failed_schedules ||= Set.new
492
+
493
+ if @failed_schedules.include?(schedule.key)
494
+ worker.log message, :debug
495
+ return
496
+ end
497
+
498
+ @failed_schedules << schedule.key
499
+ worker.log message, :error
500
+ Workhorse.on_exception.call(exception)
501
+
502
+ return
503
+ end
504
+
505
+ # Marks jobs that passed their deadline before any worker got to them.
506
+ #
507
+ # @return [Array<Workhorse::DbJob>] The jobs that were expired
508
+ # @private
509
+ def expire_due_jobs
510
+ return [] unless expiry_supported?
511
+
512
+ expired = []
513
+
514
+ Workhorse.tx_callback.call do
515
+ rel = Workhorse::DbJob.waiting.where(Workhorse::DbJob.arel_table[:expires_at].lteq(Time.now))
516
+
517
+ # Bounded, as this runs while the poller holds the global lock: a
518
+ # backlog of jobs that all expired at once - workers down over a
519
+ # weekend, or a bulk enqueue - would otherwise turn one poll into
520
+ # thousands of updates that block every other worker. The rest is
521
+ # expired by the following polls, most overdue first.
522
+ rel.order(:expires_at).limit(MAX_EXPIRIES_PER_POLL).each do |db_job|
523
+ db_job.mark_expired!
524
+ expired << db_job
525
+ worker.log "Job #{db_job.id} passed its deadline of #{db_job.expires_at} and was not run", :warn
526
+ end
527
+ end
528
+
529
+ return expired
530
+ end
531
+
532
+ # Calls {Workhorse.on_job_expired} for each expired job, keeping a failing
533
+ # callback away from the poller's own error handling, which would shut the
534
+ # worker down.
535
+ #
536
+ # @param expired [Array<Workhorse::DbJob>]
537
+ # @return [void]
538
+ # @private
539
+ def notify_expired(expired)
540
+ expired.each do |db_job|
541
+ Workhorse.on_job_expired.call(db_job)
542
+ rescue Exception => e
543
+ worker.log %(on_job_expired failed for job #{db_job.id}: #{e.message}), :error
544
+ Workhorse.on_exception.call(e)
545
+ end
546
+
547
+ return
548
+ end
549
+
313
550
  # Returns an array of {Workhorse::DbJob}s that can be started.
314
551
  # Uses complex SQL with UNIONs to respect queue ordering and limits.
315
552
  #
@@ -335,7 +572,7 @@ module Workhorse
335
572
  # any presumptions on the order.
336
573
  record_number = queue.nil? ? limit : 1
337
574
 
338
- union_parts << agnostic_limit(select, record_number)
575
+ union_parts << select.take(record_number)
339
576
  end
340
577
 
341
578
  return [] if union_parts.empty?
@@ -346,9 +583,6 @@ module Workhorse
346
583
  # contained within.
347
584
  # Additionally, each of the subselects and the final union select is given
348
585
  # an alias to comply with MySQL requirements.
349
- # These aliases are added directly instead of using Arel `as`, because it
350
- # uses the keyword 'AS' in SQL generated for Oracle, which is invalid for
351
- # table aliases.
352
586
  union_query_sql = '('
353
587
  union_query_sql += "SELECT * FROM (#{union_parts.shift.to_sql}) union_0"
354
588
  union_parts.each_with_index do |part, idx|
@@ -366,7 +600,7 @@ module Workhorse
366
600
  select = order(select)
367
601
 
368
602
  # Limit number of records
369
- select = agnostic_limit(select, limit)
603
+ select = select.take(limit)
370
604
 
371
605
  return Workhorse::DbJob.find_by_sql(select.to_sql).to_a
372
606
  end
@@ -375,12 +609,32 @@ module Workhorse
375
609
  #
376
610
  # @return [Arel::SelectManager] the select manager
377
611
  def valid_select_id
612
+ now = Time.now
613
+
378
614
  select = table.project(table[:id])
379
615
  select = select.where(table[:state].eq(:waiting))
380
- select = select.where(table[:perform_at].lteq(Time.now).or(table[:perform_at].eq(nil)))
616
+ select = select.where(table[:perform_at].lteq(now).or(table[:perform_at].eq(nil)))
617
+
618
+ # The deadline is enforced here and not only by #expire_due_jobs, which
619
+ # expires at most MAX_EXPIRIES_PER_POLL jobs per poll: a larger backlog
620
+ # would otherwise leave the surplus selectable and performed in the very
621
+ # same poll, past the deadline the caller set.
622
+ if expiry_supported?
623
+ select = select.where(table[:expires_at].gt(now).or(table[:expires_at].eq(nil)))
624
+ end
625
+
381
626
  return select
382
627
  end
383
628
 
629
+ # Returns whether this installation has the columns the expiry feature
630
+ # needs. They are absent until the migration adding them has been run.
631
+ #
632
+ # @return [Boolean]
633
+ # @private
634
+ def expiry_supported?
635
+ return Workhorse::DbJob.column_names.include?('expires_at')
636
+ end
637
+
384
638
  # Returns a fresh Arel select manager containing the id of all waiting jobs,
385
639
  # ordered with {#order}.
386
640
  #
@@ -398,17 +652,6 @@ module Workhorse
398
652
  select.order(Arel.sql('priority').asc).order(Arel.sql('created_at').asc)
399
653
  end
400
654
 
401
- # Limits the number of records
402
- #
403
- # @param select [Arel::SelectManager] the select manager on which to apply
404
- # the limit
405
- # @param number [Integer] the maximum number of records to return
406
- # @return [Arel::SelectManager] the resultant select manager
407
- def agnostic_limit(select, number)
408
- return select.where(Arel.sql('ROWNUM').lteq(number)) if @is_oracle
409
- return select.take(number)
410
- end
411
-
412
655
  # Returns an Array of queue names for which a job may be posted
413
656
  #
414
657
  # This is done in multiple steps. First, all queues with jobs that are in
@@ -453,10 +696,9 @@ module Workhorse
453
696
  queues = select.project(:queue)
454
697
 
455
698
  # Note that `select_values` is used here on purpose: `execute` does not
456
- # return a result set on every adapter (the Oracle enhanced adapter, for
457
- # instance, returns `true` for queries), while `select_values` is
699
+ # return a result set on every adapter, while `select_values` is
458
700
  # implemented in terms of `exec_query` and thus behaves the same on the
459
- # mysql2, trilogy and Oracle enhanced adapters.
701
+ # mysql2 and trilogy adapters.
460
702
  return Workhorse::DbJob.connection.select_values(queues.distinct.to_sql)
461
703
  end
462
704
  end
@@ -72,17 +72,23 @@ module Workhorse
72
72
  @size - @active_threads.value
73
73
  end
74
74
 
75
- # Waits until the pool is shut down. This will wait forever unless you
76
- # eventually call {#shutdown} (either before calling `wait` or after it in
77
- # another thread).
75
+ # Waits until the pool is shut down. Without a timeout this waits forever
76
+ # unless you eventually call {#shutdown} (either before calling `wait` or
77
+ # after it in another thread).
78
78
  #
79
- # @return [void]
80
- def wait
79
+ # @param timeout [Numeric, nil] Seconds to wait at most, or nil for no
80
+ # limit.
81
+ # @return [Boolean] Whether the pool is shut down
82
+ def wait(timeout: nil)
83
+ deadline = timeout ? Process.clock_gettime(Process::CLOCK_MONOTONIC) + timeout : nil
84
+
81
85
  # Here we use a loop-sleep combination instead of using
82
86
  # ThreadPoolExecutor's `wait_for_termination`. See issue #21 for more
83
87
  # information.
84
88
  loop do
85
- break if @executor.shutdown?
89
+ return true if @executor.shutdown?
90
+ return false if deadline && Process.clock_gettime(Process::CLOCK_MONOTONIC) >= deadline
91
+
86
92
  sleep 0.1
87
93
  end
88
94
  end