workhorse 1.5.2 → 2.0.0.rc1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. checksums.yaml +4 -4
  2. data/.github/workflows/ruby.yml +137 -1
  3. data/CHANGELOG.md +150 -0
  4. data/Gemfile +16 -1
  5. data/README.md +316 -72
  6. data/Rakefile +1 -0
  7. data/VERSION +1 -1
  8. data/bin/rubocop +5 -1
  9. data/lib/generators/workhorse/install_generator.rb +10 -1
  10. data/lib/generators/workhorse/templates/config/initializers/workhorse.rb +55 -0
  11. data/lib/generators/workhorse/templates/create_table_jobs.rb +15 -2
  12. data/lib/generators/workhorse/templates/create_table_workhorse_schedules.rb +42 -0
  13. data/lib/workhorse/daemon/shell_handler.rb +4 -1
  14. data/lib/workhorse/daemon.rb +57 -7
  15. data/lib/workhorse/db_job.rb +98 -8
  16. data/lib/workhorse/enqueuer.rb +51 -8
  17. data/lib/workhorse/jobs/cleanup_succeeded_jobs.rb +26 -8
  18. data/lib/workhorse/jobs/detect_late_schedules_job.rb +59 -0
  19. data/lib/workhorse/notifiers/base.rb +55 -0
  20. data/lib/workhorse/notifiers/file_system.rb +66 -0
  21. data/lib/workhorse/notifiers/none.rb +8 -0
  22. data/lib/workhorse/notifiers/redis.rb +227 -0
  23. data/lib/workhorse/performer.rb +29 -2
  24. data/lib/workhorse/poller.rb +303 -21
  25. data/lib/workhorse/pool.rb +12 -6
  26. data/lib/workhorse/schedule.rb +288 -0
  27. data/lib/workhorse/schedules.rb +197 -0
  28. data/lib/workhorse/worker.rb +102 -31
  29. data/lib/workhorse.rb +136 -0
  30. data/test/lib/db_schema.rb +36 -3
  31. data/test/lib/jobs.rb +29 -0
  32. data/test/lib/test_helper.rb +113 -20
  33. data/test/workhorse/daemon_test.rb +33 -0
  34. data/test/workhorse/db_job_test.rb +2 -4
  35. data/test/workhorse/notifier_test.rb +487 -0
  36. data/test/workhorse/performer_test.rb +7 -9
  37. data/test/workhorse/poller_test.rb +97 -23
  38. data/test/workhorse/schedule_test.rb +967 -0
  39. data/test/workhorse/worker_test.rb +201 -76
  40. data/workhorse.gemspec +6 -5
  41. metadata +29 -3
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 9ffbb2c24919d7b745fd87bd6a8675fdae5448a6c0e11c3af486d34baf5a9343
4
- data.tar.gz: 9cf2270f0c3a65d244d679baf56367033e314e4a8b7ed11cc17f4d07e4a6fe63
3
+ metadata.gz: 84d0f28bfeb67292f3de373115ada4ef7c2cff27b29e847af29326733ba89841
4
+ data.tar.gz: f69f1d82c2c9c15f47f87d8f8390c31e655596636383b9cadefc8c240d8f43b6
5
5
  SHA512:
6
- metadata.gz: 22b02451a64ae383b75616b9556cc05927f479f47db7d7993347df0f520d36ebfda8c7f3f4de49574c84499dfd603285bab83295df72c6100b0f6f438730d576
7
- data.tar.gz: '0549ab8a2e2683dd13b60bc85a5d2d85a260dee938a13714c45fe1233af217a433af2e71f899a7056c3afd8f503f6f94e1a5b9cee5e97bbf5fb93fed9ae75a03'
6
+ metadata.gz: 00c490f9e4aa2b57066b8b974e38286509cb156cb27f79dc7d7a355c9797338eaa8dcf33a5dc9c07e2f3bbe37878d0be5e82b7f098da9e45194ad0fe5b064d6f
7
+ data.tar.gz: e0116907c47ae31d9fa5a5626e331a503286667966aa974255e8d9f816c363fe3a9e55406e5f79c9a3da7e82abe4b0648b505de3e279992a58e1864d847edaef
@@ -40,10 +40,146 @@ jobs:
40
40
  - name: Run rake tests
41
41
  uses: nick-fields/retry@v2
42
42
  with:
43
- timeout_seconds: 120
43
+ # Not much above what one run takes on a loaded runner.
44
+ timeout_seconds: 300
44
45
  retry_on: any
45
46
  max_attempts: 3
46
47
  command: bundle exec rake test TESTOPTS='--verbose'
48
+ # A timed-out attempt can leave the suite and its daemon workers
49
+ # running - the workers call setsid, so killing the attempt does not
50
+ # reach them - and the next attempt then runs against them. KILL, as
51
+ # a worker traps TERM. Clear their pidfiles as well.
52
+ on_retry_command: |
53
+ pkill -KILL -f "^Workhorse .*: $GITHUB_WORKSPACE" || true
54
+ pkill -KILL -f "[r]ake_test_loader" || true
55
+ sleep 1
56
+ rm -f tmp/pids/*
57
+
58
+ test-oracle:
59
+ runs-on: ubuntu-latest
60
+
61
+ # Oracle gets a job of its own rather than a matrix entry: it needs a
62
+ # service container and the Instant Client, and it is run against a single
63
+ # Ruby version, as what is being covered is the database dialect rather
64
+ # than the interpreter.
65
+ services:
66
+ oracle:
67
+ image: gvenzl/oracle-free:23-slim
68
+ env:
69
+ ORACLE_PASSWORD: oracle
70
+ APP_USER: workhorse
71
+ APP_USER_PASSWORD: workhorse
72
+ ports:
73
+ - 1521:1521
74
+ options: >-
75
+ --health-cmd healthcheck.sh
76
+ --health-interval 20s
77
+ --health-timeout 10s
78
+ --health-retries 30
79
+
80
+ env:
81
+ DB_ADAPTER: oracle_enhanced
82
+ DB_NAME: FREEPDB1
83
+ DB_USERNAME: workhorse
84
+ DB_PASSWORD: workhorse
85
+ DB_HOST: localhost
86
+ DB_PORT: 1521
87
+ LD_LIBRARY_PATH: /opt/oracle/instantclient
88
+
89
+ steps:
90
+ - uses: actions/checkout@v2
91
+
92
+ # ruby-oci8 compiles against the Instant Client, so both the runtime
93
+ # library and the SDK headers are needed; basiclite alone does not carry
94
+ # oci.h.
95
+ - name: Install Oracle Instant Client
96
+ run: |
97
+ sudo apt-get update
98
+ # The package was renamed in Ubuntu 24.04, which `ubuntu-latest` now
99
+ # is; keep working on both.
100
+ sudo apt-get install -y unzip
101
+ sudo apt-get install -y libaio1 || sudo apt-get install -y libaio1t64
102
+ # The rename came with a soname change to libaio.so.1t64, while the
103
+ # Oracle client still links against libaio.so.1. Without this the
104
+ # client library cannot be linked, and ruby-oci8's version probe
105
+ # fails with a TypeError that says nothing about the cause.
106
+ if ! ldconfig -p | grep -qE 'libaio\.so\.1$|libaio\.so\.1 '; then
107
+ t64=$(ls /usr/lib/*/libaio.so.1t64 2>/dev/null | head -1)
108
+ if [ -n "$t64" ]; then sudo ln -sfn "$t64" "$(dirname "$t64")/libaio.so.1"; fi
109
+ fi
110
+ sudo mkdir -p /opt/oracle
111
+ cd /tmp
112
+ # Pinned to 21.13 rather than taking the current release: ruby-oci8
113
+ # 2.2 fails its `OCI_MAJOR_VERSION` probe against an Instant Client
114
+ # 23 header, after finding the library and the SDK correctly. A 21
115
+ # client talks to the 23 server used below.
116
+ base=https://download.oracle.com/otn_software/linux/instantclient/2113000
117
+ curl -sSLO $base/instantclient-basiclite-linux.x64-21.13.0.0.0dbru.zip
118
+ curl -sSLO $base/instantclient-sdk-linux.x64-21.13.0.0.0dbru.zip
119
+ sudo unzip -oq 'instantclient-basiclite-linux.x64-*.zip' -d /opt/oracle
120
+ sudo unzip -oq 'instantclient-sdk-linux.x64-*.zip' -d /opt/oracle
121
+ # The archives unpack into a versioned directory, whose name changes
122
+ # with every release; the fixed symlink is what LD_LIBRARY_PATH and
123
+ # the ruby-oci8 build both point at.
124
+ sudo ln -sfn "$(find /opt/oracle -maxdepth 1 -type d -name 'instantclient_*' | head -1)" /opt/oracle/instantclient
125
+ # The basiclite package ships only the versioned libraries, and
126
+ # linking a conftest against -lclntsh needs the unversioned name.
127
+ cd /opt/oracle/instantclient
128
+ for lib in libclntsh libocci; do
129
+ versioned=$(ls $lib.so.* 2>/dev/null | head -1)
130
+ if [ -n "$versioned" ]; then sudo ln -sfn "$versioned" "$lib.so"; fi
131
+ done
132
+ echo /opt/oracle/instantclient | sudo tee /etc/ld.so.conf.d/oracle-instantclient.conf
133
+ sudo ldconfig
134
+
135
+ - name: Set up Ruby
136
+ uses: ruby/setup-ruby@v1
137
+ with:
138
+ ruby-version: '3.2.1'
139
+
140
+ - name: Install bundle including the oracle group
141
+ run: |
142
+ bundle config set --local with oracle
143
+ bundle install --jobs 4 --retry 3
144
+
145
+ # extconf.rb only says to consult mkmf.log, which is otherwise lost with
146
+ # the runner.
147
+ - name: Show the native extension build log
148
+ if: failure()
149
+ run: find / -name mkmf.log -path '*ruby-oci8*' -exec cat {} +
150
+
151
+ # The global lock is DBMS_LOCK on Oracle, and execute permission on that
152
+ # package is not granted to a new schema by default. SYS owns the
153
+ # package, and only it can grant it, so this connects as SYSDBA.
154
+ - name: Grant DBMS_LOCK to the test schema
155
+ run: |
156
+ bundle exec ruby -e '
157
+ require "active_record"
158
+ require "active_record/connection_adapters/oracle_enhanced_adapter"
159
+ ActiveRecord::Base.establish_connection(
160
+ adapter: "oracle_enhanced", database: "FREEPDB1",
161
+ username: "sys", password: "oracle", privilege: "SYSDBA",
162
+ host: "localhost", port: 1521
163
+ )
164
+ ActiveRecord::Base.connection.execute("GRANT EXECUTE ON DBMS_LOCK TO workhorse")
165
+ '
166
+
167
+ - name: Run rake tests
168
+ uses: nick-fields/retry@v2
169
+ with:
170
+ timeout_seconds: 600
171
+ retry_on: any
172
+ max_attempts: 3
173
+ command: bundle exec rake test TESTOPTS='--verbose'
174
+ # A timed-out attempt can leave the suite and its daemon workers
175
+ # running - the workers call setsid, so killing the attempt does not
176
+ # reach them - and the next attempt then runs against them. KILL, as
177
+ # a worker traps TERM. Clear their pidfiles as well.
178
+ on_retry_command: |
179
+ pkill -KILL -f "^Workhorse .*: $GITHUB_WORKSPACE" || true
180
+ pkill -KILL -f "[r]ake_test_loader" || true
181
+ sleep 1
182
+ rm -f tmp/pids/*
47
183
 
48
184
  linters:
49
185
  runs-on: ubuntu-latest
data/CHANGELOG.md CHANGED
@@ -1,5 +1,155 @@
1
1
  # Workhorse Changelog
2
2
 
3
+ ## 2.0.0.rc1 - 2026-09-30
4
+
5
+ Sitrox reference: #154443.
6
+
7
+ ### Added
8
+
9
+ * Oracle 12c or later is supported again, and is now covered by CI against
10
+ `activerecord-oracle_enhanced-adapter` rather than only by hand. The global
11
+ lock uses `DBMS_LOCK`, and the generated migrations leave out the index
12
+ prefix lengths Oracle rejects while naming every index within the 30
13
+ characters it allows before 12.2. See
14
+ [Database support](README.md#database-support).
15
+
16
+ 2.0.0.rc0 dropped Oracle; anyone who took that release up and needs it can
17
+ move straight to this one. Grant the schema execute permission on
18
+ `DBMS_LOCK`, as described under Database support.
19
+
20
+ ### Fixed
21
+
22
+ * On Oracle, jobs are picked up in priority order. 1.x limited the rows it
23
+ selected by filtering on `ROWNUM`, which Oracle numbers before sorting, so
24
+ both the job taken from each queue and the jobs a poll picked up were an
25
+ arbitrary subset regardless of `priority`. Limiting now uses
26
+ `FETCH FIRST … ROWS ONLY`, which applies after the sort and is why 12c is
27
+ now the minimum.
28
+
29
+ * A poll that finds the global lock taken now waits for it with a polling
30
+ interval below half a second. The lock timeout follows the interval, and
31
+ MySQL and Oracle take it as whole seconds, so they rounded it down to not
32
+ waiting at all - only MariaDB honours a fraction. Each such poll gave up
33
+ at once and counted towards `Workhorse.max_global_lock_fails`, so several
34
+ workers contending for the lock could trigger its alarm about a crashed
35
+ worker. The timeout is now rounded up to whole seconds.
36
+
37
+ * A soft restart (`USR1`) no longer wedges a worker whose `tmp/pids` does not
38
+ exist yet, as on a fresh checkout or deployment. The worker touched its
39
+ shutdown file there after it had stopped accepting jobs, so failing on it
40
+ left a worker that took no jobs and never exited. The directory is now
41
+ created when missing, and a shutdown file that still cannot be written no
42
+ longer holds up the shutdown. The heartbeat and the memory-limit shutdown,
43
+ which write to the same directory, create it as well.
44
+
45
+ * `Workhorse.clean_stuck_jobs` works on Oracle. It is built on splitting
46
+ `locked_by` into host, PID and random component, which was written with
47
+ MySQL's `substring_index` and so raised on Oracle for as long as the option
48
+ has existed — visible only to an installation that turned it on, as it is
49
+ off by default.
50
+
51
+ ## 2.0.0.rc0 - 2026-09-29
52
+
53
+ Sitrox reference: #154443.
54
+
55
+ ### Breaking changes
56
+
57
+ * **Support for Oracle is dropped.** Workhorse supports MySQL and MariaDB
58
+ only. The Oracle branches of the global lock, the row limiting and the
59
+ generated migrations are gone, along with
60
+ `Workhorse::Poller::ORACLE_LOCK_MODE` and `ORACLE_LOCK_HANDLE`. It was never
61
+ covered by CI, so it only ever had manual verification. An Oracle
62
+ installation has no upgrade path and should stay on 1.x. See
63
+ [Database support](README.md#database-support).
64
+
65
+ * A forked daemon worker no longer runs the `at_exit` handlers registered by
66
+ the process that started it. They belong to that process, and one that waits
67
+ on threads the fork did not inherit hangs a worker which has already
68
+ finished - which then ignores `TERM`. This skips interpreter finalisation as
69
+ a whole, so buffered output is dropped too; a worker that dies of an
70
+ unhandled exception now reports it through `Workhorse.on_exception` and
71
+ exits non-zero.
72
+
73
+ ### Added
74
+
75
+ * *Scheduling*: workhorse runs jobs on a cron schedule itself, without an
76
+ external scheduler process. Each schedule owns a row in the new
77
+ `workhorse_schedules` table holding the occurrence it is waiting for, so an
78
+ occurrence whose time passes while nothing is running is not lost, and what
79
+ happens to it is a per-schedule `catch_up` policy. Timezones and daylight
80
+ saving are handled. See [Scheduling](README.md#scheduling).
81
+
82
+ * *Notifications*: enqueuing a job announces it and a waiting worker polls
83
+ straight away, rather than sleeping out its polling interval. Polling stays
84
+ the floor. `:file` suits workers sharing a filesystem with the application
85
+ and `:redis` those that do not; both are off by default. See
86
+ [Notifications](README.md#notifications).
87
+
88
+ * Job deadlines and lateness reporting: `expires_at`, `max_lateness`, the new
89
+ `expired` state, `Workhorse.on_job_expired`, `Workhorse.on_job_late` and
90
+ `Workhorse::DbJob#lateness`. A job past its deadline is expired rather than
91
+ performed. See [Lateness and deadlines](README.md#lateness-and-deadlines).
92
+
93
+ * `Workhorse::Jobs::DetectLateSchedulesJob`, which reports schedules whose
94
+ next occurrence lies well in the past. Neither callback above can fire for a
95
+ job that was never created, so this is what catches materialization having
96
+ stopped. See
97
+ [Detecting schedules that stopped](README.md#detecting-schedules-that-stopped).
98
+
99
+ * `Workhorse.shutdown_timeout`, the seconds the daemon's `stop` waits for a
100
+ worker before killing it. Defaults to 300, `nil` restores the previous
101
+ behaviour. A worker that ignores `TERM` used to leave `stop` - and whatever
102
+ waits on it, usually a deployment - looping forever.
103
+
104
+ * `Workhorse.enqueue_job_class`, the keyword arguments `expires_at:` and
105
+ `max_lateness:` on `Workhorse.enqueue` and `Workhorse.enqueue_active_job`,
106
+ `priority:` on the latter, and the scope `Workhorse::DbJob.expired`.
107
+
108
+ * The composite indexes `[state, perform_at]`, `[state, priority, created_at]`
109
+ and `[state, expires_at]` in the generated `jobs` migration, replacing the
110
+ single-column index on `state`. `rails generate workhorse:install` now emits
111
+ two migrations rather than one.
112
+
113
+ * `fugit` as a runtime dependency, for parsing cron expressions.
114
+
115
+ ### Changed
116
+
117
+ * `Workhorse::Jobs::CleanupSucceededJobs` also deletes jobs in the new
118
+ `expired` state, with a `states` argument to opt out. A schedule using
119
+ `expires_after` that regularly misses its window would otherwise grow the
120
+ jobs table without bound.
121
+
122
+ * Failures to obtain the global lock on a poll that a notification or an
123
+ instant repoll brought forward no longer count towards
124
+ `max_global_lock_fails`, and are logged at `debug`. Several workers woken by
125
+ one announcement race for the lock and all but one lose, which says nothing
126
+ about a crashed worker.
127
+
128
+ * `Workhorse::DbJob#reset!` accepts `expired` as the terminal state it is.
129
+
130
+ ### Fixed
131
+
132
+ * A deadlock between shutting a worker down and the poller posting a job.
133
+ `Worker#shutdown` held the worker's mutex while waiting for the poller
134
+ thread, which could be waiting for that same mutex in `Worker#perform`. The
135
+ worker then ignored `TERM`. A job that was locked but cannot be performed
136
+ because the worker is shutting down is now reset to `waiting` rather than
137
+ left locked, where it would have blocked its queue.
138
+
139
+ * `Worker#shutdown` raising when called concurrently, which the daemon does by
140
+ sending both `TERM` and `INT`.
141
+
142
+ ### Documentation
143
+
144
+ * PostgreSQL is documented as unsupported. Workers emit `GET_LOCK` on every
145
+ poll, which PostgreSQL does not provide, so a worker fails on its first one.
146
+ The requirements previously listed it as supported, which it never was.
147
+
148
+ ### Upgrading
149
+
150
+ Nothing breaks without migrating, but the new features need one. See
151
+ [Upgrading from 1.x](README.md#upgrading-from-1x).
152
+
3
153
  ## 1.5.2 - 2026-08-04
4
154
 
5
155
  * Fix `Poller#valid_queues` raising `NoMethodError` on the Oracle adapter. The
data/Gemfile CHANGED
@@ -11,5 +11,20 @@ gem 'minitest'
11
11
  gem 'mysql2'
12
12
  gem 'pry'
13
13
  gem 'rake'
14
- gem 'rubocop', '~> 1.60'
14
+
15
+ # Pinned to a patch range rather than given a floor: Gemfile.lock is not
16
+ # checked in, so CI resolves the newest version matching this line while a
17
+ # checkout keeps whatever it installed. With a floor, every rubocop release
18
+ # that adds a cop turns the build red for a reason nobody can reproduce
19
+ # locally.
20
+ gem 'rubocop', '~> 1.91.0'
15
21
  gem 'trilogy'
22
+
23
+ # Only needed to run the suite against Oracle, which additionally requires the
24
+ # Oracle Instant Client to be installed - building ruby-oci8 fails without it.
25
+ # The group is optional, so a plain `bundle install` skips it; enable it with
26
+ # `bundle config set --local with oracle`.
27
+ group :oracle, optional: true do
28
+ gem 'activerecord-oracle_enhanced-adapter', '~> 7.1'
29
+ gem 'ruby-oci8'
30
+ end