active_durable 0.5.0 → 0.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. checksums.yaml +4 -4
  2. data/.yardopts +8 -0
  3. data/CHANGELOG.md +114 -3
  4. data/README.md +174 -22
  5. data/app/controllers/active_durable/executions_controller.rb +4 -2
  6. data/app/helpers/active_durable/dashboard_helper.rb +11 -3
  7. data/app/views/active_durable/executions/index.html.erb +4 -3
  8. data/app/views/active_durable/executions/show.html.erb +16 -6
  9. data/app/views/layouts/active_durable/application.html.erb +7 -5
  10. data/lib/active_durable/configuration.rb +18 -0
  11. data/lib/active_durable/engine.rb +2 -4
  12. data/lib/active_durable/errors.rb +80 -5
  13. data/lib/active_durable/execution.rb +16 -2
  14. data/lib/active_durable/flow.rb +170 -22
  15. data/lib/active_durable/flow_parallel.rb +46 -7
  16. data/lib/active_durable/lease.rb +2 -0
  17. data/lib/active_durable/notebook.rb +17 -4
  18. data/lib/active_durable/open_telemetry.rb +14 -2
  19. data/lib/active_durable/operations.rb +54 -15
  20. data/lib/active_durable/parallel.rb +15 -0
  21. data/lib/active_durable/prune_job.rb +16 -0
  22. data/lib/active_durable/pruner.rb +32 -0
  23. data/lib/active_durable/record.rb +2 -0
  24. data/lib/active_durable/registry.rb +4 -0
  25. data/lib/active_durable/retry_policy.rb +2 -0
  26. data/lib/active_durable/run_job.rb +3 -0
  27. data/lib/active_durable/runner.rb +48 -6
  28. data/lib/active_durable/serializer.rb +29 -3
  29. data/lib/active_durable/signal_record.rb +13 -0
  30. data/lib/active_durable/step.rb +20 -0
  31. data/lib/active_durable/sweep_job.rb +3 -0
  32. data/lib/active_durable/sweeper.rb +26 -6
  33. data/lib/active_durable/testing.rb +9 -1
  34. data/lib/active_durable/version.rb +2 -1
  35. data/lib/active_durable.rb +137 -13
  36. data/lib/generators/active_durable/install/install_generator.rb +2 -0
  37. data/lib/generators/active_durable/install/templates/create_active_durable_tables.rb.tt +12 -6
  38. data/lib/generators/active_durable/upgrade/templates/add_active_durable_prune_index.rb.tt +14 -0
  39. data/lib/generators/active_durable/upgrade/templates/make_active_durable_ids_case_sensitive.rb.tt +72 -0
  40. data/lib/generators/active_durable/upgrade/upgrade_generator.rb +39 -0
  41. data/lib/tasks/active_durable.rake +13 -2
  42. metadata +31 -11
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 593d546eb6e8e46a70c03be3945934ed0c8a189937a5a27256478783cd345df0
4
- data.tar.gz: 3226b77ada034c6bf49830dfdb30457904e14104b430b4cfb0a9def3d3de4124
3
+ metadata.gz: ca565450baafb042b5ec363d2d7c101f24edd3281b180d43c173eaeff8dd8df8
4
+ data.tar.gz: 96ceecb07cf0d85b0d8aaf604dececee56301983d16055de408e533c53059da6
5
5
  SHA512:
6
- metadata.gz: f0f0adcff14966f0b8cf7710218a6a6df9b27152593ea4d0d29c90f1968967817b467c33b9803dac0b7aaec6ff105c291a50ecf1335f15882b1969c4d7f7d2bd
7
- data.tar.gz: 1ecba4e399e4f9a247ea9717e9727a50c86d7389aa41a4e53fe3aa9817a100e04a91ecfdc7a2dc51333b24ae9b94a666f1a3c2d5e1987132a48949970a210592
6
+ metadata.gz: a6dafa621dba36aac0429fc264922d876d378390aa7fccebf05a085bbcb1f2e001d7bf57d202075299e47cbbb20d29af86df62000adca22367bec76ab18dd16b
7
+ data.tar.gz: 7b4c90ddb5104811ab17be9532fe2c4ca029bae139eeeaafdef1bd04a222fa4c2896c34b2ce44376d3ff13222473e76276a991c125fcdbe057937d7e6a649896
data/.yardopts ADDED
@@ -0,0 +1,8 @@
1
+ --markup markdown
2
+ --readme README.md
3
+ --title "ActiveDurable"
4
+ --no-private
5
+ --hide-api private
6
+ lib/**/*.rb
7
+ -
8
+ CHANGELOG.md
data/CHANGELOG.md CHANGED
@@ -2,13 +2,110 @@
2
2
 
3
3
  All notable changes to this project are documented here. The format follows
4
4
  [Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and the project uses
5
- [Semantic Versioning](https://semver.org/). Before 1.0 the database schema may change between minor versions:
6
- regenerate the migration when upgrading.
5
+ [Semantic Versioning](https://semver.org/). Schema changes ship as new migrations: after updating the gem, run
6
+ `bin/rails generate active_durable:upgrade` and `bin/rails db:migrate`.
7
7
 
8
- ## [Unreleased]
8
+ ## [0.7.0] - 2026-10-07
9
+
10
+ Fixes from an architecture and security review. Most of them stop a saga from paying twice or undoing the wrong
11
+ thing.
12
+
13
+ Upgrading from 0.6: `bin/rails generate active_durable:upgrade && bin/rails db:migrate`. The new migration only
14
+ changes MySQL databases: it rebuilds the three tables and blocks writes to them while it runs, so run it at a quiet
15
+ time, or prune first. Note the behavior changes below: more errors block instead of undoing, `retry` no longer
16
+ resets every failed step, rerun ids are random, and `Durable.start` refuses some ids.
17
+
18
+ ### Changed
19
+
20
+ - More errors count as bugs and block the saga instead of being retried and undone: `ArgumentError`, `TypeError`,
21
+ `IndexError` and `KeyError`, `FrozenError`, `ZeroDivisionError`, `RangeError`, `NoMatchingPatternError`,
22
+ `LocalJumpError`, `RegexpError` and `EncodingError`, besides `NameError` and `NoMethodError`. The list is the new
23
+ `config.code_errors`, which takes classes or names, so an app can add its own errors or drop one.
24
+ - A step that blocks is written in the notebook as `blocked`, and runs again after `ActiveDurable.retry`. Rescuing
25
+ its error in the recipe no longer lets the saga carry on.
26
+ - `ActiveDurable.retry` gives a fresh set of attempts only to the step that blocked the saga (with its whole
27
+ `flow.parallel`) and to steps that hit a bug. Compensating, it still resets every failed undo.
28
+ - Rerun ids are `<id>~rerun-<random>` instead of `<id>~rerun-<count>`.
29
+ - `ActiveDurable.compensate` refuses `running` executions, even after their lease ran out.
30
+ - `Durable.start` refuses ids that are blank, contain `/`, a NUL character or bytes that are not UTF-8.
31
+ - The serializer converts binary strings that are valid UTF-8 (such as `Net::HTTP` bodies) and strings in other
32
+ encodings to UTF-8, and rejects NUL characters and invalid bytes, in values and keys.
33
+
34
+ ### Fixed
35
+
36
+ - `retry` ran again a failed step the recipe had already handled: with a fallback (Stripe fails, the recipe pays
37
+ with PayPal) and a later bug, retrying charged twice. `rerun` did the same; it now copies the failed steps before
38
+ the chosen one too.
39
+ - A step whose result the database refused (a NUL on PostgreSQL, bytes that are not UTF-8 on MySQL and SQLite, any
40
+ failed write) counted as a failed attempt: it ran again, then the saga was undone without undoing it. It now
41
+ blocks.
42
+ - Undoing a saga by hand skipped a step declared `undo_on_failure` that was between attempts or blocked, and the
43
+ finished branches of a `flow.parallel` stopped halfway.
44
+ - An operator could undo a saga while its worker was still inside a step longer than `lease_duration`.
45
+ - A pending signal nobody was waiting for (a duplicate webhook) made the saga enqueue itself again forever.
46
+ - Signals sent to a blocked saga were refused and lost.
47
+ - `rerun` lost the undos of the `flow.parallel` branches before the chosen step, and from a branch it ran nothing;
48
+ the dashboard no longer offers branches.
49
+ - On MySQL, ids and step names that differ only in case or accents were treated as the same. The new migration
50
+ makes those columns `utf8mb4_bin`.
51
+ - `raise ActiveRecord::Rollback` inside `flow.transaction`, an undo or a hook counted as success.
52
+ - On Rails 6.1 and 7.0 with MySQL, two concurrent `start` calls with the same id inside the app's transaction made
53
+ the second raise `RecordNotFound`.
54
+ - Rerun ids collided after pruning, and could give a new rerun the tickets of a pruned one.
55
+ - An id with `/` broke the whole dashboard list.
56
+ - `LoadError`, `NotImplementedError` and `SystemStackError` left the saga running, retried forever by the sweeper.
57
+ - A bug in an undo used up all its attempts before blocking; it now blocks at once.
58
+ - Error messages are cleaned up (invalid bytes, NUL) before they are stored.
59
+ - README: how to run Solid Queue in development, and the sweeper and cleanup entries go under the `production:` key
60
+ Rails already wrote in `config/recurring.yml`. Pasting a second `production:` key silently dropped Rails' own
61
+ `clear_solid_queue_finished_jobs` task.
62
+
63
+ ### Added
64
+
65
+ - Benchmarks in `benchmarks/`: the cost of a step and of resuming a long saga (`throughput.rb`), and many worker
66
+ processes on the same sagas, optionally killing one with SIGKILL every second (`load.rb`, `CHAOS=1`). The README
67
+ has a "Performance" section with the results on PostgreSQL, MySQL and SQLite.
68
+ - README: what counts as a bug, `config.code_errors`, and new limits: ids are forgotten after pruning, the lease and
69
+ slow steps, clocks, and secrets in error messages.
70
+
71
+ ## [0.6.0] - 2026-10-06
72
+
73
+ Usable in a real app: found by a first-time user's walkthrough, plus hooks, cleanup, upgrades and an API reference.
74
+
75
+ Upgrading from 0.5: `bin/rails generate active_durable:upgrade && bin/rails db:migrate`. Note the behavior change
76
+ below: a plain exception raised by a recipe now blocks the saga instead of undoing it; use `flow.abort!` for business
77
+ rejections.
78
+
79
+ ### Added
80
+
81
+ - `flow.on(:completed) { ... }` and `flow.on(:compensated) { ... }`: hooks to update your own records when a saga
82
+ ends. Declared before the first step, so a saga undone early still knows them. Each runs once, in a transaction
83
+ with the notebook entry that records it (exactly once for database changes, covered by the crash tester); a failing
84
+ hook blocks the execution and `ActiveDurable.retry` runs only the hook. The dashboard lists them in the notebook,
85
+ and OpenTelemetry traces them.
86
+ - Cleanup: `ActiveDurable.prune(older_than:)`, `ActiveDurable::PruneJob` and `bin/rails active_durable:prune` delete
87
+ finished executions (completed, compensated, superseded) older than `config.keep_finished_for` (30 days by
88
+ default), with their notebook and signals, in batches. Active and blocked executions are never deleted.
89
+ - An API reference: the public API has YARD docs (parameters, returns, errors, examples) and the internals are
90
+ marked `@api private`, so rubydoc.info shows only what an app calls. `.yardopts` configures it.
91
+ - `bin/rails generate active_durable:upgrade` adds the migrations a newer version needs, skipping the ones the app
92
+ already has. The first one is an index on `durable_executions (status, updated_at)` for the cleanup and the
93
+ sweeper; new installs get it from `active_durable:install`.
9
94
 
10
95
  ### Changed
11
96
 
97
+ - **A bug no longer undoes a saga.** Only a step that fails for good, or `flow.abort!`, undoes the finished steps.
98
+ Any other error raised by the recipe, and a `NameError` or `NoMethodError` inside a step (now
99
+ `ActiveDurable::CodeError`, not retried), blocks the execution instead: a typo in a deploy used to refund every
100
+ saga that woke up with it. Fix the code and call `ActiveDurable.retry`. A business rejection raised as a plain
101
+ exception in the recipe body now blocks too: use `flow.abort!` (or raise `ActiveDurable::Abort`).
102
+ - The repository moved to [webresstudio/active_durable](https://github.com/webresstudio/active_durable). Links to the
103
+ old address redirect.
104
+ - A website, in English and Spanish, with an interactive simulator of a checkout: pick what goes wrong (a crash, a
105
+ refused parcel, a declined card…) and watch the steps, the notebook and the outside world. Source in `site/`, built
106
+ by `bin/site` and published to GitHub Pages by `.github/workflows/pages.yml`. The gem's homepage and both READMEs
107
+ link to it. Its simulator also shows a bug in a deploy (the saga is blocked, nothing is undone, a retry carries it
108
+ on) and the order's own status, set by `flow.on(:compensated)`.
12
109
  - README: the quick start recipe reads top to bottom, with one-line undos and the Stripe calls in a `Payments`
13
110
  module where charging and refunding sit side by side. A new section, "In a Rails app", sets up a Rails app step by
14
111
  step: install, job backend, sweeper, the initializer with every setting, routes, where each piece of code goes
@@ -16,6 +113,20 @@ regenerate the migration when upgrading.
16
113
  before going to production. The settings moved there from "Observability".
17
114
  - Specs cover undos without arguments (`-> { ... }`), a `Method` as an undo, and `flow.abort!` inside a step, which
18
115
  skips the step's remaining retries.
116
+ - gemspec: Active Job, Active Record and Active Support are required `>= 6.1, < 9`, the versions CI tests, instead
117
+ of any version from 6.1 on. The duplicate homepage link is gone, so `gem build` no longer warns.
118
+
119
+ ### Fixed
120
+
121
+ - The rake tasks were loaded twice (Rails already loads an engine's `lib/tasks`), so each one ran twice.
122
+ - `bin/rails active_durable:sweep` with the `:async` adapter (the Rails default in development) enqueued jobs that died
123
+ with the rake process. It now runs the due executions itself.
124
+ - Dashboard: the execution id no longer widens its column (long ids wrap), and long problems take two lines, with the
125
+ full message on hover.
126
+ - README: `Payments` turns Stripe's errors into its own, so the recipe does not depend on Stripe; development with
127
+ `:async` is explained (sagas started from the console are lost until the sweeper runs); a Minitest example. The
128
+ order page shows the order's own status, written by the saga (`mark_shipped`, `flow.on(:compensated)`), instead of
129
+ the saga's status, which said "Processing…" for days after the parcel left.
19
130
 
20
131
  ## [0.5.0] - 2026-10-06
21
132
 
data/README.md CHANGED
@@ -5,22 +5,25 @@
5
5
  </p>
6
6
 
7
7
  <p align="center">
8
- <a href="https://github.com/williamromero/active_durable/actions/workflows/main.yml"><img src="https://github.com/williamromero/active_durable/actions/workflows/main.yml/badge.svg" alt="CI"></a>
8
+ <a href="https://github.com/webresstudio/active_durable/actions/workflows/main.yml"><img src="https://github.com/webresstudio/active_durable/actions/workflows/main.yml/badge.svg" alt="CI"></a>
9
9
  <img src="https://img.shields.io/badge/ruby-3.1%2B-CC342D?logo=ruby&logoColor=white" alt="Ruby 3.1 and newer">
10
10
  <img src="https://img.shields.io/badge/rails-6.1%2B-D30001?logo=rubyonrails&logoColor=white" alt="Rails 6.1 and newer">
11
11
  <img src="https://img.shields.io/badge/PostgreSQL%20%C2%B7%20MySQL%20%C2%B7%20SQLite-tested-3DD6A0" alt="PostgreSQL, MySQL and SQLite">
12
12
  <img src="https://img.shields.io/badge/no%20Redis-no%20extra%20servers-7EA6FF" alt="No Redis, no extra servers">
13
13
  <a href="LICENSE.txt"><img src="https://img.shields.io/badge/license-MIT-B08CFF" alt="MIT license"></a>
14
+ <a href="https://webresstudio.github.io/active_durable/"><img src="https://img.shields.io/badge/website-try%20the%20simulator-F2B641" alt="Website: try the interactive simulator"></a>
14
15
  </p>
15
16
 
16
17
  <p align="center">
18
+ <a href="https://webresstudio.github.io/active_durable/"><b>Website</b></a> ·
17
19
  <a href="#quick-start">Quick start</a> ·
18
20
  <a href="#in-a-rails-app">In a Rails app</a> ·
19
21
  <a href="#how-it-works">How it works</a> ·
20
22
  <a href="#the-building-blocks">Building blocks</a> ·
21
23
  <a href="#dashboard">Dashboard</a> ·
22
24
  <a href="#testing-the-crash-tester">Crash tester</a> ·
23
- <a href="#compatibility">Compatibility</a>
25
+ <a href="#compatibility">Compatibility</a> ·
26
+ <a href="https://rubydoc.info/gems/active_durable">API reference</a>
24
27
  </p>
25
28
 
26
29
  ---
@@ -61,6 +64,7 @@ Write the recipe once, in `app/sagas/`:
61
64
  # app/sagas/checkout_saga.rb
62
65
  CheckoutSaga = Durable.define(:checkout) do |flow, order_id:|
63
66
  order = Order.find(order_id)
67
+ flow.on(:compensated) { order.update!(status: "cancelled") } # once every finished step was undone
64
68
 
65
69
  # Touches only your database: committed together with its checkpoint, so it runs exactly once.
66
70
  flow.transaction :reserve_stock, undo: -> { order.release_stock! } do
@@ -71,7 +75,7 @@ CheckoutSaga = Durable.define(:checkout) do |flow, order_id:|
71
75
  # Talks to the outside world: the ticket is an idempotency key that never changes for this step.
72
76
  payment = flow.step :charge, undo: ->(charge, ticket) { Payments.refund(charge, ticket) } do |ticket|
73
77
  Payments.charge(order, ticket)
74
- rescue Stripe::CardError => e
78
+ rescue Payments::CardDeclined => e
75
79
  flow.abort!(e.message) # a declined card is not retried: the stock is released right away
76
80
  end
77
81
 
@@ -79,6 +83,7 @@ CheckoutSaga = Durable.define(:checkout) do |flow, order_id:|
79
83
  flow.pivot :dispatch do |ticket|
80
84
  { "tracking" => Carrier.ship(order.id, reference: ticket).tracking_number, "payment" => payment["id"] }
81
85
  end
86
+ flow.transaction(:mark_shipped) { order.update!(status: "shipped") } # your own status, for your pages
82
87
 
83
88
  flow.step(:confirmation_email) { OrderMailer.shipped(order.id).deliver_now && true }
84
89
  flow.sleep(:wait_for_delivery, 3.days) # no worker is held while it sleeps
@@ -91,12 +96,16 @@ The calls to Stripe live in a plain module, doing and undoing side by side. The
91
96
  ```ruby
92
97
  # app/services/payments.rb
93
98
  module Payments
99
+ class CardDeclined < StandardError; end
100
+
94
101
  def self.charge(order, ticket)
95
102
  intent = Stripe::PaymentIntent.create(
96
103
  { amount: order.total_cents, currency: "usd", customer: order.user.stripe_id, confirm: true },
97
104
  { idempotency_key: ticket }
98
105
  )
99
106
  { "id" => intent.id } # written in the notebook, so it must fit in JSON
107
+ rescue Stripe::CardError => e
108
+ raise CardDeclined, e.message # the recipe speaks your language, not Stripe's
100
109
  end
101
110
 
102
111
  def self.refund(charge, ticket)
@@ -131,6 +140,9 @@ Run the three commands from the [quick start](#quick-start). The migration creat
131
140
  exactly once only because the notebook and your data commit together. A queue in its own database, like Solid
132
141
  Queue's in Rails 8, is fine.
133
142
 
143
+ When you update the gem, run `bin/rails generate active_durable:upgrade` and then `bin/rails db:migrate`: it adds only
144
+ the migrations your app is missing, and running it twice changes nothing.
145
+
134
146
  ### 2. A job backend
135
147
 
136
148
  ActiveDurable runs on Active Job, so it uses the backend you already have. New Rails 8 apps come with Solid Queue;
@@ -145,13 +157,26 @@ config.solid_queue.connects_to = { database: { writing: :queue } }
145
157
 
146
158
  - **Workers must listen to the saga queue.** It is `:default` unless you change `config.queue_name`; if you do, add
147
159
  it to your workers (Solid Queue's `config/queue.yml`, Sidekiq's `-q`).
148
- - **In development** Rails' default `:async` adapter runs jobs inside the server, sleeps and retries included. With
149
- Solid Queue, run `bin/jobs` next to the server.
160
+ - **In development** Rails' default `:async` adapter keeps each job in the memory of the process that enqueued it.
161
+ The server runs its own, sleeps and retries included, but a saga started from the console, `bin/rails runner`, a
162
+ rake task or `db/seeds.rb` is lost when that process exits, and scheduled wake-ups are lost when the server
163
+ restarts. A minute later, `bin/rails active_durable:sweep` picks them up: with `:async` it runs them right there.
164
+ Or run Solid Queue in development too, so jobs live in the database: give `development` a `queue` database in
165
+ `config/database.yml` (like the one `production` has, with `migrations_paths: db/queue_migrate`), add the two
166
+ lines below, run `bin/rails db:prepare`, and start `bin/jobs` next to the server.
167
+
168
+ ```ruby
169
+ # config/environments/development.rb
170
+ config.active_job.queue_adapter = :solid_queue
171
+ config.solid_queue.connects_to = { database: { writing: :queue } }
172
+ ```
150
173
 
151
174
  ### 3. The sweeper
152
175
 
153
176
  The safety net: every minute it enqueues executions that lost their job, because the process died between the
154
- commit and the enqueue or a worker died holding a lease. With Solid Queue:
177
+ commit and the enqueue or a worker died holding a lease. With Solid Queue, add these entries under the
178
+ `production:` key Rails already wrote in `config/recurring.yml` (a second `production:` key would silently replace
179
+ it), and under a `development:` key too if you run Solid Queue in development:
155
180
 
156
181
  ```yaml
157
182
  # config/recurring.yml
@@ -159,11 +184,18 @@ production:
159
184
  active_durable_sweep:
160
185
  class: ActiveDurable::SweepJob
161
186
  schedule: every minute
187
+ active_durable_prune:
188
+ class: ActiveDurable::PruneJob
189
+ schedule: every day at 4am
162
190
  ```
163
191
 
164
192
  With another backend, schedule `ActiveDurable::SweepJob` in its own scheduler (GoodJob cron, sidekiq-cron), or run
165
193
  `bin/rails active_durable:sweep` from cron.
166
194
 
195
+ The second entry is the cleanup: finished executions (completed, undone or superseded) are kept for
196
+ `config.keep_finished_for`, 30 days by default, and then `ActiveDurable::PruneJob` deletes them with their notebook.
197
+ Active and blocked executions are never deleted. Without it, every saga stays in the database forever.
198
+
167
199
  ### 4. The initializer
168
200
 
169
201
  The settings, with their defaults:
@@ -179,6 +211,8 @@ ActiveDurable.configure do |config|
179
211
  config.backoff = ->(attempt) { [2**attempt, 3600].min } # or [5, 30, 300], or a number
180
212
  config.parallel_concurrency = 4 # threads per flow.parallel
181
213
  config.sweep_grace = 1.minute # the sweeper leaves executions this young alone
214
+ config.keep_finished_for = 30.days # then ActiveDurable::PruneJob deletes finished ones
215
+ config.code_errors += ["Payments::Misconfigured"] # your own errors that mean a bug (see "A bug is not a failure")
182
216
 
183
217
  # Who may open the dashboard outside development and test. With Devise (HTTP basic auth: see Dashboard):
184
218
  config.dashboard_authorize = ->(controller) { controller.request.env["warden"]&.user&.admin? }
@@ -237,7 +271,7 @@ class OrdersController < ApplicationController
237
271
  end
238
272
  end
239
273
 
240
- # app/models/order.rb
274
+ # app/models/order.rb: orders has a status column ("placed" by default) that the saga updates
241
275
  class Order < ApplicationRecord
242
276
  def checkout
243
277
  Durable.find("checkout-#{id}")
@@ -245,15 +279,16 @@ class Order < ApplicationRecord
245
279
  end
246
280
  ```
247
281
 
248
- The `id:` ties the saga to its order, so the order page can show where the saga is:
282
+ The `id:` ties the saga to its order: `order.checkout` is for your team and the dashboard. The page shows the order's
283
+ own status, which the saga writes as it goes (`mark_shipped`, `flow.on(:compensated)`), not the saga's: a shipped
284
+ order still has a saga sleeping three days before it asks for a review.
249
285
 
250
286
  ```erb
251
287
  <%# app/views/orders/show.html.erb %>
252
- <% case @order.checkout.status %>
253
- <% when "completed" %> Your order is confirmed.
254
- <% when "compensated" %> We could not complete it and refunded you.
255
- <% when "blocked" %> We are looking into it.
256
- <% else %> Processing…
288
+ <% case @order.status %>
289
+ <% when "shipped" %> Your order is on its way.
290
+ <% when "cancelled" %> We could not complete it and refunded you.
291
+ <% else %> Processing…
257
292
  <% end %>
258
293
  ```
259
294
 
@@ -275,6 +310,7 @@ the notebook, not to the job: if your backend retries the job too, the copy find
275
310
  | stop retrying a business failure | `flow.abort!`, like the declined card above |
276
311
  | see what is running | the [dashboard](#dashboard), or `ActiveDurable::RunJob` in your backend's UI |
277
312
  | hear about a stuck saga | the `blocked.active_durable` [event](#observability) |
313
+ | delete finished sagas | schedule `ActiveDurable::PruneJob` once a day (step 3) |
278
314
  | run a saga inline in tests | `ActiveDurable::Testing.drain(id)` |
279
315
  | start or wake sagas from your own jobs | call `Durable.start` or `Durable.signal` there |
280
316
 
@@ -301,6 +337,26 @@ it "charges once and confirms the order" do
301
337
  end
302
338
  ```
303
339
 
340
+ With Minitest, the Rails default:
341
+
342
+ ```ruby
343
+ # test/test_helper.rb
344
+ require "active_durable/testing"
345
+
346
+ class ActiveSupport::TestCase
347
+ setup { ActiveDurable::Testing.reset! }
348
+ end
349
+
350
+ # test/services/place_order_test.rb
351
+ class PlaceOrderTest < ActiveSupport::TestCase
352
+ test "charges once and confirms the order" do
353
+ order = PlaceOrder.call(email: "ana@example.com", total_cents: 4200)
354
+
355
+ assert_equal "completed", ActiveDurable::Testing.drain(order.checkout.id).status
356
+ end
357
+ end
358
+ ```
359
+
304
360
  `drain` runs the saga right there, without a worker. Stub Stripe as you already do, then let the
305
361
  [crash tester](#testing-the-crash-tester) kill the saga at every point.
306
362
 
@@ -308,6 +364,7 @@ end
308
364
 
309
365
  - [ ] Workers are running and listen to `config.queue_name`.
310
366
  - [ ] The sweeper runs every minute.
367
+ - [ ] The cleanup runs every day, or you keep every execution on purpose.
311
368
  - [ ] `dashboard_authorize` is set; without it the dashboard answers 403.
312
369
  - [ ] `lease_duration` is longer than your slowest step.
313
370
  - [ ] Every step that calls an outside service passes the ticket as its idempotency key.
@@ -334,8 +391,8 @@ stateDiagram-v2
334
391
  sleeping --> running: wake-up time
335
392
  waiting --> running: Durable.signal
336
393
  running --> completed: every step done
337
- running --> compensated: failure before the pivot, undos ran
338
- running --> blocked: needs a person
394
+ running --> compensated: a step failed for good, or flow.abort!, before the pivot
395
+ running --> blocked: a bug, a failure after the pivot or a failing hook
339
396
  blocked --> pending: ActiveDurable.retry
340
397
  ```
341
398
 
@@ -357,6 +414,7 @@ Three rules follow from replaying the recipe:
357
414
  | `flow.parallel(name) { \|branches\| ... }` | several steps at the same time | each branch like a step |
358
415
  | `flow.sleep(name, 3.days)` | waiting without holding a worker | — |
359
416
  | `flow.wait_for(name, timeout:)` | waiting for `Durable.signal` | a timeout fails the saga |
417
+ | `flow.on(:completed) { ... }` | updating your own records when the saga ends (also `:compensated`) | blocks; a retry runs the hook again |
360
418
 
361
419
  Steps take `undo:`, `retry:` (`3`, `false` or `{ attempts:, backoff: }`) and `undo_on_failure:`.
362
420
 
@@ -377,11 +435,38 @@ idempotency key, Stripe answers with the first result instead of charging twice.
377
435
 
378
436
  <br>
379
437
 
380
- When a step runs out of attempts, raises `ActiveDurable::Abort`, or the recipe raises, every finished step is undone,
381
- last one first. Each undo is written in the notebook too, so a crash in the middle of undoing resumes where it
438
+ When a step runs out of attempts or calls `flow.abort!`, every finished step is undone, last one first. Each undo is written in the notebook too, so a crash in the middle of undoing resumes where it
382
439
  stopped. An undo receives `(result, undo_ticket, step_ticket)` and takes as many as it declares: `-> { ... }` takes
383
440
  none. Any object that responds to `call` works too, such as `Payments.method(:refund)`.
384
441
 
442
+ **A bug is not a failure.** Undoing a saga refunds money and releases stock, so only a step that failed for good or
443
+ `flow.abort!` triggers it. These block the execution instead, without retrying, because a typo in a deploy must never
444
+ refund your customers:
445
+
446
+ - anything the recipe raises outside a step;
447
+ - inside a step, an error that means the code is wrong: `NameError` (and `NoMethodError`), `ArgumentError`,
448
+ `TypeError`, `IndexError` (and `KeyError`), `FrozenError`, `ZeroDivisionError`, `RangeError`,
449
+ `NoMatchingPatternError`, `LocalJumpError`, `RegexpError`, `EncodingError`, and the errors outside
450
+ `StandardError` such as `LoadError`, `NotImplementedError` and `SystemStackError`;
451
+ - a step whose result cannot be stored (a NUL character, bytes that are not UTF-8): it already acted.
452
+
453
+ Fix the code and call `ActiveDurable.retry(id)`, or press Retry in the dashboard, and the saga carries on from where it
454
+ stopped. Rescuing the error in the recipe does not help: the saga stops at the next step and blocks anyway. To reject
455
+ the work for a business reason, call `flow.abort!(reason)`.
456
+
457
+ The list is `config.code_errors`. Add your own errors, as a class or as a name (a name also matches subclasses, and
458
+ works before the gem that defines it is loaded), or drop one:
459
+
460
+ ```ruby
461
+ ActiveDurable.configure do |config|
462
+ config.code_errors << "Payments::Misconfigured"
463
+ config.code_errors -= ["ArgumentError"] # if a gem you call raises it for bad input
464
+ end
465
+ ```
466
+
467
+ An `ArgumentError` from bad data, like an invalid date, blocks too: a person decides. To treat one of these errors as
468
+ a failure in a single step, rescue it inside the step and raise another error, or call `flow.abort!`.
469
+
385
470
  A step that failed is not undone, because it did not happen. The exception is a step whose failure may hide a
386
471
  success, like a charge whose answer timed out: declare `undo_on_failure: true` and its undo runs with `nil` as the
387
472
  result, so it can look the outcome up with the step's ticket.
@@ -409,13 +494,39 @@ everything. After it, steps cannot declare `undo:` and are retried with backoff
409
494
 
410
495
  </details>
411
496
 
497
+ <details>
498
+ <summary><b>When a saga ends: hooks</b></summary>
499
+
500
+ <br>
501
+
502
+ `flow.on(:completed)` and `flow.on(:compensated)` run once the saga ends that way, to update your own records:
503
+
504
+ ```ruby
505
+ CheckoutSaga = Durable.define(:checkout) do |flow, order_id:|
506
+ order = Order.find(order_id)
507
+ flow.on(:completed) { order.update!(status: "delivered") }
508
+ flow.on(:compensated) { order.update!(status: "cancelled") }
509
+
510
+ flow.transaction(:reserve_stock, undo: -> { order.release_stock! }) { ... }
511
+ end
512
+ ```
513
+
514
+ Declare them before the first step: a saga undone at its first step never reaches the lines after it. `completed`
515
+ runs after the last step and `compensated` after the last undo, each in a transaction together with the notebook
516
+ entry that records it, so a hook that only touches your database runs exactly once, even if the process dies. If a
517
+ hook raises, the execution is blocked, and `ActiveDurable.retry` runs the hook again, not the steps.
518
+
519
+ For progress before the end (paid, shipped), write a step: `flow.transaction(:mark_shipped) { ... }`.
520
+
521
+ </details>
522
+
412
523
  <details>
413
524
  <summary><b>Sleeping and waiting for signals</b></summary>
414
525
 
415
526
  <br>
416
527
 
417
528
  `flow.sleep` writes the wake-up time down and releases the worker. `flow.wait_for` does the same until a signal
418
- arrives. Signals can arrive before the saga starts waiting.
529
+ arrives. Signals can arrive before the saga starts waiting, or while it is blocked: they are kept until it does.
419
530
 
420
531
  ```ruby
421
532
  kyc = flow.wait_for(:kyc_done, timeout: 2.hours)
@@ -508,13 +619,21 @@ end
508
619
  ## Operating sagas
509
620
 
510
621
  ```ruby
511
- ActiveDurable.retry("checkout-7") # blocked: try again where it stopped
622
+ ActiveDurable.retry("checkout-7") # blocked: try again the step that blocked it
512
623
  ActiveDurable.compensate("checkout-7", reason: "customer cancelled") # undo everything (only before the pivot)
513
624
  ActiveDurable.rerun("checkout-7", from: :ship) # a new execution reusing steps before :ship
625
+ ActiveDurable.prune(older_than: 30.days) # delete finished executions and their notebook
514
626
  ```
515
627
 
516
- All three refuse an execution a worker is running right now. A rerun runs the chosen step and the following ones
517
- again with new tickets, so they have effects again; a blocked original becomes `superseded`.
628
+ `retry` gives the step that blocked the saga a fresh set of attempts. Failed steps the recipe already handled stay
629
+ failed: if Stripe failed and the recipe paid with PayPal instead, a retry never runs Stripe again.
630
+
631
+ A rerun (`<id>~rerun-<random>`) keeps the steps before the chosen one, failed ones and parallel branches included,
632
+ and runs the chosen step and the following ones again with new tickets, so they have effects again. A blocked
633
+ original becomes `superseded`.
634
+
635
+ None of them touches an execution a worker is running. `compensate` refuses a `running` execution even after its
636
+ lease ran out: the worker may still be inside a slow step, and if it died, the sweeper resumes it.
518
637
 
519
638
  ## Testing: the crash tester
520
639
 
@@ -554,7 +673,7 @@ end
554
673
 
555
674
  | Event | Payload |
556
675
  | --- | --- |
557
- | `execution` / `step` / `compensation` / `undo` | `execution_id` (and `recipe`, `step`, `kind`) |
676
+ | `execution` / `step` / `compensation` / `undo` / `hook` | `execution_id` (and `recipe`, `step`, `kind`) |
558
677
  | `completed` / `compensated` | `execution_id`, `recipe` |
559
678
  | `blocked` | `execution_id`, `recipe`, `error` |
560
679
  | `retried` / `compensation_requested` / `rerun` | operator actions |
@@ -601,12 +720,45 @@ Every feature works on every version. Things your app may need on older Rails, u
601
720
  | Temporal | the Temporal server | written by hand | a Temporal cluster |
602
721
  | **ActiveDurable** | **your database** | **yes, in reverse, with a point of no return** | **nothing extra** |
603
722
 
723
+ ## Performance
724
+
725
+ Measured on an Apple M1 Ultra with Ruby 4.0.7 and Rails 8.1, each database on the same machine. The scripts are in
726
+ [`benchmarks/`](benchmarks/README.md), so you can run them on yours.
727
+
728
+ | | PostgreSQL 16 | MySQL 9.6 | SQLite 3 |
729
+ | --- | --- | --- | --- |
730
+ | A step (`flow.step`), one process | 1.8 ms | 3.0 ms | 0.45 ms |
731
+ | A step (`flow.transaction`), one process | 2.0 ms | 2.8 ms | 0.45 ms |
732
+ | Resuming a saga with 1,000 finished steps | 17 ms | 21 ms | 9 ms |
733
+ | 2,000 sagas of 5 steps, 8 worker processes | 6.9 s (290 sagas/s) | 8.1 s (246 sagas/s) | 1,000 sagas, 4 workers: 6.5 s |
734
+ | The same, killing a worker with SIGKILL every second | 9.2 s, 9 workers killed | 13.2 s, 13 killed | — |
735
+ | Duplicate charges | 0 | 0 | 0 |
736
+
737
+ A step costs about four queries: the lease check that fences the write, the notebook row and the commit that makes
738
+ it durable. In the load tests each call to the outside world takes 5 ms, so a saga spends most of its time waiting,
739
+ as in a real app. When a worker dies mid-step, another one runs that step again with the same ticket: on PostgreSQL 2
740
+ of the 2,000 charges were sent twice, and the idempotency key made them one.
741
+
742
+ The same test in a Rails 8 app with Solid Queue: 300 checkouts, every Solid Queue process killed with SIGKILL twice
743
+ while sagas were halfway through. All of them settled, the refunds and hooks ran once, and no order was charged twice.
744
+
604
745
  ## Guarantees and limits
605
746
 
606
747
  - A step runs **at least once**; with an idempotency key it has its effect once. `flow.transaction` runs exactly
607
748
  once, because its change and its checkpoint commit together.
608
749
  - One worker at a time per execution: taking an execution and every write are fenced by a lease token.
750
+ - A hook (`flow.on`) runs once; exactly once if it only touches your database.
751
+ - A bug never undoes a saga: an error in the code blocks it until you fix it and call `ActiveDurable.retry`.
609
752
  - Sagas do not isolate each other: two sagas can see each other's intermediate states.
753
+ - `Durable.start(id:)` is idempotent while the execution exists. Once it is pruned (`keep_finished_for`, 30 days),
754
+ the same id starts a new saga with the same tickets: keep finished executions longer than a webhook can be
755
+ delivered again.
756
+ - The lease is renewed on each notebook write. A step longer than `lease_duration` can run twice at the same time,
757
+ with the same ticket. Keep the workers' clocks in sync (NTP): the lease compares times written by different
758
+ machines.
759
+ - Error messages are stored in the notebook and reach the dashboard, the logs, the `blocked` event and
760
+ OpenTelemetry: keep secrets and personal data out of them.
761
+ - Ids cannot be blank or contain `/`. Ids and step names compare exactly, also on MySQL.
610
762
  - `flow.transaction` is atomic only when the notebook lives in the same database as your data.
611
763
 
612
764
  ## Development
@@ -23,15 +23,17 @@ module ActiveDurable
23
23
  @execution = Execution.find(params[:id])
24
24
  steps = @execution.steps.to_a
25
25
  @steps = steps
26
- @forward_steps = steps.reject(&:undo?).sort_by { |step| step.position.to_i }
26
+ @forward_steps = steps.select(&:forward?).sort_by { |step| step.position.to_i }
27
+ @rerun_steps = @forward_steps.select(&:position) # a parallel branch cannot be a starting point
27
28
  @undo_steps = steps.select(&:undo?)
29
+ @hook_steps = steps.select(&:hook?)
28
30
  @signals = @execution.signals.to_a
29
31
  @reruns = Execution.where(forked_from: @execution.id).order(:created_at).to_a
30
32
  end
31
33
 
32
34
  def retry_now
33
35
  ActiveDurable.retry(params[:id])
34
- redirect_to execution_path(params[:id]), notice: "Retrying. Failed steps got a fresh set of attempts."
36
+ redirect_to execution_path(params[:id]), notice: "Retrying. The step that blocked it got a fresh set of attempts."
35
37
  end
36
38
 
37
39
  def compensate
@@ -57,6 +57,13 @@ module ActiveDurable
57
57
  safe_join(name.to_s.split(/(?<=_)/), tag.wbr)
58
58
  end
59
59
 
60
+ # Ids with a slash cannot be routed; versions before 0.7 accepted them, so they are listed without a link.
61
+ def execution_link(id, **options)
62
+ return tag.span(id, title: "Ids with a slash have no page; use the console", **options) if id.include?("/")
63
+
64
+ link_to(id, execution_path(id), **options)
65
+ end
66
+
60
67
  def ticket_for(execution, step)
61
68
  "#{execution.id}:#{step.name}"
62
69
  end
@@ -75,7 +82,7 @@ module ActiveDurable
75
82
  end
76
83
 
77
84
  def can_compensate?(execution, steps)
78
- %w[blocked pending running sleeping waiting].include?(execution.status) && !execution.compensating &&
85
+ %w[blocked pending sleeping waiting].include?(execution.status) && !execution.compensating &&
79
86
  steps.none? { |step| step.kind == "pivot" && step.completed? }
80
87
  end
81
88
 
@@ -87,7 +94,7 @@ module ActiveDurable
87
94
  # their parallel step, undone steps are marked, and an active execution gets a "next" ghost block.
88
95
  def track_items(execution, steps)
89
96
  undone = steps.select { |step| step.undo? && step.completed? }.to_set { |step| step.name.delete_suffix(":undo") }
90
- branches, main = steps.reject(&:undo?).partition { |step| step.position.nil? }
97
+ branches, main = steps.select(&:forward?).partition { |step| step.position.nil? }
91
98
 
92
99
  items = main.sort_by(&:position).map do |step|
93
100
  build_item(step, undone, branches.select { |branch| branch.name.start_with?("#{step.name}/") })
@@ -109,7 +116,8 @@ module ActiveDurable
109
116
 
110
117
  def state_word(state)
111
118
  {
112
- "completed" => "done", "failed" => "failed", "retrying" => "retrying", "waiting" => "waiting for a signal",
119
+ "completed" => "done", "failed" => "failed", "blocked" => "blocked", "retrying" => "retrying",
120
+ "waiting" => "waiting for a signal",
113
121
  "sleeping" => "sleeping", "undone" => "undone", "next" => "up next", "running" => "running"
114
122
  }.fetch(state.to_s, state.to_s)
115
123
  end
@@ -42,8 +42,8 @@
42
42
  <% @executions.each do |execution| %>
43
43
  <% items = track_items(execution, @steps_by_execution.fetch(execution.id, [])) %>
44
44
  <tr data-id="<%= execution.id %>" data-status="<%= execution.status %>">
45
- <td>
46
- <%= link_to execution.id, execution_path(execution), class: "exec-link" %>
45
+ <td class="exec">
46
+ <%= execution_link(execution.id, class: "exec-link") %>
47
47
  <span class="recipe"><%= execution.recipe %>, version <%= execution.recipe_version %></span>
48
48
  </td>
49
49
  <td>
@@ -69,7 +69,8 @@
69
69
  <% if compensating_now?(execution) %><%= status_pill("compensating") %><% end %>
70
70
  </td>
71
71
  <td class="when"><%= execution.wake_at ? when_text(execution.wake_at) : tag.span("—", class: "muted") %></td>
72
- <td class="problem"><%= execution.error&.dig("message").to_s.truncate(110) %></td>
72
+ <% problem = execution.error&.dig("message").to_s %>
73
+ <td class="problem"><span title="<%= problem %>"><%= problem.truncate(110) %></span></td>
73
74
  <td class="when"><%= when_text(execution.updated_at) %></td>
74
75
  </tr>
75
76
  <% end %>