geneva_drive 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (121) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +38 -0
  3. data/MANUAL.md +1037 -355
  4. data/README.md +3 -1
  5. data/Rakefile +13 -2
  6. data/bin/md2html +96 -19
  7. data/lib/generators/geneva_drive/install/install_generator.rb +25 -0
  8. data/lib/generators/geneva_drive/install/templates/add_metadata_to_step_executions.rb +17 -0
  9. data/lib/generators/geneva_drive/install/templates/add_metadata_to_workflows.rb +17 -0
  10. data/lib/generators/geneva_drive/install/templates/add_resumable_step_support.rb +29 -0
  11. data/lib/generators/geneva_drive/install/templates/add_started_at_index_to_step_executions.rb +48 -0
  12. data/lib/generators/geneva_drive/install/templates/allow_null_hero_on_workflows.rb +66 -0
  13. data/lib/generators/geneva_drive/install/templates/create_step_executions_migration.rb +25 -5
  14. data/lib/generators/geneva_drive/install/templates/create_workflows_migration.rb +14 -3
  15. data/lib/generators/geneva_drive/install/templates/initializer.rb.tt +22 -2
  16. data/lib/geneva_drive/combined_exception_policy.rb +106 -0
  17. data/lib/geneva_drive/exception_policy.rb +354 -0
  18. data/lib/geneva_drive/executor.rb +430 -148
  19. data/lib/geneva_drive/flow_control.rb +82 -11
  20. data/lib/geneva_drive/iterable_step.rb +199 -0
  21. data/lib/geneva_drive/job_options.rb +70 -0
  22. data/lib/geneva_drive/jobs/housekeeping_job.rb +174 -52
  23. data/lib/geneva_drive/jobs/perform_step_job.rb +91 -3
  24. data/lib/geneva_drive/migration_helpers.rb +55 -20
  25. data/lib/geneva_drive/resumable_step_definition.rb +97 -0
  26. data/lib/geneva_drive/step_definition.rb +160 -40
  27. data/lib/geneva_drive/step_execution/metadata_accessor.rb +121 -0
  28. data/lib/geneva_drive/step_execution.rb +124 -2
  29. data/lib/geneva_drive/test_helpers.rb +123 -4
  30. data/lib/geneva_drive/version.rb +1 -1
  31. data/lib/geneva_drive/workflow/metadata_accessor.rb +85 -0
  32. data/lib/geneva_drive/workflow.rb +485 -73
  33. data/lib/geneva_drive.rb +100 -2
  34. data/test/dsl/class_level_on_exception_test.rb +200 -0
  35. data/test/dsl/exception_policy_test.rb +331 -0
  36. data/test/dsl/step_definition_test.rb +126 -2
  37. data/test/dsl/step_level_exception_policy_test.rb +143 -0
  38. data/test/dummy/config/storage.yml +7 -0
  39. data/test/dummy_install/config/storage.yml +7 -0
  40. data/test/jobs/housekeeping_job_test.rb +451 -5
  41. data/test/jobs/perform_step_job_double_deferral_test.rb +268 -0
  42. data/test/jobs/perform_step_job_test.rb +25 -2
  43. data/test/migration_helpers_test.rb +137 -0
  44. data/test/step_execution/metadata_accessor_test.rb +147 -0
  45. data/test/test_helper.rb +37 -0
  46. data/test/test_helper_test.rb +281 -0
  47. data/test/workflow/class_level_on_exception_integration_test.rb +313 -0
  48. data/test/workflow/composable_exception_policy_test.rb +231 -0
  49. data/test/workflow/cursor_size_limit_test.rb +75 -0
  50. data/test/workflow/executor_test.rb +100 -4
  51. data/test/workflow/external_cancel_test.rb +245 -0
  52. data/test/workflow/flow_control_test.rb +4 -3
  53. data/test/workflow/instrumentation_test.rb +28 -0
  54. data/test/workflow/max_reattempts_test.rb +51 -0
  55. data/test/workflow/pause_resume_test.rb +574 -0
  56. data/test/workflow/reattempt_on_exception_integration_test.rb +68 -0
  57. data/test/workflow/resumable_step_integration_test.rb +341 -0
  58. data/test/workflow/resumable_step_test.rb +615 -0
  59. data/test/workflow/resumable_without_migration_test.rb +103 -0
  60. data/test/workflow/resume_and_skip_test.rb +57 -41
  61. data/test/workflow/with_inline_enqueue_test.rb +137 -0
  62. data/test/workflow/workflow_test.rb +117 -2
  63. metadata +47 -61
  64. data/test/dummy/db/migrate/20241217000001_create_geneva_drive_workflows.rb +0 -68
  65. data/test/dummy/db/migrate/20241217000002_create_geneva_drive_step_executions.rb +0 -90
  66. data/test/dummy/db/migrate/20241217000003_add_finished_at_to_geneva_drive_step_executions.rb +0 -25
  67. data/test/dummy/db/migrate/20241217000004_add_error_class_name_to_geneva_drive_step_executions.rb +0 -7
  68. data/test/dummy/db/schema.rb +0 -72
  69. data/test/dummy/log/development.log +0 -112
  70. data/test/dummy/log/test.log +0 -4
  71. data/test/dummy/tmp/local_secret.txt +0 -1
  72. data/test/dummy_install/Rakefile +0 -6
  73. data/test/dummy_install/app/assets/stylesheets/application.css +0 -15
  74. data/test/dummy_install/app/controllers/application_controller.rb +0 -4
  75. data/test/dummy_install/app/helpers/application_helper.rb +0 -2
  76. data/test/dummy_install/app/jobs/application_job.rb +0 -7
  77. data/test/dummy_install/app/models/application_record.rb +0 -3
  78. data/test/dummy_install/app/models/user.rb +0 -9
  79. data/test/dummy_install/app/views/layouts/application.html.erb +0 -27
  80. data/test/dummy_install/app/views/layouts/mailer.html.erb +0 -13
  81. data/test/dummy_install/app/views/layouts/mailer.text.erb +0 -1
  82. data/test/dummy_install/app/views/pwa/manifest.json.erb +0 -22
  83. data/test/dummy_install/app/views/pwa/service-worker.js +0 -26
  84. data/test/dummy_install/bin/dev +0 -2
  85. data/test/dummy_install/bin/rails +0 -4
  86. data/test/dummy_install/bin/rake +0 -4
  87. data/test/dummy_install/bin/setup +0 -34
  88. data/test/dummy_install/config/application.rb +0 -29
  89. data/test/dummy_install/config/boot.rb +0 -5
  90. data/test/dummy_install/config/cable.yml +0 -10
  91. data/test/dummy_install/config/database.yml +0 -13
  92. data/test/dummy_install/config/environment.rb +0 -5
  93. data/test/dummy_install/config/environments/development.rb +0 -51
  94. data/test/dummy_install/config/environments/production.rb +0 -70
  95. data/test/dummy_install/config/environments/test.rb +0 -39
  96. data/test/dummy_install/config/initializers/assets.rb +0 -7
  97. data/test/dummy_install/config/initializers/content_security_policy.rb +0 -25
  98. data/test/dummy_install/config/initializers/filter_parameter_logging.rb +0 -8
  99. data/test/dummy_install/config/initializers/geneva_drive.rb +0 -76
  100. data/test/dummy_install/config/initializers/inflections.rb +0 -16
  101. data/test/dummy_install/config/locales/en.yml +0 -31
  102. data/test/dummy_install/config/puma.rb +0 -38
  103. data/test/dummy_install/config/routes.rb +0 -3
  104. data/test/dummy_install/config.ru +0 -6
  105. data/test/dummy_install/db/migrate/20241217000000_create_users.rb +0 -12
  106. data/test/dummy_install/db/migrate/20260128104738_create_geneva_drive_workflows.rb +0 -74
  107. data/test/dummy_install/db/migrate/20260128104739_create_geneva_drive_step_executions.rb +0 -97
  108. data/test/dummy_install/db/migrate/20260128104740_add_finished_at_to_geneva_drive_step_executions.rb +0 -25
  109. data/test/dummy_install/db/migrate/20260128104741_add_error_class_name_to_geneva_drive_step_executions.rb +0 -7
  110. data/test/dummy_install/db/migrate/20260128104742_add_resumable_step_support_to_geneva_drive_step_executions.rb +0 -25
  111. data/test/dummy_install/db/schema.rb +0 -76
  112. data/test/dummy_install/log/test.log +0 -4031
  113. data/test/dummy_install/public/400.html +0 -114
  114. data/test/dummy_install/public/404.html +0 -114
  115. data/test/dummy_install/public/406-unsupported-browser.html +0 -114
  116. data/test/dummy_install/public/422.html +0 -114
  117. data/test/dummy_install/public/500.html +0 -114
  118. data/test/dummy_install/public/icon.png +0 -0
  119. data/test/dummy_install/public/icon.svg +0 -3
  120. data/test/dummy_install/tmp/local_secret.txt +0 -1
  121. data/test/generators/install_generator_test.rb +0 -94
data/MANUAL.md CHANGED
@@ -1,6 +1,8 @@
1
- # GenevaDrive Manual
1
+ # Introduction
2
2
 
3
- GenevaDrive provides durable, multi-step workflows for Rails applications. We built it because the prevailing "durable function" approach — where you write an `async` function and expect the runtime to persist it mid-execution — is fundamentally broken. The graph of steps and the code that executes each step must live in separate universes. GenevaDrive delivers on that architecture.
3
+ GenevaDrive provides durable, multi-step workflows for Rails applications. The graph of steps and the code that executes each step live in separate universes — the structure is defined at class load time, the behavior runs later via ActiveJob. GenevaDrive delivers on that architecture.
4
+
5
+ # Part I — Why GenevaDrive Exists
4
6
 
5
7
  ## The Problem with Long-Running Processes
6
8
 
@@ -12,115 +14,11 @@ Imagine you need to onboard a new user. You send a welcome email, wait three day
12
14
  - **Idempotency is hard to get right.** If a job runs twice due to a queue hiccup, you might send two welcome emails or charge a credit card twice.
13
15
  - **ActiveJob does not provide a stable ID.** Retried jobs can create duplicate job IDs even though the jobs are distinct. You cannot use the job ID as an idempotency key.
14
16
 
15
- ## The problem with "durable functions"
16
-
17
- There is a rather popular approach to building durable execution systems based on the concept of "durable functions". Systems like [Temporal](https://temporal.io) and [absurd](https://lucumr.pocoo.org/2025/11/3/absurd-workflows/) as well as [Vercel Workflows](https://vercel.com/docs/workflow) take that concept quite far. It seems neat on the surface, yet deeply flawed in nature.
18
-
19
- The assumption made with those "durable functions" is that it is possible to _pretend that you have a marshalable stack._ For example, this section in a workflow function:
20
-
21
- ```js
22
- let step = 0;
23
- while (step++ < 20) {
24
- const { newMessages, finishReason } = await ctx.step("iteration", async () => {
25
- return await singleStep(messages);
26
- });
27
- messages.push(...newMessages);
28
- if (finishReason !== "tool-calls") {
29
- break;
30
- }
31
- }
32
- ```
33
-
34
- can only work if `step` gets marshaled and reinstated if the function gets resumed. Async generators (and Fibers in Ruby, and - in general - any systems based on continuations or coroutines) allow suspension and resumption, but none allow proper _serialization and revival._ If you try to encode a durable function, consisting of multiple steps, as a suspendable and resumable workflow, you essentially have 3 ways to do it:
35
-
36
- * Make your function restartable from the very beginning (idempotent)
37
- * Use a serializable system stack frame, which - usually - comes down to serializing a VM image upon suspension
38
- * Make the user write functions that only - and ever - use special facilities for accessing transients (current time, database connections, heavy resources)
39
-
40
- Most "workflow engines" do their utmost to maintain the guise of resumable functions _while not providing them._ The fact that you have to wait on a Fiber to receive an HTTP result is not very useful if the only program that can receive that result is the very process which has started that HTTP request.
41
-
42
- ### Why no modern runtime provides stack serialization
43
-
44
- The fundamental issue is that no mainstream runtime — V8, SpiderMonkey, YARV, the JVM, the CLR — provides the ability to serialize an executing call stack to bytes and revive it later. This is not an oversight. It is a deliberate engineering decision driven by hard constraints.
45
-
46
- A call stack contains pointers: return addresses, references to heap objects, handles to file descriptors and sockets, pointers into native libraries. Serializing a pointer is meaningless — the memory address 0x7fff5fbff8c0 on one machine means nothing on another, or even on the same machine after a restart. To serialize a stack, you must either:
47
-
48
- 1. **Replace all pointers with symbolic references** that can be resolved at revival time. This requires a complete indirection layer over every memory access — a performance catastrophe for general-purpose code.
49
- 2. **Serialize the entire heap along with the stack**, effectively snapshotting the whole process. This is what Smalltalk images did.
50
- 3. **Restrict the language** so that stacks never contain non-serializable values. This means no closures over native resources, no FFI, no direct system calls.
51
-
52
- Modern runtimes chose speed over serializability. JavaScript engines like V8 perform aggressive JIT compilation that inlines functions, eliminates stack frames, and stores values in machine registers. The "stack" you think exists in your `async` function is a fiction maintained for debugging — the actual execution state is scattered across registers, hidden classes, inline caches, and optimized machine code that has no stable representation.
53
-
54
- ### Systems that actually solved this
55
-
56
- True stack serialization is not impossible. It has been done — just not in environments optimized for raw speed.
57
-
58
- **Smalltalk images** are the canonical example. A Smalltalk system serializes its entire object memory, including all activation records (stack frames), to a single file. You can save an image mid-computation, quit, restart days later, and continue exactly where you left off. This works because Smalltalk controls everything: the object format, the bytecode interpreter, the garbage collector. There are no opaque pointers to external resources — or if there are, the image-saving mechanism explicitly handles them.
59
-
60
- **Erlang/OTP** takes a different approach: processes are so lightweight and isolated that you simply design for crash recovery. A process dies, its supervisor restarts it, and it reconstructs its state from durable storage. There's no pretense that you can freeze and thaw a running computation — you design for restart from the beginning.
61
-
62
- **Scheme continuations** (particularly in implementations like Chez Scheme or Gambit) can capture delimited continuations and, in some implementations, serialize them. But these implementations pay the cost: they maintain a CPS-transformed representation that is inherently slower than direct-style execution.
63
-
64
- **[Seaside](https://seaside.st/)** deserves special mention. This Smalltalk web framework, developed in the early 2000s, used continuations to model web application control flow. You could write a multi-page wizard as a single method with `call:` and `answer:` — the framework would suspend execution while waiting for user input and resume it when the response arrived. [Wee](https://github.com/mneumann/wee), a Ruby port inspired by Seaside's ideas, attempted the same trick using Ruby's `callcc`. Both frameworks demonstrated genuine continuation-based web development. But they also demonstrated its limits: Seaside required either keeping all session continuations in memory (scaling poorly) or relying on Smalltalk's image persistence (requiring the same VM instance to handle subsequent requests). Wee suffered from Ruby 1.8's notorious continuation memory leaks and remained a curiosity rather than a production tool.
65
-
66
- The fundamental problem with continuation-based web frameworks is process affinity. A suspended continuation exists in the memory of a specific process on a specific machine. When the user submits the next form, that exact process must handle the request — no load balancer can route it elsewhere, no autoscaler can spin up a fresh instance to handle the load, no deployment can replace the running code. This is incompatible with modern elastic infrastructure. Kubernetes doesn't care that your user's shopping cart continuation lives in pod `web-7f8d9c-xk2p4` — when traffic spikes, it will route requests wherever capacity exists. When you deploy, it will terminate old pods and start new ones. Your continuations die with them.
67
-
68
- ### The problem of transient resources
69
-
70
- Even if you could serialize a continuation, you would face the problem of transient resources. A continuation captures the call stack, but the call stack contains references to objects that cannot meaningfully survive process boundaries.
71
-
72
- Consider a database transaction. Your step opens a connection, begins a transaction, inserts a row, and then — mid-transaction — suspends to wait for user confirmation:
73
-
74
- ```ruby
75
- step :reserve_inventory do
76
- ActiveRecord::Base.transaction do
77
- hero.line_items.each { |item| Inventory.decrement!(item.sku, item.quantity) }
78
- # Suspend here, wait for payment confirmation...
79
- yield # In a hypothetical continuation-based system
80
- hero.update!(reserved_at: Time.current)
81
- end
82
- end
83
- ```
84
-
85
- What happens when you try to revive this continuation on a different machine, or even the same machine after a restart? The `ActiveRecord::Base.connection` object holds a socket to a PostgreSQL server. That socket is gone. You could theoretically reconnect — some systems use lazy connection resolution for exactly this reason — but reconnecting gives you a *new* connection. The transaction you started? It was rolled back the moment the original connection died. The `BEGIN` you issued exists only in the logs. The row locks you held have been released. Some other process may have already modified the rows you thought you had locked.
86
-
87
- There is no way to "re-enter" a transaction. Transactions are not addressable resources you can resume — they are ephemeral states of a connection that exist only as long as that connection lives. The same applies to file handles, HTTP connections mid-request, mutex locks, and any other resource that represents a relationship with an external system. A serialized continuation that references such resources is not a suspended computation — it is a lie about the state of the world.
88
-
89
- The lesson from these systems is clear: serializable execution state requires either total control over the runtime environment or acceptance of significant performance overhead. You cannot bolt it onto V8 or Ruby's YARV after the fact.
90
-
91
- ### Why modern developers won't build this
92
-
93
- Building a marshalable stack VM is a multi-year, multi-million-dollar undertaking. It requires:
94
-
95
- - Deep expertise in compiler construction, garbage collection, and runtime systems
96
- - Willingness to sacrifice raw performance for serializability
97
- - Long-term maintenance commitment as the underlying platform evolves
98
-
99
- The JavaScript ecosystem optimizes for different goals: startup time, peak throughput, memory efficiency on mobile devices. Google, Mozilla, and Apple compete on V8, SpiderMonkey, and JavaScriptCore benchmarks. No one is competing on "ability to serialize a running function to disk."
100
-
101
- The developers building "durable function" frameworks are, by and large, application developers — skilled in their domain, but not runtime engineers. They are building atop V8, not replacing it. They cannot make V8 serialize its internal state because V8 was never designed to expose that state. The best they can do is replay: run the function again from the start, skip the steps that already completed, and hope the interleaving of side effects is deterministic. This is not serialization. It is simulation.
102
-
103
- ### Pretending is denial
104
-
105
- When a framework claims to offer "durable functions" without true stack serialization, it is engaging in denial — not about the laws of physics, but about the semantics of their own system.
106
-
107
- Consider what happens when a "durable function" resumes. The framework:
108
-
109
- 1. Loads the function definition
110
- 2. Starts executing from the beginning
111
- 3. Intercepts calls to "step" functions and returns cached results instead of re-executing
112
-
113
- This only works if the control flow between steps is perfectly deterministic. But JavaScript (and Ruby, and Python) are not deterministic languages. The order of object keys, the behavior of `Math.random()`, the resolution of race conditions in `Promise.all()` — all of these can vary between runs. If your loop counter depends on a hash iteration order that changed between Node versions, your "resumed" function takes a different path than the original.
114
-
115
- The frameworks paper over this with restrictions: don't use randomness, don't depend on time, don't read from external systems except through blessed APIs. But these restrictions are invisible until you violate them. You write what looks like normal code, you test it, it works — and then six months later, after a Node upgrade, a resumed workflow takes a wrong turn and corrupts your data.
116
-
117
- Honesty requires admitting what the system actually provides. If execution state is not truly serialized, don't pretend it is. Make the boundaries explicit. Make the steps explicit. Make the user acknowledge, at every step boundary, that they are persisting state to a database. GenevaDrive takes this position.
118
-
119
- ### The DAG alternative
17
+ ## The DAG Approach
120
18
 
121
- Instead, it is useful to look at workflow systems in terms of _DAGs_ (directed acyclic graphs). In that vision, the code driving the DAG and the code driving the nodes in that DAG is strictly separate, and runs in different domains. While the DAG definition code is "single use" and runs during the orchestration, the actual code of the nodes _is_ required to be explicitly restartable, and is required to be explicitly idempotent. There is no pretense that "this function being wrapped in this callback will cleanly resume", and there is way less possibility to "accidentally" call something non-idempotent when defining the nodes.
19
+ Instead of pretending that a function can be suspended, serialized, and resumed — which no mainstream runtime actually supports — GenevaDrive models workflows as DAGs (directed acyclic graphs). The code driving the DAG and the code driving each node are strictly separate and run in different domains.
122
20
 
123
- A geneva_drive workflow is a DAG with a single permitted input and a single permitted output per node, nothing more. It explicitly ditches the illusion of a marshalable VM universe in favor of clarity, cohesion with the host environment (UNIX system running Ruby running Rails) and deliberately picks clarity over pretense of magic.
21
+ A GenevaDrive workflow is a DAG with a single permitted input and a single permitted output per node. It explicitly ditches the illusion of a marshalable VM universe in favor of clarity, cohesion with the host environment (UNIX system running Ruby running Rails) and deliberately picks clarity over pretense of magic.
124
22
 
125
23
  ```ruby
126
24
  class OrderFulfillmentWorkflow < GenevaDrive::Workflow # DAG context — definition time
@@ -158,6 +56,38 @@ GenevaDrive addresses these challenges with a small set of guarantees:
158
56
 
159
57
  ---
160
58
 
59
+ # Part II — Getting Started
60
+
61
+ ## Installation and Setup
62
+
63
+ ### Adding the Gem
64
+
65
+ ```bash
66
+ bundle add geneva_drive
67
+ bin/rails generate geneva_drive:install
68
+ bin/rails db:migrate
69
+ ```
70
+
71
+ The generator creates the migration required for geneva_drive to operate - you need to run all of them. When updating the gem, rerun the generator to add any migrations you may need to run as migrations get added as the gem evolves.
72
+
73
+ ### Database Tables
74
+
75
+ GenevaDrive uses a two-table design:
76
+
77
+ - **`geneva_drive_workflows`** — The workflow records. Each row represents one workflow instance with its current state, hero association, and progress tracking.
78
+ - **`geneva_drive_step_executions`** — The idempotency keys. Each row represents one attempt to execute a step, with timing, outcome, and error information.
79
+
80
+ This separation keeps the workflows table clean while maintaining a complete audit trail in step executions.
81
+
82
+ ### UUID Primary Keys
83
+
84
+ If your application uses UUID primary keys, the migrations will detect this and also use UUIDs for the foreign keys and the primary keys of the geneva_drive resources.
85
+
86
+ Note that you don't want to mix integer IDs and UUIDs in the same application. geneva_drive uses a polymorphic relation for the `hero` of the workflow, which has a typed `hero_id` column. You will thus want either all of your models to be used as heroes to have UUID primary keys or bigint primary keys, but not mix the two.
87
+
88
+ > [!WARNING]
89
+ > If you are using UUIDs, we also strongly recommend adopting a lexicographically ordered UUID flavour, at least for the geneva_drive tables. Such flavours include [UUIDv7](https://github.com/seouri/rails-uuid-pk) and [tou](https://github.com/cheddar-me/tou) - you will want your primary keys to be pre-sorted for using the admin effectively, as well as for efficient `find_each` usage with the geneva_drive tables.
90
+
161
91
  ## Core Concepts
162
92
 
163
93
  ### Workflows
@@ -249,6 +179,8 @@ end
249
179
 
250
180
  ---
251
181
 
182
+ # Part III — Defining Workflows
183
+
252
184
  ## Defining Steps
253
185
 
254
186
  ### Named Steps
@@ -342,22 +274,30 @@ The loop bodies execute once when Ruby loads the class. Each iteration adds a ne
342
274
 
343
275
  ### Instance Methods as Steps
344
276
 
345
- For complex steps, you can define instance methods and reference them with `step def`:
277
+ The step definition can be a block, but if you give the step the same name as an instance method of your workflow that method will be called instead:
346
278
 
347
279
  ```ruby
348
280
  class DataExportWorkflow < GenevaDrive::Workflow
349
- step def gather_records
350
- hero.update!(export_data: hero.exportable_records.to_json)
281
+ step :anonymize_records
282
+
283
+ def anonymize_records
284
+ Anonymizer.process_arel(hero.records)
351
285
  end
286
+ end
287
+ ```
288
+
289
+ Since `def` in modern Rubies returns the name of the method you define as a `Symbol` you can use the `step def` shorthand as well. Together with "endless methods" this can give you a very compact description:
290
+
291
+ ```ruby
292
+ class DataExportWorkflow < GenevaDrive::Workflow
293
+ step def gather_records = hero.update!(export_data: hero.exportable_records.to_json)
352
294
 
353
295
  step def write_to_storage
354
296
  Storage.write(hero.export_path, hero.export_data)
355
297
  hero.update!(exported_at: Time.current)
356
298
  end
357
299
 
358
- step def notify_user
359
- ExportMailer.complete(hero).deliver_later
360
- end
300
+ step def notify_user = ExportMailer.complete(hero).deliver_later
361
301
  end
362
302
  ```
363
303
 
@@ -385,7 +325,58 @@ end
385
325
 
386
326
  The referenced step must already be defined — you can only insert before or after steps that appear earlier in the class body.
387
327
 
388
- ---
328
+ ## Conditional Execution
329
+
330
+ ### Skipping Steps with `skip_if:`
331
+
332
+ You can declare conditions that cause a step to be skipped without entering the step body:
333
+
334
+ ```ruby
335
+ class NotificationWorkflow < GenevaDrive::Workflow
336
+ step :send_email, skip_if: -> { hero.email_unsubscribed? } do
337
+ NotificationMailer.notify(hero).deliver_later
338
+ end
339
+
340
+ step :send_sms, skip_if: :sms_disabled? do
341
+ SmsService.notify(hero)
342
+ end
343
+
344
+ private
345
+
346
+ def sms_disabled?
347
+ !hero.sms_enabled? || hero.phone.blank?
348
+ end
349
+ end
350
+ ```
351
+
352
+ The `skip_if` option accepts a lambda, a symbol (method name), or a boolean. The condition is evaluated before the step executes.
353
+
354
+ For a complete example showing conditional steps in context, see the [User Onboarding Workflow](#user-onboarding-workflow) in the appendix.
355
+
356
+ ### Blanket Cancellation with `cancel_if`
357
+
358
+ When certain conditions should cancel the entire workflow regardless of which step is running, use `cancel_if`:
359
+
360
+ ```ruby
361
+ class EngagementWorkflow < GenevaDrive::Workflow
362
+ cancel_if { hero.deactivated? }
363
+ cancel_if { hero.unsubscribed? }
364
+
365
+ step :send_week_1_email do
366
+ EngagementMailer.week_1(hero).deliver_later
367
+ end
368
+
369
+ step :send_week_2_email, wait: 7.days do
370
+ EngagementMailer.week_2(hero).deliver_later
371
+ end
372
+
373
+ step :send_week_4_email, wait: 14.days do
374
+ EngagementMailer.week_4(hero).deliver_later
375
+ end
376
+ end
377
+ ```
378
+
379
+ GenevaDrive evaluates `cancel_if` conditions before every step. If any condition returns true, the workflow cancels immediately.
389
380
 
390
381
  ## Flow Control
391
382
 
@@ -451,9 +442,48 @@ When a workflow is paused, it stays paused until you explicitly resume it:
451
442
 
452
443
  ```ruby
453
444
  workflow = FraudReviewWorkflow.find(id)
454
- workflow.resume! # Schedules the next step
445
+ workflow.resume! # Re-enqueues the scheduled step
446
+ ```
447
+
448
+ ### How Pause and Resume Work
449
+
450
+ When you call `pause!` externally on a workflow that's waiting for a scheduled step, GenevaDrive preserves the scheduled step execution rather than canceling it. This provides better timeline visibility — you can see that a step was scheduled, became overdue during the pause period, and when it eventually ran.
451
+
452
+ **Pause behavior:**
453
+ - The workflow transitions from `ready` to `paused`
454
+ - The scheduled step execution remains in `scheduled` state
455
+ - If the scheduled time passes while paused, the execution becomes "overdue" (visible in the timeline)
456
+
457
+ **Resume behavior:**
458
+ - The workflow transitions from `paused` back to `ready`
459
+ - If a scheduled execution exists:
460
+ - If still in the future → a job is enqueued with remaining wait time
461
+ - If overdue (scheduled time has passed) → a job is enqueued to run immediately
462
+ - If no scheduled execution exists (executor canceled it while paused) → a new execution is created
463
+
464
+ ```ruby
465
+ # Timeline example:
466
+ # T+0h: step_one completes, step_two scheduled for T+2h
467
+ # T+1h: pause! called (step_two still scheduled for T+2h)
468
+ # T+3h: resume! called (step_two is overdue, runs immediately)
469
+
470
+ workflow = WaitingWorkflow.create!(hero: user)
471
+ perform_next_step(workflow) # step_one runs, step_two scheduled
472
+
473
+ # Later...
474
+ workflow.pause! # step_two stays scheduled
475
+ workflow.step_executions.last.state # => "scheduled"
476
+
477
+ # Much later (after scheduled time passed)...
478
+ workflow.resume! # step_two re-enqueued to run now
455
479
  ```
456
480
 
481
+ This behavior means:
482
+ - Multiple pause/resume cycles reuse the same step execution (no duplicates)
483
+ - The scheduled_for timestamp shows when the step was originally intended to run
484
+ - Overdue steps are clearly visible — their scheduled_for is in the past
485
+ - The executor guards against duplicate execution, so multiple jobs for the same step are safe
486
+
457
487
  ### Reattempting Steps
458
488
 
459
489
  Use `reattempt!` to retry the current step, optionally after a delay:
@@ -516,134 +546,138 @@ class OrderFulfillmentWorkflow < GenevaDrive::Workflow
516
546
  end
517
547
  ```
518
548
 
519
- ---
520
-
521
- ## Workflow States
522
-
523
- ### State Machine Diagram
524
-
525
- ```mermaid
526
- stateDiagram-v2
527
- [*] --> ready: create
528
- ready --> performing: step starts
529
- performing --> ready: step completes / reattempt
530
- performing --> finished: last step completes / finished!
531
- performing --> canceled: cancel!
532
- performing --> paused: pause! / exception
533
- paused --> ready: resume!
534
- ```
535
-
536
- A workflow begins in `ready` state. When a step starts executing, it transitions to `performing`. Upon step completion, it returns to `ready` (unless it was the last step, in which case it transitions to `finished`).
537
-
538
- ### Step Execution States
539
-
540
- Step executions have their own state machine:
541
-
542
- | State | Meaning |
543
- |-------|---------|
544
- | `scheduled` | Waiting to run |
545
- | `in_progress` | Currently executing |
546
- | `completed` | Finished successfully |
547
- | `failed` | Exception occurred |
548
- | `canceled` | Canceled before execution |
549
- | `skipped` | Skipped via `skip_if` or `skip!` |
550
-
551
- ---
549
+ ## Resumable Steps
552
550
 
553
- ## Conditional Execution
551
+ A regular step must finish within a single job execution. When a step has to churn through a large collection — sending a campaign to 200 000 subscribers, syncing a paginated API, backfilling a table — that single execution becomes a liability: a deploy, a worker restart, or a queue timeout loses all progress. Resumable steps solve this with **cursor-based iteration**: the step periodically checkpoints its position into the database, and can be interrupted and continued in a later job execution from exactly where it left off.
554
552
 
555
- ### Skipping Steps with `skip_if:`
553
+ Resumable steps store their state in two extra columns on `geneva_drive_step_executions` (`cursor` and `continues_from_id`), added by the installer migrations — re-run `bin/rails generate geneva_drive:install` on an existing installation to pick them up. Until the migration runs, everything else keeps working: regular steps, pause/resume and housekeeping are unaffected, and executing an actual `resumable_step` fails with a configuration error pointing at the missing migration.
556
554
 
557
- You can declare conditions that cause a step to be skipped without entering the step body:
555
+ Define one with `resumable_step`. The block receives an `IterableStep` object (API-compatible with Rails 8.1's `ActiveJob::Continuation::Step`):
558
556
 
559
557
  ```ruby
560
- class NotificationWorkflow < GenevaDrive::Workflow
561
- step :send_email, skip_if: -> { hero.email_unsubscribed? } do
562
- NotificationMailer.notify(hero).deliver_later
558
+ class CampaignWorkflow < GenevaDrive::Workflow
559
+ step :prepare do
560
+ hero.update!(status: "sending")
563
561
  end
564
562
 
565
- step :send_sms, skip_if: :sms_disabled? do
566
- SmsService.notify(hero)
563
+ resumable_step :send_notifications do |iter|
564
+ iter.iterate_over_records(hero.subscribers) do |subscriber|
565
+ CampaignMailer.notify(hero, subscriber).deliver_later
566
+ end
567
567
  end
568
568
 
569
- private
570
-
571
- def sms_disabled?
572
- !hero.sms_enabled? || hero.phone.blank?
569
+ step :finalize do
570
+ hero.update!(status: "sent")
573
571
  end
574
572
  end
575
573
  ```
576
574
 
577
- The `skip_if` option accepts a lambda, a symbol (method name), or a boolean. The condition is evaluated before the step executes.
578
-
579
- For a complete example showing conditional steps in context, see the [User Onboarding Workflow](#user-onboarding-workflow) in the appendix.
575
+ ### The Cursor
580
576
 
581
- ### Blanket Cancellation with `cancel_if`
577
+ The cursor is a value persisted on the step execution after every checkpoint. It is whatever your iteration needs to pick up where it stopped: a record ID, a page number, an opaque API token, a date. Cursors are serialized with ActiveJob serializers, so anything ActiveJob can serialize works — including `Date` and `Time` — without manual conversion.
582
578
 
583
- When certain conditions should cancel the entire workflow regardless of which step is running, use `cancel_if`:
579
+ The cursor is a position marker, not a place to store the data being processed: it is rewritten on every checkpoint and copied to every successor execution. To keep that write path cheap, the serialized JSON is limited to 128 KB by default — exceeding it raises `GenevaDrive::CursorTooLargeError`. The limit is configurable in the initializer via `GenevaDrive.max_cursor_size` (`nil` disables the check).
584
580
 
585
581
  ```ruby
586
- class EngagementWorkflow < GenevaDrive::Workflow
587
- cancel_if { hero.deactivated? }
588
- cancel_if { hero.unsubscribed? }
589
-
590
- step :send_week_1_email do
591
- EngagementMailer.week_1(hero).deliver_later
592
- end
593
-
594
- step :send_week_2_email, wait: 7.days do
595
- EngagementMailer.week_2(hero).deliver_later
596
- end
597
-
598
- step :send_week_4_email, wait: 14.days do
599
- EngagementMailer.week_4(hero).deliver_later
582
+ resumable_step :process_records do |iter|
583
+ hero.records.where("id > ?", iter.cursor || 0).find_each do |record|
584
+ process(record)
585
+ iter.set!(record.id) # persist cursor, check for interruption
600
586
  end
601
587
  end
602
588
  ```
603
589
 
604
- GenevaDrive evaluates `cancel_if` conditions before every step. If any condition returns true, the workflow cancels immediately.
590
+ The `IterableStep` API:
605
591
 
606
- ---
592
+ | Method | Effect |
593
+ |--------|--------|
594
+ | `iter.cursor` | Current cursor value (`nil` on first run) |
595
+ | `iter.set!(value)` | Set the cursor, persist it, check for interruption |
596
+ | `iter.advance!` | Increment an integer cursor by 1 (integers only) |
597
+ | `iter.checkpoint!` | Persist cursor and check for interruption |
598
+ | `iter.resumed?` | `true` when continuing from a previous execution |
599
+ | `iter.skip_to!(value, wait: nil)` | Set the cursor and suspend immediately; `wait:` delays the continuation |
607
600
 
608
- ## The Asynchronous Execution Model
601
+ Helpers for common iteration shapes:
609
602
 
610
- ### Key Assumptions
603
+ ```ruby
604
+ # ActiveRecord relation, one record at a time (find_each under the hood)
605
+ iter.iterate_over_records(hero.subscribers) { |subscriber| ... }
611
606
 
612
- GenevaDrive steps execute asynchronously via ActiveJob. This has important implications:
607
+ # ActiveRecord relation in batches, yielding relations for bulk operations
608
+ iter.iterate_over_subrelations(hero.subscribers, batch_size: 500) do |batch|
609
+ batch.update_all(notified_at: Time.current)
610
+ end
613
611
 
614
- - **Every step runs on a different machine/process/thread.** Don't assume anything about the execution environment between steps.
615
- - **Instance variables don't persist between steps.** Each step gets a freshly-loaded workflow instance.
616
- - **The workflow is always loaded fresh from the database.** Any changes you make to the workflow or hero are persisted and reloaded.
617
- - **Steps may be separated by seconds or months.** A `wait: 30.days` step means exactly what it says.
612
+ # Stable in-memory arrays, index used as cursor
613
+ iter.iterate_over(items) { |item| ... }
614
+ ```
618
615
 
619
- ### No Shared State Between Steps
616
+ ### Chained Executions
620
617
 
621
- > [!WARNING]
622
- > Instance variables do not persist between steps. Store data in the database.
618
+ When a resumable step is interrupted, the current step execution **completes** (with outcome `continued`) and a successor execution is created, linked to its predecessor via `continues_from_id` and carrying the cursor forward. There is no special "suspended" state — each execution is a normal record with a clear start and end, so the full history of a long iteration is visible as a chain:
623
619
 
624
620
  ```ruby
625
- # WRONG - @data won't exist in next step
626
- step :fetch do
627
- @data = ExternalApi.fetch(hero.external_id)
628
- end
621
+ workflow.step_executions.where(step_name: "send_notifications").order(:created_at)
622
+ # => chunk 1 (completed/continued), chunk 2 (completed/continued), ..., chunk N (completed/success)
623
+ ```
629
624
 
630
- step :process do
631
- process(@data) # @data is nil here!
632
- end
625
+ A step is interrupted when any of these happen:
633
626
 
634
- # RIGHT - persist to the hero or another record
635
- step :fetch do
636
- hero.update!(external_data: ExternalApi.fetch(hero.external_id))
637
- end
627
+ - `max_iterations:` is reached (`resumable_step :import, max_iterations: 10_000`)
628
+ - `max_runtime:` is exceeded (`resumable_step :import, max_runtime: 5.minutes`)
629
+ - The job queue signals shutdown (e.g. Sidekiq stopping)
630
+ - The workflow is paused or canceled externally
631
+ - The step calls `skip_to!` or `suspend!` explicitly
638
632
 
639
- step :process do
640
- process(hero.external_data)
633
+ ```ruby
634
+ # Suspend explicitly, e.g. to respect a rate limit
635
+ resumable_step :sync_api do |iter|
636
+ page = iter.cursor || 1
637
+ loop do
638
+ response = ExternalApi.fetch(page: page)
639
+ suspend!(wait: response.retry_after) if response.rate_limited?
640
+ break if response.empty?
641
+ response.items.each { |item| process(item) }
642
+ page += 1
643
+ iter.set!(page)
644
+ end
641
645
  end
642
646
  ```
643
647
 
644
- You will almost never have the same `self` between steps. Treat each step as an independent unit that reads from and writes to the database.
648
+ ### Flow Control and Errors in Resumable Steps
645
649
 
646
- ---
650
+ All flow control works inside resumable steps, with cursor-aware semantics:
651
+
652
+ - `pause!` completes the current execution keeping the cursor; `resume!` continues from it.
653
+ - `reattempt!` continues from the cursor by default; `reattempt!(rewind: true)` clears the cursor and starts the iteration over.
654
+ - `cancel!`, `skip!` and `finished!` behave as in regular steps.
655
+
656
+ Exception policies (`on_exception:` on the step, class-level `on_exception`, `max_reattempts:`, `terminal_action:`, `report:`) apply exactly as for regular steps. A `:reattempt!` policy continues from the last checkpoint, so a transient failure halfway through a large collection does not redo the completed portion. When an unhandled exception pauses the workflow, `resume!` also retries the failed step from its last checkpoint.
657
+
658
+ Housekeeping recovery is cursor-aware too: a resumable execution stuck `in_progress` (dead worker) is recovered by continuing from its persisted cursor, not by restarting the iteration.
659
+
660
+ ### Writing Restart-Safe Iterations
661
+
662
+ The cursor marks the last *checkpointed* position, and one item may be re-processed if execution stops between doing the work and checkpointing. Make each iteration idempotent (e.g. guard with a uniqueness constraint or a state flag on the processed record) rather than assuming exactly-once delivery.
663
+
664
+ ### Testing Resumable Steps
665
+
666
+ `speedrun_workflow` and `speedrun_current_step` run resumable steps to completion with interruption checks disabled, following the execution chain across explicit suspensions. To exercise partial progress, use `run_iterations`:
667
+
668
+ ```ruby
669
+ test "keeps its place across interruptions" do
670
+ workflow = CampaignWorkflow.create!(hero: campaign)
671
+ perform_next_step(workflow) # :prepare
672
+
673
+ run_iterations(workflow, count: 3) # three iterations, then interrupt
674
+ assert_cursor(workflow, 3)
675
+ assert_step_has_successor(workflow, :send_notifications)
676
+
677
+ speedrun_workflow(workflow) # run the rest
678
+ assert workflow.finished?
679
+ end
680
+ ```
647
681
 
648
682
  ## Exception Handling
649
683
 
@@ -657,9 +691,9 @@ step :risky_operation do
657
691
  end
658
692
  ```
659
693
 
660
- ### Configuring Exception Policy
694
+ ### Step-Level Exception Policy
661
695
 
662
- You can change how a step handles exceptions using `on_exception`:
696
+ Use `on_exception:` on a step to control what happens when that step raises:
663
697
 
664
698
  ```ruby
665
699
  class ResilientApiWorkflow < GenevaDrive::Workflow
@@ -669,20 +703,66 @@ class ResilientApiWorkflow < GenevaDrive::Workflow
669
703
  end
670
704
  ```
671
705
 
672
- Available exception handlers:
706
+ Available actions:
673
707
 
674
708
  - `:pause!` — (default) Pause the workflow for manual review
675
709
  - `:cancel!` — Cancel the workflow
676
- - `:reattempt!` — Retry the step
710
+ - `:reattempt!` — Retry the step (creates a new step execution)
677
711
  - `:skip!` — Skip the step and continue to the next
678
712
 
713
+ ### Class-Level Exception Policy
714
+
715
+ Declare `on_exception` at the class level to set a default for all steps. Steps without an explicit `on_exception:` override inherit this policy:
716
+
717
+ ```ruby
718
+ class ResilientWorkflow < GenevaDrive::Workflow
719
+ on_exception :reattempt!, max_reattempts: 5
720
+
721
+ step :fetch_data do
722
+ ExternalApi.fetch(hero) # Inherits reattempt! policy
723
+ end
724
+
725
+ step :send_email, on_exception: :skip! do
726
+ Mailer.deliver(hero) # Step-level overrides class-level
727
+ end
728
+ end
729
+ ```
730
+
731
+ You can declare multiple class-level policies to handle different exception types:
732
+
733
+ ```ruby
734
+ class OAuthWorkflow < GenevaDrive::Workflow
735
+ # Specific: only matches OAuth2::Error and its subclasses
736
+ on_exception OAuth2::Error, action: :reattempt!, wait: 15.seconds
737
+
738
+ # Specific: cancel on permanent client errors
739
+ on_exception Google::Apis::ClientError, action: :cancel!
740
+
741
+ # Blanket: everything else gets reattempted
742
+ on_exception :reattempt!, max_reattempts: 3
743
+
744
+ step :sync do
745
+ GoogleCalendar.sync(hero)
746
+ end
747
+ end
748
+ ```
749
+
750
+ #### Precedence Rules
751
+
752
+ When a step raises an exception, GenevaDrive resolves the policy in this order:
753
+
754
+ 1. **Step-level single policy** — if the step has an explicit `on_exception:` with a single policy (symbol, `ExceptionPolicy`, or Proc), use it unconditionally
755
+ 2. **Step-level composable array** — if the step has an array of policies, walk them (specific match first, blanket fallback). If no policy in the array matches, continue to step 3
756
+ 3. **Class-level specific** — walk class-level policies (most recently defined first); use the first one whose exception class filter matches
757
+ 4. **Class-level blanket** — use the most recently defined policy with no exception class filter
758
+ 5. **Default** — pause the workflow
759
+
679
760
  ### Limiting Reattempts
680
761
 
681
- When using `on_exception: :reattempt!`, you can limit the number of consecutive reattempts before the workflow pauses. This prevents infinite retry loops when an error is persistent rather than transient.
762
+ When using `:reattempt!`, limit consecutive reattempts with `max_reattempts:`. This prevents infinite retry loops when an error is persistent:
682
763
 
683
764
  ```ruby
684
765
  class ExternalApiWorkflow < GenevaDrive::Workflow
685
- # Will pause after 5 consecutive failures
686
766
  step :sync_to_crm, on_exception: :reattempt!, max_reattempts: 5 do
687
767
  CrmApi.sync(hero)
688
768
  end
@@ -691,30 +771,244 @@ end
691
771
 
692
772
  The `max_reattempts:` option:
693
773
 
694
- - **Defaults to 100** when `on_exception: :reattempt!` is used — a safety net against infinite loops
695
- - **Set to `nil`** to disable the limit and allow unlimited reattempts
774
+ - **Defaults to 100** when `:reattempt!` is used — a safety net against infinite loops
775
+ - **Set to `nil`** to disable the limit entirely
696
776
  - **Only counts consecutive reattempts** — if the step succeeds, the count resets
697
777
  - **Does not affect manual `reattempt!` calls** — only automatic exception handling respects this limit
698
778
 
699
- When the limit is exceeded, GenevaDrive logs a warning and pauses the workflow, storing the original exception for debugging:
779
+ When the limit is exceeded, GenevaDrive logs a warning and pauses the workflow by default. Use `terminal_action:` to change this behavior:
700
780
 
701
781
  ```ruby
702
- # Unlimited reattempts (explicit opt-out of the safety limit)
703
- step :polling_step, on_exception: :reattempt!, max_reattempts: nil do
704
- check_external_status!
782
+ # Cancel the workflow instead of pausing when reattempts are exhausted
783
+ step :flaky_api, on_exception: :reattempt!, max_reattempts: 10, terminal_action: :cancel! do
784
+ FlakyService.call(hero)
705
785
  end
786
+ ```
706
787
 
707
- # Custom limit
708
- step :flaky_api, on_exception: :reattempt!, max_reattempts: 10 do
709
- FlakyService.call(hero)
788
+ `terminal_action:` accepts `:pause!` (default), `:cancel!`, or `:skip!`.
789
+
790
+ ### Reusable Exception Policies
791
+
792
+ For policies shared across multiple workflows, create an `ExceptionPolicy` object:
793
+
794
+ ```ruby
795
+ # Store in a constant or module for reuse
796
+ TRANSIENT_RETRY = GenevaDrive::ExceptionPolicy.new(
797
+ :reattempt!,
798
+ wait: 30.seconds,
799
+ max_reattempts: 5,
800
+ terminal_action: :cancel!
801
+ )
802
+
803
+ class OrderWorkflow < GenevaDrive::Workflow
804
+ step :charge_card, on_exception: TRANSIENT_RETRY do
805
+ PaymentGateway.charge(hero)
806
+ end
807
+ end
808
+
809
+ class ShippingWorkflow < GenevaDrive::Workflow
810
+ step :book_courier, on_exception: TRANSIENT_RETRY do
811
+ CourierApi.book(hero)
812
+ end
813
+ end
814
+ ```
815
+
816
+ Use the `matching:` keyword to create policies that target specific exception classes. This is especially useful when composing multiple policies into an array (see [Composable Step-Level Policies](#composable-step-level-policies)):
817
+
818
+ ```ruby
819
+ # Targets a specific exception class
820
+ TIMEOUT_RETRY = GenevaDrive::ExceptionPolicy.new(
821
+ :reattempt!,
822
+ matching: Net::OpenTimeout,
823
+ wait: 10.seconds,
824
+ max_reattempts: 5
825
+ )
826
+
827
+ # Targets multiple exception classes
828
+ OAUTH_CANCEL = GenevaDrive::ExceptionPolicy.new(
829
+ :cancel!,
830
+ matching: [OAuth2::Error, "Faraday::ConnectionFailed"]
831
+ )
832
+ ```
833
+
834
+ `matching:` accepts a class, a string (resolved lazily via `safe_constantize` — the class doesn't need to be loaded at definition time), or an array of either. A policy without `matching:` is a blanket policy that matches any exception.
835
+
836
+ You can also pass an `ExceptionPolicy` to the class-level `on_exception`. Exception class filters are set on the policy after creation:
837
+
838
+ ```ruby
839
+ class ApiWorkflow < GenevaDrive::Workflow
840
+ on_exception Timeout::Error, action: :reattempt!, wait: 10.seconds, max_reattempts: 3
841
+ on_exception :cancel! # Everything else cancels
842
+
843
+ step :call_api do
844
+ SlowApi.call(hero)
845
+ end
710
846
  end
711
847
  ```
712
848
 
713
- This option only makes sense with `on_exception: :reattempt!` — specifying `max_reattempts:` with other exception handlers will raise a `StepConfigurationError`.
849
+ ### Composable Step-Level Policies
850
+
851
+ Sometimes a single action isn't enough — you want transient errors reattempted, fatal errors to cancel, and everything else to skip. Pass an array of `ExceptionPolicy` objects to `on_exception:` to route different exception types to different actions on a single step:
852
+
853
+ ```ruby
854
+ class CalendarSyncWorkflow < GenevaDrive::Workflow
855
+ step :sync_events, on_exception: [
856
+ GenevaDrive::ExceptionPolicy.new(:reattempt!, matching: Net::OpenTimeout, wait: 10.seconds, max_reattempts: 5),
857
+ GenevaDrive::ExceptionPolicy.new(:cancel!, matching: OAuth2::Error),
858
+ GenevaDrive::ExceptionPolicy.new(:skip!) # blanket fallback for anything else
859
+ ] do
860
+ GoogleCalendar.sync(hero)
861
+ end
862
+
863
+ step :send_confirmation do
864
+ CalendarMailer.synced(hero).deliver_later
865
+ end
866
+ end
867
+ ```
868
+
869
+ Each policy in the array can use the `matching:` keyword to target specific exception classes. When an exception is raised, GenevaDrive walks the array in two passes:
870
+
871
+ 1. **Specific policies** (those with `matching:`) are checked first, in definition order. The first match wins.
872
+ 2. **Blanket policy** (the first policy without `matching:`) acts as a catchall fallback.
873
+ 3. **Fall through** — if no policy in the array matches, resolution continues at the class level (and ultimately falls back to `:pause!`).
874
+
875
+ This means you can combine a step-level array with a class-level policy for layered exception handling:
876
+
877
+ ```ruby
878
+ class RobustApiWorkflow < GenevaDrive::Workflow
879
+ on_exception :pause! # class-level fallback
880
+
881
+ step :call_api, on_exception: [
882
+ GenevaDrive::ExceptionPolicy.new(:reattempt!, matching: Timeout::Error, max_reattempts: 3)
883
+ # No blanket fallback — unmatched exceptions fall through to class-level :pause!
884
+ ] do
885
+ SlowApi.call(hero)
886
+ end
887
+ end
888
+ ```
889
+
890
+ #### Global Reattempt Cap
891
+
892
+ When multiple policies in the array have `max_reattempts:`, GenevaDrive enforces the *minimum* across all of them as a global cap on consecutive reattempts. This prevents runaway retries when different exception types alternate:
893
+
894
+ ```ruby
895
+ step :sync, on_exception: [
896
+ GenevaDrive::ExceptionPolicy.new(:reattempt!, matching: Timeout::Error, max_reattempts: 10),
897
+ GenevaDrive::ExceptionPolicy.new(:reattempt!, matching: RateLimitError, max_reattempts: 3, terminal_action: :cancel!)
898
+ ] do
899
+ ExternalApi.sync(hero)
900
+ end
901
+ ```
902
+
903
+ Here the global cap is 3. Even if every failure is a `Timeout::Error` (whose individual policy allows 10), the step stops reattempting after 3 consecutive failures. When the cap is hit, the `terminal_action` of the *matched* policy applies — so if the 4th failure is a `Timeout::Error`, the workflow pauses (the default), but if it's a `RateLimitError`, the workflow cancels.
904
+
905
+ > [!TIP]
906
+ > Store reusable policy arrays as frozen constants so you can share them across workflows:
907
+ >
908
+ > ```ruby
909
+ > module ExceptionPolicies
910
+ > EXTERNAL_API = [
911
+ > GenevaDrive::ExceptionPolicy.new(:reattempt!, matching: Net::OpenTimeout, max_reattempts: 5),
912
+ > GenevaDrive::ExceptionPolicy.new(:cancel!, matching: OAuth2::Error),
913
+ > GenevaDrive::ExceptionPolicy.new(:pause!) # blanket fallback
914
+ > ].freeze
915
+ > end
916
+ >
917
+ > class SyncWorkflow < GenevaDrive::Workflow
918
+ > step :sync, on_exception: ExceptionPolicies::EXTERNAL_API do
919
+ > ExternalApi.sync(hero)
920
+ > end
921
+ > end
922
+ > ```
923
+
924
+ ### Imperative Exception Handlers
925
+
926
+ For full control, pass a block to `on_exception`. The block receives the exception and runs in the workflow context — you must call a flow control method (`reattempt!`, `cancel!`, `pause!`, or `skip!`):
927
+
928
+ ```ruby
929
+ class SmartRetryWorkflow < GenevaDrive::Workflow
930
+ on_exception RateLimitError do |error|
931
+ reattempt! wait: error.retry_after.seconds
932
+ end
933
+
934
+ on_exception do |error|
935
+ if error.message.include?("temporary")
936
+ reattempt!
937
+ else
938
+ cancel!
939
+ end
940
+ end
941
+
942
+ step :call_api do
943
+ ExternalApi.call(hero)
944
+ end
945
+ end
946
+ ```
947
+
948
+ If the block returns without calling a flow control method, the workflow pauses.
949
+
950
+ Imperative handlers cannot be combined with `max_reattempts:`, `wait:`, or `terminal_action:` — manage that logic inside the block.
951
+
952
+ ### Controlling Error Reporting
953
+
954
+ When a step raises an exception, GenevaDrive reports it to your error tracker via `Rails.error.report`. This is the right default — you want visibility into failures. But some exceptions are expected and handled: rate limits, transient timeouts, throttling responses. Reporting these on every reattempt floods your error tracker with noise and obscures the errors that actually need attention.
955
+
956
+ The `report:` option controls when `Rails.error.report` is called. It accepts three values:
957
+
958
+ - `:always` — (default) Report every exception, regardless of what the policy does with it. This is the safest choice and preserves the behavior you're used to.
959
+ - `:never` — Never report the exception. Use this for errors that are fully expected and handled — rate limits, throttles, circuit breaker trips. The exception still triggers the policy action (reattempt, skip, etc.), but your error tracker stays clean.
960
+ - `:terminal_only` — Suppress reports while the step is being reattempted, but report when reattempts are exhausted and the `terminal_action` fires. This is the sweet spot for transient errors: you don't care about individual retries, but you want to know when the retries give up.
961
+
962
+ Use `report:` on both class-level `on_exception` and on `ExceptionPolicy` objects:
963
+
964
+ ```ruby
965
+ class CalendarSyncWorkflow < GenevaDrive::Workflow
966
+ # Rate limits are expected — never report them
967
+ on_exception Pecorino::Throttle::Throttled, report: :never do |error|
968
+ reattempt!(wait: error.retry_after.clamp(10, 600).seconds)
969
+ end
970
+
971
+ # Transient timeouts: only report when we give up
972
+ on_exception Net::OpenTimeout,
973
+ action: :reattempt!,
974
+ wait: 10.seconds,
975
+ max_reattempts: 5,
976
+ terminal_action: :cancel!,
977
+ report: :terminal_only
978
+
979
+ step :sync_events do
980
+ GoogleCalendar.sync(hero)
981
+ end
982
+ end
983
+ ```
984
+
985
+ The same option works on `ExceptionPolicy` objects for reusable policies and composable arrays:
986
+
987
+ ```ruby
988
+ RATE_LIMIT_POLICY = GenevaDrive::ExceptionPolicy.new(
989
+ :reattempt!,
990
+ matching: RateLimitError,
991
+ wait: 30.seconds,
992
+ max_reattempts: 10,
993
+ terminal_action: :pause!,
994
+ report: :terminal_only
995
+ )
996
+
997
+ step :call_api, on_exception: [
998
+ RATE_LIMIT_POLICY,
999
+ GenevaDrive::ExceptionPolicy.new(:reattempt!, matching: Timeout::Error, report: :never, max_reattempts: 3),
1000
+ GenevaDrive::ExceptionPolicy.new(:pause!) # blanket fallback, reports by default
1001
+ ] do
1002
+ ExternalApi.call(hero)
1003
+ end
1004
+ ```
1005
+
1006
+ > [!IMPORTANT]
1007
+ > The `report:` option only controls `Rails.error.report`. The exception is still re-raised after the executor commits its state transitions — your background job framework (Sidekiq, SolidQueue, etc.) will see it. If your error tracker also hooks into the job framework's error handler, you may need to configure that separately.
714
1008
 
715
1009
  ### Manual Exception Handling
716
1010
 
717
- For granular control, handle exceptions within the step:
1011
+ For the most granular control, handle exceptions directly within the step using standard Ruby `rescue`:
718
1012
 
719
1013
  ```ruby
720
1014
  class PaymentWorkflow < GenevaDrive::Workflow
@@ -731,6 +1025,8 @@ class PaymentWorkflow < GenevaDrive::Workflow
731
1025
  end
732
1026
  ```
733
1027
 
1028
+ This bypasses the exception policy entirely — the step catches the error before GenevaDrive sees it.
1029
+
734
1030
  For a complete example showing granular exception handling, see the [Payment Processing Workflow](#payment-processing-workflow) in the appendix.
735
1031
 
736
1032
  ### Recovering Paused Workflows
@@ -740,19 +1036,204 @@ Find and resume paused workflows from the console:
740
1036
  ```ruby
741
1037
  GenevaDrive::Workflow.paused.each do |workflow|
742
1038
  puts "#{workflow.class.name} ##{workflow.id}: next step is #{workflow.next_step_name}"
743
- puts " Error: #{workflow.step_executions.failed.last&.error_message}"
1039
+
1040
+ # Check if there's a scheduled execution (externally paused)
1041
+ if (scheduled = workflow.current_execution)
1042
+ overdue = scheduled.scheduled_for < Time.current
1043
+ puts " Scheduled step: #{scheduled.step_name} (#{overdue ? 'overdue' : 'future'})"
1044
+ puts " Originally scheduled for: #{scheduled.scheduled_for}"
1045
+ end
1046
+
1047
+ # Check if there's a failed execution (paused due to exception)
1048
+ if (failed = workflow.step_executions.failed.last)
1049
+ puts " Failed step: #{failed.step_name}"
1050
+ puts " Error: #{failed.error_message}"
1051
+ end
744
1052
  end
745
1053
 
746
1054
  # Resume a specific workflow
747
1055
  workflow = GenevaDrive::Workflow.find(id)
748
- workflow.resume!
1056
+ workflow.resume! # Re-enqueues existing scheduled step or creates new one
1057
+ ```
1058
+
1059
+ ## Workflow States
1060
+
1061
+ ### State Machine Diagram
1062
+
1063
+ ```mermaid
1064
+ stateDiagram-v2
1065
+ [*] --> ready: create
1066
+ ready --> performing: step starts
1067
+ performing --> ready: step completes / reattempt
1068
+ performing --> finished: last step completes / finished!
1069
+ performing --> canceled: cancel!
1070
+ performing --> paused: pause! / exception
1071
+ paused --> ready: resume!
749
1072
  ```
750
1073
 
1074
+ A workflow begins in `ready` state. When a step starts executing, it transitions to `performing`. Upon step completion, it returns to `ready` (unless it was the last step, in which case it transitions to `finished`).
1075
+
1076
+ ### Step Execution States
1077
+
1078
+ Step executions have their own state machine:
1079
+
1080
+ | State | Meaning |
1081
+ |-------|---------|
1082
+ | `scheduled` | Waiting to run |
1083
+ | `in_progress` | Currently executing |
1084
+ | `completed` | Finished successfully |
1085
+ | `failed` | Exception occurred |
1086
+ | `canceled` | Canceled before execution |
1087
+ | `skipped` | Skipped via `skip_if` or `skip!` |
1088
+
751
1089
  ---
752
1090
 
753
- ## Working with Heroes
1091
+ # Part IV — Tips and Recommendations
1092
+
1093
+ ## Keep Workflows Short-Lived
1094
+
1095
+ A GenevaDrive workflow should be a one-off process with a clear beginning and end — not a long-running loop that retries forever. The best workflows are created, execute their steps over hours or days, reach `finished` (or `canceled`), and are eventually cleaned up by the housekeeping job. If you need a recurring process for a hero, create a new workflow on a schedule rather than building an immortal one.
1096
+
1097
+ ### Why one-off workflows are better
1098
+
1099
+ **Code changes between deploys.** A workflow definition is a Ruby class. When you deploy new code, the class may have different steps, different logic, different wait times. A workflow that was created before the deploy still carries its original step pointer — it will execute the steps as they existed when it was defined. If you keep a single workflow alive for months, it accumulates an increasingly stale relationship with your codebase. A fresh workflow always runs the latest code.
1100
+
1101
+ **Step executions are append-only history.** Each step attempt creates a `StepExecution` record. This history is the audit trail for what happened and when. A workflow that runs for two weeks and processes three steps has a concise, readable history. A workflow that has been alive for six months and has reattempted the same polling step 4,000 times has an audit trail that is effectively noise. The step execution table becomes a dumping ground rather than a useful record.
1102
+
1103
+ **Terminal states tell you what succeeded and what didn't.** GenevaDrive's state machine is designed around the assumption that workflows end. A `finished` workflow is one that completed all its steps successfully. A `canceled` workflow is one that was stopped deliberately. A `paused` workflow is one that needs attention. These states are meaningful precisely because they are final. A workflow that never finishes — because it is designed to loop forever — can never be `finished`, which means you lose the ability to distinguish "working as intended" from "stuck." Your monitoring and alerting becomes guesswork.
1104
+
1105
+ ### The cron pattern
1106
+
1107
+ Instead of building a workflow that loops endlessly, create a cron job that periodically creates a new workflow for each eligible hero:
1108
+
1109
+ ```ruby
1110
+ # app/jobs/create_billing_check_workflows_job.rb
1111
+ class CreateBillingCheckWorkflowsJob < ApplicationJob
1112
+ def perform
1113
+ Account.active.find_each do |account|
1114
+ # Only create if no ongoing workflow exists for this hero
1115
+ next if BillingCheckWorkflow.ongoing.exists?(hero: account)
1116
+ BillingCheckWorkflow.create!(hero: account)
1117
+ end
1118
+ end
1119
+ end
1120
+ ```
1121
+
1122
+ Schedule this job with your cron adapter:
1123
+
1124
+ ```ruby
1125
+ # With Solid Queue recurring tasks
1126
+ billing_check:
1127
+ class: CreateBillingCheckWorkflowsJob
1128
+ schedule: every day at 6am
1129
+
1130
+ # With GoodJob cron
1131
+ GoodJob::Cron.schedule(
1132
+ cron: "0 6 * * *",
1133
+ class: "CreateBillingCheckWorkflowsJob"
1134
+ )
1135
+ ```
1136
+
1137
+ Each invocation creates a fresh workflow that runs the current code, produces a clean execution history, and terminates with a clear outcome.
1138
+
1139
+ ### Compare the two approaches
1140
+
1141
+ **The immortal workflow (don't do this):**
1142
+
1143
+ ```ruby
1144
+ class BillingCheckWorkflow < GenevaDrive::Workflow
1145
+ step :check_billing do
1146
+ if hero.payment_overdue?
1147
+ BillingMailer.overdue(hero).deliver_later
1148
+ end
1149
+ reattempt!(wait: 1.day) # Loop forever
1150
+ end
1151
+ end
1152
+ ```
1153
+
1154
+ This workflow never finishes. You cannot tell from its state whether it is working correctly or stuck. Its step execution history grows without bound. When you deploy new billing logic, existing workflows keep running the old code path. And if you need to change the check interval from daily to hourly, you must somehow update every running workflow instance.
1155
+
1156
+ **The one-off workflow (do this instead):**
1157
+
1158
+ ```ruby
1159
+ class BillingCheckWorkflow < GenevaDrive::Workflow
1160
+ step :check_payment_status do
1161
+ skip! unless hero.payment_overdue?
1162
+ hero.update!(last_overdue_notice_at: Time.current)
1163
+ end
1164
+
1165
+ step :send_overdue_notice, skip_if: -> { hero.last_overdue_notice_at.nil? } do
1166
+ BillingMailer.overdue(hero).deliver_later
1167
+ end
1168
+
1169
+ step :schedule_followup, skip_if: -> { hero.last_overdue_notice_at.nil? } do
1170
+ hero.update!(followup_needed: true)
1171
+ end
1172
+ end
1173
+ ```
1174
+
1175
+ Each day, the cron job creates a new `BillingCheckWorkflow` for each account. It runs, does its work, finishes. Tomorrow's workflow will use tomorrow's code. The execution history for each workflow is three steps long. You can query `BillingCheckWorkflow.where(hero: account).finished` to see every successful check, and `BillingCheckWorkflow.where(hero: account).paused` to find the ones that hit problems.
1176
+
1177
+ ### The partial index allows coexistence
1178
+
1179
+ You might wonder: if a cron job creates a new workflow every day, won't old workflows collide with new ones?
1180
+
1181
+ No. GenevaDrive maintains a partial unique index on `(type, hero_type, hero_id)` that only covers **ongoing** workflows — those not in `finished` or `canceled` state. This means:
1182
+
1183
+ - Only one `BillingCheckWorkflow` for a given account can be `ready`, `performing`, or `paused` at any time.
1184
+ - Once a workflow reaches `finished` or `canceled`, it drops out of the index entirely.
1185
+ - The next cron run can create a fresh workflow for the same hero without conflict.
1186
+
1187
+ This is the designed-for pattern. Yesterday's finished workflow and today's active workflow coexist in the database. The unique index prevents accidental duplicates (two active workflows for the same hero), while the housekeeping job eventually cleans up old finished workflows after your configured retention period.
1188
+
1189
+ ```ruby
1190
+ # This works because yesterday's workflow already finished
1191
+ account = Account.find(42)
1192
+ BillingCheckWorkflow.where(hero: account).finished.count # => 30 (last month)
1193
+ BillingCheckWorkflow.ongoing.where(hero: account).count # => 1 (today's)
1194
+ ```
1195
+
1196
+ The short-lived workflow pattern works _with_ the library's design rather than against it. Workflows start, do their work, and end. The database tells you exactly which ones succeeded and which ones didn't. New code applies immediately. History stays readable. This is the intended way to use GenevaDrive.
1197
+
1198
+ ## The Asynchronous Execution Model
1199
+
1200
+ ### Key Assumptions
1201
+
1202
+ GenevaDrive steps execute asynchronously via ActiveJob. This has important implications:
1203
+
1204
+ - **Every step runs on a different machine/process/thread.** Don't assume anything about the execution environment between steps.
1205
+ - **Instance variables don't persist between steps.** Each step gets a freshly-loaded workflow instance.
1206
+ - **The workflow is always loaded fresh from the database.** Any changes you make to the workflow or hero are persisted and reloaded.
1207
+ - **Steps may be separated by seconds or months.** A `wait: 30.days` step means exactly what it says.
1208
+
1209
+ ### No Shared State Between Steps
1210
+
1211
+ > [!WARNING]
1212
+ > Instance variables do not persist between steps. Store data in the database.
1213
+
1214
+ ```ruby
1215
+ # WRONG - @data won't exist in next step
1216
+ step :fetch do
1217
+ @data = ExternalApi.fetch(hero.external_id)
1218
+ end
1219
+
1220
+ step :process do
1221
+ process(@data) # @data is nil here!
1222
+ end
1223
+
1224
+ # RIGHT - persist to the hero or another record
1225
+ step :fetch do
1226
+ hero.update!(external_data: ExternalApi.fetch(hero.external_id))
1227
+ end
1228
+
1229
+ step :process do
1230
+ process(hero.external_data)
1231
+ end
1232
+ ```
1233
+
1234
+ You will almost never have the same `self` between steps. Treat each step as an independent unit that reads from and writes to the database.
754
1235
 
755
- ### Choosing the Right Hero
1236
+ ## Choosing the Right Hero
756
1237
 
757
1238
  Make the hero the specific business object being processed, not the user who owns it. This keeps workflows focused and allows multiple concurrent workflows for the same user.
758
1239
 
@@ -767,7 +1248,7 @@ InvoiceWorkflow.create!(hero: user, invoice_id: invoice.id)
767
1248
 
768
1249
  If you're processing a subscription renewal, the hero is the `Subscription`. If you're processing an order, the hero is the `Order`. The more specific your hero, the easier it is to reason about workflow state.
769
1250
 
770
- ### Associating Workflows from the Hero
1251
+ ## Associating Workflows from the Hero
771
1252
 
772
1253
  GenevaDrive uses STI (Single Table Inheritance) combined with a polymorphic `hero` association. This means a single hero can have multiple different workflow types associated with it, each representing a distinct process the hero is going through.
773
1254
 
@@ -819,126 +1300,133 @@ class User < ApplicationRecord
819
1300
  end
820
1301
  ```
821
1302
 
822
- ### Workflows Without Heroes
823
-
824
- Some workflows don't operate on a specific record — system maintenance, batch jobs, or scheduled reports:
825
-
826
- ```ruby
827
- class SystemMaintenanceWorkflow < GenevaDrive::Workflow
828
- may_proceed_without_hero!
829
-
830
- step :cleanup_temp_files do
831
- TempFileService.cleanup_older_than(7.days)
832
- end
1303
+ ## When Heroes Disappear
833
1304
 
834
- step :vacuum_database do
835
- ActiveRecord::Base.connection.execute("VACUUM ANALYZE")
836
- end
1305
+ If the hero is deleted while the workflow is running, GenevaDrive cancels the workflow automatically. This is the right default — in practice, the vast majority of workflows make no sense without their hero. If an admin manually deletes a user, the onboarding workflow for that user should not keep sending emails into the void.
837
1306
 
838
- step :notify_ops do
839
- OpsMailer.maintenance_complete.deliver_later
840
- end
841
- end
1307
+ If a step tries to access a nil hero, it raises, and the workflow pauses. This is also correct — it surfaces the problem to an operator rather than silently continuing with broken assumptions.
842
1308
 
843
- # Create without a hero
844
- SystemMaintenanceWorkflow.create!
845
- ```
1309
+ For the extremely rare case where a workflow must continue after the hero has been deleted from the database, the `may_proceed_without_hero!` escape hatch exists. You will almost never need it.
846
1310
 
847
- ### When Heroes Disappear
1311
+ ---
848
1312
 
849
- If the hero is deleted while the workflow is running, GenevaDrive cancels the workflow. This prevents steps from failing with `RecordNotFound` errors.
1313
+ # Part V — Operations
850
1314
 
851
- If your workflow needs to continue after the hero is deleted (for example, a GDPR erasure workflow), declare it explicitly:
1315
+ ## ActiveJob Integration
852
1316
 
853
- ```ruby
854
- class DataErasureWorkflow < GenevaDrive::Workflow
855
- may_proceed_without_hero!
1317
+ ### How Steps Are Scheduled
856
1318
 
857
- step :archive_for_compliance do
858
- ComplianceArchive.store(hero)
859
- end
1319
+ When a step completes, GenevaDrive creates a new `StepExecution` record and enqueues a `PerformStepJob`. The job runs after the transaction commits (using `after_all_transactions_commit`), ensuring the step execution record is visible to the job worker.
860
1320
 
861
- step :delete_from_primary do
862
- hero.destroy! # Hero no longer exists after this
863
- end
1321
+ If a step specifies `wait:`, GenevaDrive passes the delay to ActiveJob's `set(wait_until:)`. This means your queue adapter handles the scheduling — GenevaDrive doesn't implement its own timer.
864
1322
 
865
- step :purge_from_backups do
866
- # hero is nil here, but we stored the ID we need
867
- BackupService.purge(id: @archived_hero_id)
868
- end
869
- end
870
- ```
1323
+ ### Recommended Queue Adapters
871
1324
 
872
- For an example of a workflow that handles data deletion carefully, see the [Data Retention Workflow](#data-retention-workflow) in the appendix.
1325
+ We recommend using a queue adapter that co-commits with your database:
873
1326
 
874
- ---
1327
+ - **Solid Queue** — Built for Rails, uses your existing database
1328
+ - **GoodJob** — PostgreSQL-based, with excellent admin UI
1329
+ - **Gouda** — Another PostgreSQL option with simple semantics
875
1330
 
876
- ## Installation and Setup
1331
+ Co-committing matters because GenevaDrive relies on transactional guarantees. When a step completes and schedules the next step, both the state change and the job enqueue should be atomic. With co-committing adapters, if the transaction rolls back, the job is never enqueued.
877
1332
 
878
- ### Adding the Gem
1333
+ With non-transactional adapters (Sidekiq, Resque), there's a small window where the job is enqueued but the transaction hasn't committed. GenevaDrive handles this gracefully — the job will see the step execution in the wrong state and skip it — but you may see occasional log warnings.
879
1334
 
880
- ```bash
881
- bundle add geneva_drive
882
- bin/rails generate geneva_drive:install
883
- bin/rails db:migrate
884
- ```
1335
+ ### Inline Enqueueing for Bulk Operations
885
1336
 
886
- The generator creates two migrations: one for workflows and one for step executions.
1337
+ When creating many workflows at once, you may want to batch the job inserts for efficiency. Libraries like BulkEnqueue or native adapter bulk methods can significantly reduce database round-trips. However, GenevaDrive's default behavior of deferring job enqueueing to `after_all_transactions_commit` interferes with bulk enqueueing:
887
1338
 
888
- ### Database Tables
1339
+ 1. `BulkEnqueue.in_bulk { }` starts, buffer is set
1340
+ 2. For each workflow:
1341
+ - Workflow and step execution created in transaction
1342
+ - `after_all_transactions_commit` callback is **registered** (not executed)
1343
+ - Transaction commits
1344
+ 3. Bulk block ends, flushes **empty** buffer, clears buffer
1345
+ 4. **Now** `after_all_transactions_commit` callbacks fire
1346
+ 5. Jobs are enqueued individually (buffer already cleared)
889
1347
 
890
- GenevaDrive uses a two-table design:
1348
+ Use `GenevaDrive.with_inline_enqueue` to temporarily disable deferred enqueueing:
891
1349
 
892
- - **`geneva_drive_workflows`** — The workflow records. Each row represents one workflow instance with its current state, hero association, and progress tracking.
893
- - **`geneva_drive_step_executions`** — The idempotency keys. Each row represents one attempt to execute a step, with timing, outcome, and error information.
1350
+ ```ruby
1351
+ BulkEnqueue.in_bulk do
1352
+ GenevaDrive.with_inline_enqueue do
1353
+ drafts.each do |draft|
1354
+ DeliverBriefWorkflow.create!(hero: draft)
1355
+ end
1356
+ end
1357
+ end
1358
+ ```
894
1359
 
895
- This separation keeps the workflows table clean while maintaining a complete audit trail in step executions.
1360
+ Inside the block, jobs are enqueued immediately (inline) rather than being deferred to `after_all_transactions_commit`. This allows bulk enqueueing libraries to capture and batch the job inserts.
896
1361
 
897
- ### UUID Primary Keys
1362
+ > [!WARNING]
1363
+ > **Only use `with_inline_enqueue` if you have a co-committing, database-backed ActiveJob adapter** (SolidQueue, GoodJob, Gouda) running on the same database as your application. With these adapters, the job INSERT and workflow records are written in the same transaction, guaranteeing atomicity.
1364
+ >
1365
+ > If you use a non-transactional adapter (Redis-based Sidekiq, Resque, etc.), do **not** use this in production — jobs may be picked up before their associated records are committed, causing "record not found" errors.
898
1366
 
899
- If your application uses UUID primary keys, the migrations will detect this and also use UUIDs for the foreign keys and the primary keys of the geneva_drive resources.
1367
+ The method is thread-safe and won't affect other concurrent requests.
900
1368
 
901
- Note that you don't want to mix integer IDs and UUIDs in the same application.
1369
+ ### Custom Job Options
902
1370
 
903
- ---
1371
+ You can override the Active Job settings used to enqueue a workflow's step jobs at three levels: for the entire workflow class, for an individual step, or for a specific workflow instance. All three accept the same keys that Active Job's `set` method understands: `:queue`, `:priority`, `:wait`, and `:wait_until`. Unknown keys are rejected at configuration time so that typos surface immediately instead of silently falling back to defaults.
904
1372
 
905
- ## ActiveJob Integration
1373
+ ```ruby
1374
+ class HighPriorityWorkflow < GenevaDrive::Workflow
1375
+ set_step_job_options queue: :critical, priority: 0
906
1376
 
907
- ### How Steps Are Scheduled
1377
+ step :urgent_action, job_options: {queue: :urgent, priority: -1} do
1378
+ UrgentService.process!(hero)
1379
+ end
1380
+ end
1381
+ ```
908
1382
 
909
- When a step completes, GenevaDrive creates a new `StepExecution` record and enqueues a `PerformStepJob`. The job runs after the transaction commits (using `after_all_transactions_commit`), ensuring the step execution record is visible to the job worker.
1383
+ Options merge from lowest to highest precedence: **class defaults** (`set_step_job_options`) are overridden by **per-instance options** (`workflow.step_job_options = {...}`), which are in turn overridden by **per-step options** (`step ..., job_options: {...}`). The merged hash is passed straight to `PerformStepJob.set(...)` for every enqueue of that step, including reattempts and resume re-enqueueing.
910
1384
 
911
- If a step specifies `wait:`, GenevaDrive passes the delay to ActiveJob's `set(wait_until:)`. This means your queue adapter handles the scheduling — GenevaDrive doesn't implement its own timer.
1385
+ Per-instance options are stored in the workflow's `metadata` column, so they survive across step boundaries and are excluded from the dedupe uniqueness check — two workflows for the same hero cannot coexist just because their job options differ.
912
1386
 
913
- ### Recommended Queue Adapters
1387
+ #### When per-step job options matter: protecting in-flight work from a fresh backlog
914
1388
 
915
- We recommend using a queue adapter that co-commits with your database:
1389
+ Consider a workflow that downloads, processes, and cleans up a large artifact per hero:
916
1390
 
917
- - **Solid Queue** — Built for Rails, uses your existing database
918
- - **GoodJob** — PostgreSQL-based, with excellent admin UI
919
- - **Gouda** — Another PostgreSQL option with simple semantics
1391
+ ```ruby
1392
+ class RubygemsWorkflow < GenevaDrive::Workflow
1393
+ step :extract do
1394
+ # Download a large tarball to local disk
1395
+ end
920
1396
 
921
- Co-committing matters because GenevaDrive relies on transactional guarantees. When a step completes and schedules the next step, both the state change and the job enqueue should be atomic. With co-committing adapters, if the transaction rolls back, the job is never enqueued.
1397
+ step :process do
1398
+ # Do work on the extracted data
1399
+ end
922
1400
 
923
- With non-transactional adapters (Sidekiq, Resque), there's a small window where the job is enqueued but the transaction hasn't committed. GenevaDrive handles this gracefully — the job will see the step execution in the wrong state and skip it — but you may see occasional log warnings.
1401
+ step :cleanup do
1402
+ # Delete the on-disk artifact
1403
+ end
1404
+ end
1405
+ ```
924
1406
 
925
- ### Custom Job Options
1407
+ If you enqueue 50,000 of these at once, all 50,000 `:extract` jobs land on the queue at the default priority — ahead of the `:process` and `:cleanup` steps of workflows that have already downloaded something. On queue adapters where lower numbers run first, the workers happily drain `:extract` jobs first, filling the disk with unprocessed artifacts until the volume runs out and everything starts failing.
926
1408
 
927
- Override the queue or priority for all steps in a workflow:
1409
+ Bumping the later steps to a higher priority (lower number) fixes this:
928
1410
 
929
1411
  ```ruby
930
- class HighPriorityWorkflow < GenevaDrive::Workflow
931
- set_step_job_options queue: :critical, priority: 0
1412
+ class RubygemsWorkflow < GenevaDrive::Workflow
1413
+ step :extract, job_options: {priority: 10} do
1414
+ # Download a large tarball to local disk
1415
+ end
932
1416
 
933
- step :urgent_action do
934
- UrgentService.process!(hero)
1417
+ step :process do # runs at default priority — ahead of :extract
1418
+ # Do work on the extracted data
1419
+ end
1420
+
1421
+ step :cleanup do # also runs ahead of :extract
1422
+ # Delete the on-disk artifact
935
1423
  end
936
1424
  end
937
1425
  ```
938
1426
 
939
- The options are passed directly to ActiveJob's `set` method.
1427
+ Workflows that have already extracted their artifact can now drain through `:process` and `:cleanup` — releasing disk — while new `:extract` jobs wait their turn. The fleet reaches steady state instead of collapsing under its own backlog.
940
1428
 
941
- ---
1429
+ The same pattern is useful whenever a later step releases a scarce resource (disk, external quota, a database lock) that earlier steps consume: give the release-work a higher priority than the acquire-work.
942
1430
 
943
1431
  ## Housekeeping
944
1432
 
@@ -957,7 +1445,7 @@ Configure housekeeping thresholds in an initializer:
957
1445
  # config/initializers/geneva_drive.rb
958
1446
  GenevaDrive.delete_completed_workflows_after = 30.days
959
1447
  GenevaDrive.stuck_in_progress_threshold = 1.hour
960
- GenevaDrive.stuck_scheduled_threshold = 1.hour
1448
+ GenevaDrive.stuck_scheduled_threshold = 15.minutes
961
1449
  GenevaDrive.stuck_recovery_action = :reattempt # or :cancel
962
1450
  ```
963
1451
 
@@ -981,8 +1469,6 @@ GoodJob::Cron.schedule(
981
1469
  GenevaDrive::HousekeepingJob.perform_later
982
1470
  ```
983
1471
 
984
- ---
985
-
986
1472
  ## Testing
987
1473
 
988
1474
  ### Test Helpers
@@ -1046,7 +1532,29 @@ test "skips email if user unsubscribed" do
1046
1532
  end
1047
1533
  ```
1048
1534
 
1049
- ---
1535
+ ### Transactional Tests and SQLite
1536
+
1537
+ GenevaDrive defers job enqueueing to `after_all_transactions_commit` so that step execution records are visible to the job worker before it runs. This can cause problems with transactional tests on SQLite — the `after_all_transactions_commit` callback writes to the database after the inner transaction commits, but with SQLite's single-writer limitation this can conflict with the test transaction wrapper or other connections. Symptoms include `SQLite3` errors in workflow tests while the rest of the test suite works fine.
1538
+
1539
+ By default, GenevaDrive detects `Rails.env.test?` and skips the deferral, enqueueing jobs immediately. This means transactional tests work out of the box with no extra setup.
1540
+
1541
+ If you need strict after-commit semantics in a specific test (for example, to verify the exact commit-then-enqueue ordering), you can re-enable deferral:
1542
+
1543
+ ```ruby
1544
+ test "job is enqueued only after commit" do
1545
+ GenevaDrive.enqueue_after_commit = true
1546
+
1547
+ # ... test that relies on real after_commit timing ...
1548
+ ensure
1549
+ GenevaDrive.enqueue_after_commit = false
1550
+ end
1551
+ ```
1552
+
1553
+ If you are not using transactional tests at all (for example, you use DatabaseCleaner with truncation), you can set this globally in your initializer:
1554
+
1555
+ ```ruby
1556
+ GenevaDrive.enqueue_after_commit = true
1557
+ ```
1050
1558
 
1051
1559
  ## Observability
1052
1560
 
@@ -1087,13 +1595,199 @@ ActiveSupport::Notifications.subscribe("step.geneva_drive") do |event|
1087
1595
  end
1088
1596
  ```
1089
1597
 
1598
+ ### Metric gauges
1599
+
1600
+ `GenevaDrive::HousekeepingJob` reports gauges via [Measurometer](https://rubygems.org/gems/measurometer) on every run:
1601
+
1602
+ | Gauge | Tags | Value |
1603
+ |-------|------|-------|
1604
+ | `geneva_drive.<state>` | _(none)_ | Absolute count of workflows in `<state>` (e.g. `ready`, `paused`, `finished`), summed across all classes |
1605
+ | `geneva_drive.<state>` | `workflow: <ClassName>` | Absolute count of workflows in `<state>` for a single workflow class |
1606
+ | `geneva_drive.paused_ratio` | `workflow: <ClassName>` | Float `0.0..1.0` — that class's `paused` count divided by its total population (all states) |
1607
+
1608
+ `paused_ratio` is normalized on purpose: absolute paused counts are hard to alert on (50 paused of 50 is a fire; 50 of 500,000 is noise), whereas the ratio is bounded and comparable across classes. The denominator is the whole population (paused ÷ total), so it stays within `0..1` even when nothing is ongoing.
1609
+
1090
1610
  ---
1091
1611
 
1092
- ## Appendix
1612
+ # Part VI — Appendix
1613
+
1614
+ ## Appendix: Why "Durable Functions" Are a Mirage
1615
+
1616
+ There is a rather popular approach to building durable execution systems based on the concept of "durable functions". Systems like [Temporal](https://temporal.io) and [absurd](https://lucumr.pocoo.org/2025/11/3/absurd-workflows/) as well as [Vercel Workflows](https://vercel.com/docs/workflow) take that concept quite far. It seems neat on the surface, yet deeply flawed in nature.
1617
+
1618
+ The assumption made with those "durable functions" is that it is possible to _pretend that you have a marshalable stack._ For example, this section in a workflow function:
1619
+
1620
+ ```js
1621
+ let step = 0;
1622
+ while (step++ < 20) {
1623
+ const { newMessages, finishReason } = await ctx.step("iteration", async () => {
1624
+ return await singleStep(messages);
1625
+ });
1626
+ messages.push(...newMessages);
1627
+ if (finishReason !== "tool-calls") {
1628
+ break;
1629
+ }
1630
+ }
1631
+ ```
1632
+
1633
+ can only work if `step` gets marshaled and reinstated if the function gets resumed. Async generators (and Fibers in Ruby, and - in general - any systems based on continuations or coroutines) allow suspension and resumption, but none allow proper _serialization and revival._ If you try to encode a durable function, consisting of multiple steps, as a suspendable and resumable workflow, you essentially have 3 ways to do it:
1634
+
1635
+ * Make your function restartable from the very beginning (idempotent)
1636
+ * Use a serializable system stack frame, which - usually - comes down to serializing a VM image upon suspension
1637
+ * Make the user write functions that only - and ever - use special facilities for accessing transients (current time, database connections, heavy resources)
1638
+
1639
+ Most "workflow engines" do their utmost to maintain the guise of resumable functions _while not providing them._ The fact that you have to wait on a Fiber to receive an HTTP result is not very useful if the only program that can receive that result is the very process which has started that HTTP request.
1640
+
1641
+ ### Why no modern runtime provides stack serialization
1642
+
1643
+ The fundamental issue is that no mainstream runtime — V8, SpiderMonkey, YARV, the JVM, the CLR — provides the ability to serialize an executing call stack to bytes and revive it later. This is not an oversight. It is a deliberate engineering decision driven by hard constraints.
1644
+
1645
+ A call stack contains pointers: return addresses, references to heap objects, handles to file descriptors and sockets, pointers into native libraries. Serializing a pointer is meaningless — the memory address 0x7fff5fbff8c0 on one machine means nothing on another, or even on the same machine after a restart. To serialize a stack, you must either:
1646
+
1647
+ 1. **Replace all pointers with symbolic references** that can be resolved at revival time. This requires a complete indirection layer over every memory access — a performance catastrophe for general-purpose code.
1648
+ 2. **Serialize the entire heap along with the stack**, effectively snapshotting the whole process. This is what Smalltalk images did.
1649
+ 3. **Restrict the language** so that stacks never contain non-serializable values. This means no closures over native resources, no FFI, no direct system calls.
1650
+
1651
+ Modern runtimes chose speed over serializability. JavaScript engines like V8 perform aggressive JIT compilation that inlines functions, eliminates stack frames, and stores values in machine registers. The "stack" you think exists in your `async` function is a fiction maintained for debugging — the actual execution state is scattered across registers, hidden classes, inline caches, and optimized machine code that has no stable representation.
1652
+
1653
+ ### Systems that actually solved this
1654
+
1655
+ True stack serialization is not impossible. It has been done — just not in environments optimized for raw speed.
1656
+
1657
+ **Smalltalk images** are the canonical example. A Smalltalk system serializes its entire object memory, including all activation records (stack frames), to a single file. You can save an image mid-computation, quit, restart days later, and continue exactly where you left off. This works because Smalltalk controls everything: the object format, the bytecode interpreter, the garbage collector. There are no opaque pointers to external resources — or if there are, the image-saving mechanism explicitly handles them.
1658
+
1659
+ **Erlang/OTP** takes a different approach: processes are so lightweight and isolated that you simply design for crash recovery. A process dies, its supervisor restarts it, and it reconstructs its state from durable storage. There's no pretense that you can freeze and thaw a running computation — you design for restart from the beginning.
1660
+
1661
+ **Scheme continuations** (particularly in implementations like Chez Scheme or Gambit) can capture delimited continuations and, in some implementations, serialize them. But these implementations pay the cost: they maintain a CPS-transformed representation that is inherently slower than direct-style execution.
1662
+
1663
+ **[Seaside](https://seaside.st/)** deserves special mention. This Smalltalk web framework, developed in the early 2000s, used continuations to model web application control flow. You could write a multi-page wizard as a single method with `call:` and `answer:` — the framework would suspend execution while waiting for user input and resume it when the response arrived. [Wee](https://github.com/mneumann/wee), a Ruby port inspired by Seaside's ideas, attempted the same trick using Ruby's `callcc`. Both frameworks demonstrated genuine continuation-based web development. But they also demonstrated its limits: Seaside required either keeping all session continuations in memory (scaling poorly) or relying on Smalltalk's image persistence (requiring the same VM instance to handle subsequent requests). Wee suffered from Ruby 1.8's notorious continuation memory leaks and remained a curiosity rather than a production tool.
1664
+
1665
+ The fundamental problem with continuation-based web frameworks is process affinity. A suspended continuation exists in the memory of a specific process on a specific machine. When the user submits the next form, that exact process must handle the request — no load balancer can route it elsewhere, no autoscaler can spin up a fresh instance to handle the load, no deployment can replace the running code. This is incompatible with modern elastic infrastructure. Kubernetes doesn't care that your user's shopping cart continuation lives in pod `web-7f8d9c-xk2p4` — when traffic spikes, it will route requests wherever capacity exists. When you deploy, it will terminate old pods and start new ones. Your continuations die with them.
1666
+
1667
+ ### The problem of transient resources
1668
+
1669
+ Even if you could serialize a continuation, you would face the problem of transient resources. A continuation captures the call stack, but the call stack contains references to objects that cannot meaningfully survive process boundaries.
1670
+
1671
+ Consider a database transaction. Your step opens a connection, begins a transaction, inserts a row, and then — mid-transaction — suspends to wait for user confirmation:
1672
+
1673
+ ```ruby
1674
+ step :reserve_inventory do
1675
+ ActiveRecord::Base.transaction do
1676
+ hero.line_items.each { |item| Inventory.decrement!(item.sku, item.quantity) }
1677
+ # Suspend here, wait for payment confirmation...
1678
+ yield # In a hypothetical continuation-based system
1679
+ hero.update!(reserved_at: Time.current)
1680
+ end
1681
+ end
1682
+ ```
1683
+
1684
+ What happens when you try to revive this continuation on a different machine, or even the same machine after a restart? The `ActiveRecord::Base.connection` object holds a socket to a PostgreSQL server. That socket is gone. You could theoretically reconnect — some systems use lazy connection resolution for exactly this reason — but reconnecting gives you a *new* connection. The transaction you started? It was rolled back the moment the original connection died. The `BEGIN` you issued exists only in the logs. The row locks you held have been released. Some other process may have already modified the rows you thought you had locked.
1685
+
1686
+ There is no way to "re-enter" a transaction. Transactions are not addressable resources you can resume — they are ephemeral states of a connection that exist only as long as that connection lives. The same applies to file handles, HTTP connections mid-request, mutex locks, and any other resource that represents a relationship with an external system. A serialized continuation that references such resources is not a suspended computation — it is a lie about the state of the world.
1687
+
1688
+ The lesson from these systems is clear: serializable execution state requires either total control over the runtime environment or acceptance of significant performance overhead. You cannot bolt it onto V8 or Ruby's YARV after the fact.
1689
+
1690
+ ### Why modern developers won't build this
1691
+
1692
+ Building a marshalable stack VM is a multi-year, multi-million-dollar undertaking. It requires:
1693
+
1694
+ - Deep expertise in compiler construction, garbage collection, and runtime systems
1695
+ - Willingness to sacrifice raw performance for serializability
1696
+ - Long-term maintenance commitment as the underlying platform evolves
1697
+
1698
+ The JavaScript ecosystem optimizes for different goals: startup time, peak throughput, memory efficiency on mobile devices. Google, Mozilla, and Apple compete on V8, SpiderMonkey, and JavaScriptCore benchmarks. No one is competing on "ability to serialize a running function to disk."
1699
+
1700
+ The developers building "durable function" frameworks are, by and large, application developers — skilled in their domain, but not runtime engineers. They are building atop V8, not replacing it. They cannot make V8 serialize its internal state because V8 was never designed to expose that state. The best they can do is replay: run the function again from the start, skip the steps that already completed, and hope the interleaving of side effects is deterministic. This is not serialization. It is simulation.
1701
+
1702
+ ### Pretending is denial
1703
+
1704
+ When a framework claims to offer "durable functions" without true stack serialization, it is engaging in denial — not about the laws of physics, but about the semantics of their own system.
1705
+
1706
+ Consider what happens when a "durable function" resumes. The framework:
1707
+
1708
+ 1. Loads the function definition
1709
+ 2. Starts executing from the beginning
1710
+ 3. Intercepts calls to "step" functions and returns cached results instead of re-executing
1711
+
1712
+ This only works if the control flow between steps is perfectly deterministic. But JavaScript (and Ruby, and Python) are not deterministic languages. The order of object keys, the behavior of `Math.random()`, the resolution of race conditions in `Promise.all()` — all of these can vary between runs. If your loop counter depends on a hash iteration order that changed between Node versions, your "resumed" function takes a different path than the original.
1713
+
1714
+ The frameworks paper over this with restrictions: don't use randomness, don't depend on time, don't read from external systems except through blessed APIs. But these restrictions are invisible until you violate them. You write what looks like normal code, you test it, it works — and then six months later, after a Node upgrade, a resumed workflow takes a wrong turn and corrupts your data.
1715
+
1716
+ Honesty requires admitting what the system actually provides. If execution state is not truly serialized, don't pretend it is. Make the boundaries explicit. Make the steps explicit. Make the user acknowledge, at every step boundary, that they are persisting state to a database. GenevaDrive takes this position.
1717
+
1718
+ ## Appendix: Pause as an Operational Safety Valve
1719
+
1720
+ When a step raises an unhandled exception, the default behavior is to **pause** the workflow. This is not a coincidence — it is the most operationally useful thing we can do. A paused workflow is not broken. It is stopped, intact, waiting for a human to decide what happens next.
1721
+
1722
+ ### Why pause instead of letting the exception propagate, retry, or cancel?
1723
+
1724
+ Consider the alternatives:
1725
+
1726
+ - **Let the exception propagate.** The step execution job raises, and the workflow stays in `performing` state. This is bad — a `performing` workflow cannot have a new step execution started, so there is no way to advance it once the bug is fixed or the upstream problem is resolved. The workflow appears to hang indefinitely.
1727
+ - **Infinite retry** can hammer a broken external service, burn API quota, or loop on a bug that no amount of retrying will fix.
1728
+ - **Cancel** throws away all the work the workflow has already done. If five of seven steps completed successfully, canceling means starting over — or worse, leaving the system in a half-finished state with no record of what happened.
1729
+
1730
+ Pausing avoids all of these. The workflow stops advancing, but every piece of state is preserved: which step failed, what exception was raised, the full backtrace, and which step would run next. An operator can inspect the situation, fix the underlying problem (deploy a bug fix, correct bad data, restore an external service), and then decide how to proceed.
1731
+
1732
+ ### The slot stays occupied
1733
+
1734
+ A paused workflow is still **ongoing**. It keeps its slot in the unique index on `(type, hero_type, hero_id)`. This means no competing workflow can be created for the same hero while the paused one exists. If a cron job or a controller action tries to create a duplicate, the unique constraint rejects it.
1735
+
1736
+ This is a feature, not a limitation. Imagine a payment processing workflow pauses because the gateway returned an unexpected error. You do not want a second payment workflow starting for the same order while the first one is stopped mid-flight — that risks double-charging. The paused workflow holds the slot, acting as a lock that says "a human needs to look at this before anything else happens for this hero."
1737
+
1738
+ ### Unpausing a workflow
1739
+
1740
+ You have three options when a workflow is paused:
1741
+
1742
+ **`resume!`** sets the workflow back to `ready` and re-enqueues the step that was about to run. If the step was originally scheduled for some time in the future and that time has now passed, it runs immediately — the system recognizes it is overdue. If the scheduled time is still in the future, it waits the remaining duration. This is the most common recovery path: fix the root cause, then resume.
1743
+
1744
+ ```ruby
1745
+ workflow = PaymentProcessingWorkflow.find(id)
1746
+ workflow.resume! # Re-enqueues the failed step
1747
+ ```
1748
+
1749
+ **`skip!`** marks the current step as skipped and advances to the next one. This is useful when the step is no longer relevant — perhaps you resolved the issue out-of-band, or the step was attempting something that has since been handled manually.
1750
+
1751
+ ```ruby
1752
+ workflow = PaymentProcessingWorkflow.find(id)
1753
+ workflow.skip! # Advances past the failed step
1754
+ ```
1093
1755
 
1094
- ### Complete Example Workflows
1756
+ **`cancel!`** terminates the workflow entirely. The slot in the unique index is released, and a new workflow for the same hero can be created.
1095
1757
 
1096
- #### User Onboarding Workflow
1758
+ ```ruby
1759
+ workflow = PaymentProcessingWorkflow.find(id)
1760
+ workflow.cancel! # Ends the workflow, frees the hero slot
1761
+ ```
1762
+
1763
+ ### Deliberate pause from step code
1764
+
1765
+ Steps can also call `pause!` explicitly to request human intervention before the workflow continues. This is the "emergency stop button" — the step has detected a condition that requires a conscious decision rather than automatic progression.
1766
+
1767
+ The payment processing example in this manual demonstrates this pattern. When the gateway flags a transaction as potentially fraudulent, the step pauses the workflow rather than proceeding:
1768
+
1769
+ ```ruby
1770
+ step :capture_payment, wait: 1.hour do
1771
+ result = PaymentGateway.capture(
1772
+ authorization_id: hero.authorization_id,
1773
+ idempotency_key: "capture-#{hero.id}"
1774
+ )
1775
+ hero.update!(captured_at: Time.current, transaction_id: result.transaction_id)
1776
+ rescue PaymentGateway::FraudSuspected
1777
+ hero.flag_for_fraud_review!
1778
+ pause! # A human must review before we continue
1779
+ end
1780
+ ```
1781
+
1782
+ The workflow will sit in `paused` state — holding the slot, preventing duplicates — until a fraud analyst reviews the case and calls `resume!` or `cancel!`.
1783
+
1784
+ ### Pause on missing step definitions
1785
+
1786
+ There is one more scenario where pause happens automatically. If a workflow references a step name that no longer exists in the class definition — typically after a deploy that removed or renamed a step — the executor pauses the workflow rather than silently skipping or crashing. This gives operators a chance to notice the mismatch and decide whether to skip, cancel, or deploy a fix.
1787
+
1788
+ ## Complete Example Workflows
1789
+
1790
+ ### User Onboarding Workflow
1097
1791
 
1098
1792
  This workflow demonstrates named steps with wait times, skip conditions, and blanket cancellation. It guides a new user through account setup with timed reminders.
1099
1793
 
@@ -1140,7 +1834,7 @@ class UserOnboardingWorkflow < GenevaDrive::Workflow
1140
1834
  end
1141
1835
  ```
1142
1836
 
1143
- #### Payment Processing Workflow
1837
+ ### Payment Processing Workflow
1144
1838
 
1145
1839
  This workflow demonstrates manual exception handling, dynamic retry waits, and early termination. It handles the complexity of interacting with an external payment gateway.
1146
1840
 
@@ -1211,25 +1905,19 @@ class PaymentProcessingWorkflow < GenevaDrive::Workflow
1211
1905
  end
1212
1906
  ```
1213
1907
 
1214
- #### Data Retention Workflow
1908
+ ### User Erasure Workflow
1215
1909
 
1216
- This workflow demonstrates handling hero deletion mid-workflow. It processes GDPR-style data erasure requests while maintaining compliance records.
1217
-
1218
- ```ruby
1219
- class DataRetentionWorkflow < GenevaDrive::Workflow
1220
- may_proceed_without_hero!
1910
+ This workflow processes a GDPR-style data erasure request. Every step requires the hero to exist — if an admin or another process deletes the user externally before the workflow finishes, the next step will raise on `hero` access, pausing the workflow for an operator to investigate. This is the correct behavior: silent continuation with a missing user would be worse than stopping.
1221
1911
 
1222
- step :create_compliance_record do
1223
- ComplianceRecord.create!(
1224
- user_id: hero.id,
1225
- user_email: hero.email,
1226
- request_type: "erasure",
1227
- requested_at: Time.current,
1228
- workflow_id: id
1229
- )
1230
- end
1912
+ The final step destroys the hero. Because it is the last step, the workflow transitions to `finished` and there are no subsequent steps that need the hero.
1231
1913
 
1914
+ ```ruby
1915
+ class UserErasureWorkflow < GenevaDrive::Workflow
1232
1916
  step :export_to_archive do
1917
+ # Every step accesses hero directly — no nil guards.
1918
+ # If hero was deleted externally, this raises and the
1919
+ # workflow pauses. An operator can then investigate why
1920
+ # the user vanished before erasure completed.
1233
1921
  archive_data = UserDataExporter.export(hero)
1234
1922
  ComplianceArchive.store(
1235
1923
  user_id: hero.id,
@@ -1239,56 +1927,45 @@ class DataRetentionWorkflow < GenevaDrive::Workflow
1239
1927
  end
1240
1928
 
1241
1929
  step :notify_third_parties do
1242
- # Request deletion from integrated services
1243
1930
  hero.oauth_connections.each do |connection|
1244
1931
  ThirdPartyDeletionService.request(connection)
1245
1932
  end
1246
1933
  end
1247
1934
 
1248
1935
  step :delete_user_data, wait: 24.hours do
1249
- # Give third parties time to process
1250
- # hero may be nil if already deleted
1251
- return unless hero
1252
-
1936
+ # Give third parties time to process deletion requests
1253
1937
  hero.posts.destroy_all
1254
1938
  hero.comments.destroy_all
1255
1939
  hero.messages.destroy_all
1256
1940
  hero.files.each { |f| f.purge_later }
1257
1941
  end
1258
1942
 
1259
- step :delete_user_record do
1260
- return unless hero
1261
-
1262
- hero_id = hero.id
1263
- hero.destroy!
1264
-
1265
- # Update compliance record
1266
- ComplianceRecord.find_by(user_id: hero_id, workflow_id: id)
1267
- &.update!(completed_at: Time.current)
1268
- end
1269
-
1270
1943
  step :send_confirmation do
1271
- record = ComplianceRecord.find_by(workflow_id: id)
1272
- return unless record
1944
+ ErasureMailer.complete(hero.email).deliver_later
1945
+ end
1273
1946
 
1274
- ComplianceMailer.erasure_complete(record.user_email).deliver_later
1947
+ step :delete_user_record do
1948
+ # Last step — hero is destroyed, workflow finishes.
1949
+ # No subsequent steps need the hero.
1950
+ hero.destroy!
1275
1951
  end
1276
1952
  end
1277
1953
  ```
1278
1954
 
1279
- ### Quick Reference
1955
+ ## Quick Reference
1280
1956
 
1281
- #### Flow Control Methods
1957
+ ### Flow Control Methods
1282
1958
 
1283
1959
  | Method | Effect |
1284
1960
  |--------|--------|
1285
1961
  | `cancel!` | Stop workflow, mark canceled |
1286
1962
  | `pause!` | Stop workflow, await manual resume |
1287
- | `reattempt!(wait:)` | Retry current step, optionally after delay |
1963
+ | `reattempt!(wait:, rewind:)` | Retry current step; `rewind: true` clears cursor |
1288
1964
  | `skip!` | Skip current step, proceed to next |
1289
1965
  | `finished!` | Complete workflow early |
1966
+ | `suspend!(wait:)` | Interrupt resumable step, continue via successor after delay |
1290
1967
 
1291
- #### Workflow States
1968
+ ### Workflow States
1292
1969
 
1293
1970
  | State | Meaning |
1294
1971
  |-------|---------|
@@ -1298,7 +1975,7 @@ end
1298
1975
  | `canceled` | Workflow was canceled |
1299
1976
  | `paused` | Awaiting manual intervention |
1300
1977
 
1301
- #### Step Execution States
1978
+ ### Step Execution States
1302
1979
 
1303
1980
  | State | Meaning |
1304
1981
  |-------|---------|
@@ -1309,13 +1986,18 @@ end
1309
1986
  | `canceled` | Canceled before execution |
1310
1987
  | `skipped` | Skipped via `skip_if` or `skip!` |
1311
1988
 
1312
- #### Step Options
1989
+ A resumable step interrupted mid-iteration completes its execution with outcome `continued` and schedules a successor execution — there is no separate state for it.
1990
+
1991
+ ### Step Options
1313
1992
 
1314
1993
  | Option | Type | Description |
1315
1994
  |--------|------|-------------|
1316
1995
  | `wait:` | Duration | Delay before step executes |
1996
+ | `job_options:` | Hash | Options passed to Active Job's `set` method for this step |
1317
1997
  | `skip_if:` | Proc, Symbol, Boolean | Condition to skip step |
1318
1998
  | `on_exception:` | Symbol | Exception handler (`:pause!`, `:cancel!`, `:reattempt!`, `:skip!`) |
1319
1999
  | `max_reattempts:` | Integer, nil | Max consecutive reattempts before pausing (default: 100, `nil` = unlimited) |
1320
2000
  | `before_step:` | Symbol | Insert before this step |
1321
2001
  | `after_step:` | Symbol | Insert after this step |
2002
+ | `max_iterations:` | Integer | (resumable_step) Interrupt after N iterations, continue via successor |
2003
+ | `max_runtime:` | Duration | (resumable_step) Interrupt after duration elapsed, continue via successor |