ruby_reactor 0.8.2 → 0.8.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. checksums.yaml +4 -4
  2. data/.claude/skills/speckit-review/SKILL.md +324 -0
  3. data/.release-please-manifest.json +1 -1
  4. data/.specify/extensions.yml +10 -0
  5. data/.specify/feature.json +1 -1
  6. data/.specify/workflows/speckit/workflow.yml +13 -1
  7. data/.specify/workflows/workflow-registry.json +2 -2
  8. data/CHANGELOG.md +82 -0
  9. data/CLAUDE.md +2 -2
  10. data/README.md +35 -2
  11. data/lib/ruby_reactor/adapters/active_job/router.rb +19 -0
  12. data/lib/ruby_reactor/adapters/sidekiq/router.rb +21 -0
  13. data/lib/ruby_reactor/context.rb +26 -0
  14. data/lib/ruby_reactor/context_serializer.rb +4 -2
  15. data/lib/ruby_reactor/dsl/interrupt_builder.rb +14 -0
  16. data/lib/ruby_reactor/dsl/lockable.rb +76 -21
  17. data/lib/ruby_reactor/dsl/step_builder.rb +112 -1
  18. data/lib/ruby_reactor/error/async_result_pending.rb +1 -1
  19. data/lib/ruby_reactor/error/execution_parked.rb +16 -0
  20. data/lib/ruby_reactor/error/reactor_contention_park.rb +26 -0
  21. data/lib/ruby_reactor/error/step_contention_park.rb +26 -0
  22. data/lib/ruby_reactor/executor/async_step_dispatch.rb +109 -3
  23. data/lib/ruby_reactor/executor/compensation_manager.rb +99 -17
  24. data/lib/ruby_reactor/executor/ordered_lock_support.rb +76 -44
  25. data/lib/ruby_reactor/executor/result_handler.rb +31 -11
  26. data/lib/ruby_reactor/executor/retry_manager.rb +9 -1
  27. data/lib/ruby_reactor/executor/step_coordination.rb +788 -0
  28. data/lib/ruby_reactor/executor/step_executor.rb +115 -11
  29. data/lib/ruby_reactor/executor.rb +90 -20
  30. data/lib/ruby_reactor/map/element_executor.rb +24 -2
  31. data/lib/ruby_reactor/map/helpers.rb +35 -11
  32. data/lib/ruby_reactor/max_retries_exhausted_failure.rb +2 -2
  33. data/lib/ruby_reactor/open_telemetry.rb +61 -24
  34. data/lib/ruby_reactor/retry_context.rb +31 -2
  35. data/lib/ruby_reactor/rspec/helpers.rb +15 -0
  36. data/lib/ruby_reactor/rspec/matchers.rb +92 -0
  37. data/lib/ruby_reactor/rspec/test_subject.rb +7 -1
  38. data/lib/ruby_reactor/step/async_reactor_step.rb +40 -24
  39. data/lib/ruby_reactor/step/compose_step.rb +14 -3
  40. data/lib/ruby_reactor/step.rb +49 -7
  41. data/lib/ruby_reactor/step_sweeper.rb +29 -1
  42. data/lib/ruby_reactor/step_worker.rb +260 -37
  43. data/lib/ruby_reactor/version.rb +1 -1
  44. data/lib/ruby_reactor/web/api.rb +72 -7
  45. data/lib/ruby_reactor/web/coordination_serializer.rb +120 -2
  46. data/lib/ruby_reactor/web/public/assets/{index-Dw4KV4QY.js → index-CeZU-ESu.js} +9 -9
  47. data/lib/ruby_reactor/web/public/index.html +1 -1
  48. data/lib/ruby_reactor/worker.rb +56 -30
  49. data/lib/ruby_reactor.rb +27 -5
  50. data/specs/future_improvements.md +250 -0
  51. metadata +8 -28
  52. data/specs/002-step-input-contracts/checklists/requirements.md +0 -49
  53. data/specs/002-step-input-contracts/contracts/dsl-surface.md +0 -193
  54. data/specs/002-step-input-contracts/data-model.md +0 -115
  55. data/specs/002-step-input-contracts/plan.md +0 -165
  56. data/specs/002-step-input-contracts/quickstart.md +0 -170
  57. data/specs/002-step-input-contracts/research.md +0 -233
  58. data/specs/002-step-input-contracts/spec.md +0 -359
  59. data/specs/002-step-input-contracts/tasks.md +0 -367
  60. data/specs/004-inheritable-step-class/checklists/requirements.md +0 -40
  61. data/specs/004-inheritable-step-class/contracts/step-lifecycle.md +0 -85
  62. data/specs/004-inheritable-step-class/data-model.md +0 -116
  63. data/specs/004-inheritable-step-class/plan.md +0 -174
  64. data/specs/004-inheritable-step-class/quickstart.md +0 -112
  65. data/specs/004-inheritable-step-class/research.md +0 -308
  66. data/specs/004-inheritable-step-class/spec.md +0 -316
  67. data/specs/004-inheritable-step-class/tasks.md +0 -258
  68. data/specs/active_job.md +0 -259
  69. data/specs/deferred-003-step-lock-declarations/checklists/requirements.md +0 -51
  70. data/specs/deferred-003-step-lock-declarations/contracts/dsl-surface.md +0 -154
  71. data/specs/deferred-003-step-lock-declarations/data-model.md +0 -131
  72. data/specs/deferred-003-step-lock-declarations/plan.md +0 -166
  73. data/specs/deferred-003-step-lock-declarations/quickstart.md +0 -169
  74. data/specs/deferred-003-step-lock-declarations/research.md +0 -196
  75. data/specs/deferred-003-step-lock-declarations/spec.md +0 -447
  76. data/specs/deferred-003-step-lock-declarations/tasks.md +0 -572
  77. data/specs/possible_feature.md +0 -22
@@ -0,0 +1,788 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyReactor
4
+ class Executor
5
+ # Acquire/release the coordination a STEP declares (`with_lock` etc.),
6
+ # around that step's own work only — never the whole reactor. Built fresh
7
+ # per step execution, and exactly ONE instance guards any one invocation
8
+ # (research D2), in one of two modes:
9
+ #
10
+ # - reactor-driven (the default): `StepExecutor` / `StepWorker` build it
11
+ # from the `StepConfig`, so it sees the step's EFFECTIVE declarations
12
+ # (inline and class alike, in one global order), its retry policy and
13
+ # its output validator. Inside a worker, contention raises `Contended`
14
+ # for the caller to park the execution on.
15
+ # - `direct: true`: `Step.run` called by application code or by another
16
+ # step's body. Its own unit of work (FR-023): it waits the configured
17
+ # `wait:` then fails, keeps no state on the context, and never parks.
18
+ #
19
+ # Order (contracts/dsl-surface.md §3): ordered-lock gate, period
20
+ # fast-check, lock, semaphore, period re-check, rate limit, body, mark
21
+ # period. The rate limit is the LAST acquisition because it is the only
22
+ # one that cannot be handed back — nothing after it can contend, so a
23
+ # spent slot is never followed by a park. A park therefore releases
24
+ # everything this step took, exactly like reactor-level contention, and
25
+ # carries nothing across the gap but its ordered-lock position (FR-015:
26
+ # the contended step's work has not started, so there is nothing to
27
+ # protect in between).
28
+ # rubocop:disable Metrics/ClassLength
29
+ class StepCoordination
30
+ # The step's key proc raised, or returned nil/empty. The step fails
31
+ # before its body runs; the cause and the step name are both surfaced.
32
+ class KeyError < RubyReactor::Error::Base; end
33
+
34
+ # An `async_step` refused at dispatch because it declares a key this
35
+ # execution holds (`AsyncStepDispatch`). Nothing was dispatched or run.
36
+ class DispatchRefused < RubyReactor::Error::Base; end
37
+
38
+ # A semaphore slot has no hold expiry to bound a rollback wait by, so it
39
+ # waits the lock's default `ttl` instead (005 D-F1). A constant, not a
40
+ # config key — add one if someone needs it.
41
+ DEFAULT_ROLLBACK_WAIT = 60
42
+
43
+ # A `Contended`/`KeyError` raised by a DIRECT `Step.run` inside this
44
+ # step's body. The body had already started, so for THIS step it is an
45
+ # ordinary failure — compensated, never parked — and must not be
46
+ # mistaken for a failure of this step's own acquisition.
47
+ class NestedCoordinationError < StandardError; end
48
+
49
+ # Raised when a primitive cannot be acquired. Carries enough to build
50
+ # both the synchronous failure message and the worker's park decision.
51
+ class Contended < StandardError
52
+ attr_reader :primitive, :key, :step_name, :reactor_name, :original
53
+
54
+ # `message:` overrides the default wording for a re-wrap that needs to
55
+ # say something else (the contention-ceiling failure in
56
+ # `StepExecutor#contention_ceiling_failure`, a rollback that could not
57
+ # re-acquire) while staying a `Contended`.
58
+ def initialize(primitive:, key:, step_name:, reactor_name:, original:, message: nil) # rubocop:disable Metrics/ParameterLists
59
+ @primitive = primitive
60
+ @key = key
61
+ @step_name = step_name
62
+ @reactor_name = reactor_name
63
+ @original = original
64
+ default = "#{reactor_name} step :#{step_name} could not acquire #{primitive} '#{key}': #{original.message}"
65
+ super(message || default)
66
+ end
67
+
68
+ # RateLimit::ExceededError and OrderedLock::WaitError carry a precise
69
+ # hint; Lock/Semaphore::AcquisitionError do not (nil falls back to the
70
+ # configured base delay in Worker.snooze_delay).
71
+ def retry_after_seconds
72
+ original.retry_after_seconds if original.respond_to?(:retry_after_seconds)
73
+ end
74
+ end
75
+
76
+ attr_reader :step_config, :arguments, :context, :reactor_class, :middlewares
77
+
78
+ # rubocop:disable Metrics/ParameterLists
79
+ def initialize(step_config:, arguments:, context:, reactor_class:, middlewares:, direct: false)
80
+ # rubocop:enable Metrics/ParameterLists
81
+ @step_config = step_config
82
+ @arguments = arguments
83
+ @context = context
84
+ @reactor_class = reactor_class
85
+ @middlewares = middlewares
86
+ @direct = direct
87
+ end
88
+
89
+ # FR-007, in one place: a key proc that returns nil/empty or raises fails
90
+ # the step before its work runs, naming the step and the cause. Shared
91
+ # with `AsyncStepDispatch`'s deadlock guard, which computes the same key
92
+ # outside an instance and must fail the same (non-retryable) way.
93
+ def self.resolve_key(config, arguments, step_name)
94
+ key = config[:key_proc].call(arguments)
95
+ if key.nil? || key.to_s.empty?
96
+ raise KeyError.new("#{step_name}: coordination key proc returned nil/empty", step: step_name)
97
+ end
98
+
99
+ key
100
+ rescue KeyError
101
+ raise
102
+ rescue StandardError => e
103
+ raise KeyError.new(
104
+ "#{step_name}: coordination key proc raised #{e.class}: #{e.message}",
105
+ step: step_name, original_error: e
106
+ )
107
+ end
108
+
109
+ # THE forward enforcement site for a step a reactor runs — `StepExecutor`
110
+ # in-process and `StepWorker` for an `async_step` both come through
111
+ # here, so the two paths cannot drift. Arguments are validated first
112
+ # (InputValidationError before any hold, Finding 8), then the step's
113
+ # effective declarations are taken once around its body.
114
+ def self.run_step(step_config, resolved_arguments, context:, reactor_class:, middlewares:)
115
+ arguments = step_config.body_arguments(resolved_arguments, context.inputs)
116
+ body = -> { call_body(step_config, arguments, context) }
117
+ return body.call if none?(step_config)
118
+
119
+ new(step_config: step_config, arguments: arguments, context: context, reactor_class: reactor_class,
120
+ middlewares: middlewares).around_run(&body)
121
+ end
122
+
123
+ # A `Contended`/`KeyError` escaping a reactor step's body was raised by a
124
+ # nested DIRECT `Step.run` in it — whether or not this step declares
125
+ # anything itself — so it is re-raised as `NestedCoordinationError`.
126
+ def self.call_body(step_config, arguments, context)
127
+ step_config.call_body(arguments, context)
128
+ rescue Contended, KeyError => e
129
+ raise NestedCoordinationError, e.message
130
+ end
131
+ private_class_method :call_body
132
+
133
+ # True when constructing a StepCoordination would be pointless — lets
134
+ # every call site skip it with one check per step (plan: "a step
135
+ # declaring nothing pays one nil check per step").
136
+ def self.none?(step_config)
137
+ !step_config.respond_to?(:declares_coordination?) || !step_config.declares_coordination?
138
+ end
139
+
140
+ # The one piece of step state a contention park carries across the gap
141
+ # is its ordered-lock position — a park must not lose its place in line.
142
+ # When the park is instead escalated to a TERMINAL failure (the
143
+ # `lock_snooze_max_attempts` ceiling, in `StepExecutor#handle_contention`
144
+ # and `StepWorker#handle_contention`) no redelivery will consume it, so
145
+ # advance it here or every later position stalls for the full
146
+ # `poison_pill_timeout`.
147
+ def self.discard_parked_state!(context)
148
+ return unless context.is_a?(RubyReactor::Context)
149
+
150
+ stash = context.private_data.delete(:step_ordered_locks) ||
151
+ context.private_data.delete("step_ordered_locks") || {}
152
+ stash.each_value do |raw|
153
+ next unless raw.is_a?(Hash) && (raw[:key] || raw["key"])
154
+
155
+ # `failed: true`: the step this position belongs to ended in a
156
+ # failure, so strict successors must be chain-skipped, not run.
157
+ Executor::OrderedLockSupport.advance_with_retry(raw.transform_keys(&:to_sym), failed: true)
158
+ rescue StandardError => e
159
+ RubyReactor.configuration.logger.warn(
160
+ "RubyReactor could not advance parked ordered-lock position #{raw.inspect}: #{e.message}"
161
+ )
162
+ end
163
+ context.private_data.delete(:step_contention)
164
+ end
165
+
166
+ # See the class comment for the order. Each stage is a bare yield when
167
+ # its primitive is not declared. Release is the natural reverse via
168
+ # nested `ensure`s: semaphore before lock (FR-008), lock before the
169
+ # ordered-lock gate advances.
170
+ def around_run(&block)
171
+ result = ordered_lock_gate do
172
+ period_fast_check do
173
+ with_lock do
174
+ with_semaphore do
175
+ period_recheck do
176
+ rate_limited { run_body(&block) }
177
+ end
178
+ end
179
+ end
180
+ end
181
+ end
182
+ # Reaching here, or raising anything but this step's own `Contended`,
183
+ # means the step is terminal for every declaration shape.
184
+ clear_contention_state
185
+ result
186
+ # Neither is terminal: the execution parks (or, synchronously, the
187
+ # caller turns contention into a failure) and this step's contention
188
+ # counter and marker must survive the park.
189
+ rescue Contended, Error::ExecutionParked
190
+ raise
191
+ rescue StandardError
192
+ clear_contention_state
193
+ raise
194
+ end
195
+
196
+ # Re-take exclusion primitives ONLY (lock, then semaphore) for
197
+ # compensate/undo — never rate limit, period, or the ordered lock
198
+ # (contracts/dsl-surface.md §6): a forward-work quota must never
199
+ # suppress cleanup, and rollback is not part of the order forward work
200
+ # runs in. Waits the declaration's own `rollback_wait` (default: the
201
+ # lock's `ttl`, or `DEFAULT_ROLLBACK_WAIT` for a semaphore), never the
202
+ # forward `wait:` — a forward holder finishes or expires within `ttl`,
203
+ # so the undo outlasts it instead of being dropped (005 D-F1). Rollback
204
+ # never parks: the execution is already mid-failure, so in a worker the
205
+ # wait blocks the thread. A key still busy after the wait comes back as
206
+ # a `Failure(Contended)` for `CompensationManager` to report.
207
+ def around_rollback(&block)
208
+ rollback_with_lock do
209
+ rollback_with_semaphore(&block)
210
+ end
211
+ end
212
+
213
+ private
214
+
215
+ # Everything is acquired: this step is no longer waiting on anything.
216
+ def run_body
217
+ clear_contention_marker
218
+ yield
219
+ end
220
+
221
+ def rollback_with_lock
222
+ config = step_config.lock_config
223
+ return yield unless config
224
+
225
+ wait = config[:rollback_wait] || config[:ttl]
226
+ acquire_for_rollback(:lock, config, wait) do |key|
227
+ lock = RubyReactor::Lock.new(key, owner: owner, ttl: config[:ttl], wait: wait,
228
+ auto_extend: config.fetch(:auto_extend, true))
229
+ begin
230
+ lock.acquire
231
+ rescue RubyReactor::Lock::AcquisitionError => e
232
+ next [nil, e]
233
+ end
234
+ push_key(key)
235
+ emit(:lock_acquired, key)
236
+ begin
237
+ [yield, nil]
238
+ ensure
239
+ release_lock(lock)
240
+ pop_key(key)
241
+ emit(:lock_released, key)
242
+ end
243
+ end
244
+ end
245
+
246
+ def rollback_with_semaphore
247
+ config = step_config.semaphore_config
248
+ return yield unless config
249
+
250
+ limit = config[:limit]
251
+ wait = config[:rollback_wait] || DEFAULT_ROLLBACK_WAIT
252
+ acquire_for_rollback(:semaphore, config, wait) do |key|
253
+ begin
254
+ semaphore = poll_semaphore(key, limit, wait)
255
+ rescue RubyReactor::Semaphore::AcquisitionError => e
256
+ next [nil, e]
257
+ end
258
+ push_key(key) if limit == 1
259
+ emit(:semaphore_acquired, key, limit)
260
+ begin
261
+ [yield, nil]
262
+ ensure
263
+ release_semaphore(semaphore, key, limit)
264
+ end
265
+ end
266
+ end
267
+
268
+ # Polls instead of `Semaphore#acquire`'s blocking pop: the storage
269
+ # adapter shares ONE Redis connection per process, and a blocking pop
270
+ # held for up to `rollback_wait` would stall every other thread's Redis
271
+ # call — lock auto-extenders and ordered-lock heartbeats included.
272
+ def poll_semaphore(key, limit, wait)
273
+ deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + wait.to_f
274
+ loop do
275
+ semaphore = RubyReactor::Semaphore.new(key, limit: limit, wait: 0)
276
+ semaphore.acquire
277
+ return semaphore
278
+ rescue RubyReactor::Semaphore::AcquisitionError
279
+ raise if Process.clock_gettime(Process::CLOCK_MONOTONIC) >= deadline
280
+
281
+ sleep 0.1
282
+ end
283
+ end
284
+
285
+ # Shared "compute the key, run the acquire block, turn a failure into
286
+ # the rollback Failure shape" wrapper for the two rollback primitives.
287
+ # The block returns `[result, acquisition_error]`; a KeyError computing
288
+ # the key itself is reported the same way.
289
+ def acquire_for_rollback(primitive, config, wait)
290
+ key = key_for(config)
291
+ result, error = yield(key)
292
+ return result unless error
293
+
294
+ rollback_failure(primitive, key, error, wait)
295
+ rescue KeyError => e
296
+ rollback_failure(primitive, nil, e, wait)
297
+ end
298
+
299
+ # A `Contended`, not a bare string, so `CompensationManager` can report
300
+ # the key and primitive on `Failure#rollback_failures`.
301
+ def rollback_failure(primitive, key, error, wait)
302
+ RubyReactor::Failure(
303
+ Contended.new(
304
+ primitive: primitive, key: key, step_name: step_name, reactor_name: reactor_label, original: error,
305
+ message: "could not re-acquire #{primitive} '#{key}' for rollback of :#{step_name} within #{wait}s: " \
306
+ "#{error.message}"
307
+ ),
308
+ retryable: false, step_name: step_name
309
+ )
310
+ end
311
+
312
+ # Strict-ordering gate (contract §3 position 1, research D8): the
313
+ # position is assigned when the execution FIRST REACHES this step (its
314
+ # key reads step arguments, which do not exist until then) — so
315
+ # executions are ordered by arrival at the step, not by enqueue. Takes
316
+ # nothing else while waiting for a turn (D3 step 1): out of turn, this
317
+ # raises `Contended` before any other primitive is even attempted.
318
+ def ordered_lock_gate(&block)
319
+ config = step_config.ordered_lock_config
320
+ return yield unless config
321
+
322
+ info = ordered_lock_arrival_info(config)
323
+ # Nested on a key this thread is already ordered on: no nonce was
324
+ # assigned, so run ungated (see `ordered_lock_arrival_info`).
325
+ return yield unless info
326
+
327
+ # The same exhaustive classifier the reactor level uses (R-06). A
328
+ # step's gate always runs before its body, so the position has never
329
+ # started: always `fresh`.
330
+ case gate_ordered_lock(info)
331
+ when :go, :drained
332
+ with_active_ordered_key(info[:key]) { run_under_ordered_lock(info, &block) }
333
+ when :skip_chain
334
+ # Terminal (skipped, not failed) — advance it like the executor's
335
+ # reactor-level short-circuit does, or every later skipped step
336
+ # stays in flight and the sequence never drains.
337
+ Executor::OrderedLockSupport.advance_with_retry(info, failed: false)
338
+ delete_ordered_lock_stash
339
+ RubyReactor.Skipped(nil, reason: :ordered_lock_chain_failed, step_name: step_name)
340
+ when :stale
341
+ # The position belongs to a drained generation whose numbering a
342
+ # newer batch reuses (F7): the body must not run unordered. No
343
+ # advance — the epoch fence makes it a no-op — just drop the stash.
344
+ delete_ordered_lock_stash
345
+ RubyReactor.Skipped(nil, reason: :ordered_lock_stale_batch, step_name: step_name)
346
+ end
347
+ end
348
+
349
+ # A `WaitError` means "not this nonce's turn yet". A worker parks and
350
+ # keeps the position (the redelivery re-adopts the same nonce).
351
+ # Synchronously the execution is terminal, and the position never held
352
+ # the turn, so it has no failed work for successors to be protected
353
+ # from: hand it back with `failed: false` (005 D-F3), or it would stall
354
+ # every successor until the poison pill — and, with `failed: true`,
355
+ # chain-skip every strict successor forever.
356
+ def gate_ordered_lock(info)
357
+ Executor::OrderedLockSupport.gate(info, fresh: true)
358
+ rescue RubyReactor::OrderedLock::WaitError => e
359
+ unless parking?
360
+ Executor::OrderedLockSupport.advance_with_retry(info, failed: false)
361
+ delete_ordered_lock_stash
362
+ end
363
+ raise Contended.new(primitive: :ordered_lock, key: info[:key], step_name: step_name,
364
+ reactor_name: reactor_label, original: e)
365
+ end
366
+
367
+ # Same thread-local guard `Reactor#assign_ordered_lock_nonce!` and
368
+ # `OrderedLockSupport#enter_ordered_lock_scope` keep for reactor-level
369
+ # ordering: while this step holds a position on `key`, a nested
370
+ # `Reactor.run` (or step) on the same key must see it and skip assigning
371
+ # a second nonce, or the two wait on each other until the poison pill.
372
+ def with_active_ordered_key(key)
373
+ active = Executor::OrderedLockSupport.active_keys
374
+ active << key
375
+ yield
376
+ ensure
377
+ idx = active.rindex(key)
378
+ active.delete_at(idx) if idx
379
+ end
380
+
381
+ # Charged LAST, immediately before the body (see the class comment), so
382
+ # a slot is only ever spent on work that is about to run. The rescue
383
+ # covers the charge only — a `RateLimit::ExceededError` raised by the
384
+ # body itself is the body's failure, not this step's contention.
385
+ def rate_limited
386
+ config = step_config.rate_limit_config
387
+ return yield unless config
388
+
389
+ key_base, limits = rate_limit_key_and_limits(config)
390
+ charge_rate_limit(key_base, limits)
391
+ yield
392
+ end
393
+
394
+ def charge_rate_limit(key_base, limits)
395
+ RubyReactor::RateLimit.new(key_base, limits: limits).check_and_increment!
396
+ rescue RubyReactor::RateLimit::ExceededError => e
397
+ raise Contended.new(primitive: :rate_limit, key: key_base, step_name: step_name,
398
+ reactor_name: reactor_label, original: e)
399
+ end
400
+
401
+ # Named config resolves lazily against the registry (config order does
402
+ # not matter). An unregistered name is a configuration mistake, not
403
+ # contention: like a key that cannot be computed (FR-007) it fails the
404
+ # step before its work, as this step's own `KeyError` — which the
405
+ # executor and `StepWorker` both turn into a non-retryable,
406
+ # never-started failure.
407
+ def rate_limit_key_and_limits(config)
408
+ if config[:name]
409
+ [config[:name].to_s, RubyReactor.configuration.rate_limits.fetch(config[:name])]
410
+ else
411
+ [key_for(config), config[:limits]]
412
+ end
413
+ rescue RubyReactor::RateLimitRegistry::UnknownLimitError => e
414
+ raise KeyError.new("#{step_name}: #{e.message}", step: step_name, original_error: e)
415
+ end
416
+
417
+ # First arrival assigns a fresh nonce and stashes it (keyed by step name)
418
+ # so a redelivery or in-process retry of the SAME step re-reads the SAME
419
+ # nonce (T060 scenario 5) instead of cutting in line with a fresh one. A
420
+ # `Context` round-trips its `private_data` through JSON, symbolizing
421
+ # every hash key at every depth — the per-step key comes back as a
422
+ # Symbol even though it was stored as a String, so lookups check both.
423
+ def ordered_lock_arrival_info(config)
424
+ stash = ordered_lock_stash
425
+ cached = stash[stash_key] || stash[stash_key.to_sym]
426
+ return cached if cached
427
+
428
+ key = key_for(config)
429
+ # Nested under an ordered scope on the SAME key in this thread (an
430
+ # outer step, or an outer `Reactor.run`): a second nonce would never
431
+ # come up — the outer waits for this one to finish, this one waits for
432
+ # the outer to advance. Mirror `Reactor#assign_ordered_lock_nonce!`:
433
+ # skip assignment, warn, and let the inner work run ungated.
434
+ if Executor::OrderedLockSupport.active_keys.include?(key)
435
+ RubyReactor.configuration.logger.warn(
436
+ "RubyReactor: step :#{step_name} declares `with_ordered_lock` on key '#{key}', which this " \
437
+ "thread is already ordered on — nonce assignment skipped, the step runs without ordering " \
438
+ "enforcement. Use a different key, or move the nested call to a top-level invocation."
439
+ )
440
+ return nil
441
+ end
442
+
443
+ nonce, epoch = RubyReactor::OrderedLock.assign(key, ttl: config[:ttl])
444
+ info = {
445
+ key: key, nonce: nonce, epoch: epoch, poison_pill_timeout: config[:poison_pill_timeout],
446
+ ttl: config[:ttl], strict: config.fetch(:strict, true)
447
+ }
448
+ stash[stash_key] = info
449
+ info
450
+ end
451
+
452
+ # On the context only for a reactor-driven step — the one mode with a
453
+ # redelivery or retry to carry the position to. A direct call's position
454
+ # lives and dies with this instance, so it can never collide with the
455
+ # stash of the reactor step whose body made the call.
456
+ def ordered_lock_stash
457
+ return @ordered_lock_stash ||= {} unless context_state?
458
+
459
+ context.private_data[:step_ordered_locks] ||= {}
460
+ end
461
+
462
+ def delete_ordered_lock_stash
463
+ stash = context_state? ? context.private_data[:step_ordered_locks] : @ordered_lock_stash
464
+ return unless stash
465
+
466
+ stash.delete(stash_key)
467
+ stash.delete(stash_key.to_sym)
468
+ end
469
+
470
+ def stash_key
471
+ step_name.to_s
472
+ end
473
+
474
+ # Heartbeats while the body runs so a merely-slow step is not
475
+ # poison-passed by a successor (T061). The position's fate is decided
476
+ # ONCE, as `outcome`, and carried out in `ensure` — so every exit,
477
+ # including one that is not a `StandardError`, stops the heartbeat
478
+ # (005 R-07, F8). This position passed the gate, so it held the turn: a
479
+ # failure here is the chain's failure (`failed: true`, D-F3).
480
+ def run_under_ordered_lock(info)
481
+ heartbeat = Executor::OrderedLockSupport.start_heartbeat(info)
482
+ outcome = :abandoned
483
+ begin
484
+ result = yield
485
+ outcome = if retry_pending?(result)
486
+ :retry_pending
487
+ elsif chain_failed?(result)
488
+ :failed
489
+ else
490
+ :succeeded
491
+ end
492
+ result
493
+ rescue Contended
494
+ # Lock/semaphore/rate-limit contention after the gate: parked in a
495
+ # worker it keeps its place; synchronously it is terminal.
496
+ outcome = parking? ? :parked : :failed
497
+ raise
498
+ rescue Error::ExecutionParked
499
+ outcome = :parked
500
+ raise
501
+ rescue StandardError
502
+ outcome = :failed
503
+ raise
504
+ ensure
505
+ heartbeat.stop
506
+ finish_position(info, outcome)
507
+ end
508
+ end
509
+
510
+ # - :succeeded / :failed — terminal: advance (a failure records the
511
+ # strict chain marker) and drop the stash.
512
+ # - :retry_pending / :parked — `RetryManager` or the redelivery runs this
513
+ # step again and must keep its place: the stash survives, so the next
514
+ # attempt re-reads the SAME nonce. The heartbeat is stopped across the
515
+ # gap; `poison_pill_timeout` bounds it.
516
+ # - :abandoned — an exit that is not a `StandardError` (`Sidekiq::Shutdown`,
517
+ # `NoMemoryError`, ...). Not advanced: `Sidekiq::Shutdown` pushes the
518
+ # job back to run again, which must keep this place, and `failed: true`
519
+ # would poison successors for work that may yet complete. With the
520
+ # heartbeat stopped, the poison pill releases the position within
521
+ # `poison_pill_timeout`.
522
+ def finish_position(info, outcome)
523
+ return unless %i[succeeded failed].include?(outcome)
524
+
525
+ Executor::OrderedLockSupport.advance_with_retry(info, failed: outcome == :failed)
526
+ delete_ordered_lock_stash
527
+ end
528
+
529
+ # Mirrors `RetryManager#handle_failure_result`'s decision, made here one
530
+ # moment earlier: `prepare_retry_attempt` has already counted this
531
+ # attempt, so both read the same numbers and agree. A direct call has no
532
+ # retry policy of its own — it is never retried.
533
+ def retry_pending?(result)
534
+ return false unless context_state?
535
+ return false unless result.is_a?(RubyReactor::Failure) && result.retryable?
536
+
537
+ max_attempts = step_config.retry_config[:max_attempts]
538
+ return false unless max_attempts.to_i > 1
539
+
540
+ context.retry_context.can_retry_step?(step_name, max_attempts)
541
+ end
542
+
543
+ # The reactor validates a step's output AFTER `around_run` returns, so a
544
+ # body that succeeded with a contract-violating value would advance this
545
+ # position (or mark the period bucket) as successful even though the
546
+ # step is about to be turned into a failure. `step_config` is the
547
+ # reactor's `StepConfig` for every reactor-driven step, class-backed or
548
+ # inline, so its validator is always visible here; re-run it (it is a
549
+ # pure check).
550
+ def chain_failed?(result)
551
+ return true if result.is_a?(RubyReactor::Failure)
552
+ return false unless result.is_a?(RubyReactor::Success)
553
+
554
+ validator = step_config.respond_to?(:output_validator) && step_config.output_validator
555
+ return false unless validator
556
+
557
+ !validator.call(result.value).success?
558
+ rescue StandardError
559
+ false
560
+ end
561
+
562
+ def with_lock
563
+ config = step_config.lock_config
564
+ return yield unless config
565
+
566
+ key = key_for(config)
567
+ lock = RubyReactor::Lock.new(
568
+ key, owner: owner, ttl: config[:ttl], wait: wait_for(config[:wait]),
569
+ auto_extend: config.fetch(:auto_extend, true)
570
+ )
571
+ begin
572
+ lock.acquire
573
+ rescue RubyReactor::Lock::AcquisitionError => e
574
+ emit(:lock_failed, key, e)
575
+ raise Contended.new(primitive: :lock, key: key, step_name: step_name, reactor_name: reactor_label,
576
+ original: e)
577
+ end
578
+ push_key(key)
579
+ emit(:lock_acquired, key)
580
+
581
+ begin
582
+ yield
583
+ ensure
584
+ release_lock(lock)
585
+ pop_key(key)
586
+ emit(:lock_released, key)
587
+ end
588
+ end
589
+
590
+ def with_semaphore
591
+ config = step_config.semaphore_config
592
+ return yield unless config
593
+
594
+ key = key_for(config)
595
+ limit = config[:limit]
596
+ semaphore = RubyReactor::Semaphore.new(key, limit: limit, wait: wait_for(config[:wait]))
597
+ begin
598
+ semaphore.acquire
599
+ rescue RubyReactor::Semaphore::AcquisitionError => e
600
+ emit(:semaphore_failed, key, limit, e)
601
+ raise Contended.new(primitive: :semaphore, key: key, step_name: step_name, reactor_name: reactor_label,
602
+ original: e)
603
+ end
604
+ # Only a single-slot semaphore has the circular-wait shape the async
605
+ # deadlock guard can act on (T032/T034) — mirrors `Executor#acquire_semaphore`.
606
+ push_key(key) if limit == 1
607
+ emit(:semaphore_acquired, key, limit)
608
+
609
+ begin
610
+ yield
611
+ ensure
612
+ release_semaphore(semaphore, key, limit)
613
+ end
614
+ end
615
+
616
+ # Dedup window, fast pre-check (contract §3 position 2): mirrors
617
+ # `Executor#check_period_gate` — skips a step already marked without
618
+ # spending a lock/semaphore attempt on it. NOT authoritative by itself:
619
+ # two callers can both pass this check before either marks the bucket,
620
+ # so `period_recheck` repeats it UNDER the exclusion primitives, which
621
+ # is what actually closes the race.
622
+ def period_fast_check
623
+ config = step_config.period_config
624
+ return yield unless config
625
+ return skipped_for_period if storage_adapter.period_seen?(period_key(config))
626
+
627
+ yield
628
+ end
629
+
630
+ # Dedup window, re-check (contract §3 position 5): the authoritative
631
+ # check, taken under lock/semaphore and before the rate limit, so two
632
+ # racing callers serialize here, only the first marks the bucket, and a
633
+ # deduplicated step never spends a rate-limit slot.
634
+ #
635
+ # Marked only on a plain `Success` whose output the reactor will accept
636
+ # — never `Skipped`/`Halt` (no work happened), a failure, or a value the
637
+ # output contract is about to reject (`chain_failed?`), any of which
638
+ # would dedup away the next legitimate run.
639
+ def period_recheck
640
+ config = step_config.period_config
641
+ return yield unless config
642
+
643
+ key = period_key(config)
644
+ return skipped_for_period if storage_adapter.period_seen?(key)
645
+
646
+ result = yield
647
+ if plain_success?(result) && !chain_failed?(result)
648
+ storage_adapter.period_mark(key, RubyReactor::Period.ttl_seconds(config[:every]))
649
+ end
650
+ result
651
+ end
652
+
653
+ def skipped_for_period
654
+ RubyReactor.Skipped(nil, reason: :period, step_name: step_name)
655
+ end
656
+
657
+ def plain_success?(result)
658
+ result.is_a?(RubyReactor::Success) && !result.is_a?(RubyReactor::Halt) && !result.is_a?(RubyReactor::Skipped)
659
+ end
660
+
661
+ def period_key(config)
662
+ RubyReactor::Period.key(key_for(config), config[:every])
663
+ end
664
+
665
+ def storage_adapter
666
+ RubyReactor.configuration.storage_adapter
667
+ end
668
+
669
+ # Only a reactor-driven step running in a worker has a redelivery to
670
+ # park into; a direct call or a synchronous run is terminal.
671
+ def parking?
672
+ context_state? && context.inline_async_execution
673
+ end
674
+
675
+ # Whether this invocation's state (ordered-lock position, contention
676
+ # counter and marker) belongs on the context. Never for a direct call:
677
+ # its context is the CALLER's execution, whose step state it must not
678
+ # read or overwrite.
679
+ def context_state?
680
+ !@direct && context.is_a?(RubyReactor::Context)
681
+ end
682
+
683
+ # Mirrors `Executor#contention_wait`: inside a worker, fail fast instead
684
+ # of blocking the thread — the caller snoozes via `perform_in` instead.
685
+ def wait_for(configured)
686
+ parking? ? 0 : configured
687
+ end
688
+
689
+ def clear_contention_marker
690
+ context.private_data.delete(:step_contention) if context_state?
691
+ end
692
+
693
+ # Deliberately only on a terminal result, never per acquisition: a step
694
+ # declaring several primitives acquires them one at a time, so resetting
695
+ # the counter earlier would restart the `lock_snooze_max_attempts`
696
+ # budget on every redelivery that gets past the first primitive.
697
+ def clear_contention_state
698
+ return unless context_state?
699
+
700
+ clear_contention_marker
701
+ context.retry_context.clear_contention_for_step(step_name)
702
+ end
703
+
704
+ # Never raises: a release failure must not mask the step's own result
705
+ # (mirrors `Executor#release_one`).
706
+ def release_lock(lock)
707
+ released = lock.release
708
+ return if released
709
+
710
+ RubyReactor.configuration.logger.warn(
711
+ "RubyReactor lock '#{lock.key}' was not held at release time (likely TTL expired or owner changed)"
712
+ )
713
+ rescue StandardError => e
714
+ RubyReactor.configuration.logger.warn("RubyReactor failed to release lock '#{lock.key}': #{e.message}")
715
+ end
716
+
717
+ def release_semaphore(semaphore, key, limit)
718
+ semaphore.release
719
+ pop_key(key) if limit == 1
720
+ emit(:semaphore_released, key)
721
+ rescue StandardError => e
722
+ RubyReactor.configuration.logger.warn("RubyReactor failed to release semaphore '#{key}': #{e.message}")
723
+ end
724
+
725
+ # The coordination hooks are shared with reactor-level coordination;
726
+ # mark these as this step's for the duration of the call, so a
727
+ # middleware can attribute them without guessing from `current_step`.
728
+ def emit(event, *args)
729
+ return middlewares.on(event, *args, context) unless context.is_a?(RubyReactor::Context)
730
+
731
+ previous = context.coordinating_step
732
+ context.coordinating_step = step_name
733
+ begin
734
+ middlewares.on(event, *args, context)
735
+ ensure
736
+ context.coordinating_step = previous
737
+ end
738
+ end
739
+
740
+ def reactor_label
741
+ reactor_class&.name || reactor_class.inspect
742
+ end
743
+
744
+ # re-entrancy: same owner as every reactor in this execution tree
745
+ # (research D5, full rule landed in US4/T037).
746
+ def owner
747
+ @owner ||= context&.coordination_owner ||
748
+ ((context.root_context || context).context_id if context) ||
749
+ SecureRandom.uuid
750
+ end
751
+
752
+ def key_for(config)
753
+ StepCoordination.resolve_key(config, arguments, step_name)
754
+ end
755
+
756
+ # A direct call is its own unit of work: it names the step class that
757
+ # was invoked, never the caller's `current_step` — that is the step
758
+ # whose body made the call (F9).
759
+ def step_name
760
+ return step_config.name if @direct || !context.is_a?(RubyReactor::Context)
761
+
762
+ context.current_step || step_config.name.to_s
763
+ end
764
+
765
+ def push_key(key)
766
+ held_lock_keys << key
767
+ end
768
+
769
+ # Pop ONE occurrence, not every occurrence (Finding 1) — a nested hold
770
+ # on the same key (reactor + step, or step + inner step) must leave the
771
+ # outer hold's entry intact for the async deadlock guard to keep seeing
772
+ # the key while the outer scope is still open.
773
+ def pop_key(key)
774
+ keys = held_lock_keys
775
+ idx = keys.index(key)
776
+ keys.delete_at(idx) if idx
777
+ end
778
+
779
+ def held_lock_keys
780
+ root = context && (context.root_context || context)
781
+ return [] unless root
782
+
783
+ root.private_data[:held_lock_keys] ||= []
784
+ end
785
+ end
786
+ # rubocop:enable Metrics/ClassLength
787
+ end
788
+ end