ruby_reactor 0.8.2 → 0.8.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. checksums.yaml +4 -4
  2. data/.claude/skills/speckit-review/SKILL.md +324 -0
  3. data/.release-please-manifest.json +1 -1
  4. data/.specify/extensions.yml +10 -0
  5. data/.specify/feature.json +1 -1
  6. data/.specify/workflows/speckit/workflow.yml +13 -1
  7. data/.specify/workflows/workflow-registry.json +2 -2
  8. data/CHANGELOG.md +82 -0
  9. data/CLAUDE.md +2 -2
  10. data/README.md +35 -2
  11. data/lib/ruby_reactor/adapters/active_job/router.rb +19 -0
  12. data/lib/ruby_reactor/adapters/sidekiq/router.rb +21 -0
  13. data/lib/ruby_reactor/context.rb +26 -0
  14. data/lib/ruby_reactor/context_serializer.rb +4 -2
  15. data/lib/ruby_reactor/dsl/interrupt_builder.rb +14 -0
  16. data/lib/ruby_reactor/dsl/lockable.rb +76 -21
  17. data/lib/ruby_reactor/dsl/step_builder.rb +112 -1
  18. data/lib/ruby_reactor/error/async_result_pending.rb +1 -1
  19. data/lib/ruby_reactor/error/execution_parked.rb +16 -0
  20. data/lib/ruby_reactor/error/reactor_contention_park.rb +26 -0
  21. data/lib/ruby_reactor/error/step_contention_park.rb +26 -0
  22. data/lib/ruby_reactor/executor/async_step_dispatch.rb +109 -3
  23. data/lib/ruby_reactor/executor/compensation_manager.rb +99 -17
  24. data/lib/ruby_reactor/executor/ordered_lock_support.rb +76 -44
  25. data/lib/ruby_reactor/executor/result_handler.rb +31 -11
  26. data/lib/ruby_reactor/executor/retry_manager.rb +9 -1
  27. data/lib/ruby_reactor/executor/step_coordination.rb +788 -0
  28. data/lib/ruby_reactor/executor/step_executor.rb +115 -11
  29. data/lib/ruby_reactor/executor.rb +90 -20
  30. data/lib/ruby_reactor/map/element_executor.rb +24 -2
  31. data/lib/ruby_reactor/map/helpers.rb +35 -11
  32. data/lib/ruby_reactor/max_retries_exhausted_failure.rb +2 -2
  33. data/lib/ruby_reactor/open_telemetry.rb +61 -24
  34. data/lib/ruby_reactor/retry_context.rb +31 -2
  35. data/lib/ruby_reactor/rspec/helpers.rb +15 -0
  36. data/lib/ruby_reactor/rspec/matchers.rb +92 -0
  37. data/lib/ruby_reactor/rspec/test_subject.rb +7 -1
  38. data/lib/ruby_reactor/step/async_reactor_step.rb +40 -24
  39. data/lib/ruby_reactor/step/compose_step.rb +14 -3
  40. data/lib/ruby_reactor/step.rb +49 -7
  41. data/lib/ruby_reactor/step_sweeper.rb +29 -1
  42. data/lib/ruby_reactor/step_worker.rb +260 -37
  43. data/lib/ruby_reactor/version.rb +1 -1
  44. data/lib/ruby_reactor/web/api.rb +72 -7
  45. data/lib/ruby_reactor/web/coordination_serializer.rb +120 -2
  46. data/lib/ruby_reactor/web/public/assets/{index-Dw4KV4QY.js → index-CeZU-ESu.js} +9 -9
  47. data/lib/ruby_reactor/web/public/index.html +1 -1
  48. data/lib/ruby_reactor/worker.rb +56 -30
  49. data/lib/ruby_reactor.rb +27 -5
  50. data/specs/future_improvements.md +250 -0
  51. metadata +8 -28
  52. data/specs/002-step-input-contracts/checklists/requirements.md +0 -49
  53. data/specs/002-step-input-contracts/contracts/dsl-surface.md +0 -193
  54. data/specs/002-step-input-contracts/data-model.md +0 -115
  55. data/specs/002-step-input-contracts/plan.md +0 -165
  56. data/specs/002-step-input-contracts/quickstart.md +0 -170
  57. data/specs/002-step-input-contracts/research.md +0 -233
  58. data/specs/002-step-input-contracts/spec.md +0 -359
  59. data/specs/002-step-input-contracts/tasks.md +0 -367
  60. data/specs/004-inheritable-step-class/checklists/requirements.md +0 -40
  61. data/specs/004-inheritable-step-class/contracts/step-lifecycle.md +0 -85
  62. data/specs/004-inheritable-step-class/data-model.md +0 -116
  63. data/specs/004-inheritable-step-class/plan.md +0 -174
  64. data/specs/004-inheritable-step-class/quickstart.md +0 -112
  65. data/specs/004-inheritable-step-class/research.md +0 -308
  66. data/specs/004-inheritable-step-class/spec.md +0 -316
  67. data/specs/004-inheritable-step-class/tasks.md +0 -258
  68. data/specs/active_job.md +0 -259
  69. data/specs/deferred-003-step-lock-declarations/checklists/requirements.md +0 -51
  70. data/specs/deferred-003-step-lock-declarations/contracts/dsl-surface.md +0 -154
  71. data/specs/deferred-003-step-lock-declarations/data-model.md +0 -131
  72. data/specs/deferred-003-step-lock-declarations/plan.md +0 -166
  73. data/specs/deferred-003-step-lock-declarations/quickstart.md +0 -169
  74. data/specs/deferred-003-step-lock-declarations/research.md +0 -196
  75. data/specs/deferred-003-step-lock-declarations/spec.md +0 -447
  76. data/specs/deferred-003-step-lock-declarations/tasks.md +0 -572
  77. data/specs/possible_feature.md +0 -22
@@ -2,6 +2,7 @@
2
2
 
3
3
  module RubyReactor
4
4
  class Executor
5
+ # rubocop:disable Metrics/ClassLength
5
6
  class StepExecutor
6
7
  include AsyncStepDispatch
7
8
 
@@ -106,7 +107,13 @@ module RubyReactor
106
107
  end
107
108
  result
108
109
  rescue Exception => e # rubocop:disable Lint/RescueException
109
- @middlewares.on(:failed_step, step_config.name, e, @context) unless completed
110
+ # A park signal (contention, or an awaited background result) is
111
+ # "try again later", never a failure: `:snooze_step`, mirroring
112
+ # `:snooze_reactor` (005 R-04).
113
+ unless completed
114
+ event = e.is_a?(Error::ExecutionParked) ? :snooze_step : :failed_step
115
+ @middlewares.on(event, step_config.name, e, @context)
116
+ end
110
117
  raise
111
118
  end
112
119
  end
@@ -191,6 +198,24 @@ module RubyReactor
191
198
  e.step_name = step_config.name
192
199
  e.step_arguments ||= resolved_arguments
193
200
  raise
201
+ rescue Executor::StepCoordination::Contended => e
202
+ handle_contention(step_config, e, resolved_arguments)
203
+ # `KeyError`: this step's coordination could not be resolved — a key proc
204
+ # that raised or returned nil/empty, or an unregistered `with_rate_limit`
205
+ # name. Not contention: a non-retryable failure, never a park (mirrors
206
+ # how the reactor-level equivalent escalates instead of snoozing in
207
+ # `Worker#perform`).
208
+ rescue Executor::StepCoordination::KeyError => e
209
+ RubyReactor::Failure(e, step_name: step_config.name, reactor_name: @reactor_class.name,
210
+ step_arguments: resolved_arguments, inputs: @context.inputs, retryable: false)
211
+ # A park signal from a composed child (its contention, or its wait on a
212
+ # background result) is not this step failing: let it reach the
213
+ # executors above, which park their own holds, and the worker (F10).
214
+ # Nor is it a retry attempt: give back the one `prepare_retry_attempt`
215
+ # just counted, as `handle_contention` does for this step's own park.
216
+ rescue Error::ExecutionParked
217
+ @context.retry_context.decrement_attempt_for_step(step_config.name)
218
+ raise
194
219
  rescue StandardError => e
195
220
  # Identify redacted inputs
196
221
  redact_inputs = @reactor_class.inputs.select { |_, config| config[:redact] }.keys
@@ -205,6 +230,89 @@ module RubyReactor
205
230
  )
206
231
  end
207
232
 
233
+ # Contention (US3). Synchronously there is no queue to park into: the
234
+ # step fails. In a worker the execution parks at this step — by raising
235
+ # `Error::StepContentionPark`, which every executor on the stack lets
236
+ # through after parking its own holds, and which the worker (or
237
+ # `Map::ElementExecutor`) requeues once, at the top, after all of them
238
+ # have saved (005 R-01). Bounded by `lock_snooze_max_attempts` on this
239
+ # step's own contention counter, separate from its retry budget.
240
+ #
241
+ # `Context#with_step`'s `ensure` has already restored `current_step` to
242
+ # its pre-step value by the time this rescue runs, so it is pinned again
243
+ # here: it is the resume cursor the redelivery continues from.
244
+ def handle_contention(step_config, contended, resolved_arguments)
245
+ return contention_failure(step_config, contended, resolved_arguments) unless @context.inline_async_execution
246
+
247
+ @context.current_step = step_config.name
248
+ # Park evidence belongs to the async path ONLY: synchronously there is
249
+ # no requeue, the step just fails, and writing these would report a
250
+ # terminal execution as parked and leave the dashboard "waiting".
251
+ attempt = @context.retry_context.contention_attempts_for_step(step_config.name) + 1
252
+ @context.append_execution_trace(
253
+ { type: :contention_park, step: step_config.name, primitive: contended.primitive, key: contended.key,
254
+ attempt: attempt, timestamp: Time.now }
255
+ )
256
+ @context.private_data[:step_contention] = {
257
+ step: step_config.name, primitive: contended.primitive, key: contended.key, attempts: attempt,
258
+ next_attempt_at: nil
259
+ }
260
+ log_async_event(
261
+ "step_coordination.parked", step_config.name,
262
+ key: contended.key, primitive: contended.primitive, attempt: attempt
263
+ )
264
+
265
+ # A park is not a retry attempt (Finding 2): give back the one
266
+ # `prepare_retry_attempt` just counted, and count contention instead.
267
+ @context.retry_context.decrement_attempt_for_step(step_config.name)
268
+ count = @context.retry_context.increment_contention_for_step(step_config.name)
269
+ return contention_ceiling_failure(step_config, contended, count) if contention_ceiling?(contended, count)
270
+
271
+ raise Error::StepContentionPark, contended
272
+ end
273
+
274
+ # `OrderedLock::WaitError` is exempt, as in `Worker#handle_snooze`: its
275
+ # own poison-pill timeout is the only meaningful upper bound.
276
+ def contention_ceiling?(contended, count)
277
+ max = RubyReactor.configuration.lock_snooze_max_attempts
278
+ return false if contended.original.is_a?(RubyReactor::OrderedLock::WaitError)
279
+
280
+ max != :infinity && count > max
281
+ end
282
+
283
+ # Terminal: earlier parks kept this step's `:step_contention` marker
284
+ # (which would report a failed execution as still waiting on a key) and
285
+ # its un-advanced ordered-lock position (which would stall every
286
+ # successor until its poison_pill_timeout) for a redelivery that is no
287
+ # longer coming. Hand both back. Carried as a `Contended`, not a bare
288
+ # string, so `CompensationManager#step_never_started?` still recognises
289
+ # that this step's body never ran.
290
+ def contention_ceiling_failure(step_config, contended, count)
291
+ Executor::StepCoordination.discard_parked_state!(@context)
292
+ RubyReactor::Failure(
293
+ Executor::StepCoordination::Contended.new(
294
+ primitive: contended.primitive, key: contended.key, step_name: step_config.name,
295
+ reactor_name: @reactor_class.name, original: contended.original,
296
+ message: "Step '#{step_config.name}' gave up on #{contended.primitive} '#{contended.key}' after " \
297
+ "#{count} contention attempts"
298
+ ),
299
+ step_name: step_config.name, reactor_name: @reactor_class.name, retryable: false,
300
+ exception_class: contended.original.class.name
301
+ )
302
+ end
303
+
304
+ # The `Contended` itself, not its `.original`: it is what tells
305
+ # `CompensationManager#step_never_started?` that THIS step's own
306
+ # acquisition failed. A bare `Lock::AcquisitionError` could equally have
307
+ # been raised by the body (a nested `Reactor.run`), whose side effects
308
+ # must be compensated. `ResultHandler#resolve_exception_class` reports
309
+ # the cause's class.
310
+ def contention_failure(step_config, contended, resolved_arguments)
311
+ RubyReactor::Failure(contended, step_name: step_config.name, reactor_name: @reactor_class.name,
312
+ step_arguments: resolved_arguments, inputs: @context.inputs,
313
+ retryable: false, exception_class: contended.original.class.name)
314
+ end
315
+
208
316
  def execute_step_sync(step_config, resolved_arguments = nil)
209
317
  @context.with_step(step_config.name) do
210
318
  # Check conditions and guards
@@ -349,22 +457,17 @@ module RubyReactor
349
457
  { type: :run, step: step_config.name, timestamp: Time.now,
350
458
  arguments: contract ? contract.redact(arguments) : arguments }
351
459
  )
352
- if step_config.has_run_block?
353
- # Execute inline block
354
- # If no arguments are defined for the step, pass the reactor inputs as arguments
355
- args_to_pass = arguments.empty? ? @context.inputs : arguments
356
- args_to_pass = step_config.inline_contract.enforce!(args_to_pass) if step_config.inline_contract
357
- catch(StepSignals::TAG) { step_config.run_block.call(args_to_pass, @context) }
358
- elsif step_config.has_impl?
359
- # Execute step class
360
- catch(StepSignals::TAG) { step_config.impl.run(arguments, @context) }
361
- else
460
+ unless step_config.has_run_block? || step_config.has_impl?
362
461
  raise Error::ValidationError.new(
363
462
  "Step '#{step_config.name}' has no implementation",
364
463
  step: step_config.name,
365
464
  context: @context
366
465
  )
367
466
  end
467
+
468
+ Executor::StepCoordination.run_step(step_config, arguments, context: @context,
469
+ reactor_class: @reactor_class,
470
+ middlewares: @middlewares)
368
471
  end
369
472
 
370
473
  def find_context_by_id(root_context, target_id)
@@ -382,5 +485,6 @@ module RubyReactor
382
485
  nil
383
486
  end
384
487
  end
488
+ # rubocop:enable Metrics/ClassLength
385
489
  end
386
490
  end
@@ -4,6 +4,9 @@ require "English"
4
4
  require_relative "executor/input_validator"
5
5
  require_relative "executor/graph_manager"
6
6
  require_relative "executor/retry_manager"
7
+ # Before compensation_manager: its NEVER_STARTED_ERROR_CLASSES names
8
+ # StepCoordination at load time, while `class Executor` does not exist yet.
9
+ require_relative "executor/step_coordination"
7
10
  require_relative "executor/compensation_manager"
8
11
  require_relative "executor/result_handler"
9
12
  require_relative "executor/async_step_dispatch"
@@ -118,6 +121,7 @@ module RubyReactor
118
121
  return finalize_halt(halted)
119
122
  end
120
123
 
124
+ @context.admit!
121
125
  @context.status = :running
122
126
  save_context
123
127
 
@@ -137,12 +141,15 @@ module RubyReactor
137
141
  RubyReactor::RateLimitRegistry::UnknownLimitError,
138
142
  RubyReactor::OrderedLock::WaitError => e
139
143
  @contention_snooze = true
140
- raise e
141
- rescue Error::AsyncResultPending
142
- # Only reachable when this executor runs nested inside a worker (a
143
- # composed child; sync callers never park). Propagate to the ROOT
144
- # resume, which owns the park. This child's own lock/semaphore (if any)
145
- # ARE released below and re-competed for on redelivery.
144
+ raise composed_contention_park(e) || e
145
+ rescue Error::ExecutionParked
146
+ # A park signal from a step of this run (its contention, or a wait on a
147
+ # background result), reaching a composed child's or a map element's
148
+ # first run. Every executor on the stack keeps its OWN lock/semaphore
149
+ # through the gap and re-adopts it on redelivery (005 D-A2); the worker
150
+ # at the top requeues once, after all of them have saved. A synchronous
151
+ # caller never sees a park signal: nothing raises one outside a worker.
152
+ park_held_primitives! if @context.inline_async_execution
146
153
  @contention_snooze = true
147
154
  raise
148
155
  rescue StandardError => e
@@ -151,7 +158,7 @@ module RubyReactor
151
158
  completed = true
152
159
  @result
153
160
  ensure
154
- release_locks
161
+ release_locks unless @parked
155
162
  leave_ordered_lock_scope
156
163
  save_context if persist_context? && !skip_context_persist?
157
164
 
@@ -226,6 +233,10 @@ module RubyReactor
226
233
  return finalize_halt(halted)
227
234
  end
228
235
 
236
+ # Past every reactor-level gate. Idempotent for a genuine resume, which
237
+ # was admitted on its first run (or, saved before `admitted` existed,
238
+ # is marked now).
239
+ @context.admit!
229
240
  prepare_for_resume
230
241
  save_context
231
242
 
@@ -247,12 +258,14 @@ module RubyReactor
247
258
  RubyReactor::RateLimitRegistry::UnknownLimitError,
248
259
  RubyReactor::OrderedLock::WaitError => e
249
260
  @contention_snooze = true
250
- raise e
251
- rescue Error::AsyncResultPending => e
252
- # An awaited async unit is not terminal yet: park. Exclusive lock and
253
- # semaphore stay HELD (recorded on the context for the resuming job to
254
- # re-adopt); the worker snoozes the job. The context lock is still
255
- # released below — the redelivered job must be able to take it.
261
+ raise composed_contention_park(e) || e
262
+ rescue Error::ExecutionParked => e
263
+ # A step of this run parked (contention, or an awaited background result
264
+ # not terminal yet) — here, or in a composed child that already parked
265
+ # its own holds on the way through. Exclusive lock and semaphore stay
266
+ # HELD (recorded on the context for the resuming job to re-adopt); the
267
+ # worker snoozes the job. The context lock is still released below — the
268
+ # redelivered job must be able to take it.
256
269
  park_held_primitives!
257
270
  @contention_snooze = true
258
271
  raise e
@@ -370,8 +383,15 @@ module RubyReactor
370
383
  # `execute` for sync reactors, the first `resume_execution` pass for async
371
384
  # reactors. Genuine resumes never re-check (a paused reactor must not block
372
385
  # itself on resume).
386
+ #
387
+ # At most once per execution: a lock or semaphore contended right after
388
+ # the charge snoozes the job BEFORE admission, and its redelivery is still
389
+ # a first run. The `rate_limit_charged` marker rides `private_data`, so
390
+ # that redelivery skips the charge but still runs every other first-run
391
+ # gate (the period re-check above all).
373
392
  def check_rate_limit
374
393
  return unless @reactor_class.respond_to?(:rate_limit_config) && @reactor_class.rate_limit_config
394
+ return if @context.private_data[:rate_limit_charged] || @context.private_data["rate_limit_charged"]
375
395
 
376
396
  config = @reactor_class.rate_limit_config
377
397
 
@@ -386,14 +406,53 @@ module RubyReactor
386
406
  end
387
407
 
388
408
  RubyReactor::RateLimit.new(key_base, limits: limits).check_and_increment!
409
+ @context.private_data[:rate_limit_charged] = true
389
410
  end
390
411
 
391
- # True when nothing has run yet for this context — the very first execution
392
- # of the reactor, including an async reactor's first worker pass. A genuine
393
- # resume (paused, async-handed-off, or retried step) always records a
394
- # `current_step` before serializing, so it is never mistaken for a first run.
412
+ # True when this execution has not yet passed its reactor-level gates —
413
+ # the very first execution, including an async reactor's first worker
414
+ # pass. Read from the explicit `admitted` marker, so a park at any depth
415
+ # (which unwinds `with_step` and clears `current_step`) can never make a
416
+ # redelivery look fresh and re-charge its rate limit (005 R-03). The old
417
+ # inference is AND-ed in so a context saved before the marker existed
418
+ # still resumes as it did.
395
419
  def first_execution?
396
- @context.current_step.nil? && @context.intermediate_results.empty?
420
+ !@context.admitted? && @context.current_step.nil? && @context.intermediate_results.empty?
421
+ end
422
+
423
+ # A composed child's own reactor-level contention inside a worker. A
424
+ # root's reaches `Worker#perform`, which snoozes the job; a child's would
425
+ # first reach the step that composed it, which turns any error into an
426
+ # ordinary step failure (F3). So the child raises a park signal instead,
427
+ # and every executor above it keeps its holds until the worker requeues.
428
+ #
429
+ # Bounded where it is raised, like a step's contention park (005 R-05):
430
+ # the child counts its own parks, and past `lock_snooze_max_attempts` the
431
+ # contention error goes through as that step's failure, so the parent
432
+ # rolls back and releases its holds — which the worker's snooze
433
+ # escalation would do neither of. Returns nil to raise `error` unchanged.
434
+ def composed_contention_park(error)
435
+ return nil unless @context.inline_async_execution && @context.root_context
436
+ # A configuration mistake, not contention: it stays a permanent failure.
437
+ return nil if error.is_a?(RubyReactor::RateLimitRegistry::UnknownLimitError)
438
+ return nil if within_inline_map_element?
439
+
440
+ parks = (@context.private_data[:admission_parks] || @context.private_data["admission_parks"]).to_i + 1
441
+ @context.private_data[:admission_parks] = parks
442
+ max = RubyReactor.configuration.lock_snooze_max_attempts
443
+ return nil if max != :infinity && parks > max
444
+
445
+ Error::ReactorContentionPark.new(error)
446
+ end
447
+
448
+ # An inline (non-fan-out) map element has no persisted context of its own:
449
+ # a park re-runs the whole map step with fresh element contexts, repeating
450
+ # the elements that already finished and restarting the count above (005
451
+ # R-01, "Known, not changed"). Under one, contention stays a failure.
452
+ def within_inline_map_element?
453
+ ctx = @context
454
+ ctx = ctx.parent_context until ctx.nil? || (ctx.map_metadata && ctx.root_context)
455
+ !ctx.nil?
397
456
  end
398
457
 
399
458
  # Record and persist a Halt result, then return it. Shared by the
@@ -612,7 +671,7 @@ module RubyReactor
612
671
  if @acquired_semaphore
613
672
  key = @acquired_semaphore.key
614
673
  release_one("semaphore", @acquired_semaphore)
615
- held_lock_keys.delete(key)
674
+ pop_held_lock_key(key)
616
675
  middlewares.on(:semaphore_released, key, @context)
617
676
  end
618
677
  @acquired_semaphore = nil
@@ -621,11 +680,22 @@ module RubyReactor
621
680
 
622
681
  key = @acquired_lock.key
623
682
  release_one("lock", @acquired_lock)
624
- held_lock_keys.delete(key)
683
+ pop_held_lock_key(key)
625
684
  @acquired_lock = nil
626
685
  middlewares.on(:lock_released, key, @context)
627
686
  end
628
687
 
688
+ # Pop a SINGLE occurrence of `key`, not every occurrence (Finding 1).
689
+ # With a reactor and a step both holding K, the step's release must not
690
+ # erase the reactor's still-open entry — the async deadlock guard reads
691
+ # this registry and would stop seeing K held while the reactor's own
692
+ # hold is still live. `StepCoordination#pop_key` mirrors this exactly.
693
+ def pop_held_lock_key(key)
694
+ keys = held_lock_keys
695
+ idx = keys.index(key)
696
+ keys.delete_at(idx) if idx
697
+ end
698
+
629
699
  # Exclusive keys this EXECUTION currently holds, recorded on the root
630
700
  # context so a dispatching step anywhere in the tree can see the whole
631
701
  # chain. Read by the async_reactor deadlock guard; nothing else
@@ -61,7 +61,15 @@ module RubyReactor
61
61
  return if check_fail_fast?(arguments, storage)
62
62
 
63
63
  executor = Executor.new(context.reactor_class, {}, context)
64
- arguments[:serialized_context] ? executor.resume_execution : executor.execute
64
+ begin
65
+ arguments[:serialized_context] ? executor.resume_execution : executor.execute
66
+ rescue Error::ExecutionParked => e
67
+ # The element parked (a step's contention, or an awaited background
68
+ # result) and its executor already parked its own holds on the
69
+ # context: requeue the element with that context, like `Worker`
70
+ # snoozes a reactor. Not finished — no result, no counter decrement.
71
+ return requeue_parked_element(arguments, context, e)
72
+ end
65
73
 
66
74
  result = executor.result
67
75
 
@@ -75,6 +83,20 @@ module RubyReactor
75
83
  finalize_execution(arguments, storage)
76
84
  end
77
85
 
86
+ def self.requeue_parked_element(arguments, context, error)
87
+ context.middlewares&.on(:before_async_enqueue, context)
88
+ config = RubyReactor.configuration
89
+ config.async_router.perform_map_element_in(
90
+ Worker.snooze_delay(config, error),
91
+ map_id: arguments[:map_id], element_id: arguments[:element_id], index: arguments[:index],
92
+ serialized_inputs: arguments[:serialized_inputs], reactor_class_info: arguments[:reactor_class_info],
93
+ strict_ordering: arguments[:strict_ordering], parent_context_id: arguments[:parent_context_id],
94
+ parent_reactor_class_name: arguments[:parent_reactor_class_name], step_name: arguments[:step_name],
95
+ batch_size: arguments[:batch_size], serialized_context: ContextSerializer.serialize(context),
96
+ fail_fast: arguments[:fail_fast]
97
+ )
98
+ end
99
+
78
100
  def self.load_parent_context(arguments, reactor_class_name, storage)
79
101
  parent_context_data = storage.retrieve_context(arguments[:parent_context_id], reactor_class_name)
80
102
  parent_reactor_class = Object.const_get(reactor_class_name)
@@ -204,7 +226,7 @@ module RubyReactor
204
226
  RubyReactor::Map::Dispatcher.perform(next_batch_args)
205
227
  end
206
228
 
207
- private_class_method :load_parent_context, :trigger_next_batch_if_needed
229
+ private_class_method :load_parent_context, :trigger_next_batch_if_needed, :requeue_parked_element
208
230
  end
209
231
  end
210
232
  end
@@ -90,6 +90,9 @@ module RubyReactor
90
90
  executor.middlewares.on(:failed_reactor, parent_context.reactor_class.name, failure_response,
91
91
  parent_context)
92
92
  end
93
+ # This branch runs no executor loop, so nothing else persists the
94
+ # failed status: store it here.
95
+ store_parent(parent_context, storage)
93
96
  else
94
97
  parent_context.set_result(step_name_sym, final_result.value)
95
98
 
@@ -116,20 +119,41 @@ module RubyReactor
116
119
  # `after:` target here in the collector instead of the original
117
120
  # dispatching worker.
118
121
  parent_context.inline_async_execution = true
119
- executor.resume_execution
122
+ resume_parked_aware(executor, parent_context)
120
123
  end
124
+ end
125
+
126
+ # `resume_execution` persists the parent itself — under the parent's
127
+ # context lock, and deliberately NOT when it lost that lock to a live
128
+ # duplicate or replayed an already-terminal run. Storing again here, after
129
+ # the lock is released, would overwrite whatever the lock's holder wrote
130
+ # (single writer), so the resume's own save is the only one.
131
+ #
132
+ # This collector is a worker running the parent's execution, so it is
133
+ # also a final handler for park signals (005 R-01), like `Worker` and
134
+ # `ElementExecutor`: the resume has already parked the parent's holds and
135
+ # saved; hand the rest back to the parent's own worker.
136
+ #
137
+ # The parent is loaded from its own blob, so it has no `root_context`:
138
+ # a map inside a composed child resumes and requeues that child as its
139
+ # own execution, never its root. ponytail: the root is never resumed
140
+ # after such a map (already so on main); see "Fan-out map inside a
141
+ # composed child" in specs/future_improvements.md.
142
+ def resume_parked_aware(executor, parent_context)
143
+ executor.resume_execution
144
+ rescue RubyReactor::Error::ExecutionParked => e
145
+ config = RubyReactor.configuration
146
+ config.async_router.perform_in(
147
+ RubyReactor::Worker.snooze_delay(config, e), parent_context.context_id,
148
+ RubyReactor.reactor_storage_name(parent_context.reactor_class)
149
+ )
150
+ end
121
151
 
122
- # Checkpoint the ROOT, not the sub (F9/C2). When the map is embedded in a
123
- # composed sub-reactor, parent_context is the *sub*; storing only the sub
124
- # would leave the root blob stale and a rehydrate-by-root-id resume would
125
- # lose the map's completion. Resolve the root (which embeds the sub's
126
- # post-map state via composed_contexts) and store that. For a top-level
127
- # map parent_context IS the root, so this is unchanged.
128
- root = parent_context.root_context || parent_context
152
+ def store_parent(parent_context, storage)
129
153
  storage.store_context(
130
- root.context_id,
131
- ContextSerializer.serialize(root),
132
- RubyReactor.reactor_storage_name(root.reactor_class)
154
+ parent_context.context_id,
155
+ ContextSerializer.serialize(parent_context),
156
+ RubyReactor.reactor_storage_name(parent_context.reactor_class)
133
157
  )
134
158
  end
135
159
  end
@@ -7,12 +7,12 @@ module RubyReactor
7
7
  # rubocop:disable Metrics/ParameterLists
8
8
  def initialize(message, step:, attempts:, original_error: nil,
9
9
  inputs: {}, backtrace: nil, redact_inputs: [],
10
- reactor_name: nil, step_arguments: {}, validation_errors: nil)
10
+ reactor_name: nil, step_arguments: {}, validation_errors: nil, rollback_failures: nil)
11
11
  # rubocop:enable Metrics/ParameterLists
12
12
  super(message,
13
13
  step_name: step, inputs: inputs, backtrace: backtrace,
14
14
  redact_inputs: redact_inputs, reactor_name: reactor_name, step_arguments: step_arguments,
15
- validation_errors: validation_errors)
15
+ validation_errors: validation_errors, rollback_failures: rollback_failures)
16
16
  @attempts = attempts
17
17
  @original_error = original_error
18
18
  end
@@ -185,6 +185,24 @@ module RubyReactor
185
185
  span.finish
186
186
  end
187
187
 
188
+ # The step's attempt ended in a park (its contention, or an awaited
189
+ # background result): not a failure, and the redelivery opens a new span.
190
+ # Closed here, since neither `complete_step` nor `failed_step` fires for
191
+ # it and the span is keyed by step name.
192
+ def on_snooze_step(step_name, error, _context)
193
+ token = @step_tokens.delete(step_name)
194
+ ::OpenTelemetry::Context.detach(token) if token
195
+ @retry_errors.delete(step_name)
196
+
197
+ span = @step_spans.delete(step_name)
198
+ return unless span
199
+
200
+ span.set_attribute("step.status", "parked")
201
+ span.set_attribute("step.park_reason", error.class.name)
202
+ span.status = ::OpenTelemetry::Trace::Status.ok
203
+ span.finish
204
+ end
205
+
188
206
  def on_retry_attempt(step_name, attempt, error, _context)
189
207
  return unless defined?(::OpenTelemetry)
190
208
 
@@ -371,63 +389,82 @@ module RubyReactor
371
389
  RubyReactor.configuration.logger.warn("Telemetry context injection failed: #{e.message}")
372
390
  end
373
391
 
374
- def on_lock_acquired(key, _context)
392
+ def on_lock_acquired(key, context)
375
393
  span = @reactor_span
376
394
  return unless span
377
395
 
378
- span.set_attribute("reactor.lock.key", key.to_s)
379
- span.add_event("lock_acquired", attributes: { "lock.key" => key.to_s })
396
+ # The reactor span's own attributes describe the REACTOR's hold; a step's
397
+ # is recorded on its event only, so it can never overwrite them.
398
+ span.set_attribute("reactor.lock.key", key.to_s) unless coordinating_step(context)
399
+ span.add_event("lock_acquired", attributes: step_attributes(context).merge("lock.key" => key.to_s))
380
400
  end
381
401
 
382
- def on_lock_released(key, _context)
402
+ def on_lock_released(key, context)
383
403
  span = @reactor_span
384
404
  return unless span
385
405
 
386
- span.add_event("lock_released", attributes: { "lock.key" => key.to_s })
406
+ span.add_event("lock_released", attributes: step_attributes(context).merge("lock.key" => key.to_s))
387
407
  end
388
408
 
389
- def on_lock_failed(key, error, _context)
409
+ def on_lock_failed(key, error, context)
390
410
  span = @reactor_span
391
411
  return unless span
392
412
 
393
- span.add_event("lock_acquisition_failed", attributes: {
394
- "lock.key" => key.to_s,
395
- "error.message" => error.message,
396
- "error.class" => error.class.name
397
- })
413
+ span.add_event("lock_acquisition_failed", attributes: step_attributes(context).merge(
414
+ "lock.key" => key.to_s,
415
+ "error.message" => error.message,
416
+ "error.class" => error.class.name
417
+ ))
398
418
  end
399
419
 
400
- def on_semaphore_acquired(key, limit, _context)
420
+ def on_semaphore_acquired(key, limit, context)
401
421
  span = @reactor_span
402
422
  return unless span
403
423
 
404
- span.set_attribute("reactor.semaphore.key", key.to_s)
405
- span.set_attribute("reactor.semaphore.limit", limit.to_i)
424
+ unless coordinating_step(context)
425
+ span.set_attribute("reactor.semaphore.key", key.to_s)
426
+ span.set_attribute("reactor.semaphore.limit", limit.to_i)
427
+ end
406
428
  span.add_event("semaphore_acquired",
407
- attributes: { "semaphore.key" => key.to_s, "semaphore.limit" => limit.to_i })
429
+ attributes: step_attributes(context).merge("semaphore.key" => key.to_s,
430
+ "semaphore.limit" => limit.to_i))
408
431
  end
409
432
 
410
- def on_semaphore_released(key, _context)
433
+ def on_semaphore_released(key, context)
411
434
  span = @reactor_span
412
435
  return unless span
413
436
 
414
- span.add_event("semaphore_released", attributes: { "semaphore.key" => key.to_s })
437
+ span.add_event("semaphore_released", attributes: step_attributes(context).merge("semaphore.key" => key.to_s))
415
438
  end
416
439
 
417
- def on_semaphore_failed(key, limit, error, _context)
440
+ def on_semaphore_failed(key, limit, error, context)
418
441
  span = @reactor_span
419
442
  return unless span
420
443
 
421
- span.add_event("semaphore_acquisition_failed", attributes: {
422
- "semaphore.key" => key.to_s,
423
- "semaphore.limit" => limit.to_i,
424
- "error.message" => error.message,
425
- "error.class" => error.class.name
426
- })
444
+ span.add_event("semaphore_acquisition_failed", attributes: step_attributes(context).merge(
445
+ "semaphore.key" => key.to_s,
446
+ "semaphore.limit" => limit.to_i,
447
+ "error.message" => error.message,
448
+ "error.class" => error.class.name
449
+ ))
427
450
  end
428
451
 
429
452
  private
430
453
 
454
+ # US7/FR-028: attribute a step-level hold to the step that took it, on
455
+ # each EVENT (several steps can coordinate within one reactor span).
456
+ # `coordinating_step` is set by `StepCoordination` around its own hooks
457
+ # only — NOT `current_step`, which is also set while a resumed reactor
458
+ # re-takes its own reactor-level holds.
459
+ def coordinating_step(context)
460
+ context.respond_to?(:coordinating_step) ? context.coordinating_step : nil
461
+ end
462
+
463
+ def step_attributes(context)
464
+ step = coordinating_step(context)
465
+ step ? { "ruby_reactor.step" => step.to_s } : {}
466
+ end
467
+
431
468
  def extract_context(context)
432
469
  return nil unless defined?(::OpenTelemetry)
433
470