ruby_reactor 0.8.2 → 0.8.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.claude/skills/speckit-review/SKILL.md +324 -0
- data/.release-please-manifest.json +1 -1
- data/.specify/extensions.yml +10 -0
- data/.specify/feature.json +1 -1
- data/.specify/workflows/speckit/workflow.yml +13 -1
- data/.specify/workflows/workflow-registry.json +2 -2
- data/CHANGELOG.md +82 -0
- data/CLAUDE.md +2 -2
- data/README.md +35 -2
- data/lib/ruby_reactor/adapters/active_job/router.rb +19 -0
- data/lib/ruby_reactor/adapters/sidekiq/router.rb +21 -0
- data/lib/ruby_reactor/context.rb +26 -0
- data/lib/ruby_reactor/context_serializer.rb +4 -2
- data/lib/ruby_reactor/dsl/interrupt_builder.rb +14 -0
- data/lib/ruby_reactor/dsl/lockable.rb +76 -21
- data/lib/ruby_reactor/dsl/step_builder.rb +112 -1
- data/lib/ruby_reactor/error/async_result_pending.rb +1 -1
- data/lib/ruby_reactor/error/execution_parked.rb +16 -0
- data/lib/ruby_reactor/error/reactor_contention_park.rb +26 -0
- data/lib/ruby_reactor/error/step_contention_park.rb +26 -0
- data/lib/ruby_reactor/executor/async_step_dispatch.rb +109 -3
- data/lib/ruby_reactor/executor/compensation_manager.rb +99 -17
- data/lib/ruby_reactor/executor/ordered_lock_support.rb +76 -44
- data/lib/ruby_reactor/executor/result_handler.rb +31 -11
- data/lib/ruby_reactor/executor/retry_manager.rb +9 -1
- data/lib/ruby_reactor/executor/step_coordination.rb +788 -0
- data/lib/ruby_reactor/executor/step_executor.rb +115 -11
- data/lib/ruby_reactor/executor.rb +90 -20
- data/lib/ruby_reactor/map/element_executor.rb +24 -2
- data/lib/ruby_reactor/map/helpers.rb +35 -11
- data/lib/ruby_reactor/max_retries_exhausted_failure.rb +2 -2
- data/lib/ruby_reactor/open_telemetry.rb +61 -24
- data/lib/ruby_reactor/retry_context.rb +31 -2
- data/lib/ruby_reactor/rspec/helpers.rb +15 -0
- data/lib/ruby_reactor/rspec/matchers.rb +92 -0
- data/lib/ruby_reactor/rspec/test_subject.rb +7 -1
- data/lib/ruby_reactor/step/async_reactor_step.rb +40 -24
- data/lib/ruby_reactor/step/compose_step.rb +14 -3
- data/lib/ruby_reactor/step.rb +49 -7
- data/lib/ruby_reactor/step_sweeper.rb +29 -1
- data/lib/ruby_reactor/step_worker.rb +260 -37
- data/lib/ruby_reactor/version.rb +1 -1
- data/lib/ruby_reactor/web/api.rb +72 -7
- data/lib/ruby_reactor/web/coordination_serializer.rb +120 -2
- data/lib/ruby_reactor/web/public/assets/{index-Dw4KV4QY.js → index-CeZU-ESu.js} +9 -9
- data/lib/ruby_reactor/web/public/index.html +1 -1
- data/lib/ruby_reactor/worker.rb +56 -30
- data/lib/ruby_reactor.rb +27 -5
- data/specs/future_improvements.md +250 -0
- metadata +8 -28
- data/specs/002-step-input-contracts/checklists/requirements.md +0 -49
- data/specs/002-step-input-contracts/contracts/dsl-surface.md +0 -193
- data/specs/002-step-input-contracts/data-model.md +0 -115
- data/specs/002-step-input-contracts/plan.md +0 -165
- data/specs/002-step-input-contracts/quickstart.md +0 -170
- data/specs/002-step-input-contracts/research.md +0 -233
- data/specs/002-step-input-contracts/spec.md +0 -359
- data/specs/002-step-input-contracts/tasks.md +0 -367
- data/specs/004-inheritable-step-class/checklists/requirements.md +0 -40
- data/specs/004-inheritable-step-class/contracts/step-lifecycle.md +0 -85
- data/specs/004-inheritable-step-class/data-model.md +0 -116
- data/specs/004-inheritable-step-class/plan.md +0 -174
- data/specs/004-inheritable-step-class/quickstart.md +0 -112
- data/specs/004-inheritable-step-class/research.md +0 -308
- data/specs/004-inheritable-step-class/spec.md +0 -316
- data/specs/004-inheritable-step-class/tasks.md +0 -258
- data/specs/active_job.md +0 -259
- data/specs/deferred-003-step-lock-declarations/checklists/requirements.md +0 -51
- data/specs/deferred-003-step-lock-declarations/contracts/dsl-surface.md +0 -154
- data/specs/deferred-003-step-lock-declarations/data-model.md +0 -131
- data/specs/deferred-003-step-lock-declarations/plan.md +0 -166
- data/specs/deferred-003-step-lock-declarations/quickstart.md +0 -169
- data/specs/deferred-003-step-lock-declarations/research.md +0 -196
- data/specs/deferred-003-step-lock-declarations/spec.md +0 -447
- data/specs/deferred-003-step-lock-declarations/tasks.md +0 -572
- data/specs/possible_feature.md +0 -22
|
@@ -11,7 +11,23 @@ module RubyReactor
|
|
|
11
11
|
# load-bearing — the record is written BEFORE the signal, so a reader that
|
|
12
12
|
# misses the (at-most-once) signal still finds the answer on its next
|
|
13
13
|
# fallback re-check.
|
|
14
|
+
#
|
|
15
|
+
# Single writer: a context is written only by the execution that owns it,
|
|
16
|
+
# and this unit is not the parent's execution. It reads the parent's context
|
|
17
|
+
# and NEVER writes it back — the parent may be saving newer progress at this
|
|
18
|
+
# very moment, and an older snapshot written over it would revert that. The
|
|
19
|
+
# parent holds only the link written at dispatch (`:async_step_ref`); every
|
|
20
|
+
# fact about this unit — its run (arguments, attempts), its park state and
|
|
21
|
+
# its outcome — lives on its own Step Result Record, where the dashboard
|
|
22
|
+
# rebuilds it from the link. A body's changes to `context` are therefore
|
|
23
|
+
# local to this job and never persisted.
|
|
24
|
+
# rubocop:disable Metrics/ClassLength
|
|
14
25
|
class StepWorker
|
|
26
|
+
# Grace added to a park's stamped window, so ordinary queue latency on the
|
|
27
|
+
# redelivery does not make `StepSweeper` mistake a parked unit for a lost
|
|
28
|
+
# one (see `#mark_record_parked`).
|
|
29
|
+
PARK_SWEEP_GRACE = 30
|
|
30
|
+
|
|
15
31
|
class << self
|
|
16
32
|
def perform(arguments)
|
|
17
33
|
arguments = arguments.transform_keys(&:to_sym)
|
|
@@ -25,16 +41,18 @@ module RubyReactor
|
|
|
25
41
|
root_context_id: arguments[:root_context_id],
|
|
26
42
|
reactor_class_name: arguments[:reactor_class_name],
|
|
27
43
|
step_context_id: arguments[:step_context_id],
|
|
28
|
-
step_name: arguments[:step_name].to_sym
|
|
44
|
+
step_name: arguments[:step_name].to_sym,
|
|
45
|
+
contention_attempts: arguments[:contention_attempts] || 0
|
|
29
46
|
}
|
|
30
47
|
end
|
|
31
48
|
end
|
|
32
49
|
|
|
33
|
-
def initialize(root_context_id:, reactor_class_name:, step_context_id:, step_name:)
|
|
50
|
+
def initialize(root_context_id:, reactor_class_name:, step_context_id:, step_name:, contention_attempts: 0)
|
|
34
51
|
@root_context_id = root_context_id
|
|
35
52
|
@reactor_class_name = reactor_class_name
|
|
36
53
|
@step_context_id = step_context_id || root_context_id
|
|
37
54
|
@step_name = step_name
|
|
55
|
+
@contention_attempts = contention_attempts.to_i
|
|
38
56
|
end
|
|
39
57
|
|
|
40
58
|
# The lock is what makes a lost unit recoverable: the record alone cannot say
|
|
@@ -61,8 +79,15 @@ module RubyReactor
|
|
|
61
79
|
context.reactor_class&.validate_definition!
|
|
62
80
|
step_config = context.reactor_class&.steps&.[](@step_name)
|
|
63
81
|
return record_missing_step unless step_config
|
|
82
|
+
return if already_completed?(context)
|
|
64
83
|
|
|
65
84
|
complete(run_step(context, step_config), context)
|
|
85
|
+
rescue Executor::StepCoordination::Contended => e
|
|
86
|
+
handle_contention(e, context)
|
|
87
|
+
rescue Executor::StepCoordination::KeyError => e
|
|
88
|
+
log(:error, "failed", error: "#{e.class}: #{e.message}")
|
|
89
|
+
complete(RubyReactor.Failure(e, step_name: @step_name, reactor_name: @reactor_class_name, retryable: false),
|
|
90
|
+
context)
|
|
66
91
|
rescue StandardError => e
|
|
67
92
|
# The unit's failure belongs in its record, where a reader can see it.
|
|
68
93
|
# Raising instead would hand the job to the backend's retry machinery to
|
|
@@ -71,6 +96,126 @@ module RubyReactor
|
|
|
71
96
|
complete(RubyReactor.Failure(e, step_name: @step_name, reactor_name: @reactor_class_name), nil)
|
|
72
97
|
end
|
|
73
98
|
|
|
99
|
+
# Finding 6: `async_step` has no delayed re-enqueue of its own, so a
|
|
100
|
+
# contended step parks the same way a step-level contention park does
|
|
101
|
+
# elsewhere — reschedule via `perform_step_in`, bounded by
|
|
102
|
+
# `lock_snooze_max_attempts` (the `OrderedLock::WaitError` exemption
|
|
103
|
+
# mirrors `Worker#handle_snooze`). WITHOUT calling `complete`: the Step
|
|
104
|
+
# Result Record stays "dispatched" so a reader keeps waiting instead of
|
|
105
|
+
# seeing a phantom terminal state.
|
|
106
|
+
def handle_contention(contended, context = nil)
|
|
107
|
+
# `run_step`'s loop counted this round as an attempt before the body
|
|
108
|
+
# raised; a contention park is not a retry attempt, so give it back —
|
|
109
|
+
# mirrors `StepExecutor#handle_contention`. Left inflated, the count
|
|
110
|
+
# persists across redeliveries and `StepCoordination#retry_pending?`
|
|
111
|
+
# eventually reads the step as out of retries, advancing its
|
|
112
|
+
# ordered-lock position out from under an attempt still to come.
|
|
113
|
+
context&.retry_context&.decrement_attempt_for_step(@step_name)
|
|
114
|
+
|
|
115
|
+
config = RubyReactor.configuration
|
|
116
|
+
attempt = @contention_attempts + 1
|
|
117
|
+
uncapped = contended.original.is_a?(RubyReactor::OrderedLock::WaitError)
|
|
118
|
+
|
|
119
|
+
if !uncapped && config.lock_snooze_max_attempts != :infinity && attempt > config.lock_snooze_max_attempts
|
|
120
|
+
log(:warn, "contention_exhausted", key: contended.key, attempt: attempt)
|
|
121
|
+
# Terminal: an earlier park kept this step's ordered-lock position for
|
|
122
|
+
# a redelivery that is no longer coming — advance it, or every later
|
|
123
|
+
# position stalls until its poison pill. `complete` then writes a
|
|
124
|
+
# fresh terminal record, which drops the position and waiting marker.
|
|
125
|
+
Executor::StepCoordination.discard_parked_state!(context) if context
|
|
126
|
+
complete(
|
|
127
|
+
RubyReactor::Failure(
|
|
128
|
+
"async_step :#{@step_name} gave up on #{contended.primitive} '#{contended.key}' after " \
|
|
129
|
+
"#{attempt} contention attempts",
|
|
130
|
+
step_name: @step_name, reactor_name: @reactor_class_name, retryable: false,
|
|
131
|
+
exception_class: contended.original.class.name
|
|
132
|
+
), context
|
|
133
|
+
)
|
|
134
|
+
return
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
delay = RubyReactor::Worker.snooze_delay(config, contended)
|
|
138
|
+
log(:info, "parked", key: contended.key, primitive: contended.primitive, attempt: attempt, delay: delay)
|
|
139
|
+
mark_record_parked(context, delay, attempt, contended)
|
|
140
|
+
RubyReactor.configuration.async_router.perform_step_in(
|
|
141
|
+
delay, root_context_id: @root_context_id, reactor_class_name: @reactor_class_name,
|
|
142
|
+
step_context_id: @step_context_id, step_name: @step_name, contention_attempts: attempt
|
|
143
|
+
)
|
|
144
|
+
end
|
|
145
|
+
|
|
146
|
+
# The liveness lock only drops a CONCURRENT duplicate. A parked unit whose
|
|
147
|
+
# stamped window lapsed can be swept while its scheduled redelivery is
|
|
148
|
+
# merely late, so the two deliveries can run one after the other and repeat
|
|
149
|
+
# the side effect. The record is the durable answer: once it is terminal,
|
|
150
|
+
# the body must not run again.
|
|
151
|
+
#
|
|
152
|
+
# ponytail: the record check, not a durable delivery lease — it closes the
|
|
153
|
+
# repeat-after-completion case. Two deliveries that both arrive while the
|
|
154
|
+
# unit is still only parked still both run the body (serialized by the
|
|
155
|
+
# step's own coordination). Add a lease if that shows up in practice.
|
|
156
|
+
def already_completed?(context)
|
|
157
|
+
record = storage.retrieve_step_result(@step_context_id, @step_name, step_result_namespace(context))
|
|
158
|
+
return false unless record && record["status"] == "completed"
|
|
159
|
+
|
|
160
|
+
log(:info, "duplicate_dropped")
|
|
161
|
+
true
|
|
162
|
+
end
|
|
163
|
+
|
|
164
|
+
# A park releases the liveness lock (`perform`'s ensure) and leaves the
|
|
165
|
+
# Step Result Record at "dispatched" — which is EXACTLY the shape
|
|
166
|
+
# `StepSweeper` reads as "this unit's job was lost". Left unmarked it
|
|
167
|
+
# re-dispatches immediately, and when the parked redelivery then fires the
|
|
168
|
+
# body runs a second time: two jobs, sequential, so the liveness lock
|
|
169
|
+
# (which only drops CONCURRENT duplicates) never sees them collide.
|
|
170
|
+
# Stamping the window the redelivery is due in lets the sweeper tell
|
|
171
|
+
# parked from lost.
|
|
172
|
+
#
|
|
173
|
+
# ponytail: a fixed grace covers ordinary queue latency; a park whose
|
|
174
|
+
# redelivery is lost is recovered one grace period late rather than never.
|
|
175
|
+
# Make it configurable only if real queue lag exceeds it.
|
|
176
|
+
#
|
|
177
|
+
# The record is also where the unit's OWN park state lives (005 R-09):
|
|
178
|
+
# its ordered-lock position (`load_step_context` restores it on the
|
|
179
|
+
# redelivery) and what it waits on (the dashboard's "waiting"). Never the
|
|
180
|
+
# parent's root blob: the parent may be checkpointing newer progress right
|
|
181
|
+
# now, and this worker is not its writer (F5).
|
|
182
|
+
def mark_record_parked(context, delay, attempt, contended)
|
|
183
|
+
namespace = step_result_namespace(context)
|
|
184
|
+
record = storage.retrieve_step_result(@step_context_id, @step_name, namespace)
|
|
185
|
+
return unless record
|
|
186
|
+
|
|
187
|
+
record["parked_until"] = (Time.now + delay + PARK_SWEEP_GRACE).iso8601
|
|
188
|
+
# The counter lives in the job payload, which a sweeper re-dispatch
|
|
189
|
+
# rebuilds from the record alone — without it a swept park restarts at
|
|
190
|
+
# zero and `lock_snooze_max_attempts` never bites.
|
|
191
|
+
record["contention_attempts"] = attempt
|
|
192
|
+
position = ordered_lock_position(context)
|
|
193
|
+
record["ordered_lock"] = position if position
|
|
194
|
+
record["waiting"] = { "step" => @step_name, "primitive" => contended.primitive, "key" => contended.key,
|
|
195
|
+
"attempts" => attempt }
|
|
196
|
+
record.merge!(run_fields)
|
|
197
|
+
storage.store_step_result(@step_context_id, @step_name, record, namespace)
|
|
198
|
+
rescue StandardError => e
|
|
199
|
+
RubyReactor.configuration.logger.warn(
|
|
200
|
+
"RubyReactor: async_step :#{@step_name} could not mark its record parked: #{e.message}"
|
|
201
|
+
)
|
|
202
|
+
end
|
|
203
|
+
|
|
204
|
+
def ordered_lock_position(context)
|
|
205
|
+
stash = context&.private_data&.[](:step_ordered_locks) || context&.private_data&.[]("step_ordered_locks")
|
|
206
|
+
return nil unless stash
|
|
207
|
+
|
|
208
|
+
stash[@step_name.to_s] || stash[@step_name.to_sym]
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
# Records are namespaced by the reactor that OWNS the step (what
|
|
212
|
+
# `AsyncStepDispatch#async_step_class_name` wrote them under), which for a
|
|
213
|
+
# composed child is not the root this job was handed.
|
|
214
|
+
def step_result_namespace(context)
|
|
215
|
+
owner = context&.reactor_class
|
|
216
|
+
owner ? RubyReactor.reactor_storage_name(owner) : @reactor_class_name
|
|
217
|
+
end
|
|
218
|
+
|
|
74
219
|
def acquire_liveness_lock
|
|
75
220
|
# Inline testing re-enters this frame synchronously, so the lock would
|
|
76
221
|
# self-contend; it only guards cross-process delivery, impossible inline.
|
|
@@ -93,37 +238,89 @@ module RubyReactor
|
|
|
93
238
|
end
|
|
94
239
|
|
|
95
240
|
def run_step(context, step_config)
|
|
241
|
+
# A suppressed step never coordinates (FR-012) — decided before the
|
|
242
|
+
# arguments are validated or any hold is taken, exactly as
|
|
243
|
+
# `StepExecutor#execute_step_sync` orders it. `complete` persists the
|
|
244
|
+
# nil result, so the reader sees the same skipped unit a same-process
|
|
245
|
+
# step would produce.
|
|
246
|
+
unless step_config.should_run?(context)
|
|
247
|
+
log(:info, "skipped")
|
|
248
|
+
return RubyReactor.Success(nil)
|
|
249
|
+
end
|
|
250
|
+
|
|
96
251
|
arguments = resolve_arguments(step_config, context)
|
|
252
|
+
# Reactor-side `argument`/`validate_args` rules gate the step BEFORE its
|
|
253
|
+
# coordination is acquired, exactly as `StepExecutor#execute_step_sync`
|
|
254
|
+
# orders them — an async_step must not take a lock (or spend a rate-limit
|
|
255
|
+
# slot) for arguments it is about to reject.
|
|
256
|
+
invalid = validate_arguments(step_config, arguments)
|
|
257
|
+
return invalid if invalid
|
|
258
|
+
|
|
97
259
|
log(:info, "running")
|
|
260
|
+
# What `StepExecutor#run_step_implementation` records as a `:run` trace
|
|
261
|
+
# entry for a same-process step, kept on this unit's record instead (see
|
|
262
|
+
# the class comment): the dashboard resolves the step's coordination key
|
|
263
|
+
# and its inspector's arguments from it.
|
|
264
|
+
contract = step_config.input_contract
|
|
265
|
+
@run = { "started_at" => Time.now.iso8601(6),
|
|
266
|
+
"arguments" => ContextSerializer.serialize_value(contract ? contract.redact(arguments) : arguments) }
|
|
98
267
|
|
|
99
268
|
attempt = 0
|
|
100
269
|
result = nil
|
|
101
270
|
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
271
|
+
# Under `with_step`, exactly as `StepExecutor#execute_step_sync` runs a
|
|
272
|
+
# same-process step: the coordination hooks this worker fires must be
|
|
273
|
+
# attributable to the step, which reads `context.current_step`.
|
|
274
|
+
context.with_step(@step_name) do
|
|
275
|
+
loop do
|
|
276
|
+
attempt += 1
|
|
277
|
+
@run["attempts"] = attempt
|
|
278
|
+
# Mirror the count onto the context: `StepCoordination` reads
|
|
279
|
+
# `retry_context` to decide whether an ordered-lock position should be
|
|
280
|
+
# held for a pending retry, and this worker is the one path that never
|
|
281
|
+
# goes through `RetryManager#prepare_retry_attempt`.
|
|
282
|
+
context.retry_context.increment_attempt_for_step(@step_name)
|
|
283
|
+
result = execute_step_body(step_config, arguments, context)
|
|
284
|
+
break unless retry?(step_config, result, attempt)
|
|
285
|
+
|
|
286
|
+
delay = backoff_delay(step_config, attempt)
|
|
287
|
+
log(:warn, "retrying", attempt: attempt, delay: delay)
|
|
288
|
+
sleep(delay)
|
|
289
|
+
end
|
|
110
290
|
end
|
|
111
291
|
|
|
112
292
|
result
|
|
113
293
|
end
|
|
114
294
|
|
|
295
|
+
# Same check and same structured, non-retryable shape `StepExecutor`
|
|
296
|
+
# produces — the same arguments fail the same rules on every attempt.
|
|
297
|
+
def validate_arguments(step_config, arguments)
|
|
298
|
+
return nil unless step_config.args_validator
|
|
299
|
+
|
|
300
|
+
validation_result = step_config.args_validator.call(arguments)
|
|
301
|
+
return nil if validation_result.success?
|
|
302
|
+
|
|
303
|
+
error = validation_result.error
|
|
304
|
+
error.step_name = @step_name
|
|
305
|
+
error.step_arguments = arguments
|
|
306
|
+
log(:warn, "invalid_arguments", error: "#{error.class}: #{error.message}")
|
|
307
|
+
RubyReactor.Failure(error, validation_errors: error.field_errors, step_name: @step_name,
|
|
308
|
+
step_arguments: arguments, reactor_name: @reactor_class_name,
|
|
309
|
+
retryable: false)
|
|
310
|
+
end
|
|
311
|
+
|
|
312
|
+
# The same single enforcement site `StepExecutor` uses
|
|
313
|
+
# (`StepCoordination.run_step`), so the worker path cannot drift from the
|
|
314
|
+
# in-process one. `Contended`/`KeyError` — this step's own — propagate
|
|
315
|
+
# unrescued: `perform_unit` handles them (park, or a non-retryable Failure).
|
|
115
316
|
def execute_step_body(step_config, arguments, context)
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
args = step_config.inline_contract.enforce!(args) if step_config.inline_contract
|
|
120
|
-
step_config.run_block.call(args, context)
|
|
121
|
-
elsif step_config.has_impl?
|
|
122
|
-
step_config.impl.run(arguments, context)
|
|
123
|
-
else
|
|
124
|
-
RubyReactor.Failure("Step '#{@step_name}' has no implementation")
|
|
125
|
-
end
|
|
317
|
+
unless step_config.has_run_block? || step_config.has_impl?
|
|
318
|
+
return RubyReactor.Failure("Step '#{@step_name}' has no implementation")
|
|
319
|
+
end
|
|
126
320
|
|
|
321
|
+
result = Executor::StepCoordination.run_step(step_config, arguments, context: context,
|
|
322
|
+
reactor_class: context.reactor_class,
|
|
323
|
+
middlewares: context.middlewares)
|
|
127
324
|
normalize(result)
|
|
128
325
|
rescue Error::InputValidationError => e
|
|
129
326
|
# Same shape the executor builds, and never retried: the same arguments
|
|
@@ -131,6 +328,8 @@ module RubyReactor
|
|
|
131
328
|
RubyReactor.Failure(e, validation_errors: e.field_errors, step_name: @step_name,
|
|
132
329
|
step_arguments: e.step_arguments || {}, reactor_name: @reactor_class_name,
|
|
133
330
|
retryable: false)
|
|
331
|
+
rescue Executor::StepCoordination::Contended, Executor::StepCoordination::KeyError
|
|
332
|
+
raise
|
|
134
333
|
rescue StandardError => e
|
|
135
334
|
RubyReactor.Failure(e, step_name: @step_name, reactor_name: @reactor_class_name)
|
|
136
335
|
end
|
|
@@ -182,14 +381,21 @@ module RubyReactor
|
|
|
182
381
|
record["signal"] = "halt"
|
|
183
382
|
record["reason"] = result.reason
|
|
184
383
|
end
|
|
185
|
-
|
|
384
|
+
record.merge!(run_fields)
|
|
385
|
+
# Namespaced by the reactor that OWNS the step — what
|
|
386
|
+
# `AsyncStepDispatch` wrote the `dispatched` record under and what the
|
|
387
|
+
# reader's `Template::Result` looks under. For an async_step inside a
|
|
388
|
+
# composed child that is the child, not the root name this job carries.
|
|
389
|
+
storage.store_step_result(@step_context_id, @step_name, record, step_result_namespace(context || @step_context))
|
|
186
390
|
log(result.success? ? :info : :warn, result.success? ? "completed" : "completed_with_failure")
|
|
187
391
|
storage.publish(RubyReactor.async_step_channel(@step_context_id, @step_name), "done")
|
|
188
392
|
result
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
393
|
+
end
|
|
394
|
+
|
|
395
|
+
# This delivery's run, once the body was reached: `started_at`,
|
|
396
|
+
# `arguments` (redacted, serialized) and `attempts`. Empty before that.
|
|
397
|
+
def run_fields
|
|
398
|
+
@run || {}
|
|
193
399
|
end
|
|
194
400
|
|
|
195
401
|
def record_missing_parent
|
|
@@ -219,16 +425,43 @@ module RubyReactor
|
|
|
219
425
|
return nil unless data
|
|
220
426
|
|
|
221
427
|
root = ContextSerializer.deserialize_hash(data)
|
|
222
|
-
@root_context = root
|
|
223
428
|
found = find_context(root, @step_context_id)
|
|
429
|
+
# Kept so the paths that deliberately pass no context to `complete`
|
|
430
|
+
# still write the record under the owning reactor's namespace.
|
|
431
|
+
@step_context = found
|
|
432
|
+
restore_parked_position(found) if found
|
|
224
433
|
# The step runs in its own job; nothing it reaches should hand off again.
|
|
225
434
|
found&.inline_async_execution = true
|
|
435
|
+
# Per-job owner, NEVER the root context id (US4-5, research D5):
|
|
436
|
+
# ownership never crosses a process hand-off, so this job's coordination
|
|
437
|
+
# never re-enters the dispatching execution's holds and never blocks a
|
|
438
|
+
# SECOND async_step dispatch on the same key from proceeding once this
|
|
439
|
+
# one releases. Derived, not random: `StepCoordination` can detach a
|
|
440
|
+
# lock across a contention park, and `perform_step_in` redelivers this
|
|
441
|
+
# same unit — a fresh uuid per redelivery would make `lock.reattach`
|
|
442
|
+
# fail against its own detached hold until the TTL expired.
|
|
443
|
+
found&.coordination_owner = "async_step:#{@step_context_id}:#{@step_name}"
|
|
444
|
+
# `ContextSerializer` does not carry `middlewares`, and nothing else in
|
|
445
|
+
# this worker builds them — without this the step's coordination hooks
|
|
446
|
+
# fire into an empty runner, contradicting "identical hooks in a worker".
|
|
447
|
+
found&.middlewares ||= Executor.middlewares_for(found.reactor_class)
|
|
226
448
|
found
|
|
227
449
|
rescue RubyReactor::Error::DeserializationError, RubyReactor::Error::SchemaVersionError => e
|
|
228
450
|
log(:error, "parent_context_unreadable", error: "#{e.class}: #{e.message}")
|
|
229
451
|
nil
|
|
230
452
|
end
|
|
231
453
|
|
|
454
|
+
# A parked unit's ordered-lock position lives on its own record, not in
|
|
455
|
+
# the parent's blob (see `mark_record_parked`). Put it back where
|
|
456
|
+
# `StepCoordination` looks, so the redelivery re-reads the SAME nonce.
|
|
457
|
+
def restore_parked_position(context)
|
|
458
|
+
record = storage.retrieve_step_result(@step_context_id, @step_name, step_result_namespace(context))
|
|
459
|
+
position = record && record["ordered_lock"]
|
|
460
|
+
return unless position
|
|
461
|
+
|
|
462
|
+
(context.private_data[:step_ordered_locks] ||= {})[@step_name.to_s] = position.transform_keys(&:to_sym)
|
|
463
|
+
end
|
|
464
|
+
|
|
232
465
|
def find_context(context, target_id)
|
|
233
466
|
return context if context.context_id == target_id
|
|
234
467
|
|
|
@@ -241,17 +474,6 @@ module RubyReactor
|
|
|
241
474
|
nil
|
|
242
475
|
end
|
|
243
476
|
|
|
244
|
-
def save_root(_context)
|
|
245
|
-
return unless @root_context
|
|
246
|
-
|
|
247
|
-
storage.store_context(@root_context.context_id, ContextSerializer.serialize(@root_context),
|
|
248
|
-
@reactor_class_name)
|
|
249
|
-
rescue StandardError => e
|
|
250
|
-
RubyReactor.configuration.logger.warn(
|
|
251
|
-
"RubyReactor: async_step :#{@step_name} could not persist its parent context: #{e.message}"
|
|
252
|
-
)
|
|
253
|
-
end
|
|
254
|
-
|
|
255
477
|
def storage
|
|
256
478
|
RubyReactor.configuration.storage_adapter
|
|
257
479
|
end
|
|
@@ -272,4 +494,5 @@ module RubyReactor
|
|
|
272
494
|
)
|
|
273
495
|
end
|
|
274
496
|
end
|
|
497
|
+
# rubocop:enable Metrics/ClassLength
|
|
275
498
|
end
|
data/lib/ruby_reactor/version.rb
CHANGED
data/lib/ruby_reactor/web/api.rb
CHANGED
|
@@ -47,6 +47,9 @@ module RubyReactor
|
|
|
47
47
|
structure = self.class.build_structure(reactor_class) if reactor_class.respond_to?(:steps)
|
|
48
48
|
|
|
49
49
|
api_status = self.class.reactor_status(data)
|
|
50
|
+
class_name = data[:reactor_class]&.to_s
|
|
51
|
+
runs = self.class.async_step_runs(data[:composed_contexts], class_name)
|
|
52
|
+
trace = self.class.with_async_step_runs(data[:execution_trace] || [], runs)
|
|
50
53
|
|
|
51
54
|
response_data = {
|
|
52
55
|
id: data[:context_id],
|
|
@@ -55,7 +58,9 @@ module RubyReactor
|
|
|
55
58
|
current_step: data[:current_step].to_s,
|
|
56
59
|
retry_count: data[:retry_count] || 0,
|
|
57
60
|
undo_stack: data[:undo_stack] || [],
|
|
58
|
-
step_attempts:
|
|
61
|
+
step_attempts: self.class.with_async_step_attempts(
|
|
62
|
+
data.dig(:retry_context, :step_attempts) || {}, runs
|
|
63
|
+
),
|
|
59
64
|
created_at: data[:started_at],
|
|
60
65
|
inputs: data[:inputs],
|
|
61
66
|
intermediate_results: self.class.with_map_summaries(
|
|
@@ -65,15 +70,14 @@ module RubyReactor
|
|
|
65
70
|
# Once per reactor, never per step: the old per-step `async`
|
|
66
71
|
# field is gone because there is now exactly one hand-off point.
|
|
67
72
|
background_handoff: self.class.background_handoff_for(reactor_class),
|
|
68
|
-
steps:
|
|
69
|
-
composed_contexts: self.class.hydrate_composed_contexts(
|
|
70
|
-
data[:composed_contexts] || {},
|
|
71
|
-
data[:reactor_class]&.to_s
|
|
72
|
-
),
|
|
73
|
+
steps: trace,
|
|
74
|
+
composed_contexts: self.class.hydrate_composed_contexts(data[:composed_contexts] || {}, class_name),
|
|
73
75
|
coordination: CoordinationSerializer.build(
|
|
74
76
|
reactor_class,
|
|
75
77
|
inputs: data[:inputs],
|
|
76
|
-
context_id: data[:context_id]
|
|
78
|
+
context_id: data[:context_id],
|
|
79
|
+
execution_trace: trace,
|
|
80
|
+
private_data: data[:private_data] || {}
|
|
77
81
|
),
|
|
78
82
|
error: data[:failure_reason]
|
|
79
83
|
}
|
|
@@ -294,11 +298,72 @@ module RubyReactor
|
|
|
294
298
|
when "map_ref" then hydrate_map_ref(value, reactor_class_name)
|
|
295
299
|
when "async_step_ref" then hydrate_async_step_ref(value, reactor_class_name)
|
|
296
300
|
when "async_reactor_ref" then hydrate_async_reactor_ref(value)
|
|
301
|
+
when "composed" then hydrate_composed_child(value)
|
|
297
302
|
else value
|
|
298
303
|
end
|
|
299
304
|
end
|
|
300
305
|
end
|
|
301
306
|
|
|
307
|
+
# An inline `compose` child is embedded in the parent's blob, so its own
|
|
308
|
+
# `async_step` links are rebuilt the same way, one level down.
|
|
309
|
+
def self.hydrate_composed_child(value)
|
|
310
|
+
child = value[:context] || value["context"]
|
|
311
|
+
return value unless child.is_a?(RubyReactor::Context)
|
|
312
|
+
|
|
313
|
+
class_name = RubyReactor.reactor_storage_name(child.reactor_class)
|
|
314
|
+
runs = async_step_runs(child.composed_contexts, class_name)
|
|
315
|
+
view = child.to_h.merge(
|
|
316
|
+
execution_trace: with_async_step_runs(child.execution_trace, runs),
|
|
317
|
+
composed_contexts: hydrate_composed_contexts(child.composed_contexts, class_name)
|
|
318
|
+
)
|
|
319
|
+
value.merge(context: view)
|
|
320
|
+
end
|
|
321
|
+
|
|
322
|
+
# Single writer: an `async_step` unit never writes its parent, which
|
|
323
|
+
# holds only the `:async_step_ref` written at dispatch. The unit's run —
|
|
324
|
+
# arguments, attempts, when it started — is on its Step Result Record.
|
|
325
|
+
# `{ step_name => record }` for every linked unit that has started.
|
|
326
|
+
def self.async_step_runs(composed_contexts, reactor_class_name)
|
|
327
|
+
return {} unless composed_contexts.is_a?(Hash)
|
|
328
|
+
|
|
329
|
+
composed_contexts.each_with_object({}) do |(name, ref), runs|
|
|
330
|
+
next unless (ref[:type] || ref["type"]).to_s == "async_step_ref"
|
|
331
|
+
|
|
332
|
+
record = async_step_record(ref, reactor_class_name)
|
|
333
|
+
runs[name] = record if record&.key?("started_at")
|
|
334
|
+
end
|
|
335
|
+
end
|
|
336
|
+
|
|
337
|
+
def self.async_step_record(ref_data, reactor_class_name)
|
|
338
|
+
context_id = ref_data[:context_id] || ref_data["context_id"]
|
|
339
|
+
name = ref_data[:name] || ref_data["name"]
|
|
340
|
+
return nil unless context_id && name
|
|
341
|
+
|
|
342
|
+
RubyReactor.configuration.storage_adapter.retrieve_step_result(context_id, name, reactor_class_name)
|
|
343
|
+
end
|
|
344
|
+
|
|
345
|
+
# The parent's own trace, with each started unit's `:run` entry placed by
|
|
346
|
+
# its start time — the entry a same-process step writes for itself. A
|
|
347
|
+
# unit that already has one (a context saved before units stopped
|
|
348
|
+
# writing their parent) is left as it is.
|
|
349
|
+
def self.with_async_step_runs(trace, runs)
|
|
350
|
+
runs.each_with_object(trace.dup) do |(name, record), merged|
|
|
351
|
+
next if merged.any? { |e| (e[:type] || e["type"]).to_s == "run" && (e[:step] || e["step"]).to_s == name.to_s }
|
|
352
|
+
|
|
353
|
+
started = Time.iso8601(record["started_at"])
|
|
354
|
+
entry = { type: :run, step: name, timestamp: started, background: true,
|
|
355
|
+
arguments: ContextSerializer.deserialize_value(record["arguments"]) }
|
|
356
|
+
later = merged.index { |e| e[:timestamp].is_a?(Time) && e[:timestamp] > started }
|
|
357
|
+
merged.insert(later || merged.size, entry)
|
|
358
|
+
end
|
|
359
|
+
end
|
|
360
|
+
|
|
361
|
+
def self.with_async_step_attempts(step_attempts, runs)
|
|
362
|
+
runs.each_with_object(step_attempts.dup) do |(name, record), merged|
|
|
363
|
+
merged[name] = record["attempts"] if record["attempts"]
|
|
364
|
+
end
|
|
365
|
+
end
|
|
366
|
+
|
|
302
367
|
# The reference lives on the parent's context; the outcome lives in the
|
|
303
368
|
# Step Result Record. Resolve it so the dashboard can show whether the
|
|
304
369
|
# unit is still dispatched or has landed, mirroring hydrate_map_ref.
|
|
@@ -4,7 +4,7 @@ module RubyReactor
|
|
|
4
4
|
module Web
|
|
5
5
|
class CoordinationSerializer
|
|
6
6
|
class << self
|
|
7
|
-
def build(reactor_class, inputs:, context_id:)
|
|
7
|
+
def build(reactor_class, inputs:, context_id:, execution_trace: [], private_data: {})
|
|
8
8
|
return {} unless reactor_class
|
|
9
9
|
|
|
10
10
|
adapter = RubyReactor.configuration.storage_adapter
|
|
@@ -20,18 +20,122 @@ module RubyReactor
|
|
|
20
20
|
end
|
|
21
21
|
|
|
22
22
|
if reactor_class.rate_limit_config
|
|
23
|
-
result[:rate_limit] = build_rate_limit(reactor_class.rate_limit_config,
|
|
23
|
+
result[:rate_limit] = build_rate_limit(resolve_rate_limit(reactor_class.rate_limit_config),
|
|
24
|
+
normalized_inputs, adapter)
|
|
24
25
|
end
|
|
25
26
|
|
|
26
27
|
if reactor_class.period_config
|
|
27
28
|
result[:period] = build_period(reactor_class.period_config, normalized_inputs, adapter)
|
|
28
29
|
end
|
|
29
30
|
|
|
31
|
+
if reactor_class.respond_to?(:steps)
|
|
32
|
+
steps = build_steps(reactor_class, normalized_inputs, context_id, execution_trace, adapter)
|
|
33
|
+
result[:steps] = steps unless steps.empty?
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
waiting = private_data[:step_contention] || private_data["step_contention"] ||
|
|
37
|
+
async_step_waiting(reactor_class, context_id, adapter)
|
|
38
|
+
result[:waiting] = normalize_waiting(waiting) if waiting
|
|
39
|
+
|
|
30
40
|
result
|
|
31
41
|
end
|
|
32
42
|
|
|
33
43
|
private
|
|
34
44
|
|
|
45
|
+
# US7/FR-029: one row per coordinating step PER PRIMITIVE it declares,
|
|
46
|
+
# keyed to the step's OWN resolved arguments (from its latest `:run`
|
|
47
|
+
# trace entry), not the reactor's inputs. A step declaring both
|
|
48
|
+
# `with_lock` and `with_semaphore` is two rows — both gates are active,
|
|
49
|
+
# so an operator has to be able to see both. A step not yet reached is
|
|
50
|
+
# reported "pending" rather than omitted, so the list is stable.
|
|
51
|
+
def build_steps(reactor_class, inputs, context_id, execution_trace, adapter)
|
|
52
|
+
reactor_class.steps.flat_map do |name, step_config|
|
|
53
|
+
next [] unless step_config.respond_to?(:declares_coordination?) && step_config.declares_coordination?
|
|
54
|
+
|
|
55
|
+
build_step_entries(name, step_config, inputs, context_id, execution_trace, adapter)
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def build_step_entries(name, step_config, inputs, context_id, execution_trace, adapter) # rubocop:disable Metrics/ParameterLists
|
|
60
|
+
entry = latest_run_entry(execution_trace, name)
|
|
61
|
+
declarations = step_config.coordination_declarations
|
|
62
|
+
return [{ step: name.to_s, state: "pending" }] if declarations.empty?
|
|
63
|
+
|
|
64
|
+
# One PENDING row per declared primitive too, matching the reached
|
|
65
|
+
# shape below — a step declaring both a lock and a semaphore has two
|
|
66
|
+
# gates to wait on, and collapsing them to a single primitive-less
|
|
67
|
+
# row hides which.
|
|
68
|
+
if entry.nil?
|
|
69
|
+
return declarations.keys.map do |primitive|
|
|
70
|
+
{ step: name.to_s, primitive: primitive.to_s, state: "pending" }
|
|
71
|
+
end
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
# The trace records the RESOLVED arguments; the execution keyed off
|
|
75
|
+
# `coordination_arguments` of them, so the same function is applied
|
|
76
|
+
# here. A redacted value was never recorded, so a key computed from it
|
|
77
|
+
# would name a different Redis key than the one held — report it as
|
|
78
|
+
# unavailable instead of probing the wrong one.
|
|
79
|
+
traced = entry[:arguments] || entry["arguments"] || {}
|
|
80
|
+
if traced.is_a?(Hash) && traced.value?(RubyReactor::Step::InputContract::REDACTED)
|
|
81
|
+
return declarations.keys.map do |primitive|
|
|
82
|
+
{ step: name.to_s, primitive: primitive.to_s, key: nil,
|
|
83
|
+
key_error: "key unavailable: the step's arguments are redacted" }
|
|
84
|
+
end
|
|
85
|
+
end
|
|
86
|
+
args = step_config.coordination_arguments(traced.transform_keys(&:to_sym), inputs)
|
|
87
|
+
|
|
88
|
+
declarations.map do |primitive, config|
|
|
89
|
+
built = build_step_primitive(primitive, config, args, context_id, adapter)
|
|
90
|
+
{ step: name.to_s, primitive: primitive.to_s }.merge(built)
|
|
91
|
+
end
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
def build_step_primitive(primitive, config, args, context_id, adapter)
|
|
95
|
+
case primitive
|
|
96
|
+
when :lock then build_lock(config, args, context_id, adapter)
|
|
97
|
+
when :semaphore then build_semaphore(config, args, adapter)
|
|
98
|
+
when :rate_limit then build_rate_limit(resolve_rate_limit(config), args, adapter)
|
|
99
|
+
when :period then build_period(config, args, adapter)
|
|
100
|
+
else { key: resolve_key(config[:key_proc], args) }
|
|
101
|
+
end
|
|
102
|
+
rescue StandardError => e
|
|
103
|
+
{ key: nil, key_error: e.message }
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
def latest_run_entry(execution_trace, step_name)
|
|
107
|
+
Array(execution_trace).reverse_each.find do |e|
|
|
108
|
+
type = e[:type] || e["type"]
|
|
109
|
+
step = e[:step] || e["step"]
|
|
110
|
+
type.to_s == "run" && step.to_s == step_name.to_s
|
|
111
|
+
end
|
|
112
|
+
end
|
|
113
|
+
|
|
114
|
+
# A parked `async_step` keeps its "waiting on" marker on its own Step
|
|
115
|
+
# Result Record, never on the parent's context (005 R-09).
|
|
116
|
+
def async_step_waiting(reactor_class, context_id, adapter)
|
|
117
|
+
return nil unless context_id && reactor_class.respond_to?(:steps)
|
|
118
|
+
|
|
119
|
+
namespace = RubyReactor.reactor_storage_name(reactor_class)
|
|
120
|
+
reactor_class.steps.each do |name, step_config|
|
|
121
|
+
next unless step_config.respond_to?(:async_dispatch) && step_config.async_dispatch == :step
|
|
122
|
+
|
|
123
|
+
record = adapter.retrieve_step_result(context_id, name, namespace)
|
|
124
|
+
return record["waiting"] if record.is_a?(Hash) && record["waiting"]
|
|
125
|
+
end
|
|
126
|
+
nil
|
|
127
|
+
end
|
|
128
|
+
|
|
129
|
+
def normalize_waiting(waiting)
|
|
130
|
+
{
|
|
131
|
+
step: (waiting[:step] || waiting["step"]).to_s,
|
|
132
|
+
key: waiting[:key] || waiting["key"],
|
|
133
|
+
primitive: (waiting[:primitive] || waiting["primitive"]).to_s,
|
|
134
|
+
attempts: waiting[:attempts] || waiting["attempts"],
|
|
135
|
+
next_attempt_at: waiting[:next_attempt_at] || waiting["next_attempt_at"]
|
|
136
|
+
}
|
|
137
|
+
end
|
|
138
|
+
|
|
35
139
|
def normalize_inputs(inputs)
|
|
36
140
|
return {} unless inputs.is_a?(Hash)
|
|
37
141
|
|
|
@@ -134,6 +238,20 @@ module RubyReactor
|
|
|
134
238
|
}
|
|
135
239
|
end
|
|
136
240
|
|
|
241
|
+
# `with_rate_limit(:name)` stores only the name; the renderer needs the
|
|
242
|
+
# registered windows and the name-as-key the limiter actually uses
|
|
243
|
+
# (mirrors `StepCoordination#rate_limit_key_and_limits`).
|
|
244
|
+
def resolve_rate_limit(config)
|
|
245
|
+
return config unless config[:name]
|
|
246
|
+
|
|
247
|
+
name = config[:name]
|
|
248
|
+
{ limits: RubyReactor.configuration.rate_limits.fetch(name), key_proc: ->(_args) { name.to_s } }
|
|
249
|
+
rescue StandardError => e
|
|
250
|
+
# An unregistered name is a config mistake, not a reason for the
|
|
251
|
+
# dashboard to 500 — surface it as this row's key_error.
|
|
252
|
+
{ limits: [], key_proc: ->(_args) { raise e } }
|
|
253
|
+
end
|
|
254
|
+
|
|
137
255
|
def map_limits(limits)
|
|
138
256
|
Array(limits).map do |window|
|
|
139
257
|
{
|