ruby_reactor 0.8.4 → 0.8.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.release-please-manifest.json +1 -1
- data/.specify/feature.json +1 -1
- data/CHANGELOG.md +196 -0
- data/CLAUDE.md +1 -1
- data/README.md +47 -11
- data/lib/ruby_reactor/dsl/async_macros.rb +30 -1
- data/lib/ruby_reactor/dsl/async_reactor_builder.rb +12 -6
- data/lib/ruby_reactor/dsl/compose_builder.rb +12 -6
- data/lib/ruby_reactor/dsl/interrupt_builder.rb +1 -3
- data/lib/ruby_reactor/dsl/map_builder.rb +0 -2
- data/lib/ruby_reactor/dsl/step_builder.rb +91 -19
- data/lib/ruby_reactor/error/argument_resolution_error.rb +19 -0
- data/lib/ruby_reactor/error/rescuable.rb +28 -0
- data/lib/ruby_reactor/executor/compensation_manager.rb +30 -26
- data/lib/ruby_reactor/executor/result_handler.rb +18 -16
- data/lib/ruby_reactor/executor/step_coordination.rb +11 -8
- data/lib/ruby_reactor/executor/step_executor.rb +59 -49
- data/lib/ruby_reactor/executor.rb +38 -4
- data/lib/ruby_reactor/map/collector.rb +21 -11
- data/lib/ruby_reactor/map/dispatcher.rb +29 -3
- data/lib/ruby_reactor/map/element_executor.rb +9 -3
- data/lib/ruby_reactor/map/helpers.rb +32 -2
- data/lib/ruby_reactor/map/result_enumerator.rb +18 -12
- data/lib/ruby_reactor/reactor.rb +24 -0
- data/lib/ruby_reactor/rspec/matchers.rb +19 -3
- data/lib/ruby_reactor/step/compose_step.rb +7 -1
- data/lib/ruby_reactor/step/map_step.rb +109 -4
- data/lib/ruby_reactor/step.rb +7 -0
- data/lib/ruby_reactor/step_worker.rb +46 -22
- data/lib/ruby_reactor/storage/adapter.rb +4 -0
- data/lib/ruby_reactor/storage/redis_adapter.rb +9 -0
- data/lib/ruby_reactor/storage/redis_reactor_scan.rb +1 -1
- data/lib/ruby_reactor/version.rb +1 -1
- data/lib/ruby_reactor/web/api.rb +1 -1
- data/lib/ruby_reactor/web/public/assets/{index-CeZU-ESu.js → index-CQbgHtd0.js} +10 -10
- data/lib/ruby_reactor/web/public/index.html +1 -1
- data/lib/ruby_reactor/worker.rb +3 -1
- data/lib/ruby_reactor.rb +17 -6
- data/specs/007-execution-flow-analysis/analysis/README.md +147 -0
- data/specs/007-execution-flow-analysis/analysis/execution-order.md +359 -0
- data/specs/007-execution-flow-analysis/analysis/findings-and-options.md +502 -0
- data/specs/007-execution-flow-analysis/analysis/invariants.md +109 -0
- data/specs/007-execution-flow-analysis/checklists/requirements.md +39 -0
- data/specs/007-execution-flow-analysis/contracts/report-structure.md +71 -0
- data/specs/007-execution-flow-analysis/data-model.md +83 -0
- data/specs/007-execution-flow-analysis/evidence/harness.rb +229 -0
- data/specs/007-execution-flow-analysis/evidence/output.txt +333 -0
- data/specs/007-execution-flow-analysis/evidence/probes/01_plain.rb +122 -0
- data/specs/007-execution-flow-analysis/evidence/probes/02_compose.rb +182 -0
- data/specs/007-execution-flow-analysis/evidence/probes/03_map.rb +232 -0
- data/specs/007-execution-flow-analysis/evidence/probes/04_async.rb +132 -0
- data/specs/007-execution-flow-analysis/evidence/probes/05_background.rb +58 -0
- data/specs/007-execution-flow-analysis/evidence/probes/06_coordination.rb +158 -0
- data/specs/007-execution-flow-analysis/evidence/probes/07_interrupts_manual.rb +185 -0
- data/specs/007-execution-flow-analysis/evidence/run.rb +15 -0
- data/specs/007-execution-flow-analysis/plan.md +127 -0
- data/specs/007-execution-flow-analysis/quickstart.md +51 -0
- data/specs/007-execution-flow-analysis/research.md +202 -0
- data/specs/007-execution-flow-analysis/spec.md +270 -0
- data/specs/007-execution-flow-analysis/tasks.md +257 -0
- data/specs/008-rollback-reliability/checklists/requirements.md +43 -0
- data/specs/008-rollback-reliability/contracts/api-surface.md +126 -0
- data/specs/008-rollback-reliability/contracts/rollback-semantics.md +76 -0
- data/specs/008-rollback-reliability/data-model.md +139 -0
- data/specs/008-rollback-reliability/plan.md +233 -0
- data/specs/008-rollback-reliability/quickstart.md +105 -0
- data/specs/008-rollback-reliability/research.md +653 -0
- data/specs/008-rollback-reliability/spec.md +561 -0
- data/specs/008-rollback-reliability/tasks.md +1110 -0
- data/specs/future_improvements.md +48 -0
- metadata +35 -2
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# Interrupts, manual cancel/undo, failure kinds and crash re-drive:
|
|
4
|
+
# research H2 (worker), H5 contrast, H24 (pause), H26–H28.
|
|
5
|
+
|
|
6
|
+
module P
|
|
7
|
+
class Intr01 < Base
|
|
8
|
+
with_lock { |_inputs| "rk" }
|
|
9
|
+
pstep :a
|
|
10
|
+
interrupt(:approval) { wait_for :a }
|
|
11
|
+
pstep :c, after: :approval, fail: true
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
class Intr02 < Base
|
|
15
|
+
pstep :a
|
|
16
|
+
interrupt :approval do
|
|
17
|
+
wait_for :a
|
|
18
|
+
max_attempts 1
|
|
19
|
+
validate_payload { required(:ok).filled(:bool) }
|
|
20
|
+
end
|
|
21
|
+
pstep :c, after: :approval
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
class Intr03 < Base
|
|
25
|
+
pstep :a
|
|
26
|
+
interrupt(:approval) { wait_for :a }
|
|
27
|
+
pstep :c, after: :approval
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
class Intr04 < Base
|
|
31
|
+
with_lock { |_inputs| "rk" }
|
|
32
|
+
pstep :a
|
|
33
|
+
pstep :b, after: :a
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
# A worker-side step whose lock stays contended past lock_snooze_max_attempts.
|
|
37
|
+
class Edge01 < Base
|
|
38
|
+
background all: true
|
|
39
|
+
pstep :a
|
|
40
|
+
pstep(:b, after: :a) { with_lock { |_args| "sk" } }
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
class Crash < Exception; end # rubocop:disable Lint/InheritException
|
|
44
|
+
|
|
45
|
+
class Edge02 < Base
|
|
46
|
+
background all: true
|
|
47
|
+
pstep :a
|
|
48
|
+
pstep :b, after: :a do
|
|
49
|
+
run do |_inputs, _ctx|
|
|
50
|
+
Probe.rec("run:b")
|
|
51
|
+
# A killed worker is an interruption (Sidekiq::Shutdown < Interrupt);
|
|
52
|
+
# any other exception would be b's own failure and roll back (008 R-16).
|
|
53
|
+
raise Interrupt, "worker killed" if (Probe.counters[:crash] += 1) == 1
|
|
54
|
+
|
|
55
|
+
RubyReactor.Success("b")
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
pstep :c, after: :b
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
class Edge03 < Base
|
|
62
|
+
pstep :a
|
|
63
|
+
pstep :b, after: :a, fail: :raise do
|
|
64
|
+
run do |_inputs, _ctx|
|
|
65
|
+
Probe.rec("run:b")
|
|
66
|
+
raise Crash, "non-StandardError"
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
class Edge03b < Base
|
|
72
|
+
pstep :a
|
|
73
|
+
pstep :b, after: :a do
|
|
74
|
+
run do |_inputs, _ctx|
|
|
75
|
+
Probe.rec("run:b")
|
|
76
|
+
raise Interrupt, "signal"
|
|
77
|
+
end
|
|
78
|
+
end
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
class Edge05 < Base
|
|
82
|
+
pstep :a
|
|
83
|
+
pstep :b, after: :a do
|
|
84
|
+
inputs { input :x, :string }
|
|
85
|
+
argument :x, value(5)
|
|
86
|
+
end
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
class Edge06 < Base
|
|
90
|
+
input :n, :integer
|
|
91
|
+
pstep :a
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
def status_of(klass, id) = "status=#{klass.find(id).context.status}"
|
|
96
|
+
|
|
97
|
+
Probe.scenario "S-intr-01", "reactor with_lock(rk): a → interrupt → [continue] → c(fails)",
|
|
98
|
+
mode: :inline,
|
|
99
|
+
expected: %w[lock_acquired:rk run:a lock_released:lock:rk lock_acquired:rk run:c compensate:c undo:a
|
|
100
|
+
lock_released:lock:rk => failure(c)] do
|
|
101
|
+
paused = P::Intr01.run({})
|
|
102
|
+
Probe.note("first run => #{Probe.outcome(paused)}")
|
|
103
|
+
P::Intr01.continue(id: paused.execution_id, step_name: :approval, payload: { ok: true })
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
Probe.scenario "S-intr-02", "a → interrupt(validate, max_attempts 1) ← invalid payload",
|
|
107
|
+
mode: :inline, expected: %w[run:a undo:a => failure(approval)] do
|
|
108
|
+
paused = P::Intr02.run({})
|
|
109
|
+
P::Intr02.continue(id: paused.execution_id, step_name: :approval, payload: { ok: "nope" })
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
Probe.scenario "S-intr-03", "a → interrupt → Reactor.cancel",
|
|
113
|
+
mode: :inline, expected: %w[run:a => status=cancelled] do
|
|
114
|
+
paused = P::Intr03.run({})
|
|
115
|
+
P::Intr03.cancel(id: paused.execution_id, reason: "user gave up")
|
|
116
|
+
status_of(P::Intr03, paused.execution_id)
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
Probe.scenario "S-intr-04", "completed a → b; reactor lock rk held by another owner; Reactor.undo(id)",
|
|
120
|
+
mode: :inline, expected: %w[lock_acquired:rk run:a run:b lock_released:lock:rk undo:b undo:a
|
|
121
|
+
=> status=cancelled] do
|
|
122
|
+
done = P::Intr04.run({})
|
|
123
|
+
P.hold_lock("rk", "someone-else")
|
|
124
|
+
P::Intr04.undo(done.execution_id)
|
|
125
|
+
status_of(P::Intr04, done.execution_id)
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
Probe.scenario "S-edge-01", "background all: a → b(step lock held elsewhere; lock_snooze_max_attempts 2)",
|
|
129
|
+
mode: :worker, expected: %w[run:a undo:a => failure(b)] do
|
|
130
|
+
RubyReactor.configuration.lock_snooze_max_attempts = 2
|
|
131
|
+
P.hold_lock("sk", "someone-else")
|
|
132
|
+
Probe.run_async(P::Edge01)
|
|
133
|
+
ensure
|
|
134
|
+
RubyReactor.configuration.lock_snooze_max_attempts = 20
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
Probe.scenario "S-edge-02", "background all: a → b(worker crashes once) → c; job redelivered",
|
|
138
|
+
mode: :worker, expected: %w[run:a run:b crash run:b run:c => success] do
|
|
139
|
+
P::Edge02.run({})
|
|
140
|
+
job = RubyReactor::Adapters::Sidekiq::Worker.jobs.first.dup
|
|
141
|
+
begin
|
|
142
|
+
Probe.drain
|
|
143
|
+
rescue Interrupt
|
|
144
|
+
Probe.rec("crash")
|
|
145
|
+
end
|
|
146
|
+
RubyReactor::Adapters::Sidekiq::Worker.new.perform(*job["args"])
|
|
147
|
+
Probe.drain
|
|
148
|
+
P::Edge02.find(job["args"].first)
|
|
149
|
+
end
|
|
150
|
+
|
|
151
|
+
Probe.scenario "S-edge-03", "a → b(raises a non-StandardError Exception)",
|
|
152
|
+
mode: :inline, expected: %w[run:a run:b compensate:b undo:a => failure(b)] do
|
|
153
|
+
P::Edge03.run({}) # 008 R-16: the step's own failure, rolled back
|
|
154
|
+
end
|
|
155
|
+
|
|
156
|
+
Probe.scenario "S-edge-03b", "a → b(interrupted: raises Interrupt)",
|
|
157
|
+
mode: :inline, expected: %w[run:a run:b => raised(Interrupt)] do
|
|
158
|
+
reactor = P::Edge03b.new
|
|
159
|
+
reactor.run({})
|
|
160
|
+
rescue Interrupt
|
|
161
|
+
stored = RubyReactor.configuration.storage_adapter.retrieve_context(reactor.context.context_id, "P::Edge03b")
|
|
162
|
+
Probe.note("stored status=#{stored["status"]}") # 008 R-08/R-16: aborted, awaiting a manual undo
|
|
163
|
+
"raised(Interrupt)"
|
|
164
|
+
end
|
|
165
|
+
|
|
166
|
+
Probe.scenario "S-edge-04", "a → b(declares a where-condition)",
|
|
167
|
+
mode: :inline, expected: %w[=> raised(RubyReactor::Error::DeprecatedDslError)] do
|
|
168
|
+
# 008 R-15: `where`/`guard` are removed, so the class cannot be defined.
|
|
169
|
+
Class.new(P::Base) do
|
|
170
|
+
pstep :a
|
|
171
|
+
pstep(:b, after: :a) { where { |_ctx| true } }
|
|
172
|
+
end
|
|
173
|
+
rescue RubyReactor::Error::DeprecatedDslError
|
|
174
|
+
"raised(RubyReactor::Error::DeprecatedDslError)"
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
Probe.scenario "S-edge-05", "a → b(argument type check fails)",
|
|
178
|
+
mode: :inline, expected: %w[run:a undo:a => failure(b)] do
|
|
179
|
+
P::Edge05.run({})
|
|
180
|
+
end
|
|
181
|
+
|
|
182
|
+
Probe.scenario "S-edge-06", "reactor input validation fails",
|
|
183
|
+
mode: :inline, expected: %w[=> failure(?)] do
|
|
184
|
+
P::Edge06.run({ n: "not a number" })
|
|
185
|
+
end
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# bundle exec ruby specs/007-execution-flow-analysis/evidence/run.rb [| tee …/output.txt]
|
|
4
|
+
# PROBE=<substring> limits the run to matching scenario ids.
|
|
5
|
+
|
|
6
|
+
require_relative "harness"
|
|
7
|
+
|
|
8
|
+
puts "# Execution-flow probes — #{Time.now.utc.iso8601} — ruby_reactor #{RubyReactor::VERSION} " \
|
|
9
|
+
"(#{`git rev-parse --short HEAD`.strip})"
|
|
10
|
+
puts
|
|
11
|
+
|
|
12
|
+
Dir[File.join(__dir__, "probes", "*.rb")].each { |f| require f }
|
|
13
|
+
|
|
14
|
+
total = Probe.tally.values.sum
|
|
15
|
+
puts "#{total} scenarios, #{Probe.tally[:match]} match, #{Probe.tally[:mismatch]} mismatch"
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
# Implementation Plan: Execution Flow & Compensation Analysis
|
|
2
|
+
|
|
3
|
+
**Branch**: `execution_flow_analysis` | **Date**: 2026-09-26 | **Spec**: [spec.md](spec.md)
|
|
4
|
+
|
|
5
|
+
**Input**: Feature specification from `specs/007-execution-flow-analysis/spec.md`
|
|
6
|
+
|
|
7
|
+
## Summary
|
|
8
|
+
|
|
9
|
+
A documentation-only research project. It maps how RubyReactor orders forward execution and
|
|
10
|
+
rollback (compensate + undo) across plain steps, `compose`, `map` (inline and fan-out), `async_step`,
|
|
11
|
+
`async_reactor`, `background` reactors and interrupts, under locks, retries and every failure kind.
|
|
12
|
+
It answers the user's three open questions (map element rollback, rollback of earlier composed
|
|
13
|
+
reactors, map-level `compensate_all`/`compensate_each`).
|
|
14
|
+
|
|
15
|
+
Approach: read the executor source and cite it by `file:line`. Confirm each headline claim with a
|
|
16
|
+
small **probe**, a throwaway reactor run against the real test Redis through the real worker code
|
|
17
|
+
paths (Sidekiq fake mode + drain). Each probe prints its observed event sequence next to the
|
|
18
|
+
sequence the report claims. Deliverables are Markdown under this feature directory. Library code,
|
|
19
|
+
the test suite and the demo app are not touched.
|
|
20
|
+
|
|
21
|
+
## Technical Context
|
|
22
|
+
|
|
23
|
+
**Language/Version**: Markdown + Mermaid (report); Ruby >= 3.0 (probe scripts only, run against
|
|
24
|
+
`lib/` of this checkout)
|
|
25
|
+
|
|
26
|
+
**Primary Dependencies**: `ruby_reactor` (this checkout), `sidekiq/testing` fake mode, the
|
|
27
|
+
`RubyReactor::RSpec::SidekiqHelpers.drain_async_jobs` helper (plain-Ruby callable), `redis` gem
|
|
28
|
+
|
|
29
|
+
**Storage**: Test Redis at `redis://localhost:6780` (`RUBY_REACTOR_TEST_REDIS_URL` overrides), the
|
|
30
|
+
same instance `spec/spec_helper.rb` uses
|
|
31
|
+
|
|
32
|
+
**Testing**: Probes check themselves. Each scenario declares the expected event sequence and prints
|
|
33
|
+
`MATCH` / `MISMATCH` with both sequences. No RSpec files are added (FR-011).
|
|
34
|
+
|
|
35
|
+
**Target Platform**: Developer machine / CI shell with Redis reachable
|
|
36
|
+
|
|
37
|
+
**Project Type**: Library research (no runtime change)
|
|
38
|
+
|
|
39
|
+
**Performance Goals**: N/A. Full probe run should finish in under 1 minute, so it can be re-run
|
|
40
|
+
as behavior changes.
|
|
41
|
+
|
|
42
|
+
**Constraints**: Zero diff under `lib/`, `spec/`, `demo_app/`, `README.md`, `documentation/`
|
|
43
|
+
(SC-005). Probes must not sleep on real backoff (use `base_delay: 0`/tiny delays).
|
|
44
|
+
|
|
45
|
+
**Scale/Scope**: 6 constructs (step, compose, map, async_step, async_reactor, background reactor)
|
|
46
|
+
+ interrupts × failure positions (before / inside / after, nested) × modes (inline / worker) ×
|
|
47
|
+
cross-cutting conditions (reactor lock, step lock/semaphore, ordered lock, retries, failure
|
|
48
|
+
kinds). Roughly 40–60 matrix cells, 25–40 invariants.
|
|
49
|
+
|
|
50
|
+
## Constitution Check
|
|
51
|
+
|
|
52
|
+
*GATE: Must pass before Phase 0 research. Re-check after Phase 1 design.*
|
|
53
|
+
|
|
54
|
+
| Principle | Status | Note |
|
|
55
|
+
|---|---|---|
|
|
56
|
+
| I. Gem-First Design | PASS (N/A) | No `lib/` change. |
|
|
57
|
+
| II. Saga Pattern Integrity | PASS | The research exists to check this principle; findings feed later remedies. |
|
|
58
|
+
| III. Test-First with Real Infrastructure | PASS | No product code, so no Red-Green cycle. Probes use real Redis and real worker bodies, no mocks, matching the principle's intent. |
|
|
59
|
+
| IV. Observability by Default | PASS (N/A) | No runtime change. Probes observe through the shipped middleware hooks. |
|
|
60
|
+
| V. Simplicity & SemVer | PASS | No version bump, no API change. |
|
|
61
|
+
| VI. Demo-App Proof of Feature | N/A (justified) | Not a user-facing feature or public API change. No demo reactor/rake/spec. Any remedy chosen later carries its own demo. |
|
|
62
|
+
| Dev Workflow: docs task | PASS with deviation (see Complexity Tracking) | The documentation task here is an **audit**. Every README/documentation claim contradicted or left unstated by the findings is listed with `file:line` in the findings report. Editing those files is deferred to the remedy features, because behavior does not change here. |
|
|
63
|
+
|
|
64
|
+
- [x] Documentation impact identified: README.md sections "Error Handling and Compensation",
|
|
65
|
+
"async_step" / "Compensation is opt-in", "async_reactor"; `documentation/composition.md`
|
|
66
|
+
(`async_reactor` vs `compose` table); `documentation/background_and_async.md` ("Compensation
|
|
67
|
+
is opt-in", "Error Handling and Compensation"); `documentation/data_pipelines.md` (fail_fast);
|
|
68
|
+
`documentation/core_concepts.md` ("Compensation Order"); `documentation/DAG.md`. All are
|
|
69
|
+
**audited**, not edited (tasks.md carries the audit task).
|
|
70
|
+
|
|
71
|
+
**Post-design re-check (after Phase 1)**: unchanged. All gates PASS. The one deviation is tracked below.
|
|
72
|
+
|
|
73
|
+
## Project Structure
|
|
74
|
+
|
|
75
|
+
### Documentation (this feature)
|
|
76
|
+
|
|
77
|
+
```text
|
|
78
|
+
specs/007-execution-flow-analysis/
|
|
79
|
+
├── spec.md
|
|
80
|
+
├── plan.md # this file
|
|
81
|
+
├── research.md # Phase 0: method decisions + preliminary hypotheses
|
|
82
|
+
├── data-model.md # Phase 1: scenario / event / invariant / finding / option shapes
|
|
83
|
+
├── quickstart.md # Phase 1: how to run probes and validate the report
|
|
84
|
+
├── contracts/
|
|
85
|
+
│ └── report-structure.md # Phase 1: required sections + ID/label conventions
|
|
86
|
+
├── checklists/
|
|
87
|
+
│ └── requirements.md
|
|
88
|
+
├── tasks.md # Phase 2 (/speckit-tasks)
|
|
89
|
+
├── analysis/ # Deliverable (/speckit-implement)
|
|
90
|
+
│ ├── README.md # index, reading guide, answers to the 3 questions (US2)
|
|
91
|
+
│ ├── execution-order.md # construct lifecycles (FR-001) + order matrix (FR-002/003/012)
|
|
92
|
+
│ ├── invariants.md # invariants w/ status, evidence, test coverage (FR-006/008)
|
|
93
|
+
│ └── findings-and-options.md# findings + doc audit (FR-009) and options (FR-010)
|
|
94
|
+
└── evidence/ # Reproducible observations (FR-007, SC-003)
|
|
95
|
+
├── harness.rb # recorder middleware, scenario DSL, Redis/Sidekiq setup
|
|
96
|
+
├── probes/*.rb # one file per area: plain, compose, map, async, coordination, edge
|
|
97
|
+
├── run.rb # loads harness + probes, prints results
|
|
98
|
+
└── output.txt # captured run transcript cited by the report
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
### Source Code (repository root)
|
|
102
|
+
|
|
103
|
+
No source changes. Read-only inputs:
|
|
104
|
+
|
|
105
|
+
```text
|
|
106
|
+
lib/ruby_reactor/executor.rb # execute / resume_execution / lock lifetime
|
|
107
|
+
lib/ruby_reactor/executor/*.rb # step loop, retries, result handling, rollback, coordination
|
|
108
|
+
lib/ruby_reactor/step/{compose,map,async_reactor}_step.rb
|
|
109
|
+
lib/ruby_reactor/map/*.rb # dispatcher, element executor, collector, helpers
|
|
110
|
+
lib/ruby_reactor/step_worker.rb, worker.rb # async_step unit, reactor worker
|
|
111
|
+
lib/ruby_reactor/reactor.rb # run / continue / undo / cancel
|
|
112
|
+
lib/ruby_reactor/template/result.rb # async result read semantics
|
|
113
|
+
spec/** # mapped for invariant coverage (FR-008)
|
|
114
|
+
README.md, documentation/*.md # audited for claims (FR-009)
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
**Structure Decision**: Everything lives under `specs/007-execution-flow-analysis/`. The report
|
|
118
|
+
is split into four files by question type (what order? what always holds? what is wrong and what
|
|
119
|
+
could we do?), plus an index. Probes are grouped by area into a handful of files so each report
|
|
120
|
+
claim can cite `probe-id` and the transcript line.
|
|
121
|
+
|
|
122
|
+
## Complexity Tracking
|
|
123
|
+
|
|
124
|
+
| Violation | Why Needed | Simpler Alternative Rejected Because |
|
|
125
|
+
|-----------|------------|-------------------------------------|
|
|
126
|
+
| Docs task is an audit, not README/documentation edits | This feature changes no behavior. The research shows some current behavior may be unintended (e.g. map rollback is a `TODO`). | Writing current behavior into README/documentation now would present possibly buggy semantics as contract before the follow-up decision. Each remedy feature updates the docs itself. |
|
|
127
|
+
| Probe scripts (Ruby) inside a documentation-only feature | SC-003 requires reproducible observations, not only source reading. | Reading alone can't settle ordering in multi-process paths (collector, step worker). Adding RSpec files would change the test suite (FR-011). |
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# Quickstart: Reproducing & Validating the Analysis
|
|
2
|
+
|
|
3
|
+
## Prerequisites
|
|
4
|
+
|
|
5
|
+
- Ruby per `.tool-versions`, `bundle install` done at repo root.
|
|
6
|
+
- Test Redis reachable. Default `redis://localhost:6780`, the same one the spec suite uses:
|
|
7
|
+
|
|
8
|
+
```sh
|
|
9
|
+
docker run -d --name rr-test-redis -p 6780:6379 redis:7-alpine # if not already running
|
|
10
|
+
# or: export RUBY_REACTOR_TEST_REDIS_URL=redis://localhost:6379
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
Probes call `FLUSHDB` on that Redis between scenarios, like the spec suite's storage reset.
|
|
14
|
+
**Do not point them at a Redis holding data you care about.**
|
|
15
|
+
|
|
16
|
+
## Run the probes
|
|
17
|
+
|
|
18
|
+
```sh
|
|
19
|
+
bundle exec ruby specs/007-execution-flow-analysis/evidence/run.rb \
|
|
20
|
+
| tee specs/007-execution-flow-analysis/evidence/output.txt
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Optional filter: `PROBE=map bundle exec ruby …/run.rb` runs only scenario ids containing `map`.
|
|
24
|
+
|
|
25
|
+
**Expected outcome**: every block ends in `MATCH`. The tail line prints
|
|
26
|
+
`N scenarios, N match, 0 mismatch`. If a block says `MISMATCH`, behavior has drifted from the
|
|
27
|
+
report: fix the report (observation wins, see contracts/report-structure.md).
|
|
28
|
+
|
|
29
|
+
## Validate the report
|
|
30
|
+
|
|
31
|
+
```sh
|
|
32
|
+
cd specs/007-execution-flow-analysis
|
|
33
|
+
# 1. No unfilled matrix cells / placeholders (the report quotes the source's
|
|
34
|
+
# own `# TODO` comment in MapStep, so TODO is not a placeholder marker here)
|
|
35
|
+
grep -nE 'TBD|\?\?\?|\[fill' analysis/*.md # expect: no output
|
|
36
|
+
# 2. Every scenario id cited in the report resolves to a probe block
|
|
37
|
+
ID='S-(plain|compose|map|async|bg|lock|retry|intr|edge)-[0-9]+[a-z]?'
|
|
38
|
+
grep -ohE "$ID" analysis/*.md | sort -u > /tmp/cited
|
|
39
|
+
grep -oE "^== $ID" evidence/output.txt | sed 's/== //' | sort -u > /tmp/run
|
|
40
|
+
comm -23 /tmp/cited /tmp/run # expect: no output
|
|
41
|
+
# 3. Zero product diff
|
|
42
|
+
git diff --stat main -- lib spec demo_app README.md documentation # expect: empty
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## Scenarios that prove the headline answers
|
|
46
|
+
|
|
47
|
+
| Question | Scenario ids (see analysis/README.md) |
|
|
48
|
+
|---|---|
|
|
49
|
+
| Q1 map elements rollback | `S-map-*` |
|
|
50
|
+
| Q2 earlier composed reactors rollback | `S-compose-*` |
|
|
51
|
+
| Q3 map compensate_all / each gap | `S-map-*` + findings-and-options.md |
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
# Research: Execution Flow & Compensation Analysis
|
|
2
|
+
|
|
3
|
+
**Phase 0 output for** [plan.md](plan.md). No `NEEDS CLARIFICATION` remained in the Technical
|
|
4
|
+
Context. This file records the **method** decisions and the **preliminary hypotheses** that Phase 0
|
|
5
|
+
code reading produced. The hypotheses are the targets the implementation must confirm or refute
|
|
6
|
+
with evidence. They are not conclusions.
|
|
7
|
+
|
|
8
|
+
Baseline: commit `faf90e8d` (after #61 step-scoped retries and #63 inputs protection).
|
|
9
|
+
|
|
10
|
+
## Method decisions
|
|
11
|
+
|
|
12
|
+
### D1 — Evidence = source citations + executable probes against real Redis
|
|
13
|
+
|
|
14
|
+
- **Decision**: Each behavioral claim cites `file:line` [R]. Headline claims are also reproduced by
|
|
15
|
+
a probe [O] in `evidence/probes/`, run with Sidekiq in **fake** mode and drained through
|
|
16
|
+
`RubyReactor::RSpec::SidekiqHelpers.drain_async_jobs`. That runs the real `Worker`,
|
|
17
|
+
`MapElementWorker`, `MapCollectorWorker` and `StepWorker` bodies against the real test Redis.
|
|
18
|
+
- **Rationale**: Multi-process paths (collector resuming a parent, step worker writing records,
|
|
19
|
+
async reactor child) are hard to order by reading alone. Fake+drain is the same mechanism the
|
|
20
|
+
gem's own async specs use, so the observations match what the suite trusts.
|
|
21
|
+
- **Alternatives considered**: New RSpec files under `spec/` (rejected: FR-011, changes the test
|
|
22
|
+
suite). Mocked storage (rejected: Constitution III). Docker demo app (rejected: same code paths,
|
|
23
|
+
far slower to iterate). `Sidekiq::Testing.inline!` (rejected as the default: it re-enters jobs
|
|
24
|
+
synchronously inside the caller's frame and skips liveness locks, so it hides real ordering. Used
|
|
25
|
+
only where a probe needs it explicitly, and labelled).
|
|
26
|
+
|
|
27
|
+
### D2 — Trace capture: recorder middleware + step-body log
|
|
28
|
+
|
|
29
|
+
- **Decision**: One recorder collects, in order, (a) body events that probe steps write themselves
|
|
30
|
+
(`run:X`, `compensate:X`, `undo:X`), and (b) middleware events from a registered recording
|
|
31
|
+
middleware (`lock_acquired`, `lock_released`, `semaphore_*`, `retry_attempt`,
|
|
32
|
+
`start_compensation`, `start_undo`, `start_reactor`, `failed_reactor`, …). Probes also dump
|
|
33
|
+
`Failure#rollback_failures` and the context `execution_trace` where useful.
|
|
34
|
+
- **Rationale**: Body events show *what ran*. Middleware events show *where locks sit relative to
|
|
35
|
+
rollback*, which US3 AS2 asks for. Both come from shipped surfaces, with no patching.
|
|
36
|
+
- **Alternatives considered**: Monkey-patching `CompensationManager` (rejected: evidence must come
|
|
37
|
+
from the public surface). Reading only `execution_trace` (rejected: it omits lock events and
|
|
38
|
+
lives per context, so nested children would have to be stitched together).
|
|
39
|
+
|
|
40
|
+
### D3 — Execution modes covered
|
|
41
|
+
|
|
42
|
+
- **Decision**: inline (plain `Reactor.run`). Worker: `background all:` / `after:` / `before:`,
|
|
43
|
+
fan-out map, `async_step`, `async_reactor`, collector resume. Interrupt pause/continue.
|
|
44
|
+
The ActiveJob backend is **not** probed separately. Its adapters delegate to the same shared
|
|
45
|
+
bodies (`Worker`, `Map::ElementExecutor`, `Map::Collector`, `StepWorker`), and the report cites
|
|
46
|
+
that delegation [R].
|
|
47
|
+
- **Rationale**: Ordering is decided in the shared bodies. The adapters only enqueue and perform.
|
|
48
|
+
- **Alternatives considered**: Probing both backends (rejected: doubles runtime and adds no ordering
|
|
49
|
+
information).
|
|
50
|
+
|
|
51
|
+
### D4 — Vocabulary
|
|
52
|
+
|
|
53
|
+
- **Decision**: *compensate* = the failing step's own cleanup (receives the error).
|
|
54
|
+
*undo* = rollback of a previously completed step, walked in reverse completion order (receives
|
|
55
|
+
its result). *rollback* = both. *Left in place* = completed work that no rollback touches.
|
|
56
|
+
*Unit* = an `async_step` or `async_reactor` dispatch.
|
|
57
|
+
- **Rationale**: Matches `CompensationManager` and README usage, so readers can map the report
|
|
58
|
+
onto the code.
|
|
59
|
+
|
|
60
|
+
### D5 — Evidence labels
|
|
61
|
+
|
|
62
|
+
- **Decision**: `[R: path:line]` read in source. `[O: probe-id]` observed in
|
|
63
|
+
`evidence/output.txt`. `[T: spec/path:line]` covered by an existing spec. A claim with only [R]
|
|
64
|
+
is marked *by reading*.
|
|
65
|
+
|
|
66
|
+
### D6 — Invariant status
|
|
67
|
+
|
|
68
|
+
- **Decision**: `HOLDS` (evidence agrees, no counter-example found), `VIOLATED` (a reproducible
|
|
69
|
+
counter-example exists), `CONDITIONAL` (holds only under stated conditions, which are listed),
|
|
70
|
+
`UNDETERMINED` (evidence insufficient; says what would settle it).
|
|
71
|
+
|
|
72
|
+
### D7 — Finding severity
|
|
73
|
+
|
|
74
|
+
- **Decision**: **High**: completed side effects silently left in place on failure, a side effect
|
|
75
|
+
that can run twice, or rollback of work that never ran. **Medium**: the behavior is
|
|
76
|
+
defensible but differs by mode (inline vs worker) or contradicts documentation. **Low**:
|
|
77
|
+
clarity/naming. The behavior is predictable, but the DSL gives no hint of it.
|
|
78
|
+
|
|
79
|
+
### D8 — Deliverable layout
|
|
80
|
+
|
|
81
|
+
- **Decision**: See plan.md "Project Structure": four report files under `analysis/` plus
|
|
82
|
+
`evidence/`.
|
|
83
|
+
- **Rationale**: Readers come with one of three questions (what order? what always holds? what's
|
|
84
|
+
wrong?). One file per question keeps each scannable. The index carries the three direct
|
|
85
|
+
answers so SC-002 (under 5 minutes) holds.
|
|
86
|
+
|
|
87
|
+
### D9 — Documentation is audited, not edited
|
|
88
|
+
|
|
89
|
+
- **Decision**: Contradicted or missing claims are listed in `findings-and-options.md` with
|
|
90
|
+
`file:line` and the quoted text.
|
|
91
|
+
- **Rationale**: See plan.md Complexity Tracking.
|
|
92
|
+
|
|
93
|
+
### D10 — Coordination scope
|
|
94
|
+
|
|
95
|
+
- **Decision**: Reactor-level `with_lock` / `with_semaphore` / `with_ordered_lock` / rate
|
|
96
|
+
limit / period, and step-level `with_lock` / `with_semaphore` / `with_ordered_lock` /
|
|
97
|
+
`with_rate_limit` / `with_period`. Lock and semaphore are probed. Ordered lock, rate limit and
|
|
98
|
+
period are analysed by reading, with probes only where they create a distinct rollback path.
|
|
99
|
+
|
|
100
|
+
## Preliminary hypotheses (to confirm or refute in implementation)
|
|
101
|
+
|
|
102
|
+
Each hypothesis becomes a probe and/or an invariant. Source pointers are where reading found it.
|
|
103
|
+
|
|
104
|
+
### Plain steps
|
|
105
|
+
|
|
106
|
+
- **H1** Failure of step N: `compensate(N)`, then `undo(N-1 … 1)` in reverse completion order,
|
|
107
|
+
then no further steps run. `executor/compensation_manager.rb` `handle_step_failure`,
|
|
108
|
+
`rollback_completed_steps`.
|
|
109
|
+
- **H2** A step whose own lock/semaphore/rate-limit acquisition failed (never started) is **not**
|
|
110
|
+
compensated. Earlier steps are still undone. `compensation_manager.rb`
|
|
111
|
+
`NEVER_STARTED_ERROR_CLASSES`.
|
|
112
|
+
- **H3** A compensate that fails still lets prior undos run. The reactor then fails with a
|
|
113
|
+
`CompensationError`-derived message. An undo that fails does **not** stop the remaining undos.
|
|
114
|
+
Both are reported on `rollback_failures`.
|
|
115
|
+
- **H4** `Halt` stops without any rollback. `Skipped` steps never enter the undo stack.
|
|
116
|
+
- **H5** An exception that is **not** a `RubyReactor::Error::Base` and escapes outside a step body
|
|
117
|
+
(e.g. an argument `transform` or a `where`/guard raising) reaches
|
|
118
|
+
`ResultHandler#build_execution_failure`'s "unknown error" branch, which does **not** roll back.
|
|
119
|
+
Completed steps are left in place.
|
|
120
|
+
- **H6** Output-validation failure compensates the step (its side effect exists) and undoes prior
|
|
121
|
+
steps.
|
|
122
|
+
|
|
123
|
+
### Retries
|
|
124
|
+
|
|
125
|
+
- **H7** Retries run before any compensation. Compensation runs exactly once, after the last
|
|
126
|
+
attempt (`MaxRetriesExhaustedFailure`). No per-attempt compensation.
|
|
127
|
+
- **H8** Inline reactor: retry backoff `sleep`s in-process. In a worker (`background`, map
|
|
128
|
+
element): the retry is a re-enqueue (`RetryQueuedResult`), and compensation happens in whichever
|
|
129
|
+
delivery exhausts the attempts.
|
|
130
|
+
- **H9** `async_step` retries are in-worker loops (`StepWorker#run_step`). They never re-enqueue,
|
|
131
|
+
and exhaustion writes a failed record **without** calling the step's `compensate`.
|
|
132
|
+
|
|
133
|
+
### Compose
|
|
134
|
+
|
|
135
|
+
- **H10** Child failure: the child rolls itself back first (child compensate + child undos). Then
|
|
136
|
+
the parent treats the compose step as failed. `ComposeStep#compensate` runs, but is a no-op
|
|
137
|
+
because the child undo stack is already cleared, or because `current_step` is not the compose
|
|
138
|
+
name when the compensate runs outside `with_step`. Then the parent undoes its own earlier steps,
|
|
139
|
+
**including earlier compose steps**, whose `undo` replays the child's undo stack in reverse.
|
|
140
|
+
- **H11** A parent step failing **after** a completed compose undoes the compose, which undoes all
|
|
141
|
+
child steps in reverse. So yes, earlier composed reactors are rolled back.
|
|
142
|
+
- **H12** `compose` with `retries`: after a child failure the child's `intermediate_results` survive
|
|
143
|
+
the rollback. A retry *resumes* the child, and the child treats its already **undone** steps as
|
|
144
|
+
completed. Those steps are not re-run, and the retried step sees stale results.
|
|
145
|
+
Suspected High finding. Must be probed.
|
|
146
|
+
- **H13** A composed child's own reactor-level lock is re-entrant with the parent's (same root
|
|
147
|
+
owner).
|
|
148
|
+
|
|
149
|
+
### Map
|
|
150
|
+
|
|
151
|
+
- **H14** `MapStep#compensate` is a `TODO` returning `Success()`, and `MapStep` has no `undo`. So
|
|
152
|
+
(a) when element K fails fail-fast, elements 0..K-1 that already succeeded are **left in place**.
|
|
153
|
+
Only element K's own reactor rolls back its own steps. (b) When a step **after** the map fails,
|
|
154
|
+
the map's elements are **not** undone.
|
|
155
|
+
- **H15** `fail_fast false`: failed elements roll back individually (inside their own executor, or
|
|
156
|
+
via `executor.undo_all` in `ElementExecutor#handle_result`). Succeeded elements stay. The map
|
|
157
|
+
step itself succeeds and hands a `ResultEnumerator` to the collect block / dependants.
|
|
158
|
+
- **H16** Fan-out map, fail-fast: elements already dispatched keep running after the first failure
|
|
159
|
+
(only *not-yet-started* elements check `check_fail_fast?`). The collector fails the parent
|
|
160
|
+
using the first recorded failure, and later-finishing elements are neither compensated nor
|
|
161
|
+
reflected.
|
|
162
|
+
- **H17** Inline map: completed element contexts are not retained anywhere the parent could
|
|
163
|
+
reach to undo them. Even a working `MapStep#undo` would need new storage. Fan-out elements'
|
|
164
|
+
contexts are stored by id (`store_map_element_context_id`), so they *are* reachable.
|
|
165
|
+
- **H18** Map has no `retries` DSL. Retries are per inner step, inside each element.
|
|
166
|
+
|
|
167
|
+
### async_step / async_reactor
|
|
168
|
+
|
|
169
|
+
- **H19** A unit never enters the parent's undo stack. Parent rollback never touches it.
|
|
170
|
+
- **H20** A unit's failure does not fail the parent. A reader receives the `Failure` object as an
|
|
171
|
+
argument and must `fail!` explicitly. That fails the **reader** step: the reader's compensate
|
|
172
|
+
runs, then the parent's undos. The async_step's own `compensate`/`undo` blocks **never** run.
|
|
173
|
+
This contradicts `documentation/background_and_async.md` ("they run only if the failure is
|
|
174
|
+
surfaced into the parent's compensation path").
|
|
175
|
+
- **H21** An `async_reactor` child is an ordinary reactor. It rolls back its own steps on its own
|
|
176
|
+
failure, in its worker. The parent is not affected unless a reader opts in.
|
|
177
|
+
- **H22** Reader wait timeout (`AsyncWaitTimeoutError`) is an `Error::Base` raised during argument
|
|
178
|
+
resolution, so the parent rolls back its completed steps.
|
|
179
|
+
|
|
180
|
+
### background / worker
|
|
181
|
+
|
|
182
|
+
- **H23** `background after:/before:` serializes the undo stack. A worker-side failure undoes steps
|
|
183
|
+
that ran in the **calling** process too.
|
|
184
|
+
- **H24** Reactor-level lock/semaphore is held from admission to the executor's `ensure`, i.e.
|
|
185
|
+
**through** rollback. It is released on interrupt pause (not a park), and kept across parks.
|
|
186
|
+
- **H25** Step-level lock is released after the body and **re-acquired** for that step's
|
|
187
|
+
compensate/undo (`around_rollback`), with `rollback_wait`. There is a window between forward
|
|
188
|
+
release and rollback re-acquire.
|
|
189
|
+
|
|
190
|
+
### Interrupts / manual
|
|
191
|
+
|
|
192
|
+
- **H26** After a pause, the undo stack is persisted. A failure after `continue` undoes steps that
|
|
193
|
+
completed before the pause. An interrupt validation exhausting `max_attempts` calls `undo` and
|
|
194
|
+
marks the reactor failed.
|
|
195
|
+
- **H27** `Reactor.cancel` does not roll back. `Reactor.undo(id)` rolls back and cancels, and does
|
|
196
|
+
**not** take the reactor-level lock (step-level rollback locks still apply).
|
|
197
|
+
|
|
198
|
+
### Crash / re-drive
|
|
199
|
+
|
|
200
|
+
- **H28** Checkpoint after each successful step bounds a crash re-run to at most the one step in
|
|
201
|
+
flight, which can run twice (at-least-once). Its undo is recorded only if its result was
|
|
202
|
+
checkpointed.
|