flowlit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. flowlit/__init__.py +30 -0
  2. flowlit/application/__init__.py +7 -0
  3. flowlit/application/dto.py +100 -0
  4. flowlit/application/errors.py +119 -0
  5. flowlit/application/events.py +201 -0
  6. flowlit/application/executor_dispatcher.py +112 -0
  7. flowlit/application/executor_registry.py +26 -0
  8. flowlit/application/orchestrator.py +393 -0
  9. flowlit/application/ports/__init__.py +6 -0
  10. flowlit/application/ports/event_bus.py +31 -0
  11. flowlit/application/ports/execution_coordinator.py +45 -0
  12. flowlit/application/ports/executor.py +153 -0
  13. flowlit/application/ports/plan_parser.py +28 -0
  14. flowlit/application/ports/workflow_repository.py +24 -0
  15. flowlit/application/workflow_service.py +342 -0
  16. flowlit/domain/__init__.py +7 -0
  17. flowlit/domain/dag.py +99 -0
  18. flowlit/domain/errors.py +245 -0
  19. flowlit/domain/spec_resolution.py +232 -0
  20. flowlit/domain/step.py +333 -0
  21. flowlit/domain/workflow.py +746 -0
  22. flowlit/infrastructure/__init__.py +7 -0
  23. flowlit/infrastructure/event_bus/__init__.py +1 -0
  24. flowlit/infrastructure/event_bus/asyncio_event_bus.py +72 -0
  25. flowlit/infrastructure/execution/__init__.py +3 -0
  26. flowlit/infrastructure/execution/asyncio_execution_coordinator.py +140 -0
  27. flowlit/infrastructure/executors/__init__.py +11 -0
  28. flowlit/infrastructure/executors/base_executor.py +102 -0
  29. flowlit/infrastructure/executors/http_executor.py +191 -0
  30. flowlit/infrastructure/executors/noop_executor.py +40 -0
  31. flowlit/infrastructure/executors/shell_executor.py +133 -0
  32. flowlit/infrastructure/executors/sleep_executor.py +56 -0
  33. flowlit/infrastructure/repositories/__init__.py +1 -0
  34. flowlit/infrastructure/repositories/in_memory_repository.py +36 -0
  35. flowlit/infrastructure/yaml_loader/__init__.py +1 -0
  36. flowlit/infrastructure/yaml_loader/loader.py +103 -0
  37. flowlit/infrastructure/yaml_loader/schema.py +61 -0
  38. flowlit/interfaces/__init__.py +6 -0
  39. flowlit/interfaces/cli/__init__.py +1 -0
  40. flowlit/interfaces/cli/main.py +170 -0
  41. flowlit/interfaces/composition_root.py +92 -0
  42. flowlit/plugins.py +91 -0
  43. flowlit/py.typed +0 -0
  44. flowlit-0.1.0.dist-info/METADATA +198 -0
  45. flowlit-0.1.0.dist-info/RECORD +48 -0
  46. flowlit-0.1.0.dist-info/WHEEL +4 -0
  47. flowlit-0.1.0.dist-info/entry_points.txt +2 -0
  48. flowlit-0.1.0.dist-info/licenses/LICENSE +21 -0
flowlit/__init__.py ADDED
@@ -0,0 +1,30 @@
1
+ """Flowlit: a lightweight, event-driven workflow execution engine.
2
+
3
+ The codebase is organized as four layers, each only allowed to depend on
4
+ the ones below it:
5
+
6
+ interfaces (CLI, future MCP server -- thin adapters, no business logic)
7
+ v
8
+ infrastructure (YAML loading, in-memory event bus/repository, executors)
9
+ v
10
+ application (use cases, ports/Protocols, orchestration)
11
+ v
12
+ domain (Step/Workflow entities, DAG validation -- zero third-party deps)
13
+
14
+ `domain` and `application` never import from `infrastructure` or
15
+ `interfaces`. Wiring of concrete implementations happens in exactly one
16
+ place: `flowlit.interfaces.composition_root`.
17
+
18
+ Above all four layers sits one public extension seam, re-exported here
19
+ for ergonomics: `register_executor()` (see `flowlit.plugins`) lets code
20
+ that imports flowlit as a library add its own step-type executors
21
+ without forking flowlit's own source.
22
+ """
23
+
24
+ from flowlit.plugins import clear_registered_executors, register_executor, registered_executors
25
+
26
+ __all__ = [
27
+ "clear_registered_executors",
28
+ "register_executor",
29
+ "registered_executors",
30
+ ]
@@ -0,0 +1,7 @@
1
+ """Application layer: use cases, ports, and orchestration.
2
+
3
+ Depends only on `flowlit.domain` plus the standard library (`asyncio`,
4
+ `typing`). Defines the ports (`Protocol` classes) that infrastructure
5
+ adapters implement, and the `WorkflowService` use-case API that all
6
+ interface adapters (CLI, future MCP server) call through.
7
+ """
@@ -0,0 +1,100 @@
1
+ """Boundary DTOs returned by `WorkflowService`.
2
+
3
+ Plain, immutable, primitives-only dataclasses -- never a domain `Step` or
4
+ `Workflow`. Every interface adapter (CLI now, MCP server later) can
5
+ serialize these straight to JSON with `dataclasses.asdict()` without
6
+ importing anything from `flowlit.domain`.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from collections.abc import Mapping
12
+ from dataclasses import dataclass
13
+ from typing import Any
14
+
15
+
16
+ @dataclass(frozen=True)
17
+ class StepDTO:
18
+ id: str
19
+ type: str
20
+ status: str
21
+ spec: Mapping[str, Any]
22
+ """The *raw* spec, exactly as declared in the plan -- placeholders
23
+ (`${{ steps.<id>.output.<path> }}`) and all, unresolved. See
24
+ `resolved_spec` for the substituted values, once available."""
25
+ resolved_spec: Mapping[str, Any] | None
26
+ """The spec with every placeholder substituted, the same way
27
+ `StepDispatchRequested` carries it to the executor -- computed
28
+ on demand (via `Workflow.resolve_step_spec()`), not stored, so this
29
+ is always current as of *this* call, not a snapshot from dispatch
30
+ time. `None` if it can't be resolved (yet, or ever) -- most commonly
31
+ because the step is still `PENDING`, but also for e.g. a `SKIPPED`
32
+ step whose own spec referenced a dependency that never produced
33
+ output. See `flowlit.application.errors.OutputResolutionError` for
34
+ what "can't be resolved" actually means; a `None` here is never
35
+ itself an error to the caller, just "nothing to show yet."""
36
+ output: Mapping[str, Any] | None
37
+ error: str | None
38
+ optional: bool
39
+ blocking: bool
40
+ silent_failures: bool
41
+ when: str | None
42
+ """The raw `when:` condition, if this step has one -- see
43
+ `flowlit.domain.step.Step`. `None` means the step is unconditional,
44
+ not that its condition was falsy (a falsy condition shows up as
45
+ `status == "skipped"` instead)."""
46
+ when_not: str | None
47
+ """The raw `when_not:` condition, if this step has one -- negated
48
+ `when:`, same caveat about `None` not meaning "was truthy"."""
49
+ else_: frozenset[str] | None
50
+ """The step ids this step's `else:` names, if it has one -- true iff
51
+ every one of them ended up skipped by its own condition (not
52
+ cascade) -- see `skip_cause`."""
53
+ skip_cause: str | None
54
+ """Only meaningful when `status == "skipped"`, `None` otherwise:
55
+ `"condition"` if this step's own `when`/`when_not`/`else` evaluated
56
+ falsy (a deliberate, expected non-selection), `"cascade"` if it was
57
+ swept up by an upstream failure or an external `cancel_workflow()`
58
+ instead, never getting the chance to evaluate anything of its own --
59
+ see `flowlit.domain.step.SkipCause`."""
60
+ timeout_seconds: float | None
61
+ """This step's own `timeout_seconds`, if set -- see
62
+ `flowlit.domain.step.Step`. `None` means unbounded, not that a
63
+ timeout already happened (that shows up as `status == "failed"`,
64
+ with `error` saying so)."""
65
+ retry: int
66
+ """Additional attempts after the first, if any -- see
67
+ `flowlit.domain.step.Step`. `0` (the default) means no retry;
68
+ doesn't reflect how many attempts were actually made if the step is
69
+ still in flight or already terminal -- `error` says that, e.g.
70
+ "failed after 3 attempt(s): ..."."""
71
+ retry_backoff_seconds: float
72
+ """Delay before the first retry, doubling each attempt -- only
73
+ meaningful when `retry > 0`."""
74
+ started_at: str | None
75
+ """ISO 8601 (`datetime.isoformat()`), `None` if the step never ran
76
+ (e.g. it was skipped) -- see `flowlit.domain.step.Step`."""
77
+ ended_at: str | None
78
+ """ISO 8601, `None` until the step reaches a terminal status."""
79
+
80
+
81
+ @dataclass(frozen=True)
82
+ class WorkflowStatusDTO:
83
+ id: str
84
+ run_id: str
85
+ """This execution's identity, distinct from `id` (the plan's own
86
+ declared, run-independent identity) -- see
87
+ `flowlit.domain.workflow.Workflow.run_id`. What every step's
88
+ `${{ step.idempotency_key }}` is built from."""
89
+ concurrency_mode: str
90
+ """`"allow_parallel"` or `"singleton"` -- see
91
+ `flowlit.domain.workflow.ConcurrencyMode`."""
92
+ name: str
93
+ status: str
94
+ steps: tuple[StepDTO, ...]
95
+ created_at: str
96
+ """ISO 8601 -- always set; stamped once, in `Workflow.create()`."""
97
+ started_at: str | None
98
+ """ISO 8601, `None` until the workflow leaves PENDING."""
99
+ ended_at: str | None
100
+ """ISO 8601, `None` until the workflow reaches a terminal status."""
@@ -0,0 +1,119 @@
1
+ """Application-level errors.
2
+
3
+ Callers of `WorkflowService` (the CLI now, an MCP server later) only ever
4
+ need to catch these -- never a domain error, a pydantic `ValidationError`,
5
+ or a `yaml.YAMLError` directly. Adapters that produce those lower-level
6
+ errors wrap them here (via `raise ... from exc`, preserving the cause for
7
+ debugging) so the public API has one small, deliberate error surface.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+
13
+ class ApplicationError(Exception):
14
+ """Base class for all errors raised by the application layer."""
15
+
16
+
17
+ class PlanSchemaError(ApplicationError):
18
+ """The plan text isn't valid YAML, or doesn't match the plan schema
19
+ (missing/extra/mistyped fields) -- a problem with the document itself,
20
+ independent of whether its dependency graph would be valid.
21
+ """
22
+
23
+
24
+ class PlanValidationError(ApplicationError):
25
+ """The plan is well-formed YAML matching the schema, but the step
26
+ graph it describes is not a valid DAG (a cycle, or a dependency on an
27
+ unknown step id).
28
+ """
29
+
30
+
31
+ class UnknownStepTypeError(ApplicationError):
32
+ """A step's `type` has no registered executor."""
33
+
34
+ def __init__(self, step_type: str) -> None:
35
+ self.step_type = step_type
36
+ super().__init__(f"no executor registered for step type '{step_type}'")
37
+
38
+
39
+ class SpecValidationError(ApplicationError):
40
+ """A step's `spec` doesn't match its executor's own declared shape --
41
+ raised by `TypedExecutor.validate_spec()` implementations (see
42
+ `flowlit.application.ports.executor`), never a raw pydantic
43
+ `ValidationError` or any other library-specific exception, so callers
44
+ only ever need to catch this one type.
45
+
46
+ Raised (and handled) in two different places, deliberately: eagerly
47
+ in `WorkflowService.submit_plan()` for any step whose spec has no
48
+ `${{ steps.<id>.output.<path> }}` placeholders (fail the whole plan
49
+ before execution starts), and again in `ExecutorDispatcher` right
50
+ before dispatch (the only point a placeholder-bearing spec can
51
+ honestly be checked, once it's been resolved) -- there, it's just an
52
+ ordinary step failure, not a re-raise.
53
+ """
54
+
55
+
56
+ class WorkflowNotFoundError(ApplicationError):
57
+ """No workflow with the given id has been submitted (or it hasn't
58
+ been persisted yet)."""
59
+
60
+ def __init__(self, workflow_id: str) -> None:
61
+ self.workflow_id = workflow_id
62
+ super().__init__(f"no workflow found with id '{workflow_id}'")
63
+
64
+
65
+ class WorkflowAlreadyStartedError(ApplicationError):
66
+ """A workflow was asked to start (via `WorkflowService.start_workflow()`
67
+ or `run_to_completion()`) more than once. Each submitted workflow can
68
+ only be started once -- poll `get_status()`/`get_next_steps()` to
69
+ observe a run that's already in progress, rather than starting it
70
+ again.
71
+ """
72
+
73
+ def __init__(self, workflow_id: str) -> None:
74
+ self.workflow_id = workflow_id
75
+ super().__init__(f"workflow '{workflow_id}' has already been started")
76
+
77
+
78
+ class WorkflowAlreadyInFlightError(ApplicationError):
79
+ """A `concurrency.mode: singleton` plan was submitted while another
80
+ run of the same plan `id` is still PENDING or RUNNING, and this
81
+ submission doesn't match it closely enough to be treated as a
82
+ duplicate of it (see `WorkflowRunIdConflictError` for the one case
83
+ that's ambiguous rather than a plain reject).
84
+
85
+ Raised for: no `run_id` supplied at all (`submit_plan()`, which
86
+ never claims a stable identity to dedup against), or a `run_id` that
87
+ doesn't match the in-flight run's own. The caller should back off and
88
+ retry later, or -- if it actually meant to attach to the existing
89
+ run -- resubmit via `submit_plan_with_run_id()` with that run's own
90
+ `run_id` and identical plan text instead.
91
+ """
92
+
93
+ def __init__(self, plan_id: str, in_flight_handle: str) -> None:
94
+ self.plan_id = plan_id
95
+ self.in_flight_handle = in_flight_handle
96
+ super().__init__(
97
+ f"plan '{plan_id}' is a singleton and already has a run in flight: "
98
+ f"'{in_flight_handle}'"
99
+ )
100
+
101
+
102
+ class WorkflowRunIdConflictError(ApplicationError):
103
+ """A `concurrency.mode: singleton` plan was submitted with the same
104
+ `run_id` as its currently in-flight run, but different plan text.
105
+
106
+ Reusing a `run_id` is how a caller declares "this is the same
107
+ logical request as before" -- but the request itself has changed, so
108
+ silently either running the old text or the new one would be
109
+ guessing. Modeled on Stripe's own idempotency-key contract: a key
110
+ reused with a different payload is a conflict, not a dedup.
111
+ """
112
+
113
+ def __init__(self, plan_id: str, run_id: str) -> None:
114
+ self.plan_id = plan_id
115
+ self.run_id = run_id
116
+ super().__init__(
117
+ f"plan '{plan_id}' run '{run_id}' is already in flight with different plan text -- "
118
+ "reusing a run_id must resubmit the exact same plan"
119
+ )
@@ -0,0 +1,201 @@
1
+ """The events published on the `EventBus` as a workflow runs.
2
+
3
+ All immutable value objects (`@dataclass(frozen=True)`, the closest
4
+ Python equivalent to a Java record). Subscribers are keyed by these
5
+ classes' *type*, not by a string topic name -- see `ports.event_bus`.
6
+
7
+ Orchestrator Dispatcher / Executor
8
+ ------------ ---------------------
9
+ publishes StepReady (observers only)
10
+ publishes StepDispatchRequested ----------> runs the step
11
+ subscribes to
12
+ StepCompleted / StepFailed <------------- publishes StepCompleted or StepFailed
13
+ publishes StepSkipped
14
+ (cascading, for a failed
15
+ optional+blocking step's
16
+ dependents)
17
+ publishes WorkflowCompleted
18
+ or WorkflowFailed once
19
+ the workflow is settled
20
+ publishes WorkflowCancelled
21
+ (Orchestrator.cancel(),
22
+ called from outside --
23
+ not triggered by any
24
+ step's own outcome)
25
+
26
+ `StepReady` and `StepDispatchRequested` carry the same shape of fields
27
+ and are published back to back for the same step, but they're
28
+ deliberately two distinct event types, not one event with two
29
+ subscribers. Their `spec` does differ in one respect: `StepReady` always
30
+ carries the step's *raw* spec, exactly as declared in the plan
31
+ (`${{ steps.<id>.output.<path> }}` placeholders and all), since it's
32
+ announced before the orchestrator has even attempted to resolve them;
33
+ `StepDispatchRequested` carries the *resolved* spec, with every
34
+ placeholder substituted against the actual output of the steps it
35
+ references (see `flowlit.domain.spec_resolution`) -- executors only
36
+ ever see concrete values, never a placeholder syntax to worry about.
37
+
38
+ * `StepReady` is a pure announcement -- "this step's dependencies are
39
+ satisfied and it has been handed off for execution." Any number of
40
+ observers can subscribe (the CLI's progress printer today; a future
41
+ audit log or MCP notification stream later). Nothing subscribed to it
42
+ causes further work.
43
+ * `StepDispatchRequested` is the actual trigger -- "now go run it." Only
44
+ `ExecutorDispatcher` subscribes to it.
45
+
46
+ Splitting them this way makes "every observer has seen StepReady before
47
+ any executor-side effect (including its own logging) can happen"
48
+ *structural*: it falls out of the orchestrator publishing `StepReady`
49
+ before `StepDispatchRequested`, and the event bus fully delivering one
50
+ event to all of its subscribers before returning (see
51
+ `AsyncioEventBus`). It does not depend on which of several subscribers
52
+ to the same event happened to register first -- relying on that turned
53
+ out to be exactly the kind of accident-of-wiring-order bug this split
54
+ exists to rule out.
55
+
56
+ `WorkflowCompleted` fires for a successful *or* partially-degraded-but-
57
+ still-successful run alike (`WorkflowStatus.COMPLETED` or
58
+ `COMPLETED_WITH_FAILURES` -- see `flowlit.domain.workflow`); the exact
59
+ distinction is a matter of looking at the workflow's own status
60
+ afterward, not a different event.
61
+ """
62
+
63
+ from __future__ import annotations
64
+
65
+ from collections.abc import Mapping
66
+ from dataclasses import dataclass
67
+ from typing import Any
68
+
69
+
70
+ @dataclass(frozen=True)
71
+ class StepReady:
72
+ """Published by the orchestrator when a step's dependencies have all
73
+ been satisfied and it has been handed off for execution. Purely
74
+ informational -- for observers (e.g. the CLI's progress printer),
75
+ not for whatever actually runs the step. See `StepDispatchRequested`
76
+ for that.
77
+ """
78
+
79
+ workflow_id: str
80
+ step_id: str
81
+ step_type: str
82
+ spec: Mapping[str, Any]
83
+ timeout_seconds: float | None = None
84
+ """This step's `timeout_seconds` (see `flowlit.domain.step.Step`),
85
+ carried here purely for observers -- e.g. a progress printer that
86
+ wants to say "submitted, 30s budget." `StepDispatchRequested` is
87
+ what actually enforces it; see there."""
88
+ retry: int = 0
89
+ retry_backoff_seconds: float = 1.0
90
+ """This step's `retry`/`retry_backoff_seconds`, carried here purely
91
+ for observers, same spirit as `timeout_seconds` above --
92
+ `StepDispatchRequested` is what actually enforces them."""
93
+
94
+
95
+ @dataclass(frozen=True)
96
+ class StepDispatchRequested:
97
+ """Published by the orchestrator immediately after `StepReady`, once
98
+ that announcement has been fully delivered to every observer. This
99
+ is what `ExecutorDispatcher` actually subscribes to in order to run
100
+ the step -- kept separate from `StepReady` so "the step was
101
+ announced" and "the step is now being executed" can never race.
102
+ """
103
+
104
+ workflow_id: str
105
+ step_id: str
106
+ step_type: str
107
+ spec: Mapping[str, Any]
108
+ timeout_seconds: float | None = None
109
+ """If set, `ExecutorDispatcher` enforces this as a hard ceiling on
110
+ the executor's entire `execute()` call via `asyncio.wait_for()`,
111
+ uniformly across every executor type -- see `Step.timeout_seconds`
112
+ for the full contract this places on executor authors."""
113
+ retry: int = 0
114
+ retry_backoff_seconds: float = 1.0
115
+ """If `retry > 0`, `ExecutorDispatcher` retries a genuine execution
116
+ failure (an `ExecutorResult(success=False, ...)`, an unhandled
117
+ exception, or a timeout) up to this many additional times, with
118
+ `retry_backoff_seconds` delay before the first retry, doubling each
119
+ attempt -- see `Step.retry`. Each attempt gets its own fresh
120
+ `timeout_seconds` budget, not one shared across all of them."""
121
+
122
+
123
+ @dataclass(frozen=True)
124
+ class StepCompleted:
125
+ """Published by the executor dispatcher once a step's executor
126
+ reports success."""
127
+
128
+ workflow_id: str
129
+ step_id: str
130
+ output: Mapping[str, Any] | None
131
+
132
+
133
+ @dataclass(frozen=True)
134
+ class StepFailed:
135
+ """Published by the executor dispatcher when a step's executor
136
+ reports failure, raises, or when no executor is registered for its
137
+ type."""
138
+
139
+ workflow_id: str
140
+ step_id: str
141
+ error: str
142
+
143
+
144
+ @dataclass(frozen=True)
145
+ class StepSkipped:
146
+ """Published by the orchestrator for each step that
147
+ `Workflow.cascade_skip_dependents_of()` marks SKIPPED -- i.e. a step
148
+ that can never run because it (transitively) depends on a step that
149
+ failed while `optional` and `blocking`. Purely informational, same
150
+ spirit as `StepReady`."""
151
+
152
+ workflow_id: str
153
+ step_id: str
154
+ reason: str
155
+
156
+
157
+ @dataclass(frozen=True)
158
+ class StepCancelled:
159
+ """Published for each step whose task was still in flight when a
160
+ *required* step failed elsewhere and the orchestrator cancelled
161
+ every other in-flight step. Distinct from `StepSkipped`: a skipped
162
+ step never even started; a cancelled step was genuinely RUNNING and
163
+ got stopped. Purely informational -- doesn't affect the workflow's
164
+ outcome, which is already decided by the required failure that
165
+ triggered the cancellation.
166
+ """
167
+
168
+ workflow_id: str
169
+ step_id: str
170
+ reason: str
171
+
172
+
173
+ @dataclass(frozen=True)
174
+ class WorkflowCompleted:
175
+ """Published by the orchestrator once the workflow is settled with a
176
+ successful outcome -- `WorkflowStatus.COMPLETED` or
177
+ `COMPLETED_WITH_FAILURES`. Check the workflow's own status (e.g. via
178
+ `WorkflowService.get_status`) to tell which."""
179
+
180
+ workflow_id: str
181
+
182
+
183
+ @dataclass(frozen=True)
184
+ class WorkflowFailed:
185
+ """Published by the orchestrator the moment a *required* step fails
186
+ -- immediately, not waiting for other in-flight steps to settle,
187
+ since the outcome is already decided at that point."""
188
+
189
+ workflow_id: str
190
+ reason: str
191
+
192
+
193
+ @dataclass(frozen=True)
194
+ class WorkflowCancelled:
195
+ """Published by `Orchestrator.cancel()` once a workflow has been
196
+ stopped from outside -- the only one of the three "workflow is done"
197
+ events that isn't triggered by the steps' own outcomes (see
198
+ `WorkflowStatus.CANCELLED`)."""
199
+
200
+ workflow_id: str
201
+ reason: str
@@ -0,0 +1,112 @@
1
+ """`ExecutorDispatcher`: the single bus subscriber that routes
2
+ `StepDispatchRequested` events to the right executor.
3
+
4
+ Rather than every concrete `Executor` subscribing to the bus itself, one
5
+ dispatcher subscribes once and looks the executor up in the
6
+ `ExecutorRegistry` by step type. That keeps `Executor` implementations
7
+ bus-agnostic (trivially unit-testable by calling `execute()` directly)
8
+ and keeps "which step types exist" a single, obvious place to look
9
+ (the registry), rather than scattered across each executor's own
10
+ subscription call.
11
+
12
+ Deliberately subscribes to `StepDispatchRequested`, not the earlier
13
+ `StepReady` announcement -- see `events.py` for why those are two
14
+ separate event types.
15
+
16
+ This is also the single choke point every step's execution passes
17
+ through regardless of type -- which is why `Step.timeout_seconds` and
18
+ `Step.retry` are both enforced right here, via `asyncio.wait_for()` and
19
+ a plain retry loop, rather than inside any individual executor:
20
+ wrapping the call here means `noop`, `sleep`, and any custom executor
21
+ someone registers all get the same timeout/retry behavior `shell`/
22
+ `http` would otherwise be the only ones to have, with zero changes to
23
+ any of them. See `flowlit.application.ports.executor.Executor` for the
24
+ cancellation contract this places on executor authors.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import asyncio
30
+ import logging
31
+
32
+ from flowlit.application.errors import SpecValidationError, UnknownStepTypeError
33
+ from flowlit.application.events import StepCompleted, StepDispatchRequested, StepFailed
34
+ from flowlit.application.executor_registry import ExecutorRegistry
35
+ from flowlit.application.ports.event_bus import EventBus
36
+
37
+ _logger = logging.getLogger(__name__)
38
+
39
+
40
+ class ExecutorDispatcher:
41
+ def __init__(self, event_bus: EventBus, registry: ExecutorRegistry) -> None:
42
+ self._event_bus = event_bus
43
+ self._registry = registry
44
+ event_bus.subscribe(StepDispatchRequested, self._on_dispatch_requested)
45
+
46
+ async def _on_dispatch_requested(self, event: StepDispatchRequested) -> None:
47
+ try:
48
+ executor = self._registry.get(event.step_type)
49
+ except UnknownStepTypeError as exc:
50
+ # Never retried -- an unregistered step type fails identically
51
+ # on every attempt, so retrying it is pure waste.
52
+ await self._publish_failure(event, str(exc))
53
+ return
54
+
55
+ attempts = event.retry + 1
56
+ last_error = "unknown error"
57
+ for attempt in range(attempts):
58
+ try:
59
+ if event.timeout_seconds is None:
60
+ result = await executor.execute(event.step_id, event.spec)
61
+ else:
62
+ result = await asyncio.wait_for(
63
+ executor.execute(event.step_id, event.spec),
64
+ timeout=event.timeout_seconds,
65
+ )
66
+ except TimeoutError:
67
+ # asyncio.wait_for() has already cancelled and fully unwound
68
+ # the executor's execute() coroutine by the time this
69
+ # exception reaches us -- see Executor's docstring for what
70
+ # that requires of the executor itself (release its own
71
+ # resources on CancelledError, not only its own timeout path).
72
+ last_error = f"step timed out after {event.timeout_seconds} seconds"
73
+ except SpecValidationError as exc:
74
+ # Raised by a TypedExecutor's validate_spec() (see
75
+ # BaseExecutor.execute()) -- the only point a placeholder-
76
+ # bearing spec can honestly be checked, since WorkflowService.
77
+ # submit_plan()'s own eager check skips those. Never
78
+ # retried either, for the same reason as UnknownStepTypeError
79
+ # above: an invalid spec is invalid on every attempt.
80
+ await self._publish_failure(event, f"invalid spec: {exc}")
81
+ return
82
+ except Exception as exc: # noqa: BLE001 - an executor bug must fail the step, not the process
83
+ last_error = f"executor raised an unexpected error: {exc}"
84
+ else:
85
+ if result.success:
86
+ await self._event_bus.publish(
87
+ StepCompleted(event.workflow_id, event.step_id, result.output)
88
+ )
89
+ return
90
+ last_error = result.error or "executor reported failure"
91
+
92
+ if attempt < attempts - 1:
93
+ delay = event.retry_backoff_seconds * (2**attempt)
94
+ _logger.info(
95
+ "step %s: attempt %s/%s failed (%s), retrying in %ss",
96
+ event.step_id,
97
+ attempt + 1,
98
+ attempts,
99
+ last_error,
100
+ delay,
101
+ )
102
+ await asyncio.sleep(delay)
103
+
104
+ # Only mention attempt count when there actually was more than
105
+ # one -- keeps the retry=0 (default) error message identical to
106
+ # what it was before this feature existed.
107
+ if attempts > 1:
108
+ last_error = f"failed after {attempts} attempt(s): {last_error}"
109
+ await self._publish_failure(event, last_error)
110
+
111
+ async def _publish_failure(self, event: StepDispatchRequested, error: str) -> None:
112
+ await self._event_bus.publish(StepFailed(event.workflow_id, event.step_id, error))
@@ -0,0 +1,26 @@
1
+ """`ExecutorRegistry`: routes a step type to the `Executor` that handles it.
2
+
3
+ Extension pattern for adding a new step type: implement the `Executor`
4
+ port (no base class to inherit -- just the matching `execute` method),
5
+ then add one `"my_type": MyExecutor()` entry where the registry is built
6
+ in `flowlit.interfaces.composition_root`. No plugin discovery or
7
+ registration decorators for v1.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from collections.abc import Mapping
13
+
14
+ from flowlit.application.errors import UnknownStepTypeError
15
+ from flowlit.application.ports.executor import Executor
16
+
17
+
18
+ class ExecutorRegistry:
19
+ def __init__(self, executors_by_step_type: Mapping[str, Executor]) -> None:
20
+ self._executors_by_step_type = dict(executors_by_step_type)
21
+
22
+ def get(self, step_type: str) -> Executor:
23
+ try:
24
+ return self._executors_by_step_type[step_type]
25
+ except KeyError:
26
+ raise UnknownStepTypeError(step_type) from None