flowlit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- flowlit/__init__.py +30 -0
- flowlit/application/__init__.py +7 -0
- flowlit/application/dto.py +100 -0
- flowlit/application/errors.py +119 -0
- flowlit/application/events.py +201 -0
- flowlit/application/executor_dispatcher.py +112 -0
- flowlit/application/executor_registry.py +26 -0
- flowlit/application/orchestrator.py +393 -0
- flowlit/application/ports/__init__.py +6 -0
- flowlit/application/ports/event_bus.py +31 -0
- flowlit/application/ports/execution_coordinator.py +45 -0
- flowlit/application/ports/executor.py +153 -0
- flowlit/application/ports/plan_parser.py +28 -0
- flowlit/application/ports/workflow_repository.py +24 -0
- flowlit/application/workflow_service.py +342 -0
- flowlit/domain/__init__.py +7 -0
- flowlit/domain/dag.py +99 -0
- flowlit/domain/errors.py +245 -0
- flowlit/domain/spec_resolution.py +232 -0
- flowlit/domain/step.py +333 -0
- flowlit/domain/workflow.py +746 -0
- flowlit/infrastructure/__init__.py +7 -0
- flowlit/infrastructure/event_bus/__init__.py +1 -0
- flowlit/infrastructure/event_bus/asyncio_event_bus.py +72 -0
- flowlit/infrastructure/execution/__init__.py +3 -0
- flowlit/infrastructure/execution/asyncio_execution_coordinator.py +140 -0
- flowlit/infrastructure/executors/__init__.py +11 -0
- flowlit/infrastructure/executors/base_executor.py +102 -0
- flowlit/infrastructure/executors/http_executor.py +191 -0
- flowlit/infrastructure/executors/noop_executor.py +40 -0
- flowlit/infrastructure/executors/shell_executor.py +133 -0
- flowlit/infrastructure/executors/sleep_executor.py +56 -0
- flowlit/infrastructure/repositories/__init__.py +1 -0
- flowlit/infrastructure/repositories/in_memory_repository.py +36 -0
- flowlit/infrastructure/yaml_loader/__init__.py +1 -0
- flowlit/infrastructure/yaml_loader/loader.py +103 -0
- flowlit/infrastructure/yaml_loader/schema.py +61 -0
- flowlit/interfaces/__init__.py +6 -0
- flowlit/interfaces/cli/__init__.py +1 -0
- flowlit/interfaces/cli/main.py +170 -0
- flowlit/interfaces/composition_root.py +92 -0
- flowlit/plugins.py +91 -0
- flowlit/py.typed +0 -0
- flowlit-0.1.0.dist-info/METADATA +198 -0
- flowlit-0.1.0.dist-info/RECORD +48 -0
- flowlit-0.1.0.dist-info/WHEEL +4 -0
- flowlit-0.1.0.dist-info/entry_points.txt +2 -0
- flowlit-0.1.0.dist-info/licenses/LICENSE +21 -0
flowlit/__init__.py
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""Flowlit: a lightweight, event-driven workflow execution engine.
|
|
2
|
+
|
|
3
|
+
The codebase is organized as four layers, each only allowed to depend on
|
|
4
|
+
the ones below it:
|
|
5
|
+
|
|
6
|
+
interfaces (CLI, future MCP server -- thin adapters, no business logic)
|
|
7
|
+
v
|
|
8
|
+
infrastructure (YAML loading, in-memory event bus/repository, executors)
|
|
9
|
+
v
|
|
10
|
+
application (use cases, ports/Protocols, orchestration)
|
|
11
|
+
v
|
|
12
|
+
domain (Step/Workflow entities, DAG validation -- zero third-party deps)
|
|
13
|
+
|
|
14
|
+
`domain` and `application` never import from `infrastructure` or
|
|
15
|
+
`interfaces`. Wiring of concrete implementations happens in exactly one
|
|
16
|
+
place: `flowlit.interfaces.composition_root`.
|
|
17
|
+
|
|
18
|
+
Above all four layers sits one public extension seam, re-exported here
|
|
19
|
+
for ergonomics: `register_executor()` (see `flowlit.plugins`) lets code
|
|
20
|
+
that imports flowlit as a library add its own step-type executors
|
|
21
|
+
without forking flowlit's own source.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from flowlit.plugins import clear_registered_executors, register_executor, registered_executors
|
|
25
|
+
|
|
26
|
+
__all__ = [
|
|
27
|
+
"clear_registered_executors",
|
|
28
|
+
"register_executor",
|
|
29
|
+
"registered_executors",
|
|
30
|
+
]
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
"""Application layer: use cases, ports, and orchestration.
|
|
2
|
+
|
|
3
|
+
Depends only on `flowlit.domain` plus the standard library (`asyncio`,
|
|
4
|
+
`typing`). Defines the ports (`Protocol` classes) that infrastructure
|
|
5
|
+
adapters implement, and the `WorkflowService` use-case API that all
|
|
6
|
+
interface adapters (CLI, future MCP server) call through.
|
|
7
|
+
"""
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""Boundary DTOs returned by `WorkflowService`.
|
|
2
|
+
|
|
3
|
+
Plain, immutable, primitives-only dataclasses -- never a domain `Step` or
|
|
4
|
+
`Workflow`. Every interface adapter (CLI now, MCP server later) can
|
|
5
|
+
serialize these straight to JSON with `dataclasses.asdict()` without
|
|
6
|
+
importing anything from `flowlit.domain`.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from collections.abc import Mapping
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass(frozen=True)
|
|
17
|
+
class StepDTO:
|
|
18
|
+
id: str
|
|
19
|
+
type: str
|
|
20
|
+
status: str
|
|
21
|
+
spec: Mapping[str, Any]
|
|
22
|
+
"""The *raw* spec, exactly as declared in the plan -- placeholders
|
|
23
|
+
(`${{ steps.<id>.output.<path> }}`) and all, unresolved. See
|
|
24
|
+
`resolved_spec` for the substituted values, once available."""
|
|
25
|
+
resolved_spec: Mapping[str, Any] | None
|
|
26
|
+
"""The spec with every placeholder substituted, the same way
|
|
27
|
+
`StepDispatchRequested` carries it to the executor -- computed
|
|
28
|
+
on demand (via `Workflow.resolve_step_spec()`), not stored, so this
|
|
29
|
+
is always current as of *this* call, not a snapshot from dispatch
|
|
30
|
+
time. `None` if it can't be resolved (yet, or ever) -- most commonly
|
|
31
|
+
because the step is still `PENDING`, but also for e.g. a `SKIPPED`
|
|
32
|
+
step whose own spec referenced a dependency that never produced
|
|
33
|
+
output. See `flowlit.application.errors.OutputResolutionError` for
|
|
34
|
+
what "can't be resolved" actually means; a `None` here is never
|
|
35
|
+
itself an error to the caller, just "nothing to show yet."""
|
|
36
|
+
output: Mapping[str, Any] | None
|
|
37
|
+
error: str | None
|
|
38
|
+
optional: bool
|
|
39
|
+
blocking: bool
|
|
40
|
+
silent_failures: bool
|
|
41
|
+
when: str | None
|
|
42
|
+
"""The raw `when:` condition, if this step has one -- see
|
|
43
|
+
`flowlit.domain.step.Step`. `None` means the step is unconditional,
|
|
44
|
+
not that its condition was falsy (a falsy condition shows up as
|
|
45
|
+
`status == "skipped"` instead)."""
|
|
46
|
+
when_not: str | None
|
|
47
|
+
"""The raw `when_not:` condition, if this step has one -- negated
|
|
48
|
+
`when:`, same caveat about `None` not meaning "was truthy"."""
|
|
49
|
+
else_: frozenset[str] | None
|
|
50
|
+
"""The step ids this step's `else:` names, if it has one -- true iff
|
|
51
|
+
every one of them ended up skipped by its own condition (not
|
|
52
|
+
cascade) -- see `skip_cause`."""
|
|
53
|
+
skip_cause: str | None
|
|
54
|
+
"""Only meaningful when `status == "skipped"`, `None` otherwise:
|
|
55
|
+
`"condition"` if this step's own `when`/`when_not`/`else` evaluated
|
|
56
|
+
falsy (a deliberate, expected non-selection), `"cascade"` if it was
|
|
57
|
+
swept up by an upstream failure or an external `cancel_workflow()`
|
|
58
|
+
instead, never getting the chance to evaluate anything of its own --
|
|
59
|
+
see `flowlit.domain.step.SkipCause`."""
|
|
60
|
+
timeout_seconds: float | None
|
|
61
|
+
"""This step's own `timeout_seconds`, if set -- see
|
|
62
|
+
`flowlit.domain.step.Step`. `None` means unbounded, not that a
|
|
63
|
+
timeout already happened (that shows up as `status == "failed"`,
|
|
64
|
+
with `error` saying so)."""
|
|
65
|
+
retry: int
|
|
66
|
+
"""Additional attempts after the first, if any -- see
|
|
67
|
+
`flowlit.domain.step.Step`. `0` (the default) means no retry;
|
|
68
|
+
doesn't reflect how many attempts were actually made if the step is
|
|
69
|
+
still in flight or already terminal -- `error` says that, e.g.
|
|
70
|
+
"failed after 3 attempt(s): ..."."""
|
|
71
|
+
retry_backoff_seconds: float
|
|
72
|
+
"""Delay before the first retry, doubling each attempt -- only
|
|
73
|
+
meaningful when `retry > 0`."""
|
|
74
|
+
started_at: str | None
|
|
75
|
+
"""ISO 8601 (`datetime.isoformat()`), `None` if the step never ran
|
|
76
|
+
(e.g. it was skipped) -- see `flowlit.domain.step.Step`."""
|
|
77
|
+
ended_at: str | None
|
|
78
|
+
"""ISO 8601, `None` until the step reaches a terminal status."""
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclass(frozen=True)
|
|
82
|
+
class WorkflowStatusDTO:
|
|
83
|
+
id: str
|
|
84
|
+
run_id: str
|
|
85
|
+
"""This execution's identity, distinct from `id` (the plan's own
|
|
86
|
+
declared, run-independent identity) -- see
|
|
87
|
+
`flowlit.domain.workflow.Workflow.run_id`. What every step's
|
|
88
|
+
`${{ step.idempotency_key }}` is built from."""
|
|
89
|
+
concurrency_mode: str
|
|
90
|
+
"""`"allow_parallel"` or `"singleton"` -- see
|
|
91
|
+
`flowlit.domain.workflow.ConcurrencyMode`."""
|
|
92
|
+
name: str
|
|
93
|
+
status: str
|
|
94
|
+
steps: tuple[StepDTO, ...]
|
|
95
|
+
created_at: str
|
|
96
|
+
"""ISO 8601 -- always set; stamped once, in `Workflow.create()`."""
|
|
97
|
+
started_at: str | None
|
|
98
|
+
"""ISO 8601, `None` until the workflow leaves PENDING."""
|
|
99
|
+
ended_at: str | None
|
|
100
|
+
"""ISO 8601, `None` until the workflow reaches a terminal status."""
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
"""Application-level errors.
|
|
2
|
+
|
|
3
|
+
Callers of `WorkflowService` (the CLI now, an MCP server later) only ever
|
|
4
|
+
need to catch these -- never a domain error, a pydantic `ValidationError`,
|
|
5
|
+
or a `yaml.YAMLError` directly. Adapters that produce those lower-level
|
|
6
|
+
errors wrap them here (via `raise ... from exc`, preserving the cause for
|
|
7
|
+
debugging) so the public API has one small, deliberate error surface.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class ApplicationError(Exception):
|
|
14
|
+
"""Base class for all errors raised by the application layer."""
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class PlanSchemaError(ApplicationError):
|
|
18
|
+
"""The plan text isn't valid YAML, or doesn't match the plan schema
|
|
19
|
+
(missing/extra/mistyped fields) -- a problem with the document itself,
|
|
20
|
+
independent of whether its dependency graph would be valid.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class PlanValidationError(ApplicationError):
|
|
25
|
+
"""The plan is well-formed YAML matching the schema, but the step
|
|
26
|
+
graph it describes is not a valid DAG (a cycle, or a dependency on an
|
|
27
|
+
unknown step id).
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class UnknownStepTypeError(ApplicationError):
|
|
32
|
+
"""A step's `type` has no registered executor."""
|
|
33
|
+
|
|
34
|
+
def __init__(self, step_type: str) -> None:
|
|
35
|
+
self.step_type = step_type
|
|
36
|
+
super().__init__(f"no executor registered for step type '{step_type}'")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class SpecValidationError(ApplicationError):
|
|
40
|
+
"""A step's `spec` doesn't match its executor's own declared shape --
|
|
41
|
+
raised by `TypedExecutor.validate_spec()` implementations (see
|
|
42
|
+
`flowlit.application.ports.executor`), never a raw pydantic
|
|
43
|
+
`ValidationError` or any other library-specific exception, so callers
|
|
44
|
+
only ever need to catch this one type.
|
|
45
|
+
|
|
46
|
+
Raised (and handled) in two different places, deliberately: eagerly
|
|
47
|
+
in `WorkflowService.submit_plan()` for any step whose spec has no
|
|
48
|
+
`${{ steps.<id>.output.<path> }}` placeholders (fail the whole plan
|
|
49
|
+
before execution starts), and again in `ExecutorDispatcher` right
|
|
50
|
+
before dispatch (the only point a placeholder-bearing spec can
|
|
51
|
+
honestly be checked, once it's been resolved) -- there, it's just an
|
|
52
|
+
ordinary step failure, not a re-raise.
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class WorkflowNotFoundError(ApplicationError):
|
|
57
|
+
"""No workflow with the given id has been submitted (or it hasn't
|
|
58
|
+
been persisted yet)."""
|
|
59
|
+
|
|
60
|
+
def __init__(self, workflow_id: str) -> None:
|
|
61
|
+
self.workflow_id = workflow_id
|
|
62
|
+
super().__init__(f"no workflow found with id '{workflow_id}'")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class WorkflowAlreadyStartedError(ApplicationError):
|
|
66
|
+
"""A workflow was asked to start (via `WorkflowService.start_workflow()`
|
|
67
|
+
or `run_to_completion()`) more than once. Each submitted workflow can
|
|
68
|
+
only be started once -- poll `get_status()`/`get_next_steps()` to
|
|
69
|
+
observe a run that's already in progress, rather than starting it
|
|
70
|
+
again.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
def __init__(self, workflow_id: str) -> None:
|
|
74
|
+
self.workflow_id = workflow_id
|
|
75
|
+
super().__init__(f"workflow '{workflow_id}' has already been started")
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
class WorkflowAlreadyInFlightError(ApplicationError):
|
|
79
|
+
"""A `concurrency.mode: singleton` plan was submitted while another
|
|
80
|
+
run of the same plan `id` is still PENDING or RUNNING, and this
|
|
81
|
+
submission doesn't match it closely enough to be treated as a
|
|
82
|
+
duplicate of it (see `WorkflowRunIdConflictError` for the one case
|
|
83
|
+
that's ambiguous rather than a plain reject).
|
|
84
|
+
|
|
85
|
+
Raised for: no `run_id` supplied at all (`submit_plan()`, which
|
|
86
|
+
never claims a stable identity to dedup against), or a `run_id` that
|
|
87
|
+
doesn't match the in-flight run's own. The caller should back off and
|
|
88
|
+
retry later, or -- if it actually meant to attach to the existing
|
|
89
|
+
run -- resubmit via `submit_plan_with_run_id()` with that run's own
|
|
90
|
+
`run_id` and identical plan text instead.
|
|
91
|
+
"""
|
|
92
|
+
|
|
93
|
+
def __init__(self, plan_id: str, in_flight_handle: str) -> None:
|
|
94
|
+
self.plan_id = plan_id
|
|
95
|
+
self.in_flight_handle = in_flight_handle
|
|
96
|
+
super().__init__(
|
|
97
|
+
f"plan '{plan_id}' is a singleton and already has a run in flight: "
|
|
98
|
+
f"'{in_flight_handle}'"
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class WorkflowRunIdConflictError(ApplicationError):
|
|
103
|
+
"""A `concurrency.mode: singleton` plan was submitted with the same
|
|
104
|
+
`run_id` as its currently in-flight run, but different plan text.
|
|
105
|
+
|
|
106
|
+
Reusing a `run_id` is how a caller declares "this is the same
|
|
107
|
+
logical request as before" -- but the request itself has changed, so
|
|
108
|
+
silently either running the old text or the new one would be
|
|
109
|
+
guessing. Modeled on Stripe's own idempotency-key contract: a key
|
|
110
|
+
reused with a different payload is a conflict, not a dedup.
|
|
111
|
+
"""
|
|
112
|
+
|
|
113
|
+
def __init__(self, plan_id: str, run_id: str) -> None:
|
|
114
|
+
self.plan_id = plan_id
|
|
115
|
+
self.run_id = run_id
|
|
116
|
+
super().__init__(
|
|
117
|
+
f"plan '{plan_id}' run '{run_id}' is already in flight with different plan text -- "
|
|
118
|
+
"reusing a run_id must resubmit the exact same plan"
|
|
119
|
+
)
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""The events published on the `EventBus` as a workflow runs.
|
|
2
|
+
|
|
3
|
+
All immutable value objects (`@dataclass(frozen=True)`, the closest
|
|
4
|
+
Python equivalent to a Java record). Subscribers are keyed by these
|
|
5
|
+
classes' *type*, not by a string topic name -- see `ports.event_bus`.
|
|
6
|
+
|
|
7
|
+
Orchestrator Dispatcher / Executor
|
|
8
|
+
------------ ---------------------
|
|
9
|
+
publishes StepReady (observers only)
|
|
10
|
+
publishes StepDispatchRequested ----------> runs the step
|
|
11
|
+
subscribes to
|
|
12
|
+
StepCompleted / StepFailed <------------- publishes StepCompleted or StepFailed
|
|
13
|
+
publishes StepSkipped
|
|
14
|
+
(cascading, for a failed
|
|
15
|
+
optional+blocking step's
|
|
16
|
+
dependents)
|
|
17
|
+
publishes WorkflowCompleted
|
|
18
|
+
or WorkflowFailed once
|
|
19
|
+
the workflow is settled
|
|
20
|
+
publishes WorkflowCancelled
|
|
21
|
+
(Orchestrator.cancel(),
|
|
22
|
+
called from outside --
|
|
23
|
+
not triggered by any
|
|
24
|
+
step's own outcome)
|
|
25
|
+
|
|
26
|
+
`StepReady` and `StepDispatchRequested` carry the same shape of fields
|
|
27
|
+
and are published back to back for the same step, but they're
|
|
28
|
+
deliberately two distinct event types, not one event with two
|
|
29
|
+
subscribers. Their `spec` does differ in one respect: `StepReady` always
|
|
30
|
+
carries the step's *raw* spec, exactly as declared in the plan
|
|
31
|
+
(`${{ steps.<id>.output.<path> }}` placeholders and all), since it's
|
|
32
|
+
announced before the orchestrator has even attempted to resolve them;
|
|
33
|
+
`StepDispatchRequested` carries the *resolved* spec, with every
|
|
34
|
+
placeholder substituted against the actual output of the steps it
|
|
35
|
+
references (see `flowlit.domain.spec_resolution`) -- executors only
|
|
36
|
+
ever see concrete values, never a placeholder syntax to worry about.
|
|
37
|
+
|
|
38
|
+
* `StepReady` is a pure announcement -- "this step's dependencies are
|
|
39
|
+
satisfied and it has been handed off for execution." Any number of
|
|
40
|
+
observers can subscribe (the CLI's progress printer today; a future
|
|
41
|
+
audit log or MCP notification stream later). Nothing subscribed to it
|
|
42
|
+
causes further work.
|
|
43
|
+
* `StepDispatchRequested` is the actual trigger -- "now go run it." Only
|
|
44
|
+
`ExecutorDispatcher` subscribes to it.
|
|
45
|
+
|
|
46
|
+
Splitting them this way makes "every observer has seen StepReady before
|
|
47
|
+
any executor-side effect (including its own logging) can happen"
|
|
48
|
+
*structural*: it falls out of the orchestrator publishing `StepReady`
|
|
49
|
+
before `StepDispatchRequested`, and the event bus fully delivering one
|
|
50
|
+
event to all of its subscribers before returning (see
|
|
51
|
+
`AsyncioEventBus`). It does not depend on which of several subscribers
|
|
52
|
+
to the same event happened to register first -- relying on that turned
|
|
53
|
+
out to be exactly the kind of accident-of-wiring-order bug this split
|
|
54
|
+
exists to rule out.
|
|
55
|
+
|
|
56
|
+
`WorkflowCompleted` fires for a successful *or* partially-degraded-but-
|
|
57
|
+
still-successful run alike (`WorkflowStatus.COMPLETED` or
|
|
58
|
+
`COMPLETED_WITH_FAILURES` -- see `flowlit.domain.workflow`); the exact
|
|
59
|
+
distinction is a matter of looking at the workflow's own status
|
|
60
|
+
afterward, not a different event.
|
|
61
|
+
"""
|
|
62
|
+
|
|
63
|
+
from __future__ import annotations
|
|
64
|
+
|
|
65
|
+
from collections.abc import Mapping
|
|
66
|
+
from dataclasses import dataclass
|
|
67
|
+
from typing import Any
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass(frozen=True)
|
|
71
|
+
class StepReady:
|
|
72
|
+
"""Published by the orchestrator when a step's dependencies have all
|
|
73
|
+
been satisfied and it has been handed off for execution. Purely
|
|
74
|
+
informational -- for observers (e.g. the CLI's progress printer),
|
|
75
|
+
not for whatever actually runs the step. See `StepDispatchRequested`
|
|
76
|
+
for that.
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
workflow_id: str
|
|
80
|
+
step_id: str
|
|
81
|
+
step_type: str
|
|
82
|
+
spec: Mapping[str, Any]
|
|
83
|
+
timeout_seconds: float | None = None
|
|
84
|
+
"""This step's `timeout_seconds` (see `flowlit.domain.step.Step`),
|
|
85
|
+
carried here purely for observers -- e.g. a progress printer that
|
|
86
|
+
wants to say "submitted, 30s budget." `StepDispatchRequested` is
|
|
87
|
+
what actually enforces it; see there."""
|
|
88
|
+
retry: int = 0
|
|
89
|
+
retry_backoff_seconds: float = 1.0
|
|
90
|
+
"""This step's `retry`/`retry_backoff_seconds`, carried here purely
|
|
91
|
+
for observers, same spirit as `timeout_seconds` above --
|
|
92
|
+
`StepDispatchRequested` is what actually enforces them."""
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@dataclass(frozen=True)
|
|
96
|
+
class StepDispatchRequested:
|
|
97
|
+
"""Published by the orchestrator immediately after `StepReady`, once
|
|
98
|
+
that announcement has been fully delivered to every observer. This
|
|
99
|
+
is what `ExecutorDispatcher` actually subscribes to in order to run
|
|
100
|
+
the step -- kept separate from `StepReady` so "the step was
|
|
101
|
+
announced" and "the step is now being executed" can never race.
|
|
102
|
+
"""
|
|
103
|
+
|
|
104
|
+
workflow_id: str
|
|
105
|
+
step_id: str
|
|
106
|
+
step_type: str
|
|
107
|
+
spec: Mapping[str, Any]
|
|
108
|
+
timeout_seconds: float | None = None
|
|
109
|
+
"""If set, `ExecutorDispatcher` enforces this as a hard ceiling on
|
|
110
|
+
the executor's entire `execute()` call via `asyncio.wait_for()`,
|
|
111
|
+
uniformly across every executor type -- see `Step.timeout_seconds`
|
|
112
|
+
for the full contract this places on executor authors."""
|
|
113
|
+
retry: int = 0
|
|
114
|
+
retry_backoff_seconds: float = 1.0
|
|
115
|
+
"""If `retry > 0`, `ExecutorDispatcher` retries a genuine execution
|
|
116
|
+
failure (an `ExecutorResult(success=False, ...)`, an unhandled
|
|
117
|
+
exception, or a timeout) up to this many additional times, with
|
|
118
|
+
`retry_backoff_seconds` delay before the first retry, doubling each
|
|
119
|
+
attempt -- see `Step.retry`. Each attempt gets its own fresh
|
|
120
|
+
`timeout_seconds` budget, not one shared across all of them."""
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
@dataclass(frozen=True)
|
|
124
|
+
class StepCompleted:
|
|
125
|
+
"""Published by the executor dispatcher once a step's executor
|
|
126
|
+
reports success."""
|
|
127
|
+
|
|
128
|
+
workflow_id: str
|
|
129
|
+
step_id: str
|
|
130
|
+
output: Mapping[str, Any] | None
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
@dataclass(frozen=True)
|
|
134
|
+
class StepFailed:
|
|
135
|
+
"""Published by the executor dispatcher when a step's executor
|
|
136
|
+
reports failure, raises, or when no executor is registered for its
|
|
137
|
+
type."""
|
|
138
|
+
|
|
139
|
+
workflow_id: str
|
|
140
|
+
step_id: str
|
|
141
|
+
error: str
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
@dataclass(frozen=True)
|
|
145
|
+
class StepSkipped:
|
|
146
|
+
"""Published by the orchestrator for each step that
|
|
147
|
+
`Workflow.cascade_skip_dependents_of()` marks SKIPPED -- i.e. a step
|
|
148
|
+
that can never run because it (transitively) depends on a step that
|
|
149
|
+
failed while `optional` and `blocking`. Purely informational, same
|
|
150
|
+
spirit as `StepReady`."""
|
|
151
|
+
|
|
152
|
+
workflow_id: str
|
|
153
|
+
step_id: str
|
|
154
|
+
reason: str
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
@dataclass(frozen=True)
|
|
158
|
+
class StepCancelled:
|
|
159
|
+
"""Published for each step whose task was still in flight when a
|
|
160
|
+
*required* step failed elsewhere and the orchestrator cancelled
|
|
161
|
+
every other in-flight step. Distinct from `StepSkipped`: a skipped
|
|
162
|
+
step never even started; a cancelled step was genuinely RUNNING and
|
|
163
|
+
got stopped. Purely informational -- doesn't affect the workflow's
|
|
164
|
+
outcome, which is already decided by the required failure that
|
|
165
|
+
triggered the cancellation.
|
|
166
|
+
"""
|
|
167
|
+
|
|
168
|
+
workflow_id: str
|
|
169
|
+
step_id: str
|
|
170
|
+
reason: str
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
@dataclass(frozen=True)
|
|
174
|
+
class WorkflowCompleted:
|
|
175
|
+
"""Published by the orchestrator once the workflow is settled with a
|
|
176
|
+
successful outcome -- `WorkflowStatus.COMPLETED` or
|
|
177
|
+
`COMPLETED_WITH_FAILURES`. Check the workflow's own status (e.g. via
|
|
178
|
+
`WorkflowService.get_status`) to tell which."""
|
|
179
|
+
|
|
180
|
+
workflow_id: str
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
@dataclass(frozen=True)
|
|
184
|
+
class WorkflowFailed:
|
|
185
|
+
"""Published by the orchestrator the moment a *required* step fails
|
|
186
|
+
-- immediately, not waiting for other in-flight steps to settle,
|
|
187
|
+
since the outcome is already decided at that point."""
|
|
188
|
+
|
|
189
|
+
workflow_id: str
|
|
190
|
+
reason: str
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
@dataclass(frozen=True)
|
|
194
|
+
class WorkflowCancelled:
|
|
195
|
+
"""Published by `Orchestrator.cancel()` once a workflow has been
|
|
196
|
+
stopped from outside -- the only one of the three "workflow is done"
|
|
197
|
+
events that isn't triggered by the steps' own outcomes (see
|
|
198
|
+
`WorkflowStatus.CANCELLED`)."""
|
|
199
|
+
|
|
200
|
+
workflow_id: str
|
|
201
|
+
reason: str
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""`ExecutorDispatcher`: the single bus subscriber that routes
|
|
2
|
+
`StepDispatchRequested` events to the right executor.
|
|
3
|
+
|
|
4
|
+
Rather than every concrete `Executor` subscribing to the bus itself, one
|
|
5
|
+
dispatcher subscribes once and looks the executor up in the
|
|
6
|
+
`ExecutorRegistry` by step type. That keeps `Executor` implementations
|
|
7
|
+
bus-agnostic (trivially unit-testable by calling `execute()` directly)
|
|
8
|
+
and keeps "which step types exist" a single, obvious place to look
|
|
9
|
+
(the registry), rather than scattered across each executor's own
|
|
10
|
+
subscription call.
|
|
11
|
+
|
|
12
|
+
Deliberately subscribes to `StepDispatchRequested`, not the earlier
|
|
13
|
+
`StepReady` announcement -- see `events.py` for why those are two
|
|
14
|
+
separate event types.
|
|
15
|
+
|
|
16
|
+
This is also the single choke point every step's execution passes
|
|
17
|
+
through regardless of type -- which is why `Step.timeout_seconds` and
|
|
18
|
+
`Step.retry` are both enforced right here, via `asyncio.wait_for()` and
|
|
19
|
+
a plain retry loop, rather than inside any individual executor:
|
|
20
|
+
wrapping the call here means `noop`, `sleep`, and any custom executor
|
|
21
|
+
someone registers all get the same timeout/retry behavior `shell`/
|
|
22
|
+
`http` would otherwise be the only ones to have, with zero changes to
|
|
23
|
+
any of them. See `flowlit.application.ports.executor.Executor` for the
|
|
24
|
+
cancellation contract this places on executor authors.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import asyncio
|
|
30
|
+
import logging
|
|
31
|
+
|
|
32
|
+
from flowlit.application.errors import SpecValidationError, UnknownStepTypeError
|
|
33
|
+
from flowlit.application.events import StepCompleted, StepDispatchRequested, StepFailed
|
|
34
|
+
from flowlit.application.executor_registry import ExecutorRegistry
|
|
35
|
+
from flowlit.application.ports.event_bus import EventBus
|
|
36
|
+
|
|
37
|
+
_logger = logging.getLogger(__name__)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class ExecutorDispatcher:
|
|
41
|
+
def __init__(self, event_bus: EventBus, registry: ExecutorRegistry) -> None:
|
|
42
|
+
self._event_bus = event_bus
|
|
43
|
+
self._registry = registry
|
|
44
|
+
event_bus.subscribe(StepDispatchRequested, self._on_dispatch_requested)
|
|
45
|
+
|
|
46
|
+
async def _on_dispatch_requested(self, event: StepDispatchRequested) -> None:
|
|
47
|
+
try:
|
|
48
|
+
executor = self._registry.get(event.step_type)
|
|
49
|
+
except UnknownStepTypeError as exc:
|
|
50
|
+
# Never retried -- an unregistered step type fails identically
|
|
51
|
+
# on every attempt, so retrying it is pure waste.
|
|
52
|
+
await self._publish_failure(event, str(exc))
|
|
53
|
+
return
|
|
54
|
+
|
|
55
|
+
attempts = event.retry + 1
|
|
56
|
+
last_error = "unknown error"
|
|
57
|
+
for attempt in range(attempts):
|
|
58
|
+
try:
|
|
59
|
+
if event.timeout_seconds is None:
|
|
60
|
+
result = await executor.execute(event.step_id, event.spec)
|
|
61
|
+
else:
|
|
62
|
+
result = await asyncio.wait_for(
|
|
63
|
+
executor.execute(event.step_id, event.spec),
|
|
64
|
+
timeout=event.timeout_seconds,
|
|
65
|
+
)
|
|
66
|
+
except TimeoutError:
|
|
67
|
+
# asyncio.wait_for() has already cancelled and fully unwound
|
|
68
|
+
# the executor's execute() coroutine by the time this
|
|
69
|
+
# exception reaches us -- see Executor's docstring for what
|
|
70
|
+
# that requires of the executor itself (release its own
|
|
71
|
+
# resources on CancelledError, not only its own timeout path).
|
|
72
|
+
last_error = f"step timed out after {event.timeout_seconds} seconds"
|
|
73
|
+
except SpecValidationError as exc:
|
|
74
|
+
# Raised by a TypedExecutor's validate_spec() (see
|
|
75
|
+
# BaseExecutor.execute()) -- the only point a placeholder-
|
|
76
|
+
# bearing spec can honestly be checked, since WorkflowService.
|
|
77
|
+
# submit_plan()'s own eager check skips those. Never
|
|
78
|
+
# retried either, for the same reason as UnknownStepTypeError
|
|
79
|
+
# above: an invalid spec is invalid on every attempt.
|
|
80
|
+
await self._publish_failure(event, f"invalid spec: {exc}")
|
|
81
|
+
return
|
|
82
|
+
except Exception as exc: # noqa: BLE001 - an executor bug must fail the step, not the process
|
|
83
|
+
last_error = f"executor raised an unexpected error: {exc}"
|
|
84
|
+
else:
|
|
85
|
+
if result.success:
|
|
86
|
+
await self._event_bus.publish(
|
|
87
|
+
StepCompleted(event.workflow_id, event.step_id, result.output)
|
|
88
|
+
)
|
|
89
|
+
return
|
|
90
|
+
last_error = result.error or "executor reported failure"
|
|
91
|
+
|
|
92
|
+
if attempt < attempts - 1:
|
|
93
|
+
delay = event.retry_backoff_seconds * (2**attempt)
|
|
94
|
+
_logger.info(
|
|
95
|
+
"step %s: attempt %s/%s failed (%s), retrying in %ss",
|
|
96
|
+
event.step_id,
|
|
97
|
+
attempt + 1,
|
|
98
|
+
attempts,
|
|
99
|
+
last_error,
|
|
100
|
+
delay,
|
|
101
|
+
)
|
|
102
|
+
await asyncio.sleep(delay)
|
|
103
|
+
|
|
104
|
+
# Only mention attempt count when there actually was more than
|
|
105
|
+
# one -- keeps the retry=0 (default) error message identical to
|
|
106
|
+
# what it was before this feature existed.
|
|
107
|
+
if attempts > 1:
|
|
108
|
+
last_error = f"failed after {attempts} attempt(s): {last_error}"
|
|
109
|
+
await self._publish_failure(event, last_error)
|
|
110
|
+
|
|
111
|
+
async def _publish_failure(self, event: StepDispatchRequested, error: str) -> None:
|
|
112
|
+
await self._event_bus.publish(StepFailed(event.workflow_id, event.step_id, error))
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""`ExecutorRegistry`: routes a step type to the `Executor` that handles it.
|
|
2
|
+
|
|
3
|
+
Extension pattern for adding a new step type: implement the `Executor`
|
|
4
|
+
port (no base class to inherit -- just the matching `execute` method),
|
|
5
|
+
then add one `"my_type": MyExecutor()` entry where the registry is built
|
|
6
|
+
in `flowlit.interfaces.composition_root`. No plugin discovery or
|
|
7
|
+
registration decorators for v1.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from collections.abc import Mapping
|
|
13
|
+
|
|
14
|
+
from flowlit.application.errors import UnknownStepTypeError
|
|
15
|
+
from flowlit.application.ports.executor import Executor
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class ExecutorRegistry:
|
|
19
|
+
def __init__(self, executors_by_step_type: Mapping[str, Executor]) -> None:
|
|
20
|
+
self._executors_by_step_type = dict(executors_by_step_type)
|
|
21
|
+
|
|
22
|
+
def get(self, step_type: str) -> Executor:
|
|
23
|
+
try:
|
|
24
|
+
return self._executors_by_step_type[step_type]
|
|
25
|
+
except KeyError:
|
|
26
|
+
raise UnknownStepTypeError(step_type) from None
|