taskflow-meter 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- taskflow_meter/__init__.py +47 -0
- taskflow_meter/_version.py +24 -0
- taskflow_meter/api/__init__.py +35 -0
- taskflow_meter/api/asgi.py +236 -0
- taskflow_meter/api/dispatch.py +128 -0
- taskflow_meter/api/http.py +210 -0
- taskflow_meter/api/router.py +113 -0
- taskflow_meter/api/routes.py +53 -0
- taskflow_meter/api/serializers.py +137 -0
- taskflow_meter/api/service.py +189 -0
- taskflow_meter/api/sse.py +222 -0
- taskflow_meter/api/wsgi.py +145 -0
- taskflow_meter/cli.py +287 -0
- taskflow_meter/collect/__init__.py +31 -0
- taskflow_meter/collect/attachment.py +208 -0
- taskflow_meter/collect/listener.py +161 -0
- taskflow_meter/collect/pipeline.py +229 -0
- taskflow_meter/collect/progress.py +170 -0
- taskflow_meter/conf.py +173 -0
- taskflow_meter/contrib/__init__.py +18 -0
- taskflow_meter/contrib/django.py +160 -0
- taskflow_meter/contrib/fastapi.py +149 -0
- taskflow_meter/contrib/flask.py +140 -0
- taskflow_meter/contrib/paste.py +96 -0
- taskflow_meter/contrib/pecan.py +84 -0
- taskflow_meter/datasource/__init__.py +33 -0
- taskflow_meter/datasource/base.py +154 -0
- taskflow_meter/datasource/memory.py +232 -0
- taskflow_meter/datasource/persistence.py +311 -0
- taskflow_meter/datasource/sqlalchemy/__init__.py +21 -0
- taskflow_meter/datasource/sqlalchemy/migrations/env.py +68 -0
- taskflow_meter/datasource/sqlalchemy/migrations/script.py.mako +25 -0
- taskflow_meter/datasource/sqlalchemy/migrations/versions/0001_initial.py +71 -0
- taskflow_meter/datasource/sqlalchemy/models.py +63 -0
- taskflow_meter/datasource/sqlalchemy/source.py +367 -0
- taskflow_meter/diff.py +223 -0
- taskflow_meter/events.py +129 -0
- taskflow_meter/fold.py +137 -0
- taskflow_meter/meter.py +255 -0
- taskflow_meter/models.py +143 -0
- taskflow_meter/poller.py +191 -0
- taskflow_meter/py.typed +0 -0
- taskflow_meter/states.py +60 -0
- taskflow_meter/transports/__init__.py +21 -0
- taskflow_meter/transports/amqp.py +204 -0
- taskflow_meter/transports/base.py +105 -0
- taskflow_meter/transports/http.py +97 -0
- taskflow_meter/transports/memory.py +83 -0
- taskflow_meter-1.0.0.dist-info/METADATA +258 -0
- taskflow_meter-1.0.0.dist-info/RECORD +53 -0
- taskflow_meter-1.0.0.dist-info/WHEEL +4 -0
- taskflow_meter-1.0.0.dist-info/entry_points.txt +19 -0
- taskflow_meter-1.0.0.dist-info/licenses/LICENSE +176 -0
taskflow_meter/meter.py
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
2
|
+
# not use this file except in compliance with the License. You may obtain
|
|
3
|
+
# a copy of the License at
|
|
4
|
+
#
|
|
5
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
6
|
+
#
|
|
7
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
8
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
9
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
10
|
+
# License for the specific language governing permissions and limitations
|
|
11
|
+
# under the License.
|
|
12
|
+
|
|
13
|
+
"""The handle everything else hangs off.
|
|
14
|
+
|
|
15
|
+
A :class:`Meter` owns a source, optionally a store and a poller, and the
|
|
16
|
+
lifecycle that ties them together. The API layer holds one of these and
|
|
17
|
+
nothing else, which is what lets the same object serve an embedded
|
|
18
|
+
single-process dashboard and a worker reading a shared store.
|
|
19
|
+
|
|
20
|
+
Lifecycle deserves its own note, because the obvious mechanism does not
|
|
21
|
+
work. **A mounted ASGI app never receives the lifespan scope** -- the
|
|
22
|
+
host router handles it at the root and does not forward it to a mounted
|
|
23
|
+
sub-application -- so startup cannot hang off lifespan alone. Hence
|
|
24
|
+
three ways in, all safe to combine:
|
|
25
|
+
|
|
26
|
+
1. :meth:`start` / :meth:`stop`, or the context manager. What a host
|
|
27
|
+
application should call from its own lifespan or ``AppConfig.ready``.
|
|
28
|
+
2. :meth:`lifespan`, for when our ASGI app *is* the root application.
|
|
29
|
+
3. :meth:`ensure_started`, the lazy fallback a first request can call.
|
|
30
|
+
|
|
31
|
+
:meth:`start` and :meth:`stop` are reference counted, so a host that
|
|
32
|
+
starts the meter and a first request that also touches it do not fight:
|
|
33
|
+
the poller stops when the last holder lets go. :meth:`ensure_started`
|
|
34
|
+
deliberately takes no reference -- it has no paired release.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
from __future__ import annotations
|
|
38
|
+
|
|
39
|
+
import atexit
|
|
40
|
+
import contextlib
|
|
41
|
+
import logging
|
|
42
|
+
import threading
|
|
43
|
+
from collections.abc import AsyncIterator
|
|
44
|
+
from types import TracebackType
|
|
45
|
+
from typing import Self
|
|
46
|
+
|
|
47
|
+
from taskflow_meter.datasource.base import DEFAULT_EVENT_LIMIT
|
|
48
|
+
from taskflow_meter.datasource.base import DEFAULT_FLOW_LIMIT
|
|
49
|
+
from taskflow_meter.datasource.base import DataSource
|
|
50
|
+
from taskflow_meter.datasource.base import EventPage
|
|
51
|
+
from taskflow_meter.datasource.base import FlowPage
|
|
52
|
+
from taskflow_meter.datasource.base import WritableDataSource
|
|
53
|
+
from taskflow_meter.datasource.memory import MemoryDataSource
|
|
54
|
+
from taskflow_meter.models import AtomSnapshot
|
|
55
|
+
from taskflow_meter.models import FlowSnapshot
|
|
56
|
+
from taskflow_meter.poller import DEFAULT_INTERVAL
|
|
57
|
+
from taskflow_meter.poller import Poller
|
|
58
|
+
|
|
59
|
+
LOG = logging.getLogger(__name__)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class Meter:
|
|
63
|
+
"""Owns a source, a store, a poller, and their lifecycle."""
|
|
64
|
+
|
|
65
|
+
def __init__(
|
|
66
|
+
self,
|
|
67
|
+
source: DataSource,
|
|
68
|
+
*,
|
|
69
|
+
store: WritableDataSource | None = None,
|
|
70
|
+
poll: bool = True,
|
|
71
|
+
interval: float = DEFAULT_INTERVAL,
|
|
72
|
+
) -> None:
|
|
73
|
+
"""Watch ``source``, keeping what is observed in ``store``.
|
|
74
|
+
|
|
75
|
+
With ``poll`` set, a store is required and defaults to an
|
|
76
|
+
in-memory one. Without it there is no poller and no thread: the
|
|
77
|
+
meter is a read-through to ``source``, or to ``store`` when one
|
|
78
|
+
is supplied -- which is the shape an API worker takes when a
|
|
79
|
+
separate collector process keeps the store warm.
|
|
80
|
+
"""
|
|
81
|
+
self.source = source
|
|
82
|
+
self.store = (
|
|
83
|
+
store
|
|
84
|
+
if store is not None
|
|
85
|
+
else (MemoryDataSource() if poll else None)
|
|
86
|
+
)
|
|
87
|
+
self._poller: Poller | None = None
|
|
88
|
+
if poll:
|
|
89
|
+
assert self.store is not None
|
|
90
|
+
self._poller = Poller(source, self.store, interval=interval)
|
|
91
|
+
|
|
92
|
+
self._lock = threading.RLock()
|
|
93
|
+
self._holders = 0
|
|
94
|
+
self._started = False
|
|
95
|
+
|
|
96
|
+
# -- what to read from -----------------------------------------------
|
|
97
|
+
|
|
98
|
+
@property
|
|
99
|
+
def reader(self) -> DataSource:
|
|
100
|
+
"""Where queries are answered from.
|
|
101
|
+
|
|
102
|
+
The store wins when there is one, even though the source may be
|
|
103
|
+
fresher. Serving snapshots from the source while serving events
|
|
104
|
+
from the store would let a client see a snapshot newer than the
|
|
105
|
+
stream it is resuming, and silently miss the difference.
|
|
106
|
+
"""
|
|
107
|
+
return self.store if self.store is not None else self.source
|
|
108
|
+
|
|
109
|
+
@property
|
|
110
|
+
def poller(self) -> Poller | None:
|
|
111
|
+
return self._poller
|
|
112
|
+
|
|
113
|
+
@property
|
|
114
|
+
def running(self) -> bool:
|
|
115
|
+
with self._lock:
|
|
116
|
+
return self._started
|
|
117
|
+
|
|
118
|
+
# -- lifecycle -------------------------------------------------------
|
|
119
|
+
|
|
120
|
+
def start(self) -> None:
|
|
121
|
+
"""Take a reference, starting everything on the first one."""
|
|
122
|
+
with self._lock:
|
|
123
|
+
self._holders += 1
|
|
124
|
+
self._start_locked()
|
|
125
|
+
|
|
126
|
+
def stop(self) -> None:
|
|
127
|
+
"""Release a reference, stopping when the last one goes.
|
|
128
|
+
|
|
129
|
+
Extra calls are harmless: the count floors at zero rather than
|
|
130
|
+
going negative and leaving the meter unstoppable.
|
|
131
|
+
"""
|
|
132
|
+
with self._lock:
|
|
133
|
+
if self._holders > 0:
|
|
134
|
+
self._holders -= 1
|
|
135
|
+
if self._holders == 0:
|
|
136
|
+
self._stop_locked()
|
|
137
|
+
|
|
138
|
+
def ensure_started(self) -> None:
|
|
139
|
+
"""Start if not already running, without taking a reference.
|
|
140
|
+
|
|
141
|
+
For the lazy path: a first request can call this without owing a
|
|
142
|
+
matching :meth:`stop`.
|
|
143
|
+
"""
|
|
144
|
+
with self._lock:
|
|
145
|
+
self._start_locked()
|
|
146
|
+
|
|
147
|
+
def _start_locked(self) -> None:
|
|
148
|
+
if self._started:
|
|
149
|
+
return
|
|
150
|
+
self.source.start()
|
|
151
|
+
if self.store is not None:
|
|
152
|
+
self.store.start()
|
|
153
|
+
if self._poller is not None:
|
|
154
|
+
self._poller.start()
|
|
155
|
+
# Registered once per start cycle: _start_locked returns early
|
|
156
|
+
# when already started, so this cannot stack up.
|
|
157
|
+
atexit.register(self._atexit_stop)
|
|
158
|
+
self._started = True
|
|
159
|
+
|
|
160
|
+
def _stop_locked(self) -> None:
|
|
161
|
+
if not self._started:
|
|
162
|
+
return
|
|
163
|
+
self._started = False
|
|
164
|
+
if self._poller is not None:
|
|
165
|
+
self._poller.stop()
|
|
166
|
+
# Shutdown must not be abandoned halfway because one component
|
|
167
|
+
# objected, or the rest leak.
|
|
168
|
+
for closing in (self.store, self.source):
|
|
169
|
+
if closing is None:
|
|
170
|
+
continue
|
|
171
|
+
try:
|
|
172
|
+
closing.stop()
|
|
173
|
+
except Exception:
|
|
174
|
+
LOG.exception("failed stopping %r", closing)
|
|
175
|
+
# A no-op when it was never registered.
|
|
176
|
+
atexit.unregister(self._atexit_stop)
|
|
177
|
+
|
|
178
|
+
def _atexit_stop(self) -> None:
|
|
179
|
+
with self._lock:
|
|
180
|
+
self._holders = 0
|
|
181
|
+
self._stop_locked()
|
|
182
|
+
|
|
183
|
+
def __enter__(self) -> Self:
|
|
184
|
+
self.start()
|
|
185
|
+
return self
|
|
186
|
+
|
|
187
|
+
def __exit__(
|
|
188
|
+
self,
|
|
189
|
+
exc_type: type[BaseException] | None,
|
|
190
|
+
exc: BaseException | None,
|
|
191
|
+
tb: TracebackType | None,
|
|
192
|
+
) -> None:
|
|
193
|
+
self.stop()
|
|
194
|
+
|
|
195
|
+
@contextlib.asynccontextmanager
|
|
196
|
+
async def lifespan(
|
|
197
|
+
self,
|
|
198
|
+
app: object = None, # noqa: ARG002 - ASGI hands the app over
|
|
199
|
+
) -> AsyncIterator[None]:
|
|
200
|
+
"""ASGI lifespan handler, for when our app is the root one.
|
|
201
|
+
|
|
202
|
+
Useless on a *mounted* app -- see the module docstring -- so this
|
|
203
|
+
is a convenience, never the only way in.
|
|
204
|
+
"""
|
|
205
|
+
self.start()
|
|
206
|
+
try:
|
|
207
|
+
yield
|
|
208
|
+
finally:
|
|
209
|
+
self.stop()
|
|
210
|
+
|
|
211
|
+
# -- queries ---------------------------------------------------------
|
|
212
|
+
|
|
213
|
+
@property
|
|
214
|
+
def supports_events(self) -> bool:
|
|
215
|
+
return self.reader.supports_events
|
|
216
|
+
|
|
217
|
+
def list_flows(
|
|
218
|
+
self,
|
|
219
|
+
*,
|
|
220
|
+
state: str | None = None,
|
|
221
|
+
book_id: str | None = None,
|
|
222
|
+
limit: int = DEFAULT_FLOW_LIMIT,
|
|
223
|
+
marker: str | None = None,
|
|
224
|
+
) -> FlowPage:
|
|
225
|
+
return self.reader.list_flows(
|
|
226
|
+
state=state, book_id=book_id, limit=limit, marker=marker
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
def get_flow(self, run_id: str) -> FlowSnapshot | None:
|
|
230
|
+
return self.reader.get_flow(run_id)
|
|
231
|
+
|
|
232
|
+
def get_atoms(self, run_id: str) -> tuple[AtomSnapshot, ...] | None:
|
|
233
|
+
return self.reader.get_atoms(run_id)
|
|
234
|
+
|
|
235
|
+
def events_since(
|
|
236
|
+
self,
|
|
237
|
+
run_id: str,
|
|
238
|
+
*,
|
|
239
|
+
since_seq: int = 0,
|
|
240
|
+
limit: int = DEFAULT_EVENT_LIMIT,
|
|
241
|
+
) -> EventPage:
|
|
242
|
+
return self.reader.events_since(
|
|
243
|
+
run_id, since_seq=since_seq, limit=limit
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
def poll_once(self) -> int:
|
|
247
|
+
"""Run one poll on the calling thread.
|
|
248
|
+
|
|
249
|
+
Lets a caller drive the meter deterministically instead of
|
|
250
|
+
waiting on the interval.
|
|
251
|
+
"""
|
|
252
|
+
if self._poller is None:
|
|
253
|
+
msg = "this meter has no poller (poll=False)"
|
|
254
|
+
raise RuntimeError(msg)
|
|
255
|
+
return self._poller.poll_once()
|
taskflow_meter/models.py
ADDED
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
2
|
+
# not use this file except in compliance with the License. You may obtain
|
|
3
|
+
# a copy of the License at
|
|
4
|
+
#
|
|
5
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
6
|
+
#
|
|
7
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
8
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
9
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
10
|
+
# License for the specific language governing permissions and limitations
|
|
11
|
+
# under the License.
|
|
12
|
+
|
|
13
|
+
"""Point-in-time view of a flow and its atoms.
|
|
14
|
+
|
|
15
|
+
These are the types the API serialises and the diff engine compares. They
|
|
16
|
+
are deliberately plain: adapters translate taskflow's persistence models or
|
|
17
|
+
notifier callbacks *into* these, so nothing downstream needs to know which
|
|
18
|
+
producer it is talking to.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
from dataclasses import dataclass
|
|
24
|
+
from dataclasses import field
|
|
25
|
+
from typing import Any
|
|
26
|
+
|
|
27
|
+
from taskflow_meter import states
|
|
28
|
+
|
|
29
|
+
TASK = "task"
|
|
30
|
+
RETRY = "retry"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass(frozen=True, slots=True)
|
|
34
|
+
class AtomSnapshot:
|
|
35
|
+
"""What is known about a single atom at one moment."""
|
|
36
|
+
|
|
37
|
+
name: str
|
|
38
|
+
uuid: str | None = None
|
|
39
|
+
atom_type: str = TASK
|
|
40
|
+
state: str | None = None
|
|
41
|
+
intention: str | None = None
|
|
42
|
+
progress: float = 0.0
|
|
43
|
+
progress_details: dict[str, Any] | None = None
|
|
44
|
+
failure: dict[str, Any] | None = None
|
|
45
|
+
revert_failure: dict[str, Any] | None = None
|
|
46
|
+
has_result: bool = False
|
|
47
|
+
meta: dict[str, Any] = field(default_factory=dict)
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def is_finished(self) -> bool:
|
|
51
|
+
return self.state in states.ATOM_FINISH_STATES
|
|
52
|
+
|
|
53
|
+
@property
|
|
54
|
+
def is_running(self) -> bool:
|
|
55
|
+
return self.state in states.ATOM_RUNNING_STATES
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def completion(self) -> float:
|
|
59
|
+
"""This atom's contribution to flow completion, in ``[0, 1]``.
|
|
60
|
+
|
|
61
|
+
Reported progress is only trusted while the atom is running. A
|
|
62
|
+
finished atom's stored progress cannot be taken at face value:
|
|
63
|
+
taskflow sets it to 1.0 on both ``SUCCESS`` and ``REVERTED``, and
|
|
64
|
+
leaves it untouched on ``FAILURE``, so it would otherwise report a
|
|
65
|
+
reverted or failed atom as complete.
|
|
66
|
+
"""
|
|
67
|
+
if self.state in states.ATOM_COMPLETE_STATES:
|
|
68
|
+
return 1.0
|
|
69
|
+
if self.is_running:
|
|
70
|
+
return _clamp(self.progress)
|
|
71
|
+
return 0.0
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
@dataclass(frozen=True, slots=True)
|
|
75
|
+
class FlowSnapshot:
|
|
76
|
+
"""What is known about one flow run at one moment.
|
|
77
|
+
|
|
78
|
+
``observed_at`` is when *we* looked, not when the flow actually changed:
|
|
79
|
+
taskflow records no timestamps below the logbook, so nothing else is
|
|
80
|
+
available. For the live listener that is close enough to be the same
|
|
81
|
+
thing; for the poller it is only accurate to the poll interval.
|
|
82
|
+
"""
|
|
83
|
+
|
|
84
|
+
run_id: str
|
|
85
|
+
name: str = ""
|
|
86
|
+
state: str | None = None
|
|
87
|
+
book_id: str | None = None
|
|
88
|
+
book_name: str | None = None
|
|
89
|
+
observed_at: float = 0.0
|
|
90
|
+
meta: dict[str, Any] = field(default_factory=dict)
|
|
91
|
+
atoms: dict[str, AtomSnapshot] = field(default_factory=dict)
|
|
92
|
+
|
|
93
|
+
@property
|
|
94
|
+
def is_finished(self) -> bool:
|
|
95
|
+
return self.state in states.FLOW_FINISH_STATES
|
|
96
|
+
|
|
97
|
+
@property
|
|
98
|
+
def atom_names(self) -> tuple[str, ...]:
|
|
99
|
+
"""Atom names in a stable order, so output never reshuffles."""
|
|
100
|
+
return tuple(sorted(self.atoms))
|
|
101
|
+
|
|
102
|
+
def atom(self, name: str) -> AtomSnapshot | None:
|
|
103
|
+
return self.atoms.get(name)
|
|
104
|
+
|
|
105
|
+
@property
|
|
106
|
+
def running_atoms(self) -> tuple[AtomSnapshot, ...]:
|
|
107
|
+
"""The atoms executing right now, in name order.
|
|
108
|
+
|
|
109
|
+
The answer to "what is it doing?". A plural, because taskflow
|
|
110
|
+
runs unordered and graph flows in parallel, so there is often
|
|
111
|
+
more than one -- and none at all between two atoms, or once the
|
|
112
|
+
flow has finished.
|
|
113
|
+
"""
|
|
114
|
+
return tuple(
|
|
115
|
+
self.atoms[name]
|
|
116
|
+
for name in self.atom_names
|
|
117
|
+
if self.atoms[name].is_running
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
@property
|
|
121
|
+
def state_counts(self) -> dict[str, int]:
|
|
122
|
+
counts: dict[str, int] = {}
|
|
123
|
+
for atom in self.atoms.values():
|
|
124
|
+
key = atom.state if atom.state is not None else "UNKNOWN"
|
|
125
|
+
counts[key] = counts.get(key, 0) + 1
|
|
126
|
+
return counts
|
|
127
|
+
|
|
128
|
+
@property
|
|
129
|
+
def completion(self) -> float:
|
|
130
|
+
"""Unweighted mean of the atoms' completion, in ``[0, 1]``.
|
|
131
|
+
|
|
132
|
+
Every atom counts the same because taskflow gives us nothing to
|
|
133
|
+
weight them by -- no durations, no cost hints. Treat it as a rough
|
|
134
|
+
indicator, not an estimate of time remaining.
|
|
135
|
+
"""
|
|
136
|
+
if not self.atoms:
|
|
137
|
+
return 0.0
|
|
138
|
+
total = sum(atom.completion for atom in self.atoms.values())
|
|
139
|
+
return total / len(self.atoms)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _clamp(value: float, low: float = 0.0, high: float = 1.0) -> float:
|
|
143
|
+
return max(low, min(high, value))
|
taskflow_meter/poller.py
ADDED
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
2
|
+
# not use this file except in compliance with the License. You may obtain
|
|
3
|
+
# a copy of the License at
|
|
4
|
+
#
|
|
5
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
6
|
+
#
|
|
7
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
8
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
9
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
10
|
+
# License for the specific language governing permissions and limitations
|
|
11
|
+
# under the License.
|
|
12
|
+
|
|
13
|
+
"""Turn a state-only datasource into an event stream, by watching it.
|
|
14
|
+
|
|
15
|
+
The poller is the second producer. It reads whatever a source can see,
|
|
16
|
+
diffs each flow against what it saw last time, and feeds the resulting
|
|
17
|
+
events to a sink -- which is how a deployment whose persistence records
|
|
18
|
+
only current state ends up with a resumable stream.
|
|
19
|
+
|
|
20
|
+
What it cannot do is see between polls. A state a flow passed through
|
|
21
|
+
and left within one interval was never observable, and no amount of
|
|
22
|
+
diffing invents it. Shorter intervals narrow that window at the cost of
|
|
23
|
+
load; the in-process listener closes it entirely.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import logging
|
|
29
|
+
import threading
|
|
30
|
+
from collections.abc import Callable
|
|
31
|
+
from dataclasses import dataclass
|
|
32
|
+
from dataclasses import field
|
|
33
|
+
|
|
34
|
+
from taskflow_meter.datasource.base import DEFAULT_FLOW_LIMIT
|
|
35
|
+
from taskflow_meter.datasource.base import DataSource
|
|
36
|
+
from taskflow_meter.datasource.base import WritableDataSource
|
|
37
|
+
from taskflow_meter.diff import diff_flow
|
|
38
|
+
from taskflow_meter.events import SequenceAllocator
|
|
39
|
+
from taskflow_meter.models import FlowSnapshot
|
|
40
|
+
|
|
41
|
+
LOG = logging.getLogger(__name__)
|
|
42
|
+
|
|
43
|
+
#: Seconds between polls, when nothing says otherwise.
|
|
44
|
+
DEFAULT_INTERVAL = 2.0
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass(slots=True)
|
|
48
|
+
class PollStats:
|
|
49
|
+
"""Counters worth exposing on a health endpoint."""
|
|
50
|
+
|
|
51
|
+
polls: int = 0
|
|
52
|
+
events: int = 0
|
|
53
|
+
errors: int = 0
|
|
54
|
+
flows_seen: int = 0
|
|
55
|
+
last_error: str | None = field(default=None)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class Poller:
|
|
59
|
+
"""Watches ``source``, feeding what changes into ``sink``."""
|
|
60
|
+
|
|
61
|
+
def __init__(
|
|
62
|
+
self,
|
|
63
|
+
source: DataSource,
|
|
64
|
+
sink: WritableDataSource,
|
|
65
|
+
*,
|
|
66
|
+
interval: float = DEFAULT_INTERVAL,
|
|
67
|
+
allocator: SequenceAllocator | None = None,
|
|
68
|
+
page_size: int = DEFAULT_FLOW_LIMIT,
|
|
69
|
+
on_error: Callable[[Exception], None] | None = None,
|
|
70
|
+
) -> None:
|
|
71
|
+
if interval <= 0:
|
|
72
|
+
msg = "interval must be positive"
|
|
73
|
+
raise ValueError(msg)
|
|
74
|
+
if page_size < 1:
|
|
75
|
+
msg = "page_size must be at least 1"
|
|
76
|
+
raise ValueError(msg)
|
|
77
|
+
|
|
78
|
+
self.source = source
|
|
79
|
+
self.sink = sink
|
|
80
|
+
self.interval = interval
|
|
81
|
+
self.stats = PollStats()
|
|
82
|
+
|
|
83
|
+
self._allocator = allocator or SequenceAllocator()
|
|
84
|
+
self._page_size = page_size
|
|
85
|
+
self._on_error = on_error
|
|
86
|
+
# Every flow the source still reports, as last seen. Finished
|
|
87
|
+
# flows stay here on purpose: dropping one whose events were
|
|
88
|
+
# already emitted would make the next poll rediscover it and
|
|
89
|
+
# replay its whole history.
|
|
90
|
+
self._previous: dict[str, FlowSnapshot] = {}
|
|
91
|
+
self._lock = threading.Lock()
|
|
92
|
+
self._thread: threading.Thread | None = None
|
|
93
|
+
self._stopping = threading.Event()
|
|
94
|
+
|
|
95
|
+
@property
|
|
96
|
+
def allocator(self) -> SequenceAllocator:
|
|
97
|
+
return self._allocator
|
|
98
|
+
|
|
99
|
+
@property
|
|
100
|
+
def running(self) -> bool:
|
|
101
|
+
with self._lock:
|
|
102
|
+
return self._thread is not None and self._thread.is_alive()
|
|
103
|
+
|
|
104
|
+
# -- one pass --------------------------------------------------------
|
|
105
|
+
|
|
106
|
+
def poll_once(self) -> int:
|
|
107
|
+
"""Read, diff, emit. Returns how many events were produced.
|
|
108
|
+
|
|
109
|
+
Raises whatever the source or sink raises; the background loop is
|
|
110
|
+
what turns that into a counted, logged error.
|
|
111
|
+
"""
|
|
112
|
+
seen: dict[str, FlowSnapshot] = {}
|
|
113
|
+
emitted = 0
|
|
114
|
+
|
|
115
|
+
for snapshot in self._read_all():
|
|
116
|
+
seen[snapshot.run_id] = snapshot
|
|
117
|
+
previous = self._previous.get(snapshot.run_id)
|
|
118
|
+
events = diff_flow(
|
|
119
|
+
previous,
|
|
120
|
+
snapshot,
|
|
121
|
+
allocator=self._allocator,
|
|
122
|
+
ts=snapshot.observed_at,
|
|
123
|
+
)
|
|
124
|
+
if events:
|
|
125
|
+
self.sink.apply_many(events)
|
|
126
|
+
emitted += len(events)
|
|
127
|
+
|
|
128
|
+
for run_id in self._previous.keys() - seen.keys():
|
|
129
|
+
# The source stopped reporting it -- the logbook was deleted,
|
|
130
|
+
# or retention expired it. Forget it rather than pretend.
|
|
131
|
+
LOG.debug("run %s is no longer reported by the source", run_id)
|
|
132
|
+
|
|
133
|
+
self._previous = seen
|
|
134
|
+
self.stats.polls += 1
|
|
135
|
+
self.stats.events += emitted
|
|
136
|
+
self.stats.flows_seen = len(seen)
|
|
137
|
+
return emitted
|
|
138
|
+
|
|
139
|
+
def _read_all(self) -> list[FlowSnapshot]:
|
|
140
|
+
"""Walk every page the source offers."""
|
|
141
|
+
flows: list[FlowSnapshot] = []
|
|
142
|
+
marker: str | None = None
|
|
143
|
+
while True:
|
|
144
|
+
page = self.source.list_flows(limit=self._page_size, marker=marker)
|
|
145
|
+
flows.extend(page.items)
|
|
146
|
+
if not page.has_more:
|
|
147
|
+
return flows
|
|
148
|
+
marker = page.next_marker
|
|
149
|
+
|
|
150
|
+
# -- background loop -------------------------------------------------
|
|
151
|
+
|
|
152
|
+
def start(self) -> None:
|
|
153
|
+
"""Begin polling on a daemon thread. Idempotent."""
|
|
154
|
+
with self._lock:
|
|
155
|
+
if self._thread is not None and self._thread.is_alive():
|
|
156
|
+
return
|
|
157
|
+
self._stopping.clear()
|
|
158
|
+
self._thread = threading.Thread(
|
|
159
|
+
target=self._loop,
|
|
160
|
+
name="taskflow-meter-poller",
|
|
161
|
+
daemon=True,
|
|
162
|
+
)
|
|
163
|
+
self._thread.start()
|
|
164
|
+
|
|
165
|
+
def stop(self, timeout: float | None = 5.0) -> None:
|
|
166
|
+
"""Ask the loop to finish and wait for it. Idempotent."""
|
|
167
|
+
with self._lock:
|
|
168
|
+
thread = self._thread
|
|
169
|
+
self._thread = None
|
|
170
|
+
self._stopping.set()
|
|
171
|
+
if thread is not None and thread.is_alive():
|
|
172
|
+
thread.join(timeout)
|
|
173
|
+
if thread.is_alive():
|
|
174
|
+
LOG.warning("poller thread did not stop within %.1fs", timeout)
|
|
175
|
+
|
|
176
|
+
def _loop(self) -> None:
|
|
177
|
+
while not self._stopping.is_set():
|
|
178
|
+
try:
|
|
179
|
+
self.poll_once()
|
|
180
|
+
except Exception as exc:
|
|
181
|
+
# A monitoring loop that dies on one bad read stops
|
|
182
|
+
# monitoring silently, which is worse than a noisy one.
|
|
183
|
+
self.stats.errors += 1
|
|
184
|
+
self.stats.last_error = repr(exc)
|
|
185
|
+
LOG.exception("poll failed")
|
|
186
|
+
if self._on_error is not None:
|
|
187
|
+
try:
|
|
188
|
+
self._on_error(exc)
|
|
189
|
+
except Exception:
|
|
190
|
+
LOG.exception("poller error handler failed")
|
|
191
|
+
self._stopping.wait(self.interval)
|
taskflow_meter/py.typed
ADDED
|
File without changes
|
taskflow_meter/states.py
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
2
|
+
# not use this file except in compliance with the License. You may obtain
|
|
3
|
+
# a copy of the License at
|
|
4
|
+
#
|
|
5
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
6
|
+
#
|
|
7
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
8
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
9
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
10
|
+
# License for the specific language governing permissions and limitations
|
|
11
|
+
# under the License.
|
|
12
|
+
|
|
13
|
+
"""State vocabulary, derived from taskflow rather than duplicated.
|
|
14
|
+
|
|
15
|
+
Importing the values from :mod:`taskflow.states` keeps us from drifting if
|
|
16
|
+
upstream ever renames one, and keeps the groupings in a single place.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from taskflow import states as _tf
|
|
22
|
+
|
|
23
|
+
#: Flow states after which no further atom activity is expected.
|
|
24
|
+
FLOW_FINISH_STATES: frozenset[str] = frozenset(
|
|
25
|
+
{_tf.SUCCESS, _tf.FAILURE, _tf.REVERTED, _tf.SUSPENDED}
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
#: Atom states in which a result (or failure) is available. Mirrors
|
|
29
|
+
#: ``taskflow.listeners.base.FINISH_STATES``.
|
|
30
|
+
ATOM_FINISH_STATES: frozenset[str] = frozenset(
|
|
31
|
+
{_tf.SUCCESS, _tf.FAILURE, _tf.REVERTED, _tf.REVERT_FAILURE}
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
#: Atom states during which reported progress is meaningful.
|
|
35
|
+
ATOM_RUNNING_STATES: frozenset[str] = frozenset(
|
|
36
|
+
{_tf.RUNNING, _tf.REVERTING, _tf.RETRYING}
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
#: Atom states that contribute a completed unit of forward work.
|
|
40
|
+
#:
|
|
41
|
+
#: ``IGNORE`` counts because a decider has ruled the atom out: it will never
|
|
42
|
+
#: run, so treating it as outstanding would leave the flow permanently short
|
|
43
|
+
#: of 100%. ``REVERTED`` deliberately does *not* count -- taskflow sets an
|
|
44
|
+
#: atom's progress back to 1.0 when its revert finishes, which says the
|
|
45
|
+
#: revert completed, not that the work did.
|
|
46
|
+
ATOM_COMPLETE_STATES: frozenset[str] = frozenset({_tf.SUCCESS, _tf.IGNORE})
|
|
47
|
+
|
|
48
|
+
SUCCESS: str = _tf.SUCCESS
|
|
49
|
+
FAILURE: str = _tf.FAILURE
|
|
50
|
+
PENDING: str = _tf.PENDING
|
|
51
|
+
RUNNING: str = _tf.RUNNING
|
|
52
|
+
REVERTING: str = _tf.REVERTING
|
|
53
|
+
REVERTED: str = _tf.REVERTED
|
|
54
|
+
REVERT_FAILURE: str = _tf.REVERT_FAILURE
|
|
55
|
+
RETRYING: str = _tf.RETRYING
|
|
56
|
+
IGNORE: str = _tf.IGNORE
|
|
57
|
+
SUSPENDED: str = _tf.SUSPENDED
|
|
58
|
+
|
|
59
|
+
EXECUTE: str = _tf.EXECUTE
|
|
60
|
+
REVERT: str = _tf.REVERT
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
2
|
+
# not use this file except in compliance with the License. You may obtain
|
|
3
|
+
# a copy of the License at
|
|
4
|
+
#
|
|
5
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
6
|
+
#
|
|
7
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
8
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
9
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
10
|
+
# License for the specific language governing permissions and limitations
|
|
11
|
+
# under the License.
|
|
12
|
+
|
|
13
|
+
"""Transports: where events go once produced."""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from taskflow_meter.transports.base import Publisher
|
|
18
|
+
from taskflow_meter.transports.memory import DataSourcePublisher
|
|
19
|
+
from taskflow_meter.transports.memory import MemoryTransport
|
|
20
|
+
|
|
21
|
+
__all__ = ["DataSourcePublisher", "MemoryTransport", "Publisher"]
|