taskflow-meter 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- taskflow_meter/__init__.py +47 -0
- taskflow_meter/_version.py +24 -0
- taskflow_meter/api/__init__.py +35 -0
- taskflow_meter/api/asgi.py +236 -0
- taskflow_meter/api/dispatch.py +128 -0
- taskflow_meter/api/http.py +210 -0
- taskflow_meter/api/router.py +113 -0
- taskflow_meter/api/routes.py +53 -0
- taskflow_meter/api/serializers.py +137 -0
- taskflow_meter/api/service.py +189 -0
- taskflow_meter/api/sse.py +222 -0
- taskflow_meter/api/wsgi.py +145 -0
- taskflow_meter/cli.py +287 -0
- taskflow_meter/collect/__init__.py +31 -0
- taskflow_meter/collect/attachment.py +208 -0
- taskflow_meter/collect/listener.py +161 -0
- taskflow_meter/collect/pipeline.py +229 -0
- taskflow_meter/collect/progress.py +170 -0
- taskflow_meter/conf.py +173 -0
- taskflow_meter/contrib/__init__.py +18 -0
- taskflow_meter/contrib/django.py +160 -0
- taskflow_meter/contrib/fastapi.py +149 -0
- taskflow_meter/contrib/flask.py +140 -0
- taskflow_meter/contrib/paste.py +96 -0
- taskflow_meter/contrib/pecan.py +84 -0
- taskflow_meter/datasource/__init__.py +33 -0
- taskflow_meter/datasource/base.py +154 -0
- taskflow_meter/datasource/memory.py +232 -0
- taskflow_meter/datasource/persistence.py +311 -0
- taskflow_meter/datasource/sqlalchemy/__init__.py +21 -0
- taskflow_meter/datasource/sqlalchemy/migrations/env.py +68 -0
- taskflow_meter/datasource/sqlalchemy/migrations/script.py.mako +25 -0
- taskflow_meter/datasource/sqlalchemy/migrations/versions/0001_initial.py +71 -0
- taskflow_meter/datasource/sqlalchemy/models.py +63 -0
- taskflow_meter/datasource/sqlalchemy/source.py +367 -0
- taskflow_meter/diff.py +223 -0
- taskflow_meter/events.py +129 -0
- taskflow_meter/fold.py +137 -0
- taskflow_meter/meter.py +255 -0
- taskflow_meter/models.py +143 -0
- taskflow_meter/poller.py +191 -0
- taskflow_meter/py.typed +0 -0
- taskflow_meter/states.py +60 -0
- taskflow_meter/transports/__init__.py +21 -0
- taskflow_meter/transports/amqp.py +204 -0
- taskflow_meter/transports/base.py +105 -0
- taskflow_meter/transports/http.py +97 -0
- taskflow_meter/transports/memory.py +83 -0
- taskflow_meter-1.0.0.dist-info/METADATA +258 -0
- taskflow_meter-1.0.0.dist-info/RECORD +53 -0
- taskflow_meter-1.0.0.dist-info/WHEEL +4 -0
- taskflow_meter-1.0.0.dist-info/entry_points.txt +19 -0
- taskflow_meter-1.0.0.dist-info/licenses/LICENSE +176 -0
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
2
|
+
# not use this file except in compliance with the License. You may obtain
|
|
3
|
+
# a copy of the License at
|
|
4
|
+
#
|
|
5
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
6
|
+
#
|
|
7
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
8
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
9
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
10
|
+
# License for the specific language governing permissions and limitations
|
|
11
|
+
# under the License.
|
|
12
|
+
|
|
13
|
+
"""Attach to a running engine: the whole in-process path in one call.
|
|
14
|
+
|
|
15
|
+
Named ``attachment`` rather than ``attach`` because the package
|
|
16
|
+
re-exports the :func:`attach` function, and a module of the same name
|
|
17
|
+
would be shadowed by it -- leaving ``taskflow_meter.collect.attach``
|
|
18
|
+
meaning one thing to an importer and another to anything that resolves
|
|
19
|
+
attributes.
|
|
20
|
+
|
|
21
|
+
with attach(engine) as watched:
|
|
22
|
+
engine.run()
|
|
23
|
+
meter = Meter(watched.store, poll=False)
|
|
24
|
+
|
|
25
|
+
What that gets you over reading persistence is latency and shape: state
|
|
26
|
+
changes arrive as the engine makes them rather than a poll interval
|
|
27
|
+
later, per-task progress arrives the moment a task reports it, and the
|
|
28
|
+
flow's graph -- which taskflow never persists -- is emitted once up
|
|
29
|
+
front so a UI can draw the thing instead of tailing a list.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import contextlib
|
|
35
|
+
import logging
|
|
36
|
+
import time
|
|
37
|
+
from collections.abc import Callable
|
|
38
|
+
from collections.abc import Iterator
|
|
39
|
+
from collections.abc import Sequence
|
|
40
|
+
from dataclasses import dataclass
|
|
41
|
+
from dataclasses import field
|
|
42
|
+
from typing import Any
|
|
43
|
+
|
|
44
|
+
from taskflow.engines.action_engine import compiler as tf_compiler
|
|
45
|
+
|
|
46
|
+
from taskflow_meter.collect.listener import MeterListener
|
|
47
|
+
from taskflow_meter.collect.pipeline import EventPipeline
|
|
48
|
+
from taskflow_meter.collect.progress import ProgressTap
|
|
49
|
+
from taskflow_meter.datasource.base import WritableDataSource
|
|
50
|
+
from taskflow_meter.datasource.memory import MemoryDataSource
|
|
51
|
+
from taskflow_meter.events import Event
|
|
52
|
+
from taskflow_meter.events import EventKind
|
|
53
|
+
from taskflow_meter.events import SequenceAllocator
|
|
54
|
+
from taskflow_meter.transports.base import Publisher
|
|
55
|
+
from taskflow_meter.transports.memory import DataSourcePublisher
|
|
56
|
+
|
|
57
|
+
LOG = logging.getLogger(__name__)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass(slots=True)
|
|
61
|
+
class Attachment:
|
|
62
|
+
"""What is watching one engine, and where its events landed."""
|
|
63
|
+
|
|
64
|
+
engine: Any
|
|
65
|
+
store: WritableDataSource | None
|
|
66
|
+
pipeline: EventPipeline
|
|
67
|
+
listener: MeterListener
|
|
68
|
+
tap: ProgressTap
|
|
69
|
+
allocator: SequenceAllocator = field(default_factory=SequenceAllocator)
|
|
70
|
+
|
|
71
|
+
@property
|
|
72
|
+
def run_id(self) -> str:
|
|
73
|
+
return self.listener.run_id
|
|
74
|
+
|
|
75
|
+
def flush(self, timeout: float = 5.0) -> bool:
|
|
76
|
+
"""Wait for queued events to be delivered."""
|
|
77
|
+
return self.pipeline.flush(timeout)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@contextlib.contextmanager
|
|
81
|
+
def attach(
|
|
82
|
+
engine: Any,
|
|
83
|
+
*,
|
|
84
|
+
store: WritableDataSource | None = None,
|
|
85
|
+
publishers: Sequence[Publisher] = (),
|
|
86
|
+
book_id: str | None = None,
|
|
87
|
+
emit_structure: bool = True,
|
|
88
|
+
max_queue: int = 1000,
|
|
89
|
+
clock: Callable[[], float] = time.time,
|
|
90
|
+
) -> Iterator[Attachment]:
|
|
91
|
+
"""Watch ``engine`` for as long as the block runs.
|
|
92
|
+
|
|
93
|
+
An in-memory store is created unless one is given, so the simple
|
|
94
|
+
case needs no arguments. Everything is torn down on the way out,
|
|
95
|
+
including when the flow raises -- a callback left registered on
|
|
96
|
+
somebody's task object outlives the monitoring that wanted it.
|
|
97
|
+
"""
|
|
98
|
+
resolved_store = MemoryDataSource() if store is None else store
|
|
99
|
+
sinks: list[Publisher] = [DataSourcePublisher(resolved_store)]
|
|
100
|
+
sinks.extend(publishers)
|
|
101
|
+
|
|
102
|
+
pipeline = EventPipeline(sinks, max_queue=max_queue)
|
|
103
|
+
allocator = SequenceAllocator()
|
|
104
|
+
listener = MeterListener(
|
|
105
|
+
engine,
|
|
106
|
+
pipeline.submit,
|
|
107
|
+
allocator=allocator,
|
|
108
|
+
clock=clock,
|
|
109
|
+
book_id=book_id,
|
|
110
|
+
)
|
|
111
|
+
tap = ProgressTap(
|
|
112
|
+
engine,
|
|
113
|
+
pipeline.submit,
|
|
114
|
+
allocator=allocator,
|
|
115
|
+
clock=clock,
|
|
116
|
+
book_id=book_id,
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
pipeline.start()
|
|
120
|
+
try:
|
|
121
|
+
with listener:
|
|
122
|
+
tap.register()
|
|
123
|
+
try:
|
|
124
|
+
if emit_structure:
|
|
125
|
+
_emit_structure(
|
|
126
|
+
engine, pipeline.submit, allocator, clock, book_id
|
|
127
|
+
)
|
|
128
|
+
yield Attachment(
|
|
129
|
+
engine=engine,
|
|
130
|
+
store=resolved_store,
|
|
131
|
+
pipeline=pipeline,
|
|
132
|
+
listener=listener,
|
|
133
|
+
tap=tap,
|
|
134
|
+
allocator=allocator,
|
|
135
|
+
)
|
|
136
|
+
finally:
|
|
137
|
+
tap.deregister()
|
|
138
|
+
finally:
|
|
139
|
+
pipeline.stop()
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _emit_structure(
|
|
143
|
+
engine: Any,
|
|
144
|
+
emit: Callable[[Event], Any],
|
|
145
|
+
allocator: SequenceAllocator,
|
|
146
|
+
clock: Callable[[], float],
|
|
147
|
+
book_id: str | None,
|
|
148
|
+
) -> None:
|
|
149
|
+
"""Emit the flow's graph, once, before anything runs."""
|
|
150
|
+
try:
|
|
151
|
+
graph = describe_graph(engine)
|
|
152
|
+
except Exception:
|
|
153
|
+
# Topology is a bonus; failing to read it must not stop the
|
|
154
|
+
# states and progress that are the point.
|
|
155
|
+
LOG.exception("could not describe the flow's graph")
|
|
156
|
+
return
|
|
157
|
+
|
|
158
|
+
run_id = str(engine.storage.flow_uuid)
|
|
159
|
+
emit(
|
|
160
|
+
Event(
|
|
161
|
+
run_id=run_id,
|
|
162
|
+
seq=allocator.allocate(run_id),
|
|
163
|
+
ts=clock(),
|
|
164
|
+
kind=EventKind.FLOW_STRUCTURE,
|
|
165
|
+
book_id=book_id,
|
|
166
|
+
details=graph,
|
|
167
|
+
)
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def describe_graph(engine: Any) -> dict[str, Any]:
|
|
172
|
+
"""Render the compiled execution graph as plain data.
|
|
173
|
+
|
|
174
|
+
This is the thing the persistence datasource can never provide:
|
|
175
|
+
taskflow stores atoms but not the edges between them, so the shape
|
|
176
|
+
of a flow only exists in the process that compiled it.
|
|
177
|
+
"""
|
|
178
|
+
engine.compile()
|
|
179
|
+
compilation = engine.compilation
|
|
180
|
+
graph = compilation.execution_graph
|
|
181
|
+
|
|
182
|
+
nodes = [
|
|
183
|
+
{"name": _name(node), "kind": str(data.get("kind", ""))}
|
|
184
|
+
for node, data in graph.nodes(data=True)
|
|
185
|
+
]
|
|
186
|
+
edges = [
|
|
187
|
+
{"from": _name(source), "to": _name(target)}
|
|
188
|
+
for source, target in graph.edges()
|
|
189
|
+
]
|
|
190
|
+
return {
|
|
191
|
+
"nodes": sorted(nodes, key=lambda item: (item["kind"], item["name"])),
|
|
192
|
+
"edges": sorted(edges, key=lambda item: (item["from"], item["to"])),
|
|
193
|
+
"atom_count": sum(
|
|
194
|
+
1
|
|
195
|
+
for _, data in graph.nodes(data=True)
|
|
196
|
+
if data.get("kind") in tf_compiler.ATOMS
|
|
197
|
+
),
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _name(node: Any) -> str:
|
|
202
|
+
"""The name of a graph node.
|
|
203
|
+
|
|
204
|
+
Not ``str(node)``: an atom's repr is its decorated form, so a task
|
|
205
|
+
called "first" renders as '"first==1.0"' and matches nothing a
|
|
206
|
+
client knows it by.
|
|
207
|
+
"""
|
|
208
|
+
return str(getattr(node, "name", None) or node)
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
2
|
+
# not use this file except in compliance with the License. You may obtain
|
|
3
|
+
# a copy of the License at
|
|
4
|
+
#
|
|
5
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
6
|
+
#
|
|
7
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
8
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
9
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
10
|
+
# License for the specific language governing permissions and limitations
|
|
11
|
+
# under the License.
|
|
12
|
+
|
|
13
|
+
"""Watch an engine's notifiers for flow and atom state changes.
|
|
14
|
+
|
|
15
|
+
The half of the in-process path taskflow makes easy: a listener sees
|
|
16
|
+
every state transition, with the engine reporting them as they happen
|
|
17
|
+
rather than a poll interval later.
|
|
18
|
+
|
|
19
|
+
It is only half. taskflow never re-emits intra-task progress on the
|
|
20
|
+
atom notifier, so the numbers a task reports about itself arrive
|
|
21
|
+
through :mod:`taskflow_meter.collect.progress` instead.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import logging
|
|
27
|
+
import time
|
|
28
|
+
from collections.abc import Callable
|
|
29
|
+
from typing import Any
|
|
30
|
+
|
|
31
|
+
from taskflow.listeners import base as tf_listeners
|
|
32
|
+
|
|
33
|
+
from taskflow_meter.events import Event
|
|
34
|
+
from taskflow_meter.events import EventKind
|
|
35
|
+
from taskflow_meter.events import SequenceAllocator
|
|
36
|
+
from taskflow_meter.models import RETRY
|
|
37
|
+
from taskflow_meter.models import TASK
|
|
38
|
+
|
|
39
|
+
LOG = logging.getLogger(__name__)
|
|
40
|
+
|
|
41
|
+
Emit = Callable[[Event], Any]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class MeterListener(tf_listeners.Listener):
|
|
45
|
+
"""Turns an engine's notifications into events."""
|
|
46
|
+
|
|
47
|
+
def __init__(
|
|
48
|
+
self,
|
|
49
|
+
engine: Any,
|
|
50
|
+
emit: Emit,
|
|
51
|
+
*,
|
|
52
|
+
allocator: SequenceAllocator | None = None,
|
|
53
|
+
clock: Callable[[], float] = time.time,
|
|
54
|
+
book_id: str | None = None,
|
|
55
|
+
) -> None:
|
|
56
|
+
"""Watch ``engine``, handing each event to ``emit``.
|
|
57
|
+
|
|
58
|
+
``book_id`` has to be supplied because nothing on the engine
|
|
59
|
+
knows it: storage holds a flow detail and a backend, and a flow
|
|
60
|
+
detail carries no reference back to the logbook that owns it.
|
|
61
|
+
Whoever created the book has it; we would only be guessing.
|
|
62
|
+
"""
|
|
63
|
+
super().__init__(engine)
|
|
64
|
+
self._emit = emit
|
|
65
|
+
self._allocator = allocator or SequenceAllocator()
|
|
66
|
+
self._clock = clock
|
|
67
|
+
self._book_id = book_id
|
|
68
|
+
|
|
69
|
+
@property
|
|
70
|
+
def allocator(self) -> SequenceAllocator:
|
|
71
|
+
return self._allocator
|
|
72
|
+
|
|
73
|
+
@property
|
|
74
|
+
def run_id(self) -> str:
|
|
75
|
+
"""The flow detail uuid, which is how a run is identified."""
|
|
76
|
+
return str(self._engine.storage.flow_uuid)
|
|
77
|
+
|
|
78
|
+
# -- receivers -------------------------------------------------------
|
|
79
|
+
|
|
80
|
+
def _flow_receiver(self, state: str, details: dict[str, Any]) -> None:
|
|
81
|
+
self._safely(
|
|
82
|
+
EventKind.FLOW_STATE,
|
|
83
|
+
state=state,
|
|
84
|
+
old_state=details.get("old_state"),
|
|
85
|
+
extra={"flow_name": details.get("flow_name")},
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
def _task_receiver(self, state: str, details: dict[str, Any]) -> None:
|
|
89
|
+
self._atom_event(state, details, TASK, "task")
|
|
90
|
+
|
|
91
|
+
def _retry_receiver(self, state: str, details: dict[str, Any]) -> None:
|
|
92
|
+
self._atom_event(state, details, RETRY, "retry")
|
|
93
|
+
|
|
94
|
+
def _atom_event(
|
|
95
|
+
self,
|
|
96
|
+
state: str,
|
|
97
|
+
details: dict[str, Any],
|
|
98
|
+
atom_type: str,
|
|
99
|
+
prefix: str,
|
|
100
|
+
) -> None:
|
|
101
|
+
extra: dict[str, Any] = {}
|
|
102
|
+
if "result" in details:
|
|
103
|
+
# Results are arbitrary application objects and are not
|
|
104
|
+
# ours to copy around; their presence is the useful part.
|
|
105
|
+
extra["has_result"] = details["result"] is not None
|
|
106
|
+
self._safely(
|
|
107
|
+
EventKind.ATOM_STATE,
|
|
108
|
+
state=state,
|
|
109
|
+
old_state=details.get("old_state"),
|
|
110
|
+
atom_name=details.get(f"{prefix}_name"),
|
|
111
|
+
atom_uuid=details.get(f"{prefix}_uuid"),
|
|
112
|
+
atom_type=atom_type,
|
|
113
|
+
extra=extra,
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
# -- emitting --------------------------------------------------------
|
|
117
|
+
|
|
118
|
+
def _safely(self, kind: EventKind, **fields: Any) -> None:
|
|
119
|
+
"""Build and emit one event, swallowing anything that goes wrong.
|
|
120
|
+
|
|
121
|
+
This runs on the engine's thread -- an executor thread under the
|
|
122
|
+
parallel engine. An exception escaping here would surface
|
|
123
|
+
inside somebody's flow as a monitoring bug they did not ask for.
|
|
124
|
+
"""
|
|
125
|
+
try:
|
|
126
|
+
self._emit(self._build(kind, **fields))
|
|
127
|
+
except Exception:
|
|
128
|
+
LOG.exception("could not emit a %s event", kind)
|
|
129
|
+
|
|
130
|
+
def _build(
|
|
131
|
+
self,
|
|
132
|
+
kind: EventKind,
|
|
133
|
+
*,
|
|
134
|
+
state: str | None = None,
|
|
135
|
+
old_state: str | None = None,
|
|
136
|
+
atom_name: str | None = None,
|
|
137
|
+
atom_uuid: str | None = None,
|
|
138
|
+
atom_type: str | None = None,
|
|
139
|
+
extra: dict[str, Any] | None = None,
|
|
140
|
+
) -> Event:
|
|
141
|
+
run_id = self.run_id
|
|
142
|
+
details = {
|
|
143
|
+
key: value
|
|
144
|
+
for key, value in (extra or {}).items()
|
|
145
|
+
if value is not None
|
|
146
|
+
}
|
|
147
|
+
return Event(
|
|
148
|
+
run_id=run_id,
|
|
149
|
+
seq=self._allocator.allocate(run_id),
|
|
150
|
+
# Unlike the poller's, this timestamp really is when it
|
|
151
|
+
# happened: the engine is telling us as it does it.
|
|
152
|
+
ts=self._clock(),
|
|
153
|
+
kind=kind,
|
|
154
|
+
book_id=self._book_id,
|
|
155
|
+
atom_name=atom_name,
|
|
156
|
+
atom_uuid=atom_uuid,
|
|
157
|
+
atom_type=atom_type,
|
|
158
|
+
state=state,
|
|
159
|
+
old_state=old_state,
|
|
160
|
+
details=details,
|
|
161
|
+
)
|
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
# Licensed under the Apache License, Version 2.0 (the "License"); you may
|
|
2
|
+
# not use this file except in compliance with the License. You may obtain
|
|
3
|
+
# a copy of the License at
|
|
4
|
+
#
|
|
5
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
6
|
+
#
|
|
7
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
8
|
+
# distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
|
|
9
|
+
# WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
|
|
10
|
+
# License for the specific language governing permissions and limitations
|
|
11
|
+
# under the License.
|
|
12
|
+
|
|
13
|
+
"""The barrier between a running flow and everything downstream.
|
|
14
|
+
|
|
15
|
+
We are inside somebody else's task. A slow webhook, a full disk or a
|
|
16
|
+
bug in a publisher must cost that task nothing, so the rules here are
|
|
17
|
+
absolute:
|
|
18
|
+
|
|
19
|
+
* :meth:`EventPipeline.submit` never raises and never blocks. It puts
|
|
20
|
+
the event on a bounded queue and returns.
|
|
21
|
+
* All delivery happens on a sender thread. Nothing downstream runs on
|
|
22
|
+
the thread executing the task.
|
|
23
|
+
* A full queue drops, counts and logs. Blocking would stall the flow;
|
|
24
|
+
growing without limit would take the process down later, for reasons
|
|
25
|
+
nobody would connect back to monitoring.
|
|
26
|
+
* A publisher that raises is logged and counted. The next batch is
|
|
27
|
+
still attempted, because one bad delivery is not a reason to stop
|
|
28
|
+
watching.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import logging
|
|
34
|
+
import queue
|
|
35
|
+
import threading
|
|
36
|
+
from collections.abc import Callable
|
|
37
|
+
from collections.abc import Sequence
|
|
38
|
+
from dataclasses import dataclass
|
|
39
|
+
|
|
40
|
+
from taskflow_meter.events import Event
|
|
41
|
+
from taskflow_meter.transports.base import Publisher
|
|
42
|
+
|
|
43
|
+
LOG = logging.getLogger(__name__)
|
|
44
|
+
|
|
45
|
+
DEFAULT_MAX_QUEUE = 1000
|
|
46
|
+
DEFAULT_BATCH = 100
|
|
47
|
+
|
|
48
|
+
#: Pushed on the queue to wake the sender for shutdown.
|
|
49
|
+
_STOP = object()
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass(slots=True)
|
|
53
|
+
class PipelineStats:
|
|
54
|
+
"""Counters worth exposing on a health endpoint."""
|
|
55
|
+
|
|
56
|
+
submitted: int = 0
|
|
57
|
+
delivered: int = 0
|
|
58
|
+
dropped: int = 0
|
|
59
|
+
errors: int = 0
|
|
60
|
+
last_error: str | None = None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class EventPipeline:
|
|
64
|
+
"""Queues events and delivers them to publishers off-thread."""
|
|
65
|
+
|
|
66
|
+
def __init__(
|
|
67
|
+
self,
|
|
68
|
+
publishers: Sequence[Publisher],
|
|
69
|
+
*,
|
|
70
|
+
max_queue: int = DEFAULT_MAX_QUEUE,
|
|
71
|
+
batch_size: int = DEFAULT_BATCH,
|
|
72
|
+
on_drop: Callable[[Event], None] | None = None,
|
|
73
|
+
) -> None:
|
|
74
|
+
if max_queue < 1:
|
|
75
|
+
msg = "max_queue must be at least 1"
|
|
76
|
+
raise ValueError(msg)
|
|
77
|
+
if batch_size < 1:
|
|
78
|
+
msg = "batch_size must be at least 1"
|
|
79
|
+
raise ValueError(msg)
|
|
80
|
+
|
|
81
|
+
self.publishers = tuple(publishers)
|
|
82
|
+
self.batch_size = batch_size
|
|
83
|
+
self.stats = PipelineStats()
|
|
84
|
+
self._on_drop = on_drop
|
|
85
|
+
self._queue: queue.Queue[object] = queue.Queue(maxsize=max_queue)
|
|
86
|
+
self._lock = threading.Lock()
|
|
87
|
+
self._thread: threading.Thread | None = None
|
|
88
|
+
self._idle = threading.Event()
|
|
89
|
+
self._idle.set()
|
|
90
|
+
|
|
91
|
+
@property
|
|
92
|
+
def running(self) -> bool:
|
|
93
|
+
with self._lock:
|
|
94
|
+
return self._thread is not None and self._thread.is_alive()
|
|
95
|
+
|
|
96
|
+
# -- the flow's side -------------------------------------------------
|
|
97
|
+
|
|
98
|
+
def submit(self, event: Event) -> bool:
|
|
99
|
+
"""Queue an event. Returns whether *this* event was queued.
|
|
100
|
+
|
|
101
|
+
A full queue makes room by discarding the oldest, so a true
|
|
102
|
+
return does not mean nothing was lost -- ``stats.dropped`` is
|
|
103
|
+
what says that.
|
|
104
|
+
|
|
105
|
+
Called from task and executor threads. It cannot raise: a
|
|
106
|
+
monitoring bug must never fail somebody's task.
|
|
107
|
+
"""
|
|
108
|
+
self.stats.submitted += 1
|
|
109
|
+
try:
|
|
110
|
+
self._queue.put_nowait(event)
|
|
111
|
+
except queue.Full:
|
|
112
|
+
return self._make_room_for(event)
|
|
113
|
+
except Exception: # pragma: no cover - defensive
|
|
114
|
+
LOG.exception("could not queue an event")
|
|
115
|
+
return False
|
|
116
|
+
else:
|
|
117
|
+
self._idle.clear()
|
|
118
|
+
return True
|
|
119
|
+
|
|
120
|
+
def _make_room_for(self, event: Event) -> bool:
|
|
121
|
+
"""Discard the oldest to fit ``event`` in, and say so."""
|
|
122
|
+
try:
|
|
123
|
+
self._queue.get_nowait()
|
|
124
|
+
self._queue.task_done()
|
|
125
|
+
except queue.Empty: # pragma: no cover - racing a fast drain
|
|
126
|
+
pass
|
|
127
|
+
self.stats.dropped += 1
|
|
128
|
+
LOG.warning(
|
|
129
|
+
"event queue is full at %d; dropped the oldest (%d so far)",
|
|
130
|
+
self._queue.maxsize,
|
|
131
|
+
self.stats.dropped,
|
|
132
|
+
)
|
|
133
|
+
if self._on_drop is not None:
|
|
134
|
+
try:
|
|
135
|
+
self._on_drop(event)
|
|
136
|
+
except Exception:
|
|
137
|
+
LOG.exception("drop handler failed")
|
|
138
|
+
try:
|
|
139
|
+
self._queue.put_nowait(event)
|
|
140
|
+
except queue.Full: # pragma: no cover - racing another producer
|
|
141
|
+
self.stats.dropped += 1
|
|
142
|
+
return False
|
|
143
|
+
self._idle.clear()
|
|
144
|
+
return True
|
|
145
|
+
|
|
146
|
+
# -- lifecycle -------------------------------------------------------
|
|
147
|
+
|
|
148
|
+
def start(self) -> None:
|
|
149
|
+
"""Begin delivering on a daemon thread. Idempotent."""
|
|
150
|
+
with self._lock:
|
|
151
|
+
if self._thread is not None and self._thread.is_alive():
|
|
152
|
+
return
|
|
153
|
+
for publisher in self.publishers:
|
|
154
|
+
publisher.start()
|
|
155
|
+
self._thread = threading.Thread(
|
|
156
|
+
target=self._run,
|
|
157
|
+
name="taskflow-meter-pipeline",
|
|
158
|
+
daemon=True,
|
|
159
|
+
)
|
|
160
|
+
self._thread.start()
|
|
161
|
+
|
|
162
|
+
def stop(self, *, drain: bool = True, timeout: float = 5.0) -> None:
|
|
163
|
+
"""Finish delivering and shut the sender down. Idempotent."""
|
|
164
|
+
with self._lock:
|
|
165
|
+
thread = self._thread
|
|
166
|
+
self._thread = None
|
|
167
|
+
if thread is not None and thread.is_alive():
|
|
168
|
+
if drain:
|
|
169
|
+
self.flush(timeout)
|
|
170
|
+
self._queue.put(_STOP)
|
|
171
|
+
thread.join(timeout)
|
|
172
|
+
if thread.is_alive():
|
|
173
|
+
LOG.warning(
|
|
174
|
+
"pipeline thread did not stop within %.1fs", timeout
|
|
175
|
+
)
|
|
176
|
+
for publisher in self.publishers:
|
|
177
|
+
try:
|
|
178
|
+
publisher.stop()
|
|
179
|
+
except Exception:
|
|
180
|
+
LOG.exception("failed stopping %r", publisher)
|
|
181
|
+
|
|
182
|
+
def flush(self, timeout: float = 5.0) -> bool:
|
|
183
|
+
"""Wait for the queue to empty. Returns whether it did."""
|
|
184
|
+
return self._idle.wait(timeout)
|
|
185
|
+
|
|
186
|
+
# -- the sender's side -----------------------------------------------
|
|
187
|
+
|
|
188
|
+
def _run(self) -> None:
|
|
189
|
+
while True:
|
|
190
|
+
taken = [self._queue.get()]
|
|
191
|
+
taken.extend(self._take_more())
|
|
192
|
+
# The sentinel can be picked up mid-batch, not just on its
|
|
193
|
+
# own: a producer racing shutdown puts events behind it.
|
|
194
|
+
# Missing it here would leave the sender running forever.
|
|
195
|
+
stopping = any(entry is _STOP for entry in taken)
|
|
196
|
+
|
|
197
|
+
self._deliver(
|
|
198
|
+
[entry for entry in taken if isinstance(entry, Event)]
|
|
199
|
+
)
|
|
200
|
+
for _ in taken:
|
|
201
|
+
self._queue.task_done()
|
|
202
|
+
if self._queue.empty():
|
|
203
|
+
self._idle.set()
|
|
204
|
+
if stopping:
|
|
205
|
+
return
|
|
206
|
+
|
|
207
|
+
def _take_more(self) -> list[object]:
|
|
208
|
+
"""Drain what is already waiting, up to a batch."""
|
|
209
|
+
more: list[object] = []
|
|
210
|
+
while len(more) + 1 < self.batch_size:
|
|
211
|
+
try:
|
|
212
|
+
more.append(self._queue.get_nowait())
|
|
213
|
+
except queue.Empty:
|
|
214
|
+
break
|
|
215
|
+
return more
|
|
216
|
+
|
|
217
|
+
def _deliver(self, events: list[Event]) -> None:
|
|
218
|
+
if not events:
|
|
219
|
+
return
|
|
220
|
+
for publisher in self.publishers:
|
|
221
|
+
try:
|
|
222
|
+
publisher.publish(events)
|
|
223
|
+
except Exception as exc:
|
|
224
|
+
# One bad delivery is not a reason to stop watching.
|
|
225
|
+
self.stats.errors += 1
|
|
226
|
+
self.stats.last_error = repr(exc)
|
|
227
|
+
LOG.exception("publisher %r failed", publisher)
|
|
228
|
+
else:
|
|
229
|
+
self.stats.delivered += len(events)
|