taskflow-meter 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. taskflow_meter/__init__.py +47 -0
  2. taskflow_meter/_version.py +24 -0
  3. taskflow_meter/api/__init__.py +35 -0
  4. taskflow_meter/api/asgi.py +236 -0
  5. taskflow_meter/api/dispatch.py +128 -0
  6. taskflow_meter/api/http.py +210 -0
  7. taskflow_meter/api/router.py +113 -0
  8. taskflow_meter/api/routes.py +53 -0
  9. taskflow_meter/api/serializers.py +137 -0
  10. taskflow_meter/api/service.py +189 -0
  11. taskflow_meter/api/sse.py +222 -0
  12. taskflow_meter/api/wsgi.py +145 -0
  13. taskflow_meter/cli.py +287 -0
  14. taskflow_meter/collect/__init__.py +31 -0
  15. taskflow_meter/collect/attachment.py +208 -0
  16. taskflow_meter/collect/listener.py +161 -0
  17. taskflow_meter/collect/pipeline.py +229 -0
  18. taskflow_meter/collect/progress.py +170 -0
  19. taskflow_meter/conf.py +173 -0
  20. taskflow_meter/contrib/__init__.py +18 -0
  21. taskflow_meter/contrib/django.py +160 -0
  22. taskflow_meter/contrib/fastapi.py +149 -0
  23. taskflow_meter/contrib/flask.py +140 -0
  24. taskflow_meter/contrib/paste.py +96 -0
  25. taskflow_meter/contrib/pecan.py +84 -0
  26. taskflow_meter/datasource/__init__.py +33 -0
  27. taskflow_meter/datasource/base.py +154 -0
  28. taskflow_meter/datasource/memory.py +232 -0
  29. taskflow_meter/datasource/persistence.py +311 -0
  30. taskflow_meter/datasource/sqlalchemy/__init__.py +21 -0
  31. taskflow_meter/datasource/sqlalchemy/migrations/env.py +68 -0
  32. taskflow_meter/datasource/sqlalchemy/migrations/script.py.mako +25 -0
  33. taskflow_meter/datasource/sqlalchemy/migrations/versions/0001_initial.py +71 -0
  34. taskflow_meter/datasource/sqlalchemy/models.py +63 -0
  35. taskflow_meter/datasource/sqlalchemy/source.py +367 -0
  36. taskflow_meter/diff.py +223 -0
  37. taskflow_meter/events.py +129 -0
  38. taskflow_meter/fold.py +137 -0
  39. taskflow_meter/meter.py +255 -0
  40. taskflow_meter/models.py +143 -0
  41. taskflow_meter/poller.py +191 -0
  42. taskflow_meter/py.typed +0 -0
  43. taskflow_meter/states.py +60 -0
  44. taskflow_meter/transports/__init__.py +21 -0
  45. taskflow_meter/transports/amqp.py +204 -0
  46. taskflow_meter/transports/base.py +105 -0
  47. taskflow_meter/transports/http.py +97 -0
  48. taskflow_meter/transports/memory.py +83 -0
  49. taskflow_meter-1.0.0.dist-info/METADATA +258 -0
  50. taskflow_meter-1.0.0.dist-info/RECORD +53 -0
  51. taskflow_meter-1.0.0.dist-info/WHEEL +4 -0
  52. taskflow_meter-1.0.0.dist-info/entry_points.txt +19 -0
  53. taskflow_meter-1.0.0.dist-info/licenses/LICENSE +176 -0
@@ -0,0 +1,71 @@
1
+ # Licensed under the Apache License, Version 2.0 (the "License"); you may
2
+ # not use this file except in compliance with the License. You may obtain
3
+ # a copy of the License at
4
+ #
5
+ # http://www.apache.org/licenses/LICENSE-2.0
6
+ #
7
+ # Unless required by applicable law or agreed to in writing, software
8
+ # distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
9
+ # WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
10
+ # License for the specific language governing permissions and limitations
11
+ # under the License.
12
+
13
+ """The initial schema.
14
+
15
+ Revision ID: 0001
16
+ Revises:
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import sqlalchemy as sa
22
+ from alembic import op
23
+
24
+ revision = "0001"
25
+ down_revision = None
26
+ branch_labels = None
27
+ depends_on = None
28
+
29
+
30
+ def upgrade() -> None:
31
+ op.create_table(
32
+ "taskflow_meter_flows",
33
+ sa.Column("run_id", sa.String(64), primary_key=True),
34
+ sa.Column("book_id", sa.String(64), index=True),
35
+ sa.Column("book_name", sa.String(255)),
36
+ sa.Column("name", sa.String(255), nullable=False),
37
+ sa.Column("state", sa.String(32), index=True),
38
+ sa.Column("observed_at", sa.Float, nullable=False),
39
+ sa.Column("meta", sa.JSON, nullable=False),
40
+ sa.Column("atoms", sa.JSON, nullable=False),
41
+ )
42
+ op.create_index(
43
+ "ix_taskflow_meter_flows_listing",
44
+ "taskflow_meter_flows",
45
+ ["observed_at", "run_id"],
46
+ )
47
+ op.create_table(
48
+ "taskflow_meter_events",
49
+ sa.Column("run_id", sa.String(64), primary_key=True),
50
+ sa.Column("seq", sa.Integer, primary_key=True),
51
+ sa.Column("ts", sa.Float, nullable=False),
52
+ sa.Column("kind", sa.String(32), nullable=False),
53
+ sa.Column("book_id", sa.String(64)),
54
+ sa.Column("atom_name", sa.String(255)),
55
+ sa.Column("atom_uuid", sa.String(64)),
56
+ sa.Column("atom_type", sa.String(16)),
57
+ sa.Column("state", sa.String(32)),
58
+ sa.Column("old_state", sa.String(32)),
59
+ sa.Column("intention", sa.String(16)),
60
+ sa.Column("progress", sa.Float),
61
+ sa.Column("details", sa.JSON, nullable=False),
62
+ )
63
+
64
+
65
+ def downgrade() -> None:
66
+ op.drop_table("taskflow_meter_events")
67
+ op.drop_index(
68
+ "ix_taskflow_meter_flows_listing",
69
+ table_name="taskflow_meter_flows",
70
+ )
71
+ op.drop_table("taskflow_meter_flows")
@@ -0,0 +1,63 @@
1
+ # Licensed under the Apache License, Version 2.0 (the "License"); you may
2
+ # not use this file except in compliance with the License. You may obtain
3
+ # a copy of the License at
4
+ #
5
+ # http://www.apache.org/licenses/LICENSE-2.0
6
+ #
7
+ # Unless required by applicable law or agreed to in writing, software
8
+ # distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
9
+ # WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
10
+ # License for the specific language governing permissions and limitations
11
+ # under the License.
12
+
13
+ """The meter's own schema.
14
+
15
+ Two tables. A flow's atoms live as JSON inside its row rather than in
16
+ a table of their own: every read wants the whole snapshot, and nothing
17
+ in the API filters or joins on an individual atom. The cost of that
18
+ choice is that a query like "which flows have a failed atom" would need
19
+ a schema change -- worth knowing before someone needs one.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import sqlalchemy as sa
25
+
26
+ metadata = sa.MetaData()
27
+
28
+ #: The current state of every run the collector has seen.
29
+ flows = sa.Table(
30
+ "taskflow_meter_flows",
31
+ metadata,
32
+ sa.Column("run_id", sa.String(64), primary_key=True),
33
+ sa.Column("book_id", sa.String(64), index=True),
34
+ sa.Column("book_name", sa.String(255)),
35
+ sa.Column("name", sa.String(255), nullable=False, default=""),
36
+ sa.Column("state", sa.String(32), index=True),
37
+ sa.Column("observed_at", sa.Float, nullable=False),
38
+ sa.Column("meta", sa.JSON, nullable=False),
39
+ sa.Column("atoms", sa.JSON, nullable=False),
40
+ # Listing is always newest first, and paging breaks ties on run_id.
41
+ sa.Index("ix_taskflow_meter_flows_listing", "observed_at", "run_id"),
42
+ )
43
+
44
+ #: Every event, keyed so that re-applying one is a no-op rather than a
45
+ #: duplicate -- a collector that reconnects and replays must not
46
+ #: double-count.
47
+ events = sa.Table(
48
+ "taskflow_meter_events",
49
+ metadata,
50
+ sa.Column("run_id", sa.String(64), primary_key=True),
51
+ sa.Column("seq", sa.Integer, primary_key=True),
52
+ sa.Column("ts", sa.Float, nullable=False),
53
+ sa.Column("kind", sa.String(32), nullable=False),
54
+ sa.Column("book_id", sa.String(64)),
55
+ sa.Column("atom_name", sa.String(255)),
56
+ sa.Column("atom_uuid", sa.String(64)),
57
+ sa.Column("atom_type", sa.String(16)),
58
+ sa.Column("state", sa.String(32)),
59
+ sa.Column("old_state", sa.String(32)),
60
+ sa.Column("intention", sa.String(16)),
61
+ sa.Column("progress", sa.Float),
62
+ sa.Column("details", sa.JSON, nullable=False),
63
+ )
@@ -0,0 +1,367 @@
1
+ # Licensed under the Apache License, Version 2.0 (the "License"); you may
2
+ # not use this file except in compliance with the License. You may obtain
3
+ # a copy of the License at
4
+ #
5
+ # http://www.apache.org/licenses/LICENSE-2.0
6
+ #
7
+ # Unless required by applicable law or agreed to in writing, software
8
+ # distributed under the License is distributed on an "AS IS" BASIS, WITHOUT
9
+ # WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the
10
+ # License for the specific language governing permissions and limitations
11
+ # under the License.
12
+
13
+ """A datasource with a schema of its own.
14
+
15
+ Where the persistence datasource reads what taskflow happens to keep,
16
+ this one owns what it stores: the full event history, real filtering
17
+ and paging in SQL rather than a full scan, and a retention policy the
18
+ deployment sets rather than inherits.
19
+
20
+ It is the far end of the collector deployment -- one process consumes
21
+ events from a transport and writes them here, and any number of API
22
+ workers read from it with ``Meter(poll=False)``.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import contextlib
28
+ import logging
29
+ from collections.abc import Iterable
30
+ from collections.abc import Iterator
31
+ from dataclasses import asdict
32
+ from pathlib import Path
33
+ from typing import Any
34
+
35
+ import sqlalchemy as sa
36
+
37
+ from taskflow_meter.datasource.base import DEFAULT_EVENT_LIMIT
38
+ from taskflow_meter.datasource.base import DEFAULT_FLOW_LIMIT
39
+ from taskflow_meter.datasource.base import EventPage
40
+ from taskflow_meter.datasource.base import FlowPage
41
+ from taskflow_meter.datasource.base import UnknownMarkerError
42
+ from taskflow_meter.datasource.base import WritableDataSource
43
+ from taskflow_meter.datasource.sqlalchemy.models import events as events_t
44
+ from taskflow_meter.datasource.sqlalchemy.models import flows as flows_t
45
+ from taskflow_meter.datasource.sqlalchemy.models import metadata
46
+ from taskflow_meter.events import Event
47
+ from taskflow_meter.events import EventKind
48
+ from taskflow_meter.fold import contiguous_from
49
+ from taskflow_meter.fold import flow_from_event
50
+ from taskflow_meter.fold import fold
51
+ from taskflow_meter.models import AtomSnapshot
52
+ from taskflow_meter.models import FlowSnapshot
53
+
54
+ LOG = logging.getLogger(__name__)
55
+
56
+ MIGRATIONS = Path(__file__).parent / "migrations"
57
+
58
+
59
+ class SQLADataSource(WritableDataSource):
60
+ """Stores flows and their events in a database we own."""
61
+
62
+ name = "sqlalchemy"
63
+
64
+ def __init__(
65
+ self,
66
+ url: str | None = None,
67
+ *,
68
+ engine: Any = None,
69
+ create_schema: bool = False,
70
+ ) -> None:
71
+ """Connect to ``url``, or use an ``engine`` somebody else owns.
72
+
73
+ ``create_schema`` is for tests and throwaway databases; a real
74
+ deployment runs :func:`upgrade` so the schema has a version.
75
+ """
76
+ if (url is None) == (engine is None):
77
+ msg = "pass exactly one of url or engine"
78
+ raise ValueError(msg)
79
+ self._url = url
80
+ self._engine = engine
81
+ self._owns_engine = engine is None
82
+ if create_schema:
83
+ metadata.create_all(self.engine)
84
+
85
+ @property
86
+ def engine(self) -> Any:
87
+ if self._engine is None:
88
+ self._engine = sa.create_engine(str(self._url))
89
+ return self._engine
90
+
91
+ def stop(self) -> None:
92
+ if self._owns_engine and self._engine is not None:
93
+ self._engine.dispose()
94
+ self._engine = None
95
+
96
+ # -- writing ---------------------------------------------------------
97
+
98
+ def apply(self, event: Event) -> None:
99
+ self.apply_many([event])
100
+
101
+ def apply_many(self, events: Iterable[Event]) -> None:
102
+ """Fold a batch into the stored state, in one transaction.
103
+
104
+ Grouped by run so a batch touching several flows still reads and
105
+ rewrites each row once.
106
+ """
107
+ batch = list(events)
108
+ if not batch:
109
+ return
110
+
111
+ by_run: dict[str, list[Event]] = {}
112
+ for event in batch:
113
+ by_run.setdefault(event.run_id, []).append(event)
114
+
115
+ with self.engine.begin() as conn:
116
+ for run_id, run_events in by_run.items():
117
+ self._apply_run(conn, run_id, run_events)
118
+
119
+ def _apply_run(
120
+ self, conn: Any, run_id: str, run_events: list[Event]
121
+ ) -> None:
122
+ snapshot = self._load_flow(conn, run_id)
123
+ if snapshot is None:
124
+ snapshot = flow_from_event(run_events[0])
125
+ for event in run_events:
126
+ snapshot = fold(snapshot, event)
127
+ self._save_flow(conn, snapshot)
128
+
129
+ for event in run_events:
130
+ row = _event_row(event)
131
+ try:
132
+ with conn.begin_nested():
133
+ conn.execute(sa.insert(events_t).values(**row))
134
+ except sa.exc.IntegrityError:
135
+ # (run_id, seq) already stored: a collector that
136
+ # reconnected and replayed, which must be a no-op
137
+ # rather than a duplicate.
138
+ LOG.debug(
139
+ "event %d for run %s is already stored",
140
+ event.seq,
141
+ run_id,
142
+ )
143
+
144
+ def _save_flow(self, conn: Any, snapshot: FlowSnapshot) -> None:
145
+ values = {
146
+ "run_id": snapshot.run_id,
147
+ "book_id": snapshot.book_id,
148
+ "book_name": snapshot.book_name,
149
+ "name": snapshot.name,
150
+ "state": snapshot.state,
151
+ "observed_at": snapshot.observed_at,
152
+ "meta": snapshot.meta,
153
+ "atoms": {
154
+ name: asdict(atom) for name, atom in snapshot.atoms.items()
155
+ },
156
+ }
157
+ updated = conn.execute(
158
+ sa.update(flows_t)
159
+ .where(flows_t.c.run_id == snapshot.run_id)
160
+ .values(**values)
161
+ )
162
+ if updated.rowcount == 0:
163
+ conn.execute(sa.insert(flows_t).values(**values))
164
+
165
+ def forget(self, run_id: str) -> bool:
166
+ """Drop a run and its events. Returns whether it was there."""
167
+ with self.engine.begin() as conn:
168
+ conn.execute(
169
+ sa.delete(events_t).where(events_t.c.run_id == run_id)
170
+ )
171
+ deleted = conn.execute(
172
+ sa.delete(flows_t).where(flows_t.c.run_id == run_id)
173
+ )
174
+ return bool(deleted.rowcount)
175
+
176
+ def prune(self, before: float) -> int:
177
+ """Drop runs last observed before ``before``.
178
+
179
+ This datasource keeps what it is told forever otherwise -- the
180
+ deployment owns the retention policy, unlike the persistence
181
+ source, which inherits taskflow's.
182
+ """
183
+ with self.engine.begin() as conn:
184
+ stale = [
185
+ row.run_id
186
+ for row in conn.execute(
187
+ sa.select(flows_t.c.run_id).where(
188
+ flows_t.c.observed_at < before
189
+ )
190
+ )
191
+ ]
192
+ if not stale:
193
+ return 0
194
+ conn.execute(
195
+ sa.delete(events_t).where(events_t.c.run_id.in_(stale))
196
+ )
197
+ conn.execute(sa.delete(flows_t).where(flows_t.c.run_id.in_(stale)))
198
+ return len(stale)
199
+
200
+ # -- reading ---------------------------------------------------------
201
+
202
+ def get_flow(self, run_id: str) -> FlowSnapshot | None:
203
+ with self._connect() as conn:
204
+ return self._load_flow(conn, run_id)
205
+
206
+ def list_flows(
207
+ self,
208
+ *,
209
+ state: str | None = None,
210
+ book_id: str | None = None,
211
+ limit: int = DEFAULT_FLOW_LIMIT,
212
+ marker: str | None = None,
213
+ ) -> FlowPage:
214
+ if limit < 1:
215
+ msg = "limit must be at least 1"
216
+ raise ValueError(msg)
217
+
218
+ query = sa.select(flows_t)
219
+ if state is not None:
220
+ query = query.where(flows_t.c.state == state)
221
+ if book_id is not None:
222
+ query = query.where(flows_t.c.book_id == book_id)
223
+
224
+ with self._connect() as conn:
225
+ if marker is not None:
226
+ anchor = conn.execute(
227
+ sa.select(flows_t.c.observed_at, flows_t.c.run_id).where(
228
+ flows_t.c.run_id == marker
229
+ )
230
+ ).first()
231
+ if anchor is None:
232
+ msg = f"unknown paging marker: {marker!r}"
233
+ raise UnknownMarkerError(msg)
234
+ # Keyset paging in the listing order: newest first,
235
+ # ties broken by run_id ascending.
236
+ query = query.where(
237
+ sa.or_(
238
+ flows_t.c.observed_at < anchor.observed_at,
239
+ sa.and_(
240
+ flows_t.c.observed_at == anchor.observed_at,
241
+ flows_t.c.run_id > anchor.run_id,
242
+ ),
243
+ )
244
+ )
245
+
246
+ query = query.order_by(
247
+ flows_t.c.observed_at.desc(), flows_t.c.run_id.asc()
248
+ ).limit(limit + 1)
249
+ rows = list(conn.execute(query))
250
+
251
+ more = len(rows) > limit
252
+ window = [_flow_from_row(row) for row in rows[:limit]]
253
+ return FlowPage(
254
+ items=tuple(window),
255
+ next_marker=window[-1].run_id if more and window else None,
256
+ )
257
+
258
+ def events_since(
259
+ self,
260
+ run_id: str,
261
+ *,
262
+ since_seq: int = 0,
263
+ limit: int = DEFAULT_EVENT_LIMIT,
264
+ ) -> EventPage:
265
+ if limit < 1:
266
+ msg = "limit must be at least 1"
267
+ raise ValueError(msg)
268
+
269
+ with self._connect() as conn:
270
+ oldest = conn.execute(
271
+ sa.select(sa.func.min(events_t.c.seq)).where(
272
+ events_t.c.run_id == run_id
273
+ )
274
+ ).scalar()
275
+ rows = list(
276
+ conn.execute(
277
+ sa.select(events_t)
278
+ .where(
279
+ events_t.c.run_id == run_id,
280
+ events_t.c.seq > since_seq,
281
+ )
282
+ .order_by(events_t.c.seq.asc())
283
+ .limit(limit)
284
+ )
285
+ )
286
+
287
+ truncated = oldest is not None and since_seq + 1 < oldest
288
+ expected = (
289
+ oldest if truncated and oldest is not None else since_seq + 1
290
+ )
291
+ selected = contiguous_from(
292
+ [_event_from_row(row) for row in rows], expected, limit
293
+ )
294
+ return EventPage(
295
+ events=tuple(selected),
296
+ next_seq=selected[-1].seq if selected else since_seq,
297
+ oldest_seq=oldest,
298
+ truncated=truncated,
299
+ )
300
+
301
+ # -- internals -------------------------------------------------------
302
+
303
+ @contextlib.contextmanager
304
+ def _connect(self) -> Iterator[Any]:
305
+ with self.engine.connect() as conn:
306
+ yield conn
307
+
308
+ def _load_flow(self, conn: Any, run_id: str) -> FlowSnapshot | None:
309
+ row = conn.execute(
310
+ sa.select(flows_t).where(flows_t.c.run_id == run_id)
311
+ ).first()
312
+ return None if row is None else _flow_from_row(row)
313
+
314
+
315
+ def _flow_from_row(row: Any) -> FlowSnapshot:
316
+ return FlowSnapshot(
317
+ run_id=row.run_id,
318
+ name=row.name,
319
+ state=row.state,
320
+ book_id=row.book_id,
321
+ book_name=row.book_name,
322
+ observed_at=row.observed_at,
323
+ meta=dict(row.meta or {}),
324
+ atoms={
325
+ name: AtomSnapshot(**data)
326
+ for name, data in (row.atoms or {}).items()
327
+ },
328
+ )
329
+
330
+
331
+ def _event_row(event: Event) -> dict[str, Any]:
332
+ row = asdict(event)
333
+ row["kind"] = str(event.kind)
334
+ return row
335
+
336
+
337
+ def _event_from_row(row: Any) -> Event:
338
+ return Event(
339
+ run_id=row.run_id,
340
+ seq=row.seq,
341
+ ts=row.ts,
342
+ kind=EventKind(row.kind),
343
+ book_id=row.book_id,
344
+ atom_name=row.atom_name,
345
+ atom_uuid=row.atom_uuid,
346
+ atom_type=row.atom_type,
347
+ state=row.state,
348
+ old_state=row.old_state,
349
+ intention=row.intention,
350
+ progress=row.progress,
351
+ details=dict(row.details or {}),
352
+ )
353
+
354
+
355
+ def upgrade(url: str, revision: str = "head") -> None:
356
+ """Bring a database up to date, the way a deployment should.
357
+
358
+ Imported lazily: alembic is only needed by whoever runs migrations,
359
+ not by every process that reads the results.
360
+ """
361
+ from alembic import command
362
+ from alembic.config import Config
363
+
364
+ config = Config()
365
+ config.set_main_option("script_location", str(MIGRATIONS))
366
+ config.set_main_option("sqlalchemy.url", url)
367
+ command.upgrade(config, revision)