dirigent-core 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. dirigent_core/__init__.py +26 -0
  2. dirigent_core/alembic/env.py +71 -0
  3. dirigent_core/alembic/script.py.mako +26 -0
  4. dirigent_core/alembic/versions/0001_baseline_schema.py +1016 -0
  5. dirigent_core/alerting.py +805 -0
  6. dirigent_core/artifacts.py +73 -0
  7. dirigent_core/auth.py +529 -0
  8. dirigent_core/blockdocs.py +223 -0
  9. dirigent_core/config.py +421 -0
  10. dirigent_core/configdocs.py +134 -0
  11. dirigent_core/database.py +177 -0
  12. dirigent_core/directory.py +339 -0
  13. dirigent_core/documents.py +878 -0
  14. dirigent_core/documentschema.py +92 -0
  15. dirigent_core/engine/__init__.py +120 -0
  16. dirigent_core/engine/claim.py +156 -0
  17. dirigent_core/engine/context.py +420 -0
  18. dirigent_core/engine/definition.py +543 -0
  19. dirigent_core/engine/executor.py +1029 -0
  20. dirigent_core/engine/failure.py +71 -0
  21. dirigent_core/engine/recovery.py +137 -0
  22. dirigent_core/engine/references.py +239 -0
  23. dirigent_core/engine/runs.py +865 -0
  24. dirigent_core/engine/services.py +74 -0
  25. dirigent_core/engine/state.py +434 -0
  26. dirigent_core/ids.py +39 -0
  27. dirigent_core/logging.py +310 -0
  28. dirigent_core/migrations.py +98 -0
  29. dirigent_core/models.py +626 -0
  30. dirigent_core/pipelines.py +494 -0
  31. dirigent_core/plugins.py +255 -0
  32. dirigent_core/protocol.py +276 -0
  33. dirigent_core/py.typed +0 -0
  34. dirigent_core/ratelimit.py +42 -0
  35. dirigent_core/registry.py +32 -0
  36. dirigent_core/retention.py +293 -0
  37. dirigent_core/scheduler.py +491 -0
  38. dirigent_core/schemas.py +147 -0
  39. dirigent_core/secrets.py +164 -0
  40. dirigent_core/storage.py +349 -0
  41. dirigent_core/telemetry.py +402 -0
  42. dirigent_core/trigger_documents.py +193 -0
  43. dirigent_core/triggers/__init__.py +117 -0
  44. dirigent_core/triggers/backfill.py +131 -0
  45. dirigent_core/triggers/materialize.py +218 -0
  46. dirigent_core/triggers/schedules.py +531 -0
  47. dirigent_core/triggers/webhooks.py +586 -0
  48. dirigent_core/types.py +59 -0
  49. dirigent_core/worker.py +350 -0
  50. dirigent_core-0.9.0.dist-info/METADATA +29 -0
  51. dirigent_core-0.9.0.dist-info/RECORD +53 -0
  52. dirigent_core-0.9.0.dist-info/WHEEL +4 -0
  53. dirigent_core-0.9.0.dist-info/licenses/LICENSE +18 -0
@@ -0,0 +1,805 @@
1
+ """Alerting rules and channels as data, with delivery queued and retried like any other work.
2
+
3
+ A rule says one thing about one run once: the unique constraint on
4
+ ``(alert_rule_id, run_id, event)`` is the deduplication, and the throttle window is a second,
5
+ coarser guard against a flapping pipeline.
6
+ """
7
+
8
+ import asyncio
9
+ from collections.abc import AsyncGenerator, Sequence
10
+ from contextlib import asynccontextmanager, suppress
11
+ from datetime import datetime, timedelta
12
+ from typing import Final, cast
13
+ from uuid import UUID
14
+
15
+ import sqlalchemy as sa
16
+ from pydantic import BaseModel, ConfigDict, Field, JsonValue
17
+ from sqlalchemy.exc import IntegrityError
18
+ from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker
19
+
20
+ from dirigent_client.enums import AlertEvent, AlertScope, LogLevel, NotificationStatus, RunStatus
21
+ from dirigent_common import EntityName, JsonMap, format_duration
22
+ from dirigent_core.database import session_scope
23
+ from dirigent_core.engine.references import substitute
24
+ from dirigent_core.engine.services import EngineServices
25
+ from dirigent_core.logging import get_logger
26
+ from dirigent_core.models import AlertRule, Connection, LogEntry, Notification, Pipeline, Run, utcnow
27
+ from dirigent_plugin import AlertMessage, Notifier
28
+
29
+ EVENT_FOR_STATUS: dict[RunStatus, AlertEvent] = {
30
+ RunStatus.FAILED: AlertEvent.RUN_FAILED,
31
+ RunStatus.COMPLETED_WITH_ERRORS: AlertEvent.RUN_COMPLETED_WITH_ERRORS,
32
+ RunStatus.SUCCEEDED: AlertEvent.RUN_SUCCEEDED,
33
+ }
34
+
35
+ DEFAULT_TEMPLATE = "${run.pipeline} run ${run.status}"
36
+
37
+ CLAIM_LIMIT = 1
38
+
39
+ #: How many times a lease is renewed over its own length while a delivery is in flight.
40
+ RENEWALS_PER_LEASE = 3
41
+
42
+ _logger = get_logger("alerting")
43
+
44
+
45
+ class AlertError(Exception):
46
+ """An alert rule could not be declared, found, or delivered."""
47
+
48
+
49
+ class AlertRuleRequest(BaseModel):
50
+ """What it takes to declare an alert rule, from the API or the CLI."""
51
+
52
+ model_config = ConfigDict(frozen=True)
53
+
54
+ code: EntityName
55
+ name: str | None = None
56
+ description: str | None = None
57
+ event: AlertEvent
58
+ notifier: str
59
+ scope: AlertScope = AlertScope.GLOBAL
60
+ pipeline: str | None = None
61
+ connection: str | None = None
62
+ template: str | None = None
63
+ throttle: timedelta = timedelta(0)
64
+
65
+
66
+ #: Stands in for "this path names nothing", which is not the same as naming a null.
67
+ MISSING: Final = object()
68
+
69
+
70
+ def render_template(template: str, context: JsonMap) -> str:
71
+ """Render an alert's message, resolving ``${run.*}`` against the run's own facts.
72
+
73
+ An unresolvable reference is left verbatim rather than raised or blanked, so a typo in a
74
+ template never costs the message. A reference resolving to null renders as empty.
75
+ """
76
+
77
+ def one(reference: str) -> str:
78
+ resolved = _walk(context, reference.strip())
79
+ return f"${{{reference}}}" if resolved is MISSING else _as_text(cast("JsonValue", resolved))
80
+
81
+ return substitute(template, one)
82
+
83
+
84
+ def _walk(context: JsonMap, reference: str) -> object:
85
+ """Walk a dotted path into the alert context, or return MISSING when it names nothing."""
86
+ parts = [piece for piece in reference.split(".") if piece]
87
+ if not parts:
88
+ return MISSING
89
+ current: JsonValue = context
90
+ for part in parts:
91
+ if not isinstance(current, dict) or part not in current:
92
+ return MISSING
93
+ current = current[part]
94
+ return current
95
+
96
+
97
+ def _as_text(value: JsonValue) -> str:
98
+ """Render a resolved value the way a message wants to read it."""
99
+ if isinstance(value, bool):
100
+ return "true" if value else "false"
101
+ return "" if value is None else str(value)
102
+
103
+
104
+ def build_context(run: Run, pipeline: Pipeline, *, base_url: str | None = None) -> JsonMap:
105
+ """Gather the facts a template may read, under the one namespace it has.
106
+
107
+ A flat snapshot rather than a live handle: the notification is delivered later, and must
108
+ describe the run as it was when the alert fired.
109
+ """
110
+ duration = (run.finished_at - run.started_at).total_seconds() if run.finished_at and run.started_at else None
111
+ return {
112
+ "run": {
113
+ "id": str(run.id),
114
+ "status": run.status.value,
115
+ "pipeline": pipeline.code,
116
+ "error": run.error,
117
+ "params": dict(run.params),
118
+ "trigger": run.triggered_by_label or run.triggered_by_kind.value,
119
+ "started_at": run.started_at.isoformat() if run.started_at else None,
120
+ "finished_at": run.finished_at.isoformat() if run.finished_at else None,
121
+ "duration_ms": round(duration * 1000) if duration is not None else None,
122
+ "url": f"{base_url.rstrip('/')}/runs/{run.id}" if base_url else None,
123
+ }
124
+ }
125
+
126
+
127
+ async def find_rule(session: AsyncSession, code: str) -> AlertRule | None:
128
+ """Find one alert rule by code."""
129
+ found = await session.execute(sa.select(AlertRule).where(AlertRule.code == code))
130
+ return found.scalar_one_or_none()
131
+
132
+
133
+ async def list_rules(
134
+ session: AsyncSession,
135
+ *,
136
+ after: UUID | None = None,
137
+ limit: int | None = None,
138
+ ) -> list[AlertRule]:
139
+ """List every alert rule in id order, which is the order they were declared in."""
140
+ statement = sa.select(AlertRule).order_by(AlertRule.id)
141
+ if after is not None:
142
+ statement = statement.where(AlertRule.id > after)
143
+ if limit is not None:
144
+ statement = statement.limit(limit)
145
+ rows = await session.execute(statement)
146
+ return list(rows.scalars())
147
+
148
+
149
+ async def create_rule(session: AsyncSession, services: EngineServices, request: AlertRuleRequest) -> AlertRule:
150
+ """Declare an alert rule, refusing a notifier or a scope this instance cannot honour."""
151
+ if request.notifier not in services.host.notifiers:
152
+ installed = ", ".join(sorted(services.host.notifiers)) or "none are installed"
153
+ raise AlertError(f"no notifier {request.notifier!r} is installed ({installed})")
154
+ if await find_rule(session, request.code) is not None:
155
+ raise AlertError(f"an alert rule coded {request.code!r} already exists")
156
+ pipeline_id = await _scope_pipeline(session, request)
157
+ connection_id = await _connection_id(session, request.connection) if request.connection else None
158
+ rule = AlertRule(
159
+ code=request.code,
160
+ name=request.name,
161
+ description=request.description,
162
+ event=request.event,
163
+ scope=request.scope,
164
+ pipeline_id=pipeline_id,
165
+ notifier=request.notifier,
166
+ connection_id=connection_id,
167
+ template=request.template,
168
+ throttle_seconds=int(request.throttle.total_seconds()),
169
+ )
170
+ session.add(rule)
171
+ await session.flush()
172
+ _logger.info("alert rule created", rule=rule.code, alert_event=rule.event.value, notifier=rule.notifier)
173
+ return rule
174
+
175
+
176
+ async def _scope_pipeline(session: AsyncSession, request: AlertRuleRequest) -> UUID | None:
177
+ """Resolve the pipeline a rule is scoped to, refusing a scope that names nothing."""
178
+ if request.scope is AlertScope.GLOBAL:
179
+ return None
180
+ if not request.pipeline:
181
+ raise AlertError("a pipeline-scoped rule has to name the pipeline it watches")
182
+ found = await session.execute(sa.select(Pipeline).where(Pipeline.code == request.pipeline))
183
+ pipeline = found.scalar_one_or_none()
184
+ if pipeline is None:
185
+ raise AlertError(f"no pipeline coded {request.pipeline!r}")
186
+ return pipeline.id
187
+
188
+
189
+ async def _connection_id(session: AsyncSession, code: str) -> UUID:
190
+ """Resolve the connection a notifier delivers through, refusing an unknown code."""
191
+ found = await session.execute(sa.select(Connection).where(Connection.code == code))
192
+ connection = found.scalar_one_or_none()
193
+ if connection is None:
194
+ raise AlertError(f"no connection coded {code!r}")
195
+ return connection.id
196
+
197
+
198
+ async def delete_rule(session: AsyncSession, rule: AlertRule) -> None:
199
+ """Remove an alert rule; the notifications it already raised are kept."""
200
+ code = rule.code
201
+ await session.delete(rule)
202
+ await session.flush()
203
+ _logger.info("alert rule deleted", rule=code)
204
+
205
+
206
+ async def matching_rules(session: AsyncSession, event: AlertEvent, pipeline_id: UUID) -> list[AlertRule]:
207
+ """Find the live rules that want to hear about one event, throttled or not.
208
+
209
+ A paused rule matches nothing. Pausing is instance state an operator sets on the row, so
210
+ it is read here rather than folded into ``active``, which is what the rule itself declares.
211
+ """
212
+ rows = await session.execute(
213
+ sa.select(AlertRule).where(
214
+ AlertRule.event == event,
215
+ AlertRule.active.is_(True),
216
+ AlertRule.paused.is_(False),
217
+ sa.or_(AlertRule.scope == AlertScope.GLOBAL, AlertRule.pipeline_id == pipeline_id),
218
+ )
219
+ )
220
+ return list(rows.scalars())
221
+
222
+
223
+ async def set_paused(session: AsyncSession, rule: AlertRule, *, paused: bool) -> AlertRule:
224
+ """Hold a rule's deliveries, or let them resume; the rule itself is left as declared."""
225
+ rule.paused = paused
226
+ await session.flush()
227
+ _logger.info("alert rule paused" if paused else "alert rule resumed", rule=rule.code)
228
+ return rule
229
+
230
+
231
+ async def raised_at(session: AsyncSession, rule: AlertRule, pipeline_id: UUID) -> datetime | None:
232
+ """When this rule last raised anything **for this pipeline**.
233
+
234
+ A global rule watches every pipeline, and a window measured across all of them lets one
235
+ noisy pipeline silence the rest. The window is per pipeline, which is what an operator
236
+ means by "not more than one of these an hour".
237
+
238
+ Read from ``created_at``, which nothing moves: ``available_at`` is rewritten by every
239
+ delivery retry and by every recovery, so a channel that is down would widen the window
240
+ it was raised in.
241
+ """
242
+ return (
243
+ await session.execute(
244
+ sa.select(sa.func.max(Notification.created_at))
245
+ .join(Run, Run.id == Notification.run_id)
246
+ .where(Notification.alert_rule_id == rule.id, Run.pipeline_id == pipeline_id)
247
+ )
248
+ ).scalar()
249
+
250
+
251
+ async def throttled_until(session: AsyncSession, rule: AlertRule, pipeline_id: UUID, now: datetime) -> datetime | None:
252
+ """Say when this rule may next raise for this pipeline, or ``None`` if it may now.
253
+
254
+ Measured from when a message was raised, not from when it was delivered or retried, so
255
+ neither a slow notifier nor a failing one moves the window.
256
+ """
257
+ if not rule.throttle_seconds:
258
+ return None
259
+ last = await raised_at(session, rule, pipeline_id)
260
+ if last is None:
261
+ return None
262
+ opens = last + timedelta(seconds=rule.throttle_seconds)
263
+ return opens if opens > now else None
264
+
265
+
266
+ async def raise_for_run(
267
+ session: AsyncSession,
268
+ services: EngineServices,
269
+ run: Run,
270
+ event: AlertEvent,
271
+ *,
272
+ now: datetime | None = None,
273
+ ) -> list[Notification]:
274
+ """Queue whatever this run's settling owes, in the caller's outcome transaction.
275
+
276
+ Called from the same commit that settles the run, so a run cannot reach a terminal state
277
+ without its alerts having been queued.
278
+ """
279
+ moment = now or utcnow()
280
+ pipeline = await session.get(Pipeline, run.pipeline_id)
281
+ if pipeline is None: # pragma: no cover - the foreign key makes this unreachable
282
+ return []
283
+ rules = await matching_rules(session, event, run.pipeline_id)
284
+ if not rules:
285
+ return []
286
+ context = build_context(run, pipeline, base_url=services.settings.alert_base_url)
287
+ queued: list[Notification] = []
288
+ for rule in rules:
289
+ if await _already_raised(session, rule.id, run.id, event):
290
+ continue
291
+ opens = await throttled_until(session, rule, run.pipeline_id, moment)
292
+ if opens is not None:
293
+ # A suppressed alert that leaves nothing behind is indistinguishable from a rule
294
+ # that never matched, which is the hard way to learn a throttle is too wide.
295
+ session.add(
296
+ LogEntry(
297
+ run_id=run.id,
298
+ level=LogLevel.INFO,
299
+ message=f"alert {rule.code!r} suppressed by its throttle window",
300
+ fields={
301
+ "event": event.value,
302
+ "notifier": rule.notifier,
303
+ "throttle": format_duration(timedelta(seconds=rule.throttle_seconds)),
304
+ "opens_at": opens.isoformat(),
305
+ },
306
+ created_at=moment,
307
+ )
308
+ )
309
+ continue
310
+ subject = render_template(rule.template or DEFAULT_TEMPLATE, context)
311
+ notification = Notification(
312
+ alert_rule_id=rule.id,
313
+ run_id=run.id,
314
+ event=event,
315
+ notifier=rule.notifier,
316
+ connection_id=rule.connection_id,
317
+ subject=subject,
318
+ body=_body(context),
319
+ context=context,
320
+ status=NotificationStatus.PENDING,
321
+ available_at=moment,
322
+ created_at=moment,
323
+ )
324
+ # A savepoint, because two processes can reach this between the check above and the
325
+ # insert, and an IntegrityError raised here would otherwise abort the caller's whole
326
+ # transaction rather than just this insert.
327
+ try:
328
+ async with session.begin_nested():
329
+ session.add(notification)
330
+ session.add(
331
+ LogEntry(
332
+ run_id=run.id,
333
+ level=LogLevel.INFO,
334
+ message=f"alert {rule.code!r} queued for delivery through {rule.notifier!r}",
335
+ fields={"event": event.value, "notifier": rule.notifier},
336
+ created_at=moment,
337
+ )
338
+ )
339
+ except IntegrityError:
340
+ continue
341
+ rule.last_sent_at = moment
342
+ queued.append(notification)
343
+ await session.flush()
344
+ if queued:
345
+ _logger.info("alerts raised", run_id=str(run.id), alert_event=event.value, rules=len(queued))
346
+ return queued
347
+
348
+
349
+ def _body(context: JsonMap) -> str:
350
+ """Render the default body: the run's own facts, one per line, in a stable order."""
351
+ run = cast("JsonMap", context.get("run", {}))
352
+ lines = [f"{name}: {_as_text(value)}" for name, value in run.items() if value not in (None, "", {}, [])]
353
+ return "\n".join(lines)
354
+
355
+
356
+ async def _already_raised(session: AsyncSession, rule_id: UUID, run_id: UUID, event: AlertEvent) -> bool:
357
+ """Report whether this rule has already said this about this run."""
358
+ found = await session.execute(
359
+ sa.select(sa.func.count())
360
+ .select_from(Notification)
361
+ .where(
362
+ Notification.alert_rule_id == rule_id,
363
+ Notification.run_id == run_id,
364
+ Notification.event == event,
365
+ )
366
+ )
367
+ return int(found.scalar_one()) > 0
368
+
369
+
370
+ async def raise_for_status(
371
+ session: AsyncSession,
372
+ services: EngineServices,
373
+ run: Run,
374
+ status: RunStatus,
375
+ *,
376
+ now: datetime | None = None,
377
+ ) -> list[Notification]:
378
+ """Queue the alerts a terminal run status owes, if that status raises an event at all."""
379
+ event = EVENT_FOR_STATUS.get(status)
380
+ if event is None:
381
+ return []
382
+ return await raise_for_run(session, services, run, event, now=now)
383
+
384
+
385
+ async def raise_for_stuck(
386
+ session: AsyncSession,
387
+ services: EngineServices,
388
+ run_ids: Sequence[UUID],
389
+ *,
390
+ now: datetime | None = None,
391
+ ) -> int:
392
+ """Queue ``run_stuck`` alerts for the runs the sweeper found standing still.
393
+
394
+ The sweeper re-detects the same run every sweep; the once-per-rule-per-run constraint is
395
+ what keeps that to one message.
396
+ """
397
+ moment = now or utcnow()
398
+ queued = 0
399
+ for run_id in run_ids:
400
+ run = await session.get(Run, run_id)
401
+ if run is None: # pragma: no cover - the sweeper just read these ids
402
+ continue
403
+ queued += len(await raise_for_run(session, services, run, AlertEvent.RUN_STUCK, now=moment))
404
+ return queued
405
+
406
+
407
+ async def queue_test_message(
408
+ session: AsyncSession,
409
+ services: EngineServices,
410
+ *,
411
+ notifier: str,
412
+ connection: str | None = None,
413
+ subject: str = "dirigent test alert",
414
+ body: str = "This is a test message sent through the notifier surface.",
415
+ ) -> Notification:
416
+ """Queue one unattached message, which is what ``dg alerts test`` sends."""
417
+ if notifier not in services.host.notifiers:
418
+ installed = ", ".join(sorted(services.host.notifiers)) or "none are installed"
419
+ raise AlertError(f"no notifier {notifier!r} is installed ({installed})")
420
+ notification = Notification(
421
+ alert_rule_id=None,
422
+ run_id=None,
423
+ event=AlertEvent.RUN_SUCCEEDED,
424
+ notifier=notifier,
425
+ connection_id=await _connection_id(session, connection) if connection else None,
426
+ subject=subject,
427
+ body=body,
428
+ context={"run": {"pipeline": "(test)", "status": "succeeded"}},
429
+ status=NotificationStatus.PENDING,
430
+ available_at=utcnow(),
431
+ )
432
+ session.add(notification)
433
+ await session.flush()
434
+ return notification
435
+
436
+
437
+ async def claim_notification(
438
+ session: AsyncSession,
439
+ *,
440
+ owner: str,
441
+ now: datetime,
442
+ lease_seconds: int,
443
+ ) -> Notification | None:
444
+ """Claim the next due notification, the same way a worker claims an attempt."""
445
+ statement = (
446
+ sa.select(Notification)
447
+ .where(Notification.status == NotificationStatus.PENDING, Notification.available_at <= now)
448
+ .order_by(Notification.available_at, Notification.id)
449
+ .limit(CLAIM_LIMIT)
450
+ )
451
+ if session.get_bind().dialect.name == "postgresql":
452
+ statement = statement.with_for_update(skip_locked=True, of=Notification)
453
+ found = await session.execute(statement)
454
+ notification = found.scalars().first()
455
+ if notification is None:
456
+ return None
457
+ notification.status = NotificationStatus.SENDING
458
+ notification.lease_owner = owner
459
+ notification.lease_expires_at = now + timedelta(seconds=lease_seconds)
460
+ notification.attempt += 1
461
+ await session.flush()
462
+ return notification
463
+
464
+
465
+ def notifier_config(services: EngineServices, notifier: Notifier, connection: Connection | None) -> BaseModel:
466
+ """Open the connection a notifier delivers through, validated against its own model."""
467
+ if connection is None:
468
+ return notifier.config_model()
469
+ return services.secrets.decrypt_config(
470
+ notifier.config_model, connection.config, connection.secret_envelope, key_id=connection.secret_key_id
471
+ )
472
+
473
+
474
+ async def send_notification(
475
+ session: AsyncSession,
476
+ services: EngineServices,
477
+ notification: Notification,
478
+ *,
479
+ owner: str,
480
+ now: datetime | None = None,
481
+ ) -> bool:
482
+ """Deliver one claimed notification, returning whether it was delivered.
483
+
484
+ The outcome is written only while this worker still holds the lease, so a delivery that
485
+ outlasted its lease cannot overwrite the outcome of the worker the row was handed to.
486
+
487
+ THE SESSION IS COMMITTED BEFORE THE SEND, so no transaction is open while an outbound
488
+ call is in flight: on SQLite that transaction holds the one write lock, and everything
489
+ else on the instance -- the sweeper reclaiming this very row, the API, a worker settling
490
+ an attempt -- would queue behind a notifier that is slow to answer. Nothing read above
491
+ has to be held anyway, because the fence below re-reads the row it writes.
492
+ """
493
+ moment = now or utcnow()
494
+ try:
495
+ notifier = services.host.notifiers.get(notification.notifier)
496
+ if notifier is None:
497
+ raise AlertError(f"notifier {notification.notifier!r} is not installed on this worker")
498
+ connection = await session.get(Connection, notification.connection_id) if notification.connection_id else None
499
+ config = notifier_config(services, notifier, connection)
500
+ await session.commit()
501
+ await notifier.send(_message(notification), config)
502
+ except Exception as error:
503
+ if not await _holds_lease(session, notification, owner):
504
+ return False
505
+ return await _record_failure(session, services, notification, error, moment)
506
+ if not await _holds_lease(session, notification, owner):
507
+ return False
508
+ notification.status = NotificationStatus.SENT
509
+ notification.sent_at = moment
510
+ notification.error = None
511
+ notification.lease_owner = None
512
+ notification.lease_expires_at = None
513
+ _timeline(session, notification, LogLevel.INFO, f"alert delivered through {notification.notifier!r}", moment)
514
+ _logger.info("notification delivered", notification_id=str(notification.id), notifier=notification.notifier)
515
+ return True
516
+
517
+
518
+ async def _holds_lease(session: AsyncSession, notification: Notification, owner: str) -> bool:
519
+ """Re-read the row and report whether this worker may still write this outcome.
520
+
521
+ The lease is a fence, not a hint: a row the sweeper returned to the queue belongs to
522
+ whoever claimed it next, and writing into it would replace a live delivery's outcome
523
+ with the outcome of an abandoned one.
524
+ """
525
+ found = await session.execute(
526
+ sa.select(Notification.status, Notification.lease_owner).where(Notification.id == notification.id)
527
+ )
528
+ row = found.one_or_none()
529
+ if row is not None and row.status is NotificationStatus.SENDING and row.lease_owner == owner:
530
+ return True
531
+ _logger.warning(
532
+ "lease lost, notification outcome discarded",
533
+ notification_id=str(notification.id),
534
+ notifier=notification.notifier,
535
+ holder=None if row is None else row.lease_owner,
536
+ notification_status=None if row is None else row.status.value,
537
+ )
538
+ return False
539
+
540
+
541
+ async def renew_lease(
542
+ session: AsyncSession,
543
+ notification_id: UUID,
544
+ *,
545
+ owner: str,
546
+ lease_seconds: int,
547
+ now: datetime | None = None,
548
+ ) -> bool:
549
+ """Push a notification's lease out, reporting whether this worker still held it."""
550
+ moment = now or utcnow()
551
+ renewed = await session.execute(
552
+ sa.update(Notification)
553
+ .where(
554
+ Notification.id == notification_id,
555
+ Notification.status == NotificationStatus.SENDING,
556
+ Notification.lease_owner == owner,
557
+ )
558
+ .values(lease_expires_at=moment + timedelta(seconds=lease_seconds))
559
+ .returning(Notification.id)
560
+ .execution_options(synchronize_session=False)
561
+ )
562
+ return renewed.scalar_one_or_none() is not None
563
+
564
+
565
+ def _message(notification: Notification) -> AlertMessage:
566
+ """Render a stored notification as the message the notifier surface takes."""
567
+ context = notification.context or {}
568
+ run = cast("JsonMap", context.get("run", {}))
569
+ return AlertMessage(
570
+ event=notification.event.value,
571
+ subject=notification.subject,
572
+ body=notification.body,
573
+ run_id=notification.run_id,
574
+ pipeline=cast("str | None", run.get("pipeline")),
575
+ url=cast("str | None", run.get("url")),
576
+ context=cast("dict[str, JsonValue]", context),
577
+ )
578
+
579
+
580
+ async def _record_failure(
581
+ session: AsyncSession,
582
+ services: EngineServices,
583
+ notification: Notification,
584
+ error: Exception,
585
+ moment: datetime,
586
+ ) -> bool:
587
+ """Schedule another delivery, or record that this one is never going to arrive."""
588
+ message = f"{type(error).__name__}: {error}"
589
+ notification.error = message
590
+ notification.lease_owner = None
591
+ notification.lease_expires_at = None
592
+ budget = services.settings.notification_max_attempts
593
+ if notification.attempt >= budget:
594
+ notification.status = NotificationStatus.FAILED
595
+ _timeline(
596
+ session,
597
+ notification,
598
+ LogLevel.ERROR,
599
+ f"alert could not be delivered through {notification.notifier!r} after {budget} attempts: {message}",
600
+ moment,
601
+ )
602
+ _logger.error(
603
+ "notification failed terminally",
604
+ notification_id=str(notification.id),
605
+ notifier=notification.notifier,
606
+ attempts=notification.attempt,
607
+ error=message,
608
+ )
609
+ return False
610
+ delay = services.settings.notification_backoff.total_seconds() * (2 ** (notification.attempt - 1))
611
+ notification.status = NotificationStatus.PENDING
612
+ notification.available_at = moment + timedelta(seconds=delay)
613
+ _logger.warning(
614
+ "notification delivery failed, will retry",
615
+ notification_id=str(notification.id),
616
+ attempt=notification.attempt,
617
+ delay_seconds=round(delay, 1),
618
+ error=message,
619
+ )
620
+ return False
621
+
622
+
623
+ def _timeline(
624
+ session: AsyncSession,
625
+ notification: Notification,
626
+ level: LogLevel,
627
+ message: str,
628
+ moment: datetime,
629
+ ) -> None:
630
+ """Put a delivery outcome into the run's timeline."""
631
+ if notification.run_id is None:
632
+ return
633
+ session.add(
634
+ LogEntry(
635
+ run_id=notification.run_id,
636
+ level=level,
637
+ message=message,
638
+ fields={"notifier": notification.notifier, "event": notification.event.value},
639
+ created_at=moment,
640
+ )
641
+ )
642
+
643
+
644
+ async def list_notifications(
645
+ session: AsyncSession,
646
+ *,
647
+ run_id: UUID | None = None,
648
+ notification_status: NotificationStatus | None = None,
649
+ notifier: str | None = None,
650
+ after: UUID | None = None,
651
+ limit: int | None = None,
652
+ ) -> list[Notification]:
653
+ """List notifications newest first, for one run or across the instance.
654
+
655
+ The filters are the server's own, so a row they leave out is one the caller never reads
656
+ and a cursor walk does not have to be told which pages to skip.
657
+ """
658
+ statement = sa.select(Notification).order_by(Notification.id.desc())
659
+ if run_id is not None:
660
+ statement = statement.where(Notification.run_id == run_id)
661
+ if notification_status is not None:
662
+ statement = statement.where(Notification.status == notification_status)
663
+ if notifier is not None:
664
+ statement = statement.where(Notification.notifier == notifier)
665
+ if after is not None:
666
+ statement = statement.where(Notification.id < after)
667
+ if limit is not None:
668
+ statement = statement.limit(limit)
669
+ rows = await session.execute(statement)
670
+ return list(rows.scalars())
671
+
672
+
673
+ async def find_notification(session: AsyncSession, notification_id: UUID) -> Notification | None:
674
+ """Find one notification by id."""
675
+ return await session.get(Notification, notification_id)
676
+
677
+
678
+ async def retry_notification(
679
+ session: AsyncSession,
680
+ notification: Notification,
681
+ *,
682
+ now: datetime | None = None,
683
+ ) -> Notification:
684
+ """Put one notification back on the queue, due now.
685
+
686
+ An explicit retry starts the delivery over rather than adding one try to a budget that is
687
+ already spent: a row that failed terminally has no attempts left, and a retry that left the
688
+ counter where it was would fail again without calling the notifier at all. So the backoff,
689
+ the counter and the last refusal all go, and a worker claims the row on its next pass.
690
+
691
+ A row a worker is holding is left alone: its lease is live, and returning it to the queue
692
+ would hand the same message to a second worker.
693
+ """
694
+ if notification.status is NotificationStatus.SENDING:
695
+ raise AlertError("a worker is delivering this one; wait for it to finish or fail")
696
+ notification.status = NotificationStatus.PENDING
697
+ notification.available_at = now or utcnow()
698
+ notification.attempt = 0
699
+ notification.error = None
700
+ notification.sent_at = None
701
+ notification.lease_owner = None
702
+ notification.lease_expires_at = None
703
+ await session.flush()
704
+ _logger.info("notification queued again", notification_id=str(notification.id), notifier=notification.notifier)
705
+ return notification
706
+
707
+
708
+ async def recover_notifications(session: AsyncSession, *, now: datetime | None = None) -> int:
709
+ """Return notifications whose worker died back to the queue, like the lease sweeper does.
710
+
711
+ One conditional statement rather than a select and a write: a renewal or a delivery that
712
+ commits between the two would otherwise be overwritten, requeueing a notification another
713
+ worker is still sending or has already sent.
714
+ """
715
+ moment = now or utcnow()
716
+ recovered = await session.execute(
717
+ sa.update(Notification)
718
+ .where(
719
+ Notification.status == NotificationStatus.SENDING,
720
+ Notification.lease_expires_at.is_not(None),
721
+ Notification.lease_expires_at < moment,
722
+ )
723
+ .values(
724
+ status=NotificationStatus.PENDING,
725
+ available_at=moment,
726
+ lease_owner=None,
727
+ lease_expires_at=None,
728
+ )
729
+ .returning(Notification.id)
730
+ .execution_options(synchronize_session=False)
731
+ )
732
+ return len(recovered.scalars().all())
733
+
734
+
735
+ class NotificationDispatcher(BaseModel):
736
+ """The worker's alert-delivery loop: claim one, send it, repeat until the queue is empty."""
737
+
738
+ model_config = ConfigDict(arbitrary_types_allowed=True, frozen=True)
739
+
740
+ sessions: async_sessionmaker[AsyncSession]
741
+ services: EngineServices
742
+ owner: str
743
+ max_per_pass: int = Field(default=25, ge=1)
744
+
745
+ async def drain(self, *, now: datetime | None = None) -> int:
746
+ """Deliver every notification that is due, and report how many were sent.
747
+
748
+ Each delivery is its own transaction, so one undeliverable message never rolls back
749
+ the ones that went out beside it.
750
+ """
751
+ sent = 0
752
+ for _ in range(self.max_per_pass):
753
+ async with session_scope(self.sessions) as session:
754
+ moment = now or utcnow()
755
+ notification = await claim_notification(
756
+ session,
757
+ owner=self.owner,
758
+ now=moment,
759
+ lease_seconds=int(self.services.settings.notification_lease.total_seconds()),
760
+ )
761
+ if notification is None:
762
+ return sent
763
+ async with session_scope(self.sessions) as session:
764
+ claimed = await session.get(Notification, notification.id)
765
+ if claimed is None: # pragma: no cover - it was just claimed
766
+ continue
767
+ async with self._renewing(claimed.id):
768
+ delivered = await send_notification(session, self.services, claimed, owner=self.owner, now=now)
769
+ if delivered:
770
+ sent += 1
771
+ return sent
772
+
773
+ @asynccontextmanager
774
+ async def _renewing(self, notification_id: UUID) -> AsyncGenerator[None]:
775
+ """Hold a notification's lease open for as long as its delivery is in flight."""
776
+ halting = asyncio.Event()
777
+ renewing = asyncio.create_task(self._renew_until_halted(notification_id, halting))
778
+ try:
779
+ yield
780
+ finally:
781
+ halting.set()
782
+ await renewing
783
+
784
+ async def _renew_until_halted(self, notification_id: UUID, halting: asyncio.Event) -> None:
785
+ """Extend the lease on a cadence until halted, or until the row is no longer this worker's.
786
+
787
+ The halt is a flag rather than a cancel: a renewal already inside its transaction runs
788
+ to its own end, where a cancel landing in a database await strands the session's
789
+ connection, checked out of the pool and never closed.
790
+ """
791
+ lease_seconds = int(self.services.settings.notification_lease.total_seconds())
792
+ while not halting.is_set():
793
+ with suppress(TimeoutError):
794
+ await asyncio.wait_for(halting.wait(), timeout=lease_seconds / RENEWALS_PER_LEASE)
795
+ if halting.is_set():
796
+ return
797
+ try:
798
+ async with session_scope(self.sessions) as session:
799
+ if not await renew_lease(session, notification_id, owner=self.owner, lease_seconds=lease_seconds):
800
+ return
801
+ except Exception as error: # the delivery runs on; the sweeper reclaims if it outlasts the lease
802
+ _logger.warning(
803
+ "notification lease renewal failed", notification_id=str(notification_id), error=str(error)
804
+ )
805
+ return