dirigent-core 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dirigent_core/__init__.py +26 -0
- dirigent_core/alembic/env.py +71 -0
- dirigent_core/alembic/script.py.mako +26 -0
- dirigent_core/alembic/versions/0001_baseline_schema.py +1016 -0
- dirigent_core/alerting.py +805 -0
- dirigent_core/artifacts.py +73 -0
- dirigent_core/auth.py +529 -0
- dirigent_core/blockdocs.py +223 -0
- dirigent_core/config.py +421 -0
- dirigent_core/configdocs.py +134 -0
- dirigent_core/database.py +177 -0
- dirigent_core/directory.py +339 -0
- dirigent_core/documents.py +878 -0
- dirigent_core/documentschema.py +92 -0
- dirigent_core/engine/__init__.py +120 -0
- dirigent_core/engine/claim.py +156 -0
- dirigent_core/engine/context.py +420 -0
- dirigent_core/engine/definition.py +543 -0
- dirigent_core/engine/executor.py +1029 -0
- dirigent_core/engine/failure.py +71 -0
- dirigent_core/engine/recovery.py +137 -0
- dirigent_core/engine/references.py +239 -0
- dirigent_core/engine/runs.py +865 -0
- dirigent_core/engine/services.py +74 -0
- dirigent_core/engine/state.py +434 -0
- dirigent_core/ids.py +39 -0
- dirigent_core/logging.py +310 -0
- dirigent_core/migrations.py +98 -0
- dirigent_core/models.py +626 -0
- dirigent_core/pipelines.py +494 -0
- dirigent_core/plugins.py +255 -0
- dirigent_core/protocol.py +276 -0
- dirigent_core/py.typed +0 -0
- dirigent_core/ratelimit.py +42 -0
- dirigent_core/registry.py +32 -0
- dirigent_core/retention.py +293 -0
- dirigent_core/scheduler.py +491 -0
- dirigent_core/schemas.py +147 -0
- dirigent_core/secrets.py +164 -0
- dirigent_core/storage.py +349 -0
- dirigent_core/telemetry.py +402 -0
- dirigent_core/trigger_documents.py +193 -0
- dirigent_core/triggers/__init__.py +117 -0
- dirigent_core/triggers/backfill.py +131 -0
- dirigent_core/triggers/materialize.py +218 -0
- dirigent_core/triggers/schedules.py +531 -0
- dirigent_core/triggers/webhooks.py +586 -0
- dirigent_core/types.py +59 -0
- dirigent_core/worker.py +350 -0
- dirigent_core-0.9.0.dist-info/METADATA +29 -0
- dirigent_core-0.9.0.dist-info/RECORD +53 -0
- dirigent_core-0.9.0.dist-info/WHEEL +4 -0
- dirigent_core-0.9.0.dist-info/licenses/LICENSE +18 -0
|
@@ -0,0 +1,805 @@
|
|
|
1
|
+
"""Alerting rules and channels as data, with delivery queued and retried like any other work.
|
|
2
|
+
|
|
3
|
+
A rule says one thing about one run once: the unique constraint on
|
|
4
|
+
``(alert_rule_id, run_id, event)`` is the deduplication, and the throttle window is a second,
|
|
5
|
+
coarser guard against a flapping pipeline.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import asyncio
|
|
9
|
+
from collections.abc import AsyncGenerator, Sequence
|
|
10
|
+
from contextlib import asynccontextmanager, suppress
|
|
11
|
+
from datetime import datetime, timedelta
|
|
12
|
+
from typing import Final, cast
|
|
13
|
+
from uuid import UUID
|
|
14
|
+
|
|
15
|
+
import sqlalchemy as sa
|
|
16
|
+
from pydantic import BaseModel, ConfigDict, Field, JsonValue
|
|
17
|
+
from sqlalchemy.exc import IntegrityError
|
|
18
|
+
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker
|
|
19
|
+
|
|
20
|
+
from dirigent_client.enums import AlertEvent, AlertScope, LogLevel, NotificationStatus, RunStatus
|
|
21
|
+
from dirigent_common import EntityName, JsonMap, format_duration
|
|
22
|
+
from dirigent_core.database import session_scope
|
|
23
|
+
from dirigent_core.engine.references import substitute
|
|
24
|
+
from dirigent_core.engine.services import EngineServices
|
|
25
|
+
from dirigent_core.logging import get_logger
|
|
26
|
+
from dirigent_core.models import AlertRule, Connection, LogEntry, Notification, Pipeline, Run, utcnow
|
|
27
|
+
from dirigent_plugin import AlertMessage, Notifier
|
|
28
|
+
|
|
29
|
+
EVENT_FOR_STATUS: dict[RunStatus, AlertEvent] = {
|
|
30
|
+
RunStatus.FAILED: AlertEvent.RUN_FAILED,
|
|
31
|
+
RunStatus.COMPLETED_WITH_ERRORS: AlertEvent.RUN_COMPLETED_WITH_ERRORS,
|
|
32
|
+
RunStatus.SUCCEEDED: AlertEvent.RUN_SUCCEEDED,
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
DEFAULT_TEMPLATE = "${run.pipeline} run ${run.status}"
|
|
36
|
+
|
|
37
|
+
CLAIM_LIMIT = 1
|
|
38
|
+
|
|
39
|
+
#: How many times a lease is renewed over its own length while a delivery is in flight.
|
|
40
|
+
RENEWALS_PER_LEASE = 3
|
|
41
|
+
|
|
42
|
+
_logger = get_logger("alerting")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class AlertError(Exception):
|
|
46
|
+
"""An alert rule could not be declared, found, or delivered."""
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class AlertRuleRequest(BaseModel):
|
|
50
|
+
"""What it takes to declare an alert rule, from the API or the CLI."""
|
|
51
|
+
|
|
52
|
+
model_config = ConfigDict(frozen=True)
|
|
53
|
+
|
|
54
|
+
code: EntityName
|
|
55
|
+
name: str | None = None
|
|
56
|
+
description: str | None = None
|
|
57
|
+
event: AlertEvent
|
|
58
|
+
notifier: str
|
|
59
|
+
scope: AlertScope = AlertScope.GLOBAL
|
|
60
|
+
pipeline: str | None = None
|
|
61
|
+
connection: str | None = None
|
|
62
|
+
template: str | None = None
|
|
63
|
+
throttle: timedelta = timedelta(0)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
#: Stands in for "this path names nothing", which is not the same as naming a null.
|
|
67
|
+
MISSING: Final = object()
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def render_template(template: str, context: JsonMap) -> str:
|
|
71
|
+
"""Render an alert's message, resolving ``${run.*}`` against the run's own facts.
|
|
72
|
+
|
|
73
|
+
An unresolvable reference is left verbatim rather than raised or blanked, so a typo in a
|
|
74
|
+
template never costs the message. A reference resolving to null renders as empty.
|
|
75
|
+
"""
|
|
76
|
+
|
|
77
|
+
def one(reference: str) -> str:
|
|
78
|
+
resolved = _walk(context, reference.strip())
|
|
79
|
+
return f"${{{reference}}}" if resolved is MISSING else _as_text(cast("JsonValue", resolved))
|
|
80
|
+
|
|
81
|
+
return substitute(template, one)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _walk(context: JsonMap, reference: str) -> object:
|
|
85
|
+
"""Walk a dotted path into the alert context, or return MISSING when it names nothing."""
|
|
86
|
+
parts = [piece for piece in reference.split(".") if piece]
|
|
87
|
+
if not parts:
|
|
88
|
+
return MISSING
|
|
89
|
+
current: JsonValue = context
|
|
90
|
+
for part in parts:
|
|
91
|
+
if not isinstance(current, dict) or part not in current:
|
|
92
|
+
return MISSING
|
|
93
|
+
current = current[part]
|
|
94
|
+
return current
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _as_text(value: JsonValue) -> str:
|
|
98
|
+
"""Render a resolved value the way a message wants to read it."""
|
|
99
|
+
if isinstance(value, bool):
|
|
100
|
+
return "true" if value else "false"
|
|
101
|
+
return "" if value is None else str(value)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def build_context(run: Run, pipeline: Pipeline, *, base_url: str | None = None) -> JsonMap:
|
|
105
|
+
"""Gather the facts a template may read, under the one namespace it has.
|
|
106
|
+
|
|
107
|
+
A flat snapshot rather than a live handle: the notification is delivered later, and must
|
|
108
|
+
describe the run as it was when the alert fired.
|
|
109
|
+
"""
|
|
110
|
+
duration = (run.finished_at - run.started_at).total_seconds() if run.finished_at and run.started_at else None
|
|
111
|
+
return {
|
|
112
|
+
"run": {
|
|
113
|
+
"id": str(run.id),
|
|
114
|
+
"status": run.status.value,
|
|
115
|
+
"pipeline": pipeline.code,
|
|
116
|
+
"error": run.error,
|
|
117
|
+
"params": dict(run.params),
|
|
118
|
+
"trigger": run.triggered_by_label or run.triggered_by_kind.value,
|
|
119
|
+
"started_at": run.started_at.isoformat() if run.started_at else None,
|
|
120
|
+
"finished_at": run.finished_at.isoformat() if run.finished_at else None,
|
|
121
|
+
"duration_ms": round(duration * 1000) if duration is not None else None,
|
|
122
|
+
"url": f"{base_url.rstrip('/')}/runs/{run.id}" if base_url else None,
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
async def find_rule(session: AsyncSession, code: str) -> AlertRule | None:
|
|
128
|
+
"""Find one alert rule by code."""
|
|
129
|
+
found = await session.execute(sa.select(AlertRule).where(AlertRule.code == code))
|
|
130
|
+
return found.scalar_one_or_none()
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
async def list_rules(
|
|
134
|
+
session: AsyncSession,
|
|
135
|
+
*,
|
|
136
|
+
after: UUID | None = None,
|
|
137
|
+
limit: int | None = None,
|
|
138
|
+
) -> list[AlertRule]:
|
|
139
|
+
"""List every alert rule in id order, which is the order they were declared in."""
|
|
140
|
+
statement = sa.select(AlertRule).order_by(AlertRule.id)
|
|
141
|
+
if after is not None:
|
|
142
|
+
statement = statement.where(AlertRule.id > after)
|
|
143
|
+
if limit is not None:
|
|
144
|
+
statement = statement.limit(limit)
|
|
145
|
+
rows = await session.execute(statement)
|
|
146
|
+
return list(rows.scalars())
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
async def create_rule(session: AsyncSession, services: EngineServices, request: AlertRuleRequest) -> AlertRule:
|
|
150
|
+
"""Declare an alert rule, refusing a notifier or a scope this instance cannot honour."""
|
|
151
|
+
if request.notifier not in services.host.notifiers:
|
|
152
|
+
installed = ", ".join(sorted(services.host.notifiers)) or "none are installed"
|
|
153
|
+
raise AlertError(f"no notifier {request.notifier!r} is installed ({installed})")
|
|
154
|
+
if await find_rule(session, request.code) is not None:
|
|
155
|
+
raise AlertError(f"an alert rule coded {request.code!r} already exists")
|
|
156
|
+
pipeline_id = await _scope_pipeline(session, request)
|
|
157
|
+
connection_id = await _connection_id(session, request.connection) if request.connection else None
|
|
158
|
+
rule = AlertRule(
|
|
159
|
+
code=request.code,
|
|
160
|
+
name=request.name,
|
|
161
|
+
description=request.description,
|
|
162
|
+
event=request.event,
|
|
163
|
+
scope=request.scope,
|
|
164
|
+
pipeline_id=pipeline_id,
|
|
165
|
+
notifier=request.notifier,
|
|
166
|
+
connection_id=connection_id,
|
|
167
|
+
template=request.template,
|
|
168
|
+
throttle_seconds=int(request.throttle.total_seconds()),
|
|
169
|
+
)
|
|
170
|
+
session.add(rule)
|
|
171
|
+
await session.flush()
|
|
172
|
+
_logger.info("alert rule created", rule=rule.code, alert_event=rule.event.value, notifier=rule.notifier)
|
|
173
|
+
return rule
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
async def _scope_pipeline(session: AsyncSession, request: AlertRuleRequest) -> UUID | None:
|
|
177
|
+
"""Resolve the pipeline a rule is scoped to, refusing a scope that names nothing."""
|
|
178
|
+
if request.scope is AlertScope.GLOBAL:
|
|
179
|
+
return None
|
|
180
|
+
if not request.pipeline:
|
|
181
|
+
raise AlertError("a pipeline-scoped rule has to name the pipeline it watches")
|
|
182
|
+
found = await session.execute(sa.select(Pipeline).where(Pipeline.code == request.pipeline))
|
|
183
|
+
pipeline = found.scalar_one_or_none()
|
|
184
|
+
if pipeline is None:
|
|
185
|
+
raise AlertError(f"no pipeline coded {request.pipeline!r}")
|
|
186
|
+
return pipeline.id
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
async def _connection_id(session: AsyncSession, code: str) -> UUID:
|
|
190
|
+
"""Resolve the connection a notifier delivers through, refusing an unknown code."""
|
|
191
|
+
found = await session.execute(sa.select(Connection).where(Connection.code == code))
|
|
192
|
+
connection = found.scalar_one_or_none()
|
|
193
|
+
if connection is None:
|
|
194
|
+
raise AlertError(f"no connection coded {code!r}")
|
|
195
|
+
return connection.id
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
async def delete_rule(session: AsyncSession, rule: AlertRule) -> None:
|
|
199
|
+
"""Remove an alert rule; the notifications it already raised are kept."""
|
|
200
|
+
code = rule.code
|
|
201
|
+
await session.delete(rule)
|
|
202
|
+
await session.flush()
|
|
203
|
+
_logger.info("alert rule deleted", rule=code)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
async def matching_rules(session: AsyncSession, event: AlertEvent, pipeline_id: UUID) -> list[AlertRule]:
|
|
207
|
+
"""Find the live rules that want to hear about one event, throttled or not.
|
|
208
|
+
|
|
209
|
+
A paused rule matches nothing. Pausing is instance state an operator sets on the row, so
|
|
210
|
+
it is read here rather than folded into ``active``, which is what the rule itself declares.
|
|
211
|
+
"""
|
|
212
|
+
rows = await session.execute(
|
|
213
|
+
sa.select(AlertRule).where(
|
|
214
|
+
AlertRule.event == event,
|
|
215
|
+
AlertRule.active.is_(True),
|
|
216
|
+
AlertRule.paused.is_(False),
|
|
217
|
+
sa.or_(AlertRule.scope == AlertScope.GLOBAL, AlertRule.pipeline_id == pipeline_id),
|
|
218
|
+
)
|
|
219
|
+
)
|
|
220
|
+
return list(rows.scalars())
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
async def set_paused(session: AsyncSession, rule: AlertRule, *, paused: bool) -> AlertRule:
|
|
224
|
+
"""Hold a rule's deliveries, or let them resume; the rule itself is left as declared."""
|
|
225
|
+
rule.paused = paused
|
|
226
|
+
await session.flush()
|
|
227
|
+
_logger.info("alert rule paused" if paused else "alert rule resumed", rule=rule.code)
|
|
228
|
+
return rule
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
async def raised_at(session: AsyncSession, rule: AlertRule, pipeline_id: UUID) -> datetime | None:
|
|
232
|
+
"""When this rule last raised anything **for this pipeline**.
|
|
233
|
+
|
|
234
|
+
A global rule watches every pipeline, and a window measured across all of them lets one
|
|
235
|
+
noisy pipeline silence the rest. The window is per pipeline, which is what an operator
|
|
236
|
+
means by "not more than one of these an hour".
|
|
237
|
+
|
|
238
|
+
Read from ``created_at``, which nothing moves: ``available_at`` is rewritten by every
|
|
239
|
+
delivery retry and by every recovery, so a channel that is down would widen the window
|
|
240
|
+
it was raised in.
|
|
241
|
+
"""
|
|
242
|
+
return (
|
|
243
|
+
await session.execute(
|
|
244
|
+
sa.select(sa.func.max(Notification.created_at))
|
|
245
|
+
.join(Run, Run.id == Notification.run_id)
|
|
246
|
+
.where(Notification.alert_rule_id == rule.id, Run.pipeline_id == pipeline_id)
|
|
247
|
+
)
|
|
248
|
+
).scalar()
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
async def throttled_until(session: AsyncSession, rule: AlertRule, pipeline_id: UUID, now: datetime) -> datetime | None:
|
|
252
|
+
"""Say when this rule may next raise for this pipeline, or ``None`` if it may now.
|
|
253
|
+
|
|
254
|
+
Measured from when a message was raised, not from when it was delivered or retried, so
|
|
255
|
+
neither a slow notifier nor a failing one moves the window.
|
|
256
|
+
"""
|
|
257
|
+
if not rule.throttle_seconds:
|
|
258
|
+
return None
|
|
259
|
+
last = await raised_at(session, rule, pipeline_id)
|
|
260
|
+
if last is None:
|
|
261
|
+
return None
|
|
262
|
+
opens = last + timedelta(seconds=rule.throttle_seconds)
|
|
263
|
+
return opens if opens > now else None
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
async def raise_for_run(
|
|
267
|
+
session: AsyncSession,
|
|
268
|
+
services: EngineServices,
|
|
269
|
+
run: Run,
|
|
270
|
+
event: AlertEvent,
|
|
271
|
+
*,
|
|
272
|
+
now: datetime | None = None,
|
|
273
|
+
) -> list[Notification]:
|
|
274
|
+
"""Queue whatever this run's settling owes, in the caller's outcome transaction.
|
|
275
|
+
|
|
276
|
+
Called from the same commit that settles the run, so a run cannot reach a terminal state
|
|
277
|
+
without its alerts having been queued.
|
|
278
|
+
"""
|
|
279
|
+
moment = now or utcnow()
|
|
280
|
+
pipeline = await session.get(Pipeline, run.pipeline_id)
|
|
281
|
+
if pipeline is None: # pragma: no cover - the foreign key makes this unreachable
|
|
282
|
+
return []
|
|
283
|
+
rules = await matching_rules(session, event, run.pipeline_id)
|
|
284
|
+
if not rules:
|
|
285
|
+
return []
|
|
286
|
+
context = build_context(run, pipeline, base_url=services.settings.alert_base_url)
|
|
287
|
+
queued: list[Notification] = []
|
|
288
|
+
for rule in rules:
|
|
289
|
+
if await _already_raised(session, rule.id, run.id, event):
|
|
290
|
+
continue
|
|
291
|
+
opens = await throttled_until(session, rule, run.pipeline_id, moment)
|
|
292
|
+
if opens is not None:
|
|
293
|
+
# A suppressed alert that leaves nothing behind is indistinguishable from a rule
|
|
294
|
+
# that never matched, which is the hard way to learn a throttle is too wide.
|
|
295
|
+
session.add(
|
|
296
|
+
LogEntry(
|
|
297
|
+
run_id=run.id,
|
|
298
|
+
level=LogLevel.INFO,
|
|
299
|
+
message=f"alert {rule.code!r} suppressed by its throttle window",
|
|
300
|
+
fields={
|
|
301
|
+
"event": event.value,
|
|
302
|
+
"notifier": rule.notifier,
|
|
303
|
+
"throttle": format_duration(timedelta(seconds=rule.throttle_seconds)),
|
|
304
|
+
"opens_at": opens.isoformat(),
|
|
305
|
+
},
|
|
306
|
+
created_at=moment,
|
|
307
|
+
)
|
|
308
|
+
)
|
|
309
|
+
continue
|
|
310
|
+
subject = render_template(rule.template or DEFAULT_TEMPLATE, context)
|
|
311
|
+
notification = Notification(
|
|
312
|
+
alert_rule_id=rule.id,
|
|
313
|
+
run_id=run.id,
|
|
314
|
+
event=event,
|
|
315
|
+
notifier=rule.notifier,
|
|
316
|
+
connection_id=rule.connection_id,
|
|
317
|
+
subject=subject,
|
|
318
|
+
body=_body(context),
|
|
319
|
+
context=context,
|
|
320
|
+
status=NotificationStatus.PENDING,
|
|
321
|
+
available_at=moment,
|
|
322
|
+
created_at=moment,
|
|
323
|
+
)
|
|
324
|
+
# A savepoint, because two processes can reach this between the check above and the
|
|
325
|
+
# insert, and an IntegrityError raised here would otherwise abort the caller's whole
|
|
326
|
+
# transaction rather than just this insert.
|
|
327
|
+
try:
|
|
328
|
+
async with session.begin_nested():
|
|
329
|
+
session.add(notification)
|
|
330
|
+
session.add(
|
|
331
|
+
LogEntry(
|
|
332
|
+
run_id=run.id,
|
|
333
|
+
level=LogLevel.INFO,
|
|
334
|
+
message=f"alert {rule.code!r} queued for delivery through {rule.notifier!r}",
|
|
335
|
+
fields={"event": event.value, "notifier": rule.notifier},
|
|
336
|
+
created_at=moment,
|
|
337
|
+
)
|
|
338
|
+
)
|
|
339
|
+
except IntegrityError:
|
|
340
|
+
continue
|
|
341
|
+
rule.last_sent_at = moment
|
|
342
|
+
queued.append(notification)
|
|
343
|
+
await session.flush()
|
|
344
|
+
if queued:
|
|
345
|
+
_logger.info("alerts raised", run_id=str(run.id), alert_event=event.value, rules=len(queued))
|
|
346
|
+
return queued
|
|
347
|
+
|
|
348
|
+
|
|
349
|
+
def _body(context: JsonMap) -> str:
|
|
350
|
+
"""Render the default body: the run's own facts, one per line, in a stable order."""
|
|
351
|
+
run = cast("JsonMap", context.get("run", {}))
|
|
352
|
+
lines = [f"{name}: {_as_text(value)}" for name, value in run.items() if value not in (None, "", {}, [])]
|
|
353
|
+
return "\n".join(lines)
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
async def _already_raised(session: AsyncSession, rule_id: UUID, run_id: UUID, event: AlertEvent) -> bool:
|
|
357
|
+
"""Report whether this rule has already said this about this run."""
|
|
358
|
+
found = await session.execute(
|
|
359
|
+
sa.select(sa.func.count())
|
|
360
|
+
.select_from(Notification)
|
|
361
|
+
.where(
|
|
362
|
+
Notification.alert_rule_id == rule_id,
|
|
363
|
+
Notification.run_id == run_id,
|
|
364
|
+
Notification.event == event,
|
|
365
|
+
)
|
|
366
|
+
)
|
|
367
|
+
return int(found.scalar_one()) > 0
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
async def raise_for_status(
|
|
371
|
+
session: AsyncSession,
|
|
372
|
+
services: EngineServices,
|
|
373
|
+
run: Run,
|
|
374
|
+
status: RunStatus,
|
|
375
|
+
*,
|
|
376
|
+
now: datetime | None = None,
|
|
377
|
+
) -> list[Notification]:
|
|
378
|
+
"""Queue the alerts a terminal run status owes, if that status raises an event at all."""
|
|
379
|
+
event = EVENT_FOR_STATUS.get(status)
|
|
380
|
+
if event is None:
|
|
381
|
+
return []
|
|
382
|
+
return await raise_for_run(session, services, run, event, now=now)
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
async def raise_for_stuck(
|
|
386
|
+
session: AsyncSession,
|
|
387
|
+
services: EngineServices,
|
|
388
|
+
run_ids: Sequence[UUID],
|
|
389
|
+
*,
|
|
390
|
+
now: datetime | None = None,
|
|
391
|
+
) -> int:
|
|
392
|
+
"""Queue ``run_stuck`` alerts for the runs the sweeper found standing still.
|
|
393
|
+
|
|
394
|
+
The sweeper re-detects the same run every sweep; the once-per-rule-per-run constraint is
|
|
395
|
+
what keeps that to one message.
|
|
396
|
+
"""
|
|
397
|
+
moment = now or utcnow()
|
|
398
|
+
queued = 0
|
|
399
|
+
for run_id in run_ids:
|
|
400
|
+
run = await session.get(Run, run_id)
|
|
401
|
+
if run is None: # pragma: no cover - the sweeper just read these ids
|
|
402
|
+
continue
|
|
403
|
+
queued += len(await raise_for_run(session, services, run, AlertEvent.RUN_STUCK, now=moment))
|
|
404
|
+
return queued
|
|
405
|
+
|
|
406
|
+
|
|
407
|
+
async def queue_test_message(
|
|
408
|
+
session: AsyncSession,
|
|
409
|
+
services: EngineServices,
|
|
410
|
+
*,
|
|
411
|
+
notifier: str,
|
|
412
|
+
connection: str | None = None,
|
|
413
|
+
subject: str = "dirigent test alert",
|
|
414
|
+
body: str = "This is a test message sent through the notifier surface.",
|
|
415
|
+
) -> Notification:
|
|
416
|
+
"""Queue one unattached message, which is what ``dg alerts test`` sends."""
|
|
417
|
+
if notifier not in services.host.notifiers:
|
|
418
|
+
installed = ", ".join(sorted(services.host.notifiers)) or "none are installed"
|
|
419
|
+
raise AlertError(f"no notifier {notifier!r} is installed ({installed})")
|
|
420
|
+
notification = Notification(
|
|
421
|
+
alert_rule_id=None,
|
|
422
|
+
run_id=None,
|
|
423
|
+
event=AlertEvent.RUN_SUCCEEDED,
|
|
424
|
+
notifier=notifier,
|
|
425
|
+
connection_id=await _connection_id(session, connection) if connection else None,
|
|
426
|
+
subject=subject,
|
|
427
|
+
body=body,
|
|
428
|
+
context={"run": {"pipeline": "(test)", "status": "succeeded"}},
|
|
429
|
+
status=NotificationStatus.PENDING,
|
|
430
|
+
available_at=utcnow(),
|
|
431
|
+
)
|
|
432
|
+
session.add(notification)
|
|
433
|
+
await session.flush()
|
|
434
|
+
return notification
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
async def claim_notification(
|
|
438
|
+
session: AsyncSession,
|
|
439
|
+
*,
|
|
440
|
+
owner: str,
|
|
441
|
+
now: datetime,
|
|
442
|
+
lease_seconds: int,
|
|
443
|
+
) -> Notification | None:
|
|
444
|
+
"""Claim the next due notification, the same way a worker claims an attempt."""
|
|
445
|
+
statement = (
|
|
446
|
+
sa.select(Notification)
|
|
447
|
+
.where(Notification.status == NotificationStatus.PENDING, Notification.available_at <= now)
|
|
448
|
+
.order_by(Notification.available_at, Notification.id)
|
|
449
|
+
.limit(CLAIM_LIMIT)
|
|
450
|
+
)
|
|
451
|
+
if session.get_bind().dialect.name == "postgresql":
|
|
452
|
+
statement = statement.with_for_update(skip_locked=True, of=Notification)
|
|
453
|
+
found = await session.execute(statement)
|
|
454
|
+
notification = found.scalars().first()
|
|
455
|
+
if notification is None:
|
|
456
|
+
return None
|
|
457
|
+
notification.status = NotificationStatus.SENDING
|
|
458
|
+
notification.lease_owner = owner
|
|
459
|
+
notification.lease_expires_at = now + timedelta(seconds=lease_seconds)
|
|
460
|
+
notification.attempt += 1
|
|
461
|
+
await session.flush()
|
|
462
|
+
return notification
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
def notifier_config(services: EngineServices, notifier: Notifier, connection: Connection | None) -> BaseModel:
|
|
466
|
+
"""Open the connection a notifier delivers through, validated against its own model."""
|
|
467
|
+
if connection is None:
|
|
468
|
+
return notifier.config_model()
|
|
469
|
+
return services.secrets.decrypt_config(
|
|
470
|
+
notifier.config_model, connection.config, connection.secret_envelope, key_id=connection.secret_key_id
|
|
471
|
+
)
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
async def send_notification(
|
|
475
|
+
session: AsyncSession,
|
|
476
|
+
services: EngineServices,
|
|
477
|
+
notification: Notification,
|
|
478
|
+
*,
|
|
479
|
+
owner: str,
|
|
480
|
+
now: datetime | None = None,
|
|
481
|
+
) -> bool:
|
|
482
|
+
"""Deliver one claimed notification, returning whether it was delivered.
|
|
483
|
+
|
|
484
|
+
The outcome is written only while this worker still holds the lease, so a delivery that
|
|
485
|
+
outlasted its lease cannot overwrite the outcome of the worker the row was handed to.
|
|
486
|
+
|
|
487
|
+
THE SESSION IS COMMITTED BEFORE THE SEND, so no transaction is open while an outbound
|
|
488
|
+
call is in flight: on SQLite that transaction holds the one write lock, and everything
|
|
489
|
+
else on the instance -- the sweeper reclaiming this very row, the API, a worker settling
|
|
490
|
+
an attempt -- would queue behind a notifier that is slow to answer. Nothing read above
|
|
491
|
+
has to be held anyway, because the fence below re-reads the row it writes.
|
|
492
|
+
"""
|
|
493
|
+
moment = now or utcnow()
|
|
494
|
+
try:
|
|
495
|
+
notifier = services.host.notifiers.get(notification.notifier)
|
|
496
|
+
if notifier is None:
|
|
497
|
+
raise AlertError(f"notifier {notification.notifier!r} is not installed on this worker")
|
|
498
|
+
connection = await session.get(Connection, notification.connection_id) if notification.connection_id else None
|
|
499
|
+
config = notifier_config(services, notifier, connection)
|
|
500
|
+
await session.commit()
|
|
501
|
+
await notifier.send(_message(notification), config)
|
|
502
|
+
except Exception as error:
|
|
503
|
+
if not await _holds_lease(session, notification, owner):
|
|
504
|
+
return False
|
|
505
|
+
return await _record_failure(session, services, notification, error, moment)
|
|
506
|
+
if not await _holds_lease(session, notification, owner):
|
|
507
|
+
return False
|
|
508
|
+
notification.status = NotificationStatus.SENT
|
|
509
|
+
notification.sent_at = moment
|
|
510
|
+
notification.error = None
|
|
511
|
+
notification.lease_owner = None
|
|
512
|
+
notification.lease_expires_at = None
|
|
513
|
+
_timeline(session, notification, LogLevel.INFO, f"alert delivered through {notification.notifier!r}", moment)
|
|
514
|
+
_logger.info("notification delivered", notification_id=str(notification.id), notifier=notification.notifier)
|
|
515
|
+
return True
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
async def _holds_lease(session: AsyncSession, notification: Notification, owner: str) -> bool:
|
|
519
|
+
"""Re-read the row and report whether this worker may still write this outcome.
|
|
520
|
+
|
|
521
|
+
The lease is a fence, not a hint: a row the sweeper returned to the queue belongs to
|
|
522
|
+
whoever claimed it next, and writing into it would replace a live delivery's outcome
|
|
523
|
+
with the outcome of an abandoned one.
|
|
524
|
+
"""
|
|
525
|
+
found = await session.execute(
|
|
526
|
+
sa.select(Notification.status, Notification.lease_owner).where(Notification.id == notification.id)
|
|
527
|
+
)
|
|
528
|
+
row = found.one_or_none()
|
|
529
|
+
if row is not None and row.status is NotificationStatus.SENDING and row.lease_owner == owner:
|
|
530
|
+
return True
|
|
531
|
+
_logger.warning(
|
|
532
|
+
"lease lost, notification outcome discarded",
|
|
533
|
+
notification_id=str(notification.id),
|
|
534
|
+
notifier=notification.notifier,
|
|
535
|
+
holder=None if row is None else row.lease_owner,
|
|
536
|
+
notification_status=None if row is None else row.status.value,
|
|
537
|
+
)
|
|
538
|
+
return False
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
async def renew_lease(
|
|
542
|
+
session: AsyncSession,
|
|
543
|
+
notification_id: UUID,
|
|
544
|
+
*,
|
|
545
|
+
owner: str,
|
|
546
|
+
lease_seconds: int,
|
|
547
|
+
now: datetime | None = None,
|
|
548
|
+
) -> bool:
|
|
549
|
+
"""Push a notification's lease out, reporting whether this worker still held it."""
|
|
550
|
+
moment = now or utcnow()
|
|
551
|
+
renewed = await session.execute(
|
|
552
|
+
sa.update(Notification)
|
|
553
|
+
.where(
|
|
554
|
+
Notification.id == notification_id,
|
|
555
|
+
Notification.status == NotificationStatus.SENDING,
|
|
556
|
+
Notification.lease_owner == owner,
|
|
557
|
+
)
|
|
558
|
+
.values(lease_expires_at=moment + timedelta(seconds=lease_seconds))
|
|
559
|
+
.returning(Notification.id)
|
|
560
|
+
.execution_options(synchronize_session=False)
|
|
561
|
+
)
|
|
562
|
+
return renewed.scalar_one_or_none() is not None
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
def _message(notification: Notification) -> AlertMessage:
|
|
566
|
+
"""Render a stored notification as the message the notifier surface takes."""
|
|
567
|
+
context = notification.context or {}
|
|
568
|
+
run = cast("JsonMap", context.get("run", {}))
|
|
569
|
+
return AlertMessage(
|
|
570
|
+
event=notification.event.value,
|
|
571
|
+
subject=notification.subject,
|
|
572
|
+
body=notification.body,
|
|
573
|
+
run_id=notification.run_id,
|
|
574
|
+
pipeline=cast("str | None", run.get("pipeline")),
|
|
575
|
+
url=cast("str | None", run.get("url")),
|
|
576
|
+
context=cast("dict[str, JsonValue]", context),
|
|
577
|
+
)
|
|
578
|
+
|
|
579
|
+
|
|
580
|
+
async def _record_failure(
|
|
581
|
+
session: AsyncSession,
|
|
582
|
+
services: EngineServices,
|
|
583
|
+
notification: Notification,
|
|
584
|
+
error: Exception,
|
|
585
|
+
moment: datetime,
|
|
586
|
+
) -> bool:
|
|
587
|
+
"""Schedule another delivery, or record that this one is never going to arrive."""
|
|
588
|
+
message = f"{type(error).__name__}: {error}"
|
|
589
|
+
notification.error = message
|
|
590
|
+
notification.lease_owner = None
|
|
591
|
+
notification.lease_expires_at = None
|
|
592
|
+
budget = services.settings.notification_max_attempts
|
|
593
|
+
if notification.attempt >= budget:
|
|
594
|
+
notification.status = NotificationStatus.FAILED
|
|
595
|
+
_timeline(
|
|
596
|
+
session,
|
|
597
|
+
notification,
|
|
598
|
+
LogLevel.ERROR,
|
|
599
|
+
f"alert could not be delivered through {notification.notifier!r} after {budget} attempts: {message}",
|
|
600
|
+
moment,
|
|
601
|
+
)
|
|
602
|
+
_logger.error(
|
|
603
|
+
"notification failed terminally",
|
|
604
|
+
notification_id=str(notification.id),
|
|
605
|
+
notifier=notification.notifier,
|
|
606
|
+
attempts=notification.attempt,
|
|
607
|
+
error=message,
|
|
608
|
+
)
|
|
609
|
+
return False
|
|
610
|
+
delay = services.settings.notification_backoff.total_seconds() * (2 ** (notification.attempt - 1))
|
|
611
|
+
notification.status = NotificationStatus.PENDING
|
|
612
|
+
notification.available_at = moment + timedelta(seconds=delay)
|
|
613
|
+
_logger.warning(
|
|
614
|
+
"notification delivery failed, will retry",
|
|
615
|
+
notification_id=str(notification.id),
|
|
616
|
+
attempt=notification.attempt,
|
|
617
|
+
delay_seconds=round(delay, 1),
|
|
618
|
+
error=message,
|
|
619
|
+
)
|
|
620
|
+
return False
|
|
621
|
+
|
|
622
|
+
|
|
623
|
+
def _timeline(
|
|
624
|
+
session: AsyncSession,
|
|
625
|
+
notification: Notification,
|
|
626
|
+
level: LogLevel,
|
|
627
|
+
message: str,
|
|
628
|
+
moment: datetime,
|
|
629
|
+
) -> None:
|
|
630
|
+
"""Put a delivery outcome into the run's timeline."""
|
|
631
|
+
if notification.run_id is None:
|
|
632
|
+
return
|
|
633
|
+
session.add(
|
|
634
|
+
LogEntry(
|
|
635
|
+
run_id=notification.run_id,
|
|
636
|
+
level=level,
|
|
637
|
+
message=message,
|
|
638
|
+
fields={"notifier": notification.notifier, "event": notification.event.value},
|
|
639
|
+
created_at=moment,
|
|
640
|
+
)
|
|
641
|
+
)
|
|
642
|
+
|
|
643
|
+
|
|
644
|
+
async def list_notifications(
|
|
645
|
+
session: AsyncSession,
|
|
646
|
+
*,
|
|
647
|
+
run_id: UUID | None = None,
|
|
648
|
+
notification_status: NotificationStatus | None = None,
|
|
649
|
+
notifier: str | None = None,
|
|
650
|
+
after: UUID | None = None,
|
|
651
|
+
limit: int | None = None,
|
|
652
|
+
) -> list[Notification]:
|
|
653
|
+
"""List notifications newest first, for one run or across the instance.
|
|
654
|
+
|
|
655
|
+
The filters are the server's own, so a row they leave out is one the caller never reads
|
|
656
|
+
and a cursor walk does not have to be told which pages to skip.
|
|
657
|
+
"""
|
|
658
|
+
statement = sa.select(Notification).order_by(Notification.id.desc())
|
|
659
|
+
if run_id is not None:
|
|
660
|
+
statement = statement.where(Notification.run_id == run_id)
|
|
661
|
+
if notification_status is not None:
|
|
662
|
+
statement = statement.where(Notification.status == notification_status)
|
|
663
|
+
if notifier is not None:
|
|
664
|
+
statement = statement.where(Notification.notifier == notifier)
|
|
665
|
+
if after is not None:
|
|
666
|
+
statement = statement.where(Notification.id < after)
|
|
667
|
+
if limit is not None:
|
|
668
|
+
statement = statement.limit(limit)
|
|
669
|
+
rows = await session.execute(statement)
|
|
670
|
+
return list(rows.scalars())
|
|
671
|
+
|
|
672
|
+
|
|
673
|
+
async def find_notification(session: AsyncSession, notification_id: UUID) -> Notification | None:
|
|
674
|
+
"""Find one notification by id."""
|
|
675
|
+
return await session.get(Notification, notification_id)
|
|
676
|
+
|
|
677
|
+
|
|
678
|
+
async def retry_notification(
|
|
679
|
+
session: AsyncSession,
|
|
680
|
+
notification: Notification,
|
|
681
|
+
*,
|
|
682
|
+
now: datetime | None = None,
|
|
683
|
+
) -> Notification:
|
|
684
|
+
"""Put one notification back on the queue, due now.
|
|
685
|
+
|
|
686
|
+
An explicit retry starts the delivery over rather than adding one try to a budget that is
|
|
687
|
+
already spent: a row that failed terminally has no attempts left, and a retry that left the
|
|
688
|
+
counter where it was would fail again without calling the notifier at all. So the backoff,
|
|
689
|
+
the counter and the last refusal all go, and a worker claims the row on its next pass.
|
|
690
|
+
|
|
691
|
+
A row a worker is holding is left alone: its lease is live, and returning it to the queue
|
|
692
|
+
would hand the same message to a second worker.
|
|
693
|
+
"""
|
|
694
|
+
if notification.status is NotificationStatus.SENDING:
|
|
695
|
+
raise AlertError("a worker is delivering this one; wait for it to finish or fail")
|
|
696
|
+
notification.status = NotificationStatus.PENDING
|
|
697
|
+
notification.available_at = now or utcnow()
|
|
698
|
+
notification.attempt = 0
|
|
699
|
+
notification.error = None
|
|
700
|
+
notification.sent_at = None
|
|
701
|
+
notification.lease_owner = None
|
|
702
|
+
notification.lease_expires_at = None
|
|
703
|
+
await session.flush()
|
|
704
|
+
_logger.info("notification queued again", notification_id=str(notification.id), notifier=notification.notifier)
|
|
705
|
+
return notification
|
|
706
|
+
|
|
707
|
+
|
|
708
|
+
async def recover_notifications(session: AsyncSession, *, now: datetime | None = None) -> int:
|
|
709
|
+
"""Return notifications whose worker died back to the queue, like the lease sweeper does.
|
|
710
|
+
|
|
711
|
+
One conditional statement rather than a select and a write: a renewal or a delivery that
|
|
712
|
+
commits between the two would otherwise be overwritten, requeueing a notification another
|
|
713
|
+
worker is still sending or has already sent.
|
|
714
|
+
"""
|
|
715
|
+
moment = now or utcnow()
|
|
716
|
+
recovered = await session.execute(
|
|
717
|
+
sa.update(Notification)
|
|
718
|
+
.where(
|
|
719
|
+
Notification.status == NotificationStatus.SENDING,
|
|
720
|
+
Notification.lease_expires_at.is_not(None),
|
|
721
|
+
Notification.lease_expires_at < moment,
|
|
722
|
+
)
|
|
723
|
+
.values(
|
|
724
|
+
status=NotificationStatus.PENDING,
|
|
725
|
+
available_at=moment,
|
|
726
|
+
lease_owner=None,
|
|
727
|
+
lease_expires_at=None,
|
|
728
|
+
)
|
|
729
|
+
.returning(Notification.id)
|
|
730
|
+
.execution_options(synchronize_session=False)
|
|
731
|
+
)
|
|
732
|
+
return len(recovered.scalars().all())
|
|
733
|
+
|
|
734
|
+
|
|
735
|
+
class NotificationDispatcher(BaseModel):
|
|
736
|
+
"""The worker's alert-delivery loop: claim one, send it, repeat until the queue is empty."""
|
|
737
|
+
|
|
738
|
+
model_config = ConfigDict(arbitrary_types_allowed=True, frozen=True)
|
|
739
|
+
|
|
740
|
+
sessions: async_sessionmaker[AsyncSession]
|
|
741
|
+
services: EngineServices
|
|
742
|
+
owner: str
|
|
743
|
+
max_per_pass: int = Field(default=25, ge=1)
|
|
744
|
+
|
|
745
|
+
async def drain(self, *, now: datetime | None = None) -> int:
|
|
746
|
+
"""Deliver every notification that is due, and report how many were sent.
|
|
747
|
+
|
|
748
|
+
Each delivery is its own transaction, so one undeliverable message never rolls back
|
|
749
|
+
the ones that went out beside it.
|
|
750
|
+
"""
|
|
751
|
+
sent = 0
|
|
752
|
+
for _ in range(self.max_per_pass):
|
|
753
|
+
async with session_scope(self.sessions) as session:
|
|
754
|
+
moment = now or utcnow()
|
|
755
|
+
notification = await claim_notification(
|
|
756
|
+
session,
|
|
757
|
+
owner=self.owner,
|
|
758
|
+
now=moment,
|
|
759
|
+
lease_seconds=int(self.services.settings.notification_lease.total_seconds()),
|
|
760
|
+
)
|
|
761
|
+
if notification is None:
|
|
762
|
+
return sent
|
|
763
|
+
async with session_scope(self.sessions) as session:
|
|
764
|
+
claimed = await session.get(Notification, notification.id)
|
|
765
|
+
if claimed is None: # pragma: no cover - it was just claimed
|
|
766
|
+
continue
|
|
767
|
+
async with self._renewing(claimed.id):
|
|
768
|
+
delivered = await send_notification(session, self.services, claimed, owner=self.owner, now=now)
|
|
769
|
+
if delivered:
|
|
770
|
+
sent += 1
|
|
771
|
+
return sent
|
|
772
|
+
|
|
773
|
+
@asynccontextmanager
|
|
774
|
+
async def _renewing(self, notification_id: UUID) -> AsyncGenerator[None]:
|
|
775
|
+
"""Hold a notification's lease open for as long as its delivery is in flight."""
|
|
776
|
+
halting = asyncio.Event()
|
|
777
|
+
renewing = asyncio.create_task(self._renew_until_halted(notification_id, halting))
|
|
778
|
+
try:
|
|
779
|
+
yield
|
|
780
|
+
finally:
|
|
781
|
+
halting.set()
|
|
782
|
+
await renewing
|
|
783
|
+
|
|
784
|
+
async def _renew_until_halted(self, notification_id: UUID, halting: asyncio.Event) -> None:
|
|
785
|
+
"""Extend the lease on a cadence until halted, or until the row is no longer this worker's.
|
|
786
|
+
|
|
787
|
+
The halt is a flag rather than a cancel: a renewal already inside its transaction runs
|
|
788
|
+
to its own end, where a cancel landing in a database await strands the session's
|
|
789
|
+
connection, checked out of the pool and never closed.
|
|
790
|
+
"""
|
|
791
|
+
lease_seconds = int(self.services.settings.notification_lease.total_seconds())
|
|
792
|
+
while not halting.is_set():
|
|
793
|
+
with suppress(TimeoutError):
|
|
794
|
+
await asyncio.wait_for(halting.wait(), timeout=lease_seconds / RENEWALS_PER_LEASE)
|
|
795
|
+
if halting.is_set():
|
|
796
|
+
return
|
|
797
|
+
try:
|
|
798
|
+
async with session_scope(self.sessions) as session:
|
|
799
|
+
if not await renew_lease(session, notification_id, owner=self.owner, lease_seconds=lease_seconds):
|
|
800
|
+
return
|
|
801
|
+
except Exception as error: # the delivery runs on; the sweeper reclaims if it outlasts the lease
|
|
802
|
+
_logger.warning(
|
|
803
|
+
"notification lease renewal failed", notification_id=str(notification_id), error=str(error)
|
|
804
|
+
)
|
|
805
|
+
return
|