dirigent-server 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dirigent_server/__init__.py +7 -0
- dirigent_server/app.py +171 -0
- dirigent_server/dependencies.py +51 -0
- dirigent_server/errors.py +116 -0
- dirigent_server/health.py +97 -0
- dirigent_server/logging.py +5 -0
- dirigent_server/pagination.py +56 -0
- dirigent_server/py.typed +0 -0
- dirigent_server/routes/__init__.py +79 -0
- dirigent_server/routes/alerts.py +289 -0
- dirigent_server/routes/auth.py +218 -0
- dirigent_server/routes/blocks.py +39 -0
- dirigent_server/routes/connections.py +294 -0
- dirigent_server/routes/hooks.py +114 -0
- dirigent_server/routes/pipelines.py +427 -0
- dirigent_server/routes/runs.py +818 -0
- dirigent_server/routes/schema.py +27 -0
- dirigent_server/routes/schemas.py +127 -0
- dirigent_server/routes/system.py +63 -0
- dirigent_server/routes/trigger_documents.py +109 -0
- dirigent_server/routes/triggers.py +537 -0
- dirigent_server/routes/users.py +226 -0
- dirigent_server/routes/workers.py +52 -0
- dirigent_server/security.py +186 -0
- dirigent_server/static/.gitkeep +0 -0
- dirigent_server/transactions.py +34 -0
- dirigent_server/ui.py +237 -0
- dirigent_server-0.9.0.dist-info/METADATA +17 -0
- dirigent_server-0.9.0.dist-info/RECORD +32 -0
- dirigent_server-0.9.0.dist-info/WHEEL +4 -0
- dirigent_server-0.9.0.dist-info/licenses/LICENSE +18 -0
- dirigent_server-0.9.0.dist-info/licenses/THIRD_PARTY_NOTICES.md +631 -0
|
@@ -0,0 +1,818 @@
|
|
|
1
|
+
"""Runs: the list, the detail with its DAG view model, cancellation, retry, logs, events, and a report."""
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
from collections.abc import AsyncGenerator, Generator, Mapping, Sequence
|
|
5
|
+
from contextlib import contextmanager
|
|
6
|
+
from datetime import UTC, datetime, timedelta
|
|
7
|
+
from typing import Annotated
|
|
8
|
+
from uuid import UUID
|
|
9
|
+
|
|
10
|
+
import sqlalchemy as sa
|
|
11
|
+
from fastapi import APIRouter, Header, HTTPException, Query, Request, status
|
|
12
|
+
from fastapi.responses import StreamingResponse
|
|
13
|
+
from sqlalchemy.ext.asyncio import AsyncSession, async_sessionmaker
|
|
14
|
+
|
|
15
|
+
from dirigent_client.enums import AttemptStatus, LogLevel, RunItemStatus, RunStatus
|
|
16
|
+
from dirigent_client.schemas import (
|
|
17
|
+
TERMINAL_RUN_STATUSES,
|
|
18
|
+
AttemptEvent,
|
|
19
|
+
AttemptOut,
|
|
20
|
+
DagNode,
|
|
21
|
+
DagView,
|
|
22
|
+
ItemOut,
|
|
23
|
+
LogEntryOut,
|
|
24
|
+
Page,
|
|
25
|
+
RunDetail,
|
|
26
|
+
RunOut,
|
|
27
|
+
RunReport,
|
|
28
|
+
StepReport,
|
|
29
|
+
)
|
|
30
|
+
from dirigent_common.durations import DurationError, parse_duration
|
|
31
|
+
from dirigent_core import telemetry
|
|
32
|
+
from dirigent_core.database import session_scope
|
|
33
|
+
from dirigent_core.engine.definition import PipelineDefinition, load_definition
|
|
34
|
+
from dirigent_core.engine.runs import RunCreationError, cancel_run, retry_step
|
|
35
|
+
from dirigent_core.engine.state import (
|
|
36
|
+
StepCounts,
|
|
37
|
+
attempt_counts,
|
|
38
|
+
build_step_states,
|
|
39
|
+
in_execution_order,
|
|
40
|
+
item_counts,
|
|
41
|
+
step_states,
|
|
42
|
+
)
|
|
43
|
+
from dirigent_core.models import (
|
|
44
|
+
ArtifactRef,
|
|
45
|
+
LogEntry,
|
|
46
|
+
Pipeline,
|
|
47
|
+
PipelineVersion,
|
|
48
|
+
Run,
|
|
49
|
+
RunItem,
|
|
50
|
+
StepAttempt,
|
|
51
|
+
utcnow,
|
|
52
|
+
)
|
|
53
|
+
from dirigent_core.pipelines import UNWELL, carries_tag, failing_steps
|
|
54
|
+
from dirigent_core.registry import unmet_worker_tags
|
|
55
|
+
from dirigent_server.dependencies import ServicesDep, SessionDep, get_sessions
|
|
56
|
+
from dirigent_server.pagination import DEFAULT_PAGE, AfterParam, LimitParam, clip, int_cursor, uuid_cursor
|
|
57
|
+
from dirigent_server.security import OperatorDep, PrincipalDep
|
|
58
|
+
from dirigent_server.transactions import Transactional
|
|
59
|
+
|
|
60
|
+
router = APIRouter(route_class=Transactional, tags=["runs"])
|
|
61
|
+
|
|
62
|
+
DEFAULT_LOG_PAGE = 200
|
|
63
|
+
|
|
64
|
+
FOLLOW_INTERVAL_SECONDS = 0.5
|
|
65
|
+
FOLLOW_MAX_INTERVAL_SECONDS = 5.0
|
|
66
|
+
FOLLOW_BACKOFF = 1.5
|
|
67
|
+
FOLLOW_MAX_SECONDS = 3600.0
|
|
68
|
+
|
|
69
|
+
#: Each open tail is a database connection's worth of periodic queries against the pool
|
|
70
|
+
#: every request shares.
|
|
71
|
+
MAX_TAILS_PER_PRINCIPAL = 8
|
|
72
|
+
|
|
73
|
+
#: The header a browser's EventSource resends on its own, carrying the last id it saw.
|
|
74
|
+
LAST_EVENT_ID = Annotated[
|
|
75
|
+
str | None,
|
|
76
|
+
Header(alias="Last-Event-ID", description="The last log id delivered; a reconnect resumes past it."),
|
|
77
|
+
]
|
|
78
|
+
|
|
79
|
+
#: Process-local, so a cluster's total is this times the number of servers.
|
|
80
|
+
OPEN_TAILS: dict[str, int] = {}
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
async def _run_row(session: AsyncSession, run_id: UUID) -> Run:
|
|
84
|
+
"""Read a run, or say this instance has no such run."""
|
|
85
|
+
run = await session.get(Run, run_id)
|
|
86
|
+
if run is None:
|
|
87
|
+
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail=f"no run {run_id}")
|
|
88
|
+
return run
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
async def _context(session: AsyncSession, run: Run) -> tuple[Pipeline, PipelineVersion, PipelineDefinition]:
|
|
92
|
+
"""Read the pipeline and pinned version a run points at, plus its parsed definition."""
|
|
93
|
+
pipeline = await session.get(Pipeline, run.pipeline_id)
|
|
94
|
+
version = await session.get(PipelineVersion, run.pipeline_version_id)
|
|
95
|
+
if pipeline is None or version is None: # pragma: no cover - both are RESTRICT foreign keys
|
|
96
|
+
raise HTTPException(status_code=status.HTTP_409_CONFLICT, detail="the run's definition is gone")
|
|
97
|
+
return pipeline, version, load_definition(version.ordered_document)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _render_run(run: Run, pipeline: Pipeline, version: PipelineVersion, failed_step: str | None = None) -> RunOut:
|
|
101
|
+
"""Render a run row with the names a client actually wants to see."""
|
|
102
|
+
return RunOut(
|
|
103
|
+
id=run.id,
|
|
104
|
+
pipeline=pipeline.code,
|
|
105
|
+
pipeline_version=version.version,
|
|
106
|
+
status=run.status,
|
|
107
|
+
priority=run.priority,
|
|
108
|
+
params=dict(run.params),
|
|
109
|
+
triggered_by_kind=run.triggered_by_kind,
|
|
110
|
+
triggered_by_label=run.triggered_by_label,
|
|
111
|
+
trace_id=telemetry.trace_id_of(run.traceparent),
|
|
112
|
+
error=run.error,
|
|
113
|
+
failed_step=failed_step,
|
|
114
|
+
started_at=run.started_at,
|
|
115
|
+
finished_at=run.finished_at,
|
|
116
|
+
window_start=run.window_start,
|
|
117
|
+
window_end=run.window_end,
|
|
118
|
+
log_levels={pattern: LogLevel(str(level)) for pattern, level in run.log_levels.items()}
|
|
119
|
+
if run.log_levels
|
|
120
|
+
else None,
|
|
121
|
+
created_at=run.created_at,
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
async def _attempts(session: AsyncSession, run_id: UUID, step_order: Sequence[str] = ()) -> list[StepAttempt]:
|
|
126
|
+
"""Read every attempt of a run, in the order they ran."""
|
|
127
|
+
rows = await session.execute(
|
|
128
|
+
sa.select(StepAttempt, RunItem.item_index)
|
|
129
|
+
.outerjoin(RunItem, RunItem.id == StepAttempt.run_item_id)
|
|
130
|
+
.where(StepAttempt.run_id == run_id)
|
|
131
|
+
)
|
|
132
|
+
found = rows.all()
|
|
133
|
+
indexes = {attempt.id: index for attempt, index in found if index is not None}
|
|
134
|
+
return in_execution_order([attempt for attempt, _ in found], indexes, step_order)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
async def _warnings(session: AsyncSession, run_id: UUID) -> dict[str, int]:
|
|
138
|
+
"""Count the warnings and errors each step logged, which a succeeded step can still have."""
|
|
139
|
+
rows = await session.execute(
|
|
140
|
+
sa.select(LogEntry.step_name, sa.func.count())
|
|
141
|
+
.where(LogEntry.run_id == run_id, LogEntry.level.in_((LogLevel.WARNING, LogLevel.ERROR)))
|
|
142
|
+
.group_by(LogEntry.step_name)
|
|
143
|
+
)
|
|
144
|
+
return {name: count for name, count in rows.all() if name is not None}
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
async def _items(session: AsyncSession, run_id: UUID) -> list[RunItem]:
|
|
148
|
+
"""Read a run's fan-out items, in grid order."""
|
|
149
|
+
rows = await session.execute(
|
|
150
|
+
sa.select(RunItem).where(RunItem.run_id == run_id).order_by(RunItem.step_name, RunItem.item_index)
|
|
151
|
+
)
|
|
152
|
+
return list(rows.scalars())
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _human_order(definition: PipelineDefinition, ran: Sequence[str]) -> list[str]:
|
|
156
|
+
"""Name a run's steps the way a person watched them: as they ran, then as they were written.
|
|
157
|
+
|
|
158
|
+
The steps that ran arrive in execution order, which already carries the written order for
|
|
159
|
+
the steps that have not, so following them says both things at once.
|
|
160
|
+
"""
|
|
161
|
+
seen = list(dict.fromkeys(ran))
|
|
162
|
+
return [name for name in seen if name in definition.steps] + [name for name in definition.steps if name not in seen]
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _dag(
|
|
166
|
+
definition: PipelineDefinition,
|
|
167
|
+
attempts: Mapping[str, StepCounts],
|
|
168
|
+
items: Mapping[str, dict[RunItemStatus, int]],
|
|
169
|
+
) -> DagView:
|
|
170
|
+
"""Fold the pinned definition and the run's counts into the graph the UI draws.
|
|
171
|
+
|
|
172
|
+
IN THE ORDER THE STEPS WERE WRITTEN, WHICH IS THE ONE ORDER THAT DOES NOT MOVE. A graph is a
|
|
173
|
+
shape, and the same run read twice has to be the same shape: ordering the nodes by how the
|
|
174
|
+
run went would lay the boxes out one way while it is in flight and another once it settles,
|
|
175
|
+
and elk is told to respect the order it is given.
|
|
176
|
+
"""
|
|
177
|
+
states = step_states(definition, {name: counts.latest for name, counts in attempts.items()})
|
|
178
|
+
nodes: list[DagNode] = []
|
|
179
|
+
edges: list[tuple[str, str]] = []
|
|
180
|
+
for name, step in definition.steps.items():
|
|
181
|
+
of_step = items.get(name, {})
|
|
182
|
+
nodes.append(
|
|
183
|
+
DagNode(
|
|
184
|
+
code=name,
|
|
185
|
+
name=step.name,
|
|
186
|
+
block=step.block,
|
|
187
|
+
outcome=states[name].outcome.value,
|
|
188
|
+
depends_on=list(step.depends_on),
|
|
189
|
+
rule=step.rule.value,
|
|
190
|
+
fan_out=step.is_fan_out,
|
|
191
|
+
items_total=sum(of_step.values()),
|
|
192
|
+
items_failed=of_step.get(RunItemStatus.FAILED, 0),
|
|
193
|
+
attempts=attempts[name].total if name in attempts else 0,
|
|
194
|
+
)
|
|
195
|
+
)
|
|
196
|
+
edges.extend((dependency, name) for dependency in step.depends_on)
|
|
197
|
+
return DagView(nodes=nodes, edges=edges)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
@router.get("/runs", operation_id="listRuns", summary="List runs", response_model=Page[RunOut])
|
|
201
|
+
async def list_runs(
|
|
202
|
+
session: SessionDep,
|
|
203
|
+
principal: PrincipalDep,
|
|
204
|
+
pipeline: Annotated[str | None, Query(description="Only runs of this pipeline.")] = None,
|
|
205
|
+
run_status: Annotated[RunStatus | None, Query(alias="status", description="Only runs in this state.")] = None,
|
|
206
|
+
since: Annotated[str | None, Query(description="Only runs created within this window, e.g. 24h.")] = None,
|
|
207
|
+
tag: Annotated[
|
|
208
|
+
list[str] | None, Query(description="Only runs whose pipeline wears this tag; repeat it to name more.")
|
|
209
|
+
] = None,
|
|
210
|
+
after: AfterParam = None,
|
|
211
|
+
limit: LimitParam = DEFAULT_PAGE,
|
|
212
|
+
) -> Page[RunOut]:
|
|
213
|
+
"""List runs newest first, filtered by pipeline, status, tag, and how far back to look.
|
|
214
|
+
|
|
215
|
+
``tag`` repeats, and repeating it narrows. It asks the pipeline the run is of what it
|
|
216
|
+
wears now: a run pins its version, never its pipeline's tags, so retagging a pipeline
|
|
217
|
+
changes which runs this answers with.
|
|
218
|
+
"""
|
|
219
|
+
statement = (
|
|
220
|
+
sa.select(Run, Pipeline, PipelineVersion)
|
|
221
|
+
.join(Pipeline, Pipeline.id == Run.pipeline_id)
|
|
222
|
+
.join(PipelineVersion, PipelineVersion.id == Run.pipeline_version_id)
|
|
223
|
+
.order_by(Run.id.desc())
|
|
224
|
+
.limit(limit + 1)
|
|
225
|
+
)
|
|
226
|
+
if pipeline is not None:
|
|
227
|
+
statement = statement.where(Pipeline.code == pipeline)
|
|
228
|
+
if run_status is not None:
|
|
229
|
+
statement = statement.where(Run.status == run_status)
|
|
230
|
+
if since is not None:
|
|
231
|
+
statement = statement.where(Run.created_at >= utcnow() - _window(since))
|
|
232
|
+
for one in tag or ():
|
|
233
|
+
statement = statement.where(carries_tag(session, one))
|
|
234
|
+
cursor = uuid_cursor(after)
|
|
235
|
+
if cursor is not None:
|
|
236
|
+
statement = statement.where(Run.id < cursor)
|
|
237
|
+
rows = await session.execute(statement)
|
|
238
|
+
read = rows.all()
|
|
239
|
+
failing = await failing_steps(session, [run.id for run, _, _ in read if run.status in UNWELL])
|
|
240
|
+
found = [_render_run(run, name, version, failing.get(run.id)) for run, name, version in read]
|
|
241
|
+
items, following = clip(found, limit, lambda row: row.id)
|
|
242
|
+
return Page(items=items, next=following)
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _window(since: str) -> timedelta:
|
|
246
|
+
"""Parse a humane window such as ``24h``, refusing anything the format does not define."""
|
|
247
|
+
try:
|
|
248
|
+
parsed = parse_duration(since)
|
|
249
|
+
except DurationError as error:
|
|
250
|
+
raise HTTPException(status_code=status.HTTP_422_UNPROCESSABLE_CONTENT, detail=str(error)) from error
|
|
251
|
+
if not isinstance(parsed, timedelta): # pragma: no cover - parse_duration returns one or raises
|
|
252
|
+
raise HTTPException(status_code=status.HTTP_422_UNPROCESSABLE_CONTENT, detail=f"{since!r} is not a duration")
|
|
253
|
+
return parsed
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
@router.get("/runs/{run_id}", operation_id="getRun", summary="Read a run", response_model=RunDetail)
|
|
257
|
+
async def get_run(run_id: UUID, session: SessionDep, principal: PrincipalDep) -> RunDetail:
|
|
258
|
+
"""Read a run with the DAG view model, and how many items and attempts it has."""
|
|
259
|
+
run = await _run_row(session, run_id)
|
|
260
|
+
pipeline, version, definition = await _context(session, run)
|
|
261
|
+
attempts = await attempt_counts(session, run_id)
|
|
262
|
+
items = await item_counts(session, run_id)
|
|
263
|
+
unmet = await unmet_worker_tags(session, run.worker_tags) if run.status is RunStatus.QUEUED else []
|
|
264
|
+
return RunDetail(
|
|
265
|
+
run=_render_run(run, pipeline, version),
|
|
266
|
+
dag=_dag(definition, attempts, items),
|
|
267
|
+
items_total=sum(sum(of_step.values()) for of_step in items.values()),
|
|
268
|
+
attempts_total=sum(counts.total for counts in attempts.values()),
|
|
269
|
+
waiting_for_workers=unmet or None,
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
@router.get(
|
|
274
|
+
"/runs/{run_id}/items",
|
|
275
|
+
operation_id="listRunItems",
|
|
276
|
+
summary="List a run's fan-out items",
|
|
277
|
+
response_model=Page[ItemOut],
|
|
278
|
+
)
|
|
279
|
+
async def list_items(
|
|
280
|
+
run_id: UUID,
|
|
281
|
+
session: SessionDep,
|
|
282
|
+
principal: PrincipalDep,
|
|
283
|
+
after: AfterParam = None,
|
|
284
|
+
limit: LimitParam = DEFAULT_PAGE,
|
|
285
|
+
) -> Page[ItemOut]:
|
|
286
|
+
"""List a run's fan-out items in the order they were created, which is grid order."""
|
|
287
|
+
await _run_row(session, run_id)
|
|
288
|
+
statement = sa.select(RunItem).where(RunItem.run_id == run_id).order_by(RunItem.id).limit(limit + 1)
|
|
289
|
+
cursor = uuid_cursor(after)
|
|
290
|
+
if cursor is not None:
|
|
291
|
+
statement = statement.where(RunItem.id > cursor)
|
|
292
|
+
rows = list((await session.execute(statement)).scalars())
|
|
293
|
+
found = [ItemOut.model_validate(row, from_attributes=True) for row in rows]
|
|
294
|
+
items, following = clip(found, limit, lambda row: row.id)
|
|
295
|
+
return Page(items=items, next=following)
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
@router.get(
|
|
299
|
+
"/runs/{run_id}/attempts",
|
|
300
|
+
operation_id="listRunAttempts",
|
|
301
|
+
summary="List a run's attempts",
|
|
302
|
+
response_model=Page[AttemptOut],
|
|
303
|
+
)
|
|
304
|
+
async def list_attempts(
|
|
305
|
+
run_id: UUID,
|
|
306
|
+
session: SessionDep,
|
|
307
|
+
principal: PrincipalDep,
|
|
308
|
+
step: Annotated[str | None, Query(description="Only attempts of this step.")] = None,
|
|
309
|
+
attempt_status: Annotated[
|
|
310
|
+
AttemptStatus | None, Query(alias="status", description="Only attempts in this state.")
|
|
311
|
+
] = None,
|
|
312
|
+
after: AfterParam = None,
|
|
313
|
+
limit: LimitParam = DEFAULT_PAGE,
|
|
314
|
+
) -> Page[AttemptOut]:
|
|
315
|
+
"""List a run's attempts in the order they were created, filtered by step and by state."""
|
|
316
|
+
await _run_row(session, run_id)
|
|
317
|
+
statement = sa.select(StepAttempt).where(StepAttempt.run_id == run_id).order_by(StepAttempt.id).limit(limit + 1)
|
|
318
|
+
if step is not None:
|
|
319
|
+
statement = statement.where(StepAttempt.step_name == step)
|
|
320
|
+
if attempt_status is not None:
|
|
321
|
+
statement = statement.where(StepAttempt.status == attempt_status)
|
|
322
|
+
cursor = uuid_cursor(after)
|
|
323
|
+
if cursor is not None:
|
|
324
|
+
statement = statement.where(StepAttempt.id > cursor)
|
|
325
|
+
rows = list((await session.execute(statement)).scalars())
|
|
326
|
+
spilled = await _spilled(session, run_id, [row.id for row in rows])
|
|
327
|
+
found = [_render_attempt(row, spilled) for row in rows]
|
|
328
|
+
items, following = clip(found, limit, lambda row: row.id)
|
|
329
|
+
return Page(items=items, next=following)
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
async def _spilled(
|
|
333
|
+
session: AsyncSession, run_id: UUID, only: Sequence[UUID] | None = None
|
|
334
|
+
) -> dict[UUID, tuple[str, int | None]]:
|
|
335
|
+
"""Map each attempt whose output went to storage to the URI and size it went as.
|
|
336
|
+
|
|
337
|
+
An artifact that inlined is not in here: the attempt row already carries its value.
|
|
338
|
+
"""
|
|
339
|
+
statement = sa.select(ArtifactRef.step_attempt_id, ArtifactRef.uri, ArtifactRef.size_bytes).where(
|
|
340
|
+
ArtifactRef.run_id == run_id, ArtifactRef.uri.is_not(None)
|
|
341
|
+
)
|
|
342
|
+
if only is not None:
|
|
343
|
+
statement = statement.where(ArtifactRef.step_attempt_id.in_(only))
|
|
344
|
+
rows = await session.execute(statement)
|
|
345
|
+
return {found: (uri, size) for found, uri, size in rows.all() if found is not None and uri is not None}
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def _render_attempt(attempt: StepAttempt, spilled: dict[UUID, tuple[str, int | None]]) -> AttemptOut:
|
|
349
|
+
"""Render one attempt, naming the artifact its output was written to when there is one."""
|
|
350
|
+
uri, size = spilled.get(attempt.id, (None, None))
|
|
351
|
+
return AttemptOut.model_validate(attempt, from_attributes=True).model_copy(
|
|
352
|
+
update={"output_uri": uri, "output_bytes": size}
|
|
353
|
+
)
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
@router.post("/runs/{run_id}/$cancel", operation_id="cancelRun", summary="Cancel a run", response_model=RunOut)
|
|
357
|
+
async def cancel(run_id: UUID, session: SessionDep, services: ServicesDep, principal: OperatorDep) -> RunOut:
|
|
358
|
+
"""Stop what has not started, and tell the remote about what has."""
|
|
359
|
+
run = await _run_row(session, run_id)
|
|
360
|
+
pipeline, version, _ = await _context(session, run)
|
|
361
|
+
await cancel_run(session, services, run, reason=f"cancelled by {principal.label}")
|
|
362
|
+
return _render_run(run, pipeline, version)
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
@router.post(
|
|
366
|
+
"/attempts/{attempt_id}/$retry",
|
|
367
|
+
operation_id="retryAttempt",
|
|
368
|
+
summary="Retry a failed step",
|
|
369
|
+
response_model=AttemptOut,
|
|
370
|
+
status_code=status.HTTP_202_ACCEPTED,
|
|
371
|
+
)
|
|
372
|
+
async def retry(
|
|
373
|
+
attempt_id: UUID,
|
|
374
|
+
session: SessionDep,
|
|
375
|
+
services: ServicesDep,
|
|
376
|
+
principal: OperatorDep,
|
|
377
|
+
idempotency_key: Annotated[
|
|
378
|
+
str | None,
|
|
379
|
+
Header(alias="Idempotency-Key", description="Required, so a retried request creates one attempt."),
|
|
380
|
+
] = None,
|
|
381
|
+
) -> AttemptOut:
|
|
382
|
+
"""Create one manual attempt of a failed step, reading its upstream stored outputs."""
|
|
383
|
+
if not idempotency_key:
|
|
384
|
+
raise HTTPException(
|
|
385
|
+
status_code=status.HTTP_400_BAD_REQUEST,
|
|
386
|
+
detail="an Idempotency-Key header is required, so a retried request creates one attempt",
|
|
387
|
+
)
|
|
388
|
+
attempt = await session.get(StepAttempt, attempt_id)
|
|
389
|
+
if attempt is None:
|
|
390
|
+
raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail=f"no attempt {attempt_id}")
|
|
391
|
+
run = await _run_row(session, attempt.run_id)
|
|
392
|
+
try:
|
|
393
|
+
created = await retry_step(
|
|
394
|
+
session,
|
|
395
|
+
services,
|
|
396
|
+
run,
|
|
397
|
+
attempt.step_name,
|
|
398
|
+
idempotency_key=idempotency_key,
|
|
399
|
+
run_item_id=attempt.run_item_id,
|
|
400
|
+
)
|
|
401
|
+
except RunCreationError as error:
|
|
402
|
+
raise HTTPException(status_code=status.HTTP_409_CONFLICT, detail=str(error)) from error
|
|
403
|
+
return AttemptOut.model_validate(created, from_attributes=True)
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
async def _log_page(session: AsyncSession, run_id: UUID, after: int, limit: int, step: str | None) -> Page[LogEntryOut]:
|
|
407
|
+
"""Read one page of a run's log entries, in write order."""
|
|
408
|
+
statement = (
|
|
409
|
+
sa.select(LogEntry).where(LogEntry.run_id == run_id, LogEntry.id > after).order_by(LogEntry.id).limit(limit + 1)
|
|
410
|
+
)
|
|
411
|
+
if step is not None:
|
|
412
|
+
statement = statement.where(LogEntry.step_name == step)
|
|
413
|
+
rows = list((await session.execute(statement)).scalars())
|
|
414
|
+
found = [LogEntryOut.model_validate(row, from_attributes=True) for row in rows]
|
|
415
|
+
items, following = clip(found, limit, lambda entry: entry.id)
|
|
416
|
+
return Page(items=items, next=following)
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
@router.get(
|
|
420
|
+
"/runs/{run_id}/$logs",
|
|
421
|
+
operation_id="getRunLogs",
|
|
422
|
+
summary="Read a run's log entries",
|
|
423
|
+
response_model=Page[LogEntryOut],
|
|
424
|
+
responses={200: {"content": {"text/event-stream": {}}, "description": "A page, or an SSE tail."}},
|
|
425
|
+
)
|
|
426
|
+
async def read_logs(
|
|
427
|
+
run_id: UUID,
|
|
428
|
+
request: Request,
|
|
429
|
+
session: SessionDep,
|
|
430
|
+
principal: PrincipalDep,
|
|
431
|
+
after: AfterParam = None,
|
|
432
|
+
limit: LimitParam = DEFAULT_LOG_PAGE,
|
|
433
|
+
step: Annotated[str | None, Query(description="Only entries from this step.")] = None,
|
|
434
|
+
follow: Annotated[str | None, Query(description="Set to sse to stream new entries as they land.")] = None,
|
|
435
|
+
last_event_id: LAST_EVENT_ID = None,
|
|
436
|
+
) -> Page[LogEntryOut] | StreamingResponse:
|
|
437
|
+
"""Serve a page of log entries, or an SSE tail that follows the run to its end."""
|
|
438
|
+
await _run_row(session, run_id)
|
|
439
|
+
start = _resume_from(after, last_event_id)
|
|
440
|
+
if follow != "sse":
|
|
441
|
+
return await _log_page(session, run_id, start, limit, step)
|
|
442
|
+
watcher = _claim(principal.user_id)
|
|
443
|
+
return StreamingResponse(
|
|
444
|
+
_tail(get_sessions(request), run_id, start, step, watcher),
|
|
445
|
+
media_type="text/event-stream",
|
|
446
|
+
headers={"cache-control": "no-cache", "x-accel-buffering": "no"},
|
|
447
|
+
)
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
def _resume_from(after: str | None, last_event_id: str | None) -> int:
|
|
451
|
+
"""Name the log id a stream starts past: an explicit cursor, else the one a reconnect resent."""
|
|
452
|
+
if after is not None:
|
|
453
|
+
return int_cursor(after) or 0
|
|
454
|
+
return int_cursor(last_event_id, name="Last-Event-ID") or 0
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
def next_interval(current: float) -> float:
|
|
458
|
+
"""Widen a quiet stream's poll interval, up to the ceiling."""
|
|
459
|
+
return min(current * FOLLOW_BACKOFF, FOLLOW_MAX_INTERVAL_SECONDS)
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def _claim(user_id: UUID) -> str:
|
|
463
|
+
"""Name the principal a stream is opened by, refusing one that already holds the cap.
|
|
464
|
+
|
|
465
|
+
Log tails and event streams draw on one budget: what a principal costs this instance is
|
|
466
|
+
the number of poll loops it holds open, not what they carry.
|
|
467
|
+
"""
|
|
468
|
+
watcher = str(user_id)
|
|
469
|
+
if OPEN_TAILS.get(watcher, 0) >= MAX_TAILS_PER_PRINCIPAL:
|
|
470
|
+
raise HTTPException(
|
|
471
|
+
status_code=status.HTTP_429_TOO_MANY_REQUESTS,
|
|
472
|
+
detail=f"you already have {MAX_TAILS_PER_PRINCIPAL} streams open on this server",
|
|
473
|
+
headers={"retry-after": "5"},
|
|
474
|
+
)
|
|
475
|
+
return watcher
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
@contextmanager
|
|
479
|
+
def _held(watcher: str) -> Generator[None]:
|
|
480
|
+
"""Count one open stream against a principal's budget for as long as it runs."""
|
|
481
|
+
OPEN_TAILS[watcher] = OPEN_TAILS.get(watcher, 0) + 1
|
|
482
|
+
try:
|
|
483
|
+
yield
|
|
484
|
+
finally:
|
|
485
|
+
remaining = OPEN_TAILS.get(watcher, 1) - 1
|
|
486
|
+
if remaining > 0:
|
|
487
|
+
OPEN_TAILS[watcher] = remaining
|
|
488
|
+
else:
|
|
489
|
+
OPEN_TAILS.pop(watcher, None)
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
async def _tail_read(
|
|
493
|
+
sessions: async_sessionmaker[AsyncSession],
|
|
494
|
+
run_id: UUID,
|
|
495
|
+
cursor: int,
|
|
496
|
+
step: str | None,
|
|
497
|
+
) -> tuple[Page[LogEntryOut], bool]:
|
|
498
|
+
"""One poll's reads, whole or not at all.
|
|
499
|
+
|
|
500
|
+
A watcher that disconnects cancels its generator wherever it happens to be, and a
|
|
501
|
+
driver await torn down mid-read leaves the pooled connection broken for whoever draws
|
|
502
|
+
it next. Shielded by the caller, the read runs to its own end and closes its session,
|
|
503
|
+
so a cancellation only ever lands between polls.
|
|
504
|
+
"""
|
|
505
|
+
async with session_scope(sessions) as session:
|
|
506
|
+
page = await _log_page(session, run_id, cursor, DEFAULT_LOG_PAGE, step)
|
|
507
|
+
run = await session.get(Run, run_id)
|
|
508
|
+
return page, run is not None and run.status in TERMINAL_RUN_STATUSES
|
|
509
|
+
|
|
510
|
+
|
|
511
|
+
async def _tail(
|
|
512
|
+
sessions: async_sessionmaker[AsyncSession],
|
|
513
|
+
run_id: UUID,
|
|
514
|
+
after: int,
|
|
515
|
+
step: str | None,
|
|
516
|
+
watcher: str,
|
|
517
|
+
) -> AsyncGenerator[str]:
|
|
518
|
+
"""Stream a run's log entries as server-sent events until the run settles.
|
|
519
|
+
|
|
520
|
+
``end`` says the run settled and there is nothing more to read. At the wall-clock limit
|
|
521
|
+
the stream says ``expired`` instead, which a client reopens from the last id it saw.
|
|
522
|
+
"""
|
|
523
|
+
with _held(watcher):
|
|
524
|
+
cursor = after
|
|
525
|
+
waited = 0.0
|
|
526
|
+
interval = FOLLOW_INTERVAL_SECONDS
|
|
527
|
+
while waited < FOLLOW_MAX_SECONDS:
|
|
528
|
+
page, settled = await asyncio.shield(_tail_read(sessions, run_id, cursor, step))
|
|
529
|
+
for entry in page.items:
|
|
530
|
+
yield f"id: {entry.id}\nevent: log\ndata: {entry.model_dump_json()}\n\n"
|
|
531
|
+
cursor = entry.id
|
|
532
|
+
if page.next is not None:
|
|
533
|
+
interval = FOLLOW_INTERVAL_SECONDS
|
|
534
|
+
continue
|
|
535
|
+
if settled:
|
|
536
|
+
yield "event: end\ndata: {}\n\n"
|
|
537
|
+
return
|
|
538
|
+
await asyncio.sleep(interval)
|
|
539
|
+
waited += interval
|
|
540
|
+
interval = next_interval(interval)
|
|
541
|
+
yield "event: expired\ndata: {}\n\n"
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
@router.get(
|
|
545
|
+
"/runs/{run_id}/$events",
|
|
546
|
+
operation_id="getRunEvents",
|
|
547
|
+
summary="Stream a run's story",
|
|
548
|
+
response_model=None,
|
|
549
|
+
responses={200: {"content": {"text/event-stream": {}}, "description": "The run's story, as SSE."}},
|
|
550
|
+
)
|
|
551
|
+
async def read_events(
|
|
552
|
+
run_id: UUID,
|
|
553
|
+
request: Request,
|
|
554
|
+
session: SessionDep,
|
|
555
|
+
principal: PrincipalDep,
|
|
556
|
+
after: AfterParam = None,
|
|
557
|
+
last_event_id: LAST_EVENT_ID = None,
|
|
558
|
+
) -> StreamingResponse:
|
|
559
|
+
"""Stream one run's story: its attempts as they move, its log entries, and how it ended.
|
|
560
|
+
|
|
561
|
+
Each cycle reads the run's whole attempt grid, which is the read a polling client caused
|
|
562
|
+
once per poll; it is made once per watcher here and goes out as the states that changed.
|
|
563
|
+
"""
|
|
564
|
+
await _run_row(session, run_id)
|
|
565
|
+
watcher = _claim(principal.user_id)
|
|
566
|
+
return StreamingResponse(
|
|
567
|
+
_story(get_sessions(request), run_id, _resume_from(after, last_event_id), watcher),
|
|
568
|
+
media_type="text/event-stream",
|
|
569
|
+
headers={"cache-control": "no-cache", "x-accel-buffering": "no"},
|
|
570
|
+
)
|
|
571
|
+
|
|
572
|
+
|
|
573
|
+
#: What an attempt looks like from outside: a change to any of it is news to a watcher.
|
|
574
|
+
type Fingerprint = tuple[AttemptStatus, int, datetime | None, str | None, float | None]
|
|
575
|
+
|
|
576
|
+
|
|
577
|
+
def _fingerprint(attempt: StepAttempt) -> Fingerprint:
|
|
578
|
+
"""State the observable part of an attempt."""
|
|
579
|
+
return (attempt.status, attempt.attempt, attempt.finished_at, attempt.waiting_message, attempt.waiting_progress)
|
|
580
|
+
|
|
581
|
+
|
|
582
|
+
#: What a run looks like from outside: a change to any of it is news to a watcher.
|
|
583
|
+
type RunFingerprint = tuple[RunStatus, datetime | None, datetime | None, str | None]
|
|
584
|
+
|
|
585
|
+
|
|
586
|
+
def _run_fingerprint(run: RunOut) -> RunFingerprint:
|
|
587
|
+
"""State the observable part of a run."""
|
|
588
|
+
return (run.status, run.started_at, run.finished_at, run.error)
|
|
589
|
+
|
|
590
|
+
|
|
591
|
+
def _moment(when: datetime | None) -> datetime:
|
|
592
|
+
"""Read a timestamp for ordering, treating a missing one as the beginning of time."""
|
|
593
|
+
return when if when is not None else datetime.min.replace(tzinfo=UTC)
|
|
594
|
+
|
|
595
|
+
|
|
596
|
+
async def _attempt_rows(session: AsyncSession, run_id: UUID) -> list[StepAttempt]:
|
|
597
|
+
"""Read every attempt of a run by id, which is the order they were created in."""
|
|
598
|
+
rows = await session.execute(sa.select(StepAttempt).where(StepAttempt.run_id == run_id).order_by(StepAttempt.id))
|
|
599
|
+
return list(rows.scalars())
|
|
600
|
+
|
|
601
|
+
|
|
602
|
+
def _render_event(
|
|
603
|
+
attempt: StepAttempt,
|
|
604
|
+
spilled: dict[UUID, tuple[str, int | None]],
|
|
605
|
+
labels: dict[UUID, tuple[str, int]],
|
|
606
|
+
) -> AttemptEvent:
|
|
607
|
+
"""Render one attempt for the event stream, naming the fan-out element it ran for.
|
|
608
|
+
|
|
609
|
+
The element's label and its index both go out: the label is what a reader reads, and the
|
|
610
|
+
index is the fan-out order a list of the elements is put back into.
|
|
611
|
+
"""
|
|
612
|
+
uri, size = spilled.get(attempt.id, (None, None))
|
|
613
|
+
named = labels.get(attempt.run_item_id) if attempt.run_item_id is not None else None
|
|
614
|
+
item, index = named if named is not None else (None, None)
|
|
615
|
+
return AttemptEvent.model_validate(attempt, from_attributes=True).model_copy(
|
|
616
|
+
update={"output_uri": uri, "output_bytes": size, "item": item, "item_index": index}
|
|
617
|
+
)
|
|
618
|
+
|
|
619
|
+
|
|
620
|
+
async def _story_read(
|
|
621
|
+
sessions: async_sessionmaker[AsyncSession],
|
|
622
|
+
run_id: UUID,
|
|
623
|
+
cursor: int,
|
|
624
|
+
want_labels: bool | None,
|
|
625
|
+
) -> tuple[
|
|
626
|
+
list[StepAttempt],
|
|
627
|
+
dict[UUID, tuple[str, int | None]],
|
|
628
|
+
Page[LogEntryOut],
|
|
629
|
+
RunOut | None,
|
|
630
|
+
dict[UUID, tuple[str, int]] | None,
|
|
631
|
+
]:
|
|
632
|
+
"""One story poll's reads, whole or not at all; see ``_tail_read`` for why.
|
|
633
|
+
|
|
634
|
+
The run is rendered on every poll, not only once it has settled: a watcher learns that a
|
|
635
|
+
run started the same way it learns that one ended.
|
|
636
|
+
"""
|
|
637
|
+
async with session_scope(sessions) as session:
|
|
638
|
+
read_labels = None
|
|
639
|
+
if want_labels:
|
|
640
|
+
read_labels = {
|
|
641
|
+
item.id: (item.item_key.strip() or str(item.item_index), item.item_index)
|
|
642
|
+
for item in await _items(session, run_id)
|
|
643
|
+
}
|
|
644
|
+
attempts = await _attempt_rows(session, run_id)
|
|
645
|
+
spilled = await _spilled(session, run_id)
|
|
646
|
+
page = await _log_page(session, run_id, cursor, DEFAULT_LOG_PAGE, None)
|
|
647
|
+
found = (
|
|
648
|
+
await session.execute(
|
|
649
|
+
sa.select(Run, Pipeline, PipelineVersion)
|
|
650
|
+
.join(Pipeline, Pipeline.id == Run.pipeline_id)
|
|
651
|
+
.join(PipelineVersion, PipelineVersion.id == Run.pipeline_version_id)
|
|
652
|
+
.where(Run.id == run_id)
|
|
653
|
+
)
|
|
654
|
+
).all()
|
|
655
|
+
rendered = next((_render_run(run, pipeline, version) for run, pipeline, version in found), None)
|
|
656
|
+
return attempts, spilled, page, rendered, read_labels
|
|
657
|
+
|
|
658
|
+
|
|
659
|
+
def _in_order(transitions: Sequence[tuple[datetime, str]], entries: Sequence[tuple[datetime, str]]) -> list[str]:
|
|
660
|
+
"""Merge one cycle's attempt transitions into its log entries, leaving the entries in id order.
|
|
661
|
+
|
|
662
|
+
LOG ENTRIES GO OUT IN ASCENDING ID, ALWAYS. A client resumes from the highest id it has
|
|
663
|
+
been sent, so an entry delivered after a higher one is dropped and never asked for again.
|
|
664
|
+
Lines are buffered per attempt and written on a flush and again when that attempt settles,
|
|
665
|
+
so two attempts running at once commit theirs out of timestamp order: sorting a cycle by
|
|
666
|
+
timestamp is exactly what loses them.
|
|
667
|
+
|
|
668
|
+
A transition is placed before the first entry written no earlier than it, which keeps a
|
|
669
|
+
state change ahead of the lines it caused.
|
|
670
|
+
"""
|
|
671
|
+
ordered = sorted(transitions, key=lambda one: one[0])
|
|
672
|
+
merged: list[str] = []
|
|
673
|
+
placed = 0
|
|
674
|
+
for when, payload in entries:
|
|
675
|
+
while placed < len(ordered) and ordered[placed][0] <= when:
|
|
676
|
+
merged.append(ordered[placed][1])
|
|
677
|
+
placed += 1
|
|
678
|
+
merged.append(payload)
|
|
679
|
+
merged.extend(payload for _, payload in ordered[placed:])
|
|
680
|
+
return merged
|
|
681
|
+
|
|
682
|
+
|
|
683
|
+
async def _story(
|
|
684
|
+
sessions: async_sessionmaker[AsyncSession],
|
|
685
|
+
run_id: UUID,
|
|
686
|
+
after: int,
|
|
687
|
+
watcher: str,
|
|
688
|
+
) -> AsyncGenerator[str]:
|
|
689
|
+
"""Stream a run's own state and its attempt transitions and log entries, in order.
|
|
690
|
+
|
|
691
|
+
Every attempt is reported once on connect, so a watcher that joined late reads the same
|
|
692
|
+
story as one that was there from the start, without the states it missed in between.
|
|
693
|
+
|
|
694
|
+
The run goes out on connect and again whenever its own state changes, ahead of that
|
|
695
|
+
cycle's attempt transitions and log entries, so a state change leads the lines it caused.
|
|
696
|
+
Once the run settles it goes out on the settled path below instead, after the last drain.
|
|
697
|
+
|
|
698
|
+
Only a log frame carries an ``id``. The attempt and run frames are replayed on connect by
|
|
699
|
+
that same rule, so an id on one would make a reconnect resume past a state it is about to
|
|
700
|
+
be told again; the log cursor is the only position a reconnect can restore, and a client
|
|
701
|
+
dedupes the replayed attempts by their own ids.
|
|
702
|
+
|
|
703
|
+
``end`` says the run settled and follows the ``run`` frame. At the wall-clock limit the
|
|
704
|
+
stream says ``expired`` instead, which a client reopens from the last id it saw.
|
|
705
|
+
"""
|
|
706
|
+
with _held(watcher):
|
|
707
|
+
cursor = after
|
|
708
|
+
reported: dict[UUID, Fingerprint] = {}
|
|
709
|
+
reported_run: RunFingerprint | None = None
|
|
710
|
+
labels: dict[UUID, tuple[str, int]] = {}
|
|
711
|
+
loaded = False
|
|
712
|
+
drained = False
|
|
713
|
+
waited = 0.0
|
|
714
|
+
interval = FOLLOW_INTERVAL_SECONDS
|
|
715
|
+
while waited < FOLLOW_MAX_SECONDS:
|
|
716
|
+
# Shielded for the same reason _tail_read is: a disconnect lands between polls,
|
|
717
|
+
# never inside a driver await holding the pooled connection.
|
|
718
|
+
attempts, spilled, page, rendered, read_labels = await asyncio.shield(
|
|
719
|
+
_story_read(sessions, run_id, cursor, None if loaded else True)
|
|
720
|
+
)
|
|
721
|
+
if read_labels is not None:
|
|
722
|
+
labels = read_labels
|
|
723
|
+
loaded = True
|
|
724
|
+
settled = rendered is not None and rendered.status in TERMINAL_RUN_STATUSES
|
|
725
|
+
if rendered is not None and not settled and _run_fingerprint(rendered) != reported_run:
|
|
726
|
+
reported_run = _run_fingerprint(rendered)
|
|
727
|
+
yield f"event: run\ndata: {rendered.model_dump_json()}\n\n"
|
|
728
|
+
moved = [row for row in attempts if reported.get(row.id) != _fingerprint(row)]
|
|
729
|
+
reported.update({row.id: _fingerprint(row) for row in moved})
|
|
730
|
+
transitions = [
|
|
731
|
+
(
|
|
732
|
+
_moment(row.finished_at or row.started_at),
|
|
733
|
+
f"event: attempt\ndata: {_render_event(row, spilled, labels).model_dump_json()}\n\n",
|
|
734
|
+
)
|
|
735
|
+
for row in moved
|
|
736
|
+
]
|
|
737
|
+
entries = [
|
|
738
|
+
(
|
|
739
|
+
_moment(entry.created_at),
|
|
740
|
+
f"id: {entry.id}\nevent: log\ndata: {entry.model_dump_json()}\n\n",
|
|
741
|
+
)
|
|
742
|
+
for entry in page.items
|
|
743
|
+
]
|
|
744
|
+
for payload in _in_order(transitions, entries):
|
|
745
|
+
yield payload
|
|
746
|
+
if page.items:
|
|
747
|
+
cursor = page.items[-1].id
|
|
748
|
+
if page.next is not None:
|
|
749
|
+
interval = FOLLOW_INTERVAL_SECONDS
|
|
750
|
+
continue
|
|
751
|
+
if settled and rendered is not None:
|
|
752
|
+
# One more pass after the run settles, for the lines a block wrote on its way out.
|
|
753
|
+
if drained:
|
|
754
|
+
yield f"event: run\ndata: {rendered.model_dump_json()}\n\n"
|
|
755
|
+
yield "event: end\ndata: {}\n\n"
|
|
756
|
+
return
|
|
757
|
+
drained = True
|
|
758
|
+
continue
|
|
759
|
+
await asyncio.sleep(interval)
|
|
760
|
+
waited += interval
|
|
761
|
+
interval = next_interval(interval)
|
|
762
|
+
yield "event: expired\ndata: {}\n\n"
|
|
763
|
+
|
|
764
|
+
|
|
765
|
+
@router.get(
|
|
766
|
+
"/runs/{run_id}/$report",
|
|
767
|
+
operation_id="getRunReport",
|
|
768
|
+
summary="Summarise a run",
|
|
769
|
+
response_model=RunReport,
|
|
770
|
+
)
|
|
771
|
+
async def report(run_id: UUID, session: SessionDep, principal: PrincipalDep) -> RunReport:
|
|
772
|
+
"""Summarise a run: what each step amounted to, and how long the whole thing took."""
|
|
773
|
+
run = await _run_row(session, run_id)
|
|
774
|
+
pipeline, version, definition = await _context(session, run)
|
|
775
|
+
attempts = await _attempts(session, run_id, version.step_order)
|
|
776
|
+
items = await _items(session, run_id)
|
|
777
|
+
states = build_step_states(definition, attempts)
|
|
778
|
+
warned = await _warnings(session, run_id)
|
|
779
|
+
steps = [
|
|
780
|
+
StepReport(
|
|
781
|
+
step=name,
|
|
782
|
+
block=definition.steps[name].block,
|
|
783
|
+
outcome=states[name].outcome.value,
|
|
784
|
+
attempts=sum(1 for attempt in attempts if attempt.step_name == name),
|
|
785
|
+
depends_on=list(definition.steps[name].depends_on),
|
|
786
|
+
warnings=warned.get(name, 0),
|
|
787
|
+
duration_ms=_duration_ms(
|
|
788
|
+
min((a.started_at for a in attempts if a.step_name == name and a.started_at), default=None),
|
|
789
|
+
max((a.finished_at for a in attempts if a.step_name == name and a.finished_at), default=None),
|
|
790
|
+
),
|
|
791
|
+
error=next(
|
|
792
|
+
(a.error for a in attempts if a.step_name == name and a.status is AttemptStatus.FAILED and a.error),
|
|
793
|
+
None,
|
|
794
|
+
),
|
|
795
|
+
)
|
|
796
|
+
for name in _human_order(definition, [attempt.step_name for attempt in attempts])
|
|
797
|
+
]
|
|
798
|
+
return RunReport(
|
|
799
|
+
run_id=run.id,
|
|
800
|
+
pipeline=pipeline.code,
|
|
801
|
+
pipeline_version=version.version,
|
|
802
|
+
status=run.status,
|
|
803
|
+
triggered_by=run.triggered_by_label,
|
|
804
|
+
started_at=run.started_at,
|
|
805
|
+
finished_at=run.finished_at,
|
|
806
|
+
duration_ms=_duration_ms(run.started_at, run.finished_at),
|
|
807
|
+
steps=steps,
|
|
808
|
+
items_total=len(items),
|
|
809
|
+
items_failed=sum(1 for item in items if item.status is RunItemStatus.FAILED),
|
|
810
|
+
error=run.error,
|
|
811
|
+
)
|
|
812
|
+
|
|
813
|
+
|
|
814
|
+
def _duration_ms(started: datetime | None, finished: datetime | None) -> int | None:
|
|
815
|
+
"""Report how long something took, or nothing when it has not finished."""
|
|
816
|
+
if started is None or finished is None:
|
|
817
|
+
return None
|
|
818
|
+
return round((finished - started).total_seconds() * 1000)
|