agentdeck-sdk 3.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentdeck/README.md +50 -0
- agentdeck/__init__.py +51 -0
- agentdeck/adapters/__init__.py +5 -0
- agentdeck/adapters/control/__init__.py +1 -0
- agentdeck/adapters/control/memory/__init__.py +5 -0
- agentdeck/adapters/control/memory/port.py +25 -0
- agentdeck/adapters/control/sqlite/__init__.py +5 -0
- agentdeck/adapters/control/sqlite/port.py +111 -0
- agentdeck/adapters/engines/__init__.py +1 -0
- agentdeck/adapters/engines/langgraph/__init__.py +8 -0
- agentdeck/adapters/engines/langgraph/checkpointer.py +205 -0
- agentdeck/adapters/engines/langgraph/engine.py +410 -0
- agentdeck/adapters/engines/openai_agents/__init__.py +9 -0
- agentdeck/adapters/engines/openai_agents/engine.py +321 -0
- agentdeck/adapters/engines/openai_agents/reconcile.py +178 -0
- agentdeck/adapters/engines/openai_agents/runconfig.py +118 -0
- agentdeck/adapters/engines/openai_agents/sessions.py +100 -0
- agentdeck/adapters/engines/openai_agents/translate.py +124 -0
- agentdeck/adapters/engines/stub/__init__.py +5 -0
- agentdeck/adapters/engines/stub/engine.py +100 -0
- agentdeck/adapters/stores/__init__.py +1 -0
- agentdeck/adapters/stores/memory/__init__.py +5 -0
- agentdeck/adapters/stores/memory/store.py +148 -0
- agentdeck/adapters/stores/postgres/__init__.py +5 -0
- agentdeck/adapters/stores/postgres/store.py +374 -0
- agentdeck/adapters/stores/redis/__init__.py +5 -0
- agentdeck/adapters/stores/redis/store.py +359 -0
- agentdeck/adapters/stores/sqlite/__init__.py +5 -0
- agentdeck/adapters/stores/sqlite/store.py +338 -0
- agentdeck/adapters/telemetry/__init__.py +1 -0
- agentdeck/adapters/telemetry/langfuse/__init__.py +18 -0
- agentdeck/adapters/telemetry/langfuse/client.py +180 -0
- agentdeck/adapters/telemetry/langfuse/sink.py +366 -0
- agentdeck/adapters/telemetry/langfuse/trace.py +88 -0
- agentdeck/adapters/tools/__init__.py +1 -0
- agentdeck/adapters/tools/mcp/__init__.py +18 -0
- agentdeck/adapters/tools/mcp/lifecycle.py +178 -0
- agentdeck/adapters/tools/mcp/source.py +47 -0
- agentdeck/adapters/tools/mcp/transport.py +234 -0
- agentdeck/adapters/tools/mcp/wiring.py +63 -0
- agentdeck/authoring/__init__.py +22 -0
- agentdeck/authoring/agent.py +169 -0
- agentdeck/authoring/compile.py +250 -0
- agentdeck/authoring/graphs.py +145 -0
- agentdeck/authoring/hooks.py +117 -0
- agentdeck/authoring/injection.py +233 -0
- agentdeck/authoring/instructions.py +80 -0
- agentdeck/authoring/interrupts.py +37 -0
- agentdeck/authoring/nodes.py +140 -0
- agentdeck/authoring/runners/__init__.py +6 -0
- agentdeck/authoring/runners/agent.py +171 -0
- agentdeck/authoring/runners/workflow.py +99 -0
- agentdeck/authoring/skills.py +56 -0
- agentdeck/authoring/state.py +44 -0
- agentdeck/authoring/timers.py +44 -0
- agentdeck/authoring/tools.py +147 -0
- agentdeck/authoring/web_search.py +43 -0
- agentdeck/authoring/workflow.py +266 -0
- agentdeck/cli.py +56 -0
- agentdeck/composition.py +213 -0
- agentdeck/core/__init__.py +110 -0
- agentdeck/core/base.py +47 -0
- agentdeck/core/content.py +166 -0
- agentdeck/core/context.py +136 -0
- agentdeck/core/control.py +150 -0
- agentdeck/core/events.py +468 -0
- agentdeck/core/invocable.py +38 -0
- agentdeck/core/ports/__init__.py +26 -0
- agentdeck/core/ports/control.py +35 -0
- agentdeck/core/ports/engine.py +66 -0
- agentdeck/core/ports/sink.py +53 -0
- agentdeck/core/ports/store.py +170 -0
- agentdeck/core/ports/tools.py +57 -0
- agentdeck/core/reporting.py +75 -0
- agentdeck/core/status.py +75 -0
- agentdeck/deck.py +894 -0
- agentdeck/errors.py +67 -0
- agentdeck/mcp.py +82 -0
- agentdeck/observers.py +111 -0
- agentdeck/py.typed +0 -0
- agentdeck/runtime/__init__.py +1 -0
- agentdeck/runtime/capture.py +32 -0
- agentdeck/runtime/config.default.yaml +38 -0
- agentdeck/runtime/discovery.py +175 -0
- agentdeck/runtime/dispatch.py +440 -0
- agentdeck/runtime/registry.py +173 -0
- agentdeck/runtime/service.py +671 -0
- agentdeck/runtime/settings.py +568 -0
- agentdeck/serve.py +330 -0
- agentdeck/skills/__init__.py +114 -0
- agentdeck/skills/bundle.py +65 -0
- agentdeck/surfaces/__init__.py +4 -0
- agentdeck/surfaces/cli/__init__.py +7 -0
- agentdeck/surfaces/cli/chat.py +88 -0
- agentdeck/surfaces/serve/__init__.py +7 -0
- agentdeck/surfaces/serve/app.py +71 -0
- agentdeck/surfaces/serve/compat.py +212 -0
- agentdeck/surfaces/serve/workflows.py +69 -0
- agentdeck/testing.py +364 -0
- agentdeck_sdk-3.1.0.dist-info/METADATA +222 -0
- agentdeck_sdk-3.1.0.dist-info/RECORD +104 -0
- agentdeck_sdk-3.1.0.dist-info/WHEEL +4 -0
- agentdeck_sdk-3.1.0.dist-info/entry_points.txt +3 -0
- agentdeck_sdk-3.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,374 @@
|
|
|
1
|
+
"""The event log in Postgres: the same contract as ``adapters.stores.sqlite``, for workers
|
|
2
|
+
that share a database rather than a filesystem.
|
|
3
|
+
|
|
4
|
+
The SQLite store's own docstring points here for a networked deployment — WAL needs shared
|
|
5
|
+
memory across processes, so a log on NFS is unreliable, and that is exactly the shape a
|
|
6
|
+
multi-worker server has. Nothing else changes: append-only, one row per event, ``seq``
|
|
7
|
+
scoped to one run within one log, and status still derived by folding events rather than
|
|
8
|
+
stored (ADR-D5: the log is the sole source of truth).
|
|
9
|
+
|
|
10
|
+
Everything this store owns lives in its **own schema** (``agentdeck_events`` by default),
|
|
11
|
+
so a database that also holds the langgraph checkpointer's tables keeps the platform record
|
|
12
|
+
and the engine's private execution state apart — the operational separation ADR-D5 asks
|
|
13
|
+
for, expressed as the one thing Postgres can enforce.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import asyncio
|
|
19
|
+
import hashlib
|
|
20
|
+
from typing import TYPE_CHECKING, Any
|
|
21
|
+
|
|
22
|
+
import psycopg
|
|
23
|
+
from psycopg import sql
|
|
24
|
+
|
|
25
|
+
from agentdeck.core.events import Event
|
|
26
|
+
from agentdeck.core.ports import EventStorePort, RunSummary, SessionClaim
|
|
27
|
+
from agentdeck.core.status import LIFECYCLE_KINDS, TERMINAL_STATUSES, can_resume, status_of
|
|
28
|
+
from agentdeck.errors import StoreError
|
|
29
|
+
|
|
30
|
+
if TYPE_CHECKING:
|
|
31
|
+
from collections.abc import Awaitable, Callable, Sequence
|
|
32
|
+
from datetime import timedelta
|
|
33
|
+
|
|
34
|
+
from agentdeck.core.context import RunContext
|
|
35
|
+
from agentdeck.core.events import KnownPayload, RunResumed, RunStarted
|
|
36
|
+
from agentdeck.core.status import RunStatus
|
|
37
|
+
|
|
38
|
+
type Connection = psycopg.AsyncConnection[tuple[Any, ...]]
|
|
39
|
+
|
|
40
|
+
# Postgres's own wall clock, so N workers sharing one database compare one clock rather than N
|
|
41
|
+
# (ADR-D11 §4). ``clock_timestamp()`` and not ``now()``: the latter is the transaction's start
|
|
42
|
+
# time, so every event in a batch would carry the timestamp of the statement that opened it.
|
|
43
|
+
_SELECT_NOW = "SELECT clock_timestamp()"
|
|
44
|
+
|
|
45
|
+
_DEFAULT_SCHEMA = "agentdeck_events"
|
|
46
|
+
|
|
47
|
+
_SORTED_LIFECYCLE_KINDS = sorted(LIFECYCLE_KINDS)
|
|
48
|
+
|
|
49
|
+
# The SQLite store's busy timeout, in the one place Postgres spells the same idea: long
|
|
50
|
+
# enough to wait out a peer's claim (milliseconds of one transaction), short enough that a
|
|
51
|
+
# wedged holder surfaces as an error instead of hanging a request forever.
|
|
52
|
+
_LOCK_TIMEOUT_MS = 5_000
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _advisory_key(name: str) -> int:
|
|
56
|
+
"""A stable signed 64-bit lock number for ``name``.
|
|
57
|
+
|
|
58
|
+
Hashed here rather than by Postgres's own ``hashtext`` because that function is an
|
|
59
|
+
undocumented internal whose value is not promised across major versions; this one is
|
|
60
|
+
the same number in every process and every server. Two different names colliding costs
|
|
61
|
+
nothing but serializing two unrelated claims, which is why a 64-bit digest is plenty.
|
|
62
|
+
"""
|
|
63
|
+
return int.from_bytes(hashlib.blake2b(name.encode(), digest_size=8).digest(), "big", signed=True)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class PostgresEventStore(EventStorePort):
|
|
67
|
+
"""Append-only rows in one Postgres schema, reachable by every worker at once.
|
|
68
|
+
|
|
69
|
+
One connection per store instance, serialized by a lock — ``psycopg``'s async
|
|
70
|
+
connection is not safe to drive from two coroutines at once, and a single writer per
|
|
71
|
+
log is the shape the Runtime already assumes. Consequence to know when operating it:
|
|
72
|
+
anything waiting on a peer's transaction holds this process's only connection, so every
|
|
73
|
+
other call queues behind it. On the write and claim paths that wait is bounded by
|
|
74
|
+
``lock_timeout``; the one place it is not is first-use schema setup, whose lock is
|
|
75
|
+
session-scoped and taken before any transaction exists, so a peer wedged midway through
|
|
76
|
+
creating the schema blocks this instance until it finishes or its connection drops.
|
|
77
|
+
|
|
78
|
+
Every write — both conditional appends and the plain one — takes a per-log advisory lock
|
|
79
|
+
before reading or inserting anything, so two servers agree through Postgres and not
|
|
80
|
+
through Python. ``READ COMMITTED`` is load-bearing for the claims, not incidental: the
|
|
81
|
+
loser has to see the winner's committed rows after the lock is handed over, and a
|
|
82
|
+
snapshot taken at the transaction's first statement — what ``REPEATABLE READ`` would
|
|
83
|
+
give it — was taken before the winner committed. Since that first statement is the
|
|
84
|
+
``lock_timeout`` setting rather than the lock itself, the pin is the whole defence and
|
|
85
|
+
not a second layer of one; the connection pins it so a server configured otherwise
|
|
86
|
+
cannot quietly break the claim.
|
|
87
|
+
|
|
88
|
+
A ``psycopg`` exception never crosses the port: everything funnels through ``_run``
|
|
89
|
+
and reaches the caller as ``StoreError``.
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
def __init__(self, dsn: str, *, schema: str = _DEFAULT_SCHEMA) -> None:
|
|
93
|
+
self._dsn = dsn
|
|
94
|
+
self._schema = schema
|
|
95
|
+
self._conn: Connection | None = None
|
|
96
|
+
self._lock = asyncio.Lock()
|
|
97
|
+
self._setup_key = _advisory_key(f"agentdeck:events:setup:{schema}")
|
|
98
|
+
table = sql.Identifier(schema, "events")
|
|
99
|
+
# Composed once: the schema name is an identifier, so it is quoted by psycopg
|
|
100
|
+
# rather than interpolated — a schema is caller-supplied configuration.
|
|
101
|
+
self._ddl = (
|
|
102
|
+
sql.SQL("CREATE SCHEMA IF NOT EXISTS {schema}").format(schema=sql.Identifier(schema)),
|
|
103
|
+
sql.SQL(
|
|
104
|
+
"CREATE TABLE IF NOT EXISTS {table} ("
|
|
105
|
+
" id BIGSERIAL PRIMARY KEY,"
|
|
106
|
+
" namespace TEXT NOT NULL,"
|
|
107
|
+
" log_key TEXT NOT NULL,"
|
|
108
|
+
" run_id TEXT NOT NULL,"
|
|
109
|
+
" seq INTEGER NOT NULL,"
|
|
110
|
+
" data JSONB NOT NULL)"
|
|
111
|
+
).format(table=table),
|
|
112
|
+
sql.SQL("CREATE INDEX IF NOT EXISTS events_by_log ON {table} (namespace, log_key, id)").format(table=table),
|
|
113
|
+
# UNIQUE is the guard, not just the index: one seq per run is the promise consumers
|
|
114
|
+
# refetch a gap with, and a duplicate is the one corruption a gap check cannot see.
|
|
115
|
+
# ponytail: only schemas this build creates get the constraint — one from an earlier
|
|
116
|
+
# beta keeps its non-unique index, and v2 has no migration story yet.
|
|
117
|
+
sql.SQL(
|
|
118
|
+
"CREATE UNIQUE INDEX IF NOT EXISTS events_by_run ON {table} (namespace, log_key, run_id, seq)"
|
|
119
|
+
).format(table=table),
|
|
120
|
+
)
|
|
121
|
+
self._insert = sql.SQL(
|
|
122
|
+
"INSERT INTO {table} (namespace, log_key, run_id, seq, data) VALUES (%s, %s, %s, %s, %s::jsonb)"
|
|
123
|
+
).format(table=table)
|
|
124
|
+
self._select_log = sql.SQL(
|
|
125
|
+
"SELECT data FROM {table} WHERE namespace = %s AND log_key = %s ORDER BY id ASC LIMIT %s OFFSET %s"
|
|
126
|
+
).format(table=table)
|
|
127
|
+
self._select_run = sql.SQL(
|
|
128
|
+
"SELECT data FROM {table} WHERE namespace = %s AND log_key = %s AND run_id = %s AND seq >= %s ORDER BY id ASC"
|
|
129
|
+
).format(table=table)
|
|
130
|
+
self._select_last_seq = sql.SQL(
|
|
131
|
+
"SELECT MAX(seq) FROM {table} WHERE namespace = %s AND log_key = %s AND run_id = %s"
|
|
132
|
+
).format(table=table)
|
|
133
|
+
# DISTINCT ON is Postgres's version of the SQLite store's MAX(id) group-by: the
|
|
134
|
+
# newest lifecycle row of each run, one row per run, one statement.
|
|
135
|
+
self._select_last_lifecycle = sql.SQL(
|
|
136
|
+
"SELECT DISTINCT ON (log_key, run_id) log_key, run_id, data FROM {table} "
|
|
137
|
+
"WHERE namespace = %s AND data->>'kind' = ANY(%s) ORDER BY log_key, run_id, id DESC"
|
|
138
|
+
).format(table=table)
|
|
139
|
+
self._select_run_lifecycle = sql.SQL(
|
|
140
|
+
"SELECT data FROM {table} WHERE namespace = %s AND log_key = %s AND run_id = %s "
|
|
141
|
+
"AND data->>'kind' = ANY(%s) ORDER BY id DESC LIMIT 1"
|
|
142
|
+
).format(table=table)
|
|
143
|
+
self._select_log_lifecycle = sql.SQL(
|
|
144
|
+
"SELECT DISTINCT ON (run_id) run_id, data FROM {table} "
|
|
145
|
+
"WHERE namespace = %s AND log_key = %s AND data->>'kind' = ANY(%s) ORDER BY run_id, id DESC"
|
|
146
|
+
).format(table=table)
|
|
147
|
+
self._select_last_events = sql.SQL(
|
|
148
|
+
"SELECT DISTINCT ON (run_id) run_id, data FROM {table} "
|
|
149
|
+
"WHERE namespace = %s AND log_key = %s AND run_id = ANY(%s) ORDER BY run_id, id DESC"
|
|
150
|
+
).format(table=table)
|
|
151
|
+
|
|
152
|
+
async def _run[T](self, work: Callable[[Connection], Awaitable[T]], op: str) -> T:
|
|
153
|
+
"""Every statement this store runs goes through here — one caller at a time, and no
|
|
154
|
+
``psycopg`` exception escaping the port."""
|
|
155
|
+
async with self._lock:
|
|
156
|
+
try:
|
|
157
|
+
return await work(await self._ready())
|
|
158
|
+
except psycopg.Error as exc:
|
|
159
|
+
# A connection that died stays dead otherwise: drop it so the next call
|
|
160
|
+
# dials again instead of failing forever on a socket nobody is holding.
|
|
161
|
+
if self._conn is not None and self._conn.closed:
|
|
162
|
+
self._conn = None
|
|
163
|
+
raise StoreError(f"event log {op} failed: {exc}") from exc
|
|
164
|
+
|
|
165
|
+
async def _ready(self) -> Connection:
|
|
166
|
+
"""Connect and create the schema on first use — never at construction, so building
|
|
167
|
+
this store is not I/O and a composition root can wire one without a live server."""
|
|
168
|
+
if self._conn is None:
|
|
169
|
+
conn: Connection = await psycopg.AsyncConnection.connect(self._dsn, autocommit=True)
|
|
170
|
+
# Closed rather than abandoned if setup fails: this is the retried path, and a
|
|
171
|
+
# connection nothing holds a reference to holds a server backend regardless.
|
|
172
|
+
try:
|
|
173
|
+
await conn.set_isolation_level(psycopg.IsolationLevel.READ_COMMITTED)
|
|
174
|
+
# Two workers starting together would otherwise race their own CREATEs: the
|
|
175
|
+
# IF NOT EXISTS checks are not atomic against each other, and the loser gets
|
|
176
|
+
# a duplicate-object error rather than the table it asked for.
|
|
177
|
+
await conn.execute("SELECT pg_advisory_lock(%s)", (self._setup_key,))
|
|
178
|
+
try:
|
|
179
|
+
for statement in self._ddl:
|
|
180
|
+
await conn.execute(statement)
|
|
181
|
+
finally:
|
|
182
|
+
await conn.execute("SELECT pg_advisory_unlock(%s)", (self._setup_key,))
|
|
183
|
+
except BaseException:
|
|
184
|
+
await conn.close()
|
|
185
|
+
raise
|
|
186
|
+
self._conn = conn
|
|
187
|
+
return self._conn
|
|
188
|
+
|
|
189
|
+
async def append(self, log_key: str, payloads: Sequence[KnownPayload], ctx: RunContext, origin: str) -> list[Event]:
|
|
190
|
+
"""Writes take the log's lock, for the port's paging guarantee and now for ``seq``.
|
|
191
|
+
|
|
192
|
+
Row order is `BIGSERIAL`, assigned at insert and published at commit, so an
|
|
193
|
+
unlocked append can be given a *later* number than a claim's in-flight insert and
|
|
194
|
+
still commit *first* — the claim's event then appears at an offset a reader has
|
|
195
|
+
already gone past, and one of its neighbours is delivered twice. Serializing writes
|
|
196
|
+
per log is what keeps the log growing only at its end. The same lock is what makes
|
|
197
|
+
reading this run's last ``seq`` and inserting the next one indivisible.
|
|
198
|
+
"""
|
|
199
|
+
if not payloads:
|
|
200
|
+
# No lock and no transaction for a batch with nothing in it, as in the Redis store.
|
|
201
|
+
return []
|
|
202
|
+
|
|
203
|
+
async def _work(conn: Connection) -> list[Event]:
|
|
204
|
+
async with conn.transaction():
|
|
205
|
+
await self._lock_log(conn, ctx.namespace_key, log_key)
|
|
206
|
+
return await self._stamp_and_insert(conn, log_key, list(payloads), ctx, origin)
|
|
207
|
+
|
|
208
|
+
return await self._run(_work, "append")
|
|
209
|
+
|
|
210
|
+
async def read(self, log_key: str, ctx: RunContext, offset: int = 0, limit: int | None = None) -> list[Event]:
|
|
211
|
+
if limit is not None and limit < 0:
|
|
212
|
+
raise ValueError(f"limit must be None or >= 0, got {limit}")
|
|
213
|
+
|
|
214
|
+
async def _work(conn: Connection) -> list[dict[str, Any]]:
|
|
215
|
+
# LIMIT NULL is Postgres for "no limit", which is what the port's None means.
|
|
216
|
+
cursor = await conn.execute(self._select_log, (ctx.namespace_key, log_key, limit, max(offset, 0)))
|
|
217
|
+
return [row[0] for row in await cursor.fetchall()]
|
|
218
|
+
|
|
219
|
+
return [Event.model_validate(data) for data in await self._run(_work, "read")]
|
|
220
|
+
|
|
221
|
+
async def read_run(self, log_key: str, run_id: str, ctx: RunContext, from_seq: int = 0) -> list[Event]:
|
|
222
|
+
async def _work(conn: Connection) -> list[dict[str, Any]]:
|
|
223
|
+
cursor = await conn.execute(self._select_run, (ctx.namespace_key, log_key, run_id, from_seq))
|
|
224
|
+
return [row[0] for row in await cursor.fetchall()]
|
|
225
|
+
|
|
226
|
+
return [Event.model_validate(data) for data in await self._run(_work, "read_run")]
|
|
227
|
+
|
|
228
|
+
async def _stamp_and_insert(
|
|
229
|
+
self, conn: Connection, log_key: str, payloads: list[KnownPayload], ctx: RunContext, origin: str
|
|
230
|
+
) -> list[Event]:
|
|
231
|
+
"""Assign, build, insert — callable only with this log's advisory lock already held.
|
|
232
|
+
|
|
233
|
+
``ts`` is ``clock_timestamp()``, Postgres's own wall clock, so N workers sharing one
|
|
234
|
+
database compare one clock rather than N (ADR-D11 §4). ``clock_timestamp`` rather than
|
|
235
|
+
``now()``, which is the transaction's start time and would hand every event in a batch
|
|
236
|
+
the timestamp of the statement that opened it.
|
|
237
|
+
"""
|
|
238
|
+
cursor = await conn.execute(_SELECT_NOW)
|
|
239
|
+
now = (await cursor.fetchone())[0] # ty: ignore[not-subscriptable] — one-row scalar select
|
|
240
|
+
seq = await self._last_seq(conn, ctx.namespace_key, log_key, ctx.run_id)
|
|
241
|
+
events = []
|
|
242
|
+
for payload in payloads:
|
|
243
|
+
seq += 1
|
|
244
|
+
events.append(
|
|
245
|
+
Event(
|
|
246
|
+
kind=payload.kind,
|
|
247
|
+
seq=seq,
|
|
248
|
+
run_id=ctx.run_id,
|
|
249
|
+
session_id=ctx.session_id,
|
|
250
|
+
namespace=ctx.namespace,
|
|
251
|
+
origin=origin,
|
|
252
|
+
ts=now,
|
|
253
|
+
payload=payload,
|
|
254
|
+
)
|
|
255
|
+
)
|
|
256
|
+
await conn.cursor().executemany(self._insert, [_row(ctx.namespace_key, log_key, event) for event in events])
|
|
257
|
+
return events
|
|
258
|
+
|
|
259
|
+
async def claim_start(
|
|
260
|
+
self, log_key: str, opening: RunStarted, ctx: RunContext, origin: str, stale_after: timedelta
|
|
261
|
+
) -> tuple[SessionClaim, Event | None]:
|
|
262
|
+
"""The port's session claim as one transaction holding this log's advisory lock, so
|
|
263
|
+
only one of two servers can open a run on an idle session.
|
|
264
|
+
|
|
265
|
+
A refused claim is still a clean answer: the loser waits the winner's transaction
|
|
266
|
+
out and then reads the run it opened. Only a lock held past ``lock_timeout`` raises,
|
|
267
|
+
because that is a store nobody can write to rather than a session somebody took.
|
|
268
|
+
"""
|
|
269
|
+
|
|
270
|
+
async def _work(conn: Connection) -> tuple[SessionClaim, Event | None]:
|
|
271
|
+
async with conn.transaction():
|
|
272
|
+
await self._lock_log(conn, ctx.namespace_key, log_key)
|
|
273
|
+
cursor = await conn.execute(_SELECT_NOW)
|
|
274
|
+
stale_before = (await cursor.fetchone())[0] - stale_after # ty: ignore[not-subscriptable]
|
|
275
|
+
overridden: list[Event] = []
|
|
276
|
+
for _run_id, last in await self._open_runs(conn, ctx.namespace_key, log_key):
|
|
277
|
+
if last.ts > stale_before:
|
|
278
|
+
return SessionClaim(held_by=last.run_id), None
|
|
279
|
+
overridden.append(last)
|
|
280
|
+
event = (await self._stamp_and_insert(conn, log_key, [opening], ctx, origin))[0]
|
|
281
|
+
return SessionClaim(overridden=tuple(overridden)), event
|
|
282
|
+
|
|
283
|
+
return await self._run(_work, "claim_start")
|
|
284
|
+
|
|
285
|
+
async def claim_resume(
|
|
286
|
+
self, log_key: str, run_id: str, resumed: RunResumed, ctx: RunContext, origin: str
|
|
287
|
+
) -> Event | None:
|
|
288
|
+
"""The port's conditional append as one transaction holding this log's advisory
|
|
289
|
+
lock: the lock is taken before the read, so a peer cannot resume the same run in
|
|
290
|
+
the gap between this caller's check and its insert.
|
|
291
|
+
|
|
292
|
+
A loser gets its clean ``None`` — it reads the ``RUNNING`` status the winner
|
|
293
|
+
published. Only an unreachable store or a lock held past ``lock_timeout`` raises,
|
|
294
|
+
never a fabricated ``None``.
|
|
295
|
+
"""
|
|
296
|
+
if ctx.run_id != run_id:
|
|
297
|
+
raise ValueError(f"a claim on run {run_id!r} cannot be made in the context of {ctx.run_id!r}")
|
|
298
|
+
|
|
299
|
+
async def _work(conn: Connection) -> Event | None:
|
|
300
|
+
async with conn.transaction():
|
|
301
|
+
await self._lock_log(conn, ctx.namespace_key, log_key)
|
|
302
|
+
last = await self._last_lifecycle_of_run(conn, ctx.namespace_key, log_key, run_id)
|
|
303
|
+
if not can_resume(status_of([last] if last is not None else [])):
|
|
304
|
+
return None
|
|
305
|
+
return (await self._stamp_and_insert(conn, log_key, [resumed], ctx, origin))[0]
|
|
306
|
+
|
|
307
|
+
return await self._run(_work, "claim_resume")
|
|
308
|
+
|
|
309
|
+
async def list_runs(self, ctx: RunContext, status: RunStatus | None = None) -> list[RunSummary]:
|
|
310
|
+
"""Overrides the port's per-run fold: one statement returns each run's *last*
|
|
311
|
+
lifecycle row, so a listing deserializes one event per run instead of all of them."""
|
|
312
|
+
|
|
313
|
+
async def _work(conn: Connection) -> list[tuple[str, str, dict[str, Any]]]:
|
|
314
|
+
cursor = await conn.execute(self._select_last_lifecycle, (ctx.namespace_key, _SORTED_LIFECYCLE_KINDS))
|
|
315
|
+
return [(row[0], row[1], row[2]) for row in await cursor.fetchall()]
|
|
316
|
+
|
|
317
|
+
summaries = [
|
|
318
|
+
RunSummary(log_key=log_key, run_id=run_id, status=status_of([Event.model_validate(data)]))
|
|
319
|
+
for log_key, run_id, data in await self._run(_work, "list_runs")
|
|
320
|
+
]
|
|
321
|
+
return [summary for summary in summaries if status is None or summary.status is status]
|
|
322
|
+
|
|
323
|
+
async def _lock_log(self, conn: Connection, namespace: str, log_key: str) -> None:
|
|
324
|
+
"""Serialize this log's writes, and bound the wait.
|
|
325
|
+
|
|
326
|
+
The lock is per (namespace, log key) and transaction-scoped, so it is released by the
|
|
327
|
+
commit that publishes the write and never outlives a crashed worker. It is taken
|
|
328
|
+
before any read the decision depends on, which is the whole point: a lock acquired
|
|
329
|
+
afterwards would leave the same check-then-write window a plain read has.
|
|
330
|
+
"""
|
|
331
|
+
await conn.execute("SELECT set_config('lock_timeout', %s, true)", (f"{_LOCK_TIMEOUT_MS}ms",))
|
|
332
|
+
await conn.execute("SELECT pg_advisory_xact_lock(%s)", (_advisory_key(f"{namespace}\x00{log_key}"),))
|
|
333
|
+
|
|
334
|
+
async def _last_seq(self, conn: Connection, namespace: str, log_key: str, run_id: str) -> int:
|
|
335
|
+
cursor = await conn.execute(self._select_last_seq, (namespace, log_key, run_id))
|
|
336
|
+
row = await cursor.fetchone()
|
|
337
|
+
return row[0] if row is not None and row[0] is not None else -1
|
|
338
|
+
|
|
339
|
+
async def _last_lifecycle_of_run(self, conn: Connection, namespace: str, log_key: str, run_id: str) -> Event | None:
|
|
340
|
+
cursor = await conn.execute(self._select_run_lifecycle, (namespace, log_key, run_id, _SORTED_LIFECYCLE_KINDS))
|
|
341
|
+
row = await cursor.fetchone()
|
|
342
|
+
return Event.model_validate(row[0]) if row is not None else None
|
|
343
|
+
|
|
344
|
+
async def _open_runs(self, conn: Connection, namespace: str, log_key: str) -> list[tuple[str, Event]]:
|
|
345
|
+
"""Every run in this log that has recorded a transition but not a terminal one,
|
|
346
|
+
paired with its own last event — whatever kind — because that event is the run's
|
|
347
|
+
last sign of life, and silence is all that separates an abandoned run from a
|
|
348
|
+
working one.
|
|
349
|
+
"""
|
|
350
|
+
cursor = await conn.execute(self._select_log_lifecycle, (namespace, log_key, _SORTED_LIFECYCLE_KINDS))
|
|
351
|
+
open_runs = [
|
|
352
|
+
row[0]
|
|
353
|
+
for row in await cursor.fetchall()
|
|
354
|
+
if status_of([Event.model_validate(row[1])]) not in TERMINAL_STATUSES
|
|
355
|
+
]
|
|
356
|
+
if not open_runs:
|
|
357
|
+
return []
|
|
358
|
+
cursor = await conn.execute(self._select_last_events, (namespace, log_key, open_runs))
|
|
359
|
+
last_events = {row[0]: Event.model_validate(row[1]) for row in await cursor.fetchall()}
|
|
360
|
+
return [(run_id, last_events[run_id]) for run_id in open_runs]
|
|
361
|
+
|
|
362
|
+
async def aclose(self) -> None:
|
|
363
|
+
try:
|
|
364
|
+
if self._conn is not None:
|
|
365
|
+
await self._conn.close()
|
|
366
|
+
except psycopg.Error as exc:
|
|
367
|
+
raise StoreError(f"closing the event log failed: {exc}") from exc
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def _row(namespace: str, log_key: str, event: Event) -> tuple[str, str, str, int, str]:
|
|
371
|
+
return (namespace, log_key, event.run_id, event.seq, event.model_dump_json())
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
__all__ = ["PostgresEventStore"]
|