agentdeck-sdk 3.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentdeck/README.md +50 -0
- agentdeck/__init__.py +51 -0
- agentdeck/adapters/__init__.py +5 -0
- agentdeck/adapters/control/__init__.py +1 -0
- agentdeck/adapters/control/memory/__init__.py +5 -0
- agentdeck/adapters/control/memory/port.py +25 -0
- agentdeck/adapters/control/sqlite/__init__.py +5 -0
- agentdeck/adapters/control/sqlite/port.py +111 -0
- agentdeck/adapters/engines/__init__.py +1 -0
- agentdeck/adapters/engines/langgraph/__init__.py +8 -0
- agentdeck/adapters/engines/langgraph/checkpointer.py +205 -0
- agentdeck/adapters/engines/langgraph/engine.py +410 -0
- agentdeck/adapters/engines/openai_agents/__init__.py +9 -0
- agentdeck/adapters/engines/openai_agents/engine.py +321 -0
- agentdeck/adapters/engines/openai_agents/reconcile.py +178 -0
- agentdeck/adapters/engines/openai_agents/runconfig.py +118 -0
- agentdeck/adapters/engines/openai_agents/sessions.py +100 -0
- agentdeck/adapters/engines/openai_agents/translate.py +124 -0
- agentdeck/adapters/engines/stub/__init__.py +5 -0
- agentdeck/adapters/engines/stub/engine.py +100 -0
- agentdeck/adapters/stores/__init__.py +1 -0
- agentdeck/adapters/stores/memory/__init__.py +5 -0
- agentdeck/adapters/stores/memory/store.py +148 -0
- agentdeck/adapters/stores/postgres/__init__.py +5 -0
- agentdeck/adapters/stores/postgres/store.py +374 -0
- agentdeck/adapters/stores/redis/__init__.py +5 -0
- agentdeck/adapters/stores/redis/store.py +359 -0
- agentdeck/adapters/stores/sqlite/__init__.py +5 -0
- agentdeck/adapters/stores/sqlite/store.py +338 -0
- agentdeck/adapters/telemetry/__init__.py +1 -0
- agentdeck/adapters/telemetry/langfuse/__init__.py +18 -0
- agentdeck/adapters/telemetry/langfuse/client.py +180 -0
- agentdeck/adapters/telemetry/langfuse/sink.py +366 -0
- agentdeck/adapters/telemetry/langfuse/trace.py +88 -0
- agentdeck/adapters/tools/__init__.py +1 -0
- agentdeck/adapters/tools/mcp/__init__.py +18 -0
- agentdeck/adapters/tools/mcp/lifecycle.py +178 -0
- agentdeck/adapters/tools/mcp/source.py +47 -0
- agentdeck/adapters/tools/mcp/transport.py +234 -0
- agentdeck/adapters/tools/mcp/wiring.py +63 -0
- agentdeck/authoring/__init__.py +22 -0
- agentdeck/authoring/agent.py +169 -0
- agentdeck/authoring/compile.py +250 -0
- agentdeck/authoring/graphs.py +145 -0
- agentdeck/authoring/hooks.py +117 -0
- agentdeck/authoring/injection.py +233 -0
- agentdeck/authoring/instructions.py +80 -0
- agentdeck/authoring/interrupts.py +37 -0
- agentdeck/authoring/nodes.py +140 -0
- agentdeck/authoring/runners/__init__.py +6 -0
- agentdeck/authoring/runners/agent.py +171 -0
- agentdeck/authoring/runners/workflow.py +99 -0
- agentdeck/authoring/skills.py +56 -0
- agentdeck/authoring/state.py +44 -0
- agentdeck/authoring/timers.py +44 -0
- agentdeck/authoring/tools.py +147 -0
- agentdeck/authoring/web_search.py +43 -0
- agentdeck/authoring/workflow.py +266 -0
- agentdeck/cli.py +56 -0
- agentdeck/composition.py +213 -0
- agentdeck/core/__init__.py +110 -0
- agentdeck/core/base.py +47 -0
- agentdeck/core/content.py +166 -0
- agentdeck/core/context.py +136 -0
- agentdeck/core/control.py +150 -0
- agentdeck/core/events.py +468 -0
- agentdeck/core/invocable.py +38 -0
- agentdeck/core/ports/__init__.py +26 -0
- agentdeck/core/ports/control.py +35 -0
- agentdeck/core/ports/engine.py +66 -0
- agentdeck/core/ports/sink.py +53 -0
- agentdeck/core/ports/store.py +170 -0
- agentdeck/core/ports/tools.py +57 -0
- agentdeck/core/reporting.py +75 -0
- agentdeck/core/status.py +75 -0
- agentdeck/deck.py +894 -0
- agentdeck/errors.py +67 -0
- agentdeck/mcp.py +82 -0
- agentdeck/observers.py +111 -0
- agentdeck/py.typed +0 -0
- agentdeck/runtime/__init__.py +1 -0
- agentdeck/runtime/capture.py +32 -0
- agentdeck/runtime/config.default.yaml +38 -0
- agentdeck/runtime/discovery.py +175 -0
- agentdeck/runtime/dispatch.py +440 -0
- agentdeck/runtime/registry.py +173 -0
- agentdeck/runtime/service.py +671 -0
- agentdeck/runtime/settings.py +568 -0
- agentdeck/serve.py +330 -0
- agentdeck/skills/__init__.py +114 -0
- agentdeck/skills/bundle.py +65 -0
- agentdeck/surfaces/__init__.py +4 -0
- agentdeck/surfaces/cli/__init__.py +7 -0
- agentdeck/surfaces/cli/chat.py +88 -0
- agentdeck/surfaces/serve/__init__.py +7 -0
- agentdeck/surfaces/serve/app.py +71 -0
- agentdeck/surfaces/serve/compat.py +212 -0
- agentdeck/surfaces/serve/workflows.py +69 -0
- agentdeck/testing.py +364 -0
- agentdeck_sdk-3.1.0.dist-info/METADATA +222 -0
- agentdeck_sdk-3.1.0.dist-info/RECORD +104 -0
- agentdeck_sdk-3.1.0.dist-info/WHEEL +4 -0
- agentdeck_sdk-3.1.0.dist-info/entry_points.txt +3 -0
- agentdeck_sdk-3.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,366 @@
|
|
|
1
|
+
"""``EventSinkPort`` that renders the canonical event stream as Langfuse traces.
|
|
2
|
+
|
|
3
|
+
One run is one trace: ``run.started`` opens the root observation, a terminal event closes
|
|
4
|
+
it, and what happens in between becomes children — tool calls as spans, reported model
|
|
5
|
+
usage as generations, workflow node updates as points on the timeline. Nothing here asks
|
|
6
|
+
whether the run was an agent or a workflow, which is exactly why both are traced by the
|
|
7
|
+
same code: the stream is all there is to read.
|
|
8
|
+
|
|
9
|
+
Two properties of the sink contract shape the whole module. ``emit`` must return promptly,
|
|
10
|
+
so every call does in-memory work only — the Langfuse SDK buffers observations and ships
|
|
11
|
+
them from its own background thread, and nothing here awaits a round trip. And the tap is
|
|
12
|
+
lossy: any event can be missing, so an observation left open by a lost ``completed`` event
|
|
13
|
+
is closed by the next one that implies it rather than waited for, and a run whose terminal
|
|
14
|
+
event never arrives is eventually evicted instead of leaking its spans forever.
|
|
15
|
+
|
|
16
|
+
Shutdown is where the buffering is paid for: ``close`` finishes what is still open and flushes
|
|
17
|
+
the SDK's queue itself, because a batch the SDK has not shipped yet leaves the process only if
|
|
18
|
+
something asks it to.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import asyncio
|
|
24
|
+
import json
|
|
25
|
+
import logging
|
|
26
|
+
import re
|
|
27
|
+
from dataclasses import dataclass, field
|
|
28
|
+
from typing import TYPE_CHECKING, Any
|
|
29
|
+
|
|
30
|
+
from agentdeck.core.content import DataBlock, ImageBlock, ResourceBlock, TextBlock
|
|
31
|
+
from agentdeck.core.events import (
|
|
32
|
+
ControlObserved,
|
|
33
|
+
ControlRequested,
|
|
34
|
+
NodeUpdated,
|
|
35
|
+
RunCancelled,
|
|
36
|
+
RunCompleted,
|
|
37
|
+
RunFailed,
|
|
38
|
+
RunInterrupted,
|
|
39
|
+
RunPaused,
|
|
40
|
+
RunResumed,
|
|
41
|
+
RunStarted,
|
|
42
|
+
ToolCallCompleted,
|
|
43
|
+
ToolCallStarted,
|
|
44
|
+
UsageReported,
|
|
45
|
+
)
|
|
46
|
+
from agentdeck.core.ports import EventSinkPort
|
|
47
|
+
|
|
48
|
+
if TYPE_CHECKING:
|
|
49
|
+
from collections.abc import Mapping
|
|
50
|
+
|
|
51
|
+
from agentdeck.adapters.telemetry.langfuse.trace import Level, Observation, ObservationKind, Tracer
|
|
52
|
+
from agentdeck.core.content import Input
|
|
53
|
+
from agentdeck.core.events import Event, Usage
|
|
54
|
+
|
|
55
|
+
logger = logging.getLogger(__name__)
|
|
56
|
+
|
|
57
|
+
# How many runs may be mid-trace at once. A run whose terminal event the tap dropped would
|
|
58
|
+
# otherwise hold its observations for the life of the process; the oldest is closed as
|
|
59
|
+
# abandoned to make room, which reports the loss instead of accumulating it.
|
|
60
|
+
MAX_OPEN_RUNS = 512
|
|
61
|
+
|
|
62
|
+
# The same bound per run, for the same reason: a long, chatty run that keeps losing
|
|
63
|
+
# ``tool.call.completed`` would otherwise hold one unfinished span per loss, arguments and all.
|
|
64
|
+
MAX_OPEN_CALLS = 256
|
|
65
|
+
|
|
66
|
+
# What an invocable's kind means to Langfuse. A skill is a tool call by another name.
|
|
67
|
+
_OBSERVATION_OF: dict[str, ObservationKind] = {"agent": "agent", "workflow": "chain", "skill": "tool"}
|
|
68
|
+
|
|
69
|
+
# A base64 data URI anywhere in an input or an output is decoded by the Langfuse SDK and
|
|
70
|
+
# queued for upload to its media store, so one gets described here instead. Bytes leaving for
|
|
71
|
+
# a third party are this adapter's decision to make, not a default to inherit.
|
|
72
|
+
_DATA_URI = re.compile(
|
|
73
|
+
r"data:(?P<media_type>[\w.+-]+/[\w.+-]+)?(?:;[\w.+-]+=[\w.+-]+)*;base64,(?P<data>[A-Za-z0-9+/=]+)"
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@dataclass(slots=True)
|
|
78
|
+
class _OpenTrace:
|
|
79
|
+
"""A run's root observation and the tool calls still open under it."""
|
|
80
|
+
|
|
81
|
+
root: Observation
|
|
82
|
+
calls: dict[str, Observation] = field(default_factory=dict)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class LangfuseSink(EventSinkPort):
|
|
86
|
+
"""Renders each run as a Langfuse trace. Registered only when Langfuse is configured."""
|
|
87
|
+
|
|
88
|
+
def __init__(
|
|
89
|
+
self,
|
|
90
|
+
tracer: Tracer,
|
|
91
|
+
*,
|
|
92
|
+
max_open_runs: int = MAX_OPEN_RUNS,
|
|
93
|
+
max_open_calls: int = MAX_OPEN_CALLS,
|
|
94
|
+
) -> None:
|
|
95
|
+
self._tracer = tracer
|
|
96
|
+
self._max_open_runs = max_open_runs
|
|
97
|
+
self._max_open_calls = max_open_calls
|
|
98
|
+
self._open: dict[str, _OpenTrace] = {}
|
|
99
|
+
# Outlives the trace it belongs to, because a suspended run's trace closes and its
|
|
100
|
+
# continuation still has to say whose run it is. Bounded like ``_open``, and dropped
|
|
101
|
+
# once the run reaches a terminal event.
|
|
102
|
+
|
|
103
|
+
async def emit(self, event: Event) -> None:
|
|
104
|
+
"""Fold one event into its run's trace. Never suspends: no round trip is awaited."""
|
|
105
|
+
match event.payload:
|
|
106
|
+
case RunStarted() as started:
|
|
107
|
+
self._start(event, started)
|
|
108
|
+
case RunResumed() as resumed:
|
|
109
|
+
self._continue(event, resumed)
|
|
110
|
+
case ControlRequested() as requested:
|
|
111
|
+
self._control(event, f"{requested.verb} requested", requested.reason)
|
|
112
|
+
case ControlObserved() as observed:
|
|
113
|
+
self._control(event, f"{observed.verb} observed", f"at {observed.safe_point}")
|
|
114
|
+
case ToolCallStarted() as call:
|
|
115
|
+
self._tool_started(event, call)
|
|
116
|
+
case ToolCallCompleted() as call:
|
|
117
|
+
self._tool_completed(event, call)
|
|
118
|
+
case NodeUpdated() as node:
|
|
119
|
+
self._node(event, node)
|
|
120
|
+
case UsageReported() as reported:
|
|
121
|
+
self._usage(event, reported)
|
|
122
|
+
case RunCompleted() as completed:
|
|
123
|
+
self._finish(event, output=_render(completed.output), usage=completed.usage)
|
|
124
|
+
case RunFailed() as failed:
|
|
125
|
+
self._finish(
|
|
126
|
+
event,
|
|
127
|
+
level="ERROR",
|
|
128
|
+
status=f"{failed.error_code}: {failed.message}",
|
|
129
|
+
metadata={"error_code": failed.error_code, "retryable": failed.retryable},
|
|
130
|
+
)
|
|
131
|
+
case RunCancelled() as cancelled:
|
|
132
|
+
self._finish(event, level="WARNING", status=cancelled.reason or "cancelled")
|
|
133
|
+
case RunInterrupted() as interrupted:
|
|
134
|
+
# A suspended run is closed rather than held open: its answer may take days
|
|
135
|
+
# and may arrive in another process, and a trace nobody can see until then is
|
|
136
|
+
# worse than one that says "waiting". The resume opens a second root in the
|
|
137
|
+
# same trace, so both halves stay together.
|
|
138
|
+
self._finish(
|
|
139
|
+
event,
|
|
140
|
+
status=f"waiting for {interrupted.reason}",
|
|
141
|
+
metadata={"interrupt_id": interrupted.interrupt_id, "suspended": True},
|
|
142
|
+
)
|
|
143
|
+
case RunPaused() as paused:
|
|
144
|
+
self._finish(event, status=paused.reason or "paused", metadata={"suspended": True})
|
|
145
|
+
case _:
|
|
146
|
+
# Deltas, messages, artifacts, custom kinds and anything a newer writer added:
|
|
147
|
+
# a trace is not a transcript, and the event log is what holds the full record.
|
|
148
|
+
pass
|
|
149
|
+
|
|
150
|
+
async def close(self) -> None:
|
|
151
|
+
"""Close the traces still open and push the SDK's buffer out before the process ends.
|
|
152
|
+
|
|
153
|
+
Both halves matter. An observation nobody finished is never shipped at all, so a run the
|
|
154
|
+
shutdown cut short would otherwise vanish rather than show up as interrupted. And what
|
|
155
|
+
the SDK has already batched only leaves on a flush — the one it does at interpreter exit
|
|
156
|
+
never happens to a process that is killed, which is precisely when the telemetry of the
|
|
157
|
+
last few seconds is worth having.
|
|
158
|
+
"""
|
|
159
|
+
if self._open:
|
|
160
|
+
logger.warning("%d langfuse traces were still open at shutdown", len(self._open))
|
|
161
|
+
for open_trace in self._open.values():
|
|
162
|
+
self._abandon(open_trace, "the process shut down before the run ended")
|
|
163
|
+
self._open.clear()
|
|
164
|
+
# On a worker thread because the SDK's flush blocks: on the event loop it would block the
|
|
165
|
+
# very deadline that is supposed to keep a slow flush from holding shutdown open.
|
|
166
|
+
await asyncio.to_thread(self._tracer.flush)
|
|
167
|
+
|
|
168
|
+
def _start(self, event: Event, started: RunStarted) -> None:
|
|
169
|
+
self._track(
|
|
170
|
+
event.run_id,
|
|
171
|
+
self._tracer.root(
|
|
172
|
+
started.invocable,
|
|
173
|
+
kind=_OBSERVATION_OF.get(started.kind_of_invocable, "span"),
|
|
174
|
+
trace_key=event.run_id,
|
|
175
|
+
session_id=event.session_id,
|
|
176
|
+
input=_render(started.input),
|
|
177
|
+
metadata=_without_nones(
|
|
178
|
+
{
|
|
179
|
+
"run_id": event.run_id,
|
|
180
|
+
"namespace": event.namespace,
|
|
181
|
+
"invocable_kind": started.kind_of_invocable,
|
|
182
|
+
}
|
|
183
|
+
),
|
|
184
|
+
),
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
def _continue(self, event: Event, resumed: RunResumed) -> None:
|
|
188
|
+
"""Reopen the trace of a run that was suspended, as a second root under the same key.
|
|
189
|
+
|
|
190
|
+
The kind and the run's constants live in ``run.started``, which a process that only
|
|
191
|
+
picked up the resume never saw — so a continuation is a plain span and says who it
|
|
192
|
+
belongs to in its metadata.
|
|
193
|
+
|
|
194
|
+
The answer the resume carried is the continuation's input, media-described like any
|
|
195
|
+
other content: a trace of an approval flow that does not say what was approved is a
|
|
196
|
+
trace of the wrong half of the story.
|
|
197
|
+
"""
|
|
198
|
+
if event.run_id in self._open:
|
|
199
|
+
return
|
|
200
|
+
self._track(
|
|
201
|
+
event.run_id,
|
|
202
|
+
self._tracer.root(
|
|
203
|
+
event.origin,
|
|
204
|
+
kind="span",
|
|
205
|
+
trace_key=event.run_id,
|
|
206
|
+
session_id=event.session_id,
|
|
207
|
+
input=_render(resumed.value) if resumed.value is not None else None,
|
|
208
|
+
metadata={"run_id": event.run_id, "namespace": event.namespace, "resumed": True},
|
|
209
|
+
),
|
|
210
|
+
)
|
|
211
|
+
|
|
212
|
+
def _control(self, event: Event, what: str, detail: str | None) -> None:
|
|
213
|
+
"""A control phase as a point on the run's timeline.
|
|
214
|
+
|
|
215
|
+
Both phases are recorded, because the gap between them is the number an operator
|
|
216
|
+
actually asks about — "I pressed cancel and it kept going" is answered by when the run
|
|
217
|
+
reached a safe point, not by when the signal was written. A signal that arrives while
|
|
218
|
+
no trace is open here (another process opened it, or the tap dropped the opening) is
|
|
219
|
+
left to the process that has one, the same way tool spans are.
|
|
220
|
+
"""
|
|
221
|
+
open_trace = self._open.get(event.run_id)
|
|
222
|
+
if open_trace is None:
|
|
223
|
+
return
|
|
224
|
+
open_trace.root.child(event.kind, kind="span").finish(output=what, status=detail)
|
|
225
|
+
|
|
226
|
+
def _tool_started(self, event: Event, call: ToolCallStarted) -> None:
|
|
227
|
+
open_trace = self._open.get(event.run_id)
|
|
228
|
+
if open_trace is None:
|
|
229
|
+
return
|
|
230
|
+
while len(open_trace.calls) >= self._max_open_calls:
|
|
231
|
+
stale_id = next(iter(open_trace.calls))
|
|
232
|
+
_no_completion(open_trace.calls.pop(stale_id), stale_id)
|
|
233
|
+
open_trace.calls[call.call_id] = open_trace.root.child(call.tool, kind="tool", input=_without_media(call.args))
|
|
234
|
+
|
|
235
|
+
def _tool_completed(self, event: Event, call: ToolCallCompleted) -> None:
|
|
236
|
+
open_trace = self._open.get(event.run_id)
|
|
237
|
+
if open_trace is None:
|
|
238
|
+
return
|
|
239
|
+
# No open span means the dispatch dropped the ``started`` event: record the result on
|
|
240
|
+
# a span of its own rather than losing the call, and accept that it has no duration.
|
|
241
|
+
span = open_trace.calls.pop(call.call_id, None) or open_trace.root.child(call.tool, kind="tool")
|
|
242
|
+
span.finish(
|
|
243
|
+
output=_without_media(call.result_preview),
|
|
244
|
+
metadata=_without_nones(
|
|
245
|
+
{
|
|
246
|
+
"result_size": call.result_size,
|
|
247
|
+
"result_sha256": call.result_sha256,
|
|
248
|
+
"artifact_id": call.artifact_id,
|
|
249
|
+
}
|
|
250
|
+
),
|
|
251
|
+
level="ERROR" if call.error else None,
|
|
252
|
+
status=call.error,
|
|
253
|
+
)
|
|
254
|
+
|
|
255
|
+
def _node(self, event: Event, node: NodeUpdated) -> None:
|
|
256
|
+
open_trace = self._open.get(event.run_id)
|
|
257
|
+
if open_trace is None:
|
|
258
|
+
return
|
|
259
|
+
# The patched key names, never their values: which node ran and what it touched is
|
|
260
|
+
# what makes a workflow trace readable, and the state itself is not telemetry's to copy.
|
|
261
|
+
open_trace.root.child(node.node, kind="span").finish(output=sorted(node.state_patch))
|
|
262
|
+
|
|
263
|
+
def _usage(self, event: Event, reported: UsageReported) -> None:
|
|
264
|
+
open_trace = self._open.get(event.run_id)
|
|
265
|
+
if open_trace is None:
|
|
266
|
+
return
|
|
267
|
+
open_trace.root.child(reported.model, kind="generation").finish(usage=reported.usage)
|
|
268
|
+
|
|
269
|
+
def _finish(
|
|
270
|
+
self,
|
|
271
|
+
event: Event,
|
|
272
|
+
*,
|
|
273
|
+
output: Any = None,
|
|
274
|
+
metadata: Mapping[str, Any] | None = None,
|
|
275
|
+
level: Level | None = None,
|
|
276
|
+
status: str | None = None,
|
|
277
|
+
usage: Usage | None = None,
|
|
278
|
+
) -> None:
|
|
279
|
+
open_trace = self._open.pop(event.run_id, None)
|
|
280
|
+
if open_trace is None:
|
|
281
|
+
return
|
|
282
|
+
for call_id, span in open_trace.calls.items():
|
|
283
|
+
_no_completion(span, call_id)
|
|
284
|
+
if usage is not None:
|
|
285
|
+
# Langfuse accounts tokens and cost on generations only, so the run's own total
|
|
286
|
+
# gets a generation of its own. Reported per-call usage is already counted by the
|
|
287
|
+
# generations those events opened, hence the distinct name rather than a second
|
|
288
|
+
# copy of a model's.
|
|
289
|
+
open_trace.root.child("run.usage", kind="generation").finish(usage=usage)
|
|
290
|
+
open_trace.root.finish(output=output, metadata=metadata, level=level, status=status)
|
|
291
|
+
|
|
292
|
+
def _track(self, run_id: str, root: Observation) -> None:
|
|
293
|
+
while len(self._open) >= self._max_open_runs:
|
|
294
|
+
stale_id = next(iter(self._open))
|
|
295
|
+
logger.warning("langfuse trace for run %s abandoned: %d runs already open", stale_id, self._max_open_runs)
|
|
296
|
+
self._abandon(self._open.pop(stale_id), "no terminal event seen")
|
|
297
|
+
if (superseded := self._open.pop(run_id, None)) is not None:
|
|
298
|
+
# One opening per run is the Runtime's promise; a second one would otherwise leave
|
|
299
|
+
# the first root open forever, invisible in Langfuse and unaccounted for here.
|
|
300
|
+
logger.warning("langfuse trace for run %s reopened; the first one is abandoned", run_id)
|
|
301
|
+
self._abandon(superseded, "superseded by a second opening of this run")
|
|
302
|
+
self._open[run_id] = _OpenTrace(root)
|
|
303
|
+
|
|
304
|
+
def _abandon(self, open_trace: _OpenTrace, status: str) -> None:
|
|
305
|
+
for call_id, span in open_trace.calls.items():
|
|
306
|
+
_no_completion(span, call_id)
|
|
307
|
+
open_trace.root.finish(level="WARNING", status=status)
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def _no_completion(span: Observation, call_id: str) -> None:
|
|
311
|
+
span.finish(level="WARNING", status=f"no completion event for call {call_id}")
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _render(blocks: Input) -> list[str]:
|
|
315
|
+
"""Content blocks as strings a trace can carry: bytes are described, never copied."""
|
|
316
|
+
out: list[str] = []
|
|
317
|
+
for block in blocks:
|
|
318
|
+
match block:
|
|
319
|
+
case TextBlock():
|
|
320
|
+
out.append(_without_media(block.text))
|
|
321
|
+
case ImageBlock():
|
|
322
|
+
out.append(f"<image {block.media_type}, {len(block.data_b64)} base64 chars>")
|
|
323
|
+
case ResourceBlock():
|
|
324
|
+
out.append(f"<resource {block.uri}>")
|
|
325
|
+
case DataBlock():
|
|
326
|
+
out.append(json.dumps(_without_media(block.data), sort_keys=True))
|
|
327
|
+
case _:
|
|
328
|
+
# Reachable since UnknownBlock (#109): a block type this version doesn't know
|
|
329
|
+
# lands here instead of rejecting the whole event. A dropped block would read
|
|
330
|
+
# as "the run had no input", which is worse than a placeholder saying what was
|
|
331
|
+
# lost, so this names the type even though it can't render the content.
|
|
332
|
+
out.append(f"<{getattr(block, 'type', 'unknown')} block>")
|
|
333
|
+
return out
|
|
334
|
+
|
|
335
|
+
|
|
336
|
+
def _without_media(value: Any) -> Any:
|
|
337
|
+
"""``value`` with every inline base64 payload replaced by a description of it.
|
|
338
|
+
|
|
339
|
+
Tool arguments and results are engine-shaped data this adapter passes through, and the
|
|
340
|
+
Langfuse SDK decodes a data URI it finds in one and queues the bytes for upload to its
|
|
341
|
+
media store. A run that hands a tool an inline image would ship that image; describing it
|
|
342
|
+
keeps the same promise the run's own content blocks get.
|
|
343
|
+
"""
|
|
344
|
+
match value:
|
|
345
|
+
case str():
|
|
346
|
+
return _DATA_URI.sub(_describe_media, value)
|
|
347
|
+
case dict():
|
|
348
|
+
return {key: _without_media(item) for key, item in value.items()}
|
|
349
|
+
case list():
|
|
350
|
+
return [_without_media(item) for item in value]
|
|
351
|
+
case tuple():
|
|
352
|
+
return tuple(_without_media(item) for item in value)
|
|
353
|
+
case _:
|
|
354
|
+
return value
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def _describe_media(match: re.Match[str]) -> str:
|
|
358
|
+
media_type = match.group("media_type") or "application/octet-stream"
|
|
359
|
+
return f"<inline {media_type}, {len(match.group('data'))} base64 chars>"
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def _without_nones(metadata: Mapping[str, Any]) -> dict[str, Any]:
|
|
363
|
+
return {key: value for key, value in metadata.items() if value is not None}
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
__all__ = ["MAX_OPEN_CALLS", "MAX_OPEN_RUNS", "LangfuseSink"]
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""The tracing surface :class:`~agentdeck.adapters.telemetry.langfuse.sink.LangfuseSink`
|
|
2
|
+
writes to: one root observation per run, children under it, each finished exactly once.
|
|
3
|
+
|
|
4
|
+
Narrow on purpose. The event-to-observation mapping is the part of this adapter that can be
|
|
5
|
+
wrong, and stating it as two protocols is what lets a recorder stand in for Langfuse in a
|
|
6
|
+
test — no keys, no collector, no network — while the SDK stays behind one module.
|
|
7
|
+
|
|
8
|
+
Observations are opened and finished as separate calls rather than through a context
|
|
9
|
+
manager: a sink sees a run as interleaved events, so the span that opens on
|
|
10
|
+
``tool.call.started`` has to outlive the ``emit`` that opened it.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from typing import TYPE_CHECKING, Any, Literal, Protocol
|
|
16
|
+
|
|
17
|
+
if TYPE_CHECKING:
|
|
18
|
+
from collections.abc import Mapping
|
|
19
|
+
|
|
20
|
+
from agentdeck.core.events import Usage
|
|
21
|
+
|
|
22
|
+
# Langfuse's observation types, restricted to the ones a canonical event can justify.
|
|
23
|
+
ObservationKind = Literal["agent", "chain", "tool", "span", "generation"]
|
|
24
|
+
|
|
25
|
+
# Langfuse's observation levels, verbatim — an adapter that renamed them would only make
|
|
26
|
+
# the mapping to the backend's own vocabulary harder to check.
|
|
27
|
+
Level = Literal["DEBUG", "DEFAULT", "WARNING", "ERROR"]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class Observation(Protocol):
|
|
31
|
+
"""One open observation. It may open children and must be finished exactly once."""
|
|
32
|
+
|
|
33
|
+
def child(
|
|
34
|
+
self,
|
|
35
|
+
name: str,
|
|
36
|
+
*,
|
|
37
|
+
kind: ObservationKind,
|
|
38
|
+
input: Any = None,
|
|
39
|
+
metadata: Mapping[str, Any] | None = None,
|
|
40
|
+
) -> Observation:
|
|
41
|
+
"""Open a nested observation under this one."""
|
|
42
|
+
|
|
43
|
+
def finish(
|
|
44
|
+
self,
|
|
45
|
+
*,
|
|
46
|
+
output: Any = None,
|
|
47
|
+
metadata: Mapping[str, Any] | None = None,
|
|
48
|
+
level: Level | None = None,
|
|
49
|
+
status: str | None = None,
|
|
50
|
+
usage: Usage | None = None,
|
|
51
|
+
) -> None:
|
|
52
|
+
"""Close this observation, recording how it ended.
|
|
53
|
+
|
|
54
|
+
``usage`` is only accounted on a ``generation``; on any other kind the backend
|
|
55
|
+
ignores it, so a caller with a token total to report opens one.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class Tracer(Protocol):
|
|
60
|
+
"""Opens the root of a trace — the only observation that carries the trace's identity."""
|
|
61
|
+
|
|
62
|
+
def root(
|
|
63
|
+
self,
|
|
64
|
+
name: str,
|
|
65
|
+
*,
|
|
66
|
+
kind: ObservationKind,
|
|
67
|
+
trace_key: str,
|
|
68
|
+
session_id: str | None,
|
|
69
|
+
input: Any = None,
|
|
70
|
+
metadata: Mapping[str, Any] | None = None,
|
|
71
|
+
) -> Observation:
|
|
72
|
+
"""Open the root observation of the trace ``trace_key`` identifies.
|
|
73
|
+
|
|
74
|
+
Equal ``trace_key``s belong to the same trace, whichever process opened them — that
|
|
75
|
+
is what makes a run suspended in one worker and resumed in another one trace rather
|
|
76
|
+
than two.
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
def flush(self) -> None:
|
|
80
|
+
"""Ship what is still buffered, blocking until it has left or been given up on.
|
|
81
|
+
|
|
82
|
+
Called once, when the sink is closed. Blocking because the SDK's own flush is, and
|
|
83
|
+
pretending otherwise would only hide where the waiting happens: the sink puts this on a
|
|
84
|
+
worker thread so the deadline bounding it stays honest.
|
|
85
|
+
"""
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
__all__ = ["Level", "Observation", "ObservationKind", "Tracer"]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Tool sources — implementations of ``ToolSourcePort``, one directory per system."""
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""The MCP tool adapter: ``ToolSourcePort`` over an MCP server registry, transport and all."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from agentdeck.adapters.tools.mcp.lifecycle import MCPLifecycle
|
|
6
|
+
from agentdeck.adapters.tools.mcp.source import MCP_SERVER_NAMES_KEY, MCPToolSource
|
|
7
|
+
from agentdeck.adapters.tools.mcp.transport import MCPServerStreamableHttpResilient
|
|
8
|
+
from agentdeck.adapters.tools.mcp.wiring import mcp_status_banner, resolve_agent_mcp_servers, resolve_agent_mcp_status
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"MCP_SERVER_NAMES_KEY",
|
|
12
|
+
"MCPLifecycle",
|
|
13
|
+
"MCPServerStreamableHttpResilient",
|
|
14
|
+
"MCPToolSource",
|
|
15
|
+
"mcp_status_banner",
|
|
16
|
+
"resolve_agent_mcp_servers",
|
|
17
|
+
"resolve_agent_mcp_status",
|
|
18
|
+
]
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
"""MCP server registry + process lifecycle.
|
|
2
|
+
|
|
3
|
+
Agents name the servers they need; this owns how to reach them. The composition root
|
|
4
|
+
(``Deck``/``App``) hands :meth:`MCPLifecycle.configure`/:meth:`startup` the ``{name: spec}``
|
|
5
|
+
config it read from its own source — an ``MCP(...)`` capability object's ``.mcp.json``, today::
|
|
6
|
+
|
|
7
|
+
{"mcpServers": {"agentdeck": {"type": "http", "url": "http://host.docker.internal:8765/mcp"}}}
|
|
8
|
+
|
|
9
|
+
Every server connects at startup; connect failures are **soft** — the server is
|
|
10
|
+
marked unavailable and the backend still boots. Agents referencing an
|
|
11
|
+
unavailable server boot without its tools rather than crashing.
|
|
12
|
+
|
|
13
|
+
:meth:`MCPLifecycle.startup` and :meth:`MCPLifecycle.shutdown` are the composition
|
|
14
|
+
root's to call (``App`` does, in its lifespan) — resolving tools never connects.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import asyncio
|
|
20
|
+
import logging
|
|
21
|
+
from typing import TYPE_CHECKING, Any
|
|
22
|
+
|
|
23
|
+
from agentdeck.adapters.tools.mcp.transport import MCPServerStreamableHttpResilient
|
|
24
|
+
from agentdeck.errors import ConfigError
|
|
25
|
+
|
|
26
|
+
if TYPE_CHECKING:
|
|
27
|
+
from collections.abc import Sequence
|
|
28
|
+
|
|
29
|
+
from agents.mcp import MCPServer
|
|
30
|
+
|
|
31
|
+
logger = logging.getLogger(__name__)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _build_server(name: str, spec: dict[str, Any]) -> MCPServer:
|
|
35
|
+
"""Construct an ``MCPServer`` from one JSON entry (http transport only)."""
|
|
36
|
+
transport = (spec.get("type") or "http").lower()
|
|
37
|
+
if transport not in {"http", "streamable-http", "streamable_http"}:
|
|
38
|
+
raise ConfigError(f"MCP server '{name}': unsupported transport '{transport}'")
|
|
39
|
+
url = spec.get("url")
|
|
40
|
+
if not isinstance(url, str):
|
|
41
|
+
raise ConfigError(f"MCP server '{name}': missing `url` for http transport")
|
|
42
|
+
params: dict[str, Any] = {"url": url}
|
|
43
|
+
if isinstance(headers := spec.get("headers"), dict):
|
|
44
|
+
params["headers"] = headers
|
|
45
|
+
if isinstance(timeout := spec.get("timeout"), (int, float)):
|
|
46
|
+
params["timeout"] = timeout
|
|
47
|
+
return MCPServerStreamableHttpResilient(params=params, name=name)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class MCPLifecycle:
|
|
51
|
+
"""Process-wide registry of MCP servers, keyed by config name.
|
|
52
|
+
|
|
53
|
+
:meth:`startup` connects each configured server (soft per-server failure).
|
|
54
|
+
:meth:`resolve_or_pending` hands an agent its server instance.
|
|
55
|
+
:meth:`shutdown` cleans up. Idempotent; safe to configure/resolve sync.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
_servers: dict[str, MCPServer] = {}
|
|
59
|
+
_failed: dict[str, BaseException] = {}
|
|
60
|
+
_connected: set[str] = set()
|
|
61
|
+
_config: dict[str, dict[str, Any]] = {}
|
|
62
|
+
_lock: asyncio.Lock | None = None
|
|
63
|
+
|
|
64
|
+
@classmethod
|
|
65
|
+
def _ensure_lock(cls) -> asyncio.Lock:
|
|
66
|
+
if cls._lock is None:
|
|
67
|
+
cls._lock = asyncio.Lock()
|
|
68
|
+
return cls._lock
|
|
69
|
+
|
|
70
|
+
@classmethod
|
|
71
|
+
def configured_names(cls) -> Sequence[str]:
|
|
72
|
+
return tuple(cls._servers.keys())
|
|
73
|
+
|
|
74
|
+
@classmethod
|
|
75
|
+
def failed_names(cls) -> Sequence[str]:
|
|
76
|
+
return tuple(cls._failed.keys())
|
|
77
|
+
|
|
78
|
+
@classmethod
|
|
79
|
+
def is_available(cls, name: str) -> bool:
|
|
80
|
+
return name in cls._connected
|
|
81
|
+
|
|
82
|
+
@classmethod
|
|
83
|
+
def failure_reason(cls, name: str) -> BaseException | None:
|
|
84
|
+
return cls._failed.get(name)
|
|
85
|
+
|
|
86
|
+
@classmethod
|
|
87
|
+
def configure(cls, config: dict[str, dict[str, Any]] | None = None) -> None:
|
|
88
|
+
"""Load server specs without connecting. Safe to call sync (import-time builds).
|
|
89
|
+
|
|
90
|
+
``config`` is the composition root's own ``{name: spec}`` — an ``MCP(...)``
|
|
91
|
+
capability object's ``.config()``, typically. No config (or ``None``) means no
|
|
92
|
+
servers — the same fail-open rule a project with no ``.mcp.json`` has always had.
|
|
93
|
+
"""
|
|
94
|
+
config = config or {}
|
|
95
|
+
if not config:
|
|
96
|
+
logger.info("No MCP servers configured; none will be connected.")
|
|
97
|
+
cls._config = config
|
|
98
|
+
for name, spec in config.items():
|
|
99
|
+
if name in cls._servers or name in cls._failed:
|
|
100
|
+
continue
|
|
101
|
+
try:
|
|
102
|
+
cls._servers[name] = _build_server(name, spec)
|
|
103
|
+
except Exception as exc: # config errors are soft — disable this server, keep the rest
|
|
104
|
+
logger.warning("MCP server '%s' disabled (config error): %r", name, exc)
|
|
105
|
+
cls._failed[name] = exc
|
|
106
|
+
|
|
107
|
+
@classmethod
|
|
108
|
+
async def startup(cls, config: dict[str, dict[str, Any]] | None = None) -> None:
|
|
109
|
+
"""Configure (if not yet) and connect each known server; failures are soft."""
|
|
110
|
+
cls.configure(config)
|
|
111
|
+
async with cls._ensure_lock():
|
|
112
|
+
for name, server in list(cls._servers.items()):
|
|
113
|
+
if name in cls._connected or name in cls._failed:
|
|
114
|
+
continue
|
|
115
|
+
logger.info("MCPLifecycle: connecting %s", name)
|
|
116
|
+
try:
|
|
117
|
+
await server.connect()
|
|
118
|
+
except Exception as exc: # soft: a down MCP host must never fail the backend's boot
|
|
119
|
+
logger.error(
|
|
120
|
+
"MCPLifecycle: connection to '%s' FAILED — agents referencing it "
|
|
121
|
+
"boot without its tools. Cause: %r",
|
|
122
|
+
name,
|
|
123
|
+
exc,
|
|
124
|
+
)
|
|
125
|
+
cls._failed[name] = exc
|
|
126
|
+
# Soft-fail: unreachable for the rest of the process (no boot-time reconnect).
|
|
127
|
+
# A drop *after* a successful connect is handled by the transport's reconnect.
|
|
128
|
+
cls._servers.pop(name, None)
|
|
129
|
+
continue
|
|
130
|
+
cls._connected.add(name)
|
|
131
|
+
|
|
132
|
+
@classmethod
|
|
133
|
+
async def shutdown(cls) -> None:
|
|
134
|
+
async with cls._ensure_lock():
|
|
135
|
+
for name in list(cls._connected):
|
|
136
|
+
server = cls._servers.get(name)
|
|
137
|
+
if server is None:
|
|
138
|
+
cls._connected.discard(name)
|
|
139
|
+
continue
|
|
140
|
+
try:
|
|
141
|
+
await server.cleanup()
|
|
142
|
+
except Exception: # best-effort teardown — keep cleaning up the rest
|
|
143
|
+
logger.exception("MCPLifecycle: cleanup failed for '%s'", name)
|
|
144
|
+
cls._connected.discard(name)
|
|
145
|
+
|
|
146
|
+
@classmethod
|
|
147
|
+
def resolve_or_pending(cls, name: str) -> MCPServer | None:
|
|
148
|
+
"""Return the server for ``name`` regardless of connect state, or ``None``.
|
|
149
|
+
|
|
150
|
+
Lets agents declared at import time wire up the same instance the lifecycle
|
|
151
|
+
connects later. ``None`` when the name is unknown or its connect failed —
|
|
152
|
+
callers (``authoring.compile.compile_agent``) filter that out to boot with reduced capability.
|
|
153
|
+
"""
|
|
154
|
+
if name in cls._failed:
|
|
155
|
+
return None
|
|
156
|
+
if name not in cls._servers:
|
|
157
|
+
if not cls._config:
|
|
158
|
+
cls.configure()
|
|
159
|
+
if name not in cls._servers and name not in cls._failed:
|
|
160
|
+
logger.warning(
|
|
161
|
+
"MCP server '%s' not found in config; agent boots without it. Configured: %r",
|
|
162
|
+
name,
|
|
163
|
+
list(cls._servers.keys()),
|
|
164
|
+
)
|
|
165
|
+
return None
|
|
166
|
+
return cls._servers.get(name)
|
|
167
|
+
|
|
168
|
+
@classmethod
|
|
169
|
+
def reset(cls) -> None:
|
|
170
|
+
"""Tests only — clears state. Does not call cleanup()."""
|
|
171
|
+
cls._servers.clear()
|
|
172
|
+
cls._failed.clear()
|
|
173
|
+
cls._connected.clear()
|
|
174
|
+
cls._config.clear()
|
|
175
|
+
cls._lock = None
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
__all__ = ["MCPLifecycle"]
|