percolate-core 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- percolate_core/__init__.py +13 -0
- percolate_core/agentic/__init__.py +48 -0
- percolate_core/agentic/api.py +283 -0
- percolate_core/agentic/authoring.py +165 -0
- percolate_core/agentic/cli.py +425 -0
- percolate_core/agentic/contracts/__init__.py +87 -0
- percolate_core/agentic/contracts/agents.py +148 -0
- percolate_core/agentic/contracts/context.py +122 -0
- percolate_core/agentic/contracts/conversation.py +154 -0
- percolate_core/agentic/contracts/protocol.py +150 -0
- percolate_core/agentic/contracts/tools.py +102 -0
- percolate_core/agentic/db/__init__.py +32 -0
- percolate_core/agentic/db/client.py +174 -0
- percolate_core/agentic/db/repository.py +503 -0
- percolate_core/agentic/gateway.py +352 -0
- percolate_core/agentic/phases.py +110 -0
- percolate_core/agentic/runtime/__init__.py +41 -0
- percolate_core/agentic/runtime/chained_action.py +125 -0
- percolate_core/agentic/runtime/citations.py +75 -0
- percolate_core/agentic/runtime/delegation.py +137 -0
- percolate_core/agentic/runtime/engine.py +176 -0
- percolate_core/agentic/runtime/injection.py +58 -0
- percolate_core/agentic/runtime/persistence.py +314 -0
- percolate_core/agentic/runtime/protocol_adapter.py +114 -0
- percolate_core/agentic/sequencer.py +597 -0
- percolate_core/agentic/settings.py +82 -0
- percolate_core/agentic/testing.py +295 -0
- percolate_core/agentic/tools/__init__.py +24 -0
- percolate_core/agentic/tools/mcp.py +140 -0
- percolate_core/agentic/tools/openapi.py +294 -0
- percolate_core/agentic/tools/provider.py +89 -0
- percolate_core/agentic/tools/registry.py +142 -0
- percolate_core/agentic/tools/toolset.py +83 -0
- percolate_core/cli.py +64 -0
- percolate_core/content/__init__.py +5 -0
- percolate_core/content/server.py +115 -0
- percolate_core/content/storage.py +63 -0
- percolate_core/core/__init__.py +14 -0
- percolate_core/core/config.py +48 -0
- percolate_core/core/identity.py +128 -0
- percolate_core/worker/__init__.py +8 -0
- percolate_core/worker/http.py +100 -0
- percolate_core/worker/loop.py +111 -0
- percolate_core/worker/templates.py +84 -0
- percolate_core-0.1.0.dist-info/METADATA +195 -0
- percolate_core-0.1.0.dist-info/RECORD +49 -0
- percolate_core-0.1.0.dist-info/WHEEL +4 -0
- percolate_core-0.1.0.dist-info/entry_points.txt +2 -0
- percolate_core-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""percolate-core — Postgres-native subsystems.
|
|
2
|
+
|
|
3
|
+
The database is the system; these are the processes that sit in front of it.
|
|
4
|
+
Everything here connects through the same RLS-scoped roles and holds no table
|
|
5
|
+
grants: every interaction is a SECURITY DEFINER function call.
|
|
6
|
+
|
|
7
|
+
percolate_core.core connecting as the caller, configuration (always)
|
|
8
|
+
percolate_core.worker the step loop and @handler registry (always)
|
|
9
|
+
percolate_core.content the Content Server [content]
|
|
10
|
+
percolate_core.agentic the Agent Runtime [agent]
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"""p8agentic — the agent runtime speced in ``specs/agentic``.
|
|
2
|
+
|
|
3
|
+
Agents are rows in Postgres, tools are MCP/OpenAPI services reached over the
|
|
4
|
+
network, a turn's rows are written through RLS as an ordinary ``authenticated``
|
|
5
|
+
caller, AG-UI events stream out over ``NOTIFY``, and delegation is an MCP call
|
|
6
|
+
into this runtime's own agent gateway. ``dev/verify.py`` asserts all of that
|
|
7
|
+
against a live LLM.
|
|
8
|
+
|
|
9
|
+
Layout, mapped onto ``specs/agentic/brief.md``:
|
|
10
|
+
|
|
11
|
+
- ``contracts`` — the ontology as pydantic types (brief §7, ontology.md), plus
|
|
12
|
+
the two wire protocols (brief §5.1 item 2).
|
|
13
|
+
- ``db`` — a small asyncpg client and the ``Repository`` protocol every
|
|
14
|
+
read/write in this runtime goes through (brief §5: "through the same functions
|
|
15
|
+
and RLS any other client would go through, no privileged backdoor").
|
|
16
|
+
- ``tools`` — tool-server resolution and invocation: MCP is the canonical
|
|
17
|
+
shape, OpenAPI is an adapter onto it (brief §5.2). Zero built-in tools.
|
|
18
|
+
- ``runtime`` — the four components of brief §5.1, one module each, plus the
|
|
19
|
+
chained action (§4) and citation (§5.3) mechanisms.
|
|
20
|
+
- ``sequencer`` — the ordered steps of one agent turn, and the mapping from
|
|
21
|
+
``specs/agentic/sequence.md``'s phase gates onto demos that prove them.
|
|
22
|
+
- ``fakes`` — pydantic-ai fake models and an in-memory repository, so the
|
|
23
|
+
whole path above runs with no Postgres and no LLM credentials.
|
|
24
|
+
- ``cli`` — the operator surface. FastAPI wraps this later (brief §4:
|
|
25
|
+
streaming is the one piece that structurally can't be plain PostgREST).
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
__version__ = "0.0.1"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def build_fingerprint() -> str:
|
|
32
|
+
"""A short hash of this package's source on disk.
|
|
33
|
+
|
|
34
|
+
Exposed by every long-running surface (`/_build`) so a caller can tell
|
|
35
|
+
whether the process it is talking to is running the code in the working
|
|
36
|
+
tree. Three separate debugging sessions in this module's history began with
|
|
37
|
+
a confusing failure whose cause was a server started before the fix — a
|
|
38
|
+
stale process fails in the shape of a bug, and that is worth fifteen lines
|
|
39
|
+
to make impossible.
|
|
40
|
+
"""
|
|
41
|
+
import hashlib
|
|
42
|
+
from pathlib import Path
|
|
43
|
+
|
|
44
|
+
digest = hashlib.sha256()
|
|
45
|
+
for path in sorted(Path(__file__).parent.rglob("*.py")):
|
|
46
|
+
digest.update(path.name.encode())
|
|
47
|
+
digest.update(str(path.stat().st_mtime_ns).encode())
|
|
48
|
+
return digest.hexdigest()[:12]
|
|
@@ -0,0 +1,283 @@
|
|
|
1
|
+
"""The HTTP surface — one POST that streams, and one GET that replays.
|
|
2
|
+
|
|
3
|
+
``brief.md`` §4/§8 give FastAPI exactly one job here: "the SSE/WebSocket fan-out
|
|
4
|
+
(FastAPI, relaying Postgres NOTIFY)". CRUD is PostgREST's — `sessions_api`,
|
|
5
|
+
`runs_api`, `messages_api`, `agents_api` are already a REST API, and
|
|
6
|
+
hand-rolling a second one over the same tables is the thing this collection's
|
|
7
|
+
ground rules explicitly rule out. So this module is deliberately two endpoints:
|
|
8
|
+
|
|
9
|
+
POST /chat OpenAI-shaped in, AG-UI events out (SSE)
|
|
10
|
+
GET /sessions/{id}/events the same stream, for an observer or a reconnect
|
|
11
|
+
|
|
12
|
+
**Mountable, not just runnable.** The endpoints live on a ``router()`` and the
|
|
13
|
+
pool lives in a ``lifespan()``; ``build_app()`` is the twenty lines that put
|
|
14
|
+
them in an app of their own. A larger service — the ``p8`` image, which also
|
|
15
|
+
carries the Content Server — mounts the same two things under whatever prefix
|
|
16
|
+
it likes, and gets identical behaviour, because standalone mode is a *caller*
|
|
17
|
+
of the mountable pieces rather than a second implementation of them.
|
|
18
|
+
|
|
19
|
+
The second endpoint is not a convenience. It is the proof that the stream is a
|
|
20
|
+
*relay*, not a side effect of the request that started the run: a client that
|
|
21
|
+
drops mid-turn reconnects here and keeps receiving, and a second client can
|
|
22
|
+
watch a conversation it did not start. Both endpoints read from the same
|
|
23
|
+
``NOTIFY`` channel, so both see a delegation tree's parent and child events
|
|
24
|
+
interleaved, correlated by ``run_id``.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import asyncio
|
|
30
|
+
from collections.abc import AsyncIterator
|
|
31
|
+
from contextlib import asynccontextmanager
|
|
32
|
+
from uuid import UUID
|
|
33
|
+
|
|
34
|
+
from fastapi import APIRouter, FastAPI, HTTPException, Request
|
|
35
|
+
from fastapi.responses import StreamingResponse
|
|
36
|
+
|
|
37
|
+
from .contracts import ChatRequest, RequestContext, RunError, Session, to_sse
|
|
38
|
+
from .db import PgClient, PgRepository
|
|
39
|
+
from .gateway import caller_id
|
|
40
|
+
from .runtime import EventSink, channel, relay
|
|
41
|
+
from .sequencer import TurnSequencer
|
|
42
|
+
from .settings import Settings
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@asynccontextmanager
|
|
46
|
+
async def lifespan(app: FastAPI, settings: Settings | None = None) -> AsyncIterator[None]:
|
|
47
|
+
"""Own the connection pool for as long as the app is up.
|
|
48
|
+
|
|
49
|
+
A host app composes this into its own lifespan; ``build_app`` uses it
|
|
50
|
+
directly. Either way the pool is per *process*, not per request — the one
|
|
51
|
+
piece of state this module needs, kept on ``app.state`` so a mounted router
|
|
52
|
+
finds it exactly where a standalone one does.
|
|
53
|
+
"""
|
|
54
|
+
resolved = settings or Settings.from_env()
|
|
55
|
+
app.state.p8agentic_settings = resolved
|
|
56
|
+
app.state.p8agentic_client = PgClient(resolved)
|
|
57
|
+
await app.state.p8agentic_client.connect()
|
|
58
|
+
try:
|
|
59
|
+
yield
|
|
60
|
+
finally:
|
|
61
|
+
await app.state.p8agentic_client.close()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _state(request: Request) -> tuple[PgClient, Settings]:
|
|
65
|
+
"""The pool and settings, from wherever this router was mounted.
|
|
66
|
+
|
|
67
|
+
Namespaced attribute names, deliberately: in a mounted deployment
|
|
68
|
+
``app.state`` belongs to the host, and a module that claims ``state.client``
|
|
69
|
+
is a module that collides with the next one to do the same.
|
|
70
|
+
"""
|
|
71
|
+
app = request.app
|
|
72
|
+
client = getattr(app.state, "p8agentic_client", None)
|
|
73
|
+
if client is None:
|
|
74
|
+
raise HTTPException(
|
|
75
|
+
500,
|
|
76
|
+
"p8agentic is mounted but its lifespan is not installed — compose "
|
|
77
|
+
"p8agentic.api.lifespan into the host app's lifespan",
|
|
78
|
+
)
|
|
79
|
+
return client, app.state.p8agentic_settings
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def router() -> APIRouter:
|
|
83
|
+
"""The two endpoints, ready to mount under any prefix."""
|
|
84
|
+
api = APIRouter()
|
|
85
|
+
|
|
86
|
+
def _identity(request: Request) -> tuple[UUID | None, RequestContext]:
|
|
87
|
+
"""Who is calling, and what ambient context they brought.
|
|
88
|
+
|
|
89
|
+
Identity comes from the bearer token, verified with the same secret
|
|
90
|
+
PostgREST uses (``gateway.caller_id``) — never from a header a client
|
|
91
|
+
can set. Ambient context comes from ``X-P8-*`` (``brief.md`` §5.1 item
|
|
92
|
+
4) and is trusted only for what it is: context, not authorization.
|
|
93
|
+
"""
|
|
94
|
+
headers = dict(request.headers)
|
|
95
|
+
user_id = caller_id(headers)
|
|
96
|
+
context = RequestContext.from_headers(headers)
|
|
97
|
+
return user_id, context.model_copy(update={"user_id": user_id})
|
|
98
|
+
|
|
99
|
+
@api.get("/_build")
|
|
100
|
+
async def build() -> dict[str, str]:
|
|
101
|
+
"""Which source this process is running.
|
|
102
|
+
|
|
103
|
+
A stale server fails in the shape of a bug — a tool that "returned an
|
|
104
|
+
unexpected keyword argument", a run stuck in `running`. This makes
|
|
105
|
+
"you are talking to yesterday's code" a one-line answer instead of a
|
|
106
|
+
debugging session.
|
|
107
|
+
"""
|
|
108
|
+
from . import build_fingerprint
|
|
109
|
+
|
|
110
|
+
return {"build": build_fingerprint()}
|
|
111
|
+
|
|
112
|
+
@api.post("/chat")
|
|
113
|
+
async def chat(request: Request, body: ChatRequest) -> StreamingResponse:
|
|
114
|
+
"""Run one turn and stream it as AG-UI events.
|
|
115
|
+
|
|
116
|
+
``model`` names an **agent**, not an LLM — which is what lets an
|
|
117
|
+
off-the-shelf OpenAI-shaped client drive this runtime with no bespoke
|
|
118
|
+
fields (``brief.md`` §5.1 item 2). ``tools`` is accepted by the shape
|
|
119
|
+
and ignored: an agent's tool surface is whatever ``tool_servers`` its
|
|
120
|
+
row points at.
|
|
121
|
+
|
|
122
|
+
``stream: false`` is not supported and says so, rather than silently
|
|
123
|
+
buffering: this endpoint exists *because* streaming is the one thing
|
|
124
|
+
PostgREST cannot do. A caller who wants the finished answer reads
|
|
125
|
+
``messages_api``, which is a better answer than a fake non-streaming
|
|
126
|
+
mode.
|
|
127
|
+
"""
|
|
128
|
+
client, settings = _state(request)
|
|
129
|
+
user_id, context = _identity(request)
|
|
130
|
+
if user_id is None:
|
|
131
|
+
raise HTTPException(401, "a verified bearer token is required")
|
|
132
|
+
if not body.stream:
|
|
133
|
+
raise HTTPException(
|
|
134
|
+
400, "stream=false is not supported; read messages_api for the finished turn"
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
repo = PgRepository(client, user_id)
|
|
138
|
+
session = await _resume_or_open(repo, body, request)
|
|
139
|
+
|
|
140
|
+
return StreamingResponse(
|
|
141
|
+
_run_and_stream(client, repo, settings, body, context, session),
|
|
142
|
+
media_type="text/event-stream",
|
|
143
|
+
headers={
|
|
144
|
+
# The session is in the response headers as well as in every
|
|
145
|
+
# event, so a client can reconnect to GET /sessions/{id}/events
|
|
146
|
+
# without having to parse the stream it just lost.
|
|
147
|
+
"X-P8-Session-Id": str(session.id),
|
|
148
|
+
"Cache-Control": "no-cache",
|
|
149
|
+
"X-Accel-Buffering": "no",
|
|
150
|
+
},
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
@api.get("/sessions/{session_id}/events")
|
|
154
|
+
async def events(session_id: UUID, request: Request) -> StreamingResponse:
|
|
155
|
+
"""Relay a session's stream to an observer. Nothing is run."""
|
|
156
|
+
client, _settings = _state(request)
|
|
157
|
+
user_id, _ = _identity(request)
|
|
158
|
+
if user_id is None:
|
|
159
|
+
raise HTTPException(401, "a verified bearer token is required")
|
|
160
|
+
repo = PgRepository(client, user_id)
|
|
161
|
+
if await repo.session_messages(session_id) is None: # pragma: no cover
|
|
162
|
+
raise HTTPException(404, "no such session")
|
|
163
|
+
|
|
164
|
+
async def _observe() -> AsyncIterator[str]:
|
|
165
|
+
async for payload in relay(client, session_id):
|
|
166
|
+
yield to_sse(payload)
|
|
167
|
+
|
|
168
|
+
return StreamingResponse(_observe(), media_type="text/event-stream")
|
|
169
|
+
|
|
170
|
+
return api
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
DRAIN_SECONDS = 0.5
|
|
174
|
+
"""How long to keep reading the channel after the turn returns. Half a second is
|
|
175
|
+
far longer than local NOTIFY delivery and short enough that a client is not left
|
|
176
|
+
waiting on a stream that is already finished."""
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def build_app(settings: Settings | None = None) -> FastAPI:
|
|
180
|
+
"""Standalone mode: an app whose only job is to host this router.
|
|
181
|
+
|
|
182
|
+
Thin on purpose. Everything it does — install the lifespan, include the
|
|
183
|
+
router — a host app does the same way, so there is no behaviour that only
|
|
184
|
+
exists when this subsystem runs alone.
|
|
185
|
+
"""
|
|
186
|
+
resolved = settings or Settings.from_env()
|
|
187
|
+
|
|
188
|
+
@asynccontextmanager
|
|
189
|
+
async def _lifespan(app: FastAPI) -> AsyncIterator[None]:
|
|
190
|
+
async with lifespan(app, resolved):
|
|
191
|
+
yield
|
|
192
|
+
|
|
193
|
+
app = FastAPI(title="p8agentic", version="0.0.1", lifespan=_lifespan)
|
|
194
|
+
app.include_router(router())
|
|
195
|
+
return app
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
async def _resume_or_open(
|
|
199
|
+
repo: PgRepository, body: ChatRequest, request: Request
|
|
200
|
+
) -> Session:
|
|
201
|
+
"""Continue the named conversation, or start one.
|
|
202
|
+
|
|
203
|
+
The id may arrive in the body (`session_id`) **or** as the
|
|
204
|
+
`X-P8-Session-Id` header, and the header is the one that matters: the
|
|
205
|
+
inbound contract is the OpenAI chat shape (`brief.md` §5.1 item 2), which
|
|
206
|
+
has nowhere to put a thread id, so a client using an off-the-shelf SDK can
|
|
207
|
+
only reach for a header. It is the same name the response sends back, so
|
|
208
|
+
resuming is echoing what you were given.
|
|
209
|
+
|
|
210
|
+
An unknown or someone else's id fails **here**, with a status code, rather
|
|
211
|
+
than mid-stream: once the response has begun the status is already sent,
|
|
212
|
+
and "your session id was wrong" is not something a client should have to
|
|
213
|
+
parse out of an event.
|
|
214
|
+
"""
|
|
215
|
+
header = request.headers.get("x-p8-session-id")
|
|
216
|
+
session_id = body.session_id or (UUID(header) if header else None)
|
|
217
|
+
if session_id is None:
|
|
218
|
+
return await repo.create_session(Session())
|
|
219
|
+
|
|
220
|
+
session = await repo.get_session(session_id)
|
|
221
|
+
if session is None:
|
|
222
|
+
raise HTTPException(404, f"no session {session_id} — it does not exist, or is not yours")
|
|
223
|
+
return session
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
async def _run_and_stream(
|
|
227
|
+
client: PgClient,
|
|
228
|
+
repo: PgRepository,
|
|
229
|
+
settings: Settings,
|
|
230
|
+
body: ChatRequest,
|
|
231
|
+
context: RequestContext,
|
|
232
|
+
session: Session,
|
|
233
|
+
) -> AsyncIterator[str]:
|
|
234
|
+
"""Start listening, start the turn, forward until the turn is done.
|
|
235
|
+
|
|
236
|
+
The order matters: ``LISTEN`` is established *before* the run starts, or the
|
|
237
|
+
first events are published to nobody. And the events forwarded here come off
|
|
238
|
+
the channel, not out of the sequencer — which is why a delegated sub-run
|
|
239
|
+
executing in the gateway process appears in this stream at all.
|
|
240
|
+
"""
|
|
241
|
+
queue: asyncio.Queue = asyncio.Queue()
|
|
242
|
+
conn = await client.listen(channel(session.id), queue)
|
|
243
|
+
sequencer = TurnSequencer(repo, settings, sink=EventSink(repo))
|
|
244
|
+
turn = asyncio.create_task(
|
|
245
|
+
sequencer.run_turn(
|
|
246
|
+
body.model_copy(update={"session_id": session.id}), context
|
|
247
|
+
)
|
|
248
|
+
)
|
|
249
|
+
try:
|
|
250
|
+
drain_until: float | None = None
|
|
251
|
+
loop = asyncio.get_running_loop()
|
|
252
|
+
while True:
|
|
253
|
+
if turn.done():
|
|
254
|
+
# A NOTIFY is published on the *runtime's* connection and
|
|
255
|
+
# delivered on the *listener's*, so "the turn returned" does not
|
|
256
|
+
# mean "the events arrived". Closing on `done() and empty()`
|
|
257
|
+
# dropped whatever was still in flight — most often
|
|
258
|
+
# RUN_FINISHED, since it is emitted last.
|
|
259
|
+
if drain_until is None:
|
|
260
|
+
drain_until = loop.time() + DRAIN_SECONDS
|
|
261
|
+
if queue.empty() and loop.time() >= drain_until:
|
|
262
|
+
break
|
|
263
|
+
try:
|
|
264
|
+
payload = await asyncio.wait_for(queue.get(), timeout=0.05)
|
|
265
|
+
except TimeoutError:
|
|
266
|
+
continue
|
|
267
|
+
yield to_sse(payload)
|
|
268
|
+
|
|
269
|
+
if turn.exception() is not None:
|
|
270
|
+
error = turn.exception()
|
|
271
|
+
yield to_sse(
|
|
272
|
+
RunError(thread_id=session.id, message=f"{type(error).__name__}: {error}")
|
|
273
|
+
)
|
|
274
|
+
finally:
|
|
275
|
+
# A client that disconnects mid-turn does not cancel the run — the rows
|
|
276
|
+
# still land, which is the point of rows being the record. But nobody is
|
|
277
|
+
# left to await the task, so retrieve its exception rather than let it
|
|
278
|
+
# surface as "Task exception was never retrieved" at some later GC.
|
|
279
|
+
if not turn.done():
|
|
280
|
+
turn.add_done_callback(lambda t: t.exception() if not t.cancelled() else None)
|
|
281
|
+
await conn.close()
|
|
282
|
+
|
|
283
|
+
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
"""Authoring an agent: one format, and it is a JSON Schema document.
|
|
2
|
+
|
|
3
|
+
``brief.md`` §6 says an agent can be authored in code, is "fully serializable
|
|
4
|
+
the moment it's finished", and is applied as a *migration* that upserts one row.
|
|
5
|
+
This module is both ends of that, and there is deliberately only one shape.
|
|
6
|
+
|
|
7
|
+
An agent **is** the flat schema document ``model_json_schema()`` produces, with
|
|
8
|
+
agent config beside the schema fields where ``json_schema_extra`` hoists it. The
|
|
9
|
+
convention is ``../p8k8``'s (``p8/agentic/agent_schema.py``) and it is worth
|
|
10
|
+
keeping precisely for what it rules out: there is **no ``prompt`` key**. The
|
|
11
|
+
system prompt is the schema's ``description`` — the *docstring of the model
|
|
12
|
+
class* — and the structured output is its ``properties``. So authoring in Python
|
|
13
|
+
and authoring in YAML produce the same document, byte for byte:
|
|
14
|
+
|
|
15
|
+
class Support(BaseModel):
|
|
16
|
+
'''You triage incident questions for the support desk.'''
|
|
17
|
+
model_config = ConfigDict(json_schema_extra={
|
|
18
|
+
"name": "support",
|
|
19
|
+
"model": "openai:gpt-4.1-mini",
|
|
20
|
+
"tools": [{"server": "p8-rag", "tools": ["post_rpc_search_chunks"]}],
|
|
21
|
+
})
|
|
22
|
+
claim: str
|
|
23
|
+
|
|
24
|
+
Support.model_json_schema() == yaml.safe_load(open("support.yaml"))
|
|
25
|
+
|
|
26
|
+
That is §6's "authored in code, deployed as data" as a fact rather than a
|
|
27
|
+
promise — ``from_model_class`` is one line because the pydantic model *already
|
|
28
|
+
is* the interchange format. It is a format like any other, but one that carries
|
|
29
|
+
more meaning: every tool that reads JSON Schema reads an agent, and the
|
|
30
|
+
description a human writes is the description a model receives.
|
|
31
|
+
|
|
32
|
+
A second, prose-shaped format lived here for a while, on the argument that a
|
|
33
|
+
reading form and a committing form answer different questions. It is gone: its
|
|
34
|
+
top-level key was ``prompt``, which is the one thing this convention says does
|
|
35
|
+
not exist, and a reading form that contradicts the deployed form teaches the
|
|
36
|
+
wrong shape. YAML with block scalars reads fine.
|
|
37
|
+
|
|
38
|
+
One deliberate divergence from p8k8, noted rather than silently inherited: it
|
|
39
|
+
distinguishes ``structured_output: true`` (properties are the output contract)
|
|
40
|
+
from ``false`` (properties are "thinking aides" that shape the model's reasoning
|
|
41
|
+
without being returned). This runtime has one meaning for ``properties`` — the
|
|
42
|
+
output contract, which is what makes ``chained_action`` reachable (``brief.md``
|
|
43
|
+
§2). Thinking aides are a real idea and simply are not built here.
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
from __future__ import annotations
|
|
47
|
+
|
|
48
|
+
from pathlib import Path
|
|
49
|
+
from typing import Any
|
|
50
|
+
|
|
51
|
+
import yaml
|
|
52
|
+
|
|
53
|
+
from .contracts import AgentSpec, ChainedAction, ContextPolicy, ToolRef
|
|
54
|
+
|
|
55
|
+
# ---------------------------------------------------------------------------
|
|
56
|
+
# JSON Schema — the deployable form
|
|
57
|
+
# ---------------------------------------------------------------------------
|
|
58
|
+
|
|
59
|
+
def to_json_schema(agent: AgentSpec) -> dict[str, Any]:
|
|
60
|
+
"""An ``agents`` row as the flat schema document ``model_json_schema`` emits.
|
|
61
|
+
|
|
62
|
+
Flat, not nested under ``json_schema_extra``: pydantic hoists that wrapper
|
|
63
|
+
to the top level when it builds the schema, so a nested document here would
|
|
64
|
+
not match what a code-authored agent produces — and matching is the whole
|
|
65
|
+
point of the convention.
|
|
66
|
+
"""
|
|
67
|
+
doc: dict[str, Any] = {"type": "object", "name": agent.name}
|
|
68
|
+
if agent.system_prompt:
|
|
69
|
+
doc["description"] = agent.system_prompt
|
|
70
|
+
if agent.model:
|
|
71
|
+
doc["model"] = agent.model
|
|
72
|
+
if agent.tools:
|
|
73
|
+
doc["tools"] = [t.model_dump(exclude_none=True) for t in agent.tools]
|
|
74
|
+
|
|
75
|
+
schema = agent.structured_output_schema or {}
|
|
76
|
+
doc["properties"] = schema.get("properties", {})
|
|
77
|
+
if schema.get("required"):
|
|
78
|
+
doc["required"] = schema["required"]
|
|
79
|
+
|
|
80
|
+
if agent.chained_action:
|
|
81
|
+
doc["chained_action"] = agent.chained_action.model_dump()
|
|
82
|
+
if agent.summarizer_agent_id:
|
|
83
|
+
doc["summarizer_agent_id"] = str(agent.summarizer_agent_id)
|
|
84
|
+
policy = agent.context_policy.model_dump(exclude_none=True)
|
|
85
|
+
if policy:
|
|
86
|
+
doc["context_policy"] = policy
|
|
87
|
+
return doc
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def from_json_schema(doc: dict[str, Any]) -> AgentSpec:
|
|
91
|
+
"""The inverse: a schema document back into the row it describes.
|
|
92
|
+
|
|
93
|
+
``description`` is the system prompt and ``properties`` is the structured
|
|
94
|
+
output — the two mappings that make "no ``prompt`` key" work. Everything
|
|
95
|
+
else is agent config sitting at the same level, exactly where
|
|
96
|
+
``json_schema_extra`` puts it.
|
|
97
|
+
"""
|
|
98
|
+
properties = doc.get("properties") or {}
|
|
99
|
+
output = (
|
|
100
|
+
{
|
|
101
|
+
"type": "object",
|
|
102
|
+
"properties": properties,
|
|
103
|
+
"required": doc.get("required", []),
|
|
104
|
+
}
|
|
105
|
+
if properties
|
|
106
|
+
else None
|
|
107
|
+
)
|
|
108
|
+
return AgentSpec(
|
|
109
|
+
name=doc.get("name") or doc.get("title", ""),
|
|
110
|
+
system_prompt=doc.get("description", ""),
|
|
111
|
+
model=doc.get("model"),
|
|
112
|
+
tools=[ToolRef(**t) for t in doc.get("tools", [])],
|
|
113
|
+
structured_output_schema=output,
|
|
114
|
+
chained_action=(
|
|
115
|
+
ChainedAction(**doc["chained_action"]) if doc.get("chained_action") else None
|
|
116
|
+
),
|
|
117
|
+
summarizer_agent_id=doc.get("summarizer_agent_id"),
|
|
118
|
+
context_policy=ContextPolicy(**doc.get("context_policy", {})),
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def from_model_class(model_cls: type) -> AgentSpec:
|
|
123
|
+
"""A code-authored agent, deployed as data — and it really is three lines.
|
|
124
|
+
|
|
125
|
+
The docstring is the prompt, the fields are the output, ``json_schema_extra``
|
|
126
|
+
is the config. Nothing is extracted or inferred; pydantic already produces
|
|
127
|
+
the interchange format.
|
|
128
|
+
"""
|
|
129
|
+
return from_json_schema(model_cls.model_json_schema())
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
class _BlockDumper(yaml.SafeDumper):
|
|
133
|
+
"""Emit multi-line strings as block scalars.
|
|
134
|
+
|
|
135
|
+
Cosmetic, and worth the six lines: the description *is* the system prompt —
|
|
136
|
+
the part of an agent a human actually edits — and PyYAML's default folds it
|
|
137
|
+
into quoted lines with blank rows between them, unreadable in a file whose
|
|
138
|
+
whole purpose is to be read in a diff.
|
|
139
|
+
"""
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _block_str(dumper: yaml.SafeDumper, value: str) -> Any:
|
|
143
|
+
return dumper.represent_scalar(
|
|
144
|
+
"tag:yaml.org,2002:str", value, style="|" if "\n" in value else None
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
_BlockDumper.add_representer(str, _block_str)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def to_yaml(agent: AgentSpec) -> str:
|
|
152
|
+
"""The schema document as YAML — exactly what ``agent push`` accepts back."""
|
|
153
|
+
return yaml.dump(
|
|
154
|
+
to_json_schema(agent), Dumper=_BlockDumper, sort_keys=False, width=88
|
|
155
|
+
)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def load(path: Path | str, text: str | None = None) -> AgentSpec:
|
|
159
|
+
"""Read an agent definition from a file or a string.
|
|
160
|
+
|
|
161
|
+
YAML and JSON are the same document to ``yaml.safe_load``, so there is
|
|
162
|
+
nothing to dispatch on and no suffix to interpret.
|
|
163
|
+
"""
|
|
164
|
+
body = text if text is not None else Path(path).read_text()
|
|
165
|
+
return from_json_schema(yaml.safe_load(body))
|