@pylonsync/functions 0.4.22 → 0.4.23

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,13 @@
1
+ /**
2
+ * Internal functions backing the `agent()` loop. Registered by the
3
+ * runtime loader only when the app declares at least one agent.
4
+ *
5
+ * Both run under the CALLING user's auth (internal fns inherit the
6
+ * wrapping handler's auth — they are not admin), so every op verifies
7
+ * run ownership itself and stamps identity from `ctx.auth`, never from
8
+ * args. `internal: true` keeps them off the HTTP surface; the
9
+ * `__pylon_` prefix keeps their timeout out of the runner's
10
+ * wedge-probe budget.
11
+ */
12
+ import type { FnDefinition } from "./types";
13
+ export declare function registerAgentInternals(registry: Map<string, FnDefinition>): void;
@@ -0,0 +1,103 @@
1
+ /**
2
+ * `agent()` — a define-type for LLM agents with durable run state.
3
+ *
4
+ * ```ts
5
+ * // functions/researcher.ts
6
+ * export default agent({
7
+ * system: "You are a research assistant.",
8
+ * tools: {
9
+ * searchDocs: {
10
+ * description: "Search the document library",
11
+ * args: { query: v.string() },
12
+ * handler: async (ctx, { query }) => ctx.runQuery("findSimilar", { query }),
13
+ * },
14
+ * },
15
+ * });
16
+ * ```
17
+ *
18
+ * An agent compiles to an ordinary streaming ACTION (named after its
19
+ * file, callable via `streamFn("researcher", { input })`), whose
20
+ * handler runs the tool loop:
21
+ *
22
+ * 1. Create an `AgentRun` row (or load one when `runId` is passed —
23
+ * that's how a conversation continues) and append the user's
24
+ * input as an `AgentMessage`.
25
+ * 2. `ctx.llm.stream` with the declared tools; text deltas flow to
26
+ * `ctx.stream` (resumable — the run row records the stream id).
27
+ * 3. On `stop_reason === "tool_use"`: validate each tool call's
28
+ * input against its validators, run the handler, record the
29
+ * call + result as messages, loop.
30
+ * 4. Terminal: run marked completed/failed.
31
+ *
32
+ * `AgentRun` and `AgentMessage` are real synced entities (injected
33
+ * into the manifest by the SDK when any agent exists), owner-scoped by
34
+ * policy — so `db.useQuery("AgentMessage", { where: { runId } })`
35
+ * shows the transcript live on every one of the user's devices,
36
+ * including tool calls, with zero extra plumbing.
37
+ */
38
+ import type { ActionCtx, AnyValidator, FnDefinition, ValidatorSchema } from "./types";
39
+ /** One tool an agent can call. */
40
+ export interface AgentTool {
41
+ /** Shown to the model — say when to use the tool, not how it works. */
42
+ description: string;
43
+ /** Argument validators (same `v.*` schema as functions). The model's
44
+ * JSON is validated before the handler runs; invalid input becomes
45
+ * a tool_result error the model can react to. Omit for no-arg tools. */
46
+ args?: ValidatorSchema;
47
+ /** Runs with the agent action's ctx (runQuery/runMutation, llm,
48
+ * email, …). The return value is JSON-serialized into the
49
+ * tool_result the model sees. Throwing marks the result is_error —
50
+ * the model sees the message and can recover. */
51
+ handler: (ctx: ActionCtx, input: Record<string, unknown>) => unknown;
52
+ }
53
+ export interface AgentDefinition {
54
+ /** System prompt. A function receives the ctx + call args for
55
+ * per-user prompts. */
56
+ system?: string | ((ctx: ActionCtx, args: AgentCallArgs) => string);
57
+ tools?: Record<string, AgentTool>;
58
+ /** Model override (subject to the server's allowlist). */
59
+ model?: string;
60
+ /** Max model↔tool round-trips per invocation (default 16). Hitting
61
+ * the cap fails the run rather than looping forever. */
62
+ maxSteps?: number;
63
+ /** max_tokens per completion (default: server default). */
64
+ maxTokens?: number;
65
+ /** Auth gate for the action (default "user" — runs are owner-scoped,
66
+ * so an authenticated caller is the natural default). */
67
+ auth?: "user" | "admin";
68
+ /** Idle timeout in seconds (default 600; activity extends it). */
69
+ timeout?: number;
70
+ }
71
+ /** The synthesized action's args. */
72
+ export interface AgentCallArgs {
73
+ /** The user's message for this turn. */
74
+ input: string;
75
+ /** Continue an existing run (must belong to the caller and this
76
+ * agent). Omit to start a new run. */
77
+ runId?: string;
78
+ /** Optional display title, stored on new runs. */
79
+ title?: string;
80
+ }
81
+ /** What the agent action resolves with (also the `event: result`
82
+ * payload on the SSE stream). */
83
+ export interface AgentResult {
84
+ runId: string;
85
+ /** Concatenated text of the final assistant message. */
86
+ text: string;
87
+ /** Round-trips consumed. */
88
+ steps: number;
89
+ usage: {
90
+ input_tokens: number;
91
+ output_tokens: number;
92
+ };
93
+ }
94
+ /** Convert one `v.*` validator to a JSON-Schema fragment. */
95
+ export declare function validatorToJsonSchema(val: AnyValidator): Record<string, unknown>;
96
+ /** Convert a validator schema (a tool's `args`) to a JSON-Schema
97
+ * object with `required` derived from non-optional fields. */
98
+ export declare function validatorSchemaToJsonSchema(schema: ValidatorSchema): Record<string, unknown>;
99
+ /** Marker so the SDK's discoverFunctions can detect agents and inject
100
+ * the AgentRun/AgentMessage entities into the manifest. */
101
+ export declare const AGENT_MARKER = "__pylonAgent";
102
+ export declare function agent(def: AgentDefinition): FnDefinition<AgentCallArgs, AgentResult>;
103
+ export declare function isAgentDefinition(value: unknown): boolean;
package/dist/index.d.ts CHANGED
@@ -20,6 +20,8 @@
20
20
  export { query, mutation, action } from "./define";
21
21
  export { v } from "./validators";
22
22
  export { workflow } from "./workflows";
23
+ export { agent, isAgentDefinition, validatorToJsonSchema, validatorSchemaToJsonSchema, } from "./agent";
24
+ export type { AgentDefinition, AgentTool, AgentCallArgs, AgentResult, } from "./agent";
23
25
  export type { WorkflowDefinition, WorkflowRun, WorkflowRunRequest, WorkflowRunnerResponse, WorkflowStepResult, } from "./workflows";
24
26
  export { resetDb, installTestIsolation } from "./testing";
25
27
  export { slugifyName, availableSlug } from "./slugify";
package/dist/types.d.ts CHANGED
@@ -266,6 +266,15 @@ export interface DbWriter extends DbReader {
266
266
  * client disconnects; it just keeps writing.
267
267
  */
268
268
  export interface Stream {
269
+ /**
270
+ * The host-assigned resumable-stream id for THIS call, present when
271
+ * the caller connected over SSE (`streamFn`). Persist it — e.g. on a
272
+ * run row — and any device can attach to the live stream (or fetch
273
+ * the buffered replay + final result) via `resumeStream(id)` /
274
+ * `GET /api/fn-streams/<id>`. Absent for non-streaming invocations
275
+ * (plain JSON calls, scheduled jobs).
276
+ */
277
+ readonly id?: string;
269
278
  /** Write a text chunk to the client (SSE). */
270
279
  write(data: string): void;
271
280
  /** Write a typed SSE event (`event: <name>` framing on the wire). */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@pylonsync/functions",
3
- "version": "0.4.22",
3
+ "version": "0.4.23",
4
4
  "description": "TypeScript function runtime for pylon — defines server-side queries, mutations, and actions.",
5
5
  "type": "module",
6
6
  "main": "src/index.ts",
@@ -0,0 +1,161 @@
1
+ /**
2
+ * Internal functions backing the `agent()` loop. Registered by the
3
+ * runtime loader only when the app declares at least one agent.
4
+ *
5
+ * Both run under the CALLING user's auth (internal fns inherit the
6
+ * wrapping handler's auth — they are not admin), so every op verifies
7
+ * run ownership itself and stamps identity from `ctx.auth`, never from
8
+ * args. `internal: true` keeps them off the HTTP surface; the
9
+ * `__pylon_` prefix keeps their timeout out of the runner's
10
+ * wedge-probe budget.
11
+ */
12
+
13
+ import type { FnDefinition, MutationCtx, QueryCtx } from "./types";
14
+
15
+ interface RunRow {
16
+ id: string;
17
+ agent: string;
18
+ status: string;
19
+ userId: string | null;
20
+ [key: string]: unknown;
21
+ }
22
+
23
+ function nowIso(): string {
24
+ return new Date().toISOString();
25
+ }
26
+
27
+ /** Ownership fence: admins pass; otherwise the run must belong to the
28
+ * caller. Missing and foreign runs are indistinguishable (NOT_FOUND). */
29
+ function requireOwnedRun(
30
+ ctx: QueryCtx | MutationCtx,
31
+ run: Record<string, unknown> | null,
32
+ ): RunRow {
33
+ const owned =
34
+ run &&
35
+ (ctx.auth.isAdmin ||
36
+ (typeof run.userId === "string" && run.userId === ctx.auth.userId));
37
+ if (!owned) {
38
+ const err = new Error("Run not found");
39
+ (err as { code?: string }).code = "RUN_NOT_FOUND";
40
+ throw err;
41
+ }
42
+ return run as unknown as RunRow;
43
+ }
44
+
45
+ export function registerAgentInternals(
46
+ registry: Map<string, FnDefinition>,
47
+ ): void {
48
+ registry.set("__pylon_agent_read", {
49
+ type: "query",
50
+ internal: true,
51
+ auth: "user",
52
+ handler: async (ctx: QueryCtx, args: Record<string, unknown>) => {
53
+ const runId = String(args.runId ?? "");
54
+ const run = await ctx.db.get("AgentRun", runId);
55
+ const owned = requireOwnedRun(ctx, run);
56
+ const messages = await ctx.db.query("AgentMessage", {
57
+ runId,
58
+ $order: { seq: "asc" },
59
+ });
60
+ return { run: owned, messages };
61
+ },
62
+ } as unknown as FnDefinition);
63
+
64
+ registry.set("__pylon_agent_write", {
65
+ type: "mutation",
66
+ internal: true,
67
+ auth: "user",
68
+ handler: async (ctx: MutationCtx, args: Record<string, unknown>) => {
69
+ const op = String(args.op ?? "");
70
+ switch (op) {
71
+ case "createRun": {
72
+ // Owner-scoped or nothing: a null-owner row would be
73
+ // invisible to everyone (policies guard with
74
+ // `auth.userId != null`), and before that guard existed it
75
+ // was readable by ANONYMOUS callers (null == null evaluates
76
+ // true in the policy engine). Refuse instead of storing an
77
+ // orphan — cron/system invocations must call the agent under
78
+ // a service user's auth.
79
+ const userId = ctx.auth.userId;
80
+ if (typeof userId !== "string" || userId === "") {
81
+ const err = new Error(
82
+ "Agent runs require a signed-in user; invoke the agent under a service user for system/cron calls",
83
+ );
84
+ (err as { code?: string }).code = "AGENT_REQUIRES_USER";
85
+ throw err;
86
+ }
87
+ const id = await ctx.db.insert("AgentRun", {
88
+ agent: String(args.agent ?? "agent"),
89
+ status: "idle",
90
+ userId,
91
+ title: args.title ?? null,
92
+ streamId: null,
93
+ error: null,
94
+ createdAt: nowIso(),
95
+ updatedAt: nowIso(),
96
+ });
97
+ return { id };
98
+ }
99
+ case "appendMessage": {
100
+ const runId = String(args.runId ?? "");
101
+ const run = requireOwnedRun(
102
+ ctx,
103
+ await ctx.db.get("AgentRun", runId),
104
+ );
105
+ // Single-writer per run (the loop rejects concurrent turns),
106
+ // so a read-then-insert seq is race-free in practice; the
107
+ // mutation's transaction covers the rest.
108
+ const last = await ctx.db.query("AgentMessage", {
109
+ runId,
110
+ $order: { seq: "desc" },
111
+ $limit: 1,
112
+ });
113
+ const seq =
114
+ last.length > 0 ? (Number(last[0].seq) || 0) + 1 : 1;
115
+ const id = await ctx.db.insert("AgentMessage", {
116
+ runId,
117
+ userId: run.userId,
118
+ seq,
119
+ role: String(args.role ?? "user"),
120
+ content: args.content ?? null,
121
+ createdAt: nowIso(),
122
+ });
123
+ await ctx.db.update("AgentRun", runId, { updatedAt: nowIso() });
124
+ return { id, seq };
125
+ }
126
+ case "setStatus": {
127
+ const runId = String(args.runId ?? "");
128
+ const run = requireOwnedRun(ctx, await ctx.db.get("AgentRun", runId));
129
+ // The loop's pre-flight RUN_BUSY check races with itself
130
+ // across concurrent turns; this in-transaction guard is the
131
+ // authoritative claim. A "running" row older than staleMs is
132
+ // a dead generation (process crash) and may be taken over.
133
+ if (args.guardNotRunning === true && run.status === "running") {
134
+ const staleMs = Number(args.staleMs) || 0;
135
+ const updatedAt = Date.parse(String(run.updatedAt ?? "")) || 0;
136
+ if (Date.now() - updatedAt < staleMs) {
137
+ const err = new Error(
138
+ "This run is already generating — wait for it to finish",
139
+ );
140
+ (err as { code?: string }).code = "RUN_BUSY";
141
+ throw err;
142
+ }
143
+ }
144
+ const patch: Record<string, unknown> = {
145
+ status: String(args.status ?? "idle"),
146
+ updatedAt: nowIso(),
147
+ };
148
+ if (args.error !== undefined) patch.error = args.error;
149
+ if (args.streamId !== undefined) patch.streamId = args.streamId;
150
+ const updated = await ctx.db.update("AgentRun", runId, patch);
151
+ return { updated };
152
+ }
153
+ default: {
154
+ const err = new Error(`Unknown agent write op "${op}"`);
155
+ (err as { code?: string }).code = "INVALID_OP";
156
+ throw err;
157
+ }
158
+ }
159
+ },
160
+ } as unknown as FnDefinition);
161
+ }
@@ -0,0 +1,472 @@
1
+ /**
2
+ * agent() unit tests: the validator→JSON-Schema converter and the tool
3
+ * loop driven against a scripted mock ctx (no Bun runner, no LLM).
4
+ */
5
+ import { describe, expect, test } from "bun:test";
6
+ import {
7
+ agent,
8
+ isAgentDefinition,
9
+ validatorSchemaToJsonSchema,
10
+ } from "./agent";
11
+ import { v } from "./validators";
12
+ import type {
13
+ ActionCtx,
14
+ LlmCompleteRequest,
15
+ LlmCompleteResponse,
16
+ LlmStreamEvent,
17
+ } from "./types";
18
+
19
+ // ---------------------------------------------------------------------------
20
+ // validator → JSON Schema
21
+ // ---------------------------------------------------------------------------
22
+
23
+ describe("validatorSchemaToJsonSchema", () => {
24
+ test("maps every validator shape losslessly", () => {
25
+ const schema = validatorSchemaToJsonSchema({
26
+ q: v.string(),
27
+ n: v.int(),
28
+ score: v.number(),
29
+ on: v.boolean(),
30
+ target: v.id("Doc"),
31
+ tags: v.array(v.string()),
32
+ opts: v.object({ deep: v.optional(v.boolean()) }),
33
+ kind: v.union(v.literal("a"), v.literal("b")),
34
+ blob: v.json(),
35
+ maybe: v.optional(v.string()),
36
+ });
37
+ expect(schema).toEqual({
38
+ type: "object",
39
+ properties: {
40
+ q: { type: "string" },
41
+ n: { type: "integer" },
42
+ score: { type: "number" },
43
+ on: { type: "boolean" },
44
+ target: { type: "string" },
45
+ tags: { type: "array", items: { type: "string" } },
46
+ opts: {
47
+ type: "object",
48
+ properties: { deep: { type: "boolean" } },
49
+ },
50
+ kind: { anyOf: [{ const: "a" }, { const: "b" }] },
51
+ blob: {},
52
+ maybe: { type: "string" },
53
+ },
54
+ required: ["q", "n", "score", "on", "target", "tags", "opts", "kind", "blob"],
55
+ });
56
+ });
57
+
58
+ test("empty schema produces an object schema with no required", () => {
59
+ expect(validatorSchemaToJsonSchema({})).toEqual({
60
+ type: "object",
61
+ properties: {},
62
+ });
63
+ });
64
+ });
65
+
66
+ // ---------------------------------------------------------------------------
67
+ // Mock ctx harness
68
+ // ---------------------------------------------------------------------------
69
+
70
+ interface WriteOp {
71
+ op: string;
72
+ [key: string]: unknown;
73
+ }
74
+
75
+ interface ReadOverride {
76
+ run?: Record<string, unknown>;
77
+ messages?: Array<Record<string, unknown>>;
78
+ }
79
+
80
+ function mockCtx(script: LlmCompleteResponse[], read?: ReadOverride) {
81
+ const writes: WriteOp[] = [];
82
+ const streamed: Array<{ event?: string; data: string }> = [];
83
+ let call = 0;
84
+ const ctx = {
85
+ auth: { userId: "u1", isAdmin: false, tenantId: null, roles: [] },
86
+ stream: {
87
+ id: "st_test",
88
+ write(data: string) {
89
+ streamed.push({ data });
90
+ },
91
+ writeEvent(event: string, data: string) {
92
+ streamed.push({ event, data });
93
+ },
94
+ },
95
+ llm: {
96
+ async stream(
97
+ _req: LlmCompleteRequest,
98
+ onEvent: (e: LlmStreamEvent) => void,
99
+ ): Promise<LlmCompleteResponse> {
100
+ const res = script[call];
101
+ call += 1;
102
+ if (!res) throw new Error("mock LLM script exhausted");
103
+ for (const block of res.content) {
104
+ if (block.type === "text") {
105
+ onEvent({ type: "text_delta", text: block.text });
106
+ }
107
+ }
108
+ onEvent({ type: "done", stop_reason: res.stop_reason, usage: res.usage });
109
+ return res;
110
+ },
111
+ async complete() {
112
+ throw new Error("not used");
113
+ },
114
+ async embed() {
115
+ throw new Error("not used");
116
+ },
117
+ },
118
+ async runQuery(name: string, args: Record<string, unknown>) {
119
+ if (name === "__pylon_agent_read") {
120
+ return {
121
+ run: {
122
+ id: String(args.runId),
123
+ agent: "helper",
124
+ status: "completed",
125
+ ...read?.run,
126
+ },
127
+ messages: read?.messages ?? [
128
+ { id: "m1", seq: 1, role: "user", content: "earlier question" },
129
+ {
130
+ id: "m2",
131
+ seq: 2,
132
+ role: "assistant",
133
+ content: [{ type: "text", text: "earlier answer" }],
134
+ },
135
+ ],
136
+ };
137
+ }
138
+ throw new Error(`unexpected query ${name}`);
139
+ },
140
+ async runMutation(name: string, args: Record<string, unknown>) {
141
+ if (name !== "__pylon_agent_write") throw new Error(`unexpected ${name}`);
142
+ writes.push(args as WriteOp);
143
+ if (args.op === "createRun") return { id: "run_1" };
144
+ if (args.op === "appendMessage") return { id: "m", seq: writes.length };
145
+ return { updated: true };
146
+ },
147
+ error(code: string, message: string) {
148
+ const err = new Error(message);
149
+ (err as { code?: string }).code = code;
150
+ return err;
151
+ },
152
+ };
153
+ return { ctx: ctx as unknown as ActionCtx, writes, streamed };
154
+ }
155
+
156
+ const done = (text: string): LlmCompleteResponse => ({
157
+ model: "m",
158
+ content: [{ type: "text", text }],
159
+ stop_reason: "end_turn",
160
+ usage: { input_tokens: 10, output_tokens: 5 },
161
+ });
162
+
163
+ const wantsTool = (
164
+ name: string,
165
+ input: Record<string, unknown>,
166
+ ): LlmCompleteResponse => ({
167
+ model: "m",
168
+ content: [
169
+ { type: "text", text: "Let me check." },
170
+ { type: "tool_use", id: "tu_1", name, input },
171
+ ],
172
+ stop_reason: "tool_use",
173
+ usage: { input_tokens: 20, output_tokens: 8 },
174
+ });
175
+
176
+ // ---------------------------------------------------------------------------
177
+ // The loop
178
+ // ---------------------------------------------------------------------------
179
+
180
+ describe("agent loop", () => {
181
+ test("agent() produces a tagged action with the standard args", () => {
182
+ const def = agent({ system: "hi" });
183
+ expect((def as unknown as Record<string, unknown>).type).toBe("action");
184
+ expect((def as unknown as Record<string, unknown>).auth).toBe("user");
185
+ expect(isAgentDefinition(def)).toBe(true);
186
+ expect(isAgentDefinition(agent({}))).toBe(true);
187
+ expect(isAgentDefinition({ type: "action" })).toBe(false);
188
+ });
189
+
190
+ test("new run: tool round-trip, transcript order, terminal status", async () => {
191
+ const toolCalls: unknown[] = [];
192
+ const def = agent({
193
+ system: "You are helpful.",
194
+ tools: {
195
+ lookup: {
196
+ description: "Look something up",
197
+ args: { q: v.string() },
198
+ handler: async (_ctx, input) => {
199
+ toolCalls.push(input);
200
+ return { hits: 3 };
201
+ },
202
+ },
203
+ },
204
+ });
205
+ const { ctx, writes, streamed } = mockCtx([
206
+ wantsTool("lookup", { q: "pylon" }),
207
+ done("Found 3 hits."),
208
+ ]);
209
+
210
+ const result = await (def.handler as unknown as (
211
+ c: ActionCtx,
212
+ a: Record<string, unknown>,
213
+ ) => Promise<Record<string, unknown>>)(ctx, {
214
+ input: "search pylon",
215
+ __agentName: "helper",
216
+ });
217
+
218
+ expect(result.runId).toBe("run_1");
219
+ expect(result.text).toBe("Found 3 hits.");
220
+ expect(result.steps).toBe(2);
221
+ expect(result.usage).toEqual({ input_tokens: 30, output_tokens: 13 });
222
+ expect(toolCalls).toEqual([{ q: "pylon" }]);
223
+
224
+ // Write sequence: createRun, CLAIM (running, guarded, +streamId),
225
+ // user msg, assistant(tool_use), tool_result, assistant(final),
226
+ // completed. The claim precedes every message write so a losing
227
+ // racer never persists a stray turn.
228
+ expect(writes.map((w) => `${w.op}:${w.role ?? w.status ?? ""}`)).toEqual([
229
+ "createRun:",
230
+ "setStatus:running",
231
+ "appendMessage:user",
232
+ "appendMessage:assistant",
233
+ "appendMessage:user", // tool_result batch
234
+ "appendMessage:assistant",
235
+ "setStatus:completed",
236
+ ]);
237
+ const running = writes.find((w) => w.status === "running")!;
238
+ expect(running.streamId).toBe("st_test");
239
+ expect(running.guardNotRunning).toBe(true);
240
+ expect(Number(running.staleMs)).toBeGreaterThan(0);
241
+ const toolResultMsg = writes[4];
242
+ const blocks = toolResultMsg.content as Array<Record<string, unknown>>;
243
+ expect(blocks[0].type).toBe("tool_result");
244
+ expect(blocks[0].tool_use_id).toBe("tu_1");
245
+ expect(JSON.parse(String(blocks[0].content))).toEqual({ hits: 3 });
246
+
247
+ // Streaming: text deltas + a typed tool event.
248
+ expect(streamed.some((s) => s.data === "Let me check.")).toBe(true);
249
+ const toolEvent = streamed.find((s) => s.event === "tool")!;
250
+ expect(JSON.parse(toolEvent.data)).toEqual({
251
+ name: "lookup",
252
+ input: { q: "pylon" },
253
+ isError: false,
254
+ });
255
+ });
256
+
257
+ test("invalid tool input and unknown tools become is_error results", async () => {
258
+ const def = agent({
259
+ tools: {
260
+ strict: {
261
+ description: "needs a number",
262
+ args: { n: v.int() },
263
+ handler: async () => "never reached",
264
+ },
265
+ },
266
+ });
267
+ const { ctx, writes } = mockCtx([
268
+ {
269
+ model: "m",
270
+ content: [
271
+ { type: "tool_use", id: "tu_a", name: "strict", input: { n: "NaN" } },
272
+ { type: "tool_use", id: "tu_b", name: "ghost", input: {} },
273
+ ],
274
+ stop_reason: "tool_use",
275
+ usage: { input_tokens: 1, output_tokens: 1 },
276
+ },
277
+ done("Recovered."),
278
+ ]);
279
+
280
+ const result = await (def.handler as unknown as (
281
+ c: ActionCtx,
282
+ a: Record<string, unknown>,
283
+ ) => Promise<Record<string, unknown>>)(ctx, {
284
+ input: "go",
285
+ __agentName: "helper",
286
+ });
287
+ expect(result.text).toBe("Recovered.");
288
+
289
+ const toolResults = writes.find(
290
+ (w) =>
291
+ w.op === "appendMessage" &&
292
+ Array.isArray(w.content) &&
293
+ (w.content as Array<Record<string, unknown>>).some(
294
+ (b) => b.type === "tool_result",
295
+ ),
296
+ )!;
297
+ const blocks = (toolResults.content as Array<Record<string, unknown>>).filter(
298
+ (b) => b.type === "tool_result",
299
+ );
300
+ expect(blocks).toHaveLength(2);
301
+ expect(blocks.every((b) => b.is_error === true)).toBe(true);
302
+ expect(String(blocks[1].content)).toContain('Unknown tool "ghost"');
303
+ });
304
+
305
+ test("continuation loads history and rejects agent mismatch", async () => {
306
+ const def = agent({});
307
+ const { ctx, writes } = mockCtx([done("With context.")]);
308
+ const result = await (def.handler as unknown as (
309
+ c: ActionCtx,
310
+ a: Record<string, unknown>,
311
+ ) => Promise<Record<string, unknown>>)(ctx, {
312
+ input: "follow-up",
313
+ runId: "run_9",
314
+ __agentName: "helper",
315
+ });
316
+ expect(result.runId).toBe("run_9");
317
+ // No createRun for continuations.
318
+ expect(writes.some((w) => w.op === "createRun")).toBe(false);
319
+
320
+ // Wrong agent name → AGENT_MISMATCH.
321
+ const other = mockCtx([done("nope")]);
322
+ await expect(
323
+ (def.handler as unknown as (
324
+ c: ActionCtx,
325
+ a: Record<string, unknown>,
326
+ ) => Promise<unknown>)(other.ctx, {
327
+ input: "x",
328
+ runId: "run_9",
329
+ __agentName: "differentAgent",
330
+ }),
331
+ ).rejects.toThrow(/belongs to agent/);
332
+ });
333
+
334
+ test("a fresh running run is RUN_BUSY; a stale one is taken over", async () => {
335
+ const def = agent({ timeout: 600 });
336
+ const busy = mockCtx([done("never")], {
337
+ run: { status: "running", updatedAt: new Date().toISOString() },
338
+ });
339
+ await expect(
340
+ (def.handler as unknown as (
341
+ c: ActionCtx,
342
+ a: Record<string, unknown>,
343
+ ) => Promise<unknown>)(busy.ctx, {
344
+ input: "x",
345
+ runId: "run_9",
346
+ __agentName: "helper",
347
+ }),
348
+ ).rejects.toThrow(/already generating/);
349
+ expect(busy.writes).toHaveLength(0);
350
+
351
+ // updatedAt older than the timeout window → dead generation,
352
+ // continuation proceeds.
353
+ const stale = mockCtx([done("Took over.")], {
354
+ run: {
355
+ status: "running",
356
+ updatedAt: new Date(Date.now() - 700_000).toISOString(),
357
+ },
358
+ });
359
+ const result = await (def.handler as unknown as (
360
+ c: ActionCtx,
361
+ a: Record<string, unknown>,
362
+ ) => Promise<Record<string, unknown>>)(stale.ctx, {
363
+ input: "continue",
364
+ runId: "run_9",
365
+ __agentName: "helper",
366
+ });
367
+ expect(result.text).toBe("Took over.");
368
+ });
369
+
370
+ test("dangling tool_use in history is repaired with error results", async () => {
371
+ const def = agent({});
372
+ const { ctx, writes } = mockCtx([done("Recovered context.")], {
373
+ messages: [
374
+ { id: "m1", seq: 1, role: "user", content: "do the thing" },
375
+ {
376
+ id: "m2",
377
+ seq: 2,
378
+ role: "assistant",
379
+ content: [
380
+ { type: "text", text: "On it." },
381
+ { type: "tool_use", id: "tu_dead", name: "lookup", input: {} },
382
+ ],
383
+ },
384
+ ],
385
+ });
386
+ await (def.handler as unknown as (
387
+ c: ActionCtx,
388
+ a: Record<string, unknown>,
389
+ ) => Promise<unknown>)(ctx, {
390
+ input: "still there?",
391
+ runId: "run_9",
392
+ __agentName: "helper",
393
+ });
394
+ // The repair batch persists BEFORE the new user turn.
395
+ const repair = writes[1];
396
+ expect(repair.op).toBe("appendMessage");
397
+ const blocks = repair.content as Array<Record<string, unknown>>;
398
+ expect(blocks[0].type).toBe("tool_result");
399
+ expect(blocks[0].tool_use_id).toBe("tu_dead");
400
+ expect(blocks[0].is_error).toBe(true);
401
+ expect(String(blocks[0].content)).toContain("interrupted");
402
+ expect(writes[2].content).toBe("still there?");
403
+ });
404
+
405
+ test("oversized tool results are truncated before persisting", async () => {
406
+ const def = agent({
407
+ tools: {
408
+ dump: {
409
+ description: "returns a lot",
410
+ handler: async () => "x".repeat(80 * 1024),
411
+ },
412
+ },
413
+ });
414
+ const { ctx, writes } = mockCtx([
415
+ wantsTool("dump", {}),
416
+ done("Done."),
417
+ ]);
418
+ await (def.handler as unknown as (
419
+ c: ActionCtx,
420
+ a: Record<string, unknown>,
421
+ ) => Promise<unknown>)(ctx, { input: "go", __agentName: "helper" });
422
+ const toolResults = writes.find(
423
+ (w) =>
424
+ w.op === "appendMessage" &&
425
+ Array.isArray(w.content) &&
426
+ (w.content as Array<Record<string, unknown>>).some(
427
+ (b) => b.type === "tool_result",
428
+ ),
429
+ )!;
430
+ const block = (toolResults.content as Array<Record<string, unknown>>)[0];
431
+ const content = String(block.content);
432
+ expect(content.length).toBeLessThan(70 * 1024);
433
+ expect(content.endsWith("[truncated]")).toBe(true);
434
+ });
435
+
436
+ test("maxSteps overrun fails the run", async () => {
437
+ const def = agent({
438
+ maxSteps: 1,
439
+ tools: {
440
+ loop: {
441
+ description: "always called",
442
+ handler: async () => "again",
443
+ },
444
+ },
445
+ });
446
+ // The model asks for a tool every time — exceeds maxSteps=1.
447
+ const { ctx, writes } = mockCtx([
448
+ wantsTool("loop", {}),
449
+ wantsTool("loop", {}),
450
+ ]);
451
+ await expect(
452
+ (def.handler as unknown as (
453
+ c: ActionCtx,
454
+ a: Record<string, unknown>,
455
+ ) => Promise<unknown>)(ctx, { input: "go", __agentName: "helper" }),
456
+ ).rejects.toThrow(/tool round-trips/);
457
+ const failed = writes.find((w) => w.status === "failed")!;
458
+ expect(String(failed.error)).toContain("tool round-trips");
459
+ });
460
+
461
+ test("llm failure marks the run failed and rethrows", async () => {
462
+ const def = agent({});
463
+ const { ctx, writes } = mockCtx([]); // script exhausted immediately
464
+ await expect(
465
+ (def.handler as unknown as (
466
+ c: ActionCtx,
467
+ a: Record<string, unknown>,
468
+ ) => Promise<unknown>)(ctx, { input: "go", __agentName: "helper" }),
469
+ ).rejects.toThrow(/script exhausted/);
470
+ expect(writes.some((w) => w.status === "failed")).toBe(true);
471
+ });
472
+ });
package/src/agent.ts ADDED
@@ -0,0 +1,430 @@
1
+ /**
2
+ * `agent()` — a define-type for LLM agents with durable run state.
3
+ *
4
+ * ```ts
5
+ * // functions/researcher.ts
6
+ * export default agent({
7
+ * system: "You are a research assistant.",
8
+ * tools: {
9
+ * searchDocs: {
10
+ * description: "Search the document library",
11
+ * args: { query: v.string() },
12
+ * handler: async (ctx, { query }) => ctx.runQuery("findSimilar", { query }),
13
+ * },
14
+ * },
15
+ * });
16
+ * ```
17
+ *
18
+ * An agent compiles to an ordinary streaming ACTION (named after its
19
+ * file, callable via `streamFn("researcher", { input })`), whose
20
+ * handler runs the tool loop:
21
+ *
22
+ * 1. Create an `AgentRun` row (or load one when `runId` is passed —
23
+ * that's how a conversation continues) and append the user's
24
+ * input as an `AgentMessage`.
25
+ * 2. `ctx.llm.stream` with the declared tools; text deltas flow to
26
+ * `ctx.stream` (resumable — the run row records the stream id).
27
+ * 3. On `stop_reason === "tool_use"`: validate each tool call's
28
+ * input against its validators, run the handler, record the
29
+ * call + result as messages, loop.
30
+ * 4. Terminal: run marked completed/failed.
31
+ *
32
+ * `AgentRun` and `AgentMessage` are real synced entities (injected
33
+ * into the manifest by the SDK when any agent exists), owner-scoped by
34
+ * policy — so `db.useQuery("AgentMessage", { where: { runId } })`
35
+ * shows the transcript live on every one of the user's devices,
36
+ * including tool calls, with zero extra plumbing.
37
+ */
38
+
39
+ import { action } from "./define";
40
+ import { v, validateArgs } from "./validators";
41
+ import type {
42
+ ActionCtx,
43
+ AnyValidator,
44
+ FnDefinition,
45
+ LlmContentBlock,
46
+ LlmMessage,
47
+ LlmTool,
48
+ ValidatorSchema,
49
+ } from "./types";
50
+
51
+ // ---------------------------------------------------------------------------
52
+ // Public types
53
+ // ---------------------------------------------------------------------------
54
+
55
+ /** One tool an agent can call. */
56
+ export interface AgentTool {
57
+ /** Shown to the model — say when to use the tool, not how it works. */
58
+ description: string;
59
+ /** Argument validators (same `v.*` schema as functions). The model's
60
+ * JSON is validated before the handler runs; invalid input becomes
61
+ * a tool_result error the model can react to. Omit for no-arg tools. */
62
+ args?: ValidatorSchema;
63
+ /** Runs with the agent action's ctx (runQuery/runMutation, llm,
64
+ * email, …). The return value is JSON-serialized into the
65
+ * tool_result the model sees. Throwing marks the result is_error —
66
+ * the model sees the message and can recover. */
67
+ handler: (ctx: ActionCtx, input: Record<string, unknown>) => unknown;
68
+ }
69
+
70
+ export interface AgentDefinition {
71
+ /** System prompt. A function receives the ctx + call args for
72
+ * per-user prompts. */
73
+ system?: string | ((ctx: ActionCtx, args: AgentCallArgs) => string);
74
+ tools?: Record<string, AgentTool>;
75
+ /** Model override (subject to the server's allowlist). */
76
+ model?: string;
77
+ /** Max model↔tool round-trips per invocation (default 16). Hitting
78
+ * the cap fails the run rather than looping forever. */
79
+ maxSteps?: number;
80
+ /** max_tokens per completion (default: server default). */
81
+ maxTokens?: number;
82
+ /** Auth gate for the action (default "user" — runs are owner-scoped,
83
+ * so an authenticated caller is the natural default). */
84
+ auth?: "user" | "admin";
85
+ /** Idle timeout in seconds (default 600; activity extends it). */
86
+ timeout?: number;
87
+ }
88
+
89
+ /** The synthesized action's args. */
90
+ export interface AgentCallArgs {
91
+ /** The user's message for this turn. */
92
+ input: string;
93
+ /** Continue an existing run (must belong to the caller and this
94
+ * agent). Omit to start a new run. */
95
+ runId?: string;
96
+ /** Optional display title, stored on new runs. */
97
+ title?: string;
98
+ }
99
+
100
+ /** What the agent action resolves with (also the `event: result`
101
+ * payload on the SSE stream). */
102
+ export interface AgentResult {
103
+ runId: string;
104
+ /** Concatenated text of the final assistant message. */
105
+ text: string;
106
+ /** Round-trips consumed. */
107
+ steps: number;
108
+ usage: { input_tokens: number; output_tokens: number };
109
+ }
110
+
111
+ // ---------------------------------------------------------------------------
112
+ // Validators → JSON Schema (for LlmTool.input_schema)
113
+ // ---------------------------------------------------------------------------
114
+
115
+ /** Convert one `v.*` validator to a JSON-Schema fragment. */
116
+ export function validatorToJsonSchema(val: AnyValidator): Record<string, unknown> {
117
+ const t = (val as { type: string }).type;
118
+ switch (t) {
119
+ case "string":
120
+ return { type: "string" };
121
+ case "int":
122
+ return { type: "integer" };
123
+ case "number":
124
+ return { type: "number" };
125
+ case "boolean":
126
+ return { type: "boolean" };
127
+ case "null":
128
+ return { type: "null" };
129
+ case "id":
130
+ return { type: "string" };
131
+ case "literal":
132
+ return { const: (val as { value: unknown }).value };
133
+ case "array":
134
+ return {
135
+ type: "array",
136
+ items: validatorToJsonSchema((val as { items: AnyValidator }).items),
137
+ };
138
+ case "object": {
139
+ const fields = (val as { fields: ValidatorSchema }).fields ?? {};
140
+ return validatorSchemaToJsonSchema(fields);
141
+ }
142
+ case "union": {
143
+ const variants = (val as { variants: AnyValidator[] }).variants ?? [];
144
+ return { anyOf: variants.map(validatorToJsonSchema) };
145
+ }
146
+ // json / any — anything goes; the empty schema is JSON Schema's
147
+ // "any value".
148
+ default:
149
+ return {};
150
+ }
151
+ }
152
+
153
+ /** Convert a validator schema (a tool's `args`) to a JSON-Schema
154
+ * object with `required` derived from non-optional fields. */
155
+ export function validatorSchemaToJsonSchema(
156
+ schema: ValidatorSchema,
157
+ ): Record<string, unknown> {
158
+ const properties: Record<string, unknown> = {};
159
+ const required: string[] = [];
160
+ for (const [name, val] of Object.entries(schema)) {
161
+ properties[name] = validatorToJsonSchema(val as AnyValidator);
162
+ if (!(val as { optional?: boolean }).optional) {
163
+ required.push(name);
164
+ }
165
+ }
166
+ const out: Record<string, unknown> = { type: "object", properties };
167
+ if (required.length > 0) out.required = required;
168
+ return out;
169
+ }
170
+
171
+ // ---------------------------------------------------------------------------
172
+ // The define-type
173
+ // ---------------------------------------------------------------------------
174
+
175
+ /** Marker so the SDK's discoverFunctions can detect agents and inject
176
+ * the AgentRun/AgentMessage entities into the manifest. */
177
+ export const AGENT_MARKER = "__pylonAgent";
178
+
179
+ export function agent(def: AgentDefinition): FnDefinition<AgentCallArgs, AgentResult> {
180
+ const fnDef = action({
181
+ args: {
182
+ input: v.string(),
183
+ runId: v.optional(v.string()),
184
+ title: v.optional(v.string()),
185
+ },
186
+ auth: def.auth ?? "user",
187
+ timeout: def.timeout ?? 600,
188
+ handler: (ctx: ActionCtx, args: AgentCallArgs) =>
189
+ runAgentLoop(def, ctx, args),
190
+ } as never) as FnDefinition<AgentCallArgs, AgentResult>;
191
+ (fnDef as unknown as Record<string, unknown>)[AGENT_MARKER] = true;
192
+ return fnDef;
193
+ }
194
+
195
+ export function isAgentDefinition(value: unknown): boolean {
196
+ return (
197
+ typeof value === "object" &&
198
+ value !== null &&
199
+ (value as Record<string, unknown>)[AGENT_MARKER] === true
200
+ );
201
+ }
202
+
203
+ // ---------------------------------------------------------------------------
204
+ // The loop
205
+ // ---------------------------------------------------------------------------
206
+
207
+ const DEFAULT_MAX_STEPS = 16;
208
+
209
+ /** Tool results persist into AgentMessage rows and replay into every
210
+ * later completion's context — cap them so one oversized return can't
211
+ * bloat the transcript (and the sync payloads) unboundedly. */
212
+ const MAX_TOOL_RESULT_CHARS = 64 * 1024;
213
+
214
+ function truncateToolResult(s: string): string {
215
+ if (s.length <= MAX_TOOL_RESULT_CHARS) return s;
216
+ return `${s.slice(0, MAX_TOOL_RESULT_CHARS)}\n[truncated]`;
217
+ }
218
+
219
+ interface StoredMessage {
220
+ id: string;
221
+ seq: number;
222
+ role: string;
223
+ content: unknown;
224
+ }
225
+
226
+ async function runAgentLoop(
227
+ def: AgentDefinition,
228
+ ctx: ActionCtx,
229
+ args: AgentCallArgs,
230
+ ): Promise<AgentResult> {
231
+ // The action's file-inferred name isn't visible here; the internal
232
+ // write mutation derives it from this marker arg set by the registry
233
+ // loader (see registerAgentInternals). Fallback "agent".
234
+ const agentName =
235
+ (args as unknown as Record<string, unknown>).__agentName?.toString() ??
236
+ "agent";
237
+
238
+ // A run left "running" longer than the agent's timeout is a dead
239
+ // generation (the process died before the terminal status write) —
240
+ // continuations may take it over instead of being RUN_BUSY forever.
241
+ const staleMs = Math.max(def.timeout ?? 600, 60) * 1000;
242
+
243
+ // 1. Create or load the run (ownership enforced inside the internal
244
+ // fns, which run under this caller's auth).
245
+ let runId: string;
246
+ let history: LlmMessage[] = [];
247
+ if (args.runId) {
248
+ const loaded = await ctx.runQuery<{
249
+ run: { id: string; agent: string; status: string; updatedAt?: string };
250
+ messages: StoredMessage[];
251
+ }>("__pylon_agent_read", { runId: args.runId });
252
+ if (loaded.run.agent !== agentName) {
253
+ throw ctx.error(
254
+ "AGENT_MISMATCH",
255
+ `Run ${args.runId} belongs to agent "${loaded.run.agent}"`,
256
+ );
257
+ }
258
+ if (loaded.run.status === "running") {
259
+ const updatedAt = Date.parse(String(loaded.run.updatedAt ?? "")) || 0;
260
+ if (Date.now() - updatedAt < staleMs) {
261
+ throw ctx.error(
262
+ "RUN_BUSY",
263
+ "This run is already generating — wait for it to finish",
264
+ );
265
+ }
266
+ }
267
+ runId = loaded.run.id;
268
+ history = loaded.messages.map(storedToLlmMessage);
269
+ } else {
270
+ const created = await ctx.runMutation<{ id: string }>(
271
+ "__pylon_agent_write",
272
+ { op: "createRun", agent: agentName, title: args.title ?? null },
273
+ );
274
+ runId = created.id;
275
+ }
276
+
277
+ const write = (op: Record<string, unknown>) =>
278
+ ctx.runMutation<Record<string, unknown>>("__pylon_agent_write", {
279
+ ...op,
280
+ runId,
281
+ });
282
+
283
+ // 2. Claim the run FIRST (guarded inside the mutation's transaction —
284
+ // the pre-flight check above races between concurrent turns), then
285
+ // write. The stream id lets other devices attach mid-generation.
286
+ await write({
287
+ op: "setStatus",
288
+ status: "running",
289
+ streamId: ctx.stream.id ?? null,
290
+ guardNotRunning: true,
291
+ staleMs,
292
+ });
293
+
294
+ // A crash between persisting an assistant tool_use turn and its
295
+ // tool_results leaves a transcript the LLM API rejects on replay.
296
+ // Repair with synthetic error results before the new user turn.
297
+ const lastMsg = history[history.length - 1];
298
+ if (lastMsg?.role === "assistant" && Array.isArray(lastMsg.content)) {
299
+ const dangling = lastMsg.content.filter(
300
+ (b): b is Extract<LlmContentBlock, { type: "tool_use" }> =>
301
+ (b as { type?: string }).type === "tool_use",
302
+ );
303
+ if (dangling.length > 0) {
304
+ const repairs: LlmContentBlock[] = dangling.map((b) => ({
305
+ type: "tool_result",
306
+ tool_use_id: b.id,
307
+ content: "tool execution was interrupted",
308
+ is_error: true,
309
+ }));
310
+ await write({ op: "appendMessage", role: "user", content: repairs });
311
+ history.push({ role: "user", content: repairs });
312
+ }
313
+ }
314
+
315
+ await write({ op: "appendMessage", role: "user", content: args.input });
316
+ history.push({ role: "user", content: args.input });
317
+
318
+ // Tool declarations for the model.
319
+ const tools: LlmTool[] = Object.entries(def.tools ?? {}).map(
320
+ ([name, tool]) => ({
321
+ name,
322
+ description: tool.description,
323
+ input_schema: validatorSchemaToJsonSchema(tool.args ?? {}),
324
+ }),
325
+ );
326
+
327
+ const maxSteps = def.maxSteps ?? DEFAULT_MAX_STEPS;
328
+ const usage = { input_tokens: 0, output_tokens: 0 };
329
+ let finalText = "";
330
+ let steps = 0;
331
+
332
+ try {
333
+ for (;;) {
334
+ steps += 1;
335
+ if (steps > maxSteps) {
336
+ throw ctx.error(
337
+ "AGENT_MAX_STEPS",
338
+ `Agent exceeded ${maxSteps} tool round-trips`,
339
+ );
340
+ }
341
+ const system =
342
+ typeof def.system === "function" ? def.system(ctx, args) : def.system;
343
+ const res = await ctx.llm.stream(
344
+ {
345
+ messages: history,
346
+ ...(system ? { system } : {}),
347
+ ...(tools.length > 0 ? { tools } : {}),
348
+ ...(def.model ? { model: def.model } : {}),
349
+ ...(def.maxTokens ? { max_tokens: def.maxTokens } : {}),
350
+ },
351
+ (e) => {
352
+ if (e.type === "text_delta") ctx.stream.write(e.text);
353
+ },
354
+ );
355
+ usage.input_tokens += res.usage.input_tokens;
356
+ usage.output_tokens += res.usage.output_tokens;
357
+
358
+ // Persist the assistant turn exactly as the model produced it
359
+ // (text + tool_use blocks) so history replays are faithful.
360
+ await write({ op: "appendMessage", role: "assistant", content: res.content });
361
+ history.push({ role: "assistant", content: res.content });
362
+ finalText = res.content
363
+ .filter((b): b is Extract<LlmContentBlock, { type: "text" }> => b.type === "text")
364
+ .map((b) => b.text)
365
+ .join("");
366
+
367
+ if (res.stop_reason !== "tool_use") break;
368
+
369
+ // 3. Execute every requested tool; failures become is_error
370
+ // results the model can react to rather than run-fatal throws.
371
+ const results: LlmContentBlock[] = [];
372
+ for (const block of res.content) {
373
+ if (block.type !== "tool_use") continue;
374
+ const tool = def.tools?.[block.name];
375
+ let content: string;
376
+ let isError = false;
377
+ if (!tool) {
378
+ content = `Unknown tool "${block.name}"`;
379
+ isError = true;
380
+ } else {
381
+ try {
382
+ if (tool.args) {
383
+ const check = validateArgs(block.input, tool.args);
384
+ if (!check.valid) {
385
+ throw new Error(`Invalid tool input: ${check.errors.join("; ")}`);
386
+ }
387
+ }
388
+ const value = await tool.handler(ctx, block.input);
389
+ content =
390
+ typeof value === "string" ? value : JSON.stringify(value ?? null);
391
+ } catch (err) {
392
+ content = err instanceof Error ? err.message : String(err);
393
+ isError = true;
394
+ }
395
+ }
396
+ content = truncateToolResult(content);
397
+ // Announce the tool call on the stream so live UIs can render
398
+ // "using searchDocs…" without polling the message rows.
399
+ ctx.stream.writeEvent(
400
+ "tool",
401
+ JSON.stringify({ name: block.name, input: block.input, isError }),
402
+ );
403
+ results.push({
404
+ type: "tool_result",
405
+ tool_use_id: block.id,
406
+ content,
407
+ ...(isError ? { is_error: true } : {}),
408
+ });
409
+ }
410
+ await write({ op: "appendMessage", role: "user", content: results });
411
+ history.push({ role: "user", content: results });
412
+ }
413
+ } catch (err) {
414
+ const message = err instanceof Error ? err.message : String(err);
415
+ // Best-effort — the failure we surface is the loop's, not the
416
+ // bookkeeping write's.
417
+ await write({ op: "setStatus", status: "failed", error: message }).catch(
418
+ () => {},
419
+ );
420
+ throw err;
421
+ }
422
+
423
+ await write({ op: "setStatus", status: "completed" });
424
+ return { runId, text: finalText, steps, usage };
425
+ }
426
+
427
+ function storedToLlmMessage(m: StoredMessage): LlmMessage {
428
+ const role = m.role === "assistant" ? "assistant" : "user";
429
+ return { role, content: m.content as LlmMessage["content"] };
430
+ }
package/src/index.ts CHANGED
@@ -21,6 +21,18 @@
21
21
  export { query, mutation, action } from "./define";
22
22
  export { v } from "./validators";
23
23
  export { workflow } from "./workflows";
24
+ export {
25
+ agent,
26
+ isAgentDefinition,
27
+ validatorToJsonSchema,
28
+ validatorSchemaToJsonSchema,
29
+ } from "./agent";
30
+ export type {
31
+ AgentDefinition,
32
+ AgentTool,
33
+ AgentCallArgs,
34
+ AgentResult,
35
+ } from "./agent";
24
36
  export type {
25
37
  WorkflowDefinition,
26
38
  WorkflowRun,
package/src/runtime.ts CHANGED
@@ -788,8 +788,11 @@ function buildWriterOps(callId: string, unsafeOp: boolean): Omit<DbWriter, "unsa
788
788
  };
789
789
  }
790
790
 
791
- function buildStream(callId: string): Stream {
791
+ function buildStream(callId: string, streamId?: string): Stream {
792
792
  return {
793
+ // Host-assigned resumable-stream id (SSE fn path only). Persist it
794
+ // to let other devices attach via GET /api/fn-streams/<id>.
795
+ id: streamId,
793
796
  write(data: string) {
794
797
  // Stream messages are fire-and-forget; they don't get a `result` reply.
795
798
  send({ type: "stream", call_id: callId, data });
@@ -1192,7 +1195,10 @@ async function handleCall(msg: CallMessage): Promise<void> {
1192
1195
  const abort = new AbortController();
1193
1196
  callAborts.set(msg.call_id, abort);
1194
1197
 
1195
- const stream = buildStream(msg.call_id);
1198
+ const stream = buildStream(
1199
+ msg.call_id,
1200
+ (msg as { stream_id?: string }).stream_id,
1201
+ );
1196
1202
  const scheduler = buildScheduler(msg.call_id);
1197
1203
  const email = buildEmail(msg.call_id);
1198
1204
  const llm = buildLlm(msg.call_id);
@@ -1405,6 +1411,8 @@ async function main() {
1405
1411
  files = [];
1406
1412
  }
1407
1413
 
1414
+ const { isAgentDefinition, AGENT_MARKER } = await import("./agent");
1415
+ let agentsPresent = false;
1408
1416
  for (const file of files) {
1409
1417
  const name = basename(file, file.endsWith(".ts") ? ".ts" : ".js");
1410
1418
  try {
@@ -1420,6 +1428,19 @@ async function main() {
1420
1428
  typeof anyDef.type === "string" &&
1421
1429
  typeof anyDef.handler === "function"
1422
1430
  ) {
1431
+ if (isAgentDefinition(def)) {
1432
+ // The agent loop needs its own registered name (for the
1433
+ // AgentRun.agent column + continuation checks) but handlers
1434
+ // don't know their filename — inject it post-validation.
1435
+ agentsPresent = true;
1436
+ const orig = anyDef.handler as (
1437
+ ctx: unknown,
1438
+ args: Record<string, unknown>,
1439
+ ) => unknown;
1440
+ anyDef.handler = (ctx: unknown, args: Record<string, unknown>) =>
1441
+ orig(ctx, { ...args, __agentName: name });
1442
+ void AGENT_MARKER;
1443
+ }
1423
1444
  registry.set(name, def as FnDefinition);
1424
1445
  }
1425
1446
  } catch (err) {
@@ -1427,6 +1448,11 @@ async function main() {
1427
1448
  }
1428
1449
  }
1429
1450
 
1451
+ if (agentsPresent) {
1452
+ const { registerAgentInternals } = await import("./agent-internals");
1453
+ registerAgentInternals(registry);
1454
+ }
1455
+
1430
1456
  // Workflows: scan the app's workflows/ dir (sibling of functions/).
1431
1457
  // Each file default-exports a `workflow(...)`. Declared names ride the
1432
1458
  // ready handshake so the host registers them with its WorkflowEngine;
package/src/types.ts CHANGED
@@ -323,6 +323,16 @@ export interface DbWriter extends DbReader {
323
323
  * client disconnects; it just keeps writing.
324
324
  */
325
325
  export interface Stream {
326
+ /**
327
+ * The host-assigned resumable-stream id for THIS call, present when
328
+ * the caller connected over SSE (`streamFn`). Persist it — e.g. on a
329
+ * run row — and any device can attach to the live stream (or fetch
330
+ * the buffered replay + final result) via `resumeStream(id)` /
331
+ * `GET /api/fn-streams/<id>`. Absent for non-streaming invocations
332
+ * (plain JSON calls, scheduled jobs).
333
+ */
334
+ readonly id?: string;
335
+
326
336
  /** Write a text chunk to the client (SSE). */
327
337
  write(data: string): void;
328
338