@pylonsync/functions 0.4.27 → 0.4.28

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/agent.d.ts CHANGED
@@ -57,8 +57,11 @@ export interface AgentDefinition {
57
57
  tools?: Record<string, AgentTool>;
58
58
  /** Model override (subject to the server's allowlist). */
59
59
  model?: string;
60
- /** Max model↔tool round-trips per invocation (default 16). Hitting
61
- * the cap fails the run rather than looping forever. */
60
+ /** Max model↔tool round-trips per invocation (default 64). Hitting
61
+ * the cap fails the run rather than looping forever. Steering input
62
+ * drained mid-invocation extends the same invocation, so this bounds
63
+ * a steered turn too. The run row's cumulative `steps` counts every
64
+ * invocation and is not capped. */
62
65
  maxSteps?: number;
63
66
  /** max_tokens per completion (default: server default). */
64
67
  maxTokens?: number;
@@ -70,13 +73,19 @@ export interface AgentDefinition {
70
73
  }
71
74
  /** The synthesized action's args. */
72
75
  export interface AgentCallArgs {
73
- /** The user's message for this turn. */
74
- input: string;
76
+ /** The user's message for this turn. Required unless `cancel`. */
77
+ input?: string;
75
78
  /** Continue an existing run (must belong to the caller and this
76
- * agent). Omit to start a new run. */
79
+ * agent). Omit to start a new run. Sending while the run is already
80
+ * generating queues the message for that generation rather than
81
+ * refusing it — see {@link AgentResult.queued}. */
77
82
  runId?: string;
78
83
  /** Optional display title, stored on new runs. */
79
84
  title?: string;
85
+ /** Ask the run to stop. Requires `runId`, ignores `input`, and
86
+ * returns as soon as the request is recorded — a live generation
87
+ * stops at its next turn boundary. */
88
+ cancel?: boolean;
80
89
  }
81
90
  /** What the agent action resolves with (also the `event: result`
82
91
  * payload on the SSE stream). */
@@ -84,12 +93,19 @@ export interface AgentResult {
84
93
  runId: string;
85
94
  /** Concatenated text of the final assistant message. */
86
95
  text: string;
87
- /** Round-trips consumed. */
96
+ /** Round-trips consumed by THIS invocation. */
88
97
  steps: number;
89
98
  usage: {
90
99
  input_tokens: number;
91
100
  output_tokens: number;
92
101
  };
102
+ /** The message was queued onto a generation already in flight
103
+ * instead of starting a turn. No model call happened on this call;
104
+ * the running loop picks the message up at its next boundary. */
105
+ queued?: boolean;
106
+ /** The run stopped because cancel was requested. `text` holds
107
+ * whatever the model had produced by then. */
108
+ cancelled?: boolean;
93
109
  }
94
110
  /** Convert one `v.*` validator to a JSON-Schema fragment. */
95
111
  export declare function validatorToJsonSchema(val: AnyValidator): Record<string, unknown>;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@pylonsync/functions",
3
- "version": "0.4.27",
3
+ "version": "0.4.28",
4
4
  "description": "TypeScript function runtime for pylon — defines server-side queries, mutations, and actions.",
5
5
  "type": "module",
6
6
  "main": "src/index.ts",
@@ -20,10 +20,23 @@ interface RunRow {
20
20
  [key: string]: unknown;
21
21
  }
22
22
 
23
+ /** Bounds on the steering queue. A user typing at a busy run must not
24
+ * be able to grow one row without limit, and the drained text is
25
+ * replayed into model context, so it is billed. */
26
+ const MAX_PENDING_INPUTS = 32;
27
+ const MAX_PENDING_CHARS = 32 * 1024;
28
+
23
29
  function nowIso(): string {
24
30
  return new Date().toISOString();
25
31
  }
26
32
 
33
+ /** The steering queue as a string list, tolerating a null/garbled column. */
34
+ function pendingInputOf(run: Record<string, unknown>): string[] {
35
+ const raw = run.pendingInput;
36
+ if (!Array.isArray(raw)) return [];
37
+ return raw.filter((v): v is string => typeof v === "string");
38
+ }
39
+
27
40
  /** Ownership fence: admins pass; otherwise the run must belong to the
28
41
  * caller. Missing and foreign runs are indistinguishable (NOT_FOUND). */
29
42
  function requireOwnedRun(
@@ -61,6 +74,27 @@ export function registerAgentInternals(
61
74
  },
62
75
  } as unknown as FnDefinition);
63
76
 
77
+ // Liveness probe for a loop that is mid-turn. Deliberately does NOT
78
+ // load the transcript the way __pylon_agent_read does — the loop
79
+ // calls this on a timer while a tool handler runs, and pulling every
80
+ // message each time would scale the poll cost with the conversation.
81
+ registry.set("__pylon_agent_poll", {
82
+ type: "query",
83
+ internal: true,
84
+ auth: "user",
85
+ handler: async (ctx: QueryCtx, args: Record<string, unknown>) => {
86
+ const run = requireOwnedRun(
87
+ ctx,
88
+ await ctx.db.get("AgentRun", String(args.runId ?? "")),
89
+ );
90
+ return {
91
+ status: run.status,
92
+ cancelRequested: run.cancelRequested === true,
93
+ pending: pendingInputOf(run).length,
94
+ };
95
+ },
96
+ } as unknown as FnDefinition);
97
+
64
98
  registry.set("__pylon_agent_write", {
65
99
  type: "mutation",
66
100
  internal: true,
@@ -91,6 +125,9 @@ export function registerAgentInternals(
91
125
  title: args.title ?? null,
92
126
  streamId: null,
93
127
  error: null,
128
+ pendingInput: [],
129
+ cancelRequested: false,
130
+ steps: 0,
94
131
  createdAt: nowIso(),
95
132
  updatedAt: nowIso(),
96
133
  });
@@ -126,10 +163,12 @@ export function registerAgentInternals(
126
163
  case "setStatus": {
127
164
  const runId = String(args.runId ?? "");
128
165
  const run = requireOwnedRun(ctx, await ctx.db.get("AgentRun", runId));
129
- // The loop's pre-flight RUN_BUSY check races with itself
130
- // across concurrent turns; this in-transaction guard is the
131
- // authoritative claim. A "running" row older than staleMs is
132
- // a dead generation (process crash) and may be taken over.
166
+ // The authoritative claim on the run. A caller that finds
167
+ // the run live queues its message instead of reaching here,
168
+ // so this guards the narrower race: two turns that both read
169
+ // a non-running row and both try to start generating. A
170
+ // "running" row older than staleMs is a dead generation
171
+ // (process crash) and may be taken over.
133
172
  if (args.guardNotRunning === true && run.status === "running") {
134
173
  const staleMs = Number(args.staleMs) || 0;
135
174
  const updatedAt = Date.parse(String(run.updatedAt ?? "")) || 0;
@@ -141,15 +180,109 @@ export function registerAgentInternals(
141
180
  throw err;
142
181
  }
143
182
  }
183
+ const status = String(args.status ?? "idle");
144
184
  const patch: Record<string, unknown> = {
145
- status: String(args.status ?? "idle"),
185
+ status,
146
186
  updatedAt: nowIso(),
147
187
  };
148
188
  if (args.error !== undefined) patch.error = args.error;
149
189
  if (args.streamId !== undefined) patch.streamId = args.streamId;
190
+ if (args.steps !== undefined) patch.steps = Number(args.steps) || 0;
191
+ // Claiming the run starts a new generation, which supersedes
192
+ // a cancel left set by one that died before acting on it.
193
+ // Terminal statuses clear it because it has been honoured.
194
+ if (status === "running" || status === "cancelled") {
195
+ patch.cancelRequested = false;
196
+ }
150
197
  const updated = await ctx.db.update("AgentRun", runId, patch);
151
198
  return { updated };
152
199
  }
200
+ case "enqueueInput": {
201
+ // Steering: the caller sent a message while a generation was
202
+ // in flight. Queue it on the run instead of refusing it; the
203
+ // loop folds it into the transcript at the next turn
204
+ // boundary, where the message order stays replayable.
205
+ const runId = String(args.runId ?? "");
206
+ const run = requireOwnedRun(ctx, await ctx.db.get("AgentRun", runId));
207
+ const input = String(args.input ?? "");
208
+ if (input === "") {
209
+ const err = new Error("Steering input must not be empty");
210
+ (err as { code?: string }).code = "AGENT_INPUT_REQUIRED";
211
+ throw err;
212
+ }
213
+ const queue = pendingInputOf(run);
214
+ const chars =
215
+ queue.reduce((n, s) => n + s.length, 0) + input.length;
216
+ if (queue.length >= MAX_PENDING_INPUTS || chars > MAX_PENDING_CHARS) {
217
+ const err = new Error(
218
+ "Too much input queued for this run — wait for the agent to catch up",
219
+ );
220
+ (err as { code?: string }).code = "AGENT_INPUT_QUEUE_FULL";
221
+ throw err;
222
+ }
223
+ queue.push(input);
224
+ // Deliberately NOT touching updatedAt: that column is the
225
+ // liveness clock the staleness takeover reads. Bumping it
226
+ // here would let a user typing at a dead generation keep it
227
+ // looking alive forever.
228
+ await ctx.db.update("AgentRun", runId, { pendingInput: queue });
229
+ return { queued: queue.length };
230
+ }
231
+ case "drainInput": {
232
+ // One atomic turn-boundary check: take everything queued,
233
+ // report whether cancel was requested, and (at the terminal
234
+ // boundary) settle the run as completed only if nothing
235
+ // arrived in the meantime. Doing all three in one
236
+ // transaction is what closes the race where a message lands
237
+ // between the last turn and the completed write.
238
+ const runId = String(args.runId ?? "");
239
+ const run = requireOwnedRun(ctx, await ctx.db.get("AgentRun", runId));
240
+ const queue = pendingInputOf(run);
241
+ const cancelRequested = run.cancelRequested === true;
242
+ const patch: Record<string, unknown> = { updatedAt: nowIso() };
243
+ if (queue.length > 0) patch.pendingInput = [];
244
+ if (args.steps !== undefined) patch.steps = Number(args.steps) || 0;
245
+ let completed = false;
246
+ if (
247
+ queue.length === 0 &&
248
+ !cancelRequested &&
249
+ args.completeIfEmpty === true
250
+ ) {
251
+ patch.status = "completed";
252
+ completed = true;
253
+ }
254
+ await ctx.db.update("AgentRun", runId, patch);
255
+ return { input: queue, cancelRequested, completed };
256
+ }
257
+ case "requestCancel": {
258
+ const runId = String(args.runId ?? "");
259
+ const run = requireOwnedRun(ctx, await ctx.db.get("AgentRun", runId));
260
+ const agent = args.agent === undefined ? null : String(args.agent);
261
+ if (agent !== null && run.agent !== agent) {
262
+ const err = new Error(
263
+ `Run ${runId} belongs to agent "${run.agent}"`,
264
+ );
265
+ (err as { code?: string }).code = "AGENT_MISMATCH";
266
+ throw err;
267
+ }
268
+ if (run.status === "running") {
269
+ // A generation is live. Raise the flag and let the loop
270
+ // honour it at its next boundary — it owns the terminal
271
+ // write, so racing it here would garble the transcript.
272
+ // updatedAt stays put so the staleness clock keeps running.
273
+ await ctx.db.update("AgentRun", runId, { cancelRequested: true });
274
+ return { status: "running", accepted: true };
275
+ }
276
+ // Nothing is generating, so there is no loop to honour it:
277
+ // settle terminally and drop anything queued.
278
+ await ctx.db.update("AgentRun", runId, {
279
+ status: "cancelled",
280
+ cancelRequested: false,
281
+ pendingInput: [],
282
+ updatedAt: nowIso(),
283
+ });
284
+ return { status: "cancelled", accepted: true };
285
+ }
153
286
  default: {
154
287
  const err = new Error(`Unknown agent write op "${op}"`);
155
288
  (err as { code?: string }).code = "INVALID_OP";
package/src/agent.test.ts CHANGED
@@ -77,9 +77,23 @@ interface ReadOverride {
77
77
  messages?: Array<Record<string, unknown>>;
78
78
  }
79
79
 
80
+ /** Mutable stand-in for the AgentRun columns the loop steers on, so a
81
+ * test can simulate a message or a cancel landing mid-run. */
82
+ interface RunState {
83
+ pendingInput: string[];
84
+ cancelRequested: boolean;
85
+ status: string;
86
+ }
87
+
80
88
  function mockCtx(script: LlmCompleteResponse[], read?: ReadOverride) {
81
89
  const writes: WriteOp[] = [];
82
90
  const streamed: Array<{ event?: string; data: string }> = [];
91
+ const state: RunState = {
92
+ pendingInput: [],
93
+ cancelRequested: false,
94
+ status: "running",
95
+ };
96
+ let polls = 0;
83
97
  let call = 0;
84
98
  const ctx = {
85
99
  auth: { userId: "u1", isAdmin: false, tenantId: null, roles: [] },
@@ -116,6 +130,14 @@ function mockCtx(script: LlmCompleteResponse[], read?: ReadOverride) {
116
130
  },
117
131
  },
118
132
  async runQuery(name: string, args: Record<string, unknown>) {
133
+ if (name === "__pylon_agent_poll") {
134
+ polls += 1;
135
+ return {
136
+ status: state.status,
137
+ cancelRequested: state.cancelRequested,
138
+ pending: state.pendingInput.length,
139
+ };
140
+ }
119
141
  if (name === "__pylon_agent_read") {
120
142
  return {
121
143
  run: {
@@ -142,6 +164,32 @@ function mockCtx(script: LlmCompleteResponse[], read?: ReadOverride) {
142
164
  writes.push(args as WriteOp);
143
165
  if (args.op === "createRun") return { id: "run_1" };
144
166
  if (args.op === "appendMessage") return { id: "m", seq: writes.length };
167
+ // Mirrors agent-internals' transactional semantics: take the
168
+ // queue, report the cancel flag, and settle only when nothing
169
+ // arrived and no cancel is pending.
170
+ if (args.op === "drainInput") {
171
+ const input = state.pendingInput.slice();
172
+ state.pendingInput = [];
173
+ let completed = false;
174
+ if (
175
+ input.length === 0 &&
176
+ !state.cancelRequested &&
177
+ args.completeIfEmpty === true
178
+ ) {
179
+ state.status = "completed";
180
+ completed = true;
181
+ }
182
+ return { input, cancelRequested: state.cancelRequested, completed };
183
+ }
184
+ if (args.op === "enqueueInput") {
185
+ state.pendingInput.push(String(args.input));
186
+ return { queued: state.pendingInput.length };
187
+ }
188
+ if (args.op === "requestCancel") {
189
+ state.cancelRequested = true;
190
+ return { status: state.status, accepted: true };
191
+ }
192
+ if (args.op === "setStatus") state.status = String(args.status);
145
193
  return { updated: true };
146
194
  },
147
195
  error(code: string, message: string) {
@@ -150,7 +198,13 @@ function mockCtx(script: LlmCompleteResponse[], read?: ReadOverride) {
150
198
  return err;
151
199
  },
152
200
  };
153
- return { ctx: ctx as unknown as ActionCtx, writes, streamed };
201
+ return {
202
+ ctx: ctx as unknown as ActionCtx,
203
+ writes,
204
+ streamed,
205
+ state,
206
+ pollCount: () => polls,
207
+ };
154
208
  }
155
209
 
156
210
  const done = (text: string): LlmCompleteResponse => ({
@@ -222,23 +276,30 @@ describe("agent loop", () => {
222
276
  expect(toolCalls).toEqual([{ q: "pylon" }]);
223
277
 
224
278
  // Write sequence: createRun, CLAIM (running, guarded, +streamId),
225
- // user msg, assistant(tool_use), tool_result, assistant(final),
226
- // completed. The claim precedes every message write so a losing
227
- // racer never persists a stray turn.
279
+ // user msg, assistant(tool_use), the tool boundary's drain,
280
+ // tool_result, assistant(final), then the terminal drain — which
281
+ // settles the run itself, so there is no separate completed write.
282
+ // The claim precedes every message write so a losing racer never
283
+ // persists a stray turn.
228
284
  expect(writes.map((w) => `${w.op}:${w.role ?? w.status ?? ""}`)).toEqual([
229
285
  "createRun:",
230
286
  "setStatus:running",
231
287
  "appendMessage:user",
232
288
  "appendMessage:assistant",
289
+ "drainInput:",
233
290
  "appendMessage:user", // tool_result batch
234
291
  "appendMessage:assistant",
235
- "setStatus:completed",
292
+ "drainInput:",
236
293
  ]);
237
294
  const running = writes.find((w) => w.status === "running")!;
238
295
  expect(running.streamId).toBe("st_test");
239
296
  expect(running.guardNotRunning).toBe(true);
240
297
  expect(Number(running.staleMs)).toBeGreaterThan(0);
241
- const toolResultMsg = writes[4];
298
+ // The terminal drain settles the run in the same transaction that
299
+ // proves nothing was queued.
300
+ const terminal = writes[writes.length - 1];
301
+ expect(terminal.completeIfEmpty).toBe(true);
302
+ const toolResultMsg = writes[5];
242
303
  const blocks = toolResultMsg.content as Array<Record<string, unknown>>;
243
304
  expect(blocks[0].type).toBe("tool_result");
244
305
  expect(blocks[0].tool_use_id).toBe("tu_1");
@@ -331,22 +392,27 @@ describe("agent loop", () => {
331
392
  ).rejects.toThrow(/belongs to agent/);
332
393
  });
333
394
 
334
- test("a fresh running run is RUN_BUSY; a stale one is taken over", async () => {
395
+ test("a message to a live run is queued, not refused; a stale one is taken over", async () => {
335
396
  const def = agent({ timeout: 600 });
336
397
  const busy = mockCtx([done("never")], {
337
398
  run: { status: "running", updatedAt: new Date().toISOString() },
338
399
  });
339
- await expect(
340
- (def.handler as unknown as (
341
- c: ActionCtx,
342
- a: Record<string, unknown>,
343
- ) => Promise<unknown>)(busy.ctx, {
344
- input: "x",
345
- runId: "run_9",
346
- __agentName: "helper",
347
- }),
348
- ).rejects.toThrow(/already generating/);
349
- expect(busy.writes).toHaveLength(0);
400
+ const queued = await (def.handler as unknown as (
401
+ c: ActionCtx,
402
+ a: Record<string, unknown>,
403
+ ) => Promise<Record<string, unknown>>)(busy.ctx, {
404
+ input: "actually, use the other file",
405
+ runId: "run_9",
406
+ __agentName: "helper",
407
+ });
408
+ // Steering: the message lands on the queue and no turn starts.
409
+ expect(queued.queued).toBe(true);
410
+ expect(queued.runId).toBe("run_9");
411
+ expect(queued.steps).toBe(0);
412
+ expect(busy.state.pendingInput).toEqual(["actually, use the other file"]);
413
+ expect(busy.writes.map((w) => w.op)).toEqual(["enqueueInput"]);
414
+ // No claim, no transcript write — the live generation owns those.
415
+ expect(busy.writes.some((w) => w.op === "setStatus")).toBe(false);
350
416
 
351
417
  // updatedAt older than the timeout window → dead generation,
352
418
  // continuation proceeds.
@@ -391,15 +457,35 @@ describe("agent loop", () => {
391
457
  runId: "run_9",
392
458
  __agentName: "helper",
393
459
  });
394
- // The repair batch persists BEFORE the new user turn.
460
+ // The repair rides in the SAME user message as the new input.
461
+ // Two user messages in a row are rejected by the Messages API just
462
+ // as a dangling tool_use is, so splitting them would swap one
463
+ // unreplayable transcript for another.
395
464
  const repair = writes[1];
396
465
  expect(repair.op).toBe("appendMessage");
466
+ expect(repair.role).toBe("user");
397
467
  const blocks = repair.content as Array<Record<string, unknown>>;
398
468
  expect(blocks[0].type).toBe("tool_result");
399
469
  expect(blocks[0].tool_use_id).toBe("tu_dead");
400
470
  expect(blocks[0].is_error).toBe(true);
401
471
  expect(String(blocks[0].content)).toContain("interrupted");
402
- expect(writes[2].content).toBe("still there?");
472
+ expect(blocks[1]).toEqual({ type: "text", text: "still there?" });
473
+ // tool_result blocks come first — the API requires them at the
474
+ // head of the user turn that answers a tool_use.
475
+ expect(blocks.filter((b) => b.type === "tool_result")).toHaveLength(1);
476
+ });
477
+
478
+ test("with nothing to repair the opening turn stays a plain string", async () => {
479
+ const def = agent({});
480
+ const { ctx, writes } = mockCtx([done("ok")]);
481
+ await (def.handler as unknown as (
482
+ c: ActionCtx,
483
+ a: Record<string, unknown>,
484
+ ) => Promise<unknown>)(ctx, { input: "hello", __agentName: "helper" });
485
+ const opening = writes.find(
486
+ (w) => w.op === "appendMessage" && w.role === "user",
487
+ )!;
488
+ expect(opening.content).toBe("hello");
403
489
  });
404
490
 
405
491
  test("oversized tool results are truncated before persisting", async () => {
@@ -458,6 +544,294 @@ describe("agent loop", () => {
458
544
  expect(String(failed.error)).toContain("tool round-trips");
459
545
  });
460
546
 
547
+ test("queued input drains at the terminal boundary and extends the turn", async () => {
548
+ const def = agent({});
549
+ const { ctx, writes, state } = mockCtx([
550
+ done("First answer."),
551
+ done("Second answer."),
552
+ ]);
553
+ // The user types while the first response is being produced.
554
+ const originalStream = (ctx as unknown as { llm: { stream: unknown } }).llm
555
+ .stream as (...a: unknown[]) => Promise<unknown>;
556
+ let sent = false;
557
+ (ctx as unknown as { llm: { stream: unknown } }).llm.stream = async (
558
+ ...a: unknown[]
559
+ ) => {
560
+ const res = await originalStream(...a);
561
+ if (!sent) {
562
+ sent = true;
563
+ state.pendingInput.push("wait, also check the tests");
564
+ }
565
+ return res;
566
+ };
567
+
568
+ const result = await (def.handler as unknown as (
569
+ c: ActionCtx,
570
+ a: Record<string, unknown>,
571
+ ) => Promise<Record<string, unknown>>)(ctx, {
572
+ input: "go",
573
+ __agentName: "helper",
574
+ });
575
+
576
+ // The run did NOT complete after the first response — the queued
577
+ // message became the next user turn instead.
578
+ expect(result.text).toBe("Second answer.");
579
+ expect(result.steps).toBe(2);
580
+ const userTurns = writes.filter(
581
+ (w) => w.op === "appendMessage" && w.role === "user",
582
+ );
583
+ expect(userTurns.map((w) => w.content)).toEqual([
584
+ "go",
585
+ "wait, also check the tests",
586
+ ]);
587
+ expect(state.pendingInput).toEqual([]);
588
+ expect(state.status).toBe("completed");
589
+ });
590
+
591
+ test("queued input merges into the tool_result turn, results first", async () => {
592
+ const def = agent({
593
+ tools: {
594
+ lookup: {
595
+ description: "look up",
596
+ handler: async () => "found",
597
+ },
598
+ },
599
+ });
600
+ const { ctx, writes, state } = mockCtx([
601
+ wantsTool("lookup", { q: "a" }),
602
+ done("Done."),
603
+ ]);
604
+ state.pendingInput.push("focus on the second one");
605
+
606
+ await (def.handler as unknown as (
607
+ c: ActionCtx,
608
+ a: Record<string, unknown>,
609
+ ) => Promise<unknown>)(ctx, { input: "go", __agentName: "helper" });
610
+
611
+ const toolTurn = writes.find(
612
+ (w) =>
613
+ w.op === "appendMessage" &&
614
+ Array.isArray(w.content) &&
615
+ (w.content as Array<Record<string, unknown>>).some(
616
+ (b) => b.type === "tool_result",
617
+ ),
618
+ )!;
619
+ const blocks = toolTurn.content as Array<Record<string, unknown>>;
620
+ // tool_result blocks stay at the head of the turn; the steering
621
+ // text rides along behind them rather than as a second user
622
+ // message, which the Messages API would reject.
623
+ expect(blocks[0].type).toBe("tool_result");
624
+ expect(blocks[blocks.length - 1]).toEqual({
625
+ type: "text",
626
+ text: "focus on the second one",
627
+ });
628
+ });
629
+
630
+ test("cancel requested mid-run stops at the boundary and marks the run cancelled", async () => {
631
+ const def = agent({
632
+ tools: {
633
+ slow: {
634
+ description: "slow tool",
635
+ handler: async () => "done anyway",
636
+ },
637
+ },
638
+ });
639
+ const { ctx, writes, state } = mockCtx([
640
+ wantsTool("slow", {}),
641
+ done("never reached"),
642
+ ]);
643
+ state.cancelRequested = true;
644
+
645
+ const result = await (def.handler as unknown as (
646
+ c: ActionCtx,
647
+ a: Record<string, unknown>,
648
+ ) => Promise<Record<string, unknown>>)(ctx, {
649
+ input: "go",
650
+ __agentName: "helper",
651
+ });
652
+
653
+ expect(result.cancelled).toBe(true);
654
+ expect(state.status).toBe("cancelled");
655
+ // The tool_use still got its tool_result — a transcript missing
656
+ // one cannot be replayed on the next turn.
657
+ const toolTurn = writes.find(
658
+ (w) =>
659
+ w.op === "appendMessage" &&
660
+ Array.isArray(w.content) &&
661
+ (w.content as Array<Record<string, unknown>>).some(
662
+ (b) => b.type === "tool_result",
663
+ ),
664
+ )!;
665
+ expect(toolTurn).toBeDefined();
666
+ // And the loop stopped instead of asking the model again.
667
+ expect(
668
+ writes.filter((w) => w.op === "appendMessage" && w.role === "assistant"),
669
+ ).toHaveLength(1);
670
+ });
671
+
672
+ test("a cancel racing queued input keeps the queued text in the transcript", async () => {
673
+ const def = agent({});
674
+ const { ctx, writes, state } = mockCtx([done("First answer.")]);
675
+ const originalStream = (ctx as unknown as { llm: { stream: unknown } }).llm
676
+ .stream as (...a: unknown[]) => Promise<unknown>;
677
+ (ctx as unknown as { llm: { stream: unknown } }).llm.stream = async (
678
+ ...a: unknown[]
679
+ ) => {
680
+ const res = await originalStream(...a);
681
+ // Both land while the model is producing its answer.
682
+ state.pendingInput.push("one more thing");
683
+ state.cancelRequested = true;
684
+ return res;
685
+ };
686
+
687
+ const result = await (def.handler as unknown as (
688
+ c: ActionCtx,
689
+ a: Record<string, unknown>,
690
+ ) => Promise<Record<string, unknown>>)(ctx, {
691
+ input: "go",
692
+ __agentName: "helper",
693
+ });
694
+
695
+ expect(result.cancelled).toBe(true);
696
+ // The drain already took the text off the row, so dropping it here
697
+ // would lose words the user watched themselves type.
698
+ const userTurns = writes.filter(
699
+ (w) => w.op === "appendMessage" && w.role === "user",
700
+ );
701
+ expect(userTurns.map((w) => w.content)).toEqual(["go", "one more thing"]);
702
+ expect(state.status).toBe("cancelled");
703
+ });
704
+
705
+ test("cancel: true records the request and starts no turn", async () => {
706
+ const def = agent({});
707
+ const { ctx, writes, state } = mockCtx([]); // no LLM script needed
708
+ const result = await (def.handler as unknown as (
709
+ c: ActionCtx,
710
+ a: Record<string, unknown>,
711
+ ) => Promise<Record<string, unknown>>)(ctx, {
712
+ runId: "run_9",
713
+ cancel: true,
714
+ __agentName: "helper",
715
+ });
716
+ expect(result.cancelled).toBe(true);
717
+ expect(result.runId).toBe("run_9");
718
+ expect(state.cancelRequested).toBe(true);
719
+ expect(writes.map((w) => w.op)).toEqual(["requestCancel"]);
720
+ // The agent name travels with it so a run can't be stopped
721
+ // through a different agent's action.
722
+ expect(writes[0].agent).toBe("helper");
723
+ });
724
+
725
+ test("cancel without a runId, and input without either, are refused", async () => {
726
+ const def = agent({});
727
+ const call = (a: Record<string, unknown>) =>
728
+ (def.handler as unknown as (
729
+ c: ActionCtx,
730
+ a: Record<string, unknown>,
731
+ ) => Promise<unknown>)(mockCtx([]).ctx, { __agentName: "helper", ...a });
732
+ await expect(call({ cancel: true })).rejects.toThrow(/runId/);
733
+ await expect(call({})).rejects.toThrow(/input is required/);
734
+ });
735
+
736
+ test("a cancel raised while a tool runs aborts ctx.signal and skips the rest", async () => {
737
+ const seen: string[] = [];
738
+ const def = agent({
739
+ tools: {
740
+ first: {
741
+ description: "first",
742
+ handler: async (toolCtx) => {
743
+ seen.push("first");
744
+ // The cancel lands while this handler is in flight; the
745
+ // poller is what notices, so wait past one interval.
746
+ return await new Promise((resolve) => {
747
+ const timer = setInterval(() => {
748
+ if (toolCtx.signal?.aborted) {
749
+ clearInterval(timer);
750
+ resolve("aborted early");
751
+ }
752
+ }, 10);
753
+ });
754
+ },
755
+ },
756
+ second: {
757
+ description: "second",
758
+ handler: async () => {
759
+ seen.push("second");
760
+ return "ran";
761
+ },
762
+ },
763
+ },
764
+ });
765
+ const { ctx, state } = mockCtx([
766
+ {
767
+ model: "m",
768
+ content: [
769
+ { type: "tool_use", id: "tu_1", name: "first", input: {} },
770
+ { type: "tool_use", id: "tu_2", name: "second", input: {} },
771
+ ],
772
+ stop_reason: "tool_use",
773
+ usage: { input_tokens: 1, output_tokens: 1 },
774
+ },
775
+ done("never reached"),
776
+ ]);
777
+ setTimeout(() => {
778
+ state.cancelRequested = true;
779
+ }, 30);
780
+
781
+ const result = await (def.handler as unknown as (
782
+ c: ActionCtx,
783
+ a: Record<string, unknown>,
784
+ ) => Promise<Record<string, unknown>>)(ctx, {
785
+ input: "go",
786
+ __agentName: "helper",
787
+ });
788
+
789
+ expect(result.cancelled).toBe(true);
790
+ // The running handler saw the abort, and the queued one never
791
+ // started.
792
+ expect(seen).toEqual(["first"]);
793
+ expect(state.status).toBe("cancelled");
794
+ }, 10_000);
795
+
796
+ test("the default step budget is high enough for a real tool loop", async () => {
797
+ // 16 was the old default and is nowhere near enough for an agent
798
+ // that reads files before answering.
799
+ const def = agent({
800
+ tools: { step: { description: "one step", handler: async () => "ok" } },
801
+ });
802
+ const script: LlmCompleteResponse[] = [];
803
+ for (let i = 0; i < 40; i += 1) script.push(wantsTool("step", { i }));
804
+ script.push(done("Finished."));
805
+ const { ctx } = mockCtx(script);
806
+ const result = await (def.handler as unknown as (
807
+ c: ActionCtx,
808
+ a: Record<string, unknown>,
809
+ ) => Promise<Record<string, unknown>>)(ctx, {
810
+ input: "go",
811
+ __agentName: "helper",
812
+ });
813
+ expect(result.text).toBe("Finished.");
814
+ expect(result.steps).toBe(41);
815
+ });
816
+
817
+ test("cumulative steps carry across invocations", async () => {
818
+ const def = agent({});
819
+ const { ctx, writes } = mockCtx([done("Next.")], {
820
+ run: { status: "completed", steps: 7 },
821
+ });
822
+ await (def.handler as unknown as (
823
+ c: ActionCtx,
824
+ a: Record<string, unknown>,
825
+ ) => Promise<unknown>)(ctx, {
826
+ input: "again",
827
+ runId: "run_9",
828
+ __agentName: "helper",
829
+ });
830
+ // This invocation ran one step on top of the stored 7.
831
+ const drains = writes.filter((w) => w.op === "drainInput");
832
+ expect(Number(drains[drains.length - 1].steps)).toBe(8);
833
+ });
834
+
461
835
  test("llm failure marks the run failed and rethrows", async () => {
462
836
  const def = agent({});
463
837
  const { ctx, writes } = mockCtx([]); // script exhausted immediately
package/src/agent.ts CHANGED
@@ -74,8 +74,11 @@ export interface AgentDefinition {
74
74
  tools?: Record<string, AgentTool>;
75
75
  /** Model override (subject to the server's allowlist). */
76
76
  model?: string;
77
- /** Max model↔tool round-trips per invocation (default 16). Hitting
78
- * the cap fails the run rather than looping forever. */
77
+ /** Max model↔tool round-trips per invocation (default 64). Hitting
78
+ * the cap fails the run rather than looping forever. Steering input
79
+ * drained mid-invocation extends the same invocation, so this bounds
80
+ * a steered turn too. The run row's cumulative `steps` counts every
81
+ * invocation and is not capped. */
79
82
  maxSteps?: number;
80
83
  /** max_tokens per completion (default: server default). */
81
84
  maxTokens?: number;
@@ -88,13 +91,19 @@ export interface AgentDefinition {
88
91
 
89
92
  /** The synthesized action's args. */
90
93
  export interface AgentCallArgs {
91
- /** The user's message for this turn. */
92
- input: string;
94
+ /** The user's message for this turn. Required unless `cancel`. */
95
+ input?: string;
93
96
  /** Continue an existing run (must belong to the caller and this
94
- * agent). Omit to start a new run. */
97
+ * agent). Omit to start a new run. Sending while the run is already
98
+ * generating queues the message for that generation rather than
99
+ * refusing it — see {@link AgentResult.queued}. */
95
100
  runId?: string;
96
101
  /** Optional display title, stored on new runs. */
97
102
  title?: string;
103
+ /** Ask the run to stop. Requires `runId`, ignores `input`, and
104
+ * returns as soon as the request is recorded — a live generation
105
+ * stops at its next turn boundary. */
106
+ cancel?: boolean;
98
107
  }
99
108
 
100
109
  /** What the agent action resolves with (also the `event: result`
@@ -103,9 +112,16 @@ export interface AgentResult {
103
112
  runId: string;
104
113
  /** Concatenated text of the final assistant message. */
105
114
  text: string;
106
- /** Round-trips consumed. */
115
+ /** Round-trips consumed by THIS invocation. */
107
116
  steps: number;
108
117
  usage: { input_tokens: number; output_tokens: number };
118
+ /** The message was queued onto a generation already in flight
119
+ * instead of starting a turn. No model call happened on this call;
120
+ * the running loop picks the message up at its next boundary. */
121
+ queued?: boolean;
122
+ /** The run stopped because cancel was requested. `text` holds
123
+ * whatever the model had produced by then. */
124
+ cancelled?: boolean;
109
125
  }
110
126
 
111
127
  // ---------------------------------------------------------------------------
@@ -179,9 +195,10 @@ export const AGENT_MARKER = "__pylonAgent";
179
195
  export function agent(def: AgentDefinition): FnDefinition<AgentCallArgs, AgentResult> {
180
196
  const fnDef = action({
181
197
  args: {
182
- input: v.string(),
198
+ input: v.optional(v.string()),
183
199
  runId: v.optional(v.string()),
184
200
  title: v.optional(v.string()),
201
+ cancel: v.optional(v.boolean()),
185
202
  },
186
203
  auth: def.auth ?? "user",
187
204
  timeout: def.timeout ?? 600,
@@ -204,7 +221,13 @@ export function isAgentDefinition(value: unknown): boolean {
204
221
  // The loop
205
222
  // ---------------------------------------------------------------------------
206
223
 
207
- const DEFAULT_MAX_STEPS = 16;
224
+ const DEFAULT_MAX_STEPS = 64;
225
+
226
+ /** How often the loop asks whether cancel was requested while a tool
227
+ * handler is running. Only runs during a tool batch: the host blocks
228
+ * its per-call read loop for the whole of `ctx.llm.stream`, so no RPC
229
+ * the child issues during a generation would be serviced anyway. */
230
+ const CANCEL_POLL_MS = 2000;
208
231
 
209
232
  /** Tool results persist into AgentMessage rows and replay into every
210
233
  * later completion's context — cap them so one oversized return can't
@@ -223,6 +246,30 @@ interface StoredMessage {
223
246
  content: unknown;
224
247
  }
225
248
 
249
+ /** What one turn-boundary `drainInput` reports back. */
250
+ interface DrainResult {
251
+ input: string[];
252
+ cancelRequested: boolean;
253
+ completed: boolean;
254
+ }
255
+
256
+ /** Give tool handlers a ctx whose `signal` also aborts on cancel, so a
257
+ * handler that already threads `ctx.signal` into fetch stops with the
258
+ * run. Prototype-based rather than a spread: ctx's methods keep
259
+ * working when invoked on the derived object. */
260
+ function ctxWithSignal(ctx: ActionCtx, signal: AbortSignal): ActionCtx {
261
+ const combined =
262
+ ctx.signal && typeof AbortSignal.any === "function"
263
+ ? AbortSignal.any([ctx.signal, signal])
264
+ : signal;
265
+ const derived = Object.create(ctx) as ActionCtx;
266
+ Object.defineProperty(derived, "signal", {
267
+ value: combined,
268
+ enumerable: true,
269
+ });
270
+ return derived;
271
+ }
272
+
226
273
  async function runAgentLoop(
227
274
  def: AgentDefinition,
228
275
  ctx: ActionCtx,
@@ -235,18 +282,59 @@ async function runAgentLoop(
235
282
  (args as unknown as Record<string, unknown>).__agentName?.toString() ??
236
283
  "agent";
237
284
 
285
+ const noUsage = { input_tokens: 0, output_tokens: 0 };
286
+
287
+ // Cancel is a control call, not a turn: record the request and
288
+ // return. Ownership (and the agent match) is enforced inside the
289
+ // internal mutation, which runs under this caller's auth.
290
+ if (args.cancel === true) {
291
+ if (!args.runId) {
292
+ throw ctx.error(
293
+ "AGENT_CANCEL_NEEDS_RUN",
294
+ "cancel requires the runId of the run to stop",
295
+ );
296
+ }
297
+ await ctx.runMutation("__pylon_agent_write", {
298
+ op: "requestCancel",
299
+ runId: args.runId,
300
+ agent: agentName,
301
+ });
302
+ return {
303
+ runId: args.runId,
304
+ text: "",
305
+ steps: 0,
306
+ usage: noUsage,
307
+ cancelled: true,
308
+ };
309
+ }
310
+
311
+ const input = args.input ?? "";
312
+ if (input === "") {
313
+ throw ctx.error(
314
+ "AGENT_INPUT_REQUIRED",
315
+ "input is required (pass cancel: true to stop a run instead)",
316
+ );
317
+ }
318
+
238
319
  // A run left "running" longer than the agent's timeout is a dead
239
320
  // generation (the process died before the terminal status write) —
240
- // continuations may take it over instead of being RUN_BUSY forever.
321
+ // continuations may take it over instead of waiting forever.
241
322
  const staleMs = Math.max(def.timeout ?? 600, 60) * 1000;
242
323
 
243
324
  // 1. Create or load the run (ownership enforced inside the internal
244
325
  // fns, which run under this caller's auth).
245
326
  let runId: string;
246
327
  let history: LlmMessage[] = [];
328
+ let priorSteps = 0;
247
329
  if (args.runId) {
248
330
  const loaded = await ctx.runQuery<{
249
- run: { id: string; agent: string; status: string; updatedAt?: string };
331
+ run: {
332
+ id: string;
333
+ agent: string;
334
+ status: string;
335
+ updatedAt?: string;
336
+ steps?: number;
337
+ };
250
338
  messages: StoredMessage[];
251
339
  }>("__pylon_agent_read", { runId: args.runId });
252
340
  if (loaded.run.agent !== agentName) {
@@ -258,13 +346,28 @@ async function runAgentLoop(
258
346
  if (loaded.run.status === "running") {
259
347
  const updatedAt = Date.parse(String(loaded.run.updatedAt ?? "")) || 0;
260
348
  if (Date.now() - updatedAt < staleMs) {
261
- throw ctx.error(
262
- "RUN_BUSY",
263
- "This run is already generating — wait for it to finish",
349
+ // The generation is alive. Hand it the message rather than
350
+ // refusing the call: the loop drains the queue at its next
351
+ // turn boundary, which is what lets a user steer mid-run.
352
+ const queued = await ctx.runMutation<{ queued: number }>(
353
+ "__pylon_agent_write",
354
+ { op: "enqueueInput", runId: loaded.run.id, input },
355
+ );
356
+ ctx.stream.writeEvent(
357
+ "queued",
358
+ JSON.stringify({ runId: loaded.run.id, queued: queued.queued }),
264
359
  );
360
+ return {
361
+ runId: loaded.run.id,
362
+ text: "",
363
+ steps: 0,
364
+ usage: noUsage,
365
+ queued: true,
366
+ };
265
367
  }
266
368
  }
267
369
  runId = loaded.run.id;
370
+ priorSteps = Number(loaded.run.steps) || 0;
268
371
  history = loaded.messages.map(storedToLlmMessage);
269
372
  } else {
270
373
  const created = await ctx.runMutation<{ id: string }>(
@@ -293,27 +396,33 @@ async function runAgentLoop(
293
396
 
294
397
  // A crash between persisting an assistant tool_use turn and its
295
398
  // tool_results leaves a transcript the LLM API rejects on replay.
296
- // Repair with synthetic error results before the new user turn.
399
+ // Repair with synthetic error results, carried in the SAME user
400
+ // message as this turn's input: two user messages in a row are
401
+ // rejected as well, so appending the repair separately would trade
402
+ // one unreplayable transcript for another.
403
+ let repairs: LlmContentBlock[] = [];
297
404
  const lastMsg = history[history.length - 1];
298
405
  if (lastMsg?.role === "assistant" && Array.isArray(lastMsg.content)) {
299
- const dangling = lastMsg.content.filter(
300
- (b): b is Extract<LlmContentBlock, { type: "tool_use" }> =>
301
- (b as { type?: string }).type === "tool_use",
302
- );
303
- if (dangling.length > 0) {
304
- const repairs: LlmContentBlock[] = dangling.map((b) => ({
406
+ repairs = lastMsg.content
407
+ .filter(
408
+ (b): b is Extract<LlmContentBlock, { type: "tool_use" }> =>
409
+ (b as { type?: string }).type === "tool_use",
410
+ )
411
+ .map((b) => ({
305
412
  type: "tool_result",
306
413
  tool_use_id: b.id,
307
414
  content: "tool execution was interrupted",
308
415
  is_error: true,
309
416
  }));
310
- await write({ op: "appendMessage", role: "user", content: repairs });
311
- history.push({ role: "user", content: repairs });
312
- }
313
417
  }
314
-
315
- await write({ op: "appendMessage", role: "user", content: args.input });
316
- history.push({ role: "user", content: args.input });
418
+ // Plain string when there is nothing to repair — the common case
419
+ // stays a simple transcript row.
420
+ const opening: string | LlmContentBlock[] =
421
+ repairs.length > 0
422
+ ? [...repairs, { type: "text", text: input }]
423
+ : input;
424
+ await write({ op: "appendMessage", role: "user", content: opening });
425
+ history.push({ role: "user", content: opening });
317
426
 
318
427
  // Tool declarations for the model.
319
428
  const tools: LlmTool[] = Object.entries(def.tools ?? {}).map(
@@ -328,6 +437,13 @@ async function runAgentLoop(
328
437
  const usage = { input_tokens: 0, output_tokens: 0 };
329
438
  let finalText = "";
330
439
  let steps = 0;
440
+ let cancelled = false;
441
+ let settled = false;
442
+
443
+ const drain = (extra: Record<string, unknown>) =>
444
+ write({ op: "drainInput", steps: priorSteps + steps, ...extra }) as Promise<
445
+ unknown
446
+ > as Promise<DrainResult>;
331
447
 
332
448
  try {
333
449
  for (;;) {
@@ -364,16 +480,137 @@ async function runAgentLoop(
364
480
  .map((b) => b.text)
365
481
  .join("");
366
482
 
367
- if (res.stop_reason !== "tool_use") break;
483
+ if (res.stop_reason !== "tool_use") {
484
+ // Terminal boundary. Settling and draining in one transaction
485
+ // is what stops a message that lands right now from being
486
+ // stranded on a completed run.
487
+ const pending = await drain({ completeIfEmpty: true });
488
+ // Whatever was queued has already left the row, so persist it
489
+ // either way — cancelling must not silently swallow words the
490
+ // user had typed and can still see queued in the UI.
491
+ const text = pending.input.join("\n\n");
492
+ if (text !== "") {
493
+ await write({ op: "appendMessage", role: "user", content: text });
494
+ history.push({ role: "user", content: text });
495
+ }
496
+ if (pending.cancelRequested) {
497
+ cancelled = true;
498
+ break;
499
+ }
500
+ if (pending.completed) {
501
+ settled = true;
502
+ break;
503
+ }
504
+ // Steered after the model had stopped: the queued text became
505
+ // the next user turn and the same invocation keeps going.
506
+ continue;
507
+ }
368
508
 
369
509
  // 3. Execute every requested tool; failures become is_error
370
510
  // results the model can react to rather than run-fatal throws.
371
- const results: LlmContentBlock[] = [];
372
- for (const block of res.content) {
373
- if (block.type !== "tool_use") continue;
511
+ const batch = await runToolBatch(def, ctx, runId, res.content);
512
+
513
+ // Tool boundary. Every tool_use needs its tool_result even when
514
+ // the batch stopped early, or the transcript can't be replayed.
515
+ const pending = await drain({});
516
+ const stopping = pending.cancelRequested || batch.cancelled;
517
+ const blocks: LlmContentBlock[] = [...batch.results];
518
+ for (const text of pending.input) {
519
+ blocks.push({ type: "text", text });
520
+ }
521
+ await write({ op: "appendMessage", role: "user", content: blocks });
522
+ history.push({ role: "user", content: blocks });
523
+ if (stopping) {
524
+ cancelled = true;
525
+ break;
526
+ }
527
+ }
528
+ } catch (err) {
529
+ const message = err instanceof Error ? err.message : String(err);
530
+ // Best-effort — the failure we surface is the loop's, not the
531
+ // bookkeeping write's.
532
+ await write({
533
+ op: "setStatus",
534
+ status: "failed",
535
+ error: message,
536
+ steps: priorSteps + steps,
537
+ }).catch(() => {});
538
+ throw err;
539
+ }
540
+
541
+ if (cancelled) {
542
+ await write({
543
+ op: "setStatus",
544
+ status: "cancelled",
545
+ steps: priorSteps + steps,
546
+ });
547
+ return { runId, text: finalText, steps, usage, cancelled: true };
548
+ }
549
+ // `settled` means drainInput already wrote the terminal status in
550
+ // the same transaction that proved nothing was queued.
551
+ if (!settled) {
552
+ await write({
553
+ op: "setStatus",
554
+ status: "completed",
555
+ steps: priorSteps + steps,
556
+ });
557
+ }
558
+ return { runId, text: finalText, steps, usage };
559
+ }
560
+
561
+ /** Run one turn's tool calls, watching for a cancel while they run.
562
+ *
563
+ * The watch is a poll rather than a push because a cancel can be
564
+ * requested on a different machine than the loop runs on — the run
565
+ * row is the only thing both sides share. It runs only for the
566
+ * duration of the batch: during `ctx.llm.stream` the host blocks its
567
+ * per-call read loop, so an RPC issued then would not be answered. */
568
+ async function runToolBatch(
569
+ def: AgentDefinition,
570
+ ctx: ActionCtx,
571
+ runId: string,
572
+ turn: LlmContentBlock[],
573
+ ): Promise<{ results: LlmContentBlock[]; cancelled: boolean }> {
574
+ const calls = turn.filter(
575
+ (b): b is Extract<LlmContentBlock, { type: "tool_use" }> =>
576
+ b.type === "tool_use",
577
+ );
578
+ const results: LlmContentBlock[] = [];
579
+ if (calls.length === 0) return { results, cancelled: false };
580
+
581
+ const controller = new AbortController();
582
+ const toolCtx = ctxWithSignal(ctx, controller.signal);
583
+ let cancelled = false;
584
+ let polling = false;
585
+ const timer = setInterval(() => {
586
+ // Skip rather than stack: a slow poll must not queue more.
587
+ if (polling || cancelled) return;
588
+ polling = true;
589
+ void ctx
590
+ .runQuery<{ cancelRequested: boolean }>("__pylon_agent_poll", { runId })
591
+ .then((s) => {
592
+ if (s.cancelRequested && !cancelled) {
593
+ cancelled = true;
594
+ controller.abort(new Error("agent run cancelled"));
595
+ }
596
+ })
597
+ .catch(() => {})
598
+ .finally(() => {
599
+ polling = false;
600
+ });
601
+ }, CANCEL_POLL_MS);
602
+
603
+ try {
604
+ for (const block of calls) {
605
+ let content: string;
606
+ let isError = false;
607
+ if (cancelled) {
608
+ // Stop starting new work, but still answer the call so the
609
+ // model sees why it has no result.
610
+ content = "cancelled by user";
611
+ isError = true;
612
+ } else {
374
613
  const tool = def.tools?.[block.name];
375
- let content: string;
376
- let isError = false;
377
614
  if (!tool) {
378
615
  content = `Unknown tool "${block.name}"`;
379
616
  isError = true;
@@ -385,7 +622,7 @@ async function runAgentLoop(
385
622
  throw new Error(`Invalid tool input: ${check.errors.join("; ")}`);
386
623
  }
387
624
  }
388
- const value = await tool.handler(ctx, block.input);
625
+ const value = await tool.handler(toolCtx, block.input);
389
626
  content =
390
627
  typeof value === "string" ? value : JSON.stringify(value ?? null);
391
628
  } catch (err) {
@@ -393,35 +630,25 @@ async function runAgentLoop(
393
630
  isError = true;
394
631
  }
395
632
  }
396
- content = truncateToolResult(content);
397
- // Announce the tool call on the stream so live UIs can render
398
- // "using searchDocs…" without polling the message rows.
399
- ctx.stream.writeEvent(
400
- "tool",
401
- JSON.stringify({ name: block.name, input: block.input, isError }),
402
- );
403
- results.push({
404
- type: "tool_result",
405
- tool_use_id: block.id,
406
- content,
407
- ...(isError ? { is_error: true } : {}),
408
- });
409
633
  }
410
- await write({ op: "appendMessage", role: "user", content: results });
411
- history.push({ role: "user", content: results });
634
+ content = truncateToolResult(content);
635
+ // Announce the tool call on the stream so live UIs can render
636
+ // "using searchDocs…" without polling the message rows.
637
+ ctx.stream.writeEvent(
638
+ "tool",
639
+ JSON.stringify({ name: block.name, input: block.input, isError }),
640
+ );
641
+ results.push({
642
+ type: "tool_result",
643
+ tool_use_id: block.id,
644
+ content,
645
+ ...(isError ? { is_error: true } : {}),
646
+ });
412
647
  }
413
- } catch (err) {
414
- const message = err instanceof Error ? err.message : String(err);
415
- // Best-effort — the failure we surface is the loop's, not the
416
- // bookkeeping write's.
417
- await write({ op: "setStatus", status: "failed", error: message }).catch(
418
- () => {},
419
- );
420
- throw err;
648
+ } finally {
649
+ clearInterval(timer);
421
650
  }
422
-
423
- await write({ op: "setStatus", status: "completed" });
424
- return { runId, text: finalText, steps, usage };
651
+ return { results, cancelled };
425
652
  }
426
653
 
427
654
  function storedToLlmMessage(m: StoredMessage): LlmMessage {