@pylonsync/functions 0.4.27 → 0.4.29
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent.d.ts +22 -6
- package/package.json +1 -1
- package/src/agent-internals.ts +138 -5
- package/src/agent.test.ts +394 -20
- package/src/agent.ts +286 -59
package/dist/agent.d.ts
CHANGED
|
@@ -57,8 +57,11 @@ export interface AgentDefinition {
|
|
|
57
57
|
tools?: Record<string, AgentTool>;
|
|
58
58
|
/** Model override (subject to the server's allowlist). */
|
|
59
59
|
model?: string;
|
|
60
|
-
/** Max model↔tool round-trips per invocation (default
|
|
61
|
-
* the cap fails the run rather than looping forever.
|
|
60
|
+
/** Max model↔tool round-trips per invocation (default 64). Hitting
|
|
61
|
+
* the cap fails the run rather than looping forever. Steering input
|
|
62
|
+
* drained mid-invocation extends the same invocation, so this bounds
|
|
63
|
+
* a steered turn too. The run row's cumulative `steps` counts every
|
|
64
|
+
* invocation and is not capped. */
|
|
62
65
|
maxSteps?: number;
|
|
63
66
|
/** max_tokens per completion (default: server default). */
|
|
64
67
|
maxTokens?: number;
|
|
@@ -70,13 +73,19 @@ export interface AgentDefinition {
|
|
|
70
73
|
}
|
|
71
74
|
/** The synthesized action's args. */
|
|
72
75
|
export interface AgentCallArgs {
|
|
73
|
-
/** The user's message for this turn. */
|
|
74
|
-
input
|
|
76
|
+
/** The user's message for this turn. Required unless `cancel`. */
|
|
77
|
+
input?: string;
|
|
75
78
|
/** Continue an existing run (must belong to the caller and this
|
|
76
|
-
* agent). Omit to start a new run.
|
|
79
|
+
* agent). Omit to start a new run. Sending while the run is already
|
|
80
|
+
* generating queues the message for that generation rather than
|
|
81
|
+
* refusing it — see {@link AgentResult.queued}. */
|
|
77
82
|
runId?: string;
|
|
78
83
|
/** Optional display title, stored on new runs. */
|
|
79
84
|
title?: string;
|
|
85
|
+
/** Ask the run to stop. Requires `runId`, ignores `input`, and
|
|
86
|
+
* returns as soon as the request is recorded — a live generation
|
|
87
|
+
* stops at its next turn boundary. */
|
|
88
|
+
cancel?: boolean;
|
|
80
89
|
}
|
|
81
90
|
/** What the agent action resolves with (also the `event: result`
|
|
82
91
|
* payload on the SSE stream). */
|
|
@@ -84,12 +93,19 @@ export interface AgentResult {
|
|
|
84
93
|
runId: string;
|
|
85
94
|
/** Concatenated text of the final assistant message. */
|
|
86
95
|
text: string;
|
|
87
|
-
/** Round-trips consumed. */
|
|
96
|
+
/** Round-trips consumed by THIS invocation. */
|
|
88
97
|
steps: number;
|
|
89
98
|
usage: {
|
|
90
99
|
input_tokens: number;
|
|
91
100
|
output_tokens: number;
|
|
92
101
|
};
|
|
102
|
+
/** The message was queued onto a generation already in flight
|
|
103
|
+
* instead of starting a turn. No model call happened on this call;
|
|
104
|
+
* the running loop picks the message up at its next boundary. */
|
|
105
|
+
queued?: boolean;
|
|
106
|
+
/** The run stopped because cancel was requested. `text` holds
|
|
107
|
+
* whatever the model had produced by then. */
|
|
108
|
+
cancelled?: boolean;
|
|
93
109
|
}
|
|
94
110
|
/** Convert one `v.*` validator to a JSON-Schema fragment. */
|
|
95
111
|
export declare function validatorToJsonSchema(val: AnyValidator): Record<string, unknown>;
|
package/package.json
CHANGED
package/src/agent-internals.ts
CHANGED
|
@@ -20,10 +20,23 @@ interface RunRow {
|
|
|
20
20
|
[key: string]: unknown;
|
|
21
21
|
}
|
|
22
22
|
|
|
23
|
+
/** Bounds on the steering queue. A user typing at a busy run must not
|
|
24
|
+
* be able to grow one row without limit, and the drained text is
|
|
25
|
+
* replayed into model context, so it is billed. */
|
|
26
|
+
const MAX_PENDING_INPUTS = 32;
|
|
27
|
+
const MAX_PENDING_CHARS = 32 * 1024;
|
|
28
|
+
|
|
23
29
|
function nowIso(): string {
|
|
24
30
|
return new Date().toISOString();
|
|
25
31
|
}
|
|
26
32
|
|
|
33
|
+
/** The steering queue as a string list, tolerating a null/garbled column. */
|
|
34
|
+
function pendingInputOf(run: Record<string, unknown>): string[] {
|
|
35
|
+
const raw = run.pendingInput;
|
|
36
|
+
if (!Array.isArray(raw)) return [];
|
|
37
|
+
return raw.filter((v): v is string => typeof v === "string");
|
|
38
|
+
}
|
|
39
|
+
|
|
27
40
|
/** Ownership fence: admins pass; otherwise the run must belong to the
|
|
28
41
|
* caller. Missing and foreign runs are indistinguishable (NOT_FOUND). */
|
|
29
42
|
function requireOwnedRun(
|
|
@@ -61,6 +74,27 @@ export function registerAgentInternals(
|
|
|
61
74
|
},
|
|
62
75
|
} as unknown as FnDefinition);
|
|
63
76
|
|
|
77
|
+
// Liveness probe for a loop that is mid-turn. Deliberately does NOT
|
|
78
|
+
// load the transcript the way __pylon_agent_read does — the loop
|
|
79
|
+
// calls this on a timer while a tool handler runs, and pulling every
|
|
80
|
+
// message each time would scale the poll cost with the conversation.
|
|
81
|
+
registry.set("__pylon_agent_poll", {
|
|
82
|
+
type: "query",
|
|
83
|
+
internal: true,
|
|
84
|
+
auth: "user",
|
|
85
|
+
handler: async (ctx: QueryCtx, args: Record<string, unknown>) => {
|
|
86
|
+
const run = requireOwnedRun(
|
|
87
|
+
ctx,
|
|
88
|
+
await ctx.db.get("AgentRun", String(args.runId ?? "")),
|
|
89
|
+
);
|
|
90
|
+
return {
|
|
91
|
+
status: run.status,
|
|
92
|
+
cancelRequested: run.cancelRequested === true,
|
|
93
|
+
pending: pendingInputOf(run).length,
|
|
94
|
+
};
|
|
95
|
+
},
|
|
96
|
+
} as unknown as FnDefinition);
|
|
97
|
+
|
|
64
98
|
registry.set("__pylon_agent_write", {
|
|
65
99
|
type: "mutation",
|
|
66
100
|
internal: true,
|
|
@@ -91,6 +125,9 @@ export function registerAgentInternals(
|
|
|
91
125
|
title: args.title ?? null,
|
|
92
126
|
streamId: null,
|
|
93
127
|
error: null,
|
|
128
|
+
pendingInput: [],
|
|
129
|
+
cancelRequested: false,
|
|
130
|
+
steps: 0,
|
|
94
131
|
createdAt: nowIso(),
|
|
95
132
|
updatedAt: nowIso(),
|
|
96
133
|
});
|
|
@@ -126,10 +163,12 @@ export function registerAgentInternals(
|
|
|
126
163
|
case "setStatus": {
|
|
127
164
|
const runId = String(args.runId ?? "");
|
|
128
165
|
const run = requireOwnedRun(ctx, await ctx.db.get("AgentRun", runId));
|
|
129
|
-
// The
|
|
130
|
-
//
|
|
131
|
-
//
|
|
132
|
-
// a
|
|
166
|
+
// The authoritative claim on the run. A caller that finds
|
|
167
|
+
// the run live queues its message instead of reaching here,
|
|
168
|
+
// so this guards the narrower race: two turns that both read
|
|
169
|
+
// a non-running row and both try to start generating. A
|
|
170
|
+
// "running" row older than staleMs is a dead generation
|
|
171
|
+
// (process crash) and may be taken over.
|
|
133
172
|
if (args.guardNotRunning === true && run.status === "running") {
|
|
134
173
|
const staleMs = Number(args.staleMs) || 0;
|
|
135
174
|
const updatedAt = Date.parse(String(run.updatedAt ?? "")) || 0;
|
|
@@ -141,15 +180,109 @@ export function registerAgentInternals(
|
|
|
141
180
|
throw err;
|
|
142
181
|
}
|
|
143
182
|
}
|
|
183
|
+
const status = String(args.status ?? "idle");
|
|
144
184
|
const patch: Record<string, unknown> = {
|
|
145
|
-
status
|
|
185
|
+
status,
|
|
146
186
|
updatedAt: nowIso(),
|
|
147
187
|
};
|
|
148
188
|
if (args.error !== undefined) patch.error = args.error;
|
|
149
189
|
if (args.streamId !== undefined) patch.streamId = args.streamId;
|
|
190
|
+
if (args.steps !== undefined) patch.steps = Number(args.steps) || 0;
|
|
191
|
+
// Claiming the run starts a new generation, which supersedes
|
|
192
|
+
// a cancel left set by one that died before acting on it.
|
|
193
|
+
// Terminal statuses clear it because it has been honoured.
|
|
194
|
+
if (status === "running" || status === "cancelled") {
|
|
195
|
+
patch.cancelRequested = false;
|
|
196
|
+
}
|
|
150
197
|
const updated = await ctx.db.update("AgentRun", runId, patch);
|
|
151
198
|
return { updated };
|
|
152
199
|
}
|
|
200
|
+
case "enqueueInput": {
|
|
201
|
+
// Steering: the caller sent a message while a generation was
|
|
202
|
+
// in flight. Queue it on the run instead of refusing it; the
|
|
203
|
+
// loop folds it into the transcript at the next turn
|
|
204
|
+
// boundary, where the message order stays replayable.
|
|
205
|
+
const runId = String(args.runId ?? "");
|
|
206
|
+
const run = requireOwnedRun(ctx, await ctx.db.get("AgentRun", runId));
|
|
207
|
+
const input = String(args.input ?? "");
|
|
208
|
+
if (input === "") {
|
|
209
|
+
const err = new Error("Steering input must not be empty");
|
|
210
|
+
(err as { code?: string }).code = "AGENT_INPUT_REQUIRED";
|
|
211
|
+
throw err;
|
|
212
|
+
}
|
|
213
|
+
const queue = pendingInputOf(run);
|
|
214
|
+
const chars =
|
|
215
|
+
queue.reduce((n, s) => n + s.length, 0) + input.length;
|
|
216
|
+
if (queue.length >= MAX_PENDING_INPUTS || chars > MAX_PENDING_CHARS) {
|
|
217
|
+
const err = new Error(
|
|
218
|
+
"Too much input queued for this run — wait for the agent to catch up",
|
|
219
|
+
);
|
|
220
|
+
(err as { code?: string }).code = "AGENT_INPUT_QUEUE_FULL";
|
|
221
|
+
throw err;
|
|
222
|
+
}
|
|
223
|
+
queue.push(input);
|
|
224
|
+
// Deliberately NOT touching updatedAt: that column is the
|
|
225
|
+
// liveness clock the staleness takeover reads. Bumping it
|
|
226
|
+
// here would let a user typing at a dead generation keep it
|
|
227
|
+
// looking alive forever.
|
|
228
|
+
await ctx.db.update("AgentRun", runId, { pendingInput: queue });
|
|
229
|
+
return { queued: queue.length };
|
|
230
|
+
}
|
|
231
|
+
case "drainInput": {
|
|
232
|
+
// One atomic turn-boundary check: take everything queued,
|
|
233
|
+
// report whether cancel was requested, and (at the terminal
|
|
234
|
+
// boundary) settle the run as completed only if nothing
|
|
235
|
+
// arrived in the meantime. Doing all three in one
|
|
236
|
+
// transaction is what closes the race where a message lands
|
|
237
|
+
// between the last turn and the completed write.
|
|
238
|
+
const runId = String(args.runId ?? "");
|
|
239
|
+
const run = requireOwnedRun(ctx, await ctx.db.get("AgentRun", runId));
|
|
240
|
+
const queue = pendingInputOf(run);
|
|
241
|
+
const cancelRequested = run.cancelRequested === true;
|
|
242
|
+
const patch: Record<string, unknown> = { updatedAt: nowIso() };
|
|
243
|
+
if (queue.length > 0) patch.pendingInput = [];
|
|
244
|
+
if (args.steps !== undefined) patch.steps = Number(args.steps) || 0;
|
|
245
|
+
let completed = false;
|
|
246
|
+
if (
|
|
247
|
+
queue.length === 0 &&
|
|
248
|
+
!cancelRequested &&
|
|
249
|
+
args.completeIfEmpty === true
|
|
250
|
+
) {
|
|
251
|
+
patch.status = "completed";
|
|
252
|
+
completed = true;
|
|
253
|
+
}
|
|
254
|
+
await ctx.db.update("AgentRun", runId, patch);
|
|
255
|
+
return { input: queue, cancelRequested, completed };
|
|
256
|
+
}
|
|
257
|
+
case "requestCancel": {
|
|
258
|
+
const runId = String(args.runId ?? "");
|
|
259
|
+
const run = requireOwnedRun(ctx, await ctx.db.get("AgentRun", runId));
|
|
260
|
+
const agent = args.agent === undefined ? null : String(args.agent);
|
|
261
|
+
if (agent !== null && run.agent !== agent) {
|
|
262
|
+
const err = new Error(
|
|
263
|
+
`Run ${runId} belongs to agent "${run.agent}"`,
|
|
264
|
+
);
|
|
265
|
+
(err as { code?: string }).code = "AGENT_MISMATCH";
|
|
266
|
+
throw err;
|
|
267
|
+
}
|
|
268
|
+
if (run.status === "running") {
|
|
269
|
+
// A generation is live. Raise the flag and let the loop
|
|
270
|
+
// honour it at its next boundary — it owns the terminal
|
|
271
|
+
// write, so racing it here would garble the transcript.
|
|
272
|
+
// updatedAt stays put so the staleness clock keeps running.
|
|
273
|
+
await ctx.db.update("AgentRun", runId, { cancelRequested: true });
|
|
274
|
+
return { status: "running", accepted: true };
|
|
275
|
+
}
|
|
276
|
+
// Nothing is generating, so there is no loop to honour it:
|
|
277
|
+
// settle terminally and drop anything queued.
|
|
278
|
+
await ctx.db.update("AgentRun", runId, {
|
|
279
|
+
status: "cancelled",
|
|
280
|
+
cancelRequested: false,
|
|
281
|
+
pendingInput: [],
|
|
282
|
+
updatedAt: nowIso(),
|
|
283
|
+
});
|
|
284
|
+
return { status: "cancelled", accepted: true };
|
|
285
|
+
}
|
|
153
286
|
default: {
|
|
154
287
|
const err = new Error(`Unknown agent write op "${op}"`);
|
|
155
288
|
(err as { code?: string }).code = "INVALID_OP";
|
package/src/agent.test.ts
CHANGED
|
@@ -77,9 +77,23 @@ interface ReadOverride {
|
|
|
77
77
|
messages?: Array<Record<string, unknown>>;
|
|
78
78
|
}
|
|
79
79
|
|
|
80
|
+
/** Mutable stand-in for the AgentRun columns the loop steers on, so a
|
|
81
|
+
* test can simulate a message or a cancel landing mid-run. */
|
|
82
|
+
interface RunState {
|
|
83
|
+
pendingInput: string[];
|
|
84
|
+
cancelRequested: boolean;
|
|
85
|
+
status: string;
|
|
86
|
+
}
|
|
87
|
+
|
|
80
88
|
function mockCtx(script: LlmCompleteResponse[], read?: ReadOverride) {
|
|
81
89
|
const writes: WriteOp[] = [];
|
|
82
90
|
const streamed: Array<{ event?: string; data: string }> = [];
|
|
91
|
+
const state: RunState = {
|
|
92
|
+
pendingInput: [],
|
|
93
|
+
cancelRequested: false,
|
|
94
|
+
status: "running",
|
|
95
|
+
};
|
|
96
|
+
let polls = 0;
|
|
83
97
|
let call = 0;
|
|
84
98
|
const ctx = {
|
|
85
99
|
auth: { userId: "u1", isAdmin: false, tenantId: null, roles: [] },
|
|
@@ -116,6 +130,14 @@ function mockCtx(script: LlmCompleteResponse[], read?: ReadOverride) {
|
|
|
116
130
|
},
|
|
117
131
|
},
|
|
118
132
|
async runQuery(name: string, args: Record<string, unknown>) {
|
|
133
|
+
if (name === "__pylon_agent_poll") {
|
|
134
|
+
polls += 1;
|
|
135
|
+
return {
|
|
136
|
+
status: state.status,
|
|
137
|
+
cancelRequested: state.cancelRequested,
|
|
138
|
+
pending: state.pendingInput.length,
|
|
139
|
+
};
|
|
140
|
+
}
|
|
119
141
|
if (name === "__pylon_agent_read") {
|
|
120
142
|
return {
|
|
121
143
|
run: {
|
|
@@ -142,6 +164,32 @@ function mockCtx(script: LlmCompleteResponse[], read?: ReadOverride) {
|
|
|
142
164
|
writes.push(args as WriteOp);
|
|
143
165
|
if (args.op === "createRun") return { id: "run_1" };
|
|
144
166
|
if (args.op === "appendMessage") return { id: "m", seq: writes.length };
|
|
167
|
+
// Mirrors agent-internals' transactional semantics: take the
|
|
168
|
+
// queue, report the cancel flag, and settle only when nothing
|
|
169
|
+
// arrived and no cancel is pending.
|
|
170
|
+
if (args.op === "drainInput") {
|
|
171
|
+
const input = state.pendingInput.slice();
|
|
172
|
+
state.pendingInput = [];
|
|
173
|
+
let completed = false;
|
|
174
|
+
if (
|
|
175
|
+
input.length === 0 &&
|
|
176
|
+
!state.cancelRequested &&
|
|
177
|
+
args.completeIfEmpty === true
|
|
178
|
+
) {
|
|
179
|
+
state.status = "completed";
|
|
180
|
+
completed = true;
|
|
181
|
+
}
|
|
182
|
+
return { input, cancelRequested: state.cancelRequested, completed };
|
|
183
|
+
}
|
|
184
|
+
if (args.op === "enqueueInput") {
|
|
185
|
+
state.pendingInput.push(String(args.input));
|
|
186
|
+
return { queued: state.pendingInput.length };
|
|
187
|
+
}
|
|
188
|
+
if (args.op === "requestCancel") {
|
|
189
|
+
state.cancelRequested = true;
|
|
190
|
+
return { status: state.status, accepted: true };
|
|
191
|
+
}
|
|
192
|
+
if (args.op === "setStatus") state.status = String(args.status);
|
|
145
193
|
return { updated: true };
|
|
146
194
|
},
|
|
147
195
|
error(code: string, message: string) {
|
|
@@ -150,7 +198,13 @@ function mockCtx(script: LlmCompleteResponse[], read?: ReadOverride) {
|
|
|
150
198
|
return err;
|
|
151
199
|
},
|
|
152
200
|
};
|
|
153
|
-
return {
|
|
201
|
+
return {
|
|
202
|
+
ctx: ctx as unknown as ActionCtx,
|
|
203
|
+
writes,
|
|
204
|
+
streamed,
|
|
205
|
+
state,
|
|
206
|
+
pollCount: () => polls,
|
|
207
|
+
};
|
|
154
208
|
}
|
|
155
209
|
|
|
156
210
|
const done = (text: string): LlmCompleteResponse => ({
|
|
@@ -222,23 +276,30 @@ describe("agent loop", () => {
|
|
|
222
276
|
expect(toolCalls).toEqual([{ q: "pylon" }]);
|
|
223
277
|
|
|
224
278
|
// Write sequence: createRun, CLAIM (running, guarded, +streamId),
|
|
225
|
-
// user msg, assistant(tool_use),
|
|
226
|
-
//
|
|
227
|
-
//
|
|
279
|
+
// user msg, assistant(tool_use), the tool boundary's drain,
|
|
280
|
+
// tool_result, assistant(final), then the terminal drain — which
|
|
281
|
+
// settles the run itself, so there is no separate completed write.
|
|
282
|
+
// The claim precedes every message write so a losing racer never
|
|
283
|
+
// persists a stray turn.
|
|
228
284
|
expect(writes.map((w) => `${w.op}:${w.role ?? w.status ?? ""}`)).toEqual([
|
|
229
285
|
"createRun:",
|
|
230
286
|
"setStatus:running",
|
|
231
287
|
"appendMessage:user",
|
|
232
288
|
"appendMessage:assistant",
|
|
289
|
+
"drainInput:",
|
|
233
290
|
"appendMessage:user", // tool_result batch
|
|
234
291
|
"appendMessage:assistant",
|
|
235
|
-
"
|
|
292
|
+
"drainInput:",
|
|
236
293
|
]);
|
|
237
294
|
const running = writes.find((w) => w.status === "running")!;
|
|
238
295
|
expect(running.streamId).toBe("st_test");
|
|
239
296
|
expect(running.guardNotRunning).toBe(true);
|
|
240
297
|
expect(Number(running.staleMs)).toBeGreaterThan(0);
|
|
241
|
-
|
|
298
|
+
// The terminal drain settles the run in the same transaction that
|
|
299
|
+
// proves nothing was queued.
|
|
300
|
+
const terminal = writes[writes.length - 1];
|
|
301
|
+
expect(terminal.completeIfEmpty).toBe(true);
|
|
302
|
+
const toolResultMsg = writes[5];
|
|
242
303
|
const blocks = toolResultMsg.content as Array<Record<string, unknown>>;
|
|
243
304
|
expect(blocks[0].type).toBe("tool_result");
|
|
244
305
|
expect(blocks[0].tool_use_id).toBe("tu_1");
|
|
@@ -331,22 +392,27 @@ describe("agent loop", () => {
|
|
|
331
392
|
).rejects.toThrow(/belongs to agent/);
|
|
332
393
|
});
|
|
333
394
|
|
|
334
|
-
test("a
|
|
395
|
+
test("a message to a live run is queued, not refused; a stale one is taken over", async () => {
|
|
335
396
|
const def = agent({ timeout: 600 });
|
|
336
397
|
const busy = mockCtx([done("never")], {
|
|
337
398
|
run: { status: "running", updatedAt: new Date().toISOString() },
|
|
338
399
|
});
|
|
339
|
-
await
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
).
|
|
349
|
-
expect(
|
|
400
|
+
const queued = await (def.handler as unknown as (
|
|
401
|
+
c: ActionCtx,
|
|
402
|
+
a: Record<string, unknown>,
|
|
403
|
+
) => Promise<Record<string, unknown>>)(busy.ctx, {
|
|
404
|
+
input: "actually, use the other file",
|
|
405
|
+
runId: "run_9",
|
|
406
|
+
__agentName: "helper",
|
|
407
|
+
});
|
|
408
|
+
// Steering: the message lands on the queue and no turn starts.
|
|
409
|
+
expect(queued.queued).toBe(true);
|
|
410
|
+
expect(queued.runId).toBe("run_9");
|
|
411
|
+
expect(queued.steps).toBe(0);
|
|
412
|
+
expect(busy.state.pendingInput).toEqual(["actually, use the other file"]);
|
|
413
|
+
expect(busy.writes.map((w) => w.op)).toEqual(["enqueueInput"]);
|
|
414
|
+
// No claim, no transcript write — the live generation owns those.
|
|
415
|
+
expect(busy.writes.some((w) => w.op === "setStatus")).toBe(false);
|
|
350
416
|
|
|
351
417
|
// updatedAt older than the timeout window → dead generation,
|
|
352
418
|
// continuation proceeds.
|
|
@@ -391,15 +457,35 @@ describe("agent loop", () => {
|
|
|
391
457
|
runId: "run_9",
|
|
392
458
|
__agentName: "helper",
|
|
393
459
|
});
|
|
394
|
-
// The repair
|
|
460
|
+
// The repair rides in the SAME user message as the new input.
|
|
461
|
+
// Two user messages in a row are rejected by the Messages API just
|
|
462
|
+
// as a dangling tool_use is, so splitting them would swap one
|
|
463
|
+
// unreplayable transcript for another.
|
|
395
464
|
const repair = writes[1];
|
|
396
465
|
expect(repair.op).toBe("appendMessage");
|
|
466
|
+
expect(repair.role).toBe("user");
|
|
397
467
|
const blocks = repair.content as Array<Record<string, unknown>>;
|
|
398
468
|
expect(blocks[0].type).toBe("tool_result");
|
|
399
469
|
expect(blocks[0].tool_use_id).toBe("tu_dead");
|
|
400
470
|
expect(blocks[0].is_error).toBe(true);
|
|
401
471
|
expect(String(blocks[0].content)).toContain("interrupted");
|
|
402
|
-
expect(
|
|
472
|
+
expect(blocks[1]).toEqual({ type: "text", text: "still there?" });
|
|
473
|
+
// tool_result blocks come first — the API requires them at the
|
|
474
|
+
// head of the user turn that answers a tool_use.
|
|
475
|
+
expect(blocks.filter((b) => b.type === "tool_result")).toHaveLength(1);
|
|
476
|
+
});
|
|
477
|
+
|
|
478
|
+
test("with nothing to repair the opening turn stays a plain string", async () => {
|
|
479
|
+
const def = agent({});
|
|
480
|
+
const { ctx, writes } = mockCtx([done("ok")]);
|
|
481
|
+
await (def.handler as unknown as (
|
|
482
|
+
c: ActionCtx,
|
|
483
|
+
a: Record<string, unknown>,
|
|
484
|
+
) => Promise<unknown>)(ctx, { input: "hello", __agentName: "helper" });
|
|
485
|
+
const opening = writes.find(
|
|
486
|
+
(w) => w.op === "appendMessage" && w.role === "user",
|
|
487
|
+
)!;
|
|
488
|
+
expect(opening.content).toBe("hello");
|
|
403
489
|
});
|
|
404
490
|
|
|
405
491
|
test("oversized tool results are truncated before persisting", async () => {
|
|
@@ -458,6 +544,294 @@ describe("agent loop", () => {
|
|
|
458
544
|
expect(String(failed.error)).toContain("tool round-trips");
|
|
459
545
|
});
|
|
460
546
|
|
|
547
|
+
test("queued input drains at the terminal boundary and extends the turn", async () => {
|
|
548
|
+
const def = agent({});
|
|
549
|
+
const { ctx, writes, state } = mockCtx([
|
|
550
|
+
done("First answer."),
|
|
551
|
+
done("Second answer."),
|
|
552
|
+
]);
|
|
553
|
+
// The user types while the first response is being produced.
|
|
554
|
+
const originalStream = (ctx as unknown as { llm: { stream: unknown } }).llm
|
|
555
|
+
.stream as (...a: unknown[]) => Promise<unknown>;
|
|
556
|
+
let sent = false;
|
|
557
|
+
(ctx as unknown as { llm: { stream: unknown } }).llm.stream = async (
|
|
558
|
+
...a: unknown[]
|
|
559
|
+
) => {
|
|
560
|
+
const res = await originalStream(...a);
|
|
561
|
+
if (!sent) {
|
|
562
|
+
sent = true;
|
|
563
|
+
state.pendingInput.push("wait, also check the tests");
|
|
564
|
+
}
|
|
565
|
+
return res;
|
|
566
|
+
};
|
|
567
|
+
|
|
568
|
+
const result = await (def.handler as unknown as (
|
|
569
|
+
c: ActionCtx,
|
|
570
|
+
a: Record<string, unknown>,
|
|
571
|
+
) => Promise<Record<string, unknown>>)(ctx, {
|
|
572
|
+
input: "go",
|
|
573
|
+
__agentName: "helper",
|
|
574
|
+
});
|
|
575
|
+
|
|
576
|
+
// The run did NOT complete after the first response — the queued
|
|
577
|
+
// message became the next user turn instead.
|
|
578
|
+
expect(result.text).toBe("Second answer.");
|
|
579
|
+
expect(result.steps).toBe(2);
|
|
580
|
+
const userTurns = writes.filter(
|
|
581
|
+
(w) => w.op === "appendMessage" && w.role === "user",
|
|
582
|
+
);
|
|
583
|
+
expect(userTurns.map((w) => w.content)).toEqual([
|
|
584
|
+
"go",
|
|
585
|
+
"wait, also check the tests",
|
|
586
|
+
]);
|
|
587
|
+
expect(state.pendingInput).toEqual([]);
|
|
588
|
+
expect(state.status).toBe("completed");
|
|
589
|
+
});
|
|
590
|
+
|
|
591
|
+
test("queued input merges into the tool_result turn, results first", async () => {
|
|
592
|
+
const def = agent({
|
|
593
|
+
tools: {
|
|
594
|
+
lookup: {
|
|
595
|
+
description: "look up",
|
|
596
|
+
handler: async () => "found",
|
|
597
|
+
},
|
|
598
|
+
},
|
|
599
|
+
});
|
|
600
|
+
const { ctx, writes, state } = mockCtx([
|
|
601
|
+
wantsTool("lookup", { q: "a" }),
|
|
602
|
+
done("Done."),
|
|
603
|
+
]);
|
|
604
|
+
state.pendingInput.push("focus on the second one");
|
|
605
|
+
|
|
606
|
+
await (def.handler as unknown as (
|
|
607
|
+
c: ActionCtx,
|
|
608
|
+
a: Record<string, unknown>,
|
|
609
|
+
) => Promise<unknown>)(ctx, { input: "go", __agentName: "helper" });
|
|
610
|
+
|
|
611
|
+
const toolTurn = writes.find(
|
|
612
|
+
(w) =>
|
|
613
|
+
w.op === "appendMessage" &&
|
|
614
|
+
Array.isArray(w.content) &&
|
|
615
|
+
(w.content as Array<Record<string, unknown>>).some(
|
|
616
|
+
(b) => b.type === "tool_result",
|
|
617
|
+
),
|
|
618
|
+
)!;
|
|
619
|
+
const blocks = toolTurn.content as Array<Record<string, unknown>>;
|
|
620
|
+
// tool_result blocks stay at the head of the turn; the steering
|
|
621
|
+
// text rides along behind them rather than as a second user
|
|
622
|
+
// message, which the Messages API would reject.
|
|
623
|
+
expect(blocks[0].type).toBe("tool_result");
|
|
624
|
+
expect(blocks[blocks.length - 1]).toEqual({
|
|
625
|
+
type: "text",
|
|
626
|
+
text: "focus on the second one",
|
|
627
|
+
});
|
|
628
|
+
});
|
|
629
|
+
|
|
630
|
+
test("cancel requested mid-run stops at the boundary and marks the run cancelled", async () => {
|
|
631
|
+
const def = agent({
|
|
632
|
+
tools: {
|
|
633
|
+
slow: {
|
|
634
|
+
description: "slow tool",
|
|
635
|
+
handler: async () => "done anyway",
|
|
636
|
+
},
|
|
637
|
+
},
|
|
638
|
+
});
|
|
639
|
+
const { ctx, writes, state } = mockCtx([
|
|
640
|
+
wantsTool("slow", {}),
|
|
641
|
+
done("never reached"),
|
|
642
|
+
]);
|
|
643
|
+
state.cancelRequested = true;
|
|
644
|
+
|
|
645
|
+
const result = await (def.handler as unknown as (
|
|
646
|
+
c: ActionCtx,
|
|
647
|
+
a: Record<string, unknown>,
|
|
648
|
+
) => Promise<Record<string, unknown>>)(ctx, {
|
|
649
|
+
input: "go",
|
|
650
|
+
__agentName: "helper",
|
|
651
|
+
});
|
|
652
|
+
|
|
653
|
+
expect(result.cancelled).toBe(true);
|
|
654
|
+
expect(state.status).toBe("cancelled");
|
|
655
|
+
// The tool_use still got its tool_result — a transcript missing
|
|
656
|
+
// one cannot be replayed on the next turn.
|
|
657
|
+
const toolTurn = writes.find(
|
|
658
|
+
(w) =>
|
|
659
|
+
w.op === "appendMessage" &&
|
|
660
|
+
Array.isArray(w.content) &&
|
|
661
|
+
(w.content as Array<Record<string, unknown>>).some(
|
|
662
|
+
(b) => b.type === "tool_result",
|
|
663
|
+
),
|
|
664
|
+
)!;
|
|
665
|
+
expect(toolTurn).toBeDefined();
|
|
666
|
+
// And the loop stopped instead of asking the model again.
|
|
667
|
+
expect(
|
|
668
|
+
writes.filter((w) => w.op === "appendMessage" && w.role === "assistant"),
|
|
669
|
+
).toHaveLength(1);
|
|
670
|
+
});
|
|
671
|
+
|
|
672
|
+
test("a cancel racing queued input keeps the queued text in the transcript", async () => {
|
|
673
|
+
const def = agent({});
|
|
674
|
+
const { ctx, writes, state } = mockCtx([done("First answer.")]);
|
|
675
|
+
const originalStream = (ctx as unknown as { llm: { stream: unknown } }).llm
|
|
676
|
+
.stream as (...a: unknown[]) => Promise<unknown>;
|
|
677
|
+
(ctx as unknown as { llm: { stream: unknown } }).llm.stream = async (
|
|
678
|
+
...a: unknown[]
|
|
679
|
+
) => {
|
|
680
|
+
const res = await originalStream(...a);
|
|
681
|
+
// Both land while the model is producing its answer.
|
|
682
|
+
state.pendingInput.push("one more thing");
|
|
683
|
+
state.cancelRequested = true;
|
|
684
|
+
return res;
|
|
685
|
+
};
|
|
686
|
+
|
|
687
|
+
const result = await (def.handler as unknown as (
|
|
688
|
+
c: ActionCtx,
|
|
689
|
+
a: Record<string, unknown>,
|
|
690
|
+
) => Promise<Record<string, unknown>>)(ctx, {
|
|
691
|
+
input: "go",
|
|
692
|
+
__agentName: "helper",
|
|
693
|
+
});
|
|
694
|
+
|
|
695
|
+
expect(result.cancelled).toBe(true);
|
|
696
|
+
// The drain already took the text off the row, so dropping it here
|
|
697
|
+
// would lose words the user watched themselves type.
|
|
698
|
+
const userTurns = writes.filter(
|
|
699
|
+
(w) => w.op === "appendMessage" && w.role === "user",
|
|
700
|
+
);
|
|
701
|
+
expect(userTurns.map((w) => w.content)).toEqual(["go", "one more thing"]);
|
|
702
|
+
expect(state.status).toBe("cancelled");
|
|
703
|
+
});
|
|
704
|
+
|
|
705
|
+
test("cancel: true records the request and starts no turn", async () => {
|
|
706
|
+
const def = agent({});
|
|
707
|
+
const { ctx, writes, state } = mockCtx([]); // no LLM script needed
|
|
708
|
+
const result = await (def.handler as unknown as (
|
|
709
|
+
c: ActionCtx,
|
|
710
|
+
a: Record<string, unknown>,
|
|
711
|
+
) => Promise<Record<string, unknown>>)(ctx, {
|
|
712
|
+
runId: "run_9",
|
|
713
|
+
cancel: true,
|
|
714
|
+
__agentName: "helper",
|
|
715
|
+
});
|
|
716
|
+
expect(result.cancelled).toBe(true);
|
|
717
|
+
expect(result.runId).toBe("run_9");
|
|
718
|
+
expect(state.cancelRequested).toBe(true);
|
|
719
|
+
expect(writes.map((w) => w.op)).toEqual(["requestCancel"]);
|
|
720
|
+
// The agent name travels with it so a run can't be stopped
|
|
721
|
+
// through a different agent's action.
|
|
722
|
+
expect(writes[0].agent).toBe("helper");
|
|
723
|
+
});
|
|
724
|
+
|
|
725
|
+
test("cancel without a runId, and input without either, are refused", async () => {
|
|
726
|
+
const def = agent({});
|
|
727
|
+
const call = (a: Record<string, unknown>) =>
|
|
728
|
+
(def.handler as unknown as (
|
|
729
|
+
c: ActionCtx,
|
|
730
|
+
a: Record<string, unknown>,
|
|
731
|
+
) => Promise<unknown>)(mockCtx([]).ctx, { __agentName: "helper", ...a });
|
|
732
|
+
await expect(call({ cancel: true })).rejects.toThrow(/runId/);
|
|
733
|
+
await expect(call({})).rejects.toThrow(/input is required/);
|
|
734
|
+
});
|
|
735
|
+
|
|
736
|
+
test("a cancel raised while a tool runs aborts ctx.signal and skips the rest", async () => {
|
|
737
|
+
const seen: string[] = [];
|
|
738
|
+
const def = agent({
|
|
739
|
+
tools: {
|
|
740
|
+
first: {
|
|
741
|
+
description: "first",
|
|
742
|
+
handler: async (toolCtx) => {
|
|
743
|
+
seen.push("first");
|
|
744
|
+
// The cancel lands while this handler is in flight; the
|
|
745
|
+
// poller is what notices, so wait past one interval.
|
|
746
|
+
return await new Promise((resolve) => {
|
|
747
|
+
const timer = setInterval(() => {
|
|
748
|
+
if (toolCtx.signal?.aborted) {
|
|
749
|
+
clearInterval(timer);
|
|
750
|
+
resolve("aborted early");
|
|
751
|
+
}
|
|
752
|
+
}, 10);
|
|
753
|
+
});
|
|
754
|
+
},
|
|
755
|
+
},
|
|
756
|
+
second: {
|
|
757
|
+
description: "second",
|
|
758
|
+
handler: async () => {
|
|
759
|
+
seen.push("second");
|
|
760
|
+
return "ran";
|
|
761
|
+
},
|
|
762
|
+
},
|
|
763
|
+
},
|
|
764
|
+
});
|
|
765
|
+
const { ctx, state } = mockCtx([
|
|
766
|
+
{
|
|
767
|
+
model: "m",
|
|
768
|
+
content: [
|
|
769
|
+
{ type: "tool_use", id: "tu_1", name: "first", input: {} },
|
|
770
|
+
{ type: "tool_use", id: "tu_2", name: "second", input: {} },
|
|
771
|
+
],
|
|
772
|
+
stop_reason: "tool_use",
|
|
773
|
+
usage: { input_tokens: 1, output_tokens: 1 },
|
|
774
|
+
},
|
|
775
|
+
done("never reached"),
|
|
776
|
+
]);
|
|
777
|
+
setTimeout(() => {
|
|
778
|
+
state.cancelRequested = true;
|
|
779
|
+
}, 30);
|
|
780
|
+
|
|
781
|
+
const result = await (def.handler as unknown as (
|
|
782
|
+
c: ActionCtx,
|
|
783
|
+
a: Record<string, unknown>,
|
|
784
|
+
) => Promise<Record<string, unknown>>)(ctx, {
|
|
785
|
+
input: "go",
|
|
786
|
+
__agentName: "helper",
|
|
787
|
+
});
|
|
788
|
+
|
|
789
|
+
expect(result.cancelled).toBe(true);
|
|
790
|
+
// The running handler saw the abort, and the queued one never
|
|
791
|
+
// started.
|
|
792
|
+
expect(seen).toEqual(["first"]);
|
|
793
|
+
expect(state.status).toBe("cancelled");
|
|
794
|
+
}, 10_000);
|
|
795
|
+
|
|
796
|
+
test("the default step budget is high enough for a real tool loop", async () => {
|
|
797
|
+
// 16 was the old default and is nowhere near enough for an agent
|
|
798
|
+
// that reads files before answering.
|
|
799
|
+
const def = agent({
|
|
800
|
+
tools: { step: { description: "one step", handler: async () => "ok" } },
|
|
801
|
+
});
|
|
802
|
+
const script: LlmCompleteResponse[] = [];
|
|
803
|
+
for (let i = 0; i < 40; i += 1) script.push(wantsTool("step", { i }));
|
|
804
|
+
script.push(done("Finished."));
|
|
805
|
+
const { ctx } = mockCtx(script);
|
|
806
|
+
const result = await (def.handler as unknown as (
|
|
807
|
+
c: ActionCtx,
|
|
808
|
+
a: Record<string, unknown>,
|
|
809
|
+
) => Promise<Record<string, unknown>>)(ctx, {
|
|
810
|
+
input: "go",
|
|
811
|
+
__agentName: "helper",
|
|
812
|
+
});
|
|
813
|
+
expect(result.text).toBe("Finished.");
|
|
814
|
+
expect(result.steps).toBe(41);
|
|
815
|
+
});
|
|
816
|
+
|
|
817
|
+
test("cumulative steps carry across invocations", async () => {
|
|
818
|
+
const def = agent({});
|
|
819
|
+
const { ctx, writes } = mockCtx([done("Next.")], {
|
|
820
|
+
run: { status: "completed", steps: 7 },
|
|
821
|
+
});
|
|
822
|
+
await (def.handler as unknown as (
|
|
823
|
+
c: ActionCtx,
|
|
824
|
+
a: Record<string, unknown>,
|
|
825
|
+
) => Promise<unknown>)(ctx, {
|
|
826
|
+
input: "again",
|
|
827
|
+
runId: "run_9",
|
|
828
|
+
__agentName: "helper",
|
|
829
|
+
});
|
|
830
|
+
// This invocation ran one step on top of the stored 7.
|
|
831
|
+
const drains = writes.filter((w) => w.op === "drainInput");
|
|
832
|
+
expect(Number(drains[drains.length - 1].steps)).toBe(8);
|
|
833
|
+
});
|
|
834
|
+
|
|
461
835
|
test("llm failure marks the run failed and rethrows", async () => {
|
|
462
836
|
const def = agent({});
|
|
463
837
|
const { ctx, writes } = mockCtx([]); // script exhausted immediately
|
package/src/agent.ts
CHANGED
|
@@ -74,8 +74,11 @@ export interface AgentDefinition {
|
|
|
74
74
|
tools?: Record<string, AgentTool>;
|
|
75
75
|
/** Model override (subject to the server's allowlist). */
|
|
76
76
|
model?: string;
|
|
77
|
-
/** Max model↔tool round-trips per invocation (default
|
|
78
|
-
* the cap fails the run rather than looping forever.
|
|
77
|
+
/** Max model↔tool round-trips per invocation (default 64). Hitting
|
|
78
|
+
* the cap fails the run rather than looping forever. Steering input
|
|
79
|
+
* drained mid-invocation extends the same invocation, so this bounds
|
|
80
|
+
* a steered turn too. The run row's cumulative `steps` counts every
|
|
81
|
+
* invocation and is not capped. */
|
|
79
82
|
maxSteps?: number;
|
|
80
83
|
/** max_tokens per completion (default: server default). */
|
|
81
84
|
maxTokens?: number;
|
|
@@ -88,13 +91,19 @@ export interface AgentDefinition {
|
|
|
88
91
|
|
|
89
92
|
/** The synthesized action's args. */
|
|
90
93
|
export interface AgentCallArgs {
|
|
91
|
-
/** The user's message for this turn. */
|
|
92
|
-
input
|
|
94
|
+
/** The user's message for this turn. Required unless `cancel`. */
|
|
95
|
+
input?: string;
|
|
93
96
|
/** Continue an existing run (must belong to the caller and this
|
|
94
|
-
* agent). Omit to start a new run.
|
|
97
|
+
* agent). Omit to start a new run. Sending while the run is already
|
|
98
|
+
* generating queues the message for that generation rather than
|
|
99
|
+
* refusing it — see {@link AgentResult.queued}. */
|
|
95
100
|
runId?: string;
|
|
96
101
|
/** Optional display title, stored on new runs. */
|
|
97
102
|
title?: string;
|
|
103
|
+
/** Ask the run to stop. Requires `runId`, ignores `input`, and
|
|
104
|
+
* returns as soon as the request is recorded — a live generation
|
|
105
|
+
* stops at its next turn boundary. */
|
|
106
|
+
cancel?: boolean;
|
|
98
107
|
}
|
|
99
108
|
|
|
100
109
|
/** What the agent action resolves with (also the `event: result`
|
|
@@ -103,9 +112,16 @@ export interface AgentResult {
|
|
|
103
112
|
runId: string;
|
|
104
113
|
/** Concatenated text of the final assistant message. */
|
|
105
114
|
text: string;
|
|
106
|
-
/** Round-trips consumed. */
|
|
115
|
+
/** Round-trips consumed by THIS invocation. */
|
|
107
116
|
steps: number;
|
|
108
117
|
usage: { input_tokens: number; output_tokens: number };
|
|
118
|
+
/** The message was queued onto a generation already in flight
|
|
119
|
+
* instead of starting a turn. No model call happened on this call;
|
|
120
|
+
* the running loop picks the message up at its next boundary. */
|
|
121
|
+
queued?: boolean;
|
|
122
|
+
/** The run stopped because cancel was requested. `text` holds
|
|
123
|
+
* whatever the model had produced by then. */
|
|
124
|
+
cancelled?: boolean;
|
|
109
125
|
}
|
|
110
126
|
|
|
111
127
|
// ---------------------------------------------------------------------------
|
|
@@ -179,9 +195,10 @@ export const AGENT_MARKER = "__pylonAgent";
|
|
|
179
195
|
export function agent(def: AgentDefinition): FnDefinition<AgentCallArgs, AgentResult> {
|
|
180
196
|
const fnDef = action({
|
|
181
197
|
args: {
|
|
182
|
-
input: v.string(),
|
|
198
|
+
input: v.optional(v.string()),
|
|
183
199
|
runId: v.optional(v.string()),
|
|
184
200
|
title: v.optional(v.string()),
|
|
201
|
+
cancel: v.optional(v.boolean()),
|
|
185
202
|
},
|
|
186
203
|
auth: def.auth ?? "user",
|
|
187
204
|
timeout: def.timeout ?? 600,
|
|
@@ -204,7 +221,13 @@ export function isAgentDefinition(value: unknown): boolean {
|
|
|
204
221
|
// The loop
|
|
205
222
|
// ---------------------------------------------------------------------------
|
|
206
223
|
|
|
207
|
-
const DEFAULT_MAX_STEPS =
|
|
224
|
+
const DEFAULT_MAX_STEPS = 64;
|
|
225
|
+
|
|
226
|
+
/** How often the loop asks whether cancel was requested while a tool
|
|
227
|
+
* handler is running. Only runs during a tool batch: the host blocks
|
|
228
|
+
* its per-call read loop for the whole of `ctx.llm.stream`, so no RPC
|
|
229
|
+
* the child issues during a generation would be serviced anyway. */
|
|
230
|
+
const CANCEL_POLL_MS = 2000;
|
|
208
231
|
|
|
209
232
|
/** Tool results persist into AgentMessage rows and replay into every
|
|
210
233
|
* later completion's context — cap them so one oversized return can't
|
|
@@ -223,6 +246,30 @@ interface StoredMessage {
|
|
|
223
246
|
content: unknown;
|
|
224
247
|
}
|
|
225
248
|
|
|
249
|
+
/** What one turn-boundary `drainInput` reports back. */
|
|
250
|
+
interface DrainResult {
|
|
251
|
+
input: string[];
|
|
252
|
+
cancelRequested: boolean;
|
|
253
|
+
completed: boolean;
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
/** Give tool handlers a ctx whose `signal` also aborts on cancel, so a
|
|
257
|
+
* handler that already threads `ctx.signal` into fetch stops with the
|
|
258
|
+
* run. Prototype-based rather than a spread: ctx's methods keep
|
|
259
|
+
* working when invoked on the derived object. */
|
|
260
|
+
function ctxWithSignal(ctx: ActionCtx, signal: AbortSignal): ActionCtx {
|
|
261
|
+
const combined =
|
|
262
|
+
ctx.signal && typeof AbortSignal.any === "function"
|
|
263
|
+
? AbortSignal.any([ctx.signal, signal])
|
|
264
|
+
: signal;
|
|
265
|
+
const derived = Object.create(ctx) as ActionCtx;
|
|
266
|
+
Object.defineProperty(derived, "signal", {
|
|
267
|
+
value: combined,
|
|
268
|
+
enumerable: true,
|
|
269
|
+
});
|
|
270
|
+
return derived;
|
|
271
|
+
}
|
|
272
|
+
|
|
226
273
|
async function runAgentLoop(
|
|
227
274
|
def: AgentDefinition,
|
|
228
275
|
ctx: ActionCtx,
|
|
@@ -235,18 +282,59 @@ async function runAgentLoop(
|
|
|
235
282
|
(args as unknown as Record<string, unknown>).__agentName?.toString() ??
|
|
236
283
|
"agent";
|
|
237
284
|
|
|
285
|
+
const noUsage = { input_tokens: 0, output_tokens: 0 };
|
|
286
|
+
|
|
287
|
+
// Cancel is a control call, not a turn: record the request and
|
|
288
|
+
// return. Ownership (and the agent match) is enforced inside the
|
|
289
|
+
// internal mutation, which runs under this caller's auth.
|
|
290
|
+
if (args.cancel === true) {
|
|
291
|
+
if (!args.runId) {
|
|
292
|
+
throw ctx.error(
|
|
293
|
+
"AGENT_CANCEL_NEEDS_RUN",
|
|
294
|
+
"cancel requires the runId of the run to stop",
|
|
295
|
+
);
|
|
296
|
+
}
|
|
297
|
+
await ctx.runMutation("__pylon_agent_write", {
|
|
298
|
+
op: "requestCancel",
|
|
299
|
+
runId: args.runId,
|
|
300
|
+
agent: agentName,
|
|
301
|
+
});
|
|
302
|
+
return {
|
|
303
|
+
runId: args.runId,
|
|
304
|
+
text: "",
|
|
305
|
+
steps: 0,
|
|
306
|
+
usage: noUsage,
|
|
307
|
+
cancelled: true,
|
|
308
|
+
};
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
const input = args.input ?? "";
|
|
312
|
+
if (input === "") {
|
|
313
|
+
throw ctx.error(
|
|
314
|
+
"AGENT_INPUT_REQUIRED",
|
|
315
|
+
"input is required (pass cancel: true to stop a run instead)",
|
|
316
|
+
);
|
|
317
|
+
}
|
|
318
|
+
|
|
238
319
|
// A run left "running" longer than the agent's timeout is a dead
|
|
239
320
|
// generation (the process died before the terminal status write) —
|
|
240
|
-
// continuations may take it over instead of
|
|
321
|
+
// continuations may take it over instead of waiting forever.
|
|
241
322
|
const staleMs = Math.max(def.timeout ?? 600, 60) * 1000;
|
|
242
323
|
|
|
243
324
|
// 1. Create or load the run (ownership enforced inside the internal
|
|
244
325
|
// fns, which run under this caller's auth).
|
|
245
326
|
let runId: string;
|
|
246
327
|
let history: LlmMessage[] = [];
|
|
328
|
+
let priorSteps = 0;
|
|
247
329
|
if (args.runId) {
|
|
248
330
|
const loaded = await ctx.runQuery<{
|
|
249
|
-
run: {
|
|
331
|
+
run: {
|
|
332
|
+
id: string;
|
|
333
|
+
agent: string;
|
|
334
|
+
status: string;
|
|
335
|
+
updatedAt?: string;
|
|
336
|
+
steps?: number;
|
|
337
|
+
};
|
|
250
338
|
messages: StoredMessage[];
|
|
251
339
|
}>("__pylon_agent_read", { runId: args.runId });
|
|
252
340
|
if (loaded.run.agent !== agentName) {
|
|
@@ -258,13 +346,28 @@ async function runAgentLoop(
|
|
|
258
346
|
if (loaded.run.status === "running") {
|
|
259
347
|
const updatedAt = Date.parse(String(loaded.run.updatedAt ?? "")) || 0;
|
|
260
348
|
if (Date.now() - updatedAt < staleMs) {
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
349
|
+
// The generation is alive. Hand it the message rather than
|
|
350
|
+
// refusing the call: the loop drains the queue at its next
|
|
351
|
+
// turn boundary, which is what lets a user steer mid-run.
|
|
352
|
+
const queued = await ctx.runMutation<{ queued: number }>(
|
|
353
|
+
"__pylon_agent_write",
|
|
354
|
+
{ op: "enqueueInput", runId: loaded.run.id, input },
|
|
355
|
+
);
|
|
356
|
+
ctx.stream.writeEvent(
|
|
357
|
+
"queued",
|
|
358
|
+
JSON.stringify({ runId: loaded.run.id, queued: queued.queued }),
|
|
264
359
|
);
|
|
360
|
+
return {
|
|
361
|
+
runId: loaded.run.id,
|
|
362
|
+
text: "",
|
|
363
|
+
steps: 0,
|
|
364
|
+
usage: noUsage,
|
|
365
|
+
queued: true,
|
|
366
|
+
};
|
|
265
367
|
}
|
|
266
368
|
}
|
|
267
369
|
runId = loaded.run.id;
|
|
370
|
+
priorSteps = Number(loaded.run.steps) || 0;
|
|
268
371
|
history = loaded.messages.map(storedToLlmMessage);
|
|
269
372
|
} else {
|
|
270
373
|
const created = await ctx.runMutation<{ id: string }>(
|
|
@@ -293,27 +396,33 @@ async function runAgentLoop(
|
|
|
293
396
|
|
|
294
397
|
// A crash between persisting an assistant tool_use turn and its
|
|
295
398
|
// tool_results leaves a transcript the LLM API rejects on replay.
|
|
296
|
-
// Repair with synthetic error results
|
|
399
|
+
// Repair with synthetic error results, carried in the SAME user
|
|
400
|
+
// message as this turn's input: two user messages in a row are
|
|
401
|
+
// rejected as well, so appending the repair separately would trade
|
|
402
|
+
// one unreplayable transcript for another.
|
|
403
|
+
let repairs: LlmContentBlock[] = [];
|
|
297
404
|
const lastMsg = history[history.length - 1];
|
|
298
405
|
if (lastMsg?.role === "assistant" && Array.isArray(lastMsg.content)) {
|
|
299
|
-
|
|
300
|
-
(
|
|
301
|
-
(b
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
406
|
+
repairs = lastMsg.content
|
|
407
|
+
.filter(
|
|
408
|
+
(b): b is Extract<LlmContentBlock, { type: "tool_use" }> =>
|
|
409
|
+
(b as { type?: string }).type === "tool_use",
|
|
410
|
+
)
|
|
411
|
+
.map((b) => ({
|
|
305
412
|
type: "tool_result",
|
|
306
413
|
tool_use_id: b.id,
|
|
307
414
|
content: "tool execution was interrupted",
|
|
308
415
|
is_error: true,
|
|
309
416
|
}));
|
|
310
|
-
await write({ op: "appendMessage", role: "user", content: repairs });
|
|
311
|
-
history.push({ role: "user", content: repairs });
|
|
312
|
-
}
|
|
313
417
|
}
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
418
|
+
// Plain string when there is nothing to repair — the common case
|
|
419
|
+
// stays a simple transcript row.
|
|
420
|
+
const opening: string | LlmContentBlock[] =
|
|
421
|
+
repairs.length > 0
|
|
422
|
+
? [...repairs, { type: "text", text: input }]
|
|
423
|
+
: input;
|
|
424
|
+
await write({ op: "appendMessage", role: "user", content: opening });
|
|
425
|
+
history.push({ role: "user", content: opening });
|
|
317
426
|
|
|
318
427
|
// Tool declarations for the model.
|
|
319
428
|
const tools: LlmTool[] = Object.entries(def.tools ?? {}).map(
|
|
@@ -328,6 +437,13 @@ async function runAgentLoop(
|
|
|
328
437
|
const usage = { input_tokens: 0, output_tokens: 0 };
|
|
329
438
|
let finalText = "";
|
|
330
439
|
let steps = 0;
|
|
440
|
+
let cancelled = false;
|
|
441
|
+
let settled = false;
|
|
442
|
+
|
|
443
|
+
const drain = (extra: Record<string, unknown>) =>
|
|
444
|
+
write({ op: "drainInput", steps: priorSteps + steps, ...extra }) as Promise<
|
|
445
|
+
unknown
|
|
446
|
+
> as Promise<DrainResult>;
|
|
331
447
|
|
|
332
448
|
try {
|
|
333
449
|
for (;;) {
|
|
@@ -364,16 +480,137 @@ async function runAgentLoop(
|
|
|
364
480
|
.map((b) => b.text)
|
|
365
481
|
.join("");
|
|
366
482
|
|
|
367
|
-
if (res.stop_reason !== "tool_use")
|
|
483
|
+
if (res.stop_reason !== "tool_use") {
|
|
484
|
+
// Terminal boundary. Settling and draining in one transaction
|
|
485
|
+
// is what stops a message that lands right now from being
|
|
486
|
+
// stranded on a completed run.
|
|
487
|
+
const pending = await drain({ completeIfEmpty: true });
|
|
488
|
+
// Whatever was queued has already left the row, so persist it
|
|
489
|
+
// either way — cancelling must not silently swallow words the
|
|
490
|
+
// user had typed and can still see queued in the UI.
|
|
491
|
+
const text = pending.input.join("\n\n");
|
|
492
|
+
if (text !== "") {
|
|
493
|
+
await write({ op: "appendMessage", role: "user", content: text });
|
|
494
|
+
history.push({ role: "user", content: text });
|
|
495
|
+
}
|
|
496
|
+
if (pending.cancelRequested) {
|
|
497
|
+
cancelled = true;
|
|
498
|
+
break;
|
|
499
|
+
}
|
|
500
|
+
if (pending.completed) {
|
|
501
|
+
settled = true;
|
|
502
|
+
break;
|
|
503
|
+
}
|
|
504
|
+
// Steered after the model had stopped: the queued text became
|
|
505
|
+
// the next user turn and the same invocation keeps going.
|
|
506
|
+
continue;
|
|
507
|
+
}
|
|
368
508
|
|
|
369
509
|
// 3. Execute every requested tool; failures become is_error
|
|
370
510
|
// results the model can react to rather than run-fatal throws.
|
|
371
|
-
const
|
|
372
|
-
|
|
373
|
-
|
|
511
|
+
const batch = await runToolBatch(def, ctx, runId, res.content);
|
|
512
|
+
|
|
513
|
+
// Tool boundary. Every tool_use needs its tool_result even when
|
|
514
|
+
// the batch stopped early, or the transcript can't be replayed.
|
|
515
|
+
const pending = await drain({});
|
|
516
|
+
const stopping = pending.cancelRequested || batch.cancelled;
|
|
517
|
+
const blocks: LlmContentBlock[] = [...batch.results];
|
|
518
|
+
for (const text of pending.input) {
|
|
519
|
+
blocks.push({ type: "text", text });
|
|
520
|
+
}
|
|
521
|
+
await write({ op: "appendMessage", role: "user", content: blocks });
|
|
522
|
+
history.push({ role: "user", content: blocks });
|
|
523
|
+
if (stopping) {
|
|
524
|
+
cancelled = true;
|
|
525
|
+
break;
|
|
526
|
+
}
|
|
527
|
+
}
|
|
528
|
+
} catch (err) {
|
|
529
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
530
|
+
// Best-effort — the failure we surface is the loop's, not the
|
|
531
|
+
// bookkeeping write's.
|
|
532
|
+
await write({
|
|
533
|
+
op: "setStatus",
|
|
534
|
+
status: "failed",
|
|
535
|
+
error: message,
|
|
536
|
+
steps: priorSteps + steps,
|
|
537
|
+
}).catch(() => {});
|
|
538
|
+
throw err;
|
|
539
|
+
}
|
|
540
|
+
|
|
541
|
+
if (cancelled) {
|
|
542
|
+
await write({
|
|
543
|
+
op: "setStatus",
|
|
544
|
+
status: "cancelled",
|
|
545
|
+
steps: priorSteps + steps,
|
|
546
|
+
});
|
|
547
|
+
return { runId, text: finalText, steps, usage, cancelled: true };
|
|
548
|
+
}
|
|
549
|
+
// `settled` means drainInput already wrote the terminal status in
|
|
550
|
+
// the same transaction that proved nothing was queued.
|
|
551
|
+
if (!settled) {
|
|
552
|
+
await write({
|
|
553
|
+
op: "setStatus",
|
|
554
|
+
status: "completed",
|
|
555
|
+
steps: priorSteps + steps,
|
|
556
|
+
});
|
|
557
|
+
}
|
|
558
|
+
return { runId, text: finalText, steps, usage };
|
|
559
|
+
}
|
|
560
|
+
|
|
561
|
+
/** Run one turn's tool calls, watching for a cancel while they run.
|
|
562
|
+
*
|
|
563
|
+
* The watch is a poll rather than a push because a cancel can be
|
|
564
|
+
* requested on a different machine than the loop runs on — the run
|
|
565
|
+
* row is the only thing both sides share. It runs only for the
|
|
566
|
+
* duration of the batch: during `ctx.llm.stream` the host blocks its
|
|
567
|
+
* per-call read loop, so an RPC issued then would not be answered. */
|
|
568
|
+
async function runToolBatch(
|
|
569
|
+
def: AgentDefinition,
|
|
570
|
+
ctx: ActionCtx,
|
|
571
|
+
runId: string,
|
|
572
|
+
turn: LlmContentBlock[],
|
|
573
|
+
): Promise<{ results: LlmContentBlock[]; cancelled: boolean }> {
|
|
574
|
+
const calls = turn.filter(
|
|
575
|
+
(b): b is Extract<LlmContentBlock, { type: "tool_use" }> =>
|
|
576
|
+
b.type === "tool_use",
|
|
577
|
+
);
|
|
578
|
+
const results: LlmContentBlock[] = [];
|
|
579
|
+
if (calls.length === 0) return { results, cancelled: false };
|
|
580
|
+
|
|
581
|
+
const controller = new AbortController();
|
|
582
|
+
const toolCtx = ctxWithSignal(ctx, controller.signal);
|
|
583
|
+
let cancelled = false;
|
|
584
|
+
let polling = false;
|
|
585
|
+
const timer = setInterval(() => {
|
|
586
|
+
// Skip rather than stack: a slow poll must not queue more.
|
|
587
|
+
if (polling || cancelled) return;
|
|
588
|
+
polling = true;
|
|
589
|
+
void ctx
|
|
590
|
+
.runQuery<{ cancelRequested: boolean }>("__pylon_agent_poll", { runId })
|
|
591
|
+
.then((s) => {
|
|
592
|
+
if (s.cancelRequested && !cancelled) {
|
|
593
|
+
cancelled = true;
|
|
594
|
+
controller.abort(new Error("agent run cancelled"));
|
|
595
|
+
}
|
|
596
|
+
})
|
|
597
|
+
.catch(() => {})
|
|
598
|
+
.finally(() => {
|
|
599
|
+
polling = false;
|
|
600
|
+
});
|
|
601
|
+
}, CANCEL_POLL_MS);
|
|
602
|
+
|
|
603
|
+
try {
|
|
604
|
+
for (const block of calls) {
|
|
605
|
+
let content: string;
|
|
606
|
+
let isError = false;
|
|
607
|
+
if (cancelled) {
|
|
608
|
+
// Stop starting new work, but still answer the call so the
|
|
609
|
+
// model sees why it has no result.
|
|
610
|
+
content = "cancelled by user";
|
|
611
|
+
isError = true;
|
|
612
|
+
} else {
|
|
374
613
|
const tool = def.tools?.[block.name];
|
|
375
|
-
let content: string;
|
|
376
|
-
let isError = false;
|
|
377
614
|
if (!tool) {
|
|
378
615
|
content = `Unknown tool "${block.name}"`;
|
|
379
616
|
isError = true;
|
|
@@ -385,7 +622,7 @@ async function runAgentLoop(
|
|
|
385
622
|
throw new Error(`Invalid tool input: ${check.errors.join("; ")}`);
|
|
386
623
|
}
|
|
387
624
|
}
|
|
388
|
-
const value = await tool.handler(
|
|
625
|
+
const value = await tool.handler(toolCtx, block.input);
|
|
389
626
|
content =
|
|
390
627
|
typeof value === "string" ? value : JSON.stringify(value ?? null);
|
|
391
628
|
} catch (err) {
|
|
@@ -393,35 +630,25 @@ async function runAgentLoop(
|
|
|
393
630
|
isError = true;
|
|
394
631
|
}
|
|
395
632
|
}
|
|
396
|
-
content = truncateToolResult(content);
|
|
397
|
-
// Announce the tool call on the stream so live UIs can render
|
|
398
|
-
// "using searchDocs…" without polling the message rows.
|
|
399
|
-
ctx.stream.writeEvent(
|
|
400
|
-
"tool",
|
|
401
|
-
JSON.stringify({ name: block.name, input: block.input, isError }),
|
|
402
|
-
);
|
|
403
|
-
results.push({
|
|
404
|
-
type: "tool_result",
|
|
405
|
-
tool_use_id: block.id,
|
|
406
|
-
content,
|
|
407
|
-
...(isError ? { is_error: true } : {}),
|
|
408
|
-
});
|
|
409
633
|
}
|
|
410
|
-
|
|
411
|
-
|
|
634
|
+
content = truncateToolResult(content);
|
|
635
|
+
// Announce the tool call on the stream so live UIs can render
|
|
636
|
+
// "using searchDocs…" without polling the message rows.
|
|
637
|
+
ctx.stream.writeEvent(
|
|
638
|
+
"tool",
|
|
639
|
+
JSON.stringify({ name: block.name, input: block.input, isError }),
|
|
640
|
+
);
|
|
641
|
+
results.push({
|
|
642
|
+
type: "tool_result",
|
|
643
|
+
tool_use_id: block.id,
|
|
644
|
+
content,
|
|
645
|
+
...(isError ? { is_error: true } : {}),
|
|
646
|
+
});
|
|
412
647
|
}
|
|
413
|
-
}
|
|
414
|
-
|
|
415
|
-
// Best-effort — the failure we surface is the loop's, not the
|
|
416
|
-
// bookkeeping write's.
|
|
417
|
-
await write({ op: "setStatus", status: "failed", error: message }).catch(
|
|
418
|
-
() => {},
|
|
419
|
-
);
|
|
420
|
-
throw err;
|
|
648
|
+
} finally {
|
|
649
|
+
clearInterval(timer);
|
|
421
650
|
}
|
|
422
|
-
|
|
423
|
-
await write({ op: "setStatus", status: "completed" });
|
|
424
|
-
return { runId, text: finalText, steps, usage };
|
|
651
|
+
return { results, cancelled };
|
|
425
652
|
}
|
|
426
653
|
|
|
427
654
|
function storedToLlmMessage(m: StoredMessage): LlmMessage {
|