@agent-compose/sdk 0.5.0 → 0.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/active-step.d.ts +60 -0
- package/dist/agent/agent-loop-steer.test.d.ts +1 -0
- package/dist/agent/agent-loop.d.ts +46 -0
- package/dist/agent/async-queue.d.ts +29 -0
- package/dist/agent/protocol.d.ts +9 -1
- package/dist/agent/resolve-agent-id.test.d.ts +1 -0
- package/dist/agent/run-agent.d.ts +16 -3
- package/dist/agent/steer-control.d.ts +57 -0
- package/dist/agent/steer-control.test.d.ts +1 -0
- package/dist/client.d.ts +161 -0
- package/dist/index.d.ts +7 -4
- package/dist/index.js +1341 -157
- package/dist/pause/__tests__/agent-loop-checkpoint.test.d.ts +1 -0
- package/dist/pause/__tests__/checkpoint.test.d.ts +1 -0
- package/dist/pause/__tests__/errors.test.d.ts +1 -0
- package/dist/pause/__tests__/manager.test.d.ts +1 -0
- package/dist/pause/__tests__/pause-core.test.d.ts +1 -0
- package/dist/pause/__tests__/state-dir.test.d.ts +1 -0
- package/dist/pause/__tests__/wrappers.test.d.ts +1 -0
- package/dist/pause/checkpoint.d.ts +28 -0
- package/dist/pause/errors.d.ts +52 -0
- package/dist/pause/manager.d.ts +63 -0
- package/dist/pause/pause-core.d.ts +101 -0
- package/dist/pause/state-dir.d.ts +80 -0
- package/dist/pause/wrappers.d.ts +41 -0
- package/dist/request-context/request-context.d.ts +12 -0
- package/dist/runtimes/claude.d.ts +6 -0
- package/dist/runtimes/openai-desktop.d.ts +2 -0
- package/dist/runtimes/openai-desktop.js +1338 -156
- package/dist/runtimes/vercel.d.ts +12 -0
- package/dist/runtimes/vercel.js +50 -7
- package/dist/runtimes/vercel.test.d.ts +1 -0
- package/dist/sse.d.ts +2 -3
- package/dist/step-invocation/index.d.ts +2 -2
- package/dist/step-invocation/invoker.d.ts +3 -0
- package/dist/step-invocation/protocol.d.ts +12 -0
- package/dist/step-invocation/server.d.ts +1 -0
- package/dist/step-invocation/types.d.ts +40 -5
- package/dist/types/events.d.ts +9 -0
- package/dist/types/execution-context.d.ts +25 -0
- package/dist/types/protocol.d.ts +8 -0
- package/dist/types/runtime.d.ts +55 -0
- package/dist/types/sandbox.d.ts +6 -1
- package/dist/utils/schemas.d.ts +2 -0
- package/dist/workflow-steps/__tests__/pause-wiring.test.d.ts +1 -0
- package/dist/workflow-steps/index.d.ts +2 -0
- package/dist/workflow-steps/observability.d.ts +43 -11
- package/dist/workflow-steps/run-callback.d.ts +39 -0
- package/dist/workflow-steps/runner.d.ts +8 -0
- package/package.json +1 -1
- package/src/active-step.ts +124 -0
- package/src/agent/agent-loop.ts +253 -19
- package/src/agent/async-queue.ts +61 -0
- package/src/agent/protocol.ts +12 -2
- package/src/agent/run-agent.ts +184 -8
- package/src/agent/steer-control.ts +125 -0
- package/src/client.ts +277 -0
- package/src/index.ts +18 -2
- package/src/pause/checkpoint.ts +44 -0
- package/src/pause/errors.ts +70 -0
- package/src/pause/manager.ts +177 -0
- package/src/pause/pause-core.ts +267 -0
- package/src/pause/state-dir.ts +262 -0
- package/src/pause/wrappers.ts +79 -0
- package/src/request-context/request-context.ts +17 -2
- package/src/runtimes/claude.ts +101 -6
- package/src/runtimes/openai-desktop.ts +11 -0
- package/src/runtimes/vercel.ts +26 -0
- package/src/sandbox.ts +45 -17
- package/src/sse.ts +8 -6
- package/src/step-invocation/index.ts +2 -1
- package/src/step-invocation/invoker.ts +107 -29
- package/src/step-invocation/protocol.ts +16 -0
- package/src/step-invocation/server.ts +45 -12
- package/src/step-invocation/types.ts +43 -7
- package/src/tools/coding.ts +16 -5
- package/src/types/events.ts +9 -0
- package/src/types/execution-context.ts +25 -0
- package/src/types/protocol.ts +8 -0
- package/src/types/runtime.ts +52 -0
- package/src/types/sandbox.ts +10 -1
- package/src/types/workflow.ts +6 -1
- package/src/utils/bundler.ts +8 -3
- package/src/utils/schemas.ts +2 -0
- package/src/workflow-steps/index.ts +3 -0
- package/src/workflow-steps/observability.ts +84 -13
- package/src/workflow-steps/run-callback.ts +72 -0
- package/src/workflow-steps/runner.ts +70 -8
- package/dist/utils/discovery.d.ts +0 -2
- package/src/utils/discovery.ts +0 -4
package/src/tools/coding.ts
CHANGED
|
@@ -10,6 +10,13 @@ function resolvePath(path: string, cwd?: string): string {
|
|
|
10
10
|
return cwd && !isAbsolute(path) ? join(cwd, path) : path;
|
|
11
11
|
}
|
|
12
12
|
|
|
13
|
+
function formatCommandFailure(command: string, result: { exitCode: number; stdout: string; stderr: string }): string {
|
|
14
|
+
const parts = [`Command failed with exit code ${result.exitCode}: ${command}`];
|
|
15
|
+
if (result.stderr.trim()) parts.push(`stderr:\n${result.stderr}`);
|
|
16
|
+
if (result.stdout.trim()) parts.push(`stdout:\n${result.stdout}`);
|
|
17
|
+
return parts.join("\n");
|
|
18
|
+
}
|
|
19
|
+
|
|
13
20
|
async function readFile(sandbox: SandboxProvider, path: string, opts?: { offset?: number; limit?: number; cwd?: string }): Promise<string> {
|
|
14
21
|
const script = `
|
|
15
22
|
const fs = require("fs");
|
|
@@ -26,14 +33,18 @@ for (let i = start; i <= end; i++) console.log(String(i) + ": " + lines[i - 1]);
|
|
|
26
33
|
.filter(Boolean)
|
|
27
34
|
.map(q)
|
|
28
35
|
.join(" ");
|
|
29
|
-
const
|
|
30
|
-
|
|
36
|
+
const command = `node -e ${q(script)} ${args}`;
|
|
37
|
+
const result = await sandbox.commands.run(command, { cwd: opts?.cwd, timeoutMs: 30_000 });
|
|
38
|
+
if (result.exitCode !== 0) throw new Error(formatCommandFailure(command, result));
|
|
39
|
+
return result.stdout;
|
|
31
40
|
}
|
|
32
41
|
|
|
33
42
|
async function readRawFile(sandbox: SandboxProvider, path: string, cwd?: string): Promise<string> {
|
|
34
43
|
const script = `const fs = require("fs"); process.stdout.write(fs.readFileSync(process.argv[1], "utf8"));`;
|
|
35
|
-
const
|
|
36
|
-
|
|
44
|
+
const command = `node -e ${q(script)} ${q(path)}`;
|
|
45
|
+
const result = await sandbox.commands.run(command, { cwd, timeoutMs: 30_000 });
|
|
46
|
+
if (result.exitCode !== 0) throw new Error(formatCommandFailure(command, result));
|
|
47
|
+
return result.stdout;
|
|
37
48
|
}
|
|
38
49
|
|
|
39
50
|
export interface CodingTool<TInput extends Record<string, unknown> = Record<string, unknown>> {
|
|
@@ -118,7 +129,7 @@ export const bashTool: CodingTool<{
|
|
|
118
129
|
cwd: cwd ?? ctx.cwd,
|
|
119
130
|
timeoutMs: timeoutMs ?? 120_000,
|
|
120
131
|
});
|
|
121
|
-
if (result.exitCode !== 0) throw new Error(
|
|
132
|
+
if (result.exitCode !== 0) throw new Error(formatCommandFailure(command, result));
|
|
122
133
|
return result.stdout;
|
|
123
134
|
},
|
|
124
135
|
};
|
package/src/types/events.ts
CHANGED
|
@@ -15,6 +15,15 @@ export type RunEvent =
|
|
|
15
15
|
seq?: number;
|
|
16
16
|
agentId: string;
|
|
17
17
|
label: string;
|
|
18
|
+
/** Tool whitelist passed to `agent({ tools: [...] })`. Drives the
|
|
19
|
+
* per-agent "tools" badge on the dashboard. */
|
|
20
|
+
allowedTools?: string[];
|
|
21
|
+
/** Resolved model id (e.g. `claude-sonnet-4-6`). */
|
|
22
|
+
model?: string;
|
|
23
|
+
/** Short runtime self-identifier (`claude`, `openai-desktop`, …)
|
|
24
|
+
* read off `ModelExecutionContract.kind`. The dashboard maps
|
|
25
|
+
* this to a small runtime icon on the agent card header. */
|
|
26
|
+
runtimeKind?: string;
|
|
18
27
|
}
|
|
19
28
|
| {
|
|
20
29
|
event: "agent.message";
|
|
@@ -2,6 +2,8 @@
|
|
|
2
2
|
|
|
3
3
|
import type { InvokeAndWaitOptions, RunStatus } from "../client.js";
|
|
4
4
|
import type { RequestContext } from "../request-context/request-context.js";
|
|
5
|
+
import type { PauseRequest } from "../pause/pause-core.js";
|
|
6
|
+
import type { RequestDecisionRequest, WaitForEventRequest } from "../pause/wrappers.js";
|
|
5
7
|
import type { SandboxProvider } from "./sandbox.js";
|
|
6
8
|
|
|
7
9
|
/** The identity of this workflow run. */
|
|
@@ -27,4 +29,27 @@ export interface BaseExecutionContext {
|
|
|
27
29
|
setMetadata?: (data: Record<string, unknown>) => Promise<void>;
|
|
28
30
|
/** Invoke another registered workflow and wait for it to settle. */
|
|
29
31
|
invokeChild: InvokeChild;
|
|
32
|
+
/**
|
|
33
|
+
* Disk-backed memoise across pause-resume. First call runs `fn` and
|
|
34
|
+
* atomically writes the result to the sandbox; on resume the recorded
|
|
35
|
+
* value is returned and `fn` is NOT re-executed. Use for expensive
|
|
36
|
+
* deterministic transforms; for side effects, use `invokeChild`.
|
|
37
|
+
* See ADR-0006 §"`ctx.checkpoint(name, fn)` — disk-backed memoisation".
|
|
38
|
+
*/
|
|
39
|
+
checkpoint<T>(name: string, fn: () => Promise<T> | T): Promise<T>;
|
|
40
|
+
/**
|
|
41
|
+
* Pause for feedback. The step exits and the workflow waits durably until
|
|
42
|
+
* something resolves the pause (a resume call, a TTL expiry); on resume the
|
|
43
|
+
* step body re-runs from the top and this call returns the resume payload.
|
|
44
|
+
* Provide a `schema` to validate the payload, `ttlMs` + `onExpiry` to bound
|
|
45
|
+
* the wait, `correlationKey` for by-key resume. Throws PauseRequestError /
|
|
46
|
+
* PauseExpiredError / PauseSchemaError. See ADR-0006 / ADR-0011.
|
|
47
|
+
*/
|
|
48
|
+
pause<T = unknown>(req: PauseRequest<T>): Promise<T>;
|
|
49
|
+
/** Pause for a typed decision (a `schema` is required). Wrapper over `pause`. */
|
|
50
|
+
requestDecision<T>(req: RequestDecisionRequest<T>): Promise<T>;
|
|
51
|
+
/** Lightweight timed pause — resolves after `durationMs`, no snapshot. */
|
|
52
|
+
sleep(durationMs: number): Promise<void>;
|
|
53
|
+
/** Pause until an event resumes by `correlationKey`. Wrapper over `pause`. */
|
|
54
|
+
waitForEvent<T = unknown>(req: WaitForEventRequest<T>): Promise<T>;
|
|
30
55
|
}
|
package/src/types/protocol.ts
CHANGED
|
@@ -71,4 +71,12 @@ export interface AgentStatus {
|
|
|
71
71
|
completed: string[];
|
|
72
72
|
blockers: string[];
|
|
73
73
|
exit_signal: boolean;
|
|
74
|
+
/** PR 7 self-pause (honoured only for `mode: "hitl"` agents). The agent
|
|
75
|
+
* cannot proceed without a human decision: it sets this true, puts the
|
|
76
|
+
* question in `question`, and ends its turn. The loop pauses at the next
|
|
77
|
+
* boundary and injects the human's answer as the next turn. `auto` agents
|
|
78
|
+
* ignore it and keep going. Should be paired with `exit_signal: false`. */
|
|
79
|
+
needs_input?: boolean;
|
|
80
|
+
/** The question to put to the human when `needs_input` is true. */
|
|
81
|
+
question?: string;
|
|
74
82
|
}
|
package/src/types/runtime.ts
CHANGED
|
@@ -48,14 +48,66 @@ export type ToolCallGateResult =
|
|
|
48
48
|
export interface ModelExecutionContract {
|
|
49
49
|
/** True when this runtime can run `processToolCall` before tool execution. */
|
|
50
50
|
supportsToolCallProcessor?: boolean;
|
|
51
|
+
/** Short self-identifier ("claude", "openai-desktop", "vercel", …). Read
|
|
52
|
+
* by the agent loop and surfaced on `agent.spawned` so the dashboard
|
|
53
|
+
* can show a per-agent runtime icon without re-fetching template
|
|
54
|
+
* metadata. Optional — runtimes that omit it stay anonymous. */
|
|
55
|
+
kind?: string;
|
|
56
|
+
/** Resolved model id used by this runtime instance (already merged with
|
|
57
|
+
* config + runtime defaults). Surfaced on `agent.spawned` so each
|
|
58
|
+
* agent card on the Agent tab can label which model it ran against. */
|
|
59
|
+
model?: string;
|
|
51
60
|
/** Runtime-owned pre-tool gate. Adapters call the shared processor chain
|
|
52
61
|
* through this seam; the agent loop stays SDK-agnostic. */
|
|
53
62
|
gateToolCall?(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult>;
|
|
63
|
+
/**
|
|
64
|
+
* Capture runtime-private in-memory state that will not survive the
|
|
65
|
+
* runner subprocess exit. Called by the agent loop at pause time,
|
|
66
|
+
* AFTER the loop has flushed its own state to disk.
|
|
67
|
+
*
|
|
68
|
+
* Return value is opaque to the loop — whatever the runtime needs to
|
|
69
|
+
* round-trip its conversation across pause-resume. Must be JSON-
|
|
70
|
+
* serialisable; the loop atomically writes it to
|
|
71
|
+
* `/tmp/wf/state/runtime-<agentInstanceId>.json` and reads it back
|
|
72
|
+
* on resume to hand to `restoreCheckpoint`.
|
|
73
|
+
*
|
|
74
|
+
* Default (method omitted): runtime holds no instance state that
|
|
75
|
+
* needs to round-trip across pause. The shipped example is the
|
|
76
|
+
* Claude runtime — the conversation lives server-side at
|
|
77
|
+
* Anthropic, addressed by `session_id`, and the loop already holds
|
|
78
|
+
* `lastSessionId` as part of its own state. On resume the loop
|
|
79
|
+
* restores the id, the next `sendMessage` passes it through, and
|
|
80
|
+
* Anthropic resumes the server-side conversation.
|
|
81
|
+
*
|
|
82
|
+
* Runtimes that hold the conversation in-process — Vercel's
|
|
83
|
+
* `VercelRunner.messages` is the canonical case — MUST implement
|
|
84
|
+
* both hooks: the messages array is reconstructed from
|
|
85
|
+
* `response.messages` on each `streamText` and would be lost the
|
|
86
|
+
* moment the subprocess exits. See ADR-0006 §"Concrete examples
|
|
87
|
+
* for shipped runtimes" for the audit + worked examples.
|
|
88
|
+
*/
|
|
89
|
+
captureCheckpoint?(): unknown;
|
|
90
|
+
/**
|
|
91
|
+
* Restore runtime-private state previously returned by
|
|
92
|
+
* `captureCheckpoint`. Called by the agent loop on resume, AFTER
|
|
93
|
+
* the loop has restored its own state but BEFORE iterations resume.
|
|
94
|
+
*
|
|
95
|
+
* `blob` is whatever this same runtime returned at pause time. If
|
|
96
|
+
* `captureCheckpoint` is omitted, this is never called.
|
|
97
|
+
*/
|
|
98
|
+
restoreCheckpoint?(blob: unknown): void;
|
|
54
99
|
sendMessage(opts: {
|
|
55
100
|
prompt: string;
|
|
56
101
|
sessionId?: string;
|
|
57
102
|
iteration?: number;
|
|
58
103
|
signal?: AbortSignal;
|
|
104
|
+
/** Push-iterable of mid-turn user messages from outside the agent
|
|
105
|
+
* loop — e.g. dashboard chat injections. Runtimes that support
|
|
106
|
+
* streaming-input mode (Claude Agent SDK) read from this in
|
|
107
|
+
* parallel with the initial `prompt`; the SDK handles delivery
|
|
108
|
+
* at the next safe boundary. Runtimes without streaming-input
|
|
109
|
+
* support ignore this and fall back to per-iteration injection. */
|
|
110
|
+
inboxStream?: AsyncIterable<{ text: string; senderName?: string | null }>;
|
|
59
111
|
}): AsyncGenerator<AgentMessage>;
|
|
60
112
|
}
|
|
61
113
|
|
package/src/types/sandbox.ts
CHANGED
|
@@ -10,12 +10,17 @@ export interface SandboxCommandRunOptions {
|
|
|
10
10
|
envs?: Record<string, string>;
|
|
11
11
|
onStdout?: (data: string) => void;
|
|
12
12
|
onStderr?: (data: string) => void;
|
|
13
|
-
|
|
13
|
+
/** Run the command with root privileges. Vercel maps this to its native
|
|
14
|
+
* `sudo` flag; the local provider prepends `sudo`. Defaults to false.
|
|
15
|
+
* Requires the sandbox image to grant the command root (Vercel's runtimes
|
|
16
|
+
* do — passwordless). */
|
|
17
|
+
sudo?: boolean;
|
|
14
18
|
}
|
|
15
19
|
|
|
16
20
|
export interface SandboxCommandResult {
|
|
17
21
|
exitCode: number;
|
|
18
22
|
stdout: string;
|
|
23
|
+
stderr: string;
|
|
19
24
|
}
|
|
20
25
|
|
|
21
26
|
export interface SandboxProvider {
|
|
@@ -23,6 +28,10 @@ export interface SandboxProvider {
|
|
|
23
28
|
/** Working directory for the agent process. Set by onStart after environment setup. */
|
|
24
29
|
cwd?: string;
|
|
25
30
|
commands: {
|
|
31
|
+
// NOTE: does NOT throw on non-zero exit. Callers must check
|
|
32
|
+
// `result.exitCode` themselves. The `agent-env` setup workflow's
|
|
33
|
+
// `run(sb, cmd)` helper is the canonical pattern — copy it into any
|
|
34
|
+
// setup workflow that needs to fail loudly on command errors.
|
|
26
35
|
run(cmd: string, opts?: SandboxCommandRunOptions): Promise<SandboxCommandResult>;
|
|
27
36
|
};
|
|
28
37
|
files: {
|
package/src/types/workflow.ts
CHANGED
|
@@ -210,7 +210,12 @@ function compileRunForm<TOutput, TInput extends Record<string, unknown>>(
|
|
|
210
210
|
setMetadata: stepCtx.setMetadata,
|
|
211
211
|
step: stepCtx.step,
|
|
212
212
|
agentEvents: stepCtx.agentEvents,
|
|
213
|
-
|
|
213
|
+
checkpoint: stepCtx.checkpoint,
|
|
214
|
+
pause: stepCtx.pause,
|
|
215
|
+
requestDecision: stepCtx.requestDecision,
|
|
216
|
+
sleep: stepCtx.sleep,
|
|
217
|
+
waitForEvent: stepCtx.waitForEvent,
|
|
218
|
+
processors: metadata.processors ?? [],
|
|
214
219
|
};
|
|
215
220
|
const sandbox = stepCtx.sandbox;
|
|
216
221
|
if (!sandbox) throw new Error("legacy run-form workflow requires a sandbox in StepContext");
|
package/src/utils/bundler.ts
CHANGED
|
@@ -111,8 +111,9 @@ async function bundle(path: string, label: string): Promise<string> {
|
|
|
111
111
|
}
|
|
112
112
|
|
|
113
113
|
/** Dynamically import a bundled source in-process and extract fields from its
|
|
114
|
-
* default export. Safe for CLI/test contexts.
|
|
115
|
-
*
|
|
114
|
+
* default export. Safe for CLI/test contexts. Evaluation failures are
|
|
115
|
+
* surfaced with their original cause so users fix the real import/runtime
|
|
116
|
+
* problem instead of being told the default export shape is wrong. */
|
|
116
117
|
async function extractFromBundle<T>(
|
|
117
118
|
source: string,
|
|
118
119
|
label: string,
|
|
@@ -123,7 +124,11 @@ async function extractFromBundle<T>(
|
|
|
123
124
|
const loaded = await importSourceModule<Record<string, unknown>>(source, tmpPath);
|
|
124
125
|
try { return pick(loaded.mod.default); }
|
|
125
126
|
finally { await loaded.cleanup(); }
|
|
126
|
-
} catch {
|
|
127
|
+
} catch (err) {
|
|
128
|
+
throw new WorkflowSourceValidationError(
|
|
129
|
+
`Bundled ${label} could not be evaluated: ${err instanceof Error ? err.message : String(err)}`,
|
|
130
|
+
);
|
|
131
|
+
}
|
|
127
132
|
}
|
|
128
133
|
|
|
129
134
|
/**
|
package/src/utils/schemas.ts
CHANGED
|
@@ -28,3 +28,6 @@ export { WORKFLOW_BRAND } from "./types.js";
|
|
|
28
28
|
|
|
29
29
|
export { StepObservabilityCollector } from "./observability.js";
|
|
30
30
|
export type { StepObservability, SubStepEvent } from "./observability.js";
|
|
31
|
+
|
|
32
|
+
export { makeRunCallbackEmitterFromEnv } from "./run-callback.js";
|
|
33
|
+
export type { LiveAgentEventEmitter } from "./run-callback.js";
|
|
@@ -5,19 +5,34 @@
|
|
|
5
5
|
* tokenised stdout sentinel; the activity persists the snapshot to the
|
|
6
6
|
* run's metadata + lifecycle event tables.
|
|
7
7
|
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
8
|
+
* Live streaming + batch backstop — the no-double-write design:
|
|
9
|
+
*
|
|
10
|
+
* 1. `agentEvents.emit` immediately fires the optional `liveEmitter`
|
|
11
|
+
* (the runner's POST to `/internal/runs/.../events`).
|
|
12
|
+
* 2. The emitter returns `Promise<boolean>` — true means the server
|
|
13
|
+
* accepted and persisted this event, false means it failed (POST
|
|
14
|
+
* error, server 5xx, network blip).
|
|
15
|
+
* 3. The collector tracks which seqs were successfully ack'd.
|
|
16
|
+
* 4. At `snapshot()` time we await any in-flight emits (with a small
|
|
17
|
+
* grace window so the agent loop's final-burst posts can finish),
|
|
18
|
+
* then STRIP ack'd events from the returned `events` array.
|
|
19
|
+
*
|
|
20
|
+
* The result: the snapshot's `events` array only contains events
|
|
21
|
+
* that the live path didn't successfully deliver. The server's
|
|
22
|
+
* batch-flush in `persistStepObservability` becomes a true backstop
|
|
23
|
+
* for the FAILURE path — it never re-writes (and never re-notifies)
|
|
24
|
+
* the events the live route already handled. No double pg_notify,
|
|
25
|
+
* no dashboard duplicates.
|
|
26
|
+
*
|
|
27
|
+
* `metadata` and `subSteps` remain batch-only because they're
|
|
28
|
+
* naturally boundary events (no streaming benefit) and aren't
|
|
29
|
+
* written by the live route at all.
|
|
17
30
|
*/
|
|
18
31
|
|
|
19
32
|
import type { AgentLifecycleEvent } from "../agent/agent-loop.js";
|
|
20
33
|
import type { AgentEventSink } from "../types/workflow.js";
|
|
34
|
+
import type { LiveAgentEventEmitter } from "./run-callback.js";
|
|
35
|
+
import { PauseSignal, isPauseSignal } from "../pause/pause-core.js";
|
|
21
36
|
|
|
22
37
|
/** One named sub-step (from `ctx.step("name", async () => ...)`).
|
|
23
38
|
* Becomes a `workflow_substep_*` lifecycle event on the run timeline. */
|
|
@@ -39,6 +54,13 @@ export interface StepObservability {
|
|
|
39
54
|
subSteps?: SubStepEvent[];
|
|
40
55
|
}
|
|
41
56
|
|
|
57
|
+
/** Maximum time we'll wait for in-flight live-emit POSTs to settle
|
|
58
|
+
* before taking the snapshot. Tuned to be longer than a healthy POST
|
|
59
|
+
* (~50ms) but short enough that a totally-broken live path doesn't
|
|
60
|
+
* block step return — the batch backstop will handle whatever doesn't
|
|
61
|
+
* resolve in time. */
|
|
62
|
+
const LIVE_EMIT_DRAIN_TIMEOUT_MS = 1_500;
|
|
63
|
+
|
|
42
64
|
/**
|
|
43
65
|
* Append-only collector bound to a single step's `StepContext`. The
|
|
44
66
|
* step's `setMetadata` / `step` / `agentEvents` properties all point at
|
|
@@ -53,6 +75,15 @@ export class StepObservabilityCollector {
|
|
|
53
75
|
private metadata: Record<string, unknown> = {};
|
|
54
76
|
private events: AgentLifecycleEvent[] = [];
|
|
55
77
|
private subSteps: SubStepEvent[] = [];
|
|
78
|
+
private readonly liveEmitter?: LiveAgentEventEmitter;
|
|
79
|
+
/** Seqs of events the server confirmed via the live route. */
|
|
80
|
+
private readonly ackedSeqs: Set<number> = new Set();
|
|
81
|
+
/** Promises for in-flight live emits — awaited at snapshot time. */
|
|
82
|
+
private readonly inFlight: Set<Promise<void>> = new Set();
|
|
83
|
+
|
|
84
|
+
constructor(opts: { liveEmitter?: LiveAgentEventEmitter } = {}) {
|
|
85
|
+
this.liveEmitter = opts.liveEmitter;
|
|
86
|
+
}
|
|
56
87
|
|
|
57
88
|
readonly setMetadata = async (data: Record<string, unknown>): Promise<void> => {
|
|
58
89
|
Object.assign(this.metadata, data);
|
|
@@ -70,6 +101,9 @@ export class StepObservabilityCollector {
|
|
|
70
101
|
});
|
|
71
102
|
return result;
|
|
72
103
|
} catch (err) {
|
|
104
|
+
// A pause inside ctx.step is control flow, not a failed sub-step — let
|
|
105
|
+
// it propagate untouched so serveStep emits the pause sentinel.
|
|
106
|
+
if (isPauseSignal(err)) throw err;
|
|
73
107
|
this.subSteps.push({
|
|
74
108
|
name,
|
|
75
109
|
startedAt,
|
|
@@ -83,20 +117,57 @@ export class StepObservabilityCollector {
|
|
|
83
117
|
|
|
84
118
|
readonly agentEvents: AgentEventSink = {
|
|
85
119
|
emit: (event: AgentLifecycleEvent) => {
|
|
120
|
+
// `seq` is the event's index in `events`, captured BEFORE push so
|
|
121
|
+
// it matches the index the server-side batch flush uses (its
|
|
122
|
+
// `.entries()` loop). Same index → same idempotency key on the
|
|
123
|
+
// server. The live route's `acceptedSeqs` response uses this seq;
|
|
124
|
+
// we strip those from the snapshot in `snapshot()`.
|
|
125
|
+
const seq = this.events.length;
|
|
86
126
|
this.events.push(event);
|
|
127
|
+
if (this.liveEmitter) {
|
|
128
|
+
const tracked = this.liveEmitter(event, seq)
|
|
129
|
+
.then((ok) => { if (ok) this.ackedSeqs.add(seq); })
|
|
130
|
+
.catch(() => { /* leave unacked → batch backstop delivers */ });
|
|
131
|
+
this.inFlight.add(tracked);
|
|
132
|
+
// Self-clean so completed promises don't leak across long-
|
|
133
|
+
// running steps.
|
|
134
|
+
void tracked.finally(() => this.inFlight.delete(tracked));
|
|
135
|
+
}
|
|
87
136
|
},
|
|
88
137
|
};
|
|
89
138
|
|
|
139
|
+
/** Wait for in-flight live emits to settle (or timeout) so the
|
|
140
|
+
* ackedSeqs set is maximally up-to-date before we filter. Used by
|
|
141
|
+
* `snapshot()` — exposed separately for tests. */
|
|
142
|
+
private async drainInFlight(timeoutMs = LIVE_EMIT_DRAIN_TIMEOUT_MS): Promise<void> {
|
|
143
|
+
if (this.inFlight.size === 0) return;
|
|
144
|
+
await Promise.race([
|
|
145
|
+
Promise.allSettled([...this.inFlight]),
|
|
146
|
+
new Promise<void>((resolve) => setTimeout(resolve, timeoutMs)),
|
|
147
|
+
]);
|
|
148
|
+
}
|
|
149
|
+
|
|
90
150
|
/** Snapshot the accumulated state. Returns `undefined` when nothing
|
|
91
|
-
* was recorded so the wire payload can drop the field entirely.
|
|
92
|
-
|
|
151
|
+
* was recorded so the wire payload can drop the field entirely.
|
|
152
|
+
*
|
|
153
|
+
* Async because we drain in-flight live emits first. Any event the
|
|
154
|
+
* server acknowledged is REMOVED from the returned `events` array
|
|
155
|
+
* so the server-side batch flush doesn't re-write/re-notify it.
|
|
156
|
+
* Events that failed live delivery (POST error, timeout) stay in
|
|
157
|
+
* the array as the durable backstop. */
|
|
158
|
+
async snapshot(): Promise<StepObservability | undefined> {
|
|
159
|
+
await this.drainInFlight();
|
|
160
|
+
|
|
93
161
|
const hasMetadata = Object.keys(this.metadata).length > 0;
|
|
94
|
-
const hasEvents = this.events.length > 0;
|
|
95
162
|
const hasSubSteps = this.subSteps.length > 0;
|
|
163
|
+
// Filter out ack'd events — the live route already wrote them.
|
|
164
|
+
const remainingEvents = this.events.filter((_, idx) => !this.ackedSeqs.has(idx));
|
|
165
|
+
const hasEvents = remainingEvents.length > 0;
|
|
166
|
+
|
|
96
167
|
if (!hasMetadata && !hasEvents && !hasSubSteps) return undefined;
|
|
97
168
|
const result: StepObservability = {};
|
|
98
169
|
if (hasMetadata) result.metadata = { ...this.metadata };
|
|
99
|
-
if (hasEvents) result.events =
|
|
170
|
+
if (hasEvents) result.events = remainingEvents;
|
|
100
171
|
if (hasSubSteps) result.subSteps = [...this.subSteps];
|
|
101
172
|
return result;
|
|
102
173
|
}
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Runner → server live event emitter.
|
|
3
|
+
*
|
|
4
|
+
* Pairs with the server-side `/api/v1/internal/runs/:runId/steps/:stepIndex/events`
|
|
5
|
+
* route. The runner reads `AGENT_COMPOSE_URL`, `AGENT_COMPOSE_RUN_TOKEN`,
|
|
6
|
+
* and `RUN_ID` from env; if all three are present it builds a POST
|
|
7
|
+
* function that sends each agent lifecycle event to the server as
|
|
8
|
+
* soon as the agent loop emits it.
|
|
9
|
+
*
|
|
10
|
+
* Non-blocking: the agent loop doesn't await the emitter — the runtime
|
|
11
|
+
* keeps emitting events at full speed. But the emitter still returns a
|
|
12
|
+
* `Promise<boolean>` so the collector can later decide whether to
|
|
13
|
+
* include each event in the batch-end durable backstop. `true` = server
|
|
14
|
+
* accepted the event (don't re-deliver in batch); `false` = POST
|
|
15
|
+
* failed (DO re-deliver). The collector awaits these promises with a
|
|
16
|
+
* short grace window at snapshot time.
|
|
17
|
+
*
|
|
18
|
+
* This is the seam that lets us avoid the double-write problem: every
|
|
19
|
+
* successfully-ack'd live event is stripped from `observability.events`
|
|
20
|
+
* before the batch flush runs, so the durable backstop only carries
|
|
21
|
+
* events that genuinely failed live delivery. No more two paths writing
|
|
22
|
+
* the same row + double pg_notify.
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
import type { AgentLifecycleEvent } from "../agent/agent-loop.js";
|
|
26
|
+
|
|
27
|
+
export type LiveAgentEventEmitter =
|
|
28
|
+
(event: AgentLifecycleEvent, seq: number) => Promise<boolean>;
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Build a live emitter from process env + the caller-supplied stepIndex.
|
|
32
|
+
* Returns `undefined` when any required env var is missing (local tests,
|
|
33
|
+
* non-sandbox callers) so the caller can wire `undefined` straight
|
|
34
|
+
* through to the collector and get batch-only delivery without
|
|
35
|
+
* conditional plumbing.
|
|
36
|
+
*
|
|
37
|
+
* `stepIndex` is a parameter rather than an env read because
|
|
38
|
+
* `serveStep` scrubs `AC_STEP_INDEX` from `process.env` before invoking
|
|
39
|
+
* the workflow handler (to keep the transport envelope unreachable
|
|
40
|
+
* from user code); the handler still holds the parsed integer and
|
|
41
|
+
* passes it here.
|
|
42
|
+
*/
|
|
43
|
+
export function makeRunCallbackEmitterFromEnv(stepIndex: number): LiveAgentEventEmitter | undefined {
|
|
44
|
+
const baseUrl = process.env.AGENT_COMPOSE_URL;
|
|
45
|
+
const token = process.env.AGENT_COMPOSE_RUN_TOKEN;
|
|
46
|
+
const runId = process.env.RUN_ID;
|
|
47
|
+
if (!baseUrl || !token || !runId) return undefined;
|
|
48
|
+
if (!Number.isFinite(stepIndex) || stepIndex < 0) return undefined;
|
|
49
|
+
const url = `${baseUrl.replace(/\/+$/, "")}/api/v1/internal/runs/${runId}/steps/${stepIndex}/events`;
|
|
50
|
+
const headers = {
|
|
51
|
+
"content-type": "application/json",
|
|
52
|
+
"authorization": `Bearer ${token}`,
|
|
53
|
+
} as const;
|
|
54
|
+
return async (event, seq) => {
|
|
55
|
+
try {
|
|
56
|
+
const res = await fetch(url, {
|
|
57
|
+
method: "POST",
|
|
58
|
+
headers,
|
|
59
|
+
body: JSON.stringify({ events: [{ ...event, seq }] }),
|
|
60
|
+
});
|
|
61
|
+
if (!res.ok) return false;
|
|
62
|
+
// Parse the route's response to confirm THIS seq was accepted.
|
|
63
|
+
// The route returns `{ acceptedSeqs: number[] }`; absence means
|
|
64
|
+
// the server logged but didn't persist (rare — DB hiccup), and
|
|
65
|
+
// we should leave the event in the batch backstop.
|
|
66
|
+
const body = await res.json().catch(() => null) as { acceptedSeqs?: number[] } | null;
|
|
67
|
+
return Array.isArray(body?.acceptedSeqs) && body.acceptedSeqs.includes(seq);
|
|
68
|
+
} catch {
|
|
69
|
+
return false;
|
|
70
|
+
}
|
|
71
|
+
};
|
|
72
|
+
}
|
|
@@ -28,6 +28,12 @@ import type { RequestContext } from "../request-context/request-context.js";
|
|
|
28
28
|
import type { SandboxProvider } from "../types/sandbox.js";
|
|
29
29
|
import type { WorkflowRun, WorkflowCtx } from "../types/workflow.js";
|
|
30
30
|
import { StepObservabilityCollector, type StepObservability } from "./observability.js";
|
|
31
|
+
import type { LiveAgentEventEmitter } from "./run-callback.js";
|
|
32
|
+
import { scopedCheckpoint } from "../pause/checkpoint.js";
|
|
33
|
+
import { corePause, PauseSignal, isPauseSignal, type PauseRequest } from "../pause/pause-core.js";
|
|
34
|
+
import type { StepPauseRequest } from "../step-invocation/types.js";
|
|
35
|
+
import { buildPauseWrappers, type PauseFn, type KindedPauseFn } from "../pause/wrappers.js";
|
|
36
|
+
import { runWithActiveStep, setActiveStepBridge, restoreActiveStepBridge } from "../active-step.js";
|
|
31
37
|
|
|
32
38
|
export class StepValidationError extends Error {
|
|
33
39
|
readonly kind = "step-validation" as const;
|
|
@@ -84,6 +90,11 @@ export interface RunWorkflowStepsOpts<TInput, TOutput> {
|
|
|
84
90
|
onStepStarted?(stepIndex: number, stepName: string): void | Promise<void>;
|
|
85
91
|
/** Child workflow invocation implementation. Defaults to a clear unsupported error. */
|
|
86
92
|
invokeChild?: WorkflowCtx["invokeChild"];
|
|
93
|
+
/** Optional live-stream emitter for agent lifecycle events. The runner
|
|
94
|
+
* passes a fetch-based emitter wired to the per-run callback token so
|
|
95
|
+
* the dashboard sees events as the agent loop produces them; tests
|
|
96
|
+
* leave it undefined and get batch-only delivery. */
|
|
97
|
+
liveAgentEventEmitter?: LiveAgentEventEmitter;
|
|
87
98
|
}
|
|
88
99
|
|
|
89
100
|
export interface RunWorkflowStepsResult<TOutput> {
|
|
@@ -101,6 +112,8 @@ export interface RunWorkflowSingleStepOpts {
|
|
|
101
112
|
sandbox?: SandboxProvider;
|
|
102
113
|
abortSignal?: AbortSignal;
|
|
103
114
|
invokeChild?: WorkflowCtx["invokeChild"];
|
|
115
|
+
/** Optional live-stream emitter — see `RunWorkflowStepsOpts.liveAgentEventEmitter`. */
|
|
116
|
+
liveAgentEventEmitter?: LiveAgentEventEmitter;
|
|
104
117
|
}
|
|
105
118
|
|
|
106
119
|
/** Result of one step run — output plus whatever the step's observability
|
|
@@ -116,7 +129,14 @@ export async function runWorkflowSingleStep(opts: RunWorkflowSingleStepOpts): Pr
|
|
|
116
129
|
if (!step) throw new Error(`Step index ${opts.stepIndex} not found in workflow "${opts.workflow.id}"`);
|
|
117
130
|
const parsedInput = step.input.safeParse(opts.input);
|
|
118
131
|
if (!parsedInput.success) throw new StepValidationError(step.name, "input", parsedInput.error);
|
|
119
|
-
const collector = new StepObservabilityCollector();
|
|
132
|
+
const collector = new StepObservabilityCollector({ liveEmitter: opts.liveAgentEventEmitter });
|
|
133
|
+
const coord = { runId: opts.run.id, stepIndex: opts.stepIndex };
|
|
134
|
+
const pauseWithKind: KindedPauseFn = function <T>(req: PauseRequest<T>, kind: StepPauseRequest["kind"]): Promise<T> {
|
|
135
|
+
return corePause(req, coord, kind);
|
|
136
|
+
};
|
|
137
|
+
const pause: PauseFn = function <T>(req: PauseRequest<T>): Promise<T> {
|
|
138
|
+
return corePause(req, coord);
|
|
139
|
+
};
|
|
120
140
|
const stepCtx: StepContext<unknown> = {
|
|
121
141
|
input: parsedInput.data,
|
|
122
142
|
requestContext: opts.requestContext,
|
|
@@ -128,11 +148,35 @@ export async function runWorkflowSingleStep(opts: RunWorkflowSingleStepOpts): Pr
|
|
|
128
148
|
setMetadata: collector.setMetadata,
|
|
129
149
|
step: collector.step,
|
|
130
150
|
agentEvents: collector.agentEvents,
|
|
151
|
+
checkpoint: scopedCheckpoint(`step${opts.stepIndex}`),
|
|
152
|
+
pause,
|
|
153
|
+
...buildPauseWrappers(pauseWithKind),
|
|
131
154
|
};
|
|
132
|
-
|
|
155
|
+
let output: unknown;
|
|
156
|
+
try {
|
|
157
|
+
// Set the active step so `agent()` / `ctx.pause` derive deterministic
|
|
158
|
+
// ids from the un-scrubbed index (see active-step.ts). Async-local scope
|
|
159
|
+
// keeps concurrent in-process runs from clobbering each other. ALSO publish
|
|
160
|
+
// on the cross-instance bridge: a bundled workflow inlines its own SDK copy
|
|
161
|
+
// (a separate AsyncLocalStorage), so its `agent()` can't see the ALS we set
|
|
162
|
+
// here — the bridge carries the step across that boundary. One step per
|
|
163
|
+
// subprocess in the sandbox ⇒ no concurrency on the global slot.
|
|
164
|
+
const bridged = setActiveStepBridge({ stepIndex: opts.stepIndex });
|
|
165
|
+
try {
|
|
166
|
+
output = await runWithActiveStep({ stepIndex: opts.stepIndex }, () => Promise.resolve(step.run(stepCtx)));
|
|
167
|
+
} finally {
|
|
168
|
+
restoreActiveStepBridge(bridged);
|
|
169
|
+
}
|
|
170
|
+
} catch (err) {
|
|
171
|
+
if (isPauseSignal(err)) {
|
|
172
|
+
const observability = await collector.snapshot();
|
|
173
|
+
throw new PauseSignal(err.pauseId, err.pauseRequest, observability);
|
|
174
|
+
}
|
|
175
|
+
throw err;
|
|
176
|
+
}
|
|
133
177
|
const parsedOutput = step.output.safeParse(output);
|
|
134
178
|
if (!parsedOutput.success) throw new StepValidationError(step.name, "output", parsedOutput.error);
|
|
135
|
-
const observability = collector.snapshot();
|
|
179
|
+
const observability = await collector.snapshot();
|
|
136
180
|
return observability === undefined
|
|
137
181
|
? { output: parsedOutput.data }
|
|
138
182
|
: { output: parsedOutput.data, observability };
|
|
@@ -179,7 +223,14 @@ export async function runWorkflowSteps<TInput, TOutput>(
|
|
|
179
223
|
throw err;
|
|
180
224
|
}
|
|
181
225
|
|
|
182
|
-
const collector = new StepObservabilityCollector();
|
|
226
|
+
const collector = new StepObservabilityCollector({ liveEmitter: opts.liveAgentEventEmitter });
|
|
227
|
+
const coord = { runId: run.id, stepIndex: i };
|
|
228
|
+
const pauseWithKind: KindedPauseFn = function <T>(req: PauseRequest<T>, kind: StepPauseRequest["kind"]): Promise<T> {
|
|
229
|
+
return corePause(req, coord, kind);
|
|
230
|
+
};
|
|
231
|
+
const pause: PauseFn = function <T>(req: PauseRequest<T>): Promise<T> {
|
|
232
|
+
return corePause(req, coord);
|
|
233
|
+
};
|
|
183
234
|
const stepCtx: StepContext<unknown> = {
|
|
184
235
|
input: parsedStepInput.data,
|
|
185
236
|
requestContext,
|
|
@@ -191,16 +242,27 @@ export async function runWorkflowSteps<TInput, TOutput>(
|
|
|
191
242
|
setMetadata: collector.setMetadata,
|
|
192
243
|
step: collector.step,
|
|
193
244
|
agentEvents: collector.agentEvents,
|
|
245
|
+
checkpoint: scopedCheckpoint(`step${i}`),
|
|
246
|
+
pause,
|
|
247
|
+
...buildPauseWrappers(pauseWithKind),
|
|
194
248
|
};
|
|
195
249
|
|
|
196
250
|
const startedAt = Date.now();
|
|
197
251
|
let output: unknown;
|
|
198
252
|
try {
|
|
199
|
-
|
|
253
|
+
// Active step for deterministic agent/pause id derivation (active-step.ts).
|
|
254
|
+
// Async-local scope keeps concurrent in-process runs isolated.
|
|
255
|
+
output = await runWithActiveStep({ stepIndex: i }, () => Promise.resolve(step.run(stepCtx)));
|
|
200
256
|
} catch (err) {
|
|
257
|
+
// A pause is control flow, not a failure — let it propagate so serveStep
|
|
258
|
+
// emits the pause sentinel instead of recording a failed step.
|
|
259
|
+
if (isPauseSignal(err)) {
|
|
260
|
+
const observability = await collector.snapshot();
|
|
261
|
+
throw new PauseSignal(err.pauseId, err.pauseRequest, observability);
|
|
262
|
+
}
|
|
201
263
|
const wrapped = err instanceof Error ? err : new Error(String(err));
|
|
202
264
|
const durationMs = Date.now() - startedAt;
|
|
203
|
-
const observability = collector.snapshot();
|
|
265
|
+
const observability = await collector.snapshot();
|
|
204
266
|
stepResults.push({
|
|
205
267
|
name: step.name, status: "failed", error: wrapped.message, durationMs,
|
|
206
268
|
...(observability ? { observability } : {}),
|
|
@@ -213,7 +275,7 @@ export async function runWorkflowSteps<TInput, TOutput>(
|
|
|
213
275
|
if (!parsedOutput.success) {
|
|
214
276
|
const err = new StepValidationError(step.name, "output", parsedOutput.error);
|
|
215
277
|
const durationMs = Date.now() - startedAt;
|
|
216
|
-
const observability = collector.snapshot();
|
|
278
|
+
const observability = await collector.snapshot();
|
|
217
279
|
stepResults.push({
|
|
218
280
|
name: step.name, status: "failed", error: err.message, durationMs,
|
|
219
281
|
...(observability ? { observability } : {}),
|
|
@@ -224,7 +286,7 @@ export async function runWorkflowSteps<TInput, TOutput>(
|
|
|
224
286
|
|
|
225
287
|
const durationMs = Date.now() - startedAt;
|
|
226
288
|
current = parsedOutput.data;
|
|
227
|
-
const observability = collector.snapshot();
|
|
289
|
+
const observability = await collector.snapshot();
|
|
228
290
|
stepResults.push({
|
|
229
291
|
name: step.name, status: "completed", output: parsedOutput.data, durationMs,
|
|
230
292
|
...(observability ? { observability } : {}),
|
package/src/utils/discovery.ts
DELETED