@agent-compose/sdk 0.6.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +66 -39
- package/dist/agent/__tests__/runtime-json-schema.test.d.ts +10 -0
- package/dist/agent/agent-context.d.ts +21 -1
- package/dist/agent/agent-loop.d.ts +24 -1
- package/dist/client.d.ts +338 -534
- package/dist/directives.d.ts +112 -0
- package/dist/display.d.ts +242 -0
- package/dist/errors.d.ts +24 -1
- package/dist/index.d.ts +34 -13
- package/dist/index.js +2984 -861
- package/dist/pause/wrappers.d.ts +31 -9
- package/dist/processors/ask-human.d.ts +30 -0
- package/dist/processors/ask-human.test.d.ts +1 -0
- package/dist/processors/index.d.ts +1 -0
- package/dist/runtimes/_acp-client.d.ts +46 -1
- package/dist/runtimes/_cli-agent.d.ts +58 -4
- package/dist/runtimes/_jsonl-guard.d.ts +103 -0
- package/dist/runtimes/amp.d.ts +2 -2
- package/dist/runtimes/claude-code.d.ts +59 -0
- package/dist/runtimes/claude-code.test.d.ts +14 -0
- package/dist/runtimes/claude.d.ts +16 -0
- package/dist/runtimes/claude.test.d.ts +8 -0
- package/dist/runtimes/codex.d.ts +9 -3
- package/dist/runtimes/cursor.d.ts +9 -0
- package/dist/runtimes/droid.d.ts +9 -0
- package/dist/runtimes/jsonl-guard.test.d.ts +19 -0
- package/dist/runtimes/openai-desktop.js +2922 -861
- package/dist/runtimes/opencode.d.ts +25 -0
- package/dist/runtimes/vercel.js +22 -1
- package/dist/sandbox/devbox.d.ts +42 -0
- package/dist/sandbox/exec-stream.d.ts +14 -0
- package/dist/sandbox/network-policy.d.ts +100 -0
- package/dist/sandbox/provider-def.d.ts +79 -0
- package/dist/sandbox/providers/desktop.d.ts +10 -0
- package/dist/sandbox/providers/e2b.d.ts +17 -0
- package/dist/sandbox/providers/local.d.ts +11 -0
- package/dist/sandbox/providers/vercel.d.ts +18 -0
- package/dist/sandbox/registry.d.ts +45 -0
- package/dist/sandbox/sizes.d.ts +68 -0
- package/dist/sandbox.d.ts +24 -299
- package/dist/step-invocation/__tests__/foreground-recovery.test.d.ts +1 -0
- package/dist/step-invocation/invoker.d.ts +24 -1
- package/dist/step-invocation/protocol.d.ts +13 -0
- package/dist/types/api-compliance.d.ts +71 -0
- package/dist/types/api-conversations.d.ts +492 -0
- package/dist/types/api-factory.d.ts +309 -0
- package/dist/types/api-projects.d.ts +131 -0
- package/dist/types/api-runs.d.ts +377 -0
- package/dist/types/api-scopes.d.ts +102 -0
- package/dist/types/conversation-stream.d.ts +191 -0
- package/dist/types/execution-context.d.ts +12 -2
- package/dist/types/protocol.d.ts +30 -1
- package/dist/types/sandbox-environment.d.ts +8 -5
- package/dist/types/sandbox.d.ts +79 -0
- package/dist/types/workflow-metadata.d.ts +33 -8
- package/dist/types/workflow-plan.d.ts +10 -0
- package/dist/types/workflow.d.ts +18 -193
- package/dist/utils/bundler.d.ts +12 -1
- package/dist/utils/errors.d.ts +9 -1
- package/dist/workflow-steps/index.d.ts +1 -1
- package/dist/workflow-steps/observability.d.ts +8 -1
- package/dist/workflow-steps/runner.d.ts +3 -3
- package/dist/workflow-steps/step.d.ts +15 -1
- package/dist/workflow-steps/types.d.ts +19 -5
- package/dist/workflow-steps/workflow.d.ts +22 -1
- package/dist/workflows/engine.d.ts +3 -2
- package/dist/workflows/invoke-child.d.ts +2 -2
- package/package.json +1 -1
- package/src/agent/agent-context.ts +206 -16
- package/src/agent/agent-loop.ts +40 -4
- package/src/agent/run-agent.ts +9 -1
- package/src/client.ts +909 -621
- package/src/directives.ts +184 -0
- package/src/display.ts +788 -0
- package/src/errors.ts +39 -0
- package/src/index.ts +117 -10
- package/src/pause/wrappers.ts +44 -9
- package/src/processors/ask-human.ts +136 -0
- package/src/processors/index.ts +5 -0
- package/src/runtimes/_acp-client.ts +72 -3
- package/src/runtimes/_cli-agent.ts +171 -38
- package/src/runtimes/_jsonl-guard.ts +219 -0
- package/src/runtimes/claude-code.ts +246 -0
- package/src/runtimes/claude.ts +32 -2
- package/src/runtimes/codex.ts +55 -3
- package/src/runtimes/cursor.ts +59 -0
- package/src/runtimes/droid.ts +63 -0
- package/src/runtimes/openai-desktop.ts +59 -14
- package/src/runtimes/opencode.ts +61 -0
- package/src/sandbox/devbox.ts +48 -0
- package/src/sandbox/exec-stream.ts +48 -0
- package/src/sandbox/network-policy.ts +181 -0
- package/src/sandbox/provider-def.ts +94 -0
- package/src/sandbox/providers/desktop.ts +57 -0
- package/src/sandbox/providers/e2b.ts +354 -0
- package/src/sandbox/providers/local.ts +106 -0
- package/src/sandbox/providers/vercel.ts +331 -0
- package/src/sandbox/registry.ts +198 -0
- package/src/sandbox/sizes.ts +95 -0
- package/src/sandbox.ts +59 -1263
- package/src/step-invocation/invoker.ts +319 -34
- package/src/step-invocation/protocol.ts +19 -0
- package/src/types/api-compliance.ts +79 -0
- package/src/types/api-conversations.ts +522 -0
- package/src/types/api-factory.ts +336 -0
- package/src/types/api-projects.ts +140 -0
- package/src/types/api-runs.ts +412 -0
- package/src/types/api-scopes.ts +102 -0
- package/src/types/conversation-stream.ts +231 -0
- package/src/types/execution-context.ts +10 -2
- package/src/types/protocol.ts +33 -0
- package/src/types/sandbox-environment.ts +28 -9
- package/src/types/sandbox.ts +78 -0
- package/src/types/workflow-metadata.ts +35 -8
- package/src/types/workflow-plan.ts +11 -0
- package/src/types/workflow.ts +25 -280
- package/src/utils/bundler.ts +32 -5
- package/src/utils/errors.ts +34 -2
- package/src/workflow-steps/index.ts +1 -0
- package/src/workflow-steps/observability.ts +19 -8
- package/src/workflow-steps/runner.ts +4 -4
- package/src/workflow-steps/step.ts +49 -1
- package/src/workflow-steps/types.ts +20 -5
- package/src/workflow-steps/workflow.ts +22 -1
- package/src/workflows/engine.ts +3 -2
- package/src/workflows/invoke-child.ts +2 -2
|
@@ -25,7 +25,7 @@ import { randomBytes } from "node:crypto";
|
|
|
25
25
|
import { z } from "zod";
|
|
26
26
|
import { SandboxUnavailableError } from "../sandbox-errors.js";
|
|
27
27
|
import type { SandboxProvider, SandboxBackgroundProcess } from "../types/sandbox.js";
|
|
28
|
-
import { RUNNER_COMMAND, STEP_ENV, stepResultLinePrefix, stepPauseLinePrefix, requestContextPath, stepInputPath, stepResultFilePath } from "./protocol.js";
|
|
28
|
+
import { RUNNER_COMMAND, STEP_ENV, stepResultLinePrefix, stepPauseLinePrefix, requestContextPath, stepInputPath, stepResultFilePath, stepLogFilePath, stepPidFilePath } from "./protocol.js";
|
|
29
29
|
import { StepPauseRequestSchema } from "./types.js";
|
|
30
30
|
import type { StepRequest, StepResult } from "./types.js";
|
|
31
31
|
import type { StepObservability } from "../workflow-steps/observability.js";
|
|
@@ -194,6 +194,18 @@ export interface InvokeStepOptions {
|
|
|
194
194
|
* the caller's job; the activity batches and inserts at step completion. */
|
|
195
195
|
onStdout?: (line: string) => void;
|
|
196
196
|
onStderr?: (line: string) => void;
|
|
197
|
+
/** Called when the LIVE output stream fails mid-run (e.g. E2B's connect-web
|
|
198
|
+
* transport throws `received unsupported compressed output` on a large
|
|
199
|
+
* compressed frame) and the invoker falls back to recovering the result +
|
|
200
|
+
* full logs from the durable token-keyed files. The run is NOT failed — this
|
|
201
|
+
* is the hook to emit a structured alert so the degradation is visible/paged.
|
|
202
|
+
* Best-effort: keep it cheap and non-throwing. */
|
|
203
|
+
onStreamDegraded?: (info: { error: unknown; runnerPid: number; resultToken: string }) => void;
|
|
204
|
+
/** Ceiling on the durable-result recovery poll after a live-stream fault
|
|
205
|
+
* (default 45m — the wedged-runner backstop). The activity passes its step
|
|
206
|
+
* ceiling so a stream fault on a step with hours of legitimate work left
|
|
207
|
+
* recovers instead of timing out; startToClose is the real wall. */
|
|
208
|
+
recoveryDeadlineMs?: number;
|
|
197
209
|
}
|
|
198
210
|
|
|
199
211
|
/** A step running as a background command (ADR-0028). The activity races its
|
|
@@ -219,13 +231,18 @@ export interface RunningStep<TOutput = unknown> {
|
|
|
219
231
|
function makeStreamSplitters(
|
|
220
232
|
opts: Pick<InvokeStepOptions, "onStdout" | "onStderr"> | undefined,
|
|
221
233
|
resultToken: string,
|
|
222
|
-
): { onStdout?: (chunk: string) => void; onStderr?: (chunk: string) => void; flush: () => void } {
|
|
234
|
+
): { onStdout?: (chunk: string) => void; onStderr?: (chunk: string) => void; flush: () => void; stdoutConsumedChars: () => number } {
|
|
223
235
|
const resultSentinel = stepResultLinePrefix(resultToken);
|
|
224
236
|
const pauseSentinel = stepPauseLinePrefix(resultToken);
|
|
225
237
|
const isSentinel = (line: string) => line.startsWith(resultSentinel) || line.startsWith(pauseSentinel);
|
|
226
238
|
const make = (sink: ((line: string) => void) | undefined, filterSentinel: boolean) => {
|
|
227
|
-
if (!sink) return { onChunk: undefined, flush: () => {} };
|
|
239
|
+
if (!sink) return { onChunk: undefined, flush: () => {}, consumed: () => 0 };
|
|
228
240
|
let buf = "";
|
|
241
|
+
// Chars consumed as WHOLE lines (incl. each trailing "\n"), counting sentinel
|
|
242
|
+
// lines too — this is the offset into the tee'd log file up to which the live
|
|
243
|
+
// stream has already delivered. The recovery backfill starts exactly here, so
|
|
244
|
+
// an in-flight partial line is re-read whole from the file (no split, no dup).
|
|
245
|
+
let consumed = 0;
|
|
229
246
|
return {
|
|
230
247
|
onChunk: (chunk: string) => {
|
|
231
248
|
buf += chunk;
|
|
@@ -233,6 +250,7 @@ function makeStreamSplitters(
|
|
|
233
250
|
while ((nl = buf.indexOf("\n")) !== -1) {
|
|
234
251
|
const line = buf.slice(0, nl);
|
|
235
252
|
buf = buf.slice(nl + 1);
|
|
253
|
+
consumed += line.length + 1;
|
|
236
254
|
if (filterSentinel && isSentinel(line)) continue;
|
|
237
255
|
sink(line);
|
|
238
256
|
}
|
|
@@ -241,9 +259,11 @@ function makeStreamSplitters(
|
|
|
241
259
|
if (buf.length === 0) return;
|
|
242
260
|
const line = buf;
|
|
243
261
|
buf = "";
|
|
262
|
+
consumed += line.length;
|
|
244
263
|
if (filterSentinel && isSentinel(line)) return;
|
|
245
264
|
sink(line);
|
|
246
265
|
},
|
|
266
|
+
consumed: () => consumed,
|
|
247
267
|
};
|
|
248
268
|
};
|
|
249
269
|
const stdout = make(opts?.onStdout, true);
|
|
@@ -252,6 +272,7 @@ function makeStreamSplitters(
|
|
|
252
272
|
...(stdout.onChunk ? { onStdout: stdout.onChunk } : {}),
|
|
253
273
|
...(stderr.onChunk ? { onStderr: stderr.onChunk } : {}),
|
|
254
274
|
flush: () => { stdout.flush(); stderr.flush(); },
|
|
275
|
+
stdoutConsumedChars: () => stdout.consumed(),
|
|
255
276
|
};
|
|
256
277
|
}
|
|
257
278
|
|
|
@@ -307,6 +328,174 @@ async function classifyRunnerOutcome<TOutput>(
|
|
|
307
328
|
};
|
|
308
329
|
}
|
|
309
330
|
|
|
331
|
+
/** Poll the runner's durable result file (token-keyed, written immediately before
|
|
332
|
+
* exit on EVERY path — success, failure, pause) as the AUTHORITATIVE completion
|
|
333
|
+
* signal, using `/proc/<pid>` liveness to tell work-in-progress from a crash.
|
|
334
|
+
* Used wherever the live stream cannot be trusted for completion: the native
|
|
335
|
+
* resume path (a reconnected handle's `wait()` can resolve early after a VM
|
|
336
|
+
* pause/resume) AND the launch path after a live-stream transport fault.
|
|
337
|
+
* `cat`-ing the result file is itself immune to the large-frame compression bug
|
|
338
|
+
* — it is a single small JSON line, far below any compression threshold. The
|
|
339
|
+
* activity's `startToCloseTimeout` is the real upper bound; the deadline here
|
|
340
|
+
* is a backstop so a wedged runner can't leak this loop in the worker forever
|
|
341
|
+
* (default 45m; a worker-death recovery reconnecting mid-step passes the full
|
|
342
|
+
* step ceiling instead — hours of legitimate work may remain). */
|
|
343
|
+
async function pollDurableResult<TOutput>(
|
|
344
|
+
sandbox: SandboxProvider,
|
|
345
|
+
runnerPid: number,
|
|
346
|
+
resultToken: string,
|
|
347
|
+
opts?: { signal?: AbortSignal; deadlineMs?: number },
|
|
348
|
+
): Promise<StepResult<TOutput>> {
|
|
349
|
+
const signal = opts?.signal;
|
|
350
|
+
const resultFile = stepResultFilePath(resultToken);
|
|
351
|
+
const POLL_MS = 1000;
|
|
352
|
+
const deadlineMs = opts?.deadlineMs ?? 45 * 60_000;
|
|
353
|
+
const DEADLINE = Date.now() + deadlineMs;
|
|
354
|
+
// Both internal commands carry an explicit budget. Required for Vercel:
|
|
355
|
+
// without `timeoutMs` there is no hard client-side deadline in the provider's
|
|
356
|
+
// `commands.run`, so a wedged VM hangs one `cat` forever and the poll
|
|
357
|
+
// deadline never advances. A timeout throws → the `.catch(() => null)`
|
|
358
|
+
// treats it as a failed read/probe and the loop continues (a hung probe
|
|
359
|
+
// never misclassifies as dead). Harmless on E2B — a 1-line cat is ms-fast.
|
|
360
|
+
const CMD_TIMEOUT_MS = 10_000;
|
|
361
|
+
// A destroyed sandbox (e.g. run cancelled → sandbox killed mid-recovery)
|
|
362
|
+
// rejects EVERY command, which a lone `.catch(() => null)` renders
|
|
363
|
+
// indistinguishable from a live runner — the loop would spin until the
|
|
364
|
+
// deadline (hours) against a VM that no longer exists. Track consecutive
|
|
365
|
+
// iterations where BOTH the result read and the liveness probe rejected and
|
|
366
|
+
// exit with an honest runner-exit once the streak is unambiguous. 20
|
|
367
|
+
// iterations ≈ 30s–7min of continuous rejections (each command may burn its
|
|
368
|
+
// 10s budget) — long enough to ride out a transient provider-API blip,
|
|
369
|
+
// short enough that a dead sandbox doesn't hold the worker slot for hours.
|
|
370
|
+
const UNREACHABLE_LIMIT = 20;
|
|
371
|
+
let unreachableStreak = 0;
|
|
372
|
+
const readResult = async (): Promise<StepResult<TOutput> | "unreachable" | null> => {
|
|
373
|
+
const r = await sandbox.commands.run(`cat ${resultFile} 2>/dev/null || true`, { timeoutMs: CMD_TIMEOUT_MS }).catch(() => "unreachable" as const);
|
|
374
|
+
if (r === "unreachable") return "unreachable";
|
|
375
|
+
return r.stdout ? (parseStepResult<TOutput>(r.stdout, resultToken) ?? null) : null;
|
|
376
|
+
};
|
|
377
|
+
for (let attempt = 1; ; attempt++) {
|
|
378
|
+
// Aborted = the activity already resolved via the pause watcher (a re-pause);
|
|
379
|
+
// this poll's result is now unobserved. Return (never throw — the promise is
|
|
380
|
+
// no longer awaited) so it settles cleanly with no unhandled rejection.
|
|
381
|
+
if (signal?.aborted) {
|
|
382
|
+
return { ok: false, error: { kind: "runner-exit", message: "reconnectStep aborted (run re-paused)", exitCode: 1 } };
|
|
383
|
+
}
|
|
384
|
+
const fromFile = await readResult();
|
|
385
|
+
if (fromFile !== null && fromFile !== "unreachable") return fromFile;
|
|
386
|
+
|
|
387
|
+
const probe = await sandbox.commands
|
|
388
|
+
.run(`test -d /proc/${runnerPid} && echo alive || echo dead`, { timeoutMs: CMD_TIMEOUT_MS })
|
|
389
|
+
.catch(() => null);
|
|
390
|
+
if (fromFile === "unreachable" && probe === null) {
|
|
391
|
+
unreachableStreak++;
|
|
392
|
+
if (unreachableStreak >= UNREACHABLE_LIMIT) {
|
|
393
|
+
return {
|
|
394
|
+
ok: false,
|
|
395
|
+
error: {
|
|
396
|
+
kind: "runner-exit",
|
|
397
|
+
message: `sandbox stopped responding while waiting for runner pid ${runnerPid} to emit a result (${UNREACHABLE_LIMIT} consecutive failed polls)`,
|
|
398
|
+
exitCode: 1,
|
|
399
|
+
},
|
|
400
|
+
};
|
|
401
|
+
}
|
|
402
|
+
} else {
|
|
403
|
+
unreachableStreak = 0;
|
|
404
|
+
}
|
|
405
|
+
const dead = (probe?.stdout ?? "").includes("dead");
|
|
406
|
+
process.stderr.write(
|
|
407
|
+
`[recover] poll ${attempt}: result-file=absent runner=${dead ? "dead" : "alive"} pid=${runnerPid}\n`,
|
|
408
|
+
);
|
|
409
|
+
if (dead) {
|
|
410
|
+
// Close the write-then-exit race with one final read, else classify a crash.
|
|
411
|
+
const finalRead = await readResult();
|
|
412
|
+
if (finalRead !== null && finalRead !== "unreachable") return finalRead;
|
|
413
|
+
return {
|
|
414
|
+
ok: false,
|
|
415
|
+
error: { kind: "runner-exit", message: `runner pid ${runnerPid} exited without writing a result file`, exitCode: 1 },
|
|
416
|
+
};
|
|
417
|
+
}
|
|
418
|
+
if (Date.now() > DEADLINE) {
|
|
419
|
+
return {
|
|
420
|
+
ok: false,
|
|
421
|
+
error: {
|
|
422
|
+
kind: "runner-exit",
|
|
423
|
+
message: `timed out after ${Math.round(deadlineMs / 60_000)}m waiting for runner pid ${runnerPid} to emit a result`,
|
|
424
|
+
exitCode: 1,
|
|
425
|
+
},
|
|
426
|
+
};
|
|
427
|
+
}
|
|
428
|
+
await new Promise((r) => setTimeout(r, POLL_MS));
|
|
429
|
+
}
|
|
430
|
+
}
|
|
431
|
+
|
|
432
|
+
/** Recover a step's result AND its full logs after the live output stream died
|
|
433
|
+
* mid-run (the dominant cause being E2B's connect-web transport throwing on a
|
|
434
|
+
* compressed large frame — a LOG-TRANSPORT fault, not a runner failure). Waits
|
|
435
|
+
* for the runner to actually finish (`pollDurableResult`), then reads the tee'd
|
|
436
|
+
* log file back over the HTTP file transport (compression-immune) and backfills
|
|
437
|
+
* only the lines the dead stream never delivered — everything past the last
|
|
438
|
+
* whole line already emitted live (`stdoutConsumedChars`), so no duplication and
|
|
439
|
+
* no split lines. Logs BEFORE the fault are already persisted by the activity's
|
|
440
|
+
* live flush; this restores the tail so "what happened" survives intact. */
|
|
441
|
+
async function recoverLogsAndResult<TOutput>(
|
|
442
|
+
sandbox: SandboxProvider,
|
|
443
|
+
runnerPid: number,
|
|
444
|
+
resultToken: string,
|
|
445
|
+
liveSplitters: { stdoutConsumedChars: () => number },
|
|
446
|
+
opts: Pick<InvokeStepOptions, "onStdout" | "onStderr"> | undefined,
|
|
447
|
+
pollOpts?: { deadlineMs?: number },
|
|
448
|
+
): Promise<StepResult<TOutput>> {
|
|
449
|
+
const result = await pollDurableResult<TOutput>(sandbox, runnerPid, resultToken, pollOpts);
|
|
450
|
+
// Runner has exited → the tee'd log file is complete. Best-effort: a failed
|
|
451
|
+
// readback just means the recovered run keeps the live logs it already had.
|
|
452
|
+
if (sandbox.files.read && opts?.onStdout) {
|
|
453
|
+
const fullLog = await sandbox.files.read(stepLogFilePath(resultToken)).catch(() => null);
|
|
454
|
+
if (fullLog != null) {
|
|
455
|
+
const already = liveSplitters.stdoutConsumedChars();
|
|
456
|
+
const tail = fullLog.length > already ? fullLog.slice(already) : "";
|
|
457
|
+
if (tail.length > 0) {
|
|
458
|
+
const backfill = makeStreamSplitters(opts, resultToken);
|
|
459
|
+
backfill.onStdout?.(tail);
|
|
460
|
+
backfill.flush();
|
|
461
|
+
}
|
|
462
|
+
}
|
|
463
|
+
}
|
|
464
|
+
return result;
|
|
465
|
+
}
|
|
466
|
+
|
|
467
|
+
/** Read the foreground launch's pidfile back — the /proc liveness handle for
|
|
468
|
+
* the recovery poll. Returns `null` when recovery is impossible, for either
|
|
469
|
+
* reason: the sandbox cannot run ANY command (genuinely dead VM), or the
|
|
470
|
+
* pidfile never appeared. The pidfile write is the FIRST statement of the
|
|
471
|
+
* runner pipeline, so a still-absent pid after the retries means the runner
|
|
472
|
+
* never started and no result file will ever exist — recovery would poll
|
|
473
|
+
* blind for the full step ceiling and then misclassify; the caller must
|
|
474
|
+
* rethrow the original terminal error instead (fast, honest). The retries
|
|
475
|
+
* cover both a transiently-unreachable sandbox and the launch race where the
|
|
476
|
+
* stream fault fired before the VM executed the pidfile write. */
|
|
477
|
+
async function readRunnerPid(
|
|
478
|
+
sandbox: SandboxProvider,
|
|
479
|
+
resultToken: string,
|
|
480
|
+
): Promise<number | null> {
|
|
481
|
+
const ATTEMPTS = 3;
|
|
482
|
+
const RETRY_DELAY_MS = 2_000;
|
|
483
|
+
for (let attempt = 1; attempt <= ATTEMPTS; attempt++) {
|
|
484
|
+
try {
|
|
485
|
+
const r = await sandbox.commands.run(
|
|
486
|
+
`cat ${stepPidFilePath(resultToken)} 2>/dev/null || true`,
|
|
487
|
+
{ timeoutMs: 10_000 },
|
|
488
|
+
);
|
|
489
|
+
const pid = Number.parseInt(r.stdout.trim(), 10);
|
|
490
|
+
if (Number.isInteger(pid) && pid > 0) return pid;
|
|
491
|
+
} catch {
|
|
492
|
+
// Sandbox unreachable this attempt — fall through to the retry delay.
|
|
493
|
+
}
|
|
494
|
+
if (attempt < ATTEMPTS) await new Promise((r) => setTimeout(r, RETRY_DELAY_MS));
|
|
495
|
+
}
|
|
496
|
+
return null;
|
|
497
|
+
}
|
|
498
|
+
|
|
310
499
|
/** Write the step's input + request-context files and build the runner env.
|
|
311
500
|
* Shared by the foreground and background launch paths. Returns the
|
|
312
501
|
* per-invocation `resultToken` + the env map. */
|
|
@@ -337,20 +526,60 @@ export async function invokeStep<TOutput = unknown>(
|
|
|
337
526
|
|
|
338
527
|
let result: { stdout: string; stderr: string; exitCode: number };
|
|
339
528
|
try {
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
529
|
+
// Run the agent as ROOT (ADR-0019 follow-up). The factory drive (Archil)
|
|
530
|
+
// presents its S3-synced files root-owned and exposes no uid-mapped mount,
|
|
531
|
+
// so a non-root agent EACCESes on every shared file — the reason for the
|
|
532
|
+
// expensive boot-time `chmod -R` walk. As root the agent writes them
|
|
533
|
+
// directly: the walk disappears entirely. Egress is edge-enforced with NO
|
|
534
|
+
// root exemption (see sandbox.ts), so root is confined exactly like the
|
|
535
|
+
// non-root user. `HOME=/root` so the spawned `claude` finds root's skills.
|
|
536
|
+
//
|
|
537
|
+
// Same pipeline pattern as `launchStep`: tee the runner's stdout to the
|
|
538
|
+
// durable token-keyed log file so a live-stream fault can backfill the
|
|
539
|
+
// undelivered tail, and write the wrapping shell's pid (`$$`) to a pidfile
|
|
540
|
+
// first — the shell waits on the pipeline, so `/proc/<pid>` liveness is a
|
|
541
|
+
// correct "runner may still write a result" proxy for the recovery poll.
|
|
542
|
+
// `set -o pipefail` is REQUIRED: without it the pipeline's exit code is
|
|
543
|
+
// tee's (0), masking a non-zero runner exit and breaking the runner-exit
|
|
544
|
+
// classification (the provider execs via `sh -c`, where pipefail works).
|
|
545
|
+
result = await sandbox.commands.run(
|
|
546
|
+
`set -o pipefail; echo $$ > ${stepPidFilePath(resultToken)}; ${RUNNER_COMMAND} | tee ${stepLogFilePath(resultToken)}`,
|
|
547
|
+
{
|
|
548
|
+
envs: { ...envs, HOME: "/root", IS_SANDBOX: "1" },
|
|
549
|
+
sudo: true,
|
|
550
|
+
timeoutMs: 0,
|
|
551
|
+
...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
|
|
552
|
+
...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
|
|
553
|
+
},
|
|
554
|
+
);
|
|
346
555
|
} catch (e) {
|
|
347
556
|
// Typed sandbox-infrastructure failure: the engine's recovery contract
|
|
348
|
-
//
|
|
349
|
-
//
|
|
350
|
-
//
|
|
351
|
-
//
|
|
352
|
-
|
|
353
|
-
|
|
557
|
+
// keys on retryability. Its `[sandbox-unavailable:*]` message prefix must
|
|
558
|
+
// reach the workflow as the failure leaf, not ride an embedded stderr tail
|
|
559
|
+
// that a later slice(-2000) can drop. Handle before the CommandExitError
|
|
560
|
+
// downgrade below.
|
|
561
|
+
if (e instanceof SandboxUnavailableError) {
|
|
562
|
+
// Pre-launch (retryable): nothing user-side ran — propagate INTACT so the
|
|
563
|
+
// workflow layer re-provisions (identity + prefix contract; pinned test).
|
|
564
|
+
if (e.retryable) throw e;
|
|
565
|
+
// Post-launch terminal: the LIVE stream/wait call died (e.g. Vercel's raw
|
|
566
|
+
// "The operation timed out.") while the runner is very likely still alive
|
|
567
|
+
// and will write its durable result file. Recover instead of failing —
|
|
568
|
+
// NEVER re-launch the runner. A null pid means recovery is impossible
|
|
569
|
+
// (sandbox can't run ANY command, or the pidfile never appeared because
|
|
570
|
+
// the runner never actually started): rethrow the original terminal
|
|
571
|
+
// classification fast rather than polling blind for the step ceiling.
|
|
572
|
+
// We do NOT flush the live splitter here — the recovery backfills from
|
|
573
|
+
// the last whole line, re-reading any in-flight partial line whole from
|
|
574
|
+
// the durable log (avoids a split/duplicated line).
|
|
575
|
+
const runnerPid = await readRunnerPid(sandbox, resultToken);
|
|
576
|
+
if (runnerPid === null) throw e;
|
|
577
|
+
opts?.onStreamDegraded?.({ error: e, runnerPid, resultToken });
|
|
578
|
+
return recoverLogsAndResult<TOutput>(
|
|
579
|
+
sandbox, runnerPid, resultToken, splitters, opts,
|
|
580
|
+
opts?.recoveryDeadlineMs !== undefined ? { deadlineMs: opts.recoveryDeadlineMs } : undefined,
|
|
581
|
+
);
|
|
582
|
+
}
|
|
354
583
|
// Some providers (E2B) throw a CommandExitError on a non-zero exit instead of
|
|
355
584
|
// returning it. Recover stdout/stderr/exitCode from the error so we can STILL
|
|
356
585
|
// parse the runner's structured `__AC_STEP_RESULT__` payload — otherwise the
|
|
@@ -384,8 +613,25 @@ export async function launchStep<TOutput = unknown>(
|
|
|
384
613
|
}
|
|
385
614
|
const { resultToken, envs } = await prepareStepLaunch(sandbox, request, opts);
|
|
386
615
|
const splitters = makeStreamSplitters(opts, resultToken);
|
|
387
|
-
|
|
388
|
-
|
|
616
|
+
// Tee the runner's stdout to a durable, token-keyed LOG file (stderr stays a
|
|
617
|
+
// separate live stream). The live output rides E2B's connect-web command
|
|
618
|
+
// stream, which THROWS on a compressed large frame (gRPC-web cannot decode
|
|
619
|
+
// message compression) and kills the feed mid-run. The tee'd file, read back
|
|
620
|
+
// over the envd HTTP transport (compression-immune), lets `wait()` recover the
|
|
621
|
+
// full logs + result instead of failing a run that actually completed. `tee`
|
|
622
|
+
// runs in the same `bash -c` pipeline E2B already wraps the command in, so the
|
|
623
|
+
// background pid (the pause / reconnect-by-pid handle) is unchanged; the result
|
|
624
|
+
// sentinel still rides stdout (tee passes it through) and the result FILE is
|
|
625
|
+
// written by the runner directly, so the happy path is untouched. `set -o
|
|
626
|
+
// pipefail` is REQUIRED: without it the pipeline's exit code is tee's (0),
|
|
627
|
+
// masking a non-zero runner exit and breaking the runner-exit classification;
|
|
628
|
+
// pipefail propagates the runner's code (E2B execs via `bash -c`).
|
|
629
|
+
//
|
|
630
|
+
// Run the agent as ROOT — see invokeStep above. `sudo:true` → user:"root";
|
|
631
|
+
// `HOME=/root` so the spawned `claude` finds root's skills.
|
|
632
|
+
const proc = await sandbox.commands.runBackground(`set -o pipefail; ${RUNNER_COMMAND} | tee ${stepLogFilePath(resultToken)}`, {
|
|
633
|
+
envs: { ...envs, HOME: "/root", IS_SANDBOX: "1" },
|
|
634
|
+
sudo: true,
|
|
389
635
|
timeoutMs: 0,
|
|
390
636
|
...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
|
|
391
637
|
...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
|
|
@@ -394,9 +640,25 @@ export async function launchStep<TOutput = unknown>(
|
|
|
394
640
|
runnerPid: proc.pid,
|
|
395
641
|
resultToken,
|
|
396
642
|
async wait() {
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
643
|
+
try {
|
|
644
|
+
const result = await proc.wait();
|
|
645
|
+
splitters.flush();
|
|
646
|
+
return classifyRunnerOutcome<TOutput>(sandbox, result, resultToken);
|
|
647
|
+
} catch (e) {
|
|
648
|
+
// A genuine infra death must propagate so the engine re-provisions.
|
|
649
|
+
if (e instanceof SandboxUnavailableError) throw e;
|
|
650
|
+
// Otherwise the runner's LIVE output stream died while the runner itself
|
|
651
|
+
// is alive and writing its durable result + tee'd log files. The dominant
|
|
652
|
+
// cause is E2B's connect-web transport throwing on a compressed large
|
|
653
|
+
// frame ("received unsupported compressed output") — a LOG-TRANSPORT
|
|
654
|
+
// fault, NEVER a reason to fail a run that completed. Degrade the live
|
|
655
|
+
// feed, alert, and recover the result + full logs over the HTTP file
|
|
656
|
+
// transport. We do NOT flush the live splitter here — the recovery
|
|
657
|
+
// backfills from the last whole line, re-reading any in-flight partial
|
|
658
|
+
// line whole from the durable log (avoids a split/duplicated line).
|
|
659
|
+
opts?.onStreamDegraded?.({ error: e, runnerPid: proc.pid, resultToken });
|
|
660
|
+
return recoverLogsAndResult<TOutput>(sandbox, proc.pid, resultToken, splitters, opts);
|
|
661
|
+
}
|
|
400
662
|
},
|
|
401
663
|
};
|
|
402
664
|
}
|
|
@@ -414,38 +676,61 @@ export async function launchStep<TOutput = unknown>(
|
|
|
414
676
|
export async function reconnectStep<TOutput = unknown>(
|
|
415
677
|
sandbox: SandboxProvider,
|
|
416
678
|
resume: { runnerPid: number; resultToken: string; stepIndex: number },
|
|
417
|
-
opts?: Pick<InvokeStepOptions, "onStdout" | "onStderr"
|
|
679
|
+
opts?: Pick<InvokeStepOptions, "onStdout" | "onStderr"> & {
|
|
680
|
+
signal?: AbortSignal;
|
|
681
|
+
/** Ceiling on the durable-result poll (default 45m — the wedged-runner
|
|
682
|
+
* backstop). A worker-death recovery reconnecting MID-step passes the
|
|
683
|
+
* step ceiling instead: hours of legitimate work may remain, and the
|
|
684
|
+
* activity's per-attempt startToClose already bounds it server-side. */
|
|
685
|
+
pollDeadlineMs?: number;
|
|
686
|
+
},
|
|
418
687
|
): Promise<RunningStep<TOutput>> {
|
|
419
688
|
if (!sandbox.commands.connectProcess) {
|
|
420
689
|
throw new Error("reconnectStep requires a provider with background-command support (commands.connectProcess)");
|
|
421
690
|
}
|
|
422
691
|
const { runnerPid, resultToken } = resume;
|
|
423
692
|
const splitters = makeStreamSplitters(opts, resultToken);
|
|
424
|
-
|
|
693
|
+
const signal = opts?.signal;
|
|
694
|
+
const pollDeadlineMs = opts?.pollDeadlineMs;
|
|
695
|
+
|
|
696
|
+
// Re-attach to the suspended runner for LIVE stdout streaming only. This is
|
|
697
|
+
// best-effort: a reconnect that throws after the native resume just means no
|
|
698
|
+
// live feed for the dashboard — completion is read from the durable result
|
|
699
|
+
// file below, NOT from this handle.
|
|
700
|
+
let attached: SandboxBackgroundProcess | null = null;
|
|
425
701
|
try {
|
|
426
|
-
|
|
702
|
+
attached = await sandbox.commands.connectProcess(runnerPid, {
|
|
427
703
|
...(splitters.onStdout ? { onStdout: splitters.onStdout } : {}),
|
|
428
704
|
...(splitters.onStderr ? { onStderr: splitters.onStderr } : {}),
|
|
429
705
|
});
|
|
430
706
|
} catch (e) {
|
|
431
|
-
// A sandbox-infrastructure failure must propagate intact (recovery
|
|
432
|
-
// contract). Anything else means the runner already exited during resume
|
|
433
|
-
// (we re-attached too late) — its durable result file is written, so the
|
|
434
|
-
// handle's wait() classifies from the file below.
|
|
435
707
|
if (e instanceof SandboxUnavailableError) throw e;
|
|
436
708
|
}
|
|
437
|
-
|
|
709
|
+
|
|
438
710
|
return {
|
|
439
711
|
runnerPid,
|
|
440
712
|
resultToken,
|
|
441
713
|
async wait() {
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
714
|
+
// CRITICAL: do NOT trust `connect(pid).wait()` on the resume path. After a
|
|
715
|
+
// native VM pause/resume, E2B's re-attached handle can resolve its wait()
|
|
716
|
+
// EARLY — the reconnected stdout stream closes while the runner process is
|
|
717
|
+
// still alive and working — returning a default `{exitCode:0}`. That made
|
|
718
|
+
// the worker report "exited 0 without emitting" and force-kill a runner
|
|
719
|
+
// that was mid-iteration. So ignore the handle for completion and poll the
|
|
720
|
+
// runner's durable result file + `/proc` liveness instead — the same
|
|
721
|
+
// authoritative signal the launch-path recovery uses (`pollDurableResult`).
|
|
722
|
+
try {
|
|
723
|
+
const result = await pollDurableResult<TOutput>(sandbox, runnerPid, resultToken, {
|
|
724
|
+
...(signal ? { signal } : {}),
|
|
725
|
+
...(pollDeadlineMs !== undefined ? { deadlineMs: pollDeadlineMs } : {}),
|
|
726
|
+
});
|
|
727
|
+
splitters.flush();
|
|
728
|
+
return result;
|
|
729
|
+
} finally {
|
|
730
|
+
// Release the streaming handle (keeps it un-GC'd for the duration of the
|
|
731
|
+
// poll above; a dangling reconnect after a re-pause is cleaned up here).
|
|
732
|
+
await attached?.kill().catch(() => {});
|
|
446
733
|
}
|
|
447
|
-
splitters.flush();
|
|
448
|
-
return classifyRunnerOutcome<TOutput>(sandbox, output, resultToken);
|
|
449
734
|
},
|
|
450
735
|
};
|
|
451
736
|
}
|
|
@@ -80,3 +80,22 @@ export function requestContextPath(stepIndex: number): string {
|
|
|
80
80
|
export function stepResultFilePath(token: string): string {
|
|
81
81
|
return `/tmp/wf/step-result-${token}.json`;
|
|
82
82
|
}
|
|
83
|
+
|
|
84
|
+
/** Sandbox-side pidfile the FOREGROUND launch writes (`echo $$ > …`) before
|
|
85
|
+
* spawning the runner pipeline — the wrapping shell's pid. The recovery path
|
|
86
|
+
* reads it back to probe `/proc/<pid>` liveness when the live stream/wait
|
|
87
|
+
* call dies mid-step (providers with no background-process pid — Vercel). */
|
|
88
|
+
export function stepPidFilePath(token: string): string {
|
|
89
|
+
return `/tmp/wf/step-pid-${token}.pid`;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/** Sandbox-side path where the runner's stdout is tee'd as a durable LOG file,
|
|
93
|
+
* keyed by the per-invocation token. The live output rides E2B's connect-web
|
|
94
|
+
* command stream, which THROWS on a compressed large frame (gRPC-web cannot
|
|
95
|
+
* decode message compression) and kills the feed mid-run. This file, read back
|
|
96
|
+
* over the envd HTTP file transport (compression-immune), lets the invoker
|
|
97
|
+
* recover the FULL logs after such a fault instead of losing the tail — the
|
|
98
|
+
* log-side analogue of `stepResultFilePath` for the result. */
|
|
99
|
+
export function stepLogFilePath(token: string): string {
|
|
100
|
+
return `/tmp/wf/step-log-${token}.log`;
|
|
101
|
+
}
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Break-glass compliance session wire types (ADR-0051).
|
|
3
|
+
*
|
|
4
|
+
* A team admin has no default access to a scoped (Private/Shared) document.
|
|
5
|
+
* To view one they don't hold a grant on, they open a READ-ONLY,
|
|
6
|
+
* owner-approved, time-boxed compliance session over a scope (one factory, or
|
|
7
|
+
* the whole team). These are the shapes the server's `/api/v1/compliance/*`
|
|
8
|
+
* routes emit — `client.ts` is a thin typed wrapper over them.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
export type ComplianceScopeKind = "factory" | "team";
|
|
12
|
+
export type ComplianceStatus = "pending" | "active" | "expired" | "revoked";
|
|
13
|
+
|
|
14
|
+
/** One break-glass compliance session. `expiresAt` is null while pending —
|
|
15
|
+
* the TTL clock starts at approval, not request. */
|
|
16
|
+
export interface ComplianceSession {
|
|
17
|
+
id: string;
|
|
18
|
+
teamId: string;
|
|
19
|
+
requestedByUserId: string | null;
|
|
20
|
+
requestedByLabel: string | null;
|
|
21
|
+
approvedByUserId: string | null;
|
|
22
|
+
approvedByLabel: string | null;
|
|
23
|
+
reason: string;
|
|
24
|
+
scopeKind: ComplianceScopeKind;
|
|
25
|
+
/** The factory this session covers (factory scope); null for team scope. */
|
|
26
|
+
scopeRef: string | null;
|
|
27
|
+
status: ComplianceStatus;
|
|
28
|
+
ttlSeconds: number;
|
|
29
|
+
expiresAt: string | null;
|
|
30
|
+
revokedByUserId: string | null;
|
|
31
|
+
revokedByLabel: string | null;
|
|
32
|
+
createdAt: string;
|
|
33
|
+
approvedAt: string | null;
|
|
34
|
+
revokedAt: string | null;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/** One per-document access recorded under an active session — the individual
|
|
38
|
+
* audit trail the session's reasoned entry authorizes. */
|
|
39
|
+
export interface ComplianceAccess {
|
|
40
|
+
id: string;
|
|
41
|
+
sessionId: string;
|
|
42
|
+
/** Nulled when the accessed file is later deleted (ON DELETE SET NULL);
|
|
43
|
+
* `filePath` is the durable snapshot that survives. */
|
|
44
|
+
fileId: string | null;
|
|
45
|
+
filePath: string;
|
|
46
|
+
accessedByUserId: string | null;
|
|
47
|
+
accessedByLabel: string | null;
|
|
48
|
+
accessedAt: string;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
export interface RequestComplianceSessionInput {
|
|
52
|
+
scopeKind: ComplianceScopeKind;
|
|
53
|
+
/** Required for factory scope; must be omitted for team scope. */
|
|
54
|
+
scopeRef?: string | null;
|
|
55
|
+
reason: string;
|
|
56
|
+
/** Session lifetime once approved — 15 min floor, 7 day ceiling. */
|
|
57
|
+
ttlSeconds: number;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export interface ListComplianceSessionsOptions {
|
|
61
|
+
status?: ComplianceStatus;
|
|
62
|
+
limit?: number;
|
|
63
|
+
cursor?: string;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
export interface ComplianceSessionsPage {
|
|
67
|
+
sessions: ComplianceSession[];
|
|
68
|
+
nextCursor: string | null;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
export interface ListComplianceAccessesOptions {
|
|
72
|
+
limit?: number;
|
|
73
|
+
cursor?: string;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
export interface ComplianceAccessesPage {
|
|
77
|
+
accesses: ComplianceAccess[];
|
|
78
|
+
nextCursor: string | null;
|
|
79
|
+
}
|