@stigmer/runner 3.0.8-dev.20260613085218 → 3.0.9-dev.20260615145121

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/dist/.build-fingerprint +1 -1
  2. package/dist/activities/execute-cursor/hook-script.d.ts +23 -12
  3. package/dist/activities/execute-cursor/hook-script.js +85 -51
  4. package/dist/activities/execute-cursor/hook-script.js.map +1 -1
  5. package/dist/activities/execute-cursor/index.js +220 -79
  6. package/dist/activities/execute-cursor/index.js.map +1 -1
  7. package/dist/activities/execute-cursor/message-translator.d.ts +51 -9
  8. package/dist/activities/execute-cursor/message-translator.js +146 -20
  9. package/dist/activities/execute-cursor/message-translator.js.map +1 -1
  10. package/dist/activities/execute-cursor/persist-decision.d.ts +42 -0
  11. package/dist/activities/execute-cursor/persist-decision.js +30 -0
  12. package/dist/activities/execute-cursor/persist-decision.js.map +1 -0
  13. package/dist/activities/execute-cursor/prompt-builder.d.ts +25 -0
  14. package/dist/activities/execute-cursor/prompt-builder.js +54 -0
  15. package/dist/activities/execute-cursor/prompt-builder.js.map +1 -1
  16. package/dist/activities/execute-cursor/workspace-setup.d.ts +8 -2
  17. package/dist/activities/execute-cursor/workspace-setup.js +62 -30
  18. package/dist/activities/execute-cursor/workspace-setup.js.map +1 -1
  19. package/dist/activities/execute-deep-agent/index.js +15 -5
  20. package/dist/activities/execute-deep-agent/index.js.map +1 -1
  21. package/dist/activities/execute-deep-agent/status-builder-shared.d.ts +0 -1
  22. package/dist/activities/execute-deep-agent/status-builder-shared.js +32 -8
  23. package/dist/activities/execute-deep-agent/status-builder-shared.js.map +1 -1
  24. package/dist/activities/execute-deep-agent/status-builder.js +4 -5
  25. package/dist/activities/execute-deep-agent/status-builder.js.map +1 -1
  26. package/dist/activities/execute-deep-agent/streaming-v3.js +4 -5
  27. package/dist/activities/execute-deep-agent/streaming-v3.js.map +1 -1
  28. package/dist/activities/execute-deep-agent/streaming.d.ts +9 -1
  29. package/dist/activities/execute-deep-agent/streaming.js +4 -5
  30. package/dist/activities/execute-deep-agent/streaming.js.map +1 -1
  31. package/dist/activities/execute-deep-agent/subagent-tracker.js +4 -5
  32. package/dist/activities/execute-deep-agent/subagent-tracker.js.map +1 -1
  33. package/dist/activities/execute-deep-agent/v3-status-builder.js +6 -5
  34. package/dist/activities/execute-deep-agent/v3-status-builder.js.map +1 -1
  35. package/dist/config.d.ts +21 -0
  36. package/dist/config.js +12 -0
  37. package/dist/config.js.map +1 -1
  38. package/dist/in-flight.d.ts +35 -0
  39. package/dist/in-flight.js +61 -0
  40. package/dist/in-flight.js.map +1 -0
  41. package/dist/runner-manager.d.ts +2 -0
  42. package/dist/runner-manager.js +90 -29
  43. package/dist/runner-manager.js.map +1 -1
  44. package/dist/runner.d.ts +2 -0
  45. package/dist/runner.js +2 -0
  46. package/dist/runner.js.map +1 -1
  47. package/dist/shared/grpc-retry.d.ts +9 -20
  48. package/dist/shared/grpc-retry.js +9 -52
  49. package/dist/shared/grpc-retry.js.map +1 -1
  50. package/dist/shared/stall-watchdog.d.ts +68 -0
  51. package/dist/shared/stall-watchdog.js +102 -0
  52. package/dist/shared/stall-watchdog.js.map +1 -0
  53. package/dist/shared/status-offload.d.ts +84 -0
  54. package/dist/shared/status-offload.js +292 -0
  55. package/dist/shared/status-offload.js.map +1 -0
  56. package/dist/shared/status.d.ts +34 -3
  57. package/dist/shared/status.js +102 -9
  58. package/dist/shared/status.js.map +1 -1
  59. package/dist/{activities/execute-deep-agent → shared}/streaming-scheduler.d.ts +4 -0
  60. package/dist/{activities/execute-deep-agent → shared}/streaming-scheduler.js +4 -0
  61. package/dist/shared/streaming-scheduler.js.map +1 -0
  62. package/package.json +2 -2
  63. package/src/__tests__/config.test.ts +8 -0
  64. package/src/__tests__/in-flight.test.ts +84 -0
  65. package/src/activities/__tests__/classify-tool-approvals.test.ts +1 -0
  66. package/src/activities/__tests__/discover-mcp-server.test.ts +1 -0
  67. package/src/activities/execute-cursor/__tests__/build-prompt.test.ts +74 -0
  68. package/src/activities/execute-cursor/__tests__/hook-script.test.ts +90 -15
  69. package/src/activities/execute-cursor/__tests__/message-translator.test.ts +124 -12
  70. package/src/activities/execute-cursor/__tests__/persist-decision.test.ts +99 -0
  71. package/src/activities/execute-cursor/__tests__/tool-result-image.test.ts +244 -0
  72. package/src/activities/execute-cursor/__tests__/workspace-setup.test.ts +53 -4
  73. package/src/activities/execute-cursor/hook-script.ts +85 -51
  74. package/src/activities/execute-cursor/index.ts +187 -38
  75. package/src/activities/execute-cursor/message-translator.ts +146 -20
  76. package/src/activities/execute-cursor/persist-decision.ts +54 -0
  77. package/src/activities/execute-cursor/prompt-builder.ts +59 -0
  78. package/src/activities/execute-cursor/workspace-setup.ts +76 -44
  79. package/src/activities/execute-deep-agent/__tests__/index.test.ts +1 -0
  80. package/src/activities/execute-deep-agent/__tests__/status-builder-shared.test.ts +66 -0
  81. package/src/activities/execute-deep-agent/__tests__/status-builder.test.ts +6 -3
  82. package/src/activities/execute-deep-agent/__tests__/streaming-v3.test.ts +70 -0
  83. package/src/activities/execute-deep-agent/index.ts +17 -5
  84. package/src/activities/execute-deep-agent/status-builder-shared.ts +27 -5
  85. package/src/activities/execute-deep-agent/status-builder.ts +3 -5
  86. package/src/activities/execute-deep-agent/streaming-v3.ts +5 -5
  87. package/src/activities/execute-deep-agent/streaming.ts +14 -5
  88. package/src/activities/execute-deep-agent/subagent-tracker.ts +4 -5
  89. package/src/activities/execute-deep-agent/v3-status-builder.ts +5 -5
  90. package/src/config.ts +27 -0
  91. package/src/in-flight.ts +71 -0
  92. package/src/runner-manager.ts +127 -33
  93. package/src/runner.ts +6 -0
  94. package/src/shared/__tests__/artifact-storage.test.ts +1 -0
  95. package/src/shared/__tests__/grpc-retry-extended.test.ts +6 -144
  96. package/src/shared/__tests__/grpc-retry.test.ts +5 -134
  97. package/src/shared/__tests__/stall-watchdog.test.ts +193 -0
  98. package/src/shared/__tests__/status-offload.test.ts +256 -0
  99. package/src/shared/__tests__/status.test.ts +199 -0
  100. package/src/shared/grpc-retry.ts +9 -72
  101. package/src/shared/stall-watchdog.ts +122 -0
  102. package/src/shared/status-offload.ts +342 -0
  103. package/src/shared/status.ts +142 -8
  104. package/src/{activities/execute-deep-agent → shared}/streaming-scheduler.ts +4 -0
  105. package/dist/activities/execute-deep-agent/streaming-scheduler.js.map +0 -1
  106. /package/src/{activities/execute-deep-agent → shared}/__tests__/streaming-scheduler.test.ts +0 -0
@@ -1,20 +1,15 @@
1
1
  /**
2
- * Exponential-backoff retry wrapper for gRPC status persistence.
2
+ * gRPC error classification for status persistence retries.
3
3
  *
4
- * Wraps the raw StigmerClient.updateStatus() call with retry logic
5
- * that classifies gRPC error codes as retryable vs terminal. A failed
6
- * status update must never crash the streaming loop — on permanent
7
- * failure, the signal falls back to UNSPECIFIED.
8
- *
9
- * Used by the ExecuteDeepAgent streaming loop. The simpler
10
- * fire-and-forget persistStatus in shared/status.ts is preserved
11
- * for ExecuteCursor and non-streaming callers.
4
+ * Status persistence flows through a single chokepoint — `persistStatus` in
5
+ * status.ts — which uses these helpers to decide whether a failed
6
+ * `updateStatus` should back off and retry (transient transport errors) or
7
+ * fail fast (deterministic errors). This module is intentionally just the
8
+ * classification policy + its options type, kept small and separately tested;
9
+ * the persist loop that consumes it lives with the rest of the persist logic.
12
10
  */
13
11
 
14
12
  import { ConnectError, Code } from "@connectrpc/connect";
15
- import { ExecutionControlSignal } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/enum_pb";
16
- import type { AgentExecutionStatus } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/api_pb";
17
- import type { StigmerClient } from "../client/stigmer-client.js";
18
13
 
19
14
  export interface RetryOptions {
20
15
  /** Base delay before the first retry (ms). Default: 100. */
@@ -38,6 +33,7 @@ const TERMINAL_CODES = new Set<Code>([
38
33
  Code.PermissionDenied,
39
34
  ]);
40
35
 
36
+ /** True for transient transport errors that are worth retrying with backoff. */
41
37
  export function isRetryableError(err: unknown): boolean {
42
38
  if (err instanceof ConnectError) {
43
39
  return RETRYABLE_CODES.has(err.code);
@@ -45,69 +41,10 @@ export function isRetryableError(err: unknown): boolean {
45
41
  return false;
46
42
  }
47
43
 
44
+ /** True for deterministic errors that retrying cannot fix. */
48
45
  export function isTerminalError(err: unknown): boolean {
49
46
  if (err instanceof ConnectError) {
50
47
  return TERMINAL_CODES.has(err.code);
51
48
  }
52
49
  return false;
53
50
  }
54
-
55
- function defaultDelay(ms: number): Promise<void> {
56
- return new Promise(resolve => setTimeout(resolve, ms));
57
- }
58
-
59
- /**
60
- * Persist execution status with exponential-backoff retry.
61
- *
62
- * Returns the ExecutionControlSignal from the server on success.
63
- * On permanent or exhausted-retry failure, logs the error and
64
- * returns UNSPECIFIED (never throws).
65
- */
66
- export async function persistWithRetry(
67
- client: StigmerClient,
68
- executionId: string,
69
- status: AgentExecutionStatus,
70
- options?: RetryOptions,
71
- ): Promise<ExecutionControlSignal> {
72
- const baseDelay = options?.baseDelayMs ?? 100;
73
- const factor = options?.backoffFactor ?? 2;
74
- const maxRetries = options?.maxRetries ?? 3;
75
- const delay = options?.delayFn ?? defaultDelay;
76
-
77
- let lastError: unknown;
78
-
79
- for (let attempt = 0; attempt <= maxRetries; attempt++) {
80
- try {
81
- const response = await client.updateStatus(executionId, status);
82
- return response.signal;
83
- } catch (err: unknown) {
84
- lastError = err;
85
-
86
- if (isTerminalError(err)) {
87
- const code = (err as ConnectError).code;
88
- console.error(
89
- `[grpc-retry] Terminal error persisting status for ${executionId}: ` +
90
- `code=${Code[code]} (attempt ${attempt + 1}/${maxRetries + 1})`,
91
- );
92
- return ExecutionControlSignal.UNSPECIFIED;
93
- }
94
-
95
- if (!isRetryableError(err) || attempt === maxRetries) {
96
- break;
97
- }
98
-
99
- const delayMs = baseDelay * Math.pow(factor, attempt);
100
- console.warn(
101
- `[grpc-retry] Retryable error for ${executionId}: ` +
102
- `code=${err instanceof ConnectError ? Code[err.code] : "unknown"} ` +
103
- `(attempt ${attempt + 1}/${maxRetries + 1}, retry in ${delayMs}ms)`,
104
- );
105
- await delay(delayMs);
106
- }
107
- }
108
-
109
- console.error(
110
- `[grpc-retry] All retries exhausted for ${executionId}: ${lastError}`,
111
- );
112
- return ExecutionControlSignal.UNSPECIFIED;
113
- }
@@ -0,0 +1,122 @@
1
+ /**
2
+ * Progress-based stall watchdog.
3
+ *
4
+ * Long-running agent turns are kept alive against Temporal by a periodic
5
+ * keep-alive heartbeat (see {@link ./heartbeat.ts}). That heartbeat proves the
6
+ * runner *process* is alive, but it says nothing about whether the agent is
7
+ * making *progress*: a turn that wedges inside the harness stream loop (a tool
8
+ * call that never returns, a model connection that silently dies) keeps
9
+ * heartbeating forever and never times out. The execution then sits at
10
+ * EXECUTION_IN_PROGRESS indefinitely.
11
+ *
12
+ * This watchdog closes that gap. It is an *out-of-band* timer: callers report
13
+ * progress via {@link StallWatchdog.recordActivity} on every stream event AND
14
+ * every token delta, and the watchdog fires `onStall` once if no progress is
15
+ * reported for `stallMs`. "Out-of-band" is the deliberate correctness property
16
+ * — the check runs on its own timer rather than inside the consumer's loop, so
17
+ * it catches a true "no new events for N minutes" hang. An in-band check that
18
+ * only runs when the loop advances cannot fire while the loop is blocked
19
+ * awaiting the next event, which is exactly the wedge we need to detect.
20
+ *
21
+ * This module is the convergence target for stall detection across harnesses:
22
+ * the Cursor harness wires it here, and the native deep-agent harness has its
23
+ * own (harness-local, in-band) check that should later converge onto this one.
24
+ */
25
+
26
+ /**
27
+ * Canonical default stall window. Harnesses may override with a larger value
28
+ * when their tool calls routinely run longer without emitting stream activity.
29
+ */
30
+ export const DEFAULT_STALL_TIMEOUT_MS = 120_000;
31
+
32
+ /**
33
+ * Thrown / reported when an agent stream makes no progress for longer than the
34
+ * configured stall window. Carries the observed idle duration so callers can
35
+ * build an actionable, recognizable error message.
36
+ */
37
+ export class StallTimeoutError extends Error {
38
+ constructor(
39
+ public readonly stalledMs: number,
40
+ detail?: string,
41
+ ) {
42
+ super(
43
+ `Agent stream stalled: no activity for ${Math.round(stalledMs / 1000)}s` +
44
+ (detail ? ` (${detail})` : ""),
45
+ );
46
+ this.name = "StallTimeoutError";
47
+ }
48
+ }
49
+
50
+ /**
51
+ * Recognizable prefix on every stall-induced failure message. Single source of
52
+ * truth so callers (and any future UI/log keying) match on one constant.
53
+ */
54
+ export const STALL_ERROR_PREFIX = "[StallTimeoutError]";
55
+
56
+ /**
57
+ * Build the user-facing failure text for a stall. Keeps the canonical wording
58
+ * (prefix + actionable "Retry or resume.") next to the error it describes.
59
+ */
60
+ export function formatStallFailure(error: StallTimeoutError): string {
61
+ return `${STALL_ERROR_PREFIX} ${error.message}. Retry or resume.`;
62
+ }
63
+
64
+ export interface StallWatchdog {
65
+ /** Reset the idle timer. Call on every stream event AND every token delta. */
66
+ recordActivity(): void;
67
+ /** Disarm the watchdog. Idempotent; safe to call in a `finally`. */
68
+ stop(): void;
69
+ }
70
+
71
+ /**
72
+ * Start an out-of-band stall watchdog.
73
+ *
74
+ * @param stallMs Idle window after which `onStall` fires.
75
+ * @param onStall Invoked at most once with the observed idle duration (ms)
76
+ * when `Date.now() - lastActivityAt` exceeds `stallMs`. The
77
+ * watchdog disarms itself before invoking, so `onStall` runs
78
+ * exactly once even if its handler is slow.
79
+ *
80
+ * The poll interval is `stallMs / 4` capped at 15s: frequent enough to detect a
81
+ * stall promptly without busy-looping, and bounded so a large `stallMs` still
82
+ * polls on a sane cadence.
83
+ */
84
+ export function startStallWatchdog(
85
+ stallMs: number,
86
+ onStall: (idleMs: number) => void,
87
+ ): StallWatchdog {
88
+ let lastActivityAt = Date.now();
89
+ let fired = false;
90
+ let timer: ReturnType<typeof setInterval> | undefined;
91
+
92
+ const tickMs = Math.min(Math.max(Math.floor(stallMs / 4), 1), 15_000);
93
+
94
+ const disarm = (): void => {
95
+ if (timer !== undefined) {
96
+ clearInterval(timer);
97
+ timer = undefined;
98
+ }
99
+ };
100
+
101
+ timer = setInterval(() => {
102
+ if (fired) return;
103
+ const idleMs = Date.now() - lastActivityAt;
104
+ if (idleMs >= stallMs) {
105
+ fired = true;
106
+ disarm();
107
+ onStall(idleMs);
108
+ }
109
+ }, tickMs);
110
+ // Do not keep the event loop alive solely for the watchdog.
111
+ timer.unref?.();
112
+
113
+ return {
114
+ recordActivity(): void {
115
+ lastActivityAt = Date.now();
116
+ },
117
+ stop(): void {
118
+ fired = true;
119
+ disarm();
120
+ },
121
+ };
122
+ }
@@ -0,0 +1,342 @@
1
+ /**
2
+ * Size-bounding guard for the persisted AgentExecutionStatus payload.
3
+ *
4
+ * Tool outputs (an MCP screenshot's base64 image, a giant accessibility-tree
5
+ * dump, a multi-MB shell log, a huge file write) are stored inline in
6
+ * `ToolCall.result`/`args_preview`, which live in `status.messages` and are
7
+ * re-serialized whole on every `persistStatus` -> `updateStatus` gRPC call.
8
+ * Left unchecked, a single large result pushes the message past the server's
9
+ * 4 MiB gRPC receive cap; the call fails with `resource_exhausted`, progress
10
+ * stops persisting, and the live UI freezes mid-execution.
11
+ *
12
+ * This module enforces, at the single persist boundary, that the payload stays
13
+ * under the limit:
14
+ *
15
+ * 1. offloadOversizedToolOutputs — per-tool-call, SIZE-DRIVEN (not keyed on
16
+ * tool type): any result over the byte threshold is spilled to artifact
17
+ * storage and replaced with a short head plus a typed ToolCallOutputRef.
18
+ * Images are uploaded as their decoded bytes (so the UI can render an
19
+ * <img>); other large output is uploaded as text with a preview head.
20
+ * Idempotent and content-hash-deduped so the throttled, repeated persists
21
+ * (and result re-inflation by mergeToolCallEvent) upload each blob once.
22
+ *
23
+ * 2. enforceStatusSizeLimit — an aggregate, type-agnostic backstop that runs
24
+ * even when no artifact storage is available: if the encoded status still
25
+ * exceeds a soft cap (comfortably under 4 MiB), it elides the largest
26
+ * remaining inline fields in place until the payload fits.
27
+ *
28
+ * Both operate ONLY on what is persisted/streamed; the agent's working context
29
+ * is managed by the harness/SDK separately, so reasoning is unaffected.
30
+ */
31
+
32
+ import { createHash } from "node:crypto";
33
+ import { create, toBinary } from "@bufbuild/protobuf";
34
+ import {
35
+ AgentExecutionStatusSchema,
36
+ } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/api_pb";
37
+ import type { AgentExecutionStatus } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/api_pb";
38
+ import { ToolCallOutputRefSchema } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/message_pb";
39
+ import type { ToolCall } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/message_pb";
40
+ import type { ArtifactStorage } from "./artifact-storage.js";
41
+
42
+ /**
43
+ * A single tool output (result or args_preview) larger than this many bytes is
44
+ * offloaded to artifact storage instead of inlined into the persisted status.
45
+ * 256 KiB is generous for ordinary tool output yet far below the 4 MiB cap, so
46
+ * even a handful of large-but-sub-threshold results cannot aggregate past it.
47
+ */
48
+ export const INLINE_TOOL_OUTPUT_MAX_BYTES = 256 * 1024;
49
+
50
+ /** Head of an offloaded text result kept inline for an at-a-glance preview. */
51
+ export const TEXT_PREVIEW_HEAD_CHARS = 4_000;
52
+
53
+ /**
54
+ * Soft cap for the whole encoded status. Kept comfortably under the 4 MiB gRPC
55
+ * receive limit so the aggregate backstop trims before the server ever rejects.
56
+ */
57
+ export const STATUS_PAYLOAD_SOFT_LIMIT_BYTES = 3 * 1024 * 1024;
58
+
59
+ /**
60
+ * Tighter cap used only after the server has already rejected a payload as too
61
+ * large, to maximize the chance the retry succeeds.
62
+ */
63
+ export const STATUS_PAYLOAD_HARD_LIMIT_BYTES = 2 * 1024 * 1024;
64
+
65
+ /** Marker left in place of an aggregate-elided inline field. */
66
+ const ELISION_MARKER = "[output elided to keep status under the size limit]";
67
+
68
+ /** Below this size an inline field is not worth eliding (the marker is ~50B). */
69
+ const ELISION_MIN_BYTES = 1_024;
70
+
71
+ export interface ToolOutputOffloadContext {
72
+ readonly artifactStorage: ArtifactStorage;
73
+ readonly executionId: string;
74
+ /** Override the per-result byte threshold (tests use a small value). */
75
+ readonly maxInlineBytes?: number;
76
+ }
77
+
78
+ interface ImagePayload {
79
+ readonly base64: string;
80
+ readonly mimeType: string;
81
+ }
82
+
83
+ function byteLen(s: string | undefined): number {
84
+ return s ? Buffer.byteLength(s, "utf8") : 0;
85
+ }
86
+
87
+ function sha256(s: string): string {
88
+ return createHash("sha256").update(s, "utf8").digest("hex");
89
+ }
90
+
91
+ function headChars(s: string, n: number): string {
92
+ return s.length <= n ? s : s.slice(0, n);
93
+ }
94
+
95
+ function formatBytes(n: number): string {
96
+ if (n >= 1024 * 1024) return `${(n / (1024 * 1024)).toFixed(1)} MB`;
97
+ if (n >= 1024) return `${(n / 1024).toFixed(1)} KB`;
98
+ return `${n} B`;
99
+ }
100
+
101
+ function extFromMime(mimeType: string): string {
102
+ switch (mimeType) {
103
+ case "image/png": return "png";
104
+ case "image/jpeg": return "jpg";
105
+ case "image/gif": return "gif";
106
+ case "image/webp": return "webp";
107
+ case "image/svg+xml": return "svg";
108
+ default: return "bin";
109
+ }
110
+ }
111
+
112
+ function matchDataUrl(s: string): ImagePayload | null {
113
+ const m = s.match(/^data:(image\/[a-zA-Z0-9.+-]+);base64,([\s\S]+)$/);
114
+ if (!m) return null;
115
+ return { mimeType: m[1], base64: m[2].replace(/\s+/g, "") };
116
+ }
117
+
118
+ function contentBlocks(parsed: unknown): unknown[] {
119
+ if (Array.isArray(parsed)) return parsed;
120
+ if (parsed && typeof parsed === "object") {
121
+ const obj = parsed as Record<string, unknown>;
122
+ if (Array.isArray(obj.content)) return obj.content;
123
+ // Defensive: a serialized LangChain ToolMessage envelope nests its blocks
124
+ // under kwargs.content. The deep-agent extractor already normalizes image
125
+ // results to a top-level array (status-builder-shared.ts), so this branch is
126
+ // insurance against future shape drift — not the primary path.
127
+ const kwargs = obj.kwargs;
128
+ if (kwargs && typeof kwargs === "object" && Array.isArray((kwargs as Record<string, unknown>).content)) {
129
+ return (kwargs as Record<string, unknown>).content as unknown[];
130
+ }
131
+ }
132
+ return [];
133
+ }
134
+
135
+ /**
136
+ * Best-effort extraction of an inline base64 image from a tool result string.
137
+ * Handles a raw data URL, MCP image content blocks ({type:"image", data,
138
+ * mimeType}) and OpenAI-style image_url blocks. Returns null for non-image or
139
+ * unparseable results so the caller falls back to text offload.
140
+ */
141
+ export function detectImagePayload(result: string): ImagePayload | null {
142
+ const direct = matchDataUrl(result.trim());
143
+ if (direct) return direct;
144
+
145
+ let parsed: unknown;
146
+ try {
147
+ parsed = JSON.parse(result);
148
+ } catch {
149
+ return null;
150
+ }
151
+
152
+ for (const block of contentBlocks(parsed)) {
153
+ if (!block || typeof block !== "object") continue;
154
+ const obj = block as Record<string, unknown>;
155
+ const type = typeof obj.type === "string" ? obj.type : "";
156
+ if (type !== "image" && type !== "image_url") continue;
157
+
158
+ const raw =
159
+ typeof obj.data === "string" ? obj.data :
160
+ typeof obj.image === "string" ? obj.image :
161
+ undefined;
162
+ if (raw) {
163
+ const asUrl = matchDataUrl(raw);
164
+ if (asUrl) return asUrl;
165
+ const mimeType =
166
+ typeof obj.mimeType === "string" ? obj.mimeType :
167
+ typeof obj.mime_type === "string" ? obj.mime_type :
168
+ "image/png";
169
+ return { mimeType, base64: raw.replace(/\s+/g, "") };
170
+ }
171
+
172
+ const imageUrl = obj.image_url;
173
+ if (imageUrl && typeof imageUrl === "object") {
174
+ const url = (imageUrl as Record<string, unknown>).url;
175
+ if (typeof url === "string") {
176
+ const asUrl = matchDataUrl(url);
177
+ if (asUrl) return asUrl;
178
+ }
179
+ }
180
+ }
181
+ return null;
182
+ }
183
+
184
+ function collapsedResultFor(ref: { isImage: boolean; sizeBytes: bigint; truncatedPreview: string }): string {
185
+ if (ref.isImage) {
186
+ return `[image output — ${formatBytes(Number(ref.sizeBytes))}, view inline]`;
187
+ }
188
+ const tail = `\n\n[output truncated — ${formatBytes(Number(ref.sizeBytes))} total; view full output]`;
189
+ return ref.truncatedPreview + tail;
190
+ }
191
+
192
+ async function maybeOffloadToolCall(
193
+ tc: ToolCall,
194
+ ctx: ToolOutputOffloadContext,
195
+ maxBytes: number,
196
+ ): Promise<void> {
197
+ const result = tc.result;
198
+ if (!result) return;
199
+
200
+ const hash = sha256(result);
201
+
202
+ // Idempotent: the same content was already offloaded on a prior persist (or
203
+ // mergeToolCallEvent re-inflated `result` with identical bytes). Re-collapse
204
+ // the inline copy without re-uploading.
205
+ if (tc.outputRef && tc.outputRef.contentHash === hash) {
206
+ tc.result = collapsedResultFor(tc.outputRef);
207
+ return;
208
+ }
209
+
210
+ // Images are offloaded regardless of size, BEFORE the size gate below: a
211
+ // `ToolCallOutputRef` is the only path the UI has to render an image inline,
212
+ // so even a sub-threshold screenshot must be lifted out of `result` into a
213
+ // renderable ref. Non-image text only offloads when it exceeds maxBytes.
214
+ const image = detectImagePayload(result);
215
+ if (image) {
216
+ const bytes = Buffer.from(image.base64, "base64");
217
+ const key = `artifacts/${ctx.executionId}/toolcalls/${tc.id}.${extFromMime(image.mimeType)}`;
218
+ await ctx.artifactStorage.upload(key, bytes, image.mimeType);
219
+ const downloadUrl = await ctx.artifactStorage.getDownloadUrl(key);
220
+ tc.outputRef = create(ToolCallOutputRefSchema, {
221
+ storageKey: key,
222
+ downloadUrl,
223
+ sizeBytes: BigInt(bytes.length),
224
+ contentHash: hash,
225
+ mimeType: image.mimeType,
226
+ isImage: true,
227
+ truncatedPreview: "",
228
+ });
229
+ tc.result = collapsedResultFor(tc.outputRef);
230
+ return;
231
+ }
232
+
233
+ // Non-image text below the inline budget stays inline; only oversized text is
234
+ // spilled (with a head preview) to keep the persisted payload bounded.
235
+ if (byteLen(result) <= maxBytes) return;
236
+
237
+ const content = Buffer.from(result, "utf8");
238
+ const key = `artifacts/${ctx.executionId}/toolcalls/${tc.id}.txt`;
239
+ await ctx.artifactStorage.upload(key, content, "text/plain");
240
+ const downloadUrl = await ctx.artifactStorage.getDownloadUrl(key);
241
+ tc.outputRef = create(ToolCallOutputRefSchema, {
242
+ storageKey: key,
243
+ downloadUrl,
244
+ sizeBytes: BigInt(content.length),
245
+ contentHash: hash,
246
+ mimeType: "text/plain",
247
+ isImage: false,
248
+ truncatedPreview: headChars(result, TEXT_PREVIEW_HEAD_CHARS),
249
+ });
250
+ tc.result = collapsedResultFor(tc.outputRef);
251
+ }
252
+
253
+ /**
254
+ * Offload every oversized tool result in the status to artifact storage,
255
+ * replacing the inline value with a short head + ToolCallOutputRef. Per-tool
256
+ * failures fall back to an inline truncation (a bounded result beats a failed
257
+ * persist) and never throw, so a storage hiccup cannot fail the execution.
258
+ */
259
+ export async function offloadOversizedToolOutputs(
260
+ status: AgentExecutionStatus,
261
+ ctx: ToolOutputOffloadContext,
262
+ ): Promise<void> {
263
+ const maxBytes = ctx.maxInlineBytes ?? INLINE_TOOL_OUTPUT_MAX_BYTES;
264
+ for (const msg of status.messages) {
265
+ for (const tc of msg.toolCalls) {
266
+ try {
267
+ await maybeOffloadToolCall(tc, ctx, maxBytes);
268
+ } catch (err) {
269
+ const original = tc.result ?? "";
270
+ tc.result =
271
+ headChars(original, TEXT_PREVIEW_HEAD_CHARS) +
272
+ `\n\n[output truncated — offload failed: ${err instanceof Error ? err.message : String(err)}]`;
273
+ console.warn(
274
+ `[status-offload] execution=${ctx.executionId} tool=${tc.name} ` +
275
+ `offload failed (non-fatal); truncated inline`,
276
+ );
277
+ }
278
+ }
279
+ }
280
+ }
281
+
282
+ function encodedSize(status: AgentExecutionStatus): number {
283
+ return toBinary(AgentExecutionStatusSchema, status).length;
284
+ }
285
+
286
+ /**
287
+ * Aggregate, type-agnostic backstop. If the encoded status exceeds
288
+ * `softLimitBytes`, elide the largest inline tool fields (and, as a last
289
+ * resort, message content) in place until it fits. Returns true if anything
290
+ * was elided. This guarantees a bounded payload even without artifact storage
291
+ * (e.g. offload disabled) or when many medium results sum past the limit.
292
+ */
293
+ export function enforceStatusSizeLimit(
294
+ status: AgentExecutionStatus,
295
+ softLimitBytes: number = STATUS_PAYLOAD_SOFT_LIMIT_BYTES,
296
+ ): boolean {
297
+ if (encodedSize(status) <= softLimitBytes) return false;
298
+
299
+ const toolCalls: ToolCall[] = [];
300
+ for (const msg of status.messages) {
301
+ for (const tc of msg.toolCalls) toolCalls.push(tc);
302
+ }
303
+ // Largest inline footprint first so we shed the most bytes per elision.
304
+ toolCalls.sort(
305
+ (a, b) =>
306
+ byteLen(b.result) + byteLen(b.argsPreview) -
307
+ (byteLen(a.result) + byteLen(a.argsPreview)),
308
+ );
309
+
310
+ let elidedAny = false;
311
+ for (const tc of toolCalls) {
312
+ if (encodedSize(status) <= softLimitBytes) return elidedAny;
313
+ if (byteLen(tc.result) > ELISION_MIN_BYTES) {
314
+ tc.result = ELISION_MARKER;
315
+ elidedAny = true;
316
+ }
317
+ if (byteLen(tc.argsPreview) > ELISION_MIN_BYTES) {
318
+ tc.argsPreview = ELISION_MARKER;
319
+ elidedAny = true;
320
+ }
321
+ if (tc.args !== undefined) {
322
+ tc.args = undefined;
323
+ elidedAny = true;
324
+ }
325
+ }
326
+
327
+ // Last resort: oversized message content (e.g. a huge AI response).
328
+ if (encodedSize(status) > softLimitBytes) {
329
+ const byContent = [...status.messages].sort(
330
+ (a, b) => byteLen(b.content) - byteLen(a.content),
331
+ );
332
+ for (const msg of byContent) {
333
+ if (encodedSize(status) <= softLimitBytes) break;
334
+ if (byteLen(msg.content) > ELISION_MIN_BYTES) {
335
+ msg.content = headChars(msg.content, TEXT_PREVIEW_HEAD_CHARS) + `\n\n${ELISION_MARKER}`;
336
+ elidedAny = true;
337
+ }
338
+ }
339
+ }
340
+
341
+ return elidedAny;
342
+ }