@stigmer/runner 3.0.8-dev.20260613085218 → 3.0.9-dev.20260615145121

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/dist/.build-fingerprint +1 -1
  2. package/dist/activities/execute-cursor/hook-script.d.ts +23 -12
  3. package/dist/activities/execute-cursor/hook-script.js +85 -51
  4. package/dist/activities/execute-cursor/hook-script.js.map +1 -1
  5. package/dist/activities/execute-cursor/index.js +220 -79
  6. package/dist/activities/execute-cursor/index.js.map +1 -1
  7. package/dist/activities/execute-cursor/message-translator.d.ts +51 -9
  8. package/dist/activities/execute-cursor/message-translator.js +146 -20
  9. package/dist/activities/execute-cursor/message-translator.js.map +1 -1
  10. package/dist/activities/execute-cursor/persist-decision.d.ts +42 -0
  11. package/dist/activities/execute-cursor/persist-decision.js +30 -0
  12. package/dist/activities/execute-cursor/persist-decision.js.map +1 -0
  13. package/dist/activities/execute-cursor/prompt-builder.d.ts +25 -0
  14. package/dist/activities/execute-cursor/prompt-builder.js +54 -0
  15. package/dist/activities/execute-cursor/prompt-builder.js.map +1 -1
  16. package/dist/activities/execute-cursor/workspace-setup.d.ts +8 -2
  17. package/dist/activities/execute-cursor/workspace-setup.js +62 -30
  18. package/dist/activities/execute-cursor/workspace-setup.js.map +1 -1
  19. package/dist/activities/execute-deep-agent/index.js +15 -5
  20. package/dist/activities/execute-deep-agent/index.js.map +1 -1
  21. package/dist/activities/execute-deep-agent/status-builder-shared.d.ts +0 -1
  22. package/dist/activities/execute-deep-agent/status-builder-shared.js +32 -8
  23. package/dist/activities/execute-deep-agent/status-builder-shared.js.map +1 -1
  24. package/dist/activities/execute-deep-agent/status-builder.js +4 -5
  25. package/dist/activities/execute-deep-agent/status-builder.js.map +1 -1
  26. package/dist/activities/execute-deep-agent/streaming-v3.js +4 -5
  27. package/dist/activities/execute-deep-agent/streaming-v3.js.map +1 -1
  28. package/dist/activities/execute-deep-agent/streaming.d.ts +9 -1
  29. package/dist/activities/execute-deep-agent/streaming.js +4 -5
  30. package/dist/activities/execute-deep-agent/streaming.js.map +1 -1
  31. package/dist/activities/execute-deep-agent/subagent-tracker.js +4 -5
  32. package/dist/activities/execute-deep-agent/subagent-tracker.js.map +1 -1
  33. package/dist/activities/execute-deep-agent/v3-status-builder.js +6 -5
  34. package/dist/activities/execute-deep-agent/v3-status-builder.js.map +1 -1
  35. package/dist/config.d.ts +21 -0
  36. package/dist/config.js +12 -0
  37. package/dist/config.js.map +1 -1
  38. package/dist/in-flight.d.ts +35 -0
  39. package/dist/in-flight.js +61 -0
  40. package/dist/in-flight.js.map +1 -0
  41. package/dist/runner-manager.d.ts +2 -0
  42. package/dist/runner-manager.js +90 -29
  43. package/dist/runner-manager.js.map +1 -1
  44. package/dist/runner.d.ts +2 -0
  45. package/dist/runner.js +2 -0
  46. package/dist/runner.js.map +1 -1
  47. package/dist/shared/grpc-retry.d.ts +9 -20
  48. package/dist/shared/grpc-retry.js +9 -52
  49. package/dist/shared/grpc-retry.js.map +1 -1
  50. package/dist/shared/stall-watchdog.d.ts +68 -0
  51. package/dist/shared/stall-watchdog.js +102 -0
  52. package/dist/shared/stall-watchdog.js.map +1 -0
  53. package/dist/shared/status-offload.d.ts +84 -0
  54. package/dist/shared/status-offload.js +292 -0
  55. package/dist/shared/status-offload.js.map +1 -0
  56. package/dist/shared/status.d.ts +34 -3
  57. package/dist/shared/status.js +102 -9
  58. package/dist/shared/status.js.map +1 -1
  59. package/dist/{activities/execute-deep-agent → shared}/streaming-scheduler.d.ts +4 -0
  60. package/dist/{activities/execute-deep-agent → shared}/streaming-scheduler.js +4 -0
  61. package/dist/shared/streaming-scheduler.js.map +1 -0
  62. package/package.json +2 -2
  63. package/src/__tests__/config.test.ts +8 -0
  64. package/src/__tests__/in-flight.test.ts +84 -0
  65. package/src/activities/__tests__/classify-tool-approvals.test.ts +1 -0
  66. package/src/activities/__tests__/discover-mcp-server.test.ts +1 -0
  67. package/src/activities/execute-cursor/__tests__/build-prompt.test.ts +74 -0
  68. package/src/activities/execute-cursor/__tests__/hook-script.test.ts +90 -15
  69. package/src/activities/execute-cursor/__tests__/message-translator.test.ts +124 -12
  70. package/src/activities/execute-cursor/__tests__/persist-decision.test.ts +99 -0
  71. package/src/activities/execute-cursor/__tests__/tool-result-image.test.ts +244 -0
  72. package/src/activities/execute-cursor/__tests__/workspace-setup.test.ts +53 -4
  73. package/src/activities/execute-cursor/hook-script.ts +85 -51
  74. package/src/activities/execute-cursor/index.ts +187 -38
  75. package/src/activities/execute-cursor/message-translator.ts +146 -20
  76. package/src/activities/execute-cursor/persist-decision.ts +54 -0
  77. package/src/activities/execute-cursor/prompt-builder.ts +59 -0
  78. package/src/activities/execute-cursor/workspace-setup.ts +76 -44
  79. package/src/activities/execute-deep-agent/__tests__/index.test.ts +1 -0
  80. package/src/activities/execute-deep-agent/__tests__/status-builder-shared.test.ts +66 -0
  81. package/src/activities/execute-deep-agent/__tests__/status-builder.test.ts +6 -3
  82. package/src/activities/execute-deep-agent/__tests__/streaming-v3.test.ts +70 -0
  83. package/src/activities/execute-deep-agent/index.ts +17 -5
  84. package/src/activities/execute-deep-agent/status-builder-shared.ts +27 -5
  85. package/src/activities/execute-deep-agent/status-builder.ts +3 -5
  86. package/src/activities/execute-deep-agent/streaming-v3.ts +5 -5
  87. package/src/activities/execute-deep-agent/streaming.ts +14 -5
  88. package/src/activities/execute-deep-agent/subagent-tracker.ts +4 -5
  89. package/src/activities/execute-deep-agent/v3-status-builder.ts +5 -5
  90. package/src/config.ts +27 -0
  91. package/src/in-flight.ts +71 -0
  92. package/src/runner-manager.ts +127 -33
  93. package/src/runner.ts +6 -0
  94. package/src/shared/__tests__/artifact-storage.test.ts +1 -0
  95. package/src/shared/__tests__/grpc-retry-extended.test.ts +6 -144
  96. package/src/shared/__tests__/grpc-retry.test.ts +5 -134
  97. package/src/shared/__tests__/stall-watchdog.test.ts +193 -0
  98. package/src/shared/__tests__/status-offload.test.ts +256 -0
  99. package/src/shared/__tests__/status.test.ts +199 -0
  100. package/src/shared/grpc-retry.ts +9 -72
  101. package/src/shared/stall-watchdog.ts +122 -0
  102. package/src/shared/status-offload.ts +342 -0
  103. package/src/shared/status.ts +142 -8
  104. package/src/{activities/execute-deep-agent → shared}/streaming-scheduler.ts +4 -0
  105. package/dist/activities/execute-deep-agent/streaming-scheduler.js.map +0 -1
  106. /package/src/{activities/execute-deep-agent → shared}/__tests__/streaming-scheduler.test.ts +0 -0
@@ -58,6 +58,18 @@ function assistantEvent(
58
58
  };
59
59
  }
60
60
 
61
+ function thinkingEvent(
62
+ runId: string,
63
+ text: string,
64
+ ): Extract<SDKMessage, { type: "thinking" }> {
65
+ return {
66
+ type: "thinking",
67
+ agent_id: "agent-1",
68
+ run_id: runId,
69
+ text,
70
+ };
71
+ }
72
+
61
73
  function countToolCallsWithId(messages: AgentMessage[], callId: string): number {
62
74
  let count = 0;
63
75
  for (const msg of messages) {
@@ -237,9 +249,9 @@ describe("MessageAccumulator tool call status transitions", () => {
237
249
  expect(acc.subAgentExecutions[0].status).toBe(SubAgentStatus.SUB_AGENT_IN_PROGRESS);
238
250
  });
239
251
 
240
- it("subAgentDirty starts false and is set when a sub-agent is created", () => {
252
+ it("isDirty starts false and is set when a sub-agent is created", () => {
241
253
  const acc = new MessageAccumulator([]);
242
- expect(acc.subAgentDirty).toBe(false);
254
+ expect(acc.isDirty).toBe(false);
243
255
 
244
256
  acc.trackSubAgentExecution(
245
257
  toolCallEvent("tc-dirty1", "task", "running", "r1", {
@@ -247,10 +259,10 @@ describe("MessageAccumulator tool call status transitions", () => {
247
259
  }),
248
260
  );
249
261
 
250
- expect(acc.subAgentDirty).toBe(true);
262
+ expect(acc.isDirty).toBe(true);
251
263
  });
252
264
 
253
- it("markSubAgentPersisted clears the dirty flag, and an update re-marks it", () => {
265
+ it("markPersisted clears the dirty flag, and a sub-agent update re-marks it", () => {
254
266
  const acc = new MessageAccumulator([]);
255
267
 
256
268
  acc.trackSubAgentExecution(
@@ -258,17 +270,17 @@ describe("MessageAccumulator tool call status transitions", () => {
258
270
  args: { description: "Research", prompt: "Go" },
259
271
  }),
260
272
  );
261
- expect(acc.subAgentDirty).toBe(true);
273
+ expect(acc.isDirty).toBe(true);
262
274
 
263
- acc.markSubAgentPersisted();
264
- expect(acc.subAgentDirty).toBe(false);
275
+ acc.markPersisted();
276
+ expect(acc.isDirty).toBe(false);
265
277
 
266
278
  // The completion transition (IN_PROGRESS -> COMPLETED) must re-mark dirty
267
279
  // so the terminal sub-agent state is persisted to the live stream.
268
280
  acc.trackSubAgentExecution(
269
281
  toolCallEvent("tc-dirty2", "task", "completed", "r1", { result: "Done" }),
270
282
  );
271
- expect(acc.subAgentDirty).toBe(true);
283
+ expect(acc.isDirty).toBe(true);
272
284
  expect(acc.subAgentExecutions[0].status).toBe(SubAgentStatus.SUB_AGENT_COMPLETED);
273
285
  });
274
286
 
@@ -291,8 +303,8 @@ describe("MessageAccumulator tool call status transitions", () => {
291
303
  toolCallEvent("tc-done", "task", "completed", "r1", { result: "Found it" }),
292
304
  );
293
305
 
294
- acc.markSubAgentPersisted();
295
- expect(acc.subAgentDirty).toBe(false);
306
+ acc.markPersisted();
307
+ expect(acc.isDirty).toBe(false);
296
308
 
297
309
  acc.cancelInProgressSubAgents();
298
310
 
@@ -303,14 +315,14 @@ describe("MessageAccumulator tool call status transitions", () => {
303
315
  expect(running.completedAt).not.toBe("");
304
316
  expect(done.status).toBe(SubAgentStatus.SUB_AGENT_COMPLETED);
305
317
  // The transition must mark dirty so the cancellation is persisted.
306
- expect(acc.subAgentDirty).toBe(true);
318
+ expect(acc.isDirty).toBe(true);
307
319
  });
308
320
 
309
321
  it("cancelInProgressSubAgents is a no-op when there are no running sub-agents", () => {
310
322
  const acc = new MessageAccumulator([]);
311
323
  acc.cancelInProgressSubAgents();
312
324
  expect(acc.subAgentExecutions).toHaveLength(0);
313
- expect(acc.subAgentDirty).toBe(false);
325
+ expect(acc.isDirty).toBe(false);
314
326
  });
315
327
 
316
328
  it("sub-agent name extraction handles object subagentType", () => {
@@ -710,6 +722,106 @@ describe("MessageAccumulator tool call status transitions", () => {
710
722
  });
711
723
  });
712
724
 
725
+ // Issue #179: the live thinking/tool-call trace was starved because the Cursor
726
+ // loop's persist cadence had no force-flush for tool-call lifecycle. The
727
+ // accumulator's isDirty signal is now the force-flush source: it MUST fire on
728
+ // the discrete, user-visible events (tool start, tool finish, sub-agent
729
+ // delegation) and MUST NOT fire on high-frequency token deltas (assistant
730
+ // text, model thinking), which ride the StreamingUpdateScheduler time cadence.
731
+ describe("streaming force-flush signal (issue #179 cadence)", () => {
732
+ it("flags dirty the instant a tool call is created", () => {
733
+ const acc = new MessageAccumulator([]);
734
+ acc.processEvent(assistantEvent("r1", "Let me search."));
735
+ expect(acc.isDirty).toBe(false); // assistant text alone does not force-flush
736
+
737
+ acc.processEvent(toolCallEvent("tc-1", "Shell", "running", "r1"));
738
+ expect(acc.isDirty).toBe(true);
739
+ });
740
+
741
+ it("flags dirty when a tool call transitions to a terminal status", () => {
742
+ const acc = new MessageAccumulator([]);
743
+ acc.processEvent(assistantEvent("r1", "Running."));
744
+ acc.processEvent(toolCallEvent("tc-1", "Shell", "running", "r1"));
745
+ acc.markPersisted();
746
+ expect(acc.isDirty).toBe(false);
747
+
748
+ acc.processEvent(toolCallEvent("tc-1", "Shell", "completed", "r1", { result: "OK" }));
749
+ expect(acc.isDirty).toBe(true);
750
+ });
751
+
752
+ it("does NOT re-flag dirty on a redundant terminal re-emit", () => {
753
+ const acc = new MessageAccumulator([]);
754
+ acc.processEvent(assistantEvent("r1", "Running."));
755
+ acc.processEvent(toolCallEvent("tc-1", "read", "running", "r1"));
756
+ acc.processEvent(toolCallEvent("tc-1", "read", "completed", "r1", { result: "data" }));
757
+ acc.markPersisted();
758
+ expect(acc.isDirty).toBe(false);
759
+
760
+ // An already-terminal call re-emitting is noise, not a state change.
761
+ acc.processEvent(toolCallEvent("tc-1", "read", "completed", "r1", { result: "" }));
762
+ expect(acc.isDirty).toBe(false);
763
+ });
764
+
765
+ it("does NOT flag dirty on model thinking deltas (they ride the scheduler cadence)", () => {
766
+ const acc = new MessageAccumulator([]);
767
+ acc.processEvent(thinkingEvent("r1", "Let me reason about this"));
768
+ acc.processEvent(thinkingEvent("r1", " step by step..."));
769
+ expect(acc.isDirty).toBe(false);
770
+ });
771
+
772
+ it("does NOT flag dirty on assistant text deltas", () => {
773
+ const acc = new MessageAccumulator([]);
774
+ acc.processEvent(assistantEvent("r1", "Here is "));
775
+ acc.processEvent(assistantEvent("r1", "the answer."));
776
+ expect(acc.isDirty).toBe(false);
777
+ });
778
+
779
+ it("suppressed todo tools do NOT flag dirty (TodoTracker owns that signal)", () => {
780
+ const acc = new MessageAccumulator([]);
781
+ acc.processEvent(assistantEvent("r1", "Planning."));
782
+ acc.processEvent(toolCallEvent("tc-todo", "updateTodos", "running", "r1", {
783
+ args: { todos: [{ content: "Step 1", status: "pending" }] },
784
+ }));
785
+ expect(acc.isDirty).toBe(false);
786
+ });
787
+
788
+ it("markPersisted clears a tool-call dirty flag", () => {
789
+ const acc = new MessageAccumulator([]);
790
+ acc.processEvent(assistantEvent("r1", "Running."));
791
+ acc.processEvent(toolCallEvent("tc-1", "Shell", "running", "r1"));
792
+ expect(acc.isDirty).toBe(true);
793
+
794
+ acc.markPersisted();
795
+ expect(acc.isDirty).toBe(false);
796
+ });
797
+
798
+ // The user-facing symptom: a short thinking+tool turn (< 20 stream events).
799
+ // The old `eventCount % 20` gate never fired, so the whole trace landed only
800
+ // at the final persist. With the force-flush signal, the tool lifecycle is
801
+ // observable mid-turn — each discrete event leaves isDirty set for the loop.
802
+ it("a short thinking+tool turn produces force-flush points mid-turn", () => {
803
+ const acc = new MessageAccumulator([]);
804
+ const flushPoints: string[] = [];
805
+ const step = (label: string, ev: SDKMessage) => {
806
+ acc.processEvent(ev);
807
+ if (acc.isDirty) {
808
+ flushPoints.push(label);
809
+ acc.markPersisted();
810
+ }
811
+ };
812
+
813
+ step("thinking", thinkingEvent("r1", "Thinking about the task..."));
814
+ step("assistant", assistantEvent("r1", "I'll check the file."));
815
+ step("tool-start", toolCallEvent("tc-1", "read", "running", "r1"));
816
+ step("tool-done", toolCallEvent("tc-1", "read", "completed", "r1", { result: "contents" }));
817
+ step("assistant-2", assistantEvent("r1", "Done."));
818
+
819
+ // Tool start and tool completion are the discrete moments the live UI must
820
+ // see immediately; thinking/assistant deltas are carried by the scheduler.
821
+ expect(flushPoints).toEqual(["tool-start", "tool-done"]);
822
+ });
823
+ });
824
+
713
825
  describe("cancelInProgressSubAgentProtos standalone", () => {
714
826
  it("cancels IN_PROGRESS/PENDING protos in place and reports whether anything changed", () => {
715
827
  const running = create(SubAgentExecutionSchema, {
@@ -0,0 +1,99 @@
1
+ /**
2
+ * Unit tests for the Cursor streaming persist decision (issue #179).
3
+ *
4
+ * Validates the converged cadence: discrete force-flush signals persist
5
+ * immediately, while high-frequency token deltas ride the shared
6
+ * StreamingUpdateScheduler's time cadence (rather than the old, time-blind
7
+ * `eventCount % 20` gate that starved short turns).
8
+ */
9
+
10
+ import { describe, it, expect } from "vitest";
11
+ import {
12
+ StreamingUpdateScheduler,
13
+ type StreamingConfig,
14
+ } from "../../../shared/streaming-scheduler.js";
15
+ import { shouldPersistStreamingStatus } from "../persist-decision.js";
16
+
17
+ const CONFIG: StreamingConfig = {
18
+ minIntervalMs: 500,
19
+ maxIntervalMs: 5000,
20
+ burstThreshold: 50,
21
+ };
22
+
23
+ const NO_FORCE = {
24
+ deltaEnricherDirty: false,
25
+ todosDirty: false,
26
+ contentDirty: false,
27
+ } as const;
28
+
29
+ describe("shouldPersistStreamingStatus", () => {
30
+ it("persists on the first event via the scheduler (kills the opaque-wait at turn start)", () => {
31
+ const scheduler = new StreamingUpdateScheduler(CONFIG, 0);
32
+ expect(shouldPersistStreamingStatus(NO_FORCE, scheduler, 1, 0)).toBe(true);
33
+ });
34
+
35
+ it("does not persist a sub-500ms token delta with no force signal", () => {
36
+ const scheduler = new StreamingUpdateScheduler(CONFIG, 0);
37
+ scheduler.markUpdateSent(1, 0); // first flush already sent
38
+
39
+ // 100ms later, one more delta event — too soon for the time cadence.
40
+ expect(shouldPersistStreamingStatus(NO_FORCE, scheduler, 2, 100)).toBe(false);
41
+ });
42
+
43
+ it("persists a token delta once the 500ms cadence elapses", () => {
44
+ const scheduler = new StreamingUpdateScheduler(CONFIG, 0);
45
+ scheduler.markUpdateSent(1, 0);
46
+
47
+ expect(shouldPersistStreamingStatus(NO_FORCE, scheduler, 2, 500)).toBe(true);
48
+ });
49
+
50
+ it.each([
51
+ ["contentDirty (tool call / sub-agent)", { ...NO_FORCE, contentDirty: true }],
52
+ ["todosDirty", { ...NO_FORCE, todosDirty: true }],
53
+ ["deltaEnricherDirty (shell output)", { ...NO_FORCE, deltaEnricherDirty: true }],
54
+ ])("force-flushes immediately on %s, bypassing the time cadence", (_label, signals) => {
55
+ const scheduler = new StreamingUpdateScheduler(CONFIG, 0);
56
+ scheduler.markUpdateSent(1, 0);
57
+
58
+ // Only 1ms after the last flush: the scheduler alone would say no...
59
+ expect(shouldPersistStreamingStatus(NO_FORCE, scheduler, 2, 1)).toBe(false);
60
+ // ...but any discrete force signal flushes now.
61
+ expect(shouldPersistStreamingStatus(signals, scheduler, 2, 1)).toBe(true);
62
+ });
63
+
64
+ // The #179 starvation scenario: a short turn of < 20 events. Under the old
65
+ // `eventCount % 20` gate, a tool call appearing at event 3 (and completing at
66
+ // event 4) would never be persisted until the unconditional final flush. With
67
+ // the converged decision, the tool lifecycle force-flushes the instant it
68
+ // happens, regardless of how few events the turn has.
69
+ it("surfaces a tool call on a short (<20 event) turn", () => {
70
+ const scheduler = new StreamingUpdateScheduler(CONFIG, 0);
71
+ const flushes: number[] = [];
72
+ let clock = 0;
73
+
74
+ // Helper mirroring the loop: decide, then mark both clocks on a flush.
75
+ const tick = (eventCount: number, contentDirty: boolean) => {
76
+ const persist = shouldPersistStreamingStatus(
77
+ { ...NO_FORCE, contentDirty },
78
+ scheduler,
79
+ eventCount,
80
+ clock,
81
+ );
82
+ if (persist) {
83
+ flushes.push(eventCount);
84
+ scheduler.markUpdateSent(eventCount, clock);
85
+ }
86
+ clock += 50; // 50ms between events — well under the 500ms floor
87
+ };
88
+
89
+ tick(1, false); // assistant delta — first-event flush
90
+ tick(2, false); // thinking delta — within 500ms, no flush
91
+ tick(3, true); // tool call starts — force-flush
92
+ tick(4, true); // tool call completes — force-flush
93
+ tick(5, false); // closing assistant delta — within 500ms, no flush
94
+
95
+ // The tool lifecycle (events 3 and 4) is observed live; event 1 is the
96
+ // first-event flush. Events 2 and 5 correctly ride the cadence.
97
+ expect(flushes).toEqual([1, 3, 4]);
98
+ });
99
+ });
@@ -0,0 +1,244 @@
1
+ /**
2
+ * Unit tests for Cursor MCP image-result normalization.
3
+ *
4
+ * The Cursor SDK wraps an MCP tool result as
5
+ * { status, value: { content: [ { text:{text} }, { image:{ data, mimeType } } ] } }
6
+ * where image `data` is a Node Buffer-JSON ({ type:"Buffer", data:number[] }).
7
+ * The translator must re-emit that as the canonical top-level content-block
8
+ * array the shared persist-time offload consumes, so a screenshot lands as a
9
+ * renderable image ToolCallOutputRef instead of text/plain. These tests pin
10
+ * that normalization and confirm it flows end-to-end through the offload.
11
+ */
12
+
13
+ import { describe, it, expect, vi } from "vitest";
14
+ import { create } from "@bufbuild/protobuf";
15
+ import { AgentExecutionStatusSchema } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/api_pb";
16
+ import { AgentMessageSchema } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/message_pb";
17
+ import type { SDKMessage } from "@cursor/sdk";
18
+ import type { ArtifactStorage } from "../../../shared/artifact-storage.js";
19
+ import {
20
+ offloadOversizedToolOutputs,
21
+ detectImagePayload,
22
+ } from "../../../shared/status-offload.js";
23
+ import {
24
+ toResultString,
25
+ canonicalizeImageResult,
26
+ buildToolCallProto,
27
+ MessageAccumulator,
28
+ } from "../message-translator.js";
29
+ import type { AgentMessage } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/message_pb";
30
+
31
+ // PNG signature + a little payload, the way the Cursor SDK serializes bytes.
32
+ const PNG_BYTES = [137, 80, 78, 71, 13, 10, 26, 10, 0, 0, 0, 13, 73, 72, 68, 82];
33
+ const PNG_BASE64 = Buffer.from(PNG_BYTES).toString("base64");
34
+
35
+ function cursorImageEnvelope(text = "App=com.example") {
36
+ return {
37
+ status: "success",
38
+ value: {
39
+ content: [
40
+ { text: { text } },
41
+ { image: { data: { type: "Buffer", data: PNG_BYTES }, mimeType: "image/png" } },
42
+ ],
43
+ isError: false,
44
+ },
45
+ };
46
+ }
47
+
48
+ describe("canonicalizeImageResult", () => {
49
+ it("converts a Cursor image envelope (Buffer-JSON) to the canonical array", () => {
50
+ const out = canonicalizeImageResult(cursorImageEnvelope("App=Slack"));
51
+ expect(out).toBeDefined();
52
+ expect(JSON.parse(out!)).toEqual([
53
+ { type: "text", text: "App=Slack" },
54
+ { type: "image", data: PNG_BASE64, mimeType: "image/png" },
55
+ ]);
56
+ });
57
+
58
+ it("handles a bare { content: [...] } envelope (no status/value wrapper)", () => {
59
+ const out = canonicalizeImageResult({
60
+ content: [{ image: { data: { type: "Buffer", data: PNG_BYTES }, mimeType: "image/png" } }],
61
+ });
62
+ expect(JSON.parse(out!)).toEqual([
63
+ { type: "image", data: PNG_BASE64, mimeType: "image/png" },
64
+ ]);
65
+ });
66
+
67
+ it("accepts an already-base64 image data string", () => {
68
+ const out = canonicalizeImageResult({
69
+ content: [{ image: { data: PNG_BASE64, mimeType: "image/png" } }],
70
+ });
71
+ expect(JSON.parse(out!)).toEqual([
72
+ { type: "image", data: PNG_BASE64, mimeType: "image/png" },
73
+ ]);
74
+ });
75
+
76
+ it("accepts a data: URL image data string", () => {
77
+ const out = canonicalizeImageResult({
78
+ content: [{ image: { data: `data:image/png;base64,${PNG_BASE64}`, mimeType: "image/png" } }],
79
+ });
80
+ expect(JSON.parse(out!)).toEqual([
81
+ { type: "image", data: PNG_BASE64, mimeType: "image/png" },
82
+ ]);
83
+ });
84
+
85
+ it("defaults mimeType to image/png when absent", () => {
86
+ const out = canonicalizeImageResult({
87
+ content: [{ image: { data: { type: "Buffer", data: PNG_BYTES } } }],
88
+ });
89
+ expect(JSON.parse(out!)[0]).toEqual({ type: "image", data: PNG_BASE64, mimeType: "image/png" });
90
+ });
91
+
92
+ it("returns undefined for a text-only envelope (no transformation)", () => {
93
+ expect(canonicalizeImageResult({ status: "success", value: { content: [{ text: { text: "hi" } }] } }))
94
+ .toBeUndefined();
95
+ });
96
+
97
+ it("returns undefined when there is no content array", () => {
98
+ expect(canonicalizeImageResult({ status: "success", value: { stdout: "ok" } })).toBeUndefined();
99
+ expect(canonicalizeImageResult("plain string")).toBeUndefined();
100
+ expect(canonicalizeImageResult(null)).toBeUndefined();
101
+ });
102
+ });
103
+
104
+ describe("toResultString", () => {
105
+ it("passes a string through unchanged", () => {
106
+ expect(toResultString("just logs")).toBe("just logs");
107
+ });
108
+
109
+ it("returns empty string for an absent result", () => {
110
+ expect(toResultString(null)).toBe("");
111
+ expect(toResultString(undefined)).toBe("");
112
+ });
113
+
114
+ it("JSON.stringifies a non-image object unchanged (text-only envelope)", () => {
115
+ const env = { status: "success", value: { content: [{ text: { text: "hi" } }] } };
116
+ expect(toResultString(env)).toBe(JSON.stringify(env));
117
+ });
118
+
119
+ it("normalizes an image envelope to the canonical array", () => {
120
+ const out = toResultString(cursorImageEnvelope("App=X"));
121
+ expect(JSON.parse(out)).toEqual([
122
+ { type: "text", text: "App=X" },
123
+ { type: "image", data: PNG_BASE64, mimeType: "image/png" },
124
+ ]);
125
+ });
126
+ });
127
+
128
+ describe("buildToolCallProto image normalization", () => {
129
+ it("produces a result the shared offload detector recognizes as an image", () => {
130
+ const event = {
131
+ type: "tool_call",
132
+ agent_id: "a1",
133
+ run_id: "r1",
134
+ call_id: "tc-img",
135
+ name: "mcp",
136
+ status: "completed",
137
+ args: { providerIdentifier: "open-computer-use", toolName: "get_app_state", args: {} },
138
+ result: cursorImageEnvelope(),
139
+ } as unknown as Extract<SDKMessage, { type: "tool_call" }>;
140
+
141
+ const tc = buildToolCallProto(event);
142
+ expect(tc.name).toBe("get_app_state");
143
+ const img = detectImagePayload(tc.result);
144
+ expect(img).not.toBeNull();
145
+ expect(img?.mimeType).toBe("image/png");
146
+ expect(img?.base64).toBe(PNG_BASE64);
147
+ // The bloated Buffer-JSON must not survive into the persisted result.
148
+ expect(tc.result).not.toContain('"Buffer"');
149
+ });
150
+ });
151
+
152
+ describe("cursor image flows through the persist-time offload", () => {
153
+ it("offloads the screenshot as an image ref with no inline bytes", async () => {
154
+ const uploads: { key: string; contentType?: string }[] = [];
155
+ const storage: ArtifactStorage = {
156
+ upload: vi.fn(async (key: string, _content: Buffer, contentType?: string) => {
157
+ uploads.push({ key, contentType });
158
+ return key;
159
+ }),
160
+ getDownloadUrl: vi.fn(async (key: string) => `https://artifacts.local/${key}`),
161
+ exists: vi.fn(async () => true),
162
+ };
163
+
164
+ const event = {
165
+ type: "tool_call",
166
+ agent_id: "a1",
167
+ run_id: "r1",
168
+ call_id: "tc-img",
169
+ name: "mcp",
170
+ status: "completed",
171
+ args: { providerIdentifier: "open-computer-use", toolName: "get_app_state", args: {} },
172
+ result: cursorImageEnvelope(),
173
+ } as unknown as Extract<SDKMessage, { type: "tool_call" }>;
174
+
175
+ const tc = buildToolCallProto(event);
176
+ const status = create(AgentExecutionStatusSchema, {
177
+ messages: [create(AgentMessageSchema, { toolCalls: [tc] })],
178
+ });
179
+
180
+ await offloadOversizedToolOutputs(status, { artifactStorage: storage, executionId: "exec-1" });
181
+
182
+ const out = status.messages[0].toolCalls[0];
183
+ expect(out.outputRef).toBeDefined();
184
+ expect(out.outputRef!.isImage).toBe(true);
185
+ expect(out.outputRef!.mimeType).toBe("image/png");
186
+ expect(out.outputRef!.storageKey.endsWith(".png")).toBe(true);
187
+ expect(uploads[0]?.contentType).toBe("image/png");
188
+ // Inline result is collapsed; no base64/Buffer bytes remain in the status.
189
+ expect(out.result).not.toContain(PNG_BASE64);
190
+ expect(out.result).not.toContain('"Buffer"');
191
+ });
192
+ });
193
+
194
+ describe("sub-agent image normalization (extractConversationSteps)", () => {
195
+ it("normalizes a screenshot returned inside a sub-agent toolCall step", () => {
196
+ const messages: AgentMessage[] = [];
197
+ const acc = new MessageAccumulator(messages);
198
+
199
+ const running = {
200
+ type: "tool_call",
201
+ agent_id: "a1",
202
+ run_id: "r1",
203
+ call_id: "tc-sub-img",
204
+ name: "task",
205
+ status: "running",
206
+ args: { description: "screenshot", prompt: "capture" },
207
+ } as unknown as Extract<SDKMessage, { type: "tool_call" }>;
208
+ acc.processEvent(running);
209
+ acc.trackSubAgentExecution(running);
210
+
211
+ const completed = {
212
+ type: "tool_call",
213
+ agent_id: "a1",
214
+ run_id: "r1",
215
+ call_id: "tc-sub-img",
216
+ name: "task",
217
+ status: "completed",
218
+ args: { description: "screenshot", prompt: "capture" },
219
+ result: {
220
+ status: "success",
221
+ value: {
222
+ conversationSteps: [
223
+ {
224
+ type: "toolCall",
225
+ message: {
226
+ type: "get_app_state",
227
+ args: {},
228
+ result: cursorImageEnvelope("App=SubAgent"),
229
+ },
230
+ },
231
+ ],
232
+ },
233
+ },
234
+ } as unknown as Extract<SDKMessage, { type: "tool_call" }>;
235
+
236
+ acc.processEvent(completed);
237
+ acc.trackSubAgentExecution(completed);
238
+
239
+ const sub = acc.subAgentExecutions[0];
240
+ const subToolResult = sub.messages[0].toolCalls[0].result;
241
+ const img = detectImagePayload(subToolResult);
242
+ expect(img?.base64).toBe(PNG_BASE64);
243
+ });
244
+ });
@@ -43,9 +43,12 @@ function freshRoot(): string {
43
43
  const stigmerScript = (root: string) =>
44
44
  join(root, ".stigmer", "sessions", "ses-1", "hitl", "stigmer-approval.sh");
45
45
 
46
+ // Single preToolUse registration — the common shape in these tests.
47
+ const pre = (scriptPath: string) => [{ event: "preToolUse", scriptPath }];
48
+
46
49
  describe("buildMergedConfig", () => {
47
50
  it("writes a standalone config and restores by delete when no hooks.json exists", () => {
48
- const { merged, restoreTo } = buildMergedConfig(null, "/abs/hitl/stigmer-approval.sh");
51
+ const { merged, restoreTo } = buildMergedConfig(null, pre("/abs/hitl/stigmer-approval.sh"));
49
52
  const parsed = JSON.parse(merged);
50
53
  expect(parsed.hooks.preToolUse).toHaveLength(1);
51
54
  expect(parsed.hooks.preToolUse[0].command).toBe("/abs/hitl/stigmer-approval.sh");
@@ -54,6 +57,47 @@ describe("buildMergedConfig", () => {
54
57
  expect(restoreTo).toBeNull();
55
58
  });
56
59
 
60
+ it("registers multiple events (preToolUse + beforeMCPExecution) and restores by delete", () => {
61
+ const { merged, restoreTo } = buildMergedConfig(null, [
62
+ { event: "preToolUse", scriptPath: "/abs/hitl/stigmer-approval.sh" },
63
+ { event: "beforeMCPExecution", scriptPath: "/abs/hitl/stigmer-mcp-capture.sh" },
64
+ ]);
65
+ const parsed = JSON.parse(merged);
66
+ expect(parsed.hooks.preToolUse[0].command).toBe("/abs/hitl/stigmer-approval.sh");
67
+ expect(parsed.hooks.beforeMCPExecution[0].command).toBe("/abs/hitl/stigmer-mcp-capture.sh");
68
+ expect(parsed.hooks.beforeMCPExecution[0].failClosed).toBe(true);
69
+ expect(restoreTo).toBeNull();
70
+ });
71
+
72
+ it("merges into both event arrays and strips stale Stigmer entries from each on restore", () => {
73
+ const root = "/abs";
74
+ const stalePre = join(root, ".stigmer", "sessions", "ses-1", "hitl", "stigmer-approval.sh");
75
+ const staleMcp = join(root, ".stigmer", "sessions", "ses-1", "hitl", "stigmer-mcp-capture.sh");
76
+ const original = JSON.stringify({
77
+ version: 1,
78
+ hooks: {
79
+ preToolUse: [{ command: "./user.sh" }, { command: stalePre, failClosed: true }],
80
+ beforeMCPExecution: [{ command: staleMcp, failClosed: true }],
81
+ },
82
+ });
83
+ const freshPre = join(root, ".stigmer", "sessions", "ses-2", "hitl", "stigmer-approval.sh");
84
+ const freshMcp = join(root, ".stigmer", "sessions", "ses-2", "hitl", "stigmer-mcp-capture.sh");
85
+
86
+ const { merged, restoreTo } = buildMergedConfig(original, [
87
+ { event: "preToolUse", scriptPath: freshPre },
88
+ { event: "beforeMCPExecution", scriptPath: freshMcp },
89
+ ]);
90
+
91
+ const m = JSON.parse(merged);
92
+ expect(m.hooks.preToolUse.map((e: any) => e.command)).toEqual(["./user.sh", freshPre]);
93
+ expect(m.hooks.beforeMCPExecution.map((e: any) => e.command)).toEqual([freshMcp]);
94
+
95
+ // Restore is self-healing: every stale Stigmer entry is removed from both.
96
+ const r = JSON.parse(restoreTo!);
97
+ expect(r.hooks.preToolUse).toEqual([{ command: "./user.sh" }]);
98
+ expect(r.hooks.beforeMCPExecution).toEqual([]);
99
+ });
100
+
57
101
  it("merges with a user's hooks.json and restores the original bytes verbatim", () => {
58
102
  const original = JSON.stringify(
59
103
  {
@@ -67,7 +111,7 @@ describe("buildMergedConfig", () => {
67
111
  2,
68
112
  );
69
113
  const script = "/abs/.stigmer/sessions/ses-1/hitl/stigmer-approval.sh";
70
- const { merged, restoreTo } = buildMergedConfig(original, script);
114
+ const { merged, restoreTo } = buildMergedConfig(original, pre(script));
71
115
  const parsed = JSON.parse(merged);
72
116
 
73
117
  // Our entry is appended; the user's preToolUse hook is preserved...
@@ -93,7 +137,7 @@ describe("buildMergedConfig", () => {
93
137
  },
94
138
  });
95
139
  const fresh = join(root, ".stigmer", "sessions", "ses-2", "hitl", "stigmer-approval.sh");
96
- const { merged, restoreTo } = buildMergedConfig(original, fresh);
140
+ const { merged, restoreTo } = buildMergedConfig(original, pre(fresh));
97
141
 
98
142
  const mergedParsed = JSON.parse(merged);
99
143
  // No duplicate: user entry + exactly one fresh Stigmer entry.
@@ -109,7 +153,7 @@ describe("buildMergedConfig", () => {
109
153
 
110
154
  it("replaces an unparseable hooks.json for the turn but restores its exact bytes", () => {
111
155
  const garbage = "{ this is not json ";
112
- const { merged, restoreTo } = buildMergedConfig(garbage, "/abs/hitl/stigmer-approval.sh");
156
+ const { merged, restoreTo } = buildMergedConfig(garbage, pre("/abs/hitl/stigmer-approval.sh"));
113
157
  // We still install a working gate for the turn...
114
158
  expect(JSON.parse(merged).hooks.preToolUse).toHaveLength(1);
115
159
  // ...and never "fix" the user's file: restore their exact original bytes.
@@ -147,6 +191,11 @@ describe("installHitlGate / removeHitlGate", () => {
147
191
  const command = hooksJson.hooks.preToolUse[0].command;
148
192
  expect(command).toBe(join(hitlDir, "stigmer-approval.sh"));
149
193
  expect(command.startsWith("/")).toBe(true);
194
+ // The SAME script gates MCP via beforeMCPExecution (preToolUse does not
195
+ // enforce MCP); the script branches internally on hook_event_name.
196
+ expect(hooksJson.hooks.beforeMCPExecution[0].command).toBe(
197
+ join(hitlDir, "stigmer-approval.sh"),
198
+ );
150
199
  // The workspace holds no relocated artifacts.
151
200
  expect(existsSync(join(workspaceRoot, ".cursor", "hooks"))).toBe(false);
152
201
  });