@stigmer/runner 3.0.8-dev.20260613085218 → 3.0.9-dev.20260615145121
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/.build-fingerprint +1 -1
- package/dist/activities/execute-cursor/hook-script.d.ts +23 -12
- package/dist/activities/execute-cursor/hook-script.js +85 -51
- package/dist/activities/execute-cursor/hook-script.js.map +1 -1
- package/dist/activities/execute-cursor/index.js +220 -79
- package/dist/activities/execute-cursor/index.js.map +1 -1
- package/dist/activities/execute-cursor/message-translator.d.ts +51 -9
- package/dist/activities/execute-cursor/message-translator.js +146 -20
- package/dist/activities/execute-cursor/message-translator.js.map +1 -1
- package/dist/activities/execute-cursor/persist-decision.d.ts +42 -0
- package/dist/activities/execute-cursor/persist-decision.js +30 -0
- package/dist/activities/execute-cursor/persist-decision.js.map +1 -0
- package/dist/activities/execute-cursor/prompt-builder.d.ts +25 -0
- package/dist/activities/execute-cursor/prompt-builder.js +54 -0
- package/dist/activities/execute-cursor/prompt-builder.js.map +1 -1
- package/dist/activities/execute-cursor/workspace-setup.d.ts +8 -2
- package/dist/activities/execute-cursor/workspace-setup.js +62 -30
- package/dist/activities/execute-cursor/workspace-setup.js.map +1 -1
- package/dist/activities/execute-deep-agent/index.js +15 -5
- package/dist/activities/execute-deep-agent/index.js.map +1 -1
- package/dist/activities/execute-deep-agent/status-builder-shared.d.ts +0 -1
- package/dist/activities/execute-deep-agent/status-builder-shared.js +32 -8
- package/dist/activities/execute-deep-agent/status-builder-shared.js.map +1 -1
- package/dist/activities/execute-deep-agent/status-builder.js +4 -5
- package/dist/activities/execute-deep-agent/status-builder.js.map +1 -1
- package/dist/activities/execute-deep-agent/streaming-v3.js +4 -5
- package/dist/activities/execute-deep-agent/streaming-v3.js.map +1 -1
- package/dist/activities/execute-deep-agent/streaming.d.ts +9 -1
- package/dist/activities/execute-deep-agent/streaming.js +4 -5
- package/dist/activities/execute-deep-agent/streaming.js.map +1 -1
- package/dist/activities/execute-deep-agent/subagent-tracker.js +4 -5
- package/dist/activities/execute-deep-agent/subagent-tracker.js.map +1 -1
- package/dist/activities/execute-deep-agent/v3-status-builder.js +6 -5
- package/dist/activities/execute-deep-agent/v3-status-builder.js.map +1 -1
- package/dist/config.d.ts +21 -0
- package/dist/config.js +12 -0
- package/dist/config.js.map +1 -1
- package/dist/in-flight.d.ts +35 -0
- package/dist/in-flight.js +61 -0
- package/dist/in-flight.js.map +1 -0
- package/dist/runner-manager.d.ts +2 -0
- package/dist/runner-manager.js +90 -29
- package/dist/runner-manager.js.map +1 -1
- package/dist/runner.d.ts +2 -0
- package/dist/runner.js +2 -0
- package/dist/runner.js.map +1 -1
- package/dist/shared/grpc-retry.d.ts +9 -20
- package/dist/shared/grpc-retry.js +9 -52
- package/dist/shared/grpc-retry.js.map +1 -1
- package/dist/shared/stall-watchdog.d.ts +68 -0
- package/dist/shared/stall-watchdog.js +102 -0
- package/dist/shared/stall-watchdog.js.map +1 -0
- package/dist/shared/status-offload.d.ts +84 -0
- package/dist/shared/status-offload.js +292 -0
- package/dist/shared/status-offload.js.map +1 -0
- package/dist/shared/status.d.ts +34 -3
- package/dist/shared/status.js +102 -9
- package/dist/shared/status.js.map +1 -1
- package/dist/{activities/execute-deep-agent → shared}/streaming-scheduler.d.ts +4 -0
- package/dist/{activities/execute-deep-agent → shared}/streaming-scheduler.js +4 -0
- package/dist/shared/streaming-scheduler.js.map +1 -0
- package/package.json +2 -2
- package/src/__tests__/config.test.ts +8 -0
- package/src/__tests__/in-flight.test.ts +84 -0
- package/src/activities/__tests__/classify-tool-approvals.test.ts +1 -0
- package/src/activities/__tests__/discover-mcp-server.test.ts +1 -0
- package/src/activities/execute-cursor/__tests__/build-prompt.test.ts +74 -0
- package/src/activities/execute-cursor/__tests__/hook-script.test.ts +90 -15
- package/src/activities/execute-cursor/__tests__/message-translator.test.ts +124 -12
- package/src/activities/execute-cursor/__tests__/persist-decision.test.ts +99 -0
- package/src/activities/execute-cursor/__tests__/tool-result-image.test.ts +244 -0
- package/src/activities/execute-cursor/__tests__/workspace-setup.test.ts +53 -4
- package/src/activities/execute-cursor/hook-script.ts +85 -51
- package/src/activities/execute-cursor/index.ts +187 -38
- package/src/activities/execute-cursor/message-translator.ts +146 -20
- package/src/activities/execute-cursor/persist-decision.ts +54 -0
- package/src/activities/execute-cursor/prompt-builder.ts +59 -0
- package/src/activities/execute-cursor/workspace-setup.ts +76 -44
- package/src/activities/execute-deep-agent/__tests__/index.test.ts +1 -0
- package/src/activities/execute-deep-agent/__tests__/status-builder-shared.test.ts +66 -0
- package/src/activities/execute-deep-agent/__tests__/status-builder.test.ts +6 -3
- package/src/activities/execute-deep-agent/__tests__/streaming-v3.test.ts +70 -0
- package/src/activities/execute-deep-agent/index.ts +17 -5
- package/src/activities/execute-deep-agent/status-builder-shared.ts +27 -5
- package/src/activities/execute-deep-agent/status-builder.ts +3 -5
- package/src/activities/execute-deep-agent/streaming-v3.ts +5 -5
- package/src/activities/execute-deep-agent/streaming.ts +14 -5
- package/src/activities/execute-deep-agent/subagent-tracker.ts +4 -5
- package/src/activities/execute-deep-agent/v3-status-builder.ts +5 -5
- package/src/config.ts +27 -0
- package/src/in-flight.ts +71 -0
- package/src/runner-manager.ts +127 -33
- package/src/runner.ts +6 -0
- package/src/shared/__tests__/artifact-storage.test.ts +1 -0
- package/src/shared/__tests__/grpc-retry-extended.test.ts +6 -144
- package/src/shared/__tests__/grpc-retry.test.ts +5 -134
- package/src/shared/__tests__/stall-watchdog.test.ts +193 -0
- package/src/shared/__tests__/status-offload.test.ts +256 -0
- package/src/shared/__tests__/status.test.ts +199 -0
- package/src/shared/grpc-retry.ts +9 -72
- package/src/shared/stall-watchdog.ts +122 -0
- package/src/shared/status-offload.ts +342 -0
- package/src/shared/status.ts +142 -8
- package/src/{activities/execute-deep-agent → shared}/streaming-scheduler.ts +4 -0
- package/dist/activities/execute-deep-agent/streaming-scheduler.js.map +0 -1
- /package/src/{activities/execute-deep-agent → shared}/__tests__/streaming-scheduler.test.ts +0 -0
|
@@ -58,6 +58,18 @@ function assistantEvent(
|
|
|
58
58
|
};
|
|
59
59
|
}
|
|
60
60
|
|
|
61
|
+
function thinkingEvent(
|
|
62
|
+
runId: string,
|
|
63
|
+
text: string,
|
|
64
|
+
): Extract<SDKMessage, { type: "thinking" }> {
|
|
65
|
+
return {
|
|
66
|
+
type: "thinking",
|
|
67
|
+
agent_id: "agent-1",
|
|
68
|
+
run_id: runId,
|
|
69
|
+
text,
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
|
|
61
73
|
function countToolCallsWithId(messages: AgentMessage[], callId: string): number {
|
|
62
74
|
let count = 0;
|
|
63
75
|
for (const msg of messages) {
|
|
@@ -237,9 +249,9 @@ describe("MessageAccumulator tool call status transitions", () => {
|
|
|
237
249
|
expect(acc.subAgentExecutions[0].status).toBe(SubAgentStatus.SUB_AGENT_IN_PROGRESS);
|
|
238
250
|
});
|
|
239
251
|
|
|
240
|
-
it("
|
|
252
|
+
it("isDirty starts false and is set when a sub-agent is created", () => {
|
|
241
253
|
const acc = new MessageAccumulator([]);
|
|
242
|
-
expect(acc.
|
|
254
|
+
expect(acc.isDirty).toBe(false);
|
|
243
255
|
|
|
244
256
|
acc.trackSubAgentExecution(
|
|
245
257
|
toolCallEvent("tc-dirty1", "task", "running", "r1", {
|
|
@@ -247,10 +259,10 @@ describe("MessageAccumulator tool call status transitions", () => {
|
|
|
247
259
|
}),
|
|
248
260
|
);
|
|
249
261
|
|
|
250
|
-
expect(acc.
|
|
262
|
+
expect(acc.isDirty).toBe(true);
|
|
251
263
|
});
|
|
252
264
|
|
|
253
|
-
it("
|
|
265
|
+
it("markPersisted clears the dirty flag, and a sub-agent update re-marks it", () => {
|
|
254
266
|
const acc = new MessageAccumulator([]);
|
|
255
267
|
|
|
256
268
|
acc.trackSubAgentExecution(
|
|
@@ -258,17 +270,17 @@ describe("MessageAccumulator tool call status transitions", () => {
|
|
|
258
270
|
args: { description: "Research", prompt: "Go" },
|
|
259
271
|
}),
|
|
260
272
|
);
|
|
261
|
-
expect(acc.
|
|
273
|
+
expect(acc.isDirty).toBe(true);
|
|
262
274
|
|
|
263
|
-
acc.
|
|
264
|
-
expect(acc.
|
|
275
|
+
acc.markPersisted();
|
|
276
|
+
expect(acc.isDirty).toBe(false);
|
|
265
277
|
|
|
266
278
|
// The completion transition (IN_PROGRESS -> COMPLETED) must re-mark dirty
|
|
267
279
|
// so the terminal sub-agent state is persisted to the live stream.
|
|
268
280
|
acc.trackSubAgentExecution(
|
|
269
281
|
toolCallEvent("tc-dirty2", "task", "completed", "r1", { result: "Done" }),
|
|
270
282
|
);
|
|
271
|
-
expect(acc.
|
|
283
|
+
expect(acc.isDirty).toBe(true);
|
|
272
284
|
expect(acc.subAgentExecutions[0].status).toBe(SubAgentStatus.SUB_AGENT_COMPLETED);
|
|
273
285
|
});
|
|
274
286
|
|
|
@@ -291,8 +303,8 @@ describe("MessageAccumulator tool call status transitions", () => {
|
|
|
291
303
|
toolCallEvent("tc-done", "task", "completed", "r1", { result: "Found it" }),
|
|
292
304
|
);
|
|
293
305
|
|
|
294
|
-
acc.
|
|
295
|
-
expect(acc.
|
|
306
|
+
acc.markPersisted();
|
|
307
|
+
expect(acc.isDirty).toBe(false);
|
|
296
308
|
|
|
297
309
|
acc.cancelInProgressSubAgents();
|
|
298
310
|
|
|
@@ -303,14 +315,14 @@ describe("MessageAccumulator tool call status transitions", () => {
|
|
|
303
315
|
expect(running.completedAt).not.toBe("");
|
|
304
316
|
expect(done.status).toBe(SubAgentStatus.SUB_AGENT_COMPLETED);
|
|
305
317
|
// The transition must mark dirty so the cancellation is persisted.
|
|
306
|
-
expect(acc.
|
|
318
|
+
expect(acc.isDirty).toBe(true);
|
|
307
319
|
});
|
|
308
320
|
|
|
309
321
|
it("cancelInProgressSubAgents is a no-op when there are no running sub-agents", () => {
|
|
310
322
|
const acc = new MessageAccumulator([]);
|
|
311
323
|
acc.cancelInProgressSubAgents();
|
|
312
324
|
expect(acc.subAgentExecutions).toHaveLength(0);
|
|
313
|
-
expect(acc.
|
|
325
|
+
expect(acc.isDirty).toBe(false);
|
|
314
326
|
});
|
|
315
327
|
|
|
316
328
|
it("sub-agent name extraction handles object subagentType", () => {
|
|
@@ -710,6 +722,106 @@ describe("MessageAccumulator tool call status transitions", () => {
|
|
|
710
722
|
});
|
|
711
723
|
});
|
|
712
724
|
|
|
725
|
+
// Issue #179: the live thinking/tool-call trace was starved because the Cursor
|
|
726
|
+
// loop's persist cadence had no force-flush for tool-call lifecycle. The
|
|
727
|
+
// accumulator's isDirty signal is now the force-flush source: it MUST fire on
|
|
728
|
+
// the discrete, user-visible events (tool start, tool finish, sub-agent
|
|
729
|
+
// delegation) and MUST NOT fire on high-frequency token deltas (assistant
|
|
730
|
+
// text, model thinking), which ride the StreamingUpdateScheduler time cadence.
|
|
731
|
+
describe("streaming force-flush signal (issue #179 cadence)", () => {
|
|
732
|
+
it("flags dirty the instant a tool call is created", () => {
|
|
733
|
+
const acc = new MessageAccumulator([]);
|
|
734
|
+
acc.processEvent(assistantEvent("r1", "Let me search."));
|
|
735
|
+
expect(acc.isDirty).toBe(false); // assistant text alone does not force-flush
|
|
736
|
+
|
|
737
|
+
acc.processEvent(toolCallEvent("tc-1", "Shell", "running", "r1"));
|
|
738
|
+
expect(acc.isDirty).toBe(true);
|
|
739
|
+
});
|
|
740
|
+
|
|
741
|
+
it("flags dirty when a tool call transitions to a terminal status", () => {
|
|
742
|
+
const acc = new MessageAccumulator([]);
|
|
743
|
+
acc.processEvent(assistantEvent("r1", "Running."));
|
|
744
|
+
acc.processEvent(toolCallEvent("tc-1", "Shell", "running", "r1"));
|
|
745
|
+
acc.markPersisted();
|
|
746
|
+
expect(acc.isDirty).toBe(false);
|
|
747
|
+
|
|
748
|
+
acc.processEvent(toolCallEvent("tc-1", "Shell", "completed", "r1", { result: "OK" }));
|
|
749
|
+
expect(acc.isDirty).toBe(true);
|
|
750
|
+
});
|
|
751
|
+
|
|
752
|
+
it("does NOT re-flag dirty on a redundant terminal re-emit", () => {
|
|
753
|
+
const acc = new MessageAccumulator([]);
|
|
754
|
+
acc.processEvent(assistantEvent("r1", "Running."));
|
|
755
|
+
acc.processEvent(toolCallEvent("tc-1", "read", "running", "r1"));
|
|
756
|
+
acc.processEvent(toolCallEvent("tc-1", "read", "completed", "r1", { result: "data" }));
|
|
757
|
+
acc.markPersisted();
|
|
758
|
+
expect(acc.isDirty).toBe(false);
|
|
759
|
+
|
|
760
|
+
// An already-terminal call re-emitting is noise, not a state change.
|
|
761
|
+
acc.processEvent(toolCallEvent("tc-1", "read", "completed", "r1", { result: "" }));
|
|
762
|
+
expect(acc.isDirty).toBe(false);
|
|
763
|
+
});
|
|
764
|
+
|
|
765
|
+
it("does NOT flag dirty on model thinking deltas (they ride the scheduler cadence)", () => {
|
|
766
|
+
const acc = new MessageAccumulator([]);
|
|
767
|
+
acc.processEvent(thinkingEvent("r1", "Let me reason about this"));
|
|
768
|
+
acc.processEvent(thinkingEvent("r1", " step by step..."));
|
|
769
|
+
expect(acc.isDirty).toBe(false);
|
|
770
|
+
});
|
|
771
|
+
|
|
772
|
+
it("does NOT flag dirty on assistant text deltas", () => {
|
|
773
|
+
const acc = new MessageAccumulator([]);
|
|
774
|
+
acc.processEvent(assistantEvent("r1", "Here is "));
|
|
775
|
+
acc.processEvent(assistantEvent("r1", "the answer."));
|
|
776
|
+
expect(acc.isDirty).toBe(false);
|
|
777
|
+
});
|
|
778
|
+
|
|
779
|
+
it("suppressed todo tools do NOT flag dirty (TodoTracker owns that signal)", () => {
|
|
780
|
+
const acc = new MessageAccumulator([]);
|
|
781
|
+
acc.processEvent(assistantEvent("r1", "Planning."));
|
|
782
|
+
acc.processEvent(toolCallEvent("tc-todo", "updateTodos", "running", "r1", {
|
|
783
|
+
args: { todos: [{ content: "Step 1", status: "pending" }] },
|
|
784
|
+
}));
|
|
785
|
+
expect(acc.isDirty).toBe(false);
|
|
786
|
+
});
|
|
787
|
+
|
|
788
|
+
it("markPersisted clears a tool-call dirty flag", () => {
|
|
789
|
+
const acc = new MessageAccumulator([]);
|
|
790
|
+
acc.processEvent(assistantEvent("r1", "Running."));
|
|
791
|
+
acc.processEvent(toolCallEvent("tc-1", "Shell", "running", "r1"));
|
|
792
|
+
expect(acc.isDirty).toBe(true);
|
|
793
|
+
|
|
794
|
+
acc.markPersisted();
|
|
795
|
+
expect(acc.isDirty).toBe(false);
|
|
796
|
+
});
|
|
797
|
+
|
|
798
|
+
// The user-facing symptom: a short thinking+tool turn (< 20 stream events).
|
|
799
|
+
// The old `eventCount % 20` gate never fired, so the whole trace landed only
|
|
800
|
+
// at the final persist. With the force-flush signal, the tool lifecycle is
|
|
801
|
+
// observable mid-turn — each discrete event leaves isDirty set for the loop.
|
|
802
|
+
it("a short thinking+tool turn produces force-flush points mid-turn", () => {
|
|
803
|
+
const acc = new MessageAccumulator([]);
|
|
804
|
+
const flushPoints: string[] = [];
|
|
805
|
+
const step = (label: string, ev: SDKMessage) => {
|
|
806
|
+
acc.processEvent(ev);
|
|
807
|
+
if (acc.isDirty) {
|
|
808
|
+
flushPoints.push(label);
|
|
809
|
+
acc.markPersisted();
|
|
810
|
+
}
|
|
811
|
+
};
|
|
812
|
+
|
|
813
|
+
step("thinking", thinkingEvent("r1", "Thinking about the task..."));
|
|
814
|
+
step("assistant", assistantEvent("r1", "I'll check the file."));
|
|
815
|
+
step("tool-start", toolCallEvent("tc-1", "read", "running", "r1"));
|
|
816
|
+
step("tool-done", toolCallEvent("tc-1", "read", "completed", "r1", { result: "contents" }));
|
|
817
|
+
step("assistant-2", assistantEvent("r1", "Done."));
|
|
818
|
+
|
|
819
|
+
// Tool start and tool completion are the discrete moments the live UI must
|
|
820
|
+
// see immediately; thinking/assistant deltas are carried by the scheduler.
|
|
821
|
+
expect(flushPoints).toEqual(["tool-start", "tool-done"]);
|
|
822
|
+
});
|
|
823
|
+
});
|
|
824
|
+
|
|
713
825
|
describe("cancelInProgressSubAgentProtos standalone", () => {
|
|
714
826
|
it("cancels IN_PROGRESS/PENDING protos in place and reports whether anything changed", () => {
|
|
715
827
|
const running = create(SubAgentExecutionSchema, {
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Unit tests for the Cursor streaming persist decision (issue #179).
|
|
3
|
+
*
|
|
4
|
+
* Validates the converged cadence: discrete force-flush signals persist
|
|
5
|
+
* immediately, while high-frequency token deltas ride the shared
|
|
6
|
+
* StreamingUpdateScheduler's time cadence (rather than the old, time-blind
|
|
7
|
+
* `eventCount % 20` gate that starved short turns).
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { describe, it, expect } from "vitest";
|
|
11
|
+
import {
|
|
12
|
+
StreamingUpdateScheduler,
|
|
13
|
+
type StreamingConfig,
|
|
14
|
+
} from "../../../shared/streaming-scheduler.js";
|
|
15
|
+
import { shouldPersistStreamingStatus } from "../persist-decision.js";
|
|
16
|
+
|
|
17
|
+
const CONFIG: StreamingConfig = {
|
|
18
|
+
minIntervalMs: 500,
|
|
19
|
+
maxIntervalMs: 5000,
|
|
20
|
+
burstThreshold: 50,
|
|
21
|
+
};
|
|
22
|
+
|
|
23
|
+
const NO_FORCE = {
|
|
24
|
+
deltaEnricherDirty: false,
|
|
25
|
+
todosDirty: false,
|
|
26
|
+
contentDirty: false,
|
|
27
|
+
} as const;
|
|
28
|
+
|
|
29
|
+
describe("shouldPersistStreamingStatus", () => {
|
|
30
|
+
it("persists on the first event via the scheduler (kills the opaque-wait at turn start)", () => {
|
|
31
|
+
const scheduler = new StreamingUpdateScheduler(CONFIG, 0);
|
|
32
|
+
expect(shouldPersistStreamingStatus(NO_FORCE, scheduler, 1, 0)).toBe(true);
|
|
33
|
+
});
|
|
34
|
+
|
|
35
|
+
it("does not persist a sub-500ms token delta with no force signal", () => {
|
|
36
|
+
const scheduler = new StreamingUpdateScheduler(CONFIG, 0);
|
|
37
|
+
scheduler.markUpdateSent(1, 0); // first flush already sent
|
|
38
|
+
|
|
39
|
+
// 100ms later, one more delta event — too soon for the time cadence.
|
|
40
|
+
expect(shouldPersistStreamingStatus(NO_FORCE, scheduler, 2, 100)).toBe(false);
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
it("persists a token delta once the 500ms cadence elapses", () => {
|
|
44
|
+
const scheduler = new StreamingUpdateScheduler(CONFIG, 0);
|
|
45
|
+
scheduler.markUpdateSent(1, 0);
|
|
46
|
+
|
|
47
|
+
expect(shouldPersistStreamingStatus(NO_FORCE, scheduler, 2, 500)).toBe(true);
|
|
48
|
+
});
|
|
49
|
+
|
|
50
|
+
it.each([
|
|
51
|
+
["contentDirty (tool call / sub-agent)", { ...NO_FORCE, contentDirty: true }],
|
|
52
|
+
["todosDirty", { ...NO_FORCE, todosDirty: true }],
|
|
53
|
+
["deltaEnricherDirty (shell output)", { ...NO_FORCE, deltaEnricherDirty: true }],
|
|
54
|
+
])("force-flushes immediately on %s, bypassing the time cadence", (_label, signals) => {
|
|
55
|
+
const scheduler = new StreamingUpdateScheduler(CONFIG, 0);
|
|
56
|
+
scheduler.markUpdateSent(1, 0);
|
|
57
|
+
|
|
58
|
+
// Only 1ms after the last flush: the scheduler alone would say no...
|
|
59
|
+
expect(shouldPersistStreamingStatus(NO_FORCE, scheduler, 2, 1)).toBe(false);
|
|
60
|
+
// ...but any discrete force signal flushes now.
|
|
61
|
+
expect(shouldPersistStreamingStatus(signals, scheduler, 2, 1)).toBe(true);
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
// The #179 starvation scenario: a short turn of < 20 events. Under the old
|
|
65
|
+
// `eventCount % 20` gate, a tool call appearing at event 3 (and completing at
|
|
66
|
+
// event 4) would never be persisted until the unconditional final flush. With
|
|
67
|
+
// the converged decision, the tool lifecycle force-flushes the instant it
|
|
68
|
+
// happens, regardless of how few events the turn has.
|
|
69
|
+
it("surfaces a tool call on a short (<20 event) turn", () => {
|
|
70
|
+
const scheduler = new StreamingUpdateScheduler(CONFIG, 0);
|
|
71
|
+
const flushes: number[] = [];
|
|
72
|
+
let clock = 0;
|
|
73
|
+
|
|
74
|
+
// Helper mirroring the loop: decide, then mark both clocks on a flush.
|
|
75
|
+
const tick = (eventCount: number, contentDirty: boolean) => {
|
|
76
|
+
const persist = shouldPersistStreamingStatus(
|
|
77
|
+
{ ...NO_FORCE, contentDirty },
|
|
78
|
+
scheduler,
|
|
79
|
+
eventCount,
|
|
80
|
+
clock,
|
|
81
|
+
);
|
|
82
|
+
if (persist) {
|
|
83
|
+
flushes.push(eventCount);
|
|
84
|
+
scheduler.markUpdateSent(eventCount, clock);
|
|
85
|
+
}
|
|
86
|
+
clock += 50; // 50ms between events — well under the 500ms floor
|
|
87
|
+
};
|
|
88
|
+
|
|
89
|
+
tick(1, false); // assistant delta — first-event flush
|
|
90
|
+
tick(2, false); // thinking delta — within 500ms, no flush
|
|
91
|
+
tick(3, true); // tool call starts — force-flush
|
|
92
|
+
tick(4, true); // tool call completes — force-flush
|
|
93
|
+
tick(5, false); // closing assistant delta — within 500ms, no flush
|
|
94
|
+
|
|
95
|
+
// The tool lifecycle (events 3 and 4) is observed live; event 1 is the
|
|
96
|
+
// first-event flush. Events 2 and 5 correctly ride the cadence.
|
|
97
|
+
expect(flushes).toEqual([1, 3, 4]);
|
|
98
|
+
});
|
|
99
|
+
});
|
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Unit tests for Cursor MCP image-result normalization.
|
|
3
|
+
*
|
|
4
|
+
* The Cursor SDK wraps an MCP tool result as
|
|
5
|
+
* { status, value: { content: [ { text:{text} }, { image:{ data, mimeType } } ] } }
|
|
6
|
+
* where image `data` is a Node Buffer-JSON ({ type:"Buffer", data:number[] }).
|
|
7
|
+
* The translator must re-emit that as the canonical top-level content-block
|
|
8
|
+
* array the shared persist-time offload consumes, so a screenshot lands as a
|
|
9
|
+
* renderable image ToolCallOutputRef instead of text/plain. These tests pin
|
|
10
|
+
* that normalization and confirm it flows end-to-end through the offload.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { describe, it, expect, vi } from "vitest";
|
|
14
|
+
import { create } from "@bufbuild/protobuf";
|
|
15
|
+
import { AgentExecutionStatusSchema } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/api_pb";
|
|
16
|
+
import { AgentMessageSchema } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/message_pb";
|
|
17
|
+
import type { SDKMessage } from "@cursor/sdk";
|
|
18
|
+
import type { ArtifactStorage } from "../../../shared/artifact-storage.js";
|
|
19
|
+
import {
|
|
20
|
+
offloadOversizedToolOutputs,
|
|
21
|
+
detectImagePayload,
|
|
22
|
+
} from "../../../shared/status-offload.js";
|
|
23
|
+
import {
|
|
24
|
+
toResultString,
|
|
25
|
+
canonicalizeImageResult,
|
|
26
|
+
buildToolCallProto,
|
|
27
|
+
MessageAccumulator,
|
|
28
|
+
} from "../message-translator.js";
|
|
29
|
+
import type { AgentMessage } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/message_pb";
|
|
30
|
+
|
|
31
|
+
// PNG signature + a little payload, the way the Cursor SDK serializes bytes.
|
|
32
|
+
const PNG_BYTES = [137, 80, 78, 71, 13, 10, 26, 10, 0, 0, 0, 13, 73, 72, 68, 82];
|
|
33
|
+
const PNG_BASE64 = Buffer.from(PNG_BYTES).toString("base64");
|
|
34
|
+
|
|
35
|
+
function cursorImageEnvelope(text = "App=com.example") {
|
|
36
|
+
return {
|
|
37
|
+
status: "success",
|
|
38
|
+
value: {
|
|
39
|
+
content: [
|
|
40
|
+
{ text: { text } },
|
|
41
|
+
{ image: { data: { type: "Buffer", data: PNG_BYTES }, mimeType: "image/png" } },
|
|
42
|
+
],
|
|
43
|
+
isError: false,
|
|
44
|
+
},
|
|
45
|
+
};
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
describe("canonicalizeImageResult", () => {
|
|
49
|
+
it("converts a Cursor image envelope (Buffer-JSON) to the canonical array", () => {
|
|
50
|
+
const out = canonicalizeImageResult(cursorImageEnvelope("App=Slack"));
|
|
51
|
+
expect(out).toBeDefined();
|
|
52
|
+
expect(JSON.parse(out!)).toEqual([
|
|
53
|
+
{ type: "text", text: "App=Slack" },
|
|
54
|
+
{ type: "image", data: PNG_BASE64, mimeType: "image/png" },
|
|
55
|
+
]);
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
it("handles a bare { content: [...] } envelope (no status/value wrapper)", () => {
|
|
59
|
+
const out = canonicalizeImageResult({
|
|
60
|
+
content: [{ image: { data: { type: "Buffer", data: PNG_BYTES }, mimeType: "image/png" } }],
|
|
61
|
+
});
|
|
62
|
+
expect(JSON.parse(out!)).toEqual([
|
|
63
|
+
{ type: "image", data: PNG_BASE64, mimeType: "image/png" },
|
|
64
|
+
]);
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
it("accepts an already-base64 image data string", () => {
|
|
68
|
+
const out = canonicalizeImageResult({
|
|
69
|
+
content: [{ image: { data: PNG_BASE64, mimeType: "image/png" } }],
|
|
70
|
+
});
|
|
71
|
+
expect(JSON.parse(out!)).toEqual([
|
|
72
|
+
{ type: "image", data: PNG_BASE64, mimeType: "image/png" },
|
|
73
|
+
]);
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
it("accepts a data: URL image data string", () => {
|
|
77
|
+
const out = canonicalizeImageResult({
|
|
78
|
+
content: [{ image: { data: `data:image/png;base64,${PNG_BASE64}`, mimeType: "image/png" } }],
|
|
79
|
+
});
|
|
80
|
+
expect(JSON.parse(out!)).toEqual([
|
|
81
|
+
{ type: "image", data: PNG_BASE64, mimeType: "image/png" },
|
|
82
|
+
]);
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
it("defaults mimeType to image/png when absent", () => {
|
|
86
|
+
const out = canonicalizeImageResult({
|
|
87
|
+
content: [{ image: { data: { type: "Buffer", data: PNG_BYTES } } }],
|
|
88
|
+
});
|
|
89
|
+
expect(JSON.parse(out!)[0]).toEqual({ type: "image", data: PNG_BASE64, mimeType: "image/png" });
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
it("returns undefined for a text-only envelope (no transformation)", () => {
|
|
93
|
+
expect(canonicalizeImageResult({ status: "success", value: { content: [{ text: { text: "hi" } }] } }))
|
|
94
|
+
.toBeUndefined();
|
|
95
|
+
});
|
|
96
|
+
|
|
97
|
+
it("returns undefined when there is no content array", () => {
|
|
98
|
+
expect(canonicalizeImageResult({ status: "success", value: { stdout: "ok" } })).toBeUndefined();
|
|
99
|
+
expect(canonicalizeImageResult("plain string")).toBeUndefined();
|
|
100
|
+
expect(canonicalizeImageResult(null)).toBeUndefined();
|
|
101
|
+
});
|
|
102
|
+
});
|
|
103
|
+
|
|
104
|
+
describe("toResultString", () => {
|
|
105
|
+
it("passes a string through unchanged", () => {
|
|
106
|
+
expect(toResultString("just logs")).toBe("just logs");
|
|
107
|
+
});
|
|
108
|
+
|
|
109
|
+
it("returns empty string for an absent result", () => {
|
|
110
|
+
expect(toResultString(null)).toBe("");
|
|
111
|
+
expect(toResultString(undefined)).toBe("");
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
it("JSON.stringifies a non-image object unchanged (text-only envelope)", () => {
|
|
115
|
+
const env = { status: "success", value: { content: [{ text: { text: "hi" } }] } };
|
|
116
|
+
expect(toResultString(env)).toBe(JSON.stringify(env));
|
|
117
|
+
});
|
|
118
|
+
|
|
119
|
+
it("normalizes an image envelope to the canonical array", () => {
|
|
120
|
+
const out = toResultString(cursorImageEnvelope("App=X"));
|
|
121
|
+
expect(JSON.parse(out)).toEqual([
|
|
122
|
+
{ type: "text", text: "App=X" },
|
|
123
|
+
{ type: "image", data: PNG_BASE64, mimeType: "image/png" },
|
|
124
|
+
]);
|
|
125
|
+
});
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
describe("buildToolCallProto image normalization", () => {
|
|
129
|
+
it("produces a result the shared offload detector recognizes as an image", () => {
|
|
130
|
+
const event = {
|
|
131
|
+
type: "tool_call",
|
|
132
|
+
agent_id: "a1",
|
|
133
|
+
run_id: "r1",
|
|
134
|
+
call_id: "tc-img",
|
|
135
|
+
name: "mcp",
|
|
136
|
+
status: "completed",
|
|
137
|
+
args: { providerIdentifier: "open-computer-use", toolName: "get_app_state", args: {} },
|
|
138
|
+
result: cursorImageEnvelope(),
|
|
139
|
+
} as unknown as Extract<SDKMessage, { type: "tool_call" }>;
|
|
140
|
+
|
|
141
|
+
const tc = buildToolCallProto(event);
|
|
142
|
+
expect(tc.name).toBe("get_app_state");
|
|
143
|
+
const img = detectImagePayload(tc.result);
|
|
144
|
+
expect(img).not.toBeNull();
|
|
145
|
+
expect(img?.mimeType).toBe("image/png");
|
|
146
|
+
expect(img?.base64).toBe(PNG_BASE64);
|
|
147
|
+
// The bloated Buffer-JSON must not survive into the persisted result.
|
|
148
|
+
expect(tc.result).not.toContain('"Buffer"');
|
|
149
|
+
});
|
|
150
|
+
});
|
|
151
|
+
|
|
152
|
+
describe("cursor image flows through the persist-time offload", () => {
|
|
153
|
+
it("offloads the screenshot as an image ref with no inline bytes", async () => {
|
|
154
|
+
const uploads: { key: string; contentType?: string }[] = [];
|
|
155
|
+
const storage: ArtifactStorage = {
|
|
156
|
+
upload: vi.fn(async (key: string, _content: Buffer, contentType?: string) => {
|
|
157
|
+
uploads.push({ key, contentType });
|
|
158
|
+
return key;
|
|
159
|
+
}),
|
|
160
|
+
getDownloadUrl: vi.fn(async (key: string) => `https://artifacts.local/${key}`),
|
|
161
|
+
exists: vi.fn(async () => true),
|
|
162
|
+
};
|
|
163
|
+
|
|
164
|
+
const event = {
|
|
165
|
+
type: "tool_call",
|
|
166
|
+
agent_id: "a1",
|
|
167
|
+
run_id: "r1",
|
|
168
|
+
call_id: "tc-img",
|
|
169
|
+
name: "mcp",
|
|
170
|
+
status: "completed",
|
|
171
|
+
args: { providerIdentifier: "open-computer-use", toolName: "get_app_state", args: {} },
|
|
172
|
+
result: cursorImageEnvelope(),
|
|
173
|
+
} as unknown as Extract<SDKMessage, { type: "tool_call" }>;
|
|
174
|
+
|
|
175
|
+
const tc = buildToolCallProto(event);
|
|
176
|
+
const status = create(AgentExecutionStatusSchema, {
|
|
177
|
+
messages: [create(AgentMessageSchema, { toolCalls: [tc] })],
|
|
178
|
+
});
|
|
179
|
+
|
|
180
|
+
await offloadOversizedToolOutputs(status, { artifactStorage: storage, executionId: "exec-1" });
|
|
181
|
+
|
|
182
|
+
const out = status.messages[0].toolCalls[0];
|
|
183
|
+
expect(out.outputRef).toBeDefined();
|
|
184
|
+
expect(out.outputRef!.isImage).toBe(true);
|
|
185
|
+
expect(out.outputRef!.mimeType).toBe("image/png");
|
|
186
|
+
expect(out.outputRef!.storageKey.endsWith(".png")).toBe(true);
|
|
187
|
+
expect(uploads[0]?.contentType).toBe("image/png");
|
|
188
|
+
// Inline result is collapsed; no base64/Buffer bytes remain in the status.
|
|
189
|
+
expect(out.result).not.toContain(PNG_BASE64);
|
|
190
|
+
expect(out.result).not.toContain('"Buffer"');
|
|
191
|
+
});
|
|
192
|
+
});
|
|
193
|
+
|
|
194
|
+
describe("sub-agent image normalization (extractConversationSteps)", () => {
|
|
195
|
+
it("normalizes a screenshot returned inside a sub-agent toolCall step", () => {
|
|
196
|
+
const messages: AgentMessage[] = [];
|
|
197
|
+
const acc = new MessageAccumulator(messages);
|
|
198
|
+
|
|
199
|
+
const running = {
|
|
200
|
+
type: "tool_call",
|
|
201
|
+
agent_id: "a1",
|
|
202
|
+
run_id: "r1",
|
|
203
|
+
call_id: "tc-sub-img",
|
|
204
|
+
name: "task",
|
|
205
|
+
status: "running",
|
|
206
|
+
args: { description: "screenshot", prompt: "capture" },
|
|
207
|
+
} as unknown as Extract<SDKMessage, { type: "tool_call" }>;
|
|
208
|
+
acc.processEvent(running);
|
|
209
|
+
acc.trackSubAgentExecution(running);
|
|
210
|
+
|
|
211
|
+
const completed = {
|
|
212
|
+
type: "tool_call",
|
|
213
|
+
agent_id: "a1",
|
|
214
|
+
run_id: "r1",
|
|
215
|
+
call_id: "tc-sub-img",
|
|
216
|
+
name: "task",
|
|
217
|
+
status: "completed",
|
|
218
|
+
args: { description: "screenshot", prompt: "capture" },
|
|
219
|
+
result: {
|
|
220
|
+
status: "success",
|
|
221
|
+
value: {
|
|
222
|
+
conversationSteps: [
|
|
223
|
+
{
|
|
224
|
+
type: "toolCall",
|
|
225
|
+
message: {
|
|
226
|
+
type: "get_app_state",
|
|
227
|
+
args: {},
|
|
228
|
+
result: cursorImageEnvelope("App=SubAgent"),
|
|
229
|
+
},
|
|
230
|
+
},
|
|
231
|
+
],
|
|
232
|
+
},
|
|
233
|
+
},
|
|
234
|
+
} as unknown as Extract<SDKMessage, { type: "tool_call" }>;
|
|
235
|
+
|
|
236
|
+
acc.processEvent(completed);
|
|
237
|
+
acc.trackSubAgentExecution(completed);
|
|
238
|
+
|
|
239
|
+
const sub = acc.subAgentExecutions[0];
|
|
240
|
+
const subToolResult = sub.messages[0].toolCalls[0].result;
|
|
241
|
+
const img = detectImagePayload(subToolResult);
|
|
242
|
+
expect(img?.base64).toBe(PNG_BASE64);
|
|
243
|
+
});
|
|
244
|
+
});
|
|
@@ -43,9 +43,12 @@ function freshRoot(): string {
|
|
|
43
43
|
const stigmerScript = (root: string) =>
|
|
44
44
|
join(root, ".stigmer", "sessions", "ses-1", "hitl", "stigmer-approval.sh");
|
|
45
45
|
|
|
46
|
+
// Single preToolUse registration — the common shape in these tests.
|
|
47
|
+
const pre = (scriptPath: string) => [{ event: "preToolUse", scriptPath }];
|
|
48
|
+
|
|
46
49
|
describe("buildMergedConfig", () => {
|
|
47
50
|
it("writes a standalone config and restores by delete when no hooks.json exists", () => {
|
|
48
|
-
const { merged, restoreTo } = buildMergedConfig(null, "/abs/hitl/stigmer-approval.sh");
|
|
51
|
+
const { merged, restoreTo } = buildMergedConfig(null, pre("/abs/hitl/stigmer-approval.sh"));
|
|
49
52
|
const parsed = JSON.parse(merged);
|
|
50
53
|
expect(parsed.hooks.preToolUse).toHaveLength(1);
|
|
51
54
|
expect(parsed.hooks.preToolUse[0].command).toBe("/abs/hitl/stigmer-approval.sh");
|
|
@@ -54,6 +57,47 @@ describe("buildMergedConfig", () => {
|
|
|
54
57
|
expect(restoreTo).toBeNull();
|
|
55
58
|
});
|
|
56
59
|
|
|
60
|
+
it("registers multiple events (preToolUse + beforeMCPExecution) and restores by delete", () => {
|
|
61
|
+
const { merged, restoreTo } = buildMergedConfig(null, [
|
|
62
|
+
{ event: "preToolUse", scriptPath: "/abs/hitl/stigmer-approval.sh" },
|
|
63
|
+
{ event: "beforeMCPExecution", scriptPath: "/abs/hitl/stigmer-mcp-capture.sh" },
|
|
64
|
+
]);
|
|
65
|
+
const parsed = JSON.parse(merged);
|
|
66
|
+
expect(parsed.hooks.preToolUse[0].command).toBe("/abs/hitl/stigmer-approval.sh");
|
|
67
|
+
expect(parsed.hooks.beforeMCPExecution[0].command).toBe("/abs/hitl/stigmer-mcp-capture.sh");
|
|
68
|
+
expect(parsed.hooks.beforeMCPExecution[0].failClosed).toBe(true);
|
|
69
|
+
expect(restoreTo).toBeNull();
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
it("merges into both event arrays and strips stale Stigmer entries from each on restore", () => {
|
|
73
|
+
const root = "/abs";
|
|
74
|
+
const stalePre = join(root, ".stigmer", "sessions", "ses-1", "hitl", "stigmer-approval.sh");
|
|
75
|
+
const staleMcp = join(root, ".stigmer", "sessions", "ses-1", "hitl", "stigmer-mcp-capture.sh");
|
|
76
|
+
const original = JSON.stringify({
|
|
77
|
+
version: 1,
|
|
78
|
+
hooks: {
|
|
79
|
+
preToolUse: [{ command: "./user.sh" }, { command: stalePre, failClosed: true }],
|
|
80
|
+
beforeMCPExecution: [{ command: staleMcp, failClosed: true }],
|
|
81
|
+
},
|
|
82
|
+
});
|
|
83
|
+
const freshPre = join(root, ".stigmer", "sessions", "ses-2", "hitl", "stigmer-approval.sh");
|
|
84
|
+
const freshMcp = join(root, ".stigmer", "sessions", "ses-2", "hitl", "stigmer-mcp-capture.sh");
|
|
85
|
+
|
|
86
|
+
const { merged, restoreTo } = buildMergedConfig(original, [
|
|
87
|
+
{ event: "preToolUse", scriptPath: freshPre },
|
|
88
|
+
{ event: "beforeMCPExecution", scriptPath: freshMcp },
|
|
89
|
+
]);
|
|
90
|
+
|
|
91
|
+
const m = JSON.parse(merged);
|
|
92
|
+
expect(m.hooks.preToolUse.map((e: any) => e.command)).toEqual(["./user.sh", freshPre]);
|
|
93
|
+
expect(m.hooks.beforeMCPExecution.map((e: any) => e.command)).toEqual([freshMcp]);
|
|
94
|
+
|
|
95
|
+
// Restore is self-healing: every stale Stigmer entry is removed from both.
|
|
96
|
+
const r = JSON.parse(restoreTo!);
|
|
97
|
+
expect(r.hooks.preToolUse).toEqual([{ command: "./user.sh" }]);
|
|
98
|
+
expect(r.hooks.beforeMCPExecution).toEqual([]);
|
|
99
|
+
});
|
|
100
|
+
|
|
57
101
|
it("merges with a user's hooks.json and restores the original bytes verbatim", () => {
|
|
58
102
|
const original = JSON.stringify(
|
|
59
103
|
{
|
|
@@ -67,7 +111,7 @@ describe("buildMergedConfig", () => {
|
|
|
67
111
|
2,
|
|
68
112
|
);
|
|
69
113
|
const script = "/abs/.stigmer/sessions/ses-1/hitl/stigmer-approval.sh";
|
|
70
|
-
const { merged, restoreTo } = buildMergedConfig(original, script);
|
|
114
|
+
const { merged, restoreTo } = buildMergedConfig(original, pre(script));
|
|
71
115
|
const parsed = JSON.parse(merged);
|
|
72
116
|
|
|
73
117
|
// Our entry is appended; the user's preToolUse hook is preserved...
|
|
@@ -93,7 +137,7 @@ describe("buildMergedConfig", () => {
|
|
|
93
137
|
},
|
|
94
138
|
});
|
|
95
139
|
const fresh = join(root, ".stigmer", "sessions", "ses-2", "hitl", "stigmer-approval.sh");
|
|
96
|
-
const { merged, restoreTo } = buildMergedConfig(original, fresh);
|
|
140
|
+
const { merged, restoreTo } = buildMergedConfig(original, pre(fresh));
|
|
97
141
|
|
|
98
142
|
const mergedParsed = JSON.parse(merged);
|
|
99
143
|
// No duplicate: user entry + exactly one fresh Stigmer entry.
|
|
@@ -109,7 +153,7 @@ describe("buildMergedConfig", () => {
|
|
|
109
153
|
|
|
110
154
|
it("replaces an unparseable hooks.json for the turn but restores its exact bytes", () => {
|
|
111
155
|
const garbage = "{ this is not json ";
|
|
112
|
-
const { merged, restoreTo } = buildMergedConfig(garbage, "/abs/hitl/stigmer-approval.sh");
|
|
156
|
+
const { merged, restoreTo } = buildMergedConfig(garbage, pre("/abs/hitl/stigmer-approval.sh"));
|
|
113
157
|
// We still install a working gate for the turn...
|
|
114
158
|
expect(JSON.parse(merged).hooks.preToolUse).toHaveLength(1);
|
|
115
159
|
// ...and never "fix" the user's file: restore their exact original bytes.
|
|
@@ -147,6 +191,11 @@ describe("installHitlGate / removeHitlGate", () => {
|
|
|
147
191
|
const command = hooksJson.hooks.preToolUse[0].command;
|
|
148
192
|
expect(command).toBe(join(hitlDir, "stigmer-approval.sh"));
|
|
149
193
|
expect(command.startsWith("/")).toBe(true);
|
|
194
|
+
// The SAME script gates MCP via beforeMCPExecution (preToolUse does not
|
|
195
|
+
// enforce MCP); the script branches internally on hook_event_name.
|
|
196
|
+
expect(hooksJson.hooks.beforeMCPExecution[0].command).toBe(
|
|
197
|
+
join(hitlDir, "stigmer-approval.sh"),
|
|
198
|
+
);
|
|
150
199
|
// The workspace holds no relocated artifacts.
|
|
151
200
|
expect(existsSync(join(workspaceRoot, ".cursor", "hooks"))).toBe(false);
|
|
152
201
|
});
|