@stigmer/runner 3.10.0 → 3.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (260) hide show
  1. package/README.md +12 -1
  2. package/dist/.build-fingerprint +1 -1
  3. package/dist/activities/call-llm.js +9 -10
  4. package/dist/activities/call-llm.js.map +1 -1
  5. package/dist/activities/classify-tool-approvals.d.ts +2 -1
  6. package/dist/activities/classify-tool-approvals.js +28 -2
  7. package/dist/activities/classify-tool-approvals.js.map +1 -1
  8. package/dist/activities/discover-mcp-server.d.ts +32 -0
  9. package/dist/activities/discover-mcp-server.js +162 -27
  10. package/dist/activities/discover-mcp-server.js.map +1 -1
  11. package/dist/activities/execute-cursor/__test-utils__/cursor-hook-harness.d.ts +8 -0
  12. package/dist/activities/execute-cursor/__test-utils__/cursor-hook-harness.js +1 -1
  13. package/dist/activities/execute-cursor/__test-utils__/cursor-hook-harness.js.map +1 -1
  14. package/dist/activities/execute-cursor/approval-state.d.ts +28 -2
  15. package/dist/activities/execute-cursor/approval-state.js +7 -1
  16. package/dist/activities/execute-cursor/approval-state.js.map +1 -1
  17. package/dist/activities/execute-cursor/attachment-resolver.d.ts +14 -0
  18. package/dist/activities/execute-cursor/attachment-resolver.js +18 -4
  19. package/dist/activities/execute-cursor/attachment-resolver.js.map +1 -1
  20. package/dist/activities/execute-cursor/blueprint-resolver.d.ts +1 -9
  21. package/dist/activities/execute-cursor/blueprint-resolver.js +6 -22
  22. package/dist/activities/execute-cursor/blueprint-resolver.js.map +1 -1
  23. package/dist/activities/execute-cursor/env-resolver.js +3 -1
  24. package/dist/activities/execute-cursor/env-resolver.js.map +1 -1
  25. package/dist/activities/execute-cursor/error-classifier.d.ts +40 -3
  26. package/dist/activities/execute-cursor/error-classifier.js +81 -3
  27. package/dist/activities/execute-cursor/error-classifier.js.map +1 -1
  28. package/dist/activities/execute-cursor/extract-structured-output.d.ts +29 -0
  29. package/dist/activities/execute-cursor/extract-structured-output.js +58 -0
  30. package/dist/activities/execute-cursor/extract-structured-output.js.map +1 -0
  31. package/dist/activities/execute-cursor/hook-script.d.ts +14 -3
  32. package/dist/activities/execute-cursor/hook-script.js +72 -10
  33. package/dist/activities/execute-cursor/hook-script.js.map +1 -1
  34. package/dist/activities/execute-cursor/index.d.ts +5 -1
  35. package/dist/activities/execute-cursor/index.js +51 -57
  36. package/dist/activities/execute-cursor/index.js.map +1 -1
  37. package/dist/activities/execute-cursor/mcp-resolver.d.ts +24 -1
  38. package/dist/activities/execute-cursor/mcp-resolver.js +5 -2
  39. package/dist/activities/execute-cursor/mcp-resolver.js.map +1 -1
  40. package/dist/activities/execute-cursor/prompt-builder.d.ts +18 -4
  41. package/dist/activities/execute-cursor/prompt-builder.js +12 -7
  42. package/dist/activities/execute-cursor/prompt-builder.js.map +1 -1
  43. package/dist/activities/execute-cursor/turn-stream.js +4 -1
  44. package/dist/activities/execute-cursor/turn-stream.js.map +1 -1
  45. package/dist/activities/execute-deep-agent/attachment-injector.d.ts +18 -1
  46. package/dist/activities/execute-deep-agent/attachment-injector.js +68 -23
  47. package/dist/activities/execute-deep-agent/attachment-injector.js.map +1 -1
  48. package/dist/activities/execute-deep-agent/environment.js +3 -1
  49. package/dist/activities/execute-deep-agent/environment.js.map +1 -1
  50. package/dist/activities/execute-deep-agent/index.js +15 -0
  51. package/dist/activities/execute-deep-agent/index.js.map +1 -1
  52. package/dist/activities/execute-deep-agent/prompt-builder.d.ts +7 -7
  53. package/dist/activities/execute-deep-agent/prompt-builder.js +8 -2
  54. package/dist/activities/execute-deep-agent/prompt-builder.js.map +1 -1
  55. package/dist/activities/execute-deep-agent/setup.d.ts +10 -0
  56. package/dist/activities/execute-deep-agent/setup.js +55 -23
  57. package/dist/activities/execute-deep-agent/setup.js.map +1 -1
  58. package/dist/activities/execute-deep-agent/subagent-transformer.d.ts +18 -1
  59. package/dist/activities/execute-deep-agent/subagent-transformer.js +8 -1
  60. package/dist/activities/execute-deep-agent/subagent-transformer.js.map +1 -1
  61. package/dist/activities/execute-deep-agent/subagent-wiring.d.ts +11 -4
  62. package/dist/activities/execute-deep-agent/subagent-wiring.js +13 -4
  63. package/dist/activities/execute-deep-agent/subagent-wiring.js.map +1 -1
  64. package/dist/activities/hydrate-workflow-execution.js +3 -1
  65. package/dist/activities/hydrate-workflow-execution.js.map +1 -1
  66. package/dist/activities/workflow-event-activities.d.ts +28 -10
  67. package/dist/activities/workflow-event-activities.js +87 -58
  68. package/dist/activities/workflow-event-activities.js.map +1 -1
  69. package/dist/claimcheck/payload-codec.js +21 -1
  70. package/dist/claimcheck/payload-codec.js.map +1 -1
  71. package/dist/client/stigmer-client.d.ts +9 -4
  72. package/dist/client/stigmer-client.js +28 -15
  73. package/dist/client/stigmer-client.js.map +1 -1
  74. package/dist/encryption/config.d.ts +32 -0
  75. package/dist/encryption/config.js +68 -0
  76. package/dist/encryption/config.js.map +1 -0
  77. package/dist/encryption/index.d.ts +3 -0
  78. package/dist/encryption/index.js +3 -0
  79. package/dist/encryption/index.js.map +1 -0
  80. package/dist/encryption/payload-codec.d.ts +41 -0
  81. package/dist/encryption/payload-codec.js +130 -0
  82. package/dist/encryption/payload-codec.js.map +1 -0
  83. package/dist/payload-codecs.d.ts +16 -0
  84. package/dist/payload-codecs.js +38 -0
  85. package/dist/payload-codecs.js.map +1 -0
  86. package/dist/preflight.d.ts +31 -0
  87. package/dist/preflight.js +43 -0
  88. package/dist/preflight.js.map +1 -1
  89. package/dist/runner-manager.js +5 -15
  90. package/dist/runner-manager.js.map +1 -1
  91. package/dist/runner.js +5 -16
  92. package/dist/runner.js.map +1 -1
  93. package/dist/shared/approval-policy.d.ts +9 -3
  94. package/dist/shared/approval-policy.js +15 -6
  95. package/dist/shared/approval-policy.js.map +1 -1
  96. package/dist/shared/attachment-naming.d.ts +53 -0
  97. package/dist/shared/attachment-naming.js +59 -0
  98. package/dist/shared/attachment-naming.js.map +1 -0
  99. package/dist/shared/caller-identity.d.ts +23 -2
  100. package/dist/shared/caller-identity.js +36 -5
  101. package/dist/shared/caller-identity.js.map +1 -1
  102. package/dist/shared/channel-attachment.js +1 -0
  103. package/dist/shared/channel-attachment.js.map +1 -1
  104. package/dist/shared/checkpointer/http-saver.d.ts +42 -1
  105. package/dist/shared/checkpointer/http-saver.js +96 -8
  106. package/dist/shared/checkpointer/http-saver.js.map +1 -1
  107. package/dist/shared/conversation-attachment.js +1 -0
  108. package/dist/shared/conversation-attachment.js.map +1 -1
  109. package/dist/shared/datastore-attachment.d.ts +50 -7
  110. package/dist/shared/datastore-attachment.js +93 -11
  111. package/dist/shared/datastore-attachment.js.map +1 -1
  112. package/dist/shared/http-retry.d.ts +43 -0
  113. package/dist/shared/http-retry.js +50 -0
  114. package/dist/shared/http-retry.js.map +1 -0
  115. package/dist/shared/llm-backend.d.ts +275 -0
  116. package/dist/shared/llm-backend.js +425 -0
  117. package/dist/shared/llm-backend.js.map +1 -0
  118. package/dist/shared/llm-proxy.d.ts +8 -0
  119. package/dist/shared/llm-proxy.js +15 -0
  120. package/dist/shared/llm-proxy.js.map +1 -1
  121. package/dist/shared/mcp-enabled-tools.d.ts +57 -0
  122. package/dist/shared/mcp-enabled-tools.js +86 -0
  123. package/dist/shared/mcp-enabled-tools.js.map +1 -0
  124. package/dist/shared/mcp-manager.d.ts +3 -1
  125. package/dist/shared/mcp-manager.js +17 -4
  126. package/dist/shared/mcp-manager.js.map +1 -1
  127. package/dist/shared/mcp-resolver.d.ts +39 -2
  128. package/dist/shared/mcp-resolver.js +38 -2
  129. package/dist/shared/mcp-resolver.js.map +1 -1
  130. package/dist/shared/model-client.d.ts +12 -5
  131. package/dist/shared/model-client.js +138 -18
  132. package/dist/shared/model-client.js.map +1 -1
  133. package/dist/shared/model-error.js +198 -5
  134. package/dist/shared/model-error.js.map +1 -1
  135. package/dist/shared/plan-mode-permissions.d.ts +26 -0
  136. package/dist/shared/plan-mode-permissions.js +28 -0
  137. package/dist/shared/plan-mode-permissions.js.map +1 -0
  138. package/dist/worker.d.ts +2 -1
  139. package/dist/worker.js +2 -4
  140. package/dist/worker.js.map +1 -1
  141. package/dist/workflow-engine/types.d.ts +18 -0
  142. package/dist/workflow-engine/types.js.map +1 -1
  143. package/dist/workflows/call-agent-orchestrator.d.ts +9 -0
  144. package/dist/workflows/call-agent-orchestrator.js +1 -0
  145. package/dist/workflows/call-agent-orchestrator.js.map +1 -1
  146. package/dist/workflows/connect-mcp-server.js +7 -0
  147. package/dist/workflows/connect-mcp-server.js.map +1 -1
  148. package/dist/workflows/engine-core.js +23 -2
  149. package/dist/workflows/engine-core.js.map +1 -1
  150. package/dist/workflows/execute-from-execution.d.ts +1 -1
  151. package/dist/workflows/execute-from-execution.js +11 -1
  152. package/dist/workflows/execute-from-execution.js.map +1 -1
  153. package/package.json +8 -2
  154. package/src/__tests__/claimcheck-codec.test.ts +36 -0
  155. package/src/__tests__/encryption-codec.test.ts +234 -0
  156. package/src/__tests__/fixtures/encrypted-payload-fixture.json +15 -0
  157. package/src/__tests__/history-encryption-e2e.test.ts +243 -0
  158. package/src/__tests__/preflight.test.ts +50 -2
  159. package/src/activities/__tests__/call-llm.test.ts +75 -0
  160. package/src/activities/__tests__/classify-tool-approvals.test.ts +117 -1
  161. package/src/activities/__tests__/discover-mcp-server.hang.test.ts +103 -0
  162. package/src/activities/__tests__/discover-mcp-server.test.ts +203 -0
  163. package/src/activities/__tests__/workflow-event-activities.test.ts +107 -8
  164. package/src/activities/call-llm.ts +9 -16
  165. package/src/activities/classify-tool-approvals.ts +34 -4
  166. package/src/activities/discover-mcp-server.ts +190 -32
  167. package/src/activities/execute-cursor/__test-utils__/cursor-hook-harness.ts +9 -0
  168. package/src/activities/execute-cursor/__tests__/approval-gate.test.ts +14 -0
  169. package/src/activities/execute-cursor/__tests__/attachment-resolver.test.ts +53 -0
  170. package/src/activities/execute-cursor/__tests__/build-prompt.test.ts +40 -14
  171. package/src/activities/execute-cursor/__tests__/error-classifier-extraction.test.ts +208 -0
  172. package/src/activities/execute-cursor/__tests__/extract-structured-output.test.ts +120 -0
  173. package/src/activities/execute-cursor/__tests__/hook-script.test.ts +93 -0
  174. package/src/activities/execute-cursor/__tests__/mcp-resolver.test.ts +125 -0
  175. package/src/activities/execute-cursor/__tests__/prompt-builder-delegation.test.ts +1 -1
  176. package/src/activities/execute-cursor/__tests__/turn-stream.test.ts +13 -0
  177. package/src/activities/execute-cursor/approval-state.ts +30 -1
  178. package/src/activities/execute-cursor/attachment-resolver.ts +31 -3
  179. package/src/activities/execute-cursor/blueprint-resolver.ts +7 -27
  180. package/src/activities/execute-cursor/env-resolver.ts +3 -1
  181. package/src/activities/execute-cursor/error-classifier.ts +91 -4
  182. package/src/activities/execute-cursor/extract-structured-output.ts +72 -0
  183. package/src/activities/execute-cursor/hook-script.ts +74 -10
  184. package/src/activities/execute-cursor/index.ts +55 -71
  185. package/src/activities/execute-cursor/mcp-resolver.ts +36 -2
  186. package/src/activities/execute-cursor/prompt-builder.ts +34 -9
  187. package/src/activities/execute-cursor/turn-stream.ts +5 -2
  188. package/src/activities/execute-deep-agent/__tests__/attachment-injector.test.ts +110 -8
  189. package/src/activities/execute-deep-agent/__tests__/datastore-degradation.test.ts +104 -0
  190. package/src/activities/execute-deep-agent/__tests__/hitl-reject.test.ts +2 -0
  191. package/src/activities/execute-deep-agent/__tests__/hitl-resume-approve-all.test.ts +1 -0
  192. package/src/activities/execute-deep-agent/__tests__/hitl-resume-history.test.ts +1 -0
  193. package/src/activities/execute-deep-agent/__tests__/prompt-builder.test.ts +34 -5
  194. package/src/activities/execute-deep-agent/__tests__/sequential-gate-resume.test.ts +1 -0
  195. package/src/activities/execute-deep-agent/__tests__/subagent-plan-mode-permissions.test.ts +173 -0
  196. package/src/activities/execute-deep-agent/__tests__/subagent-wiring.test.ts +12 -7
  197. package/src/activities/execute-deep-agent/attachment-injector.ts +94 -30
  198. package/src/activities/execute-deep-agent/environment.ts +3 -1
  199. package/src/activities/execute-deep-agent/index.ts +20 -0
  200. package/src/activities/execute-deep-agent/prompt-builder.ts +20 -10
  201. package/src/activities/execute-deep-agent/setup.ts +76 -28
  202. package/src/activities/execute-deep-agent/subagent-transformer.ts +23 -1
  203. package/src/activities/execute-deep-agent/subagent-wiring.ts +14 -4
  204. package/src/activities/hydrate-workflow-execution.ts +3 -1
  205. package/src/activities/workflow-event-activities.ts +96 -69
  206. package/src/claimcheck/payload-codec.ts +33 -1
  207. package/src/client/__tests__/stigmer-client.test.ts +8 -8
  208. package/src/client/stigmer-client.ts +32 -18
  209. package/src/encryption/config.ts +91 -0
  210. package/src/encryption/index.ts +3 -0
  211. package/src/encryption/payload-codec.ts +152 -0
  212. package/src/payload-codecs.ts +56 -0
  213. package/src/preflight.ts +45 -0
  214. package/src/runner-manager.ts +6 -24
  215. package/src/runner.ts +6 -25
  216. package/src/shared/__tests__/approval-policy.test.ts +82 -39
  217. package/src/shared/__tests__/attachment-naming.test.ts +159 -0
  218. package/src/shared/__tests__/bedrock-adapter.test.ts +213 -0
  219. package/src/shared/__tests__/bedrock-seam.test.ts +390 -0
  220. package/src/shared/__tests__/caller-identity.test.ts +25 -0
  221. package/src/shared/__tests__/channel-attachment.test.ts +1 -1
  222. package/src/shared/__tests__/connect-backfill.test.ts +1 -0
  223. package/src/shared/__tests__/conversation-attachment.test.ts +1 -1
  224. package/src/shared/__tests__/datastore-attachment.test.ts +129 -1
  225. package/src/shared/__tests__/foundry-adapter.test.ts +276 -0
  226. package/src/shared/__tests__/foundry-seam.test.ts +482 -0
  227. package/src/shared/__tests__/http-retry.test.ts +67 -0
  228. package/src/shared/__tests__/llm-backend.test.ts +616 -0
  229. package/src/shared/__tests__/mcp-enabled-tools.test.ts +86 -0
  230. package/src/shared/__tests__/mcp-manager.test.ts +84 -2
  231. package/src/shared/__tests__/mcp-resolver.test.ts +146 -3
  232. package/src/shared/__tests__/model-client.test.ts +154 -0
  233. package/src/shared/__tests__/model-error.test.ts +289 -1
  234. package/src/shared/__tests__/synthesized-attachment.test.ts +1 -0
  235. package/src/shared/__tests__/vertex-adapter.test.ts +169 -0
  236. package/src/shared/__tests__/vertex-seam.test.ts +295 -0
  237. package/src/shared/approval-policy.ts +14 -7
  238. package/src/shared/attachment-naming.ts +78 -0
  239. package/src/shared/caller-identity.ts +40 -5
  240. package/src/shared/channel-attachment.ts +1 -0
  241. package/src/shared/checkpointer/__tests__/http-saver.test.ts +196 -1
  242. package/src/shared/checkpointer/http-saver.ts +117 -9
  243. package/src/shared/conversation-attachment.ts +1 -0
  244. package/src/shared/datastore-attachment.ts +106 -11
  245. package/src/shared/http-retry.ts +50 -0
  246. package/src/shared/llm-backend.ts +544 -0
  247. package/src/shared/llm-proxy.ts +15 -0
  248. package/src/shared/mcp-enabled-tools.ts +105 -0
  249. package/src/shared/mcp-manager.ts +21 -4
  250. package/src/shared/mcp-resolver.ts +73 -2
  251. package/src/shared/model-client.ts +161 -19
  252. package/src/shared/model-error.ts +222 -4
  253. package/src/shared/plan-mode-permissions.ts +30 -0
  254. package/src/worker.ts +4 -5
  255. package/src/workflow-engine/types.ts +18 -0
  256. package/src/workflows/__tests__/execute-serverless-workflow.test.ts +68 -2
  257. package/src/workflows/call-agent-orchestrator.ts +10 -0
  258. package/src/workflows/connect-mcp-server.ts +7 -0
  259. package/src/workflows/engine-core.ts +23 -2
  260. package/src/workflows/execute-from-execution.ts +12 -2
@@ -0,0 +1,125 @@
1
+ import { describe, it, expect, vi, beforeEach } from "vitest";
2
+ import { resolveMcpServers } from "../mcp-resolver.js";
3
+
4
+ function makeUsage(
5
+ slug: string,
6
+ enabledTools: string[] = [],
7
+ toolApprovalOverrides: Array<{ toolName: string; requiresApproval: boolean }> = [],
8
+ ) {
9
+ return {
10
+ mcpServerRef: { slug, org: "test-org", kind: 0 },
11
+ enabledTools,
12
+ toolApprovalOverrides,
13
+ } as any;
14
+ }
15
+
16
+ function httpMcpServer(slug: string, defaultEnabledTools: string[] = []) {
17
+ return {
18
+ metadata: { id: `id-${slug}`, slug },
19
+ spec: {
20
+ serverType: { case: "http", value: { url: "https://mcp.example.com/mcp", headers: {} } },
21
+ env: {},
22
+ defaultEnabledTools,
23
+ },
24
+ status: undefined,
25
+ } as any;
26
+ }
27
+
28
+ function clientReturning(serversBySlug: Record<string, unknown>) {
29
+ return {
30
+ getMcpServerByReference: vi.fn(async (ref: { slug: string }) => {
31
+ const server = serversBySlug[ref.slug];
32
+ if (!server) throw new Error(`not found: ${ref.slug}`);
33
+ return server;
34
+ }),
35
+ } as any;
36
+ }
37
+
38
+ // The effective-list semantics live in shared/mcp-enabled-tools.ts (tested
39
+ // there); these tests pin the CURSOR resolver's threading — the near-duplicate
40
+ // of shared/mcp-resolver.ts that must mirror it until oss#387 consolidates.
41
+ describe("resolveMcpServers (cursor) — enabled_tools threading (issue #350)", () => {
42
+ beforeEach(() => {
43
+ vi.restoreAllMocks();
44
+ vi.spyOn(console, "warn").mockImplementation(() => {});
45
+ });
46
+
47
+ it("carries the usage's enabled_tools as the effective allow-list", async () => {
48
+ const client = clientReturning({ github: httpMcpServer("github") });
49
+
50
+ const result = await resolveMcpServers(
51
+ client, [makeUsage("github", ["create_pr"])], {}, "stdio-forbidden",
52
+ );
53
+
54
+ expect(result.resolvedServers[0].enabledTools).toEqual(["create_pr"]);
55
+ });
56
+
57
+ it("falls back to default_enabled_tools for an empty usage list", async () => {
58
+ const client = clientReturning({
59
+ github: httpMcpServer("github", ["search_code"]),
60
+ });
61
+
62
+ const result = await resolveMcpServers(
63
+ client, [makeUsage("github")], {}, "stdio-forbidden",
64
+ );
65
+
66
+ expect(result.resolvedServers[0].enabledTools).toEqual(["search_code"]);
67
+ });
68
+
69
+ it("resolves unrestricted (absent field) when both lists are empty", async () => {
70
+ const client = clientReturning({ github: httpMcpServer("github") });
71
+
72
+ const result = await resolveMcpServers(
73
+ client, [makeUsage("github")], {}, "stdio-forbidden",
74
+ );
75
+
76
+ expect(result.resolvedServers[0].enabledTools).toBeUndefined();
77
+ });
78
+
79
+ it("never narrows the Cursor SDK config — the SDK has no allow-list field; enforcement is the hook's disabled arm", async () => {
80
+ const client = clientReturning({ github: httpMcpServer("github") });
81
+
82
+ const result = await resolveMcpServers(
83
+ client, [makeUsage("github", ["create_pr"])], {}, "stdio-forbidden",
84
+ );
85
+
86
+ expect(result.cursorConfig.github).toEqual({
87
+ type: "http",
88
+ url: "https://mcp.example.com/mcp",
89
+ headers: undefined,
90
+ });
91
+ });
92
+ });
93
+
94
+ describe("resolveMcpServers (cursor) — tool_approval_overrides threading (issue #349)", () => {
95
+ beforeEach(() => {
96
+ vi.restoreAllMocks();
97
+ vi.spyOn(console, "warn").mockImplementation(() => {});
98
+ });
99
+
100
+ it("carries the usage's overrides on its own resolved server only", async () => {
101
+ // Riding the server is the scoping mechanism: an override can no longer
102
+ // reach a same-named tool on another server, because it never exists
103
+ // anywhere but its own server's object.
104
+ const client = clientReturning({
105
+ github: httpMcpServer("github"),
106
+ slack: httpMcpServer("slack"),
107
+ });
108
+
109
+ const result = await resolveMcpServers(
110
+ client,
111
+ [
112
+ makeUsage("github", [], [{ toolName: "delete_item", requiresApproval: false }]),
113
+ makeUsage("slack"),
114
+ ],
115
+ {},
116
+ "stdio-forbidden",
117
+ );
118
+
119
+ const bySlug = new Map(result.resolvedServers.map((s) => [s.slug, s]));
120
+ expect(bySlug.get("github")!.toolApprovalOverrides).toEqual([
121
+ { toolName: "delete_item", requiresApproval: false },
122
+ ]);
123
+ expect(bySlug.get("slack")!.toolApprovalOverrides).toEqual([]);
124
+ });
125
+ });
@@ -70,7 +70,7 @@ describe("buildEnhancedPrompt delegation integration", () => {
70
70
  skills: [],
71
71
  subAgents: [],
72
72
  workspaceFileRefs: [],
73
- attachmentPaths: [],
73
+ attachments: [],
74
74
  };
75
75
 
76
76
  it("includes exploration guidance when a workspace dir is present", () => {
@@ -222,6 +222,19 @@ describe("consumeCursorTurnStream", () => {
222
222
  expect(state.streamErrorMessage).toBe("boom");
223
223
  });
224
224
 
225
+ it("does not capture a non-string stream ERROR message (oss#299 hardening)", async () => {
226
+ // message is untyped at runtime; a structured value assigned here would
227
+ // crash classifyText downstream (.toLowerCase() on a non-string).
228
+ const { deps, state } = buildDeps();
229
+
230
+ await consumeCursorTurnStream(
231
+ mockRun([ev({ type: "status", status: "ERROR", message: { code: 14 } })]),
232
+ deps,
233
+ );
234
+
235
+ expect(state.streamErrorMessage).toBeUndefined();
236
+ });
237
+
225
238
  describe("first-denial early stop", () => {
226
239
  let hitlDir: string;
227
240
 
@@ -13,6 +13,7 @@
13
13
  * "mcpToolPolicies": {
14
14
  * "apply_cloud_resource": { "requiresApproval": true, "message": "..." }
15
15
  * },
16
+ * "mcpServerEnabledTools": { "planton": ["get_cloud_resource"] },
16
17
  * "approvedGrants": [{ "toolName": "edit", "mcpServerSlug": "", "key": "write", "salient": "a.txt", "contentDigest": "<sha256>" }],
17
18
  * "approvedGrantTokens": ["<base64(key\nsalient[\ncontentDigest])>"]
18
19
  * }
@@ -174,6 +175,21 @@ export interface ApprovalStateFile {
174
175
  */
175
176
  leasedCategories: string[];
176
177
  mcpToolPolicies: Record<string, McpToolPolicyEntry>;
178
+ /**
179
+ * Per-server effective enabled_tools allow-lists (issue #350), keyed by
180
+ * MCP server slug — ONLY restricted servers appear (an absent slug means
181
+ * unrestricted, so the common case stays an empty object). The Cursor SDK
182
+ * config cannot hide a server's tools, so the hook enforces the manifest
183
+ * instead: on beforeMCPExecution it matches the payload's mcp_server_name
184
+ * against this map and denies a non-listed tool with the non-pausing,
185
+ * permanent "disabled" kind — BEFORE autoApproveAll and grants, because
186
+ * enabled_tools is a capability manifest, not an approval gate (no bypass
187
+ * may resurrect a disabled tool, and no human may be offered "approve" on
188
+ * one). Unlike mcpToolPolicies (name-keyed, server-blind), this map is
189
+ * server-scoped: the hook payload carries the server identity, so equal
190
+ * tool names on different servers cannot cross-grant.
191
+ */
192
+ mcpServerEnabledTools: Record<string, string[]>;
177
193
  approvedGrants: ApprovalGrant[];
178
194
  approvedGrantTokens: string[];
179
195
  /**
@@ -381,6 +397,8 @@ function parseArgs(argsPreview: string): Record<string, unknown> | undefined {
381
397
  * - leasedCategories: built-in categories with a run-lifetime lease
382
398
  * - mcpToolPolicies: per-tool policy for MCP tools requiring approval (leased
383
399
  * servers are already absent — dropped upstream by mergeApprovalPolicies)
400
+ * - mcpServerEnabledTools: per-server enabled_tools allow-lists (issue #350,
401
+ * restricted servers only) for the hook's permanent "disabled" arm
384
402
  * - approvedGrants / approvedGrantTokens: tools approved in the current HITL
385
403
  * cycle, allowed through on reinvocation
386
404
  *
@@ -396,6 +414,7 @@ export function buildApprovalState(
396
414
  captureIgnored = false,
397
415
  gitWorkspace = true,
398
416
  unattendedSkip = false,
417
+ mcpServerEnabledTools: Record<string, string[]> = {},
399
418
  ): ApprovalStateFile {
400
419
  const approvedGrants = grants ?? [];
401
420
 
@@ -411,6 +430,7 @@ export function buildApprovalState(
411
430
  autoApproveAll: globalBypass,
412
431
  leasedCategories: [...leasedCategories],
413
432
  mcpToolPolicies,
433
+ mcpServerEnabledTools,
414
434
  approvedGrants,
415
435
  // The hook matches a tool call's PRIMARY token (content when it can compute a
416
436
  // digest from tool_input, else coarse). A content-identified grant authorizes
@@ -474,12 +494,18 @@ const DENIAL_LEDGER_FILE = "denials.jsonl";
474
494
  * classification may never have run).
475
495
  * - `fail-closed` — the approval state file was missing, so everything gated
476
496
  * denied. A turn-level "the gate itself was broken" fact.
497
+ * - `disabled` — the agent's enabled_tools manifest excludes this MCP
498
+ * tool (issue #350). Permanent for the run and
499
+ * mode-independent: NOT an approval (a human must never be
500
+ * offered "approve" on a manifest-disabled tool), so it is
501
+ * non-pausing and the model adapts — the same consumer
502
+ * semantics as `secret`.
477
503
  *
478
504
  * An unknown kind string is preserved as-is: it is treated as non-pausing (an
479
505
  * unknown deny must never manufacture an approval) but still attributes the
480
506
  * blocked call to our own hook.
481
507
  */
482
- export type DenialKind = "approval" | "unattended" | "secret" | "capture-error" | "fail-closed";
508
+ export type DenialKind = "approval" | "unattended" | "secret" | "capture-error" | "fail-closed" | "disabled";
483
509
 
484
510
  /** The one kind that pauses the run for user approval. */
485
511
  export const APPROVAL_DENIAL_KIND: DenialKind = "approval";
@@ -487,6 +513,9 @@ export const APPROVAL_DENIAL_KIND: DenialKind = "approval";
487
513
  /** The unattended-mode resolution kind (non-pausing; stamped SKIPPED). */
488
514
  export const UNATTENDED_DENIAL_KIND: DenialKind = "unattended";
489
515
 
516
+ /** The enabled_tools manifest denial kind (non-pausing, permanent; issue #350). */
517
+ export const DISABLED_DENIAL_KIND: DenialKind = "disabled";
518
+
490
519
  /**
491
520
  * One denial recorded by the preToolUse hook. `token` is the call's identity in
492
521
  * the same space as grantToken() (base64 of `toolName \n salientArg`), used to
@@ -22,6 +22,13 @@
22
22
  * directives are built from the RESOLVED paths, so prompt and filesystem can
23
23
  * never disagree.
24
24
  *
25
+ * Duplicate filenames are renamed, never overwritten (issue #364): because
26
+ * placement keys purely on the filename, two attachments with the same name
27
+ * contend for one path — the later one takes the platform's `stem-2.ext`
28
+ * rename (shared/attachment-naming.ts, same semantics as the deep-agent
29
+ * injector and the React composer) and the rename is disclosed in the
30
+ * prompt's `<input_files>` section via {@link ResolvedAttachment.renamedFrom}.
31
+ *
25
32
  * Error model: fail-hard, matching the native harness's attachment injector.
26
33
  * Attachments are explicit user inputs — an execution that silently runs
27
34
  * without one produces silently incorrect results (the "plan file wasn't
@@ -33,6 +40,7 @@ import { mkdir, copyFile, readFile, stat, writeFile } from "node:fs/promises";
33
40
  import { join, basename } from "node:path";
34
41
  import type { Attachment } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/spec_pb";
35
42
  import type { ArtifactStorage } from "../../shared/artifact-storage.js";
43
+ import { allocateUniqueName } from "../../shared/attachment-naming.js";
36
44
  import {
37
45
  isVisionCandidate,
38
46
  type VisionBudget,
@@ -46,9 +54,16 @@ import { ensureStigmerSymlink, STIGMER_LOCAL_STATE_DIR } from "../../shared/work
46
54
  const INPUTS_SUBDIR = "inputs";
47
55
 
48
56
  export interface ResolvedAttachment {
57
+ /** The final on-disk basename — after any duplicate rename. */
49
58
  filename: string;
50
59
  /** Workspace-relative path the agent reads (`.stigmer/inputs/{filename}`). */
51
60
  relativePath: string;
61
+ /**
62
+ * The attachment's original filename, present only when a duplicate name
63
+ * was renamed (shared/attachment-naming.ts) — rendered as disclosure in
64
+ * the prompt's `<input_files>` section.
65
+ */
66
+ renamedFrom?: string;
52
67
  /** Present when the attachment was accepted into the turn's vision payload. */
53
68
  vision?: VisionImage;
54
69
  /**
@@ -111,9 +126,13 @@ export async function resolveAttachments(
111
126
  // it, but only when the agent has skills).
112
127
  await ensureStigmerSymlink(options.primaryWorkspaceDir, platformDir);
113
128
 
129
+ // Placement keys purely on the filename, so this set is the whole
130
+ // collision domain — sequential resolution means each attachment sees
131
+ // every name claimed before it (see module doc on duplicate handling).
132
+ const takenNames = new Set<string>();
114
133
  const results: ResolvedAttachment[] = [];
115
134
  for (const attachment of attachments) {
116
- results.push(await resolveAttachment(attachment, inputsDir, options));
135
+ results.push(await resolveAttachment(attachment, inputsDir, takenNames, options));
117
136
  }
118
137
 
119
138
  console.log(
@@ -127,11 +146,15 @@ export async function resolveAttachments(
127
146
  async function resolveAttachment(
128
147
  attachment: Attachment,
129
148
  inputsDir: string,
149
+ takenNames: Set<string>,
130
150
  options: AttachmentResolverOptions,
131
151
  ): Promise<ResolvedAttachment> {
132
152
  // Local-mode fast path: the file is already on this machine's disk.
133
153
  if (options.mode === "local" && attachment.localPath) {
134
- const filename = safeInputName(attachment.filename || attachment.localPath);
154
+ const { name: filename, renamedFrom } = allocateUniqueName(
155
+ safeInputName(attachment.filename || attachment.localPath),
156
+ takenNames,
157
+ );
135
158
  let vision: VisionOutcome | undefined;
136
159
  try {
137
160
  vision = await materializeLocalFile(attachment, filename, inputsDir, options.visionBudget);
@@ -145,6 +168,7 @@ async function resolveAttachment(
145
168
  return {
146
169
  filename,
147
170
  relativePath: join(STIGMER_LOCAL_STATE_DIR, INPUTS_SUBDIR, filename),
171
+ ...(renamedFrom !== undefined ? { renamedFrom } : {}),
148
172
  ...visionOutcomeFields(vision),
149
173
  };
150
174
  }
@@ -164,7 +188,10 @@ async function resolveAttachment(
164
188
  );
165
189
  }
166
190
 
167
- const filename = safeInputName(attachment.filename || attachment.storageKey);
191
+ const { name: filename, renamedFrom } = allocateUniqueName(
192
+ safeInputName(attachment.filename || attachment.storageKey),
193
+ takenNames,
194
+ );
168
195
  let content: Buffer;
169
196
  try {
170
197
  content = await options.storage.download(attachment.storageKey);
@@ -185,6 +212,7 @@ async function resolveAttachment(
185
212
  return {
186
213
  filename,
187
214
  relativePath: join(STIGMER_LOCAL_STATE_DIR, INPUTS_SUBDIR, filename),
215
+ ...(renamedFrom !== undefined ? { renamedFrom } : {}),
188
216
  ...visionOutcomeFields(vision),
189
217
  };
190
218
  }
@@ -20,6 +20,13 @@ import type { SessionSpec } from "@stigmer/protos/ai/stigmer/agentic/session/v1/
20
20
  import type { WorkspaceEntry } from "@stigmer/protos/ai/stigmer/agentic/session/v1/workspace_pb";
21
21
  import type { ApiResourceReference } from "@stigmer/protos/ai/stigmer/commons/apiresource/io_pb";
22
22
  import type { CloudRepo } from "./session-lifecycle.js";
23
+ import { mergeMcpServerUsages } from "../../shared/mcp-resolver.js";
24
+
25
+ // Both harnesses must merge agent + session usages identically (session wins
26
+ // per slug — the usage whose enabled_tools the enforcement honors), so the
27
+ // merge lives in shared/mcp-resolver.ts. Re-exported here for its historical
28
+ // home alongside mergeSkillRefs.
29
+ export { mergeMcpServerUsages } from "../../shared/mcp-resolver.js";
23
30
 
24
31
  /**
25
32
  * Path segments that identify runner-internal directories. Any workspace dir
@@ -130,33 +137,6 @@ export function resolveCloudRepos(workspaceEntries: WorkspaceEntry[]): CloudRepo
130
137
  // MCP and skill merging
131
138
  // ---------------------------------------------------------------------------
132
139
 
133
- /**
134
- * Merge MCP server usages from agent (base) and session (overlay).
135
- *
136
- * Replicates session_context_merge.py::merge_mcp_server_usages():
137
- * - Agent-level usages are the base set
138
- * - Session-level usages extend or override by mcp_server_ref.slug
139
- * - If both reference the same slug, session-level takes precedence
140
- */
141
- export function mergeMcpServerUsages(
142
- agentUsages: McpServerUsage[],
143
- sessionUsages: McpServerUsage[],
144
- ): McpServerUsage[] {
145
- const bySlug = new Map<string, McpServerUsage>();
146
-
147
- for (const usage of agentUsages) {
148
- const slug = usage.mcpServerRef?.slug;
149
- if (slug) bySlug.set(slug, usage);
150
- }
151
-
152
- for (const usage of sessionUsages) {
153
- const slug = usage.mcpServerRef?.slug;
154
- if (slug) bySlug.set(slug, usage);
155
- }
156
-
157
- return [...bySlug.values()];
158
- }
159
-
160
140
  /**
161
141
  * Merge skill refs from agent and session.
162
142
  *
@@ -25,7 +25,9 @@ export async function resolveExecutionEnv(
25
25
  ): Promise<EnvResult> {
26
26
  // A desktop runner exchanges its bootstrap credential for a token scoped to
27
27
  // this execution's session, so cloud's decrypt gate binds the read (#156).
28
- // No-op for cloud sandbox and OSS runners.
28
+ // No-op for cloud sandbox and OSS runners. A failed exchange throws and
29
+ // fails the activity: the bootstrap credential no longer decrypts
30
+ // (stigmer-cloud#218), so proceeding would resolve redacted placeholders.
29
31
  const scopedToken = await client.acquireScopedRunnerToken({
30
32
  agentExecutionId: executionId,
31
33
  });
@@ -7,8 +7,9 @@
7
7
  * 3. Surface isRetryable for future workflow-level retry decisions
8
8
  *
9
9
  * Error detail can come from several sources (in priority order):
10
- * - A thrown CursorSdkError's structured fields (highest fidelity)
11
- * - run.wait().result (SDK-provided string, often bare/generic)
10
+ * - Structured fields, either from a thrown CursorSdkError or lifted from a
11
+ * structured run.wait() error value (highest fidelity)
12
+ * - run.wait() error text (SDK-provided string, often bare/generic)
12
13
  * - SDKStatusMessage with status "ERROR" from the stream
13
14
  * - ConnectError captured from process unhandledRejection
14
15
  * - Text extracted from the failing run.conversation() turn
@@ -39,7 +40,8 @@ export interface ClassifiedError {
39
40
 
40
41
  /**
41
42
  * Structured fields lifted from a thrown CursorSdkError (errors.d.ts:
42
- * { code, status, isRetryable, cause, endpoint, requestId, operation }).
43
+ * { code, status, isRetryable, cause, endpoint, requestId, operation }) or
44
+ * from a structured run.wait() error value (see extractRunErrorSources).
43
45
  * Only the fields used for classification are retained.
44
46
  */
45
47
  export interface SdkErrorFields {
@@ -48,6 +50,87 @@ export interface SdkErrorFields {
48
50
  message?: string;
49
51
  }
50
52
 
53
+ /**
54
+ * What String() produces for any plain object. Carries zero signal, so it is
55
+ * refused everywhere: extraction never emits it, and classifyFromSources
56
+ * treats it as absent should any other producer leak it through.
57
+ */
58
+ const OBJECT_JUNK_STRING = "[object Object]";
59
+
60
+ /**
61
+ * Error detail lifted from a run.wait() result, split by fidelity: structured
62
+ * values land in sdkError, plain text in sdkResultFields. At most one of the
63
+ * two is set; both undefined means the result carried nothing usable and the
64
+ * classifier's lower-priority sources should decide.
65
+ */
66
+ export interface RunErrorSources {
67
+ sdkError: SdkErrorFields | undefined;
68
+ sdkResultFields: string | undefined;
69
+ }
70
+
71
+ const NO_RUN_ERROR_SOURCES: RunErrorSources = {
72
+ sdkError: undefined,
73
+ sdkResultFields: undefined,
74
+ };
75
+
76
+ /**
77
+ * Lift error detail from a run.wait() result whose status is "error".
78
+ *
79
+ * The SDK types RunResult.result as `string`, but structured values (Error
80
+ * instances, { code, message } objects) have been observed at runtime, both
81
+ * in `result` and in the undeclared error/message/reason fields (oss#299).
82
+ * A bare String() on those yields "[object Object]", which both destroys the
83
+ * message users see and — worse — shadows the lower-priority classifier
84
+ * sources (stream, rejection, conversation introspection) that often hold
85
+ * the real reason.
86
+ *
87
+ * Walks the candidate fields in order and answers from the FIRST one that
88
+ * yields usable content: strings keep flowing to the string channel
89
+ * (sdkResultFields), structured values are lifted into the same structured
90
+ * channel a thrown CursorSdkError uses (sdkError), and hopeless values are
91
+ * skipped so a later candidate — or the classifier's fallback sources — can
92
+ * win. (The previous `??` chain stopped at the first non-nullish value, so a
93
+ * hopeless object or empty string hid usable text one field later.)
94
+ *
95
+ * Deliberately NO JSON.stringify fallback for unrecognized object shapes:
96
+ * the error arm already logs the raw result in full server-side, and a JSON
97
+ * blob shown to the user would shadow the introspection sources that exist
98
+ * precisely to recover the real reason.
99
+ */
100
+ export function extractRunErrorSources(result: unknown): RunErrorSources {
101
+ if (result === null || typeof result !== "object") return NO_RUN_ERROR_SOURCES;
102
+ const r = result as Record<string, unknown>;
103
+ for (const candidate of [r.result, r.error, r.message, r.reason]) {
104
+ const extracted = extractFromCandidate(candidate);
105
+ if (extracted) return extracted;
106
+ }
107
+ return NO_RUN_ERROR_SOURCES;
108
+ }
109
+
110
+ function extractFromCandidate(v: unknown): RunErrorSources | undefined {
111
+ if (v == null) return undefined;
112
+ if (typeof v === "string") {
113
+ if (v.length === 0 || v === OBJECT_JUNK_STRING) return undefined;
114
+ return { sdkError: undefined, sdkResultFields: v };
115
+ }
116
+ if (typeof v === "object") {
117
+ // Covers Error instances too: their message (and, on SDK/Node error
118
+ // shapes, code/status) are readable as plain properties.
119
+ const o = v as Record<string, unknown>;
120
+ const fields: SdkErrorFields = {};
121
+ if (typeof o.code === "string" && o.code.length > 0) fields.code = o.code;
122
+ if (typeof o.status === "number") fields.status = o.status;
123
+ if (typeof o.message === "string" && o.message.length > 0) fields.message = o.message;
124
+ if (fields.code !== undefined || fields.status !== undefined || fields.message !== undefined) {
125
+ return { sdkError: fields, sdkResultFields: undefined };
126
+ }
127
+ return undefined;
128
+ }
129
+ // Remaining primitives (number, boolean, ...) stringify losslessly.
130
+ const text = String(v);
131
+ return text.length > 0 ? { sdkError: undefined, sdkResultFields: text } : undefined;
132
+ }
133
+
51
134
  const AUTH_PATTERNS = [
52
135
  "unauthenticated", "unauthorized", "401", "forbidden",
53
136
  "permission_denied", "invalid api key", "not logged in",
@@ -168,7 +251,11 @@ function classifyFromSources(opts: SynthesizeErrorOpts): ClassifiedError {
168
251
  }
169
252
 
170
253
  if (opts.sdkResultFields) {
171
- const isBareGeneric = opts.sdkResultFields === "Cursor run failed";
254
+ // "Cursor run failed" is the SDK's bare generic; "[object Object]" is
255
+ // String()-coerced junk from any producer that bypassed the shape-aware
256
+ // extraction. Neither carries signal — fall through to better sources.
257
+ const isBareGeneric = opts.sdkResultFields === "Cursor run failed"
258
+ || opts.sdkResultFields === OBJECT_JUNK_STRING;
172
259
  if (!isBareGeneric) {
173
260
  const { category, retryable } = classifyText(opts.sdkResultFields);
174
261
  return {
@@ -0,0 +1,72 @@
1
+ /**
2
+ * Tier-2 structured-output extraction for the Cursor harness.
3
+ *
4
+ * When tier-1 text extraction (shared/extract-json.ts) cannot find JSON in
5
+ * the agent's free-text response, this tier asks an economy-tier LLM to
6
+ * extract it via withStructuredOutput (function-calling), which guarantees
7
+ * schema-conformant output through the API's tool-use mechanism.
8
+ *
9
+ * Lives in its own module (rather than inside execute-cursor/index.ts) so
10
+ * the LangChain construction path stays out of the Cursor activity's module
11
+ * graph until a run actually needs tier 2 — index.ts imports this module
12
+ * lazily at the call site, mirroring its tier-1 import, which is what
13
+ * bundle-slim's deferred evaluation preserves.
14
+ */
15
+
16
+ import type { Config } from "../../config.js";
17
+ import { getEconomyModel } from "../../shared/model-registry.js";
18
+ import { buildChatModel } from "../../shared/model-client.js";
19
+ import { checkDirectCredentials } from "../../shared/llm-backend.js";
20
+ import { tryInferProvider } from "../../shared/llm-proxy.js";
21
+ import { jsonSchemaToZod } from "../../shared/json-schema-to-zod.js";
22
+
23
+ /**
24
+ * Extract structured data from an agent's free-text response using an
25
+ * economy-tier LLM with withStructuredOutput (function-calling).
26
+ *
27
+ * Construction (registry-id resolution, provider inference, proxy wiring) is
28
+ * delegated to the shared buildChatModel so the economy model's registry id
29
+ * is always resolved to a provider API id before the call.
30
+ *
31
+ * Throws when no LLM is reachable (no proxy and no credential path for the
32
+ * extraction model's provider) — the caller treats any throw here as "tier 2
33
+ * unavailable", logs it, and returns the agent's text without structured
34
+ * output, so the failure mode is a diagnosable log line, never a lost run.
35
+ */
36
+ export async function extractStructuredOutput(
37
+ agentResponse: string,
38
+ schema: Record<string, unknown>,
39
+ config: Config,
40
+ primaryModel: string,
41
+ ): Promise<unknown | null> {
42
+ const extractionModel = await getEconomyModel(primaryModel);
43
+ const proxyEndpoint = config.proxyEndpoint ?? undefined;
44
+
45
+ if (!proxyEndpoint) {
46
+ const provider = tryInferProvider(extractionModel);
47
+ const missing = provider === null ? null : checkDirectCredentials(provider);
48
+ if (missing !== null) {
49
+ throw new Error(
50
+ `Structured-output extraction needs the ${provider} model ` +
51
+ `'${extractionModel}' but has no credential path. ${missing}`,
52
+ );
53
+ }
54
+ }
55
+
56
+ const { model: llm } = await buildChatModel({
57
+ modelName: extractionModel,
58
+ proxyEndpoint,
59
+ stigmerToken: config.stigmerToken ?? undefined,
60
+ maxTokens: 4096,
61
+ });
62
+
63
+ const zodSchema = jsonSchemaToZod(schema);
64
+ const structured = llm.withStructuredOutput(zodSchema);
65
+
66
+ const result = await structured.invoke([
67
+ { role: "system", content: "Extract the structured data from the agent's response. Return only the data that matches the schema." },
68
+ { role: "user", content: agentResponse },
69
+ ]);
70
+
71
+ return result ?? null;
72
+ }