@stigmer/runner 3.0.8-dev.20260613085218 → 3.0.9-dev.20260615145121

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/dist/.build-fingerprint +1 -1
  2. package/dist/activities/execute-cursor/hook-script.d.ts +23 -12
  3. package/dist/activities/execute-cursor/hook-script.js +85 -51
  4. package/dist/activities/execute-cursor/hook-script.js.map +1 -1
  5. package/dist/activities/execute-cursor/index.js +220 -79
  6. package/dist/activities/execute-cursor/index.js.map +1 -1
  7. package/dist/activities/execute-cursor/message-translator.d.ts +51 -9
  8. package/dist/activities/execute-cursor/message-translator.js +146 -20
  9. package/dist/activities/execute-cursor/message-translator.js.map +1 -1
  10. package/dist/activities/execute-cursor/persist-decision.d.ts +42 -0
  11. package/dist/activities/execute-cursor/persist-decision.js +30 -0
  12. package/dist/activities/execute-cursor/persist-decision.js.map +1 -0
  13. package/dist/activities/execute-cursor/prompt-builder.d.ts +25 -0
  14. package/dist/activities/execute-cursor/prompt-builder.js +54 -0
  15. package/dist/activities/execute-cursor/prompt-builder.js.map +1 -1
  16. package/dist/activities/execute-cursor/workspace-setup.d.ts +8 -2
  17. package/dist/activities/execute-cursor/workspace-setup.js +62 -30
  18. package/dist/activities/execute-cursor/workspace-setup.js.map +1 -1
  19. package/dist/activities/execute-deep-agent/index.js +15 -5
  20. package/dist/activities/execute-deep-agent/index.js.map +1 -1
  21. package/dist/activities/execute-deep-agent/status-builder-shared.d.ts +0 -1
  22. package/dist/activities/execute-deep-agent/status-builder-shared.js +32 -8
  23. package/dist/activities/execute-deep-agent/status-builder-shared.js.map +1 -1
  24. package/dist/activities/execute-deep-agent/status-builder.js +4 -5
  25. package/dist/activities/execute-deep-agent/status-builder.js.map +1 -1
  26. package/dist/activities/execute-deep-agent/streaming-v3.js +4 -5
  27. package/dist/activities/execute-deep-agent/streaming-v3.js.map +1 -1
  28. package/dist/activities/execute-deep-agent/streaming.d.ts +9 -1
  29. package/dist/activities/execute-deep-agent/streaming.js +4 -5
  30. package/dist/activities/execute-deep-agent/streaming.js.map +1 -1
  31. package/dist/activities/execute-deep-agent/subagent-tracker.js +4 -5
  32. package/dist/activities/execute-deep-agent/subagent-tracker.js.map +1 -1
  33. package/dist/activities/execute-deep-agent/v3-status-builder.js +6 -5
  34. package/dist/activities/execute-deep-agent/v3-status-builder.js.map +1 -1
  35. package/dist/config.d.ts +21 -0
  36. package/dist/config.js +12 -0
  37. package/dist/config.js.map +1 -1
  38. package/dist/in-flight.d.ts +35 -0
  39. package/dist/in-flight.js +61 -0
  40. package/dist/in-flight.js.map +1 -0
  41. package/dist/runner-manager.d.ts +2 -0
  42. package/dist/runner-manager.js +90 -29
  43. package/dist/runner-manager.js.map +1 -1
  44. package/dist/runner.d.ts +2 -0
  45. package/dist/runner.js +2 -0
  46. package/dist/runner.js.map +1 -1
  47. package/dist/shared/grpc-retry.d.ts +9 -20
  48. package/dist/shared/grpc-retry.js +9 -52
  49. package/dist/shared/grpc-retry.js.map +1 -1
  50. package/dist/shared/stall-watchdog.d.ts +68 -0
  51. package/dist/shared/stall-watchdog.js +102 -0
  52. package/dist/shared/stall-watchdog.js.map +1 -0
  53. package/dist/shared/status-offload.d.ts +84 -0
  54. package/dist/shared/status-offload.js +292 -0
  55. package/dist/shared/status-offload.js.map +1 -0
  56. package/dist/shared/status.d.ts +34 -3
  57. package/dist/shared/status.js +102 -9
  58. package/dist/shared/status.js.map +1 -1
  59. package/dist/{activities/execute-deep-agent → shared}/streaming-scheduler.d.ts +4 -0
  60. package/dist/{activities/execute-deep-agent → shared}/streaming-scheduler.js +4 -0
  61. package/dist/shared/streaming-scheduler.js.map +1 -0
  62. package/package.json +2 -2
  63. package/src/__tests__/config.test.ts +8 -0
  64. package/src/__tests__/in-flight.test.ts +84 -0
  65. package/src/activities/__tests__/classify-tool-approvals.test.ts +1 -0
  66. package/src/activities/__tests__/discover-mcp-server.test.ts +1 -0
  67. package/src/activities/execute-cursor/__tests__/build-prompt.test.ts +74 -0
  68. package/src/activities/execute-cursor/__tests__/hook-script.test.ts +90 -15
  69. package/src/activities/execute-cursor/__tests__/message-translator.test.ts +124 -12
  70. package/src/activities/execute-cursor/__tests__/persist-decision.test.ts +99 -0
  71. package/src/activities/execute-cursor/__tests__/tool-result-image.test.ts +244 -0
  72. package/src/activities/execute-cursor/__tests__/workspace-setup.test.ts +53 -4
  73. package/src/activities/execute-cursor/hook-script.ts +85 -51
  74. package/src/activities/execute-cursor/index.ts +187 -38
  75. package/src/activities/execute-cursor/message-translator.ts +146 -20
  76. package/src/activities/execute-cursor/persist-decision.ts +54 -0
  77. package/src/activities/execute-cursor/prompt-builder.ts +59 -0
  78. package/src/activities/execute-cursor/workspace-setup.ts +76 -44
  79. package/src/activities/execute-deep-agent/__tests__/index.test.ts +1 -0
  80. package/src/activities/execute-deep-agent/__tests__/status-builder-shared.test.ts +66 -0
  81. package/src/activities/execute-deep-agent/__tests__/status-builder.test.ts +6 -3
  82. package/src/activities/execute-deep-agent/__tests__/streaming-v3.test.ts +70 -0
  83. package/src/activities/execute-deep-agent/index.ts +17 -5
  84. package/src/activities/execute-deep-agent/status-builder-shared.ts +27 -5
  85. package/src/activities/execute-deep-agent/status-builder.ts +3 -5
  86. package/src/activities/execute-deep-agent/streaming-v3.ts +5 -5
  87. package/src/activities/execute-deep-agent/streaming.ts +14 -5
  88. package/src/activities/execute-deep-agent/subagent-tracker.ts +4 -5
  89. package/src/activities/execute-deep-agent/v3-status-builder.ts +5 -5
  90. package/src/config.ts +27 -0
  91. package/src/in-flight.ts +71 -0
  92. package/src/runner-manager.ts +127 -33
  93. package/src/runner.ts +6 -0
  94. package/src/shared/__tests__/artifact-storage.test.ts +1 -0
  95. package/src/shared/__tests__/grpc-retry-extended.test.ts +6 -144
  96. package/src/shared/__tests__/grpc-retry.test.ts +5 -134
  97. package/src/shared/__tests__/stall-watchdog.test.ts +193 -0
  98. package/src/shared/__tests__/status-offload.test.ts +256 -0
  99. package/src/shared/__tests__/status.test.ts +199 -0
  100. package/src/shared/grpc-retry.ts +9 -72
  101. package/src/shared/stall-watchdog.ts +122 -0
  102. package/src/shared/status-offload.ts +342 -0
  103. package/src/shared/status.ts +142 -8
  104. package/src/{activities/execute-deep-agent → shared}/streaming-scheduler.ts +4 -0
  105. package/dist/activities/execute-deep-agent/streaming-scheduler.js.map +0 -1
  106. /package/src/{activities/execute-deep-agent → shared}/__tests__/streaming-scheduler.test.ts +0 -0
@@ -1,15 +1,23 @@
1
1
  /**
2
- * Template for the preToolUse hook script that Cursor spawns.
2
+ * Template for the approval hook script that Cursor spawns.
3
3
  *
4
- * This module doesn't execute as a hook itself — it generates the shell
5
- * script that is written to .cursor/hooks/stigmer-approval.sh. That script
6
- * is invoked by Cursor for every tool call via the preToolUse hook.
4
+ * This module doesn't execute as a hook itself — it generates the shell script
5
+ * written to the HITL dir as stigmer-approval.sh. Cursor invokes that ONE script
6
+ * for TWO events (registered in .cursor/hooks.json by workspace-setup.ts):
7
+ * - `preToolUse` — fires for built-in tools (Write/Shell/Delete/…).
8
+ * - `beforeMCPExecution` — the only event Cursor enforces for MCP tool calls;
9
+ * `preToolUse` does NOT gate MCP (confirmed by a live
10
+ * payload capture). MCP is therefore gated in exactly
11
+ * ONE place, so a denial is never double-recorded.
12
+ * The script branches on the payload's `hook_event_name`: MCP tools are gated on
13
+ * the beforeMCPExecution invocation, built-ins on the preToolUse invocation.
7
14
  *
8
15
  * The hook script:
9
16
  * 1. Reads the tool call JSON from stdin
10
17
  * 2. Reads the approval state JSON file written by the cursor-runner
11
- * 3. Evaluates the policy: auto-approve, approved grants (reinvocation),
12
- * gated built-in tools, MCP require-approval policies
18
+ * 3. Evaluates the policy: auto-approve, approved grants (reinvocation), then —
19
+ * by event — gated built-in tools (preToolUse) or MCP require-approval
20
+ * policies (beforeMCPExecution)
13
21
  * 4. On a deny, appends the call's identity token to the denial ledger
14
22
  * (stigmer-denials.jsonl) so the runner can mark the gated tool call as
15
23
  * WAITING_APPROVAL — the hook is the only place the deny decision is made,
@@ -49,23 +57,35 @@
49
57
  * Policy evaluation order (first match wins). The model is "gate the dangerous
50
58
  * set, allow the rest" — matching the native harness and avoiding denial of
51
59
  * auto-approved MCP tools (which are absent from mcpToolPolicies):
52
- * 1. autoApproveAll → allow
53
- * 2. Gated built-in (category non-empty):
54
- * a. identity token in approvedGrantTokens → allow (reinvocation grant)
60
+ * 0. Scope guard: not the runner's own agent → allow (never touch the ledger)
61
+ * 1. Missing state file → deny (fail-closed); autoApproveAll → allow
62
+ * 2. beforeMCPExecution event → MCP tool present in mcpToolPolicies
63
+ * (require-approval):
64
+ * a. name token in approvedGrantTokens → allow (reinvocation grant)
55
65
  * b. otherwise → record denial, deny
56
- * 3. MCP tool present in mcpToolPolicies (require-approval):
57
- * a. name token in approvedGrantTokens → allow
66
+ * (auto-approved / unlisted MCP tools fall through → allow)
67
+ * 3. preToolUse event → gated built-in (category non-empty):
68
+ * a. identity token in approvedGrantTokens → allow (reinvocation grant)
58
69
  * b. otherwise → record denial, deny
59
- * 4. Everything else (read-only built-ins, auto-approved MCP, unknown) → allow
70
+ * (read-only / ungated built-ins fall through → allow)
60
71
  */
61
72
 
62
73
  import { SALIENT_ARG_FIELDS, getBuiltInGatedCategories } from "./approval-policy.js";
63
74
 
75
+ // Shown to the model when the gate denies a tool call. It must NOT teach the
76
+ // model to ask for permission in prose or to "stop and wait" — that framing
77
+ // makes the model narrate approval requests instead of invoking tools (the
78
+ // platform's approval surface is driven by tool invocation; see
79
+ // formatToolApprovalProtocol in prompt-builder.ts). Instead it tells the model
80
+ // the approval is automatic and that it should not retry or work around THIS
81
+ // action. Embedded verbatim into the generated hook script inside a
82
+ // single-quoted bash echo of a JSON object, so the text must contain no double
83
+ // quotes, apostrophes, or backslashes.
64
84
  const APPROVAL_REQUIRED_AGENT_MESSAGE =
65
- "STIGMER_APPROVAL_REQUIRED: This tool call requires user approval before " +
66
- "execution. Do not attempt alternative approaches or workarounds (including " +
67
- "shell commands). Stop and wait — the execution will resume after the user " +
68
- "reviews and approves this tool call.";
85
+ "This action has been submitted to the user for approval automatically; you " +
86
+ "do not need to ask for permission. Do not retry it or attempt a workaround " +
87
+ "for this action. The platform will resume you automatically after the user " +
88
+ "responds — continue with the rest of the task.";
69
89
 
70
90
  /**
71
91
  * Build the bash `case` arms that map an incoming hook `tool_name` to its
@@ -91,9 +111,11 @@ function buildCategoryCaseArms(): string {
91
111
  * Build the inline Node.js identity extractor embedded in the hook script.
92
112
  *
93
113
  * Parses the hook's stdin JSON properly (the bash fallback's grep truncates
94
- * string values at the first escaped quote) and emits four lines:
95
- * tool_name, canonical category, identity token, and MCP name-token. The token
96
- * encodings must stay byte-identical to grantToken() in approval-state.ts.
114
+ * string values at the first escaped quote) and emits five lines: tool_name,
115
+ * canonical category, identity token, MCP name-token, and hook_event_name (the
116
+ * event discriminator: `preToolUse` for built-ins, `beforeMCPExecution` for MCP).
117
+ * The token encodings must stay byte-identical to grantToken() in
118
+ * approval-state.ts.
97
119
  *
98
120
  * Authored as a single-quoted bash string, so the JS must not contain single
99
121
  * quotes. The category map and salient field list are baked from
@@ -115,7 +137,8 @@ function buildNodeIdentityScript(): string {
115
137
  `let s="";`,
116
138
  `for(const f of ${fields}){const v=a[f];if(typeof v==="string"&&v){s=v;break;}}`,
117
139
  `const b=(x)=>Buffer.from(x,"utf8").toString("base64");`,
118
- `process.stdout.write(name+"\\n"+cat+"\\n"+b(cat+"\\n"+s)+"\\n"+b(name+"\\n"));`,
140
+ `const ev=typeof t.hook_event_name==="string"?t.hook_event_name:"";`,
141
+ `process.stdout.write(name+"\\n"+cat+"\\n"+b(cat+"\\n"+s)+"\\n"+b(name+"\\n")+"\\n"+ev);`,
119
142
  ].join("");
120
143
  }
121
144
 
@@ -156,13 +179,15 @@ export function generateHookScript(
156
179
  const categoryCaseArms = buildCategoryCaseArms();
157
180
  const nodeIdentityScript = buildNodeIdentityScript();
158
181
  return `#!/bin/bash
159
- # Stigmer HITL approval hook for Cursor preToolUse
182
+ # Stigmer HITL approval hook for Cursor (preToolUse + beforeMCPExecution).
160
183
  # Generated by cursor-runner — do not edit manually.
161
184
  #
162
- # Reads tool call from stdin (JSON), checks approval state file, returns a
163
- # permission decision on stdout (JSON). On a deny, appends the call's canonical
164
- # identity token to the denial ledger so the runner can mark the gated tool call
165
- # as WAITING_APPROVAL. See hook-script.ts for the cross-taxonomy identity design.
185
+ # Reads a tool call from stdin (JSON), checks the approval state file, returns a
186
+ # permission decision on stdout (JSON). Branches on hook_event_name: MCP tools
187
+ # are gated on beforeMCPExecution, built-ins on preToolUse (preToolUse does not
188
+ # enforce MCP). On a deny, appends the call's canonical identity token to the
189
+ # denial ledger so the runner can mark the gated tool call as WAITING_APPROVAL.
190
+ # See hook-script.ts for the cross-taxonomy identity design.
166
191
 
167
192
  set -euo pipefail
168
193
 
@@ -219,6 +244,7 @@ if [ -n "$IDENTITY" ]; then
219
244
  CATEGORY=$(printf '%s\\n' "$IDENTITY" | sed -n 2p)
220
245
  TOKEN=$(printf '%s\\n' "$IDENTITY" | sed -n 3p)
221
246
  MCP_TOKEN=$(printf '%s\\n' "$IDENTITY" | sed -n 4p)
247
+ HOOK_EVENT=$(printf '%s\\n' "$IDENTITY" | sed -n 5p)
222
248
  else
223
249
  # Fallback when the Node binary cannot run: grep/cut extraction. Best-effort
224
250
  # only — '"field":"[^"]*"' truncates at the first JSON-escaped quote, so the
@@ -227,6 +253,7 @@ else
227
253
  # Every extraction ends with '|| true': under 'set -e' a non-matching grep
228
254
  # would otherwise abort the script and emit no decision.
229
255
  TOOL_NAME=$(echo "$INPUT" | grep -o '"tool_name":"[^"]*"' | head -1 | cut -d'"' -f4 || true)
256
+ HOOK_EVENT=$(echo "$INPUT" | grep -o '"hook_event_name":"[^"]*"' | head -1 | cut -d'"' -f4 || true)
230
257
  SALIENT=""
231
258
  for field in ${salientFields}; do
232
259
  v=$(echo "$INPUT" | grep -o "\\"$field\\":\\"[^\\"]*\\"" | head -1 | cut -d'"' -f4 || true)
@@ -262,7 +289,37 @@ record_denial() {
262
289
  echo '{"toolName":"'"$TOOL_NAME"'","token":"'"$1"'"}' >> "$LEDGER_FILE" 2>/dev/null || true
263
290
  }
264
291
 
265
- # --- 2. Gated built-in tools (category non-empty) ---
292
+ # --- 2. MCP tools (beforeMCPExecution event) ---
293
+ # preToolUse does NOT enforce gating for MCP calls — beforeMCPExecution does — so
294
+ # MCP is gated here and ONLY here (never double-recorded). mcpToolPolicies holds
295
+ # only require-approval tools (auto-approved MCP tools are absent), so presence
296
+ # means "deny" unless an entry is explicitly false. MCP tool names are consistent
297
+ # across the hook and the stream, so the identity token is name-only:
298
+ # base64("$TOOL_NAME\\n").
299
+ if [ "$HOOK_EVENT" = "beforeMCPExecution" ]; then
300
+ if echo "$STATE" | grep -q "\\"mcpToolPolicies\\"" && [ -n "$TOOL_NAME" ]; then
301
+ TOOL_POLICY=$(echo "$STATE" | grep -o "\\"$TOOL_NAME\\":{[^}]*}" | head -1 || true)
302
+ if [ -n "$TOOL_POLICY" ] && ! echo "$TOOL_POLICY" | grep -q '"requiresApproval":false'; then
303
+ # Reinvocation grant: this tool was approved earlier → allow.
304
+ if echo "$STATE" | grep -qF "\\"$MCP_TOKEN\\""; then
305
+ echo '{"permission":"allow"}'
306
+ exit 0
307
+ fi
308
+ MSG=$(echo "$TOOL_POLICY" | grep -o '"message":"[^"]*"' | head -1 | cut -d'"' -f4 || true)
309
+ if [ -z "$MSG" ]; then
310
+ MSG="Tool requires approval: $TOOL_NAME"
311
+ fi
312
+ record_denial "$MCP_TOKEN"
313
+ echo '{"permission":"deny","agent_message":"${APPROVAL_REQUIRED_AGENT_MESSAGE}","user_message":"'"$MSG"'"}'
314
+ exit 0
315
+ fi
316
+ fi
317
+ # Auto-approved or unlisted MCP tool → allow.
318
+ echo '{"permission":"allow"}'
319
+ exit 0
320
+ fi
321
+
322
+ # --- 3. Gated built-in tools (preToolUse event, category non-empty) ---
266
323
  if [ -n "$CATEGORY" ]; then
267
324
  # Reinvocation grant: this exact resource was approved earlier → allow.
268
325
  if echo "$STATE" | grep -qF "\\"$TOKEN\\""; then
@@ -274,32 +331,9 @@ if [ -n "$CATEGORY" ]; then
274
331
  exit 0
275
332
  fi
276
333
 
277
- # --- 3. MCP tools that require approval → deny ---
278
- # mcpToolPolicies holds only require-approval tools (auto-approved MCP tools are
279
- # absent), so presence means "deny" unless an entry is explicitly false. MCP tool
280
- # names are consistent across the hook and the stream, so the identity token is
281
- # name-only: base64("$TOOL_NAME\\n").
282
- if echo "$STATE" | grep -q "\\"mcpToolPolicies\\"" && [ -n "$TOOL_NAME" ]; then
283
- TOOL_POLICY=$(echo "$STATE" | grep -o "\\"$TOOL_NAME\\":{[^}]*}" | head -1 || true)
284
- if [ -n "$TOOL_POLICY" ] && ! echo "$TOOL_POLICY" | grep -q '"requiresApproval":false'; then
285
- if echo "$STATE" | grep -qF "\\"$MCP_TOKEN\\""; then
286
- echo '{"permission":"allow"}'
287
- exit 0
288
- fi
289
- MSG=$(echo "$TOOL_POLICY" | grep -o '"message":"[^"]*"' | head -1 | cut -d'"' -f4 || true)
290
- if [ -z "$MSG" ]; then
291
- MSG="Tool requires approval: $TOOL_NAME"
292
- fi
293
- record_denial "$MCP_TOKEN"
294
- echo '{"permission":"deny","agent_message":"${APPROVAL_REQUIRED_AGENT_MESSAGE}","user_message":"'"$MSG"'"}'
295
- exit 0
296
- fi
297
- fi
298
-
299
334
  # --- 4. Everything else → allow ---
300
- # Read-only built-ins, auto-approved MCP tools, and anything not explicitly
301
- # gated. Fail-open mirrors the native harness (gate the dangerous set, allow the
302
- # rest) and prevents denying auto-approved MCP tools the state cannot enumerate.
335
+ # Read-only built-ins and anything not explicitly gated. Fail-open mirrors the
336
+ # native harness (gate the dangerous set, allow the rest).
303
337
  echo '{"permission":"allow"}'
304
338
  exit 0
305
339
  `;