@stigmer/runner 3.0.9-dev.20260616060535 → 3.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (328) hide show
  1. package/dist/.build-fingerprint +1 -1
  2. package/dist/__test-utils__/approval-contract/types.d.ts +174 -0
  3. package/dist/__test-utils__/approval-contract/types.js +24 -0
  4. package/dist/__test-utils__/approval-contract/types.js.map +1 -0
  5. package/dist/activities/call-agent-status.d.ts +19 -1
  6. package/dist/activities/call-agent-status.js +64 -5
  7. package/dist/activities/call-agent-status.js.map +1 -1
  8. package/dist/activities/call-llm.js +19 -53
  9. package/dist/activities/call-llm.js.map +1 -1
  10. package/dist/activities/classify-tool-approvals.d.ts +24 -0
  11. package/dist/activities/classify-tool-approvals.js +69 -17
  12. package/dist/activities/classify-tool-approvals.js.map +1 -1
  13. package/dist/activities/discover-mcp-server.d.ts +7 -0
  14. package/dist/activities/discover-mcp-server.js +11 -1
  15. package/dist/activities/discover-mcp-server.js.map +1 -1
  16. package/dist/activities/execute-cursor/__test-utils__/cursor-hook-harness.d.ts +140 -0
  17. package/dist/activities/execute-cursor/__test-utils__/cursor-hook-harness.js +130 -0
  18. package/dist/activities/execute-cursor/__test-utils__/cursor-hook-harness.js.map +1 -0
  19. package/dist/activities/execute-cursor/__test-utils__/gateway-substrate.d.ts +18 -0
  20. package/dist/activities/execute-cursor/__test-utils__/gateway-substrate.js +123 -0
  21. package/dist/activities/execute-cursor/__test-utils__/gateway-substrate.js.map +1 -0
  22. package/dist/activities/execute-cursor/approval-policy.d.ts +17 -61
  23. package/dist/activities/execute-cursor/approval-policy.js +8 -120
  24. package/dist/activities/execute-cursor/approval-policy.js.map +1 -1
  25. package/dist/activities/execute-cursor/approval-state.d.ts +219 -9
  26. package/dist/activities/execute-cursor/approval-state.js +233 -17
  27. package/dist/activities/execute-cursor/approval-state.js.map +1 -1
  28. package/dist/activities/execute-cursor/capture-flow.d.ts +127 -0
  29. package/dist/activities/execute-cursor/capture-flow.js +234 -0
  30. package/dist/activities/execute-cursor/capture-flow.js.map +1 -0
  31. package/dist/activities/execute-cursor/cas-observations.d.ts +96 -0
  32. package/dist/activities/execute-cursor/cas-observations.js +184 -0
  33. package/dist/activities/execute-cursor/cas-observations.js.map +1 -0
  34. package/dist/activities/execute-cursor/command-provenance.d.ts +62 -0
  35. package/dist/activities/execute-cursor/command-provenance.js +128 -0
  36. package/dist/activities/execute-cursor/command-provenance.js.map +1 -0
  37. package/dist/activities/execute-cursor/exact-apply.d.ts +110 -0
  38. package/dist/activities/execute-cursor/exact-apply.js +204 -0
  39. package/dist/activities/execute-cursor/exact-apply.js.map +1 -0
  40. package/dist/activities/execute-cursor/hook-script.d.ts +53 -24
  41. package/dist/activities/execute-cursor/hook-script.js +310 -47
  42. package/dist/activities/execute-cursor/hook-script.js.map +1 -1
  43. package/dist/activities/execute-cursor/index.d.ts +8 -1
  44. package/dist/activities/execute-cursor/index.js +500 -64
  45. package/dist/activities/execute-cursor/index.js.map +1 -1
  46. package/dist/activities/execute-cursor/message-translator.d.ts +258 -11
  47. package/dist/activities/execute-cursor/message-translator.js +836 -90
  48. package/dist/activities/execute-cursor/message-translator.js.map +1 -1
  49. package/dist/activities/execute-cursor/prompt-builder.d.ts +20 -23
  50. package/dist/activities/execute-cursor/prompt-builder.js +74 -15
  51. package/dist/activities/execute-cursor/prompt-builder.js.map +1 -1
  52. package/dist/activities/execute-cursor/workspace-setup.d.ts +17 -0
  53. package/dist/activities/execute-cursor/workspace-setup.js +212 -33
  54. package/dist/activities/execute-cursor/workspace-setup.js.map +1 -1
  55. package/dist/activities/execute-deep-agent/__test-utils__/gateway-substrate.d.ts +19 -0
  56. package/dist/activities/execute-deep-agent/__test-utils__/gateway-substrate.js +143 -0
  57. package/dist/activities/execute-deep-agent/__test-utils__/gateway-substrate.js.map +1 -0
  58. package/dist/activities/execute-deep-agent/__test-utils__/scripted-model.d.ts +88 -0
  59. package/dist/activities/execute-deep-agent/__test-utils__/scripted-model.js +81 -0
  60. package/dist/activities/execute-deep-agent/__test-utils__/scripted-model.js.map +1 -0
  61. package/dist/activities/execute-deep-agent/approval-file-change.d.ts +47 -0
  62. package/dist/activities/execute-deep-agent/approval-file-change.js +68 -0
  63. package/dist/activities/execute-deep-agent/approval-file-change.js.map +1 -0
  64. package/dist/activities/execute-deep-agent/attachment-injector.d.ts +8 -1
  65. package/dist/activities/execute-deep-agent/attachment-injector.js +7 -7
  66. package/dist/activities/execute-deep-agent/attachment-injector.js.map +1 -1
  67. package/dist/activities/execute-deep-agent/cas-capture-backend.d.ts +42 -0
  68. package/dist/activities/execute-deep-agent/cas-capture-backend.js +47 -0
  69. package/dist/activities/execute-deep-agent/cas-capture-backend.js.map +1 -0
  70. package/dist/activities/execute-deep-agent/cas-capture-observer.d.ts +79 -0
  71. package/dist/activities/execute-deep-agent/cas-capture-observer.js +112 -0
  72. package/dist/activities/execute-deep-agent/cas-capture-observer.js.map +1 -0
  73. package/dist/activities/execute-deep-agent/hitl.d.ts +10 -0
  74. package/dist/activities/execute-deep-agent/hitl.js +5 -1
  75. package/dist/activities/execute-deep-agent/hitl.js.map +1 -1
  76. package/dist/activities/execute-deep-agent/index.d.ts +2 -1
  77. package/dist/activities/execute-deep-agent/index.js +370 -56
  78. package/dist/activities/execute-deep-agent/index.js.map +1 -1
  79. package/dist/activities/execute-deep-agent/inline-publisher.d.ts +7 -1
  80. package/dist/activities/execute-deep-agent/inline-publisher.js +23 -2
  81. package/dist/activities/execute-deep-agent/inline-publisher.js.map +1 -1
  82. package/dist/activities/execute-deep-agent/setup.d.ts +53 -2
  83. package/dist/activities/execute-deep-agent/setup.js +149 -92
  84. package/dist/activities/execute-deep-agent/setup.js.map +1 -1
  85. package/dist/activities/execute-deep-agent/stamp-flowed-rows.d.ts +36 -0
  86. package/dist/activities/execute-deep-agent/stamp-flowed-rows.js +56 -0
  87. package/dist/activities/execute-deep-agent/stamp-flowed-rows.js.map +1 -0
  88. package/dist/activities/execute-deep-agent/status-builder-shared.d.ts +34 -1
  89. package/dist/activities/execute-deep-agent/status-builder-shared.js +26 -25
  90. package/dist/activities/execute-deep-agent/status-builder-shared.js.map +1 -1
  91. package/dist/activities/execute-deep-agent/status-builder.d.ts +11 -5
  92. package/dist/activities/execute-deep-agent/status-builder.js +6 -2
  93. package/dist/activities/execute-deep-agent/status-builder.js.map +1 -1
  94. package/dist/activities/execute-deep-agent/streaming-side-effects.js +2 -19
  95. package/dist/activities/execute-deep-agent/streaming-side-effects.js.map +1 -1
  96. package/dist/activities/execute-deep-agent/streaming.js +3 -15
  97. package/dist/activities/execute-deep-agent/streaming.js.map +1 -1
  98. package/dist/activities/execute-deep-agent/subagent-transformer.d.ts +25 -7
  99. package/dist/activities/execute-deep-agent/subagent-transformer.js +23 -7
  100. package/dist/activities/execute-deep-agent/subagent-transformer.js.map +1 -1
  101. package/dist/activities/execute-deep-agent/subagent-wiring.d.ts +30 -3
  102. package/dist/activities/execute-deep-agent/subagent-wiring.js +29 -3
  103. package/dist/activities/execute-deep-agent/subagent-wiring.js.map +1 -1
  104. package/dist/activities/execute-deep-agent/v3-status-builder.js +6 -2
  105. package/dist/activities/execute-deep-agent/v3-status-builder.js.map +1 -1
  106. package/dist/claimcheck/payload-codec.js +9 -5
  107. package/dist/claimcheck/payload-codec.js.map +1 -1
  108. package/dist/client/stigmer-client.d.ts +2 -0
  109. package/dist/client/stigmer-client.js +2 -0
  110. package/dist/client/stigmer-client.js.map +1 -1
  111. package/dist/middleware/approval-gate.d.ts +85 -4
  112. package/dist/middleware/approval-gate.js +165 -38
  113. package/dist/middleware/approval-gate.js.map +1 -1
  114. package/dist/middleware/types.d.ts +2 -5
  115. package/dist/shared/activity-input.d.ts +43 -0
  116. package/dist/shared/activity-input.js +17 -0
  117. package/dist/shared/activity-input.js.map +1 -0
  118. package/dist/shared/approval-canonicalize.d.ts +19 -0
  119. package/dist/shared/approval-canonicalize.js +119 -0
  120. package/dist/shared/approval-canonicalize.js.map +1 -0
  121. package/dist/shared/approval-fingerprint.d.ts +106 -0
  122. package/dist/shared/approval-fingerprint.js +113 -0
  123. package/dist/shared/approval-fingerprint.js.map +1 -0
  124. package/dist/shared/approval-policy.d.ts +182 -12
  125. package/dist/shared/approval-policy.js +213 -27
  126. package/dist/shared/approval-policy.js.map +1 -1
  127. package/dist/shared/args-preview.d.ts +52 -0
  128. package/dist/shared/args-preview.js +93 -0
  129. package/dist/shared/args-preview.js.map +1 -0
  130. package/dist/shared/artifact-storage.d.ts +19 -1
  131. package/dist/shared/artifact-storage.js +48 -11
  132. package/dist/shared/artifact-storage.js.map +1 -1
  133. package/dist/shared/file-change.d.ts +44 -0
  134. package/dist/shared/file-change.js +57 -0
  135. package/dist/shared/file-change.js.map +1 -0
  136. package/dist/shared/file-tools.d.ts +107 -0
  137. package/dist/shared/file-tools.js +168 -0
  138. package/dist/shared/file-tools.js.map +1 -0
  139. package/dist/shared/filereview/capture.d.ts +202 -0
  140. package/dist/shared/filereview/capture.js +498 -0
  141. package/dist/shared/filereview/capture.js.map +1 -0
  142. package/dist/shared/filereview/cas-substrate.d.ts +190 -0
  143. package/dist/shared/filereview/cas-substrate.js +284 -0
  144. package/dist/shared/filereview/cas-substrate.js.map +1 -0
  145. package/dist/shared/filereview/digest.d.ts +40 -0
  146. package/dist/shared/filereview/digest.js +66 -0
  147. package/dist/shared/filereview/digest.js.map +1 -0
  148. package/dist/shared/filereview/events.d.ts +170 -0
  149. package/dist/shared/filereview/events.js +298 -0
  150. package/dist/shared/filereview/events.js.map +1 -0
  151. package/dist/shared/filereview/git-substrate.d.ts +175 -0
  152. package/dist/shared/filereview/git-substrate.js +439 -0
  153. package/dist/shared/filereview/git-substrate.js.map +1 -0
  154. package/dist/shared/filereview/index.d.ts +11 -0
  155. package/dist/shared/filereview/index.js +12 -0
  156. package/dist/shared/filereview/index.js.map +1 -0
  157. package/dist/shared/filereview/secret-paths.d.ts +63 -0
  158. package/dist/shared/filereview/secret-paths.js +105 -0
  159. package/dist/shared/filereview/secret-paths.js.map +1 -0
  160. package/dist/shared/fingerprint-secret.d.ts +26 -0
  161. package/dist/shared/fingerprint-secret.js +47 -0
  162. package/dist/shared/fingerprint-secret.js.map +1 -0
  163. package/dist/shared/model-client.d.ts +51 -0
  164. package/dist/shared/model-client.js +77 -0
  165. package/dist/shared/model-client.js.map +1 -0
  166. package/dist/shared/plan-artifact.js +0 -2
  167. package/dist/shared/plan-artifact.js.map +1 -1
  168. package/dist/shared/status-offload.d.ts +83 -9
  169. package/dist/shared/status-offload.js +399 -79
  170. package/dist/shared/status-offload.js.map +1 -1
  171. package/dist/shared/status.js +14 -1
  172. package/dist/shared/status.js.map +1 -1
  173. package/dist/shared/tool-kind.d.ts +19 -0
  174. package/dist/shared/tool-kind.js +13 -0
  175. package/dist/shared/tool-kind.js.map +1 -1
  176. package/dist/shared/tool-row.d.ts +88 -0
  177. package/dist/shared/tool-row.js +127 -0
  178. package/dist/shared/tool-row.js.map +1 -0
  179. package/dist/shared/workspace/platform-dir.d.ts +25 -0
  180. package/dist/shared/workspace/platform-dir.js +38 -2
  181. package/dist/shared/workspace/platform-dir.js.map +1 -1
  182. package/dist/workflows/call-agent-orchestrator.js +56 -7
  183. package/dist/workflows/call-agent-orchestrator.js.map +1 -1
  184. package/dist/workflows/connect-mcp-server.d.ts +50 -0
  185. package/dist/workflows/connect-mcp-server.js +136 -15
  186. package/dist/workflows/connect-mcp-server.js.map +1 -1
  187. package/dist/workflows/types.d.ts +8 -0
  188. package/package.json +2 -2
  189. package/src/__test-utils__/approval-contract/contract.ts +224 -0
  190. package/src/__test-utils__/approval-contract/types.ts +179 -0
  191. package/src/__test-utils__/fake-artifact-storage.ts +72 -0
  192. package/src/__tests__/approval-gateway-contract.test.ts +29 -0
  193. package/src/__tests__/claimcheck-codec.test.ts +16 -53
  194. package/src/__tests__/golden-e2e.test.ts +2 -0
  195. package/src/__tests__/runner-token-coordinator.test.ts +3 -3
  196. package/src/activities/__tests__/call-agent-status.test.ts +135 -0
  197. package/src/activities/__tests__/call-llm.test.ts +1 -1
  198. package/src/activities/__tests__/classify-tool-approvals.test.ts +208 -1
  199. package/src/activities/__tests__/discover-mcp-server.test.ts +30 -0
  200. package/src/activities/__tests__/workflow-event-activities.test.ts +2 -1
  201. package/src/activities/call-agent-status.ts +74 -4
  202. package/src/activities/call-llm.ts +18 -63
  203. package/src/activities/classify-tool-approvals.ts +101 -19
  204. package/src/activities/discover-mcp-server.ts +29 -1
  205. package/src/activities/execute-cursor/__test-utils__/cursor-hook-harness.ts +216 -0
  206. package/src/activities/execute-cursor/__test-utils__/gateway-substrate.ts +148 -0
  207. package/src/activities/execute-cursor/__tests__/approval-gate.test.ts +41 -9
  208. package/src/activities/execute-cursor/__tests__/approval-state.test.ts +292 -0
  209. package/src/activities/execute-cursor/__tests__/build-prompt.test.ts +68 -1
  210. package/src/activities/execute-cursor/__tests__/capture-flow.test.ts +1005 -0
  211. package/src/activities/execute-cursor/__tests__/cas-observations.test.ts +187 -0
  212. package/src/activities/execute-cursor/__tests__/coarse-fingerprint.test.ts +97 -0
  213. package/src/activities/execute-cursor/__tests__/command-provenance.test.ts +240 -0
  214. package/src/activities/execute-cursor/__tests__/deny-gate-exact-apply.test.ts +203 -0
  215. package/src/activities/execute-cursor/__tests__/exact-apply.test.ts +375 -0
  216. package/src/activities/execute-cursor/__tests__/hitl-ledger.test.ts +1294 -24
  217. package/src/activities/execute-cursor/__tests__/hitl-resume-history.test.ts +446 -0
  218. package/src/activities/execute-cursor/__tests__/hook-script.test.ts +384 -110
  219. package/src/activities/execute-cursor/__tests__/message-translator.test.ts +171 -25
  220. package/src/activities/execute-cursor/__tests__/sequential-gate-resume.test.ts +189 -0
  221. package/src/activities/execute-cursor/__tests__/tool-result-image.test.ts +44 -23
  222. package/src/activities/execute-cursor/__tests__/workspace-setup.test.ts +190 -10
  223. package/src/activities/execute-cursor/approval-policy.ts +28 -159
  224. package/src/activities/execute-cursor/approval-state.ts +366 -18
  225. package/src/activities/execute-cursor/capture-flow.ts +323 -0
  226. package/src/activities/execute-cursor/cas-observations.ts +204 -0
  227. package/src/activities/execute-cursor/command-provenance.ts +168 -0
  228. package/src/activities/execute-cursor/exact-apply.ts +253 -0
  229. package/src/activities/execute-cursor/hook-script.ts +317 -51
  230. package/src/activities/execute-cursor/index.ts +575 -67
  231. package/src/activities/execute-cursor/message-translator.ts +963 -89
  232. package/src/activities/execute-cursor/prompt-builder.ts +80 -14
  233. package/src/activities/execute-cursor/workspace-setup.ts +257 -42
  234. package/src/activities/execute-deep-agent/__test-utils__/gateway-substrate.ts +180 -0
  235. package/src/activities/execute-deep-agent/__test-utils__/scripted-model.ts +134 -0
  236. package/src/activities/execute-deep-agent/__tests__/approval-file-change.test.ts +84 -0
  237. package/src/activities/execute-deep-agent/__tests__/attachment-injector.test.ts +11 -24
  238. package/src/activities/execute-deep-agent/__tests__/cas-capture-backend.test.ts +64 -0
  239. package/src/activities/execute-deep-agent/__tests__/cas-capture-observer.test.ts +163 -0
  240. package/src/activities/execute-deep-agent/__tests__/hitl-integration.test.ts +8 -5
  241. package/src/activities/execute-deep-agent/__tests__/hitl-resume-approve-all.test.ts +342 -0
  242. package/src/activities/execute-deep-agent/__tests__/hitl-resume-history.test.ts +2 -5
  243. package/src/activities/execute-deep-agent/__tests__/inline-publisher.test.ts +31 -13
  244. package/src/activities/execute-deep-agent/__tests__/sequential-gate-resume.test.ts +349 -0
  245. package/src/activities/execute-deep-agent/__tests__/stamp-flowed-rows.test.ts +119 -0
  246. package/src/activities/execute-deep-agent/__tests__/status-builder.test.ts +12 -11
  247. package/src/activities/execute-deep-agent/__tests__/streaming-v3.test.ts +9 -9
  248. package/src/activities/execute-deep-agent/__tests__/subagent-approval-propagation.test.ts +160 -0
  249. package/src/activities/execute-deep-agent/__tests__/subagent-gitignored-capture.test.ts +213 -0
  250. package/src/activities/execute-deep-agent/__tests__/subagent-transformer.test.ts +3 -6
  251. package/src/activities/execute-deep-agent/__tests__/subagent-wiring.test.ts +84 -1
  252. package/src/activities/execute-deep-agent/__tests__/v3-status-builder.test.ts +4 -1
  253. package/src/activities/execute-deep-agent/approval-file-change.ts +80 -0
  254. package/src/activities/execute-deep-agent/attachment-injector.ts +20 -11
  255. package/src/activities/execute-deep-agent/cas-capture-backend.ts +66 -0
  256. package/src/activities/execute-deep-agent/cas-capture-observer.ts +125 -0
  257. package/src/activities/execute-deep-agent/hitl.ts +15 -1
  258. package/src/activities/execute-deep-agent/index.ts +434 -64
  259. package/src/activities/execute-deep-agent/inline-publisher.ts +27 -4
  260. package/src/activities/execute-deep-agent/setup.ts +223 -125
  261. package/src/activities/execute-deep-agent/stamp-flowed-rows.ts +64 -0
  262. package/src/activities/execute-deep-agent/status-builder-shared.ts +62 -23
  263. package/src/activities/execute-deep-agent/status-builder.ts +19 -7
  264. package/src/activities/execute-deep-agent/streaming-side-effects.ts +2 -16
  265. package/src/activities/execute-deep-agent/streaming.ts +3 -13
  266. package/src/activities/execute-deep-agent/subagent-transformer.ts +53 -13
  267. package/src/activities/execute-deep-agent/subagent-wiring.ts +50 -3
  268. package/src/activities/execute-deep-agent/v3-status-builder.ts +8 -2
  269. package/src/claimcheck/payload-codec.ts +8 -8
  270. package/src/client/stigmer-client.ts +9 -1
  271. package/src/middleware/__tests__/approval-gate.test.ts +488 -4
  272. package/src/middleware/approval-gate.ts +247 -38
  273. package/src/middleware/types.ts +5 -5
  274. package/src/shared/__tests__/activity-input.test.ts +78 -0
  275. package/src/shared/__tests__/approval-canonicalize.test.ts +106 -0
  276. package/src/shared/__tests__/approval-fingerprint.test.ts +115 -0
  277. package/src/shared/__tests__/approval-policy.test.ts +274 -40
  278. package/src/shared/__tests__/args-preview.test.ts +78 -0
  279. package/src/shared/__tests__/artifact-storage-extended.test.ts +62 -10
  280. package/src/shared/__tests__/artifact-storage.test.ts +123 -11
  281. package/src/shared/__tests__/file-change.test.ts +85 -0
  282. package/src/shared/__tests__/file-tools.test.ts +90 -0
  283. package/src/shared/__tests__/fingerprint-secret.test.ts +51 -0
  284. package/src/shared/__tests__/lease-scope-corpus.test.ts +56 -0
  285. package/src/shared/__tests__/model-client.test.ts +162 -0
  286. package/src/shared/__tests__/plan-artifact.test.ts +11 -26
  287. package/src/shared/__tests__/policy-source-corpus.test.ts +58 -0
  288. package/src/shared/__tests__/status-offload.test.ts +573 -16
  289. package/src/shared/__tests__/status.test.ts +4 -5
  290. package/src/shared/__tests__/tool-kind.test.ts +24 -1
  291. package/src/shared/__tests__/tool-row.test.ts +221 -0
  292. package/src/shared/activity-input.ts +57 -0
  293. package/src/shared/approval-canonicalize.ts +159 -0
  294. package/src/shared/approval-fingerprint.ts +148 -0
  295. package/src/shared/approval-policy.ts +303 -27
  296. package/src/shared/args-preview.ts +98 -0
  297. package/src/shared/artifact-storage.ts +62 -11
  298. package/src/shared/checkpointer/__tests__/http-saver.test.ts +1 -2
  299. package/src/shared/file-change.ts +64 -0
  300. package/src/shared/file-tools.ts +169 -0
  301. package/src/shared/filereview/__tests__/capture.test.ts +856 -0
  302. package/src/shared/filereview/__tests__/cas-substrate.test.ts +404 -0
  303. package/src/shared/filereview/__tests__/digest.test.ts +100 -0
  304. package/src/shared/filereview/__tests__/events.test.ts +245 -0
  305. package/src/shared/filereview/__tests__/git-substrate.test.ts +362 -0
  306. package/src/shared/filereview/__tests__/proxy-reconcile.test.ts +286 -0
  307. package/src/shared/filereview/__tests__/secret-paths.test.ts +121 -0
  308. package/src/shared/filereview/capture.ts +727 -0
  309. package/src/shared/filereview/cas-substrate.ts +401 -0
  310. package/src/shared/filereview/digest.ts +83 -0
  311. package/src/shared/filereview/events.ts +449 -0
  312. package/src/shared/filereview/git-substrate.ts +555 -0
  313. package/src/shared/filereview/index.ts +60 -0
  314. package/src/shared/filereview/secret-paths.ts +121 -0
  315. package/src/shared/fingerprint-secret.ts +53 -0
  316. package/src/shared/model-client.ts +122 -0
  317. package/src/shared/plan-artifact.ts +0 -2
  318. package/src/shared/status-offload.ts +433 -77
  319. package/src/shared/status.ts +13 -0
  320. package/src/shared/tool-kind.ts +33 -0
  321. package/src/shared/tool-row.ts +135 -0
  322. package/src/shared/workspace/platform-dir.ts +41 -2
  323. package/src/workflow-engine/__tests__/golden-execution.test.ts +35 -18
  324. package/src/workflow-engine/__tests__/tasks/try.test.ts +1 -1
  325. package/src/workflows/__tests__/connect-mcp-server.test.ts +304 -29
  326. package/src/workflows/call-agent-orchestrator.ts +53 -6
  327. package/src/workflows/connect-mcp-server.ts +179 -24
  328. package/src/workflows/types.ts +8 -0
@@ -3,8 +3,10 @@
3
3
  *
4
4
  * Tool outputs (an MCP screenshot's base64 image, a giant accessibility-tree
5
5
  * dump, a multi-MB shell log, a huge file write) are stored inline in
6
- * `ToolCall.result`/`args_preview`, which live in `status.messages` and are
7
- * re-serialized whole on every `persistStatus` -> `updateStatus` gRPC call.
6
+ * `ToolCall.result`/`args_preview`, which live in `status.messages` AND in
7
+ * each `status.sub_agent_executions[].messages` (a delegated tool call's
8
+ * result is just as unbounded), and are re-serialized whole on every
9
+ * `persistStatus` -> `updateStatus` gRPC call.
8
10
  * Left unchecked, a single large result pushes the message past the server's
9
11
  * 4 MiB gRPC receive cap; the call fails with `resource_exhausted`, progress
10
12
  * stops persisting, and the live UI freezes mid-execution.
@@ -20,12 +22,22 @@
20
22
  * Idempotent and content-hash-deduped so the throttled, repeated persists
21
23
  * (and result re-inflation by mergeToolCallEvent) upload each blob once.
22
24
  *
23
- * 2. enforceStatusSizeLimit — an aggregate, type-agnostic backstop that runs
25
+ * 2. offloadCandidateChangesToFit — an aggregate, storage-backed step for the
26
+ * file-review ledger: if the status still exceeds the soft cap after (1),
27
+ * it offloads the largest still-inline captured before/after bodies to
28
+ * retrievable refs (biggest-first) until it fits, so a captured file stays
29
+ * REVIEWABLE (the UI lazily fetches the ref) instead of being dropped. The
30
+ * persisted body is a display projection — reconcile sources bytes from the
31
+ * git refs / CAS manifest, never this body — so this is correctness-neutral.
32
+ *
33
+ * 3. enforceStatusSizeLimit — an aggregate, type-agnostic backstop that runs
24
34
  * even when no artifact storage is available: if the encoded status still
25
35
  * exceeds a soft cap (comfortably under 4 MiB), it elides the largest
26
- * remaining inline fields in place until the payload fits.
36
+ * remaining inline fields in place until the payload fits. For file-review
37
+ * bodies this is the LAST resort (no storage, or (2) could not free enough):
38
+ * the body is dropped and the file marked SIZE_ELIDED / incomplete.
27
39
  *
28
- * Both operate ONLY on what is persisted/streamed; the agent's working context
40
+ * All operate ONLY on what is persisted/streamed; the agent's working context
29
41
  * is managed by the harness/SDK separately, so reasoning is unaffected.
30
42
  */
31
43
 
@@ -36,8 +48,11 @@ import {
36
48
  } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/api_pb";
37
49
  import type { AgentExecutionStatus } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/api_pb";
38
50
  import { ToolCallOutputRefSchema } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/message_pb";
39
- import type { ToolCall } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/message_pb";
51
+ import type { AgentMessage, FileContent, ToolCall } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/message_pb";
52
+ import type { CapturedFileChange } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/filereview_pb";
53
+ import { FileReviewBlockReason } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/enum_pb";
40
54
  import type { ArtifactStorage } from "./artifact-storage.js";
55
+ import { deriveDiffCompleteness } from "./filereview/events.js";
41
56
 
42
57
  /**
43
58
  * A single tool output (result or args_preview) larger than this many bytes is
@@ -47,6 +62,14 @@ import type { ArtifactStorage } from "./artifact-storage.js";
47
62
  */
48
63
  export const INLINE_TOOL_OUTPUT_MAX_BYTES = 256 * 1024;
49
64
 
65
+ /**
66
+ * A single FileChange before/after body larger than this is offloaded to
67
+ * artifact storage. Smaller than the 256 KiB tool-output cap because a file
68
+ * change can carry two bodies (before + after) per change and several changes
69
+ * per tool call, so the aggregate would otherwise climb quickly.
70
+ */
71
+ export const INLINE_FILE_CONTENT_MAX_BYTES = 128 * 1024;
72
+
50
73
  /** Head of an offloaded text result kept inline for an at-a-glance preview. */
51
74
  export const TEXT_PREVIEW_HEAD_CHARS = 4_000;
52
75
 
@@ -62,8 +85,14 @@ export const STATUS_PAYLOAD_SOFT_LIMIT_BYTES = 3 * 1024 * 1024;
62
85
  */
63
86
  export const STATUS_PAYLOAD_HARD_LIMIT_BYTES = 2 * 1024 * 1024;
64
87
 
65
- /** Marker left in place of an aggregate-elided inline field. */
66
- const ELISION_MARKER = "[output elided to keep status under the size limit]";
88
+ /**
89
+ * Marker left in place of an aggregate-elided inline field. Exported so the
90
+ * resume-time exact-apply resolver can recognize (and refuse to write) an elided
91
+ * body rather than corrupting a file with the marker text — the lossy elision is
92
+ * the one case where the exact approved bytes are unrecoverable and exact-apply
93
+ * must fall back. The two sides cannot drift because they share this constant.
94
+ */
95
+ export const ELISION_MARKER = "[output elided to keep status under the size limit]";
67
96
 
68
97
  /** Below this size an inline field is not worth eliding (the marker is ~50B). */
69
98
  const ELISION_MIN_BYTES = 1_024;
@@ -73,6 +102,8 @@ export interface ToolOutputOffloadContext {
73
102
  readonly executionId: string;
74
103
  /** Override the per-result byte threshold (tests use a small value). */
75
104
  readonly maxInlineBytes?: number;
105
+ /** Override the per-file-content-body byte threshold (tests use a small value). */
106
+ readonly maxInlineFileBytes?: number;
76
107
  }
77
108
 
78
109
  interface ImagePayload {
@@ -98,6 +129,29 @@ function formatBytes(n: number): string {
98
129
  return `${n} B`;
99
130
  }
100
131
 
132
+ /**
133
+ * Every message list carried on the status: the parent transcript plus each
134
+ * sub-agent's nested transcript. Sub-agent tool calls hold real results (a
135
+ * delegated read/grep/screenshot can be as large as a top-level one), so every
136
+ * size-bounding pass in this module walks THIS set — a location covered by
137
+ * offload but not elision (or vice versa) would be a hole in the persist
138
+ * boundary's bounded-payload guarantee.
139
+ */
140
+ function allMessageLists(status: AgentExecutionStatus): readonly (readonly AgentMessage[])[] {
141
+ return [status.messages, ...status.subAgentExecutions.map((sa) => sa.messages)];
142
+ }
143
+
144
+ /** Every tool call in the status, across the parent and all sub-agents. */
145
+ function allToolCalls(status: AgentExecutionStatus): ToolCall[] {
146
+ const out: ToolCall[] = [];
147
+ for (const messages of allMessageLists(status)) {
148
+ for (const msg of messages) {
149
+ for (const tc of msg.toolCalls) out.push(tc);
150
+ }
151
+ }
152
+ return out;
153
+ }
154
+
101
155
  function extFromMime(mimeType: string): string {
102
156
  switch (mimeType) {
103
157
  case "image/png": return "png";
@@ -109,65 +163,94 @@ function extFromMime(mimeType: string): string {
109
163
  }
110
164
  }
111
165
 
166
+ /** Match an exact `data:image/...;base64,...` URL (the whole string). */
112
167
  function matchDataUrl(s: string): ImagePayload | null {
113
168
  const m = s.match(/^data:(image\/[a-zA-Z0-9.+-]+);base64,([\s\S]+)$/);
114
169
  if (!m) return null;
115
170
  return { mimeType: m[1], base64: m[2].replace(/\s+/g, "") };
116
171
  }
117
172
 
118
- function contentBlocks(parsed: unknown): unknown[] {
119
- if (Array.isArray(parsed)) return parsed;
120
- if (parsed && typeof parsed === "object") {
121
- const obj = parsed as Record<string, unknown>;
122
- if (Array.isArray(obj.content)) return obj.content;
123
- // Defensive: a serialized LangChain ToolMessage envelope nests its blocks
124
- // under kwargs.content. The deep-agent extractor already normalizes image
125
- // results to a top-level array (status-builder-shared.ts), so this branch is
126
- // insurance against future shape drift — not the primary path.
127
- const kwargs = obj.kwargs;
128
- if (kwargs && typeof kwargs === "object" && Array.isArray((kwargs as Record<string, unknown>).content)) {
129
- return (kwargs as Record<string, unknown>).content as unknown[];
130
- }
131
- }
132
- return [];
173
+ /**
174
+ * Find a `data:image/...;base64,...` URL anywhere in a string — whether it IS
175
+ * the whole result or is embedded in surrounding text/JSON. A data URL is an
176
+ * unambiguous image signal, so scanning is safe (no false positives on ordinary
177
+ * text). The base64 run is bounded by the first non-base64 character (e.g. a
178
+ * closing JSON quote), so an embedded URL is extracted cleanly.
179
+ */
180
+ function findDataUrlInString(s: string): ImagePayload | null {
181
+ const m = s.match(/data:(image\/[a-zA-Z0-9.+-]+);base64,([A-Za-z0-9+/=\s]+)/);
182
+ if (!m) return null;
183
+ return { mimeType: m[1], base64: m[2].replace(/\s+/g, "") };
133
184
  }
134
185
 
135
186
  /**
136
- * Best-effort extraction of an inline base64 image from a tool result string.
137
- * Handles a raw data URL, MCP image content blocks ({type:"image", data,
138
- * mimeType}) and OpenAI-style image_url blocks. Returns null for non-image or
139
- * unparseable results so the caller falls back to text offload.
187
+ * Build an ImagePayload from a block's `data` + optional mime hint. Accepts the
188
+ * shapes the Cursor SDK actually delivers for an image block's `data`:
189
+ * - Node Buffer-JSON ({ type:"Buffer", data:number[] }) the runtime shape
190
+ * confirmed from a real cursor-harness get_app_state result. (The SDK's .d.ts
191
+ * types `data` as a string, but at runtime image bytes serialize as
192
+ * Buffer-JSON, so both must be handled.)
193
+ * - a `data:` URL string, or
194
+ * - a bare base64 string.
195
+ * Anything else (a file path, a number) yields null, so only an explicit image
196
+ * signal ever matches. The mime hint defaults to image/png when absent — Cursor
197
+ * MCP image blocks frequently omit mimeType.
140
198
  */
141
- export function detectImagePayload(result: string): ImagePayload | null {
142
- const direct = matchDataUrl(result.trim());
143
- if (direct) return direct;
199
+ function imageFromData(data: unknown, mime: unknown): ImagePayload | null {
200
+ const mimeType = typeof mime === "string" && mime.length > 0 ? mime : "image/png";
144
201
 
145
- let parsed: unknown;
146
- try {
147
- parsed = JSON.parse(result);
148
- } catch {
202
+ if (data && typeof data === "object") {
203
+ const d = data as Record<string, unknown>;
204
+ if (d.type === "Buffer" && Array.isArray(d.data)) {
205
+ try {
206
+ const base64 = Buffer.from(d.data as number[]).toString("base64");
207
+ return base64.length > 0 ? { mimeType, base64 } : null;
208
+ } catch {
209
+ return null;
210
+ }
211
+ }
149
212
  return null;
150
213
  }
151
214
 
152
- for (const block of contentBlocks(parsed)) {
153
- if (!block || typeof block !== "object") continue;
154
- const obj = block as Record<string, unknown>;
155
- const type = typeof obj.type === "string" ? obj.type : "";
156
- if (type !== "image" && type !== "image_url") continue;
157
-
158
- const raw =
159
- typeof obj.data === "string" ? obj.data :
160
- typeof obj.image === "string" ? obj.image :
161
- undefined;
162
- if (raw) {
163
- const asUrl = matchDataUrl(raw);
164
- if (asUrl) return asUrl;
165
- const mimeType =
166
- typeof obj.mimeType === "string" ? obj.mimeType :
167
- typeof obj.mime_type === "string" ? obj.mime_type :
168
- "image/png";
169
- return { mimeType, base64: raw.replace(/\s+/g, "") };
170
- }
215
+ if (typeof data !== "string" || data.length === 0) return null;
216
+ const asUrl = matchDataUrl(data);
217
+ if (asUrl) return asUrl;
218
+ return { mimeType, base64: data.replace(/\s+/g, "") };
219
+ }
220
+
221
+ /**
222
+ * Extract an image from a single object IF it is a recognized image block.
223
+ * Recognized shapes, all of which carry an explicit image marker (so an
224
+ * arbitrary object never matches):
225
+ * - Cursor SDK MCP block: { image: { data: <base64|dataUrl>, mimeType? } }
226
+ * - Anthropic/MCP-style block: { type: "image", data: <base64>, mimeType? }
227
+ * - OpenAI-style block: { type: "image_url", image_url: { url: <dataUrl> } }
228
+ * (also tolerates { type: "image", image: <base64> })
229
+ *
230
+ * Returns null when no image marker is present, leaving the recursive walk to
231
+ * keep searching siblings/children.
232
+ */
233
+ function imageFromBlock(obj: Record<string, unknown>): ImagePayload | null {
234
+ // Cursor SDK shape: the image rides under a nested `image` object. This is the
235
+ // exact shape @cursor/sdk uses for MCP image content (see conversation-types),
236
+ // which canonicalizeImageResult normalizes — but only when the result reaches
237
+ // it as an object. A result delivered already-serialized (a string) bypasses
238
+ // that, so detection must recognize this shape directly.
239
+ if (obj.image && typeof obj.image === "object") {
240
+ const img = obj.image as Record<string, unknown>;
241
+ const payload = imageFromData(img.data, img.mimeType ?? img.mime_type);
242
+ if (payload) return payload;
243
+ }
244
+
245
+ const type = typeof obj.type === "string" ? obj.type : "";
246
+ if (type === "image" || type === "image_url") {
247
+ const inline = imageFromData(
248
+ typeof obj.data === "string" ? obj.data
249
+ : typeof obj.image === "string" ? obj.image
250
+ : undefined,
251
+ obj.mimeType ?? obj.mime_type,
252
+ );
253
+ if (inline) return inline;
171
254
 
172
255
  const imageUrl = obj.image_url;
173
256
  if (imageUrl && typeof imageUrl === "object") {
@@ -181,6 +264,67 @@ export function detectImagePayload(result: string): ImagePayload | null {
181
264
  return null;
182
265
  }
183
266
 
267
+ /**
268
+ * Walk a parsed JSON value depth-first, returning the first recognized image
269
+ * block. Recursing (rather than only checking the top level or a `content`
270
+ * array) is what makes detection robust to HOW a harness wraps the image: a
271
+ * multimodal MCP result may arrive as a top-level array, under `value.content`,
272
+ * under `kwargs.content`, or nested deeper still. Because imageFromBlock
273
+ * requires an explicit image marker, the walk never misclassifies ordinary
274
+ * nested data as an image.
275
+ */
276
+ function findImageInValue(value: unknown): ImagePayload | null {
277
+ if (Array.isArray(value)) {
278
+ for (const item of value) {
279
+ const found = findImageInValue(item);
280
+ if (found) return found;
281
+ }
282
+ return null;
283
+ }
284
+ if (value && typeof value === "object") {
285
+ const obj = value as Record<string, unknown>;
286
+ const direct = imageFromBlock(obj);
287
+ if (direct) return direct;
288
+ for (const child of Object.values(obj)) {
289
+ const found = findImageInValue(child);
290
+ if (found) return found;
291
+ }
292
+ }
293
+ return null;
294
+ }
295
+
296
+ /**
297
+ * Best-effort extraction of an inline image from a tool result string.
298
+ *
299
+ * An MCP tool that returns an image (e.g. a computer-use screenshot) must be
300
+ * lifted into a renderable `ToolCallOutputRef`; otherwise it persists as text
301
+ * and the UI shows raw JSON / "view full output" instead of the picture. The
302
+ * image can arrive in many wrappers depending on the harness and whether the
303
+ * result was pre-serialized, so detection looks for an UNAMBIGUOUS image signal
304
+ * rather than a fixed envelope position:
305
+ *
306
+ * 1. a `data:image/*;base64,...` URL anywhere in the string, then
307
+ * 2. a recognized image block at any depth of a JSON result
308
+ * (see {@link imageFromBlock}).
309
+ *
310
+ * Returns null for non-image or unparseable results so the caller falls back to
311
+ * text offload. It deliberately does NOT treat a bare base64 string with no
312
+ * image marker as an image — that would misclassify legitimate large text
313
+ * (logs, base64-encoded files) as pictures.
314
+ */
315
+ export function detectImagePayload(result: string): ImagePayload | null {
316
+ const embeddedUrl = findDataUrlInString(result);
317
+ if (embeddedUrl) return embeddedUrl;
318
+
319
+ let parsed: unknown;
320
+ try {
321
+ parsed = JSON.parse(result);
322
+ } catch {
323
+ return null;
324
+ }
325
+ return findImageInValue(parsed);
326
+ }
327
+
184
328
  function collapsedResultFor(ref: { isImage: boolean; sizeBytes: bigint; truncatedPreview: string }): string {
185
329
  if (ref.isImage) {
186
330
  return `[image output — ${formatBytes(Number(ref.sizeBytes))}, view inline]`;
@@ -216,10 +360,8 @@ async function maybeOffloadToolCall(
216
360
  const bytes = Buffer.from(image.base64, "base64");
217
361
  const key = `artifacts/${ctx.executionId}/toolcalls/${tc.id}.${extFromMime(image.mimeType)}`;
218
362
  await ctx.artifactStorage.upload(key, bytes, image.mimeType);
219
- const downloadUrl = await ctx.artifactStorage.getDownloadUrl(key);
220
363
  tc.outputRef = create(ToolCallOutputRefSchema, {
221
364
  storageKey: key,
222
- downloadUrl,
223
365
  sizeBytes: BigInt(bytes.length),
224
366
  contentHash: hash,
225
367
  mimeType: image.mimeType,
@@ -237,10 +379,8 @@ async function maybeOffloadToolCall(
237
379
  const content = Buffer.from(result, "utf8");
238
380
  const key = `artifacts/${ctx.executionId}/toolcalls/${tc.id}.txt`;
239
381
  await ctx.artifactStorage.upload(key, content, "text/plain");
240
- const downloadUrl = await ctx.artifactStorage.getDownloadUrl(key);
241
382
  tc.outputRef = create(ToolCallOutputRefSchema, {
242
383
  storageKey: key,
243
- downloadUrl,
244
384
  sizeBytes: BigInt(content.length),
245
385
  contentHash: hash,
246
386
  mimeType: "text/plain",
@@ -250,39 +390,216 @@ async function maybeOffloadToolCall(
250
390
  tc.result = collapsedResultFor(tc.outputRef);
251
391
  }
252
392
 
393
+ /**
394
+ * Spill one side (before/after) of a file change to artifact storage when its
395
+ * inline body exceeds the cap, replacing the inline body with a ref carrying a
396
+ * head preview. A side that is absent, already a ref (offloaded on a prior
397
+ * persist), or under the cap is left untouched — the `case === "ref"` check
398
+ * makes this idempotent across the throttled, repeated persists.
399
+ */
400
+ async function maybeOffloadFileContent(
401
+ content: FileContent | undefined,
402
+ key: string,
403
+ ctx: ToolOutputOffloadContext,
404
+ maxBytes: number,
405
+ ): Promise<void> {
406
+ if (!content || content.body.case !== "inline") return;
407
+ const text = content.body.value;
408
+ if (byteLen(text) <= maxBytes) return;
409
+
410
+ const bytes = Buffer.from(text, "utf8");
411
+ await ctx.artifactStorage.upload(key, bytes, "text/plain");
412
+ content.body = {
413
+ case: "ref",
414
+ value: create(ToolCallOutputRefSchema, {
415
+ storageKey: key,
416
+ sizeBytes: BigInt(bytes.length),
417
+ contentHash: sha256(text),
418
+ mimeType: "text/plain",
419
+ isImage: false,
420
+ truncatedPreview: headChars(text, TEXT_PREVIEW_HEAD_CHARS),
421
+ }),
422
+ };
423
+ }
424
+
425
+ /**
426
+ * Every captured file change carried on a CANDIDATE_CAPTURED event in the
427
+ * file_review ledger. The before/after bodies live HERE (not on tool calls)
428
+ * under the apply-then-review model, so the persist-boundary size guards must
429
+ * cover this location too — otherwise a large captured file silently pushes the
430
+ * status past the gRPC cap (the freeze this module exists to prevent).
431
+ */
432
+ function candidateChanges(status: AgentExecutionStatus): CapturedFileChange[] {
433
+ const out: CapturedFileChange[] = [];
434
+ for (const ev of status.fileReviewEventStream?.events ?? []) {
435
+ if (ev.payload.case === "candidateCaptured") {
436
+ out.push(...ev.payload.value.changes);
437
+ }
438
+ }
439
+ return out;
440
+ }
441
+
442
+ /** Offload oversized before/after bodies of every captured file-review change. */
443
+ async function maybeOffloadCandidateChanges(
444
+ status: AgentExecutionStatus,
445
+ ctx: ToolOutputOffloadContext,
446
+ maxBytes: number,
447
+ ): Promise<void> {
448
+ const changes = candidateChanges(status);
449
+ for (const change of changes) {
450
+ const base = `artifacts/${ctx.executionId}/filereview/${change.id}`;
451
+ await maybeOffloadFileContent(change.before, `${base}.before.txt`, ctx, maxBytes);
452
+ await maybeOffloadFileContent(change.after, `${base}.after.txt`, ctx, maxBytes);
453
+ }
454
+ }
455
+
253
456
  /**
254
457
  * Offload every oversized tool result in the status to artifact storage,
255
458
  * replacing the inline value with a short head + ToolCallOutputRef. Per-tool
256
459
  * failures fall back to an inline truncation (a bounded result beats a failed
257
460
  * persist) and never throw, so a storage hiccup cannot fail the execution.
461
+ *
462
+ * File-review before/after bodies (on the CANDIDATE_CAPTURED ledger events) are
463
+ * offloaded in the same pass — a tool can produce a small result yet a large
464
+ * captured diff. A file-review offload failure is non-fatal — the body stays
465
+ * inline and the aggregate backstop (enforceStatusSizeLimit) drops it if needed.
258
466
  */
259
467
  export async function offloadOversizedToolOutputs(
260
468
  status: AgentExecutionStatus,
261
469
  ctx: ToolOutputOffloadContext,
262
470
  ): Promise<void> {
263
471
  const maxBytes = ctx.maxInlineBytes ?? INLINE_TOOL_OUTPUT_MAX_BYTES;
264
- for (const msg of status.messages) {
265
- for (const tc of msg.toolCalls) {
266
- try {
267
- await maybeOffloadToolCall(tc, ctx, maxBytes);
268
- } catch (err) {
269
- const original = tc.result ?? "";
270
- tc.result =
271
- headChars(original, TEXT_PREVIEW_HEAD_CHARS) +
272
- `\n\n[output truncated — offload failed: ${err instanceof Error ? err.message : String(err)}]`;
273
- console.warn(
274
- `[status-offload] execution=${ctx.executionId} tool=${tc.name} ` +
275
- `offload failed (non-fatal); truncated inline`,
276
- );
277
- }
472
+ const maxFileBytes = ctx.maxInlineFileBytes ?? INLINE_FILE_CONTENT_MAX_BYTES;
473
+ for (const tc of allToolCalls(status)) {
474
+ try {
475
+ await maybeOffloadToolCall(tc, ctx, maxBytes);
476
+ } catch (err) {
477
+ const original = tc.result ?? "";
478
+ tc.result =
479
+ headChars(original, TEXT_PREVIEW_HEAD_CHARS) +
480
+ `\n\n[output truncated — offload failed: ${err instanceof Error ? err.message : String(err)}]`;
481
+ console.warn(
482
+ `[status-offload] execution=${ctx.executionId} tool=${tc.name} ` +
483
+ `offload failed (non-fatal); truncated inline`,
484
+ );
278
485
  }
279
486
  }
487
+
488
+ // File-review ledger: offload oversized captured before/after bodies the same
489
+ // way (the proto's contract is "offloaded before the candidate event is
490
+ // persisted"). A failure is non-fatal — the body stays inline and the
491
+ // mark-incomplete backstop handles it without corrupting the ledger.
492
+ try {
493
+ await maybeOffloadCandidateChanges(status, ctx, maxFileBytes);
494
+ } catch {
495
+ console.warn(
496
+ `[status-offload] execution=${ctx.executionId} ` +
497
+ `file-review change offload failed (non-fatal); left inline for the size backstop`,
498
+ );
499
+ }
500
+ }
501
+
502
+ /**
503
+ * Aggregate-budget offload of file-review candidate bodies (async, storage-backed).
504
+ *
505
+ * The per-item pass ({@link offloadOversizedToolOutputs}) only offloads a body
506
+ * over {@link INLINE_FILE_CONTENT_MAX_BYTES}; many mid-sized captured files (each
507
+ * under that per-file cap) can still sum the whole status past the soft limit.
508
+ * Rather than let the storage-less backstop ({@link enforceStatusSizeLimit}) DROP
509
+ * those bodies — which sets `diff_complete=false` and blocks approval, turning a
510
+ * reviewable change discard-only — this step offloads the largest still-inline
511
+ * captured bodies to retrievable refs (biggest-first) until the status fits. The
512
+ * review UI then lazily fetches each ref via getArtifactContent exactly as it
513
+ * already does for a >128 KiB file, and the change stays reviewable
514
+ * (`diff_complete` is untouched, so the set rollup is unchanged).
515
+ *
516
+ * Correctness: the persisted before/after body is a DISPLAY projection —
517
+ * reconcile sources the approved bytes from the pinned git refs / CAS manifest
518
+ * and verifies them against the enforcement digests (`before_sha256`/
519
+ * `after_sha256`), never this body (see {@link ../filereview/capture.js}). So
520
+ * converting a body to a ref (or, in the backstop, dropping it) cannot change
521
+ * what is applied on approval.
522
+ *
523
+ * Non-fatal per side: a storage failure leaves that body inline for the backstop
524
+ * to drop. Returns true if any body was offloaded. Called only for a status that
525
+ * actually carries file-review events (the caller guards on that), and a no-op
526
+ * (single encode) when the status already fits.
527
+ */
528
+ export async function offloadCandidateChangesToFit(
529
+ status: AgentExecutionStatus,
530
+ ctx: ToolOutputOffloadContext,
531
+ softLimitBytes: number = STATUS_PAYLOAD_SOFT_LIMIT_BYTES,
532
+ ): Promise<boolean> {
533
+ if (encodedSize(status) <= softLimitBytes) return false;
534
+
535
+ // Every still-inline captured side worth offloading, paired with its stable
536
+ // artifact key (identical to maybeOffloadCandidateChanges so a later persist is
537
+ // idempotent) and its byte size. ELISION_MIN_BYTES is the same "worth it"
538
+ // threshold the drop backstop uses, so the two agree on what is large enough.
539
+ interface InlineCandidateSide {
540
+ readonly content: FileContent;
541
+ readonly key: string;
542
+ readonly bytes: number;
543
+ }
544
+ const sides: InlineCandidateSide[] = [];
545
+ for (const change of candidateChanges(status)) {
546
+ const base = `artifacts/${ctx.executionId}/filereview/${change.id}`;
547
+ if (change.before?.body.case === "inline" && byteLen(change.before.body.value) > ELISION_MIN_BYTES) {
548
+ sides.push({ content: change.before, key: `${base}.before.txt`, bytes: byteLen(change.before.body.value) });
549
+ }
550
+ if (change.after?.body.case === "inline" && byteLen(change.after.body.value) > ELISION_MIN_BYTES) {
551
+ sides.push({ content: change.after, key: `${base}.after.txt`, bytes: byteLen(change.after.body.value) });
552
+ }
553
+ }
554
+
555
+ // Largest first: shed the most bytes per upload and offload the fewest bodies.
556
+ sides.sort((a, b) => b.bytes - a.bytes);
557
+
558
+ let offloadedAny = false;
559
+ for (const side of sides) {
560
+ if (encodedSize(status) <= softLimitBytes) break;
561
+ try {
562
+ // Pre-filtered to inline & over the threshold, so this always offloads;
563
+ // the shared helper keeps the ref shape + the `case === "ref"` idempotency.
564
+ await maybeOffloadFileContent(side.content, side.key, ctx, ELISION_MIN_BYTES);
565
+ offloadedAny = true;
566
+ } catch {
567
+ // Non-fatal: leave this body inline for enforceStatusSizeLimit to drop.
568
+ console.warn(
569
+ `[status-offload] execution=${ctx.executionId} ` +
570
+ `aggregate file-review offload failed for ${side.key} (non-fatal); ` +
571
+ `left inline for the size backstop`,
572
+ );
573
+ }
574
+ }
575
+
576
+ return offloadedAny;
280
577
  }
281
578
 
282
579
  function encodedSize(status: AgentExecutionStatus): number {
283
580
  return toBinary(AgentExecutionStatusSchema, status).length;
284
581
  }
285
582
 
583
+ /**
584
+ * Drop a captured change's oversized inline before/after bodies (replacing them
585
+ * with nothing), returning true if either side was large enough to drop. Used by
586
+ * the backstop for file-review bodies, where overwriting with the elision marker
587
+ * would corrupt the authoritative content — dropping + marking incomplete is the
588
+ * safe alternative (bytes are re-sourced from refs/re-capture on reconcile).
589
+ */
590
+ function dropInlineBodiesIfLarge(change: CapturedFileChange): boolean {
591
+ let dropped = false;
592
+ if (change.before?.body.case === "inline" && byteLen(change.before.body.value) > ELISION_MIN_BYTES) {
593
+ change.before = undefined;
594
+ dropped = true;
595
+ }
596
+ if (change.after?.body.case === "inline" && byteLen(change.after.body.value) > ELISION_MIN_BYTES) {
597
+ change.after = undefined;
598
+ dropped = true;
599
+ }
600
+ return dropped;
601
+ }
602
+
286
603
  /**
287
604
  * Aggregate, type-agnostic backstop. If the encoded status exceeds
288
605
  * `softLimitBytes`, elide the largest inline tool fields (and, as a last
@@ -296,10 +613,7 @@ export function enforceStatusSizeLimit(
296
613
  ): boolean {
297
614
  if (encodedSize(status) <= softLimitBytes) return false;
298
615
 
299
- const toolCalls: ToolCall[] = [];
300
- for (const msg of status.messages) {
301
- for (const tc of msg.toolCalls) toolCalls.push(tc);
302
- }
616
+ const toolCalls = allToolCalls(status);
303
617
  // Largest inline footprint first so we shed the most bytes per elision.
304
618
  toolCalls.sort(
305
619
  (a, b) =>
@@ -324,9 +638,51 @@ export function enforceStatusSizeLimit(
324
638
  }
325
639
  }
326
640
 
327
- // Last resort: oversized message content (e.g. a huge AI response).
641
+ // File-review ledger bodies the storage-less LAST resort. When storage is
642
+ // available, offloadCandidateChangesToFit has already turned oversized captured
643
+ // bodies into retrievable refs (kept reviewable); a body still inline here means
644
+ // there was no storage, or offloading everything still did not free enough.
645
+ // Unlike a tool output, we never overwrite the captured body with the elision
646
+ // marker — the review renders this body, so a marker would show corrupt content;
647
+ // instead we DROP it and mark the file incomplete (SIZE_ELIDED). Reconcile
648
+ // sources bytes from the git refs / CAS manifest (never this display body), and
649
+ // the review surface blocks approval of an incomplete diff. This trades
650
+ // reviewability for a bounded payload, never correctness.
651
+ if (encodedSize(status) > softLimitBytes) {
652
+ for (const ev of status.fileReviewEventStream?.events ?? []) {
653
+ if (encodedSize(status) <= softLimitBytes) break;
654
+ if (ev.payload.case !== "candidateCaptured") continue;
655
+ const candidate = ev.payload.value;
656
+ let markedAny = false;
657
+ for (const change of candidate.changes) {
658
+ if (encodedSize(status) <= softLimitBytes) break;
659
+ if (dropInlineBodiesIfLarge(change)) {
660
+ change.diffComplete = false;
661
+ // Record the honest cause so the review UI distinguishes a size-elided
662
+ // diff from a secret-withheld one (doc 15). Don't overwrite a reason a
663
+ // more specific producer already set (e.g. SECRET_WITHHELD).
664
+ if (change.blockedReason === FileReviewBlockReason.UNSPECIFIED) {
665
+ change.blockedReason = FileReviewBlockReason.SIZE_ELIDED;
666
+ }
667
+ markedAny = true;
668
+ elidedAny = true;
669
+ }
670
+ }
671
+ if (markedAny) {
672
+ // Re-derive the rollup from the now-elided changes via the single shared
673
+ // rule. A dropped inline body is non-binary incomplete, so the set
674
+ // downgrades to PARTIAL_BLOCKED (a BINARY_SUMMARY_ONLY set that loses a
675
+ // text body is no longer binary-only); computing it here keeps the rule
676
+ // in one place instead of hardcoding the outcome.
677
+ candidate.diffCompleteness = deriveDiffCompleteness(candidate.changes);
678
+ }
679
+ }
680
+ }
681
+
682
+ // Last resort: oversized message content (e.g. a huge AI response),
683
+ // parent and sub-agent alike.
328
684
  if (encodedSize(status) > softLimitBytes) {
329
- const byContent = [...status.messages].sort(
685
+ const byContent = allMessageLists(status).flat().sort(
330
686
  (a, b) => byteLen(b.content) - byteLen(a.content),
331
687
  );
332
688
  for (const msg of byContent) {