patchwork-os 0.2.0-beta.9.canary.92 → 1.1.0-beta.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (739) hide show
  1. package/README.bridge.md +15 -16
  2. package/README.md +46 -458
  3. package/dist/activationMetrics.js +2 -3
  4. package/dist/activationMetrics.js.map +1 -1
  5. package/dist/activityLog.d.ts +1 -0
  6. package/dist/activityLog.js +8 -3
  7. package/dist/activityLog.js.map +1 -1
  8. package/dist/adapters/gemini.js +2 -2
  9. package/dist/adapters/gemini.js.map +1 -1
  10. package/dist/adapters/grok.js +6 -1
  11. package/dist/adapters/grok.js.map +1 -1
  12. package/dist/adapters/openai.js +4 -0
  13. package/dist/adapters/openai.js.map +1 -1
  14. package/dist/approvalHttp.d.ts +9 -0
  15. package/dist/approvalHttp.js +210 -171
  16. package/dist/approvalHttp.js.map +1 -1
  17. package/dist/approvalKpi.d.ts +102 -0
  18. package/dist/approvalKpi.js +282 -0
  19. package/dist/approvalKpi.js.map +1 -0
  20. package/dist/approvalQueue.d.ts +1 -1
  21. package/dist/approvalQueue.js.map +1 -1
  22. package/dist/automation.d.ts +45 -4
  23. package/dist/automation.js +123 -15
  24. package/dist/automation.js.map +1 -1
  25. package/dist/bridge.d.ts +10 -0
  26. package/dist/bridge.js +225 -46
  27. package/dist/bridge.js.map +1 -1
  28. package/dist/bridgeLockDiscovery.d.ts +16 -1
  29. package/dist/bridgeLockDiscovery.js +38 -4
  30. package/dist/bridgeLockDiscovery.js.map +1 -1
  31. package/dist/bridgeToolsRules.js +3 -18
  32. package/dist/bridgeToolsRules.js.map +1 -1
  33. package/dist/claudeAuthHttp.d.ts +25 -0
  34. package/dist/claudeAuthHttp.js +219 -0
  35. package/dist/claudeAuthHttp.js.map +1 -0
  36. package/dist/claudeDriver.js +4 -1
  37. package/dist/claudeDriver.js.map +1 -1
  38. package/dist/claudeOrchestrator.d.ts +15 -0
  39. package/dist/claudeOrchestrator.js +44 -0
  40. package/dist/claudeOrchestrator.js.map +1 -1
  41. package/dist/commands/connect.d.ts +47 -0
  42. package/dist/commands/connect.js +419 -0
  43. package/dist/commands/connect.js.map +1 -0
  44. package/dist/commands/install.js +3 -10
  45. package/dist/commands/install.js.map +1 -1
  46. package/dist/commands/launchd.d.ts +7 -0
  47. package/dist/commands/launchd.js +20 -2
  48. package/dist/commands/launchd.js.map +1 -1
  49. package/dist/commands/patchworkInit.d.ts +7 -0
  50. package/dist/commands/patchworkInit.js +26 -0
  51. package/dist/commands/patchworkInit.js.map +1 -1
  52. package/dist/commands/recipe.d.ts +59 -4
  53. package/dist/commands/recipe.js +206 -5
  54. package/dist/commands/recipe.js.map +1 -1
  55. package/dist/commands/recipeInstall.d.ts +9 -0
  56. package/dist/commands/recipeInstall.js +56 -3
  57. package/dist/commands/recipeInstall.js.map +1 -1
  58. package/dist/commands/task.d.ts +25 -0
  59. package/dist/commands/task.js +61 -41
  60. package/dist/commands/task.js.map +1 -1
  61. package/dist/commands/tokenEfficiency.d.ts +4 -0
  62. package/dist/commands/tokenEfficiency.js +23 -26
  63. package/dist/commands/tokenEfficiency.js.map +1 -1
  64. package/dist/commands/tracesExport.d.ts +15 -1
  65. package/dist/commands/tracesExport.js +39 -5
  66. package/dist/commands/tracesExport.js.map +1 -1
  67. package/dist/config.js +16 -0
  68. package/dist/config.js.map +1 -1
  69. package/dist/connectorRoutes.js +613 -62
  70. package/dist/connectorRoutes.js.map +1 -1
  71. package/dist/connectors/asana.js +9 -3
  72. package/dist/connectors/asana.js.map +1 -1
  73. package/dist/connectors/baseConnector.d.ts +16 -0
  74. package/dist/connectors/baseConnector.js +2 -0
  75. package/dist/connectors/baseConnector.js.map +1 -1
  76. package/dist/connectors/caldiy.js +34 -0
  77. package/dist/connectors/caldiy.js.map +1 -1
  78. package/dist/connectors/confluence.js +9 -2
  79. package/dist/connectors/confluence.js.map +1 -1
  80. package/dist/connectors/connectorActivity.d.ts +24 -0
  81. package/dist/connectors/connectorActivity.js +32 -0
  82. package/dist/connectors/connectorActivity.js.map +1 -0
  83. package/dist/connectors/connectorRedirectUri.js +22 -2
  84. package/dist/connectors/connectorRedirectUri.js.map +1 -1
  85. package/dist/connectors/connectorRegistry.d.ts +0 -3
  86. package/dist/connectors/connectorRegistry.js +8 -8
  87. package/dist/connectors/connectorRegistry.js.map +1 -1
  88. package/dist/connectors/datadog.js +8 -1
  89. package/dist/connectors/datadog.js.map +1 -1
  90. package/dist/connectors/discord.js +4 -3
  91. package/dist/connectors/discord.js.map +1 -1
  92. package/dist/connectors/elasticsearch.js +22 -0
  93. package/dist/connectors/elasticsearch.js.map +1 -1
  94. package/dist/connectors/github.d.ts +38 -0
  95. package/dist/connectors/github.js +91 -5
  96. package/dist/connectors/github.js.map +1 -1
  97. package/dist/connectors/gitlab.js +19 -16
  98. package/dist/connectors/gitlab.js.map +1 -1
  99. package/dist/connectors/gmail.d.ts +5 -0
  100. package/dist/connectors/gmail.js +53 -11
  101. package/dist/connectors/gmail.js.map +1 -1
  102. package/dist/connectors/googleCalendar.js +5 -4
  103. package/dist/connectors/googleCalendar.js.map +1 -1
  104. package/dist/connectors/googleDocs.js +7 -6
  105. package/dist/connectors/googleDocs.js.map +1 -1
  106. package/dist/connectors/googleDrive.js +15 -8
  107. package/dist/connectors/googleDrive.js.map +1 -1
  108. package/dist/connectors/grafana.js +39 -3
  109. package/dist/connectors/grafana.js.map +1 -1
  110. package/dist/connectors/jira.d.ts +1 -1
  111. package/dist/connectors/jira.js +19 -4
  112. package/dist/connectors/jira.js.map +1 -1
  113. package/dist/connectors/linear.js +2 -2
  114. package/dist/connectors/linear.js.map +1 -1
  115. package/dist/connectors/mcpClient.d.ts +29 -1
  116. package/dist/connectors/mcpClient.js +114 -44
  117. package/dist/connectors/mcpClient.js.map +1 -1
  118. package/dist/connectors/mcpOAuth.d.ts +1 -0
  119. package/dist/connectors/mcpOAuth.js +49 -4
  120. package/dist/connectors/mcpOAuth.js.map +1 -1
  121. package/dist/connectors/monday.js +5 -4
  122. package/dist/connectors/monday.js.map +1 -1
  123. package/dist/connectors/mongodb.js +32 -5
  124. package/dist/connectors/mongodb.js.map +1 -1
  125. package/dist/connectors/notion.d.ts +1 -1
  126. package/dist/connectors/notion.js +22 -6
  127. package/dist/connectors/notion.js.map +1 -1
  128. package/dist/connectors/oauthError.d.ts +17 -0
  129. package/dist/connectors/oauthError.js +50 -0
  130. package/dist/connectors/oauthError.js.map +1 -0
  131. package/dist/connectors/postgres.d.ts +7 -0
  132. package/dist/connectors/postgres.js +32 -2
  133. package/dist/connectors/postgres.js.map +1 -1
  134. package/dist/connectors/posthog.js +26 -0
  135. package/dist/connectors/posthog.js.map +1 -1
  136. package/dist/connectors/redis.d.ts +1 -0
  137. package/dist/connectors/redis.js +47 -20
  138. package/dist/connectors/redis.js.map +1 -1
  139. package/dist/connectors/salesforce.js +66 -5
  140. package/dist/connectors/salesforce.js.map +1 -1
  141. package/dist/connectors/sentry.js +2 -2
  142. package/dist/connectors/sentry.js.map +1 -1
  143. package/dist/connectors/slack.js +37 -14
  144. package/dist/connectors/slack.js.map +1 -1
  145. package/dist/connectors/snowflake.js +24 -0
  146. package/dist/connectors/snowflake.js.map +1 -1
  147. package/dist/connectors/stripe.js +9 -2
  148. package/dist/connectors/stripe.js.map +1 -1
  149. package/dist/connectors/supabase.js +34 -0
  150. package/dist/connectors/supabase.js.map +1 -1
  151. package/dist/connectors/tokenStorage.d.ts +20 -0
  152. package/dist/connectors/tokenStorage.js +196 -64
  153. package/dist/connectors/tokenStorage.js.map +1 -1
  154. package/dist/connectors/woocommerce.js +34 -0
  155. package/dist/connectors/woocommerce.js.map +1 -1
  156. package/dist/copilot/parseIntent.d.ts +58 -0
  157. package/dist/copilot/parseIntent.js +149 -0
  158. package/dist/copilot/parseIntent.js.map +1 -0
  159. package/dist/decisionReplay.d.ts +8 -6
  160. package/dist/decisionReplay.js +35 -11
  161. package/dist/decisionReplay.js.map +1 -1
  162. package/dist/decisionTraceLog.d.ts +9 -0
  163. package/dist/decisionTraceLog.js +6 -0
  164. package/dist/decisionTraceLog.js.map +1 -1
  165. package/dist/drivers/claude/api.js +49 -7
  166. package/dist/drivers/claude/api.js.map +1 -1
  167. package/dist/drivers/claude/envSanitizer.d.ts +9 -1
  168. package/dist/drivers/claude/envSanitizer.js +41 -3
  169. package/dist/drivers/claude/envSanitizer.js.map +1 -1
  170. package/dist/drivers/claude/streamParser.d.ts +13 -0
  171. package/dist/drivers/claude/streamParser.js.map +1 -1
  172. package/dist/drivers/claude/subprocess.js +131 -16
  173. package/dist/drivers/claude/subprocess.js.map +1 -1
  174. package/dist/drivers/claude/subprocessSettings.d.ts +1 -1
  175. package/dist/drivers/claude/subprocessSettings.js +18 -4
  176. package/dist/drivers/claude/subprocessSettings.js.map +1 -1
  177. package/dist/drivers/gemini/index.d.ts +13 -0
  178. package/dist/drivers/gemini/index.js +211 -67
  179. package/dist/drivers/gemini/index.js.map +1 -1
  180. package/dist/drivers/local/index.d.ts +2 -17
  181. package/dist/drivers/local/index.js +5 -92
  182. package/dist/drivers/local/index.js.map +1 -1
  183. package/dist/drivers/openai/index.js +69 -9
  184. package/dist/drivers/openai/index.js.map +1 -1
  185. package/dist/drivers/outputCap.d.ts +27 -0
  186. package/dist/drivers/outputCap.js +50 -0
  187. package/dist/drivers/outputCap.js.map +1 -0
  188. package/dist/embeddings/cosine.d.ts +23 -0
  189. package/dist/embeddings/cosine.js +59 -0
  190. package/dist/embeddings/cosine.js.map +1 -0
  191. package/dist/embeddings/index.d.ts +19 -0
  192. package/dist/embeddings/index.js +35 -0
  193. package/dist/embeddings/index.js.map +1 -0
  194. package/dist/embeddings/localEmbeddings.d.ts +29 -0
  195. package/dist/embeddings/localEmbeddings.js +103 -0
  196. package/dist/embeddings/localEmbeddings.js.map +1 -0
  197. package/dist/embeddings/localEndpoints.d.ts +24 -0
  198. package/dist/embeddings/localEndpoints.js +37 -0
  199. package/dist/embeddings/localEndpoints.js.map +1 -0
  200. package/dist/embeddings/types.d.ts +32 -0
  201. package/dist/embeddings/types.js +12 -0
  202. package/dist/embeddings/types.js.map +1 -0
  203. package/dist/featureFlags.d.ts +18 -0
  204. package/dist/featureFlags.js +32 -0
  205. package/dist/featureFlags.js.map +1 -1
  206. package/dist/fp/automationInterpreter.js +118 -39
  207. package/dist/fp/automationInterpreter.js.map +1 -1
  208. package/dist/fp/automationProgram.d.ts +10 -0
  209. package/dist/fp/automationProgram.js.map +1 -1
  210. package/dist/fp/automationState.d.ts +6 -0
  211. package/dist/fp/automationState.js +38 -0
  212. package/dist/fp/automationState.js.map +1 -1
  213. package/dist/fp/commandDescription.d.ts +6 -0
  214. package/dist/fp/commandDescription.js +18 -5
  215. package/dist/fp/commandDescription.js.map +1 -1
  216. package/dist/fp/interpreterContext.d.ts +16 -5
  217. package/dist/fp/interpreterContext.js +59 -3
  218. package/dist/fp/interpreterContext.js.map +1 -1
  219. package/dist/fp/policyParser.js +18 -8
  220. package/dist/fp/policyParser.js.map +1 -1
  221. package/dist/haltPushDispatch.d.ts +4 -0
  222. package/dist/haltPushDispatch.js +7 -22
  223. package/dist/haltPushDispatch.js.map +1 -1
  224. package/dist/inboxRoutes.js +2 -2
  225. package/dist/inboxRoutes.js.map +1 -1
  226. package/dist/index.js +672 -135
  227. package/dist/index.js.map +1 -1
  228. package/dist/installGuard.js +6 -0
  229. package/dist/installGuard.js.map +1 -1
  230. package/dist/localEndpointGuard.d.ts +25 -0
  231. package/dist/localEndpointGuard.js +101 -0
  232. package/dist/localEndpointGuard.js.map +1 -0
  233. package/dist/mcpRoutes.js +1 -1
  234. package/dist/mcpRoutes.js.map +1 -1
  235. package/dist/oauth.d.ts +13 -0
  236. package/dist/oauth.js +53 -9
  237. package/dist/oauth.js.map +1 -1
  238. package/dist/orchestrator/childBridgeClient.d.ts +7 -0
  239. package/dist/orchestrator/childBridgeClient.js +42 -1
  240. package/dist/orchestrator/childBridgeClient.js.map +1 -1
  241. package/dist/orchestrator/orchestratorBridge.d.ts +33 -0
  242. package/dist/orchestrator/orchestratorBridge.js +75 -9
  243. package/dist/orchestrator/orchestratorBridge.js.map +1 -1
  244. package/dist/patchworkConfig.d.ts +4 -0
  245. package/dist/patchworkConfig.js +33 -4
  246. package/dist/patchworkConfig.js.map +1 -1
  247. package/dist/probe.js +39 -5
  248. package/dist/probe.js.map +1 -1
  249. package/dist/recipeOrchestration.d.ts +59 -0
  250. package/dist/recipeOrchestration.js +314 -60
  251. package/dist/recipeOrchestration.js.map +1 -1
  252. package/dist/recipeRoutes.d.ts +93 -0
  253. package/dist/recipeRoutes.js +664 -27
  254. package/dist/recipeRoutes.js.map +1 -1
  255. package/dist/recipes/RecipeOrchestrator.d.ts +14 -0
  256. package/dist/recipes/RecipeOrchestrator.js +42 -2
  257. package/dist/recipes/RecipeOrchestrator.js.map +1 -1
  258. package/dist/recipes/agentExecutor.d.ts +51 -2
  259. package/dist/recipes/agentExecutor.js +86 -13
  260. package/dist/recipes/agentExecutor.js.map +1 -1
  261. package/dist/recipes/chainedRunner.d.ts +111 -3
  262. package/dist/recipes/chainedRunner.js +440 -45
  263. package/dist/recipes/chainedRunner.js.map +1 -1
  264. package/dist/recipes/compiler.js +42 -29
  265. package/dist/recipes/compiler.js.map +1 -1
  266. package/dist/recipes/connectorPreflight.js +31 -0
  267. package/dist/recipes/connectorPreflight.js.map +1 -1
  268. package/dist/recipes/dependencyGraph.d.ts +16 -0
  269. package/dist/recipes/dependencyGraph.js +37 -2
  270. package/dist/recipes/dependencyGraph.js.map +1 -1
  271. package/dist/recipes/eventTriggerPrograms.d.ts +36 -0
  272. package/dist/recipes/eventTriggerPrograms.js +158 -0
  273. package/dist/recipes/eventTriggerPrograms.js.map +1 -0
  274. package/dist/recipes/githubInstallSource.d.ts +9 -2
  275. package/dist/recipes/githubInstallSource.js +36 -6
  276. package/dist/recipes/githubInstallSource.js.map +1 -1
  277. package/dist/recipes/haltCategory.d.ts +10 -0
  278. package/dist/recipes/haltCategory.js +11 -1
  279. package/dist/recipes/haltCategory.js.map +1 -1
  280. package/dist/recipes/idempotencyKey.d.ts +30 -3
  281. package/dist/recipes/idempotencyKey.js +89 -5
  282. package/dist/recipes/idempotencyKey.js.map +1 -1
  283. package/dist/recipes/installer.js +12 -35
  284. package/dist/recipes/installer.js.map +1 -1
  285. package/dist/recipes/judgeVerdict.d.ts +11 -1
  286. package/dist/recipes/judgeVerdict.js +47 -7
  287. package/dist/recipes/judgeVerdict.js.map +1 -1
  288. package/dist/recipes/names.d.ts +5 -0
  289. package/dist/recipes/names.js +10 -5
  290. package/dist/recipes/names.js.map +1 -1
  291. package/dist/recipes/parser.js +74 -10
  292. package/dist/recipes/parser.js.map +1 -1
  293. package/dist/recipes/pricing/costRouter.d.ts +43 -0
  294. package/dist/recipes/pricing/costRouter.js +44 -0
  295. package/dist/recipes/pricing/costRouter.js.map +1 -0
  296. package/dist/recipes/pricing/priceTable.d.ts +76 -0
  297. package/dist/recipes/pricing/priceTable.js +144 -0
  298. package/dist/recipes/pricing/priceTable.js.map +1 -0
  299. package/dist/recipes/replayRun.js +5 -2
  300. package/dist/recipes/replayRun.js.map +1 -1
  301. package/dist/recipes/resolveRecipePath.js +15 -1
  302. package/dist/recipes/resolveRecipePath.js.map +1 -1
  303. package/dist/recipes/runBudget.d.ts +93 -32
  304. package/dist/recipes/runBudget.js +213 -49
  305. package/dist/recipes/runBudget.js.map +1 -1
  306. package/dist/recipes/runRegistry.d.ts +32 -0
  307. package/dist/recipes/runRegistry.js +54 -0
  308. package/dist/recipes/runRegistry.js.map +1 -0
  309. package/dist/recipes/scheduler.js +6 -2
  310. package/dist/recipes/scheduler.js.map +1 -1
  311. package/dist/recipes/schema.d.ts +59 -6
  312. package/dist/recipes/schemaGenerator.d.ts +9 -0
  313. package/dist/recipes/schemaGenerator.js +446 -77
  314. package/dist/recipes/schemaGenerator.js.map +1 -1
  315. package/dist/recipes/simulation/aggregateRunRisk.d.ts +46 -0
  316. package/dist/recipes/simulation/aggregateRunRisk.js +120 -0
  317. package/dist/recipes/simulation/aggregateRunRisk.js.map +1 -0
  318. package/dist/recipes/simulation/costProjector.d.ts +32 -0
  319. package/dist/recipes/simulation/costProjector.js +194 -0
  320. package/dist/recipes/simulation/costProjector.js.map +1 -0
  321. package/dist/recipes/simulation/sideEffects.d.ts +32 -0
  322. package/dist/recipes/simulation/sideEffects.js +62 -0
  323. package/dist/recipes/simulation/sideEffects.js.map +1 -0
  324. package/dist/recipes/simulation/simulate.d.ts +36 -0
  325. package/dist/recipes/simulation/simulate.js +264 -0
  326. package/dist/recipes/simulation/simulate.js.map +1 -0
  327. package/dist/recipes/simulation/simulateMockedRun.d.ts +52 -0
  328. package/dist/recipes/simulation/simulateMockedRun.js +84 -0
  329. package/dist/recipes/simulation/simulateMockedRun.js.map +1 -0
  330. package/dist/recipes/simulation/synthesizeMockedOutputs.d.ts +31 -0
  331. package/dist/recipes/simulation/synthesizeMockedOutputs.js +50 -0
  332. package/dist/recipes/simulation/synthesizeMockedOutputs.js.map +1 -0
  333. package/dist/recipes/simulation/types.d.ts +198 -0
  334. package/dist/recipes/simulation/types.js +30 -0
  335. package/dist/recipes/simulation/types.js.map +1 -0
  336. package/dist/recipes/stepObservation.d.ts +12 -0
  337. package/dist/recipes/stepObservation.js +21 -1
  338. package/dist/recipes/stepObservation.js.map +1 -1
  339. package/dist/recipes/toolRegistry.js +61 -24
  340. package/dist/recipes/toolRegistry.js.map +1 -1
  341. package/dist/recipes/tools/airtable.d.ts +15 -0
  342. package/dist/recipes/tools/airtable.js +240 -0
  343. package/dist/recipes/tools/airtable.js.map +1 -0
  344. package/dist/recipes/tools/caldiy.d.ts +13 -0
  345. package/dist/recipes/tools/caldiy.js +215 -0
  346. package/dist/recipes/tools/caldiy.js.map +1 -0
  347. package/dist/recipes/tools/circleci.d.ts +10 -0
  348. package/dist/recipes/tools/circleci.js +204 -0
  349. package/dist/recipes/tools/circleci.js.map +1 -0
  350. package/dist/recipes/tools/cloudflare.d.ts +13 -0
  351. package/dist/recipes/tools/cloudflare.js +211 -0
  352. package/dist/recipes/tools/cloudflare.js.map +1 -0
  353. package/dist/recipes/tools/confluence.js +11 -10
  354. package/dist/recipes/tools/confluence.js.map +1 -1
  355. package/dist/recipes/tools/datadog.js +13 -12
  356. package/dist/recipes/tools/datadog.js.map +1 -1
  357. package/dist/recipes/tools/docs.d.ts +18 -0
  358. package/dist/recipes/tools/docs.js +95 -0
  359. package/dist/recipes/tools/docs.js.map +1 -0
  360. package/dist/recipes/tools/elasticsearch.d.ts +11 -0
  361. package/dist/recipes/tools/elasticsearch.js +157 -0
  362. package/dist/recipes/tools/elasticsearch.js.map +1 -0
  363. package/dist/recipes/tools/fanOut.js +4 -4
  364. package/dist/recipes/tools/fanOut.js.map +1 -1
  365. package/dist/recipes/tools/figma.d.ts +12 -0
  366. package/dist/recipes/tools/figma.js +195 -0
  367. package/dist/recipes/tools/figma.js.map +1 -0
  368. package/dist/recipes/tools/file.js +43 -22
  369. package/dist/recipes/tools/file.js.map +1 -1
  370. package/dist/recipes/tools/github.js +162 -5
  371. package/dist/recipes/tools/github.js.map +1 -1
  372. package/dist/recipes/tools/gmail.js +45 -33
  373. package/dist/recipes/tools/gmail.js.map +1 -1
  374. package/dist/recipes/tools/grafana.d.ts +11 -0
  375. package/dist/recipes/tools/grafana.js +216 -0
  376. package/dist/recipes/tools/grafana.js.map +1 -0
  377. package/dist/recipes/tools/http.d.ts +1 -1
  378. package/dist/recipes/tools/http.js +10 -28
  379. package/dist/recipes/tools/http.js.map +1 -1
  380. package/dist/recipes/tools/hubspot.js +13 -12
  381. package/dist/recipes/tools/hubspot.js.map +1 -1
  382. package/dist/recipes/tools/index.d.ts +28 -0
  383. package/dist/recipes/tools/index.js +28 -0
  384. package/dist/recipes/tools/index.js.map +1 -1
  385. package/dist/recipes/tools/intercom.js +11 -10
  386. package/dist/recipes/tools/intercom.js.map +1 -1
  387. package/dist/recipes/tools/jira.js +13 -0
  388. package/dist/recipes/tools/jira.js.map +1 -1
  389. package/dist/recipes/tools/meetingNotes.js +2 -2
  390. package/dist/recipes/tools/monday.d.ts +22 -0
  391. package/dist/recipes/tools/monday.js +242 -0
  392. package/dist/recipes/tools/monday.js.map +1 -0
  393. package/dist/recipes/tools/mongodb.d.ts +19 -0
  394. package/dist/recipes/tools/mongodb.js +194 -0
  395. package/dist/recipes/tools/mongodb.js.map +1 -0
  396. package/dist/recipes/tools/notion.js +11 -10
  397. package/dist/recipes/tools/notion.js.map +1 -1
  398. package/dist/recipes/tools/obsidian.d.ts +15 -0
  399. package/dist/recipes/tools/obsidian.js +173 -0
  400. package/dist/recipes/tools/obsidian.js.map +1 -0
  401. package/dist/recipes/tools/outcomes.d.ts +15 -0
  402. package/dist/recipes/tools/outcomes.js +111 -0
  403. package/dist/recipes/tools/outcomes.js.map +1 -0
  404. package/dist/recipes/tools/paystack.d.ts +11 -0
  405. package/dist/recipes/tools/paystack.js +212 -0
  406. package/dist/recipes/tools/paystack.js.map +1 -0
  407. package/dist/recipes/tools/pipedrive.d.ts +16 -0
  408. package/dist/recipes/tools/pipedrive.js +234 -0
  409. package/dist/recipes/tools/pipedrive.js.map +1 -0
  410. package/dist/recipes/tools/postgres.d.ts +15 -0
  411. package/dist/recipes/tools/postgres.js +186 -0
  412. package/dist/recipes/tools/postgres.js.map +1 -0
  413. package/dist/recipes/tools/posthog.d.ts +16 -0
  414. package/dist/recipes/tools/posthog.js +219 -0
  415. package/dist/recipes/tools/posthog.js.map +1 -0
  416. package/dist/recipes/tools/redis.d.ts +9 -0
  417. package/dist/recipes/tools/redis.js +141 -0
  418. package/dist/recipes/tools/redis.js.map +1 -0
  419. package/dist/recipes/tools/resend.d.ts +8 -0
  420. package/dist/recipes/tools/resend.js +153 -0
  421. package/dist/recipes/tools/resend.js.map +1 -0
  422. package/dist/recipes/tools/salesforce.d.ts +16 -0
  423. package/dist/recipes/tools/salesforce.js +184 -0
  424. package/dist/recipes/tools/salesforce.js.map +1 -0
  425. package/dist/recipes/tools/sendgrid.d.ts +9 -0
  426. package/dist/recipes/tools/sendgrid.js +175 -0
  427. package/dist/recipes/tools/sendgrid.js.map +1 -0
  428. package/dist/recipes/tools/shopify.d.ts +16 -0
  429. package/dist/recipes/tools/shopify.js +266 -0
  430. package/dist/recipes/tools/shopify.js.map +1 -0
  431. package/dist/recipes/tools/snowflake.d.ts +16 -0
  432. package/dist/recipes/tools/snowflake.js +173 -0
  433. package/dist/recipes/tools/snowflake.js.map +1 -0
  434. package/dist/recipes/tools/stripe.js +13 -12
  435. package/dist/recipes/tools/stripe.js.map +1 -1
  436. package/dist/recipes/tools/supabase.d.ts +9 -0
  437. package/dist/recipes/tools/supabase.js +133 -0
  438. package/dist/recipes/tools/supabase.js.map +1 -0
  439. package/dist/recipes/tools/todoist.d.ts +15 -0
  440. package/dist/recipes/tools/todoist.js +228 -0
  441. package/dist/recipes/tools/todoist.js.map +1 -0
  442. package/dist/recipes/tools/twilio.d.ts +11 -0
  443. package/dist/recipes/tools/twilio.js +180 -0
  444. package/dist/recipes/tools/twilio.js.map +1 -0
  445. package/dist/recipes/tools/vercel.d.ts +9 -0
  446. package/dist/recipes/tools/vercel.js +146 -0
  447. package/dist/recipes/tools/vercel.js.map +1 -0
  448. package/dist/recipes/tools/webflow.d.ts +11 -0
  449. package/dist/recipes/tools/webflow.js +244 -0
  450. package/dist/recipes/tools/webflow.js.map +1 -0
  451. package/dist/recipes/tools/woocommerce.d.ts +18 -0
  452. package/dist/recipes/tools/woocommerce.js +260 -0
  453. package/dist/recipes/tools/woocommerce.js.map +1 -0
  454. package/dist/recipes/tools/wrapConnectorExecute.d.ts +25 -0
  455. package/dist/recipes/tools/wrapConnectorExecute.js +37 -0
  456. package/dist/recipes/tools/wrapConnectorExecute.js.map +1 -0
  457. package/dist/recipes/tools/zendesk.js +11 -10
  458. package/dist/recipes/tools/zendesk.js.map +1 -1
  459. package/dist/recipes/triggerVars.d.ts +15 -0
  460. package/dist/recipes/triggerVars.js +57 -0
  461. package/dist/recipes/triggerVars.js.map +1 -0
  462. package/dist/recipes/validation.d.ts +1 -1
  463. package/dist/recipes/validation.js +536 -12
  464. package/dist/recipes/validation.js.map +1 -1
  465. package/dist/recipes/yamlPositions.d.ts +1 -1
  466. package/dist/recipes/yamlPositions.js +7 -2
  467. package/dist/recipes/yamlPositions.js.map +1 -1
  468. package/dist/recipes/yamlRunner.d.ts +216 -4
  469. package/dist/recipes/yamlRunner.js +1030 -96
  470. package/dist/recipes/yamlRunner.js.map +1 -1
  471. package/dist/recipesHttp.d.ts +5 -0
  472. package/dist/recipesHttp.js +68 -6
  473. package/dist/recipesHttp.js.map +1 -1
  474. package/dist/resources.js +14 -1
  475. package/dist/resources.js.map +1 -1
  476. package/dist/riskSignals.d.ts +42 -0
  477. package/dist/riskSignals.js +197 -0
  478. package/dist/riskSignals.js.map +1 -0
  479. package/dist/riskTier.d.ts +9 -0
  480. package/dist/riskTier.js +48 -0
  481. package/dist/riskTier.js.map +1 -1
  482. package/dist/runLog.d.ts +53 -0
  483. package/dist/runLog.js +75 -16
  484. package/dist/runLog.js.map +1 -1
  485. package/dist/schemas/recipe.v1.json +360 -16
  486. package/dist/server.d.ts +60 -0
  487. package/dist/server.js +192 -31
  488. package/dist/server.js.map +1 -1
  489. package/dist/sessionCheckpoint.js +4 -1
  490. package/dist/sessionCheckpoint.js.map +1 -1
  491. package/dist/sessionDetail.d.ts +43 -0
  492. package/dist/sessionDetail.js +59 -0
  493. package/dist/sessionDetail.js.map +1 -0
  494. package/dist/settingsEnv.d.ts +47 -0
  495. package/dist/settingsEnv.js +88 -0
  496. package/dist/settingsEnv.js.map +1 -0
  497. package/dist/streamableHttp.js +44 -10
  498. package/dist/streamableHttp.js.map +1 -1
  499. package/dist/ta/backtest/abnday.d.ts +53 -0
  500. package/dist/ta/backtest/abnday.js +123 -0
  501. package/dist/ta/backtest/abnday.js.map +1 -0
  502. package/dist/ta/backtest/directional.d.ts +48 -0
  503. package/dist/ta/backtest/directional.js +103 -0
  504. package/dist/ta/backtest/directional.js.map +1 -0
  505. package/dist/ta/backtest/fundx.d.ts +67 -0
  506. package/dist/ta/backtest/fundx.js +264 -0
  507. package/dist/ta/backtest/fundx.js.map +1 -0
  508. package/dist/ta/backtest/fvg.d.ts +62 -0
  509. package/dist/ta/backtest/fvg.js +236 -0
  510. package/dist/ta/backtest/fvg.js.map +1 -0
  511. package/dist/ta/backtest/ichimoku.d.ts +41 -0
  512. package/dist/ta/backtest/ichimoku.js +146 -0
  513. package/dist/ta/backtest/ichimoku.js.map +1 -0
  514. package/dist/ta/backtest/ingest.d.ts +20 -0
  515. package/dist/ta/backtest/ingest.js +104 -0
  516. package/dist/ta/backtest/ingest.js.map +1 -0
  517. package/dist/ta/backtest/msb.d.ts +49 -0
  518. package/dist/ta/backtest/msb.js +162 -0
  519. package/dist/ta/backtest/msb.js.map +1 -0
  520. package/dist/ta/backtest/scoring.d.ts +33 -0
  521. package/dist/ta/backtest/scoring.js +82 -0
  522. package/dist/ta/backtest/scoring.js.map +1 -0
  523. package/dist/ta/backtest/summary.d.ts +20 -0
  524. package/dist/ta/backtest/summary.js +42 -0
  525. package/dist/ta/backtest/summary.js.map +1 -0
  526. package/dist/ta/backtest/tsmom.d.ts +72 -0
  527. package/dist/ta/backtest/tsmom.js +202 -0
  528. package/dist/ta/backtest/tsmom.js.map +1 -0
  529. package/dist/ta/backtest/volsq.d.ts +34 -0
  530. package/dist/ta/backtest/volsq.js +154 -0
  531. package/dist/ta/backtest/volsq.js.map +1 -0
  532. package/dist/ta/backtest/walkForward.d.ts +45 -0
  533. package/dist/ta/backtest/walkForward.js +136 -0
  534. package/dist/ta/backtest/walkForward.js.map +1 -0
  535. package/dist/ta/cycles.d.ts +15 -0
  536. package/dist/ta/cycles.js +31 -0
  537. package/dist/ta/cycles.js.map +1 -0
  538. package/dist/ta/desk/accrualEmitter.d.ts +59 -0
  539. package/dist/ta/desk/accrualEmitter.js +136 -0
  540. package/dist/ta/desk/accrualEmitter.js.map +1 -0
  541. package/dist/ta/desk/cellBacktest.d.ts +179 -0
  542. package/dist/ta/desk/cellBacktest.js +514 -0
  543. package/dist/ta/desk/cellBacktest.js.map +1 -0
  544. package/dist/ta/desk/cells/wpMaRejection.d.ts +49 -0
  545. package/dist/ta/desk/cells/wpMaRejection.js +141 -0
  546. package/dist/ta/desk/cells/wpMaRejection.js.map +1 -0
  547. package/dist/ta/desk/cells/wpVolumeClimax.d.ts +57 -0
  548. package/dist/ta/desk/cells/wpVolumeClimax.js +193 -0
  549. package/dist/ta/desk/cells/wpVolumeClimax.js.map +1 -0
  550. package/dist/ta/desk/collectors.d.ts +18 -0
  551. package/dist/ta/desk/collectors.js +450 -0
  552. package/dist/ta/desk/collectors.js.map +1 -0
  553. package/dist/ta/desk/contract.d.ts +51 -0
  554. package/dist/ta/desk/contract.js +261 -0
  555. package/dist/ta/desk/contract.js.map +1 -0
  556. package/dist/ta/desk/deskLedger.d.ts +36 -0
  557. package/dist/ta/desk/deskLedger.js +239 -0
  558. package/dist/ta/desk/deskLedger.js.map +1 -0
  559. package/dist/ta/desk/liqTape.d.ts +37 -0
  560. package/dist/ta/desk/liqTape.js +99 -0
  561. package/dist/ta/desk/liqTape.js.map +1 -0
  562. package/dist/ta/desk/post.d.ts +20 -0
  563. package/dist/ta/desk/post.js +65 -0
  564. package/dist/ta/desk/post.js.map +1 -0
  565. package/dist/ta/desk/render.d.ts +25 -0
  566. package/dist/ta/desk/render.js +398 -0
  567. package/dist/ta/desk/render.js.map +1 -0
  568. package/dist/ta/desk/surfaces.d.ts +216 -0
  569. package/dist/ta/desk/surfaces.js +472 -0
  570. package/dist/ta/desk/surfaces.js.map +1 -0
  571. package/dist/ta/desk/types.d.ts +174 -0
  572. package/dist/ta/desk/types.js +104 -0
  573. package/dist/ta/desk/types.js.map +1 -0
  574. package/dist/ta/index.d.ts +23 -0
  575. package/dist/ta/index.js +54 -0
  576. package/dist/ta/index.js.map +1 -0
  577. package/dist/ta/ledger.d.ts +44 -0
  578. package/dist/ta/ledger.js +95 -0
  579. package/dist/ta/ledger.js.map +1 -0
  580. package/dist/ta/levels.d.ts +21 -0
  581. package/dist/ta/levels.js +41 -0
  582. package/dist/ta/levels.js.map +1 -0
  583. package/dist/ta/swings.d.ts +30 -0
  584. package/dist/ta/swings.js +83 -0
  585. package/dist/ta/swings.js.map +1 -0
  586. package/dist/ta/types.d.ts +81 -0
  587. package/dist/ta/types.js +20 -0
  588. package/dist/ta/types.js.map +1 -0
  589. package/dist/telemetry.js +20 -9
  590. package/dist/telemetry.js.map +1 -1
  591. package/dist/tokenUsageTracker.d.ts +2 -1
  592. package/dist/tokenUsageTracker.js +38 -20
  593. package/dist/tokenUsageTracker.js.map +1 -1
  594. package/dist/tools/contextBundle.js +13 -9
  595. package/dist/tools/contextBundle.js.map +1 -1
  596. package/dist/tools/ctxGetTaskContext.js +8 -3
  597. package/dist/tools/ctxGetTaskContext.js.map +1 -1
  598. package/dist/tools/ctxQueryTraces.d.ts +18 -1
  599. package/dist/tools/ctxQueryTraces.js +176 -27
  600. package/dist/tools/ctxQueryTraces.js.map +1 -1
  601. package/dist/tools/enrichCommit.js +5 -2
  602. package/dist/tools/enrichCommit.js.map +1 -1
  603. package/dist/tools/getDependencyTree.js +2 -2
  604. package/dist/tools/getDependencyTree.js.map +1 -1
  605. package/dist/tools/getDiagnostics.js +3 -0
  606. package/dist/tools/getDiagnostics.js.map +1 -1
  607. package/dist/tools/getDocumentSymbols.js +5 -1
  608. package/dist/tools/getDocumentSymbols.js.map +1 -1
  609. package/dist/tools/getGitHotspots.js +12 -1
  610. package/dist/tools/getGitHotspots.js.map +1 -1
  611. package/dist/tools/getGitLog.js +5 -2
  612. package/dist/tools/getGitLog.js.map +1 -1
  613. package/dist/tools/getGitStatus.js +6 -1
  614. package/dist/tools/getGitStatus.js.map +1 -1
  615. package/dist/tools/getProjectContext.js +3 -1
  616. package/dist/tools/getProjectContext.js.map +1 -1
  617. package/dist/tools/getToolCapabilities.js +39 -4
  618. package/dist/tools/getToolCapabilities.js.map +1 -1
  619. package/dist/tools/gitWrite.js +12 -3
  620. package/dist/tools/gitWrite.js.map +1 -1
  621. package/dist/tools/github/composite.d.ts +1 -1
  622. package/dist/tools/github/composite.js +7 -7
  623. package/dist/tools/github/composite.js.map +1 -1
  624. package/dist/tools/github/pr.d.ts +6 -6
  625. package/dist/tools/github/pr.js +18 -11
  626. package/dist/tools/github/pr.js.map +1 -1
  627. package/dist/tools/httpClient.js +55 -3
  628. package/dist/tools/httpClient.js.map +1 -1
  629. package/dist/tools/index.d.ts +2 -1
  630. package/dist/tools/index.js +19 -12
  631. package/dist/tools/index.js.map +1 -1
  632. package/dist/tools/lsp.d.ts +6 -0
  633. package/dist/tools/lsp.js +81 -12
  634. package/dist/tools/lsp.js.map +1 -1
  635. package/dist/tools/previewEdit.js +16 -1
  636. package/dist/tools/previewEdit.js.map +1 -1
  637. package/dist/tools/resumeClaudeTask.js +17 -0
  638. package/dist/tools/resumeClaudeTask.js.map +1 -1
  639. package/dist/tools/searchAndReplace.js +9 -5
  640. package/dist/tools/searchAndReplace.js.map +1 -1
  641. package/dist/tools/searchTools.d.ts +26 -0
  642. package/dist/tools/searchTools.js +42 -2
  643. package/dist/tools/searchTools.js.map +1 -1
  644. package/dist/tools/terminal.js +23 -51
  645. package/dist/tools/terminal.js.map +1 -1
  646. package/dist/tools/transaction.d.ts +9 -1
  647. package/dist/tools/transaction.js +59 -4
  648. package/dist/tools/transaction.js.map +1 -1
  649. package/dist/tools/utils.d.ts +16 -0
  650. package/dist/tools/utils.js +77 -7
  651. package/dist/tools/utils.js.map +1 -1
  652. package/dist/transport.d.ts +31 -0
  653. package/dist/transport.js +117 -14
  654. package/dist/transport.js.map +1 -1
  655. package/dist/wireHaltPushDispatch.js +12 -0
  656. package/dist/wireHaltPushDispatch.js.map +1 -1
  657. package/dist/workerGateDecisionLog.d.ts +141 -0
  658. package/dist/workerGateDecisionLog.js +361 -0
  659. package/dist/workerGateDecisionLog.js.map +1 -0
  660. package/dist/workers/actionClass.d.ts +38 -0
  661. package/dist/workers/actionClass.js +155 -0
  662. package/dist/workers/actionClass.js.map +1 -0
  663. package/dist/workers/backtest.d.ts +68 -0
  664. package/dist/workers/backtest.js +113 -0
  665. package/dist/workers/backtest.js.map +1 -0
  666. package/dist/workers/contextRisk.d.ts +35 -0
  667. package/dist/workers/contextRisk.js +20 -0
  668. package/dist/workers/contextRisk.js.map +1 -0
  669. package/dist/workers/contextRiskScorer.d.ts +48 -0
  670. package/dist/workers/contextRiskScorer.js +94 -0
  671. package/dist/workers/contextRiskScorer.js.map +1 -0
  672. package/dist/workers/graduation.d.ts +61 -0
  673. package/dist/workers/graduation.js +112 -0
  674. package/dist/workers/graduation.js.map +1 -0
  675. package/dist/workers/outcomeStore.d.ts +98 -0
  676. package/dist/workers/outcomeStore.js +146 -0
  677. package/dist/workers/outcomeStore.js.map +1 -0
  678. package/dist/workers/outcomesCli.d.ts +34 -0
  679. package/dist/workers/outcomesCli.js +91 -0
  680. package/dist/workers/outcomesCli.js.map +1 -0
  681. package/dist/workers/runWorkerShadow.d.ts +90 -0
  682. package/dist/workers/runWorkerShadow.js +323 -0
  683. package/dist/workers/runWorkerShadow.js.map +1 -0
  684. package/dist/workers/shadowGate.d.ts +30 -0
  685. package/dist/workers/shadowGate.js +58 -0
  686. package/dist/workers/shadowGate.js.map +1 -0
  687. package/dist/workers/shadowObserver.d.ts +165 -0
  688. package/dist/workers/shadowObserver.js +203 -0
  689. package/dist/workers/shadowObserver.js.map +1 -0
  690. package/dist/workers/shadowReport.d.ts +16 -0
  691. package/dist/workers/shadowReport.js +61 -0
  692. package/dist/workers/shadowReport.js.map +1 -0
  693. package/dist/workers/shadowRun.d.ts +38 -0
  694. package/dist/workers/shadowRun.js +44 -0
  695. package/dist/workers/shadowRun.js.map +1 -0
  696. package/dist/workers/trustLevel.d.ts +82 -0
  697. package/dist/workers/trustLevel.js +100 -0
  698. package/dist/workers/trustLevel.js.map +1 -0
  699. package/dist/workers/worker.d.ts +49 -0
  700. package/dist/workers/worker.js +88 -0
  701. package/dist/workers/worker.js.map +1 -0
  702. package/dist/workers/workerGate.d.ts +101 -0
  703. package/dist/workers/workerGate.js +210 -0
  704. package/dist/workers/workerGate.js.map +1 -0
  705. package/dist/workers/workerLevelStore.d.ts +33 -0
  706. package/dist/workers/workerLevelStore.js +87 -0
  707. package/dist/workers/workerLevelStore.js.map +1 -0
  708. package/dist/workers/workerLoader.d.ts +7 -0
  709. package/dist/workers/workerLoader.js +31 -0
  710. package/dist/workers/workerLoader.js.map +1 -0
  711. package/dist/writeFileAtomic.js +40 -13
  712. package/dist/writeFileAtomic.js.map +1 -1
  713. package/package.json +14 -6
  714. package/scripts/mcp-stdio-shim.cjs +247 -87
  715. package/templates/CLAUDE.bridge.md +1 -1
  716. package/templates/recipes/dependency-bump.yaml +33 -0
  717. package/templates/recipes/incident-to-pr.yaml +187 -0
  718. package/templates/recipes/outcome-ingester.yaml +53 -0
  719. package/templates/recipes/release-notes.yaml +30 -0
  720. package/templates/recipes/triage-failing-tests-autofile.yaml +152 -0
  721. package/templates/recipes/triage-failing-tests.yaml +38 -0
  722. package/templates/workers/dependency-upkeep.worker.yaml +26 -0
  723. package/templates/workers/release-notes.worker.yaml +17 -0
  724. package/templates/workers/test-guardian.worker.yaml +24 -0
  725. package/deploy/README.md +0 -172
  726. package/deploy/bootstrap-new-vps.sh +0 -364
  727. package/deploy/bootstrap-vps.sh +0 -192
  728. package/deploy/claude-ide-bridge.service.template +0 -67
  729. package/deploy/claude-ide-bridge@.service +0 -31
  730. package/deploy/deploy-dashboard.sh +0 -198
  731. package/deploy/deploy-landing.sh +0 -136
  732. package/deploy/ecosystem.config.js.example +0 -36
  733. package/deploy/install-vps-service.sh +0 -240
  734. package/deploy/macos/README.md +0 -153
  735. package/deploy/macos/com.patchwork.bridge.plist.template +0 -54
  736. package/deploy/macos/com.patchwork.tunnel.plist.template +0 -76
  737. package/deploy/macos/install-mac-bridge.sh +0 -244
  738. package/deploy/macos/uninstall-mac-bridge.sh +0 -22
  739. package/deploy/nginx-claude-bridge.conf.template +0 -129
@@ -31,8 +31,10 @@ import { fileURLToPath } from "node:url";
31
31
  import { parse as parseYaml } from "yaml";
32
32
  import { captureFixture } from "../connectors/fixtureRecorder.js";
33
33
  import { sanitizeEnv } from "../drivers/claude/envSanitizer.js";
34
+ import { isLoopbackOrPrivateEndpoint } from "../localEndpointGuard.js";
34
35
  import { loadConfig as loadPatchworkConfigSync } from "../patchworkConfig.js";
35
36
  import { findYamlRecipePath } from "../recipesHttp.js";
37
+ import { classifyTool } from "../riskTier.js";
36
38
  /**
37
39
  * Local alias for `sanitizeParsedJson` from `src/sanitizeParsedJson.ts`.
38
40
  * Kept under the old name so the existing callsites in this file don't
@@ -42,14 +44,18 @@ import { findYamlRecipePath } from "../recipesHttp.js";
42
44
  */
43
45
  import { sanitizeParsedJson as sanitizeParsed } from "../sanitizeParsedJson.js";
44
46
  import { ensureCmdShim } from "../winShim.js";
47
+ import { mergeAgentDisallowedTools } from "../workers/workerGate.js";
45
48
  import { executeAgent as _executeAgent, } from "./agentExecutor.js";
46
49
  import { categoriseHaltReason } from "./haltCategory.js";
47
50
  import { assertValidManualRunId, deriveScopeKey, WriteEffectLedger, } from "./idempotencyKey.js";
48
51
  import { buildJudgeArtefactBlock, JUDGE_PROMPT_SUFFIX, parseJudgeVerdict, } from "./judgeVerdict.js";
49
52
  import { defaultDeprecationWarn, normalizeRecipeForRuntime, } from "./migrations/index.js";
53
+ import { costRouter } from "./pricing/costRouter.js";
54
+ import { loadPriceTable, costUsd as priceCostUsd, } from "./pricing/priceTable.js";
50
55
  import { resolveRecipePath } from "./resolveRecipePath.js";
51
56
  import { RunBudget } from "./runBudget.js";
52
- import { detectSilentFail } from "./stepObservation.js";
57
+ import { registerRun, unregisterRun } from "./runRegistry.js";
58
+ import { detectSilentFail, redactSecretsForPrompt } from "./stepObservation.js";
53
59
  // Import tool registry and trigger tool self-registration
54
60
  import { applyToolOutputContext, executeTool, getTool, hasTool, registerPluginTools, } from "./toolRegistry.js";
55
61
  import { resolveWorkspaceRoot } from "./workspaceRoot.js";
@@ -134,6 +140,13 @@ export function evaluateExpect(result, expect) {
134
140
  * without schema assertions don't pay the import/compile cost.
135
141
  */
136
142
  let _stepExpectAjv;
143
+ // Process-scoped probe cache for `claude --version`. Avoids spawning the .cmd
144
+ // shim (300–700 ms on Windows) on every recipe step when no claudeFn is
145
+ // configured. Exported for tests that need to reset between cases.
146
+ let _claudeCliProbeCache;
147
+ export function resetProbeCliCache() {
148
+ _claudeCliProbeCache = undefined;
149
+ }
137
150
  async function getStepExpectAjv() {
138
151
  if (!_stepExpectAjv) {
139
152
  const { createAjv2020 } = await import("../ajv2020.js");
@@ -314,6 +327,14 @@ export async function loadRecipeServers(specs) {
314
327
  debug: (_msg) => { },
315
328
  };
316
329
  for (const spec of toLoad) {
330
+ // Mark the spec as loaded OPTIMISTICALLY before the async load so two
331
+ // concurrent recipe runs sharing a `servers:` spec don't both pass the
332
+ // `filter` dedup above and double-register the same plugin tools (the
333
+ // registry does not guard re-registration). On failure we remove it so a
334
+ // later run can retry.
335
+ if (loadedPluginSpecs.has(spec))
336
+ continue;
337
+ loadedPluginSpecs.add(spec);
317
338
  try {
318
339
  const loaded = await loadPluginsFull([spec], minimalConfig, minimalLogger);
319
340
  let toolCount = 0;
@@ -325,39 +346,172 @@ export async function loadRecipeServers(specs) {
325
346
  }));
326
347
  toolCount += registerPluginTools(pluginTools);
327
348
  }
328
- loadedPluginSpecs.add(spec);
329
349
  if (toolCount > 0) {
330
350
  console.info(`[recipe servers] loaded "${spec}" — ${toolCount} tool(s) registered`);
331
351
  }
332
352
  }
333
353
  catch (err) {
354
+ loadedPluginSpecs.delete(spec);
334
355
  console.warn(`[recipe servers] failed to load "${spec}": ${err instanceof Error ? err.message : String(err)}`);
335
356
  }
336
357
  }
337
358
  }
359
+ /**
360
+ * P1 cost/token corpus — drivers that incur real, metered, per-token API
361
+ * billing (the only ones whose spend is real money and thus priceable here).
362
+ * Mirrors `BILLABLE_DRIVERS` in runBudget.ts (kept local — that set is private
363
+ * and runBudget.ts is enforcement-critical / must not be modified for P1).
364
+ * `local` reports usage but costs no real money, so it is NOT billable.
365
+ */
366
+ const COST_BILLABLE_DRIVERS = new Set([
367
+ "anthropic",
368
+ "openai",
369
+ "grok",
370
+ "gemini",
371
+ "gemini-api",
372
+ ]);
373
+ function newStepUsageAccumulator() {
374
+ return { inputTokens: 0, outputTokens: 0, measured: false };
375
+ }
376
+ /**
377
+ * Fold one agent call's usage into a per-step accumulator. Adds tokens when
378
+ * `usage` is present; adds USD only when the served model is billable AND
379
+ * present in the price table (NEVER a `0` placeholder for the unpriced case).
380
+ */
381
+ function accumulateAgentUsage(acc, usage, servedBy, priceTable) {
382
+ if (!usage)
383
+ return;
384
+ acc.measured = true;
385
+ acc.inputTokens += usage.inputTokens;
386
+ acc.outputTokens += usage.outputTokens;
387
+ const driver = servedBy?.driver;
388
+ const model = servedBy?.model;
389
+ if (driver && model && COST_BILLABLE_DRIVERS.has(driver)) {
390
+ const cost = priceCostUsd(model, usage, priceTable);
391
+ if (typeof cost === "number") {
392
+ acc.costUsd = (acc.costUsd ?? 0) + cost;
393
+ }
394
+ }
395
+ }
396
+ /**
397
+ * Build the optional token fields for a step result from its accumulator.
398
+ * Returns an empty object (no fields) when the step reported no usage, so a
399
+ * tool step or unmeasured-driver step round-trips with the fields ABSENT.
400
+ */
401
+ function stepUsageFields(acc) {
402
+ if (!acc.measured)
403
+ return {};
404
+ return {
405
+ inputTokens: acc.inputTokens,
406
+ outputTokens: acc.outputTokens,
407
+ ...(typeof acc.costUsd === "number" ? { costUsd: acc.costUsd } : {}),
408
+ };
409
+ }
410
+ /**
411
+ * P1 — single-call usage → persisted-step usage fields. Exported for the
412
+ * chained runner, whose agent steps make exactly one agent call (no
413
+ * judge→refine loop), so a per-call computation suffices. Returns undefined
414
+ * when the driver reported no usage (fields stay ABSENT). `costUsd` set only
415
+ * for a billable driver + priced model; never a `0` placeholder.
416
+ */
417
+ export function computeAgentCallUsage(usage, servedBy, priceTable = loadPriceTable()) {
418
+ if (!usage)
419
+ return undefined;
420
+ const driver = servedBy?.driver;
421
+ const model = servedBy?.model;
422
+ let cost;
423
+ if (driver && model && COST_BILLABLE_DRIVERS.has(driver)) {
424
+ const c = priceCostUsd(model, usage, priceTable);
425
+ if (typeof c === "number")
426
+ cost = c;
427
+ }
428
+ return {
429
+ inputTokens: usage.inputTokens,
430
+ outputTokens: usage.outputTokens,
431
+ ...(typeof cost === "number" ? { costUsd: cost } : {}),
432
+ };
433
+ }
434
+ function newRunUsageAccumulator() {
435
+ return { inputTokens: 0, outputTokens: 0, measured: false };
436
+ }
437
+ function foldStepIntoRun(run, step) {
438
+ if (!step.measured)
439
+ return;
440
+ run.measured = true;
441
+ run.inputTokens += step.inputTokens;
442
+ run.outputTokens += step.outputTokens;
443
+ if (typeof step.costUsd === "number") {
444
+ run.costUsd = (run.costUsd ?? 0) + step.costUsd;
445
+ }
446
+ }
447
+ /** Build the optional `tokenTotals` for a run, or undefined when none measured. */
448
+ function runTokenTotals(run) {
449
+ if (!run.measured)
450
+ return undefined;
451
+ return {
452
+ inputTokens: run.inputTokens,
453
+ outputTokens: run.outputTokens,
454
+ ...(typeof run.costUsd === "number" ? { costUsd: run.costUsd } : {}),
455
+ };
456
+ }
457
+ /**
458
+ * Extract ONLY the env vars a recipe explicitly declares via a
459
+ * `context: [{ type: "env", keys: [...] }]` block. Both the flat runner AND the
460
+ * chained/replay paths MUST use this so undeclared process-level secrets never
461
+ * reach `{{env.X}}` template expressions.
462
+ *
463
+ * Audit 2026-06-08 (recipe-support-3): the chained dispatch and replay paths
464
+ * previously spread the entire `process.env` into the template context, silently
465
+ * diverging from the flat runner's allowlist and exposing every process secret
466
+ * (API keys, OAuth/connector tokens, TLS material) to any chained recipe author.
467
+ */
468
+ export function declaredRecipeEnv(recipe, processEnv = process.env) {
469
+ const out = {};
470
+ const blocks = recipe?.context;
471
+ if (!Array.isArray(blocks))
472
+ return out;
473
+ for (const block of blocks) {
474
+ const b = block;
475
+ if (b.type === "env" && Array.isArray(b.keys)) {
476
+ for (const key of b.keys) {
477
+ if (typeof key !== "string")
478
+ continue;
479
+ const v = processEnv[key];
480
+ if (v !== undefined)
481
+ out[key] = v;
482
+ }
483
+ }
484
+ }
485
+ return out;
486
+ }
338
487
  export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
339
488
  if (recipe.servers?.length) {
340
489
  await loadRecipeServers(recipe.servers);
341
490
  }
342
491
  const now = deps.now ? deps.now() : new Date();
343
- // Resolve recipe-level context blocks (type: env) into seed context
344
- const envCtx = {};
345
- if (Array.isArray(recipe.context)) {
346
- for (const block of recipe
347
- .context ?? []) {
348
- const b = block;
349
- if (b.type === "env" && Array.isArray(b.keys)) {
350
- for (const key of b.keys) {
351
- const v = process.env[key];
352
- if (v !== undefined)
353
- envCtx[key] = v;
354
- }
355
- }
356
- }
357
- }
492
+ // Resolve recipe-level context blocks (type: env) into seed context via the
493
+ // shared declared-keys allowlist (also used by the chained/replay paths).
494
+ const envCtx = declaredRecipeEnv(recipe);
495
+ // SECRETS-IN-VARS: track which ctx keys came from a `type: env` block so the
496
+ // agent (LLM-facing) prompt can redact them. Their raw values still flow to
497
+ // TOOL steps (an http header / DB password legitimately needs the secret),
498
+ // but they must never reach the model verbatim — the secure default is
499
+ // redaction. See PR body / docs/recipe-feature-investigation-2026-06-05.md.
500
+ const secretKeys = new Set(Object.keys(envCtx));
501
+ const iso = now.toISOString();
358
502
  const ctx = {
359
- date: now.toISOString().slice(0, 10),
503
+ date: iso.slice(0, 10),
360
504
  time: now.toTimeString().slice(0, 5),
505
+ // Built-in date/time tokens, injected (not phantom) so {{YYYY-MM-DD}} etc.
506
+ // render real values at run time AND pass template-ref lint. Keep in sync
507
+ // with builtinKeys in validation.ts. (audit 2026-06-10 recipe-validation-1)
508
+ YYYY: iso.slice(0, 4),
509
+ "YYYY-MM": iso.slice(0, 7),
510
+ "YYYY-MM-DD": iso.slice(0, 10),
511
+ ISO_NOW: iso,
512
+ HH: iso.slice(11, 13),
513
+ MM: iso.slice(14, 16),
514
+ SS: iso.slice(17, 19),
361
515
  ...envCtx,
362
516
  ...seedContext,
363
517
  };
@@ -470,10 +624,38 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
470
624
  // Non-fatal — run-log failures must never break recipe execution.
471
625
  }
472
626
  }
627
+ // Register this run so POST /runs/:seq/cancel can abort it (H11).
628
+ // Mirrors chainedRunner.ts:1277 — only the top-level run registers.
629
+ const runController = runSeq !== undefined ? registerRun(runSeq) : undefined;
630
+ // L1 (review #1028): the LIVE cancel handle is runController.signal (aborted
631
+ // by POST /runs/:seq/cancel); deps.signal is the external caller signal
632
+ // (absent on the production flat path). Combine both so a cancelled run aborts
633
+ // a pending approval wait instead of hanging the full TTL — forwarding only
634
+ // deps.signal left the flat path's L1 goal unmet. Mirrors the dual-signal
635
+ // next-step check below.
636
+ const effectiveRunSignal = runController?.signal && deps.signal
637
+ ? AbortSignal.any([runController.signal, deps.signal])
638
+ : (runController?.signal ?? deps.signal);
473
639
  const outputs = [];
474
640
  const stepResults = [];
641
+ // P1 cost/token corpus. The price table is loaded once per run (fail-open).
642
+ // `currentStepUsage` accumulates usage across all agent calls of the CURRENT
643
+ // agent step (including judge→refine re-runs via `runAgentText`); `runUsage`
644
+ // sums measured steps into the run-level total.
645
+ const priceTable = loadPriceTable();
646
+ const runUsage = newRunUsageAccumulator();
647
+ let currentStepUsage = newStepUsageAccumulator();
475
648
  let stepsRun = 0;
476
649
  let runError;
650
+ // Bug (2): the flat runner historically recorded the first non-optional
651
+ // failure in `runError` but kept executing later steps — diverging from
652
+ // chainedRunner, which aborts on a fatal failure. This flag is set ONLY
653
+ // when a failure is fatal (non-optional AND fail-open semantics do not
654
+ // apply via step.optional / on_error.fallback=log_only|deliver_original).
655
+ // The loop checks it at the top and breaks, matching chainedRunner's
656
+ // abort-on-failure contract. Fail-open failures never set it, so
657
+ // log_only/deliver_original/optional steps still let the run continue.
658
+ let haltAfterFailure = false;
477
659
  // Live-tail SSE broadcaster. Wrapped in a try/catch on every call so a
478
660
  // misbehaving listener can never break the run (mirrors chainedRunner).
479
661
  // No-ops when `activityLog` isn't wired (CLI runs, tests, mocks).
@@ -543,6 +725,245 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
543
725
  ts: Date.now(),
544
726
  });
545
727
  };
728
+ // ── OPT-IN judge → refine loop (helper closure) ──────────────────────────
729
+ //
730
+ // ⚠️ INVARIANT DEPARTURE — this drives a bounded revise→re-judge loop and
731
+ // MAY gate the run on exhaustion. It departs the augment-only invariant in
732
+ // judgeVerdict.ts, but is reachable ONLY when the judge step opts in via
733
+ // `agent.max_revisions > 0`. The augment-only PR3a path is untouched.
734
+ //
735
+ // `runAgentText` mirrors the main agent path's text processing exactly
736
+ // (strip leading narration, then JSON-fence parse + sanitize, else use the
737
+ // raw string) so a revised draft commits to ctx the same way a first-pass
738
+ // agent step would. It returns `{ value, ok }`; `ok: false` signals a
739
+ // failed / silent-fail / empty agent response — the caller stops the loop
740
+ // and treats it as exhausted (we don't re-judge a non-result).
741
+ const runAgentText = async (prompt, driver, model, mcpAccess, downshift, providerOptions,
742
+ // P0-5: carry the reviewed/judge step's opt-in tool sandbox into refine-loop
743
+ // re-runs so a sandboxed step STAYS sandboxed across revisions/re-judges.
744
+ sandboxOpts) => {
745
+ // Phase 4: route revisions too, so a downshift on the reviewed step also
746
+ // applies to its refine-loop re-runs (no-op when downshift is absent).
747
+ const routed = resolveRouting({ driver, model }, downshift, prompt, runBudget);
748
+ const agentReturn = await _executeAgent({
749
+ prompt,
750
+ driver: routed.driver === "api" ? "anthropic" : routed.driver,
751
+ model: routed.model,
752
+ ...(mcpAccess !== undefined && { mcpAccess }),
753
+ ...(sandboxOpts?.sandbox !== undefined && {
754
+ sandbox: sandboxOpts.sandbox,
755
+ }),
756
+ ...(sandboxOpts?.tools !== undefined && {
757
+ allowedTools: sandboxOpts.tools,
758
+ }),
759
+ // Worker.autonomy: a sandboxed step STAYS sandboxed across re-runs AND
760
+ // inherits the worker's agent-step deny list (same merge as the primary
761
+ // agent branch), so refine-loop re-runs can't bypass the gate either.
762
+ ...(() => {
763
+ const merged = mergeAgentDisallowedTools(sandboxOpts?.disallowedTools, deps.agentDisallowedTools);
764
+ return merged !== undefined ? { disallowedTools: merged } : {};
765
+ })(),
766
+ // Fail closed if a worker sandbox can't be enforced on the chosen driver.
767
+ ...(deps.agentDisallowedTools?.length && { enforceSandbox: true }),
768
+ ...(providerOptions && { providerOptions }),
769
+ }, buildAgentExecutorDeps(stepDeps, deps));
770
+ runBudget.reconcile(
771
+ // Prefer the driver executeAgent actually resolved+ran; fall back to
772
+ // the routed value only when servedBy is absent (non-executeAgent
773
+ // callers). Stops auto-detected runs being mis-attributed to "auto".
774
+ agentReturn.servedBy?.driver ??
775
+ (routed.driver === "api" ? "anthropic" : (routed.driver ?? "auto")), agentReturn.usage,
776
+ // Resolved model for USD pricing (Phase 3). Absent → unpriced → the USD
777
+ // cap fails open for this call.
778
+ agentReturn.servedBy?.model,
779
+ // Char counts for the opt-in unmeasured-driver ≈$ estimate (warn-only).
780
+ { inputChars: prompt.length, outputChars: agentReturn.text.length });
781
+ // P1: fold this refine-loop agent call into the current step's usage.
782
+ accumulateAgentUsage(currentStepUsage, agentReturn.usage, agentReturn.servedBy, priceTable);
783
+ const text = agentReturn.text;
784
+ // Same failure detection as the main agent branch: explicit failure
785
+ // marker or silent-fail patterns ⇒ not a usable result.
786
+ if (text.startsWith("[agent step failed:") || detectSilentFail(text)) {
787
+ return { value: text, ok: false };
788
+ }
789
+ const stripped = stripLeadingNarration(text);
790
+ if (!stripped.trim()) {
791
+ return { value: stripped, ok: false };
792
+ }
793
+ try {
794
+ const jsonMatch = /```(?:json)?\s*([\s\S]*?)```/.exec(stripped) ?? [
795
+ null,
796
+ stripped,
797
+ ];
798
+ const parsed = sanitizeParsed(JSON.parse((jsonMatch[1] ?? "").trim()));
799
+ return { value: parsed, ok: true };
800
+ }
801
+ catch {
802
+ return { value: stripped, ok: true };
803
+ }
804
+ };
805
+ const runJudgeRefineLoop = async (params) => {
806
+ const { agentCfg, reviewsKey, maxRevisions, judgeStepId, firstVerdict, judgeStepResult, failOpenAgent, } = params;
807
+ // Find the agent step whose output the judge reviews. A judge that
808
+ // reviews a tool step or a seed var (no agent to re-run) cannot be
809
+ // refined — skip the loop gracefully, leaving the augment-only verdict
810
+ // already stashed on the judge step result untouched.
811
+ const reviewedStep = recipe.steps.find((s) => s.agent && (s.agent.into ?? "agent_output") === reviewsKey);
812
+ if (!reviewedStep?.agent) {
813
+ return { haltAfterFailure: false };
814
+ }
815
+ const reviewedAgent = reviewedStep.agent;
816
+ let currentVerdict = firstVerdict;
817
+ let revisions = 0;
818
+ while (revisions < maxRevisions &&
819
+ currentVerdict.verdict === "request_changes") {
820
+ // Budget gate: never exceed the run's token budget. If admission is
821
+ // refused, stop early (treat as exhausted) — the budget halt is
822
+ // surfaced by the next top-of-loop admission check for later steps.
823
+ const admission = runBudget.admit();
824
+ if (!admission.admitted) {
825
+ break;
826
+ }
827
+ // REVISE: re-run the reviewed agent with the prior draft + fixList.
828
+ const priorDraft = ctx[reviewsKey];
829
+ const fixList = currentVerdict.fixList ?? [];
830
+ const revisionBlock = `\n\n<revision-request>\n` +
831
+ `A reviewer requested changes to your previous draft. Address every` +
832
+ ` item, then return the full revised draft only.\n\n` +
833
+ `<previous-draft>\n${typeof priorDraft === "string" ? priorDraft : JSON.stringify(priorDraft, null, 2)}\n</previous-draft>\n\n` +
834
+ `<fix-list>\n${fixList.length > 0 ? fixList.map((f) => `- ${f}`).join("\n") : "- (no explicit fix list provided)"}\n</fix-list>\n` +
835
+ `</revision-request>`;
836
+ const revisionPrompt = render(reviewedAgent.prompt, redactSecretsForPrompt(ctx, secretKeys)) +
837
+ revisionBlock;
838
+ // Quality-aware escalation: on the Nth revision, re-run the reviewed step
839
+ // with the Nth more-capable candidate (`escalate[revisions]`) instead of
840
+ // the base model — local/cheap first, escalate to cloud only when the
841
+ // judge keeps rejecting. `revisions` is 0-based here (incremented after
842
+ // the re-judge below). When escalating we drop `downshift` for this call:
843
+ // escalation means "go stronger", so it must not be re-downshifted.
844
+ const escalateTo = reviewedAgent.escalate?.[revisions];
845
+ const revised = await runAgentText(revisionPrompt, escalateTo?.driver ?? reviewedAgent.driver, escalateTo?.model ?? reviewedAgent.model, reviewedAgent.mcpAccess, escalateTo ? undefined : reviewedAgent.downshift, undefined,
846
+ // P0-5: the revision re-runs the REVIEWED step → keep its sandbox.
847
+ {
848
+ ...(reviewedAgent.sandbox !== undefined && {
849
+ sandbox: reviewedAgent.sandbox,
850
+ }),
851
+ ...(reviewedAgent.tools !== undefined && {
852
+ tools: reviewedAgent.tools,
853
+ }),
854
+ ...(reviewedAgent.disallowedTools !== undefined && {
855
+ disallowedTools: reviewedAgent.disallowedTools,
856
+ }),
857
+ });
858
+ if (!revised.ok) {
859
+ // A failed / empty revision can't be re-judged — stop and treat the
860
+ // loop as exhausted with the last good verdict still in place.
861
+ break;
862
+ }
863
+ // R4 #1 (HIGH): stage the revised draft locally — do NOT commit to ctx
864
+ // yet. Committing before the verdict resolves leaves an UNAPPROVED draft
865
+ // in ctx on any loop break (unparseable verdict, budget denial, failed
866
+ // re-judge), so downstream steps treat it as approved. The revised value
867
+ // is only promoted to ctx once a verdict accepts it (approve, or
868
+ // exhaustion with on_exhausted: "proceed").
869
+ const pendingRevised = revised.value;
870
+ // Budget gate (post-revise, pre-re-judge): the revise call may have
871
+ // exhausted the token budget. Check again before firing the re-judge so
872
+ // we don't make one extra LLM call over budget — audit 2026-06-03 LOW #2.
873
+ const postReviseAdmission = runBudget.admit();
874
+ if (!postReviseAdmission.admitted) {
875
+ // Audit 2026-06-08 (recipe-flat-1): the revision was produced but the
876
+ // budget ran out before we could re-judge it. On on_exhausted:"proceed"
877
+ // the user opted to accept best-effort output, so promote the revision
878
+ // instead of silently discarding it and keeping the stale pre-revision
879
+ // draft. On "halt" the run errors below, so leave ctx untouched.
880
+ if ((agentCfg.on_exhausted ?? "halt") === "proceed") {
881
+ ctx[reviewsKey] = pendingRevised;
882
+ }
883
+ break;
884
+ }
885
+ // RE-JUDGE: rebuild the judge prompt against the revised artefact. The
886
+ // judge reviews the STAGED draft, not ctx (which still holds the prior
887
+ // accepted value).
888
+ const reJudgePrompt = render(agentCfg.prompt, redactSecretsForPrompt(ctx, secretKeys)) +
889
+ buildJudgeArtefactBlock(pendingRevised) +
890
+ JUDGE_PROMPT_SUFFIX;
891
+ const judged = await runAgentText(reJudgePrompt, agentCfg.driver, agentCfg.model, agentCfg.mcpAccess,
892
+ // M32: pass the judge step's downshift so cost-aware routing applies
893
+ // to re-judge calls in the refine loop, not just the initial judge.
894
+ agentCfg.downshift,
895
+ // Re-judge is a judge call → enforce JSON on supporting drivers.
896
+ { responseFormat: { type: "json_object" } },
897
+ // P0-5: the re-judge re-runs the JUDGE step → keep its sandbox.
898
+ {
899
+ ...(agentCfg.sandbox !== undefined && { sandbox: agentCfg.sandbox }),
900
+ ...(agentCfg.tools !== undefined && { tools: agentCfg.tools }),
901
+ ...(agentCfg.disallowedTools !== undefined && {
902
+ disallowedTools: agentCfg.disallowedTools,
903
+ }),
904
+ });
905
+ if (!judged.ok) {
906
+ // Audit 2026-06-03 (MEDIUM #17): a failed / silent-fail / empty
907
+ // RE-JUDGE can't yield a trustworthy verdict. Mirror the revise-
908
+ // failure break above: stop and KEEP the last good verdict. Parsing
909
+ // the failure/empty text would have produced a bogus verdict (usually
910
+ // "unparseable"), silently dropping the request_changes signal and
911
+ // skipping the on_exhausted gate — the run would proceed as if the
912
+ // (unvalidated) revised draft had been approved.
913
+ break;
914
+ }
915
+ const judgedText = typeof judged.value === "string"
916
+ ? judged.value
917
+ : JSON.stringify(judged.value);
918
+ currentVerdict = parseJudgeVerdict(stripLeadingNarration(judgedText));
919
+ revisions++;
920
+ // R4 #2 (HIGH): an UNPARSEABLE verdict exits the while-loop (only
921
+ // "request_changes" continues it), but the exhaustion gate below fires
922
+ // ONLY on "request_changes" — so an unparseable verdict would leave the
923
+ // run 'ok' with the unvalidated draft never committed and no error.
924
+ // Treat it as a hard, non-ok stop (distinct from the failed-re-judge
925
+ // break above, which keeps the prior good verdict). Do NOT promote the
926
+ // staged draft.
927
+ if (currentVerdict.verdict === "unparseable") {
928
+ const reason = `judge "${judgeStepId}" returned an unparseable verdict after revision`;
929
+ judgeStepResult.judgeVerdict = currentVerdict;
930
+ judgeStepResult.revisions = revisions;
931
+ judgeStepResult.status = "error";
932
+ judgeStepResult.error = reason;
933
+ judgeStepResult.haltReason = reason;
934
+ judgeStepResult.haltCategory = "judge_revisions_exhausted";
935
+ return {
936
+ runError: reason,
937
+ haltAfterFailure: !failOpenAgent,
938
+ };
939
+ }
940
+ // R4 #1: verdict accepted the revision (approve, or non-exhausted
941
+ // continuation). Promote the staged draft to ctx so downstream steps and
942
+ // the next iteration see the improved, judged value.
943
+ ctx[reviewsKey] = pendingRevised;
944
+ }
945
+ // Record the FINAL verdict + the revision count on the judge step result.
946
+ judgeStepResult.judgeVerdict = currentVerdict;
947
+ judgeStepResult.revisions = revisions;
948
+ // EXHAUSTION: still requesting changes after the loop.
949
+ if (currentVerdict.verdict === "request_changes") {
950
+ const onExhausted = agentCfg.on_exhausted ?? "halt";
951
+ if (onExhausted === "halt") {
952
+ const reason = `judge "${judgeStepId}" did not approve after ${maxRevisions} revisions`;
953
+ judgeStepResult.status = "error";
954
+ judgeStepResult.error = reason;
955
+ judgeStepResult.haltReason = reason;
956
+ judgeStepResult.haltCategory = "judge_revisions_exhausted";
957
+ return {
958
+ runError: reason,
959
+ // Respect fail-open like other agent failures.
960
+ haltAfterFailure: !failOpenAgent,
961
+ };
962
+ }
963
+ // "proceed": leave status ok, keep the recorded (unapproved) verdict.
964
+ }
965
+ return { haltAfterFailure: false };
966
+ };
546
967
  // The step loop is wrapped so an uncaught throw from any unguarded
547
968
  // call site (a `when`/prompt render on a malformed step, a path-jail
548
969
  // re-check, etc.) cannot escape `runYamlRecipe` and strand the
@@ -551,6 +972,26 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
551
972
  // finalization path, which marks the run "error".
552
973
  try {
553
974
  for (const step of recipe.steps) {
975
+ // Bug (2): abort on a prior fatal failure. chainedRunner throws (and
976
+ // stops) when a non-optional step fails; the flat runner used to keep
977
+ // going. Break here so later steps don't run on top of a failed
978
+ // dependency. Fail-open failures (step.optional / on_error.fallback=
979
+ // log_only|deliver_original) never set `haltAfterFailure`, so they
980
+ // still let the run continue exactly as before.
981
+ if (haltAfterFailure)
982
+ break;
983
+ // Run-level cancel: abort when the registry controller fires (H11) OR
984
+ // when a caller-provided signal is aborted (#850 parity — external
985
+ // cancellation, e.g. POST /runs/:seq/cancel). An in-flight step is
986
+ // allowed to finish; the next step is not dispatched.
987
+ if (runController?.signal.aborted || deps.signal?.aborted) {
988
+ runError = runError ?? "recipe run cancelled";
989
+ break;
990
+ }
991
+ // Pick up a `~/.patchwork/prices.json` update mid-run for long-running
992
+ // recipes (honours the refreshPrices() contract). No-op unless a usdMax
993
+ // cap is set; never disturbs injected (unit-test) price tables.
994
+ runBudget.refreshPrices();
554
995
  const stepIdForEmit = step.into ?? step.agent?.into ?? `step_${stepsRun}`;
555
996
  const stepTs = Date.now();
556
997
  stepStartTs.set(stepIdForEmit, stepTs);
@@ -567,13 +1008,23 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
567
1008
  // A falsy guard records the step as `skipped`, increments stepsRun, and
568
1009
  // continues — it is NOT a failure. Bridge-dev iMessage recipes rely on
569
1010
  // this to suppress the iMessage agent step when phone is empty.
570
- if (typeof step.when === "string" && step.when.length > 0) {
571
- const rendered = render(step.when, ctx).trim().toLowerCase();
572
- const truthy = !!rendered &&
573
- rendered !== "0" &&
574
- rendered !== "false" &&
575
- rendered !== "null" &&
576
- rendered !== "undefined";
1011
+ if (step.when === false ||
1012
+ (typeof step.when === "string" && step.when.length > 0)) {
1013
+ const rendered = step.when === false
1014
+ ? "false"
1015
+ : render(step.when, ctx).trim().toLowerCase();
1016
+ // Falsy if the WHOLE value is a falsy token OR its LAST token is. The
1017
+ // last-token check is what makes a `when` fed an agent's free-text
1018
+ // decision gate correctly: a step like `decide_file` emits paragraphs of
1019
+ // reasoning that END in "true"/"false" (into: should_file), and a bare
1020
+ // "non-empty ⇒ truthy" check treated that prose as truthy and ran the
1021
+ // guarded step even on a "false" verdict. Single-token guards (the common
1022
+ // case — {{phone}}, {{repo}}, "0") are unchanged: last token === whole
1023
+ // value. Trailing punctuation/backticks/quotes are stripped so `` `false`. ``
1024
+ // still reads false.
1025
+ const FALSY = new Set(["", "0", "false", "null", "undefined"]);
1026
+ const lastToken = (rendered.split(/\s+/).pop() ?? "").replace(/[^a-z0-9]/g, "");
1027
+ const truthy = !!rendered && !FALSY.has(rendered) && !FALSY.has(lastToken);
577
1028
  if (!truthy) {
578
1029
  const skipId = step.into ?? step.agent?.into ?? `step_${stepsRun}`;
579
1030
  stepResults.push({
@@ -596,6 +1047,80 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
596
1047
  continue;
597
1048
  }
598
1049
  }
1050
+ // Bug (3): per-recipe token budget gates ALL step types, not just
1051
+ // agent steps. The admission check used to live inside the
1052
+ // `if (step.agent)` branch, so once the budget was breached the run
1053
+ // kept executing tool steps unbounded. Gate here — after the `when:`
1054
+ // guard resolves truthy, before the agent/tool split — so a breach
1055
+ // halts the run regardless of the next step's kind. Subscription
1056
+ // drivers report no usage and fail open inside RunBudget, so this is
1057
+ // a no-op until a measured agent step actually breaches the cap.
1058
+ const budgetAdmission = runBudget.admit();
1059
+ if (!budgetAdmission.admitted) {
1060
+ const reason = budgetAdmission.reason ??
1061
+ "Run exceeded its token budget — budget_exceeded.";
1062
+ runError = runError ?? reason;
1063
+ haltAfterFailure = true;
1064
+ const budgetStepId = step.into ?? step.agent?.into ?? `step_${stepsRun}`;
1065
+ stepResults.push({
1066
+ id: budgetStepId,
1067
+ tool: step.agent ? "agent" : step.tool,
1068
+ status: "error",
1069
+ error: reason,
1070
+ haltReason: reason,
1071
+ haltCategory: "budget_exceeded",
1072
+ durationMs: 0,
1073
+ });
1074
+ stepsRun++;
1075
+ persistLiveStepResults();
1076
+ emitStepDone(stepIdForEmit);
1077
+ continue;
1078
+ }
1079
+ // M3 — flat-runner approval gate. Safe-by-default: engages for
1080
+ // `manual`-triggered runs (cron/webhook/recipe runs never block
1081
+ // mid-flight) and only when the bridge injected `requireApprovalFn`
1082
+ // (i.e. approvalGate != "off"). Per-recipe opt-out via
1083
+ // `requireApproval: false`. The injected fn applies the tier threshold
1084
+ // itself and returns `true` for steps that don't need sign-off; a
1085
+ // `false` result is an explicit human rejection → halt the run.
1086
+ //
1087
+ // worker.autonomy: when `gateAutomatedRuns` is set the gate ALSO engages
1088
+ // on automated triggers (that's how workers run), and `requireApprovalFn`
1089
+ // is the worker-aware fn — reversible actions pass, risky-unearned ones
1090
+ // queue. Off → manual-only, byte-identical to pre-flip behaviour.
1091
+ if (deps.requireApprovalFn &&
1092
+ (recipeTriggerKind === "manual" || deps.gateAutomatedRuns) &&
1093
+ recipe.requireApproval !== false) {
1094
+ const approvalToolId = step.agent ? "agent" : (step.tool ?? "unknown");
1095
+ const approved = await deps.requireApprovalFn({
1096
+ toolId: approvalToolId,
1097
+ tier: classifyTool(approvalToolId),
1098
+ summary: step.agent
1099
+ ? `agent step${step.agent.into ? ` → ${step.agent.into}` : ""}`
1100
+ : `tool ${approvalToolId}`,
1101
+ params: step.agent ? undefined : step,
1102
+ ...(effectiveRunSignal && { signal: effectiveRunSignal }), // L1
1103
+ });
1104
+ if (!approved) {
1105
+ const reason = `Step rejected by approval gate — approval_rejected.`;
1106
+ runError = runError ?? reason;
1107
+ haltAfterFailure = true;
1108
+ const rejId = step.into ?? step.agent?.into ?? `step_${stepsRun}`;
1109
+ stepResults.push({
1110
+ id: rejId,
1111
+ tool: step.agent ? "agent" : step.tool,
1112
+ status: "error",
1113
+ error: reason,
1114
+ haltReason: reason,
1115
+ haltCategory: "approval_rejected",
1116
+ durationMs: 0,
1117
+ });
1118
+ stepsRun++;
1119
+ persistLiveStepResults();
1120
+ emitStepDone(stepIdForEmit);
1121
+ continue;
1122
+ }
1123
+ }
599
1124
  // Handle agent steps separately
600
1125
  if (step.agent) {
601
1126
  const agentCfg = step.agent;
@@ -603,7 +1128,7 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
603
1128
  // PR3a: judge prompt convention. Append the structured-verdict
604
1129
  // suffix and, when `reviews: <stepId>` is set, inject the
605
1130
  // upstream step's output as an <artefact> block.
606
- let renderedPrompt = render(agentCfg.prompt, ctx);
1131
+ let renderedPrompt = render(agentCfg.prompt, redactSecretsForPrompt(ctx, secretKeys));
607
1132
  if (isJudge) {
608
1133
  if (agentCfg.reviews) {
609
1134
  renderedPrompt += buildJudgeArtefactBlock(ctx[agentCfg.reviews]);
@@ -613,44 +1138,81 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
613
1138
  const intoKey = agentCfg.into ?? "agent_output";
614
1139
  const stepId = intoKey;
615
1140
  const stepStart = Date.now();
1141
+ // P1: fresh per-step usage accumulator for this agent step (and any
1142
+ // judge→refine re-runs it spawns via runAgentText, which share it).
1143
+ currentStepUsage = newStepUsageAccumulator();
616
1144
  let agentResult;
617
- // PR2b: per-recipe token budget. Admission check before dispatch;
618
- // reconcile actual consumption after. Subscription drivers
619
- // (Claude CLI, provider subprocess) report `usage === undefined`
620
- // — `RunBudget.reconcile` records a fail-open warning per driver
621
- // per run and continues.
622
- const admission = runBudget.admit();
623
- if (!admission.admitted) {
624
- const reason = admission.reason ??
625
- "Run exceeded its token budget — budget_exceeded.";
626
- runError = runError ?? reason;
627
- stepResults.push({
628
- id: stepId,
629
- tool: "agent",
630
- status: "error",
631
- error: reason,
632
- haltReason: reason,
633
- haltCategory: "budget_exceeded",
634
- durationMs: 0,
635
- });
636
- stepsRun++;
637
- persistLiveStepResults();
638
- emitStepDone(stepIdForEmit);
639
- continue;
640
- }
1145
+ // Bug (2): fail-open semantics for THIS agent step. Mirrors the
1146
+ // tool-branch `failOpen` (step.optional OR recipe-level
1147
+ // on_error.fallback=log_only|deliver_original). Used to decide
1148
+ // whether an agent failure is fatal (sets `haltAfterFailure`, which
1149
+ // aborts the run at the next loop top) or fail-open (records the
1150
+ // error but lets the run continue, as before).
1151
+ const agentFallback = recipe.on_error?.fallback;
1152
+ const agentFallbackFailOpen = agentFallback === "log_only" || agentFallback === "deliver_original";
1153
+ const failOpenAgent = step.optional === true || agentFallbackFailOpen;
1154
+ // PR2b: per-recipe token budget. Admission is now checked once at the
1155
+ // top of the loop (Bug (3)) so it gates tool steps too; here we only
1156
+ // reconcile actual consumption after the call. Subscription drivers
1157
+ // (Claude CLI, provider subprocess) report `usage === undefined` —
1158
+ // `RunBudget.reconcile` records a fail-open warning per driver per
1159
+ // run and continues.
641
1160
  try {
1161
+ // Phase 4: opt-in cost-aware routing. No-op (returns preferred) when
1162
+ // the step has no `downshift` list or no USD cap is set.
1163
+ const routed = resolveRouting({ driver: agentCfg.driver, model: agentCfg.model }, agentCfg.downshift, renderedPrompt, runBudget);
1164
+ // Worker.autonomy: fold the worker's agent-step sandbox into this
1165
+ // step's own deny list so the subprocess can't bypass the gate.
1166
+ const agentDisallowed = mergeAgentDisallowedTools(agentCfg.disallowedTools, deps.agentDisallowedTools);
642
1167
  const agentReturn = await _executeAgent({
643
1168
  prompt: renderedPrompt,
644
- driver: agentCfg.driver === "api" ? "anthropic" : agentCfg.driver,
645
- model: agentCfg.model,
1169
+ driver: routed.driver === "api" ? "anthropic" : routed.driver,
1170
+ model: routed.model,
646
1171
  ...(agentCfg.mcpAccess !== undefined && {
647
1172
  mcpAccess: agentCfg.mcpAccess,
648
1173
  }),
1174
+ // P0-5 opt-in tool sandbox: thread sandbox + allow/deny lists onto
1175
+ // the executor input so the subprocess driver can enforce them via
1176
+ // --allowed-tools / --disallowed-tools / --permission-mode dontAsk.
1177
+ ...(agentCfg.sandbox !== undefined && {
1178
+ sandbox: agentCfg.sandbox,
1179
+ }),
1180
+ ...(agentCfg.tools !== undefined && {
1181
+ allowedTools: agentCfg.tools,
1182
+ }),
1183
+ ...(agentDisallowed !== undefined && {
1184
+ disallowedTools: agentDisallowed,
1185
+ }),
1186
+ // Worker sandbox is enforceable only on the subprocess driver;
1187
+ // fail closed on any other driver rather than run un-sandboxed.
1188
+ ...(deps.agentDisallowedTools?.length && {
1189
+ enforceSandbox: true,
1190
+ }),
1191
+ // Constrained decoding: enforce a pure-JSON verdict on judge steps
1192
+ // (OpenAI-compatible drivers honor it; others ignore it). Pairs
1193
+ // with the pure-JSON JUDGE_PROMPT_SUFFIX + tolerant parser.
1194
+ ...(isJudge && {
1195
+ providerOptions: { responseFormat: { type: "json_object" } },
1196
+ }),
649
1197
  }, buildAgentExecutorDeps(stepDeps, deps));
650
1198
  agentResult = agentReturn.text;
651
- runBudget.reconcile(agentCfg.driver === "api"
652
- ? "anthropic"
653
- : (agentCfg.driver ?? "auto"), agentReturn.usage);
1199
+ runBudget.reconcile(
1200
+ // Prefer the driver executeAgent actually resolved+ran; the routed
1201
+ // value is only the fallback for non-executeAgent callers (it is
1202
+ // often undefined → previously logged "auto").
1203
+ agentReturn.servedBy?.driver ??
1204
+ (routed.driver === "api"
1205
+ ? "anthropic"
1206
+ : (routed.driver ?? "auto")), agentReturn.usage,
1207
+ // Resolved model for USD pricing (Phase 3); absent → fail open.
1208
+ agentReturn.servedBy?.model,
1209
+ // Char counts for the opt-in unmeasured-driver ≈$ estimate.
1210
+ {
1211
+ inputChars: renderedPrompt.length,
1212
+ outputChars: agentReturn.text.length,
1213
+ });
1214
+ // P1: fold this primary agent call into the current step's usage.
1215
+ accumulateAgentUsage(currentStepUsage, agentReturn.usage, agentReturn.servedBy, priceTable);
654
1216
  // Catch both `[agent step failed: ...]` (existing) and the
655
1217
  // silent-fail patterns `[agent step skipped: ...]` etc. via the
656
1218
  // shared detector. Per-step opt-out via `silentFailDetection: false`.
@@ -663,6 +1225,8 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
663
1225
  ? `silent-fail detected (${agentSilentFail.reason}): ${agentSilentFail.matched}`
664
1226
  : agentResult;
665
1227
  runError = runError ?? reason;
1228
+ if (!failOpenAgent)
1229
+ haltAfterFailure = true;
666
1230
  stepResults.push({
667
1231
  id: stepId,
668
1232
  tool: "agent",
@@ -680,6 +1244,8 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
680
1244
  if (!stripped.trim()) {
681
1245
  const errMsg = `[agent step failed: ${agentCfg.driver ?? "agent"} returned only narration or whitespace — no content]`;
682
1246
  runError = runError ?? errMsg;
1247
+ if (!failOpenAgent)
1248
+ haltAfterFailure = true;
683
1249
  stepResults.push({
684
1250
  id: stepId,
685
1251
  tool: "agent",
@@ -695,12 +1261,15 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
695
1261
  try {
696
1262
  const jsonMatch = /```(?:json)?\s*([\s\S]*?)```/.exec(stripped) ?? [null, stripped];
697
1263
  const parsed = sanitizeParsed(JSON.parse((jsonMatch[1] ?? "").trim()));
698
- ctx[intoKey] = parsed;
1264
+ if (!isJudge)
1265
+ ctx[intoKey] = parsed;
699
1266
  }
700
1267
  catch {
701
- ctx[intoKey] = stripped;
1268
+ if (!isJudge)
1269
+ ctx[intoKey] = stripped;
702
1270
  }
703
- outputs.push(intoKey);
1271
+ if (!isJudge)
1272
+ outputs.push(intoKey);
704
1273
  // PR3a: parse + stash the judge verdict on the step result.
705
1274
  // Augment-only: a `request_changes` verdict still yields
706
1275
  // `status: "ok"`. The verdict surfaces via the runlog +
@@ -708,13 +1277,43 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
708
1277
  const judgeVerdict = isJudge
709
1278
  ? parseJudgeVerdict(stripped)
710
1279
  : undefined;
711
- stepResults.push({
1280
+ const judgeStepResult = {
712
1281
  id: stepId,
713
1282
  tool: "agent",
714
1283
  status: "ok",
715
1284
  ...(judgeVerdict !== undefined && { judgeVerdict }),
716
1285
  durationMs: Date.now() - stepStart,
717
- });
1286
+ };
1287
+ stepResults.push(judgeStepResult);
1288
+ // ── OPT-IN judge → refine loop ───────────────────────────────
1289
+ // ⚠️ INVARIANT DEPARTURE: when the judge step opts in via
1290
+ // `max_revisions > 0`, a `request_changes` verdict now DRIVES a
1291
+ // bounded revise→re-judge loop instead of merely stashing the
1292
+ // verdict. This deliberately departs the augment-only invariant
1293
+ // (see judgeVerdict.ts) — but ONLY when the opt-in fields are
1294
+ // present. With them absent the block below is skipped entirely
1295
+ // and behavior is byte-identical to the PR3a augment-only path.
1296
+ if (isJudge &&
1297
+ agentCfg.reviews &&
1298
+ typeof agentCfg.max_revisions === "number" &&
1299
+ agentCfg.max_revisions > 0 &&
1300
+ judgeVerdict?.verdict === "request_changes") {
1301
+ const loopOutcome = await runJudgeRefineLoop({
1302
+ agentCfg,
1303
+ reviewsKey: agentCfg.reviews,
1304
+ maxRevisions: agentCfg.max_revisions,
1305
+ judgeStepId: stepId,
1306
+ firstVerdict: judgeVerdict,
1307
+ judgeStepResult,
1308
+ failOpenAgent,
1309
+ });
1310
+ if (loopOutcome.runError !== undefined) {
1311
+ runError = runError ?? loopOutcome.runError;
1312
+ }
1313
+ if (loopOutcome.haltAfterFailure) {
1314
+ haltAfterFailure = true;
1315
+ }
1316
+ }
718
1317
  // Slice 2 — per-step expect eval. Runs on the value just
719
1318
  // committed to ctx[intoKey]. Halt failure flips the just-pushed
720
1319
  // result to error and rolls back the ctx commit so downstream
@@ -730,11 +1329,9 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
730
1329
  last.error = `expect_failed: ${failures.join("; ")}`;
731
1330
  last.haltReason = `expect_failed in step "${stepId}": ${failures.join("; ")}`;
732
1331
  last.haltCategory = "expect_failed";
733
- const fbk = recipe.on_error?.fallback;
734
- const fbkOpen = fbk === "log_only" || fbk === "deliver_original";
735
- const failOpenAgent = step.optional === true || fbkOpen;
736
1332
  if (!failOpenAgent) {
737
1333
  runError = runError ?? last.haltReason;
1334
+ haltAfterFailure = true;
738
1335
  }
739
1336
  delete ctx[intoKey];
740
1337
  }
@@ -750,6 +1347,8 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
750
1347
  catch (err) {
751
1348
  const msg = err instanceof Error ? err.message : String(err);
752
1349
  runError = runError ?? `agent step "${stepId}" failed: ${msg}`;
1350
+ if (!failOpenAgent)
1351
+ haltAfterFailure = true;
753
1352
  stepResults.push({
754
1353
  id: stepId,
755
1354
  tool: "agent",
@@ -760,6 +1359,14 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
760
1359
  durationMs: Date.now() - stepStart,
761
1360
  });
762
1361
  }
1362
+ // P1: attach this agent step's summed token usage (across primary +
1363
+ // any judge→refine re-runs) to the result just pushed, and fold it
1364
+ // into the run-level total. Fields are ABSENT when no usage measured.
1365
+ const pushedAgentResult = stepResults[stepResults.length - 1];
1366
+ if (pushedAgentResult) {
1367
+ Object.assign(pushedAgentResult, stepUsageFields(currentStepUsage));
1368
+ }
1369
+ foldStepIntoRun(runUsage, currentStepUsage);
763
1370
  stepsRun++;
764
1371
  persistLiveStepResults();
765
1372
  emitStepDone(stepIdForEmit);
@@ -768,10 +1375,22 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
768
1375
  const stepStart = Date.now();
769
1376
  const stepId = step.into ?? `step_${stepsRun}`;
770
1377
  // Resolve retry policy: step-level overrides recipe-level.
771
- const retryCount = step.retry ?? recipe.on_error?.retry ?? 0;
1378
+ // Clamp to 0 as a safety net against negative values slipping past
1379
+ // schema validation (M31: negative retry loops 0 times, skipping step).
1380
+ const retryCount = Math.max(0, step.retry ?? recipe.on_error?.retry ?? 0);
772
1381
  const retryDelayMs = step.retryDelay ?? recipe.on_error?.retryDelay ?? 1000;
773
1382
  let result = null;
774
1383
  let stepError;
1384
+ // Bug (2): distinguish a HARD tool error (a thrown error or a
1385
+ // `{ok:false}` JSON envelope) from a SOFT silent-fail detection
1386
+ // (`{count:0,error}` connector envelopes, string placeholders). Only
1387
+ // hard failures abort the run; soft silent-fail detections keep the
1388
+ // run going so connector health-check recipes can still deliver the
1389
+ // degraded payload downstream (a long-standing, tested contract —
1390
+ // see "linear.list_issues — returns error payload" tests). Silent-fail
1391
+ // detection is an observability augment; it was never meant to gate
1392
+ // delivery for these envelopes.
1393
+ let stepErrorIsSilentFail = false;
775
1394
  let thrownError;
776
1395
  let thrownErrorCode;
777
1396
  for (let attempt = 0; attempt <= retryCount; attempt++) {
@@ -779,6 +1398,7 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
779
1398
  await new Promise((r) => setTimeout(r, retryDelayMs));
780
1399
  }
781
1400
  stepError = undefined;
1401
+ stepErrorIsSilentFail = false;
782
1402
  thrownError = undefined;
783
1403
  thrownErrorCode = undefined;
784
1404
  try {
@@ -835,6 +1455,7 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
835
1455
  const detected = detectSilentFail(result);
836
1456
  if (detected) {
837
1457
  stepError = `silent-fail detected (${detected.reason}): ${detected.matched}`;
1458
+ stepErrorIsSilentFail = true;
838
1459
  }
839
1460
  }
840
1461
  }
@@ -850,6 +1471,21 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
850
1471
  }
851
1472
  if (!stepError && !thrownError)
852
1473
  break;
1474
+ // Audit 2026-06-10 recipe-runners-2: do NOT retry on a step_timeout.
1475
+ // The timed-out attempt's underlying tool call keeps running in the
1476
+ // background (Promise.race only abandons the wait, it does not cancel
1477
+ // the call). Re-issuing the step here is, at best, pointless — for a
1478
+ // write tool the in-flight idempotency ledger short-circuits the retry
1479
+ // to the SAME promise (no second side effect, but also no progress) —
1480
+ // and, at worst, a second side effect for any tool the ledger cannot
1481
+ // dedup (non-write tools, or a write tool whose first attempt already
1482
+ // committed its effect then threw). A true cancel needs an AbortSignal
1483
+ // threaded through every tool/connector call, which is out of scope
1484
+ // here; until then, refusing to retry on timeout is the safe contract.
1485
+ // (Genuine transient failures — non-timeout throws / {ok:false} — still
1486
+ // retry below.)
1487
+ if (thrownError?.startsWith("step_timeout:"))
1488
+ break;
853
1489
  }
854
1490
  // Recipe-level fallback: log_only / deliver_original treat step failure
855
1491
  // as non-fatal (fail-open) — same semantics as step-level optional: true.
@@ -872,6 +1508,7 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
872
1508
  });
873
1509
  if (!failOpen) {
874
1510
  runError = runError ?? `${step.tool} failed: ${thrownError}`;
1511
+ haltAfterFailure = true;
875
1512
  }
876
1513
  else if (fallbackFailOpen && !step.optional) {
877
1514
  console.warn(`step ${stepId} failed but on_error.fallback=${fallback} — treating as non-fatal: ${thrownError}`);
@@ -880,6 +1517,24 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
880
1517
  else {
881
1518
  const finalStatus = result === null ? "skipped" : stepError ? "error" : "ok";
882
1519
  const retryNote = retryCount > 0 ? ` after ${retryCount + 1} attempts` : "";
1520
+ // Outcome attribution: capture the filed-issue URL on github.create_issue
1521
+ // steps so trust-replay can look up the issue's eventual disposition in
1522
+ // the outcome store (confirmed/junk/unknown). Targeted to this one tool
1523
+ // only — storing all tool outputs would bloat the run log unnecessarily.
1524
+ let stepOutput;
1525
+ if (finalStatus === "ok" &&
1526
+ result !== null &&
1527
+ step.tool === "github.create_issue") {
1528
+ try {
1529
+ const parsed = JSON.parse(result);
1530
+ if (typeof parsed.url === "string") {
1531
+ stepOutput = { url: parsed.url, issueNumber: parsed.issueNumber };
1532
+ }
1533
+ }
1534
+ catch {
1535
+ /* non-JSON or missing url — skip output capture */
1536
+ }
1537
+ }
883
1538
  stepResults.push({
884
1539
  id: stepId,
885
1540
  tool: step.tool,
@@ -891,11 +1546,17 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
891
1546
  haltCategory: "tool_error",
892
1547
  }
893
1548
  : {}),
1549
+ ...(stepOutput !== undefined ? { output: stepOutput } : {}),
894
1550
  durationMs: Date.now() - stepStart,
895
1551
  });
896
1552
  if (stepError) {
897
1553
  if (!failOpen) {
898
1554
  runError = runError ?? `${step.tool} failed: ${stepError}`;
1555
+ // Soft silent-fail detections (connector error envelopes) record
1556
+ // the error but must NOT abort the run — see the
1557
+ // `stepErrorIsSilentFail` note above. Hard `{ok:false}` errors do.
1558
+ if (!stepErrorIsSilentFail)
1559
+ haltAfterFailure = true;
899
1560
  }
900
1561
  else if (fallbackFailOpen && !step.optional) {
901
1562
  console.warn(`step ${stepId} failed but on_error.fallback=${fallback} — treating as non-fatal: ${stepError}`);
@@ -932,6 +1593,7 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
932
1593
  last.haltCategory = "expect_failed";
933
1594
  if (!failOpen) {
934
1595
  runError = runError ?? last.haltReason;
1596
+ haltAfterFailure = true;
935
1597
  }
936
1598
  result = null;
937
1599
  }
@@ -967,6 +1629,13 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
967
1629
  const msg = err instanceof Error ? err.message : String(err);
968
1630
  runError = runError ?? `recipe run aborted: ${msg}`;
969
1631
  }
1632
+ finally {
1633
+ // Drop the run from the registry (success, failure, or cancel) so
1634
+ // the seq can't be cancelled post-hoc and the map doesn't leak (H11).
1635
+ if (runController !== undefined && runSeq !== undefined) {
1636
+ unregisterRun(runSeq);
1637
+ }
1638
+ }
970
1639
  // Evaluate expect block before persisting so failures are stored in the
971
1640
  // run log. Guarded: a throw here must not skip finalization and strand
972
1641
  // the run at "running".
@@ -998,8 +1667,21 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
998
1667
  ...(s.haltReason ? { haltReason: s.haltReason } : {}),
999
1668
  ...(s.haltCategory ? { haltCategory: s.haltCategory } : {}),
1000
1669
  ...(s.judgeVerdict ? { judgeVerdict: s.judgeVerdict } : {}),
1670
+ // P1: carry per-step token usage through to the persisted run row.
1671
+ // Absent for tool / unmeasured-driver steps (round-trips unchanged).
1672
+ ...(typeof s.inputTokens === "number"
1673
+ ? { inputTokens: s.inputTokens }
1674
+ : {}),
1675
+ ...(typeof s.outputTokens === "number"
1676
+ ? { outputTokens: s.outputTokens }
1677
+ : {}),
1678
+ ...(typeof s.costUsd === "number" ? { costUsd: s.costUsd } : {}),
1001
1679
  durationMs: s.durationMs,
1002
1680
  }));
1681
+ // P1: run-level token aggregate + budget totals (latter only when a
1682
+ // budget was configured — never persist all-zero no-budget totals).
1683
+ const tokenTotals = runTokenTotals(runUsage);
1684
+ const budgetTotals = recipe.budget ? runBudget.totals() : undefined;
1003
1685
  if (deps.runLog && runSeq !== undefined) {
1004
1686
  deps.runLog.completeRun(runSeq, {
1005
1687
  status: runError ? "error" : "done",
@@ -1010,6 +1692,11 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
1010
1692
  ...(runError !== undefined && { errorMessage: runError }),
1011
1693
  ...(assertionFailures.length > 0 ? { assertionFailures } : {}),
1012
1694
  ...(inboxOutputs.length > 0 ? { inboxOutputs } : {}),
1695
+ ...(runBudget.finalWarnings().length > 0
1696
+ ? { budgetWarnings: runBudget.finalWarnings() }
1697
+ : {}),
1698
+ ...(tokenTotals ? { tokenTotals } : {}),
1699
+ ...(budgetTotals ? { budgetTotals } : {}),
1013
1700
  });
1014
1701
  emit("recipe_done", {
1015
1702
  runSeq,
@@ -1047,6 +1734,8 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
1047
1734
  stepResults: finalStepResults,
1048
1735
  ...(assertionFailures.length > 0 ? { assertionFailures } : {}),
1049
1736
  ...(inboxOutputs.length > 0 ? { inboxOutputs } : {}),
1737
+ ...(tokenTotals ? { tokenTotals } : {}),
1738
+ ...(budgetTotals ? { budgetTotals } : {}),
1050
1739
  });
1051
1740
  }
1052
1741
  }
@@ -1093,6 +1782,14 @@ export async function runYamlRecipe(recipe, deps = {}, seedContext = {}) {
1093
1782
  stepResults,
1094
1783
  errorMessage: runError,
1095
1784
  ...(assertionFailures.length > 0 ? { assertionFailures } : {}),
1785
+ ...(runBudget.finalWarnings().length > 0
1786
+ ? { budgetWarnings: runBudget.finalWarnings() }
1787
+ : {}),
1788
+ // P1: forward run-level token aggregate to callers / persisters.
1789
+ ...(() => {
1790
+ const tt = runTokenTotals(runUsage);
1791
+ return tt ? { tokenTotals: tt } : {};
1792
+ })(),
1096
1793
  };
1097
1794
  }
1098
1795
  export async function executeStep(step, ctx, deps) {
@@ -1296,6 +1993,18 @@ export function defaultGitStaleBranches(days, workdir) {
1296
1993
  return "(git branches unavailable)";
1297
1994
  }
1298
1995
  }
1996
+ /**
1997
+ * True when running under the vitest harness (same VITEST / NODE_ENV signal
1998
+ * `src/recipes/migrations/index.ts` guards on). Used only to DEFAULT `testMode` on so a
1999
+ * bare `runYamlRecipe(...)` in a unit test never appends a synthetic row to the
2000
+ * operator's real `~/.patchwork/runs.jsonl` — which is also the de-facto
2001
+ * worker-trust store and rotates at 1 MB / 10k lines, so test rows would evict
2002
+ * real trust evidence and pollute every operator halt surface. An explicit
2003
+ * `deps.testMode` (true or false) always wins over this default.
2004
+ */
2005
+ function isVitestEnv() {
2006
+ return process.env.VITEST != null || process.env.NODE_ENV === "test";
2007
+ }
1299
2008
  /** Resolve all RunnerDeps to concrete StepDeps with production defaults filled in. */
1300
2009
  function resolveStepDeps(deps, scope) {
1301
2010
  const workdir = deps.workdir ?? process.cwd();
@@ -1354,7 +2063,8 @@ function resolveStepDeps(deps, scope) {
1354
2063
  return getValidAccessToken();
1355
2064
  }),
1356
2065
  logDir: deps.logDir,
1357
- testMode: deps.testMode ?? false,
2066
+ activityLog: deps.activityLog,
2067
+ testMode: deps.testMode ?? isVitestEnv(),
1358
2068
  // PR5a/b: per-attempt idempotency ledger. Disk-backed when
1359
2069
  // `ledgerDir` + `manualRunId` + recipe name are all available so a
1360
2070
  // retry of the same logical attempt re-uses prior records (resume
@@ -1383,12 +2093,19 @@ function buildAgentExecutorDeps(stepDeps, runnerDeps, claudeCodeFnOverride) {
1383
2093
  const claudeCliFn = claudeCodeFnOverride ?? stepDeps.claudeCodeFn;
1384
2094
  return {
1385
2095
  anthropicFn: async (prompt, model) => toAgentResult(await stepDeps.claudeFn(prompt, model)),
1386
- providerDriverFn: async (driver, prompt, model) => toAgentResult(await stepDeps.providerDriverFn(driver, prompt, model)),
2096
+ providerDriverFn: async (driver, prompt, model, providerOptions) => toAgentResult(
2097
+ // Keep the 3-arg call shape when unconstrained (backward-compatible
2098
+ // with deps.providerDriverFn mocks that assert exact arity).
2099
+ providerOptions
2100
+ ? await stepDeps.providerDriverFn(driver, prompt, model, providerOptions)
2101
+ : await stepDeps.providerDriverFn(driver, prompt, model)),
1387
2102
  claudeCliFn: async (prompt, opts) => toAgentResult(await claudeCliFn(prompt, opts)),
1388
2103
  localFn: async (prompt, model) => toAgentResult(await stepDeps.localFn(prompt, model)),
1389
2104
  probeClaudeCli: () => {
1390
2105
  if (runnerDeps.claudeFn !== undefined)
1391
2106
  return false;
2107
+ if (_claudeCliProbeCache !== undefined)
2108
+ return _claudeCliProbeCache.result;
1392
2109
  // Use the same resolution as defaultClaudeCodeFn so the auto-detect
1393
2110
  // branch in agentExecutor.ts doesn't probe "claude" via PATH and
1394
2111
  // then later fail to spawn the configured override (or vice versa).
@@ -1396,7 +2113,8 @@ function buildAgentExecutorDeps(stepDeps, runnerDeps, claudeCodeFnOverride) {
1396
2113
  encoding: "utf-8",
1397
2114
  timeout: 5000,
1398
2115
  });
1399
- return !probe.error;
2116
+ _claudeCliProbeCache = { result: !probe.error };
2117
+ return _claudeCliProbeCache.result;
1400
2118
  },
1401
2119
  loadPatchworkConfig: () => {
1402
2120
  // Synchronous static import — earlier `require()` form silently failed
@@ -1454,19 +2172,35 @@ export function defaultClaudeCodeFn(prompt, opts) {
1454
2172
  if (opts?.mcpAccess === true) {
1455
2173
  return Promise.resolve("[agent step failed: recipe_mcp_unsupported — defaultClaudeCodeFn does not support mcpAccess:true; route via SubprocessDriver or unset the mcpAccess flag on this step]");
1456
2174
  }
2175
+ // P0-5 opt-in tool sandbox on the `recipe run --local` / non-bridge path.
2176
+ // Without this the sandbox would be silently ignored here (a one-path gap);
2177
+ // mirror the SubprocessDriver argv rule (§3): filter argv-injection values,
2178
+ // run in --permission-mode dontAsk + --allowed-tools when sandbox is active,
2179
+ // and always apply --disallowed-tools regardless of mode.
2180
+ const sandboxAllowed = (Array.isArray(opts?.allowedTools) ? opts.allowedTools : []).filter((t) => typeof t === "string" && t.length > 0 && !t.startsWith("-"));
2181
+ const sandboxDenied = (Array.isArray(opts?.disallowedTools) ? opts.disallowedTools : []).filter((t) => typeof t === "string" && t.length > 0 && !t.startsWith("-"));
2182
+ const localArgs = [
2183
+ "-p",
2184
+ prompt,
2185
+ // --strict-mcp-config: never load ~/.claude.json or .mcp.json. Recipes
2186
+ // are sandboxed by default (mcpAccess defaults to false above). This
2187
+ // also prevents accidental session attachment when the parent process
2188
+ // had a bridge MCP entry in ~/.claude.json.
2189
+ "--strict-mcp-config",
2190
+ "--system-prompt",
2191
+ "You are a helpful assistant processing a recipe task. Use ONLY the data explicitly provided in the user message — treat it as ground truth. Do not call tools to look up git history, emails, or any other information; all necessary data is already included.",
2192
+ "--no-session-persistence",
2193
+ ];
2194
+ if (opts?.sandbox === true && sandboxAllowed.length > 0) {
2195
+ localArgs.push("--permission-mode", "dontAsk");
2196
+ localArgs.push("--allowed-tools", ...sandboxAllowed);
2197
+ }
2198
+ // Deny rules apply in ANY mode.
2199
+ if (sandboxDenied.length > 0) {
2200
+ localArgs.push("--disallowed-tools", ...sandboxDenied);
2201
+ }
1457
2202
  try {
1458
- const result = spawnSync(binary, [
1459
- "-p",
1460
- prompt,
1461
- // --strict-mcp-config: never load ~/.claude.json or .mcp.json. Recipes
1462
- // are sandboxed by default (mcpAccess defaults to false above). This
1463
- // also prevents accidental session attachment when the parent process
1464
- // had a bridge MCP entry in ~/.claude.json.
1465
- "--strict-mcp-config",
1466
- "--system-prompt",
1467
- "You are a helpful assistant processing a recipe task. Use ONLY the data explicitly provided in the user message — treat it as ground truth. Do not call tools to look up git history, emails, or any other information; all necessary data is already included.",
1468
- "--no-session-persistence",
1469
- ], {
2203
+ const result = spawnSync(binary, localArgs, {
1470
2204
  cwd: workspace.path,
1471
2205
  // sanitizeEnv strips CLAUDECODE / CLAUDE_CODE_* / MCP_* from the
1472
2206
  // child so the spawn doesn't re-authenticate as, or nest under,
@@ -1493,10 +2227,74 @@ export function defaultClaudeCodeFn(prompt, opts) {
1493
2227
  return Promise.resolve(`[agent step failed: ${err instanceof Error ? err.message : String(err)}]`);
1494
2228
  }
1495
2229
  }
2230
+ /**
2231
+ * Map a driver's `providerMeta` to AgentUsage. Returns undefined unless BOTH
2232
+ * token counts are present as numbers — a half-populated count would mislead
2233
+ * RunBudget. Pure + exported for tests.
2234
+ */
2235
+ export function providerMetaToUsage(meta) {
2236
+ if (!meta)
2237
+ return undefined;
2238
+ const inputTokens = meta.inputTokens;
2239
+ const outputTokens = meta.outputTokens;
2240
+ if (typeof inputTokens === "number" && typeof outputTokens === "number") {
2241
+ // Reject NaN/Infinity/negative counts: a negative count would price to a
2242
+ // negative cost and silently *reduce* usdSpent, defeating the usdMax cap.
2243
+ if (!Number.isFinite(inputTokens) ||
2244
+ inputTokens < 0 ||
2245
+ !Number.isFinite(outputTokens) ||
2246
+ outputTokens < 0) {
2247
+ return undefined;
2248
+ }
2249
+ return { inputTokens, outputTokens };
2250
+ }
2251
+ return undefined;
2252
+ }
2253
+ const ROUTER_CHARS_PER_TOKEN = 4;
2254
+ /**
2255
+ * Empirical output:input token ratio used for pre-dispatch cost estimates.
2256
+ * LLMs typically produce far fewer output tokens than they consume on input for
2257
+ * most agentic tasks (completion, classification, summarisation). The old 1:1
2258
+ * assumption made models appear 2–5× more expensive than reality, causing
2259
+ * unnecessary downshifts to cheaper models.
2260
+ *
2261
+ * 0.3 is a deliberately-conservative upper bound (real ratios are often 0.1–0.2
2262
+ * for short-form steps). Using a higher-than-typical value avoids under-estimating
2263
+ * cost and over-spending, while still being far more accurate than 1:1.
2264
+ *
2265
+ * The real cost is always reconciled after the call (see the cost-routing ADR),
2266
+ * so this estimate only affects routing decisions, never final billing.
2267
+ */
2268
+ const ROUTER_OUTPUT_RATIO = 0.3;
2269
+ /**
2270
+ * Apply opt-in cost-aware routing (Phase 4) to choose the driver/model for an
2271
+ * agent dispatch. Returns `preferred` UNCHANGED when there is no downshift list
2272
+ * or no USD cap is set (byte-identical to no routing). The output-token figure
2273
+ * uses a 0.3:1 output:input estimate (conservative upper bound; the 1:1 default
2274
+ * doubled apparent cost and caused unnecessary model downshifts — audit
2275
+ * 2026-06-03 LOW #7). The real cost is reconciled after the call.
2276
+ * Exported for unit testing.
2277
+ */
2278
+ export function resolveRouting(preferred, downshift, promptText, budget) {
2279
+ if (!downshift || downshift.length === 0)
2280
+ return preferred;
2281
+ const remainingUsd = budget.remainingUsd();
2282
+ if (remainingUsd === undefined)
2283
+ return preferred; // no USD cap → no routing
2284
+ const estInputTokens = Math.ceil(promptText.length / ROUTER_CHARS_PER_TOKEN);
2285
+ // Fix (audit 2026-06-03 LOW #7): use a realistic 0.3:1 output:input ratio
2286
+ // instead of 1:1. LLMs produce far fewer output tokens than input for most
2287
+ // tasks; 0.3 is a conservative upper bound that avoids under-estimating cost.
2288
+ const estOutputTokens = Math.ceil(estInputTokens * ROUTER_OUTPUT_RATIO);
2289
+ return costRouter(preferred, downshift, {
2290
+ remainingUsd,
2291
+ quote: (driver, model) => budget.quoteUsd(driver, model, estInputTokens, estOutputTokens),
2292
+ });
2293
+ }
1496
2294
  /** Returns a providerDriverFn with a per-run driver cache (not shared across runs). */
1497
- function makeProviderDriverFn() {
2295
+ export function makeProviderDriverFn() {
1498
2296
  const cache = new Map();
1499
- return async function defaultProviderDriverFn(driverName, prompt, model) {
2297
+ return async function defaultProviderDriverFn(driverName, prompt, model, providerOptions) {
1500
2298
  try {
1501
2299
  let driver = cache.get(driverName);
1502
2300
  if (!driver) {
@@ -1520,14 +2318,45 @@ function makeProviderDriverFn() {
1520
2318
  startupTimeoutMs,
1521
2319
  signal: controller.signal,
1522
2320
  model,
2321
+ ...(providerOptions && { providerOptions }),
1523
2322
  });
1524
2323
  if (result.exitCode !== undefined && result.exitCode !== 0) {
1525
2324
  const detail = result.stderrTail ?? result.text ?? "";
1526
2325
  return `[agent step failed: ${driverName} exited ${result.exitCode}${detail ? ` — ${detail.slice(0, 200)}` : ""}]`;
1527
2326
  }
2327
+ // API drivers (OpenAI / Grok) never set exitCode. On failure they
2328
+ // resolve with `{ text: "", wasAborted?/errorMessage }` — surface the
2329
+ // real cause (timeout / 401 / 429) instead of the generic
2330
+ // "empty output" branch below, which swallows the actual reason.
2331
+ if (result.wasAborted) {
2332
+ return `[agent step failed: ${driverName} timed out or was cancelled]`;
2333
+ }
2334
+ if (result.errorMessage) {
2335
+ return `[agent step failed: ${driverName} — ${result.errorMessage.slice(0, 200)}]`;
2336
+ }
1528
2337
  if (!result.text) {
1529
2338
  return `[agent step failed: ${driverName} returned empty output (possible timeout or auth error)]`;
1530
2339
  }
2340
+ // Forward token usage (when the driver reported it) so RunBudget can
2341
+ // enforce a real budget for openai/grok/gemini instead of failing
2342
+ // open. No usage → bare string, normalised to {text} downstream.
2343
+ const usage = providerMetaToUsage(result.providerMeta);
2344
+ // Carry the model the driver ACTUALLY resolved+billed (providerMeta.
2345
+ // model, e.g. openai's "gpt-4o" default when the step omitted model) so
2346
+ // RunBudget prices the real model. executeAgent's stamp() is idempotent
2347
+ // — it preserves this servedBy rather than re-deriving from raw input.
2348
+ const resolvedModel = typeof result.providerMeta?.model === "string"
2349
+ ? result.providerMeta.model
2350
+ : undefined;
2351
+ if (usage || resolvedModel) {
2352
+ return {
2353
+ text: result.text,
2354
+ ...(usage ? { usage } : {}),
2355
+ ...(resolvedModel
2356
+ ? { servedBy: { driver: driverName, model: resolvedModel } }
2357
+ : {}),
2358
+ };
2359
+ }
1531
2360
  return result.text;
1532
2361
  }
1533
2362
  finally {
@@ -1539,13 +2368,30 @@ function makeProviderDriverFn() {
1539
2368
  }
1540
2369
  };
1541
2370
  }
1542
- async function defaultClaudeFn(prompt, model) {
2371
+ /** Default Anthropic API request timeout. Mirrors the provider path (300s). */
2372
+ const DEFAULT_CLAUDE_API_TIMEOUT_MS = 300_000;
2373
+ /**
2374
+ * R4 #4 (HIGH): default max output tokens. The old hard-coded 1024 silently
2375
+ * truncated structured JSON (judge verdicts, multi-field agent outputs).
2376
+ */
2377
+ const DEFAULT_CLAUDE_MAX_TOKENS = 4096;
2378
+ export async function defaultClaudeFn(prompt, model, opts) {
1543
2379
  const apiKey = process.env.ANTHROPIC_API_KEY;
1544
2380
  if (!apiKey)
1545
2381
  return { text: "[agent step skipped: ANTHROPIC_API_KEY not set]" };
2382
+ const maxTokens = typeof opts?.maxTokens === "number" && opts.maxTokens > 0
2383
+ ? opts.maxTokens
2384
+ : DEFAULT_CLAUDE_MAX_TOKENS;
2385
+ // R4 #3 (HIGH): abort a stalled gateway instead of hanging the run forever.
2386
+ const timeoutMs = typeof opts?.timeoutMs === "number" && opts.timeoutMs > 0
2387
+ ? opts.timeoutMs
2388
+ : DEFAULT_CLAUDE_API_TIMEOUT_MS;
2389
+ const controller = new AbortController();
2390
+ const timeout = setTimeout(() => controller.abort(), timeoutMs);
1546
2391
  try {
1547
2392
  const res = await fetch("https://api.anthropic.com/v1/messages", {
1548
2393
  method: "POST",
2394
+ signal: controller.signal,
1549
2395
  headers: {
1550
2396
  "x-api-key": apiKey,
1551
2397
  "anthropic-version": "2023-06-01",
@@ -1553,7 +2399,7 @@ async function defaultClaudeFn(prompt, model) {
1553
2399
  },
1554
2400
  body: JSON.stringify({
1555
2401
  model,
1556
- max_tokens: 1024,
2402
+ max_tokens: maxTokens,
1557
2403
  messages: [
1558
2404
  {
1559
2405
  role: "user",
@@ -1571,25 +2417,57 @@ async function defaultClaudeFn(prompt, model) {
1571
2417
  // versions) and downstream (subscription/CLI driver returns
1572
2418
  // undefined here).
1573
2419
  const data = (await res.json());
1574
- const text = data.content?.[0]?.text ?? "[agent step failed: empty response]";
2420
+ let text = data.content?.[0]?.text ?? "[agent step failed: empty response]";
2421
+ // R4 #4: detect+warn when the response was cut off at the token cap so a
2422
+ // truncated (likely unparseable) JSON payload isn't silently trusted.
2423
+ if (data.stop_reason === "max_tokens") {
2424
+ text = `[warning: response truncated at max_tokens=${maxTokens}; raise max_tokens]\n${text}`;
2425
+ }
1575
2426
  const inputTokens = data.usage?.input_tokens;
1576
2427
  const outputTokens = data.usage?.output_tokens;
1577
- if (typeof inputTokens === "number" && typeof outputTokens === "number") {
2428
+ if (typeof inputTokens === "number" &&
2429
+ typeof outputTokens === "number" &&
2430
+ Number.isFinite(inputTokens) &&
2431
+ inputTokens >= 0 &&
2432
+ Number.isFinite(outputTokens) &&
2433
+ outputTokens >= 0) {
1578
2434
  return { text, usage: { inputTokens, outputTokens } };
1579
2435
  }
1580
2436
  return { text };
1581
2437
  }
1582
2438
  catch (err) {
2439
+ const aborted = controller.signal.aborted ||
2440
+ (err instanceof Error && err.name === "AbortError");
2441
+ if (aborted) {
2442
+ return {
2443
+ text: `[agent step failed: Anthropic API request timed out after ${timeoutMs}ms]`,
2444
+ };
2445
+ }
1583
2446
  return {
1584
2447
  text: `[agent step failed: ${err instanceof Error ? err.message : String(err)}]`,
1585
2448
  };
1586
2449
  }
2450
+ finally {
2451
+ clearTimeout(timeout);
2452
+ }
1587
2453
  }
1588
- async function defaultLocalFn(prompt, model) {
2454
+ export async function defaultLocalFn(prompt, model) {
1589
2455
  try {
1590
2456
  const { createLocalAdapter } = await import("../adapters/local.js");
1591
2457
  const { loadConfig: loadPatchworkConfig } = await import("../patchworkConfig.js");
1592
2458
  const cfg = loadPatchworkConfig();
2459
+ // Anti-SSRF: the local adapter streams the prompt to `cfg.localEndpoint`
2460
+ // (dashboard/config-controlled). A `driver: local` recipe must not be
2461
+ // able to POST the prompt to an arbitrary public host. Mirror the
2462
+ // LocalApiDriver gate (src/drivers/local/index.ts): reject any non
2463
+ // loopback/private endpoint unless LOCAL_ENDPOINT_ALLOW_REMOTE=1.
2464
+ if (cfg.localEndpoint &&
2465
+ process.env.LOCAL_ENDPOINT_ALLOW_REMOTE !== "1" &&
2466
+ !isLoopbackOrPrivateEndpoint(cfg.localEndpoint)) {
2467
+ return {
2468
+ text: "[agent step failed: localEndpoint is a public host; set LOCAL_ENDPOINT_ALLOW_REMOTE=1 to override]",
2469
+ };
2470
+ }
1593
2471
  const adapter = createLocalAdapter({
1594
2472
  endpoint: cfg.localEndpoint,
1595
2473
  defaultModel: cfg.localModel ?? model,
@@ -1665,17 +2543,38 @@ export function buildChainedDeps(runnerDeps, claudeCodeFnOverride) {
1665
2543
  const result = await executeStep(step, {}, stepDeps);
1666
2544
  return result ?? "";
1667
2545
  };
1668
- const executeAgent = async (prompt, model, driver, mcpAccess) => {
1669
- // chainedRunner's AgentExecutor contract still returns a plain string —
1670
- // PR2b's token-budget consumer will plug in here as well, but for now
1671
- // we discard `.usage`.
1672
- const result = await _executeAgent({
2546
+ const executeAgent = async (prompt, model, driver, opts) => {
2547
+ // Surface the FULL AgentResult (text + usage + servedBy) so the chained
2548
+ // runner can reconcile real spend against the run budget — alignment with
2549
+ // the flat path, which already reads `.usage`. (Previously this closure
2550
+ // discarded everything but `.text`, leaving the chained path's budget
2551
+ // unenforced — the S1 SECURITY finding.)
2552
+ //
2553
+ // P0-5 + parity fix: the prior 4th param was `mcpAccess?: boolean`, but the
2554
+ // AgentExecutor type was 3-arg and the chained call site passed only 3 args
2555
+ // → chained recipes silently dropped mcpAccess (and would have dropped the
2556
+ // new sandbox fields too). Threading an opts object closes both gaps.
2557
+ return _executeAgent({
1673
2558
  prompt,
1674
2559
  model,
1675
- driver,
1676
- ...(mcpAccess !== undefined && { mcpAccess }),
2560
+ driver: driver === "api" ? "anthropic" : driver,
2561
+ ...(opts?.mcpAccess !== undefined && { mcpAccess: opts.mcpAccess }),
2562
+ ...(opts?.sandbox !== undefined && { sandbox: opts.sandbox }),
2563
+ ...(opts?.allowedTools !== undefined && {
2564
+ allowedTools: opts.allowedTools,
2565
+ }),
2566
+ // Worker.autonomy: single chokepoint for the CHAINED path — fold the
2567
+ // worker's agent-step deny list into every chained agent call so the
2568
+ // subprocess can't bypass the per-step gate (mirrors the flat branch).
2569
+ ...(() => {
2570
+ const merged = mergeAgentDisallowedTools(opts?.disallowedTools, runnerDeps.agentDisallowedTools);
2571
+ return merged !== undefined ? { disallowedTools: merged } : {};
2572
+ })(),
2573
+ // Fail closed if a worker sandbox can't be enforced on the chosen driver.
2574
+ ...(runnerDeps.agentDisallowedTools?.length && {
2575
+ enforceSandbox: true,
2576
+ }),
1677
2577
  }, buildAgentExecutorDeps(stepDeps, runnerDeps, claudeCodeFnOverride));
1678
- return result.text;
1679
2578
  };
1680
2579
  // ---------------------------------------------------------------------
1681
2580
  // BEGIN A-PR2 EDIT BLOCK — `loadNestedRecipe` jail (dogfood F-04).
@@ -1768,7 +2667,17 @@ export function buildChainedDeps(runnerDeps, claudeCodeFnOverride) {
1768
2667
  }
1769
2668
  return null;
1770
2669
  };
1771
- return { executeTool, executeAgent, loadNestedRecipe };
2670
+ return {
2671
+ executeTool,
2672
+ executeAgent,
2673
+ loadNestedRecipe,
2674
+ // Tier-1 #4 (audit 2026-06-22): forward the approval gate into the chained
2675
+ // path so it is no longer flat-only. Undefined when the bridge didn't
2676
+ // inject one (approvalGate == "off") — the chained gate then no-ops.
2677
+ ...(runnerDeps.requireApprovalFn && {
2678
+ requireApprovalFn: runnerDeps.requireApprovalFn,
2679
+ }),
2680
+ };
1772
2681
  }
1773
2682
  /**
1774
2683
  * Dispatch a loaded recipe to the appropriate runner.
@@ -1787,10 +2696,21 @@ export async function dispatchRecipe(recipe, deps, seedContext = {}) {
1787
2696
  const chainedRecipe = recipe;
1788
2697
  const now = deps.now ? deps.now() : new Date();
1789
2698
  const options = {
2699
+ // Audit 2026-06-08 (recipe-support-3): only the recipe's declared env
2700
+ // keys reach the template context — NOT the full process.env. Parity with
2701
+ // the flat runner; prevents undeclared-secret exposure via {{env.X}}.
1790
2702
  env: {
1791
- ...process.env,
2703
+ ...declaredRecipeEnv(chainedRecipe),
1792
2704
  DATE: now.toISOString().slice(0, 10),
1793
2705
  TIME: now.toTimeString().slice(0, 5),
2706
+ // Built-in date/time tokens (parity with the flat runner ctx + lint).
2707
+ YYYY: now.toISOString().slice(0, 4),
2708
+ "YYYY-MM": now.toISOString().slice(0, 7),
2709
+ "YYYY-MM-DD": now.toISOString().slice(0, 10),
2710
+ ISO_NOW: now.toISOString(),
2711
+ HH: now.toISOString().slice(11, 13),
2712
+ MM: now.toISOString().slice(14, 16),
2713
+ SS: now.toISOString().slice(17, 19),
1794
2714
  ...seedContext,
1795
2715
  },
1796
2716
  maxConcurrency: Math.max(1, chainedRecipe.maxConcurrency ?? 4),
@@ -1804,6 +2724,20 @@ export async function dispatchRecipe(recipe, deps, seedContext = {}) {
1804
2724
  activityLog: deps.chainedOptions?.activityLog,
1805
2725
  mockedOutputs: deps.chainedOptions?.mockedOutputs,
1806
2726
  taskIdPrefix: deps.chainedOptions?.taskIdPrefix,
2727
+ // Parity (#850): forward the run-level budget, price table, and
2728
+ // cancellation signal that the chained runner honours. Without these the
2729
+ // chained path silently diverged from the flat path —
2730
+ // - `budget` lets a caller inject a shared RunBudget (and is the
2731
+ // hook the chained runner uses to enforce usdMax).
2732
+ // - `priceTable` reuses an already-loaded table instead of forcing the
2733
+ // chained RunBudget to re-load it from disk.
2734
+ // - `signal` wires AbortSignal-based cancellation into the run; a
2735
+ // pre-aborted signal now prevents dispatch on the
2736
+ // chained path too (parity target — flat cancellation
2737
+ // is still a separate gap).
2738
+ budget: deps.chainedOptions?.budget,
2739
+ priceTable: deps.chainedOptions?.priceTable,
2740
+ signal: deps.chainedOptions?.signal,
1807
2741
  };
1808
2742
  if (!deps.chainedDeps) {
1809
2743
  throw new Error("chainedDeps required for chained recipes (provide executeTool, executeAgent, loadNestedRecipe)");