infraweaver 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (524) hide show
  1. package/LICENSE +661 -0
  2. package/README.md +244 -0
  3. package/dist/agents/claude.d.ts +113 -0
  4. package/dist/agents/claudePretoolGate.d.ts +137 -0
  5. package/dist/agents/dispatchMetrics.d.ts +77 -0
  6. package/dist/agents/gateServer.d.ts +7 -0
  7. package/dist/agents/index.d.ts +6 -0
  8. package/dist/agents/nativeFsDenies.d.ts +46 -0
  9. package/dist/agents/opencode.d.ts +284 -0
  10. package/dist/agents/opencodePlugin.d.ts +96 -0
  11. package/dist/agents/opencodeShared.d.ts +40 -0
  12. package/dist/agents/postRun.d.ts +126 -0
  13. package/dist/agents/reviewer.d.ts +41 -0
  14. package/dist/agents/sessionLabeler.d.ts +97 -0
  15. package/dist/agents/shared.d.ts +246 -0
  16. package/dist/agents/subagentModels.d.ts +19 -0
  17. package/dist/agents/tokenQuota.d.ts +118 -0
  18. package/dist/agents/writeGateSource.d.ts +22 -0
  19. package/dist/agents/writePolicy.d.ts +79 -0
  20. package/dist/brand.d.ts +67 -0
  21. package/dist/cli.mjs +247846 -0
  22. package/dist/external.d.ts +227 -0
  23. package/dist/i18n/scaffolding.d.ts +59 -0
  24. package/dist/index.d.ts +6 -0
  25. package/dist/index.js +247061 -0
  26. package/dist/internal/index.d.ts +18 -0
  27. package/dist/internal.js +2398 -0
  28. package/dist/lifecycle.d.ts +2 -0
  29. package/dist/main.d.ts +8 -0
  30. package/dist/mcp/arkConfig.d.ts +1 -0
  31. package/dist/mcp/assess.d.ts +157 -0
  32. package/dist/mcp/capabilityContext.d.ts +71 -0
  33. package/dist/mcp/changeSummary.d.ts +52 -0
  34. package/dist/mcp/checkSuite.d.ts +27 -0
  35. package/dist/mcp/checkout.d.ts +92 -0
  36. package/dist/mcp/comment.d.ts +127 -0
  37. package/dist/mcp/commitInfo.d.ts +11 -0
  38. package/dist/mcp/crosswalk.d.ts +178 -0
  39. package/dist/mcp/crosswalkDigest.d.ts +1 -0
  40. package/dist/mcp/cyberEssentials.d.ts +24 -0
  41. package/dist/mcp/dashboard.d.ts +125 -0
  42. package/dist/mcp/dependencies.d.ts +12 -0
  43. package/dist/mcp/frameworks.d.ts +74 -0
  44. package/dist/mcp/geminiSanitizer.d.ts +28 -0
  45. package/dist/mcp/git.d.ts +60 -0
  46. package/dist/mcp/guardrails.d.ts +222 -0
  47. package/dist/mcp/issue.d.ts +20 -0
  48. package/dist/mcp/issueComments.d.ts +11 -0
  49. package/dist/mcp/issueEvents.d.ts +11 -0
  50. package/dist/mcp/issueInfo.d.ts +11 -0
  51. package/dist/mcp/labels.d.ts +14 -0
  52. package/dist/mcp/localContext.d.ts +26 -0
  53. package/dist/mcp/moduleExtraction.d.ts +79 -0
  54. package/dist/mcp/moduleTests.d.ts +106 -0
  55. package/dist/mcp/modules.d.ts +198 -0
  56. package/dist/mcp/output.d.ts +16 -0
  57. package/dist/mcp/pathSafety.d.ts +14 -0
  58. package/dist/mcp/policy.d.ts +50 -0
  59. package/dist/mcp/pr.d.ts +64 -0
  60. package/dist/mcp/prInfo.d.ts +11 -0
  61. package/dist/mcp/providerSchema.d.ts +52 -0
  62. package/dist/mcp/review.d.ts +212 -0
  63. package/dist/mcp/reviewComments.d.ts +245 -0
  64. package/dist/mcp/roots.d.ts +60 -0
  65. package/dist/mcp/scope.d.ts +26 -0
  66. package/dist/mcp/selectMode.d.ts +20 -0
  67. package/dist/mcp/server.d.ts +61 -0
  68. package/dist/mcp/shared.d.ts +62 -0
  69. package/dist/mcp/shell.d.ts +58 -0
  70. package/dist/mcp/staleFix.d.ts +138 -0
  71. package/dist/mcp/terraform/azurePrices.d.ts +8 -0
  72. package/dist/mcp/terraform/baseline.d.ts +57 -0
  73. package/dist/mcp/terraform/concernResult.d.ts +38 -0
  74. package/dist/mcp/terraform/cost.d.ts +55 -0
  75. package/dist/mcp/terraform/cspm.d.ts +80 -0
  76. package/dist/mcp/terraform/currency.d.ts +110 -0
  77. package/dist/mcp/terraform/decisions.d.ts +178 -0
  78. package/dist/mcp/terraform/eidas.d.ts +100 -0
  79. package/dist/mcp/terraform/evidence.d.ts +343 -0
  80. package/dist/mcp/terraform/findings.d.ts +76 -0
  81. package/dist/mcp/terraform/fixMemory.d.ts +81 -0
  82. package/dist/mcp/terraform/governance.d.ts +56 -0
  83. package/dist/mcp/terraform/hcl.d.ts +115 -0
  84. package/dist/mcp/terraform/iacLanguages.d.ts +120 -0
  85. package/dist/mcp/terraform/idleFloor.d.ts +78 -0
  86. package/dist/mcp/terraform/keyless.d.ts +68 -0
  87. package/dist/mcp/terraform/moduleDocs.d.ts +56 -0
  88. package/dist/mcp/terraform/nativeRules.d.ts +38 -0
  89. package/dist/mcp/terraform/nativeScan.d.ts +11 -0
  90. package/dist/mcp/terraform/noiseProfile.d.ts +100 -0
  91. package/dist/mcp/terraform/normalization.d.ts +52 -0
  92. package/dist/mcp/terraform/oscal.d.ts +103 -0
  93. package/dist/mcp/terraform/packs/azurerm-3-4.d.ts +2 -0
  94. package/dist/mcp/terraform/packs/cloudflare-4-5.d.ts +2 -0
  95. package/dist/mcp/terraform/paths.d.ts +28 -0
  96. package/dist/mcp/terraform/plan.d.ts +157 -0
  97. package/dist/mcp/terraform/planDrift.d.ts +199 -0
  98. package/dist/mcp/terraform/policyAuthor.d.ts +191 -0
  99. package/dist/mcp/terraform/prowlerOcsf.d.ts +136 -0
  100. package/dist/mcp/terraform/refactor.d.ts +183 -0
  101. package/dist/mcp/terraform/registry.d.ts +108 -0
  102. package/dist/mcp/terraform/risk.d.ts +41 -0
  103. package/dist/mcp/terraform/scanSession.d.ts +116 -0
  104. package/dist/mcp/terraform/scannerCache.d.ts +15 -0
  105. package/dist/mcp/terraform/scannerEgress.d.ts +72 -0
  106. package/dist/mcp/terraform/scannerJson.d.ts +13 -0
  107. package/dist/mcp/terraform/scanners.d.ts +205 -0
  108. package/dist/mcp/terraform/signing.d.ts +61 -0
  109. package/dist/mcp/terraform/stateLocking.d.ts +18 -0
  110. package/dist/mcp/terraform/subprocess.d.ts +43 -0
  111. package/dist/mcp/terraform/suppressions.d.ts +90 -0
  112. package/dist/mcp/terraform/taxonomy.d.ts +39 -0
  113. package/dist/mcp/terraform/tfquery.d.ts +47 -0
  114. package/dist/mcp/terraform/toolchainIntegrity.d.ts +82 -0
  115. package/dist/mcp/terraform/tools/authorPolicy.d.ts +59 -0
  116. package/dist/mcp/terraform/tools/detectPlanDrift.d.ts +20 -0
  117. package/dist/mcp/terraform/tools/emitCost.d.ts +11 -0
  118. package/dist/mcp/terraform/tools/emitSarif.d.ts +14 -0
  119. package/dist/mcp/terraform/tools/fixMemory.d.ts +19 -0
  120. package/dist/mcp/terraform/tools/infracostDiff.d.ts +22 -0
  121. package/dist/mcp/terraform/tools/ingestExternalFindings.d.ts +21 -0
  122. package/dist/mcp/terraform/tools/moduleLookup.d.ts +14 -0
  123. package/dist/mcp/terraform/tools/moduleSearch.d.ts +23 -0
  124. package/dist/mcp/terraform/tools/plan.d.ts +59 -0
  125. package/dist/mcp/terraform/tools/providerUpgrade.d.ts +55 -0
  126. package/dist/mcp/terraform/tools/readFindings.d.ts +17 -0
  127. package/dist/mcp/terraform/tools/scan.d.ts +114 -0
  128. package/dist/mcp/terraform/tools/validate.d.ts +31 -0
  129. package/dist/mcp/terraform/tools/verifyRemediation.d.ts +31 -0
  130. package/dist/mcp/terraform/tools/versionCurrency.d.ts +5 -0
  131. package/dist/mcp/terraform/tools.d.ts +16 -0
  132. package/dist/mcp/terraform/transformPacks.d.ts +267 -0
  133. package/dist/mcp/terraform/types.d.ts +226 -0
  134. package/dist/mcp/terraform/verification.d.ts +81 -0
  135. package/dist/mcp/terraform/vex.d.ts +115 -0
  136. package/dist/mcp/terraform/writeOnly.d.ts +67 -0
  137. package/dist/mcp/terraform.d.ts +51 -0
  138. package/dist/mcp/terratest.d.ts +85 -0
  139. package/dist/mcp/untrustedContent.d.ts +76 -0
  140. package/dist/mcp/upload.d.ts +8 -0
  141. package/dist/models.d.ts +181 -0
  142. package/dist/modes/address-reviews.d.ts +2 -0
  143. package/dist/modes/assess.d.ts +2 -0
  144. package/dist/modes/build.d.ts +2 -0
  145. package/dist/modes/compliance-audit.d.ts +2 -0
  146. package/dist/modes/cost-optimization.d.ts +2 -0
  147. package/dist/modes/drift-detect.d.ts +2 -0
  148. package/dist/modes/fix.d.ts +2 -0
  149. package/dist/modes/incremental-review.d.ts +2 -0
  150. package/dist/modes/index.d.ts +24 -0
  151. package/dist/modes/modernize-deprecated.d.ts +2 -0
  152. package/dist/modes/plan.d.ts +2 -0
  153. package/dist/modes/policy-gate.d.ts +2 -0
  154. package/dist/modes/prFormats.d.ts +3 -0
  155. package/dist/modes/refactor.d.ts +2 -0
  156. package/dist/modes/refresh-remediation.d.ts +2 -0
  157. package/dist/modes/remediate-and-refactor.d.ts +2 -0
  158. package/dist/modes/remediate.d.ts +2 -0
  159. package/dist/modes/resolve-conflicts.d.ts +2 -0
  160. package/dist/modes/review.d.ts +2 -0
  161. package/dist/modes/summarize-pr.d.ts +2 -0
  162. package/dist/modes/task.d.ts +2 -0
  163. package/dist/modes/terraform-code-review.d.ts +2 -0
  164. package/dist/modes/types.d.ts +9 -0
  165. package/dist/modes/update-dependencies.d.ts +2 -0
  166. package/dist/prep/index.d.ts +7 -0
  167. package/dist/prep/installNodeDependencies.d.ts +2 -0
  168. package/dist/prep/installPythonDependencies.d.ts +2 -0
  169. package/dist/prep/types.d.ts +31 -0
  170. package/dist/reviewQuality.d.ts +107 -0
  171. package/dist/skills/terraform-best-practices/SKILL.md +379 -0
  172. package/dist/toolState.d.ts +123 -0
  173. package/dist/utils/activity.d.ts +40 -0
  174. package/dist/utils/agent.d.ts +47 -0
  175. package/dist/utils/agentHangReport.d.ts +38 -0
  176. package/dist/utils/annotations.d.ts +17 -0
  177. package/dist/utils/apiFetch.d.ts +11 -0
  178. package/dist/utils/apiKeys.d.ts +48 -0
  179. package/dist/utils/apiUrl.d.ts +28 -0
  180. package/dist/utils/assets.d.ts +8 -0
  181. package/dist/utils/baseRefConfig.d.ts +121 -0
  182. package/dist/utils/billingErrors.d.ts +85 -0
  183. package/dist/utils/body.d.ts +34 -0
  184. package/dist/utils/buildInfraweaverFooter.d.ts +29 -0
  185. package/dist/utils/byokFallback.d.ts +85 -0
  186. package/dist/utils/changeImpact.d.ts +101 -0
  187. package/dist/utils/claudeSubscription.d.ts +30 -0
  188. package/dist/utils/cli.d.ts +10 -0
  189. package/dist/utils/cloudClient.d.ts +33 -0
  190. package/dist/utils/cloudFetchCore.d.ts +42 -0
  191. package/dist/utils/cloudReport.d.ts +114 -0
  192. package/dist/utils/codexHome.d.ts +29 -0
  193. package/dist/utils/codexOAuth.d.ts +60 -0
  194. package/dist/utils/diffCoverage.d.ts +63 -0
  195. package/dist/utils/env.d.ts +46 -0
  196. package/dist/utils/errorReport.d.ts +17 -0
  197. package/dist/utils/exitHandler.d.ts +8 -0
  198. package/dist/utils/fixDoubleEscapedString.d.ts +1 -0
  199. package/dist/utils/gitAuth.d.ts +84 -0
  200. package/dist/utils/gitAuthServer.d.ts +24 -0
  201. package/dist/utils/github.d.ts +104 -0
  202. package/dist/utils/globals.d.ts +3 -0
  203. package/dist/utils/infraweaverConfig.d.ts +56 -0
  204. package/dist/utils/install.d.ts +37 -0
  205. package/dist/utils/instructions.d.ts +48 -0
  206. package/dist/utils/keylessOidc.d.ts +28 -0
  207. package/dist/utils/leapingComment.d.ts +11 -0
  208. package/dist/utils/learnings.d.ts +62 -0
  209. package/dist/utils/learningsTruncate.d.ts +25 -0
  210. package/dist/utils/lifecycle.d.ts +57 -0
  211. package/dist/utils/log.d.ts +111 -0
  212. package/dist/utils/modelResolver.d.ts +63 -0
  213. package/dist/utils/moduleFetch.d.ts +39 -0
  214. package/dist/utils/normalizeEnv.d.ts +30 -0
  215. package/dist/utils/openCodeModels.d.ts +11 -0
  216. package/dist/utils/openaiCompatible.d.ts +26 -0
  217. package/dist/utils/opentofuMcp.d.ts +50 -0
  218. package/dist/utils/overrides.d.ts +40 -0
  219. package/dist/utils/packageManager.d.ts +49 -0
  220. package/dist/utils/patchWorkflowRunFields.d.ts +29 -0
  221. package/dist/utils/payload.d.ts +249 -0
  222. package/dist/utils/prSummary.d.ts +61 -0
  223. package/dist/utils/presets.d.ts +36 -0
  224. package/dist/utils/progressComment.d.ts +146 -0
  225. package/dist/utils/providerErrors.d.ts +31 -0
  226. package/dist/utils/proxyModel.d.ts +76 -0
  227. package/dist/utils/rangeDiff.d.ts +51 -0
  228. package/dist/utils/redact.d.ts +33 -0
  229. package/dist/utils/registryAuth.d.ts +52 -0
  230. package/dist/utils/remediationCommand.d.ts +58 -0
  231. package/dist/utils/retry.d.ts +38 -0
  232. package/dist/utils/reviewCleanup.d.ts +14 -0
  233. package/dist/utils/run.d.ts +9 -0
  234. package/dist/utils/runContext.d.ts +112 -0
  235. package/dist/utils/runContextData.d.ts +42 -0
  236. package/dist/utils/runErrorRenderer.d.ts +64 -0
  237. package/dist/utils/runLifecycle.d.ts +89 -0
  238. package/dist/utils/runStartupLog.d.ts +15 -0
  239. package/dist/utils/secrets.d.ts +34 -0
  240. package/dist/utils/setup.d.ts +90 -0
  241. package/dist/utils/shell.d.ts +44 -0
  242. package/dist/utils/skills.d.ts +10 -0
  243. package/dist/utils/subprocess.d.ts +81 -0
  244. package/dist/utils/terraformMcp.d.ts +46 -0
  245. package/dist/utils/time.d.ts +15 -0
  246. package/dist/utils/timer.d.ts +23 -0
  247. package/dist/utils/todoTracking.d.ts +16 -0
  248. package/dist/utils/token.d.ts +52 -0
  249. package/dist/utils/toolLicensing.d.ts +56 -0
  250. package/dist/utils/toolSelection.d.ts +73 -0
  251. package/dist/utils/toon.d.ts +16 -0
  252. package/dist/utils/version.d.ts +2 -0
  253. package/dist/utils/versioning.d.ts +7 -0
  254. package/dist/utils/vertex.d.ts +16 -0
  255. package/dist/utils/workflow.d.ts +13 -0
  256. package/package.json +132 -0
  257. package/src/agents/claude.ts +1588 -0
  258. package/src/agents/claudePretoolGate.ts +286 -0
  259. package/src/agents/dispatchMetrics.ts +162 -0
  260. package/src/agents/gateServer.ts +178 -0
  261. package/src/agents/index.ts +10 -0
  262. package/src/agents/nativeFsDenies.ts +164 -0
  263. package/src/agents/opencode.ts +1600 -0
  264. package/src/agents/opencodePlugin.ts +304 -0
  265. package/src/agents/opencodeShared.ts +134 -0
  266. package/src/agents/postRun.ts +615 -0
  267. package/src/agents/reviewer.ts +134 -0
  268. package/src/agents/sessionLabeler.ts +185 -0
  269. package/src/agents/shared.ts +377 -0
  270. package/src/agents/subagentModels.ts +40 -0
  271. package/src/agents/tokenQuota.ts +203 -0
  272. package/src/agents/writeGateSource.ts +148 -0
  273. package/src/agents/writePolicy.ts +132 -0
  274. package/src/brand.ts +84 -0
  275. package/src/cli.ts +117 -0
  276. package/src/commands/gha.ts +188 -0
  277. package/src/commands/mcp.ts +126 -0
  278. package/src/commands/verifyEvidence.ts +100 -0
  279. package/src/entry.ts +7 -0
  280. package/src/entryPost.ts +109 -0
  281. package/src/external.ts +308 -0
  282. package/src/i18n/scaffolding.ts +191 -0
  283. package/src/index.ts +7 -0
  284. package/src/internal/index.ts +74 -0
  285. package/src/lifecycle.ts +2 -0
  286. package/src/main.ts +815 -0
  287. package/src/mcp/__fixtures__/infraweaver-scratch-pr-49-review-3485940013.json +110 -0
  288. package/src/mcp/__fixtures__/infraweaver-scratch-pr-64-review-3531000326.json +14 -0
  289. package/src/mcp/__fixtures__/infraweaver-test-repo-pr-1.diff.json +67 -0
  290. package/src/mcp/__snapshots__/checkout.test.ts.snap +109 -0
  291. package/src/mcp/__snapshots__/reviewComments.test.ts.snap +71 -0
  292. package/src/mcp/arkConfig.ts +7 -0
  293. package/src/mcp/assess.ts +723 -0
  294. package/src/mcp/capabilityContext.ts +80 -0
  295. package/src/mcp/changeSummary.ts +145 -0
  296. package/src/mcp/checkSuite.ts +261 -0
  297. package/src/mcp/checkout.ts +1015 -0
  298. package/src/mcp/comment.ts +683 -0
  299. package/src/mcp/commitInfo.ts +57 -0
  300. package/src/mcp/crosswalk.ts +1253 -0
  301. package/src/mcp/crosswalkDigest.ts +5 -0
  302. package/src/mcp/cyberEssentials.ts +60 -0
  303. package/src/mcp/dashboard.ts +232 -0
  304. package/src/mcp/dependencies.ts +190 -0
  305. package/src/mcp/frameworks.ts +236 -0
  306. package/src/mcp/geminiSanitizer.ts +212 -0
  307. package/src/mcp/git.ts +1116 -0
  308. package/src/mcp/guardrails.ts +771 -0
  309. package/src/mcp/issue.ts +74 -0
  310. package/src/mcp/issueComments.ts +48 -0
  311. package/src/mcp/issueEvents.ts +100 -0
  312. package/src/mcp/issueInfo.ts +72 -0
  313. package/src/mcp/labels.ts +35 -0
  314. package/src/mcp/localContext.ts +65 -0
  315. package/src/mcp/localServer.ts +217 -0
  316. package/src/mcp/moduleExtraction.ts +368 -0
  317. package/src/mcp/moduleTests.ts +421 -0
  318. package/src/mcp/modules.ts +752 -0
  319. package/src/mcp/output.ts +71 -0
  320. package/src/mcp/pathSafety.ts +28 -0
  321. package/src/mcp/policy.ts +226 -0
  322. package/src/mcp/pr.ts +238 -0
  323. package/src/mcp/prInfo.ts +91 -0
  324. package/src/mcp/providerSchema.ts +175 -0
  325. package/src/mcp/review.ts +1081 -0
  326. package/src/mcp/reviewComments.ts +1137 -0
  327. package/src/mcp/roots.ts +217 -0
  328. package/src/mcp/scope.ts +82 -0
  329. package/src/mcp/selectMode.ts +207 -0
  330. package/src/mcp/server.ts +498 -0
  331. package/src/mcp/shared.ts +120 -0
  332. package/src/mcp/shell.ts +631 -0
  333. package/src/mcp/staleFix.ts +557 -0
  334. package/src/mcp/terraform/__snapshots__/moduleDocs.test.ts.snap +40 -0
  335. package/src/mcp/terraform/azurePrices.ts +38 -0
  336. package/src/mcp/terraform/baseline.ts +172 -0
  337. package/src/mcp/terraform/concernResult.ts +111 -0
  338. package/src/mcp/terraform/cost.ts +175 -0
  339. package/src/mcp/terraform/cspm.ts +326 -0
  340. package/src/mcp/terraform/currency.ts +354 -0
  341. package/src/mcp/terraform/decisions.ts +590 -0
  342. package/src/mcp/terraform/eidas.ts +134 -0
  343. package/src/mcp/terraform/evidence.ts +771 -0
  344. package/src/mcp/terraform/findings.ts +383 -0
  345. package/src/mcp/terraform/fixMemory.ts +253 -0
  346. package/src/mcp/terraform/governance.ts +231 -0
  347. package/src/mcp/terraform/hcl.ts +273 -0
  348. package/src/mcp/terraform/iacLanguages.ts +431 -0
  349. package/src/mcp/terraform/idleFloor.ts +304 -0
  350. package/src/mcp/terraform/keyless.ts +123 -0
  351. package/src/mcp/terraform/moduleDocs.ts +303 -0
  352. package/src/mcp/terraform/nativeRules.ts +152 -0
  353. package/src/mcp/terraform/nativeScan.ts +67 -0
  354. package/src/mcp/terraform/noiseProfile.ts +183 -0
  355. package/src/mcp/terraform/normalization.ts +265 -0
  356. package/src/mcp/terraform/oscal.ts +283 -0
  357. package/src/mcp/terraform/packs/azurerm-3-4.ts +382 -0
  358. package/src/mcp/terraform/packs/cloudflare-4-5.ts +114 -0
  359. package/src/mcp/terraform/paths.ts +59 -0
  360. package/src/mcp/terraform/plan.ts +377 -0
  361. package/src/mcp/terraform/planDrift.ts +520 -0
  362. package/src/mcp/terraform/policyAuthor.ts +886 -0
  363. package/src/mcp/terraform/prowlerOcsf.ts +380 -0
  364. package/src/mcp/terraform/refactor.ts +879 -0
  365. package/src/mcp/terraform/registry.ts +317 -0
  366. package/src/mcp/terraform/risk.ts +99 -0
  367. package/src/mcp/terraform/scanSession.ts +242 -0
  368. package/src/mcp/terraform/scannerCache.ts +86 -0
  369. package/src/mcp/terraform/scannerEgress.ts +141 -0
  370. package/src/mcp/terraform/scannerJson.ts +22 -0
  371. package/src/mcp/terraform/scanners.ts +1134 -0
  372. package/src/mcp/terraform/signing.ts +151 -0
  373. package/src/mcp/terraform/stateLocking.ts +106 -0
  374. package/src/mcp/terraform/subprocess.ts +102 -0
  375. package/src/mcp/terraform/suppressions.ts +551 -0
  376. package/src/mcp/terraform/taxonomy.ts +0 -0
  377. package/src/mcp/terraform/tfquery.ts +159 -0
  378. package/src/mcp/terraform/toolchainIntegrity.ts +151 -0
  379. package/src/mcp/terraform/tools/authorPolicy.ts +185 -0
  380. package/src/mcp/terraform/tools/detectPlanDrift.ts +279 -0
  381. package/src/mcp/terraform/tools/emitCost.ts +130 -0
  382. package/src/mcp/terraform/tools/emitSarif.ts +108 -0
  383. package/src/mcp/terraform/tools/fixMemory.ts +82 -0
  384. package/src/mcp/terraform/tools/infracostDiff.ts +110 -0
  385. package/src/mcp/terraform/tools/ingestExternalFindings.ts +259 -0
  386. package/src/mcp/terraform/tools/moduleLookup.ts +86 -0
  387. package/src/mcp/terraform/tools/moduleSearch.ts +46 -0
  388. package/src/mcp/terraform/tools/plan.ts +384 -0
  389. package/src/mcp/terraform/tools/providerUpgrade.ts +217 -0
  390. package/src/mcp/terraform/tools/readFindings.ts +138 -0
  391. package/src/mcp/terraform/tools/scan.ts +303 -0
  392. package/src/mcp/terraform/tools/validate.ts +160 -0
  393. package/src/mcp/terraform/tools/verifyRemediation.ts +190 -0
  394. package/src/mcp/terraform/tools/versionCurrency.ts +64 -0
  395. package/src/mcp/terraform/tools.ts +62 -0
  396. package/src/mcp/terraform/transformPacks.ts +594 -0
  397. package/src/mcp/terraform/types.ts +382 -0
  398. package/src/mcp/terraform/verification.ts +150 -0
  399. package/src/mcp/terraform/vex.ts +367 -0
  400. package/src/mcp/terraform/writeOnly.ts +252 -0
  401. package/src/mcp/terraform.ts +52 -0
  402. package/src/mcp/terratest.ts +215 -0
  403. package/src/mcp/untrustedContent.ts +126 -0
  404. package/src/mcp/upload.ts +123 -0
  405. package/src/models.ts +753 -0
  406. package/src/modes/address-reviews.ts +35 -0
  407. package/src/modes/assess.ts +27 -0
  408. package/src/modes/build.ts +86 -0
  409. package/src/modes/compliance-audit.ts +25 -0
  410. package/src/modes/cost-optimization.ts +36 -0
  411. package/src/modes/drift-detect.ts +31 -0
  412. package/src/modes/fix.ts +29 -0
  413. package/src/modes/incremental-review.ts +103 -0
  414. package/src/modes/index.ts +94 -0
  415. package/src/modes/modernize-deprecated.ts +36 -0
  416. package/src/modes/plan.ts +21 -0
  417. package/src/modes/policy-gate.ts +27 -0
  418. package/src/modes/prFormats.ts +325 -0
  419. package/src/modes/refactor.ts +74 -0
  420. package/src/modes/refresh-remediation.ts +40 -0
  421. package/src/modes/remediate-and-refactor.ts +42 -0
  422. package/src/modes/remediate.ts +78 -0
  423. package/src/modes/resolve-conflicts.ts +32 -0
  424. package/src/modes/review.ts +135 -0
  425. package/src/modes/summarize-pr.ts +42 -0
  426. package/src/modes/task.ts +26 -0
  427. package/src/modes/terraform-code-review.ts +68 -0
  428. package/src/modes/types.ts +17 -0
  429. package/src/modes/update-dependencies.ts +44 -0
  430. package/src/prep/index.ts +96 -0
  431. package/src/prep/installNodeDependencies.ts +251 -0
  432. package/src/prep/installPythonDependencies.ts +235 -0
  433. package/src/prep/types.ts +38 -0
  434. package/src/reviewQuality.ts +237 -0
  435. package/src/runCli.ts +335 -0
  436. package/src/skills/terraform-best-practices/SKILL.md +379 -0
  437. package/src/toolState.ts +241 -0
  438. package/src/utils/activity.ts +210 -0
  439. package/src/utils/agent.ts +245 -0
  440. package/src/utils/agentHangReport.ts +180 -0
  441. package/src/utils/annotations.ts +56 -0
  442. package/src/utils/apiFetch.ts +19 -0
  443. package/src/utils/apiKeys.ts +305 -0
  444. package/src/utils/apiUrl.ts +42 -0
  445. package/src/utils/assets.ts +123 -0
  446. package/src/utils/baseRefConfig.ts +254 -0
  447. package/src/utils/billingErrors.ts +210 -0
  448. package/src/utils/body.ts +168 -0
  449. package/src/utils/buildInfraweaverFooter.ts +104 -0
  450. package/src/utils/byokFallback.ts +137 -0
  451. package/src/utils/changeImpact.ts +330 -0
  452. package/src/utils/claudeSubscription.ts +93 -0
  453. package/src/utils/cli.ts +36 -0
  454. package/src/utils/cloudClient.ts +57 -0
  455. package/src/utils/cloudFetchCore.ts +139 -0
  456. package/src/utils/cloudReport.ts +316 -0
  457. package/src/utils/codexHome.ts +204 -0
  458. package/src/utils/codexOAuth.ts +154 -0
  459. package/src/utils/codexRefreshDetect.ts +36 -0
  460. package/src/utils/diffCoverage.ts +404 -0
  461. package/src/utils/env.ts +84 -0
  462. package/src/utils/errorReport.ts +94 -0
  463. package/src/utils/exitHandler.ts +35 -0
  464. package/src/utils/fixDoubleEscapedString.ts +9 -0
  465. package/src/utils/ghaCore.ts +13 -0
  466. package/src/utils/gitAuth.ts +268 -0
  467. package/src/utils/gitAuthServer.ts +186 -0
  468. package/src/utils/github.ts +676 -0
  469. package/src/utils/globals.ts +9 -0
  470. package/src/utils/humanEditCapture.ts +198 -0
  471. package/src/utils/infraweaverConfig.ts +202 -0
  472. package/src/utils/install.ts +279 -0
  473. package/src/utils/instructions.ts +640 -0
  474. package/src/utils/keylessOidc.ts +47 -0
  475. package/src/utils/leapingComment.ts +20 -0
  476. package/src/utils/learnings.ts +145 -0
  477. package/src/utils/learningsTruncate.ts +42 -0
  478. package/src/utils/lifecycle.ts +198 -0
  479. package/src/utils/log.ts +452 -0
  480. package/src/utils/modelResolver.ts +105 -0
  481. package/src/utils/moduleFetch.ts +138 -0
  482. package/src/utils/normalizeEnv.ts +106 -0
  483. package/src/utils/openCodeModels.ts +86 -0
  484. package/src/utils/openaiCompatible.ts +52 -0
  485. package/src/utils/opentofuMcp.ts +61 -0
  486. package/src/utils/overrides.ts +100 -0
  487. package/src/utils/packageManager.ts +257 -0
  488. package/src/utils/patchWorkflowRunFields.ts +173 -0
  489. package/src/utils/payload.ts +1206 -0
  490. package/src/utils/prSummary.ts +148 -0
  491. package/src/utils/presets.ts +105 -0
  492. package/src/utils/progressComment.ts +266 -0
  493. package/src/utils/providerErrors.ts +189 -0
  494. package/src/utils/proxyModel.ts +239 -0
  495. package/src/utils/rangeDiff.ts +182 -0
  496. package/src/utils/redact.ts +86 -0
  497. package/src/utils/registryAuth.ts +115 -0
  498. package/src/utils/remediationCommand.ts +144 -0
  499. package/src/utils/retry.ts +180 -0
  500. package/src/utils/reviewCleanup.ts +116 -0
  501. package/src/utils/run.ts +100 -0
  502. package/src/utils/runContext.ts +276 -0
  503. package/src/utils/runContextData.ts +121 -0
  504. package/src/utils/runErrorRenderer.ts +282 -0
  505. package/src/utils/runFixture.ts +91 -0
  506. package/src/utils/runLifecycle.ts +360 -0
  507. package/src/utils/runStartupLog.ts +67 -0
  508. package/src/utils/secrets.ts +208 -0
  509. package/src/utils/setup.ts +368 -0
  510. package/src/utils/shell.ts +132 -0
  511. package/src/utils/skills.ts +67 -0
  512. package/src/utils/subprocess.ts +474 -0
  513. package/src/utils/terraformMcp.ts +99 -0
  514. package/src/utils/time.ts +59 -0
  515. package/src/utils/timer.ts +72 -0
  516. package/src/utils/todoTracking.ts +168 -0
  517. package/src/utils/token.ts +294 -0
  518. package/src/utils/toolLicensing.ts +152 -0
  519. package/src/utils/toolSelection.ts +239 -0
  520. package/src/utils/toon.ts +76 -0
  521. package/src/utils/version.ts +10 -0
  522. package/src/utils/versioning.ts +44 -0
  523. package/src/utils/vertex.ts +94 -0
  524. package/src/utils/workflow.ts +25 -0
@@ -0,0 +1,1588 @@
1
+ /**
2
+ * Claude Code agent — secure harness around the `claude` CLI.
3
+ *
4
+ * mirrors the opencode harness's security model:
5
+ * - native exec tools (Bash, Monitor, REPL, Workflow) blocked via BOTH
6
+ * --disallowedTools AND managed-settings.json `permissions.deny` (the agent
7
+ * cannot shell out / run code outside the MCP shell). the managed-settings
8
+ * deny is the authoritative, bypass-immune layer: `--disallowedTools` alone
9
+ * (a `cliArg`-source deny) was observed to leak under
10
+ * `--dangerously-skip-permissions`, surfacing a secret env marker via the
11
+ * native Bash tool. managed-settings denies are `policySettings`-source,
12
+ * highest precedence, and survive bypassPermissions mode.
13
+ * - managed-settings.json: filesystem sandbox — deny /proc, /sys reads
14
+ * - MCP ShellTool provides restricted shell (filtered env, no secrets)
15
+ * - MCP server injected via --mcp-config (not replacing project config)
16
+ * - ASKPASS handles git auth separately (token never in subprocess env)
17
+ *
18
+ * the agent process itself gets full env (needs LLM API keys, PATH, etc.).
19
+ * security is enforced at the tool layer, not the process layer.
20
+ */
21
+ import { execFileSync } from "node:child_process";
22
+ import { chmodSync, mkdirSync, writeFileSync } from "node:fs";
23
+ import { join } from "node:path";
24
+ import { performance } from "node:perf_hooks";
25
+ import { setTimeout as sleep } from "node:timers/promises";
26
+ import {
27
+ buildClaudePretoolGateSettings,
28
+ buildClaudePretoolGateSource,
29
+ CLAUDE_PRETOOL_GATE_FILENAME,
30
+ } from "#app/agents/claudePretoolGate";
31
+ import { startGateServer } from "#app/agents/gateServer";
32
+ import {
33
+ GIT_NATIVE_READ_DENY_CLAUDE,
34
+ GIT_NATIVE_WRITE_DENY_CLAUDE,
35
+ harnessTmpdirWriteDenyClaude,
36
+ runnerHomeWriteDenyClaude,
37
+ } from "#app/agents/nativeFsDenies";
38
+ import { finalizeAgentResult } from "#app/agents/postRun";
39
+ import {
40
+ REVIEWER_AGENT_NAME,
41
+ REVIEWER_SYSTEM_PROMPT,
42
+ } from "#app/agents/reviewer";
43
+ import { writePolicyPath } from "#app/agents/writePolicy";
44
+ import type { DispatchRecorder } from "#app/agents/dispatchMetrics";
45
+ import {
46
+ formatWithLabel,
47
+ ORCHESTRATOR_LABEL,
48
+ SessionLabeler,
49
+ } from "#app/agents/sessionLabeler";
50
+ import {
51
+ type AgentResult,
52
+ type AgentRunContext,
53
+ type AgentUsage,
54
+ agent,
55
+ agentHomeEnv,
56
+ baseAgentEnv,
57
+ buildAgentUsage,
58
+ logTokenTable,
59
+ mergeAgentUsage,
60
+ MAX_STDERR_LINES,
61
+ MCP_SERVER_TOKEN_ENV,
62
+ } from "#app/agents/shared";
63
+ import {
64
+ type RunTokenQuota,
65
+ TokenQuotaExceededError,
66
+ } from "#app/agents/tokenQuota";
67
+ import { PRODUCT_NAME } from "#app/brand";
68
+ import { infraweaverMcpName } from "#app/external";
69
+ import {
70
+ BEDROCK_MODEL_ID_ENV,
71
+ isBedrockAnthropicId,
72
+ isVertexAnthropicId,
73
+ VERTEX_MODEL_ID_ENV,
74
+ } from "#app/models";
75
+ import {
76
+ AGENT_ACTIVITY_TIMEOUT_MS,
77
+ getIdleMs,
78
+ markActivity,
79
+ } from "#app/utils/activity";
80
+ import { preflightClaudeSubscription } from "#app/utils/claudeSubscription";
81
+ import { formatJsonValue, log } from "#app/utils/cli";
82
+ import { installFromNpmTarball } from "#app/utils/install";
83
+ import { findProviderErrorMatch } from "#app/utils/providerErrors";
84
+ import { installBundledSkills } from "#app/utils/skills";
85
+ import {
86
+ DEFAULT_MAX_RETAINED_BYTES,
87
+ SPAWN_ACTIVITY_TIMEOUT_CODE,
88
+ SpawnTimeoutError,
89
+ spawn,
90
+ TailBuffer,
91
+ } from "#app/utils/subprocess";
92
+ import {
93
+ OPENTOFU_MCP_SERVER_NAME,
94
+ resolveOpenTofuMcp,
95
+ } from "#app/utils/opentofuMcp";
96
+ import {
97
+ resolveTerraformMcp,
98
+ TERRAFORM_MCP_SERVER_NAME,
99
+ } from "#app/utils/terraformMcp";
100
+ import { ThinkingTimer } from "#app/utils/timer";
101
+ import type { TodoTracker } from "#app/utils/todoTracking";
102
+ import { getDevDependencyVersion } from "#app/utils/version";
103
+ import { applyClaudeVertexEnv } from "#app/utils/vertex";
104
+
105
+ async function installClaudeCli(): Promise<string> {
106
+ return await installFromNpmTarball({
107
+ packageName: "@anthropic-ai/claude-code",
108
+ version: getDevDependencyVersion("@anthropic-ai/claude-code"),
109
+ // 2.1.113+ ships a native binary (bin/claude.exe) instead of cli.js; the
110
+ // package postinstall copies it from the platform optionalDependency, so we
111
+ // need installDependencies to run that postinstall.
112
+ executablePath: "bin/claude.exe",
113
+ installDependencies: true,
114
+ });
115
+ }
116
+
117
+ /**
118
+ * Native claude-code tools that execute arbitrary shell/code and therefore
119
+ * bypass Infraweaver's security boundary (the restricted MCP `shell` tool with a
120
+ * filtered, secret-free env). These run inside the agent process with full env,
121
+ * so leaving any of them enabled defeats both `shell: "disabled"` AND the
122
+ * env-filtering that the MCP shell relies on even when shell is enabled.
123
+ *
124
+ * As of claude-code 2.1.150 the exec surface is no longer just `Bash`:
125
+ * - `Monitor` runs a shell command/script (the `command` field)
126
+ * - `REPL` runs arbitrary JavaScript (can `require("node:child_process")`)
127
+ * - `Workflow` orchestrates subagents/pipelines that can reach the above
128
+ * Each is denied at top level and inside `Agent(...)` (Task subagents), mirroring
129
+ * the existing `Bash` / `Agent(Bash)` pair. Denying a tool that isn't registered
130
+ * in a given run is a harmless no-op, so this list is also forward-safe.
131
+ *
132
+ * `CLAUDE_EXEC_TOOL_DENY_RULES` is wired into TWO surfaces: `--disallowedTools`
133
+ * (removes the tools from the advertised list) and managed-settings.json
134
+ * `permissions.deny` (the authoritative, bypass-immune deny — see
135
+ * buildManagedSettings). The flag alone proved insufficient: under
136
+ * `--dangerously-skip-permissions` the native Bash tool ran despite
137
+ * `--disallowedTools Bash`, leaking a per-run secret marker.
138
+ */
139
+ const CLAUDE_EXEC_TOOLS = ["Bash", "Monitor", "REPL", "Workflow"] as const;
140
+ export const CLAUDE_EXEC_TOOL_DENY_RULES = [
141
+ ...CLAUDE_EXEC_TOOLS,
142
+ ...CLAUDE_EXEC_TOOLS.map((t) => `Agent(${t})`),
143
+ ];
144
+ const CLAUDE_DISALLOWED_TOOLS = CLAUDE_EXEC_TOOL_DENY_RULES.join(",");
145
+
146
+ // ── config ─────────────────────────────────────────────────────────────────────
147
+
148
+ // Claude Code expands `${VAR}` in .mcp.json values, including HTTP-server
149
+ // `headers` (code.claude.com/docs/en/mcp-configuration "Environment variable
150
+ // expansion"), so the on-disk mcp.json carries only this placeholder — never the
151
+ // raw token, which travels via MCP_SERVER_TOKEN_ENV on the agent's spawn env.
152
+ const INFRAWEAVER_MCP_AUTH_HEADER = `Bearer \${${MCP_SERVER_TOKEN_ENV}}`;
153
+
154
+ export function writeMcpConfig(ctx: AgentRunContext): string {
155
+ const configDir = join(ctx.tmpdir, ".claude");
156
+ mkdirSync(configDir, { recursive: true });
157
+ const configPath = join(configDir, "mcp.json");
158
+ // opt-in second server: HashiCorp's terraform-mcp-server (registry
159
+ // toolset, docker stdio) for live module/provider knowledge.
160
+ const terraformMcp = resolveTerraformMcp(ctx.payload);
161
+ if (terraformMcp.kind === "docker_missing")
162
+ log.info(`» ${terraformMcp.note}`);
163
+ // opt-in sibling: the OpenTofu registry MCP server (hosted HTTP, no docker) —
164
+ // OpenTofu Registry knowledge for an OpenTofu-first repo.
165
+ const opentofuMcp = resolveOpenTofuMcp(ctx.payload);
166
+ writeFileSync(
167
+ configPath,
168
+ JSON.stringify({
169
+ mcpServers: {
170
+ [infraweaverMcpName]: {
171
+ type: "http",
172
+ url: ctx.mcpServerUrl,
173
+ headers: { Authorization: INFRAWEAVER_MCP_AUTH_HEADER },
174
+ },
175
+ ...(terraformMcp.kind === "available"
176
+ ? {
177
+ [TERRAFORM_MCP_SERVER_NAME]: {
178
+ type: "stdio",
179
+ command: terraformMcp.command,
180
+ args: terraformMcp.args,
181
+ },
182
+ }
183
+ : {}),
184
+ ...(opentofuMcp.kind === "available"
185
+ ? {
186
+ [OPENTOFU_MCP_SERVER_NAME]: {
187
+ type: "http",
188
+ url: opentofuMcp.url,
189
+ },
190
+ }
191
+ : {}),
192
+ },
193
+ }),
194
+ );
195
+ return configPath;
196
+ }
197
+
198
+ /**
199
+ * Drop the PreToolUse gate script + its `--settings` JSON into the per-run
200
+ * tmpdir and return the absolute path to the settings file. The script
201
+ * blocks state-mutating MCP tool calls when `agent_id` is non-empty (i.e.,
202
+ * the call originates inside a Task/Agent subagent dispatch). See
203
+ * action/agents/claudePretoolGate.ts for the contract.
204
+ *
205
+ * Two paths register the gate:
206
+ * 1. flag settings (`--settings <path>`) — covers non-CI runs (`pnpm dev:run`,
207
+ * local dev) where `installManagedSettings` is a no-op.
208
+ * 2. managed settings (/etc/claude-code/managed-settings.json) — covers CI,
209
+ * where `allowManagedHooksOnly: true` filters flag-settings hooks. The
210
+ * same hook entry is embedded in `buildManagedSettings` below.
211
+ *
212
+ * The flag settings also carry the native exec-tool `permissions.deny`
213
+ * (via `buildClaudePretoolGateSettings`) so non-CI runs (where managed
214
+ * settings are absent) still block native Bash et al. at a settings-source
215
+ * deny, not just the `--disallowedTools` cliArg deny that proved leaky under
216
+ * `--dangerously-skip-permissions`.
217
+ */
218
+ function writePretoolGateAssets(
219
+ ctx: AgentRunContext,
220
+ stopHookPath?: string,
221
+ ): {
222
+ scriptPath: string;
223
+ settingsPath: string;
224
+ } {
225
+ const scriptPath = join(ctx.tmpdir, CLAUDE_PRETOOL_GATE_FILENAME);
226
+ writeFileSync(
227
+ scriptPath,
228
+ buildClaudePretoolGateSource(
229
+ ctx.subagentDeniedTools,
230
+ writePolicyPath(ctx.tmpdir),
231
+ ),
232
+ );
233
+ chmodSync(scriptPath, 0o755);
234
+ const settingsPath = join(ctx.tmpdir, "infraweaver-claude-settings.json");
235
+ const settings = buildClaudePretoolGateSettings(
236
+ scriptPath,
237
+ CLAUDE_EXEC_TOOL_DENY_RULES,
238
+ );
239
+ // the harness-owned tmpdir files (this gate script, the Stop hook, the
240
+ // write-policy JSON) are executed/re-read during the run, so a native write
241
+ // to any of them is command execution — deny in the flag settings too so
242
+ // non-CI runs (no managed settings) carry the same fence.
243
+ settings.permissions.deny.push(...harnessTmpdirWriteDenyClaude(ctx.tmpdir));
244
+ if (stopHookPath) {
245
+ // register the Stop hook here as well as in managed settings: in CI
246
+ // `allowManagedHooksOnly` filters this flag-settings copy (no double
247
+ // fire), while non-CI runs — where managed settings are never installed —
248
+ // otherwise have no Stop hook at all, making gate retries and reflection
249
+ // dead locally while finalizeAgentResult still hard-fails on their
250
+ // absence.
251
+ (settings.hooks as Record<string, unknown>).Stop = [
252
+ { hooks: [{ type: "command", command: stopHookPath }] },
253
+ ];
254
+ }
255
+ writeFileSync(settingsPath, JSON.stringify(settings));
256
+ return { scriptPath, settingsPath };
257
+ }
258
+
259
+ /**
260
+ * Build the `--agents` JSON definition for the `loupe` subagent.
261
+ *
262
+ * The Claude Code path always runs against an Anthropic model (see
263
+ * resolveAgent), so we hardcode the cheaper-sibling downshift: lenses run
264
+ * on Sonnet, the orchestrator stays on whatever model `--model` was passed.
265
+ *
266
+ * Per-call model override is also possible (Task tool's `model` arg accepts
267
+ * 'sonnet' | 'opus' | 'haiku') and takes precedence over what's set here —
268
+ * we don't pass it; the per-subagent `model` field is the right default.
269
+ *
270
+ * The `model` is route-aware: the bare Anthropic-API id `claude-sonnet-5`
271
+ * is only valid on the direct API. On Bedrock (`eu.anthropic.…`) and Vertex
272
+ * (`@`-versioned) routes that id is rejected, and claude-code does not
273
+ * translate per-agent model names — so on those routes we omit `model`
274
+ * entirely and the lens inherits the orchestrator's (already valid) model.
275
+ *
276
+ * Kept in lockstep with the `claude-sonnet` alias in models.ts (the opencode
277
+ * path derives the same target via deriveSubagentModels); bump both together.
278
+ *
279
+ * The non-mutative + non-recursive contract is enforced by the prose system
280
+ * prompt baked into the agent — see action/agents/reviewer.ts for why we
281
+ * no longer wire per-agent `disallowedTools` here.
282
+ */
283
+ export function buildAgentsJson(routeAware?: {
284
+ isBedrock: boolean;
285
+ isVertex: boolean;
286
+ }): string {
287
+ const inheritOrchestratorModel =
288
+ routeAware?.isBedrock || routeAware?.isVertex;
289
+ const agents = {
290
+ [REVIEWER_AGENT_NAME]: {
291
+ description:
292
+ "Read-only review subagent for lens-based code review (correctness, security, billing-subsystem, etc.). " +
293
+ "Reads only — no writes, no state-changing shell or MCP calls, no nested subagent dispatch.",
294
+ prompt: REVIEWER_SYSTEM_PROMPT,
295
+ // omit on non-direct-API routes so the id can't be rejected by the provider.
296
+ ...(inheritOrchestratorModel ? {} : { model: "claude-sonnet-5" }),
297
+ },
298
+ };
299
+ return JSON.stringify(agents);
300
+ }
301
+
302
+ // ── model helpers ─────────────────────────────────────────────────────────────
303
+
304
+ // claude CLI expects bare model names (e.g. "claude-sonnet-5"), not provider-prefixed specifiers
305
+ export function stripProviderPrefix(specifier: string): string {
306
+ const slashIndex = specifier.indexOf("/");
307
+ return slashIndex > 0 ? specifier.slice(slashIndex + 1) : specifier;
308
+ }
309
+
310
+ // `high` is the model's tuned default ("equivalent to not setting the parameter"
311
+ // per Anthropic docs). `max` is "absolute maximum capability with no constraints
312
+ // on token spending" — meaningfully slower and burns more thinking budget per
313
+ // turn. We default everyone to `high`; PRs that genuinely need full-send can
314
+ // opt in via a future per-run override rather than paying the wall-time cost on
315
+ // every Opus run.
316
+ function resolveEffort(_model: string | undefined): "high" {
317
+ return "high";
318
+ }
319
+
320
+ // ── NDJSON event types ─────────────────────────────────────────────────────────
321
+
322
+ interface ContentBlock {
323
+ type: string;
324
+ text?: string;
325
+ id?: string;
326
+ name?: string;
327
+ input?: unknown;
328
+ tool_use_id?: string;
329
+ content?: string | unknown;
330
+ is_error?: boolean;
331
+ [key: string]: unknown;
332
+ }
333
+
334
+ // SDK schema (per claude-agent-sdk docs) puts `session_id` and
335
+ // `parent_tool_use_id` at the top level of every Assistant/User/System/Result
336
+ // message, not inside `message`. Subagent events carry a non-null
337
+ // `parent_tool_use_id` pointing at the orchestrator's Task/Agent tool_use id.
338
+ interface ClaudeSystemEvent {
339
+ type: "system";
340
+ session_id?: string;
341
+ parent_tool_use_id?: string | null;
342
+ [key: string]: unknown;
343
+ }
344
+
345
+ interface ClaudeAssistantEvent {
346
+ type: "assistant";
347
+ session_id?: string;
348
+ parent_tool_use_id?: string | null;
349
+ message?: {
350
+ role?: string;
351
+ content?: ContentBlock[];
352
+ model?: string;
353
+ usage?: {
354
+ input_tokens?: number;
355
+ output_tokens?: number;
356
+ cache_creation_input_tokens?: number;
357
+ cache_read_input_tokens?: number;
358
+ };
359
+ [key: string]: unknown;
360
+ };
361
+ [key: string]: unknown;
362
+ }
363
+
364
+ interface ClaudeUserEvent {
365
+ type: "user";
366
+ session_id?: string;
367
+ parent_tool_use_id?: string | null;
368
+ message?: {
369
+ role?: string;
370
+ content?: ContentBlock[];
371
+ [key: string]: unknown;
372
+ };
373
+ [key: string]: unknown;
374
+ }
375
+
376
+ interface ClaudeResultEvent {
377
+ type: "result";
378
+ subtype?: string;
379
+ // claude CLI sets `is_error: true` (alongside `subtype: "success"`) when
380
+ // an upstream provider fails mid-stream. `api_error_status` carries the
381
+ // provider HTTP status (e.g. 401 for invalid API key). per the official
382
+ // SDK types, `api_error_status` is `number | null`, and the `error_*`
383
+ // subtypes carry their actionable payload in `errors: string[]` instead
384
+ // of `result`.
385
+ is_error?: boolean;
386
+ api_error_status?: number | null;
387
+ errors?: string[];
388
+ result?: string;
389
+ session_id?: string;
390
+ num_turns?: number;
391
+ total_cost_usd?: number;
392
+ total_input_tokens?: number;
393
+ total_output_tokens?: number;
394
+ usage?: {
395
+ input_tokens?: number;
396
+ output_tokens?: number;
397
+ cache_read_input_tokens?: number;
398
+ cache_creation_input_tokens?: number;
399
+ };
400
+ [key: string]: unknown;
401
+ }
402
+
403
+ // additional event types emitted by Claude CLI (handled as no-ops / debug)
404
+ interface ClaudeStreamEvent {
405
+ type: "stream_event";
406
+ [key: string]: unknown;
407
+ }
408
+ interface ClaudeToolProgressEvent {
409
+ type: "tool_progress";
410
+ [key: string]: unknown;
411
+ }
412
+ interface ClaudeToolUseSummaryEvent {
413
+ type: "tool_use_summary";
414
+ [key: string]: unknown;
415
+ }
416
+ interface ClaudeAuthStatusEvent {
417
+ type: "auth_status";
418
+ [key: string]: unknown;
419
+ }
420
+
421
+ type ClaudeEvent =
422
+ | ClaudeSystemEvent
423
+ | ClaudeAssistantEvent
424
+ | ClaudeUserEvent
425
+ | ClaudeResultEvent
426
+ | ClaudeStreamEvent
427
+ | ClaudeToolProgressEvent
428
+ | ClaudeToolUseSummaryEvent
429
+ | ClaudeAuthStatusEvent;
430
+
431
+ // ── runner ──────────────────────────────────────────────────────────────────────
432
+
433
+ type RunParams = {
434
+ label: string;
435
+ cmd: string;
436
+ args: string[];
437
+ cwd: string;
438
+ env: Record<string, string | undefined>;
439
+ todoTracker?: TodoTracker | undefined;
440
+ onActivityTimeout?: (() => void) | undefined;
441
+ onToolUse?:
442
+ | ((event: { toolName: string; input: unknown }) => void)
443
+ | undefined;
444
+ /** per-run token ceiling (LLM10). `label` is the accumulator's source key, so
445
+ * a `--resume` retry is charged on top of the attempt it resumes rather than
446
+ * replacing it. Absent → nothing to enforce. */
447
+ tokenQuota?: RunTokenQuota | undefined;
448
+ /** specialist-routing watch (agents/dispatchMetrics.ts). Passed as the bare
449
+ * recorder rather than the whole `toolState` — this runner drives a
450
+ * subprocess and has no business reading run state it cannot act on.
451
+ * Optional for the same reason `tokenQuota` is: subagent and test call sites
452
+ * have nothing to record. */
453
+ dispatchMetrics?: DispatchRecorder | undefined;
454
+ };
455
+
456
+ type ClaudeRunResult = AgentResult & {
457
+ sessionId?: string | undefined;
458
+ /** set when the failure was a synthetic-stop with a retryable provider
459
+ * status — the session is intact and worth a `--resume` retry. */
460
+ retryableTransient?: boolean | undefined;
461
+ };
462
+
463
+ /**
464
+ * Provider statuses worth a resume-retry: rate limit (429) and upstream 5xx
465
+ * (500–599, which includes Anthropic's 529 "overloaded"). Deterministic
466
+ * failures (401 auth, 400 validation) are excluded — resuming would just
467
+ * replay the same error at full cost.
468
+ */
469
+ export function isRetryableApiStatus(
470
+ status: number | null | undefined,
471
+ ): boolean {
472
+ if (typeof status !== "number") return false;
473
+ return status === 429 || (status >= 500 && status < 600);
474
+ }
475
+
476
+ /**
477
+ * Backoff schedule for transient-failure resume retries (one entry per
478
+ * retry). Bounded at two: a blip clears within seconds; anything that
479
+ * survives ~20s of backoff is an outage the run cannot wait out.
480
+ */
481
+ const TRANSIENT_RESUME_RETRY_DELAYS_MS: readonly number[] = [5_000, 15_000];
482
+
483
+ /**
484
+ * Continue prompt for a `--resume` retry after a transient provider failure.
485
+ * The interrupted turn's tail may or may not have been applied server-side,
486
+ * so the prompt asks the model to re-check before repeating its last action.
487
+ */
488
+ export const TRANSIENT_RESUME_PROMPT =
489
+ "SESSION INTERRUPTED — the previous turn was cut short by a transient provider error (rate limit or upstream outage), not by anything you did. " +
490
+ "Continue the task from where you left off. Re-check the result of your last action before repeating it — it may or may not have completed.";
491
+
492
+ /**
493
+ * Return the tail of `text` capped at `maxCodeUnits` UTF-16 code units,
494
+ * dropping any partial first line. used in the exit-non-zero stdout fallback
495
+ * so we never surface a truncated NDJSON event to operators —
496
+ * `result.stdout.slice(-2048)` would otherwise cut mid-line and produce a
497
+ * syntactically broken JSON fragment. code units rather than bytes because
498
+ * `String.prototype.slice` operates on UTF-16 units; for multi-byte UTF-8
499
+ * content the effective byte budget can be up to 4× the nominal limit.
500
+ */
501
+ export function tailLines(text: string, maxCodeUnits: number): string {
502
+ if (text.length <= maxCodeUnits) return text;
503
+ const tail = text.slice(-maxCodeUnits);
504
+ const firstNewline = tail.indexOf("\n");
505
+ // if no newline in window or it's at the very start, return as-is;
506
+ // otherwise drop the partial first line.
507
+ return firstNewline > 0 && firstNewline < tail.length - 1
508
+ ? tail.slice(firstNewline + 1)
509
+ : tail;
510
+ }
511
+
512
+ /** Monotonic id so each `runClaude` invocation charges the quota under its own
513
+ * source key. Process-local and never persisted — see `quotaSource` below. */
514
+ let quotaSourceCounter = 0;
515
+ function nextQuotaSourceId(): number {
516
+ quotaSourceCounter += 1;
517
+ return quotaSourceCounter;
518
+ }
519
+
520
+ export async function runClaude(params: RunParams): Promise<ClaudeRunResult> {
521
+ const startTime = performance.now();
522
+ let eventCount = 0;
523
+
524
+ // per-session labeler so parallel subagent log lines can be differentiated.
525
+ // claude-agent-sdk runs subagents inside the orchestrator's session — they
526
+ // share `session_id` — and stamps every subagent message with a non-null
527
+ // `parent_tool_use_id` pointing at the Agent tool_use that spawned them.
528
+ // we bind each Agent tool_use id to its dispatched label up front, then
529
+ // labelFor short-circuits to the direct mapping when parent_tool_use_id is
530
+ // set. orchestrator events (parent_tool_use_id === null) flow through the
531
+ // sessionID path and bind to ORCHESTRATOR_LABEL on first sighting.
532
+ const labeler = new SessionLabeler();
533
+ function eventLabel(event: {
534
+ session_id?: string;
535
+ parent_tool_use_id?: string | null;
536
+ }): string {
537
+ return labeler.labelFor(
538
+ event.session_id ?? null,
539
+ event.parent_tool_use_id ?? null,
540
+ );
541
+ }
542
+ function withLabel(label: string, message: string): string {
543
+ return label === ORCHESTRATOR_LABEL
544
+ ? message
545
+ : formatWithLabel(label, message);
546
+ }
547
+
548
+ // one ThinkingTimer per session — sharing a single timer across sessions
549
+ // conflated cross-session interleaving as parent thinking time. each timer
550
+ // formats its log lines through the session label so attribution is visible.
551
+ const thinkingTimers = new Map<string, ThinkingTimer>();
552
+ function timerFor(label: string): ThinkingTimer {
553
+ let t = thinkingTimers.get(label);
554
+ if (!t) {
555
+ const formatLine = (line: string) =>
556
+ label === ORCHESTRATOR_LABEL ? line : formatWithLabel(label, line);
557
+ t = new ThinkingTimer(formatLine);
558
+ thinkingTimers.set(label, t);
559
+ }
560
+ return t;
561
+ }
562
+
563
+ let finalOutput = "";
564
+ let sessionId: string | undefined;
565
+ let resultErrorSubtype: string | null = null;
566
+ // captures the structured error string from a result event with
567
+ // `is_error: true` (e.g. mid-stream provider auth failures the CLI
568
+ // surfaces as `subtype: "success"` synthetic-stop events, or the
569
+ // `errors[]` array from `error_*` subtypes). preferred over raw
570
+ // stdout/stderr in the exit-non-zero path so the GitHub Actions
571
+ // `##[error]` line shows the actionable message instead of an 8KB+
572
+ // NDJSON dump.
573
+ let lastResultError: string | null = null;
574
+ // set only for synthetic-stop `subtype: "success"` + `is_error: true`
575
+ // events, where `accumulatedTokens` from prior `assistant` events is
576
+ // stale and logging it would mislead operators into thinking billable
577
+ // tokens were spent on a successful turn. deliberately NOT set for
578
+ // `error_max_turns` / `error_during_execution` / `error_*` subtypes
579
+ // because those runs genuinely consumed tokens and operators need
580
+ // billing visibility for them.
581
+ let syntheticStopFailure = false;
582
+ // provider HTTP status from the synthetic-stop result event, used to decide
583
+ // whether the failure is worth a `--resume` retry (see isRetryableApiStatus).
584
+ let lastApiErrorStatus: number | null = null;
585
+ let accumulatedTokens = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 };
586
+ // Claude CLI reports a single end-of-run `total_cost_usd` on the result
587
+ // event. per-message events don't carry cost, so there's nothing to sum —
588
+ // we just capture the final value when it arrives.
589
+ let accumulatedCostUsd = 0;
590
+ let tokensLogged = false;
591
+
592
+ // LLM10. The activity watchdog bounds silence; a remediation loop that keeps
593
+ // emitting tokens is never silent, so it needs its own bound. Aborting the
594
+ // spawn is what actually stops consumption — returning a flag would let the
595
+ // CLI keep burning tokens until it happened to finish.
596
+ const quotaAbort = new AbortController();
597
+ // The source key must be unique PER INVOCATION, not per label: a `--resume`
598
+ // retry reuses the label but starts a fresh accumulator from zero, and the
599
+ // quota charges only increases — so a shared key would make every retry's
600
+ // tokens free until it happened to exceed the previous attempt's total.
601
+ const quotaSource = `${params.label}#${nextQuotaSourceId()}`;
602
+ function chargeTokenQuota(): void {
603
+ const quota = params.tokenQuota;
604
+ if (!quota || quotaAbort.signal.aborted) return;
605
+ if (!quota.observe(quotaSource, accumulatedTokens)) return;
606
+ log.info(`» ${params.label}: ${quota.gap()}`);
607
+ quotaAbort.abort(new TokenQuotaExceededError(quota.gap()));
608
+ }
609
+
610
+ function buildUsage(): AgentUsage | undefined {
611
+ return buildAgentUsage({
612
+ agent: "claude",
613
+ input: accumulatedTokens.input,
614
+ output: accumulatedTokens.output,
615
+ cacheRead: accumulatedTokens.cacheRead,
616
+ cacheWrite: accumulatedTokens.cacheWrite,
617
+ costUsd: accumulatedCostUsd,
618
+ });
619
+ }
620
+
621
+ const handlers = {
622
+ system: (event: ClaudeSystemEvent) => {
623
+ // claude-agent-sdk only emits system:init for the top-level query, so
624
+ // this binds the orchestrator label and never appears in subagent flow.
625
+ // we still route through eventLabel so a subagent system event (if the
626
+ // SDK ever adds one) wouldn't go silently misattributed.
627
+ const label = eventLabel(event);
628
+ log.debug(withLabel(label, `» ${params.label} system event`));
629
+ },
630
+ assistant: (event: ClaudeAssistantEvent) => {
631
+ const content = event.message?.content;
632
+ if (!content) return;
633
+
634
+ const label = eventLabel(event);
635
+ const boxTitle =
636
+ label === ORCHESTRATOR_LABEL
637
+ ? params.label
638
+ : `${params.label} [${label}]`;
639
+
640
+ for (const block of content) {
641
+ if (block.type === "text" && block.text?.trim()) {
642
+ const message = block.text.trim();
643
+ log.box(message, { title: boxTitle });
644
+ // only the orchestrator's text becomes the run's "output" — subagent
645
+ // report-back text would otherwise clobber the parent's final answer.
646
+ if (label === ORCHESTRATOR_LABEL) {
647
+ finalOutput = message;
648
+ }
649
+ } else if (block.type === "tool_use") {
650
+ const toolName = block.name || "unknown";
651
+ // only the orchestrator's tool calls feed the diff-coverage tracker —
652
+ // a specialist subagent reading the diff cannot satisfy the primary
653
+ // reviewer's coverage obligation. Matches the opencode path's
654
+ // `isOrchestrator` gate on ctx.onToolUse.
655
+ if (params.onToolUse && label === ORCHESTRATOR_LABEL) {
656
+ params.onToolUse({
657
+ toolName,
658
+ input: block.input,
659
+ });
660
+ }
661
+ timerFor(label).markToolCall();
662
+ const inputFormatted = formatJsonValue(block.input || {});
663
+ const toolCallLine =
664
+ inputFormatted !== "{}"
665
+ ? `» ${toolName}(${inputFormatted})`
666
+ : `» ${toolName}()`;
667
+ log.info(withLabel(label, toolCallLine));
668
+
669
+ // when the orchestrator dispatches a subagent, bind the Agent
670
+ // tool_use id to the dispatched label so future events carrying
671
+ // `parent_tool_use_id === block.id` resolve directly to the right
672
+ // lens. v2.1.63+ renamed the tool to "Agent"; older versions
673
+ // emitted "Task". match both for forward-compat.
674
+ if (
675
+ (toolName === "Task" || toolName === "Agent") &&
676
+ block.input &&
677
+ typeof block.input === "object"
678
+ ) {
679
+ const taskInput = block.input as {
680
+ description?: string;
681
+ subagent_type?: string;
682
+ prompt?: string;
683
+ };
684
+ const dispatchedLabel = labeler.recordTaskDispatch(
685
+ taskInput,
686
+ block.id ?? null,
687
+ );
688
+ // the specialist-routing watch. counted here, at the Task call, not
689
+ // at completion: a dispatch that errors out still cost the routing
690
+ // decision and usually the tokens. See agents/dispatchMetrics.ts.
691
+ params.dispatchMetrics?.recordDispatch(dispatchedLabel);
692
+ log.info(
693
+ withLabel(
694
+ label,
695
+ `» dispatching subagent: ${dispatchedLabel}` +
696
+ (taskInput.subagent_type
697
+ ? ` (subagent_type=${taskInput.subagent_type})`
698
+ : ""),
699
+ ),
700
+ );
701
+ }
702
+
703
+ // agent's explicit MCP report_progress takes priority over todo tracking
704
+ if (toolName.includes("report_progress") && params.todoTracker) {
705
+ log.debug("» report_progress detected, disabling todo tracking");
706
+ params.todoTracker.cancel();
707
+ }
708
+
709
+ // parse TodoWrite events for live progress tracking. only honor the
710
+ // orchestrator's todos — subagents emit their own todo lists which
711
+ // would otherwise clobber the visible progress comment.
712
+ if (
713
+ toolName === "TodoWrite" &&
714
+ params.todoTracker?.enabled &&
715
+ label === ORCHESTRATOR_LABEL
716
+ ) {
717
+ params.todoTracker.update(block.input);
718
+ }
719
+ }
720
+ }
721
+
722
+ // accumulate per-message usage if available. capture cache fields too
723
+ // so the fallback token table (used when no final `result` event fires)
724
+ // still reports the full breakdown instead of silently dropping cache.
725
+ const msgUsage = event.message?.usage;
726
+ if (msgUsage) {
727
+ accumulatedTokens.input += msgUsage.input_tokens || 0;
728
+ accumulatedTokens.output += msgUsage.output_tokens || 0;
729
+ accumulatedTokens.cacheRead += msgUsage.cache_read_input_tokens || 0;
730
+ accumulatedTokens.cacheWrite +=
731
+ msgUsage.cache_creation_input_tokens || 0;
732
+ // attribute the same message to orchestrator-vs-subagent. `label`
733
+ // already resolved above from session_id / parent_tool_use_id, so this
734
+ // costs one addition and no extra bookkeeping — and per-message is the
735
+ // only granularity at which the two are separable at all (the terminal
736
+ // `result` event reports one run-wide total).
737
+ params.dispatchMetrics?.recordTokens(
738
+ label,
739
+ (msgUsage.input_tokens || 0) +
740
+ (msgUsage.cache_read_input_tokens || 0) +
741
+ (msgUsage.cache_creation_input_tokens || 0) +
742
+ (msgUsage.output_tokens || 0),
743
+ );
744
+ // per-message is the only point at which a runaway is still cheap to
745
+ // stop; the `result` event arrives after the tokens are already spent.
746
+ chargeTokenQuota();
747
+ }
748
+ },
749
+ user: (event: ClaudeUserEvent) => {
750
+ const content = event.message?.content;
751
+ if (!content) return;
752
+
753
+ const label = eventLabel(event);
754
+
755
+ for (const block of content) {
756
+ if (typeof block === "string") continue;
757
+ if (block.type === "tool_result") {
758
+ timerFor(label).markToolResult();
759
+
760
+ const outputContent =
761
+ typeof block.content === "string"
762
+ ? block.content
763
+ : Array.isArray(block.content)
764
+ ? (block.content as unknown[])
765
+ .map((entry: unknown) =>
766
+ typeof entry === "string"
767
+ ? entry
768
+ : typeof entry === "object" &&
769
+ entry !== null &&
770
+ "text" in entry
771
+ ? String((entry as { text: unknown }).text)
772
+ : JSON.stringify(entry),
773
+ )
774
+ .join("\n")
775
+ : String(block.content);
776
+
777
+ if (block.is_error) {
778
+ log.info(withLabel(label, `» tool error: ${outputContent}`));
779
+ } else {
780
+ log.debug(withLabel(label, `» tool output: ${outputContent}`));
781
+ }
782
+ }
783
+ }
784
+ },
785
+ result: (event: ClaudeResultEvent) => {
786
+ if (event.session_id) sessionId = event.session_id;
787
+ const subtype = event.subtype || "unknown";
788
+ const numTurns = event.num_turns || 0;
789
+
790
+ // claude CLI emits synthetic-stop result events with `subtype: "success"`
791
+ // but `is_error: true` when an upstream provider fails mid-stream (e.g.
792
+ // 401 from anthropic). short-circuit before the usage/token-table path
793
+ // so we don't log a usage table for a failed attempt and so downstream
794
+ // (`resultErrorSubtype` branch) surfaces the structured error. gated on
795
+ // `subtype === "success"` because the `error_*` subtypes also set
796
+ // `is_error: true` but carry their payload in `errors: string[]` and
797
+ // are handled by the dedicated branches below.
798
+ if (event.is_error === true && subtype === "success") {
799
+ const apiStatus = event.api_error_status;
800
+ lastResultError =
801
+ event.result?.trim() ||
802
+ `claude reported is_error=true with no result text (api_error_status=${apiStatus ?? "unknown"})`;
803
+ resultErrorSubtype = subtype;
804
+ syntheticStopFailure = true;
805
+ lastApiErrorStatus = typeof apiStatus === "number" ? apiStatus : null;
806
+ log.info(
807
+ `» ${params.label} result error: subtype=${subtype}, api_error_status=${apiStatus ?? "unknown"}, message=${lastResultError}`,
808
+ );
809
+ return;
810
+ }
811
+
812
+ if (subtype === "success") {
813
+ // extract detailed usage from result event (most accurate source).
814
+ // note: `input` here is non-cached input tokens only, matching the
815
+ // semantics of OpenCode's step_finish.tokens.input — the logTokenTable
816
+ // helper sums Input + Cache Read + Cache Write + Output into the Total
817
+ // column so consumers get the real billable figure.
818
+ const usage = event.usage;
819
+ const inputTokens = usage?.input_tokens || 0;
820
+ const cacheRead = usage?.cache_read_input_tokens || 0;
821
+ const cacheWrite = usage?.cache_creation_input_tokens || 0;
822
+ const outputTokens = usage?.output_tokens || 0;
823
+ // guard against NaN/Infinity from malformed CLI output poisoning the total
824
+ const costUsd =
825
+ typeof event.total_cost_usd === "number" &&
826
+ Number.isFinite(event.total_cost_usd)
827
+ ? event.total_cost_usd
828
+ : 0;
829
+
830
+ accumulatedTokens = {
831
+ input: inputTokens,
832
+ output: outputTokens,
833
+ cacheRead,
834
+ cacheWrite,
835
+ };
836
+ accumulatedCostUsd = costUsd;
837
+ // the authoritative figure REPLACES the per-message accumulation above,
838
+ // so the quota charges only the increase (see tokenQuota.ts). This turn
839
+ // is already over — recording it keeps a multi-attempt run's total
840
+ // honest, which is what bounds the NEXT attempt.
841
+ chargeTokenQuota();
842
+
843
+ log.info(
844
+ `» ${params.label} result: subtype=${subtype}, turns=${numTurns}`,
845
+ );
846
+
847
+ if (!tokensLogged) {
848
+ logTokenTable({
849
+ input: inputTokens,
850
+ cacheRead,
851
+ cacheWrite,
852
+ output: outputTokens,
853
+ costUsd,
854
+ });
855
+ tokensLogged = true;
856
+ }
857
+ } else if (subtype === "error_max_turns") {
858
+ resultErrorSubtype = subtype;
859
+ lastResultError = event.errors?.join("\n").trim() || null;
860
+ log.info(
861
+ `» ${params.label} max turns reached: ${JSON.stringify(event)}`,
862
+ );
863
+ } else if (subtype === "error_during_execution") {
864
+ resultErrorSubtype = subtype;
865
+ lastResultError = event.errors?.join("\n").trim() || null;
866
+ log.info(`» ${params.label} execution error: ${JSON.stringify(event)}`);
867
+ } else if (subtype.startsWith("error")) {
868
+ resultErrorSubtype = subtype;
869
+ lastResultError = event.errors?.join("\n").trim() || null;
870
+ log.info(
871
+ `» ${params.label} result: subtype=${subtype}, data=${JSON.stringify(event)}`,
872
+ );
873
+ } else {
874
+ log.info(
875
+ `» ${params.label} result: subtype=${subtype}, data=${JSON.stringify(event)}`,
876
+ );
877
+ }
878
+
879
+ if (event.result?.trim()) {
880
+ finalOutput = event.result.trim();
881
+ }
882
+ },
883
+ // additional Claude CLI event types — debug-logged only
884
+ stream_event: () => {},
885
+ tool_progress: () => {},
886
+ tool_use_summary: () => {},
887
+ auth_status: () => {},
888
+ };
889
+
890
+ const recentStderr: string[] = [];
891
+ // ring buffer of recent non-JSON stdout lines. Claude CLI prints
892
+ // human-readable TTY chrome (status bubbles, quota notices, etc.)
893
+ // alongside the NDJSON event stream. when the CLI exits non-zero without
894
+ // emitting a structured error event, these lines are the only actionable
895
+ // signal — preferring them over the NDJSON tail keeps progress comments
896
+ // readable. issue #643.
897
+ const recentNonJsonStdout: string[] = [];
898
+
899
+ let lastProviderError: string | null = null;
900
+
901
+ // capped accumulator — see opencode.ts for rationale (issue #680).
902
+ const output = new TailBuffer(DEFAULT_MAX_RETAINED_BYTES);
903
+ let stdoutBuffer = "";
904
+
905
+ // parse + dispatch a single already-trimmed NDJSON line. shared between the
906
+ // streaming chunk handler and the post-exit flush of the final unterminated
907
+ // line, so a `result` event on a stream that ends without a trailing newline
908
+ // is never dropped.
909
+ const dispatchLine = (trimmed: string): void => {
910
+ let event: ClaudeEvent;
911
+ try {
912
+ event = JSON.parse(trimmed) as ClaudeEvent;
913
+ } catch {
914
+ log.debug(`» non-JSON stdout line: ${trimmed.substring(0, 200)}`);
915
+ recentNonJsonStdout.push(trimmed);
916
+ if (recentNonJsonStdout.length > MAX_STDERR_LINES)
917
+ recentNonJsonStdout.shift();
918
+ return;
919
+ }
920
+
921
+ eventCount++;
922
+ log.debug(JSON.stringify(event, null, 2));
923
+
924
+ const handler = handlers[event.type as keyof typeof handlers];
925
+ if (!handler) {
926
+ log.debug(`» ${params.label} event (unhandled): type=${event.type}`);
927
+ return;
928
+ }
929
+ try {
930
+ (handler as (e: ClaudeEvent) => void)(event);
931
+ } catch (err) {
932
+ log.info(
933
+ `» ${params.label} handler for type=${event.type} threw: ${err instanceof Error ? err.message : String(err)}`,
934
+ );
935
+ }
936
+ };
937
+
938
+ try {
939
+ const result = await spawn({
940
+ cmd: params.cmd,
941
+ args: params.args,
942
+ cwd: params.cwd,
943
+ env: params.env,
944
+ // flat agent idle budget — long synchronous MCP tool calls (issue #760)
945
+ // sit well under it, so no per-toolcall suspend bracketing is needed.
946
+ activityTimeout: AGENT_ACTIVITY_TIMEOUT_MS,
947
+ onActivityTimeout: params.onActivityTimeout,
948
+ // LLM10: kills the CLI (and its group) the moment the run's token
949
+ // ceiling is reached. See chargeTokenQuota above.
950
+ abortSignal: quotaAbort.signal,
951
+ stdio: ["ignore", "pipe", "pipe"],
952
+ // run claude in its own process group so SIGKILL on activity timeout /
953
+ // outer cancellation reaches any subprocesses it spawns (rg, file
954
+ // watchers, mcp transports, etc). claude (2.1.113+) is now a native
955
+ // binary like opencode-ai/bin/opencode, so detached + killGroup is
956
+ // required to avoid orphaning the binary and its children.
957
+ killGroup: true,
958
+ // claude already drains every chunk via onStdout (NDJSON parsing) and
959
+ // onStderr (recentStderr ring buffer). retaining a second copy in the
960
+ // spawn wrapper would grow unbounded for long sessions and previously
961
+ // crashed the wrapper with RangeError. see issue #680.
962
+ retain: "none",
963
+ onStdout: async (chunk) => {
964
+ // capture idle BEFORE markActivity resets the clock, so the gap that
965
+ // elapsed while the agent was silent (inside a long tool call) is
966
+ // actually observable — reading it after markActivity always yields ~0.
967
+ const idleBeforeChunk = getIdleMs();
968
+ const text = chunk.toString();
969
+ output.append(text);
970
+ markActivity();
971
+
972
+ if (idleBeforeChunk > 10000) {
973
+ log.info(
974
+ `» no activity for ${(idleBeforeChunk / 1000).toFixed(1)}s (${params.label} may be processing internally) (${eventCount} events processed so far)`,
975
+ );
976
+ }
977
+
978
+ stdoutBuffer += text;
979
+ const lines = stdoutBuffer.split("\n");
980
+ stdoutBuffer = lines.pop() || "";
981
+
982
+ for (const line of lines) {
983
+ const trimmed = line.trim();
984
+ if (!trimmed) continue;
985
+ dispatchLine(trimmed);
986
+ }
987
+ },
988
+ onStderr: (chunk) => {
989
+ const trimmed = chunk.trim();
990
+ if (!trimmed) return;
991
+
992
+ recentStderr.push(trimmed);
993
+ if (recentStderr.length > MAX_STDERR_LINES) recentStderr.shift();
994
+
995
+ const match = findProviderErrorMatch(trimmed);
996
+ if (match) {
997
+ lastProviderError = match.label;
998
+ log.info(
999
+ `» provider error detected (${match.label}): ${match.excerpt}`,
1000
+ );
1001
+ } else {
1002
+ log.debug(trimmed);
1003
+ }
1004
+ },
1005
+ });
1006
+
1007
+ // flush a final line that arrived without a trailing newline. the CLI's
1008
+ // terminal `{"type":"result",…}` event can land here if the stream ends
1009
+ // (or is SIGKILLed) between the write and its `\n` — dropping it would lose
1010
+ // the authoritative usage/cost and the structured is_error payload.
1011
+ const tail = stdoutBuffer.trim();
1012
+ if (tail) dispatchLine(tail);
1013
+
1014
+ if (result.exitCode === 0) {
1015
+ await params.todoTracker?.flush();
1016
+ } else {
1017
+ params.todoTracker?.cancel();
1018
+ }
1019
+
1020
+ const duration = performance.now() - startTime;
1021
+ log.info(
1022
+ `» ${params.label} completed in ${Math.round(duration)}ms with exit code ${result.exitCode}`,
1023
+ );
1024
+
1025
+ if (eventCount === 0) {
1026
+ const stderrContext = recentStderr.join("\n");
1027
+ const diagnosis = lastProviderError
1028
+ ? `provider error: ${lastProviderError}`
1029
+ : "unknown cause (no stdout events received)";
1030
+ log.info(`» ${params.label} produced 0 events (${diagnosis})`);
1031
+ if (stderrContext) log.info(`» last stderr output:\n${stderrContext}`);
1032
+ }
1033
+
1034
+ // skip the fallback token table only for the synthetic-stop
1035
+ // `subtype: "success"` + `is_error: true` case: `accumulatedTokens` from
1036
+ // prior `assistant` events is stale there and logging it would mislead
1037
+ // operators into thinking billable tokens were spent on a successful turn.
1038
+ // `error_max_turns` / `error_during_execution` / `error_*` subtypes
1039
+ // represent runs that genuinely consumed tokens, so they still get the
1040
+ // table for billing visibility.
1041
+ if (
1042
+ !tokensLogged &&
1043
+ !syntheticStopFailure &&
1044
+ (accumulatedTokens.input > 0 ||
1045
+ accumulatedTokens.output > 0 ||
1046
+ accumulatedTokens.cacheRead > 0 ||
1047
+ accumulatedTokens.cacheWrite > 0)
1048
+ ) {
1049
+ logTokenTable({ ...accumulatedTokens, costUsd: accumulatedCostUsd });
1050
+ tokensLogged = true;
1051
+ }
1052
+
1053
+ const usage = buildUsage();
1054
+ const retryableTransient =
1055
+ syntheticStopFailure && isRetryableApiStatus(lastApiErrorStatus);
1056
+
1057
+ if (result.exitCode !== 0) {
1058
+ const errorContext = lastProviderError ? ` (${lastProviderError})` : "";
1059
+ // prefer the structured `lastResultError` (parsed from a result event
1060
+ // with `is_error: true`) over raw stdout. raw stdout is the full NDJSON
1061
+ // event stream — dumping it into a GitHub Actions `##[error]` line both
1062
+ // hides the actionable provider message and pollutes the run log. cap
1063
+ // the stdout fallback to the last 2KB so it stays readable when neither
1064
+ // a structured error nor stderr is available.
1065
+ //
1066
+ // result.stdout / result.stderr are empty because we pass retain:"none"
1067
+ // to spawn (see issue #680); the agent layer keeps its own bounded
1068
+ // mirrors via `output` (TailBuffer) and `recentStderr` (ring buffer).
1069
+ const stdoutSnapshot = output.toString();
1070
+ const stderrSnapshot = recentStderr.join("\n");
1071
+ const truncatedStdout = stdoutSnapshot
1072
+ ? tailLines(stdoutSnapshot, 2048)
1073
+ : "";
1074
+ // prefer non-JSON stdout (human-readable TTY chrome the CLI prints,
1075
+ // including status bubbles and quota notices) over the raw NDJSON
1076
+ // tail. when the CLI exits 1 without emitting `is_error` (issue #643),
1077
+ // the NDJSON fallback would otherwise dump 2KB of `system/init` events
1078
+ // into the progress comment with no mention of the actual cause.
1079
+ const nonJsonStdoutSnapshot = recentNonJsonStdout.join("\n");
1080
+ const errorMessage =
1081
+ lastResultError ||
1082
+ stderrSnapshot ||
1083
+ nonJsonStdoutSnapshot ||
1084
+ truncatedStdout ||
1085
+ `unknown error - no output from Claude CLI${errorContext}`;
1086
+ log.error(
1087
+ `${params.label} exited with code ${result.exitCode}${errorContext}: ${errorMessage}`,
1088
+ );
1089
+ log.debug(`stdout: ${stdoutSnapshot.substring(0, 500)}`);
1090
+ log.debug(`stderr: ${stderrSnapshot.substring(0, 500)}`);
1091
+ return {
1092
+ success: false,
1093
+ output: finalOutput || stdoutSnapshot,
1094
+ error: errorMessage,
1095
+ usage,
1096
+ sessionId,
1097
+ retryableTransient,
1098
+ };
1099
+ }
1100
+
1101
+ if (eventCount === 0 && lastProviderError) {
1102
+ return {
1103
+ success: false,
1104
+ output: finalOutput || output.toString(),
1105
+ error: `provider error: ${lastProviderError}`,
1106
+ usage,
1107
+ sessionId,
1108
+ };
1109
+ }
1110
+
1111
+ if (resultErrorSubtype) {
1112
+ return {
1113
+ success: false,
1114
+ output: finalOutput || output.toString(),
1115
+ error: lastResultError || `result subtype: ${resultErrorSubtype}`,
1116
+ usage,
1117
+ sessionId,
1118
+ retryableTransient,
1119
+ };
1120
+ }
1121
+
1122
+ return {
1123
+ success: true,
1124
+ output: finalOutput || output.toString(),
1125
+ usage,
1126
+ sessionId,
1127
+ };
1128
+ } catch (error) {
1129
+ params.todoTracker?.cancel();
1130
+ const duration = performance.now() - startTime;
1131
+ const errorMessage = error instanceof Error ? error.message : String(error);
1132
+ const isActivityTimeout =
1133
+ error instanceof SpawnTimeoutError &&
1134
+ error.code === SPAWN_ACTIVITY_TIMEOUT_CODE;
1135
+
1136
+ // A quota abort is a decision we made, not a fault to diagnose. Appending
1137
+ // the hang diagnosis ("N events processed before the hang") would misdirect
1138
+ // the operator into looking for a stall that never happened.
1139
+ if (error instanceof TokenQuotaExceededError) {
1140
+ log.info(
1141
+ `» ${params.label} stopped after ${(duration / 1000).toFixed(1)}s: ${errorMessage}`,
1142
+ );
1143
+ return {
1144
+ success: false,
1145
+ output: finalOutput || output.toString(),
1146
+ error: errorMessage,
1147
+ usage: buildUsage(),
1148
+ sessionId,
1149
+ };
1150
+ }
1151
+
1152
+ const stderrContext = recentStderr.slice(-10).join("\n");
1153
+ const diagnosis = lastProviderError
1154
+ ? `likely cause: ${lastProviderError}`
1155
+ : eventCount === 0
1156
+ ? "Claude produced 0 stdout events - check if the API is reachable"
1157
+ : `${eventCount} events were processed before the hang`;
1158
+
1159
+ log.info(
1160
+ `» ${params.label} ${isActivityTimeout ? "hung" : "failed"} after ${(duration / 1000).toFixed(1)}s: ${errorMessage}`,
1161
+ );
1162
+ log.info(`» diagnosis: ${diagnosis}`);
1163
+ if (stderrContext)
1164
+ log.info(
1165
+ `» recent stderr (last ${Math.min(recentStderr.length, 10)} lines):\n${stderrContext}`,
1166
+ );
1167
+
1168
+ return {
1169
+ success: false,
1170
+ output: finalOutput || output.toString(),
1171
+ error: `${errorMessage} [${diagnosis}]`,
1172
+ usage: buildUsage(),
1173
+ sessionId,
1174
+ };
1175
+ }
1176
+ }
1177
+
1178
+ // ── managed settings ────────────────────────────────────────────────────────────
1179
+
1180
+ const MANAGED_SETTINGS_DIR = "/etc/claude-code";
1181
+ const MANAGED_SETTINGS_PATH = `${MANAGED_SETTINGS_DIR}/managed-settings.json`;
1182
+
1183
+ // managed-settings.json has absolute highest precedence in Claude Code's config hierarchy.
1184
+ // it cannot be overridden by user, project, or local settings — safe against malicious PRs.
1185
+ //
1186
+ // permissions.deny blocks native tools (Read, Grep, Edit, Glob) from accessing /proc and /sys,
1187
+ // the git surfaces (blanket Edit(.git/**) write deny + narrow .git/config read deny — see
1188
+ // nativeFsDenies.ts), and any path passed in via ctx.secretDenyPaths (codex auth dir, vertex
1189
+ // creds dir, etc.).
1190
+ // sandbox.filesystem.denyRead blocks the Bash tool sandbox from reading those paths.
1191
+ // allowManagedPermissionRulesOnly prevents malicious PRs from adding allow rules that override
1192
+ // our deny rules — safe in CI because --dangerously-skip-permissions makes allow/ask irrelevant.
1193
+ // allowManagedHooksOnly prevents malicious project hooks from bypassing deny rules.
1194
+ // Per Claude Code permissions docs, Read(...) deny ALSO blocks file-reading Bash commands
1195
+ // (cat, head, tail, sed) and survives bypassPermissions mode. See wiki/security.md and
1196
+ // wiki/codex-auth.md.
1197
+
1198
+ /**
1199
+ * env var carrying the gate-server URL to the Claude subprocess. the Stop
1200
+ * hook curls it on every stop; an absent value disables the hook (e.g.
1201
+ * non-CI local dev paths that don't install managed settings either).
1202
+ */
1203
+ const STOP_HOOK_GATE_URL_ENV = "INFRAWEAVER_GATE_URL";
1204
+ // `_TOKEN` suffix is intentional: filterEnv() strips it from the agent's shell
1205
+ // sandbox, so only the Stop hook (a child of this process) can authenticate to
1206
+ // the gate server. See gateServer.ts.
1207
+ const STOP_HOOK_GATE_TOKEN_ENV = "INFRAWEAVER_GATE_TOKEN";
1208
+
1209
+ /**
1210
+ * managed Stop hook. swaps the old `--resume <sessionId>` follow-up
1211
+ * subprocesses (reflection + every gate retry — cost audit on PR #792
1212
+ * showed reflection alone burned ~$0.85 / 111K cache_write per Opus run,
1213
+ * almost all of it wasted re-running `getAttachmentMessages` in the fresh
1214
+ * process) for a `{decision: "block", reason: ...}` injection inside the
1215
+ * live `queryLoop`. existing session context is already in the prompt
1216
+ * cache so only the new reason text is fresh cache_write.
1217
+ *
1218
+ * the script is intentionally minimal — all decision logic lives in the
1219
+ * sidecar gate server (`gateServer.ts`), which reads live `ctx.toolState`
1220
+ * mutations from the same process the MCP server runs in. budget +
1221
+ * one-shot tracking lives there too, so re-fires across multiple stops in
1222
+ * one session are safe. claude-code's 8-consecutive-block override is the
1223
+ * last-line backstop.
1224
+ */
1225
+ export function buildStopHookScript(): string {
1226
+ return [
1227
+ "#!/usr/bin/env bash",
1228
+ "set -euo pipefail",
1229
+ `url="\${${STOP_HOOK_GATE_URL_ENV}:-}"`,
1230
+ `tok="\${${STOP_HOOK_GATE_TOKEN_ENV}:-}"`,
1231
+ 'if [ -z "$url" ]; then exit 0; fi',
1232
+ // jq is a hard dependency of every later line; under `set -euo pipefail` a
1233
+ // missing binary would abort mid-pipeline as an unexplained exit 1 (a
1234
+ // silent allow). fail the same way, but say why — the gate server's
1235
+ // delivery accounting (commit-on-flush) keeps its state consistent either
1236
+ // way, this is purely for operator visibility.
1237
+ 'command -v jq >/dev/null 2>&1 || { echo "infraweaver-stop-hook: jq not found — allowing stop" >&2; exit 0; }',
1238
+ "cat >/dev/null",
1239
+ // a curl failure (timeout, refused) falls back to an allow. the gate
1240
+ // server only commits budget/one-shot state once the response has flushed
1241
+ // to this curl, so an undelivered decision is re-decided on the next stop
1242
+ // rather than silently lost.
1243
+ 'response=$(curl -fsS --max-time 30 -H "Authorization: Bearer $tok" "$url" 2>/dev/null || printf \'{"block":false}\')',
1244
+ // tolerate an unparseable body (proxy interference, truncation): fall
1245
+ // back to an allow with a stderr note instead of dying under `set -e`.
1246
+ 'block=$(printf "%s" "$response" | jq -r ".block // false" 2>/dev/null || { echo "infraweaver-stop-hook: unparseable gate response — allowing stop" >&2; printf "false"; })',
1247
+ 'if [ "$block" != "true" ]; then exit 0; fi',
1248
+ 'reason=$(printf "%s" "$response" | jq -r ".reason // \\"\\"")',
1249
+ 'if [ -z "$reason" ]; then exit 0; fi',
1250
+ 'jq -n --arg reason "$reason" \'{decision: "block", reason: $reason}\'',
1251
+ "",
1252
+ ].join("\n");
1253
+ }
1254
+
1255
+ export interface ManagedSettingsParams {
1256
+ ctx: AgentRunContext;
1257
+ stopHookPath: string | null;
1258
+ pretoolGateScriptPath: string;
1259
+ }
1260
+
1261
+ export function buildManagedSettings(
1262
+ params: ManagedSettingsParams,
1263
+ ): Record<string, unknown> {
1264
+ const secretDenyPaths = params.ctx.secretDenyPaths ?? [];
1265
+ const toolDeny = secretDenyPaths.flatMap((path) => [
1266
+ `Read(${path}/**)`,
1267
+ `Read(/${path}/**)`,
1268
+ `Grep(${path}/**)`,
1269
+ `Grep(/${path}/**)`,
1270
+ `Edit(${path}/**)`,
1271
+ `Edit(/${path}/**)`,
1272
+ `Glob(${path}/**)`,
1273
+ `Glob(/${path}/**)`,
1274
+ ]);
1275
+ // single builder for both the PreToolUse gate hook and the native exec-tool
1276
+ // deny — both fields are consumed here (and identically in the flag-settings
1277
+ // path via writePretoolGateAssets), keeping CLAUDE_EXEC_TOOL_DENY_RULES the
1278
+ // single source.
1279
+ const gate = buildClaudePretoolGateSettings(
1280
+ params.pretoolGateScriptPath,
1281
+ CLAUDE_EXEC_TOOL_DENY_RULES,
1282
+ );
1283
+ const base: Record<string, unknown> = {
1284
+ allowManagedPermissionRulesOnly: true,
1285
+ allowManagedHooksOnly: true,
1286
+ permissions: {
1287
+ deny: [
1288
+ // native exec tools — the authoritative, bypass-immune deny.
1289
+ // `--disallowedTools` (a cliArg-source deny) leaked under
1290
+ // `--dangerously-skip-permissions`; policySettings denies survive
1291
+ // bypassPermissions mode. covers top-level + Agent(...) subagent use.
1292
+ ...gate.permissions.deny,
1293
+ "Read(//proc/**)",
1294
+ "Read(//sys/**)",
1295
+ "Grep(//proc/**)",
1296
+ "Grep(//sys/**)",
1297
+ "Edit(//proc/**)",
1298
+ "Edit(//sys/**)",
1299
+ "Glob(//proc/**)",
1300
+ "Glob(//sys/**)",
1301
+ // git surfaces — blanket Edit(.git/**) write deny (nothing legit
1302
+ // writes .git via native tools; real commits go through MCP git tools
1303
+ // outside this gate) + narrow Read/Grep/Glob(.git/config) read deny.
1304
+ // mirrors opencode's edit-blanket / read-narrow split. canonical:
1305
+ // action/agents/nativeFsDenies.ts.
1306
+ ...GIT_NATIVE_WRITE_DENY_CLAUDE,
1307
+ ...GIT_NATIVE_READ_DENY_CLAUDE,
1308
+ ...runnerHomeWriteDenyClaude(process.env.HOME),
1309
+ ...toolDeny,
1310
+ ],
1311
+ },
1312
+ sandbox: {
1313
+ filesystem: {
1314
+ denyRead: ["/proc", "/sys", ...secretDenyPaths],
1315
+ },
1316
+ },
1317
+ };
1318
+ // PreToolUse gate replicated into managed settings so it survives the
1319
+ // `allowManagedHooksOnly: true` policy gate (see
1320
+ // src/utils/hooks/hooksConfigSnapshot.ts in claude-code source). the Stop
1321
+ // hook (gate-server retries) is layered into the same `hooks` object when
1322
+ // present so both fire under managed settings.
1323
+ const hooks: Record<string, unknown> = {
1324
+ ...gate.hooks,
1325
+ };
1326
+ if (params.stopHookPath) {
1327
+ hooks.Stop = [
1328
+ {
1329
+ hooks: [{ type: "command", command: params.stopHookPath }],
1330
+ },
1331
+ ];
1332
+ }
1333
+ base.hooks = hooks;
1334
+ return base;
1335
+ }
1336
+
1337
+ function installManagedSettings(params: ManagedSettingsParams): void {
1338
+ if (process.env.CI !== "true") return;
1339
+
1340
+ const content = JSON.stringify(buildManagedSettings(params), null, 2);
1341
+ try {
1342
+ execFileSync("sudo", ["mkdir", "-p", MANAGED_SETTINGS_DIR]);
1343
+ execFileSync("sudo", ["tee", MANAGED_SETTINGS_PATH], {
1344
+ input: content,
1345
+ stdio: ["pipe", "ignore", "pipe"],
1346
+ });
1347
+ log.debug(`» wrote managed settings to ${MANAGED_SETTINGS_PATH}`);
1348
+ } catch (err) {
1349
+ // fail closed: managed settings carry the bypass-immune layer
1350
+ // (allowManagedHooksOnly, /proc + secret denyRead, the .git write fence).
1351
+ // Without them the run would proceed under --dangerously-skip-permissions
1352
+ // with only the flag-settings deny, so a malicious PR's project hooks
1353
+ // could execute and Read(/proc/self/environ) could exfiltrate API keys.
1354
+ // A CI run that cannot install this layer must not start.
1355
+ throw new Error(
1356
+ `failed to install managed settings (the bypass-immune security layer) — refusing to run without it: ${err instanceof Error ? err.message : String(err)}`,
1357
+ );
1358
+ }
1359
+ }
1360
+
1361
+ // ── agent ───────────────────────────────────────────────────────────────────────
1362
+
1363
+ export const claude = agent({
1364
+ name: "claude",
1365
+ install: installClaudeCli,
1366
+ run: async (ctx) => {
1367
+ const cliPath = await installClaudeCli();
1368
+
1369
+ const specifier = ctx.resolvedModel;
1370
+ // claude-code on Bedrock takes the bare AWS model ID — no provider prefix
1371
+ // to strip, since the ID is already in `provider.model` form (e.g.
1372
+ // `eu.anthropic.claude-opus-4-7`). detect via the env-var sentinel: if
1373
+ // BEDROCK_MODEL_ID is set and matches the resolved specifier, this is a
1374
+ // bedrock route. see `wiki/model-resolution.md` for the routing pattern.
1375
+ const bedrockModelId = process.env[BEDROCK_MODEL_ID_ENV]?.trim();
1376
+ const isBedrockRoute =
1377
+ specifier !== undefined &&
1378
+ bedrockModelId !== undefined &&
1379
+ bedrockModelId === specifier &&
1380
+ isBedrockAnthropicId(specifier);
1381
+ const vertexModelId = process.env[VERTEX_MODEL_ID_ENV]?.trim();
1382
+ const isVertexRoute =
1383
+ specifier !== undefined &&
1384
+ vertexModelId !== undefined &&
1385
+ vertexModelId === specifier &&
1386
+ isVertexAnthropicId(specifier);
1387
+ const model = !specifier
1388
+ ? undefined
1389
+ : isBedrockRoute
1390
+ ? specifier
1391
+ : isVertexRoute
1392
+ ? undefined
1393
+ : stripProviderPrefix(specifier);
1394
+
1395
+ const homeEnv = agentHomeEnv(ctx.tmpdir);
1396
+
1397
+ mkdirSync(join(homeEnv.XDG_CONFIG_HOME, "claude"), { recursive: true });
1398
+
1399
+ installBundledSkills({ home: homeEnv.HOME });
1400
+
1401
+ const mcpConfigPath = writeMcpConfig(ctx);
1402
+ const effort = resolveEffort(model);
1403
+
1404
+ // reflection + every gate retry (dirty tree, unsubmitted review, summary
1405
+ // stale) move from post-exit `--resume <sessionId>` subprocesses to a
1406
+ // managed Stop hook that curls a sidecar gate server. see
1407
+ // `buildStopHookScript` for the cost rationale (PR #792 audit) and
1408
+ // `gateServer.ts` for the decision policy. Written before the gate assets
1409
+ // so its path can be registered in the flag settings too (non-CI parity).
1410
+ const stopHookPath = join(ctx.tmpdir, "infraweaver-stop-hook.sh");
1411
+ writeFileSync(stopHookPath, buildStopHookScript(), { mode: 0o755 });
1412
+
1413
+ // PreToolUse gate that hard-blocks state-mutating MCP tool calls from
1414
+ // subagents (the `agent_id` field is non-empty in the hook input only
1415
+ // for subagent-originated calls — verified against
1416
+ // yasasbanukaofficial/claude-code src/utils/hooks.ts createBaseHookInput).
1417
+ // Wired via two surfaces so it fires in both CI and local (see
1418
+ // writePretoolGateAssets / buildManagedSettings comments).
1419
+ const pretoolGate = writePretoolGateAssets(ctx, stopHookPath);
1420
+
1421
+ installManagedSettings({
1422
+ ctx,
1423
+ stopHookPath,
1424
+ pretoolGateScriptPath: pretoolGate.scriptPath,
1425
+ });
1426
+
1427
+ // base args shared between initial run and continue runs
1428
+ const baseArgs = [
1429
+ "--output-format",
1430
+ "stream-json",
1431
+ "--dangerously-skip-permissions",
1432
+ "--mcp-config",
1433
+ mcpConfigPath,
1434
+ "--settings",
1435
+ pretoolGate.settingsPath,
1436
+ "--verbose",
1437
+ "--effort",
1438
+ effort,
1439
+ "--disallowedTools",
1440
+ CLAUDE_DISALLOWED_TOOLS,
1441
+ "--agents",
1442
+ buildAgentsJson({ isBedrock: isBedrockRoute, isVertex: isVertexRoute }),
1443
+ ];
1444
+
1445
+ if (model) {
1446
+ baseArgs.push("--model", model);
1447
+ }
1448
+
1449
+ // agent process gets full env — needs LLM API keys, PATH, locale, etc.
1450
+ // security is enforced via managed-settings.json, --disallowedTools (native exec tools), and MCP tool filtering.
1451
+ //
1452
+ // bedrock route: claude-code reads `CLAUDE_CODE_USE_BEDROCK=1` to switch
1453
+ // its provider implementation from the direct Anthropic API to Bedrock.
1454
+ // AWS_BEARER_TOKEN_BEDROCK / AWS_ACCESS_KEY_ID + AWS_SECRET_ACCESS_KEY +
1455
+ // AWS_REGION are already in process.env from the workflow's `env:` block.
1456
+ // see https://docs.claude.com/en/docs/claude-code/amazon-bedrock.
1457
+ //
1458
+ // we only force CLAUDE_CODE_USE_BEDROCK=1 when this is a Infraweaver-routed
1459
+ // bedrock run; if the user has set the env var manually for some other
1460
+ // reason (e.g. always-Bedrock org policy), `...process.env` already
1461
+ // carries it through and we don't disturb it.
1462
+ const repoDir = process.cwd();
1463
+
1464
+ // PWD is pinned to the repo dir — see baseAgentEnv for why.
1465
+ const env: NodeJS.ProcessEnv = baseAgentEnv(homeEnv, repoDir);
1466
+ // Claude Code auto-compacts the session at its detected model window: 200K
1467
+ // stays 200K, a 1M-context model uses 500K. Setting this keeps long
1468
+ // Remediate/Build runs from hitting the hard context limit mid-task. Respect
1469
+ // an operator override; revalidate when the pinned CLI or model windows change.
1470
+ env.CLAUDE_CODE_AUTO_COMPACT_WINDOW ||= "500000";
1471
+ if (isBedrockRoute) {
1472
+ env.CLAUDE_CODE_USE_BEDROCK = "1";
1473
+ }
1474
+ if (isVertexRoute) {
1475
+ applyClaudeVertexEnv(env);
1476
+ env.ANTHROPIC_MODEL = specifier;
1477
+ }
1478
+
1479
+ // claude-code's `Vw()` resolver prefers ANTHROPIC_API_KEY over the OAuth
1480
+ // token when both are set, so we strip the API key to fall through to the
1481
+ // Max-subscription path. bedrock route uses AWS creds and is excluded.
1482
+ // the strip is gated on a 1-token preflight: an exhausted (session/weekly
1483
+ // limit) or revoked subscription would otherwise kill the run at its first
1484
+ // model call with a working API key sitting unused in env.
1485
+ if (
1486
+ env.CLAUDE_CODE_OAUTH_TOKEN &&
1487
+ !isBedrockRoute &&
1488
+ env.ANTHROPIC_API_KEY
1489
+ ) {
1490
+ const preflight = await preflightClaudeSubscription({
1491
+ token: env.CLAUDE_CODE_OAUTH_TOKEN,
1492
+ model,
1493
+ });
1494
+ if (preflight.usable) {
1495
+ log.debug(
1496
+ "» CLAUDE_CODE_OAUTH_TOKEN present — stripping ANTHROPIC_API_KEY from Claude Code env so the OAuth subscription is used",
1497
+ );
1498
+ delete env.ANTHROPIC_API_KEY;
1499
+ } else {
1500
+ log.info(
1501
+ `» Claude subscription unusable (${preflight.reason}) — falling back to ANTHROPIC_API_KEY`,
1502
+ );
1503
+ delete env.CLAUDE_CODE_OAUTH_TOKEN;
1504
+ }
1505
+ }
1506
+
1507
+ log.info(`» effort: ${effort}`);
1508
+ log.debug(
1509
+ `» starting ${PRODUCT_NAME} (Claude Code): ${cliPath} ${baseArgs.join(" ")}`,
1510
+ );
1511
+ log.debug(`» working directory: ${repoDir}`);
1512
+
1513
+ // gate server lives only as long as the claude subprocess does. the
1514
+ // Stop hook curls `gateServer.url` and turns the response into its
1515
+ // `{decision: "block", reason}` payload (or exits 0 to allow stop).
1516
+ await using gateServer = await startGateServer(ctx);
1517
+
1518
+ const runParams = {
1519
+ label: PRODUCT_NAME,
1520
+ cmd: cliPath,
1521
+ cwd: repoDir,
1522
+ env: {
1523
+ ...env,
1524
+ [STOP_HOOK_GATE_URL_ENV]: gateServer.url,
1525
+ [STOP_HOOK_GATE_TOKEN_ENV]: gateServer.token,
1526
+ // the MCP client (this Claude process) expands ${INFRAWEAVER_MCP_TOKEN}
1527
+ // in mcp.json headers from here; filterEnv() strips it from the MCP
1528
+ // shell sandbox so a sandboxed command can't read it.
1529
+ [MCP_SERVER_TOKEN_ENV]: ctx.mcpServerToken,
1530
+ },
1531
+ todoTracker: ctx.todoTracker,
1532
+ onActivityTimeout: ctx.onActivityTimeout,
1533
+ onToolUse: ctx.onToolUse,
1534
+ tokenQuota: ctx.tokenQuota,
1535
+ // spread into the `--resume` attempt too, so a retry's dispatches and
1536
+ // tokens land in the same run-level tally rather than being lost.
1537
+ dispatchMetrics: ctx.toolState.dispatchMetrics,
1538
+ };
1539
+
1540
+ let result = await runClaude({
1541
+ ...runParams,
1542
+ args: [...baseArgs, "-p", ctx.instructions.full],
1543
+ });
1544
+
1545
+ // bounded resume-retry: a mid-run synthetic-stop with a retryable provider
1546
+ // status (429 / 5xx / 529) would otherwise discard the whole session even
1547
+ // though the CLI supports `--resume` and the session context is intact — a
1548
+ // 90-minute run must not die on one transient blip. each `--resume`
1549
+ // invocation emits a fresh session_id, so the resume target is refreshed
1550
+ // from every attempt. usage is merged across attempts for billing.
1551
+ let resumeSessionId = result.sessionId;
1552
+ for (const delayMs of TRANSIENT_RESUME_RETRY_DELAYS_MS) {
1553
+ if (
1554
+ result.success ||
1555
+ result.retryableTransient !== true ||
1556
+ !resumeSessionId
1557
+ )
1558
+ break;
1559
+ log.info(
1560
+ `» transient provider failure (${result.error ?? "unknown"}) — resuming session ${resumeSessionId} in ${Math.round(delayMs / 1000)}s`,
1561
+ );
1562
+ await sleep(delayMs);
1563
+ const attempt = await runClaude({
1564
+ ...runParams,
1565
+ args: [
1566
+ ...baseArgs,
1567
+ "--resume",
1568
+ resumeSessionId,
1569
+ "-p",
1570
+ TRANSIENT_RESUME_PROMPT,
1571
+ ],
1572
+ });
1573
+ result = {
1574
+ ...attempt,
1575
+ usage: mergeAgentUsage(result.usage, attempt.usage),
1576
+ };
1577
+ resumeSessionId = attempt.sessionId ?? resumeSessionId;
1578
+ }
1579
+
1580
+ // every follow-up turn (reflection + gate retries) has already happened
1581
+ // inside this single subprocess via the Stop hook, so usage aggregation
1582
+ // and resume orchestration are no-ops. all that remains is the terminal
1583
+ // hard-fail render: when the budget exhausted with `stopHook` /
1584
+ // `unsubmittedReview` still failing, flip `success` to false with the
1585
+ // same error shape `runPostRunRetryLoop` produced pre-migration.
1586
+ return finalizeAgentResult({ ctx, result });
1587
+ },
1588
+ });