@oneuptime/common 14.0.9 → 14.0.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (383) hide show
  1. package/Models/DatabaseModels/AutoRemediationSuggestion.ts +60 -0
  2. package/Models/DatabaseModels/CephCluster.ts +213 -0
  3. package/Models/DatabaseModels/DatabaseServer.ts +213 -0
  4. package/Models/DatabaseModels/DockerHost.ts +213 -0
  5. package/Models/DatabaseModels/DockerSwarmCluster.ts +213 -0
  6. package/Models/DatabaseModels/GlobalConfig.ts +1 -1
  7. package/Models/DatabaseModels/Host.ts +213 -0
  8. package/Models/DatabaseModels/Index.ts +2 -0
  9. package/Models/DatabaseModels/PodmanHost.ts +213 -0
  10. package/Models/DatabaseModels/ProxmoxCluster.ts +213 -0
  11. package/Models/DatabaseModels/ResourceAiAgent.ts +419 -0
  12. package/Models/DatabaseModels/RunnerJob.ts +144 -0
  13. package/Models/DatabaseModels/VMwareVCenter.ts +213 -0
  14. package/Server/API/AutoRemediationAPI.ts +392 -1
  15. package/Server/API/ResourceAiAccessAPI.ts +1669 -0
  16. package/Server/Infrastructure/Postgres/SchemaMigrations/1796300000000-AddResourceAiAgents.ts +423 -0
  17. package/Server/Infrastructure/Postgres/SchemaMigrations/Index.ts +2 -0
  18. package/Server/Infrastructure/Semaphore.ts +22 -0
  19. package/Server/Middleware/TelemetryIngest.ts +15 -5
  20. package/Server/Services/AnalyticsDatabaseService.ts +45 -1
  21. package/Server/Services/AutoRemediationRuleEngineService.ts +940 -112
  22. package/Server/Services/CephClusterService.ts +109 -1
  23. package/Server/Services/DatabaseServerService.ts +116 -1
  24. package/Server/Services/DockerHostService.ts +109 -1
  25. package/Server/Services/DockerSwarmClusterService.ts +113 -1
  26. package/Server/Services/HostService.ts +126 -4
  27. package/Server/Services/MetricRecordingRuleService.ts +47 -0
  28. package/Server/Services/PodmanHostService.ts +109 -1
  29. package/Server/Services/ProxmoxClusterService.ts +110 -1
  30. package/Server/Services/ResourceAiAccessService.ts +1439 -0
  31. package/Server/Services/ResourceAiAgentJobService.ts +202 -0
  32. package/Server/Services/ResourceAiAgentService.ts +2432 -0
  33. package/Server/Services/RunnerJobService.ts +611 -4
  34. package/Server/Services/TelemetryUsageBillingService.ts +28 -0
  35. package/Server/Services/TraceRecordingRuleService.ts +33 -0
  36. package/Server/Services/VMwareVCenterService.ts +110 -1
  37. package/Server/Utils/AI/Remediation/RemediationCommandTools.ts +1434 -32
  38. package/Server/Utils/AI/Remediation/RemediationExecutionRunner.ts +935 -21
  39. package/Server/Utils/AI/Remediation/RemediationPlanRunner.ts +25 -1
  40. package/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.ts +577 -0
  41. package/Server/Utils/AI/ResourceAccess/ResourceAccessContext.ts +271 -0
  42. package/Server/Utils/AI/ResourceAccess/ResourceAccessToolNames.ts +45 -0
  43. package/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.ts +964 -0
  44. package/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.ts +333 -0
  45. package/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.ts +873 -0
  46. package/Server/Utils/AI/SRE/AIInvestigationEngine.ts +72 -2
  47. package/Server/Utils/AI/SRE/AlertInvestigationRunner.ts +72 -5
  48. package/Server/Utils/AI/SRE/IncidentInvestigationRunner.ts +72 -5
  49. package/Server/Utils/AutoRemediation/CommandPlanExecutor.ts +396 -13
  50. package/Server/Utils/AutoRemediation/RemediationVerifier.ts +35 -2
  51. package/Server/Utils/Database/ProjectScopedReferenceValidator.ts +9 -1
  52. package/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.ts +920 -0
  53. package/Server/Utils/SessionReplay/SessionReplayUsage.ts +84 -6
  54. package/Server/Utils/Workspace/MicrosoftTeams/Actions/Alert.ts +35 -15
  55. package/Server/Utils/Workspace/MicrosoftTeams/Actions/AlertEpisode.ts +33 -15
  56. package/Server/Utils/Workspace/MicrosoftTeams/Actions/Auth.ts +21 -1
  57. package/Server/Utils/Workspace/MicrosoftTeams/Actions/Incident.ts +298 -224
  58. package/Server/Utils/Workspace/MicrosoftTeams/Actions/IncidentEpisode.ts +35 -15
  59. package/Server/Utils/Workspace/MicrosoftTeams/Actions/ScheduledMaintenance.ts +389 -175
  60. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeams.ts +366 -143
  61. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.ts +163 -0
  62. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.ts +396 -0
  63. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.ts +418 -0
  64. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.ts +152 -0
  65. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.ts +251 -0
  66. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.ts +214 -0
  67. package/Tests/App/Dashboard/ClusterAccessNotice.test.tsx +93 -0
  68. package/Tests/App/Dashboard/DatabaseDocumentationMarkdown.test.ts +18 -6
  69. package/Tests/App/Dashboard/InvestigationInfrastructureTools.test.tsx +506 -0
  70. package/Tests/App/Dashboard/RemediationSuggestionCardDescription.test.tsx +209 -0
  71. package/Tests/App/Dashboard/RemediationSuggestionCardResource.test.tsx +317 -0
  72. package/Tests/App/Dashboard/ResourceAiAccessSettingsUtil.test.ts +921 -0
  73. package/Tests/App/Dashboard/ResourceAiAgentInstall.test.ts +860 -0
  74. package/Tests/App/Dashboard/ResourceAiAgentPage.test.tsx +1721 -0
  75. package/Tests/App/Dashboard/ResourceAiAgentStatus.test.ts +1002 -0
  76. package/Tests/App/Dashboard/ResourceAiInsightsPage.test.tsx +931 -0
  77. package/Tests/App/Dashboard/ResourceAiNavigation.test.tsx +575 -0
  78. package/Tests/App/Dashboard/RunbookStepTypeMaps.test.ts +38 -7
  79. package/Tests/Models/DatabaseModels/DatabaseServerModels.test.ts +33 -1
  80. package/Tests/Models/DatabaseModels/ResourceAiAccessColumns.test.ts +738 -0
  81. package/Tests/Models/DatabaseModels/ResourceAiAgentModel.test.ts +842 -0
  82. package/Tests/Server/API/AutoRemediationApproveResourceRound.test.ts +835 -0
  83. package/Tests/Server/API/ResourceAiAccessAPI.test.ts +2730 -0
  84. package/Tests/Server/Infrastructure/Postgres/AddDatabaseServerTablesMigration.test.ts +60 -0
  85. package/Tests/Server/Infrastructure/Postgres/AddResourceAiAgentsMigration.test.ts +659 -0
  86. package/Tests/Server/Infrastructure/SemaphoreMutex.test.ts +31 -0
  87. package/Tests/Server/Middleware/TelemetryIngestBrowserKey.test.ts +63 -5
  88. package/Tests/Server/Middleware/TelemetryIngestKubernetesAgentRunnerPinnedKey.test.ts +103 -5
  89. package/Tests/Server/Middleware/TelemetryIngestKubernetesAgentRunnerRateLimit.test.ts +19 -1
  90. package/Tests/Server/Services/AutoRemediationResourceRuleEngine.test.ts +1279 -0
  91. package/Tests/Server/Services/DatabaseServerService.test.ts +55 -0
  92. package/Tests/Server/Services/GroupTelemetryUsageExcludeNames.test.ts +299 -0
  93. package/Tests/Server/Services/HostServiceFindOrCreateMemo.test.ts +44 -0
  94. package/Tests/Server/Services/MonitorProbeServiceIntervalScheduling.test.ts +45 -6
  95. package/Tests/Server/Services/RecordingRuleReservedMetricName.test.ts +171 -0
  96. package/Tests/Server/Services/ResourceAiAccessChangeAuthorization.test.ts +256 -0
  97. package/Tests/Server/Services/ResourceAiAccessService.test.ts +1373 -0
  98. package/Tests/Server/Services/ResourceAiAgentJobService.test.ts +375 -0
  99. package/Tests/Server/Services/ResourceAiAgentServiceHelpers.test.ts +1114 -0
  100. package/Tests/Server/Services/ResourceAiAgentServiceLifecycle.test.ts +1050 -0
  101. package/Tests/Server/Services/ResourceAiAgentServiceRegister.test.ts +2251 -0
  102. package/Tests/Server/Services/ResourceAiSettingsCreate.test.ts +373 -0
  103. package/Tests/Server/Services/ResourceAiSettingsPermission.test.ts +1042 -0
  104. package/Tests/Server/Services/ResourceServiceDeleteCleansUpAi.test.ts +319 -0
  105. package/Tests/Server/Services/RunnerJobEnqueueKubectl.test.ts +105 -0
  106. package/Tests/Server/Services/RunnerJobEnqueueResourceCommand.test.ts +986 -0
  107. package/Tests/Server/Services/RunnerJobResourceCommandLane.test.ts +211 -0
  108. package/Tests/Server/Services/RunnerJobResourceCommandTimeoutAndRedaction.test.ts +352 -0
  109. package/Tests/Server/Services/TelemetryUsageBillingSloExclusion.test.ts +218 -0
  110. package/Tests/Server/TestingUtils/Services/FakeRunnerJobCount.ts +145 -0
  111. package/Tests/Server/Utils/AI/InvestigationInfrastructureAccessWiring.test.ts +453 -0
  112. package/Tests/Server/Utils/AI/InvestigationInfrastructureReport.test.ts +721 -0
  113. package/Tests/Server/Utils/AI/RemediationCommandTools.test.ts +94 -1
  114. package/Tests/Server/Utils/AI/RemediationCommandToolsResource.test.ts +1530 -0
  115. package/Tests/Server/Utils/AI/RemediationExecutionRunnerResourceMode.test.ts +1249 -0
  116. package/Tests/Server/Utils/AI/RemediationPlanRunner.test.ts +117 -0
  117. package/Tests/Server/Utils/AI/RemediationResourceCopyParity.test.ts +463 -0
  118. package/Tests/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.test.ts +722 -0
  119. package/Tests/Server/Utils/AI/ResourceAccess/ResourceAccessContext.test.ts +266 -0
  120. package/Tests/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.test.ts +1220 -0
  121. package/Tests/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.test.ts +569 -0
  122. package/Tests/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.test.ts +830 -0
  123. package/Tests/Server/Utils/AutoRemediation/CommandPlanExecutorResource.test.ts +717 -0
  124. package/Tests/Server/Utils/AutoRemediation/RemediationVerifierResourceFollowUp.test.ts +338 -0
  125. package/Tests/Server/Utils/Monitor/Criteria/SessionReplayBudgetTemplateCriteria.test.ts +422 -0
  126. package/Tests/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.test.ts +1647 -0
  127. package/Tests/Server/Utils/SessionReplay/SessionReplayUsage.test.ts +143 -1
  128. package/Tests/Server/Utils/Telemetry/TelemetryIngestionKeyGuard.test.ts +33 -3
  129. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsAccountNotLinked.test.ts +1093 -0
  130. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.test.ts +1049 -0
  131. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.test.ts +1676 -0
  132. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCards.test.ts +2884 -0
  133. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.test.ts +2490 -0
  134. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateSubmit.test.ts +2293 -0
  135. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateSubmitServerTimezone.test.ts +386 -0
  136. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.test.ts +1477 -0
  137. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.test.ts +1967 -0
  138. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsStripHtmlTags.test.ts +92 -0
  139. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsSubmittedFormRemoval.test.ts +1819 -0
  140. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.test.ts +1033 -0
  141. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezoneServerZone.test.ts +610 -0
  142. package/Tests/Server/Utils/Workspace/MicrosoftTeamsBotMessageHandling.test.ts +2610 -0
  143. package/Tests/Server/Utils/Workspace/MicrosoftTeamsCreateCommandsEndToEnd.test.ts +4672 -0
  144. package/Tests/Server/Utils/Workspace/WorkspaceCreateProjectReferences.test.ts +190 -65
  145. package/Tests/Types/AutoRemediation/AiRemediationCommandPlan.test.ts +10 -4
  146. package/Tests/Types/AutoRemediation/AiRemediationCommandPlanResourceCommand.test.ts +305 -0
  147. package/Tests/Types/Monitor/Recommendation/MonitorRecommendationCatalog.test.ts +277 -13
  148. package/Tests/Types/Monitor/Recommendation/MonitorRecommendationUtil.test.ts +113 -0
  149. package/Tests/Types/Monitor/RumAlertTemplates.test.ts +773 -2
  150. package/Tests/Types/ResourceAiAgent/AiResourceType.test.ts +419 -0
  151. package/Tests/Types/ResourceAiAgent/ResourceAiAccess.test.ts +517 -0
  152. package/Tests/Types/Runbook/RunbookStepType.test.ts +147 -8
  153. package/Tests/UI/Rum/RecordingHealthDashboard.test.tsx +769 -1
  154. package/Tests/Utils/AiRemediation/Resource/CephCommandPolicy.test.ts +1653 -0
  155. package/Tests/Utils/AiRemediation/Resource/CephOutputRedaction.test.ts +204 -0
  156. package/Tests/Utils/AiRemediation/Resource/DatabaseCommandPolicy.test.ts +1077 -0
  157. package/Tests/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.test.ts +558 -0
  158. package/Tests/Utils/AiRemediation/Resource/DatabaseQueryRedactor.test.ts +834 -0
  159. package/Tests/Utils/AiRemediation/Resource/DockerCliGrammar.test.ts +747 -0
  160. package/Tests/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.test.ts +1259 -0
  161. package/Tests/Utils/AiRemediation/Resource/DockerOutputRedaction.test.ts +386 -0
  162. package/Tests/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.test.ts +782 -0
  163. package/Tests/Utils/AiRemediation/Resource/GovcCommandPolicy.test.ts +2259 -0
  164. package/Tests/Utils/AiRemediation/Resource/HostCommandPolicy.test.ts +2019 -0
  165. package/Tests/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.test.ts +2497 -0
  166. package/Tests/Utils/AiRemediation/Resource/ResourceCommandPolicy.test.ts +1305 -0
  167. package/Tests/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.test.ts +585 -0
  168. package/Tests/Utils/AiRemediation/Resource/ResourceOutputRedactor.test.ts +334 -0
  169. package/Tests/Utils/AiRemediation/Resource/ResourceOutputRedactorCopyParity.test.ts +98 -0
  170. package/Tests/Utils/AiRemediation/Resource/ResourcePolicyImportClosure.test.ts +391 -0
  171. package/Tests/Utils/AiRemediation/ResourceAiAgentPolicyCopyParity.test.ts +497 -0
  172. package/Tests/Utils/SessionReplay/SessionReplayBudgetMetricType.test.ts +383 -0
  173. package/Types/AI/ResourceAiAccessApi.ts +140 -0
  174. package/Types/AI/ResourceAiAccessPermissions.ts +126 -0
  175. package/Types/AutoRemediation/AiRemediationCommandPlan.ts +183 -2
  176. package/Types/Monitor/Recommendation/MonitorRecommendationCatalog.ts +55 -22
  177. package/Types/Monitor/Recommendation/MonitorRecommendationTypes.ts +35 -3
  178. package/Types/Monitor/RumAlertTemplates.ts +351 -4
  179. package/Types/ResourceAiAgent/AiResourceType.ts +310 -0
  180. package/Types/ResourceAiAgent/ResourceAiAccess.ts +574 -0
  181. package/Types/Rum/SessionReplayBudgetMetricType.ts +47 -0
  182. package/Types/Runbook/RunbookStepType.ts +29 -0
  183. package/Types/Telemetry/TelemetryIngestSurface.ts +14 -2
  184. package/Utils/AI/InvestigationReport.ts +121 -15
  185. package/Utils/AiRemediation/Resource/CephCommandPolicy.ts +1936 -0
  186. package/Utils/AiRemediation/Resource/DatabaseCommandPolicy.ts +166 -0
  187. package/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.ts +1532 -0
  188. package/Utils/AiRemediation/Resource/DatabaseQueryRedactor.ts +1394 -0
  189. package/Utils/AiRemediation/Resource/DockerCliGrammar.ts +1839 -0
  190. package/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.ts +490 -0
  191. package/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.ts +687 -0
  192. package/Utils/AiRemediation/Resource/GovcCommandPolicy.ts +2096 -0
  193. package/Utils/AiRemediation/Resource/HostCommandPolicy.ts +2982 -0
  194. package/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.ts +1679 -0
  195. package/Utils/AiRemediation/Resource/ResourceCommandPolicy.ts +851 -0
  196. package/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.ts +593 -0
  197. package/Utils/AiRemediation/Resource/ResourceOutputRedactor.ts +2812 -0
  198. package/Utils/SessionReplay/SessionReplayBudgetMetricType.ts +184 -0
  199. package/build/dist/Models/DatabaseModels/AutoRemediationSuggestion.js +62 -0
  200. package/build/dist/Models/DatabaseModels/AutoRemediationSuggestion.js.map +1 -1
  201. package/build/dist/Models/DatabaseModels/CephCluster.js +219 -0
  202. package/build/dist/Models/DatabaseModels/CephCluster.js.map +1 -1
  203. package/build/dist/Models/DatabaseModels/DatabaseServer.js +219 -0
  204. package/build/dist/Models/DatabaseModels/DatabaseServer.js.map +1 -1
  205. package/build/dist/Models/DatabaseModels/DockerHost.js +219 -0
  206. package/build/dist/Models/DatabaseModels/DockerHost.js.map +1 -1
  207. package/build/dist/Models/DatabaseModels/DockerSwarmCluster.js +219 -0
  208. package/build/dist/Models/DatabaseModels/DockerSwarmCluster.js.map +1 -1
  209. package/build/dist/Models/DatabaseModels/GlobalConfig.js +1 -1
  210. package/build/dist/Models/DatabaseModels/GlobalConfig.js.map +1 -1
  211. package/build/dist/Models/DatabaseModels/Host.js +219 -0
  212. package/build/dist/Models/DatabaseModels/Host.js.map +1 -1
  213. package/build/dist/Models/DatabaseModels/Index.js +2 -0
  214. package/build/dist/Models/DatabaseModels/Index.js.map +1 -1
  215. package/build/dist/Models/DatabaseModels/PodmanHost.js +219 -0
  216. package/build/dist/Models/DatabaseModels/PodmanHost.js.map +1 -1
  217. package/build/dist/Models/DatabaseModels/ProxmoxCluster.js +219 -0
  218. package/build/dist/Models/DatabaseModels/ProxmoxCluster.js.map +1 -1
  219. package/build/dist/Models/DatabaseModels/ResourceAiAgent.js +439 -0
  220. package/build/dist/Models/DatabaseModels/ResourceAiAgent.js.map +1 -0
  221. package/build/dist/Models/DatabaseModels/RunnerJob.js +145 -0
  222. package/build/dist/Models/DatabaseModels/RunnerJob.js.map +1 -1
  223. package/build/dist/Models/DatabaseModels/VMwareVCenter.js +219 -0
  224. package/build/dist/Models/DatabaseModels/VMwareVCenter.js.map +1 -1
  225. package/build/dist/Server/API/AutoRemediationAPI.js +231 -1
  226. package/build/dist/Server/API/AutoRemediationAPI.js.map +1 -1
  227. package/build/dist/Server/API/ResourceAiAccessAPI.js +1119 -0
  228. package/build/dist/Server/API/ResourceAiAccessAPI.js.map +1 -0
  229. package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/1796300000000-AddResourceAiAgents.js +168 -0
  230. package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/1796300000000-AddResourceAiAgents.js.map +1 -0
  231. package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/Index.js +2 -0
  232. package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/Index.js.map +1 -1
  233. package/build/dist/Server/Infrastructure/Semaphore.js +6 -0
  234. package/build/dist/Server/Infrastructure/Semaphore.js.map +1 -1
  235. package/build/dist/Server/Middleware/TelemetryIngest.js +12 -5
  236. package/build/dist/Server/Middleware/TelemetryIngest.js.map +1 -1
  237. package/build/dist/Server/Services/AnalyticsDatabaseService.js +26 -1
  238. package/build/dist/Server/Services/AnalyticsDatabaseService.js.map +1 -1
  239. package/build/dist/Server/Services/AutoRemediationRuleEngineService.js +601 -3
  240. package/build/dist/Server/Services/AutoRemediationRuleEngineService.js.map +1 -1
  241. package/build/dist/Server/Services/CephClusterService.js +104 -0
  242. package/build/dist/Server/Services/CephClusterService.js.map +1 -1
  243. package/build/dist/Server/Services/DatabaseServerService.js +99 -0
  244. package/build/dist/Server/Services/DatabaseServerService.js.map +1 -1
  245. package/build/dist/Server/Services/DockerHostService.js +104 -0
  246. package/build/dist/Server/Services/DockerHostService.js.map +1 -1
  247. package/build/dist/Server/Services/DockerSwarmClusterService.js +104 -0
  248. package/build/dist/Server/Services/DockerSwarmClusterService.js.map +1 -1
  249. package/build/dist/Server/Services/HostService.js +114 -2
  250. package/build/dist/Server/Services/HostService.js.map +1 -1
  251. package/build/dist/Server/Services/MetricRecordingRuleService.js +46 -0
  252. package/build/dist/Server/Services/MetricRecordingRuleService.js.map +1 -1
  253. package/build/dist/Server/Services/PodmanHostService.js +104 -0
  254. package/build/dist/Server/Services/PodmanHostService.js.map +1 -1
  255. package/build/dist/Server/Services/ProxmoxClusterService.js +104 -0
  256. package/build/dist/Server/Services/ProxmoxClusterService.js.map +1 -1
  257. package/build/dist/Server/Services/ResourceAiAccessService.js +1028 -0
  258. package/build/dist/Server/Services/ResourceAiAccessService.js.map +1 -0
  259. package/build/dist/Server/Services/ResourceAiAgentJobService.js +173 -0
  260. package/build/dist/Server/Services/ResourceAiAgentJobService.js.map +1 -0
  261. package/build/dist/Server/Services/ResourceAiAgentService.js +1687 -0
  262. package/build/dist/Server/Services/ResourceAiAgentService.js.map +1 -0
  263. package/build/dist/Server/Services/RunnerJobService.js +344 -5
  264. package/build/dist/Server/Services/RunnerJobService.js.map +1 -1
  265. package/build/dist/Server/Services/TelemetryUsageBillingService.js +26 -0
  266. package/build/dist/Server/Services/TelemetryUsageBillingService.js.map +1 -1
  267. package/build/dist/Server/Services/TraceRecordingRuleService.js +37 -0
  268. package/build/dist/Server/Services/TraceRecordingRuleService.js.map +1 -1
  269. package/build/dist/Server/Services/VMwareVCenterService.js +104 -0
  270. package/build/dist/Server/Services/VMwareVCenterService.js.map +1 -1
  271. package/build/dist/Server/Utils/AI/Remediation/RemediationCommandTools.js +1029 -30
  272. package/build/dist/Server/Utils/AI/Remediation/RemediationCommandTools.js.map +1 -1
  273. package/build/dist/Server/Utils/AI/Remediation/RemediationExecutionRunner.js +633 -28
  274. package/build/dist/Server/Utils/AI/Remediation/RemediationExecutionRunner.js.map +1 -1
  275. package/build/dist/Server/Utils/AI/Remediation/RemediationPlanRunner.js +21 -1
  276. package/build/dist/Server/Utils/AI/Remediation/RemediationPlanRunner.js.map +1 -1
  277. package/build/dist/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.js +330 -0
  278. package/build/dist/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.js.map +1 -0
  279. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessContext.js +160 -0
  280. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessContext.js.map +1 -0
  281. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessToolNames.js +33 -0
  282. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessToolNames.js.map +1 -0
  283. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.js +640 -0
  284. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.js.map +1 -0
  285. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.js +226 -0
  286. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.js.map +1 -0
  287. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.js +600 -0
  288. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.js.map +1 -0
  289. package/build/dist/Server/Utils/AI/SRE/AIInvestigationEngine.js +50 -4
  290. package/build/dist/Server/Utils/AI/SRE/AIInvestigationEngine.js.map +1 -1
  291. package/build/dist/Server/Utils/AI/SRE/AlertInvestigationRunner.js +55 -4
  292. package/build/dist/Server/Utils/AI/SRE/AlertInvestigationRunner.js.map +1 -1
  293. package/build/dist/Server/Utils/AI/SRE/IncidentInvestigationRunner.js +55 -4
  294. package/build/dist/Server/Utils/AI/SRE/IncidentInvestigationRunner.js.map +1 -1
  295. package/build/dist/Server/Utils/AutoRemediation/CommandPlanExecutor.js +290 -13
  296. package/build/dist/Server/Utils/AutoRemediation/CommandPlanExecutor.js.map +1 -1
  297. package/build/dist/Server/Utils/AutoRemediation/RemediationVerifier.js +29 -1
  298. package/build/dist/Server/Utils/AutoRemediation/RemediationVerifier.js.map +1 -1
  299. package/build/dist/Server/Utils/Database/ProjectScopedReferenceValidator.js +9 -1
  300. package/build/dist/Server/Utils/Database/ProjectScopedReferenceValidator.js.map +1 -1
  301. package/build/dist/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.js +627 -0
  302. package/build/dist/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.js.map +1 -0
  303. package/build/dist/Server/Utils/SessionReplay/SessionReplayUsage.js +60 -5
  304. package/build/dist/Server/Utils/SessionReplay/SessionReplayUsage.js.map +1 -1
  305. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Alert.js +19 -15
  306. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Alert.js.map +1 -1
  307. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/AlertEpisode.js +19 -15
  308. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/AlertEpisode.js.map +1 -1
  309. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Auth.js +17 -1
  310. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Auth.js.map +1 -1
  311. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Incident.js +183 -209
  312. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Incident.js.map +1 -1
  313. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/IncidentEpisode.js +19 -15
  314. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/IncidentEpisode.js.map +1 -1
  315. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/ScheduledMaintenance.js +233 -167
  316. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/ScheduledMaintenance.js.map +1 -1
  317. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeams.js +263 -110
  318. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeams.js.map +1 -1
  319. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.js +129 -0
  320. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.js.map +1 -0
  321. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.js +266 -0
  322. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.js.map +1 -0
  323. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.js +258 -0
  324. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.js.map +1 -0
  325. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.js +104 -0
  326. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.js.map +1 -0
  327. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.js +183 -0
  328. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.js.map +1 -0
  329. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.js +123 -0
  330. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.js.map +1 -0
  331. package/build/dist/Types/AI/ResourceAiAccessApi.js +30 -0
  332. package/build/dist/Types/AI/ResourceAiAccessApi.js.map +1 -0
  333. package/build/dist/Types/AI/ResourceAiAccessPermissions.js +98 -0
  334. package/build/dist/Types/AI/ResourceAiAccessPermissions.js.map +1 -0
  335. package/build/dist/Types/AutoRemediation/AiRemediationCommandPlan.js +116 -29
  336. package/build/dist/Types/AutoRemediation/AiRemediationCommandPlan.js.map +1 -1
  337. package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationCatalog.js +47 -21
  338. package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationCatalog.js.map +1 -1
  339. package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationTypes.js +5 -0
  340. package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationTypes.js.map +1 -1
  341. package/build/dist/Types/Monitor/RumAlertTemplates.js +199 -3
  342. package/build/dist/Types/Monitor/RumAlertTemplates.js.map +1 -1
  343. package/build/dist/Types/ResourceAiAgent/AiResourceType.js +242 -0
  344. package/build/dist/Types/ResourceAiAgent/AiResourceType.js.map +1 -0
  345. package/build/dist/Types/ResourceAiAgent/ResourceAiAccess.js +275 -0
  346. package/build/dist/Types/ResourceAiAgent/ResourceAiAccess.js.map +1 -0
  347. package/build/dist/Types/Rum/SessionReplayBudgetMetricType.js +45 -0
  348. package/build/dist/Types/Rum/SessionReplayBudgetMetricType.js.map +1 -0
  349. package/build/dist/Types/Runbook/RunbookStepType.js +27 -0
  350. package/build/dist/Types/Runbook/RunbookStepType.js.map +1 -1
  351. package/build/dist/Types/Telemetry/TelemetryIngestSurface.js +13 -2
  352. package/build/dist/Types/Telemetry/TelemetryIngestSurface.js.map +1 -1
  353. package/build/dist/Utils/AI/InvestigationReport.js +71 -11
  354. package/build/dist/Utils/AI/InvestigationReport.js.map +1 -1
  355. package/build/dist/Utils/AiRemediation/Resource/CephCommandPolicy.js +1462 -0
  356. package/build/dist/Utils/AiRemediation/Resource/CephCommandPolicy.js.map +1 -0
  357. package/build/dist/Utils/AiRemediation/Resource/DatabaseCommandPolicy.js +122 -0
  358. package/build/dist/Utils/AiRemediation/Resource/DatabaseCommandPolicy.js.map +1 -0
  359. package/build/dist/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.js +1139 -0
  360. package/build/dist/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.js.map +1 -0
  361. package/build/dist/Utils/AiRemediation/Resource/DatabaseQueryRedactor.js +1060 -0
  362. package/build/dist/Utils/AiRemediation/Resource/DatabaseQueryRedactor.js.map +1 -0
  363. package/build/dist/Utils/AiRemediation/Resource/DockerCliGrammar.js +1301 -0
  364. package/build/dist/Utils/AiRemediation/Resource/DockerCliGrammar.js.map +1 -0
  365. package/build/dist/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.js +370 -0
  366. package/build/dist/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.js.map +1 -0
  367. package/build/dist/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.js +498 -0
  368. package/build/dist/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.js.map +1 -0
  369. package/build/dist/Utils/AiRemediation/Resource/GovcCommandPolicy.js +1565 -0
  370. package/build/dist/Utils/AiRemediation/Resource/GovcCommandPolicy.js.map +1 -0
  371. package/build/dist/Utils/AiRemediation/Resource/HostCommandPolicy.js +2206 -0
  372. package/build/dist/Utils/AiRemediation/Resource/HostCommandPolicy.js.map +1 -0
  373. package/build/dist/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.js +1129 -0
  374. package/build/dist/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.js.map +1 -0
  375. package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicy.js +565 -0
  376. package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicy.js.map +1 -0
  377. package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.js +408 -0
  378. package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.js.map +1 -0
  379. package/build/dist/Utils/AiRemediation/Resource/ResourceOutputRedactor.js +1910 -0
  380. package/build/dist/Utils/AiRemediation/Resource/ResourceOutputRedactor.js.map +1 -0
  381. package/build/dist/Utils/SessionReplay/SessionReplayBudgetMetricType.js +160 -0
  382. package/build/dist/Utils/SessionReplay/SessionReplayBudgetMetricType.js.map +1 -0
  383. package/package.json +1 -1
@@ -0,0 +1,1530 @@
1
+ import RemediationCommandToolkit, {
2
+ INLINE_COMMAND_STEP_ID_PREFIX,
3
+ RemediationCommandNeedingApproval,
4
+ RemediationCommandToolkitOptions,
5
+ } from "../../../../Server/Utils/AI/Remediation/RemediationCommandTools";
6
+ import { ObservabilityAssistantExtraTool } from "../../../../Server/Utils/AI/Chat/ObservabilityAssistant";
7
+ import { ToolCallOutcome } from "../../../../Server/Utils/AI/Toolbox/Index";
8
+ import AIRunService from "../../../../Server/Services/AIRunService";
9
+ import AutoRemediationSuggestionService from "../../../../Server/Services/AutoRemediationSuggestionService";
10
+ import ResourceAiAccessService from "../../../../Server/Services/ResourceAiAccessService";
11
+ import RunnerJobService, {
12
+ MAX_AI_RESOURCE_COMMAND_JOBS_PER_PROJECT_PER_HOUR,
13
+ ResourceCommandEnqueueRefusedException,
14
+ } from "../../../../Server/Services/RunnerJobService";
15
+ import RunnerService from "../../../../Server/Services/RunnerService";
16
+ import { MAX_AUTO_EXECUTIONS_PER_RULE_PER_HOUR } from "../../../../Server/Services/AutoRemediationRuleEngineService";
17
+ import Semaphore, {
18
+ SemaphoreMutex,
19
+ } from "../../../../Server/Infrastructure/Semaphore";
20
+ import logger from "../../../../Server/Utils/Logger";
21
+ import AutoRemediationSuggestion from "../../../../Models/DatabaseModels/AutoRemediationSuggestion";
22
+ import RunnerJob from "../../../../Models/DatabaseModels/RunnerJob";
23
+ import AutoRemediationSuggestionStatus from "../../../../Types/AutoRemediation/AutoRemediationSuggestionStatus";
24
+ import AutoRemediationVerificationStatus from "../../../../Types/AutoRemediation/AutoRemediationVerificationStatus";
25
+ import {
26
+ AiRemediationCommand,
27
+ AiRemediationCommandExecutionStatus,
28
+ AiRemediationCommandPlan,
29
+ AiRemediationCommandPolicyVerdict,
30
+ MAX_PLAN_COMMANDS,
31
+ RESOURCE_ALWAYS_ASKS_SUMMARY,
32
+ RESOURCE_NEVER_RUNS_SUMMARY,
33
+ } from "../../../../Types/AutoRemediation/AiRemediationCommandPlan";
34
+ import AiResourceType from "../../../../Types/ResourceAiAgent/AiResourceType";
35
+ import {
36
+ MAX_RESOURCE_COMMAND_TIMEOUT_MS,
37
+ ResourceAiAccessStatus,
38
+ ResourceAiAgentPosture,
39
+ ResourceAiRemediationMode,
40
+ ResourceCommandTier,
41
+ } from "../../../../Types/ResourceAiAgent/ResourceAiAccess";
42
+ import RunbookStepType from "../../../../Types/Runbook/RunbookStepType";
43
+ import RunnerJobOrigin from "../../../../Types/Runbook/RunnerJobOrigin";
44
+ import RunnerJobStatus from "../../../../Types/Runbook/RunnerJobStatus";
45
+ import ResourceCommandPolicy from "../../../../Utils/AiRemediation/Resource/ResourceCommandPolicy";
46
+ import { RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME } from "../../../../Server/Utils/AI/ResourceAccess/ResourceAccessToolNames";
47
+ import { JSONObject } from "../../../../Types/JSON";
48
+ import ObjectID from "../../../../Types/ObjectID";
49
+ import PositiveNumber from "../../../../Types/PositiveNumber";
50
+ import OneUptimeDate from "../../../../Types/Date";
51
+ import { afterEach, beforeEach, describe, expect, it } from "@jest/globals";
52
+
53
+ /*
54
+ * Contract under test — the remediation toolkit on a RESOURCE round (a
55
+ * Docker or Podman host, a Docker Swarm, Proxmox, VMware or Ceph cluster, a
56
+ * database server, a host), mirroring what the kubectl branch guarantees on
57
+ * a cluster round:
58
+ *
59
+ * - the round offers ResourceCommand steps on its resource and nothing else:
60
+ * no Runner, no cluster, no credential; the schema, the targets list and
61
+ * the tool descriptions say so, and carry the resource's own write guide;
62
+ * - a read sent through the execute tool is refused and pointed at
63
+ * run_infrastructure_command (it would count as an executed fix);
64
+ * - Denied never runs; the ladder (evaluateForAutoExecution with the
65
+ * resource's allowlist, bypass = BypassApproval) decides what runs
66
+ * unattended, and what a human's click would let run is KEPT for the
67
+ * round's proposal — riskier changes, what always needs a human, a mode
68
+ * changed to "ask", a tripped breaker, another round holding the resource;
69
+ * - the resource is re-read LIVE before every change: fixes turned off, not
70
+ * ready, deleted or reached through another agent revoke it (and drop what
71
+ * was kept); a failed read refuses without revoking; the live mode and
72
+ * allowlist decide;
73
+ * - the agent's reported write scope (read-only, protected targets,
74
+ * ONEUPTIME_AI_WRITE_TARGETS) refuses a change, or its rollback, before it
75
+ * is composed — never kept;
76
+ * - the run's first change takes a per-resource breaker slot under the
77
+ * resource's own lock; later changes reuse it;
78
+ * - the command is enqueued through enqueueAiResourceCommand with the
79
+ * inline step id and the resource's agent, recorded BEFORE and named on
80
+ * the record before the wait, and read like every resource job: NotRun is
81
+ * off the record, Unknown stays as "may have run".
82
+ */
83
+
84
+ const PROJECT_ID: ObjectID = new ObjectID(
85
+ "22222222-2222-4222-8222-222222222222",
86
+ );
87
+ const RUN_ID: ObjectID = new ObjectID("88888888-8888-4888-8888-888888888888");
88
+ const SUGGESTION_ID: ObjectID = new ObjectID(
89
+ "77777777-7777-4777-8777-777777777777",
90
+ );
91
+ const OTHER_SUGGESTION_ID: ObjectID = new ObjectID(
92
+ "79797979-7979-4979-8979-797979797979",
93
+ );
94
+ const RESOURCE_ID: string = "33333333-3333-4333-8333-333333333333";
95
+ const AGENT_ID: string = "44444444-4444-4444-8444-444444444444";
96
+ const OTHER_AGENT_ID: string = "45454545-4545-4545-8545-454545454545";
97
+ const JOB_ID: ObjectID = new ObjectID("66666666-6666-4666-8666-666666666666");
98
+
99
+ const SAFE_WRITE: string = "docker restart web";
100
+ const SAFE_UNDO: string = "docker start web";
101
+ const RISKY_WRITE: string = "docker stop web";
102
+ const READ: string = "docker ps -a";
103
+
104
+ function posture(
105
+ overrides: Partial<ResourceAiAgentPosture> = {},
106
+ ): ResourceAiAgentPosture {
107
+ return {
108
+ resourceType: AiResourceType.DockerHost,
109
+ resourceIdentifier: "web-1",
110
+ allowWrites: true,
111
+ writeTargets: [],
112
+ protectedTargets: ["oneuptime-ai-agent"],
113
+ reachable: true,
114
+ ...overrides,
115
+ };
116
+ }
117
+
118
+ function resource(
119
+ overrides: Partial<ResourceAiAccessStatus> = {},
120
+ postureOverrides: Partial<ResourceAiAgentPosture> = {},
121
+ ): ResourceAiAccessStatus {
122
+ const resourceType: AiResourceType =
123
+ overrides.resourceType || AiResourceType.DockerHost;
124
+
125
+ return {
126
+ resourceType,
127
+ resourceId: RESOURCE_ID,
128
+ resourceName: "web-1",
129
+ isAiInvestigationEnabled: true,
130
+ aiRemediationMode: ResourceAiRemediationMode.Automatic,
131
+ aiCommandAllowlist: [],
132
+ agent: {
133
+ agentId: AGENT_ID,
134
+ connectionStatus: "connected",
135
+ isOnline: true,
136
+ posture: posture({ resourceType, ...postureOverrides }),
137
+ },
138
+ gaps: [],
139
+ isInvestigationReady: true,
140
+ isRemediationReady: true,
141
+ ...overrides,
142
+ };
143
+ }
144
+
145
+ function buildToolkit(
146
+ overrides: Partial<RemediationCommandToolkitOptions> = {},
147
+ ): RemediationCommandToolkit {
148
+ return new RemediationCommandToolkit({
149
+ projectId: PROJECT_ID,
150
+ aiRunId: RUN_ID,
151
+ suggestionId: SUGGESTION_ID,
152
+ mode: "FullAuto",
153
+ allowlistPatterns: [],
154
+ allowedRunnerIds: [],
155
+ clusterTargets: [],
156
+ resourceTargets: [resource()],
157
+ proposesRefusedCommands: true,
158
+ resourceHold: { anyOrder: false },
159
+ ...overrides,
160
+ });
161
+ }
162
+
163
+ function tool(
164
+ toolkit: RemediationCommandToolkit,
165
+ name: string,
166
+ ): ObservabilityAssistantExtraTool {
167
+ const found: ObservabilityAssistantExtraTool | undefined = toolkit
168
+ .buildTools()
169
+ .find((candidate: ObservabilityAssistantExtraTool) => {
170
+ return candidate.definition.name === name;
171
+ });
172
+ if (!found) {
173
+ throw new Error(`${name} is not offered.`);
174
+ }
175
+ return found;
176
+ }
177
+
178
+ function execute(
179
+ toolkit: RemediationCommandToolkit,
180
+ args: JSONObject,
181
+ ): Promise<ToolCallOutcome> {
182
+ return tool(toolkit, "execute_remediation_command").execute(args);
183
+ }
184
+
185
+ function resourceArgs(
186
+ overrides: Partial<Record<string, unknown>> = {},
187
+ ): JSONObject {
188
+ return {
189
+ stepType: "ResourceCommand",
190
+ resourceId: RESOURCE_ID,
191
+ command: SAFE_WRITE,
192
+ rationale: "the web container is wedged after an OOM",
193
+ expectedEffect: "web serves requests again",
194
+ ...overrides,
195
+ } as JSONObject;
196
+ }
197
+
198
+ function fakeJob(overrides: Partial<Record<string, unknown>> = {}): RunnerJob {
199
+ return {
200
+ id: JOB_ID,
201
+ _id: JOB_ID.toString(),
202
+ status: RunnerJobStatus.Succeeded,
203
+ exitCode: 0,
204
+ output: "web",
205
+ payload: { displayCommand: SAFE_WRITE, program: "docker" },
206
+ ...overrides,
207
+ } as unknown as RunnerJob;
208
+ }
209
+
210
+ let enqueue: jest.SpyInstance;
211
+ let persist: jest.SpyInstance;
212
+ let liveStatus: jest.SpyInstance;
213
+ let lock: jest.SpyInstance;
214
+ let release: jest.SpyInstance;
215
+ let poll: jest.SpyInstance;
216
+ let jobCount: jest.SpyInstance;
217
+ let suggestionCount: jest.SpyInstance;
218
+ let suggestionFindBy: jest.SpyInstance;
219
+ let recordOutcome: jest.SpyInstance;
220
+ let claimRead: jest.SpyInstance;
221
+
222
+ function mockHappyPath(): void {
223
+ jest.spyOn(logger, "error").mockImplementation((): void => {
224
+ return undefined;
225
+ });
226
+ jest.spyOn(logger, "warn").mockImplementation((): void => {
227
+ return undefined;
228
+ });
229
+ jest
230
+ .spyOn(AutoRemediationSuggestionService, "findOneById")
231
+ .mockResolvedValue({
232
+ id: SUGGESTION_ID,
233
+ _id: SUGGESTION_ID.toString(),
234
+ status: AutoRemediationSuggestionStatus.Planning,
235
+ } as unknown as AutoRemediationSuggestion);
236
+ persist = jest
237
+ .spyOn(AutoRemediationSuggestionService, "updateOneById")
238
+ .mockResolvedValue(undefined as never);
239
+ jobCount = jest
240
+ .spyOn(RunnerJobService, "countBy")
241
+ .mockResolvedValue(new PositiveNumber(0));
242
+ enqueue = jest
243
+ .spyOn(RunnerJobService, "enqueueAiResourceCommand")
244
+ .mockResolvedValue(fakeJob({ status: RunnerJobStatus.Pending }));
245
+ poll = jest
246
+ .spyOn(RunnerJobService, "pollUntilTerminal")
247
+ .mockResolvedValue(fakeJob());
248
+ claimRead = jest.spyOn(RunnerJobService, "findOneById").mockResolvedValue({
249
+ id: JOB_ID,
250
+ claimedAt: new Date(),
251
+ startedAt: new Date(),
252
+ } as unknown as RunnerJob);
253
+ jest.spyOn(AIRunService, "updateOneBy").mockResolvedValue(undefined as never);
254
+ recordOutcome = jest
255
+ .spyOn(ResourceAiAccessService, "recordCommandOutcome")
256
+ .mockResolvedValue(undefined);
257
+ jest
258
+ .spyOn(RunnerService, "getOnlineAiCommandRunnersForProject")
259
+ .mockResolvedValue([]);
260
+ liveStatus = jest
261
+ .spyOn(ResourceAiAccessService, "getStatusForResource")
262
+ .mockResolvedValue(resource());
263
+ lock = jest
264
+ .spyOn(Semaphore, "lock")
265
+ .mockResolvedValue({ id: "resource-lock" } as unknown as SemaphoreMutex);
266
+ release = jest.spyOn(Semaphore, "release").mockResolvedValue(undefined);
267
+ suggestionCount = jest
268
+ .spyOn(AutoRemediationSuggestionService, "countBy")
269
+ .mockResolvedValue(new PositiveNumber(0));
270
+ suggestionFindBy = jest
271
+ .spyOn(AutoRemediationSuggestionService, "findBy")
272
+ .mockResolvedValue([]);
273
+ jest.spyOn(RunnerJobService, "findBy").mockResolvedValue([]);
274
+ }
275
+
276
+ function expectNothingRanOrRecorded(
277
+ toolkit: RemediationCommandToolkit,
278
+ outcome: ToolCallOutcome,
279
+ ): void {
280
+ expect(outcome.success).toBe(false);
281
+ expect(enqueue).not.toHaveBeenCalled();
282
+ expect(persist).not.toHaveBeenCalled();
283
+ expect(toolkit.getExecutedCommands()).toHaveLength(0);
284
+ }
285
+
286
+ function kept(
287
+ toolkit: RemediationCommandToolkit,
288
+ ): Array<RemediationCommandNeedingApproval> {
289
+ return toolkit.getCommandsNeedingApproval();
290
+ }
291
+
292
+ beforeEach(() => {
293
+ mockHappyPath();
294
+ });
295
+
296
+ afterEach(() => {
297
+ jest.restoreAllMocks();
298
+ });
299
+
300
+ describe("the tools a resource round offers", () => {
301
+ it("offers the targets list and the execute tool in FullAuto — ResourceCommand on the resource only", () => {
302
+ const toolkit: RemediationCommandToolkit = buildToolkit();
303
+
304
+ expect(toolkit.isResourceRound()).toBe(true);
305
+ expect(
306
+ toolkit.buildTools().map((candidate: ObservabilityAssistantExtraTool) => {
307
+ return candidate.definition.name;
308
+ }),
309
+ ).toEqual(["list_command_targets", "execute_remediation_command"]);
310
+
311
+ const executeTool: ObservabilityAssistantExtraTool = tool(
312
+ toolkit,
313
+ "execute_remediation_command",
314
+ );
315
+ const schema: JSONObject = executeTool.definition.inputSchema as JSONObject;
316
+ const properties: JSONObject = schema["properties"] as JSONObject;
317
+
318
+ expect((properties["stepType"] as JSONObject)["enum"]).toEqual([
319
+ "ResourceCommand",
320
+ ]);
321
+ expect(Object.keys(properties).sort()).toEqual(
322
+ [
323
+ "command",
324
+ "expectedEffect",
325
+ "rationale",
326
+ "resourceId",
327
+ "rollbackCommand",
328
+ "stepType",
329
+ "timeoutInMs",
330
+ ].sort(),
331
+ );
332
+ expect(schema["required"]).toContain("resourceId");
333
+
334
+ const description: string = executeTool.definition.description;
335
+ expect(description).toContain("stepType ResourceCommand");
336
+ expect(description).toContain(RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME);
337
+ expect(description).toContain(RESOURCE_ALWAYS_ASKS_SUMMARY);
338
+ expect(description).toContain(RESOURCE_NEVER_RUNS_SUMMARY);
339
+ expect(description).toContain(
340
+ ResourceCommandPolicy.getWriteCommandGuide(AiResourceType.DockerHost),
341
+ );
342
+ expect(description).toContain(`resourceId: ${RESOURCE_ID}`);
343
+ expect(description).not.toContain("kubectl");
344
+
345
+ expect(
346
+ tool(toolkit, "list_command_targets").definition.description,
347
+ ).toContain("the infrastructure resource this round is about");
348
+ });
349
+
350
+ it("offers the propose tool in Suggest, with the same ResourceCommand schema and the write guide", () => {
351
+ const toolkit: RemediationCommandToolkit = buildToolkit({
352
+ mode: "Suggest",
353
+ });
354
+
355
+ const proposeTool: ObservabilityAssistantExtraTool = tool(
356
+ toolkit,
357
+ "propose_remediation_commands",
358
+ );
359
+ const items: JSONObject = (
360
+ (proposeTool.definition.inputSchema as JSONObject)[
361
+ "properties"
362
+ ] as JSONObject
363
+ )["commands"] as JSONObject;
364
+ const itemSchema: JSONObject = items["items"] as JSONObject;
365
+
366
+ expect(
367
+ ((itemSchema["properties"] as JSONObject)["stepType"] as JSONObject)[
368
+ "enum"
369
+ ],
370
+ ).toEqual(["ResourceCommand"]);
371
+ expect(itemSchema["required"]).toContain("resourceId");
372
+ expect(proposeTool.definition.description).toContain(
373
+ ResourceCommandPolicy.getWriteCommandGuide(AiResourceType.DockerHost),
374
+ );
375
+ expect(proposeTool.definition.description).toContain(
376
+ RESOURCE_NEVER_RUNS_SUMMARY,
377
+ );
378
+ });
379
+
380
+ it("a round without resource targets is not a resource round, and its schema is unchanged", () => {
381
+ const toolkit: RemediationCommandToolkit = buildToolkit({
382
+ resourceTargets: undefined,
383
+ });
384
+
385
+ expect(toolkit.isResourceRound()).toBe(false);
386
+ const properties: JSONObject = (
387
+ tool(toolkit, "execute_remediation_command").definition
388
+ .inputSchema as JSONObject
389
+ )["properties"] as JSONObject;
390
+ expect((properties["stepType"] as JSONObject)["enum"]).toEqual([
391
+ "Bash",
392
+ "SSH",
393
+ "Kubectl",
394
+ ]);
395
+ expect(properties["resourceId"]).toBeUndefined();
396
+ });
397
+
398
+ /*
399
+ * A Kubernetes or rule round never offered ResourceCommand: to it the
400
+ * step type is as unknown as before the resource lane existed, and its
401
+ * refusal lists exactly the step types it offers — the words it always
402
+ * had ("stepType must be one of: Bash, SSH, Kubectl.").
403
+ */
404
+ it("a non-resource round refuses a ResourceCommand step with the original words", async () => {
405
+ const toolkit: RemediationCommandToolkit = buildToolkit({
406
+ resourceTargets: [],
407
+ });
408
+
409
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
410
+
411
+ expect(outcome.textForLlm).toBe(
412
+ "stepType must be one of: Bash, SSH, Kubectl.",
413
+ );
414
+ expectNothingRanOrRecorded(toolkit, outcome);
415
+ });
416
+
417
+ it("a non-resource round refuses an unknown step type naming only Bash, SSH and Kubectl — never ResourceCommand", async () => {
418
+ const toolkit: RemediationCommandToolkit = buildToolkit({
419
+ resourceTargets: [],
420
+ });
421
+
422
+ const outcome: ToolCallOutcome = await execute(
423
+ toolkit,
424
+ resourceArgs({ stepType: "kubectl" }),
425
+ );
426
+
427
+ expect(outcome.textForLlm).toBe(
428
+ "stepType must be one of: Bash, SSH, Kubectl.",
429
+ );
430
+ expectNothingRanOrRecorded(toolkit, outcome);
431
+ });
432
+
433
+ it("a resource round refuses an unknown step type naming only ResourceCommand", async () => {
434
+ const toolkit: RemediationCommandToolkit = buildToolkit();
435
+
436
+ const outcome: ToolCallOutcome = await execute(
437
+ toolkit,
438
+ resourceArgs({ stepType: "Docker" }),
439
+ );
440
+
441
+ expect(outcome.textForLlm).toBe(
442
+ "stepType must be one of: ResourceCommand.",
443
+ );
444
+ expectNothingRanOrRecorded(toolkit, outcome);
445
+ });
446
+ });
447
+
448
+ describe("list_command_targets on a resource round", () => {
449
+ it("lists the resource, its programs, mode, write scope and allowlist — and no Runner", async () => {
450
+ const toolkit: RemediationCommandToolkit = buildToolkit({
451
+ resourceTargets: [resource({ aiCommandAllowlist: ["docker stop web"] })],
452
+ });
453
+
454
+ const outcome: ToolCallOutcome = await tool(
455
+ toolkit,
456
+ "list_command_targets",
457
+ ).execute({});
458
+
459
+ expect(outcome.success).toBe(true);
460
+ expect(outcome.result?.rowCount).toBe(1);
461
+ expect(outcome.textForLlm).toContain("Resource");
462
+ expect(outcome.textForLlm).toContain(RESOURCE_ID);
463
+ expect(outcome.textForLlm).toContain("ResourceCommand");
464
+ expect(outcome.textForLlm).toContain("its Docker AI agent");
465
+ expect(outcome.textForLlm).toContain("Automatic:");
466
+ expect(outcome.textForLlm).toContain("docker stop web");
467
+ expect(outcome.textForLlm).toContain("oneuptime-ai-agent");
468
+ expect(
469
+ RunnerService.getOnlineAiCommandRunnersForProject,
470
+ ).not.toHaveBeenCalled();
471
+ });
472
+
473
+ it("says so when the resource no longer allows remediation", async () => {
474
+ const toolkit: RemediationCommandToolkit = buildToolkit({
475
+ resourceTargets: [resource({ isRemediationReady: false })],
476
+ });
477
+
478
+ const outcome: ToolCallOutcome = await tool(
479
+ toolkit,
480
+ "list_command_targets",
481
+ ).execute({});
482
+
483
+ expect(outcome.result?.rowCount).toBe(0);
484
+ expect(outcome.textForLlm).toContain("no longer allows AI remediation");
485
+ });
486
+ });
487
+
488
+ describe("execute_remediation_command on a resource round", () => {
489
+ it("runs a safe change through the resource chokepoint, recorded before it runs", async () => {
490
+ const toolkit: RemediationCommandToolkit = buildToolkit();
491
+
492
+ const outcome: ToolCallOutcome = await execute(
493
+ toolkit,
494
+ resourceArgs({ rollbackCommand: SAFE_UNDO, timeoutInMs: 45000 }),
495
+ );
496
+
497
+ expect(outcome.success).toBe(true);
498
+ expect(enqueue).toHaveBeenCalledTimes(1);
499
+ expect(enqueue.mock.calls[0]![0]).toEqual({
500
+ projectId: PROJECT_ID,
501
+ aiRunId: RUN_ID,
502
+ origin: RunnerJobOrigin.AiRemediation,
503
+ autoRemediationSuggestionId: SUGGESTION_ID,
504
+ resourceType: AiResourceType.DockerHost,
505
+ resourceId: new ObjectID(RESOURCE_ID),
506
+ stepId: `${INLINE_COMMAND_STEP_ID_PREFIX}1`,
507
+ targetResourceAiAgentId: new ObjectID(AGENT_ID),
508
+ command: SAFE_WRITE,
509
+ timeoutInMs: 45000,
510
+ claimTimeoutInMs: 60000,
511
+ });
512
+
513
+ // The durable record lands before the job exists.
514
+ expect(persist.mock.invocationCallOrder[0]!).toBeLessThan(
515
+ enqueue.mock.invocationCallOrder[0]!,
516
+ );
517
+
518
+ const executed: AiRemediationCommand = toolkit.getExecutedCommands()[0]!;
519
+ expect(executed).toMatchObject({
520
+ sequence: 1,
521
+ stepType: RunbookStepType.ResourceCommand,
522
+ runnerId: AGENT_ID,
523
+ runnerNameSnapshot: "Docker AI agent",
524
+ resourceType: AiResourceType.DockerHost,
525
+ resourceId: RESOURCE_ID,
526
+ resourceNameSnapshot: "web-1",
527
+ resourceCommandTier: ResourceCommandTier.SafeWrite,
528
+ command: SAFE_WRITE,
529
+ rollbackCommand: SAFE_UNDO,
530
+ policyVerdict: AiRemediationCommandPolicyVerdict.AutoApproved,
531
+ wasAutoExecuted: true,
532
+ });
533
+ expect(executed.credentialId).toBeUndefined();
534
+ expect(executed.kubernetesClusterId).toBeUndefined();
535
+ expect(executed.execution).toMatchObject({
536
+ status: AiRemediationCommandExecutionStatus.Succeeded,
537
+ runnerJobId: JOB_ID.toString(),
538
+ exitCode: 0,
539
+ });
540
+
541
+ expect(outcome.result?.citationLabel).toBe(
542
+ 'Executed on Docker host "web-1": docker restart web',
543
+ );
544
+ expect(outcome.textForLlm).toContain(
545
+ '<tool_result source="untrusted_resource_output">',
546
+ );
547
+ expect(recordOutcome).toHaveBeenCalledWith({
548
+ resourceType: AiResourceType.DockerHost,
549
+ resourceId: new ObjectID(RESOURCE_ID),
550
+ succeeded: true,
551
+ errorMessage: undefined,
552
+ });
553
+
554
+ // The job id was on the record before the wait began.
555
+ const recorded: Array<JSONObject> = persist.mock.calls.map(
556
+ (call: Array<unknown>): JSONObject => {
557
+ return (call[0] as { data: { commandPlan: JSONObject } }).data
558
+ .commandPlan;
559
+ },
560
+ );
561
+ expect(
562
+ recorded.some((plan: JSONObject): boolean => {
563
+ const commands: Array<JSONObject> = plan[
564
+ "commands"
565
+ ] as Array<JSONObject>;
566
+ return (
567
+ (commands[0]!["execution"] as JSONObject)["runnerJobId"] ===
568
+ JOB_ID.toString() &&
569
+ (commands[0]!["execution"] as JSONObject)["status"] ===
570
+ AiRemediationCommandExecutionStatus.Pending
571
+ );
572
+ }),
573
+ ).toBe(true);
574
+ });
575
+
576
+ it("takes the resource's own breaker lock for its first change, releases it once the job exists, and not again for the next change", async () => {
577
+ const toolkit: RemediationCommandToolkit = buildToolkit();
578
+
579
+ await execute(toolkit, resourceArgs());
580
+
581
+ expect(lock).toHaveBeenCalledTimes(1);
582
+ expect(lock.mock.calls[0]![0]).toMatchObject({
583
+ namespace: "AutoRemediationResourceBreaker",
584
+ key: `DockerHost:${RESOURCE_ID}`,
585
+ });
586
+ expect(release).toHaveBeenCalledTimes(1);
587
+ expect(release.mock.invocationCallOrder[0]!).toBeGreaterThan(
588
+ enqueue.mock.invocationCallOrder[0]!,
589
+ );
590
+ expect(release.mock.invocationCallOrder[0]!).toBeLessThan(
591
+ poll.mock.invocationCallOrder[0]!,
592
+ );
593
+
594
+ await execute(toolkit, resourceArgs({ command: "docker restart api" }));
595
+
596
+ expect(lock).toHaveBeenCalledTimes(1);
597
+ expect(enqueue).toHaveBeenCalledTimes(2);
598
+ expect(enqueue.mock.calls[1]![0]).toMatchObject({
599
+ stepId: `${INLINE_COMMAND_STEP_ID_PREFIX}2`,
600
+ });
601
+ });
602
+
603
+ it("refuses a read and points at run_infrastructure_command", async () => {
604
+ const toolkit: RemediationCommandToolkit = buildToolkit();
605
+
606
+ const outcome: ToolCallOutcome = await execute(
607
+ toolkit,
608
+ resourceArgs({ command: READ }),
609
+ );
610
+
611
+ expect(outcome.textForLlm).toContain("is read-only");
612
+ expect(outcome.textForLlm).toContain(RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME);
613
+ expectNothingRanOrRecorded(toolkit, outcome);
614
+ expect(kept(toolkit)).toHaveLength(0);
615
+ });
616
+
617
+ it.each([
618
+ ["docker exec web sh"],
619
+ ["docker rm -f web"],
620
+ ["docker restart web; rm -rf /"],
621
+ ["sudo docker restart web"],
622
+ ["systemctl restart nginx"],
623
+ ])("refuses a denied command outright: %s", async (command: string) => {
624
+ const toolkit: RemediationCommandToolkit = buildToolkit();
625
+
626
+ const outcome: ToolCallOutcome = await execute(
627
+ toolkit,
628
+ resourceArgs({ command }),
629
+ );
630
+
631
+ expect(outcome.textForLlm).toContain(
632
+ "Denied by the Docker host command policy",
633
+ );
634
+ expect(outcome.textForLlm).toContain("can never run");
635
+ expectNothingRanOrRecorded(toolkit, outcome);
636
+ expect(kept(toolkit)).toHaveLength(0);
637
+ });
638
+
639
+ it("refuses a step that names a credential: the agent never gets one", async () => {
640
+ const toolkit: RemediationCommandToolkit = buildToolkit();
641
+
642
+ const outcome: ToolCallOutcome = await execute(
643
+ toolkit,
644
+ resourceArgs({ credentialId: "55555555-5555-4555-8555-555555555555" }),
645
+ );
646
+
647
+ expect(outcome.textForLlm).toContain("never carries a credential");
648
+ expectNothingRanOrRecorded(toolkit, outcome);
649
+ });
650
+
651
+ it.each([
652
+ ["an unknown resourceId", { resourceId: OTHER_AGENT_ID }],
653
+ ["no resourceId", { resourceId: undefined }],
654
+ ])(
655
+ "refuses %s",
656
+ async (_label: string, overrides: Record<string, unknown>) => {
657
+ const toolkit: RemediationCommandToolkit = buildToolkit();
658
+
659
+ const outcome: ToolCallOutcome = await execute(
660
+ toolkit,
661
+ resourceArgs(overrides),
662
+ );
663
+
664
+ expect(outcome.textForLlm).toContain("resourceId is required");
665
+ expectNothingRanOrRecorded(toolkit, outcome);
666
+ },
667
+ );
668
+
669
+ it("matches the resourceId case-insensitively", async () => {
670
+ const toolkit: RemediationCommandToolkit = buildToolkit();
671
+
672
+ const outcome: ToolCallOutcome = await execute(
673
+ toolkit,
674
+ resourceArgs({ resourceId: RESOURCE_ID.toUpperCase() }),
675
+ );
676
+
677
+ expect(outcome.success).toBe(true);
678
+ expect(enqueue).toHaveBeenCalledTimes(1);
679
+ });
680
+
681
+ it.each([["Bash"], ["SSH"], ["Kubectl"]])(
682
+ "refuses a %s step: a resource round runs resource commands only",
683
+ async (stepType: string) => {
684
+ const toolkit: RemediationCommandToolkit = buildToolkit();
685
+
686
+ const outcome: ToolCallOutcome = await execute(
687
+ toolkit,
688
+ resourceArgs({ stepType, runnerId: AGENT_ID }),
689
+ );
690
+
691
+ expect(outcome.textForLlm).toContain("use stepType ResourceCommand");
692
+ expectNothingRanOrRecorded(toolkit, outcome);
693
+ },
694
+ );
695
+
696
+ it("refuses a riskier change on an Automatic resource and KEEPS it for the proposal", async () => {
697
+ const toolkit: RemediationCommandToolkit = buildToolkit();
698
+
699
+ const outcome: ToolCallOutcome = await execute(
700
+ toolkit,
701
+ resourceArgs({ command: RISKY_WRITE, rollbackCommand: SAFE_UNDO }),
702
+ );
703
+
704
+ expect(outcome.textForLlm).toContain("The command was NOT executed");
705
+ expect(outcome.textForLlm).toContain(
706
+ "OneUptime AI proposes it for one-click approval",
707
+ );
708
+ expectNothingRanOrRecorded(toolkit, outcome);
709
+
710
+ const keptCommands: Array<RemediationCommandNeedingApproval> =
711
+ kept(toolkit);
712
+ expect(keptCommands).toHaveLength(1);
713
+ expect(keptCommands[0]!.reason).toContain("riskier change");
714
+ expect(keptCommands[0]!.command).toMatchObject({
715
+ stepType: RunbookStepType.ResourceCommand,
716
+ command: RISKY_WRITE,
717
+ resourceCommandTier: ResourceCommandTier.RiskyWrite,
718
+ policyVerdict: AiRemediationCommandPolicyVerdict.RequiresApproval,
719
+ wasAutoExecuted: false,
720
+ });
721
+
722
+ // The same change twice is kept once.
723
+ await execute(
724
+ toolkit,
725
+ resourceArgs({ command: RISKY_WRITE, rollbackCommand: SAFE_UNDO }),
726
+ );
727
+ expect(kept(toolkit)).toHaveLength(1);
728
+ });
729
+
730
+ it("runs a riskier change the resource's allowlist names", async () => {
731
+ liveStatus.mockResolvedValue(
732
+ resource({ aiCommandAllowlist: ["docker stop *"] }),
733
+ );
734
+ const toolkit: RemediationCommandToolkit = buildToolkit({
735
+ resourceTargets: [resource({ aiCommandAllowlist: ["docker stop *"] })],
736
+ });
737
+
738
+ const outcome: ToolCallOutcome = await execute(
739
+ toolkit,
740
+ resourceArgs({ command: RISKY_WRITE }),
741
+ );
742
+
743
+ expect(outcome.success).toBe(true);
744
+ expect(enqueue).toHaveBeenCalledTimes(1);
745
+ expect(toolkit.getExecutedCommands()[0]!.policyVerdict).toBe(
746
+ AiRemediationCommandPolicyVerdict.AutoApproved,
747
+ );
748
+ });
749
+
750
+ it("uses the LIVE allowlist, not the run's snapshot", async () => {
751
+ liveStatus.mockResolvedValue(
752
+ resource({ aiCommandAllowlist: ["docker stop web"] }),
753
+ );
754
+ const toolkit: RemediationCommandToolkit = buildToolkit();
755
+
756
+ const outcome: ToolCallOutcome = await execute(
757
+ toolkit,
758
+ resourceArgs({ command: RISKY_WRITE }),
759
+ );
760
+
761
+ expect(outcome.success).toBe(true);
762
+ expect(enqueue).toHaveBeenCalledTimes(1);
763
+ });
764
+
765
+ it("runs a riskier change on a BypassApproval resource, rollback included", async () => {
766
+ const bypass: ResourceAiAccessStatus = resource({
767
+ aiRemediationMode: ResourceAiRemediationMode.BypassApproval,
768
+ });
769
+ liveStatus.mockResolvedValue(bypass);
770
+ const toolkit: RemediationCommandToolkit = buildToolkit({
771
+ resourceTargets: [bypass],
772
+ });
773
+
774
+ const outcome: ToolCallOutcome = await execute(
775
+ toolkit,
776
+ resourceArgs({
777
+ command: "docker start web",
778
+ rollbackCommand: RISKY_WRITE,
779
+ }),
780
+ );
781
+
782
+ expect(outcome.success).toBe(true);
783
+ expect(enqueue).toHaveBeenCalledTimes(1);
784
+ expect(toolkit.getExecutedCommands()[0]!.rollbackCommand).toBe(RISKY_WRITE);
785
+ });
786
+
787
+ it("keeps a change that always needs a human even on a bypassed, allowlisted resource", async () => {
788
+ const host: ResourceAiAccessStatus = resource({
789
+ resourceType: AiResourceType.Host,
790
+ aiRemediationMode: ResourceAiRemediationMode.BypassApproval,
791
+ aiCommandAllowlist: ["kill -TERM 1234"],
792
+ });
793
+ liveStatus.mockResolvedValue(host);
794
+ const toolkit: RemediationCommandToolkit = buildToolkit({
795
+ resourceTargets: [host],
796
+ });
797
+
798
+ const outcome: ToolCallOutcome = await execute(
799
+ toolkit,
800
+ resourceArgs({ command: "kill -TERM 1234" }),
801
+ );
802
+
803
+ expect(outcome.textForLlm).toContain("always needs a human, in every mode");
804
+ expectNothingRanOrRecorded(toolkit, outcome);
805
+ expect(kept(toolkit)).toHaveLength(1);
806
+ expect(kept(toolkit)[0]!.reason).toContain("always needs a human");
807
+ });
808
+
809
+ it("keeps every change on a resource that asks for approval", async () => {
810
+ const askFirst: ResourceAiAccessStatus = resource({
811
+ aiRemediationMode: ResourceAiRemediationMode.RequireApproval,
812
+ });
813
+ liveStatus.mockResolvedValue(askFirst);
814
+ const toolkit: RemediationCommandToolkit = buildToolkit({
815
+ resourceTargets: [askFirst],
816
+ });
817
+
818
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
819
+
820
+ expect(outcome.textForLlm).toContain(
821
+ 'Docker host "web-1" requires human approval for every change',
822
+ );
823
+ expectNothingRanOrRecorded(toolkit, outcome);
824
+ expect(kept(toolkit)[0]!.reason).toBe(
825
+ 'Docker host "web-1" asks for approval of every change',
826
+ );
827
+ });
828
+
829
+ it("keeps a change when the operator moved the resource to ask for approval mid-run", async () => {
830
+ liveStatus.mockResolvedValue(
831
+ resource({
832
+ aiRemediationMode: ResourceAiRemediationMode.RequireApproval,
833
+ }),
834
+ );
835
+ const toolkit: RemediationCommandToolkit = buildToolkit();
836
+
837
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
838
+
839
+ expect(outcome.textForLlm).toContain(
840
+ "was changed to ask for approval during this run",
841
+ );
842
+ expectNothingRanOrRecorded(toolkit, outcome);
843
+ expect(kept(toolkit)[0]!.reason).toContain(
844
+ "was changed to ask for approval during the round",
845
+ );
846
+ });
847
+
848
+ /*
849
+ * Automatic runs only a signal's FIRST round unattended. A follow-up that
850
+ * started under Bypass approval must stop running changes once the
851
+ * operator moves the resource to Automatic mid-run — Automatic is still
852
+ * an "unattended" mode, but not for round 2.
853
+ */
854
+ it("keeps a change on a follow-up round when the operator moved the resource from Bypass approval to Automatic mid-run", async () => {
855
+ liveStatus.mockResolvedValue(
856
+ resource({ aiRemediationMode: ResourceAiRemediationMode.Automatic }),
857
+ );
858
+ const toolkit: RemediationCommandToolkit = buildToolkit({
859
+ resourceTargets: [
860
+ resource({
861
+ aiRemediationMode: ResourceAiRemediationMode.BypassApproval,
862
+ }),
863
+ ],
864
+ resourceRoundNumber: 2,
865
+ });
866
+
867
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
868
+
869
+ expect(outcome.textForLlm).toContain(
870
+ "was changed to Automatic during this run, and Automatic asks for approval of every change after a signal's first round (this is round 2)",
871
+ );
872
+ expectNothingRanOrRecorded(toolkit, outcome);
873
+ expect(kept(toolkit)[0]!.reason).toContain(
874
+ "was changed to Automatic during the round",
875
+ );
876
+ });
877
+
878
+ it("negative control: the same move on round 1 keeps running safe changes (Automatic runs round 1 unattended)", async () => {
879
+ liveStatus.mockResolvedValue(
880
+ resource({ aiRemediationMode: ResourceAiRemediationMode.Automatic }),
881
+ );
882
+ const toolkit: RemediationCommandToolkit = buildToolkit({
883
+ resourceTargets: [
884
+ resource({
885
+ aiRemediationMode: ResourceAiRemediationMode.BypassApproval,
886
+ }),
887
+ ],
888
+ resourceRoundNumber: 1,
889
+ });
890
+
891
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
892
+
893
+ expect(outcome.success).toBe(true);
894
+ expect(enqueue).toHaveBeenCalledTimes(1);
895
+ });
896
+
897
+ it("negative control: a follow-up round on a resource still on Bypass approval keeps running", async () => {
898
+ const bypass: ResourceAiAccessStatus = resource({
899
+ aiRemediationMode: ResourceAiRemediationMode.BypassApproval,
900
+ });
901
+ liveStatus.mockResolvedValue(bypass);
902
+ const toolkit: RemediationCommandToolkit = buildToolkit({
903
+ resourceTargets: [bypass],
904
+ resourceRoundNumber: 2,
905
+ });
906
+
907
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
908
+
909
+ expect(outcome.success).toBe(true);
910
+ expect(enqueue).toHaveBeenCalledTimes(1);
911
+ });
912
+
913
+ it.each([
914
+ [
915
+ "fixes turned off",
916
+ resource({
917
+ isRemediationReady: false,
918
+ aiRemediationMode: ResourceAiRemediationMode.Disabled,
919
+ gaps: [
920
+ {
921
+ code: "remediation_disabled",
922
+ title: "AI fixes are turned off for this Docker host",
923
+ nextStep: "Turn them on.",
924
+ blocksInvestigation: false,
925
+ blocksRemediation: true,
926
+ },
927
+ ],
928
+ }),
929
+ "no longer allows AI remediation (AI fixes are turned off for this Docker host)",
930
+ ],
931
+ [
932
+ "the agent was replaced",
933
+ resource({
934
+ agent: {
935
+ agentId: OTHER_AGENT_ID,
936
+ connectionStatus: "connected",
937
+ isOnline: true,
938
+ posture: posture(),
939
+ },
940
+ }),
941
+ "is no longer reached through the AI agent this run started with",
942
+ ],
943
+ ["the resource was deleted", null, "no longer exists in this project"],
944
+ ])(
945
+ "revokes the resource when %s — and drops what was kept for it",
946
+ async (
947
+ _label: string,
948
+ live: ResourceAiAccessStatus | null,
949
+ expected: string,
950
+ ) => {
951
+ const toolkit: RemediationCommandToolkit = buildToolkit();
952
+
953
+ // Kept while the resource still allowed remediation.
954
+ await execute(toolkit, resourceArgs({ command: RISKY_WRITE }));
955
+ expect(kept(toolkit)).toHaveLength(1);
956
+
957
+ liveStatus.mockResolvedValue(live);
958
+
959
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
960
+
961
+ expect(outcome.textForLlm).toContain(expected);
962
+ expect(outcome.textForLlm).toContain(
963
+ "Do NOT run any further command on this Docker host",
964
+ );
965
+ expectNothingRanOrRecorded(toolkit, outcome);
966
+ expect(kept(toolkit)).toHaveLength(0);
967
+ expect(toolkit.getResourceTargets()).toHaveLength(0);
968
+ },
969
+ );
970
+
971
+ it("refuses without revoking when the live status cannot be read", async () => {
972
+ liveStatus.mockRejectedValue(new Error("db down"));
973
+ const toolkit: RemediationCommandToolkit = buildToolkit();
974
+
975
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
976
+
977
+ expect(outcome.textForLlm).toContain(
978
+ 'Could not confirm that Docker host "web-1" still allows AI remediation',
979
+ );
980
+ expectNothingRanOrRecorded(toolkit, outcome);
981
+ expect(kept(toolkit)).toHaveLength(0);
982
+ expect(toolkit.getResourceTargets()).toHaveLength(1);
983
+ });
984
+
985
+ it.each([
986
+ [
987
+ "a protected target",
988
+ { protectedTargets: ["web"] },
989
+ "which the Docker AI agent protects",
990
+ ],
991
+ [
992
+ "a target outside ONEUPTIME_AI_WRITE_TARGETS",
993
+ { writeTargets: ["api*"] },
994
+ "outside the targets the Docker AI agent may change (ONEUPTIME_AI_WRITE_TARGETS=api*)",
995
+ ],
996
+ ])(
997
+ "refuses — never keeps — a change to %s",
998
+ async (
999
+ _label: string,
1000
+ postureOverrides: Partial<ResourceAiAgentPosture>,
1001
+ expected: string,
1002
+ ) => {
1003
+ const scoped: ResourceAiAccessStatus = resource({}, postureOverrides);
1004
+ liveStatus.mockResolvedValue(scoped);
1005
+ const toolkit: RemediationCommandToolkit = buildToolkit({
1006
+ resourceTargets: [scoped],
1007
+ });
1008
+
1009
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1010
+
1011
+ expect(outcome.textForLlm).toContain(expected);
1012
+ expect(outcome.textForLlm).toContain(
1013
+ "The command was neither run nor recorded.",
1014
+ );
1015
+ expectNothingRanOrRecorded(toolkit, outcome);
1016
+ expect(kept(toolkit)).toHaveLength(0);
1017
+ },
1018
+ );
1019
+
1020
+ it("refuses when the agent's LIVE posture narrowed the write scope mid-run", async () => {
1021
+ liveStatus.mockResolvedValue(resource({}, { protectedTargets: ["web"] }));
1022
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1023
+
1024
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1025
+
1026
+ expect(outcome.textForLlm).toContain("which the Docker AI agent protects");
1027
+ expectNothingRanOrRecorded(toolkit, outcome);
1028
+ });
1029
+
1030
+ it("refuses a change whose rollback the agent would refuse", async () => {
1031
+ const scoped: ResourceAiAccessStatus = resource(
1032
+ {},
1033
+ { writeTargets: ["web"] },
1034
+ );
1035
+ liveStatus.mockResolvedValue(scoped);
1036
+ const toolkit: RemediationCommandToolkit = buildToolkit({
1037
+ resourceTargets: [scoped],
1038
+ });
1039
+
1040
+ const outcome: ToolCallOutcome = await execute(
1041
+ toolkit,
1042
+ resourceArgs({ rollbackCommand: "docker start api" }),
1043
+ );
1044
+
1045
+ expect(outcome.textForLlm).toContain(
1046
+ "The rollbackCommand would be refused when it has to run",
1047
+ );
1048
+ expectNothingRanOrRecorded(toolkit, outcome);
1049
+ });
1050
+
1051
+ it.each([
1052
+ [
1053
+ "a riskier rollback on an Automatic resource",
1054
+ { command: SAFE_UNDO, rollbackCommand: RISKY_WRITE },
1055
+ "is a risky change",
1056
+ ],
1057
+ [
1058
+ "a denied rollback",
1059
+ { rollbackCommand: "docker rm -f web" },
1060
+ "The rollbackCommand is denied",
1061
+ ],
1062
+ ])(
1063
+ "refuses %s",
1064
+ async (
1065
+ _label: string,
1066
+ overrides: Record<string, unknown>,
1067
+ expected: string,
1068
+ ) => {
1069
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1070
+
1071
+ const outcome: ToolCallOutcome = await execute(
1072
+ toolkit,
1073
+ resourceArgs(overrides),
1074
+ );
1075
+
1076
+ expect(outcome.textForLlm).toContain(expected);
1077
+ expectNothingRanOrRecorded(toolkit, outcome);
1078
+ },
1079
+ );
1080
+
1081
+ it("refuses a rollback that always needs a human, even on a bypassed resource", async () => {
1082
+ const host: ResourceAiAccessStatus = resource({
1083
+ resourceType: AiResourceType.Host,
1084
+ aiRemediationMode: ResourceAiRemediationMode.BypassApproval,
1085
+ });
1086
+ liveStatus.mockResolvedValue(host);
1087
+ const toolkit: RemediationCommandToolkit = buildToolkit({
1088
+ resourceTargets: [host],
1089
+ });
1090
+
1091
+ const outcome: ToolCallOutcome = await execute(
1092
+ toolkit,
1093
+ resourceArgs({
1094
+ command: "systemctl restart nginx",
1095
+ rollbackCommand: "kill -TERM 1234",
1096
+ }),
1097
+ );
1098
+
1099
+ expect(outcome.textForLlm).toContain("always needs a human");
1100
+ expect(outcome.textForLlm).toContain("rollbacks run unattended");
1101
+ expectNothingRanOrRecorded(toolkit, outcome);
1102
+ });
1103
+
1104
+ it("keeps the change when the resource's hourly breaker has tripped", async () => {
1105
+ suggestionCount.mockResolvedValue(
1106
+ new PositiveNumber(MAX_AUTO_EXECUTIONS_PER_RULE_PER_HOUR),
1107
+ );
1108
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1109
+
1110
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1111
+
1112
+ expect(outcome.textForLlm).toContain(
1113
+ 'The hourly circuit breaker for Docker host "web-1" tripped',
1114
+ );
1115
+ expect(enqueue).not.toHaveBeenCalled();
1116
+ expect(release).toHaveBeenCalledTimes(1);
1117
+ expect(kept(toolkit)[0]!.reason).toContain("hourly circuit breaker");
1118
+ });
1119
+
1120
+ it("keeps the change when the breaker lock cannot be had", async () => {
1121
+ lock.mockRejectedValue(new Error("redis down"));
1122
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1123
+
1124
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1125
+
1126
+ expect(outcome.textForLlm).toContain(
1127
+ 'Could not check the hourly limit on unattended AI fixes for Docker host "web-1"',
1128
+ );
1129
+ expect(enqueue).not.toHaveBeenCalled();
1130
+ expect(kept(toolkit)[0]!.reason).toContain("could not be checked");
1131
+ });
1132
+
1133
+ it("keeps the change when another round holds the resource", async () => {
1134
+ suggestionFindBy.mockImplementation(
1135
+ async (args: unknown): Promise<Array<AutoRemediationSuggestion>> => {
1136
+ const query: Record<string, unknown> = (
1137
+ args as { query: Record<string, unknown> }
1138
+ ).query;
1139
+ // The hold read (the breaker's in-flight read filters on Planning).
1140
+ if (typeof query["status"] === "string") {
1141
+ return [];
1142
+ }
1143
+ return [
1144
+ {
1145
+ id: OTHER_SUGGESTION_ID,
1146
+ status: AutoRemediationSuggestionStatus.AutoExecuted,
1147
+ verificationStatus: AutoRemediationVerificationStatus.Pending,
1148
+ verificationDeadlineAt: OneUptimeDate.addRemoveMinutes(
1149
+ OneUptimeDate.getCurrentDate(),
1150
+ 5,
1151
+ ),
1152
+ } as unknown as AutoRemediationSuggestion,
1153
+ ];
1154
+ },
1155
+ );
1156
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1157
+
1158
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1159
+
1160
+ expect(outcome.textForLlm).toContain(
1161
+ 'Another OneUptime AI run on Docker host "web-1" applied a fix that is still being verified',
1162
+ );
1163
+ expect(enqueue).not.toHaveBeenCalled();
1164
+ expect(release).toHaveBeenCalledTimes(1);
1165
+ expect(kept(toolkit)[0]!.reason).toContain("another OneUptime AI run");
1166
+ });
1167
+
1168
+ it("refuses at the project's hourly limit on infrastructure commands, counted on ResourceCommand rows", async () => {
1169
+ jobCount.mockResolvedValue(
1170
+ new PositiveNumber(MAX_AI_RESOURCE_COMMAND_JOBS_PER_PROJECT_PER_HOUR),
1171
+ );
1172
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1173
+
1174
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1175
+
1176
+ expect(outcome.textForLlm).toContain(
1177
+ "hourly limit on AI commands on its infrastructure",
1178
+ );
1179
+ expect(enqueue).not.toHaveBeenCalled();
1180
+ expect(
1181
+ (jobCount.mock.calls[0]![0] as { query: Record<string, unknown> }).query,
1182
+ ).toMatchObject({
1183
+ projectId: PROJECT_ID,
1184
+ origin: RunnerJobOrigin.AiRemediation,
1185
+ stepType: RunbookStepType.ResourceCommand,
1186
+ });
1187
+ });
1188
+
1189
+ it("stops the moment a human dismissed the suggestion", async () => {
1190
+ jest
1191
+ .spyOn(AutoRemediationSuggestionService, "findOneById")
1192
+ .mockResolvedValue({
1193
+ id: SUGGESTION_ID,
1194
+ status: AutoRemediationSuggestionStatus.Dismissed,
1195
+ } as unknown as AutoRemediationSuggestion);
1196
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1197
+
1198
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1199
+
1200
+ expect(outcome.textForLlm).toContain("dismissed or already settled");
1201
+ expectNothingRanOrRecorded(toolkit, outcome);
1202
+ });
1203
+
1204
+ it("clamps the timeout to the resource agent's maximum", async () => {
1205
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1206
+
1207
+ await execute(toolkit, resourceArgs({ timeoutInMs: 290_000 }));
1208
+
1209
+ expect(enqueue.mock.calls[0]![0]).toMatchObject({
1210
+ timeoutInMs: MAX_RESOURCE_COMMAND_TIMEOUT_MS,
1211
+ });
1212
+ });
1213
+
1214
+ it("takes a command that certainly never ran off the record, and says so", async () => {
1215
+ poll.mockResolvedValue(
1216
+ fakeJob({
1217
+ status: RunnerJobStatus.Failed,
1218
+ exitCode: undefined,
1219
+ output: "",
1220
+ errorMessage: "Refused: the agent is read-only.",
1221
+ }),
1222
+ );
1223
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1224
+
1225
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1226
+
1227
+ expect(outcome.success).toBe(false);
1228
+ expect(outcome.textForLlm).toContain(
1229
+ '"docker restart web" did NOT run on Docker host "web-1": Refused: the agent is read-only.',
1230
+ );
1231
+ expect(outcome.textForLlm).toContain("Nothing changed on the Docker host");
1232
+ expect(toolkit.getExecutedCommands()).toHaveLength(0);
1233
+ // The job exists: its sequence is never reused.
1234
+ await execute(toolkit, resourceArgs());
1235
+ expect(enqueue.mock.calls[1]![0]).toMatchObject({
1236
+ stepId: `${INLINE_COMMAND_STEP_ID_PREFIX}2`,
1237
+ });
1238
+ });
1239
+
1240
+ it("tells the model the agent is not picking commands up when none claimed the job", async () => {
1241
+ poll.mockResolvedValue(
1242
+ fakeJob({
1243
+ status: RunnerJobStatus.TimedOut,
1244
+ exitCode: undefined,
1245
+ output: "",
1246
+ errorMessage: undefined,
1247
+ }),
1248
+ );
1249
+ claimRead.mockResolvedValue({ id: JOB_ID } as unknown as RunnerJob);
1250
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1251
+
1252
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1253
+
1254
+ expect(outcome.textForLlm).toContain("its AI agent is not picking them up");
1255
+ expect(toolkit.getExecutedCommands()).toHaveLength(0);
1256
+ });
1257
+
1258
+ it("a command the chokepoint refused never ran", async () => {
1259
+ enqueue.mockRejectedValue(
1260
+ new ResourceCommandEnqueueRefusedException(
1261
+ "switch_off",
1262
+ 'AI fixes are turned off for Docker host "web-1", so this command was not enqueued.',
1263
+ ),
1264
+ );
1265
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1266
+
1267
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1268
+
1269
+ expect(outcome.success).toBe(false);
1270
+ expect(outcome.textForLlm).toContain("did NOT run");
1271
+ expect(outcome.textForLlm).toContain("AI fixes are turned off");
1272
+ expect(toolkit.getExecutedCommands()).toHaveLength(0);
1273
+ });
1274
+
1275
+ it("a command the agent took with no result back stays on the record as 'may have run'", async () => {
1276
+ poll.mockResolvedValue(
1277
+ fakeJob({
1278
+ status: RunnerJobStatus.TimedOut,
1279
+ exitCode: undefined,
1280
+ output: "",
1281
+ errorMessage: undefined,
1282
+ }),
1283
+ );
1284
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1285
+
1286
+ const outcome: ToolCallOutcome = await execute(
1287
+ toolkit,
1288
+ resourceArgs({ rollbackCommand: SAFE_UNDO }),
1289
+ );
1290
+
1291
+ expect(outcome.success).toBe(true);
1292
+ expect(outcome.textForLlm).toContain(
1293
+ 'RESULT UNKNOWN on Docker host "web-1"',
1294
+ );
1295
+ expect(outcome.textForLlm).toContain(RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME);
1296
+ expect(outcome.textForLlm).toContain(
1297
+ "its rollbackCommand is not run blind",
1298
+ );
1299
+ expect(outcome.result?.citationLabel).toBe(
1300
+ 'Sent to Docker host "web-1", result unknown: docker restart web',
1301
+ );
1302
+ const executed: AiRemediationCommand = toolkit.getExecutedCommands()[0]!;
1303
+ expect(executed.execution?.status).toBe(
1304
+ AiRemediationCommandExecutionStatus.Failed,
1305
+ );
1306
+ expect(executed.execution?.runnerJobId).toBe(JOB_ID.toString());
1307
+ });
1308
+
1309
+ it("a broken wait after the enqueue is 'may have run' too", async () => {
1310
+ poll.mockRejectedValue(new Error("lost the database connection"));
1311
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1312
+
1313
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1314
+
1315
+ expect(outcome.textForLlm).toContain("RESULT UNKNOWN");
1316
+ expect(outcome.textForLlm).toContain("Waiting for its result failed");
1317
+ expect(toolkit.getExecutedCommands()[0]!.execution?.runnerJobId).toBe(
1318
+ JOB_ID.toString(),
1319
+ );
1320
+ });
1321
+
1322
+ it("a program the agent killed at its timeout may have changed the resource", async () => {
1323
+ poll.mockResolvedValue(
1324
+ fakeJob({
1325
+ status: RunnerJobStatus.Failed,
1326
+ exitCode: undefined,
1327
+ output: "",
1328
+ errorMessage: "Killed (timeout 30s)",
1329
+ }),
1330
+ );
1331
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1332
+
1333
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1334
+
1335
+ expect(outcome.textForLlm).toContain("MAY have changed the resource");
1336
+ expect(toolkit.getExecutedCommands()).toHaveLength(1);
1337
+ });
1338
+
1339
+ it("a change that ran and failed is on the record, and only an access failure becomes the resource's last error", async () => {
1340
+ poll.mockResolvedValue(
1341
+ fakeJob({
1342
+ status: RunnerJobStatus.Failed,
1343
+ exitCode: 1,
1344
+ output: "[stderr]\nError response from daemon: No such container: web",
1345
+ errorMessage: "Exit code 1",
1346
+ }),
1347
+ );
1348
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1349
+
1350
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1351
+
1352
+ expect(outcome.success).toBe(true);
1353
+ expect(outcome.textForLlm).toContain("FAILED");
1354
+ expect(toolkit.getExecutedCommands()[0]!.execution?.status).toBe(
1355
+ AiRemediationCommandExecutionStatus.Failed,
1356
+ );
1357
+ expect(recordOutcome).not.toHaveBeenCalled();
1358
+ });
1359
+
1360
+ it(`sends at most the run's command budget`, async () => {
1361
+ const toolkit: RemediationCommandToolkit = buildToolkit();
1362
+
1363
+ for (let i: number = 0; i < 5; i++) {
1364
+ await execute(toolkit, resourceArgs());
1365
+ }
1366
+
1367
+ const outcome: ToolCallOutcome = await execute(toolkit, resourceArgs());
1368
+
1369
+ expect(outcome.textForLlm).toContain("command budget");
1370
+ expect(enqueue).toHaveBeenCalledTimes(5);
1371
+ });
1372
+ });
1373
+
1374
+ describe("propose_remediation_commands on a resource round", () => {
1375
+ function propose(
1376
+ toolkit: RemediationCommandToolkit,
1377
+ commands: Array<JSONObject>,
1378
+ ): Promise<ToolCallOutcome> {
1379
+ return tool(toolkit, "propose_remediation_commands").execute({
1380
+ commands,
1381
+ } as JSONObject);
1382
+ }
1383
+
1384
+ it("records a plan of resource commands, every one RequiresApproval, with the resource and the tier", async () => {
1385
+ const toolkit: RemediationCommandToolkit = buildToolkit({
1386
+ mode: "Suggest",
1387
+ });
1388
+
1389
+ const outcome: ToolCallOutcome = await propose(toolkit, [
1390
+ resourceArgs({ command: RISKY_WRITE, rollbackCommand: SAFE_UNDO }),
1391
+ resourceArgs({ command: SAFE_WRITE }),
1392
+ ]);
1393
+
1394
+ expect(outcome.success).toBe(true);
1395
+ const plan: AiRemediationCommandPlan = toolkit.getProposedPlan()!;
1396
+ expect(plan.commands).toHaveLength(2);
1397
+ expect(plan.commands[0]).toMatchObject({
1398
+ sequence: 1,
1399
+ stepType: RunbookStepType.ResourceCommand,
1400
+ runnerId: AGENT_ID,
1401
+ runnerNameSnapshot: "Docker AI agent",
1402
+ resourceType: AiResourceType.DockerHost,
1403
+ resourceId: RESOURCE_ID,
1404
+ resourceNameSnapshot: "web-1",
1405
+ resourceCommandTier: ResourceCommandTier.RiskyWrite,
1406
+ policyVerdict: AiRemediationCommandPolicyVerdict.RequiresApproval,
1407
+ rollbackCommand: SAFE_UNDO,
1408
+ });
1409
+ expect(plan.commands[1]!.policyVerdict).toBe(
1410
+ AiRemediationCommandPolicyVerdict.RequiresApproval,
1411
+ );
1412
+ // Nothing is enqueued while planning.
1413
+ expect(enqueue).not.toHaveBeenCalled();
1414
+ expect(liveStatus).not.toHaveBeenCalled();
1415
+ });
1416
+
1417
+ it("refuses the whole plan when a step is a read, denied, credentialed or out of scope", async () => {
1418
+ const toolkit: RemediationCommandToolkit = buildToolkit({
1419
+ mode: "Suggest",
1420
+ resourceTargets: [resource({}, { protectedTargets: ["api"] })],
1421
+ });
1422
+
1423
+ const outcome: ToolCallOutcome = await propose(toolkit, [
1424
+ resourceArgs({ command: READ }),
1425
+ resourceArgs({ command: "docker exec web sh" }),
1426
+ resourceArgs({ credentialId: "55555555-5555-4555-8555-555555555555" }),
1427
+ resourceArgs({ command: "docker restart api" }),
1428
+ resourceArgs(),
1429
+ ]);
1430
+
1431
+ expect(outcome.success).toBe(false);
1432
+ expect(outcome.textForLlm).toContain("The plan was NOT recorded");
1433
+ expect(outcome.textForLlm).toContain("Command 1:");
1434
+ expect(outcome.textForLlm).toContain("is read-only — it is not a fix");
1435
+ expect(outcome.textForLlm).toContain("Command 2: Denied");
1436
+ expect(outcome.textForLlm).toContain("Command 3: A ResourceCommand never");
1437
+ expect(outcome.textForLlm).toContain("Command 4:");
1438
+ expect(outcome.textForLlm).toContain("protects");
1439
+ expect(outcome.textForLlm).not.toContain("Command 5:");
1440
+ expect(toolkit.getProposedPlan()).toBeNull();
1441
+ });
1442
+
1443
+ it(`caps a plan at ${MAX_PLAN_COMMANDS} commands`, async () => {
1444
+ const toolkit: RemediationCommandToolkit = buildToolkit({
1445
+ mode: "Suggest",
1446
+ });
1447
+ const commands: Array<JSONObject> = [];
1448
+ for (let i: number = 0; i <= MAX_PLAN_COMMANDS; i++) {
1449
+ commands.push(resourceArgs());
1450
+ }
1451
+
1452
+ const outcome: ToolCallOutcome = await propose(toolkit, commands);
1453
+
1454
+ expect(outcome.success).toBe(false);
1455
+ expect(toolkit.getProposedPlan()).toBeNull();
1456
+ });
1457
+ });
1458
+
1459
+ describe("the resource write scope, as the toolkit reads the agent's posture", () => {
1460
+ it("never refuses a read or a denied command (the policy handles the latter)", () => {
1461
+ const readOnly: ResourceAiAccessStatus = resource(
1462
+ {},
1463
+ { allowWrites: false },
1464
+ );
1465
+ expect(
1466
+ RemediationCommandToolkit.getResourceWriteScopeRefusal({
1467
+ resource: readOnly,
1468
+ command: READ,
1469
+ }),
1470
+ ).toBeNull();
1471
+ expect(
1472
+ RemediationCommandToolkit.getResourceWriteScopeRefusal({
1473
+ resource: readOnly,
1474
+ command: "docker exec web sh",
1475
+ }),
1476
+ ).toBeNull();
1477
+ });
1478
+
1479
+ it("refuses every write of a read-only agent, naming the switch", () => {
1480
+ expect(
1481
+ RemediationCommandToolkit.getResourceWriteScopeRefusal({
1482
+ resource: resource({}, { allowWrites: false }),
1483
+ command: SAFE_WRITE,
1484
+ }),
1485
+ ).toContain("ONEUPTIME_AI_ALLOW_WRITES=true");
1486
+ });
1487
+
1488
+ it("refuses every write when the agent reported no posture", () => {
1489
+ const status: ResourceAiAccessStatus = resource();
1490
+ status.agent!.posture = null;
1491
+
1492
+ expect(
1493
+ RemediationCommandToolkit.getResourceWriteScopeRefusal({
1494
+ resource: status,
1495
+ command: SAFE_WRITE,
1496
+ }),
1497
+ ).toContain("read-only");
1498
+ expect(RemediationCommandToolkit.describeResourceWriteScope(status)).toBe(
1499
+ "not reported by the agent yet, so it runs no change",
1500
+ );
1501
+ });
1502
+
1503
+ it("describes the scope in the words of the agent's settings", () => {
1504
+ expect(
1505
+ RemediationCommandToolkit.describeResourceWriteScope(
1506
+ resource({}, { allowWrites: false }),
1507
+ ),
1508
+ ).toContain(
1509
+ "read-only (ONEUPTIME_AI_ALLOW_WRITES is not true on the agent)",
1510
+ );
1511
+ expect(
1512
+ RemediationCommandToolkit.describeResourceWriteScope(
1513
+ resource({}, { writeTargets: ["web*", "api"] }),
1514
+ ),
1515
+ ).toContain(
1516
+ "changes only targets matching ONEUPTIME_AI_WRITE_TARGETS=web*,api",
1517
+ );
1518
+ expect(
1519
+ RemediationCommandToolkit.describeResourceWriteScope(resource()),
1520
+ ).toContain("never its protected targets (oneuptime-ai-agent)");
1521
+ });
1522
+
1523
+ it("names where the agent's write access is set, for an approval refusal", () => {
1524
+ expect(
1525
+ RemediationCommandToolkit.getResourceScopeRefusalNextStep(
1526
+ AiResourceType.Host,
1527
+ ),
1528
+ ).toContain("Host AI agent's write access");
1529
+ });
1530
+ });