@oneuptime/common 14.0.9 → 14.0.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (383) hide show
  1. package/Models/DatabaseModels/AutoRemediationSuggestion.ts +60 -0
  2. package/Models/DatabaseModels/CephCluster.ts +213 -0
  3. package/Models/DatabaseModels/DatabaseServer.ts +213 -0
  4. package/Models/DatabaseModels/DockerHost.ts +213 -0
  5. package/Models/DatabaseModels/DockerSwarmCluster.ts +213 -0
  6. package/Models/DatabaseModels/GlobalConfig.ts +1 -1
  7. package/Models/DatabaseModels/Host.ts +213 -0
  8. package/Models/DatabaseModels/Index.ts +2 -0
  9. package/Models/DatabaseModels/PodmanHost.ts +213 -0
  10. package/Models/DatabaseModels/ProxmoxCluster.ts +213 -0
  11. package/Models/DatabaseModels/ResourceAiAgent.ts +419 -0
  12. package/Models/DatabaseModels/RunnerJob.ts +144 -0
  13. package/Models/DatabaseModels/VMwareVCenter.ts +213 -0
  14. package/Server/API/AutoRemediationAPI.ts +392 -1
  15. package/Server/API/ResourceAiAccessAPI.ts +1669 -0
  16. package/Server/Infrastructure/Postgres/SchemaMigrations/1796300000000-AddResourceAiAgents.ts +423 -0
  17. package/Server/Infrastructure/Postgres/SchemaMigrations/Index.ts +2 -0
  18. package/Server/Infrastructure/Semaphore.ts +22 -0
  19. package/Server/Middleware/TelemetryIngest.ts +15 -5
  20. package/Server/Services/AnalyticsDatabaseService.ts +45 -1
  21. package/Server/Services/AutoRemediationRuleEngineService.ts +940 -112
  22. package/Server/Services/CephClusterService.ts +109 -1
  23. package/Server/Services/DatabaseServerService.ts +116 -1
  24. package/Server/Services/DockerHostService.ts +109 -1
  25. package/Server/Services/DockerSwarmClusterService.ts +113 -1
  26. package/Server/Services/HostService.ts +126 -4
  27. package/Server/Services/MetricRecordingRuleService.ts +47 -0
  28. package/Server/Services/PodmanHostService.ts +109 -1
  29. package/Server/Services/ProxmoxClusterService.ts +110 -1
  30. package/Server/Services/ResourceAiAccessService.ts +1439 -0
  31. package/Server/Services/ResourceAiAgentJobService.ts +202 -0
  32. package/Server/Services/ResourceAiAgentService.ts +2432 -0
  33. package/Server/Services/RunnerJobService.ts +611 -4
  34. package/Server/Services/TelemetryUsageBillingService.ts +28 -0
  35. package/Server/Services/TraceRecordingRuleService.ts +33 -0
  36. package/Server/Services/VMwareVCenterService.ts +110 -1
  37. package/Server/Utils/AI/Remediation/RemediationCommandTools.ts +1434 -32
  38. package/Server/Utils/AI/Remediation/RemediationExecutionRunner.ts +935 -21
  39. package/Server/Utils/AI/Remediation/RemediationPlanRunner.ts +25 -1
  40. package/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.ts +577 -0
  41. package/Server/Utils/AI/ResourceAccess/ResourceAccessContext.ts +271 -0
  42. package/Server/Utils/AI/ResourceAccess/ResourceAccessToolNames.ts +45 -0
  43. package/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.ts +964 -0
  44. package/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.ts +333 -0
  45. package/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.ts +873 -0
  46. package/Server/Utils/AI/SRE/AIInvestigationEngine.ts +72 -2
  47. package/Server/Utils/AI/SRE/AlertInvestigationRunner.ts +72 -5
  48. package/Server/Utils/AI/SRE/IncidentInvestigationRunner.ts +72 -5
  49. package/Server/Utils/AutoRemediation/CommandPlanExecutor.ts +396 -13
  50. package/Server/Utils/AutoRemediation/RemediationVerifier.ts +35 -2
  51. package/Server/Utils/Database/ProjectScopedReferenceValidator.ts +9 -1
  52. package/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.ts +920 -0
  53. package/Server/Utils/SessionReplay/SessionReplayUsage.ts +84 -6
  54. package/Server/Utils/Workspace/MicrosoftTeams/Actions/Alert.ts +35 -15
  55. package/Server/Utils/Workspace/MicrosoftTeams/Actions/AlertEpisode.ts +33 -15
  56. package/Server/Utils/Workspace/MicrosoftTeams/Actions/Auth.ts +21 -1
  57. package/Server/Utils/Workspace/MicrosoftTeams/Actions/Incident.ts +298 -224
  58. package/Server/Utils/Workspace/MicrosoftTeams/Actions/IncidentEpisode.ts +35 -15
  59. package/Server/Utils/Workspace/MicrosoftTeams/Actions/ScheduledMaintenance.ts +389 -175
  60. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeams.ts +366 -143
  61. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.ts +163 -0
  62. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.ts +396 -0
  63. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.ts +418 -0
  64. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.ts +152 -0
  65. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.ts +251 -0
  66. package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.ts +214 -0
  67. package/Tests/App/Dashboard/ClusterAccessNotice.test.tsx +93 -0
  68. package/Tests/App/Dashboard/DatabaseDocumentationMarkdown.test.ts +18 -6
  69. package/Tests/App/Dashboard/InvestigationInfrastructureTools.test.tsx +506 -0
  70. package/Tests/App/Dashboard/RemediationSuggestionCardDescription.test.tsx +209 -0
  71. package/Tests/App/Dashboard/RemediationSuggestionCardResource.test.tsx +317 -0
  72. package/Tests/App/Dashboard/ResourceAiAccessSettingsUtil.test.ts +921 -0
  73. package/Tests/App/Dashboard/ResourceAiAgentInstall.test.ts +860 -0
  74. package/Tests/App/Dashboard/ResourceAiAgentPage.test.tsx +1721 -0
  75. package/Tests/App/Dashboard/ResourceAiAgentStatus.test.ts +1002 -0
  76. package/Tests/App/Dashboard/ResourceAiInsightsPage.test.tsx +931 -0
  77. package/Tests/App/Dashboard/ResourceAiNavigation.test.tsx +575 -0
  78. package/Tests/App/Dashboard/RunbookStepTypeMaps.test.ts +38 -7
  79. package/Tests/Models/DatabaseModels/DatabaseServerModels.test.ts +33 -1
  80. package/Tests/Models/DatabaseModels/ResourceAiAccessColumns.test.ts +738 -0
  81. package/Tests/Models/DatabaseModels/ResourceAiAgentModel.test.ts +842 -0
  82. package/Tests/Server/API/AutoRemediationApproveResourceRound.test.ts +835 -0
  83. package/Tests/Server/API/ResourceAiAccessAPI.test.ts +2730 -0
  84. package/Tests/Server/Infrastructure/Postgres/AddDatabaseServerTablesMigration.test.ts +60 -0
  85. package/Tests/Server/Infrastructure/Postgres/AddResourceAiAgentsMigration.test.ts +659 -0
  86. package/Tests/Server/Infrastructure/SemaphoreMutex.test.ts +31 -0
  87. package/Tests/Server/Middleware/TelemetryIngestBrowserKey.test.ts +63 -5
  88. package/Tests/Server/Middleware/TelemetryIngestKubernetesAgentRunnerPinnedKey.test.ts +103 -5
  89. package/Tests/Server/Middleware/TelemetryIngestKubernetesAgentRunnerRateLimit.test.ts +19 -1
  90. package/Tests/Server/Services/AutoRemediationResourceRuleEngine.test.ts +1279 -0
  91. package/Tests/Server/Services/DatabaseServerService.test.ts +55 -0
  92. package/Tests/Server/Services/GroupTelemetryUsageExcludeNames.test.ts +299 -0
  93. package/Tests/Server/Services/HostServiceFindOrCreateMemo.test.ts +44 -0
  94. package/Tests/Server/Services/MonitorProbeServiceIntervalScheduling.test.ts +45 -6
  95. package/Tests/Server/Services/RecordingRuleReservedMetricName.test.ts +171 -0
  96. package/Tests/Server/Services/ResourceAiAccessChangeAuthorization.test.ts +256 -0
  97. package/Tests/Server/Services/ResourceAiAccessService.test.ts +1373 -0
  98. package/Tests/Server/Services/ResourceAiAgentJobService.test.ts +375 -0
  99. package/Tests/Server/Services/ResourceAiAgentServiceHelpers.test.ts +1114 -0
  100. package/Tests/Server/Services/ResourceAiAgentServiceLifecycle.test.ts +1050 -0
  101. package/Tests/Server/Services/ResourceAiAgentServiceRegister.test.ts +2251 -0
  102. package/Tests/Server/Services/ResourceAiSettingsCreate.test.ts +373 -0
  103. package/Tests/Server/Services/ResourceAiSettingsPermission.test.ts +1042 -0
  104. package/Tests/Server/Services/ResourceServiceDeleteCleansUpAi.test.ts +319 -0
  105. package/Tests/Server/Services/RunnerJobEnqueueKubectl.test.ts +105 -0
  106. package/Tests/Server/Services/RunnerJobEnqueueResourceCommand.test.ts +986 -0
  107. package/Tests/Server/Services/RunnerJobResourceCommandLane.test.ts +211 -0
  108. package/Tests/Server/Services/RunnerJobResourceCommandTimeoutAndRedaction.test.ts +352 -0
  109. package/Tests/Server/Services/TelemetryUsageBillingSloExclusion.test.ts +218 -0
  110. package/Tests/Server/TestingUtils/Services/FakeRunnerJobCount.ts +145 -0
  111. package/Tests/Server/Utils/AI/InvestigationInfrastructureAccessWiring.test.ts +453 -0
  112. package/Tests/Server/Utils/AI/InvestigationInfrastructureReport.test.ts +721 -0
  113. package/Tests/Server/Utils/AI/RemediationCommandTools.test.ts +94 -1
  114. package/Tests/Server/Utils/AI/RemediationCommandToolsResource.test.ts +1530 -0
  115. package/Tests/Server/Utils/AI/RemediationExecutionRunnerResourceMode.test.ts +1249 -0
  116. package/Tests/Server/Utils/AI/RemediationPlanRunner.test.ts +117 -0
  117. package/Tests/Server/Utils/AI/RemediationResourceCopyParity.test.ts +463 -0
  118. package/Tests/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.test.ts +722 -0
  119. package/Tests/Server/Utils/AI/ResourceAccess/ResourceAccessContext.test.ts +266 -0
  120. package/Tests/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.test.ts +1220 -0
  121. package/Tests/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.test.ts +569 -0
  122. package/Tests/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.test.ts +830 -0
  123. package/Tests/Server/Utils/AutoRemediation/CommandPlanExecutorResource.test.ts +717 -0
  124. package/Tests/Server/Utils/AutoRemediation/RemediationVerifierResourceFollowUp.test.ts +338 -0
  125. package/Tests/Server/Utils/Monitor/Criteria/SessionReplayBudgetTemplateCriteria.test.ts +422 -0
  126. package/Tests/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.test.ts +1647 -0
  127. package/Tests/Server/Utils/SessionReplay/SessionReplayUsage.test.ts +143 -1
  128. package/Tests/Server/Utils/Telemetry/TelemetryIngestionKeyGuard.test.ts +33 -3
  129. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsAccountNotLinked.test.ts +1093 -0
  130. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.test.ts +1049 -0
  131. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.test.ts +1676 -0
  132. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCards.test.ts +2884 -0
  133. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.test.ts +2490 -0
  134. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateSubmit.test.ts +2293 -0
  135. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateSubmitServerTimezone.test.ts +386 -0
  136. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.test.ts +1477 -0
  137. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.test.ts +1967 -0
  138. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsStripHtmlTags.test.ts +92 -0
  139. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsSubmittedFormRemoval.test.ts +1819 -0
  140. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.test.ts +1033 -0
  141. package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezoneServerZone.test.ts +610 -0
  142. package/Tests/Server/Utils/Workspace/MicrosoftTeamsBotMessageHandling.test.ts +2610 -0
  143. package/Tests/Server/Utils/Workspace/MicrosoftTeamsCreateCommandsEndToEnd.test.ts +4672 -0
  144. package/Tests/Server/Utils/Workspace/WorkspaceCreateProjectReferences.test.ts +190 -65
  145. package/Tests/Types/AutoRemediation/AiRemediationCommandPlan.test.ts +10 -4
  146. package/Tests/Types/AutoRemediation/AiRemediationCommandPlanResourceCommand.test.ts +305 -0
  147. package/Tests/Types/Monitor/Recommendation/MonitorRecommendationCatalog.test.ts +277 -13
  148. package/Tests/Types/Monitor/Recommendation/MonitorRecommendationUtil.test.ts +113 -0
  149. package/Tests/Types/Monitor/RumAlertTemplates.test.ts +773 -2
  150. package/Tests/Types/ResourceAiAgent/AiResourceType.test.ts +419 -0
  151. package/Tests/Types/ResourceAiAgent/ResourceAiAccess.test.ts +517 -0
  152. package/Tests/Types/Runbook/RunbookStepType.test.ts +147 -8
  153. package/Tests/UI/Rum/RecordingHealthDashboard.test.tsx +769 -1
  154. package/Tests/Utils/AiRemediation/Resource/CephCommandPolicy.test.ts +1653 -0
  155. package/Tests/Utils/AiRemediation/Resource/CephOutputRedaction.test.ts +204 -0
  156. package/Tests/Utils/AiRemediation/Resource/DatabaseCommandPolicy.test.ts +1077 -0
  157. package/Tests/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.test.ts +558 -0
  158. package/Tests/Utils/AiRemediation/Resource/DatabaseQueryRedactor.test.ts +834 -0
  159. package/Tests/Utils/AiRemediation/Resource/DockerCliGrammar.test.ts +747 -0
  160. package/Tests/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.test.ts +1259 -0
  161. package/Tests/Utils/AiRemediation/Resource/DockerOutputRedaction.test.ts +386 -0
  162. package/Tests/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.test.ts +782 -0
  163. package/Tests/Utils/AiRemediation/Resource/GovcCommandPolicy.test.ts +2259 -0
  164. package/Tests/Utils/AiRemediation/Resource/HostCommandPolicy.test.ts +2019 -0
  165. package/Tests/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.test.ts +2497 -0
  166. package/Tests/Utils/AiRemediation/Resource/ResourceCommandPolicy.test.ts +1305 -0
  167. package/Tests/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.test.ts +585 -0
  168. package/Tests/Utils/AiRemediation/Resource/ResourceOutputRedactor.test.ts +334 -0
  169. package/Tests/Utils/AiRemediation/Resource/ResourceOutputRedactorCopyParity.test.ts +98 -0
  170. package/Tests/Utils/AiRemediation/Resource/ResourcePolicyImportClosure.test.ts +391 -0
  171. package/Tests/Utils/AiRemediation/ResourceAiAgentPolicyCopyParity.test.ts +497 -0
  172. package/Tests/Utils/SessionReplay/SessionReplayBudgetMetricType.test.ts +383 -0
  173. package/Types/AI/ResourceAiAccessApi.ts +140 -0
  174. package/Types/AI/ResourceAiAccessPermissions.ts +126 -0
  175. package/Types/AutoRemediation/AiRemediationCommandPlan.ts +183 -2
  176. package/Types/Monitor/Recommendation/MonitorRecommendationCatalog.ts +55 -22
  177. package/Types/Monitor/Recommendation/MonitorRecommendationTypes.ts +35 -3
  178. package/Types/Monitor/RumAlertTemplates.ts +351 -4
  179. package/Types/ResourceAiAgent/AiResourceType.ts +310 -0
  180. package/Types/ResourceAiAgent/ResourceAiAccess.ts +574 -0
  181. package/Types/Rum/SessionReplayBudgetMetricType.ts +47 -0
  182. package/Types/Runbook/RunbookStepType.ts +29 -0
  183. package/Types/Telemetry/TelemetryIngestSurface.ts +14 -2
  184. package/Utils/AI/InvestigationReport.ts +121 -15
  185. package/Utils/AiRemediation/Resource/CephCommandPolicy.ts +1936 -0
  186. package/Utils/AiRemediation/Resource/DatabaseCommandPolicy.ts +166 -0
  187. package/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.ts +1532 -0
  188. package/Utils/AiRemediation/Resource/DatabaseQueryRedactor.ts +1394 -0
  189. package/Utils/AiRemediation/Resource/DockerCliGrammar.ts +1839 -0
  190. package/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.ts +490 -0
  191. package/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.ts +687 -0
  192. package/Utils/AiRemediation/Resource/GovcCommandPolicy.ts +2096 -0
  193. package/Utils/AiRemediation/Resource/HostCommandPolicy.ts +2982 -0
  194. package/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.ts +1679 -0
  195. package/Utils/AiRemediation/Resource/ResourceCommandPolicy.ts +851 -0
  196. package/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.ts +593 -0
  197. package/Utils/AiRemediation/Resource/ResourceOutputRedactor.ts +2812 -0
  198. package/Utils/SessionReplay/SessionReplayBudgetMetricType.ts +184 -0
  199. package/build/dist/Models/DatabaseModels/AutoRemediationSuggestion.js +62 -0
  200. package/build/dist/Models/DatabaseModels/AutoRemediationSuggestion.js.map +1 -1
  201. package/build/dist/Models/DatabaseModels/CephCluster.js +219 -0
  202. package/build/dist/Models/DatabaseModels/CephCluster.js.map +1 -1
  203. package/build/dist/Models/DatabaseModels/DatabaseServer.js +219 -0
  204. package/build/dist/Models/DatabaseModels/DatabaseServer.js.map +1 -1
  205. package/build/dist/Models/DatabaseModels/DockerHost.js +219 -0
  206. package/build/dist/Models/DatabaseModels/DockerHost.js.map +1 -1
  207. package/build/dist/Models/DatabaseModels/DockerSwarmCluster.js +219 -0
  208. package/build/dist/Models/DatabaseModels/DockerSwarmCluster.js.map +1 -1
  209. package/build/dist/Models/DatabaseModels/GlobalConfig.js +1 -1
  210. package/build/dist/Models/DatabaseModels/GlobalConfig.js.map +1 -1
  211. package/build/dist/Models/DatabaseModels/Host.js +219 -0
  212. package/build/dist/Models/DatabaseModels/Host.js.map +1 -1
  213. package/build/dist/Models/DatabaseModels/Index.js +2 -0
  214. package/build/dist/Models/DatabaseModels/Index.js.map +1 -1
  215. package/build/dist/Models/DatabaseModels/PodmanHost.js +219 -0
  216. package/build/dist/Models/DatabaseModels/PodmanHost.js.map +1 -1
  217. package/build/dist/Models/DatabaseModels/ProxmoxCluster.js +219 -0
  218. package/build/dist/Models/DatabaseModels/ProxmoxCluster.js.map +1 -1
  219. package/build/dist/Models/DatabaseModels/ResourceAiAgent.js +439 -0
  220. package/build/dist/Models/DatabaseModels/ResourceAiAgent.js.map +1 -0
  221. package/build/dist/Models/DatabaseModels/RunnerJob.js +145 -0
  222. package/build/dist/Models/DatabaseModels/RunnerJob.js.map +1 -1
  223. package/build/dist/Models/DatabaseModels/VMwareVCenter.js +219 -0
  224. package/build/dist/Models/DatabaseModels/VMwareVCenter.js.map +1 -1
  225. package/build/dist/Server/API/AutoRemediationAPI.js +231 -1
  226. package/build/dist/Server/API/AutoRemediationAPI.js.map +1 -1
  227. package/build/dist/Server/API/ResourceAiAccessAPI.js +1119 -0
  228. package/build/dist/Server/API/ResourceAiAccessAPI.js.map +1 -0
  229. package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/1796300000000-AddResourceAiAgents.js +168 -0
  230. package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/1796300000000-AddResourceAiAgents.js.map +1 -0
  231. package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/Index.js +2 -0
  232. package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/Index.js.map +1 -1
  233. package/build/dist/Server/Infrastructure/Semaphore.js +6 -0
  234. package/build/dist/Server/Infrastructure/Semaphore.js.map +1 -1
  235. package/build/dist/Server/Middleware/TelemetryIngest.js +12 -5
  236. package/build/dist/Server/Middleware/TelemetryIngest.js.map +1 -1
  237. package/build/dist/Server/Services/AnalyticsDatabaseService.js +26 -1
  238. package/build/dist/Server/Services/AnalyticsDatabaseService.js.map +1 -1
  239. package/build/dist/Server/Services/AutoRemediationRuleEngineService.js +601 -3
  240. package/build/dist/Server/Services/AutoRemediationRuleEngineService.js.map +1 -1
  241. package/build/dist/Server/Services/CephClusterService.js +104 -0
  242. package/build/dist/Server/Services/CephClusterService.js.map +1 -1
  243. package/build/dist/Server/Services/DatabaseServerService.js +99 -0
  244. package/build/dist/Server/Services/DatabaseServerService.js.map +1 -1
  245. package/build/dist/Server/Services/DockerHostService.js +104 -0
  246. package/build/dist/Server/Services/DockerHostService.js.map +1 -1
  247. package/build/dist/Server/Services/DockerSwarmClusterService.js +104 -0
  248. package/build/dist/Server/Services/DockerSwarmClusterService.js.map +1 -1
  249. package/build/dist/Server/Services/HostService.js +114 -2
  250. package/build/dist/Server/Services/HostService.js.map +1 -1
  251. package/build/dist/Server/Services/MetricRecordingRuleService.js +46 -0
  252. package/build/dist/Server/Services/MetricRecordingRuleService.js.map +1 -1
  253. package/build/dist/Server/Services/PodmanHostService.js +104 -0
  254. package/build/dist/Server/Services/PodmanHostService.js.map +1 -1
  255. package/build/dist/Server/Services/ProxmoxClusterService.js +104 -0
  256. package/build/dist/Server/Services/ProxmoxClusterService.js.map +1 -1
  257. package/build/dist/Server/Services/ResourceAiAccessService.js +1028 -0
  258. package/build/dist/Server/Services/ResourceAiAccessService.js.map +1 -0
  259. package/build/dist/Server/Services/ResourceAiAgentJobService.js +173 -0
  260. package/build/dist/Server/Services/ResourceAiAgentJobService.js.map +1 -0
  261. package/build/dist/Server/Services/ResourceAiAgentService.js +1687 -0
  262. package/build/dist/Server/Services/ResourceAiAgentService.js.map +1 -0
  263. package/build/dist/Server/Services/RunnerJobService.js +344 -5
  264. package/build/dist/Server/Services/RunnerJobService.js.map +1 -1
  265. package/build/dist/Server/Services/TelemetryUsageBillingService.js +26 -0
  266. package/build/dist/Server/Services/TelemetryUsageBillingService.js.map +1 -1
  267. package/build/dist/Server/Services/TraceRecordingRuleService.js +37 -0
  268. package/build/dist/Server/Services/TraceRecordingRuleService.js.map +1 -1
  269. package/build/dist/Server/Services/VMwareVCenterService.js +104 -0
  270. package/build/dist/Server/Services/VMwareVCenterService.js.map +1 -1
  271. package/build/dist/Server/Utils/AI/Remediation/RemediationCommandTools.js +1029 -30
  272. package/build/dist/Server/Utils/AI/Remediation/RemediationCommandTools.js.map +1 -1
  273. package/build/dist/Server/Utils/AI/Remediation/RemediationExecutionRunner.js +633 -28
  274. package/build/dist/Server/Utils/AI/Remediation/RemediationExecutionRunner.js.map +1 -1
  275. package/build/dist/Server/Utils/AI/Remediation/RemediationPlanRunner.js +21 -1
  276. package/build/dist/Server/Utils/AI/Remediation/RemediationPlanRunner.js.map +1 -1
  277. package/build/dist/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.js +330 -0
  278. package/build/dist/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.js.map +1 -0
  279. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessContext.js +160 -0
  280. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessContext.js.map +1 -0
  281. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessToolNames.js +33 -0
  282. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessToolNames.js.map +1 -0
  283. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.js +640 -0
  284. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.js.map +1 -0
  285. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.js +226 -0
  286. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.js.map +1 -0
  287. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.js +600 -0
  288. package/build/dist/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.js.map +1 -0
  289. package/build/dist/Server/Utils/AI/SRE/AIInvestigationEngine.js +50 -4
  290. package/build/dist/Server/Utils/AI/SRE/AIInvestigationEngine.js.map +1 -1
  291. package/build/dist/Server/Utils/AI/SRE/AlertInvestigationRunner.js +55 -4
  292. package/build/dist/Server/Utils/AI/SRE/AlertInvestigationRunner.js.map +1 -1
  293. package/build/dist/Server/Utils/AI/SRE/IncidentInvestigationRunner.js +55 -4
  294. package/build/dist/Server/Utils/AI/SRE/IncidentInvestigationRunner.js.map +1 -1
  295. package/build/dist/Server/Utils/AutoRemediation/CommandPlanExecutor.js +290 -13
  296. package/build/dist/Server/Utils/AutoRemediation/CommandPlanExecutor.js.map +1 -1
  297. package/build/dist/Server/Utils/AutoRemediation/RemediationVerifier.js +29 -1
  298. package/build/dist/Server/Utils/AutoRemediation/RemediationVerifier.js.map +1 -1
  299. package/build/dist/Server/Utils/Database/ProjectScopedReferenceValidator.js +9 -1
  300. package/build/dist/Server/Utils/Database/ProjectScopedReferenceValidator.js.map +1 -1
  301. package/build/dist/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.js +627 -0
  302. package/build/dist/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.js.map +1 -0
  303. package/build/dist/Server/Utils/SessionReplay/SessionReplayUsage.js +60 -5
  304. package/build/dist/Server/Utils/SessionReplay/SessionReplayUsage.js.map +1 -1
  305. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Alert.js +19 -15
  306. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Alert.js.map +1 -1
  307. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/AlertEpisode.js +19 -15
  308. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/AlertEpisode.js.map +1 -1
  309. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Auth.js +17 -1
  310. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Auth.js.map +1 -1
  311. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Incident.js +183 -209
  312. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Incident.js.map +1 -1
  313. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/IncidentEpisode.js +19 -15
  314. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/IncidentEpisode.js.map +1 -1
  315. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/ScheduledMaintenance.js +233 -167
  316. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/ScheduledMaintenance.js.map +1 -1
  317. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeams.js +263 -110
  318. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeams.js.map +1 -1
  319. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.js +129 -0
  320. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.js.map +1 -0
  321. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.js +266 -0
  322. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.js.map +1 -0
  323. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.js +258 -0
  324. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.js.map +1 -0
  325. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.js +104 -0
  326. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.js.map +1 -0
  327. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.js +183 -0
  328. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.js.map +1 -0
  329. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.js +123 -0
  330. package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.js.map +1 -0
  331. package/build/dist/Types/AI/ResourceAiAccessApi.js +30 -0
  332. package/build/dist/Types/AI/ResourceAiAccessApi.js.map +1 -0
  333. package/build/dist/Types/AI/ResourceAiAccessPermissions.js +98 -0
  334. package/build/dist/Types/AI/ResourceAiAccessPermissions.js.map +1 -0
  335. package/build/dist/Types/AutoRemediation/AiRemediationCommandPlan.js +116 -29
  336. package/build/dist/Types/AutoRemediation/AiRemediationCommandPlan.js.map +1 -1
  337. package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationCatalog.js +47 -21
  338. package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationCatalog.js.map +1 -1
  339. package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationTypes.js +5 -0
  340. package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationTypes.js.map +1 -1
  341. package/build/dist/Types/Monitor/RumAlertTemplates.js +199 -3
  342. package/build/dist/Types/Monitor/RumAlertTemplates.js.map +1 -1
  343. package/build/dist/Types/ResourceAiAgent/AiResourceType.js +242 -0
  344. package/build/dist/Types/ResourceAiAgent/AiResourceType.js.map +1 -0
  345. package/build/dist/Types/ResourceAiAgent/ResourceAiAccess.js +275 -0
  346. package/build/dist/Types/ResourceAiAgent/ResourceAiAccess.js.map +1 -0
  347. package/build/dist/Types/Rum/SessionReplayBudgetMetricType.js +45 -0
  348. package/build/dist/Types/Rum/SessionReplayBudgetMetricType.js.map +1 -0
  349. package/build/dist/Types/Runbook/RunbookStepType.js +27 -0
  350. package/build/dist/Types/Runbook/RunbookStepType.js.map +1 -1
  351. package/build/dist/Types/Telemetry/TelemetryIngestSurface.js +13 -2
  352. package/build/dist/Types/Telemetry/TelemetryIngestSurface.js.map +1 -1
  353. package/build/dist/Utils/AI/InvestigationReport.js +71 -11
  354. package/build/dist/Utils/AI/InvestigationReport.js.map +1 -1
  355. package/build/dist/Utils/AiRemediation/Resource/CephCommandPolicy.js +1462 -0
  356. package/build/dist/Utils/AiRemediation/Resource/CephCommandPolicy.js.map +1 -0
  357. package/build/dist/Utils/AiRemediation/Resource/DatabaseCommandPolicy.js +122 -0
  358. package/build/dist/Utils/AiRemediation/Resource/DatabaseCommandPolicy.js.map +1 -0
  359. package/build/dist/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.js +1139 -0
  360. package/build/dist/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.js.map +1 -0
  361. package/build/dist/Utils/AiRemediation/Resource/DatabaseQueryRedactor.js +1060 -0
  362. package/build/dist/Utils/AiRemediation/Resource/DatabaseQueryRedactor.js.map +1 -0
  363. package/build/dist/Utils/AiRemediation/Resource/DockerCliGrammar.js +1301 -0
  364. package/build/dist/Utils/AiRemediation/Resource/DockerCliGrammar.js.map +1 -0
  365. package/build/dist/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.js +370 -0
  366. package/build/dist/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.js.map +1 -0
  367. package/build/dist/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.js +498 -0
  368. package/build/dist/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.js.map +1 -0
  369. package/build/dist/Utils/AiRemediation/Resource/GovcCommandPolicy.js +1565 -0
  370. package/build/dist/Utils/AiRemediation/Resource/GovcCommandPolicy.js.map +1 -0
  371. package/build/dist/Utils/AiRemediation/Resource/HostCommandPolicy.js +2206 -0
  372. package/build/dist/Utils/AiRemediation/Resource/HostCommandPolicy.js.map +1 -0
  373. package/build/dist/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.js +1129 -0
  374. package/build/dist/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.js.map +1 -0
  375. package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicy.js +565 -0
  376. package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicy.js.map +1 -0
  377. package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.js +408 -0
  378. package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.js.map +1 -0
  379. package/build/dist/Utils/AiRemediation/Resource/ResourceOutputRedactor.js +1910 -0
  380. package/build/dist/Utils/AiRemediation/Resource/ResourceOutputRedactor.js.map +1 -0
  381. package/build/dist/Utils/SessionReplay/SessionReplayBudgetMetricType.js +160 -0
  382. package/build/dist/Utils/SessionReplay/SessionReplayBudgetMetricType.js.map +1 -0
  383. package/package.json +1 -1
@@ -682,6 +682,12 @@ export default class RemediationPlanRunner {
682
682
  suggestionType: true,
683
683
  commandPlan: true,
684
684
  verificationWindowMinutes: true,
685
+ /*
686
+ * So a resource round's notes name its resource: this sweep's
687
+ * own (describeStrandedSource) and settleInterruptedExecution's.
688
+ */
689
+ resourceType: true,
690
+ resourceId: true,
685
691
  },
686
692
  limit: 100,
687
693
  skip: 0,
@@ -745,7 +751,7 @@ export default class RemediationPlanRunner {
745
751
 
746
752
  await this.postFeedItem({
747
753
  suggestion,
748
- markdown: `⚡ **Auto Remediation Rule "${suggestion.ruleNameSnapshot || "Auto Remediation Rule"}": AI planning did not complete** (${reason}) — no runbook was proposed.`,
754
+ markdown: `⚡ **${this.describeStrandedSource(suggestion)}: AI planning did not complete** (${reason}) — no runbook was proposed.`,
749
755
  pingWorkspace: false,
750
756
  });
751
757
  } catch (error) {
@@ -756,6 +762,24 @@ export default class RemediationPlanRunner {
756
762
  }
757
763
  }
758
764
 
765
+ /*
766
+ * How the sweeper's feed note names a stranded round. A resource round
767
+ * (a Docker host's, a database server's, ...) names its resource —
768
+ * 'AI remediation for Docker host "web-1"' — as the execution runner's
769
+ * notes do, never "Auto Remediation Rule ...": only a resource round
770
+ * carries resourceType and resourceId (a rule round never sets them).
771
+ * Every other round keeps its label unchanged.
772
+ */
773
+ private static describeStrandedSource(
774
+ suggestion: AutoRemediationSuggestion,
775
+ ): string {
776
+ if (suggestion.resourceType && suggestion.resourceId) {
777
+ return suggestion.ruleNameSnapshot || "AI remediation";
778
+ }
779
+
780
+ return `Auto Remediation Rule "${suggestion.ruleNameSnapshot || "Auto Remediation Rule"}"`;
781
+ }
782
+
759
783
  // Why a Planning suggestion counts as stranded — or null if it does not.
760
784
  private static async getStrandedReason(
761
785
  suggestion: AutoRemediationSuggestion,
@@ -0,0 +1,577 @@
1
+ import ObjectID from "../../../../Types/ObjectID";
2
+ import { JSONObject } from "../../../../Types/JSON";
3
+ import RunnerJobOrigin from "../../../../Types/Runbook/RunnerJobOrigin";
4
+ import AiResourceType, {
5
+ AI_RESOURCE_TYPE_INFO,
6
+ ALL_AI_RESOURCE_TYPES,
7
+ AiResourceTypeInfo,
8
+ isAiResourceType,
9
+ } from "../../../../Types/ResourceAiAgent/AiResourceType";
10
+ import {
11
+ DEFAULT_RESOURCE_COMMAND_TIMEOUT_MS,
12
+ MAX_RESOURCE_COMMANDS_PER_INVESTIGATION,
13
+ MAX_RESOURCE_COMMAND_TIMEOUT_MS,
14
+ ResourceAiAccessStatus,
15
+ ResourceCommandTier,
16
+ } from "../../../../Types/ResourceAiAgent/ResourceAiAccess";
17
+ import ResourceCommandPolicy from "../../../../Utils/AiRemediation/Resource/ResourceCommandPolicy";
18
+ import { ResourceCommandPolicyResult } from "../../../../Utils/AiRemediation/Resource/ResourceCommandPolicyCore";
19
+ import { redactResourceCommandOutput } from "../../../../Utils/AiRemediation/Resource/ResourceOutputRedactor";
20
+ import KubectlWaitBudget, {
21
+ KubectlWaitBudgetResult,
22
+ MIN_KUBECTL_TIMEOUT_MS,
23
+ } from "../../../../Utils/AiRemediation/KubectlWaitBudget";
24
+ import { describeResourceNoun } from "../../../Services/ResourceAiAccessService";
25
+ import { ToolCallOutcome } from "../Toolbox/Index";
26
+ import { ToolArgs } from "../Toolbox/ToolTypes";
27
+ import { ObservabilityAssistantExtraTool } from "../Chat/ObservabilityAssistant";
28
+ import ResourceCommandJobRunner, {
29
+ RESOURCE_COMMAND_CLAIM_TIMEOUT_MS,
30
+ RESOURCE_COMMAND_OUTPUT_TRUNCATED_SUFFIX,
31
+ ResourceCommandJobOutcome,
32
+ ResourceCommandRunState,
33
+ } from "./ResourceCommandJobRunner";
34
+ import {
35
+ INFRASTRUCTURE_RESULT_UNKNOWN_EVENT_PREFIX,
36
+ LIST_INFRASTRUCTURE_ACCESS_TOOL_NAME,
37
+ RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME,
38
+ } from "./ResourceAccessToolNames";
39
+ import ResourceAccessContext, {
40
+ describeResource,
41
+ } from "./ResourceAccessContext";
42
+
43
+ /*
44
+ * The run-scoped, READ-ONLY infrastructure command toolkit for an
45
+ * investigation — the sibling of KubectlInvestigationToolkit for every
46
+ * resource with a resource AI agent (Docker and Podman hosts, Docker Swarm,
47
+ * Proxmox, VMware and Ceph clusters, database servers, hosts).
48
+ *
49
+ * One instance backs one AIRun. It is handed to the agent loop as
50
+ * extraTools and only ever offers resources the access service called
51
+ * investigation-ready. Three independent guards keep it read-only: the
52
+ * resource command policy here refuses anything but the Read tier, the
53
+ * enqueue chokepoint (RunnerJobService.enqueueAiResourceCommand) refuses a
54
+ * non-Read investigation-origin job, and the agent refuses the same before
55
+ * it runs anything. "Read-only — nothing in your systems was changed" stays
56
+ * literally true.
57
+ *
58
+ * What else the toolkit owns, exactly as the kubectl one does:
59
+ * - the run's wall clock: every command's claim window and timeout are
60
+ * planned against the run's deadline (KubectlWaitBudget, which is
61
+ * generic in all but name) and a command the budget can no longer hold
62
+ * is refused before anything is enqueued;
63
+ * - what the model sees: output reaches it only through the shared
64
+ * ResourceCommandJobRunner redaction;
65
+ * - what counts as evidence: a command that never ran is a failed call
66
+ * that mints no citation, one whose result never came back is "result
67
+ * unknown" (never "did not run"), and one that ran and exited non-zero
68
+ * is still evidence, cited with rowCount 0;
69
+ * - a per-agent breaker: once one command goes unclaimed, that resource's
70
+ * agent is unreachable for the rest of the run.
71
+ */
72
+
73
+ export interface InfrastructureInvestigationToolkitOptions {
74
+ projectId: ObjectID;
75
+ aiRunId: ObjectID;
76
+ resources: Array<ResourceAiAccessStatus>;
77
+ maxCommands?: number | undefined;
78
+ /*
79
+ * Which readiness admits a resource. Investigations require the
80
+ * investigation switch; a remediation run may read a resource it is
81
+ * allowed to fix even when its operator left investigation off.
82
+ */
83
+ readinessCheck?: "investigation" | "remediation" | undefined;
84
+ /*
85
+ * When the run's wall-clock budget ends (epoch milliseconds). Every
86
+ * command's wait is planned to end before it; absent, commands are only
87
+ * bounded by their own timeouts.
88
+ */
89
+ runDeadlineAtMs?: number | undefined;
90
+ }
91
+
92
+ // A reason that already ends a sentence.
93
+ const SENTENCE_END_PATTERN: RegExp = /[.!?]$/;
94
+
95
+ // The agent an unclaimed command tripped the breaker for.
96
+ interface UnreachableAgent {
97
+ // The claim window the unclaimed command was given.
98
+ claimWindowMs: number;
99
+ }
100
+
101
+ export default class InfrastructureInvestigationToolkit {
102
+ private options: InfrastructureInvestigationToolkitOptions;
103
+ private commandsRun: number = 0;
104
+ // Agents (by id) that left a command unclaimed in this run.
105
+ private unreachableAgents: Map<string, UnreachableAgent> = new Map<
106
+ string,
107
+ UnreachableAgent
108
+ >();
109
+
110
+ public constructor(options: InfrastructureInvestigationToolkitOptions) {
111
+ this.options = options;
112
+ }
113
+
114
+ public getReadyResources(): Array<ResourceAiAccessStatus> {
115
+ return this.options.resources.filter(
116
+ (resource: ResourceAiAccessStatus): boolean => {
117
+ const isReady: boolean =
118
+ this.options.readinessCheck === "remediation"
119
+ ? resource.isRemediationReady
120
+ : resource.isInvestigationReady;
121
+
122
+ return (
123
+ isReady &&
124
+ isAiResourceType(resource.resourceType) &&
125
+ resource.agent !== null &&
126
+ Boolean(resource.agent.agentId)
127
+ );
128
+ },
129
+ );
130
+ }
131
+
132
+ public getCommandsRun(): number {
133
+ return this.commandsRun;
134
+ }
135
+
136
+ // Whether this resource's agent tripped the breaker.
137
+ public isResourceUnreachable(resourceId: string): boolean {
138
+ const resource: ResourceAiAccessStatus | undefined =
139
+ this.options.resources.find(
140
+ (candidate: ResourceAiAccessStatus): boolean => {
141
+ return candidate.resourceId === resourceId;
142
+ },
143
+ );
144
+
145
+ return Boolean(
146
+ resource?.agent && this.unreachableAgents.has(resource.agent.agentId),
147
+ );
148
+ }
149
+
150
+ /*
151
+ * No ready resource, no tools: the model is told in its context why each
152
+ * resource is unreachable, and a tool that always fails would only burn
153
+ * its budget.
154
+ */
155
+ public buildTools(): Array<ObservabilityAssistantExtraTool> {
156
+ if (this.getReadyResources().length === 0) {
157
+ return [];
158
+ }
159
+
160
+ return [this.buildListAccessTool(), this.buildRunCommandTool()];
161
+ }
162
+
163
+ // One line per ready resource: what it is, its id, and its programs.
164
+ private describeReadyResources(): string {
165
+ return this.getReadyResources()
166
+ .map((resource: ResourceAiAccessStatus): string => {
167
+ const info: AiResourceTypeInfo =
168
+ AI_RESOURCE_TYPE_INFO[resource.resourceType];
169
+
170
+ return `- ${describeResource(resource)} — resourceId: ${resource.resourceId} — programs: ${info.programs.join(
171
+ ", ",
172
+ )}`;
173
+ })
174
+ .join("\n");
175
+ }
176
+
177
+ /*
178
+ * The read-command cheat-sheet of every tool policy the ready resources
179
+ * use, once per policy (a Docker host and a Podman host share one).
180
+ */
181
+ private describeReadCommandGuides(): string {
182
+ const presentTypes: Array<AiResourceType> = ALL_AI_RESOURCE_TYPES.filter(
183
+ (type: AiResourceType): boolean => {
184
+ return this.getReadyResources().some(
185
+ (resource: ResourceAiAccessStatus): boolean => {
186
+ return resource.resourceType === type;
187
+ },
188
+ );
189
+ },
190
+ );
191
+
192
+ const byPolicy: Map<string, Array<AiResourceType>> = new Map<
193
+ string,
194
+ Array<AiResourceType>
195
+ >();
196
+
197
+ for (const type of presentTypes) {
198
+ const policyName: string = ResourceCommandPolicy.getToolPolicy(type).name;
199
+ byPolicy.set(policyName, [...(byPolicy.get(policyName) || []), type]);
200
+ }
201
+
202
+ const sections: Array<string> = [];
203
+
204
+ for (const types of byPolicy.values()) {
205
+ const heading: string = types
206
+ .map((type: AiResourceType): string => {
207
+ return AI_RESOURCE_TYPE_INFO[type].displayName;
208
+ })
209
+ .join(" / ");
210
+
211
+ sections.push(
212
+ `${heading} — read-only commands:\n${ResourceCommandPolicy.getReadCommandGuide(
213
+ types[0]!,
214
+ )}`,
215
+ );
216
+ }
217
+
218
+ return sections.join("\n\n");
219
+ }
220
+
221
+ private buildListAccessTool(): ObservabilityAssistantExtraTool {
222
+ return {
223
+ definition: {
224
+ name: LIST_INFRASTRUCTURE_ACCESS_TOOL_NAME,
225
+ description: `List the infrastructure resources (Docker and Podman hosts, Docker Swarm, Proxmox, VMware and Ceph clusters, database servers, hosts) linked to this signal that OneUptime AI may inspect with read-only commands through their AI agents, with their resourceId and the programs each one runs. Call this once before ${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME} if you are unsure which resource to target.`,
226
+ inputSchema: { type: "object", properties: {} },
227
+ },
228
+ execute: async (): Promise<ToolCallOutcome> => {
229
+ const ready: Array<ResourceAiAccessStatus> = this.getReadyResources();
230
+
231
+ const text: string = ready
232
+ .map((resource: ResourceAiAccessStatus): string => {
233
+ const info: AiResourceTypeInfo =
234
+ AI_RESOURCE_TYPE_INFO[resource.resourceType];
235
+
236
+ return `- resourceId: ${resource.resourceId} — ${describeResource(
237
+ resource,
238
+ )} — ${
239
+ this.isResourceUnreachable(resource.resourceId)
240
+ ? `UNREACHABLE for the rest of this investigation: its ${info.agentDisplayName} did not pick up an earlier command. Do not call ${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME} on it again.`
241
+ : `read-only commands via its ${info.agentDisplayName} (programs: ${info.programs.join(
242
+ ", ",
243
+ )})`
244
+ }`;
245
+ })
246
+ .join("\n");
247
+
248
+ return {
249
+ success: true,
250
+ textForLlm: text,
251
+ result: {
252
+ dataForLlm: text,
253
+ rowCount: ready.length,
254
+ citationLabel: "Infrastructure OneUptime AI can inspect",
255
+ redactionCount: 0,
256
+ isTruncated: false,
257
+ },
258
+ };
259
+ },
260
+ };
261
+ }
262
+
263
+ private buildRunCommandTool(): ObservabilityAssistantExtraTool {
264
+ const maxCommands: number =
265
+ this.options.maxCommands ?? MAX_RESOURCE_COMMANDS_PER_INVESTIGATION;
266
+
267
+ return {
268
+ definition: {
269
+ name: RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME,
270
+ description: `Run ONE read-only command on a linked infrastructure resource through its AI agent and get its output. The resources you may inspect:\n${this.describeReadyResources()}\n\nWhat each kind of resource accepts:\n\n${this.describeReadCommandGuides()}\n\nOne command per call, written as the program followed by its arguments — never a shell line: pipes, redirects, ;, &&, $( ) and sudo are refused. Anything that changes a resource is refused here — this is an investigation — and so is anything that would read credentials; credential-looking values (passwords, tokens, keys, connection strings) are redacted from every output before you see it, so do not spend commands on them. Keep output small (limit lines and tails). At most ${maxCommands} commands per investigation.`,
271
+ inputSchema: {
272
+ type: "object",
273
+ properties: {
274
+ resourceId: {
275
+ type: "string",
276
+ description: `The resourceId of the linked resource to inspect (from the context or ${LIST_INFRASTRUCTURE_ACCESS_TOOL_NAME}).`,
277
+ },
278
+ command: {
279
+ type: "string",
280
+ description:
281
+ 'The command, starting with one of the resource\'s programs, e.g. "docker ps -a", "docker logs --tail 100 web", "systemctl status nginx --no-pager", "pvesh get /cluster/resources", "ceph health detail" or "db sessions --limit 20". One line, no shell operators.',
282
+ },
283
+ rationale: {
284
+ type: "string",
285
+ description:
286
+ "One sentence on what you expect this command to tell you — shown to humans on the timeline.",
287
+ },
288
+ timeoutInMs: {
289
+ type: "number",
290
+ description: `Timeout in milliseconds (default ${DEFAULT_RESOURCE_COMMAND_TIMEOUT_MS}, max ${MAX_RESOURCE_COMMAND_TIMEOUT_MS}). Shortened automatically when the investigation's time budget is nearly spent.`,
291
+ },
292
+ },
293
+ required: ["resourceId", "command", "rationale"],
294
+ },
295
+ },
296
+ execute: async (args: JSONObject): Promise<ToolCallOutcome> => {
297
+ return this.runCommand(args, maxCommands);
298
+ },
299
+ };
300
+ }
301
+
302
+ private async runCommand(
303
+ args: JSONObject,
304
+ maxCommands: number,
305
+ ): Promise<ToolCallOutcome> {
306
+ if (this.commandsRun >= maxCommands) {
307
+ return this.failure(
308
+ `The per-investigation infrastructure command budget (${maxCommands} commands) is spent. Finish your analysis with what you have.`,
309
+ );
310
+ }
311
+
312
+ const resourceIdRaw: string | undefined = ToolArgs.getString(
313
+ args,
314
+ "resourceId",
315
+ );
316
+ const resource: ResourceAiAccessStatus | undefined =
317
+ this.getReadyResources().find(
318
+ (candidate: ResourceAiAccessStatus): boolean => {
319
+ return (
320
+ Boolean(resourceIdRaw) &&
321
+ candidate.resourceId.toLowerCase() === resourceIdRaw!.toLowerCase()
322
+ );
323
+ },
324
+ );
325
+
326
+ if (!resource || !resource.agent) {
327
+ return this.failure(
328
+ `resourceId is not one of the resources OneUptime AI may inspect for this signal. Use ${LIST_INFRASTRUCTURE_ACCESS_TOOL_NAME}.`,
329
+ );
330
+ }
331
+
332
+ const info: AiResourceTypeInfo =
333
+ AI_RESOURCE_TYPE_INFO[resource.resourceType];
334
+ const label: string = describeResource(resource);
335
+
336
+ const command: string = ToolArgs.getString(args, "command") || "";
337
+
338
+ if (!command) {
339
+ return this.failure("command is required.");
340
+ }
341
+
342
+ const policy: ResourceCommandPolicyResult =
343
+ ResourceCommandPolicy.evaluateCommand({
344
+ resourceType: resource.resourceType,
345
+ command,
346
+ });
347
+
348
+ if (policy.tier === ResourceCommandTier.Denied) {
349
+ return this.failure(
350
+ `Refused by the ${info.displayName} command policy: ${policy.reason}. Nothing was run. What ${label} accepts:\n${ResourceCommandPolicy.getReadCommandGuide(
351
+ resource.resourceType,
352
+ )}`,
353
+ `Refused by the ${info.displayName} command policy on ${label}. Nothing was run.`,
354
+ );
355
+ }
356
+
357
+ if (policy.tier !== ResourceCommandTier.Read) {
358
+ return this.failure(
359
+ `"${policy.displayCommand}" would change ${label} (${policy.tier}) and this is a read-only investigation, so it was NOT run. If a change is the fix, put it in your Suggested next steps for a human. Read-only commands for it:\n${ResourceCommandPolicy.getReadCommandGuide(
360
+ resource.resourceType,
361
+ )}`,
362
+ `"${policy.displayCommand}" would change ${label} and this is a read-only investigation, so it was not run.`,
363
+ );
364
+ }
365
+
366
+ /*
367
+ * The breaker: checked before the budget is planned, a command is spent
368
+ * or anything is enqueued, so a dead agent costs one claim window per
369
+ * run rather than one per command.
370
+ */
371
+ const tripped: UnreachableAgent | undefined = this.unreachableAgents.get(
372
+ resource.agent.agentId,
373
+ );
374
+
375
+ if (tripped !== undefined) {
376
+ return this.failure(
377
+ `The ${info.agentDisplayName} of ${label} did not pick up an earlier command within ${InfrastructureInvestigationToolkit.describeSeconds(
378
+ tripped.claimWindowMs,
379
+ )}, so the ${describeResourceNoun(resource.resourceType)} is treated as unreachable for the rest of this investigation. Nothing was run. Do not call ${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME} on it again: continue with OneUptime telemetry and say in **${ResourceAccessContext.REPORT_SECTION_HEADING}** that its ${info.agentDisplayName} did not respond.`,
380
+ `No command was run on ${label}: its ${info.agentDisplayName} did not pick up an earlier command, so it was unreachable for the rest of this investigation.`,
381
+ );
382
+ }
383
+
384
+ const requestedTimeoutInMs: number = ToolArgs.getNumber(
385
+ args,
386
+ "timeoutInMs",
387
+ {
388
+ defaultValue: DEFAULT_RESOURCE_COMMAND_TIMEOUT_MS,
389
+ min: MIN_KUBECTL_TIMEOUT_MS,
390
+ max: MAX_RESOURCE_COMMAND_TIMEOUT_MS,
391
+ },
392
+ );
393
+
394
+ /*
395
+ * Planned before the command counts against the budget or anything is
396
+ * enqueued: a command the run's wall clock can no longer hold is
397
+ * refused outright rather than started and abandoned.
398
+ */
399
+ const budget: KubectlWaitBudgetResult = KubectlWaitBudget.plan({
400
+ requestedTimeoutInMs,
401
+ maxClaimTimeoutInMs: RESOURCE_COMMAND_CLAIM_TIMEOUT_MS,
402
+ deadlineAtMs: this.options.runDeadlineAtMs,
403
+ });
404
+
405
+ if (!budget.ok) {
406
+ return this.failure(
407
+ `Not enough time is left in this investigation's budget to run another infrastructure command (about ${InfrastructureInvestigationToolkit.describeSeconds(
408
+ budget.refusal.remainingBudgetMs,
409
+ )} remain; a command needs at least ${InfrastructureInvestigationToolkit.describeSeconds(
410
+ budget.refusal.minimumBudgetMs,
411
+ )} including the wait for the agent). Nothing was run. Finish your analysis with what you have.`,
412
+ );
413
+ }
414
+
415
+ /*
416
+ * Counted once a job is enqueued, whether or not the agent then runs
417
+ * it: the per-investigation cap bounds the agent work this run can ask
418
+ * for, not the commands that happened to succeed.
419
+ */
420
+ this.commandsRun++;
421
+
422
+ let outcome: ResourceCommandJobOutcome;
423
+
424
+ try {
425
+ outcome = await ResourceCommandJobRunner.run({
426
+ projectId: this.options.projectId,
427
+ aiRunId: this.options.aiRunId,
428
+ origin: RunnerJobOrigin.AiInvestigation,
429
+ resourceType: resource.resourceType,
430
+ resourceId: new ObjectID(resource.resourceId),
431
+ targetResourceAiAgentId: new ObjectID(resource.agent.agentId),
432
+ /*
433
+ * The command exactly as evaluated above: the chokepoint and the
434
+ * agent evaluate the same text, so all three verdicts agree.
435
+ */
436
+ command,
437
+ stepId: `ai-investigation-resource-${this.commandsRun}`,
438
+ timeoutInMs: budget.plan.timeoutInMs,
439
+ claimTimeoutInMs: budget.plan.claimTimeoutInMs,
440
+ });
441
+ } catch (error) {
442
+ const message: string = redactResourceCommandOutput({
443
+ resourceType: resource.resourceType,
444
+ program: policy.program,
445
+ text: error instanceof Error ? error.message : String(error),
446
+ }).trim();
447
+
448
+ return this.failure(
449
+ `The command could not be run on ${label}: ${message} Continue with OneUptime telemetry and mention this in your report.`,
450
+ `No command was run on ${label}: it was refused before it reached the ${info.agentDisplayName}.`,
451
+ );
452
+ }
453
+
454
+ if (outcome.claimTimedOut === true) {
455
+ this.unreachableAgents.set(resource.agent.agentId, {
456
+ claimWindowMs: budget.plan.claimTimeoutInMs,
457
+ });
458
+
459
+ return this.failure(
460
+ `The command was NOT run on ${label}: its ${info.agentDisplayName} did not pick up "${outcome.displayCommand}" within ${InfrastructureInvestigationToolkit.describeSeconds(
461
+ budget.plan.claimTimeoutInMs,
462
+ )} (it may be offline, restarting or busy). The ${describeResourceNoun(
463
+ resource.resourceType,
464
+ )} is treated as unreachable for the rest of this investigation — do not call ${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME} on it again. Continue with OneUptime telemetry and say in **${ResourceAccessContext.REPORT_SECTION_HEADING}** that its ${info.agentDisplayName} did not respond.`,
465
+ `No command was run on ${label}: the ${info.agentDisplayName} did not pick up the command in time.`,
466
+ );
467
+ }
468
+
469
+ /*
470
+ * The agent took the command and no result came back: it may have run
471
+ * or not. There is no output to cite, so it is a failed call — but it
472
+ * is never recorded as "did not run", and the panel counts it apart.
473
+ */
474
+ if (outcome.runState === ResourceCommandRunState.Unknown) {
475
+ const reason: string = InfrastructureInvestigationToolkit.toSentence(
476
+ redactResourceCommandOutput({
477
+ resourceType: resource.resourceType,
478
+ program: policy.program,
479
+ text: outcome.errorMessage || "No result came back for this command.",
480
+ }),
481
+ );
482
+
483
+ return this.failure(
484
+ `No result came back from ${label} for "${outcome.displayCommand}": ${reason} Nothing from this command is evidence. Continue with OneUptime telemetry and mention in **${ResourceAccessContext.REPORT_SECTION_HEADING}** that its ${info.agentDisplayName} stopped responding.`,
485
+ `${INFRASTRUCTURE_RESULT_UNKNOWN_EVENT_PREFIX} the ${info.agentDisplayName} of ${label} took the command, but no result came back, so whether it ran is unknown.`,
486
+ );
487
+ }
488
+
489
+ /*
490
+ * Nothing reached the program, so there is nothing to cite: a refusal
491
+ * by the server or the agent, or a program that could not start. The
492
+ * model reads why; the persisted event carries only the category,
493
+ * because the agent's reason can name its configuration and every
494
+ * reader of the incident sees the event.
495
+ */
496
+ if (outcome.executed === false) {
497
+ const reason: string = InfrastructureInvestigationToolkit.toSentence(
498
+ redactResourceCommandOutput({
499
+ resourceType: resource.resourceType,
500
+ program: policy.program,
501
+ text:
502
+ outcome.errorMessage ||
503
+ "The job ended without running the command.",
504
+ }),
505
+ );
506
+
507
+ return this.failure(
508
+ `No result came back from ${label} for "${outcome.displayCommand}": ${reason} Nothing from this command is evidence. Continue with OneUptime telemetry and mention in **${ResourceAccessContext.REPORT_SECTION_HEADING}** that the command could not run.`,
509
+ `No result came back from ${label}: the command was refused, or the ${info.agentDisplayName} did not run it.`,
510
+ );
511
+ }
512
+
513
+ const text: string = ResourceCommandJobRunner.describeForLlm({
514
+ outcome,
515
+ resourceType: resource.resourceType,
516
+ });
517
+
518
+ return {
519
+ success: true,
520
+ textForLlm: text,
521
+ result: {
522
+ dataForLlm: text,
523
+ rowCount: outcome.succeeded ? 1 : 0,
524
+ citationLabel: InfrastructureInvestigationToolkit.getCitationLabel({
525
+ displayCommand: outcome.displayCommand,
526
+ resource,
527
+ }),
528
+ redactionCount: outcome.redactionCount ?? 0,
529
+ isTruncated:
530
+ outcome.isTruncated ??
531
+ outcome.output.endsWith(
532
+ RESOURCE_COMMAND_OUTPUT_TRUNCATED_SUFFIX.trim(),
533
+ ),
534
+ },
535
+ };
536
+ }
537
+
538
+ /*
539
+ * How a command that ran is cited: `docker ps -a` on Docker host "web-1".
540
+ * The report's evidence list and the panel show it as is.
541
+ */
542
+ public static getCitationLabel(data: {
543
+ displayCommand: string;
544
+ resource: ResourceAiAccessStatus;
545
+ }): string {
546
+ const displayName: string = isAiResourceType(data.resource.resourceType)
547
+ ? AI_RESOURCE_TYPE_INFO[data.resource.resourceType].displayName
548
+ : "resource";
549
+
550
+ return `\`${data.displayCommand}\` on ${displayName} "${data.resource.resourceName}"`;
551
+ }
552
+
553
+ // A redacted reason that ends a sentence.
554
+ private static toSentence(text: string): string {
555
+ const trimmed: string = text.trim();
556
+
557
+ return SENTENCE_END_PATTERN.test(trimmed) ? trimmed : `${trimmed}.`;
558
+ }
559
+
560
+ private static describeSeconds(milliseconds: number): string {
561
+ return `${Math.max(0, Math.round(milliseconds / 1000))}s`;
562
+ }
563
+
564
+ /*
565
+ * `errorMessage` is what the run's persisted event (and so every reader
566
+ * of the incident's investigation panel) carries; it defaults to the
567
+ * model's text, which is right for failures that name nothing beyond the
568
+ * command and the resource.
569
+ */
570
+ private failure(text: string, errorMessage?: string): ToolCallOutcome {
571
+ return {
572
+ success: false,
573
+ textForLlm: text,
574
+ errorMessage: errorMessage ?? text,
575
+ };
576
+ }
577
+ }