@oneuptime/common 14.0.9 → 14.0.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/Models/DatabaseModels/AutoRemediationSuggestion.ts +60 -0
- package/Models/DatabaseModels/CephCluster.ts +213 -0
- package/Models/DatabaseModels/DatabaseServer.ts +213 -0
- package/Models/DatabaseModels/DockerHost.ts +213 -0
- package/Models/DatabaseModels/DockerSwarmCluster.ts +213 -0
- package/Models/DatabaseModels/GlobalConfig.ts +1 -1
- package/Models/DatabaseModels/Host.ts +213 -0
- package/Models/DatabaseModels/Index.ts +2 -0
- package/Models/DatabaseModels/PodmanHost.ts +213 -0
- package/Models/DatabaseModels/ProxmoxCluster.ts +213 -0
- package/Models/DatabaseModels/ResourceAiAgent.ts +419 -0
- package/Models/DatabaseModels/RunnerJob.ts +144 -0
- package/Models/DatabaseModels/VMwareVCenter.ts +213 -0
- package/Server/API/AutoRemediationAPI.ts +392 -1
- package/Server/API/ResourceAiAccessAPI.ts +1669 -0
- package/Server/Infrastructure/Postgres/SchemaMigrations/1796300000000-AddResourceAiAgents.ts +423 -0
- package/Server/Infrastructure/Postgres/SchemaMigrations/Index.ts +2 -0
- package/Server/Infrastructure/Semaphore.ts +22 -0
- package/Server/Middleware/TelemetryIngest.ts +15 -5
- package/Server/Services/AnalyticsDatabaseService.ts +45 -1
- package/Server/Services/AutoRemediationRuleEngineService.ts +940 -112
- package/Server/Services/CephClusterService.ts +109 -1
- package/Server/Services/DatabaseServerService.ts +116 -1
- package/Server/Services/DockerHostService.ts +109 -1
- package/Server/Services/DockerSwarmClusterService.ts +113 -1
- package/Server/Services/HostService.ts +126 -4
- package/Server/Services/MetricRecordingRuleService.ts +47 -0
- package/Server/Services/PodmanHostService.ts +109 -1
- package/Server/Services/ProxmoxClusterService.ts +110 -1
- package/Server/Services/ResourceAiAccessService.ts +1439 -0
- package/Server/Services/ResourceAiAgentJobService.ts +202 -0
- package/Server/Services/ResourceAiAgentService.ts +2432 -0
- package/Server/Services/RunnerJobService.ts +611 -4
- package/Server/Services/TelemetryUsageBillingService.ts +28 -0
- package/Server/Services/TraceRecordingRuleService.ts +33 -0
- package/Server/Services/VMwareVCenterService.ts +110 -1
- package/Server/Utils/AI/Remediation/RemediationCommandTools.ts +1434 -32
- package/Server/Utils/AI/Remediation/RemediationExecutionRunner.ts +935 -21
- package/Server/Utils/AI/Remediation/RemediationPlanRunner.ts +25 -1
- package/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.ts +577 -0
- package/Server/Utils/AI/ResourceAccess/ResourceAccessContext.ts +271 -0
- package/Server/Utils/AI/ResourceAccess/ResourceAccessToolNames.ts +45 -0
- package/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.ts +964 -0
- package/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.ts +333 -0
- package/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.ts +873 -0
- package/Server/Utils/AI/SRE/AIInvestigationEngine.ts +72 -2
- package/Server/Utils/AI/SRE/AlertInvestigationRunner.ts +72 -5
- package/Server/Utils/AI/SRE/IncidentInvestigationRunner.ts +72 -5
- package/Server/Utils/AutoRemediation/CommandPlanExecutor.ts +396 -13
- package/Server/Utils/AutoRemediation/RemediationVerifier.ts +35 -2
- package/Server/Utils/Database/ProjectScopedReferenceValidator.ts +9 -1
- package/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.ts +920 -0
- package/Server/Utils/SessionReplay/SessionReplayUsage.ts +84 -6
- package/Server/Utils/Workspace/MicrosoftTeams/Actions/Alert.ts +35 -15
- package/Server/Utils/Workspace/MicrosoftTeams/Actions/AlertEpisode.ts +33 -15
- package/Server/Utils/Workspace/MicrosoftTeams/Actions/Auth.ts +21 -1
- package/Server/Utils/Workspace/MicrosoftTeams/Actions/Incident.ts +298 -224
- package/Server/Utils/Workspace/MicrosoftTeams/Actions/IncidentEpisode.ts +35 -15
- package/Server/Utils/Workspace/MicrosoftTeams/Actions/ScheduledMaintenance.ts +389 -175
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeams.ts +366 -143
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.ts +163 -0
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.ts +396 -0
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.ts +418 -0
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.ts +152 -0
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.ts +251 -0
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.ts +214 -0
- package/Tests/App/Dashboard/ClusterAccessNotice.test.tsx +93 -0
- package/Tests/App/Dashboard/DatabaseDocumentationMarkdown.test.ts +18 -6
- package/Tests/App/Dashboard/InvestigationInfrastructureTools.test.tsx +506 -0
- package/Tests/App/Dashboard/RemediationSuggestionCardDescription.test.tsx +209 -0
- package/Tests/App/Dashboard/RemediationSuggestionCardResource.test.tsx +317 -0
- package/Tests/App/Dashboard/ResourceAiAccessSettingsUtil.test.ts +921 -0
- package/Tests/App/Dashboard/ResourceAiAgentInstall.test.ts +860 -0
- package/Tests/App/Dashboard/ResourceAiAgentPage.test.tsx +1721 -0
- package/Tests/App/Dashboard/ResourceAiAgentStatus.test.ts +1002 -0
- package/Tests/App/Dashboard/ResourceAiInsightsPage.test.tsx +931 -0
- package/Tests/App/Dashboard/ResourceAiNavigation.test.tsx +575 -0
- package/Tests/App/Dashboard/RunbookStepTypeMaps.test.ts +38 -7
- package/Tests/Models/DatabaseModels/DatabaseServerModels.test.ts +33 -1
- package/Tests/Models/DatabaseModels/ResourceAiAccessColumns.test.ts +738 -0
- package/Tests/Models/DatabaseModels/ResourceAiAgentModel.test.ts +842 -0
- package/Tests/Server/API/AutoRemediationApproveResourceRound.test.ts +835 -0
- package/Tests/Server/API/ResourceAiAccessAPI.test.ts +2730 -0
- package/Tests/Server/Infrastructure/Postgres/AddDatabaseServerTablesMigration.test.ts +60 -0
- package/Tests/Server/Infrastructure/Postgres/AddResourceAiAgentsMigration.test.ts +659 -0
- package/Tests/Server/Infrastructure/SemaphoreMutex.test.ts +31 -0
- package/Tests/Server/Middleware/TelemetryIngestBrowserKey.test.ts +63 -5
- package/Tests/Server/Middleware/TelemetryIngestKubernetesAgentRunnerPinnedKey.test.ts +103 -5
- package/Tests/Server/Middleware/TelemetryIngestKubernetesAgentRunnerRateLimit.test.ts +19 -1
- package/Tests/Server/Services/AutoRemediationResourceRuleEngine.test.ts +1279 -0
- package/Tests/Server/Services/DatabaseServerService.test.ts +55 -0
- package/Tests/Server/Services/GroupTelemetryUsageExcludeNames.test.ts +299 -0
- package/Tests/Server/Services/HostServiceFindOrCreateMemo.test.ts +44 -0
- package/Tests/Server/Services/MonitorProbeServiceIntervalScheduling.test.ts +45 -6
- package/Tests/Server/Services/RecordingRuleReservedMetricName.test.ts +171 -0
- package/Tests/Server/Services/ResourceAiAccessChangeAuthorization.test.ts +256 -0
- package/Tests/Server/Services/ResourceAiAccessService.test.ts +1373 -0
- package/Tests/Server/Services/ResourceAiAgentJobService.test.ts +375 -0
- package/Tests/Server/Services/ResourceAiAgentServiceHelpers.test.ts +1114 -0
- package/Tests/Server/Services/ResourceAiAgentServiceLifecycle.test.ts +1050 -0
- package/Tests/Server/Services/ResourceAiAgentServiceRegister.test.ts +2251 -0
- package/Tests/Server/Services/ResourceAiSettingsCreate.test.ts +373 -0
- package/Tests/Server/Services/ResourceAiSettingsPermission.test.ts +1042 -0
- package/Tests/Server/Services/ResourceServiceDeleteCleansUpAi.test.ts +319 -0
- package/Tests/Server/Services/RunnerJobEnqueueKubectl.test.ts +105 -0
- package/Tests/Server/Services/RunnerJobEnqueueResourceCommand.test.ts +986 -0
- package/Tests/Server/Services/RunnerJobResourceCommandLane.test.ts +211 -0
- package/Tests/Server/Services/RunnerJobResourceCommandTimeoutAndRedaction.test.ts +352 -0
- package/Tests/Server/Services/TelemetryUsageBillingSloExclusion.test.ts +218 -0
- package/Tests/Server/TestingUtils/Services/FakeRunnerJobCount.ts +145 -0
- package/Tests/Server/Utils/AI/InvestigationInfrastructureAccessWiring.test.ts +453 -0
- package/Tests/Server/Utils/AI/InvestigationInfrastructureReport.test.ts +721 -0
- package/Tests/Server/Utils/AI/RemediationCommandTools.test.ts +94 -1
- package/Tests/Server/Utils/AI/RemediationCommandToolsResource.test.ts +1530 -0
- package/Tests/Server/Utils/AI/RemediationExecutionRunnerResourceMode.test.ts +1249 -0
- package/Tests/Server/Utils/AI/RemediationPlanRunner.test.ts +117 -0
- package/Tests/Server/Utils/AI/RemediationResourceCopyParity.test.ts +463 -0
- package/Tests/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.test.ts +722 -0
- package/Tests/Server/Utils/AI/ResourceAccess/ResourceAccessContext.test.ts +266 -0
- package/Tests/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.test.ts +1220 -0
- package/Tests/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.test.ts +569 -0
- package/Tests/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.test.ts +830 -0
- package/Tests/Server/Utils/AutoRemediation/CommandPlanExecutorResource.test.ts +717 -0
- package/Tests/Server/Utils/AutoRemediation/RemediationVerifierResourceFollowUp.test.ts +338 -0
- package/Tests/Server/Utils/Monitor/Criteria/SessionReplayBudgetTemplateCriteria.test.ts +422 -0
- package/Tests/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.test.ts +1647 -0
- package/Tests/Server/Utils/SessionReplay/SessionReplayUsage.test.ts +143 -1
- package/Tests/Server/Utils/Telemetry/TelemetryIngestionKeyGuard.test.ts +33 -3
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsAccountNotLinked.test.ts +1093 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.test.ts +1049 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.test.ts +1676 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCards.test.ts +2884 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.test.ts +2490 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateSubmit.test.ts +2293 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateSubmitServerTimezone.test.ts +386 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.test.ts +1477 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.test.ts +1967 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsStripHtmlTags.test.ts +92 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsSubmittedFormRemoval.test.ts +1819 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.test.ts +1033 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezoneServerZone.test.ts +610 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeamsBotMessageHandling.test.ts +2610 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeamsCreateCommandsEndToEnd.test.ts +4672 -0
- package/Tests/Server/Utils/Workspace/WorkspaceCreateProjectReferences.test.ts +190 -65
- package/Tests/Types/AutoRemediation/AiRemediationCommandPlan.test.ts +10 -4
- package/Tests/Types/AutoRemediation/AiRemediationCommandPlanResourceCommand.test.ts +305 -0
- package/Tests/Types/Monitor/Recommendation/MonitorRecommendationCatalog.test.ts +277 -13
- package/Tests/Types/Monitor/Recommendation/MonitorRecommendationUtil.test.ts +113 -0
- package/Tests/Types/Monitor/RumAlertTemplates.test.ts +773 -2
- package/Tests/Types/ResourceAiAgent/AiResourceType.test.ts +419 -0
- package/Tests/Types/ResourceAiAgent/ResourceAiAccess.test.ts +517 -0
- package/Tests/Types/Runbook/RunbookStepType.test.ts +147 -8
- package/Tests/UI/Rum/RecordingHealthDashboard.test.tsx +769 -1
- package/Tests/Utils/AiRemediation/Resource/CephCommandPolicy.test.ts +1653 -0
- package/Tests/Utils/AiRemediation/Resource/CephOutputRedaction.test.ts +204 -0
- package/Tests/Utils/AiRemediation/Resource/DatabaseCommandPolicy.test.ts +1077 -0
- package/Tests/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.test.ts +558 -0
- package/Tests/Utils/AiRemediation/Resource/DatabaseQueryRedactor.test.ts +834 -0
- package/Tests/Utils/AiRemediation/Resource/DockerCliGrammar.test.ts +747 -0
- package/Tests/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.test.ts +1259 -0
- package/Tests/Utils/AiRemediation/Resource/DockerOutputRedaction.test.ts +386 -0
- package/Tests/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.test.ts +782 -0
- package/Tests/Utils/AiRemediation/Resource/GovcCommandPolicy.test.ts +2259 -0
- package/Tests/Utils/AiRemediation/Resource/HostCommandPolicy.test.ts +2019 -0
- package/Tests/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.test.ts +2497 -0
- package/Tests/Utils/AiRemediation/Resource/ResourceCommandPolicy.test.ts +1305 -0
- package/Tests/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.test.ts +585 -0
- package/Tests/Utils/AiRemediation/Resource/ResourceOutputRedactor.test.ts +334 -0
- package/Tests/Utils/AiRemediation/Resource/ResourceOutputRedactorCopyParity.test.ts +98 -0
- package/Tests/Utils/AiRemediation/Resource/ResourcePolicyImportClosure.test.ts +391 -0
- package/Tests/Utils/AiRemediation/ResourceAiAgentPolicyCopyParity.test.ts +497 -0
- package/Tests/Utils/SessionReplay/SessionReplayBudgetMetricType.test.ts +383 -0
- package/Types/AI/ResourceAiAccessApi.ts +140 -0
- package/Types/AI/ResourceAiAccessPermissions.ts +126 -0
- package/Types/AutoRemediation/AiRemediationCommandPlan.ts +183 -2
- package/Types/Monitor/Recommendation/MonitorRecommendationCatalog.ts +55 -22
- package/Types/Monitor/Recommendation/MonitorRecommendationTypes.ts +35 -3
- package/Types/Monitor/RumAlertTemplates.ts +351 -4
- package/Types/ResourceAiAgent/AiResourceType.ts +310 -0
- package/Types/ResourceAiAgent/ResourceAiAccess.ts +574 -0
- package/Types/Rum/SessionReplayBudgetMetricType.ts +47 -0
- package/Types/Runbook/RunbookStepType.ts +29 -0
- package/Types/Telemetry/TelemetryIngestSurface.ts +14 -2
- package/Utils/AI/InvestigationReport.ts +121 -15
- package/Utils/AiRemediation/Resource/CephCommandPolicy.ts +1936 -0
- package/Utils/AiRemediation/Resource/DatabaseCommandPolicy.ts +166 -0
- package/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.ts +1532 -0
- package/Utils/AiRemediation/Resource/DatabaseQueryRedactor.ts +1394 -0
- package/Utils/AiRemediation/Resource/DockerCliGrammar.ts +1839 -0
- package/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.ts +490 -0
- package/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.ts +687 -0
- package/Utils/AiRemediation/Resource/GovcCommandPolicy.ts +2096 -0
- package/Utils/AiRemediation/Resource/HostCommandPolicy.ts +2982 -0
- package/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.ts +1679 -0
- package/Utils/AiRemediation/Resource/ResourceCommandPolicy.ts +851 -0
- package/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.ts +593 -0
- package/Utils/AiRemediation/Resource/ResourceOutputRedactor.ts +2812 -0
- package/Utils/SessionReplay/SessionReplayBudgetMetricType.ts +184 -0
- package/build/dist/Models/DatabaseModels/AutoRemediationSuggestion.js +62 -0
- package/build/dist/Models/DatabaseModels/AutoRemediationSuggestion.js.map +1 -1
- package/build/dist/Models/DatabaseModels/CephCluster.js +219 -0
- package/build/dist/Models/DatabaseModels/CephCluster.js.map +1 -1
- package/build/dist/Models/DatabaseModels/DatabaseServer.js +219 -0
- package/build/dist/Models/DatabaseModels/DatabaseServer.js.map +1 -1
- package/build/dist/Models/DatabaseModels/DockerHost.js +219 -0
- package/build/dist/Models/DatabaseModels/DockerHost.js.map +1 -1
- package/build/dist/Models/DatabaseModels/DockerSwarmCluster.js +219 -0
- package/build/dist/Models/DatabaseModels/DockerSwarmCluster.js.map +1 -1
- package/build/dist/Models/DatabaseModels/GlobalConfig.js +1 -1
- package/build/dist/Models/DatabaseModels/GlobalConfig.js.map +1 -1
- package/build/dist/Models/DatabaseModels/Host.js +219 -0
- package/build/dist/Models/DatabaseModels/Host.js.map +1 -1
- package/build/dist/Models/DatabaseModels/Index.js +2 -0
- package/build/dist/Models/DatabaseModels/Index.js.map +1 -1
- package/build/dist/Models/DatabaseModels/PodmanHost.js +219 -0
- package/build/dist/Models/DatabaseModels/PodmanHost.js.map +1 -1
- package/build/dist/Models/DatabaseModels/ProxmoxCluster.js +219 -0
- package/build/dist/Models/DatabaseModels/ProxmoxCluster.js.map +1 -1
- package/build/dist/Models/DatabaseModels/ResourceAiAgent.js +439 -0
- package/build/dist/Models/DatabaseModels/ResourceAiAgent.js.map +1 -0
- package/build/dist/Models/DatabaseModels/RunnerJob.js +145 -0
- package/build/dist/Models/DatabaseModels/RunnerJob.js.map +1 -1
- package/build/dist/Models/DatabaseModels/VMwareVCenter.js +219 -0
- package/build/dist/Models/DatabaseModels/VMwareVCenter.js.map +1 -1
- package/build/dist/Server/API/AutoRemediationAPI.js +231 -1
- package/build/dist/Server/API/AutoRemediationAPI.js.map +1 -1
- package/build/dist/Server/API/ResourceAiAccessAPI.js +1119 -0
- package/build/dist/Server/API/ResourceAiAccessAPI.js.map +1 -0
- package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/1796300000000-AddResourceAiAgents.js +168 -0
- package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/1796300000000-AddResourceAiAgents.js.map +1 -0
- package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/Index.js +2 -0
- package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/Index.js.map +1 -1
- package/build/dist/Server/Infrastructure/Semaphore.js +6 -0
- package/build/dist/Server/Infrastructure/Semaphore.js.map +1 -1
- package/build/dist/Server/Middleware/TelemetryIngest.js +12 -5
- package/build/dist/Server/Middleware/TelemetryIngest.js.map +1 -1
- package/build/dist/Server/Services/AnalyticsDatabaseService.js +26 -1
- package/build/dist/Server/Services/AnalyticsDatabaseService.js.map +1 -1
- package/build/dist/Server/Services/AutoRemediationRuleEngineService.js +601 -3
- package/build/dist/Server/Services/AutoRemediationRuleEngineService.js.map +1 -1
- package/build/dist/Server/Services/CephClusterService.js +104 -0
- package/build/dist/Server/Services/CephClusterService.js.map +1 -1
- package/build/dist/Server/Services/DatabaseServerService.js +99 -0
- package/build/dist/Server/Services/DatabaseServerService.js.map +1 -1
- package/build/dist/Server/Services/DockerHostService.js +104 -0
- package/build/dist/Server/Services/DockerHostService.js.map +1 -1
- package/build/dist/Server/Services/DockerSwarmClusterService.js +104 -0
- package/build/dist/Server/Services/DockerSwarmClusterService.js.map +1 -1
- package/build/dist/Server/Services/HostService.js +114 -2
- package/build/dist/Server/Services/HostService.js.map +1 -1
- package/build/dist/Server/Services/MetricRecordingRuleService.js +46 -0
- package/build/dist/Server/Services/MetricRecordingRuleService.js.map +1 -1
- package/build/dist/Server/Services/PodmanHostService.js +104 -0
- package/build/dist/Server/Services/PodmanHostService.js.map +1 -1
- package/build/dist/Server/Services/ProxmoxClusterService.js +104 -0
- package/build/dist/Server/Services/ProxmoxClusterService.js.map +1 -1
- package/build/dist/Server/Services/ResourceAiAccessService.js +1028 -0
- package/build/dist/Server/Services/ResourceAiAccessService.js.map +1 -0
- package/build/dist/Server/Services/ResourceAiAgentJobService.js +173 -0
- package/build/dist/Server/Services/ResourceAiAgentJobService.js.map +1 -0
- package/build/dist/Server/Services/ResourceAiAgentService.js +1687 -0
- package/build/dist/Server/Services/ResourceAiAgentService.js.map +1 -0
- package/build/dist/Server/Services/RunnerJobService.js +344 -5
- package/build/dist/Server/Services/RunnerJobService.js.map +1 -1
- package/build/dist/Server/Services/TelemetryUsageBillingService.js +26 -0
- package/build/dist/Server/Services/TelemetryUsageBillingService.js.map +1 -1
- package/build/dist/Server/Services/TraceRecordingRuleService.js +37 -0
- package/build/dist/Server/Services/TraceRecordingRuleService.js.map +1 -1
- package/build/dist/Server/Services/VMwareVCenterService.js +104 -0
- package/build/dist/Server/Services/VMwareVCenterService.js.map +1 -1
- package/build/dist/Server/Utils/AI/Remediation/RemediationCommandTools.js +1029 -30
- package/build/dist/Server/Utils/AI/Remediation/RemediationCommandTools.js.map +1 -1
- package/build/dist/Server/Utils/AI/Remediation/RemediationExecutionRunner.js +633 -28
- package/build/dist/Server/Utils/AI/Remediation/RemediationExecutionRunner.js.map +1 -1
- package/build/dist/Server/Utils/AI/Remediation/RemediationPlanRunner.js +21 -1
- package/build/dist/Server/Utils/AI/Remediation/RemediationPlanRunner.js.map +1 -1
- package/build/dist/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.js +330 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.js.map +1 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessContext.js +160 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessContext.js.map +1 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessToolNames.js +33 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessToolNames.js.map +1 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.js +640 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.js.map +1 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.js +226 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.js.map +1 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.js +600 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.js.map +1 -0
- package/build/dist/Server/Utils/AI/SRE/AIInvestigationEngine.js +50 -4
- package/build/dist/Server/Utils/AI/SRE/AIInvestigationEngine.js.map +1 -1
- package/build/dist/Server/Utils/AI/SRE/AlertInvestigationRunner.js +55 -4
- package/build/dist/Server/Utils/AI/SRE/AlertInvestigationRunner.js.map +1 -1
- package/build/dist/Server/Utils/AI/SRE/IncidentInvestigationRunner.js +55 -4
- package/build/dist/Server/Utils/AI/SRE/IncidentInvestigationRunner.js.map +1 -1
- package/build/dist/Server/Utils/AutoRemediation/CommandPlanExecutor.js +290 -13
- package/build/dist/Server/Utils/AutoRemediation/CommandPlanExecutor.js.map +1 -1
- package/build/dist/Server/Utils/AutoRemediation/RemediationVerifier.js +29 -1
- package/build/dist/Server/Utils/AutoRemediation/RemediationVerifier.js.map +1 -1
- package/build/dist/Server/Utils/Database/ProjectScopedReferenceValidator.js +9 -1
- package/build/dist/Server/Utils/Database/ProjectScopedReferenceValidator.js.map +1 -1
- package/build/dist/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.js +627 -0
- package/build/dist/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.js.map +1 -0
- package/build/dist/Server/Utils/SessionReplay/SessionReplayUsage.js +60 -5
- package/build/dist/Server/Utils/SessionReplay/SessionReplayUsage.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Alert.js +19 -15
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Alert.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/AlertEpisode.js +19 -15
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/AlertEpisode.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Auth.js +17 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Auth.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Incident.js +183 -209
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Incident.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/IncidentEpisode.js +19 -15
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/IncidentEpisode.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/ScheduledMaintenance.js +233 -167
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/ScheduledMaintenance.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeams.js +263 -110
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeams.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.js +129 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.js.map +1 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.js +266 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.js.map +1 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.js +258 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.js.map +1 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.js +104 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.js.map +1 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.js +183 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.js.map +1 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.js +123 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.js.map +1 -0
- package/build/dist/Types/AI/ResourceAiAccessApi.js +30 -0
- package/build/dist/Types/AI/ResourceAiAccessApi.js.map +1 -0
- package/build/dist/Types/AI/ResourceAiAccessPermissions.js +98 -0
- package/build/dist/Types/AI/ResourceAiAccessPermissions.js.map +1 -0
- package/build/dist/Types/AutoRemediation/AiRemediationCommandPlan.js +116 -29
- package/build/dist/Types/AutoRemediation/AiRemediationCommandPlan.js.map +1 -1
- package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationCatalog.js +47 -21
- package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationCatalog.js.map +1 -1
- package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationTypes.js +5 -0
- package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationTypes.js.map +1 -1
- package/build/dist/Types/Monitor/RumAlertTemplates.js +199 -3
- package/build/dist/Types/Monitor/RumAlertTemplates.js.map +1 -1
- package/build/dist/Types/ResourceAiAgent/AiResourceType.js +242 -0
- package/build/dist/Types/ResourceAiAgent/AiResourceType.js.map +1 -0
- package/build/dist/Types/ResourceAiAgent/ResourceAiAccess.js +275 -0
- package/build/dist/Types/ResourceAiAgent/ResourceAiAccess.js.map +1 -0
- package/build/dist/Types/Rum/SessionReplayBudgetMetricType.js +45 -0
- package/build/dist/Types/Rum/SessionReplayBudgetMetricType.js.map +1 -0
- package/build/dist/Types/Runbook/RunbookStepType.js +27 -0
- package/build/dist/Types/Runbook/RunbookStepType.js.map +1 -1
- package/build/dist/Types/Telemetry/TelemetryIngestSurface.js +13 -2
- package/build/dist/Types/Telemetry/TelemetryIngestSurface.js.map +1 -1
- package/build/dist/Utils/AI/InvestigationReport.js +71 -11
- package/build/dist/Utils/AI/InvestigationReport.js.map +1 -1
- package/build/dist/Utils/AiRemediation/Resource/CephCommandPolicy.js +1462 -0
- package/build/dist/Utils/AiRemediation/Resource/CephCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/DatabaseCommandPolicy.js +122 -0
- package/build/dist/Utils/AiRemediation/Resource/DatabaseCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.js +1139 -0
- package/build/dist/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/DatabaseQueryRedactor.js +1060 -0
- package/build/dist/Utils/AiRemediation/Resource/DatabaseQueryRedactor.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/DockerCliGrammar.js +1301 -0
- package/build/dist/Utils/AiRemediation/Resource/DockerCliGrammar.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.js +370 -0
- package/build/dist/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.js +498 -0
- package/build/dist/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/GovcCommandPolicy.js +1565 -0
- package/build/dist/Utils/AiRemediation/Resource/GovcCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/HostCommandPolicy.js +2206 -0
- package/build/dist/Utils/AiRemediation/Resource/HostCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.js +1129 -0
- package/build/dist/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicy.js +565 -0
- package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.js +408 -0
- package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/ResourceOutputRedactor.js +1910 -0
- package/build/dist/Utils/AiRemediation/Resource/ResourceOutputRedactor.js.map +1 -0
- package/build/dist/Utils/SessionReplay/SessionReplayBudgetMetricType.js +160 -0
- package/build/dist/Utils/SessionReplay/SessionReplayBudgetMetricType.js.map +1 -0
- package/package.json +1 -1
|
@@ -24,10 +24,47 @@ import {
|
|
|
24
24
|
MAX_COMMAND_TIMEOUT_MS,
|
|
25
25
|
MAX_PLAN_COMMANDS,
|
|
26
26
|
MIN_COMMAND_TIMEOUT_MS,
|
|
27
|
+
RESOURCE_ALLOWLIST_SUMMARY,
|
|
28
|
+
RESOURCE_ALWAYS_ASKS_SUMMARY,
|
|
29
|
+
RESOURCE_NEVER_RUNS_SUMMARY,
|
|
30
|
+
RESOURCE_RISKIER_CHANGES_SUMMARY,
|
|
31
|
+
RESOURCE_SAFE_CHANGES_SUMMARY,
|
|
32
|
+
RESOURCE_UNATTENDED_ROUND_BECOMES_PROPOSAL_SUMMARY,
|
|
27
33
|
getKubectlAlwaysAsksSummary,
|
|
28
34
|
getKubectlRiskierChangesSummary,
|
|
29
35
|
getKubectlSafeChangesSummary,
|
|
30
36
|
} from "../../../../Types/AutoRemediation/AiRemediationCommandPlan";
|
|
37
|
+
import AiResourceType, {
|
|
38
|
+
AI_RESOURCE_TYPE_INFO,
|
|
39
|
+
ALL_AI_RESOURCE_TYPES,
|
|
40
|
+
AiResourceTypeInfo,
|
|
41
|
+
isAiResourceType,
|
|
42
|
+
} from "../../../../Types/ResourceAiAgent/AiResourceType";
|
|
43
|
+
import {
|
|
44
|
+
MAX_RESOURCE_COMMAND_TIMEOUT_MS,
|
|
45
|
+
RESOURCE_AI_ALLOW_WRITES_ENV,
|
|
46
|
+
RESOURCE_AI_WRITE_TARGETS_ENV,
|
|
47
|
+
ResourceAiAccessGap,
|
|
48
|
+
ResourceAiAccessStatus,
|
|
49
|
+
ResourceAiAgentPosture,
|
|
50
|
+
ResourceAiRemediationMode,
|
|
51
|
+
ResourceCommandTier,
|
|
52
|
+
isUnattendedResourceRemediationMode,
|
|
53
|
+
} from "../../../../Types/ResourceAiAgent/ResourceAiAccess";
|
|
54
|
+
import ResourceCommandPolicy from "../../../../Utils/AiRemediation/Resource/ResourceCommandPolicy";
|
|
55
|
+
import {
|
|
56
|
+
ResourceAutoExecutionVerdict,
|
|
57
|
+
ResourceCommandPolicyResult,
|
|
58
|
+
} from "../../../../Utils/AiRemediation/Resource/ResourceCommandPolicyCore";
|
|
59
|
+
import ResourceAiAccessService, {
|
|
60
|
+
describeResourceNoun,
|
|
61
|
+
} from "../../../Services/ResourceAiAccessService";
|
|
62
|
+
import ResourceCommandJobRunner, {
|
|
63
|
+
RESOURCE_COMMAND_CLAIM_TIMEOUT_MS,
|
|
64
|
+
ResourceCommandJobOutcome,
|
|
65
|
+
ResourceCommandRunState,
|
|
66
|
+
} from "../ResourceAccess/ResourceCommandJobRunner";
|
|
67
|
+
import { RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME } from "../ResourceAccess/ResourceAccessToolNames";
|
|
31
68
|
import {
|
|
32
69
|
KUBECTL_ALLOW_NODE_OPERATIONS_ENV,
|
|
33
70
|
KUBECTL_WRITE_NAMESPACES_ENV,
|
|
@@ -61,6 +98,11 @@ import AutoRemediationRuleEngineService, {
|
|
|
61
98
|
ClusterBreakerState,
|
|
62
99
|
ClusterRoundHold,
|
|
63
100
|
MAX_AUTO_EXECUTIONS_PER_RULE_PER_HOUR,
|
|
101
|
+
RESOURCE_BREAKER_LOCK_NAMESPACE,
|
|
102
|
+
ResourceBreakerState,
|
|
103
|
+
ResourceRoundHold,
|
|
104
|
+
doesResourceModeRunRoundUnattended,
|
|
105
|
+
getResourceBreakerLockKey,
|
|
64
106
|
} from "../../../Services/AutoRemediationRuleEngineService";
|
|
65
107
|
import AutoRemediationSuggestionService from "../../../Services/AutoRemediationSuggestionService";
|
|
66
108
|
import KubernetesClusterAiAccessService from "../../../Services/KubernetesClusterAiAccessService";
|
|
@@ -69,6 +111,7 @@ import RunbookCredentialService from "../../../Services/RunbookCredentialService
|
|
|
69
111
|
import RunnerJobService, {
|
|
70
112
|
isTerminalAgentJobStatus,
|
|
71
113
|
MAX_AI_COMMAND_JOBS_PER_PROJECT_PER_HOUR,
|
|
114
|
+
MAX_AI_RESOURCE_COMMAND_JOBS_PER_PROJECT_PER_HOUR,
|
|
72
115
|
} from "../../../Services/RunnerJobService";
|
|
73
116
|
import RunnerService from "../../../Services/RunnerService";
|
|
74
117
|
import QueryHelper from "../../../Types/Database/QueryHelper";
|
|
@@ -213,6 +256,39 @@ const CLUSTER_BREAKER_LOCK_NAMESPACE: string = "AutoRemediationClusterBreaker";
|
|
|
213
256
|
const CLUSTER_BREAKER_LOCK_TIMEOUT_MS: number = 60_000;
|
|
214
257
|
const CLUSTER_BREAKER_LOCK_ACQUIRE_TIMEOUT_MS: number = 20_000;
|
|
215
258
|
|
|
259
|
+
/*
|
|
260
|
+
* What the model is told when a resource command ran but did not finish
|
|
261
|
+
* (no exit code): the change may have landed before the agent stopped it.
|
|
262
|
+
*/
|
|
263
|
+
const RESOURCE_UNFINISHED_CHECK_FIRST: string = `The command did not finish (the resource's AI agent stopped it before it reported an exit code), so it MAY have changed the resource. Before you reissue this command, or run anything that depends on it, check with a read (${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME}) whether it took effect. Do NOT resend it blindly.`;
|
|
264
|
+
|
|
265
|
+
/*
|
|
266
|
+
* The key a resource target, and the commands aimed at it, are matched by:
|
|
267
|
+
* the type and the (case-insensitive) id.
|
|
268
|
+
*/
|
|
269
|
+
function getResourceKey(
|
|
270
|
+
resourceType: string | undefined,
|
|
271
|
+
resourceId: string | undefined,
|
|
272
|
+
): string {
|
|
273
|
+
if (!resourceType || !resourceId) {
|
|
274
|
+
return "";
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
return `${resourceType}:${resourceId.toLowerCase()}`;
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
// 'Docker host "web-1"' — how the toolkit names a resource in copy.
|
|
281
|
+
function describeResourceLabel(data: {
|
|
282
|
+
resourceType?: string | undefined;
|
|
283
|
+
resourceName?: string | undefined;
|
|
284
|
+
}): string {
|
|
285
|
+
return `${
|
|
286
|
+
isAiResourceType(data.resourceType)
|
|
287
|
+
? describeResourceNoun(data.resourceType)
|
|
288
|
+
: "resource"
|
|
289
|
+
} "${data.resourceName || "(unknown)"}"`;
|
|
290
|
+
}
|
|
291
|
+
|
|
216
292
|
export type RemediationCommandMode = "Suggest" | "FullAuto";
|
|
217
293
|
|
|
218
294
|
/*
|
|
@@ -315,6 +391,37 @@ export interface RemediationCommandToolkitOptions {
|
|
|
315
391
|
| undefined;
|
|
316
392
|
}
|
|
317
393
|
| undefined;
|
|
394
|
+
/*
|
|
395
|
+
* A resource round's one target: an infrastructure resource (a Docker or
|
|
396
|
+
* Podman host, a Docker Swarm, Proxmox, VMware or Ceph cluster, a
|
|
397
|
+
* database server, a host) whose AI page allows remediation, reached
|
|
398
|
+
* through its resource AI agent. Present (non-empty) only on a resource
|
|
399
|
+
* round, which then offers ResourceCommand steps — and only those.
|
|
400
|
+
*/
|
|
401
|
+
resourceTargets?: Array<ResourceAiAccessStatus> | undefined;
|
|
402
|
+
/*
|
|
403
|
+
* clusterHold for a resource: how the run's FIRST change on the resource
|
|
404
|
+
* checks, under the per-resource breaker lock, that no other AI run holds
|
|
405
|
+
* it (AutoRemediationRuleEngineService.findRoundHoldingResource).
|
|
406
|
+
*/
|
|
407
|
+
resourceHold?:
|
|
408
|
+
| {
|
|
409
|
+
anyOrder: boolean;
|
|
410
|
+
subject?:
|
|
411
|
+
| {
|
|
412
|
+
incidentId?: ObjectID | undefined;
|
|
413
|
+
alertId?: ObjectID | undefined;
|
|
414
|
+
}
|
|
415
|
+
| undefined;
|
|
416
|
+
}
|
|
417
|
+
| undefined;
|
|
418
|
+
/*
|
|
419
|
+
* A resource round's number for its signal (1 when absent). The live
|
|
420
|
+
* mode is re-checked against it before every change: Automatic runs only
|
|
421
|
+
* round 1 unattended, so a follow-up that started under Bypass approval
|
|
422
|
+
* stops running changes once the resource is on Automatic.
|
|
423
|
+
*/
|
|
424
|
+
resourceRoundNumber?: number | undefined;
|
|
318
425
|
}
|
|
319
426
|
|
|
320
427
|
interface CommandArgsParseResult {
|
|
@@ -334,6 +441,11 @@ export default class RemediationCommandToolkit {
|
|
|
334
441
|
* treats them as if they had never been targets.
|
|
335
442
|
*/
|
|
336
443
|
private revokedClusterIds: Set<string> = new Set<string>();
|
|
444
|
+
/*
|
|
445
|
+
* The same for resource targets (by getResourceKey): their AI page
|
|
446
|
+
* stopped allowing this run to change them mid-run.
|
|
447
|
+
*/
|
|
448
|
+
private revokedResourceKeys: Set<string> = new Set<string>();
|
|
337
449
|
/*
|
|
338
450
|
* Commands this run sent to a Runner that never reached kubectl. They are
|
|
339
451
|
* not executed commands, but their jobs exist: their sequence numbers
|
|
@@ -364,8 +476,40 @@ export default class RemediationCommandToolkit {
|
|
|
364
476
|
public getCommandsNeedingApproval(): Array<RemediationCommandNeedingApproval> {
|
|
365
477
|
return this.commandsNeedingApproval.filter(
|
|
366
478
|
(kept: RemediationCommandNeedingApproval) => {
|
|
367
|
-
return
|
|
368
|
-
kept.command.kubernetesClusterId || ""
|
|
479
|
+
return (
|
|
480
|
+
!this.revokedClusterIds.has(kept.command.kubernetesClusterId || "") &&
|
|
481
|
+
!this.revokedResourceKeys.has(
|
|
482
|
+
getResourceKey(kept.command.resourceType, kept.command.resourceId),
|
|
483
|
+
)
|
|
484
|
+
);
|
|
485
|
+
},
|
|
486
|
+
);
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
/*
|
|
490
|
+
* Is this a resource round? Fixed at construction (from the targets it was
|
|
491
|
+
* given, not the ones still allowed), so the tools the model was handed
|
|
492
|
+
* never change shape mid-run.
|
|
493
|
+
*/
|
|
494
|
+
public isResourceRound(): boolean {
|
|
495
|
+
return (this.options.resourceTargets || []).length > 0;
|
|
496
|
+
}
|
|
497
|
+
|
|
498
|
+
/*
|
|
499
|
+
* The resources this run may still change: remediation-ready, with an AI
|
|
500
|
+
* agent, and not revoked by a live re-check.
|
|
501
|
+
*/
|
|
502
|
+
public getResourceTargets(): Array<ResourceAiAccessStatus> {
|
|
503
|
+
return (this.options.resourceTargets || []).filter(
|
|
504
|
+
(resource: ResourceAiAccessStatus): boolean => {
|
|
505
|
+
return (
|
|
506
|
+
resource.isRemediationReady &&
|
|
507
|
+
isAiResourceType(resource.resourceType) &&
|
|
508
|
+
resource.agent !== null &&
|
|
509
|
+
Boolean(resource.agent?.agentId) &&
|
|
510
|
+
!this.revokedResourceKeys.has(
|
|
511
|
+
getResourceKey(resource.resourceType, resource.resourceId),
|
|
512
|
+
)
|
|
369
513
|
);
|
|
370
514
|
},
|
|
371
515
|
);
|
|
@@ -407,8 +551,9 @@ export default class RemediationCommandToolkit {
|
|
|
407
551
|
return {
|
|
408
552
|
definition: {
|
|
409
553
|
name: "list_command_targets",
|
|
410
|
-
description:
|
|
411
|
-
"List where this remediation may run commands:
|
|
554
|
+
description: this.isResourceRound()
|
|
555
|
+
? "List where this remediation may run commands: the infrastructure resource this round is about (stepType ResourceCommand runs ONE command on it through its own AI agent), with its resourceId, programs, remediation mode, write scope and command allowlist. Call this before composing any command."
|
|
556
|
+
: "List where this remediation may run commands: Runners (Bash runs on the Runner's host; SSH runs on an assigned credential's host) and Kubernetes clusters (Kubectl runs through the cluster's Kubernetes AI agent or Runner). Call this before composing any command.",
|
|
412
557
|
inputSchema: {
|
|
413
558
|
type: "object",
|
|
414
559
|
properties: {},
|
|
@@ -525,6 +670,49 @@ export default class RemediationCommandToolkit {
|
|
|
525
670
|
});
|
|
526
671
|
}
|
|
527
672
|
|
|
673
|
+
for (const resource of this.getResourceTargets()) {
|
|
674
|
+
const info: AiResourceTypeInfo =
|
|
675
|
+
AI_RESOURCE_TYPE_INFO[resource.resourceType];
|
|
676
|
+
|
|
677
|
+
rows.push({
|
|
678
|
+
targetType: "Resource",
|
|
679
|
+
resourceType: resource.resourceType,
|
|
680
|
+
resourceId: resource.resourceId,
|
|
681
|
+
name: resource.resourceName,
|
|
682
|
+
stepTypes: "ResourceCommand",
|
|
683
|
+
via: `its ${info.agentDisplayName}`,
|
|
684
|
+
programs: info.programs.join(", "),
|
|
685
|
+
// One idea per field, as for clusters (the serializer caps each).
|
|
686
|
+
remediationMode: this.describeResourceModeForLlm(resource),
|
|
687
|
+
safeChanges: RESOURCE_SAFE_CHANGES_SUMMARY,
|
|
688
|
+
riskierChanges: RESOURCE_RISKIER_CHANGES_SUMMARY,
|
|
689
|
+
alwaysNeedsAHuman: RESOURCE_ALWAYS_ASKS_SUMMARY,
|
|
690
|
+
neverRuns: RESOURCE_NEVER_RUNS_SUMMARY,
|
|
691
|
+
writeScope:
|
|
692
|
+
RemediationCommandToolkit.describeResourceWriteScope(resource),
|
|
693
|
+
commandAllowlist:
|
|
694
|
+
resource.aiCommandAllowlist.length > 0
|
|
695
|
+
? resource.aiCommandAllowlist.join(" | ")
|
|
696
|
+
: "(none)",
|
|
697
|
+
allowlistMatching: RESOURCE_ALLOWLIST_SUMMARY,
|
|
698
|
+
});
|
|
699
|
+
}
|
|
700
|
+
|
|
701
|
+
if (rows.length === 0 && this.isResourceRound()) {
|
|
702
|
+
return {
|
|
703
|
+
success: true,
|
|
704
|
+
textForLlm:
|
|
705
|
+
"The infrastructure resource this round is about no longer allows AI remediation (its AI agent page changed during this run, or its AI agent went away). You cannot run or propose commands — say so in your analysis.",
|
|
706
|
+
result: {
|
|
707
|
+
dataForLlm: "(no command targets available)",
|
|
708
|
+
rowCount: 0,
|
|
709
|
+
citationLabel: "Available command targets",
|
|
710
|
+
redactionCount: 0,
|
|
711
|
+
isTruncated: false,
|
|
712
|
+
},
|
|
713
|
+
};
|
|
714
|
+
}
|
|
715
|
+
|
|
528
716
|
if (rows.length === 0) {
|
|
529
717
|
return {
|
|
530
718
|
success: true,
|
|
@@ -586,6 +774,126 @@ export default class RemediationCommandToolkit {
|
|
|
586
774
|
return "RequireApproval: every kubectl change is proposed for one-click approval";
|
|
587
775
|
}
|
|
588
776
|
|
|
777
|
+
/*
|
|
778
|
+
* describeClusterModeForLlm for a resource: the canonical
|
|
779
|
+
* ResourceAiRemediationMode semantics in one field of at most the
|
|
780
|
+
* serializer's 500 characters.
|
|
781
|
+
*/
|
|
782
|
+
private describeResourceModeForLlm(resource: ResourceAiAccessStatus): string {
|
|
783
|
+
if (
|
|
784
|
+
resource.aiRemediationMode === ResourceAiRemediationMode.BypassApproval
|
|
785
|
+
) {
|
|
786
|
+
return `BypassApproval: AI does not ask — every change the policy allows, safeChanges AND riskierChanges, runs without a human, except alwaysNeedsAHuman${
|
|
787
|
+
this.options.proposesRefusedCommands
|
|
788
|
+
? " (submit such a change anyway: it is refused, recorded, and proposed for one-click approval when this round ends if no other change ran)"
|
|
789
|
+
: ""
|
|
790
|
+
}; and ${RESOURCE_UNATTENDED_ROUND_BECOMES_PROPOSAL_SUMMARY}`;
|
|
791
|
+
}
|
|
792
|
+
|
|
793
|
+
if (resource.aiRemediationMode === ResourceAiRemediationMode.Automatic) {
|
|
794
|
+
return `Automatic: safeChanges run without a human; riskierChanges never run inline unless the commandAllowlist names their exact shape, and alwaysNeedsAHuman never does — ${
|
|
795
|
+
this.options.proposesRefusedCommands
|
|
796
|
+
? "submit it anyway: it is refused, recorded, and proposed for one-click approval when this round ends if no other change ran; put it in your written recommendations too"
|
|
797
|
+
: "put it in your written recommendations for a human"
|
|
798
|
+
}; and ${RESOURCE_UNATTENDED_ROUND_BECOMES_PROPOSAL_SUMMARY}`;
|
|
799
|
+
}
|
|
800
|
+
|
|
801
|
+
return "RequireApproval: every change is proposed for one-click approval";
|
|
802
|
+
}
|
|
803
|
+
|
|
804
|
+
/*
|
|
805
|
+
* Where the resource's AI agent lets AI-composed writes land, as it last
|
|
806
|
+
* reported (its posture): read-only unless ONEUPTIME_AI_ALLOW_WRITES is
|
|
807
|
+
* true, never its protected targets, and — with ONEUPTIME_AI_WRITE_TARGETS
|
|
808
|
+
* set — only the targets its globs name.
|
|
809
|
+
*/
|
|
810
|
+
public static describeResourceWriteScope(
|
|
811
|
+
resource: ResourceAiAccessStatus,
|
|
812
|
+
): string {
|
|
813
|
+
const posture: ResourceAiAgentPosture | null | undefined =
|
|
814
|
+
resource.agent?.posture;
|
|
815
|
+
|
|
816
|
+
if (!posture) {
|
|
817
|
+
return "not reported by the agent yet, so it runs no change";
|
|
818
|
+
}
|
|
819
|
+
|
|
820
|
+
if (posture.allowWrites !== true) {
|
|
821
|
+
return `read-only (${RESOURCE_AI_ALLOW_WRITES_ENV} is not true on the agent): it refuses every change`;
|
|
822
|
+
}
|
|
823
|
+
|
|
824
|
+
const parts: Array<string> = [
|
|
825
|
+
posture.writeTargets.length > 0
|
|
826
|
+
? `changes only targets matching ${RESOURCE_AI_WRITE_TARGETS_ENV}=${posture.writeTargets.join(",")} — name each target exactly`
|
|
827
|
+
: "changes any target its command policy allows",
|
|
828
|
+
];
|
|
829
|
+
|
|
830
|
+
if (posture.protectedTargets.length > 0) {
|
|
831
|
+
parts.push(
|
|
832
|
+
`never its protected targets (${posture.protectedTargets
|
|
833
|
+
.slice(0, 10)
|
|
834
|
+
.join(", ")}): itself and what it runs in`,
|
|
835
|
+
);
|
|
836
|
+
}
|
|
837
|
+
|
|
838
|
+
return parts.join("; ");
|
|
839
|
+
}
|
|
840
|
+
|
|
841
|
+
/*
|
|
842
|
+
* Why the resource's AI agent would refuse this write, going by the
|
|
843
|
+
* posture it reported — or null when it would not. The rule is the
|
|
844
|
+
* agent's own (ResourceCommandPolicy.getWriteScopeRefusal: read-only
|
|
845
|
+
* unless ONEUPTIME_AI_ALLOW_WRITES is true, never a protected target,
|
|
846
|
+
* only the ONEUPTIME_AI_WRITE_TARGETS globs when set), the one the enqueue
|
|
847
|
+
* chokepoint asks too. Reads and Denied commands are not this rule's
|
|
848
|
+
* business (the policy refuses the latter on its own).
|
|
849
|
+
*/
|
|
850
|
+
public static getResourceWriteScopeRefusal(data: {
|
|
851
|
+
resource: ResourceAiAccessStatus;
|
|
852
|
+
command: string;
|
|
853
|
+
}): string | null {
|
|
854
|
+
if (!isAiResourceType(data.resource.resourceType)) {
|
|
855
|
+
return "The resource type is unknown, so no change can run on it.";
|
|
856
|
+
}
|
|
857
|
+
|
|
858
|
+
const policy: ResourceCommandPolicyResult =
|
|
859
|
+
ResourceCommandPolicy.evaluateCommand({
|
|
860
|
+
resourceType: data.resource.resourceType,
|
|
861
|
+
command: data.command,
|
|
862
|
+
});
|
|
863
|
+
|
|
864
|
+
if (
|
|
865
|
+
policy.tier === ResourceCommandTier.Read ||
|
|
866
|
+
policy.tier === ResourceCommandTier.Denied
|
|
867
|
+
) {
|
|
868
|
+
return null;
|
|
869
|
+
}
|
|
870
|
+
|
|
871
|
+
const posture: ResourceAiAgentPosture | null | undefined =
|
|
872
|
+
data.resource.agent?.posture;
|
|
873
|
+
|
|
874
|
+
return ResourceCommandPolicy.getWriteScopeRefusal({
|
|
875
|
+
result: policy,
|
|
876
|
+
allowWrites: posture?.allowWrites === true,
|
|
877
|
+
writeTargets: posture?.writeTargets || [],
|
|
878
|
+
protectedTargets: posture?.protectedTargets || [],
|
|
879
|
+
resourceType: data.resource.resourceType,
|
|
880
|
+
});
|
|
881
|
+
}
|
|
882
|
+
|
|
883
|
+
/*
|
|
884
|
+
* What to do about a resource write-scope refusal, for the approve route
|
|
885
|
+
* and the model: the agent's own environment decides.
|
|
886
|
+
*/
|
|
887
|
+
public static getResourceScopeRefusalNextStep(
|
|
888
|
+
resourceType: AiResourceType,
|
|
889
|
+
): string {
|
|
890
|
+
const agentName: string = isAiResourceType(resourceType)
|
|
891
|
+
? AI_RESOURCE_TYPE_INFO[resourceType].agentDisplayName
|
|
892
|
+
: "resource's AI agent";
|
|
893
|
+
|
|
894
|
+
return `Dismiss the suggestion and let a new plan be composed, or change the ${agentName}'s write access (${RESOURCE_AI_ALLOW_WRITES_ENV} and ${RESOURCE_AI_WRITE_TARGETS_ENV} in its environment), restart it, and approve again.`;
|
|
895
|
+
}
|
|
896
|
+
|
|
589
897
|
/*
|
|
590
898
|
* Which kubectl changes the model may be offered on this cluster. A
|
|
591
899
|
* Runner that reported node operations off refuses every one, approved
|
|
@@ -932,6 +1240,29 @@ export default class RemediationCommandToolkit {
|
|
|
932
1240
|
*/
|
|
933
1241
|
|
|
934
1242
|
private buildExecuteTool(): ObservabilityAssistantExtraTool {
|
|
1243
|
+
if (this.isResourceRound()) {
|
|
1244
|
+
return {
|
|
1245
|
+
definition: {
|
|
1246
|
+
name: "execute_remediation_command",
|
|
1247
|
+
description: this.describeResourceExecuteTool(),
|
|
1248
|
+
inputSchema: {
|
|
1249
|
+
type: "object",
|
|
1250
|
+
properties: this.buildCommandSchemaProperties(),
|
|
1251
|
+
required: [
|
|
1252
|
+
"stepType",
|
|
1253
|
+
"resourceId",
|
|
1254
|
+
"command",
|
|
1255
|
+
"rationale",
|
|
1256
|
+
"expectedEffect",
|
|
1257
|
+
],
|
|
1258
|
+
},
|
|
1259
|
+
},
|
|
1260
|
+
execute: async (args: JSONObject): Promise<ToolCallOutcome> => {
|
|
1261
|
+
return this.executeCommand(args);
|
|
1262
|
+
},
|
|
1263
|
+
};
|
|
1264
|
+
}
|
|
1265
|
+
|
|
935
1266
|
return {
|
|
936
1267
|
definition: {
|
|
937
1268
|
name: "execute_remediation_command",
|
|
@@ -952,7 +1283,109 @@ export default class RemediationCommandToolkit {
|
|
|
952
1283
|
};
|
|
953
1284
|
}
|
|
954
1285
|
|
|
1286
|
+
/*
|
|
1287
|
+
* What each resource this round may change accepts, by tier — its tool
|
|
1288
|
+
* policy's own guide, once per policy.
|
|
1289
|
+
*/
|
|
1290
|
+
private describeResourceWriteGuides(): string {
|
|
1291
|
+
const sections: Array<string> = [];
|
|
1292
|
+
const seenPolicies: Set<string> = new Set<string>();
|
|
1293
|
+
|
|
1294
|
+
for (const type of ALL_AI_RESOURCE_TYPES) {
|
|
1295
|
+
const resources: Array<ResourceAiAccessStatus> = (
|
|
1296
|
+
this.options.resourceTargets || []
|
|
1297
|
+
).filter((resource: ResourceAiAccessStatus): boolean => {
|
|
1298
|
+
return resource.resourceType === type;
|
|
1299
|
+
});
|
|
1300
|
+
|
|
1301
|
+
if (resources.length === 0) {
|
|
1302
|
+
continue;
|
|
1303
|
+
}
|
|
1304
|
+
|
|
1305
|
+
const policyName: string = ResourceCommandPolicy.getToolPolicy(type).name;
|
|
1306
|
+
|
|
1307
|
+
if (seenPolicies.has(policyName)) {
|
|
1308
|
+
continue;
|
|
1309
|
+
}
|
|
1310
|
+
|
|
1311
|
+
seenPolicies.add(policyName);
|
|
1312
|
+
|
|
1313
|
+
sections.push(
|
|
1314
|
+
`${resources
|
|
1315
|
+
.map((resource: ResourceAiAccessStatus): string => {
|
|
1316
|
+
return `${AI_RESOURCE_TYPE_INFO[type].displayName} "${resource.resourceName}" (resourceId: ${resource.resourceId})`;
|
|
1317
|
+
})
|
|
1318
|
+
.join(
|
|
1319
|
+
", ",
|
|
1320
|
+
)} — changes it accepts:\n${ResourceCommandPolicy.getWriteCommandGuide(
|
|
1321
|
+
type,
|
|
1322
|
+
)}`,
|
|
1323
|
+
);
|
|
1324
|
+
}
|
|
1325
|
+
|
|
1326
|
+
return sections.join("\n\n");
|
|
1327
|
+
}
|
|
1328
|
+
|
|
1329
|
+
// execute_remediation_command's description on a resource round.
|
|
1330
|
+
private describeResourceExecuteTool(): string {
|
|
1331
|
+
return `Execute ONE remediation change immediately on the infrastructure resource this round is about, through its own AI agent: stepType ResourceCommand with the resource's resourceId. One command per call, written as the program followed by its arguments — never a shell line. This tool is for CHANGES only — a read-only command goes through ${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME} and is refused here. Safe changes run (${RESOURCE_SAFE_CHANGES_SUMMARY}). Riskier changes (${RESOURCE_RISKIER_CHANGES_SUMMARY}) are refused unless the resource's command allowlist names their exact shape (${RESOURCE_ALLOWLIST_SUMMARY}) or the resource bypasses approvals — ${
|
|
1332
|
+
this.options.proposesRefusedCommands
|
|
1333
|
+
? "a refused riskier change is recorded and proposed for one-click approval when this round ends, provided no other change ran"
|
|
1334
|
+
: "put those in your recommendations"
|
|
1335
|
+
}. Whatever the mode, ${RESOURCE_ALWAYS_ASKS_SUMMARY}; ${RESOURCE_UNATTENDED_ROUND_BECOMES_PROPOSAL_SUMMARY}; ${RESOURCE_NEVER_RUNS_SUMMARY}. A change the resource's AI agent would refuse (it runs read-only, or the target is protected or outside its writeScope in list_command_targets) is refused before it runs. At most ${MAX_AUTO_EXECUTED_COMMANDS_PER_RUN} commands may be sent per remediation. Provide a rollbackCommand whenever the command changes state and an undo exists.\n\n${this.describeResourceWriteGuides()}`;
|
|
1336
|
+
}
|
|
1337
|
+
|
|
1338
|
+
/*
|
|
1339
|
+
* The command schema of a resource round: ResourceCommand only, on the
|
|
1340
|
+
* round's resource — never a Runner, a cluster or a credential.
|
|
1341
|
+
*/
|
|
1342
|
+
private buildResourceCommandSchemaProperties(): JSONObject {
|
|
1343
|
+
return {
|
|
1344
|
+
resourceId: {
|
|
1345
|
+
type: "string",
|
|
1346
|
+
description:
|
|
1347
|
+
"The resource to change: its resourceId from list_command_targets.",
|
|
1348
|
+
},
|
|
1349
|
+
stepType: {
|
|
1350
|
+
type: "string",
|
|
1351
|
+
enum: ["ResourceCommand"],
|
|
1352
|
+
description:
|
|
1353
|
+
"ResourceCommand runs ONE command on the resource through its own AI agent.",
|
|
1354
|
+
},
|
|
1355
|
+
command: {
|
|
1356
|
+
type: "string",
|
|
1357
|
+
description:
|
|
1358
|
+
'The exact command, one line starting with one of the resource\'s programs (list_command_targets), e.g. "docker restart web", "docker service update --force api", "systemctl restart nginx", "pvesh create /nodes/pve1/qemu/100/status/start", "govc vm.power -on /DC/vm/web-01", "ceph osd in 3" or "db cancel-query 4242". No pipes, redirects, chaining, substitution or sudo.',
|
|
1359
|
+
},
|
|
1360
|
+
timeoutInMs: {
|
|
1361
|
+
type: "number",
|
|
1362
|
+
description: `Execution timeout in milliseconds (default ${Math.min(
|
|
1363
|
+
DEFAULT_COMMAND_TIMEOUT_MS,
|
|
1364
|
+
MAX_RESOURCE_COMMAND_TIMEOUT_MS,
|
|
1365
|
+
)}, max ${MAX_RESOURCE_COMMAND_TIMEOUT_MS}).`,
|
|
1366
|
+
},
|
|
1367
|
+
rationale: {
|
|
1368
|
+
type: "string",
|
|
1369
|
+
description:
|
|
1370
|
+
"Why this command remediates the incident — shown to humans verbatim.",
|
|
1371
|
+
},
|
|
1372
|
+
expectedEffect: {
|
|
1373
|
+
type: "string",
|
|
1374
|
+
description: "What you expect to observe if it works.",
|
|
1375
|
+
},
|
|
1376
|
+
rollbackCommand: {
|
|
1377
|
+
type: "string",
|
|
1378
|
+
description:
|
|
1379
|
+
"Optional undo command for the same resource, run unattended if verification later fails. Must pass the same policy, and — unless the resource bypasses approvals — be a safe change on ONE named object, e.g. docker start <container> after docker stop <container>, or systemctl start <unit> after systemctl stop <unit>.",
|
|
1380
|
+
},
|
|
1381
|
+
};
|
|
1382
|
+
}
|
|
1383
|
+
|
|
955
1384
|
private buildCommandSchemaProperties(): JSONObject {
|
|
1385
|
+
if (this.isResourceRound()) {
|
|
1386
|
+
return this.buildResourceCommandSchemaProperties();
|
|
1387
|
+
}
|
|
1388
|
+
|
|
956
1389
|
return {
|
|
957
1390
|
runnerId: {
|
|
958
1391
|
type: "string",
|
|
@@ -1074,6 +1507,44 @@ export default class RemediationCommandToolkit {
|
|
|
1074
1507
|
}
|
|
1075
1508
|
}
|
|
1076
1509
|
|
|
1510
|
+
if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
1511
|
+
/*
|
|
1512
|
+
* Reads belong to run_infrastructure_command, for the reason they
|
|
1513
|
+
* belong to run_kubectl on a cluster: sent here, a read would count
|
|
1514
|
+
* as an executed fix, take a breaker slot and could auto-resolve the
|
|
1515
|
+
* signal in the name of a change that changed nothing.
|
|
1516
|
+
*/
|
|
1517
|
+
if (command.resourceCommandTier === ResourceCommandTier.Read) {
|
|
1518
|
+
return this.failure(
|
|
1519
|
+
`"${command.command}" is read-only. Run it with ${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME}, which does not count as a remediation command — this tool is for changes only. Nothing was executed.`,
|
|
1520
|
+
);
|
|
1521
|
+
}
|
|
1522
|
+
|
|
1523
|
+
// The resource as its AI page stands NOW, as for a cluster.
|
|
1524
|
+
const liveRefusal: FullAutoRefusal | null =
|
|
1525
|
+
await this.refreshResourceTarget(command);
|
|
1526
|
+
|
|
1527
|
+
if (liveRefusal) {
|
|
1528
|
+
this.recordNeedingApproval(command, liveRefusal);
|
|
1529
|
+
return this.failure(liveRefusal.text);
|
|
1530
|
+
}
|
|
1531
|
+
|
|
1532
|
+
/*
|
|
1533
|
+
* The agent's write scope as it reports it NOW: a write it would
|
|
1534
|
+
* refuse is never enqueued — nor kept for a proposal, which no click
|
|
1535
|
+
* could make runnable.
|
|
1536
|
+
*/
|
|
1537
|
+
const liveResource: ResourceAiAccessStatus | undefined =
|
|
1538
|
+
this.findResourceTarget(command.resourceType, command.resourceId);
|
|
1539
|
+
const scopeRefusal: string | null = liveResource
|
|
1540
|
+
? this.getResourceCommandScopeRefusal(liveResource, command)
|
|
1541
|
+
: null;
|
|
1542
|
+
|
|
1543
|
+
if (scopeRefusal) {
|
|
1544
|
+
return this.failure(scopeRefusal);
|
|
1545
|
+
}
|
|
1546
|
+
}
|
|
1547
|
+
|
|
1077
1548
|
/*
|
|
1078
1549
|
* FullAuto gate: the full policy. Anything that is not AutoApproved is
|
|
1079
1550
|
* refused here; the model is told why so it can pick an allowlisted
|
|
@@ -1088,22 +1559,59 @@ export default class RemediationCommandToolkit {
|
|
|
1088
1559
|
|
|
1089
1560
|
command.policyVerdict = AiRemediationCommandPolicyVerdict.AutoApproved;
|
|
1090
1561
|
|
|
1091
|
-
|
|
1092
|
-
|
|
1093
|
-
|
|
1094
|
-
|
|
1095
|
-
|
|
1096
|
-
|
|
1097
|
-
|
|
1098
|
-
|
|
1099
|
-
|
|
1100
|
-
|
|
1101
|
-
|
|
1562
|
+
if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
1563
|
+
/*
|
|
1564
|
+
* The resource lane's own hourly brake, counted the way the enqueue
|
|
1565
|
+
* chokepoint counts it (ResourceCommand rows only): pre-checked here
|
|
1566
|
+
* so the model gets a useful refusal instead of a thrown one.
|
|
1567
|
+
*/
|
|
1568
|
+
const resourceJobsInLastHour: number = (
|
|
1569
|
+
await RunnerJobService.countBy({
|
|
1570
|
+
query: {
|
|
1571
|
+
projectId: this.options.projectId,
|
|
1572
|
+
origin: RunnerJobOrigin.AiRemediation,
|
|
1573
|
+
stepType: RunbookStepType.ResourceCommand,
|
|
1574
|
+
createdAt: QueryHelper.greaterThan(
|
|
1575
|
+
OneUptimeDate.getSomeHoursAgo(1),
|
|
1576
|
+
),
|
|
1577
|
+
},
|
|
1578
|
+
props: { isRoot: true },
|
|
1579
|
+
})
|
|
1580
|
+
).toNumber();
|
|
1102
1581
|
|
|
1103
|
-
|
|
1104
|
-
|
|
1105
|
-
|
|
1106
|
-
)
|
|
1582
|
+
if (
|
|
1583
|
+
resourceJobsInLastHour >=
|
|
1584
|
+
MAX_AI_RESOURCE_COMMAND_JOBS_PER_PROJECT_PER_HOUR
|
|
1585
|
+
) {
|
|
1586
|
+
return this.failure(
|
|
1587
|
+
`This project has hit its hourly limit on AI commands on its infrastructure (${MAX_AI_RESOURCE_COMMAND_JOBS_PER_PROJECT_PER_HOUR}). No further commands can run this hour. Summarize and hand off to a human.`,
|
|
1588
|
+
);
|
|
1589
|
+
}
|
|
1590
|
+
} else {
|
|
1591
|
+
/*
|
|
1592
|
+
* Project-wide hourly storm brake across the kubectl and Runner AI
|
|
1593
|
+
* command jobs, counted the way the enqueue chokepoints count it:
|
|
1594
|
+
* resource commands have their own brake above and never count here.
|
|
1595
|
+
*/
|
|
1596
|
+
const jobsInLastHour: number = (
|
|
1597
|
+
await RunnerJobService.countBy({
|
|
1598
|
+
query: {
|
|
1599
|
+
projectId: this.options.projectId,
|
|
1600
|
+
origin: RunnerJobOrigin.AiRemediation,
|
|
1601
|
+
stepType: QueryHelper.notEquals(RunbookStepType.ResourceCommand),
|
|
1602
|
+
createdAt: QueryHelper.greaterThan(
|
|
1603
|
+
OneUptimeDate.getSomeHoursAgo(1),
|
|
1604
|
+
),
|
|
1605
|
+
},
|
|
1606
|
+
props: { isRoot: true },
|
|
1607
|
+
})
|
|
1608
|
+
).toNumber();
|
|
1609
|
+
|
|
1610
|
+
if (jobsInLastHour >= MAX_AI_COMMAND_JOBS_PER_PROJECT_PER_HOUR) {
|
|
1611
|
+
return this.failure(
|
|
1612
|
+
`This project has hit its hourly AI-command limit (${MAX_AI_COMMAND_JOBS_PER_PROJECT_PER_HOUR}). No further commands can run this hour. Summarize and hand off to a human.`,
|
|
1613
|
+
);
|
|
1614
|
+
}
|
|
1107
1615
|
}
|
|
1108
1616
|
|
|
1109
1617
|
/*
|
|
@@ -1129,6 +1637,22 @@ export default class RemediationCommandToolkit {
|
|
|
1129
1637
|
clusterSlotLock = reservation.mutex;
|
|
1130
1638
|
}
|
|
1131
1639
|
|
|
1640
|
+
// The same for the run's first change on a resource.
|
|
1641
|
+
if (
|
|
1642
|
+
command.stepType === RunbookStepType.ResourceCommand &&
|
|
1643
|
+
!this.hasChangedResource(command.resourceType, command.resourceId)
|
|
1644
|
+
) {
|
|
1645
|
+
const reservation: ClusterSlotReservation =
|
|
1646
|
+
await this.reserveResourceSlot(command);
|
|
1647
|
+
|
|
1648
|
+
if (reservation.refusal) {
|
|
1649
|
+
this.recordNeedingApproval(command, reservation.refusal);
|
|
1650
|
+
return this.failure(reservation.refusal.text);
|
|
1651
|
+
}
|
|
1652
|
+
|
|
1653
|
+
clusterSlotLock = reservation.mutex;
|
|
1654
|
+
}
|
|
1655
|
+
|
|
1132
1656
|
const releaseClusterSlotLock: () => Promise<void> =
|
|
1133
1657
|
async (): Promise<void> => {
|
|
1134
1658
|
if (!clusterSlotLock) {
|
|
@@ -1194,6 +1718,8 @@ export default class RemediationCommandToolkit {
|
|
|
1194
1718
|
* whose wait broke): it may have run. Its citation says so.
|
|
1195
1719
|
*/
|
|
1196
1720
|
let kubectlResultUnknown: boolean = false;
|
|
1721
|
+
// The same for a resource command the agent took with no result back.
|
|
1722
|
+
let resourceResultUnknown: boolean = false;
|
|
1197
1723
|
|
|
1198
1724
|
try {
|
|
1199
1725
|
if (command.stepType === RunbookStepType.Kubectl) {
|
|
@@ -1306,8 +1832,103 @@ export default class RemediationCommandToolkit {
|
|
|
1306
1832
|
} else {
|
|
1307
1833
|
outcomeText = KubectlJobRunner.describeForLlm(outcome);
|
|
1308
1834
|
}
|
|
1309
|
-
} else {
|
|
1310
|
-
|
|
1835
|
+
} else if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
1836
|
+
/*
|
|
1837
|
+
* The kubectl lane's shape for a resource: enqueue through the
|
|
1838
|
+
* resource chokepoint (which re-runs the policy, the binding, the
|
|
1839
|
+
* switch and the agent's write scope), name the job on the record
|
|
1840
|
+
* BEFORE the wait, release the breaker slot, wait, and read the
|
|
1841
|
+
* finished job exactly as the investigation lane reads it
|
|
1842
|
+
* (ResourceCommandJobRunner: run state, redaction, access failure).
|
|
1843
|
+
*/
|
|
1844
|
+
const resourceType: AiResourceType =
|
|
1845
|
+
command.resourceType as AiResourceType;
|
|
1846
|
+
const resourceId: ObjectID = new ObjectID(command.resourceId!);
|
|
1847
|
+
|
|
1848
|
+
const job: RunnerJob = await RunnerJobService.enqueueAiResourceCommand({
|
|
1849
|
+
projectId: this.options.projectId,
|
|
1850
|
+
aiRunId: this.options.aiRunId,
|
|
1851
|
+
origin: RunnerJobOrigin.AiRemediation,
|
|
1852
|
+
autoRemediationSuggestionId: this.options.suggestionId,
|
|
1853
|
+
resourceType,
|
|
1854
|
+
resourceId,
|
|
1855
|
+
stepId: `${INLINE_COMMAND_STEP_ID_PREFIX}${command.sequence}`,
|
|
1856
|
+
targetResourceAiAgentId: new ObjectID(command.runnerId),
|
|
1857
|
+
command: command.command,
|
|
1858
|
+
timeoutInMs: command.timeoutInMs,
|
|
1859
|
+
claimTimeoutInMs: RESOURCE_COMMAND_CLAIM_TIMEOUT_MS,
|
|
1860
|
+
});
|
|
1861
|
+
|
|
1862
|
+
command.execution.runnerJobId = job.id?.toString();
|
|
1863
|
+
await this.persistPlanProgress();
|
|
1864
|
+
|
|
1865
|
+
// The job row is the breaker reservation: the next holder counts it.
|
|
1866
|
+
await afterEnqueue();
|
|
1867
|
+
|
|
1868
|
+
const terminalJob: RunnerJob = await this.waitForJobWithHeartbeat({
|
|
1869
|
+
jobId: job.id!,
|
|
1870
|
+
claimTimeoutInMs: RESOURCE_COMMAND_CLAIM_TIMEOUT_MS,
|
|
1871
|
+
executionTimeoutInMs: command.timeoutInMs,
|
|
1872
|
+
});
|
|
1873
|
+
|
|
1874
|
+
const outcome: ResourceCommandJobOutcome =
|
|
1875
|
+
await ResourceCommandJobRunner.readFinishedJob({
|
|
1876
|
+
job,
|
|
1877
|
+
terminalJob,
|
|
1878
|
+
command: command.command,
|
|
1879
|
+
resourceType,
|
|
1880
|
+
claimTimeoutInMs: RESOURCE_COMMAND_CLAIM_TIMEOUT_MS,
|
|
1881
|
+
executionTimeoutInMs: command.timeoutInMs,
|
|
1882
|
+
});
|
|
1883
|
+
|
|
1884
|
+
await ResourceCommandJobRunner.recordOutcomeOnResource({
|
|
1885
|
+
resourceType,
|
|
1886
|
+
resourceId,
|
|
1887
|
+
outcome,
|
|
1888
|
+
});
|
|
1889
|
+
|
|
1890
|
+
// Certainly never ran: a failed tool call, off the record.
|
|
1891
|
+
if (outcome.runState === ResourceCommandRunState.NotRun) {
|
|
1892
|
+
return await this.settleNeverRan(command, {
|
|
1893
|
+
displayCommand: outcome.displayCommand,
|
|
1894
|
+
errorMessage: outcome.errorMessage,
|
|
1895
|
+
claimTimedOut: outcome.claimTimedOut === true,
|
|
1896
|
+
});
|
|
1897
|
+
}
|
|
1898
|
+
|
|
1899
|
+
command.execution.status = outcome.succeeded
|
|
1900
|
+
? AiRemediationCommandExecutionStatus.Succeeded
|
|
1901
|
+
: AiRemediationCommandExecutionStatus.Failed;
|
|
1902
|
+
command.execution.completedAt =
|
|
1903
|
+
OneUptimeDate.getCurrentDate().toISOString();
|
|
1904
|
+
command.execution.exitCode = outcome.exitCode;
|
|
1905
|
+
command.execution.output = outcome.output;
|
|
1906
|
+
redactionCount = outcome.redactionCount ?? 0;
|
|
1907
|
+
isTruncated = outcome.isTruncated ?? false;
|
|
1908
|
+
if (!outcome.succeeded) {
|
|
1909
|
+
command.execution.errorMessage = outcome.errorMessage;
|
|
1910
|
+
}
|
|
1911
|
+
|
|
1912
|
+
if (outcome.runState === ResourceCommandRunState.Unknown) {
|
|
1913
|
+
resourceResultUnknown = true;
|
|
1914
|
+
outcomeText = this.describeResultUnknown(command, {
|
|
1915
|
+
displayCommand: outcome.displayCommand,
|
|
1916
|
+
reason: outcome.errorMessage,
|
|
1917
|
+
});
|
|
1918
|
+
} else if (!outcome.succeeded && typeof outcome.exitCode !== "number") {
|
|
1919
|
+
// The program ran but was stopped before it finished.
|
|
1920
|
+
outcomeText = `${ResourceCommandJobRunner.describeForLlm({
|
|
1921
|
+
outcome,
|
|
1922
|
+
resourceType,
|
|
1923
|
+
})}\n${RESOURCE_UNFINISHED_CHECK_FIRST}`;
|
|
1924
|
+
} else {
|
|
1925
|
+
outcomeText = ResourceCommandJobRunner.describeForLlm({
|
|
1926
|
+
outcome,
|
|
1927
|
+
resourceType,
|
|
1928
|
+
});
|
|
1929
|
+
}
|
|
1930
|
+
} else {
|
|
1931
|
+
const job: RunnerJob = await RunnerJobService.enqueueAiCommand({
|
|
1311
1932
|
projectId: this.options.projectId,
|
|
1312
1933
|
aiRunId: this.options.aiRunId,
|
|
1313
1934
|
autoRemediationSuggestionId: this.options.suggestionId,
|
|
@@ -1380,6 +2001,22 @@ export default class RemediationCommandToolkit {
|
|
|
1380
2001
|
});
|
|
1381
2002
|
}
|
|
1382
2003
|
|
|
2004
|
+
/*
|
|
2005
|
+
* The resource chokepoint refused the command (its policy, binding,
|
|
2006
|
+
* switch, brake and write-scope checks run before the row is
|
|
2007
|
+
* written): no job exists, so it certainly never ran.
|
|
2008
|
+
*/
|
|
2009
|
+
if (
|
|
2010
|
+
command.stepType === RunbookStepType.ResourceCommand &&
|
|
2011
|
+
!command.execution.runnerJobId
|
|
2012
|
+
) {
|
|
2013
|
+
return await this.settleNeverRan(command, {
|
|
2014
|
+
displayCommand: command.command,
|
|
2015
|
+
errorMessage: this.redactResourceText(command, message),
|
|
2016
|
+
claimTimedOut: false,
|
|
2017
|
+
});
|
|
2018
|
+
}
|
|
2019
|
+
|
|
1383
2020
|
command.execution.status = AiRemediationCommandExecutionStatus.Failed;
|
|
1384
2021
|
command.execution.completedAt =
|
|
1385
2022
|
OneUptimeDate.getCurrentDate().toISOString();
|
|
@@ -1396,6 +2033,17 @@ export default class RemediationCommandToolkit {
|
|
|
1396
2033
|
displayCommand: command.command,
|
|
1397
2034
|
reason: `Waiting for its result failed: ${message}`,
|
|
1398
2035
|
});
|
|
2036
|
+
} else if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
2037
|
+
// The same for a resource's agent.
|
|
2038
|
+
command.execution.errorMessage = this.redactResourceText(
|
|
2039
|
+
command,
|
|
2040
|
+
message,
|
|
2041
|
+
);
|
|
2042
|
+
resourceResultUnknown = true;
|
|
2043
|
+
outcomeText = this.describeResultUnknown(command, {
|
|
2044
|
+
displayCommand: command.command,
|
|
2045
|
+
reason: `Waiting for its result failed: ${message}`,
|
|
2046
|
+
});
|
|
1399
2047
|
} else {
|
|
1400
2048
|
outcomeText = `Command FAILED before completion: ${message}`;
|
|
1401
2049
|
}
|
|
@@ -1414,13 +2062,46 @@ export default class RemediationCommandToolkit {
|
|
|
1414
2062
|
? kubectlResultUnknown
|
|
1415
2063
|
? `Sent to cluster "${command.kubernetesClusterNameSnapshot}", result unknown: ${this.summarizeCommand(command.command)}`
|
|
1416
2064
|
: `Executed on cluster "${command.kubernetesClusterNameSnapshot}": ${this.summarizeCommand(command.command)}`
|
|
1417
|
-
:
|
|
2065
|
+
: command.stepType === RunbookStepType.ResourceCommand
|
|
2066
|
+
? resourceResultUnknown
|
|
2067
|
+
? `Sent to ${this.describeCommandResource(command)}, result unknown: ${this.summarizeCommand(command.command)}`
|
|
2068
|
+
: `Executed on ${this.describeCommandResource(command)}: ${this.summarizeCommand(command.command)}`
|
|
2069
|
+
: `Executed on Runner "${command.runnerNameSnapshot}": ${this.summarizeCommand(command.command)}`,
|
|
1418
2070
|
redactionCount,
|
|
1419
2071
|
isTruncated,
|
|
1420
2072
|
},
|
|
1421
2073
|
};
|
|
1422
2074
|
}
|
|
1423
2075
|
|
|
2076
|
+
// 'Docker host "web-1"' for a ResourceCommand step, from its snapshot.
|
|
2077
|
+
private describeCommandResource(command: AiRemediationCommand): string {
|
|
2078
|
+
const info: AiResourceTypeInfo | null = isAiResourceType(
|
|
2079
|
+
command.resourceType,
|
|
2080
|
+
)
|
|
2081
|
+
? AI_RESOURCE_TYPE_INFO[command.resourceType]
|
|
2082
|
+
: null;
|
|
2083
|
+
|
|
2084
|
+
return `${info ? info.displayName : "resource"} "${
|
|
2085
|
+
command.resourceNameSnapshot || command.resourceId || "(unknown)"
|
|
2086
|
+
}"`;
|
|
2087
|
+
}
|
|
2088
|
+
|
|
2089
|
+
// A resource program's text, through the resource redaction.
|
|
2090
|
+
private redactResourceText(
|
|
2091
|
+
command: AiRemediationCommand,
|
|
2092
|
+
text: string,
|
|
2093
|
+
): string {
|
|
2094
|
+
if (!isAiResourceType(command.resourceType)) {
|
|
2095
|
+
return ToolResultSerializer.redact(text).text;
|
|
2096
|
+
}
|
|
2097
|
+
|
|
2098
|
+
return ResourceCommandJobRunner.redactAndCap({
|
|
2099
|
+
output: text,
|
|
2100
|
+
resourceType: command.resourceType,
|
|
2101
|
+
program: command.command.trim().split(/\s+/)[0] || "",
|
|
2102
|
+
}).text;
|
|
2103
|
+
}
|
|
2104
|
+
|
|
1424
2105
|
/*
|
|
1425
2106
|
* A kubectl command whose job certainly never reached kubectl (no Runner
|
|
1426
2107
|
* claimed it, the server or the Runner refused it, kubectl could not
|
|
@@ -1444,6 +2125,23 @@ export default class RemediationCommandToolkit {
|
|
|
1444
2125
|
this.neverRanCount += 1;
|
|
1445
2126
|
await this.persistPlanProgress();
|
|
1446
2127
|
|
|
2128
|
+
if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
2129
|
+
const label: string = this.describeCommandResource(command);
|
|
2130
|
+
const noun: string = isAiResourceType(command.resourceType)
|
|
2131
|
+
? describeResourceNoun(command.resourceType)
|
|
2132
|
+
: "resource";
|
|
2133
|
+
|
|
2134
|
+
return this.failure(
|
|
2135
|
+
`"${data.displayCommand}" did NOT run on ${label}: ${
|
|
2136
|
+
data.errorMessage || "the job never reached the resource's AI agent."
|
|
2137
|
+
} Nothing changed on the ${noun} and nothing was recorded as executed. ${
|
|
2138
|
+
data.claimTimedOut
|
|
2139
|
+
? `Do NOT send more commands to this ${noun} in this run — its AI agent is not picking them up; say so in your analysis.`
|
|
2140
|
+
: "Do NOT resend the same command; fix what the refusal names, or put the change in your recommendations for a human."
|
|
2141
|
+
}`,
|
|
2142
|
+
);
|
|
2143
|
+
}
|
|
2144
|
+
|
|
1447
2145
|
return this.failure(
|
|
1448
2146
|
`"${data.displayCommand}" did NOT run on cluster "${
|
|
1449
2147
|
command.kubernetesClusterNameSnapshot || command.kubernetesClusterId
|
|
@@ -1465,6 +2163,31 @@ export default class RemediationCommandToolkit {
|
|
|
1465
2163
|
command: AiRemediationCommand,
|
|
1466
2164
|
data: { displayCommand: string; reason: string | undefined },
|
|
1467
2165
|
): string {
|
|
2166
|
+
if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
2167
|
+
const resourceReason: string = this.redactResourceText(
|
|
2168
|
+
command,
|
|
2169
|
+
(data.reason || "No result came back for this command.").trim(),
|
|
2170
|
+
).trim();
|
|
2171
|
+
const noun: string = isAiResourceType(command.resourceType)
|
|
2172
|
+
? describeResourceNoun(command.resourceType)
|
|
2173
|
+
: "resource";
|
|
2174
|
+
|
|
2175
|
+
return [
|
|
2176
|
+
`${data.displayCommand}`,
|
|
2177
|
+
`RESULT UNKNOWN on ${this.describeCommandResource(command)}: ${
|
|
2178
|
+
SENTENCE_END_PATTERN.test(resourceReason)
|
|
2179
|
+
? resourceReason
|
|
2180
|
+
: `${resourceReason}.`
|
|
2181
|
+
}`,
|
|
2182
|
+
`The command reached the ${noun}'s AI agent, so it MAY have run and changed the ${noun}. It stays on this round's record as a command that may have run, and verification judges it${
|
|
2183
|
+
command.rollbackCommand
|
|
2184
|
+
? "; if the service does not recover, its rollbackCommand is not run blind — a human is asked to check and undo it"
|
|
2185
|
+
: ""
|
|
2186
|
+
}.`,
|
|
2187
|
+
`Before you reissue this command, or run anything that depends on it, check with a read (${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME}) whether it took effect. Do NOT resend it blindly.`,
|
|
2188
|
+
].join("\n");
|
|
2189
|
+
}
|
|
2190
|
+
|
|
1468
2191
|
const reason: string = KubectlOutputRedactor.redact(
|
|
1469
2192
|
(data.reason || "No result came back for this command.").trim(),
|
|
1470
2193
|
).text;
|
|
@@ -1522,6 +2245,40 @@ export default class RemediationCommandToolkit {
|
|
|
1522
2245
|
return null;
|
|
1523
2246
|
}
|
|
1524
2247
|
|
|
2248
|
+
/*
|
|
2249
|
+
* getCommandScopeRefusal for a resource: whether its AI agent would
|
|
2250
|
+
* refuse this command or its rollback (getResourceWriteScopeRefusal),
|
|
2251
|
+
* worded for the model.
|
|
2252
|
+
*/
|
|
2253
|
+
private getResourceCommandScopeRefusal(
|
|
2254
|
+
resource: ResourceAiAccessStatus,
|
|
2255
|
+
command: Pick<AiRemediationCommand, "command" | "rollbackCommand">,
|
|
2256
|
+
): string | null {
|
|
2257
|
+
const forward: string | null =
|
|
2258
|
+
RemediationCommandToolkit.getResourceWriteScopeRefusal({
|
|
2259
|
+
resource,
|
|
2260
|
+
command: command.command,
|
|
2261
|
+
});
|
|
2262
|
+
|
|
2263
|
+
if (forward) {
|
|
2264
|
+
return `${forward} The command was neither run nor recorded.`;
|
|
2265
|
+
}
|
|
2266
|
+
|
|
2267
|
+
if (command.rollbackCommand) {
|
|
2268
|
+
const rollback: string | null =
|
|
2269
|
+
RemediationCommandToolkit.getResourceWriteScopeRefusal({
|
|
2270
|
+
resource,
|
|
2271
|
+
command: command.rollbackCommand,
|
|
2272
|
+
});
|
|
2273
|
+
|
|
2274
|
+
if (rollback) {
|
|
2275
|
+
return `The rollbackCommand would be refused when it has to run: ${rollback} The command was neither run nor recorded — give a rollback the agent may run, or omit it.`;
|
|
2276
|
+
}
|
|
2277
|
+
}
|
|
2278
|
+
|
|
2279
|
+
return null;
|
|
2280
|
+
}
|
|
2281
|
+
|
|
1525
2282
|
/*
|
|
1526
2283
|
* Null when the command may auto-execute in FullAuto; otherwise why not.
|
|
1527
2284
|
* Bash/SSH: denylist, structural guard and the rule allowlist, for the
|
|
@@ -1603,6 +2360,10 @@ export default class RemediationCommandToolkit {
|
|
|
1603
2360
|
return null;
|
|
1604
2361
|
}
|
|
1605
2362
|
|
|
2363
|
+
if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
2364
|
+
return this.getResourceFullAutoRefusal(command);
|
|
2365
|
+
}
|
|
2366
|
+
|
|
1606
2367
|
const policy: CommandPolicyResult = CommandPolicy.evaluateCommand({
|
|
1607
2368
|
command: command.command,
|
|
1608
2369
|
allowlistPatterns: this.options.allowlistPatterns,
|
|
@@ -1706,6 +2467,123 @@ export default class RemediationCommandToolkit {
|
|
|
1706
2467
|
: "Include this action in your final recommendations for a human.";
|
|
1707
2468
|
}
|
|
1708
2469
|
|
|
2470
|
+
/*
|
|
2471
|
+
* The FullAuto gate for a resource command — the kubectl branch of
|
|
2472
|
+
* getFullAutoRefusal, on the resource command policy: the resource's
|
|
2473
|
+
* (live) mode must run unattended, the ladder
|
|
2474
|
+
* (ResourceCommandPolicy.evaluateForAutoExecution with the resource's
|
|
2475
|
+
* allowlist, bypass = BypassApproval) must auto-approve the command, and
|
|
2476
|
+
* its rollback — which runs unattended — must auto-approve too. A refusal
|
|
2477
|
+
* a human's click would lift carries an approvalReason.
|
|
2478
|
+
*/
|
|
2479
|
+
private getResourceFullAutoRefusal(
|
|
2480
|
+
command: AiRemediationCommand,
|
|
2481
|
+
): FullAutoRefusal | null {
|
|
2482
|
+
const resource: ResourceAiAccessStatus | undefined =
|
|
2483
|
+
this.findResourceTarget(command.resourceType, command.resourceId);
|
|
2484
|
+
|
|
2485
|
+
if (!resource) {
|
|
2486
|
+
return {
|
|
2487
|
+
text: "The resource is no longer a valid target. Use list_command_targets.",
|
|
2488
|
+
};
|
|
2489
|
+
}
|
|
2490
|
+
|
|
2491
|
+
const label: string = describeResourceLabel(resource);
|
|
2492
|
+
|
|
2493
|
+
if (!isUnattendedResourceRemediationMode(resource.aiRemediationMode)) {
|
|
2494
|
+
return {
|
|
2495
|
+
text: `${label.charAt(0).toUpperCase()}${label.slice(1)} requires human approval for every change, so nothing can execute inline in this run. ${this.describeWhereRefusedChangesGo()}`,
|
|
2496
|
+
approvalReason: `${label} asks for approval of every change`,
|
|
2497
|
+
};
|
|
2498
|
+
}
|
|
2499
|
+
|
|
2500
|
+
const bypassApproval: boolean =
|
|
2501
|
+
resource.aiRemediationMode === ResourceAiRemediationMode.BypassApproval;
|
|
2502
|
+
|
|
2503
|
+
const verdict: ResourceAutoExecutionVerdict =
|
|
2504
|
+
ResourceCommandPolicy.evaluateForAutoExecution({
|
|
2505
|
+
resourceType: resource.resourceType,
|
|
2506
|
+
command: command.command,
|
|
2507
|
+
allowlistPatterns: resource.aiCommandAllowlist,
|
|
2508
|
+
bypassApproval,
|
|
2509
|
+
});
|
|
2510
|
+
|
|
2511
|
+
if (verdict.verdict !== AiRemediationCommandPolicyVerdict.AutoApproved) {
|
|
2512
|
+
if (verdict.verdict === AiRemediationCommandPolicyVerdict.Denied) {
|
|
2513
|
+
return {
|
|
2514
|
+
text: `${verdict.reason} The command was NOT executed.`,
|
|
2515
|
+
};
|
|
2516
|
+
}
|
|
2517
|
+
|
|
2518
|
+
return this.describeResourceNeedsAHuman({
|
|
2519
|
+
resource,
|
|
2520
|
+
verdict,
|
|
2521
|
+
bypassApproval,
|
|
2522
|
+
});
|
|
2523
|
+
}
|
|
2524
|
+
|
|
2525
|
+
if (command.rollbackCommand) {
|
|
2526
|
+
const rollbackVerdict: ResourceAutoExecutionVerdict =
|
|
2527
|
+
ResourceCommandPolicy.evaluateForAutoExecution({
|
|
2528
|
+
resourceType: resource.resourceType,
|
|
2529
|
+
command: command.rollbackCommand,
|
|
2530
|
+
allowlistPatterns: resource.aiCommandAllowlist,
|
|
2531
|
+
bypassApproval,
|
|
2532
|
+
});
|
|
2533
|
+
|
|
2534
|
+
if (
|
|
2535
|
+
rollbackVerdict.verdict !==
|
|
2536
|
+
AiRemediationCommandPolicyVerdict.AutoApproved
|
|
2537
|
+
) {
|
|
2538
|
+
return {
|
|
2539
|
+
text: `The rollbackCommand does not qualify for automatic execution: ${rollbackVerdict.reason} Nothing was executed. Provide a safe rollback on ONE named object (for example the start that undoes a stop), or omit it.`,
|
|
2540
|
+
};
|
|
2541
|
+
}
|
|
2542
|
+
}
|
|
2543
|
+
|
|
2544
|
+
return null;
|
|
2545
|
+
}
|
|
2546
|
+
|
|
2547
|
+
/*
|
|
2548
|
+
* describeNeedsAHuman for a resource: named for what actually holds the
|
|
2549
|
+
* change back — a change the policy says always needs a human (in every
|
|
2550
|
+
* mode, Bypass approval and the allowlist included), or a riskier change
|
|
2551
|
+
* on a resource that runs only safe changes on its own.
|
|
2552
|
+
*/
|
|
2553
|
+
private describeResourceNeedsAHuman(data: {
|
|
2554
|
+
resource: ResourceAiAccessStatus;
|
|
2555
|
+
verdict: ResourceAutoExecutionVerdict;
|
|
2556
|
+
bypassApproval: boolean;
|
|
2557
|
+
}): FullAutoRefusal {
|
|
2558
|
+
const { resource, verdict } = data;
|
|
2559
|
+
const label: string = describeResourceLabel(resource);
|
|
2560
|
+
const whereItGoes: string = this.describeWhereRefusedChangesGo();
|
|
2561
|
+
|
|
2562
|
+
if (verdict.requiresHuman === true) {
|
|
2563
|
+
return {
|
|
2564
|
+
text: `${verdict.reason} The command was NOT executed: this change always needs a human, in every mode — Bypass approval and the ${describeResourceNoun(
|
|
2565
|
+
resource.resourceType,
|
|
2566
|
+
)}'s allowlist included. ${whereItGoes} Do NOT hunt for a worse substitute that would need a human just the same.`,
|
|
2567
|
+
approvalReason: `the command policy of ${label} says this change always needs a human, in every mode`,
|
|
2568
|
+
};
|
|
2569
|
+
}
|
|
2570
|
+
|
|
2571
|
+
if (
|
|
2572
|
+
verdict.tier === ResourceCommandTier.RiskyWrite &&
|
|
2573
|
+
!data.bypassApproval
|
|
2574
|
+
) {
|
|
2575
|
+
return {
|
|
2576
|
+
text: `${verdict.reason} The command was NOT executed. ${whereItGoes} Do NOT hunt for a worse safe substitute; use a safe change (${RESOURCE_SAFE_CHANGES_SUMMARY}) only when it is genuinely the right fix.`,
|
|
2577
|
+
approvalReason: `it is a riskier change (${verdict.tier}), and ${label} runs only safe changes on its own unless its command allowlist names the exact command`,
|
|
2578
|
+
};
|
|
2579
|
+
}
|
|
2580
|
+
|
|
2581
|
+
return {
|
|
2582
|
+
text: `${verdict.reason} The command was NOT executed. ${whereItGoes}`,
|
|
2583
|
+
approvalReason: `${label} does not allow it without a human: ${verdict.reason}`,
|
|
2584
|
+
};
|
|
2585
|
+
}
|
|
2586
|
+
|
|
1709
2587
|
/*
|
|
1710
2588
|
* Keep a kubectl change refused only for want of a human's click, so a
|
|
1711
2589
|
* cluster round that executes nothing can propose it when it settles.
|
|
@@ -1717,9 +2595,11 @@ export default class RemediationCommandToolkit {
|
|
|
1717
2595
|
command: AiRemediationCommand,
|
|
1718
2596
|
refusal: FullAutoRefusal,
|
|
1719
2597
|
): void {
|
|
2598
|
+
// Kubectl changes on a cluster, and resource commands on a resource.
|
|
1720
2599
|
if (
|
|
1721
2600
|
!refusal.approvalReason ||
|
|
1722
|
-
command.stepType !== RunbookStepType.Kubectl
|
|
2601
|
+
(command.stepType !== RunbookStepType.Kubectl &&
|
|
2602
|
+
command.stepType !== RunbookStepType.ResourceCommand)
|
|
1723
2603
|
) {
|
|
1724
2604
|
return;
|
|
1725
2605
|
}
|
|
@@ -1728,6 +2608,8 @@ export default class RemediationCommandToolkit {
|
|
|
1728
2608
|
(kept: RemediationCommandNeedingApproval) => {
|
|
1729
2609
|
return (
|
|
1730
2610
|
kept.command.kubernetesClusterId === command.kubernetesClusterId &&
|
|
2611
|
+
getResourceKey(kept.command.resourceType, kept.command.resourceId) ===
|
|
2612
|
+
getResourceKey(command.resourceType, command.resourceId) &&
|
|
1731
2613
|
kept.command.command === command.command
|
|
1732
2614
|
);
|
|
1733
2615
|
},
|
|
@@ -1989,6 +2871,283 @@ export default class RemediationCommandToolkit {
|
|
|
1989
2871
|
}
|
|
1990
2872
|
}
|
|
1991
2873
|
|
|
2874
|
+
/*
|
|
2875
|
+
* refreshClusterTarget for a resource: re-read the resource from its AI
|
|
2876
|
+
* page before every inline change. Refuses — and stops the run from
|
|
2877
|
+
* changing that resource again — when fixes were turned off or lost
|
|
2878
|
+
* readiness (project switches, the agent going offline or read-only
|
|
2879
|
+
* included: the status folds them in), when the resource is now reached
|
|
2880
|
+
* through another AI agent than the one this run's command names (the
|
|
2881
|
+
* agent was reset or replaced), or when the status cannot be read (fail
|
|
2882
|
+
* closed). Otherwise the snapshot is replaced with the live status, so the
|
|
2883
|
+
* verdict that follows uses the CURRENT mode and allowlist.
|
|
2884
|
+
*/
|
|
2885
|
+
private async refreshResourceTarget(
|
|
2886
|
+
command: AiRemediationCommand,
|
|
2887
|
+
): Promise<FullAutoRefusal | null> {
|
|
2888
|
+
const key: string = getResourceKey(
|
|
2889
|
+
command.resourceType,
|
|
2890
|
+
command.resourceId,
|
|
2891
|
+
);
|
|
2892
|
+
const label: string = describeResourceLabel({
|
|
2893
|
+
resourceType: command.resourceType,
|
|
2894
|
+
resourceName: command.resourceNameSnapshot || command.resourceId,
|
|
2895
|
+
});
|
|
2896
|
+
const noun: string = isAiResourceType(command.resourceType)
|
|
2897
|
+
? describeResourceNoun(command.resourceType)
|
|
2898
|
+
: "resource";
|
|
2899
|
+
const stopText: string = `The command was NOT executed. Do NOT run any further command on this ${noun} in this run; summarize what happened and put the fix in your final recommendations.`;
|
|
2900
|
+
|
|
2901
|
+
if (
|
|
2902
|
+
!isAiResourceType(command.resourceType) ||
|
|
2903
|
+
!command.resourceId ||
|
|
2904
|
+
!ObjectID.isValidUUID(command.resourceId)
|
|
2905
|
+
) {
|
|
2906
|
+
return {
|
|
2907
|
+
text: `The command names no valid resource. ${stopText}`,
|
|
2908
|
+
};
|
|
2909
|
+
}
|
|
2910
|
+
|
|
2911
|
+
let status: ResourceAiAccessStatus | null;
|
|
2912
|
+
|
|
2913
|
+
try {
|
|
2914
|
+
status = await ResourceAiAccessService.getStatusForResource({
|
|
2915
|
+
projectId: this.options.projectId,
|
|
2916
|
+
resourceType: command.resourceType,
|
|
2917
|
+
resourceId: new ObjectID(command.resourceId),
|
|
2918
|
+
});
|
|
2919
|
+
} catch (error) {
|
|
2920
|
+
logger.error(
|
|
2921
|
+
`RemediationCommandToolkit: could not re-read the AI access of ${command.resourceType} ${command.resourceId} before an inline change; refusing it: ${error}`,
|
|
2922
|
+
);
|
|
2923
|
+
return {
|
|
2924
|
+
text: `Could not confirm that ${label} still allows AI remediation. ${stopText}`,
|
|
2925
|
+
};
|
|
2926
|
+
}
|
|
2927
|
+
|
|
2928
|
+
if (!status) {
|
|
2929
|
+
this.revokeResource(key);
|
|
2930
|
+
return {
|
|
2931
|
+
text: `${label.charAt(0).toUpperCase()}${label.slice(1)} no longer exists in this project. ${stopText}`,
|
|
2932
|
+
};
|
|
2933
|
+
}
|
|
2934
|
+
|
|
2935
|
+
const liveLabel: string = describeResourceLabel(status);
|
|
2936
|
+
|
|
2937
|
+
if (!status.isRemediationReady) {
|
|
2938
|
+
this.revokeResource(key);
|
|
2939
|
+
const gap: ResourceAiAccessGap | undefined = status.gaps.find(
|
|
2940
|
+
(candidate: ResourceAiAccessGap): boolean => {
|
|
2941
|
+
return candidate.blocksRemediation;
|
|
2942
|
+
},
|
|
2943
|
+
);
|
|
2944
|
+
return {
|
|
2945
|
+
text: `${liveLabel.charAt(0).toUpperCase()}${liveLabel.slice(1)} no longer allows AI remediation${
|
|
2946
|
+
gap ? ` (${gap.title})` : ""
|
|
2947
|
+
} — its AI agent page changed during this run. ${stopText}`,
|
|
2948
|
+
};
|
|
2949
|
+
}
|
|
2950
|
+
|
|
2951
|
+
if (!status.agent || status.agent.agentId !== command.runnerId) {
|
|
2952
|
+
this.revokeResource(key);
|
|
2953
|
+
return {
|
|
2954
|
+
text: `${liveLabel.charAt(0).toUpperCase()}${liveLabel.slice(1)} is no longer reached through the AI agent this run started with (its agent was reset or replaced). ${stopText}`,
|
|
2955
|
+
};
|
|
2956
|
+
}
|
|
2957
|
+
|
|
2958
|
+
const snapshot: ResourceAiAccessStatus | undefined =
|
|
2959
|
+
this.findResourceTarget(command.resourceType, command.resourceId);
|
|
2960
|
+
|
|
2961
|
+
this.replaceResourceTarget(status);
|
|
2962
|
+
|
|
2963
|
+
const round: number = this.options.resourceRoundNumber || 1;
|
|
2964
|
+
|
|
2965
|
+
if (
|
|
2966
|
+
snapshot &&
|
|
2967
|
+
doesResourceModeRunRoundUnattended(snapshot.aiRemediationMode, round) &&
|
|
2968
|
+
!doesResourceModeRunRoundUnattended(status.aiRemediationMode, round)
|
|
2969
|
+
) {
|
|
2970
|
+
/*
|
|
2971
|
+
* On a follow-up round, Automatic is still an unattended mode — but
|
|
2972
|
+
* one that asks for every round after the first, so it stops this
|
|
2973
|
+
* round's unattended changes like a move to "ask for approval" does.
|
|
2974
|
+
*/
|
|
2975
|
+
if (
|
|
2976
|
+
isUnattendedResourceRemediationMode(status.aiRemediationMode) &&
|
|
2977
|
+
round > 1
|
|
2978
|
+
) {
|
|
2979
|
+
return {
|
|
2980
|
+
text: `The AI remediation mode of ${liveLabel} was changed to Automatic during this run, and Automatic asks for approval of every change after a signal's first round (this is round ${round}), so no change runs on it unattended any more. The command was NOT executed. ${this.describeWhereRefusedChangesGo()} Do NOT try other changes on this ${noun}.`,
|
|
2981
|
+
approvalReason: `the AI remediation mode of ${liveLabel} was changed to Automatic during the round, which asks for approval after a signal's first round`,
|
|
2982
|
+
};
|
|
2983
|
+
}
|
|
2984
|
+
|
|
2985
|
+
return {
|
|
2986
|
+
text: `The AI remediation mode of ${liveLabel} was changed to ask for approval during this run, so no change runs on it unattended any more. The command was NOT executed. ${this.describeWhereRefusedChangesGo()} Do NOT try other changes on this ${noun}.`,
|
|
2987
|
+
approvalReason: `the AI remediation mode of ${liveLabel} was changed to ask for approval during the round`,
|
|
2988
|
+
};
|
|
2989
|
+
}
|
|
2990
|
+
|
|
2991
|
+
return null;
|
|
2992
|
+
}
|
|
2993
|
+
|
|
2994
|
+
/*
|
|
2995
|
+
* revokeCluster for a resource: the rest of the run treats it as never
|
|
2996
|
+
* having been a target, and what was kept for its proposal is dropped.
|
|
2997
|
+
*/
|
|
2998
|
+
private revokeResource(key: string): void {
|
|
2999
|
+
this.revokedResourceKeys.add(key);
|
|
3000
|
+
this.commandsNeedingApproval = this.commandsNeedingApproval.filter(
|
|
3001
|
+
(kept: RemediationCommandNeedingApproval): boolean => {
|
|
3002
|
+
return (
|
|
3003
|
+
getResourceKey(kept.command.resourceType, kept.command.resourceId) !==
|
|
3004
|
+
key
|
|
3005
|
+
);
|
|
3006
|
+
},
|
|
3007
|
+
);
|
|
3008
|
+
}
|
|
3009
|
+
|
|
3010
|
+
private replaceResourceTarget(status: ResourceAiAccessStatus): void {
|
|
3011
|
+
const key: string = getResourceKey(status.resourceType, status.resourceId);
|
|
3012
|
+
|
|
3013
|
+
this.options.resourceTargets = (this.options.resourceTargets || []).map(
|
|
3014
|
+
(resource: ResourceAiAccessStatus): ResourceAiAccessStatus => {
|
|
3015
|
+
return getResourceKey(resource.resourceType, resource.resourceId) ===
|
|
3016
|
+
key
|
|
3017
|
+
? status
|
|
3018
|
+
: resource;
|
|
3019
|
+
},
|
|
3020
|
+
);
|
|
3021
|
+
}
|
|
3022
|
+
|
|
3023
|
+
// Does this run already hold a slot on the resource (an inline job there)?
|
|
3024
|
+
private hasChangedResource(
|
|
3025
|
+
resourceType: string | undefined,
|
|
3026
|
+
resourceId: string | undefined,
|
|
3027
|
+
): boolean {
|
|
3028
|
+
const key: string = getResourceKey(resourceType, resourceId);
|
|
3029
|
+
|
|
3030
|
+
return this.executedCommands.some(
|
|
3031
|
+
(executed: AiRemediationCommand): boolean => {
|
|
3032
|
+
return (
|
|
3033
|
+
executed.stepType === RunbookStepType.ResourceCommand &&
|
|
3034
|
+
getResourceKey(executed.resourceType, executed.resourceId) === key &&
|
|
3035
|
+
Boolean(executed.execution?.runnerJobId)
|
|
3036
|
+
);
|
|
3037
|
+
},
|
|
3038
|
+
);
|
|
3039
|
+
}
|
|
3040
|
+
|
|
3041
|
+
/*
|
|
3042
|
+
* reserveClusterSlot for a resource: one of the resource's hourly
|
|
3043
|
+
* unattended slots, taken under the per-resource breaker lock
|
|
3044
|
+
* (RESOURCE_BREAKER_LOCK_NAMESPACE, keyed by type and id, never a
|
|
3045
|
+
* cluster's key), which is returned HELD until this run's job row exists.
|
|
3046
|
+
* Under the same lock, no other AI run may hold the resource
|
|
3047
|
+
* (resourceHold). Fails closed.
|
|
3048
|
+
*/
|
|
3049
|
+
private async reserveResourceSlot(
|
|
3050
|
+
command: AiRemediationCommand,
|
|
3051
|
+
): Promise<ClusterSlotReservation> {
|
|
3052
|
+
const resourceType: AiResourceType = command.resourceType as AiResourceType;
|
|
3053
|
+
const resourceId: string = command.resourceId || "";
|
|
3054
|
+
const label: string = describeResourceLabel({
|
|
3055
|
+
resourceType,
|
|
3056
|
+
resourceName: command.resourceNameSnapshot || resourceId,
|
|
3057
|
+
});
|
|
3058
|
+
const noun: string = isAiResourceType(resourceType)
|
|
3059
|
+
? describeResourceNoun(resourceType)
|
|
3060
|
+
: "resource";
|
|
3061
|
+
const couldNotCheck: FullAutoRefusal = {
|
|
3062
|
+
text: `Could not check the hourly limit on unattended AI fixes for ${label}, so the command was NOT executed. ${this.describeWhereRefusedChangesGo()}`,
|
|
3063
|
+
approvalReason: `the hourly limit on unattended AI fixes for ${label} could not be checked`,
|
|
3064
|
+
};
|
|
3065
|
+
|
|
3066
|
+
if (!isAiResourceType(resourceType) || !resourceId) {
|
|
3067
|
+
return { mutex: null, refusal: couldNotCheck };
|
|
3068
|
+
}
|
|
3069
|
+
|
|
3070
|
+
let mutex: SemaphoreMutex | null = null;
|
|
3071
|
+
|
|
3072
|
+
try {
|
|
3073
|
+
mutex = await Semaphore.lock({
|
|
3074
|
+
key: getResourceBreakerLockKey(resourceType, resourceId),
|
|
3075
|
+
namespace: RESOURCE_BREAKER_LOCK_NAMESPACE,
|
|
3076
|
+
lockTimeout: CLUSTER_BREAKER_LOCK_TIMEOUT_MS,
|
|
3077
|
+
acquireTimeout: CLUSTER_BREAKER_LOCK_ACQUIRE_TIMEOUT_MS,
|
|
3078
|
+
});
|
|
3079
|
+
} catch (error) {
|
|
3080
|
+
logger.error(
|
|
3081
|
+
`RemediationCommandToolkit: could not take the circuit-breaker lock of ${resourceType} ${resourceId}; refusing the inline change: ${error}`,
|
|
3082
|
+
);
|
|
3083
|
+
return { mutex: null, refusal: couldNotCheck };
|
|
3084
|
+
}
|
|
3085
|
+
|
|
3086
|
+
try {
|
|
3087
|
+
const breaker: ResourceBreakerState =
|
|
3088
|
+
await AutoRemediationRuleEngineService.getResourceBreakerState({
|
|
3089
|
+
resourceType,
|
|
3090
|
+
resourceId,
|
|
3091
|
+
projectId: this.options.projectId,
|
|
3092
|
+
forRound: {
|
|
3093
|
+
suggestionId: this.options.suggestionId,
|
|
3094
|
+
createdAt: this.options.suggestionCreatedAt,
|
|
3095
|
+
},
|
|
3096
|
+
});
|
|
3097
|
+
|
|
3098
|
+
if (!breaker.hasHeadroom) {
|
|
3099
|
+
await RemediationCommandToolkit.releaseLock(mutex);
|
|
3100
|
+
logger.warn(
|
|
3101
|
+
`RemediationCommandToolkit: ${resourceType} ${resourceId} hit its hourly circuit breaker (${breaker.autoExecutedInWindow} unattended AI fixes); refusing an inline change.`,
|
|
3102
|
+
);
|
|
3103
|
+
return {
|
|
3104
|
+
mutex: null,
|
|
3105
|
+
refusal: {
|
|
3106
|
+
text: `The hourly circuit breaker for ${label} tripped: it already had ${breaker.autoExecutedInWindow} unattended AI fix(es) in the last hour (the limit is ${MAX_AUTO_EXECUTIONS_PER_RULE_PER_HOUR}). The command was NOT executed. ${this.describeWhereRefusedChangesGo()}`,
|
|
3107
|
+
approvalReason: `the hourly circuit breaker for ${label} tripped (${breaker.autoExecutedInWindow} unattended AI fixes in the last hour)`,
|
|
3108
|
+
},
|
|
3109
|
+
};
|
|
3110
|
+
}
|
|
3111
|
+
|
|
3112
|
+
if (this.options.resourceHold) {
|
|
3113
|
+
const hold: ResourceRoundHold | null =
|
|
3114
|
+
await AutoRemediationRuleEngineService.findRoundHoldingResource({
|
|
3115
|
+
projectId: this.options.projectId,
|
|
3116
|
+
resourceType,
|
|
3117
|
+
resourceId,
|
|
3118
|
+
forRound: {
|
|
3119
|
+
suggestionId: this.options.suggestionId,
|
|
3120
|
+
createdAt: this.options.suggestionCreatedAt,
|
|
3121
|
+
},
|
|
3122
|
+
anyOrder: this.options.resourceHold.anyOrder,
|
|
3123
|
+
subject: this.options.resourceHold.subject,
|
|
3124
|
+
});
|
|
3125
|
+
|
|
3126
|
+
if (hold) {
|
|
3127
|
+
await RemediationCommandToolkit.releaseLock(mutex);
|
|
3128
|
+
logger.warn(
|
|
3129
|
+
`RemediationCommandToolkit: another AI run (${hold.suggestionId}) on ${resourceType} ${resourceId} ${hold.description}; refusing an inline change.`,
|
|
3130
|
+
);
|
|
3131
|
+
return {
|
|
3132
|
+
mutex: null,
|
|
3133
|
+
refusal: {
|
|
3134
|
+
text: `Another OneUptime AI run on ${label} ${hold.description}, so this run may not change the ${noun} too — two unattended fixes on one ${noun} verify and roll back on top of each other. The command was NOT executed. ${this.describeWhereRefusedChangesGo()} Do NOT try other changes on this ${noun}.`,
|
|
3135
|
+
approvalReason: `another OneUptime AI run on ${label} ${hold.description}`,
|
|
3136
|
+
},
|
|
3137
|
+
};
|
|
3138
|
+
}
|
|
3139
|
+
}
|
|
3140
|
+
|
|
3141
|
+
return { mutex };
|
|
3142
|
+
} catch (error) {
|
|
3143
|
+
await RemediationCommandToolkit.releaseLock(mutex);
|
|
3144
|
+
logger.error(
|
|
3145
|
+
`RemediationCommandToolkit: circuit-breaker or in-flight run check failed for ${resourceType} ${resourceId}; refusing the inline change: ${error}`,
|
|
3146
|
+
);
|
|
3147
|
+
return { mutex: null, refusal: couldNotCheck };
|
|
3148
|
+
}
|
|
3149
|
+
}
|
|
3150
|
+
|
|
1992
3151
|
private static async releaseLock(mutex: SemaphoreMutex): Promise<void> {
|
|
1993
3152
|
try {
|
|
1994
3153
|
await Semaphore.release(mutex);
|
|
@@ -2023,7 +3182,9 @@ export default class RemediationCommandToolkit {
|
|
|
2023
3182
|
return {
|
|
2024
3183
|
definition: {
|
|
2025
3184
|
name: "propose_remediation_commands",
|
|
2026
|
-
description:
|
|
3185
|
+
description: this.isResourceRound()
|
|
3186
|
+
? `Propose an ordered plan of at most ${MAX_PLAN_COMMANDS} remediation commands on the infrastructure resource this round is about (stepType ResourceCommand with its resourceId) for one-click human approval. Nothing executes until a human approves the whole plan. Call this at most once with your final plan (a later call replaces the earlier one). One command per step, written as the program followed by its arguments — never a shell line; a read-only command is not a fix (run it with ${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME} instead). Provide a rollbackCommand for every state-changing command that has an undo — it runs unattended, so it must be a safe change on ONE named object. The commands run through the resource's own AI agent, and a change it would refuse (it runs read-only, or the target is protected or outside its writeScope in list_command_targets) is refused; ${RESOURCE_NEVER_RUNS_SUMMARY}.\n\n${this.describeResourceWriteGuides()}`
|
|
3187
|
+
: `Propose an ordered plan of at most ${MAX_PLAN_COMMANDS} remediation commands for one-click human approval. Nothing executes until a human approves the whole plan. Call this at most once with your final plan (a later call replaces the earlier one). Provide a rollbackCommand for every state-changing command that has an undo (for Kubectl, e.g. kubectl rollout undo deployment/<name> -n <namespace>). Kubectl commands run through the cluster's Kubernetes AI agent or Runner, and a write outside its writeScope (list_command_targets) is refused; ${KUBECTL_NEVER_RUNS_SUMMARY}.`,
|
|
2027
3188
|
inputSchema: {
|
|
2028
3189
|
type: "object",
|
|
2029
3190
|
properties: {
|
|
@@ -2034,12 +3195,15 @@ export default class RemediationCommandToolkit {
|
|
|
2034
3195
|
items: {
|
|
2035
3196
|
type: "object",
|
|
2036
3197
|
properties: this.buildCommandSchemaProperties(),
|
|
2037
|
-
required:
|
|
2038
|
-
|
|
2039
|
-
|
|
2040
|
-
|
|
2041
|
-
|
|
2042
|
-
|
|
3198
|
+
required: this.isResourceRound()
|
|
3199
|
+
? [
|
|
3200
|
+
"stepType",
|
|
3201
|
+
"resourceId",
|
|
3202
|
+
"command",
|
|
3203
|
+
"rationale",
|
|
3204
|
+
"expectedEffect",
|
|
3205
|
+
]
|
|
3206
|
+
: ["stepType", "command", "rationale", "expectedEffect"],
|
|
2043
3207
|
},
|
|
2044
3208
|
},
|
|
2045
3209
|
},
|
|
@@ -2101,6 +3265,21 @@ export default class RemediationCommandToolkit {
|
|
|
2101
3265
|
*/
|
|
2102
3266
|
parsed.command.policyVerdict =
|
|
2103
3267
|
AiRemediationCommandPolicyVerdict.RequiresApproval;
|
|
3268
|
+
} else if (parsed.command.stepType === RunbookStepType.ResourceCommand) {
|
|
3269
|
+
/*
|
|
3270
|
+
* The same for a resource command: a proposal runs only after a
|
|
3271
|
+
* click, so it is RequiresApproval whatever the resource's mode —
|
|
3272
|
+
* and a read is not a fix a human needs to approve.
|
|
3273
|
+
*/
|
|
3274
|
+
if (parsed.command.resourceCommandTier === ResourceCommandTier.Read) {
|
|
3275
|
+
problems.push(
|
|
3276
|
+
`Command ${i + 1}: "${parsed.command.command}" is read-only — it is not a fix. Run it with ${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME} and propose only the change.`,
|
|
3277
|
+
);
|
|
3278
|
+
continue;
|
|
3279
|
+
}
|
|
3280
|
+
|
|
3281
|
+
parsed.command.policyVerdict =
|
|
3282
|
+
AiRemediationCommandPolicyVerdict.RequiresApproval;
|
|
2104
3283
|
} else {
|
|
2105
3284
|
/*
|
|
2106
3285
|
* Informational verdict for the approval card: AutoApproved
|
|
@@ -2164,9 +3343,37 @@ export default class RemediationCommandToolkit {
|
|
|
2164
3343
|
const stepTypeRaw: string = ToolArgs.getString(args, "stepType") || "";
|
|
2165
3344
|
const stepType: RunbookStepType = stepTypeRaw as RunbookStepType;
|
|
2166
3345
|
|
|
2167
|
-
|
|
3346
|
+
/*
|
|
3347
|
+
* The step types this round offers — what its schema's stepType enum
|
|
3348
|
+
* lists: ResourceCommand on a resource round; Bash, SSH and Kubectl on
|
|
3349
|
+
* every other round, where ResourceCommand is as unknown as it always
|
|
3350
|
+
* was (so a Kubernetes or rule round's refusal keeps its words).
|
|
3351
|
+
*/
|
|
3352
|
+
const offeredStepTypes: Array<RunbookStepType> = this.isResourceRound()
|
|
3353
|
+
? [RunbookStepType.ResourceCommand]
|
|
3354
|
+
: AI_COMMAND_STEP_TYPES.filter((type: RunbookStepType): boolean => {
|
|
3355
|
+
return type !== RunbookStepType.ResourceCommand;
|
|
3356
|
+
});
|
|
3357
|
+
|
|
3358
|
+
if (
|
|
3359
|
+
!AI_COMMAND_STEP_TYPES.includes(stepType) ||
|
|
3360
|
+
(!this.isResourceRound() && stepType === RunbookStepType.ResourceCommand)
|
|
3361
|
+
) {
|
|
3362
|
+
return {
|
|
3363
|
+
errorText: `stepType must be one of: ${offeredStepTypes.join(", ")}.`,
|
|
3364
|
+
};
|
|
3365
|
+
}
|
|
3366
|
+
|
|
3367
|
+
/*
|
|
3368
|
+
* A resource round runs resource commands on its resource and nothing
|
|
3369
|
+
* else: it was given no Runner and no cluster.
|
|
3370
|
+
*/
|
|
3371
|
+
if (
|
|
3372
|
+
this.isResourceRound() &&
|
|
3373
|
+
stepType !== RunbookStepType.ResourceCommand
|
|
3374
|
+
) {
|
|
2168
3375
|
return {
|
|
2169
|
-
errorText: `
|
|
3376
|
+
errorText: `This remediation round is about one infrastructure resource: use stepType ResourceCommand with its resourceId from list_command_targets. ${stepType} is not available here.`,
|
|
2170
3377
|
};
|
|
2171
3378
|
}
|
|
2172
3379
|
|
|
@@ -2209,6 +3416,31 @@ export default class RemediationCommandToolkit {
|
|
|
2209
3416
|
});
|
|
2210
3417
|
}
|
|
2211
3418
|
|
|
3419
|
+
/*
|
|
3420
|
+
* A resource command runs on the resource's own AI agent, never on a
|
|
3421
|
+
* Runner, so it must never fall through to the Bash/SSH path below.
|
|
3422
|
+
* Only a resource round (resourceTargets) offers it — any other round
|
|
3423
|
+
* refused it with the step types it offers, above; this is the belt
|
|
3424
|
+
* and braces.
|
|
3425
|
+
*/
|
|
3426
|
+
if (stepType === RunbookStepType.ResourceCommand) {
|
|
3427
|
+
if (!this.isResourceRound()) {
|
|
3428
|
+
return {
|
|
3429
|
+
errorText: `stepType must be one of: ${offeredStepTypes.join(", ")}.`,
|
|
3430
|
+
};
|
|
3431
|
+
}
|
|
3432
|
+
|
|
3433
|
+
return this.parseResourceCommand({
|
|
3434
|
+
args,
|
|
3435
|
+
sequence,
|
|
3436
|
+
commandText,
|
|
3437
|
+
rollbackCommand,
|
|
3438
|
+
timeoutInMs,
|
|
3439
|
+
rationale,
|
|
3440
|
+
expectedEffect,
|
|
3441
|
+
});
|
|
3442
|
+
}
|
|
3443
|
+
|
|
2212
3444
|
const denyReason: string | null = CommandPolicy.getDenyReason(commandText);
|
|
2213
3445
|
if (denyReason) {
|
|
2214
3446
|
return {
|
|
@@ -2466,6 +3698,176 @@ export default class RemediationCommandToolkit {
|
|
|
2466
3698
|
);
|
|
2467
3699
|
}
|
|
2468
3700
|
|
|
3701
|
+
/*
|
|
3702
|
+
* ResourceCommand: the target is the round's resource, the AI agent is
|
|
3703
|
+
* whichever one its AI page reports online, and no credential ever
|
|
3704
|
+
* travels. The model only names a resource it was shown (by resourceId);
|
|
3705
|
+
* the tier the resource command policy gives the command is recorded for
|
|
3706
|
+
* the card, and a command, or a rollback, the agent's write scope would
|
|
3707
|
+
* refuse is never composed. Denied never is either.
|
|
3708
|
+
*/
|
|
3709
|
+
private parseResourceCommand(data: {
|
|
3710
|
+
args: JSONObject;
|
|
3711
|
+
sequence: number;
|
|
3712
|
+
commandText: string;
|
|
3713
|
+
rollbackCommand: string | undefined;
|
|
3714
|
+
timeoutInMs: number;
|
|
3715
|
+
rationale: string;
|
|
3716
|
+
expectedEffect: string;
|
|
3717
|
+
}): CommandArgsParseResult {
|
|
3718
|
+
const resourceIdRaw: string | undefined = ToolArgs.getString(
|
|
3719
|
+
data.args,
|
|
3720
|
+
"resourceId",
|
|
3721
|
+
);
|
|
3722
|
+
|
|
3723
|
+
const resource: ResourceAiAccessStatus | undefined = resourceIdRaw
|
|
3724
|
+
? this.getResourceTargets().find(
|
|
3725
|
+
(candidate: ResourceAiAccessStatus): boolean => {
|
|
3726
|
+
return (
|
|
3727
|
+
candidate.resourceId.toLowerCase() === resourceIdRaw.toLowerCase()
|
|
3728
|
+
);
|
|
3729
|
+
},
|
|
3730
|
+
)
|
|
3731
|
+
: undefined;
|
|
3732
|
+
|
|
3733
|
+
if (!resource || !resource.agent) {
|
|
3734
|
+
return {
|
|
3735
|
+
errorText:
|
|
3736
|
+
"resourceId is required for ResourceCommand and must be the resource from list_command_targets that allows AI remediation.",
|
|
3737
|
+
};
|
|
3738
|
+
}
|
|
3739
|
+
|
|
3740
|
+
/*
|
|
3741
|
+
* A resource's agent is never given a credential by OneUptime: a step
|
|
3742
|
+
* that names one is refused, never silently stripped.
|
|
3743
|
+
*/
|
|
3744
|
+
const credentialIdRaw: string | undefined = ToolArgs.getString(
|
|
3745
|
+
data.args,
|
|
3746
|
+
"credentialId",
|
|
3747
|
+
);
|
|
3748
|
+
|
|
3749
|
+
if (credentialIdRaw) {
|
|
3750
|
+
return {
|
|
3751
|
+
errorText: `A ResourceCommand never carries a credential: the ${
|
|
3752
|
+
AI_RESOURCE_TYPE_INFO[resource.resourceType].agentDisplayName
|
|
3753
|
+
} uses only the credentials in its own environment. Omit credentialId.`,
|
|
3754
|
+
};
|
|
3755
|
+
}
|
|
3756
|
+
|
|
3757
|
+
const info: AiResourceTypeInfo =
|
|
3758
|
+
AI_RESOURCE_TYPE_INFO[resource.resourceType];
|
|
3759
|
+
|
|
3760
|
+
const policy: ResourceCommandPolicyResult =
|
|
3761
|
+
ResourceCommandPolicy.evaluateCommand({
|
|
3762
|
+
resourceType: resource.resourceType,
|
|
3763
|
+
command: data.commandText,
|
|
3764
|
+
});
|
|
3765
|
+
|
|
3766
|
+
if (policy.tier === ResourceCommandTier.Denied) {
|
|
3767
|
+
return {
|
|
3768
|
+
errorText: `Denied by the ${info.displayName} command policy: ${policy.reason}. This command can never run, even with human approval — take a different approach.`,
|
|
3769
|
+
};
|
|
3770
|
+
}
|
|
3771
|
+
|
|
3772
|
+
let rollbackDisplay: string | undefined = undefined;
|
|
3773
|
+
|
|
3774
|
+
if (data.rollbackCommand) {
|
|
3775
|
+
const rollbackPolicy: ResourceCommandPolicyResult =
|
|
3776
|
+
ResourceCommandPolicy.evaluateCommand({
|
|
3777
|
+
resourceType: resource.resourceType,
|
|
3778
|
+
command: data.rollbackCommand,
|
|
3779
|
+
});
|
|
3780
|
+
|
|
3781
|
+
if (rollbackPolicy.tier === ResourceCommandTier.Denied) {
|
|
3782
|
+
return {
|
|
3783
|
+
errorText: `The rollbackCommand is denied by the ${info.displayName} command policy: ${rollbackPolicy.reason}. Provide a safe rollback or omit it.`,
|
|
3784
|
+
};
|
|
3785
|
+
}
|
|
3786
|
+
|
|
3787
|
+
/*
|
|
3788
|
+
* A rollback runs unattended after verification fails: a change that
|
|
3789
|
+
* always needs a human can never be one, and a riskier one only on a
|
|
3790
|
+
* resource whose operator bypassed approvals.
|
|
3791
|
+
*/
|
|
3792
|
+
if (rollbackPolicy.requiresHuman === true) {
|
|
3793
|
+
return {
|
|
3794
|
+
errorText: `The rollbackCommand "${rollbackPolicy.displayCommand}" always needs a human (${rollbackPolicy.reason}), and rollbacks run unattended. Use a safe undo for ONE named object instead, or omit it.`,
|
|
3795
|
+
};
|
|
3796
|
+
}
|
|
3797
|
+
|
|
3798
|
+
if (
|
|
3799
|
+
rollbackPolicy.tier === ResourceCommandTier.RiskyWrite &&
|
|
3800
|
+
resource.aiRemediationMode !== ResourceAiRemediationMode.BypassApproval
|
|
3801
|
+
) {
|
|
3802
|
+
return {
|
|
3803
|
+
errorText: `The rollbackCommand "${rollbackPolicy.displayCommand}" is a risky change (${rollbackPolicy.reason}) and rollbacks run unattended. Use a safe undo for ONE named object instead (for example the start that undoes a stop), or omit it.`,
|
|
3804
|
+
};
|
|
3805
|
+
}
|
|
3806
|
+
|
|
3807
|
+
rollbackDisplay = rollbackPolicy.displayCommand;
|
|
3808
|
+
}
|
|
3809
|
+
|
|
3810
|
+
// A write the agent has said it will refuse is never composed.
|
|
3811
|
+
const scopeRefusal: string | null = this.getResourceCommandScopeRefusal(
|
|
3812
|
+
resource,
|
|
3813
|
+
{
|
|
3814
|
+
command: policy.displayCommand,
|
|
3815
|
+
rollbackCommand: rollbackDisplay,
|
|
3816
|
+
},
|
|
3817
|
+
);
|
|
3818
|
+
|
|
3819
|
+
if (scopeRefusal) {
|
|
3820
|
+
return { errorText: scopeRefusal };
|
|
3821
|
+
}
|
|
3822
|
+
|
|
3823
|
+
return {
|
|
3824
|
+
command: {
|
|
3825
|
+
sequence: data.sequence,
|
|
3826
|
+
stepType: RunbookStepType.ResourceCommand,
|
|
3827
|
+
/*
|
|
3828
|
+
* The access target is the resource's AI agent (its row id and
|
|
3829
|
+
* display name), as a Kubectl step through the Kubernetes AI agent
|
|
3830
|
+
* names that agent.
|
|
3831
|
+
*/
|
|
3832
|
+
runnerId: resource.agent.agentId,
|
|
3833
|
+
runnerNameSnapshot: info.agentDisplayName,
|
|
3834
|
+
resourceType: resource.resourceType,
|
|
3835
|
+
resourceId: resource.resourceId,
|
|
3836
|
+
resourceNameSnapshot: resource.resourceName,
|
|
3837
|
+
resourceCommandTier: policy.tier,
|
|
3838
|
+
// Stored in the canonical rendered form so the card shows exactly what runs.
|
|
3839
|
+
command: policy.displayCommand,
|
|
3840
|
+
timeoutInMs: Math.min(
|
|
3841
|
+
data.timeoutInMs,
|
|
3842
|
+
MAX_RESOURCE_COMMAND_TIMEOUT_MS,
|
|
3843
|
+
),
|
|
3844
|
+
rationale: data.rationale,
|
|
3845
|
+
expectedEffect: data.expectedEffect,
|
|
3846
|
+
rollbackCommand: rollbackDisplay,
|
|
3847
|
+
policyVerdict: AiRemediationCommandPolicyVerdict.RequiresApproval,
|
|
3848
|
+
},
|
|
3849
|
+
};
|
|
3850
|
+
}
|
|
3851
|
+
|
|
3852
|
+
private findResourceTarget(
|
|
3853
|
+
resourceType: string | undefined,
|
|
3854
|
+
resourceId: string | undefined,
|
|
3855
|
+
): ResourceAiAccessStatus | undefined {
|
|
3856
|
+
const key: string = getResourceKey(resourceType, resourceId);
|
|
3857
|
+
|
|
3858
|
+
if (!key) {
|
|
3859
|
+
return undefined;
|
|
3860
|
+
}
|
|
3861
|
+
|
|
3862
|
+
return this.getResourceTargets().find(
|
|
3863
|
+
(resource: ResourceAiAccessStatus): boolean => {
|
|
3864
|
+
return (
|
|
3865
|
+
getResourceKey(resource.resourceType, resource.resourceId) === key
|
|
3866
|
+
);
|
|
3867
|
+
},
|
|
3868
|
+
);
|
|
3869
|
+
}
|
|
3870
|
+
|
|
2469
3871
|
/*
|
|
2470
3872
|
* Wait for the RunnerJob to reach a terminal state while keeping the
|
|
2471
3873
|
* AIRun's heartbeat fresh — a command may legitimately take minutes, and
|