@oneuptime/common 14.0.9 → 14.0.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/Models/DatabaseModels/AutoRemediationSuggestion.ts +60 -0
- package/Models/DatabaseModels/CephCluster.ts +213 -0
- package/Models/DatabaseModels/DatabaseServer.ts +213 -0
- package/Models/DatabaseModels/DockerHost.ts +213 -0
- package/Models/DatabaseModels/DockerSwarmCluster.ts +213 -0
- package/Models/DatabaseModels/GlobalConfig.ts +1 -1
- package/Models/DatabaseModels/Host.ts +213 -0
- package/Models/DatabaseModels/Index.ts +2 -0
- package/Models/DatabaseModels/PodmanHost.ts +213 -0
- package/Models/DatabaseModels/ProxmoxCluster.ts +213 -0
- package/Models/DatabaseModels/ResourceAiAgent.ts +419 -0
- package/Models/DatabaseModels/RunnerJob.ts +144 -0
- package/Models/DatabaseModels/VMwareVCenter.ts +213 -0
- package/Server/API/AutoRemediationAPI.ts +392 -1
- package/Server/API/ResourceAiAccessAPI.ts +1669 -0
- package/Server/Infrastructure/Postgres/SchemaMigrations/1796300000000-AddResourceAiAgents.ts +423 -0
- package/Server/Infrastructure/Postgres/SchemaMigrations/Index.ts +2 -0
- package/Server/Infrastructure/Semaphore.ts +22 -0
- package/Server/Middleware/TelemetryIngest.ts +15 -5
- package/Server/Services/AnalyticsDatabaseService.ts +45 -1
- package/Server/Services/AutoRemediationRuleEngineService.ts +940 -112
- package/Server/Services/CephClusterService.ts +109 -1
- package/Server/Services/DatabaseServerService.ts +116 -1
- package/Server/Services/DockerHostService.ts +109 -1
- package/Server/Services/DockerSwarmClusterService.ts +113 -1
- package/Server/Services/HostService.ts +126 -4
- package/Server/Services/MetricRecordingRuleService.ts +47 -0
- package/Server/Services/PodmanHostService.ts +109 -1
- package/Server/Services/ProxmoxClusterService.ts +110 -1
- package/Server/Services/ResourceAiAccessService.ts +1439 -0
- package/Server/Services/ResourceAiAgentJobService.ts +202 -0
- package/Server/Services/ResourceAiAgentService.ts +2432 -0
- package/Server/Services/RunnerJobService.ts +611 -4
- package/Server/Services/TelemetryUsageBillingService.ts +28 -0
- package/Server/Services/TraceRecordingRuleService.ts +33 -0
- package/Server/Services/VMwareVCenterService.ts +110 -1
- package/Server/Utils/AI/Remediation/RemediationCommandTools.ts +1434 -32
- package/Server/Utils/AI/Remediation/RemediationExecutionRunner.ts +935 -21
- package/Server/Utils/AI/Remediation/RemediationPlanRunner.ts +25 -1
- package/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.ts +577 -0
- package/Server/Utils/AI/ResourceAccess/ResourceAccessContext.ts +271 -0
- package/Server/Utils/AI/ResourceAccess/ResourceAccessToolNames.ts +45 -0
- package/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.ts +964 -0
- package/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.ts +333 -0
- package/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.ts +873 -0
- package/Server/Utils/AI/SRE/AIInvestigationEngine.ts +72 -2
- package/Server/Utils/AI/SRE/AlertInvestigationRunner.ts +72 -5
- package/Server/Utils/AI/SRE/IncidentInvestigationRunner.ts +72 -5
- package/Server/Utils/AutoRemediation/CommandPlanExecutor.ts +396 -13
- package/Server/Utils/AutoRemediation/RemediationVerifier.ts +35 -2
- package/Server/Utils/Database/ProjectScopedReferenceValidator.ts +9 -1
- package/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.ts +920 -0
- package/Server/Utils/SessionReplay/SessionReplayUsage.ts +84 -6
- package/Server/Utils/Workspace/MicrosoftTeams/Actions/Alert.ts +35 -15
- package/Server/Utils/Workspace/MicrosoftTeams/Actions/AlertEpisode.ts +33 -15
- package/Server/Utils/Workspace/MicrosoftTeams/Actions/Auth.ts +21 -1
- package/Server/Utils/Workspace/MicrosoftTeams/Actions/Incident.ts +298 -224
- package/Server/Utils/Workspace/MicrosoftTeams/Actions/IncidentEpisode.ts +35 -15
- package/Server/Utils/Workspace/MicrosoftTeams/Actions/ScheduledMaintenance.ts +389 -175
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeams.ts +366 -143
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.ts +163 -0
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.ts +396 -0
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.ts +418 -0
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.ts +152 -0
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.ts +251 -0
- package/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.ts +214 -0
- package/Tests/App/Dashboard/ClusterAccessNotice.test.tsx +93 -0
- package/Tests/App/Dashboard/DatabaseDocumentationMarkdown.test.ts +18 -6
- package/Tests/App/Dashboard/InvestigationInfrastructureTools.test.tsx +506 -0
- package/Tests/App/Dashboard/RemediationSuggestionCardDescription.test.tsx +209 -0
- package/Tests/App/Dashboard/RemediationSuggestionCardResource.test.tsx +317 -0
- package/Tests/App/Dashboard/ResourceAiAccessSettingsUtil.test.ts +921 -0
- package/Tests/App/Dashboard/ResourceAiAgentInstall.test.ts +860 -0
- package/Tests/App/Dashboard/ResourceAiAgentPage.test.tsx +1721 -0
- package/Tests/App/Dashboard/ResourceAiAgentStatus.test.ts +1002 -0
- package/Tests/App/Dashboard/ResourceAiInsightsPage.test.tsx +931 -0
- package/Tests/App/Dashboard/ResourceAiNavigation.test.tsx +575 -0
- package/Tests/App/Dashboard/RunbookStepTypeMaps.test.ts +38 -7
- package/Tests/Models/DatabaseModels/DatabaseServerModels.test.ts +33 -1
- package/Tests/Models/DatabaseModels/ResourceAiAccessColumns.test.ts +738 -0
- package/Tests/Models/DatabaseModels/ResourceAiAgentModel.test.ts +842 -0
- package/Tests/Server/API/AutoRemediationApproveResourceRound.test.ts +835 -0
- package/Tests/Server/API/ResourceAiAccessAPI.test.ts +2730 -0
- package/Tests/Server/Infrastructure/Postgres/AddDatabaseServerTablesMigration.test.ts +60 -0
- package/Tests/Server/Infrastructure/Postgres/AddResourceAiAgentsMigration.test.ts +659 -0
- package/Tests/Server/Infrastructure/SemaphoreMutex.test.ts +31 -0
- package/Tests/Server/Middleware/TelemetryIngestBrowserKey.test.ts +63 -5
- package/Tests/Server/Middleware/TelemetryIngestKubernetesAgentRunnerPinnedKey.test.ts +103 -5
- package/Tests/Server/Middleware/TelemetryIngestKubernetesAgentRunnerRateLimit.test.ts +19 -1
- package/Tests/Server/Services/AutoRemediationResourceRuleEngine.test.ts +1279 -0
- package/Tests/Server/Services/DatabaseServerService.test.ts +55 -0
- package/Tests/Server/Services/GroupTelemetryUsageExcludeNames.test.ts +299 -0
- package/Tests/Server/Services/HostServiceFindOrCreateMemo.test.ts +44 -0
- package/Tests/Server/Services/MonitorProbeServiceIntervalScheduling.test.ts +45 -6
- package/Tests/Server/Services/RecordingRuleReservedMetricName.test.ts +171 -0
- package/Tests/Server/Services/ResourceAiAccessChangeAuthorization.test.ts +256 -0
- package/Tests/Server/Services/ResourceAiAccessService.test.ts +1373 -0
- package/Tests/Server/Services/ResourceAiAgentJobService.test.ts +375 -0
- package/Tests/Server/Services/ResourceAiAgentServiceHelpers.test.ts +1114 -0
- package/Tests/Server/Services/ResourceAiAgentServiceLifecycle.test.ts +1050 -0
- package/Tests/Server/Services/ResourceAiAgentServiceRegister.test.ts +2251 -0
- package/Tests/Server/Services/ResourceAiSettingsCreate.test.ts +373 -0
- package/Tests/Server/Services/ResourceAiSettingsPermission.test.ts +1042 -0
- package/Tests/Server/Services/ResourceServiceDeleteCleansUpAi.test.ts +319 -0
- package/Tests/Server/Services/RunnerJobEnqueueKubectl.test.ts +105 -0
- package/Tests/Server/Services/RunnerJobEnqueueResourceCommand.test.ts +986 -0
- package/Tests/Server/Services/RunnerJobResourceCommandLane.test.ts +211 -0
- package/Tests/Server/Services/RunnerJobResourceCommandTimeoutAndRedaction.test.ts +352 -0
- package/Tests/Server/Services/TelemetryUsageBillingSloExclusion.test.ts +218 -0
- package/Tests/Server/TestingUtils/Services/FakeRunnerJobCount.ts +145 -0
- package/Tests/Server/Utils/AI/InvestigationInfrastructureAccessWiring.test.ts +453 -0
- package/Tests/Server/Utils/AI/InvestigationInfrastructureReport.test.ts +721 -0
- package/Tests/Server/Utils/AI/RemediationCommandTools.test.ts +94 -1
- package/Tests/Server/Utils/AI/RemediationCommandToolsResource.test.ts +1530 -0
- package/Tests/Server/Utils/AI/RemediationExecutionRunnerResourceMode.test.ts +1249 -0
- package/Tests/Server/Utils/AI/RemediationPlanRunner.test.ts +117 -0
- package/Tests/Server/Utils/AI/RemediationResourceCopyParity.test.ts +463 -0
- package/Tests/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.test.ts +722 -0
- package/Tests/Server/Utils/AI/ResourceAccess/ResourceAccessContext.test.ts +266 -0
- package/Tests/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.test.ts +1220 -0
- package/Tests/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.test.ts +569 -0
- package/Tests/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.test.ts +830 -0
- package/Tests/Server/Utils/AutoRemediation/CommandPlanExecutorResource.test.ts +717 -0
- package/Tests/Server/Utils/AutoRemediation/RemediationVerifierResourceFollowUp.test.ts +338 -0
- package/Tests/Server/Utils/Monitor/Criteria/SessionReplayBudgetTemplateCriteria.test.ts +422 -0
- package/Tests/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.test.ts +1647 -0
- package/Tests/Server/Utils/SessionReplay/SessionReplayUsage.test.ts +143 -1
- package/Tests/Server/Utils/Telemetry/TelemetryIngestionKeyGuard.test.ts +33 -3
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsAccountNotLinked.test.ts +1093 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.test.ts +1049 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.test.ts +1676 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCards.test.ts +2884 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.test.ts +2490 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateSubmit.test.ts +2293 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateSubmitServerTimezone.test.ts +386 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.test.ts +1477 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.test.ts +1967 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsStripHtmlTags.test.ts +92 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsSubmittedFormRemoval.test.ts +1819 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.test.ts +1033 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezoneServerZone.test.ts +610 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeamsBotMessageHandling.test.ts +2610 -0
- package/Tests/Server/Utils/Workspace/MicrosoftTeamsCreateCommandsEndToEnd.test.ts +4672 -0
- package/Tests/Server/Utils/Workspace/WorkspaceCreateProjectReferences.test.ts +190 -65
- package/Tests/Types/AutoRemediation/AiRemediationCommandPlan.test.ts +10 -4
- package/Tests/Types/AutoRemediation/AiRemediationCommandPlanResourceCommand.test.ts +305 -0
- package/Tests/Types/Monitor/Recommendation/MonitorRecommendationCatalog.test.ts +277 -13
- package/Tests/Types/Monitor/Recommendation/MonitorRecommendationUtil.test.ts +113 -0
- package/Tests/Types/Monitor/RumAlertTemplates.test.ts +773 -2
- package/Tests/Types/ResourceAiAgent/AiResourceType.test.ts +419 -0
- package/Tests/Types/ResourceAiAgent/ResourceAiAccess.test.ts +517 -0
- package/Tests/Types/Runbook/RunbookStepType.test.ts +147 -8
- package/Tests/UI/Rum/RecordingHealthDashboard.test.tsx +769 -1
- package/Tests/Utils/AiRemediation/Resource/CephCommandPolicy.test.ts +1653 -0
- package/Tests/Utils/AiRemediation/Resource/CephOutputRedaction.test.ts +204 -0
- package/Tests/Utils/AiRemediation/Resource/DatabaseCommandPolicy.test.ts +1077 -0
- package/Tests/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.test.ts +558 -0
- package/Tests/Utils/AiRemediation/Resource/DatabaseQueryRedactor.test.ts +834 -0
- package/Tests/Utils/AiRemediation/Resource/DockerCliGrammar.test.ts +747 -0
- package/Tests/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.test.ts +1259 -0
- package/Tests/Utils/AiRemediation/Resource/DockerOutputRedaction.test.ts +386 -0
- package/Tests/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.test.ts +782 -0
- package/Tests/Utils/AiRemediation/Resource/GovcCommandPolicy.test.ts +2259 -0
- package/Tests/Utils/AiRemediation/Resource/HostCommandPolicy.test.ts +2019 -0
- package/Tests/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.test.ts +2497 -0
- package/Tests/Utils/AiRemediation/Resource/ResourceCommandPolicy.test.ts +1305 -0
- package/Tests/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.test.ts +585 -0
- package/Tests/Utils/AiRemediation/Resource/ResourceOutputRedactor.test.ts +334 -0
- package/Tests/Utils/AiRemediation/Resource/ResourceOutputRedactorCopyParity.test.ts +98 -0
- package/Tests/Utils/AiRemediation/Resource/ResourcePolicyImportClosure.test.ts +391 -0
- package/Tests/Utils/AiRemediation/ResourceAiAgentPolicyCopyParity.test.ts +497 -0
- package/Tests/Utils/SessionReplay/SessionReplayBudgetMetricType.test.ts +383 -0
- package/Types/AI/ResourceAiAccessApi.ts +140 -0
- package/Types/AI/ResourceAiAccessPermissions.ts +126 -0
- package/Types/AutoRemediation/AiRemediationCommandPlan.ts +183 -2
- package/Types/Monitor/Recommendation/MonitorRecommendationCatalog.ts +55 -22
- package/Types/Monitor/Recommendation/MonitorRecommendationTypes.ts +35 -3
- package/Types/Monitor/RumAlertTemplates.ts +351 -4
- package/Types/ResourceAiAgent/AiResourceType.ts +310 -0
- package/Types/ResourceAiAgent/ResourceAiAccess.ts +574 -0
- package/Types/Rum/SessionReplayBudgetMetricType.ts +47 -0
- package/Types/Runbook/RunbookStepType.ts +29 -0
- package/Types/Telemetry/TelemetryIngestSurface.ts +14 -2
- package/Utils/AI/InvestigationReport.ts +121 -15
- package/Utils/AiRemediation/Resource/CephCommandPolicy.ts +1936 -0
- package/Utils/AiRemediation/Resource/DatabaseCommandPolicy.ts +166 -0
- package/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.ts +1532 -0
- package/Utils/AiRemediation/Resource/DatabaseQueryRedactor.ts +1394 -0
- package/Utils/AiRemediation/Resource/DockerCliGrammar.ts +1839 -0
- package/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.ts +490 -0
- package/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.ts +687 -0
- package/Utils/AiRemediation/Resource/GovcCommandPolicy.ts +2096 -0
- package/Utils/AiRemediation/Resource/HostCommandPolicy.ts +2982 -0
- package/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.ts +1679 -0
- package/Utils/AiRemediation/Resource/ResourceCommandPolicy.ts +851 -0
- package/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.ts +593 -0
- package/Utils/AiRemediation/Resource/ResourceOutputRedactor.ts +2812 -0
- package/Utils/SessionReplay/SessionReplayBudgetMetricType.ts +184 -0
- package/build/dist/Models/DatabaseModels/AutoRemediationSuggestion.js +62 -0
- package/build/dist/Models/DatabaseModels/AutoRemediationSuggestion.js.map +1 -1
- package/build/dist/Models/DatabaseModels/CephCluster.js +219 -0
- package/build/dist/Models/DatabaseModels/CephCluster.js.map +1 -1
- package/build/dist/Models/DatabaseModels/DatabaseServer.js +219 -0
- package/build/dist/Models/DatabaseModels/DatabaseServer.js.map +1 -1
- package/build/dist/Models/DatabaseModels/DockerHost.js +219 -0
- package/build/dist/Models/DatabaseModels/DockerHost.js.map +1 -1
- package/build/dist/Models/DatabaseModels/DockerSwarmCluster.js +219 -0
- package/build/dist/Models/DatabaseModels/DockerSwarmCluster.js.map +1 -1
- package/build/dist/Models/DatabaseModels/GlobalConfig.js +1 -1
- package/build/dist/Models/DatabaseModels/GlobalConfig.js.map +1 -1
- package/build/dist/Models/DatabaseModels/Host.js +219 -0
- package/build/dist/Models/DatabaseModels/Host.js.map +1 -1
- package/build/dist/Models/DatabaseModels/Index.js +2 -0
- package/build/dist/Models/DatabaseModels/Index.js.map +1 -1
- package/build/dist/Models/DatabaseModels/PodmanHost.js +219 -0
- package/build/dist/Models/DatabaseModels/PodmanHost.js.map +1 -1
- package/build/dist/Models/DatabaseModels/ProxmoxCluster.js +219 -0
- package/build/dist/Models/DatabaseModels/ProxmoxCluster.js.map +1 -1
- package/build/dist/Models/DatabaseModels/ResourceAiAgent.js +439 -0
- package/build/dist/Models/DatabaseModels/ResourceAiAgent.js.map +1 -0
- package/build/dist/Models/DatabaseModels/RunnerJob.js +145 -0
- package/build/dist/Models/DatabaseModels/RunnerJob.js.map +1 -1
- package/build/dist/Models/DatabaseModels/VMwareVCenter.js +219 -0
- package/build/dist/Models/DatabaseModels/VMwareVCenter.js.map +1 -1
- package/build/dist/Server/API/AutoRemediationAPI.js +231 -1
- package/build/dist/Server/API/AutoRemediationAPI.js.map +1 -1
- package/build/dist/Server/API/ResourceAiAccessAPI.js +1119 -0
- package/build/dist/Server/API/ResourceAiAccessAPI.js.map +1 -0
- package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/1796300000000-AddResourceAiAgents.js +168 -0
- package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/1796300000000-AddResourceAiAgents.js.map +1 -0
- package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/Index.js +2 -0
- package/build/dist/Server/Infrastructure/Postgres/SchemaMigrations/Index.js.map +1 -1
- package/build/dist/Server/Infrastructure/Semaphore.js +6 -0
- package/build/dist/Server/Infrastructure/Semaphore.js.map +1 -1
- package/build/dist/Server/Middleware/TelemetryIngest.js +12 -5
- package/build/dist/Server/Middleware/TelemetryIngest.js.map +1 -1
- package/build/dist/Server/Services/AnalyticsDatabaseService.js +26 -1
- package/build/dist/Server/Services/AnalyticsDatabaseService.js.map +1 -1
- package/build/dist/Server/Services/AutoRemediationRuleEngineService.js +601 -3
- package/build/dist/Server/Services/AutoRemediationRuleEngineService.js.map +1 -1
- package/build/dist/Server/Services/CephClusterService.js +104 -0
- package/build/dist/Server/Services/CephClusterService.js.map +1 -1
- package/build/dist/Server/Services/DatabaseServerService.js +99 -0
- package/build/dist/Server/Services/DatabaseServerService.js.map +1 -1
- package/build/dist/Server/Services/DockerHostService.js +104 -0
- package/build/dist/Server/Services/DockerHostService.js.map +1 -1
- package/build/dist/Server/Services/DockerSwarmClusterService.js +104 -0
- package/build/dist/Server/Services/DockerSwarmClusterService.js.map +1 -1
- package/build/dist/Server/Services/HostService.js +114 -2
- package/build/dist/Server/Services/HostService.js.map +1 -1
- package/build/dist/Server/Services/MetricRecordingRuleService.js +46 -0
- package/build/dist/Server/Services/MetricRecordingRuleService.js.map +1 -1
- package/build/dist/Server/Services/PodmanHostService.js +104 -0
- package/build/dist/Server/Services/PodmanHostService.js.map +1 -1
- package/build/dist/Server/Services/ProxmoxClusterService.js +104 -0
- package/build/dist/Server/Services/ProxmoxClusterService.js.map +1 -1
- package/build/dist/Server/Services/ResourceAiAccessService.js +1028 -0
- package/build/dist/Server/Services/ResourceAiAccessService.js.map +1 -0
- package/build/dist/Server/Services/ResourceAiAgentJobService.js +173 -0
- package/build/dist/Server/Services/ResourceAiAgentJobService.js.map +1 -0
- package/build/dist/Server/Services/ResourceAiAgentService.js +1687 -0
- package/build/dist/Server/Services/ResourceAiAgentService.js.map +1 -0
- package/build/dist/Server/Services/RunnerJobService.js +344 -5
- package/build/dist/Server/Services/RunnerJobService.js.map +1 -1
- package/build/dist/Server/Services/TelemetryUsageBillingService.js +26 -0
- package/build/dist/Server/Services/TelemetryUsageBillingService.js.map +1 -1
- package/build/dist/Server/Services/TraceRecordingRuleService.js +37 -0
- package/build/dist/Server/Services/TraceRecordingRuleService.js.map +1 -1
- package/build/dist/Server/Services/VMwareVCenterService.js +104 -0
- package/build/dist/Server/Services/VMwareVCenterService.js.map +1 -1
- package/build/dist/Server/Utils/AI/Remediation/RemediationCommandTools.js +1029 -30
- package/build/dist/Server/Utils/AI/Remediation/RemediationCommandTools.js.map +1 -1
- package/build/dist/Server/Utils/AI/Remediation/RemediationExecutionRunner.js +633 -28
- package/build/dist/Server/Utils/AI/Remediation/RemediationExecutionRunner.js.map +1 -1
- package/build/dist/Server/Utils/AI/Remediation/RemediationPlanRunner.js +21 -1
- package/build/dist/Server/Utils/AI/Remediation/RemediationPlanRunner.js.map +1 -1
- package/build/dist/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.js +330 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/InfrastructureInvestigationToolkit.js.map +1 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessContext.js +160 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessContext.js.map +1 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessToolNames.js +33 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAccessToolNames.js.map +1 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.js +640 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiAccessSettings.js.map +1 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.js +226 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceAiDeleteCleanup.js.map +1 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.js +600 -0
- package/build/dist/Server/Utils/AI/ResourceAccess/ResourceCommandJobRunner.js.map +1 -0
- package/build/dist/Server/Utils/AI/SRE/AIInvestigationEngine.js +50 -4
- package/build/dist/Server/Utils/AI/SRE/AIInvestigationEngine.js.map +1 -1
- package/build/dist/Server/Utils/AI/SRE/AlertInvestigationRunner.js +55 -4
- package/build/dist/Server/Utils/AI/SRE/AlertInvestigationRunner.js.map +1 -1
- package/build/dist/Server/Utils/AI/SRE/IncidentInvestigationRunner.js +55 -4
- package/build/dist/Server/Utils/AI/SRE/IncidentInvestigationRunner.js.map +1 -1
- package/build/dist/Server/Utils/AutoRemediation/CommandPlanExecutor.js +290 -13
- package/build/dist/Server/Utils/AutoRemediation/CommandPlanExecutor.js.map +1 -1
- package/build/dist/Server/Utils/AutoRemediation/RemediationVerifier.js +29 -1
- package/build/dist/Server/Utils/AutoRemediation/RemediationVerifier.js.map +1 -1
- package/build/dist/Server/Utils/Database/ProjectScopedReferenceValidator.js +9 -1
- package/build/dist/Server/Utils/Database/ProjectScopedReferenceValidator.js.map +1 -1
- package/build/dist/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.js +627 -0
- package/build/dist/Server/Utils/SessionReplay/SessionReplayBudgetMetrics.js.map +1 -0
- package/build/dist/Server/Utils/SessionReplay/SessionReplayUsage.js +60 -5
- package/build/dist/Server/Utils/SessionReplay/SessionReplayUsage.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Alert.js +19 -15
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Alert.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/AlertEpisode.js +19 -15
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/AlertEpisode.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Auth.js +17 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Auth.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Incident.js +183 -209
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/Incident.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/IncidentEpisode.js +19 -15
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/IncidentEpisode.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/ScheduledMaintenance.js +233 -167
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/Actions/ScheduledMaintenance.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeams.js +263 -110
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeams.js.map +1 -1
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.js +129 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsActivityDeduplicator.js.map +1 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.js +266 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCardChoices.js.map +1 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.js +258 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsCreateCommands.js.map +1 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.js +104 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsMessageSize.js.map +1 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.js +183 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsReplies.js.map +1 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.js +123 -0
- package/build/dist/Server/Utils/Workspace/MicrosoftTeams/MicrosoftTeamsTimezone.js.map +1 -0
- package/build/dist/Types/AI/ResourceAiAccessApi.js +30 -0
- package/build/dist/Types/AI/ResourceAiAccessApi.js.map +1 -0
- package/build/dist/Types/AI/ResourceAiAccessPermissions.js +98 -0
- package/build/dist/Types/AI/ResourceAiAccessPermissions.js.map +1 -0
- package/build/dist/Types/AutoRemediation/AiRemediationCommandPlan.js +116 -29
- package/build/dist/Types/AutoRemediation/AiRemediationCommandPlan.js.map +1 -1
- package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationCatalog.js +47 -21
- package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationCatalog.js.map +1 -1
- package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationTypes.js +5 -0
- package/build/dist/Types/Monitor/Recommendation/MonitorRecommendationTypes.js.map +1 -1
- package/build/dist/Types/Monitor/RumAlertTemplates.js +199 -3
- package/build/dist/Types/Monitor/RumAlertTemplates.js.map +1 -1
- package/build/dist/Types/ResourceAiAgent/AiResourceType.js +242 -0
- package/build/dist/Types/ResourceAiAgent/AiResourceType.js.map +1 -0
- package/build/dist/Types/ResourceAiAgent/ResourceAiAccess.js +275 -0
- package/build/dist/Types/ResourceAiAgent/ResourceAiAccess.js.map +1 -0
- package/build/dist/Types/Rum/SessionReplayBudgetMetricType.js +45 -0
- package/build/dist/Types/Rum/SessionReplayBudgetMetricType.js.map +1 -0
- package/build/dist/Types/Runbook/RunbookStepType.js +27 -0
- package/build/dist/Types/Runbook/RunbookStepType.js.map +1 -1
- package/build/dist/Types/Telemetry/TelemetryIngestSurface.js +13 -2
- package/build/dist/Types/Telemetry/TelemetryIngestSurface.js.map +1 -1
- package/build/dist/Utils/AI/InvestigationReport.js +71 -11
- package/build/dist/Utils/AI/InvestigationReport.js.map +1 -1
- package/build/dist/Utils/AiRemediation/Resource/CephCommandPolicy.js +1462 -0
- package/build/dist/Utils/AiRemediation/Resource/CephCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/DatabaseCommandPolicy.js +122 -0
- package/build/dist/Utils/AiRemediation/Resource/DatabaseCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.js +1139 -0
- package/build/dist/Utils/AiRemediation/Resource/DatabaseDiagnosticCatalog.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/DatabaseQueryRedactor.js +1060 -0
- package/build/dist/Utils/AiRemediation/Resource/DatabaseQueryRedactor.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/DockerCliGrammar.js +1301 -0
- package/build/dist/Utils/AiRemediation/Resource/DockerCliGrammar.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.js +370 -0
- package/build/dist/Utils/AiRemediation/Resource/DockerEngineCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.js +498 -0
- package/build/dist/Utils/AiRemediation/Resource/DockerSwarmCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/GovcCommandPolicy.js +1565 -0
- package/build/dist/Utils/AiRemediation/Resource/GovcCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/HostCommandPolicy.js +2206 -0
- package/build/dist/Utils/AiRemediation/Resource/HostCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.js +1129 -0
- package/build/dist/Utils/AiRemediation/Resource/ProxmoxCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicy.js +565 -0
- package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicy.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.js +408 -0
- package/build/dist/Utils/AiRemediation/Resource/ResourceCommandPolicyCore.js.map +1 -0
- package/build/dist/Utils/AiRemediation/Resource/ResourceOutputRedactor.js +1910 -0
- package/build/dist/Utils/AiRemediation/Resource/ResourceOutputRedactor.js.map +1 -0
- package/build/dist/Utils/SessionReplay/SessionReplayBudgetMetricType.js +160 -0
- package/build/dist/Utils/SessionReplay/SessionReplayBudgetMetricType.js.map +1 -0
- package/package.json +1 -1
|
@@ -5,18 +5,24 @@ import RunnerJobOrigin from "../../../../Types/Runbook/RunnerJobOrigin";
|
|
|
5
5
|
import RunnerJobStatus from "../../../../Types/Runbook/RunnerJobStatus";
|
|
6
6
|
import AIRunStatus from "../../../../Types/AI/AIRunStatus";
|
|
7
7
|
import AutoRemediationSuggestionStatus from "../../../../Types/AutoRemediation/AutoRemediationSuggestionStatus";
|
|
8
|
-
import { AI_COMMAND_STEP_TYPES, AiRemediationCommandExecutionStatus, AiRemediationCommandPolicyVerdict, AiRemediationCommandPlanUtil, DEFAULT_COMMAND_TIMEOUT_MS, INLINE_COMMAND_STEP_ID_PREFIX, KUBECTL_ALLOWLIST_SUMMARY, KUBECTL_ALWAYS_ASKS_SUMMARY, KUBECTL_NEVER_RUNS_SUMMARY, KUBECTL_RISKIER_CHANGES_SUMMARY, KUBECTL_SAFE_CHANGES_SUMMARY, MAX_COMMAND_TIMEOUT_MS, MAX_PLAN_COMMANDS, MIN_COMMAND_TIMEOUT_MS, getKubectlAlwaysAsksSummary, getKubectlRiskierChangesSummary, getKubectlSafeChangesSummary, } from "../../../../Types/AutoRemediation/AiRemediationCommandPlan";
|
|
8
|
+
import { AI_COMMAND_STEP_TYPES, AiRemediationCommandExecutionStatus, AiRemediationCommandPolicyVerdict, AiRemediationCommandPlanUtil, DEFAULT_COMMAND_TIMEOUT_MS, INLINE_COMMAND_STEP_ID_PREFIX, KUBECTL_ALLOWLIST_SUMMARY, KUBECTL_ALWAYS_ASKS_SUMMARY, KUBECTL_NEVER_RUNS_SUMMARY, KUBECTL_RISKIER_CHANGES_SUMMARY, KUBECTL_SAFE_CHANGES_SUMMARY, MAX_COMMAND_TIMEOUT_MS, MAX_PLAN_COMMANDS, MIN_COMMAND_TIMEOUT_MS, RESOURCE_ALLOWLIST_SUMMARY, RESOURCE_ALWAYS_ASKS_SUMMARY, RESOURCE_NEVER_RUNS_SUMMARY, RESOURCE_RISKIER_CHANGES_SUMMARY, RESOURCE_SAFE_CHANGES_SUMMARY, RESOURCE_UNATTENDED_ROUND_BECOMES_PROPOSAL_SUMMARY, getKubectlAlwaysAsksSummary, getKubectlRiskierChangesSummary, getKubectlSafeChangesSummary, } from "../../../../Types/AutoRemediation/AiRemediationCommandPlan";
|
|
9
|
+
import { AI_RESOURCE_TYPE_INFO, ALL_AI_RESOURCE_TYPES, isAiResourceType, } from "../../../../Types/ResourceAiAgent/AiResourceType";
|
|
10
|
+
import { MAX_RESOURCE_COMMAND_TIMEOUT_MS, RESOURCE_AI_ALLOW_WRITES_ENV, RESOURCE_AI_WRITE_TARGETS_ENV, ResourceAiRemediationMode, ResourceCommandTier, isUnattendedResourceRemediationMode, } from "../../../../Types/ResourceAiAgent/ResourceAiAccess";
|
|
11
|
+
import ResourceCommandPolicy from "../../../../Utils/AiRemediation/Resource/ResourceCommandPolicy";
|
|
12
|
+
import ResourceAiAccessService, { describeResourceNoun, } from "../../../Services/ResourceAiAccessService";
|
|
13
|
+
import ResourceCommandJobRunner, { RESOURCE_COMMAND_CLAIM_TIMEOUT_MS, ResourceCommandRunState, } from "../ResourceAccess/ResourceCommandJobRunner";
|
|
14
|
+
import { RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME } from "../ResourceAccess/ResourceAccessToolNames";
|
|
9
15
|
import { KUBECTL_ALLOW_NODE_OPERATIONS_ENV, KUBECTL_WRITE_NAMESPACES_ENV, KubectlCommandTier, KubernetesAiRemediationMode, isKubernetesAgentRunnerName, isKubernetesAgentRunnerPosture, getKubernetesAiAccessTargetKind, isUnattendedRemediationMode, parseKubernetesRunnerPosture, } from "../../../../Types/Kubernetes/KubernetesClusterAiAccess";
|
|
10
16
|
import CommandPolicy from "../../../../Utils/AiRemediation/CommandPolicy";
|
|
11
17
|
import KubectlPolicy from "../../../../Utils/AiRemediation/KubectlPolicy";
|
|
12
18
|
import KubectlWriteScope from "../../../../Utils/AiRemediation/KubectlWriteScope";
|
|
13
19
|
import AIRunService from "../../../Services/AIRunService";
|
|
14
|
-
import AutoRemediationRuleEngineService, { MAX_AUTO_EXECUTIONS_PER_RULE_PER_HOUR, } from "../../../Services/AutoRemediationRuleEngineService";
|
|
20
|
+
import AutoRemediationRuleEngineService, { MAX_AUTO_EXECUTIONS_PER_RULE_PER_HOUR, RESOURCE_BREAKER_LOCK_NAMESPACE, doesResourceModeRunRoundUnattended, getResourceBreakerLockKey, } from "../../../Services/AutoRemediationRuleEngineService";
|
|
15
21
|
import AutoRemediationSuggestionService from "../../../Services/AutoRemediationSuggestionService";
|
|
16
22
|
import KubernetesClusterAiAccessService from "../../../Services/KubernetesClusterAiAccessService";
|
|
17
23
|
import Semaphore from "../../../Infrastructure/Semaphore";
|
|
18
24
|
import RunbookCredentialService from "../../../Services/RunbookCredentialService";
|
|
19
|
-
import RunnerJobService, { isTerminalAgentJobStatus, MAX_AI_COMMAND_JOBS_PER_PROJECT_PER_HOUR, } from "../../../Services/RunnerJobService";
|
|
25
|
+
import RunnerJobService, { isTerminalAgentJobStatus, MAX_AI_COMMAND_JOBS_PER_PROJECT_PER_HOUR, MAX_AI_RESOURCE_COMMAND_JOBS_PER_PROJECT_PER_HOUR, } from "../../../Services/RunnerJobService";
|
|
20
26
|
import RunnerService from "../../../Services/RunnerService";
|
|
21
27
|
import QueryHelper from "../../../Types/Database/QueryHelper";
|
|
22
28
|
import { ToolArgs } from "../Toolbox/ToolTypes";
|
|
@@ -138,6 +144,27 @@ const KUBECTL_UNFINISHED_CHECK_FIRST = "kubectl did not finish (it was stopped b
|
|
|
138
144
|
const CLUSTER_BREAKER_LOCK_NAMESPACE = "AutoRemediationClusterBreaker";
|
|
139
145
|
const CLUSTER_BREAKER_LOCK_TIMEOUT_MS = 60000;
|
|
140
146
|
const CLUSTER_BREAKER_LOCK_ACQUIRE_TIMEOUT_MS = 20000;
|
|
147
|
+
/*
|
|
148
|
+
* What the model is told when a resource command ran but did not finish
|
|
149
|
+
* (no exit code): the change may have landed before the agent stopped it.
|
|
150
|
+
*/
|
|
151
|
+
const RESOURCE_UNFINISHED_CHECK_FIRST = `The command did not finish (the resource's AI agent stopped it before it reported an exit code), so it MAY have changed the resource. Before you reissue this command, or run anything that depends on it, check with a read (${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME}) whether it took effect. Do NOT resend it blindly.`;
|
|
152
|
+
/*
|
|
153
|
+
* The key a resource target, and the commands aimed at it, are matched by:
|
|
154
|
+
* the type and the (case-insensitive) id.
|
|
155
|
+
*/
|
|
156
|
+
function getResourceKey(resourceType, resourceId) {
|
|
157
|
+
if (!resourceType || !resourceId) {
|
|
158
|
+
return "";
|
|
159
|
+
}
|
|
160
|
+
return `${resourceType}:${resourceId.toLowerCase()}`;
|
|
161
|
+
}
|
|
162
|
+
// 'Docker host "web-1"' — how the toolkit names a resource in copy.
|
|
163
|
+
function describeResourceLabel(data) {
|
|
164
|
+
return `${isAiResourceType(data.resourceType)
|
|
165
|
+
? describeResourceNoun(data.resourceType)
|
|
166
|
+
: "resource"} "${data.resourceName || "(unknown)"}"`;
|
|
167
|
+
}
|
|
141
168
|
/*
|
|
142
169
|
* Verbs that never run on their own in any mode, whatever the allowlist
|
|
143
170
|
* says. A patch of a Node never does either; the policy marks all three
|
|
@@ -173,6 +200,11 @@ export default class RemediationCommandToolkit {
|
|
|
173
200
|
* treats them as if they had never been targets.
|
|
174
201
|
*/
|
|
175
202
|
this.revokedClusterIds = new Set();
|
|
203
|
+
/*
|
|
204
|
+
* The same for resource targets (by getResourceKey): their AI page
|
|
205
|
+
* stopped allowing this run to change them mid-run.
|
|
206
|
+
*/
|
|
207
|
+
this.revokedResourceKeys = new Set();
|
|
176
208
|
/*
|
|
177
209
|
* Commands this run sent to a Runner that never reached kubectl. They are
|
|
178
210
|
* not executed commands, but their jobs exist: their sequence numbers
|
|
@@ -197,7 +229,30 @@ export default class RemediationCommandToolkit {
|
|
|
197
229
|
*/
|
|
198
230
|
getCommandsNeedingApproval() {
|
|
199
231
|
return this.commandsNeedingApproval.filter((kept) => {
|
|
200
|
-
return !this.revokedClusterIds.has(kept.command.kubernetesClusterId || "")
|
|
232
|
+
return (!this.revokedClusterIds.has(kept.command.kubernetesClusterId || "") &&
|
|
233
|
+
!this.revokedResourceKeys.has(getResourceKey(kept.command.resourceType, kept.command.resourceId)));
|
|
234
|
+
});
|
|
235
|
+
}
|
|
236
|
+
/*
|
|
237
|
+
* Is this a resource round? Fixed at construction (from the targets it was
|
|
238
|
+
* given, not the ones still allowed), so the tools the model was handed
|
|
239
|
+
* never change shape mid-run.
|
|
240
|
+
*/
|
|
241
|
+
isResourceRound() {
|
|
242
|
+
return (this.options.resourceTargets || []).length > 0;
|
|
243
|
+
}
|
|
244
|
+
/*
|
|
245
|
+
* The resources this run may still change: remediation-ready, with an AI
|
|
246
|
+
* agent, and not revoked by a live re-check.
|
|
247
|
+
*/
|
|
248
|
+
getResourceTargets() {
|
|
249
|
+
return (this.options.resourceTargets || []).filter((resource) => {
|
|
250
|
+
var _a;
|
|
251
|
+
return (resource.isRemediationReady &&
|
|
252
|
+
isAiResourceType(resource.resourceType) &&
|
|
253
|
+
resource.agent !== null &&
|
|
254
|
+
Boolean((_a = resource.agent) === null || _a === void 0 ? void 0 : _a.agentId) &&
|
|
255
|
+
!this.revokedResourceKeys.has(getResourceKey(resource.resourceType, resource.resourceId)));
|
|
201
256
|
});
|
|
202
257
|
}
|
|
203
258
|
getClusterTargets() {
|
|
@@ -228,7 +283,9 @@ export default class RemediationCommandToolkit {
|
|
|
228
283
|
return {
|
|
229
284
|
definition: {
|
|
230
285
|
name: "list_command_targets",
|
|
231
|
-
description:
|
|
286
|
+
description: this.isResourceRound()
|
|
287
|
+
? "List where this remediation may run commands: the infrastructure resource this round is about (stepType ResourceCommand runs ONE command on it through its own AI agent), with its resourceId, programs, remediation mode, write scope and command allowlist. Call this before composing any command."
|
|
288
|
+
: "List where this remediation may run commands: Runners (Bash runs on the Runner's host; SSH runs on an assigned credential's host) and Kubernetes clusters (Kubectl runs through the cluster's Kubernetes AI agent or Runner). Call this before composing any command.",
|
|
232
289
|
inputSchema: {
|
|
233
290
|
type: "object",
|
|
234
291
|
properties: {},
|
|
@@ -312,6 +369,42 @@ export default class RemediationCommandToolkit {
|
|
|
312
369
|
? cluster.kubectlAllowlist.join(" | ")
|
|
313
370
|
: "(none)", allowlistMatching: KUBECTL_ALLOWLIST_SUMMARY }));
|
|
314
371
|
}
|
|
372
|
+
for (const resource of this.getResourceTargets()) {
|
|
373
|
+
const info = AI_RESOURCE_TYPE_INFO[resource.resourceType];
|
|
374
|
+
rows.push({
|
|
375
|
+
targetType: "Resource",
|
|
376
|
+
resourceType: resource.resourceType,
|
|
377
|
+
resourceId: resource.resourceId,
|
|
378
|
+
name: resource.resourceName,
|
|
379
|
+
stepTypes: "ResourceCommand",
|
|
380
|
+
via: `its ${info.agentDisplayName}`,
|
|
381
|
+
programs: info.programs.join(", "),
|
|
382
|
+
// One idea per field, as for clusters (the serializer caps each).
|
|
383
|
+
remediationMode: this.describeResourceModeForLlm(resource),
|
|
384
|
+
safeChanges: RESOURCE_SAFE_CHANGES_SUMMARY,
|
|
385
|
+
riskierChanges: RESOURCE_RISKIER_CHANGES_SUMMARY,
|
|
386
|
+
alwaysNeedsAHuman: RESOURCE_ALWAYS_ASKS_SUMMARY,
|
|
387
|
+
neverRuns: RESOURCE_NEVER_RUNS_SUMMARY,
|
|
388
|
+
writeScope: RemediationCommandToolkit.describeResourceWriteScope(resource),
|
|
389
|
+
commandAllowlist: resource.aiCommandAllowlist.length > 0
|
|
390
|
+
? resource.aiCommandAllowlist.join(" | ")
|
|
391
|
+
: "(none)",
|
|
392
|
+
allowlistMatching: RESOURCE_ALLOWLIST_SUMMARY,
|
|
393
|
+
});
|
|
394
|
+
}
|
|
395
|
+
if (rows.length === 0 && this.isResourceRound()) {
|
|
396
|
+
return {
|
|
397
|
+
success: true,
|
|
398
|
+
textForLlm: "The infrastructure resource this round is about no longer allows AI remediation (its AI agent page changed during this run, or its AI agent went away). You cannot run or propose commands — say so in your analysis.",
|
|
399
|
+
result: {
|
|
400
|
+
dataForLlm: "(no command targets available)",
|
|
401
|
+
rowCount: 0,
|
|
402
|
+
citationLabel: "Available command targets",
|
|
403
|
+
redactionCount: 0,
|
|
404
|
+
isTruncated: false,
|
|
405
|
+
},
|
|
406
|
+
};
|
|
407
|
+
}
|
|
315
408
|
if (rows.length === 0) {
|
|
316
409
|
return {
|
|
317
410
|
success: true,
|
|
@@ -357,6 +450,92 @@ export default class RemediationCommandToolkit {
|
|
|
357
450
|
}
|
|
358
451
|
return "RequireApproval: every kubectl change is proposed for one-click approval";
|
|
359
452
|
}
|
|
453
|
+
/*
|
|
454
|
+
* describeClusterModeForLlm for a resource: the canonical
|
|
455
|
+
* ResourceAiRemediationMode semantics in one field of at most the
|
|
456
|
+
* serializer's 500 characters.
|
|
457
|
+
*/
|
|
458
|
+
describeResourceModeForLlm(resource) {
|
|
459
|
+
if (resource.aiRemediationMode === ResourceAiRemediationMode.BypassApproval) {
|
|
460
|
+
return `BypassApproval: AI does not ask — every change the policy allows, safeChanges AND riskierChanges, runs without a human, except alwaysNeedsAHuman${this.options.proposesRefusedCommands
|
|
461
|
+
? " (submit such a change anyway: it is refused, recorded, and proposed for one-click approval when this round ends if no other change ran)"
|
|
462
|
+
: ""}; and ${RESOURCE_UNATTENDED_ROUND_BECOMES_PROPOSAL_SUMMARY}`;
|
|
463
|
+
}
|
|
464
|
+
if (resource.aiRemediationMode === ResourceAiRemediationMode.Automatic) {
|
|
465
|
+
return `Automatic: safeChanges run without a human; riskierChanges never run inline unless the commandAllowlist names their exact shape, and alwaysNeedsAHuman never does — ${this.options.proposesRefusedCommands
|
|
466
|
+
? "submit it anyway: it is refused, recorded, and proposed for one-click approval when this round ends if no other change ran; put it in your written recommendations too"
|
|
467
|
+
: "put it in your written recommendations for a human"}; and ${RESOURCE_UNATTENDED_ROUND_BECOMES_PROPOSAL_SUMMARY}`;
|
|
468
|
+
}
|
|
469
|
+
return "RequireApproval: every change is proposed for one-click approval";
|
|
470
|
+
}
|
|
471
|
+
/*
|
|
472
|
+
* Where the resource's AI agent lets AI-composed writes land, as it last
|
|
473
|
+
* reported (its posture): read-only unless ONEUPTIME_AI_ALLOW_WRITES is
|
|
474
|
+
* true, never its protected targets, and — with ONEUPTIME_AI_WRITE_TARGETS
|
|
475
|
+
* set — only the targets its globs name.
|
|
476
|
+
*/
|
|
477
|
+
static describeResourceWriteScope(resource) {
|
|
478
|
+
var _a;
|
|
479
|
+
const posture = (_a = resource.agent) === null || _a === void 0 ? void 0 : _a.posture;
|
|
480
|
+
if (!posture) {
|
|
481
|
+
return "not reported by the agent yet, so it runs no change";
|
|
482
|
+
}
|
|
483
|
+
if (posture.allowWrites !== true) {
|
|
484
|
+
return `read-only (${RESOURCE_AI_ALLOW_WRITES_ENV} is not true on the agent): it refuses every change`;
|
|
485
|
+
}
|
|
486
|
+
const parts = [
|
|
487
|
+
posture.writeTargets.length > 0
|
|
488
|
+
? `changes only targets matching ${RESOURCE_AI_WRITE_TARGETS_ENV}=${posture.writeTargets.join(",")} — name each target exactly`
|
|
489
|
+
: "changes any target its command policy allows",
|
|
490
|
+
];
|
|
491
|
+
if (posture.protectedTargets.length > 0) {
|
|
492
|
+
parts.push(`never its protected targets (${posture.protectedTargets
|
|
493
|
+
.slice(0, 10)
|
|
494
|
+
.join(", ")}): itself and what it runs in`);
|
|
495
|
+
}
|
|
496
|
+
return parts.join("; ");
|
|
497
|
+
}
|
|
498
|
+
/*
|
|
499
|
+
* Why the resource's AI agent would refuse this write, going by the
|
|
500
|
+
* posture it reported — or null when it would not. The rule is the
|
|
501
|
+
* agent's own (ResourceCommandPolicy.getWriteScopeRefusal: read-only
|
|
502
|
+
* unless ONEUPTIME_AI_ALLOW_WRITES is true, never a protected target,
|
|
503
|
+
* only the ONEUPTIME_AI_WRITE_TARGETS globs when set), the one the enqueue
|
|
504
|
+
* chokepoint asks too. Reads and Denied commands are not this rule's
|
|
505
|
+
* business (the policy refuses the latter on its own).
|
|
506
|
+
*/
|
|
507
|
+
static getResourceWriteScopeRefusal(data) {
|
|
508
|
+
var _a;
|
|
509
|
+
if (!isAiResourceType(data.resource.resourceType)) {
|
|
510
|
+
return "The resource type is unknown, so no change can run on it.";
|
|
511
|
+
}
|
|
512
|
+
const policy = ResourceCommandPolicy.evaluateCommand({
|
|
513
|
+
resourceType: data.resource.resourceType,
|
|
514
|
+
command: data.command,
|
|
515
|
+
});
|
|
516
|
+
if (policy.tier === ResourceCommandTier.Read ||
|
|
517
|
+
policy.tier === ResourceCommandTier.Denied) {
|
|
518
|
+
return null;
|
|
519
|
+
}
|
|
520
|
+
const posture = (_a = data.resource.agent) === null || _a === void 0 ? void 0 : _a.posture;
|
|
521
|
+
return ResourceCommandPolicy.getWriteScopeRefusal({
|
|
522
|
+
result: policy,
|
|
523
|
+
allowWrites: (posture === null || posture === void 0 ? void 0 : posture.allowWrites) === true,
|
|
524
|
+
writeTargets: (posture === null || posture === void 0 ? void 0 : posture.writeTargets) || [],
|
|
525
|
+
protectedTargets: (posture === null || posture === void 0 ? void 0 : posture.protectedTargets) || [],
|
|
526
|
+
resourceType: data.resource.resourceType,
|
|
527
|
+
});
|
|
528
|
+
}
|
|
529
|
+
/*
|
|
530
|
+
* What to do about a resource write-scope refusal, for the approve route
|
|
531
|
+
* and the model: the agent's own environment decides.
|
|
532
|
+
*/
|
|
533
|
+
static getResourceScopeRefusalNextStep(resourceType) {
|
|
534
|
+
const agentName = isAiResourceType(resourceType)
|
|
535
|
+
? AI_RESOURCE_TYPE_INFO[resourceType].agentDisplayName
|
|
536
|
+
: "resource's AI agent";
|
|
537
|
+
return `Dismiss the suggestion and let a new plan be composed, or change the ${agentName}'s write access (${RESOURCE_AI_ALLOW_WRITES_ENV} and ${RESOURCE_AI_WRITE_TARGETS_ENV} in its environment), restart it, and approve again.`;
|
|
538
|
+
}
|
|
360
539
|
/*
|
|
361
540
|
* Which kubectl changes the model may be offered on this cluster. A
|
|
362
541
|
* Runner that reported node operations off refuses every one, approved
|
|
@@ -605,6 +784,28 @@ export default class RemediationCommandToolkit {
|
|
|
605
784
|
* ------------------------------------------------------------------
|
|
606
785
|
*/
|
|
607
786
|
buildExecuteTool() {
|
|
787
|
+
if (this.isResourceRound()) {
|
|
788
|
+
return {
|
|
789
|
+
definition: {
|
|
790
|
+
name: "execute_remediation_command",
|
|
791
|
+
description: this.describeResourceExecuteTool(),
|
|
792
|
+
inputSchema: {
|
|
793
|
+
type: "object",
|
|
794
|
+
properties: this.buildCommandSchemaProperties(),
|
|
795
|
+
required: [
|
|
796
|
+
"stepType",
|
|
797
|
+
"resourceId",
|
|
798
|
+
"command",
|
|
799
|
+
"rationale",
|
|
800
|
+
"expectedEffect",
|
|
801
|
+
],
|
|
802
|
+
},
|
|
803
|
+
},
|
|
804
|
+
execute: async (args) => {
|
|
805
|
+
return this.executeCommand(args);
|
|
806
|
+
},
|
|
807
|
+
};
|
|
808
|
+
}
|
|
608
809
|
return {
|
|
609
810
|
definition: {
|
|
610
811
|
name: "execute_remediation_command",
|
|
@@ -622,7 +823,80 @@ export default class RemediationCommandToolkit {
|
|
|
622
823
|
},
|
|
623
824
|
};
|
|
624
825
|
}
|
|
826
|
+
/*
|
|
827
|
+
* What each resource this round may change accepts, by tier — its tool
|
|
828
|
+
* policy's own guide, once per policy.
|
|
829
|
+
*/
|
|
830
|
+
describeResourceWriteGuides() {
|
|
831
|
+
const sections = [];
|
|
832
|
+
const seenPolicies = new Set();
|
|
833
|
+
for (const type of ALL_AI_RESOURCE_TYPES) {
|
|
834
|
+
const resources = (this.options.resourceTargets || []).filter((resource) => {
|
|
835
|
+
return resource.resourceType === type;
|
|
836
|
+
});
|
|
837
|
+
if (resources.length === 0) {
|
|
838
|
+
continue;
|
|
839
|
+
}
|
|
840
|
+
const policyName = ResourceCommandPolicy.getToolPolicy(type).name;
|
|
841
|
+
if (seenPolicies.has(policyName)) {
|
|
842
|
+
continue;
|
|
843
|
+
}
|
|
844
|
+
seenPolicies.add(policyName);
|
|
845
|
+
sections.push(`${resources
|
|
846
|
+
.map((resource) => {
|
|
847
|
+
return `${AI_RESOURCE_TYPE_INFO[type].displayName} "${resource.resourceName}" (resourceId: ${resource.resourceId})`;
|
|
848
|
+
})
|
|
849
|
+
.join(", ")} — changes it accepts:\n${ResourceCommandPolicy.getWriteCommandGuide(type)}`);
|
|
850
|
+
}
|
|
851
|
+
return sections.join("\n\n");
|
|
852
|
+
}
|
|
853
|
+
// execute_remediation_command's description on a resource round.
|
|
854
|
+
describeResourceExecuteTool() {
|
|
855
|
+
return `Execute ONE remediation change immediately on the infrastructure resource this round is about, through its own AI agent: stepType ResourceCommand with the resource's resourceId. One command per call, written as the program followed by its arguments — never a shell line. This tool is for CHANGES only — a read-only command goes through ${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME} and is refused here. Safe changes run (${RESOURCE_SAFE_CHANGES_SUMMARY}). Riskier changes (${RESOURCE_RISKIER_CHANGES_SUMMARY}) are refused unless the resource's command allowlist names their exact shape (${RESOURCE_ALLOWLIST_SUMMARY}) or the resource bypasses approvals — ${this.options.proposesRefusedCommands
|
|
856
|
+
? "a refused riskier change is recorded and proposed for one-click approval when this round ends, provided no other change ran"
|
|
857
|
+
: "put those in your recommendations"}. Whatever the mode, ${RESOURCE_ALWAYS_ASKS_SUMMARY}; ${RESOURCE_UNATTENDED_ROUND_BECOMES_PROPOSAL_SUMMARY}; ${RESOURCE_NEVER_RUNS_SUMMARY}. A change the resource's AI agent would refuse (it runs read-only, or the target is protected or outside its writeScope in list_command_targets) is refused before it runs. At most ${MAX_AUTO_EXECUTED_COMMANDS_PER_RUN} commands may be sent per remediation. Provide a rollbackCommand whenever the command changes state and an undo exists.\n\n${this.describeResourceWriteGuides()}`;
|
|
858
|
+
}
|
|
859
|
+
/*
|
|
860
|
+
* The command schema of a resource round: ResourceCommand only, on the
|
|
861
|
+
* round's resource — never a Runner, a cluster or a credential.
|
|
862
|
+
*/
|
|
863
|
+
buildResourceCommandSchemaProperties() {
|
|
864
|
+
return {
|
|
865
|
+
resourceId: {
|
|
866
|
+
type: "string",
|
|
867
|
+
description: "The resource to change: its resourceId from list_command_targets.",
|
|
868
|
+
},
|
|
869
|
+
stepType: {
|
|
870
|
+
type: "string",
|
|
871
|
+
enum: ["ResourceCommand"],
|
|
872
|
+
description: "ResourceCommand runs ONE command on the resource through its own AI agent.",
|
|
873
|
+
},
|
|
874
|
+
command: {
|
|
875
|
+
type: "string",
|
|
876
|
+
description: 'The exact command, one line starting with one of the resource\'s programs (list_command_targets), e.g. "docker restart web", "docker service update --force api", "systemctl restart nginx", "pvesh create /nodes/pve1/qemu/100/status/start", "govc vm.power -on /DC/vm/web-01", "ceph osd in 3" or "db cancel-query 4242". No pipes, redirects, chaining, substitution or sudo.',
|
|
877
|
+
},
|
|
878
|
+
timeoutInMs: {
|
|
879
|
+
type: "number",
|
|
880
|
+
description: `Execution timeout in milliseconds (default ${Math.min(DEFAULT_COMMAND_TIMEOUT_MS, MAX_RESOURCE_COMMAND_TIMEOUT_MS)}, max ${MAX_RESOURCE_COMMAND_TIMEOUT_MS}).`,
|
|
881
|
+
},
|
|
882
|
+
rationale: {
|
|
883
|
+
type: "string",
|
|
884
|
+
description: "Why this command remediates the incident — shown to humans verbatim.",
|
|
885
|
+
},
|
|
886
|
+
expectedEffect: {
|
|
887
|
+
type: "string",
|
|
888
|
+
description: "What you expect to observe if it works.",
|
|
889
|
+
},
|
|
890
|
+
rollbackCommand: {
|
|
891
|
+
type: "string",
|
|
892
|
+
description: "Optional undo command for the same resource, run unattended if verification later fails. Must pass the same policy, and — unless the resource bypasses approvals — be a safe change on ONE named object, e.g. docker start <container> after docker stop <container>, or systemctl start <unit> after systemctl stop <unit>.",
|
|
893
|
+
},
|
|
894
|
+
};
|
|
895
|
+
}
|
|
625
896
|
buildCommandSchemaProperties() {
|
|
897
|
+
if (this.isResourceRound()) {
|
|
898
|
+
return this.buildResourceCommandSchemaProperties();
|
|
899
|
+
}
|
|
626
900
|
return {
|
|
627
901
|
runnerId: {
|
|
628
902
|
type: "string",
|
|
@@ -714,6 +988,35 @@ export default class RemediationCommandToolkit {
|
|
|
714
988
|
return this.failure(scopeRefusal);
|
|
715
989
|
}
|
|
716
990
|
}
|
|
991
|
+
if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
992
|
+
/*
|
|
993
|
+
* Reads belong to run_infrastructure_command, for the reason they
|
|
994
|
+
* belong to run_kubectl on a cluster: sent here, a read would count
|
|
995
|
+
* as an executed fix, take a breaker slot and could auto-resolve the
|
|
996
|
+
* signal in the name of a change that changed nothing.
|
|
997
|
+
*/
|
|
998
|
+
if (command.resourceCommandTier === ResourceCommandTier.Read) {
|
|
999
|
+
return this.failure(`"${command.command}" is read-only. Run it with ${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME}, which does not count as a remediation command — this tool is for changes only. Nothing was executed.`);
|
|
1000
|
+
}
|
|
1001
|
+
// The resource as its AI page stands NOW, as for a cluster.
|
|
1002
|
+
const liveRefusal = await this.refreshResourceTarget(command);
|
|
1003
|
+
if (liveRefusal) {
|
|
1004
|
+
this.recordNeedingApproval(command, liveRefusal);
|
|
1005
|
+
return this.failure(liveRefusal.text);
|
|
1006
|
+
}
|
|
1007
|
+
/*
|
|
1008
|
+
* The agent's write scope as it reports it NOW: a write it would
|
|
1009
|
+
* refuse is never enqueued — nor kept for a proposal, which no click
|
|
1010
|
+
* could make runnable.
|
|
1011
|
+
*/
|
|
1012
|
+
const liveResource = this.findResourceTarget(command.resourceType, command.resourceId);
|
|
1013
|
+
const scopeRefusal = liveResource
|
|
1014
|
+
? this.getResourceCommandScopeRefusal(liveResource, command)
|
|
1015
|
+
: null;
|
|
1016
|
+
if (scopeRefusal) {
|
|
1017
|
+
return this.failure(scopeRefusal);
|
|
1018
|
+
}
|
|
1019
|
+
}
|
|
717
1020
|
/*
|
|
718
1021
|
* FullAuto gate: the full policy. Anything that is not AutoApproved is
|
|
719
1022
|
* refused here; the model is told why so it can pick an allowlisted
|
|
@@ -725,17 +1028,44 @@ export default class RemediationCommandToolkit {
|
|
|
725
1028
|
return this.failure(gateFailure.text);
|
|
726
1029
|
}
|
|
727
1030
|
command.policyVerdict = AiRemediationCommandPolicyVerdict.AutoApproved;
|
|
728
|
-
|
|
729
|
-
|
|
730
|
-
|
|
731
|
-
|
|
732
|
-
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
1031
|
+
if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
1032
|
+
/*
|
|
1033
|
+
* The resource lane's own hourly brake, counted the way the enqueue
|
|
1034
|
+
* chokepoint counts it (ResourceCommand rows only): pre-checked here
|
|
1035
|
+
* so the model gets a useful refusal instead of a thrown one.
|
|
1036
|
+
*/
|
|
1037
|
+
const resourceJobsInLastHour = (await RunnerJobService.countBy({
|
|
1038
|
+
query: {
|
|
1039
|
+
projectId: this.options.projectId,
|
|
1040
|
+
origin: RunnerJobOrigin.AiRemediation,
|
|
1041
|
+
stepType: RunbookStepType.ResourceCommand,
|
|
1042
|
+
createdAt: QueryHelper.greaterThan(OneUptimeDate.getSomeHoursAgo(1)),
|
|
1043
|
+
},
|
|
1044
|
+
props: { isRoot: true },
|
|
1045
|
+
})).toNumber();
|
|
1046
|
+
if (resourceJobsInLastHour >=
|
|
1047
|
+
MAX_AI_RESOURCE_COMMAND_JOBS_PER_PROJECT_PER_HOUR) {
|
|
1048
|
+
return this.failure(`This project has hit its hourly limit on AI commands on its infrastructure (${MAX_AI_RESOURCE_COMMAND_JOBS_PER_PROJECT_PER_HOUR}). No further commands can run this hour. Summarize and hand off to a human.`);
|
|
1049
|
+
}
|
|
1050
|
+
}
|
|
1051
|
+
else {
|
|
1052
|
+
/*
|
|
1053
|
+
* Project-wide hourly storm brake across the kubectl and Runner AI
|
|
1054
|
+
* command jobs, counted the way the enqueue chokepoints count it:
|
|
1055
|
+
* resource commands have their own brake above and never count here.
|
|
1056
|
+
*/
|
|
1057
|
+
const jobsInLastHour = (await RunnerJobService.countBy({
|
|
1058
|
+
query: {
|
|
1059
|
+
projectId: this.options.projectId,
|
|
1060
|
+
origin: RunnerJobOrigin.AiRemediation,
|
|
1061
|
+
stepType: QueryHelper.notEquals(RunbookStepType.ResourceCommand),
|
|
1062
|
+
createdAt: QueryHelper.greaterThan(OneUptimeDate.getSomeHoursAgo(1)),
|
|
1063
|
+
},
|
|
1064
|
+
props: { isRoot: true },
|
|
1065
|
+
})).toNumber();
|
|
1066
|
+
if (jobsInLastHour >= MAX_AI_COMMAND_JOBS_PER_PROJECT_PER_HOUR) {
|
|
1067
|
+
return this.failure(`This project has hit its hourly AI-command limit (${MAX_AI_COMMAND_JOBS_PER_PROJECT_PER_HOUR}). No further commands can run this hour. Summarize and hand off to a human.`);
|
|
1068
|
+
}
|
|
739
1069
|
}
|
|
740
1070
|
/*
|
|
741
1071
|
* The first change this run makes on a cluster takes one of the
|
|
@@ -753,6 +1083,16 @@ export default class RemediationCommandToolkit {
|
|
|
753
1083
|
}
|
|
754
1084
|
clusterSlotLock = reservation.mutex;
|
|
755
1085
|
}
|
|
1086
|
+
// The same for the run's first change on a resource.
|
|
1087
|
+
if (command.stepType === RunbookStepType.ResourceCommand &&
|
|
1088
|
+
!this.hasChangedResource(command.resourceType, command.resourceId)) {
|
|
1089
|
+
const reservation = await this.reserveResourceSlot(command);
|
|
1090
|
+
if (reservation.refusal) {
|
|
1091
|
+
this.recordNeedingApproval(command, reservation.refusal);
|
|
1092
|
+
return this.failure(reservation.refusal.text);
|
|
1093
|
+
}
|
|
1094
|
+
clusterSlotLock = reservation.mutex;
|
|
1095
|
+
}
|
|
756
1096
|
const releaseClusterSlotLock = async () => {
|
|
757
1097
|
if (!clusterSlotLock) {
|
|
758
1098
|
return;
|
|
@@ -773,7 +1113,7 @@ export default class RemediationCommandToolkit {
|
|
|
773
1113
|
* the job exists (and its id is on the record) — before the wait.
|
|
774
1114
|
*/
|
|
775
1115
|
async recordAndRun(command, afterEnqueue) {
|
|
776
|
-
var _a, _b, _c, _d, _e;
|
|
1116
|
+
var _a, _b, _c, _d, _e, _f, _g, _h;
|
|
777
1117
|
command.wasAutoExecuted = true;
|
|
778
1118
|
command.execution = {
|
|
779
1119
|
status: AiRemediationCommandExecutionStatus.Pending,
|
|
@@ -808,6 +1148,8 @@ export default class RemediationCommandToolkit {
|
|
|
808
1148
|
* whose wait broke): it may have run. Its citation says so.
|
|
809
1149
|
*/
|
|
810
1150
|
let kubectlResultUnknown = false;
|
|
1151
|
+
// The same for a resource command the agent took with no result back.
|
|
1152
|
+
let resourceResultUnknown = false;
|
|
811
1153
|
try {
|
|
812
1154
|
if (command.stepType === RunbookStepType.Kubectl) {
|
|
813
1155
|
/*
|
|
@@ -910,6 +1252,93 @@ export default class RemediationCommandToolkit {
|
|
|
910
1252
|
outcomeText = KubectlJobRunner.describeForLlm(outcome);
|
|
911
1253
|
}
|
|
912
1254
|
}
|
|
1255
|
+
else if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
1256
|
+
/*
|
|
1257
|
+
* The kubectl lane's shape for a resource: enqueue through the
|
|
1258
|
+
* resource chokepoint (which re-runs the policy, the binding, the
|
|
1259
|
+
* switch and the agent's write scope), name the job on the record
|
|
1260
|
+
* BEFORE the wait, release the breaker slot, wait, and read the
|
|
1261
|
+
* finished job exactly as the investigation lane reads it
|
|
1262
|
+
* (ResourceCommandJobRunner: run state, redaction, access failure).
|
|
1263
|
+
*/
|
|
1264
|
+
const resourceType = command.resourceType;
|
|
1265
|
+
const resourceId = new ObjectID(command.resourceId);
|
|
1266
|
+
const job = await RunnerJobService.enqueueAiResourceCommand({
|
|
1267
|
+
projectId: this.options.projectId,
|
|
1268
|
+
aiRunId: this.options.aiRunId,
|
|
1269
|
+
origin: RunnerJobOrigin.AiRemediation,
|
|
1270
|
+
autoRemediationSuggestionId: this.options.suggestionId,
|
|
1271
|
+
resourceType,
|
|
1272
|
+
resourceId,
|
|
1273
|
+
stepId: `${INLINE_COMMAND_STEP_ID_PREFIX}${command.sequence}`,
|
|
1274
|
+
targetResourceAiAgentId: new ObjectID(command.runnerId),
|
|
1275
|
+
command: command.command,
|
|
1276
|
+
timeoutInMs: command.timeoutInMs,
|
|
1277
|
+
claimTimeoutInMs: RESOURCE_COMMAND_CLAIM_TIMEOUT_MS,
|
|
1278
|
+
});
|
|
1279
|
+
command.execution.runnerJobId = (_d = job.id) === null || _d === void 0 ? void 0 : _d.toString();
|
|
1280
|
+
await this.persistPlanProgress();
|
|
1281
|
+
// The job row is the breaker reservation: the next holder counts it.
|
|
1282
|
+
await afterEnqueue();
|
|
1283
|
+
const terminalJob = await this.waitForJobWithHeartbeat({
|
|
1284
|
+
jobId: job.id,
|
|
1285
|
+
claimTimeoutInMs: RESOURCE_COMMAND_CLAIM_TIMEOUT_MS,
|
|
1286
|
+
executionTimeoutInMs: command.timeoutInMs,
|
|
1287
|
+
});
|
|
1288
|
+
const outcome = await ResourceCommandJobRunner.readFinishedJob({
|
|
1289
|
+
job,
|
|
1290
|
+
terminalJob,
|
|
1291
|
+
command: command.command,
|
|
1292
|
+
resourceType,
|
|
1293
|
+
claimTimeoutInMs: RESOURCE_COMMAND_CLAIM_TIMEOUT_MS,
|
|
1294
|
+
executionTimeoutInMs: command.timeoutInMs,
|
|
1295
|
+
});
|
|
1296
|
+
await ResourceCommandJobRunner.recordOutcomeOnResource({
|
|
1297
|
+
resourceType,
|
|
1298
|
+
resourceId,
|
|
1299
|
+
outcome,
|
|
1300
|
+
});
|
|
1301
|
+
// Certainly never ran: a failed tool call, off the record.
|
|
1302
|
+
if (outcome.runState === ResourceCommandRunState.NotRun) {
|
|
1303
|
+
return await this.settleNeverRan(command, {
|
|
1304
|
+
displayCommand: outcome.displayCommand,
|
|
1305
|
+
errorMessage: outcome.errorMessage,
|
|
1306
|
+
claimTimedOut: outcome.claimTimedOut === true,
|
|
1307
|
+
});
|
|
1308
|
+
}
|
|
1309
|
+
command.execution.status = outcome.succeeded
|
|
1310
|
+
? AiRemediationCommandExecutionStatus.Succeeded
|
|
1311
|
+
: AiRemediationCommandExecutionStatus.Failed;
|
|
1312
|
+
command.execution.completedAt =
|
|
1313
|
+
OneUptimeDate.getCurrentDate().toISOString();
|
|
1314
|
+
command.execution.exitCode = outcome.exitCode;
|
|
1315
|
+
command.execution.output = outcome.output;
|
|
1316
|
+
redactionCount = (_e = outcome.redactionCount) !== null && _e !== void 0 ? _e : 0;
|
|
1317
|
+
isTruncated = (_f = outcome.isTruncated) !== null && _f !== void 0 ? _f : false;
|
|
1318
|
+
if (!outcome.succeeded) {
|
|
1319
|
+
command.execution.errorMessage = outcome.errorMessage;
|
|
1320
|
+
}
|
|
1321
|
+
if (outcome.runState === ResourceCommandRunState.Unknown) {
|
|
1322
|
+
resourceResultUnknown = true;
|
|
1323
|
+
outcomeText = this.describeResultUnknown(command, {
|
|
1324
|
+
displayCommand: outcome.displayCommand,
|
|
1325
|
+
reason: outcome.errorMessage,
|
|
1326
|
+
});
|
|
1327
|
+
}
|
|
1328
|
+
else if (!outcome.succeeded && typeof outcome.exitCode !== "number") {
|
|
1329
|
+
// The program ran but was stopped before it finished.
|
|
1330
|
+
outcomeText = `${ResourceCommandJobRunner.describeForLlm({
|
|
1331
|
+
outcome,
|
|
1332
|
+
resourceType,
|
|
1333
|
+
})}\n${RESOURCE_UNFINISHED_CHECK_FIRST}`;
|
|
1334
|
+
}
|
|
1335
|
+
else {
|
|
1336
|
+
outcomeText = ResourceCommandJobRunner.describeForLlm({
|
|
1337
|
+
outcome,
|
|
1338
|
+
resourceType,
|
|
1339
|
+
});
|
|
1340
|
+
}
|
|
1341
|
+
}
|
|
913
1342
|
else {
|
|
914
1343
|
const job = await RunnerJobService.enqueueAiCommand({
|
|
915
1344
|
projectId: this.options.projectId,
|
|
@@ -924,7 +1353,7 @@ export default class RemediationCommandToolkit {
|
|
|
924
1353
|
claimTimeoutInMs: AI_COMMAND_CLAIM_TIMEOUT_MS,
|
|
925
1354
|
});
|
|
926
1355
|
// Same rule as the kubectl lane: the job id lands before the wait.
|
|
927
|
-
command.execution.runnerJobId = (
|
|
1356
|
+
command.execution.runnerJobId = (_g = job.id) === null || _g === void 0 ? void 0 : _g.toString();
|
|
928
1357
|
await this.persistPlanProgress();
|
|
929
1358
|
const terminalJob = await this.waitForJobWithHeartbeat({
|
|
930
1359
|
jobId: job.id,
|
|
@@ -948,7 +1377,7 @@ export default class RemediationCommandToolkit {
|
|
|
948
1377
|
`Command ended with status ${terminalJob.status}.`;
|
|
949
1378
|
}
|
|
950
1379
|
outcomeText = [
|
|
951
|
-
`Command ${succeeded ? "SUCCEEDED" : "FAILED"} (exit code: ${(
|
|
1380
|
+
`Command ${succeeded ? "SUCCEEDED" : "FAILED"} (exit code: ${(_h = terminalJob.exitCode) !== null && _h !== void 0 ? _h : "n/a"}${succeeded ? "" : `, error: ${command.execution.errorMessage}`}).`,
|
|
952
1381
|
`<tool_result source="untrusted_command_output">`,
|
|
953
1382
|
command.execution.output || "(no output)",
|
|
954
1383
|
`</tool_result>`,
|
|
@@ -972,6 +1401,19 @@ export default class RemediationCommandToolkit {
|
|
|
972
1401
|
claimTimedOut: false,
|
|
973
1402
|
});
|
|
974
1403
|
}
|
|
1404
|
+
/*
|
|
1405
|
+
* The resource chokepoint refused the command (its policy, binding,
|
|
1406
|
+
* switch, brake and write-scope checks run before the row is
|
|
1407
|
+
* written): no job exists, so it certainly never ran.
|
|
1408
|
+
*/
|
|
1409
|
+
if (command.stepType === RunbookStepType.ResourceCommand &&
|
|
1410
|
+
!command.execution.runnerJobId) {
|
|
1411
|
+
return await this.settleNeverRan(command, {
|
|
1412
|
+
displayCommand: command.command,
|
|
1413
|
+
errorMessage: this.redactResourceText(command, message),
|
|
1414
|
+
claimTimedOut: false,
|
|
1415
|
+
});
|
|
1416
|
+
}
|
|
975
1417
|
command.execution.status = AiRemediationCommandExecutionStatus.Failed;
|
|
976
1418
|
command.execution.completedAt =
|
|
977
1419
|
OneUptimeDate.getCurrentDate().toISOString();
|
|
@@ -988,6 +1430,15 @@ export default class RemediationCommandToolkit {
|
|
|
988
1430
|
reason: `Waiting for its result failed: ${message}`,
|
|
989
1431
|
});
|
|
990
1432
|
}
|
|
1433
|
+
else if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
1434
|
+
// The same for a resource's agent.
|
|
1435
|
+
command.execution.errorMessage = this.redactResourceText(command, message);
|
|
1436
|
+
resourceResultUnknown = true;
|
|
1437
|
+
outcomeText = this.describeResultUnknown(command, {
|
|
1438
|
+
displayCommand: command.command,
|
|
1439
|
+
reason: `Waiting for its result failed: ${message}`,
|
|
1440
|
+
});
|
|
1441
|
+
}
|
|
991
1442
|
else {
|
|
992
1443
|
outcomeText = `Command FAILED before completion: ${message}`;
|
|
993
1444
|
}
|
|
@@ -1003,12 +1454,34 @@ export default class RemediationCommandToolkit {
|
|
|
1003
1454
|
? kubectlResultUnknown
|
|
1004
1455
|
? `Sent to cluster "${command.kubernetesClusterNameSnapshot}", result unknown: ${this.summarizeCommand(command.command)}`
|
|
1005
1456
|
: `Executed on cluster "${command.kubernetesClusterNameSnapshot}": ${this.summarizeCommand(command.command)}`
|
|
1006
|
-
:
|
|
1457
|
+
: command.stepType === RunbookStepType.ResourceCommand
|
|
1458
|
+
? resourceResultUnknown
|
|
1459
|
+
? `Sent to ${this.describeCommandResource(command)}, result unknown: ${this.summarizeCommand(command.command)}`
|
|
1460
|
+
: `Executed on ${this.describeCommandResource(command)}: ${this.summarizeCommand(command.command)}`
|
|
1461
|
+
: `Executed on Runner "${command.runnerNameSnapshot}": ${this.summarizeCommand(command.command)}`,
|
|
1007
1462
|
redactionCount,
|
|
1008
1463
|
isTruncated,
|
|
1009
1464
|
},
|
|
1010
1465
|
};
|
|
1011
1466
|
}
|
|
1467
|
+
// 'Docker host "web-1"' for a ResourceCommand step, from its snapshot.
|
|
1468
|
+
describeCommandResource(command) {
|
|
1469
|
+
const info = isAiResourceType(command.resourceType)
|
|
1470
|
+
? AI_RESOURCE_TYPE_INFO[command.resourceType]
|
|
1471
|
+
: null;
|
|
1472
|
+
return `${info ? info.displayName : "resource"} "${command.resourceNameSnapshot || command.resourceId || "(unknown)"}"`;
|
|
1473
|
+
}
|
|
1474
|
+
// A resource program's text, through the resource redaction.
|
|
1475
|
+
redactResourceText(command, text) {
|
|
1476
|
+
if (!isAiResourceType(command.resourceType)) {
|
|
1477
|
+
return ToolResultSerializer.redact(text).text;
|
|
1478
|
+
}
|
|
1479
|
+
return ResourceCommandJobRunner.redactAndCap({
|
|
1480
|
+
output: text,
|
|
1481
|
+
resourceType: command.resourceType,
|
|
1482
|
+
program: command.command.trim().split(/\s+/)[0] || "",
|
|
1483
|
+
}).text;
|
|
1484
|
+
}
|
|
1012
1485
|
/*
|
|
1013
1486
|
* A kubectl command whose job certainly never reached kubectl (no Runner
|
|
1014
1487
|
* claimed it, the server or the Runner refused it, kubectl could not
|
|
@@ -1022,6 +1495,15 @@ export default class RemediationCommandToolkit {
|
|
|
1022
1495
|
});
|
|
1023
1496
|
this.neverRanCount += 1;
|
|
1024
1497
|
await this.persistPlanProgress();
|
|
1498
|
+
if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
1499
|
+
const label = this.describeCommandResource(command);
|
|
1500
|
+
const noun = isAiResourceType(command.resourceType)
|
|
1501
|
+
? describeResourceNoun(command.resourceType)
|
|
1502
|
+
: "resource";
|
|
1503
|
+
return this.failure(`"${data.displayCommand}" did NOT run on ${label}: ${data.errorMessage || "the job never reached the resource's AI agent."} Nothing changed on the ${noun} and nothing was recorded as executed. ${data.claimTimedOut
|
|
1504
|
+
? `Do NOT send more commands to this ${noun} in this run — its AI agent is not picking them up; say so in your analysis.`
|
|
1505
|
+
: "Do NOT resend the same command; fix what the refusal names, or put the change in your recommendations for a human."}`);
|
|
1506
|
+
}
|
|
1025
1507
|
return this.failure(`"${data.displayCommand}" did NOT run on cluster "${command.kubernetesClusterNameSnapshot || command.kubernetesClusterId}": ${data.errorMessage || "the job never reached kubectl."} Nothing changed on the cluster and nothing was recorded as executed. ${data.claimTimedOut
|
|
1026
1508
|
? "Do NOT send more commands to this cluster in this run — its Runner is not picking them up; say so in your analysis."
|
|
1027
1509
|
: "Do NOT resend the same command; fix what the refusal names, or put the change in your recommendations for a human."}`);
|
|
@@ -1033,6 +1515,22 @@ export default class RemediationCommandToolkit {
|
|
|
1033
1515
|
* model checks with a read before it reissues it or builds on it.
|
|
1034
1516
|
*/
|
|
1035
1517
|
describeResultUnknown(command, data) {
|
|
1518
|
+
if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
1519
|
+
const resourceReason = this.redactResourceText(command, (data.reason || "No result came back for this command.").trim()).trim();
|
|
1520
|
+
const noun = isAiResourceType(command.resourceType)
|
|
1521
|
+
? describeResourceNoun(command.resourceType)
|
|
1522
|
+
: "resource";
|
|
1523
|
+
return [
|
|
1524
|
+
`${data.displayCommand}`,
|
|
1525
|
+
`RESULT UNKNOWN on ${this.describeCommandResource(command)}: ${SENTENCE_END_PATTERN.test(resourceReason)
|
|
1526
|
+
? resourceReason
|
|
1527
|
+
: `${resourceReason}.`}`,
|
|
1528
|
+
`The command reached the ${noun}'s AI agent, so it MAY have run and changed the ${noun}. It stays on this round's record as a command that may have run, and verification judges it${command.rollbackCommand
|
|
1529
|
+
? "; if the service does not recover, its rollbackCommand is not run blind — a human is asked to check and undo it"
|
|
1530
|
+
: ""}.`,
|
|
1531
|
+
`Before you reissue this command, or run anything that depends on it, check with a read (${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME}) whether it took effect. Do NOT resend it blindly.`,
|
|
1532
|
+
].join("\n");
|
|
1533
|
+
}
|
|
1036
1534
|
const reason = KubectlOutputRedactor.redact((data.reason || "No result came back for this command.").trim()).text;
|
|
1037
1535
|
return [
|
|
1038
1536
|
`${data.displayCommand}`,
|
|
@@ -1071,6 +1569,30 @@ export default class RemediationCommandToolkit {
|
|
|
1071
1569
|
}
|
|
1072
1570
|
return null;
|
|
1073
1571
|
}
|
|
1572
|
+
/*
|
|
1573
|
+
* getCommandScopeRefusal for a resource: whether its AI agent would
|
|
1574
|
+
* refuse this command or its rollback (getResourceWriteScopeRefusal),
|
|
1575
|
+
* worded for the model.
|
|
1576
|
+
*/
|
|
1577
|
+
getResourceCommandScopeRefusal(resource, command) {
|
|
1578
|
+
const forward = RemediationCommandToolkit.getResourceWriteScopeRefusal({
|
|
1579
|
+
resource,
|
|
1580
|
+
command: command.command,
|
|
1581
|
+
});
|
|
1582
|
+
if (forward) {
|
|
1583
|
+
return `${forward} The command was neither run nor recorded.`;
|
|
1584
|
+
}
|
|
1585
|
+
if (command.rollbackCommand) {
|
|
1586
|
+
const rollback = RemediationCommandToolkit.getResourceWriteScopeRefusal({
|
|
1587
|
+
resource,
|
|
1588
|
+
command: command.rollbackCommand,
|
|
1589
|
+
});
|
|
1590
|
+
if (rollback) {
|
|
1591
|
+
return `The rollbackCommand would be refused when it has to run: ${rollback} The command was neither run nor recorded — give a rollback the agent may run, or omit it.`;
|
|
1592
|
+
}
|
|
1593
|
+
}
|
|
1594
|
+
return null;
|
|
1595
|
+
}
|
|
1074
1596
|
/*
|
|
1075
1597
|
* Null when the command may auto-execute in FullAuto; otherwise why not.
|
|
1076
1598
|
* Bash/SSH: denylist, structural guard and the rule allowlist, for the
|
|
@@ -1134,6 +1656,9 @@ export default class RemediationCommandToolkit {
|
|
|
1134
1656
|
}
|
|
1135
1657
|
return null;
|
|
1136
1658
|
}
|
|
1659
|
+
if (command.stepType === RunbookStepType.ResourceCommand) {
|
|
1660
|
+
return this.getResourceFullAutoRefusal(command);
|
|
1661
|
+
}
|
|
1137
1662
|
const policy = CommandPolicy.evaluateCommand({
|
|
1138
1663
|
command: command.command,
|
|
1139
1664
|
allowlistPatterns: this.options.allowlistPatterns,
|
|
@@ -1213,6 +1738,92 @@ export default class RemediationCommandToolkit {
|
|
|
1213
1738
|
? "It is recorded for a human: if you run no other change in this round, OneUptime AI proposes it for one-click approval when the round ends. Also put it in your final recommendations."
|
|
1214
1739
|
: "Include this action in your final recommendations for a human.";
|
|
1215
1740
|
}
|
|
1741
|
+
/*
|
|
1742
|
+
* The FullAuto gate for a resource command — the kubectl branch of
|
|
1743
|
+
* getFullAutoRefusal, on the resource command policy: the resource's
|
|
1744
|
+
* (live) mode must run unattended, the ladder
|
|
1745
|
+
* (ResourceCommandPolicy.evaluateForAutoExecution with the resource's
|
|
1746
|
+
* allowlist, bypass = BypassApproval) must auto-approve the command, and
|
|
1747
|
+
* its rollback — which runs unattended — must auto-approve too. A refusal
|
|
1748
|
+
* a human's click would lift carries an approvalReason.
|
|
1749
|
+
*/
|
|
1750
|
+
getResourceFullAutoRefusal(command) {
|
|
1751
|
+
const resource = this.findResourceTarget(command.resourceType, command.resourceId);
|
|
1752
|
+
if (!resource) {
|
|
1753
|
+
return {
|
|
1754
|
+
text: "The resource is no longer a valid target. Use list_command_targets.",
|
|
1755
|
+
};
|
|
1756
|
+
}
|
|
1757
|
+
const label = describeResourceLabel(resource);
|
|
1758
|
+
if (!isUnattendedResourceRemediationMode(resource.aiRemediationMode)) {
|
|
1759
|
+
return {
|
|
1760
|
+
text: `${label.charAt(0).toUpperCase()}${label.slice(1)} requires human approval for every change, so nothing can execute inline in this run. ${this.describeWhereRefusedChangesGo()}`,
|
|
1761
|
+
approvalReason: `${label} asks for approval of every change`,
|
|
1762
|
+
};
|
|
1763
|
+
}
|
|
1764
|
+
const bypassApproval = resource.aiRemediationMode === ResourceAiRemediationMode.BypassApproval;
|
|
1765
|
+
const verdict = ResourceCommandPolicy.evaluateForAutoExecution({
|
|
1766
|
+
resourceType: resource.resourceType,
|
|
1767
|
+
command: command.command,
|
|
1768
|
+
allowlistPatterns: resource.aiCommandAllowlist,
|
|
1769
|
+
bypassApproval,
|
|
1770
|
+
});
|
|
1771
|
+
if (verdict.verdict !== AiRemediationCommandPolicyVerdict.AutoApproved) {
|
|
1772
|
+
if (verdict.verdict === AiRemediationCommandPolicyVerdict.Denied) {
|
|
1773
|
+
return {
|
|
1774
|
+
text: `${verdict.reason} The command was NOT executed.`,
|
|
1775
|
+
};
|
|
1776
|
+
}
|
|
1777
|
+
return this.describeResourceNeedsAHuman({
|
|
1778
|
+
resource,
|
|
1779
|
+
verdict,
|
|
1780
|
+
bypassApproval,
|
|
1781
|
+
});
|
|
1782
|
+
}
|
|
1783
|
+
if (command.rollbackCommand) {
|
|
1784
|
+
const rollbackVerdict = ResourceCommandPolicy.evaluateForAutoExecution({
|
|
1785
|
+
resourceType: resource.resourceType,
|
|
1786
|
+
command: command.rollbackCommand,
|
|
1787
|
+
allowlistPatterns: resource.aiCommandAllowlist,
|
|
1788
|
+
bypassApproval,
|
|
1789
|
+
});
|
|
1790
|
+
if (rollbackVerdict.verdict !==
|
|
1791
|
+
AiRemediationCommandPolicyVerdict.AutoApproved) {
|
|
1792
|
+
return {
|
|
1793
|
+
text: `The rollbackCommand does not qualify for automatic execution: ${rollbackVerdict.reason} Nothing was executed. Provide a safe rollback on ONE named object (for example the start that undoes a stop), or omit it.`,
|
|
1794
|
+
};
|
|
1795
|
+
}
|
|
1796
|
+
}
|
|
1797
|
+
return null;
|
|
1798
|
+
}
|
|
1799
|
+
/*
|
|
1800
|
+
* describeNeedsAHuman for a resource: named for what actually holds the
|
|
1801
|
+
* change back — a change the policy says always needs a human (in every
|
|
1802
|
+
* mode, Bypass approval and the allowlist included), or a riskier change
|
|
1803
|
+
* on a resource that runs only safe changes on its own.
|
|
1804
|
+
*/
|
|
1805
|
+
describeResourceNeedsAHuman(data) {
|
|
1806
|
+
const { resource, verdict } = data;
|
|
1807
|
+
const label = describeResourceLabel(resource);
|
|
1808
|
+
const whereItGoes = this.describeWhereRefusedChangesGo();
|
|
1809
|
+
if (verdict.requiresHuman === true) {
|
|
1810
|
+
return {
|
|
1811
|
+
text: `${verdict.reason} The command was NOT executed: this change always needs a human, in every mode — Bypass approval and the ${describeResourceNoun(resource.resourceType)}'s allowlist included. ${whereItGoes} Do NOT hunt for a worse substitute that would need a human just the same.`,
|
|
1812
|
+
approvalReason: `the command policy of ${label} says this change always needs a human, in every mode`,
|
|
1813
|
+
};
|
|
1814
|
+
}
|
|
1815
|
+
if (verdict.tier === ResourceCommandTier.RiskyWrite &&
|
|
1816
|
+
!data.bypassApproval) {
|
|
1817
|
+
return {
|
|
1818
|
+
text: `${verdict.reason} The command was NOT executed. ${whereItGoes} Do NOT hunt for a worse safe substitute; use a safe change (${RESOURCE_SAFE_CHANGES_SUMMARY}) only when it is genuinely the right fix.`,
|
|
1819
|
+
approvalReason: `it is a riskier change (${verdict.tier}), and ${label} runs only safe changes on its own unless its command allowlist names the exact command`,
|
|
1820
|
+
};
|
|
1821
|
+
}
|
|
1822
|
+
return {
|
|
1823
|
+
text: `${verdict.reason} The command was NOT executed. ${whereItGoes}`,
|
|
1824
|
+
approvalReason: `${label} does not allow it without a human: ${verdict.reason}`,
|
|
1825
|
+
};
|
|
1826
|
+
}
|
|
1216
1827
|
/*
|
|
1217
1828
|
* Keep a kubectl change refused only for want of a human's click, so a
|
|
1218
1829
|
* cluster round that executes nothing can propose it when it settles.
|
|
@@ -1221,12 +1832,16 @@ export default class RemediationCommandToolkit {
|
|
|
1221
1832
|
* once; at most a plan's worth is kept.
|
|
1222
1833
|
*/
|
|
1223
1834
|
recordNeedingApproval(command, refusal) {
|
|
1835
|
+
// Kubectl changes on a cluster, and resource commands on a resource.
|
|
1224
1836
|
if (!refusal.approvalReason ||
|
|
1225
|
-
command.stepType !== RunbookStepType.Kubectl
|
|
1837
|
+
(command.stepType !== RunbookStepType.Kubectl &&
|
|
1838
|
+
command.stepType !== RunbookStepType.ResourceCommand)) {
|
|
1226
1839
|
return;
|
|
1227
1840
|
}
|
|
1228
1841
|
const alreadyKept = this.commandsNeedingApproval.some((kept) => {
|
|
1229
1842
|
return (kept.command.kubernetesClusterId === command.kubernetesClusterId &&
|
|
1843
|
+
getResourceKey(kept.command.resourceType, kept.command.resourceId) ===
|
|
1844
|
+
getResourceKey(command.resourceType, command.resourceId) &&
|
|
1230
1845
|
kept.command.command === command.command);
|
|
1231
1846
|
});
|
|
1232
1847
|
if (alreadyKept ||
|
|
@@ -1426,6 +2041,216 @@ export default class RemediationCommandToolkit {
|
|
|
1426
2041
|
return { mutex: null, refusal: couldNotCheck };
|
|
1427
2042
|
}
|
|
1428
2043
|
}
|
|
2044
|
+
/*
|
|
2045
|
+
* refreshClusterTarget for a resource: re-read the resource from its AI
|
|
2046
|
+
* page before every inline change. Refuses — and stops the run from
|
|
2047
|
+
* changing that resource again — when fixes were turned off or lost
|
|
2048
|
+
* readiness (project switches, the agent going offline or read-only
|
|
2049
|
+
* included: the status folds them in), when the resource is now reached
|
|
2050
|
+
* through another AI agent than the one this run's command names (the
|
|
2051
|
+
* agent was reset or replaced), or when the status cannot be read (fail
|
|
2052
|
+
* closed). Otherwise the snapshot is replaced with the live status, so the
|
|
2053
|
+
* verdict that follows uses the CURRENT mode and allowlist.
|
|
2054
|
+
*/
|
|
2055
|
+
async refreshResourceTarget(command) {
|
|
2056
|
+
const key = getResourceKey(command.resourceType, command.resourceId);
|
|
2057
|
+
const label = describeResourceLabel({
|
|
2058
|
+
resourceType: command.resourceType,
|
|
2059
|
+
resourceName: command.resourceNameSnapshot || command.resourceId,
|
|
2060
|
+
});
|
|
2061
|
+
const noun = isAiResourceType(command.resourceType)
|
|
2062
|
+
? describeResourceNoun(command.resourceType)
|
|
2063
|
+
: "resource";
|
|
2064
|
+
const stopText = `The command was NOT executed. Do NOT run any further command on this ${noun} in this run; summarize what happened and put the fix in your final recommendations.`;
|
|
2065
|
+
if (!isAiResourceType(command.resourceType) ||
|
|
2066
|
+
!command.resourceId ||
|
|
2067
|
+
!ObjectID.isValidUUID(command.resourceId)) {
|
|
2068
|
+
return {
|
|
2069
|
+
text: `The command names no valid resource. ${stopText}`,
|
|
2070
|
+
};
|
|
2071
|
+
}
|
|
2072
|
+
let status;
|
|
2073
|
+
try {
|
|
2074
|
+
status = await ResourceAiAccessService.getStatusForResource({
|
|
2075
|
+
projectId: this.options.projectId,
|
|
2076
|
+
resourceType: command.resourceType,
|
|
2077
|
+
resourceId: new ObjectID(command.resourceId),
|
|
2078
|
+
});
|
|
2079
|
+
}
|
|
2080
|
+
catch (error) {
|
|
2081
|
+
logger.error(`RemediationCommandToolkit: could not re-read the AI access of ${command.resourceType} ${command.resourceId} before an inline change; refusing it: ${error}`);
|
|
2082
|
+
return {
|
|
2083
|
+
text: `Could not confirm that ${label} still allows AI remediation. ${stopText}`,
|
|
2084
|
+
};
|
|
2085
|
+
}
|
|
2086
|
+
if (!status) {
|
|
2087
|
+
this.revokeResource(key);
|
|
2088
|
+
return {
|
|
2089
|
+
text: `${label.charAt(0).toUpperCase()}${label.slice(1)} no longer exists in this project. ${stopText}`,
|
|
2090
|
+
};
|
|
2091
|
+
}
|
|
2092
|
+
const liveLabel = describeResourceLabel(status);
|
|
2093
|
+
if (!status.isRemediationReady) {
|
|
2094
|
+
this.revokeResource(key);
|
|
2095
|
+
const gap = status.gaps.find((candidate) => {
|
|
2096
|
+
return candidate.blocksRemediation;
|
|
2097
|
+
});
|
|
2098
|
+
return {
|
|
2099
|
+
text: `${liveLabel.charAt(0).toUpperCase()}${liveLabel.slice(1)} no longer allows AI remediation${gap ? ` (${gap.title})` : ""} — its AI agent page changed during this run. ${stopText}`,
|
|
2100
|
+
};
|
|
2101
|
+
}
|
|
2102
|
+
if (!status.agent || status.agent.agentId !== command.runnerId) {
|
|
2103
|
+
this.revokeResource(key);
|
|
2104
|
+
return {
|
|
2105
|
+
text: `${liveLabel.charAt(0).toUpperCase()}${liveLabel.slice(1)} is no longer reached through the AI agent this run started with (its agent was reset or replaced). ${stopText}`,
|
|
2106
|
+
};
|
|
2107
|
+
}
|
|
2108
|
+
const snapshot = this.findResourceTarget(command.resourceType, command.resourceId);
|
|
2109
|
+
this.replaceResourceTarget(status);
|
|
2110
|
+
const round = this.options.resourceRoundNumber || 1;
|
|
2111
|
+
if (snapshot &&
|
|
2112
|
+
doesResourceModeRunRoundUnattended(snapshot.aiRemediationMode, round) &&
|
|
2113
|
+
!doesResourceModeRunRoundUnattended(status.aiRemediationMode, round)) {
|
|
2114
|
+
/*
|
|
2115
|
+
* On a follow-up round, Automatic is still an unattended mode — but
|
|
2116
|
+
* one that asks for every round after the first, so it stops this
|
|
2117
|
+
* round's unattended changes like a move to "ask for approval" does.
|
|
2118
|
+
*/
|
|
2119
|
+
if (isUnattendedResourceRemediationMode(status.aiRemediationMode) &&
|
|
2120
|
+
round > 1) {
|
|
2121
|
+
return {
|
|
2122
|
+
text: `The AI remediation mode of ${liveLabel} was changed to Automatic during this run, and Automatic asks for approval of every change after a signal's first round (this is round ${round}), so no change runs on it unattended any more. The command was NOT executed. ${this.describeWhereRefusedChangesGo()} Do NOT try other changes on this ${noun}.`,
|
|
2123
|
+
approvalReason: `the AI remediation mode of ${liveLabel} was changed to Automatic during the round, which asks for approval after a signal's first round`,
|
|
2124
|
+
};
|
|
2125
|
+
}
|
|
2126
|
+
return {
|
|
2127
|
+
text: `The AI remediation mode of ${liveLabel} was changed to ask for approval during this run, so no change runs on it unattended any more. The command was NOT executed. ${this.describeWhereRefusedChangesGo()} Do NOT try other changes on this ${noun}.`,
|
|
2128
|
+
approvalReason: `the AI remediation mode of ${liveLabel} was changed to ask for approval during the round`,
|
|
2129
|
+
};
|
|
2130
|
+
}
|
|
2131
|
+
return null;
|
|
2132
|
+
}
|
|
2133
|
+
/*
|
|
2134
|
+
* revokeCluster for a resource: the rest of the run treats it as never
|
|
2135
|
+
* having been a target, and what was kept for its proposal is dropped.
|
|
2136
|
+
*/
|
|
2137
|
+
revokeResource(key) {
|
|
2138
|
+
this.revokedResourceKeys.add(key);
|
|
2139
|
+
this.commandsNeedingApproval = this.commandsNeedingApproval.filter((kept) => {
|
|
2140
|
+
return (getResourceKey(kept.command.resourceType, kept.command.resourceId) !==
|
|
2141
|
+
key);
|
|
2142
|
+
});
|
|
2143
|
+
}
|
|
2144
|
+
replaceResourceTarget(status) {
|
|
2145
|
+
const key = getResourceKey(status.resourceType, status.resourceId);
|
|
2146
|
+
this.options.resourceTargets = (this.options.resourceTargets || []).map((resource) => {
|
|
2147
|
+
return getResourceKey(resource.resourceType, resource.resourceId) ===
|
|
2148
|
+
key
|
|
2149
|
+
? status
|
|
2150
|
+
: resource;
|
|
2151
|
+
});
|
|
2152
|
+
}
|
|
2153
|
+
// Does this run already hold a slot on the resource (an inline job there)?
|
|
2154
|
+
hasChangedResource(resourceType, resourceId) {
|
|
2155
|
+
const key = getResourceKey(resourceType, resourceId);
|
|
2156
|
+
return this.executedCommands.some((executed) => {
|
|
2157
|
+
var _a;
|
|
2158
|
+
return (executed.stepType === RunbookStepType.ResourceCommand &&
|
|
2159
|
+
getResourceKey(executed.resourceType, executed.resourceId) === key &&
|
|
2160
|
+
Boolean((_a = executed.execution) === null || _a === void 0 ? void 0 : _a.runnerJobId));
|
|
2161
|
+
});
|
|
2162
|
+
}
|
|
2163
|
+
/*
|
|
2164
|
+
* reserveClusterSlot for a resource: one of the resource's hourly
|
|
2165
|
+
* unattended slots, taken under the per-resource breaker lock
|
|
2166
|
+
* (RESOURCE_BREAKER_LOCK_NAMESPACE, keyed by type and id, never a
|
|
2167
|
+
* cluster's key), which is returned HELD until this run's job row exists.
|
|
2168
|
+
* Under the same lock, no other AI run may hold the resource
|
|
2169
|
+
* (resourceHold). Fails closed.
|
|
2170
|
+
*/
|
|
2171
|
+
async reserveResourceSlot(command) {
|
|
2172
|
+
const resourceType = command.resourceType;
|
|
2173
|
+
const resourceId = command.resourceId || "";
|
|
2174
|
+
const label = describeResourceLabel({
|
|
2175
|
+
resourceType,
|
|
2176
|
+
resourceName: command.resourceNameSnapshot || resourceId,
|
|
2177
|
+
});
|
|
2178
|
+
const noun = isAiResourceType(resourceType)
|
|
2179
|
+
? describeResourceNoun(resourceType)
|
|
2180
|
+
: "resource";
|
|
2181
|
+
const couldNotCheck = {
|
|
2182
|
+
text: `Could not check the hourly limit on unattended AI fixes for ${label}, so the command was NOT executed. ${this.describeWhereRefusedChangesGo()}`,
|
|
2183
|
+
approvalReason: `the hourly limit on unattended AI fixes for ${label} could not be checked`,
|
|
2184
|
+
};
|
|
2185
|
+
if (!isAiResourceType(resourceType) || !resourceId) {
|
|
2186
|
+
return { mutex: null, refusal: couldNotCheck };
|
|
2187
|
+
}
|
|
2188
|
+
let mutex = null;
|
|
2189
|
+
try {
|
|
2190
|
+
mutex = await Semaphore.lock({
|
|
2191
|
+
key: getResourceBreakerLockKey(resourceType, resourceId),
|
|
2192
|
+
namespace: RESOURCE_BREAKER_LOCK_NAMESPACE,
|
|
2193
|
+
lockTimeout: CLUSTER_BREAKER_LOCK_TIMEOUT_MS,
|
|
2194
|
+
acquireTimeout: CLUSTER_BREAKER_LOCK_ACQUIRE_TIMEOUT_MS,
|
|
2195
|
+
});
|
|
2196
|
+
}
|
|
2197
|
+
catch (error) {
|
|
2198
|
+
logger.error(`RemediationCommandToolkit: could not take the circuit-breaker lock of ${resourceType} ${resourceId}; refusing the inline change: ${error}`);
|
|
2199
|
+
return { mutex: null, refusal: couldNotCheck };
|
|
2200
|
+
}
|
|
2201
|
+
try {
|
|
2202
|
+
const breaker = await AutoRemediationRuleEngineService.getResourceBreakerState({
|
|
2203
|
+
resourceType,
|
|
2204
|
+
resourceId,
|
|
2205
|
+
projectId: this.options.projectId,
|
|
2206
|
+
forRound: {
|
|
2207
|
+
suggestionId: this.options.suggestionId,
|
|
2208
|
+
createdAt: this.options.suggestionCreatedAt,
|
|
2209
|
+
},
|
|
2210
|
+
});
|
|
2211
|
+
if (!breaker.hasHeadroom) {
|
|
2212
|
+
await RemediationCommandToolkit.releaseLock(mutex);
|
|
2213
|
+
logger.warn(`RemediationCommandToolkit: ${resourceType} ${resourceId} hit its hourly circuit breaker (${breaker.autoExecutedInWindow} unattended AI fixes); refusing an inline change.`);
|
|
2214
|
+
return {
|
|
2215
|
+
mutex: null,
|
|
2216
|
+
refusal: {
|
|
2217
|
+
text: `The hourly circuit breaker for ${label} tripped: it already had ${breaker.autoExecutedInWindow} unattended AI fix(es) in the last hour (the limit is ${MAX_AUTO_EXECUTIONS_PER_RULE_PER_HOUR}). The command was NOT executed. ${this.describeWhereRefusedChangesGo()}`,
|
|
2218
|
+
approvalReason: `the hourly circuit breaker for ${label} tripped (${breaker.autoExecutedInWindow} unattended AI fixes in the last hour)`,
|
|
2219
|
+
},
|
|
2220
|
+
};
|
|
2221
|
+
}
|
|
2222
|
+
if (this.options.resourceHold) {
|
|
2223
|
+
const hold = await AutoRemediationRuleEngineService.findRoundHoldingResource({
|
|
2224
|
+
projectId: this.options.projectId,
|
|
2225
|
+
resourceType,
|
|
2226
|
+
resourceId,
|
|
2227
|
+
forRound: {
|
|
2228
|
+
suggestionId: this.options.suggestionId,
|
|
2229
|
+
createdAt: this.options.suggestionCreatedAt,
|
|
2230
|
+
},
|
|
2231
|
+
anyOrder: this.options.resourceHold.anyOrder,
|
|
2232
|
+
subject: this.options.resourceHold.subject,
|
|
2233
|
+
});
|
|
2234
|
+
if (hold) {
|
|
2235
|
+
await RemediationCommandToolkit.releaseLock(mutex);
|
|
2236
|
+
logger.warn(`RemediationCommandToolkit: another AI run (${hold.suggestionId}) on ${resourceType} ${resourceId} ${hold.description}; refusing an inline change.`);
|
|
2237
|
+
return {
|
|
2238
|
+
mutex: null,
|
|
2239
|
+
refusal: {
|
|
2240
|
+
text: `Another OneUptime AI run on ${label} ${hold.description}, so this run may not change the ${noun} too — two unattended fixes on one ${noun} verify and roll back on top of each other. The command was NOT executed. ${this.describeWhereRefusedChangesGo()} Do NOT try other changes on this ${noun}.`,
|
|
2241
|
+
approvalReason: `another OneUptime AI run on ${label} ${hold.description}`,
|
|
2242
|
+
},
|
|
2243
|
+
};
|
|
2244
|
+
}
|
|
2245
|
+
}
|
|
2246
|
+
return { mutex };
|
|
2247
|
+
}
|
|
2248
|
+
catch (error) {
|
|
2249
|
+
await RemediationCommandToolkit.releaseLock(mutex);
|
|
2250
|
+
logger.error(`RemediationCommandToolkit: circuit-breaker or in-flight run check failed for ${resourceType} ${resourceId}; refusing the inline change: ${error}`);
|
|
2251
|
+
return { mutex: null, refusal: couldNotCheck };
|
|
2252
|
+
}
|
|
2253
|
+
}
|
|
1429
2254
|
static async releaseLock(mutex) {
|
|
1430
2255
|
try {
|
|
1431
2256
|
await Semaphore.release(mutex);
|
|
@@ -1452,7 +2277,9 @@ export default class RemediationCommandToolkit {
|
|
|
1452
2277
|
return {
|
|
1453
2278
|
definition: {
|
|
1454
2279
|
name: "propose_remediation_commands",
|
|
1455
|
-
description:
|
|
2280
|
+
description: this.isResourceRound()
|
|
2281
|
+
? `Propose an ordered plan of at most ${MAX_PLAN_COMMANDS} remediation commands on the infrastructure resource this round is about (stepType ResourceCommand with its resourceId) for one-click human approval. Nothing executes until a human approves the whole plan. Call this at most once with your final plan (a later call replaces the earlier one). One command per step, written as the program followed by its arguments — never a shell line; a read-only command is not a fix (run it with ${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME} instead). Provide a rollbackCommand for every state-changing command that has an undo — it runs unattended, so it must be a safe change on ONE named object. The commands run through the resource's own AI agent, and a change it would refuse (it runs read-only, or the target is protected or outside its writeScope in list_command_targets) is refused; ${RESOURCE_NEVER_RUNS_SUMMARY}.\n\n${this.describeResourceWriteGuides()}`
|
|
2282
|
+
: `Propose an ordered plan of at most ${MAX_PLAN_COMMANDS} remediation commands for one-click human approval. Nothing executes until a human approves the whole plan. Call this at most once with your final plan (a later call replaces the earlier one). Provide a rollbackCommand for every state-changing command that has an undo (for Kubectl, e.g. kubectl rollout undo deployment/<name> -n <namespace>). Kubectl commands run through the cluster's Kubernetes AI agent or Runner, and a write outside its writeScope (list_command_targets) is refused; ${KUBECTL_NEVER_RUNS_SUMMARY}.`,
|
|
1456
2283
|
inputSchema: {
|
|
1457
2284
|
type: "object",
|
|
1458
2285
|
properties: {
|
|
@@ -1462,12 +2289,15 @@ export default class RemediationCommandToolkit {
|
|
|
1462
2289
|
items: {
|
|
1463
2290
|
type: "object",
|
|
1464
2291
|
properties: this.buildCommandSchemaProperties(),
|
|
1465
|
-
required:
|
|
1466
|
-
|
|
1467
|
-
|
|
1468
|
-
|
|
1469
|
-
|
|
1470
|
-
|
|
2292
|
+
required: this.isResourceRound()
|
|
2293
|
+
? [
|
|
2294
|
+
"stepType",
|
|
2295
|
+
"resourceId",
|
|
2296
|
+
"command",
|
|
2297
|
+
"rationale",
|
|
2298
|
+
"expectedEffect",
|
|
2299
|
+
]
|
|
2300
|
+
: ["stepType", "command", "rationale", "expectedEffect"],
|
|
1471
2301
|
},
|
|
1472
2302
|
},
|
|
1473
2303
|
},
|
|
@@ -1515,6 +2345,19 @@ export default class RemediationCommandToolkit {
|
|
|
1515
2345
|
parsed.command.policyVerdict =
|
|
1516
2346
|
AiRemediationCommandPolicyVerdict.RequiresApproval;
|
|
1517
2347
|
}
|
|
2348
|
+
else if (parsed.command.stepType === RunbookStepType.ResourceCommand) {
|
|
2349
|
+
/*
|
|
2350
|
+
* The same for a resource command: a proposal runs only after a
|
|
2351
|
+
* click, so it is RequiresApproval whatever the resource's mode —
|
|
2352
|
+
* and a read is not a fix a human needs to approve.
|
|
2353
|
+
*/
|
|
2354
|
+
if (parsed.command.resourceCommandTier === ResourceCommandTier.Read) {
|
|
2355
|
+
problems.push(`Command ${i + 1}: "${parsed.command.command}" is read-only — it is not a fix. Run it with ${RUN_INFRASTRUCTURE_COMMAND_TOOL_NAME} and propose only the change.`);
|
|
2356
|
+
continue;
|
|
2357
|
+
}
|
|
2358
|
+
parsed.command.policyVerdict =
|
|
2359
|
+
AiRemediationCommandPolicyVerdict.RequiresApproval;
|
|
2360
|
+
}
|
|
1518
2361
|
else {
|
|
1519
2362
|
/*
|
|
1520
2363
|
* Informational verdict for the approval card: AutoApproved
|
|
@@ -1564,9 +2407,31 @@ export default class RemediationCommandToolkit {
|
|
|
1564
2407
|
async parseAndValidateCommand(args, sequence) {
|
|
1565
2408
|
const stepTypeRaw = ToolArgs.getString(args, "stepType") || "";
|
|
1566
2409
|
const stepType = stepTypeRaw;
|
|
1567
|
-
|
|
2410
|
+
/*
|
|
2411
|
+
* The step types this round offers — what its schema's stepType enum
|
|
2412
|
+
* lists: ResourceCommand on a resource round; Bash, SSH and Kubectl on
|
|
2413
|
+
* every other round, where ResourceCommand is as unknown as it always
|
|
2414
|
+
* was (so a Kubernetes or rule round's refusal keeps its words).
|
|
2415
|
+
*/
|
|
2416
|
+
const offeredStepTypes = this.isResourceRound()
|
|
2417
|
+
? [RunbookStepType.ResourceCommand]
|
|
2418
|
+
: AI_COMMAND_STEP_TYPES.filter((type) => {
|
|
2419
|
+
return type !== RunbookStepType.ResourceCommand;
|
|
2420
|
+
});
|
|
2421
|
+
if (!AI_COMMAND_STEP_TYPES.includes(stepType) ||
|
|
2422
|
+
(!this.isResourceRound() && stepType === RunbookStepType.ResourceCommand)) {
|
|
1568
2423
|
return {
|
|
1569
|
-
errorText: `stepType must be one of: ${
|
|
2424
|
+
errorText: `stepType must be one of: ${offeredStepTypes.join(", ")}.`,
|
|
2425
|
+
};
|
|
2426
|
+
}
|
|
2427
|
+
/*
|
|
2428
|
+
* A resource round runs resource commands on its resource and nothing
|
|
2429
|
+
* else: it was given no Runner and no cluster.
|
|
2430
|
+
*/
|
|
2431
|
+
if (this.isResourceRound() &&
|
|
2432
|
+
stepType !== RunbookStepType.ResourceCommand) {
|
|
2433
|
+
return {
|
|
2434
|
+
errorText: `This remediation round is about one infrastructure resource: use stepType ResourceCommand with its resourceId from list_command_targets. ${stepType} is not available here.`,
|
|
1570
2435
|
};
|
|
1571
2436
|
}
|
|
1572
2437
|
const commandText = ToolArgs.getString(args, "command") || "";
|
|
@@ -1597,6 +2462,29 @@ export default class RemediationCommandToolkit {
|
|
|
1597
2462
|
expectedEffect,
|
|
1598
2463
|
});
|
|
1599
2464
|
}
|
|
2465
|
+
/*
|
|
2466
|
+
* A resource command runs on the resource's own AI agent, never on a
|
|
2467
|
+
* Runner, so it must never fall through to the Bash/SSH path below.
|
|
2468
|
+
* Only a resource round (resourceTargets) offers it — any other round
|
|
2469
|
+
* refused it with the step types it offers, above; this is the belt
|
|
2470
|
+
* and braces.
|
|
2471
|
+
*/
|
|
2472
|
+
if (stepType === RunbookStepType.ResourceCommand) {
|
|
2473
|
+
if (!this.isResourceRound()) {
|
|
2474
|
+
return {
|
|
2475
|
+
errorText: `stepType must be one of: ${offeredStepTypes.join(", ")}.`,
|
|
2476
|
+
};
|
|
2477
|
+
}
|
|
2478
|
+
return this.parseResourceCommand({
|
|
2479
|
+
args,
|
|
2480
|
+
sequence,
|
|
2481
|
+
commandText,
|
|
2482
|
+
rollbackCommand,
|
|
2483
|
+
timeoutInMs,
|
|
2484
|
+
rationale,
|
|
2485
|
+
expectedEffect,
|
|
2486
|
+
});
|
|
2487
|
+
}
|
|
1600
2488
|
const denyReason = CommandPolicy.getDenyReason(commandText);
|
|
1601
2489
|
if (denyReason) {
|
|
1602
2490
|
return {
|
|
@@ -1786,6 +2674,117 @@ export default class RemediationCommandToolkit {
|
|
|
1786
2674
|
return cluster.clusterId === clusterId;
|
|
1787
2675
|
});
|
|
1788
2676
|
}
|
|
2677
|
+
/*
|
|
2678
|
+
* ResourceCommand: the target is the round's resource, the AI agent is
|
|
2679
|
+
* whichever one its AI page reports online, and no credential ever
|
|
2680
|
+
* travels. The model only names a resource it was shown (by resourceId);
|
|
2681
|
+
* the tier the resource command policy gives the command is recorded for
|
|
2682
|
+
* the card, and a command, or a rollback, the agent's write scope would
|
|
2683
|
+
* refuse is never composed. Denied never is either.
|
|
2684
|
+
*/
|
|
2685
|
+
parseResourceCommand(data) {
|
|
2686
|
+
const resourceIdRaw = ToolArgs.getString(data.args, "resourceId");
|
|
2687
|
+
const resource = resourceIdRaw
|
|
2688
|
+
? this.getResourceTargets().find((candidate) => {
|
|
2689
|
+
return (candidate.resourceId.toLowerCase() === resourceIdRaw.toLowerCase());
|
|
2690
|
+
})
|
|
2691
|
+
: undefined;
|
|
2692
|
+
if (!resource || !resource.agent) {
|
|
2693
|
+
return {
|
|
2694
|
+
errorText: "resourceId is required for ResourceCommand and must be the resource from list_command_targets that allows AI remediation.",
|
|
2695
|
+
};
|
|
2696
|
+
}
|
|
2697
|
+
/*
|
|
2698
|
+
* A resource's agent is never given a credential by OneUptime: a step
|
|
2699
|
+
* that names one is refused, never silently stripped.
|
|
2700
|
+
*/
|
|
2701
|
+
const credentialIdRaw = ToolArgs.getString(data.args, "credentialId");
|
|
2702
|
+
if (credentialIdRaw) {
|
|
2703
|
+
return {
|
|
2704
|
+
errorText: `A ResourceCommand never carries a credential: the ${AI_RESOURCE_TYPE_INFO[resource.resourceType].agentDisplayName} uses only the credentials in its own environment. Omit credentialId.`,
|
|
2705
|
+
};
|
|
2706
|
+
}
|
|
2707
|
+
const info = AI_RESOURCE_TYPE_INFO[resource.resourceType];
|
|
2708
|
+
const policy = ResourceCommandPolicy.evaluateCommand({
|
|
2709
|
+
resourceType: resource.resourceType,
|
|
2710
|
+
command: data.commandText,
|
|
2711
|
+
});
|
|
2712
|
+
if (policy.tier === ResourceCommandTier.Denied) {
|
|
2713
|
+
return {
|
|
2714
|
+
errorText: `Denied by the ${info.displayName} command policy: ${policy.reason}. This command can never run, even with human approval — take a different approach.`,
|
|
2715
|
+
};
|
|
2716
|
+
}
|
|
2717
|
+
let rollbackDisplay = undefined;
|
|
2718
|
+
if (data.rollbackCommand) {
|
|
2719
|
+
const rollbackPolicy = ResourceCommandPolicy.evaluateCommand({
|
|
2720
|
+
resourceType: resource.resourceType,
|
|
2721
|
+
command: data.rollbackCommand,
|
|
2722
|
+
});
|
|
2723
|
+
if (rollbackPolicy.tier === ResourceCommandTier.Denied) {
|
|
2724
|
+
return {
|
|
2725
|
+
errorText: `The rollbackCommand is denied by the ${info.displayName} command policy: ${rollbackPolicy.reason}. Provide a safe rollback or omit it.`,
|
|
2726
|
+
};
|
|
2727
|
+
}
|
|
2728
|
+
/*
|
|
2729
|
+
* A rollback runs unattended after verification fails: a change that
|
|
2730
|
+
* always needs a human can never be one, and a riskier one only on a
|
|
2731
|
+
* resource whose operator bypassed approvals.
|
|
2732
|
+
*/
|
|
2733
|
+
if (rollbackPolicy.requiresHuman === true) {
|
|
2734
|
+
return {
|
|
2735
|
+
errorText: `The rollbackCommand "${rollbackPolicy.displayCommand}" always needs a human (${rollbackPolicy.reason}), and rollbacks run unattended. Use a safe undo for ONE named object instead, or omit it.`,
|
|
2736
|
+
};
|
|
2737
|
+
}
|
|
2738
|
+
if (rollbackPolicy.tier === ResourceCommandTier.RiskyWrite &&
|
|
2739
|
+
resource.aiRemediationMode !== ResourceAiRemediationMode.BypassApproval) {
|
|
2740
|
+
return {
|
|
2741
|
+
errorText: `The rollbackCommand "${rollbackPolicy.displayCommand}" is a risky change (${rollbackPolicy.reason}) and rollbacks run unattended. Use a safe undo for ONE named object instead (for example the start that undoes a stop), or omit it.`,
|
|
2742
|
+
};
|
|
2743
|
+
}
|
|
2744
|
+
rollbackDisplay = rollbackPolicy.displayCommand;
|
|
2745
|
+
}
|
|
2746
|
+
// A write the agent has said it will refuse is never composed.
|
|
2747
|
+
const scopeRefusal = this.getResourceCommandScopeRefusal(resource, {
|
|
2748
|
+
command: policy.displayCommand,
|
|
2749
|
+
rollbackCommand: rollbackDisplay,
|
|
2750
|
+
});
|
|
2751
|
+
if (scopeRefusal) {
|
|
2752
|
+
return { errorText: scopeRefusal };
|
|
2753
|
+
}
|
|
2754
|
+
return {
|
|
2755
|
+
command: {
|
|
2756
|
+
sequence: data.sequence,
|
|
2757
|
+
stepType: RunbookStepType.ResourceCommand,
|
|
2758
|
+
/*
|
|
2759
|
+
* The access target is the resource's AI agent (its row id and
|
|
2760
|
+
* display name), as a Kubectl step through the Kubernetes AI agent
|
|
2761
|
+
* names that agent.
|
|
2762
|
+
*/
|
|
2763
|
+
runnerId: resource.agent.agentId,
|
|
2764
|
+
runnerNameSnapshot: info.agentDisplayName,
|
|
2765
|
+
resourceType: resource.resourceType,
|
|
2766
|
+
resourceId: resource.resourceId,
|
|
2767
|
+
resourceNameSnapshot: resource.resourceName,
|
|
2768
|
+
resourceCommandTier: policy.tier,
|
|
2769
|
+
// Stored in the canonical rendered form so the card shows exactly what runs.
|
|
2770
|
+
command: policy.displayCommand,
|
|
2771
|
+
timeoutInMs: Math.min(data.timeoutInMs, MAX_RESOURCE_COMMAND_TIMEOUT_MS),
|
|
2772
|
+
rationale: data.rationale,
|
|
2773
|
+
expectedEffect: data.expectedEffect,
|
|
2774
|
+
rollbackCommand: rollbackDisplay,
|
|
2775
|
+
policyVerdict: AiRemediationCommandPolicyVerdict.RequiresApproval,
|
|
2776
|
+
},
|
|
2777
|
+
};
|
|
2778
|
+
}
|
|
2779
|
+
findResourceTarget(resourceType, resourceId) {
|
|
2780
|
+
const key = getResourceKey(resourceType, resourceId);
|
|
2781
|
+
if (!key) {
|
|
2782
|
+
return undefined;
|
|
2783
|
+
}
|
|
2784
|
+
return this.getResourceTargets().find((resource) => {
|
|
2785
|
+
return (getResourceKey(resource.resourceType, resource.resourceId) === key);
|
|
2786
|
+
});
|
|
2787
|
+
}
|
|
1789
2788
|
/*
|
|
1790
2789
|
* Wait for the RunnerJob to reach a terminal state while keeping the
|
|
1791
2790
|
* AIRun's heartbeat fresh — a command may legitimately take minutes, and
|