@rizom/ops 0.2.0-alpha.305 → 0.2.0-alpha.307

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,6 +6,7 @@ import { type RunCommand as OpsRunCommand } from "./run-subprocess";
6
6
  import { type SshKeygen } from "./ssh-key-bootstrap";
7
7
  import type { UserRunner } from "./user-runner";
8
8
  import type { cleanupDirectorySyncStress, runDeployedDirectorySyncStress, verifyDirectorySyncStressAccess } from "./directory-sync-stress-system";
9
+ import type { cleanupHealthWatchdogSmoke, runHealthWatchdogSmoke } from "./health-watchdog-smoke";
9
10
  export interface CommandResult {
10
11
  success: boolean;
11
12
  message?: string;
@@ -24,6 +25,8 @@ export interface CommandDependencies extends LoadPilotRegistryOptions {
24
25
  directorySyncStressRunner?: typeof runDeployedDirectorySyncStress | undefined;
25
26
  directorySyncStressAccessRunner?: typeof verifyDirectorySyncStressAccess | undefined;
26
27
  directorySyncStressCleanupRunner?: typeof cleanupDirectorySyncStress | undefined;
28
+ healthWatchdogSmokeRunner?: typeof runHealthWatchdogSmoke | undefined;
29
+ healthWatchdogSmokeCleanupRunner?: typeof cleanupHealthWatchdogSmoke | undefined;
27
30
  }
28
31
  export declare const globalFlags: FlagDefinitions;
29
32
  export declare const commands: readonly CommandDefinition<CommandDependencies, CommandResult>[];
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@rizom/ops",
3
- "version": "0.2.0-alpha.305",
3
+ "version": "0.2.0-alpha.307",
4
4
  "description": "Operator CLI for managing private brain fleet registry repos",
5
5
  "keywords": [
6
6
  "brains",
@@ -0,0 +1,95 @@
1
+ name: Health Watchdog Smoke
2
+
3
+ on:
4
+ workflow_dispatch:
5
+ inputs:
6
+ handle:
7
+ description: Smoke fleet handle only
8
+ required: true
9
+ type: string
10
+ default: smoke
11
+ confirm:
12
+ description: Type watchdog-smoke:<handle> to confirm temporary container restarts
13
+ required: true
14
+ type: string
15
+
16
+ permissions:
17
+ contents: read
18
+
19
+ concurrency:
20
+ group: deploy-${{ inputs.handle }}
21
+ cancel-in-progress: false
22
+
23
+ env:
24
+ WATCHDOG_SMOKE_RUN_ID: gha-${{ github.run_id }}-${{ github.run_attempt }}
25
+
26
+ jobs:
27
+ smoke:
28
+ runs-on: ubuntu-latest
29
+ timeout-minutes: 20
30
+ steps:
31
+ - uses: actions/checkout@v5
32
+
33
+ - uses: oven-sh/setup-bun@v2
34
+
35
+ - name: Install operator tooling
36
+ run: bun install
37
+
38
+ - name: Load Bitwarden environment through Varlock
39
+ uses: ./.github/actions/varlock-env
40
+ env:
41
+ BWS_ACCESS_TOKEN: ${{ secrets.BWS_ACCESS_TOKEN }}
42
+
43
+ - name: Run fleet watchdog isolation smoke
44
+ env:
45
+ HANDLE_INPUT: ${{ inputs.handle }}
46
+ CONFIRM_INPUT: ${{ inputs.confirm }}
47
+ run: |
48
+ bunx brains-ops smoke:health-watchdog "$GITHUB_WORKSPACE" "$HANDLE_INPUT" \
49
+ --confirm "$CONFIRM_INPUT" \
50
+ --run-id "$WATCHDOG_SMOKE_RUN_ID" \
51
+ --artifacts-dir "$RUNNER_TEMP/health-watchdog-smoke"
52
+
53
+ - name: Upload watchdog smoke evidence
54
+ if: always()
55
+ uses: actions/upload-artifact@v4
56
+ with:
57
+ name: health-watchdog-smoke-${{ inputs.handle }}-${{ github.run_id }}-${{ github.run_attempt }}
58
+ path: ${{ runner.temp }}/health-watchdog-smoke
59
+ if-no-files-found: warn
60
+
61
+ cleanup:
62
+ needs: smoke
63
+ if: always()
64
+ runs-on: ubuntu-latest
65
+ timeout-minutes: 10
66
+ steps:
67
+ - uses: actions/checkout@v5
68
+
69
+ - uses: oven-sh/setup-bun@v2
70
+
71
+ - name: Install operator tooling
72
+ run: bun install
73
+
74
+ - name: Load Bitwarden environment through Varlock
75
+ uses: ./.github/actions/varlock-env
76
+ env:
77
+ BWS_ACCESS_TOKEN: ${{ secrets.BWS_ACCESS_TOKEN }}
78
+
79
+ - name: Remove residual watchdog smoke fixtures
80
+ env:
81
+ HANDLE_INPUT: ${{ inputs.handle }}
82
+ CONFIRM_INPUT: ${{ inputs.confirm }}
83
+ run: |
84
+ bunx brains-ops smoke:health-watchdog:cleanup "$GITHUB_WORKSPACE" "$HANDLE_INPUT" \
85
+ --confirm "$CONFIRM_INPUT" \
86
+ --run-id "$WATCHDOG_SMOKE_RUN_ID" \
87
+ --artifacts-dir "$RUNNER_TEMP/health-watchdog-smoke-cleanup"
88
+
89
+ - name: Upload watchdog smoke cleanup evidence
90
+ if: always()
91
+ uses: actions/upload-artifact@v4
92
+ with:
93
+ name: health-watchdog-smoke-cleanup-${{ inputs.handle }}-${{ github.run_id }}-${{ github.run_attempt }}
94
+ path: ${{ runner.temp }}/health-watchdog-smoke-cleanup
95
+ if-no-files-found: warn
@@ -10,6 +10,7 @@ Treat these as checked-in deploy artifacts in the pilot repo:
10
10
  - `.github/workflows/build.yml`
11
11
  - `.github/workflows/deploy.yml`
12
12
  - `.github/workflows/directory-sync-stress.yml`
13
+ - `.github/workflows/health-watchdog-smoke.yml`
13
14
  - `.github/workflows/reconcile.yml`
14
15
 
15
16
  `.env.schema` is the single source of truth for required and sensitive deploy vars.
@@ -88,6 +89,14 @@ The workflow loads operator credentials through Bitwarden/Varlock, but it is sep
88
89
 
89
90
  Treat any gated health failure, restart, OOM, residual probe, or entity-baseline drift as a failed gate. Do not restart the target during measurement. Recovery is a separate operator action after evidence collection.
90
91
 
92
+ ## Health watchdog smoke gate
93
+
94
+ Use the manual `Health Watchdog Smoke` workflow only after the `smoke` fleet user has completed a normal Deploy run. The workflow shares the `deploy-<handle>` concurrency group, resolves the server from pilot desired state through Hetzner, loads the existing deploy SSH key through Bitwarden/Varlock, and refuses targets whose handle and domain do not identify smoke. Confirm with `watchdog-smoke:<handle>`.
95
+
96
+ The smoke does not deploy or replace the installed systemd units. It verifies that the installed watchdog exactly matches the packaged canonical payload, then exercises that installed script. It also verifies the deployed rover image label, exact-label selector, active timer, and `/health/live` Docker healthcheck. While holding the timer's global lock, it creates temporary containers for eligible unhealthy, unrelated service-labelled unhealthy, false-labelled unhealthy, and eligible healthy cases. It then requires exactly three eligible restarts followed by budget suppression, diagnostics and state only for the eligible fixture, and no restart of the deployed rover or ineligible fixtures.
97
+
98
+ The primary job uploads remote incident, state, and command evidence. An independent `if: always()` cleanup job removes deterministic fixture names and remote temporary files using the same workflow run ID. Treat missing evidence, unexpected eligibility, deployed-container movement, cleanup failure, or any non-smoke target rejection as a failed gate.
99
+
91
100
  ## Stale deploy lock recovery
92
101
 
93
102
  Kamal intentionally leaves its remote deploy lock in place when a deployment is cancelled or interrupted. Confirm that no deployment for the user is still active before releasing the lock, then use the deploy workflow's explicit recovery input: