kampodine 0.3.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,127 @@
1
+ #!/usr/bin/env bash
2
+ # metrics.sh — ONE-SHOT VM snapshot over SSH (no continuous monitoring —
3
+ # Beszel owns that; this is the CLI triage view). One SSH round-trip per
4
+ # snapshot: load + CPUs, memory, disk WITH an images/data/other breakdown,
5
+ # per-container CPU/MEM (podman stats), uptime, top-5 processes by RSS.
6
+ #
7
+ # The remote command is busybox-safe (cat/echo/nproc/df/du/podman/ps/sort/
8
+ # head — see METRICS_REMOTE_CMD in common.sh); all parsing/rendering happens
9
+ # client-side (metrics_* helpers in common.sh — the SAME snapshot folds into
10
+ # `status --verbose`). Disk threshold reuses the deploy gate's verdict logic
11
+ # (disk_verdict): --warn-disk <pct> (default 90) exits 1 when usage >= pct.
12
+ #
13
+ # Usage:
14
+ # kampodine metrics [--profile <name>] [--host <user@ip>] [--ssh-key <path>]
15
+ # [--warn-disk <pct>] [--watch <sec>] [--count <n>]
16
+ set -euo pipefail
17
+
18
+ HERE="$(cd "$(dirname "$0")" && pwd)"
19
+ # shellcheck source=common.sh
20
+ source "$HERE/common.sh"
21
+
22
+ KAMPODINE_HOST="${KAMPODINE_HOST:-}"
23
+ if [ -n "${KAMPODINE_SSH_KEY:-}" ]; then
24
+ SSH_KEY="$KAMPODINE_SSH_KEY"
25
+ else
26
+ SSH_KEY=""
27
+ fi
28
+ PROFILE_FLAG=""
29
+ HOST_FLAG=""
30
+ KEY_FLAG=""
31
+ WARN_DISK="90"
32
+ WATCH=""
33
+ COUNT=""
34
+
35
+ say() { printf '\033[1;36m[metrics]\033[0m %s\n' "$*"; }
36
+ die() { printf '\033[1;31m[metrics] FAIL:\033[0m %s\n' "$*" >&2; exit 1; }
37
+ usage() {
38
+ local code="${1:-0}"
39
+ printf 'Usage:\n'
40
+ grep '^# kampodine metrics' "$0" | sed 's/^# //'
41
+ cat <<'EOF'
42
+
43
+ One SSH round-trip per snapshot; busybox-safe remote commands; rendering is
44
+ client-side. NO continuous monitoring (Beszel owns that) — one-shot CLI view;
45
+ --watch N re-snapshots every N seconds (Ctrl-C ends), --count M bounds the
46
+ iterations. --warn-disk <pct> (default 90) exits 1 when root disk usage is at
47
+ or above the threshold — the SAME verdict logic as deploy's fail-closed gate.
48
+ Host resolution: --host | --profile <name> | KAMPODINE_PROFILE |
49
+ config defaultProfile | KAMPODINE_HOST.
50
+
51
+ Examples:
52
+ kampodine metrics # profile/host from config or env
53
+ kampodine metrics --host root@203.0.113.10 # explicit target
54
+ kampodine metrics --profile prod # per-instance profile
55
+ kampodine metrics --warn-disk 85 # exit 1 at >= 85% (default 90)
56
+ kampodine metrics --watch 10 --count 6 # snapshot every 10s, 6 times
57
+ kampodine status --verbose # the same snapshot inside status
58
+ EOF
59
+ exit "$code"
60
+ }
61
+
62
+ while [[ $# -gt 0 ]]; do
63
+ case "$1" in
64
+ --profile) PROFILE_FLAG="$2"; shift 2 ;;
65
+ --host) HOST_FLAG="$2"; KAMPODINE_HOST="$2"; shift 2 ;;
66
+ --ssh-key) KEY_FLAG="$2"; SSH_KEY="$2"; shift 2 ;;
67
+ --warn-disk) WARN_DISK="$2"; shift 2 ;;
68
+ --watch) WATCH="$2"; shift 2 ;;
69
+ --count) COUNT="$2"; shift 2 ;;
70
+ -h|--help) usage 0 ;;
71
+ *) die "unknown argument: $1 (--help)" ;;
72
+ esac
73
+ done
74
+
75
+ # profile resolution (flag > KAMPODINE_PROFILE > defaultProfile); an explicit
76
+ # --host/--ssh-key always beats the profile
77
+ profile_resolve "$PROFILE_FLAG" "$HOST_FLAG" "$KEY_FLAG"
78
+ SSH_KEY="${KEY_FLAG:-${KAMPODINE_SSH_KEY:-}}"
79
+
80
+ [[ "$WARN_DISK" =~ ^[0-9]+$ && "$WARN_DISK" -ge 1 && "$WARN_DISK" -le 100 ]] \
81
+ || die "--warn-disk must be a percentage 1-100 (got: $WARN_DISK)"
82
+ [[ -z "$WATCH" || "$WATCH" =~ ^[1-9][0-9]*$ ]] || die "--watch must be a positive-integer number of seconds (got: $WATCH)"
83
+ [[ -z "$COUNT" || "$COUNT" =~ ^[1-9][0-9]*$ ]] || die "--count must be a positive integer (got: $COUNT)"
84
+ [[ -n "$KAMPODINE_HOST" ]] || die "target required: --host root@<ip>, --profile <name>, KAMPODINE_PROFILE, config defaultProfile, or KAMPODINE_HOST=root@<ip>"
85
+
86
+ SSH_ARGS=(-o ConnectTimeout=10 -o BatchMode=yes)
87
+ [[ -n "$SSH_KEY" ]] && SSH_ARGS+=(-i "$SSH_KEY")
88
+ # $1 is a composed remote command — client-side expansion is the design.
89
+ # shellcheck disable=SC2029
90
+ vm() { ssh "${SSH_ARGS[@]}" "$KAMPODINE_HOST" "$1"; }
91
+
92
+ # --- macOS ssh-agent quirk (first run from a fresh machine) -------------------
93
+ if [[ "$(uname -s)" == "Darwin" ]]; then
94
+ SSH_AUTH_SOCK="$(launchctl getenv SSH_AUTH_SOCK 2>/dev/null || true)"
95
+ export SSH_AUTH_SOCK
96
+ fi
97
+
98
+ snapshot() {
99
+ local raw pct verdict
100
+ raw="$(vm "$METRICS_REMOTE_CMD")" || die "cannot reach VM $KAMPODINE_HOST (ssh failed) — nothing to report"
101
+ say "== metrics ($KAMPODINE_HOST) =="
102
+ metrics_render "$raw"
103
+ pct="$(disk_pct_from_df "$M_DISK")"
104
+ verdict="$(disk_verdict "$pct" "$WARN_DISK")"
105
+ case "$verdict" in
106
+ fail)
107
+ printf 'WARNING: root disk at %s%% (>= --warn-disk %s%%) — free space first: kampodine deploy prune --dry-run\n' \
108
+ "${pct:-?}" "$WARN_DISK"
109
+ return 1
110
+ ;;
111
+ warn)
112
+ printf 'WARNING: root disk above 90%% (%s%%) — old sha-tagged deploy images pile up; reclaim: kampodine deploy prune --dry-run\n' "$pct"
113
+ ;;
114
+ esac
115
+ return 0
116
+ }
117
+
118
+ n=0
119
+ while :; do
120
+ n=$(( n + 1 ))
121
+ if ! snapshot; then
122
+ exit 1
123
+ fi
124
+ [[ -n "$COUNT" && "$n" -ge "$COUNT" ]] && exit 0
125
+ [[ -z "$WATCH" ]] && exit 0
126
+ sleep "$WATCH"
127
+ done
@@ -9,13 +9,16 @@
9
9
  # are COLLECTED (files are independent; one bad tenant must not block the others)
10
10
  # and the script exits 1 if any failed — the deploy's smoke gate catches a red migration.
11
11
  # Runs ON the VM from the repo checkout (also fine locally against .tmp data):
12
- # kampodine migrate [--allow-running]
12
+ # kampodine migrate [--allow-running] [--profile <name>]
13
13
  set -euo pipefail
14
14
 
15
15
  SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
16
16
  REPO_ROOT="${REPO_ROOT:-$(cd "${SCRIPT_DIR}/../../.." && pwd)}"
17
17
  TENANT_DIR="${TENANT_DIR:-/data/tenants}"
18
- SERVICE="${SERVICE:-esellar-api}"
18
+ SERVICE="${SERVICE:-kampodine-api}"
19
+ HERE="$SCRIPT_DIR"
20
+ # shellcheck source=common.sh
21
+ source "$HERE/common.sh"
19
22
 
20
23
  log() { printf '[migrate-all] %s\n' "$*"; }
21
24
  fail() { printf '[migrate-all][FAIL] %s\n' "$*" >&2; exit 1; }
@@ -25,7 +28,7 @@ Usage:
25
28
  kampodine migrate [--allow-running]
26
29
 
27
30
  Examples:
28
- kampodine migrate # stop the API first (rc-service esellar-api stop)
31
+ kampodine migrate # stop the API first (rc-service kampodine-api stop)
29
32
  kampodine migrate --allow-running # deliberate: migrate while the API serves (SQLITE_BUSY risk)
30
33
 
31
34
  Runs drizzle migrations over the libSQL dbs under TENANT_DIR (default
@@ -37,13 +40,25 @@ EOF
37
40
  }
38
41
 
39
42
  ALLOW_RUNNING=0
43
+ PROFILE_FLAG=""
44
+ PENDING_PROFILE=0
40
45
  for arg in "$@"; do
46
+ if [[ "$PENDING_PROFILE" -eq 1 ]]; then
47
+ PROFILE_FLAG="$arg"
48
+ PENDING_PROFILE=0
49
+ continue
50
+ fi
41
51
  case "$arg" in
42
52
  --allow-running) ALLOW_RUNNING=1 ;;
43
- -h|--help) usage ;;
53
+ --profile) PENDING_PROFILE=1 ;;
54
+ --help|-h) usage ;;
44
55
  *) fail "unknown argument: $arg (--help)" ;;
45
56
  esac
46
57
  done
58
+ [[ "$PENDING_PROFILE" -eq 0 ]] || fail "--profile requires a value"
59
+ # profile selection is validated for uniformity — migrate consumes no
60
+ # host/sshKey fields (it runs ON the VM from the repo checkout)
61
+ profile_resolve "$PROFILE_FLAG" "" ""
47
62
 
48
63
  [[ -f "${REPO_ROOT}/apps/api/scripts/libsql-migrate/migrate-db.ts" ]] \
49
64
  || fail "repo root not found at ${REPO_ROOT} (set REPO_ROOT=... if the checkout lives elsewhere)"
@@ -51,11 +66,30 @@ command -v pnpm >/dev/null 2>&1 || fail "pnpm not found in PATH"
51
66
  [[ -d "$TENANT_DIR" ]] || fail "tenant dir ${TENANT_DIR} does not exist"
52
67
 
53
68
  # Writing schema while the API serves traffic risks SQLITE_BUSY on a single-writer
54
- # engine — the service must be stopped first (rc-service esellar-api stop). Override exists
69
+ # engine — the service must be stopped first (rc-service ${SERVICE} stop). Override exists
55
70
  # for deliberate local use.
56
- if systemctl is-active --quiet "$SERVICE" 2>/dev/null; then
71
+ #
72
+ # Init-aware detection (fixes the DEAD guard, 2026-10-08): this script runs ON
73
+ # the VM, whose init is OpenRC (Alpine ships no systemd) — a bare `systemctl
74
+ # is-active` always failed there, so the stop-first guard NEVER fired and
75
+ # migrations could run under the serving API. Mirror common.sh's detection
76
+ # order: rc-service (openrc) first, then systemctl (systemd hosts). On a host
77
+ # with neither (local dev machines), state is unknowable — proceed with a
78
+ # loud warning rather than blocking local runs.
79
+ service_running() {
80
+ if command -v rc-service >/dev/null 2>&1; then
81
+ rc-service "$SERVICE" status >/dev/null 2>&1
82
+ elif command -v systemctl >/dev/null 2>&1; then
83
+ systemctl is-active --quiet "$SERVICE"
84
+ else
85
+ log "service state unknown (neither rc-service nor systemctl found) — assuming ${SERVICE} is not running"
86
+ return 1
87
+ fi
88
+ }
89
+
90
+ if service_running; then
57
91
  if [[ $ALLOW_RUNNING -ne 1 ]]; then
58
- fail "${SERVICE} is running — stop it first (rc-service esellar-api stop) or pass --allow-running"
92
+ fail "${SERVICE} is running — stop it first (rc-service ${SERVICE} stop) or pass --allow-running"
59
93
  fi
60
94
  log "WARNING: migrating while ${SERVICE} is running (--allow-running)"
61
95
  fi
package/scripts/status.sh CHANGED
@@ -1,44 +1,164 @@
1
1
  #!/usr/bin/env bash
2
- # status.sh — one command: what is LIVE (through the proxy) + the blue/green
3
- # pair view (instances, reserved IP, health).
2
+ # status.sh — one command: deployment count (ledger), what is LIVE (through
3
+ # the proxy), the VM's disk/image/service state (with --host), and the
4
+ # blue/green pair view.
4
5
  #
5
6
  # Usage:
6
- # kampodine status
7
+ # kampodine status [--host root@<ip>] [--ssh-key <path>]
7
8
  #
8
- # Env: APP_HOST_HEADER (default: app.example.com), OCI_PROFILE, OCI_COMPARTMENT.
9
+ # Without --host: live health + the deployment count (all hosts). With
10
+ # --host: adds VM disk usage (LOUD warning above 90%), the sha-tagged image
11
+ # count with a reclaimable-by-prune estimate, and service states
12
+ # (kampodine-api, kamal-proxy, walshipper if present).
13
+ #
14
+ # Env: APP_HOST_HEADER (default: app.example.com), KAMPODINE_HOST,
15
+ # KAMPODINE_SSH_KEY, OCI_PROFILE, OCI_COMPARTMENT.
9
16
  set -euo pipefail
10
17
  HERE="$(cd "$(dirname "$0")" && pwd)"
11
- HOST="${APP_HOST_HEADER:-app.example.com}"
18
+ # shellcheck source=common.sh
19
+ source "$HERE/common.sh"
20
+
21
+ HOST_HEADER="${PROFILE_PROXY_HOST:-${APP_HOST_HEADER:-app.example.com}}"
22
+ IMAGE="127.0.0.1:5000/kampodine-api"
23
+ DISK_PATH="/var/lib/containers"
24
+ KAMPODINE_HOST="${KAMPODINE_HOST:-}"
25
+ if [ -n "${KAMPODINE_SSH_KEY:-}" ]; then
26
+ SSH_KEY="$KAMPODINE_SSH_KEY"
27
+ else
28
+ SSH_KEY=""
29
+ fi
30
+ PROFILE_FLAG=""
31
+ HOST_FLAG=""
32
+ KEY_FLAG=""
33
+ VERBOSE=0
12
34
 
13
35
  say() { printf '%s\n' "$*"; }
14
36
  die() { printf '[status] FAIL: %s\n' "$*" >&2; exit 1; }
15
37
  usage() {
16
38
  cat <<'EOF'
17
39
  Usage:
18
- kampodine status
40
+ kampodine status [--host root@<ip>] [--profile <name>] [--ssh-key <path>] [--verbose]
19
41
 
20
42
  Examples:
21
- kampodine status # live health (proxy host) + the blue/green pair view
43
+ kampodine status # deployments (ledger) + live health + blue/green pair
44
+ kampodine status --host root@203.0.113.10 # + VM disk usage, image/prune estimate, service states
45
+ kampodine status --profile prod # resolve host/key from a config profile
46
+ kampodine status --host root@203.0.113.10 --verbose # + the full metrics snapshot
47
+ # (load, memory, disk breakdown, containers, top procs)
22
48
 
23
- Env: APP_HOST_HEADER (default app.example.com), OCI_PROFILE, OCI_COMPARTMENT.
49
+ Env: APP_HOST_HEADER (default app.example.com), KAMPODINE_HOST,
50
+ KAMPODINE_SSH_KEY, KAMPODINE_PROFILE, OCI_PROFILE, OCI_COMPARTMENT.
24
51
  EOF
25
52
  exit 0
26
53
  }
27
54
 
28
55
  while [[ $# -gt 0 ]]; do
29
56
  case "$1" in
57
+ --host) HOST_FLAG="$2"; KAMPODINE_HOST="$2"; shift 2 ;;
58
+ --ssh-key) KEY_FLAG="$2"; SSH_KEY="$2"; shift 2 ;;
59
+ --profile) PROFILE_FLAG="$2"; shift 2 ;;
60
+ --verbose) VERBOSE=1; shift ;;
30
61
  -h|--help) usage ;;
31
62
  *) die "unknown argument: $1 (--help)" ;;
32
63
  esac
33
64
  done
34
65
 
35
- say "== live (through the proxy: https://$HOST) =="
36
- if body="$(curl -sf -m 8 "https://$HOST/api/auth/ok" 2>/dev/null)"; then
66
+ # profile resolution (flag > KAMPODINE_PROFILE > defaultProfile); an explicit
67
+ # --host/--ssh-key flag always beats the profile
68
+ profile_resolve "$PROFILE_FLAG" "$HOST_FLAG" "$KEY_FLAG"
69
+ SSH_KEY="${KEY_FLAG:-${KAMPODINE_SSH_KEY:-}}"
70
+
71
+ # --- deployment count (the owner's "total deployments", always visible) -------
72
+ if [[ -n "$KAMPODINE_HOST" ]]; then
73
+ say "== deployments (ledger: $(ledger_file)) =="
74
+ ledger_total_line "$(ledger_count "$KAMPODINE_HOST")" "deployments to $KAMPODINE_HOST"
75
+ else
76
+ say "== deployments (ledger: $(ledger_file)) =="
77
+ ledger_total_line "$(ledger_count)" "across all hosts — pass --host root@<ip> for VM state"
78
+ if [[ $VERBOSE -eq 1 ]]; then
79
+ say "metrics snapshot: needs a target — add --host root@<ip> (or --profile <name>)"
80
+ fi
81
+ fi
82
+ say
83
+
84
+ say "== live (through the proxy: https://$HOST_HEADER) =="
85
+ if body="$(curl -sf -m 8 "https://$HOST_HEADER/api/auth/ok" 2>/dev/null)"; then
37
86
  say "HEALTH OK: $body"
38
- say "build-id : $(curl -sf -m 8 "https://$HOST/build-id.txt" 2>/dev/null || echo unreachable)"
87
+ say "build-id : $(curl -sf -m 8 "https://$HOST_HEADER/build-id.txt" 2>/dev/null || echo unreachable)"
39
88
  else
40
89
  say "UNREACHABLE or unhealthy — check the VM (ssh) and kamal-proxy"
41
90
  fi
91
+
92
+ if [[ -n "$KAMPODINE_HOST" ]]; then
93
+ SSH_ARGS=(-o ConnectTimeout=10 -o BatchMode=yes)
94
+ [[ -n "$SSH_KEY" ]] && SSH_ARGS+=(-i "$SSH_KEY")
95
+ # $1 is a composed remote command — client-side expansion is the design.
96
+ # shellcheck disable=SC2029
97
+ vm() { ssh "${SSH_ARGS[@]}" "$KAMPODINE_HOST" "$1"; }
98
+ if [[ "$(uname -s)" == "Darwin" ]]; then
99
+ SSH_AUTH_SOCK="$(launchctl getenv SSH_AUTH_SOCK 2>/dev/null || true)"
100
+ export SSH_AUTH_SOCK
101
+ fi
102
+
103
+ say
104
+ say "== VM ($KAMPODINE_HOST) =="
105
+ df_out="$(vm "df -P $DISK_PATH 2>/dev/null" || true)"
106
+ pct="$(disk_pct_from_df "$df_out")"
107
+ case "$(disk_verdict "$pct" "")" in
108
+ ok) say "disk : ${pct}% used on $DISK_PATH" ;;
109
+ warn)
110
+ say "disk : ${pct}% used on $DISK_PATH"
111
+ say "WARNING : VM disk above 90% — old sha-tagged deploy images pile up (~1GB each); reclaim: kampodine deploy prune --dry-run"
112
+ ;;
113
+ *) say "disk : unknown (df unreadable)" ;;
114
+ esac
115
+
116
+ images_raw="$(vm "podman images --format '{{.Tag}}|{{.CreatedAt}}|{{.Size}}' $IMAGE 2>/dev/null" || true)"
117
+ images="$(printf '%s\n' "$images_raw" | grep -E '^[0-9a-f]{7,40}\|' || true)"
118
+ running="$(vm "podman ps --format '{{.Image}}' 2>/dev/null" || true)"
119
+ img_count="$(printf '%s\n' "$images" | grep -c . || true)"
120
+ reclaim="$(prune_select "$images" "$running" 2 || true)"
121
+ reclaim_n="0"
122
+ reclaim_tags="(none)"
123
+ reclaim_size="~0 MB"
124
+ if [[ -n "$reclaim" ]]; then
125
+ reclaim_n="$(printf '%s\n' "$reclaim" | grep -c . || true)"
126
+ reclaim_tags="$(printf '%s\n' "$reclaim" | cut -d'|' -f1 | paste -sd, -)"
127
+ reclaim_size="$(sum_sizes_human "$(printf '%s\n' "$reclaim" | cut -d'|' -f2)")"
128
+ fi
129
+ say "images : ${img_count:-0} sha-tagged deploy image(s); prune would remove ${reclaim_n} ${reclaim_tags} (${reclaim_size}): kampodine deploy prune --dry-run"
130
+
131
+ # init detection (groundwork for non-Alpine targets): the roll call keeps
132
+ # the historical shape on Alpine (byte-identical "for s in … rc-service …")
133
+ # and swaps the probe command on systemd hosts.
134
+ init_detect vm >/dev/null 2>&1 || true
135
+ case "$INIT_SYSTEM" in
136
+ systemd) svc_probe='systemctl status "$s"' ;;
137
+ *) svc_probe='rc-service "$s" status' ;;
138
+ esac
139
+ say "services:"
140
+ vm "for s in kampodine-api kamal-proxy walshipper; do if $svc_probe >/dev/null 2>&1; then echo \"\$s: running\"; else echo \"\$s: not-running\"; fi; done" \
141
+ || say "(service roll call failed)"
142
+
143
+ if [[ $VERBOSE -eq 1 ]]; then
144
+ say
145
+ say "== metrics snapshot =="
146
+ if metrics_raw="$(vm "$METRICS_REMOTE_CMD" 2>/dev/null)"; then
147
+ metrics_render "$metrics_raw"
148
+ metrics_pct="$(disk_pct_from_df "$M_DISK")"
149
+ if [[ -n "$metrics_pct" ]] && [[ "$metrics_pct" -gt 90 ]]; then
150
+ say "WARNING : root disk above 90% — reclaim: kampodine deploy prune --dry-run"
151
+ fi
152
+ else
153
+ say "(metrics snapshot failed — VM unreachable?)"
154
+ fi
155
+ fi
156
+ fi
157
+
42
158
  say
43
159
  say "== blue/green pair =="
44
- exec bash "$HERE/bluegreen.sh" status
160
+ if [[ -n "$PROFILE_FLAG" ]]; then
161
+ exec bash "$HERE/bluegreen.sh" status --profile "$PROFILE_FLAG"
162
+ else
163
+ exec bash "$HERE/bluegreen.sh" status
164
+ fi