@leverege/build-tools 2.114.0-DEVOP-578.2 → 2.114.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +148 -5
- package/package.json +2 -2
- package/src/helm-charts/llm-prompt-server/helmup.bootstrap +52 -0
- package/src/helmup.sh +37 -6
- package/src/k8s/commands/cryo.mjs +741 -48
- package/src/k8s/commands/exe.mjs +1 -5
- package/src/k8s/k8s.mjs +2 -3
- package/src/k8s/lib/kubectl.mjs +41 -10
- package/src/service-man/config/service-accounts/llm-prompt-server.json +2 -1
|
@@ -1,49 +1,138 @@
|
|
|
1
|
+
// Hibernate and wake phases are inherently sequential — each kubectl/gcloud call
|
|
2
|
+
// must complete before the next begins to maintain dependency order. Every
|
|
3
|
+
// await-in-loop in this file is intentional.
|
|
4
|
+
/* eslint-disable no-await-in-loop */
|
|
5
|
+
|
|
6
|
+
import fs from 'node:fs/promises'
|
|
7
|
+
|
|
1
8
|
import { Command } from 'commander'
|
|
2
9
|
import chalk from 'chalk'
|
|
3
10
|
import { $ } from 'zx'
|
|
4
11
|
|
|
5
|
-
import { log, errorExit, sleep, warning, proceed,
|
|
12
|
+
import { log, errorExit, sleep, warning, proceed, parseYamlFile } from '../../Utils.mjs'
|
|
6
13
|
import { overwhelmGate } from '../lib/gate.mjs'
|
|
14
|
+
import { listDeployments, namespaceExists, listStatefulSets, getReplicaCount } from '../lib/kubectl.mjs'
|
|
7
15
|
|
|
8
16
|
$.verbose = false
|
|
9
17
|
|
|
10
|
-
// ──
|
|
18
|
+
// ── Known service categories (from helmup.sh, in hibernate order) ─────────
|
|
19
|
+
// Within each phase, services are listed in the order they should be stopped
|
|
20
|
+
// (reverse of bootstrap/deploy order within that phase).
|
|
11
21
|
|
|
12
|
-
const
|
|
13
|
-
|
|
14
|
-
|
|
22
|
+
const PLATFORM_TRANSPONDERS = [ 'transponder-tsdb', 'transponder-rt', 'transponder-bq', 'transponder-dh' ]
|
|
23
|
+
const PLATFORM_SUPPORT = [ 'rule-engine', 'resource-server', 'messenger', 'llm-prompt-server', 'imagine', 'emailer', 'db-curator' ]
|
|
24
|
+
const PLATFORM_CORE = [ 'scheduler', 'message-processor', 'api-server', 'authz-server' ]
|
|
25
|
+
const AUXILIARY_KNOWN = [ 'vms-server', 'vin-decoder-server', 'reason', 'pusher', 'push-notifier', 'pubsub-pulse', 'pgbouncer', 'postgres-pgbouncer', 'overdose', 'geotile-server', 'fota-server', 'argocd' ]
|
|
26
|
+
const SYSTEM_DEPLOY = [ 'reloader-reloader', 'reflector' ]
|
|
27
|
+
|
|
28
|
+
const ALL_KNOWN = new Set( [
|
|
29
|
+
...PLATFORM_TRANSPONDERS, ...PLATFORM_SUPPORT, ...PLATFORM_CORE,
|
|
30
|
+
...AUXILIARY_KNOWN, ...SYSTEM_DEPLOY,
|
|
31
|
+
] )
|
|
15
32
|
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
33
|
+
// ── Config ─────────────────────────────────────────────────────────────────
|
|
34
|
+
|
|
35
|
+
const loadConfig = async ( configPath ) => {
|
|
36
|
+
let config
|
|
37
|
+
try {
|
|
38
|
+
config = await parseYamlFile( configPath )
|
|
39
|
+
} catch {
|
|
40
|
+
errorExit( `Config file not found or invalid YAML: ${configPath}` )
|
|
19
41
|
}
|
|
20
|
-
if ( !deploymentPhases || !nodePools ) {
|
|
42
|
+
if ( !config?.deploymentPhases || !config?.nodePools ) {
|
|
21
43
|
errorExit( 'Config must include deploymentPhases and nodePools' )
|
|
22
44
|
}
|
|
23
45
|
return config
|
|
24
46
|
}
|
|
25
47
|
|
|
26
|
-
|
|
48
|
+
const overwhelmKey = async key => ( await $`overwhelm -k ${key}` ).stdout.trim()
|
|
49
|
+
|
|
50
|
+
const getClusterEnv = async () => ( {
|
|
51
|
+
project : await overwhelmKey( 'PROJECT_ID' ),
|
|
52
|
+
name : await overwhelmKey( 'CLUSTER_NAME' ),
|
|
53
|
+
region : await overwhelmKey( 'GCE_REGION' ),
|
|
54
|
+
} )
|
|
55
|
+
|
|
56
|
+
// ── Display ────────────────────────────────────────────────────────────────
|
|
27
57
|
|
|
28
58
|
const banner = ( text ) => {
|
|
29
59
|
const bar = '─'.repeat( Math.max( 0, 56 - text.length ) )
|
|
30
60
|
log( chalk.cyan.bold( `\n── ${text} ${bar}` ) )
|
|
31
61
|
}
|
|
32
62
|
|
|
33
|
-
// ── kubectl
|
|
63
|
+
// ── kubectl ────────────────────────────────────────────────────────────────
|
|
64
|
+
|
|
65
|
+
const scaleDeployments = async ( names, replicas, namespace ) => {
|
|
66
|
+
return $`kubectl scale deployment ${names} --replicas=${replicas} -n ${namespace}`
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
const scaleStatefulSets = async ( names, replicas, namespace ) => {
|
|
70
|
+
return $`kubectl scale statefulset ${names} --replicas=${replicas} -n ${namespace}`
|
|
71
|
+
}
|
|
34
72
|
|
|
35
|
-
const
|
|
36
|
-
$`kubectl scale deployment ${names} --replicas=${replicas} -n ${namespace}`
|
|
73
|
+
const CNPG_CRD = 'clusters.postgresql.cnpg.io'
|
|
37
74
|
|
|
38
|
-
const
|
|
39
|
-
|
|
75
|
+
const cnpgHibernate = async ( name, namespace ) => {
|
|
76
|
+
log( ` cnpg hibernate on: ${name} (${namespace})` )
|
|
77
|
+
await $`kubectl cnpg hibernate on -n ${namespace} ${name}`
|
|
78
|
+
}
|
|
40
79
|
|
|
41
|
-
const
|
|
42
|
-
log(
|
|
80
|
+
const cnpgWake = async ( name, namespace, timeoutSecs = 300 ) => {
|
|
81
|
+
log( ` cnpg hibernate off: ${name} (${namespace})` )
|
|
43
82
|
try {
|
|
44
|
-
await $`kubectl
|
|
83
|
+
await $`kubectl cnpg hibernate off -n ${namespace} ${name}`
|
|
84
|
+
} catch ( err ) {
|
|
85
|
+
if ( err.stderr?.includes( 'not found' ) ) {
|
|
86
|
+
log( ` ${name} — cluster not found, skipping` )
|
|
87
|
+
return
|
|
88
|
+
}
|
|
89
|
+
throw err
|
|
90
|
+
}
|
|
91
|
+
log( ` waiting for ${name} to be Ready (up to ${timeoutSecs}s)...` )
|
|
92
|
+
try {
|
|
93
|
+
await $`kubectl wait ${CNPG_CRD} ${name} -n ${namespace} --for=condition=Ready --timeout=${timeoutSecs}s`
|
|
45
94
|
} catch {
|
|
46
|
-
warning(
|
|
95
|
+
warning( `${name} not Ready after ${timeoutSecs}s — proceeding, but poolers may fail to connect.` )
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
const anyCnpgClusters = async ( namespace, retries = 4, delayMs = 10_000 ) => {
|
|
100
|
+
for ( let i = 0; i <= retries; i++ ) {
|
|
101
|
+
try {
|
|
102
|
+
const result = await $`kubectl get ${CNPG_CRD} -n ${namespace} --no-headers -o name`
|
|
103
|
+
if ( result.stdout.trim().length > 0 ) return true
|
|
104
|
+
} catch {
|
|
105
|
+
// webhook may not be registered yet — retry
|
|
106
|
+
}
|
|
107
|
+
if ( i < retries ) {
|
|
108
|
+
log( ` CNPG webhook not ready, retrying in ${delayMs / 1000}s… (${i + 1}/${retries})` )
|
|
109
|
+
await sleep( delayMs )
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
return false
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
const filterExisting = async ( kind, names, namespace ) => {
|
|
116
|
+
try {
|
|
117
|
+
const result = await $`kubectl get ${kind} -n ${namespace} --no-headers -o name`
|
|
118
|
+
const existing = new Set(
|
|
119
|
+
result.stdout.trim().split( '\n' ).filter( Boolean )
|
|
120
|
+
.map( n => n.replace( /^[^/]+\//, '' ) )
|
|
121
|
+
)
|
|
122
|
+
return names.filter( n => existing.has( n ) )
|
|
123
|
+
} catch {
|
|
124
|
+
return []
|
|
125
|
+
}
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
const waitForPodsGone = async ( namespaces, timeoutSecs = 300 ) => {
|
|
129
|
+
log( `\nWaiting for pods to terminate in [${namespaces.join( ', ' )}] (up to ${timeoutSecs}s)...` )
|
|
130
|
+
for ( const ns of namespaces ) {
|
|
131
|
+
try {
|
|
132
|
+
await $`kubectl wait pod --all -n ${ns} --for=delete --timeout=${timeoutSecs}s`
|
|
133
|
+
} catch {
|
|
134
|
+
warning( `Some pods still terminating in ${ns} after ${timeoutSecs}s — proceeding.` )
|
|
135
|
+
}
|
|
47
136
|
}
|
|
48
137
|
}
|
|
49
138
|
|
|
@@ -52,26 +141,23 @@ const waitForNodesReady = async ( timeoutSecs = 600 ) => {
|
|
|
52
141
|
await $`kubectl wait node --all --for=condition=Ready --timeout=${timeoutSecs}s`
|
|
53
142
|
}
|
|
54
143
|
|
|
55
|
-
// ── gcloud
|
|
144
|
+
// ── gcloud ─────────────────────────────────────────────────────────────────
|
|
56
145
|
|
|
57
146
|
const updateAutoscalerMin = async ( pool, cluster, min ) => {
|
|
58
147
|
log( ` ${pool.name}: autoscaler min → ${min}` )
|
|
59
|
-
|
|
60
|
-
await $`gcloud container node-pools update ${pool.name} --cluster=${name} --project=${project} --region=${region} --enable-autoscaling --min-nodes=${min} --max-nodes=${pool.autoscaler.max} --quiet`
|
|
148
|
+
await $`gcloud container node-pools update ${pool.name} --cluster=${cluster.name} --project=${cluster.project} --region=${cluster.region} --enable-autoscaling --min-nodes=${min} --max-nodes=${pool.autoscaler.max} --quiet`
|
|
61
149
|
}
|
|
62
150
|
|
|
63
151
|
const resizePool = async ( pool, cluster, size ) => {
|
|
64
152
|
log( ` ${pool.name} → ${size} node(s)` )
|
|
65
|
-
|
|
66
|
-
await $`gcloud container clusters resize ${name} --node-pool=${pool.name} --num-nodes=${size} --project=${project} --region=${region} --quiet`
|
|
153
|
+
await $`gcloud container clusters resize ${cluster.name} --node-pool=${pool.name} --num-nodes=${size} --project=${cluster.project} --region=${cluster.region} --quiet`
|
|
67
154
|
}
|
|
68
155
|
|
|
69
|
-
// ── Hibernate
|
|
156
|
+
// ── Hibernate ──────────────────────────────────────────────────────────────
|
|
70
157
|
|
|
71
158
|
const runHibernate = async ( configPath ) => {
|
|
72
159
|
const config = await loadConfig( configPath )
|
|
73
160
|
const {
|
|
74
|
-
cluster,
|
|
75
161
|
namespace = 'default',
|
|
76
162
|
phasePauseMs = 10_000,
|
|
77
163
|
deploymentPhases,
|
|
@@ -80,27 +166,104 @@ const runHibernate = async ( configPath ) => {
|
|
|
80
166
|
} = config
|
|
81
167
|
|
|
82
168
|
await overwhelmGate()
|
|
169
|
+
const cluster = await getClusterEnv()
|
|
170
|
+
|
|
83
171
|
await proceed( `Hibernate ${chalk.yellow.bold( cluster.name )}? All workloads scale to 0 and node pools drain.` )
|
|
84
172
|
|
|
173
|
+
// deferShutdown phases (e.g. CNPG operator) are held back until after
|
|
174
|
+
// statefulset phases so their webhooks remain available for cnpg hibernate on.
|
|
175
|
+
const deferredDeployPhases = []
|
|
176
|
+
|
|
85
177
|
for ( const group of deploymentPhases ) {
|
|
178
|
+
if ( group.deferShutdown ) {
|
|
179
|
+
deferredDeployPhases.push( group )
|
|
180
|
+
continue
|
|
181
|
+
}
|
|
182
|
+
const phaseNamespace = group.namespace ?? namespace
|
|
183
|
+
const members = group.optional ?
|
|
184
|
+
await filterExisting( 'deployment', group.members, phaseNamespace ) :
|
|
185
|
+
group.members
|
|
186
|
+
if ( !members.length ) {
|
|
187
|
+
log( ` ${group.name} — not present, skipping` )
|
|
188
|
+
continue
|
|
189
|
+
}
|
|
86
190
|
banner( `Deployments: ${group.name}` )
|
|
87
|
-
await scaleDeployments(
|
|
191
|
+
await scaleDeployments( members, 0, phaseNamespace )
|
|
88
192
|
await sleep( group.pauseMs ?? phasePauseMs )
|
|
89
193
|
}
|
|
90
194
|
|
|
195
|
+
// Before statefulset phases, ensure any deferShutdown phases are actually
|
|
196
|
+
// running — a prior failed run may have left them at 0, which would cause
|
|
197
|
+
// CNPG webhook calls to fail. Wait for rollout (not just a fixed sleep)
|
|
198
|
+
// so the webhook is registered before cnpg hibernate on is called.
|
|
199
|
+
if ( deferredDeployPhases.length && statefulsetPhases.some( p => p.type === 'cnpg' ) ) {
|
|
200
|
+
banner( 'Ensuring deferred services are running' )
|
|
201
|
+
for ( const group of deferredDeployPhases ) {
|
|
202
|
+
const phaseNamespace = group.namespace ?? namespace
|
|
203
|
+
log( ` ${group.name} → 1` )
|
|
204
|
+
await scaleDeployments( group.members, 1, phaseNamespace )
|
|
205
|
+
if ( group.waitForReady ) {
|
|
206
|
+
log( ` waiting for ${group.name} to be ready...` )
|
|
207
|
+
for ( const member of group.members ) {
|
|
208
|
+
try {
|
|
209
|
+
await $`kubectl rollout status deployment ${member} -n ${phaseNamespace} --timeout=120s`
|
|
210
|
+
} catch {
|
|
211
|
+
warning( `${member} not ready after 120s — proceeding` )
|
|
212
|
+
}
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
|
|
91
218
|
if ( statefulsetPhases.length ) {
|
|
92
219
|
for ( const group of statefulsetPhases ) {
|
|
220
|
+
const phaseNamespace = group.namespace ?? namespace
|
|
221
|
+
const kind = group.type === 'cnpg' ? CNPG_CRD : 'statefulset'
|
|
222
|
+
const members = group.optional ?
|
|
223
|
+
await filterExisting( kind, group.members, phaseNamespace ) :
|
|
224
|
+
group.members
|
|
225
|
+
if ( !members.length ) {
|
|
226
|
+
log( ` ${group.name} — not present, skipping` )
|
|
227
|
+
continue
|
|
228
|
+
}
|
|
93
229
|
banner( `StatefulSets: ${group.name}` )
|
|
94
|
-
|
|
230
|
+
if ( group.type === 'cnpg' ) {
|
|
231
|
+
for ( const member of members ) {
|
|
232
|
+
await cnpgHibernate( member, phaseNamespace )
|
|
233
|
+
}
|
|
234
|
+
} else {
|
|
235
|
+
await scaleStatefulSets( members, 0, phaseNamespace )
|
|
236
|
+
}
|
|
95
237
|
await sleep( group.pauseMs ?? phasePauseMs )
|
|
96
238
|
}
|
|
97
239
|
} else {
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
240
|
+
warning( 'statefulsetPhases is empty — StatefulSet shutdown skipped.' )
|
|
241
|
+
warning( 'Define phases in cluster-config.yaml when ready to hibernate database workloads.' )
|
|
242
|
+
warning( 'Note: CNPG clusters require kubectl cnpg hibernate, not kubectl scale.' )
|
|
101
243
|
}
|
|
102
244
|
|
|
103
|
-
|
|
245
|
+
if ( deferredDeployPhases.length ) {
|
|
246
|
+
for ( const group of deferredDeployPhases ) {
|
|
247
|
+
const phaseNamespace = group.namespace ?? namespace
|
|
248
|
+
const members = group.optional ?
|
|
249
|
+
await filterExisting( 'deployment', group.members, phaseNamespace ) :
|
|
250
|
+
group.members
|
|
251
|
+
if ( !members.length ) {
|
|
252
|
+
log( ` ${group.name} — not present, skipping` )
|
|
253
|
+
continue
|
|
254
|
+
}
|
|
255
|
+
banner( `Deployments: ${group.name}` )
|
|
256
|
+
await scaleDeployments( members, 0, phaseNamespace )
|
|
257
|
+
await sleep( group.pauseMs ?? phasePauseMs )
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
const namespacesToDrain = [ ...new Set( [
|
|
262
|
+
namespace,
|
|
263
|
+
...deploymentPhases.map( g => g.namespace ?? namespace ),
|
|
264
|
+
...statefulsetPhases.map( g => g.namespace ?? namespace ),
|
|
265
|
+
] ) ]
|
|
266
|
+
await waitForPodsGone( namespacesToDrain )
|
|
104
267
|
|
|
105
268
|
const poolsWithAutoscaler = nodePools.filter( p => p.autoscaler )
|
|
106
269
|
if ( poolsWithAutoscaler.length ) {
|
|
@@ -116,19 +279,23 @@ const runHibernate = async ( configPath ) => {
|
|
|
116
279
|
}
|
|
117
280
|
|
|
118
281
|
log( chalk.green( '\nCluster hibernated. Control plane (~$0.10/hr) continues running.' ) )
|
|
119
|
-
log( `To wake: ${chalk.cyan(
|
|
282
|
+
log( `To wake: ${chalk.cyan( 'k8s cryo wake' )}` )
|
|
120
283
|
}
|
|
121
284
|
|
|
122
|
-
// ── Wake
|
|
285
|
+
// ── Wake ───────────────────────────────────────────────────────────────────
|
|
123
286
|
|
|
124
287
|
const runWake = async ( configPath ) => {
|
|
125
288
|
const config = await loadConfig( configPath )
|
|
126
|
-
const {
|
|
289
|
+
const {
|
|
290
|
+
namespace = 'default',
|
|
291
|
+
nodePools,
|
|
292
|
+
deploymentPhases = [],
|
|
293
|
+
statefulsetPhases = [],
|
|
294
|
+
} = config
|
|
127
295
|
|
|
128
296
|
await overwhelmGate()
|
|
297
|
+
const cluster = await getClusterEnv()
|
|
129
298
|
|
|
130
|
-
// Restore autoscaler minimums before resizing so the autoscaler
|
|
131
|
-
// doesn't immediately scale back down the nodes we're bringing up.
|
|
132
299
|
const poolsWithAutoscaler = nodePools.filter( p => p.autoscaler )
|
|
133
300
|
if ( poolsWithAutoscaler.length ) {
|
|
134
301
|
banner( 'Restoring autoscaler minimums' )
|
|
@@ -139,26 +306,552 @@ const runWake = async ( configPath ) => {
|
|
|
139
306
|
|
|
140
307
|
banner( 'Restoring node pools' )
|
|
141
308
|
for ( const pool of nodePools ) {
|
|
142
|
-
if ( pool.restoreSize
|
|
143
|
-
await resizePool( pool, cluster, pool.restoreSize )
|
|
144
|
-
} else {
|
|
309
|
+
if ( pool.restoreSize <= 0 ) {
|
|
145
310
|
log( ` ${pool.name} → skipped (autoscaler brings up on demand)` )
|
|
311
|
+
continue
|
|
146
312
|
}
|
|
313
|
+
// Autoscaled pools start at 1 node per zone — enough for initial scheduling.
|
|
314
|
+
// The autoscaler brings up additional nodes as workloads are placed.
|
|
315
|
+
const initialSize = pool.autoscaler ? 1 : pool.restoreSize
|
|
316
|
+
await resizePool( pool, cluster, initialSize )
|
|
147
317
|
}
|
|
148
318
|
|
|
149
319
|
await waitForNodesReady()
|
|
150
320
|
|
|
151
|
-
|
|
321
|
+
// Build a unified wake list from both phase arrays, tagged with kind.
|
|
322
|
+
// Phases at the same wakeOrder run in parallel; levels are processed in order.
|
|
323
|
+
// Helm won't restore manually-scaled-to-0 resources on its own — it only
|
|
324
|
+
// patches fields that changed between old and new manifest — so cryo must
|
|
325
|
+
// restore replica counts explicitly before helmup reconciles.
|
|
326
|
+
const wakeList = [
|
|
327
|
+
...deploymentPhases.map( p => ( { ...p, phaseKind: 'deployment' } ) ),
|
|
328
|
+
...statefulsetPhases.map( p => ( { ...p, phaseKind: 'statefulset' } ) ),
|
|
329
|
+
]
|
|
330
|
+
|
|
331
|
+
const wakeMap = new Map()
|
|
332
|
+
for ( const group of wakeList ) {
|
|
333
|
+
const order = group.wakeOrder ?? Infinity
|
|
334
|
+
if ( !wakeMap.has( order ) ) wakeMap.set( order, [] )
|
|
335
|
+
wakeMap.get( order ).push( group )
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
const wakeGroup = async ( group ) => {
|
|
339
|
+
const phaseNamespace = group.namespace ?? namespace
|
|
340
|
+
|
|
341
|
+
if ( group.phaseKind === 'statefulset' && group.type === 'cnpg' ) {
|
|
342
|
+
// Attempt hibernate off unconditionally — cnpgWake handles not-found gracefully.
|
|
343
|
+
// filterExisting is unreliable here: hibernated clusters are invisible when the
|
|
344
|
+
// operator is down, and the operator may not have been running during a prior failed run.
|
|
345
|
+
for ( const member of group.members ) {
|
|
346
|
+
await cnpgWake( member, phaseNamespace )
|
|
347
|
+
}
|
|
348
|
+
return
|
|
349
|
+
}
|
|
350
|
+
|
|
351
|
+
if ( group.requiresCnpg && !await anyCnpgClusters( phaseNamespace ) ) {
|
|
352
|
+
log( ` ${group.name} — no CNPG clusters in ${phaseNamespace}, skipping` )
|
|
353
|
+
return
|
|
354
|
+
}
|
|
355
|
+
|
|
356
|
+
const kind = group.phaseKind === 'statefulset' ? 'statefulset' : 'deployment'
|
|
357
|
+
const members = group.optional ?
|
|
358
|
+
await filterExisting( kind, group.members, phaseNamespace ) :
|
|
359
|
+
group.members
|
|
360
|
+
if ( !members.length ) {
|
|
361
|
+
log( ` ${group.name} — not present, skipping` )
|
|
362
|
+
return
|
|
363
|
+
}
|
|
364
|
+
const replicas = group.restoreReplicas ?? 1
|
|
365
|
+
log( ` ${group.name} → ${replicas}` )
|
|
366
|
+
if ( group.phaseKind === 'statefulset' ) {
|
|
367
|
+
await scaleStatefulSets( members, replicas, phaseNamespace )
|
|
368
|
+
} else {
|
|
369
|
+
await scaleDeployments( members, replicas, phaseNamespace )
|
|
370
|
+
}
|
|
371
|
+
if ( group.waitForReady ) {
|
|
372
|
+
log( ` waiting for ${group.name} to be ready...` )
|
|
373
|
+
for ( const member of members ) {
|
|
374
|
+
try {
|
|
375
|
+
await $`kubectl rollout status deployment ${member} -n ${phaseNamespace} --timeout=120s`
|
|
376
|
+
} catch {
|
|
377
|
+
warning( `${member} not ready after 120s — proceeding` )
|
|
378
|
+
}
|
|
379
|
+
}
|
|
380
|
+
}
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
banner( 'Restoring workloads' )
|
|
384
|
+
const levels = [ ...wakeMap.keys() ].sort( ( a, b ) => a - b )
|
|
385
|
+
for ( const level of levels ) {
|
|
386
|
+
await Promise.all( wakeMap.get( level ).map( wakeGroup ) )
|
|
387
|
+
}
|
|
388
|
+
|
|
389
|
+
log( chalk.green( '\nCluster restored. Run helmup to reconcile any config or version changes.' ) )
|
|
390
|
+
}
|
|
391
|
+
|
|
392
|
+
// ── Init: YAML generation ──────────────────────────────────────────────────
|
|
393
|
+
|
|
394
|
+
const buildConfigYaml = ( { clusterName, namespace, phasePauseMs, deployPhases, stsPhases, pools } ) => {
|
|
395
|
+
const bar = label => ` # ── ${label} ${'─'.repeat( Math.max( 0, 57 - label.length ) )}`
|
|
396
|
+
|
|
397
|
+
const serializePhase = ( phase ) => {
|
|
398
|
+
const ns = phase.namespace && phase.namespace !== namespace ?
|
|
399
|
+
`\n namespace: ${phase.namespace}` : ''
|
|
400
|
+
const type = phase.type ? `\n type: ${phase.type}` : ''
|
|
401
|
+
const opt = phase.optional ? '\n optional: true' : ''
|
|
402
|
+
const rcnpg = phase.requiresCnpg ? '\n requiresCnpg: true' : ''
|
|
403
|
+
const wo = phase.wakeOrder != null ? `\n wakeOrder: ${phase.wakeOrder}` : ''
|
|
404
|
+
const defer = phase.deferShutdown ? '\n deferShutdown: true' : ''
|
|
405
|
+
const waitRdy = phase.waitForReady ? '\n waitForReady: true' : ''
|
|
406
|
+
const restore = phase.restoreReplicas != null && phase.restoreReplicas !== 1 ?
|
|
407
|
+
`\n restoreReplicas: ${phase.restoreReplicas}` : ''
|
|
408
|
+
const pauseMs = phase.pauseMs ? `\n pauseMs: ${phase.pauseMs}` : ''
|
|
409
|
+
const members = phase.members.map( m => ` - ${m}` ).join( '\n' )
|
|
410
|
+
return ` - name: "${phase.name}"${type}${ns}${opt}${rcnpg}${wo}${defer}${waitRdy}${restore}${pauseMs}\n members:\n${members}`
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
const phaseBlock = ( phase ) => {
|
|
414
|
+
const header = phase.header ? `\n${bar( phase.header )}\n` : ''
|
|
415
|
+
const comment = phase.comment ? ` # ${phase.comment}\n` : ''
|
|
416
|
+
return `${header}${comment}${serializePhase( phase )}`
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
const deployBlocks = deployPhases.filter( p => p.members?.length ).map( phaseBlock ).join( '\n' )
|
|
420
|
+
|
|
421
|
+
const stsContent = stsPhases.filter( p => p.members?.length )
|
|
422
|
+
const stsBlocks = stsContent.map( phaseBlock ).join( '\n' )
|
|
423
|
+
const hasCnpg = stsContent.some( p => p.type === 'cnpg' )
|
|
424
|
+
|
|
425
|
+
const poolBlocks = pools.map( ( pool ) => {
|
|
426
|
+
const auto = pool.autoscaler ?
|
|
427
|
+
`\n autoscaler:\n min: ${pool.autoscaler.min}\n max: ${pool.autoscaler.max}` : ''
|
|
428
|
+
return ` - name: ${pool.name}\n restoreSize: ${pool.restoreSize}${auto}`
|
|
429
|
+
} ).join( '\n' )
|
|
430
|
+
|
|
431
|
+
const cnpgNote = hasCnpg ?
|
|
432
|
+
'\n# CNPG clusters use type: cnpg → kubectl cnpg hibernate on/off (not kubectl scale).' : ''
|
|
433
|
+
|
|
434
|
+
return `# k8s-cryo.yaml — cluster hibernation / wake configuration for ${clusterName}
|
|
435
|
+
# Cluster identity (project, name, region) is read from overwhelm at runtime.
|
|
436
|
+
# Generated by: k8s cryo init — review before committing.
|
|
437
|
+
#
|
|
438
|
+
# Phases execute top → bottom during hibernate.
|
|
439
|
+
# wakeOrder controls the wake sequence independently of hibernate order.
|
|
440
|
+
|
|
441
|
+
namespace: ${namespace}
|
|
442
|
+
phasePauseMs: ${phasePauseMs}
|
|
443
|
+
|
|
444
|
+
deploymentPhases:
|
|
445
|
+
${deployBlocks}
|
|
446
|
+
|
|
447
|
+
# StatefulSet shutdown phases — executed after all deployments are down.${cnpgNote}
|
|
448
|
+
${stsContent.length ? `statefulsetPhases:\n${stsBlocks}` : 'statefulsetPhases: [] # TODO: add StatefulSet phases before hibernating'}
|
|
449
|
+
|
|
450
|
+
# Node pools — restoreSize is nodes per zone (regional cluster × 3 zones).
|
|
451
|
+
# Pools with restoreSize: 0 are managed on demand by the cluster autoscaler.
|
|
452
|
+
nodePools:
|
|
453
|
+
${poolBlocks}
|
|
454
|
+
`
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
// ── Init ───────────────────────────────────────────────────────────────────
|
|
458
|
+
|
|
459
|
+
const runInit = async ( outputPath, opts ) => {
|
|
460
|
+
const { namespace, force } = opts
|
|
461
|
+
|
|
462
|
+
if ( !force ) {
|
|
463
|
+
try {
|
|
464
|
+
await fs.access( outputPath )
|
|
465
|
+
errorExit( `${outputPath} already exists — use --force to overwrite` )
|
|
466
|
+
} catch ( err ) {
|
|
467
|
+
if ( err.code !== 'ENOENT' ) errorExit( `Cannot access ${outputPath}: ${err.message}` )
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
|
|
471
|
+
await overwhelmGate()
|
|
472
|
+
const cluster = await getClusterEnv()
|
|
473
|
+
|
|
474
|
+
log( `\nScanning ${chalk.cyan( cluster.name )} (${cluster.project})…` )
|
|
475
|
+
|
|
476
|
+
// ── Default namespace deployments ─────────────────────────────────────────
|
|
477
|
+
|
|
478
|
+
banner( 'Listing deployments' )
|
|
479
|
+
const allDeploys = await listDeployments( namespace )
|
|
480
|
+
log( ` Found ${allDeploys.length} deployment(s) in ${namespace}` )
|
|
481
|
+
|
|
482
|
+
const deployed = new Set( allDeploys )
|
|
483
|
+
const transponders = PLATFORM_TRANSPONDERS.filter( d => deployed.has( d ) )
|
|
484
|
+
const platformSupport = PLATFORM_SUPPORT.filter( d => deployed.has( d ) )
|
|
485
|
+
const platformCore = PLATFORM_CORE.filter( d => deployed.has( d ) )
|
|
486
|
+
const auxiliary = AUXILIARY_KNOWN.filter( d => deployed.has( d ) )
|
|
487
|
+
const systemDeploy = SYSTEM_DEPLOY.filter( d => deployed.has( d ) )
|
|
488
|
+
const project = allDeploys.filter( d => !ALL_KNOWN.has( d ) )
|
|
489
|
+
|
|
490
|
+
if ( project.length ) {
|
|
491
|
+
log( ` → Project / product (review placement): ${project.join( ', ' )}` )
|
|
492
|
+
}
|
|
493
|
+
|
|
494
|
+
// ── System namespace scans ────────────────────────────────────────────────
|
|
495
|
+
|
|
496
|
+
banner( 'Scanning system namespaces' )
|
|
497
|
+
|
|
498
|
+
// Returns deployments present in ns from candidates; skips if namespace absent.
|
|
499
|
+
const scanNs = async ( ns, candidates ) => {
|
|
500
|
+
if ( !await namespaceExists( ns ) ) return []
|
|
501
|
+
const found = await filterExisting( 'deployment', candidates, ns )
|
|
502
|
+
log( found.length ?
|
|
503
|
+
` ${ns}: ${found.join( ', ' )}` :
|
|
504
|
+
` ${ns}: (empty)` )
|
|
505
|
+
return found
|
|
506
|
+
}
|
|
507
|
+
|
|
508
|
+
// CNPG — operator, poolers, cluster CRDs
|
|
509
|
+
const cnpgSystemExists = await namespaceExists( 'cnpg-system' )
|
|
510
|
+
let cnpgOperatorMembers = []
|
|
511
|
+
let cnpgPoolers = []
|
|
512
|
+
let cnpgClusters = []
|
|
513
|
+
|
|
514
|
+
if ( cnpgSystemExists ) {
|
|
515
|
+
cnpgOperatorMembers = await filterExisting(
|
|
516
|
+
'deployment', [ 'barman-cloud', 'cnpg-controller-manager' ], 'cnpg-system'
|
|
517
|
+
)
|
|
518
|
+
log( ` cnpg-system: ${cnpgOperatorMembers.join( ', ' ) || '(none)'}` )
|
|
519
|
+
|
|
520
|
+
if ( await namespaceExists( 'cnpg-operands' ) ) {
|
|
521
|
+
const operandDeploys = await listDeployments( 'cnpg-operands' )
|
|
522
|
+
cnpgPoolers = operandDeploys.filter( d => d.endsWith( '-pool-rw' ) || d.endsWith( '-pool-ro' ) )
|
|
523
|
+
if ( cnpgPoolers.length ) log( ` cnpg-operands poolers: ${cnpgPoolers.join( ', ' )}` )
|
|
524
|
+
|
|
525
|
+
try {
|
|
526
|
+
const r = await $`kubectl get ${CNPG_CRD} -n cnpg-operands --no-headers -o name`
|
|
527
|
+
cnpgClusters = r.stdout.trim().split( '\n' ).filter( Boolean )
|
|
528
|
+
.map( n => n.replace( /^[^/]+\//, '' ) )
|
|
529
|
+
if ( cnpgClusters.length ) log( ` cnpg-operands clusters: ${cnpgClusters.join( ', ' )}` )
|
|
530
|
+
} catch {
|
|
531
|
+
warning( 'Could not list CNPG clusters — operator may not be running' )
|
|
532
|
+
}
|
|
533
|
+
}
|
|
534
|
+
}
|
|
535
|
+
|
|
536
|
+
const traefikDeploys = await scanNs( 'traefik', [ 'traefik' ] )
|
|
537
|
+
const certDeploys = await scanNs( 'cert-manager',
|
|
538
|
+
[ 'cert-manager', 'cert-manager-cainjector', 'cert-manager-webhook' ] )
|
|
539
|
+
const esoDeploys = await scanNs( 'external-secrets',
|
|
540
|
+
[ 'external-secrets', 'external-secrets-cert-controller', 'external-secrets-webhook' ] )
|
|
541
|
+
const veleroDeploys = await scanNs( 'velero', [ 'velero' ] )
|
|
542
|
+
const estafetteDeploys = await scanNs( 'estafette', [ 'estafette-gke-preemptible-killer' ] )
|
|
543
|
+
const promDeploys = await scanNs( 'prometheus', [
|
|
544
|
+
'prometheus-stack-elasticsearch8-metrics',
|
|
545
|
+
'prometheus-stack-grafana',
|
|
546
|
+
'prometheus-stack-kube-prom-operator',
|
|
547
|
+
'prometheus-stack-kube-state-metrics',
|
|
548
|
+
'prometheus-stack-stackdriver-metrics',
|
|
549
|
+
] )
|
|
550
|
+
const monitoringDeploys = await scanNs( 'monitoring', [
|
|
551
|
+
'grafana',
|
|
552
|
+
'prometheus-server',
|
|
553
|
+
'prometheus-kube-state-metrics',
|
|
554
|
+
'elasticsearch8-exporter',
|
|
555
|
+
'stackdriver-exporter-prometheus-stackdriver-exporter',
|
|
556
|
+
] )
|
|
557
|
+
|
|
558
|
+
// ── StatefulSet scans ─────────────────────────────────────────────────────
|
|
559
|
+
|
|
560
|
+
banner( 'Scanning StatefulSets' )
|
|
561
|
+
|
|
562
|
+
const defaultSts = new Set( await listStatefulSets( namespace ) )
|
|
563
|
+
const hasValkey = defaultSts.has( 'valkey-primary' )
|
|
564
|
+
const hasRedis = !hasValkey && defaultSts.has( 'redis-master' )
|
|
565
|
+
const hasPostgres = defaultSts.has( 'postgres-postgresql' )
|
|
566
|
+
const hasTsdbMaster = defaultSts.has( 'timescale-db-postgresql-master' )
|
|
567
|
+
const hasTsdbSlave = defaultSts.has( 'timescale-db-postgresql-slave' )
|
|
568
|
+
|
|
569
|
+
log( ` ${namespace}: ${[ ...defaultSts ].join( ', ' ) || '(none)'}` )
|
|
570
|
+
|
|
571
|
+
const tsdbSlaveReplicas = hasTsdbSlave ?
|
|
572
|
+
await getReplicaCount( 'statefulset', 'timescale-db-postgresql-slave', namespace ) : 2
|
|
573
|
+
|
|
574
|
+
const hasValkeyReplicas = defaultSts.has( 'valkey-replicas' )
|
|
575
|
+
const valkeyReplicas = hasValkeyReplicas ?
|
|
576
|
+
await getReplicaCount( 'statefulset', 'valkey-replicas', namespace ) : 1
|
|
577
|
+
|
|
578
|
+
const elasticSts = await listStatefulSets( 'elastic' )
|
|
579
|
+
const esKnown = [ 'elasticsearch8-coordinating', 'elasticsearch8-ingest',
|
|
580
|
+
'elasticsearch8-data', 'elasticsearch8-master' ]
|
|
581
|
+
const esMembers = esKnown.filter( s => elasticSts.includes( s ) )
|
|
582
|
+
const esDataMembers = esMembers.filter( s => s.includes( 'data' ) )
|
|
583
|
+
const esOtherMembers = esMembers.filter( s => !s.includes( 'data' ) )
|
|
584
|
+
const esDataReplicas = esDataMembers.length ?
|
|
585
|
+
await getReplicaCount( 'statefulset', esDataMembers[0], 'elastic' ) : 3
|
|
586
|
+
const esOtherReplicas = esOtherMembers.length ?
|
|
587
|
+
await getReplicaCount( 'statefulset', esOtherMembers[0], 'elastic' ) : 3
|
|
588
|
+
if ( esMembers.length ) {
|
|
589
|
+
log( ` elastic: ${esDataMembers.join( ', ' )} (×${esDataReplicas}), ${esOtherMembers.join( ', ' )} (×${esOtherReplicas})` )
|
|
590
|
+
}
|
|
591
|
+
|
|
592
|
+
const promSts = new Set( await listStatefulSets( 'prometheus' ) )
|
|
593
|
+
const promStsMembers = [
|
|
594
|
+
'prometheus-prometheus-stack-kube-prom-prometheus',
|
|
595
|
+
'alertmanager-prometheus-stack-kube-prom-alertmanager',
|
|
596
|
+
].filter( s => promSts.has( s ) )
|
|
597
|
+
if ( promStsMembers.length ) log( ` prometheus: ${promStsMembers.join( ', ' )}` )
|
|
598
|
+
|
|
599
|
+
const monitoringSts = new Set( await listStatefulSets( 'monitoring' ) )
|
|
600
|
+
const monitoringStsMembers = [ 'prometheus-alertmanager' ]
|
|
601
|
+
.filter( s => monitoringSts.has( s ) )
|
|
602
|
+
if ( monitoringStsMembers.length ) log( ` monitoring: ${monitoringStsMembers.join( ', ' )}` )
|
|
603
|
+
|
|
604
|
+
// ── Node pools ────────────────────────────────────────────────────────────
|
|
605
|
+
|
|
606
|
+
banner( 'Listing node pools' )
|
|
607
|
+
const poolsRaw = JSON.parse(
|
|
608
|
+
( await $`gcloud container node-pools list --cluster=${cluster.name} --project=${cluster.project} --region=${cluster.region} --format=json` ).stdout
|
|
609
|
+
)
|
|
610
|
+
log( ` Found ${poolsRaw.length} node pool(s)` )
|
|
611
|
+
|
|
612
|
+
const pools = poolsRaw.map( ( pool ) => {
|
|
613
|
+
const auto = pool.autoscaling
|
|
614
|
+
const isPreemptible = pool.config?.spot || pool.config?.preemptible ||
|
|
615
|
+
pool.name.includes( 'preemptible' ) || pool.name.includes( 'spot' )
|
|
616
|
+
const hasAutoscalerMin = auto?.enabled && ( auto.minNodeCount ?? 0 ) > 0
|
|
617
|
+
|
|
618
|
+
let restoreSize
|
|
619
|
+
if ( isPreemptible ) restoreSize = 0
|
|
620
|
+
else if ( hasAutoscalerMin ) restoreSize = auto.minNodeCount
|
|
621
|
+
else restoreSize = pool.initialNodeCount ?? 1
|
|
622
|
+
|
|
623
|
+
const entry = { name: pool.name, restoreSize }
|
|
624
|
+
if ( hasAutoscalerMin ) entry.autoscaler = { min: auto.minNodeCount, max: auto.maxNodeCount }
|
|
625
|
+
return entry
|
|
626
|
+
} )
|
|
627
|
+
|
|
628
|
+
// ── Build phase lists ─────────────────────────────────────────────────────
|
|
629
|
+
|
|
630
|
+
const deployPhases = [
|
|
631
|
+
project.length && {
|
|
632
|
+
header : 'PROJECT / PRODUCT',
|
|
633
|
+
comment : 'Stopped first — depend on the platform but nothing depends on them.',
|
|
634
|
+
name : 'Project',
|
|
635
|
+
pauseMs : 15_000,
|
|
636
|
+
members : project,
|
|
637
|
+
},
|
|
638
|
+
auxiliary.length && {
|
|
639
|
+
header : 'AUXILIARY',
|
|
640
|
+
name : 'Auxiliary',
|
|
641
|
+
optional : true,
|
|
642
|
+
members : auxiliary,
|
|
643
|
+
},
|
|
644
|
+
transponders.length && {
|
|
645
|
+
header : 'PLATFORM — transponders',
|
|
646
|
+
comment : 'Data pipeline — last platform tier deployed, so first to hibernate.',
|
|
647
|
+
name : 'Platform — transponders',
|
|
648
|
+
members : transponders,
|
|
649
|
+
},
|
|
650
|
+
platformSupport.length && {
|
|
651
|
+
header : 'PLATFORM — support',
|
|
652
|
+
name : 'Platform — support',
|
|
653
|
+
members : platformSupport,
|
|
654
|
+
},
|
|
655
|
+
platformCore.length && {
|
|
656
|
+
header : 'PLATFORM — core',
|
|
657
|
+
comment : 'First platform tier deployed — last among deployments to hibernate.',
|
|
658
|
+
name : 'Platform — core',
|
|
659
|
+
members : platformCore,
|
|
660
|
+
},
|
|
661
|
+
systemDeploy.length && {
|
|
662
|
+
header : 'SYSTEM — deployments',
|
|
663
|
+
name : 'System — deployments',
|
|
664
|
+
members : systemDeploy,
|
|
665
|
+
},
|
|
666
|
+
cnpgPoolers.length && {
|
|
667
|
+
header : 'SYSTEM — CNPG poolers',
|
|
668
|
+
comment : 'Must come down before CNPG databases are hibernated.',
|
|
669
|
+
name : 'System — CNPG poolers',
|
|
670
|
+
namespace : 'cnpg-operands',
|
|
671
|
+
optional : true,
|
|
672
|
+
requiresCnpg : true,
|
|
673
|
+
wakeOrder : 40,
|
|
674
|
+
members : cnpgPoolers,
|
|
675
|
+
},
|
|
676
|
+
cnpgOperatorMembers.length && {
|
|
677
|
+
header : 'SYSTEM — CNPG operator',
|
|
678
|
+
comment : 'Must wake before CNPG clusters (webhook dependency).',
|
|
679
|
+
name : 'System — CNPG operator',
|
|
680
|
+
namespace : 'cnpg-system',
|
|
681
|
+
wakeOrder : 30,
|
|
682
|
+
deferShutdown : true,
|
|
683
|
+
waitForReady : true,
|
|
684
|
+
members : cnpgOperatorMembers,
|
|
685
|
+
},
|
|
686
|
+
veleroDeploys.length && {
|
|
687
|
+
header : 'SYSTEM — velero',
|
|
688
|
+
name : 'System — velero',
|
|
689
|
+
namespace : 'velero',
|
|
690
|
+
optional : true,
|
|
691
|
+
members : veleroDeploys,
|
|
692
|
+
},
|
|
693
|
+
estafetteDeploys.length && {
|
|
694
|
+
header : 'SYSTEM — estafette',
|
|
695
|
+
name : 'System — estafette',
|
|
696
|
+
namespace : 'estafette',
|
|
697
|
+
optional : true,
|
|
698
|
+
members : estafetteDeploys,
|
|
699
|
+
},
|
|
700
|
+
monitoringDeploys.length && {
|
|
701
|
+
header : 'SYSTEM — monitoring (legacy)',
|
|
702
|
+
comment : 'Legacy prometheus-community/prometheus + grafana/grafana in monitoring ns.',
|
|
703
|
+
name : 'System — monitoring (legacy)',
|
|
704
|
+
namespace : 'monitoring',
|
|
705
|
+
optional : true,
|
|
706
|
+
members : monitoringDeploys,
|
|
707
|
+
},
|
|
708
|
+
promDeploys.length && {
|
|
709
|
+
header : 'SYSTEM — prometheus',
|
|
710
|
+
name : 'System — prometheus',
|
|
711
|
+
namespace : 'prometheus',
|
|
712
|
+
optional : true,
|
|
713
|
+
members : promDeploys,
|
|
714
|
+
},
|
|
715
|
+
traefikDeploys.length && {
|
|
716
|
+
header : 'SYSTEM — traefik',
|
|
717
|
+
name : 'System — traefik',
|
|
718
|
+
namespace : 'traefik',
|
|
719
|
+
optional : true,
|
|
720
|
+
members : traefikDeploys,
|
|
721
|
+
},
|
|
722
|
+
certDeploys.length && {
|
|
723
|
+
header : 'SYSTEM — cert-manager',
|
|
724
|
+
name : 'System — cert-manager',
|
|
725
|
+
namespace : 'cert-manager',
|
|
726
|
+
optional : true,
|
|
727
|
+
wakeOrder : 20,
|
|
728
|
+
members : certDeploys,
|
|
729
|
+
},
|
|
730
|
+
esoDeploys.length && {
|
|
731
|
+
header : 'SYSTEM — external-secrets',
|
|
732
|
+
name : 'System — external-secrets',
|
|
733
|
+
namespace : 'external-secrets',
|
|
734
|
+
optional : true,
|
|
735
|
+
wakeOrder : 20,
|
|
736
|
+
members : esoDeploys,
|
|
737
|
+
},
|
|
738
|
+
].filter( Boolean )
|
|
739
|
+
|
|
740
|
+
const stsPhases = [
|
|
741
|
+
...cnpgClusters.map( ( name, i ) => ( {
|
|
742
|
+
header : i === 0 ? 'CNPG clusters' : undefined,
|
|
743
|
+
name : `CNPG — ${name.replace( /^cnpg-db-/, '' )}`,
|
|
744
|
+
type : 'cnpg',
|
|
745
|
+
namespace : 'cnpg-operands',
|
|
746
|
+
optional : true,
|
|
747
|
+
wakeOrder : 35,
|
|
748
|
+
members : [ name ],
|
|
749
|
+
} ) ),
|
|
750
|
+
hasPostgres && {
|
|
751
|
+
header : 'Legacy postgres',
|
|
752
|
+
name : 'System — postgres',
|
|
753
|
+
wakeOrder : 10,
|
|
754
|
+
members : [ 'postgres-postgresql' ],
|
|
755
|
+
},
|
|
756
|
+
hasTsdbMaster && {
|
|
757
|
+
header : 'TimescaleDB',
|
|
758
|
+
name : 'System — timescale-db (master)',
|
|
759
|
+
wakeOrder : 10,
|
|
760
|
+
members : [ 'timescale-db-postgresql-master' ],
|
|
761
|
+
},
|
|
762
|
+
hasTsdbSlave && {
|
|
763
|
+
name : 'System — timescale-db (slave)',
|
|
764
|
+
wakeOrder : 10,
|
|
765
|
+
restoreReplicas : tsdbSlaveReplicas,
|
|
766
|
+
members : [ 'timescale-db-postgresql-slave' ],
|
|
767
|
+
},
|
|
768
|
+
hasValkey && {
|
|
769
|
+
header : 'Valkey',
|
|
770
|
+
name : 'System — valkey',
|
|
771
|
+
wakeOrder : 10,
|
|
772
|
+
restoreReplicas : valkeyReplicas !== 1 ? valkeyReplicas : undefined,
|
|
773
|
+
members : [ 'valkey-primary', ...( hasValkeyReplicas ? [ 'valkey-replicas' ] : [] ) ],
|
|
774
|
+
},
|
|
775
|
+
hasRedis && {
|
|
776
|
+
header : 'Redis',
|
|
777
|
+
name : 'System — redis',
|
|
778
|
+
wakeOrder : 10,
|
|
779
|
+
members : [
|
|
780
|
+
'redis-master',
|
|
781
|
+
...( defaultSts.has( 'redis-replicas' ) ? [ 'redis-replicas' ] : [] ),
|
|
782
|
+
],
|
|
783
|
+
},
|
|
784
|
+
promStsMembers.length && {
|
|
785
|
+
header : 'Prometheus',
|
|
786
|
+
name : 'System — prometheus',
|
|
787
|
+
namespace : 'prometheus',
|
|
788
|
+
optional : true,
|
|
789
|
+
members : promStsMembers,
|
|
790
|
+
},
|
|
791
|
+
monitoringStsMembers.length && {
|
|
792
|
+
header : 'Monitoring (legacy)',
|
|
793
|
+
name : 'System — monitoring (legacy)',
|
|
794
|
+
namespace : 'monitoring',
|
|
795
|
+
optional : true,
|
|
796
|
+
members : monitoringStsMembers,
|
|
797
|
+
},
|
|
798
|
+
esDataMembers.length && {
|
|
799
|
+
header : 'Elasticsearch',
|
|
800
|
+
comment : 'Bitnami chart — no operator, plain kubectl scale.',
|
|
801
|
+
name : 'System — elasticsearch (data)',
|
|
802
|
+
namespace : 'elastic',
|
|
803
|
+
wakeOrder : 10,
|
|
804
|
+
restoreReplicas : esDataReplicas,
|
|
805
|
+
members : esDataMembers,
|
|
806
|
+
},
|
|
807
|
+
esOtherMembers.length && {
|
|
808
|
+
name : 'System — elasticsearch',
|
|
809
|
+
namespace : 'elastic',
|
|
810
|
+
wakeOrder : 10,
|
|
811
|
+
restoreReplicas : esOtherReplicas,
|
|
812
|
+
members : esOtherMembers,
|
|
813
|
+
},
|
|
814
|
+
].filter( Boolean )
|
|
815
|
+
|
|
816
|
+
// ── Write ─────────────────────────────────────────────────────────────────
|
|
817
|
+
|
|
818
|
+
const yaml = buildConfigYaml( {
|
|
819
|
+
clusterName : cluster.name,
|
|
820
|
+
namespace,
|
|
821
|
+
phasePauseMs : 10_000,
|
|
822
|
+
deployPhases,
|
|
823
|
+
stsPhases,
|
|
824
|
+
pools,
|
|
825
|
+
} )
|
|
826
|
+
|
|
827
|
+
// eslint-disable-next-line security/detect-non-literal-fs-filename
|
|
828
|
+
await fs.writeFile( outputPath, yaml, 'utf8' )
|
|
829
|
+
|
|
830
|
+
log( chalk.green( `\nWrote ${outputPath}` ) )
|
|
831
|
+
log( 'Review the file — especially the Project phase and StatefulSet replica counts — then commit.' )
|
|
152
832
|
}
|
|
153
833
|
|
|
154
|
-
// ── Commands
|
|
834
|
+
// ── Commands ───────────────────────────────────────────────────────────────
|
|
155
835
|
|
|
156
|
-
|
|
836
|
+
const hibernateCommand = new Command( 'hibernate' )
|
|
157
837
|
.description( 'Scale all workloads to zero and drain all node pools' )
|
|
158
|
-
.argument( '[config]', 'path to
|
|
159
|
-
.action( async
|
|
838
|
+
.argument( '[config]', 'path to k8s-cryo.yaml', './k8s-cryo.yaml' )
|
|
839
|
+
.action( async config => runHibernate( config ).catch( errorExit ) )
|
|
160
840
|
|
|
161
|
-
|
|
841
|
+
const wakeCommand = new Command( 'wake' )
|
|
162
842
|
.description( 'Restore node pools from hibernation (then run helmup)' )
|
|
163
|
-
.argument( '[config]', 'path to
|
|
164
|
-
.action( async
|
|
843
|
+
.argument( '[config]', 'path to k8s-cryo.yaml', './k8s-cryo.yaml' )
|
|
844
|
+
.action( async config => runWake( config ).catch( errorExit ) )
|
|
845
|
+
|
|
846
|
+
const initCommand = new Command( 'init' )
|
|
847
|
+
.description( 'Generate a starter k8s-cryo.yaml from live cluster state' )
|
|
848
|
+
.argument( '[output]', 'output file path', './k8s-cryo.yaml' )
|
|
849
|
+
.option( '-n, --namespace <ns>', 'namespace to scan for deployments', 'default' )
|
|
850
|
+
.option( '-f, --force', 'overwrite existing file' )
|
|
851
|
+
.action( async ( output, opts ) => runInit( output, opts ).catch( errorExit ) )
|
|
852
|
+
|
|
853
|
+
export const cryoCommand = new Command( 'cryo' )
|
|
854
|
+
.description( 'Hibernate, wake, and initialize cluster configurations' )
|
|
855
|
+
.addCommand( initCommand )
|
|
856
|
+
.addCommand( hibernateCommand )
|
|
857
|
+
.addCommand( wakeCommand )
|