@leverege/build-tools 2.114.0-DEVOP-578.2 → 2.114.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,49 +1,138 @@
1
+ // Hibernate and wake phases are inherently sequential — each kubectl/gcloud call
2
+ // must complete before the next begins to maintain dependency order. Every
3
+ // await-in-loop in this file is intentional.
4
+ /* eslint-disable no-await-in-loop */
5
+
6
+ import fs from 'node:fs/promises'
7
+
1
8
  import { Command } from 'commander'
2
9
  import chalk from 'chalk'
3
10
  import { $ } from 'zx'
4
11
 
5
- import { log, errorExit, sleep, warning, proceed, parseJsonFile } from '../../Utils.mjs'
12
+ import { log, errorExit, sleep, warning, proceed, parseYamlFile } from '../../Utils.mjs'
6
13
  import { overwhelmGate } from '../lib/gate.mjs'
14
+ import { listDeployments, namespaceExists, listStatefulSets, getReplicaCount } from '../lib/kubectl.mjs'
7
15
 
8
16
  $.verbose = false
9
17
 
10
- // ── Config ────────────────────────────────────────────────────────────────
18
+ // ── Known service categories (from helmup.sh, in hibernate order) ─────────
19
+ // Within each phase, services are listed in the order they should be stopped
20
+ // (reverse of bootstrap/deploy order within that phase).
11
21
 
12
- const loadConfig = async ( configPath ) => {
13
- const config = await parseJsonFile( configPath )
14
- if ( !config ) errorExit( `Config file not found: ${configPath}` )
22
+ const PLATFORM_TRANSPONDERS = [ 'transponder-tsdb', 'transponder-rt', 'transponder-bq', 'transponder-dh' ]
23
+ const PLATFORM_SUPPORT = [ 'rule-engine', 'resource-server', 'messenger', 'llm-prompt-server', 'imagine', 'emailer', 'db-curator' ]
24
+ const PLATFORM_CORE = [ 'scheduler', 'message-processor', 'api-server', 'authz-server' ]
25
+ const AUXILIARY_KNOWN = [ 'vms-server', 'vin-decoder-server', 'reason', 'pusher', 'push-notifier', 'pubsub-pulse', 'pgbouncer', 'postgres-pgbouncer', 'overdose', 'geotile-server', 'fota-server', 'argocd' ]
26
+ const SYSTEM_DEPLOY = [ 'reloader-reloader', 'reflector' ]
27
+
28
+ const ALL_KNOWN = new Set( [
29
+ ...PLATFORM_TRANSPONDERS, ...PLATFORM_SUPPORT, ...PLATFORM_CORE,
30
+ ...AUXILIARY_KNOWN, ...SYSTEM_DEPLOY,
31
+ ] )
15
32
 
16
- const { cluster, deploymentPhases, nodePools } = config
17
- if ( !cluster?.project || !cluster?.name || !cluster?.region ) {
18
- errorExit( 'Config must include cluster.project, cluster.name, and cluster.region' )
33
+ // ── Config ─────────────────────────────────────────────────────────────────
34
+
35
+ const loadConfig = async ( configPath ) => {
36
+ let config
37
+ try {
38
+ config = await parseYamlFile( configPath )
39
+ } catch {
40
+ errorExit( `Config file not found or invalid YAML: ${configPath}` )
19
41
  }
20
- if ( !deploymentPhases || !nodePools ) {
42
+ if ( !config?.deploymentPhases || !config?.nodePools ) {
21
43
  errorExit( 'Config must include deploymentPhases and nodePools' )
22
44
  }
23
45
  return config
24
46
  }
25
47
 
26
- // ── Display ───────────────────────────────────────────────────────────────
48
+ const overwhelmKey = async key => ( await $`overwhelm -k ${key}` ).stdout.trim()
49
+
50
+ const getClusterEnv = async () => ( {
51
+ project : await overwhelmKey( 'PROJECT_ID' ),
52
+ name : await overwhelmKey( 'CLUSTER_NAME' ),
53
+ region : await overwhelmKey( 'GCE_REGION' ),
54
+ } )
55
+
56
+ // ── Display ────────────────────────────────────────────────────────────────
27
57
 
28
58
  const banner = ( text ) => {
29
59
  const bar = '─'.repeat( Math.max( 0, 56 - text.length ) )
30
60
  log( chalk.cyan.bold( `\n── ${text} ${bar}` ) )
31
61
  }
32
62
 
33
- // ── kubectl ───────────────────────────────────────────────────────────────
63
+ // ── kubectl ────────────────────────────────────────────────────────────────
64
+
65
+ const scaleDeployments = async ( names, replicas, namespace ) => {
66
+ return $`kubectl scale deployment ${names} --replicas=${replicas} -n ${namespace}`
67
+ }
68
+
69
+ const scaleStatefulSets = async ( names, replicas, namespace ) => {
70
+ return $`kubectl scale statefulset ${names} --replicas=${replicas} -n ${namespace}`
71
+ }
34
72
 
35
- const scaleDeployments = async ( names, replicas, namespace ) =>
36
- $`kubectl scale deployment ${names} --replicas=${replicas} -n ${namespace}`
73
+ const CNPG_CRD = 'clusters.postgresql.cnpg.io'
37
74
 
38
- const scaleStatefulSets = async ( names, replicas, namespace ) =>
39
- $`kubectl scale statefulset ${names} --replicas=${replicas} -n ${namespace}`
75
+ const cnpgHibernate = async ( name, namespace ) => {
76
+ log( ` cnpg hibernate on: ${name} (${namespace})` )
77
+ await $`kubectl cnpg hibernate on -n ${namespace} ${name}`
78
+ }
40
79
 
41
- const waitForPodsGone = async ( namespace, timeoutSecs = 300 ) => {
42
- log( `\nWaiting for pods to terminate (up to ${timeoutSecs}s)...` )
80
+ const cnpgWake = async ( name, namespace, timeoutSecs = 300 ) => {
81
+ log( ` cnpg hibernate off: ${name} (${namespace})` )
43
82
  try {
44
- await $`kubectl wait pod --all -n ${namespace} --for=delete --timeout=${timeoutSecs}s`
83
+ await $`kubectl cnpg hibernate off -n ${namespace} ${name}`
84
+ } catch ( err ) {
85
+ if ( err.stderr?.includes( 'not found' ) ) {
86
+ log( ` ${name} — cluster not found, skipping` )
87
+ return
88
+ }
89
+ throw err
90
+ }
91
+ log( ` waiting for ${name} to be Ready (up to ${timeoutSecs}s)...` )
92
+ try {
93
+ await $`kubectl wait ${CNPG_CRD} ${name} -n ${namespace} --for=condition=Ready --timeout=${timeoutSecs}s`
45
94
  } catch {
46
- warning( `Some pods still terminating after ${timeoutSecs}s — proceeding.` )
95
+ warning( `${name} not Ready after ${timeoutSecs}s — proceeding, but poolers may fail to connect.` )
96
+ }
97
+ }
98
+
99
+ const anyCnpgClusters = async ( namespace, retries = 4, delayMs = 10_000 ) => {
100
+ for ( let i = 0; i <= retries; i++ ) {
101
+ try {
102
+ const result = await $`kubectl get ${CNPG_CRD} -n ${namespace} --no-headers -o name`
103
+ if ( result.stdout.trim().length > 0 ) return true
104
+ } catch {
105
+ // webhook may not be registered yet — retry
106
+ }
107
+ if ( i < retries ) {
108
+ log( ` CNPG webhook not ready, retrying in ${delayMs / 1000}s… (${i + 1}/${retries})` )
109
+ await sleep( delayMs )
110
+ }
111
+ }
112
+ return false
113
+ }
114
+
115
+ const filterExisting = async ( kind, names, namespace ) => {
116
+ try {
117
+ const result = await $`kubectl get ${kind} -n ${namespace} --no-headers -o name`
118
+ const existing = new Set(
119
+ result.stdout.trim().split( '\n' ).filter( Boolean )
120
+ .map( n => n.replace( /^[^/]+\//, '' ) )
121
+ )
122
+ return names.filter( n => existing.has( n ) )
123
+ } catch {
124
+ return []
125
+ }
126
+ }
127
+
128
+ const waitForPodsGone = async ( namespaces, timeoutSecs = 300 ) => {
129
+ log( `\nWaiting for pods to terminate in [${namespaces.join( ', ' )}] (up to ${timeoutSecs}s)...` )
130
+ for ( const ns of namespaces ) {
131
+ try {
132
+ await $`kubectl wait pod --all -n ${ns} --for=delete --timeout=${timeoutSecs}s`
133
+ } catch {
134
+ warning( `Some pods still terminating in ${ns} after ${timeoutSecs}s — proceeding.` )
135
+ }
47
136
  }
48
137
  }
49
138
 
@@ -52,26 +141,23 @@ const waitForNodesReady = async ( timeoutSecs = 600 ) => {
52
141
  await $`kubectl wait node --all --for=condition=Ready --timeout=${timeoutSecs}s`
53
142
  }
54
143
 
55
- // ── gcloud ────────────────────────────────────────────────────────────────
144
+ // ── gcloud ─────────────────────────────────────────────────────────────────
56
145
 
57
146
  const updateAutoscalerMin = async ( pool, cluster, min ) => {
58
147
  log( ` ${pool.name}: autoscaler min → ${min}` )
59
- const { name, project, region } = cluster
60
- await $`gcloud container node-pools update ${pool.name} --cluster=${name} --project=${project} --region=${region} --enable-autoscaling --min-nodes=${min} --max-nodes=${pool.autoscaler.max} --quiet`
148
+ await $`gcloud container node-pools update ${pool.name} --cluster=${cluster.name} --project=${cluster.project} --region=${cluster.region} --enable-autoscaling --min-nodes=${min} --max-nodes=${pool.autoscaler.max} --quiet`
61
149
  }
62
150
 
63
151
  const resizePool = async ( pool, cluster, size ) => {
64
152
  log( ` ${pool.name} → ${size} node(s)` )
65
- const { name, project, region } = cluster
66
- await $`gcloud container clusters resize ${name} --node-pool=${pool.name} --num-nodes=${size} --project=${project} --region=${region} --quiet`
153
+ await $`gcloud container clusters resize ${cluster.name} --node-pool=${pool.name} --num-nodes=${size} --project=${cluster.project} --region=${cluster.region} --quiet`
67
154
  }
68
155
 
69
- // ── Hibernate ─────────────────────────────────────────────────────────────
156
+ // ── Hibernate ──────────────────────────────────────────────────────────────
70
157
 
71
158
  const runHibernate = async ( configPath ) => {
72
159
  const config = await loadConfig( configPath )
73
160
  const {
74
- cluster,
75
161
  namespace = 'default',
76
162
  phasePauseMs = 10_000,
77
163
  deploymentPhases,
@@ -80,27 +166,104 @@ const runHibernate = async ( configPath ) => {
80
166
  } = config
81
167
 
82
168
  await overwhelmGate()
169
+ const cluster = await getClusterEnv()
170
+
83
171
  await proceed( `Hibernate ${chalk.yellow.bold( cluster.name )}? All workloads scale to 0 and node pools drain.` )
84
172
 
173
+ // deferShutdown phases (e.g. CNPG operator) are held back until after
174
+ // statefulset phases so their webhooks remain available for cnpg hibernate on.
175
+ const deferredDeployPhases = []
176
+
85
177
  for ( const group of deploymentPhases ) {
178
+ if ( group.deferShutdown ) {
179
+ deferredDeployPhases.push( group )
180
+ continue
181
+ }
182
+ const phaseNamespace = group.namespace ?? namespace
183
+ const members = group.optional ?
184
+ await filterExisting( 'deployment', group.members, phaseNamespace ) :
185
+ group.members
186
+ if ( !members.length ) {
187
+ log( ` ${group.name} — not present, skipping` )
188
+ continue
189
+ }
86
190
  banner( `Deployments: ${group.name}` )
87
- await scaleDeployments( group.members, 0, namespace )
191
+ await scaleDeployments( members, 0, phaseNamespace )
88
192
  await sleep( group.pauseMs ?? phasePauseMs )
89
193
  }
90
194
 
195
+ // Before statefulset phases, ensure any deferShutdown phases are actually
196
+ // running — a prior failed run may have left them at 0, which would cause
197
+ // CNPG webhook calls to fail. Wait for rollout (not just a fixed sleep)
198
+ // so the webhook is registered before cnpg hibernate on is called.
199
+ if ( deferredDeployPhases.length && statefulsetPhases.some( p => p.type === 'cnpg' ) ) {
200
+ banner( 'Ensuring deferred services are running' )
201
+ for ( const group of deferredDeployPhases ) {
202
+ const phaseNamespace = group.namespace ?? namespace
203
+ log( ` ${group.name} → 1` )
204
+ await scaleDeployments( group.members, 1, phaseNamespace )
205
+ if ( group.waitForReady ) {
206
+ log( ` waiting for ${group.name} to be ready...` )
207
+ for ( const member of group.members ) {
208
+ try {
209
+ await $`kubectl rollout status deployment ${member} -n ${phaseNamespace} --timeout=120s`
210
+ } catch {
211
+ warning( `${member} not ready after 120s — proceeding` )
212
+ }
213
+ }
214
+ }
215
+ }
216
+ }
217
+
91
218
  if ( statefulsetPhases.length ) {
92
219
  for ( const group of statefulsetPhases ) {
220
+ const phaseNamespace = group.namespace ?? namespace
221
+ const kind = group.type === 'cnpg' ? CNPG_CRD : 'statefulset'
222
+ const members = group.optional ?
223
+ await filterExisting( kind, group.members, phaseNamespace ) :
224
+ group.members
225
+ if ( !members.length ) {
226
+ log( ` ${group.name} — not present, skipping` )
227
+ continue
228
+ }
93
229
  banner( `StatefulSets: ${group.name}` )
94
- await scaleStatefulSets( group.members, 0, namespace )
230
+ if ( group.type === 'cnpg' ) {
231
+ for ( const member of members ) {
232
+ await cnpgHibernate( member, phaseNamespace )
233
+ }
234
+ } else {
235
+ await scaleStatefulSets( members, 0, phaseNamespace )
236
+ }
95
237
  await sleep( group.pauseMs ?? phasePauseMs )
96
238
  }
97
239
  } else {
98
- // no ordered phases defined yet — scale all StatefulSets together
99
- banner( 'StatefulSets' )
100
- await $`kubectl scale statefulset --all -n ${namespace} --replicas=0`
240
+ warning( 'statefulsetPhases is empty — StatefulSet shutdown skipped.' )
241
+ warning( 'Define phases in cluster-config.yaml when ready to hibernate database workloads.' )
242
+ warning( 'Note: CNPG clusters require kubectl cnpg hibernate, not kubectl scale.' )
101
243
  }
102
244
 
103
- await waitForPodsGone( namespace )
245
+ if ( deferredDeployPhases.length ) {
246
+ for ( const group of deferredDeployPhases ) {
247
+ const phaseNamespace = group.namespace ?? namespace
248
+ const members = group.optional ?
249
+ await filterExisting( 'deployment', group.members, phaseNamespace ) :
250
+ group.members
251
+ if ( !members.length ) {
252
+ log( ` ${group.name} — not present, skipping` )
253
+ continue
254
+ }
255
+ banner( `Deployments: ${group.name}` )
256
+ await scaleDeployments( members, 0, phaseNamespace )
257
+ await sleep( group.pauseMs ?? phasePauseMs )
258
+ }
259
+ }
260
+
261
+ const namespacesToDrain = [ ...new Set( [
262
+ namespace,
263
+ ...deploymentPhases.map( g => g.namespace ?? namespace ),
264
+ ...statefulsetPhases.map( g => g.namespace ?? namespace ),
265
+ ] ) ]
266
+ await waitForPodsGone( namespacesToDrain )
104
267
 
105
268
  const poolsWithAutoscaler = nodePools.filter( p => p.autoscaler )
106
269
  if ( poolsWithAutoscaler.length ) {
@@ -116,19 +279,23 @@ const runHibernate = async ( configPath ) => {
116
279
  }
117
280
 
118
281
  log( chalk.green( '\nCluster hibernated. Control plane (~$0.10/hr) continues running.' ) )
119
- log( `To wake: ${chalk.cyan( `k8s wake ${configPath}` )}` )
282
+ log( `To wake: ${chalk.cyan( 'k8s cryo wake' )}` )
120
283
  }
121
284
 
122
- // ── Wake ──────────────────────────────────────────────────────────────────
285
+ // ── Wake ───────────────────────────────────────────────────────────────────
123
286
 
124
287
  const runWake = async ( configPath ) => {
125
288
  const config = await loadConfig( configPath )
126
- const { cluster, nodePools } = config
289
+ const {
290
+ namespace = 'default',
291
+ nodePools,
292
+ deploymentPhases = [],
293
+ statefulsetPhases = [],
294
+ } = config
127
295
 
128
296
  await overwhelmGate()
297
+ const cluster = await getClusterEnv()
129
298
 
130
- // Restore autoscaler minimums before resizing so the autoscaler
131
- // doesn't immediately scale back down the nodes we're bringing up.
132
299
  const poolsWithAutoscaler = nodePools.filter( p => p.autoscaler )
133
300
  if ( poolsWithAutoscaler.length ) {
134
301
  banner( 'Restoring autoscaler minimums' )
@@ -139,26 +306,552 @@ const runWake = async ( configPath ) => {
139
306
 
140
307
  banner( 'Restoring node pools' )
141
308
  for ( const pool of nodePools ) {
142
- if ( pool.restoreSize > 0 ) {
143
- await resizePool( pool, cluster, pool.restoreSize )
144
- } else {
309
+ if ( pool.restoreSize <= 0 ) {
145
310
  log( ` ${pool.name} → skipped (autoscaler brings up on demand)` )
311
+ continue
146
312
  }
313
+ // Autoscaled pools start at 1 node per zone — enough for initial scheduling.
314
+ // The autoscaler brings up additional nodes as workloads are placed.
315
+ const initialSize = pool.autoscaler ? 1 : pool.restoreSize
316
+ await resizePool( pool, cluster, initialSize )
147
317
  }
148
318
 
149
319
  await waitForNodesReady()
150
320
 
151
- log( chalk.green( '\nNodes ready. Run helmup to restore all services.' ) )
321
+ // Build a unified wake list from both phase arrays, tagged with kind.
322
+ // Phases at the same wakeOrder run in parallel; levels are processed in order.
323
+ // Helm won't restore manually-scaled-to-0 resources on its own — it only
324
+ // patches fields that changed between old and new manifest — so cryo must
325
+ // restore replica counts explicitly before helmup reconciles.
326
+ const wakeList = [
327
+ ...deploymentPhases.map( p => ( { ...p, phaseKind: 'deployment' } ) ),
328
+ ...statefulsetPhases.map( p => ( { ...p, phaseKind: 'statefulset' } ) ),
329
+ ]
330
+
331
+ const wakeMap = new Map()
332
+ for ( const group of wakeList ) {
333
+ const order = group.wakeOrder ?? Infinity
334
+ if ( !wakeMap.has( order ) ) wakeMap.set( order, [] )
335
+ wakeMap.get( order ).push( group )
336
+ }
337
+
338
+ const wakeGroup = async ( group ) => {
339
+ const phaseNamespace = group.namespace ?? namespace
340
+
341
+ if ( group.phaseKind === 'statefulset' && group.type === 'cnpg' ) {
342
+ // Attempt hibernate off unconditionally — cnpgWake handles not-found gracefully.
343
+ // filterExisting is unreliable here: hibernated clusters are invisible when the
344
+ // operator is down, and the operator may not have been running during a prior failed run.
345
+ for ( const member of group.members ) {
346
+ await cnpgWake( member, phaseNamespace )
347
+ }
348
+ return
349
+ }
350
+
351
+ if ( group.requiresCnpg && !await anyCnpgClusters( phaseNamespace ) ) {
352
+ log( ` ${group.name} — no CNPG clusters in ${phaseNamespace}, skipping` )
353
+ return
354
+ }
355
+
356
+ const kind = group.phaseKind === 'statefulset' ? 'statefulset' : 'deployment'
357
+ const members = group.optional ?
358
+ await filterExisting( kind, group.members, phaseNamespace ) :
359
+ group.members
360
+ if ( !members.length ) {
361
+ log( ` ${group.name} — not present, skipping` )
362
+ return
363
+ }
364
+ const replicas = group.restoreReplicas ?? 1
365
+ log( ` ${group.name} → ${replicas}` )
366
+ if ( group.phaseKind === 'statefulset' ) {
367
+ await scaleStatefulSets( members, replicas, phaseNamespace )
368
+ } else {
369
+ await scaleDeployments( members, replicas, phaseNamespace )
370
+ }
371
+ if ( group.waitForReady ) {
372
+ log( ` waiting for ${group.name} to be ready...` )
373
+ for ( const member of members ) {
374
+ try {
375
+ await $`kubectl rollout status deployment ${member} -n ${phaseNamespace} --timeout=120s`
376
+ } catch {
377
+ warning( `${member} not ready after 120s — proceeding` )
378
+ }
379
+ }
380
+ }
381
+ }
382
+
383
+ banner( 'Restoring workloads' )
384
+ const levels = [ ...wakeMap.keys() ].sort( ( a, b ) => a - b )
385
+ for ( const level of levels ) {
386
+ await Promise.all( wakeMap.get( level ).map( wakeGroup ) )
387
+ }
388
+
389
+ log( chalk.green( '\nCluster restored. Run helmup to reconcile any config or version changes.' ) )
390
+ }
391
+
392
+ // ── Init: YAML generation ──────────────────────────────────────────────────
393
+
394
+ const buildConfigYaml = ( { clusterName, namespace, phasePauseMs, deployPhases, stsPhases, pools } ) => {
395
+ const bar = label => ` # ── ${label} ${'─'.repeat( Math.max( 0, 57 - label.length ) )}`
396
+
397
+ const serializePhase = ( phase ) => {
398
+ const ns = phase.namespace && phase.namespace !== namespace ?
399
+ `\n namespace: ${phase.namespace}` : ''
400
+ const type = phase.type ? `\n type: ${phase.type}` : ''
401
+ const opt = phase.optional ? '\n optional: true' : ''
402
+ const rcnpg = phase.requiresCnpg ? '\n requiresCnpg: true' : ''
403
+ const wo = phase.wakeOrder != null ? `\n wakeOrder: ${phase.wakeOrder}` : ''
404
+ const defer = phase.deferShutdown ? '\n deferShutdown: true' : ''
405
+ const waitRdy = phase.waitForReady ? '\n waitForReady: true' : ''
406
+ const restore = phase.restoreReplicas != null && phase.restoreReplicas !== 1 ?
407
+ `\n restoreReplicas: ${phase.restoreReplicas}` : ''
408
+ const pauseMs = phase.pauseMs ? `\n pauseMs: ${phase.pauseMs}` : ''
409
+ const members = phase.members.map( m => ` - ${m}` ).join( '\n' )
410
+ return ` - name: "${phase.name}"${type}${ns}${opt}${rcnpg}${wo}${defer}${waitRdy}${restore}${pauseMs}\n members:\n${members}`
411
+ }
412
+
413
+ const phaseBlock = ( phase ) => {
414
+ const header = phase.header ? `\n${bar( phase.header )}\n` : ''
415
+ const comment = phase.comment ? ` # ${phase.comment}\n` : ''
416
+ return `${header}${comment}${serializePhase( phase )}`
417
+ }
418
+
419
+ const deployBlocks = deployPhases.filter( p => p.members?.length ).map( phaseBlock ).join( '\n' )
420
+
421
+ const stsContent = stsPhases.filter( p => p.members?.length )
422
+ const stsBlocks = stsContent.map( phaseBlock ).join( '\n' )
423
+ const hasCnpg = stsContent.some( p => p.type === 'cnpg' )
424
+
425
+ const poolBlocks = pools.map( ( pool ) => {
426
+ const auto = pool.autoscaler ?
427
+ `\n autoscaler:\n min: ${pool.autoscaler.min}\n max: ${pool.autoscaler.max}` : ''
428
+ return ` - name: ${pool.name}\n restoreSize: ${pool.restoreSize}${auto}`
429
+ } ).join( '\n' )
430
+
431
+ const cnpgNote = hasCnpg ?
432
+ '\n# CNPG clusters use type: cnpg → kubectl cnpg hibernate on/off (not kubectl scale).' : ''
433
+
434
+ return `# k8s-cryo.yaml — cluster hibernation / wake configuration for ${clusterName}
435
+ # Cluster identity (project, name, region) is read from overwhelm at runtime.
436
+ # Generated by: k8s cryo init — review before committing.
437
+ #
438
+ # Phases execute top → bottom during hibernate.
439
+ # wakeOrder controls the wake sequence independently of hibernate order.
440
+
441
+ namespace: ${namespace}
442
+ phasePauseMs: ${phasePauseMs}
443
+
444
+ deploymentPhases:
445
+ ${deployBlocks}
446
+
447
+ # StatefulSet shutdown phases — executed after all deployments are down.${cnpgNote}
448
+ ${stsContent.length ? `statefulsetPhases:\n${stsBlocks}` : 'statefulsetPhases: [] # TODO: add StatefulSet phases before hibernating'}
449
+
450
+ # Node pools — restoreSize is nodes per zone (regional cluster × 3 zones).
451
+ # Pools with restoreSize: 0 are managed on demand by the cluster autoscaler.
452
+ nodePools:
453
+ ${poolBlocks}
454
+ `
455
+ }
456
+
457
+ // ── Init ───────────────────────────────────────────────────────────────────
458
+
459
+ const runInit = async ( outputPath, opts ) => {
460
+ const { namespace, force } = opts
461
+
462
+ if ( !force ) {
463
+ try {
464
+ await fs.access( outputPath )
465
+ errorExit( `${outputPath} already exists — use --force to overwrite` )
466
+ } catch ( err ) {
467
+ if ( err.code !== 'ENOENT' ) errorExit( `Cannot access ${outputPath}: ${err.message}` )
468
+ }
469
+ }
470
+
471
+ await overwhelmGate()
472
+ const cluster = await getClusterEnv()
473
+
474
+ log( `\nScanning ${chalk.cyan( cluster.name )} (${cluster.project})…` )
475
+
476
+ // ── Default namespace deployments ─────────────────────────────────────────
477
+
478
+ banner( 'Listing deployments' )
479
+ const allDeploys = await listDeployments( namespace )
480
+ log( ` Found ${allDeploys.length} deployment(s) in ${namespace}` )
481
+
482
+ const deployed = new Set( allDeploys )
483
+ const transponders = PLATFORM_TRANSPONDERS.filter( d => deployed.has( d ) )
484
+ const platformSupport = PLATFORM_SUPPORT.filter( d => deployed.has( d ) )
485
+ const platformCore = PLATFORM_CORE.filter( d => deployed.has( d ) )
486
+ const auxiliary = AUXILIARY_KNOWN.filter( d => deployed.has( d ) )
487
+ const systemDeploy = SYSTEM_DEPLOY.filter( d => deployed.has( d ) )
488
+ const project = allDeploys.filter( d => !ALL_KNOWN.has( d ) )
489
+
490
+ if ( project.length ) {
491
+ log( ` → Project / product (review placement): ${project.join( ', ' )}` )
492
+ }
493
+
494
+ // ── System namespace scans ────────────────────────────────────────────────
495
+
496
+ banner( 'Scanning system namespaces' )
497
+
498
+ // Returns deployments present in ns from candidates; skips if namespace absent.
499
+ const scanNs = async ( ns, candidates ) => {
500
+ if ( !await namespaceExists( ns ) ) return []
501
+ const found = await filterExisting( 'deployment', candidates, ns )
502
+ log( found.length ?
503
+ ` ${ns}: ${found.join( ', ' )}` :
504
+ ` ${ns}: (empty)` )
505
+ return found
506
+ }
507
+
508
+ // CNPG — operator, poolers, cluster CRDs
509
+ const cnpgSystemExists = await namespaceExists( 'cnpg-system' )
510
+ let cnpgOperatorMembers = []
511
+ let cnpgPoolers = []
512
+ let cnpgClusters = []
513
+
514
+ if ( cnpgSystemExists ) {
515
+ cnpgOperatorMembers = await filterExisting(
516
+ 'deployment', [ 'barman-cloud', 'cnpg-controller-manager' ], 'cnpg-system'
517
+ )
518
+ log( ` cnpg-system: ${cnpgOperatorMembers.join( ', ' ) || '(none)'}` )
519
+
520
+ if ( await namespaceExists( 'cnpg-operands' ) ) {
521
+ const operandDeploys = await listDeployments( 'cnpg-operands' )
522
+ cnpgPoolers = operandDeploys.filter( d => d.endsWith( '-pool-rw' ) || d.endsWith( '-pool-ro' ) )
523
+ if ( cnpgPoolers.length ) log( ` cnpg-operands poolers: ${cnpgPoolers.join( ', ' )}` )
524
+
525
+ try {
526
+ const r = await $`kubectl get ${CNPG_CRD} -n cnpg-operands --no-headers -o name`
527
+ cnpgClusters = r.stdout.trim().split( '\n' ).filter( Boolean )
528
+ .map( n => n.replace( /^[^/]+\//, '' ) )
529
+ if ( cnpgClusters.length ) log( ` cnpg-operands clusters: ${cnpgClusters.join( ', ' )}` )
530
+ } catch {
531
+ warning( 'Could not list CNPG clusters — operator may not be running' )
532
+ }
533
+ }
534
+ }
535
+
536
+ const traefikDeploys = await scanNs( 'traefik', [ 'traefik' ] )
537
+ const certDeploys = await scanNs( 'cert-manager',
538
+ [ 'cert-manager', 'cert-manager-cainjector', 'cert-manager-webhook' ] )
539
+ const esoDeploys = await scanNs( 'external-secrets',
540
+ [ 'external-secrets', 'external-secrets-cert-controller', 'external-secrets-webhook' ] )
541
+ const veleroDeploys = await scanNs( 'velero', [ 'velero' ] )
542
+ const estafetteDeploys = await scanNs( 'estafette', [ 'estafette-gke-preemptible-killer' ] )
543
+ const promDeploys = await scanNs( 'prometheus', [
544
+ 'prometheus-stack-elasticsearch8-metrics',
545
+ 'prometheus-stack-grafana',
546
+ 'prometheus-stack-kube-prom-operator',
547
+ 'prometheus-stack-kube-state-metrics',
548
+ 'prometheus-stack-stackdriver-metrics',
549
+ ] )
550
+ const monitoringDeploys = await scanNs( 'monitoring', [
551
+ 'grafana',
552
+ 'prometheus-server',
553
+ 'prometheus-kube-state-metrics',
554
+ 'elasticsearch8-exporter',
555
+ 'stackdriver-exporter-prometheus-stackdriver-exporter',
556
+ ] )
557
+
558
+ // ── StatefulSet scans ─────────────────────────────────────────────────────
559
+
560
+ banner( 'Scanning StatefulSets' )
561
+
562
+ const defaultSts = new Set( await listStatefulSets( namespace ) )
563
+ const hasValkey = defaultSts.has( 'valkey-primary' )
564
+ const hasRedis = !hasValkey && defaultSts.has( 'redis-master' )
565
+ const hasPostgres = defaultSts.has( 'postgres-postgresql' )
566
+ const hasTsdbMaster = defaultSts.has( 'timescale-db-postgresql-master' )
567
+ const hasTsdbSlave = defaultSts.has( 'timescale-db-postgresql-slave' )
568
+
569
+ log( ` ${namespace}: ${[ ...defaultSts ].join( ', ' ) || '(none)'}` )
570
+
571
+ const tsdbSlaveReplicas = hasTsdbSlave ?
572
+ await getReplicaCount( 'statefulset', 'timescale-db-postgresql-slave', namespace ) : 2
573
+
574
+ const hasValkeyReplicas = defaultSts.has( 'valkey-replicas' )
575
+ const valkeyReplicas = hasValkeyReplicas ?
576
+ await getReplicaCount( 'statefulset', 'valkey-replicas', namespace ) : 1
577
+
578
+ const elasticSts = await listStatefulSets( 'elastic' )
579
+ const esKnown = [ 'elasticsearch8-coordinating', 'elasticsearch8-ingest',
580
+ 'elasticsearch8-data', 'elasticsearch8-master' ]
581
+ const esMembers = esKnown.filter( s => elasticSts.includes( s ) )
582
+ const esDataMembers = esMembers.filter( s => s.includes( 'data' ) )
583
+ const esOtherMembers = esMembers.filter( s => !s.includes( 'data' ) )
584
+ const esDataReplicas = esDataMembers.length ?
585
+ await getReplicaCount( 'statefulset', esDataMembers[0], 'elastic' ) : 3
586
+ const esOtherReplicas = esOtherMembers.length ?
587
+ await getReplicaCount( 'statefulset', esOtherMembers[0], 'elastic' ) : 3
588
+ if ( esMembers.length ) {
589
+ log( ` elastic: ${esDataMembers.join( ', ' )} (×${esDataReplicas}), ${esOtherMembers.join( ', ' )} (×${esOtherReplicas})` )
590
+ }
591
+
592
+ const promSts = new Set( await listStatefulSets( 'prometheus' ) )
593
+ const promStsMembers = [
594
+ 'prometheus-prometheus-stack-kube-prom-prometheus',
595
+ 'alertmanager-prometheus-stack-kube-prom-alertmanager',
596
+ ].filter( s => promSts.has( s ) )
597
+ if ( promStsMembers.length ) log( ` prometheus: ${promStsMembers.join( ', ' )}` )
598
+
599
+ const monitoringSts = new Set( await listStatefulSets( 'monitoring' ) )
600
+ const monitoringStsMembers = [ 'prometheus-alertmanager' ]
601
+ .filter( s => monitoringSts.has( s ) )
602
+ if ( monitoringStsMembers.length ) log( ` monitoring: ${monitoringStsMembers.join( ', ' )}` )
603
+
604
+ // ── Node pools ────────────────────────────────────────────────────────────
605
+
606
+ banner( 'Listing node pools' )
607
+ const poolsRaw = JSON.parse(
608
+ ( await $`gcloud container node-pools list --cluster=${cluster.name} --project=${cluster.project} --region=${cluster.region} --format=json` ).stdout
609
+ )
610
+ log( ` Found ${poolsRaw.length} node pool(s)` )
611
+
612
+ const pools = poolsRaw.map( ( pool ) => {
613
+ const auto = pool.autoscaling
614
+ const isPreemptible = pool.config?.spot || pool.config?.preemptible ||
615
+ pool.name.includes( 'preemptible' ) || pool.name.includes( 'spot' )
616
+ const hasAutoscalerMin = auto?.enabled && ( auto.minNodeCount ?? 0 ) > 0
617
+
618
+ let restoreSize
619
+ if ( isPreemptible ) restoreSize = 0
620
+ else if ( hasAutoscalerMin ) restoreSize = auto.minNodeCount
621
+ else restoreSize = pool.initialNodeCount ?? 1
622
+
623
+ const entry = { name: pool.name, restoreSize }
624
+ if ( hasAutoscalerMin ) entry.autoscaler = { min: auto.minNodeCount, max: auto.maxNodeCount }
625
+ return entry
626
+ } )
627
+
628
+ // ── Build phase lists ─────────────────────────────────────────────────────
629
+
630
+ const deployPhases = [
631
+ project.length && {
632
+ header : 'PROJECT / PRODUCT',
633
+ comment : 'Stopped first — depend on the platform but nothing depends on them.',
634
+ name : 'Project',
635
+ pauseMs : 15_000,
636
+ members : project,
637
+ },
638
+ auxiliary.length && {
639
+ header : 'AUXILIARY',
640
+ name : 'Auxiliary',
641
+ optional : true,
642
+ members : auxiliary,
643
+ },
644
+ transponders.length && {
645
+ header : 'PLATFORM — transponders',
646
+ comment : 'Data pipeline — last platform tier deployed, so first to hibernate.',
647
+ name : 'Platform — transponders',
648
+ members : transponders,
649
+ },
650
+ platformSupport.length && {
651
+ header : 'PLATFORM — support',
652
+ name : 'Platform — support',
653
+ members : platformSupport,
654
+ },
655
+ platformCore.length && {
656
+ header : 'PLATFORM — core',
657
+ comment : 'First platform tier deployed — last among deployments to hibernate.',
658
+ name : 'Platform — core',
659
+ members : platformCore,
660
+ },
661
+ systemDeploy.length && {
662
+ header : 'SYSTEM — deployments',
663
+ name : 'System — deployments',
664
+ members : systemDeploy,
665
+ },
666
+ cnpgPoolers.length && {
667
+ header : 'SYSTEM — CNPG poolers',
668
+ comment : 'Must come down before CNPG databases are hibernated.',
669
+ name : 'System — CNPG poolers',
670
+ namespace : 'cnpg-operands',
671
+ optional : true,
672
+ requiresCnpg : true,
673
+ wakeOrder : 40,
674
+ members : cnpgPoolers,
675
+ },
676
+ cnpgOperatorMembers.length && {
677
+ header : 'SYSTEM — CNPG operator',
678
+ comment : 'Must wake before CNPG clusters (webhook dependency).',
679
+ name : 'System — CNPG operator',
680
+ namespace : 'cnpg-system',
681
+ wakeOrder : 30,
682
+ deferShutdown : true,
683
+ waitForReady : true,
684
+ members : cnpgOperatorMembers,
685
+ },
686
+ veleroDeploys.length && {
687
+ header : 'SYSTEM — velero',
688
+ name : 'System — velero',
689
+ namespace : 'velero',
690
+ optional : true,
691
+ members : veleroDeploys,
692
+ },
693
+ estafetteDeploys.length && {
694
+ header : 'SYSTEM — estafette',
695
+ name : 'System — estafette',
696
+ namespace : 'estafette',
697
+ optional : true,
698
+ members : estafetteDeploys,
699
+ },
700
+ monitoringDeploys.length && {
701
+ header : 'SYSTEM — monitoring (legacy)',
702
+ comment : 'Legacy prometheus-community/prometheus + grafana/grafana in monitoring ns.',
703
+ name : 'System — monitoring (legacy)',
704
+ namespace : 'monitoring',
705
+ optional : true,
706
+ members : monitoringDeploys,
707
+ },
708
+ promDeploys.length && {
709
+ header : 'SYSTEM — prometheus',
710
+ name : 'System — prometheus',
711
+ namespace : 'prometheus',
712
+ optional : true,
713
+ members : promDeploys,
714
+ },
715
+ traefikDeploys.length && {
716
+ header : 'SYSTEM — traefik',
717
+ name : 'System — traefik',
718
+ namespace : 'traefik',
719
+ optional : true,
720
+ members : traefikDeploys,
721
+ },
722
+ certDeploys.length && {
723
+ header : 'SYSTEM — cert-manager',
724
+ name : 'System — cert-manager',
725
+ namespace : 'cert-manager',
726
+ optional : true,
727
+ wakeOrder : 20,
728
+ members : certDeploys,
729
+ },
730
+ esoDeploys.length && {
731
+ header : 'SYSTEM — external-secrets',
732
+ name : 'System — external-secrets',
733
+ namespace : 'external-secrets',
734
+ optional : true,
735
+ wakeOrder : 20,
736
+ members : esoDeploys,
737
+ },
738
+ ].filter( Boolean )
739
+
740
+ const stsPhases = [
741
+ ...cnpgClusters.map( ( name, i ) => ( {
742
+ header : i === 0 ? 'CNPG clusters' : undefined,
743
+ name : `CNPG — ${name.replace( /^cnpg-db-/, '' )}`,
744
+ type : 'cnpg',
745
+ namespace : 'cnpg-operands',
746
+ optional : true,
747
+ wakeOrder : 35,
748
+ members : [ name ],
749
+ } ) ),
750
+ hasPostgres && {
751
+ header : 'Legacy postgres',
752
+ name : 'System — postgres',
753
+ wakeOrder : 10,
754
+ members : [ 'postgres-postgresql' ],
755
+ },
756
+ hasTsdbMaster && {
757
+ header : 'TimescaleDB',
758
+ name : 'System — timescale-db (master)',
759
+ wakeOrder : 10,
760
+ members : [ 'timescale-db-postgresql-master' ],
761
+ },
762
+ hasTsdbSlave && {
763
+ name : 'System — timescale-db (slave)',
764
+ wakeOrder : 10,
765
+ restoreReplicas : tsdbSlaveReplicas,
766
+ members : [ 'timescale-db-postgresql-slave' ],
767
+ },
768
+ hasValkey && {
769
+ header : 'Valkey',
770
+ name : 'System — valkey',
771
+ wakeOrder : 10,
772
+ restoreReplicas : valkeyReplicas !== 1 ? valkeyReplicas : undefined,
773
+ members : [ 'valkey-primary', ...( hasValkeyReplicas ? [ 'valkey-replicas' ] : [] ) ],
774
+ },
775
+ hasRedis && {
776
+ header : 'Redis',
777
+ name : 'System — redis',
778
+ wakeOrder : 10,
779
+ members : [
780
+ 'redis-master',
781
+ ...( defaultSts.has( 'redis-replicas' ) ? [ 'redis-replicas' ] : [] ),
782
+ ],
783
+ },
784
+ promStsMembers.length && {
785
+ header : 'Prometheus',
786
+ name : 'System — prometheus',
787
+ namespace : 'prometheus',
788
+ optional : true,
789
+ members : promStsMembers,
790
+ },
791
+ monitoringStsMembers.length && {
792
+ header : 'Monitoring (legacy)',
793
+ name : 'System — monitoring (legacy)',
794
+ namespace : 'monitoring',
795
+ optional : true,
796
+ members : monitoringStsMembers,
797
+ },
798
+ esDataMembers.length && {
799
+ header : 'Elasticsearch',
800
+ comment : 'Bitnami chart — no operator, plain kubectl scale.',
801
+ name : 'System — elasticsearch (data)',
802
+ namespace : 'elastic',
803
+ wakeOrder : 10,
804
+ restoreReplicas : esDataReplicas,
805
+ members : esDataMembers,
806
+ },
807
+ esOtherMembers.length && {
808
+ name : 'System — elasticsearch',
809
+ namespace : 'elastic',
810
+ wakeOrder : 10,
811
+ restoreReplicas : esOtherReplicas,
812
+ members : esOtherMembers,
813
+ },
814
+ ].filter( Boolean )
815
+
816
+ // ── Write ─────────────────────────────────────────────────────────────────
817
+
818
+ const yaml = buildConfigYaml( {
819
+ clusterName : cluster.name,
820
+ namespace,
821
+ phasePauseMs : 10_000,
822
+ deployPhases,
823
+ stsPhases,
824
+ pools,
825
+ } )
826
+
827
+ // eslint-disable-next-line security/detect-non-literal-fs-filename
828
+ await fs.writeFile( outputPath, yaml, 'utf8' )
829
+
830
+ log( chalk.green( `\nWrote ${outputPath}` ) )
831
+ log( 'Review the file — especially the Project phase and StatefulSet replica counts — then commit.' )
152
832
  }
153
833
 
154
- // ── Commands ──────────────────────────────────────────────────────────────
834
+ // ── Commands ───────────────────────────────────────────────────────────────
155
835
 
156
- export const hibernateCommand = new Command( 'hibernate' )
836
+ const hibernateCommand = new Command( 'hibernate' )
157
837
  .description( 'Scale all workloads to zero and drain all node pools' )
158
- .argument( '[config]', 'path to cluster-config.json', './cluster-config.json' )
159
- .action( async ( config ) => runHibernate( config ).catch( errorExit ) )
838
+ .argument( '[config]', 'path to k8s-cryo.yaml', './k8s-cryo.yaml' )
839
+ .action( async config => runHibernate( config ).catch( errorExit ) )
160
840
 
161
- export const wakeCommand = new Command( 'wake' )
841
+ const wakeCommand = new Command( 'wake' )
162
842
  .description( 'Restore node pools from hibernation (then run helmup)' )
163
- .argument( '[config]', 'path to cluster-config.json', './cluster-config.json' )
164
- .action( async ( config ) => runWake( config ).catch( errorExit ) )
843
+ .argument( '[config]', 'path to k8s-cryo.yaml', './k8s-cryo.yaml' )
844
+ .action( async config => runWake( config ).catch( errorExit ) )
845
+
846
+ const initCommand = new Command( 'init' )
847
+ .description( 'Generate a starter k8s-cryo.yaml from live cluster state' )
848
+ .argument( '[output]', 'output file path', './k8s-cryo.yaml' )
849
+ .option( '-n, --namespace <ns>', 'namespace to scan for deployments', 'default' )
850
+ .option( '-f, --force', 'overwrite existing file' )
851
+ .action( async ( output, opts ) => runInit( output, opts ).catch( errorExit ) )
852
+
853
+ export const cryoCommand = new Command( 'cryo' )
854
+ .description( 'Hibernate, wake, and initialize cluster configurations' )
855
+ .addCommand( initCommand )
856
+ .addCommand( hibernateCommand )
857
+ .addCommand( wakeCommand )