@leverege/build-tools 2.114.0-DEVOP-578.1 → 2.114.0-DEVOP-578.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@leverege/build-tools",
3
- "version": "2.114.0-DEVOP-578.1",
3
+ "version": "2.114.0-DEVOP-578.2",
4
4
  "description": "A collection of build / support tools for Leverege developers",
5
5
  "main": "index.js",
6
6
  "repository": {
@@ -68,6 +68,7 @@
68
68
  "service-man": "src/service-man/service-man.mjs",
69
69
  "snapshot-cnpg": "src/snapshot-cnpg.mjs",
70
70
  "sync-chart-version": "src/sync-chart-version.mjs",
71
+ "throwup": "src/helmup.sh",
71
72
  "upload-model": "src/upload-model.sh",
72
73
  "yarn-build": "src/yarn-build.mjs"
73
74
  },
@@ -96,7 +97,7 @@
96
97
  "commander": "^15.0.0",
97
98
  "deepmerge": "^4.3.1",
98
99
  "enquirer": "^2.4.1",
99
- "execa": "^10.0.0",
100
+ "execa": "^10.0.1",
100
101
  "find-yarn-workspace-root": "^2.0.0",
101
102
  "glob": "^13.0.6",
102
103
  "googleapis": "^173.0.0",
@@ -125,8 +126,8 @@
125
126
  "@leverege/eslint-config-leverege": "^9",
126
127
  "chai": "^6.2.2",
127
128
  "eslint": "^9",
128
- "mocha": "^11.7.6",
129
- "npm": "^12.0.1"
129
+ "mocha": "^11.8.0",
130
+ "npm": "^12.0.2"
130
131
  },
131
132
  "allowScripts": {
132
133
  "unrs-resolver@1.12.2": true
package/src/bash-funcs CHANGED
@@ -120,6 +120,18 @@ function getManagedSecret() {
120
120
  echo "$SECRET"
121
121
  }
122
122
 
123
+ # Fail fast when gcloud cannot mint tokens. Downstream tools capture gcloud
124
+ # output, so reauthentication cannot prompt there; surfacing it here lets
125
+ # gcloud reauth interactively on the TTY when policy allows. stdout goes to a
126
+ # literal /dev/null (NOT $DEVNULL, which is /dev/stderr in debug mode) so the
127
+ # access tokens are never printed.
128
+ function errorIfStaleGcloudAuth() {
129
+ gcloud auth print-access-token > /dev/null
130
+ exitOnError $? "GCP credentials expired. Please run: `color g 'gcloud auth login'`"
131
+ gcloud auth application-default print-access-token > /dev/null
132
+ exitOnError $? "GCP application-default credentials expired. Please run: `color g 'gcloud auth application-default login'`"
133
+ }
134
+
123
135
  function getProjectIdFromK8sContext() {
124
136
  local currContext=`kubectl config current-context`
125
137
 
@@ -286,7 +298,7 @@ function ifPluggedIn() {
286
298
  local PLUGIN=
287
299
  HELMUP="helm upgrade --install $SERVICE $CHARTLOC -f $SERVICE/values.yaml $CHARTVER $HELM_WHAT"
288
300
  case "$INVOKED_AS" in
289
- "helmup"|"helmwhat"|"helmcycle"|"helminit" )
301
+ "helmup"|"helmwhat"|"helmcycle"|"helminit"|"throwup" )
290
302
  PLUGIN="$1/helmup.plugin"
291
303
  ;;
292
304
  "helmdn" )
@@ -6,12 +6,8 @@ orbs:
6
6
  workflows:
7
7
  build-and-verify:
8
8
  jobs:
9
- - leverege/install-dependencies:
10
- context: npm
11
9
  - leverege/code-analysis:
12
10
  context: npm
13
- requires:
14
- - leverege/install-dependencies
15
11
  - leverege/code-coverage:
16
12
  context:
17
13
  - npm
@@ -19,12 +15,8 @@ workflows:
19
15
  coverage-threshold: 70 # percentage threshold for functions, lines, and statements
20
16
  enable-redis: false # set to true if tests require a live Redis connection
21
17
  cache-config: '{"type":"inMemory"}' # sets CACHE_CONFIG env var - options: inMemory, redis
22
- requires:
23
- - leverege/install-dependencies
24
18
  - leverege/generate-docs:
25
19
  context: npm
26
- requires:
27
- - leverege/install-dependencies
28
20
  - leverege/mirror-to-github:
29
21
  context: github-mirror
30
22
  # github-team: platform # available teams: platform, utilities, pitcrew, recovr
@@ -0,0 +1,45 @@
1
+ image:
2
+ tag:
3
+
4
+ config:
5
+ LOG_LEVEL: "WARNING"
6
+ GCP_PROJECT_ID: "${PROJECT_ID}" # needed for python secrets manager?
7
+ GCP_SECRET_MANAGER_NAMESPACE: "VMS_SERVER"
8
+ IMAGINE_API_HOST: "http://api-server:8181"
9
+ IMAGINE_API_PROJECT_ID: "set-in-values-local"
10
+ IMAGINE_API_SYSTEM_ID: "set-in-values-local"
11
+ IMAGINE_API_KEY: "set-in-values-local"
12
+ IMAGINE_API_AUDIENCE: "${PROJECT_ID}-aud"
13
+ IMAGINE_SERVICE_SYNC_ENABLED: "true"
14
+ IMAGINE_CAMERA_BLUEPRINT_ID: "set-in-values-local"
15
+ VMS_SERVER_HOST: "http://vms-server:8080"
16
+ VIDEO_PLAYBACK_SERVICE_VERSION: "20"
17
+ VIDEO_MANAGEMENT_SERVICE_VERSION: "2"
18
+ USE_SHARED_METRICS: "true"
19
+ VIDEO_SEGMENTS_UPLOAD_TIMEOUT: "60"
20
+ PLAYLIST_OPEN_ENDED_CACHE_ENABLED: "true"
21
+ SEGMENT_RECONCILER_ENABLED: "true"
22
+ SEGMENT_RECONCILER_INTERVAL_SECONDS: "300"
23
+ SEGMENT_RECONCILER_MIN_AGE_SECONDS: "300"
24
+ SEGMENT_RECONCILER_LOOKBACK_HOURS: "24"
25
+
26
+ # run on the preemptible node pool (base chart adds the target-env=preemptible
27
+ # nodeAffinity, toleration and safe-to-evict annotation)
28
+ service:
29
+ preemptible: true
30
+
31
+ serviceMonitor:
32
+ enabled: true
33
+ interval: "60s"
34
+ namespace: "default"
35
+
36
+ resources:
37
+ requests:
38
+ cpu: 500m
39
+ memory: 512Mi
40
+
41
+ autoscaling:
42
+ enabled: true
43
+ minReplicas: 2
44
+ maxReplicas: 16
45
+ averageCPU: 75
package/src/helmup.sh CHANGED
@@ -82,6 +82,9 @@ BAD_IDEA
82
82
  exit 1
83
83
  fi
84
84
 
85
+ # stop here rather than half way through a deploy if the creds cannot mint tokens
86
+ errorIfStaleGcloudAuth
87
+
85
88
  # artifact-registry => https://console.cloud.google.com/artifacts?project=leverege-registry
86
89
  REGISTRY_DOMAIN="us-docker.pkg.dev"
87
90
  # open container initiative (oci) => https://opencontainers.org/
@@ -92,6 +95,9 @@ gcloud auth application-default print-access-token | helm registry login \
92
95
 
93
96
  # make sure we have a current npmrc file
94
97
  refresh-npm-token
98
+ exitOnError $? "refresh-npm-token failed - the rc files were not refreshed"
99
+
100
+ HELM_EMOJI=$( [ "$INVOKED_AS" == "throwup" ] && echo "🤮" || echo "⎈" )
95
101
 
96
102
  # removed geotile-server from standard platform 3/20/22 - low usage
97
103
  PLATFORM=(
@@ -100,6 +106,7 @@ PLATFORM=(
100
106
  db-curator
101
107
  emailer
102
108
  imagine
109
+ llm-prompt-server
103
110
  message-processor
104
111
  messenger
105
112
  resource-server
@@ -122,6 +129,7 @@ AUXILIARY=(
122
129
  reason
123
130
  transponder-dh
124
131
  vin-decoder-server
132
+ vms-server
125
133
  )
126
134
 
127
135
  # There's a chicken / egg situation that arises with new clusters in that
@@ -622,7 +630,7 @@ HELMUP_BANNER
622
630
  function doHelmup() {
623
631
  local culprit="$(gcloud config get account)"
624
632
  local deployT="$(date -u +'%Y-%m-%dT%H:%M:%SZ')"
625
- local doing="Helming => `color g \"$*\"`"
633
+ local doing="Helming => $HELM_EMOJI `color g \"$*\"` $HELM_EMOJI"
626
634
  local from=
627
635
  [ ! -z "$FROM_VERSIONS" ] && from="`color y \"<= from Versions.json\"`"
628
636
  printf "$doing $from\n"
@@ -0,0 +1,164 @@
1
+ import { Command } from 'commander'
2
+ import chalk from 'chalk'
3
+ import { $ } from 'zx'
4
+
5
+ import { log, errorExit, sleep, warning, proceed, parseJsonFile } from '../../Utils.mjs'
6
+ import { overwhelmGate } from '../lib/gate.mjs'
7
+
8
+ $.verbose = false
9
+
10
+ // ── Config ────────────────────────────────────────────────────────────────
11
+
12
+ const loadConfig = async ( configPath ) => {
13
+ const config = await parseJsonFile( configPath )
14
+ if ( !config ) errorExit( `Config file not found: ${configPath}` )
15
+
16
+ const { cluster, deploymentPhases, nodePools } = config
17
+ if ( !cluster?.project || !cluster?.name || !cluster?.region ) {
18
+ errorExit( 'Config must include cluster.project, cluster.name, and cluster.region' )
19
+ }
20
+ if ( !deploymentPhases || !nodePools ) {
21
+ errorExit( 'Config must include deploymentPhases and nodePools' )
22
+ }
23
+ return config
24
+ }
25
+
26
+ // ── Display ───────────────────────────────────────────────────────────────
27
+
28
+ const banner = ( text ) => {
29
+ const bar = '─'.repeat( Math.max( 0, 56 - text.length ) )
30
+ log( chalk.cyan.bold( `\n── ${text} ${bar}` ) )
31
+ }
32
+
33
+ // ── kubectl ───────────────────────────────────────────────────────────────
34
+
35
+ const scaleDeployments = async ( names, replicas, namespace ) =>
36
+ $`kubectl scale deployment ${names} --replicas=${replicas} -n ${namespace}`
37
+
38
+ const scaleStatefulSets = async ( names, replicas, namespace ) =>
39
+ $`kubectl scale statefulset ${names} --replicas=${replicas} -n ${namespace}`
40
+
41
+ const waitForPodsGone = async ( namespace, timeoutSecs = 300 ) => {
42
+ log( `\nWaiting for pods to terminate (up to ${timeoutSecs}s)...` )
43
+ try {
44
+ await $`kubectl wait pod --all -n ${namespace} --for=delete --timeout=${timeoutSecs}s`
45
+ } catch {
46
+ warning( `Some pods still terminating after ${timeoutSecs}s — proceeding.` )
47
+ }
48
+ }
49
+
50
+ const waitForNodesReady = async ( timeoutSecs = 600 ) => {
51
+ log( `\nWaiting for nodes to become Ready (up to ${timeoutSecs}s)...` )
52
+ await $`kubectl wait node --all --for=condition=Ready --timeout=${timeoutSecs}s`
53
+ }
54
+
55
+ // ── gcloud ────────────────────────────────────────────────────────────────
56
+
57
+ const updateAutoscalerMin = async ( pool, cluster, min ) => {
58
+ log( ` ${pool.name}: autoscaler min → ${min}` )
59
+ const { name, project, region } = cluster
60
+ await $`gcloud container node-pools update ${pool.name} --cluster=${name} --project=${project} --region=${region} --enable-autoscaling --min-nodes=${min} --max-nodes=${pool.autoscaler.max} --quiet`
61
+ }
62
+
63
+ const resizePool = async ( pool, cluster, size ) => {
64
+ log( ` ${pool.name} → ${size} node(s)` )
65
+ const { name, project, region } = cluster
66
+ await $`gcloud container clusters resize ${name} --node-pool=${pool.name} --num-nodes=${size} --project=${project} --region=${region} --quiet`
67
+ }
68
+
69
+ // ── Hibernate ─────────────────────────────────────────────────────────────
70
+
71
+ const runHibernate = async ( configPath ) => {
72
+ const config = await loadConfig( configPath )
73
+ const {
74
+ cluster,
75
+ namespace = 'default',
76
+ phasePauseMs = 10_000,
77
+ deploymentPhases,
78
+ statefulsetPhases = [],
79
+ nodePools,
80
+ } = config
81
+
82
+ await overwhelmGate()
83
+ await proceed( `Hibernate ${chalk.yellow.bold( cluster.name )}? All workloads scale to 0 and node pools drain.` )
84
+
85
+ for ( const group of deploymentPhases ) {
86
+ banner( `Deployments: ${group.name}` )
87
+ await scaleDeployments( group.members, 0, namespace )
88
+ await sleep( group.pauseMs ?? phasePauseMs )
89
+ }
90
+
91
+ if ( statefulsetPhases.length ) {
92
+ for ( const group of statefulsetPhases ) {
93
+ banner( `StatefulSets: ${group.name}` )
94
+ await scaleStatefulSets( group.members, 0, namespace )
95
+ await sleep( group.pauseMs ?? phasePauseMs )
96
+ }
97
+ } else {
98
+ // no ordered phases defined yet — scale all StatefulSets together
99
+ banner( 'StatefulSets' )
100
+ await $`kubectl scale statefulset --all -n ${namespace} --replicas=0`
101
+ }
102
+
103
+ await waitForPodsGone( namespace )
104
+
105
+ const poolsWithAutoscaler = nodePools.filter( p => p.autoscaler )
106
+ if ( poolsWithAutoscaler.length ) {
107
+ banner( 'Clearing autoscaler minimums' )
108
+ for ( const pool of poolsWithAutoscaler ) {
109
+ await updateAutoscalerMin( pool, cluster, 0 )
110
+ }
111
+ }
112
+
113
+ banner( 'Draining node pools' )
114
+ for ( const pool of nodePools ) {
115
+ await resizePool( pool, cluster, 0 )
116
+ }
117
+
118
+ log( chalk.green( '\nCluster hibernated. Control plane (~$0.10/hr) continues running.' ) )
119
+ log( `To wake: ${chalk.cyan( `k8s wake ${configPath}` )}` )
120
+ }
121
+
122
+ // ── Wake ──────────────────────────────────────────────────────────────────
123
+
124
+ const runWake = async ( configPath ) => {
125
+ const config = await loadConfig( configPath )
126
+ const { cluster, nodePools } = config
127
+
128
+ await overwhelmGate()
129
+
130
+ // Restore autoscaler minimums before resizing so the autoscaler
131
+ // doesn't immediately scale back down the nodes we're bringing up.
132
+ const poolsWithAutoscaler = nodePools.filter( p => p.autoscaler )
133
+ if ( poolsWithAutoscaler.length ) {
134
+ banner( 'Restoring autoscaler minimums' )
135
+ for ( const pool of poolsWithAutoscaler ) {
136
+ await updateAutoscalerMin( pool, cluster, pool.autoscaler.min )
137
+ }
138
+ }
139
+
140
+ banner( 'Restoring node pools' )
141
+ for ( const pool of nodePools ) {
142
+ if ( pool.restoreSize > 0 ) {
143
+ await resizePool( pool, cluster, pool.restoreSize )
144
+ } else {
145
+ log( ` ${pool.name} → skipped (autoscaler brings up on demand)` )
146
+ }
147
+ }
148
+
149
+ await waitForNodesReady()
150
+
151
+ log( chalk.green( '\nNodes ready. Run helmup to restore all services.' ) )
152
+ }
153
+
154
+ // ── Commands ──────────────────────────────────────────────────────────────
155
+
156
+ export const hibernateCommand = new Command( 'hibernate' )
157
+ .description( 'Scale all workloads to zero and drain all node pools' )
158
+ .argument( '[config]', 'path to cluster-config.json', './cluster-config.json' )
159
+ .action( async ( config ) => runHibernate( config ).catch( errorExit ) )
160
+
161
+ export const wakeCommand = new Command( 'wake' )
162
+ .description( 'Restore node pools from hibernation (then run helmup)' )
163
+ .argument( '[config]', 'path to cluster-config.json', './cluster-config.json' )
164
+ .action( async ( config ) => runWake( config ).catch( errorExit ) )
package/src/k8s/k8s.mjs CHANGED
@@ -8,6 +8,7 @@ import { scaleCommand } from './commands/scale.mjs'
8
8
  import { rollCommand } from './commands/roll.mjs'
9
9
  import { logCommand } from './commands/log.mjs'
10
10
  import { completionCommand } from './commands/completion.mjs'
11
+ import { hibernateCommand, wakeCommand } from './commands/cryo.mjs'
11
12
 
12
13
  $.verbose = false
13
14
 
@@ -20,4 +21,6 @@ program
20
21
  .addCommand( rollCommand )
21
22
  .addCommand( logCommand )
22
23
  .addCommand( completionCommand )
24
+ .addCommand( hibernateCommand )
25
+ .addCommand( wakeCommand )
23
26
  .parse()
@@ -0,0 +1,8 @@
1
+ {
2
+ "roles" : [
3
+ "leveregeBigQueryReader",
4
+ "leveregePubSub",
5
+ "leveregeSecretsViewer",
6
+ "leveregeStorageObjectAdmin"
7
+ ]
8
+ }