amsd-pipeline 1.9.0 → 1.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/install.sh +159 -3
  2. package/package.json +1 -1
package/install.sh CHANGED
@@ -93,6 +93,18 @@ if [ "$UNINSTALL" = "1" ]; then
93
93
  exit 1
94
94
  fi
95
95
  fi
96
+ # STOP THE RUNNER HOST FIRST — it is NOT docker-related (a plain host process), so it must not
97
+ # be skipped by the "no container runtime, nothing to uninstall" exit below.
98
+ _UN_RH_PIDFILE="$_UN_ROOT/launch-dashboard/.runner-host.pid"
99
+ if [ -f "$_UN_RH_PIDFILE" ]; then
100
+ _UN_RH_PID="$(cat "$_UN_RH_PIDFILE" 2>/dev/null)"
101
+ if [ -n "$_UN_RH_PID" ] && kill -0 "$_UN_RH_PID" 2>/dev/null; then
102
+ kill "$_UN_RH_PID" 2>/dev/null
103
+ _ok "stopped runner-host (pid $_UN_RH_PID)"
104
+ fi
105
+ rm -f "$_UN_RH_PIDFILE"
106
+ fi
107
+
96
108
  . "$INSTALLER_DIR/lib/isolated-compose-identity.sh"
97
109
  if [ -f "$INSTALLER_DIR/lib/container-runtime.sh" ]; then
98
110
  . "$INSTALLER_DIR/lib/container-runtime.sh"
@@ -204,14 +216,33 @@ if [ -n "$DEST" ]; then
204
216
  [ -n "$_excl" ] && _RUN_STATE_EXCLUDES+=("$_excl")
205
217
  done < <(run_state_exclude_args "$INSTALLER_DIR/run-state-paths.json")
206
218
  fi
219
+
220
+ # AN UPDATE MUST NEVER OVERWRITE AN OPERATOR'S EXISTING PROJECT CONFIG EITHER — a DIFFERENT
221
+ # mechanism from the excludes above, deliberately: a blanket exclude would also block a
222
+ # brand-new project's config.env from ever being extracted the first time a later ref adds
223
+ # one. Snapshot whatever already exists now, extract, then restore it over what the ref just
224
+ # wrote — so a first install (or a genuinely new project) is unaffected either way.
225
+ _OPCFG_TMP=""
226
+ if [ -f "$INSTALLER_DIR/operator-config-paths.json" ]; then
227
+ . "$INSTALLER_DIR/lib/preserve-operator-config.sh"
228
+ _OPCFG_TMP="$(mktemp -d)"
229
+ snapshot_operator_config "$DEST" "$INSTALLER_DIR/operator-config-paths.json" "$_OPCFG_TMP"
230
+ fi
231
+
207
232
  # The ${arr[@]+"${arr[@]}"} form, not bare "${arr[@]}": bash <4.4 (macOS ships 3.2 by default,
208
233
  # GPLv3 licensing) throws "unbound variable" under `set -u` expanding an empty array the plain
209
234
  # way. This form is safe on every bash this installer might run under.
210
235
  if ! git -C "$_GIT_ROOT" archive "$_PKG_REF" \
211
236
  | tar -x -C "$DEST" "${_RUN_STATE_EXCLUDES[@]+"${_RUN_STATE_EXCLUDES[@]}"}"; then
212
237
  _bad "packaging '$_PKG_REF' into $DEST failed"
238
+ [ -n "$_OPCFG_TMP" ] && rm -rf "$_OPCFG_TMP"
213
239
  exit 1
214
240
  fi
241
+
242
+ if [ -n "$_OPCFG_TMP" ]; then
243
+ restore_operator_config "$DEST" "$_OPCFG_TMP"
244
+ rm -rf "$_OPCFG_TMP"
245
+ fi
215
246
  _ok "packaged $_PKG_REF into $DEST"
216
247
  ROOT="$DEST"
217
248
  CONFIG="$ROOT/orchestrations/config"
@@ -519,6 +550,24 @@ esac
519
550
 
520
551
  # ── Dashboards: OPTIONAL, and never a reason to fail ────────────────────────
521
552
  _head "Dashboards (optional)"
553
+
554
+ # BUILD BEFORE UP, OR THE HEALTHCHECK FAILS BY CONSTRUCTION. orchestrations/dashboards/live/ is
555
+ # gitignored — eleventy's own build output, never tracked — so on every fresh install it starts
556
+ # EMPTY. agent-monitor's healthcheck probes `/`, nginx has no index and autoindex is off, so a
557
+ # brand-new install was 403-unhealthy FOREVER regardless of docker, subnet or port: found live
558
+ # 2026-09-03, grafana (depends_on agent-monitor: condition service_healthy) never even started —
559
+ # stuck at "Created". Only a completed pipeline run (or this build) ever populated live/ before.
560
+ #
561
+ # Only when there is something to build FROM (src/ present, same test the Build section above
562
+ # uses) — a packaged, src/-less install ships no eleventy at all.
563
+ if [ -d "$ROOT/src" ] && [ -f "$ROOT/package.json" ]; then
564
+ if (cd "$ROOT" && npm run dashboards:build --silent >/dev/null 2>&1); then
565
+ _ok "dashboards built — agent-monitor has real content to serve"
566
+ else
567
+ _warn "dashboards:build failed — agent-monitor's healthcheck may fail until a run populates it"
568
+ fi
569
+ fi
570
+
522
571
  # THE PROBE ASKS THE RESOLVED RUNTIME. It said `docker` literally, so on a podman-only machine the
523
572
  # installer announced "runtime: podman" and then started nothing — the report and the behaviour
524
573
  # disagreeing, which is this file's recurring defect.
@@ -615,25 +664,70 @@ else
615
664
  . "$INSTALLER_DIR/lib/wait-for-health.sh"
616
665
  . "$INSTALLER_DIR/lib/isolated-compose-identity.sh"
617
666
 
618
- # A REAL PASSWORD IS A DECISION ONLY A HUMAN MAKES — never synthesized here. Mirrors the root
619
- # .env handling: copy the template so there is something to fill in, never invent a secret.
667
+ # A VENDOR/API CREDENTIAL IS A DECISION ONLY A HUMAN MAKES — never synthesized (see the root
668
+ # .env handling: copy the template, never invent a secret). LAUNCH_PASSWORD is a DIFFERENT
669
+ # risk class: it gates a loopback-only local UI, not a billed vendor account or a shared
670
+ # system — a blank one previously meant install.sh already knew this stack could not start
671
+ # (it had just written this exact warning) and then attempted `up -d` anyway, hard-failing on
672
+ # compose's `${LAUNCH_PASSWORD:?...}` interpolation instead of the warning it already gave.
673
+ # Operator decision 2026-09-03: generate one so the dashboard starts unattended; changeable
674
+ # any time by editing launch-dashboard/.env directly.
620
675
  if [ ! -f "$LAUNCH_DIR/.env" ]; then
621
676
  if [ -f "$LAUNCH_DIR/.env.example" ]; then
622
677
  cp "$LAUNCH_DIR/.env.example" "$LAUNCH_DIR/.env"
623
- _warn "launch-dashboard/.env created from .env.example — FILL IN LAUNCH_PASSWORD before it can start"
678
+ _GENERATED_PW="$("$NODE_BIN" -e 'process.stdout.write(require("crypto").randomBytes(18).toString("base64url"))' 2>/dev/null)"
679
+ if [ -n "$_GENERATED_PW" ]; then
680
+ # REPLACE the template's blank line in place — never append a second
681
+ # LAUNCH_PASSWORD= key. Both parse fine (bash sourcing takes the last one) but a
682
+ # duplicate key is a needless trap for whoever reads this file by hand next.
683
+ if grep -q '^LAUNCH_PASSWORD=' "$LAUNCH_DIR/.env"; then
684
+ _LD_TMP="$(mktemp)"
685
+ sed "s|^LAUNCH_PASSWORD=.*|LAUNCH_PASSWORD=$_GENERATED_PW|" "$LAUNCH_DIR/.env" > "$_LD_TMP" \
686
+ && mv "$_LD_TMP" "$LAUNCH_DIR/.env"
687
+ else
688
+ printf '\nLAUNCH_PASSWORD=%s\n' "$_GENERATED_PW" >> "$LAUNCH_DIR/.env"
689
+ fi
690
+ _ok "launch-dashboard/.env created with a generated LAUNCH_PASSWORD"
691
+ # SHOWN ONCE, HERE — otherwise the only way to learn it is to already know to go
692
+ # read the file by hand, which is exactly the gap an operator hit live: the
693
+ # dashboard was up and healthy with no way to log into it from the install output
694
+ # alone. Also saved in launch-dashboard/.env for every time after this one.
695
+ printf ' LAUNCH_PASSWORD: %s\n' "$_GENERATED_PW"
696
+ printf ' (also saved in launch-dashboard/.env — edit that file to change it)\n'
697
+ else
698
+ _warn "launch-dashboard/.env created from .env.example — FILL IN LAUNCH_PASSWORD before it can start"
699
+ fi
624
700
  else
625
701
  _bad "launch-dashboard/.env is missing and there is no .env.example to create one from"
626
702
  FAILED=1
627
703
  fi
628
704
  fi
629
705
 
706
+ # PRE-CREATE THE BIND-MOUNT SOURCES, AS THE HOST USER, BEFORE DOCKER EVER SEES THEM.
707
+ #
708
+ # The compose file's own comment above (services.launch-api) already names this exact trap: a
709
+ # bind mount onto a directory that doesn't exist yet gets auto-created BY DOCKER, as root — and
710
+ # launch-api's own `user: "${LAUNCH_UID:-1000}:..."` then cannot write to it. Found live
711
+ # 2026-09-03 against a genuinely fresh install: "unable to open database file", launch-api
712
+ # crash-looping, nginx's launch-ui reporting "host not found in upstream" as a downstream
713
+ # symptom of the crash — same bug CLASS as the dashboards live/ directory fixed above, here for
714
+ # ./data and ./spool specifically.
715
+ mkdir -p "$LAUNCH_DIR/data" "$LAUNCH_DIR/spool"
716
+
630
717
  _LD_PORT="$(grep -E '^LAUNCH_UI_PORT=' "$LAUNCH_DIR/.env" 2>/dev/null | tail -1 | cut -d= -f2)"
631
718
  _LD_PORT="${_LD_PORT:-8099}"
632
719
  _LD_PROJECT="$(isolated_project_name "$ROOT" launch)"
633
720
  _LD_HEALTH_URL="http://localhost:${_LD_PORT}/api/health"
634
721
 
722
+ _LD_PW="$(grep -E '^LAUNCH_PASSWORD=' "$LAUNCH_DIR/.env" 2>/dev/null | tail -1 | cut -d= -f2-)"
635
723
  if [ ! -f "$LAUNCH_DIR/.env" ]; then
636
724
  LAUNCH_STATUS=failed
725
+ elif [ -z "$_LD_PW" ] && [ "$CHECK_ONLY" != "1" ]; then
726
+ # KNOWN ALREADY, NEVER A SURPRISE CRASH. A .env from before LAUNCH_PASSWORD was
727
+ # auto-generated (or one an operator deliberately blanked) still fails compose's
728
+ # `${LAUNCH_PASSWORD:?...}` interpolation — skip the attempt instead of hitting it.
729
+ LAUNCH_STATUS=failed
730
+ _warn "launch-dashboard/.env has no LAUNCH_PASSWORD — skipping start; set one and re-run to bring it up"
637
731
  elif [ "$CHECK_ONLY" = "1" ]; then
638
732
  if wait_for_health "$_LD_HEALTH_URL" 3 1; then
639
733
  LAUNCH_STATUS=up
@@ -692,6 +786,68 @@ else
692
786
  fi
693
787
  fi
694
788
 
789
+ # ── Runner host: what actually launches a pipeline run from the dashboard ────
790
+ # "the install script must start all services" (operator, 2026-09-04) — a saved launch request
791
+ # sat "pending" forever with nothing polling for it, because nothing ever started this.
792
+ #
793
+ # NOT DOCKERIZED, DELIBERATELY — same as runner-host.js's own header says: "a container cannot
794
+ # exec a host process." It spawns the pipeline's real launcher on the HOST, which needs git access
795
+ # to the codeline root, the claude/codemie-claude CLI's host auth (~/.claude — a container has none
796
+ # of this unless it were bind-mounted in), and host git credentials for anything that pushes.
797
+ # Containerizing the poll loop alone is easy (the spool it watches is already bind-mounted into
798
+ # launch-api); containerizing what it SPAWNS on a hit would mean containerizing the whole pipeline.
799
+ _head "Runner host (launches pipeline runs the dashboard queues)"
800
+ if [ "$LAUNCH_STATUS" = "up" ]; then
801
+ _RH_PIDFILE="$LAUNCH_DIR/.runner-host.pid"
802
+ _RH_LOG="$LAUNCH_DIR/.runner-host.log"
803
+ _RH_OLD_PID=""
804
+ [ -f "$_RH_PIDFILE" ] && _RH_OLD_PID="$(cat "$_RH_PIDFILE" 2>/dev/null)"
805
+ if [ -n "$_RH_OLD_PID" ] && kill -0 "$_RH_OLD_PID" 2>/dev/null; then
806
+ _ok "already running (pid $_RH_OLD_PID)"
807
+ else
808
+ # setsid, NEVER nohup — found live: nohup here made install.sh hang forever whenever its
809
+ # own stdio is piped (any parent that captures its output, including this test suite).
810
+ # setsid fully detaches into a new session (immune to SIGHUP by construction, survives the
811
+ # launching shell/terminal closing — the WSL-restart case this exists for). Falls back to
812
+ # a plain backgrounded process on a host with no setsid (macOS ships none by default).
813
+ #
814
+ # `</dev/null >>log 2>&1` on the command ALONE was still not enough — the daemon kept the
815
+ # pipe to install.sh's own stdout open regardless (Node's spawn() never saw 'close', even
816
+ # though every byte of real output arrived and install.sh itself had long since exited).
817
+ # bash forking a background job inherits ALL open fds, not just 0/1/2; a plain per-command
818
+ # redirect only dup2's those three. `exec` with no command applies the redirect to the
819
+ # CURRENT shell — including whatever else it inherited — before the second `exec` replaces
820
+ # that shell's own process image with the daemon, so nothing is left holding the pipe open.
821
+ _RH_DAEMONIZE="setsid"
822
+ command -v setsid >/dev/null 2>&1 || _RH_DAEMONIZE=""
823
+ # launch-dashboard/.env MUST BE SOURCED HERE. Docker Compose auto-loads a .env file next
824
+ # to the compose file into the CONTAINER's environment; a bare host process gets none of
825
+ # that for free. Found live: runner-host.js's own config.js hard-requires LAUNCH_PASSWORD
826
+ # from process.env ("gates a button that spends real money") and crashed instantly with it
827
+ # unset, even though the value was sitting right there in the file the whole time.
828
+ # SPOOL_DIR's default ('/spool') is the CONTAINER's bind-mount path — correct for
829
+ # launch-api running inside docker, meaningless for a bare host process. Found live, right
830
+ # after the LAUNCH_PASSWORD fix above stopped masking it: EACCES on mkdir '/spool/requests'
831
+ # (no permission to create a directory at the filesystem root). The real, same, host
832
+ # directory this container has bind-mounted as /spool is $LAUNCH_DIR/spool.
833
+ ( exec </dev/null >>"$_RH_LOG" 2>&1
834
+ cd "$ROOT" && set -a && . "$LAUNCH_DIR/.env" 2>/dev/null; set +a
835
+ EPAM_HOME="$ROOT" SPOOL_DIR="$LAUNCH_DIR/spool" RUNS_DB="$LAUNCH_DIR/data/runs.db" \
836
+ exec $_RH_DAEMONIZE "$NODE_BIN" "$LAUNCH_DIR/backend/src/runner-host.js" ) &
837
+ echo $! > "$_RH_PIDFILE"
838
+ sleep 0.3
839
+ _RH_NEW_PID="$(cat "$_RH_PIDFILE" 2>/dev/null)"
840
+ if [ -n "$_RH_NEW_PID" ] && kill -0 "$_RH_NEW_PID" 2>/dev/null; then
841
+ _ok "started (pid $_RH_NEW_PID, log: $_RH_LOG)"
842
+ else
843
+ _bad "runner-host failed to start — see $_RH_LOG"
844
+ FAILED=1
845
+ fi
846
+ fi
847
+ else
848
+ _ok "skipped — launch dashboard status is '$LAUNCH_STATUS', nothing to poll for"
849
+ fi
850
+
695
851
  # ── The command people will actually type ───────────────────────────────────
696
852
  # ── What this install IS ──────────────────────────────────────────────────────
697
853
  # An install whose mode can only be inferred from which containers happen to be running is an
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "amsd-pipeline",
3
- "version": "1.9.0",
3
+ "version": "1.10.0",
4
4
  "description": "Installer for the amsd-pipeline orchestration stack. Clones, packages and provisions the full stack with one command — no separate git clone step.",
5
5
  "bin": {
6
6
  "amsd-pipeline": "bin/amsd-pipeline.js"