amsd-pipeline 1.10.0 → 1.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/install.sh +68 -75
  2. package/package.json +1 -1
package/install.sh CHANGED
@@ -96,13 +96,18 @@ if [ "$UNINSTALL" = "1" ]; then
96
96
  # STOP THE RUNNER HOST FIRST — it is NOT docker-related (a plain host process), so it must not
97
97
  # be skipped by the "no container runtime, nothing to uninstall" exit below.
98
98
  _UN_RH_PIDFILE="$_UN_ROOT/launch-dashboard/.runner-host.pid"
99
- if [ -f "$_UN_RH_PIDFILE" ]; then
100
- _UN_RH_PID="$(cat "$_UN_RH_PIDFILE" 2>/dev/null)"
101
- if [ -n "$_UN_RH_PID" ] && kill -0 "$_UN_RH_PID" 2>/dev/null; then
102
- kill "$_UN_RH_PID" 2>/dev/null
103
- _ok "stopped runner-host (pid $_UN_RH_PID)"
104
- fi
105
- rm -f "$_UN_RH_PIDFILE"
99
+ _UN_RH_WAS_RUNNING=""
100
+ [ -f "$_UN_RH_PIDFILE" ] && _UN_RH_WAS_RUNNING="$(cat "$_UN_RH_PIDFILE" 2>/dev/null)"
101
+ . "$INSTALLER_DIR/lib/runner-host-control.sh"
102
+ if stop_runner_host "$_UN_ROOT/launch-dashboard" && [ -n "$_UN_RH_WAS_RUNNING" ]; then
103
+ _ok "stopped runner-host (pid $_UN_RH_WAS_RUNNING)"
104
+ fi
105
+ _UN_SW_PIDFILE="$_UN_ROOT/orchestrations/dashboards/.snapshot-watch.pid"
106
+ _UN_SW_WAS_RUNNING=""
107
+ [ -f "$_UN_SW_PIDFILE" ] && _UN_SW_WAS_RUNNING="$(cat "$_UN_SW_PIDFILE" 2>/dev/null)"
108
+ . "$INSTALLER_DIR/lib/snapshot-watch-control.sh"
109
+ if stop_snapshot_watch "$_UN_ROOT" && [ -n "$_UN_SW_WAS_RUNNING" ]; then
110
+ _ok "stopped snapshot-watch (pid $_UN_SW_WAS_RUNNING)"
106
111
  fi
107
112
 
108
113
  . "$INSTALLER_DIR/lib/isolated-compose-identity.sh"
@@ -626,6 +631,19 @@ compose_up() {
626
631
  rm -f "$_log" 2>/dev/null
627
632
  return 1
628
633
  fi
634
+ # PERSISTED so `pipeline-services.sh --start` can bring this exact stack back up later
635
+ # (after a WSL restart, a deliberate stop) WITHOUT re-rolling a different subnet/port —
636
+ # `down` (no -v) removes the network, so a later `up` with no env at all would fall back to
637
+ # the compose file's own default subnet and could collide with the dev stack or another
638
+ # install. Written fresh on every successful compose_up() — this IS the current identity.
639
+ {
640
+ printf 'OBS_PROJECT=%s\n' "$_OBS_PROJECT"
641
+ printf 'OBS_SUBNET=%s\n' "$_subnet"
642
+ printf 'OBS_CLICKHOUSE_PORT=%s\n' "$((8123 + _off))"
643
+ printf 'OBS_LANGFUSE_PORT=%s\n' "$((3100 + _off))"
644
+ printf 'OBS_DASHBOARD_PORT=%s\n' "$((8092 + _off))"
645
+ printf 'OBS_GRAFANA_PORT=%s\n' "$((3001 + _off))"
646
+ } > "$ROOT/.pipeline-services-state.env"
629
647
  rm -f "$_log" 2>/dev/null
630
648
  return 0
631
649
  }
@@ -639,6 +657,19 @@ case "$USE_DOCKER" in
639
657
  else _warn "no container runtime is running — dashboards unavailable, THE PIPELINE STILL RUNS"; fi ;;
640
658
  esac
641
659
 
660
+ # ── Snapshot watch: keeps agent-monitor's build-info.json fresh ─────────────
661
+ # A HOST process, same class as runner-host.js: it writes orchestrations/dashboards/live/
662
+ # build-info.json on the host filesystem, which agent-monitor's nginx container only reads via a
663
+ # bind mount. Without it, a run's own pre-flight fails 3 checks every time — found live
664
+ # 2026-09-04 against a genuinely fresh install ("snapshot-watch.js is NOT running").
665
+ _head "Snapshot watch (keeps the dashboard's build-info.json fresh)"
666
+ if [ "$CHECK_ONLY" != "1" ] && [ -f "$ROOT/orchestrations/scripts/snapshot-watch.js" ]; then
667
+ . "$INSTALLER_DIR/lib/snapshot-watch-control.sh"
668
+ start_snapshot_watch "$ROOT" || FAILED=1
669
+ else
670
+ _ok "skipped"
671
+ fi
672
+
642
673
  # ── Launch dashboard: OPTIONAL, and must be genuinely UP when it claims to be ─
643
674
  # THIS CANNOT DEPEND ON A HUMAN OR AN LLM DOING IT BY HAND. A rebuild-and-restart done manually
644
675
  # once is a rebuild-and-restart that must be done manually every time — this makes it the same
@@ -670,33 +701,32 @@ else
670
701
  # system — a blank one previously meant install.sh already knew this stack could not start
671
702
  # (it had just written this exact warning) and then attempted `up -d` anyway, hard-failing on
672
703
  # compose's `${LAUNCH_PASSWORD:?...}` interpolation instead of the warning it already gave.
673
- # Operator decision 2026-09-03: generate one so the dashboard starts unattended; changeable
674
- # any time by editing launch-dashboard/.env directly.
704
+ #
705
+ # Operator decision 2026-09-04, SUPERSEDING the 2026-09-03 random-generation decision: a
706
+ # RANDOM value made every install's password unknowable without reading the file, and
707
+ # unrecoverable once a running container had it in memory — copying a newer .env over an
708
+ # older one (exactly what moving credentials between two installs looks like) silently
709
+ # desynced the file from the live process, and the only fix was a manual docker restart.
710
+ # A FIXED, KNOWN default removes the ambiguity entirely: every fresh install starts on the
711
+ # same well-known password, printed here and in the file either way, and the change-password
712
+ # flow (Flutter UI, not this script) is how an operator actually secures it afterward.
713
+ _DEFAULT_LAUNCH_PW="abcd1234"
675
714
  if [ ! -f "$LAUNCH_DIR/.env" ]; then
676
715
  if [ -f "$LAUNCH_DIR/.env.example" ]; then
677
716
  cp "$LAUNCH_DIR/.env.example" "$LAUNCH_DIR/.env"
678
- _GENERATED_PW="$("$NODE_BIN" -e 'process.stdout.write(require("crypto").randomBytes(18).toString("base64url"))' 2>/dev/null)"
679
- if [ -n "$_GENERATED_PW" ]; then
680
- # REPLACE the template's blank line in place — never append a second
681
- # LAUNCH_PASSWORD= key. Both parse fine (bash sourcing takes the last one) but a
682
- # duplicate key is a needless trap for whoever reads this file by hand next.
683
- if grep -q '^LAUNCH_PASSWORD=' "$LAUNCH_DIR/.env"; then
684
- _LD_TMP="$(mktemp)"
685
- sed "s|^LAUNCH_PASSWORD=.*|LAUNCH_PASSWORD=$_GENERATED_PW|" "$LAUNCH_DIR/.env" > "$_LD_TMP" \
686
- && mv "$_LD_TMP" "$LAUNCH_DIR/.env"
687
- else
688
- printf '\nLAUNCH_PASSWORD=%s\n' "$_GENERATED_PW" >> "$LAUNCH_DIR/.env"
689
- fi
690
- _ok "launch-dashboard/.env created with a generated LAUNCH_PASSWORD"
691
- # SHOWN ONCE, HERE — otherwise the only way to learn it is to already know to go
692
- # read the file by hand, which is exactly the gap an operator hit live: the
693
- # dashboard was up and healthy with no way to log into it from the install output
694
- # alone. Also saved in launch-dashboard/.env for every time after this one.
695
- printf ' LAUNCH_PASSWORD: %s\n' "$_GENERATED_PW"
696
- printf ' (also saved in launch-dashboard/.env — edit that file to change it)\n'
717
+ # REPLACE the template's blank line in place — never append a second
718
+ # LAUNCH_PASSWORD= key. Both parse fine (bash sourcing takes the last one) but a
719
+ # duplicate key is a needless trap for whoever reads this file by hand next.
720
+ if grep -q '^LAUNCH_PASSWORD=' "$LAUNCH_DIR/.env"; then
721
+ _LD_TMP="$(mktemp)"
722
+ sed "s|^LAUNCH_PASSWORD=.*|LAUNCH_PASSWORD=$_DEFAULT_LAUNCH_PW|" "$LAUNCH_DIR/.env" > "$_LD_TMP" \
723
+ && mv "$_LD_TMP" "$LAUNCH_DIR/.env"
697
724
  else
698
- _warn "launch-dashboard/.env created from .env.example — FILL IN LAUNCH_PASSWORD before it can start"
725
+ printf '\nLAUNCH_PASSWORD=%s\n' "$_DEFAULT_LAUNCH_PW" >> "$LAUNCH_DIR/.env"
699
726
  fi
727
+ _ok "launch-dashboard/.env created with the default LAUNCH_PASSWORD"
728
+ printf ' LAUNCH_PASSWORD: %s\n' "$_DEFAULT_LAUNCH_PW"
729
+ printf ' CHANGE THIS after your first login — it is the same on every fresh install.\n'
700
730
  else
701
731
  _bad "launch-dashboard/.env is missing and there is no .env.example to create one from"
702
732
  FAILED=1
@@ -777,6 +807,13 @@ else
777
807
  elif wait_for_health "$_LD_HEALTH_URL" "$LAUNCH_HEALTH_TRIES" "$LAUNCH_HEALTH_INTERVAL"; then
778
808
  LAUNCH_STATUS=up
779
809
  _ok "up and healthy at $_LD_HEALTH_URL (project: $_LD_PROJECT, subnet: $_LD_SUBNET)"
810
+ # Same reason as the observability stack's own state file — appended, not truncated:
811
+ # that one is always written first in a single install.sh run.
812
+ {
813
+ printf 'LAUNCH_PROJECT=%s\n' "$_LD_PROJECT"
814
+ printf 'LAUNCH_SUBNET=%s\n' "$_LD_SUBNET"
815
+ printf 'LAUNCH_UI_PORT=%s\n' "$_LD_PORT"
816
+ } >> "$ROOT/.pipeline-services-state.env"
780
817
  else
781
818
  LAUNCH_STATUS=unhealthy
782
819
  _bad "containers started but never answered healthy at $_LD_HEALTH_URL"
@@ -798,52 +835,8 @@ fi
798
835
  # launch-api); containerizing what it SPAWNS on a hit would mean containerizing the whole pipeline.
799
836
  _head "Runner host (launches pipeline runs the dashboard queues)"
800
837
  if [ "$LAUNCH_STATUS" = "up" ]; then
801
- _RH_PIDFILE="$LAUNCH_DIR/.runner-host.pid"
802
- _RH_LOG="$LAUNCH_DIR/.runner-host.log"
803
- _RH_OLD_PID=""
804
- [ -f "$_RH_PIDFILE" ] && _RH_OLD_PID="$(cat "$_RH_PIDFILE" 2>/dev/null)"
805
- if [ -n "$_RH_OLD_PID" ] && kill -0 "$_RH_OLD_PID" 2>/dev/null; then
806
- _ok "already running (pid $_RH_OLD_PID)"
807
- else
808
- # setsid, NEVER nohup — found live: nohup here made install.sh hang forever whenever its
809
- # own stdio is piped (any parent that captures its output, including this test suite).
810
- # setsid fully detaches into a new session (immune to SIGHUP by construction, survives the
811
- # launching shell/terminal closing — the WSL-restart case this exists for). Falls back to
812
- # a plain backgrounded process on a host with no setsid (macOS ships none by default).
813
- #
814
- # `</dev/null >>log 2>&1` on the command ALONE was still not enough — the daemon kept the
815
- # pipe to install.sh's own stdout open regardless (Node's spawn() never saw 'close', even
816
- # though every byte of real output arrived and install.sh itself had long since exited).
817
- # bash forking a background job inherits ALL open fds, not just 0/1/2; a plain per-command
818
- # redirect only dup2's those three. `exec` with no command applies the redirect to the
819
- # CURRENT shell — including whatever else it inherited — before the second `exec` replaces
820
- # that shell's own process image with the daemon, so nothing is left holding the pipe open.
821
- _RH_DAEMONIZE="setsid"
822
- command -v setsid >/dev/null 2>&1 || _RH_DAEMONIZE=""
823
- # launch-dashboard/.env MUST BE SOURCED HERE. Docker Compose auto-loads a .env file next
824
- # to the compose file into the CONTAINER's environment; a bare host process gets none of
825
- # that for free. Found live: runner-host.js's own config.js hard-requires LAUNCH_PASSWORD
826
- # from process.env ("gates a button that spends real money") and crashed instantly with it
827
- # unset, even though the value was sitting right there in the file the whole time.
828
- # SPOOL_DIR's default ('/spool') is the CONTAINER's bind-mount path — correct for
829
- # launch-api running inside docker, meaningless for a bare host process. Found live, right
830
- # after the LAUNCH_PASSWORD fix above stopped masking it: EACCES on mkdir '/spool/requests'
831
- # (no permission to create a directory at the filesystem root). The real, same, host
832
- # directory this container has bind-mounted as /spool is $LAUNCH_DIR/spool.
833
- ( exec </dev/null >>"$_RH_LOG" 2>&1
834
- cd "$ROOT" && set -a && . "$LAUNCH_DIR/.env" 2>/dev/null; set +a
835
- EPAM_HOME="$ROOT" SPOOL_DIR="$LAUNCH_DIR/spool" RUNS_DB="$LAUNCH_DIR/data/runs.db" \
836
- exec $_RH_DAEMONIZE "$NODE_BIN" "$LAUNCH_DIR/backend/src/runner-host.js" ) &
837
- echo $! > "$_RH_PIDFILE"
838
- sleep 0.3
839
- _RH_NEW_PID="$(cat "$_RH_PIDFILE" 2>/dev/null)"
840
- if [ -n "$_RH_NEW_PID" ] && kill -0 "$_RH_NEW_PID" 2>/dev/null; then
841
- _ok "started (pid $_RH_NEW_PID, log: $_RH_LOG)"
842
- else
843
- _bad "runner-host failed to start — see $_RH_LOG"
844
- FAILED=1
845
- fi
846
- fi
838
+ . "$INSTALLER_DIR/lib/runner-host-control.sh"
839
+ start_runner_host "$ROOT" "$LAUNCH_DIR" || FAILED=1
847
840
  else
848
841
  _ok "skipped — launch dashboard status is '$LAUNCH_STATUS', nothing to poll for"
849
842
  fi
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "amsd-pipeline",
3
- "version": "1.10.0",
3
+ "version": "1.12.0",
4
4
  "description": "Installer for the amsd-pipeline orchestration stack. Clones, packages and provisions the full stack with one command — no separate git clone step.",
5
5
  "bin": {
6
6
  "amsd-pipeline": "bin/amsd-pipeline.js"