amsd-pipeline 1.9.0 → 1.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/install.sh +132 -3
  2. package/package.json +1 -1
package/install.sh CHANGED
@@ -93,6 +93,16 @@ if [ "$UNINSTALL" = "1" ]; then
93
93
  exit 1
94
94
  fi
95
95
  fi
96
+ # STOP THE RUNNER HOST FIRST — it is NOT docker-related (a plain host process), so it must not
97
+ # be skipped by the "no container runtime, nothing to uninstall" exit below.
98
+ _UN_RH_PIDFILE="$_UN_ROOT/launch-dashboard/.runner-host.pid"
99
+ _UN_RH_WAS_RUNNING=""
100
+ [ -f "$_UN_RH_PIDFILE" ] && _UN_RH_WAS_RUNNING="$(cat "$_UN_RH_PIDFILE" 2>/dev/null)"
101
+ . "$INSTALLER_DIR/lib/runner-host-control.sh"
102
+ if stop_runner_host "$_UN_ROOT/launch-dashboard" && [ -n "$_UN_RH_WAS_RUNNING" ]; then
103
+ _ok "stopped runner-host (pid $_UN_RH_WAS_RUNNING)"
104
+ fi
105
+
96
106
  . "$INSTALLER_DIR/lib/isolated-compose-identity.sh"
97
107
  if [ -f "$INSTALLER_DIR/lib/container-runtime.sh" ]; then
98
108
  . "$INSTALLER_DIR/lib/container-runtime.sh"
@@ -204,14 +214,33 @@ if [ -n "$DEST" ]; then
204
214
  [ -n "$_excl" ] && _RUN_STATE_EXCLUDES+=("$_excl")
205
215
  done < <(run_state_exclude_args "$INSTALLER_DIR/run-state-paths.json")
206
216
  fi
217
+
218
+ # AN UPDATE MUST NEVER OVERWRITE AN OPERATOR'S EXISTING PROJECT CONFIG EITHER — a DIFFERENT
219
+ # mechanism from the excludes above, deliberately: a blanket exclude would also block a
220
+ # brand-new project's config.env from ever being extracted the first time a later ref adds
221
+ # one. Snapshot whatever already exists now, extract, then restore it over what the ref just
222
+ # wrote — so a first install (or a genuinely new project) is unaffected either way.
223
+ _OPCFG_TMP=""
224
+ if [ -f "$INSTALLER_DIR/operator-config-paths.json" ]; then
225
+ . "$INSTALLER_DIR/lib/preserve-operator-config.sh"
226
+ _OPCFG_TMP="$(mktemp -d)"
227
+ snapshot_operator_config "$DEST" "$INSTALLER_DIR/operator-config-paths.json" "$_OPCFG_TMP"
228
+ fi
229
+
207
230
  # The ${arr[@]+"${arr[@]}"} form, not bare "${arr[@]}": bash <4.4 (macOS ships 3.2 by default,
208
231
  # GPLv3 licensing) throws "unbound variable" under `set -u` expanding an empty array the plain
209
232
  # way. This form is safe on every bash this installer might run under.
210
233
  if ! git -C "$_GIT_ROOT" archive "$_PKG_REF" \
211
234
  | tar -x -C "$DEST" "${_RUN_STATE_EXCLUDES[@]+"${_RUN_STATE_EXCLUDES[@]}"}"; then
212
235
  _bad "packaging '$_PKG_REF' into $DEST failed"
236
+ [ -n "$_OPCFG_TMP" ] && rm -rf "$_OPCFG_TMP"
213
237
  exit 1
214
238
  fi
239
+
240
+ if [ -n "$_OPCFG_TMP" ]; then
241
+ restore_operator_config "$DEST" "$_OPCFG_TMP"
242
+ rm -rf "$_OPCFG_TMP"
243
+ fi
215
244
  _ok "packaged $_PKG_REF into $DEST"
216
245
  ROOT="$DEST"
217
246
  CONFIG="$ROOT/orchestrations/config"
@@ -519,6 +548,24 @@ esac
519
548
 
520
549
  # ── Dashboards: OPTIONAL, and never a reason to fail ────────────────────────
521
550
  _head "Dashboards (optional)"
551
+
552
+ # BUILD BEFORE UP, OR THE HEALTHCHECK FAILS BY CONSTRUCTION. orchestrations/dashboards/live/ is
553
+ # gitignored — eleventy's own build output, never tracked — so on every fresh install it starts
554
+ # EMPTY. agent-monitor's healthcheck probes `/`, nginx has no index and autoindex is off, so a
555
+ # brand-new install was 403-unhealthy FOREVER regardless of docker, subnet or port: found live
556
+ # 2026-09-03, grafana (depends_on agent-monitor: condition service_healthy) never even started —
557
+ # stuck at "Created". Only a completed pipeline run (or this build) ever populated live/ before.
558
+ #
559
+ # Only when there is something to build FROM (src/ present, same test the Build section above
560
+ # uses) — a packaged, src/-less install ships no eleventy at all.
561
+ if [ -d "$ROOT/src" ] && [ -f "$ROOT/package.json" ]; then
562
+ if (cd "$ROOT" && npm run dashboards:build --silent >/dev/null 2>&1); then
563
+ _ok "dashboards built — agent-monitor has real content to serve"
564
+ else
565
+ _warn "dashboards:build failed — agent-monitor's healthcheck may fail until a run populates it"
566
+ fi
567
+ fi
568
+
522
569
  # THE PROBE ASKS THE RESOLVED RUNTIME. It said `docker` literally, so on a podman-only machine the
523
570
  # installer announced "runtime: podman" and then started nothing — the report and the behaviour
524
571
  # disagreeing, which is this file's recurring defect.
@@ -577,6 +624,19 @@ compose_up() {
577
624
  rm -f "$_log" 2>/dev/null
578
625
  return 1
579
626
  fi
627
+ # PERSISTED so `pipeline-services.sh --start` can bring this exact stack back up later
628
+ # (after a WSL restart, a deliberate stop) WITHOUT re-rolling a different subnet/port —
629
+ # `down` (no -v) removes the network, so a later `up` with no env at all would fall back to
630
+ # the compose file's own default subnet and could collide with the dev stack or another
631
+ # install. Written fresh on every successful compose_up() — this IS the current identity.
632
+ {
633
+ printf 'OBS_PROJECT=%s\n' "$_OBS_PROJECT"
634
+ printf 'OBS_SUBNET=%s\n' "$_subnet"
635
+ printf 'OBS_CLICKHOUSE_PORT=%s\n' "$((8123 + _off))"
636
+ printf 'OBS_LANGFUSE_PORT=%s\n' "$((3100 + _off))"
637
+ printf 'OBS_DASHBOARD_PORT=%s\n' "$((8092 + _off))"
638
+ printf 'OBS_GRAFANA_PORT=%s\n' "$((3001 + _off))"
639
+ } > "$ROOT/.pipeline-services-state.env"
580
640
  rm -f "$_log" 2>/dev/null
581
641
  return 0
582
642
  }
@@ -615,25 +675,69 @@ else
615
675
  . "$INSTALLER_DIR/lib/wait-for-health.sh"
616
676
  . "$INSTALLER_DIR/lib/isolated-compose-identity.sh"
617
677
 
618
- # A REAL PASSWORD IS A DECISION ONLY A HUMAN MAKES — never synthesized here. Mirrors the root
619
- # .env handling: copy the template so there is something to fill in, never invent a secret.
678
+ # A VENDOR/API CREDENTIAL IS A DECISION ONLY A HUMAN MAKES — never synthesized (see the root
679
+ # .env handling: copy the template, never invent a secret). LAUNCH_PASSWORD is a DIFFERENT
680
+ # risk class: it gates a loopback-only local UI, not a billed vendor account or a shared
681
+ # system — a blank one previously meant install.sh already knew this stack could not start
682
+ # (it had just written this exact warning) and then attempted `up -d` anyway, hard-failing on
683
+ # compose's `${LAUNCH_PASSWORD:?...}` interpolation instead of the warning it already gave.
684
+ #
685
+ # Operator decision 2026-09-04, SUPERSEDING the 2026-09-03 random-generation decision: a
686
+ # RANDOM value made every install's password unknowable without reading the file, and
687
+ # unrecoverable once a running container had it in memory — copying a newer .env over an
688
+ # older one (exactly what moving credentials between two installs looks like) silently
689
+ # desynced the file from the live process, and the only fix was a manual docker restart.
690
+ # A FIXED, KNOWN default removes the ambiguity entirely: every fresh install starts on the
691
+ # same well-known password, printed here and in the file either way, and the change-password
692
+ # flow (Flutter UI, not this script) is how an operator actually secures it afterward.
693
+ _DEFAULT_LAUNCH_PW="abcd1234"
620
694
  if [ ! -f "$LAUNCH_DIR/.env" ]; then
621
695
  if [ -f "$LAUNCH_DIR/.env.example" ]; then
622
696
  cp "$LAUNCH_DIR/.env.example" "$LAUNCH_DIR/.env"
623
- _warn "launch-dashboard/.env created from .env.example — FILL IN LAUNCH_PASSWORD before it can start"
697
+ # REPLACE the template's blank line in place — never append a second
698
+ # LAUNCH_PASSWORD= key. Both parse fine (bash sourcing takes the last one) but a
699
+ # duplicate key is a needless trap for whoever reads this file by hand next.
700
+ if grep -q '^LAUNCH_PASSWORD=' "$LAUNCH_DIR/.env"; then
701
+ _LD_TMP="$(mktemp)"
702
+ sed "s|^LAUNCH_PASSWORD=.*|LAUNCH_PASSWORD=$_DEFAULT_LAUNCH_PW|" "$LAUNCH_DIR/.env" > "$_LD_TMP" \
703
+ && mv "$_LD_TMP" "$LAUNCH_DIR/.env"
704
+ else
705
+ printf '\nLAUNCH_PASSWORD=%s\n' "$_DEFAULT_LAUNCH_PW" >> "$LAUNCH_DIR/.env"
706
+ fi
707
+ _ok "launch-dashboard/.env created with the default LAUNCH_PASSWORD"
708
+ printf ' LAUNCH_PASSWORD: %s\n' "$_DEFAULT_LAUNCH_PW"
709
+ printf ' CHANGE THIS after your first login — it is the same on every fresh install.\n'
624
710
  else
625
711
  _bad "launch-dashboard/.env is missing and there is no .env.example to create one from"
626
712
  FAILED=1
627
713
  fi
628
714
  fi
629
715
 
716
+ # PRE-CREATE THE BIND-MOUNT SOURCES, AS THE HOST USER, BEFORE DOCKER EVER SEES THEM.
717
+ #
718
+ # The compose file's own comment above (services.launch-api) already names this exact trap: a
719
+ # bind mount onto a directory that doesn't exist yet gets auto-created BY DOCKER, as root — and
720
+ # launch-api's own `user: "${LAUNCH_UID:-1000}:..."` then cannot write to it. Found live
721
+ # 2026-09-03 against a genuinely fresh install: "unable to open database file", launch-api
722
+ # crash-looping, nginx's launch-ui reporting "host not found in upstream" as a downstream
723
+ # symptom of the crash — same bug CLASS as the dashboards live/ directory fixed above, here for
724
+ # ./data and ./spool specifically.
725
+ mkdir -p "$LAUNCH_DIR/data" "$LAUNCH_DIR/spool"
726
+
630
727
  _LD_PORT="$(grep -E '^LAUNCH_UI_PORT=' "$LAUNCH_DIR/.env" 2>/dev/null | tail -1 | cut -d= -f2)"
631
728
  _LD_PORT="${_LD_PORT:-8099}"
632
729
  _LD_PROJECT="$(isolated_project_name "$ROOT" launch)"
633
730
  _LD_HEALTH_URL="http://localhost:${_LD_PORT}/api/health"
634
731
 
732
+ _LD_PW="$(grep -E '^LAUNCH_PASSWORD=' "$LAUNCH_DIR/.env" 2>/dev/null | tail -1 | cut -d= -f2-)"
635
733
  if [ ! -f "$LAUNCH_DIR/.env" ]; then
636
734
  LAUNCH_STATUS=failed
735
+ elif [ -z "$_LD_PW" ] && [ "$CHECK_ONLY" != "1" ]; then
736
+ # KNOWN ALREADY, NEVER A SURPRISE CRASH. A .env from before LAUNCH_PASSWORD was
737
+ # auto-generated (or one an operator deliberately blanked) still fails compose's
738
+ # `${LAUNCH_PASSWORD:?...}` interpolation — skip the attempt instead of hitting it.
739
+ LAUNCH_STATUS=failed
740
+ _warn "launch-dashboard/.env has no LAUNCH_PASSWORD — skipping start; set one and re-run to bring it up"
637
741
  elif [ "$CHECK_ONLY" = "1" ]; then
638
742
  if wait_for_health "$_LD_HEALTH_URL" 3 1; then
639
743
  LAUNCH_STATUS=up
@@ -683,6 +787,13 @@ else
683
787
  elif wait_for_health "$_LD_HEALTH_URL" "$LAUNCH_HEALTH_TRIES" "$LAUNCH_HEALTH_INTERVAL"; then
684
788
  LAUNCH_STATUS=up
685
789
  _ok "up and healthy at $_LD_HEALTH_URL (project: $_LD_PROJECT, subnet: $_LD_SUBNET)"
790
+ # Same reason as the observability stack's own state file — appended, not truncated:
791
+ # that one is always written first in a single install.sh run.
792
+ {
793
+ printf 'LAUNCH_PROJECT=%s\n' "$_LD_PROJECT"
794
+ printf 'LAUNCH_SUBNET=%s\n' "$_LD_SUBNET"
795
+ printf 'LAUNCH_UI_PORT=%s\n' "$_LD_PORT"
796
+ } >> "$ROOT/.pipeline-services-state.env"
686
797
  else
687
798
  LAUNCH_STATUS=unhealthy
688
799
  _bad "containers started but never answered healthy at $_LD_HEALTH_URL"
@@ -692,6 +803,24 @@ else
692
803
  fi
693
804
  fi
694
805
 
806
+ # ── Runner host: what actually launches a pipeline run from the dashboard ────
807
+ # "the install script must start all services" (operator, 2026-09-04) — a saved launch request
808
+ # sat "pending" forever with nothing polling for it, because nothing ever started this.
809
+ #
810
+ # NOT DOCKERIZED, DELIBERATELY — same as runner-host.js's own header says: "a container cannot
811
+ # exec a host process." It spawns the pipeline's real launcher on the HOST, which needs git access
812
+ # to the codeline root, the claude/codemie-claude CLI's host auth (~/.claude — a container has none
813
+ # of this unless it were bind-mounted in), and host git credentials for anything that pushes.
814
+ # Containerizing the poll loop alone is easy (the spool it watches is already bind-mounted into
815
+ # launch-api); containerizing what it SPAWNS on a hit would mean containerizing the whole pipeline.
816
+ _head "Runner host (launches pipeline runs the dashboard queues)"
817
+ if [ "$LAUNCH_STATUS" = "up" ]; then
818
+ . "$INSTALLER_DIR/lib/runner-host-control.sh"
819
+ start_runner_host "$ROOT" "$LAUNCH_DIR" || FAILED=1
820
+ else
821
+ _ok "skipped — launch dashboard status is '$LAUNCH_STATUS', nothing to poll for"
822
+ fi
823
+
695
824
  # ── The command people will actually type ───────────────────────────────────
696
825
  # ── What this install IS ──────────────────────────────────────────────────────
697
826
  # An install whose mode can only be inferred from which containers happen to be running is an
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "amsd-pipeline",
3
- "version": "1.9.0",
3
+ "version": "1.11.0",
4
4
  "description": "Installer for the amsd-pipeline orchestration stack. Clones, packages and provisions the full stack with one command — no separate git clone step.",
5
5
  "bin": {
6
6
  "amsd-pipeline": "bin/amsd-pipeline.js"