amsd-pipeline 1.9.0 → 1.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/install.sh +132 -3
- package/package.json +1 -1
package/install.sh
CHANGED
|
@@ -93,6 +93,16 @@ if [ "$UNINSTALL" = "1" ]; then
|
|
|
93
93
|
exit 1
|
|
94
94
|
fi
|
|
95
95
|
fi
|
|
96
|
+
# STOP THE RUNNER HOST FIRST — it is NOT docker-related (a plain host process), so it must not
|
|
97
|
+
# be skipped by the "no container runtime, nothing to uninstall" exit below.
|
|
98
|
+
_UN_RH_PIDFILE="$_UN_ROOT/launch-dashboard/.runner-host.pid"
|
|
99
|
+
_UN_RH_WAS_RUNNING=""
|
|
100
|
+
[ -f "$_UN_RH_PIDFILE" ] && _UN_RH_WAS_RUNNING="$(cat "$_UN_RH_PIDFILE" 2>/dev/null)"
|
|
101
|
+
. "$INSTALLER_DIR/lib/runner-host-control.sh"
|
|
102
|
+
if stop_runner_host "$_UN_ROOT/launch-dashboard" && [ -n "$_UN_RH_WAS_RUNNING" ]; then
|
|
103
|
+
_ok "stopped runner-host (pid $_UN_RH_WAS_RUNNING)"
|
|
104
|
+
fi
|
|
105
|
+
|
|
96
106
|
. "$INSTALLER_DIR/lib/isolated-compose-identity.sh"
|
|
97
107
|
if [ -f "$INSTALLER_DIR/lib/container-runtime.sh" ]; then
|
|
98
108
|
. "$INSTALLER_DIR/lib/container-runtime.sh"
|
|
@@ -204,14 +214,33 @@ if [ -n "$DEST" ]; then
|
|
|
204
214
|
[ -n "$_excl" ] && _RUN_STATE_EXCLUDES+=("$_excl")
|
|
205
215
|
done < <(run_state_exclude_args "$INSTALLER_DIR/run-state-paths.json")
|
|
206
216
|
fi
|
|
217
|
+
|
|
218
|
+
# AN UPDATE MUST NEVER OVERWRITE AN OPERATOR'S EXISTING PROJECT CONFIG EITHER — a DIFFERENT
|
|
219
|
+
# mechanism from the excludes above, deliberately: a blanket exclude would also block a
|
|
220
|
+
# brand-new project's config.env from ever being extracted the first time a later ref adds
|
|
221
|
+
# one. Snapshot whatever already exists now, extract, then restore it over what the ref just
|
|
222
|
+
# wrote — so a first install (or a genuinely new project) is unaffected either way.
|
|
223
|
+
_OPCFG_TMP=""
|
|
224
|
+
if [ -f "$INSTALLER_DIR/operator-config-paths.json" ]; then
|
|
225
|
+
. "$INSTALLER_DIR/lib/preserve-operator-config.sh"
|
|
226
|
+
_OPCFG_TMP="$(mktemp -d)"
|
|
227
|
+
snapshot_operator_config "$DEST" "$INSTALLER_DIR/operator-config-paths.json" "$_OPCFG_TMP"
|
|
228
|
+
fi
|
|
229
|
+
|
|
207
230
|
# The ${arr[@]+"${arr[@]}"} form, not bare "${arr[@]}": bash <4.4 (macOS ships 3.2 by default,
|
|
208
231
|
# GPLv3 licensing) throws "unbound variable" under `set -u` expanding an empty array the plain
|
|
209
232
|
# way. This form is safe on every bash this installer might run under.
|
|
210
233
|
if ! git -C "$_GIT_ROOT" archive "$_PKG_REF" \
|
|
211
234
|
| tar -x -C "$DEST" "${_RUN_STATE_EXCLUDES[@]+"${_RUN_STATE_EXCLUDES[@]}"}"; then
|
|
212
235
|
_bad "packaging '$_PKG_REF' into $DEST failed"
|
|
236
|
+
[ -n "$_OPCFG_TMP" ] && rm -rf "$_OPCFG_TMP"
|
|
213
237
|
exit 1
|
|
214
238
|
fi
|
|
239
|
+
|
|
240
|
+
if [ -n "$_OPCFG_TMP" ]; then
|
|
241
|
+
restore_operator_config "$DEST" "$_OPCFG_TMP"
|
|
242
|
+
rm -rf "$_OPCFG_TMP"
|
|
243
|
+
fi
|
|
215
244
|
_ok "packaged $_PKG_REF into $DEST"
|
|
216
245
|
ROOT="$DEST"
|
|
217
246
|
CONFIG="$ROOT/orchestrations/config"
|
|
@@ -519,6 +548,24 @@ esac
|
|
|
519
548
|
|
|
520
549
|
# ── Dashboards: OPTIONAL, and never a reason to fail ────────────────────────
|
|
521
550
|
_head "Dashboards (optional)"
|
|
551
|
+
|
|
552
|
+
# BUILD BEFORE UP, OR THE HEALTHCHECK FAILS BY CONSTRUCTION. orchestrations/dashboards/live/ is
|
|
553
|
+
# gitignored — eleventy's own build output, never tracked — so on every fresh install it starts
|
|
554
|
+
# EMPTY. agent-monitor's healthcheck probes `/`, nginx has no index and autoindex is off, so a
|
|
555
|
+
# brand-new install was 403-unhealthy FOREVER regardless of docker, subnet or port: found live
|
|
556
|
+
# 2026-09-03, grafana (depends_on agent-monitor: condition service_healthy) never even started —
|
|
557
|
+
# stuck at "Created". Only a completed pipeline run (or this build) ever populated live/ before.
|
|
558
|
+
#
|
|
559
|
+
# Only when there is something to build FROM (src/ present, same test the Build section above
|
|
560
|
+
# uses) — a packaged, src/-less install ships no eleventy at all.
|
|
561
|
+
if [ -d "$ROOT/src" ] && [ -f "$ROOT/package.json" ]; then
|
|
562
|
+
if (cd "$ROOT" && npm run dashboards:build --silent >/dev/null 2>&1); then
|
|
563
|
+
_ok "dashboards built — agent-monitor has real content to serve"
|
|
564
|
+
else
|
|
565
|
+
_warn "dashboards:build failed — agent-monitor's healthcheck may fail until a run populates it"
|
|
566
|
+
fi
|
|
567
|
+
fi
|
|
568
|
+
|
|
522
569
|
# THE PROBE ASKS THE RESOLVED RUNTIME. It said `docker` literally, so on a podman-only machine the
|
|
523
570
|
# installer announced "runtime: podman" and then started nothing — the report and the behaviour
|
|
524
571
|
# disagreeing, which is this file's recurring defect.
|
|
@@ -577,6 +624,19 @@ compose_up() {
|
|
|
577
624
|
rm -f "$_log" 2>/dev/null
|
|
578
625
|
return 1
|
|
579
626
|
fi
|
|
627
|
+
# PERSISTED so `pipeline-services.sh --start` can bring this exact stack back up later
|
|
628
|
+
# (after a WSL restart, a deliberate stop) WITHOUT re-rolling a different subnet/port —
|
|
629
|
+
# `down` (no -v) removes the network, so a later `up` with no env at all would fall back to
|
|
630
|
+
# the compose file's own default subnet and could collide with the dev stack or another
|
|
631
|
+
# install. Written fresh on every successful compose_up() — this IS the current identity.
|
|
632
|
+
{
|
|
633
|
+
printf 'OBS_PROJECT=%s\n' "$_OBS_PROJECT"
|
|
634
|
+
printf 'OBS_SUBNET=%s\n' "$_subnet"
|
|
635
|
+
printf 'OBS_CLICKHOUSE_PORT=%s\n' "$((8123 + _off))"
|
|
636
|
+
printf 'OBS_LANGFUSE_PORT=%s\n' "$((3100 + _off))"
|
|
637
|
+
printf 'OBS_DASHBOARD_PORT=%s\n' "$((8092 + _off))"
|
|
638
|
+
printf 'OBS_GRAFANA_PORT=%s\n' "$((3001 + _off))"
|
|
639
|
+
} > "$ROOT/.pipeline-services-state.env"
|
|
580
640
|
rm -f "$_log" 2>/dev/null
|
|
581
641
|
return 0
|
|
582
642
|
}
|
|
@@ -615,25 +675,69 @@ else
|
|
|
615
675
|
. "$INSTALLER_DIR/lib/wait-for-health.sh"
|
|
616
676
|
. "$INSTALLER_DIR/lib/isolated-compose-identity.sh"
|
|
617
677
|
|
|
618
|
-
# A
|
|
619
|
-
# .env handling: copy the template
|
|
678
|
+
# A VENDOR/API CREDENTIAL IS A DECISION ONLY A HUMAN MAKES — never synthesized (see the root
|
|
679
|
+
# .env handling: copy the template, never invent a secret). LAUNCH_PASSWORD is a DIFFERENT
|
|
680
|
+
# risk class: it gates a loopback-only local UI, not a billed vendor account or a shared
|
|
681
|
+
# system — a blank one previously meant install.sh already knew this stack could not start
|
|
682
|
+
# (it had just written this exact warning) and then attempted `up -d` anyway, hard-failing on
|
|
683
|
+
# compose's `${LAUNCH_PASSWORD:?...}` interpolation instead of the warning it already gave.
|
|
684
|
+
#
|
|
685
|
+
# Operator decision 2026-09-04, SUPERSEDING the 2026-09-03 random-generation decision: a
|
|
686
|
+
# RANDOM value made every install's password unknowable without reading the file, and
|
|
687
|
+
# unrecoverable once a running container had it in memory — copying a newer .env over an
|
|
688
|
+
# older one (exactly what moving credentials between two installs looks like) silently
|
|
689
|
+
# desynced the file from the live process, and the only fix was a manual docker restart.
|
|
690
|
+
# A FIXED, KNOWN default removes the ambiguity entirely: every fresh install starts on the
|
|
691
|
+
# same well-known password, printed here and in the file either way, and the change-password
|
|
692
|
+
# flow (Flutter UI, not this script) is how an operator actually secures it afterward.
|
|
693
|
+
_DEFAULT_LAUNCH_PW="abcd1234"
|
|
620
694
|
if [ ! -f "$LAUNCH_DIR/.env" ]; then
|
|
621
695
|
if [ -f "$LAUNCH_DIR/.env.example" ]; then
|
|
622
696
|
cp "$LAUNCH_DIR/.env.example" "$LAUNCH_DIR/.env"
|
|
623
|
-
|
|
697
|
+
# REPLACE the template's blank line in place — never append a second
|
|
698
|
+
# LAUNCH_PASSWORD= key. Both parse fine (bash sourcing takes the last one) but a
|
|
699
|
+
# duplicate key is a needless trap for whoever reads this file by hand next.
|
|
700
|
+
if grep -q '^LAUNCH_PASSWORD=' "$LAUNCH_DIR/.env"; then
|
|
701
|
+
_LD_TMP="$(mktemp)"
|
|
702
|
+
sed "s|^LAUNCH_PASSWORD=.*|LAUNCH_PASSWORD=$_DEFAULT_LAUNCH_PW|" "$LAUNCH_DIR/.env" > "$_LD_TMP" \
|
|
703
|
+
&& mv "$_LD_TMP" "$LAUNCH_DIR/.env"
|
|
704
|
+
else
|
|
705
|
+
printf '\nLAUNCH_PASSWORD=%s\n' "$_DEFAULT_LAUNCH_PW" >> "$LAUNCH_DIR/.env"
|
|
706
|
+
fi
|
|
707
|
+
_ok "launch-dashboard/.env created with the default LAUNCH_PASSWORD"
|
|
708
|
+
printf ' LAUNCH_PASSWORD: %s\n' "$_DEFAULT_LAUNCH_PW"
|
|
709
|
+
printf ' CHANGE THIS after your first login — it is the same on every fresh install.\n'
|
|
624
710
|
else
|
|
625
711
|
_bad "launch-dashboard/.env is missing and there is no .env.example to create one from"
|
|
626
712
|
FAILED=1
|
|
627
713
|
fi
|
|
628
714
|
fi
|
|
629
715
|
|
|
716
|
+
# PRE-CREATE THE BIND-MOUNT SOURCES, AS THE HOST USER, BEFORE DOCKER EVER SEES THEM.
|
|
717
|
+
#
|
|
718
|
+
# The compose file's own comment above (services.launch-api) already names this exact trap: a
|
|
719
|
+
# bind mount onto a directory that doesn't exist yet gets auto-created BY DOCKER, as root — and
|
|
720
|
+
# launch-api's own `user: "${LAUNCH_UID:-1000}:..."` then cannot write to it. Found live
|
|
721
|
+
# 2026-09-03 against a genuinely fresh install: "unable to open database file", launch-api
|
|
722
|
+
# crash-looping, nginx's launch-ui reporting "host not found in upstream" as a downstream
|
|
723
|
+
# symptom of the crash — same bug CLASS as the dashboards live/ directory fixed above, here for
|
|
724
|
+
# ./data and ./spool specifically.
|
|
725
|
+
mkdir -p "$LAUNCH_DIR/data" "$LAUNCH_DIR/spool"
|
|
726
|
+
|
|
630
727
|
_LD_PORT="$(grep -E '^LAUNCH_UI_PORT=' "$LAUNCH_DIR/.env" 2>/dev/null | tail -1 | cut -d= -f2)"
|
|
631
728
|
_LD_PORT="${_LD_PORT:-8099}"
|
|
632
729
|
_LD_PROJECT="$(isolated_project_name "$ROOT" launch)"
|
|
633
730
|
_LD_HEALTH_URL="http://localhost:${_LD_PORT}/api/health"
|
|
634
731
|
|
|
732
|
+
_LD_PW="$(grep -E '^LAUNCH_PASSWORD=' "$LAUNCH_DIR/.env" 2>/dev/null | tail -1 | cut -d= -f2-)"
|
|
635
733
|
if [ ! -f "$LAUNCH_DIR/.env" ]; then
|
|
636
734
|
LAUNCH_STATUS=failed
|
|
735
|
+
elif [ -z "$_LD_PW" ] && [ "$CHECK_ONLY" != "1" ]; then
|
|
736
|
+
# KNOWN ALREADY, NEVER A SURPRISE CRASH. A .env from before LAUNCH_PASSWORD was
|
|
737
|
+
# auto-generated (or one an operator deliberately blanked) still fails compose's
|
|
738
|
+
# `${LAUNCH_PASSWORD:?...}` interpolation — skip the attempt instead of hitting it.
|
|
739
|
+
LAUNCH_STATUS=failed
|
|
740
|
+
_warn "launch-dashboard/.env has no LAUNCH_PASSWORD — skipping start; set one and re-run to bring it up"
|
|
637
741
|
elif [ "$CHECK_ONLY" = "1" ]; then
|
|
638
742
|
if wait_for_health "$_LD_HEALTH_URL" 3 1; then
|
|
639
743
|
LAUNCH_STATUS=up
|
|
@@ -683,6 +787,13 @@ else
|
|
|
683
787
|
elif wait_for_health "$_LD_HEALTH_URL" "$LAUNCH_HEALTH_TRIES" "$LAUNCH_HEALTH_INTERVAL"; then
|
|
684
788
|
LAUNCH_STATUS=up
|
|
685
789
|
_ok "up and healthy at $_LD_HEALTH_URL (project: $_LD_PROJECT, subnet: $_LD_SUBNET)"
|
|
790
|
+
# Same reason as the observability stack's own state file — appended, not truncated:
|
|
791
|
+
# that one is always written first in a single install.sh run.
|
|
792
|
+
{
|
|
793
|
+
printf 'LAUNCH_PROJECT=%s\n' "$_LD_PROJECT"
|
|
794
|
+
printf 'LAUNCH_SUBNET=%s\n' "$_LD_SUBNET"
|
|
795
|
+
printf 'LAUNCH_UI_PORT=%s\n' "$_LD_PORT"
|
|
796
|
+
} >> "$ROOT/.pipeline-services-state.env"
|
|
686
797
|
else
|
|
687
798
|
LAUNCH_STATUS=unhealthy
|
|
688
799
|
_bad "containers started but never answered healthy at $_LD_HEALTH_URL"
|
|
@@ -692,6 +803,24 @@ else
|
|
|
692
803
|
fi
|
|
693
804
|
fi
|
|
694
805
|
|
|
806
|
+
# ── Runner host: what actually launches a pipeline run from the dashboard ────
|
|
807
|
+
# "the install script must start all services" (operator, 2026-09-04) — a saved launch request
|
|
808
|
+
# sat "pending" forever with nothing polling for it, because nothing ever started this.
|
|
809
|
+
#
|
|
810
|
+
# NOT DOCKERIZED, DELIBERATELY — same as runner-host.js's own header says: "a container cannot
|
|
811
|
+
# exec a host process." It spawns the pipeline's real launcher on the HOST, which needs git access
|
|
812
|
+
# to the codeline root, the claude/codemie-claude CLI's host auth (~/.claude — a container has none
|
|
813
|
+
# of this unless it were bind-mounted in), and host git credentials for anything that pushes.
|
|
814
|
+
# Containerizing the poll loop alone is easy (the spool it watches is already bind-mounted into
|
|
815
|
+
# launch-api); containerizing what it SPAWNS on a hit would mean containerizing the whole pipeline.
|
|
816
|
+
_head "Runner host (launches pipeline runs the dashboard queues)"
|
|
817
|
+
if [ "$LAUNCH_STATUS" = "up" ]; then
|
|
818
|
+
. "$INSTALLER_DIR/lib/runner-host-control.sh"
|
|
819
|
+
start_runner_host "$ROOT" "$LAUNCH_DIR" || FAILED=1
|
|
820
|
+
else
|
|
821
|
+
_ok "skipped — launch dashboard status is '$LAUNCH_STATUS', nothing to poll for"
|
|
822
|
+
fi
|
|
823
|
+
|
|
695
824
|
# ── The command people will actually type ───────────────────────────────────
|
|
696
825
|
# ── What this install IS ──────────────────────────────────────────────────────
|
|
697
826
|
# An install whose mode can only be inferred from which containers happen to be running is an
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "amsd-pipeline",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.11.0",
|
|
4
4
|
"description": "Installer for the amsd-pipeline orchestration stack. Clones, packages and provisions the full stack with one command — no separate git clone step.",
|
|
5
5
|
"bin": {
|
|
6
6
|
"amsd-pipeline": "bin/amsd-pipeline.js"
|