@bongos/core 1.19.679 → 1.19.680
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.bongos-core.json +24 -14
- package/docs/adr/0279-an-upgrade-is-proven-by-the-served-version-not-the-health-check.md +48 -0
- package/docs/adr/README.md +1 -0
- package/docs/module-api-changelog.md +2 -0
- package/docs/recipes/instance-service-restart.md +89 -0
- package/package-lock.json +2 -2
- package/package.json +1 -1
- package/scripts/gds/upgrade.js +153 -17
- package/src/module-api.js +1 -1
- package/tests/upgrade.mjs +259 -0
package/.bongos-core.json
CHANGED
|
@@ -2,22 +2,22 @@
|
|
|
2
2
|
"artifact": "bongos-core",
|
|
3
3
|
"manifest_schema": 1,
|
|
4
4
|
"generator": "scripts/gds/package-core.js",
|
|
5
|
-
"core_version": "1.19.
|
|
6
|
-
"core_contract": "1.19.
|
|
7
|
-
"source_commit": "
|
|
5
|
+
"core_version": "1.19.680",
|
|
6
|
+
"core_contract": "1.19.680",
|
|
7
|
+
"source_commit": "e022c08bc3285327348cd482d38e0659e2ae8b3d",
|
|
8
8
|
"source_ref": "HEAD",
|
|
9
|
-
"built_at": "2026-09-
|
|
9
|
+
"built_at": "2026-09-12T01:35:27.332Z",
|
|
10
10
|
"redaction": {
|
|
11
11
|
"model": "docs-redacted+functional-verbatim",
|
|
12
|
-
"docs_redacted":
|
|
12
|
+
"docs_redacted": 476,
|
|
13
13
|
"agent_docs_stubbed": 24,
|
|
14
14
|
"functional_verbatim": 2124,
|
|
15
15
|
"rules": 3,
|
|
16
16
|
"gate_literals": 3,
|
|
17
17
|
"gate": "passed"
|
|
18
18
|
},
|
|
19
|
-
"file_count":
|
|
20
|
-
"tree_sha256": "
|
|
19
|
+
"file_count": 2624,
|
|
20
|
+
"tree_sha256": "912d1481802f12085195156ce06ab53a39ca1e159441ebee3efe401ddbedf56d",
|
|
21
21
|
"files": [
|
|
22
22
|
{
|
|
23
23
|
"path": ".claude/skills/ask-for-help/SKILL.md",
|
|
@@ -1904,10 +1904,15 @@
|
|
|
1904
1904
|
"mode": "0000644",
|
|
1905
1905
|
"sha256": "3d941c0e1f1ce5618fa6861a720a4ac1c1988495c5f17aed77a348a4a3b7b718"
|
|
1906
1906
|
},
|
|
1907
|
+
{
|
|
1908
|
+
"path": "docs/adr/0279-an-upgrade-is-proven-by-the-served-version-not-the-health-check.md",
|
|
1909
|
+
"mode": "0000644",
|
|
1910
|
+
"sha256": "46e65d06711ad356563b01b824c416d03cecf135f9f69a60e469710e91cb3f62"
|
|
1911
|
+
},
|
|
1907
1912
|
{
|
|
1908
1913
|
"path": "docs/adr/README.md",
|
|
1909
1914
|
"mode": "0000644",
|
|
1910
|
-
"sha256": "
|
|
1915
|
+
"sha256": "6766d2b792da65e6cebc79adf4261b710d6f5ce113a8a84953cf1c643e20a0b7"
|
|
1911
1916
|
},
|
|
1912
1917
|
{
|
|
1913
1918
|
"path": "docs/api-reference.md",
|
|
@@ -2792,7 +2797,7 @@
|
|
|
2792
2797
|
{
|
|
2793
2798
|
"path": "docs/module-api-changelog.md",
|
|
2794
2799
|
"mode": "0000644",
|
|
2795
|
-
"sha256": "
|
|
2800
|
+
"sha256": "592be55bbe17442266b4478ca962fcc0bdc4cbd66718c6b4af13a1cca911cc3e"
|
|
2796
2801
|
},
|
|
2797
2802
|
{
|
|
2798
2803
|
"path": "docs/modules-contract.md",
|
|
@@ -2949,6 +2954,11 @@
|
|
|
2949
2954
|
"mode": "0000644",
|
|
2950
2955
|
"sha256": "259148d6e612209502b3f46e82e57c62da049fac4d883755e151be37de92fbf7"
|
|
2951
2956
|
},
|
|
2957
|
+
{
|
|
2958
|
+
"path": "docs/recipes/instance-service-restart.md",
|
|
2959
|
+
"mode": "0000644",
|
|
2960
|
+
"sha256": "77b4f5e695ee839b6e9f7e192af6f13114069a114cc2f63167582a4a8768f4ce"
|
|
2961
|
+
},
|
|
2952
2962
|
{
|
|
2953
2963
|
"path": "docs/recipes/local-dev.md",
|
|
2954
2964
|
"mode": "0000644",
|
|
@@ -7767,12 +7777,12 @@
|
|
|
7767
7777
|
{
|
|
7768
7778
|
"path": "package-lock.json",
|
|
7769
7779
|
"mode": "0000644",
|
|
7770
|
-
"sha256": "
|
|
7780
|
+
"sha256": "f0fd6366abe9523c64332de417406bfe0ab6f6872abc6e651debbe5f1941cad0"
|
|
7771
7781
|
},
|
|
7772
7782
|
{
|
|
7773
7783
|
"path": "package.json",
|
|
7774
7784
|
"mode": "0000644",
|
|
7775
|
-
"sha256": "
|
|
7785
|
+
"sha256": "60cff4653b02fd38c3c7fb8673cb423a57f8360955578a6a345ccaef9bf1ac9a"
|
|
7776
7786
|
},
|
|
7777
7787
|
{
|
|
7778
7788
|
"path": "public-docs/index.html",
|
|
@@ -9102,7 +9112,7 @@
|
|
|
9102
9112
|
{
|
|
9103
9113
|
"path": "scripts/gds/upgrade.js",
|
|
9104
9114
|
"mode": "0000644",
|
|
9105
|
-
"sha256": "
|
|
9115
|
+
"sha256": "642d0ade55c203d6b07478a07d3c21c26fa4635dc8a92e4fe989434e635abbbb"
|
|
9106
9116
|
},
|
|
9107
9117
|
{
|
|
9108
9118
|
"path": "scripts/gds/validate-design.js",
|
|
@@ -9532,7 +9542,7 @@
|
|
|
9532
9542
|
{
|
|
9533
9543
|
"path": "src/module-api.js",
|
|
9534
9544
|
"mode": "0000644",
|
|
9535
|
-
"sha256": "
|
|
9545
|
+
"sha256": "19014335bfaede6fba2e6ff79ba1fe746a2149345e25448b2ed0176bdb8a6649"
|
|
9536
9546
|
},
|
|
9537
9547
|
{
|
|
9538
9548
|
"path": "src/module-loader/catalog.js",
|
|
@@ -12977,7 +12987,7 @@
|
|
|
12977
12987
|
{
|
|
12978
12988
|
"path": "tests/upgrade.mjs",
|
|
12979
12989
|
"mode": "0000644",
|
|
12980
|
-
"sha256": "
|
|
12990
|
+
"sha256": "106735a9ab28b4a23e2f25269030d982798da3fa39e31d40a1ec78277ac971bc"
|
|
12981
12991
|
},
|
|
12982
12992
|
{
|
|
12983
12993
|
"path": "tests/upgrade_persist_pin.mjs",
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# ADR 0279 — An upgrade is proven by the served version, not by a health check
|
|
2
|
+
|
|
3
|
+
- **Status:** accepted
|
|
4
|
+
- **Date:** 2026-09-11
|
|
5
|
+
- **Task:** [task 1002884](https://cloudbongos.com/builders#/task/1002884) (BONGOS-V2, goal 1000090 — *Working area 4, Bongos Core distribution*), from idea 1000682
|
|
6
|
+
- **Extends** the mandatory post-restart health check of task 1002222 (audit H9, finding F5) and the auto-rollback of task 2149, whose unattended policy is [ADR 0136](<redacted>.md) §3. The health check stays; this says why it was never sufficient on its own.
|
|
7
|
+
|
|
8
|
+
## Context
|
|
9
|
+
|
|
10
|
+
On 2026-08-11 the auto-upgrade sweep on cloudbongos.com printed `✓ upgrade complete — core 1.19.13 → 1.19.56`, wrote a success row to the `core_upgrades` ledger and exited 0. Live kept serving **1.19.13**. Every monitoring surface agreed the upgrade had landed. None of them had looked at the running process.
|
|
11
|
+
|
|
12
|
+
Three things had to line up, and all three were in the shipped code:
|
|
13
|
+
|
|
14
|
+
1. **The restart failed, and that was a warning.** The sweep ran `systemctl restart cloudbongos.service` as user `lars` with no TTY. The unit is a `User=` unit, so the restart needed authorization polkit could only get interactively — `Interactive authentication required`, exit 1. `upgrade.js` logged `! restart command failed` and **fell through to verification**.
|
|
15
|
+
2. **The version check read the wrong thing.** `readInstalledCoreVersion()` reads `node_modules/@bongos/core/package.json` — the version on **disk**. `npm install` had genuinely put 1.19.56 there. Disk was right; the process was not.
|
|
16
|
+
3. **The health check could not tell the difference.** `pollHealth()` asks whether *something* answers with a 2xx. The old process was up and perfectly healthy, so it answered. A health check confirms a port is being served; it cannot confirm *what* is serving it.
|
|
17
|
+
|
|
18
|
+
Each check was individually reasonable. Together they were a closed loop that could not observe the failure they existed to catch, which is the worst shape a check can have: it reports success it has not proven, and a lying tool is more expensive than a broken one, because nothing goes looking.
|
|
19
|
+
|
|
20
|
+
The damage outlived the incident. `.claude/scheduled-tasks/core-update-subscription/subscribe.js` routes around `bongos upgrade` in favour of `go-live.js` and says so in a comment — *"upgrade.js's own health check is documented to false-pass by polling the still-running OLD process"*. A known-lying tool had become something the rest of the system worked around instead of fixing.
|
|
21
|
+
|
|
22
|
+
## Decision
|
|
23
|
+
|
|
24
|
+
**A bump is confirmed by asking the running process what version it is. Anything less is unconfirmed, and unconfirmed is not success.**
|
|
25
|
+
|
|
26
|
+
Three changes in `scripts/gds/upgrade.js`:
|
|
27
|
+
|
|
28
|
+
1. **A failed restart fails the upgrade.** It enters the same auto-rollback path as a failed install, migrate or health check, and returns non-zero. Rolling back rather than merely erroring keeps disk and process consistent: the previous core goes back on disk, matching the one that is (still) running. Leaving disk ahead of the process is the state that made the original incident invisible.
|
|
29
|
+
|
|
30
|
+
2. **The served version is read back from `/version`.** That route reports the live process's `coreVersion` (`serve-internal.js`, since 1.17.2) and is exempt from the prelaunch gate, so it is readable wherever `/healthz` is. The URL defaults to `/version` on the health URL's origin and can be overridden with `--version-url`. Deriving rather than requiring a new flag is deliberate: every existing call site passes only `--health-url`, and a check that must be opted into is off exactly where it is needed.
|
|
31
|
+
|
|
32
|
+
3. **A mismatch fails; an unreadable endpoint warns.** These are different facts and are treated differently. *"It says 1.19.13 and I asked for 1.19.56"* is proof of failure — roll back. *"It would not tell me"* is absence of proof: an instance may not expose the endpoint, and refusing every such bump would be a worse regression than the false-pass we are removing. Passing `--version-url` explicitly asks for proof, so there an unreadable endpoint **is** a failure — which is what the unattended subscription lane now does, since its roster already carries the exact URL.
|
|
33
|
+
|
|
34
|
+
**On escalation: `sudo -n`, not a hand-placed polkit rule.** `restartService()` retries a failed `systemctl restart` through `sudo -n` when not root — the idiom `dev.js` and `dev-lib.js` already use, and consistent with `provision.js`, which prefixes its own `systemctl` calls with `sudo`. `upgrade.js` was the one place that restarted a service without escalating, which is why only it hit the wall. The live mitigation at the time was a hand-written `/etc/polkit-1/rules.d/<redacted>.rules` on one box; that rule is **superseded**, because a rebuilt box inherits code and does not inherit hand-placed `/etc` files. `-n` fails fast rather than hanging on a password prompt, so the retry costs nothing where it isn't permitted. The polkit route is documented in [`docs/recipes/instance-service-restart.md`](../recipes/instance-service-restart.md) for hosts that prefer it or that deliberately withhold sudo.
|
|
35
|
+
|
|
36
|
+
## Consequences
|
|
37
|
+
|
|
38
|
+
- An upgrade that cannot restart the service now **fails loudly and reverts**, where it used to warn and claim success. This is the point, but it means a host with neither passwordless sudo nor a polkit rule will start seeing upgrades fail that previously "passed" — correctly. The error names both fixes.
|
|
39
|
+
- `--no-health-check` remains the single documented escape hatch and now waives the served-version read-back too, rather than adding a second flag to reason about.
|
|
40
|
+
- `subscribe.js` no longer needs to describe the direct-upgrade lane as having "no independent read-back"; the comment and the log line were corrected with the fix, since a stale comment about a fixed bug is how the next session re-learns a lie.
|
|
41
|
+
- The rollback path calls `restartService()` too, so it also escalates — a rollback on a box that needs sudo can now actually restart onto the restored core instead of silently leaving the failure in place.
|
|
42
|
+
|
|
43
|
+
## Rejected
|
|
44
|
+
|
|
45
|
+
- **Comparing `startedAt` instead of `coreVersion`.** A restart that swaps nothing still moves `startedAt` if the unit bounced, and a process that never restarted keeps an old one — but `startedAt` cannot distinguish "restarted onto the same (old) core" from "restarted onto the new one". `coreVersion` answers the actual question.
|
|
46
|
+
- **Making an unreadable endpoint fatal by default.** Fails closed in the wrong direction: it would break upgrades on instances that never served `/version`, to catch a case that a fatal restart failure already catches.
|
|
47
|
+
- **Requiring `--version-url` everywhere.** The check would then be absent from every current call site — the same "off where it matters" failure in a new costume.
|
|
48
|
+
- **Keeping the polkit rule as the answer.** It works, but it lives in `/etc` on one machine. The task's own framing is that a rebuilt box must inherit the fix; only code does that.
|
package/docs/adr/README.md
CHANGED
|
@@ -370,3 +370,4 @@ This keeps the decision history honest and traceable.
|
|
|
370
370
|
| 0276 | [**The skill-listing budget cannot hold 64 skills and every trigger, so the number is ratcheted and the choice goes to the owner** ([task 1003620](https://cloudbongos.com/builders#/task/1003620) · goal 1000095 — *Working area 6*, criterion `wa6-role-experience`). Every SKILL.md description is resident in EVERY session for every craft, and `skill-lint` budgets the whole listing at 8,000 chars. It has been trimmed twice ([#1003548], [#1003587]) and grown back both times, because the 400-char per-description aim was a WARNING THAT ACCUMULATES. This task rewrote 49 of the 50 `.claude/skills/` descriptions — 17 warnings to 0, listing 25,624 → 19,316 chars, about 1,570 tokens returned to every session — with every trigger phrase preserved and CHECKED MECHANICALLY (a checker diffed the quoted phrases against HEAD and caught five real losses, all restored; four dropped strings were UI labels, not triggers). **8,000 was not reached, and the arithmetic says it cannot be:** across 64 skills the names (880) plus the mandatory quoted triggers (~5,370) are an irreducible ~6,250, leaving ~27 chars per skill to say what the skill DOES. Reaching the budget means trigger-only descriptions — trading truncation for the loss of the semantic signal routing leans on when the user’s words do not literally match a trigger. **Decision: ratchet `skill_listing_chars` at 19,316** in `fitness-ratchets.js` (fails open if the linter cannot load), so it cannot grow while the real question is open; the baseline is deliberately ABOVE skill-lint’s budget and is not a claim the budget is met. **Open and owner-gated:** delist skills (64 is a menu), scale the budget with the roster, or accept trigger-only text — recommendation is delist then rescale, filed as a blocker. `fitness.js`’s over-budget warning STAYS, as the visible trace of that question. Rejected: raising `LISTING_BUDGET_CHARS` to silence the warning (deletes the signal, not the debt); re-cutting the 14 module ui-design descriptions a week after [#1003587] wrote them, for ~2,000 chars toward a target still 9,000 away.](<redacted>.md) | skills / context budget / routing |
|
|
371
371
|
| 0277 | [**A box is "in use" only while a human is attached, and the claim expires** ([task 1003507](https://cloudbongos.com/builders#/task/1003507) · goal 1000095 — *Working area 6*, criterion `wa6-role-experience`). `sweep-idle` skipped any box with `claude_active = true` at ANY age, and `claude_active` is a LATCH, not a level: only a heartbeat ping writes it, and `infra/box-heartbeat.sh` exits WITHOUT pinging when it sees nothing — so silence, the very signal the sweep exists to act on, could never clear it. Reproduced against the shipped selector: skipped at `idleMinutes` of 10, 120, 1440 and **5,256,000** (ten years). Not reaped late; never. The second leg is that `load > 0.2` kept `last_activity_at` bumping every 5 minutes anyway (a devcontainer plus an idle `claude` clears that floor on its own), so EITHER leg alone kept a box alive — which is why bounding only the veto looks like a fix and is not: the incident box's heartbeat was FRESH. Measured: `example-owner`'s box up since 2026-08-12, tmux `otb` UNATTACHED since 2026-08-15 18:01 UTC, `claude` burning 16 min of CPU across 25h of wall clock, **$20.21** of mostly-unattended compute. A THIRD defect, found while testing this and confirmed on clean `origin/main`, made the sweep inert regardless: `loadDeps()` read `_deps` before anything declared it, so it threw `ReferenceError` on first call and every command resolving a DigitalOcean client went with it — including `sweep-idle --apply`, which builds that client before the park loop. **The idle sweep could not park any box at all**, hidden because the suite's only `apply: true` test relied on the veto emptying the idle set before `makeDo()` was reached: one bug shielded by the other. **Decision: a box may not stay active longer than `BOX_UNATTENDED_MAX_HOURS` (12) without evidence a HUMAN was attached.** The heartbeat already computed that signal and folded it into one boolean; it now reports `attached` separately (login session, inbound SSH, open ttyd, or an ATTACHED tmux client via `#{session_attached}` — the signal that separates this box from a working one) and the server stamps `last_attached_at`. `claude_active` keeps its veto but it EXPIRES, bounded by a `claude_active_since` edge stamp. `pgrep -x claude` and the load floor remain reasons the box PINGS, never evidence anyone is THERE — a running process is not a person, and that distinction is the whole decision. The two `NULL` defaults deliberately DISAGREE: `last_attached_at` NULL means "no data" and keeps a pre-1003507 box on the old behaviour (core_238 pointedly does NOT backfill it — a backfilled `now()` starts a clock nothing can advance and parks every un-upgraded box one cap later, and box source sync is not prompt: idea 1000745 records 257 commits behind for three days), while `claude_active_since` NULL is REFUSED because an unknown latch age is the forever-latch itself, and is backfilled so the state is unreachable after deploy. 12h because parking is reversible since task 1002726 (snapshot kept), so a false positive costs one wake against $20.21 for no cap. Both clocks clear at park/wake/deprovision, or a woken box inherits an expired latch and is parked instantly — the fix reintroducing the bug from the far side. Rejected: tracking attachment INSTEAD of bounding the latch (the silent box keeps `true` forever, so the reported hole survives); bounding the latch alone (built first, and the verification probe caught it — the fresh heartbeat meant lifting the veto changed nothing); deleting the load floor (it is what makes an autonomous run count); measuring from `active_since` (that is uptime — parks a box worked on for days); raising `BOX_IDLE_MINUTES` (no threshold reaches an unbounded veto).](<redacted>.md) | dev box / cost / idle sweep |
|
|
372
372
|
| 0278 | [**A gated project still takes applications, and the exemption is scoped to the verb** ([task 1003525](https://cloudbongos.com/builders#/task/1003525) · the apply write itself in [task 1003624](https://cloudbongos.com/builders#/task/1003624) · goal 1000106 — *Working area 1, Project creation*; owner decision 2026-09-11). A project's owner sets who may SEE it (`platformVisibility`, [ADR 0192](<redacted>.md)) and who may JOIN it (`joinability`, [ADR 0194](<redacted>.md)) independently — and set to their middle values, members-only AND apply-to-join, the project took no applications at all: the member door refused every cookie-less request with `401` before the public `POST <api>/access-requests` could answer, because that write was not on the exempt list. The two settings composed into **"nobody can apply"**, which nobody chose. It survived because nothing LIED about it — the hub's join box relayed the project's own `401` honestly as `members_only`, and the hall's landing, where the apply form lives, is itself behind the door; the composition was simply unreachable. Found by the R14 proof ([task 1002333](https://cloudbongos.com/builders#/task/1002333)). ADR 0192 §3 had fixed the exempt list at "the door, the manifest, the probes and the downloads — and nothing wider" and left widening it as an owner call, which is what this is. **Decision: yes — and BOTH halves are exempted, each scoped to one path and one verb.** `POST <api>/access-requests` (it grants nothing — an application is a row in a queue the owner still reviews, [ADR 0201](<redacted>.md), already public on every non-gated project) and `GET <api>/access-requests/status` (without it the answer is half an answer: `bongos login` cannot re-poll the device flow after a `not_approved` — the `device_code` is spent — so an applicant would file a request and then wait on an approval they can never observe). `EXEMPT` entries may now be `{ re, methods }` beside the bare `RegExp`s, and `isExempt(path, method)` takes the verb as an OPTIONAL second argument that **fails closed** for a scoped entry when none is given, so the one-argument static callers cannot accidentally widen. **The verb is load-bearing, not tidiness:** the bare `GET` on `<api>/access-requests` is the OWNER'S QUEUE (`requireBuilder` + `access_request.review`), the surface listing would-be builders by name with their vouch state — a path-only exemption would have silently taken the member door off the front of it, leaving one layer where there were two, and the queue's own `requirePermission` still holding is exactly what makes that loss easy to miss. **It opens no oracle the gate was closing:** the status route's boolean twin `GET <api>/auth/web/admission-status` is ALREADY reachable on a gated project inside the `auth/*` subtree §3 exempts whole (§3 records that cost in as many words), and the two share ONE per-IP budget on purpose ([ADR 0209](<redacted>.md)) so neither can be alternated against the other. What it DOES add, stated as the honest cost: applicant detail — `pending`/`dismissed`/`none` over the twin's bare `admitted`. Whether that answer should collapse is ADR 0209's still-open owner question and is deliberately NOT decided here. No hub change: `joinRelayOutcome` maps the RELAYED status, so it carries the project's real answer the moment the `401` stops. Rejected: "gated means gated" — hide *Apply to join* and say so in the manage blurb (coherent, and the call went the other way); exempting the path without the verb; exempting the write alone; collapsing the status response while the route happened to be open (that is how a deferred decision gets made by accident).](<redacted>.md) | project visibility / join door / member door |
|
|
373
|
+
| 0279 | [**An upgrade is proven by the served version, not by a health check** ([task 1002884](https://cloudbongos.com/builders#/task/1002884) · goal 1000090 — *Working area 4, Bongos Core distribution*; from idea 1000682). On 2026-08-11 the auto-upgrade sweep printed `✓ upgrade complete — core 1.19.13 → 1.19.56`, wrote a success row to `core_upgrades` and exited 0 while live kept serving **1.19.13**. Three shipped checks formed a closed loop that could not see the failure they existed to catch: the `systemctl restart` failed with `Interactive authentication required` (a `User=` unit, no TTY) and `upgrade.js` treated it as a WARNING and fell through; `readInstalledCoreVersion()` then confirmed the version on **disk**, where `npm install` had correctly put it; and `pollHealth()` got a 200 from the **still-running old process**, because a health check confirms a port is served, never *what* serves it. A lying tool is worse than a broken one — nothing goes looking. The damage outlived the incident: `subscribe.js` had already routed the unattended lane around `bongos upgrade` in favour of `go-live.js`, citing this false-pass in a comment. **Decision: a bump is confirmed by asking the running process what version it is.** (1) A failed restart enters the same auto-rollback path as a failed install/migrate/health and exits non-zero — rolling back rather than merely erroring keeps disk and process consistent, since disk-ahead-of-process is the state that made the incident invisible. (2) The served version is read back from `/version` (`coreVersion`, since 1.17.2, prelaunch-gate exempt), defaulting to the `--health-url` origin and overridable via `--version-url` — derived rather than opt-in because every existing call site passes only `--health-url`, and a check you must opt into is off exactly where it is needed. (3) A **mismatch** fails (proof of failure → roll back); an **unreadable** endpoint only warns (absence of proof — refusing every such bump would regress harder than the false-pass), except under an explicit `--version-url`, which asks for proof and therefore gets a failure. That strict mode is what the unattended subscription lane now passes, its roster already carrying the URL. **Escalation is `sudo -n`, not a hand-placed polkit rule:** `restartService()` retries a failed restart through `sudo -n` when not root — the idiom `dev.js`/`dev-lib.js` already use and consistent with `provision.js`, `upgrade.js` having been the one place that restarted without escalating. The live `/etc/polkit-1/rules.d/<redacted>.rules` mitigation is superseded: a rebuilt box inherits code, not hand-placed `/etc` files (polkit route kept in [`docs/recipes/instance-service-restart.md`](../recipes/instance-service-restart.md)). `--no-health-check` stays the single escape hatch and now waives the read-back too. Rejected: comparing `startedAt` (cannot distinguish a restart onto the same old core from one onto the new); making an unreadable endpoint fatal by default; requiring `--version-url` everywhere (absent from every current call site — the same "off where it matters" failure in a new costume); keeping polkit as the answer.](<redacted>.md) | core distribution / upgrade verification |
|
|
@@ -1817,5 +1817,7 @@ is load-bearing: the script throws rather than guess if it is missing, and
|
|
|
1817
1817
|
landed since 1.19.677 with no explicit bump. run 34653458123. (task 1002620)
|
|
1818
1818
|
1.19.679 — CI auto-patch (publish-on-merge, ADR 0161): carrier for merges
|
|
1819
1819
|
landed since 1.19.678 with no explicit bump. run 34657605820. (task 1002620)
|
|
1820
|
+
1.19.680 — CI auto-patch (publish-on-merge, ADR 0161): carrier for merges
|
|
1821
|
+
landed since 1.19.679 with no explicit bump. run 34665216511. (task 1002620)
|
|
1820
1822
|
---------------------------------------------------------------------------
|
|
1821
1823
|
```
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
# Recipe — letting `bongos upgrade` restart the instance service unattended
|
|
2
|
+
|
|
3
|
+
> [ADR 0279](../adr/<redacted>.md) · [task 1002884](https://cloudbongos.com/builders#/task/1002884). A core bump only counts once the **running process** is the new core, so `bongos upgrade` has to be able to restart the service without a human at a keyboard. This is what to do when it cannot.
|
|
4
|
+
|
|
5
|
+
## The symptom
|
|
6
|
+
|
|
7
|
+
```
|
|
8
|
+
! restart command failed (exit 1) — restart the service manually.
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
or, in the journal for the unit:
|
|
12
|
+
|
|
13
|
+
```
|
|
14
|
+
Interactive authentication required.
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
Since task 1002884 this **fails the bump and rolls it back** rather than warning and carrying on. That is the fix working: before it, the run went on to health-check the still-running old process, got a 200, and reported `✓ upgrade complete` for a core that was never served.
|
|
18
|
+
|
|
19
|
+
## Why it happens
|
|
20
|
+
|
|
21
|
+
A provisioned instance runs under a `User=` systemd unit. `systemctl restart <unit>` from the owning (non-root) user is a privileged action, so systemd asks polkit, and polkit wants to authenticate a human. In an interactive shell you get a password prompt. In the unattended sweep there is no TTY, so polkit refuses outright and `systemctl` exits 1.
|
|
22
|
+
|
|
23
|
+
## The fix, in order of preference
|
|
24
|
+
|
|
25
|
+
### 1. Passwordless sudo (what the code already tries)
|
|
26
|
+
|
|
27
|
+
`upgrade.js` retries a failed `systemctl restart` through `sudo -n systemctl restart <unit>` when it is not running as root — the same escalation `dev.js`, `dev-lib.js` and `provision.js` use. If the service owner has passwordless sudo, **nothing needs configuring** and the bump succeeds on the retry (the log says `✓ restarted <unit> (via sudo -n)`).
|
|
28
|
+
|
|
29
|
+
Check whether the owner already has it:
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
sudo -n -l
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
A `sudo: a password is required` means no. Note the `-n`: without it this command blocks on a prompt, which is exactly the failure being diagnosed.
|
|
36
|
+
|
|
37
|
+
### 2. Scoped sudoers, if blanket sudo is too much
|
|
38
|
+
|
|
39
|
+
Grant only the restart, not everything. Write it with `visudo -f` so a syntax error cannot lock the host out:
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
sudo visudo -f /etc/sudoers.d/50-<slug>-restart
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
```
|
|
46
|
+
<owner> ALL=(root) NOPASSWD: <redacted> restart <slug>.service
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Then confirm the retry path works, non-interactively:
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
sudo -n systemctl restart <slug>.service && echo OK
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
### 3. A polkit rule, for hosts that deliberately withhold sudo
|
|
56
|
+
|
|
57
|
+
Equivalent outcome, different mechanism — use it when policy forbids a sudoers entry.
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
sudo tee /etc/polkit-1/rules.d/50-<slug>-restart.rules >/dev/null <<'RULE'
|
|
61
|
+
polkit.addRule(function(action, subject) {
|
|
62
|
+
if (action.id == "org.freedesktop.systemd1.manage-units" &&
|
|
63
|
+
action.lookup("unit") == "<slug>.service" &&
|
|
64
|
+
subject.user == "<owner>") {
|
|
65
|
+
return polkit.Result.YES;
|
|
66
|
+
}
|
|
67
|
+
});
|
|
68
|
+
RULE
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
No daemon reload is needed; polkit picks up rules files on the next check.
|
|
72
|
+
|
|
73
|
+
> **This file is not inherited by a rebuilt box.** It lives in `/etc` on one machine. A reprovision restores code, not hand-placed host config — which is precisely why the `sudo -n` retry lives in the core and this page is the fallback rather than the answer. If you reach for this route, record it in the instance's own provisioning notes or the box will re-discover the stall.
|
|
74
|
+
|
|
75
|
+
## Verifying it actually worked
|
|
76
|
+
|
|
77
|
+
Do not trust the exit code alone — that is the failure mode this whole page exists because of. Ask the running process what it is:
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
curl -s https://<instance>/version
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
`coreVersion` in that response is the live process's core. If it still reads the old version after a "successful" restart, the process did not swap, and `bongos upgrade` will now catch that itself and roll back.
|
|
84
|
+
|
|
85
|
+
## Related
|
|
86
|
+
|
|
87
|
+
- [ADR 0279](../adr/<redacted>.md) — why a health check cannot confirm an upgrade
|
|
88
|
+
- [ADR 0136](../adr/<redacted>.md) §3 — the unattended subscription lane's health-gate + auto-rollback policy
|
|
89
|
+
- [`docs/recipes/ops-gotchas.md`](ops-gotchas.md) — the wider set of deploy traps
|
package/package-lock.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@bongos/core",
|
|
3
|
-
"version": "1.19.
|
|
3
|
+
"version": "1.19.680",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "@bongos/core",
|
|
9
|
-
"version": "1.19.
|
|
9
|
+
"version": "1.19.680",
|
|
10
10
|
"license": "AGPL-3.0-or-later",
|
|
11
11
|
"dependencies": {
|
|
12
12
|
"express": "^4.21.2",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@bongos/core",
|
|
3
|
-
"version": "1.19.
|
|
3
|
+
"version": "1.19.680",
|
|
4
4
|
"description": "Cloud Bongos — the AI-first build platform core (GDS + platform surfaces + module system), installed as a versioned dependency (ADR 0108).",
|
|
5
5
|
"license": "AGPL-3.0-or-later",
|
|
6
6
|
"main": "src/platform-server.js",
|
package/scripts/gds/upgrade.js
CHANGED
|
@@ -521,7 +521,20 @@ function healDocAssets({ coreRoot }, run = spawnSync, fsImpl = fs) {
|
|
|
521
521
|
return { ok: failed.length === 0, ran, failed, missing };
|
|
522
522
|
}
|
|
523
523
|
|
|
524
|
-
|
|
524
|
+
// Restart the instance service so the newly-installed core becomes the one being SERVED.
|
|
525
|
+
// Escalation is the whole difficulty here (task 1002884). A provisioned instance runs under a
|
|
526
|
+
// `User=` systemd unit, and the unattended sweep that restarts it has no TTY — so a bare
|
|
527
|
+
// `systemctl restart` from the owning (non-root) user returns "Interactive authentication
|
|
528
|
+
// required" and exits 1. We retry once through `sudo -n`, the same idiom dev.js and dev-lib.js
|
|
529
|
+
// already use for privileged systemctl; `-n` fails fast rather than blocking on a password
|
|
530
|
+
// prompt, so the retry costs nothing when it isn't permitted. An operator who prefers polkit
|
|
531
|
+
// over sudo can install the rule in docs/recipes/instance-service-restart.md instead — either
|
|
532
|
+
// path makes this return ok.
|
|
533
|
+
// `--restart-cmd` is NOT escalated: an explicit command is the operator's own contract.
|
|
534
|
+
// Trust boundary: `service` comes from `--service` or the subscription roster — operator-supplied
|
|
535
|
+
// config, never request input — and the escalation reaches only as far as the host's own sudoers
|
|
536
|
+
// or polkit scoping allows. Do not wire a caller that derives `service` from anything untrusted.
|
|
537
|
+
function restartService({ service, restartCmd }, run = spawnSync, deps = {}) {
|
|
525
538
|
if (restartCmd) {
|
|
526
539
|
const parts = restartCmd.split(/\s+/);
|
|
527
540
|
const r = run(parts[0], parts.slice(1), { stdio: 'inherit' });
|
|
@@ -529,7 +542,14 @@ function restartService({ service, restartCmd }, run = spawnSync) {
|
|
|
529
542
|
}
|
|
530
543
|
if (service) {
|
|
531
544
|
const r = run('systemctl', ['restart', service], { stdio: 'inherit' });
|
|
532
|
-
|
|
545
|
+
if (r && !r.error && r.status === 0) return { ok: true, code: 0 };
|
|
546
|
+
const isRoot = deps.isRoot !== undefined
|
|
547
|
+
? !!deps.isRoot
|
|
548
|
+
: (typeof process.getuid === 'function' && process.getuid() === 0);
|
|
549
|
+
if (isRoot) return { ok: false, code: r && r.status };
|
|
550
|
+
const s = run('sudo', ['-n', 'systemctl', 'restart', service], { stdio: 'inherit' });
|
|
551
|
+
const ok = !!s && !s.error && s.status === 0;
|
|
552
|
+
return { ok, code: ok ? 0 : (s && s.status), triedSudo: true, directCode: r && r.status };
|
|
533
553
|
}
|
|
534
554
|
return { ok: null, skipped: true };
|
|
535
555
|
}
|
|
@@ -622,6 +642,68 @@ async function pollHealth(healthUrl, deps = {}) {
|
|
|
622
642
|
}
|
|
623
643
|
}
|
|
624
644
|
|
|
645
|
+
// ---- served-version verification (task 1002884) -----------------------------
|
|
646
|
+
// The gap this closes: after a restart, `readInstalledCoreVersion` reads node_modules on DISK
|
|
647
|
+
// and `pollHealth` proves only that *something* answers the port. Neither can tell "the new core
|
|
648
|
+
// is up" from "the old core never went down" — so when a restart silently failed on
|
|
649
|
+
// cloudbongos.com (2026-08-11), the health poll hit the STILL-RUNNING old process, got 200, and
|
|
650
|
+
// the bump reported "✓ upgrade complete" while live kept serving the previous core.
|
|
651
|
+
// `/version` reports the LIVE process's coreVersion (serve-internal.js, since 1.17.2), which is
|
|
652
|
+
// the one signal that distinguishes the two.
|
|
653
|
+
|
|
654
|
+
// Default the version endpoint to /version on the same origin as the health URL. Deriving rather
|
|
655
|
+
// than demanding a second flag is deliberate: every existing call site passes only --health-url,
|
|
656
|
+
// and a check that has to be opted into is off exactly where it is needed. Returns null when the
|
|
657
|
+
// health URL isn't parseable as one.
|
|
658
|
+
function deriveVersionUrl(healthUrl) {
|
|
659
|
+
try {
|
|
660
|
+
const u = new URL(healthUrl);
|
|
661
|
+
u.pathname = '/version';
|
|
662
|
+
u.search = '';
|
|
663
|
+
u.hash = '';
|
|
664
|
+
return u.toString();
|
|
665
|
+
} catch { return null; }
|
|
666
|
+
}
|
|
667
|
+
|
|
668
|
+
// Poll the version endpoint until it yields a parseable coreVersion or the budget elapses. Runs
|
|
669
|
+
// AFTER the health poll, so the service is already answering and the budget is short — this is
|
|
670
|
+
// re-reading a live endpoint, not waiting for a boot. A body without a usable coreVersion is
|
|
671
|
+
// "unreadable", NOT a mismatch: the caller must be able to tell "it says the wrong version"
|
|
672
|
+
// (a real failure) from "it wouldn't tell me" (unconfirmed), because only the first is proof.
|
|
673
|
+
async function pollServedVersion(versionUrl, deps = {}) {
|
|
674
|
+
const fetchImpl = deps.fetch || fetch;
|
|
675
|
+
const sleep = deps.sleep || ((ms) => new Promise((resolve) => setTimeout(resolve, ms)));
|
|
676
|
+
const now = deps.now || Date.now;
|
|
677
|
+
// Default to the health budget when a caller (or a test) has already narrowed it, else 15s.
|
|
678
|
+
const budgetMs = deps.versionBudgetMs != null
|
|
679
|
+
? deps.versionBudgetMs
|
|
680
|
+
: (deps.healthBudgetMs != null ? deps.healthBudgetMs : 15_000);
|
|
681
|
+
// A hard attempt cap alongside the clock: `now` is injectable, and a frozen or non-advancing
|
|
682
|
+
// clock would otherwise spin this loop forever instead of giving up.
|
|
683
|
+
const maxAttempts = deps.versionMaxAttempts != null ? deps.versionMaxAttempts : 12;
|
|
684
|
+
const deadline = now() + budgetMs;
|
|
685
|
+
let attempt = 0;
|
|
686
|
+
let last = { ok: false, error: 'not attempted' };
|
|
687
|
+
for (;;) {
|
|
688
|
+
attempt++;
|
|
689
|
+
try {
|
|
690
|
+
const res = await fetchImpl(versionUrl, { signal: AbortSignal.timeout(5_000) });
|
|
691
|
+
if (!res || !res.ok) last = { ok: false, error: `status ${res && res.status}` };
|
|
692
|
+
else if (typeof res.json !== 'function') last = { ok: false, error: 'response carried no JSON body' };
|
|
693
|
+
else {
|
|
694
|
+
const body = await res.json();
|
|
695
|
+
const version = body && typeof body.coreVersion === 'string' ? body.coreVersion : null;
|
|
696
|
+
if (version) return { ok: true, version, attempts: attempt };
|
|
697
|
+
last = { ok: false, error: 'no coreVersion in the response body' };
|
|
698
|
+
}
|
|
699
|
+
} catch (e) {
|
|
700
|
+
last = { ok: false, error: e.message };
|
|
701
|
+
}
|
|
702
|
+
if (now() >= deadline || attempt >= maxAttempts) return { ...last, attempts: attempt, timedOut: true };
|
|
703
|
+
await sleep(Math.min(2_000, 500 * attempt));
|
|
704
|
+
}
|
|
705
|
+
}
|
|
706
|
+
|
|
625
707
|
// ---- integrity-pin verification (task 1002216 / audit H8, finding F1) --------
|
|
626
708
|
// The packager (package-core.js) computes tree_sha256 = canonicalTreeHash over the
|
|
627
709
|
// redacted core files (+ the two synthesized files) and records it, the per-file
|
|
@@ -738,7 +820,9 @@ async function rollback({ instanceDir, snapshot, fromVersion, toVersion, opts },
|
|
|
738
820
|
else err(` ! reinstall of previous core failed (${inst.reason || 'exit ' + inst.code}) — the instance may need manual repair`);
|
|
739
821
|
|
|
740
822
|
// 3. restart back onto the previous core.
|
|
741
|
-
|
|
823
|
+
// `deps` is passed here for the same reason the forward path passes it: this restart runs
|
|
824
|
+
// AFTER a failure, so its sudo-escalation is the one that matters most and must be testable.
|
|
825
|
+
const rs = restartService({ service: opts.service, restartCmd: opts.restartCmd }, run, deps);
|
|
742
826
|
if (rs.skipped) log(' • restart: no --service/--restart-cmd — restart the instance service manually to load the reverted core.');
|
|
743
827
|
else if (!rs.ok) err(` ! restart after rollback failed (exit ${rs.code}) — restart the service manually.`);
|
|
744
828
|
else log(` ✓ restarted ${opts.service || 'service'} on the previous core`);
|
|
@@ -961,10 +1045,24 @@ async function runUpgrade(opts, deps = {}) {
|
|
|
961
1045
|
} else log(' • doc regen skipped (--skip-regen-docs)');
|
|
962
1046
|
|
|
963
1047
|
// 6. restart.
|
|
964
|
-
const rs = restartService({ service: opts.service, restartCmd: opts.restartCmd }, run);
|
|
1048
|
+
const rs = restartService({ service: opts.service, restartCmd: opts.restartCmd }, run, deps);
|
|
965
1049
|
if (rs.skipped) log(` • restart: no --service/--restart-cmd — restart the instance service to load core ${targetVersion}.`);
|
|
966
|
-
else if (!rs.ok)
|
|
967
|
-
|
|
1050
|
+
else if (!rs.ok) {
|
|
1051
|
+
// task 1002884: a failed restart is an upgrade FAILURE, not a warning. It used to fall
|
|
1052
|
+
// through to the verification below — but the old process is still up, so the health poll
|
|
1053
|
+
// passed against IT and the bump wrote a success row for a core that was never served.
|
|
1054
|
+
// Failing here also keeps disk and process consistent: rollback puts the previous core back
|
|
1055
|
+
// on disk, matching the one that is (still) running.
|
|
1056
|
+
err(` ✖ restart failed (exit ${rs.code})${rs.triedSudo ? ` — \`systemctl restart\` exited ${rs.directCode} and \`sudo -n systemctl restart\` could not escalate` : ''}`);
|
|
1057
|
+
err(' The previous core is still the one being served. Grant the restart a non-interactive path');
|
|
1058
|
+
err(' (sudoers NOPASSWD or a polkit rule — docs/recipes/instance-service-restart.md), then re-run.');
|
|
1059
|
+
const error = `restart failed (exit ${rs.code}) — the instance is still running the previous core`;
|
|
1060
|
+
if (rollbackOnFailure && snapshot) {
|
|
1061
|
+
const rb = await doRollback('restart failed');
|
|
1062
|
+
return { ok: false, error, fromVersion, toVersion: targetVersion, restartOk: false, rolledBack: true, rollback: rb };
|
|
1063
|
+
}
|
|
1064
|
+
return { ok: false, error, fromVersion, toVersion: targetVersion, restartOk: false };
|
|
1065
|
+
} else log(` ✓ restarted ${opts.service || 'service'}${rs.triedSudo ? ' (via sudo -n)' : ''}`);
|
|
968
1066
|
|
|
969
1067
|
// 7. verify: installed version == target, optional /healthz, ledger row.
|
|
970
1068
|
const installed = readInstalledCoreVersion(instanceDir, fsImpl);
|
|
@@ -985,12 +1083,48 @@ async function runUpgrade(opts, deps = {}) {
|
|
|
985
1083
|
log(' • no managed restart + no --health-url — new core installed but not confirmed serving; restart the instance + health-check it to complete the bump.');
|
|
986
1084
|
}
|
|
987
1085
|
|
|
1086
|
+
// 7b. Confirm the SERVED process actually swapped (task 1002884). Everything above this point
|
|
1087
|
+
// can be true of an instance still running the OLD core: the pin moved, node_modules holds
|
|
1088
|
+
// the new version, and /healthz answers 200 — from the process that never went down. Only
|
|
1089
|
+
// the live coreVersion settles it.
|
|
1090
|
+
// A definite MISMATCH is a failure (same treatment as a failed health check). Being unable
|
|
1091
|
+
// to READ the endpoint is not: an instance may not expose it, and refusing every such bump
|
|
1092
|
+
// would be a worse regression than the false-pass. Passing --version-url explicitly asks
|
|
1093
|
+
// for proof, so there an unreadable endpoint IS a failure.
|
|
1094
|
+
let servedOk = null;
|
|
1095
|
+
let servedVersion = null;
|
|
1096
|
+
const versionUrl = opts.versionUrl || (opts.healthUrl ? deriveVersionUrl(opts.healthUrl) : null);
|
|
1097
|
+
if (opts.healthCheck === false) {
|
|
1098
|
+
// --no-health-check is the documented opt-out from confirming the bump serves; it covers this.
|
|
1099
|
+
} else if (versionUrl && healthOk !== false) {
|
|
1100
|
+
const sv = await pollServedVersion(versionUrl, deps);
|
|
1101
|
+
servedVersion = sv.version || null;
|
|
1102
|
+
if (sv.ok && servedVersion === targetVersion) {
|
|
1103
|
+
servedOk = true;
|
|
1104
|
+
log(` ✓ served core version: ${servedVersion} (${versionUrl})`);
|
|
1105
|
+
} else if (sv.ok) {
|
|
1106
|
+
servedOk = false;
|
|
1107
|
+
err(` ! served core version is ${servedVersion} but the target is ${targetVersion} (${versionUrl})`);
|
|
1108
|
+
err(' The restart did not swap the running process — the old core is still being served.');
|
|
1109
|
+
} else if (opts.versionUrl) {
|
|
1110
|
+
servedOk = false;
|
|
1111
|
+
err(` ! could not read ${versionUrl} (${sv.error}) — --version-url was given, so this bump cannot be confirmed`);
|
|
1112
|
+
} else {
|
|
1113
|
+
err(` ! could not read ${versionUrl} (${sv.error}) — this bump is NOT confirmed against the running process.`);
|
|
1114
|
+
err(' Pass --version-url <url> if this instance serves its version elsewhere.');
|
|
1115
|
+
}
|
|
1116
|
+
}
|
|
1117
|
+
|
|
988
1118
|
// A failed post-restart health check means the new core is serving broken — auto-revert (task
|
|
989
|
-
// 2149) instead of recording an "upgraded" row + leaving the instance down.
|
|
990
|
-
//
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
|
|
1119
|
+
// 2149) instead of recording an "upgraded" row + leaving the instance down. A served-version
|
|
1120
|
+
// mismatch means the new core is not serving AT ALL (task 1002884) and reverts the same way.
|
|
1121
|
+
// The rollback writes its OWN rolled_back ledger row, so skip the normal insert below.
|
|
1122
|
+
if ((healthOk === false || servedOk === false) && rollbackOnFailure && snapshot) {
|
|
1123
|
+
const why = healthOk === false
|
|
1124
|
+
? 'health check failed after restart'
|
|
1125
|
+
: `served core version did not change (still ${servedVersion || 'unconfirmed'}, expected ${targetVersion})`;
|
|
1126
|
+
const rb = await doRollback(why);
|
|
1127
|
+
return { ok: false, error: `${why} — rolled back to ${snapshot.prevVersion || 'the previous core'}`, fromVersion, toVersion: targetVersion, versionOk, healthOk, servedOk, servedVersion, rolledBack: true, rollback: rb };
|
|
994
1128
|
}
|
|
995
1129
|
|
|
996
1130
|
const ledger = await recordLedger({ fromVersion, toVersion: targetVersion, sourceCommit: opts.sourceCommit, note: opts.note }, deps);
|
|
@@ -1010,9 +1144,9 @@ async function runUpgrade(opts, deps = {}) {
|
|
|
1010
1144
|
dryRun: opts.dryRun,
|
|
1011
1145
|
}, run, log, err);
|
|
1012
1146
|
|
|
1013
|
-
const ok = versionOk && healthOk !== false;
|
|
1147
|
+
const ok = versionOk && healthOk !== false && servedOk !== false;
|
|
1014
1148
|
log(ok ? `\n✓ upgrade complete — core ${fromVersion || '(none)'} → ${targetVersion}` : `\n! upgrade finished with warnings — review above.`);
|
|
1015
|
-
return { ok, fromVersion, toVersion: targetVersion, versionOk, healthOk, ledger: ledger.ok, pin };
|
|
1149
|
+
return { ok, fromVersion, toVersion: targetVersion, versionOk, healthOk, servedOk, servedVersion, ledger: ledger.ok, pin };
|
|
1016
1150
|
}
|
|
1017
1151
|
|
|
1018
1152
|
function parseArgs(argv) {
|
|
@@ -1023,6 +1157,7 @@ function parseArgs(argv) {
|
|
|
1023
1157
|
service: arg('--service', argv),
|
|
1024
1158
|
restartCmd: arg('--restart-cmd', argv),
|
|
1025
1159
|
healthUrl: arg('--health-url', argv),
|
|
1160
|
+
versionUrl: arg('--version-url', argv), // task 1002884: read the LIVE process's coreVersion back; defaults to /version on the health URL's origin
|
|
1026
1161
|
sourceCommit: arg('--source-commit', argv),
|
|
1027
1162
|
note: arg('--note', argv),
|
|
1028
1163
|
pinPath: arg('--pin-path', argv),
|
|
@@ -1058,7 +1193,8 @@ async function main(argv = process.argv.slice(2)) {
|
|
|
1058
1193
|
' --service <unit> systemd unit to restart after applying',
|
|
1059
1194
|
' --restart-cmd <cmd> explicit restart command (overrides --service)',
|
|
1060
1195
|
' --health-url <url> poll after restart (~30s) until 2xx; a still-failing check fails the bump (auto-rollback). REQUIRED for a bump that restarts the service, unless --no-health-check.',
|
|
1061
|
-
' --
|
|
1196
|
+
' --version-url <url> read the RUNNING process\'s coreVersion back after restart and require it to equal --to (default: /version on the --health-url origin). A health check alone cannot tell a new core from an old one that never went down.',
|
|
1197
|
+
' --no-health-check explicitly skip the post-restart health confirmation (default: the check is ON — a bump that restarts the service must confirm it serves); also skips the served-version read-back',
|
|
1062
1198
|
' --skip-migrate do not run `npm run migrate`',
|
|
1063
1199
|
' --skip-materialize do not refresh .claude/',
|
|
1064
1200
|
' --skip-regen-docs do not regenerate the OpenAPI spec + typed client from the new core',
|
|
@@ -1069,8 +1205,8 @@ async function main(argv = process.argv.slice(2)) {
|
|
|
1069
1205
|
' --dry-run show the plan; change nothing',
|
|
1070
1206
|
' --force skip the clean-tree + same-version guards',
|
|
1071
1207
|
'',
|
|
1072
|
-
'Full bump: pre-flight → bump pin → npm install → verify integrity pin → migrate → materialize .claude → regen API docs+client → restart → verify.',
|
|
1073
|
-
'On a failure after the pin moves (install/pin-verify/migrate/health), auto-rollback reverts the pin, reinstalls the previous core, restarts, and records a rolled_back ledger row (the DB is not rolled back — core migrations are additive).',
|
|
1208
|
+
'Full bump: pre-flight → bump pin → npm install → verify integrity pin → migrate → materialize .claude → regen API docs+client → restart → verify (health + served coreVersion).',
|
|
1209
|
+
'On a failure after the pin moves (install/pin-verify/migrate/restart/health/served-version), auto-rollback reverts the pin, reinstalls the previous core, restarts, and records a rolled_back ledger row (the DB is not rolled back — core migrations are additive).',
|
|
1074
1210
|
].join('\n') + '\n');
|
|
1075
1211
|
return 0;
|
|
1076
1212
|
}
|
|
@@ -1087,7 +1223,7 @@ async function main(argv = process.argv.slice(2)) {
|
|
|
1087
1223
|
module.exports = {
|
|
1088
1224
|
runUpgrade, parseArgs, main, rollback,
|
|
1089
1225
|
versionFromTgzPath, versionFromDep, readInstalledCoreVersion, readPinnedCoreVersion, writePin, writeRegistryPin, restorePin,
|
|
1090
|
-
vendorTarball, gitTreeClean, persistPin, dirtyPinFiles, gitCurrentBranch, PIN_FILES, preflightDbIdentity, resolveInstanceDb, preflightModules, reportModulePreflight, npmInstall, runMigrate, regenerateApiArtifacts, regenerateNavDocs, NAV_WHOLE_FILE_GENERATORS, missingDocAssets, healDocAssets, DOC_ASSET_GENERATORS, restartService, recordLedger, pollHealth,
|
|
1226
|
+
vendorTarball, gitTreeClean, persistPin, dirtyPinFiles, gitCurrentBranch, PIN_FILES, preflightDbIdentity, resolveInstanceDb, preflightModules, reportModulePreflight, npmInstall, runMigrate, regenerateApiArtifacts, regenerateNavDocs, NAV_WHOLE_FILE_GENERATORS, missingDocAssets, healDocAssets, DOC_ASSET_GENERATORS, restartService, recordLedger, pollHealth, deriveVersionUrl, pollServedVersion,
|
|
1091
1227
|
readManifest, resolveReferenceManifest, verifyInstalledPin, // task 1002216 (audit H8) — integrity-pin verification
|
|
1092
1228
|
CORE_PKG, ARTIFACT,
|
|
1093
1229
|
};
|
package/src/module-api.js
CHANGED
|
@@ -71,7 +71,7 @@ const { responsibilityFor, ROLE_RESPONSIBILITIES } = require('./role-responsibil
|
|
|
71
71
|
// there. scripts/gds/bump-version.js still rewrites the literal below; it appends
|
|
72
72
|
// the entry to that file. Look for a version's history there, not here.
|
|
73
73
|
// ---------------------------------------------------------------------------
|
|
74
|
-
const CORE_VERSION = '1.19.
|
|
74
|
+
const CORE_VERSION = '1.19.680'; // CI auto-patch carrier (ADR 0161); changelog: docs/module-api-changelog.md
|
|
75
75
|
|
|
76
76
|
// A namespaced logger so a module's log lines are attributable + consistent.
|
|
77
77
|
// Usage: const log = api.logger('dev-box'); log.info('mounted');
|
package/tests/upgrade.mjs
CHANGED
|
@@ -1091,6 +1091,265 @@ t('runUpgrade: refuses BEFORE the pin moves when the instance names no database'
|
|
|
1091
1091
|
assert.match(readFileSync(join(dir, 'package.json'), 'utf8'), /"@bongos\/core": "1\.0\.0"/);
|
|
1092
1092
|
});
|
|
1093
1093
|
|
|
1094
|
+
// ---- task 1002884: an upgrade may not declare success it has not proven -----
|
|
1095
|
+
// The 2026-08-11 cloudbongos.com incident, as executable tests. The sweep's `systemctl restart`
|
|
1096
|
+
// failed with "Interactive authentication required", upgrade.js logged it as a warning and
|
|
1097
|
+
// carried on, the health poll hit the STILL-RUNNING old process and got 200, and the run
|
|
1098
|
+
// reported "✓ upgrade complete — core 1.19.13 → 1.19.56" while live served 1.19.13.
|
|
1099
|
+
// Both halves are covered: a failed restart must be fatal, and a served version that did not
|
|
1100
|
+
// change must be caught even when the restart reported success.
|
|
1101
|
+
|
|
1102
|
+
// A run shim whose `systemctl restart` fails, optionally letting `sudo -n systemctl restart` win.
|
|
1103
|
+
function restartFailingRun(calls, { dir, sudoWorks = false } = {}) {
|
|
1104
|
+
const base = pinAwareRun(calls, { dir });
|
|
1105
|
+
return (cmd, args = []) => {
|
|
1106
|
+
if (cmd === 'systemctl' && args[0] === 'restart') { calls.push(`systemctl ${args.join(' ')}`); return { status: 1 }; }
|
|
1107
|
+
if (cmd === 'sudo') { calls.push(`sudo ${args.join(' ')}`); return { status: sudoWorks ? 0 : 1 }; }
|
|
1108
|
+
return base(cmd, args);
|
|
1109
|
+
};
|
|
1110
|
+
}
|
|
1111
|
+
|
|
1112
|
+
// A /version endpoint that reports whatever the live process is "serving".
|
|
1113
|
+
function versionFetch(serving, { healthOk = true } = {}) {
|
|
1114
|
+
return async (url) => {
|
|
1115
|
+
if (String(url).endsWith('/version')) return { ok: true, status: 200, json: async () => ({ coreVersion: serving }) };
|
|
1116
|
+
return { ok: healthOk, status: healthOk ? 200 : 503 };
|
|
1117
|
+
};
|
|
1118
|
+
}
|
|
1119
|
+
|
|
1120
|
+
t('restartService: a non-root systemctl failure retries through `sudo -n` and can succeed there', () => {
|
|
1121
|
+
const calls = [];
|
|
1122
|
+
const run = (cmd, args = []) => { calls.push(`${cmd} ${args.join(' ')}`); return { status: cmd === 'sudo' ? 0 : 1 }; };
|
|
1123
|
+
const r = u.restartService({ service: 'demo.service' }, run, { isRoot: false });
|
|
1124
|
+
assert.equal(r.ok, true, 'the sudo -n retry restarted the service');
|
|
1125
|
+
assert.equal(r.triedSudo, true);
|
|
1126
|
+
assert.equal(r.directCode, 1, 'the direct attempt is reported alongside');
|
|
1127
|
+
assert.deepEqual(calls, ['systemctl restart demo.service', 'sudo -n systemctl restart demo.service']);
|
|
1128
|
+
});
|
|
1129
|
+
|
|
1130
|
+
t('restartService: as root there is no sudo retry (nothing to escalate to)', () => {
|
|
1131
|
+
const calls = [];
|
|
1132
|
+
const run = (cmd, args = []) => { calls.push(`${cmd} ${args.join(' ')}`); return { status: 1 }; };
|
|
1133
|
+
const r = u.restartService({ service: 'demo.service' }, run, { isRoot: true });
|
|
1134
|
+
assert.equal(r.ok, false);
|
|
1135
|
+
assert.ok(!r.triedSudo, 'root does not shell out to sudo');
|
|
1136
|
+
assert.deepEqual(calls, ['systemctl restart demo.service']);
|
|
1137
|
+
});
|
|
1138
|
+
|
|
1139
|
+
t('restartService: an explicit --restart-cmd is never escalated (the operator owns that contract)', () => {
|
|
1140
|
+
const calls = [];
|
|
1141
|
+
const run = (cmd, args = []) => { calls.push(`${cmd} ${args.join(' ')}`); return { status: 1 }; };
|
|
1142
|
+
const r = u.restartService({ restartCmd: 'deploy restart' }, run, { isRoot: false });
|
|
1143
|
+
assert.equal(r.ok, false);
|
|
1144
|
+
assert.deepEqual(calls, ['deploy restart'], 'no sudo retry for an explicit command');
|
|
1145
|
+
});
|
|
1146
|
+
|
|
1147
|
+
t('the 2026-08-11 incident: a failed restart FAILS the bump even though /healthz answers 200', async () => {
|
|
1148
|
+
const src = mkdtempSync(join(tmpdir(), 'src-'));
|
|
1149
|
+
writeFileSync(join(src, 'bongos-core-1.15.0.tgz'), 'TGZ');
|
|
1150
|
+
const dir = scratchConsumer({ pinned: '1.14.1', installed: '1.14.1' });
|
|
1151
|
+
const calls = [], queries = [];
|
|
1152
|
+
const res = await u.runUpgrade(
|
|
1153
|
+
{ to: '1.15.0', from: join(src, 'bongos-core-1.15.0.tgz'), instance: dir, service: 'demo.service', healthUrl: 'http://x/healthz' },
|
|
1154
|
+
{
|
|
1155
|
+
log: () => {}, err: () => {},
|
|
1156
|
+
run: restartFailingRun(calls, { dir, sudoWorks: false }),
|
|
1157
|
+
isRoot: false,
|
|
1158
|
+
materialize: () => ({ skills: 1 }), regenerateApiArtifacts: () => ({ ok: true, ran: [] }),
|
|
1159
|
+
databaseUrl: 'postgres://x/y', pg: recordingPg(queries),
|
|
1160
|
+
// the old process is still up and healthy — the exact false-pass this task exists to kill
|
|
1161
|
+
fetch: async () => ({ ok: true, status: 200 }),
|
|
1162
|
+
sleep: async () => {}, now: () => 0, healthBudgetMs: 0,
|
|
1163
|
+
}
|
|
1164
|
+
);
|
|
1165
|
+
assert.equal(res.ok, false, 'a failed restart is not a successful upgrade');
|
|
1166
|
+
assert.equal(res.restartOk, false);
|
|
1167
|
+
assert.match(res.error, /restart failed/i);
|
|
1168
|
+
assert.match(res.error, /still running the previous core/i);
|
|
1169
|
+
assert.equal(res.rolledBack, true, 'the bump reverts rather than leaving disk ahead of the process');
|
|
1170
|
+
assert.equal(u.readPinnedCoreVersion(dir), '1.14.1', 'pin reverted');
|
|
1171
|
+
assert.equal(queries.length, 1, 'no phantom "upgraded" ledger row');
|
|
1172
|
+
assert.match(queries[0].params[3], /^rolled_back: restart failed/);
|
|
1173
|
+
assert.ok(calls.includes('sudo -n systemctl restart demo.service'), 'it tried to escalate before giving up');
|
|
1174
|
+
});
|
|
1175
|
+
|
|
1176
|
+
t('a restart that fails directly but succeeds under sudo -n carries the bump through', async () => {
|
|
1177
|
+
const src = mkdtempSync(join(tmpdir(), 'src-'));
|
|
1178
|
+
writeFileSync(join(src, 'bongos-core-1.15.0.tgz'), 'TGZ');
|
|
1179
|
+
const dir = scratchConsumer({ pinned: '1.14.1', installed: '1.14.1' });
|
|
1180
|
+
const calls = [], queries = [];
|
|
1181
|
+
const res = await u.runUpgrade(
|
|
1182
|
+
{ to: '1.15.0', from: join(src, 'bongos-core-1.15.0.tgz'), instance: dir, service: 'demo.service', healthUrl: 'http://x/healthz' },
|
|
1183
|
+
{
|
|
1184
|
+
log: () => {}, err: () => {},
|
|
1185
|
+
run: restartFailingRun(calls, { dir, sudoWorks: true }),
|
|
1186
|
+
isRoot: false,
|
|
1187
|
+
materialize: () => ({ skills: 1 }), regenerateApiArtifacts: () => ({ ok: true, ran: [] }),
|
|
1188
|
+
databaseUrl: 'postgres://x/y', pg: recordingPg(queries),
|
|
1189
|
+
fetch: versionFetch('1.15.0'),
|
|
1190
|
+
sleep: async () => {}, now: () => 0, healthBudgetMs: 0,
|
|
1191
|
+
}
|
|
1192
|
+
);
|
|
1193
|
+
assert.equal(res.ok, true, 'escalating is a success, not a warning');
|
|
1194
|
+
assert.equal(res.servedOk, true);
|
|
1195
|
+
assert.equal(res.servedVersion, '1.15.0');
|
|
1196
|
+
assert.equal(queries.length, 1);
|
|
1197
|
+
assert.ok(!/rolled_back/.test(queries[0].params[3] || ''), 'a real upgraded row, not a reversal');
|
|
1198
|
+
});
|
|
1199
|
+
|
|
1200
|
+
t('served-version mismatch: restart reports success but the OLD core is still serving → fail + rollback', async () => {
|
|
1201
|
+
const src = mkdtempSync(join(tmpdir(), 'src-'));
|
|
1202
|
+
writeFileSync(join(src, 'bongos-core-1.15.0.tgz'), 'TGZ');
|
|
1203
|
+
const dir = scratchConsumer({ pinned: '1.14.1', installed: '1.14.1' });
|
|
1204
|
+
const calls = [], queries = [];
|
|
1205
|
+
const res = await u.runUpgrade(
|
|
1206
|
+
{ to: '1.15.0', from: join(src, 'bongos-core-1.15.0.tgz'), instance: dir, service: 'demo.service', healthUrl: 'http://x/healthz' },
|
|
1207
|
+
{
|
|
1208
|
+
log: () => {}, err: () => {},
|
|
1209
|
+
run: pinAwareRun(calls, { dir }), // systemctl exits 0 — nothing else can catch this
|
|
1210
|
+
materialize: () => ({ skills: 1 }), regenerateApiArtifacts: () => ({ ok: true, ran: [] }),
|
|
1211
|
+
databaseUrl: 'postgres://x/y', pg: recordingPg(queries),
|
|
1212
|
+
fetch: versionFetch('1.14.1'), // the process never swapped
|
|
1213
|
+
sleep: async () => {}, now: () => 0, healthBudgetMs: 0,
|
|
1214
|
+
}
|
|
1215
|
+
);
|
|
1216
|
+
assert.equal(res.ok, false, 'disk-and-health agreement is not proof the new core is serving');
|
|
1217
|
+
assert.equal(res.servedOk, false);
|
|
1218
|
+
assert.equal(res.servedVersion, '1.14.1');
|
|
1219
|
+
assert.equal(res.rolledBack, true);
|
|
1220
|
+
assert.equal(queries.length, 1, 'no phantom "upgraded" row');
|
|
1221
|
+
assert.match(queries[0].params[3], /^rolled_back: served core version did not change/);
|
|
1222
|
+
});
|
|
1223
|
+
|
|
1224
|
+
t('served-version match: the bump is confirmed against the running process', async () => {
|
|
1225
|
+
const src = mkdtempSync(join(tmpdir(), 'src-'));
|
|
1226
|
+
writeFileSync(join(src, 'bongos-core-1.15.0.tgz'), 'TGZ');
|
|
1227
|
+
const dir = scratchConsumer({ pinned: '1.14.1', installed: '1.14.1' });
|
|
1228
|
+
const urls = [];
|
|
1229
|
+
const res = await u.runUpgrade(
|
|
1230
|
+
{ to: '1.15.0', from: join(src, 'bongos-core-1.15.0.tgz'), instance: dir, service: 'demo.service', healthUrl: 'http://x/healthz' },
|
|
1231
|
+
{
|
|
1232
|
+
log: () => {}, err: () => {},
|
|
1233
|
+
run: pinAwareRun([], { dir }),
|
|
1234
|
+
materialize: () => ({ skills: 1 }), regenerateApiArtifacts: () => ({ ok: true, ran: [] }),
|
|
1235
|
+
fetch: async (url) => { urls.push(String(url)); return versionFetch('1.15.0')(url); },
|
|
1236
|
+
sleep: async () => {}, now: () => 0, healthBudgetMs: 0,
|
|
1237
|
+
}
|
|
1238
|
+
);
|
|
1239
|
+
assert.equal(res.ok, true);
|
|
1240
|
+
assert.equal(res.servedOk, true);
|
|
1241
|
+
assert.ok(urls.includes('http://x/version'), 'the version URL is derived from the health URL origin');
|
|
1242
|
+
});
|
|
1243
|
+
|
|
1244
|
+
t('an unreadable DERIVED version endpoint warns but does not fail the bump (no regression)', async () => {
|
|
1245
|
+
const src = mkdtempSync(join(tmpdir(), 'src-'));
|
|
1246
|
+
writeFileSync(join(src, 'bongos-core-1.15.0.tgz'), 'TGZ');
|
|
1247
|
+
const dir = scratchConsumer({ pinned: '1.14.1', installed: '1.14.1' });
|
|
1248
|
+
const warnings = [];
|
|
1249
|
+
const res = await u.runUpgrade(
|
|
1250
|
+
{ to: '1.15.0', from: join(src, 'bongos-core-1.15.0.tgz'), instance: dir, service: 'demo.service', healthUrl: 'http://x/healthz' },
|
|
1251
|
+
{
|
|
1252
|
+
log: () => {}, err: (m) => warnings.push(String(m)),
|
|
1253
|
+
run: pinAwareRun([], { dir }),
|
|
1254
|
+
materialize: () => ({ skills: 1 }), regenerateApiArtifacts: () => ({ ok: true, ran: [] }),
|
|
1255
|
+
fetch: async (url) => (String(url).endsWith('/version') ? { ok: false, status: 404 } : { ok: true, status: 200 }),
|
|
1256
|
+
sleep: async () => {}, now: () => 0, healthBudgetMs: 0,
|
|
1257
|
+
}
|
|
1258
|
+
);
|
|
1259
|
+
assert.equal(res.ok, true, 'an instance that does not serve /version can still be upgraded');
|
|
1260
|
+
assert.equal(res.servedOk, null, 'unconfirmed is neither pass nor fail');
|
|
1261
|
+
assert.ok(warnings.some((w) => /NOT confirmed against the running process/.test(w)), 'but it says so loudly');
|
|
1262
|
+
});
|
|
1263
|
+
|
|
1264
|
+
t('an unreadable EXPLICIT --version-url fails the bump (proof was asked for)', async () => {
|
|
1265
|
+
const src = mkdtempSync(join(tmpdir(), 'src-'));
|
|
1266
|
+
writeFileSync(join(src, 'bongos-core-1.15.0.tgz'), 'TGZ');
|
|
1267
|
+
const dir = scratchConsumer({ pinned: '1.14.1', installed: '1.14.1' });
|
|
1268
|
+
const res = await u.runUpgrade(
|
|
1269
|
+
{ to: '1.15.0', from: join(src, 'bongos-core-1.15.0.tgz'), instance: dir, service: 'demo.service', healthUrl: 'http://x/healthz', versionUrl: 'http://x/v', rollbackOnFailure: false },
|
|
1270
|
+
{
|
|
1271
|
+
log: () => {}, err: () => {},
|
|
1272
|
+
run: pinAwareRun([], { dir }),
|
|
1273
|
+
materialize: () => ({ skills: 1 }), regenerateApiArtifacts: () => ({ ok: true, ran: [] }),
|
|
1274
|
+
fetch: async (url) => (String(url) === 'http://x/v' ? { ok: false, status: 500 } : { ok: true, status: 200 }),
|
|
1275
|
+
sleep: async () => {}, now: () => 0, healthBudgetMs: 0,
|
|
1276
|
+
}
|
|
1277
|
+
);
|
|
1278
|
+
assert.equal(res.ok, false);
|
|
1279
|
+
assert.equal(res.servedOk, false);
|
|
1280
|
+
});
|
|
1281
|
+
|
|
1282
|
+
t('--no-health-check also opts out of the served-version read-back (one documented escape hatch)', async () => {
|
|
1283
|
+
const src = mkdtempSync(join(tmpdir(), 'src-'));
|
|
1284
|
+
writeFileSync(join(src, 'bongos-core-1.15.0.tgz'), 'TGZ');
|
|
1285
|
+
const dir = scratchConsumer({ pinned: '1.14.1', installed: '1.14.1' });
|
|
1286
|
+
let fetched = false;
|
|
1287
|
+
const res = await u.runUpgrade(
|
|
1288
|
+
{ to: '1.15.0', from: join(src, 'bongos-core-1.15.0.tgz'), instance: dir, service: 'demo.service', healthCheck: false },
|
|
1289
|
+
{
|
|
1290
|
+
log: () => {}, err: () => {},
|
|
1291
|
+
run: pinAwareRun([], { dir }),
|
|
1292
|
+
materialize: () => ({ skills: 1 }), regenerateApiArtifacts: () => ({ ok: true, ran: [] }),
|
|
1293
|
+
fetch: async () => { fetched = true; return { ok: true, status: 200 }; },
|
|
1294
|
+
}
|
|
1295
|
+
);
|
|
1296
|
+
assert.equal(res.ok, true);
|
|
1297
|
+
assert.equal(fetched, false, 'nothing is polled when the confirmation is explicitly waived');
|
|
1298
|
+
});
|
|
1299
|
+
|
|
1300
|
+
t('the ROLLBACK restart escalates too — the restart after a failure is the one that matters', async () => {
|
|
1301
|
+
const src = mkdtempSync(join(tmpdir(), 'src-'));
|
|
1302
|
+
writeFileSync(join(src, 'bongos-core-1.15.0.tgz'), 'TGZ');
|
|
1303
|
+
const dir = scratchConsumer({ pinned: '1.14.1', installed: '1.14.1' });
|
|
1304
|
+
const calls = [], queries = [];
|
|
1305
|
+
// The forward restart works; the health check then fails, and the rollback's restart is the
|
|
1306
|
+
// one that needs sudo. Without deps reaching rollback's restartService there is no seam to
|
|
1307
|
+
// assert this on, and the escalation that runs after a failure would go untested.
|
|
1308
|
+
let restarts = 0;
|
|
1309
|
+
const base = pinAwareRun(calls, { dir });
|
|
1310
|
+
const run = (cmd, args = []) => {
|
|
1311
|
+
if (cmd === 'systemctl' && args[0] === 'restart') {
|
|
1312
|
+
restarts++;
|
|
1313
|
+
calls.push(`systemctl ${args.join(' ')}`);
|
|
1314
|
+
return { status: restarts === 1 ? 0 : 1 }; // forward ok, rollback needs escalation
|
|
1315
|
+
}
|
|
1316
|
+
if (cmd === 'sudo') { calls.push(`sudo ${args.join(' ')}`); return { status: 0 }; }
|
|
1317
|
+
return base(cmd, args);
|
|
1318
|
+
};
|
|
1319
|
+
const res = await u.runUpgrade(
|
|
1320
|
+
{ to: '1.15.0', from: join(src, 'bongos-core-1.15.0.tgz'), instance: dir, service: 'demo.service', healthUrl: 'http://x/healthz' },
|
|
1321
|
+
{
|
|
1322
|
+
log: () => {}, err: () => {},
|
|
1323
|
+
run, isRoot: false,
|
|
1324
|
+
materialize: () => ({ skills: 1 }), regenerateApiArtifacts: () => ({ ok: true, ran: [] }),
|
|
1325
|
+
databaseUrl: 'postgres://x/y', pg: recordingPg(queries),
|
|
1326
|
+
fetch: async () => ({ ok: false, status: 503 }), // health fails forward AND after rollback
|
|
1327
|
+
sleep: async () => {}, now: () => 0, healthBudgetMs: 0,
|
|
1328
|
+
}
|
|
1329
|
+
);
|
|
1330
|
+
assert.equal(res.ok, false);
|
|
1331
|
+
assert.equal(res.rolledBack, true);
|
|
1332
|
+
assert.equal(res.rollback.restartOk, true, 'the rollback restart succeeded via sudo -n');
|
|
1333
|
+
assert.ok(calls.includes('sudo -n systemctl restart demo.service'), 'the rollback path escalated');
|
|
1334
|
+
assert.equal(u.readInstalledCoreVersion(dir), '1.14.1', 'the previous core is back on disk');
|
|
1335
|
+
});
|
|
1336
|
+
|
|
1337
|
+
t('deriveVersionUrl: /version on the health URL origin, query + fragment dropped', () => {
|
|
1338
|
+
assert.equal(u.deriveVersionUrl('http://127.0.0.1:3002/healthz'), 'http://127.0.0.1:3002/version');
|
|
1339
|
+
assert.equal(u.deriveVersionUrl('https://demo.example.com/healthz?probe=1#x'), 'https://demo.example.com/version');
|
|
1340
|
+
assert.equal(u.deriveVersionUrl('not a url'), null);
|
|
1341
|
+
});
|
|
1342
|
+
|
|
1343
|
+
t('pollServedVersion: gives up rather than spinning when the clock never advances', async () => {
|
|
1344
|
+
const r = await u.pollServedVersion('http://x/version', {
|
|
1345
|
+
fetch: async () => ({ ok: true, status: 200, json: async () => ({}) }), // never carries a coreVersion
|
|
1346
|
+
sleep: async () => {}, now: () => 0, versionBudgetMs: 1_000, versionMaxAttempts: 3,
|
|
1347
|
+
});
|
|
1348
|
+
assert.equal(r.ok, false);
|
|
1349
|
+
assert.equal(r.attempts, 3, 'the attempt cap bounds a frozen clock');
|
|
1350
|
+
assert.match(r.error, /no coreVersion/);
|
|
1351
|
+
});
|
|
1352
|
+
|
|
1094
1353
|
await Promise.all(pending);
|
|
1095
1354
|
console.log(`\nupgrade: ${passed} passed, ${failed} failed`);
|
|
1096
1355
|
process.exit(failed ? 1 : 0);
|