@bongos/core 1.20.30 → 1.20.32
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.bongos-core.json +50 -20
- package/docs/module-api-changelog.md +4 -0
- package/docs/recipes/upgrading-the-core.md +1 -0
- package/modules/provisioning/core-upgrade.js +13 -1
- package/modules/provisioning/migrations/provisioning_031_intent_not_before.sql +19 -0
- package/modules/provisioning/migrations/provisioning_032_intent_remedied_reason.sql +19 -0
- package/modules/provisioning/provisioning.js +1 -1
- package/package-lock.json +2 -2
- package/package.json +1 -1
- package/release-notes.json +12 -0
- package/scripts/gds/intent-retry.js +52 -0
- package/scripts/gds/provision-core-upgrade.js +10 -0
- package/scripts/gds/provision.js +12 -4
- package/scripts/gds/upgrade-outcome.js +5 -1
- package/scripts/gds/wedge-remedy.js +157 -0
- package/src/module-api.js +1 -1
- package/tests/core_upgrade_door.mjs +35 -1
- package/tests/intent_retry.mjs +57 -0
- package/tests/provision.mjs +41 -16
- package/tests/wedge_remedy.mjs +268 -0
package/.bongos-core.json
CHANGED
|
@@ -2,22 +2,22 @@
|
|
|
2
2
|
"artifact": "bongos-core",
|
|
3
3
|
"manifest_schema": 1,
|
|
4
4
|
"generator": "scripts/gds/package-core.js",
|
|
5
|
-
"core_version": "1.20.
|
|
6
|
-
"core_contract": "1.20.
|
|
7
|
-
"source_commit": "
|
|
5
|
+
"core_version": "1.20.32",
|
|
6
|
+
"core_contract": "1.20.32",
|
|
7
|
+
"source_commit": "22743b01c499a1cc3bd285e9b25d6efe937aa8a7",
|
|
8
8
|
"source_ref": "HEAD",
|
|
9
|
-
"built_at": "2026-09-30T21:
|
|
9
|
+
"built_at": "2026-09-30T21:45:12.402Z",
|
|
10
10
|
"redaction": {
|
|
11
11
|
"model": "docs-redacted+functional-verbatim",
|
|
12
12
|
"docs_redacted": 562,
|
|
13
13
|
"agent_docs_stubbed": 25,
|
|
14
|
-
"functional_verbatim":
|
|
14
|
+
"functional_verbatim": 2669,
|
|
15
15
|
"rules": 3,
|
|
16
16
|
"gate_literals": 3,
|
|
17
17
|
"gate": "passed"
|
|
18
18
|
},
|
|
19
|
-
"file_count":
|
|
20
|
-
"tree_sha256": "
|
|
19
|
+
"file_count": 3257,
|
|
20
|
+
"tree_sha256": "7de0ec4831b06324127256eef3791a481ef07b33654cd0ca8b2cc058ec0e14ef",
|
|
21
21
|
"files": [
|
|
22
22
|
{
|
|
23
23
|
"path": ".claude/skills/ask-for-help/SKILL.md",
|
|
@@ -2802,7 +2802,7 @@
|
|
|
2802
2802
|
{
|
|
2803
2803
|
"path": "docs/module-api-changelog.md",
|
|
2804
2804
|
"mode": "0000644",
|
|
2805
|
-
"sha256": "
|
|
2805
|
+
"sha256": "bd97221b57256af19ec7b09b897cce082e54fca07e69500ccf485feba106317f"
|
|
2806
2806
|
},
|
|
2807
2807
|
{
|
|
2808
2808
|
"path": "docs/modules-contract.md",
|
|
@@ -3057,7 +3057,7 @@
|
|
|
3057
3057
|
{
|
|
3058
3058
|
"path": "docs/recipes/upgrading-the-core.md",
|
|
3059
3059
|
"mode": "0000644",
|
|
3060
|
-
"sha256": "
|
|
3060
|
+
"sha256": "6d44cfdcd7fa92d3b6c6514af69356d6e986d5443219251cd091685918e4e1f2"
|
|
3061
3061
|
},
|
|
3062
3062
|
{
|
|
3063
3063
|
"path": "docs/recipes/windows-builders.md",
|
|
@@ -7282,7 +7282,7 @@
|
|
|
7282
7282
|
{
|
|
7283
7283
|
"path": "modules/provisioning/core-upgrade.js",
|
|
7284
7284
|
"mode": "0000644",
|
|
7285
|
-
"sha256": "
|
|
7285
|
+
"sha256": "5467ab673dfad5969dec0029188d0be18ea82732a26f99bca720de6680e1d618"
|
|
7286
7286
|
},
|
|
7287
7287
|
{
|
|
7288
7288
|
"path": "modules/provisioning/cost-ledger.js",
|
|
@@ -7464,6 +7464,16 @@
|
|
|
7464
7464
|
"mode": "0000644",
|
|
7465
7465
|
"sha256": "7287181b4b42d3e391cd97e19ffa5fbc76026e81ad6f54648eb12a69dfea55a1"
|
|
7466
7466
|
},
|
|
7467
|
+
{
|
|
7468
|
+
"path": "modules/provisioning/migrations/provisioning_031_intent_not_before.sql",
|
|
7469
|
+
"mode": "0000644",
|
|
7470
|
+
"sha256": "a93d8946cbe4631b2af64b2e3638dca188981032a08f5d29d63e4d0adb22d235"
|
|
7471
|
+
},
|
|
7472
|
+
{
|
|
7473
|
+
"path": "modules/provisioning/migrations/provisioning_032_intent_remedied_reason.sql",
|
|
7474
|
+
"mode": "0000644",
|
|
7475
|
+
"sha256": "73794821e3beca0129a435747e89dd5755b26f054dca0d8ca0b421392abf7127"
|
|
7476
|
+
},
|
|
7467
7477
|
{
|
|
7468
7478
|
"path": "modules/provisioning/module.json",
|
|
7469
7479
|
"mode": "0000644",
|
|
@@ -7512,7 +7522,7 @@
|
|
|
7512
7522
|
{
|
|
7513
7523
|
"path": "modules/provisioning/provisioning.js",
|
|
7514
7524
|
"mode": "0000644",
|
|
7515
|
-
"sha256": "
|
|
7525
|
+
"sha256": "fb4f09b621654eebe595dcbca305773ea11ed063506a6e4116391e43566a540d"
|
|
7516
7526
|
},
|
|
7517
7527
|
{
|
|
7518
7528
|
"path": "modules/provisioning/public-refusal.js",
|
|
@@ -8982,12 +8992,12 @@
|
|
|
8982
8992
|
{
|
|
8983
8993
|
"path": "package-lock.json",
|
|
8984
8994
|
"mode": "0000644",
|
|
8985
|
-
"sha256": "
|
|
8995
|
+
"sha256": "cd508994cbd9d44ef721787e0eb7c803bfeda98459587d53ba726a36cff2581e"
|
|
8986
8996
|
},
|
|
8987
8997
|
{
|
|
8988
8998
|
"path": "package.json",
|
|
8989
8999
|
"mode": "0000644",
|
|
8990
|
-
"sha256": "
|
|
9000
|
+
"sha256": "5c53762b7665282aeca491b037cb2c02ee8c6def6c00018262596235d02bbb64"
|
|
8991
9001
|
},
|
|
8992
9002
|
{
|
|
8993
9003
|
"path": "public-docs/index.html",
|
|
@@ -9007,7 +9017,7 @@
|
|
|
9007
9017
|
{
|
|
9008
9018
|
"path": "release-notes.json",
|
|
9009
9019
|
"mode": "0000644",
|
|
9010
|
-
"sha256": "
|
|
9020
|
+
"sha256": "1f7d261d95f2fbaf1d0e438e9c3ad7f13773c5b56d349263815d5be03ebab401"
|
|
9011
9021
|
},
|
|
9012
9022
|
{
|
|
9013
9023
|
"path": "scripts/bongos-mcp.js",
|
|
@@ -9739,6 +9749,11 @@
|
|
|
9739
9749
|
"mode": "0000644",
|
|
9740
9750
|
"sha256": "7b2402543c3b1eadbafc14907e6dc6689b434ac65fb9c709115dd81a92eb13eb"
|
|
9741
9751
|
},
|
|
9752
|
+
{
|
|
9753
|
+
"path": "scripts/gds/intent-retry.js",
|
|
9754
|
+
"mode": "0000644",
|
|
9755
|
+
"sha256": "056622ecd8327527f4fb98b37ce8af90778d42c816455431ed3dd2f76800a75b"
|
|
9756
|
+
},
|
|
9742
9757
|
{
|
|
9743
9758
|
"path": "scripts/gds/key-card.js",
|
|
9744
9759
|
"mode": "0000644",
|
|
@@ -9967,7 +9982,7 @@
|
|
|
9967
9982
|
{
|
|
9968
9983
|
"path": "scripts/gds/provision-core-upgrade.js",
|
|
9969
9984
|
"mode": "0000644",
|
|
9970
|
-
"sha256": "
|
|
9985
|
+
"sha256": "53904eb7a91f7f0c7c80c3caaceed29d6317652d6830e1f575198b0ba9f5158f"
|
|
9971
9986
|
},
|
|
9972
9987
|
{
|
|
9973
9988
|
"path": "scripts/gds/provision-disconnect.js",
|
|
@@ -10022,7 +10037,7 @@
|
|
|
10022
10037
|
{
|
|
10023
10038
|
"path": "scripts/gds/provision.js",
|
|
10024
10039
|
"mode": "0000644",
|
|
10025
|
-
"sha256": "
|
|
10040
|
+
"sha256": "dd972cc62ed8282af5f1239a3b728761fb06a414120cac8a39b38838e8301f7d"
|
|
10026
10041
|
},
|
|
10027
10042
|
{
|
|
10028
10043
|
"path": "scripts/gds/publish-credential-check.js",
|
|
@@ -10637,7 +10652,7 @@
|
|
|
10637
10652
|
{
|
|
10638
10653
|
"path": "scripts/gds/upgrade-outcome.js",
|
|
10639
10654
|
"mode": "0000644",
|
|
10640
|
-
"sha256": "
|
|
10655
|
+
"sha256": "3bd19ba68a09a6739379bdd626bcf5b93addccdf1cd8774e20857dc9a66686fb"
|
|
10641
10656
|
},
|
|
10642
10657
|
{
|
|
10643
10658
|
"path": "scripts/gds/upgrade.js",
|
|
@@ -10669,6 +10684,11 @@
|
|
|
10669
10684
|
"mode": "0000644",
|
|
10670
10685
|
"sha256": "cf034e9c4c5887dde47951272dba1e31b9b94a212640181de1cc3086e225b357"
|
|
10671
10686
|
},
|
|
10687
|
+
{
|
|
10688
|
+
"path": "scripts/gds/wedge-remedy.js",
|
|
10689
|
+
"mode": "0000644",
|
|
10690
|
+
"sha256": "abe6ecedc808f3f57ef2dc5e9f953792ae26f39625425952a8ce529dbf6de576"
|
|
10691
|
+
},
|
|
10672
10692
|
{
|
|
10673
10693
|
"path": "scripts/gds/worktree-claim-guard.js",
|
|
10674
10694
|
"mode": "0000644",
|
|
@@ -11132,7 +11152,7 @@
|
|
|
11132
11152
|
{
|
|
11133
11153
|
"path": "src/module-api.js",
|
|
11134
11154
|
"mode": "0000644",
|
|
11135
|
-
"sha256": "
|
|
11155
|
+
"sha256": "a1f03380a933a2c9fdcd63f55f0152bf7d36101135d171c729f7cd79c9ef0ec4"
|
|
11136
11156
|
},
|
|
11137
11157
|
{
|
|
11138
11158
|
"path": "src/module-loader/catalog.js",
|
|
@@ -12077,7 +12097,7 @@
|
|
|
12077
12097
|
{
|
|
12078
12098
|
"path": "tests/core_upgrade_door.mjs",
|
|
12079
12099
|
"mode": "0000644",
|
|
12080
|
-
"sha256": "
|
|
12100
|
+
"sha256": "8c8b03b8a3bb83b0c733726f3aa1256b6af5ac61c76e6af465fe2bc4f792ea6b"
|
|
12081
12101
|
},
|
|
12082
12102
|
{
|
|
12083
12103
|
"path": "tests/core_upgrade_owner_door.mjs",
|
|
@@ -13744,6 +13764,11 @@
|
|
|
13744
13764
|
"mode": "0000644",
|
|
13745
13765
|
"sha256": "9e69603c05fae6c39e2db7336d6569327f17365ff46ef05b1074943dc34a49e7"
|
|
13746
13766
|
},
|
|
13767
|
+
{
|
|
13768
|
+
"path": "tests/intent_retry.mjs",
|
|
13769
|
+
"mode": "0000644",
|
|
13770
|
+
"sha256": "750745cf866dcf1012f2677881738aa64ae2b8ceaed078ab688612a8b0d2208c"
|
|
13771
|
+
},
|
|
13747
13772
|
{
|
|
13748
13773
|
"path": "tests/interaction_posture.mjs",
|
|
13749
13774
|
"mode": "0000644",
|
|
@@ -14702,7 +14727,7 @@
|
|
|
14702
14727
|
{
|
|
14703
14728
|
"path": "tests/provision.mjs",
|
|
14704
14729
|
"mode": "0000644",
|
|
14705
|
-
"sha256": "
|
|
14730
|
+
"sha256": "46b2235ee186ba304c18f5cf3002a9e5311f7bda11fa230bbada8a7a43bec3cd"
|
|
14706
14731
|
},
|
|
14707
14732
|
{
|
|
14708
14733
|
"path": "tests/provision_account_identity.mjs",
|
|
@@ -16214,6 +16239,11 @@
|
|
|
16214
16239
|
"mode": "0000644",
|
|
16215
16240
|
"sha256": "456b20334164a710f59116d6a6c4b85090dc0e65668284d0a736e6ee1df5dea6"
|
|
16216
16241
|
},
|
|
16242
|
+
{
|
|
16243
|
+
"path": "tests/wedge_remedy.mjs",
|
|
16244
|
+
"mode": "0000644",
|
|
16245
|
+
"sha256": "9810e6b2cf5ec77fad5af12b195a3ff4919143b146ff43f106c67c4090c61c25"
|
|
16246
|
+
},
|
|
16217
16247
|
{
|
|
16218
16248
|
"path": "tests/wizard_born_first_board.mjs",
|
|
16219
16249
|
"mode": "0000644",
|
|
@@ -2717,5 +2717,9 @@ is load-bearing: the script throws rather than guess if it is missing, and
|
|
|
2717
2717
|
landed since 1.20.28 with no explicit bump. run 36775277990. (task 1002620)
|
|
2718
2718
|
1.20.30 — CI auto-patch (publish-on-merge, ADR 0161): carrier for merges
|
|
2719
2719
|
landed since 1.20.29 with no explicit bump. run 36777226204. (task 1002620)
|
|
2720
|
+
1.20.31 — CI auto-patch (publish-on-merge, ADR 0161): carrier for merges
|
|
2721
|
+
landed since 1.20.30 with no explicit bump. run 36779142919. (task 1002620)
|
|
2722
|
+
1.20.32 — CI auto-patch (publish-on-merge, ADR 0161): carrier for merges
|
|
2723
|
+
landed since 1.20.31 with no explicit bump. run 36781343271. (task 1002620)
|
|
2720
2724
|
---------------------------------------------------------------------------
|
|
2721
2725
|
```
|
|
@@ -179,6 +179,7 @@ running a go-live.
|
|
|
179
179
|
|
|
180
180
|
## Gotchas
|
|
181
181
|
|
|
182
|
+
- **The Deploy door fixes three known snags by itself, once** (task 1004448, `scripts/gds/wedge-remedy.js`). When a core move through `/deploy` fails on a **dirty tree** (it commits the tracked changes inside the project), a **full disk** (it prunes that project's own nightly dumps down to the newest 7) or a **database behind its code** (it migrates that database with its own `PGDATABASE` before the code moves), the runner applies the fix, records a `core-upgrade-remedy` event on the project, and tries the move once more on the next tick. If the retry fails for the same reason, or the fix itself fails, the move is retired as `unknown` (`remedy_did_not_hold` / `remedy_failed`) for a person — never a second fix. The manual fixes below still apply to the unattended nightly sweep, which does not go through the door.
|
|
182
183
|
- **`--registry` needs no credential at all** (task 1003878). `@bongos/core` is **public** on npm (task 1003872), so `bongos upgrade --registry` resolves it from the default registry like any other dependency — no token, no login, and no instance `.npmrc`. An install failure on this path is therefore a *real* failure (network, registry outage, or a version that was never published), never a missing credential.
|
|
183
184
|
*History, so the symptom is recognisable on an old instance:* the core used to publish private, and the instance `.npmrc` read `_authToken=<redacted> npm gives that per-project key precedence, so with the env var unset it expanded EMPTY and **masked a perfectly valid literal token in `~/.npmrc`** — npm went anonymous, the private core 404'd, and the error blamed a credential the box actually held. Task 1002754 worked around it by resolving the token in-process; task 1003878 deleted the whole mechanism along with the `.npmrc` that caused it. An instance still carrying a committed `.npmrc` from a pre-1003878 scaffold can safely delete it.
|
|
184
185
|
- **A green `publish` run is not evidence of a publish.** The publish-on-merge lane succeeds on its no-op paths too (cascade terminator, armed-gate). Read the run's **job summary**, which now states the verdict explicitly: `PUBLISHED @bongos/core@<v>` vs `NO-OP — nothing was published this run`.
|
|
@@ -128,6 +128,16 @@ function coreUpgradeFault(inst, to, live = null) {
|
|
|
128
128
|
async function enqueueCoreUpgrade(provisioning, db, instanceId, to, requestedBy, action = 'core-upgrade') {
|
|
129
129
|
const { intent, created } = await provisioning.enqueueIntent(db, instanceId, action, requestedBy, to);
|
|
130
130
|
if (created) return { intent, created: true };
|
|
131
|
+
// THE SAME MOVE, WAITING OUT A RETRY (task 1004447). A transient failure re-pends the
|
|
132
|
+
// intent with a not_before up to six hours away, and it keeps the single open slot all
|
|
133
|
+
// that time — so without this, the owner pressing Move again was told "a run is pending,
|
|
134
|
+
// try again later" about the very move they were asking for. Same action, same target,
|
|
135
|
+
// still pending: bring its next attempt forward instead. Anything else stays a conflict.
|
|
136
|
+
if (intent && intent.state === 'pending' && intent.action === action && String(intent.target_version || '') === String(to)
|
|
137
|
+
&& intent.not_before && new Date(intent.not_before).getTime() > Date.now()) {
|
|
138
|
+
await db.query("UPDATE provisioning_intents SET not_before = NULL, updated_at = now() WHERE id = $1 AND state = 'pending'", [intent.id]);
|
|
139
|
+
return { intent: { ...intent, not_before: null }, created: false, retryNow: true };
|
|
140
|
+
}
|
|
131
141
|
return { intent, created: false, conflict: 'open_intent' };
|
|
132
142
|
}
|
|
133
143
|
|
|
@@ -336,7 +346,9 @@ function previewFromSnapshot(inst, snap, openIntent, lastMove) {
|
|
|
336
346
|
// and a stale pass is worse than no answer because it reads as a guarantee.
|
|
337
347
|
preflight: preflightForOffer(pf, served, recommended),
|
|
338
348
|
preflight_at: pf && recommended && pf.target === recommended ? (snap.preflight_at || null) : null,
|
|
339
|
-
|
|
349
|
+
// not_before: a failed move waiting out its retry backoff (task 1004447) — the page can
|
|
350
|
+
// say "trying again at …" instead of an unexplained "pending" for up to six hours.
|
|
351
|
+
open_intent: openIntent ? { action: openIntent.action, state: openIntent.state, target_version: openIntent.target_version || null, not_before: openIntent.not_before || null, last_error: openIntent.not_before && openIntent.last_error ? String(openIntent.last_error).slice(0, 600) : null } : null,
|
|
340
352
|
// THE LAST MOVE, WHATEVER BECAME OF IT (task 1004145).
|
|
341
353
|
//
|
|
342
354
|
// `open_intent` is `state IN ('pending','running')` — by the index's own definition.
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
-- provisioning_031_intent_not_before.sql — a failed intent waits before its next attempt
|
|
2
|
+
-- (task 1004447, goal 1000090 deploy self-heal).
|
|
3
|
+
--
|
|
4
|
+
-- WHY. A failed intent was re-pended with no delay, and the drain loop claims the oldest
|
|
5
|
+
-- pending row — so one runner tick ran the same failing move up to MAX_INTENT_ATTEMPTS
|
|
6
|
+
-- times back to back, restarting the service each time a pin had moved, and then gave up.
|
|
7
|
+
-- A registry blip or a release still publishing needs minutes, not milliseconds.
|
|
8
|
+
--
|
|
9
|
+
-- not_before: the earliest time the runner may claim this row again. NULL = now (every row
|
|
10
|
+
-- written before this column, and every fresh request). claimNextIntent skips a row whose
|
|
11
|
+
-- not_before is still in the future; scripts/gds/provision.js sets it on each re-pend.
|
|
12
|
+
--
|
|
13
|
+
-- Additive and namespaced (ADR 0083): no down-migration, idempotent — safe to re-run.
|
|
14
|
+
|
|
15
|
+
BEGIN;
|
|
16
|
+
|
|
17
|
+
ALTER TABLE provisioning_intents ADD COLUMN IF NOT EXISTS not_before timestamptz;
|
|
18
|
+
|
|
19
|
+
COMMIT;
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
-- provisioning_032_intent_remedied_reason.sql — which known snag the runner already fixed
|
|
2
|
+
-- for a core move (task 1004448, goal 1000090 deploy self-heal).
|
|
3
|
+
--
|
|
4
|
+
-- WHY. A core move that fails on a known snag (a dirty checkout, a full disk, a database
|
|
5
|
+
-- behind its code) now has the fix applied by the runner and is tried once more
|
|
6
|
+
-- (scripts/gds/wedge-remedy.js). "Once" has to survive between runner ticks, so the reason
|
|
7
|
+
-- the fix was applied for is kept on the intent. A retry that fails on the SAME reason is
|
|
8
|
+
-- retired for a person instead of fixed again — never a loop.
|
|
9
|
+
--
|
|
10
|
+
-- remedied_reason: a reason code from scripts/gds/upgrade-outcome.js (fixed vocabulary,
|
|
11
|
+
-- never builder text). NULL = no fix has been applied for this intent.
|
|
12
|
+
--
|
|
13
|
+
-- Additive and namespaced (ADR 0083): no down-migration, idempotent — safe to re-run.
|
|
14
|
+
|
|
15
|
+
BEGIN;
|
|
16
|
+
|
|
17
|
+
ALTER TABLE provisioning_intents ADD COLUMN IF NOT EXISTS remedied_reason text;
|
|
18
|
+
|
|
19
|
+
COMMIT;
|
|
@@ -1254,7 +1254,7 @@ async function claimNextIntent(db) {
|
|
|
1254
1254
|
SET state = 'running', attempts = attempts + 1, updated_at = now()
|
|
1255
1255
|
WHERE id = (
|
|
1256
1256
|
SELECT id FROM provisioning_intents
|
|
1257
|
-
WHERE state = 'pending'
|
|
1257
|
+
WHERE state = 'pending' AND (not_before IS NULL OR not_before <= now()) -- a re-pended failure waits out its backoff (task 1004447)
|
|
1258
1258
|
ORDER BY created_at
|
|
1259
1259
|
FOR UPDATE SKIP LOCKED
|
|
1260
1260
|
LIMIT 1
|
package/package-lock.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@bongos/core",
|
|
3
|
-
"version": "1.20.
|
|
3
|
+
"version": "1.20.32",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "@bongos/core",
|
|
9
|
-
"version": "1.20.
|
|
9
|
+
"version": "1.20.32",
|
|
10
10
|
"license": "AGPL-3.0-or-later",
|
|
11
11
|
"dependencies": {
|
|
12
12
|
"express": "^4.21.2",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@bongos/core",
|
|
3
|
-
"version": "1.20.
|
|
3
|
+
"version": "1.20.32",
|
|
4
4
|
"description": "Cloud Bongos — the AI-first build platform core (GDS + platform surfaces + module system), installed as a versioned dependency (ADR 0108).",
|
|
5
5
|
"license": "AGPL-3.0-or-later",
|
|
6
6
|
"main": "src/platform-server.js",
|
package/release-notes.json
CHANGED
|
@@ -8252,5 +8252,17 @@
|
|
|
8252
8252
|
"id": "1004446",
|
|
8253
8253
|
"text": "When a software update fails, the system now records what kind of failure it was — a temporary glitch, a known snag, a rule only the owner can change, or something unknown — and tells the truth about it instead of always sayin"
|
|
8254
8254
|
}
|
|
8255
|
+
],
|
|
8256
|
+
"1.20.31": [
|
|
8257
|
+
{
|
|
8258
|
+
"id": "1004447",
|
|
8259
|
+
"text": "A software update that fails for a temporary reason now waits and tries again later — after 15 minutes, then an hour, then six hours — instead of hammering the same failing step three times in a row. Failures that retrying can"
|
|
8260
|
+
}
|
|
8261
|
+
],
|
|
8262
|
+
"1.20.32": [
|
|
8263
|
+
{
|
|
8264
|
+
"id": "1004448",
|
|
8265
|
+
"text": "When an update to a project gets stuck on one of three known problems (leftover file changes, a full disk, or a database that is behind), the system now fixes it by itself and tries again once. If that doesn't work, it stops a"
|
|
8266
|
+
}
|
|
8255
8267
|
]
|
|
8256
8268
|
}
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
//
|
|
3
|
+
// scripts/gds/intent-retry.js — when a failed provisioning intent is tried again, and when it
|
|
4
|
+
// is retired instead (task 1004447, goal 1000090 deploy self-heal).
|
|
5
|
+
//
|
|
6
|
+
// WHY. The drain loop re-pended a failure with no delay and claims the oldest pending row, so
|
|
7
|
+
// one tick ran the same failing move up to MAX_INTENT_ATTEMPTS times back to back — seconds
|
|
8
|
+
// apart, restarting the service after every pin move — and then gave up. That is the wrong
|
|
9
|
+
// answer for each kind of failure upgrade-outcome.js now names:
|
|
10
|
+
// transient — the registry blipped, the release is still publishing: WAIT, then retry
|
|
11
|
+
// (about 15 minutes, 1 hour, 6 hours), which is what a person would do.
|
|
12
|
+
// known_wedge, needs_decision, unknown — trying the same thing again cannot change the
|
|
13
|
+
// answer, so the intent is RETIRED at once and the failure left standing
|
|
14
|
+
// for the next step (a fix, the owner, a person) to act on.
|
|
15
|
+
// A failure with no class (every action other than a core move) keeps its old attempt budget,
|
|
16
|
+
// but each retry now waits DEFAULT_RETRY_DELAY_MS — at least the next tick, never the same one.
|
|
17
|
+
//
|
|
18
|
+
// PURE: no I/O. The drain loop (scripts/gds/provision.js cmdRunIntents) asks and obeys.
|
|
19
|
+
|
|
20
|
+
const MINUTE = 60 * 1000;
|
|
21
|
+
const TRANSIENT_BACKOFF_MS = Object.freeze([15 * MINUTE, 60 * MINUTE, 6 * 60 * MINUTE]);
|
|
22
|
+
const DEFAULT_RETRY_DELAY_MS = MINUTE;
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* What to do with an intent whose run just failed. PURE.
|
|
26
|
+
* @param {object} p
|
|
27
|
+
* @param {number} p.attempts the attempt that just ran (claimNextIntent increments first)
|
|
28
|
+
* @param {object} [p.failure] `{ class, reason }` from the result, when classified
|
|
29
|
+
* @param {boolean} [p.terminal] the leg said it must not be retried unattended
|
|
30
|
+
* @param {number} p.maxAttempts the runner's budget for unclassified failures
|
|
31
|
+
* @returns {{ action: 'retire' } | { action: 'retry', delayMs: number }}
|
|
32
|
+
*/
|
|
33
|
+
function retryDecision({ attempts, failure, terminal, maxAttempts }) {
|
|
34
|
+
if (terminal) return { action: 'retire' };
|
|
35
|
+
if (failure && failure.class) {
|
|
36
|
+
if (failure.class !== 'transient') return { action: 'retire' };
|
|
37
|
+
const i = Math.max(0, (attempts || 1) - 1);
|
|
38
|
+
return i < TRANSIENT_BACKOFF_MS.length ? { action: 'retry', delayMs: TRANSIENT_BACKOFF_MS[i] } : { action: 'retire' };
|
|
39
|
+
}
|
|
40
|
+
return (attempts || 0) >= maxAttempts ? { action: 'retire' } : { action: 'retry', delayMs: DEFAULT_RETRY_DELAY_MS };
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* The attempt ceiling checked BEFORE an intent runs. A row waiting out a transient backoff
|
|
45
|
+
* has one attempt per backoff step plus the first, so the pre-run guard must not retire it
|
|
46
|
+
* on the default budget before its last scheduled try. PURE.
|
|
47
|
+
*/
|
|
48
|
+
function maxAttemptsFor(intent, maxAttempts) {
|
|
49
|
+
return intent && intent.failure_class === 'transient' ? Math.max(maxAttempts, TRANSIENT_BACKOFF_MS.length + 1) : maxAttempts;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
module.exports = { TRANSIENT_BACKOFF_MS, DEFAULT_RETRY_DELAY_MS, retryDecision, maxAttemptsFor };
|
|
@@ -241,6 +241,16 @@ async function coreUpgradeInstance(inst, deps, intent) {
|
|
|
241
241
|
log(` already serving core ${to} — nothing to move`);
|
|
242
242
|
return { ok: true, noop: true };
|
|
243
243
|
}
|
|
244
|
+
// ---- 2b. is the database current for the code it runs NOW (task 1004448) -------------
|
|
245
|
+
// A database behind its own code is already failing every page that touches the missing
|
|
246
|
+
// tables (the 2026-07-04 shape), and moving the code on top of it tangles a catch-up
|
|
247
|
+
// migration into a pin move. So it is refused as a known snag, and wedge-remedy.js
|
|
248
|
+
// migrates that database and tries the move once more. An answer it cannot read proceeds.
|
|
249
|
+
if (apply) {
|
|
250
|
+
const probe = boxExec(versionCmd(inst), { allowFail: true });
|
|
251
|
+
const pending = parseSchemaPending(probe && !probe.softFailed ? (probe.stdout || '') : null);
|
|
252
|
+
if (pending > 0) return failed('schema_pending', `the project's database is ${pending} migration(s) behind the code it is running now — it has to catch up before the core moves`);
|
|
253
|
+
}
|
|
244
254
|
|
|
245
255
|
// ---- 3. drive the proven path -----------------------------------------------------
|
|
246
256
|
// Exactly the invocation docs/recipes/core-release-pipeline.md gate 3 documents, plus
|
package/scripts/gds/provision.js
CHANGED
|
@@ -46,6 +46,7 @@ const path = require('node:path');
|
|
|
46
46
|
const cp = require('node:child_process');
|
|
47
47
|
const crypto = require('node:crypto');
|
|
48
48
|
|
|
49
|
+
const retry = require('./intent-retry.js'); // when a failed intent is tried again (task 1004447)
|
|
49
50
|
const { CONFIG, MANIFEST_UA, softFailResult, MANIFEST_VERIFY_INTERVAL_MS, MANIFEST_VERIFY_TRIES, MAX_INTENT_ATTEMPTS, REPO_ROOT, coreVersionSafe, hasFlag, loadDeps, oauthSecret, provisionerBotEmail } = require('./provision-config.js');
|
|
50
51
|
const { accountPreflight, appUser, alreadyScaffolded, buildCorePinRefreshCommit, coreCheckoutHasGit, dbCreateCmd, dbName, dbRoleCmd, dbRoleCmdShown, dbRoleGrantsCmd, ensureInstanceUserCmd, generateDbPassword, grantInstanceRepoReadCmd, instanceDbRole, instanceStateDir, instanceUser, deployKeyPath, deployKeyTitle, ensurePrivateRepoAccess, installedCorePackDir, installedCoreTarball, instanceInitSpec, instanceRepoRemote, migrateCmd, onboardMode, ownerLoginOf, parseTargetRef, planCorePinRefresh, refreshStandaloneCorePin, resolveOwnerGithubToken, resolveVendorableCoreTarball, safeVersionLabel, scaffoldStandaloneRepo, seedFirstVersionCmd, standaloneInstallCmd, standaloneMigrateCmd, standalonePullCmd, standaloneRegenDocsCmd, standaloneRoot } = require('./provision-repo.js');
|
|
51
52
|
const { backupScriptPath, backupService, backupServicePath, backupTimer, backupTimerPath, backupUnitName, instanceManifestCmd, originEnvVarsFor, serviceUnit, serviceUnitPath, settingsConsumed, settingsEnvVarsFor, renamedEnvCarry, upsertEnvVars, webEnvBody, webEnvPath } = require('./provision-units.js');
|
|
@@ -1171,8 +1172,8 @@ async function cmdRunIntents(deps) {
|
|
|
1171
1172
|
const intent = await provisioning.claimNextIntent(db);
|
|
1172
1173
|
if (!intent) break;
|
|
1173
1174
|
ran++;
|
|
1174
|
-
if (intent.attempts > MAX_INTENT_ATTEMPTS) {
|
|
1175
|
-
await provisioning.resolveIntent(db, intent.id, 'error', `exceeded ${MAX_INTENT_ATTEMPTS} attempts`);
|
|
1175
|
+
if (intent.attempts > retry.maxAttemptsFor(intent, MAX_INTENT_ATTEMPTS)) { // a transient core move gets one try per backoff step (task 1004447)
|
|
1176
|
+
await provisioning.resolveIntent(db, intent.id, 'error', `exceeded ${retry.maxAttemptsFor(intent, MAX_INTENT_ATTEMPTS)} attempts`);
|
|
1176
1177
|
errors++; log(` ✖ intent #${intent.id} retired (too many attempts)`); continue;
|
|
1177
1178
|
}
|
|
1178
1179
|
try {
|
|
@@ -1192,7 +1193,14 @@ async function cmdRunIntents(deps) {
|
|
|
1192
1193
|
else if (intent.action === 'repo-private' || intent.action === 'render-standup') result = intent.action === 'repo-private' ? await require('./provision-repo-private.js').repoPrivateInstance(inst, deps) : await require('./provision-render.js').renderStandupInstance(inst, deps, intent); // wire the deploy key, THEN flip (task 1004193) · spend the borrowed Render key once (task 1004183); required here — this file is at the size ratchet
|
|
1193
1194
|
else result = { ok: false, error: `unknown_action '${intent.action}' — runner core ${coreVersionSafe() || '?'} does not know it (runner too old?); refusing to guess` };
|
|
1194
1195
|
if (result && result.ok === false) {
|
|
1195
|
-
|
|
1196
|
+
// WAIT OR RETIRE (task 1004447, intent-retry.js): a transient failure waits ~15m/1h/6h; a
|
|
1197
|
+
// classified non-transient one is retired at once (retrying cannot change its answer);
|
|
1198
|
+
// an unclassified one keeps its budget but never re-runs in the same tick.
|
|
1199
|
+
const fix = await require('./wedge-remedy.js').remedyAfterFailure({ intent, inst, result, deps }); // a KNOWN snag: fix it, try once more (task 1004448)
|
|
1200
|
+
if (fix.repended) { errors++; continue; }
|
|
1201
|
+
Object.assign(result, { failure: fix.failure, error: fix.error });
|
|
1202
|
+
const next = retry.retryDecision({ attempts: intent.attempts, failure: result.failure, terminal: result.terminal, maxAttempts: MAX_INTENT_ATTEMPTS });
|
|
1203
|
+
if (next.action === 'retire') { // terminal: a leg that must not be retried unattended (the Render leg's key is already gone — task 1004183)
|
|
1196
1204
|
await provisioning.resolveIntent(db, intent.id, 'error', result.error || 'op failed');
|
|
1197
1205
|
// A retired intent stops silently otherwise (F19): record the reason
|
|
1198
1206
|
// on the instance WITHOUT changing its lifecycle status — passing the
|
|
@@ -1202,7 +1210,7 @@ async function cmdRunIntents(deps) {
|
|
|
1202
1210
|
await provisioning.setInstanceStatus(db, intent.instance_id, inst.status, { error_note: result.error || 'op failed', error_note_action: intent.action }).catch(() => {});
|
|
1203
1211
|
if (result.failure) await recordFailureClass(db, intent.id, result.failure); // WHAT KIND of failure, beside the sentence (task 1004446)
|
|
1204
1212
|
} else {
|
|
1205
|
-
await db.query(`UPDATE provisioning_intents SET state='pending', last_error=$2, updated_at=now() WHERE id=$1`, [intent.id, result.error || 'op failed']);
|
|
1213
|
+
await db.query(`UPDATE provisioning_intents SET state='pending', last_error=$2, not_before = now() + ($3 * interval '1 millisecond'), updated_at=now() WHERE id=$1`, [intent.id, result.error || 'op failed', next.delayMs]);
|
|
1206
1214
|
if (result.failure) await recordFailureClass(db, intent.id, result.failure);
|
|
1207
1215
|
}
|
|
1208
1216
|
errors++;
|
|
@@ -19,7 +19,7 @@
|
|
|
19
19
|
//
|
|
20
20
|
// THE FOUR CLASSES the rest of the self-heal chain keys on:
|
|
21
21
|
// transient — try again later and it may simply work (registry blip, not published yet)
|
|
22
|
-
// known_wedge — a known snag with a known fix the runner
|
|
22
|
+
// known_wedge — a known snag with a known fix the runner applies itself, once (wedge-remedy.js)
|
|
23
23
|
// needs_decision — a rule the owner set is refusing it; only the owner can change that
|
|
24
24
|
// unknown — anything else, and every rollback that failed; a person should look
|
|
25
25
|
//
|
|
@@ -42,6 +42,7 @@ const REASON_CLASS = Object.freeze({
|
|
|
42
42
|
bad_slug: 'unknown',
|
|
43
43
|
installed_unreadable: 'unknown',
|
|
44
44
|
served_mismatch: 'unknown',
|
|
45
|
+
schema_pending: 'known_wedge', // the database is behind the code it runs NOW (task 1004448)
|
|
45
46
|
// upgrade.js refusals, before the pin moves
|
|
46
47
|
dirty_tree: 'known_wedge',
|
|
47
48
|
downgrade: 'needs_decision',
|
|
@@ -59,6 +60,9 @@ const REASON_CLASS = Object.freeze({
|
|
|
59
60
|
migrate: 'unknown',
|
|
60
61
|
restart: 'unknown',
|
|
61
62
|
health: 'unknown',
|
|
63
|
+
// the runner's one automatic fix did not clear it, or itself failed (task 1004448)
|
|
64
|
+
remedy_did_not_hold: 'unknown',
|
|
65
|
+
remedy_failed: 'unknown',
|
|
62
66
|
// the one that must never be retried unattended
|
|
63
67
|
rollback_failed: 'unknown',
|
|
64
68
|
});
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
//
|
|
3
|
+
// scripts/gds/wedge-remedy.js — a core move that failed on a KNOWN snag fixes the snag and
|
|
4
|
+
// tries once more (task 1004448, goal 1000090 deploy self-heal).
|
|
5
|
+
//
|
|
6
|
+
// WHY. upgrade-outcome.js names three failures that are not a fault in the release and not a
|
|
7
|
+
// decision the owner has to make, but a mess on the box with one known fix — the fixes the
|
|
8
|
+
// owner has been doing over SSH (docs/recipes/upgrading-the-core.md, Gotchas):
|
|
9
|
+
// dirty_tree — the project's checkout has uncommitted changes to tracked files (the
|
|
10
|
+
// upgrade's own rewrites, left behind by an earlier move). Fix: commit them
|
|
11
|
+
// inside the project.
|
|
12
|
+
// disk_full — the move ran out of disk. Fix: prune this project's OWN nightly
|
|
13
|
+
// pg_dumps down to the newest few (the only thing on the box this runner
|
|
14
|
+
// knows is safe to delete — they are copies, and the newest are kept).
|
|
15
|
+
// schema_pending — the project's database is behind the code it is running NOW. Fix:
|
|
16
|
+
// migrate that database, with its own PGDATABASE, before the code moves.
|
|
17
|
+
// intent-retry.js retires every known_wedge at once, so on its own the failure only waited
|
|
18
|
+
// for a person. This file runs the fix, records it as an event on the project, and re-pends
|
|
19
|
+
// the move for ONE more try.
|
|
20
|
+
//
|
|
21
|
+
// AT MOST ONCE PER FAILURE. The reason a fix was applied for is stored on the intent
|
|
22
|
+
// (provisioning_intents.remedied_reason, migration provisioning_032). If the retry fails on
|
|
23
|
+
// the SAME reason, the fix did not hold: the move is retired as `unknown` /
|
|
24
|
+
// remedy_did_not_hold, which is the class that reaches a person — never a second fix, never
|
|
25
|
+
// a loop. A fix that itself fails is retired the same way (remedy_failed). A different
|
|
26
|
+
// known reason on the retry gets its own one fix (a disk full after a dirty tree is a new
|
|
27
|
+
// snag, not the same one twice); the attempt budget still bounds the whole run.
|
|
28
|
+
//
|
|
29
|
+
// 'tagged-not-published' needs nothing here: it is `not_published`, a transient failure that
|
|
30
|
+
// intent-retry.js already waits out.
|
|
31
|
+
//
|
|
32
|
+
// planRemedy is PURE (the command, never run). remedyAfterFailure does the I/O through the
|
|
33
|
+
// runner's own exec and db, so every fix is testable with an injected exec.
|
|
34
|
+
|
|
35
|
+
const { CONFIG } = require('./provision-config.js');
|
|
36
|
+
const { dbName, selfInstanceRoot, standaloneRoot } = require('./provision-repo.js');
|
|
37
|
+
const { migratePlan, CORE_MIGRATE_SCRIPT } = require('./upgrade-migrate.js');
|
|
38
|
+
const { isValidSlug } = require('../../modules/provisioning/provisioning.js');
|
|
39
|
+
const outcome = require('./upgrade-outcome.js');
|
|
40
|
+
const { DEFAULT_RETRY_DELAY_MS } = require('./intent-retry.js');
|
|
41
|
+
|
|
42
|
+
const REMEDY_REASONS = Object.freeze(['dirty_tree', 'disk_full', 'schema_pending']);
|
|
43
|
+
// The newest dumps a disk-full prune always leaves behind: a week of nightlies. The nightly
|
|
44
|
+
// job's own retention is 30 days; this only ever cuts below that when the disk is full.
|
|
45
|
+
const DUMP_KEEP_FLOOR = 7;
|
|
46
|
+
const REMEDY_EVENT = 'core-upgrade-remedy';
|
|
47
|
+
// A value that reaches a shell string below. Narrower than what Postgres or a path allow, on
|
|
48
|
+
// purpose: anything outside it is refused, never quoted.
|
|
49
|
+
const SHELL_SAFE_RE = /^[A-Za-z0-9._/-]{1,200}$/;
|
|
50
|
+
const DB_NAME_RE = /^[A-Za-z0-9_-]{1,63}$/;
|
|
51
|
+
const ACCOUNT_RE = /^[a-z_][a-z0-9_-]{0,31}$/;
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* The fix for one known snag, as a command to run on the box. PURE.
|
|
55
|
+
* @param {object} inst the provisioning_instances row
|
|
56
|
+
* @param {string} reason the failure's reason code
|
|
57
|
+
* @param {object} [o] { privileged, runAs, instanceDir, selfRoot, backupDir, migrate }
|
|
58
|
+
* @returns {{ reason, cmd, cwd, env, describe } | { refused: string } | null} null: no known fix
|
|
59
|
+
*/
|
|
60
|
+
function planRemedy(inst, reason, o = {}) {
|
|
61
|
+
if (!REMEDY_REASONS.includes(reason)) return null;
|
|
62
|
+
if (!inst || !isValidSlug(inst.slug)) return { refused: 'the project name is not a valid slug' };
|
|
63
|
+
const dir = o.instanceDir || (inst.hosting_shape === 'cloud-host' ? standaloneRoot(inst) : (o.selfRoot || selfInstanceRoot)());
|
|
64
|
+
if (!SHELL_SAFE_RE.test(dir)) return { refused: 'the project folder is not a path this runner will put in a command' };
|
|
65
|
+
const db = dbName(inst);
|
|
66
|
+
if (!DB_NAME_RE.test(db)) return { refused: 'the project database name is not one this runner will put in a command' };
|
|
67
|
+
|
|
68
|
+
if (reason === 'dirty_tree') {
|
|
69
|
+
// TRACKED files only (`add -u`): the snag is the upgrade's own rewrites of tracked files.
|
|
70
|
+
// A new untracked file is not swept into the owner's history on a guess — if that is the
|
|
71
|
+
// dirt, the retry fails the same way and a person looks. Runs as the runner, which owns
|
|
72
|
+
// the checkout (pushUpgradePin's shape), with hooks off and a runner identity so the
|
|
73
|
+
// commit says who made it.
|
|
74
|
+
const g = `git -C ${dir} -c core.hooksPath=/dev/null -c user.name=bongos-runner -c user.email=runner@bongos.invalid`;
|
|
75
|
+
return {
|
|
76
|
+
reason, cwd: null, env: null,
|
|
77
|
+
cmd: `${g} add -u && ${g} commit -m "chore: commit changes left in the working tree (automatic fix before a core move)"`,
|
|
78
|
+
describe: `committed the uncommitted changes in ${dir} so the core move could run`,
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
if (reason === 'disk_full') {
|
|
82
|
+
const backupDir = o.backupDir || CONFIG.backupDir;
|
|
83
|
+
if (!SHELL_SAFE_RE.test(backupDir)) return { refused: 'the backup folder is not a path this runner will put in a command' };
|
|
84
|
+
// EXACTLY db-backup-nightly.sh's name, so this never touches a sibling project's dumps
|
|
85
|
+
// (`test` must not prune `test-nk`'s — task 1003948). The timestamp sorts by name, so
|
|
86
|
+
// `sort | head -n -K` drops all but the newest K. Nothing is deleted when K or fewer exist.
|
|
87
|
+
const D = '[0-9]';
|
|
88
|
+
const glob = `${db}-${D}${D}${D}${D}-${D}${D}-${D}${D}T${D}${D}${D}${D}${D}${D}Z.sql.gz`;
|
|
89
|
+
return {
|
|
90
|
+
reason, cwd: null, env: null,
|
|
91
|
+
cmd: `find ${backupDir} -maxdepth 1 -type f -name '${glob}' | sort | head -n -${DUMP_KEEP_FLOOR} | xargs -r rm -f --`,
|
|
92
|
+
describe: `pruned ${db}'s old database backups in ${backupDir}, keeping the newest ${DUMP_KEEP_FLOOR}`,
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
// schema_pending — the project's OWN migrate command (a declared script wins, else the
|
|
96
|
+
// core's migrate.sh: upgrade-migrate.js), aimed at its OWN database. migrate.sh defaults to
|
|
97
|
+
// production when PGDATABASE is unset, so it is always set, never inherited.
|
|
98
|
+
const plan = (o.migrate || migratePlan)(dir);
|
|
99
|
+
const run = plan.via === 'npm-script' ? 'npm run migrate' : `bash ${CORE_MIGRATE_SCRIPT}`;
|
|
100
|
+
const initCwd = plan.via === 'npm-script' ? '' : `INIT_CWD=${dir} `;
|
|
101
|
+
if (o.privileged) {
|
|
102
|
+
if (!ACCOUNT_RE.test(String(o.runAs || ''))) return { refused: 'no account to run the migration as' };
|
|
103
|
+
return { reason, cwd: dir, env: null, cmd: `sudo -u ${o.runAs} env PGDATABASE=${db} ${initCwd}${run}`, describe: `migrated ${db} to match the code it is running` };
|
|
104
|
+
}
|
|
105
|
+
return { reason, cwd: dir, env: { PGDATABASE: db, ...(initCwd ? { INIT_CWD: dir } : {}) }, cmd: run, describe: `migrated ${db} to match the code it is running` };
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// The account the project runs as, probed the way the move itself probes (task 1004128).
|
|
109
|
+
function runAsFor(inst, deps) {
|
|
110
|
+
try { return require('./provision-core-upgrade.js').resolveRunAs(inst, deps).runAs; } catch { return null; }
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* After a core move failed: apply the known fix, record it, and re-pend the move once.
|
|
115
|
+
*
|
|
116
|
+
* Returns `{ repended: true }` when the move will be tried again, otherwise
|
|
117
|
+
* `{ repended: false, failure, error }` — the failure the drain loop should RETIRE with,
|
|
118
|
+
* which is the original one when there is no fix for it, and `unknown` when a fix was
|
|
119
|
+
* already spent on this reason or the fix itself failed.
|
|
120
|
+
*/
|
|
121
|
+
async function remedyAfterFailure({ intent, inst, result, deps }) {
|
|
122
|
+
const failure = result && result.failure;
|
|
123
|
+
const reason = failure && failure.reason;
|
|
124
|
+
const none = { repended: false, failure, error: result && result.error };
|
|
125
|
+
if (!intent || intent.action !== 'core-upgrade' || !failure || failure.class !== 'known_wedge') return none;
|
|
126
|
+
const { db, provisioning, log = () => {}, apply } = deps;
|
|
127
|
+
if (!apply) return none;
|
|
128
|
+
if (intent.remedied_reason === reason) {
|
|
129
|
+
return {
|
|
130
|
+
repended: false, failure: outcome.classifyFailure('remedy_did_not_hold'),
|
|
131
|
+
error: `${result.error || 'core move failed'} — the automatic fix for this (${reason}) was already tried once and did not clear it, so a person needs to look`,
|
|
132
|
+
};
|
|
133
|
+
}
|
|
134
|
+
const boxExec = inst.hosting_shape === 'cloud-host' ? (deps.controlExec || deps.exec) : deps.exec;
|
|
135
|
+
const plan = planRemedy(inst, reason, { privileged: deps.privileged, runAs: reason === 'schema_pending' && deps.privileged ? runAsFor(inst, deps) : null, selfRoot: deps.selfInstanceRoot, migrate: deps.migratePlan });
|
|
136
|
+
if (!plan) return none;
|
|
137
|
+
const giveUp = async (why) => {
|
|
138
|
+
await provisioning.recordEvent(db, { instanceId: inst.id, ownerBuilderId: inst.owner_builder_id, event: REMEDY_EVENT, detail: `could not fix ${reason} automatically: ${why}`, actor: 'runner' }).catch(() => {});
|
|
139
|
+
return { repended: false, failure: outcome.classifyFailure('remedy_failed'), error: `${result.error || 'core move failed'} — the automatic fix for ${reason} did not work (${why}), so a person needs to look` };
|
|
140
|
+
};
|
|
141
|
+
if (plan.refused) return giveUp(plan.refused);
|
|
142
|
+
log(` [remedy] ${reason}: ${plan.cmd}`);
|
|
143
|
+
const r = boxExec(plan.cmd, { allowFail: true, ...(plan.cwd ? { cwd: plan.cwd } : {}), ...(plan.env ? { env: plan.env } : {}) });
|
|
144
|
+
if (!r || r.softFailed === true || r.ok === false) {
|
|
145
|
+
const said = String((r && (r.stderr || r.stdout || r.error)) || 'the command failed').trim().split('\n').pop().slice(0, 160);
|
|
146
|
+
return giveUp(said);
|
|
147
|
+
}
|
|
148
|
+
await provisioning.recordEvent(db, { instanceId: inst.id, ownerBuilderId: inst.owner_builder_id, event: REMEDY_EVENT, detail: `${plan.describe}; trying the move to ${intent.target_version} once more`, actor: 'runner' }).catch(() => {});
|
|
149
|
+
await db.query(
|
|
150
|
+
`UPDATE provisioning_intents SET state='pending', last_error=$2, remedied_reason=$3, failure_class=$4, failure_reason=$5,
|
|
151
|
+
not_before = now() + ($6 * interval '1 millisecond'), updated_at=now() WHERE id=$1`,
|
|
152
|
+
[intent.id, result.error || 'core move failed', reason, failure.class, failure.reason, DEFAULT_RETRY_DELAY_MS]);
|
|
153
|
+
log(` ✓ fixed ${reason} on ${inst.slug}; the move to ${intent.target_version} will be tried once more`);
|
|
154
|
+
return { repended: true };
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
module.exports = { REMEDY_REASONS, DUMP_KEEP_FLOOR, REMEDY_EVENT, planRemedy, remedyAfterFailure };
|
package/src/module-api.js
CHANGED
|
@@ -75,7 +75,7 @@ const { responsibilityFor, ROLE_RESPONSIBILITIES } = require('./role-responsibil
|
|
|
75
75
|
// MAJOR (see allowBoxScope below): passes the request through untouched.
|
|
76
76
|
function deprecatedNoopMiddleware(_req, _res, next) { next(); }
|
|
77
77
|
|
|
78
|
-
const CORE_VERSION = '1.20.
|
|
78
|
+
const CORE_VERSION = '1.20.32'; // CI auto-patch carrier (ADR 0161); changelog: docs/module-api-changelog.md
|
|
79
79
|
|
|
80
80
|
// A namespaced logger so a module's log lines are attributable + consistent.
|
|
81
81
|
// Usage: const log = api.logger('discord'); log.info('mounted');
|
|
@@ -431,7 +431,7 @@ await t('schema_pending passes through as a count, and UNKNOWN stays null rather
|
|
|
431
431
|
await t('an open intent is surfaced with its target, so the page can say what is already running', () => {
|
|
432
432
|
const p = CU.previewFromSnapshot(previewInst, { served_version: '1.19.758', available: [], recommended: null },
|
|
433
433
|
{ action: 'core-upgrade', state: 'running', target_version: '1.19.762' });
|
|
434
|
-
assert.deepEqual(p.open_intent, { action: 'core-upgrade', state: 'running', target_version: '1.19.762' });
|
|
434
|
+
assert.deepEqual(p.open_intent, { action: 'core-upgrade', state: 'running', target_version: '1.19.762', not_before: null, last_error: null });
|
|
435
435
|
});
|
|
436
436
|
|
|
437
437
|
console.log('\nthe deploy page is gated on the same atom as its routes:');
|
|
@@ -973,5 +973,39 @@ await t('a failed move carries its class to the page; a move that worked carries
|
|
|
973
973
|
assert.equal(CU.previewFromSnapshot(ACTIVE, null, null, old).last_move.failure, null, 'a row from before the columns');
|
|
974
974
|
});
|
|
975
975
|
|
|
976
|
+
console.log('\na move waiting out its retry (task 1004447):');
|
|
977
|
+
|
|
978
|
+
const waiting = (over = {}) => ({ id: 5, action: 'core-upgrade', state: 'pending', target_version: '1.20.1', not_before: new Date(Date.now() + 3600e3).toISOString(), ...over });
|
|
979
|
+
const enqueueAgainst = async (open, to = '1.20.1') => {
|
|
980
|
+
const writes = [];
|
|
981
|
+
const provisioning = { enqueueIntent: async () => ({ intent: open, created: false }) };
|
|
982
|
+
const db = { query: async (sql, params) => { writes.push({ sql, params }); return { rows: [] }; } };
|
|
983
|
+
return { r: await CU.enqueueCoreUpgrade(provisioning, db, 7, to, 'api:archon:42'), writes };
|
|
984
|
+
};
|
|
985
|
+
|
|
986
|
+
await t('pressing Move for the SAME waiting move brings its retry forward instead of refusing', async () => {
|
|
987
|
+
const { r, writes } = await enqueueAgainst(waiting());
|
|
988
|
+
assert.equal(r.retryNow, true);
|
|
989
|
+
assert.ok(!r.conflict);
|
|
990
|
+
assert.match(writes[0].sql, /SET not_before = NULL/);
|
|
991
|
+
assert.deepEqual(writes[0].params, [5]);
|
|
992
|
+
});
|
|
993
|
+
|
|
994
|
+
await t('a DIFFERENT target, a running move, or one not waiting is still a conflict', async () => {
|
|
995
|
+
for (const open of [waiting({ target_version: '1.20.2' }), waiting({ state: 'running' }), waiting({ not_before: null }), waiting({ action: 'restart' })]) {
|
|
996
|
+
const { r, writes } = await enqueueAgainst(open);
|
|
997
|
+
assert.equal(r.conflict, 'open_intent');
|
|
998
|
+
assert.equal(writes.length, 0);
|
|
999
|
+
}
|
|
1000
|
+
});
|
|
1001
|
+
|
|
1002
|
+
await t('the preview says when a waiting move will try again, and why it is waiting', () => {
|
|
1003
|
+
const p = CU.previewFromSnapshot(ACTIVE, null, waiting({ last_error: 'core 1.20.1 is not published yet' }), null);
|
|
1004
|
+
assert.ok(p.open_intent.not_before);
|
|
1005
|
+
assert.match(p.open_intent.last_error, /not published yet/);
|
|
1006
|
+
assert.equal(CU.previewFromSnapshot(ACTIVE, null, waiting({ not_before: null, last_error: 'x' }), null).open_intent.last_error, null,
|
|
1007
|
+
'an ordinary pending intent shows no error');
|
|
1008
|
+
});
|
|
1009
|
+
|
|
976
1010
|
console.log(`\ncore_upgrade_door: ${passed} passed, ${failed} failed`);
|
|
977
1011
|
if (failed) process.exit(1);
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
// tests/intent_retry.mjs — a failed intent waits, or is retired, by what kind of failure it
|
|
2
|
+
// was (task 1004447, goal 1000090 deploy self-heal).
|
|
3
|
+
//
|
|
4
|
+
// WHAT IS AT RISK. One runner tick used to run the same failing core move up to three times
|
|
5
|
+
// back to back, restarting the service each time, then give up. These cases pin the policy
|
|
6
|
+
// that replaced it; tests/provision.mjs drives the drain loop with it.
|
|
7
|
+
//
|
|
8
|
+
// Run: node tests/intent_retry.mjs
|
|
9
|
+
|
|
10
|
+
import { strict as assert } from 'node:assert';
|
|
11
|
+
import { retryDecision, maxAttemptsFor, TRANSIENT_BACKOFF_MS, DEFAULT_RETRY_DELAY_MS } from '../scripts/gds/intent-retry.js';
|
|
12
|
+
|
|
13
|
+
let passed = 0;
|
|
14
|
+
let failed = 0;
|
|
15
|
+
const t = async (name, fn) => {
|
|
16
|
+
try { await fn(); passed += 1; console.log(` PASS ${name}`); }
|
|
17
|
+
catch (e) { failed += 1; console.log(` FAIL ${name}\n ${e.message}`); }
|
|
18
|
+
};
|
|
19
|
+
const MIN = 60 * 1000;
|
|
20
|
+
const transient = { class: 'transient', reason: 'not_published' };
|
|
21
|
+
|
|
22
|
+
await t('a transient failure waits about 15 minutes, then 1 hour, then 6 hours', () => {
|
|
23
|
+
assert.deepEqual([...TRANSIENT_BACKOFF_MS], [15 * MIN, 60 * MIN, 360 * MIN]);
|
|
24
|
+
for (const [attempts, ms] of [[1, 15 * MIN], [2, 60 * MIN], [3, 360 * MIN]]) {
|
|
25
|
+
assert.deepEqual(retryDecision({ attempts, failure: transient, maxAttempts: 3 }), { action: 'retry', delayMs: ms });
|
|
26
|
+
}
|
|
27
|
+
});
|
|
28
|
+
|
|
29
|
+
await t('and is retired after its last scheduled try', () => {
|
|
30
|
+
assert.deepEqual(retryDecision({ attempts: 4, failure: transient, maxAttempts: 3 }), { action: 'retire' });
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
await t('every NON-transient class is retired at once — retrying cannot change the answer', () => {
|
|
34
|
+
for (const c of ['known_wedge', 'needs_decision', 'unknown']) {
|
|
35
|
+
assert.deepEqual(retryDecision({ attempts: 1, failure: { class: c, reason: 'x' }, maxAttempts: 3 }), { action: 'retire' }, c);
|
|
36
|
+
}
|
|
37
|
+
});
|
|
38
|
+
|
|
39
|
+
await t('terminal always retires, even a transient class', () => {
|
|
40
|
+
assert.deepEqual(retryDecision({ attempts: 1, failure: transient, terminal: true, maxAttempts: 3 }), { action: 'retire' });
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
await t('an UNCLASSIFIED failure keeps its budget, but never retries in the same tick', () => {
|
|
44
|
+
assert.deepEqual(retryDecision({ attempts: 1, maxAttempts: 3 }), { action: 'retry', delayMs: DEFAULT_RETRY_DELAY_MS });
|
|
45
|
+
assert.ok(DEFAULT_RETRY_DELAY_MS >= MIN, 'at least a minute: the runner ticks about every two');
|
|
46
|
+
assert.deepEqual(retryDecision({ attempts: 3, maxAttempts: 3 }), { action: 'retire' });
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
await t('the pre-run guard gives a transient row room for its last scheduled try', () => {
|
|
50
|
+
assert.equal(maxAttemptsFor({ failure_class: 'transient' }, 3), 4);
|
|
51
|
+
assert.equal(maxAttemptsFor({ failure_class: 'unknown' }, 3), 3);
|
|
52
|
+
assert.equal(maxAttemptsFor({}, 3), 3);
|
|
53
|
+
assert.equal(maxAttemptsFor({ failure_class: 'transient' }, 9), 9, 'never lowers a configured budget');
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
console.log(`\nintent_retry: ${passed} passed, ${failed} failed`);
|
|
57
|
+
if (failed) process.exit(1);
|
package/tests/provision.mjs
CHANGED
|
@@ -2564,38 +2564,63 @@ await ta('F19: a NON-final failure re-pends the intent and writes nothing on the
|
|
|
2564
2564
|
assert.equal(statusWrites.length, 0, 'no instance write until the intent actually retires');
|
|
2565
2565
|
});
|
|
2566
2566
|
|
|
2567
|
-
// A failed core move stores WHAT KIND of failure beside its sentence (task 1004446)
|
|
2568
|
-
|
|
2569
|
-
|
|
2567
|
+
// A failed core move stores WHAT KIND of failure beside its sentence (task 1004446), and
|
|
2568
|
+
// waits or retires by that kind (task 1004447). The queue here is STATEFUL — a fake
|
|
2569
|
+
// claimNextIntent that honours not_before the way the real query does — so "one tick runs a
|
|
2570
|
+
// failing move at most once" is measured, not assumed.
|
|
2571
|
+
const queueRun = async ({ attempts = 0, transient = false } = {}) => {
|
|
2572
|
+
const row = { id: 11, instance_id: 18, action: 'core-upgrade', target_version: '1.20.1', attempts, state: 'pending', not_before: null, failure_class: attempts && transient ? 'transient' : null }; // a re-tried row carries the class its last attempt wrote
|
|
2573
|
+
let runs = 0; const resolved = []; const classWrites = []; const repends = [];
|
|
2570
2574
|
const deps = {
|
|
2571
2575
|
apply: true, log: () => {},
|
|
2572
|
-
db: { query: async (sql, params) => {
|
|
2576
|
+
db: { query: async (sql, params) => {
|
|
2577
|
+
if (/SET failure_class/.test(sql)) { classWrites.push(params); row.failure_class = params[1]; }
|
|
2578
|
+
if (/SET state='pending'/.test(sql)) { repends.push(params); row.state = 'pending'; row.not_before = new Date(Date.now() + params[2]); }
|
|
2579
|
+
return { rows: [] };
|
|
2580
|
+
} },
|
|
2573
2581
|
provisioning: {
|
|
2574
|
-
claimNextIntent: async () =>
|
|
2575
|
-
|
|
2576
|
-
|
|
2577
|
-
|
|
2578
|
-
|
|
2582
|
+
claimNextIntent: async () => {
|
|
2583
|
+
if (row.state !== 'pending' || (row.not_before && row.not_before > new Date())) return null;
|
|
2584
|
+
row.state = 'running'; row.attempts += 1; runs += 1; return { ...row };
|
|
2585
|
+
},
|
|
2586
|
+
// 'dedicated' is a shape the runner cannot move (needs_decision); the control plane with
|
|
2587
|
+
// an unreadable registry is the transient case. Neither needs an exec.
|
|
2588
|
+
getInstanceById: async () => (transient ? { ...activeStandalone, hosting_shape: 'control-plane' } : { ...activeStandalone, hosting_shape: 'dedicated' }),
|
|
2589
|
+
resolveIntent: async (_db, id, state) => { resolved.push({ id, state }); row.state = state; },
|
|
2579
2590
|
setInstanceStatus: async () => {}, recordEvent: async () => {},
|
|
2580
2591
|
claimNextOAuthManifest: async () => null, resolveOAuthManifest: async () => {},
|
|
2581
2592
|
},
|
|
2582
2593
|
exec: () => ({ ok: true }), writeFile: () => ({ ok: true }),
|
|
2594
|
+
readInstalledCoreVersion: () => '1.19.1081', selfInstanceRoot: () => '/srv/cb',
|
|
2595
|
+
updateChannel: { DEFAULT_CHANNEL: 'patch', normalizeChannel: (c) => c || 'patch', channelAllows: () => true, listAvailableVersions: () => ({ ok: false, versions: [] }) },
|
|
2583
2596
|
checkDnsTokenScope: async () => ({ reachable: true, zone: 'z' }), checkCaddyWiring: () => null,
|
|
2584
2597
|
};
|
|
2585
2598
|
await P.cmdRunIntents(deps);
|
|
2586
|
-
return { resolved, classWrites };
|
|
2599
|
+
return { runs, resolved, classWrites, repends, row };
|
|
2587
2600
|
};
|
|
2588
2601
|
|
|
2589
|
-
await ta('a
|
|
2590
|
-
const { resolved, classWrites } = await
|
|
2591
|
-
assert.equal(
|
|
2602
|
+
await ta('a needs-decision core move is RETIRED at once with its class — retrying cannot change it', async () => {
|
|
2603
|
+
const { runs, resolved, classWrites } = await queueRun();
|
|
2604
|
+
assert.equal(runs, 1);
|
|
2605
|
+
assert.equal(resolved[0].state, 'error');
|
|
2592
2606
|
assert.deepEqual(classWrites, [[11, 'needs_decision', 'shape_not_automated']]);
|
|
2593
2607
|
});
|
|
2594
2608
|
|
|
2595
|
-
await ta('a
|
|
2596
|
-
const { resolved, classWrites } = await
|
|
2609
|
+
await ta('a TRANSIENT core move runs ONCE per tick and waits ~15 minutes before its next try', async () => {
|
|
2610
|
+
const { runs, resolved, repends, classWrites } = await queueRun({ transient: true });
|
|
2611
|
+
assert.equal(runs, 1, 'the same failing move is not re-run inside one tick');
|
|
2612
|
+
assert.equal(resolved.length, 0, 'not retired — it waits');
|
|
2613
|
+
assert.equal(repends.length, 1);
|
|
2614
|
+
assert.equal(repends[0][2], 15 * 60 * 1000, 'not_before = now + 15 minutes');
|
|
2615
|
+
assert.deepEqual(classWrites, [[11, 'transient', 'registry_unreadable']]);
|
|
2616
|
+
});
|
|
2617
|
+
|
|
2618
|
+
await ta('a transient move on its LAST scheduled try (the 6-hour one) is retired after it', async () => {
|
|
2619
|
+
const { runs, resolved } = await queueRun({ transient: true, attempts: 3 });
|
|
2620
|
+
// attempt 4 RUNS (the pre-run guard allows it for a transient row) and is then retired.
|
|
2621
|
+
assert.equal(runs, 1, 'the 6-hour try actually ran — not retired unrun by the pre-run guard');
|
|
2622
|
+
assert.equal(resolved.length, 1);
|
|
2597
2623
|
assert.equal(resolved[0].state, 'error');
|
|
2598
|
-
assert.deepEqual(classWrites, [[11, 'needs_decision', 'shape_not_automated']]);
|
|
2599
2624
|
});
|
|
2600
2625
|
|
|
2601
2626
|
console.log('\nreview-panel hardening (task 1002720 — the loopback/latch/birth-note seams):');
|
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
// tests/wedge_remedy.mjs — a core move that failed on a KNOWN snag fixes the snag and tries
|
|
2
|
+
// once more, and never twice (task 1004448, goal 1000090 deploy self-heal).
|
|
3
|
+
//
|
|
4
|
+
// WHAT IS AT RISK. Each fix runs a command on a live project's box: a commit in the owner's
|
|
5
|
+
// checkout, a delete among database backups, a migration against a database. The cases pin
|
|
6
|
+
// WHAT each command is (and what it refuses to build), that it runs at most once per
|
|
7
|
+
// failure, and that a fix which did not hold reaches a person instead of looping. The
|
|
8
|
+
// disk-full prune is also run for real against a temp folder, because a `find` pattern is
|
|
9
|
+
// the kind of thing that reads right and deletes a sibling project's dumps.
|
|
10
|
+
//
|
|
11
|
+
// Run: node tests/wedge_remedy.mjs
|
|
12
|
+
|
|
13
|
+
import { strict as assert } from 'node:assert';
|
|
14
|
+
import { createRequire } from 'node:module';
|
|
15
|
+
import { spawnSync } from 'node:child_process';
|
|
16
|
+
import fs from 'node:fs';
|
|
17
|
+
import os from 'node:os';
|
|
18
|
+
import path from 'node:path';
|
|
19
|
+
|
|
20
|
+
process.env.PROVISION_APP_USER = process.env.PROVISION_APP_USER || 'lars'; // appUser() refuses a Windows account name
|
|
21
|
+
const require = createRequire(import.meta.url);
|
|
22
|
+
const W = require('../scripts/gds/wedge-remedy.js');
|
|
23
|
+
const R = require('../scripts/gds/provision-core-upgrade.js');
|
|
24
|
+
const P = require('../scripts/gds/provision.js');
|
|
25
|
+
const { REASON_CLASS } = require('../scripts/gds/upgrade-outcome.js');
|
|
26
|
+
const { DEFAULT_RETRY_DELAY_MS } = require('../scripts/gds/intent-retry.js');
|
|
27
|
+
|
|
28
|
+
let passed = 0;
|
|
29
|
+
let failed = 0;
|
|
30
|
+
const t = async (name, fn) => {
|
|
31
|
+
try { await fn(); passed += 1; console.log(` PASS ${name}`); }
|
|
32
|
+
catch (e) { failed += 1; console.log(` FAIL ${name}\n ${e.message}`); }
|
|
33
|
+
};
|
|
34
|
+
|
|
35
|
+
const inst = (over = {}) => ({ id: 7, slug: 'charter', db_name: 'charter', port: 3010, status: 'active', hosting_shape: 'co-tenant', owner_builder_id: 1, ...over });
|
|
36
|
+
const intent = (over = {}) => ({ id: 99, action: 'core-upgrade', target_version: '1.20.31', attempts: 1, remedied_reason: null, ...over });
|
|
37
|
+
const wedge = (reason) => ({ ok: false, error: `core upgrade to 1.20.31 was refused (${reason})`, failure: { class: 'known_wedge', reason } });
|
|
38
|
+
const selfRoot = () => '/srv/charter';
|
|
39
|
+
|
|
40
|
+
function harness({ execFails = false } = {}) {
|
|
41
|
+
const ran = []; const events = []; const queries = [];
|
|
42
|
+
const deps = {
|
|
43
|
+
apply: true, privileged: false, log: () => {}, selfInstanceRoot: selfRoot,
|
|
44
|
+
exec: (cmd, opts) => { ran.push({ cmd, ...opts }); return execFails ? { ok: false, softFailed: true, stderr: 'fatal: nope' } : { ok: true, stdout: '' }; },
|
|
45
|
+
db: { query: async (sql, params) => { queries.push({ sql, params }); return { rows: [] }; } },
|
|
46
|
+
provisioning: { recordEvent: async (_db, e) => { events.push(e); } },
|
|
47
|
+
migratePlan: () => ({ via: 'npm-script' }),
|
|
48
|
+
};
|
|
49
|
+
return { deps, ran, events, queries };
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
console.log('\nthe fix for each known snag (planRemedy):');
|
|
53
|
+
|
|
54
|
+
await t('every reason with a fix is a known_wedge in the classifier — and back', () => {
|
|
55
|
+
assert.deepEqual([...W.REMEDY_REASONS].sort(), ['dirty_tree', 'disk_full', 'schema_pending']);
|
|
56
|
+
for (const r of W.REMEDY_REASONS) assert.equal(REASON_CLASS[r], 'known_wedge', r);
|
|
57
|
+
for (const [r, c] of Object.entries(REASON_CLASS)) if (c === 'known_wedge') assert.ok(W.REMEDY_REASONS.includes(r), `${r} is known_wedge but has no fix — it would be retired with nothing done`);
|
|
58
|
+
assert.equal(REASON_CLASS.remedy_did_not_hold, 'unknown', 'a fix that did not hold reaches a person');
|
|
59
|
+
assert.equal(REASON_CLASS.remedy_failed, 'unknown');
|
|
60
|
+
});
|
|
61
|
+
|
|
62
|
+
await t('dirty tree: commits TRACKED changes inside the project, hooks off, as the runner', () => {
|
|
63
|
+
const p = W.planRemedy(inst(), 'dirty_tree', { selfRoot });
|
|
64
|
+
assert.match(p.cmd, /^git -C \/srv\/charter -c core\.hooksPath=\/dev\/null .* add -u && git -C \/srv\/charter .* commit -m /);
|
|
65
|
+
assert.doesNotMatch(p.cmd, /add -A|add \./, 'an untracked file is never swept into the owner\'s history on a guess');
|
|
66
|
+
assert.doesNotMatch(p.cmd, /sudo/, 'the runner owns the checkout (pushUpgradePin\'s shape)');
|
|
67
|
+
});
|
|
68
|
+
|
|
69
|
+
await t('a cloud-host project is fixed in ITS folder, not the control plane\'s', () => {
|
|
70
|
+
const p = W.planRemedy(inst({ hosting_shape: 'cloud-host' }), 'dirty_tree', { selfRoot });
|
|
71
|
+
assert.match(p.cmd, /git -C \/\S+\/charter /);
|
|
72
|
+
assert.doesNotMatch(p.cmd, /\/srv\/charter/);
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
await t('disk full: prunes only THIS database\'s dumps, keeping the newest week', () => {
|
|
76
|
+
const p = W.planRemedy(inst(), 'disk_full', { selfRoot, backupDir: '/var/backups/gds' });
|
|
77
|
+
assert.match(p.cmd, /^find \/var\/backups\/gds -maxdepth 1 -type f -name 'charter-\[0-9\]/);
|
|
78
|
+
assert.match(p.cmd, new RegExp(`head -n -${W.DUMP_KEEP_FLOOR} \\| xargs -r rm -f --$`));
|
|
79
|
+
assert.equal(W.DUMP_KEEP_FLOOR, 7);
|
|
80
|
+
});
|
|
81
|
+
|
|
82
|
+
await t('schema pending: migrates the project\'s OWN database, PGDATABASE always set', () => {
|
|
83
|
+
const plain = W.planRemedy(inst(), 'schema_pending', { selfRoot, migrate: () => ({ via: 'npm-script' }) });
|
|
84
|
+
assert.equal(plain.cmd, 'npm run migrate');
|
|
85
|
+
assert.equal(plain.cwd, '/srv/charter');
|
|
86
|
+
assert.deepEqual(plain.env, { PGDATABASE: 'charter' });
|
|
87
|
+
const adopted = W.planRemedy(inst(), 'schema_pending', { selfRoot, migrate: () => ({ via: 'core-script' }) });
|
|
88
|
+
assert.equal(adopted.cmd, 'bash node_modules/@bongos/core/scripts/migrate.sh');
|
|
89
|
+
assert.deepEqual(adopted.env, { PGDATABASE: 'charter', INIT_CWD: '/srv/charter' });
|
|
90
|
+
const priv = W.planRemedy(inst(), 'schema_pending', { selfRoot, privileged: true, runAs: 'bongos-charter', migrate: () => ({ via: 'npm-script' }) });
|
|
91
|
+
assert.equal(priv.cmd, 'sudo -u bongos-charter env PGDATABASE=charter npm run migrate', 'sudo resets env, so PGDATABASE rides the command');
|
|
92
|
+
assert.ok(W.planRemedy(inst(), 'schema_pending', { selfRoot, privileged: true, runAs: null }).refused);
|
|
93
|
+
});
|
|
94
|
+
|
|
95
|
+
await t('no command is built from a name that is not shell-safe', () => {
|
|
96
|
+
assert.ok(W.planRemedy(inst({ slug: 'a;rm -rf' }), 'dirty_tree', { selfRoot }).refused);
|
|
97
|
+
assert.ok(W.planRemedy(inst({ db_name: 'x; drop' }), 'disk_full', { selfRoot }).refused);
|
|
98
|
+
assert.ok(W.planRemedy(inst(), 'dirty_tree', { selfRoot: () => '/srv/a b' }).refused);
|
|
99
|
+
assert.ok(W.planRemedy(inst(), 'disk_full', { selfRoot, backupDir: '/tmp/$(x)' }).refused);
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
await t('a reason with no fix gets no command', () => {
|
|
103
|
+
for (const r of ['artist_gate', 'downgrade', 'health', 'not_published', 'nope']) assert.equal(W.planRemedy(inst(), r, { selfRoot }), null, r);
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
console.log('\nthe prune, run for real:');
|
|
107
|
+
|
|
108
|
+
const bash = spawnSync('bash', ['-c', 'echo ok'], { encoding: 'utf8' });
|
|
109
|
+
await t('keeps the newest 7 of this project\'s dumps and never touches a sibling\'s', () => {
|
|
110
|
+
if (bash.status !== 0) { console.log(' (skipped: no bash here)'); return; }
|
|
111
|
+
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'wedge-prune-'));
|
|
112
|
+
try {
|
|
113
|
+
const mine = []; for (let d = 1; d <= 10; d++) mine.push(`charter-2026-09-${String(d).padStart(2, '0')}T030000Z.sql.gz`);
|
|
114
|
+
const others = ['charter-nk-2026-09-01T030000Z.sql.gz', 'test-2026-09-01T030000Z.sql.gz', 'charter-notes.txt'];
|
|
115
|
+
for (const f of [...mine, ...others]) fs.writeFileSync(path.join(dir, f), 'x');
|
|
116
|
+
const p = W.planRemedy(inst(), 'disk_full', { selfRoot, backupDir: '.' }); // run IN the folder: a temp path can hold a `~`
|
|
117
|
+
const r = spawnSync('bash', ['-c', p.cmd], { encoding: 'utf8', cwd: dir });
|
|
118
|
+
assert.equal(r.status, 0, r.stderr);
|
|
119
|
+
const left = fs.readdirSync(dir).sort();
|
|
120
|
+
assert.deepEqual(left, [...others, ...mine.slice(3)].sort());
|
|
121
|
+
// Nothing is deleted when the floor is not exceeded.
|
|
122
|
+
const again = spawnSync('bash', ['-c', p.cmd], { encoding: 'utf8', cwd: dir });
|
|
123
|
+
assert.equal(again.status, 0);
|
|
124
|
+
assert.equal(fs.readdirSync(dir).length, left.length);
|
|
125
|
+
} finally { fs.rmSync(dir, { recursive: true, force: true }); }
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
console.log('\nonce per failure (remedyAfterFailure):');
|
|
129
|
+
|
|
130
|
+
for (const reason of W.REMEDY_REASONS) {
|
|
131
|
+
await t(`${reason}: runs the fix, records an event, re-pends the move with the reason stored`, async () => {
|
|
132
|
+
const h = harness();
|
|
133
|
+
const r = await W.remedyAfterFailure({ intent: intent(), inst: inst(), result: wedge(reason), deps: h.deps });
|
|
134
|
+
assert.deepEqual(r, { repended: true });
|
|
135
|
+
assert.equal(h.ran.length, 1);
|
|
136
|
+
assert.equal(h.events.length, 1);
|
|
137
|
+
assert.equal(h.events[0].event, W.REMEDY_EVENT);
|
|
138
|
+
assert.match(h.events[0].detail, /trying the move to 1\.20\.31 once more/);
|
|
139
|
+
const up = h.queries.find((q) => /SET state='pending'/.test(q.sql));
|
|
140
|
+
assert.ok(up, 're-pended');
|
|
141
|
+
assert.match(up.sql, /remedied_reason=\$3/);
|
|
142
|
+
assert.deepEqual(up.params, [99, wedge(reason).error, reason, 'known_wedge', reason, DEFAULT_RETRY_DELAY_MS]);
|
|
143
|
+
});
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
await t('the same reason again after a fix: no second fix — retired as unknown for a person', async () => {
|
|
147
|
+
const h = harness();
|
|
148
|
+
const r = await W.remedyAfterFailure({ intent: intent({ remedied_reason: 'dirty_tree' }), inst: inst(), result: wedge('dirty_tree'), deps: h.deps });
|
|
149
|
+
assert.equal(r.repended, false);
|
|
150
|
+
assert.deepEqual(r.failure, { class: 'unknown', reason: 'remedy_did_not_hold' });
|
|
151
|
+
assert.match(r.error, /already tried once/);
|
|
152
|
+
assert.equal(h.ran.length, 0, 'nothing ran');
|
|
153
|
+
assert.equal(h.queries.length, 0);
|
|
154
|
+
});
|
|
155
|
+
|
|
156
|
+
await t('a DIFFERENT known snag on the retry gets its own one fix', async () => {
|
|
157
|
+
const h = harness();
|
|
158
|
+
const r = await W.remedyAfterFailure({ intent: intent({ remedied_reason: 'dirty_tree' }), inst: inst(), result: wedge('disk_full'), deps: h.deps });
|
|
159
|
+
assert.equal(r.repended, true);
|
|
160
|
+
});
|
|
161
|
+
|
|
162
|
+
await t('a fix that fails is recorded and retired as unknown — never re-pended', async () => {
|
|
163
|
+
const h = harness({ execFails: true });
|
|
164
|
+
const r = await W.remedyAfterFailure({ intent: intent(), inst: inst(), result: wedge('dirty_tree'), deps: h.deps });
|
|
165
|
+
assert.equal(r.repended, false);
|
|
166
|
+
assert.deepEqual(r.failure, { class: 'unknown', reason: 'remedy_failed' });
|
|
167
|
+
assert.match(r.error, /did not work \(fatal: nope\)/);
|
|
168
|
+
assert.equal(h.events.length, 1);
|
|
169
|
+
assert.match(h.events[0].detail, /could not fix dirty_tree/);
|
|
170
|
+
assert.equal(h.queries.length, 0);
|
|
171
|
+
});
|
|
172
|
+
|
|
173
|
+
await t('leaves everything else alone: other classes, other actions, and a dry run', async () => {
|
|
174
|
+
for (const [i, res] of [[intent(), { ok: false, error: 'x', failure: { class: 'transient', reason: 'not_published' } }],
|
|
175
|
+
[intent(), { ok: false, error: 'x', failure: { class: 'needs_decision', reason: 'artist_gate' } }],
|
|
176
|
+
[intent(), { ok: false, error: 'x' }],
|
|
177
|
+
[intent({ action: 'restart' }), wedge('dirty_tree')]]) {
|
|
178
|
+
const h = harness();
|
|
179
|
+
const r = await W.remedyAfterFailure({ intent: i, inst: inst(), result: res, deps: h.deps });
|
|
180
|
+
assert.equal(r.repended, false);
|
|
181
|
+
assert.deepEqual(r.failure, res.failure);
|
|
182
|
+
assert.equal(h.ran.length, 0);
|
|
183
|
+
}
|
|
184
|
+
const h = harness(); h.deps.apply = false;
|
|
185
|
+
assert.equal((await W.remedyAfterFailure({ intent: intent(), inst: inst(), result: wedge('dirty_tree'), deps: h.deps })).repended, false);
|
|
186
|
+
assert.equal(h.ran.length, 0);
|
|
187
|
+
});
|
|
188
|
+
|
|
189
|
+
console.log('\nthe move refuses on a database behind its code:');
|
|
190
|
+
|
|
191
|
+
await t('schemaPending > 0 on the box refuses as schema_pending BEFORE the upgrade runs', async () => {
|
|
192
|
+
const ran = [];
|
|
193
|
+
const deps = {
|
|
194
|
+
apply: true, privileged: false, log: () => {}, sleep: async () => {},
|
|
195
|
+
exec: (cmd) => { ran.push(cmd); return { ok: true, stdout: /\/version/.test(cmd) ? JSON.stringify({ coreVersion: '1.20.30', schemaPending: 2 }) : '' }; },
|
|
196
|
+
db: { query: async () => ({ rows: [] }) }, provisioning: { recordEvent: async () => {} },
|
|
197
|
+
readInstalledCoreVersion: () => '1.20.30', selfInstanceRoot: () => '/srv/cb',
|
|
198
|
+
updateChannel: { DEFAULT_CHANNEL: 'patch', normalizeChannel: (c) => c || 'patch', channelAllows: () => true, listAvailableVersions: () => ({ ok: true, versions: ['1.20.30', '1.20.31'] }) },
|
|
199
|
+
};
|
|
200
|
+
const r = await R.coreUpgradeInstance(inst({ hosting_shape: 'control-plane', slug: 'cloudbongos' }), deps, intent());
|
|
201
|
+
assert.equal(r.ok, false);
|
|
202
|
+
assert.deepEqual(r.failure, { class: 'known_wedge', reason: 'schema_pending' });
|
|
203
|
+
assert.match(r.error, /2 migration\(s\) behind/);
|
|
204
|
+
assert.ok(!ran.some((c) => /upgrade\.js/.test(c)), 'the pin never moved');
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
console.log('\nthe drain loop, end to end:');
|
|
208
|
+
|
|
209
|
+
// A stateful fake queue, the shape tests/provision.mjs uses: claimNextIntent honours
|
|
210
|
+
// not_before, so "fixed, then tried once more on a LATER tick" is measured.
|
|
211
|
+
async function drain({ row, reply }) {
|
|
212
|
+
const ran = []; const events = []; const resolved = []; let runs = 0;
|
|
213
|
+
const deps = {
|
|
214
|
+
apply: true, privileged: false, log: () => {}, sleep: async () => {},
|
|
215
|
+
db: { query: async (sql, params) => {
|
|
216
|
+
if (/SET state='pending'/.test(sql)) { row.state = 'pending'; row.not_before = new Date(Date.now() + params[params.length - 1]); if (/remedied_reason/.test(sql)) row.remedied_reason = params[2]; }
|
|
217
|
+
if (/SET failure_class/.test(sql)) { row.failure_class = params[1]; row.failure_reason = params[2]; }
|
|
218
|
+
return { rows: [] };
|
|
219
|
+
} },
|
|
220
|
+
provisioning: {
|
|
221
|
+
claimNextIntent: async () => {
|
|
222
|
+
if (row.state !== 'pending' || (row.not_before && row.not_before > new Date())) return null;
|
|
223
|
+
row.state = 'running'; row.attempts += 1; runs += 1; return { ...row };
|
|
224
|
+
},
|
|
225
|
+
getInstanceById: async () => inst({ hosting_shape: 'control-plane', slug: 'cloudbongos', db_name: 'cloudbongos' }),
|
|
226
|
+
resolveIntent: async (_db, id, state, note) => { resolved.push({ id, state, note }); row.state = state; },
|
|
227
|
+
setInstanceStatus: async () => {}, recordEvent: async (_db, e) => { events.push(e); },
|
|
228
|
+
claimNextOAuthManifest: async () => null, resolveOAuthManifest: async () => {},
|
|
229
|
+
},
|
|
230
|
+
exec: (cmd, opts) => { ran.push(cmd); return reply(cmd, opts); },
|
|
231
|
+
writeFile: () => ({ ok: true }),
|
|
232
|
+
readInstalledCoreVersion: () => '1.20.30', selfInstanceRoot: () => '/srv/cb',
|
|
233
|
+
updateChannel: { DEFAULT_CHANNEL: 'patch', normalizeChannel: (c) => c || 'patch', channelAllows: () => true, listAvailableVersions: () => ({ ok: true, versions: ['1.20.30', '1.20.31'] }) },
|
|
234
|
+
checkDnsTokenScope: async () => ({ reachable: true, zone: 'z' }), checkCaddyWiring: () => null,
|
|
235
|
+
};
|
|
236
|
+
await P.cmdRunIntents(deps);
|
|
237
|
+
return { ran, events, resolved, runs, row };
|
|
238
|
+
}
|
|
239
|
+
const dirtyUpgrade = (cmd) => {
|
|
240
|
+
if (/upgrade\.js/.test(cmd)) { const e = new Error('Command failed'); e.stderr = '✖ working tree not clean — commit/stash first\nBONGOS_UPGRADE_OUTCOME {"stage":"refused","reason":"dirty_tree"}\n'; throw e; }
|
|
241
|
+
return { ok: true, stdout: /\/version/.test(cmd) ? JSON.stringify({ coreVersion: '1.20.30', schemaPending: 0 }) : '' };
|
|
242
|
+
};
|
|
243
|
+
|
|
244
|
+
await t('a dirty tree is committed, the move waits for the next tick, and is NOT retired', async () => {
|
|
245
|
+
const row = { id: 11, instance_id: 7, action: 'core-upgrade', target_version: '1.20.31', attempts: 0, state: 'pending', not_before: null, remedied_reason: null };
|
|
246
|
+
const d = await drain({ row, reply: dirtyUpgrade });
|
|
247
|
+
assert.equal(d.runs, 1, 'the retry is not run in the same tick');
|
|
248
|
+
assert.equal(d.resolved.length, 0, 'not retired');
|
|
249
|
+
assert.equal(d.row.state, 'pending');
|
|
250
|
+
assert.equal(d.row.remedied_reason, 'dirty_tree');
|
|
251
|
+
assert.ok(d.ran.some((c) => /add -u && .* commit/.test(c)), 'the commit ran');
|
|
252
|
+
assert.ok(d.events.some((e) => e.event === W.REMEDY_EVENT));
|
|
253
|
+
});
|
|
254
|
+
|
|
255
|
+
await t('the retry fails dirty AGAIN: retired as unknown/remedy_did_not_hold, no second commit', async () => {
|
|
256
|
+
const row = { id: 11, instance_id: 7, action: 'core-upgrade', target_version: '1.20.31', attempts: 1, state: 'pending', not_before: null, remedied_reason: 'dirty_tree' };
|
|
257
|
+
const d = await drain({ row, reply: dirtyUpgrade });
|
|
258
|
+
assert.equal(d.runs, 1);
|
|
259
|
+
assert.equal(d.resolved.length, 1);
|
|
260
|
+
assert.equal(d.resolved[0].state, 'error');
|
|
261
|
+
assert.match(d.resolved[0].note, /already tried once/);
|
|
262
|
+
assert.equal(d.row.failure_class, 'unknown');
|
|
263
|
+
assert.equal(d.row.failure_reason, 'remedy_did_not_hold');
|
|
264
|
+
assert.ok(!d.ran.some((c) => /commit -m/.test(c)), 'no second fix');
|
|
265
|
+
});
|
|
266
|
+
|
|
267
|
+
console.log(`\nwedge_remedy: ${passed} passed, ${failed} failed`);
|
|
268
|
+
if (failed) process.exit(1);
|