@bongos/core 1.20.31 → 1.20.33
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.bongos-core.json +51 -26
- package/docs/module-api-changelog.md +4 -0
- package/docs/recipes/upgrading-the-core.md +2 -0
- package/modules/autonomy/cadence.js +9 -0
- package/modules/autonomy/digest.js +5 -2
- package/modules/provisioning/migrations/provisioning_032_intent_remedied_reason.sql +19 -0
- package/package-lock.json +2 -2
- package/package.json +1 -1
- package/release-notes.json +16 -0
- package/scripts/gds/autobongos-run.js +155 -44
- package/scripts/gds/claim.js +10 -2
- package/scripts/gds/move-escalation.js +141 -0
- package/scripts/gds/provision-core-upgrade.js +10 -0
- package/scripts/gds/provision.js +5 -1
- package/scripts/gds/upgrade-outcome.js +5 -1
- package/scripts/gds/wedge-remedy.js +157 -0
- package/src/module-api.js +1 -1
- package/tests/autobongos_cadence.mjs +9 -0
- package/tests/autobongos_loop.mjs +142 -1
- package/tests/autonomy_digest.mjs +10 -0
- package/tests/claim_error_surface.mjs +7 -0
- package/tests/cli_exit_no_abort.mjs +7 -2
- package/tests/move_escalation.mjs +191 -0
- package/tests/plain_cards.mjs +2 -2
- package/tests/wedge_remedy.mjs +268 -0
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
//
|
|
3
|
+
// scripts/gds/move-escalation.js — a core move nobody can heal reaches a person: one blocker
|
|
4
|
+
// and one Discord notice per failed move, closed by itself when a later move lands (task
|
|
5
|
+
// 1004449, goal 1000090 deploy self-heal).
|
|
6
|
+
//
|
|
7
|
+
// WHY. After task 1004448 a failed /deploy move ends one of three ways: it waits and retries
|
|
8
|
+
// (transient), it is fixed and retried (known_wedge), or it is retired. A retired
|
|
9
|
+
// needs_decision move is the owner's own rule refusing it, and the deploy page says so. But a
|
|
10
|
+
// move retired as `unknown` — an unlisted failure, a fix that did not hold, and every
|
|
11
|
+
// rollback that FAILED (the project may be down) — only wrote a line on the intent, which
|
|
12
|
+
// nobody reads until they open the page. That is exactly the failure that needs a person.
|
|
13
|
+
//
|
|
14
|
+
// EDGE-TRIGGERED, WITHOUT NEW STATE. The blocker's (source, source_ref) is unique, and
|
|
15
|
+
// source_ref names the INTENT: `core-move:<instance id>:<intent id>`. An intent retires once,
|
|
16
|
+
// so a failed move files one blocker however many retries led up to it, and a re-run of the
|
|
17
|
+
// same retire files nothing. The Discord notice is posted only when the insert actually
|
|
18
|
+
// created the row, so it is exactly as edge-triggered as the blocker.
|
|
19
|
+
//
|
|
20
|
+
// CLOSED BY SUCCESS. A later core move on the same project that lands resolves every open
|
|
21
|
+
// blocker this file filed for that project, with a note saying which move fixed it, and
|
|
22
|
+
// posts one all-clear. A person can still resolve one by hand; this never reopens it.
|
|
23
|
+
//
|
|
24
|
+
// The Discord copy names a project only when it is public on the platform's map (the rule
|
|
25
|
+
// liveness-notify.js keeps), and never carries upgrade output — that stays in the blocker,
|
|
26
|
+
// inside the trust boundary. Writes go through the runner's own db handle (the hub
|
|
27
|
+
// database, where the blockers table lives), not modules/ideas/blockers.js's shared pool,
|
|
28
|
+
// which this CLI process never opens; the INSERT is that file's createBlocker, verbatim.
|
|
29
|
+
|
|
30
|
+
const { execFile } = require('child_process');
|
|
31
|
+
const { REPO_ROOT } = require('./provision-config.js');
|
|
32
|
+
|
|
33
|
+
const BLOCKER_SOURCE = 'provisioning';
|
|
34
|
+
const REF_PREFIX = 'core-move';
|
|
35
|
+
const DETAIL_MAX = 1500;
|
|
36
|
+
|
|
37
|
+
/** Does this retired move need a person? PURE. */
|
|
38
|
+
function needsPerson({ intent, failure, terminal }) {
|
|
39
|
+
if (!intent || intent.action !== 'core-upgrade') return false;
|
|
40
|
+
return !!terminal || !!(failure && (failure.class === 'unknown' || failure.reason === 'rollback_failed'));
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
function sourceRef(inst, intent) { return `${REF_PREFIX}:${inst.id}:${intent.id}`; }
|
|
44
|
+
|
|
45
|
+
function projectName(inst, provisioning) {
|
|
46
|
+
try { return provisioning.effectiveSettings(inst).visibility === 'public' ? inst.slug : null; } catch { return null; }
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/** The blocker a failed move files. PURE. */
|
|
50
|
+
function blockerFor({ inst, intent, failure, error }) {
|
|
51
|
+
const reason = (failure && failure.reason) || 'unknown';
|
|
52
|
+
const down = reason === 'rollback_failed';
|
|
53
|
+
const to = intent.target_version || '(no target)';
|
|
54
|
+
const detail = String(error || '').slice(0, DETAIL_MAX).replace(/```/g, "'''");
|
|
55
|
+
return {
|
|
56
|
+
title: down
|
|
57
|
+
? `${inst.slug}: the core move to ${to} failed AND its rollback did not finish — the project may be down`
|
|
58
|
+
: `${inst.slug}: the core move to ${to} failed and the runner cannot fix it`,
|
|
59
|
+
bodyMd: [
|
|
60
|
+
`**Project:** ${inst.slug} (instance ${inst.id}) · **target:** ${to} · **reason:** \`${reason}\` · **move:** intent ${intent.id}`,
|
|
61
|
+
'',
|
|
62
|
+
down
|
|
63
|
+
? 'The move changed the project, failed, and the automatic rollback also failed. It may be down or on a half-installed core. Check it before anything else.'
|
|
64
|
+
: 'The runner retired this move: it is not a known snag it can fix, or its one automatic fix did not work. Nothing is being retried.',
|
|
65
|
+
'',
|
|
66
|
+
'What the move reported:',
|
|
67
|
+
'```',
|
|
68
|
+
detail || '(nothing)',
|
|
69
|
+
'```',
|
|
70
|
+
'',
|
|
71
|
+
'This closes by itself when a later core move on this project succeeds.',
|
|
72
|
+
].join('\n'),
|
|
73
|
+
source: BLOCKER_SOURCE,
|
|
74
|
+
sourceRef: sourceRef(inst, intent),
|
|
75
|
+
};
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/** The owner-facing Discord line. `name` null → described, never identified. PURE. */
|
|
79
|
+
function noticeText({ kind, name, to, down }) {
|
|
80
|
+
const who = name ? `**${name}**` : 'A private project';
|
|
81
|
+
if (kind === 'fixed') return `✅ ${who} took its core update${to ? ` to ${to}` : ''}; the earlier failed update is closed.`;
|
|
82
|
+
return down
|
|
83
|
+
? `🚨 ${who}: a core update failed and could not be undone. The project may be down. A blocker is open with the details.`
|
|
84
|
+
: `⚠️ ${who}: a core update${to ? ` to ${to}` : ''} failed in a way the platform cannot fix by itself. A blocker is open with the details. Nothing is being retried.`;
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
// Fire-and-forget, the liveness-notify.js postNotice shape: CHAT_DRY_RUN=1 short-circuits and
|
|
88
|
+
// a Discord hiccup never stalls the drain.
|
|
89
|
+
function postNotice(text, { log = console.log } = {}) {
|
|
90
|
+
if (!text) return;
|
|
91
|
+
if (process.env.CHAT_DRY_RUN === '1') { log(`[core-move] [broadcast dry-run] would post: ${text}`); return; }
|
|
92
|
+
try {
|
|
93
|
+
execFile('node', ['scripts/discord/post.js', text], { cwd: REPO_ROOT, timeout: 10_000 }, (err) => {
|
|
94
|
+
if (err) log(`[core-move] notice broadcast failed (non-blocking): ${err.message}`);
|
|
95
|
+
});
|
|
96
|
+
} catch (e) { log(`[core-move] notice broadcast failed (non-blocking): ${(e && e.message) || e}`); }
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** After a move is retired: file its blocker and notice, once. Never throws. */
|
|
100
|
+
async function afterRetire({ intent, inst, result, deps }) {
|
|
101
|
+
const failure = result && result.failure;
|
|
102
|
+
if (!deps.apply || !needsPerson({ intent, failure, terminal: result && result.terminal })) return { filed: false };
|
|
103
|
+
const b = blockerFor({ inst, intent, failure, error: result.error });
|
|
104
|
+
try {
|
|
105
|
+
const { rows } = await deps.db.query(
|
|
106
|
+
`INSERT INTO blockers (title, body_md, source, source_ref)
|
|
107
|
+
VALUES ($1, $2, $3, $4)
|
|
108
|
+
ON CONFLICT (source, source_ref) WHERE source IS NOT NULL AND source_ref IS NOT NULL
|
|
109
|
+
DO NOTHING
|
|
110
|
+
RETURNING id`, [b.title, b.bodyMd, b.source, b.sourceRef]);
|
|
111
|
+
if (!rows || !rows[0]) return { filed: false };
|
|
112
|
+
const down = (failure && failure.reason) === 'rollback_failed';
|
|
113
|
+
(deps.postNotice || postNotice)(noticeText({ kind: 'failed', name: projectName(inst, deps.provisioning), to: intent.target_version, down }), { log: deps.log });
|
|
114
|
+
(deps.log || (() => {}))(` ⚑ filed blocker ${rows[0].id} for ${inst.slug}'s failed move to ${intent.target_version}`);
|
|
115
|
+
return { filed: true, blockerId: rows[0].id };
|
|
116
|
+
} catch (e) {
|
|
117
|
+
(deps.log || (() => {}))(` ⚠ could not file a blocker for ${inst.slug}'s failed move: ${(e && e.message) || e}`);
|
|
118
|
+
return { filed: false };
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/** After a move LANDS: close every open blocker this file filed for the project. Never throws. */
|
|
123
|
+
async function afterSuccess({ intent, inst, result, deps }) {
|
|
124
|
+
if (!deps.apply || !intent || intent.action !== 'core-upgrade' || !result || result.ok === false || result.noop) return { closed: 0 };
|
|
125
|
+
const to = result.served || intent.target_version;
|
|
126
|
+
try {
|
|
127
|
+
const { rows } = await deps.db.query(
|
|
128
|
+
`UPDATE blockers SET status = 'resolved', resolved_at = now(), resolution_note = $3
|
|
129
|
+
WHERE status = 'open' AND source = $1 AND source_ref LIKE $2
|
|
130
|
+
RETURNING id`,
|
|
131
|
+
[BLOCKER_SOURCE, `${REF_PREFIX}:${inst.id}:%`, `closed automatically: a later core move to ${to} succeeded (intent ${intent.id})`]);
|
|
132
|
+
const n = (rows || []).length;
|
|
133
|
+
if (n) (deps.postNotice || postNotice)(noticeText({ kind: 'fixed', name: projectName(inst, deps.provisioning), to }), { log: deps.log });
|
|
134
|
+
return { closed: n };
|
|
135
|
+
} catch (e) {
|
|
136
|
+
(deps.log || (() => {}))(` ⚠ could not close ${inst.slug}'s core-move blockers: ${(e && e.message) || e}`);
|
|
137
|
+
return { closed: 0 };
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
module.exports = { BLOCKER_SOURCE, REF_PREFIX, needsPerson, sourceRef, blockerFor, noticeText, postNotice, afterRetire, afterSuccess };
|
|
@@ -241,6 +241,16 @@ async function coreUpgradeInstance(inst, deps, intent) {
|
|
|
241
241
|
log(` already serving core ${to} — nothing to move`);
|
|
242
242
|
return { ok: true, noop: true };
|
|
243
243
|
}
|
|
244
|
+
// ---- 2b. is the database current for the code it runs NOW (task 1004448) -------------
|
|
245
|
+
// A database behind its own code is already failing every page that touches the missing
|
|
246
|
+
// tables (the 2026-07-04 shape), and moving the code on top of it tangles a catch-up
|
|
247
|
+
// migration into a pin move. So it is refused as a known snag, and wedge-remedy.js
|
|
248
|
+
// migrates that database and tries the move once more. An answer it cannot read proceeds.
|
|
249
|
+
if (apply) {
|
|
250
|
+
const probe = boxExec(versionCmd(inst), { allowFail: true });
|
|
251
|
+
const pending = parseSchemaPending(probe && !probe.softFailed ? (probe.stdout || '') : null);
|
|
252
|
+
if (pending > 0) return failed('schema_pending', `the project's database is ${pending} migration(s) behind the code it is running now — it has to catch up before the core moves`);
|
|
253
|
+
}
|
|
244
254
|
|
|
245
255
|
// ---- 3. drive the proven path -----------------------------------------------------
|
|
246
256
|
// Exactly the invocation docs/recipes/core-release-pipeline.md gate 3 documents, plus
|
package/scripts/gds/provision.js
CHANGED
|
@@ -1196,6 +1196,9 @@ async function cmdRunIntents(deps) {
|
|
|
1196
1196
|
// WAIT OR RETIRE (task 1004447, intent-retry.js): a transient failure waits ~15m/1h/6h; a
|
|
1197
1197
|
// classified non-transient one is retired at once (retrying cannot change its answer);
|
|
1198
1198
|
// an unclassified one keeps its budget but never re-runs in the same tick.
|
|
1199
|
+
const fix = await require('./wedge-remedy.js').remedyAfterFailure({ intent, inst, result, deps }); // a KNOWN snag: fix it, try once more (task 1004448)
|
|
1200
|
+
if (fix.repended) { errors++; continue; }
|
|
1201
|
+
Object.assign(result, { failure: fix.failure, error: fix.error });
|
|
1199
1202
|
const next = retry.retryDecision({ attempts: intent.attempts, failure: result.failure, terminal: result.terminal, maxAttempts: MAX_INTENT_ATTEMPTS });
|
|
1200
1203
|
if (next.action === 'retire') { // terminal: a leg that must not be retried unattended (the Render leg's key is already gone — task 1004183)
|
|
1201
1204
|
await provisioning.resolveIntent(db, intent.id, 'error', result.error || 'op failed');
|
|
@@ -1206,12 +1209,13 @@ async function cmdRunIntents(deps) {
|
|
|
1206
1209
|
// can say why the work stopped.
|
|
1207
1210
|
await provisioning.setInstanceStatus(db, intent.instance_id, inst.status, { error_note: result.error || 'op failed', error_note_action: intent.action }).catch(() => {});
|
|
1208
1211
|
if (result.failure) await recordFailureClass(db, intent.id, result.failure); // WHAT KIND of failure, beside the sentence (task 1004446)
|
|
1212
|
+
await require('./move-escalation.js').afterRetire({ intent, inst, result, deps }); // nobody can heal it: one blocker + one notice per move (task 1004449)
|
|
1209
1213
|
} else {
|
|
1210
1214
|
await db.query(`UPDATE provisioning_intents SET state='pending', last_error=$2, not_before = now() + ($3 * interval '1 millisecond'), updated_at=now() WHERE id=$1`, [intent.id, result.error || 'op failed', next.delayMs]);
|
|
1211
1215
|
if (result.failure) await recordFailureClass(db, intent.id, result.failure);
|
|
1212
1216
|
}
|
|
1213
1217
|
errors++;
|
|
1214
|
-
} else { await provisioning.resolveIntent(db, intent.id, 'done', (result && result.doneNote) || null); ok++; } // done WITH a note: it landed, and this is what it could not finish (task 1004065)
|
|
1218
|
+
} else { await provisioning.resolveIntent(db, intent.id, 'done', (result && result.doneNote) || null); ok++; await require('./move-escalation.js').afterSuccess({ intent, inst, result, deps }); } // done WITH a note: it landed, and this is what it could not finish (task 1004065)
|
|
1215
1219
|
} catch (e) {
|
|
1216
1220
|
await provisioning.resolveIntent(db, intent.id, 'error', e.message).catch(() => {});
|
|
1217
1221
|
// Surface the failure on the instance too, so a stuck instance is visible.
|
|
@@ -19,7 +19,7 @@
|
|
|
19
19
|
//
|
|
20
20
|
// THE FOUR CLASSES the rest of the self-heal chain keys on:
|
|
21
21
|
// transient — try again later and it may simply work (registry blip, not published yet)
|
|
22
|
-
// known_wedge — a known snag with a known fix the runner
|
|
22
|
+
// known_wedge — a known snag with a known fix the runner applies itself, once (wedge-remedy.js)
|
|
23
23
|
// needs_decision — a rule the owner set is refusing it; only the owner can change that
|
|
24
24
|
// unknown — anything else, and every rollback that failed; a person should look
|
|
25
25
|
//
|
|
@@ -42,6 +42,7 @@ const REASON_CLASS = Object.freeze({
|
|
|
42
42
|
bad_slug: 'unknown',
|
|
43
43
|
installed_unreadable: 'unknown',
|
|
44
44
|
served_mismatch: 'unknown',
|
|
45
|
+
schema_pending: 'known_wedge', // the database is behind the code it runs NOW (task 1004448)
|
|
45
46
|
// upgrade.js refusals, before the pin moves
|
|
46
47
|
dirty_tree: 'known_wedge',
|
|
47
48
|
downgrade: 'needs_decision',
|
|
@@ -59,6 +60,9 @@ const REASON_CLASS = Object.freeze({
|
|
|
59
60
|
migrate: 'unknown',
|
|
60
61
|
restart: 'unknown',
|
|
61
62
|
health: 'unknown',
|
|
63
|
+
// the runner's one automatic fix did not clear it, or itself failed (task 1004448)
|
|
64
|
+
remedy_did_not_hold: 'unknown',
|
|
65
|
+
remedy_failed: 'unknown',
|
|
62
66
|
// the one that must never be retried unattended
|
|
63
67
|
rollback_failed: 'unknown',
|
|
64
68
|
});
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
//
|
|
3
|
+
// scripts/gds/wedge-remedy.js — a core move that failed on a KNOWN snag fixes the snag and
|
|
4
|
+
// tries once more (task 1004448, goal 1000090 deploy self-heal).
|
|
5
|
+
//
|
|
6
|
+
// WHY. upgrade-outcome.js names three failures that are not a fault in the release and not a
|
|
7
|
+
// decision the owner has to make, but a mess on the box with one known fix — the fixes the
|
|
8
|
+
// owner has been doing over SSH (docs/recipes/upgrading-the-core.md, Gotchas):
|
|
9
|
+
// dirty_tree — the project's checkout has uncommitted changes to tracked files (the
|
|
10
|
+
// upgrade's own rewrites, left behind by an earlier move). Fix: commit them
|
|
11
|
+
// inside the project.
|
|
12
|
+
// disk_full — the move ran out of disk. Fix: prune this project's OWN nightly
|
|
13
|
+
// pg_dumps down to the newest few (the only thing on the box this runner
|
|
14
|
+
// knows is safe to delete — they are copies, and the newest are kept).
|
|
15
|
+
// schema_pending — the project's database is behind the code it is running NOW. Fix:
|
|
16
|
+
// migrate that database, with its own PGDATABASE, before the code moves.
|
|
17
|
+
// intent-retry.js retires every known_wedge at once, so on its own the failure only waited
|
|
18
|
+
// for a person. This file runs the fix, records it as an event on the project, and re-pends
|
|
19
|
+
// the move for ONE more try.
|
|
20
|
+
//
|
|
21
|
+
// AT MOST ONCE PER FAILURE. The reason a fix was applied for is stored on the intent
|
|
22
|
+
// (provisioning_intents.remedied_reason, migration provisioning_032). If the retry fails on
|
|
23
|
+
// the SAME reason, the fix did not hold: the move is retired as `unknown` /
|
|
24
|
+
// remedy_did_not_hold, which is the class that reaches a person — never a second fix, never
|
|
25
|
+
// a loop. A fix that itself fails is retired the same way (remedy_failed). A different
|
|
26
|
+
// known reason on the retry gets its own one fix (a disk full after a dirty tree is a new
|
|
27
|
+
// snag, not the same one twice); the attempt budget still bounds the whole run.
|
|
28
|
+
//
|
|
29
|
+
// 'tagged-not-published' needs nothing here: it is `not_published`, a transient failure that
|
|
30
|
+
// intent-retry.js already waits out.
|
|
31
|
+
//
|
|
32
|
+
// planRemedy is PURE (the command, never run). remedyAfterFailure does the I/O through the
|
|
33
|
+
// runner's own exec and db, so every fix is testable with an injected exec.
|
|
34
|
+
|
|
35
|
+
const { CONFIG } = require('./provision-config.js');
|
|
36
|
+
const { dbName, selfInstanceRoot, standaloneRoot } = require('./provision-repo.js');
|
|
37
|
+
const { migratePlan, CORE_MIGRATE_SCRIPT } = require('./upgrade-migrate.js');
|
|
38
|
+
const { isValidSlug } = require('../../modules/provisioning/provisioning.js');
|
|
39
|
+
const outcome = require('./upgrade-outcome.js');
|
|
40
|
+
const { DEFAULT_RETRY_DELAY_MS } = require('./intent-retry.js');
|
|
41
|
+
|
|
42
|
+
const REMEDY_REASONS = Object.freeze(['dirty_tree', 'disk_full', 'schema_pending']);
|
|
43
|
+
// The newest dumps a disk-full prune always leaves behind: a week of nightlies. The nightly
|
|
44
|
+
// job's own retention is 30 days; this only ever cuts below that when the disk is full.
|
|
45
|
+
const DUMP_KEEP_FLOOR = 7;
|
|
46
|
+
const REMEDY_EVENT = 'core-upgrade-remedy';
|
|
47
|
+
// A value that reaches a shell string below. Narrower than what Postgres or a path allow, on
|
|
48
|
+
// purpose: anything outside it is refused, never quoted.
|
|
49
|
+
const SHELL_SAFE_RE = /^[A-Za-z0-9._/-]{1,200}$/;
|
|
50
|
+
const DB_NAME_RE = /^[A-Za-z0-9_-]{1,63}$/;
|
|
51
|
+
const ACCOUNT_RE = /^[a-z_][a-z0-9_-]{0,31}$/;
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* The fix for one known snag, as a command to run on the box. PURE.
|
|
55
|
+
* @param {object} inst the provisioning_instances row
|
|
56
|
+
* @param {string} reason the failure's reason code
|
|
57
|
+
* @param {object} [o] { privileged, runAs, instanceDir, selfRoot, backupDir, migrate }
|
|
58
|
+
* @returns {{ reason, cmd, cwd, env, describe } | { refused: string } | null} null: no known fix
|
|
59
|
+
*/
|
|
60
|
+
function planRemedy(inst, reason, o = {}) {
|
|
61
|
+
if (!REMEDY_REASONS.includes(reason)) return null;
|
|
62
|
+
if (!inst || !isValidSlug(inst.slug)) return { refused: 'the project name is not a valid slug' };
|
|
63
|
+
const dir = o.instanceDir || (inst.hosting_shape === 'cloud-host' ? standaloneRoot(inst) : (o.selfRoot || selfInstanceRoot)());
|
|
64
|
+
if (!SHELL_SAFE_RE.test(dir)) return { refused: 'the project folder is not a path this runner will put in a command' };
|
|
65
|
+
const db = dbName(inst);
|
|
66
|
+
if (!DB_NAME_RE.test(db)) return { refused: 'the project database name is not one this runner will put in a command' };
|
|
67
|
+
|
|
68
|
+
if (reason === 'dirty_tree') {
|
|
69
|
+
// TRACKED files only (`add -u`): the snag is the upgrade's own rewrites of tracked files.
|
|
70
|
+
// A new untracked file is not swept into the owner's history on a guess — if that is the
|
|
71
|
+
// dirt, the retry fails the same way and a person looks. Runs as the runner, which owns
|
|
72
|
+
// the checkout (pushUpgradePin's shape), with hooks off and a runner identity so the
|
|
73
|
+
// commit says who made it.
|
|
74
|
+
const g = `git -C ${dir} -c core.hooksPath=/dev/null -c user.name=bongos-runner -c user.email=runner@bongos.invalid`;
|
|
75
|
+
return {
|
|
76
|
+
reason, cwd: null, env: null,
|
|
77
|
+
cmd: `${g} add -u && ${g} commit -m "chore: commit changes left in the working tree (automatic fix before a core move)"`,
|
|
78
|
+
describe: `committed the uncommitted changes in ${dir} so the core move could run`,
|
|
79
|
+
};
|
|
80
|
+
}
|
|
81
|
+
if (reason === 'disk_full') {
|
|
82
|
+
const backupDir = o.backupDir || CONFIG.backupDir;
|
|
83
|
+
if (!SHELL_SAFE_RE.test(backupDir)) return { refused: 'the backup folder is not a path this runner will put in a command' };
|
|
84
|
+
// EXACTLY db-backup-nightly.sh's name, so this never touches a sibling project's dumps
|
|
85
|
+
// (`test` must not prune `test-nk`'s — task 1003948). The timestamp sorts by name, so
|
|
86
|
+
// `sort | head -n -K` drops all but the newest K. Nothing is deleted when K or fewer exist.
|
|
87
|
+
const D = '[0-9]';
|
|
88
|
+
const glob = `${db}-${D}${D}${D}${D}-${D}${D}-${D}${D}T${D}${D}${D}${D}${D}${D}Z.sql.gz`;
|
|
89
|
+
return {
|
|
90
|
+
reason, cwd: null, env: null,
|
|
91
|
+
cmd: `find ${backupDir} -maxdepth 1 -type f -name '${glob}' | sort | head -n -${DUMP_KEEP_FLOOR} | xargs -r rm -f --`,
|
|
92
|
+
describe: `pruned ${db}'s old database backups in ${backupDir}, keeping the newest ${DUMP_KEEP_FLOOR}`,
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
// schema_pending — the project's OWN migrate command (a declared script wins, else the
|
|
96
|
+
// core's migrate.sh: upgrade-migrate.js), aimed at its OWN database. migrate.sh defaults to
|
|
97
|
+
// production when PGDATABASE is unset, so it is always set, never inherited.
|
|
98
|
+
const plan = (o.migrate || migratePlan)(dir);
|
|
99
|
+
const run = plan.via === 'npm-script' ? 'npm run migrate' : `bash ${CORE_MIGRATE_SCRIPT}`;
|
|
100
|
+
const initCwd = plan.via === 'npm-script' ? '' : `INIT_CWD=${dir} `;
|
|
101
|
+
if (o.privileged) {
|
|
102
|
+
if (!ACCOUNT_RE.test(String(o.runAs || ''))) return { refused: 'no account to run the migration as' };
|
|
103
|
+
return { reason, cwd: dir, env: null, cmd: `sudo -u ${o.runAs} env PGDATABASE=${db} ${initCwd}${run}`, describe: `migrated ${db} to match the code it is running` };
|
|
104
|
+
}
|
|
105
|
+
return { reason, cwd: dir, env: { PGDATABASE: db, ...(initCwd ? { INIT_CWD: dir } : {}) }, cmd: run, describe: `migrated ${db} to match the code it is running` };
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
// The account the project runs as, probed the way the move itself probes (task 1004128).
|
|
109
|
+
function runAsFor(inst, deps) {
|
|
110
|
+
try { return require('./provision-core-upgrade.js').resolveRunAs(inst, deps).runAs; } catch { return null; }
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* After a core move failed: apply the known fix, record it, and re-pend the move once.
|
|
115
|
+
*
|
|
116
|
+
* Returns `{ repended: true }` when the move will be tried again, otherwise
|
|
117
|
+
* `{ repended: false, failure, error }` — the failure the drain loop should RETIRE with,
|
|
118
|
+
* which is the original one when there is no fix for it, and `unknown` when a fix was
|
|
119
|
+
* already spent on this reason or the fix itself failed.
|
|
120
|
+
*/
|
|
121
|
+
async function remedyAfterFailure({ intent, inst, result, deps }) {
|
|
122
|
+
const failure = result && result.failure;
|
|
123
|
+
const reason = failure && failure.reason;
|
|
124
|
+
const none = { repended: false, failure, error: result && result.error };
|
|
125
|
+
if (!intent || intent.action !== 'core-upgrade' || !failure || failure.class !== 'known_wedge') return none;
|
|
126
|
+
const { db, provisioning, log = () => {}, apply } = deps;
|
|
127
|
+
if (!apply) return none;
|
|
128
|
+
if (intent.remedied_reason === reason) {
|
|
129
|
+
return {
|
|
130
|
+
repended: false, failure: outcome.classifyFailure('remedy_did_not_hold'),
|
|
131
|
+
error: `${result.error || 'core move failed'} — the automatic fix for this (${reason}) was already tried once and did not clear it, so a person needs to look`,
|
|
132
|
+
};
|
|
133
|
+
}
|
|
134
|
+
const boxExec = inst.hosting_shape === 'cloud-host' ? (deps.controlExec || deps.exec) : deps.exec;
|
|
135
|
+
const plan = planRemedy(inst, reason, { privileged: deps.privileged, runAs: reason === 'schema_pending' && deps.privileged ? runAsFor(inst, deps) : null, selfRoot: deps.selfInstanceRoot, migrate: deps.migratePlan });
|
|
136
|
+
if (!plan) return none;
|
|
137
|
+
const giveUp = async (why) => {
|
|
138
|
+
await provisioning.recordEvent(db, { instanceId: inst.id, ownerBuilderId: inst.owner_builder_id, event: REMEDY_EVENT, detail: `could not fix ${reason} automatically: ${why}`, actor: 'runner' }).catch(() => {});
|
|
139
|
+
return { repended: false, failure: outcome.classifyFailure('remedy_failed'), error: `${result.error || 'core move failed'} — the automatic fix for ${reason} did not work (${why}), so a person needs to look` };
|
|
140
|
+
};
|
|
141
|
+
if (plan.refused) return giveUp(plan.refused);
|
|
142
|
+
log(` [remedy] ${reason}: ${plan.cmd}`);
|
|
143
|
+
const r = boxExec(plan.cmd, { allowFail: true, ...(plan.cwd ? { cwd: plan.cwd } : {}), ...(plan.env ? { env: plan.env } : {}) });
|
|
144
|
+
if (!r || r.softFailed === true || r.ok === false) {
|
|
145
|
+
const said = String((r && (r.stderr || r.stdout || r.error)) || 'the command failed').trim().split('\n').pop().slice(0, 160);
|
|
146
|
+
return giveUp(said);
|
|
147
|
+
}
|
|
148
|
+
await provisioning.recordEvent(db, { instanceId: inst.id, ownerBuilderId: inst.owner_builder_id, event: REMEDY_EVENT, detail: `${plan.describe}; trying the move to ${intent.target_version} once more`, actor: 'runner' }).catch(() => {});
|
|
149
|
+
await db.query(
|
|
150
|
+
`UPDATE provisioning_intents SET state='pending', last_error=$2, remedied_reason=$3, failure_class=$4, failure_reason=$5,
|
|
151
|
+
not_before = now() + ($6 * interval '1 millisecond'), updated_at=now() WHERE id=$1`,
|
|
152
|
+
[intent.id, result.error || 'core move failed', reason, failure.class, failure.reason, DEFAULT_RETRY_DELAY_MS]);
|
|
153
|
+
log(` ✓ fixed ${reason} on ${inst.slug}; the move to ${intent.target_version} will be tried once more`);
|
|
154
|
+
return { repended: true };
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
module.exports = { REMEDY_REASONS, DUMP_KEEP_FLOOR, REMEDY_EVENT, planRemedy, remedyAfterFailure };
|
package/src/module-api.js
CHANGED
|
@@ -75,7 +75,7 @@ const { responsibilityFor, ROLE_RESPONSIBILITIES } = require('./role-responsibil
|
|
|
75
75
|
// MAJOR (see allowBoxScope below): passes the request through untouched.
|
|
76
76
|
function deprecatedNoopMiddleware(_req, _res, next) { next(); }
|
|
77
77
|
|
|
78
|
-
const CORE_VERSION = '1.20.
|
|
78
|
+
const CORE_VERSION = '1.20.33'; // CI auto-patch carrier (ADR 0161); changelog: docs/module-api-changelog.md
|
|
79
79
|
|
|
80
80
|
// A namespaced logger so a module's log lines are attributable + consistent.
|
|
81
81
|
// Usage: const log = api.logger('discord'); log.info('mounted');
|
|
@@ -40,6 +40,15 @@ const walk = (classes, cfg) => {
|
|
|
40
40
|
|
|
41
41
|
// ── classification ──────────────────────────────────────────────────────────────
|
|
42
42
|
|
|
43
|
+
test('a queue-wide gate is a wait, not a failure (task 1004396)', () => {
|
|
44
|
+
// The strand behind it is somebody's unlanded ship, not this machine breaking.
|
|
45
|
+
// Escalating it would slow the runner down for a fault it cannot fix, and the
|
|
46
|
+
// wait already wakes early when main moves, which is what the strand landing does.
|
|
47
|
+
const seen = classifyEvent({ event: 'queue_gated', waiting_on: ['777'], reason: 'x' });
|
|
48
|
+
assert.equal(seen.class, 'quiet');
|
|
49
|
+
assert.match(seen.detail, /777/);
|
|
50
|
+
});
|
|
51
|
+
|
|
43
52
|
test('a worked task counts as progress only when the LEDGER agreed', () => {
|
|
44
53
|
assert.equal(classifyEvent({ event: 'worked', verified: true }).class, 'progress');
|
|
45
54
|
// A worker's claim the ledger never saw is not progress; counting it would hold
|
|
@@ -237,6 +237,12 @@ test('an unreadable claimable feed does not halt everything', async () => {
|
|
|
237
237
|
assert.equal(got.task.id, 9001, 'the not_claimable reason stays quiet when the feed cannot be read');
|
|
238
238
|
});
|
|
239
239
|
|
|
240
|
+
test('pickTask passes over a task the runner has set aside (task 1004396)', async () => {
|
|
241
|
+
const api = fakeApi({ tasks: [ready({ id: 9001 }), ready({ id: 9002 })], claimable: [{ id: 9001 }, { id: 9002 }] });
|
|
242
|
+
const got = await runner.pickTask([1000119], { api, rank: 'archon', exclude: new Set(['9001']) });
|
|
243
|
+
assert.equal(got.task.id, 9002, 'a refused task must not stay at the head of the order');
|
|
244
|
+
});
|
|
245
|
+
|
|
240
246
|
test('a goal whose tasks cannot be read is reported, not silently skipped', async () => {
|
|
241
247
|
const api = fakeApi({ tasks: [], claimable: [] });
|
|
242
248
|
api.tasks.getTasks = async () => { throw new Error('boom'); };
|
|
@@ -273,7 +279,11 @@ function harness(over = {}) {
|
|
|
273
279
|
// The LEDGER, faked. iteration() now reads the task back from Bongos after
|
|
274
280
|
// every worker (task 1003903) instead of believing the worker's own verdict,
|
|
275
281
|
// so a test that does not say what the ledger holds is not describing a run.
|
|
276
|
-
api: {},
|
|
282
|
+
api: over.api || {},
|
|
283
|
+
// Per-test runner memory (task 1004396): refused tasks and a queue gate
|
|
284
|
+
// outlive one iteration, so a shared module-level copy would leak between tests.
|
|
285
|
+
state: over.state || runner.newRunnerState(),
|
|
286
|
+
now: over.now,
|
|
277
287
|
verifyDeps: {
|
|
278
288
|
task: over.ledger !== undefined ? over.ledger : { id: 4242, status: 'shipped', updated_at: new Date().toISOString() },
|
|
279
289
|
probeArtifact: over.probeArtifact || (async () => ({ checked: true, onMain: true, sha: 'a'.repeat(40), subject: 'x (task 4242)' })),
|
|
@@ -422,6 +432,137 @@ test('failureReason: stdout is used when stderr is empty, whitespace is collapse
|
|
|
422
432
|
assert.equal(runner.failureReason({ code: 1, stderr: 'a'.repeat(400) }, 'x').length, 400);
|
|
423
433
|
});
|
|
424
434
|
|
|
435
|
+
// ── task 1004396: a refusal is not a reason to stop, unless it refuses everything ──
|
|
436
|
+
//
|
|
437
|
+
// Owner ruling (2026-09-30): a refused claim must not put the runner into waiting
|
|
438
|
+
// by default. It sets that task aside and tries the next one. It waits only when
|
|
439
|
+
// there is a block AND nothing else it can claim — which is exactly the shape of
|
|
440
|
+
// the queue-wide gate (a stranded confirmed task refuses EVERY claim).
|
|
441
|
+
|
|
442
|
+
const taskA = { ...aTask, id: 4242 };
|
|
443
|
+
const taskB = { ...aTask, id: 4343, title: 'The next thing' };
|
|
444
|
+
// A picker that honours the exclusion set, the way the real pickTask now does.
|
|
445
|
+
const pickAorB = async (_goals, pd = {}) => {
|
|
446
|
+
const ex = pd.exclude || new Set();
|
|
447
|
+
for (const t of [taskA, taskB]) if (!ex.has(String(t.id))) return { task: t, goalId: 1000119, skipped: [] };
|
|
448
|
+
return { none: true, skipped: [] };
|
|
449
|
+
};
|
|
450
|
+
const scriptOf = (args) => String(args[0]).split(/[\\/]/).pop();
|
|
451
|
+
const gatedCard = [
|
|
452
|
+
'Your queue is gated — 1 confirmed task(s) are waiting on your rebase:',
|
|
453
|
+
' - task 777 A stranded thing',
|
|
454
|
+
' flagged: strand:branch_modifies_executed_code',
|
|
455
|
+
].join(String.fromCharCode(10));
|
|
456
|
+
|
|
457
|
+
test('a task refused on its own is set aside and the NEXT task is worked in the same pass', async () => {
|
|
458
|
+
const claims = [];
|
|
459
|
+
const h = harness({
|
|
460
|
+
pickTask: pickAorB,
|
|
461
|
+
ledger: { id: 4343, status: 'shipped', updated_at: new Date().toISOString() },
|
|
462
|
+
run: async (bin, args) => {
|
|
463
|
+
if (scriptOf(args) === 'claim.js') {
|
|
464
|
+
claims.push(args[1]);
|
|
465
|
+
return args[1] === '4242' ? { ok: false, code: 1, stdout: '', stderr: 'DEPS_NOT_SHIPPED' } : { ok: true, code: 0, stdout: '', stderr: '' };
|
|
466
|
+
}
|
|
467
|
+
return { ok: true, code: 0, stdout: '', stderr: '' };
|
|
468
|
+
},
|
|
469
|
+
});
|
|
470
|
+
const row = await runner.iteration(opts, h.deps);
|
|
471
|
+
assert.deepEqual(claims, ['4242', '4343'], 'the refusal must not end the pass');
|
|
472
|
+
assert.equal(row.event, 'worked');
|
|
473
|
+
assert.equal(row.task_id, 4343);
|
|
474
|
+
const refused = h.rows.find((r) => r.event === 'claim_failed');
|
|
475
|
+
assert.equal(refused.task_id, 4242, 'the refusal is still logged, with its reason');
|
|
476
|
+
assert.ok(refused.set_aside_s > 0, 'and says the task is set aside');
|
|
477
|
+
});
|
|
478
|
+
|
|
479
|
+
test('a set-aside task is not re-tried on the next wake, and comes back once the set-aside expires', async () => {
|
|
480
|
+
let t = 1_000_000;
|
|
481
|
+
const state = runner.newRunnerState();
|
|
482
|
+
const claims = [];
|
|
483
|
+
const mk = () => harness({
|
|
484
|
+
state, now: () => t, pickTask: pickAorB,
|
|
485
|
+
run: async (bin, args) => {
|
|
486
|
+
if (scriptOf(args) === 'claim.js') { claims.push(args[1]); return { ok: false, code: 1, stdout: '', stderr: 'nope' }; }
|
|
487
|
+
return { ok: true, code: 0, stdout: '', stderr: '' };
|
|
488
|
+
},
|
|
489
|
+
});
|
|
490
|
+
await runner.iteration(opts, mk().deps);
|
|
491
|
+
assert.deepEqual(claims, ['4242', '4343']);
|
|
492
|
+
const second = await runner.iteration(opts, mk().deps);
|
|
493
|
+
assert.deepEqual(claims, ['4242', '4343'], 'no claim is re-attempted while both are set aside');
|
|
494
|
+
assert.equal(second.event, 'nothing_claimable');
|
|
495
|
+
assert.deepEqual(second.set_aside.map((s) => s.id).sort(), ['4242', '4343']);
|
|
496
|
+
t += runner.CLAIM_SET_ASIDE_MS + 1;
|
|
497
|
+
await runner.iteration(opts, mk().deps);
|
|
498
|
+
assert.deepEqual(claims.slice(2), ['4242', '4343'], 'after the set-aside they are tried again');
|
|
499
|
+
});
|
|
500
|
+
|
|
501
|
+
test('when EVERY candidate is refused, the last failure is returned so a real outage still escalates', async () => {
|
|
502
|
+
const h = harness({
|
|
503
|
+
pickTask: pickAorB,
|
|
504
|
+
run: async (bin, args) => (scriptOf(args) === 'claim.js'
|
|
505
|
+
? { ok: false, code: 1, stdout: '', stderr: 'fetch failed' }
|
|
506
|
+
: { ok: true, code: 0, stdout: '', stderr: '' }),
|
|
507
|
+
});
|
|
508
|
+
const row = await runner.iteration(opts, h.deps);
|
|
509
|
+
assert.equal(row.event, 'claim_failed', 'returning "nothing claimable" here would hide a network outage as idle');
|
|
510
|
+
assert.equal(h.rows.filter((r) => r.event === 'claim_failed').length, 2);
|
|
511
|
+
});
|
|
512
|
+
|
|
513
|
+
test('a queue-wide gate WAITS, names the stranded task, and makes no claim until it lands', async () => {
|
|
514
|
+
const state = runner.newRunnerState();
|
|
515
|
+
let status = 'confirmed';
|
|
516
|
+
let gated = true;
|
|
517
|
+
const claims = [];
|
|
518
|
+
const api = { tasks: { getTasksId: async ({ id }) => ({ ok: true, data: { task: { id, status } } }) } };
|
|
519
|
+
const mk = () => harness({
|
|
520
|
+
state, api, pickTask: pickAorB,
|
|
521
|
+
run: async (bin, args) => {
|
|
522
|
+
if (scriptOf(args) === 'claim.js') {
|
|
523
|
+
claims.push(args[1]);
|
|
524
|
+
return gated ? { ok: false, code: runner.QUEUE_GATED_EXIT, stdout: '', stderr: gatedCard } : { ok: true, code: 0, stdout: '', stderr: '' };
|
|
525
|
+
}
|
|
526
|
+
return { ok: true, code: 0, stdout: '', stderr: '' };
|
|
527
|
+
},
|
|
528
|
+
});
|
|
529
|
+
|
|
530
|
+
const first = await runner.iteration(opts, mk().deps);
|
|
531
|
+
assert.equal(first.event, 'queue_gated');
|
|
532
|
+
assert.deepEqual(first.waiting_on, ['777'], 'the wait names what it is waiting on');
|
|
533
|
+
assert.deepEqual(claims, ['4242'], 'a gate refuses every claim, so trying the next task is pointless');
|
|
534
|
+
|
|
535
|
+
const second = await runner.iteration(opts, mk().deps);
|
|
536
|
+
assert.equal(second.event, 'queue_gated');
|
|
537
|
+
assert.deepEqual(claims, ['4242'], 'no claim is attempted while the strand is still confirmed');
|
|
538
|
+
|
|
539
|
+
status = 'shipped'; gated = false;
|
|
540
|
+
const third = await runner.iteration(opts, mk().deps);
|
|
541
|
+
assert.equal(third.event, 'worked', 'once the strand lands the runner resumes by itself');
|
|
542
|
+
assert.deepEqual(claims, ['4242', '4242']);
|
|
543
|
+
assert.equal(state.queueGate, null, 'the gate is forgotten');
|
|
544
|
+
});
|
|
545
|
+
|
|
546
|
+
test('a gate whose stranded task cannot be read back keeps waiting rather than guessing', async () => {
|
|
547
|
+
const state = runner.newRunnerState();
|
|
548
|
+
state.queueGate = { owed: ['777'], since: Date.now(), reason: 'x' };
|
|
549
|
+
let claimed = false;
|
|
550
|
+
const h = harness({
|
|
551
|
+
state, pickTask: pickAorB,
|
|
552
|
+
api: { tasks: { getTasksId: async () => ({ ok: false, status: 502 }) } },
|
|
553
|
+
run: async (bin, args) => { if (scriptOf(args) === 'claim.js') claimed = true; return { ok: true, code: 0, stdout: '', stderr: '' }; },
|
|
554
|
+
});
|
|
555
|
+
const row = await runner.iteration(opts, h.deps);
|
|
556
|
+
assert.equal(row.event, 'queue_gated');
|
|
557
|
+
assert.equal(claimed, false);
|
|
558
|
+
});
|
|
559
|
+
|
|
560
|
+
test('the runner and claim.js agree on the queue-gated exit code', () => {
|
|
561
|
+
const claim = require('../scripts/gds/claim.js');
|
|
562
|
+
assert.equal(runner.QUEUE_GATED_EXIT, claim.QUEUE_GATED_EXIT);
|
|
563
|
+
assert.notEqual(runner.QUEUE_GATED_EXIT, 1, 'it must differ from the generic refusal, or the runner cannot tell them apart');
|
|
564
|
+
});
|
|
565
|
+
|
|
425
566
|
test('a worker that cannot finish frees the claim; one that really shipped does not', async () => {
|
|
426
567
|
// Which of these happens is now decided by the LEDGER, not by the worker's own
|
|
427
568
|
// verdict: a stuck worker leaves the task still `active`, a finished one leaves
|
|
@@ -97,6 +97,16 @@ test('shipped is the LEDGER’s answer, never the worker’s', () => {
|
|
|
97
97
|
assert.deepEqual(d.onHold.ranButDidNotShip.map((r) => r.shape), ['still_held', 'shipped_no_artifact']);
|
|
98
98
|
});
|
|
99
99
|
|
|
100
|
+
test('a queue-gate wait is a hold that names the stranded task, not an unrecognised event (task 1004396)', () => {
|
|
101
|
+
const d = buildDigest([
|
|
102
|
+
{ at: at(0), event: 'queue_gated', waiting_on: ['777'], reason: 'every claim is refused until task 777 lands — waiting, not claiming (x)' },
|
|
103
|
+
{ at: at(5), event: 'queue_gate_cleared', waited_on: ['777'], waited_s: 300 },
|
|
104
|
+
], { nowEpochS: NOW_S });
|
|
105
|
+
assert.equal(d.runner.unrecognised, 0);
|
|
106
|
+
assert.equal(d.onHold.held.length, 1);
|
|
107
|
+
assert.match(d.onHold.held[0].reason, /task 777/);
|
|
108
|
+
});
|
|
109
|
+
|
|
100
110
|
test('an unworkable queue is reported as a queue problem, not as breakage', () => {
|
|
101
111
|
const d = buildDigest([{ at: at(0), event: 'nothing_claimable', skipped: [{ id: 9, reasons: ['needs_migration'] }] }], { nowEpochS: NOW_S });
|
|
102
112
|
assert.equal(d.onHold.unworkable.length, 1);
|
|
@@ -231,6 +231,13 @@ t('REBASE_REQUIRED: names the owed tasks and relays the server hint', () => {
|
|
|
231
231
|
'must not fall through to the bare code echo');
|
|
232
232
|
});
|
|
233
233
|
|
|
234
|
+
t('REBASE_REQUIRED exits with its OWN code, so an unattended caller can tell a queue gate from a task refusal (task 1004396)', () => {
|
|
235
|
+
const { QUEUE_GATED_EXIT } = require('../scripts/gds/claim.js');
|
|
236
|
+
assert.equal(formatClaimFailure(envelope('REBASE_REQUIRED', {}), 5).exit, QUEUE_GATED_EXIT);
|
|
237
|
+
assert.notEqual(QUEUE_GATED_EXIT, 1);
|
|
238
|
+
assert.equal(formatClaimFailure(envelope('ALREADY_CLAIMED', {}), 5).exit, 1, 'a per-task refusal keeps exit 1');
|
|
239
|
+
});
|
|
240
|
+
|
|
234
241
|
t('REBASE_REQUIRED: still guides when the server sends no hint and no task list', () => {
|
|
235
242
|
const out = text(formatClaimFailure(envelope('REBASE_REQUIRED', {}), 5));
|
|
236
243
|
assert.match(out, /gated/i, 'must say the queue is gated');
|
|
@@ -348,7 +348,12 @@ test('land-watch.js exits 0 after a REAL request — the teardown that aborted 3
|
|
|
348
348
|
// throwaway session, so no real API is touched and the developer's own config is
|
|
349
349
|
// never read or written. `--here` because CI may check out the MAIN checkout,
|
|
350
350
|
// where the worktree guard would refuse at exit 2 before any request is made.
|
|
351
|
-
|
|
351
|
+
// The stub refuses with REBASE_REQUIRED, the queue-wide gate, which exits with its
|
|
352
|
+
// own code since task 1004396 (QUEUE_GATED_EXIT) so an unattended caller can tell a
|
|
353
|
+
// gate from a task refusal. The abort this guards against shows as 127 either way.
|
|
354
|
+
const { QUEUE_GATED_EXIT } = createRequire(import.meta.url)('../scripts/gds/claim.js');
|
|
355
|
+
|
|
356
|
+
test('claim.js exits with its refusal code without aborting on a REFUSED claim — measured 3/3 aborting before', async () => {
|
|
352
357
|
const { proc, port } = await startStub({ claimStatus: 409 });
|
|
353
358
|
try {
|
|
354
359
|
const home = makeHome({ apiBase: `http://127.0.0.1:${port}` });
|
|
@@ -361,7 +366,7 @@ test('claim.js exits 1 without aborting on a REFUSED claim — measured 3/3 abor
|
|
|
361
366
|
});
|
|
362
367
|
assert.notEqual(r.signal, 'SIGTERM', `run ${i + 1} timed out — draining left a handle armed`);
|
|
363
368
|
assert.doesNotMatch(r.stderr || '', ABORT_RE, `run ${i + 1} aborted natively:\n${r.stderr}`);
|
|
364
|
-
assert.equal(r.status,
|
|
369
|
+
assert.equal(r.status, QUEUE_GATED_EXIT, `a queue-gated claim must exit ${QUEUE_GATED_EXIT}, got ${r.status} (127 = the abort); stderr: ${r.stderr}`);
|
|
365
370
|
// End-to-end proof that the new REBASE_REQUIRED rendering reaches a terminal,
|
|
366
371
|
// not just the unit test: the stub refuses with that exact envelope.
|
|
367
372
|
//
|