@bongos/core 1.20.32 → 1.20.34

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -40,7 +40,7 @@ const os = require('node:os');
40
40
  const seq = require('./sequence.js');
41
41
  const gauge = require('./autonomy-gauge.js');
42
42
  const {
43
- buildWorkerArgs, buildWorkerPrompt, classifyRun, DEFAULT_MODEL, DEFAULT_WORKER_TIMEOUT_MS,
43
+ buildWorkerArgs, buildWorkerPrompt, classifyRun, usageLimitHit, DEFAULT_MODEL, DEFAULT_WORKER_TIMEOUT_MS,
44
44
  } = require('./autobongos-loop.js');
45
45
  const { verifyShip, failureReason } = require('./autobongos-verify.js');
46
46
  const fence = require('../../modules/autonomy/fence.js');
@@ -95,55 +95,6 @@ function logPath() {
95
95
  try { return require('../../src/instance-config.js').configPath('autobongos-runs.jsonl'); }
96
96
  catch (_) { return path.join(REPO_ROOT, 'autobongos-runs.jsonl'); }
97
97
  }
98
- // countWorkedSince — how many tasks this runner has worked since an epoch-second
99
- // mark. Reads the run log, because that is the only record that survives a
100
- // restart, and the cap it feeds has to hold across one (a runner that forgets
101
- // its count on every crash has no cap at all).
102
- //
103
- // BOUNDED ON BOTH AXES, because this runs every iteration of a loop that never
104
- // exits and the log only grows. It reads the TAIL rather than the file, and it
105
- // scans BACKWARDS and stops at the first row older than the mark — the rows that
106
- // can possibly count are the newest ones, so the common case touches a handful
107
- // of lines whatever the log's size.
108
- //
109
- // A torn or missing log counts as zero rather than refusing: the fence and the
110
- // gauge are the gates, this is a bound on top of them, and a bound that bricks
111
- // the runner when its own log is unreadable is worse than one that occasionally
112
- // allows an extra task.
113
- const WORKED_SCAN_TAIL_BYTES = 1024 * 1024;
114
- const NEWLINE = String.fromCharCode(10);
115
-
116
- function countWorkedSince(sinceEpochS) {
117
- if (!Number.isFinite(sinceEpochS) || sinceEpochS <= 0) return 0;
118
- let text;
119
- try {
120
- const p = logPath();
121
- const { size } = fs.statSync(p);
122
- const from = Math.max(0, size - WORKED_SCAN_TAIL_BYTES);
123
- const fd = fs.openSync(p, 'r');
124
- try {
125
- const buf = Buffer.alloc(Math.min(size, WORKED_SCAN_TAIL_BYTES));
126
- fs.readSync(fd, buf, 0, buf.length, from);
127
- text = buf.toString('utf8');
128
- } finally { fs.closeSync(fd); }
129
- // A tail read can land mid-line; drop the first partial one.
130
- if (from > 0) text = text.slice(text.indexOf(NEWLINE) + 1);
131
- } catch (_) { return 0; }
132
-
133
- const lines = text.split(NEWLINE);
134
- let n = 0;
135
- for (let i = lines.length - 1; i >= 0; i--) {
136
- const line = lines[i];
137
- if (!line.trim()) continue;
138
- let row;
139
- try { row = JSON.parse(line); } catch (_) { continue; } // a torn line costs one row, not the count
140
- const at = Date.parse(row.at);
141
- if (!Number.isFinite(at)) continue;
142
- if (at / 1000 < sinceEpochS) break; // the log is append-ordered: nothing older can count
143
- if (row.event === 'worked') n += 1;
144
- }
145
- return n;
146
- }
147
98
 
148
99
  function record(entry) {
149
100
  const row = { at: new Date().toISOString(), ...entry };
@@ -189,7 +140,12 @@ async function pickTask(goals, deps = {}) {
189
140
 
190
141
  const candidates = Array.from(byId.values()).filter((t) => !seq.TERMINAL.has(t.status));
191
142
  const { order } = seq.topoOrder(candidates);
192
- const { next, skipped } = seq.pickNext(order, { rank, claimableIds });
143
+ // Tasks this process set aside after a refused claim (task 1004396) drop out
144
+ // AFTER ordering, so a dependent of one is still halted by its own unmet
145
+ // dependency rather than promoted by the task's absence.
146
+ const exclude = deps.exclude || null;
147
+ const open = exclude && exclude.size ? order.filter((t) => !exclude.has(String(t.id))) : order;
148
+ const { next, skipped } = seq.pickNext(open, { rank, claimableIds });
193
149
  if (next) return { task: next, goalId, skipped };
194
150
  for (const sk of skipped) allSkipped.push({ ...sk, goal_id: goalId });
195
151
  }
@@ -369,6 +325,80 @@ async function readFence(deps = {}) {
369
325
  return { raw, graderBypassed };
370
326
  }
371
327
 
328
+ // ── what one process remembers between iterations (task 1004396) ────────────
329
+ //
330
+ // Owner ruling, 2026-09-30: a refused claim must not put the runner into waiting
331
+ // by default. It sets THAT task aside and tries the next one. It waits only when
332
+ // there is a block and nothing else it can claim.
333
+ //
334
+ // The two refusals are told apart by claim.js's exit code, because the message is
335
+ // prose and the first cut of this read prose. The run that found this burned
336
+ // ~30 attempts on task 1004254, whose refusal quoted "Your work changes the
337
+ // tools…" — which reads like a per-task protected-path verdict and was in fact
338
+ // the plain-words line of a QUEUE-WIDE gate about a different, stranded task.
339
+ // Setting 1004254 aside would have moved the runner on to the next task, which
340
+ // the same gate refuses too.
341
+ //
342
+ // QUEUE_GATED_EXIT is pinned against claim.js's own value in a test rather than
343
+ // required from it: requiring the CLI would load its whole dependency graph into
344
+ // a process that runs for weeks, to read one integer.
345
+ const QUEUE_GATED_EXIT = 4;
346
+ // How long a task refused ON ITS OWN is left alone. Not forever: most per-task
347
+ // refusals clear (a dependency ships, a claim is released, a blocker resolves),
348
+ // and a process that remembered them for its whole life would shelve work it can
349
+ // do. Half an hour is several wakes at full speed, so a refused head of the queue
350
+ // costs one attempt per half hour instead of one per wake.
351
+ const CLAIM_SET_ASIDE_MS = 30 * 60 * 1000;
352
+ // A bound on claim attempts in ONE pass, so a queue of refusals cannot turn one
353
+ // iteration into dozens of worktree creations.
354
+ const MAX_CLAIM_ATTEMPTS = 5;
355
+ // A limit message with no machine-readable reset (the CLI's human forms name a
356
+ // clock time in a zone we would have to guess). Half an hour is short against a
357
+ // five-hour window and long against a spawn that is cut off in two seconds, so a
358
+ // guess that is wrong costs one wasted spawn per half hour, not one per wake.
359
+ const LIMIT_UNKNOWN_RESET_MS = 30 * 60 * 1000;
360
+
361
+ function newRunnerState() {
362
+ return { setAside: new Map(), queueGate: null, limitUntil: null };
363
+ }
364
+ const RUNNER_STATE = newRunnerState();
365
+
366
+ // The stranded task ids a queue-gated refusal names (" - task 777 Title"). An
367
+ // empty list is a legal answer: the gate is still a gate, the runner simply has
368
+ // nothing to poll and tries one claim per wake instead.
369
+ function owedTaskIds(text) {
370
+ const ids = [];
371
+ const re = /^\s*-\s*task\s+(\d+)/gm;
372
+ let m;
373
+ while ((m = re.exec(String(text || ''))) !== null) ids.push(m[1]);
374
+ return ids;
375
+ }
376
+
377
+ // Does the gate still hold? Asked of the LEDGER, one read per stranded task, and
378
+ // never by attempting a claim: a claim attempt per wake is the loop this replaces.
379
+ // An unreadable task keeps the wait — guessing "cleared" would restart the
380
+ // attempts on a read failure, the opposite of backing off.
381
+ async function queueGateHolds(queueGate, deps = {}) {
382
+ if (!queueGate.owed.length) return false;
383
+ const api = deps.api && deps.api.tasks ? deps.api : await (require('./cli-lib').cliClient)();
384
+ for (const id of queueGate.owed) {
385
+ let r = null;
386
+ try { r = await api.tasks.getTasksId({ id: Number(id) }); } catch (_) { return true; }
387
+ if (!r || !r.ok || !r.data || !r.data.task) return true;
388
+ if (r.data.task.status === 'confirmed') return true;
389
+ }
390
+ return false;
391
+ }
392
+
393
+ function queueGatedRow(queueGate, now) {
394
+ const on = queueGate.owed.length ? queueGate.owed.map((id) => `task ${id}`).join(', ') : 'a stranded confirmed task';
395
+ return {
396
+ event: 'queue_gated', waiting_on: queueGate.owed,
397
+ waited_s: Math.max(0, Math.round((now - queueGate.since) / 1000)),
398
+ reason: `every claim is refused until ${on} lands — waiting, not claiming (${queueGate.reason})`,
399
+ };
400
+ }
401
+
372
402
  async function iteration(opts, deps = {}) {
373
403
  const runCmd = deps.run || run;
374
404
  const spawnWork = deps.spawnWorker || spawnWorker;
@@ -400,88 +430,116 @@ async function iteration(opts, deps = {}) {
400
430
  return log({ event: 'hold', go: false, unknown: !!g.decision.unknown, reason: g.decision.reason, sleep_until: g.decision.sleepUntil ?? null });
401
431
  }
402
432
 
403
- // 1b. A DERIVED gauge buys a BOUNDED amount of work (task 1004178).
404
- //
405
- // `derived` means the last real reading's window had already reset, so the
406
- // baseline is a retired observation rather than a measurement. That is sound
407
- // for the FIVE-hour window, which turns over during a night. It does nothing
408
- // for the SEVEN-day one, which does not — so the weekly pace check, the thing
409
- // the module's notes call what actually binds a heavy week, is simply absent
410
- // on a derived reading. A stale weekly figure cannot substitute: usage only
411
- // grows, so an under-reported value makes the check more permissive, and that
412
- // is the wrong direction to be wrong in.
413
- //
414
- // So the bound is a task count, not a percentage: on one derived window, do
415
- // AUTOBONGOS_DERIVED_TASK_CAP tasks and then hold until a real reading exists.
416
- // A terminal session writes one; the owner opening a terminal in the morning
417
- // is what lifts it. Default 2 — each worker is wall-clock bounded at 90
418
- // minutes, so two of them is well inside a five-hour window even at worst, and
419
- // the cap only has to be small enough that being wrong is survivable.
420
- if (g.decision.derived) {
421
- const cap = Number(process.env.AUTOBONGOS_DERIVED_TASK_CAP ?? 2);
422
- const since = Number(g.decision.derivedSince) || 0;
423
- const done = countWorkedSince(since);
424
- if (Number.isFinite(cap) && cap >= 0 && done >= cap) {
433
+ // 1b. The usage limit, once HIT (task 1004454). Owner ruling 2026-09-30: no
434
+ // fixed cap on how much work a reading buys — the gauge above is read before
435
+ // every task and a derived reading proceeds — and hitting the limit is
436
+ // acceptable, so long as the runner then waits for the reset and resumes by
437
+ // itself. A worker cut off by the limit records when it resets; until then no
438
+ // task is taken, because every worker spawned would be cut off the same way.
439
+ const state = deps.state || RUNNER_STATE;
440
+ const now = deps.now ? deps.now() : Date.now();
441
+ if (state.limitUntil) {
442
+ if (now < state.limitUntil) {
425
443
  return log({
426
- event: 'hold', go: false, unknown: false, derived: true, worked_since_derived: done,
427
- reason: `derived gauge: ${done} task(s) already done since the window reset and the cap is ${cap} — holding for a real reading (a terminal session writes one)`,
428
- sleep_until: null,
444
+ event: 'hold', go: false, unknown: false, usage_limit: true,
445
+ reason: 'the subscription usage limit was hit — sleeping to its reset, then resuming by itself',
446
+ sleep_until: Math.round(state.limitUntil / 1000),
429
447
  });
430
448
  }
449
+ log({ event: 'usage_limit_reset', was_until: new Date(state.limitUntil).toISOString() });
450
+ state.limitUntil = null;
431
451
  }
432
452
 
433
- // 2. Pick — from the ALLOWLISTED goals, never from what the command line asked
434
- // for. gate.goals is already the intersection.
435
- const picked = await pick(gate.goals, deps.pickDeps || {});
436
- if (picked.error) return log({ event: 'pick_failed', reason: picked.error });
437
- if (picked.none) {
438
- return log({
439
- event: 'nothing_claimable', goals: gate.goals,
440
- // Carried so the cadence can tell an EMPTY queue from one where every task
441
- // halts — the reasons are the matrix's own codes, safe to print.
442
- skipped: (picked.skipped || []).map((sk) => ({ id: sk.id, goal_id: sk.goal_id, reasons: (sk.reasons || []).map((r) => r.code) })),
443
- });
444
- }
445
- const task = picked.task;
446
-
447
- // 2b. The per-TASK half of the fence. sequence.js's halt matrix already refuses
448
- // a protected-path task, and this re-asks anyway — not from distrust of the
449
- // matrix, but because the matrix is a shared component with its own callers and
450
- // the fence must not be a property of whoever happens to be picking. A gate that
451
- // only holds while a collaborator keeps calling it is not a gate.
452
- //
453
- // Both halves read the task's DECLARED touches[], which is a claim, not a fact —
454
- // see the header of modules/autonomy/fence.js for why that is acceptable as the
455
- // OUTER of three layers and would not be as the only one.
456
- const taskGate = fence.decideTask({
457
- task, goals: gate.goals, protectedHits: matchProtected(task.touches || []),
458
- });
459
- if (!taskGate.go) {
460
- return log({ event: 'fenced', code: taskGate.code, reason: taskGate.reason, task_id: task.id, goal_id: picked.goalId, globs: taskGate.globs || [] });
453
+ // 1c. A QUEUE-WIDE gate from an earlier pass (task 1004396). While the
454
+ // stranded task it named is still `confirmed`, every claim would be refused, so
455
+ // none is attempted. The wait wakes early when main moves (waitOrJump), which is
456
+ // exactly what the strand landing does.
457
+ if (state.queueGate) {
458
+ if (await queueGateHolds(state.queueGate, deps)) return log(queueGatedRow(state.queueGate, now));
459
+ log({ event: 'queue_gate_cleared', waited_on: state.queueGate.owed, waited_s: Math.max(0, Math.round((now - state.queueGate.since) / 1000)) });
460
+ state.queueGate = null;
461
461
  }
462
+ for (const [id, until] of state.setAside) if (until <= now) state.setAside.delete(id);
463
+
464
+ // 2–4, up to MAX_CLAIM_ATTEMPTS times: pick, fence, tree, claim. A task refused
465
+ // on its own is set aside and the NEXT one is tried in the same pass.
466
+ let task = null; let picked = null; let wtName = null; let wtPath = null;
467
+ let lastRefusal = null;
468
+ for (let attempt = 0; attempt < MAX_CLAIM_ATTEMPTS && !task; attempt++) {
469
+ const exclude = new Set(state.setAside.keys());
470
+ // 2. Pick — from the ALLOWLISTED goals, never from what the command line asked
471
+ // for. gate.goals is already the intersection.
472
+ const got = await pick(gate.goals, { ...(deps.pickDeps || {}), exclude });
473
+ if (got.error) return log({ event: 'pick_failed', reason: got.error });
474
+ // A picker that hands back a set-aside task is treated as having none left:
475
+ // the exclusion is this function's contract, not the picker's courtesy.
476
+ if (got.none || exclude.has(String(got.task.id))) {
477
+ // Everything left was refused in THIS pass: return the last refusal, not
478
+ // "nothing claimable". Every claim failing at once is also what an outage
479
+ // looks like, and it must still reach the failure ladder (task 1004179).
480
+ if (lastRefusal) return lastRefusal;
481
+ return log({
482
+ event: 'nothing_claimable', goals: gate.goals,
483
+ // Carried so the cadence can tell an EMPTY queue from one where every task
484
+ // halts — the reasons are the matrix's own codes, safe to print.
485
+ skipped: (got.skipped || []).map((sk) => ({ id: sk.id, goal_id: sk.goal_id, reasons: (sk.reasons || []).map((r) => r.code) })),
486
+ set_aside: Array.from(state.setAside, ([id, until]) => ({ id, until: new Date(until).toISOString() })),
487
+ });
488
+ }
489
+ const candidate = got.task;
490
+
491
+ // 2b. The per-TASK half of the fence. sequence.js's halt matrix already refuses
492
+ // a protected-path task, and this re-asks anyway — not from distrust of the
493
+ // matrix, but because the matrix is a shared component with its own callers and
494
+ // the fence must not be a property of whoever happens to be picking. A gate that
495
+ // only holds while a collaborator keeps calling it is not a gate.
496
+ //
497
+ // Both halves read the task's DECLARED touches[], which is a claim, not a fact —
498
+ // see the header of modules/autonomy/fence.js for why that is acceptable as the
499
+ // OUTER of three layers and would not be as the only one.
500
+ const taskGate = fence.decideTask({
501
+ task: candidate, goals: gate.goals, protectedHits: matchProtected(candidate.touches || []),
502
+ });
503
+ if (!taskGate.go) {
504
+ return log({ event: 'fenced', code: taskGate.code, reason: taskGate.reason, task_id: candidate.id, goal_id: got.goalId, globs: taskGate.globs || [] });
505
+ }
462
506
 
463
- if (opts.dryRun) return log({ event: 'dry_run', task_id: task.id, title: task.title, goal_id: picked.goalId });
507
+ if (opts.dryRun) return log({ event: 'dry_run', task_id: candidate.id, title: candidate.title, goal_id: got.goalId });
464
508
 
465
- // 3. The task's own working tree.
466
- const wtName = deps.worktreeName ? deps.worktreeName(task.id) : worktreeName(task.id);
467
- const made = await runCmd(process.execPath, [path.join('scripts', 'gds', 'worktree.js'), 'add', wtName, '--base', 'origin/main']);
468
- if (!made.ok) {
469
- return log({ event: 'worktree_failed', task_id: task.id, worktree: wtName, reason: failureReason(made, 'worktree.js') });
470
- }
471
- const wtPath = path.join(REPO_ROOT, '.claude', 'worktrees', wtName);
472
-
473
- // 4. Claim FROM that tree, so claim.js records it as the claim's tree and the
474
- // ship lock later agrees with where the work actually is.
475
- const claimed = await runCmd(process.execPath, [path.join(REPO_ROOT, 'scripts', 'gds', 'claim.js'), String(task.id)], { cwd: wtPath });
476
- if (!claimed.ok) {
477
- await runCmd(process.execPath, [path.join('scripts', 'gds', 'worktree.js'), 'remove', wtName]);
478
- return log({ event: 'claim_failed', task_id: task.id, worktree: wtName, reason: failureReason(claimed, 'claim.js') });
509
+ // 3. The task's own working tree.
510
+ const name = deps.worktreeName ? deps.worktreeName(candidate.id) : worktreeName(candidate.id);
511
+ const made = await runCmd(process.execPath, [path.join('scripts', 'gds', 'worktree.js'), 'add', name, '--base', 'origin/main']);
512
+ if (!made.ok) {
513
+ return log({ event: 'worktree_failed', task_id: candidate.id, worktree: name, reason: failureReason(made, 'worktree.js') });
514
+ }
515
+ const treePath = path.join(REPO_ROOT, '.claude', 'worktrees', name);
516
+
517
+ // 4. Claim FROM that tree, so claim.js records it as the claim's tree and the
518
+ // ship lock later agrees with where the work actually is.
519
+ const claimed = await runCmd(process.execPath, [path.join(REPO_ROOT, 'scripts', 'gds', 'claim.js'), String(candidate.id)], { cwd: treePath });
520
+ if (!claimed.ok) {
521
+ await runCmd(process.execPath, [path.join('scripts', 'gds', 'worktree.js'), 'remove', name]);
522
+ const reason = failureReason(claimed, 'claim.js');
523
+ if (claimed.code === QUEUE_GATED_EXIT) {
524
+ state.queueGate = { owed: owedTaskIds(`${claimed.stderr || ''}\n${claimed.stdout || ''}`), since: now, reason };
525
+ return log({ ...queueGatedRow(state.queueGate, now), task_id: candidate.id, worktree: name });
526
+ }
527
+ state.setAside.set(String(candidate.id), now + CLAIM_SET_ASIDE_MS);
528
+ lastRefusal = log({ event: 'claim_failed', task_id: candidate.id, worktree: name, reason, set_aside_s: CLAIM_SET_ASIDE_MS / 1000 });
529
+ continue;
530
+ }
531
+ task = candidate; picked = got; wtName = name; wtPath = treePath;
479
532
  }
533
+ if (!task) return lastRefusal;
480
534
 
481
535
  // 5. Work, in that tree.
482
536
  const started = Date.now();
483
537
  const res = await spawnWork({ task, model: opts.model, timeoutMs: opts.timeoutMs, cwd: wtPath, deps });
484
538
  const verdict = classifyRun(res);
539
+ // Cut off by the usage limit? Recorded here so the NEXT iteration sleeps to the
540
+ // reset instead of spawning a worker that will be cut off the same way.
541
+ const limit = verdict.outcome === 'claims_shipped' ? null : usageLimitHit({ envelope: res.envelope, stderr: res.stderr });
542
+ if (limit) state.limitUntil = limit.resetAt ? limit.resetAt * 1000 : now + LIMIT_UNKNOWN_RESET_MS;
485
543
 
486
544
  // 6. VERIFY FROM THE LEDGER (task 1003903). Read the task back from Bongos and
487
545
  // let IT say what happened — for every outcome, not only a claimed ship. The
@@ -513,6 +571,7 @@ async function iteration(opts, deps = {}) {
513
571
  session_id: res.sessionId || null, duration_s: Math.round((Date.now() - started) / 1000),
514
572
  cost_usd: verdict.costUsd ?? null,
515
573
  output_truncated: !!res.truncated,
574
+ ...(limit ? { usage_limit: { reset_at: limit.resetAt } } : {}),
516
575
  undetermined_decisions: verdict.verdict ? verdict.verdict.undetermined_decisions : null,
517
576
  // What the LEDGER says, kept separate from what the worker said, so the run
518
577
  // log can be read afterwards without having to trust either one alone.
@@ -912,6 +971,15 @@ async function forever(opts, deps = {}) {
912
971
  // corpse.
913
972
  await beat({ mode: state.mode, last_event: row.event, working_task_id: Number(row.task_id) || undefined });
914
973
  const seen = cadence.classifyEvent(row);
974
+ // A hold that names its reset SECOND waits until that second (task 1004454).
975
+ // classifyEvent carries it as an absolute epoch; the cadence has no clock of
976
+ // its own, so the conversion to a wait lives here beside the one clock read.
977
+ // Without it every hold waited the fixed idle and the "wake at the reset"
978
+ // the goal asks for was only ever as accurate as the poll.
979
+ if (Number.isFinite(Number(seen.sleepUntil)) && seen.sleepUntil !== null) {
980
+ const nowS = Math.floor((deps.now ? deps.now() : Date.now()) / 1000);
981
+ seen.sleepUntilS = Math.max(0, Number(seen.sleepUntil) - nowS);
982
+ }
915
983
  state = cadence.nextCadence(state, seen, opts.cadence);
916
984
  log({ event: 'cadence', mode: state.mode, consecutive_failures: state.consecutiveFailures, wait_s: state.waitS, class: seen.class, why: state.why });
917
985
  // A burst continues only while work is actually landing. Anything else ends
@@ -972,7 +1040,8 @@ if (require.main === module) {
972
1040
 
973
1041
  module.exports = {
974
1042
  readArgv, pickTask, iteration, record, logPath, spawnWorker, workerEnv, WORKER_ENV_ALLOW, resolveWorkerBin,
975
- worktreeName, readFence, reconcileLeftoverClaims, pollSignals, waitOrJump, forever, CLAIM_PREFIX, countWorkedSince, codeDrifted, loadedHead, UPGRADE_EXIT_CODE, MIN_UPTIME_BEFORE_UPGRADE_MS,
1043
+ worktreeName, readFence, reconcileLeftoverClaims, pollSignals, waitOrJump, forever, CLAIM_PREFIX, codeDrifted, loadedHead, UPGRADE_EXIT_CODE, MIN_UPTIME_BEFORE_UPGRADE_MS,
976
1044
  heartbeatPath, writeHeartbeat, readHeartbeat, pidAlive, anotherRunnerIsAlive, HEARTBEAT_STALE_MS,
977
1045
  isRunnerTree, RUNNER_TREE_RE, postHeartbeat, RUNNER_STARTED_AT, failureReason,
1046
+ newRunnerState, LIMIT_UNKNOWN_RESET_MS, QUEUE_GATED_EXIT, CLAIM_SET_ASIDE_MS, MAX_CLAIM_ATTEMPTS, owedTaskIds,
978
1047
  };
@@ -282,6 +282,14 @@ function claimErrorInfo(r) {
282
282
  // control characters that repaint a terminal, and a runaway one can bury the
283
283
  // guidance it was meant to add. Clip to one line, strip C0/C1 controls, cap length.
284
284
  const MAX_LISTED_TASKS = 20;
285
+
286
+ // QUEUE_GATED_EXIT — the one refusal that is about the QUEUE, not the task
287
+ // (task 1004396). REBASE_REQUIRED refuses every claim until a stranded confirmed
288
+ // task lands, so an unattended caller that reads it as "this task is refused"
289
+ // sets aside task after task for a fault none of them has. Every other refusal
290
+ // keeps exit 1. The autobongos runner is the consumer and pins the same value
291
+ // in a test.
292
+ const QUEUE_GATED_EXIT = 4;
285
293
  function clip(s, max = 160) {
286
294
  if (s == null) return '';
287
295
  // eslint-disable-next-line no-control-regex
@@ -370,7 +378,7 @@ function claimFailureGuidance({ code, message, d }, r, taskId, ctx) {
370
378
  if (t.reason) lines.push(` flagged: ${clip(t.reason)}`);
371
379
  }
372
380
  lines.push(` ${clip(d.hint, 600) || 'Clear it with /strand-fix N, or rebase the branch and re-ship — or /builder-release it to give up on one.'}`);
373
- return { lines, exit: 1 };
381
+ return { lines, exit: QUEUE_GATED_EXIT };
374
382
  }
375
383
  if (code === 'ALREADY_CLAIMED') {
376
384
  lines.push(`Task #${taskId} was just claimed by another session. Try a different task.`);
@@ -771,4 +779,4 @@ if (require.main === module) {
771
779
  });
772
780
  }
773
781
 
774
- module.exports = { formatClaimFailure, claimErrorInfo, goalMembershipOfferPlan, materializeModeSkills, printRoleLine, worktreeDetectionNote };
782
+ module.exports = { formatClaimFailure, claimErrorInfo, goalMembershipOfferPlan, materializeModeSkills, printRoleLine, worktreeDetectionNote, QUEUE_GATED_EXIT };
@@ -0,0 +1,141 @@
1
+ 'use strict';
2
+ //
3
+ // scripts/gds/move-escalation.js — a core move nobody can heal reaches a person: one blocker
4
+ // and one Discord notice per failed move, closed by itself when a later move lands (task
5
+ // 1004449, goal 1000090 deploy self-heal).
6
+ //
7
+ // WHY. After task 1004448 a failed /deploy move ends one of three ways: it waits and retries
8
+ // (transient), it is fixed and retried (known_wedge), or it is retired. A retired
9
+ // needs_decision move is the owner's own rule refusing it, and the deploy page says so. But a
10
+ // move retired as `unknown` — an unlisted failure, a fix that did not hold, and every
11
+ // rollback that FAILED (the project may be down) — only wrote a line on the intent, which
12
+ // nobody reads until they open the page. That is exactly the failure that needs a person.
13
+ //
14
+ // EDGE-TRIGGERED, WITHOUT NEW STATE. The blocker's (source, source_ref) is unique, and
15
+ // source_ref names the INTENT: `core-move:<instance id>:<intent id>`. An intent retires once,
16
+ // so a failed move files one blocker however many retries led up to it, and a re-run of the
17
+ // same retire files nothing. The Discord notice is posted only when the insert actually
18
+ // created the row, so it is exactly as edge-triggered as the blocker.
19
+ //
20
+ // CLOSED BY SUCCESS. A later core move on the same project that lands resolves every open
21
+ // blocker this file filed for that project, with a note saying which move fixed it, and
22
+ // posts one all-clear. A person can still resolve one by hand; this never reopens it.
23
+ //
24
+ // The Discord copy names a project only when it is public on the platform's map (the rule
25
+ // liveness-notify.js keeps), and never carries upgrade output — that stays in the blocker,
26
+ // inside the trust boundary. Writes go through the runner's own db handle (the hub
27
+ // database, where the blockers table lives), not modules/ideas/blockers.js's shared pool,
28
+ // which this CLI process never opens; the INSERT is that file's createBlocker, verbatim.
29
+
30
+ const { execFile } = require('child_process');
31
+ const { REPO_ROOT } = require('./provision-config.js');
32
+
33
+ const BLOCKER_SOURCE = 'provisioning';
34
+ const REF_PREFIX = 'core-move';
35
+ const DETAIL_MAX = 1500;
36
+
37
+ /** Does this retired move need a person? PURE. */
38
+ function needsPerson({ intent, failure, terminal }) {
39
+ if (!intent || intent.action !== 'core-upgrade') return false;
40
+ return !!terminal || !!(failure && (failure.class === 'unknown' || failure.reason === 'rollback_failed'));
41
+ }
42
+
43
+ function sourceRef(inst, intent) { return `${REF_PREFIX}:${inst.id}:${intent.id}`; }
44
+
45
+ function projectName(inst, provisioning) {
46
+ try { return provisioning.effectiveSettings(inst).visibility === 'public' ? inst.slug : null; } catch { return null; }
47
+ }
48
+
49
+ /** The blocker a failed move files. PURE. */
50
+ function blockerFor({ inst, intent, failure, error }) {
51
+ const reason = (failure && failure.reason) || 'unknown';
52
+ const down = reason === 'rollback_failed';
53
+ const to = intent.target_version || '(no target)';
54
+ const detail = String(error || '').slice(0, DETAIL_MAX).replace(/```/g, "'''");
55
+ return {
56
+ title: down
57
+ ? `${inst.slug}: the core move to ${to} failed AND its rollback did not finish — the project may be down`
58
+ : `${inst.slug}: the core move to ${to} failed and the runner cannot fix it`,
59
+ bodyMd: [
60
+ `**Project:** ${inst.slug} (instance ${inst.id}) · **target:** ${to} · **reason:** \`${reason}\` · **move:** intent ${intent.id}`,
61
+ '',
62
+ down
63
+ ? 'The move changed the project, failed, and the automatic rollback also failed. It may be down or on a half-installed core. Check it before anything else.'
64
+ : 'The runner retired this move: it is not a known snag it can fix, or its one automatic fix did not work. Nothing is being retried.',
65
+ '',
66
+ 'What the move reported:',
67
+ '```',
68
+ detail || '(nothing)',
69
+ '```',
70
+ '',
71
+ 'This closes by itself when a later core move on this project succeeds.',
72
+ ].join('\n'),
73
+ source: BLOCKER_SOURCE,
74
+ sourceRef: sourceRef(inst, intent),
75
+ };
76
+ }
77
+
78
+ /** The owner-facing Discord line. `name` null → described, never identified. PURE. */
79
+ function noticeText({ kind, name, to, down }) {
80
+ const who = name ? `**${name}**` : 'A private project';
81
+ if (kind === 'fixed') return `✅ ${who} took its core update${to ? ` to ${to}` : ''}; the earlier failed update is closed.`;
82
+ return down
83
+ ? `🚨 ${who}: a core update failed and could not be undone. The project may be down. A blocker is open with the details.`
84
+ : `⚠️ ${who}: a core update${to ? ` to ${to}` : ''} failed in a way the platform cannot fix by itself. A blocker is open with the details. Nothing is being retried.`;
85
+ }
86
+
87
+ // Fire-and-forget, the liveness-notify.js postNotice shape: CHAT_DRY_RUN=1 short-circuits and
88
+ // a Discord hiccup never stalls the drain.
89
+ function postNotice(text, { log = console.log } = {}) {
90
+ if (!text) return;
91
+ if (process.env.CHAT_DRY_RUN === '1') { log(`[core-move] [broadcast dry-run] would post: ${text}`); return; }
92
+ try {
93
+ execFile('node', ['scripts/discord/post.js', text], { cwd: REPO_ROOT, timeout: 10_000 }, (err) => {
94
+ if (err) log(`[core-move] notice broadcast failed (non-blocking): ${err.message}`);
95
+ });
96
+ } catch (e) { log(`[core-move] notice broadcast failed (non-blocking): ${(e && e.message) || e}`); }
97
+ }
98
+
99
+ /** After a move is retired: file its blocker and notice, once. Never throws. */
100
+ async function afterRetire({ intent, inst, result, deps }) {
101
+ const failure = result && result.failure;
102
+ if (!deps.apply || !needsPerson({ intent, failure, terminal: result && result.terminal })) return { filed: false };
103
+ const b = blockerFor({ inst, intent, failure, error: result.error });
104
+ try {
105
+ const { rows } = await deps.db.query(
106
+ `INSERT INTO blockers (title, body_md, source, source_ref)
107
+ VALUES ($1, $2, $3, $4)
108
+ ON CONFLICT (source, source_ref) WHERE source IS NOT NULL AND source_ref IS NOT NULL
109
+ DO NOTHING
110
+ RETURNING id`, [b.title, b.bodyMd, b.source, b.sourceRef]);
111
+ if (!rows || !rows[0]) return { filed: false };
112
+ const down = (failure && failure.reason) === 'rollback_failed';
113
+ (deps.postNotice || postNotice)(noticeText({ kind: 'failed', name: projectName(inst, deps.provisioning), to: intent.target_version, down }), { log: deps.log });
114
+ (deps.log || (() => {}))(` ⚑ filed blocker ${rows[0].id} for ${inst.slug}'s failed move to ${intent.target_version}`);
115
+ return { filed: true, blockerId: rows[0].id };
116
+ } catch (e) {
117
+ (deps.log || (() => {}))(` ⚠ could not file a blocker for ${inst.slug}'s failed move: ${(e && e.message) || e}`);
118
+ return { filed: false };
119
+ }
120
+ }
121
+
122
+ /** After a move LANDS: close every open blocker this file filed for the project. Never throws. */
123
+ async function afterSuccess({ intent, inst, result, deps }) {
124
+ if (!deps.apply || !intent || intent.action !== 'core-upgrade' || !result || result.ok === false || result.noop) return { closed: 0 };
125
+ const to = result.served || intent.target_version;
126
+ try {
127
+ const { rows } = await deps.db.query(
128
+ `UPDATE blockers SET status = 'resolved', resolved_at = now(), resolution_note = $3
129
+ WHERE status = 'open' AND source = $1 AND source_ref LIKE $2
130
+ RETURNING id`,
131
+ [BLOCKER_SOURCE, `${REF_PREFIX}:${inst.id}:%`, `closed automatically: a later core move to ${to} succeeded (intent ${intent.id})`]);
132
+ const n = (rows || []).length;
133
+ if (n) (deps.postNotice || postNotice)(noticeText({ kind: 'fixed', name: projectName(inst, deps.provisioning), to }), { log: deps.log });
134
+ return { closed: n };
135
+ } catch (e) {
136
+ (deps.log || (() => {}))(` ⚠ could not close ${inst.slug}'s core-move blockers: ${(e && e.message) || e}`);
137
+ return { closed: 0 };
138
+ }
139
+ }
140
+
141
+ module.exports = { BLOCKER_SOURCE, REF_PREFIX, needsPerson, sourceRef, blockerFor, noticeText, postNotice, afterRetire, afterSuccess };
@@ -1209,12 +1209,13 @@ async function cmdRunIntents(deps) {
1209
1209
  // can say why the work stopped.
1210
1210
  await provisioning.setInstanceStatus(db, intent.instance_id, inst.status, { error_note: result.error || 'op failed', error_note_action: intent.action }).catch(() => {});
1211
1211
  if (result.failure) await recordFailureClass(db, intent.id, result.failure); // WHAT KIND of failure, beside the sentence (task 1004446)
1212
+ await require('./move-escalation.js').afterRetire({ intent, inst, result, deps }); // nobody can heal it: one blocker + one notice per move (task 1004449)
1212
1213
  } else {
1213
1214
  await db.query(`UPDATE provisioning_intents SET state='pending', last_error=$2, not_before = now() + ($3 * interval '1 millisecond'), updated_at=now() WHERE id=$1`, [intent.id, result.error || 'op failed', next.delayMs]);
1214
1215
  if (result.failure) await recordFailureClass(db, intent.id, result.failure);
1215
1216
  }
1216
1217
  errors++;
1217
- } else { await provisioning.resolveIntent(db, intent.id, 'done', (result && result.doneNote) || null); ok++; } // done WITH a note: it landed, and this is what it could not finish (task 1004065)
1218
+ } else { await provisioning.resolveIntent(db, intent.id, 'done', (result && result.doneNote) || null); ok++; await require('./move-escalation.js').afterSuccess({ intent, inst, result, deps }); } // done WITH a note: it landed, and this is what it could not finish (task 1004065)
1218
1219
  } catch (e) {
1219
1220
  await provisioning.resolveIntent(db, intent.id, 'error', e.message).catch(() => {});
1220
1221
  // Surface the failure on the instance too, so a stuck instance is visible.
package/src/module-api.js CHANGED
@@ -75,7 +75,7 @@ const { responsibilityFor, ROLE_RESPONSIBILITIES } = require('./role-responsibil
75
75
  // MAJOR (see allowBoxScope below): passes the request through untouched.
76
76
  function deprecatedNoopMiddleware(_req, _res, next) { next(); }
77
77
 
78
- const CORE_VERSION = '1.20.32'; // CI auto-patch carrier (ADR 0161); changelog: docs/module-api-changelog.md
78
+ const CORE_VERSION = '1.20.34'; // CI auto-patch carrier (ADR 0161); changelog: docs/module-api-changelog.md
79
79
 
80
80
  // A namespaced logger so a module's log lines are attributable + consistent.
81
81
  // Usage: const log = api.logger('discord'); log.info('mounted');
@@ -40,6 +40,15 @@ const walk = (classes, cfg) => {
40
40
 
41
41
  // ── classification ──────────────────────────────────────────────────────────────
42
42
 
43
+ test('a queue-wide gate is a wait, not a failure (task 1004396)', () => {
44
+ // The strand behind it is somebody's unlanded ship, not this machine breaking.
45
+ // Escalating it would slow the runner down for a fault it cannot fix, and the
46
+ // wait already wakes early when main moves, which is what the strand landing does.
47
+ const seen = classifyEvent({ event: 'queue_gated', waiting_on: ['777'], reason: 'x' });
48
+ assert.equal(seen.class, 'quiet');
49
+ assert.match(seen.detail, /777/);
50
+ });
51
+
43
52
  test('a worked task counts as progress only when the LEDGER agreed', () => {
44
53
  assert.equal(classifyEvent({ event: 'worked', verified: true }).class, 'progress');
45
54
  // A worker's claim the ledger never saw is not progress; counting it would hold
@@ -284,6 +293,33 @@ function loopHarness({ rows, maxLoops = 6 }) {
284
293
  return { deps, waits: emitted, recorded, opts: { goals: [], maxTasks: 3, maxLoops } };
285
294
  }
286
295
 
296
+ // ── the usage limit: a wait to a known second, not a failure (task 1004454) ──
297
+
298
+ test('a worker cut off by the usage limit is quiet, not a failed launch', () => {
299
+ // It exits in seconds having spent nothing — exactly workerNeverRan()'s shape —
300
+ // but the machine is fine: the window is spent. Escalating it would put the
301
+ // runner into probing for a limit the owner has said is acceptable to hit.
302
+ const seen = classifyEvent({ event: 'worked', outcome: 'worker_failed', duration_s: 2, cost_usd: null, reason: 'x', usage_limit: { reset_at: 1790300000 } });
303
+ assert.equal(seen.class, 'quiet');
304
+ assert.equal(seen.sleepUntil, 1790300000);
305
+ });
306
+
307
+ test('a hold with a reset time makes the loop wait until THAT second, not a fixed idle', async () => {
308
+ const nowS = 1_790_290_000;
309
+ const h = loopHarness({ rows: [{ event: 'hold', go: false, reason: 'five-hour burn at the ceiling', sleep_until: nowS + 3600 }], maxLoops: 1 });
310
+ h.deps.now = () => nowS * 1000;
311
+ await runner.forever(h.opts, h.deps);
312
+ assert.equal(h.waits[0].waitS, 3600, 'the wake is the reset second the gauge named');
313
+ });
314
+
315
+ test('a hold whose reset has already passed waits no time at all', async () => {
316
+ const nowS = 1_790_290_000;
317
+ const h = loopHarness({ rows: [{ event: 'hold', go: false, reason: 'x', sleep_until: nowS - 5 }], maxLoops: 1 });
318
+ h.deps.now = () => nowS * 1000;
319
+ await runner.forever(h.opts, h.deps);
320
+ assert.equal(h.waits[0].waitS, 0);
321
+ });
322
+
287
323
  test('PULLING THE NETWORK: the loop backs off and never returns', async () => {
288
324
  const h = loopHarness({ rows: [{ event: 'pick_failed', reason: 'fetch failed ECONNREFUSED' }] });
289
325
  await runner.forever(h.opts, h.deps);