@bongos/core 1.21.4 → 1.21.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (126) hide show
  1. package/.bongos-core.json +233 -108
  2. package/clients/bongos-client/README.md +1 -1
  3. package/clients/bongos-client/bongos-client.global.js +2 -0
  4. package/clients/bongos-client/index.cjs +2 -0
  5. package/clients/bongos-client/index.d.ts +3 -1
  6. package/clients/bongos-client/index.mjs +2 -0
  7. package/docs/adr/0361-merges-publish-as-candidates-release-decides-what-others-are-offered.md +1 -0
  8. package/docs/adr/0362-the-governor-docket-is-collected-from-contributed-sources.md +34 -0
  9. package/docs/adr/README.md +1 -0
  10. package/docs/api/openapi.json +37 -2
  11. package/docs/api-reference.md +4 -3
  12. package/docs/copy-inventory.md +541 -658
  13. package/docs/copy-registry.json +1205 -2340
  14. package/docs/module-api-changelog.md +4 -0
  15. package/docs/onboarding/diagrams/03-drachmae-karma.mmd +1 -1
  16. package/docs/onboarding/diagrams/assertions.json +1 -1
  17. package/docs/page-inventory.json +47 -28
  18. package/docs/page-readings.json +1128 -1339
  19. package/modules/autonomy/config-idle.js +192 -0
  20. package/modules/autonomy/governor-docket.js +101 -0
  21. package/modules/autonomy/routes/autonomy.js +27 -0
  22. package/modules/autonomy/runner-health.js +22 -1
  23. package/modules/discord/board-broadcast.js +3 -1
  24. package/modules/discord/craft-broadcast.js +107 -0
  25. package/modules/discord/routes/discord.js +10 -0
  26. package/modules/discord/ship-broadcast.js +30 -2
  27. package/modules/economy/reward.js +3 -0
  28. package/modules/government/board.js +20 -5
  29. package/modules/government/docket-vocab.js +82 -0
  30. package/modules/government/docket.js +458 -0
  31. package/modules/government/routes/government.js +34 -0
  32. package/modules/hall-ui/public/city-draw.js +438 -0
  33. package/modules/hall-ui/public/city.css +110 -0
  34. package/modules/hall-ui/public/city.html +71 -0
  35. package/modules/hall-ui/public/city.js +177 -0
  36. package/modules/hall-ui/public/city.states.json +30 -0
  37. package/modules/hall-ui/public/gate.js +11 -4
  38. package/modules/hall-ui/public/genesis-home.js +9 -10
  39. package/modules/hall-ui/public/profile.css +4 -0
  40. package/modules/hall-ui/public/profile.js +25 -1
  41. package/modules/hall-ui/public/settings-autobongos.js +19 -1
  42. package/modules/hall-ui/public/settings.css +26 -0
  43. package/modules/hall-ui/public/settings.html +8 -0
  44. package/modules/hall-ui/public/settings.js +43 -3
  45. package/modules/ideas/governor-docket.js +93 -0
  46. package/modules/ideas/routes/blockers.js +5 -0
  47. package/modules/lifecycle/governor-docket.js +192 -0
  48. package/modules/lifecycle/routes/gate-approvals.js +78 -33
  49. package/modules/lifecycle/routes/lifecycle.js +3 -0
  50. package/modules/onboarding/routes/access-requests.js +68 -27
  51. package/modules/platform-identity/craft-rollup.js +72 -0
  52. package/modules/platform-identity/migrations/platform_identity_029_activity_crafts.sql +23 -0
  53. package/modules/platform-identity/platform-identity.js +12 -6
  54. package/modules/platform-identity/routes/sso.js +4 -0
  55. package/modules/platform-identity/tests/platform-identity.mjs +1 -1
  56. package/modules/provisioning/core-upgrade.js +34 -7
  57. package/modules/provisioning/module.json +2 -1
  58. package/modules/provisioning/seams.js +2 -0
  59. package/modules/public-landing/public/assets/cosmos.css +6 -50
  60. package/modules/public-landing/public/projects.html +1170 -1659
  61. package/modules/public-landing/public/projects.probes.json +18 -18
  62. package/modules/public-landing/public/projects.states.json +24 -31
  63. package/modules/security/routes/reports.js +39 -0
  64. package/package-lock.json +2 -2
  65. package/package.json +1 -1
  66. package/release-notes.json +72 -0
  67. package/scripts/gds/autobongos-grade-cap.js +187 -0
  68. package/scripts/gds/autobongos-loop.js +4 -0
  69. package/scripts/gds/autobongos-run.js +154 -13
  70. package/scripts/gds/autobongos-verify.js +211 -14
  71. package/scripts/gds/provision-core-upgrade.js +14 -2
  72. package/scripts/gds/provision-repo.js +77 -13
  73. package/scripts/gds/ship-flow.js +5 -0
  74. package/scripts/gds/ship-preflight-steps.js +2 -1
  75. package/scripts/gds/ship.js +10 -0
  76. package/scripts/gds/update-sweep.js +24 -8
  77. package/scripts/gds/upgrade-outcome.js +1 -0
  78. package/src/bongos/auth-admission.js +95 -13
  79. package/src/bongos/db-kernel.js +1 -1
  80. package/src/bongos/module-rechecks.js +77 -0
  81. package/src/bongos/route-rank-check.js +3 -0
  82. package/src/bongos/routes/core-update.js +68 -2
  83. package/src/bongos/routes/modules.js +17 -0
  84. package/src/bongos/serve-internal.js +14 -0
  85. package/src/bongos/software-update.js +3 -2
  86. package/src/module-api.js +1 -1
  87. package/tests/activity_rollup_order_db.mjs +3 -1
  88. package/tests/activity_snapshots_db.mjs +2 -0
  89. package/tests/autobongos_grade_cap.mjs +125 -0
  90. package/tests/autobongos_loop.mjs +230 -1
  91. package/tests/autobongos_verify.mjs +239 -3
  92. package/tests/autonomy_config_idle.mjs +272 -0
  93. package/tests/core_update_banner.mjs +29 -0
  94. package/tests/core_upgrade_door.mjs +31 -1
  95. package/tests/core_upgrade_runner.mjs +20 -0
  96. package/tests/discord_craft_broadcast.mjs +162 -0
  97. package/tests/genesis_home.mjs +1 -1
  98. package/tests/government_routes.mjs +3 -1
  99. package/tests/governor_docket.mjs +552 -0
  100. package/tests/hall_audit.mjs +10 -1
  101. package/tests/hall_city.mjs +369 -0
  102. package/tests/hall_page_gate_map.mjs +5 -0
  103. package/tests/hub_craft_rollup.mjs +350 -0
  104. package/tests/onboard.mjs +10 -4
  105. package/tests/platform_boot.mjs +1 -1
  106. package/tests/profile_rollup_consent.mjs +2 -0
  107. package/tests/profile_route.mjs +16 -2
  108. package/tests/projects_hub.mjs +196 -89
  109. package/tests/projects_hub_app_status.mjs +3 -1
  110. package/tests/projects_hub_app_step.mjs +72 -48
  111. package/tests/projects_hub_dns_ready.mjs +10 -2
  112. package/tests/projects_hub_look.mjs +32 -22
  113. package/tests/projects_hub_module_picker.mjs +112 -62
  114. package/tests/projects_hub_pre_uat.mjs +48 -41
  115. package/tests/provision.mjs +78 -0
  116. package/tests/public_landing_projects.mjs +12 -6
  117. package/tests/settings_main_role.mjs +190 -0
  118. package/tests/software_update.mjs +56 -1
  119. package/tests/update_subscription_engine.mjs +55 -3
  120. package/tests/wizard_demo.mjs +37 -11
  121. package/tests/wizard_draft_resume.mjs +34 -22
  122. package/tests/wizard_front_door.mjs +173 -103
  123. package/tests/wizard_intent_resume.mjs +28 -25
  124. package/tests/wizard_physics.mjs +94 -76
  125. package/tests/wizard_physics_more.mjs +428 -277
  126. package/tests/wizard_six_screens.mjs +126 -0
@@ -0,0 +1,187 @@
1
+ 'use strict';
2
+
3
+ // scripts/gds/autobongos-grade-cap.js — the grade cap for ONE autonomous run
4
+ // (task 1004536).
5
+ //
6
+ // THE RULE (owner, 2026-10-02): a task that fails the grade twice goes to the
7
+ // owner instead of being retried. In one worker run a task gets its first grade
8
+ // plus at most ONE re-grade. Evidence for why: grader-refusal loops were 48% of
9
+ // all Autobongos spend for zero ships — task 1004512 cost $63.81 over three grade
10
+ // rounds, task 1003564 $13.74. The worker prompt already said "return
11
+ // cannot_complete if the grader refuses", and the worker ignored it. A prompt
12
+ // line is not a limit.
13
+ //
14
+ // WHERE IT IS ENFORCED — one boundary, one courtesy:
15
+ // 1. THE BOUNDARY is the supervisor (autobongos-run.js), and its count comes
16
+ // from the SERVER: every graded round is POSTed to /tasks/:id/grade and
17
+ // appended to `grade_attempts` (an append-only table the worker has no write
18
+ // path to). The supervisor notes the newest attempt before spawning, counts
19
+ // the rounds after it (roundsFromAttempts), kills a worker still running
20
+ // once two graded FAILs exist, and ends the run as needs_human whatever the
21
+ // worker's verdict says. Nothing on the worker's machine — env, files —
22
+ // feeds this count, so a worker cannot erase it. (A round the worker never
23
+ // POSTs cannot advance the task either, so it is not a way around the cap.)
24
+ // 2. THE COURTESY is ship.js refusing to START a third grade (checkGradeCap at
25
+ // its entry, before any flag is read). The worker's own process holds that
26
+ // check's inputs — a run ledger named in env — so it is NOT the boundary; a
27
+ // worker could clear the variable. It exists because it is free and it stops
28
+ // an honest-but-stubborn worker before a third panel is paid for, rather
29
+ // than up to one supervisor poll later.
30
+ //
31
+ // The local ledger is ACTIVE ONLY when AUTOBONGOS_GRADE_LEDGER is set, which only
32
+ // the supervisor does. A human running ship.js by hand is never capped.
33
+ //
34
+ // A GRADER OUTAGE DOES NOT COUNT. `signals.panel_outcome === 'unavailable'` means
35
+ // the panel could not run (#943) — no findings, nothing to fix, not a quality
36
+ // verdict. Counting it would hand a task to the owner because the grader was down.
37
+
38
+ const fs = require('fs');
39
+ const path = require('path');
40
+
41
+ const LEDGER_ENV = 'AUTOBONGOS_GRADE_LEDGER';
42
+ // The first grade + ONE re-grade. Not configurable on purpose: a cap the worker
43
+ // could raise through env would be a cap it could talk its way past.
44
+ const GRADE_CAP = 2;
45
+ // ship.js's exit code for a capped run. Distinct from 1 (generic) and 2 (usage /
46
+ // refused input) so a supervisor or a test can tell the cap from any other stop.
47
+ const REFUSED_EXIT = 6;
48
+
49
+ function ledgerPathFromEnv(env = process.env) {
50
+ const p = env[LEDGER_ENV];
51
+ return p && String(p).trim() ? String(p).trim() : null;
52
+ }
53
+
54
+ // isOutage — the grader never ran. Mirrors the #943 test the ship card and
55
+ // ship-grade-steps use (signals.panel_outcome === 'unavailable').
56
+ function isOutage(grade) {
57
+ return !!(grade && grade.signals && grade.signals.panel_outcome === 'unavailable');
58
+ }
59
+
60
+ // readLedger — every row, oldest first. A missing or unreadable file is an empty
61
+ // ledger; a malformed line is skipped rather than fatal.
62
+ function readLedger(file) {
63
+ if (!file) return [];
64
+ let raw = '';
65
+ try { raw = fs.readFileSync(file, 'utf8'); } catch (_) { return []; }
66
+ const rows = [];
67
+ for (const line of raw.split(/\r?\n/)) {
68
+ if (!line.trim()) continue;
69
+ try { rows.push(JSON.parse(line)); } catch (_) { /* skip a torn line */ }
70
+ }
71
+ return rows;
72
+ }
73
+
74
+ // gradeCapState — pure over ledger rows: how many GRADED rounds (outages
75
+ // excluded) this task has had in the run, and whether the cap is reached.
76
+ function gradeCapState(rows, taskId) {
77
+ const mine = (rows || []).filter((r) => r && String(r.task_id) === String(taskId));
78
+ const graded = mine.filter((r) => !r.outage);
79
+ const fails = graded.filter((r) => r.passed !== true);
80
+ const passed = graded.some((r) => r.passed === true);
81
+ return {
82
+ graded: graded.length,
83
+ fails: fails.length,
84
+ outages: mine.length - graded.length,
85
+ passed,
86
+ capped: graded.length >= GRADE_CAP,
87
+ // The run should be handed to the owner: the cap is reached and none passed.
88
+ exhausted: !passed && fails.length >= GRADE_CAP,
89
+ };
90
+ }
91
+
92
+ // checkGradeCap — may ship.js start a grade for this task? `{ ok: true }` when no
93
+ // ledger is in play (a human ship) or the cap has room.
94
+ function checkGradeCap({ taskId, env = process.env } = {}) {
95
+ const file = ledgerPathFromEnv(env);
96
+ if (!file) return { ok: true, active: false };
97
+ const state = gradeCapState(readLedger(file), taskId);
98
+ // `exhausted`, not `capped`: a task whose second round PASSED still needs
99
+ // ship.js to resume its merge leg, and that must not be refused.
100
+ if (!state.exhausted) return { ok: true, active: true, state };
101
+ return {
102
+ ok: false, active: true, state,
103
+ lines: [
104
+ `Ship refused: task ${taskId} has already been graded ${state.graded} time(s) in this autonomous run — the cap is the first grade plus ONE re-grade (task 1004536).`,
105
+ 'A task that fails the grade twice goes to the owner instead of being retried.',
106
+ 'Stop now and return cannot_complete, quoting the grader\'s findings — the supervisor files the blocker.',
107
+ '(A GRADER UNAVAILABLE outcome is an outage and was not counted.)',
108
+ ],
109
+ };
110
+ }
111
+
112
+ // recordGradeOutcome — append one graded round. Best-effort: a write failure
113
+ // must never fail the ship that just graded.
114
+ function recordGradeOutcome({ taskId, grade, env = process.env, now = () => new Date().toISOString() } = {}) {
115
+ const file = ledgerPathFromEnv(env);
116
+ if (!file || !grade) return false;
117
+ const row = {
118
+ task_id: Number(taskId),
119
+ at: now(),
120
+ passed: grade.passed === true,
121
+ outage: isOutage(grade),
122
+ score: grade.score ?? null,
123
+ };
124
+ try {
125
+ fs.mkdirSync(path.dirname(file), { recursive: true });
126
+ fs.appendFileSync(file, `${JSON.stringify(row)}\n`);
127
+ return true;
128
+ } catch (_) { return false; }
129
+ }
130
+
131
+ // ── the server's count (the boundary) ───────────────────────────────────────────
132
+ //
133
+ // `GET /tasks/:id?include=grade,grade_attempts` serves each recorded round as
134
+ // { attempted_at, passed, workers: [{ kind, verdict, error, severities }] }.
135
+ // The grader's own outage test (#943, grader.js) is "every GATING worker failed
136
+ // to run"; the Narc is advisory (#481), so it is left out of that test the same
137
+ // way. A round with no gating workers listed is not called an outage — it is
138
+ // counted, because an unknown shape must not silently buy another round.
139
+ const ADVISORY_KINDS = new Set(['narc']);
140
+ function attemptIsOutage(attempt) {
141
+ const ws = Array.isArray(attempt && attempt.workers) ? attempt.workers : [];
142
+ const gating = ws.filter((w) => w && !ADVISORY_KINDS.has(w.kind));
143
+ return gating.length > 0 && gating.every((w) => w.error != null && w.error !== '');
144
+ }
145
+
146
+ // newestAttemptAt — the baseline, read BEFORE the worker starts. Rounds are
147
+ // counted by being newer than this server timestamp, so the supervisor's own
148
+ // clock never enters the comparison.
149
+ function newestAttemptAt(attempts) {
150
+ let newest = null;
151
+ for (const a of attempts || []) {
152
+ const t = Date.parse(a && a.attempted_at);
153
+ if (Number.isFinite(t) && (newest === null || t > newest)) newest = t;
154
+ }
155
+ return newest;
156
+ }
157
+
158
+ // roundsFromAttempts — the server's rounds after the baseline, in ledger-row
159
+ // shape so gradeCapState reads both sources the same way.
160
+ function roundsFromAttempts(attempts, taskId, afterMs = null) {
161
+ const out = [];
162
+ for (const a of attempts || []) {
163
+ const t = Date.parse(a && a.attempted_at);
164
+ if (!Number.isFinite(t) || (afterMs !== null && t <= afterMs)) continue;
165
+ out.push({ task_id: Number(taskId), at: a.attempted_at, passed: a.passed === true, outage: attemptIsOutage(a) });
166
+ }
167
+ return out;
168
+ }
169
+
170
+ // readServerAttempts — the task's recorded rounds, or null when unreadable. Null
171
+ // is "unknown", never "zero": the caller falls back to the local ledger.
172
+ async function readServerAttempts(api, taskId) {
173
+ try {
174
+ if (!api || !api.tasks || typeof api.tasks.getTasksId !== 'function') return null;
175
+ const r = await api.tasks.getTasksId({ id: Number(taskId), query: { include: 'grade,grade_attempts' } });
176
+ if (!r || !r.ok || !r.data) return null;
177
+ const grade = r.data.grade;
178
+ if (!grade) return []; // never graded: a real zero
179
+ return Array.isArray(grade.attempts) ? grade.attempts : null;
180
+ } catch (_) { return null; }
181
+ }
182
+
183
+ module.exports = {
184
+ LEDGER_ENV, GRADE_CAP, REFUSED_EXIT,
185
+ ledgerPathFromEnv, isOutage, readLedger, gradeCapState, checkGradeCap, recordGradeOutcome,
186
+ attemptIsOutage, newestAttemptAt, roundsFromAttempts, readServerAttempts,
187
+ };
@@ -103,6 +103,10 @@ function buildWorkerPrompt({ task, packPath = 'docs/packs/engineer.md' } = {}) {
103
103
  '',
104
104
  'Ship it yourself when the work is done and verified: `node scripts/gds/ship.js <id> --no-surprise --notes \"...\" --summary \"...\"`. Do NOT pass --skip-grade and do not force any gate; if the grader or the merge refuses, that is the system protecting main — return cannot_complete and say so.',
105
105
  '',
106
+ // task 1004536: stated here so the worker does not waste a session trying —
107
+ // but ship.js and the supervisor enforce it, because this line alone was ignored.
108
+ 'GRADE CAP: this run gets the first grade plus at most ONE re-grade (`node scripts/gds/ship.js <id> --regrade`). If the grade fails twice, STOP: do not fix and re-grade again — return cannot_complete with the grader\'s findings quoted, and the task goes to the owner. ship.js refuses a third grade in this run and the supervisor ends it. A "GRADER UNAVAILABLE" outcome is an outage, not a fail, and does not count toward the two.',
109
+ '',
106
110
  'When you are done, return the structured verdict. `undetermined_decisions` must list every choice you made that the spec did not settle — retry counts, naming, error shapes, anything a reviewer would want to have been asked about. An empty list is a claim that the spec determined everything; do not make it lightly.',
107
111
  ].join('\n');
108
112
  }
@@ -42,7 +42,8 @@ const gauge = require('./autonomy-gauge.js');
42
42
  const {
43
43
  buildWorkerArgs, buildWorkerPrompt, classifyRun, usageLimitHit, DEFAULT_MODEL, DEFAULT_WORKER_TIMEOUT_MS,
44
44
  } = require('./autobongos-loop.js');
45
- const { verifyShip, failureReason } = require('./autobongos-verify.js');
45
+ const { verifyShip, failureReason, recheckLands, pendingLandIds } = require('./autobongos-verify.js');
46
+ const gradeCap = require('./autobongos-grade-cap.js');
46
47
  const fence = require('../../modules/autonomy/fence.js');
47
48
  // The ADR 0043 protected-path registry, reused rather than re-derived. scripts/
48
49
  // sits outside the kernel boundary, so this direct core require is allowed here
@@ -51,6 +52,7 @@ const fence = require('../../modules/autonomy/fence.js');
51
52
  const { matchProtected } = require('../../src/bongos/permission-path-check');
52
53
  const { resolveEnv, envPrefix, LEGACY_ENV_PREFIXES } = require('../../src/instance-config');
53
54
  const cadence = require('../../modules/autonomy/cadence.js');
55
+ const configIdle = require('../../modules/autonomy/config-idle.js');
54
56
 
55
57
  const REPO_ROOT = path.resolve(__dirname, '..', '..');
56
58
 
@@ -106,6 +108,58 @@ function record(entry) {
106
108
  return row;
107
109
  }
108
110
 
111
+ // readRunLogRows — the run log back as rows, newest last. Only the TAIL is read
112
+ // (a log that runs for weeks must not be slurped whole every loop), and a line
113
+ // that will not parse — the half-written last line of a kill -9 — is skipped.
114
+ function readRunLogRows(p = logPath(), maxBytes = 4 * 1024 * 1024) {
115
+ let text = '';
116
+ try {
117
+ const size = fs.statSync(p).size;
118
+ const fd = fs.openSync(p, 'r');
119
+ try {
120
+ const len = Math.min(size, maxBytes);
121
+ const buf = Buffer.alloc(len);
122
+ fs.readSync(fd, buf, 0, len, size - len);
123
+ text = buf.toString('utf8');
124
+ } finally { fs.closeSync(fd); }
125
+ } catch (_) { return []; }
126
+ const rows = [];
127
+ for (const line of text.split('\n')) {
128
+ if (!line.trim()) continue;
129
+ try { rows.push(JSON.parse(line)); } catch (_) { /* partial line */ }
130
+ }
131
+ return rows;
132
+ }
133
+
134
+ // recheckPendingLands — once per loop (task 1004538). Which tasks are waiting on
135
+ // a land comes from the run log on disk and when they shipped comes from the
136
+ // ledger, so a restart loses neither. Open `shipped_no_artifact` blockers are read
137
+ // from the database and resolved once main carries the commit. Every outcome is a
138
+ // row, so the next pass reads it back.
139
+ //
140
+ // The blocker sweep runs at most once per LAND_SWEEP_MS per process (and at once
141
+ // after a restart): a false alarm closing within the hour is plenty, and the
142
+ // open-blocker list need not be read on every loop. Pending lands are re-probed
143
+ // every loop — and cost nothing when there are none.
144
+ const LAND_SWEEP_MS = 60 * 60 * 1000;
145
+ async function recheckPendingLands(deps = {}) {
146
+ const log = deps.record || record;
147
+ const st = deps.state || RUNNER_STATE;
148
+ const now = deps.now ? deps.now() : Date.now();
149
+ const sweepBlockers = !st.landSweepAt || now - st.landSweepAt >= LAND_SWEEP_MS;
150
+ const ids = pendingLandIds(readRunLogRows());
151
+ if (!ids.length && !sweepBlockers) return { rechecked: [], resolved: [], errors: [] };
152
+ const out = await recheckLands({ pendingIds: ids, sweepBlockers }, { ...(deps.api ? { api: deps.api } : {}), ...(deps.landDeps || {}) });
153
+ if (sweepBlockers) st.landSweepAt = now;
154
+ for (const r of out.rechecked) log({ event: 'land_recheck', ...r });
155
+ for (const r of out.resolved) {
156
+ log({ event: 'land_blocker_resolved', task_id: r.task_id, blocker_id: r.blocker_id, sha: r.sha,
157
+ reason: `main now carries ${r.sha} naming task ${r.task_id} — the no-artifact blocker was a slow land, resolved` });
158
+ }
159
+ if (out.errors.length) log({ event: 'land_recheck_failed', errors: out.errors.slice(0, 10) });
160
+ return out;
161
+ }
162
+
109
163
  // pickedFrom — the row fields that say where a pick came from (task 1004453).
110
164
  // `priority` is true only when the fence named a priority goal AND this pick came
111
165
  // from it, so "the priority goal ran dry and the runner moved on" reads as
@@ -246,9 +300,41 @@ function resolveWorkerBin(env = process.env, platform = process.platform) {
246
300
  return { bin: 'claude', shell: true };
247
301
  }
248
302
 
303
+ // gradeLedgerPath — the run's LOCAL ledger (task 1004536): the file ship.js reads
304
+ // to refuse a third grade early. A fresh file per run, under the instance config
305
+ // dir, never inside the task's tree. It is the courtesy half of the cap — see the
306
+ // header of autobongos-grade-cap.js; the boundary is the server count below.
307
+ const GRADE_CAP_WATCH_MS = 30_000;
308
+ function gradeLedgerPath(taskId) {
309
+ const id = Number(taskId);
310
+ if (!Number.isSafeInteger(id) || id <= 0) throw new Error(`gradeLedgerPath: task id must be a positive integer, got ${taskId}`);
311
+ const name = path.join('autobongos-grades', `${id}-${Date.now()}-${crypto.randomBytes(3).toString('hex')}.jsonl`);
312
+ try { return require('../../src/instance-config.js').configPath(name); }
313
+ catch (_) { return path.join(os.tmpdir(), name); }
314
+ }
315
+
316
+ // gradeCapReader — how the supervisor counts this run's graded rounds.
317
+ //
318
+ // The SERVER's grade_attempts is the count that matters: the worker has no write
319
+ // path to it, so clearing an env var or deleting a file cannot reset it. The
320
+ // baseline is the newest attempt recorded BEFORE the worker starts; only rounds
321
+ // after it belong to this run. When the server cannot be read, the local ledger
322
+ // is the fallback — and either source showing the cap exhausted is enough.
323
+ async function gradeCapReader({ api, taskId, ledger }) {
324
+ const before = await gradeCap.readServerAttempts(api, taskId);
325
+ const afterMs = before ? gradeCap.newestAttemptAt(before) : null;
326
+ return async function read() {
327
+ const local = gradeCap.gradeCapState(gradeCap.readLedger(ledger), taskId);
328
+ const attempts = before ? await gradeCap.readServerAttempts(api, taskId) : null;
329
+ if (!attempts) return { ...local, source: 'local' };
330
+ const server = gradeCap.gradeCapState(gradeCap.roundsFromAttempts(attempts, taskId, afterMs), taskId);
331
+ return server.exhausted || !local.exhausted ? { ...server, source: 'server' } : { ...local, source: 'local' };
332
+ };
333
+ }
334
+
249
335
  // spawnWorker — one headless session, bounded by WALL CLOCK, in the task's own
250
336
  // working tree.
251
- async function spawnWorker({ task, model, timeoutMs, cwd, deps = {} }) {
337
+ async function spawnWorker({ task, model, timeoutMs, cwd, gradeLedger = null, capCheck = null, deps = {} }) {
252
338
  const spawnFn = deps.spawn || spawn;
253
339
  const sessionId = deps.sessionId || crypto.randomUUID(); // pinned, so the session can be resumed
254
340
  const { bin, shell: workerShell } = deps.resolveBin ? deps.resolveBin() : resolveWorkerBin();
@@ -267,22 +353,42 @@ async function spawnWorker({ task, model, timeoutMs, cwd, deps = {} }) {
267
353
  stdio: ['pipe', 'pipe', 'pipe'],
268
354
  shell: workerShell,
269
355
  windowsHide: true,
270
- env: workerEnv(),
356
+ // The run's local grade ledger (task 1004536): ship.js appends each graded
357
+ // round to it and refuses a third early. The worker can see this variable,
358
+ // so it is the courtesy half only — capCheck below is the boundary.
359
+ env: workerEnv(process.env, gradeLedger ? { [gradeCap.LEDGER_ENV]: gradeLedger } : {}),
271
360
  });
272
361
  } catch (e) { return resolve({ spawnError: e.message }); }
273
362
 
274
- let stdout = ''; let stderr = ''; let timedOut = false; let truncated = false;
363
+ let stdout = ''; let stderr = ''; let timedOut = false; let truncated = false; let gradeCapped = false;
275
364
  const cap = (cur, d) => {
276
365
  if (cur.length >= MAX_WORKER_OUTPUT) { truncated = true; return cur; }
277
366
  return cur + String(d).slice(0, MAX_WORKER_OUTPUT - cur.length);
278
367
  };
279
368
  const timer = setTimeout(() => { timedOut = true; try { child.kill('SIGKILL'); } catch (_) {} }, timeoutMs);
369
+ // task 1004536: once the second graded FAIL is recorded the run is over — the
370
+ // task goes to the owner. capCheck reads the SERVER's grade_attempts (see
371
+ // gradeCapReader), so a worker that cleared the ledger env or deleted the file
372
+ // is still stopped. One read at a time: a slow API must not stack polls.
373
+ let capBusy = false;
374
+ const capWatch = capCheck ? setInterval(async () => {
375
+ if (capBusy || gradeCapped) return;
376
+ capBusy = true;
377
+ try {
378
+ if ((await capCheck()).exhausted) {
379
+ gradeCapped = true; clearInterval(capWatch);
380
+ try { child.kill('SIGKILL'); } catch (_) {}
381
+ }
382
+ } catch (_) { /* an unreadable count is retried next tick; the post-run read still decides */ }
383
+ capBusy = false;
384
+ }, deps.capWatchMs || GRADE_CAP_WATCH_MS) : null;
385
+ const stopWatch = () => { if (capWatch) clearInterval(capWatch); };
280
386
  child.stdout.on('data', (d) => { stdout = cap(stdout, d); });
281
387
  child.stderr.on('data', (d) => { stderr = cap(stderr, d); });
282
- child.on('error', (e) => { clearTimeout(timer); resolve({ spawnError: e.message, sessionId }); });
388
+ child.on('error', (e) => { clearTimeout(timer); stopWatch(); resolve({ spawnError: e.message, sessionId }); });
283
389
  child.on('close', (code) => {
284
- clearTimeout(timer);
285
- resolve({ envelope: stdout, exitCode: code, timedOut, truncated, sessionId, stderr: stderr.slice(-2000) });
390
+ clearTimeout(timer); stopWatch();
391
+ resolve({ envelope: stdout, exitCode: code, timedOut, truncated, gradeCapped, sessionId, stderr: stderr.slice(-2000) });
286
392
  });
287
393
  child.stdin.end(prompt);
288
394
  });
@@ -368,7 +474,7 @@ const MAX_CLAIM_ATTEMPTS = 5;
368
474
  const LIMIT_UNKNOWN_RESET_MS = 30 * 60 * 1000;
369
475
 
370
476
  function newRunnerState() {
371
- return { setAside: new Map(), queueGate: null, limitUntil: null, diskHead: null, mainHead: null };
477
+ return { setAside: new Map(), queueGate: null, limitUntil: null, diskHead: null, mainHead: null, landSweepAt: null };
372
478
  }
373
479
  const RUNNER_STATE = newRunnerState();
374
480
 
@@ -543,8 +649,24 @@ async function iteration(opts, deps = {}) {
543
649
 
544
650
  // 5. Work, in that tree.
545
651
  const started = Date.now();
546
- const res = await spawnWork({ task, model: opts.model, timeoutMs: opts.timeoutMs, cwd: wtPath, deps });
547
- const verdict = classifyRun(res);
652
+ const gradeLedger = deps.gradeLedgerPath ? deps.gradeLedgerPath(task.id) : gradeLedgerPath(task.id);
653
+ // Same client rule as verifyShip below: an injected api is used as given.
654
+ let capApi = deps.api || null;
655
+ if (!capApi) { try { capApi = await (require('./cli-lib').cliClient)(); } catch (_) { capApi = null; } }
656
+ const capCheck = await gradeCapReader({ api: capApi, taskId: task.id, ledger: gradeLedger });
657
+ const res = await spawnWork({ task, model: opts.model, timeoutMs: opts.timeoutMs, cwd: wtPath, gradeLedger, capCheck, deps });
658
+ let verdict = classifyRun(res);
659
+ // 5b. THE GRADE CAP (task 1004536). The owner's rule — a task that fails the
660
+ // grade twice goes to the owner — read from the run's ledger, not from the
661
+ // worker's verdict: the worker was told this in its prompt and kept re-grading.
662
+ const grades = await capCheck();
663
+ try { fs.rmSync(gradeLedger, { force: true }); } catch (_) {} // its counts travel in the run-log row
664
+ if (grades && grades.exhausted) {
665
+ verdict = {
666
+ ...verdict, outcome: 'needs_human', releaseClaim: true,
667
+ reason: `the grade failed ${grades.fails} times in this run — handed to the owner instead of re-graded again (task 1004536)`,
668
+ };
669
+ }
548
670
  // Cut off by the usage limit? Recorded here so the NEXT iteration sleeps to the
549
671
  // reset instead of spawning a worker that will be cut off the same way.
550
672
  const limit = verdict.outcome === 'claims_shipped' ? null : usageLimitHit({ envelope: res.envelope, stderr: res.stderr });
@@ -580,6 +702,7 @@ async function iteration(opts, deps = {}) {
580
702
  outcome: verdict.outcome, reason: verdict.reason,
581
703
  session_id: res.sessionId || null, duration_s: Math.round((Date.now() - started) / 1000),
582
704
  cost_usd: verdict.costUsd ?? null,
705
+ ...(grades && (grades.graded || grades.outages) ? { grade_rounds: grades.graded, grade_outages: grades.outages, grade_cap_reached: grades.exhausted, grade_count_source: grades.source } : {}),
583
706
  output_truncated: !!res.truncated,
584
707
  ...(limit ? { usage_limit: { reset_at: limit.resetAt } } : {}),
585
708
  undetermined_decisions: verdict.verdict ? verdict.verdict.undetermined_decisions : null,
@@ -1054,7 +1177,14 @@ async function forever(opts, deps = {}) {
1054
1177
  // down the thing it watches is strictly worse than a gap in the log, and the gap
1055
1178
  // reports itself anyway: the hall reads the AGE of the last check-in, so a
1056
1179
  // failed beat shows up as silence, which is the signal.
1057
- const beat = async (fields) => {
1180
+ // How long this runner has been idle on a setting the owner can fix (task
1181
+ // 1004539). Past that setting's grace, every beat that is not about a task says
1182
+ // `config_idle:<code>` instead of "waiting", so the hall shows the fix rather
1183
+ // than "alive" for ten hours. Any row that is not that refusal ends the stretch.
1184
+ let idle = null;
1185
+ const clock = () => (deps.now ? deps.now() : Date.now());
1186
+ const beat = async (rawFields) => {
1187
+ const fields = configIdle.markBeat(rawFields, idle, clock());
1058
1188
  // The runner state is passed, not defaulted: the file names `disk_head` from
1059
1189
  // it (task 1004406), and an injected state must reach the beat that reports it.
1060
1190
  try { beatFile(Date.now(), deps.state || RUNNER_STATE); } catch (_) { /* a runner that cannot write its file still runs */ }
@@ -1081,6 +1211,15 @@ async function forever(opts, deps = {}) {
1081
1211
  // Every loop, so a crash leaves a heartbeat that goes stale on its own — the
1082
1212
  // runner never has to notice it is dying for its death to become visible.
1083
1213
  await beat({ mode: state.mode, consecutive_failures: state.consecutiveFailures, last_event: 'loop' });
1214
+ // Slow lands (task 1004538): re-probe shipped tasks still waiting on main, and
1215
+ // close no-artifact blockers whose commit has since landed. Armed only by
1216
+ // main() (or an injected stub), like refreshInPlace, so no test harness that
1217
+ // predates it can fall through to the network. It can never stop the loop.
1218
+ if (opts.recheckLands || deps.recheckLands) {
1219
+ try { await (deps.recheckLands || recheckPendingLands)(deps); } catch (e) {
1220
+ log({ event: 'land_recheck_failed', errors: [e && e.message ? e.message : String(e)] });
1221
+ }
1222
+ }
1084
1223
  // In `probing` the runner takes exactly ONE task and only after its wait, so
1085
1224
  // a broken window costs a handful of cheap requests rather than a night of
1086
1225
  // failing spawns. `maxTasks` still bounds a full-speed burst.
@@ -1090,6 +1229,7 @@ async function forever(opts, deps = {}) {
1090
1229
  for (let i = 0; i < budget; i++) {
1091
1230
  row = await iterate(opts, deps);
1092
1231
  emit(row);
1232
+ idle = configIdle.trackIdle(idle, row, clock());
1093
1233
  // Carry the task id so the hall can tell "quiet because it is building
1094
1234
  // task 1234" from "quiet because it is dead". A worker may hold the loop for
1095
1235
  // ninety minutes, and without this every honest hour of work looks like a
@@ -1161,7 +1301,7 @@ async function main(argv) {
1161
1301
  // it lived on the machine being fenced; the database owns that now, and an empty
1162
1302
  // `--goal` means "every allowlisted goal" rather than "anything claimable". The
1163
1303
  // refusal did not disappear — it moved somewhere the runner host cannot edit.
1164
- if (opts.forever) return forever({ ...opts, refreshInPlace: true });
1304
+ if (opts.forever) return forever({ ...opts, refreshInPlace: true, recheckLands: true });
1165
1305
 
1166
1306
  // The one-shot path, kept for an operator running a single task by hand. It
1167
1307
  // still reconciles first: a leftover claim is a leftover claim whoever is
@@ -1186,8 +1326,9 @@ if (require.main === module) {
1186
1326
 
1187
1327
  module.exports = {
1188
1328
  readArgv, pickTask, iteration, record, logPath, spawnWorker, workerEnv, WORKER_ENV_ALLOW, resolveWorkerBin,
1189
- worktreeName, readFence, reconcileLeftoverClaims, pollSignals, waitOrJump, forever, CLAIM_PREFIX, codeDrifted, loadedHead, UPGRADE_EXIT_CODE, MIN_UPTIME_BEFORE_UPGRADE_MS,
1329
+ worktreeName, readFence, gradeLedgerPath, gradeCapReader, reconcileLeftoverClaims, pollSignals, waitOrJump, forever, CLAIM_PREFIX, codeDrifted, loadedHead, UPGRADE_EXIT_CODE, MIN_UPTIME_BEFORE_UPGRADE_MS,
1190
1330
  heartbeatPath, writeHeartbeat, readHeartbeat, pidAlive, anotherRunnerIsAlive, HEARTBEAT_STALE_MS,
1191
1331
  isRunnerTree, RUNNER_TREE_RE, postHeartbeat, claudeAccountLabel, RUNNER_STARTED_AT, failureReason,
1192
1332
  newRunnerState, loadedFiles, supervisorTouched, refreshCheckout, LIMIT_UNKNOWN_RESET_MS, QUEUE_GATED_EXIT, CLAIM_SET_ASIDE_MS, MAX_CLAIM_ATTEMPTS, owedTaskIds,
1333
+ readRunLogRows, recheckPendingLands, LAND_SWEEP_MS,
1193
1334
  };