@bongos/core 1.21.5 → 1.21.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/.bongos-core.json +152 -72
  2. package/clients/bongos-client/index.d.ts +1 -1
  3. package/docs/adr/0361-merges-publish-as-candidates-release-decides-what-others-are-offered.md +1 -0
  4. package/docs/api/openapi.json +2 -1
  5. package/docs/api-reference.md +1 -1
  6. package/docs/copy-inventory.md +156 -144
  7. package/docs/copy-registry.json +286 -158
  8. package/docs/module-api-changelog.md +4 -0
  9. package/docs/onboarding/diagrams/03-drachmae-karma.mmd +1 -1
  10. package/docs/onboarding/diagrams/assertions.json +1 -1
  11. package/docs/page-inventory.json +35 -6
  12. package/docs/page-readings.json +1056 -1013
  13. package/modules/autonomy/config-idle.js +192 -0
  14. package/modules/autonomy/routes/autonomy.js +22 -0
  15. package/modules/autonomy/runner-health.js +22 -1
  16. package/modules/hall-ui/public/city-draw.js +438 -0
  17. package/modules/hall-ui/public/city.css +110 -0
  18. package/modules/hall-ui/public/city.html +71 -0
  19. package/modules/hall-ui/public/city.js +177 -0
  20. package/modules/hall-ui/public/city.states.json +30 -0
  21. package/modules/hall-ui/public/gate.js +11 -4
  22. package/modules/hall-ui/public/profile.css +4 -0
  23. package/modules/hall-ui/public/profile.js +25 -1
  24. package/modules/hall-ui/public/settings-autobongos.js +19 -1
  25. package/modules/hall-ui/public/settings.css +26 -0
  26. package/modules/hall-ui/public/settings.html +8 -0
  27. package/modules/hall-ui/public/settings.js +43 -3
  28. package/modules/platform-identity/craft-rollup.js +72 -0
  29. package/modules/platform-identity/migrations/platform_identity_029_activity_crafts.sql +23 -0
  30. package/modules/platform-identity/platform-identity.js +12 -6
  31. package/modules/platform-identity/routes/sso.js +4 -0
  32. package/modules/platform-identity/tests/platform-identity.mjs +1 -1
  33. package/modules/provisioning/core-upgrade.js +34 -7
  34. package/modules/provisioning/module.json +2 -1
  35. package/modules/provisioning/seams.js +2 -0
  36. package/modules/public-landing/public/projects-hall-live.states.json +216 -0
  37. package/modules/public-landing/public/projects.html +43 -13
  38. package/modules/public-landing/public/projects.states.json +2 -1
  39. package/modules/ui-design/kit/fixtures/provisioning-instances-hall-live.json +106 -0
  40. package/package-lock.json +2 -2
  41. package/package.json +1 -1
  42. package/release-notes.json +64 -0
  43. package/scripts/gds/autobongos-grade-cap.js +187 -0
  44. package/scripts/gds/autobongos-loop.js +4 -0
  45. package/scripts/gds/autobongos-run.js +154 -13
  46. package/scripts/gds/autobongos-verify.js +211 -14
  47. package/scripts/gds/provision-core-upgrade.js +14 -2
  48. package/scripts/gds/provision-repo.js +77 -13
  49. package/scripts/gds/ship-flow.js +5 -0
  50. package/scripts/gds/ship-preflight-steps.js +2 -1
  51. package/scripts/gds/ship.js +10 -0
  52. package/scripts/gds/update-sweep.js +24 -8
  53. package/scripts/gds/upgrade-outcome.js +1 -0
  54. package/src/bongos/auth-admission.js +95 -13
  55. package/src/bongos/db-kernel.js +1 -1
  56. package/src/bongos/routes/core-update.js +27 -2
  57. package/src/bongos/serve-internal.js +14 -0
  58. package/src/bongos/software-update.js +3 -2
  59. package/src/module-api.js +1 -1
  60. package/tests/activity_rollup_order_db.mjs +3 -1
  61. package/tests/activity_snapshots_db.mjs +2 -0
  62. package/tests/autobongos_grade_cap.mjs +125 -0
  63. package/tests/autobongos_loop.mjs +230 -1
  64. package/tests/autobongos_verify.mjs +239 -3
  65. package/tests/autonomy_config_idle.mjs +272 -0
  66. package/tests/core_update_banner.mjs +29 -0
  67. package/tests/core_upgrade_door.mjs +31 -1
  68. package/tests/core_upgrade_runner.mjs +20 -0
  69. package/tests/hall_audit.mjs +10 -1
  70. package/tests/hall_city.mjs +369 -0
  71. package/tests/hall_page_gate_map.mjs +5 -0
  72. package/tests/hub_craft_rollup.mjs +350 -0
  73. package/tests/profile_rollup_consent.mjs +2 -0
  74. package/tests/profile_route.mjs +16 -2
  75. package/tests/project_invite_ui.mjs +65 -7
  76. package/tests/projects_hub_dns_ready.mjs +7 -1
  77. package/tests/provision.mjs +78 -0
  78. package/tests/settings_main_role.mjs +190 -0
  79. package/tests/software_update.mjs +56 -1
  80. package/tests/ui_design_kit.mjs +1 -1
  81. package/tests/update_subscription_engine.mjs +55 -3
  82. package/tests/wizard_physics_more.mjs +19 -0
@@ -132,6 +132,69 @@ async function recentShipsForRollup(builderId) {
132
132
  }));
133
133
  }
134
134
 
135
+ // Each craft by its own work, for the hub's profile tiles (task 1004440 /
136
+ // WA6.RP10, goal 1000095). Until now the rollup said "Roles" from the declared
137
+ // preference alone and counted only shipped tasks, so an artist or an ideator who
138
+ // had done real work read on the hub as an engineer with nothing shipped.
139
+ //
140
+ // A craft is EARNED when this builder holds credits in it — the same rule the
141
+ // local profile uses to show a role (src/bongos/role-stats.js, task 1004433),
142
+ // read through the same ports. This is a kernel file, so it reads those ports
143
+ // by key rather than importing role-stats (ADR 0091 §1). For each earned craft it
144
+ // sends that craft's headline counts; an unearned craft sends nothing at all.
145
+ // Counts and sums only: no idea title, no page wording, no task list.
146
+ //
147
+ // `disciplines` becomes the earned crafts, in the builder's preferred order;
148
+ // with nothing earned (or no economy module to ask) it falls back to the declared
149
+ // preference, exactly as before. A port that cannot be asked leaves its numbers
150
+ // null ("could not ask"), never 0. Never throws: a failed read drops its craft's
151
+ // numbers, not the report.
152
+ const ROLLUP_CRAFTS = Object.freeze(['engineer', 'artist', 'ideator']);
153
+ const countOrNull = (v) => (Number.isInteger(v) && v >= 0 ? v : null);
154
+ async function readCraftRollup(builderId, { shipped = 0, preferred = [] } = {}) {
155
+ const fallback = { counts: null, disciplines: Array.isArray(preferred) ? preferred : [] };
156
+ try {
157
+ const reward = seams.resolveOptional('reward');
158
+ const byCraftMap = reward && typeof reward.creditsByCraft === 'function'
159
+ ? await reward.creditsByCraft([builderId]) : null;
160
+ const byCraft = byCraftMap ? byCraftMap.get(String(builderId)) : null;
161
+ if (!byCraft) return fallback;
162
+ const prefCrafts = fallback.disciplines
163
+ .map((d) => (typeof reward.craftForPreference === 'function' ? reward.craftForPreference(d) : d))
164
+ .filter((c) => ROLLUP_CRAFTS.includes(c));
165
+ const earned = [...prefCrafts, ...ROLLUP_CRAFTS]
166
+ .filter((c, i, a) => a.indexOf(c) === i && Number(byCraft[c]) > 0);
167
+ if (!earned.length) return fallback;
168
+ const ask = async (port) => {
169
+ const fn = seams.resolveOptional(port);
170
+ if (typeof fn !== 'function') return null;
171
+ try { return (await fn(builderId)) || null; } catch { return null; }
172
+ };
173
+ const [artist, ideator] = await Promise.all([
174
+ earned.includes('artist') ? ask('copy-desk.artistStats') : null,
175
+ earned.includes('ideator') ? ask('ideas.ideatorStats') : null,
176
+ ]);
177
+ const credits = (c) => countOrNull(Math.round(Number(byCraft[c])));
178
+ const counts = {};
179
+ if (earned.includes('engineer')) counts.engineer = { shipped: countOrNull(shipped), credits: credits('engineer') };
180
+ if (earned.includes('artist')) {
181
+ counts.artist = { pages_approved: countOrNull(artist && artist.pages_approved), credits: credits('artist') };
182
+ }
183
+ if (earned.includes('ideator')) {
184
+ counts.ideator = {
185
+ filed: countOrNull(ideator && ideator.filed),
186
+ full_ideas: countOrNull(ideator && ideator.full_ideas),
187
+ ratified: countOrNull(ideator && ideator.ratified),
188
+ credits: credits('ideator'),
189
+ };
190
+ }
191
+ return { counts, disciplines: earned };
192
+ } catch (err) {
193
+ log.info('[auth] craft rollup read failed (non-blocking):', err.message);
194
+ return fallback;
195
+ }
196
+ }
197
+
135
198
  // Best-effort cross-project activity rollup to the hub (ADR 0141 §4): report this
136
199
  // builder's descriptive talent snapshot (credits, ships, karma, rank, disciplines,
137
200
  // and since task 1002803 their last few shipped works) for the cross-project
@@ -162,26 +225,45 @@ async function reportActivityRollup({ builderId, githubId = null }) {
162
225
  const row = b.rows[0];
163
226
  const ghId = githubId ?? row.github_id;
164
227
  if (!/^\d+$/.test(String(ghId ?? ''))) return;
165
- const shipped = await pool.query('SELECT count(*)::int AS n FROM tasks WHERE shipped_by = $1', [builderId]);
228
+ // SHIPPED works only (task 1004440): a count with no status filter also
229
+ // counted work that was handed back after shipping. The same predicate as the
230
+ // instance's own profile (lifecycle countShippedTasksForBuilder).
231
+ const shipped = await pool.query(
232
+ `SELECT count(*)::int AS n FROM tasks WHERE shipped_by = $1 AND status = 'shipped'`,
233
+ [builderId],
234
+ );
235
+ const shippedCount = shipped.rows[0] ? shipped.rows[0].n : 0;
166
236
  const recentShips = await recentShipsForRollup(builderId);
167
- await fetch(`${idp.origin}/api/bongos/sso/activity/rollup`, {
237
+ const craft = await readCraftRollup(builderId, { shipped: shippedCount, preferred: row.preferred_disciplines });
238
+ const body = {
239
+ client_id: idp.clientId,
240
+ client_secret: idp.clientSecret,
241
+ github_id: ghId,
242
+ credits: Number(row.total_credits) || 0,
243
+ tasks_shipped: shippedCount,
244
+ karma: Number(row.karma) || 0,
245
+ rank: row.rank || null,
246
+ disciplines: craft.disciplines,
247
+ recent_ships: recentShips,
248
+ };
249
+ if (craft.counts) body.crafts = craft.counts;
250
+ const send = (payload) => fetch(`${idp.origin}/api/bongos/sso/activity/rollup`, {
168
251
  method: 'POST',
169
252
  headers: {
170
253
  'Content-Type': 'application/json', Accept: 'application/json', 'User-Agent': userAgent(),
171
254
  'Bongos-Report-As-Of': asOf,
172
255
  },
173
- body: JSON.stringify({
174
- client_id: idp.clientId,
175
- client_secret: idp.clientSecret,
176
- github_id: ghId,
177
- credits: Number(row.total_credits) || 0,
178
- tasks_shipped: shipped.rows[0] ? shipped.rows[0].n : 0,
179
- karma: Number(row.karma) || 0,
180
- rank: row.rank || null,
181
- disciplines: Array.isArray(row.preferred_disciplines) ? row.preferred_disciplines : [],
182
- recent_ships: recentShips,
183
- }),
256
+ body: JSON.stringify(payload),
184
257
  });
258
+ const res = await send(body);
259
+ // A hub older than `crafts` validates the body strictly and refuses the
260
+ // unknown field with a 400, which would lose the whole report. So a refused
261
+ // report that carried crafts goes again once without them: the totals still
262
+ // land, and the craft numbers arrive once that hub upgrades.
263
+ if (body.crafts && res && res.status === 400) {
264
+ const { crafts: _unsent, ...older } = body;
265
+ await send(older);
266
+ }
185
267
  } catch (err) {
186
268
  log.info('[auth] hub activity rollup failed (non-blocking):', err.message);
187
269
  }
@@ -55,7 +55,7 @@ async function getBuilderById(id) {
55
55
  // AFTER INSERT triggers on credit_log + karma_log respectively.
56
56
  const { rows } = await pool.query(
57
57
  `SELECT id, github_id, github_login, display_name, avatar_url, total_credits, karma, created_at, rank,
58
- status, deactivated_at, reactivated_at, deactivated_reason, preferred_disciplines, system_role,
58
+ status, deactivated_at, reactivated_at, deactivated_reason, preferred_disciplines, main_discipline, system_role,
59
59
  anthropic_email, anthropic_user_id, monthly_budget_usd
60
60
  FROM builders WHERE id = $1`,
61
61
  [id]
@@ -129,7 +129,27 @@ function contributeCoreUpdateToDocket(read = readCoreUpdate) {
129
129
  });
130
130
  }
131
131
 
132
+ // This hall's own update rule, from the provisioning module when it runs (task 1004535):
133
+ // the rule set on /deploy, which the runner and the sweep already follow (ADR 0357: the
134
+ // /deploy row beats the sweep's roster file, which lives in the box's own checkout and is
135
+ // not this server's to read). No module, no row, or a failed read → null, and the readers
136
+ // fall back to the platform default exactly as before.
137
+ function storedSelfChannel() {
138
+ const read = seams.resolveOptional('provisioning.selfUpdateChannel');
139
+ return typeof read === 'function' ? read() : null;
140
+ }
141
+
142
+ // { channel, source } — WHICH rule the answer was computed under, so the page can say so:
143
+ // 'deploy' (the project's stored rule) or 'default' (none could be read). Never throws.
144
+ async function resolveChannel(selfChannel) {
145
+ let channel = null;
146
+ try { channel = (await selfChannel()) || null; } catch { channel = null; }
147
+ return { channel, source: channel ? 'deploy' : 'default' };
148
+ }
149
+
132
150
  function buildCoreUpdateRouter({
151
+ coreUpdate = readCoreUpdate,
152
+ selfChannel = storedSelfChannel,
133
153
  upgrades = readCoreUpgrades,
134
154
  softwareUpdate = readSoftwareUpdate,
135
155
  canUpdate = holdsCorePin,
@@ -155,7 +175,8 @@ function buildCoreUpdateRouter({
155
175
  try { running = require('../../module-api').CORE_VERSION; } catch { /* version unknown */ }
156
176
  // No `deploy_url` any more (task 1004296): the banner on a hall without a door of its
157
177
  // own links to Settings → Software update, and the hand-off lives on GET /software-update.
158
- res.json({ ok: true, update: await readCoreUpdate({ running }) });
178
+ const { channel, source } = await resolveChannel(selfChannel);
179
+ res.json({ ok: true, update: { ...(await coreUpdate({ running, channel })), channel_source: source } });
159
180
  }, { errorCode: 'core_update_failed', message: false }),
160
181
  );
161
182
 
@@ -186,7 +207,11 @@ function buildCoreUpdateRouter({
186
207
  asyncHandler('GET /software-update', async (req, res) => {
187
208
  let running = null;
188
209
  try { running = require('../../module-api').CORE_VERSION; } catch { /* version unknown */ }
189
- const [update, can] = await Promise.all([softwareUpdate({ running }), canUpdate(req)]);
210
+ // canUpdate does not depend on the rule, so it runs beside the rule read, not after it.
211
+ const [update, can] = await Promise.all([
212
+ resolveChannel(selfChannel).then(async ({ channel, source }) => ({ ...(await softwareUpdate({ running, channel })), channel_source: source })),
213
+ canUpdate(req),
214
+ ]);
190
215
  res.json({ ok: true, update, door: updateDoor(doorInputs()), can_update: can === true });
191
216
  }, { errorCode: 'software_update_failed', message: false }),
192
217
  );
@@ -941,6 +941,16 @@ function mountInternalSurfaces(app) {
941
941
  // statically, and a comment in that gap drops the regex out of the map, which
942
942
  // makes the page test as UNGATED while looking gated here. Prose goes above.
943
943
  const BOARD_ROOM_PAGE_RE = /^\/builders\/board-room(?:\.html)?\/?$/;
944
+ // /city — the Governor City's home (task 1004524, goal 1000125). Gated on the
945
+ // `board.vote.cast` atom, like the Board Room, because that is the atom its one
946
+ // governed read (GET /government/docket) checks: the page's reach equals the
947
+ // right to read what it draws, so no holder is refused a city they could read
948
+ // and no one else is served its shell. Its two goal reads are requireBuilder,
949
+ // a wider floor.
950
+ //
951
+ // NB as at PROJECT_SETTINGS_PAGE_RE: nothing between the `=` and the literal
952
+ // (tests/hall_page_gate_map.mjs parses this file statically). Prose goes above.
953
+ const CITY_PAGE_RE = /^\/builders\/city(?:\.html)?\/?$/;
944
954
  // /deploy — the owner's deploy door (task 1003159, ADR 0293). Gated on the
945
955
  // `core.pin.move` atom, the same key its preview and its move both check, so the
946
956
  // shell's reach equals the right to press the button. Archon-floor and system:true,
@@ -1023,6 +1033,10 @@ function mountInternalSurfaces(app) {
1023
1033
  if (BOARD_ROOM_PAGE_RE.test(norm)) {
1024
1034
  return gdsAuth.requireBoardVotePage(req, res, next);
1025
1035
  }
1036
+ // task 1004524: the Governor City, on the docket's own atom (see CITY_PAGE_RE).
1037
+ if (CITY_PAGE_RE.test(norm)) {
1038
+ return gdsAuth.requireBoardVotePage(req, res, next);
1039
+ }
1026
1040
  // ADR 0293: the deploy door, on the same atom its two routes check.
1027
1041
  if (DEPLOY_PAGE_RE.test(norm)) {
1028
1042
  return gdsAuth.requireCorePinPage(req, res, next);
@@ -18,8 +18,9 @@
18
18
  //
19
19
  // THE CHANNEL RULE IS NOT RESTATED HERE, as in core-update.js: `resolveChannelTarget` is the
20
20
  // upgrade runner's own function, so this never offers a version the runner would refuse.
21
- // No channel is passed, exactly as the banner passes none: an instance does not store its
22
- // own channel, and the runner's default is what it would get.
21
+ // The channel is the project's own update rule when the route can read it (the rule set on
22
+ // /deploy, via the provisioning.selfUpdateChannel seam — task 1004535), else the runner's
23
+ // default. The banner (core-update.js) is handed the same rule, so the two cannot disagree.
23
24
  //
24
25
  // A READING, NEVER A GUESS. `reading` is 'ok' only when the registry answered; "up to date"
25
26
  // is `offered: null` on an ok reading and nothing else. Notes that could not be read say so,
package/src/module-api.js CHANGED
@@ -75,7 +75,7 @@ const { responsibilityFor, ROLE_RESPONSIBILITIES } = require('./role-responsibil
75
75
  // MAJOR (see allowBoxScope below): passes the request through untouched.
76
76
  function deprecatedNoopMiddleware(_req, _res, next) { next(); }
77
77
 
78
- const CORE_VERSION = '1.21.5'; // CI auto-patch carrier (ADR 0161); changelog: docs/module-api-changelog.md
78
+ const CORE_VERSION = '1.21.7'; // CI auto-patch carrier (ADR 0161); changelog: docs/module-api-changelog.md
79
79
 
80
80
  // A namespaced logger so a module's log lines are attributable + consistent.
81
81
  // Usage: const log = api.logger('discord'); log.info('mounted');
@@ -74,7 +74,9 @@ const same = (actual, expected, msg) => { assert.deepEqual(actual, expected, msg
74
74
 
75
75
  try {
76
76
  for (const f of ['platform_identity_005_builder_activity.sql', 'platform_identity_012_recent_ships.sql',
77
- 'platform_identity_022_activity_snapshots.sql', 'platform_identity_026_activity_reported_as_of.sql']) {
77
+ 'platform_identity_022_activity_snapshots.sql', 'platform_identity_026_activity_reported_as_of.sql',
78
+ // upsertActivity also writes crafts (029, task 1004440): 42703 without it.
79
+ 'platform_identity_029_activity_crafts.sql']) {
78
80
  await q(migrationSql(f));
79
81
  }
80
82
  await q(migrationSql('platform_identity_026_activity_reported_as_of.sql')); // idempotent: replays cleanly
@@ -99,6 +99,8 @@ try {
99
99
  await q(migrationSql('platform_identity_022_activity_snapshots.sql'));
100
100
  // upsertActivity also writes reported_as_of (task 1004268) — same reason as 012 above.
101
101
  await q(migrationSql('platform_identity_026_activity_reported_as_of.sql'));
102
+ // ...and each craft's counts (029, task 1004440), same reason again.
103
+ await q(migrationSql('platform_identity_029_activity_crafts.sql'));
102
104
  await cleanup();
103
105
 
104
106
  // 1 — the writer snapshots the LIVE row, and only when it changed.
@@ -0,0 +1,125 @@
1
+ // tests/autobongos_grade_cap.mjs — the grade cap for one autonomous run (task 1004536).
2
+ //
3
+ // Owner rule, 2026-10-02: a task that fails the grade twice goes to the owner
4
+ // instead of being retried — the first grade plus at most ONE re-grade per
5
+ // worker run. The worker prompt already asked for this and was ignored (task
6
+ // 1004512 spent $63.81 over three rounds), so the cap is enforced in ship.js,
7
+ // where the panel spend happens, and read back by the supervisor.
8
+
9
+ import assert from 'node:assert/strict';
10
+ import { test } from 'node:test';
11
+ import { createRequire } from 'node:module';
12
+ import { spawnSync } from 'node:child_process';
13
+ import fs from 'node:fs';
14
+ import os from 'node:os';
15
+ import path from 'node:path';
16
+ import { fileURLToPath } from 'node:url';
17
+
18
+ const require = createRequire(import.meta.url);
19
+ const cap = require('../scripts/gds/autobongos-grade-cap.js');
20
+ const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
21
+
22
+ const fail = { passed: false, score: 5.6, signals: { panel_outcome: 'fail' } };
23
+ const pass = { passed: true, score: 8.1, signals: { panel_outcome: 'pass' } };
24
+ const outage = { passed: false, score: 0, signals: { panel_outcome: 'unavailable' } };
25
+
26
+ function tmpLedger() {
27
+ const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'grade-cap-'));
28
+ return path.join(dir, 'run.jsonl');
29
+ }
30
+ function ledgerWith(grades, taskId = 4242) {
31
+ const file = tmpLedger();
32
+ const env = { [cap.LEDGER_ENV]: file };
33
+ for (const g of grades) assert.equal(cap.recordGradeOutcome({ taskId, grade: g, env }), true);
34
+ return { file, env };
35
+ }
36
+
37
+ test('the cap is the first grade plus ONE re-grade', () => {
38
+ assert.equal(cap.GRADE_CAP, 2);
39
+ });
40
+
41
+ test('one failed grade leaves room for exactly one re-grade', () => {
42
+ const { env } = ledgerWith([fail]);
43
+ assert.equal(cap.checkGradeCap({ taskId: 4242, env }).ok, true);
44
+ });
45
+
46
+ test('a third grade attempt after two FAILs is refused, and says to hand it to the owner', () => {
47
+ const { env } = ledgerWith([fail, fail]);
48
+ const r = cap.checkGradeCap({ taskId: 4242, env });
49
+ assert.equal(r.ok, false);
50
+ assert.equal(r.state.exhausted, true);
51
+ assert.match(r.lines.join('\n'), /cannot_complete/);
52
+ });
53
+
54
+ test('a grader-UNAVAILABLE outcome does not count toward the two', () => {
55
+ const { env, file } = ledgerWith([fail, outage, outage]);
56
+ const r = cap.checkGradeCap({ taskId: 4242, env });
57
+ assert.equal(r.ok, true, 'an outage is not a quality verdict — the panel never ran');
58
+ const s = cap.gradeCapState(cap.readLedger(file), 4242);
59
+ assert.equal(s.graded, 1);
60
+ assert.equal(s.outages, 2);
61
+ assert.equal(s.exhausted, false);
62
+ });
63
+
64
+ test('a second round that PASSED is not refused — its merge leg still has to run', () => {
65
+ const { env } = ledgerWith([fail, pass]);
66
+ assert.equal(cap.checkGradeCap({ taskId: 4242, env }).ok, true);
67
+ });
68
+
69
+ test('rounds are counted per task, and a human ship (no ledger) is never capped', () => {
70
+ const { env } = ledgerWith([fail, fail], 1111);
71
+ assert.equal(cap.checkGradeCap({ taskId: 4242, env }).ok, true, 'another task\'s fails are not this one\'s');
72
+ assert.deepEqual(cap.checkGradeCap({ taskId: 4242, env: {} }), { ok: true, active: false });
73
+ assert.equal(cap.recordGradeOutcome({ taskId: 4242, grade: fail, env: {} }), false, 'nothing is written without a ledger');
74
+ });
75
+
76
+ test('a torn or missing ledger reads as empty rather than throwing', () => {
77
+ assert.deepEqual(cap.readLedger(path.join(os.tmpdir(), 'no-such-ledger-1004536.jsonl')), []);
78
+ const file = tmpLedger();
79
+ fs.writeFileSync(file, '{"task_id":4242,"passed":false,"outage":false}\n{"task_id":42');
80
+ assert.equal(cap.readLedger(file).length, 1);
81
+ });
82
+
83
+ test('server attempts: an outage is every GATING worker erroring; the advisory narc does not decide it', () => {
84
+ const w = (kind, error) => ({ kind, verdict: error ? null : 'fail', error, severities: [] });
85
+ assert.equal(cap.attemptIsOutage({ workers: [w('narc', null), w('quality', 'X'), w('hacker', 'X'), w('efficiency', 'X')] }), true);
86
+ assert.equal(cap.attemptIsOutage({ workers: [w('narc', 'X'), w('quality', null), w('hacker', 'X'), w('efficiency', 'X')] }), false, 'one gating worker ran: a real verdict');
87
+ assert.equal(cap.attemptIsOutage({ workers: [] }), false, 'an unknown shape is counted, never a free round');
88
+ });
89
+
90
+ test('server attempts: only rounds newer than the pre-run baseline belong to this run', () => {
91
+ const attempts = [
92
+ { attempted_at: '2026-10-01T00:00:00Z', passed: false, workers: [] },
93
+ { attempted_at: '2026-10-02T00:00:00Z', passed: false, workers: [] },
94
+ ];
95
+ const baseline = cap.newestAttemptAt(attempts.slice(0, 1));
96
+ const rows = cap.roundsFromAttempts(attempts, 4242, baseline);
97
+ assert.equal(rows.length, 1);
98
+ assert.equal(cap.newestAttemptAt([]), null);
99
+ assert.equal(cap.roundsFromAttempts(attempts, 4242, null).length, 2);
100
+ });
101
+
102
+ test('readServerAttempts: unreadable is null (unknown), never graded is [] (a real zero)', async () => {
103
+ assert.equal(await cap.readServerAttempts({}, 4242), null);
104
+ assert.equal(await cap.readServerAttempts({ tasks: { getTasksId: async () => ({ ok: false, status: 502 }) } }, 4242), null);
105
+ assert.deepEqual(await cap.readServerAttempts({ tasks: { getTasksId: async () => ({ ok: true, data: { task: {}, grade: null } }) } }, 4242), []);
106
+ });
107
+
108
+ // The early refusal (the courtesy half): ship.js, with any flag. It must refuse BEFORE a
109
+ // session is loaded or a panel started — the panel spend is what the cap saves.
110
+ for (const flags of [['--regrade'], [], ['--skip-grade']]) {
111
+ test(`ship.js ${flags.join(' ') || '(plain re-ship)'} refuses a third grade in an autonomous run`, () => {
112
+ const { env } = ledgerWith([fail, fail]);
113
+ const home = fs.mkdtempSync(path.join(os.tmpdir(), 'grade-cap-home-'));
114
+ const res = spawnSync(process.execPath, [path.join(REPO, 'scripts', 'gds', 'ship.js'), '4242', ...flags], {
115
+ cwd: REPO,
116
+ encoding: 'utf8',
117
+ timeout: 30000,
118
+ // No session and a dead API base: if the refusal did NOT fire first, the
119
+ // run fails some other way and the assertions below say so.
120
+ env: { ...process.env, ...env, HOME: home, USERPROFILE: home, XDG_CONFIG_HOME: home, BONGOS_API_BASE: 'http://127.0.0.1:9' },
121
+ });
122
+ assert.equal(res.status, cap.REFUSED_EXIT, `expected the cap's exit code; stderr:\n${res.stderr}`);
123
+ assert.match(res.stderr, /graded 2 time\(s\) in this autonomous run/);
124
+ });
125
+ }
@@ -619,9 +619,22 @@ test('a mid-ship strand is named in the run log and blocked, but not released',
619
619
  }
620
620
  });
621
621
 
622
- test('a ship with no artifact on main is recorded as unverified', async () => {
622
+ test('a fresh ship with nothing on main yet is pending_land: unverified, no blocker (task 1004538)', async () => {
623
+ // The harness ledger shipped just now — well inside the land grace window.
623
624
  const h = harness({ probeArtifact: async () => ({ checked: true, onMain: false, sha: null, subject: null }) });
624
625
  const row = await runner.iteration(opts, h.deps);
626
+ assert.equal(row.ledger_shape, 'pending_land', 'the run log row is what the next loop re-probes from');
627
+ assert.equal(row.verified, false);
628
+ assert.deepEqual(h.blockers, [], 'a slow land is not an owner problem yet');
629
+ });
630
+
631
+ test('a ship with no artifact on main is recorded as unverified', async () => {
632
+ const h = harness({
633
+ probeArtifact: async () => ({ checked: true, onMain: false, sha: null, subject: null }),
634
+ // Shipped two days ago: past the land grace window.
635
+ ledger: { id: 4242, status: 'shipped', shipped_at: new Date(Date.now() - 48 * 3600_000).toISOString() },
636
+ });
637
+ const row = await runner.iteration(opts, h.deps);
625
638
  assert.equal(row.ledger_status, 'shipped');
626
639
  assert.equal(row.ledger_shape, 'shipped_no_artifact');
627
640
  assert.equal(row.verified, false);
@@ -1379,3 +1392,219 @@ test('the server heartbeat carries the goal the current task came from', async (
1379
1392
  assert.ok(b, 'the beat after a worked row must name its task');
1380
1393
  assert.equal(b.working_goal_id, 1000090);
1381
1394
  });
1395
+
1396
+ // ── the grade cap (task 1004536) ────────────────────────────────────────────────
1397
+ // Owner rule: a task that fails the grade twice goes to the owner. The prompt
1398
+ // asked for it and was ignored (task 1004512: $63.81 over three rounds), so the
1399
+ // supervisor reads the run's grade ledger itself.
1400
+ const gradeCapLib = require('../scripts/gds/autobongos-grade-cap.js');
1401
+ const fakeGrade = (passed, outage = false) => ({ passed, score: passed ? 8 : 5, signals: { panel_outcome: outage ? 'unavailable' : (passed ? 'pass' : 'fail') } });
1402
+ function ledgerHarness(grades, { ledgerStatus = 'completed' } = {}) {
1403
+ const file = join(mkdtempSync(join(tmpdir(), 'abgcap-')), 'run.jsonl');
1404
+ let handed = null;
1405
+ const h = harness({
1406
+ ledger: { id: 4242, status: ledgerStatus, updated_at: new Date().toISOString() },
1407
+ spawnWorker: async (a) => {
1408
+ handed = a.gradeLedger;
1409
+ // What ship.js does inside the worker: one row per graded round.
1410
+ for (const g of grades) gradeCapLib.recordGradeOutcome({ taskId: 4242, grade: g, env: { [gradeCapLib.LEDGER_ENV]: a.gradeLedger } });
1411
+ // ...and the worker IGNORING its brief, claiming a ship anyway.
1412
+ return { envelope: { is_error: false, result: '', structured_output: {
1413
+ outcome: 'shipped', task_id: 4242, what_happened: 'fixed and re-graded', undetermined_decisions: [],
1414
+ } }, exitCode: 0 };
1415
+ },
1416
+ });
1417
+ h.deps.gradeLedgerPath = () => file;
1418
+ return { h, file, handed: () => handed };
1419
+ }
1420
+
1421
+ test('the brief states the grade cap: one re-grade, then cannot_complete; an outage does not count', () => {
1422
+ const p = buildWorkerPrompt({ task: { id: 7, title: 't', description: 'd' } });
1423
+ assert.match(p, /at most ONE re-grade/);
1424
+ assert.match(p, /fails twice, STOP/);
1425
+ assert.match(p, /GRADER UNAVAILABLE[^.]*does not count/);
1426
+ });
1427
+
1428
+ test('two graded FAILs in one run end it as needs_human, whatever the worker claims', async () => {
1429
+ const { h, file, handed } = ledgerHarness([fakeGrade(false), fakeGrade(false)]);
1430
+ const row = await runner.iteration(opts, h.deps);
1431
+ assert.equal(handed(), file, 'the supervisor picks the ledger and hands it to the worker');
1432
+ assert.equal(row.outcome, 'needs_human', 'the worker said "shipped"; the ledger says the grade failed twice');
1433
+ assert.match(row.reason, /failed 2 times/);
1434
+ assert.equal(row.grade_rounds, 2);
1435
+ assert.equal(row.grade_cap_reached, true);
1436
+ assert.equal(h.blockers.length, 1, 'the owner gets a blocker — the existing needs_human path');
1437
+ assert.throws(() => readFileSync(file), /ENOENT/, 'the per-run ledger is removed once read; its counts are in the row');
1438
+ });
1439
+
1440
+ test('a grader outage between two rounds does not end the run', async () => {
1441
+ const { h } = ledgerHarness([fakeGrade(false), fakeGrade(false, true), fakeGrade(true)], { ledgerStatus: 'shipped' });
1442
+ const row = await runner.iteration(opts, h.deps);
1443
+ assert.equal(row.outcome, 'claims_shipped');
1444
+ assert.equal(row.grade_rounds, 2);
1445
+ assert.equal(row.grade_outages, 1);
1446
+ assert.equal(row.grade_cap_reached, false);
1447
+ });
1448
+
1449
+ // The grader's hacker finding on the first cut: the local ledger lives in the
1450
+ // worker's env and filesystem, so a worker can clear it. The supervisor's count
1451
+ // comes from the server's grade_attempts, which the worker cannot write.
1452
+ const attempt = (at, passed, { outage = false } = {}) => ({
1453
+ attempted_at: at, passed,
1454
+ workers: ['narc', 'quality', 'hacker', 'efficiency'].map((kind) => ({
1455
+ kind, verdict: outage ? null : (passed ? 'pass' : 'fail'), error: outage && kind !== 'narc' ? 'SUBAGENT_SPAWN_FAILED' : null, severities: [],
1456
+ })),
1457
+ });
1458
+ function serverApi(before, during) {
1459
+ let calls = 0;
1460
+ return { tasks: { getTasksId: async ({ query }) => {
1461
+ assert.match(query.include, /grade_attempts/);
1462
+ const attempts = calls++ === 0 ? before : during;
1463
+ return { ok: true, status: 200, data: { task: { id: 4242 }, grade: { attempts } } };
1464
+ } } };
1465
+ }
1466
+
1467
+ test('a worker that clears the local ledger is still capped by the server count', async () => {
1468
+ const old = attempt('2026-10-01T09:00:00.000Z', false); // a PREVIOUS run's round — not this run's
1469
+ const h = harness({
1470
+ ledger: { id: 4242, status: 'completed', updated_at: new Date().toISOString() },
1471
+ api: serverApi([old], [old, attempt('2026-10-02T10:00:00.000Z', false), attempt('2026-10-02T10:20:00.000Z', false)]),
1472
+ // Records NOTHING locally — as if it ran ship.js with the ledger env cleared.
1473
+ spawnWorker: async () => ({ envelope: { is_error: false, result: '', structured_output: {
1474
+ outcome: 'shipped', task_id: 4242, what_happened: 'kept going', undetermined_decisions: [],
1475
+ } }, exitCode: 0 }),
1476
+ });
1477
+ h.deps.gradeLedgerPath = () => join(mkdtempSync(join(tmpdir(), 'abgcap-')), 'run.jsonl');
1478
+ const row = await runner.iteration(opts, h.deps);
1479
+ assert.equal(row.outcome, 'needs_human');
1480
+ assert.equal(row.grade_count_source, 'server');
1481
+ assert.equal(row.grade_rounds, 2, 'the round from before this run started is not counted');
1482
+ assert.equal(row.grade_cap_reached, true);
1483
+ });
1484
+
1485
+ test('server rounds where every gating worker errored are outages, not fails', async () => {
1486
+ const h = harness({
1487
+ ledger: { id: 4242, status: 'completed', updated_at: new Date().toISOString() },
1488
+ api: serverApi([], [attempt('2026-10-02T10:00:00.000Z', false), attempt('2026-10-02T10:05:00.000Z', false, { outage: true })]),
1489
+ });
1490
+ h.deps.gradeLedgerPath = () => join(mkdtempSync(join(tmpdir(), 'abgcap-')), 'run.jsonl');
1491
+ const row = await runner.iteration(opts, h.deps);
1492
+ assert.notEqual(row.outcome, 'needs_human');
1493
+ assert.equal(row.grade_rounds, 1);
1494
+ assert.equal(row.grade_outages, 1);
1495
+ assert.equal(row.grade_cap_reached, false);
1496
+ });
1497
+
1498
+ test('spawnWorker hands the ledger to the worker in env and stops it once the cap is reached', async () => {
1499
+ const { EventEmitter } = await import('node:events');
1500
+ const file = join(mkdtempSync(join(tmpdir(), 'abgcap-')), 'run.jsonl');
1501
+ let env = null; let killed = false;
1502
+ const spawn = (_bin, _args, o) => {
1503
+ env = o.env;
1504
+ const child = new EventEmitter();
1505
+ child.stdout = new EventEmitter(); child.stderr = new EventEmitter();
1506
+ child.stdin = { end: () => {
1507
+ // The worker's ship.js records its second graded FAIL, then the worker keeps going.
1508
+ for (const g of [fakeGrade(false), fakeGrade(false)]) gradeCapLib.recordGradeOutcome({ taskId: 4242, grade: g, env: o.env });
1509
+ } };
1510
+ child.kill = () => { killed = true; setImmediate(() => child.emit('close', null)); };
1511
+ return child;
1512
+ };
1513
+ const capCheck = async () => gradeCapLib.gradeCapState(gradeCapLib.readLedger(file), 4242);
1514
+ const res = await runner.spawnWorker({
1515
+ task: aTask, model: 'm', timeoutMs: 60_000, cwd: process.cwd(), gradeLedger: file, capCheck,
1516
+ deps: { spawn, resolveBin: () => ({ bin: 'claude', shell: false }), sessionId: 's', capWatchMs: 5 },
1517
+ });
1518
+ assert.equal(env[gradeCapLib.LEDGER_ENV], file);
1519
+ assert.equal(killed, true, 'a worker past the cap is stopped, not left to keep paying for rounds');
1520
+ assert.equal(res.gradeCapped, true);
1521
+ assert.equal(res.timedOut, false);
1522
+ });
1523
+
1524
+ // ── slow lands are re-checked every loop, across restarts (task 1004538) ──────
1525
+
1526
+ test('forever re-checks pending lands once per loop, and a failing check never stops it', async () => {
1527
+ let calls = 0;
1528
+ const rows = [];
1529
+ const code = await runner.forever({ goals: [], maxTasks: 1, maxLoops: 3 }, {
1530
+ iteration: async () => ({ event: 'hold', reason: 'test' }),
1531
+ reconcileLeftoverClaims: async () => ({ event: 'boot_reconcile', ok: true }),
1532
+ waitOrJump: async () => ({ waited: 0, jumped: false }),
1533
+ writeHeartbeat: () => true,
1534
+ postHeartbeat: async () => true,
1535
+ emit: () => {},
1536
+ record: (r) => { rows.push(r); return r; },
1537
+ codeDrifted: async () => false,
1538
+ loadedHead: async () => null,
1539
+ recheckLands: async () => { calls += 1; if (calls === 2) throw new Error('fetch failed'); return {}; },
1540
+ });
1541
+ assert.equal(code, 0);
1542
+ assert.equal(calls, 3, 'every loop, not just at boot');
1543
+ assert.ok(rows.some((r) => r.event === 'land_recheck_failed' && /fetch failed/.test(r.errors.join(' '))));
1544
+ });
1545
+
1546
+ test('main() arms the land re-check for --forever', () => {
1547
+ const src = readFileSync(new URL('../scripts/gds/autobongos-run.js', import.meta.url), 'utf8');
1548
+ const line = src.split('\n').find((l) => l.includes('if (opts.forever) return forever('));
1549
+ assert.match(line, /recheckLands: true/, 'without this the runner never re-probes a slow land');
1550
+ });
1551
+
1552
+ test('a pending land survives a restart: the run log names it, the ledger clocks it, the blocker closes', async () => {
1553
+ // A FRESH process: nothing in memory. Only the run log on disk says task 77
1554
+ // was pending, and only the database says blocker 500 is open for task 78.
1555
+ const dir = mkdtempSync(join(tmpdir(), 'autobongos-lands-'));
1556
+ process.env.AUTOBONGOS_LOG_FILE = join(dir, 'runs.jsonl');
1557
+ try {
1558
+ writeFileSync(process.env.AUTOBONGOS_LOG_FILE, [
1559
+ JSON.stringify({ event: 'worked', task_id: '77', ledger_shape: 'pending_land' }),
1560
+ JSON.stringify({ event: 'worked', task_id: '76', ledger_shape: 'shipped' }),
1561
+ '{"event":"worked","task_id":"75","ledger_sh', // the half line a kill -9 leaves
1562
+ ].join('\n'));
1563
+ const recorded = [];
1564
+ const resolved = [];
1565
+ const probed = [];
1566
+ const api = {
1567
+ tasks: { getTasksId: async ({ id }) => ({ ok: true, data: { task: { id: String(id), status: 'shipped', shipped_at: new Date(Date.now() - 3600_000).toISOString() } } }) },
1568
+ blockers: {
1569
+ getBlockers: async () => ({ ok: true, data: { blockers: [{ id: '500', source: 'autobongos', source_ref: 'task-78-shipped_no_artifact', status: 'open' }] } }),
1570
+ postBlockersIdResolve: async ({ id }) => { resolved.push(String(id)); return { ok: true, data: {} }; },
1571
+ },
1572
+ };
1573
+ await runner.recheckPendingLands({
1574
+ api,
1575
+ state: runner.newRunnerState(),
1576
+ record: (r) => { recorded.push(r); return r; },
1577
+ landDeps: { probeArtifact: async (id) => { probed.push(String(id)); return { checked: true, onMain: true, sha: 'abc123def456', subject: `Merge from x/task-${id}` }; } },
1578
+ });
1579
+ assert.deepEqual(probed.sort(), ['77', '78'], 'only the pending task and the open blocker are probed');
1580
+ const re = recorded.find((r) => r.event === 'land_recheck');
1581
+ assert.equal(re.task_id, '77');
1582
+ assert.equal(re.ledger_shape, 'shipped', 'the land arrived, so the next pass drops it from the pending list');
1583
+ assert.deepEqual(resolved, ['500']);
1584
+ assert.ok(recorded.some((r) => r.event === 'land_blocker_resolved' && r.blocker_id === '500'));
1585
+ } finally { delete process.env.AUTOBONGOS_LOG_FILE; }
1586
+ });
1587
+
1588
+ test('the blocker sweep runs at most hourly; with nothing pending, the loops between cost nothing', async () => {
1589
+ const dir = mkdtempSync(join(tmpdir(), 'autobongos-sweep-'));
1590
+ process.env.AUTOBONGOS_LOG_FILE = join(dir, 'runs.jsonl'); // empty: nothing pending
1591
+ try {
1592
+ let lists = 0;
1593
+ let now = 1_000_000_000_000;
1594
+ const deps = {
1595
+ state: runner.newRunnerState(),
1596
+ now: () => now,
1597
+ record: (r) => r,
1598
+ api: { tasks: {}, blockers: { getBlockers: async () => { lists += 1; return { ok: true, data: { blockers: [] } }; } } },
1599
+ landDeps: { probeArtifact: async () => notOnMainRow },
1600
+ };
1601
+ const notOnMainRow = { checked: true, onMain: false, sha: null, subject: null };
1602
+ await runner.recheckPendingLands(deps);
1603
+ now += 5 * 60 * 1000;
1604
+ await runner.recheckPendingLands(deps);
1605
+ assert.equal(lists, 1, 'a second sweep five minutes later is wasted I/O');
1606
+ now += runner.LAND_SWEEP_MS;
1607
+ await runner.recheckPendingLands(deps);
1608
+ assert.equal(lists, 2, 'an hour on, it sweeps again');
1609
+ } finally { delete process.env.AUTOBONGOS_LOG_FILE; }
1610
+ });