agent-dealer 1.2.7 → 1.2.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/bundle/server/dist/adapters/agent-deck-bind.js +16 -6
  2. package/bundle/server/dist/adapters/agent-deck-bind.test.js +74 -0
  3. package/bundle/server/dist/adapters/agent-health.js +14 -3
  4. package/bundle/server/dist/adapters/github.js +4 -1
  5. package/bundle/server/dist/adapters/muse-capability.js +443 -24
  6. package/bundle/server/dist/adapters/muse-capability.test.js +469 -25
  7. package/bundle/server/dist/adapters/muse-visual-qa.js +114 -0
  8. package/bundle/server/dist/adapters/muse-visual-qa.test.js +68 -0
  9. package/bundle/server/dist/capacity/muse-probe.js +56 -9
  10. package/bundle/server/dist/capacity/muse-probe.test.js +188 -1
  11. package/bundle/server/dist/coordinator/admission.js +14 -1
  12. package/bundle/server/dist/coordinator/admission.test.js +197 -4
  13. package/bundle/server/dist/coordinator/auto-merge.integration.test.js +276 -0
  14. package/bundle/server/dist/coordinator/auto-merge.js +39 -3
  15. package/bundle/server/dist/coordinator/commands.js +90 -5
  16. package/bundle/server/dist/coordinator/deck-outage.integration.test.js +4 -2
  17. package/bundle/server/dist/coordinator/developer-effect.js +79 -1
  18. package/bundle/server/dist/coordinator/execution-report.js +6 -0
  19. package/bundle/server/dist/coordinator/execution-report.test.js +7 -0
  20. package/bundle/server/dist/coordinator/failure-cause.js +64 -1
  21. package/bundle/server/dist/coordinator/failure-cause.test.js +92 -1
  22. package/bundle/server/dist/coordinator/failure-reason.js +7 -0
  23. package/bundle/server/dist/coordinator/failure-reason.test.js +14 -0
  24. package/bundle/server/dist/coordinator/human-resolution.js +17 -1
  25. package/bundle/server/dist/coordinator/merge-conflict-sync.js +647 -0
  26. package/bundle/server/dist/coordinator/merge-conflict-sync.test.js +121 -0
  27. package/bundle/server/dist/coordinator/muse-developer.integration.test.js +9 -4
  28. package/bundle/server/dist/coordinator/muse-spawn.js +122 -29
  29. package/bundle/server/dist/coordinator/muse-spawn.test.js +210 -0
  30. package/bundle/server/dist/coordinator/playbook-feedback.js +690 -0
  31. package/bundle/server/dist/coordinator/playbook-feedback.test.js +702 -0
  32. package/bundle/server/dist/coordinator/prompts-execution-contract.test.js +107 -0
  33. package/bundle/server/dist/coordinator/prompts.js +78 -2
  34. package/bundle/server/dist/coordinator/prompts.test.js +83 -0
  35. package/bundle/server/dist/coordinator/reflect-trigger.js +33 -169
  36. package/bundle/server/dist/coordinator/reflect-trigger.test.js +149 -200
  37. package/bundle/server/dist/coordinator/reviewer-effect.js +10 -1
  38. package/bundle/server/dist/coordinator/reviewer-result.js +6 -0
  39. package/bundle/server/dist/coordinator/routing.test.js +14 -0
  40. package/bundle/server/dist/coordinator/session-timeouts.js +30 -0
  41. package/bundle/server/dist/coordinator/usage-cap.integration.test.js +1 -1
  42. package/bundle/server/dist/coordinator/worker-loop.js +18 -4
  43. package/bundle/server/dist/db/index.js +5 -0
  44. package/bundle/server/dist/db/schema.sql +3 -0
  45. package/bundle/server/dist/docs-execution-analysis.test.js +1 -0
  46. package/bundle/server/dist/repository/artifacts-for-issue.js +3 -3
  47. package/bundle/server/dist/repository/human-actions.js +14 -0
  48. package/bundle/server/dist/repository/issues.js +29 -4
  49. package/bundle/server/dist/repository/worker-sessions.js +51 -2
  50. package/bundle/server/dist/routes/human-actions.js +10 -8
  51. package/bundle/server/dist/routes/issues-execution-contract.test.js +321 -0
  52. package/bundle/server/dist/routes/issues.js +19 -2
  53. package/bundle/server/dist/runners/muse-code-jsonl.js +106 -11
  54. package/bundle/server/dist/runners/muse-config-core.js +18 -1
  55. package/bundle/server/dist/runners/muse-config.test.js +37 -0
  56. package/bundle/server/dist/runners/muse-serve-session.js +5 -0
  57. package/bundle/server/dist/runners/spawn-cli.js +116 -0
  58. package/bundle/server/dist/runners/spawn-cli.test.js +124 -0
  59. package/bundle/server/package.json +2 -2
  60. package/bundle/server/static-ui/assets/{index-yLyxRd-7.js → index-B6SVCzMR.js} +14 -14
  61. package/bundle/server/static-ui/assets/{index-DyAJNyfV.css → index-K_YcYkQU.css} +1 -1
  62. package/bundle/server/static-ui/index.html +2 -2
  63. package/bundle/shared/dist/execution-analysis.d.ts +22 -22
  64. package/bundle/shared/dist/execution-contract.d.ts +117 -0
  65. package/bundle/shared/dist/execution-contract.js +307 -0
  66. package/bundle/shared/dist/execution-contract.test.d.ts +2 -0
  67. package/bundle/shared/dist/execution-contract.test.js +499 -0
  68. package/bundle/shared/dist/execution-report.d.ts +9 -9
  69. package/bundle/shared/dist/execution-report.js +2 -0
  70. package/bundle/shared/dist/failure-cause.d.ts +4 -4
  71. package/bundle/shared/dist/failure-cause.js +2 -0
  72. package/bundle/shared/dist/human-actions.d.ts +4 -4
  73. package/bundle/shared/dist/human-actions.js +9 -0
  74. package/bundle/shared/dist/index.d.ts +85 -84
  75. package/bundle/shared/dist/index.js +3 -0
  76. package/bundle/shared/dist/issues.d.ts +146 -16
  77. package/bundle/shared/dist/issues.js +8 -0
  78. package/bundle/shared/dist/issues.test.js +1 -0
  79. package/bundle/shared/dist/outbound-draft.d.ts +12 -12
  80. package/bundle/shared/dist/worker-sessions.d.ts +10 -0
  81. package/bundle/shared/dist/worker-sessions.js +8 -0
  82. package/bundle/shared/dist/worker-sessions.test.js +1 -0
  83. package/bundle/shared/dist/workflow.d.ts +4 -4
  84. package/bundle/shared/dist/workflow.js +10 -0
  85. package/bundle/shared/dist/workflow.test.js +2 -0
  86. package/bundle/shared/package.json +1 -1
  87. package/package.json +1 -1
@@ -24,7 +24,10 @@ after(() => {
24
24
  // best-effort — a failed rm must not fail the suite
25
25
  }
26
26
  });
27
- const { museCapabilityIssues, museCapabilityCheckInFlight, parseMuseVersion, settleMuseCapabilityCheckForTests, setMuseCapabilityProbeForTests, resetMuseCapabilityStateForTests, ageMuseCapabilityCheckForTests, defaultMuseCapabilityProbe, } = await import("./muse-capability.js");
27
+ const { museCapabilityIssues, museCapabilityCheckInFlight, parseMuseVersion, settleMuseCapabilityCheckForTests, setMuseCapabilityProbeForTests, resetMuseCapabilityStateForTests, ageMuseCapabilityCheckForTests, ensureMuseCapabilityEscalation, recordMuseCapabilityOverride, museCapabilitySafetyNetAfterSession, countMuseShellToolCalls, museCapabilityRequestId, defaultMuseCapabilityProbe, } = await import("./muse-capability.js");
28
+ const { migrate, getDb } = await import("../db/index.js");
29
+ const { listHumanActionsByRequestId, resolveHumanAction } = await import("../repository/human-actions.js");
30
+ migrate();
28
31
  const OLD = "1.3.0-R3401.1";
29
32
  const NEW = "1.4.0-R4161.1";
30
33
  const STATE = path.join(process.env.AGENT_DEALER_HOME, "muse-capability.json");
@@ -48,16 +51,63 @@ async function check(version) {
48
51
  await settleMuseCapabilityCheckForTests();
49
52
  return { first, settled: museCapabilityIssues(version) };
50
53
  }
54
+ /** Drive `version` to three consecutive errors (attempts=3, exhausted). */
55
+ async function exhaust(version) {
56
+ museCapabilityIssues(version);
57
+ await settleMuseCapabilityCheckForTests();
58
+ ageMuseCapabilityCheckForTests(61_000);
59
+ museCapabilityIssues(version);
60
+ await settleMuseCapabilityCheckForTests();
61
+ ageMuseCapabilityCheckForTests(122_000);
62
+ museCapabilityIssues(version);
63
+ await settleMuseCapabilityCheckForTests();
64
+ }
65
+ /** A `muse --version` stub reporting `version`; returns a restore function. */
66
+ function stubMuseVersion(version) {
67
+ const dir = fs.mkdtempSync(path.join(os.tmpdir(), "dealer-muse-verstub-"));
68
+ const versionFile = path.join(dir, "version");
69
+ fs.writeFileSync(versionFile, `Muse Code ${version.split("-")[0]} (${version})\n`);
70
+ const bin = path.join(dir, "muse");
71
+ fs.writeFileSync(bin, `#!/bin/sh\ncat ${JSON.stringify(versionFile)}\n`);
72
+ fs.chmodSync(bin, 0o755);
73
+ const prev = process.env.MUSE_CLI;
74
+ process.env.MUSE_CLI = bin;
75
+ return () => {
76
+ if (prev === undefined)
77
+ delete process.env.MUSE_CLI;
78
+ else
79
+ process.env.MUSE_CLI = prev;
80
+ };
81
+ }
82
+ /** A real issue row — `muse_capability` escalations are FK-bound to issues. */
83
+ function makeIssue() {
84
+ const id = randomUUID();
85
+ const now = new Date().toISOString();
86
+ getDb()
87
+ .prepare(`INSERT INTO issues (id, source, title, repo, base_branch, status, current_owner, created_at, updated_at)
88
+ VALUES (?, 'manual', 'muse capability check', 'dealer-test', 'main', 'ready', 'dealer', ?, ?)`)
89
+ .run(id, now, now);
90
+ return id;
91
+ }
92
+ /** A session-log file from JSON lines (plain strings pass through verbatim). */
93
+ function writeLog(lines) {
94
+ const p = path.join(fs.mkdtempSync(path.join(os.tmpdir(), "dealer-muse-log-")), "session.ndjson");
95
+ fs.writeFileSync(p, `${lines.map((l) => (typeof l === "string" ? l : JSON.stringify(l))).join("\n")}\n`);
96
+ return p;
97
+ }
51
98
  describe("muse-capability", { concurrency: false }, () => {
52
99
  beforeEach(() => {
53
100
  resetMuseCapabilityStateForTests();
54
101
  setMuseCapabilityProbeForTests(null);
102
+ getDb().exec("DELETE FROM human_actions");
55
103
  });
56
104
  test("parseMuseVersion reads the build id from `muse --version`", () => {
57
105
  assert.equal(parseMuseVersion("Muse Code 1.3.0 (1.3.0-R3401.1)\n"), OLD);
58
106
  assert.equal(parseMuseVersion("1.4.0-R4161.1\n"), NEW);
59
107
  assert.equal(parseMuseVersion(" \n"), null);
60
108
  });
109
+ // NOT-308: with no confirmed baseline at all (fresh install) the unknown still
110
+ // blocks — fail closed until the first version is confirmed.
61
111
  test("a version not yet checked blocks while its one-time check runs (never assumed capable)", async () => {
62
112
  let release;
63
113
  setMuseCapabilityProbeForTests(() => new Promise((resolve) => (release = () => resolve({ status: "capable" }))));
@@ -72,6 +122,23 @@ describe("muse-capability", { concurrency: false }, () => {
72
122
  assert.equal(museCapabilityCheckInFlight(), false);
73
123
  assert.deepEqual(museCapabilityIssues(NEW), []);
74
124
  });
125
+ // NOT-308: with a confirmed baseline, an unchecked version never blocks — the
126
+ // one-time check runs in the background while admission proceeds on the baseline.
127
+ test("a version not yet checked does NOT block while a confirmed baseline exists", async () => {
128
+ stubProbe({ [OLD]: { status: "capable" } });
129
+ await check(OLD);
130
+ let release;
131
+ setMuseCapabilityProbeForTests(() => new Promise((resolve) => (release = () => resolve({ status: "capable" }))));
132
+ assert.deepEqual(museCapabilityIssues(NEW), []);
133
+ assert.equal(museCapabilityCheckInFlight(), true);
134
+ // Repeated polls while in flight stay unblocked and start no second probe.
135
+ assert.deepEqual(museCapabilityIssues(NEW), []);
136
+ assert.equal(museCapabilityCheckInFlight(), true);
137
+ release();
138
+ await settleMuseCapabilityCheckForTests();
139
+ assert.deepEqual(museCapabilityIssues(NEW), []);
140
+ assert.equal(JSON.parse(fs.readFileSync(STATE, "utf8")).confirmedVersion, NEW);
141
+ });
75
142
  // (a) same version as last check → no re-probe, cached result reused.
76
143
  test("the same version is checked once; later reads reuse the cached result", async () => {
77
144
  const calls = stubProbe({ [OLD]: { status: "capable" } });
@@ -97,7 +164,7 @@ describe("muse-capability", { concurrency: false }, () => {
97
164
  const calls = stubProbe({ [OLD]: { status: "capable" }, [NEW]: { status: "capable" } });
98
165
  await check(OLD);
99
166
  const { first, settled } = await check(NEW);
100
- assert.match(first[0].message, new RegExp(`Muse Code updated ${OLD} → ${NEW}: verifying`));
167
+ assert.deepEqual(first, [], "the in-flight check does not block on a baseline");
101
168
  assert.deepEqual(settled, []);
102
169
  assert.equal(calls.get(NEW), 1);
103
170
  const saved = JSON.parse(fs.readFileSync(STATE, "utf8"));
@@ -132,39 +199,65 @@ describe("muse-capability", { concurrency: false }, () => {
132
199
  await check(NEW);
133
200
  assert.deepEqual((await check(FIXED)).settled, []);
134
201
  });
135
- // (d) the check itself errors / times out → blocked with a distinct "could not verify" message.
202
+ // NOT-308 (d): an inconclusive check (throws / times out) with a confirmed baseline
203
+ // never blocks — admission proceeds on the baseline while the retry runs out.
136
204
  for (const [label, result] of [
137
205
  ["throws", new Error("spawn EACCES")],
138
206
  ["times out", { status: "error", detail: `probe session on ${NEW} timed out` }],
139
207
  ]) {
140
- test(`a check that ${label} fails closed with a distinct "could not verify" message`, async () => {
208
+ test(`a check that ${label} does NOT block while a confirmed baseline exists`, async () => {
141
209
  const calls = stubProbe({ [OLD]: { status: "capable" }, [NEW]: result });
142
210
  await check(OLD);
143
- const { settled } = await check(NEW);
144
- assert.deepEqual(settled.map((i) => i.code), ["runtime_capability"]);
145
- assert.match(settled[0].message, new RegExp(`^Could not verify Muse Code developer shell/write access after version change \\(${OLD} → ${NEW}\\)`));
146
- assert.doesNotMatch(settled[0].message, /no longer get/);
147
- // Not re-probed before the retry backoff, and the block stays up meanwhile.
148
- assert.deepEqual(museCapabilityIssues(NEW), settled);
211
+ const { first, settled } = await check(NEW);
212
+ assert.deepEqual(first, [], "the in-flight check does not block on a baseline");
213
+ assert.deepEqual(settled, [], "an inconclusive result does not block on a baseline");
214
+ // Not re-probed before the retry backoff.
215
+ assert.deepEqual(museCapabilityIssues(NEW), []);
149
216
  assert.equal(calls.get(NEW), 1);
150
217
  assert.equal(JSON.parse(fs.readFileSync(STATE, "utf8")).confirmedVersion, OLD);
151
218
  });
152
219
  }
153
- test("a could-not-verify result is retried after the backoff, staying blocked until it passes", async () => {
154
- let next = { status: "error", detail: "timed out" };
155
- let calls = 0;
156
- setMuseCapabilityProbeForTests(async () => {
157
- calls += 1;
158
- return next;
220
+ // NOT-308: inconclusive results retry at 1 min, then 2 min — then the version is
221
+ // exhausted (3 attempts), escalates, and is never re-probed (not even past 4 min,
222
+ // and never on the old 10-minute flat timer). Admission stays unblocked throughout.
223
+ test("inconclusive results back off 1 min, then 2 min, then stop after 3 attempts", async () => {
224
+ const calls = stubProbe({
225
+ [OLD]: { status: "capable" },
226
+ [NEW]: { status: "error", detail: "timed out" },
159
227
  });
160
- await check(NEW);
161
- next = { status: "capable" };
162
- ageMuseCapabilityCheckForTests(11 * 60_000);
163
- const retrying = museCapabilityIssues(NEW);
164
- assert.match(retrying[0].message, /Could not verify/);
228
+ await check(OLD);
229
+ assert.equal(calls.get(OLD), 1);
230
+ // Attempt 1 settles inconclusive; admission stays open.
231
+ museCapabilityIssues(NEW);
165
232
  await settleMuseCapabilityCheckForTests();
166
- assert.equal(calls, 2);
167
233
  assert.deepEqual(museCapabilityIssues(NEW), []);
234
+ assert.equal(calls.get(NEW), 1);
235
+ assert.equal(JSON.parse(fs.readFileSync(STATE, "utf8")).lastChecked.attempts, 1);
236
+ // No retry before 1 min, retry once it elapses (never the old 10 min flat wait —
237
+ // 61s must already re-probe).
238
+ ageMuseCapabilityCheckForTests(59_000);
239
+ museCapabilityIssues(NEW);
240
+ assert.equal(calls.get(NEW), 1);
241
+ ageMuseCapabilityCheckForTests(2_000);
242
+ museCapabilityIssues(NEW);
243
+ await settleMuseCapabilityCheckForTests();
244
+ assert.equal(calls.get(NEW), 2);
245
+ assert.deepEqual(museCapabilityIssues(NEW), []);
246
+ assert.equal(JSON.parse(fs.readFileSync(STATE, "utf8")).lastChecked.attempts, 2);
247
+ // Attempt 2 backs off 2 min: 1 more minute is not enough, 2 are.
248
+ ageMuseCapabilityCheckForTests(61_000);
249
+ museCapabilityIssues(NEW);
250
+ assert.equal(calls.get(NEW), 2);
251
+ ageMuseCapabilityCheckForTests(61_000);
252
+ museCapabilityIssues(NEW);
253
+ await settleMuseCapabilityCheckForTests();
254
+ assert.equal(calls.get(NEW), 3);
255
+ assert.deepEqual(museCapabilityIssues(NEW), []);
256
+ assert.equal(JSON.parse(fs.readFileSync(STATE, "utf8")).lastChecked.attempts, 3);
257
+ // Exhausted: no fourth probe ever — not after 4 min, not after 10.
258
+ ageMuseCapabilityCheckForTests(10 * 60_000);
259
+ assert.deepEqual(museCapabilityIssues(NEW), []);
260
+ assert.equal(calls.get(NEW), 3, "no further automatic probes after exhaustion");
168
261
  });
169
262
  test("consecutive updates name the exact previous version, not the last confirmed baseline", async () => {
170
263
  const LATER = "1.4.0-R4302.1";
@@ -176,23 +269,374 @@ describe("muse-capability", { concurrency: false }, () => {
176
269
  await check(OLD);
177
270
  await check(NEW);
178
271
  const { first, settled } = await check(LATER);
179
- assert.match(first[0].message, new RegExp(`^Muse Code updated ${NEW} → ${LATER}: verifying`));
272
+ assert.deepEqual(first, [], "the in-flight check does not block on a baseline");
180
273
  assert.match(settled[0].message, new RegExp(`^Muse Code updated ${NEW} → ${LATER}: developer sessions no longer get`));
181
274
  assert.doesNotMatch(settled[0].message, new RegExp(OLD.replace(/\./g, "\\.")));
182
275
  assert.equal(JSON.parse(fs.readFileSync(STATE, "utf8")).confirmedVersion, OLD);
183
276
  });
277
+ // NOT-308: without a baseline the inconclusive block still names the exact previous
278
+ // version (with one it never blocks, so there is no message to name it).
184
279
  test("consecutive could-not-verify updates also name the exact previous version", async () => {
185
280
  const LATER = "1.4.0-R4302.1";
186
281
  stubProbe({
187
- [OLD]: { status: "capable" },
188
282
  [NEW]: { status: "error", detail: "timed out" },
189
283
  [LATER]: { status: "error", detail: "timed out" },
190
284
  });
191
- await check(OLD);
192
285
  await check(NEW);
193
286
  const { settled } = await check(LATER);
287
+ assert.deepEqual(settled.map((i) => i.code), ["runtime_capability"]);
194
288
  assert.match(settled[0].message, new RegExp(`after version change \\(${NEW} → ${LATER}\\)`));
195
289
  });
290
+ // NOT-308: fresh-install inconclusive results fail closed until the first version is
291
+ // confirmed — and an exhausted one says it will not retry, since nothing will.
292
+ test("with no confirmed baseline, in-flight and errored checks keep admission blocked", async () => {
293
+ let release;
294
+ setMuseCapabilityProbeForTests(() => new Promise((resolve) => (release = () => resolve({ status: "capable" }))));
295
+ assert.deepEqual(museCapabilityIssues(NEW).map((i) => i.code), ["runtime_capability"]);
296
+ release();
297
+ await settleMuseCapabilityCheckForTests();
298
+ assert.deepEqual(museCapabilityIssues(NEW), []);
299
+ resetMuseCapabilityStateForTests();
300
+ stubProbe({ [NEW]: { status: "error", detail: "timed out" } });
301
+ const { settled } = await check(NEW);
302
+ assert.deepEqual(settled.map((i) => i.code), ["runtime_capability"]);
303
+ assert.match(settled[0].message, /Could not verify Muse Code developer shell\/write access/);
304
+ assert.match(settled[0].message, /the check retries automatically/);
305
+ assert.equal(JSON.parse(fs.readFileSync(STATE, "utf8")).confirmedVersion, null);
306
+ });
307
+ // NOT-308: three consecutive errors escalate exactly one "could not verify" action
308
+ // (dedupe asserted by polling twice, from two issues); admission stays unblocked.
309
+ test("an exhausted version escalates exactly one 'could not verify' action", async () => {
310
+ stubProbe({
311
+ [OLD]: { status: "capable" },
312
+ [NEW]: { status: "error", detail: "timed out" },
313
+ });
314
+ await check(OLD);
315
+ await exhaust(NEW);
316
+ assert.equal(JSON.parse(fs.readFileSync(STATE, "utf8")).lastChecked.attempts, 3);
317
+ const issueId = makeIssue();
318
+ const first = ensureMuseCapabilityEscalation(issueId);
319
+ assert.ok(first, "exhaustion raises an action");
320
+ assert.equal(first.actionType, "muse_capability");
321
+ assert.equal(first.issueId, issueId);
322
+ assert.equal(first.requestId, museCapabilityRequestId(NEW, "unverified"));
323
+ assert.match(first.reason, new RegExp(`^Could not verify Muse Code developer shell/write access for ${NEW} after 3 inconclusive checks`));
324
+ assert.match(first.reason, /last probe ran \d+s/);
325
+ assert.match(first.reason, new RegExp(`continues on the last confirmed baseline ${OLD}`));
326
+ assert.match(first.reason, /no further automatic checks/);
327
+ assert.deepEqual(JSON.parse(first.responseOptionsJson), [
328
+ { choice: "acknowledge", label: "Acknowledge — keep working on baseline" },
329
+ ]);
330
+ // A second poll — even from another issue — finds the open action, never a copy.
331
+ assert.equal(ensureMuseCapabilityEscalation(issueId)?.id, first.id);
332
+ assert.equal(ensureMuseCapabilityEscalation(randomUUID())?.id, first.id);
333
+ assert.equal(listHumanActionsByRequestId("muse_capability", museCapabilityRequestId(NEW, "unverified")).filter((a) => a.status === "open").length, 1);
334
+ assert.deepEqual(museCapabilityIssues(NEW), [], "admission stays on the baseline");
335
+ });
336
+ // NOT-308: a confirmed loss escalates exactly one action naming from → to plus the
337
+ // capability; the block lifts once the operator acknowledges that version.
338
+ test("a missing version escalates once with both versions named; acknowledge lifts the block", async () => {
339
+ const calls = stubProbe({
340
+ [OLD]: { status: "capable" },
341
+ [NEW]: { status: "missing", detail: "probe session completed without running its shell command" },
342
+ });
343
+ await check(OLD);
344
+ const { settled } = await check(NEW);
345
+ assert.deepEqual(settled.map((i) => i.code), ["runtime_capability"]);
346
+ const issueId = makeIssue();
347
+ const action = ensureMuseCapabilityEscalation(issueId);
348
+ assert.ok(action, "a missing verdict raises an action");
349
+ assert.equal(action.actionType, "muse_capability");
350
+ assert.match(action.reason, new RegExp(`^Muse Code updated ${OLD} → ${NEW}: developer sessions no longer get shell/write access`));
351
+ assert.match(action.reason, /Last probe ran \d+s/);
352
+ assert.match(action.reason, new RegExp(`admission is blocked for ${NEW}`));
353
+ assert.match(action.question, new RegExp(`Acknowledge to admit developers on ${NEW} anyway`));
354
+ assert.match(action.question, new RegExp(`pin/roll back Muse to ${OLD}`));
355
+ assert.deepEqual(JSON.parse(action.responseOptionsJson), [
356
+ { choice: "acknowledge", label: `Acknowledge — admit on ${NEW}` },
357
+ ]);
358
+ const evidence = JSON.parse(action.evidenceJson);
359
+ assert.equal(evidence.kind, "missing");
360
+ assert.equal(evidence.version, NEW);
361
+ // Deduped per version, not per poll.
362
+ assert.equal(ensureMuseCapabilityEscalation(issueId)?.id, action.id);
363
+ assert.equal(ensureMuseCapabilityEscalation(randomUUID())?.id, action.id);
364
+ assert.equal(calls.get(NEW), 1, "a missing verdict is conclusive: never re-probed");
365
+ // The block stands until the operator acknowledges this version.
366
+ assert.deepEqual(museCapabilityIssues(NEW).map((i) => i.code), ["runtime_capability"]);
367
+ recordMuseCapabilityOverride(NEW);
368
+ assert.deepEqual(museCapabilityIssues(NEW), [], "acknowledge lifts the block");
369
+ assert.equal(ensureMuseCapabilityEscalation(issueId), null, "no re-raise after acknowledge");
370
+ });
371
+ // NOT-308 repair round 3: aborting or closing the issue that holds a `missing`
372
+ // escalation resolves its open action with a lifecycle reason (never a choice) —
373
+ // that is not the operator's capability decision, so the next poll must re-raise
374
+ // a fresh action, never record a silent override that lifts the block.
375
+ test("a missing escalation closed by abort/close re-raises instead of overriding", async () => {
376
+ for (const reason of ["aborted_by_user", "closed_by_operator"]) {
377
+ resetMuseCapabilityStateForTests();
378
+ getDb().exec("DELETE FROM human_actions");
379
+ stubProbe({
380
+ [OLD]: { status: "capable" },
381
+ [NEW]: { status: "missing", detail: "probe session completed without running its shell command" },
382
+ });
383
+ await check(OLD);
384
+ await check(NEW);
385
+ const issueId = makeIssue();
386
+ const action = ensureMuseCapabilityEscalation(issueId);
387
+ assert.ok(action, `a missing verdict raises an action (${reason})`);
388
+ // What abortIssue / closeReadyIssue do to every open action on the issue.
389
+ resolveHumanAction(action.id, "test", { reason });
390
+ // No override recorded and the block still stands.
391
+ assert.ok(!JSON.parse(fs.readFileSync(STATE, "utf8")).overriddenVersions.includes(NEW), `no silent override after ${reason}`);
392
+ assert.deepEqual(museCapabilityIssues(NEW).map((i) => i.code), ["runtime_capability"], `block stands after ${reason}`);
393
+ // The next poll — from another issue, as after the holding issue closes —
394
+ // raises a fresh action instead of admitting.
395
+ const fresh = ensureMuseCapabilityEscalation(makeIssue());
396
+ assert.ok(fresh, `re-raised after ${reason}`);
397
+ assert.notEqual(fresh.id, action.id, "a fresh action, not the resolved one");
398
+ assert.equal(fresh.status, "open");
399
+ assert.equal(fresh.requestId, museCapabilityRequestId(NEW, "missing"));
400
+ // And polling again dedupes onto the fresh open action, never a third copy.
401
+ assert.equal(ensureMuseCapabilityEscalation(makeIssue())?.id, fresh.id);
402
+ assert.equal(listHumanActionsByRequestId("muse_capability", museCapabilityRequestId(NEW, "missing")).filter((a) => a.status === "open").length, 1, "exactly one open missing action");
403
+ }
404
+ });
405
+ // NOT-308 repair round 2: one version can produce both verdicts — an exhausted error
406
+ // escalates "unverified", then a safety-net probe past exhaustion returns `missing`.
407
+ // The missing verdict must raise its own action naming from → to and the lost
408
+ // capability; dismissing the unverified action must never override the missing block.
409
+ test("a missing verdict after an unverified one raises its own action; dismissing unverified never lifts missing", async () => {
410
+ let verdict = { status: "error", detail: "timed out" };
411
+ const calls = new Map();
412
+ setMuseCapabilityProbeForTests(async (version) => {
413
+ calls.set(version, (calls.get(version) ?? 0) + 1);
414
+ if (version === OLD)
415
+ return { status: "capable" };
416
+ return verdict;
417
+ });
418
+ await check(OLD);
419
+ await exhaust(NEW);
420
+ assert.equal(calls.get(NEW), 3);
421
+ const issueId = makeIssue();
422
+ const unverified = ensureMuseCapabilityEscalation(issueId);
423
+ assert.ok(unverified, "exhaustion raises the unverified action");
424
+ assert.equal(unverified.requestId, museCapabilityRequestId(NEW, "unverified"));
425
+ // The operator dismisses "could not verify" — records no override (dedicated path).
426
+ const { resolveHumanActionAndAdvance } = await import("../coordinator/commands.js");
427
+ assert.equal(resolveHumanActionAndAdvance(unverified.id, "test", "acknowledge").ok, true);
428
+ // Fresh evidence: a dirty, shell-less session forces a probe past exhaustion, which
429
+ // now returns `missing`.
430
+ verdict = { status: "missing", detail: "probe session completed without running its shell command" };
431
+ const restore = stubMuseVersion(NEW);
432
+ try {
433
+ const log = writeLog([{ type: "tool_call", name: "read_file" }]);
434
+ assert.equal(museCapabilitySafetyNetAfterSession({ issueId, runtime: "muse_code", logPath: log, dirty: true }), "probed");
435
+ await settleMuseCapabilityCheckForTests();
436
+ const missing = listHumanActionsByRequestId("muse_capability", museCapabilityRequestId(NEW, "missing"));
437
+ assert.equal(missing.length, 1, "the missing verdict raises its own action");
438
+ assert.equal(missing[0].status, "open");
439
+ assert.match(missing[0].reason, new RegExp(`^Muse Code updated ${OLD} → ${NEW}: developer sessions no longer get shell/write access`));
440
+ assert.match(missing[0].reason, new RegExp(`admission is blocked for ${NEW}`));
441
+ // The gate blocks, and the earlier dismissal did not override this verdict.
442
+ assert.deepEqual(museCapabilityIssues(NEW).map((i) => i.code), ["runtime_capability"]);
443
+ assert.ok(!JSON.parse(fs.readFileSync(STATE, "utf8")).overriddenVersions.includes(NEW));
444
+ // Polling finds the missing action — never the dismissed unverified one, and the
445
+ // resolved unverified row does not suppress the missing escalation.
446
+ assert.equal(ensureMuseCapabilityEscalation(issueId)?.id, missing[0].id);
447
+ }
448
+ finally {
449
+ restore();
450
+ }
451
+ });
452
+ // NOT-308 repair round 2: with no confirmed baseline, an exhausted version keeps
453
+ // retrying at the maximum backoff — stopping would block admission permanently with
454
+ // no path to unblock — and the escalation says blocked, never "continues on none".
455
+ test("with no baseline, an exhausted version keeps retrying and says blocked", async () => {
456
+ const calls = stubProbe({ [NEW]: { status: "error", detail: "timed out" } });
457
+ await exhaust(NEW);
458
+ assert.equal(calls.get(NEW), 3);
459
+ assert.equal(JSON.parse(fs.readFileSync(STATE, "utf8")).lastChecked.attempts, 3);
460
+ // Still blocked, but the message promises a retry — not "no further checks".
461
+ const blocked = museCapabilityIssues(NEW);
462
+ assert.deepEqual(blocked.map((i) => i.code), ["runtime_capability"]);
463
+ assert.match(blocked[0].message, /the check retries automatically/);
464
+ assert.doesNotMatch(blocked[0].message, /no further automatic checks/);
465
+ // Past exhaustion the check still re-fires once the maximum backoff elapses.
466
+ ageMuseCapabilityCheckForTests(241_000);
467
+ museCapabilityIssues(NEW);
468
+ await settleMuseCapabilityCheckForTests();
469
+ assert.equal(calls.get(NEW), 4, "no-baseline exhaustion keeps retrying at the maximum backoff");
470
+ assert.deepEqual(museCapabilityIssues(NEW).map((i) => i.code), ["runtime_capability"]);
471
+ const issueId = makeIssue();
472
+ const action = ensureMuseCapabilityEscalation(issueId);
473
+ assert.ok(action, "exhaustion still escalates for a human decision");
474
+ assert.equal(action.requestId, museCapabilityRequestId(NEW, "unverified"));
475
+ assert.match(action.reason, /admission is blocked \(no confirmed baseline yet\)/);
476
+ assert.doesNotMatch(action.reason, /continues on the last confirmed baseline/);
477
+ assert.doesNotMatch(action.question, /roll back to none/);
478
+ // Dedupe still holds: a second poll finds the open action, never a copy.
479
+ assert.equal(ensureMuseCapabilityEscalation(issueId)?.id, action.id);
480
+ });
481
+ // NOT-308 repair round 2: a fresh-install `missing` names no "none" rollback target.
482
+ test("a fresh-install missing escalation never suggests rolling back to none", async () => {
483
+ stubProbe({ [NEW]: { status: "missing", detail: "no shell" } });
484
+ const { settled } = await check(NEW);
485
+ assert.deepEqual(settled.map((i) => i.code), ["runtime_capability"]);
486
+ const action = ensureMuseCapabilityEscalation(makeIssue());
487
+ assert.ok(action);
488
+ assert.doesNotMatch(action.question, /to none/);
489
+ assert.match(action.question, /Pin\/roll back Muse outside Dealer to a working build/);
490
+ });
491
+ // NOT-308: the override is version-scoped — a later regression still blocks and
492
+ // escalates on its own.
493
+ test("an acknowledged version stays overridden while a later regressed version still blocks", async () => {
494
+ const LATER = "1.4.0-R4302.1";
495
+ stubProbe({
496
+ [OLD]: { status: "capable" },
497
+ [NEW]: { status: "missing", detail: "no shell" },
498
+ [LATER]: { status: "missing", detail: "no shell" },
499
+ });
500
+ await check(OLD);
501
+ await check(NEW);
502
+ const issueId = makeIssue();
503
+ ensureMuseCapabilityEscalation(issueId);
504
+ recordMuseCapabilityOverride(NEW);
505
+ assert.deepEqual(museCapabilityIssues(NEW), []);
506
+ const { settled } = await check(LATER);
507
+ assert.match(settled[0].message, new RegExp(`^Muse Code updated ${NEW} → ${LATER}: developer sessions no longer get`));
508
+ const later = ensureMuseCapabilityEscalation(issueId);
509
+ assert.ok(later, "the later regression raises its own action");
510
+ assert.equal(later.requestId, museCapabilityRequestId(LATER, "missing"));
511
+ assert.match(later.reason, new RegExp(`admission is blocked for ${LATER}`));
512
+ });
513
+ test("countMuseShellToolCalls counts normalized and raw shell calls, null when unknown", () => {
514
+ assert.equal(countMuseShellToolCalls(writeLog([
515
+ { type: "system", subtype: "init" },
516
+ { type: "tool_call", name: "read_file" },
517
+ { type: "tool_call", name: "bash" },
518
+ { type: "assistant", message: { content: [{ type: "text", text: "hi" }] } },
519
+ { type: "result", result: "done" },
520
+ ])), 1);
521
+ assert.equal(countMuseShellToolCalls(writeLog([
522
+ {
523
+ payload_type: "task.lifecycle.side_effect_intent",
524
+ payload: { event: { operation: "tool:bash" } },
525
+ },
526
+ { payload_type: "tool.result", payload: { call_id: "c" } },
527
+ ])), 1);
528
+ assert.equal(countMuseShellToolCalls(writeLog([{ type: "tool_call", name: "write_file" }])), 0);
529
+ assert.equal(countMuseShellToolCalls(writeLog(["not json", ""])), null);
530
+ assert.equal(countMuseShellToolCalls(path.join(os.tmpdir(), `dealer-muse-nope-${randomUUID()}`)), null);
531
+ });
532
+ // NOT-308 safety net: a dirty, shell-less session ending on an unconfirmed version
533
+ // probes immediately (no backoff wait) and a `missing` verdict escalates on the issue.
534
+ test("safety net: dirty + zero-shell on an unconfirmed version probes immediately, missing escalates", async () => {
535
+ const calls = stubProbe({
536
+ [OLD]: { status: "capable" },
537
+ [NEW]: { status: "missing", detail: "probe session completed without running its shell command" },
538
+ });
539
+ await check(OLD);
540
+ const restore = stubMuseVersion(NEW);
541
+ try {
542
+ const issueId = makeIssue();
543
+ const log = writeLog([
544
+ { type: "tool_call", name: "read_file" },
545
+ { type: "assistant", message: { content: [{ type: "text", text: "edited" }] } },
546
+ ]);
547
+ const result = museCapabilitySafetyNetAfterSession({
548
+ issueId,
549
+ runtime: "muse_code",
550
+ logPath: log,
551
+ dirty: true,
552
+ });
553
+ assert.equal(result, "probed");
554
+ assert.equal(calls.get(NEW), 1, "probes immediately — never waits out a backoff");
555
+ await settleMuseCapabilityCheckForTests();
556
+ const actions = listHumanActionsByRequestId("muse_capability", museCapabilityRequestId(NEW, "missing"));
557
+ assert.equal(actions.length, 1, "the missing verdict escalates on the watched issue");
558
+ assert.equal(actions[0].issueId, issueId);
559
+ assert.equal(JSON.parse(actions[0].evidenceJson).kind, "missing");
560
+ assert.deepEqual(museCapabilityIssues(NEW).map((i) => i.code), ["runtime_capability"]);
561
+ }
562
+ finally {
563
+ restore();
564
+ }
565
+ });
566
+ test("safety net bypasses the error backoff", async () => {
567
+ const calls = stubProbe({
568
+ [OLD]: { status: "capable" },
569
+ [NEW]: { status: "error", detail: "timed out" },
570
+ });
571
+ await check(OLD);
572
+ await check(NEW);
573
+ assert.equal(calls.get(NEW), 1, "attempt 1 done, backoff unelapsed");
574
+ const restore = stubMuseVersion(NEW);
575
+ try {
576
+ const log = writeLog([{ type: "tool_call", name: "read_file" }]);
577
+ assert.equal(museCapabilitySafetyNetAfterSession({
578
+ issueId: randomUUID(),
579
+ runtime: "muse_code",
580
+ logPath: log,
581
+ dirty: true,
582
+ }), "probed");
583
+ assert.equal(calls.get(NEW), 2, "probes now instead of waiting out the backoff");
584
+ await settleMuseCapabilityCheckForTests();
585
+ assert.equal(JSON.parse(fs.readFileSync(STATE, "utf8")).lastChecked.attempts, 2);
586
+ }
587
+ finally {
588
+ restore();
589
+ }
590
+ });
591
+ test("safety net skips clean trees, shell-using sessions, other runtimes, and confirmed versions", async () => {
592
+ const calls = stubProbe({
593
+ [OLD]: { status: "capable" },
594
+ [NEW]: { status: "capable" },
595
+ });
596
+ await check(OLD);
597
+ await check(NEW);
598
+ const restore = stubMuseVersion(NEW);
599
+ try {
600
+ const shellLog = writeLog([{ type: "tool_call", name: "bash" }]);
601
+ const bareLog = writeLog([{ type: "tool_call", name: "read_file" }]);
602
+ assert.equal(museCapabilitySafetyNetAfterSession({
603
+ issueId: randomUUID(),
604
+ runtime: "muse_code",
605
+ logPath: bareLog,
606
+ dirty: false,
607
+ }), "skipped");
608
+ assert.equal(museCapabilitySafetyNetAfterSession({
609
+ issueId: randomUUID(),
610
+ runtime: "muse_code",
611
+ logPath: shellLog,
612
+ dirty: true,
613
+ }), "skipped");
614
+ assert.equal(museCapabilitySafetyNetAfterSession({
615
+ issueId: randomUUID(),
616
+ runtime: "codex_local",
617
+ logPath: bareLog,
618
+ dirty: true,
619
+ }), "skipped");
620
+ assert.equal(museCapabilitySafetyNetAfterSession({
621
+ issueId: randomUUID(),
622
+ runtime: "muse_code",
623
+ logPath: bareLog,
624
+ dirty: true,
625
+ }), "skipped", "the confirmed version needs no verification");
626
+ assert.equal(museCapabilitySafetyNetAfterSession({
627
+ issueId: randomUUID(),
628
+ runtime: "muse_code",
629
+ logPath: "/nonexistent/session.ndjson",
630
+ dirty: true,
631
+ }), "skipped");
632
+ assert.equal(calls.get(NEW), 1, "no safety-net probe fired");
633
+ assert.equal(listHumanActionsByRequestId("muse_capability", museCapabilityRequestId(NEW, "missing")).length, 0);
634
+ assert.equal(listHumanActionsByRequestId("muse_capability", museCapabilityRequestId(NEW, "unverified")).length, 0);
635
+ }
636
+ finally {
637
+ restore();
638
+ }
639
+ });
196
640
  test("a version reported mid-probe is checked once, after the running probe, and the stale verdict is discarded", async () => {
197
641
  const LATER = "1.4.0-R4302.1";
198
642
  const calls = [];