agent-dealer 1.2.4 → 1.2.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bundle/server/dist/adapters/agent-deck-bind.js +33 -3
- package/bundle/server/dist/adapters/agent-deck-bind.test.js +60 -2
- package/bundle/server/dist/adapters/agent-health.js +28 -6
- package/bundle/server/dist/adapters/agent-health.test.js +66 -11
- package/bundle/server/dist/adapters/git-worktree.js +187 -18
- package/bundle/server/dist/adapters/git-worktree.test.js +293 -0
- package/bundle/server/dist/adapters/github.js +110 -12
- package/bundle/server/dist/adapters/github.test.js +274 -3
- package/bundle/server/dist/adapters/muse-capability.js +356 -0
- package/bundle/server/dist/adapters/muse-capability.test.js +464 -0
- package/bundle/server/dist/capacity/claude-local-cache.js +266 -81
- package/bundle/server/dist/capacity/claude-local-cache.test.js +335 -36
- package/bundle/server/dist/capacity/muse-host.js +92 -32
- package/bundle/server/dist/capacity/muse-host.test.js +130 -4
- package/bundle/server/dist/coordinator/admission.test.js +85 -0
- package/bundle/server/dist/coordinator/args.js +4 -3
- package/bundle/server/dist/coordinator/checkpoint.js +5 -3
- package/bundle/server/dist/coordinator/checkpoint.test.js +23 -0
- package/bundle/server/dist/coordinator/commands.js +100 -12
- package/bundle/server/dist/coordinator/developer-effect.js +128 -16
- package/bundle/server/dist/coordinator/developer-effect.test.js +291 -12
- package/bundle/server/dist/coordinator/human-resolution.js +33 -1
- package/bundle/server/dist/coordinator/muse-developer.integration.test.js +291 -55
- package/bundle/server/dist/coordinator/muse-spawn.js +72 -164
- package/bundle/server/dist/coordinator/projection.js +1 -0
- package/bundle/server/dist/coordinator/prompts.js +25 -6
- package/bundle/server/dist/coordinator/prompts.test.js +36 -8
- package/bundle/server/dist/coordinator/routing.js +1 -0
- package/bundle/server/dist/coordinator/routing.test.js +20 -0
- package/bundle/server/dist/coordinator/spawn.js +9 -1
- package/bundle/server/dist/coordinator/worktree-cwd-guard.js +48 -0
- package/bundle/server/dist/coordinator/worktree-cwd-guard.test.js +83 -0
- package/bundle/server/dist/coordinator/worktree-owner-liveness.js +6 -1
- package/bundle/server/dist/coordinator/worktree-owner-liveness.test.js +2 -1
- package/bundle/server/dist/repository/human-actions.js +13 -0
- package/bundle/server/dist/routes/human-actions.js +10 -1
- package/bundle/server/dist/routes/human-actions.test.js +117 -1
- package/bundle/server/dist/routes/index.js +12 -5
- package/bundle/server/dist/routes/runtime-capacity.test.js +3 -2
- package/bundle/server/dist/routes/version.js +30 -0
- package/bundle/server/dist/routes/version.test.js +21 -0
- package/bundle/server/dist/runners/muse-config-core.js +63 -16
- package/bundle/server/dist/runners/muse-config.js +1 -1
- package/bundle/server/dist/runners/muse-config.test.js +135 -29
- package/bundle/server/dist/runners/spawn-cli.js +17 -3
- package/bundle/server/dist/runners/spawn-cli.test.js +59 -0
- package/bundle/server/package.json +2 -2
- package/bundle/server/static-ui/assets/{index-CYqRh_-S.css → index-BII-LgB8.css} +1 -1
- package/bundle/server/static-ui/assets/index-CoDzZcsz.js +60 -0
- package/bundle/server/static-ui/index.html +2 -2
- package/bundle/shared/dist/agents.d.ts +15 -15
- package/bundle/shared/dist/agents.js +6 -0
- package/bundle/shared/dist/index.d.ts +7 -7
- package/bundle/shared/package.json +1 -1
- package/dist/doctor.d.ts +7 -3
- package/dist/doctor.js +21 -13
- package/dist/doctor.test.js +9 -8
- package/dist/managed/index.d.ts +1 -1
- package/dist/managed/index.js +1 -1
- package/dist/managed/updater.d.ts +16 -6
- package/dist/managed/updater.js +28 -11
- package/dist/managed-update-restart.test.d.ts +1 -0
- package/dist/managed-update-restart.test.js +227 -0
- package/dist/ports.d.ts +6 -0
- package/dist/ports.js +10 -0
- package/dist/runtime-state.d.ts +10 -0
- package/dist/runtime-state.js +26 -0
- package/dist/start.js +67 -2
- package/dist/status.js +30 -2
- package/dist/update-check.js +3 -6
- package/package.json +1 -1
- package/bundle/server/static-ui/assets/index-UO4lHZw4.js +0 -60
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// packages/server/src/adapters/github.test.ts
|
|
2
2
|
import { test } from "node:test";
|
|
3
3
|
import assert from "node:assert/strict";
|
|
4
|
-
import { parsePrView, summarizeChecks, pollPrChecks, createGithubAdapter, fetchChecksFailureEvidence, sanitizeCiText, sanitizeUrl, extractActionsRunId, buildFailureExcerpt, formatChecksFailureDetails, CHECKS_EVIDENCE_MAX_EXCERPT_CHARS, CHECKS_FAILURE_GENERIC_REASON, PR_VIEW_FIELDS, } from "./github.js";
|
|
4
|
+
import { parsePrView, summarizeChecks, pollPrChecks, createGithubAdapter, fetchChecksFailureEvidence, sanitizeCiText, sanitizeUrl, extractActionsRunId, buildFailureExcerpt, formatChecksFailureDetails, CHECKS_EVIDENCE_MAX_EXCERPT_CHARS, CHECKS_EVIDENCE_MAX_EXCERPT_LINES, CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN, CHECKS_EVIDENCE_GH_LOG_MAX_BUFFER, CHECKS_FAILURE_GENERIC_REASON, PR_VIEW_FIELDS, } from "./github.js";
|
|
5
5
|
test("parsePrView extracts the ground-truth handoff fields, including draft status", () => {
|
|
6
6
|
const view = parsePrView(JSON.stringify({
|
|
7
7
|
number: 7,
|
|
@@ -34,15 +34,17 @@ test("summarizeChecks accepts neutral/skipped as success but fails closed on an
|
|
|
34
34
|
/** Records every `gh` invocation and returns responses off a queue — never calls real `gh`. */
|
|
35
35
|
function queuedExec(responses) {
|
|
36
36
|
const calls = [];
|
|
37
|
+
const execOpts = [];
|
|
37
38
|
const queue = [...responses];
|
|
38
|
-
const exec = async (args) => {
|
|
39
|
+
const exec = async (args, opts) => {
|
|
39
40
|
calls.push(args);
|
|
41
|
+
execOpts.push(opts);
|
|
40
42
|
const next = queue.shift() ?? { stdout: "" };
|
|
41
43
|
if (next.error != null)
|
|
42
44
|
throw Object.assign(new Error(next.error), { stderr: next.error });
|
|
43
45
|
return { stdout: next.stdout ?? "" };
|
|
44
46
|
};
|
|
45
|
-
return { exec, calls };
|
|
47
|
+
return { exec, calls, execOpts };
|
|
46
48
|
}
|
|
47
49
|
test("viewPr looks up the PR explicitly by branch, never a bare `gh pr view`", async () => {
|
|
48
50
|
const { exec, calls } = queuedExec([
|
|
@@ -426,6 +428,275 @@ test("NOT-252: extractActionsRunId and sanitizeUrl helpers", async () => {
|
|
|
426
428
|
assert.equal(sanitizeUrl("https://example.com/a?b=1#c"), "https://example.com/a");
|
|
427
429
|
assert.equal(sanitizeUrl("https://example.com/a"), "https://example.com/a");
|
|
428
430
|
});
|
|
431
|
+
// --- NOT-276: search-before-truncate — an early failure must survive excerpt building ---
|
|
432
|
+
// Replays the NOT-273 incident shape (Actions run 36330128633 "Unit tests" step):
|
|
433
|
+
// 1667 TAP lines, `not ok 355/356` + `cancelledByParent` at ~21% through the log,
|
|
434
|
+
// followed by ~1300 passing-test lines. Under the old last-20K-chars pre-truncation
|
|
435
|
+
// the failure sat outside the kept tail and the excerpt showed only passing
|
|
436
|
+
// tail noise; with search-before-truncate it must surface the failure instead.
|
|
437
|
+
// Faithful-scale substitute for the real `gh run view 36330128633 --log-failed`
|
|
438
|
+
// replay (no network/gh in this sandbox — the PR description must still record a
|
|
439
|
+
// replay against the saved real "Unit tests" log showing `not ok 355`/`356`).
|
|
440
|
+
// Crucially, ~21 passing lines BEFORE the failure carry realistic
|
|
441
|
+
// failure-pattern words in their test names (real TAP `ok` lines do this — e.g.
|
|
442
|
+
// "handles error ..."), spaced >13 lines apart so each is an isolated hit region:
|
|
443
|
+
// without hit-density ranking, those early weak hits fill the 80-line budget
|
|
444
|
+
// ahead of the real failure cluster (round-2 blocking finding). The dense
|
|
445
|
+
// failure block (failureType x2 + error: x1 within 9 lines) must outrank them.
|
|
446
|
+
test("NOT-276: failure line before the last 20K chars still reaches the excerpt", async () => {
|
|
447
|
+
const early = [];
|
|
448
|
+
for (let n = 1; n <= 354; n++) {
|
|
449
|
+
// Isolated pattern hits every 16 lines (> 2x the 6-line context radius, so
|
|
450
|
+
// windows never merge): ~22 weak hits precede the real failure.
|
|
451
|
+
if (n % 16 === 0)
|
|
452
|
+
early.push(`ok ${n} - handles error output for test ${n}`);
|
|
453
|
+
else if (n === 353)
|
|
454
|
+
early.push("ok 353 - reports error when child fails to spawn");
|
|
455
|
+
else if (n === 354)
|
|
456
|
+
early.push("ok 354 - cleans up after failure");
|
|
457
|
+
else
|
|
458
|
+
early.push(`ok ${n} - passing test number ${n}`);
|
|
459
|
+
}
|
|
460
|
+
const failureBlock = [
|
|
461
|
+
"not ok 355 - coordinator spawns child with explicit cwd",
|
|
462
|
+
" ---",
|
|
463
|
+
" failureType: 'cancelledByParent'",
|
|
464
|
+
" error: test cancelled by parent",
|
|
465
|
+
" ---",
|
|
466
|
+
"not ok 356 - coordinator spawns child with explicit cwd (2)",
|
|
467
|
+
" ---",
|
|
468
|
+
" failureType: 'cancelledByParent'",
|
|
469
|
+
" ---",
|
|
470
|
+
].join("\n");
|
|
471
|
+
const filler = Array.from({ length: 1667 - 356 }, (_, i) => {
|
|
472
|
+
const n = 357 + i;
|
|
473
|
+
if (i % 200 === 0)
|
|
474
|
+
return `ok ${n} - handles error output for test ${n}`;
|
|
475
|
+
if (i % 150 === 0)
|
|
476
|
+
return `ok ${n} - cleans up after failure ${n}`;
|
|
477
|
+
return `ok ${n} - passing test number ${n}`;
|
|
478
|
+
}).join("\n");
|
|
479
|
+
const log = `${early.join("\n")}\n${failureBlock}\n${filler}`;
|
|
480
|
+
// Guard the test's premise: the failure really does sit outside the old 20K tail window.
|
|
481
|
+
assert.ok(log.indexOf("not ok 355") < log.length - 20_000);
|
|
482
|
+
const { exec } = queuedExec([
|
|
483
|
+
{ stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "36330128633")]) },
|
|
484
|
+
{ stdout: log },
|
|
485
|
+
]);
|
|
486
|
+
const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 160, expectedHeadSha: HEAD_SHA });
|
|
487
|
+
assert.ok(evidence);
|
|
488
|
+
assert.match(evidence.excerpt, /not ok 355/);
|
|
489
|
+
assert.match(evidence.excerpt, /not ok 356/);
|
|
490
|
+
assert.match(evidence.excerpt, /cancelledByParent/);
|
|
491
|
+
assert.doesNotMatch(evidence.excerpt, /passing test number 1667/);
|
|
492
|
+
assert.ok(evidence.excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
|
|
493
|
+
assert.ok(evidence.excerpt.split("\n").length <= CHECKS_EVIDENCE_MAX_EXCERPT_LINES);
|
|
494
|
+
});
|
|
495
|
+
// NOT-276 round-2 replay stand-in (blocking finding: the real `gh run view
|
|
496
|
+
// 36330128633 --log-failed` replay needs network/gh, unavailable in CI/sandbox,
|
|
497
|
+
// so this mirrors its exact byte shape at faithful scale instead): every line
|
|
498
|
+
// carries the real `<job>\t<step>\t<timestamp> ` prefix, 1667 TAP results with
|
|
499
|
+
// `not ok 355`/`356` + `cancelledByParent` at ~21% through, ~24 passing lines
|
|
500
|
+
// before the failure carrying failure-pattern words in their names, >20K
|
|
501
|
+
// chars of passing-test tail after it, and the node:test TAP summary block at
|
|
502
|
+
// the true end of the log (the root-cause mechanism: node:test runs to
|
|
503
|
+
// completion, so `# fail 2` — a weak tail hit — sits after thousands of passing
|
|
504
|
+
// lines). The excerpt must surface the early strong failure, not the tail.
|
|
505
|
+
test("NOT-276 round-2 replay: full-scale Actions-prefixed log with an early `not ok` failure", async () => {
|
|
506
|
+
const bodies = [];
|
|
507
|
+
for (let n = 1; n <= 354; n++) {
|
|
508
|
+
if (n % 16 === 0)
|
|
509
|
+
bodies.push(`ok ${n} - handles error output for test ${n}`);
|
|
510
|
+
else if (n === 353)
|
|
511
|
+
bodies.push("ok 353 - reports error when child fails to spawn");
|
|
512
|
+
else if (n === 354)
|
|
513
|
+
bodies.push("ok 354 - cleans up after failure");
|
|
514
|
+
else
|
|
515
|
+
bodies.push(`ok ${n} - passing test number ${n}`);
|
|
516
|
+
}
|
|
517
|
+
bodies.push("not ok 355 - coordinator spawns child with explicit cwd", " ---", " failureType: 'cancelledByParent'", " error: test cancelled by parent", " ---", "not ok 356 - coordinator spawns child with explicit cwd (2)", " ---", " failureType: 'cancelledByParent'", " ---");
|
|
518
|
+
for (let n = 357; n <= 1667; n++) {
|
|
519
|
+
const i = n - 357;
|
|
520
|
+
if (i % 200 === 0)
|
|
521
|
+
bodies.push(`ok ${n} - handles error output for test ${n}`);
|
|
522
|
+
else if (i % 150 === 0)
|
|
523
|
+
bodies.push(`ok ${n} - cleans up after failure ${n}`);
|
|
524
|
+
else
|
|
525
|
+
bodies.push(`ok ${n} - passing test number ${n}`);
|
|
526
|
+
}
|
|
527
|
+
// node:test's own end-of-run TAP summary, as in the real incident log — its
|
|
528
|
+
// `# fail 2` line is a weak failure-pattern hit in the tail that must not
|
|
529
|
+
// outrank the early strong `not ok` failure (round-2 blocking finding).
|
|
530
|
+
bodies.push("# tests 1667", "# suites 4", "# pass 1663", "# fail 2", "# cancelled 2", "# skipped 0", "# todo 0", "# duration_ms 184213.015");
|
|
531
|
+
const log = bodies
|
|
532
|
+
.map((b, i) => `verify\tUnit tests\t2026-09-27T16:${String(Math.floor(i / 60)).padStart(2, "0")}:${String(i % 60).padStart(2, "0")}Z ${b}`)
|
|
533
|
+
.join("\n");
|
|
534
|
+
// Guard the test's premise: the failure really does sit outside the old 20K tail window.
|
|
535
|
+
assert.ok(log.indexOf("not ok 355") < log.length - 20_000);
|
|
536
|
+
const { exec } = queuedExec([
|
|
537
|
+
{ stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "36330128633")]) },
|
|
538
|
+
{ stdout: log },
|
|
539
|
+
]);
|
|
540
|
+
const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 160, expectedHeadSha: HEAD_SHA });
|
|
541
|
+
assert.ok(evidence);
|
|
542
|
+
assert.match(evidence.excerpt, /not ok 355/);
|
|
543
|
+
assert.match(evidence.excerpt, /not ok 356/);
|
|
544
|
+
assert.match(evidence.excerpt, /cancelledByParent/);
|
|
545
|
+
assert.doesNotMatch(evidence.excerpt, /passing test number 1667/);
|
|
546
|
+
assert.ok(evidence.excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
|
|
547
|
+
assert.ok(evidence.excerpt.split("\n").length <= CHECKS_EVIDENCE_MAX_EXCERPT_LINES);
|
|
548
|
+
});
|
|
549
|
+
test("NOT-276: dense failure cluster outranks sparse earlier weak hits (unit level)", async () => {
|
|
550
|
+
// Minimal direct proof of the round-2 ranking: 30 isolated weak hits precede one
|
|
551
|
+
// dense 3-hit failure cluster; the excerpt must contain the cluster.
|
|
552
|
+
const lines = [];
|
|
553
|
+
for (let g = 0; g < 30; g++) {
|
|
554
|
+
for (let k = 0; k < 15; k++)
|
|
555
|
+
lines.push(`ok ${g * 16 + k + 1} - passing test number ${g * 16 + k + 1}`);
|
|
556
|
+
lines.push(`ok ${g * 16 + 16} - handles error output for test ${g * 16 + 16}`);
|
|
557
|
+
}
|
|
558
|
+
lines.push("not ok 999 - real failure here", " failureType: 'cancelledByParent'", " error: test cancelled by parent");
|
|
559
|
+
for (let n = 1000; n < 1100; n++)
|
|
560
|
+
lines.push(`ok ${n} - passing test number ${n}`);
|
|
561
|
+
const { excerpt } = buildFailureExcerpt(lines.join("\n"));
|
|
562
|
+
assert.match(excerpt, /not ok 999/);
|
|
563
|
+
assert.match(excerpt, /cancelledByParent/);
|
|
564
|
+
});
|
|
565
|
+
// NOT-276 round 3 (real PR #163 review, verified against the actual saved
|
|
566
|
+
// `gh run view 36330128633 --log-failed` output): plain hit-density ranking still
|
|
567
|
+
// missed the real failure two different ways. (1) `gh run view --log-failed`
|
|
568
|
+
// prefixes every line with `<job>\t<step>\t<timestamp> ` before the tool's own
|
|
569
|
+
// output, so an anchored `^\s*not ok\b` pattern never matches the real thing — only
|
|
570
|
+
// an unanchored one does. (2) This repo's own coordinator tests are *about*
|
|
571
|
+
// escalation/conflict/timeout handling, so a cluster of unrelated passing tests can
|
|
572
|
+
// out-rank the real (sparse) failure under density alone; an explicit marker
|
|
573
|
+
// (`not ok`, `##[error]`) must dominate ranking regardless of density.
|
|
574
|
+
test("NOT-276 round 3: a GitHub Actions log-line prefix must not hide the `not ok` marker", async () => {
|
|
575
|
+
const prefix = (ts) => `verify\tUnit tests\t${ts}Z `;
|
|
576
|
+
const early = Array.from({ length: 60 }, (_, i) => `${prefix(`2026-09-27T16:07:${String(i).padStart(2, "0")}.0000000`)}ok ${i + 1} - passing test number ${i + 1}`).join("\n");
|
|
577
|
+
const failureBlock = [
|
|
578
|
+
`${prefix("2026-09-27T16:07:31.3990644")}not ok 355 - NOT-273: the serve child spawns with an explicit cwd`,
|
|
579
|
+
`${prefix("2026-09-27T16:07:31.3991406")} failureType: 'cancelledByParent'`,
|
|
580
|
+
`${prefix("2026-09-27T16:07:31.3992743")}not ok 356 - no credential short-circuits to missing without spawning`,
|
|
581
|
+
`${prefix("2026-09-27T16:07:31.3993384")} failureType: 'cancelledByParent'`,
|
|
582
|
+
].join("\n");
|
|
583
|
+
const late = Array.from({ length: 60 }, (_, i) => `${prefix(`2026-09-27T16:08:${String(i).padStart(2, "0")}.0000000`)}ok ${357 + i} - passing test number ${357 + i}`).join("\n");
|
|
584
|
+
const log = `${early}\n${failureBlock}\n${late}\n${prefix("2026-09-27T16:09:02.6611414")}##[error]Process completed with exit code 1.`;
|
|
585
|
+
const { excerpt } = buildFailureExcerpt(log);
|
|
586
|
+
assert.match(excerpt, /not ok 355/);
|
|
587
|
+
assert.match(excerpt, /not ok 356/);
|
|
588
|
+
assert.match(excerpt, /cancelledByParent/);
|
|
589
|
+
});
|
|
590
|
+
// NOT-276 round 3: a lower-ranked region taken first can still be chronologically
|
|
591
|
+
// *earlier* than a higher-ranked one; if the excerpt were char-capped by slicing
|
|
592
|
+
// the reassembled string's tail (ignoring rank), the earlier low-priority region
|
|
593
|
+
// could crowd the later high-priority region's content out of the character
|
|
594
|
+
// budget even though line-budget ranking correctly preferred the real failure.
|
|
595
|
+
test("NOT-276 round 3: an earlier low-priority region must not crowd the real failure out of the character budget", async () => {
|
|
596
|
+
// ~12 weak hits (one per line, "failed"/"error"), long lines: ~12 * 460 ≈ 5500 chars.
|
|
597
|
+
const earlyWeakCluster = Array.from({ length: 12 }, (_, i) => `ok ${i + 1} - reports a handled failure/error path for scenario ${i + 1} ` + "x".repeat(400)).join("\n");
|
|
598
|
+
// The real failure: 2 strong `not ok` hits, short lines (~450 chars total).
|
|
599
|
+
const realFailure = [
|
|
600
|
+
"not ok 999 - the real failure",
|
|
601
|
+
" failureType: 'cancelledByParent'",
|
|
602
|
+
"not ok 1000 - a second real failure",
|
|
603
|
+
" failureType: 'cancelledByParent'",
|
|
604
|
+
].join("\n");
|
|
605
|
+
// A gap of ordinary passing lines keeps the two clusters as separate merged
|
|
606
|
+
// windows (matching the real incident: the weak cluster and the real failure
|
|
607
|
+
// sat ~2300 lines apart) rather than merging into one oversized window, which
|
|
608
|
+
// would exercise a different, narrower edge case than the one under test here.
|
|
609
|
+
const gap = Array.from({ length: 20 }, (_, i) => `ok ${900 + i} - passing test number ${900 + i}`).join("\n");
|
|
610
|
+
const log = `${earlyWeakCluster}\n${gap}\n${realFailure}`;
|
|
611
|
+
assert.ok(earlyWeakCluster.length + realFailure.length > CHECKS_EVIDENCE_MAX_EXCERPT_CHARS, "test premise: both regions together exceed the char budget");
|
|
612
|
+
const { excerpt, truncated } = buildFailureExcerpt(log);
|
|
613
|
+
assert.match(excerpt, /not ok 999/);
|
|
614
|
+
assert.match(excerpt, /not ok 1000/);
|
|
615
|
+
assert.match(excerpt, /cancelledByParent/);
|
|
616
|
+
assert.ok(excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
|
|
617
|
+
assert.equal(truncated, true, "dropping the lower-ranked region must be reported as truncation");
|
|
618
|
+
});
|
|
619
|
+
test("NOT-276: an exact line-budget fill still reports a later failure region as truncated", async () => {
|
|
620
|
+
// The A and B windows contain 39 and 40 lines respectively. With the separator
|
|
621
|
+
// between them, taking both fills the 80-line budget exactly. Region C is a real
|
|
622
|
+
// failure that must be omitted, and that omission must set truncated=true.
|
|
623
|
+
const ordinary = (label, count) => Array.from({ length: count }, (_, i) => `ok ${label}-${i + 1} - passing test`);
|
|
624
|
+
const regionA = Array.from({ length: 27 }, (_, i) => `not ok A-${i + 1} - strong failure A`);
|
|
625
|
+
const regionB = Array.from({ length: 28 }, (_, i) => `not ok B-${i + 1} - strong failure B`);
|
|
626
|
+
const lines = [
|
|
627
|
+
...ordinary("prefix", 6),
|
|
628
|
+
...regionA,
|
|
629
|
+
...ordinary("gap-a-b", 13),
|
|
630
|
+
...regionB,
|
|
631
|
+
...ordinary("gap-b-c", 13),
|
|
632
|
+
"not ok C-1 - later real failure",
|
|
633
|
+
...ordinary("suffix", 6),
|
|
634
|
+
];
|
|
635
|
+
const { excerpt, truncated } = buildFailureExcerpt(lines.join("\n"));
|
|
636
|
+
assert.equal(excerpt.split("\n").length, CHECKS_EVIDENCE_MAX_EXCERPT_LINES);
|
|
637
|
+
assert.match(excerpt, /not ok A-1/);
|
|
638
|
+
assert.match(excerpt, /not ok B-1/);
|
|
639
|
+
assert.doesNotMatch(excerpt, /not ok C-1/);
|
|
640
|
+
assert.equal(truncated, true, "the omitted third failure region must be reported as truncation");
|
|
641
|
+
});
|
|
642
|
+
test("NOT-276: log with no failure-pattern hits still falls back to the tail, unchanged", async () => {
|
|
643
|
+
// NB: this filler must stay free of FAILURE_LINE_PATTERN words — that absence is the no-hit premise.
|
|
644
|
+
const log = Array.from({ length: 200 }, (_, i) => `ok ${i + 1} - passing test number ${i + 1}`).join("\n");
|
|
645
|
+
const { exec } = queuedExec([
|
|
646
|
+
{ stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "111")]) },
|
|
647
|
+
{ stdout: log },
|
|
648
|
+
]);
|
|
649
|
+
const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 42, expectedHeadSha: HEAD_SHA });
|
|
650
|
+
assert.ok(evidence);
|
|
651
|
+
// The fetch path must feed the full log through exactly as buildFailureExcerpt sees it.
|
|
652
|
+
assert.equal(evidence.excerpt, buildFailureExcerpt(log).excerpt);
|
|
653
|
+
const excerptLines = evidence.excerpt.split("\n");
|
|
654
|
+
assert.equal(excerptLines.length, CHECKS_EVIDENCE_MAX_EXCERPT_LINES);
|
|
655
|
+
assert.match(excerptLines[0] ?? "", /passing test number 121/);
|
|
656
|
+
assert.match(excerptLines[excerptLines.length - 1] ?? "", /passing test number 200/);
|
|
657
|
+
assert.equal(evidence.excerptTruncated, true);
|
|
658
|
+
assert.ok(evidence.excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
|
|
659
|
+
});
|
|
660
|
+
test("NOT-276: log larger than the raw memory ceiling stays bounded without crashing", async () => {
|
|
661
|
+
// Short lines on purpose: the 80-line tail must fit under the 4,000-char
|
|
662
|
+
// excerpt cap, otherwise the char cap (not the memory ceiling) would cut the
|
|
663
|
+
// asserted last line and the test would prove nothing about the tail.
|
|
664
|
+
const line = (i) => `ok ${i} - pad pad ${i}`;
|
|
665
|
+
const targetLen = CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN + 500_000;
|
|
666
|
+
const count = Math.ceil(targetLen / 20);
|
|
667
|
+
const parts = new Array(count);
|
|
668
|
+
for (let i = 0; i < count; i++)
|
|
669
|
+
parts[i] = line(i);
|
|
670
|
+
const log = parts.join("\n");
|
|
671
|
+
assert.ok(log.length > CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN);
|
|
672
|
+
const { exec } = queuedExec([
|
|
673
|
+
{ stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "111")]) },
|
|
674
|
+
{ stdout: log },
|
|
675
|
+
]);
|
|
676
|
+
const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 42, expectedHeadSha: HEAD_SHA });
|
|
677
|
+
assert.ok(evidence);
|
|
678
|
+
assert.ok(evidence.excerpt.length <= CHECKS_EVIDENCE_MAX_EXCERPT_CHARS);
|
|
679
|
+
// The excerpt is built from the kept tail portion, never the discarded head.
|
|
680
|
+
assert.equal(evidence.excerpt, buildFailureExcerpt(log.slice(-CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN)).excerpt);
|
|
681
|
+
assert.match(evidence.excerpt, new RegExp(`pad pad ${count - 1}`));
|
|
682
|
+
});
|
|
683
|
+
test("NOT-276 round 3: run-log fetch carries an explicit maxBuffer above the raw ceiling", async () => {
|
|
684
|
+
// Node's execFile defaults to ~1 MiB maxBuffer, which would reject any larger
|
|
685
|
+
// --log-failed output with ERR_CHILD_PROCESS_STDIO_MAXBUFFER before the 2MB
|
|
686
|
+
// raw-log ceiling applies. The fetch must pass an explicit buffer above the
|
|
687
|
+
// ceiling so the ceiling — not the process buffer — bounds memory.
|
|
688
|
+
assert.ok(CHECKS_EVIDENCE_GH_LOG_MAX_BUFFER > CHECKS_EVIDENCE_MAX_RAW_LOG_CHARS_PER_RUN, "maxBuffer must sit above the raw-log ceiling");
|
|
689
|
+
const { exec, calls, execOpts } = queuedExec([
|
|
690
|
+
{ stdout: prViewWithRollup([actionsCheck("verify", "FAILURE", "111")]) },
|
|
691
|
+
{ stdout: "verify log\nError: boom\n" },
|
|
692
|
+
]);
|
|
693
|
+
const evidence = await fetchChecksFailureEvidence(exec, { cwd: "/repo", number: 42, expectedHeadSha: HEAD_SHA });
|
|
694
|
+
assert.ok(evidence);
|
|
695
|
+
assert.deepEqual(calls[1], ["run", "view", "111", "--log-failed"]);
|
|
696
|
+
assert.equal(execOpts[1]?.maxBuffer, CHECKS_EVIDENCE_GH_LOG_MAX_BUFFER);
|
|
697
|
+
// The small pr-view lookup needs no oversized buffer.
|
|
698
|
+
assert.ok(execOpts[0]?.maxBuffer == null, "pr view must not carry the log buffer");
|
|
699
|
+
});
|
|
429
700
|
test("NOT-252: formatChecksFailureDetails labels the excerpt as untrusted, not instructions", async () => {
|
|
430
701
|
const details = formatChecksFailureDetails({
|
|
431
702
|
headSha: HEAD_SHA,
|
|
@@ -0,0 +1,356 @@
|
|
|
1
|
+
// packages/server/src/adapters/muse-capability.ts
|
|
2
|
+
//
|
|
3
|
+
// NOT-277: Muse Code auto-updates itself (expected, never fought). An update once silently stopped
|
|
4
|
+
// granting developer sessions the shell/write tool, and every developer round after it surfaced
|
|
5
|
+
// only as a generic `dirty_worktree` escalation. So whenever `muse --version` reports a version not
|
|
6
|
+
// yet checked, one real developer-posture session is run whose only path to success is a shell
|
|
7
|
+
// call; the result is cached per version (and persisted, so a restart does not re-bill it):
|
|
8
|
+
//
|
|
9
|
+
// - capable → recorded as the confirmed baseline, admission proceeds with no manual step;
|
|
10
|
+
// - missing → developer admission blocked, message names old → new version + capability;
|
|
11
|
+
// - error → the check could not complete: blocked with a distinct "could not verify"
|
|
12
|
+
// message (fail closed), retried after a backoff;
|
|
13
|
+
// - in flight → blocked while the one-time check runs (never assumed capable).
|
|
14
|
+
//
|
|
15
|
+
// The probe is a fresh `muse exec` of the on-disk binary (never the long-lived serve host, which can
|
|
16
|
+
// still be the pre-update build). Checks are serialized: a version reported mid-probe is checked
|
|
17
|
+
// once the running probe settles, and a verdict for a version no longer reported is discarded.
|
|
18
|
+
// Messages name the exact update — the previously reported version → the current one.
|
|
19
|
+
//
|
|
20
|
+
// Deliberately NOT covered: NOT-177's security-enforcement evidence (`mcp_tool_allowlist_enforcement`,
|
|
21
|
+
// `cron_tool_disable`, pinned in the runners' Muse config core) stays manually re-validated as before.
|
|
22
|
+
import { execFileSync } from "node:child_process";
|
|
23
|
+
import { createHash, randomUUID } from "node:crypto";
|
|
24
|
+
import fs from "node:fs";
|
|
25
|
+
import path from "node:path";
|
|
26
|
+
import { DEVELOPER_ROLE_CEILING } from "@agent-dealer/shared";
|
|
27
|
+
import { MUSE_CLI_ENV, resolveMuseBin } from "../cli-env.js";
|
|
28
|
+
import { getDataDir } from "../db/index.js";
|
|
29
|
+
/** A probe that could not complete is retried after this long (the block stays up meanwhile). */
|
|
30
|
+
const ERROR_RETRY_MS = 10 * 60_000;
|
|
31
|
+
const PROBE_TIMEOUT_MS = 5 * 60_000;
|
|
32
|
+
const PROBE_MAX_MODEL_STEPS = 20;
|
|
33
|
+
const STATE_FILE = "muse-capability.json";
|
|
34
|
+
const CAPABILITY = "shell/write access";
|
|
35
|
+
let state = null;
|
|
36
|
+
/** At most one probe at a time (serialized); a newer version is chained after it settles. */
|
|
37
|
+
let inFlight = null;
|
|
38
|
+
/** Bumped whenever a check settles, so callers can tell a result landed mid-read. */
|
|
39
|
+
let settledCount = 0;
|
|
40
|
+
/** Bumped by resets so a probe started before a reset cannot write into the fresh state. */
|
|
41
|
+
let generation = 0;
|
|
42
|
+
let probeImpl = defaultMuseCapabilityProbe;
|
|
43
|
+
/** Latest caller hook; a chained check reports through it too. */
|
|
44
|
+
let settledHook = () => { };
|
|
45
|
+
function statePath() {
|
|
46
|
+
return path.join(getDataDir(), STATE_FILE);
|
|
47
|
+
}
|
|
48
|
+
function loadState() {
|
|
49
|
+
if (state)
|
|
50
|
+
return state;
|
|
51
|
+
try {
|
|
52
|
+
const parsed = JSON.parse(fs.readFileSync(statePath(), "utf8"));
|
|
53
|
+
const lastChecked = parsed.lastChecked && typeof parsed.lastChecked.version === "string" ? parsed.lastChecked : null;
|
|
54
|
+
const current = parsed.current && typeof parsed.current.version === "string"
|
|
55
|
+
? { version: parsed.current.version, from: typeof parsed.current.from === "string" ? parsed.current.from : null }
|
|
56
|
+
: null;
|
|
57
|
+
state = {
|
|
58
|
+
confirmedVersion: typeof parsed.confirmedVersion === "string" ? parsed.confirmedVersion : null,
|
|
59
|
+
current,
|
|
60
|
+
lastChecked: lastChecked && lastChecked.version === current?.version ? lastChecked : null,
|
|
61
|
+
};
|
|
62
|
+
}
|
|
63
|
+
catch {
|
|
64
|
+
state = { confirmedVersion: null, current: null, lastChecked: null };
|
|
65
|
+
}
|
|
66
|
+
return state;
|
|
67
|
+
}
|
|
68
|
+
function saveState(next) {
|
|
69
|
+
state = next;
|
|
70
|
+
try {
|
|
71
|
+
fs.writeFileSync(statePath(), `${JSON.stringify(next, null, 2)}\n`);
|
|
72
|
+
}
|
|
73
|
+
catch (err) {
|
|
74
|
+
// In-memory state still gates this process; only restart persistence is lost.
|
|
75
|
+
console.warn(`[muse-capability] could not persist ${statePath()}: ${String(err)}`);
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
/** Record `version` as the one Muse now reports; a change remembers what it changed from. */
|
|
79
|
+
function observe(version) {
|
|
80
|
+
const s = loadState();
|
|
81
|
+
if (s.current?.version === version)
|
|
82
|
+
return s;
|
|
83
|
+
const from = s.current?.version ?? s.confirmedVersion;
|
|
84
|
+
saveState({ ...s, current: { version, from: from !== version ? from : null }, lastChecked: null });
|
|
85
|
+
return state;
|
|
86
|
+
}
|
|
87
|
+
/** `Muse Code 1.3.0 (1.3.0-R3401.1)` → `1.3.0-R3401.1`; otherwise the last non-empty line. */
|
|
88
|
+
export function parseMuseVersion(output) {
|
|
89
|
+
const paren = /\(([^()\s]+)\)\s*$/m.exec(output.trim());
|
|
90
|
+
if (paren)
|
|
91
|
+
return paren[1];
|
|
92
|
+
const last = output.trim().split("\n").map((l) => l.trim()).filter(Boolean).pop();
|
|
93
|
+
return last ?? null;
|
|
94
|
+
}
|
|
95
|
+
function transition(from, to) {
|
|
96
|
+
return from && from !== to ? `Muse Code updated ${from} → ${to}` : `Muse Code ${to}`;
|
|
97
|
+
}
|
|
98
|
+
function issueFor(checked, from) {
|
|
99
|
+
if (checked.status === "capable")
|
|
100
|
+
return [];
|
|
101
|
+
const head = transition(from, checked.version);
|
|
102
|
+
if (checked.status === "missing") {
|
|
103
|
+
return [
|
|
104
|
+
{
|
|
105
|
+
code: "runtime_capability",
|
|
106
|
+
message: `${head}: developer sessions no longer get ${CAPABILITY} (${checked.detail}) — ` +
|
|
107
|
+
`developer admission blocked until a Muse Code version passes the check`,
|
|
108
|
+
},
|
|
109
|
+
];
|
|
110
|
+
}
|
|
111
|
+
return [
|
|
112
|
+
{
|
|
113
|
+
code: "runtime_capability",
|
|
114
|
+
message: `Could not verify Muse Code developer ${CAPABILITY} after version change ` +
|
|
115
|
+
`(${from && from !== checked.version ? `${from} → ` : ""}${checked.version}): ` +
|
|
116
|
+
`${checked.detail} — developer admission blocked; the check retries automatically`,
|
|
117
|
+
},
|
|
118
|
+
];
|
|
119
|
+
}
|
|
120
|
+
function verifyingIssue(version, from) {
|
|
121
|
+
return [
|
|
122
|
+
{
|
|
123
|
+
code: "runtime_capability",
|
|
124
|
+
message: `${transition(from, version)}: verifying developer ${CAPABILITY} ` +
|
|
125
|
+
`(one-time check for this version) — developer admission waits for the result`,
|
|
126
|
+
},
|
|
127
|
+
];
|
|
128
|
+
}
|
|
129
|
+
/** Current version is unchecked, or its could-not-verify result is due a retry. */
|
|
130
|
+
function checkDue(s) {
|
|
131
|
+
if (!s.current)
|
|
132
|
+
return false;
|
|
133
|
+
const checked = s.lastChecked;
|
|
134
|
+
if (!checked)
|
|
135
|
+
return true;
|
|
136
|
+
return checked.status === "error" && Date.now() - checked.checkedAt >= ERROR_RETRY_MS;
|
|
137
|
+
}
|
|
138
|
+
function startCheck(version) {
|
|
139
|
+
const gen = generation;
|
|
140
|
+
const promise = (async () => {
|
|
141
|
+
let result;
|
|
142
|
+
try {
|
|
143
|
+
result = await probeImpl(version);
|
|
144
|
+
}
|
|
145
|
+
catch (err) {
|
|
146
|
+
result = { status: "error", detail: err instanceof Error ? err.message : String(err) };
|
|
147
|
+
}
|
|
148
|
+
if (gen !== generation)
|
|
149
|
+
return;
|
|
150
|
+
const prev = loadState();
|
|
151
|
+
// Muse moved on while this probe ran: its verdict is about a build no longer reported, so it
|
|
152
|
+
// must not overwrite state for the newer one (which is checked next, below).
|
|
153
|
+
if (prev.current?.version !== version)
|
|
154
|
+
return;
|
|
155
|
+
const checked = { version, checkedAt: Date.now(), ...result };
|
|
156
|
+
saveState({
|
|
157
|
+
...prev,
|
|
158
|
+
confirmedVersion: result.status === "capable" ? version : prev.confirmedVersion,
|
|
159
|
+
lastChecked: checked,
|
|
160
|
+
});
|
|
161
|
+
const from = prev.current.from;
|
|
162
|
+
if (result.status === "capable") {
|
|
163
|
+
console.log(`[muse-capability] ${transition(from, version)}: developer ${CAPABILITY} confirmed`);
|
|
164
|
+
}
|
|
165
|
+
else {
|
|
166
|
+
console.warn(`[muse-capability] ${issueFor(checked, from)[0].message}`);
|
|
167
|
+
}
|
|
168
|
+
})().finally(() => {
|
|
169
|
+
if (inFlight?.promise === promise)
|
|
170
|
+
inFlight = null;
|
|
171
|
+
if (gen !== generation)
|
|
172
|
+
return;
|
|
173
|
+
settledCount += 1;
|
|
174
|
+
const s = loadState();
|
|
175
|
+
if (!inFlight && s.current && s.current.version !== version && checkDue(s))
|
|
176
|
+
startCheck(s.current.version);
|
|
177
|
+
settledHook();
|
|
178
|
+
});
|
|
179
|
+
inFlight = { version, promise };
|
|
180
|
+
}
|
|
181
|
+
/**
|
|
182
|
+
* Capability issues for the currently reported Muse version. Never awaits the probe: a version not
|
|
183
|
+
* yet checked starts one background check and blocks until it settles. Checks are serialized — a
|
|
184
|
+
* version observed while another is being probed waits for that probe, then is checked once.
|
|
185
|
+
* `onSettled` lets the caller drop its health cache so admission unblocks as soon as the result is in.
|
|
186
|
+
*/
|
|
187
|
+
export function museCapabilityIssues(version, onSettled = () => { }) {
|
|
188
|
+
settledHook = onSettled;
|
|
189
|
+
const s = observe(version);
|
|
190
|
+
const from = s.current.from;
|
|
191
|
+
if (checkDue(s) && !inFlight)
|
|
192
|
+
startCheck(version);
|
|
193
|
+
if (s.lastChecked)
|
|
194
|
+
return issueFor(s.lastChecked, from); // incl. error retry: keep the block up
|
|
195
|
+
return verifyingIssue(version, from);
|
|
196
|
+
}
|
|
197
|
+
/** True while a capability check is running (callers use a short health-cache TTL meanwhile). */
|
|
198
|
+
export function museCapabilityCheckInFlight() {
|
|
199
|
+
return inFlight !== null;
|
|
200
|
+
}
|
|
201
|
+
/**
|
|
202
|
+
* Changes each time a check settles. A health read that spans a settle must not cache what it
|
|
203
|
+
* read — it may be the "verifying" block the settle just superseded.
|
|
204
|
+
*/
|
|
205
|
+
export function museCapabilitySettleCount() {
|
|
206
|
+
return settledCount;
|
|
207
|
+
}
|
|
208
|
+
/** Tests: resolve once any in-flight capability check has settled. */
|
|
209
|
+
export async function settleMuseCapabilityCheckForTests() {
|
|
210
|
+
while (inFlight)
|
|
211
|
+
await inFlight.promise;
|
|
212
|
+
}
|
|
213
|
+
/** Tests: replace the real `muse` probe. Pass `null` to restore it. */
|
|
214
|
+
export function setMuseCapabilityProbeForTests(fn) {
|
|
215
|
+
probeImpl = fn ?? defaultMuseCapabilityProbe;
|
|
216
|
+
}
|
|
217
|
+
/** Tests: forget in-memory and persisted state; any in-flight probe result is discarded. */
|
|
218
|
+
export function resetMuseCapabilityStateForTests() {
|
|
219
|
+
generation += 1;
|
|
220
|
+
inFlight = null;
|
|
221
|
+
state = null;
|
|
222
|
+
settledHook = () => { };
|
|
223
|
+
fs.rmSync(statePath(), { force: true });
|
|
224
|
+
}
|
|
225
|
+
/** Tests: backdate the last check so the error-retry path can be exercised without waiting. */
|
|
226
|
+
export function ageMuseCapabilityCheckForTests(ms) {
|
|
227
|
+
const s = loadState();
|
|
228
|
+
if (s.lastChecked)
|
|
229
|
+
saveState({ ...s, lastChecked: { ...s.lastChecked, checkedAt: s.lastChecked.checkedAt - ms } });
|
|
230
|
+
}
|
|
231
|
+
/** `git hash-object` of a string — what `probe.sh` writes; not computable without running it. */
|
|
232
|
+
function gitBlobSha1(content) {
|
|
233
|
+
return createHash("sha1").update(`blob ${Buffer.byteLength(content)}\0${content}`).digest("hex");
|
|
234
|
+
}
|
|
235
|
+
/** What the on-disk `muse` reports now (null when it cannot say). */
|
|
236
|
+
function reportedMuseVersion() {
|
|
237
|
+
try {
|
|
238
|
+
const out = execFileSync(resolveMuseBin(), ["--version"], {
|
|
239
|
+
encoding: "utf8",
|
|
240
|
+
timeout: 30_000,
|
|
241
|
+
env: { ...process.env, ...MUSE_CLI_ENV },
|
|
242
|
+
stdio: ["ignore", "pipe", "pipe"],
|
|
243
|
+
});
|
|
244
|
+
return parseMuseVersion(out);
|
|
245
|
+
}
|
|
246
|
+
catch {
|
|
247
|
+
return null;
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
export async function defaultMuseCapabilityProbe(version, opts = {}) {
|
|
251
|
+
const [{ runMuseDeveloperSession }, { getAgentDeckMcpUrl, checkAgentDeckHealth, fetchDecks }] = await Promise.all([
|
|
252
|
+
import("../coordinator/muse-spawn.js"),
|
|
253
|
+
import("./agent-deck.js"),
|
|
254
|
+
]);
|
|
255
|
+
const { verifyWorkerDeckConnection } = await import("./agent-deck-bind.js");
|
|
256
|
+
const listDecks = opts.listDecks ?? fetchDecks;
|
|
257
|
+
const verifyDeck = opts.verifyDeck ??
|
|
258
|
+
((args) => verifyWorkerDeckConnection({ deckId: args.deckId, worktreePath: args.worktreePath, playbookIds: [] }));
|
|
259
|
+
// Decouple the paid capability check from deck reachability: when the deck is down, fail
|
|
260
|
+
// closed here — before any child exists — instead of burning a model session that could
|
|
261
|
+
// only fail on its required deck server. The deck gate (`deck_offline`) already blocks
|
|
262
|
+
// admission meanwhile, and the error-retry path re-runs this cheap check, not a turn.
|
|
263
|
+
if (!(await checkAgentDeckHealth())) {
|
|
264
|
+
return {
|
|
265
|
+
status: "error",
|
|
266
|
+
detail: `Agent Deck is unreachable — capability check for ${version} not run (no model session spent)`,
|
|
267
|
+
};
|
|
268
|
+
}
|
|
269
|
+
const scratchParent = path.join(getDataDir(), ".temporal");
|
|
270
|
+
fs.mkdirSync(scratchParent, { recursive: true });
|
|
271
|
+
const dir = fs.mkdtempSync(path.join(scratchParent, "muse-capability-"));
|
|
272
|
+
try {
|
|
273
|
+
execFileSync("git", ["init", "-q"], { cwd: dir, stdio: "ignore" });
|
|
274
|
+
const nonce = randomUUID();
|
|
275
|
+
fs.writeFileSync(path.join(dir, "probe.sh"), `#!/bin/sh\nprintf '%s' '${nonce}' | git hash-object --stdin > result.txt\n`);
|
|
276
|
+
// The probe session must bind a real deck: a synthetic id could never pass the
|
|
277
|
+
// production `get_bound_deck` identity check, so it would either burn a paid turn
|
|
278
|
+
// that only fails on its required deck server, or run on an unselected deck. List
|
|
279
|
+
// the live decks and preflight the bound one before spawning — a deck that answers
|
|
280
|
+
// but rejects this id fails closed here, with no model session spent.
|
|
281
|
+
const listed = await listDecks();
|
|
282
|
+
if (!listed.ok) {
|
|
283
|
+
return {
|
|
284
|
+
status: "error",
|
|
285
|
+
detail: `Agent Deck deck list unavailable (${listed.message}) — capability check for ${version} not run (no model session spent)`,
|
|
286
|
+
};
|
|
287
|
+
}
|
|
288
|
+
const probeDeckId = listed.decks[0]?.id;
|
|
289
|
+
if (!probeDeckId) {
|
|
290
|
+
return {
|
|
291
|
+
status: "error",
|
|
292
|
+
detail: `Agent Deck listed no decks — capability check for ${version} not run (no model session spent)`,
|
|
293
|
+
};
|
|
294
|
+
}
|
|
295
|
+
const verified = await verifyDeck({ deckId: probeDeckId, worktreePath: dir });
|
|
296
|
+
if (!verified.ok) {
|
|
297
|
+
return {
|
|
298
|
+
status: "error",
|
|
299
|
+
detail: verified.kind === "deck_unavailable"
|
|
300
|
+
? `Agent Deck is unreachable for deck ${probeDeckId} (${verified.reason}) — capability check for ${version} not run (no model session spent)`
|
|
301
|
+
: `Agent Deck rejected probe deck ${probeDeckId} (${verified.reason}) — capability check for ${version} not run (no model session spent)`,
|
|
302
|
+
};
|
|
303
|
+
}
|
|
304
|
+
const run = await runMuseDeveloperSession({
|
|
305
|
+
sessionId: randomUUID(),
|
|
306
|
+
runtime: "muse_code",
|
|
307
|
+
policy: DEVELOPER_ROLE_CEILING,
|
|
308
|
+
model: null,
|
|
309
|
+
deckId: probeDeckId,
|
|
310
|
+
agentDeckUrl: `${getAgentDeckMcpUrl().replace(/\/mcp\/?$/, "")}/mcp`,
|
|
311
|
+
maxModelSteps: PROBE_MAX_MODEL_STEPS,
|
|
312
|
+
prompt: "Dealer capability check. Using your shell tool, run exactly this command in the current " +
|
|
313
|
+
"directory: sh probe.sh\nDo not create or edit result.txt any other way. When the command " +
|
|
314
|
+
"has finished, reply with the single word DONE.",
|
|
315
|
+
cwd: dir,
|
|
316
|
+
timeoutMs: opts.timeoutMs ?? PROBE_TIMEOUT_MS,
|
|
317
|
+
logPath: path.join(dir, "probe.ndjson"),
|
|
318
|
+
});
|
|
319
|
+
const after = reportedMuseVersion();
|
|
320
|
+
if (after !== version) {
|
|
321
|
+
return {
|
|
322
|
+
status: "error",
|
|
323
|
+
detail: `muse reported ${after ?? "no version"} after probing ${version} (changed during the check)`,
|
|
324
|
+
};
|
|
325
|
+
}
|
|
326
|
+
let written = null;
|
|
327
|
+
try {
|
|
328
|
+
written = fs.readFileSync(path.join(dir, "result.txt"), "utf8").trim();
|
|
329
|
+
}
|
|
330
|
+
catch {
|
|
331
|
+
written = null;
|
|
332
|
+
}
|
|
333
|
+
// A session that did not complete cleanly is never a verdict, even if the shell ran first.
|
|
334
|
+
if (run.timedOut)
|
|
335
|
+
return { status: "error", detail: `probe session on ${version} timed out` };
|
|
336
|
+
const failure = run.muse?.failure;
|
|
337
|
+
if (failure)
|
|
338
|
+
return { status: "error", detail: `probe session failed (${failure.kind}: ${failure.message})` };
|
|
339
|
+
if (run.exitCode !== 0)
|
|
340
|
+
return { status: "error", detail: `probe session exited ${run.exitCode}` };
|
|
341
|
+
if (written === gitBlobSha1(nonce))
|
|
342
|
+
return { status: "capable" };
|
|
343
|
+
return {
|
|
344
|
+
status: "missing",
|
|
345
|
+
detail: written === null
|
|
346
|
+
? "probe session completed without running its shell command"
|
|
347
|
+
: "probe session completed but the shell command's output was not produced",
|
|
348
|
+
};
|
|
349
|
+
}
|
|
350
|
+
catch (err) {
|
|
351
|
+
return { status: "error", detail: `probe could not run: ${err instanceof Error ? err.message : String(err)}` };
|
|
352
|
+
}
|
|
353
|
+
finally {
|
|
354
|
+
fs.rmSync(dir, { recursive: true, force: true });
|
|
355
|
+
}
|
|
356
|
+
}
|