agent-dealer 1.2.7 → 1.2.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. package/bundle/server/dist/adapters/agent-deck-bind.js +16 -6
  2. package/bundle/server/dist/adapters/agent-deck-bind.test.js +74 -0
  3. package/bundle/server/dist/adapters/agent-health.js +14 -3
  4. package/bundle/server/dist/adapters/github.js +4 -1
  5. package/bundle/server/dist/adapters/muse-capability.js +443 -24
  6. package/bundle/server/dist/adapters/muse-capability.test.js +469 -25
  7. package/bundle/server/dist/adapters/muse-visual-qa.js +114 -0
  8. package/bundle/server/dist/adapters/muse-visual-qa.test.js +68 -0
  9. package/bundle/server/dist/capacity/muse-probe.js +56 -9
  10. package/bundle/server/dist/capacity/muse-probe.test.js +188 -1
  11. package/bundle/server/dist/coordinator/admission.js +14 -1
  12. package/bundle/server/dist/coordinator/admission.test.js +197 -4
  13. package/bundle/server/dist/coordinator/auto-merge.integration.test.js +276 -0
  14. package/bundle/server/dist/coordinator/auto-merge.js +39 -3
  15. package/bundle/server/dist/coordinator/commands.js +90 -5
  16. package/bundle/server/dist/coordinator/deck-outage.integration.test.js +4 -2
  17. package/bundle/server/dist/coordinator/developer-effect.js +79 -1
  18. package/bundle/server/dist/coordinator/execution-report.js +6 -0
  19. package/bundle/server/dist/coordinator/execution-report.test.js +7 -0
  20. package/bundle/server/dist/coordinator/failure-cause.js +64 -1
  21. package/bundle/server/dist/coordinator/failure-cause.test.js +92 -1
  22. package/bundle/server/dist/coordinator/failure-reason.js +7 -0
  23. package/bundle/server/dist/coordinator/failure-reason.test.js +14 -0
  24. package/bundle/server/dist/coordinator/human-resolution.js +17 -1
  25. package/bundle/server/dist/coordinator/merge-conflict-sync.js +647 -0
  26. package/bundle/server/dist/coordinator/merge-conflict-sync.test.js +121 -0
  27. package/bundle/server/dist/coordinator/muse-developer.integration.test.js +9 -4
  28. package/bundle/server/dist/coordinator/muse-spawn.js +122 -29
  29. package/bundle/server/dist/coordinator/muse-spawn.test.js +210 -0
  30. package/bundle/server/dist/coordinator/playbook-feedback.js +690 -0
  31. package/bundle/server/dist/coordinator/playbook-feedback.test.js +702 -0
  32. package/bundle/server/dist/coordinator/prompts-execution-contract.test.js +107 -0
  33. package/bundle/server/dist/coordinator/prompts.js +78 -2
  34. package/bundle/server/dist/coordinator/prompts.test.js +83 -0
  35. package/bundle/server/dist/coordinator/reflect-trigger.js +33 -169
  36. package/bundle/server/dist/coordinator/reflect-trigger.test.js +149 -200
  37. package/bundle/server/dist/coordinator/reviewer-effect.js +10 -1
  38. package/bundle/server/dist/coordinator/reviewer-result.js +6 -0
  39. package/bundle/server/dist/coordinator/routing.test.js +14 -0
  40. package/bundle/server/dist/coordinator/session-timeouts.js +30 -0
  41. package/bundle/server/dist/coordinator/usage-cap.integration.test.js +1 -1
  42. package/bundle/server/dist/coordinator/worker-loop.js +18 -4
  43. package/bundle/server/dist/db/index.js +5 -0
  44. package/bundle/server/dist/db/schema.sql +3 -0
  45. package/bundle/server/dist/docs-execution-analysis.test.js +1 -0
  46. package/bundle/server/dist/repository/artifacts-for-issue.js +3 -3
  47. package/bundle/server/dist/repository/human-actions.js +14 -0
  48. package/bundle/server/dist/repository/issues.js +29 -4
  49. package/bundle/server/dist/repository/worker-sessions.js +51 -2
  50. package/bundle/server/dist/routes/human-actions.js +10 -8
  51. package/bundle/server/dist/routes/issues-execution-contract.test.js +321 -0
  52. package/bundle/server/dist/routes/issues.js +19 -2
  53. package/bundle/server/dist/runners/muse-code-jsonl.js +106 -11
  54. package/bundle/server/dist/runners/muse-config-core.js +18 -1
  55. package/bundle/server/dist/runners/muse-config.test.js +37 -0
  56. package/bundle/server/dist/runners/muse-serve-session.js +5 -0
  57. package/bundle/server/dist/runners/spawn-cli.js +116 -0
  58. package/bundle/server/dist/runners/spawn-cli.test.js +124 -0
  59. package/bundle/server/package.json +2 -2
  60. package/bundle/server/static-ui/assets/{index-yLyxRd-7.js → index-B6SVCzMR.js} +14 -14
  61. package/bundle/server/static-ui/assets/{index-DyAJNyfV.css → index-K_YcYkQU.css} +1 -1
  62. package/bundle/server/static-ui/index.html +2 -2
  63. package/bundle/shared/dist/execution-analysis.d.ts +22 -22
  64. package/bundle/shared/dist/execution-contract.d.ts +117 -0
  65. package/bundle/shared/dist/execution-contract.js +307 -0
  66. package/bundle/shared/dist/execution-contract.test.d.ts +2 -0
  67. package/bundle/shared/dist/execution-contract.test.js +499 -0
  68. package/bundle/shared/dist/execution-report.d.ts +9 -9
  69. package/bundle/shared/dist/execution-report.js +2 -0
  70. package/bundle/shared/dist/failure-cause.d.ts +4 -4
  71. package/bundle/shared/dist/failure-cause.js +2 -0
  72. package/bundle/shared/dist/human-actions.d.ts +4 -4
  73. package/bundle/shared/dist/human-actions.js +9 -0
  74. package/bundle/shared/dist/index.d.ts +85 -84
  75. package/bundle/shared/dist/index.js +3 -0
  76. package/bundle/shared/dist/issues.d.ts +146 -16
  77. package/bundle/shared/dist/issues.js +8 -0
  78. package/bundle/shared/dist/issues.test.js +1 -0
  79. package/bundle/shared/dist/outbound-draft.d.ts +12 -12
  80. package/bundle/shared/dist/worker-sessions.d.ts +10 -0
  81. package/bundle/shared/dist/worker-sessions.js +8 -0
  82. package/bundle/shared/dist/worker-sessions.test.js +1 -0
  83. package/bundle/shared/dist/workflow.d.ts +4 -4
  84. package/bundle/shared/dist/workflow.js +10 -0
  85. package/bundle/shared/dist/workflow.test.js +2 -0
  86. package/bundle/shared/package.json +1 -1
  87. package/package.json +1 -1
@@ -7,7 +7,7 @@
7
7
  // duplicate delivery is a no-op. The effect *work* itself runs elsewhere, through a leased
8
8
  // worker whose structured result comes back into applyCompletion.
9
9
  import fs from "node:fs";
10
- import { canTransitionIssue } from "@agent-dealer/shared";
10
+ import { canTransitionIssue, tryCompileContract } from "@agent-dealer/shared";
11
11
  import { getDb } from "../db/index.js";
12
12
  import { getIssue, incrementIssueRound, incrementIssueInfraAttempts, resetIssueInfraAttempts, grantReviewRetry, transitionIssue, } from "../repository/issues.js";
13
13
  import { appendWorkflowEvent, completeWorkflowInstance, getActiveWorkflowInstance, getWorkflowInstance, listWorkflowEventsForIssue, startWorkflowInstance, WorkflowAlreadyActiveError, } from "../repository/workflow-events.js";
@@ -17,7 +17,8 @@ import { ensureIssueRepoCheckout } from "../adapters/managed-repo.js";
17
17
  import { reconcileFinding, resolveFindingsAbsentFromRound } from "../repository/findings.js";
18
18
  import { normalizeReviewerResult } from "./reviewer-result.js";
19
19
  import { getAgent } from "../repository/agents.js";
20
- import { githubIssuesSync } from "../adapters/agent-health.js";
20
+ import { githubIssuesSync, invalidateMuseHealthCache } from "../adapters/agent-health.js";
21
+ import { recordMuseCapabilityOverride } from "../adapters/muse-capability.js";
21
22
  import { createIssueArtifact, latestIssueArtifact } from "../repository/artifacts.js";
22
23
  import { cancelWorkItem, enqueueWorkItem, finishWorkItem, getWorkItem, listWorkItemsForIssue, } from "../repository/work-items.js";
23
24
  import { completeSession, getWorkerSession, listWorkerSessionsForIssue } from "../repository/worker-sessions.js";
@@ -53,7 +54,13 @@ export function getTaskSnapshot(issue) {
53
54
  const artifact = latestIssueArtifact(issue.id, TASK_SNAPSHOT_ARTIFACT_KIND);
54
55
  if (artifact?.contentJson) {
55
56
  try {
56
- return JSON.parse(artifact.contentJson);
57
+ const parsed = JSON.parse(artifact.contentJson);
58
+ // Snapshots frozen before NOT-306 carry no contract — derive it from the
59
+ // frozen source description so old workflows still render what they ran.
60
+ if (parsed.executionContract === undefined) {
61
+ return { ...parsed, executionContract: tryCompileContract(parsed.description) };
62
+ }
63
+ return parsed;
57
64
  }
58
65
  catch {
59
66
  // fall through to the live-field fallback below
@@ -66,6 +73,7 @@ export function getTaskSnapshot(issue) {
66
73
  repo: issue.repo,
67
74
  baseBranch: issue.baseBranch,
68
75
  workflowVersion: WORKFLOW_VERSION,
76
+ executionContract: tryCompileContract(issue.description),
69
77
  };
70
78
  }
71
79
  /**
@@ -83,9 +91,15 @@ export function canEditParkedIssue(issue) {
83
91
  return false;
84
92
  return !listWorkItemsForIssue(issue.id).some((w) => w.status === "pending" || w.status === "leased");
85
93
  }
86
- /** Re-freezes the snapshot when the operator edited title/description/criteria; returns the changed fields (empty = no-op). */
94
+ /**
95
+ * Re-freezes the snapshot when the operator edited title/description/criteria
96
+ * (or the contract compiled from the description); returns the changed fields
97
+ * (empty = no-op). Source and compiled contract always re-freeze atomically —
98
+ * one artifact write carries both, so they can never diverge.
99
+ */
87
100
  function refreshTaskSnapshotIfEdited(issue) {
88
101
  const frozen = getTaskSnapshot(issue);
102
+ const liveContract = tryCompileContract(issue.description);
89
103
  const changed = [];
90
104
  if (frozen.title !== issue.title)
91
105
  changed.push("title");
@@ -93,6 +107,9 @@ function refreshTaskSnapshotIfEdited(issue) {
93
107
  changed.push("description");
94
108
  if (frozen.acceptanceCriteria !== (issue.acceptanceCriteria ?? ""))
95
109
  changed.push("acceptanceCriteria");
110
+ if (JSON.stringify(frozen.executionContract ?? null) !== JSON.stringify(liveContract ?? null)) {
111
+ changed.push("executionContract");
112
+ }
96
113
  if (changed.length > 0)
97
114
  freezeTaskSnapshot(issue);
98
115
  return changed;
@@ -109,6 +126,7 @@ function freezeTaskSnapshot(issue) {
109
126
  repo: issue.repo,
110
127
  baseBranch: issue.baseBranch,
111
128
  workflowVersion: WORKFLOW_VERSION,
129
+ executionContract: tryCompileContract(issue.description),
112
130
  },
113
131
  });
114
132
  }
@@ -973,6 +991,11 @@ function questionFor(actionType, reason, resumeAsReviewer = false, mergeFailure
973
991
  // Run-scoped action type directly with its own question text (NOT-95). Case exists
974
992
  // only so this function stays total over HumanActionType.
975
993
  throw new Error("outbound_delivery_interaction_required is not raised through applyEffect");
994
+ case "muse_capability":
995
+ // Never raised through applyEffect — muse-capability.ts creates this action
996
+ // directly with its own question text (NOT-308). Case exists only so this
997
+ // function stays total over HumanActionType.
998
+ throw new Error("muse_capability is not raised through applyEffect");
976
999
  }
977
1000
  }
978
1001
  /** Exported for db/migrate-to-issues.ts, which must populate the same response options on
@@ -1014,6 +1037,11 @@ export function responseOptionsFor(actionType, resumeAsReviewer = false, opts =
1014
1037
  case "outbound_delivery_interaction_required":
1015
1038
  // Never actually raised through applyEffect — see questionFor's identical case.
1016
1039
  throw new Error("outbound_delivery_interaction_required is not raised through applyEffect");
1040
+ case "muse_capability":
1041
+ // NOT-308: created with its own stored options (muse-capability.ts) — acknowledge
1042
+ // admits on that Muse version (missing) or dismisses (unverified). Never raised
1043
+ // through applyEffect; case exists only so this function stays total.
1044
+ return [{ choice: "acknowledge", label: "Acknowledge" }];
1017
1045
  }
1018
1046
  }
1019
1047
  function parseContinuationPreview(json) {
@@ -1052,6 +1080,61 @@ export function resolveHumanActionAndAdvance(actionId, resolvedBy, choice, opts)
1052
1080
  const normalizedNote = normalizeResolutionNote(opts?.note);
1053
1081
  if (!normalizedNote.ok)
1054
1082
  return { ok: false, code: 400, error: normalizedNote.error };
1083
+ // NOT-308: a Muse capability escalation resolves without touching the workflow — it
1084
+ // has no round to spend and no merge to park, and parseHumanResolution (below)
1085
+ // deliberately rejects it since there is no workflow outcome to map. Acknowledging a
1086
+ // `missing` verdict records a per-version override so the admission gate stops
1087
+ // blocking that version; acknowledging `unverified` just dismisses. Works with or
1088
+ // without an active instance (a blocked admission never has one).
1089
+ if (action.actionType === "muse_capability") {
1090
+ if (choice !== "acknowledge") {
1091
+ return { ok: false, code: 400, error: `Invalid choice "${choice}" for muse_capability` };
1092
+ }
1093
+ if (!action.issueId)
1094
+ return { ok: false, code: 500, error: "Human action has no issue" };
1095
+ const museIssue = getIssue(action.issueId);
1096
+ if (!museIssue)
1097
+ return { ok: false, code: 404, error: "Issue not found" };
1098
+ const museInstance = getActiveWorkflowInstance(action.issueId);
1099
+ let evidence = null;
1100
+ try {
1101
+ evidence = action.evidenceJson
1102
+ ? JSON.parse(action.evidenceJson)
1103
+ : null;
1104
+ }
1105
+ catch {
1106
+ evidence = null;
1107
+ }
1108
+ // File first, then the row: if the resolve below fails the action stays open and
1109
+ // the operator retries idempotently; the reverse order could leave a resolved
1110
+ // action whose version still blocks with no way to re-acknowledge.
1111
+ if (evidence?.kind === "missing" && typeof evidence.version === "string" && evidence.version) {
1112
+ recordMuseCapabilityOverride(evidence.version);
1113
+ }
1114
+ // The lifted block must be visible on the next health read, not after the cache TTL.
1115
+ invalidateMuseHealthCache();
1116
+ return getDb().transaction(() => {
1117
+ resolveHumanAction(actionId, resolvedBy, { choice });
1118
+ appendWorkflowEvent({
1119
+ issueId: museIssue.id,
1120
+ ...(museInstance ? { workflowInstanceId: museInstance.id } : {}),
1121
+ type: "human_action.resolved",
1122
+ actorType: "human",
1123
+ actorRef: resolvedBy,
1124
+ stage: museIssue.status,
1125
+ round: museIssue.currentRound,
1126
+ payload: { actionType: action.actionType, choice },
1127
+ });
1128
+ return {
1129
+ ok: true,
1130
+ issueStatus: museIssue.status,
1131
+ nextWorkItemId: null,
1132
+ instanceCompleted: false,
1133
+ restarted: false,
1134
+ triggerReflect: false,
1135
+ };
1136
+ })();
1137
+ }
1055
1138
  const resolution = parseHumanResolution(action.actionType, choice, normalizedNote.note);
1056
1139
  if (!resolution) {
1057
1140
  return { ok: false, code: 400, error: `Invalid choice "${choice}" for ${action.actionType}` };
@@ -1446,7 +1529,9 @@ export async function resolveHumanActionAndAdvanceAsync(actionId, resolvedBy, ch
1446
1529
  return {
1447
1530
  ok: true,
1448
1531
  issueStatus: merged.issueStatus,
1449
- nextWorkItemId: null,
1532
+ // NOT-310: a retry_merge that hit a textual conflict queues a repair round —
1533
+ // report it like any other queued round instead of dropping it.
1534
+ nextWorkItemId: merged.nextWorkItemId,
1450
1535
  instanceCompleted: merged.instanceCompleted,
1451
1536
  restarted: false,
1452
1537
  triggerReflect: merged.triggerReflect,
@@ -31,15 +31,17 @@ const { deckOutageBackoffMs } = await import("./deck-outage-config.js");
31
31
  const UNREACHABLE = "Agent Deck is unreachable — fetch failed";
32
32
  before(() => migrate());
33
33
  beforeEach(() => {
34
+ // NOT-305: session-linked playbook_use_receipt artifacts reference worker_sessions,
35
+ // so child rows (artifacts, usage_events) go before their parents.
34
36
  getDb().exec(`
35
37
  DELETE FROM review_publications;
38
+ DELETE FROM artifacts;
36
39
  DELETE FROM work_items;
37
40
  DELETE FROM human_actions;
38
41
  DELETE FROM workflow_events;
39
42
  DELETE FROM findings;
40
- DELETE FROM worker_sessions;
41
- DELETE FROM artifacts;
42
43
  DELETE FROM usage_events;
44
+ DELETE FROM worker_sessions;
43
45
  DELETE FROM workflow_instances;
44
46
  DELETE FROM issues;
45
47
  DELETE FROM runtime_availability;
@@ -27,7 +27,10 @@ import { ensureIssueRepoCheckout, roleWorktreePathForResolution, resolveCheckout
27
27
  import { baseRefCandidates, inspectBranchProgress } from "./branch-progress.js";
28
28
  import { prepareWorkerDeckConnection, releaseWorkerDeckConnection, verifyWorkerDeckConnection } from "../adapters/agent-deck-bind.js";
29
29
  import { realGithubAdapter, pollPrChecks } from "../adapters/github.js";
30
- import { getWorkerSession, patchRunningSession, recordSessionProcess, setSessionInputSha } from "../repository/worker-sessions.js";
30
+ import { museCapabilitySafetyNetAfterSession } from "../adapters/muse-capability.js";
31
+ import { getOrAssignSessionCorrelationId, getWorkerSession, mergeSessionMetadata, patchRunningSession, recordSessionProcess, setSessionInputSha } from "../repository/worker-sessions.js";
32
+ import { museIdleMinutes, museStallMetadata } from "./muse-spawn.js";
33
+ import { museIdleTimeoutMs } from "./session-timeouts.js";
31
34
  import { COORDINATOR_PROCESS_OWNER, readProcessStartTime } from "./process-liveness.js";
32
35
  import { checkDeveloperWorktreeOwnerLiveness } from "./worktree-owner-liveness.js";
33
36
  import { developerSessionTimeoutMs } from "./session-timeouts.js";
@@ -87,6 +90,23 @@ async function crashOrTimeoutOutcome(opts) {
87
90
  ...(opts.logPath ? { logPath: opts.logPath } : {}),
88
91
  };
89
92
  }
93
+ /**
94
+ * NOT-307: idle-kill description for the timeout reason, or null for anything else
95
+ * (wall-clock timeout, crash, success). Minutes prefer the measured last-activity
96
+ * time and fall back to the configured bound; the last tool name rides along so the
97
+ * recorded reason — and the `muse_no_progress` classifier — names the stall point.
98
+ */
99
+ function idleCrashInfoFor(spawned) {
100
+ if (!spawned.idleTimedOut)
101
+ return null;
102
+ return {
103
+ minutes: museIdleMinutes({
104
+ lastActivityAt: spawned.muse?.lastActivityAt ?? null,
105
+ idleTimeoutMs: museIdleTimeoutMs(),
106
+ }),
107
+ lastToolName: spawned.muse?.lastToolName ?? null,
108
+ };
109
+ }
90
110
  export const developerEffectConfig = {
91
111
  get sessionTimeoutMs() {
92
112
  return developerSessionTimeoutMs();
@@ -601,6 +621,9 @@ export async function runDeveloperEffect(ctx, deps = defaultDeps) {
601
621
  }
602
622
  const session = getWorkerSession(sessionId);
603
623
  const snapshot = parseProfileSnapshot(session?.profileSnapshotJson);
624
+ // NOT-305: opaque per-session Deck correlation UUID, persisted before spawn and
625
+ // carried as observability metadata in every runtime launch config below.
626
+ const deckCorrelationId = session?.deckCorrelationId ?? getOrAssignSessionCorrelationId(sessionId);
604
627
  let taskSnapshot = getTaskSnapshot(issue);
605
628
  const runtime = snapshot?.runtime ?? "claude_code";
606
629
  const round = workItem.round;
@@ -807,6 +830,7 @@ export async function runDeveloperEffect(ctx, deps = defaultDeps) {
807
830
  worktreePath,
808
831
  runtime,
809
832
  policy,
833
+ correlationId: deckCorrelationId,
810
834
  verifyCallTool: deps.deckCallTool,
811
835
  });
812
836
  if (!prepared.ok) {
@@ -862,6 +886,22 @@ export async function runDeveloperEffect(ctx, deps = defaultDeps) {
862
886
  const scopeDecisionNote = typeof payload.scopeDecisionNote === "string" && payload.scopeDecisionNote.trim()
863
887
  ? payload.scopeDecisionNote.trim()
864
888
  : undefined;
889
+ // NOT-310: the conflict directive rides only the work item queued by the merge
890
+ // path — exactly this round reads it; nothing is re-read from the DB. Shaped
891
+ // defensively: a malformed directive renders nothing rather than breaking the
892
+ // prompt (conflictRepairSection also guards on empty base/branch).
893
+ const conflictRepair = payload.conflictRepair &&
894
+ typeof payload.conflictRepair === "object" &&
895
+ typeof payload.conflictRepair.baseBranch === "string" &&
896
+ typeof payload.conflictRepair.branch === "string"
897
+ ? {
898
+ baseBranch: payload.conflictRepair.baseBranch,
899
+ branch: payload.conflictRepair.branch,
900
+ files: Array.isArray(payload.conflictRepair.files)
901
+ ? payload.conflictRepair.files.filter((f) => typeof f === "string")
902
+ : [],
903
+ }
904
+ : undefined;
865
905
  let priorConclusion;
866
906
  let priorVerificationReceipt;
867
907
  if (retryReason) {
@@ -897,6 +937,7 @@ export async function runDeveloperEffect(ctx, deps = defaultDeps) {
897
937
  museDeveloper: isMuse,
898
938
  guidance: guidance.length ? guidance : undefined,
899
939
  scopeDecisionNote,
940
+ conflictRepair,
900
941
  });
901
942
  // NOT-83 review finding: the session is marked `running` (worker-loop.ts) before this
902
943
  // handler ever runs, so an abort landing during worktree setup/deck-bind above already
@@ -998,6 +1039,9 @@ export async function runDeveloperEffect(ctx, deps = defaultDeps) {
998
1039
  // NOT-278: frozen profile deck → the Muse exec attempt (deck headers + pre-spawn
999
1040
  // verify). Other runtimes carry their deck via mcpConfigPath/mcpEnv and ignore this.
1000
1041
  deckId: snapshot?.deckId ?? null,
1042
+ // NOT-305: opaque per-session Deck correlation UUID → the Muse settings
1043
+ // `x-agent-deck-correlation-id` observability header.
1044
+ deckCorrelationId,
1001
1045
  prompt,
1002
1046
  cwd: worktreePath,
1003
1047
  timeoutMs: developerEffectConfig.sessionTimeoutMs,
@@ -1069,6 +1113,18 @@ export async function runDeveloperEffect(ctx, deps = defaultDeps) {
1069
1113
  // here on (usage, receipt, push, PR verify, checks poll) is post-spawn and must
1070
1114
  // not be mislabeled as a session that could not start.
1071
1115
  sessionStarted = true;
1116
+ // NOT-307: stall evidence for every Muse session (timed out or not) — merged
1117
+ // into metadata_json while the row is still running, so the classifier and the
1118
+ // operator can read last-activity evidence whatever the outcome. Best-effort:
1119
+ // evidence must never fail the attempt it describes.
1120
+ if (spawned.muse) {
1121
+ try {
1122
+ mergeSessionMetadata(sessionId, museStallMetadata(spawned.muse));
1123
+ }
1124
+ catch {
1125
+ // ignore
1126
+ }
1127
+ }
1072
1128
  // NOT-169: exactly one agent.completed per spawned process, at child exit and before
1073
1129
  // any receipt mining, usage extraction, or validation. Raw outcome only — no
1074
1130
  // failure classification.
@@ -1094,6 +1150,26 @@ export async function runDeveloperEffect(ctx, deps = defaultDeps) {
1094
1150
  // ignore
1095
1151
  }
1096
1152
  }
1153
+ // NOT-308 safety net: a Muse session that left the worktree dirty without a
1154
+ // single shell tool call is what a shell-less build looks like from the outside —
1155
+ // when the running version is not yet the confirmed baseline, re-verify it in the
1156
+ // background immediately (bypassing the error backoff) and escalate a `missing`
1157
+ // verdict on this issue. Fire-and-forget: never throws, never changes this outcome.
1158
+ // Dirtiness is read here, before salvage/commit handling below can clean the tree.
1159
+ if (runtime === "muse_code") {
1160
+ try {
1161
+ const dirtyAtSessionEnd = !(await isWorktreeClean(worktreePath).catch(() => true));
1162
+ museCapabilitySafetyNetAfterSession({
1163
+ issueId: issue.id,
1164
+ runtime,
1165
+ logPath: spawned.logPath,
1166
+ dirty: dirtyAtSessionEnd,
1167
+ });
1168
+ }
1169
+ catch {
1170
+ // Observational only — the session outcome stands on its own.
1171
+ }
1172
+ }
1097
1173
  // NOT-130: record suite evidence even when the session later fails/times out — an
1098
1174
  // interrupted-but-verified tip must carry the receipt into the retry prompt.
1099
1175
  const verificationReceipt = await persistVerificationReceiptIfAny({
@@ -1287,6 +1363,7 @@ export async function runDeveloperEffect(ctx, deps = defaultDeps) {
1287
1363
  timedOut: Boolean(spawned.timedOut),
1288
1364
  logPath: spawned.logPath,
1289
1365
  runtime,
1366
+ idle: idleCrashInfoFor(spawned),
1290
1367
  });
1291
1368
  const salvageNote = `Salvaged uncommitted work as ${salvaged.message} (${salvaged.commitSha.slice(0, 7)}).`;
1292
1369
  const crashOutcome = await crashOrTimeoutOutcome({
@@ -1341,6 +1418,7 @@ export async function runDeveloperEffect(ctx, deps = defaultDeps) {
1341
1418
  timedOut: Boolean(spawned.timedOut),
1342
1419
  logPath: spawned.logPath,
1343
1420
  runtime,
1421
+ idle: idleCrashInfoFor(spawned),
1344
1422
  }),
1345
1423
  worktreePath,
1346
1424
  logPath: spawned.logPath,
@@ -151,6 +151,12 @@ export function classifyFailureReason(reason) {
151
151
  const text = (reason ?? "").trim();
152
152
  if (!text)
153
153
  return { code: "unknown", domain: "unknown", confidence: "low" };
154
+ // NOT-307: an idle-watchdog kill names itself ("made no progress … idle timeout").
155
+ // Checked before the timeout rules: the reason contains both "timeout" and "tool"
156
+ // ("last tool: …"), which would otherwise misread as tool_test_timeout.
157
+ if (/made no progress|killed on the idle timeout/i.test(text)) {
158
+ return { code: "muse_no_progress", domain: "infrastructure", confidence: "high" };
159
+ }
154
160
  // A bare timeout with no tool/test in flight is unknown by rule — do not guess.
155
161
  if (/timed? ?out|deadline/i.test(text) && !/tool|test|jest|vitest|pytest|subprocess|command/i.test(text)) {
156
162
  return { code: "unknown", domain: "unknown", confidence: "low" };
@@ -88,6 +88,13 @@ test("classifier keeps ambiguous timeouts unknown", () => {
88
88
  assert.equal(classifyFailureReason(null).code, "unknown");
89
89
  assert.equal(classifyFailureReason("").code, "unknown");
90
90
  });
91
+ test("NOT-307: classifier reads an idle-watchdog kill as muse_no_progress", () => {
92
+ const reason = "Developer session made no progress for 20 minutes. (last tool: npm_test) Killed on the idle timeout.";
93
+ const got = classifyFailureReason(reason);
94
+ assert.equal(got.code, "muse_no_progress");
95
+ assert.equal(got.domain, "infrastructure");
96
+ assert.equal(got.confidence, "high");
97
+ });
91
98
  test("classifier anchors keep author/release/unrelated spawn out of buckets", () => {
92
99
  // `auth` must not match "author"; `lease` must not match "release"/"please".
93
100
  assert.equal(classifyFailureReason("the author released a fix").code, "unknown");
@@ -329,6 +329,27 @@ export function classifyAttemptFailure(input) {
329
329
  rawReason: rawReason || "Agent Deck failure.",
330
330
  });
331
331
  }
332
+ // NOT-307: an idle-watchdog kill is a specific, high-confidence infrastructure
333
+ // cause — never `unknown`. It sorts after log evidence (an auth/429 already on
334
+ // record names an earlier, more specific root cause for the same stall) but
335
+ // before the outcome-level timeout guess, so a hung tool/test in flight does not
336
+ // demote the stall to `tool_test_timeout`. The raw reason carries the idle
337
+ // minutes and the last tool name, reusing the recorded reason when it already
338
+ // names them (reasonForSessionCrash writes them at kill time).
339
+ if (input.idleTimedOut) {
340
+ const minutes = typeof input.idleForMs === "number" && Number.isFinite(input.idleForMs) && input.idleForMs >= 0
341
+ ? Math.max(1, Math.round(input.idleForMs / 60_000))
342
+ : null;
343
+ const alreadySpecific = /made no progress/i.test(rawReason);
344
+ signals.push({
345
+ code: "muse_no_progress",
346
+ confidence: "high",
347
+ evidenceSource: "outcome_kind",
348
+ rawReason: alreadySpecific
349
+ ? rawReason
350
+ : `Muse developer session made no progress${minutes !== null ? ` for ${minutes} minute${minutes === 1 ? "" : "s"}` : ""} (last tool: ${input.lastToolName ?? "none"}).${rawReason ? ` ${rawReason}` : ""}`,
351
+ });
352
+ }
332
353
  const outcomeSignal = detectOutcomeSignal(outcomeKind, rawReason, {
333
354
  timedOut,
334
355
  toolInFlight,
@@ -413,6 +434,36 @@ function agentCompletedEvidence(workerSessionId) {
413
434
  }
414
435
  return out;
415
436
  }
437
+ /**
438
+ * NOT-307: idle-watchdog evidence read back out of `worker_sessions.metadata_json`
439
+ * (written by the developer effect for every Muse session). Silence duration is
440
+ * measured against `occurredAtMs` — the kill/completion time — from the recorded
441
+ * `lastActivityAt`. Anything unparseable reads as no idle evidence, never a throw.
442
+ */
443
+ export function museStallFromMetadata(metadataJson, occurredAtMs) {
444
+ const none = { idleTimedOut: false, idleForMs: null, lastToolName: null };
445
+ if (!metadataJson)
446
+ return none;
447
+ try {
448
+ const parsed = JSON.parse(metadataJson);
449
+ if (!parsed || typeof parsed !== "object" || Array.isArray(parsed))
450
+ return none;
451
+ const rec = parsed;
452
+ if (rec.idleTimedOut !== true)
453
+ return none;
454
+ const lastToolName = typeof rec.lastToolName === "string" ? rec.lastToolName : null;
455
+ let idleForMs = null;
456
+ if (occurredAtMs !== null && Number.isFinite(occurredAtMs) && typeof rec.lastActivityAt === "string") {
457
+ const at = Date.parse(rec.lastActivityAt);
458
+ if (Number.isFinite(at) && occurredAtMs >= at)
459
+ idleForMs = occurredAtMs - at;
460
+ }
461
+ return { idleTimedOut: true, idleForMs, lastToolName };
462
+ }
463
+ catch {
464
+ return none;
465
+ }
466
+ }
416
467
  /**
417
468
  * Backfill-on-read for legacy rows: classify from existing session error/log
418
469
  * evidence without mutating any append-only event. Always quality "inferred" —
@@ -421,6 +472,9 @@ function agentCompletedEvidence(workerSessionId) {
421
472
  export function backfillCausesForSession(session, opts) {
422
473
  const agent = agentCompletedEvidence(session.id);
423
474
  const outcomeKind = opts?.outcomeKind ?? null;
475
+ const occurredAt = opts?.occurredAt ?? session.completedAt ?? session.updatedAt;
476
+ const occurredAtMs = occurredAt ? Date.parse(occurredAt) : NaN;
477
+ const stall = museStallFromMetadata(session.metadataJson ?? null, Number.isFinite(occurredAtMs) ? occurredAtMs : null);
424
478
  return classifyAttemptFailure({
425
479
  outcomeKind,
426
480
  sessionErrorJson: session.errorJson,
@@ -428,8 +482,11 @@ export function backfillCausesForSession(session, opts) {
428
482
  runtime: session.runtime,
429
483
  exitCode: session.exitCode ?? agent.exitCode,
430
484
  timedOut: outcomeKind === "timed_out" || agent.timedOut,
485
+ idleTimedOut: stall.idleTimedOut || undefined,
486
+ idleForMs: stall.idleForMs,
487
+ lastToolName: stall.lastToolName,
431
488
  hostSuspended: agent.hostSuspended,
432
- occurredAt: opts?.occurredAt ?? session.completedAt ?? session.updatedAt,
489
+ occurredAt,
433
490
  eventCursor: opts?.eventCursor ?? null,
434
491
  sessionId: session.id,
435
492
  eventId: opts?.eventId ?? null,
@@ -449,6 +506,9 @@ export function recordCausesForWorkerFailedEvent(opts) {
449
506
  const session = opts.event.workerSessionId ? getWorkerSession(opts.event.workerSessionId) : null;
450
507
  const agent = agentCompletedEvidence(opts.event.workerSessionId);
451
508
  const outcomeTimedOut = opts.outcomeKind === "timed_out";
509
+ const occurredAtMs = Date.parse(opts.event.ts);
510
+ // NOT-307: an idle kill's stall evidence rides on the session row's metadata.
511
+ const stall = museStallFromMetadata(session?.metadataJson ?? null, Number.isFinite(occurredAtMs) ? occurredAtMs : null);
452
512
  const causes = classifyAttemptFailure({
453
513
  outcomeKind: opts.outcomeKind,
454
514
  outcomeReason: opts.outcomeReason,
@@ -458,6 +518,9 @@ export function recordCausesForWorkerFailedEvent(opts) {
458
518
  runtime: session?.runtime ?? undefined,
459
519
  exitCode: session?.exitCode ?? agent.exitCode,
460
520
  timedOut: outcomeTimedOut || agent.timedOut,
521
+ idleTimedOut: stall.idleTimedOut || undefined,
522
+ idleForMs: stall.idleForMs,
523
+ lastToolName: stall.lastToolName,
461
524
  hostSuspended: agent.hostSuspended,
462
525
  recovery: opts.recovery,
463
526
  occurredAt: opts.event.ts,
@@ -8,7 +8,7 @@ import assert from "node:assert/strict";
8
8
  import fs from "node:fs";
9
9
  import os from "node:os";
10
10
  import path from "node:path";
11
- import { classifyAttemptFailure, orderAttemptCauses, } from "./failure-cause.js";
11
+ import { backfillCausesForSession, classifyAttemptFailure, museStallFromMetadata, orderAttemptCauses, } from "./failure-cause.js";
12
12
  import { PRESUMED_DEAD_REASON } from "./failure-reason.js";
13
13
  function writeLog(body) {
14
14
  const dir = fs.mkdtempSync(path.join(os.tmpdir(), "dealer-failure-cause-"));
@@ -274,6 +274,97 @@ test("reclaimed session with provider/auth log evidence keeps it primary", () =>
274
274
  assert.equal(reclaimedAuth[0].primary, true);
275
275
  assert.equal(reclaimedAuth.find((c) => c.code === "coordinator_crash")?.primary, false);
276
276
  });
277
+ // NOT-307: an idle-watchdog kill classifies as muse_no_progress (infrastructure,
278
+ // high confidence) instead of unknown — with the idle minutes and last tool name
279
+ // in the raw reason — even when a tool was in flight (a hung tool is still a stall).
280
+ test("idle-watchdog kill classifies as muse_no_progress, not unknown", () => {
281
+ const idle = primary(classifyAttemptFailure({
282
+ outcomeKind: "timed_out",
283
+ outcomeReason: "Developer session made no progress for 20 minutes. (last tool: npm_test) Killed on the idle timeout.",
284
+ timedOut: true,
285
+ idleTimedOut: true,
286
+ idleForMs: 20 * 60_000,
287
+ lastToolName: "npm_test",
288
+ }));
289
+ assert.equal(idle.code, "muse_no_progress");
290
+ assert.equal(idle.domain, "infrastructure");
291
+ assert.equal(idle.confidence, "high");
292
+ assert.match(idle.rawReason, /20 minutes/);
293
+ assert.match(idle.rawReason, /npm_test/);
294
+ });
295
+ test("idle kill beats the tool_test_timeout guess and the unknown fallback", () => {
296
+ // A hung test in flight plus an idle kill: still the stall, not the tool timeout.
297
+ const hungTool = primary(classifyAttemptFailure({
298
+ outcomeKind: "timed_out",
299
+ outcomeReason: "vitest run timed out after 30000ms",
300
+ timedOut: true,
301
+ toolInFlight: true,
302
+ idleTimedOut: true,
303
+ idleForMs: 21 * 60_000,
304
+ lastToolName: "vitest",
305
+ }));
306
+ assert.equal(hungTool.code, "muse_no_progress");
307
+ // A bare timed_out with no idle evidence still classifies as before (unknown here).
308
+ const plain = primary(classifyAttemptFailure({ outcomeKind: "timed_out", timedOut: true }));
309
+ assert.equal(plain.code, "unknown");
310
+ });
311
+ test("idle kill composes the reason when the recorded text lacks minutes and tool", () => {
312
+ const composed = primary(classifyAttemptFailure({
313
+ outcomeKind: "timed_out",
314
+ outcomeReason: "Developer session timed out.",
315
+ timedOut: true,
316
+ idleTimedOut: true,
317
+ idleForMs: 22 * 60_000,
318
+ lastToolName: null,
319
+ }));
320
+ assert.equal(composed.code, "muse_no_progress");
321
+ assert.match(composed.rawReason, /22 minutes/);
322
+ assert.match(composed.rawReason, /last tool: none/);
323
+ assert.match(composed.rawReason, /Developer session timed out\./);
324
+ });
325
+ // NOT-307: stall evidence round-trips through worker_sessions.metadata_json —
326
+ // the record/backfill paths read the same five fields the developer effect wrote.
327
+ test("museStallFromMetadata reads idle evidence, and ignores anything else", () => {
328
+ const at = Date.parse("2026-10-01T19:40:00.000Z");
329
+ const stall = museStallFromMetadata(JSON.stringify({
330
+ lastActivityAt: "2026-10-01T19:20:00.000Z",
331
+ toolCallCount: 24,
332
+ lastToolName: "npm_test",
333
+ firstOutputMs: 900,
334
+ idleTimedOut: true,
335
+ }), at);
336
+ assert.equal(stall.idleTimedOut, true);
337
+ assert.equal(stall.idleForMs, 20 * 60_000);
338
+ assert.equal(stall.lastToolName, "npm_test");
339
+ assert.deepEqual(museStallFromMetadata(null, at).idleTimedOut, false);
340
+ assert.deepEqual(museStallFromMetadata("not-json{", at).idleTimedOut, false);
341
+ assert.deepEqual(museStallFromMetadata(JSON.stringify({ idleTimedOut: false }), at).idleTimedOut, false, "a clean session's metadata is not idle evidence");
342
+ assert.deepEqual(museStallFromMetadata(JSON.stringify({ idleTimedOut: true }), null).idleForMs, null, "no kill time means no measured silence");
343
+ });
344
+ test("backfill over an idle-kill row classifies muse_no_progress", () => {
345
+ const causes = backfillCausesForSession({
346
+ id: "s-idle-backfill",
347
+ errorJson: JSON.stringify({
348
+ reason: "Developer session made no progress for 20 minutes. (last tool: npm_test) Killed on the idle timeout.",
349
+ }),
350
+ logPath: null,
351
+ runtime: "muse_code",
352
+ exitCode: 124,
353
+ completedAt: "2026-10-01T19:40:00.000Z",
354
+ updatedAt: "2026-10-01T19:40:00.000Z",
355
+ metadataJson: JSON.stringify({
356
+ lastActivityAt: "2026-10-01T19:20:00.000Z",
357
+ toolCallCount: 24,
358
+ lastToolName: "npm_test",
359
+ firstOutputMs: 900,
360
+ idleTimedOut: true,
361
+ }),
362
+ }, { outcomeKind: "timed_out" });
363
+ const found = primary(causes);
364
+ assert.equal(found.code, "muse_no_progress");
365
+ assert.equal(found.quality, "inferred");
366
+ assert.match(found.rawReason, /20 minutes/);
367
+ });
277
368
  test("usage-capped and tool-timeout evidence map to capacity/task", () => {
278
369
  const capped = primary(classifyAttemptFailure({ outcomeKind: "usage_capped", outcomeReason: "claude_code usage capped — plan limit rejected" }));
279
370
  assert.equal(capped.code, "provider_capacity_rate_limit");
@@ -184,6 +184,13 @@ export function reasonForSessionCrash(opts) {
184
184
  const classified = classifyRunnerLogFailure(opts.logPath, opts.runtime);
185
185
  if (classified)
186
186
  return classified;
187
+ if (opts.timedOut && opts.idle) {
188
+ const mins = typeof opts.idle.minutes === "number" && Number.isFinite(opts.idle.minutes)
189
+ ? ` for ${opts.idle.minutes} minute${opts.idle.minutes === 1 ? "" : "s"}`
190
+ : "";
191
+ const tool = opts.idle.lastToolName ? ` (last tool: ${opts.idle.lastToolName})` : " (no tool call seen)";
192
+ return `Developer session made no progress${mins}.${tool} Killed on the idle timeout.`;
193
+ }
187
194
  return opts.timedOut ? "Developer session timed out." : "Developer session failed or crashed.";
188
195
  }
189
196
  function outcomeExplicitReason(outcome) {
@@ -206,6 +206,20 @@ test("NOT-133: a worker.failed event for an auth death carries the auth reason",
206
206
  test("reasonForSessionCrash falls back when log is clean", () => {
207
207
  assert.equal(reasonForSessionCrash({ timedOut: false, logPath: writeLog('{"type":"result"}\n') }), "Developer session failed or crashed.");
208
208
  });
209
+ test("NOT-307: reasonForSessionCrash names the idle minutes and last tool on an idle kill", () => {
210
+ assert.equal(reasonForSessionCrash({
211
+ timedOut: true,
212
+ logPath: writeLog('{"type":"result"}\n'),
213
+ idle: { minutes: 20, lastToolName: "npm_test" },
214
+ }), "Developer session made no progress for 20 minutes. (last tool: npm_test) Killed on the idle timeout.");
215
+ assert.equal(reasonForSessionCrash({
216
+ timedOut: true,
217
+ logPath: writeLog('{"type":"result"}\n'),
218
+ idle: { minutes: null, lastToolName: null },
219
+ }), "Developer session made no progress. (no tool call seen) Killed on the idle timeout.");
220
+ // Wall-clock timeouts read exactly as before when no idle info is given.
221
+ assert.equal(reasonForSessionCrash({ timedOut: true, logPath: writeLog('{"type":"result"}\n') }), "Developer session timed out.");
222
+ });
209
223
  test("parseErrorJsonReason reads recovery presumed-dead shape", () => {
210
224
  assert.equal(parseErrorJsonReason(JSON.stringify({ reason: PRESUMED_DEAD_REASON })), PRESUMED_DEAD_REASON);
211
225
  });