agent-dealer 1.2.15 → 1.2.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/bundle/server/dist/adapters/agent-health.js +21 -0
  2. package/bundle/server/dist/coordinator/auto-merge.js +8 -1
  3. package/bundle/server/dist/coordinator/commands.js +153 -13
  4. package/bundle/server/dist/coordinator/human-resolution.js +4 -0
  5. package/bundle/server/dist/coordinator/human-resolution.test.js +292 -2
  6. package/bundle/server/dist/coordinator/projection.js +26 -3
  7. package/bundle/server/dist/coordinator/projection.test.js +8 -1
  8. package/bundle/server/dist/coordinator/routing.js +96 -0
  9. package/bundle/server/dist/coordinator/routing.test.js +163 -0
  10. package/bundle/server/dist/coordinator/runtime-auth-park.js +97 -0
  11. package/bundle/server/dist/coordinator/worker-loop.js +15 -0
  12. package/bundle/server/dist/docs-host-awake.test.js +16 -0
  13. package/bundle/server/dist/index.js +14 -1
  14. package/bundle/server/dist/power/host-awake-lifecycle.js +30 -0
  15. package/bundle/server/dist/power/host-awake-lifecycle.test.js +181 -0
  16. package/bundle/server/dist/power/host-awake.js +162 -0
  17. package/bundle/server/dist/power/host-awake.test.js +192 -0
  18. package/bundle/server/dist/power/sleep-timer.js +86 -0
  19. package/bundle/server/dist/power/sleep-timer.test.js +79 -0
  20. package/bundle/server/dist/power/status.js +31 -0
  21. package/bundle/server/dist/power/status.test.js +60 -0
  22. package/bundle/server/dist/routes/index.js +3 -0
  23. package/bundle/server/package.json +4 -3
  24. package/bundle/server/static-ui/assets/index-DL8mEXzN.css +1 -0
  25. package/bundle/server/static-ui/assets/{index-DVyqblie.js → index-FCHzrnLr.js} +13 -13
  26. package/bundle/server/static-ui/index.html +2 -2
  27. package/bundle/shared/package.json +1 -1
  28. package/dist/bin.js +3 -0
  29. package/dist/node-hardening.d.ts +9 -0
  30. package/dist/node-hardening.js +38 -0
  31. package/dist/ports.js +54 -14
  32. package/dist/stop-einval.test.d.ts +1 -0
  33. package/dist/stop-einval.test.js +215 -0
  34. package/dist/stop.d.ts +6 -1
  35. package/dist/stop.js +18 -7
  36. package/dist/version-output.test.js +23 -16
  37. package/package.json +1 -1
  38. package/bundle/server/static-ui/assets/index-COYJxD1y.css +0 -1
@@ -104,6 +104,14 @@ let runCommandImpl = defaultRunCommand;
104
104
  export function setRunCommandForTests(fn) {
105
105
  runCommandImpl = fn ?? defaultRunCommand;
106
106
  }
107
+ let runtimeIssuesUncachedForTests = null;
108
+ /**
109
+ * Replace the live auth probe in coordinator tests. Pass `null` to restore the real probe.
110
+ * Cleared by {@link clearAgentHealthCaches}.
111
+ */
112
+ export function setRuntimeIssuesUncachedForTests(fn) {
113
+ runtimeIssuesUncachedForTests = fn;
114
+ }
107
115
  function runCommand(cmd, args, timeoutMs = DEFAULT_PROBE_TIMEOUT_MS, env) {
108
116
  return runCommandImpl(cmd, args, timeoutMs, env);
109
117
  }
@@ -122,6 +130,7 @@ export function clearAgentHealthCaches() {
122
130
  githubIssueCache = null;
123
131
  cursorSoftFailStreak = 0;
124
132
  cursorLastHealthyAt = null;
133
+ runtimeIssuesUncachedForTests = null;
125
134
  resetMuseCapabilityStateForTests();
126
135
  }
127
136
  function isSoftCursorProbeIssue(issue) {
@@ -256,6 +265,9 @@ async function museRuntimeIssues() {
256
265
  }
257
266
  /** Exported for direct testing — bypasses the 60s cache in runtimeIssues(). */
258
267
  export async function runtimeIssuesUncached(runtime) {
268
+ if (runtimeIssuesUncachedForTests) {
269
+ return await runtimeIssuesUncachedForTests(runtime);
270
+ }
259
271
  const issues = [];
260
272
  if (runtime === "claude_code") {
261
273
  if (!claudeBinExists()) {
@@ -542,6 +554,15 @@ export async function healthForAgent(agent, agentDeckOnline, runtimeIssuesByRunt
542
554
  };
543
555
  }
544
556
  export async function listAgentsWithHealth(agents) {
557
+ // NOT-369: ensure the one-time AC sleep-timer notice is computed for the health surface
558
+ // (informational only — never attached as an AgentHealthIssue / never blocks admission).
559
+ try {
560
+ const { ensureSleepTimerCheckedAtStartup } = await import("../power/sleep-timer.js");
561
+ ensureSleepTimerCheckedAtStartup();
562
+ }
563
+ catch {
564
+ // Best-effort: health listing must not fail over power checks.
565
+ }
545
566
  const agentDeckOnline = await checkAgentDeckHealth();
546
567
  const mcpRegistration = checkAgentDeckMcpRegistration();
547
568
  const needsDeckAccess = agents.some((a) => a.deckId);
@@ -32,6 +32,7 @@ import { MERGE_FAILURE_EVIDENCE_KEY, MERGE_FAILURE_RESPONSE_OPTIONS } from "./hu
32
32
  import { isMergeConflictFailure, runMergeConflictSync } from "./merge-conflict-sync.js";
33
33
  import { triggerLinearPostMerge } from "./linear-merge-verify.js";
34
34
  import { startBaseAdvancedScan } from "./base-advanced-scan.js";
35
+ import { runHeldWork } from "../power/host-awake-lifecycle.js";
35
36
  import { OPERATOR_VERIFICATION_RESPONSE_OPTIONS, formatOperatorCriteria, getOperatorCriteriaForIssue, hasOperatorVerificationForHead, operatorVerificationRequestId, } from "./operator-criteria.js";
36
37
  const run = promisify(execFile);
37
38
  /** Bound each `gh` shell-out so a hang cannot freeze the coordinator process. */
@@ -173,6 +174,11 @@ export function isAutoMergeInFlight(issueId) {
173
174
  export function clearFinalizeInflightForTests() {
174
175
  finalizeInflight.clear();
175
176
  }
177
+ /** Test seam: replace finalizeAutoMergeOnce inside the host-awake hold (lifecycle tests). */
178
+ let finalizeOnceForTests = null;
179
+ export function setFinalizeAutoMergeOnceForTests(fn) {
180
+ finalizeOnceForTests = fn;
181
+ }
176
182
  /**
177
183
  * Completes an auto-merge parked in `final_review` / system ownership after reviewer
178
184
  * approve. Success → done + reflect; failure → needs_human + policy_escalation.
@@ -183,7 +189,8 @@ export function finalizeAutoMerge(issueId) {
183
189
  const existing = finalizeInflight.get(issueId);
184
190
  if (existing)
185
191
  return existing;
186
- const promise = finalizeAutoMergeOnce(issueId).finally(() => {
192
+ // NOT-369: hold idle-sleep for the merge/publish step; release when it settles.
193
+ const promise = runHeldWork(() => (finalizeOnceForTests ?? finalizeAutoMergeOnce)(issueId)).finally(() => {
187
194
  if (finalizeInflight.get(issueId) === promise) {
188
195
  finalizeInflight.delete(issueId);
189
196
  }
@@ -7,7 +7,7 @@
7
7
  // duplicate delivery is a no-op. The effect *work* itself runs elsewhere, through a leased
8
8
  // worker whose structured result comes back into applyCompletion.
9
9
  import fs from "node:fs";
10
- import { canTransitionIssue, ExecutionContractError, tryCompileContract } from "@agent-dealer/shared";
10
+ import { canTransitionIssue, ExecutionContractError, isCursorKeychainStuckOutput, runtimeAuthClassificationForLog, tryCompileContract, } from "@agent-dealer/shared";
11
11
  import { getDb } from "../db/index.js";
12
12
  import { getIssue, incrementIssueRound, incrementIssueInfraAttempts, incrementIssueCiAttempts, resetIssueInfraAttempts, resetIssueCiAttempts, grantReviewRetry, transitionIssue, updateIssue, } from "../repository/issues.js";
13
13
  import { appendWorkflowEvent, completeWorkflowInstance, getActiveWorkflowInstance, getWorkflowInstance, listWorkflowEventsForIssue, startWorkflowInstance, WorkflowAlreadyActiveError, } from "../repository/workflow-events.js";
@@ -17,7 +17,7 @@ import { ensureIssueRepoCheckout } from "../adapters/managed-repo.js";
17
17
  import { reconcileFinding, resolveFindingsAbsentFromRound } from "../repository/findings.js";
18
18
  import { normalizeReviewerResult } from "./reviewer-result.js";
19
19
  import { getAgent } from "../repository/agents.js";
20
- import { githubIssuesSync, invalidateMuseHealthCache } from "../adapters/agent-health.js";
20
+ import { githubIssuesSync, invalidateMuseHealthCache, runtimeIssuesUncached } from "../adapters/agent-health.js";
21
21
  import { recordMuseCapabilityOverride } from "../adapters/muse-capability.js";
22
22
  import { createIssueArtifact, latestIssueArtifact } from "../repository/artifacts.js";
23
23
  import { listSourceAttachments } from "../repository/source-attachments.js";
@@ -26,11 +26,12 @@ import { completeSession, getActiveWorkerSessionForIssue, getWorkerSession, list
26
26
  import { killRunProcess } from "../runners/spawn-cli.js";
27
27
  import { buildProfileSnapshot, serializeProfileSnapshot } from "./profile-snapshot.js";
28
28
  import { workerSessionPayload } from "./session-progress.js";
29
- import { reasonForWorkerFailedEvent } from "./failure-reason.js";
30
- import { recordCausesForWorkerFailedEvent } from "./failure-cause.js";
29
+ import { reasonForWorkerFailedEvent, readSpawnLogFailureText } from "./failure-reason.js";
30
+ import { classifyAttemptFailure, recordCausesForWorkerFailedEvent } from "./failure-cause.js";
31
31
  import { routeDeveloperOutcome, routeReviewerOutcome, } from "./routing.js";
32
32
  import { projectDeveloperRoute, projectReviewerRoute } from "./projection.js";
33
33
  import { MERGE_CONFLICT_FILES_EVIDENCE_KEY, MERGE_FAILURE_EVIDENCE_KEY, MERGE_FAILURE_RESPONSE_OPTIONS, PUSH_DIVERGENCE_EVIDENCE_KEY, PUSH_DIVERGENCE_RESPONSE_OPTIONS, WORKTREE_BLOCKER_EVIDENCE_KEY, normalizeResolutionNote, parseHumanResolution, resolveHumanActionOutcome, } from "./human-resolution.js";
34
+ import { AUTH_TRANSIENT_RETRY_PAYLOAD_KEY, RUNTIME_AUTH_PARK_EVIDENCE_KEY, authProbeConfirmsFailure, defaultRemediationForRuntime, gateRuntimeAuthParkResume, parseRuntimeAuthParkEvidence, remediationFromProbe, } from "./runtime-auth-park.js";
34
35
  import { AUTO_MERGE_INTENT, finalizeAutoMerge } from "./auto-merge.js";
35
36
  import { triggerLinearPostMerge } from "./linear-merge-verify.js";
36
37
  import { conflictRepairSpent, queueConflictRepairRound } from "./merge-conflict-sync.js";
@@ -482,6 +483,25 @@ export async function applyCompletion(workItemId, leaseToken, outcome) {
482
483
  if (outcome.kind === "base_conflict") {
483
484
  return applyBaseConflictCompletion(workItemId, leaseToken, outcome);
484
485
  }
486
+ // NOT-368: classify + live-probe outside the write txn (network/CLI must never hold
487
+ // the SQLite lock). Developer and reviewer session deaths both participate — medium /
488
+ // unattributed keep today's infra-retry path inside the role's route function.
489
+ let authFailure;
490
+ const authKind = getWorkItem(workItemId)?.kind;
491
+ if ((outcome.kind === "session_failed" || outcome.kind === "timed_out") &&
492
+ (authKind === "developer" || authKind === "reviewer")) {
493
+ try {
494
+ authFailure = await prepareAuthFailureRouting(workItemId, {
495
+ kind: outcome.kind,
496
+ reason: "reason" in outcome ? outcome.reason : undefined,
497
+ logPath: "logPath" in outcome ? outcome.logPath : undefined,
498
+ });
499
+ }
500
+ catch (err) {
501
+ console.error("[coordinator] prepareAuthFailureRouting", workItemId, err);
502
+ authFailure = undefined;
503
+ }
504
+ }
485
505
  const routed = getDb().transaction(() => {
486
506
  const before = getWorkItem(workItemId);
487
507
  if (!before)
@@ -499,7 +519,7 @@ export async function applyCompletion(workItemId, leaseToken, outcome) {
499
519
  const item = finishWorkItem(workItemId, leaseToken, { status: "done", result: outcome });
500
520
  if (!item)
501
521
  return { applied: false, reason: "lease_lost" };
502
- return routeAppliedOutcome(issue, instance, item, outcome);
522
+ return routeAppliedOutcome(issue, instance, item, outcome, authFailure);
503
523
  })();
504
524
  if (routed.applied && routed.pendingAutoMerge) {
505
525
  // Await async `gh` merge outside the routing txn — never block the event loop with spawnSync.
@@ -507,6 +527,74 @@ export async function applyCompletion(workItemId, leaseToken, outcome) {
507
527
  }
508
528
  return routed;
509
529
  }
530
+ /**
531
+ * NOT-368: when a developer or reviewer session_failed / timed_out with a high-confidence
532
+ * attributed auth cause, run the runtime's live auth probe and gather consecutive-park /
533
+ * transient-retry state for the role's route function.
534
+ */
535
+ async function prepareAuthFailureRouting(workItemId, outcome) {
536
+ const item = getWorkItem(workItemId);
537
+ if (!item || (item.kind !== "developer" && item.kind !== "reviewer"))
538
+ return undefined;
539
+ const session = item.workerSessionId ? getWorkerSession(item.workerSessionId) : null;
540
+ const logPath = session?.logPath ?? outcome.logPath ?? null;
541
+ const runtimeHint = (session?.runtime ?? null);
542
+ const causes = classifyAttemptFailure({
543
+ outcomeKind: outcome.kind,
544
+ outcomeReason: outcome.reason ?? null,
545
+ logPath,
546
+ runtime: runtimeHint,
547
+ });
548
+ const primary = causes.find((c) => c.primary) ?? causes[0];
549
+ if (!primary || primary.code !== "authentication_configuration")
550
+ return undefined;
551
+ // Medium / unattributed ("Not logged in" with no vendor) keep the generic infra retry.
552
+ if (primary.confidence !== "high")
553
+ return undefined;
554
+ const haystack = `${readSpawnLogFailureText(logPath)}\n${outcome.reason ?? ""}`;
555
+ const classified = isCursorKeychainStuckOutput(haystack)
556
+ ? { runtime: "cursor_local", issue: { code: "cursor_keychain", message: defaultRemediationForRuntime("cursor_local", true) } }
557
+ : runtimeAuthClassificationForLog(haystack, runtimeHint);
558
+ const runtime = classified?.runtime ?? runtimeHint;
559
+ if (!runtime)
560
+ return undefined;
561
+ const remediation = classified?.issue.message?.trim() ||
562
+ defaultRemediationForRuntime(runtime, classified?.issue.code === "cursor_keychain");
563
+ const probeIssues = await runtimeIssuesUncached(runtime);
564
+ const probeStillFailing = authProbeConfirmsFailure(probeIssues);
565
+ const probeRemediation = remediationFromProbe(probeIssues, remediation);
566
+ return {
567
+ confidence: "high",
568
+ runtime,
569
+ remediation: probeStillFailing ? probeRemediation : remediation,
570
+ rawCause: primary.rawReason,
571
+ probeStillFailing,
572
+ consecutiveAuthParks: countConsecutiveAuthParks(item.issueId),
573
+ authTransientRetrySpent: issueSpentAuthTransientRetry(item.issueId),
574
+ };
575
+ }
576
+ /** Count trailing auth-park human actions on the issue (newest streak). */
577
+ function countConsecutiveAuthParks(issueId) {
578
+ const actions = listHumanActionsForIssue(issueId);
579
+ let n = 0;
580
+ for (let i = actions.length - 1; i >= 0; i--) {
581
+ if (parseRuntimeAuthParkEvidence(actions[i].evidenceJson) == null)
582
+ break;
583
+ n++;
584
+ }
585
+ return n;
586
+ }
587
+ /** True when any prior developer/reviewer work item on this issue spent the transient auth retry. */
588
+ function issueSpentAuthTransientRetry(issueId) {
589
+ for (const w of listWorkItemsForIssue(issueId)) {
590
+ if (w.kind !== "developer" && w.kind !== "reviewer")
591
+ continue;
592
+ const payload = parseWorkItemPayload(w.payloadJson);
593
+ if (payload[AUTH_TRANSIENT_RETRY_PAYLOAD_KEY] === true)
594
+ return true;
595
+ }
596
+ return false;
597
+ }
510
598
  function parseWorkItemPayload(json) {
511
599
  if (!json)
512
600
  return {};
@@ -906,10 +994,10 @@ export function routeCapEscalation(issue, instance, item, cap) {
906
994
  instanceCompleted: false,
907
995
  };
908
996
  }
909
- export function routeAppliedOutcome(issue, instance, item, outcome) {
997
+ export function routeAppliedOutcome(issue, instance, item, outcome, authFailure) {
910
998
  return item.kind === "developer"
911
- ? applyDeveloper(issue, instance, item, outcome)
912
- : applyReviewer(issue, instance, item, outcome);
999
+ ? applyDeveloper(issue, instance, item, outcome, authFailure)
1000
+ : applyReviewer(issue, instance, item, outcome, authFailure);
913
1001
  }
914
1002
  function eventEmitter(issue, instance, workerSessionId, stage, round) {
915
1003
  let causation = null;
@@ -955,7 +1043,7 @@ function applyProjectionTransition(issue, projection, patch) {
955
1043
  ...patch,
956
1044
  });
957
1045
  }
958
- function applyDeveloper(issue, instance, item, outcome) {
1046
+ function applyDeveloper(issue, instance, item, outcome, authFailure) {
959
1047
  const route = routeDeveloperOutcome(outcome, {
960
1048
  currentRound: issue.currentRound,
961
1049
  maxReviewRounds: issue.maxReviewRounds,
@@ -963,6 +1051,7 @@ function applyDeveloper(issue, instance, item, outcome) {
963
1051
  maxInfraAttempts: issue.maxInfraAttempts,
964
1052
  ciAttempts: issue.ciAttempts,
965
1053
  maxCiAttempts: issue.maxCiAttempts,
1054
+ ...(authFailure ? { authFailure } : {}),
966
1055
  });
967
1056
  const { projection, effect, advance } = projectDeveloperRoute(route, issue.status, issue.currentRound);
968
1057
  const ev = eventEmitter(issue, instance, item.workerSessionId, projection.issueStatus, issue.currentRound);
@@ -1051,13 +1140,14 @@ function applyDeveloper(issue, instance, item, outcome) {
1051
1140
  const issueNow = getIssue(issue.id);
1052
1141
  return applyEffect(issue, instance, effect, route, issueNow, ev, item.id);
1053
1142
  }
1054
- function applyReviewer(issue, instance, item, outcome) {
1143
+ function applyReviewer(issue, instance, item, outcome, authFailure) {
1055
1144
  let route = routeReviewerOutcome(outcome, {
1056
1145
  currentRound: issue.currentRound,
1057
1146
  maxReviewRounds: issue.maxReviewRounds,
1058
1147
  infraAttempts: issue.infraAttempts,
1059
1148
  maxInfraAttempts: issue.maxInfraAttempts,
1060
1149
  autoMerge: issue.autoMerge,
1150
+ ...(authFailure ? { authFailure } : {}),
1061
1151
  }, issue.headSha);
1062
1152
  // NOT-184: a repair loop that keeps finding new blocking issues on the same file is not
1063
1153
  // converging — hand it to a human instead of queuing another developer round.
@@ -1240,6 +1330,8 @@ function applyEffect(issue, instance, effect, route, issueNow, ev, causativeItem
1240
1330
  ...(effect.retryReason ? { retryReason: effect.retryReason } : {}),
1241
1331
  ...(effect.publishOnly ? { publishOnly: true } : {}),
1242
1332
  ...(effect.branch ? { branch: effect.branch } : {}),
1333
+ // NOT-368: persist that this infra retry spent the one transient auth retry.
1334
+ ...(effect.authTransientRetry ? { [AUTH_TRANSIENT_RETRY_PAYLOAD_KEY]: true } : {}),
1243
1335
  profileSnapshot: queuedProfileSnapshot(issue, kind),
1244
1336
  },
1245
1337
  idempotencyKey,
@@ -1261,6 +1353,10 @@ function applyEffect(issue, instance, effect, route, issueNow, ev, causativeItem
1261
1353
  const pushDivergence = actionType === "policy_escalation" && !resumeAsReviewer
1262
1354
  ? (effect.pushDivergence ?? null)
1263
1355
  : null;
1356
+ // NOT-368: confirmed runtime login park — resolve re-probes before re-queueing.
1357
+ // Reviewer-origin parks keep resumeAsReviewer (re-queue reviewer) AND auth-park
1358
+ // evidence (so resolve still re-probes); the two are not mutually exclusive.
1359
+ const runtimeAuthPark = actionType === "policy_escalation" ? (effect.runtimeAuthPark ?? null) : null;
1264
1360
  // NOT-280: an unchanged worktree blocker (same fingerprint) lands on the action it
1265
1361
  // already raised — still open, or reopened when a Resume changed nothing — instead of
1266
1362
  // opening an identical one on every Resume.
@@ -1284,9 +1380,11 @@ function applyEffect(issue, instance, effect, route, issueNow, ev, causativeItem
1284
1380
  ? { review: reviewerOutcome.result, ...(nonConvergence ? { nonConvergence } : {}) }
1285
1381
  : pushDivergence
1286
1382
  ? { [PUSH_DIVERGENCE_EVIDENCE_KEY]: pushDivergence }
1287
- : blockerFingerprint
1288
- ? { [WORKTREE_BLOCKER_EVIDENCE_KEY]: { fingerprint: blockerFingerprint } }
1289
- : undefined,
1383
+ : runtimeAuthPark
1384
+ ? { [RUNTIME_AUTH_PARK_EVIDENCE_KEY]: runtimeAuthPark }
1385
+ : blockerFingerprint
1386
+ ? { [WORKTREE_BLOCKER_EVIDENCE_KEY]: { fingerprint: blockerFingerprint } }
1387
+ : undefined,
1290
1388
  // issueNow.headSha, not issue.headSha: a stale outcome that itself exhausted the
1291
1389
  // infra budget already patched the newly observed head onto the issue above — the
1292
1390
  // pre-transition issue param would still carry the stale SHA a "resume" must not reuse.
@@ -2051,6 +2149,48 @@ export async function resolveHumanActionAndAdvanceAsync(actionId, resolvedBy, ch
2051
2149
  resumeLiveHeadSha = null;
2052
2150
  }
2053
2151
  }
2152
+ // NOT-368: re-probe before resolving a runtime-auth park. A still-failing probe keeps
2153
+ // the action open (no spawn); only a passing probe proceeds to resume on the salvaged
2154
+ // branch without charging a round or infra attempt.
2155
+ if (choice === "resume") {
2156
+ const authParkAction = getHumanAction(actionId);
2157
+ const authParkEvidence = authParkAction
2158
+ ? parseRuntimeAuthParkEvidence(authParkAction.evidenceJson)
2159
+ : null;
2160
+ if (authParkAction?.status === "open" && authParkEvidence) {
2161
+ let probeIssues = [];
2162
+ try {
2163
+ probeIssues = await runtimeIssuesUncached(authParkEvidence.runtime);
2164
+ }
2165
+ catch (err) {
2166
+ console.error("[coordinator] auth-park resolve probe", actionId, err);
2167
+ probeIssues = [
2168
+ {
2169
+ code: "runtime_auth",
2170
+ message: authParkEvidence.remediation,
2171
+ },
2172
+ ];
2173
+ }
2174
+ const gate = gateRuntimeAuthParkResume({
2175
+ evidence: authParkEvidence,
2176
+ probeStillFailing: authProbeConfirmsFailure(probeIssues),
2177
+ probeRemediation: remediationFromProbe(probeIssues, authParkEvidence.remediation),
2178
+ });
2179
+ if (!gate.proceed) {
2180
+ // Refresh the open action's reason so the operator sees the remediation again.
2181
+ try {
2182
+ updateOpenHumanAction(actionId, {
2183
+ reason: gate.message,
2184
+ question: questionFor("policy_escalation", gate.message, false),
2185
+ });
2186
+ }
2187
+ catch {
2188
+ // Best-effort — the 409 below is what keeps the action open.
2189
+ }
2190
+ return { ok: false, code: 409, error: gate.message };
2191
+ }
2192
+ }
2193
+ }
2054
2194
  // NOT-196: the `gh` PR-state read runs here, outside any DB transaction — the sync
2055
2195
  // core below only consumes the pre-read state. Only close choices on issues with a
2056
2196
  // PR number pay for the call; anything unreadable resolves to "unknown" (or
@@ -1,3 +1,7 @@
1
+ // NOT-368: auth-park evidence + resolve-time re-probe gate live in runtime-auth-park.ts;
2
+ // re-exported here so the park/resume contract stays discoverable next to NOT-93's
3
+ // deck_interaction_required resume (same roundKind: infra, no review-round spend).
4
+ export { AUTH_TRANSIENT_RETRY_PAYLOAD_KEY, MAX_CONSECUTIVE_AUTH_PARKS, RUNTIME_AUTH_PARK_EVIDENCE_KEY, authProbeConfirmsFailure, gateRuntimeAuthParkResume, isRuntimeAuthParkAction, parseRuntimeAuthParkEvidence, remediationFromProbe, } from "./runtime-auth-park.js";
1
5
  /**
2
6
  * NOT-194: the stored response options for a merge-failure policy_escalation. Shared by
3
7
  * auto-merge.ts (which raises it) and commands.ts's responseOptionsFor merge-failure
@@ -1,6 +1,65 @@
1
- import { test } from "node:test";
1
+ import { test, before, beforeEach } from "node:test";
2
2
  import assert from "node:assert/strict";
3
- import { resolveHumanActionOutcome, parseHumanResolution } from "./human-resolution.js";
3
+ import fs from "node:fs";
4
+ import os from "node:os";
5
+ import path from "node:path";
6
+ import { fileURLToPath } from "node:url";
7
+ import { resolveHumanActionOutcome, parseHumanResolution, gateRuntimeAuthParkResume, authProbeConfirmsFailure, remediationFromProbe, RUNTIME_AUTH_PARK_EVIDENCE_KEY, parseRuntimeAuthParkEvidence, } from "./human-resolution.js";
8
+ import { BUILTIN_AGENT_CLAUDE_ID, BUILTIN_AGENT_CURSOR_ID, CURSOR_AUTH_REMEDIATION, CURSOR_KEYCHAIN_REMEDIATION, MUSE_AUTH_REMEDIATION, } from "@agent-dealer/shared";
9
+ process.env.AGENT_DEALER_HOME = fs.mkdtempSync(path.join(os.tmpdir(), "dealer-human-res-"));
10
+ const { migrate, getDb } = await import("../db/index.js");
11
+ const { createIssue, getIssue, transitionIssue } = await import("../repository/issues.js");
12
+ const { listHumanActionsForIssue, getHumanAction } = await import("../repository/human-actions.js");
13
+ const { claimWorkItem, listWorkItemsForIssue, getWorkItem } = await import("../repository/work-items.js");
14
+ const { setRuntimeIssuesUncachedForTests } = await import("../adapters/agent-health.js");
15
+ const { stubManagedCloneForTests } = await import("../adapters/managed-repo.js");
16
+ const { startWorkflow, applyCompletion, resolveHumanActionAndAdvanceAsync, } = await import("./commands.js");
17
+ const FIXTURE_DIR = path.join(path.dirname(fileURLToPath(import.meta.url)), "../../../shared/src/fixtures/runtime-auth");
18
+ const MUSE_AUTH_LOG = fs.readFileSync(path.join(FIXTURE_DIR, "muse-exec-missing-credentials.txt"), "utf8");
19
+ /** NOT-114 keychain has no capture file — same reconstruction as runtime-auth-health / routing tests. */
20
+ const CURSOR_KEYCHAIN_STDERR = `Cursor couldn't save your login to the macOS keychain (errSecDuplicateItem, security exit code 45).
21
+ The keychain item is stuck. Delete it and sign in again:
22
+ security delete-generic-password -s cursor-access-token -a cursor-user
23
+ agent login
24
+ `;
25
+ before(() => migrate());
26
+ beforeEach(() => {
27
+ getDb().exec("DELETE FROM work_items");
28
+ setRuntimeIssuesUncachedForTests(null);
29
+ stubManagedCloneForTests("acme/app");
30
+ });
31
+ function newIssue(opts = {}) {
32
+ return createIssue({
33
+ title: "Auth park me",
34
+ description: "d",
35
+ acceptanceCriteria: "It works",
36
+ repo: "acme/app",
37
+ developerAgentId: BUILTIN_AGENT_CLAUDE_ID,
38
+ reviewerAgentId: BUILTIN_AGENT_CURSOR_ID,
39
+ baseBranch: "main",
40
+ maxReviewRounds: 3,
41
+ maxInfraAttempts: opts.maxInfraAttempts ?? 3,
42
+ source: "manual",
43
+ }).id;
44
+ }
45
+ function claim(issueId) {
46
+ const item = claimWorkItem(`test-${issueId}`, { leaseMs: 60_000 });
47
+ assert.ok(item && item.issueId === issueId, "expected to lease this issue's work item");
48
+ return item;
49
+ }
50
+ async function complete(issueId, outcome) {
51
+ const item = claim(issueId);
52
+ return { item, result: await applyCompletion(item.id, item.leaseToken, outcome) };
53
+ }
54
+ function openAuthPark(issueId) {
55
+ return listHumanActionsForIssue(issueId).find((a) => a.status === "open" && parseRuntimeAuthParkEvidence(a.evidenceJson) != null);
56
+ }
57
+ function stubProbeFailing(remediation = MUSE_AUTH_REMEDIATION) {
58
+ setRuntimeIssuesUncachedForTests(async () => [{ code: "runtime_auth", message: remediation }]);
59
+ }
60
+ function stubProbePassing() {
61
+ setRuntimeIssuesUncachedForTests(async () => []);
62
+ }
4
63
  // Reviewer finding #8: an unrecognized choice must be rejected, not silently treated as "close".
5
64
  test("parseHumanResolution rejects a choice not in that action type's allowed set", () => {
6
65
  assert.equal(parseHumanResolution("final_review", "bogus"), null);
@@ -115,3 +174,234 @@ test("operator_verification repair queues another repair round; verified/waive n
115
174
  assert.equal(result.roundKind, "review");
116
175
  assert.throws(() => resolveHumanActionOutcome({ actionType: "operator_verification", choice: "verified", note: "ok" }), /Unrecognized operator_verification choice/);
117
176
  });
177
+ // --- NOT-368: resolve-time auth-park re-probe ---
178
+ const AUTH_PARK_EVIDENCE = {
179
+ runtime: "muse_code",
180
+ remediation: MUSE_AUTH_REMEDIATION,
181
+ rawCause: "missing meta credentials: run muse login or set META_API_KEY",
182
+ consecutivePark: 1,
183
+ };
184
+ test("NOT-368: authProbeConfirmsFailure is true for runtime_auth and cursor_keychain only", () => {
185
+ assert.equal(authProbeConfirmsFailure([{ code: "runtime_auth", message: MUSE_AUTH_REMEDIATION }]), true);
186
+ assert.equal(authProbeConfirmsFailure([{ code: "cursor_keychain", message: "keychain stuck" }]), true);
187
+ assert.equal(authProbeConfirmsFailure([{ code: "cli_missing", message: "install muse" }]), false);
188
+ assert.equal(authProbeConfirmsFailure([]), false);
189
+ });
190
+ test("NOT-368: remediationFromProbe keeps keychain text when probe only reports runtime_auth", () => {
191
+ assert.equal(remediationFromProbe([{ code: "runtime_auth", message: CURSOR_AUTH_REMEDIATION }], CURSOR_KEYCHAIN_REMEDIATION), CURSOR_KEYCHAIN_REMEDIATION);
192
+ assert.equal(remediationFromProbe([{ code: "cursor_keychain", message: CURSOR_KEYCHAIN_REMEDIATION }], "stale fallback"), CURSOR_KEYCHAIN_REMEDIATION);
193
+ assert.equal(remediationFromProbe([{ code: "runtime_auth", message: "probe says login" }], MUSE_AUTH_REMEDIATION), "probe says login");
194
+ assert.equal(remediationFromProbe([], CURSOR_KEYCHAIN_REMEDIATION), CURSOR_KEYCHAIN_REMEDIATION);
195
+ });
196
+ test("NOT-368: gate helper — still-failing probe refuses resume", () => {
197
+ const gate = gateRuntimeAuthParkResume({
198
+ evidence: AUTH_PARK_EVIDENCE,
199
+ probeStillFailing: true,
200
+ probeRemediation: MUSE_AUTH_REMEDIATION,
201
+ });
202
+ assert.equal(gate.proceed, false);
203
+ if (!gate.proceed) {
204
+ assert.match(gate.message, /still not authenticated/i);
205
+ assert.match(gate.remediation, /muse login|META_API_KEY/i);
206
+ }
207
+ });
208
+ test("NOT-368: gate helper — passing probe allows resume (infra roundKind, no charge)", () => {
209
+ const gate = gateRuntimeAuthParkResume({
210
+ evidence: AUTH_PARK_EVIDENCE,
211
+ probeStillFailing: false,
212
+ });
213
+ assert.deepStrictEqual(gate, { proceed: true });
214
+ const resume = resolveHumanActionOutcome({ actionType: "policy_escalation", choice: "resume" });
215
+ assert.equal(resume.issueStatus, "developing");
216
+ assert.equal(resume.startNewRound, true);
217
+ assert.equal(resume.roundKind, "infra");
218
+ assert.equal(RUNTIME_AUTH_PARK_EVIDENCE_KEY, "runtimeAuthPark");
219
+ });
220
+ // --- NOT-368: coordinator-level apply + resolve (probe stubbed) ---
221
+ test("NOT-368: confirmed auth park persists remediation, does not charge infraAttempts", async () => {
222
+ stubProbeFailing();
223
+ const issueId = newIssue();
224
+ startWorkflow(issueId);
225
+ transitionIssue(issueId, "developing", { branch: "issue-auth-salvage" });
226
+ const infraBefore = getIssue(issueId).infraAttempts;
227
+ const { result } = await complete(issueId, {
228
+ kind: "session_failed",
229
+ reason: MUSE_AUTH_LOG.trim().slice(0, 300),
230
+ });
231
+ assert.equal(result.applied, true);
232
+ assert.equal(getIssue(issueId).status, "needs_human");
233
+ assert.equal(getIssue(issueId).infraAttempts, infraBefore, "park must not charge infra");
234
+ const action = openAuthPark(issueId);
235
+ assert.ok(action, "expected open auth-park human action");
236
+ assert.match(action.reason, /muse login|META_API_KEY/i);
237
+ const evidence = parseRuntimeAuthParkEvidence(action.evidenceJson);
238
+ assert.ok(evidence);
239
+ assert.equal(evidence.runtime, "muse_code");
240
+ assert.match(evidence.remediation, /muse login|META_API_KEY/i);
241
+ assert.equal(listWorkItemsForIssue(issueId).filter((w) => w.status === "pending").length, 0);
242
+ });
243
+ test("NOT-368: keychain classification + runtime_auth probe keeps keychain remediation (park + resolve)", async () => {
244
+ // Live status often confirms failure as ordinary runtime_auth; that must not replace
245
+ // the classifier's keychain deletion instructions (park or Resume rewrite).
246
+ setRuntimeIssuesUncachedForTests(async () => [
247
+ { code: "runtime_auth", message: CURSOR_AUTH_REMEDIATION },
248
+ ]);
249
+ const issueId = newIssue();
250
+ startWorkflow(issueId);
251
+ const infraBefore = getIssue(issueId).infraAttempts;
252
+ await complete(issueId, {
253
+ kind: "session_failed",
254
+ reason: CURSOR_KEYCHAIN_STDERR.trim().slice(0, 400),
255
+ });
256
+ assert.equal(getIssue(issueId).status, "needs_human");
257
+ assert.equal(getIssue(issueId).infraAttempts, infraBefore);
258
+ const action = openAuthPark(issueId);
259
+ assert.ok(action, "expected open auth-park human action");
260
+ assert.match(action.reason, /delete-generic-password|errSecDuplicateItem/i);
261
+ assert.doesNotMatch(action.reason, /CURSOR_API_KEY/);
262
+ const evidence = parseRuntimeAuthParkEvidence(action.evidenceJson);
263
+ assert.ok(evidence);
264
+ assert.equal(evidence.runtime, "cursor_local");
265
+ assert.equal(evidence.remediation, CURSOR_KEYCHAIN_REMEDIATION);
266
+ const resolved = await resolveHumanActionAndAdvanceAsync(action.id, "op", "resume");
267
+ assert.equal(resolved.ok, false);
268
+ assert.equal(resolved.code, 409);
269
+ assert.match(resolved.error, /delete-generic-password|errSecDuplicateItem/i);
270
+ assert.doesNotMatch(resolved.error, /CURSOR_API_KEY/);
271
+ assert.match(getHumanAction(action.id).reason, /delete-generic-password|errSecDuplicateItem/i);
272
+ assert.equal(getHumanAction(action.id).status, "open");
273
+ });
274
+ test("NOT-368: resolve with still-failing probe keeps action open and spawns nothing", async () => {
275
+ stubProbeFailing();
276
+ const issueId = newIssue();
277
+ startWorkflow(issueId);
278
+ await complete(issueId, {
279
+ kind: "session_failed",
280
+ reason: MUSE_AUTH_LOG.trim().slice(0, 300),
281
+ });
282
+ const action = openAuthPark(issueId);
283
+ const pendingBefore = listWorkItemsForIssue(issueId).filter((w) => w.status === "pending").length;
284
+ const infraBefore = getIssue(issueId).infraAttempts;
285
+ const resolved = await resolveHumanActionAndAdvanceAsync(action.id, "op", "resume");
286
+ assert.equal(resolved.ok, false);
287
+ assert.equal(resolved.code, 409);
288
+ assert.match(resolved.error, /still not authenticated/i);
289
+ assert.equal(getHumanAction(action.id).status, "open");
290
+ assert.equal(getIssue(issueId).status, "needs_human");
291
+ assert.equal(getIssue(issueId).infraAttempts, infraBefore);
292
+ assert.equal(listWorkItemsForIssue(issueId).filter((w) => w.status === "pending").length, pendingBefore);
293
+ });
294
+ test("NOT-368: resolve with passing probe re-queues developer on salvaged branch with no infra charge", async () => {
295
+ stubProbeFailing();
296
+ const issueId = newIssue();
297
+ startWorkflow(issueId);
298
+ const salvageBranch = "issue-auth-salvage-resume";
299
+ transitionIssue(issueId, "developing", { branch: salvageBranch });
300
+ await complete(issueId, {
301
+ kind: "session_failed",
302
+ reason: MUSE_AUTH_LOG.trim().slice(0, 300),
303
+ });
304
+ const action = openAuthPark(issueId);
305
+ const roundBefore = getIssue(issueId).currentRound;
306
+ const infraAtPark = getIssue(issueId).infraAttempts;
307
+ stubProbePassing();
308
+ const resolved = await resolveHumanActionAndAdvanceAsync(action.id, "op", "resume");
309
+ assert.equal(resolved.ok, true);
310
+ assert.equal(getHumanAction(action.id).status, "resolved");
311
+ assert.equal(getIssue(issueId).status, "developing");
312
+ assert.equal(getIssue(issueId).branch, salvageBranch);
313
+ assert.equal(getIssue(issueId).currentRound, roundBefore, "resume must not spend a review round");
314
+ // Infra resume resets the counter (NOT-93) — it must not land above the park-time value
315
+ // via a charge, and typically returns to 0.
316
+ assert.ok(getIssue(issueId).infraAttempts <= infraAtPark);
317
+ const pending = listWorkItemsForIssue(issueId).filter((w) => w.status === "pending" && w.kind === "developer");
318
+ assert.equal(pending.length, 1);
319
+ assert.equal(getWorkItem(pending[0].id).kind, "developer");
320
+ });
321
+ test("NOT-368: probe-passing auth failure charges infra by exactly 1; second parks even when probe passes", async () => {
322
+ stubProbePassing();
323
+ const issueId = newIssue();
324
+ startWorkflow(issueId);
325
+ assert.equal(getIssue(issueId).infraAttempts, 0);
326
+ await complete(issueId, {
327
+ kind: "session_failed",
328
+ reason: MUSE_AUTH_LOG.trim().slice(0, 300),
329
+ });
330
+ assert.equal(getIssue(issueId).infraAttempts, 1, "transient auth retry charges exactly +1");
331
+ assert.equal(getIssue(issueId).status, "developing");
332
+ const retry = listWorkItemsForIssue(issueId).find((w) => w.status === "pending" && w.kind === "developer");
333
+ assert.ok(retry);
334
+ const payload = JSON.parse(retry.payloadJson);
335
+ assert.equal(payload.authTransientRetry, true);
336
+ // Second high-confidence auth failure parks even though the probe still passes.
337
+ await complete(issueId, {
338
+ kind: "session_failed",
339
+ reason: MUSE_AUTH_LOG.trim().slice(0, 300),
340
+ });
341
+ assert.equal(getIssue(issueId).status, "needs_human");
342
+ assert.equal(getIssue(issueId).infraAttempts, 1, "park after transient retry must not charge again");
343
+ assert.ok(openAuthPark(issueId));
344
+ });
345
+ test("NOT-368: four consecutive confirmed auth parks escalate on the 4th (persisted streak)", async () => {
346
+ const issueId = newIssue({ maxInfraAttempts: 5 });
347
+ startWorkflow(issueId);
348
+ for (let park = 1; park <= 3; park++) {
349
+ stubProbeFailing();
350
+ await complete(issueId, {
351
+ kind: "session_failed",
352
+ reason: MUSE_AUTH_LOG.trim().slice(0, 300),
353
+ });
354
+ const action = openAuthPark(issueId);
355
+ assert.ok(action, `expected auth park #${park}`);
356
+ const evidence = parseRuntimeAuthParkEvidence(action.evidenceJson);
357
+ assert.equal(evidence.consecutivePark, park);
358
+ assert.equal(getIssue(issueId).infraAttempts, 0);
359
+ stubProbePassing();
360
+ const resolved = await resolveHumanActionAndAdvanceAsync(action.id, "op", "resume");
361
+ assert.equal(resolved.ok, true, `resume after park #${park}`);
362
+ }
363
+ stubProbeFailing();
364
+ await complete(issueId, {
365
+ kind: "session_failed",
366
+ reason: MUSE_AUTH_LOG.trim().slice(0, 300),
367
+ });
368
+ assert.equal(getIssue(issueId).status, "needs_human");
369
+ assert.equal(getIssue(issueId).infraAttempts, 0);
370
+ const fourth = listHumanActionsForIssue(issueId).find((a) => a.status === "open");
371
+ assert.ok(fourth);
372
+ assert.match(fourth.reason, /Repeated .* login failures/i);
373
+ assert.equal(parseRuntimeAuthParkEvidence(fourth.evidenceJson), null, "4th outcome is a distinct escalation, not another auth park");
374
+ });
375
+ test("NOT-368: reviewer session_failed with confirmed auth parks (resumeAsReviewer + evidence)", async () => {
376
+ stubProbeFailing();
377
+ const issueId = newIssue();
378
+ startWorkflow(issueId);
379
+ await complete(issueId, {
380
+ kind: "clean_handoff",
381
+ branch: "issue-reviewer-auth",
382
+ headSha: "abc123",
383
+ baseSha: "base1",
384
+ prNumber: 42,
385
+ prUrl: "https://gh/pr/42",
386
+ });
387
+ assert.equal(getIssue(issueId).status, "reviewing");
388
+ const infraBefore = getIssue(issueId).infraAttempts;
389
+ await complete(issueId, {
390
+ kind: "session_failed",
391
+ reason: MUSE_AUTH_LOG.trim().slice(0, 300),
392
+ });
393
+ assert.equal(getIssue(issueId).status, "needs_human");
394
+ assert.equal(getIssue(issueId).infraAttempts, infraBefore);
395
+ const action = openAuthPark(issueId);
396
+ assert.ok(action, "reviewer auth failure must park with runtimeAuthPark evidence");
397
+ assert.match(action.reason, /muse login|META_API_KEY/i);
398
+ const preview = JSON.parse(action.continuationPreviewJson);
399
+ assert.equal(preview.resumeRole, "reviewer");
400
+ assert.equal(preview.resumeHeadSha, "abc123");
401
+ stubProbePassing();
402
+ const resolved = await resolveHumanActionAndAdvanceAsync(action.id, "op", "resume");
403
+ assert.equal(resolved.ok, true);
404
+ assert.equal(getIssue(issueId).status, "reviewing");
405
+ const pending = listWorkItemsForIssue(issueId).filter((w) => w.status === "pending" && w.kind === "reviewer");
406
+ assert.equal(pending.length, 1);
407
+ });