agent-dealer 1.2.14 → 1.2.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/bundle/server/dist/adapters/agent-health.js +21 -0
  2. package/bundle/server/dist/capacity/claude-events.js +25 -5
  3. package/bundle/server/dist/capacity/claude-local-cache.js +159 -13
  4. package/bundle/server/dist/capacity/claude-local-cache.test.js +191 -3
  5. package/bundle/server/dist/coordinator/auto-merge.js +8 -1
  6. package/bundle/server/dist/coordinator/commands.js +153 -13
  7. package/bundle/server/dist/coordinator/delete-issue.js +473 -0
  8. package/bundle/server/dist/coordinator/human-resolution.js +4 -0
  9. package/bundle/server/dist/coordinator/human-resolution.test.js +292 -2
  10. package/bundle/server/dist/coordinator/projection.js +26 -3
  11. package/bundle/server/dist/coordinator/projection.test.js +8 -1
  12. package/bundle/server/dist/coordinator/routing.js +96 -0
  13. package/bundle/server/dist/coordinator/routing.test.js +163 -0
  14. package/bundle/server/dist/coordinator/runtime-auth-park.js +97 -0
  15. package/bundle/server/dist/coordinator/source-attachments.js +24 -1
  16. package/bundle/server/dist/coordinator/source-attachments.test.js +112 -0
  17. package/bundle/server/dist/coordinator/worker-loop.js +15 -0
  18. package/bundle/server/dist/db/index.js +4 -0
  19. package/bundle/server/dist/db/schema.sql +4 -0
  20. package/bundle/server/dist/docs-host-awake.test.js +16 -0
  21. package/bundle/server/dist/index.js +14 -1
  22. package/bundle/server/dist/power/host-awake-lifecycle.js +30 -0
  23. package/bundle/server/dist/power/host-awake-lifecycle.test.js +181 -0
  24. package/bundle/server/dist/power/host-awake.js +162 -0
  25. package/bundle/server/dist/power/host-awake.test.js +192 -0
  26. package/bundle/server/dist/power/sleep-timer.js +86 -0
  27. package/bundle/server/dist/power/sleep-timer.test.js +79 -0
  28. package/bundle/server/dist/power/status.js +31 -0
  29. package/bundle/server/dist/power/status.test.js +60 -0
  30. package/bundle/server/dist/repository/runtime-capacity.js +19 -3
  31. package/bundle/server/dist/routes/index.js +3 -0
  32. package/bundle/server/dist/routes/issues-delete.test.js +666 -0
  33. package/bundle/server/dist/routes/issues-source-attachments.test.js +36 -0
  34. package/bundle/server/dist/routes/issues.js +20 -0
  35. package/bundle/server/dist/routes/runtime-capacity.test.js +35 -0
  36. package/bundle/server/package.json +4 -3
  37. package/bundle/server/static-ui/assets/index-DL8mEXzN.css +1 -0
  38. package/bundle/server/static-ui/assets/index-FCHzrnLr.js +64 -0
  39. package/bundle/server/static-ui/index.html +2 -2
  40. package/bundle/shared/dist/runtime-capacity.d.ts +147 -0
  41. package/bundle/shared/dist/runtime-capacity.js +15 -0
  42. package/bundle/shared/package.json +1 -1
  43. package/dist/bin.js +3 -0
  44. package/dist/node-hardening.d.ts +9 -0
  45. package/dist/node-hardening.js +38 -0
  46. package/dist/ports.js +54 -14
  47. package/dist/stop-einval.test.d.ts +1 -0
  48. package/dist/stop-einval.test.js +215 -0
  49. package/dist/stop.d.ts +6 -1
  50. package/dist/stop.js +18 -7
  51. package/dist/version-output.test.js +23 -16
  52. package/package.json +1 -1
  53. package/bundle/server/static-ui/assets/index-COYJxD1y.css +0 -1
  54. package/bundle/server/static-ui/assets/index-DDbJ9ZU7.js +0 -64
@@ -104,6 +104,14 @@ let runCommandImpl = defaultRunCommand;
104
104
  export function setRunCommandForTests(fn) {
105
105
  runCommandImpl = fn ?? defaultRunCommand;
106
106
  }
107
+ let runtimeIssuesUncachedForTests = null;
108
+ /**
109
+ * Replace the live auth probe in coordinator tests. Pass `null` to restore the real probe.
110
+ * Cleared by {@link clearAgentHealthCaches}.
111
+ */
112
+ export function setRuntimeIssuesUncachedForTests(fn) {
113
+ runtimeIssuesUncachedForTests = fn;
114
+ }
107
115
  function runCommand(cmd, args, timeoutMs = DEFAULT_PROBE_TIMEOUT_MS, env) {
108
116
  return runCommandImpl(cmd, args, timeoutMs, env);
109
117
  }
@@ -122,6 +130,7 @@ export function clearAgentHealthCaches() {
122
130
  githubIssueCache = null;
123
131
  cursorSoftFailStreak = 0;
124
132
  cursorLastHealthyAt = null;
133
+ runtimeIssuesUncachedForTests = null;
125
134
  resetMuseCapabilityStateForTests();
126
135
  }
127
136
  function isSoftCursorProbeIssue(issue) {
@@ -256,6 +265,9 @@ async function museRuntimeIssues() {
256
265
  }
257
266
  /** Exported for direct testing — bypasses the 60s cache in runtimeIssues(). */
258
267
  export async function runtimeIssuesUncached(runtime) {
268
+ if (runtimeIssuesUncachedForTests) {
269
+ return await runtimeIssuesUncachedForTests(runtime);
270
+ }
259
271
  const issues = [];
260
272
  if (runtime === "claude_code") {
261
273
  if (!claudeBinExists()) {
@@ -542,6 +554,15 @@ export async function healthForAgent(agent, agentDeckOnline, runtimeIssuesByRunt
542
554
  };
543
555
  }
544
556
  export async function listAgentsWithHealth(agents) {
557
+ // NOT-369: ensure the one-time AC sleep-timer notice is computed for the health surface
558
+ // (informational only — never attached as an AgentHealthIssue / never blocks admission).
559
+ try {
560
+ const { ensureSleepTimerCheckedAtStartup } = await import("../power/sleep-timer.js");
561
+ ensureSleepTimerCheckedAtStartup();
562
+ }
563
+ catch {
564
+ // Best-effort: health listing must not fail over power checks.
565
+ }
545
566
  const agentDeckOnline = await checkAgentDeckHealth();
546
567
  const mcpRegistration = checkAgentDeckMcpRegistration();
547
568
  const needsDeckAccess = agents.some((a) => a.deckId);
@@ -319,6 +319,15 @@ export function extractClaudeCapacityFromEvents(events, nowMs = Date.now(), fall
319
319
  return null;
320
320
  return { runtime: CLAUDE_RUNTIME, windows: [...byKey.values()], unavailable: [] };
321
321
  }
322
+ let claudeCriticalReadingSeq = 0;
323
+ /**
324
+ * Monotonic count of successful Claude 5H/1W writes from any source (probe,
325
+ * local cache, live session). The probe's in-memory failure streak compares
326
+ * against it so a non-probe success breaks the streak too (NOT-366).
327
+ */
328
+ export function claudeCriticalReadingWriteSeq() {
329
+ return claudeCriticalReadingSeq;
330
+ }
322
331
  /**
323
332
  * Persist Claude window readings as normalized capacity snapshots
324
333
  * (per-window upserts — siblings not in this observation are left untouched,
@@ -338,21 +347,32 @@ export function recordClaudeWindowReadings(readings, runtime, nowMs = Date.now()
338
347
  const persistable = normalized.filter((n) => n.unavailableReason === null && n.remainingPercent !== null);
339
348
  if (persistable.length === 0)
340
349
  return null;
341
- const storedObservedAt = new Map(listCapacitySnapshots(runtime).map((row) => [row.windowKey, row.observedAt]));
350
+ const stored = new Map(listCapacitySnapshots(runtime).map((row) => [row.windowKey, row]));
342
351
  const fresh = persistable.filter((w) => {
343
- const prev = storedObservedAt.get(w.windowKey);
344
- if (!prev)
352
+ const prevRow = stored.get(w.windowKey);
353
+ if (!prevRow)
345
354
  return true;
346
- const prevMs = Date.parse(prev);
355
+ // NOT-366: an unavailable row (probe failure recorded over a stale
356
+ // reading) keeps the last-good value and its `observedAt` only for this
357
+ // arbitration. With no last-good value any reading wins; otherwise only
358
+ // a strictly newer one does — re-ingesting the same stale cache sample
359
+ // every poll must not wipe the recorded failure reason.
360
+ const unavailable = prevRow.source === "unavailable" || prevRow.unavailableReason !== null;
361
+ if (unavailable && prevRow.remainingPercent === null)
362
+ return true;
363
+ const prevMs = Date.parse(prevRow.observedAt);
347
364
  const nextMs = Date.parse(w.observedAt);
348
365
  if (!Number.isFinite(prevMs))
349
366
  return true;
350
367
  if (!Number.isFinite(nextMs))
351
368
  return false;
352
- return nextMs >= prevMs;
369
+ return unavailable ? nextMs > prevMs : nextMs >= prevMs;
353
370
  });
354
371
  if (fresh.length === 0)
355
372
  return 0;
373
+ if (fresh.some((w) => w.criticalRole === "five_hour" || w.criticalRole === "weekly")) {
374
+ claudeCriticalReadingSeq += 1;
375
+ }
356
376
  recordCapacitySnapshots(runtime, fresh.map((w) => ({
357
377
  windowKey: w.windowKey,
358
378
  providerBucket: w.providerBucket,
@@ -73,7 +73,11 @@
73
73
  // probe's `usage_report.rate_limits.limits[]` is this exact same shape and
74
74
  // shares the same role mapping (`limitEntryRole`).
75
75
  //
76
- // Probe argv (verified live at 2.1.283):
76
+ // Probe argv (first verified at 2.1.283; re-validated live at 2.1.292 on
77
+ // 2026-10-07 — NOT-366: stream-json still carries `usage_report` on the
78
+ // synthetic assistant event; `--output-format json` prints only the result
79
+ // event and so never shows it; a non-subscription auth source yields a cost
80
+ // summary with no report at all. See docs/RUNTIME_CAPACITY.md):
77
81
  // - `-p "/usage"` — the fixed local slash-command; never interpolated, never
78
82
  // a natural-language prompt. Resolved entirely locally: no model call, no
79
83
  // tokens, no cost.
@@ -107,9 +111,10 @@ import path from "node:path";
107
111
  import { getDataDir } from "../db/index.js";
108
112
  import { resolveClaudeBin } from "../cli-env.js";
109
113
  import { parseNdjson } from "../runners/stream-json.js";
110
- import { listCapacitySnapshots } from "../repository/runtime-capacity.js";
114
+ import { listCapacitySnapshots, recordCapacitySnapshots } from "../repository/runtime-capacity.js";
115
+ import { deriveWindowLabel, isWindowKnown, } from "@agent-dealer/shared";
111
116
  import { configuredCapacityRuntimes } from "./service.js";
112
- import { CLAUDE_RUNTIME, extractClaudeCapacityFromEvents, normalizeClaudeResetsAt, recordClaudeCapacityFromEvents, recordClaudeWindowReadings, } from "./claude-events.js";
117
+ import { CLAUDE_RUNTIME, claudeCriticalReadingWriteSeq, extractClaudeCapacityFromEvents, normalizeClaudeResetsAt, recordClaudeCapacityFromEvents, recordClaudeWindowReadings, } from "./claude-events.js";
113
118
  /** Override for the Claude cache file (tests, smoke). */
114
119
  export const CLAUDE_CACHE_FILE_ENV = "AGENT_DEALER_CLAUDE_CACHE_FILE";
115
120
  /**
@@ -545,6 +550,17 @@ export function defaultProbeRunner(bin, argv, opts) {
545
550
  });
546
551
  });
547
552
  }
553
+ function probeAuthSource(events) {
554
+ for (const e of events) {
555
+ if (e.type !== "system" || e.subtype !== "init")
556
+ continue;
557
+ const src = e.apiKeySource;
558
+ // Allow-list the shape so only a source *name* can ever be recorded.
559
+ if (typeof src === "string" && /^[A-Za-z0-9_.-]{1,64}$/.test(src))
560
+ return src;
561
+ }
562
+ return null;
563
+ }
548
564
  function probeCostFromEvents(events) {
549
565
  for (let i = events.length - 1; i >= 0; i--) {
550
566
  const e = events[i];
@@ -735,6 +751,7 @@ export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
735
751
  events = [];
736
752
  }
737
753
  const costUsd = probeCostFromEvents(events);
754
+ const authSource = probeAuthSource(events);
738
755
  // Fail-closed structural checks (reviewer-requested, 2026-09-27, PR #165):
739
756
  // the whole design rests on `/usage` resolving as a local command that
740
757
  // never reaches the model. If either assumption is violated — no local-
@@ -748,9 +765,9 @@ export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
748
765
  // logged as a distinct failure kind) instead of silently becoming a
749
766
  // recurring paid probe again.
750
767
  if (!hasLocalUsageCommandMarker(events))
751
- return fail("not_local_command", { costUsd });
768
+ return fail("not_local_command", { costUsd, authSource });
752
769
  if (costUsd !== 0)
753
- return fail("unexpected_cost", { costUsd });
770
+ return fail("unexpected_cost", { costUsd, authSource });
754
771
  // Primary signal: the `/usage` local command's own structured result.
755
772
  // Present on every successful run (verified live) — most reliable source.
756
773
  let usageReportRoles = new Set();
@@ -801,14 +818,16 @@ export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
801
818
  // Advisory — stream coverage below still counts.
802
819
  }
803
820
  const covered = new Set([...usageReportRoles, ...streamRoles, ...cacheRoles]);
804
- if (spawnResult.exitCode !== 0 && covered.size === 0)
805
- return fail("nonzero_exit", { costUsd });
821
+ if (spawnResult.exitCode !== 0 && covered.size === 0) {
822
+ return fail("nonzero_exit", { costUsd, authSource });
823
+ }
806
824
  if (!covered.has("five_hour") || !covered.has("weekly")) {
807
- // A `/usage` run should always carry both roles directly in
808
- // `usage_report.rate_limits.limits[]` (verified live) — a `no_windows`
809
- // streak here means something changed upstream and is worth revisiting,
810
- // but since the refresh is free it is not itself a cost problem.
811
- return fail("no_windows", { costUsd });
825
+ // A subscription-authenticated `/usage` run carries both roles directly
826
+ // in `usage_report.rate_limits.limits[]` (re-verified live at 2.1.292).
827
+ // `no_windows` means the report was absent — e.g. an API-key auth source
828
+ // (see `authSource`), or the CLI's own plan-limits fetch failing — and
829
+ // is persisted as the window's unavailable reason (NOT-366).
830
+ return fail("no_windows", { costUsd, authSource });
812
831
  }
813
832
  const out = {
814
833
  ok: true,
@@ -818,6 +837,7 @@ export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
818
837
  timedOut: false,
819
838
  durationMs,
820
839
  failureKind: null,
840
+ authSource,
821
841
  };
822
842
  logProbeOutcome(out);
823
843
  appendProbeDiagnostic({
@@ -830,6 +850,104 @@ export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
830
850
  });
831
851
  return out;
832
852
  }
853
+ // ---------------------------------------------------------------------------
854
+ // Unavailable-reason persistence (NOT-366)
855
+ // ---------------------------------------------------------------------------
856
+ /**
857
+ * Operator-readable cause for a failed refresh. Deliberately prose: the
858
+ * internal `ProbeFailureKind` identifiers stay in the diagnostic log and
859
+ * never reach the stored row, the API, or the header (NOT-266).
860
+ */
861
+ export function claudeProbeFailureMessage(kind, authSource = null) {
862
+ switch (kind) {
863
+ case "spawn":
864
+ return "the Claude CLI could not be started for the /usage refresh";
865
+ case "timeout":
866
+ return "the Claude /usage refresh timed out";
867
+ case "nonzero_exit":
868
+ return "the Claude /usage refresh exited with an error";
869
+ case "no_windows":
870
+ if (authSource !== null && authSource !== "none") {
871
+ return (`Claude /usage reported no plan limits: the CLI is authenticated via ${authSource}, ` +
872
+ "which has no 5H/1W subscription windows");
873
+ }
874
+ return "Claude /usage reported no 5H/1W plan limits";
875
+ case "not_local_command":
876
+ return "Claude /usage did not run as a local command, so its result was not trusted";
877
+ case "unexpected_cost":
878
+ return "the Claude /usage refresh could not be verified as free, so its result was rejected";
879
+ }
880
+ }
881
+ function failureUnavailableReason(kind) {
882
+ // A payload arrived but yielded no trusted windows vs. no payload at all.
883
+ return kind === "no_windows" || kind === "not_local_command" || kind === "unexpected_cost"
884
+ ? "unparsable"
885
+ : "missing";
886
+ }
887
+ /**
888
+ * Record a failed refresh on every critical window that has no current
889
+ * reading, so absence of data is stored with its cause instead of inferred
890
+ * from an empty header. A window still carrying a current reading is left
891
+ * alone (a fresh reading always wins). An existing stale row keeps its
892
+ * last-good value and `observedAt` — only so newer-wins arbitration in
893
+ * `recordClaudeWindowReadings` still works — and is flagged
894
+ * `source: unavailable`, so read-time classification never shows it as a
895
+ * number. The streak (count + first failure) continues from whatever is
896
+ * already stored, so it survives a server restart; any successful reading
897
+ * rewrites the row with no detail, clearing it.
898
+ */
899
+ export function recordClaudeAcquisitionFailure(result, nowMs = Date.now(), inMemoryStreak = { count: 1, firstFailureMs: nowMs }) {
900
+ const kind = result.failureKind;
901
+ if (!kind)
902
+ return 0;
903
+ const rows = new Map(listCapacitySnapshots(CLAUDE_RUNTIME).map((r) => [r.windowKey, r]));
904
+ let prevCount = 0;
905
+ let firstMs = inMemoryStreak.firstFailureMs ?? nowMs;
906
+ for (const identity of Object.values(CACHE_WINDOW_IDENTITIES)) {
907
+ const detail = rows.get(identity.windowKey)?.unavailableDetail;
908
+ if (!detail)
909
+ continue;
910
+ prevCount = Math.max(prevCount, detail.consecutiveFailures);
911
+ const ms = Date.parse(detail.firstFailureAt);
912
+ if (Number.isFinite(ms))
913
+ firstMs = Math.min(firstMs, ms);
914
+ }
915
+ const nowIso = new Date(nowMs).toISOString();
916
+ const unavailableDetail = {
917
+ message: claudeProbeFailureMessage(kind, result.authSource ?? null),
918
+ consecutiveFailures: Math.max(prevCount + 1, inMemoryStreak.count, 1),
919
+ firstFailureAt: new Date(Math.min(firstMs, nowMs)).toISOString(),
920
+ lastFailureAt: nowIso,
921
+ };
922
+ const writes = Object.values(CACHE_WINDOW_IDENTITIES).flatMap((identity) => {
923
+ const row = rows.get(identity.windowKey);
924
+ if (row && isWindowKnown(row, nowMs))
925
+ return [];
926
+ return [
927
+ {
928
+ windowKey: identity.windowKey,
929
+ providerBucket: identity.providerBucket,
930
+ durationMinutes: identity.durationMinutes,
931
+ displayLabel: row?.displayLabel ?? deriveWindowLabel(identity.durationMinutes, identity.providerBucket),
932
+ usedValue: row?.usedValue ?? null,
933
+ usedUnit: row?.usedUnit ?? null,
934
+ remainingPercent: row?.remainingPercent ?? null,
935
+ resetAt: row?.resetAt ?? null,
936
+ observedAt: row?.observedAt ?? nowIso,
937
+ freshUntil: row?.freshUntil ?? null,
938
+ expiresAt: row?.expiresAt ?? null,
939
+ source: "unavailable",
940
+ unavailableReason: failureUnavailableReason(kind),
941
+ evidenceRef: CLAUDE_PROBE_EVIDENCE_REF,
942
+ criticalRole: identity.criticalRole,
943
+ unavailableDetail,
944
+ },
945
+ ];
946
+ });
947
+ if (writes.length > 0)
948
+ recordCapacitySnapshots(CLAUDE_RUNTIME, writes);
949
+ return writes.length;
950
+ }
833
951
  /**
834
952
  * Newest valid account-wide observation per critical window (event or cache
835
953
  * — both share window keys and critical roles). A row counts when it
@@ -877,11 +995,15 @@ export function newestValidClaudeObservationMs(nowMs = Date.now()) {
877
995
  let claudeProbeInFlight = null;
878
996
  let lastClaudeProbeAttemptMs = 0;
879
997
  let consecutiveClaudeProbeFailures = 0;
998
+ let firstClaudeProbeFailureMs = null;
999
+ /** Success-write sequence seen when the current failure streak began. */
1000
+ let claudeProbeStreakReadingSeq = 0;
880
1001
  /** Test helper — clear single-flight, attempt, and backoff state. */
881
1002
  export function resetClaudeCapacityRefreshState() {
882
1003
  claudeProbeInFlight = null;
883
1004
  lastClaudeProbeAttemptMs = 0;
884
1005
  consecutiveClaudeProbeFailures = 0;
1006
+ firstClaudeProbeFailureMs = null;
885
1007
  }
886
1008
  /** Test helper — observe backoff/attempt state without spawning. */
887
1009
  export function claudeProbeRefreshStateForTests() {
@@ -939,7 +1061,31 @@ export async function maybeProbeClaudeCapacity(nowMs = Date.now(), opts = {}) {
939
1061
  claudeProbeInFlight = run;
940
1062
  try {
941
1063
  const result = await run;
942
- consecutiveClaudeProbeFailures = result.ok ? 0 : consecutiveClaudeProbeFailures + 1;
1064
+ if (result.ok) {
1065
+ consecutiveClaudeProbeFailures = 0;
1066
+ firstClaudeProbeFailureMs = null;
1067
+ }
1068
+ else {
1069
+ // Any successful 5H/1W reading since the streak began (local cache
1070
+ // or a live session, not just this probe) ends it: start fresh.
1071
+ const seq = claudeCriticalReadingWriteSeq();
1072
+ if (consecutiveClaudeProbeFailures === 0 || seq !== claudeProbeStreakReadingSeq) {
1073
+ consecutiveClaudeProbeFailures = 0;
1074
+ firstClaudeProbeFailureMs = null;
1075
+ claudeProbeStreakReadingSeq = seq;
1076
+ }
1077
+ consecutiveClaudeProbeFailures += 1;
1078
+ firstClaudeProbeFailureMs ??= nowMs;
1079
+ try {
1080
+ recordClaudeAcquisitionFailure(result, nowMs, {
1081
+ count: consecutiveClaudeProbeFailures,
1082
+ firstFailureMs: firstClaudeProbeFailureMs,
1083
+ });
1084
+ }
1085
+ catch {
1086
+ // Advisory — the failure is still in the diagnostic log.
1087
+ }
1088
+ }
943
1089
  return {
944
1090
  probed: true,
945
1091
  reason: "completed",
@@ -17,7 +17,7 @@ const { listCapacitySnapshots, clearAllCapacitySnapshots, } = await import("../r
17
17
  const { parseNdjson } = await import("../runners/stream-json.js");
18
18
  const { getRuntimeCapacitySnapshot } = await import("./service.js");
19
19
  const { recordClaudeCapacityFromEvents, recordClaudeWindowReadings } = await import("./claude-events.js");
20
- const { CLAUDE_CACHE_FILE_ENV, CLAUDE_CAPACITY_REFRESH_ENV, CLAUDE_PROBE_STALE_AFTER_MS, buildClaudeProbeArgv, claudeCacheFilePath, extractClaudeCacheSubtree, ingestClaudeLocalCache, isClaudePaidFallbackEnabled, maybeProbeClaudeCapacity, newestValidClaudeObservationMs, newestValidClaudeObservationMsByRole, parseClaudeCachedUtilization, probeDiagnosticLogPath, readClaudeLocalCache, refreshClaudeCapacityIfStale, resetClaudeCapacityRefreshState, runClaudeCapacityProbe, } = await import("./claude-local-cache.js");
20
+ const { CLAUDE_CACHE_FILE_ENV, CLAUDE_CAPACITY_REFRESH_ENV, CLAUDE_PROBE_STALE_AFTER_MS, buildClaudeProbeArgv, claudeCacheFilePath, extractClaudeCacheSubtree, ingestClaudeLocalCache, isClaudePaidFallbackEnabled, maybeProbeClaudeCapacity, newestValidClaudeObservationMs, newestValidClaudeObservationMsByRole, parseClaudeCachedUtilization, probeDiagnosticLogPath, readClaudeLocalCache, recordClaudeAcquisitionFailure, refreshClaudeCapacityIfStale, resetClaudeCapacityRefreshState, runClaudeCapacityProbe, } = await import("./claude-local-cache.js");
21
21
  const NOW_MS = Date.parse("2026-09-20T12:00:00.000Z");
22
22
  const FIVE_HOUR_RESET_SEC = Math.floor(NOW_MS / 1000) + 2 * 3600;
23
23
  const SEVEN_DAY_RESET_SEC = Math.floor(NOW_MS / 1000) + 3 * 24 * 3600;
@@ -96,7 +96,7 @@ const throwingRunner = () => {
96
96
  * need direct per-role rows. Resets stay in the future so rows count as
97
97
  * valid observations at NOW_MS.
98
98
  */
99
- function seedWindowAges(fiveHourAgeMs, weeklyAgeMs, nowMs = NOW_MS) {
99
+ function seedWindowAges(fiveHourAgeMs, weeklyAgeMs, nowMs = NOW_MS, expectedPersisted) {
100
100
  const readings = [];
101
101
  const role = (five, ageMs) => ({
102
102
  windowKey: five ? "claude_unified_five_hour" : "claude_unified_seven_day",
@@ -116,7 +116,7 @@ function seedWindowAges(fiveHourAgeMs, weeklyAgeMs, nowMs = NOW_MS) {
116
116
  readings.push(role(true, fiveHourAgeMs));
117
117
  if (weeklyAgeMs !== null)
118
118
  readings.push(role(false, weeklyAgeMs));
119
- assert.equal(recordClaudeWindowReadings(readings, "claude_code", nowMs), readings.length);
119
+ assert.equal(recordClaudeWindowReadings(readings, "claude_code", nowMs), expectedPersisted ?? readings.length);
120
120
  }
121
121
  function successRunner(atMs = NOW_MS) {
122
122
  return async () => ({
@@ -821,3 +821,191 @@ test("a full ~/.claude.json-shaped file yields 5H/1W from the subtree only", ()
821
821
  assert.equal(claude.windows.find((w) => w.displayLabel === "5H").remainingPercent, 91);
822
822
  assert.equal(claude.windows.find((w) => w.displayLabel === "1W").remainingPercent, 77);
823
823
  });
824
+ // ---------------------------------------------------------------------------
825
+ // NOT-366: the re-validated `/usage` contract and unavailable-reason rows
826
+ // ---------------------------------------------------------------------------
827
+ /**
828
+ * Sanitized shape of a real `claude -p "/usage"` stream-json run at 2.1.292
829
+ * (Dealer's exact probe argv, re-validated 2026-10-07): `system/init`, then
830
+ * the synthetic assistant event carrying `local_command_run` and
831
+ * `usage_report.rate_limits.limits[]`, then a `result` event with
832
+ * `local_command: "usage"`, `num_turns: 0`, `total_cost_usd: 0` — and no
833
+ * `usage_report` of its own (which is why `--output-format json`, printing
834
+ * only the result, never shows it). `apiKeySource` other than `none` yields
835
+ * the cost-summary-only variant: no `usage_report` at all.
836
+ */
837
+ function liveUsageStream(observedMs, opts = {}) {
838
+ const withReport = opts.withReport ?? true;
839
+ const events = [
840
+ { type: "system", subtype: "init", apiKeySource: opts.apiKeySource ?? "none", claude_code_version: "2.1.292" },
841
+ {
842
+ type: "assistant",
843
+ message: { model: "<synthetic>", role: "assistant", content: [{ type: "text", text: "<omitted>" }] },
844
+ local_command_source: "<omitted>",
845
+ timestamp: new Date(observedMs).toISOString(),
846
+ local_command_run: { command: "usage", args: "" },
847
+ ...(withReport
848
+ ? {
849
+ usage_report: {
850
+ session: { total_cost_usd: 0, total_api_duration_ms: 0 },
851
+ rate_limits: {
852
+ limits: [
853
+ {
854
+ kind: "session",
855
+ group: "session",
856
+ percent: 6,
857
+ resets_at: new Date(observedMs + 4 * 3600_000).toISOString().replace("Z", "535018+00:00"),
858
+ scope: null,
859
+ severity: "normal",
860
+ is_active: true,
861
+ },
862
+ {
863
+ kind: "weekly_all",
864
+ group: "weekly",
865
+ percent: 5,
866
+ resets_at: new Date(observedMs + 5 * 86_400_000).toISOString().replace("Z", "535047+00:00"),
867
+ scope: null,
868
+ severity: "normal",
869
+ is_active: false,
870
+ },
871
+ ],
872
+ extra_usage: { is_enabled: false },
873
+ },
874
+ },
875
+ }
876
+ : {}),
877
+ },
878
+ { type: "result", subtype: "success", is_error: false, num_turns: 0, total_cost_usd: 0, local_command: "usage" },
879
+ ];
880
+ return events.map((e) => JSON.stringify(e)).join("\n");
881
+ }
882
+ function streamRunner(stdout, atMs) {
883
+ return async () => ({ stdout: stdout(atMs()), exitCode: 0, timedOut: false, spawnError: null });
884
+ }
885
+ test("re-validated 2.1.292 /usage stream yields both five_hour and weekly", async () => {
886
+ const { extractClaudeUsageReportLimits } = await import("./claude-local-cache.js");
887
+ const readings = extractClaudeUsageReportLimits(parseNdjson(liveUsageStream(NOW_MS)), NOW_MS);
888
+ assert.ok(readings);
889
+ assert.deepEqual(readings.map((w) => w.criticalRole).sort(), ["five_hour", "weekly"]);
890
+ // The full probe accepts it: marker present, exactly $0, both roles.
891
+ const result = await runClaudeCapacityProbe(NOW_MS, {
892
+ runner: streamRunner((ms) => liveUsageStream(ms), () => NOW_MS),
893
+ bin: "/fake/claude",
894
+ });
895
+ assert.equal(result.ok, true);
896
+ assert.equal(result.authSource, "none");
897
+ const snap = getRuntimeCapacitySnapshot(NOW_MS + 1000);
898
+ const claude = snap.runtimes.find((r) => r.runtime === "claude_code");
899
+ assert.equal(claude.windows.find((w) => w.criticalRole === "five_hour").remainingPercent, 94);
900
+ assert.equal(claude.windows.find((w) => w.criticalRole === "weekly").remainingPercent, 95);
901
+ });
902
+ test("no_windows failure persists an unavailable row; streak rises to 3 then a success clears it", async () => {
903
+ const failing = streamRunner((ms) => liveUsageStream(ms, { withReport: false }), () => clock);
904
+ let clock = NOW_MS;
905
+ const firstIso = new Date(NOW_MS).toISOString();
906
+ // Backoff after n failures: 14m · 2^n → attempts at +0, +29m, +86m, +199m.
907
+ const attemptAt = [NOW_MS, NOW_MS + 29 * 60_000, NOW_MS + 86 * 60_000];
908
+ for (const [i, at] of attemptAt.entries()) {
909
+ clock = at;
910
+ const outcome = await maybeProbeClaudeCapacity(at, { runner: failing, bin: "/fake/claude" });
911
+ assert.equal(outcome.probed, true);
912
+ assert.equal(outcome.failureKind, "no_windows");
913
+ const rows = listCapacitySnapshots("claude_code");
914
+ assert.equal(rows.length, 2, "one stored row per critical window");
915
+ for (const row of rows) {
916
+ assert.ok(row.unavailableReason, "unavailable_reason is populated");
917
+ assert.equal(row.source, "unavailable");
918
+ assert.ok(row.unavailableDetail);
919
+ assert.equal(row.unavailableDetail.consecutiveFailures, i + 1);
920
+ assert.equal(row.unavailableDetail.firstFailureAt, firstIso);
921
+ assert.equal(row.unavailableDetail.lastFailureAt, new Date(at).toISOString());
922
+ assert.match(row.unavailableDetail.message, /no 5H\/1W plan limits/);
923
+ }
924
+ }
925
+ // Read model: N/A with the operator text, never the internal identifier.
926
+ const failedSnap = getRuntimeCapacitySnapshot(clock);
927
+ const failedClaude = failedSnap.runtimes.find((r) => r.runtime === "claude_code");
928
+ assert.ok(failedClaude.windows.every((w) => w.remainingPercent === null && w.unavailableDetail));
929
+ assert.ok(!JSON.stringify(failedSnap).includes("no_windows"));
930
+ clock = NOW_MS + 199 * 60_000;
931
+ const ok = await maybeProbeClaudeCapacity(clock, {
932
+ runner: streamRunner((ms) => liveUsageStream(ms), () => clock),
933
+ bin: "/fake/claude",
934
+ });
935
+ assert.equal(ok.ok, true);
936
+ for (const row of listCapacitySnapshots("claude_code")) {
937
+ assert.equal(row.unavailableReason, null);
938
+ assert.equal(row.unavailableDetail, null);
939
+ assert.equal(row.source, "observed_event");
940
+ }
941
+ // The next failure starts a new streak at 1.
942
+ clock = NOW_MS + 260 * 60_000;
943
+ await maybeProbeClaudeCapacity(clock, { runner: failing, bin: "/fake/claude" });
944
+ for (const row of listCapacitySnapshots("claude_code")) {
945
+ assert.equal(row.unavailableDetail.consecutiveFailures, 1);
946
+ assert.equal(row.unavailableDetail.firstFailureAt, new Date(clock).toISOString());
947
+ }
948
+ });
949
+ test("a live-session reading between failures restarts the streak instead of inflating it", async () => {
950
+ let clock = NOW_MS;
951
+ const failing = streamRunner((ms) => liveUsageStream(ms, { withReport: false }), () => clock);
952
+ for (const at of [NOW_MS, NOW_MS + 29 * 60_000, NOW_MS + 86 * 60_000]) {
953
+ clock = at;
954
+ await maybeProbeClaudeCapacity(at, { runner: failing, bin: "/fake/claude" });
955
+ }
956
+ assert.equal(listCapacitySnapshots("claude_code")[0].unavailableDetail.consecutiveFailures, 3);
957
+ // A live session (not the probe) lands both windows and clears the stored detail.
958
+ seedWindowAges(0, 0, NOW_MS + 100 * 60_000);
959
+ for (const row of listCapacitySnapshots("claude_code"))
960
+ assert.equal(row.unavailableDetail, null);
961
+ // Once that reading lapses (5H reset passed, 1W expired), the next failure is #1.
962
+ clock = NOW_MS + 200 * 60_000;
963
+ const outcome = await maybeProbeClaudeCapacity(clock, { runner: failing, bin: "/fake/claude" });
964
+ assert.equal(outcome.failureKind, "no_windows");
965
+ const rows = listCapacitySnapshots("claude_code");
966
+ assert.equal(rows.length, 2);
967
+ for (const row of rows) {
968
+ assert.equal(row.unavailableDetail.consecutiveFailures, 1);
969
+ assert.equal(row.unavailableDetail.firstFailureAt, new Date(clock).toISOString());
970
+ }
971
+ });
972
+ test("an API-key auth source is named in the stored reason", async () => {
973
+ const outcome = await maybeProbeClaudeCapacity(NOW_MS, {
974
+ runner: streamRunner((ms) => liveUsageStream(ms, { withReport: false, apiKeySource: "ANTHROPIC_API_KEY" }), () => NOW_MS),
975
+ bin: "/fake/claude",
976
+ });
977
+ assert.equal(outcome.failureKind, "no_windows");
978
+ const row = listCapacitySnapshots("claude_code")[0];
979
+ assert.match(row.unavailableDetail.message, /authenticated via ANTHROPIC_API_KEY/);
980
+ });
981
+ test("a failure leaves a current reading alone and keeps a stale one's value for arbitration", async () => {
982
+ // 5H current (10m old), 1W stale (20m old — past the 15m freshness).
983
+ seedWindowAges(10 * 60_000, 20 * 60_000);
984
+ assert.equal(recordClaudeAcquisitionFailure({ failureKind: "timeout", authSource: null }, NOW_MS), 1);
985
+ const rows = listCapacitySnapshots("claude_code");
986
+ const five = rows.find((r) => r.criticalRole === "five_hour");
987
+ const weekly = rows.find((r) => r.criticalRole === "weekly");
988
+ assert.equal(five.unavailableReason, null);
989
+ assert.equal(five.source, "observed_event");
990
+ assert.equal(weekly.unavailableReason, "missing");
991
+ assert.equal(weekly.unavailableDetail.message, "the Claude /usage refresh timed out");
992
+ assert.equal(weekly.observedAt, new Date(NOW_MS - 20 * 60_000).toISOString());
993
+ // Re-ingesting the same stale sample (every poll does) must not wipe the reason.
994
+ seedWindowAges(null, 20 * 60_000, NOW_MS, 0);
995
+ assert.equal(listCapacitySnapshots("claude_code").find((r) => r.criticalRole === "weekly").unavailableReason, "missing");
996
+ });
997
+ test("a fresh observed_event reading wins over a stored unavailable row and leaves no reason", () => {
998
+ // Failure recorded at NOW with no prior reading at all.
999
+ assert.equal(recordClaudeAcquisitionFailure({ failureKind: "no_windows", authSource: null }, NOW_MS), 2);
1000
+ // A live session's reading observed 2 minutes before the failure stamp.
1001
+ seedWindowAges(2 * 60_000, 2 * 60_000);
1002
+ const snap = getRuntimeCapacitySnapshot(NOW_MS + 60_000);
1003
+ const claude = snap.runtimes.find((r) => r.runtime === "claude_code");
1004
+ for (const w of claude.windows) {
1005
+ assert.ok(w.remainingPercent !== null, "the rendered window shows the value");
1006
+ assert.equal(w.unavailableReason, null);
1007
+ assert.equal(w.unavailableDetail ?? null, null);
1008
+ assert.equal(w.source, "observed_event");
1009
+ }
1010
+ assert.equal(claude.unavailableReason, null);
1011
+ });
@@ -32,6 +32,7 @@ import { MERGE_FAILURE_EVIDENCE_KEY, MERGE_FAILURE_RESPONSE_OPTIONS } from "./hu
32
32
  import { isMergeConflictFailure, runMergeConflictSync } from "./merge-conflict-sync.js";
33
33
  import { triggerLinearPostMerge } from "./linear-merge-verify.js";
34
34
  import { startBaseAdvancedScan } from "./base-advanced-scan.js";
35
+ import { runHeldWork } from "../power/host-awake-lifecycle.js";
35
36
  import { OPERATOR_VERIFICATION_RESPONSE_OPTIONS, formatOperatorCriteria, getOperatorCriteriaForIssue, hasOperatorVerificationForHead, operatorVerificationRequestId, } from "./operator-criteria.js";
36
37
  const run = promisify(execFile);
37
38
  /** Bound each `gh` shell-out so a hang cannot freeze the coordinator process. */
@@ -173,6 +174,11 @@ export function isAutoMergeInFlight(issueId) {
173
174
  export function clearFinalizeInflightForTests() {
174
175
  finalizeInflight.clear();
175
176
  }
177
+ /** Test seam: replace finalizeAutoMergeOnce inside the host-awake hold (lifecycle tests). */
178
+ let finalizeOnceForTests = null;
179
+ export function setFinalizeAutoMergeOnceForTests(fn) {
180
+ finalizeOnceForTests = fn;
181
+ }
176
182
  /**
177
183
  * Completes an auto-merge parked in `final_review` / system ownership after reviewer
178
184
  * approve. Success → done + reflect; failure → needs_human + policy_escalation.
@@ -183,7 +189,8 @@ export function finalizeAutoMerge(issueId) {
183
189
  const existing = finalizeInflight.get(issueId);
184
190
  if (existing)
185
191
  return existing;
186
- const promise = finalizeAutoMergeOnce(issueId).finally(() => {
192
+ // NOT-369: hold idle-sleep for the merge/publish step; release when it settles.
193
+ const promise = runHeldWork(() => (finalizeOnceForTests ?? finalizeAutoMergeOnce)(issueId)).finally(() => {
187
194
  if (finalizeInflight.get(issueId) === promise) {
188
195
  finalizeInflight.delete(issueId);
189
196
  }