agent-dealer 1.2.4 → 1.2.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/bundle/server/dist/adapters/agent-deck-bind.js +33 -3
  2. package/bundle/server/dist/adapters/agent-deck-bind.test.js +60 -2
  3. package/bundle/server/dist/adapters/agent-health.js +28 -6
  4. package/bundle/server/dist/adapters/agent-health.test.js +66 -11
  5. package/bundle/server/dist/adapters/git-worktree.js +187 -18
  6. package/bundle/server/dist/adapters/git-worktree.test.js +293 -0
  7. package/bundle/server/dist/adapters/github.js +110 -12
  8. package/bundle/server/dist/adapters/github.test.js +274 -3
  9. package/bundle/server/dist/adapters/muse-capability.js +356 -0
  10. package/bundle/server/dist/adapters/muse-capability.test.js +464 -0
  11. package/bundle/server/dist/capacity/claude-local-cache.js +266 -81
  12. package/bundle/server/dist/capacity/claude-local-cache.test.js +335 -36
  13. package/bundle/server/dist/capacity/muse-host.js +92 -32
  14. package/bundle/server/dist/capacity/muse-host.test.js +130 -4
  15. package/bundle/server/dist/coordinator/admission.test.js +85 -0
  16. package/bundle/server/dist/coordinator/args.js +4 -3
  17. package/bundle/server/dist/coordinator/checkpoint.js +5 -3
  18. package/bundle/server/dist/coordinator/checkpoint.test.js +23 -0
  19. package/bundle/server/dist/coordinator/commands.js +100 -12
  20. package/bundle/server/dist/coordinator/developer-effect.js +128 -16
  21. package/bundle/server/dist/coordinator/developer-effect.test.js +291 -12
  22. package/bundle/server/dist/coordinator/human-resolution.js +33 -1
  23. package/bundle/server/dist/coordinator/muse-developer.integration.test.js +291 -55
  24. package/bundle/server/dist/coordinator/muse-spawn.js +72 -164
  25. package/bundle/server/dist/coordinator/projection.js +1 -0
  26. package/bundle/server/dist/coordinator/prompts.js +25 -6
  27. package/bundle/server/dist/coordinator/prompts.test.js +36 -8
  28. package/bundle/server/dist/coordinator/routing.js +1 -0
  29. package/bundle/server/dist/coordinator/routing.test.js +20 -0
  30. package/bundle/server/dist/coordinator/spawn.js +9 -1
  31. package/bundle/server/dist/coordinator/worktree-cwd-guard.js +48 -0
  32. package/bundle/server/dist/coordinator/worktree-cwd-guard.test.js +83 -0
  33. package/bundle/server/dist/coordinator/worktree-owner-liveness.js +6 -1
  34. package/bundle/server/dist/coordinator/worktree-owner-liveness.test.js +2 -1
  35. package/bundle/server/dist/repository/human-actions.js +13 -0
  36. package/bundle/server/dist/routes/human-actions.js +10 -1
  37. package/bundle/server/dist/routes/human-actions.test.js +117 -1
  38. package/bundle/server/dist/routes/index.js +12 -5
  39. package/bundle/server/dist/routes/runtime-capacity.test.js +3 -2
  40. package/bundle/server/dist/routes/version.js +30 -0
  41. package/bundle/server/dist/routes/version.test.js +21 -0
  42. package/bundle/server/dist/runners/muse-config-core.js +63 -16
  43. package/bundle/server/dist/runners/muse-config.js +1 -1
  44. package/bundle/server/dist/runners/muse-config.test.js +135 -29
  45. package/bundle/server/dist/runners/spawn-cli.js +17 -3
  46. package/bundle/server/dist/runners/spawn-cli.test.js +59 -0
  47. package/bundle/server/package.json +2 -2
  48. package/bundle/server/static-ui/assets/{index-CYqRh_-S.css → index-BII-LgB8.css} +1 -1
  49. package/bundle/server/static-ui/assets/index-CoDzZcsz.js +60 -0
  50. package/bundle/server/static-ui/index.html +2 -2
  51. package/bundle/shared/dist/agents.d.ts +15 -15
  52. package/bundle/shared/dist/agents.js +6 -0
  53. package/bundle/shared/dist/index.d.ts +7 -7
  54. package/bundle/shared/package.json +1 -1
  55. package/dist/doctor.d.ts +7 -3
  56. package/dist/doctor.js +21 -13
  57. package/dist/doctor.test.js +9 -8
  58. package/dist/managed/index.d.ts +1 -1
  59. package/dist/managed/index.js +1 -1
  60. package/dist/managed/updater.d.ts +16 -6
  61. package/dist/managed/updater.js +28 -11
  62. package/dist/managed-update-restart.test.d.ts +1 -0
  63. package/dist/managed-update-restart.test.js +227 -0
  64. package/dist/ports.d.ts +6 -0
  65. package/dist/ports.js +10 -0
  66. package/dist/runtime-state.d.ts +10 -0
  67. package/dist/runtime-state.js +26 -0
  68. package/dist/start.js +67 -2
  69. package/dist/status.js +30 -2
  70. package/dist/update-check.js +3 -6
  71. package/package.json +1 -1
  72. package/bundle/server/static-ui/assets/index-UO4lHZw4.js +0 -60
@@ -1,14 +1,11 @@
1
1
  // packages/server/src/capacity/claude-local-cache.ts
2
2
  //
3
- // NOT-268: Claude account capacity — local-first source ladder with a
4
- // one-hour paid fallback.
3
+ // NOT-268: Claude account capacity — local-first source ladder with a free
4
+ // `/usage` refresh (corrected 2026-09-27; see below for what changed and why).
5
5
  //
6
6
  // Goal: keep Claude 5H/1W useful between Dealer runs. Production's last
7
7
  // Dealer-observed sample can be days old (no recent Dealer-managed Claude
8
- // run), while Claude Code itself maintains exact provider usage in the
9
- // `cachedUsageUtilization` key of `~/.claude.json` on every run — including
10
- // interactive runs outside Dealer. So the freshest valid observation wins
11
- // from this ladder:
8
+ // run). The ladder:
12
9
  //
13
10
  // 1. Existing Dealer `rate_limit_event` ingestion (claude-events.ts — kept
14
11
  // as-is, session-end, per-window newer-wins).
@@ -16,9 +13,39 @@
16
13
  // `~/.claude.json` (this module — free, ingested on every capacity
17
14
  // read; only that subtree is ever parsed, the rest of the config —
18
15
  // accountUuid, email, credentials, projects — is never retained).
19
- // 3. One minimal bounded paid probe when every valid 5H/1W observation is
20
- // older than 60 minutes. This is the default; operators can explicitly
21
- // disable it with `AGENT_DEALER_CLAUDE_CAPACITY_REFRESH=off`.
16
+ // 3. One minimal FREE refresh when either valid 5H/1W observation is
17
+ // missing or at least 14 minutes old: `claude -p "/usage"` (NOT-281 —
18
+ // the 15-minute display freshness would otherwise lapse into N/A while
19
+ // the old 60-minute trigger waited). This is the default behavior.
20
+ // Set `AGENT_DEALER_CLAUDE_CAPACITY_REFRESH=off` to disable it entirely
21
+ // — reading capacity then never spawns Claude. Unrecognized values also
22
+ // fail closed (stay disabled).
23
+ //
24
+ // 2026-09-27 correction (what changed and why): the original design for rung
25
+ // 3 spawned a real one-turn model prompt (`claude -p "Reply with exactly: ok"
26
+ // --model haiku ...`), assuming any `claude` invocation would emit a
27
+ // `rate_limit_event` and/or refresh rung 2's cache file. A live proof against
28
+ // a real account falsified both assumptions: three attempts each overspent
29
+ // the $0.01 cap on ambient context alone (~11K cache-creation tokens before
30
+ // the model could even answer), none emitted a `rate_limit_event`, and the
31
+ // cache file was untouched afterward — a decompiled trace of the installed
32
+ // CLI showed the cache write (`Juo()`) lives behind the interactive
33
+ // usage/plan-limits fetch, not the ordinary chat-turn path. That path
34
+ // initially looked unreachable from a headless probe, until a second live
35
+ // test found the actual trigger: passing the **local slash-command**
36
+ // `/usage` as the `-p` prompt. Claude Code resolves `/usage` as a local
37
+ // command — no model call, `total_cost_usd: 0`, ~300ms — and its NDJSON
38
+ // output carries the exact structured result Dealer needs directly, at
39
+ // `usage_report.rate_limits.limits[]` (`{ kind: "session"|"weekly_all",
40
+ // percent, resets_at, ... }` — the same shape rung 2 already parses from the
41
+ // cache file's `limits[]`). It also performs the identical write rung 2
42
+ // reads, confirmed by `cachedUsageUtilization.fetchedAtMs` changing on every
43
+ // run. Verified reproducible across repeated live invocations. This is a
44
+ // documented, user-facing CLI command (listed in the session's own
45
+ // `slash_commands`), not an internal/undocumented surface, and it costs
46
+ // nothing — so the refresh is back to default-on, and the diagnostic log
47
+ // records real ~$0 outcomes instead of a `no_windows` failure streak. See
48
+ // docs/RUNTIME_CAPACITY.md for the full writeup and both live proofs.
22
49
  //
23
50
  // Both file sources write the same `claude_unified_five_hour` /
24
51
  // `claude_unified_seven_day` window keys through the shared
@@ -26,7 +53,7 @@
26
53
  // automatic per window and an older cache can never clobber a newer event
27
54
  // (or vice versa). `source` stays `observed_event` for both; `evidenceRef`
28
55
  // tells them apart (`claude-session:unified-windows` vs
29
- // `claude-cache:cachedUsageUtilization` vs `claude-probe:minimal-print`).
56
+ // `claude-cache:cachedUsageUtilization` vs `claude-probe:usage-command`).
30
57
  //
31
58
  // Local-cache safety: only the `cachedUsageUtilization` subtree
32
59
  // (`fetchedAtMs`, `utilization.five_hour`, `utilization.seven_day`,
@@ -42,34 +69,37 @@
42
69
  // in (1, 100] read as percent. Anything else is malformed — rejected.
43
70
  // - `limits[]`: `{ kind|group: "session"|"weekly_all", percent: <0–100>,
44
71
  // resets_at: <ISO-8601>, ... }`. Model-specific and overage entries are
45
- // dropped — only the account-wide pair is ever normalized.
72
+ // dropped — only the account-wide pair is ever normalized. The `/usage`
73
+ // probe's `usage_report.rate_limits.limits[]` is this exact same shape and
74
+ // shares the same role mapping (`limitEntryRole`).
46
75
  //
47
- // Probe argv (verified live at 2.1.283 — `claude -p --max-turns 1 --model
48
- // <bogus>` parses flags and fails only on model resolution, spending
49
- // nothing, even though `--help` hides the flag):
50
- // - `--model haiku` — the cheapest supported alias (matches
51
- // runners/models.ts `haiku` "latest alias").
52
- // - `--max-turns 1` — hard one-turn bound, belt-and-braces with the
53
- // structural bound below.
54
- // - `--tools ""` + `--strict-mcp-config` (with no `--mcp-config`) — no tools,
55
- // no MCP. With no tools the model cannot continue past its first response,
56
- // the fixed minimal prompt asks for a single word, and
57
- // `--max-budget-usd 0.01` hard-caps spend.
76
+ // Probe argv (verified live at 2.1.283):
77
+ // - `-p "/usage"` — the fixed local slash-command; never interpolated, never
78
+ // a natural-language prompt. Resolved entirely locally: no model call, no
79
+ // tokens, no cost.
80
+ // - `--model haiku` + `--max-turns 1` + `--tools ""` — structural safeguards
81
+ // kept even though `/usage` never reaches the model on 2.1.283: verified
82
+ // live that they do not break local resolution, and they bound the
83
+ // (unobserved) case where a different CLI build makes `/usage` fall
84
+ // through to a real prompt (2026-09-27 review hardening — see
85
+ // `buildClaudeProbeArgv`).
86
+ // - `--strict-mcp-config` (with no `--mcp-config`) — no MCP servers loaded.
58
87
  // - `--no-session-persistence` — the probe leaves no resumable session.
59
88
  // - `--output-format stream-json` (+ `--verbose`, matching Dealer's own
60
- // stream-json parsing) so `rate_limit_event`s can be ingested.
89
+ // stream-json parsing) so the `usage_report` and any `rate_limit_event`s
90
+ // can be ingested.
91
+ // - `--max-budget-usd 0.01` — defensive belt-and-braces only: normal cost is
92
+ // exactly $0. Not sufficient alone (ambient context can blow the cap
93
+ // before the check fires — see live proof #1 below), so the result is
94
+ // also checked for the local-command marker and exactly-zero cost before
95
+ // it counts as success (`runClaudeCapacityProbe`).
61
96
  // - `--bare` is deliberately NOT used: it restricts auth to
62
97
  // ANTHROPIC_API_KEY/apiKeyHelper and would bypass the account's OAuth
63
- // login — the probe must bill to the account whose capacity it measures.
98
+ // login — the probe must read the capacity of the account it measures.
64
99
  // - Ambient settings are kept (auth must resolve); no worktree is created
65
100
  // (cwd is the OS temp dir) and nothing touches Dealer workflow/session
66
101
  // rows, worktrees, commits, PRs, or queue events — the probe spawns
67
102
  // `claude` directly, never through the coordinator.
68
- //
69
- // If a live proof ever shows the minimal probe does not reliably emit 5H/1W
70
- // (neither in its stream nor via the cache side effect), the probe is a
71
- // recurring paid no-op: disable the fallback and revise this ticket instead of
72
- // shipping it. The `no_windows` diagnostic below exists to make that visible.
73
103
  import { spawn } from "node:child_process";
74
104
  import fs from "node:fs";
75
105
  import os from "node:os";
@@ -82,7 +112,12 @@ import { configuredCapacityRuntimes } from "./service.js";
82
112
  import { CLAUDE_RUNTIME, extractClaudeCapacityFromEvents, normalizeClaudeResetsAt, recordClaudeCapacityFromEvents, recordClaudeWindowReadings, } from "./claude-events.js";
83
113
  /** Override for the Claude cache file (tests, smoke). */
84
114
  export const CLAUDE_CACHE_FILE_ENV = "AGENT_DEALER_CLAUDE_CACHE_FILE";
85
- /** Paid-fallback setting. Unset defaults to PAID_AFTER_1H; `off` disables it. */
115
+ /**
116
+ * Refresh setting. Default ON (unset/empty, or the historical explicit
117
+ * `paid-after-1h` value — kept accepted for backward compat with 1.2.4
118
+ * configs, though the refresh is no longer paid). Only `off` (or any other
119
+ * unrecognized value) disables it, failing closed on typos.
120
+ */
86
121
  export const CLAUDE_CAPACITY_REFRESH_ENV = "AGENT_DEALER_CLAUDE_CAPACITY_REFRESH";
87
122
  export const CLAUDE_CAPACITY_REFRESH_PAID_VALUE = "paid-after-1h";
88
123
  export const CLAUDE_CAPACITY_REFRESH_OFF_VALUE = "off";
@@ -91,17 +126,29 @@ export const CLAUDE_CAPACITY_PROBE_TIMEOUT_ENV = "AGENT_DEALER_CLAUDE_PROBE_TIME
91
126
  export const CLAUDE_PROBE_TIMEOUT_MS_DEFAULT = 90_000;
92
127
  /** Static evidence pointers — never credentials or raw payloads. */
93
128
  export const CLAUDE_CACHE_EVIDENCE_REF = "claude-cache:cachedUsageUtilization";
94
- export const CLAUDE_PROBE_EVIDENCE_REF = "claude-probe:minimal-print";
95
- /** Cheapest supported model alias for the probe (cf. runners/models.ts). */
129
+ export const CLAUDE_PROBE_EVIDENCE_REF = "claude-probe:usage-command";
130
+ /** Fixed local slash-command — asserted in tests; never interpolated, never
131
+ * sent to the model (Claude Code resolves it locally, no API call). */
132
+ export const CLAUDE_PROBE_PROMPT = "/usage";
133
+ /** Cheapest supported model alias — structural safeguard only; see module
134
+ * header on why this stays even though `/usage` never reaches the model. */
96
135
  export const CLAUDE_PROBE_MODEL = "haiku";
97
- /** Fixed minimal prompt — asserted in tests; never interpolated. */
98
- export const CLAUDE_PROBE_PROMPT = "Reply with exactly: ok";
99
- /** Hard spend cap for the probe (USD). */
136
+ /** Defensive spend cap (USD) — normal cost is $0; see module header. */
100
137
  export const CLAUDE_PROBE_MAX_BUDGET_USD = 0.01;
101
- /** A 5H/1W observation newer than this suppresses the paid probe. */
102
- export const CLAUDE_PROBE_STALE_AFTER_MS = 60 * 60 * 1000;
103
- /** Minimum gap between probe attempts (per account); failures back off. */
104
- export const CLAUDE_PROBE_ATTEMPT_COOLDOWN_MS = 60 * 60 * 1000;
138
+ /**
139
+ * Either critical window (5H or 1W) at least this old triggers the refresh.
140
+ * 14 minutes — one minute inside the 15-minute display `freshUntil` (NOT-281)
141
+ * so the asynchronous single-flight `/usage` run normally completes before
142
+ * the UI would mark the reading stale. Checked per window, never as a
143
+ * newest-of-pair: one fresh sibling must not suppress its stale twin.
144
+ */
145
+ export const CLAUDE_PROBE_STALE_AFTER_MS = 14 * 60 * 1000;
146
+ /**
147
+ * Minimum gap between probe attempts (per account). Aligned with the trigger
148
+ * above: a healthy observation causes at most one refresh per 14 minutes;
149
+ * failures back off exponentially (14m → 28m → 56m → ~2h, capped at 8h).
150
+ */
151
+ export const CLAUDE_PROBE_ATTEMPT_COOLDOWN_MS = 14 * 60 * 1000;
105
152
  export const CLAUDE_PROBE_BACKOFF_CAP_MS = 8 * 60 * 60 * 1000;
106
153
  /** Account-wide window identities for the local cache (exact, lowercase). */
107
154
  const FIVE_HOUR_LIMIT_NAMES = new Set(["five_hour", "session"]);
@@ -160,9 +207,23 @@ export function claudeProbeTimeoutMs() {
160
207
  return CLAUDE_PROBE_TIMEOUT_MS_DEFAULT;
161
208
  }
162
209
  /**
163
- * Fixed probe argv. Pinned by tests: the fixed minimal prompt, cheapest
164
- * model, `--max-turns 1`, no tools, no MCP, no session persistence, stream
165
- * JSON, ≤$0.01 budget.
210
+ * Fixed probe argv. Pinned by tests: the fixed local slash-command, cheapest
211
+ * model, hard one-turn bound, no tools, no MCP, no session persistence,
212
+ * stream JSON, defensive ≤$0.01 budget.
213
+ *
214
+ * Reviewer-requested hardening (2026-09-27, PR #165): `--model`/`--max-turns`/
215
+ * `--tools` were originally dropped as "meaningless" because `/usage` never
216
+ * reaches the model on 2.1.283 — but that is an empirical fact about one CLI
217
+ * version, not a contract. Verified live that keeping all three does not
218
+ * break local resolution (still `$0`, still `local_command: usage`), so they
219
+ * stay as structural bounds: if some other CLI build ever makes `/usage`
220
+ * fall through to a real prompt, the model is the cheapest alias, gets
221
+ * exactly one turn, and has no tools to call — the same belt-and-braces the
222
+ * original (abandoned) paid-turn design relied on. `--max-budget-usd` alone
223
+ * is not sufficient (this repo's own live proof showed ambient context can
224
+ * blow the cap before the check fires); see `runClaudeCapacityProbe` for the
225
+ * matching fail-closed checks on the result (local-command marker present,
226
+ * cost must be exactly $0).
166
227
  */
167
228
  export function buildClaudeProbeArgv() {
168
229
  return [
@@ -409,12 +470,14 @@ export function ingestClaudeLocalCache(nowMs = Date.now()) {
409
470
  return recordClaudeWindowReadings(result.windows, CLAUDE_RUNTIME, nowMs);
410
471
  }
411
472
  // ---------------------------------------------------------------------------
412
- // Paid fallback probe (enabled by default, explicit off switch)
473
+ // Free `/usage` refresh (default on)
413
474
  // ---------------------------------------------------------------------------
414
475
  /**
415
- * Paid probing defaults on when the setting is absent/empty. The documented
416
- * `paid-after-1h` value is accepted explicitly; `off` and unrecognized values
417
- * disable spending so a typo never silently changes the configured policy.
476
+ * Refresh defaults ON when the setting is absent/empty. The historical
477
+ * `paid-after-1h` value is still accepted explicitly (it now just means
478
+ * "enabled" — the refresh costs nothing, see module header); `off` and
479
+ * unrecognized values disable it so a typo never silently changes the
480
+ * configured policy.
418
481
  */
419
482
  export function isClaudePaidFallbackEnabled() {
420
483
  const setting = process.env[CLAUDE_CAPACITY_REFRESH_ENV];
@@ -493,6 +556,70 @@ function probeCostFromEvents(events) {
493
556
  }
494
557
  return null;
495
558
  }
559
+ /**
560
+ * True only when the stream shows `/usage` actually resolved as Claude
561
+ * Code's local command (`local_command_run.command === "usage"`) — the
562
+ * structural marker that the probe never reached the model. Absence means
563
+ * the CLI build in use does not behave like 2.1.283; the caller must not
564
+ * trust any windows the stream happens to carry (`not_local_command`).
565
+ */
566
+ function hasLocalUsageCommandMarker(events) {
567
+ return events.some((e) => {
568
+ const run = e.local_command_run;
569
+ if (!run || typeof run !== "object")
570
+ return false;
571
+ return run.command === "usage";
572
+ });
573
+ }
574
+ /**
575
+ * Extract account-wide 5H/1W readings directly from a `/usage` probe's
576
+ * stream: the local-command result event carries `usage_report.rate_limits.
577
+ * limits[]` in the exact shape (and role mapping via `limitEntryRole`) as
578
+ * the cache file's `limits[]` — see module header. This is the primary,
579
+ * most-reliable success signal for the probe (present on every successful
580
+ * `/usage` run, verified live); the rate_limit_event and cache-re-read
581
+ * checks below stay as additional, non-exclusive corroboration. Returns
582
+ * null when no usable account-wide window is present — never fabricated.
583
+ */
584
+ export function extractClaudeUsageReportLimits(events, nowMs, evidenceRef = CLAUDE_PROBE_EVIDENCE_REF) {
585
+ let limits = null;
586
+ let eventTimestamp = null;
587
+ for (const e of events) {
588
+ const report = e.usage_report;
589
+ if (!report || typeof report !== "object")
590
+ continue;
591
+ const rateLimits = report.rate_limits;
592
+ if (!rateLimits || typeof rateLimits !== "object")
593
+ continue;
594
+ const l = rateLimits.limits;
595
+ if (Array.isArray(l)) {
596
+ limits = l;
597
+ const ts = e.timestamp;
598
+ eventTimestamp = typeof ts === "string" ? ts : null;
599
+ }
600
+ }
601
+ if (!limits)
602
+ return null;
603
+ const parsedTsMs = eventTimestamp !== null ? Date.parse(eventTimestamp) : Number.NaN;
604
+ const observedMs = Number.isFinite(parsedTsMs) ? Math.min(parsedTsMs, nowMs) : nowMs;
605
+ const observedAt = new Date(observedMs).toISOString();
606
+ const byRole = new Map();
607
+ for (const item of limits) {
608
+ if (!item || typeof item !== "object" || Array.isArray(item))
609
+ continue;
610
+ const w = item;
611
+ const name = pickString([w.kind, w.group, w.name, w.window, w.bucket, w.key, w.id]);
612
+ if (!name)
613
+ continue;
614
+ const role = limitEntryRole(name);
615
+ if (!role || byRole.has(role))
616
+ continue;
617
+ const reading = cacheWindowReading(role, item, observedAt, observedMs, evidenceRef);
618
+ if (reading)
619
+ byRole.set(role, reading);
620
+ }
621
+ return byRole.size > 0 ? [...byRole.values()] : null;
622
+ }
496
623
  function rolesCoveredByReadings(readings) {
497
624
  const out = new Set();
498
625
  for (const w of readings) {
@@ -536,12 +663,18 @@ function appendProbeDiagnostic(entry) {
536
663
  }
537
664
  }
538
665
  /**
539
- * Run one minimal bounded probe and ingest whatever 5H/1W it yields — first
540
- * the stream's `rate_limit_event`s, then a re-read of the local cache (the
541
- * probe run itself refreshes Claude's own cache file, so the side effect
542
- * counts even when the stream carries no windows). Success means the union
543
- * covers both critical roles. Never throws; never creates Dealer
544
- * workflow/session rows, worktrees, commits, PRs, or queue events.
666
+ * Run one minimal FREE `/usage` probe and ingest whatever 5H/1W it yields.
667
+ * Fails closed before touching any of it unless the stream both carries the
668
+ * local-command marker (`local_command_run.command === "usage"`) and reports
669
+ * exactly $0 cost — the structural proof `/usage` actually resolved locally
670
+ * rather than falling through to a real (billable) model turn. Once that
671
+ * holds, success ingests primarily the local command's own
672
+ * `usage_report.rate_limits.limits[]` (present on every successful run),
673
+ * plus any `rate_limit_event` the stream happens to carry and a re-read of
674
+ * the local cache (the probe run itself refreshes Claude's own cache file)
675
+ * as non-exclusive corroboration; success means the union covers both
676
+ * critical roles. Never throws; never creates Dealer workflow/session rows,
677
+ * worktrees, commits, PRs, or queue events.
545
678
  */
546
679
  export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
547
680
  const startedRealMs = Date.now();
@@ -583,8 +716,8 @@ export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
583
716
  appendProbeDiagnostic({
584
717
  ts: new Date(nowMs).toISOString(),
585
718
  event: "claude_capacity_probe",
586
- trigger: "stale_60m",
587
- model: CLAUDE_PROBE_MODEL,
719
+ trigger: "stale_14m",
720
+ probeCommand: CLAUDE_PROBE_PROMPT,
588
721
  budgetUsd: CLAUDE_PROBE_MAX_BUDGET_USD,
589
722
  ...out,
590
723
  });
@@ -602,7 +735,38 @@ export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
602
735
  events = [];
603
736
  }
604
737
  const costUsd = probeCostFromEvents(events);
605
- // Ingest the stream first (event timestamps clamp to nowMs inside).
738
+ // Fail-closed structural checks (reviewer-requested, 2026-09-27, PR #165):
739
+ // the whole design rests on `/usage` resolving as a local command that
740
+ // never reaches the model. If either assumption is violated — no local-
741
+ // command marker in the stream, or cost not provably exactly $0 — reject
742
+ // the run entirely before ingesting anything from it, rather than trusting
743
+ // whatever windows a real (unexpected) model turn happened to produce.
744
+ // `costUsd !== 0` (not `> 0`) is deliberate: `null` — a missing or
745
+ // unparsable `total_cost_usd` — is not evidence of zero cost either, and
746
+ // must fail closed exactly like a confirmed charge (second review round).
747
+ // This also makes a future CLI-behavior change loud (backoff engages,
748
+ // logged as a distinct failure kind) instead of silently becoming a
749
+ // recurring paid probe again.
750
+ if (!hasLocalUsageCommandMarker(events))
751
+ return fail("not_local_command", { costUsd });
752
+ if (costUsd !== 0)
753
+ return fail("unexpected_cost", { costUsd });
754
+ // Primary signal: the `/usage` local command's own structured result.
755
+ // Present on every successful run (verified live) — most reliable source.
756
+ let usageReportRoles = new Set();
757
+ try {
758
+ const usageWindows = extractClaudeUsageReportLimits(events, nowMs);
759
+ if (usageWindows) {
760
+ usageReportRoles = rolesCoveredByReadings(usageWindows);
761
+ recordClaudeWindowReadings(usageWindows, CLAUDE_RUNTIME, nowMs);
762
+ }
763
+ }
764
+ catch {
765
+ // A broken stream must not fail the probe path — other checks below
766
+ // may still yield fresh rows.
767
+ }
768
+ // Secondary corroboration: any `rate_limit_event` the stream happens to
769
+ // carry (kept from the original NOT-248 event path; harmless if absent).
606
770
  let streamRoles = new Set();
607
771
  try {
608
772
  const extracted = extractClaudeCapacityFromEvents(events, nowMs, nowMs);
@@ -636,13 +800,14 @@ export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
636
800
  catch {
637
801
  // Advisory — stream coverage below still counts.
638
802
  }
639
- const covered = new Set([...streamRoles, ...cacheRoles]);
803
+ const covered = new Set([...usageReportRoles, ...streamRoles, ...cacheRoles]);
640
804
  if (spawnResult.exitCode !== 0 && covered.size === 0)
641
805
  return fail("nonzero_exit", { costUsd });
642
806
  if (!covered.has("five_hour") || !covered.has("weekly")) {
643
- // The probe spent money but produced no 5H/1W pair — a `no_windows`
644
- // streak is the signal to disable the opt-in and revise the ticket
645
- // rather than ship a recurring paid no-op.
807
+ // A `/usage` run should always carry both roles directly in
808
+ // `usage_report.rate_limits.limits[]` (verified live) — a `no_windows`
809
+ // streak here means something changed upstream and is worth revisiting,
810
+ // but since the refresh is free it is not itself a cost problem.
646
811
  return fail("no_windows", { costUsd });
647
812
  }
648
813
  const out = {
@@ -658,24 +823,27 @@ export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
658
823
  appendProbeDiagnostic({
659
824
  ts: new Date(nowMs).toISOString(),
660
825
  event: "claude_capacity_probe",
661
- trigger: "stale_60m",
662
- model: CLAUDE_PROBE_MODEL,
826
+ trigger: "stale_14m",
827
+ probeCommand: CLAUDE_PROBE_PROMPT,
663
828
  budgetUsd: CLAUDE_PROBE_MAX_BUDGET_USD,
664
829
  ...out,
665
830
  });
666
831
  return out;
667
832
  }
668
- // ---------------------------------------------------------------------------
669
- // Freshness gate, single-flight, backoff
670
- // ---------------------------------------------------------------------------
671
833
  /**
672
- * Newest valid account-wide 5H/1W observation (event or cache — both share
673
- * window keys and critical roles). A row counts when it carries a remaining
674
- * %, its reset is absent or future, and its observed time is not in the
675
- * future. Returns null when no such sample exists.
834
+ * Newest valid account-wide observation per critical window (event or cache
835
+ * — both share window keys and critical roles). A row counts when it
836
+ * carries a remaining %, its reset is absent or future, and its observed
837
+ * time is not in the future. A role maps to null when it has no such
838
+ * sample (missing, expired, or unparsable). Split per window on purpose
839
+ * (NOT-281): the refresh gate must see a stale twin hiding behind a fresh
840
+ * sibling, which a newest-of-pair maximum cannot show.
676
841
  */
677
- export function newestValidClaudeObservationMs(nowMs = Date.now()) {
678
- let max = null;
842
+ export function newestValidClaudeObservationMsByRole(nowMs = Date.now()) {
843
+ const out = {
844
+ five_hour: null,
845
+ weekly: null,
846
+ };
679
847
  for (const row of listCapacitySnapshots(CLAUDE_RUNTIME)) {
680
848
  if (row.criticalRole !== "five_hour" && row.criticalRole !== "weekly")
681
849
  continue;
@@ -689,9 +857,22 @@ export function newestValidClaudeObservationMs(nowMs = Date.now()) {
689
857
  const observedMs = Date.parse(row.observedAt);
690
858
  if (!Number.isFinite(observedMs) || observedMs > nowMs)
691
859
  continue;
692
- max = max === null ? observedMs : Math.max(max, observedMs);
860
+ const prev = out[row.criticalRole];
861
+ out[row.criticalRole] = prev === null ? observedMs : Math.max(prev, observedMs);
693
862
  }
694
- return max;
863
+ return out;
864
+ }
865
+ /**
866
+ * Newest valid account-wide 5H/1W observation across both critical windows.
867
+ * Kept for diagnostics; the refresh gate uses the per-role variant above.
868
+ */
869
+ export function newestValidClaudeObservationMs(nowMs = Date.now()) {
870
+ const byRole = newestValidClaudeObservationMsByRole(nowMs);
871
+ if (byRole.five_hour === null)
872
+ return byRole.weekly;
873
+ if (byRole.weekly === null)
874
+ return byRole.five_hour;
875
+ return Math.max(byRole.five_hour, byRole.weekly);
695
876
  }
696
877
  let claudeProbeInFlight = null;
697
878
  let lastClaudeProbeAttemptMs = 0;
@@ -715,11 +896,13 @@ function probeCooldownMs() {
715
896
  return Math.min(CLAUDE_PROBE_ATTEMPT_COOLDOWN_MS * 2 ** shift, CLAUDE_PROBE_BACKOFF_CAP_MS);
716
897
  }
717
898
  /**
718
- * On-demand paid fallback: when `claude_code` is configured and no valid
719
- * 5H/1W observation is newer than 60 minutes, run one minimal bounded probe
720
- * (single-flight across concurrent readers; at most one attempt per account
721
- * per 60 minutes, backing off exponentially on failure). Explicitly disabled
722
- * is a strict no-op — no spawn, no spend. Never throws, never touches Dealer
899
+ * On-demand free `/usage` refresh: when `claude_code` is configured and
900
+ * either critical window (5H or 1W) is missing or at least 14 minutes old,
901
+ * run one minimal bounded probe (single-flight across concurrent readers;
902
+ * at most one attempt per account per 14 minutes while healthy, backing
903
+ * off exponentially on failure). Per-window on purpose (NOT-281): a fresh
904
+ * sibling must never suppress its stale twin. Explicitly disabled is a
905
+ * strict no-op — no spawn at all. Never throws, never touches Dealer
723
906
  * workflow/session state or `runtime_availability`.
724
907
  */
725
908
  export async function maybeProbeClaudeCapacity(nowMs = Date.now(), opts = {}) {
@@ -729,8 +912,10 @@ export async function maybeProbeClaudeCapacity(nowMs = Date.now(), opts = {}) {
729
912
  if (!configuredCapacityRuntimes().includes(CLAUDE_RUNTIME)) {
730
913
  return { probed: false, reason: "unconfigured" };
731
914
  }
732
- const newest = newestValidClaudeObservationMs(nowMs);
733
- if (newest !== null && nowMs - newest < CLAUDE_PROBE_STALE_AFTER_MS) {
915
+ const byRole = newestValidClaudeObservationMsByRole(nowMs);
916
+ const fiveFresh = byRole.five_hour !== null && nowMs - byRole.five_hour < CLAUDE_PROBE_STALE_AFTER_MS;
917
+ const weeklyFresh = byRole.weekly !== null && nowMs - byRole.weekly < CLAUDE_PROBE_STALE_AFTER_MS;
918
+ if (fiveFresh && weeklyFresh) {
734
919
  return { probed: false, reason: "fresh" };
735
920
  }
736
921
  // Single-flight first: readers arriving while a probe runs share it
@@ -775,8 +960,8 @@ export async function maybeProbeClaudeCapacity(nowMs = Date.now(), opts = {}) {
775
960
  }
776
961
  /**
777
962
  * Full on-demand refresh for `GET /api/runtime-capacity`: ingest the free
778
- * local cache first, then consider the paid probe. Best-effort — never
779
- * throws. The route serves the stored snapshot regardless.
963
+ * local cache first, then consider the free `/usage` probe. Best-effort —
964
+ * never throws. The route serves the stored snapshot regardless.
780
965
  */
781
966
  export async function refreshClaudeCapacityIfStale(nowMs = Date.now(), opts = {}) {
782
967
  // Skip the file read entirely when no Claude account is configured: