agent-dealer 1.2.4 → 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bundle/server/dist/adapters/agent-health.js +25 -2
- package/bundle/server/dist/adapters/agent-health.test.js +55 -2
- package/bundle/server/dist/adapters/github.js +110 -12
- package/bundle/server/dist/adapters/github.test.js +274 -3
- package/bundle/server/dist/adapters/muse-capability.js +330 -0
- package/bundle/server/dist/adapters/muse-capability.test.js +378 -0
- package/bundle/server/dist/capacity/claude-local-cache.js +217 -64
- package/bundle/server/dist/capacity/claude-local-cache.test.js +174 -29
- package/bundle/server/dist/capacity/muse-host.js +60 -2
- package/bundle/server/dist/capacity/muse-host.test.js +125 -1
- package/bundle/server/dist/coordinator/admission.test.js +85 -0
- package/bundle/server/dist/coordinator/commands.js +35 -8
- package/bundle/server/dist/coordinator/developer-effect.js +40 -6
- package/bundle/server/dist/coordinator/developer-effect.test.js +121 -12
- package/bundle/server/dist/coordinator/human-resolution.js +31 -1
- package/bundle/server/dist/coordinator/muse-spawn.js +12 -2
- package/bundle/server/dist/coordinator/prompts.js +15 -0
- package/bundle/server/dist/coordinator/prompts.test.js +20 -0
- package/bundle/server/dist/coordinator/spawn.js +7 -0
- package/bundle/server/dist/coordinator/worktree-cwd-guard.js +48 -0
- package/bundle/server/dist/coordinator/worktree-cwd-guard.test.js +83 -0
- package/bundle/server/dist/routes/human-actions.js +10 -1
- package/bundle/server/dist/routes/human-actions.test.js +117 -1
- package/bundle/server/dist/routes/index.js +9 -5
- package/bundle/server/dist/routes/runtime-capacity.test.js +3 -2
- package/bundle/server/package.json +2 -2
- package/bundle/server/static-ui/assets/{index-CYqRh_-S.css → index-BII-LgB8.css} +1 -1
- package/bundle/server/static-ui/assets/index-Bq8wWpZm.js +60 -0
- package/bundle/server/static-ui/index.html +2 -2
- package/bundle/shared/dist/agents.d.ts +15 -15
- package/bundle/shared/dist/agents.js +6 -0
- package/bundle/shared/dist/index.d.ts +7 -7
- package/bundle/shared/package.json +1 -1
- package/dist/doctor.d.ts +7 -3
- package/dist/doctor.js +17 -12
- package/dist/doctor.test.js +9 -8
- package/package.json +1 -1
- package/bundle/server/static-ui/assets/index-UO4lHZw4.js +0 -60
|
@@ -1,14 +1,11 @@
|
|
|
1
1
|
// packages/server/src/capacity/claude-local-cache.ts
|
|
2
2
|
//
|
|
3
|
-
// NOT-268: Claude account capacity — local-first source ladder with a
|
|
4
|
-
//
|
|
3
|
+
// NOT-268: Claude account capacity — local-first source ladder with a free
|
|
4
|
+
// `/usage` refresh (corrected 2026-09-27; see below for what changed and why).
|
|
5
5
|
//
|
|
6
6
|
// Goal: keep Claude 5H/1W useful between Dealer runs. Production's last
|
|
7
7
|
// Dealer-observed sample can be days old (no recent Dealer-managed Claude
|
|
8
|
-
// run)
|
|
9
|
-
// `cachedUsageUtilization` key of `~/.claude.json` on every run — including
|
|
10
|
-
// interactive runs outside Dealer. So the freshest valid observation wins
|
|
11
|
-
// from this ladder:
|
|
8
|
+
// run). The ladder:
|
|
12
9
|
//
|
|
13
10
|
// 1. Existing Dealer `rate_limit_event` ingestion (claude-events.ts — kept
|
|
14
11
|
// as-is, session-end, per-window newer-wins).
|
|
@@ -16,9 +13,37 @@
|
|
|
16
13
|
// `~/.claude.json` (this module — free, ingested on every capacity
|
|
17
14
|
// read; only that subtree is ever parsed, the rest of the config —
|
|
18
15
|
// accountUuid, email, credentials, projects — is never retained).
|
|
19
|
-
// 3. One minimal
|
|
20
|
-
//
|
|
21
|
-
//
|
|
16
|
+
// 3. One minimal FREE refresh when every valid 5H/1W observation is older
|
|
17
|
+
// than 60 minutes: `claude -p "/usage"`. This is the default behavior.
|
|
18
|
+
// Set `AGENT_DEALER_CLAUDE_CAPACITY_REFRESH=off` to disable it entirely
|
|
19
|
+
// — reading capacity then never spawns Claude. Unrecognized values also
|
|
20
|
+
// fail closed (stay disabled).
|
|
21
|
+
//
|
|
22
|
+
// 2026-09-27 correction (what changed and why): the original design for rung
|
|
23
|
+
// 3 spawned a real one-turn model prompt (`claude -p "Reply with exactly: ok"
|
|
24
|
+
// --model haiku ...`), assuming any `claude` invocation would emit a
|
|
25
|
+
// `rate_limit_event` and/or refresh rung 2's cache file. A live proof against
|
|
26
|
+
// a real account falsified both assumptions: three attempts each overspent
|
|
27
|
+
// the $0.01 cap on ambient context alone (~11K cache-creation tokens before
|
|
28
|
+
// the model could even answer), none emitted a `rate_limit_event`, and the
|
|
29
|
+
// cache file was untouched afterward — a decompiled trace of the installed
|
|
30
|
+
// CLI showed the cache write (`Juo()`) lives behind the interactive
|
|
31
|
+
// usage/plan-limits fetch, not the ordinary chat-turn path. That path
|
|
32
|
+
// initially looked unreachable from a headless probe, until a second live
|
|
33
|
+
// test found the actual trigger: passing the **local slash-command**
|
|
34
|
+
// `/usage` as the `-p` prompt. Claude Code resolves `/usage` as a local
|
|
35
|
+
// command — no model call, `total_cost_usd: 0`, ~300ms — and its NDJSON
|
|
36
|
+
// output carries the exact structured result Dealer needs directly, at
|
|
37
|
+
// `usage_report.rate_limits.limits[]` (`{ kind: "session"|"weekly_all",
|
|
38
|
+
// percent, resets_at, ... }` — the same shape rung 2 already parses from the
|
|
39
|
+
// cache file's `limits[]`). It also performs the identical write rung 2
|
|
40
|
+
// reads, confirmed by `cachedUsageUtilization.fetchedAtMs` changing on every
|
|
41
|
+
// run. Verified reproducible across repeated live invocations. This is a
|
|
42
|
+
// documented, user-facing CLI command (listed in the session's own
|
|
43
|
+
// `slash_commands`), not an internal/undocumented surface, and it costs
|
|
44
|
+
// nothing — so the refresh is back to default-on, and the diagnostic log
|
|
45
|
+
// records real ~$0 outcomes instead of a `no_windows` failure streak. See
|
|
46
|
+
// docs/RUNTIME_CAPACITY.md for the full writeup and both live proofs.
|
|
22
47
|
//
|
|
23
48
|
// Both file sources write the same `claude_unified_five_hour` /
|
|
24
49
|
// `claude_unified_seven_day` window keys through the shared
|
|
@@ -26,7 +51,7 @@
|
|
|
26
51
|
// automatic per window and an older cache can never clobber a newer event
|
|
27
52
|
// (or vice versa). `source` stays `observed_event` for both; `evidenceRef`
|
|
28
53
|
// tells them apart (`claude-session:unified-windows` vs
|
|
29
|
-
// `claude-cache:cachedUsageUtilization` vs `claude-probe:
|
|
54
|
+
// `claude-cache:cachedUsageUtilization` vs `claude-probe:usage-command`).
|
|
30
55
|
//
|
|
31
56
|
// Local-cache safety: only the `cachedUsageUtilization` subtree
|
|
32
57
|
// (`fetchedAtMs`, `utilization.five_hour`, `utilization.seven_day`,
|
|
@@ -42,34 +67,37 @@
|
|
|
42
67
|
// in (1, 100] read as percent. Anything else is malformed — rejected.
|
|
43
68
|
// - `limits[]`: `{ kind|group: "session"|"weekly_all", percent: <0–100>,
|
|
44
69
|
// resets_at: <ISO-8601>, ... }`. Model-specific and overage entries are
|
|
45
|
-
// dropped — only the account-wide pair is ever normalized.
|
|
70
|
+
// dropped — only the account-wide pair is ever normalized. The `/usage`
|
|
71
|
+
// probe's `usage_report.rate_limits.limits[]` is this exact same shape and
|
|
72
|
+
// shares the same role mapping (`limitEntryRole`).
|
|
46
73
|
//
|
|
47
|
-
// Probe argv (verified live at 2.1.283
|
|
48
|
-
//
|
|
49
|
-
//
|
|
50
|
-
//
|
|
51
|
-
//
|
|
52
|
-
//
|
|
53
|
-
//
|
|
54
|
-
//
|
|
55
|
-
//
|
|
56
|
-
//
|
|
57
|
-
//
|
|
74
|
+
// Probe argv (verified live at 2.1.283):
|
|
75
|
+
// - `-p "/usage"` — the fixed local slash-command; never interpolated, never
|
|
76
|
+
// a natural-language prompt. Resolved entirely locally: no model call, no
|
|
77
|
+
// tokens, no cost.
|
|
78
|
+
// - `--model haiku` + `--max-turns 1` + `--tools ""` — structural safeguards
|
|
79
|
+
// kept even though `/usage` never reaches the model on 2.1.283: verified
|
|
80
|
+
// live that they do not break local resolution, and they bound the
|
|
81
|
+
// (unobserved) case where a different CLI build makes `/usage` fall
|
|
82
|
+
// through to a real prompt (2026-09-27 review hardening — see
|
|
83
|
+
// `buildClaudeProbeArgv`).
|
|
84
|
+
// - `--strict-mcp-config` (with no `--mcp-config`) — no MCP servers loaded.
|
|
58
85
|
// - `--no-session-persistence` — the probe leaves no resumable session.
|
|
59
86
|
// - `--output-format stream-json` (+ `--verbose`, matching Dealer's own
|
|
60
|
-
// stream-json parsing) so `
|
|
87
|
+
// stream-json parsing) so the `usage_report` and any `rate_limit_event`s
|
|
88
|
+
// can be ingested.
|
|
89
|
+
// - `--max-budget-usd 0.01` — defensive belt-and-braces only: normal cost is
|
|
90
|
+
// exactly $0. Not sufficient alone (ambient context can blow the cap
|
|
91
|
+
// before the check fires — see live proof #1 below), so the result is
|
|
92
|
+
// also checked for the local-command marker and exactly-zero cost before
|
|
93
|
+
// it counts as success (`runClaudeCapacityProbe`).
|
|
61
94
|
// - `--bare` is deliberately NOT used: it restricts auth to
|
|
62
95
|
// ANTHROPIC_API_KEY/apiKeyHelper and would bypass the account's OAuth
|
|
63
|
-
// login — the probe must
|
|
96
|
+
// login — the probe must read the capacity of the account it measures.
|
|
64
97
|
// - Ambient settings are kept (auth must resolve); no worktree is created
|
|
65
98
|
// (cwd is the OS temp dir) and nothing touches Dealer workflow/session
|
|
66
99
|
// rows, worktrees, commits, PRs, or queue events — the probe spawns
|
|
67
100
|
// `claude` directly, never through the coordinator.
|
|
68
|
-
//
|
|
69
|
-
// If a live proof ever shows the minimal probe does not reliably emit 5H/1W
|
|
70
|
-
// (neither in its stream nor via the cache side effect), the probe is a
|
|
71
|
-
// recurring paid no-op: disable the fallback and revise this ticket instead of
|
|
72
|
-
// shipping it. The `no_windows` diagnostic below exists to make that visible.
|
|
73
101
|
import { spawn } from "node:child_process";
|
|
74
102
|
import fs from "node:fs";
|
|
75
103
|
import os from "node:os";
|
|
@@ -82,7 +110,12 @@ import { configuredCapacityRuntimes } from "./service.js";
|
|
|
82
110
|
import { CLAUDE_RUNTIME, extractClaudeCapacityFromEvents, normalizeClaudeResetsAt, recordClaudeCapacityFromEvents, recordClaudeWindowReadings, } from "./claude-events.js";
|
|
83
111
|
/** Override for the Claude cache file (tests, smoke). */
|
|
84
112
|
export const CLAUDE_CACHE_FILE_ENV = "AGENT_DEALER_CLAUDE_CACHE_FILE";
|
|
85
|
-
/**
|
|
113
|
+
/**
|
|
114
|
+
* Refresh setting. Default ON (unset/empty, or the historical explicit
|
|
115
|
+
* `paid-after-1h` value — kept accepted for backward compat with 1.2.4
|
|
116
|
+
* configs, though the refresh is no longer paid). Only `off` (or any other
|
|
117
|
+
* unrecognized value) disables it, failing closed on typos.
|
|
118
|
+
*/
|
|
86
119
|
export const CLAUDE_CAPACITY_REFRESH_ENV = "AGENT_DEALER_CLAUDE_CAPACITY_REFRESH";
|
|
87
120
|
export const CLAUDE_CAPACITY_REFRESH_PAID_VALUE = "paid-after-1h";
|
|
88
121
|
export const CLAUDE_CAPACITY_REFRESH_OFF_VALUE = "off";
|
|
@@ -91,14 +124,16 @@ export const CLAUDE_CAPACITY_PROBE_TIMEOUT_ENV = "AGENT_DEALER_CLAUDE_PROBE_TIME
|
|
|
91
124
|
export const CLAUDE_PROBE_TIMEOUT_MS_DEFAULT = 90_000;
|
|
92
125
|
/** Static evidence pointers — never credentials or raw payloads. */
|
|
93
126
|
export const CLAUDE_CACHE_EVIDENCE_REF = "claude-cache:cachedUsageUtilization";
|
|
94
|
-
export const CLAUDE_PROBE_EVIDENCE_REF = "claude-probe:
|
|
95
|
-
/**
|
|
127
|
+
export const CLAUDE_PROBE_EVIDENCE_REF = "claude-probe:usage-command";
|
|
128
|
+
/** Fixed local slash-command — asserted in tests; never interpolated, never
|
|
129
|
+
* sent to the model (Claude Code resolves it locally, no API call). */
|
|
130
|
+
export const CLAUDE_PROBE_PROMPT = "/usage";
|
|
131
|
+
/** Cheapest supported model alias — structural safeguard only; see module
|
|
132
|
+
* header on why this stays even though `/usage` never reaches the model. */
|
|
96
133
|
export const CLAUDE_PROBE_MODEL = "haiku";
|
|
97
|
-
/**
|
|
98
|
-
export const CLAUDE_PROBE_PROMPT = "Reply with exactly: ok";
|
|
99
|
-
/** Hard spend cap for the probe (USD). */
|
|
134
|
+
/** Defensive spend cap (USD) — normal cost is $0; see module header. */
|
|
100
135
|
export const CLAUDE_PROBE_MAX_BUDGET_USD = 0.01;
|
|
101
|
-
/** A 5H/1W observation newer than this suppresses the
|
|
136
|
+
/** A 5H/1W observation newer than this suppresses the refresh. */
|
|
102
137
|
export const CLAUDE_PROBE_STALE_AFTER_MS = 60 * 60 * 1000;
|
|
103
138
|
/** Minimum gap between probe attempts (per account); failures back off. */
|
|
104
139
|
export const CLAUDE_PROBE_ATTEMPT_COOLDOWN_MS = 60 * 60 * 1000;
|
|
@@ -160,9 +195,23 @@ export function claudeProbeTimeoutMs() {
|
|
|
160
195
|
return CLAUDE_PROBE_TIMEOUT_MS_DEFAULT;
|
|
161
196
|
}
|
|
162
197
|
/**
|
|
163
|
-
* Fixed probe argv. Pinned by tests: the fixed
|
|
164
|
-
* model,
|
|
165
|
-
* JSON, ≤$0.01 budget.
|
|
198
|
+
* Fixed probe argv. Pinned by tests: the fixed local slash-command, cheapest
|
|
199
|
+
* model, hard one-turn bound, no tools, no MCP, no session persistence,
|
|
200
|
+
* stream JSON, defensive ≤$0.01 budget.
|
|
201
|
+
*
|
|
202
|
+
* Reviewer-requested hardening (2026-09-27, PR #165): `--model`/`--max-turns`/
|
|
203
|
+
* `--tools` were originally dropped as "meaningless" because `/usage` never
|
|
204
|
+
* reaches the model on 2.1.283 — but that is an empirical fact about one CLI
|
|
205
|
+
* version, not a contract. Verified live that keeping all three does not
|
|
206
|
+
* break local resolution (still `$0`, still `local_command: usage`), so they
|
|
207
|
+
* stay as structural bounds: if some other CLI build ever makes `/usage`
|
|
208
|
+
* fall through to a real prompt, the model is the cheapest alias, gets
|
|
209
|
+
* exactly one turn, and has no tools to call — the same belt-and-braces the
|
|
210
|
+
* original (abandoned) paid-turn design relied on. `--max-budget-usd` alone
|
|
211
|
+
* is not sufficient (this repo's own live proof showed ambient context can
|
|
212
|
+
* blow the cap before the check fires); see `runClaudeCapacityProbe` for the
|
|
213
|
+
* matching fail-closed checks on the result (local-command marker present,
|
|
214
|
+
* cost must be exactly $0).
|
|
166
215
|
*/
|
|
167
216
|
export function buildClaudeProbeArgv() {
|
|
168
217
|
return [
|
|
@@ -409,12 +458,14 @@ export function ingestClaudeLocalCache(nowMs = Date.now()) {
|
|
|
409
458
|
return recordClaudeWindowReadings(result.windows, CLAUDE_RUNTIME, nowMs);
|
|
410
459
|
}
|
|
411
460
|
// ---------------------------------------------------------------------------
|
|
412
|
-
//
|
|
461
|
+
// Free `/usage` refresh (default on)
|
|
413
462
|
// ---------------------------------------------------------------------------
|
|
414
463
|
/**
|
|
415
|
-
*
|
|
416
|
-
* `paid-after-1h` value is accepted explicitly
|
|
417
|
-
*
|
|
464
|
+
* Refresh defaults ON when the setting is absent/empty. The historical
|
|
465
|
+
* `paid-after-1h` value is still accepted explicitly (it now just means
|
|
466
|
+
* "enabled" — the refresh costs nothing, see module header); `off` and
|
|
467
|
+
* unrecognized values disable it so a typo never silently changes the
|
|
468
|
+
* configured policy.
|
|
418
469
|
*/
|
|
419
470
|
export function isClaudePaidFallbackEnabled() {
|
|
420
471
|
const setting = process.env[CLAUDE_CAPACITY_REFRESH_ENV];
|
|
@@ -493,6 +544,70 @@ function probeCostFromEvents(events) {
|
|
|
493
544
|
}
|
|
494
545
|
return null;
|
|
495
546
|
}
|
|
547
|
+
/**
|
|
548
|
+
* True only when the stream shows `/usage` actually resolved as Claude
|
|
549
|
+
* Code's local command (`local_command_run.command === "usage"`) — the
|
|
550
|
+
* structural marker that the probe never reached the model. Absence means
|
|
551
|
+
* the CLI build in use does not behave like 2.1.283; the caller must not
|
|
552
|
+
* trust any windows the stream happens to carry (`not_local_command`).
|
|
553
|
+
*/
|
|
554
|
+
function hasLocalUsageCommandMarker(events) {
|
|
555
|
+
return events.some((e) => {
|
|
556
|
+
const run = e.local_command_run;
|
|
557
|
+
if (!run || typeof run !== "object")
|
|
558
|
+
return false;
|
|
559
|
+
return run.command === "usage";
|
|
560
|
+
});
|
|
561
|
+
}
|
|
562
|
+
/**
|
|
563
|
+
* Extract account-wide 5H/1W readings directly from a `/usage` probe's
|
|
564
|
+
* stream: the local-command result event carries `usage_report.rate_limits.
|
|
565
|
+
* limits[]` in the exact shape (and role mapping via `limitEntryRole`) as
|
|
566
|
+
* the cache file's `limits[]` — see module header. This is the primary,
|
|
567
|
+
* most-reliable success signal for the probe (present on every successful
|
|
568
|
+
* `/usage` run, verified live); the rate_limit_event and cache-re-read
|
|
569
|
+
* checks below stay as additional, non-exclusive corroboration. Returns
|
|
570
|
+
* null when no usable account-wide window is present — never fabricated.
|
|
571
|
+
*/
|
|
572
|
+
export function extractClaudeUsageReportLimits(events, nowMs, evidenceRef = CLAUDE_PROBE_EVIDENCE_REF) {
|
|
573
|
+
let limits = null;
|
|
574
|
+
let eventTimestamp = null;
|
|
575
|
+
for (const e of events) {
|
|
576
|
+
const report = e.usage_report;
|
|
577
|
+
if (!report || typeof report !== "object")
|
|
578
|
+
continue;
|
|
579
|
+
const rateLimits = report.rate_limits;
|
|
580
|
+
if (!rateLimits || typeof rateLimits !== "object")
|
|
581
|
+
continue;
|
|
582
|
+
const l = rateLimits.limits;
|
|
583
|
+
if (Array.isArray(l)) {
|
|
584
|
+
limits = l;
|
|
585
|
+
const ts = e.timestamp;
|
|
586
|
+
eventTimestamp = typeof ts === "string" ? ts : null;
|
|
587
|
+
}
|
|
588
|
+
}
|
|
589
|
+
if (!limits)
|
|
590
|
+
return null;
|
|
591
|
+
const parsedTsMs = eventTimestamp !== null ? Date.parse(eventTimestamp) : Number.NaN;
|
|
592
|
+
const observedMs = Number.isFinite(parsedTsMs) ? Math.min(parsedTsMs, nowMs) : nowMs;
|
|
593
|
+
const observedAt = new Date(observedMs).toISOString();
|
|
594
|
+
const byRole = new Map();
|
|
595
|
+
for (const item of limits) {
|
|
596
|
+
if (!item || typeof item !== "object" || Array.isArray(item))
|
|
597
|
+
continue;
|
|
598
|
+
const w = item;
|
|
599
|
+
const name = pickString([w.kind, w.group, w.name, w.window, w.bucket, w.key, w.id]);
|
|
600
|
+
if (!name)
|
|
601
|
+
continue;
|
|
602
|
+
const role = limitEntryRole(name);
|
|
603
|
+
if (!role || byRole.has(role))
|
|
604
|
+
continue;
|
|
605
|
+
const reading = cacheWindowReading(role, item, observedAt, observedMs, evidenceRef);
|
|
606
|
+
if (reading)
|
|
607
|
+
byRole.set(role, reading);
|
|
608
|
+
}
|
|
609
|
+
return byRole.size > 0 ? [...byRole.values()] : null;
|
|
610
|
+
}
|
|
496
611
|
function rolesCoveredByReadings(readings) {
|
|
497
612
|
const out = new Set();
|
|
498
613
|
for (const w of readings) {
|
|
@@ -536,12 +651,18 @@ function appendProbeDiagnostic(entry) {
|
|
|
536
651
|
}
|
|
537
652
|
}
|
|
538
653
|
/**
|
|
539
|
-
* Run one minimal
|
|
540
|
-
*
|
|
541
|
-
*
|
|
542
|
-
*
|
|
543
|
-
*
|
|
544
|
-
*
|
|
654
|
+
* Run one minimal FREE `/usage` probe and ingest whatever 5H/1W it yields.
|
|
655
|
+
* Fails closed before touching any of it unless the stream both carries the
|
|
656
|
+
* local-command marker (`local_command_run.command === "usage"`) and reports
|
|
657
|
+
* exactly $0 cost — the structural proof `/usage` actually resolved locally
|
|
658
|
+
* rather than falling through to a real (billable) model turn. Once that
|
|
659
|
+
* holds, success ingests primarily the local command's own
|
|
660
|
+
* `usage_report.rate_limits.limits[]` (present on every successful run),
|
|
661
|
+
* plus any `rate_limit_event` the stream happens to carry and a re-read of
|
|
662
|
+
* the local cache (the probe run itself refreshes Claude's own cache file)
|
|
663
|
+
* as non-exclusive corroboration; success means the union covers both
|
|
664
|
+
* critical roles. Never throws; never creates Dealer workflow/session rows,
|
|
665
|
+
* worktrees, commits, PRs, or queue events.
|
|
545
666
|
*/
|
|
546
667
|
export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
|
|
547
668
|
const startedRealMs = Date.now();
|
|
@@ -584,7 +705,7 @@ export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
|
|
|
584
705
|
ts: new Date(nowMs).toISOString(),
|
|
585
706
|
event: "claude_capacity_probe",
|
|
586
707
|
trigger: "stale_60m",
|
|
587
|
-
|
|
708
|
+
probeCommand: CLAUDE_PROBE_PROMPT,
|
|
588
709
|
budgetUsd: CLAUDE_PROBE_MAX_BUDGET_USD,
|
|
589
710
|
...out,
|
|
590
711
|
});
|
|
@@ -602,7 +723,38 @@ export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
|
|
|
602
723
|
events = [];
|
|
603
724
|
}
|
|
604
725
|
const costUsd = probeCostFromEvents(events);
|
|
605
|
-
//
|
|
726
|
+
// Fail-closed structural checks (reviewer-requested, 2026-09-27, PR #165):
|
|
727
|
+
// the whole design rests on `/usage` resolving as a local command that
|
|
728
|
+
// never reaches the model. If either assumption is violated — no local-
|
|
729
|
+
// command marker in the stream, or cost not provably exactly $0 — reject
|
|
730
|
+
// the run entirely before ingesting anything from it, rather than trusting
|
|
731
|
+
// whatever windows a real (unexpected) model turn happened to produce.
|
|
732
|
+
// `costUsd !== 0` (not `> 0`) is deliberate: `null` — a missing or
|
|
733
|
+
// unparsable `total_cost_usd` — is not evidence of zero cost either, and
|
|
734
|
+
// must fail closed exactly like a confirmed charge (second review round).
|
|
735
|
+
// This also makes a future CLI-behavior change loud (backoff engages,
|
|
736
|
+
// logged as a distinct failure kind) instead of silently becoming a
|
|
737
|
+
// recurring paid probe again.
|
|
738
|
+
if (!hasLocalUsageCommandMarker(events))
|
|
739
|
+
return fail("not_local_command", { costUsd });
|
|
740
|
+
if (costUsd !== 0)
|
|
741
|
+
return fail("unexpected_cost", { costUsd });
|
|
742
|
+
// Primary signal: the `/usage` local command's own structured result.
|
|
743
|
+
// Present on every successful run (verified live) — most reliable source.
|
|
744
|
+
let usageReportRoles = new Set();
|
|
745
|
+
try {
|
|
746
|
+
const usageWindows = extractClaudeUsageReportLimits(events, nowMs);
|
|
747
|
+
if (usageWindows) {
|
|
748
|
+
usageReportRoles = rolesCoveredByReadings(usageWindows);
|
|
749
|
+
recordClaudeWindowReadings(usageWindows, CLAUDE_RUNTIME, nowMs);
|
|
750
|
+
}
|
|
751
|
+
}
|
|
752
|
+
catch {
|
|
753
|
+
// A broken stream must not fail the probe path — other checks below
|
|
754
|
+
// may still yield fresh rows.
|
|
755
|
+
}
|
|
756
|
+
// Secondary corroboration: any `rate_limit_event` the stream happens to
|
|
757
|
+
// carry (kept from the original NOT-248 event path; harmless if absent).
|
|
606
758
|
let streamRoles = new Set();
|
|
607
759
|
try {
|
|
608
760
|
const extracted = extractClaudeCapacityFromEvents(events, nowMs, nowMs);
|
|
@@ -636,13 +788,14 @@ export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
|
|
|
636
788
|
catch {
|
|
637
789
|
// Advisory — stream coverage below still counts.
|
|
638
790
|
}
|
|
639
|
-
const covered = new Set([...streamRoles, ...cacheRoles]);
|
|
791
|
+
const covered = new Set([...usageReportRoles, ...streamRoles, ...cacheRoles]);
|
|
640
792
|
if (spawnResult.exitCode !== 0 && covered.size === 0)
|
|
641
793
|
return fail("nonzero_exit", { costUsd });
|
|
642
794
|
if (!covered.has("five_hour") || !covered.has("weekly")) {
|
|
643
|
-
//
|
|
644
|
-
//
|
|
645
|
-
//
|
|
795
|
+
// A `/usage` run should always carry both roles directly in
|
|
796
|
+
// `usage_report.rate_limits.limits[]` (verified live) — a `no_windows`
|
|
797
|
+
// streak here means something changed upstream and is worth revisiting,
|
|
798
|
+
// but since the refresh is free it is not itself a cost problem.
|
|
646
799
|
return fail("no_windows", { costUsd });
|
|
647
800
|
}
|
|
648
801
|
const out = {
|
|
@@ -659,7 +812,7 @@ export async function runClaudeCapacityProbe(nowMs = Date.now(), opts = {}) {
|
|
|
659
812
|
ts: new Date(nowMs).toISOString(),
|
|
660
813
|
event: "claude_capacity_probe",
|
|
661
814
|
trigger: "stale_60m",
|
|
662
|
-
|
|
815
|
+
probeCommand: CLAUDE_PROBE_PROMPT,
|
|
663
816
|
budgetUsd: CLAUDE_PROBE_MAX_BUDGET_USD,
|
|
664
817
|
...out,
|
|
665
818
|
});
|
|
@@ -715,12 +868,12 @@ function probeCooldownMs() {
|
|
|
715
868
|
return Math.min(CLAUDE_PROBE_ATTEMPT_COOLDOWN_MS * 2 ** shift, CLAUDE_PROBE_BACKOFF_CAP_MS);
|
|
716
869
|
}
|
|
717
870
|
/**
|
|
718
|
-
* On-demand
|
|
719
|
-
* 5H/1W observation is newer than 60 minutes, run one minimal bounded
|
|
720
|
-
* (single-flight across concurrent readers; at most one attempt per
|
|
721
|
-
* per 60 minutes, backing off exponentially on failure). Explicitly
|
|
722
|
-
* is a strict no-op — no spawn
|
|
723
|
-
* workflow/session state or `runtime_availability`.
|
|
871
|
+
* On-demand free `/usage` refresh: when `claude_code` is configured and no
|
|
872
|
+
* valid 5H/1W observation is newer than 60 minutes, run one minimal bounded
|
|
873
|
+
* probe (single-flight across concurrent readers; at most one attempt per
|
|
874
|
+
* account per 60 minutes, backing off exponentially on failure). Explicitly
|
|
875
|
+
* disabled is a strict no-op — no spawn at all. Never throws, never touches
|
|
876
|
+
* Dealer workflow/session state or `runtime_availability`.
|
|
724
877
|
*/
|
|
725
878
|
export async function maybeProbeClaudeCapacity(nowMs = Date.now(), opts = {}) {
|
|
726
879
|
try {
|
|
@@ -775,8 +928,8 @@ export async function maybeProbeClaudeCapacity(nowMs = Date.now(), opts = {}) {
|
|
|
775
928
|
}
|
|
776
929
|
/**
|
|
777
930
|
* Full on-demand refresh for `GET /api/runtime-capacity`: ingest the free
|
|
778
|
-
* local cache first, then consider the
|
|
779
|
-
* throws. The route serves the stored snapshot regardless.
|
|
931
|
+
* local cache first, then consider the free `/usage` probe. Best-effort —
|
|
932
|
+
* never throws. The route serves the stored snapshot regardless.
|
|
780
933
|
*/
|
|
781
934
|
export async function refreshClaudeCapacityIfStale(nowMs = Date.now(), opts = {}) {
|
|
782
935
|
// Skip the file read entirely when no Claude account is configured:
|