agent-dealer 1.2.4 → 1.2.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/bundle/server/dist/adapters/agent-deck-bind.js +33 -3
  2. package/bundle/server/dist/adapters/agent-deck-bind.test.js +60 -2
  3. package/bundle/server/dist/adapters/agent-health.js +28 -6
  4. package/bundle/server/dist/adapters/agent-health.test.js +66 -11
  5. package/bundle/server/dist/adapters/git-worktree.js +187 -18
  6. package/bundle/server/dist/adapters/git-worktree.test.js +293 -0
  7. package/bundle/server/dist/adapters/github.js +110 -12
  8. package/bundle/server/dist/adapters/github.test.js +274 -3
  9. package/bundle/server/dist/adapters/muse-capability.js +356 -0
  10. package/bundle/server/dist/adapters/muse-capability.test.js +464 -0
  11. package/bundle/server/dist/capacity/claude-local-cache.js +266 -81
  12. package/bundle/server/dist/capacity/claude-local-cache.test.js +335 -36
  13. package/bundle/server/dist/capacity/muse-host.js +92 -32
  14. package/bundle/server/dist/capacity/muse-host.test.js +130 -4
  15. package/bundle/server/dist/coordinator/admission.test.js +85 -0
  16. package/bundle/server/dist/coordinator/args.js +4 -3
  17. package/bundle/server/dist/coordinator/checkpoint.js +5 -3
  18. package/bundle/server/dist/coordinator/checkpoint.test.js +23 -0
  19. package/bundle/server/dist/coordinator/commands.js +100 -12
  20. package/bundle/server/dist/coordinator/developer-effect.js +128 -16
  21. package/bundle/server/dist/coordinator/developer-effect.test.js +291 -12
  22. package/bundle/server/dist/coordinator/human-resolution.js +33 -1
  23. package/bundle/server/dist/coordinator/muse-developer.integration.test.js +291 -55
  24. package/bundle/server/dist/coordinator/muse-spawn.js +72 -164
  25. package/bundle/server/dist/coordinator/projection.js +1 -0
  26. package/bundle/server/dist/coordinator/prompts.js +25 -6
  27. package/bundle/server/dist/coordinator/prompts.test.js +36 -8
  28. package/bundle/server/dist/coordinator/routing.js +1 -0
  29. package/bundle/server/dist/coordinator/routing.test.js +20 -0
  30. package/bundle/server/dist/coordinator/spawn.js +9 -1
  31. package/bundle/server/dist/coordinator/worktree-cwd-guard.js +48 -0
  32. package/bundle/server/dist/coordinator/worktree-cwd-guard.test.js +83 -0
  33. package/bundle/server/dist/coordinator/worktree-owner-liveness.js +6 -1
  34. package/bundle/server/dist/coordinator/worktree-owner-liveness.test.js +2 -1
  35. package/bundle/server/dist/repository/human-actions.js +13 -0
  36. package/bundle/server/dist/routes/human-actions.js +10 -1
  37. package/bundle/server/dist/routes/human-actions.test.js +117 -1
  38. package/bundle/server/dist/routes/index.js +12 -5
  39. package/bundle/server/dist/routes/runtime-capacity.test.js +3 -2
  40. package/bundle/server/dist/routes/version.js +30 -0
  41. package/bundle/server/dist/routes/version.test.js +21 -0
  42. package/bundle/server/dist/runners/muse-config-core.js +63 -16
  43. package/bundle/server/dist/runners/muse-config.js +1 -1
  44. package/bundle/server/dist/runners/muse-config.test.js +135 -29
  45. package/bundle/server/dist/runners/spawn-cli.js +17 -3
  46. package/bundle/server/dist/runners/spawn-cli.test.js +59 -0
  47. package/bundle/server/package.json +2 -2
  48. package/bundle/server/static-ui/assets/{index-CYqRh_-S.css → index-BII-LgB8.css} +1 -1
  49. package/bundle/server/static-ui/assets/index-CoDzZcsz.js +60 -0
  50. package/bundle/server/static-ui/index.html +2 -2
  51. package/bundle/shared/dist/agents.d.ts +15 -15
  52. package/bundle/shared/dist/agents.js +6 -0
  53. package/bundle/shared/dist/index.d.ts +7 -7
  54. package/bundle/shared/package.json +1 -1
  55. package/dist/doctor.d.ts +7 -3
  56. package/dist/doctor.js +21 -13
  57. package/dist/doctor.test.js +9 -8
  58. package/dist/managed/index.d.ts +1 -1
  59. package/dist/managed/index.js +1 -1
  60. package/dist/managed/updater.d.ts +16 -6
  61. package/dist/managed/updater.js +28 -11
  62. package/dist/managed-update-restart.test.d.ts +1 -0
  63. package/dist/managed-update-restart.test.js +227 -0
  64. package/dist/ports.d.ts +6 -0
  65. package/dist/ports.js +10 -0
  66. package/dist/runtime-state.d.ts +10 -0
  67. package/dist/runtime-state.js +26 -0
  68. package/dist/start.js +67 -2
  69. package/dist/status.js +30 -2
  70. package/dist/update-check.js +3 -6
  71. package/package.json +1 -1
  72. package/bundle/server/static-ui/assets/index-UO4lHZw4.js +0 -60
@@ -1,8 +1,10 @@
1
1
  // packages/server/src/capacity/claude-local-cache.test.ts
2
2
  //
3
3
  // NOT-268: Claude capacity local-first ladder — local cache 5H/1W plus a
4
- // one-hour paid fallback. No test here performs a live provider request:
5
- // every probe spawn goes through an injected fake runner.
4
+ // free `/usage` refresh; NOT-281 retargeted the trigger to either window at
5
+ // least 14 minutes old (per-window, inside the 15-minute display freshness).
6
+ // No test here performs a live provider request: every probe spawn goes
7
+ // through an injected fake runner.
6
8
  import { test, before, beforeEach, afterEach } from "node:test";
7
9
  import assert from "node:assert/strict";
8
10
  import fs from "node:fs";
@@ -14,8 +16,8 @@ const { createAgent } = await import("../repository/agents.js");
14
16
  const { listCapacitySnapshots, clearAllCapacitySnapshots, } = await import("../repository/runtime-capacity.js");
15
17
  const { parseNdjson } = await import("../runners/stream-json.js");
16
18
  const { getRuntimeCapacitySnapshot } = await import("./service.js");
17
- const { recordClaudeCapacityFromEvents } = await import("./claude-events.js");
18
- const { CLAUDE_CACHE_FILE_ENV, CLAUDE_CAPACITY_REFRESH_ENV, buildClaudeProbeArgv, claudeCacheFilePath, extractClaudeCacheSubtree, ingestClaudeLocalCache, isClaudePaidFallbackEnabled, maybeProbeClaudeCapacity, newestValidClaudeObservationMs, parseClaudeCachedUtilization, probeDiagnosticLogPath, readClaudeLocalCache, refreshClaudeCapacityIfStale, resetClaudeCapacityRefreshState, runClaudeCapacityProbe, } = await import("./claude-local-cache.js");
19
+ const { recordClaudeCapacityFromEvents, recordClaudeWindowReadings } = await import("./claude-events.js");
20
+ const { CLAUDE_CACHE_FILE_ENV, CLAUDE_CAPACITY_REFRESH_ENV, CLAUDE_PROBE_STALE_AFTER_MS, buildClaudeProbeArgv, claudeCacheFilePath, extractClaudeCacheSubtree, ingestClaudeLocalCache, isClaudePaidFallbackEnabled, maybeProbeClaudeCapacity, newestValidClaudeObservationMs, newestValidClaudeObservationMsByRole, parseClaudeCachedUtilization, probeDiagnosticLogPath, readClaudeLocalCache, refreshClaudeCapacityIfStale, resetClaudeCapacityRefreshState, runClaudeCapacityProbe, } = await import("./claude-local-cache.js");
19
21
  const NOW_MS = Date.parse("2026-09-20T12:00:00.000Z");
20
22
  const FIVE_HOUR_RESET_SEC = Math.floor(NOW_MS / 1000) + 2 * 3600;
21
23
  const SEVEN_DAY_RESET_SEC = Math.floor(NOW_MS / 1000) + 3 * 24 * 3600;
@@ -42,17 +44,39 @@ function fullCacheFixture(fetchedAtMs) {
42
44
  },
43
45
  });
44
46
  }
45
- function probeStreamFixture(observedIso, costUsd = 0.0003) {
47
+ // Mirrors a real `claude -p "/usage"` stream at 2.1.283: a local-command
48
+ // result event carrying `usage_report.rate_limits.limits[]` (verified live
49
+ // 2026-09-27 — see claude-local-cache.ts module header), followed by the
50
+ // terminal `result` event. Real runs cost $0 (no model call); `costUsd`
51
+ // defaults to 0 but stays overridable for the cost-plumbing assertions.
52
+ function probeStreamFixture(observedIso, costUsd = 0) {
46
53
  return [
47
54
  JSON.stringify({
48
- type: "rate_limit_event",
55
+ type: "assistant",
49
56
  timestamp: observedIso,
50
- rate_limit_info: {
51
- status: "allowed",
52
- unifiedWindows: [
53
- { window: "five_hour", utilization: 0.2, resetsAt: FIVE_HOUR_RESET_SEC },
54
- { window: "seven_day", utilization: 0.4, resetsAt: SEVEN_DAY_RESET_SEC },
55
- ],
57
+ local_command_run: { command: "usage", args: "" },
58
+ usage_report: {
59
+ session: { total_cost_usd: 0 },
60
+ rate_limits: {
61
+ limits: [
62
+ {
63
+ kind: "session",
64
+ group: "session",
65
+ percent: 20,
66
+ resets_at: new Date(FIVE_HOUR_RESET_SEC * 1000).toISOString(),
67
+ severity: "normal",
68
+ is_active: false,
69
+ },
70
+ {
71
+ kind: "weekly_all",
72
+ group: "weekly",
73
+ percent: 40,
74
+ resets_at: new Date(SEVEN_DAY_RESET_SEC * 1000).toISOString(),
75
+ severity: "normal",
76
+ is_active: true,
77
+ },
78
+ ],
79
+ },
56
80
  },
57
81
  }),
58
82
  JSON.stringify({
@@ -66,6 +90,42 @@ function probeStreamFixture(observedIso, costUsd = 0.0003) {
66
90
  const throwingRunner = () => {
67
91
  throw new Error("probe runner must not be called while disabled/fresh");
68
92
  };
93
+ /**
94
+ * Seed the snapshot store with per-window ages (NOT-281): the shared cache
95
+ * fixture always stamps both windows together, so split-freshness cases
96
+ * need direct per-role rows. Resets stay in the future so rows count as
97
+ * valid observations at NOW_MS.
98
+ */
99
+ function seedWindowAges(fiveHourAgeMs, weeklyAgeMs, nowMs = NOW_MS) {
100
+ const readings = [];
101
+ const role = (five, ageMs) => ({
102
+ windowKey: five ? "claude_unified_five_hour" : "claude_unified_seven_day",
103
+ providerBucket: five ? "five_hour" : "seven_day",
104
+ durationMinutes: five ? 300 : 10080,
105
+ providerLabel: five ? "five_hour" : "seven_day",
106
+ usedValue: five ? 0.17 : 0.42,
107
+ usedUnit: "fraction",
108
+ usedFraction: five ? 0.17 : 0.42,
109
+ resetAt: new Date((five ? FIVE_HOUR_RESET_SEC : SEVEN_DAY_RESET_SEC) * 1000).toISOString(),
110
+ observedAt: new Date(nowMs - ageMs).toISOString(),
111
+ source: "observed_event",
112
+ evidenceRef: "test:split-window-seed",
113
+ criticalRole: five ? "five_hour" : "weekly",
114
+ });
115
+ if (fiveHourAgeMs !== null)
116
+ readings.push(role(true, fiveHourAgeMs));
117
+ if (weeklyAgeMs !== null)
118
+ readings.push(role(false, weeklyAgeMs));
119
+ assert.equal(recordClaudeWindowReadings(readings, "claude_code", nowMs), readings.length);
120
+ }
121
+ function successRunner(atMs = NOW_MS) {
122
+ return async () => ({
123
+ stdout: probeStreamFixture(new Date(atMs).toISOString()),
124
+ exitCode: 0,
125
+ timedOut: false,
126
+ spawnError: null,
127
+ });
128
+ }
69
129
  before(() => {
70
130
  migrate();
71
131
  createAgent({
@@ -263,7 +323,7 @@ test("a 2-hour-old cache ingests with its true age and reads expired, never live
263
323
  assert.equal(w.unavailableReason, "expired");
264
324
  }
265
325
  });
266
- test("paid fallback defaults on; explicit off and unrecognized values never spawn", async () => {
326
+ test("refresh defaults on; explicit off and unrecognized values never spawn", async () => {
267
327
  for (const value of [undefined, "", "paid-after-1h"]) {
268
328
  if (value === undefined)
269
329
  delete process.env[CLAUDE_CAPACITY_REFRESH_ENV];
@@ -277,17 +337,33 @@ test("paid fallback defaults on; explicit off and unrecognized values never spaw
277
337
  const outcome = await maybeProbeClaudeCapacity(NOW_MS, { runner: throwingRunner });
278
338
  assert.deepEqual(outcome, { probed: false, reason: "disabled" });
279
339
  }
280
- process.env[CLAUDE_CAPACITY_REFRESH_ENV] = "paid-after-1h";
340
+ delete process.env[CLAUDE_CAPACITY_REFRESH_ENV];
281
341
  assert.equal(isClaudePaidFallbackEnabled(), true);
282
342
  });
283
- test("default-on fallback suppresses a fresh sample; stale data triggers exactly one", async () => {
284
- delete process.env[CLAUDE_CAPACITY_REFRESH_ENV];
343
+ test("extractClaudeUsageReportLimits reads the /usage local-command result", async () => {
344
+ const { extractClaudeUsageReportLimits } = await import("./claude-local-cache.js");
345
+ const observedIso = new Date(NOW_MS).toISOString();
346
+ const events = probeStreamFixture(observedIso).split("\n").map((l) => JSON.parse(l));
347
+ const readings = extractClaudeUsageReportLimits(events, NOW_MS);
348
+ assert.ok(readings);
349
+ assert.equal(readings.length, 2);
350
+ const fiveHour = readings.find((w) => w.criticalRole === "five_hour");
351
+ const weekly = readings.find((w) => w.criticalRole === "weekly");
352
+ assert.equal(fiveHour.usedPercent, 20);
353
+ assert.equal(weekly.usedPercent, 40);
354
+ assert.equal(fiveHour.observedAt, observedIso);
355
+ assert.equal(fiveHour.evidenceRef, "claude-probe:usage-command");
356
+ // No usage_report anywhere in the stream — nothing fabricated.
357
+ assert.equal(extractClaudeUsageReportLimits([{ type: "result" }], NOW_MS), null);
358
+ });
359
+ test("default-on refresh suppresses a fresh sample; stale data triggers exactly one", async () => {
360
+ assert.equal(CLAUDE_PROBE_STALE_AFTER_MS, 14 * 60_000);
285
361
  fs.writeFileSync(cacheFile, fullCacheFixture(NOW_MS - 10 * 60_000));
286
362
  assert.equal(ingestClaudeLocalCache(NOW_MS), 2);
287
- assert.ok((newestValidClaudeObservationMs(NOW_MS) ?? 0) > NOW_MS - 60 * 60_000);
363
+ assert.ok((newestValidClaudeObservationMs(NOW_MS) ?? 0) > NOW_MS - CLAUDE_PROBE_STALE_AFTER_MS);
288
364
  const fresh = await maybeProbeClaudeCapacity(NOW_MS, { runner: throwingRunner });
289
365
  assert.deepEqual(fresh, { probed: false, reason: "fresh" });
290
- // All samples ≥60 minutes old: five concurrent readers share one probe.
366
+ // All samples past the 14-minute trigger: five concurrent readers share one probe.
291
367
  clearAllCapacitySnapshots();
292
368
  resetClaudeCapacityRefreshState();
293
369
  fs.writeFileSync(cacheFile, fullCacheFixture(NOW_MS - 61 * 60_000));
@@ -304,8 +380,122 @@ test("default-on fallback suppresses a fresh sample; stale data triggers exactly
304
380
  assert.equal(outcomes.filter((o) => o.reason === "shared").length, 4);
305
381
  assert.ok(outcomes.every((o) => o.probed && o.ok));
306
382
  });
307
- test("probe argv pins the fixed prompt, cheapest model, no tools/MCP, one turn, ≤$0.01", async () => {
308
- process.env[CLAUDE_CAPACITY_REFRESH_ENV] = "paid-after-1h";
383
+ test("a stale twin triggers one refresh even when its sibling is newer", async () => {
384
+ // NOT-281: the old newest-of-pair gate let one fresh window suppress the
385
+ // refresh while its sibling went stale. Freshness is per window now.
386
+ seedWindowAges(5 * 60_000, 20 * 60_000);
387
+ const byRole = newestValidClaudeObservationMsByRole(NOW_MS);
388
+ assert.equal(byRole.five_hour, NOW_MS - 5 * 60_000);
389
+ assert.equal(byRole.weekly, NOW_MS - 20 * 60_000);
390
+ let calls = 0;
391
+ const runner = async () => {
392
+ calls++;
393
+ return {
394
+ stdout: probeStreamFixture(new Date(NOW_MS).toISOString()),
395
+ exitCode: 0,
396
+ timedOut: false,
397
+ spawnError: null,
398
+ };
399
+ };
400
+ const outcome = await maybeProbeClaudeCapacity(NOW_MS, { runner });
401
+ assert.equal(calls, 1);
402
+ assert.equal(outcome.probed, true);
403
+ assert.equal(outcome.reason, "completed");
404
+ assert.equal(outcome.ok, true);
405
+ // On success both windows are fresh again …
406
+ const after = newestValidClaudeObservationMsByRole(NOW_MS);
407
+ assert.equal(after.five_hour, NOW_MS);
408
+ assert.equal(after.weekly, NOW_MS);
409
+ // … and the next API/UI poll (a minute later, well inside the ordinary
410
+ // 15-minute stale boundary) reports known values with no new spawn.
411
+ const poll = await maybeProbeClaudeCapacity(NOW_MS + 60_000, { runner: throwingRunner });
412
+ assert.deepEqual(poll, { probed: false, reason: "fresh" });
413
+ const snap = getRuntimeCapacitySnapshot(NOW_MS + 60_000);
414
+ const claude = snap.runtimes.find((r) => r.runtime === "claude_code");
415
+ assert.equal(claude.windows.find((w) => w.displayLabel === "5H").remainingPercent, 80);
416
+ assert.equal(claude.windows.find((w) => w.displayLabel === "1W").remainingPercent, 60);
417
+ });
418
+ test("a missing window triggers a refresh even when the sibling is newer", async () => {
419
+ seedWindowAges(5 * 60_000, null);
420
+ assert.deepEqual(newestValidClaudeObservationMsByRole(NOW_MS), {
421
+ five_hour: NOW_MS - 5 * 60_000,
422
+ weekly: null,
423
+ });
424
+ let calls = 0;
425
+ const runner = async () => {
426
+ calls++;
427
+ return {
428
+ stdout: probeStreamFixture(new Date(NOW_MS).toISOString()),
429
+ exitCode: 0,
430
+ timedOut: false,
431
+ spawnError: null,
432
+ };
433
+ };
434
+ const outcome = await maybeProbeClaudeCapacity(NOW_MS, { runner });
435
+ assert.equal(calls, 1);
436
+ assert.equal(outcome.ok, true);
437
+ assert.deepEqual(newestValidClaudeObservationMsByRole(NOW_MS), {
438
+ five_hour: NOW_MS,
439
+ weekly: NOW_MS,
440
+ });
441
+ });
442
+ test("the freshness boundary is 14 minutes: just under stays quiet, exactly at triggers", async () => {
443
+ seedWindowAges(13 * 60_000, 13 * 60_000);
444
+ const fresh = await maybeProbeClaudeCapacity(NOW_MS, { runner: throwingRunner });
445
+ assert.deepEqual(fresh, { probed: false, reason: "fresh" });
446
+ clearAllCapacitySnapshots();
447
+ resetClaudeCapacityRefreshState();
448
+ seedWindowAges(14 * 60_000, 0);
449
+ const stale = await maybeProbeClaudeCapacity(NOW_MS, { runner: successRunner() });
450
+ assert.equal(stale.probed, true);
451
+ assert.equal(stale.reason, "completed");
452
+ assert.equal(stale.ok, true);
453
+ });
454
+ test("a healthy refresh buys 14 minutes: no second attempt inside the interval", async () => {
455
+ seedWindowAges(20 * 60_000, 20 * 60_000);
456
+ let calls = 0;
457
+ const runner = async () => {
458
+ calls++;
459
+ return {
460
+ stdout: probeStreamFixture(new Date(NOW_MS).toISOString()),
461
+ exitCode: 0,
462
+ timedOut: false,
463
+ spawnError: null,
464
+ };
465
+ };
466
+ const first = await maybeProbeClaudeCapacity(NOW_MS, { runner });
467
+ assert.equal(first.ok, true);
468
+ assert.equal(calls, 1);
469
+ // A 5-second-poll cadence sees fresh rows and spawns nothing.
470
+ const poll = await maybeProbeClaudeCapacity(NOW_MS + 5_000, { runner: throwingRunner });
471
+ assert.deepEqual(poll, { probed: false, reason: "fresh" });
472
+ // Even if the rows go missing mid-interval (e.g. a reset passes), the
473
+ // attempt cooldown stamped by the healthy run holds — at most one
474
+ // refresh per 14 minutes, never a retry per poll.
475
+ clearAllCapacitySnapshots();
476
+ const missing = await maybeProbeClaudeCapacity(NOW_MS + 5 * 60_000, { runner: throwingRunner });
477
+ assert.deepEqual(missing, { probed: false, reason: "backoff" });
478
+ assert.equal(calls, 1);
479
+ // Past the interval the gate opens again.
480
+ const later = await maybeProbeClaudeCapacity(NOW_MS + 15 * 60_000, { runner });
481
+ assert.equal(later.probed, true);
482
+ assert.equal(calls, 2);
483
+ });
484
+ test("disabled refresh never spawns and stale readings stay honestly N/A", async () => {
485
+ process.env[CLAUDE_CAPACITY_REFRESH_ENV] = "off";
486
+ // 20 minutes old: past the 15-minute display freshness, inside expiry.
487
+ fs.writeFileSync(cacheFile, fullCacheFixture(NOW_MS - 20 * 60_000));
488
+ assert.equal(ingestClaudeLocalCache(NOW_MS), 2);
489
+ const outcome = await maybeProbeClaudeCapacity(NOW_MS, { runner: throwingRunner });
490
+ assert.deepEqual(outcome, { probed: false, reason: "disabled" });
491
+ const snap = getRuntimeCapacitySnapshot(NOW_MS);
492
+ const claude = snap.runtimes.find((r) => r.runtime === "claude_code");
493
+ for (const w of claude.windows) {
494
+ assert.equal(w.remainingPercent, null);
495
+ assert.equal(w.unavailableReason, "stale");
496
+ }
497
+ });
498
+ test("probe argv pins the fixed /usage command plus structural model-turn safeguards", async () => {
309
499
  fs.writeFileSync(cacheFile, fullCacheFixture(NOW_MS - 61 * 60_000));
310
500
  assert.equal(ingestClaudeLocalCache(NOW_MS), 2);
311
501
  let seenBin = "";
@@ -322,7 +512,7 @@ test("probe argv pins the fixed prompt, cheapest model, no tools/MCP, one turn,
322
512
  assert.equal(seenBin, "/fake/claude");
323
513
  assert.deepEqual(seenArgv, [
324
514
  "-p",
325
- "Reply with exactly: ok",
515
+ "/usage",
326
516
  "--model",
327
517
  "haiku",
328
518
  "--max-turns",
@@ -337,13 +527,14 @@ test("probe argv pins the fixed prompt, cheapest model, no tools/MCP, one turn,
337
527
  "--max-budget-usd",
338
528
  "0.01",
339
529
  ]);
340
- // Semantic pins behind the exact match: cheapest model, hard one-turn
341
- // bound, tools fully off, MCP restricted to none passed, stream JSON,
342
- // hard budget cap.
530
+ // Semantic pins behind the exact match: no MCP config passed, cheapest
531
+ // model + hard one-turn bound + no tools kept as structural safeguards
532
+ // (2026-09-27 review hardening) even though `/usage` never reaches the
533
+ // model on 2.1.283, plus the defensive budget cap.
534
+ assert.ok(!seenArgv.includes("--mcp-config"));
343
535
  assert.equal(seenArgv[seenArgv.indexOf("--model") + 1], "haiku");
344
536
  assert.equal(seenArgv[seenArgv.indexOf("--max-turns") + 1], "1");
345
537
  assert.equal(seenArgv[seenArgv.indexOf("--tools") + 1], "");
346
- assert.ok(!seenArgv.includes("--mcp-config"));
347
538
  const budget = Number(seenArgv[seenArgv.indexOf("--max-budget-usd") + 1]);
348
539
  assert.ok(Number.isFinite(budget) && budget <= 0.01);
349
540
  assert.equal(seenArgv.filter((a) => a === "-p").length, 1);
@@ -357,7 +548,7 @@ test("probe success updates both 5H and 1W and records cost without prompt/outpu
357
548
  assert.equal(ingestClaudeLocalCache(NOW_MS), 2);
358
549
  const marker = "probe-stream-marker-must-never-reach-log";
359
550
  const runner = async () => ({
360
- stdout: `${probeStreamFixture(new Date(NOW_MS).toISOString(), 0.0007)}\n${JSON.stringify({ type: "assistant", text: marker })}`,
551
+ stdout: `${probeStreamFixture(new Date(NOW_MS).toISOString())}\n${JSON.stringify({ type: "assistant", text: marker })}`,
361
552
  exitCode: 0,
362
553
  timedOut: false,
363
554
  spawnError: null,
@@ -365,28 +556,134 @@ test("probe success updates both 5H and 1W and records cost without prompt/outpu
365
556
  const result = await runClaudeCapacityProbe(NOW_MS, { runner });
366
557
  assert.equal(result.ok, true);
367
558
  assert.equal(result.failureKind, null);
368
- assert.equal(result.costUsd, 0.0007);
559
+ assert.equal(result.costUsd, 0);
369
560
  assert.deepEqual(result.windowsUpdated, ["claude_unified_five_hour", "claude_unified_seven_day"]);
370
561
  const rows = listCapacitySnapshots("claude_code");
371
562
  assert.equal(rows.find((r) => r.providerBucket === "five_hour").remainingPercent, 80);
372
563
  assert.equal(rows.find((r) => r.providerBucket === "seven_day").remainingPercent, 60);
373
564
  assert.equal(rows.find((r) => r.providerBucket === "five_hour").observedAt, new Date(NOW_MS).toISOString());
374
565
  const logText = fs.readFileSync(probeDiagnosticLogPath(), "utf8");
375
- assert.ok(logText.includes('"costUsd":0.0007'));
566
+ assert.ok(logText.includes('"costUsd":0'));
376
567
  assert.ok(!logText.includes("Reply with exactly"));
377
568
  assert.ok(!logText.includes(marker));
378
569
  });
379
- test("probe success via the cache side effect when the stream carries no windows", async () => {
570
+ test("probe rejects and ingests nothing when the result reports nonzero cost", async () => {
571
+ // Reviewer-requested hardening (PR #165): any charge means /usage did not
572
+ // resolve locally as expected — reject the whole run rather than trusting
573
+ // windows a real (unexpected) model turn happened to produce.
574
+ process.env[CLAUDE_CAPACITY_REFRESH_ENV] = "paid-after-1h";
575
+ fs.writeFileSync(cacheFile, fullCacheFixture(NOW_MS - 61 * 60_000));
576
+ assert.equal(ingestClaudeLocalCache(NOW_MS), 2);
577
+ const runner = async () => ({
578
+ stdout: probeStreamFixture(new Date(NOW_MS).toISOString(), 0.0007),
579
+ exitCode: 0,
580
+ timedOut: false,
581
+ spawnError: null,
582
+ });
583
+ const result = await runClaudeCapacityProbe(NOW_MS, { runner });
584
+ assert.equal(result.ok, false);
585
+ assert.equal(result.failureKind, "unexpected_cost");
586
+ assert.equal(result.costUsd, 0.0007);
587
+ assert.deepEqual(result.windowsUpdated, []);
588
+ // Last-good rows from before the rejected run are untouched.
589
+ const rows = listCapacitySnapshots("claude_code");
590
+ assert.equal(rows.find((r) => r.providerBucket === "five_hour").remainingPercent, 83);
591
+ });
592
+ test("probe rejects and ingests nothing when cost cannot be verified as zero", async () => {
593
+ // Reviewer-requested hardening, round 2 (PR #165): a stream with the
594
+ // local-command marker and well-formed windows, but a `result` event that
595
+ // omits `total_cost_usd` (costUsd === null), must fail closed exactly like
596
+ // a confirmed positive charge — null is not evidence of zero cost.
597
+ process.env[CLAUDE_CAPACITY_REFRESH_ENV] = "paid-after-1h";
598
+ fs.writeFileSync(cacheFile, fullCacheFixture(NOW_MS - 61 * 60_000));
599
+ assert.equal(ingestClaudeLocalCache(NOW_MS), 2);
600
+ const runner = async () => ({
601
+ stdout: [
602
+ JSON.stringify({
603
+ type: "assistant",
604
+ timestamp: new Date(NOW_MS).toISOString(),
605
+ local_command_run: { command: "usage", args: "" },
606
+ usage_report: {
607
+ rate_limits: {
608
+ limits: [
609
+ { kind: "session", percent: 20, resets_at: new Date(FIVE_HOUR_RESET_SEC * 1000).toISOString() },
610
+ { kind: "weekly_all", percent: 40, resets_at: new Date(SEVEN_DAY_RESET_SEC * 1000).toISOString() },
611
+ ],
612
+ },
613
+ },
614
+ }),
615
+ // No `result` event at all — total_cost_usd is unknowable.
616
+ ].join("\n"),
617
+ exitCode: 0,
618
+ timedOut: false,
619
+ spawnError: null,
620
+ });
621
+ const result = await runClaudeCapacityProbe(NOW_MS, { runner });
622
+ assert.equal(result.ok, false);
623
+ assert.equal(result.failureKind, "unexpected_cost");
624
+ assert.equal(result.costUsd, null);
625
+ assert.deepEqual(result.windowsUpdated, []);
626
+ const rows = listCapacitySnapshots("claude_code");
627
+ assert.equal(rows.find((r) => r.providerBucket === "five_hour").remainingPercent, 83);
628
+ });
629
+ test("probe rejects and ingests nothing when the local-command marker is absent", async () => {
630
+ // Reviewer-requested hardening (PR #165): a stream that never shows
631
+ // `/usage` resolving locally must not be trusted, even if it happens to
632
+ // carry a well-formed usage_report (e.g. a different CLI build echoing it
633
+ // from a real turn) or a fresh cache-file side effect.
634
+ process.env[CLAUDE_CAPACITY_REFRESH_ENV] = "paid-after-1h";
635
+ fs.writeFileSync(cacheFile, fullCacheFixture(NOW_MS - 61 * 60_000));
636
+ assert.equal(ingestClaudeLocalCache(NOW_MS), 2);
637
+ const runner = async () => {
638
+ fs.writeFileSync(cacheFile, fullCacheFixture(NOW_MS + 5_000));
639
+ return {
640
+ stdout: [
641
+ JSON.stringify({
642
+ type: "assistant",
643
+ timestamp: new Date(NOW_MS).toISOString(),
644
+ usage_report: {
645
+ rate_limits: { limits: [{ kind: "session", percent: 20, resets_at: null }] },
646
+ },
647
+ }),
648
+ JSON.stringify({ type: "result", is_error: false, total_cost_usd: 0 }),
649
+ ].join("\n"),
650
+ exitCode: 0,
651
+ timedOut: false,
652
+ spawnError: null,
653
+ };
654
+ };
655
+ const result = await runClaudeCapacityProbe(NOW_MS, { runner });
656
+ assert.equal(result.ok, false);
657
+ assert.equal(result.failureKind, "not_local_command");
658
+ assert.deepEqual(result.windowsUpdated, []);
659
+ const rows = listCapacitySnapshots("claude_code");
660
+ assert.equal(rows.find((r) => r.providerBucket === "five_hour").remainingPercent, 83);
661
+ });
662
+ test("probe success via the cache side effect when usage_report carries no windows", async () => {
380
663
  process.env[CLAUDE_CAPACITY_REFRESH_ENV] = "paid-after-1h";
381
664
  fs.writeFileSync(cacheFile, fullCacheFixture(NOW_MS - 61 * 60_000));
382
- // The stream is empty but the probe run itself refreshes Claude's own
383
- // cache file — the re-read must count as success.
665
+ // The local-command marker is present (so the run is trusted) and cost is
666
+ // $0, but usage_report carries no usable limits[] — the cache re-read
667
+ // (the probe run itself refreshes Claude's own cache file) still counts
668
+ // as corroborating success.
384
669
  const runner = async () => {
385
670
  // A real probe refreshes Claude's own cache file while it runs, so the
386
671
  // new `fetchedAtMs` is later than the pre-spawn `NOW_MS` stamp — the
387
672
  // post-spawn re-read must accept it, not reject it as future.
388
673
  fs.writeFileSync(cacheFile, fullCacheFixture(NOW_MS + 5_000));
389
- return { stdout: "", exitCode: 0, timedOut: false, spawnError: null };
674
+ return {
675
+ stdout: [
676
+ JSON.stringify({
677
+ type: "assistant",
678
+ timestamp: new Date(NOW_MS).toISOString(),
679
+ local_command_run: { command: "usage", args: "" },
680
+ }),
681
+ JSON.stringify({ type: "result", is_error: false, total_cost_usd: 0 }),
682
+ ].join("\n"),
683
+ exitCode: 0,
684
+ timedOut: false,
685
+ spawnError: null,
686
+ };
390
687
  };
391
688
  const result = await runClaudeCapacityProbe(NOW_MS, { runner });
392
689
  assert.equal(result.ok, true);
@@ -408,7 +705,9 @@ test("probe failure preserves last-good rows and enters bounded backoff", async
408
705
  const first = await maybeProbeClaudeCapacity(NOW_MS, { runner: failing });
409
706
  assert.equal(first.probed, true);
410
707
  assert.equal(first.ok, false);
411
- assert.equal(first.failureKind, "nonzero_exit");
708
+ // Unparsable stdout ("not-json") means no local-command marker is found —
709
+ // that check runs before the nonzero-exit fallback.
710
+ assert.equal(first.failureKind, "not_local_command");
412
711
  // Last-good rows are untouched by the failed probe (0.17 used → 83 left).
413
712
  const rows = listCapacitySnapshots("claude_code");
414
713
  assert.equal(rows.find((r) => r.providerBucket === "five_hour").remainingPercent, 83);
@@ -416,10 +715,10 @@ test("probe failure preserves last-good rows and enters bounded backoff", async
416
715
  const second = await maybeProbeClaudeCapacity(NOW_MS + 60_000, { runner: failing });
417
716
  assert.deepEqual(second, { probed: false, reason: "backoff" });
418
717
  assert.equal(calls, 1);
419
- // Exponential backoff: one failure doubles the 60-minute cooldown.
420
- const during = await maybeProbeClaudeCapacity(NOW_MS + 61 * 60_000, { runner: failing });
718
+ // Exponential backoff: one failure doubles the 14-minute cooldown (28m).
719
+ const during = await maybeProbeClaudeCapacity(NOW_MS + 15 * 60_000, { runner: failing });
421
720
  assert.deepEqual(during, { probed: false, reason: "backoff" });
422
- const after = await maybeProbeClaudeCapacity(NOW_MS + 121 * 60_000, { runner: failing });
721
+ const after = await maybeProbeClaudeCapacity(NOW_MS + 29 * 60_000, { runner: failing });
423
722
  assert.equal(after.probed, true);
424
723
  assert.equal(calls, 2);
425
724
  });