agent-dealer 1.2.3 → 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/bundle/server/dist/adapters/agent-health.js +25 -2
  2. package/bundle/server/dist/adapters/agent-health.test.js +55 -2
  3. package/bundle/server/dist/adapters/github.js +110 -12
  4. package/bundle/server/dist/adapters/github.test.js +274 -3
  5. package/bundle/server/dist/adapters/muse-capability.js +330 -0
  6. package/bundle/server/dist/adapters/muse-capability.test.js +378 -0
  7. package/bundle/server/dist/capacity/claude-local-cache.js +222 -63
  8. package/bundle/server/dist/capacity/claude-local-cache.test.js +179 -30
  9. package/bundle/server/dist/capacity/muse-host.js +111 -23
  10. package/bundle/server/dist/capacity/muse-host.test.js +125 -1
  11. package/bundle/server/dist/capacity/muse-lifecycle.test.js +2 -3
  12. package/bundle/server/dist/capacity/muse-probe.js +250 -0
  13. package/bundle/server/dist/capacity/muse-probe.test.js +183 -0
  14. package/bundle/server/dist/capacity/muse.js +16 -6
  15. package/bundle/server/dist/coordinator/admission.test.js +85 -0
  16. package/bundle/server/dist/coordinator/commands.js +35 -8
  17. package/bundle/server/dist/coordinator/developer-effect.js +40 -6
  18. package/bundle/server/dist/coordinator/developer-effect.test.js +121 -12
  19. package/bundle/server/dist/coordinator/human-resolution.js +31 -1
  20. package/bundle/server/dist/coordinator/muse-spawn.js +12 -2
  21. package/bundle/server/dist/coordinator/prompts.js +15 -0
  22. package/bundle/server/dist/coordinator/prompts.test.js +20 -0
  23. package/bundle/server/dist/coordinator/spawn.js +7 -0
  24. package/bundle/server/dist/coordinator/worktree-cwd-guard.js +48 -0
  25. package/bundle/server/dist/coordinator/worktree-cwd-guard.test.js +83 -0
  26. package/bundle/server/dist/routes/human-actions.js +10 -1
  27. package/bundle/server/dist/routes/human-actions.test.js +117 -1
  28. package/bundle/server/dist/routes/index.js +17 -14
  29. package/bundle/server/dist/routes/runtime-capacity.test.js +11 -3
  30. package/bundle/server/dist/runners/muse-serve-session.js +7 -6
  31. package/bundle/server/package.json +2 -2
  32. package/bundle/server/static-ui/assets/{index-CYqRh_-S.css → index-BII-LgB8.css} +1 -1
  33. package/bundle/server/static-ui/assets/index-Bq8wWpZm.js +60 -0
  34. package/bundle/server/static-ui/index.html +2 -2
  35. package/bundle/shared/dist/agents.d.ts +15 -15
  36. package/bundle/shared/dist/agents.js +6 -0
  37. package/bundle/shared/dist/index.d.ts +7 -7
  38. package/bundle/shared/package.json +1 -1
  39. package/dist/doctor.d.ts +22 -3
  40. package/dist/doctor.js +46 -12
  41. package/dist/doctor.test.js +41 -10
  42. package/package.json +1 -1
  43. package/bundle/server/static-ui/assets/index-UO4lHZw4.js +0 -60
@@ -0,0 +1,48 @@
1
+ // packages/server/src/coordinator/worktree-cwd-guard.ts
2
+ //
3
+ // NOT-273: defensive guard on the developer/reviewer spawn path. The worktree
4
+ // resolution (`resolveDeveloperWorktree` / `roleWorktreePathForResolution`) is
5
+ // correct today, but nothing asserts the resolved cwd before a child process
6
+ // is spawned — a regression there would run agent CLIs against the wrong
7
+ // directory (in the worst case the primary repo checkout itself, where a
8
+ // worker's `git add . && git commit` lands real commits on the checkout).
9
+ // Every real developer/reviewer spawn calls `assertWorktreeCwd` first and
10
+ // throws instead of spawning when the cwd is not a recognized
11
+ // coordinator-managed worktree path.
12
+ import path from "node:path";
13
+ /** Basename shape the worktree manager itself uses: `<sessionId>-<role>`. */
14
+ const ROLE_SUFFIX_RE = /-(developer|reviewer)$/;
15
+ /**
16
+ * Parent directory names under which coordinator-managed role worktrees live:
17
+ * - legacy layout: `<repo>/.agent-dealer-worktrees/<sessionId>-<role>`
18
+ * - managed layout (NOT-149): `<executionRoot>/worktrees/<identity>/<sessionId>-<role>`
19
+ * - dev-mode equivalent: `<dataDir>/worktrees/...`
20
+ */
21
+ const WORKTREE_PARENT_NAMES = new Set([".agent-dealer-worktrees", "worktrees"]);
22
+ /**
23
+ * True when `cwd` looks like a coordinator-managed role worktree path: its
24
+ * basename carries the `-developer`/`-reviewer` role suffix AND one of its
25
+ * parent directories is a recognized worktrees root. In particular the
26
+ * primary repo root itself never matches (no role suffix, no worktrees
27
+ * parent), nor does the ambient `process.cwd()` of a dev checkout.
28
+ */
29
+ export function isManagedWorktreeCwd(cwd) {
30
+ if (typeof cwd !== "string" || cwd.length === 0)
31
+ return false;
32
+ const normalized = path.normalize(cwd);
33
+ if (!ROLE_SUFFIX_RE.test(path.basename(normalized)))
34
+ return false;
35
+ const segments = normalized.split(path.sep).filter(Boolean);
36
+ return segments.slice(0, -1).some((segment) => WORKTREE_PARENT_NAMES.has(segment));
37
+ }
38
+ /**
39
+ * Throw instead of spawning when `cwd` is not a recognized managed-worktree
40
+ * path. Called at the top of `realDeveloperSpawn`/`realReviewerSpawn`, before
41
+ * any child process exists — the effects turn this pre-spawn throw into a
42
+ * `session_failed` "could not start" outcome, never a run against the repo.
43
+ */
44
+ export function assertWorktreeCwd(cwd) {
45
+ if (!isManagedWorktreeCwd(cwd)) {
46
+ throw new Error(`Refusing to spawn a worker outside a managed worktree (NOT-273): ${cwd}`);
47
+ }
48
+ }
@@ -0,0 +1,83 @@
1
+ // packages/server/src/coordinator/worktree-cwd-guard.test.ts
2
+ //
3
+ // NOT-273: the developer/reviewer spawn path refuses to spawn outside a
4
+ // recognized coordinator-managed worktree — in particular the primary repo
5
+ // root itself — before any child process exists.
6
+ import { test } from "node:test";
7
+ import assert from "node:assert/strict";
8
+ import fs from "node:fs";
9
+ import os from "node:os";
10
+ import path from "node:path";
11
+ const { assertWorktreeCwd, isManagedWorktreeCwd } = await import("./worktree-cwd-guard.js");
12
+ const { realDeveloperSpawn, realReviewerSpawn } = await import("./spawn.js");
13
+ const SESSION_ID = "00000000-0000-4000-8000-000000000001";
14
+ test("real worktree-shaped paths pass the guard", () => {
15
+ // Legacy layout: <repo>/.agent-dealer-worktrees/<sessionId>-<role>.
16
+ assert.equal(isManagedWorktreeCwd(`/repo/.agent-dealer-worktrees/${SESSION_ID}-developer`), true);
17
+ assert.equal(isManagedWorktreeCwd(`/repo/.agent-dealer-worktrees/${SESSION_ID}-reviewer`), true);
18
+ // Managed layout (NOT-149): <executionRoot>/worktrees/<identity>/<sessionId>-<role>.
19
+ assert.equal(isManagedWorktreeCwd(`/data/execution/worktrees/github.com/acme/app/${SESSION_ID}-developer`), true);
20
+ assert.equal(isManagedWorktreeCwd(`/data/execution/worktrees/github.com/acme/app/${SESSION_ID}-reviewer`), true);
21
+ // Dev-mode equivalent under a dev data dir.
22
+ assert.equal(isManagedWorktreeCwd(path.join(os.homedir(), ".agent-dealer-dev", "worktrees", "github.com", "acme", "app", `${SESSION_ID}-developer`)), true);
23
+ assert.doesNotThrow(() => assertWorktreeCwd(`/repo/.agent-dealer-worktrees/${SESSION_ID}-developer`));
24
+ });
25
+ test("the primary repo root and other non-worktree paths are rejected", () => {
26
+ const repoRoot = fs.mkdtempSync(path.join(os.tmpdir(), "dealer-guard-repo-"));
27
+ // A plain checkout dir (what the ambient process cwd is in a normal dev
28
+ // checkout). Deliberately synthetic, never `process.cwd()` itself: the
29
+ // runner's own cwd is environment-dependent and may itself be
30
+ // worktree-shaped (e.g. a `<uuid>-developer` dir under a `worktrees`
31
+ // parent), which the guard correctly accepts per the naming convention.
32
+ const checkoutDir = path.join(repoRoot, "checkout");
33
+ fs.mkdirSync(checkoutDir);
34
+ try {
35
+ assert.equal(isManagedWorktreeCwd(repoRoot), false, "primary repo root itself");
36
+ assert.equal(isManagedWorktreeCwd(checkoutDir), false, "ambient-checkout-shaped cwd");
37
+ assert.equal(isManagedWorktreeCwd(os.tmpdir()), false, "bare temp dir");
38
+ assert.equal(isManagedWorktreeCwd(path.join(os.tmpdir(), `${SESSION_ID}-developer`)), false, "role suffix without a worktrees parent");
39
+ assert.equal(isManagedWorktreeCwd("/repo/.agent-dealer-worktrees"), false, "worktrees root without a session dir");
40
+ assert.equal(isManagedWorktreeCwd(""), false, "empty cwd");
41
+ assert.throws(() => assertWorktreeCwd(repoRoot), /managed worktree/);
42
+ assert.throws(() => assertWorktreeCwd(checkoutDir), /managed worktree/);
43
+ }
44
+ finally {
45
+ fs.rmSync(repoRoot, { recursive: true, force: true });
46
+ }
47
+ });
48
+ test("realDeveloperSpawn refuses the primary repo root before any process is spawned", async () => {
49
+ const repoRoot = fs.mkdtempSync(path.join(os.tmpdir(), "dealer-guard-dev-"));
50
+ try {
51
+ await assert.rejects(realDeveloperSpawn({
52
+ sessionId: SESSION_ID,
53
+ runtime: "claude_code",
54
+ policy: { worktreeWrite: true },
55
+ model: null,
56
+ prompt: "Implement it",
57
+ cwd: repoRoot,
58
+ timeoutMs: 1000,
59
+ }), /managed worktree/, "repo-root cwd throws instead of spawning");
60
+ }
61
+ finally {
62
+ fs.rmSync(repoRoot, { recursive: true, force: true });
63
+ }
64
+ });
65
+ test("realReviewerSpawn refuses the primary repo root before any process is spawned", async () => {
66
+ // Synthetic repo-root stand-in, never `process.cwd()`: the runner's own
67
+ // cwd is environment-dependent and may itself be worktree-shaped.
68
+ const repoRoot = fs.mkdtempSync(path.join(os.tmpdir(), "dealer-guard-rev-"));
69
+ try {
70
+ await assert.rejects(realReviewerSpawn({
71
+ sessionId: SESSION_ID,
72
+ runtime: "claude_code",
73
+ policy: { worktreeWrite: false },
74
+ model: null,
75
+ prompt: "Review it",
76
+ cwd: repoRoot,
77
+ timeoutMs: 1000,
78
+ }), /managed worktree/, "repo-root cwd throws instead of spawning");
79
+ }
80
+ finally {
81
+ fs.rmSync(repoRoot, { recursive: true, force: true });
82
+ }
83
+ });
@@ -1,5 +1,6 @@
1
1
  import { getHumanAction, listOpenHumanActions } from "../repository/human-actions.js";
2
2
  import { resolveHumanActionAndAdvanceAsync } from "../coordinator/commands.js";
3
+ import { normalizeResolutionNote } from "../coordinator/human-resolution.js";
3
4
  import { triggerIssueReflect, resolveReflectionInteractionAction } from "../coordinator/reflect-trigger.js";
4
5
  import { resolveOutboundDeliveryAction } from "../queue/approve-deliver.js";
5
6
  export async function registerHumanActionRoutes(app) {
@@ -14,6 +15,12 @@ export async function registerHumanActionRoutes(app) {
14
15
  if (!resolvedBy || !choice) {
15
16
  return reply.status(400).send({ error: "resolvedBy and choice are required" });
16
17
  }
18
+ // NOT-272: optional human note for a product_scope_decision resolution — validated
19
+ // the same way as resolvedBy/choice above (present-but-not-a-string 400s, never 500s).
20
+ const normalizedNote = normalizeResolutionNote(body?.note);
21
+ if (!normalizedNote.ok) {
22
+ return reply.status(400).send({ error: normalizedNote.error });
23
+ }
17
24
  // Read before resolving — the issue id this action belongs to, needed for the reflect
18
25
  // trigger below and stable regardless of how resolution turns out.
19
26
  const action = getHumanAction(id);
@@ -45,7 +52,9 @@ export async function registerHumanActionRoutes(app) {
45
52
  }
46
53
  // Awaits undraft+merge when final_review:complete (NOT-102) — sync resolve alone would
47
54
  // only park at AUTO_MERGE_INTENT and leave the PR draft.
48
- const result = await resolveHumanActionAndAdvanceAsync(id, resolvedBy, choice);
55
+ const result = await resolveHumanActionAndAdvanceAsync(id, resolvedBy, choice, {
56
+ note: normalizedNote.note,
57
+ });
49
58
  if (!result.ok)
50
59
  return reply.status(result.code).send({ error: result.error });
51
60
  // Reflect is a best-effort network call to Agent Deck (health check + a sequential
@@ -12,7 +12,7 @@ const { createHumanAction, getHumanAction, listHumanActionsForIssue } = await im
12
12
  const { registerHumanActionRoutes } = await import("./human-actions.js");
13
13
  const { startWorkflow, applyCompletion } = await import("../coordinator/commands.js");
14
14
  const { ReviewerResult } = await import("../coordinator/reviewer-result.js");
15
- const { claimWorkItem } = await import("../repository/work-items.js");
15
+ const { claimWorkItem, getWorkItem } = await import("../repository/work-items.js");
16
16
  const { listArtifactsForIssue } = await import("../repository/artifacts-for-issue.js");
17
17
  const { createRun, getRun, transitionRun, addArtifact, updateRunFields } = await import("../repository/runs.js");
18
18
  const { pendingSendCount, getPendingOutboundDraft } = await import("../repository/outbound-drafts.js");
@@ -284,3 +284,119 @@ test("POST resolve 400s on a choice not valid for reflection_interaction_require
284
284
  assert.equal(res.statusCode, 400);
285
285
  await app.close();
286
286
  });
287
+ // NOT-272: seeds.
288
+ function seedPreStartScopeDecision() {
289
+ const issue = createIssue({ title: "Scope gated", repo: "acme/app", baseBranch: "main", developerAgentId: BUILTIN_AGENT_CLAUDE_ID, reviewerAgentId: BUILTIN_AGENT_CURSOR_ID, acceptanceCriteria: "Works", maxReviewRounds: 3, maxInfraAttempts: 3, source: "manual" });
290
+ const action = createHumanAction({
291
+ issueId: issue.id,
292
+ actionType: "product_scope_decision",
293
+ reason: "Scope unclear before start",
294
+ question: "Is this in scope?",
295
+ responseOptions: [{ choice: "resume", label: "Resume development" }],
296
+ });
297
+ return { issue, action };
298
+ }
299
+ /** Drives a real issue to a mid-workflow product_scope_decision via a reviewer
300
+ * escalate verdict carrying a productScopeQuestion (NOT-150 routing). */
301
+ async function seedMidWorkflowScopeDecision() {
302
+ const issue = createIssue({ title: "Scope escalated", repo: "acme/app", baseBranch: "main", developerAgentId: BUILTIN_AGENT_CLAUDE_ID, reviewerAgentId: BUILTIN_AGENT_CURSOR_ID, acceptanceCriteria: "Works", maxReviewRounds: 3, maxInfraAttempts: 3, source: "manual" });
303
+ const start = startWorkflow(issue.id);
304
+ assert.equal(start.ok, true);
305
+ const devItem = claimWorkItem(`scope-note-test-${issue.id}-dev`, { leaseMs: 60_000 });
306
+ await applyCompletion(devItem.id, devItem.leaseToken, {
307
+ kind: "clean_handoff",
308
+ branch: `issue-${issue.id}`,
309
+ headSha: "a".repeat(40),
310
+ baseSha: "b".repeat(40),
311
+ prNumber: 1,
312
+ prUrl: "https://gh/pr/1"
313
+ });
314
+ const reviewItem = claimWorkItem(`scope-note-test-${issue.id}-rev`, { leaseMs: 60_000 });
315
+ await applyCompletion(reviewItem.id, reviewItem.leaseToken, {
316
+ kind: "verdict",
317
+ result: ReviewerResult.parse({
318
+ verdict: "escalated",
319
+ baseSha: "b".repeat(40),
320
+ headSha: "a".repeat(40),
321
+ acceptanceCriteriaAssessment: "unclear scope",
322
+ evidenceAssessment: "fine",
323
+ findings: [],
324
+ risks: [],
325
+ productScopeQuestion: "Is the runner migration in scope?"
326
+ })
327
+ });
328
+ const humanAction = listHumanActionsForIssue(issue.id).find((a) => a.actionType === "product_scope_decision");
329
+ assert.ok(humanAction, "expected a product_scope_decision human action");
330
+ return { issueId: issue.id, actionId: humanAction.id };
331
+ }
332
+ test("NOT-272: POST resolve 400s (not 500) on a non-string note", async () => {
333
+ const app = await buildApp();
334
+ const { action } = seedPreStartScopeDecision();
335
+ const res = await app.inject({ method: "POST", url: `/api/human-actions/${action.id}/resolve`, payload: { resolvedBy: "yusuke", choice: "resume", note: 123 } });
336
+ assert.equal(res.statusCode, 400);
337
+ assert.match(res.json().error, /note must be a string/);
338
+ assert.equal(getHumanAction(action.id).status, "open", "a rejected resolve must leave the action open");
339
+ await app.close();
340
+ });
341
+ test("NOT-272: POST resolve 400s on an over-long note", async () => {
342
+ const app = await buildApp();
343
+ const { action } = seedPreStartScopeDecision();
344
+ const res = await app.inject({ method: "POST", url: `/api/human-actions/${action.id}/resolve`, payload: { resolvedBy: "yusuke", choice: "resume", note: "x".repeat(4001) } });
345
+ assert.equal(res.statusCode, 400);
346
+ assert.match(res.json().error, /at most 4000/);
347
+ assert.equal(getHumanAction(action.id).status, "open");
348
+ await app.close();
349
+ });
350
+ test("NOT-272: pre-start product_scope_decision resolve with a note persists it and carries it on the round-1 payload", async () => {
351
+ const app = await buildApp();
352
+ const { action } = seedPreStartScopeDecision();
353
+ const note = "The runner migration IS in scope — proceed with it.";
354
+ const res = await app.inject({ method: "POST", url: `/api/human-actions/${action.id}/resolve`, payload: { resolvedBy: "yusuke", choice: "resume", note: ` ${note} ` } });
355
+ assert.equal(res.statusCode, 200);
356
+ const body = res.json();
357
+ assert.equal(body.restarted, true);
358
+ assert.ok(body.nextWorkItemId);
359
+ const stored = getHumanAction(action.id);
360
+ assert.equal(stored.status, "resolved");
361
+ assert.deepEqual(JSON.parse(stored.resolutionJson), { choice: "resume", note }, "note round-trips exactly (trimmed)");
362
+ const payload = JSON.parse(getWorkItem(body.nextWorkItemId).payloadJson);
363
+ assert.equal(payload.scopeDecisionNote, note, "the very next round's payload carries the note");
364
+ await app.close();
365
+ });
366
+ test("NOT-272: mid-workflow product_scope_decision resolve with a note persists it and carries it on the next developer payload", async () => {
367
+ const app = await buildApp();
368
+ const { issueId, actionId } = await seedMidWorkflowScopeDecision();
369
+ const note = "Out of scope for this ticket — finish the runner work already here, nothing more.";
370
+ const res = await app.inject({ method: "POST", url: `/api/human-actions/${actionId}/resolve`, payload: { resolvedBy: "yusuke", choice: "resume", note } });
371
+ assert.equal(res.statusCode, 200);
372
+ const body = res.json();
373
+ assert.equal(body.issueStatus, "developing");
374
+ assert.ok(body.nextWorkItemId);
375
+ const stored = getHumanAction(actionId);
376
+ assert.deepEqual(JSON.parse(stored.resolutionJson), { choice: "resume", note });
377
+ const payload = JSON.parse(getWorkItem(body.nextWorkItemId).payloadJson);
378
+ assert.equal(payload.scopeDecisionNote, note);
379
+ assert.equal(getIssue(issueId).status, "developing");
380
+ await app.close();
381
+ });
382
+ test("NOT-272: resolving a product_scope_decision without a note stores exactly today's record and queues a noteless payload", async () => {
383
+ const app = await buildApp();
384
+ const { action } = seedPreStartScopeDecision();
385
+ const res = await app.inject({ method: "POST", url: `/api/human-actions/${action.id}/resolve`, payload: { resolvedBy: "yusuke", choice: "resume" } });
386
+ assert.equal(res.statusCode, 200);
387
+ const body = res.json();
388
+ const stored = getHumanAction(action.id);
389
+ assert.equal(stored.resolutionJson, JSON.stringify({ choice: "resume" }));
390
+ const payload = JSON.parse(getWorkItem(body.nextWorkItemId).payloadJson);
391
+ assert.ok(!("scopeDecisionNote" in payload), "no new payload key without a note");
392
+ await app.close();
393
+ });
394
+ test("NOT-272: a note sent with any other action type is dropped, never stored", async () => {
395
+ const app = await buildApp();
396
+ const { actionId } = await seedRealIssueAwaitingFinalReview();
397
+ const res = await app.inject({ method: "POST", url: `/api/human-actions/${actionId}/resolve`, payload: { resolvedBy: "yusuke", choice: "close", note: "should not persist" } });
398
+ assert.equal(res.statusCode, 200);
399
+ const stored = getHumanAction(actionId);
400
+ assert.equal(stored.resolutionJson, JSON.stringify({ choice: "close" }));
401
+ await app.close();
402
+ });
@@ -8,7 +8,7 @@ import { getLinearUsageSnapshot } from "../adapters/linear-graphql.js";
8
8
  import { listRuntimeModels } from "../runners/models.js";
9
9
  import { configuredCapacityRuntimes, getRuntimeCapacitySnapshot } from "../capacity/service.js";
10
10
  import { refreshClaudeCapacityIfStale } from "../capacity/claude-local-cache.js";
11
- import { maybeRefreshMuseCapacityFromHost } from "../capacity/muse-host.js";
11
+ import { refreshMuseCapacityIfStale } from "../capacity/muse-probe.js";
12
12
  import { refreshCodexCapacityIfStale } from "../capacity/codex-app-server.js";
13
13
  import { getCursorTeamBillingSnapshot, refreshCursorTeamBillingIfStale, } from "../capacity/cursor-team.js";
14
14
  import { getCursorIndividualBillingSnapshot, refreshCursorIndividualBillingIfStale, refreshCursorIndividualCapacityIfStale, } from "../capacity/cursor-individual.js";
@@ -63,13 +63,12 @@ export async function registerRoutes(app) {
63
63
  // NOT-245: provider-neutral capacity read model — one entry per configured
64
64
  // runtime account with its windows, freshness, and explicit unavailable
65
65
  // reasons. Normalized snapshots only; evidence stays server-side.
66
- // NOT-270: the only production trigger for Muse capacity — a throttled
67
- // (default 5 min), bounded, best-effort refresh through the server-owned
68
- // long-lived `muse serve` host when `muse_code` is configured. The
69
- // refresh runs in the background without blocking the read: GET serves
70
- // the last-known snapshot immediately so a slow or hanging host (bounded
71
- // by the capacity timeout) can never stall the Agents page. Failures
72
- // preserve last-good rows as N/A and never fail the read.
66
+ // Muse capacity: first take the free throttled read from the server-owned
67
+ // execution host. If there is still no complete observation under an hour
68
+ // old, the default-on bounded fallback runs one minimal turn on a dedicated
69
+ // restricted host and reads 5H/1W from that same host. This stays in the
70
+ // background: GET serves last-good immediately and never waits on model
71
+ // work. `AGENT_DEALER_MUSE_CAPACITY_REFRESH=off` disables paid fallback.
73
72
  // NOT-246: on-demand Codex refresh — when the stored Codex snapshot is
74
73
  // stale the read triggers one bounded, non-billable App Server poll
75
74
  // (single-flight, never health rows, never throws); fresh snapshots and
@@ -77,11 +76,15 @@ export async function registerRoutes(app) {
77
76
  // subprocess.
78
77
  // NOT-268: Claude local-first ladder — one background refresh ingests
79
78
  // Claude Code's own free cache (plain file read, never a spawn) and then
80
- // considers the paid probe without blocking the read: a bounded Haiku
81
- // probe fires at most once per hour and only under the explicit
82
- // `AGENT_DEALER_CLAUDE_CAPACITY_REFRESH=paid-after-1h` opt-in when every
83
- // valid 5H/1W observation is older than 60 minutes. Disabled is a strict
84
- // no-op (no spawn, no spend). Failures never break the read below.
79
+ // considers the free `/usage` refresh without blocking the read: a
80
+ // bounded `claude -p "/usage"` call fires at most once per hour, only
81
+ // when every valid 5H/1W observation is older than 60 minutes, and only
82
+ // when enabled (default on; `AGENT_DEALER_CLAUDE_CAPACITY_REFRESH=off`
83
+ // disables it). `/usage` is a local slash-command — no model call, $0
84
+ // cost, ~300ms, verified live 2026-09-27 (the original design spawned a
85
+ // real model turn and was fixed after a live proof showed it never
86
+ // worked; see docs/RUNTIME_CAPACITY.md). Failures never break the read
87
+ // below.
85
88
  app.get("/api/runtime-capacity", async () => {
86
89
  try {
87
90
  void refreshClaudeCapacityIfStale().catch(() => {
@@ -93,7 +96,7 @@ export async function registerRoutes(app) {
93
96
  }
94
97
  try {
95
98
  if (configuredCapacityRuntimes().includes("muse_code")) {
96
- void maybeRefreshMuseCapacityFromHost().catch(() => {
99
+ void refreshMuseCapacityIfStale().catch(() => {
97
100
  // Best-effort: failures preserve last-good rows via the ingest path.
98
101
  });
99
102
  }
@@ -15,6 +15,13 @@ process.env.AGENT_DEALER_SKIP_AGENT_HEALTH = "1";
15
15
  // Never spawn a real provider from route tests: the on-demand Codex refresh is
16
16
  // covered against the fake App Server in codex-app-server.test.ts.
17
17
  process.env.AGENT_DEALER_CODEX_CAPACITY_REFRESH = "off";
18
+ // Claude's `/usage` refresh is free and default-on in production (fixed
19
+ // 2026-09-27 — see docs/RUNTIME_CAPACITY.md); still disabled here so route
20
+ // tests never spawn a real `claude` subprocess.
21
+ process.env.AGENT_DEALER_CLAUDE_CAPACITY_REFRESH = "off";
22
+ // Muse's one-hour fallback is also default-on; route tests cover the free
23
+ // host trigger only and must never run a real model turn.
24
+ process.env.AGENT_DEALER_MUSE_CAPACITY_REFRESH = "off";
18
25
  // NOT-268: the route also ingests the real ~/.claude.json cache on every read. Point it at a
19
26
  // path that never exists so a developer's own Claude usage never leaks extra windows into the
20
27
  // fixture-controlled assertions below (this passed in CI, which has no such file, but failed
@@ -62,9 +69,10 @@ test("GET /api/runtime-capacity returns normalized entries without evidence", as
62
69
  assert.ok(!raw.includes("evidence"), "no evidence pointers leak to the browser");
63
70
  await app.close();
64
71
  });
65
- test("GET triggers the Muse refresh when muse_code is configured", async () => {
66
- // NOT-270: the route is the only production trigger for Muse capacity
67
- // (owned host). No credential here (env key removed, empty login dir),
72
+ test("GET triggers the free Muse refresh when muse_code is configured", async () => {
73
+ // The route is the production trigger for Muse capacity. This suite sets
74
+ // the paid fallback off, so it exercises only the free owned-host read.
75
+ // No credential here (env key removed, empty login dir),
68
76
  // so the refresh short-circuits to a `missing` sentinel without spawning
69
77
  // anything live.
70
78
  clearAllCapacitySnapshots();
@@ -1,12 +1,13 @@
1
1
  // packages/server/src/runners/muse-serve-session.ts
2
2
  //
3
- // NOT-270: run a real Dealer Muse developer turn through the server-owned
4
- // `muse serve` host (the NOT-269 proven lifecycle). Only a host that
3
+ // NOT-270: run a Muse turn through a caller-owned `muse serve` host (the
4
+ // NOT-269 proven lifecycle). Only a host that
5
5
  // observes the account's provider traffic can answer `usage/read`, so the
6
- // observation opportunity IS the real session: `session/start` +
7
- // `turn/start` on the owned host, then the session-boundary `usage/read`
8
- // (the existing refresh hook) populates 5H/1W. No synthetic prompt is ever
9
- // sent — the turn below carries the genuine Dealer developer prompt.
6
+ // observation opportunity is a `session/start` + `turn/start` on that host.
7
+ // Production callers use this for genuine Dealer developer work and, when
8
+ // capacity has been unavailable for an hour, the dedicated bounded capacity
9
+ // probe. The capacity probe owns a separate restricted host and shuts it down
10
+ // after its final `usage/read`.
10
11
  //
11
12
  // Wire contract (stable MSP, Muse 1.4.x):
12
13
  // session/start {commandId UUIDv7, workspaceRoot, modelId, approvalMode}
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@agent-dealer/server",
3
- "version": "1.2.3",
3
+ "version": "1.2.5",
4
4
  "type": "module",
5
5
  "main": "dist/index.js",
6
6
  "files": [
@@ -22,7 +22,7 @@
22
22
  "typecheck": "tsc --noEmit"
23
23
  },
24
24
  "dependencies": {
25
- "@agent-dealer/shared": "1.2.3",
25
+ "@agent-dealer/shared": "1.2.5",
26
26
  "@fastify/cors": "^11.0.1",
27
27
  "@fastify/static": "^8.2.0",
28
28
  "@modelcontextprotocol/sdk": "^1.29.0",