jules-orchestrator-kit 0.53.0 → 0.54.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/git.mjs CHANGED
@@ -1,5 +1,5 @@
1
1
  import { execFileSync, execSync } from "node:child_process";
2
- import { readFileSync, existsSync } from "node:fs";
2
+ import { readFileSync, existsSync, statSync } from "node:fs";
3
3
  import { join, delimiter } from "node:path";
4
4
  import { normalizePath } from "./config.mjs";
5
5
 
@@ -394,9 +394,104 @@ export function diffText(root = process.cwd(), base = "main", mode = "committed"
394
394
  return git(["diff", `${resolvedRef}...HEAD`], { cwd: root, raw: true });
395
395
  }
396
396
 
397
+ /**
398
+ * Paths git summarised as binary in this diff, with the size of what changed.
399
+ *
400
+ * `git diff` prints one 43-byte line for a binary file — "Binary files a/x and
401
+ * b/x differ" — regardless of whether x grew by a byte or by half a megabyte.
402
+ * Everything downstream reads the diff *text*, so a binary file is invisible to
403
+ * both the payload governor and the secret scanner: a 500 KB blob measured 250
404
+ * bytes, and a credential with a leading NUL byte was never looked at.
405
+ *
406
+ * Sizes come from the object git actually recorded where there is one, and from
407
+ * the working file otherwise, so both `committed` and `working-tree` modes get a
408
+ * real number.
409
+ *
410
+ * @param {string} root
411
+ * @param {string} base
412
+ * @param {string} mode
413
+ * @returns {Array<{ file: string, bytes: number }>}
414
+ */
415
+ export function binaryDiffEntries(root = process.cwd(), base = "main", mode = "committed") {
416
+ const resolvedRef = resolveBase(root, base);
417
+ const ranges =
418
+ mode === "working-tree" || mode === "working"
419
+ ? [[`${resolvedRef}...HEAD`], ["HEAD"]]
420
+ : [[`${resolvedRef}...HEAD`]];
421
+
422
+ const entries = new Map();
423
+ for (const range of ranges) {
424
+ let raw = "";
425
+ try {
426
+ raw = git(["diff", "--raw", "--no-renames", "-z", ...range], { cwd: root, raw: true, ignoreError: true }) || "";
427
+ } catch (_) {
428
+ continue;
429
+ }
430
+ // `--raw -z` emits ":<srcmode> <dstmode> <srcsha> <dstsha> <status>\0<path>\0".
431
+ const fields = raw.split("\0").filter(Boolean);
432
+ for (let i = 0; i < fields.length; i++) {
433
+ const meta = fields[i];
434
+ if (!meta.startsWith(":")) continue;
435
+ const parts = meta.slice(1).split(/\s+/);
436
+ const dstSha = parts[3];
437
+ const status = (parts[4] || "").charAt(0);
438
+ const file = fields[i + 1];
439
+ i += 1;
440
+ if (!file || status === "D") continue;
441
+
442
+ let bytes = 0;
443
+ if (dstSha && !/^0+$/.test(dstSha)) {
444
+ const size = git(["cat-file", "-s", dstSha], { cwd: root, ignoreError: true });
445
+ bytes = Number(size) || 0;
446
+ }
447
+ // An unstaged change has an all-zero destination sha; the working file is
448
+ // the only place its size exists.
449
+ if (!bytes) {
450
+ try {
451
+ bytes = statSync(join(root, file)).size;
452
+ } catch (_) {
453
+ bytes = 0;
454
+ }
455
+ }
456
+ // Keep the largest observation: the same path can appear in both ranges.
457
+ entries.set(file, Math.max(entries.get(file) || 0, bytes));
458
+ }
459
+ }
460
+
461
+ // Only the paths git itself refused to render as text are relevant; a file
462
+ // that diffed normally is already counted in the diff text.
463
+ const binaryPaths = new Set();
464
+ const text = diffText(root, base, mode);
465
+ for (const line of text.split("\n")) {
466
+ const m = line.match(/^Binary files (?:a\/(.+) and )?(?:b\/(.+)|\/dev\/null) differ$/);
467
+ if (m) binaryPaths.add(normalizePath(m[2] || m[1] || ""));
468
+ const m2 = line.match(/^Binary files \/dev\/null and b\/(.+) differ$/);
469
+ if (m2) binaryPaths.add(normalizePath(m2[1]));
470
+ }
471
+
472
+ return [...entries.entries()]
473
+ .filter(([file]) => binaryPaths.has(normalizePath(file)))
474
+ .map(([file, bytes]) => ({ file, bytes }));
475
+ }
476
+
477
+ /**
478
+ * Total bytes this change actually carries.
479
+ *
480
+ * The diff text plus the real size of every binary blob it only summarised.
481
+ * Without the second term the payload governor could be walked straight past
482
+ * with a committed binary of any size.
483
+ */
397
484
  export function diffBytes(root = process.cwd(), base = "main", mode = "committed") {
398
485
  const text = diffText(root, base, mode);
399
- return Buffer.byteLength(text, "utf-8");
486
+ let bytes = Buffer.byteLength(text, "utf-8");
487
+ try {
488
+ for (const entry of binaryDiffEntries(root, base, mode)) bytes += entry.bytes;
489
+ } catch (_) {
490
+ // A payload figure that is too low is the dangerous direction, but throwing
491
+ // here would break every gate on a repo git cannot describe. The text-only
492
+ // number is still returned.
493
+ }
494
+ return bytes;
400
495
  }
401
496
 
402
497
  export function showFromOrigin(root = process.cwd(), base = "main", filePath = "") {
@@ -454,6 +549,85 @@ export function parseGitHubRepo(remoteUrl) {
454
549
  * @param {string} [root=process.cwd()]
455
550
  * @returns {string}
456
551
  */
552
+ /**
553
+ * Work out which branch this repository actually treats as its base.
554
+ *
555
+ * `main` was hardcoded as the scaffolded default, which is wrong the moment
556
+ * `git init` picks `master` (still the default on many installed gits) or the
557
+ * project standardised on `develop`. The failure was not graceful: the first
558
+ * `agentctl check` could not resolve the base ref and rejected with exit 1.
559
+ *
560
+ * Resolution order, strongest evidence first:
561
+ * 1. `origin/HEAD` — the remote states its own default branch.
562
+ * 2. A local `main`, then `master` — the conventional names, preferred over
563
+ * the checked-out branch so running `init` from a feature branch does not
564
+ * record that feature branch as the base for everything after.
565
+ * 3. The current branch — covers a fresh `git init` with no commits, where
566
+ * HEAD points at an unborn branch that `--show-current` still names.
567
+ * 4. `main`, when there is no git information at all to go on.
568
+ *
569
+ * @param {string} [root=process.cwd()]
570
+ * @returns {string}
571
+ */
572
+ /**
573
+ * Split a list of repo-relative paths into the ones git already tracks and the
574
+ * ones it does not.
575
+ *
576
+ * Used to tell an onboarding mistake apart from an agent overstepping. Both
577
+ * arrive as the same exit-3 scope violation on the same protected paths, but
578
+ * "the files `init` just wrote are not committed yet" and "something edited the
579
+ * rules it is governed by" need opposite advice, and the operator who has just
580
+ * met the tool gets the wrong half if the two are not distinguished.
581
+ *
582
+ * @param {string} root
583
+ * @param {string[]} files - repo-relative paths
584
+ * @returns {{ tracked: string[], untracked: string[] }}
585
+ */
586
+ export function partitionTracked(root = process.cwd(), files = []) {
587
+ const list = files.filter((f) => typeof f === "string" && f && !f.startsWith("-"));
588
+ if (list.length === 0) return { tracked: [], untracked: [] };
589
+ let out = "";
590
+ try {
591
+ out = git(["ls-files", "--", ...list], { cwd: root, ignoreError: true }) || "";
592
+ } catch (_) {
593
+ // Without an answer, claim nothing is tracked-or-not: callers fall back to
594
+ // the generic advice rather than to a guess.
595
+ return { tracked: [], untracked: [] };
596
+ }
597
+ const tracked = new Set(out.split("\n").map((l) => l.trim()).filter(Boolean));
598
+ return {
599
+ tracked: list.filter((f) => tracked.has(f)),
600
+ untracked: list.filter((f) => !tracked.has(f)),
601
+ };
602
+ }
603
+
604
+ export function detectDefaultBranch(root = process.cwd()) {
605
+ // `git()` returns trimmed stdout, and "" for a non-zero exit under
606
+ // ignoreError — there is no status field to read.
607
+ const ask = (args) => {
608
+ try {
609
+ return git(args, { cwd: root, ignoreError: true }) || "";
610
+ } catch (_) {
611
+ return "";
612
+ }
613
+ };
614
+
615
+ const remoteHead = ask(["symbolic-ref", "--quiet", "--short", "refs/remotes/origin/HEAD"]);
616
+ if (remoteHead.startsWith("origin/")) {
617
+ const name = remoteHead.slice("origin/".length).trim();
618
+ if (name) return name;
619
+ }
620
+
621
+ for (const candidate of ["main", "master"]) {
622
+ if (ask(["rev-parse", "--verify", "--quiet", `refs/heads/${candidate}`])) return candidate;
623
+ }
624
+
625
+ const current = ask(["branch", "--show-current"]);
626
+ if (current) return current;
627
+
628
+ return "main";
629
+ }
630
+
457
631
  export function resolveGitRemoteOrigin(root = process.cwd()) {
458
632
  try {
459
633
  const raw = git(["config", "--get", "remote.origin.url"], { cwd: root, ignoreError: true });
@@ -0,0 +1,29 @@
1
+ /**
2
+ * Whether a command invocation is asking to be walked through a wizard, or has
3
+ * already said everything the wizard would ask.
4
+ *
5
+ * `agentctl task create --title X --prompt Y` states the whole task and then
6
+ * opened the interactive wizard anyway, which re-asked for the title and the
7
+ * prompt in a terminal and hung a CI job that had no terminal to answer with.
8
+ * Neither `--yes` nor `--non-interactive` was passed, so the CLI concluded
9
+ * "interactive" from the absence of a flag rather than from the presence of the
10
+ * answers.
11
+ *
12
+ * Kept as a pure function, out of the CLI script, because that script executes
13
+ * on import and so cannot be exercised from a test.
14
+ *
15
+ * @param {object} input
16
+ * @param {boolean} [input.interactive] - `--interactive` was passed explicitly
17
+ * @param {boolean} [input.nonInteractive] - `--non-interactive` / `--no-interactive`
18
+ * @param {boolean} [input.yes] - `--yes`
19
+ * @param {boolean} [input.fullySpecified] - every field the wizard would ask for is present
20
+ * @returns {boolean|undefined} true/false to force a mode, undefined to let the
21
+ * callee decide from whether stdin is a TTY
22
+ */
23
+ export function resolveWizardInteractivity(input = {}) {
24
+ // An explicit `--interactive` always wins: the flags may be intended as seeds
25
+ // to edit rather than as a complete invocation.
26
+ if (input.interactive === true) return true;
27
+ if (input.nonInteractive || input.yes || input.fullySpecified) return false;
28
+ return undefined;
29
+ }
@@ -70,35 +70,44 @@ export const COMMAND_REGISTRY = [
70
70
  shortcuts: ["d", "doc"],
71
71
  examples: [
72
72
  "agentctl doctor",
73
- "agentctl doctor --interactive",
74
- "agentctl doctor --fix safe --yes",
73
+ "agentctl doctor --probe",
75
74
  "agentctl doctor --json",
76
75
  ],
76
+ // `--interactive`, `--fix` and `--yes` were listed here and implemented
77
+ // nowhere: the report carries remediation entries, but nothing applies
78
+ // them. Advertising a flag the command silently ignores is the same defect
79
+ // as documenting `queue` as a read-only browser. They come back to this
80
+ // list when an apply step exists.
77
81
  flags: [
78
- { name: "interactive", type: "boolean", description: "Open full-screen diagnostic matrix" },
79
- { name: "fix", type: "string", description: "Apply fix classes (e.g. 'safe' or check IDs)" },
80
- { name: "probe", type: "boolean", description: "Enable active network/execution probes" },
82
+ { name: "probe", type: "boolean", description: "Actively start the provider CLI to check it answers, rather than only finding it on PATH" },
81
83
  { name: "json", type: "boolean", description: "Output structured JSON doctor report" },
82
- { name: "yes", type: "boolean", description: "Bypass interactive confirmation for safe fixes" },
83
84
  ],
84
85
  },
85
86
  {
86
87
  id: "queue",
88
+ // This described a passive viewer — "Browse and manage", `mutates: false`,
89
+ // `risk: low` — while the handler runs the queue: it dispatches every task
90
+ // to the provider and spends budget. Someone reading `--help` before their
91
+ // first run was told the opposite of what the command does.
87
92
  path: ["queue"],
88
93
  title: "queue",
89
- description: "Browse and manage canonical task queue",
94
+ description: "Execute pending task envelopes: dispatches each to the provider and moves it out of the queue",
90
95
  category: "Operate",
91
- mutates: false,
92
- risk: "low",
96
+ mutates: true,
97
+ risk: "moderate",
93
98
  interactive: "optional",
94
99
  requiresRepository: true,
95
100
  shortcuts: ["q"],
96
101
  examples: [
102
+ "agentctl queue --dry-run",
97
103
  "agentctl queue",
98
- "agentctl queue --interactive",
104
+ "agentctl queue --dag --concurrency 3",
99
105
  "agentctl queue --json",
100
106
  ],
101
107
  flags: [
108
+ { name: "dag", type: "boolean", description: "Resolve depends-on order via Kahn's algorithm before running" },
109
+ { name: "concurrency", type: "string", description: "Parallel worker slots (defaults to limits.concurrency)" },
110
+ { name: "dry-run", type: "boolean", description: "Report what would run without dispatching or moving anything" },
102
111
  { name: "interactive", type: "boolean", description: "Open full-screen queue dashboard" },
103
112
  { name: "json", type: "boolean", description: "Output structured JSON queue snapshot" },
104
113
  { name: "limit", type: "string", description: "Maximum tasks to include in snapshot" },
@@ -3,7 +3,7 @@ import { join, resolve } from "node:path";
3
3
  import { createHash } from "node:crypto";
4
4
  import { execFileSync, spawnSync } from "node:child_process";
5
5
  import { loadConfig } from "../config.mjs";
6
- import { probeProvider, detectAvailableProviders } from "../provider-readiness.mjs";
6
+ import { probeProvider, detectAvailableProviders, probeProviderLiveness } from "../provider-readiness.mjs";
7
7
  import { resolveConcurrency } from "../budget.mjs";
8
8
 
9
9
  /**
@@ -443,31 +443,53 @@ export async function runDoctorChecks(options = {}) {
443
443
  // Fall back to the default; config.present already reports a broken config.
444
444
  }
445
445
  const providerProbe = probeProvider(configuredProvider);
446
+ // A green row here used to read as "the provider works", when all it ever
447
+ // checked was a name on PATH or a variable in the environment. An installed
448
+ // CLI whose account has no entitlement passes both and then fails on the
449
+ // first dispatch, so the row has to say what it actually verified.
450
+ const liveness = activeProbe ? probeProviderLiveness(configuredProvider) : null;
451
+ const scopeNote =
452
+ providerProbe.kind === "exec"
453
+ ? "Checked: the binary is on PATH. Not checked: whether the CLI is signed in — only a dispatch can show that."
454
+ : "Checked: a credential is present in the environment. Not checked: whether the provider accepts it.";
455
+ const livenessFailed = Boolean(liveness && liveness.attempted && !liveness.ok);
456
+
446
457
  addResult({
447
458
  id: "provider.key",
448
459
  category: "Provider",
449
460
  title: `Provider Readiness (${providerProbe.name})`,
450
- status: providerProbe.ready ? "pass" : "warn",
451
- severity: providerProbe.ready ? "info" : "high",
461
+ alwaysShowSummary: true,
462
+ status: providerProbe.ready && !livenessFailed ? "pass" : "warn",
463
+ severity: providerProbe.ready && !livenessFailed ? "info" : "high",
452
464
  // Naming the variable, not the value: an operator who wonders which key a
453
465
  // dispatch will use should not have to echo a secret to find out.
454
- summary: `${providerProbe.label} — ${providerProbe.reason}`,
455
- remediation: providerProbe.ready
456
- ? []
457
- : [
458
- {
459
- summary: providerProbe.remedy,
460
- risk: "low",
461
- automatic: false,
462
- requiresProbe: false,
463
- },
464
- ],
466
+ summary: [
467
+ `${providerProbe.label} — ${providerProbe.reason}`,
468
+ liveness && liveness.attempted ? liveness.detail : null,
469
+ providerProbe.ready ? scopeNote : null,
470
+ !activeProbe && providerProbe.kind === "exec" ? "Run `agentctl doctor --probe` to start the CLI and confirm it answers." : null,
471
+ ]
472
+ .filter(Boolean)
473
+ .join(" "),
474
+ remediation:
475
+ providerProbe.ready && !livenessFailed
476
+ ? []
477
+ : [
478
+ {
479
+ summary: livenessFailed ? `The CLI is installed but did not run cleanly: ${liveness.detail}` : providerProbe.remedy,
480
+ risk: "low",
481
+ automatic: false,
482
+ requiresProbe: !providerProbe.ready ? false : true,
483
+ },
484
+ ],
465
485
  evidence: [
466
486
  { label: "provider", value: providerProbe.name, sensitive: false },
467
487
  { label: "providerKind", value: providerProbe.kind, sensitive: false },
468
488
  { label: "ready", value: providerProbe.ready, sensitive: false },
469
489
  { label: "keySource", value: providerProbe.keySource || "", sensitive: false },
470
490
  { label: "binaryFound", value: Boolean(providerProbe.binPath), sensitive: false },
491
+ { label: "livenessProbed", value: Boolean(liveness && liveness.attempted), sensitive: false },
492
+ { label: "livenessOk", value: liveness ? liveness.ok : null, sensitive: false },
471
493
  ],
472
494
  });
473
495
 
@@ -11,14 +11,32 @@ export class IdeScaffoldError extends Error {
11
11
 
12
12
  /**
13
13
  * Scaffolds IDE integration configuration files for Cursor, VS Code, and Claude Desktop.
14
+ *
15
+ * These files land in directories a project may already be using (`.cursor/`,
16
+ * `.vscode/`), and the merge logic preserves what is there — but `--dry-run`
17
+ * was accepted by the CLI and never reached here, so the only way to find out
18
+ * what would be touched was to let it happen.
19
+ *
14
20
  * @param {string} target - 'cursor' | 'vscode' | 'claude' | 'all'
15
21
  * @param {object} [options]
22
+ * @param {string} [options.root]
23
+ * @param {boolean} [options.dryRun] - report the files without writing them
16
24
  * @returns {object} Summary of scaffolded files
17
25
  */
18
26
  export function scaffoldIdeConfig(target = "all", options = {}) {
19
27
  const root = options.root || resolveRoot();
28
+ const dryRun = Boolean(options.dryRun);
20
29
  const validTargets = new Set(["cursor", "vscode", "claude", "all"]);
21
30
 
31
+ /** Write unless this is a rehearsal; the directory is only made when writing. */
32
+ const writeUnlessRehearsing = (dir, file, contents) => {
33
+ if (dryRun) return;
34
+ try {
35
+ mkdirSync(dir, { recursive: true });
36
+ } catch (_) {}
37
+ writeFileSync(file, contents, "utf-8");
38
+ };
39
+
22
40
  const normTarget = (target || "all").toLowerCase();
23
41
  if (!validTargets.has(normTarget)) {
24
42
  throw new IdeScaffoldError(`Invalid target '${target}'. Allowed targets: cursor, vscode, claude, all`);
@@ -29,10 +47,6 @@ export function scaffoldIdeConfig(target = "all", options = {}) {
29
47
  // Target: Cursor (.cursor/mcp.json)
30
48
  if (normTarget === "cursor" || normTarget === "all") {
31
49
  const cursorDir = join(root, ".cursor");
32
- try {
33
- mkdirSync(cursorDir, { recursive: true });
34
- } catch (_) {}
35
-
36
50
  const cursorConfigPath = join(cursorDir, "mcp.json");
37
51
  let existingConfig = {};
38
52
  if (existsSync(cursorConfigPath)) {
@@ -52,17 +66,13 @@ export function scaffoldIdeConfig(target = "all", options = {}) {
52
66
  },
53
67
  };
54
68
 
55
- writeFileSync(cursorConfigPath, JSON.stringify(updatedCursorConfig, null, 2), "utf-8");
69
+ writeUnlessRehearsing(cursorDir, cursorConfigPath, JSON.stringify(updatedCursorConfig, null, 2));
56
70
  results.push({ target: "cursor", file: ".cursor/mcp.json", ok: true });
57
71
  }
58
72
 
59
73
  // Target: VS Code (.vscode/tasks.json)
60
74
  if (normTarget === "vscode" || normTarget === "all") {
61
75
  const vscodeDir = join(root, ".vscode");
62
- try {
63
- mkdirSync(vscodeDir, { recursive: true });
64
- } catch (_) {}
65
-
66
76
  const vscodeTasksPath = join(vscodeDir, "tasks.json");
67
77
  let existingConfig = { version: "2.0.0", tasks: [] };
68
78
  if (existsSync(vscodeTasksPath)) {
@@ -110,17 +120,13 @@ export function scaffoldIdeConfig(target = "all", options = {}) {
110
120
  tasks: updatedTasks,
111
121
  };
112
122
 
113
- writeFileSync(vscodeTasksPath, JSON.stringify(updatedVsCodeConfig, null, 2), "utf-8");
123
+ writeUnlessRehearsing(vscodeDir, vscodeTasksPath, JSON.stringify(updatedVsCodeConfig, null, 2));
114
124
  results.push({ target: "vscode", file: ".vscode/tasks.json", ok: true });
115
125
  }
116
126
 
117
127
  // Target: Claude Desktop (claude_desktop_config.json snippet helper)
118
128
  if (normTarget === "claude" || normTarget === "all") {
119
129
  const agentDir = join(root, ".agent");
120
- try {
121
- mkdirSync(agentDir, { recursive: true });
122
- } catch (_) {}
123
-
124
130
  const snippetPath = join(agentDir, "claude_desktop_config.snippet.json");
125
131
  const claudeSnippet = {
126
132
  mcpServers: {
@@ -131,9 +137,9 @@ export function scaffoldIdeConfig(target = "all", options = {}) {
131
137
  },
132
138
  };
133
139
 
134
- writeFileSync(snippetPath, JSON.stringify(claudeSnippet, null, 2), "utf-8");
140
+ writeUnlessRehearsing(agentDir, snippetPath, JSON.stringify(claudeSnippet, null, 2));
135
141
  results.push({ target: "claude", file: ".agent/claude_desktop_config.snippet.json", ok: true });
136
142
  }
137
143
 
138
- return { ok: true, target: normTarget, results };
144
+ return { ok: true, target: normTarget, dryRun, results };
139
145
  }
@@ -0,0 +1,38 @@
1
+ /**
2
+ * Choose what to show the operator from a failed stage's captured streams.
3
+ *
4
+ * stderr is where a failing suite says what it expected; stdout is where
5
+ * several runners (node:test among them) put the whole report — assertion text
6
+ * and stack included.
7
+ *
8
+ * Preferring stderr outright discarded all of that whenever stderr held so much
9
+ * as the spawn wrapper's own "Command failed: npm test", which it always does.
10
+ * The operator was shown a line naming the command they had just typed, and
11
+ * nothing about which test failed or why. A wrapper line is not diagnostic
12
+ * output: when that is all stderr has, stdout is the report.
13
+ *
14
+ * Lives in src/ rather than in the CLI script so it can be imported and tested
15
+ * without executing the CLI, which runs on import.
16
+ *
17
+ * @param {{ stdout?: string, stderr?: string }} failure
18
+ * @returns {string}
19
+ */
20
+ export function selectFailureOutput(failure = {}) {
21
+ const rawStderr = (failure.stderr || "").trim();
22
+ const rawStdout = (failure.stdout || "").trim();
23
+ if (!rawStderr) return rawStdout;
24
+
25
+ const stderrIsWrapperOnly = rawStderr
26
+ .split("\n")
27
+ .every((line) => line.trim() === "" || /^(command failed|error: command failed|npm err!?)/i.test(line.trim()));
28
+
29
+ if (stderrIsWrapperOnly && rawStdout) return rawStdout;
30
+ if (rawStdout && rawStdout !== rawStderr) {
31
+ // Both carry something real, so show both rather than guessing which runner
32
+ // this is. stderr goes last on purpose: the caller keeps only the tail, and
33
+ // when stderr has real content it is nearly always the failure itself —
34
+ // putting the runner's banner after it would spend the cap on noise.
35
+ return `${rawStdout}\n${rawStderr}`;
36
+ }
37
+ return rawStderr;
38
+ }
@@ -1,5 +1,7 @@
1
1
  import { existsSync, statSync } from "node:fs";
2
2
  import { join, delimiter } from "node:path";
3
+ import { spawnSync } from "node:child_process";
4
+ import { resolveWindowsSpawn } from "./git.mjs";
3
5
 
4
6
  /**
5
7
  * What each provider actually needs before a dispatch can succeed.
@@ -172,6 +174,7 @@ export function probeProvider(name, opts = {}) {
172
174
  return {
173
175
  name: descriptor.name,
174
176
  kind: descriptor.kind,
177
+ bin: descriptor.bin,
175
178
  label: descriptor.label,
176
179
  ready,
177
180
  known: true,
@@ -182,6 +185,72 @@ export function probeProvider(name, opts = {}) {
182
185
  };
183
186
  }
184
187
 
188
+ /**
189
+ * Actually run the provider's CLI, rather than only finding it on PATH.
190
+ *
191
+ * `probeProvider` deliberately spawns nothing — it is called from a bare
192
+ * `agentctl` invocation and from `doctor`, both of which must stay instant. But
193
+ * a binary on PATH is a weak claim: a `gemini` that is installed and whose
194
+ * account has no access still reports ready, `doctor` still says 11 passed, and
195
+ * the first thing that disagrees is a dispatch that dies. This is the check
196
+ * that can disagree earlier, so it is opt-in (`agentctl doctor --probe`).
197
+ *
198
+ * It proves the binary starts and answers, not that the account is entitled —
199
+ * no CLI exposes "am I authorised" without doing work — but the common failures
200
+ * (broken install, wrong architecture, a CLI that refuses to start unauthenticated)
201
+ * surface here instead of mid-repair.
202
+ *
203
+ * @param {string} name
204
+ * @param {object} [opts]
205
+ * @param {NodeJS.ProcessEnv} [opts.env=process.env]
206
+ * @param {number} [opts.timeoutMs=8000]
207
+ * @returns {{ name: string, attempted: boolean, ok: boolean, detail: string }}
208
+ */
209
+ export function probeProviderLiveness(name, opts = {}) {
210
+ const env = opts.env || process.env;
211
+ const base = probeProvider(name, { env });
212
+ if (base.kind !== "exec" || !base.binPath) {
213
+ return {
214
+ name: base.name,
215
+ attempted: false,
216
+ ok: base.ready,
217
+ detail: base.kind === "http" ? "Hosted provider — a credential cannot be validated without spending a request." : base.reason,
218
+ };
219
+ }
220
+
221
+ try {
222
+ // A global npm install puts a `.cmd` shim on PATH, and since the fix for
223
+ // CVE-2024-27980 Node refuses to spawn one directly — it comes back EINVAL,
224
+ // which would report every Windows CLI as broken. `resolveWindowsSpawn` is
225
+ // the same routing `runCmd` already uses: native .exe direct, everything
226
+ // else through cmd.exe with the argv quoted the way cmd.exe parses it back.
227
+ const win = resolveWindowsSpawn(base.binPath, ["--version"], env);
228
+ const res = win
229
+ ? spawnSync(win.file, win.args, {
230
+ encoding: "utf-8",
231
+ timeout: Number.isFinite(opts.timeoutMs) ? opts.timeoutMs : 8000,
232
+ env,
233
+ windowsVerbatimArguments: win.verbatim,
234
+ })
235
+ : spawnSync(base.binPath, ["--version"], {
236
+ encoding: "utf-8",
237
+ timeout: Number.isFinite(opts.timeoutMs) ? opts.timeoutMs : 8000,
238
+ env,
239
+ });
240
+ if (res.error) {
241
+ return { name: base.name, attempted: true, ok: false, detail: `\`${base.bin || base.name} --version\` could not run: ${res.error.message}` };
242
+ }
243
+ if (res.status !== 0) {
244
+ const why = ((res.stderr || res.stdout || "").trim().split("\n")[0] || `exit ${res.status}`).slice(0, 200);
245
+ return { name: base.name, attempted: true, ok: false, detail: `\`--version\` exited ${res.status}: ${why}` };
246
+ }
247
+ const version = (res.stdout || "").trim().split("\n")[0].slice(0, 80);
248
+ return { name: base.name, attempted: true, ok: true, detail: version ? `CLI responds: ${version}` : "CLI responds." };
249
+ } catch (err) {
250
+ return { name: base.name, attempted: true, ok: false, detail: `Probe failed: ${err.message}` };
251
+ }
252
+ }
253
+
185
254
  /**
186
255
  * Probe every built-in provider, ready ones first, in preference order.
187
256
  *