jules-orchestrator-kit 0.70.0 → 0.72.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -543,6 +543,115 @@ export const COMMAND_REGISTRY = [
543
543
  { name: "json", type: "boolean", description: "Output JSON structured bootstrap result (-j)" },
544
544
  ],
545
545
  },
546
+ {
547
+ id: "pr-harvest",
548
+ path: ["pr", "harvest"],
549
+ title: "pr harvest",
550
+ description: "Scan, audit and auto-merge verified agent pull requests",
551
+ category: "Operate",
552
+ mutates: true,
553
+ risk: "moderate",
554
+ interactive: "never",
555
+ requiresRepository: true,
556
+ shortcuts: ["harvest", "pr"],
557
+ examples: [
558
+ "agentctl pr harvest",
559
+ "agentctl pr harvest --auto",
560
+ "agentctl pr harvest --dry-run",
561
+ "agentctl pr harvest --json",
562
+ ],
563
+ flags: [
564
+ { name: "tier", type: "string", description: "Filter PRs by tier label" },
565
+ { name: "limit", type: "string", description: "Maximum PRs to harvest" },
566
+ { name: "auto", type: "boolean", description: "Automatically merge qualifying green PRs" },
567
+ { name: "merge", type: "boolean", description: "Merge matching PRs" },
568
+ { name: "allow-no-checks", type: "boolean", description: "Allow PR merge when no CI checks are configured" },
569
+ { name: "dry-run", type: "boolean", description: "Simulate PR harvest without merging (-d)" },
570
+ { name: "json", type: "boolean", description: "Output structured JSON harvest report (-j)" },
571
+ ],
572
+ },
573
+ {
574
+ id: "session-get",
575
+ path: ["session", "get"],
576
+ title: "session get",
577
+ description: "Retrieve remote execution status for a session ID",
578
+ category: "Inspect",
579
+ mutates: false,
580
+ risk: "low",
581
+ interactive: "never",
582
+ requiresRepository: true,
583
+ shortcuts: ["session status", "session"],
584
+ examples: [
585
+ "agentctl session get <sessionId>",
586
+ "agentctl session get <sessionId> --dry-run",
587
+ "agentctl session get <sessionId> --json",
588
+ ],
589
+ flags: [
590
+ { name: "dry-run", type: "boolean", description: "Simulate session retrieval (-d)" },
591
+ { name: "json", type: "boolean", description: "Output structured JSON session data (-j)" },
592
+ ],
593
+ },
594
+ {
595
+ id: "plan-approve",
596
+ path: ["plan", "approve"],
597
+ title: "plan approve",
598
+ description: "Approve a pending execution plan for an agent session",
599
+ category: "Operate",
600
+ mutates: true,
601
+ risk: "moderate",
602
+ interactive: "never",
603
+ requiresRepository: true,
604
+ shortcuts: ["approve", "plan"],
605
+ examples: [
606
+ "agentctl plan approve <sessionId>",
607
+ "agentctl plan approve <sessionId> --dry-run",
608
+ "agentctl approve <sessionId>",
609
+ ],
610
+ flags: [
611
+ { name: "dry-run", type: "boolean", description: "Simulate plan approval (-d)" },
612
+ { name: "json", type: "boolean", description: "Output structured JSON result (-j)" },
613
+ ],
614
+ },
615
+ {
616
+ id: "lock",
617
+ path: ["lock"],
618
+ title: "lock",
619
+ description: "Multi-agent coordination locks: acquire, release, or view file status",
620
+ category: "Operate",
621
+ mutates: true,
622
+ risk: "low",
623
+ interactive: "never",
624
+ requiresRepository: true,
625
+ shortcuts: [],
626
+ examples: [
627
+ "agentctl lock status",
628
+ "agentctl lock acquire agent-1 task-1 src/main.js",
629
+ "agentctl lock release task-1",
630
+ ],
631
+ flags: [
632
+ { name: "json", type: "boolean", description: "Output structured JSON result (-j)" },
633
+ ],
634
+ },
635
+ {
636
+ id: "evidence",
637
+ path: ["evidence"],
638
+ title: "evidence",
639
+ description: "Inspect and verify cryptographic evidence manifests",
640
+ category: "Inspect",
641
+ mutates: false,
642
+ risk: "low",
643
+ interactive: "never",
644
+ requiresRepository: true,
645
+ shortcuts: [],
646
+ examples: [
647
+ "agentctl evidence status",
648
+ "agentctl evidence verify",
649
+ "agentctl evidence export --json",
650
+ ],
651
+ flags: [
652
+ { name: "json", type: "boolean", description: "Output structured JSON result (-j)" },
653
+ ],
654
+ },
546
655
  ];
547
656
 
548
657
  /**
@@ -33,8 +33,7 @@ const COUNT_PATTERNS = [
33
33
  // pytest — "collected 12 items", "12 passed", "no tests ran in 0.01s"
34
34
  { name: "pytest", re: /^\s*collected\s+(\d+)\s+items?/m },
35
35
  { name: "pytest", re: /=+\s*(\d+)\s+passed/m },
36
- // cargo — "running 7 tests"
37
- { name: "cargo", re: /^\s*running\s+(\d+)\s+tests?\s*$/m },
36
+ // Cargo multi-target output is aggregated above COUNT_PATTERNS (F15)
38
37
  // jest / vitest — "Tests: 12 passed, 12 total"
39
38
  { name: "jest", re: /^\s*Tests:\s+.*?(\d+)\s+total\s*$/m },
40
39
  // mocha — "12 passing"
@@ -88,10 +87,13 @@ const EXPLICIT_ZERO = [
88
87
  { name: "gradle", re: /^>\s*Task\s+:\S*test\S*\s+NO-SOURCE\s*$/mi },
89
88
  { name: "ctest", re: /\bNo tests were found\b/i },
90
89
  { name: "flutter", re: /\bNo tests ran\.?/i },
90
+ { name: "go", re: /\[no tests to run\]/ },
91
91
  ];
92
92
 
93
93
  /** Go prints this per package that has no test files at all. */
94
94
  const GO_NO_TEST_FILES = /\[no test files\]/;
95
+ /** Go prints this when test files exist but test selection (-run) matched nothing. */
96
+ const GO_NO_TESTS_TO_RUN = /\[no tests to run\]/;
95
97
  /**
96
98
  * Any sign that a Go package did run tests.
97
99
  *
@@ -111,10 +113,78 @@ const GO_RAN_SOMETHING = /^(?:(?:ok|FAIL)\s+(?!\d+\s)\S+|---\s+(?:PASS|FAIL|SKIP
111
113
  * `count` is null when no recognised runner stated one — which is not a
112
114
  * finding, only an absence of evidence.
113
115
  */
116
+ /**
117
+ * Did the command write nothing at all, on either stream?
118
+ *
119
+ * This is the one absence that is evidence rather than the lack of it. The
120
+ * one-sided floor exists because an unrecognised runner states no count, and
121
+ * hard-redding every runner not on the list would be worse than the hole it
122
+ * closes. But an unrecognised runner still *prints*: dots, a summary line,
123
+ * a package name, something. Zero bytes on both streams is not a dialect the
124
+ * list has yet to learn — it is a command that ran nothing.
125
+ *
126
+ * `pnpm -r test` on a workspace whose packages declare no test script is the
127
+ * shape that made this necessary: it exits 0, writes nothing anywhere, and
128
+ * was indistinguishable from a full suite by every signal the gate had.
129
+ *
130
+ * Resolved here and consumed in two places — the gate's floor and `init`'s
131
+ * oracle probe — because writing the rule once in each is how this project
132
+ * has repeatedly ended up with two answers to one question.
133
+ */
134
+ export function producedNoOutput(stdout = "", stderr = "") {
135
+ return `${stdout || ""}${stderr || ""}`.trim() === "";
136
+ }
137
+
138
+ /**
139
+ * Does this command claim to run a test suite?
140
+ *
141
+ * Silence alone cannot carry the verdict, and the first version of this rule
142
+ * assumed it could. `node --check index.js`, `tsc --noEmit`, `go vet ./...`
143
+ * and `python3 -m compileall -q .` all exit 0 having printed nothing — and
144
+ * they are honest static gates, two of which this kit writes itself for
145
+ * repositories that have no suite yet. Failing on silence alone hard-redded
146
+ * every one of them, which is the same first-run rejection of correct code
147
+ * that the whole collection floor is careful to avoid.
148
+ *
149
+ * Nothing in the *output* separates `pnpm -r test` from `tsc --noEmit`; both
150
+ * are empty. The difference is in what the command says it is. So this reads
151
+ * the command, the same way `isPlaceholderTestScript` does: a command that is
152
+ * recognisably a suite invocation and printed nothing ran no suite, while a
153
+ * static checker that printed nothing did exactly what it promised.
154
+ *
155
+ * One-sided in the safe direction, like everything else here. An unrecognised
156
+ * command is not treated as a suite, so an unusual runner keeps its advisory
157
+ * pass rather than becoming a hard red.
158
+ */
159
+ const TEST_SUITE_COMMAND =
160
+ /(?:^|\s|\/)(?:pytest|jest|vitest|mocha|ava|karma|jasmine|nyc|c8|tap|tape|rspec|minitest|phpunit|behave|nose2?|ginkgo|gotestsum|nextest)\b|\b(?:go|cargo|swift|dart|flutter|deno|bun|dotnet|mix|lein|sbt|gradlew?|mvn)\s+test\b|\bunittest\b|-m\s+(?:pytest|unittest)\b|\bnode\s+--test\b|(?:^|&&|;|\|)\s*(?:npm|pnpm|yarn|bun|npx)\b[^&;|]*?\btest\b/;
161
+
162
+ export function looksLikeTestSuiteCommand(cmd) {
163
+ return typeof cmd === "string" && TEST_SUITE_COMMAND.test(cmd);
164
+ }
165
+
114
166
  export function parseCollectedTests(stdout = "", stderr = "") {
115
167
  const text = `${stdout || ""}\n${stderr || ""}`;
116
168
  if (!text.trim()) return { count: null, runner: null };
117
169
 
170
+ // Pytest --collect-only states "collected N items" and "N tests collected",
171
+ // but executed 0 tests. A run that executed tests reports passed/failed/skipped.
172
+ if (
173
+ /=+\s*\d+\s+tests? collected\b/i.test(text) &&
174
+ !/=+\s*.*?(?:\d+\s+(?:passed|failed|skipped))\b/i.test(text)
175
+ ) {
176
+ return { count: 0, runner: "pytest" };
177
+ }
178
+
179
+ // Cargo states "running N tests" per target (lib, bin, integration tests, doc tests).
180
+ // A multi-target run with 0 unit tests and 58 integration tests must aggregate all
181
+ // targets rather than stopping at the first target-local zero (F15).
182
+ const cargoMatches = [...text.matchAll(/^\s*running\s+(\d+)\s+tests?\s*$/gm)];
183
+ if (cargoMatches.length > 0) {
184
+ const totalCargoTests = cargoMatches.reduce((sum, m) => sum + Number(m[1]), 0);
185
+ return { count: totalCargoTests, runner: "cargo" };
186
+ }
187
+
118
188
  // A stated count wins over a phrase that merely resembles one.
119
189
  //
120
190
  // `EXPLICIT_ZERO` used to be consulted first, so any output containing the
@@ -133,7 +203,13 @@ export function parseCollectedTests(stdout = "", stderr = "") {
133
203
 
134
204
  // Go states absence per package rather than as a count, so it needs its own
135
205
  // pass before the generic patterns.
136
- if (GO_NO_TEST_FILES.test(text) || GO_RAN_SOMETHING.test(text)) {
206
+ if (GO_NO_TEST_FILES.test(text) || GO_NO_TESTS_TO_RUN.test(text) || GO_RAN_SOMETHING.test(text)) {
207
+ const lines = text.split(/\r?\n/).map((l) => l.trim()).filter(Boolean);
208
+ const pkgLines = lines.filter((l) => /^(?:ok|FAIL|\?)\s+/.test(l));
209
+ const allEmpty =
210
+ pkgLines.length > 0 &&
211
+ pkgLines.every((l) => GO_NO_TEST_FILES.test(l) || GO_NO_TESTS_TO_RUN.test(l));
212
+ if (allEmpty) return { count: 0, runner: "go" };
137
213
  // Only a run where *no* package did anything is a zero: a monorepo where
138
214
  // one package has no tests and three do is a normal, healthy repository.
139
215
  if (!GO_RAN_SOMETHING.test(text)) return { count: 0, runner: "go" };
@@ -168,6 +244,30 @@ export function checkCollectionFloor(testResult, opts = {}) {
168
244
  return { ok: true, count: null, runner: null, reason: null };
169
245
  }
170
246
 
247
+ // A command that says it runs a suite, and printed nothing, ran no suite.
248
+ //
249
+ // Both halves are required. Silence alone would hard-red `tsc --noEmit` and
250
+ // `python3 -m compileall`, which are honest static gates this kit generates
251
+ // itself; the command shape alone would say nothing, because a real suite
252
+ // prints. Together they are decidable, and they are exactly `pnpm -r test`
253
+ // on a workspace whose packages declare no test script.
254
+ if (looksLikeTestSuiteCommand(testResult.command) && producedNoOutput(testResult.stdout, testResult.stderr)) {
255
+ return {
256
+ ok: false,
257
+ count: 0,
258
+ runner: null,
259
+ silent: true,
260
+ reason:
261
+ `The verification command ${testResult.command ? `${JSON.stringify(testResult.command)} ` : ""}` +
262
+ `exited 0 and wrote nothing at all — no test names, no summary, no count. ` +
263
+ `Every test runner prints something, so this command ran no suite, and approving this change ` +
264
+ `would certify nothing. A workspace command such as \`pnpm -r test\` does this when no package ` +
265
+ `declares a test script. Point verify.test at the suite that covers this repository ` +
266
+ `(often the root script rather than the recursive one), or — if this repository intentionally ` +
267
+ `uses only the scope and secret phases — set verify.required: false, which says so on the record.`,
268
+ };
269
+ }
270
+
171
271
  const { count, runner } = parseCollectedTests(testResult.stdout, testResult.stderr);
172
272
 
173
273
  // Deliberately one-sided: only a *stated* zero fails, because failing on