jules-orchestrator-kit 0.70.0 → 0.72.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agent/rules/jules-protocol.md +0 -1
- package/CHANGELOG.md +66 -0
- package/JULES_RULES_TEMPLATE.md +0 -1
- package/README.md +56 -1
- package/ROADMAP_V1.md +31 -6
- package/bin/agentctl.mjs +101 -8
- package/package.json +1 -1
- package/scripts/guard-reach-check.mjs +74 -8
- package/src/config.mjs +434 -5
- package/src/coverage.mjs +85 -13
- package/src/engine.mjs +106 -130
- package/src/git.mjs +104 -1
- package/src/guard-policy.mjs +581 -0
- package/src/ops/command-registry.mjs +109 -0
- package/src/ops/test-collection.mjs +103 -3
- package/src/security.mjs +554 -4
- package/src/stack-detector.mjs +73 -14
- package/src/test-paths.mjs +18 -1
- package/src/wizard-init.mjs +54 -27
- package/src/wizard-oracle.mjs +1 -1
- package/src/wizard-task.mjs +15 -1
|
@@ -543,6 +543,115 @@ export const COMMAND_REGISTRY = [
|
|
|
543
543
|
{ name: "json", type: "boolean", description: "Output JSON structured bootstrap result (-j)" },
|
|
544
544
|
],
|
|
545
545
|
},
|
|
546
|
+
{
|
|
547
|
+
id: "pr-harvest",
|
|
548
|
+
path: ["pr", "harvest"],
|
|
549
|
+
title: "pr harvest",
|
|
550
|
+
description: "Scan, audit and auto-merge verified agent pull requests",
|
|
551
|
+
category: "Operate",
|
|
552
|
+
mutates: true,
|
|
553
|
+
risk: "moderate",
|
|
554
|
+
interactive: "never",
|
|
555
|
+
requiresRepository: true,
|
|
556
|
+
shortcuts: ["harvest", "pr"],
|
|
557
|
+
examples: [
|
|
558
|
+
"agentctl pr harvest",
|
|
559
|
+
"agentctl pr harvest --auto",
|
|
560
|
+
"agentctl pr harvest --dry-run",
|
|
561
|
+
"agentctl pr harvest --json",
|
|
562
|
+
],
|
|
563
|
+
flags: [
|
|
564
|
+
{ name: "tier", type: "string", description: "Filter PRs by tier label" },
|
|
565
|
+
{ name: "limit", type: "string", description: "Maximum PRs to harvest" },
|
|
566
|
+
{ name: "auto", type: "boolean", description: "Automatically merge qualifying green PRs" },
|
|
567
|
+
{ name: "merge", type: "boolean", description: "Merge matching PRs" },
|
|
568
|
+
{ name: "allow-no-checks", type: "boolean", description: "Allow PR merge when no CI checks are configured" },
|
|
569
|
+
{ name: "dry-run", type: "boolean", description: "Simulate PR harvest without merging (-d)" },
|
|
570
|
+
{ name: "json", type: "boolean", description: "Output structured JSON harvest report (-j)" },
|
|
571
|
+
],
|
|
572
|
+
},
|
|
573
|
+
{
|
|
574
|
+
id: "session-get",
|
|
575
|
+
path: ["session", "get"],
|
|
576
|
+
title: "session get",
|
|
577
|
+
description: "Retrieve remote execution status for a session ID",
|
|
578
|
+
category: "Inspect",
|
|
579
|
+
mutates: false,
|
|
580
|
+
risk: "low",
|
|
581
|
+
interactive: "never",
|
|
582
|
+
requiresRepository: true,
|
|
583
|
+
shortcuts: ["session status", "session"],
|
|
584
|
+
examples: [
|
|
585
|
+
"agentctl session get <sessionId>",
|
|
586
|
+
"agentctl session get <sessionId> --dry-run",
|
|
587
|
+
"agentctl session get <sessionId> --json",
|
|
588
|
+
],
|
|
589
|
+
flags: [
|
|
590
|
+
{ name: "dry-run", type: "boolean", description: "Simulate session retrieval (-d)" },
|
|
591
|
+
{ name: "json", type: "boolean", description: "Output structured JSON session data (-j)" },
|
|
592
|
+
],
|
|
593
|
+
},
|
|
594
|
+
{
|
|
595
|
+
id: "plan-approve",
|
|
596
|
+
path: ["plan", "approve"],
|
|
597
|
+
title: "plan approve",
|
|
598
|
+
description: "Approve a pending execution plan for an agent session",
|
|
599
|
+
category: "Operate",
|
|
600
|
+
mutates: true,
|
|
601
|
+
risk: "moderate",
|
|
602
|
+
interactive: "never",
|
|
603
|
+
requiresRepository: true,
|
|
604
|
+
shortcuts: ["approve", "plan"],
|
|
605
|
+
examples: [
|
|
606
|
+
"agentctl plan approve <sessionId>",
|
|
607
|
+
"agentctl plan approve <sessionId> --dry-run",
|
|
608
|
+
"agentctl approve <sessionId>",
|
|
609
|
+
],
|
|
610
|
+
flags: [
|
|
611
|
+
{ name: "dry-run", type: "boolean", description: "Simulate plan approval (-d)" },
|
|
612
|
+
{ name: "json", type: "boolean", description: "Output structured JSON result (-j)" },
|
|
613
|
+
],
|
|
614
|
+
},
|
|
615
|
+
{
|
|
616
|
+
id: "lock",
|
|
617
|
+
path: ["lock"],
|
|
618
|
+
title: "lock",
|
|
619
|
+
description: "Multi-agent coordination locks: acquire, release, or view file status",
|
|
620
|
+
category: "Operate",
|
|
621
|
+
mutates: true,
|
|
622
|
+
risk: "low",
|
|
623
|
+
interactive: "never",
|
|
624
|
+
requiresRepository: true,
|
|
625
|
+
shortcuts: [],
|
|
626
|
+
examples: [
|
|
627
|
+
"agentctl lock status",
|
|
628
|
+
"agentctl lock acquire agent-1 task-1 src/main.js",
|
|
629
|
+
"agentctl lock release task-1",
|
|
630
|
+
],
|
|
631
|
+
flags: [
|
|
632
|
+
{ name: "json", type: "boolean", description: "Output structured JSON result (-j)" },
|
|
633
|
+
],
|
|
634
|
+
},
|
|
635
|
+
{
|
|
636
|
+
id: "evidence",
|
|
637
|
+
path: ["evidence"],
|
|
638
|
+
title: "evidence",
|
|
639
|
+
description: "Inspect and verify cryptographic evidence manifests",
|
|
640
|
+
category: "Inspect",
|
|
641
|
+
mutates: false,
|
|
642
|
+
risk: "low",
|
|
643
|
+
interactive: "never",
|
|
644
|
+
requiresRepository: true,
|
|
645
|
+
shortcuts: [],
|
|
646
|
+
examples: [
|
|
647
|
+
"agentctl evidence status",
|
|
648
|
+
"agentctl evidence verify",
|
|
649
|
+
"agentctl evidence export --json",
|
|
650
|
+
],
|
|
651
|
+
flags: [
|
|
652
|
+
{ name: "json", type: "boolean", description: "Output structured JSON result (-j)" },
|
|
653
|
+
],
|
|
654
|
+
},
|
|
546
655
|
];
|
|
547
656
|
|
|
548
657
|
/**
|
|
@@ -33,8 +33,7 @@ const COUNT_PATTERNS = [
|
|
|
33
33
|
// pytest — "collected 12 items", "12 passed", "no tests ran in 0.01s"
|
|
34
34
|
{ name: "pytest", re: /^\s*collected\s+(\d+)\s+items?/m },
|
|
35
35
|
{ name: "pytest", re: /=+\s*(\d+)\s+passed/m },
|
|
36
|
-
//
|
|
37
|
-
{ name: "cargo", re: /^\s*running\s+(\d+)\s+tests?\s*$/m },
|
|
36
|
+
// Cargo multi-target output is aggregated above COUNT_PATTERNS (F15)
|
|
38
37
|
// jest / vitest — "Tests: 12 passed, 12 total"
|
|
39
38
|
{ name: "jest", re: /^\s*Tests:\s+.*?(\d+)\s+total\s*$/m },
|
|
40
39
|
// mocha — "12 passing"
|
|
@@ -88,10 +87,13 @@ const EXPLICIT_ZERO = [
|
|
|
88
87
|
{ name: "gradle", re: /^>\s*Task\s+:\S*test\S*\s+NO-SOURCE\s*$/mi },
|
|
89
88
|
{ name: "ctest", re: /\bNo tests were found\b/i },
|
|
90
89
|
{ name: "flutter", re: /\bNo tests ran\.?/i },
|
|
90
|
+
{ name: "go", re: /\[no tests to run\]/ },
|
|
91
91
|
];
|
|
92
92
|
|
|
93
93
|
/** Go prints this per package that has no test files at all. */
|
|
94
94
|
const GO_NO_TEST_FILES = /\[no test files\]/;
|
|
95
|
+
/** Go prints this when test files exist but test selection (-run) matched nothing. */
|
|
96
|
+
const GO_NO_TESTS_TO_RUN = /\[no tests to run\]/;
|
|
95
97
|
/**
|
|
96
98
|
* Any sign that a Go package did run tests.
|
|
97
99
|
*
|
|
@@ -111,10 +113,78 @@ const GO_RAN_SOMETHING = /^(?:(?:ok|FAIL)\s+(?!\d+\s)\S+|---\s+(?:PASS|FAIL|SKIP
|
|
|
111
113
|
* `count` is null when no recognised runner stated one — which is not a
|
|
112
114
|
* finding, only an absence of evidence.
|
|
113
115
|
*/
|
|
116
|
+
/**
|
|
117
|
+
* Did the command write nothing at all, on either stream?
|
|
118
|
+
*
|
|
119
|
+
* This is the one absence that is evidence rather than the lack of it. The
|
|
120
|
+
* one-sided floor exists because an unrecognised runner states no count, and
|
|
121
|
+
* hard-redding every runner not on the list would be worse than the hole it
|
|
122
|
+
* closes. But an unrecognised runner still *prints*: dots, a summary line,
|
|
123
|
+
* a package name, something. Zero bytes on both streams is not a dialect the
|
|
124
|
+
* list has yet to learn — it is a command that ran nothing.
|
|
125
|
+
*
|
|
126
|
+
* `pnpm -r test` on a workspace whose packages declare no test script is the
|
|
127
|
+
* shape that made this necessary: it exits 0, writes nothing anywhere, and
|
|
128
|
+
* was indistinguishable from a full suite by every signal the gate had.
|
|
129
|
+
*
|
|
130
|
+
* Resolved here and consumed in two places — the gate's floor and `init`'s
|
|
131
|
+
* oracle probe — because writing the rule once in each is how this project
|
|
132
|
+
* has repeatedly ended up with two answers to one question.
|
|
133
|
+
*/
|
|
134
|
+
export function producedNoOutput(stdout = "", stderr = "") {
|
|
135
|
+
return `${stdout || ""}${stderr || ""}`.trim() === "";
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/**
|
|
139
|
+
* Does this command claim to run a test suite?
|
|
140
|
+
*
|
|
141
|
+
* Silence alone cannot carry the verdict, and the first version of this rule
|
|
142
|
+
* assumed it could. `node --check index.js`, `tsc --noEmit`, `go vet ./...`
|
|
143
|
+
* and `python3 -m compileall -q .` all exit 0 having printed nothing — and
|
|
144
|
+
* they are honest static gates, two of which this kit writes itself for
|
|
145
|
+
* repositories that have no suite yet. Failing on silence alone hard-redded
|
|
146
|
+
* every one of them, which is the same first-run rejection of correct code
|
|
147
|
+
* that the whole collection floor is careful to avoid.
|
|
148
|
+
*
|
|
149
|
+
* Nothing in the *output* separates `pnpm -r test` from `tsc --noEmit`; both
|
|
150
|
+
* are empty. The difference is in what the command says it is. So this reads
|
|
151
|
+
* the command, the same way `isPlaceholderTestScript` does: a command that is
|
|
152
|
+
* recognisably a suite invocation and printed nothing ran no suite, while a
|
|
153
|
+
* static checker that printed nothing did exactly what it promised.
|
|
154
|
+
*
|
|
155
|
+
* One-sided in the safe direction, like everything else here. An unrecognised
|
|
156
|
+
* command is not treated as a suite, so an unusual runner keeps its advisory
|
|
157
|
+
* pass rather than becoming a hard red.
|
|
158
|
+
*/
|
|
159
|
+
const TEST_SUITE_COMMAND =
|
|
160
|
+
/(?:^|\s|\/)(?:pytest|jest|vitest|mocha|ava|karma|jasmine|nyc|c8|tap|tape|rspec|minitest|phpunit|behave|nose2?|ginkgo|gotestsum|nextest)\b|\b(?:go|cargo|swift|dart|flutter|deno|bun|dotnet|mix|lein|sbt|gradlew?|mvn)\s+test\b|\bunittest\b|-m\s+(?:pytest|unittest)\b|\bnode\s+--test\b|(?:^|&&|;|\|)\s*(?:npm|pnpm|yarn|bun|npx)\b[^&;|]*?\btest\b/;
|
|
161
|
+
|
|
162
|
+
export function looksLikeTestSuiteCommand(cmd) {
|
|
163
|
+
return typeof cmd === "string" && TEST_SUITE_COMMAND.test(cmd);
|
|
164
|
+
}
|
|
165
|
+
|
|
114
166
|
export function parseCollectedTests(stdout = "", stderr = "") {
|
|
115
167
|
const text = `${stdout || ""}\n${stderr || ""}`;
|
|
116
168
|
if (!text.trim()) return { count: null, runner: null };
|
|
117
169
|
|
|
170
|
+
// Pytest --collect-only states "collected N items" and "N tests collected",
|
|
171
|
+
// but executed 0 tests. A run that executed tests reports passed/failed/skipped.
|
|
172
|
+
if (
|
|
173
|
+
/=+\s*\d+\s+tests? collected\b/i.test(text) &&
|
|
174
|
+
!/=+\s*.*?(?:\d+\s+(?:passed|failed|skipped))\b/i.test(text)
|
|
175
|
+
) {
|
|
176
|
+
return { count: 0, runner: "pytest" };
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
// Cargo states "running N tests" per target (lib, bin, integration tests, doc tests).
|
|
180
|
+
// A multi-target run with 0 unit tests and 58 integration tests must aggregate all
|
|
181
|
+
// targets rather than stopping at the first target-local zero (F15).
|
|
182
|
+
const cargoMatches = [...text.matchAll(/^\s*running\s+(\d+)\s+tests?\s*$/gm)];
|
|
183
|
+
if (cargoMatches.length > 0) {
|
|
184
|
+
const totalCargoTests = cargoMatches.reduce((sum, m) => sum + Number(m[1]), 0);
|
|
185
|
+
return { count: totalCargoTests, runner: "cargo" };
|
|
186
|
+
}
|
|
187
|
+
|
|
118
188
|
// A stated count wins over a phrase that merely resembles one.
|
|
119
189
|
//
|
|
120
190
|
// `EXPLICIT_ZERO` used to be consulted first, so any output containing the
|
|
@@ -133,7 +203,13 @@ export function parseCollectedTests(stdout = "", stderr = "") {
|
|
|
133
203
|
|
|
134
204
|
// Go states absence per package rather than as a count, so it needs its own
|
|
135
205
|
// pass before the generic patterns.
|
|
136
|
-
if (GO_NO_TEST_FILES.test(text) || GO_RAN_SOMETHING.test(text)) {
|
|
206
|
+
if (GO_NO_TEST_FILES.test(text) || GO_NO_TESTS_TO_RUN.test(text) || GO_RAN_SOMETHING.test(text)) {
|
|
207
|
+
const lines = text.split(/\r?\n/).map((l) => l.trim()).filter(Boolean);
|
|
208
|
+
const pkgLines = lines.filter((l) => /^(?:ok|FAIL|\?)\s+/.test(l));
|
|
209
|
+
const allEmpty =
|
|
210
|
+
pkgLines.length > 0 &&
|
|
211
|
+
pkgLines.every((l) => GO_NO_TEST_FILES.test(l) || GO_NO_TESTS_TO_RUN.test(l));
|
|
212
|
+
if (allEmpty) return { count: 0, runner: "go" };
|
|
137
213
|
// Only a run where *no* package did anything is a zero: a monorepo where
|
|
138
214
|
// one package has no tests and three do is a normal, healthy repository.
|
|
139
215
|
if (!GO_RAN_SOMETHING.test(text)) return { count: 0, runner: "go" };
|
|
@@ -168,6 +244,30 @@ export function checkCollectionFloor(testResult, opts = {}) {
|
|
|
168
244
|
return { ok: true, count: null, runner: null, reason: null };
|
|
169
245
|
}
|
|
170
246
|
|
|
247
|
+
// A command that says it runs a suite, and printed nothing, ran no suite.
|
|
248
|
+
//
|
|
249
|
+
// Both halves are required. Silence alone would hard-red `tsc --noEmit` and
|
|
250
|
+
// `python3 -m compileall`, which are honest static gates this kit generates
|
|
251
|
+
// itself; the command shape alone would say nothing, because a real suite
|
|
252
|
+
// prints. Together they are decidable, and they are exactly `pnpm -r test`
|
|
253
|
+
// on a workspace whose packages declare no test script.
|
|
254
|
+
if (looksLikeTestSuiteCommand(testResult.command) && producedNoOutput(testResult.stdout, testResult.stderr)) {
|
|
255
|
+
return {
|
|
256
|
+
ok: false,
|
|
257
|
+
count: 0,
|
|
258
|
+
runner: null,
|
|
259
|
+
silent: true,
|
|
260
|
+
reason:
|
|
261
|
+
`The verification command ${testResult.command ? `${JSON.stringify(testResult.command)} ` : ""}` +
|
|
262
|
+
`exited 0 and wrote nothing at all — no test names, no summary, no count. ` +
|
|
263
|
+
`Every test runner prints something, so this command ran no suite, and approving this change ` +
|
|
264
|
+
`would certify nothing. A workspace command such as \`pnpm -r test\` does this when no package ` +
|
|
265
|
+
`declares a test script. Point verify.test at the suite that covers this repository ` +
|
|
266
|
+
`(often the root script rather than the recursive one), or — if this repository intentionally ` +
|
|
267
|
+
`uses only the scope and secret phases — set verify.required: false, which says so on the record.`,
|
|
268
|
+
};
|
|
269
|
+
}
|
|
270
|
+
|
|
171
271
|
const { count, runner } = parseCollectedTests(testResult.stdout, testResult.stderr);
|
|
172
272
|
|
|
173
273
|
// Deliberately one-sided: only a *stated* zero fails, because failing on
|