jules-orchestrator-kit 0.64.0 → 0.66.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -3
- package/bin/agentctl.mjs +18 -2
- package/package.json +1 -1
- package/scripts/guard-reach-check.mjs +19 -3
- package/src/engine.mjs +56 -1
- package/src/guard-policy.mjs +72 -0
- package/src/memory.mjs +0 -0
- package/src/ops/test-collection.mjs +44 -10
- package/src/security.mjs +119 -6
- package/src/stack-detector.mjs +50 -0
- package/src/wizard-init.mjs +61 -11
package/README.md
CHANGED
|
@@ -49,12 +49,15 @@
|
|
|
49
49
|
<a id="quickstart"></a>
|
|
50
50
|
## Quickstart
|
|
51
51
|
|
|
52
|
-
Get running in any repository in 3 commands
|
|
52
|
+
Get running in any repository in 3 commands. `init` asks seven questions and
|
|
53
|
+
fills in a sensible answer for each; `--yes` accepts all of them, detects the
|
|
54
|
+
stack, and probes the test command it picked before writing it down.
|
|
53
55
|
|
|
54
56
|
```bash
|
|
55
57
|
# 1. Scaffold config, AGENTS.md, role prompts and guardrails
|
|
56
58
|
# (auto-detects Python, Rust, Go, Node, PHP, etc.)
|
|
57
|
-
|
|
59
|
+
# Drop --yes to choose provider, plan, profile and workflows yourself.
|
|
60
|
+
npx jules-orchestrator-kit init --yes
|
|
58
61
|
```
|
|
59
62
|
|
|
60
63
|
```bash
|
|
@@ -204,7 +207,7 @@ To maximize PR merge rates, dispatch tasks according to deterministic boundaries
|
|
|
204
207
|
* **Fail-Closed Security & Secret Redaction:** Evaluates explicit Deny rules before Allow rules against canonicalized, case-folded paths. Redacts high-entropy keys and base64-encoded credentials (such as Kubernetes `Secret` manifests).
|
|
205
208
|
* **Complexity & Cost Router:** Zero-dependency heuristic classifier (`src/router.mjs`) routing mechanical tasks to lightweight models while reserving primary models for complex refactors, with a `node --check` syntax-verification gate that transparently escalates a FAST-tier result to the primary provider if it left broken JS on disk.
|
|
206
209
|
* **Terminal UI & Diagnostic Matrix (`agentctl doctor`):** Interactive terminal dashboard, task sidecar manager, and automated transactional self-repair.
|
|
207
|
-
* **Verified Test Suite:** Tested with **
|
|
210
|
+
* **Verified Test Suite:** Tested with **1133 unit tests across 163 suites passing in < 15.0s**.
|
|
208
211
|
|
|
209
212
|
<br/>
|
|
210
213
|
|
package/bin/agentctl.mjs
CHANGED
|
@@ -429,6 +429,14 @@ async function main() {
|
|
|
429
429
|
if (p.violations) {
|
|
430
430
|
p.violations.forEach((v) => console.log(` - Violation: ${v.file} (Rule: ${v.rule})`));
|
|
431
431
|
}
|
|
432
|
+
if (p.unverified) {
|
|
433
|
+
console.log(` - Unverified: ${p.unverified}`);
|
|
434
|
+
}
|
|
435
|
+
if (p.setup) {
|
|
436
|
+
console.log(` - Setup: accepted ${p.setup.length} gate scaffold file(s) this repository did not have yet`);
|
|
437
|
+
p.setup.forEach((f) => console.log(` ${f}`));
|
|
438
|
+
console.log(` Commit them to the base branch and the full protect rules apply from then on.`);
|
|
439
|
+
}
|
|
432
440
|
if (p.findings) {
|
|
433
441
|
p.findings.forEach((f) => console.log(` - [${f.severity}] ${f.type}: ${f.description}`));
|
|
434
442
|
}
|
|
@@ -2662,13 +2670,21 @@ async function main() {
|
|
|
2662
2670
|
const taskId = values.task || "unknown";
|
|
2663
2671
|
const agent = values.agent || "jules";
|
|
2664
2672
|
|
|
2673
|
+
const { CONFIRM_AFTER } = await import("../src/memory.mjs");
|
|
2665
2674
|
const res = harvestFailure(root, { exitCode, logPath, diffText, taskId, agent });
|
|
2666
2675
|
if (values.json) {
|
|
2667
2676
|
console.log(JSON.stringify(res, null, 2));
|
|
2668
2677
|
} else {
|
|
2669
2678
|
if (res.status === "HARVESTED") {
|
|
2670
|
-
|
|
2671
|
-
|
|
2679
|
+
// No "Solution:" line any more, because there is no solution: the
|
|
2680
|
+
// repair loop exhausting its budget is the definition of not having
|
|
2681
|
+
// found one. It used to print a fixed sentence here and store it as
|
|
2682
|
+
// if it were a remedy.
|
|
2683
|
+
console.log(`🌾 Recorded failure observation: ${res.candidate.trigger}`);
|
|
2684
|
+
console.log(` Seen ${res.occurrences}× — ${res.confirmed ? "stated as a rule" : `not injected into prompts until it recurs ${CONFIRM_AFTER}×`}`);
|
|
2685
|
+
if (!res.confirmed) {
|
|
2686
|
+
console.log(` Record an actual fix for it with: agentctl learning add "<trigger>" "<solution>"`);
|
|
2687
|
+
}
|
|
2672
2688
|
} else {
|
|
2673
2689
|
console.log(`⚠️ Harvest rejected: ${res.reason}`);
|
|
2674
2690
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "jules-orchestrator-kit",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.66.0",
|
|
4
4
|
"description": "Zero-dependency safety gatekeeper, test oracle generator, and multi-agent coordination protocol for autonomous coding agents — Google Jules, Claude Code, Codex and Gemini CLI.",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
@@ -49,8 +49,10 @@ import { parseCollectedTests } from "../src/ops/test-collection.mjs";
|
|
|
49
49
|
import {
|
|
50
50
|
TEST_PATH_CASES,
|
|
51
51
|
TAMPER_CANARIES,
|
|
52
|
+
MULTILINE_CANARIES,
|
|
52
53
|
PREDICATE_MUTANTS,
|
|
53
54
|
EMPTY_RUN_CANARIES,
|
|
55
|
+
COUNTED_RUN_CANARIES,
|
|
54
56
|
SCOPE_CANARIES,
|
|
55
57
|
INNOCENT_EDITS,
|
|
56
58
|
UNREADABLE_DIALECTS,
|
|
@@ -67,6 +69,9 @@ import {
|
|
|
67
69
|
function canaryDiff(c) {
|
|
68
70
|
const ctx = c.context || "// context";
|
|
69
71
|
const lines = [`--- a/${c.file}`, `+++ b/${c.file}`, "@@ -1,20 +1,20 @@", ` ${ctx}`];
|
|
72
|
+
// Unchanged lines the change sits inside. An assertion whose keyword is on
|
|
73
|
+
// one of these is the shape the line-level denominator could not see.
|
|
74
|
+
for (const l of c.lead || []) lines.push(` ${l}`);
|
|
70
75
|
for (const l of c.removed) lines.push(`-${l}`);
|
|
71
76
|
for (const l of c.added) lines.push(`+${l}`);
|
|
72
77
|
lines.push(` ${ctx}`);
|
|
@@ -105,6 +110,17 @@ const add = (name, ok, detail) => {
|
|
|
105
110
|
|
|
106
111
|
{
|
|
107
112
|
const missed = EMPTY_RUN_CANARIES.filter((c) => parseCollectedTests(c.output, "").count !== 0);
|
|
113
|
+
const undercounted = COUNTED_RUN_CANARIES.filter((c) => {
|
|
114
|
+
const n = parseCollectedTests(c.output, "").count;
|
|
115
|
+
return n === null || n < c.atLeast;
|
|
116
|
+
});
|
|
117
|
+
add(
|
|
118
|
+
"policy: a stated count is never read as empty",
|
|
119
|
+
undercounted.length === 0,
|
|
120
|
+
undercounted.length
|
|
121
|
+
? undercounted.map((c) => `${c.id} (${c.why})`).join("; ")
|
|
122
|
+
: `${COUNTED_RUN_CANARIES.length} healthy runs counted, not rejected`
|
|
123
|
+
);
|
|
108
124
|
add(
|
|
109
125
|
"policy: empty-run detection",
|
|
110
126
|
missed.length === 0,
|
|
@@ -118,7 +134,7 @@ const canaryResults = new Map();
|
|
|
118
134
|
const silent = [];
|
|
119
135
|
const noDenominator = [];
|
|
120
136
|
const noAssertions = [];
|
|
121
|
-
for (const c of TAMPER_CANARIES) {
|
|
137
|
+
for (const c of [...TAMPER_CANARIES, ...MULTILINE_CANARIES]) {
|
|
122
138
|
const res = checkTestTampering(canaryDiff(c));
|
|
123
139
|
const hit = (res.violations || []).some((v) => v.type === c.expect);
|
|
124
140
|
canaryResults.set(c.id, hit);
|
|
@@ -132,7 +148,7 @@ const canaryResults = new Map();
|
|
|
132
148
|
noAssertions.push(`${c.id} (${res.assertionsSeen} assertions parsed)`);
|
|
133
149
|
}
|
|
134
150
|
}
|
|
135
|
-
add("canaries: every tamper rule still fires", silent.length === 0, silent.length ? silent.join("; ") : `${TAMPER_CANARIES.length} canaries red as required`);
|
|
151
|
+
add("canaries: every tamper rule still fires", silent.length === 0, silent.length ? silent.join("; ") : `${TAMPER_CANARIES.length + MULTILINE_CANARIES.length} canaries red as required`);
|
|
136
152
|
add("canaries: every finding carries a denominator", noDenominator.length === 0, noDenominator.length ? noDenominator.join(", ") : "inputsSeen > 0 on every hit");
|
|
137
153
|
add("canaries: assertion rules parsed an assertion", noAssertions.length === 0, noAssertions.length ? noAssertions.join(", ") : "assertionsSeen > 0 on every assertion finding");
|
|
138
154
|
}
|
|
@@ -174,7 +190,7 @@ const canaryResults = new Map();
|
|
|
174
190
|
const survivors = [];
|
|
175
191
|
for (const mutant of PREDICATE_MUTANTS) {
|
|
176
192
|
let killed = false;
|
|
177
|
-
for (const c of TAMPER_CANARIES) {
|
|
193
|
+
for (const c of [...TAMPER_CANARIES, ...MULTILINE_CANARIES]) {
|
|
178
194
|
// Only canaries the healthy predicate catches can kill a mutant.
|
|
179
195
|
if (!canaryResults.get(c.id)) continue;
|
|
180
196
|
const res = checkTestTampering(canaryDiff(c), { isTestPath: mutant.fn });
|
package/src/engine.mjs
CHANGED
|
@@ -204,6 +204,16 @@ export async function gate(opts = {}) {
|
|
|
204
204
|
// Read from the base commit like every other trusted field: an
|
|
205
205
|
// uncommitted `required: false` must not be able to switch the gate off.
|
|
206
206
|
required: parsed.verify?.required !== undefined ? parsed.verify.required !== false : config.verify.required !== false,
|
|
207
|
+
// The floor the collection check applies. Omitting it here meant
|
|
208
|
+
// `verify.minTests` was silently dropped and always defaulted to 1
|
|
209
|
+
// — while the failure message told the operator to set exactly
|
|
210
|
+
// that. A remediation hint that does nothing is worse than none.
|
|
211
|
+
minTests:
|
|
212
|
+
parsed.verify?.minTests !== undefined
|
|
213
|
+
? parsed.verify.minTests
|
|
214
|
+
: parsed.verify?.min_tests !== undefined
|
|
215
|
+
? parsed.verify.min_tests
|
|
216
|
+
: config.verify.minTests,
|
|
207
217
|
scope: parsed.verify?.scope || config.verify.scope || "global",
|
|
208
218
|
timeoutMs: parsed.verify?.timeoutMs || parsed.verify?.timeout_ms || config.verify.timeoutMs,
|
|
209
219
|
};
|
|
@@ -249,7 +259,45 @@ export async function gate(opts = {}) {
|
|
|
249
259
|
violation.file = link;
|
|
250
260
|
}
|
|
251
261
|
|
|
252
|
-
|
|
262
|
+
// Bootstrap: the files that bring a repository under the gate are not agent
|
|
263
|
+
// edits to the gate.
|
|
264
|
+
//
|
|
265
|
+
// `init` writes `.agent/**` and then tells the user to commit it. Doing
|
|
266
|
+
// exactly that produced Exit 3 on the very first run, because the base
|
|
267
|
+
// branch does not have the commit yet and every scaffolded path matches
|
|
268
|
+
// BUILTIN_PROTECT or BUILTIN_DENY. The advice printed alongside it was
|
|
269
|
+
// `--allow-protected` — so a newcomer's first lesson was how to switch the
|
|
270
|
+
// scope guard off. A gate that refuses its own installation is not strict,
|
|
271
|
+
// it is broken.
|
|
272
|
+
//
|
|
273
|
+
// Narrow on purpose, and only where it cannot weaken anything: the base
|
|
274
|
+
// commit must have no gate config at all — in which case `trustedScope` is
|
|
275
|
+
// already built-ins only and nothing in the added files is trusted — and
|
|
276
|
+
// every violating path must be scaffold that the base does not have. A
|
|
277
|
+
// repository already under the gate keeps the full rule, so an agent still
|
|
278
|
+
// cannot touch the policy it is governed by.
|
|
279
|
+
let acceptedScaffold = [];
|
|
280
|
+
if (!scopeResult.ok && !trustedConfigRaw) {
|
|
281
|
+
const violations = scopeResult.violations || [];
|
|
282
|
+
const isScaffold = (f) => typeof f === "string" && f.replace(/\\/g, "/").startsWith(".agent/");
|
|
283
|
+
if (
|
|
284
|
+
violations.length > 0 &&
|
|
285
|
+
violations.every((v) => isScaffold(v.file) && showFromOrigin(root, base, v.file) === null)
|
|
286
|
+
) {
|
|
287
|
+
acceptedScaffold = violations.map((v) => v.file);
|
|
288
|
+
scopeResult.violations = [];
|
|
289
|
+
scopeResult.ok = true;
|
|
290
|
+
}
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
phases.push({
|
|
294
|
+
phase: "scope",
|
|
295
|
+
ok: scopeResult.ok,
|
|
296
|
+
violations: scopeResult.violations,
|
|
297
|
+
// Reported, never silent: the operator has to see that the gate accepted
|
|
298
|
+
// files it would otherwise have blocked, and why.
|
|
299
|
+
...(acceptedScaffold.length > 0 ? { setup: acceptedScaffold } : {}),
|
|
300
|
+
});
|
|
253
301
|
appendTelemetry(root, "gate_phase", { phase: "scope", ok: scopeResult.ok });
|
|
254
302
|
if (progressBus && progressToken) {
|
|
255
303
|
progressBus.reportProgress(progressToken, 25, 100, "Phase 1/4: Scope Guard verification complete");
|
|
@@ -644,6 +692,13 @@ export async function gate(opts = {}) {
|
|
|
644
692
|
postTestHash,
|
|
645
693
|
tamperDetected: testTampered,
|
|
646
694
|
},
|
|
695
|
+
// Verified by exit code alone. Not a failure, and not an approval either:
|
|
696
|
+
// an unrecognised runner passes on purpose, but the operator has to be
|
|
697
|
+
// able to tell that apart from a suite the gate actually counted.
|
|
698
|
+
...(collectionFloor.unverified ? { unverified: collectionFloor.note } : {}),
|
|
699
|
+
...(collectionFloor.count !== null && collectionFloor.count !== undefined
|
|
700
|
+
? { testsCollected: collectionFloor.count, runner: collectionFloor.runner }
|
|
701
|
+
: {}),
|
|
647
702
|
});
|
|
648
703
|
phases.push({
|
|
649
704
|
phase: "evidence",
|
package/src/guard-policy.mjs
CHANGED
|
@@ -382,6 +382,15 @@ export const INNOCENT_EDITS = [
|
|
|
382
382
|
added: ['const calc = require("../src/calc");'],
|
|
383
383
|
why: "`require(` is an import, not a claim — the loose net must not read it as one",
|
|
384
384
|
},
|
|
385
|
+
{
|
|
386
|
+
id: "reword-comment-inside-assertion/node",
|
|
387
|
+
file: "test/calc.test.js",
|
|
388
|
+
context: "// arithmetic",
|
|
389
|
+
lead: [" assert.equal(", " add(1, 2),"],
|
|
390
|
+
removed: [" 3, // the answer", " );"],
|
|
391
|
+
added: [" 3, // the correct answer", " );"],
|
|
392
|
+
why: "line comments were never stripped — `copyCode(i)` left `pending` at the comment and the copy after the loop put it back",
|
|
393
|
+
},
|
|
385
394
|
{
|
|
386
395
|
id: "rename-helper/node",
|
|
387
396
|
file: "test/calc.test.js",
|
|
@@ -479,3 +488,66 @@ IMPORT_EXTRACTION_CASES.push(
|
|
|
479
488
|
why: "same, in the other place examples live",
|
|
480
489
|
}
|
|
481
490
|
);
|
|
491
|
+
|
|
492
|
+
/**
|
|
493
|
+
* Runs that stated a count, which must never be read as empty.
|
|
494
|
+
*
|
|
495
|
+
* The floor was written to be one-sided — only a *stated* zero fails — and
|
|
496
|
+
* then a phrase was allowed to outrank a statement. A healthy 190-test TAP
|
|
497
|
+
* suite whose one skipped fixture printed `# SKIP no tests found` was
|
|
498
|
+
* rejected as empty and attributed to Jest, in a repository that does not
|
|
499
|
+
* use Jest. A false red on a correct repository is how a user learns the
|
|
500
|
+
* gate is broken and turns it off.
|
|
501
|
+
*/
|
|
502
|
+
export const COUNTED_RUN_CANARIES = [
|
|
503
|
+
{
|
|
504
|
+
id: "tap with a skip message",
|
|
505
|
+
output: "TAP version 13\n# Subtest: performance\n # SKIP no tests found\nok 1 - performance # SKIP\n1..191\n# tests 191\n# pass 190\n# skip 1",
|
|
506
|
+
atLeast: 1,
|
|
507
|
+
why: "`no tests found` inside a skip comment is not a statement about the run",
|
|
508
|
+
},
|
|
509
|
+
{
|
|
510
|
+
id: "pytest mentioning an empty module",
|
|
511
|
+
output: "collected 12 items\n\ntests/test_a.py ............\n\n12 passed in 0.3s",
|
|
512
|
+
atLeast: 1,
|
|
513
|
+
why: "a stated count is present and must win",
|
|
514
|
+
},
|
|
515
|
+
{
|
|
516
|
+
id: "go, verbose, two tests",
|
|
517
|
+
output: "--- PASS: TestAdd (0.00s)\n--- PASS: TestSub (0.00s)\nok \texample.com/lib\t0.004s",
|
|
518
|
+
atLeast: 1,
|
|
519
|
+
why: "Go's own `ok <package>` line, which a bare `^ok\\s` confused with TAP's `ok 1 - name`",
|
|
520
|
+
},
|
|
521
|
+
];
|
|
522
|
+
|
|
523
|
+
/**
|
|
524
|
+
* Changes that sit inside an assertion whose keyword never appears in the diff.
|
|
525
|
+
*
|
|
526
|
+
* Counting `+`/`-` lines answered zero here, and nothing among the changed
|
|
527
|
+
* lines looked assertion-shaped either, so the guard reported a clean PASS on
|
|
528
|
+
* a five-element expected list rewritten to one element to match broken
|
|
529
|
+
* output. Measured on a real repository: five green phases, `APPROVED`.
|
|
530
|
+
*
|
|
531
|
+
* The statement machinery that pairs rewrites already assembles context lines
|
|
532
|
+
* together with changed ones. The denominator has to use it.
|
|
533
|
+
*/
|
|
534
|
+
export const MULTILINE_CANARIES = [
|
|
535
|
+
{
|
|
536
|
+
id: "expectation/python-multiline",
|
|
537
|
+
file: "tests/test_headers.py",
|
|
538
|
+
context: "# headers",
|
|
539
|
+
lead: [" def test_parse(self):", " self.assertEqual(", " _parse_http_header(x),"],
|
|
540
|
+
removed: [' [("text/xml", {}),', ' ("text/plain", {}),', ' ("more", {})])'],
|
|
541
|
+
added: [' [("text/xml", {"tampered": True})])'],
|
|
542
|
+
expect: "ASSERTION_EXPECTATION_CHANGED",
|
|
543
|
+
},
|
|
544
|
+
{
|
|
545
|
+
id: "expectation/js-multiline",
|
|
546
|
+
file: "test/headers.test.js",
|
|
547
|
+
context: "// headers",
|
|
548
|
+
lead: [" assert.deepStrictEqual(", " parse(input),"],
|
|
549
|
+
removed: [" [{ type: \"a\" }, { type: \"b\" }, { type: \"c\" }]", " );"],
|
|
550
|
+
added: [" [{ type: \"a\" }]", " );"],
|
|
551
|
+
expect: "ASSERTION_EXPECTATION_CHANGED",
|
|
552
|
+
},
|
|
553
|
+
];
|
package/src/memory.mjs
CHANGED
|
Binary file
|
|
@@ -71,8 +71,15 @@ const EXPLICIT_ZERO = [
|
|
|
71
71
|
|
|
72
72
|
/** Go prints this per package that has no test files at all. */
|
|
73
73
|
const GO_NO_TEST_FILES = /\[no test files\]/;
|
|
74
|
-
/**
|
|
75
|
-
|
|
74
|
+
/**
|
|
75
|
+
* Any sign that a Go package did run tests.
|
|
76
|
+
*
|
|
77
|
+
* The negative lookahead is what separates Go from TAP. `ok 1 - performance`
|
|
78
|
+
* is a TAP result line and `ok example.com/lib 0.004s` is a Go package
|
|
79
|
+
* summary, and a bare `^ok\s` matched both — so a 190-test TAP suite was
|
|
80
|
+
* classified as Go and reported as having stated no count at all.
|
|
81
|
+
*/
|
|
82
|
+
const GO_RAN_SOMETHING = /^(?:(?:ok|FAIL)\s+(?!\d+\s)\S+|---\s+(?:PASS|FAIL|SKIP):?\s)/m;
|
|
76
83
|
|
|
77
84
|
/**
|
|
78
85
|
* Read a collected-test count out of a runner's output.
|
|
@@ -87,8 +94,20 @@ export function parseCollectedTests(stdout = "", stderr = "") {
|
|
|
87
94
|
const text = `${stdout || ""}\n${stderr || ""}`;
|
|
88
95
|
if (!text.trim()) return { count: null, runner: null };
|
|
89
96
|
|
|
90
|
-
|
|
91
|
-
|
|
97
|
+
// A stated count wins over a phrase that merely resembles one.
|
|
98
|
+
//
|
|
99
|
+
// `EXPLICIT_ZERO` used to be consulted first, so any output containing the
|
|
100
|
+
// words "no tests found" was read as a zero — including a healthy TAP run
|
|
101
|
+
// of 190 passing tests whose one skipped fixture printed
|
|
102
|
+
// `# SKIP no tests found`. The gate rejected the suite as empty and named
|
|
103
|
+
// Jest as the runner, in a repository that does not use Jest. A phrase
|
|
104
|
+
// appears anywhere in a stream; a count is stated deliberately, so the
|
|
105
|
+
// count is the better witness and has to be asked first.
|
|
106
|
+
for (const rule of COUNT_PATTERNS) {
|
|
107
|
+
const m = rule.re.exec(text);
|
|
108
|
+
if (!m) continue;
|
|
109
|
+
const n = Number(m[1]);
|
|
110
|
+
if (Number.isFinite(n)) return { count: n, runner: rule.name };
|
|
92
111
|
}
|
|
93
112
|
|
|
94
113
|
// Go states absence per package rather than as a count, so it needs its own
|
|
@@ -104,11 +123,9 @@ export function parseCollectedTests(stdout = "", stderr = "") {
|
|
|
104
123
|
return { count: perTest && perTest.length > 0 ? perTest.length : null, runner: "go" };
|
|
105
124
|
}
|
|
106
125
|
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
if (
|
|
110
|
-
const n = Number(m[1]);
|
|
111
|
-
if (Number.isFinite(n)) return { count: n, runner: rule.name };
|
|
126
|
+
// Only now: no runner stated a number, so a declared absence is all there is.
|
|
127
|
+
for (const rule of EXPLICIT_ZERO) {
|
|
128
|
+
if (rule.re.test(text)) return { count: 0, runner: rule.name };
|
|
112
129
|
}
|
|
113
130
|
|
|
114
131
|
return { count: null, runner: null };
|
|
@@ -131,7 +148,24 @@ export function checkCollectionFloor(testResult, opts = {}) {
|
|
|
131
148
|
}
|
|
132
149
|
|
|
133
150
|
const { count, runner } = parseCollectedTests(testResult.stdout, testResult.stderr);
|
|
134
|
-
|
|
151
|
+
|
|
152
|
+
// Deliberately one-sided: only a *stated* zero fails, because failing on
|
|
153
|
+
// "I could not tell" would break every runner not on the list. But passing
|
|
154
|
+
// and saying nothing makes `echo "all tests passed"` indistinguishable from
|
|
155
|
+
// a real suite, so the absence of evidence is reported as an absence.
|
|
156
|
+
if (count === null) {
|
|
157
|
+
return {
|
|
158
|
+
ok: true,
|
|
159
|
+
count: null,
|
|
160
|
+
runner: null,
|
|
161
|
+
reason: null,
|
|
162
|
+
unverified: true,
|
|
163
|
+
note:
|
|
164
|
+
`The verification command exited 0, but no recognised test runner stated how many tests it ran, ` +
|
|
165
|
+
`so the gate cannot tell a full suite from a command that ran nothing. Verified by exit code alone.`,
|
|
166
|
+
};
|
|
167
|
+
}
|
|
168
|
+
if (count >= minTests) {
|
|
135
169
|
return { ok: true, count, runner, reason: null };
|
|
136
170
|
}
|
|
137
171
|
|
package/src/security.mjs
CHANGED
|
@@ -1523,8 +1523,13 @@ function stripComments(text, lang) {
|
|
|
1523
1523
|
continue;
|
|
1524
1524
|
}
|
|
1525
1525
|
|
|
1526
|
-
|
|
1527
|
-
|
|
1526
|
+
// `pending = n` is the whole fix. `copyCode(i)` copies the code up to
|
|
1527
|
+
// the comment and leaves `pending` sitting at its start; the
|
|
1528
|
+
// `copyCode(n)` after this loop then copied the comment straight back
|
|
1529
|
+
// in, so no line comment has ever been stripped. Block comments were,
|
|
1530
|
+
// which is why `/* … */` behaved and `// …` did not.
|
|
1531
|
+
if (c === "/" && c2 === "/") { copyCode(i); pending = n; break; }
|
|
1532
|
+
if (lang === "python" && c === "#") { copyCode(i); pending = n; break; }
|
|
1528
1533
|
if (c === "/" && c2 === "*") {
|
|
1529
1534
|
copyCode(i);
|
|
1530
1535
|
state.block = 1;
|
|
@@ -1929,6 +1934,9 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
|
|
|
1929
1934
|
}
|
|
1930
1935
|
|
|
1931
1936
|
const pairs = [];
|
|
1937
|
+
const pairedOld = new Set();
|
|
1938
|
+
const pairedNew = new Set();
|
|
1939
|
+
const cancelled = new Set();
|
|
1932
1940
|
for (const [shape, olds] of oldByShape) {
|
|
1933
1941
|
const news = newByShape.get(shape) || [];
|
|
1934
1942
|
|
|
@@ -1946,20 +1954,69 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
|
|
|
1946
1954
|
const survivingOld = [];
|
|
1947
1955
|
for (const o of olds) {
|
|
1948
1956
|
const twin = survivingNew.findIndex((n) => n.canon === o.canon);
|
|
1949
|
-
if (twin === -1)
|
|
1950
|
-
|
|
1957
|
+
if (twin === -1) {
|
|
1958
|
+
survivingOld.push(o);
|
|
1959
|
+
} else {
|
|
1960
|
+
// Present unchanged on both sides: this assertion did not change,
|
|
1961
|
+
// and the argument-level pass below must not be allowed to pair it
|
|
1962
|
+
// with something else and call that a rewrite.
|
|
1963
|
+
cancelled.add(o);
|
|
1964
|
+
cancelled.add(survivingNew[twin]);
|
|
1965
|
+
survivingNew.splice(twin, 1);
|
|
1966
|
+
}
|
|
1951
1967
|
}
|
|
1952
1968
|
|
|
1953
1969
|
const k = Math.min(survivingOld.length, survivingNew.length);
|
|
1954
1970
|
for (let t = 0; t < k; t++) {
|
|
1955
1971
|
const r = survivingOld[t];
|
|
1956
1972
|
const a = survivingNew[t];
|
|
1973
|
+
// Whatever this pass decides about a pair — reported, or deliberately
|
|
1974
|
+
// let go as a reorder or a reworded message — is the decision. The
|
|
1975
|
+
// argument-level pass below exists only for statements whose shapes
|
|
1976
|
+
// differ so much that they never met in a bucket here; letting it
|
|
1977
|
+
// re-open a case that was already judged turned a comment edit into a
|
|
1978
|
+
// rewritten expectation.
|
|
1979
|
+
pairedOld.add(r);
|
|
1980
|
+
pairedNew.add(a);
|
|
1957
1981
|
if (r.canon === a.canon) continue;
|
|
1958
1982
|
if (differsOnlyInMessage(r.clean, a.clean, lang)) continue;
|
|
1959
1983
|
pairs.push({ r: r.s, a: a.s });
|
|
1960
1984
|
}
|
|
1961
1985
|
}
|
|
1962
1986
|
|
|
1987
|
+
// Same assertion, same subject, different expected value.
|
|
1988
|
+
//
|
|
1989
|
+
// Shape pairing compares the statement with its literals blanked, so it
|
|
1990
|
+
// only ever matched assertions whose structure survived the edit. Shrink
|
|
1991
|
+
// a five-element expected list to one element to match broken output and
|
|
1992
|
+
// the two images land in different shape buckets, never pair, and the
|
|
1993
|
+
// rewrite is not reported at all — measured on a real repository, where
|
|
1994
|
+
// it collected five green phases.
|
|
1995
|
+
//
|
|
1996
|
+
// The arguments are the better witness here: when both sides call the
|
|
1997
|
+
// same assertion with the same number of arguments and the *subject*
|
|
1998
|
+
// argument is untouched, what changed is what the test expects of it.
|
|
1999
|
+
for (const r of oldCands) {
|
|
2000
|
+
if (pairedOld.has(r) || cancelled.has(r)) continue;
|
|
2001
|
+
const ra = splitAssertionArgs(r.clean, lang);
|
|
2002
|
+
if (!ra || ra.length < 2) continue;
|
|
2003
|
+
for (const a of newCands) {
|
|
2004
|
+
if (pairedNew.has(a) || cancelled.has(a)) continue;
|
|
2005
|
+
const aa = splitAssertionArgs(a.clean, lang);
|
|
2006
|
+
if (!aa || aa.length !== ra.length) continue;
|
|
2007
|
+
const same = (i) => ra[i].replace(/\s+/g, "") === aa[i].replace(/\s+/g, "");
|
|
2008
|
+
// The subject has to be the same expression, or these are two
|
|
2009
|
+
// different assertions that merely resemble each other.
|
|
2010
|
+
if (!same(0)) continue;
|
|
2011
|
+
if (ra.every((_, i) => same(i))) continue;
|
|
2012
|
+
if (differsOnlyInMessage(r.clean, a.clean, lang)) continue;
|
|
2013
|
+
pairs.push({ r: r.s, a: a.s });
|
|
2014
|
+
pairedOld.add(r);
|
|
2015
|
+
pairedNew.add(a);
|
|
2016
|
+
break;
|
|
2017
|
+
}
|
|
2018
|
+
}
|
|
2019
|
+
|
|
1963
2020
|
// Zero-context hunk: each image is a single fragment and the assertion
|
|
1964
2021
|
// keyword may sit outside the hunk entirely. The fragment pair is taken
|
|
1965
2022
|
// only when both sides normalize to the same shape *and* that shape
|
|
@@ -2388,21 +2445,57 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2388
2445
|
// understood. `UNREADABLE` is the state that has no business being silent
|
|
2389
2446
|
// — assertion-shaped lines were present and none of them parsed, which
|
|
2390
2447
|
// means this repository speaks a dialect the guard does not.
|
|
2448
|
+
// An assertion is a statement, not a line.
|
|
2449
|
+
//
|
|
2450
|
+
// Counting `+`/`-` lines missed the commonest shape in every language with
|
|
2451
|
+
// multi-line calls: `self.assertEqual(` sits on an unchanged context line
|
|
2452
|
+
// and only its argument lines are edited. Nothing among the changed lines
|
|
2453
|
+
// matched an assertion pattern, nothing looked assertion-shaped either, so
|
|
2454
|
+
// the guard reported `assertionsSeen: 0` and — because no line looked
|
|
2455
|
+
// suspicious — a clean PASS. A five-element expected list rewritten to one
|
|
2456
|
+
// element to match broken output sailed through five green phases.
|
|
2457
|
+
//
|
|
2458
|
+
// The statement machinery that already exists for pairing knows better:
|
|
2459
|
+
// it assembles context lines together with changed ones. Ask it.
|
|
2460
|
+
for (const [file, stats] of fileAssertions.entries()) {
|
|
2461
|
+
const lang = langForTestFile(file);
|
|
2462
|
+
let touched = 0;
|
|
2463
|
+
for (const hunk of stats.hunks) {
|
|
2464
|
+
const oldSlice = [];
|
|
2465
|
+
const newSlice = [];
|
|
2466
|
+
for (const L of hunk.lines) {
|
|
2467
|
+
if (L.kind !== "+") oldSlice.push(L);
|
|
2468
|
+
if (L.kind !== "-") newSlice.push(L);
|
|
2469
|
+
}
|
|
2470
|
+
for (const stmts of [assembleStatements(oldSlice, lang), assembleStatements(newSlice, lang)]) {
|
|
2471
|
+
for (const st of stmts) {
|
|
2472
|
+
const changed = (st.removedLines?.length || 0) + (st.addedLines?.length || 0) > 0;
|
|
2473
|
+
if (changed && ASSERTION_PATTERN.test(stripComments(st.text, lang))) touched++;
|
|
2474
|
+
}
|
|
2475
|
+
}
|
|
2476
|
+
}
|
|
2477
|
+
stats.statementAssertions = touched;
|
|
2478
|
+
}
|
|
2479
|
+
|
|
2391
2480
|
let examined = 0;
|
|
2392
2481
|
let assertionsSeen = 0;
|
|
2393
2482
|
const unreadable = [];
|
|
2394
2483
|
for (const [file, stats] of fileAssertions.entries()) {
|
|
2395
2484
|
examined += stats.examined;
|
|
2396
|
-
assertionsSeen += stats.recognised;
|
|
2485
|
+
assertionsSeen += Math.max(stats.recognised, stats.statementAssertions || 0);
|
|
2397
2486
|
if (stats.unreadable.length > 0) {
|
|
2398
2487
|
unreadable.push({ file, count: stats.unreadable.length, samples: stats.unreadable.slice(0, 3) });
|
|
2399
2488
|
}
|
|
2400
2489
|
}
|
|
2401
2490
|
|
|
2491
|
+
// Changed lines inside a test file, none of them recognisable as part of an
|
|
2492
|
+
// assertion, is not the same as "checked and clean" — it is the state where
|
|
2493
|
+
// this guard has nothing to say. Saying nothing and saying "approved" have
|
|
2494
|
+
// to look different, which is the whole reason `status` exists.
|
|
2402
2495
|
const status =
|
|
2403
2496
|
reported.length > 0
|
|
2404
2497
|
? "FAIL"
|
|
2405
|
-
: assertionsSeen === 0 && unreadable.length > 0
|
|
2498
|
+
: assertionsSeen === 0 && (unreadable.length > 0 || examined > 0)
|
|
2406
2499
|
? "UNREADABLE"
|
|
2407
2500
|
: examined > 0
|
|
2408
2501
|
? "PASS"
|
|
@@ -2480,6 +2573,26 @@ export function scanDiff(diffTextStr = "", options = {}) {
|
|
|
2480
2573
|
}
|
|
2481
2574
|
}
|
|
2482
2575
|
|
|
2576
|
+
// The gate calls `scanDiff`, not `assertTestIntegrity` — so wiring the
|
|
2577
|
+
// dialect warning into the latter meant it reached nobody. The guard
|
|
2578
|
+
// computed `UNREADABLE`, and the operator was shown an unblemished pass.
|
|
2579
|
+
// A boundary that is not reported is not a boundary.
|
|
2580
|
+
if (tamperingRes.status === "UNREADABLE") {
|
|
2581
|
+
const where = (tamperingRes.unreadable || []).map((u) => u.file);
|
|
2582
|
+
const sample = tamperingRes.unreadable?.[0]?.samples?.[0];
|
|
2583
|
+
findings.push({
|
|
2584
|
+
severity: "MEDIUM",
|
|
2585
|
+
type: "TEST_DIALECT_UNREADABLE",
|
|
2586
|
+
file: where[0] ?? null,
|
|
2587
|
+
line: null,
|
|
2588
|
+
description:
|
|
2589
|
+
`Test Tamper Guard: changed ${tamperingRes.inputsSeen} line(s) in ${tamperingRes.filesSeen} test file(s) ` +
|
|
2590
|
+
`and recognised no assertion among them${sample ? ` (e.g. ${JSON.stringify(sample)})` : ""}. ` +
|
|
2591
|
+
`This change was NOT checked for tampering. That is not a failure — an unlisted assertion library is ` +
|
|
2592
|
+
`normal — but it is not an approval either, so it is reported rather than passed silently.`,
|
|
2593
|
+
});
|
|
2594
|
+
}
|
|
2595
|
+
|
|
2483
2596
|
return {
|
|
2484
2597
|
ok: secretsOk && edgeRes.ok && crossPkgRes.ok && tamperingRes.ok,
|
|
2485
2598
|
findings,
|
package/src/stack-detector.mjs
CHANGED
|
@@ -188,6 +188,56 @@ export function detectEdgeRuntime(projectRoot = process.cwd()) {
|
|
|
188
188
|
/**
|
|
189
189
|
* Detects 24+ polyglot stacks and container environments.
|
|
190
190
|
*/
|
|
191
|
+
/**
|
|
192
|
+
* Test commands worth trying, best first, when the detected one does not run.
|
|
193
|
+
*
|
|
194
|
+
* `init` probes the command it picked. On a repository whose Makefile
|
|
195
|
+
* declares a `test` target that needs a build environment the machine does
|
|
196
|
+
* not have, that probe failed, printed `Oracle verification probe failed`,
|
|
197
|
+
* and the wizard wrote the broken command into the config anyway — in a
|
|
198
|
+
* repository where `pytest` was on PATH and all 360 tests passed in 1.3s.
|
|
199
|
+
* Measuring something and then ignoring the measurement is worse than not
|
|
200
|
+
* measuring: it produces a hard red on day one, which is how a user learns
|
|
201
|
+
* the gate is broken and turns it off.
|
|
202
|
+
*
|
|
203
|
+
* Kept deliberately generic — a per-ecosystem convention, never a per-project
|
|
204
|
+
* or per-provider guess.
|
|
205
|
+
*
|
|
206
|
+
* @param {string} root
|
|
207
|
+
* @param {string} [detected] - the command detection chose; always first.
|
|
208
|
+
* @returns {string[]} ordered, de-duplicated candidates
|
|
209
|
+
*/
|
|
210
|
+
export function oracleCandidates(root = process.cwd(), detected = "") {
|
|
211
|
+
const out = [];
|
|
212
|
+
const push = (c) => {
|
|
213
|
+
const v = (c || "").trim();
|
|
214
|
+
if (v && !out.includes(v) && !isPlaceholderTestScript(v)) out.push(v);
|
|
215
|
+
};
|
|
216
|
+
const has = (f) => existsSync(join(root, f));
|
|
217
|
+
|
|
218
|
+
push(detected);
|
|
219
|
+
|
|
220
|
+
if (has("package.json")) {
|
|
221
|
+
try {
|
|
222
|
+
const pkg = JSON.parse(readFileSync(join(root, "package.json"), "utf-8"));
|
|
223
|
+
if (pkg.scripts?.test && !isPlaceholderTestScript(pkg.scripts.test)) push("npm test");
|
|
224
|
+
} catch (_) {}
|
|
225
|
+
}
|
|
226
|
+
if (has("pytest.ini") || has("pyproject.toml") || has("setup.py") || has("tox.ini") || has("setup.cfg")) {
|
|
227
|
+
push(pytestCmd());
|
|
228
|
+
}
|
|
229
|
+
if (has("Cargo.toml")) push("cargo test");
|
|
230
|
+
if (has("go.mod")) push("go test ./...");
|
|
231
|
+
if (has("Gemfile")) push("bundle exec rspec");
|
|
232
|
+
if (has("composer.json")) push("./vendor/bin/phpunit");
|
|
233
|
+
if (has("pom.xml")) push("mvn -q test");
|
|
234
|
+
if (has("build.gradle") || has("build.gradle.kts")) push("./gradlew test");
|
|
235
|
+
if (has("pubspec.yaml")) push("dart test");
|
|
236
|
+
if (has("Package.swift")) push("swift test");
|
|
237
|
+
|
|
238
|
+
return out;
|
|
239
|
+
}
|
|
240
|
+
|
|
191
241
|
export function detectPolyglotStack(projectRoot = process.cwd()) {
|
|
192
242
|
const edgeInfo = detectEdgeRuntime(projectRoot);
|
|
193
243
|
const isDevcontainer = existsSync(join(projectRoot, ".devcontainer", "devcontainer.json"));
|
package/src/wizard-init.mjs
CHANGED
|
@@ -3,7 +3,7 @@ import { join } from "node:path";
|
|
|
3
3
|
import { parseYaml, TIER_PRESETS, VENDOR_TIERS, FALLBACK_TIER } from "./config.mjs";
|
|
4
4
|
import { suggestProvider, detectAvailableProviders } from "./provider-readiness.mjs";
|
|
5
5
|
import { detectDefaultBranch } from "./git.mjs";
|
|
6
|
-
import { resolveWorkspaceBoundary } from "./stack-detector.mjs";
|
|
6
|
+
import { resolveWorkspaceBoundary, oracleCandidates } from "./stack-detector.mjs";
|
|
7
7
|
import { PROFILE_NAMES, PROFILE_DESCRIPTIONS } from "./profiles.mjs";
|
|
8
8
|
import { detectStackOracles, runVerificationProbe } from "./wizard-oracle.mjs";
|
|
9
9
|
import { select, multiSelect, input, confirm, spinner, isTTY } from "./tui.mjs";
|
|
@@ -302,6 +302,51 @@ export function loadPresets(root = process.cwd()) {
|
|
|
302
302
|
* @param {object} [options]
|
|
303
303
|
* @returns {Promise<{ ok: boolean, configPath: string, plan: object }>}
|
|
304
304
|
*/
|
|
305
|
+
/**
|
|
306
|
+
* Probe the chosen test command, and take detection's next choice if it fails.
|
|
307
|
+
*
|
|
308
|
+
* Runs on the non-interactive path too. `--yes` means "do not ask me", not
|
|
309
|
+
* "do not check" — and the user who is not watching is exactly the one who
|
|
310
|
+
* cannot notice that the command written into their config does not run.
|
|
311
|
+
* Before this, the probe lived inside the interactive branch, so
|
|
312
|
+
* `agentctl init --yes` wrote `make test` into a repository where `make test`
|
|
313
|
+
* exits 2 and `npm test` passes, and the first gate run was a hard red.
|
|
314
|
+
*
|
|
315
|
+
* @returns {Promise<string>} the command to save
|
|
316
|
+
*/
|
|
317
|
+
async function resolveRunnableOracle(root, testCmd, options = {}) {
|
|
318
|
+
if (!testCmd) return testCmd;
|
|
319
|
+
const probeSp = spinner(`Probing oracle: ${testCmd}`, options);
|
|
320
|
+
const probeRes = await runVerificationProbe(testCmd, root);
|
|
321
|
+
if (probeRes.ok) {
|
|
322
|
+
probeSp.stop(`Oracle verified successfully (${probeRes.durationMs}ms)`);
|
|
323
|
+
return testCmd;
|
|
324
|
+
}
|
|
325
|
+
probeSp.fail(`Oracle verification probe failed (Exit ${probeRes.code})`);
|
|
326
|
+
|
|
327
|
+
const alternates = oracleCandidates(root, testCmd).filter((c) => c !== testCmd).slice(0, 3);
|
|
328
|
+
for (const cand of alternates) {
|
|
329
|
+
const altSp = spinner(`Trying ${cand}`, options);
|
|
330
|
+
const altRes = await runVerificationProbe(cand, root);
|
|
331
|
+
if (altRes.ok) {
|
|
332
|
+
altSp.stop(`${cand} runs here (${altRes.durationMs}ms) — using it instead`);
|
|
333
|
+
return cand;
|
|
334
|
+
}
|
|
335
|
+
altSp.fail(`${cand} also failed (Exit ${altRes.code})`);
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
// Nothing runs. Say so in terms the user can act on, rather than leaving a
|
|
339
|
+
// failed spinner to scroll past and a broken command in the config.
|
|
340
|
+
const out = options.stdout || process.stdout;
|
|
341
|
+
out.write("\n");
|
|
342
|
+
out.write(" \u26a0\ufe0f No test command could be run in this environment.\n");
|
|
343
|
+
out.write(` Keeping "${testCmd}" \u2014 the gate will fail until it runs here.\n`);
|
|
344
|
+
out.write(" Point verify.test in .agent/config.yml at a command that works,\n");
|
|
345
|
+
out.write(" or, if this repository genuinely has no suite, set\n");
|
|
346
|
+
out.write(" verify.required: false deliberately rather than by accident.\n\n");
|
|
347
|
+
return testCmd;
|
|
348
|
+
}
|
|
349
|
+
|
|
305
350
|
export async function runInitWizard(root = process.cwd(), options = {}) {
|
|
306
351
|
const interactive = options.interactive !== false && isTTY(options.stdin || process.stdin);
|
|
307
352
|
|
|
@@ -322,6 +367,7 @@ export async function runInitWizard(root = process.cwd(), options = {}) {
|
|
|
322
367
|
let selectedProvider = options.provider || existingConfig.provider;
|
|
323
368
|
let selectedProfile = options.profile || existingConfig.verify?.profile;
|
|
324
369
|
let testCmd = options.testCmd;
|
|
370
|
+
let probeInteractive = null;
|
|
325
371
|
let buildCmd = options.buildCmd;
|
|
326
372
|
let selectedPresets = options.presets;
|
|
327
373
|
|
|
@@ -402,16 +448,20 @@ export async function runInitWizard(root = process.cwd(), options = {}) {
|
|
|
402
448
|
|
|
403
449
|
selectedPresets = await multiSelect(presetOptions, "Select Autonomous Workflows to Enable", options);
|
|
404
450
|
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
451
|
+
probeInteractive = await confirm("Run verification probe on test command before saving?", true, options);
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
// The probe runs whether or not anyone was asked: interactive users can
|
|
455
|
+
// decline it, but silence from `--yes` is not a decline.
|
|
456
|
+
if (probeInteractive !== false && options.probe !== false) {
|
|
457
|
+
// Resolve the command the way planInit will, or there is nothing to
|
|
458
|
+
// probe: on the headless path `testCmd` stays undefined until planInit
|
|
459
|
+
// fills it in from detection, so the probe silently examined nothing —
|
|
460
|
+
// the exact fail-open shape this project keeps finding in itself.
|
|
461
|
+
const effective =
|
|
462
|
+
testCmd || existingConfig.verify?.test || detectStackOracles(root)?.candidates?.testCmd || "";
|
|
463
|
+
const adopted = await resolveRunnableOracle(root, effective, options);
|
|
464
|
+
if (adopted) testCmd = adopted;
|
|
415
465
|
}
|
|
416
466
|
|
|
417
467
|
// `...options` first, for the same reason as in wizard-task.mjs: spreading it
|