jules-orchestrator-kit 0.56.0 → 0.58.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -3
- package/bin/agentctl.mjs +125 -17
- package/package.json +1 -1
- package/src/assertions.mjs +5 -0
- package/src/config.mjs +73 -1
- package/src/dag-engine.mjs +1 -1
- package/src/engine.mjs +1 -1
- package/src/evidence.mjs +27 -4
- package/src/git.mjs +66 -7
- package/src/mutation.mjs +11 -1
- package/src/ops/command-registry.mjs +12 -8
- package/src/ops/tdd-generator.mjs +93 -24
- package/src/security.mjs +120 -12
- package/src/stack-detector.mjs +40 -3
- package/src/state.mjs +72 -16
package/README.md
CHANGED
|
@@ -204,7 +204,7 @@ To maximize PR merge rates, dispatch tasks according to deterministic boundaries
|
|
|
204
204
|
* **Fail-Closed Security & Secret Redaction:** Evaluates explicit Deny rules before Allow rules against canonicalized, case-folded paths. Redacts high-entropy keys and base64-encoded credentials (such as Kubernetes `Secret` manifests).
|
|
205
205
|
* **Complexity & Cost Router:** Zero-dependency heuristic classifier (`src/router.mjs`) routing mechanical tasks to lightweight models while reserving primary models for complex refactors, with a `node --check` syntax-verification gate that transparently escalates a FAST-tier result to the primary provider if it left broken JS on disk.
|
|
206
206
|
* **Terminal UI & Diagnostic Matrix (`agentctl doctor`):** Interactive terminal dashboard, task sidecar manager, and automated transactional self-repair.
|
|
207
|
-
* **Verified Test Suite:** Tested with **
|
|
207
|
+
* **Verified Test Suite:** Tested with **895 unit tests across 128 suites passing in < 15.0s**.
|
|
208
208
|
|
|
209
209
|
<br/>
|
|
210
210
|
|
|
@@ -237,7 +237,7 @@ To maximize PR merge rates, dispatch tasks according to deterministic boundaries
|
|
|
237
237
|
| `doctor` | `agentctl doctor [--probe] [--json]` | Diagnostic check runner. `--probe` additionally starts the configured provider's CLI to confirm it answers, rather than only finding it on `PATH`. | `0` (Healthy), `1` (Failures) |
|
|
238
238
|
| `queue` | `agentctl queue [--dag] [--concurrency <n>] [--dry-run] [--json]` | Consumes and executes task envelopes in `.agent/jules-queue/` with Kahn's DAG dependency resolution. Non-task files (manifests, `README.md`) are skipped, and `--dry-run` previews without moving anything. | `0` (Complete) |
|
|
239
239
|
| `swarm` | `agentctl swarm [--json]` | Runs parallel multi-agent swarm across worker slots with PID liveness detection. | `0` (Complete) |
|
|
240
|
-
| `check` / `gate` / `audit`| `agentctl check [--mode working-tree] [--fix] [--json] [--json-report <path>]` | Runs security, secret scanning, rules budget audit, and tiered verification gates (with declarative assertion support) against working tree or branch. | `0` (Approved), `1` (Budget/Arg), `3` (Scope), `4` (Verify), `5` (Diff >75K), `6` (Secret), `8` (Flaky) |
|
|
240
|
+
| `check` / `gate` / `audit`| `agentctl check [--mode working-tree] [--fix] [--allow-protected] [--allow-test-modifications] [--json] [--json-report <path>]` | Runs security, secret scanning, rules budget audit, and tiered verification gates (with declarative assertion support) against working tree or branch. | `0` (Approved), `1` (Budget/Arg), `3` (Scope), `4` (Verify), `5` (Diff >75K), `6` (Secret), `8` (Flaky) |
|
|
241
241
|
| `mutate` / `mutation` | `agentctl mutate [--min-score <n>] [--max-mutants <n>] [--cmd <testCmd>] [--json]` | Runs zero-dependency diff mutation testing harness on changed hunks with operator inversion and safety rollback. | `0` (Passed), `1` (Score Low) |
|
|
242
242
|
| `coverage` | `agentctl coverage [--min <pct>] [--cmd <testCmd>] [--base <ref>] [--json]` | Runs native zero-dependency V8 diff coverage check against added diff lines. | `0` (Passed), `1` (Low Coverage) |
|
|
243
243
|
| `probe` / `stability` | `agentctl probe [--repeat <n>] [--min <passRate>] [--cmd <testCmd>] [--json]` | Probes test suite flakiness across N consecutive iterations with oscillation detection. | `0` (Passed), `1` (Flaky) |
|
|
@@ -249,7 +249,7 @@ To maximize PR merge rates, dispatch tasks according to deterministic boundaries
|
|
|
249
249
|
| `resume` | `agentctl resume <sessionId> --response "<reply>"` | Streams engineer response back into active Google Jules warm session context window. | `0` (Resumed), `1` (Error) |
|
|
250
250
|
| `test-gen` | `agentctl test-gen --title <t> --spec <s> [--run]` | Scaffolds falsifiable unit tests, verifies RED failure state, and locks test in `scope.deny`. | `0` (Scaffolded/Red) |
|
|
251
251
|
| `dashboard` | `agentctl dashboard [port]` | Starts zero-dependency local HTTP telemetry and audit visualizer dashboard. | `0` (Running) |
|
|
252
|
-
| `evidence` | `agentctl evidence <generate\|verify\|show>` | Generates, verifies, or prints SHA-256
|
|
252
|
+
| `evidence` | `agentctl evidence <generate\|verify\|show>` | Generates, verifies, or prints SHA-256 evidence manifests (unkeyed digests: tamper-evident, not signed) with test-tamper locking. | `0` (Verified), `1` (Tamper) |
|
|
253
253
|
| `flaky` | `agentctl flaky <status\|heal\|reset>` | Manages Wilson-quarantined tests (Exit Code 8) and dispatches automated anti-flakiness healing swarms. | `0` (Healed/Listed) |
|
|
254
254
|
| `mcp` | `agentctl mcp` | Starts stdio Model Context Protocol (MCP) server for Claude, Cursor, and Antigravity. | `0` / Stdio stream |
|
|
255
255
|
| `mcp init` | `agentctl mcp init [--target cursor\|vscode\|claude\|all]` | 1-click config scaffolding for Cursor (`.cursor/mcp.json`), VS Code tasks (`tasks.json`), and Claude Desktop. | `0` (Scaffolded) |
|
package/bin/agentctl.mjs
CHANGED
|
@@ -61,7 +61,7 @@ Commands:
|
|
|
61
61
|
swarm Run parallel task swarm
|
|
62
62
|
mcp Start stdio Model Context Protocol (MCP) server
|
|
63
63
|
clean Clean stale branches, worktrees, locks, and ledgers
|
|
64
|
-
lock <action> Manage mutex locks (acquire | release | status)
|
|
64
|
+
lock <action> Manage mutex locks (acquire [--ttl <min>] [--pid <n>] | release | status)
|
|
65
65
|
doctor Run system diagnostics and stack resolution checks
|
|
66
66
|
providers List agent providers and whether this machine can reach them (--json)
|
|
67
67
|
provider set <name> Switch the active provider in .agent/config.yml
|
|
@@ -383,6 +383,12 @@ async function main() {
|
|
|
383
383
|
committed: { type: "boolean" },
|
|
384
384
|
fix: { type: "boolean" },
|
|
385
385
|
"allow-protected": { type: "boolean" },
|
|
386
|
+
// The tamper guard has always had an override — `allowTestModifications`
|
|
387
|
+
// — and it was reachable only from JavaScript. So a legitimate change
|
|
388
|
+
// of spec, which necessarily rewrites what a test expects, hit a
|
|
389
|
+
// CRITICAL finding at exit 6 with no documented way past it. A guard
|
|
390
|
+
// with no override is not a guard, it is an outage.
|
|
391
|
+
"allow-test-modifications": { type: "boolean" },
|
|
386
392
|
json: { type: "boolean", short: "j" },
|
|
387
393
|
"json-report": { type: "string" },
|
|
388
394
|
"dry-run": { type: "boolean", short: "d" },
|
|
@@ -402,6 +408,7 @@ async function main() {
|
|
|
402
408
|
mode: selectedMode,
|
|
403
409
|
fix: values.fix,
|
|
404
410
|
allowProtected: values["allow-protected"],
|
|
411
|
+
allowTestModifications: values["allow-test-modifications"],
|
|
405
412
|
jsonReport: values["json-report"],
|
|
406
413
|
});
|
|
407
414
|
|
|
@@ -570,8 +577,16 @@ async function main() {
|
|
|
570
577
|
if (values.committed) selectedMode = "committed";
|
|
571
578
|
if (values["working-tree"]) selectedMode = "working-tree";
|
|
572
579
|
|
|
573
|
-
|
|
574
|
-
|
|
580
|
+
// `Number(x) || default` swallows a legitimate zero: `--min-score 0`,
|
|
581
|
+
// the way to run the harness for its report without a threshold, silently
|
|
582
|
+
// enforced 80. Both flags already carry a parseArgs default, so the only
|
|
583
|
+
// thing left to guard against is a value that is not a number at all.
|
|
584
|
+
const parseNumericFlag = (raw, fallback) => {
|
|
585
|
+
const n = Number(raw);
|
|
586
|
+
return Number.isFinite(n) ? n : fallback;
|
|
587
|
+
};
|
|
588
|
+
const minScore = parseNumericFlag(values["min-score"], 80);
|
|
589
|
+
const maxMutants = parseNumericFlag(values["max-mutants"], 20);
|
|
575
590
|
const testCmd = values.cmd || config.verify?.test || "npm test";
|
|
576
591
|
|
|
577
592
|
if (!values.json) {
|
|
@@ -597,7 +612,11 @@ async function main() {
|
|
|
597
612
|
console.log(` • Mutants Killed (Fail) : ${report.killedMutants} ✅`);
|
|
598
613
|
console.log(` • Mutants Survived (Pass) : ${report.survivedMutants} ${report.survivedMutants > 0 ? "⚠️" : ""}`);
|
|
599
614
|
console.log(` • Errors / Timeouts : ${report.errorMutants}`);
|
|
600
|
-
console.log(
|
|
615
|
+
console.log(
|
|
616
|
+
report.scored === false
|
|
617
|
+
? ` • Mutation Score : n/a — ${report.reason}`
|
|
618
|
+
: ` • Mutation Score : ${report.mutationScore}% (Required: ${report.minScore}%)`
|
|
619
|
+
);
|
|
601
620
|
console.log(` • Duration : ${report.durationMs}ms\n`);
|
|
602
621
|
|
|
603
622
|
if (report.survivors.length > 0) {
|
|
@@ -1069,7 +1088,29 @@ async function main() {
|
|
|
1069
1088
|
});
|
|
1070
1089
|
|
|
1071
1090
|
const queueDir = getQueueDir(root);
|
|
1072
|
-
|
|
1091
|
+
// The DAG runner accepts `.json` and `.task` envelopes as well as
|
|
1092
|
+
// Markdown, and does its own discovery — but this count gated whether it
|
|
1093
|
+
// was ever called, using the Markdown-only filter. A queue holding only
|
|
1094
|
+
// JSON envelopes reported "0 queued task(s)" and did nothing, with no
|
|
1095
|
+
// indication that the files were there and understood.
|
|
1096
|
+
const queueEntries = readdirSync(queueDir);
|
|
1097
|
+
let files;
|
|
1098
|
+
if (values.dag) {
|
|
1099
|
+
const { isDagTaskFile } = await import("../src/dag-engine.mjs");
|
|
1100
|
+
files = queueEntries.filter((f) => {
|
|
1101
|
+
if (f === "completed" || f.startsWith(".")) return false;
|
|
1102
|
+
if (!/\.(md|json|task)$/.test(f)) return false;
|
|
1103
|
+
let content = "";
|
|
1104
|
+
try {
|
|
1105
|
+
content = readFileSync(join(queueDir, f), "utf-8");
|
|
1106
|
+
} catch (_) {
|
|
1107
|
+
return false;
|
|
1108
|
+
}
|
|
1109
|
+
return isDagTaskFile(f, content, isTaskFile);
|
|
1110
|
+
});
|
|
1111
|
+
} else {
|
|
1112
|
+
files = queueEntries.filter((f) => isTaskFile(f, queueDir));
|
|
1113
|
+
}
|
|
1073
1114
|
console.log(`Found ${files.length} queued task(s) in .agent/jules-queue/`);
|
|
1074
1115
|
if (files.length > 0) {
|
|
1075
1116
|
const concurrency = values.concurrency ? Number(values.concurrency) : undefined;
|
|
@@ -1092,11 +1133,30 @@ async function main() {
|
|
|
1092
1133
|
}
|
|
1093
1134
|
|
|
1094
1135
|
case "swarm": {
|
|
1095
|
-
|
|
1136
|
+
// This case had no parseArgs at all: `--json` and `--dry-run` were
|
|
1137
|
+
// accepted by the shell, documented in the registry, and silently
|
|
1138
|
+
// discarded — so a rehearsal dispatched for real and a script asking for
|
|
1139
|
+
// JSON got decorated prose.
|
|
1140
|
+
const { values } = parseArgs({
|
|
1141
|
+
args: args.slice(1),
|
|
1142
|
+
options: {
|
|
1143
|
+
concurrency: { type: "string", short: "c" },
|
|
1144
|
+
"dry-run": { type: "boolean", short: "d" },
|
|
1145
|
+
json: { type: "boolean", short: "j" },
|
|
1146
|
+
},
|
|
1147
|
+
allowPositionals: true,
|
|
1148
|
+
strict: false,
|
|
1149
|
+
});
|
|
1150
|
+
|
|
1096
1151
|
const queueDir = getQueueDir(root);
|
|
1097
1152
|
const files = readdirSync(queueDir).filter((f) => isTaskFile(f, queueDir));
|
|
1153
|
+
if (!values.json) console.log("🚀 Running Swarm Orchestrator...");
|
|
1098
1154
|
if (files.length === 0) {
|
|
1099
|
-
|
|
1155
|
+
if (values.json) {
|
|
1156
|
+
console.log(JSON.stringify({ ok: true, processed: 0, results: [], dryRun: Boolean(values["dry-run"]) }, null, 2));
|
|
1157
|
+
} else {
|
|
1158
|
+
console.log("No pending tasks found for swarm.");
|
|
1159
|
+
}
|
|
1100
1160
|
process.exit(0);
|
|
1101
1161
|
}
|
|
1102
1162
|
const tasks = files.map((f) => ({
|
|
@@ -1104,7 +1164,16 @@ async function main() {
|
|
|
1104
1164
|
title: f.replace(/\.md$/, ""),
|
|
1105
1165
|
prompt: readFileSync(join(queueDir, f), "utf-8"),
|
|
1106
1166
|
}));
|
|
1107
|
-
const results = await run(tasks, {
|
|
1167
|
+
const results = await run(tasks, {
|
|
1168
|
+
root,
|
|
1169
|
+
config,
|
|
1170
|
+
concurrency: values.concurrency ? Number(values.concurrency) : config.limits.concurrency || 3,
|
|
1171
|
+
dryRun: values["dry-run"],
|
|
1172
|
+
});
|
|
1173
|
+
if (values.json) {
|
|
1174
|
+
console.log(JSON.stringify({ ok: true, dryRun: Boolean(values["dry-run"]), ...results }, null, 2));
|
|
1175
|
+
process.exit((results.results || []).some((r) => r && r.ok === false) ? 1 : 0);
|
|
1176
|
+
}
|
|
1108
1177
|
process.exit(reportRunOutcome(results));
|
|
1109
1178
|
break;
|
|
1110
1179
|
}
|
|
@@ -1216,14 +1285,46 @@ async function main() {
|
|
|
1216
1285
|
case "lock": {
|
|
1217
1286
|
const action = args[1];
|
|
1218
1287
|
if (action === "acquire") {
|
|
1219
|
-
|
|
1220
|
-
|
|
1221
|
-
|
|
1222
|
-
|
|
1288
|
+
// A lock taken from the command line outlives the command. The process
|
|
1289
|
+
// that runs `lock acquire` writes the record and exits; the agent it
|
|
1290
|
+
// speaks for is somewhere else entirely. So this is a time-bounded
|
|
1291
|
+
// lease, not a pid the next acquire can test for liveness — that test
|
|
1292
|
+
// always said "dead" and handed the same files to the next caller.
|
|
1293
|
+
//
|
|
1294
|
+
// `--pid` is the escape hatch for a caller that *does* have a durable
|
|
1295
|
+
// process to point at: the lock then releases itself when that process
|
|
1296
|
+
// dies, exactly as an in-process acquire does.
|
|
1297
|
+
const flagIdx = args.findIndex((a, i) => i >= 2 && a.startsWith("--"));
|
|
1298
|
+
const positional = flagIdx === -1 ? args.slice(2) : args.slice(2, flagIdx);
|
|
1299
|
+
const { values } = parseArgs({
|
|
1300
|
+
args: flagIdx === -1 ? [] : args.slice(flagIdx),
|
|
1301
|
+
options: {
|
|
1302
|
+
ttl: { type: "string" },
|
|
1303
|
+
pid: { type: "string" },
|
|
1304
|
+
},
|
|
1305
|
+
allowPositionals: false,
|
|
1306
|
+
});
|
|
1307
|
+
const agent = positional[0] || "agent";
|
|
1308
|
+
const taskId = positional[1] || "task-1";
|
|
1309
|
+
const filePaths = positional.slice(2);
|
|
1310
|
+
const ttlMinutes = Number(values.ttl);
|
|
1311
|
+
const ownerPid = Number(values.pid);
|
|
1312
|
+
const bindsToProcess = Number.isInteger(ownerPid) && ownerPid > 0;
|
|
1313
|
+
const res = acquireLock(agent, taskId, filePaths, root, {
|
|
1314
|
+
lease: !bindsToProcess,
|
|
1315
|
+
ownerPid: bindsToProcess ? ownerPid : undefined,
|
|
1316
|
+
ttlMs: Number.isFinite(ttlMinutes) && ttlMinutes > 0 ? ttlMinutes * 60_000 : undefined,
|
|
1317
|
+
});
|
|
1223
1318
|
if (res.ok) {
|
|
1224
|
-
|
|
1319
|
+
const held = filePaths.length ? ` (${filePaths.length} path${filePaths.length === 1 ? "" : "s"})` : "";
|
|
1320
|
+
const until = bindsToProcess ? `bound to pid ${ownerPid}` : `expires ${new Date(Date.now() + (Number.isFinite(ttlMinutes) && ttlMinutes > 0 ? ttlMinutes : 120) * 60_000).toISOString()}`;
|
|
1321
|
+
console.log(`✅ Acquired lock for ${taskId}${held} — ${until}`);
|
|
1225
1322
|
} else {
|
|
1226
|
-
|
|
1323
|
+
const overlap = Array.isArray(res.conflictingFiles) && res.conflictingFiles.length
|
|
1324
|
+
? `\n Contested paths: ${res.conflictingFiles.join(", ")}`
|
|
1325
|
+
: "";
|
|
1326
|
+
const until = res.expiresAt ? `\n Held until ${res.expiresAt} (or until \`agentctl lock release ${res.taskId}\`).` : "";
|
|
1327
|
+
console.log(`❌ Lock conflict detected: held by ${res.holder} (task ${res.taskId})${overlap}${until}`);
|
|
1227
1328
|
process.exit(1);
|
|
1228
1329
|
}
|
|
1229
1330
|
} else if (action === "release") {
|
|
@@ -2610,9 +2711,16 @@ async function main() {
|
|
|
2610
2711
|
if (values.json) {
|
|
2611
2712
|
console.log(JSON.stringify(res, null, 2));
|
|
2612
2713
|
} else {
|
|
2613
|
-
|
|
2714
|
+
// "Signature" was the wrong word and the wrong promise. There is no
|
|
2715
|
+
// key anywhere in this system: the value is a SHA-256 digest of the
|
|
2716
|
+
// manifest, so anyone who can write the file can also recompute it
|
|
2717
|
+
// and have `evidence verify` agree. What it proves is that the
|
|
2718
|
+
// manifest has not been edited *since* it was written — tamper
|
|
2719
|
+
// evidence, not authorship. Calling that a signature invites an
|
|
2720
|
+
// operator to trust it for something it cannot do.
|
|
2721
|
+
console.log(`\n🛡️ Evidence Manifest Generated!`);
|
|
2614
2722
|
console.log(` Manifest ID : ${res.manifest.manifestId}`);
|
|
2615
|
-
console.log(`
|
|
2723
|
+
console.log(` Digest : ${res.manifest.evidenceHash}`);
|
|
2616
2724
|
console.log(` Location : ${res.manifestPath}`);
|
|
2617
2725
|
console.log(` Test Files : ${res.manifest.testIntegrity.testFileCount}`);
|
|
2618
2726
|
console.log(` Tampered : ${res.manifest.testIntegrity.tamperDetected ? "YES (FAILED)" : "NO (VERIFIED)"}\n`);
|
|
@@ -2628,7 +2736,7 @@ async function main() {
|
|
|
2628
2736
|
if (res.ok) {
|
|
2629
2737
|
console.log(`\n✅ Evidence Verification PASSED`);
|
|
2630
2738
|
console.log(` Manifest ID : ${res.manifestId}`);
|
|
2631
|
-
console.log(`
|
|
2739
|
+
console.log(` Digest : ${res.evidenceHash}\n`);
|
|
2632
2740
|
} else {
|
|
2633
2741
|
console.error(`\n❌ Evidence Verification FAILED`);
|
|
2634
2742
|
console.error(` Reason : ${res.reason}`);
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "jules-orchestrator-kit",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.58.0",
|
|
4
4
|
"description": "Zero-dependency safety gatekeeper, test oracle generator, and multi-agent coordination protocol for autonomous coding agents — Google Jules, Claude Code, Codex and Gemini CLI.",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
package/src/assertions.mjs
CHANGED
|
@@ -513,6 +513,11 @@ export function assertMutation(config = {}, root = process.cwd()) {
|
|
|
513
513
|
});
|
|
514
514
|
|
|
515
515
|
const diagnostics = [];
|
|
516
|
+
if (report.scored === false) {
|
|
517
|
+
// Not a failure, but not a pass worth trusting either: say so, so a green
|
|
518
|
+
// stage cannot be read as "the diff survived mutation testing".
|
|
519
|
+
diagnostics.push(report.reason || "No mutants could be generated from the added lines.");
|
|
520
|
+
}
|
|
516
521
|
if (!report.ok) {
|
|
517
522
|
diagnostics.push(
|
|
518
523
|
`Mutation score ${report.mutationScore}% below required threshold of ${minScore}% (${report.killedMutants}/${report.totalMutants} killed, ${report.survivedMutants} survived).`
|
package/src/config.mjs
CHANGED
|
@@ -28,6 +28,34 @@ const DEFAULTS = {
|
|
|
28
28
|
baseBranch: "main",
|
|
29
29
|
};
|
|
30
30
|
|
|
31
|
+
// A CI definition is code the forge runs with the repository's own
|
|
32
|
+
// credentials, on the next push, without review. `.github/**` was the only one
|
|
33
|
+
// listed, so the gate's answer depended on which forge the project happened to
|
|
34
|
+
// use: an identical exfiltration job placed in `.gitlab-ci.yml` was approved
|
|
35
|
+
// where `.github/workflows/x.yml` was rejected. A safety gate that is only
|
|
36
|
+
// safe on GitHub is not a safety gate.
|
|
37
|
+
const CI_DEFINITIONS = [
|
|
38
|
+
".github/**",
|
|
39
|
+
".gitlab-ci.yml",
|
|
40
|
+
"**/.gitlab-ci.yml",
|
|
41
|
+
".gitlab/ci/**",
|
|
42
|
+
".circleci/**",
|
|
43
|
+
"azure-pipelines.yml",
|
|
44
|
+
"azure-pipelines.yaml",
|
|
45
|
+
"**/azure-pipelines.yml",
|
|
46
|
+
"Jenkinsfile",
|
|
47
|
+
"**/Jenkinsfile",
|
|
48
|
+
".travis.yml",
|
|
49
|
+
".drone.yml",
|
|
50
|
+
"bitbucket-pipelines.yml",
|
|
51
|
+
".buildkite/**",
|
|
52
|
+
".woodpecker.yml",
|
|
53
|
+
".woodpecker/**",
|
|
54
|
+
"appveyor.yml",
|
|
55
|
+
".teamcity/**",
|
|
56
|
+
".githooks/**",
|
|
57
|
+
];
|
|
58
|
+
|
|
31
59
|
export const BUILTIN_DENY = [
|
|
32
60
|
".git/**",
|
|
33
61
|
"**/.env",
|
|
@@ -36,7 +64,7 @@ export const BUILTIN_DENY = [
|
|
|
36
64
|
"**/*.key",
|
|
37
65
|
"**/id_rsa*",
|
|
38
66
|
".agent/jules-queue/**",
|
|
39
|
-
|
|
67
|
+
...CI_DEFINITIONS,
|
|
40
68
|
];
|
|
41
69
|
|
|
42
70
|
export const BUILTIN_PROTECT = [
|
|
@@ -59,6 +87,50 @@ export const BUILTIN_PROTECT = [
|
|
|
59
87
|
"**/Makefile",
|
|
60
88
|
"**/.npmrc",
|
|
61
89
|
"**/.netrc",
|
|
90
|
+
|
|
91
|
+
// Build definitions that execute arbitrary code at install or build time.
|
|
92
|
+
// The same tier as package.json: legitimately edited, so not denied, but not
|
|
93
|
+
// edited quietly either.
|
|
94
|
+
"setup.py",
|
|
95
|
+
"**/setup.py",
|
|
96
|
+
"setup.cfg",
|
|
97
|
+
"**/setup.cfg",
|
|
98
|
+
"build.rs",
|
|
99
|
+
"**/build.rs",
|
|
100
|
+
"pom.xml",
|
|
101
|
+
"**/pom.xml",
|
|
102
|
+
"build.gradle",
|
|
103
|
+
"**/build.gradle",
|
|
104
|
+
"build.gradle.kts",
|
|
105
|
+
"**/build.gradle.kts",
|
|
106
|
+
"CMakeLists.txt",
|
|
107
|
+
"**/CMakeLists.txt",
|
|
108
|
+
"Gemfile",
|
|
109
|
+
"**/Gemfile",
|
|
110
|
+
"Dockerfile",
|
|
111
|
+
"**/Dockerfile",
|
|
112
|
+
|
|
113
|
+
// Test-runner configuration decides which tests run and what counts as a
|
|
114
|
+
// pass. Rewriting it is the cheapest way to make a suite green without
|
|
115
|
+
// touching a single assertion — `--passWithNoTests`, an added ignore
|
|
116
|
+
// pattern, a conftest.py fixture that stubs the thing under test — and the
|
|
117
|
+
// tamper guard reads test *files*, so none of it was visible to anything.
|
|
118
|
+
"conftest.py",
|
|
119
|
+
"**/conftest.py",
|
|
120
|
+
"pytest.ini",
|
|
121
|
+
"**/pytest.ini",
|
|
122
|
+
"tox.ini",
|
|
123
|
+
"**/tox.ini",
|
|
124
|
+
"jest.config.*",
|
|
125
|
+
"**/jest.config.*",
|
|
126
|
+
"vitest.config.*",
|
|
127
|
+
"**/vitest.config.*",
|
|
128
|
+
"**/.mocharc.*",
|
|
129
|
+
"karma.conf.*",
|
|
130
|
+
"**/karma.conf.*",
|
|
131
|
+
"phpunit.xml",
|
|
132
|
+
"**/phpunit.xml",
|
|
133
|
+
"**/.nycrc*",
|
|
62
134
|
];
|
|
63
135
|
|
|
64
136
|
const BLOCKED_KEYS = new Set(["__proto__", "constructor", "prototype"]);
|
package/src/dag-engine.mjs
CHANGED
|
@@ -431,7 +431,7 @@ export function resolveAffectedTests(modifiedFiles = [], options = {}) {
|
|
|
431
431
|
* @param {Function} isTaskFile - `isTaskFile` from engine.mjs, passed in to avoid a circular import.
|
|
432
432
|
* @returns {boolean}
|
|
433
433
|
*/
|
|
434
|
-
function isDagTaskFile(fileName, content, isTaskFile) {
|
|
434
|
+
export function isDagTaskFile(fileName, content, isTaskFile) {
|
|
435
435
|
if (fileName.endsWith(".task")) return true;
|
|
436
436
|
if (fileName.endsWith(".md")) return isTaskFile(fileName, content);
|
|
437
437
|
if (fileName.endsWith(".json")) {
|
package/src/engine.mjs
CHANGED
|
@@ -274,7 +274,7 @@ export async function gate(opts = {}) {
|
|
|
274
274
|
}
|
|
275
275
|
|
|
276
276
|
// Phase 3: Diff Secret Scanner & Security Checks
|
|
277
|
-
const secretResult = scanDiff(diffStr, { root });
|
|
277
|
+
const secretResult = scanDiff(diffStr, { root, allowTestModifications: opts.allowTestModifications === true });
|
|
278
278
|
// A binary file reaches the scanner as one summary line, so its contents were
|
|
279
279
|
// never looked at — a NUL byte in front of a token was enough to hide it.
|
|
280
280
|
// Inspect those files directly and fold the verdict in.
|
package/src/evidence.mjs
CHANGED
|
@@ -54,6 +54,30 @@ export function computeFileHash(filePath) {
|
|
|
54
54
|
* @param {string} [baseDir]
|
|
55
55
|
* @returns {string[]}
|
|
56
56
|
*/
|
|
57
|
+
/**
|
|
58
|
+
* Directories whose contents a build or a test run regenerates.
|
|
59
|
+
*
|
|
60
|
+
* These are not hashed. The test-integrity hash is taken before and after the
|
|
61
|
+
* verification command and any difference is reported as tampering — so a
|
|
62
|
+
* runner that writes its own cache *inside* the test tree accused itself. That
|
|
63
|
+
* is exactly what pytest does: it drops `tests/__pycache__/*.pyc` on first
|
|
64
|
+
* collection, the post-run hash no longer matched the pre-run hash, and every
|
|
65
|
+
* Python project on earth failed the gate with "test files changed during the
|
|
66
|
+
* run". A false accusation of tampering is worse than a missed one; it teaches
|
|
67
|
+
* the user to pass --allow-test-modifications by reflex.
|
|
68
|
+
*/
|
|
69
|
+
const ARTIFACT_DIRS = new Set([
|
|
70
|
+
".git", "node_modules", "target", "vendor",
|
|
71
|
+
"__pycache__", ".pytest_cache", ".mypy_cache", ".ruff_cache", ".tox",
|
|
72
|
+
".venv", "venv", ".gradle", ".nyc_output", "coverage", "htmlcov",
|
|
73
|
+
".next", ".nuxt", ".turbo", ".cache", ".parcel-cache",
|
|
74
|
+
]);
|
|
75
|
+
|
|
76
|
+
/** Compiled output. Regenerated from the sources that are hashed instead. */
|
|
77
|
+
const ARTIFACT_EXTENSIONS = new Set([
|
|
78
|
+
".pyc", ".pyo", ".pyd", ".class", ".o", ".obj", ".so", ".dll", ".dylib", ".exe", ".log",
|
|
79
|
+
]);
|
|
80
|
+
|
|
57
81
|
export function findFilesRecursively(dir, baseDir = dir) {
|
|
58
82
|
if (!existsSync(dir)) return [];
|
|
59
83
|
const results = [];
|
|
@@ -61,12 +85,11 @@ export function findFilesRecursively(dir, baseDir = dir) {
|
|
|
61
85
|
const entries = readdirSync(dir, { withFileTypes: true });
|
|
62
86
|
for (const entry of entries) {
|
|
63
87
|
const fullPath = join(dir, entry.name);
|
|
64
|
-
if (
|
|
65
|
-
continue;
|
|
66
|
-
}
|
|
88
|
+
if (ARTIFACT_DIRS.has(entry.name)) continue;
|
|
67
89
|
if (entry.isDirectory()) {
|
|
68
90
|
results.push(...findFilesRecursively(fullPath, baseDir));
|
|
69
91
|
} else if (entry.isFile()) {
|
|
92
|
+
if (ARTIFACT_EXTENSIONS.has(extname(entry.name).toLowerCase())) continue;
|
|
70
93
|
results.push(normalizePath(relative(baseDir, fullPath)));
|
|
71
94
|
}
|
|
72
95
|
}
|
|
@@ -579,7 +602,7 @@ export function generateEvidenceMarkdown(manifest) {
|
|
|
579
602
|
const lines = [
|
|
580
603
|
"### 🛡️ Autonomous Verification & Evidence Proof",
|
|
581
604
|
"",
|
|
582
|
-
`> **Evidence Manifest**: \`${manifest.manifestId}\` | **
|
|
605
|
+
`> **Evidence Manifest**: \`${manifest.manifestId}\` | **Digest**: \`${manifest.evidenceHash?.slice(0, 16)}...\``,
|
|
583
606
|
"",
|
|
584
607
|
"| Gate / Policy | Status | Evidence Record |",
|
|
585
608
|
"| :--- | :---: | :--- |",
|
package/src/git.mjs
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { execFileSync, execSync } from "node:child_process";
|
|
2
|
-
import { readFileSync, existsSync, statSync, readlinkSync } from "node:fs";
|
|
2
|
+
import { readFileSync, existsSync, statSync, lstatSync, readlinkSync } from "node:fs";
|
|
3
3
|
import { join, delimiter } from "node:path";
|
|
4
4
|
import { normalizePath, canonicalizePath } from "./config.mjs";
|
|
5
5
|
|
|
@@ -382,10 +382,38 @@ export function diffText(root = process.cwd(), base = "main", mode = "committed"
|
|
|
382
382
|
for (const file of untrackedFiles) {
|
|
383
383
|
try {
|
|
384
384
|
const fullPath = join(root, file);
|
|
385
|
+
|
|
386
|
+
// A symlink is its target *path*, never its target's contents — that is
|
|
387
|
+
// how git stores one and how `git diff` renders one. Reading through
|
|
388
|
+
// the link instead pulled whatever it pointed at into the diff, so
|
|
389
|
+
// `ln -s /etc/passwd notes.md` shipped that file's contents to the
|
|
390
|
+
// provider as ordinary added lines under an innocent name. Emit the
|
|
391
|
+
// link the way a committed symlink is emitted, and let checkScope judge
|
|
392
|
+
// the target it resolves to.
|
|
393
|
+
let linkStat = null;
|
|
394
|
+
try {
|
|
395
|
+
linkStat = lstatSync(fullPath);
|
|
396
|
+
} catch (_) {}
|
|
397
|
+
if (linkStat && linkStat.isSymbolicLink()) {
|
|
398
|
+
const target = readlinkSync(fullPath);
|
|
399
|
+
untrackedDiff += `\ndiff --git a/${file} b/${file}\nnew file mode 120000\n--- /dev/null\n+++ b/${file}\n`;
|
|
400
|
+
untrackedDiff += `@@ -0,0 +1 @@\n+${target}\n\\n`;
|
|
401
|
+
continue;
|
|
402
|
+
}
|
|
403
|
+
|
|
385
404
|
if (existsSync(fullPath)) {
|
|
386
405
|
const content = readFileSync(fullPath, "utf-8");
|
|
406
|
+
const addedLines = content.split(/\r?\n/);
|
|
407
|
+
// The `@@` header is not decoration. Every consumer that needs to know
|
|
408
|
+
// *which* line an addition is on parses it — the mutation harness and
|
|
409
|
+
// the V8 diff-coverage mapper both walk hunks — so a synthetic diff
|
|
410
|
+
// without one made untracked files invisible to them. Both silently
|
|
411
|
+
// reported nothing to do for a brand-new file, which is precisely
|
|
412
|
+
// where untested code arrives. The secret scanner never noticed
|
|
413
|
+
// because it only reads `+` lines.
|
|
387
414
|
untrackedDiff += `\ndiff --git a/${file} b/${file}\nnew file mode 100644\n--- /dev/null\n+++ b/${file}\n`;
|
|
388
|
-
untrackedDiff +=
|
|
415
|
+
untrackedDiff += `@@ -0,0 +1,${addedLines.length} @@\n`;
|
|
416
|
+
untrackedDiff += addedLines.map((line) => `+${line}`).join("\n") + "\n";
|
|
389
417
|
}
|
|
390
418
|
} catch (_) {}
|
|
391
419
|
}
|
|
@@ -493,17 +521,48 @@ export function symlinkChanges(root = process.cwd(), base = "main", mode = "comm
|
|
|
493
521
|
}
|
|
494
522
|
if (!target) continue;
|
|
495
523
|
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
? normalizePath(target)
|
|
499
|
-
: canonicalizePath(linkDir ? `${linkDir}/${target}` : target);
|
|
524
|
+
results.push({ link: entry.file, target: resolveLinkTarget(entry.file, target) });
|
|
525
|
+
}
|
|
500
526
|
|
|
501
|
-
|
|
527
|
+
// An untracked symlink appears in no `git diff --raw` at all, so in
|
|
528
|
+
// working-tree mode — the default for `agentctl check` — the loop above sees
|
|
529
|
+
// nothing. `ln -s /etc/os-release leak.txt` therefore reached checkScope as
|
|
530
|
+
// the plain name `leak.txt` and passed; committing the identical link to a
|
|
531
|
+
// branch and checking that was caught. The gate's answer must not depend on
|
|
532
|
+
// whether the attacker ran `git add`.
|
|
533
|
+
if (mode === "working-tree" || mode === "working") {
|
|
534
|
+
const untrackedRaw =
|
|
535
|
+
git(["-c", "core.quotePath=false", "ls-files", "-z", "--others", "--exclude-standard"], {
|
|
536
|
+
cwd: root,
|
|
537
|
+
raw: true,
|
|
538
|
+
ignoreError: true,
|
|
539
|
+
}) || "";
|
|
540
|
+
for (const rel of untrackedRaw.split("\0").map(normalizePath).filter(Boolean)) {
|
|
541
|
+
if (seen.has(rel)) continue;
|
|
542
|
+
let target = "";
|
|
543
|
+
try {
|
|
544
|
+
if (!lstatSync(join(root, rel)).isSymbolicLink()) continue;
|
|
545
|
+
target = readlinkSync(join(root, rel));
|
|
546
|
+
} catch (_) {
|
|
547
|
+
continue;
|
|
548
|
+
}
|
|
549
|
+
if (!target) continue;
|
|
550
|
+
seen.add(rel);
|
|
551
|
+
results.push({ link: rel, target: resolveLinkTarget(rel, target) });
|
|
552
|
+
}
|
|
502
553
|
}
|
|
503
554
|
|
|
504
555
|
return results;
|
|
505
556
|
}
|
|
506
557
|
|
|
558
|
+
/** Where a symlink at `link` pointing at `target` lands, as a repo-relative or absolute path. */
|
|
559
|
+
function resolveLinkTarget(link, target) {
|
|
560
|
+
const linkDir = normalizePath(link).split("/").slice(0, -1).join("/");
|
|
561
|
+
return normalizePath(target).startsWith("/")
|
|
562
|
+
? normalizePath(target)
|
|
563
|
+
: canonicalizePath(linkDir ? `${linkDir}/${target}` : target);
|
|
564
|
+
}
|
|
565
|
+
|
|
507
566
|
export function binaryDiffEntries(root = process.cwd(), base = "main", mode = "committed") {
|
|
508
567
|
const entries = new Map();
|
|
509
568
|
for (const entry of parseRawDiff(root, base, mode)) {
|
package/src/mutation.mjs
CHANGED
|
@@ -676,13 +676,23 @@ export function runMutationTest(options = {}) {
|
|
|
676
676
|
const selectedMutants = candidates.slice(0, maxMutants);
|
|
677
677
|
|
|
678
678
|
if (selectedMutants.length === 0) {
|
|
679
|
+
// A diff with no mutable operators yields no mutants, and reporting that as
|
|
680
|
+
// "100%" told operators their untested code had a perfect score — the exact
|
|
681
|
+
// false confidence the harness exists to remove. There is no score to
|
|
682
|
+
// report, so there is none: `mutationScore` is null and `reason` says why.
|
|
683
|
+
//
|
|
684
|
+
// `ok` stays true. Nothing was falsifiable, so nothing failed to be
|
|
685
|
+
// falsified; failing here would block every diff that only adds imports,
|
|
686
|
+
// constants or markdown, and a gate that cries wolf gets switched off.
|
|
679
687
|
return {
|
|
680
688
|
ok: true,
|
|
681
689
|
totalMutants: 0,
|
|
682
690
|
killedMutants: 0,
|
|
683
691
|
survivedMutants: 0,
|
|
684
692
|
errorMutants: 0,
|
|
685
|
-
mutationScore:
|
|
693
|
+
mutationScore: null,
|
|
694
|
+
scored: false,
|
|
695
|
+
reason: "No mutable operators in the added lines — nothing to falsify, so no score was computed.",
|
|
686
696
|
minScore,
|
|
687
697
|
results: [],
|
|
688
698
|
survivors: [],
|
|
@@ -115,23 +115,27 @@ export const COMMAND_REGISTRY = [
|
|
|
115
115
|
},
|
|
116
116
|
{
|
|
117
117
|
id: "swarm",
|
|
118
|
+
// Described as an inspector — `mutates: false`, `risk: low` — while the
|
|
119
|
+
// handler dispatches every queued task in parallel and spends budget.
|
|
120
|
+
// `--interactive` was advertised and implemented nowhere.
|
|
118
121
|
path: ["swarm"],
|
|
119
122
|
title: "swarm",
|
|
120
|
-
description: "
|
|
123
|
+
description: "Dispatch every queued task in parallel across worker slots",
|
|
121
124
|
category: "Operate",
|
|
122
|
-
mutates:
|
|
123
|
-
risk: "
|
|
124
|
-
interactive: "
|
|
125
|
+
mutates: true,
|
|
126
|
+
risk: "moderate",
|
|
127
|
+
interactive: "never",
|
|
125
128
|
requiresRepository: true,
|
|
126
129
|
shortcuts: ["s"],
|
|
127
130
|
examples: [
|
|
131
|
+
"agentctl swarm --dry-run",
|
|
128
132
|
"agentctl swarm",
|
|
129
|
-
"agentctl swarm --
|
|
130
|
-
"agentctl swarm --json",
|
|
133
|
+
"agentctl swarm --concurrency 3 --json",
|
|
131
134
|
],
|
|
132
135
|
flags: [
|
|
133
|
-
{ name: "
|
|
134
|
-
{ name: "
|
|
136
|
+
{ name: "concurrency", type: "string", description: "Parallel worker slots (defaults to limits.concurrency)" },
|
|
137
|
+
{ name: "dry-run", type: "boolean", description: "Report what would run without dispatching" },
|
|
138
|
+
{ name: "json", type: "boolean", description: "Output structured JSON swarm result" },
|
|
135
139
|
],
|
|
136
140
|
},
|
|
137
141
|
{
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { writeFileSync, mkdirSync } from "node:fs";
|
|
2
|
-
import { join
|
|
2
|
+
import { join } from "node:path";
|
|
3
3
|
import { detectPolyglotStack } from "../stack-detector.mjs";
|
|
4
4
|
import { resolveRoot } from "../config.mjs";
|
|
5
5
|
import { runCmd } from "../git.mjs";
|
|
@@ -26,45 +26,100 @@ export function scaffoldTddTest(spec = {}, options = {}) {
|
|
|
26
26
|
const details = spec.spec || "Feature requirement specification assertion.";
|
|
27
27
|
|
|
28
28
|
const stack = detectPolyglotStack(root);
|
|
29
|
-
const testDir = join(root, "test");
|
|
30
|
-
try {
|
|
31
|
-
mkdirSync(testDir, { recursive: true });
|
|
32
|
-
} catch (_) {}
|
|
33
29
|
|
|
34
|
-
|
|
35
|
-
|
|
30
|
+
// The generated oracle has to be written in the language its runner speaks.
|
|
31
|
+
//
|
|
32
|
+
// This used to emit a Node test file for every stack and then, in a Python
|
|
33
|
+
// project, run `pytest generated-x.test.mjs`. pytest exits 4 on a file it
|
|
34
|
+
// cannot collect, and the cycle read any non-zero exit as RED — so it
|
|
35
|
+
// reported a verified failing test, and locked an uncollectable file into
|
|
36
|
+
// scope.deny, having proven nothing at all.
|
|
37
|
+
const identifier = title.replace(/-/g, "_");
|
|
38
|
+
|
|
39
|
+
// The one string that must appear in the runner's output for the failure to
|
|
40
|
+
// be the *assertion* failing rather than the file never being collected.
|
|
41
|
+
const redMarker = `TDD Assertion Failed: Requirement '${title}' is not yet implemented.`;
|
|
42
|
+
const comment = (prefix) => details.split("\n").map((line) => `${prefix} ${line}`).join("\n");
|
|
43
|
+
|
|
44
|
+
let relDir = "test";
|
|
45
|
+
let fileName = `generated-${title}.test.mjs`;
|
|
46
|
+
let codeContent;
|
|
47
|
+
let testCmdParts;
|
|
48
|
+
let testCmdStr;
|
|
49
|
+
|
|
50
|
+
if (stack.stack === "python" || stack.stack === "django") {
|
|
51
|
+
fileName = `test_generated_${identifier}.py`;
|
|
52
|
+
codeContent = `def test_tdd_oracle_${identifier}():
|
|
53
|
+
# REQUIREMENT SPECIFICATION:
|
|
54
|
+
${comment(" #")}
|
|
55
|
+
is_implemented = False
|
|
56
|
+
assert is_implemented, "${redMarker}"
|
|
57
|
+
`;
|
|
58
|
+
testCmdParts = ["pytest", `${relDir}/${fileName}`];
|
|
59
|
+
testCmdStr = `pytest ${relDir}/${fileName}`;
|
|
60
|
+
} else if (stack.stack === "cargo") {
|
|
61
|
+
// Cargo discovers integration tests in tests/, not test/.
|
|
62
|
+
relDir = "tests";
|
|
63
|
+
fileName = `generated_${identifier}.rs`;
|
|
64
|
+
codeContent = `#[test]
|
|
65
|
+
fn tdd_oracle_${identifier}() {
|
|
66
|
+
// REQUIREMENT SPECIFICATION:
|
|
67
|
+
${comment(" //")}
|
|
68
|
+
let is_implemented = false;
|
|
69
|
+
assert!(is_implemented, "${redMarker}");
|
|
70
|
+
}
|
|
71
|
+
`;
|
|
72
|
+
testCmdParts = ["cargo", "test", "--test", `generated_${identifier}`];
|
|
73
|
+
testCmdStr = `cargo test --test generated_${identifier}`;
|
|
74
|
+
} else if (stack.stack === "go") {
|
|
75
|
+
fileName = `generated_${identifier}_test.go`;
|
|
76
|
+
codeContent = `package ${relDir}
|
|
36
77
|
|
|
37
|
-
|
|
78
|
+
import "testing"
|
|
79
|
+
|
|
80
|
+
func TestTddOracle${identifier.replace(/_/g, "")}(t *testing.T) {
|
|
81
|
+
// REQUIREMENT SPECIFICATION:
|
|
82
|
+
${comment("\t//")}
|
|
83
|
+
isImplemented := false
|
|
84
|
+
if !isImplemented {
|
|
85
|
+
t.Fatalf("${redMarker}")
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
`;
|
|
89
|
+
testCmdParts = ["go", "test", `./${relDir}/`];
|
|
90
|
+
testCmdStr = `go test ./${relDir}/`;
|
|
91
|
+
} else {
|
|
92
|
+
codeContent = `import test from "node:test";
|
|
38
93
|
import assert from "node:assert/strict";
|
|
39
94
|
|
|
40
95
|
test("TDD Oracle: ${title}", () => {
|
|
41
96
|
// REQUIREMENT SPECIFICATION:
|
|
42
|
-
|
|
97
|
+
${comment(" //")}
|
|
43
98
|
|
|
44
99
|
const isImplemented = false;
|
|
45
|
-
assert.equal(isImplemented, true, "
|
|
100
|
+
assert.equal(isImplemented, true, "${redMarker}");
|
|
46
101
|
});
|
|
47
102
|
`;
|
|
103
|
+
testCmdParts = ["node", "--test", join(root, relDir, fileName)];
|
|
104
|
+
testCmdStr = `node --test ${relDir}/${fileName}`;
|
|
105
|
+
}
|
|
48
106
|
|
|
49
|
-
|
|
107
|
+
const testDir = join(root, relDir);
|
|
108
|
+
try {
|
|
109
|
+
mkdirSync(testDir, { recursive: true });
|
|
110
|
+
} catch (_) {}
|
|
50
111
|
|
|
51
|
-
const
|
|
52
|
-
|
|
53
|
-
let testCmdStr = `node --test ${relativePath}`;
|
|
54
|
-
if (stack.stack === "python") {
|
|
55
|
-
testCmd = ["pytest", filePath];
|
|
56
|
-
testCmdStr = `pytest ${relativePath}`;
|
|
57
|
-
} else if (stack.stack === "cargo") {
|
|
58
|
-
testCmd = ["cargo", "test", "--test", title];
|
|
59
|
-
testCmdStr = `cargo test --test ${title}`;
|
|
60
|
-
}
|
|
112
|
+
const filePath = join(testDir, fileName);
|
|
113
|
+
writeFileSync(filePath, codeContent, "utf-8");
|
|
61
114
|
|
|
62
115
|
return {
|
|
63
116
|
filePath,
|
|
64
|
-
relativePath
|
|
117
|
+
relativePath: `${relDir}/${fileName}`,
|
|
65
118
|
codeContent,
|
|
66
|
-
testCmd,
|
|
119
|
+
testCmd: testCmdParts,
|
|
67
120
|
testCmdStr,
|
|
121
|
+
stack: stack.stack,
|
|
122
|
+
redMarker,
|
|
68
123
|
};
|
|
69
124
|
}
|
|
70
125
|
|
|
@@ -98,7 +153,21 @@ export async function runTddCycle(spec = {}, options = {}) {
|
|
|
98
153
|
|
|
99
154
|
if (redOutput.status === 0) {
|
|
100
155
|
throw new TddError(
|
|
101
|
-
`TDD RED check failed: Test '
|
|
156
|
+
`TDD RED check failed: Test '${scaffolded.relativePath}' passed initially! TDD tests must be falsifiable and fail before implementation.`
|
|
157
|
+
);
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// A non-zero exit is not proof the assertion ran. `pytest` exits 4 on a file
|
|
161
|
+
// it cannot collect and 5 when it collects nothing; a runner pointed at a
|
|
162
|
+
// file it does not understand exits non-zero for that reason alone. Reading
|
|
163
|
+
// either as RED is how a Node test file in a Python project came to be
|
|
164
|
+
// reported as a verified failing oracle. The generated assertion carries a
|
|
165
|
+
// marker; if the runner never reached it, the marker is not in the output.
|
|
166
|
+
const redText = `${redOutput.stdout || ""}\n${redOutput.stderr || ""}`;
|
|
167
|
+
if (scaffolded.redMarker && !redText.includes(scaffolded.redMarker)) {
|
|
168
|
+
throw new TddError(
|
|
169
|
+
`TDD RED check inconclusive: '${scaffolded.testCmdStr}' exited ${redOutput.status} without reaching the generated assertion, so nothing was proven falsifiable. ` +
|
|
170
|
+
`The runner most likely could not collect '${scaffolded.relativePath}' (detected stack: ${scaffolded.stack}). Output:\n${redText.trim().slice(0, 500)}`
|
|
102
171
|
);
|
|
103
172
|
}
|
|
104
173
|
|
package/src/security.mjs
CHANGED
|
@@ -838,9 +838,56 @@ function splitDiffByFile(diffText) {
|
|
|
838
838
|
|
|
839
839
|
const CANDIDATE_TOKEN_REGEX = /[A-Za-z0-9_-]{24,}/g;
|
|
840
840
|
|
|
841
|
+
/** A whole data: URI, base64-encoded or not. Its payload is not a credential. */
|
|
842
|
+
const DATA_URI_TOKEN = /\bdata:[a-z0-9.+-]+\/[a-z0-9.+-]*(?:;[a-z0-9.+=-]+)*,[^\s"'`<>)\]}]*/gi;
|
|
843
|
+
|
|
844
|
+
/** A subresource-integrity digest. High entropy by construction, public by design. */
|
|
845
|
+
const INTEGRITY_TOKEN = /\bsha(?:256|384|512)-[A-Za-z0-9+/=]+/gi;
|
|
846
|
+
|
|
847
|
+
/** A URL, from its scheme to the first character that cannot be part of one. */
|
|
848
|
+
const URL_TOKEN = /\b[a-z][a-z0-9+.-]*:\/\/[^\s"'`<>)\]},;]+/gi;
|
|
849
|
+
|
|
850
|
+
/**
|
|
851
|
+
* Remove from a line the noise that made URLs worth ignoring, and keep the
|
|
852
|
+
* parts of a URL that can carry a credential.
|
|
853
|
+
*
|
|
854
|
+
* This used to be `if (rawLine.includes("://")) continue;` — one substring
|
|
855
|
+
* anywhere on the line switched off entropy analysis for the entire line. So
|
|
856
|
+
* the scanner caught a bare 32-character key and let the identical key through
|
|
857
|
+
* the moment a comment carrying any http link sat beside it. An agent does not
|
|
858
|
+
* need to know why that works to stumble into it; a fetch call and its endpoint
|
|
859
|
+
* on one line is ordinary code.
|
|
860
|
+
*
|
|
861
|
+
* What actually justified the skip is narrower: a CDN path segment or an npm
|
|
862
|
+
* integrity hash looks exactly like a secret and is neither. Those are dropped
|
|
863
|
+
* here. A URL's userinfo and its query values are the opposite — `?api_key=…`
|
|
864
|
+
* and `//user:password@host` are where credentials genuinely hide — so they
|
|
865
|
+
* are carried over and scanned on their own.
|
|
866
|
+
*/
|
|
867
|
+
function stripEntropyNoise(rawLine) {
|
|
868
|
+
const carried = [];
|
|
869
|
+
let line = rawLine.replace(DATA_URI_TOKEN, " ").replace(INTEGRITY_TOKEN, " ");
|
|
870
|
+
line = line.replace(URL_TOKEN, (url) => {
|
|
871
|
+
const afterScheme = url.slice(url.indexOf("://") + 3);
|
|
872
|
+
const authority = afterScheme.split(/[/?#]/)[0];
|
|
873
|
+
const at = authority.lastIndexOf("@");
|
|
874
|
+
if (at > 0) carried.push(authority.slice(0, at));
|
|
875
|
+
const q = url.indexOf("?");
|
|
876
|
+
if (q !== -1) {
|
|
877
|
+
for (const pair of url.slice(q + 1).split(/[&;#]/)) {
|
|
878
|
+
const eq = pair.indexOf("=");
|
|
879
|
+
if (eq !== -1) carried.push(pair.slice(eq + 1));
|
|
880
|
+
}
|
|
881
|
+
}
|
|
882
|
+
return " ";
|
|
883
|
+
});
|
|
884
|
+
return carried.length ? `${line} ${carried.join(" ")}` : line;
|
|
885
|
+
}
|
|
886
|
+
|
|
841
887
|
/**
|
|
842
888
|
* Checks for high-entropy continuous tokens (>= 24 chars, entropy > 4.5) on added lines.
|
|
843
|
-
*
|
|
889
|
+
* Strips URLs, data: URIs, SRI hashes (sha512-, sha256-, sha384-) and skips lockfiles
|
|
890
|
+
* to eliminate false positives.
|
|
844
891
|
*
|
|
845
892
|
* @param {string} text - Text to scan
|
|
846
893
|
* @param {string|null} [file=null] - File path associated with the text
|
|
@@ -863,19 +910,11 @@ export function hasHighEntropyToken(text = "", file = null) {
|
|
|
863
910
|
|
|
864
911
|
const lines = text.split("\n");
|
|
865
912
|
for (const rawLine of lines) {
|
|
866
|
-
|
|
867
|
-
rawLine.includes("://") ||
|
|
868
|
-
rawLine.includes("data:image/") ||
|
|
869
|
-
rawLine.includes("sha512-") ||
|
|
870
|
-
rawLine.includes("sha256-") ||
|
|
871
|
-
rawLine.includes("sha384-")
|
|
872
|
-
) {
|
|
873
|
-
continue;
|
|
874
|
-
}
|
|
913
|
+
const line = stripEntropyNoise(rawLine);
|
|
875
914
|
|
|
876
915
|
CANDIDATE_TOKEN_REGEX.lastIndex = 0;
|
|
877
916
|
let match;
|
|
878
|
-
while ((match = CANDIDATE_TOKEN_REGEX.exec(
|
|
917
|
+
while ((match = CANDIDATE_TOKEN_REGEX.exec(line)) !== null) {
|
|
879
918
|
const token = match[0];
|
|
880
919
|
if (token.startsWith("sha512-") || token.startsWith("sha256-")) continue;
|
|
881
920
|
if (token.length >= 24) {
|
|
@@ -1138,6 +1177,22 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
1138
1177
|
);
|
|
1139
1178
|
const isSpecificAssertion = (str) => SPECIFIC_ASSERTION.test(str);
|
|
1140
1179
|
|
|
1180
|
+
/**
|
|
1181
|
+
* An assertion with every literal value replaced by a placeholder.
|
|
1182
|
+
*
|
|
1183
|
+
* Two lines that normalize to the same string are the same assertion about
|
|
1184
|
+
* the same expression; whatever differs between them is a value.
|
|
1185
|
+
*/
|
|
1186
|
+
const blankLiterals = (str) =>
|
|
1187
|
+
str
|
|
1188
|
+
.replace(/(['"`])(?:\\.|(?!\1)[^\\])*\1/g, "\u0000S")
|
|
1189
|
+
// The sign belongs to the literal: without it `3` and `-1` normalized to
|
|
1190
|
+
// different shapes and the rewritten expectation was never paired.
|
|
1191
|
+
.replace(/(?<![\w$])-?\d+(?:\.\d+)?\b/g, "\u0000N")
|
|
1192
|
+
.replace(/\b(?:true|false|null|undefined|None|True|False|nil)\b/g, "\u0000B")
|
|
1193
|
+
.replace(/\s+/g, " ")
|
|
1194
|
+
.trim();
|
|
1195
|
+
|
|
1141
1196
|
const fileAssertions = new Map();
|
|
1142
1197
|
|
|
1143
1198
|
for (let i = 0; i < lines.length; i++) {
|
|
@@ -1169,7 +1224,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
1169
1224
|
}
|
|
1170
1225
|
|
|
1171
1226
|
if (!fileAssertions.has(currentFile)) {
|
|
1172
|
-
fileAssertions.set(currentFile, { removed: [], added: 0, removedSpecific: [], addedSpecific: 0 });
|
|
1227
|
+
fileAssertions.set(currentFile, { removed: [], added: 0, addedTexts: [], removedSpecific: [], addedSpecific: 0 });
|
|
1173
1228
|
}
|
|
1174
1229
|
const fileStats = fileAssertions.get(currentFile);
|
|
1175
1230
|
|
|
@@ -1226,6 +1281,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
1226
1281
|
// If valid non-vacuous, non-commented assertion is added, increment added count
|
|
1227
1282
|
if (!isVacuous && !isCommented && !isCommentLine(addedText) && ASSERTION_PATTERN.test(addedText)) {
|
|
1228
1283
|
fileStats.added++;
|
|
1284
|
+
fileStats.addedTexts.push({ line: currentNewLineNo, text: addedText });
|
|
1229
1285
|
if (isSpecificAssertion(addedText)) fileStats.addedSpecific++;
|
|
1230
1286
|
}
|
|
1231
1287
|
|
|
@@ -1237,6 +1293,58 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
1237
1293
|
}
|
|
1238
1294
|
|
|
1239
1295
|
for (const [file, stats] of fileAssertions.entries()) {
|
|
1296
|
+
// An expectation that was rewritten rather than removed.
|
|
1297
|
+
//
|
|
1298
|
+
// Counting assertions cannot see this one: `assert.equal(add(1,2), 3)`
|
|
1299
|
+
// becoming `assert.equal(add(1,2), -1)` takes one specific assertion out
|
|
1300
|
+
// and puts one specific assertion back, so every total stayed level and the
|
|
1301
|
+
// guard said nothing — while the suite went from checking that addition
|
|
1302
|
+
// works to certifying that it is broken. It is the single cheapest way to
|
|
1303
|
+
// make a red suite green, and the one this tool exists to refuse.
|
|
1304
|
+
//
|
|
1305
|
+
// The signal is that the two lines are the *same assertion* with different
|
|
1306
|
+
// values in it: identical once every literal is blanked out, different
|
|
1307
|
+
// before. That does not distinguish an attack from a deliberate change of
|
|
1308
|
+
// spec — nothing can, from the diff alone — so this reports rather than
|
|
1309
|
+
// decides, and `--allow-test-modifications` is the answer when the new
|
|
1310
|
+
// expectation is the correct one.
|
|
1311
|
+
// Shapes are computed once per added line, not once per (removed, added)
|
|
1312
|
+
// pair: a large test refactor is O(n²) comparisons and the normalisation is
|
|
1313
|
+
// the expensive half of each one.
|
|
1314
|
+
const unpairedAdded = stats.addedTexts
|
|
1315
|
+
.filter((a) => isSpecificAssertion(a.text))
|
|
1316
|
+
.map((a) => ({ ...a, shape: blankLiterals(a.text) }));
|
|
1317
|
+
for (const removedItem of stats.removed.slice()) {
|
|
1318
|
+
if (!isSpecificAssertion(removedItem.text)) continue;
|
|
1319
|
+
const shape = blankLiterals(removedItem.text);
|
|
1320
|
+
const idx = unpairedAdded.findIndex(
|
|
1321
|
+
(a) => a.shape === shape && a.text.trim() !== removedItem.text.trim()
|
|
1322
|
+
);
|
|
1323
|
+
if (idx === -1) continue;
|
|
1324
|
+
const addedItem = unpairedAdded.splice(idx, 1)[0];
|
|
1325
|
+
|
|
1326
|
+
violations.push({
|
|
1327
|
+
file,
|
|
1328
|
+
line: addedItem.line ?? removedItem.line,
|
|
1329
|
+
type: "ASSERTION_EXPECTATION_CHANGED",
|
|
1330
|
+
reason:
|
|
1331
|
+
`Test Tamper Guard: Expected value rewritten in ${file}${addedItem.line ? `:${addedItem.line}` : ""} — ` +
|
|
1332
|
+
`"${removedItem.text.trim()}" became "${addedItem.text.trim()}". ` +
|
|
1333
|
+
`If the new value is the correct one, re-run with --allow-test-modifications.`,
|
|
1334
|
+
});
|
|
1335
|
+
|
|
1336
|
+
// Both sides are accounted for here, so they must not also feed the
|
|
1337
|
+
// count-based checks below — the same line reported twice under two
|
|
1338
|
+
// names tells the operator nothing extra.
|
|
1339
|
+
stats.removed.splice(stats.removed.indexOf(removedItem), 1);
|
|
1340
|
+
const rsIdx = stats.removedSpecific.indexOf(removedItem);
|
|
1341
|
+
if (rsIdx !== -1) {
|
|
1342
|
+
stats.removedSpecific.splice(rsIdx, 1);
|
|
1343
|
+
stats.addedSpecific--;
|
|
1344
|
+
}
|
|
1345
|
+
stats.added--;
|
|
1346
|
+
}
|
|
1347
|
+
|
|
1240
1348
|
if (stats.removed.length > stats.added) {
|
|
1241
1349
|
const unreplaced = stats.removed.slice(stats.added);
|
|
1242
1350
|
for (const item of unreplaced) {
|
package/src/stack-detector.mjs
CHANGED
|
@@ -1,5 +1,42 @@
|
|
|
1
1
|
import { readFileSync, writeFileSync, existsSync, readdirSync, mkdirSync } from "node:fs";
|
|
2
2
|
import { join, relative } from "node:path";
|
|
3
|
+
import { whichBinary } from "./provider-readiness.mjs";
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* The Python interpreter to invoke, by whatever name this machine has it under.
|
|
7
|
+
*
|
|
8
|
+
* `python3` does not exist on a default Windows install (the launcher is `py`,
|
|
9
|
+
* the Store package is `python`), so a hardcoded `python3` made every Python
|
|
10
|
+
* command in the kit unrunnable there.
|
|
11
|
+
*/
|
|
12
|
+
function pythonBin(env = process.env) {
|
|
13
|
+
for (const name of ["python3", "python", "py"]) {
|
|
14
|
+
if (whichBinary(name, env)) return name;
|
|
15
|
+
}
|
|
16
|
+
return "python3";
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* The command that runs a pytest suite.
|
|
21
|
+
*
|
|
22
|
+
* Not `pytest`. Invoked as a bare console script, pytest does not put the
|
|
23
|
+
* working directory on `sys.path`; invoked as `-m`, Python does. So the most
|
|
24
|
+
* ordinary Python layout there is — a module at the root, its test under
|
|
25
|
+
* `tests/` importing it — collapsed at collection with `ModuleNotFoundError:
|
|
26
|
+
* No module named 'calc'`, and the gate reported exit 4 on a project whose
|
|
27
|
+
* suite passes perfectly when the developer types `python3 -m pytest`. A
|
|
28
|
+
* first-run rejection of correct code is the most expensive failure this tool
|
|
29
|
+
* can produce: it teaches the user the gate is wrong.
|
|
30
|
+
*
|
|
31
|
+
* Falls back to the bare console script only when no interpreter can be found
|
|
32
|
+
* to host the module.
|
|
33
|
+
*/
|
|
34
|
+
export function pytestCmd(env = process.env) {
|
|
35
|
+
for (const name of ["python3", "python", "py"]) {
|
|
36
|
+
if (whichBinary(name, env)) return `${name} -m pytest`;
|
|
37
|
+
}
|
|
38
|
+
return "pytest";
|
|
39
|
+
}
|
|
3
40
|
|
|
4
41
|
/**
|
|
5
42
|
* Generate a zero-dependency smoke test file using node:test for untested JS/Generic repos.
|
|
@@ -186,7 +223,7 @@ export function detectPolyglotStack(projectRoot = process.cwd()) {
|
|
|
186
223
|
}
|
|
187
224
|
if (existsSync(join(projectRoot, "pyproject.toml")) || existsSync(join(projectRoot, "requirements.txt")) || existsSync(join(projectRoot, "setup.py"))) {
|
|
188
225
|
const triggerFile = existsSync(join(projectRoot, "pyproject.toml")) ? "pyproject.toml" : existsSync(join(projectRoot, "requirements.txt")) ? "requirements.txt" : "setup.py";
|
|
189
|
-
return { ...container, stack: "python", testCmd:
|
|
226
|
+
return { ...container, stack: "python", testCmd: pytestCmd(), buildCmd: `${pythonBin()} -m compileall -q .`, triggerFile };
|
|
190
227
|
}
|
|
191
228
|
if (existsSync(join(projectRoot, "mix.exs"))) {
|
|
192
229
|
return { ...container, stack: "mix", testCmd: "mix test", buildCmd: "mix compile", triggerFile: "mix.exs" };
|
|
@@ -270,7 +307,7 @@ export function detectPolyglotStack(projectRoot = process.cwd()) {
|
|
|
270
307
|
}
|
|
271
308
|
const pyFile = rootFiles.find((f) => f.endsWith(".py"));
|
|
272
309
|
if (pyFile) {
|
|
273
|
-
return { ...container, stack: "python", testCmd:
|
|
310
|
+
return { ...container, stack: "python", testCmd: pytestCmd(), buildCmd: `${pythonBin()} -m compileall -q .`, triggerFile: pyFile };
|
|
274
311
|
}
|
|
275
312
|
} catch (_) {}
|
|
276
313
|
|
|
@@ -699,7 +736,7 @@ export function bootstrapZeroTestRepo(root = process.cwd(), options = {}) {
|
|
|
699
736
|
if (detected.stack === "php" || detected.stack === "laravel" || detected.stack === "wordpress") {
|
|
700
737
|
testCmd = 'php -l $(find . -name "*.php" -not -path "./vendor/*" -not -path "./node_modules/*")';
|
|
701
738
|
} else if (detected.stack === "python") {
|
|
702
|
-
testCmd =
|
|
739
|
+
testCmd = `${pythonBin()} -m compileall -q -x '^\\./(\\.\\venv|venv|node_modules|\\.git)/' .`;
|
|
703
740
|
} else if (detected.stack === "dotnet") {
|
|
704
741
|
testCmd = "dotnet build --no-incremental --nologo";
|
|
705
742
|
} else if (detected.stack === "cargo") {
|
package/src/state.mjs
CHANGED
|
@@ -530,6 +530,54 @@ export function isPidAlive(pid, expectedStartTime = null) {
|
|
|
530
530
|
return true;
|
|
531
531
|
}
|
|
532
532
|
|
|
533
|
+
export const LOCK_TTL_MS = 7200000;
|
|
534
|
+
|
|
535
|
+
/**
|
|
536
|
+
* The epoch-ms instant a lock record stops holding.
|
|
537
|
+
*
|
|
538
|
+
* Records written before leases existed carry only `acquiredAt`, so they keep
|
|
539
|
+
* the two-hour window they were reaped on before.
|
|
540
|
+
*/
|
|
541
|
+
function lockExpiry(record) {
|
|
542
|
+
if (!record) return 0;
|
|
543
|
+
if (record.expiresAt) {
|
|
544
|
+
const t = new Date(record.expiresAt).getTime();
|
|
545
|
+
if (Number.isFinite(t)) return t;
|
|
546
|
+
}
|
|
547
|
+
if (record.acquiredAt) {
|
|
548
|
+
const t = new Date(record.acquiredAt).getTime();
|
|
549
|
+
if (Number.isFinite(t)) return t + LOCK_TTL_MS;
|
|
550
|
+
}
|
|
551
|
+
return Infinity;
|
|
552
|
+
}
|
|
553
|
+
|
|
554
|
+
/**
|
|
555
|
+
* Is this lock still held?
|
|
556
|
+
*
|
|
557
|
+
* There are two kinds of holder and only one of them has a process worth
|
|
558
|
+
* asking about.
|
|
559
|
+
*
|
|
560
|
+
* An in-process caller — the engine, the swarm — holds the lock for as long as
|
|
561
|
+
* it runs, so its pid is a real witness: if it died the lock is garbage, and
|
|
562
|
+
* reaping on a dead pid is what stops a crash from wedging the repo for the
|
|
563
|
+
* full TTL.
|
|
564
|
+
*
|
|
565
|
+
* `agentctl lock acquire` is the opposite. That process writes the file and
|
|
566
|
+
* exits *by design* — the agent it speaks for lives in some other process, on
|
|
567
|
+
* another machine, or has not started yet. Asking whether the CLI that wrote
|
|
568
|
+
* the record is alive therefore always answered "no", so the very next acquire
|
|
569
|
+
* reaped the lock and handed the same files to a second agent, telling both
|
|
570
|
+
* they had exclusive access. Two holders who each believe they are alone is
|
|
571
|
+
* worse than no lock at all. A record written that way marks itself `leased`,
|
|
572
|
+
* and then only its expiry decides.
|
|
573
|
+
*/
|
|
574
|
+
export function isLockLive(record) {
|
|
575
|
+
if (!record) return false;
|
|
576
|
+
if (Date.now() >= lockExpiry(record)) return false;
|
|
577
|
+
if (record.leased === true) return true;
|
|
578
|
+
return isPidAlive(record.pid, record.processStartTime ?? record.starttime ?? null);
|
|
579
|
+
}
|
|
580
|
+
|
|
533
581
|
function assertSafeTaskId(taskId) {
|
|
534
582
|
if (typeof taskId !== "string" || !/^[A-Za-z0-9][A-Za-z0-9._-]*$/.test(taskId) || basename(taskId) !== taskId) {
|
|
535
583
|
throw new Error("Invalid task id for lock path");
|
|
@@ -546,15 +594,16 @@ export function acquireLock(agentName, taskId, files = [], rootOrOpts = resolveR
|
|
|
546
594
|
if (existsSync(lockFile)) {
|
|
547
595
|
try {
|
|
548
596
|
const existing = JSON.parse(readFileSync(lockFile, "utf-8"));
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
597
|
+
if (isLockLive(existing)) {
|
|
598
|
+
return {
|
|
599
|
+
ok: false,
|
|
600
|
+
holder: existing.agent,
|
|
601
|
+
taskId,
|
|
602
|
+
pid: existing.pid,
|
|
603
|
+
expiresAt: existing.expiresAt || null,
|
|
604
|
+
};
|
|
557
605
|
}
|
|
606
|
+
try { unlinkSync(lockFile); } catch (_) {}
|
|
558
607
|
} catch (_) {
|
|
559
608
|
try { unlinkSync(lockFile); } catch (_) {}
|
|
560
609
|
}
|
|
@@ -573,11 +622,7 @@ export function acquireLock(agentName, taskId, files = [], rootOrOpts = resolveR
|
|
|
573
622
|
if (requested.size > 0) {
|
|
574
623
|
for (const held of lockStatus(root)) {
|
|
575
624
|
if (!held || held.taskId === taskId) continue;
|
|
576
|
-
|
|
577
|
-
const stillHeld =
|
|
578
|
-
isPidAlive(held.pid, recordedStart) &&
|
|
579
|
-
!(held.acquiredAt && Date.now() - new Date(held.acquiredAt).getTime() > 7200000);
|
|
580
|
-
if (!stillHeld) continue;
|
|
625
|
+
if (!isLockLive(held)) continue;
|
|
581
626
|
|
|
582
627
|
const overlap = (Array.isArray(held.files) ? held.files : [])
|
|
583
628
|
.map((f) => normalizePath(f))
|
|
@@ -588,13 +633,22 @@ export function acquireLock(agentName, taskId, files = [], rootOrOpts = resolveR
|
|
|
588
633
|
holder: held.agent,
|
|
589
634
|
taskId: held.taskId,
|
|
590
635
|
pid: held.pid,
|
|
636
|
+
expiresAt: held.expiresAt || null,
|
|
591
637
|
conflictingFiles: overlap,
|
|
592
638
|
};
|
|
593
639
|
}
|
|
594
640
|
}
|
|
595
641
|
}
|
|
596
642
|
|
|
597
|
-
|
|
643
|
+
// `leased` says the holder is not this process. A caller that will stay
|
|
644
|
+
// alive for the duration (the engine, the swarm) leaves it off and gets pid
|
|
645
|
+
// liveness; a one-shot CLI sets it and gets a plain time-bounded lease. See
|
|
646
|
+
// isLockLive.
|
|
647
|
+
const leased = opts?.lease === true;
|
|
648
|
+
const ownerPid = Number.isInteger(opts?.ownerPid) && opts.ownerPid > 0 ? opts.ownerPid : process.pid;
|
|
649
|
+
const ttlMs = Number.isFinite(opts?.ttlMs) && opts.ttlMs > 0 ? opts.ttlMs : LOCK_TTL_MS;
|
|
650
|
+
const startTime = getProcessStartTime(ownerPid);
|
|
651
|
+
const acquiredAt = new Date();
|
|
598
652
|
const concurrencyGroup = String(opts?.concurrencyGroup || opts?.concurrency_group || "").trim();
|
|
599
653
|
const payload = {
|
|
600
654
|
agent: agentName,
|
|
@@ -602,12 +656,14 @@ export function acquireLock(agentName, taskId, files = [], rootOrOpts = resolveR
|
|
|
602
656
|
branch,
|
|
603
657
|
files,
|
|
604
658
|
concurrencyGroup,
|
|
605
|
-
pid:
|
|
659
|
+
pid: ownerPid,
|
|
606
660
|
processStartTime: startTime,
|
|
607
661
|
starttime: startTime,
|
|
662
|
+
leased,
|
|
608
663
|
nonce: randomUUID(),
|
|
609
664
|
hostname: hostname(),
|
|
610
|
-
acquiredAt:
|
|
665
|
+
acquiredAt: acquiredAt.toISOString(),
|
|
666
|
+
expiresAt: new Date(acquiredAt.getTime() + ttlMs).toISOString(),
|
|
611
667
|
};
|
|
612
668
|
|
|
613
669
|
let fd;
|