jules-orchestrator-kit 0.69.0 → 0.71.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +53 -0
- package/README.md +1 -1
- package/ROADMAP_V1.md +18 -3
- package/bin/agentctl.mjs +61 -1
- package/package.json +1 -1
- package/scripts/guard-reach-check.mjs +46 -3
- package/src/config.mjs +88 -3
- package/src/engine.mjs +139 -24
- package/src/guard-policy.mjs +174 -0
- package/src/ops/test-collection.mjs +74 -0
- package/src/security.mjs +113 -2
- package/src/session-ops.mjs +117 -10
- package/src/stack-detector.mjs +52 -14
- package/src/wizard-init.mjs +38 -15
- package/src/wizard-task.mjs +15 -1
package/src/session-ops.mjs
CHANGED
|
@@ -31,6 +31,106 @@ export function parseAgeDuration(duration) {
|
|
|
31
31
|
}
|
|
32
32
|
}
|
|
33
33
|
|
|
34
|
+
/**
|
|
35
|
+
* Output shapes that mean "this command failed" even when the runner exited 0.
|
|
36
|
+
*
|
|
37
|
+
* Kept deliberately narrow: this list is only consulted for a `bashOutput`
|
|
38
|
+
* whose `exitCode` is `0` or absent, where the alternative is reporting
|
|
39
|
+
* nothing at all. A false positive here costs a few hundred characters of
|
|
40
|
+
* retry prompt; a false negative costs the retry its only evidence.
|
|
41
|
+
*/
|
|
42
|
+
const BASH_FAILURE_HINTS = [
|
|
43
|
+
/(^|\n)\s*not ok\b/i,
|
|
44
|
+
/(^|\n)# fail [1-9]/i,
|
|
45
|
+
/\b[1-9]\d* failed\b/i,
|
|
46
|
+
/\b[1-9]\d* failing\b/i,
|
|
47
|
+
/\bAssertionError\b/,
|
|
48
|
+
/\bFAILED\b/,
|
|
49
|
+
/\bTraceback \(most recent call last\)/,
|
|
50
|
+
/\bpanic: /,
|
|
51
|
+
/\berror:/i,
|
|
52
|
+
];
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Collects the diagnostics a session actually carries.
|
|
56
|
+
*
|
|
57
|
+
* The documented Activity type — the Jules API types reference, transcribed
|
|
58
|
+
* in `docs/jules-quality-plan.md` — puts command output under
|
|
59
|
+
* `artifacts[].bashOutput.{command,output,exitCode}` and the failure reason
|
|
60
|
+
* under `sessionFailed.reason`. Neither `act.error` nor `act.executionOutput`
|
|
61
|
+
* — the only two fields this file used to read — exists in that schema, so a
|
|
62
|
+
* real failure came back as the generic fallback sentence and the retry session
|
|
63
|
+
* was dispatched without the assertion it existed to fix.
|
|
64
|
+
*
|
|
65
|
+
* Blocks are returned highest-signal first, because `retrySession` truncates
|
|
66
|
+
* the joined result to 4000 characters from the front: the ordering decides
|
|
67
|
+
* which evidence survives the cut.
|
|
68
|
+
*
|
|
69
|
+
* The legacy spellings are still read. They cost nothing, and an unrecognised
|
|
70
|
+
* provider shape that does carry an error is better served by it than by the
|
|
71
|
+
* fallback sentence.
|
|
72
|
+
*
|
|
73
|
+
* @param {Array<object>} activities - Activities as returned by `listActivities`.
|
|
74
|
+
* @returns {Array<{ source: string, text: string }>} Highest-signal first.
|
|
75
|
+
*/
|
|
76
|
+
export function extractFailureDiagnostics(activities) {
|
|
77
|
+
const list = Array.isArray(activities) ? activities : [];
|
|
78
|
+
const failingBash = [];
|
|
79
|
+
const failureReasons = [];
|
|
80
|
+
const suspiciousBash = [];
|
|
81
|
+
const legacy = [];
|
|
82
|
+
const agentNotes = [];
|
|
83
|
+
const seen = new Set();
|
|
84
|
+
|
|
85
|
+
const push = (bucket, source, text) => {
|
|
86
|
+
const clean = String(text ?? "").trim();
|
|
87
|
+
if (!clean) return;
|
|
88
|
+
const key = `${source}\u0000${clean}`;
|
|
89
|
+
if (seen.has(key)) return;
|
|
90
|
+
seen.add(key);
|
|
91
|
+
bucket.push({ source, text: clean });
|
|
92
|
+
};
|
|
93
|
+
|
|
94
|
+
for (const act of list) {
|
|
95
|
+
if (!act || typeof act !== "object") continue;
|
|
96
|
+
|
|
97
|
+
const artifacts = Array.isArray(act.artifacts) ? act.artifacts : [];
|
|
98
|
+
for (const art of artifacts) {
|
|
99
|
+
const bash = art && typeof art === "object" ? art.bashOutput : null;
|
|
100
|
+
if (!bash || typeof bash !== "object") continue;
|
|
101
|
+
const command = typeof bash.command === "string" ? bash.command : "";
|
|
102
|
+
const output = typeof bash.output === "string" ? bash.output : "";
|
|
103
|
+
if (!command && !output) continue;
|
|
104
|
+
const rendered = `$ ${command || "(unknown command)"}\n${output || "(no output)"}`;
|
|
105
|
+
const exitCode = bash.exitCode;
|
|
106
|
+
if (typeof exitCode === "number" && exitCode !== 0) {
|
|
107
|
+
push(failingBash, "bashOutput", `${rendered}\n(exit code ${exitCode})`);
|
|
108
|
+
} else if (BASH_FAILURE_HINTS.some((re) => re.test(output))) {
|
|
109
|
+
const codeNote = typeof exitCode === "number" ? String(exitCode) : "not reported";
|
|
110
|
+
push(suspiciousBash, "bashOutput", `${rendered}\n(exit code ${codeNote})`);
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
const failed = act.sessionFailed && typeof act.sessionFailed === "object" ? act.sessionFailed : null;
|
|
115
|
+
if (failed && typeof failed.reason === "string") {
|
|
116
|
+
push(failureReasons, "sessionFailed", failed.reason);
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
if (act.error) {
|
|
120
|
+
push(legacy, "legacy.error", typeof act.error === "string" ? act.error : JSON.stringify(act.error));
|
|
121
|
+
}
|
|
122
|
+
if (act.executionOutput && (act.exitCode !== 0 || act.status === "FAILED")) {
|
|
123
|
+
push(legacy, "legacy.executionOutput", act.executionOutput);
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
if (act.originator === "agent" && typeof act.description === "string") {
|
|
127
|
+
push(agentNotes, "agentMessage", act.description);
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
return [...failingBash, ...failureReasons, ...suspiciousBash, ...legacy, ...agentNotes];
|
|
132
|
+
}
|
|
133
|
+
|
|
34
134
|
/**
|
|
35
135
|
* Extracts git diff patch, PR details, and modified files from a Jules session.
|
|
36
136
|
*
|
|
@@ -213,7 +313,7 @@ export async function applySessionPatch(sessionId, opts = {}) {
|
|
|
213
313
|
*
|
|
214
314
|
* @param {string} sessionId
|
|
215
315
|
* @param {object} opts
|
|
216
|
-
* @returns {Promise<{ ok: boolean, originalSessionId: string, newSession: object, failureReason: string }>}
|
|
316
|
+
* @returns {Promise<{ ok: boolean, originalSessionId: string, newSession: object, failureReason: string, diagnosticsFound: number, diagnosticSources: string[] }>}
|
|
217
317
|
*/
|
|
218
318
|
export async function retrySession(sessionId, opts = {}) {
|
|
219
319
|
if (!sessionId || typeof sessionId !== "string") {
|
|
@@ -227,18 +327,19 @@ export async function retrySession(sessionId, opts = {}) {
|
|
|
227
327
|
const activitiesRes = await provider.listActivities(sessionId, opts).catch(() => ({ activities: [] }));
|
|
228
328
|
const activities = activitiesRes.activities || [];
|
|
229
329
|
|
|
230
|
-
// Extract failure diagnostics
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
errors.push(act.executionOutput);
|
|
236
|
-
}
|
|
237
|
-
}
|
|
330
|
+
// Extract failure diagnostics. `extractFailureDiagnostics` reads the fields
|
|
331
|
+
// the API actually documents (artifacts[].bashOutput, sessionFailed.reason)
|
|
332
|
+
// and returns them highest-signal first, so the 4000-character cut below
|
|
333
|
+
// drops the least useful evidence rather than the assertion that failed.
|
|
334
|
+
const diagnostics = extractFailureDiagnostics(activities);
|
|
238
335
|
|
|
239
336
|
const rawPrompt = session?.raw?.prompt || session?.prompt || "";
|
|
240
337
|
const title = session?.raw?.title || `Retry of Session ${sessionId}`;
|
|
241
|
-
const failureReason =
|
|
338
|
+
const failureReason =
|
|
339
|
+
diagnostics
|
|
340
|
+
.map((d) => d.text)
|
|
341
|
+
.join("\n")
|
|
342
|
+
.slice(0, 4000) || "Previous session did not complete cleanly.";
|
|
242
343
|
|
|
243
344
|
let synthesizedPrompt = rawPrompt;
|
|
244
345
|
if (opts.withFailure !== false && failureReason) {
|
|
@@ -264,6 +365,12 @@ export async function retrySession(sessionId, opts = {}) {
|
|
|
264
365
|
originalSessionId: sessionId,
|
|
265
366
|
newSession: dispatchRes,
|
|
266
367
|
failureReason,
|
|
368
|
+
// Non-zero only when the session carried evidence the retry could act on.
|
|
369
|
+
// A zero here with a non-empty `failureReason` means the fallback sentence
|
|
370
|
+
// was sent — the distinction the CLI and any telemetry need, because the
|
|
371
|
+
// two look identical in `failureReason` alone.
|
|
372
|
+
diagnosticsFound: diagnostics.length,
|
|
373
|
+
diagnosticSources: diagnostics.map((d) => d.source),
|
|
267
374
|
};
|
|
268
375
|
}
|
|
269
376
|
|
package/src/stack-detector.mjs
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { readFileSync, writeFileSync, existsSync, readdirSync, mkdirSync } from "node:fs";
|
|
2
2
|
import { join, relative } from "node:path";
|
|
3
3
|
import { whichBinary } from "./provider-readiness.mjs";
|
|
4
|
+
import { yamlScalar } from "./config.mjs";
|
|
4
5
|
|
|
5
6
|
/**
|
|
6
7
|
* The Python interpreter to invoke, by whatever name this machine has it under.
|
|
@@ -31,11 +32,48 @@ function pythonBin(env = process.env) {
|
|
|
31
32
|
* Falls back to the bare console script only when no interpreter can be found
|
|
32
33
|
* to host the module.
|
|
33
34
|
*/
|
|
34
|
-
export function pytestCmd(env = process.env) {
|
|
35
|
+
export function pytestCmd(env = process.env, root = null) {
|
|
36
|
+
const prefix = root && isSrcLayout(root) ? "PYTHONPATH=src " : "";
|
|
35
37
|
for (const name of ["python3", "python", "py"]) {
|
|
36
|
-
if (whichBinary(name, env)) return `${name} -m pytest`;
|
|
38
|
+
if (whichBinary(name, env)) return `${prefix}${name} -m pytest`;
|
|
39
|
+
}
|
|
40
|
+
return `${prefix}pytest`;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Does this repository keep its package under `src/` rather than at the root?
|
|
45
|
+
*
|
|
46
|
+
* The sibling of the bug above, and a worse one. `-m` puts the *working
|
|
47
|
+
* directory* on `sys.path` — which is the fix for a module at the root, and
|
|
48
|
+
* no help at all when the package lives in `src/`. There, `import iniconfig`
|
|
49
|
+
* finds nothing in the working directory and falls through to whatever is
|
|
50
|
+
* installed in site-packages. The suite then runs green against a *different
|
|
51
|
+
* copy of the library than the one in the diff*: measured on `pytest-dev/
|
|
52
|
+
* iniconfig`, `_parse.py` gutted to `return False`, 49 tests passed, and the
|
|
53
|
+
* gate returned APPROVED (Exit 0).
|
|
54
|
+
*
|
|
55
|
+
* That is this project's worst failure shape — a check that examined
|
|
56
|
+
* something other than the thing under review, reporting a pass — and no
|
|
57
|
+
* amount of counting collected tests can see it, because the tests really
|
|
58
|
+
* did run. Only the import path can.
|
|
59
|
+
*
|
|
60
|
+
* `PYTHONPATH=src` is the ordinary spelling and puts the working tree first
|
|
61
|
+
* whether or not the package is also installed. Leading assignments are
|
|
62
|
+
* peeled into the child's environment by `runCommand`, so this needs no shell.
|
|
63
|
+
*
|
|
64
|
+
* Keyed on `__init__.py` under `src/`, which is the marker of a Python
|
|
65
|
+
* package and not of a Rust, C or JavaScript `src/` directory.
|
|
66
|
+
*/
|
|
67
|
+
export function isSrcLayout(root) {
|
|
68
|
+
const src = join(root, "src");
|
|
69
|
+
if (!existsSync(src)) return false;
|
|
70
|
+
try {
|
|
71
|
+
return readdirSync(src, { withFileTypes: true }).some(
|
|
72
|
+
(e) => e.isDirectory() && existsSync(join(src, e.name, "__init__.py"))
|
|
73
|
+
);
|
|
74
|
+
} catch (_) {
|
|
75
|
+
return false;
|
|
37
76
|
}
|
|
38
|
-
return "pytest";
|
|
39
77
|
}
|
|
40
78
|
|
|
41
79
|
/**
|
|
@@ -234,7 +272,7 @@ export function oracleCandidates(root = process.cwd(), detected = "") {
|
|
|
234
272
|
} catch (_) {}
|
|
235
273
|
}
|
|
236
274
|
if (has("pytest.ini") || has("pyproject.toml") || has("setup.py") || has("tox.ini") || has("setup.cfg")) {
|
|
237
|
-
push(pytestCmd());
|
|
275
|
+
push(pytestCmd(process.env, root));
|
|
238
276
|
}
|
|
239
277
|
if (has("Cargo.toml")) push("cargo test");
|
|
240
278
|
if (has("go.mod")) push("go test ./...");
|
|
@@ -386,7 +424,7 @@ export function detectPolyglotStack(projectRoot = process.cwd()) {
|
|
|
386
424
|
}
|
|
387
425
|
if (existsSync(join(projectRoot, "pyproject.toml")) || existsSync(join(projectRoot, "requirements.txt")) || existsSync(join(projectRoot, "setup.py"))) {
|
|
388
426
|
const triggerFile = existsSync(join(projectRoot, "pyproject.toml")) ? "pyproject.toml" : existsSync(join(projectRoot, "requirements.txt")) ? "requirements.txt" : "setup.py";
|
|
389
|
-
return { ...container, stack: "python", testCmd: pytestCmd(), buildCmd: `${pythonBin()} -m compileall -q .`, triggerFile };
|
|
427
|
+
return { ...container, stack: "python", testCmd: pytestCmd(process.env, projectRoot), buildCmd: `${pythonBin()} -m compileall -q .`, triggerFile };
|
|
390
428
|
}
|
|
391
429
|
if (existsSync(join(projectRoot, "mix.exs"))) {
|
|
392
430
|
return { ...container, stack: "mix", testCmd: "mix test", buildCmd: "mix compile", triggerFile: "mix.exs" };
|
|
@@ -470,7 +508,7 @@ export function detectPolyglotStack(projectRoot = process.cwd()) {
|
|
|
470
508
|
}
|
|
471
509
|
const pyFile = rootFiles.find((f) => f.endsWith(".py"));
|
|
472
510
|
if (pyFile) {
|
|
473
|
-
return { ...container, stack: "python", testCmd: pytestCmd(), buildCmd: `${pythonBin()} -m compileall -q .`, triggerFile: pyFile };
|
|
511
|
+
return { ...container, stack: "python", testCmd: pytestCmd(process.env, projectRoot), buildCmd: `${pythonBin()} -m compileall -q .`, triggerFile: pyFile };
|
|
474
512
|
}
|
|
475
513
|
} catch (_) {}
|
|
476
514
|
|
|
@@ -966,19 +1004,19 @@ export function bootstrapZeroTestRepo(root = process.cwd(), options = {}) {
|
|
|
966
1004
|
try {
|
|
967
1005
|
let rawConfig = readFileSync(configPath, "utf-8");
|
|
968
1006
|
if (/^\s*test:\s*.*$/m.test(rawConfig)) {
|
|
969
|
-
rawConfig = rawConfig.replace(/^\s*test:\s*.*$/m, ` test:
|
|
1007
|
+
rawConfig = rawConfig.replace(/^\s*test:\s*.*$/m, () => ` test: ${yamlScalar(testCmd)}`);
|
|
970
1008
|
} else if (/^\s*verify:\s*$/m.test(rawConfig)) {
|
|
971
|
-
rawConfig = rawConfig.replace(/^\s*verify:\s*$/m, `verify:\n test:
|
|
1009
|
+
rawConfig = rawConfig.replace(/^\s*verify:\s*$/m, () => `verify:\n test: ${yamlScalar(testCmd)}`);
|
|
972
1010
|
} else {
|
|
973
|
-
rawConfig += `\nverify:\n test:
|
|
1011
|
+
rawConfig += `\nverify:\n test: ${yamlScalar(testCmd)}\n`;
|
|
974
1012
|
}
|
|
975
1013
|
if (detected.buildCmd && /^\s*build:\s*["']?["']?\s*$/m.test(rawConfig)) {
|
|
976
|
-
rawConfig = rawConfig.replace(/^\s*build:\s*.*$/m, ` build:
|
|
1014
|
+
rawConfig = rawConfig.replace(/^\s*build:\s*.*$/m, () => ` build: ${yamlScalar(detected.buildCmd)}`);
|
|
977
1015
|
}
|
|
978
1016
|
writeFileSync(configPath, rawConfig, "utf-8");
|
|
979
1017
|
} catch (_) {}
|
|
980
1018
|
} else {
|
|
981
|
-
const cfg = `version: 1\nprovider: jules\ntier: free\nverify:\n test:
|
|
1019
|
+
const cfg = `version: 1\nprovider: jules\ntier: free\nverify:\n test: ${yamlScalar(testCmd)}\n build: ${yamlScalar(detected.buildCmd || "")}\nlimits:\n diff_kb: 75\n daily_tasks: 15\n repair_attempts: 3\nbranch_prefix: agent/\nbase_branch: main\n`;
|
|
982
1020
|
writeFileSync(configPath, cfg, "utf-8");
|
|
983
1021
|
}
|
|
984
1022
|
|
|
@@ -987,12 +1025,12 @@ export function bootstrapZeroTestRepo(root = process.cwd(), options = {}) {
|
|
|
987
1025
|
try {
|
|
988
1026
|
let rawJules = readFileSync(julesPath, "utf-8");
|
|
989
1027
|
if (/^\s*test_cmd:\s*.*$/m.test(rawJules)) {
|
|
990
|
-
rawJules = rawJules.replace(/^\s*test_cmd:\s*.*$/m, `test_cmd:
|
|
1028
|
+
rawJules = rawJules.replace(/^\s*test_cmd:\s*.*$/m, () => `test_cmd: ${yamlScalar(testCmd)}`);
|
|
991
1029
|
} else {
|
|
992
|
-
rawJules += `\ntest_cmd:
|
|
1030
|
+
rawJules += `\ntest_cmd: ${yamlScalar(testCmd)}\n`;
|
|
993
1031
|
}
|
|
994
1032
|
if (detected.buildCmd && /^\s*build_cmd:\s*["']?["']?\s*$/m.test(rawJules)) {
|
|
995
|
-
rawJules = rawJules.replace(/^\s*build_cmd:\s*.*$/m, `build_cmd:
|
|
1033
|
+
rawJules = rawJules.replace(/^\s*build_cmd:\s*.*$/m, () => `build_cmd: ${yamlScalar(detected.buildCmd)}`);
|
|
996
1034
|
}
|
|
997
1035
|
writeFileSync(julesPath, rawJules, "utf-8");
|
|
998
1036
|
} catch (_) {}
|
package/src/wizard-init.mjs
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
import { existsSync, readFileSync, writeFileSync, openSync, fsyncSync, closeSync, renameSync, mkdirSync, readdirSync } from "node:fs";
|
|
2
2
|
import { join } from "node:path";
|
|
3
|
-
import { parseYaml, TIER_PRESETS, VENDOR_TIERS, FALLBACK_TIER } from "./config.mjs";
|
|
3
|
+
import { parseYaml, yamlScalar, TIER_PRESETS, VENDOR_TIERS, FALLBACK_TIER } from "./config.mjs";
|
|
4
4
|
import { suggestProvider, detectAvailableProviders } from "./provider-readiness.mjs";
|
|
5
5
|
import { detectDefaultBranch } from "./git.mjs";
|
|
6
6
|
import { resolveWorkspaceBoundary, oracleCandidates } from "./stack-detector.mjs";
|
|
7
7
|
import { PROFILE_NAMES, PROFILE_DESCRIPTIONS } from "./profiles.mjs";
|
|
8
8
|
import { detectStackOracles, runVerificationProbe } from "./wizard-oracle.mjs";
|
|
9
|
-
import { parseCollectedTests } from "./ops/test-collection.mjs";
|
|
9
|
+
import { parseCollectedTests, producedNoOutput, looksLikeTestSuiteCommand } from "./ops/test-collection.mjs";
|
|
10
10
|
import { select, multiSelect, input, confirm, spinner, isTTY } from "./tui.mjs";
|
|
11
11
|
import { KIT_VERSION } from "./version.mjs";
|
|
12
12
|
|
|
@@ -201,6 +201,19 @@ export function planInit(root = process.cwd(), options = {}) {
|
|
|
201
201
|
? `\nlimits:\n concurrency: ${limits.concurrency}\n daily_tasks: ${limits.daily_tasks}\n stagger_ms: ${limits.stagger_ms}\n diff_kb: ${limits.diff_kb}\n`
|
|
202
202
|
: "";
|
|
203
203
|
|
|
204
|
+
// A generated comment must not begin with an ESLint directive keyword.
|
|
205
|
+
//
|
|
206
|
+
// `global`, `globals`, `exported`, `eslint`, `eslint-disable` and friends are
|
|
207
|
+
// configuration when they open a comment — in any language ESLint has a
|
|
208
|
+
// parser for, YAML included. This template began a line with "global runs
|
|
209
|
+
// the ...", which ESLint read as `/* global runs, the, ... */`: a declaration
|
|
210
|
+
// of globals named after each word of the sentence. Measured on
|
|
211
|
+
// `unjs/unimport`, that produced 18 `no-unused-vars` errors quoting
|
|
212
|
+
// individual English words back at the user, on a file `init` had written
|
|
213
|
+
// thirty seconds earlier.
|
|
214
|
+
//
|
|
215
|
+
// The word is unavoidable — `global` is the name of the setting being
|
|
216
|
+
// explained — so the sentence leads with the key instead.
|
|
204
217
|
const configYaml = `# Agent Orchestrator Kit Config (v${KIT_VERSION})
|
|
205
218
|
# provider: jules | claude-code | codex | gemini-flash (agentctl providers)
|
|
206
219
|
version: 1
|
|
@@ -212,16 +225,16 @@ ${limitsBlock}
|
|
|
212
225
|
verify:
|
|
213
226
|
# minimal | standard | max — see: agentctl profile --list
|
|
214
227
|
profile: ${profile}
|
|
215
|
-
# global runs the repository
|
|
216
|
-
# to their sub-projects
|
|
228
|
+
# scope: global runs the commands this repository declares; affected
|
|
229
|
+
# resolves changed files to their sub-projects, running only those suites
|
|
217
230
|
scope: ${verifyScope}
|
|
218
|
-
test:
|
|
231
|
+
test: ${yamlScalar(verify.test)}
|
|
219
232
|
# How long a verification stage may run before the gate kills it (default
|
|
220
233
|
# 300000). Raise it for a suite that legitimately takes longer.
|
|
221
234
|
timeout_ms: 300000
|
|
222
|
-
build:
|
|
223
|
-
lint:
|
|
224
|
-
typecheck:
|
|
235
|
+
build: ${yamlScalar(verify.build)}
|
|
236
|
+
lint: ${yamlScalar(verify.lint)}
|
|
237
|
+
typecheck: ${yamlScalar(verify.typecheck)}
|
|
225
238
|
|
|
226
239
|
presets:
|
|
227
240
|
${selectedPresets.map((p) => ` - ${p}`).join("\n")}
|
|
@@ -246,11 +259,11 @@ ${selectedPresets.map((p) => ` - ${p}`).join("\n")}
|
|
|
246
259
|
|
|
247
260
|
const julesYaml = `# Google Jules Repository Configuration (Version 2)
|
|
248
261
|
version: 2
|
|
249
|
-
test_cmd:
|
|
250
|
-
build_cmd:
|
|
262
|
+
test_cmd: ${yamlScalar(verify.test)}
|
|
263
|
+
build_cmd: ${yamlScalar(verify.build)}
|
|
251
264
|
forbidden_paths:
|
|
252
|
-
${forbiddenPaths.map((p) => ` -
|
|
253
|
-
allow_paths: ${allowPaths.length > 0 ? "\n" + allowPaths.map((p) => ` -
|
|
265
|
+
${forbiddenPaths.map((p) => ` - ${yamlScalar(p)}`).join("\n")}
|
|
266
|
+
allow_paths: ${allowPaths.length > 0 ? "\n" + allowPaths.map((p) => ` - ${yamlScalar(p)}`).join("\n") : "[]"}
|
|
254
267
|
`;
|
|
255
268
|
|
|
256
269
|
return {
|
|
@@ -330,8 +343,18 @@ export function loadPresets(root = process.cwd()) {
|
|
|
330
343
|
* cost of rejecting a candidate is trying the next one, where at gate time it
|
|
331
344
|
* would be a hard red on a repository that is fine.
|
|
332
345
|
*/
|
|
333
|
-
function probeVerdict(probeRes) {
|
|
346
|
+
function probeVerdict(probeRes, cmd) {
|
|
334
347
|
if (!probeRes.ok) return "failed";
|
|
348
|
+
// Writing nothing at all is `empty`, not `silent`. The distinction is the
|
|
349
|
+
// whole point: `silent` is the forgiving bucket that keeps a command the
|
|
350
|
+
// guard could not read, and `pnpm -r test` landed in it because it prints
|
|
351
|
+
// no output to be unreadable. So the candidate this verdict was introduced
|
|
352
|
+
// to reject was the one case it waved through, and every repository
|
|
353
|
+
// scaffolded on such a workspace kept it.
|
|
354
|
+
// Same rule as the gate's floor, from the same predicate: a command that
|
|
355
|
+
// claims to run a suite and printed nothing ran none. A static gate that
|
|
356
|
+
// printed nothing did what it promised, so it keeps the forgiving verdict.
|
|
357
|
+
if (looksLikeTestSuiteCommand(cmd) && producedNoOutput(probeRes.stdout, probeRes.stderr)) return "empty";
|
|
335
358
|
const { count } = parseCollectedTests(probeRes.stdout, probeRes.stderr);
|
|
336
359
|
if (count === null) return "silent";
|
|
337
360
|
return count > 0 ? "ran" : "empty";
|
|
@@ -341,7 +364,7 @@ async function resolveRunnableOracle(root, testCmd, options = {}) {
|
|
|
341
364
|
if (!testCmd) return testCmd;
|
|
342
365
|
const probeSp = spinner(`Probing oracle: ${testCmd}`, options);
|
|
343
366
|
const probeRes = await runVerificationProbe(testCmd, root);
|
|
344
|
-
const verdict = probeVerdict(probeRes);
|
|
367
|
+
const verdict = probeVerdict(probeRes, testCmd);
|
|
345
368
|
if (verdict === "ran") {
|
|
346
369
|
probeSp.stop(`Oracle verified successfully (${probeRes.durationMs}ms)`);
|
|
347
370
|
return testCmd;
|
|
@@ -364,7 +387,7 @@ async function resolveRunnableOracle(root, testCmd, options = {}) {
|
|
|
364
387
|
for (const cand of alternates) {
|
|
365
388
|
const altSp = spinner(`Trying ${cand}`, options);
|
|
366
389
|
const altRes = await runVerificationProbe(cand, root);
|
|
367
|
-
const altVerdict = probeVerdict(altRes);
|
|
390
|
+
const altVerdict = probeVerdict(altRes, cand);
|
|
368
391
|
if (altVerdict === "ran") {
|
|
369
392
|
altSp.stop(`${cand} runs here (${altRes.durationMs}ms) — using it instead`);
|
|
370
393
|
return cand;
|
package/src/wizard-task.mjs
CHANGED
|
@@ -261,7 +261,21 @@ ${fullPrompt}
|
|
|
261
261
|
* @returns {Promise<{ ok: boolean, dryRun: boolean, taskFile: string, written: boolean, plan: object }>}
|
|
262
262
|
*/
|
|
263
263
|
export async function runTaskCreateWizard(root = process.cwd(), options = {}) {
|
|
264
|
-
|
|
264
|
+
// `-p` is the documented way to skip the questions, so it has to skip them.
|
|
265
|
+
//
|
|
266
|
+
// README: "Pass the prompt to skip straight to review:
|
|
267
|
+
// npx jules-orchestrator-kit task create -p 'Refactor the invoice module'".
|
|
268
|
+
// Interactivity was decided by `isTTY` alone, and `-p` was consulted only by
|
|
269
|
+
// the TODO-import branch below — so in a real terminal the advertised
|
|
270
|
+
// quickstart stopped at "? Task Title" and waited for a keypress forever,
|
|
271
|
+
// then asked for the instructions it had already been handed.
|
|
272
|
+
//
|
|
273
|
+
// It looked fine under test because a non-TTY run takes the headless path and
|
|
274
|
+
// never asks. That is the same defect the flag itself has: one rule, two
|
|
275
|
+
// paths, and only the path nobody was watching kept the old answer. Making
|
|
276
|
+
// `-p` mean the headless path everywhere makes the two agree by construction.
|
|
277
|
+
const promptSupplied = typeof options.prompt === "string" && options.prompt.trim() !== "";
|
|
278
|
+
const interactive = options.interactive !== false && !promptSupplied && isTTY(options.stdin || process.stdin);
|
|
265
279
|
|
|
266
280
|
let title = options.title;
|
|
267
281
|
let promptText = options.prompt;
|