jules-orchestrator-kit 0.67.0 → 0.69.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,6 +6,7 @@ import { detectDefaultBranch } from "./git.mjs";
6
6
  import { resolveWorkspaceBoundary, oracleCandidates } from "./stack-detector.mjs";
7
7
  import { PROFILE_NAMES, PROFILE_DESCRIPTIONS } from "./profiles.mjs";
8
8
  import { detectStackOracles, runVerificationProbe } from "./wizard-oracle.mjs";
9
+ import { parseCollectedTests } from "./ops/test-collection.mjs";
9
10
  import { select, multiSelect, input, confirm, spinner, isTTY } from "./tui.mjs";
10
11
  import { KIT_VERSION } from "./version.mjs";
11
12
 
@@ -215,6 +216,9 @@ verify:
215
216
  # to their sub-projects and runs only those suites (monorepos)
216
217
  scope: ${verifyScope}
217
218
  test: "${verify.test}"
219
+ # How long a verification stage may run before the gate kills it (default
220
+ # 300000). Raise it for a suite that legitimately takes longer.
221
+ timeout_ms: 300000
218
222
  build: "${verify.build}"
219
223
  lint: "${verify.lint}"
220
224
  typecheck: "${verify.typecheck}"
@@ -314,25 +318,77 @@ export function loadPresets(root = process.cwd()) {
314
318
  *
315
319
  * @returns {Promise<string>} the command to save
316
320
  */
321
+ /**
322
+ * How much a probe actually proved.
323
+ *
324
+ * Exit 0 is the weakest of the three answers. `pnpm -r test` on a workspace
325
+ * whose packages declare no test script exits 0, prints nothing, and runs
326
+ * nothing — and it was accepted here as a verified oracle, which left the
327
+ * repository configured to approve every future change against silence.
328
+ *
329
+ * Choosing a command is the right moment to be strict about this: at init the
330
+ * cost of rejecting a candidate is trying the next one, where at gate time it
331
+ * would be a hard red on a repository that is fine.
332
+ */
333
+ function probeVerdict(probeRes) {
334
+ if (!probeRes.ok) return "failed";
335
+ const { count } = parseCollectedTests(probeRes.stdout, probeRes.stderr);
336
+ if (count === null) return "silent";
337
+ return count > 0 ? "ran" : "empty";
338
+ }
339
+
317
340
  async function resolveRunnableOracle(root, testCmd, options = {}) {
318
341
  if (!testCmd) return testCmd;
319
342
  const probeSp = spinner(`Probing oracle: ${testCmd}`, options);
320
343
  const probeRes = await runVerificationProbe(testCmd, root);
321
- if (probeRes.ok) {
344
+ const verdict = probeVerdict(probeRes);
345
+ if (verdict === "ran") {
322
346
  probeSp.stop(`Oracle verified successfully (${probeRes.durationMs}ms)`);
323
347
  return testCmd;
324
348
  }
325
- probeSp.fail(`Oracle verification probe failed (Exit ${probeRes.code})`);
349
+ if (verdict === "failed") {
350
+ probeSp.fail(`Oracle verification probe failed (Exit ${probeRes.code})`);
351
+ } else {
352
+ probeSp.fail(
353
+ verdict === "empty"
354
+ ? `"${testCmd}" exited 0 but ran no tests — looking for a command that does`
355
+ : `"${testCmd}" exited 0 without stating how many tests it ran — looking for a better one`
356
+ );
357
+ }
326
358
 
327
359
  const alternates = oracleCandidates(root, testCmd).filter((c) => c !== testCmd).slice(0, 3);
360
+ // Two passes: prefer a command that proves it ran something, and only then
361
+ // settle for one that merely exits 0. Falling back on the first exit-0
362
+ // candidate would reintroduce exactly the silence this rejects.
363
+ const settled = [];
328
364
  for (const cand of alternates) {
329
365
  const altSp = spinner(`Trying ${cand}`, options);
330
366
  const altRes = await runVerificationProbe(cand, root);
331
- if (altRes.ok) {
367
+ const altVerdict = probeVerdict(altRes);
368
+ if (altVerdict === "ran") {
332
369
  altSp.stop(`${cand} runs here (${altRes.durationMs}ms) — using it instead`);
333
370
  return cand;
334
371
  }
335
- altSp.fail(`${cand} also failed (Exit ${altRes.code})`);
372
+ if (altVerdict === "silent") {
373
+ altSp.fail(`${cand} exited 0 without stating a test count`);
374
+ settled.push(cand);
375
+ continue;
376
+ }
377
+ altSp.fail(altVerdict === "empty" ? `${cand} ran no tests` : `${cand} also failed (Exit ${altRes.code})`);
378
+ }
379
+
380
+ // Nothing proved it ran a suite. A command that at least starts and exits 0
381
+ // still beats one that does not run at all, so it is used — and said out
382
+ // loud, because the gate will only be able to check it by exit code.
383
+ if (settled.length > 0 || verdict === "silent") {
384
+ const chosen = verdict === "silent" ? testCmd : settled[0];
385
+ const out = options.stdout || process.stdout;
386
+ out.write("\n");
387
+ out.write(` \u26a0\ufe0f "${chosen}" exits 0 but states no test count.\n`);
388
+ out.write(" The gate can verify it by exit code alone, which cannot tell a\n");
389
+ out.write(" full suite from a command that ran nothing. If this repository\n");
390
+ out.write(" has a suite, point verify.test at it in .agent/config.yml.\n\n");
391
+ return chosen;
336
392
  }
337
393
 
338
394
  // Nothing runs. Say so in terms the user can act on, rather than leaving a