jules-orchestrator-kit 0.71.0 → 0.72.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -49,6 +49,13 @@ export const TEST_PATH_CASES = [
49
49
  { path: "test/Token.t.sol", expected: true, why: "foundry" },
50
50
  // Monorepo position
51
51
  { path: "packages/api/test/handler.test.js", expected: true, why: "monorepo package" },
52
+ // The canonical root test file for the supported Node runners. AVA runs
53
+ // root `test.js` with no configuration; `node --test test.js` does too.
54
+ // P-Limit's whole suite lived in one and the guard watched none of it.
55
+ { path: "test.js", expected: true, why: "node root canonical — F01's blind spot" },
56
+ { path: "test.mjs", expected: true, why: "node root, ESM" },
57
+ { path: "test.ts", expected: true, why: "node root, TypeScript" },
58
+ { path: "test.cjs", expected: true, why: "node root, CommonJS" },
52
59
 
53
60
  // Near-misses: segment match, not substring
54
61
  { path: "latest/build.js", expected: false, why: "`latest` is not `test`" },
@@ -57,6 +64,15 @@ export const TEST_PATH_CASES = [
57
64
  { path: "src/attestation.js", expected: false, why: "contains 'test', is not one" },
58
65
  { path: "src/index.js", expected: false, why: "ordinary source" },
59
66
  { path: "docs/testing-guide.md", expected: false, why: "documentation about testing" },
67
+ // The bare-name rule is Node-family only: `test.py` is not pytest (pytest
68
+ // collects `test_*.py`), `test.go` is not a Go test (`*_test.go`), and
69
+ // `test.rb` is neither RSpec nor Minitest.
70
+ { path: "test.py", expected: false, why: "pytest collects test_*.py, never a bare test.py" },
71
+ { path: "test.go", expected: false, why: "go tests are *_test.go" },
72
+ { path: "test.rb", expected: false, why: "ruby suites are *_spec.rb or test_*.rb" },
73
+ { path: "test.helpers.js", expected: false, why: "a support module named after tests" },
74
+ { path: "tests.js", expected: false, why: "plural name, ordinary source" },
75
+ { path: "testing.js", expected: false, why: "contains the word, is not the file" },
60
76
  ];
61
77
 
62
78
  /**
@@ -386,9 +402,211 @@ export const TAMPER_CANARIES = [
386
402
  added: [" throw XCTSkip(\"not now\")"],
387
403
  expect: "TEST_SKIP_INJECTION",
388
404
  },
405
+ // F01: P-Limit's root test.js is a supported runner's canonical test
406
+ // file. Every attack below had to be invisible inside it before the
407
+ // classifier knew the file existed — deletion, vacuous replacement, and
408
+ // an expectation rewritten to fit broken output.
409
+ {
410
+ id: "removal/node-root",
411
+ file: "test.js",
412
+ context: "// pending counts",
413
+ removed: ["\tt.is(limit.activeCount, 0);"],
414
+ added: [],
415
+ expect: "ASSERTION_REMOVAL",
416
+ },
417
+ {
418
+ id: "vacuous/node-root",
419
+ file: "test.js",
420
+ context: "// pending counts",
421
+ removed: ["\tt.is(limit.activeCount, 0);"],
422
+ added: ["\tt.true(true);"],
423
+ expect: "VACUOUS_ASSERTION",
424
+ },
425
+ {
426
+ id: "expectation/node-root",
427
+ file: "test.js",
428
+ context: "// pending counts",
429
+ removed: ["\tt.is(limit.pendingCount, 0);"],
430
+ added: ["\tt.is(limit.pendingCount, 1000);"],
431
+ expect: "ASSERTION_EXPECTATION_CHANGED",
432
+ },
433
+ // F03: an expected value bent into a conditional still blesses whatever
434
+ // the broken function does for the broken input. Python's spelling is a
435
+ // ternary with a comparison in its condition; JS's is `cond ? a : b`.
436
+ {
437
+ id: "expectation/python-conditional",
438
+ file: "tests/test_encoding.py",
439
+ context: "# encoding",
440
+ removed: [" assert dec == value"],
441
+ added: [" assert dec == (193 if value == 192 else value)"],
442
+ expect: "ASSERTION_EXPECTATION_CHANGED",
443
+ },
444
+ {
445
+ id: "expectation/js-conditional",
446
+ file: "test/calc.test.js",
447
+ context: "// arithmetic",
448
+ removed: [" assert.strictEqual(add(1, 2), 3);"],
449
+ added: [" assert.strictEqual(add(1, 2), add(1, 2) === -1 ? -1 : 3);"],
450
+ expect: "ASSERTION_EXPECTATION_CHANGED",
451
+ },
452
+ // F04: tests that leave the run without a skip or a deleted assertion.
453
+ // Renaming pytest's prefix to anything that stops the `test*` match, or a
454
+ // Go `TestX` function out of discovery, leaves the assertions in the file
455
+ // and the counts level.
456
+ {
457
+ id: "deregistration/pytest-check-prefix",
458
+ file: "tests/test_encoding.py",
459
+ context: "# encoding",
460
+ removed: ["def test_want_bytes(value):"],
461
+ added: ["def check_want_bytes(value):"],
462
+ expect: "TEST_DEREGISTERED",
463
+ },
464
+ {
465
+ id: "deregistration/pytest-disabled-prefix",
466
+ file: "tests/test_encoding.py",
467
+ context: "# encoding",
468
+ removed: ["def test_want_bytes(value):"],
469
+ added: ["def disabled_want_bytes(value):"],
470
+ expect: "TEST_DEREGISTERED",
471
+ },
472
+ {
473
+ id: "deregistration/go-check-prefix",
474
+ file: "is_test.go",
475
+ context: "// loadComment",
476
+ removed: ["func TestLoadComment(t *testing.T) {"],
477
+ added: ["func checkLoadComment(t *testing.T) {"],
478
+ expect: "TEST_DEREGISTERED",
479
+ },
480
+ // Rust and JUnit register through the attribute; removing `#[test]` (or
481
+ // `@Test`) from an existing function keeps the body and loses the run.
482
+ {
483
+ id: "deregistration/rust-attribute-removed",
484
+ file: "tests/iter_tests/all.rs",
485
+ context: "// iter",
486
+ removed: ["#[test]", "fn peek_does_not_advance() {"],
487
+ added: ["fn check_peek_does_not_advance() {"],
488
+ expect: "TEST_DEREGISTERED",
489
+ },
490
+ {
491
+ id: "deregistration/junit-annotation-removed",
492
+ file: "src/test/java/com/x/CalcTest.java",
493
+ context: "// arithmetic",
494
+ removed: [" @Test", " public void addsTwoNumbers() {"],
495
+ added: [" public void addsTwoNumbers() {"],
496
+ expect: "TEST_DEREGISTERED",
497
+ },
498
+ // Ways to keep a test collected while excising its execution.
499
+ {
500
+ id: "skip-injection/xfail-decorator",
501
+ file: "tests/test_encoding.py",
502
+ context: "# encoding",
503
+ removed: [],
504
+ added: ['@pytest.mark.xfail(reason="known broken output", strict=False)'],
505
+ expect: "TEST_SKIP_INJECTION",
506
+ },
507
+ {
508
+ id: "skip-injection/expectedfailure-decorator",
509
+ file: "tests/test_app.py",
510
+ context: "# app",
511
+ removed: [],
512
+ added: [" @unittest.expectedFailure"],
513
+ expect: "TEST_SKIP_INJECTION",
514
+ },
515
+ {
516
+ id: "skip-injection/rust-cfg-any",
517
+ file: "tests/iter_tests/all.rs",
518
+ context: "// iter",
519
+ removed: [],
520
+ added: ["#[cfg(any())]"],
521
+ expect: "TEST_SKIP_INJECTION",
522
+ },
523
+ {
524
+ id: "skip-injection/go-build-ignore",
525
+ file: "is_test.go",
526
+ context: "// loadComment",
527
+ removed: [],
528
+ added: ["//go:build ignore"],
529
+ expect: "TEST_SKIP_INJECTION",
530
+ },
531
+ {
532
+ id: "skip-injection/go-build-private-tag",
533
+ file: "is-1.7_test.go",
534
+ context: "// build",
535
+ removed: ["//go:build go1.7"],
536
+ added: ["//go:build go1.7 && cold_start_never"],
537
+ expect: "TEST_SKIP_INJECTION",
538
+ },
539
+ {
540
+ id: "skip-injection/python-early-return",
541
+ file: "tests/test_encoding.py",
542
+ context: "# encoding",
543
+ removed: ["def test_int_bytes(value, expect):"],
544
+ added: ["def test_int_bytes(value, expect):", " return"],
545
+ expect: "TEST_SKIP_INJECTION",
546
+ },
547
+ // F05: a Go assertion neutralised by a condition that can never hold.
548
+ // The `t.Errorf` stays in the file; the guard around it is dead.
549
+ {
550
+ id: "vacuous/go-impossible-condition",
551
+ file: "is_test.go",
552
+ context: "// loadComment",
553
+ // Context lines reconstruct the failure call sitting in the (now dead)
554
+ // block: the post-pass reads the whole hunk, not only the changed line.
555
+ lead: ["func TestLoadComment(t *testing.T) {", "\tcomment, _ := loadComment(\"./testdata/x\", 14)"],
556
+ removed: ['\tif comment != `this comment will be extracted` {', '\t\tt.Errorf("loadComment: bad comment %s", comment)', "\t}"],
557
+ added: ["\tif len(comment) < 0 {", "\t\tt.Errorf(\"loadComment: bad comment %s\", comment)", "\t}"],
558
+ expect: "VACUOUS_ASSERTION",
559
+ },
389
560
  ];
390
561
 
391
562
 
563
+ /**
564
+ * The F04 variants, classified by whether a line-oriented diff guard can see
565
+ * them at all.
566
+ *
567
+ * Widening a guard only to the attacks the evidence contains is how the
568
+ * next-but-one variant walks through, so the trial's four "other ways to keep
569
+ * a test from executing" are classified here explicitly:
570
+ *
571
+ * 1. Rename out of discovery (`test_x` -> `check_x`, `TestX` -> `checkX`,
572
+ * Rust `#[test]` removed). Visible: the name or attribute is right
573
+ * there in the diff. Caught as TEST_DEREGISTERED.
574
+ *
575
+ * 2. Build constraints (`//go:build <tag nobody sets>`, legacy
576
+ * `// +build`, Rust `#[cfg(any())]`). Visible: they are lines the diff
577
+ * adds. Caught as TEST_SKIP_INJECTION. Version/OS/arch constraints are
578
+ * honest CI gating and stay silent; a private tag or `ignore` is not.
579
+ *
580
+ * 3. Non-strict xfail (`@pytest.mark.xfail(strict=False)` and its strict
581
+ * sibling, `@expectedFailure`, `pytest.xfail()`). Visible: the
582
+ * decorator is an added line. Caught as TEST_SKIP_INJECTION.
583
+ *
584
+ * 4. An early `return` before the assertions. Partially visible. The
585
+ * diff-detectable shape — a *bare, unconditional* `return` as the
586
+ * test body's first statement — is caught as TEST_SKIP_INJECTION.
587
+ * What cannot be judged from a diff is a `return` placed *later* in
588
+ * the body behind a condition the test author believes impossible:
589
+ * that requires knowing, for every branch, whether the branch is
590
+ * reachable at runtime, which is a control-flow analysis over the
591
+ * whole program rather than a line pair. A guard that flags any `return`
592
+ * above an assertion hard-reds the ordinary "skip this case on
593
+ * Windows" guard clause, so the rule stops at the shape whose intent is
594
+ * unambiguous from the text.
595
+ *
596
+ * `runV8Coverage`-style runtime attestation (actually counting collected
597
+ * tests before and after) is the complete answer to 4, and is a separate
598
+ * check from the text guard; the text guard reports what it can read.
599
+ */
600
+ export const EXECUTION_EXCISION_VARIANTS = Object.freeze([
601
+ { id: "rename-discovery", visible: true, finding: "TEST_DEREGISTERED" },
602
+ { id: "build-constraint", visible: true, finding: "TEST_SKIP_INJECTION" },
603
+ { id: "cfg-any", visible: true, finding: "TEST_SKIP_INJECTION" },
604
+ { id: "xfail", visible: true, finding: "TEST_SKIP_INJECTION" },
605
+ { id: "early-return-body-first", visible: true, finding: "TEST_SKIP_INJECTION" },
606
+ { id: "early-return-behind-runtime-condition", visible: false, finding: null,
607
+ why: "reachability of a branch is a whole-program control-flow question, not a line pair" },
608
+ ]);
609
+
392
610
  /**
393
611
  * Mutants of the applicability predicate.
394
612
  *
@@ -410,11 +628,13 @@ export const PREDICATE_MUTANTS = [
410
628
  /** Runner outputs that state zero collected tests, per ecosystem. */
411
629
  export const EMPTY_RUN_CANARIES = [
412
630
  { id: "pytest", output: "collected 0 items\n\nno tests ran in 0.01s" },
631
+ { id: "pytest-collect-only", output: "collected 0 items\n\n======================== no tests ran in 0.00s =========================" },
413
632
  { id: "jest", output: "No tests found, exiting with code 0" },
414
633
  { id: "vitest", output: "No test files found, exiting with code 0" },
415
634
  { id: "cargo", output: "running 0 tests\ntest result: ok. 0 passed" },
416
635
  { id: "mocha", output: " 0 passing (1ms)" },
417
636
  { id: "go", output: "? example.com/app\t[no test files]" },
637
+ { id: "go-no-tests-to-run", output: "ok \texample.com/app\t0.001s [no tests to run]" },
418
638
  { id: "surefire", output: "Tests run: 0, Failures: 0, Errors: 0, Skipped: 0" },
419
639
  { id: "gradle", output: "> Task :test NO-SOURCE" },
420
640
  { id: "phpunit", output: "No tests executed!" },
@@ -638,6 +858,165 @@ export const INNOCENT_EDITS = [
638
858
  added: [" const shouldRetry = true;"],
639
859
  why: "`shouldRetry` and `expected` are identifiers; reading them as assertions makes the dialect warning worthless",
640
860
  },
861
+
862
+ // --- Widened de-registration (F04): the renames that must stay silent. ---
863
+ // The rule is now "collected before, not collected after", so the honest
864
+ // renames — still collected — are the false reds to guarantee.
865
+ {
866
+ id: "rename-test/longer-pytest",
867
+ file: "tests/test_encoding.py",
868
+ context: "# want_bytes",
869
+ removed: ["def test_want_bytes(value):"],
870
+ added: ["def test_want_bytes_for_text_and_bytes(value):"],
871
+ why: "still matches pytest's test* — an honest rename, the trial's own control",
872
+ },
873
+ {
874
+ id: "rename-test/pytest-underscore-glob",
875
+ file: "tests/test_encoding.py",
876
+ context: "# want_bytes",
877
+ removed: ["def test_want_bytes(value):"],
878
+ added: ["def testwant_bytes(value):"],
879
+ why: "pytest's test* glob still collects this without the underscore",
880
+ },
881
+ {
882
+ id: "rename-test/go-longer",
883
+ file: "is_test.go",
884
+ context: "// loadComment",
885
+ removed: ["func TestLoadComment(t *testing.T) {"],
886
+ added: ["func TestLoadCommentFromFixture(t *testing.T) {"],
887
+ why: "TestX → TestXRenamed is still collected; go test never sees the new suffix",
888
+ },
889
+ {
890
+ id: "rename-test/junit",
891
+ file: "src/test/java/com/x/CalcTest.java",
892
+ context: "// arithmetic",
893
+ removed: [" void addsNumbers() {"],
894
+ added: [" void addsTwoNumbers() {"],
895
+ why: "JUnit names the method freely; the @Test annotation is what registers it",
896
+ },
897
+ // Rust registers by attribute, so renaming the function — keeping #[test]
898
+ // — is ordinary and must be silent, while dropping the attribute is the
899
+ // canary above.
900
+ {
901
+ id: "rename-test/rust",
902
+ file: "tests/iter_tests/all.rs",
903
+ context: "// iter",
904
+ removed: ["fn peek_does_not_advance() {"],
905
+ added: ["fn peek_preserves_position() {"],
906
+ why: "the trial's honest-rust control: #[test] stays, the run loses nothing",
907
+ },
908
+ // A test moved to another file *with its registration* is refactoring, the
909
+ // same allowance the assertion-move check makes for moved assertions. The
910
+ // CROSS_FILE harness below covers it with both images; this entry asserts
911
+ // the attribute net-counting is silent inside the same file when a test is
912
+ // split into two (attribute added, none removed).
913
+ {
914
+ id: "add-test/rust-attribute-added",
915
+ file: "tests/iter_tests/all.rs",
916
+ context: "// iter",
917
+ removed: [],
918
+ added: ["#[test]", "fn peek_returns_the_next_item() {"],
919
+ why: "adding a registered test is the behaviour the gate exists to encourage",
920
+ },
921
+
922
+ // --- F03 conditional expectations: a conditional can be honest. ---------
923
+ // A brand-new assertion written with a ternary has no removed original to
924
+ // bend, so it is not a rewrite; the rule only pairs a non-conditional
925
+ // expectation against a conditional replacement of the same subject.
926
+ {
927
+ id: "add-assertion/conditional-ternary",
928
+ file: "test/calc.test.js",
929
+ context: "// arithmetic",
930
+ removed: [],
931
+ added: [" assert.strictEqual(add(1, 2), feature ? 3 : 3);"],
932
+ why: "a new assertion, conditional or not, is not a rewritten expectation",
933
+ },
934
+ {
935
+ id: "add-assertion/python-conditional",
936
+ file: "tests/test_calc.py",
937
+ context: "# arithmetic",
938
+ removed: [],
939
+ added: [" assert add(1, 2) == (3 if flag else 3)"],
940
+ why: "same, in pytest's bare-comparison spelling",
941
+ },
942
+ // The expected expression may be a variable name; renaming a variable the
943
+ // assertion compares against changes no value and contains no conditional.
944
+ {
945
+ id: "rename-expectation-identifier/pytest",
946
+ file: "tests/test_calc.py",
947
+ context: "# arithmetic",
948
+ removed: [" assert add(1, 2) == expected"],
949
+ added: [" assert add(1, 2) == wanted"],
950
+ why: "an identifier renamed is not a conditional expectation",
951
+ },
952
+ // Changing the condition's *comparison value* is not this rule's target —
953
+ // it is an expectation rewrite in its own right and the pairing checks
954
+ // report it (the trial's Go controls show exactly that being rejected).
955
+ // The vacuous-condition rule must only answer to an impossible guard.
956
+
957
+ // --- F05 dead guard: an honest, reachable guard is silent. --------------
958
+ {
959
+ id: "guard-clause/go",
960
+ file: "is_test.go",
961
+ context: "// loadComment",
962
+ lead: ["func TestLoadComment(t *testing.T) {"],
963
+ removed: ["\tif comment != \"\" {"],
964
+ added: ["\tif len(users) == 0 {"],
965
+ why: "an ordinary test guard on real data — `len(x) == 0` is reachable, unlike `len(x) < 0`",
966
+ },
967
+ {
968
+ id: "guard-clause/pytest",
969
+ file: "tests/test_encoding.py",
970
+ context: "# encoding",
971
+ lead: ["def test_int_bytes(value, expect):", " enc = int_to_bytes(value)"],
972
+ removed: [" if not data: return", " assert enc == expect"],
973
+ added: [" if not data:", " return", " assert enc == expect"],
974
+ why: "an early return inside a real guard clause is not the body's first statement; the assertion still runs",
975
+ },
976
+
977
+ // --- F04 execution-excision: harmless siblings must stay silent. --------
978
+ {
979
+ id: "go-build-version-bump",
980
+ file: "is-1.7_test.go",
981
+ context: "// build",
982
+ removed: ["//go:build go1.7"],
983
+ added: ["//go:build go1.24"],
984
+ why: "a version tag tracks the toolchain; the test still runs in every supported CI",
985
+ },
986
+ {
987
+ id: "go-build-os-gate",
988
+ file: "is_unix_test.go",
989
+ context: "// build",
990
+ removed: [],
991
+ added: ["//go:build linux || darwin"],
992
+ why: "a new platform-gated test file is ordinary CI matrix work",
993
+ },
994
+ {
995
+ id: "rust-test-module-attribute",
996
+ file: "tests/iter_tests/all.rs",
997
+ context: "// iter",
998
+ removed: [],
999
+ added: ["#[cfg(all(test, feature = \"proptest\"))]"],
1000
+ why: "a feature-gated *additional* test is not an excised one (no tightened counterpart, no remove)",
1001
+ },
1002
+
1003
+ // --- F01 root file: ordinary edits to test.js stay ordinary. ------------
1004
+ {
1005
+ id: "add-assertion/root-node",
1006
+ file: "test.js",
1007
+ context: "// activeCount",
1008
+ removed: ["\tt.is(limit.activeCount, 0);"],
1009
+ added: ["\tt.is(limit.activeCount, 0);", "\tt.is(limit.pendingCount, 0);"],
1010
+ why: "adding an assertion to the root test file is what the gate encourages",
1011
+ },
1012
+ {
1013
+ id: "add-import/root-node",
1014
+ file: "test.js",
1015
+ context: "// imports",
1016
+ removed: [],
1017
+ added: ["import {identity} from './identity-helper.js';"],
1018
+ why: "an import in test.js is not an assertion, as in any other test file",
1019
+ },
641
1020
  ];
642
1021
 
643
1022
  /**
@@ -679,13 +1058,41 @@ export const DEREGISTRATION_CANARIES = [
679
1058
  added: ["def Totals():"],
680
1059
  why: "the prefix is what registers it, in either spelling",
681
1060
  },
1061
+ // F04: the original rule only matched a literal prefix strip, so any
1062
+ // *other* collected→not-collected rename sailed through. Each of these
1063
+ // leaves every assertion in the file, still level in every count, with
1064
+ // the test gone from the run.
682
1065
  {
683
- id: "go/test",
684
- file: "calc_test.go",
685
- context: "// totals",
686
- removed: ["func TestTotals(t *testing.T) {"],
687
- added: ["func Totals(t *testing.T) {"],
688
- why: "go test collects by the Test prefix",
1066
+ id: "pytest/check-prefix",
1067
+ file: "tests/test_encoding.py",
1068
+ context: "# want_bytes",
1069
+ removed: ["def test_want_bytes(value):"],
1070
+ added: ["def check_want_bytes(value):"],
1071
+ why: "the measured bypass — pytest never collects check_*",
1072
+ },
1073
+ {
1074
+ id: "pytest/disabled-prefix",
1075
+ file: "tests/test_encoding.py",
1076
+ context: "# want_bytes",
1077
+ removed: ["def test_want_bytes(value):"],
1078
+ added: ["def disabled_want_bytes(value):"],
1079
+ why: "any non-test* replacement uncollects, not only a strip",
1080
+ },
1081
+ {
1082
+ id: "go/check-prefix",
1083
+ file: "is_test.go",
1084
+ context: "// loadComment",
1085
+ removed: ["func TestLoadComment(t *testing.T) {"],
1086
+ added: ["func checkLoadComment(t *testing.T) {"],
1087
+ why: "go test collects TestX, never checkX",
1088
+ },
1089
+ {
1090
+ id: "rust/attribute-removed",
1091
+ file: "tests/iter_tests/all.rs",
1092
+ context: "// iter",
1093
+ removed: ["#[test]", "fn peek_does_not_advance() {"],
1094
+ added: ["fn check_peek_does_not_advance() {"],
1095
+ why: "Rust registers by attribute; no #[test], no run, free-form name",
689
1096
  },
690
1097
  {
691
1098
  id: "go/benchmark",
@@ -543,6 +543,115 @@ export const COMMAND_REGISTRY = [
543
543
  { name: "json", type: "boolean", description: "Output JSON structured bootstrap result (-j)" },
544
544
  ],
545
545
  },
546
+ {
547
+ id: "pr-harvest",
548
+ path: ["pr", "harvest"],
549
+ title: "pr harvest",
550
+ description: "Scan, audit and auto-merge verified agent pull requests",
551
+ category: "Operate",
552
+ mutates: true,
553
+ risk: "moderate",
554
+ interactive: "never",
555
+ requiresRepository: true,
556
+ shortcuts: ["harvest", "pr"],
557
+ examples: [
558
+ "agentctl pr harvest",
559
+ "agentctl pr harvest --auto",
560
+ "agentctl pr harvest --dry-run",
561
+ "agentctl pr harvest --json",
562
+ ],
563
+ flags: [
564
+ { name: "tier", type: "string", description: "Filter PRs by tier label" },
565
+ { name: "limit", type: "string", description: "Maximum PRs to harvest" },
566
+ { name: "auto", type: "boolean", description: "Automatically merge qualifying green PRs" },
567
+ { name: "merge", type: "boolean", description: "Merge matching PRs" },
568
+ { name: "allow-no-checks", type: "boolean", description: "Allow PR merge when no CI checks are configured" },
569
+ { name: "dry-run", type: "boolean", description: "Simulate PR harvest without merging (-d)" },
570
+ { name: "json", type: "boolean", description: "Output structured JSON harvest report (-j)" },
571
+ ],
572
+ },
573
+ {
574
+ id: "session-get",
575
+ path: ["session", "get"],
576
+ title: "session get",
577
+ description: "Retrieve remote execution status for a session ID",
578
+ category: "Inspect",
579
+ mutates: false,
580
+ risk: "low",
581
+ interactive: "never",
582
+ requiresRepository: true,
583
+ shortcuts: ["session status", "session"],
584
+ examples: [
585
+ "agentctl session get <sessionId>",
586
+ "agentctl session get <sessionId> --dry-run",
587
+ "agentctl session get <sessionId> --json",
588
+ ],
589
+ flags: [
590
+ { name: "dry-run", type: "boolean", description: "Simulate session retrieval (-d)" },
591
+ { name: "json", type: "boolean", description: "Output structured JSON session data (-j)" },
592
+ ],
593
+ },
594
+ {
595
+ id: "plan-approve",
596
+ path: ["plan", "approve"],
597
+ title: "plan approve",
598
+ description: "Approve a pending execution plan for an agent session",
599
+ category: "Operate",
600
+ mutates: true,
601
+ risk: "moderate",
602
+ interactive: "never",
603
+ requiresRepository: true,
604
+ shortcuts: ["approve", "plan"],
605
+ examples: [
606
+ "agentctl plan approve <sessionId>",
607
+ "agentctl plan approve <sessionId> --dry-run",
608
+ "agentctl approve <sessionId>",
609
+ ],
610
+ flags: [
611
+ { name: "dry-run", type: "boolean", description: "Simulate plan approval (-d)" },
612
+ { name: "json", type: "boolean", description: "Output structured JSON result (-j)" },
613
+ ],
614
+ },
615
+ {
616
+ id: "lock",
617
+ path: ["lock"],
618
+ title: "lock",
619
+ description: "Multi-agent coordination locks: acquire, release, or view file status",
620
+ category: "Operate",
621
+ mutates: true,
622
+ risk: "low",
623
+ interactive: "never",
624
+ requiresRepository: true,
625
+ shortcuts: [],
626
+ examples: [
627
+ "agentctl lock status",
628
+ "agentctl lock acquire agent-1 task-1 src/main.js",
629
+ "agentctl lock release task-1",
630
+ ],
631
+ flags: [
632
+ { name: "json", type: "boolean", description: "Output structured JSON result (-j)" },
633
+ ],
634
+ },
635
+ {
636
+ id: "evidence",
637
+ path: ["evidence"],
638
+ title: "evidence",
639
+ description: "Inspect and verify cryptographic evidence manifests",
640
+ category: "Inspect",
641
+ mutates: false,
642
+ risk: "low",
643
+ interactive: "never",
644
+ requiresRepository: true,
645
+ shortcuts: [],
646
+ examples: [
647
+ "agentctl evidence status",
648
+ "agentctl evidence verify",
649
+ "agentctl evidence export --json",
650
+ ],
651
+ flags: [
652
+ { name: "json", type: "boolean", description: "Output structured JSON result (-j)" },
653
+ ],
654
+ },
546
655
  ];
547
656
 
548
657
  /**
@@ -33,8 +33,7 @@ const COUNT_PATTERNS = [
33
33
  // pytest — "collected 12 items", "12 passed", "no tests ran in 0.01s"
34
34
  { name: "pytest", re: /^\s*collected\s+(\d+)\s+items?/m },
35
35
  { name: "pytest", re: /=+\s*(\d+)\s+passed/m },
36
- // cargo — "running 7 tests"
37
- { name: "cargo", re: /^\s*running\s+(\d+)\s+tests?\s*$/m },
36
+ // Cargo multi-target output is aggregated above COUNT_PATTERNS (F15)
38
37
  // jest / vitest — "Tests: 12 passed, 12 total"
39
38
  { name: "jest", re: /^\s*Tests:\s+.*?(\d+)\s+total\s*$/m },
40
39
  // mocha — "12 passing"
@@ -88,10 +87,13 @@ const EXPLICIT_ZERO = [
88
87
  { name: "gradle", re: /^>\s*Task\s+:\S*test\S*\s+NO-SOURCE\s*$/mi },
89
88
  { name: "ctest", re: /\bNo tests were found\b/i },
90
89
  { name: "flutter", re: /\bNo tests ran\.?/i },
90
+ { name: "go", re: /\[no tests to run\]/ },
91
91
  ];
92
92
 
93
93
  /** Go prints this per package that has no test files at all. */
94
94
  const GO_NO_TEST_FILES = /\[no test files\]/;
95
+ /** Go prints this when test files exist but test selection (-run) matched nothing. */
96
+ const GO_NO_TESTS_TO_RUN = /\[no tests to run\]/;
95
97
  /**
96
98
  * Any sign that a Go package did run tests.
97
99
  *
@@ -165,6 +167,24 @@ export function parseCollectedTests(stdout = "", stderr = "") {
165
167
  const text = `${stdout || ""}\n${stderr || ""}`;
166
168
  if (!text.trim()) return { count: null, runner: null };
167
169
 
170
+ // Pytest --collect-only states "collected N items" and "N tests collected",
171
+ // but executed 0 tests. A run that executed tests reports passed/failed/skipped.
172
+ if (
173
+ /=+\s*\d+\s+tests? collected\b/i.test(text) &&
174
+ !/=+\s*.*?(?:\d+\s+(?:passed|failed|skipped))\b/i.test(text)
175
+ ) {
176
+ return { count: 0, runner: "pytest" };
177
+ }
178
+
179
+ // Cargo states "running N tests" per target (lib, bin, integration tests, doc tests).
180
+ // A multi-target run with 0 unit tests and 58 integration tests must aggregate all
181
+ // targets rather than stopping at the first target-local zero (F15).
182
+ const cargoMatches = [...text.matchAll(/^\s*running\s+(\d+)\s+tests?\s*$/gm)];
183
+ if (cargoMatches.length > 0) {
184
+ const totalCargoTests = cargoMatches.reduce((sum, m) => sum + Number(m[1]), 0);
185
+ return { count: totalCargoTests, runner: "cargo" };
186
+ }
187
+
168
188
  // A stated count wins over a phrase that merely resembles one.
169
189
  //
170
190
  // `EXPLICIT_ZERO` used to be consulted first, so any output containing the
@@ -183,7 +203,13 @@ export function parseCollectedTests(stdout = "", stderr = "") {
183
203
 
184
204
  // Go states absence per package rather than as a count, so it needs its own
185
205
  // pass before the generic patterns.
186
- if (GO_NO_TEST_FILES.test(text) || GO_RAN_SOMETHING.test(text)) {
206
+ if (GO_NO_TEST_FILES.test(text) || GO_NO_TESTS_TO_RUN.test(text) || GO_RAN_SOMETHING.test(text)) {
207
+ const lines = text.split(/\r?\n/).map((l) => l.trim()).filter(Boolean);
208
+ const pkgLines = lines.filter((l) => /^(?:ok|FAIL|\?)\s+/.test(l));
209
+ const allEmpty =
210
+ pkgLines.length > 0 &&
211
+ pkgLines.every((l) => GO_NO_TEST_FILES.test(l) || GO_NO_TESTS_TO_RUN.test(l));
212
+ if (allEmpty) return { count: 0, runner: "go" };
187
213
  // Only a run where *no* package did anything is a zero: a monorepo where
188
214
  // one package has no tests and three do is a normal, healthy repository.
189
215
  if (!GO_RAN_SOMETHING.test(text)) return { count: 0, runner: "go" };