jules-orchestrator-kit 0.69.0 → 0.71.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/engine.mjs CHANGED
@@ -459,7 +459,28 @@ export async function gate(opts = {}) {
459
459
  let assertMetrics = {};
460
460
 
461
461
  if (isAssertion) {
462
- const assertRes = runAssertion(stage, root);
462
+ // The loosening flags have to reach here too.
463
+ //
464
+ // `scanDiff` above is given `allowTestChanges` and honours it, while
465
+ // the `assert:test-integrity` stage received only its own stage object
466
+ // and re-ran the same guard with none of them. So an override was
467
+ // accepted by one phase and ignored by the next: `--allow-test-change
468
+ // deregistration` turned the SECRETS phase green and then failed the
469
+ // run at the anti-tamper stage, with a flag hint the operator had
470
+ // already followed. One rule, two places, and the second kept the old
471
+ // answer — for the eighth time in this project's history, which is why
472
+ // the regression test asserts the verdict a caller receives rather
473
+ // than the behaviour of either site.
474
+ const assertRes = runAssertion(
475
+ {
476
+ ...stage,
477
+ allowTestModifications: opts.allowTestModifications === true,
478
+ allowTestChanges: opts.allowTestChanges,
479
+ tamperGuard: trustedVerify.tamperGuard,
480
+ allowUnreadableTests: opts.allowUnreadableTests === true,
481
+ },
482
+ root
483
+ );
463
484
  durationMs = assertRes.metrics?.durationMs ?? (Date.now() - startTime);
464
485
  stdoutRedacted = assertRes.stdout || "";
465
486
  stderrRedacted = assertRes.stderr || "";
@@ -894,19 +915,49 @@ export async function repair(failure, opts = {}) {
894
915
  break;
895
916
  }
896
917
 
897
- attempts.push({ n, session, ok: true });
898
- appendTelemetry(root, "ooda_repair_attempt", { attempt: n, ok: true });
899
-
900
918
  // Poll async provider for terminal session state before executing re-verification gates
919
+ let pollVerdict = null;
901
920
  if (provider && session) {
902
- await pollSessionState(provider, session, {
921
+ pollVerdict = await pollSessionState(provider, session, {
903
922
  root,
904
923
  dryRun: opts.dryRun,
905
924
  pollIntervalMs: opts.pollIntervalMs,
906
925
  maxPollAttempts: opts.maxPollAttempts,
907
926
  });
927
+
928
+ // The gate below is the authority on whether the change works, so it runs
929
+ // either way. But a non-terminal session means it is about to judge a
930
+ // tree the agent may not have finished writing, and saying nothing here
931
+ // turns that into a confusing cascade of repair attempts against a
932
+ // half-applied patch.
933
+ if (pollVerdict && pollVerdict.terminal !== true) {
934
+ const why = pollVerdict.blockedOn
935
+ ? `waiting on an actor (${pollVerdict.blockedOn})`
936
+ : pollVerdict.unreachable
937
+ ? "the provider stopped answering"
938
+ : `still ${pollVerdict.status} when the poll budget ran out`;
939
+ console.warn(
940
+ `[SESSION_NOT_TERMINAL] Session ${session.id} is ${why}. Re-verification is running against a tree the agent may not have finished writing.`
941
+ );
942
+ appendTelemetry(root, "session_not_terminal", {
943
+ attempt: n,
944
+ sessionId: session.id,
945
+ status: pollVerdict.status,
946
+ blockedOn: pollVerdict.blockedOn || null,
947
+ timedOut: Boolean(pollVerdict.timedOut),
948
+ unreachable: Boolean(pollVerdict.unreachable),
949
+ });
950
+ }
908
951
  }
909
952
 
953
+ attempts.push({
954
+ n,
955
+ session,
956
+ ok: true,
957
+ poll: pollVerdict ? { status: pollVerdict.status, terminal: pollVerdict.terminal } : null,
958
+ });
959
+ appendTelemetry(root, "ooda_repair_attempt", { attempt: n, ok: true });
960
+
910
961
  // Re-verify after repair attempt
911
962
  const gateRes = await gate({ root, config, fix: false, progressBus, progressToken });
912
963
  if (gateRes.ok) {
@@ -1000,10 +1051,38 @@ export async function checkTaskPremise(task = {}, opts = {}) {
1000
1051
  }
1001
1052
 
1002
1053
  /**
1003
- * Polls the provider until the session terminates in COMPLETED, FAILED, or reaches timeout.
1054
+ * Session states the API documents as final. The full `SessionState` enum is
1055
+ * transcribed in `docs/jules-quality-plan.md`.
1056
+ */
1057
+ export const TERMINAL_SESSION_STATES = new Set(["COMPLETED", "FAILED"]);
1058
+
1059
+ /**
1060
+ * Session states that cannot advance without an actor — a human approving a
1061
+ * plan, a human answering a question, or whoever paused it resuming it.
1062
+ *
1063
+ * Polling these is not waiting, it is spending the budget: the state cannot
1064
+ * change on its own no matter how long the loop runs.
1065
+ */
1066
+ export const BLOCKING_SESSION_STATES = new Set(["AWAITING_PLAN_APPROVAL", "AWAITING_USER_FEEDBACK", "PAUSED"]);
1067
+
1068
+ /**
1069
+ * Polls the provider until the session reaches a terminal state, blocks on an
1070
+ * actor, or the poll budget runs out.
1071
+ *
1072
+ * The verdict is explicit about which of those three happened, because the
1073
+ * caller runs the verification gate on the assumption that the agent has
1074
+ * finished writing. A previous version returned `COMPLETED` for every
1075
+ * non-terminal exit — a session sitting in `AWAITING_USER_FEEDBACK` and one
1076
+ * still `IN_PROGRESS` when the budget expired were both reported as success,
1077
+ * which is the same shape this project exists to refuse.
1078
+ *
1079
+ * @returns {Promise<object>} Always carries `status` and `terminal`. Non-terminal
1080
+ * exits additionally carry one of `blockedOn`, `timedOut` or `unreachable`.
1081
+ * `status` is never synthesised: it is the last state the provider reported,
1082
+ * or `UNKNOWN` when it never answered.
1004
1083
  */
1005
1084
  export async function pollSessionState(provider, session, opts = {}) {
1006
- if (!session || !session.id) return { status: "COMPLETED" };
1085
+ if (!session || !session.id) return { status: "UNKNOWN", terminal: false, unpolled: true };
1007
1086
  const initialStatus = String(session.status || session.state || "").toUpperCase();
1008
1087
  if (
1009
1088
  initialStatus === "COMPLETED" ||
@@ -1013,13 +1092,20 @@ export async function pollSessionState(provider, session, opts = {}) {
1013
1092
  session.id.startsWith("mock-") ||
1014
1093
  session.id.startsWith("dry-run-")
1015
1094
  ) {
1016
- return { status: initialStatus || "COMPLETED" };
1095
+ // A dry run has no session to watch, so it reports the outcome a real
1096
+ // dispatch would have had to earn. Flagged as simulated so a caller can
1097
+ // tell the two apart; a terminal state already reported by the provider is
1098
+ // a fact, not a simulation, and is returned as itself.
1099
+ const status = initialStatus || "COMPLETED";
1100
+ return { status, terminal: TERMINAL_SESSION_STATES.has(status), simulated: true, polls: 0 };
1017
1101
  }
1018
1102
 
1019
1103
  const maxAttempts = opts.maxPollAttempts || 30;
1020
1104
  const pollIntervalMs = opts.pollIntervalMs || 1000;
1021
1105
  const timeoutMs = opts.pollTimeoutMs || 300000;
1022
1106
  const startTime = Date.now();
1107
+ let lastStatus = "";
1108
+ let polls = 0;
1023
1109
 
1024
1110
  for (let attempt = 0; attempt < maxAttempts; attempt++) {
1025
1111
  if (Date.now() - startTime > timeoutMs) break;
@@ -1035,30 +1121,59 @@ export async function pollSessionState(provider, session, opts = {}) {
1035
1121
  } catch (_) {}
1036
1122
  }
1037
1123
 
1038
- if (currentSession) {
1039
- const status = String(currentSession.status || currentSession.state || "").toUpperCase();
1040
- if (
1041
- (status === "AWAITING_PLAN_APPROVAL" || status === "PENDING_APPROVAL") &&
1042
- (opts.autoApprovePlan || opts.autoApprove || session.autoApprovePlan)
1043
- ) {
1044
- if (provider && typeof provider.approvePlan === "function") {
1045
- try {
1046
- await provider.approvePlan(session.id, opts);
1047
- } catch (_) {}
1124
+ if (!currentSession) {
1125
+ return {
1126
+ ...session,
1127
+ status: lastStatus || "UNKNOWN",
1128
+ terminal: false,
1129
+ unreachable: true,
1130
+ polls,
1131
+ };
1132
+ }
1133
+
1134
+ polls += 1;
1135
+ const status = String(currentSession.status || currentSession.state || "").toUpperCase();
1136
+ if (status) lastStatus = status;
1137
+
1138
+ if (TERMINAL_SESSION_STATES.has(status)) {
1139
+ return { ...currentSession, status, terminal: true, polls };
1140
+ }
1141
+
1142
+ if (BLOCKING_SESSION_STATES.has(status)) {
1143
+ // A pending plan approval is the one blocking state this loop is allowed
1144
+ // to resolve itself, and only when the caller said so.
1145
+ const isPlanApproval = status === "AWAITING_PLAN_APPROVAL";
1146
+ const wantsAutoApprove = Boolean(opts.autoApprovePlan || opts.autoApprove || session.autoApprovePlan);
1147
+ if (isPlanApproval && wantsAutoApprove && provider && typeof provider.approvePlan === "function") {
1148
+ try {
1149
+ await provider.approvePlan(session.id, opts);
1150
+ await new Promise((resolve) => setTimeout(resolve, pollIntervalMs));
1151
+ continue;
1152
+ } catch (err) {
1153
+ return {
1154
+ ...currentSession,
1155
+ status,
1156
+ terminal: false,
1157
+ blockedOn: status,
1158
+ approvePlanError: err && err.message ? err.message : String(err),
1159
+ polls,
1160
+ };
1048
1161
  }
1049
1162
  }
1050
1163
 
1051
- if (status === "COMPLETED" || status === "FAILED") {
1052
- return { ...currentSession, status };
1053
- }
1054
- } else {
1055
- break;
1164
+ return { ...currentSession, status, terminal: false, blockedOn: status, polls };
1056
1165
  }
1057
1166
 
1058
1167
  await new Promise((resolve) => setTimeout(resolve, pollIntervalMs));
1059
1168
  }
1060
1169
 
1061
- return { ...session, status: String(session.status || "COMPLETED").toUpperCase() };
1170
+ return {
1171
+ ...session,
1172
+ status: lastStatus || "UNKNOWN",
1173
+ terminal: false,
1174
+ timedOut: true,
1175
+ polls,
1176
+ };
1062
1177
  }
1063
1178
 
1064
1179
  function buildRepairPrompt(failure, attempt, _config, extraPromptDirective = null) {
@@ -454,6 +454,57 @@ export const SCOPE_CANARIES = [
454
454
  * from never showed it.
455
455
  */
456
456
  export const INNOCENT_EDITS = [
457
+ // Everything below this comment was a CRITICAL rejection until v0.71.0.
458
+ //
459
+ // Not through any rule that examined them: `UNREADABLE` was reached whenever
460
+ // a test file had changed lines and none of them parsed as an assertion, so
461
+ // the verdict was the same for a repository speaking an unsupported dialect
462
+ // and for one where the edit simply was not an assertion. Adding an import
463
+ // was a CRITICAL block. The finding carried no file, no line and no sample,
464
+ // because there was nothing to name — and it advised a pytest repository
465
+ // that its assertion library might be unsupported, from a list naming pytest.
466
+ //
467
+ // These are ordinary work. Every one of them has to be silent.
468
+ {
469
+ id: "add-import/pytest",
470
+ file: "testing/test_iniconfig.py",
471
+ context: "# fixtures",
472
+ removed: [],
473
+ added: ["import os"],
474
+ why: "adding an import to a test file is not an assertion and not tampering",
475
+ },
476
+ {
477
+ id: "rename-test-multiline/pytest",
478
+ file: "testing/test_iniconfig.py",
479
+ context: "# parsing",
480
+ removed: ["def test_parse_strips_inline_comments() -> None:"],
481
+ added: ["def test_parse_strips_inline_comments_from_continuations() -> None:"],
482
+ why: "still collected by pytest, so nothing left the run",
483
+ },
484
+ {
485
+ id: "docstring/pytest",
486
+ file: "testing/test_iniconfig.py",
487
+ context: "# parsing",
488
+ removed: [' """Check comments."""'],
489
+ added: [' """Check that inline comments are stripped."""'],
490
+ why: "prose about the test, not a value it asserts",
491
+ },
492
+ {
493
+ id: "type-annotation/pytest",
494
+ file: "testing/test_iniconfig.py",
495
+ context: "# fixtures",
496
+ removed: ["def make_config(data):"],
497
+ added: ["def make_config(data: str) -> IniConfig:"],
498
+ why: "a helper's signature, typed; no claim changed",
499
+ },
500
+ {
501
+ id: "add-fixture/go",
502
+ file: "calc_test.go",
503
+ context: "// helpers",
504
+ removed: [],
505
+ added: ["var cases = []int{1, 2, 3}"],
506
+ why: "test data added, nothing asserted or unasserted",
507
+ },
457
508
  // Renaming a test is one of the most ordinary edits there is. On the
458
509
  // one-line form the name blanked to the same shape as its replacement, the
459
510
  // two paired, and the pair was reported as a rewritten expectation.
@@ -598,6 +649,129 @@ export const INNOCENT_EDITS = [
598
649
  * will always end somewhere; what must never happen again is that the edge
599
650
  * is silent.
600
651
  */
652
+ /**
653
+ * Renames that delete a test from the run without deleting a line of it.
654
+ *
655
+ * pytest collects `test_*`, Go collects `Test*`. For those runners the name is
656
+ * the registration, so `def test_totals` → `def totals` removes the test as
657
+ * completely as deleting the file, and every count in the tamper guard stays
658
+ * level: nothing removed, nothing weakened, nothing rewritten.
659
+ *
660
+ * These were caught only as a side effect of the blanket that rejected every
661
+ * unrecognised edit to a test file — which also rejected adding an import, and
662
+ * whose printed remedy switched the real checks off. They are their own
663
+ * finding now, so narrowing that blanket costs nothing.
664
+ */
665
+ export const DEREGISTRATION_CANARIES = [
666
+ {
667
+ id: "pytest/underscore",
668
+ file: "tests/test_totals.py",
669
+ context: "# totals",
670
+ removed: ["def test_totals_round_to_cents():"],
671
+ added: ["def totals_round_to_cents():"],
672
+ why: "pytest collects by the test_ prefix, so this test no longer runs",
673
+ },
674
+ {
675
+ id: "pytest/camel",
676
+ file: "tests/test_totals.py",
677
+ context: "# totals",
678
+ removed: ["def testTotals():"],
679
+ added: ["def Totals():"],
680
+ why: "the prefix is what registers it, in either spelling",
681
+ },
682
+ {
683
+ id: "go/test",
684
+ file: "calc_test.go",
685
+ context: "// totals",
686
+ removed: ["func TestTotals(t *testing.T) {"],
687
+ added: ["func Totals(t *testing.T) {"],
688
+ why: "go test collects by the Test prefix",
689
+ },
690
+ {
691
+ id: "go/benchmark",
692
+ file: "calc_test.go",
693
+ context: "// totals",
694
+ removed: ["func BenchmarkTotals(b *testing.B) {"],
695
+ added: ["func Totals(b *testing.B) {"],
696
+ why: "the same rule for the other collected prefixes",
697
+ },
698
+ ];
699
+
700
+ /**
701
+ * A verification command that exits 0 having written nothing at all.
702
+ *
703
+ * The one absence that is evidence. The floor is otherwise deliberately
704
+ * one-sided — an unrecognised runner states no count and passes, because
705
+ * hard-redding every runner not on the list would be worse than the hole it
706
+ * closes — but an unrecognised runner still prints something. Zero bytes on
707
+ * both streams is a command that ran nothing, and `pnpm -r test` on a
708
+ * workspace whose packages declare no test script is exactly that: it was
709
+ * indistinguishable from a full suite by every signal the gate had.
710
+ */
711
+ export const SILENT_RUN_CANARIES = [
712
+ { id: "pnpm -r test", command: "pnpm -r test", stdout: "", stderr: "" },
713
+ { id: "npm test", command: "npm test", stdout: "", stderr: "" },
714
+ { id: "yarn workspaces test", command: "yarn workspaces foreach run test", stdout: "", stderr: "" },
715
+ { id: "pytest", command: "python3 -m pytest", stdout: "", stderr: "" },
716
+ { id: "go test", command: "go test ./...", stdout: "", stderr: "" },
717
+ { id: "whitespace is not output", command: "npm test", stdout: " \n", stderr: "\n" },
718
+ ];
719
+
720
+ /**
721
+ * Honest static gates that print nothing, and must keep passing.
722
+ *
723
+ * The counterweight, and the reason the rule above reads the command as well
724
+ * as the output. `tsc --noEmit`, `node --check`, `go vet` and `compileall`
725
+ * all exit 0 in silence when they succeed — that is what success looks like
726
+ * for a checker — and two of them are commands this kit writes itself for a
727
+ * repository that has no suite yet. A rule keyed on silence alone hard-redded
728
+ * every one of them, which is exactly the first-run rejection of correct code
729
+ * that the collection floor is otherwise so careful to avoid.
730
+ */
731
+ export const SILENT_STATIC_GATES = [
732
+ { id: "tsc", command: "tsc --noEmit" },
733
+ { id: "node --check", command: "node --check index.js" },
734
+ { id: "go vet", command: "go vet ./..." },
735
+ { id: "compileall", command: "python3 -m compileall -q ." },
736
+ { id: "generated parse gate", command: "node --check src/index.mjs" },
737
+ ];
738
+
739
+ /**
740
+ * Values that must survive a trip through the emitter and back.
741
+ *
742
+ * `yamlScalar` writes the manifests and `parseYaml` reads them, and a pair
743
+ * like that disagreeing about one character is this project's recurring
744
+ * defect with both halves in a single module. The parser opened a comment at
745
+ * the first `#` on a line, quoted or not, so `verify.test` containing a hash
746
+ * was silently truncated on the way in and the gate ran a command the user
747
+ * never wrote — reporting on it as if it were theirs.
748
+ *
749
+ * The corpus is the hard cases on purpose: hashes, colons, leading stars,
750
+ * apostrophes, the empty string, and the words YAML would otherwise read as
751
+ * booleans.
752
+ */
753
+ export const YAML_ROUNDTRIP_CASES = [
754
+ "pnpm -r test",
755
+ "PYTHONPATH=src python3 -m pytest",
756
+ 'pytest -k "not #slow"',
757
+ "has #hash",
758
+ "# leading hash",
759
+ "a: b",
760
+ "**/*.pem",
761
+ "**/.env.*",
762
+ ".github/**",
763
+ "",
764
+ "true",
765
+ "yes",
766
+ "1.5",
767
+ "it's fine",
768
+ "it's #1",
769
+ "O'Brien",
770
+ "agent/",
771
+ " leading space",
772
+ "trailing space ",
773
+ ];
774
+
601
775
  export const UNREADABLE_DIALECTS = [
602
776
  {
603
777
  id: "hspec",
@@ -111,6 +111,56 @@ const GO_RAN_SOMETHING = /^(?:(?:ok|FAIL)\s+(?!\d+\s)\S+|---\s+(?:PASS|FAIL|SKIP
111
111
  * `count` is null when no recognised runner stated one — which is not a
112
112
  * finding, only an absence of evidence.
113
113
  */
114
+ /**
115
+ * Did the command write nothing at all, on either stream?
116
+ *
117
+ * This is the one absence that is evidence rather than the lack of it. The
118
+ * one-sided floor exists because an unrecognised runner states no count, and
119
+ * hard-redding every runner not on the list would be worse than the hole it
120
+ * closes. But an unrecognised runner still *prints*: dots, a summary line,
121
+ * a package name, something. Zero bytes on both streams is not a dialect the
122
+ * list has yet to learn — it is a command that ran nothing.
123
+ *
124
+ * `pnpm -r test` on a workspace whose packages declare no test script is the
125
+ * shape that made this necessary: it exits 0, writes nothing anywhere, and
126
+ * was indistinguishable from a full suite by every signal the gate had.
127
+ *
128
+ * Resolved here and consumed in two places — the gate's floor and `init`'s
129
+ * oracle probe — because writing the rule once in each is how this project
130
+ * has repeatedly ended up with two answers to one question.
131
+ */
132
+ export function producedNoOutput(stdout = "", stderr = "") {
133
+ return `${stdout || ""}${stderr || ""}`.trim() === "";
134
+ }
135
+
136
+ /**
137
+ * Does this command claim to run a test suite?
138
+ *
139
+ * Silence alone cannot carry the verdict, and the first version of this rule
140
+ * assumed it could. `node --check index.js`, `tsc --noEmit`, `go vet ./...`
141
+ * and `python3 -m compileall -q .` all exit 0 having printed nothing — and
142
+ * they are honest static gates, two of which this kit writes itself for
143
+ * repositories that have no suite yet. Failing on silence alone hard-redded
144
+ * every one of them, which is the same first-run rejection of correct code
145
+ * that the whole collection floor is careful to avoid.
146
+ *
147
+ * Nothing in the *output* separates `pnpm -r test` from `tsc --noEmit`; both
148
+ * are empty. The difference is in what the command says it is. So this reads
149
+ * the command, the same way `isPlaceholderTestScript` does: a command that is
150
+ * recognisably a suite invocation and printed nothing ran no suite, while a
151
+ * static checker that printed nothing did exactly what it promised.
152
+ *
153
+ * One-sided in the safe direction, like everything else here. An unrecognised
154
+ * command is not treated as a suite, so an unusual runner keeps its advisory
155
+ * pass rather than becoming a hard red.
156
+ */
157
+ const TEST_SUITE_COMMAND =
158
+ /(?:^|\s|\/)(?:pytest|jest|vitest|mocha|ava|karma|jasmine|nyc|c8|tap|tape|rspec|minitest|phpunit|behave|nose2?|ginkgo|gotestsum|nextest)\b|\b(?:go|cargo|swift|dart|flutter|deno|bun|dotnet|mix|lein|sbt|gradlew?|mvn)\s+test\b|\bunittest\b|-m\s+(?:pytest|unittest)\b|\bnode\s+--test\b|(?:^|&&|;|\|)\s*(?:npm|pnpm|yarn|bun|npx)\b[^&;|]*?\btest\b/;
159
+
160
+ export function looksLikeTestSuiteCommand(cmd) {
161
+ return typeof cmd === "string" && TEST_SUITE_COMMAND.test(cmd);
162
+ }
163
+
114
164
  export function parseCollectedTests(stdout = "", stderr = "") {
115
165
  const text = `${stdout || ""}\n${stderr || ""}`;
116
166
  if (!text.trim()) return { count: null, runner: null };
@@ -168,6 +218,30 @@ export function checkCollectionFloor(testResult, opts = {}) {
168
218
  return { ok: true, count: null, runner: null, reason: null };
169
219
  }
170
220
 
221
+ // A command that says it runs a suite, and printed nothing, ran no suite.
222
+ //
223
+ // Both halves are required. Silence alone would hard-red `tsc --noEmit` and
224
+ // `python3 -m compileall`, which are honest static gates this kit generates
225
+ // itself; the command shape alone would say nothing, because a real suite
226
+ // prints. Together they are decidable, and they are exactly `pnpm -r test`
227
+ // on a workspace whose packages declare no test script.
228
+ if (looksLikeTestSuiteCommand(testResult.command) && producedNoOutput(testResult.stdout, testResult.stderr)) {
229
+ return {
230
+ ok: false,
231
+ count: 0,
232
+ runner: null,
233
+ silent: true,
234
+ reason:
235
+ `The verification command ${testResult.command ? `${JSON.stringify(testResult.command)} ` : ""}` +
236
+ `exited 0 and wrote nothing at all — no test names, no summary, no count. ` +
237
+ `Every test runner prints something, so this command ran no suite, and approving this change ` +
238
+ `would certify nothing. A workspace command such as \`pnpm -r test\` does this when no package ` +
239
+ `declares a test script. Point verify.test at the suite that covers this repository ` +
240
+ `(often the root script rather than the recursive one), or — if this repository intentionally ` +
241
+ `uses only the scope and secret phases — set verify.required: false, which says so on the record.`,
242
+ };
243
+ }
244
+
171
245
  const { count, runner } = parseCollectedTests(testResult.stdout, testResult.stderr);
172
246
 
173
247
  // Deliberately one-sided: only a *stated* zero fails, because failing on
package/src/security.mjs CHANGED
@@ -1875,6 +1875,51 @@ function splitTrailingMessage(clean) {
1875
1875
  return { head: clean.slice(0, lastComma), msg: tail };
1876
1876
  }
1877
1877
 
1878
+ /**
1879
+ * Test declarations that a runner finds by the *name* of the function.
1880
+ *
1881
+ * pytest collects `def test_*`, Go collects `func Test*`, and unittest and
1882
+ * Minitest collect `def test_*` off the case class. For those runners the
1883
+ * name is not prose — it is the registration. Renaming `test_totals` to
1884
+ * `totals` deletes the test from the run as completely as removing the file,
1885
+ * and the diff shows a rename.
1886
+ *
1887
+ * Only these name-driven runners are listed. `it("...")`, `#[test]` and
1888
+ * `@Test` register by call, attribute or annotation, so renaming what they
1889
+ * declare removes nothing, and the ordinary rename rules already cover them.
1890
+ */
1891
+ const NAME_REGISTERED_DECLS = [
1892
+ { lang: "python", re: /^\s*(?:async\s+)?def\s+([A-Za-z_]\w*)\s*\(/, discovered: /^test/i },
1893
+ { lang: "go", re: /^\s*func\s+([A-Za-z_]\w*)\s*\(/, discovered: /^(?:Test|Benchmark|Fuzz|Example)/ },
1894
+ ];
1895
+
1896
+ /**
1897
+ * The declared name on this line, and whether the runner would collect it.
1898
+ *
1899
+ * @returns {{ name: string, collected: boolean }|null}
1900
+ */
1901
+ function declaredTestName(text) {
1902
+ for (const rule of NAME_REGISTERED_DECLS) {
1903
+ const m = rule.re.exec(text);
1904
+ if (m) return { name: m[1], collected: rule.discovered.test(m[1]) };
1905
+ }
1906
+ return null;
1907
+ }
1908
+
1909
+ /**
1910
+ * Is `after` the same declaration as `before` with its discovery prefix gone?
1911
+ *
1912
+ * Exact on the remainder, deliberately. `test_totals` → `totals` is a
1913
+ * de-registration; `test_totals` → `test_totals_rounded` is a rename and must
1914
+ * stay silent, which is the false red this check exists alongside rather than
1915
+ * instead of.
1916
+ */
1917
+ function isDeregistration(before, after) {
1918
+ if (!before.collected || after.collected) return false;
1919
+ const stripped = before.name.replace(/^test[_-]?/i, "").replace(/^(?:Test|Benchmark|Fuzz|Example)/, "");
1920
+ return stripped.length > 0 && stripped === after.name;
1921
+ }
1922
+
1878
1923
  // A test declaration whose first argument is the test's name. The name is
1879
1924
  // prose about the test, not a value the test asserts — `test("adds", ...)`
1880
1925
  // renamed to `test("adds positives", ...)` is the rename the diff says it is.
@@ -2211,6 +2256,7 @@ export const TAMPER_KINDS = new Map([
2211
2256
  ["ASSERTION_REMOVAL", "removal"],
2212
2257
  ["ASSERTION_WEAKENED", "weakening"],
2213
2258
  ["ASSERTION_EXPECTATION_CHANGED", "expectation"],
2259
+ ["TEST_DEREGISTERED", "deregistration"],
2214
2260
  ]);
2215
2261
 
2216
2262
  /** Every kind name, for CLI validation and help text. */
@@ -2406,7 +2452,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
2406
2452
  }
2407
2453
 
2408
2454
  if (!fileAssertions.has(currentFile)) {
2409
- fileAssertions.set(currentFile, { removed: [], added: 0, addedTexts: [], removedSpecific: [], addedSpecific: 0, hunks: [], examined: 0, recognised: 0, unreadable: [] });
2455
+ fileAssertions.set(currentFile, { removed: [], added: 0, addedTexts: [], removedSpecific: [], addedSpecific: 0, hunks: [], examined: 0, recognised: 0, unreadable: [], declRemoved: [], declAdded: [] });
2410
2456
  }
2411
2457
  const fileStats = fileAssertions.get(currentFile);
2412
2458
  if (pendingHunk) {
@@ -2419,6 +2465,10 @@ export function checkTestTampering(diffOrText = "", options = {}) {
2419
2465
  const deletedText = line.slice(1);
2420
2466
  if (hunk) hunk.lines.push({ kind: "-", text: deletedText, oldNo: currentOldLineNo, newNo: null });
2421
2467
  countExamined(fileStats, deletedText);
2468
+ if (!isCommentLine(deletedText)) {
2469
+ const decl = declaredTestName(deletedText);
2470
+ if (decl) fileStats.declRemoved.push({ ...decl, line: currentOldLineNo, text: deletedText });
2471
+ }
2422
2472
  if (!isCommentLine(deletedText) && ASSERTION_PATTERN.test(deletedText)) {
2423
2473
  fileStats.removed.push({ line: currentOldLineNo, text: deletedText });
2424
2474
  if (isSpecificAssertion(deletedText)) {
@@ -2430,6 +2480,10 @@ export function checkTestTampering(diffOrText = "", options = {}) {
2430
2480
  const addedText = line.slice(1);
2431
2481
  if (hunk) hunk.lines.push({ kind: "+", text: addedText, oldNo: null, newNo: currentNewLineNo });
2432
2482
  countExamined(fileStats, addedText);
2483
+ if (!isCommentLine(addedText)) {
2484
+ const decl = declaredTestName(addedText);
2485
+ if (decl) fileStats.declAdded.push({ ...decl, line: currentNewLineNo, text: addedText });
2486
+ }
2433
2487
  let isVacuous = false;
2434
2488
 
2435
2489
  // Check skip injections
@@ -2563,6 +2617,40 @@ export function checkTestTampering(diffOrText = "", options = {}) {
2563
2617
  }
2564
2618
  }
2565
2619
 
2620
+ // A test renamed out of its runner's discovery convention.
2621
+ //
2622
+ // pytest collects `test_*` and nothing else, so `def test_totals` becoming
2623
+ // `def totals` deletes the test from every future run while leaving it in
2624
+ // the file, fully written, with all its assertions intact. Every count in
2625
+ // this guard stays level: nothing was removed, weakened or rewritten.
2626
+ //
2627
+ // Until now this was caught only by accident, as a side effect of the
2628
+ // blanket that blocked every unrecognised edit to a test file — which also
2629
+ // blocked adding an import, and whose printed remedy (`tamperGuard: "warn"`)
2630
+ // switched off the real checks along with the blanket. Narrowing that blanket
2631
+ // is what makes this its own finding, with its own name and its own remedy.
2632
+ for (const [file, stats] of fileAssertions.entries()) {
2633
+ const takenAdds = new Set();
2634
+ for (const before of stats.declRemoved || []) {
2635
+ if (!before.collected) continue;
2636
+ const idx = (stats.declAdded || []).findIndex((after, i) => !takenAdds.has(i) && isDeregistration(before, after));
2637
+ if (idx === -1) continue;
2638
+ takenAdds.add(idx);
2639
+ const after = stats.declAdded[idx];
2640
+ violations.push({
2641
+ file,
2642
+ line: after.line ?? before.line,
2643
+ type: "TEST_DEREGISTERED",
2644
+ reason:
2645
+ `Test Tamper Guard: ${JSON.stringify(before.name)} was renamed to ${JSON.stringify(after.name)} in ${file}` +
2646
+ `${after.line ? `:${after.line}` : ""}. The runner collects tests by name, so the test still exists in ` +
2647
+ `the file and no longer runs — the same effect as deleting it, with none of the signs. ` +
2648
+ `If the test is genuinely obsolete, delete it; if it is being turned into a helper, say so with ` +
2649
+ `--allow-test-change deregistration.`,
2650
+ });
2651
+ }
2652
+ }
2653
+
2566
2654
  for (const [file, stats] of fileAssertions.entries()) {
2567
2655
  // An expectation that was rewritten rather than removed.
2568
2656
  //
@@ -2668,10 +2756,33 @@ export function checkTestTampering(diffOrText = "", options = {}) {
2668
2756
  // assertion, is not the same as "checked and clean" — it is the state where
2669
2757
  // this guard has nothing to say. Saying nothing and saying "approved" have
2670
2758
  // to look different, which is the whole reason `status` exists.
2759
+ // `unreadable` is the evidence, and it is required.
2760
+ //
2761
+ // `|| examined > 0` used to stand here, and it threw away the distinction
2762
+ // this whole apparatus exists to draw. `ASSERTION_SHAPED` and `unreadable[]`
2763
+ // were built to separate "assertion-shaped lines were present and none of
2764
+ // them parsed" — a dialect the guard cannot read — from "there were no
2765
+ // assertions in these lines at all", which is most ordinary work on a test
2766
+ // file. That clause collapsed the two, so *any* changed substantive line in
2767
+ // a test file with no recognised assertion became a CRITICAL block:
2768
+ // measured on `pytest-dev/iniconfig`, renaming a test function did it, and
2769
+ // so did adding `import os`.
2770
+ //
2771
+ // The tell was in the finding itself: it carried `file: null`, `line: null`
2772
+ // and no sample, because `unreadable` was empty — the guard blocked while
2773
+ // holding no evidence of anything, and advised a pytest repository that its
2774
+ // assertion library might be unsupported, from a list that names pytest.
2775
+ //
2776
+ // Nothing is weakened by requiring the evidence. A removed or rewritten
2777
+ // assertion is a recognised assertion line, so it raises `assertionsSeen`
2778
+ // and goes to the ordinary removal and weakening checks; it never reached
2779
+ // this branch. What is lost is only the blanket, and a blanket that fires
2780
+ // on `import os` teaches its way around itself: the remedy it printed was
2781
+ // `tamperGuard: "warn"`, which switches the real guard off too.
2671
2782
  const status =
2672
2783
  reported.length > 0
2673
2784
  ? "FAIL"
2674
- : assertionsSeen === 0 && (unreadable.length > 0 || examined > 0)
2785
+ : assertionsSeen === 0 && unreadable.length > 0
2675
2786
  ? "UNREADABLE"
2676
2787
  : examined > 0
2677
2788
  ? "PASS"