jules-orchestrator-kit 0.69.0 → 0.71.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +53 -0
- package/README.md +1 -1
- package/ROADMAP_V1.md +18 -3
- package/bin/agentctl.mjs +61 -1
- package/package.json +1 -1
- package/scripts/guard-reach-check.mjs +46 -3
- package/src/config.mjs +88 -3
- package/src/engine.mjs +139 -24
- package/src/guard-policy.mjs +174 -0
- package/src/ops/test-collection.mjs +74 -0
- package/src/security.mjs +113 -2
- package/src/session-ops.mjs +117 -10
- package/src/stack-detector.mjs +52 -14
- package/src/wizard-init.mjs +38 -15
- package/src/wizard-task.mjs +15 -1
package/src/engine.mjs
CHANGED
|
@@ -459,7 +459,28 @@ export async function gate(opts = {}) {
|
|
|
459
459
|
let assertMetrics = {};
|
|
460
460
|
|
|
461
461
|
if (isAssertion) {
|
|
462
|
-
|
|
462
|
+
// The loosening flags have to reach here too.
|
|
463
|
+
//
|
|
464
|
+
// `scanDiff` above is given `allowTestChanges` and honours it, while
|
|
465
|
+
// the `assert:test-integrity` stage received only its own stage object
|
|
466
|
+
// and re-ran the same guard with none of them. So an override was
|
|
467
|
+
// accepted by one phase and ignored by the next: `--allow-test-change
|
|
468
|
+
// deregistration` turned the SECRETS phase green and then failed the
|
|
469
|
+
// run at the anti-tamper stage, with a flag hint the operator had
|
|
470
|
+
// already followed. One rule, two places, and the second kept the old
|
|
471
|
+
// answer — for the eighth time in this project's history, which is why
|
|
472
|
+
// the regression test asserts the verdict a caller receives rather
|
|
473
|
+
// than the behaviour of either site.
|
|
474
|
+
const assertRes = runAssertion(
|
|
475
|
+
{
|
|
476
|
+
...stage,
|
|
477
|
+
allowTestModifications: opts.allowTestModifications === true,
|
|
478
|
+
allowTestChanges: opts.allowTestChanges,
|
|
479
|
+
tamperGuard: trustedVerify.tamperGuard,
|
|
480
|
+
allowUnreadableTests: opts.allowUnreadableTests === true,
|
|
481
|
+
},
|
|
482
|
+
root
|
|
483
|
+
);
|
|
463
484
|
durationMs = assertRes.metrics?.durationMs ?? (Date.now() - startTime);
|
|
464
485
|
stdoutRedacted = assertRes.stdout || "";
|
|
465
486
|
stderrRedacted = assertRes.stderr || "";
|
|
@@ -894,19 +915,49 @@ export async function repair(failure, opts = {}) {
|
|
|
894
915
|
break;
|
|
895
916
|
}
|
|
896
917
|
|
|
897
|
-
attempts.push({ n, session, ok: true });
|
|
898
|
-
appendTelemetry(root, "ooda_repair_attempt", { attempt: n, ok: true });
|
|
899
|
-
|
|
900
918
|
// Poll async provider for terminal session state before executing re-verification gates
|
|
919
|
+
let pollVerdict = null;
|
|
901
920
|
if (provider && session) {
|
|
902
|
-
await pollSessionState(provider, session, {
|
|
921
|
+
pollVerdict = await pollSessionState(provider, session, {
|
|
903
922
|
root,
|
|
904
923
|
dryRun: opts.dryRun,
|
|
905
924
|
pollIntervalMs: opts.pollIntervalMs,
|
|
906
925
|
maxPollAttempts: opts.maxPollAttempts,
|
|
907
926
|
});
|
|
927
|
+
|
|
928
|
+
// The gate below is the authority on whether the change works, so it runs
|
|
929
|
+
// either way. But a non-terminal session means it is about to judge a
|
|
930
|
+
// tree the agent may not have finished writing, and saying nothing here
|
|
931
|
+
// turns that into a confusing cascade of repair attempts against a
|
|
932
|
+
// half-applied patch.
|
|
933
|
+
if (pollVerdict && pollVerdict.terminal !== true) {
|
|
934
|
+
const why = pollVerdict.blockedOn
|
|
935
|
+
? `waiting on an actor (${pollVerdict.blockedOn})`
|
|
936
|
+
: pollVerdict.unreachable
|
|
937
|
+
? "the provider stopped answering"
|
|
938
|
+
: `still ${pollVerdict.status} when the poll budget ran out`;
|
|
939
|
+
console.warn(
|
|
940
|
+
`[SESSION_NOT_TERMINAL] Session ${session.id} is ${why}. Re-verification is running against a tree the agent may not have finished writing.`
|
|
941
|
+
);
|
|
942
|
+
appendTelemetry(root, "session_not_terminal", {
|
|
943
|
+
attempt: n,
|
|
944
|
+
sessionId: session.id,
|
|
945
|
+
status: pollVerdict.status,
|
|
946
|
+
blockedOn: pollVerdict.blockedOn || null,
|
|
947
|
+
timedOut: Boolean(pollVerdict.timedOut),
|
|
948
|
+
unreachable: Boolean(pollVerdict.unreachable),
|
|
949
|
+
});
|
|
950
|
+
}
|
|
908
951
|
}
|
|
909
952
|
|
|
953
|
+
attempts.push({
|
|
954
|
+
n,
|
|
955
|
+
session,
|
|
956
|
+
ok: true,
|
|
957
|
+
poll: pollVerdict ? { status: pollVerdict.status, terminal: pollVerdict.terminal } : null,
|
|
958
|
+
});
|
|
959
|
+
appendTelemetry(root, "ooda_repair_attempt", { attempt: n, ok: true });
|
|
960
|
+
|
|
910
961
|
// Re-verify after repair attempt
|
|
911
962
|
const gateRes = await gate({ root, config, fix: false, progressBus, progressToken });
|
|
912
963
|
if (gateRes.ok) {
|
|
@@ -1000,10 +1051,38 @@ export async function checkTaskPremise(task = {}, opts = {}) {
|
|
|
1000
1051
|
}
|
|
1001
1052
|
|
|
1002
1053
|
/**
|
|
1003
|
-
*
|
|
1054
|
+
* Session states the API documents as final. The full `SessionState` enum is
|
|
1055
|
+
* transcribed in `docs/jules-quality-plan.md`.
|
|
1056
|
+
*/
|
|
1057
|
+
export const TERMINAL_SESSION_STATES = new Set(["COMPLETED", "FAILED"]);
|
|
1058
|
+
|
|
1059
|
+
/**
|
|
1060
|
+
* Session states that cannot advance without an actor — a human approving a
|
|
1061
|
+
* plan, a human answering a question, or whoever paused it resuming it.
|
|
1062
|
+
*
|
|
1063
|
+
* Polling these is not waiting, it is spending the budget: the state cannot
|
|
1064
|
+
* change on its own no matter how long the loop runs.
|
|
1065
|
+
*/
|
|
1066
|
+
export const BLOCKING_SESSION_STATES = new Set(["AWAITING_PLAN_APPROVAL", "AWAITING_USER_FEEDBACK", "PAUSED"]);
|
|
1067
|
+
|
|
1068
|
+
/**
|
|
1069
|
+
* Polls the provider until the session reaches a terminal state, blocks on an
|
|
1070
|
+
* actor, or the poll budget runs out.
|
|
1071
|
+
*
|
|
1072
|
+
* The verdict is explicit about which of those three happened, because the
|
|
1073
|
+
* caller runs the verification gate on the assumption that the agent has
|
|
1074
|
+
* finished writing. A previous version returned `COMPLETED` for every
|
|
1075
|
+
* non-terminal exit — a session sitting in `AWAITING_USER_FEEDBACK` and one
|
|
1076
|
+
* still `IN_PROGRESS` when the budget expired were both reported as success,
|
|
1077
|
+
* which is the same shape this project exists to refuse.
|
|
1078
|
+
*
|
|
1079
|
+
* @returns {Promise<object>} Always carries `status` and `terminal`. Non-terminal
|
|
1080
|
+
* exits additionally carry one of `blockedOn`, `timedOut` or `unreachable`.
|
|
1081
|
+
* `status` is never synthesised: it is the last state the provider reported,
|
|
1082
|
+
* or `UNKNOWN` when it never answered.
|
|
1004
1083
|
*/
|
|
1005
1084
|
export async function pollSessionState(provider, session, opts = {}) {
|
|
1006
|
-
if (!session || !session.id) return { status: "
|
|
1085
|
+
if (!session || !session.id) return { status: "UNKNOWN", terminal: false, unpolled: true };
|
|
1007
1086
|
const initialStatus = String(session.status || session.state || "").toUpperCase();
|
|
1008
1087
|
if (
|
|
1009
1088
|
initialStatus === "COMPLETED" ||
|
|
@@ -1013,13 +1092,20 @@ export async function pollSessionState(provider, session, opts = {}) {
|
|
|
1013
1092
|
session.id.startsWith("mock-") ||
|
|
1014
1093
|
session.id.startsWith("dry-run-")
|
|
1015
1094
|
) {
|
|
1016
|
-
|
|
1095
|
+
// A dry run has no session to watch, so it reports the outcome a real
|
|
1096
|
+
// dispatch would have had to earn. Flagged as simulated so a caller can
|
|
1097
|
+
// tell the two apart; a terminal state already reported by the provider is
|
|
1098
|
+
// a fact, not a simulation, and is returned as itself.
|
|
1099
|
+
const status = initialStatus || "COMPLETED";
|
|
1100
|
+
return { status, terminal: TERMINAL_SESSION_STATES.has(status), simulated: true, polls: 0 };
|
|
1017
1101
|
}
|
|
1018
1102
|
|
|
1019
1103
|
const maxAttempts = opts.maxPollAttempts || 30;
|
|
1020
1104
|
const pollIntervalMs = opts.pollIntervalMs || 1000;
|
|
1021
1105
|
const timeoutMs = opts.pollTimeoutMs || 300000;
|
|
1022
1106
|
const startTime = Date.now();
|
|
1107
|
+
let lastStatus = "";
|
|
1108
|
+
let polls = 0;
|
|
1023
1109
|
|
|
1024
1110
|
for (let attempt = 0; attempt < maxAttempts; attempt++) {
|
|
1025
1111
|
if (Date.now() - startTime > timeoutMs) break;
|
|
@@ -1035,30 +1121,59 @@ export async function pollSessionState(provider, session, opts = {}) {
|
|
|
1035
1121
|
} catch (_) {}
|
|
1036
1122
|
}
|
|
1037
1123
|
|
|
1038
|
-
if (currentSession) {
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
|
|
1042
|
-
|
|
1043
|
-
|
|
1044
|
-
|
|
1045
|
-
|
|
1046
|
-
|
|
1047
|
-
|
|
1124
|
+
if (!currentSession) {
|
|
1125
|
+
return {
|
|
1126
|
+
...session,
|
|
1127
|
+
status: lastStatus || "UNKNOWN",
|
|
1128
|
+
terminal: false,
|
|
1129
|
+
unreachable: true,
|
|
1130
|
+
polls,
|
|
1131
|
+
};
|
|
1132
|
+
}
|
|
1133
|
+
|
|
1134
|
+
polls += 1;
|
|
1135
|
+
const status = String(currentSession.status || currentSession.state || "").toUpperCase();
|
|
1136
|
+
if (status) lastStatus = status;
|
|
1137
|
+
|
|
1138
|
+
if (TERMINAL_SESSION_STATES.has(status)) {
|
|
1139
|
+
return { ...currentSession, status, terminal: true, polls };
|
|
1140
|
+
}
|
|
1141
|
+
|
|
1142
|
+
if (BLOCKING_SESSION_STATES.has(status)) {
|
|
1143
|
+
// A pending plan approval is the one blocking state this loop is allowed
|
|
1144
|
+
// to resolve itself, and only when the caller said so.
|
|
1145
|
+
const isPlanApproval = status === "AWAITING_PLAN_APPROVAL";
|
|
1146
|
+
const wantsAutoApprove = Boolean(opts.autoApprovePlan || opts.autoApprove || session.autoApprovePlan);
|
|
1147
|
+
if (isPlanApproval && wantsAutoApprove && provider && typeof provider.approvePlan === "function") {
|
|
1148
|
+
try {
|
|
1149
|
+
await provider.approvePlan(session.id, opts);
|
|
1150
|
+
await new Promise((resolve) => setTimeout(resolve, pollIntervalMs));
|
|
1151
|
+
continue;
|
|
1152
|
+
} catch (err) {
|
|
1153
|
+
return {
|
|
1154
|
+
...currentSession,
|
|
1155
|
+
status,
|
|
1156
|
+
terminal: false,
|
|
1157
|
+
blockedOn: status,
|
|
1158
|
+
approvePlanError: err && err.message ? err.message : String(err),
|
|
1159
|
+
polls,
|
|
1160
|
+
};
|
|
1048
1161
|
}
|
|
1049
1162
|
}
|
|
1050
1163
|
|
|
1051
|
-
|
|
1052
|
-
return { ...currentSession, status };
|
|
1053
|
-
}
|
|
1054
|
-
} else {
|
|
1055
|
-
break;
|
|
1164
|
+
return { ...currentSession, status, terminal: false, blockedOn: status, polls };
|
|
1056
1165
|
}
|
|
1057
1166
|
|
|
1058
1167
|
await new Promise((resolve) => setTimeout(resolve, pollIntervalMs));
|
|
1059
1168
|
}
|
|
1060
1169
|
|
|
1061
|
-
return {
|
|
1170
|
+
return {
|
|
1171
|
+
...session,
|
|
1172
|
+
status: lastStatus || "UNKNOWN",
|
|
1173
|
+
terminal: false,
|
|
1174
|
+
timedOut: true,
|
|
1175
|
+
polls,
|
|
1176
|
+
};
|
|
1062
1177
|
}
|
|
1063
1178
|
|
|
1064
1179
|
function buildRepairPrompt(failure, attempt, _config, extraPromptDirective = null) {
|
package/src/guard-policy.mjs
CHANGED
|
@@ -454,6 +454,57 @@ export const SCOPE_CANARIES = [
|
|
|
454
454
|
* from never showed it.
|
|
455
455
|
*/
|
|
456
456
|
export const INNOCENT_EDITS = [
|
|
457
|
+
// Everything below this comment was a CRITICAL rejection until v0.71.0.
|
|
458
|
+
//
|
|
459
|
+
// Not through any rule that examined them: `UNREADABLE` was reached whenever
|
|
460
|
+
// a test file had changed lines and none of them parsed as an assertion, so
|
|
461
|
+
// the verdict was the same for a repository speaking an unsupported dialect
|
|
462
|
+
// and for one where the edit simply was not an assertion. Adding an import
|
|
463
|
+
// was a CRITICAL block. The finding carried no file, no line and no sample,
|
|
464
|
+
// because there was nothing to name — and it advised a pytest repository
|
|
465
|
+
// that its assertion library might be unsupported, from a list naming pytest.
|
|
466
|
+
//
|
|
467
|
+
// These are ordinary work. Every one of them has to be silent.
|
|
468
|
+
{
|
|
469
|
+
id: "add-import/pytest",
|
|
470
|
+
file: "testing/test_iniconfig.py",
|
|
471
|
+
context: "# fixtures",
|
|
472
|
+
removed: [],
|
|
473
|
+
added: ["import os"],
|
|
474
|
+
why: "adding an import to a test file is not an assertion and not tampering",
|
|
475
|
+
},
|
|
476
|
+
{
|
|
477
|
+
id: "rename-test-multiline/pytest",
|
|
478
|
+
file: "testing/test_iniconfig.py",
|
|
479
|
+
context: "# parsing",
|
|
480
|
+
removed: ["def test_parse_strips_inline_comments() -> None:"],
|
|
481
|
+
added: ["def test_parse_strips_inline_comments_from_continuations() -> None:"],
|
|
482
|
+
why: "still collected by pytest, so nothing left the run",
|
|
483
|
+
},
|
|
484
|
+
{
|
|
485
|
+
id: "docstring/pytest",
|
|
486
|
+
file: "testing/test_iniconfig.py",
|
|
487
|
+
context: "# parsing",
|
|
488
|
+
removed: [' """Check comments."""'],
|
|
489
|
+
added: [' """Check that inline comments are stripped."""'],
|
|
490
|
+
why: "prose about the test, not a value it asserts",
|
|
491
|
+
},
|
|
492
|
+
{
|
|
493
|
+
id: "type-annotation/pytest",
|
|
494
|
+
file: "testing/test_iniconfig.py",
|
|
495
|
+
context: "# fixtures",
|
|
496
|
+
removed: ["def make_config(data):"],
|
|
497
|
+
added: ["def make_config(data: str) -> IniConfig:"],
|
|
498
|
+
why: "a helper's signature, typed; no claim changed",
|
|
499
|
+
},
|
|
500
|
+
{
|
|
501
|
+
id: "add-fixture/go",
|
|
502
|
+
file: "calc_test.go",
|
|
503
|
+
context: "// helpers",
|
|
504
|
+
removed: [],
|
|
505
|
+
added: ["var cases = []int{1, 2, 3}"],
|
|
506
|
+
why: "test data added, nothing asserted or unasserted",
|
|
507
|
+
},
|
|
457
508
|
// Renaming a test is one of the most ordinary edits there is. On the
|
|
458
509
|
// one-line form the name blanked to the same shape as its replacement, the
|
|
459
510
|
// two paired, and the pair was reported as a rewritten expectation.
|
|
@@ -598,6 +649,129 @@ export const INNOCENT_EDITS = [
|
|
|
598
649
|
* will always end somewhere; what must never happen again is that the edge
|
|
599
650
|
* is silent.
|
|
600
651
|
*/
|
|
652
|
+
/**
|
|
653
|
+
* Renames that delete a test from the run without deleting a line of it.
|
|
654
|
+
*
|
|
655
|
+
* pytest collects `test_*`, Go collects `Test*`. For those runners the name is
|
|
656
|
+
* the registration, so `def test_totals` → `def totals` removes the test as
|
|
657
|
+
* completely as deleting the file, and every count in the tamper guard stays
|
|
658
|
+
* level: nothing removed, nothing weakened, nothing rewritten.
|
|
659
|
+
*
|
|
660
|
+
* These were caught only as a side effect of the blanket that rejected every
|
|
661
|
+
* unrecognised edit to a test file — which also rejected adding an import, and
|
|
662
|
+
* whose printed remedy switched the real checks off. They are their own
|
|
663
|
+
* finding now, so narrowing that blanket costs nothing.
|
|
664
|
+
*/
|
|
665
|
+
export const DEREGISTRATION_CANARIES = [
|
|
666
|
+
{
|
|
667
|
+
id: "pytest/underscore",
|
|
668
|
+
file: "tests/test_totals.py",
|
|
669
|
+
context: "# totals",
|
|
670
|
+
removed: ["def test_totals_round_to_cents():"],
|
|
671
|
+
added: ["def totals_round_to_cents():"],
|
|
672
|
+
why: "pytest collects by the test_ prefix, so this test no longer runs",
|
|
673
|
+
},
|
|
674
|
+
{
|
|
675
|
+
id: "pytest/camel",
|
|
676
|
+
file: "tests/test_totals.py",
|
|
677
|
+
context: "# totals",
|
|
678
|
+
removed: ["def testTotals():"],
|
|
679
|
+
added: ["def Totals():"],
|
|
680
|
+
why: "the prefix is what registers it, in either spelling",
|
|
681
|
+
},
|
|
682
|
+
{
|
|
683
|
+
id: "go/test",
|
|
684
|
+
file: "calc_test.go",
|
|
685
|
+
context: "// totals",
|
|
686
|
+
removed: ["func TestTotals(t *testing.T) {"],
|
|
687
|
+
added: ["func Totals(t *testing.T) {"],
|
|
688
|
+
why: "go test collects by the Test prefix",
|
|
689
|
+
},
|
|
690
|
+
{
|
|
691
|
+
id: "go/benchmark",
|
|
692
|
+
file: "calc_test.go",
|
|
693
|
+
context: "// totals",
|
|
694
|
+
removed: ["func BenchmarkTotals(b *testing.B) {"],
|
|
695
|
+
added: ["func Totals(b *testing.B) {"],
|
|
696
|
+
why: "the same rule for the other collected prefixes",
|
|
697
|
+
},
|
|
698
|
+
];
|
|
699
|
+
|
|
700
|
+
/**
|
|
701
|
+
* A verification command that exits 0 having written nothing at all.
|
|
702
|
+
*
|
|
703
|
+
* The one absence that is evidence. The floor is otherwise deliberately
|
|
704
|
+
* one-sided — an unrecognised runner states no count and passes, because
|
|
705
|
+
* hard-redding every runner not on the list would be worse than the hole it
|
|
706
|
+
* closes — but an unrecognised runner still prints something. Zero bytes on
|
|
707
|
+
* both streams is a command that ran nothing, and `pnpm -r test` on a
|
|
708
|
+
* workspace whose packages declare no test script is exactly that: it was
|
|
709
|
+
* indistinguishable from a full suite by every signal the gate had.
|
|
710
|
+
*/
|
|
711
|
+
export const SILENT_RUN_CANARIES = [
|
|
712
|
+
{ id: "pnpm -r test", command: "pnpm -r test", stdout: "", stderr: "" },
|
|
713
|
+
{ id: "npm test", command: "npm test", stdout: "", stderr: "" },
|
|
714
|
+
{ id: "yarn workspaces test", command: "yarn workspaces foreach run test", stdout: "", stderr: "" },
|
|
715
|
+
{ id: "pytest", command: "python3 -m pytest", stdout: "", stderr: "" },
|
|
716
|
+
{ id: "go test", command: "go test ./...", stdout: "", stderr: "" },
|
|
717
|
+
{ id: "whitespace is not output", command: "npm test", stdout: " \n", stderr: "\n" },
|
|
718
|
+
];
|
|
719
|
+
|
|
720
|
+
/**
|
|
721
|
+
* Honest static gates that print nothing, and must keep passing.
|
|
722
|
+
*
|
|
723
|
+
* The counterweight, and the reason the rule above reads the command as well
|
|
724
|
+
* as the output. `tsc --noEmit`, `node --check`, `go vet` and `compileall`
|
|
725
|
+
* all exit 0 in silence when they succeed — that is what success looks like
|
|
726
|
+
* for a checker — and two of them are commands this kit writes itself for a
|
|
727
|
+
* repository that has no suite yet. A rule keyed on silence alone hard-redded
|
|
728
|
+
* every one of them, which is exactly the first-run rejection of correct code
|
|
729
|
+
* that the collection floor is otherwise so careful to avoid.
|
|
730
|
+
*/
|
|
731
|
+
export const SILENT_STATIC_GATES = [
|
|
732
|
+
{ id: "tsc", command: "tsc --noEmit" },
|
|
733
|
+
{ id: "node --check", command: "node --check index.js" },
|
|
734
|
+
{ id: "go vet", command: "go vet ./..." },
|
|
735
|
+
{ id: "compileall", command: "python3 -m compileall -q ." },
|
|
736
|
+
{ id: "generated parse gate", command: "node --check src/index.mjs" },
|
|
737
|
+
];
|
|
738
|
+
|
|
739
|
+
/**
|
|
740
|
+
* Values that must survive a trip through the emitter and back.
|
|
741
|
+
*
|
|
742
|
+
* `yamlScalar` writes the manifests and `parseYaml` reads them, and a pair
|
|
743
|
+
* like that disagreeing about one character is this project's recurring
|
|
744
|
+
* defect with both halves in a single module. The parser opened a comment at
|
|
745
|
+
* the first `#` on a line, quoted or not, so `verify.test` containing a hash
|
|
746
|
+
* was silently truncated on the way in and the gate ran a command the user
|
|
747
|
+
* never wrote — reporting on it as if it were theirs.
|
|
748
|
+
*
|
|
749
|
+
* The corpus is the hard cases on purpose: hashes, colons, leading stars,
|
|
750
|
+
* apostrophes, the empty string, and the words YAML would otherwise read as
|
|
751
|
+
* booleans.
|
|
752
|
+
*/
|
|
753
|
+
export const YAML_ROUNDTRIP_CASES = [
|
|
754
|
+
"pnpm -r test",
|
|
755
|
+
"PYTHONPATH=src python3 -m pytest",
|
|
756
|
+
'pytest -k "not #slow"',
|
|
757
|
+
"has #hash",
|
|
758
|
+
"# leading hash",
|
|
759
|
+
"a: b",
|
|
760
|
+
"**/*.pem",
|
|
761
|
+
"**/.env.*",
|
|
762
|
+
".github/**",
|
|
763
|
+
"",
|
|
764
|
+
"true",
|
|
765
|
+
"yes",
|
|
766
|
+
"1.5",
|
|
767
|
+
"it's fine",
|
|
768
|
+
"it's #1",
|
|
769
|
+
"O'Brien",
|
|
770
|
+
"agent/",
|
|
771
|
+
" leading space",
|
|
772
|
+
"trailing space ",
|
|
773
|
+
];
|
|
774
|
+
|
|
601
775
|
export const UNREADABLE_DIALECTS = [
|
|
602
776
|
{
|
|
603
777
|
id: "hspec",
|
|
@@ -111,6 +111,56 @@ const GO_RAN_SOMETHING = /^(?:(?:ok|FAIL)\s+(?!\d+\s)\S+|---\s+(?:PASS|FAIL|SKIP
|
|
|
111
111
|
* `count` is null when no recognised runner stated one — which is not a
|
|
112
112
|
* finding, only an absence of evidence.
|
|
113
113
|
*/
|
|
114
|
+
/**
|
|
115
|
+
* Did the command write nothing at all, on either stream?
|
|
116
|
+
*
|
|
117
|
+
* This is the one absence that is evidence rather than the lack of it. The
|
|
118
|
+
* one-sided floor exists because an unrecognised runner states no count, and
|
|
119
|
+
* hard-redding every runner not on the list would be worse than the hole it
|
|
120
|
+
* closes. But an unrecognised runner still *prints*: dots, a summary line,
|
|
121
|
+
* a package name, something. Zero bytes on both streams is not a dialect the
|
|
122
|
+
* list has yet to learn — it is a command that ran nothing.
|
|
123
|
+
*
|
|
124
|
+
* `pnpm -r test` on a workspace whose packages declare no test script is the
|
|
125
|
+
* shape that made this necessary: it exits 0, writes nothing anywhere, and
|
|
126
|
+
* was indistinguishable from a full suite by every signal the gate had.
|
|
127
|
+
*
|
|
128
|
+
* Resolved here and consumed in two places — the gate's floor and `init`'s
|
|
129
|
+
* oracle probe — because writing the rule once in each is how this project
|
|
130
|
+
* has repeatedly ended up with two answers to one question.
|
|
131
|
+
*/
|
|
132
|
+
export function producedNoOutput(stdout = "", stderr = "") {
|
|
133
|
+
return `${stdout || ""}${stderr || ""}`.trim() === "";
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/**
|
|
137
|
+
* Does this command claim to run a test suite?
|
|
138
|
+
*
|
|
139
|
+
* Silence alone cannot carry the verdict, and the first version of this rule
|
|
140
|
+
* assumed it could. `node --check index.js`, `tsc --noEmit`, `go vet ./...`
|
|
141
|
+
* and `python3 -m compileall -q .` all exit 0 having printed nothing — and
|
|
142
|
+
* they are honest static gates, two of which this kit writes itself for
|
|
143
|
+
* repositories that have no suite yet. Failing on silence alone hard-redded
|
|
144
|
+
* every one of them, which is the same first-run rejection of correct code
|
|
145
|
+
* that the whole collection floor is careful to avoid.
|
|
146
|
+
*
|
|
147
|
+
* Nothing in the *output* separates `pnpm -r test` from `tsc --noEmit`; both
|
|
148
|
+
* are empty. The difference is in what the command says it is. So this reads
|
|
149
|
+
* the command, the same way `isPlaceholderTestScript` does: a command that is
|
|
150
|
+
* recognisably a suite invocation and printed nothing ran no suite, while a
|
|
151
|
+
* static checker that printed nothing did exactly what it promised.
|
|
152
|
+
*
|
|
153
|
+
* One-sided in the safe direction, like everything else here. An unrecognised
|
|
154
|
+
* command is not treated as a suite, so an unusual runner keeps its advisory
|
|
155
|
+
* pass rather than becoming a hard red.
|
|
156
|
+
*/
|
|
157
|
+
const TEST_SUITE_COMMAND =
|
|
158
|
+
/(?:^|\s|\/)(?:pytest|jest|vitest|mocha|ava|karma|jasmine|nyc|c8|tap|tape|rspec|minitest|phpunit|behave|nose2?|ginkgo|gotestsum|nextest)\b|\b(?:go|cargo|swift|dart|flutter|deno|bun|dotnet|mix|lein|sbt|gradlew?|mvn)\s+test\b|\bunittest\b|-m\s+(?:pytest|unittest)\b|\bnode\s+--test\b|(?:^|&&|;|\|)\s*(?:npm|pnpm|yarn|bun|npx)\b[^&;|]*?\btest\b/;
|
|
159
|
+
|
|
160
|
+
export function looksLikeTestSuiteCommand(cmd) {
|
|
161
|
+
return typeof cmd === "string" && TEST_SUITE_COMMAND.test(cmd);
|
|
162
|
+
}
|
|
163
|
+
|
|
114
164
|
export function parseCollectedTests(stdout = "", stderr = "") {
|
|
115
165
|
const text = `${stdout || ""}\n${stderr || ""}`;
|
|
116
166
|
if (!text.trim()) return { count: null, runner: null };
|
|
@@ -168,6 +218,30 @@ export function checkCollectionFloor(testResult, opts = {}) {
|
|
|
168
218
|
return { ok: true, count: null, runner: null, reason: null };
|
|
169
219
|
}
|
|
170
220
|
|
|
221
|
+
// A command that says it runs a suite, and printed nothing, ran no suite.
|
|
222
|
+
//
|
|
223
|
+
// Both halves are required. Silence alone would hard-red `tsc --noEmit` and
|
|
224
|
+
// `python3 -m compileall`, which are honest static gates this kit generates
|
|
225
|
+
// itself; the command shape alone would say nothing, because a real suite
|
|
226
|
+
// prints. Together they are decidable, and they are exactly `pnpm -r test`
|
|
227
|
+
// on a workspace whose packages declare no test script.
|
|
228
|
+
if (looksLikeTestSuiteCommand(testResult.command) && producedNoOutput(testResult.stdout, testResult.stderr)) {
|
|
229
|
+
return {
|
|
230
|
+
ok: false,
|
|
231
|
+
count: 0,
|
|
232
|
+
runner: null,
|
|
233
|
+
silent: true,
|
|
234
|
+
reason:
|
|
235
|
+
`The verification command ${testResult.command ? `${JSON.stringify(testResult.command)} ` : ""}` +
|
|
236
|
+
`exited 0 and wrote nothing at all — no test names, no summary, no count. ` +
|
|
237
|
+
`Every test runner prints something, so this command ran no suite, and approving this change ` +
|
|
238
|
+
`would certify nothing. A workspace command such as \`pnpm -r test\` does this when no package ` +
|
|
239
|
+
`declares a test script. Point verify.test at the suite that covers this repository ` +
|
|
240
|
+
`(often the root script rather than the recursive one), or — if this repository intentionally ` +
|
|
241
|
+
`uses only the scope and secret phases — set verify.required: false, which says so on the record.`,
|
|
242
|
+
};
|
|
243
|
+
}
|
|
244
|
+
|
|
171
245
|
const { count, runner } = parseCollectedTests(testResult.stdout, testResult.stderr);
|
|
172
246
|
|
|
173
247
|
// Deliberately one-sided: only a *stated* zero fails, because failing on
|
package/src/security.mjs
CHANGED
|
@@ -1875,6 +1875,51 @@ function splitTrailingMessage(clean) {
|
|
|
1875
1875
|
return { head: clean.slice(0, lastComma), msg: tail };
|
|
1876
1876
|
}
|
|
1877
1877
|
|
|
1878
|
+
/**
|
|
1879
|
+
* Test declarations that a runner finds by the *name* of the function.
|
|
1880
|
+
*
|
|
1881
|
+
* pytest collects `def test_*`, Go collects `func Test*`, and unittest and
|
|
1882
|
+
* Minitest collect `def test_*` off the case class. For those runners the
|
|
1883
|
+
* name is not prose — it is the registration. Renaming `test_totals` to
|
|
1884
|
+
* `totals` deletes the test from the run as completely as removing the file,
|
|
1885
|
+
* and the diff shows a rename.
|
|
1886
|
+
*
|
|
1887
|
+
* Only these name-driven runners are listed. `it("...")`, `#[test]` and
|
|
1888
|
+
* `@Test` register by call, attribute or annotation, so renaming what they
|
|
1889
|
+
* declare removes nothing, and the ordinary rename rules already cover them.
|
|
1890
|
+
*/
|
|
1891
|
+
const NAME_REGISTERED_DECLS = [
|
|
1892
|
+
{ lang: "python", re: /^\s*(?:async\s+)?def\s+([A-Za-z_]\w*)\s*\(/, discovered: /^test/i },
|
|
1893
|
+
{ lang: "go", re: /^\s*func\s+([A-Za-z_]\w*)\s*\(/, discovered: /^(?:Test|Benchmark|Fuzz|Example)/ },
|
|
1894
|
+
];
|
|
1895
|
+
|
|
1896
|
+
/**
|
|
1897
|
+
* The declared name on this line, and whether the runner would collect it.
|
|
1898
|
+
*
|
|
1899
|
+
* @returns {{ name: string, collected: boolean }|null}
|
|
1900
|
+
*/
|
|
1901
|
+
function declaredTestName(text) {
|
|
1902
|
+
for (const rule of NAME_REGISTERED_DECLS) {
|
|
1903
|
+
const m = rule.re.exec(text);
|
|
1904
|
+
if (m) return { name: m[1], collected: rule.discovered.test(m[1]) };
|
|
1905
|
+
}
|
|
1906
|
+
return null;
|
|
1907
|
+
}
|
|
1908
|
+
|
|
1909
|
+
/**
|
|
1910
|
+
* Is `after` the same declaration as `before` with its discovery prefix gone?
|
|
1911
|
+
*
|
|
1912
|
+
* Exact on the remainder, deliberately. `test_totals` → `totals` is a
|
|
1913
|
+
* de-registration; `test_totals` → `test_totals_rounded` is a rename and must
|
|
1914
|
+
* stay silent, which is the false red this check exists alongside rather than
|
|
1915
|
+
* instead of.
|
|
1916
|
+
*/
|
|
1917
|
+
function isDeregistration(before, after) {
|
|
1918
|
+
if (!before.collected || after.collected) return false;
|
|
1919
|
+
const stripped = before.name.replace(/^test[_-]?/i, "").replace(/^(?:Test|Benchmark|Fuzz|Example)/, "");
|
|
1920
|
+
return stripped.length > 0 && stripped === after.name;
|
|
1921
|
+
}
|
|
1922
|
+
|
|
1878
1923
|
// A test declaration whose first argument is the test's name. The name is
|
|
1879
1924
|
// prose about the test, not a value the test asserts — `test("adds", ...)`
|
|
1880
1925
|
// renamed to `test("adds positives", ...)` is the rename the diff says it is.
|
|
@@ -2211,6 +2256,7 @@ export const TAMPER_KINDS = new Map([
|
|
|
2211
2256
|
["ASSERTION_REMOVAL", "removal"],
|
|
2212
2257
|
["ASSERTION_WEAKENED", "weakening"],
|
|
2213
2258
|
["ASSERTION_EXPECTATION_CHANGED", "expectation"],
|
|
2259
|
+
["TEST_DEREGISTERED", "deregistration"],
|
|
2214
2260
|
]);
|
|
2215
2261
|
|
|
2216
2262
|
/** Every kind name, for CLI validation and help text. */
|
|
@@ -2406,7 +2452,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2406
2452
|
}
|
|
2407
2453
|
|
|
2408
2454
|
if (!fileAssertions.has(currentFile)) {
|
|
2409
|
-
fileAssertions.set(currentFile, { removed: [], added: 0, addedTexts: [], removedSpecific: [], addedSpecific: 0, hunks: [], examined: 0, recognised: 0, unreadable: [] });
|
|
2455
|
+
fileAssertions.set(currentFile, { removed: [], added: 0, addedTexts: [], removedSpecific: [], addedSpecific: 0, hunks: [], examined: 0, recognised: 0, unreadable: [], declRemoved: [], declAdded: [] });
|
|
2410
2456
|
}
|
|
2411
2457
|
const fileStats = fileAssertions.get(currentFile);
|
|
2412
2458
|
if (pendingHunk) {
|
|
@@ -2419,6 +2465,10 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2419
2465
|
const deletedText = line.slice(1);
|
|
2420
2466
|
if (hunk) hunk.lines.push({ kind: "-", text: deletedText, oldNo: currentOldLineNo, newNo: null });
|
|
2421
2467
|
countExamined(fileStats, deletedText);
|
|
2468
|
+
if (!isCommentLine(deletedText)) {
|
|
2469
|
+
const decl = declaredTestName(deletedText);
|
|
2470
|
+
if (decl) fileStats.declRemoved.push({ ...decl, line: currentOldLineNo, text: deletedText });
|
|
2471
|
+
}
|
|
2422
2472
|
if (!isCommentLine(deletedText) && ASSERTION_PATTERN.test(deletedText)) {
|
|
2423
2473
|
fileStats.removed.push({ line: currentOldLineNo, text: deletedText });
|
|
2424
2474
|
if (isSpecificAssertion(deletedText)) {
|
|
@@ -2430,6 +2480,10 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2430
2480
|
const addedText = line.slice(1);
|
|
2431
2481
|
if (hunk) hunk.lines.push({ kind: "+", text: addedText, oldNo: null, newNo: currentNewLineNo });
|
|
2432
2482
|
countExamined(fileStats, addedText);
|
|
2483
|
+
if (!isCommentLine(addedText)) {
|
|
2484
|
+
const decl = declaredTestName(addedText);
|
|
2485
|
+
if (decl) fileStats.declAdded.push({ ...decl, line: currentNewLineNo, text: addedText });
|
|
2486
|
+
}
|
|
2433
2487
|
let isVacuous = false;
|
|
2434
2488
|
|
|
2435
2489
|
// Check skip injections
|
|
@@ -2563,6 +2617,40 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2563
2617
|
}
|
|
2564
2618
|
}
|
|
2565
2619
|
|
|
2620
|
+
// A test renamed out of its runner's discovery convention.
|
|
2621
|
+
//
|
|
2622
|
+
// pytest collects `test_*` and nothing else, so `def test_totals` becoming
|
|
2623
|
+
// `def totals` deletes the test from every future run while leaving it in
|
|
2624
|
+
// the file, fully written, with all its assertions intact. Every count in
|
|
2625
|
+
// this guard stays level: nothing was removed, weakened or rewritten.
|
|
2626
|
+
//
|
|
2627
|
+
// Until now this was caught only by accident, as a side effect of the
|
|
2628
|
+
// blanket that blocked every unrecognised edit to a test file — which also
|
|
2629
|
+
// blocked adding an import, and whose printed remedy (`tamperGuard: "warn"`)
|
|
2630
|
+
// switched off the real checks along with the blanket. Narrowing that blanket
|
|
2631
|
+
// is what makes this its own finding, with its own name and its own remedy.
|
|
2632
|
+
for (const [file, stats] of fileAssertions.entries()) {
|
|
2633
|
+
const takenAdds = new Set();
|
|
2634
|
+
for (const before of stats.declRemoved || []) {
|
|
2635
|
+
if (!before.collected) continue;
|
|
2636
|
+
const idx = (stats.declAdded || []).findIndex((after, i) => !takenAdds.has(i) && isDeregistration(before, after));
|
|
2637
|
+
if (idx === -1) continue;
|
|
2638
|
+
takenAdds.add(idx);
|
|
2639
|
+
const after = stats.declAdded[idx];
|
|
2640
|
+
violations.push({
|
|
2641
|
+
file,
|
|
2642
|
+
line: after.line ?? before.line,
|
|
2643
|
+
type: "TEST_DEREGISTERED",
|
|
2644
|
+
reason:
|
|
2645
|
+
`Test Tamper Guard: ${JSON.stringify(before.name)} was renamed to ${JSON.stringify(after.name)} in ${file}` +
|
|
2646
|
+
`${after.line ? `:${after.line}` : ""}. The runner collects tests by name, so the test still exists in ` +
|
|
2647
|
+
`the file and no longer runs — the same effect as deleting it, with none of the signs. ` +
|
|
2648
|
+
`If the test is genuinely obsolete, delete it; if it is being turned into a helper, say so with ` +
|
|
2649
|
+
`--allow-test-change deregistration.`,
|
|
2650
|
+
});
|
|
2651
|
+
}
|
|
2652
|
+
}
|
|
2653
|
+
|
|
2566
2654
|
for (const [file, stats] of fileAssertions.entries()) {
|
|
2567
2655
|
// An expectation that was rewritten rather than removed.
|
|
2568
2656
|
//
|
|
@@ -2668,10 +2756,33 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2668
2756
|
// assertion, is not the same as "checked and clean" — it is the state where
|
|
2669
2757
|
// this guard has nothing to say. Saying nothing and saying "approved" have
|
|
2670
2758
|
// to look different, which is the whole reason `status` exists.
|
|
2759
|
+
// `unreadable` is the evidence, and it is required.
|
|
2760
|
+
//
|
|
2761
|
+
// `|| examined > 0` used to stand here, and it threw away the distinction
|
|
2762
|
+
// this whole apparatus exists to draw. `ASSERTION_SHAPED` and `unreadable[]`
|
|
2763
|
+
// were built to separate "assertion-shaped lines were present and none of
|
|
2764
|
+
// them parsed" — a dialect the guard cannot read — from "there were no
|
|
2765
|
+
// assertions in these lines at all", which is most ordinary work on a test
|
|
2766
|
+
// file. That clause collapsed the two, so *any* changed substantive line in
|
|
2767
|
+
// a test file with no recognised assertion became a CRITICAL block:
|
|
2768
|
+
// measured on `pytest-dev/iniconfig`, renaming a test function did it, and
|
|
2769
|
+
// so did adding `import os`.
|
|
2770
|
+
//
|
|
2771
|
+
// The tell was in the finding itself: it carried `file: null`, `line: null`
|
|
2772
|
+
// and no sample, because `unreadable` was empty — the guard blocked while
|
|
2773
|
+
// holding no evidence of anything, and advised a pytest repository that its
|
|
2774
|
+
// assertion library might be unsupported, from a list that names pytest.
|
|
2775
|
+
//
|
|
2776
|
+
// Nothing is weakened by requiring the evidence. A removed or rewritten
|
|
2777
|
+
// assertion is a recognised assertion line, so it raises `assertionsSeen`
|
|
2778
|
+
// and goes to the ordinary removal and weakening checks; it never reached
|
|
2779
|
+
// this branch. What is lost is only the blanket, and a blanket that fires
|
|
2780
|
+
// on `import os` teaches its way around itself: the remedy it printed was
|
|
2781
|
+
// `tamperGuard: "warn"`, which switches the real guard off too.
|
|
2671
2782
|
const status =
|
|
2672
2783
|
reported.length > 0
|
|
2673
2784
|
? "FAIL"
|
|
2674
|
-
: assertionsSeen === 0 &&
|
|
2785
|
+
: assertionsSeen === 0 && unreadable.length > 0
|
|
2675
2786
|
? "UNREADABLE"
|
|
2676
2787
|
: examined > 0
|
|
2677
2788
|
? "PASS"
|