claude-dev-env 2.5.0 → 2.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CLAUDE.md +20 -57
- package/_shared/pr-loop/scripts/code_rules_gate.py +2 -1
- package/_shared/pr-loop/scripts/code_rules_gate_parts/CLAUDE.md +12 -2
- package/_shared/pr-loop/scripts/code_rules_gate_parts/baseline_import_isolation.py +309 -0
- package/_shared/pr-loop/scripts/code_rules_gate_parts/staged_test_regression.py +540 -0
- package/_shared/pr-loop/scripts/code_rules_gate_parts/staged_test_running.py +206 -70
- package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/__init__.py +1 -0
- package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/_repo_test_helpers.py +76 -0
- package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/test_baseline_import_isolation.py +248 -0
- package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/test_staged_test_regression.py +309 -0
- package/_shared/pr-loop/scripts/code_rules_gate_parts/tests/test_staged_test_running.py +91 -58
- package/_shared/pr-loop/scripts/pr_loop_shared_constants/code_rules_gate_constants.py +202 -0
- package/agents/CLAUDE.md +1 -1
- package/agents/code-verifier.md +36 -7
- package/bin/codex-compat.mjs +104 -0
- package/bin/codex-compat.test.mjs +51 -0
- package/codex-capability-map.json +13 -0
- package/docs/CODE_RULES.md +2 -0
- package/docs/codex-compatibility.md +25 -0
- package/docs/nas-ssh-invocation.md +96 -12
- package/docs/references/code-review-enforcement.md +31 -6
- package/hooks/blocking/CLAUDE.md +3 -0
- package/hooks/blocking/config/code_review_enforcement_constants.py +40 -10
- package/hooks/blocking/config/test_code_review_enforcement_constants.py +56 -3
- package/hooks/blocking/eli11_reply_enforcer.py +479 -0
- package/hooks/blocking/gh_body_arg_blocker.py +1 -1
- package/hooks/blocking/nas_ssh_binary_enforcer.py +8 -46
- package/hooks/blocking/shell_substitution_blocker.py +129 -0
- package/hooks/blocking/state_description_blocker.py +1 -1
- package/hooks/blocking/stop_dispatcher.py +1 -1
- package/hooks/blocking/test_bash_pre_tool_use_dispatcher.py +2 -3
- package/hooks/blocking/test_eli11_reply_enforcer.py +457 -0
- package/hooks/blocking/test_shell_substitution_blocker.py +124 -0
- package/hooks/blocking/test_stop_dispatcher.py +23 -0
- package/hooks/blocking/test_unscoped_search_blocker.py +102 -0
- package/hooks/blocking/test_verdict_directory_write_blocker.py +4 -8
- package/hooks/blocking/unscoped_search_blocker.py +391 -0
- package/hooks/git-hooks/CLAUDE.md +3 -0
- package/hooks/git-hooks/conftest.py +30 -0
- package/hooks/git-hooks/gate_utils.py +2 -2
- package/hooks/git-hooks/git_hooks_constants/__init__.py +41 -2
- package/hooks/git-hooks/pre_push.py +75 -4
- package/hooks/git-hooks/pre_push_base_reference.py +166 -0
- package/hooks/git-hooks/test_config.py +0 -15
- package/hooks/git-hooks/test_gate_utils.py +3 -15
- package/hooks/git-hooks/test_pre_commit.py +1 -15
- package/hooks/git-hooks/test_pre_push.py +236 -27
- package/hooks/git-hooks/test_pre_push_base_reference.py +339 -0
- package/hooks/hooks.json +0 -12
- package/hooks/hooks_constants/CLAUDE.md +5 -1
- package/hooks/hooks_constants/bash_pre_tool_use_dispatcher_constants.py +4 -4
- package/hooks/hooks_constants/eli11_reply_enforcer_constants.py +101 -0
- package/hooks/hooks_constants/nas_ssh_binary_enforcer_constants.py +2 -8
- package/hooks/hooks_constants/shell_command_segments.py +82 -0
- package/hooks/hooks_constants/shell_substitution_blocker_constants.py +67 -0
- package/hooks/hooks_constants/stop_dispatcher_constants.py +1 -0
- package/hooks/hooks_constants/test_bash_pre_tool_use_dispatcher_constants.py +5 -6
- package/hooks/hooks_constants/test_stop_dispatcher_constants.py +1 -0
- package/hooks/hooks_constants/unscoped_search_blocker_constants.py +153 -0
- package/package.json +4 -2
- package/rules/CLAUDE.md +17 -23
- package/rules/agent-spawn-protocol.md +6 -6
- package/rules/anti-corollary-tests.md +1 -1
- package/rules/bdd.md +1 -1
- package/rules/cleanup-temp-files.md +10 -4
- package/rules/code-standards.md +7 -0
- package/rules/conservative-action.md +1 -5
- package/rules/context7.md +0 -4
- package/rules/destructive-commands.md +47 -0
- package/rules/doc-inventory-integrity.md +48 -0
- package/rules/doc-prose-cuts.md +58 -0
- package/rules/docstring-prose-matches-implementation.md +10 -2
- package/rules/durable-post-artifacts.md +0 -4
- package/rules/eli11-replies.md +31 -0
- package/rules/explore-thoroughly.md +4 -4
- package/rules/falsify-before-green.md +68 -0
- package/rules/file-global-constants.md +1 -1
- package/rules/filesystem-search.md +51 -0
- package/rules/gh-cli-conventions.md +27 -0
- package/rules/git-workflow.md +26 -0
- package/rules/hedging-claims.md +9 -0
- package/rules/long-horizon-autonomy.md +0 -4
- package/rules/measurement-denominators.md +48 -0
- package/rules/nas-ssh-invocation.md +23 -5
- package/rules/parallel-tools.md +2 -2
- package/rules/plain-illustrative-docstrings.md +3 -7
- package/rules/plain-language.md +2 -0
- package/rules/proof-of-work-pr-comments.md +0 -4
- package/rules/re-stage-before-commit.md +2 -0
- package/rules/research-mode.md +10 -0
- package/rules/shell-invocation.md +21 -0
- package/rules/testing.md +4 -0
- package/rules/verified-commit-gate-skip.md +3 -27
- package/rules/verify-before-asking.md +5 -0
- package/rules/windows-filesystem-safe.md +1 -1
- package/rules/workers-done-before-complete.md +4 -0
- package/scripts/Migrate-ShellPolicy.ps1 +1 -1
- package/scripts/codex_capability_bridge.py +171 -0
- package/scripts/codex_compat_materializer.py +1087 -0
- package/scripts/codex_compat_watcher.py +502 -0
- package/scripts/dev_env_scripts_constants/code_review_constants.py +37 -0
- package/scripts/invoke_code_review.py +11 -4
- package/scripts/sync_to_cursor/rules.py +0 -10
- package/scripts/test_invoke_code_review.py +143 -0
- package/scripts/test_invoke_code_review_chain.py +1 -1
- package/scripts/test_invoke_code_review_contract.py +1 -1
- package/scripts/tests/test_code_review_constants.py +80 -0
- package/scripts/tests/test_codex_capability_bridge.py +91 -0
- package/scripts/tests/test_codex_compat_materializer.py +632 -0
- package/scripts/tests/test_codex_compat_watcher.py +599 -0
- package/scripts/tests/test_sync_to_cursor.py +0 -1
- package/skills/autoconverge/workflow/converge.mjs +1 -1
- package/skills/bugteam/reference/copilot-gap-analysis.md +1 -1
- package/skills/condensing-instructions/SKILL.md +42 -51
- package/skills/fresh-branch/CLAUDE.md +1 -1
- package/skills/fresh-branch/SKILL.md +5 -6
- package/skills/fresh-branch/scripts/create_fresh_branch.py +42 -24
- package/skills/fresh-branch/scripts/fresh_branch_scripts_constants/fresh_branch_cli_constants.py +1 -3
- package/skills/fresh-branch/scripts/test_create_fresh_branch.py +30 -126
- package/skills/orchestrator/SKILL.md +23 -9
- package/skills/orchestrator-refresh/SKILL.md +20 -1
- package/skills/privacy-hygiene/reference/sweep-procedure.md +1 -1
- package/skills/session-log/SKILL.md +1 -1
- package/rules/claude-md-orphan-file.md +0 -28
- package/rules/cleanup-command-forms.md +0 -23
- package/rules/code-reviews.md +0 -11
- package/rules/env-var-table-code-drift.md +0 -10
- package/rules/gh-body-file.md +0 -5
- package/rules/gh-paginate.md +0 -3
- package/rules/hook-prose-matches-detector.md +0 -15
- package/rules/no-historical-clutter.md +0 -26
- package/rules/no-inline-destructive-literals.md +0 -9
- package/rules/no-justification-noise.md +0 -61
- package/rules/package-inventory-stale-entry.md +0 -25
- package/rules/right-sized-engineering.md +0 -28
- package/rules/self-contained-docs.md +0 -17
- package/rules/shell-invocation-policy.md +0 -5
- package/rules/state-what-is.md +0 -25
- package/rules/tdd.md +0 -7
|
@@ -0,0 +1,540 @@
|
|
|
1
|
+
"""Block a commit only on a test failure the staged change itself introduces.
|
|
2
|
+
|
|
3
|
+
::
|
|
4
|
+
|
|
5
|
+
staged run: pkg/test_a.py -> 2 failures (test_x, test_y)
|
|
6
|
+
baseline run (HEAD worktree): pkg/test_a.py -> 1 failure (test_x)
|
|
7
|
+
regression = staged - baseline = {test_y} -> blocks
|
|
8
|
+
test_x was already red before this change -> does not block
|
|
9
|
+
|
|
10
|
+
A staged test group that fails is re-run against the code as it stood before
|
|
11
|
+
the change: a throwaway detached-HEAD worktree is created under the OS temp
|
|
12
|
+
root, the same group runs there with its paths remapped into that tree, and
|
|
13
|
+
the two failure sets are diffed by (classname, name) identity read from each
|
|
14
|
+
run's JUnit XML report. Only a failure absent from the baseline run blocks the
|
|
15
|
+
commit. The user's own working tree and index are never moved, so a crashed
|
|
16
|
+
run leaves every staged edit exactly where it was.
|
|
17
|
+
|
|
18
|
+
Moving the working directory alone would leave an absolute import route — a
|
|
19
|
+
``PYTHONPATH`` entry under the repository root, or an editable install pointing
|
|
20
|
+
at it — still feeding staged code to the baseline, whose identical failure
|
|
21
|
+
would then cancel a real regression out. ``baseline_import_isolation`` moves
|
|
22
|
+
every repository import root into the baseline worktree and reports which
|
|
23
|
+
modules the baseline actually loaded from the user's tree; a baseline that read
|
|
24
|
+
staged code is discarded, and every staged failure in that group blocks.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import subprocess
|
|
30
|
+
import sys
|
|
31
|
+
import tempfile
|
|
32
|
+
import xml.etree.ElementTree as ElementTree
|
|
33
|
+
from dataclasses import dataclass
|
|
34
|
+
from pathlib import Path
|
|
35
|
+
from typing import Iterable
|
|
36
|
+
|
|
37
|
+
from pr_loop_shared_constants.code_rules_gate_constants import (
|
|
38
|
+
ALL_GIT_HEAD_EXISTS_ARGS,
|
|
39
|
+
BASELINE_LEAK_PLUGIN_DIRECTORY_NAME,
|
|
40
|
+
BASELINE_LEAK_REPORT_FILENAME,
|
|
41
|
+
ALL_GIT_WORKTREE_ADD_DETACH_ARGS,
|
|
42
|
+
ALL_GIT_WORKTREE_PRUNE_ARGS,
|
|
43
|
+
ALL_GIT_WORKTREE_REMOVE_FORCE_ARGS,
|
|
44
|
+
GIT_HEAD_REVISION,
|
|
45
|
+
JUNIT_XML_CLASSNAME_ATTRIBUTE,
|
|
46
|
+
JUNIT_XML_ERROR_TAG,
|
|
47
|
+
JUNIT_XML_FAILURE_TAG,
|
|
48
|
+
JUNIT_XML_MISSING_ATTRIBUTE_FALLBACK,
|
|
49
|
+
JUNIT_XML_NAME_ATTRIBUTE,
|
|
50
|
+
JUNIT_XML_TESTCASE_TAG,
|
|
51
|
+
REGRESSION_BASELINE_IMPORT_LEAK_MESSAGE,
|
|
52
|
+
REGRESSION_BASELINE_JUNIT_SUBDIRECTORY_NAME,
|
|
53
|
+
REGRESSION_BASELINE_LEAK_UNREPORTED_MESSAGE,
|
|
54
|
+
REGRESSION_BASELINE_WORKTREE_DIRECTORY_NAME,
|
|
55
|
+
REGRESSION_BASELINE_WORKTREE_TEMP_DIRECTORY_PREFIX,
|
|
56
|
+
REGRESSION_GROUP_FAILURE_MESSAGE,
|
|
57
|
+
REGRESSION_JUNIT_TEMP_DIRECTORY_PREFIX,
|
|
58
|
+
REGRESSION_NO_BASELINE_MESSAGE,
|
|
59
|
+
REGRESSION_PRE_EXISTING_FAILURE_BYPASSED_MESSAGE,
|
|
60
|
+
REGRESSION_STAGED_JUNIT_SUBDIRECTORY_NAME,
|
|
61
|
+
REGRESSION_WORKTREE_ADD_FAILED_MESSAGE,
|
|
62
|
+
REGRESSION_WORKTREE_REMOVE_FAILED_MESSAGE,
|
|
63
|
+
STAGED_TEST_FAILURE_HEADER,
|
|
64
|
+
)
|
|
65
|
+
from terminology_sweep import repository_environment
|
|
66
|
+
|
|
67
|
+
from code_rules_gate_parts import baseline_import_isolation, staged_test_running
|
|
68
|
+
|
|
69
|
+
TestIdentity = tuple[str, str]
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass(frozen=True)
|
|
73
|
+
class GroupOutcome:
|
|
74
|
+
"""One test group's run result.
|
|
75
|
+
|
|
76
|
+
Attributes:
|
|
77
|
+
exit_code: The pytest exit code from the run.
|
|
78
|
+
failing_identities: The (classname, name) of every failed or errored
|
|
79
|
+
testcase, read from that run's JUnit XML report.
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
exit_code: int
|
|
83
|
+
failing_identities: frozenset[TestIdentity]
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _run_git(
|
|
87
|
+
repository_root: Path, all_git_arguments: tuple[str, ...]
|
|
88
|
+
) -> subprocess.CompletedProcess[str]:
|
|
89
|
+
"""Run one git subcommand in *repository_root* with the gate's scrubbed environment."""
|
|
90
|
+
return subprocess.run(
|
|
91
|
+
["git", "-C", str(repository_root), *all_git_arguments],
|
|
92
|
+
capture_output=True,
|
|
93
|
+
text=True,
|
|
94
|
+
check=False,
|
|
95
|
+
env=repository_environment(),
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _head_exists(repository_root: Path) -> bool:
|
|
100
|
+
"""Return True when the repository has a prior commit to compare against."""
|
|
101
|
+
return _run_git(repository_root, ALL_GIT_HEAD_EXISTS_ARGS).returncode == 0
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _junit_failure_identities(junit_xml_dir: Path) -> frozenset[TestIdentity]:
|
|
105
|
+
"""Return the (classname, name) of every failed/errored testcase under a report directory."""
|
|
106
|
+
identities: set[TestIdentity] = set()
|
|
107
|
+
if not junit_xml_dir.is_dir():
|
|
108
|
+
return frozenset(identities)
|
|
109
|
+
for each_report_path in junit_xml_dir.glob("*.xml"):
|
|
110
|
+
try:
|
|
111
|
+
report_root = ElementTree.parse(each_report_path).getroot()
|
|
112
|
+
except ElementTree.ParseError:
|
|
113
|
+
continue
|
|
114
|
+
for each_testcase in report_root.iter(JUNIT_XML_TESTCASE_TAG):
|
|
115
|
+
identities.update(_failing_identity_for_testcase(each_testcase))
|
|
116
|
+
return frozenset(identities)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _failing_identity_for_testcase(testcase: ElementTree.Element) -> list[TestIdentity]:
|
|
120
|
+
"""Return the testcase's (classname, name) as a one-item list when it failed, else empty."""
|
|
121
|
+
has_failed = (
|
|
122
|
+
testcase.find(JUNIT_XML_FAILURE_TAG) is not None
|
|
123
|
+
or testcase.find(JUNIT_XML_ERROR_TAG) is not None
|
|
124
|
+
)
|
|
125
|
+
if not has_failed:
|
|
126
|
+
return []
|
|
127
|
+
return [
|
|
128
|
+
(
|
|
129
|
+
testcase.get(JUNIT_XML_CLASSNAME_ATTRIBUTE, JUNIT_XML_MISSING_ATTRIBUTE_FALLBACK),
|
|
130
|
+
testcase.get(JUNIT_XML_NAME_ATTRIBUTE, JUNIT_XML_MISSING_ATTRIBUTE_FALLBACK),
|
|
131
|
+
)
|
|
132
|
+
]
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _run_group_and_collect(
|
|
136
|
+
group_root: Path,
|
|
137
|
+
all_group_test_paths: list[Path],
|
|
138
|
+
repository_root: Path,
|
|
139
|
+
junit_xml_dir: Path,
|
|
140
|
+
environment: dict[str, str] | None = None,
|
|
141
|
+
) -> GroupOutcome:
|
|
142
|
+
"""Run one test group and return its exit code with its failing test identities.
|
|
143
|
+
|
|
144
|
+
Args:
|
|
145
|
+
group_root: The pytest working directory for this run.
|
|
146
|
+
all_group_test_paths: The collection targets to pass pytest.
|
|
147
|
+
repository_root: The repository root the interpreter resolves against.
|
|
148
|
+
junit_xml_dir: Where this run's JUnit XML reports are written.
|
|
149
|
+
environment: When given, the environment the run uses; the baseline run
|
|
150
|
+
passes one that resolves imports inside the baseline worktree.
|
|
151
|
+
|
|
152
|
+
Returns:
|
|
153
|
+
The run's exit code alongside every failing test identity it reported.
|
|
154
|
+
"""
|
|
155
|
+
junit_xml_dir.mkdir(parents=True, exist_ok=True)
|
|
156
|
+
exit_code = staged_test_running._run_pytest_for_group(
|
|
157
|
+
group_root,
|
|
158
|
+
all_group_test_paths,
|
|
159
|
+
repository_root,
|
|
160
|
+
junit_xml_dir=junit_xml_dir,
|
|
161
|
+
environment=environment,
|
|
162
|
+
)
|
|
163
|
+
return GroupOutcome(exit_code, _junit_failure_identities(junit_xml_dir))
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _existing_group_targets(all_group_test_paths: list[Path]) -> list[Path]:
|
|
167
|
+
"""Return the given test paths that exist on disk.
|
|
168
|
+
|
|
169
|
+
Applied to paths already remapped into the baseline worktree: a test file
|
|
170
|
+
staged for the first time has no counterpart at HEAD, so its remapped path
|
|
171
|
+
is absent from that worktree and drops out of the baseline run entirely —
|
|
172
|
+
every one of its failures is then, correctly, treated as new by the
|
|
173
|
+
staged/baseline set difference.
|
|
174
|
+
"""
|
|
175
|
+
return [each_path for each_path in all_group_test_paths if each_path.is_file()]
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _path_under_baseline_worktree(
|
|
179
|
+
original_path: Path, repository_root: Path, baseline_worktree: Path
|
|
180
|
+
) -> Path:
|
|
181
|
+
"""Return *original_path* rebased onto the baseline worktree.
|
|
182
|
+
|
|
183
|
+
::
|
|
184
|
+
|
|
185
|
+
repository_root = /repo
|
|
186
|
+
baseline_worktree = /tmp/code_rules_gate_baseline_ab/tree
|
|
187
|
+
ok: /repo/pkg_a/test_alpha.py -> /tmp/.../tree/pkg_a/test_alpha.py
|
|
188
|
+
|
|
189
|
+
The baseline run happens inside a separate checkout, so every group root
|
|
190
|
+
and every collection target moves with it.
|
|
191
|
+
|
|
192
|
+
Args:
|
|
193
|
+
original_path: A path under the user's repository root.
|
|
194
|
+
repository_root: The repository root *original_path* is relative to.
|
|
195
|
+
baseline_worktree: The detached-HEAD worktree the baseline run uses.
|
|
196
|
+
|
|
197
|
+
Returns:
|
|
198
|
+
The same repository-relative location inside *baseline_worktree*.
|
|
199
|
+
"""
|
|
200
|
+
repository_relative_path = original_path.resolve().relative_to(repository_root.resolve())
|
|
201
|
+
return baseline_worktree / repository_relative_path
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _baseline_group_targets(
|
|
205
|
+
all_group_test_paths: list[Path], repository_root: Path, baseline_worktree: Path
|
|
206
|
+
) -> list[Path]:
|
|
207
|
+
"""Return the group's collection targets, rebased into the baseline worktree."""
|
|
208
|
+
return _existing_group_targets(
|
|
209
|
+
[
|
|
210
|
+
_path_under_baseline_worktree(each_path, repository_root, baseline_worktree)
|
|
211
|
+
for each_path in all_group_test_paths
|
|
212
|
+
]
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
@dataclass(frozen=True)
|
|
217
|
+
class BaselineRunContext:
|
|
218
|
+
"""What every baseline group run in one gate run shares.
|
|
219
|
+
|
|
220
|
+
Attributes:
|
|
221
|
+
worktree: The detached-HEAD worktree the baseline runs happen in.
|
|
222
|
+
all_import_roots: Every directory the interpreter resolves imported
|
|
223
|
+
packages from, including the roots an editable install registers.
|
|
224
|
+
plugin_directory: Where the import-origin reporting plugin was written.
|
|
225
|
+
staged_environment: The environment the staged runs used.
|
|
226
|
+
"""
|
|
227
|
+
|
|
228
|
+
worktree: Path
|
|
229
|
+
all_import_roots: list[Path]
|
|
230
|
+
plugin_directory: Path
|
|
231
|
+
staged_environment: dict[str, str]
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _baseline_run_context(
|
|
235
|
+
repository_root: Path, baseline_worktree: Path, junit_root: Path
|
|
236
|
+
) -> BaselineRunContext:
|
|
237
|
+
"""Install the leak plugin and probe the interpreter's import roots once per gate run."""
|
|
238
|
+
plugin_directory = junit_root / BASELINE_LEAK_PLUGIN_DIRECTORY_NAME
|
|
239
|
+
baseline_import_isolation.install_leak_plugin(plugin_directory)
|
|
240
|
+
staged_environment = staged_test_running._staged_pytest_environment()
|
|
241
|
+
all_import_roots = baseline_import_isolation.discover_import_roots(
|
|
242
|
+
staged_test_running._resolve_gate_python_executable(repository_root),
|
|
243
|
+
staged_environment,
|
|
244
|
+
baseline_worktree.parent,
|
|
245
|
+
)
|
|
246
|
+
return BaselineRunContext(
|
|
247
|
+
baseline_worktree, all_import_roots, plugin_directory, staged_environment
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _trusted_baseline_outcome(
|
|
252
|
+
group_root: Path, baseline_outcome: GroupOutcome, leak_report_path: Path
|
|
253
|
+
) -> GroupOutcome:
|
|
254
|
+
"""Return the baseline outcome, emptied of failures when the run read the user's own tree.
|
|
255
|
+
|
|
256
|
+
::
|
|
257
|
+
|
|
258
|
+
ok: report [] -> baseline kept; its failures cancel staged ones
|
|
259
|
+
flag: report [/repo/pkg/foo.py] -> baseline discarded; every staged failure blocks
|
|
260
|
+
flag: no report at all -> baseline discarded; nothing was proven
|
|
261
|
+
|
|
262
|
+
A baseline that imported staged code fails the same way the staged run did,
|
|
263
|
+
which would cancel a real regression out. Emptying its failure set makes
|
|
264
|
+
every staged failure count as new, so the commit blocks instead.
|
|
265
|
+
"""
|
|
266
|
+
all_leaked_modules = baseline_import_isolation.modules_imported_from_primary_tree(
|
|
267
|
+
leak_report_path
|
|
268
|
+
)
|
|
269
|
+
if all_leaked_modules is None:
|
|
270
|
+
sys.stderr.write(
|
|
271
|
+
REGRESSION_BASELINE_LEAK_UNREPORTED_MESSAGE.format(group_root=group_root) + "\n"
|
|
272
|
+
)
|
|
273
|
+
return GroupOutcome(baseline_outcome.exit_code, frozenset())
|
|
274
|
+
if not all_leaked_modules:
|
|
275
|
+
return baseline_outcome
|
|
276
|
+
sys.stderr.write(
|
|
277
|
+
REGRESSION_BASELINE_IMPORT_LEAK_MESSAGE.format(
|
|
278
|
+
group_root=group_root,
|
|
279
|
+
count=len(all_leaked_modules),
|
|
280
|
+
first_module=all_leaked_modules[0],
|
|
281
|
+
)
|
|
282
|
+
+ "\n"
|
|
283
|
+
)
|
|
284
|
+
return GroupOutcome(baseline_outcome.exit_code, frozenset())
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def _baseline_group_outcome(
|
|
288
|
+
group_root: Path,
|
|
289
|
+
baseline_targets: list[Path],
|
|
290
|
+
repository_root: Path,
|
|
291
|
+
group_junit_dir: Path,
|
|
292
|
+
context: BaselineRunContext,
|
|
293
|
+
) -> GroupOutcome:
|
|
294
|
+
"""Run one group inside the baseline worktree and keep the result only when it stayed clean."""
|
|
295
|
+
leak_report_path = group_junit_dir / BASELINE_LEAK_REPORT_FILENAME
|
|
296
|
+
baseline_environment = baseline_import_isolation.baseline_pytest_environment(
|
|
297
|
+
context.staged_environment,
|
|
298
|
+
repository_root,
|
|
299
|
+
context.worktree,
|
|
300
|
+
context.all_import_roots,
|
|
301
|
+
context.plugin_directory,
|
|
302
|
+
leak_report_path,
|
|
303
|
+
)
|
|
304
|
+
baseline_outcome = _run_group_and_collect(
|
|
305
|
+
_path_under_baseline_worktree(group_root, repository_root, context.worktree),
|
|
306
|
+
baseline_targets,
|
|
307
|
+
repository_root,
|
|
308
|
+
group_junit_dir,
|
|
309
|
+
environment=baseline_environment,
|
|
310
|
+
)
|
|
311
|
+
return _trusted_baseline_outcome(group_root, baseline_outcome, leak_report_path)
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _baseline_outcomes_for_failing_groups(
|
|
315
|
+
repository_root: Path,
|
|
316
|
+
failing_group_test_paths: dict[Path, list[Path]],
|
|
317
|
+
junit_root: Path,
|
|
318
|
+
baseline_worktree: Path,
|
|
319
|
+
) -> dict[Path, GroupOutcome]:
|
|
320
|
+
"""Run, inside the baseline worktree, only the groups whose staged run failed.
|
|
321
|
+
|
|
322
|
+
Each group's root and collection targets are rebased into
|
|
323
|
+
*baseline_worktree*, and so is every import root that resolves inside the
|
|
324
|
+
user's repository, so an absolute import route never serves staged code to
|
|
325
|
+
the baseline. *repository_root* stays the user's own repository, because
|
|
326
|
+
the interpreter resolves against the project venv that lives there. Every
|
|
327
|
+
outcome stays keyed by the original group root, which is the key the caller
|
|
328
|
+
looks the staged outcome up by.
|
|
329
|
+
"""
|
|
330
|
+
context = _baseline_run_context(repository_root, baseline_worktree, junit_root)
|
|
331
|
+
baseline_outcomes: dict[Path, GroupOutcome] = {}
|
|
332
|
+
for group_index, (group_root, all_group_test_paths) in enumerate(
|
|
333
|
+
sorted(failing_group_test_paths.items())
|
|
334
|
+
):
|
|
335
|
+
baseline_targets = _baseline_group_targets(
|
|
336
|
+
all_group_test_paths, repository_root, baseline_worktree
|
|
337
|
+
)
|
|
338
|
+
if not baseline_targets:
|
|
339
|
+
baseline_outcomes[group_root] = GroupOutcome(0, frozenset())
|
|
340
|
+
continue
|
|
341
|
+
group_junit_dir = (
|
|
342
|
+
junit_root / REGRESSION_BASELINE_JUNIT_SUBDIRECTORY_NAME / str(group_index)
|
|
343
|
+
)
|
|
344
|
+
baseline_outcomes[group_root] = _baseline_group_outcome(
|
|
345
|
+
group_root, baseline_targets, repository_root, group_junit_dir, context
|
|
346
|
+
)
|
|
347
|
+
return baseline_outcomes
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def _report_group_outcome(
|
|
351
|
+
group_root: Path, staged_outcome: GroupOutcome, baseline_failing: frozenset[TestIdentity]
|
|
352
|
+
) -> int:
|
|
353
|
+
"""Compare one group's staged failures against its baseline and report the result.
|
|
354
|
+
|
|
355
|
+
Returns:
|
|
356
|
+
0 when every staged failure was already present at the baseline; the
|
|
357
|
+
staged exit code otherwise.
|
|
358
|
+
"""
|
|
359
|
+
regression_identities = staged_outcome.failing_identities - baseline_failing
|
|
360
|
+
if not regression_identities:
|
|
361
|
+
sys.stderr.write(
|
|
362
|
+
REGRESSION_PRE_EXISTING_FAILURE_BYPASSED_MESSAGE.format(
|
|
363
|
+
group_root=group_root, count=len(staged_outcome.failing_identities)
|
|
364
|
+
)
|
|
365
|
+
+ "\n"
|
|
366
|
+
)
|
|
367
|
+
return 0
|
|
368
|
+
sys.stderr.write(
|
|
369
|
+
REGRESSION_GROUP_FAILURE_MESSAGE.format(
|
|
370
|
+
group_root=group_root, count=len(regression_identities)
|
|
371
|
+
)
|
|
372
|
+
+ "\n"
|
|
373
|
+
)
|
|
374
|
+
return staged_outcome.exit_code
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
def _first_nonzero(all_exit_codes: Iterable[int]) -> int:
|
|
378
|
+
"""Return the first non-zero value in *all_exit_codes*, or 0 when none is non-zero."""
|
|
379
|
+
for each_exit_code in all_exit_codes:
|
|
380
|
+
if each_exit_code != 0:
|
|
381
|
+
return each_exit_code
|
|
382
|
+
return 0
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def _add_baseline_worktree(repository_root: Path, baseline_worktree: Path) -> bool:
|
|
386
|
+
"""Attach a detached-HEAD checkout at *baseline_worktree*; True when git created it."""
|
|
387
|
+
worktree_added = _run_git(
|
|
388
|
+
repository_root,
|
|
389
|
+
(*ALL_GIT_WORKTREE_ADD_DETACH_ARGS, str(baseline_worktree), GIT_HEAD_REVISION),
|
|
390
|
+
)
|
|
391
|
+
return worktree_added.returncode == 0
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def _remove_baseline_worktree(repository_root: Path, baseline_worktree: Path) -> None:
|
|
395
|
+
"""Detach and delete the baseline worktree, pruning its registration when removal fails."""
|
|
396
|
+
worktree_removed = _run_git(
|
|
397
|
+
repository_root, (*ALL_GIT_WORKTREE_REMOVE_FORCE_ARGS, str(baseline_worktree))
|
|
398
|
+
)
|
|
399
|
+
if worktree_removed.returncode == 0:
|
|
400
|
+
return
|
|
401
|
+
_run_git(repository_root, ALL_GIT_WORKTREE_PRUNE_ARGS)
|
|
402
|
+
sys.stderr.write(REGRESSION_WORKTREE_REMOVE_FAILED_MESSAGE + "\n")
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def _score_groups_against_baseline(
|
|
406
|
+
failing_group_test_paths: dict[Path, list[Path]],
|
|
407
|
+
staged_outcomes: dict[Path, GroupOutcome],
|
|
408
|
+
baseline_outcomes: dict[Path, GroupOutcome],
|
|
409
|
+
) -> int:
|
|
410
|
+
"""Report each group's staged-minus-baseline diff and return the first blocking code."""
|
|
411
|
+
return _first_nonzero(
|
|
412
|
+
_report_group_outcome(
|
|
413
|
+
group_root,
|
|
414
|
+
staged_outcomes[group_root],
|
|
415
|
+
baseline_outcomes.get(group_root, GroupOutcome(0, frozenset())).failing_identities,
|
|
416
|
+
)
|
|
417
|
+
for group_root in sorted(failing_group_test_paths)
|
|
418
|
+
)
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def _run_regression_gate(
|
|
422
|
+
repository_root: Path,
|
|
423
|
+
failing_group_test_paths: dict[Path, list[Path]],
|
|
424
|
+
staged_outcomes: dict[Path, GroupOutcome],
|
|
425
|
+
junit_root: Path,
|
|
426
|
+
) -> int:
|
|
427
|
+
"""Re-run the failing groups in a throwaway HEAD worktree and score the diff.
|
|
428
|
+
|
|
429
|
+
The user's own working tree and index stay where they are for the whole
|
|
430
|
+
run: the baseline lives in its own detached checkout under the OS temp
|
|
431
|
+
root, and that checkout is removed before the gate returns.
|
|
432
|
+
"""
|
|
433
|
+
with tempfile.TemporaryDirectory(
|
|
434
|
+
prefix=REGRESSION_BASELINE_WORKTREE_TEMP_DIRECTORY_PREFIX, ignore_cleanup_errors=True
|
|
435
|
+
) as baseline_parent_text:
|
|
436
|
+
baseline_worktree = (
|
|
437
|
+
Path(baseline_parent_text) / REGRESSION_BASELINE_WORKTREE_DIRECTORY_NAME
|
|
438
|
+
)
|
|
439
|
+
if not _add_baseline_worktree(repository_root, baseline_worktree):
|
|
440
|
+
sys.stderr.write(REGRESSION_WORKTREE_ADD_FAILED_MESSAGE + "\n")
|
|
441
|
+
return _first_nonzero(outcome.exit_code for outcome in staged_outcomes.values())
|
|
442
|
+
try:
|
|
443
|
+
baseline_outcomes = _baseline_outcomes_for_failing_groups(
|
|
444
|
+
repository_root, failing_group_test_paths, junit_root, baseline_worktree
|
|
445
|
+
)
|
|
446
|
+
finally:
|
|
447
|
+
_remove_baseline_worktree(repository_root, baseline_worktree)
|
|
448
|
+
return _score_groups_against_baseline(
|
|
449
|
+
failing_group_test_paths, staged_outcomes, baseline_outcomes
|
|
450
|
+
)
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def _run_staged_groups(
|
|
454
|
+
all_tests_by_root: dict[Path, list[Path]], repository_root: Path, junit_root: Path
|
|
455
|
+
) -> dict[Path, GroupOutcome]:
|
|
456
|
+
"""Run every group once under the staged (working-tree) state."""
|
|
457
|
+
staged_outcomes: dict[Path, GroupOutcome] = {}
|
|
458
|
+
for group_index, group_root in enumerate(sorted(all_tests_by_root)):
|
|
459
|
+
group_junit_dir = (
|
|
460
|
+
junit_root / REGRESSION_STAGED_JUNIT_SUBDIRECTORY_NAME / str(group_index)
|
|
461
|
+
)
|
|
462
|
+
staged_outcomes[group_root] = _run_group_and_collect(
|
|
463
|
+
group_root, all_tests_by_root[group_root], repository_root, group_junit_dir
|
|
464
|
+
)
|
|
465
|
+
return staged_outcomes
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def run_grouped_tests_with_regression_gate(
|
|
469
|
+
all_tests_by_root: dict[Path, list[Path]], repository_root: Path
|
|
470
|
+
) -> int:
|
|
471
|
+
"""Run every staged test group and block only on failures the staged change introduces.
|
|
472
|
+
|
|
473
|
+
Every group runs once under the staged state. A group that passes needs no
|
|
474
|
+
further check. A group that fails is re-run against the HEAD baseline (a
|
|
475
|
+
throwaway detached-HEAD worktree under the OS temp root), and only a
|
|
476
|
+
failure absent from that baseline run blocks the commit — a failure
|
|
477
|
+
already present before this change does not.
|
|
478
|
+
|
|
479
|
+
Args:
|
|
480
|
+
all_tests_by_root: Staged test files grouped by owning pytest-config root.
|
|
481
|
+
repository_root: The repository root the staged test files belong to.
|
|
482
|
+
|
|
483
|
+
Returns:
|
|
484
|
+
0 when every group passes, or when every failing group's failures are
|
|
485
|
+
all pre-existing at the baseline. The first group with a genuine
|
|
486
|
+
regression's exit code otherwise.
|
|
487
|
+
"""
|
|
488
|
+
with tempfile.TemporaryDirectory(
|
|
489
|
+
prefix=REGRESSION_JUNIT_TEMP_DIRECTORY_PREFIX
|
|
490
|
+
) as junit_root_text:
|
|
491
|
+
junit_root = Path(junit_root_text)
|
|
492
|
+
staged_outcomes = _run_staged_groups(all_tests_by_root, repository_root, junit_root)
|
|
493
|
+
failing_group_test_paths = {
|
|
494
|
+
group_root: all_tests_by_root[group_root]
|
|
495
|
+
for group_root, outcome in staged_outcomes.items()
|
|
496
|
+
if outcome.exit_code != 0
|
|
497
|
+
}
|
|
498
|
+
if not failing_group_test_paths:
|
|
499
|
+
return 0
|
|
500
|
+
if not _head_exists(repository_root):
|
|
501
|
+
sys.stderr.write(REGRESSION_NO_BASELINE_MESSAGE + "\n")
|
|
502
|
+
first_failing_exit_code = _first_nonzero(
|
|
503
|
+
_report_group_outcome(group_root, staged_outcomes[group_root], frozenset())
|
|
504
|
+
for group_root in sorted(failing_group_test_paths)
|
|
505
|
+
)
|
|
506
|
+
else:
|
|
507
|
+
first_failing_exit_code = _run_regression_gate(
|
|
508
|
+
repository_root, failing_group_test_paths, staged_outcomes, junit_root
|
|
509
|
+
)
|
|
510
|
+
if first_failing_exit_code != 0:
|
|
511
|
+
sys.stderr.write(STAGED_TEST_FAILURE_HEADER + "\n")
|
|
512
|
+
return first_failing_exit_code
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
def run_staged_test_files(repository_root: Path) -> int:
|
|
516
|
+
"""Discover the staged test files and run them under the regression gate.
|
|
517
|
+
|
|
518
|
+
``conftest.py`` files are excluded from collection targets. Pytest still
|
|
519
|
+
loads them automatically when a nearby staged test runs under the same
|
|
520
|
+
owning root. A group whose staged run fails is re-checked against the
|
|
521
|
+
pre-staged baseline: a failure already present before the staged change
|
|
522
|
+
never blocks, only a failure the staged change introduces does.
|
|
523
|
+
|
|
524
|
+
Args:
|
|
525
|
+
repository_root: The repository root the staged test files belong to.
|
|
526
|
+
|
|
527
|
+
Returns:
|
|
528
|
+
0 when no collectable test file is staged, when every group collects no
|
|
529
|
+
tests, when every group passes, or when every failing group's failures
|
|
530
|
+
are all pre-existing at the baseline. The first group with a genuine
|
|
531
|
+
regression's exit code otherwise.
|
|
532
|
+
"""
|
|
533
|
+
all_test_paths = staged_test_running._staged_test_file_paths(repository_root)
|
|
534
|
+
all_pytest_targets = staged_test_running._pytest_target_paths(all_test_paths)
|
|
535
|
+
if not all_pytest_targets:
|
|
536
|
+
return 0
|
|
537
|
+
all_tests_by_root = staged_test_running._group_staged_tests_by_root(
|
|
538
|
+
all_pytest_targets, repository_root
|
|
539
|
+
)
|
|
540
|
+
return run_grouped_tests_with_regression_gate(all_tests_by_root, repository_root)
|