patchahead 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- patchahead/__init__.py +8 -0
- patchahead/analysis/__init__.py +52 -0
- patchahead/analysis/edits.py +143 -0
- patchahead/analysis/index.py +203 -0
- patchahead/analysis/python_ast.py +457 -0
- patchahead/apidiff/__init__.py +23 -0
- patchahead/apidiff/compare.py +366 -0
- patchahead/apidiff/download.py +95 -0
- patchahead/apidiff/surface.py +337 -0
- patchahead/ci.py +301 -0
- patchahead/cli.py +627 -0
- patchahead/config.py +284 -0
- patchahead/demo/__init__.py +256 -0
- patchahead/demo/fixtures/changes/field-rename.md +14 -0
- patchahead/demo/fixtures/changes/invoice-field-rename.md +21 -0
- patchahead/demo/fixtures/changes/kwarg-rename.md +14 -0
- patchahead/demo/fixtures/changes/method-rename.md +12 -0
- patchahead/demo/fixtures/changes/pagination-cursor.json +24 -0
- patchahead/demo/fixtures/changes/pagination-cursor.md +20 -0
- patchahead/demo/fixtures/changes/sdk-v2.md +31 -0
- patchahead/demo/fixtures/orders-service/README.md +51 -0
- patchahead/demo/fixtures/orders-service/app/__init__.py +0 -0
- patchahead/demo/fixtures/orders-service/app/client.py +15 -0
- patchahead/demo/fixtures/orders-service/app/models.py +10 -0
- patchahead/demo/fixtures/orders-service/app/order_report.py +24 -0
- patchahead/demo/fixtures/orders-service/app/order_sync.py +21 -0
- patchahead/demo/fixtures/orders-service/conftest.py +6 -0
- patchahead/demo/fixtures/orders-service/pyproject.toml +16 -0
- patchahead/demo/fixtures/orders-service/tests/test_client.py +14 -0
- patchahead/demo/fixtures/orders-service/tests/test_order_report.py +24 -0
- patchahead/demo/fixtures/orders-service/tests/test_order_sync.py +11 -0
- patchahead/demo/fixtures/orders-service/upstream/__init__.py +0 -0
- patchahead/demo/fixtures/orders-service/upstream/api_v1.py +34 -0
- patchahead/demo/fixtures/orders-service/upstream/api_v2.py +56 -0
- patchahead/demo/serve.py +189 -0
- patchahead/domain/__init__.py +67 -0
- patchahead/domain/change.py +269 -0
- patchahead/domain/completeness.py +91 -0
- patchahead/domain/impact.py +248 -0
- patchahead/domain/patch.py +81 -0
- patchahead/domain/plan.py +170 -0
- patchahead/domain/result.py +210 -0
- patchahead/domain/validation.py +200 -0
- patchahead/engine.py +609 -0
- patchahead/handlers/__init__.py +35 -0
- patchahead/handlers/base.py +211 -0
- patchahead/handlers/field_rename.py +425 -0
- patchahead/handlers/kwarg_rename.py +201 -0
- patchahead/handlers/method_rename.py +608 -0
- patchahead/handlers/pagination.py +582 -0
- patchahead/ingest/__init__.py +32 -0
- patchahead/ingest/base.py +102 -0
- patchahead/ingest/markdown.py +1138 -0
- patchahead/ingest/structured.py +218 -0
- patchahead/llm/__init__.py +28 -0
- patchahead/llm/client.py +152 -0
- patchahead/llm/proposer.py +620 -0
- patchahead/observability.py +223 -0
- patchahead/reporting.py +451 -0
- patchahead/testing/__init__.py +22 -0
- patchahead/testing/discovery.py +113 -0
- patchahead/testing/runner.py +138 -0
- patchahead/validation/__init__.py +5 -0
- patchahead/validation/completeness.py +265 -0
- patchahead/validation/engine.py +531 -0
- patchahead/web/__init__.py +13 -0
- patchahead/web/server.py +279 -0
- patchahead/web/static/index.html +650 -0
- patchahead/workspace.py +382 -0
- patchahead-0.3.0.dist-info/METADATA +368 -0
- patchahead-0.3.0.dist-info/RECORD +75 -0
- patchahead-0.3.0.dist-info/WHEEL +5 -0
- patchahead-0.3.0.dist-info/entry_points.txt +2 -0
- patchahead-0.3.0.dist-info/licenses/LICENSE +21 -0
- patchahead-0.3.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,531 @@
|
|
|
1
|
+
"""The validation engine: five gates, in order, producing a structured verdict.
|
|
2
|
+
|
|
3
|
+
This is the subsystem that decides whether a migration worked. Nothing else is
|
|
4
|
+
permitted to: ``MigrationResult.succeeded`` is defined as "validation passed".
|
|
5
|
+
|
|
6
|
+
Gate order is deliberate. Syntax and scope are near-instant and decisive, and
|
|
7
|
+
they run *before* anything executes repository code -- so a proposal that
|
|
8
|
+
produced invalid Python or touched 40 unrelated files never gets as far as
|
|
9
|
+
running a test command.
|
|
10
|
+
|
|
11
|
+
=== ==================== ======================================================
|
|
12
|
+
# Gate Fails when
|
|
13
|
+
=== ==================== ======================================================
|
|
14
|
+
1 ``syntax`` A modified file no longer parses as Python.
|
|
15
|
+
2 ``scope`` Files outside the plan changed, or the change is
|
|
16
|
+
larger than the configured limits.
|
|
17
|
+
3 ``targeted_tests`` The tests mapped to the changed modules fail.
|
|
18
|
+
4 ``regression_tests`` The patch broke a test that passed before it.
|
|
19
|
+
5 ``migration_assertion`` Never. It reports evidence, not breakage.
|
|
20
|
+
=== ==================== ======================================================
|
|
21
|
+
|
|
22
|
+
Gate 5 is the one that distinguishes a migration from a no-op. A patch can leave
|
|
23
|
+
a green suite green without having fixed anything; this gate asserts that the
|
|
24
|
+
specific breakage the change describes was real before the patch and gone after.
|
|
25
|
+
|
|
26
|
+
It is the one gate that cannot fail. Its question is "is there red-to-green
|
|
27
|
+
evidence", so the only answers are PASSED and SKIPPED-because-there-is-none: a
|
|
28
|
+
suite that was already green, a runner that never started, a repository whose
|
|
29
|
+
remaining failures were failing before the patch too. Evidence of *breakage* is
|
|
30
|
+
gate 4's to report, and a patch that broke something must be reported once, by
|
|
31
|
+
the gate that measured it, rather than twice under two explanations that do not
|
|
32
|
+
agree with each other.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
from __future__ import annotations
|
|
36
|
+
|
|
37
|
+
import logging
|
|
38
|
+
import time
|
|
39
|
+
from dataclasses import dataclass, field
|
|
40
|
+
|
|
41
|
+
from patchahead.analysis import edits as edit_utils
|
|
42
|
+
from patchahead.analysis.index import is_test_path
|
|
43
|
+
from patchahead.config import Config
|
|
44
|
+
from patchahead.domain.patch import PatchProposal
|
|
45
|
+
from patchahead.domain.validation import (
|
|
46
|
+
GateName,
|
|
47
|
+
GateResult,
|
|
48
|
+
GateStatus,
|
|
49
|
+
TestRun,
|
|
50
|
+
ValidationResult,
|
|
51
|
+
)
|
|
52
|
+
from patchahead.testing import discovery, runner
|
|
53
|
+
from patchahead.workspace import Workspace
|
|
54
|
+
|
|
55
|
+
log = logging.getLogger(__name__)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass
|
|
59
|
+
class ValidationOptions:
|
|
60
|
+
"""What the caller wants validated."""
|
|
61
|
+
|
|
62
|
+
#: Run gates that execute repository code. ``--no-tests`` sets this False.
|
|
63
|
+
run_tests: bool = True
|
|
64
|
+
#: Targeted-test result from before patching, for the migration assertion.
|
|
65
|
+
baseline: TestRun | None = None
|
|
66
|
+
#: Full-suite result from before patching, so the regression gate can tell
|
|
67
|
+
#: a test this patch broke from one that was already failing.
|
|
68
|
+
full_baseline: TestRun | None = None
|
|
69
|
+
#: Test command override; defaults to the repository's configured command.
|
|
70
|
+
test_command: str = ""
|
|
71
|
+
#: Files already modified in the workspace by an earlier change in the same
|
|
72
|
+
#: run. The scope gate judges a proposal by what *it* changed, not by what
|
|
73
|
+
#: the workspace has accumulated.
|
|
74
|
+
preexisting_changes: set[str] = field(default_factory=set)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class ValidationEngine:
|
|
78
|
+
"""Runs the gates against a patched workspace."""
|
|
79
|
+
|
|
80
|
+
def __init__(self, config: Config) -> None:
|
|
81
|
+
self.config = config
|
|
82
|
+
|
|
83
|
+
def validate(
|
|
84
|
+
self,
|
|
85
|
+
proposal: PatchProposal,
|
|
86
|
+
workspace: Workspace,
|
|
87
|
+
options: ValidationOptions | None = None,
|
|
88
|
+
) -> ValidationResult:
|
|
89
|
+
options = options or ValidationOptions()
|
|
90
|
+
result = ValidationResult()
|
|
91
|
+
|
|
92
|
+
result.gates.append(self._syntax_gate(proposal, workspace))
|
|
93
|
+
result.gates.append(self._scope_gate(proposal, workspace, options.preexisting_changes))
|
|
94
|
+
|
|
95
|
+
# Do not execute repository code if the patch is already known bad.
|
|
96
|
+
if any(gate.failed for gate in result.gates):
|
|
97
|
+
reason = "skipped because an earlier gate failed"
|
|
98
|
+
for name in (
|
|
99
|
+
GateName.TARGETED_TESTS,
|
|
100
|
+
GateName.REGRESSION_TESTS,
|
|
101
|
+
GateName.MIGRATION_ASSERTION,
|
|
102
|
+
):
|
|
103
|
+
result.gates.append(GateResult(name=name, status=GateStatus.SKIPPED, detail=reason))
|
|
104
|
+
return result
|
|
105
|
+
|
|
106
|
+
if not options.run_tests:
|
|
107
|
+
reason = "skipped: test execution was disabled (--no-tests)"
|
|
108
|
+
for name in (
|
|
109
|
+
GateName.TARGETED_TESTS,
|
|
110
|
+
GateName.REGRESSION_TESTS,
|
|
111
|
+
GateName.MIGRATION_ASSERTION,
|
|
112
|
+
):
|
|
113
|
+
result.gates.append(GateResult(name=name, status=GateStatus.SKIPPED, detail=reason))
|
|
114
|
+
return result
|
|
115
|
+
|
|
116
|
+
command = options.test_command or self.config.test_command
|
|
117
|
+
targeted = self._targeted_gate(proposal, workspace, command)
|
|
118
|
+
result.gates.append(targeted)
|
|
119
|
+
regression = self._regression_gate(workspace, command, options.full_baseline)
|
|
120
|
+
result.gates.append(regression)
|
|
121
|
+
edited_tests = {path for path in proposal.changed_files if is_test_path(path)}
|
|
122
|
+
result.gates.append(
|
|
123
|
+
self._assertion_gate(
|
|
124
|
+
targeted, regression, options.baseline, options.full_baseline, edited_tests
|
|
125
|
+
)
|
|
126
|
+
)
|
|
127
|
+
return result
|
|
128
|
+
|
|
129
|
+
# -- gate 1: syntax ----------------------------------------------------
|
|
130
|
+
|
|
131
|
+
def _syntax_gate(self, proposal: PatchProposal, workspace: Workspace) -> GateResult:
|
|
132
|
+
start = time.perf_counter()
|
|
133
|
+
broken: list[str] = []
|
|
134
|
+
for file_edit in proposal.files:
|
|
135
|
+
if not file_edit.changed:
|
|
136
|
+
continue
|
|
137
|
+
ok, error = edit_utils.is_parseable(file_edit.new_source, file_edit.path)
|
|
138
|
+
if not ok:
|
|
139
|
+
broken.append(error)
|
|
140
|
+
|
|
141
|
+
duration = int((time.perf_counter() - start) * 1000)
|
|
142
|
+
if broken:
|
|
143
|
+
return GateResult(
|
|
144
|
+
name=GateName.SYNTAX,
|
|
145
|
+
status=GateStatus.FAILED,
|
|
146
|
+
detail="patched file(s) no longer parse: " + "; ".join(broken),
|
|
147
|
+
duration_ms=duration,
|
|
148
|
+
)
|
|
149
|
+
changed = len(proposal.changed_files)
|
|
150
|
+
if changed == 0:
|
|
151
|
+
return GateResult(
|
|
152
|
+
name=GateName.SYNTAX,
|
|
153
|
+
status=GateStatus.SKIPPED,
|
|
154
|
+
detail="no files were modified",
|
|
155
|
+
duration_ms=duration,
|
|
156
|
+
)
|
|
157
|
+
return GateResult(
|
|
158
|
+
name=GateName.SYNTAX,
|
|
159
|
+
status=GateStatus.PASSED,
|
|
160
|
+
detail=f"{changed} modified file(s) parse as valid Python",
|
|
161
|
+
duration_ms=duration,
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
# -- gate 2: scope -----------------------------------------------------
|
|
165
|
+
|
|
166
|
+
def _scope_gate(
|
|
167
|
+
self,
|
|
168
|
+
proposal: PatchProposal,
|
|
169
|
+
workspace: Workspace,
|
|
170
|
+
preexisting: set[str] | None = None,
|
|
171
|
+
) -> GateResult:
|
|
172
|
+
"""Nothing outside the plan changed, and the change is not oversized.
|
|
173
|
+
|
|
174
|
+
This is the gate that contains an LLM. A model asked to fix one function
|
|
175
|
+
can reformat a file, "improve" a neighbour, or rewrite an import block;
|
|
176
|
+
the plan names exactly which files were supposed to change, and anything
|
|
177
|
+
else is a failure regardless of how good the diff looks.
|
|
178
|
+
"""
|
|
179
|
+
start = time.perf_counter()
|
|
180
|
+
planned = set(proposal.plan.target_files)
|
|
181
|
+
actual = set(workspace.changed_files()) - set(preexisting or ())
|
|
182
|
+
unexpected = sorted(actual - planned)
|
|
183
|
+
duration = int((time.perf_counter() - start) * 1000)
|
|
184
|
+
|
|
185
|
+
if not actual:
|
|
186
|
+
# Nothing changed, so there is nothing to judge the scope of. This
|
|
187
|
+
# must not read as a pass: a proposal that modified no files has not
|
|
188
|
+
# migrated anything, and with every other gate skipped a PASSED here
|
|
189
|
+
# would make an empty patch look validated.
|
|
190
|
+
return GateResult(
|
|
191
|
+
name=GateName.SCOPE,
|
|
192
|
+
status=GateStatus.SKIPPED,
|
|
193
|
+
detail="no files were modified",
|
|
194
|
+
duration_ms=duration,
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
if unexpected:
|
|
198
|
+
return GateResult(
|
|
199
|
+
name=GateName.SCOPE,
|
|
200
|
+
status=GateStatus.FAILED,
|
|
201
|
+
detail=(
|
|
202
|
+
f"{len(unexpected)} file(s) changed that the plan did not name: "
|
|
203
|
+
+ ", ".join(unexpected[:5])
|
|
204
|
+
+ (" ..." if len(unexpected) > 5 else "")
|
|
205
|
+
),
|
|
206
|
+
duration_ms=duration,
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
limit = self.config.max_changed_files
|
|
210
|
+
if limit and len(actual) > limit:
|
|
211
|
+
return GateResult(
|
|
212
|
+
name=GateName.SCOPE,
|
|
213
|
+
status=GateStatus.FAILED,
|
|
214
|
+
detail=(
|
|
215
|
+
f"{len(actual)} files changed, above the `max_changed_files` limit of {limit}"
|
|
216
|
+
),
|
|
217
|
+
duration_ms=duration,
|
|
218
|
+
)
|
|
219
|
+
|
|
220
|
+
diff_lines = proposal.diff_line_count
|
|
221
|
+
diff_limit = self.config.max_diff_lines
|
|
222
|
+
if diff_limit and diff_lines > diff_limit:
|
|
223
|
+
return GateResult(
|
|
224
|
+
name=GateName.SCOPE,
|
|
225
|
+
status=GateStatus.FAILED,
|
|
226
|
+
detail=(
|
|
227
|
+
f"the diff changes {diff_lines} lines, above the "
|
|
228
|
+
f"`max_diff_lines` limit of {diff_limit}"
|
|
229
|
+
),
|
|
230
|
+
duration_ms=duration,
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
return GateResult(
|
|
234
|
+
name=GateName.SCOPE,
|
|
235
|
+
status=GateStatus.PASSED,
|
|
236
|
+
detail=(
|
|
237
|
+
f"{len(actual)} file(s) changed, all named by the plan; {diff_lines} diff line(s)"
|
|
238
|
+
),
|
|
239
|
+
duration_ms=duration,
|
|
240
|
+
)
|
|
241
|
+
|
|
242
|
+
# -- gate 3: targeted tests -------------------------------------------
|
|
243
|
+
|
|
244
|
+
def _targeted_gate(
|
|
245
|
+
self, proposal: PatchProposal, workspace: Workspace, command: str
|
|
246
|
+
) -> GateResult:
|
|
247
|
+
expected = proposal.plan.expected_tests
|
|
248
|
+
if not expected:
|
|
249
|
+
return GateResult(
|
|
250
|
+
name=GateName.TARGETED_TESTS,
|
|
251
|
+
status=GateStatus.SKIPPED,
|
|
252
|
+
detail=(
|
|
253
|
+
"no tests could be mapped to the changed modules; the "
|
|
254
|
+
"regression gate runs the full suite instead"
|
|
255
|
+
),
|
|
256
|
+
)
|
|
257
|
+
|
|
258
|
+
scoped = discovery.scoped_command(command, expected)
|
|
259
|
+
if scoped == command:
|
|
260
|
+
return GateResult(
|
|
261
|
+
name=GateName.TARGETED_TESTS,
|
|
262
|
+
status=GateStatus.SKIPPED,
|
|
263
|
+
detail=(
|
|
264
|
+
f"the configured test command (`{command}`) cannot be narrowed "
|
|
265
|
+
f"to specific files; the regression gate covers these tests"
|
|
266
|
+
),
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
run = runner.run_tests(workspace, scoped, timeout=self.config.test_timeout_seconds)
|
|
270
|
+
if run.errored:
|
|
271
|
+
# The command could not start, or collected nothing. That is "could
|
|
272
|
+
# not verify", not "verified and failed", and both test gates have to
|
|
273
|
+
# say so in the same words: a runner that is missing is one fact, and
|
|
274
|
+
# a gate that called it a failure while the other called it a skip
|
|
275
|
+
# was how the same fact ended up reported as a code regression. The
|
|
276
|
+
# run ends as `patched_unverified` rather than `migrated`, because no
|
|
277
|
+
# test gate actually ran.
|
|
278
|
+
return GateResult(
|
|
279
|
+
name=GateName.TARGETED_TESTS,
|
|
280
|
+
status=GateStatus.SKIPPED,
|
|
281
|
+
detail=f"the targeted tests did not run: {run.summary}",
|
|
282
|
+
duration_ms=run.duration_ms,
|
|
283
|
+
test_run=run,
|
|
284
|
+
)
|
|
285
|
+
return GateResult(
|
|
286
|
+
name=GateName.TARGETED_TESTS,
|
|
287
|
+
status=GateStatus.PASSED if run.passed else GateStatus.FAILED,
|
|
288
|
+
detail=f"{', '.join(expected)}: {run.summary}",
|
|
289
|
+
duration_ms=run.duration_ms,
|
|
290
|
+
test_run=run,
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
# -- gate 4: regression -----------------------------------------------
|
|
294
|
+
|
|
295
|
+
def _regression_gate(
|
|
296
|
+
self, workspace: Workspace, command: str, baseline: TestRun | None
|
|
297
|
+
) -> GateResult:
|
|
298
|
+
"""Did this patch break anything that was working?
|
|
299
|
+
|
|
300
|
+
"Regression" means *newly* failing, so the gate compares the failing set
|
|
301
|
+
against a full-suite run from before the patch. A test that was already
|
|
302
|
+
red stays red without failing this gate -- a repository broken by three
|
|
303
|
+
upstream changes must still be able to migrate the first one. Only tests
|
|
304
|
+
this patch turned from passing to failing count.
|
|
305
|
+
|
|
306
|
+
Without a baseline the gate falls back to requiring a fully green suite,
|
|
307
|
+
which is the only safe reading when there is nothing to compare to.
|
|
308
|
+
"""
|
|
309
|
+
run = runner.run_tests(workspace, command, timeout=self.config.test_timeout_seconds)
|
|
310
|
+
if run.errored:
|
|
311
|
+
return GateResult(
|
|
312
|
+
name=GateName.REGRESSION_TESTS,
|
|
313
|
+
status=GateStatus.SKIPPED,
|
|
314
|
+
detail=f"the full suite did not run: {run.summary}",
|
|
315
|
+
duration_ms=run.duration_ms,
|
|
316
|
+
test_run=run,
|
|
317
|
+
)
|
|
318
|
+
if run.passed:
|
|
319
|
+
return GateResult(
|
|
320
|
+
name=GateName.REGRESSION_TESTS,
|
|
321
|
+
status=GateStatus.PASSED,
|
|
322
|
+
detail=f"`{command}`: {run.summary}",
|
|
323
|
+
duration_ms=run.duration_ms,
|
|
324
|
+
test_run=run,
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
if baseline is None or baseline.errored:
|
|
328
|
+
return GateResult(
|
|
329
|
+
name=GateName.REGRESSION_TESTS,
|
|
330
|
+
status=GateStatus.FAILED,
|
|
331
|
+
detail=(
|
|
332
|
+
f"`{command}`: {run.summary} (no pre-patch baseline was available, "
|
|
333
|
+
f"so any failure is treated as a regression)"
|
|
334
|
+
),
|
|
335
|
+
duration_ms=run.duration_ms,
|
|
336
|
+
test_run=run,
|
|
337
|
+
)
|
|
338
|
+
|
|
339
|
+
already_failing = set(baseline.failing_tests)
|
|
340
|
+
now_failing = set(run.failing_tests)
|
|
341
|
+
new_failures = sorted(now_failing - already_failing)
|
|
342
|
+
|
|
343
|
+
if new_failures:
|
|
344
|
+
return GateResult(
|
|
345
|
+
name=GateName.REGRESSION_TESTS,
|
|
346
|
+
status=GateStatus.FAILED,
|
|
347
|
+
detail=(
|
|
348
|
+
f"the patch broke {len(new_failures)} test(s) that passed before: "
|
|
349
|
+
+ ", ".join(new_failures[:5])
|
|
350
|
+
+ (" ..." if len(new_failures) > 5 else "")
|
|
351
|
+
),
|
|
352
|
+
duration_ms=run.duration_ms,
|
|
353
|
+
test_run=run,
|
|
354
|
+
)
|
|
355
|
+
|
|
356
|
+
if not baseline.failing_tests and not run.failing_tests:
|
|
357
|
+
# Both runs failed without naming tests -- a collection or import
|
|
358
|
+
# error, say. Not something to wave through as "pre-existing".
|
|
359
|
+
return GateResult(
|
|
360
|
+
name=GateName.REGRESSION_TESTS,
|
|
361
|
+
status=GateStatus.FAILED,
|
|
362
|
+
detail=(
|
|
363
|
+
f"`{command}`: {run.summary}; the failure names no specific test, "
|
|
364
|
+
f"so it cannot be attributed to pre-existing breakage"
|
|
365
|
+
),
|
|
366
|
+
duration_ms=run.duration_ms,
|
|
367
|
+
test_run=run,
|
|
368
|
+
)
|
|
369
|
+
|
|
370
|
+
fixed = sorted(already_failing - now_failing)
|
|
371
|
+
return GateResult(
|
|
372
|
+
name=GateName.REGRESSION_TESTS,
|
|
373
|
+
status=GateStatus.PASSED,
|
|
374
|
+
detail=(
|
|
375
|
+
f"no new failures. {len(now_failing)} test(s) were already failing "
|
|
376
|
+
f"before this patch (unrelated upstream breakage)"
|
|
377
|
+
+ (f"; this patch fixed {len(fixed)}" if fixed else "")
|
|
378
|
+
),
|
|
379
|
+
duration_ms=run.duration_ms,
|
|
380
|
+
test_run=run,
|
|
381
|
+
)
|
|
382
|
+
|
|
383
|
+
# -- gate 5: migration assertion --------------------------------------
|
|
384
|
+
|
|
385
|
+
def _assertion_gate(
|
|
386
|
+
self,
|
|
387
|
+
targeted: GateResult,
|
|
388
|
+
regression: GateResult,
|
|
389
|
+
baseline: TestRun | None,
|
|
390
|
+
full_baseline: TestRun | None,
|
|
391
|
+
edited_tests: set[str] | None = None,
|
|
392
|
+
) -> GateResult:
|
|
393
|
+
"""Did the specific breakage actually get fixed?
|
|
394
|
+
|
|
395
|
+
Requires evidence on both sides: a test failed before the patch, and
|
|
396
|
+
passes after it. This is the gate that separates "we changed some code"
|
|
397
|
+
from "we migrated something", and
|
|
398
|
+
:attr:`~patchahead.domain.validation.ValidationResult.verified` is
|
|
399
|
+
defined in terms of it.
|
|
400
|
+
|
|
401
|
+
It prefers the targeted gate's evidence and falls back to the full
|
|
402
|
+
suite, because a repository whose test command cannot be narrowed to
|
|
403
|
+
specific files still produces perfectly good red-to-green evidence --
|
|
404
|
+
it is just spread across the whole run. A scope is only usable when both
|
|
405
|
+
of its runs completed; one that could not run is skipped over, and if no
|
|
406
|
+
scope is usable the gate says which ones were missing and why.
|
|
407
|
+
"""
|
|
408
|
+
reasons: list[str] = []
|
|
409
|
+
for scope, before, gate in (
|
|
410
|
+
("targeted", baseline, targeted),
|
|
411
|
+
("full suite", full_baseline, regression),
|
|
412
|
+
):
|
|
413
|
+
after = gate.test_run
|
|
414
|
+
if after is None and gate.status is GateStatus.SKIPPED:
|
|
415
|
+
# The gate declined to run at all (no mapped tests, a command
|
|
416
|
+
# that cannot be narrowed). Not a problem worth reporting here;
|
|
417
|
+
# the other scope may still carry the evidence.
|
|
418
|
+
continue
|
|
419
|
+
if after is None or after.errored:
|
|
420
|
+
reasons.append(
|
|
421
|
+
f"the post-patch {scope} run did not complete"
|
|
422
|
+
+ (f": {after.summary}" if after is not None else "")
|
|
423
|
+
)
|
|
424
|
+
continue
|
|
425
|
+
if before is None:
|
|
426
|
+
reasons.append(f"no pre-patch {scope} run was recorded")
|
|
427
|
+
continue
|
|
428
|
+
if before.errored:
|
|
429
|
+
reasons.append(f"the pre-patch {scope} run did not complete: {before.summary}")
|
|
430
|
+
continue
|
|
431
|
+
return self._compare_runs(scope, before, after, edited_tests or set())
|
|
432
|
+
|
|
433
|
+
detail = "; ".join(dict.fromkeys(reasons)) or "no before/after test evidence is available"
|
|
434
|
+
return GateResult(
|
|
435
|
+
name=GateName.MIGRATION_ASSERTION,
|
|
436
|
+
status=GateStatus.SKIPPED,
|
|
437
|
+
detail=f"{detail}, so this patch is unverified",
|
|
438
|
+
)
|
|
439
|
+
|
|
440
|
+
@staticmethod
|
|
441
|
+
def _compare_runs(
|
|
442
|
+
scope: str, before: TestRun, after: TestRun, edited_tests: set[str] | None = None
|
|
443
|
+
) -> GateResult:
|
|
444
|
+
"""Decide what one usable pair of runs evidences.
|
|
445
|
+
|
|
446
|
+
A test in a file this patch edited does not count. If the patch
|
|
447
|
+
rewrote a test, that test going green shows the rewrite is consistent
|
|
448
|
+
with itself, not that the migration works -- it cannot vouch for the
|
|
449
|
+
change it is part of. Only tests the patch left alone are evidence.
|
|
450
|
+
|
|
451
|
+
Exactly one shape is a pass: something that was failing before the patch
|
|
452
|
+
passes after it. Everything else is SKIPPED, never FAILED, because this
|
|
453
|
+
gate asks "is there migration evidence" and the answer "no" is an
|
|
454
|
+
absence of proof rather than proof of breakage. A patch that genuinely
|
|
455
|
+
broke something is the regression gate's verdict to give; giving it
|
|
456
|
+
again here, under a different explanation, is how two gates end up
|
|
457
|
+
contradicting each other about the same test run.
|
|
458
|
+
"""
|
|
459
|
+
|
|
460
|
+
def verdict(status: GateStatus, detail: str) -> GateResult:
|
|
461
|
+
return GateResult(name=GateName.MIGRATION_ASSERTION, status=status, detail=detail)
|
|
462
|
+
|
|
463
|
+
if before.passed:
|
|
464
|
+
return verdict(
|
|
465
|
+
GateStatus.SKIPPED,
|
|
466
|
+
f"the {scope} tests already passed before the patch, so this run "
|
|
467
|
+
f"cannot evidence that the migration fixed anything. The patch "
|
|
468
|
+
f"may still be correct; these tests do not cover the change.",
|
|
469
|
+
)
|
|
470
|
+
|
|
471
|
+
edited = edited_tests or set()
|
|
472
|
+
fixed = sorted(set(before.failing_tests) - set(after.failing_tests))
|
|
473
|
+
broken = sorted(set(after.failing_tests) - set(before.failing_tests))
|
|
474
|
+
if edited:
|
|
475
|
+
vouching = [test for test in fixed if test.split("::", 1)[0] not in edited]
|
|
476
|
+
excluded = sorted(set(fixed) - set(vouching))
|
|
477
|
+
if not vouching:
|
|
478
|
+
shown = ", ".join(sorted(edited)[:3])
|
|
479
|
+
why = (
|
|
480
|
+
f"the only {scope} tests that went from failing to passing are in "
|
|
481
|
+
f"files this patch edited ({shown})"
|
|
482
|
+
if excluded
|
|
483
|
+
else f"this patch edited test files ({shown}), and the {scope} run "
|
|
484
|
+
f"does not say which tests went from failing to passing"
|
|
485
|
+
)
|
|
486
|
+
return verdict(
|
|
487
|
+
GateStatus.SKIPPED,
|
|
488
|
+
f"{why}; a test the patch rewrote cannot vouch for the rewrite, "
|
|
489
|
+
f"so this patch is unverified",
|
|
490
|
+
)
|
|
491
|
+
fixed = vouching
|
|
492
|
+
named = ", ".join(fixed[:3]) + (" ..." if len(fixed) > 3 else "")
|
|
493
|
+
|
|
494
|
+
if after.passed:
|
|
495
|
+
# Red before, green after. Whatever was failing passes now, named or
|
|
496
|
+
# not -- which is what makes this branch the one that works for a
|
|
497
|
+
# runner whose output PatchAhead cannot parse test names out of.
|
|
498
|
+
return verdict(
|
|
499
|
+
GateStatus.PASSED,
|
|
500
|
+
f"the {scope} tests failed before the patch and pass after it "
|
|
501
|
+
f"({named or before.summary})",
|
|
502
|
+
)
|
|
503
|
+
|
|
504
|
+
if not fixed:
|
|
505
|
+
# Still red, and nothing that was red is green. The patch may be
|
|
506
|
+
# incomplete or simply irrelevant to this failure; either way there
|
|
507
|
+
# is nothing here to verify a migration with.
|
|
508
|
+
return verdict(
|
|
509
|
+
GateStatus.SKIPPED,
|
|
510
|
+
f"the {scope} tests still fail after the patch and none of the "
|
|
511
|
+
f"tests that were failing before it pass now, so this run does "
|
|
512
|
+
f"not evidence a migration",
|
|
513
|
+
)
|
|
514
|
+
|
|
515
|
+
if broken:
|
|
516
|
+
# Some red went green and some green went red. The repair is real
|
|
517
|
+
# but so is the damage, and calling this verified would let the
|
|
518
|
+
# headline read "migrated" over a regression.
|
|
519
|
+
return verdict(
|
|
520
|
+
GateStatus.SKIPPED,
|
|
521
|
+
f"the patch repaired {len(fixed)} previously failing {scope} "
|
|
522
|
+
f"test(s) ({named}) but {len(broken)} test(s) that passed before "
|
|
523
|
+
f"it now fail; the regression gate reports those",
|
|
524
|
+
)
|
|
525
|
+
|
|
526
|
+
return verdict(
|
|
527
|
+
GateStatus.PASSED,
|
|
528
|
+
f"{len(fixed)} {scope} test(s) that failed before the patch now pass "
|
|
529
|
+
f"({named}); the {len(after.failing_tests)} still failing were "
|
|
530
|
+
f"already failing before it",
|
|
531
|
+
)
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""The optional web UI: a view over the engine, served on localhost.
|
|
2
|
+
|
|
3
|
+
Lives inside the package rather than beside it so that the HTML ships with the
|
|
4
|
+
wheel. ``patchahead demo`` has to work from an install in an empty directory,
|
|
5
|
+
and a template that only exists in a git checkout does not.
|
|
6
|
+
|
|
7
|
+
The extra for this module is ``[web]`` (FastAPI and uvicorn). The demo needs
|
|
8
|
+
``[demo]``, which adds the pytest that turns a patch into a verified migration.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from patchahead.web.server import create_app, main, static_root
|
|
12
|
+
|
|
13
|
+
__all__ = ["create_app", "main", "static_root"]
|