pytest-failure-instrumentation 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pytest_failure_instrumentation-0.1.0/LICENSE +21 -0
- pytest_failure_instrumentation-0.1.0/PKG-INFO +650 -0
- pytest_failure_instrumentation-0.1.0/README.md +618 -0
- pytest_failure_instrumentation-0.1.0/pyproject.toml +54 -0
- pytest_failure_instrumentation-0.1.0/setup.cfg +4 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/__init__.py +10 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/analysis/__init__.py +1 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/analysis/attribution.py +98 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/analysis/classify.py +109 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/analysis/collection.py +288 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/analysis/exit_status.py +72 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/analysis/fingerprint.py +31 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/analysis/severity.py +36 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/analysis/stall.py +122 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/capture/__init__.py +6 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/capture/crash_stack.py +159 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/capture/events.py +71 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/capture/heartbeat.py +69 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/capture/memory.py +133 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/capture/recorder.py +162 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/capture/state.py +98 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/config.py +109 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/hookspec.py +50 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/incidents/__init__.py +5 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/incidents/base.py +177 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/incidents/collection.py +370 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/incidents/death.py +124 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/incidents/engine.py +281 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/incidents/internal_error.py +112 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/incidents/registry.py +65 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/incidents/stall.py +169 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/incidents/summary.py +91 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/plugin.py +51 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/probes/__init__.py +27 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/probes/capabilities.py +33 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/probes/memory.py +261 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/probes/platform_flags.py +23 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/probes/process.py +137 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation/probes/stacks.py +32 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation.egg-info/PKG-INFO +650 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation.egg-info/SOURCES.txt +56 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation.egg-info/dependency_links.txt +1 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation.egg-info/entry_points.txt +2 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation.egg-info/requires.txt +8 -0
- pytest_failure_instrumentation-0.1.0/src/pytest_failure_instrumentation.egg-info/top_level.txt +1 -0
- pytest_failure_instrumentation-0.1.0/tests/test_attribution.py +113 -0
- pytest_failure_instrumentation-0.1.0/tests/test_classify.py +131 -0
- pytest_failure_instrumentation-0.1.0/tests/test_collection.py +95 -0
- pytest_failure_instrumentation-0.1.0/tests/test_collection_analysis.py +211 -0
- pytest_failure_instrumentation-0.1.0/tests/test_crash_stack.py +112 -0
- pytest_failure_instrumentation-0.1.0/tests/test_exit_status.py +63 -0
- pytest_failure_instrumentation-0.1.0/tests/test_internal_error.py +62 -0
- pytest_failure_instrumentation-0.1.0/tests/test_models.py +303 -0
- pytest_failure_instrumentation-0.1.0/tests/test_probes.py +121 -0
- pytest_failure_instrumentation-0.1.0/tests/test_run_summary.py +71 -0
- pytest_failure_instrumentation-0.1.0/tests/test_stall.py +173 -0
- pytest_failure_instrumentation-0.1.0/tests/test_stall_analysis.py +70 -0
- pytest_failure_instrumentation-0.1.0/tests/test_worker_death.py +251 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Heknon
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,650 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pytest-failure-instrumentation
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Attribute the pytest failures that leave no trace: internal errors, and xdist worker deaths, stalls and collection mismatches
|
|
5
|
+
Author: Heknon
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/Heknon/pytest-failure-instrumentation
|
|
8
|
+
Project-URL: Issues, https://github.com/Heknon/pytest-failure-instrumentation/issues
|
|
9
|
+
Project-URL: Changelog, https://github.com/Heknon/pytest-failure-instrumentation/releases
|
|
10
|
+
Keywords: pytest,xdist,crash,oom,observability,instrumentation
|
|
11
|
+
Classifier: Framework :: Pytest
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Topic :: Software Development :: Testing
|
|
21
|
+
Classifier: Topic :: System :: Monitoring
|
|
22
|
+
Requires-Python: >=3.9
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Requires-Dist: pytest>=7.0
|
|
26
|
+
Requires-Dist: pydantic>=2.0
|
|
27
|
+
Provides-Extra: test
|
|
28
|
+
Requires-Dist: pytest-xdist>=3.0; extra == "test"
|
|
29
|
+
Provides-Extra: psutil
|
|
30
|
+
Requires-Dist: psutil>=5.9; extra == "psutil"
|
|
31
|
+
Dynamic: license-file
|
|
32
|
+
|
|
33
|
+
# pytest-failure-instrumentation
|
|
34
|
+
|
|
35
|
+
[](https://github.com/Heknon/pytest-failure-instrumentation/actions/workflows/ci.yml)
|
|
36
|
+
|
|
37
|
+
Most test failures explain themselves. A `pytest_runtest_makereport` gives you
|
|
38
|
+
an assertion, a traceback and a node id, and there is nothing left to
|
|
39
|
+
investigate.
|
|
40
|
+
|
|
41
|
+
The failures that happen *outside* the call phase explain nothing. A worker is
|
|
42
|
+
killed, a run wedges, workers disagree about which tests exist, pytest raises
|
|
43
|
+
inside its own machinery — and what reaches your reporting is a placeholder
|
|
44
|
+
string, a line in a log, or nothing at all. This plugin records what those
|
|
45
|
+
failures cannot say for themselves, works out whose code is responsible, and
|
|
46
|
+
hands you one structured incident per problem.
|
|
47
|
+
|
|
48
|
+
```
|
|
49
|
+
[worker_death] NATIVE_CRASH severity=critical owner=product
|
|
50
|
+
blamed on engine.py:6 in native_call
|
|
51
|
+
in flight test_crashes.py::test_crashes phase=call started=1 finished=0
|
|
52
|
+
· died while running test_crashes.py::test_crashes (call)
|
|
53
|
+
· exit status -11 - SIGSEGV: segmentation fault in native code (pid 805, via waitid)
|
|
54
|
+
· the worker wrote a stack before dying
|
|
55
|
+
· segmentation fault in native code
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
## The problem
|
|
59
|
+
|
|
60
|
+
When a pytest-xdist worker dies, this is the whole report:
|
|
61
|
+
|
|
62
|
+
```
|
|
63
|
+
[gw7] node down: Not properly terminated
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
An OOM kill, a segfault in a C extension and a stray `os._exit(1)` are
|
|
67
|
+
indistinguishable at that point. Not because nobody bothered to print the
|
|
68
|
+
difference — because by the time anything can ask, the difference is gone.
|
|
69
|
+
There are three independent reasons, and all three have to be worked around
|
|
70
|
+
separately.
|
|
71
|
+
|
|
72
|
+
**1. The cause never leaves the process.** SIGSEGV and SIGKILL end a process
|
|
73
|
+
without running any Python. No `finally`, no `atexit`, no `__del__`, no
|
|
74
|
+
`pytest_sessionfinish`. Whatever the worker knew about what it was doing, it
|
|
75
|
+
knew only in memory, and that memory is gone. Anything you want to know
|
|
76
|
+
afterwards has to have been written down *before* — while the run was healthy
|
|
77
|
+
and the cost of writing it lands on every passing test.
|
|
78
|
+
|
|
79
|
+
**2. The exit status is read and thrown away.** The kernel keeps one number
|
|
80
|
+
that separates all these cases, and the parent process is the only thing
|
|
81
|
+
allowed to read it. execnet does read it — `Group.terminate` calls
|
|
82
|
+
`gw._io.wait()` in `execnet/multi.py` — and discards the return value. Nothing
|
|
83
|
+
in xdist asks for it either.
|
|
84
|
+
|
|
85
|
+
**3. There is no field to put it in.** In `xdist/workermanage.py`,
|
|
86
|
+
`process_from_remote` handles the channel closing: it asks execnet for a remote
|
|
87
|
+
error, and when there is none — because the remote never got to send anything —
|
|
88
|
+
substitutes a literal string:
|
|
89
|
+
|
|
90
|
+
```python
|
|
91
|
+
err = "Not properly terminated" # lost connection?
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
That string is then passed to `pytest_testnodedown(node, error)` as the error.
|
|
95
|
+
The hook is not withholding the cause. By the time it fires, the placeholder is
|
|
96
|
+
genuinely all that exists.
|
|
97
|
+
|
|
98
|
+
### Why you never see a `MemoryError`
|
|
99
|
+
|
|
100
|
+
The most common way a worker dies is also the one Python is least able to
|
|
101
|
+
report. On Linux, `malloc` returning successfully is not a promise that the
|
|
102
|
+
memory exists — overcommit hands out address space and resolves it on first
|
|
103
|
+
touch. There is no allocation failure for CPython to raise `MemoryError` from.
|
|
104
|
+
The process is killed later, from outside, with SIGKILL, which cannot be
|
|
105
|
+
caught, blocked or handled.
|
|
106
|
+
|
|
107
|
+
And the exit status is `-9` for *all* of it: the kernel OOM killer, a cgroup
|
|
108
|
+
limit, a cancelled CI job, a `kill -9` from a stray script. There is no
|
|
109
|
+
distinct code for "out of memory". The only in-process evidence that separates
|
|
110
|
+
them is the cgroup v2 `memory.events` `oom_kill` counter, which is why
|
|
111
|
+
`OOM_KILLED` is claimed only when that counter moved during this run, and
|
|
112
|
+
`SIGKILLED` — "something killed it, and here is what that could have been" —
|
|
113
|
+
when it did not.
|
|
114
|
+
|
|
115
|
+
### The failures that reach no hook at all
|
|
116
|
+
|
|
117
|
+
Worker death at least fires a hook. Three others do not.
|
|
118
|
+
|
|
119
|
+
**A worker that stalls.** `pytest_testnodedown` needs a dead process; a wedged
|
|
120
|
+
one is alive. And the controller hears from a worker only when a phase
|
|
121
|
+
*completes*, so from outside, a twenty-minute test and a deadlock are the same
|
|
122
|
+
event: nothing. The run does not fail — it never ends, and CI kills the job an
|
|
123
|
+
hour later with no artifact naming a test.
|
|
124
|
+
|
|
125
|
+
**Workers that collected different tests.** xdist notices, writes a unified
|
|
126
|
+
diff per differing worker into its own log, and aborts. Nothing structured
|
|
127
|
+
reaches a hook. With sixty workers and one odd node that is fifty-nine complete
|
|
128
|
+
diffs, every one of them naming the majority as the deviation.
|
|
129
|
+
|
|
130
|
+
**An internal error.** pytest sets `ExitCode.INTERNAL_ERROR`, which is not in
|
|
131
|
+
`summary_exit_codes` in `_pytest/terminal.py` — so `pytest_terminal_summary`
|
|
132
|
+
never fires for it. Under xdist it is worse: a worker's internal error is
|
|
133
|
+
relayed to the controller as a flat string and re-raised there, so the
|
|
134
|
+
`INTERNALERROR>` block you read names xdist's frame, not the failure.
|
|
135
|
+
|
|
136
|
+
## Who this is for
|
|
137
|
+
|
|
138
|
+
**You ship a library into other people's test suites.** Their run dies and the
|
|
139
|
+
bug report names your package. Nothing in the output can confirm or refute it,
|
|
140
|
+
and "cannot reproduce" is not an answer anyone accepts. `owner` is the field
|
|
141
|
+
that settles it — `product`, `third-party`, `customer-code` or `runtime` — and
|
|
142
|
+
it comes from a stack, not from a guess.
|
|
143
|
+
|
|
144
|
+
**You own CI for a large suite.** Runs fail with no test named. Was that the
|
|
145
|
+
OOM killer, or the runner getting reclaimed mid-job? `-9` is identical either
|
|
146
|
+
way, and the answer decides whether you buy more memory or file a ticket with
|
|
147
|
+
your CI vendor.
|
|
148
|
+
|
|
149
|
+
**Your suite hangs sometimes.** Nothing fails, the job times out, and there is
|
|
150
|
+
no evidence at all because the process that would produce it is the one that is
|
|
151
|
+
stuck. `worker_stall` names the test, says whether the thread is blocked or the
|
|
152
|
+
whole process is frozen, and prints the stack of the thread actually stuck.
|
|
153
|
+
|
|
154
|
+
**You run enough workers that they disagree.** A conftest keyed on an
|
|
155
|
+
environment variable, a machine-dependent skip, a plugin that collects
|
|
156
|
+
conditionally. The run aborts and the reason is buried in N−1 diffs.
|
|
157
|
+
|
|
158
|
+
**You are collecting failures across machines you do not control.**
|
|
159
|
+
`fingerprint` groups recurrences so one defect on twelve workers is one row
|
|
160
|
+
with a count, and `capabilities` records what each machine could measure — so a
|
|
161
|
+
missing memory figure on a customer's Windows box reads as "unmeasurable here"
|
|
162
|
+
rather than "fine".
|
|
163
|
+
|
|
164
|
+
## A worked example: xdist #1362
|
|
165
|
+
|
|
166
|
+
A worker dies in the window after it has sent its collection but before
|
|
167
|
+
scheduling begins, while a second worker is still collecting. The stale entry
|
|
168
|
+
in `registered_collections` is never cleaned up, and the run dies with a
|
|
169
|
+
`KeyError` naming an object rather than a problem
|
|
170
|
+
([pytest-dev/pytest-xdist#1362](https://github.com/pytest-dev/pytest-xdist/issues/1362)).
|
|
171
|
+
|
|
172
|
+
This is what the plugin reports for it — two incidents, because two things went
|
|
173
|
+
wrong:
|
|
174
|
+
|
|
175
|
+
```
|
|
176
|
+
[worker_death] SIGKILLED severity=needs-triage owner=unknown
|
|
177
|
+
no test in flight started=0 finished=0
|
|
178
|
+
· died before running any test (startup or collection)
|
|
179
|
+
· exit status -9 - SIGKILL: uncatchable kill (OOM killer or external kill) (pid 21780, via waitid)
|
|
180
|
+
· resident memory 31 MB at last checkpoint
|
|
181
|
+
· SIGKILL with no cgroup OOM event: a host-level OOM killer, a container or CI cancellation, or an external kill
|
|
182
|
+
|
|
183
|
+
[internal_error] INTERNAL_ERROR severity=high owner=runtime run-ending
|
|
184
|
+
blamed on loadscope.py:275 in _assign_work_unit
|
|
185
|
+
KeyError: <WorkerController gw1>
|
|
186
|
+
· raised on the controller itself and captured first-hand
|
|
187
|
+
· raised above informational: a framework-owned defect that ended the run - no test is at fault, so nothing else will surface it
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
`owner=runtime` is the load-bearing part. No test is at fault and no worker is
|
|
191
|
+
at fault, so nothing else in the run will ever surface this — which is exactly
|
|
192
|
+
why it is the one case where a framework defect is raised *above*
|
|
193
|
+
informational.
|
|
194
|
+
|
|
195
|
+
## Install
|
|
196
|
+
|
|
197
|
+
```console
|
|
198
|
+
pip install pytest-failure-instrumentation
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
It registers itself as a `pytest11` entry point. Implement one hook to receive
|
|
202
|
+
what it finds:
|
|
203
|
+
|
|
204
|
+
```python
|
|
205
|
+
# conftest.py, or your own plugin
|
|
206
|
+
def pytest_failure_incident(incident):
|
|
207
|
+
database.save(incident.model_dump())
|
|
208
|
+
alerts.send(str(incident))
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
Tell it which packages are yours, so a failing frame in your code can be told
|
|
212
|
+
from one in a dependency or in the customer's own tests:
|
|
213
|
+
|
|
214
|
+
```ini
|
|
215
|
+
[pytest]
|
|
216
|
+
failure_packages = yourcore, yourcore_ext
|
|
217
|
+
failure_product_version = 4.2.0
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
Without the hook it still writes its evidence to `.pytest-failures/`.
|
|
221
|
+
Disable it entirely with `-p no:failure_instrumentation`.
|
|
222
|
+
|
|
223
|
+
## What you get
|
|
224
|
+
|
|
225
|
+
`incident` is a pydantic model, one class per kind, discriminated on
|
|
226
|
+
`incident.kind`. A segfault's resident memory and a run summary's exit code
|
|
227
|
+
have nothing to say to each other, so they are not fields of the same object:
|
|
228
|
+
|
|
229
|
+
| `kind` | Model | Raised on |
|
|
230
|
+
|---|---|---|
|
|
231
|
+
| `worker_death` | `WorkerDeathIncident` | needs xdist |
|
|
232
|
+
| `worker_stall` | `WorkerStallIncident` | needs xdist |
|
|
233
|
+
| `collection_mismatch` | `CollectionMismatchIncident` | needs xdist |
|
|
234
|
+
| `internal_error` | `InternalErrorIncident` | any run |
|
|
235
|
+
| `run_summary` | `RunSummaryIncident` | any run |
|
|
236
|
+
|
|
237
|
+
The last two are not distributed problems, so the plugin registers whether or
|
|
238
|
+
not you run under xdist and a plain `pytest` gets both.
|
|
239
|
+
|
|
240
|
+
They share `verdict`, `confidence`, `severity`, `owner`, `fingerprint`,
|
|
241
|
+
`run_id`, `worker` and `evidence`. `str(incident)` is the alert text — every
|
|
242
|
+
block quoted in this README is what it prints. A stored row comes back as the
|
|
243
|
+
model it was written from, and the union is a schema you can migrate a table
|
|
244
|
+
against:
|
|
245
|
+
|
|
246
|
+
```python
|
|
247
|
+
from pytest_failure_instrumentation.incidents import registry
|
|
248
|
+
|
|
249
|
+
incident = registry.parse(json.loads(row)) # -> WorkerDeathIncident, ...
|
|
250
|
+
registry.json_schema()
|
|
251
|
+
```
|
|
252
|
+
|
|
253
|
+
**`owner`** is the field that settles arguments — `product`, `third-party`,
|
|
254
|
+
`customer-code`, `runtime`, or `unknown`. It comes from walking outward past
|
|
255
|
+
runtime frames to the first one that belongs to somebody: the deepest frame is
|
|
256
|
+
usually `ctypes.string_at`, which tells nobody anything. A stack with no owned
|
|
257
|
+
frame at all is not unknown — it is a positive finding that the framework
|
|
258
|
+
itself failed.
|
|
259
|
+
|
|
260
|
+
**`severity`** follows from ownership rather than from how loud the failure
|
|
261
|
+
was, so a customer's segfaulting test does not page you. The exception is a
|
|
262
|
+
framework defect that ends the run, above.
|
|
263
|
+
|
|
264
|
+
**`fingerprint`** is stable across runs and excludes worker id, pid and
|
|
265
|
+
timings, so one defect on twelve workers is one incident with a count.
|
|
266
|
+
|
|
267
|
+
**`capabilities`** says what the machine could measure, so an absent figure is
|
|
268
|
+
never read as a healthy one.
|
|
269
|
+
|
|
270
|
+
**`suspect_owner`** is kept apart from `owner` on purpose. When no stack names
|
|
271
|
+
anybody, the test that was in flight is a lead worth recording — but a guess
|
|
272
|
+
must never sit in the column a reader takes for a finding.
|
|
273
|
+
|
|
274
|
+
## Verdicts
|
|
275
|
+
|
|
276
|
+
### A worker died
|
|
277
|
+
|
|
278
|
+
| Verdict | Told apart by |
|
|
279
|
+
|---|---|
|
|
280
|
+
| `OOM_KILLED` | `-9` **and** the cgroup OOM counter moved |
|
|
281
|
+
| `SIGKILLED` | `-9`, counter flat — host OOM, CI cancellation, external kill |
|
|
282
|
+
| `NATIVE_CRASH` | SIGSEGV/SIGABRT/SIGBUS/SIGILL/SIGFPE, or a Windows NTSTATUS |
|
|
283
|
+
| `SIGNAL_<n>` | SIGTERM/SIGINT/SIGHUP — a request to stop, not a defect |
|
|
284
|
+
| `SELF_EXIT` | clean exit code, no signal |
|
|
285
|
+
| `PROBABLY_SIGNALLED` | exit code 128–191, a wrapper ate the signal |
|
|
286
|
+
| `UNKNOWN` | no status obtainable (remote gateway) |
|
|
287
|
+
|
|
288
|
+
### A worker stalled
|
|
289
|
+
|
|
290
|
+
Silence proves nothing on its own: the controller hears from a worker only when
|
|
291
|
+
a phase *completes*, so a twenty-minute test and a deadlock look identical from
|
|
292
|
+
outside. What separates them is the worker's own heartbeat, which carries CPU
|
|
293
|
+
time.
|
|
294
|
+
|
|
295
|
+
| Verdict | Heartbeat | CPU | Means |
|
|
296
|
+
|---|---|---|---|
|
|
297
|
+
| `STALLED_BLOCKED` | alive | none | the test thread is waiting on something that is not coming |
|
|
298
|
+
| `STALLED_FROZEN` | stopped | — | native code is holding the GIL, or the process is stopped |
|
|
299
|
+
| `STALLED_SILENT` | never ran | — | the watchdog is off, so there is no passive evidence either way |
|
|
300
|
+
| *(not reported)* | alive | burning | slow, not stuck |
|
|
301
|
+
|
|
302
|
+
The verdict is reached from beats already on disk. A stack is asked for
|
|
303
|
+
*afterwards*, once the decision is made, because asking a wedged process a
|
|
304
|
+
question can change its answer — see below.
|
|
305
|
+
|
|
306
|
+
### Workers collected different tests
|
|
307
|
+
|
|
308
|
+
| Verdict | Means |
|
|
309
|
+
|---|---|
|
|
310
|
+
| `COLLECTION_MEMBERSHIP_DIFFERS` | a test exists on one machine and not another |
|
|
311
|
+
| `COLLECTION_ORDER_DIFFERS` | same tests, different sequence — fatal too, since xdist addresses tests by position |
|
|
312
|
+
| `COLLECTION_PARAMETERS_UNSTABLE` | same tests, different *parameter values* — a parametrize that is not deterministic |
|
|
313
|
+
|
|
314
|
+
Sixty workers never produce sixty collections. They produce two or three
|
|
315
|
+
*variants*, so this reports one row per variant, measured against the largest:
|
|
316
|
+
|
|
317
|
+
```
|
|
318
|
+
[collection_mismatch] COLLECTION_MEMBERSHIP_DIFFERS severity=needs-triage owner=unknown run-ending
|
|
319
|
+
no stack; suspect customer-code (owner of a module the workers disagreed about (test_collect.py))
|
|
320
|
+
2 workers produced 2 different collections
|
|
321
|
+
baseline: 1 worker collected 3 tests, and everything below is measured against that list
|
|
322
|
+
1 worker is missing 1 test, in test_collect.py (gw1)
|
|
323
|
+
- test_collect.py::test_two
|
|
324
|
+
whole collections written to .pytest-failures; the difference above travels in the incident
|
|
325
|
+
· xdist addresses tests by position rather than by id, so any difference between the lists is fatal - a reordering as much as a missing test
|
|
326
|
+
· the initial collections disagreed, so the run was aborted
|
|
327
|
+
```
|
|
328
|
+
|
|
329
|
+
At sixty workers it stays the same shape, because the row count follows the
|
|
330
|
+
number of *variants* rather than the number of workers:
|
|
331
|
+
|
|
332
|
+
```
|
|
333
|
+
58 workers produced 3 different collections
|
|
334
|
+
baseline: 55 workers collected 300 tests, and everything below is measured against that list
|
|
335
|
+
2 workers are missing 1 test, in test_payments.py (gw41, gw58)
|
|
336
|
+
- test_payments.py::test_case_017
|
|
337
|
+
1 worker has 6 extra tests, in test_legacy.py (gw17)
|
|
338
|
+
+ test_legacy.py::test_extra_0
|
|
339
|
+
+ test_legacy.py::test_extra_1
|
|
340
|
+
+ test_legacy.py::test_extra_2
|
|
341
|
+
and 3 more
|
|
342
|
+
```
|
|
343
|
+
|
|
344
|
+
Read it as: **how many distinct opinions existed, which workers held each, and
|
|
345
|
+
how the minority differs from the majority.** Magnitude leads each line and
|
|
346
|
+
identity follows, samples use diff notation, and a truncated sample always says
|
|
347
|
+
how much it withheld — a sample that looks like the whole story is worse than
|
|
348
|
+
no sample at all.
|
|
349
|
+
|
|
350
|
+
**The whole difference travels in the payload**, not just the three ids the
|
|
351
|
+
text prints. `missing` and `extra` carry every differing node id, up to 500 per
|
|
352
|
+
side, with `missing_count` and `extra_count` as the true totals so you can see
|
|
353
|
+
whether that cap was reached. The distinction matters: a *collection* is
|
|
354
|
+
unbounded — sixty workers times fifty thousand node ids is hundreds of
|
|
355
|
+
megabytes — but a *difference* is almost always one test or one module's worth.
|
|
356
|
+
Only the digest is held per worker, and the whole collections are written to
|
|
357
|
+
`collection-<digest>.txt` for whoever still has the machine. That file is on a
|
|
358
|
+
runner which may be gone by the time anyone reads the alert, which is exactly
|
|
359
|
+
why the difference itself does not live there.
|
|
360
|
+
|
|
361
|
+
An order difference instead reports where the two lists first disagree, which
|
|
362
|
+
is the one fact a unified diff of a reordered list destroys.
|
|
363
|
+
|
|
364
|
+
**A parametrize whose values are drawn at collection time** — `random`, a
|
|
365
|
+
timestamp, an unordered set — gives every worker a different id for the same
|
|
366
|
+
test, and reported as membership that reads as thousands of tests appearing and
|
|
367
|
+
disappearing. It is caught by asking a second question: are these the same
|
|
368
|
+
tests once the parameters are stripped from the ids? When they are, the report
|
|
369
|
+
names the parametrized tests responsible, drops the per-variant rows — which
|
|
370
|
+
would otherwise be one near-identical block per worker — and prints what each
|
|
371
|
+
of a few workers actually collected:
|
|
372
|
+
|
|
373
|
+
```
|
|
374
|
+
[collection_mismatch] COLLECTION_PARAMETERS_UNSTABLE severity=needs-triage run-ending
|
|
375
|
+
6 workers produced 6 different collections
|
|
376
|
+
the tests are the same on every worker - only the parameter values differ, so these are not tests appearing and disappearing
|
|
377
|
+
test_billing.py::test_invoice
|
|
378
|
+
gw0 collected acct-1791, acct-3471, acct-6305, acct-7468
|
|
379
|
+
gw1 collected acct-2186, acct-2542, acct-6991, acct-9779
|
|
380
|
+
gw2 collected acct-1614, acct-1950, acct-4517, acct-9313
|
|
381
|
+
compare those values: a parametrize evaluated at collection time - a random number, a timestamp, an unordered set, a call to something live - gives every worker a different id for the same test, and xdist requires the ids to match
|
|
382
|
+
```
|
|
383
|
+
|
|
384
|
+
The values are the diagnosis. Naming the test says where to look; three rows
|
|
385
|
+
of disjoint account ids say a fetch is running at collection time, and three
|
|
386
|
+
rows of floating-point noise say a random number is. Neither is apparent from
|
|
387
|
+
one worker's list, which is the only thing xdist ever shows you.
|
|
388
|
+
|
|
389
|
+
That case is also why full id lists are held for only the first few variants.
|
|
390
|
+
"A handful of variants" is the assumption the whole design rests on, and
|
|
391
|
+
unstable ids turn it into one variant per worker. Past that limit a variant is
|
|
392
|
+
reported as `not compared` rather than diffed against a list nobody kept —
|
|
393
|
+
comparing two absent lists reports "the same tests in a different order", which
|
|
394
|
+
is a finding invented out of missing data.
|
|
395
|
+
|
|
396
|
+
A mismatch is run-ending *usually*, not always: xdist aborts when the initial
|
|
397
|
+
collections disagree, but silently drops a worker that registers a differing
|
|
398
|
+
collection after scheduling has begun. The run then continues one worker short,
|
|
399
|
+
and `run_ending` reflects which of the two happened.
|
|
400
|
+
|
|
401
|
+
## How it knows
|
|
402
|
+
|
|
403
|
+
**A fixed-size state file.** Which test and phase is open right now is written
|
|
404
|
+
to a 256-byte slot with `os.pwrite` — one syscall, no append, no growth, and a
|
|
405
|
+
file that is the same size after a million tests as after one. That is what
|
|
406
|
+
separates "died in teardown" from "died mid-call": pytest's own `logfinish`
|
|
407
|
+
fires only after the whole protocol, so it cannot tell them apart.
|
|
408
|
+
|
|
409
|
+
**The exit status, taken from the OS.** Where a `Popen` object survives, its
|
|
410
|
+
return code. Otherwise `waitid(P_PID, pid, WEXITED | WNOWAIT | WNOHANG)` —
|
|
411
|
+
`WNOWAIT` reads the status *without consuming it*, so execnet's own reaping
|
|
412
|
+
still works afterwards and nothing is broken by looking. Only a parent may do
|
|
413
|
+
this, which is why it happens on the controller, and why a remote gateway
|
|
414
|
+
honestly reports `UNKNOWN` rather than guessing. macOS does not expose
|
|
415
|
+
`os.waitid` at all and falls back to the `Popen` object, which is why
|
|
416
|
+
`capabilities` records the mechanism that answered rather than the one the
|
|
417
|
+
platform was assumed to have. On Windows the code is normalised to its
|
|
418
|
+
unsigned form first: an NTSTATUS is above 2³¹, so `0xC000013A` arrives signed
|
|
419
|
+
or unsigned depending on who answered — and a negative status means "killed by
|
|
420
|
+
signal N" to everything downstream.
|
|
421
|
+
|
|
422
|
+
**faulthandler, pointed at a per-worker file.** pytest's own faulthandler
|
|
423
|
+
plugin enables at configure time with `trylast`, aimed at shared stderr where
|
|
424
|
+
every worker's output interleaves. This claims the handler back afterwards, in
|
|
425
|
+
`pytest_sessionstart`. Its C handler is async-signal-safe and writes *while the
|
|
426
|
+
GIL is held* — which is the case that matters, since native code holding the
|
|
427
|
+
GIL is exactly what a frozen worker looks like.
|
|
428
|
+
|
|
429
|
+
**Choosing the right thread out of a dump.** A dump written with
|
|
430
|
+
`all_threads=True` holds every thread, and the first one printed in a pytest
|
|
431
|
+
worker is this plugin's own heartbeat thread; the second is execnet's receiver.
|
|
432
|
+
Reporting the first section would blame the instrumentation for the failure it
|
|
433
|
+
came to explain. The section reported is the one the fault or signal reached
|
|
434
|
+
(`Current thread`), else the one carrying pytest's runtest protocol, else
|
|
435
|
+
anything that is not ours.
|
|
436
|
+
|
|
437
|
+
**A watchdog the worker arms on itself.** The obvious design has the controller
|
|
438
|
+
signal a stalled worker and let faulthandler answer. It has two flaws: Windows
|
|
439
|
+
has no `SIGUSR1` and `os.kill` there cannot deliver one, and on POSIX the
|
|
440
|
+
signal *perturbs the subject* — PEP 475 makes Python retry on `EINTR`, but a C
|
|
441
|
+
extension blocked in a raw syscall need not, so it returns early and the stall
|
|
442
|
+
being measured disappears. So each test arms
|
|
443
|
+
`faulthandler.dump_traceback_later` instead. It works on every platform,
|
|
444
|
+
interrupts no syscall, and still dumps while native code holds the GIL. The
|
|
445
|
+
signal path remains as an extra, for asking an already-diagnosed worker for a
|
|
446
|
+
fresher stack.
|
|
447
|
+
|
|
448
|
+
**A heartbeat carrying CPU time.** One line every five seconds per worker,
|
|
449
|
+
bounded by wall-clock rather than by how many tests run. `time.process_time()`
|
|
450
|
+
in each beat is what turns silence into a verdict: alive with no CPU is
|
|
451
|
+
blocked, stopped is frozen, alive and burning is a slow test that must be
|
|
452
|
+
reported as nothing at all.
|
|
453
|
+
|
|
454
|
+
**Evidence written before it is needed.** Every mechanism above puts its output
|
|
455
|
+
on disk during the healthy part of the run, because a process that is about to
|
|
456
|
+
be killed gets no warning. The controller reads files, never the corpse.
|
|
457
|
+
|
|
458
|
+
## Cost
|
|
459
|
+
|
|
460
|
+
A passing test must cost as close to nothing as possible, because that is the
|
|
461
|
+
overwhelming majority of what runs.
|
|
462
|
+
|
|
463
|
+
- Per test: two fixed-size writes to a file that never grows, plus arming a
|
|
464
|
+
watchdog timer (~78 µs). No append log, no `/proc` read, no allocation
|
|
465
|
+
tracking.
|
|
466
|
+
- Per 5 seconds, per worker: one heartbeat carrying CPU time and resident
|
|
467
|
+
memory.
|
|
468
|
+
- Off by default: `tracemalloc` (needed to attribute an OOM kill to a source
|
|
469
|
+
line) and the live-object census — walking the heap on a worker near its
|
|
470
|
+
ceiling is exactly the instrumentation that makes things worse.
|
|
471
|
+
- pydantic is imported on the controller, and only when xdist is active. A
|
|
472
|
+
worker never loads it, so nothing about the per-test path changed when the
|
|
473
|
+
payload became typed.
|
|
474
|
+
- Nothing in the reporting path may raise. A failure while gathering an
|
|
475
|
+
incident degrades it to what survived, because an exception in a reporting
|
|
476
|
+
hook becomes an `INTERNALERROR` that ends the customer's run.
|
|
477
|
+
|
|
478
|
+
## Settings
|
|
479
|
+
|
|
480
|
+
| Setting | Default | Purpose |
|
|
481
|
+
|---|---|---|
|
|
482
|
+
| `failure_packages` | — | Your top-level packages, for attribution |
|
|
483
|
+
| `failure_directory` | `.pytest-failures` | Where evidence is written |
|
|
484
|
+
| `failure_watchdog` | `true` | Memory and liveness sampling |
|
|
485
|
+
| `failure_heartbeat_interval` | `5.0` | Seconds between liveness beats |
|
|
486
|
+
| `failure_tracemalloc_depth` | `0` | 1 names the allocating line for OOM attribution |
|
|
487
|
+
| `failure_object_census` | `false` | Count live objects at a high-water mark |
|
|
488
|
+
| `failure_high_water_mb` | auto | Memory mark for a snapshot; defaults to a share of the discovered limit |
|
|
489
|
+
| `failure_memory_limit_mb` | `0` | Soft cap (POSIX) turning an OOM kill into a `MemoryError` |
|
|
490
|
+
| `failure_slow_test_seconds` | `120` | A test outliving this dumps its own stack |
|
|
491
|
+
| `failure_stall_seconds` | `300` | Silence before a stall is assessed |
|
|
492
|
+
| `failure_stack_probe` | `true` | Ask a diagnosed stalled worker for a fresh stack (POSIX) |
|
|
493
|
+
|
|
494
|
+
`failure_memory_limit_mb` is worth a note: an `RLIMIT_AS` cap makes the
|
|
495
|
+
allocation fail *inside* the process, so you get a `MemoryError` with a
|
|
496
|
+
traceback and a node id instead of an uncatchable kill with neither. It costs
|
|
497
|
+
you a hard ceiling per worker, which is why it is opt-in.
|
|
498
|
+
|
|
499
|
+
## Platform coverage
|
|
500
|
+
|
|
501
|
+
| Capability | Linux | macOS | Windows |
|
|
502
|
+
|---|---|---|---|
|
|
503
|
+
| Test in flight, phase, exit status | yes | yes | yes |
|
|
504
|
+
| Crash stack | yes | yes | yes |
|
|
505
|
+
| Stack from a *slow or hung* test | yes | yes | yes |
|
|
506
|
+
| Current memory | procfs | psutil, else peak only | psapi |
|
|
507
|
+
| Container limit, OOM counter | yes | n/a | n/a — no OOM killer |
|
|
508
|
+
| On-demand stack from a stalled worker | yes | yes | no |
|
|
509
|
+
|
|
510
|
+
Two Windows differences are worth knowing about, because they change what you
|
|
511
|
+
will see rather than how it is reported.
|
|
512
|
+
|
|
513
|
+
ctypes wraps every foreign function call in structured exception handling, so
|
|
514
|
+
an access violation raised *through ctypes* comes back as an `OSError` and the
|
|
515
|
+
worker survives it. A fault inside a real C extension still ends the process —
|
|
516
|
+
but the reproduction that segfaults a worker on Linux may simply fail a test on
|
|
517
|
+
Windows.
|
|
518
|
+
|
|
519
|
+
And a Windows process that dies from a fault reports an NTSTATUS as its exit
|
|
520
|
+
code rather than a signal, while `abort()` reports plain `3` — the same code a
|
|
521
|
+
deliberate `os._exit(3)` gives. What separates a crash from a clean exit there
|
|
522
|
+
is whether a dump was written, not the exit status, which is why the crash
|
|
523
|
+
stack is evidence in its own right rather than a decoration on the verdict.
|
|
524
|
+
|
|
525
|
+
`psutil` is never required, only ever an upgrade: `pip install
|
|
526
|
+
pytest-failure-instrumentation[psutil]`.
|
|
527
|
+
|
|
528
|
+
## Tests
|
|
529
|
+
|
|
530
|
+
```console
|
|
531
|
+
pip install -e ".[test]"
|
|
532
|
+
pytest
|
|
533
|
+
```
|
|
534
|
+
|
|
535
|
+
The integration tests run a real pytest in a subprocess through `pytester`,
|
|
536
|
+
crash or wedge a worker for real, and read back what the plugin raised — so
|
|
537
|
+
they exercise the mechanism rather than a mock of it. Every one of them also
|
|
538
|
+
round-trips its incidents through `registry.parse` and asserts `model_dump()`
|
|
539
|
+
equals the stored row, which makes the payload contract a property of every
|
|
540
|
+
scenario rather than a test of its own.
|
|
541
|
+
|
|
542
|
+
CI runs the suite on Linux, macOS and Windows across Python 3.9–3.13 — every
|
|
543
|
+
platform path in the table above is executed on the platform it was written
|
|
544
|
+
for. The
|
|
545
|
+
probes are platform code — procfs, psapi, `waitid`, `GetExitCodeProcess`,
|
|
546
|
+
cgroup counters — and none of the Windows or macOS paths can be exercised on a
|
|
547
|
+
Linux runner, which is the whole reason the matrix exists. Two axes matter as
|
|
548
|
+
much as the operating system, so each gets its own job:
|
|
549
|
+
|
|
550
|
+
- **without `psutil`**, which is what most people actually have. Every probe
|
|
551
|
+
has to degrade to a declared "unavailable" rather than to a wrong number.
|
|
552
|
+
- **without `pytest-xdist`**, where `pytest_testnodedown` has no hookspec at
|
|
553
|
+
all and an unspecced hookimpl is a registration error — the failure mode that
|
|
554
|
+
once made a plain `pytest` run report nothing.
|
|
555
|
+
|
|
556
|
+
## Releasing
|
|
557
|
+
|
|
558
|
+
Tag the commit and the rest runs itself:
|
|
559
|
+
|
|
560
|
+
```console
|
|
561
|
+
git tag v0.2.0 && git push origin v0.2.0
|
|
562
|
+
```
|
|
563
|
+
|
|
564
|
+
The tag is the only input. `.github/workflows/release.yml` builds the sdist and
|
|
565
|
+
wheel, refuses to continue if the tag disagrees with the version in
|
|
566
|
+
`pyproject.toml`, installs the **built wheel** on Linux, macOS and Windows and
|
|
567
|
+
runs the whole suite against it, publishes to PyPI, and then creates the GitHub
|
|
568
|
+
release with the artifacts attached.
|
|
569
|
+
|
|
570
|
+
The wheel is tested rather than the checkout because this plugin is one entry
|
|
571
|
+
point. If packaging drops it the import still succeeds, the suite still passes,
|
|
572
|
+
and nothing is instrumented at all — the one failure mode a green test run
|
|
573
|
+
cannot rule out. So the release explicitly asserts the entry point exists and
|
|
574
|
+
that the package under test came from `site-packages`.
|
|
575
|
+
|
|
576
|
+
### Credentials
|
|
577
|
+
|
|
578
|
+
There is no API token to create and no secret to add to the repository.
|
|
579
|
+
Publishing uses [trusted publishing](https://docs.pypi.org/trusted-publishers/):
|
|
580
|
+
PyPI verifies this workflow's OIDC identity at upload time, so nothing
|
|
581
|
+
long-lived exists to leak or rotate. `GITHUB_TOKEN` is supplied by Actions
|
|
582
|
+
automatically.
|
|
583
|
+
|
|
584
|
+
What it does need is configuration, once, on each side.
|
|
585
|
+
|
|
586
|
+
**On PyPI** — *Your account → Publishing*. The project does not exist there
|
|
587
|
+
yet, so this is an **"Add a new pending publisher"**, not a setting on an
|
|
588
|
+
existing project; a pending publisher is how a first upload is authorised for a
|
|
589
|
+
name nobody has claimed. It becomes a normal publisher after that first
|
|
590
|
+
release.
|
|
591
|
+
|
|
592
|
+
| Field | Value |
|
|
593
|
+
|---|---|
|
|
594
|
+
| PyPI project name | `pytest-failure-instrumentation` |
|
|
595
|
+
| Owner | `Heknon` |
|
|
596
|
+
| Repository name | `pytest-failure-instrumentation` |
|
|
597
|
+
| Workflow name | `release.yml` |
|
|
598
|
+
| Environment name | `pypi` |
|
|
599
|
+
|
|
600
|
+
**On GitHub** — *Settings → Environments → New environment*, named `pypi`.
|
|
601
|
+
Under it, tick **Required reviewers** and add yourself. That is the manual gate:
|
|
602
|
+
the run pauses before anything reaches PyPI, shows you the tag it is about to
|
|
603
|
+
publish, and waits. Nothing is uploaded until someone approves, and waiting does
|
|
604
|
+
not consume the job's timeout.
|
|
605
|
+
|
|
606
|
+
Worth setting at the same time, under *Deployment branches and tags*: restrict
|
|
607
|
+
the environment to the tag pattern `v*`, so the only thing that can ever reach
|
|
608
|
+
PyPI is a tagged commit.
|
|
609
|
+
|
|
610
|
+
**TestPyPI** is a separate site with a separate account, so rehearsing needs its
|
|
611
|
+
own pending publisher at test.pypi.org with the environment named `testpypi`.
|
|
612
|
+
Leave that environment without reviewers — the point of a rehearsal is that it
|
|
613
|
+
does not need one.
|
|
614
|
+
|
|
615
|
+
## Licence
|
|
616
|
+
|
|
617
|
+
MIT — see [LICENSE](LICENSE). Declared as an SPDX expression under
|
|
618
|
+
[PEP 639](https://peps.python.org/pep-0639/) rather than a classifier, since
|
|
619
|
+
PyPI rejects a distribution carrying both.
|
|
620
|
+
|
|
621
|
+
## Status
|
|
622
|
+
|
|
623
|
+
All five kinds and every verdict in the tables above are covered, on all three
|
|
624
|
+
platforms.
|
|
625
|
+
|
|
626
|
+
Most are produced for real: a worker is crashed, killed, signalled, wedged or
|
|
627
|
+
made to disagree about its collection, and the incident is read back from the
|
|
628
|
+
hook. Two cannot be, by anyone: `OOM_KILLED` needs a kernel that has just
|
|
629
|
+
killed something, and `UNKNOWN` needs a remote gateway with no local process to
|
|
630
|
+
query. Those branches are exercised against a constructed incident instead — as
|
|
631
|
+
are the Windows NTSTATUS decodes, which additionally run against a process that
|
|
632
|
+
really exits with one.
|
|
633
|
+
|
|
634
|
+
The opt-in paths are covered too: the memory ceiling turning an uncatchable
|
|
635
|
+
kill into a `MemoryError` that names the test, and the high-water snapshot
|
|
636
|
+
naming the line holding the memory.
|
|
637
|
+
|
|
638
|
+
The probes are also called directly, because in normal use they shadow each
|
|
639
|
+
other — psutil answers before psapi, and execnet's `Popen` before `waitid` — so
|
|
640
|
+
the fallbacks a customer's machine actually runs were never being executed.
|
|
641
|
+
That includes the claim `WNOWAIT` rests on: the status is read, and the process
|
|
642
|
+
is still reapable afterwards with the same answer.
|
|
643
|
+
|
|
644
|
+
The first cross-platform run paid for itself twice. It found that a Windows
|
|
645
|
+
`\Lib\` in `sysconfig` and a `\lib\` in a traceback made every stdlib frame
|
|
646
|
+
look like nobody's code, so a blocked test was blamed on `threading.py` and
|
|
647
|
+
then on the customer who called it — a runtime frame reported as customer code,
|
|
648
|
+
which is the one direction this must never fail in. Only the 3.9 cell caught
|
|
649
|
+
it. And it found that ctypes cannot raise an uncaught fault on Windows at all,
|
|
650
|
+
which is a fact about what users will see rather than about the plugin.
|