pytest-timing 0.3.0__tar.gz → 0.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/ARCHITECTURE.md +34 -16
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/CHANGELOG.md +31 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/PKG-INFO +30 -10
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/README.md +29 -9
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/pyproject.toml +1 -1
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/__init__.py +1 -1
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/admission.py +2 -1
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/collector.py +19 -5
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/model.py +19 -3
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/plugin.py +15 -6
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/render/ascii.py +14 -2
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/schedule.py +54 -17
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/static/report.html +184 -27
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/xdist_scheduler.py +50 -15
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_ascii.py +13 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_collector.py +13 -1
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_html.py +52 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_memory_gate.py +132 -28
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_runtime_admission.py +5 -1
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/uv.lock +1 -1
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/.github/workflows/ci.yml +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/.github/workflows/release.yml +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/.gitignore +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/LICENSE +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/benchmarks/cpu_bench.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/docs/report.png +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/examples/test_demo.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/cli.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/demand.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/fixtures.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/outputs.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/render/__init__.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/render/html.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/render/trace.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/static/__init__.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/telemetry.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/src/pytest_timing/xdist_compat.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/conftest.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/fixtures/legacy_complete_false.json +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/fixtures/legacy_complete_true.json +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/fixtures/legacy_no_flag.json +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_admission.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_benchmarks.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_cli.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_cpu.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_fixtures.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_memory.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_plugin.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_schedule.py +0 -0
- {pytest_timing-0.3.0 → pytest_timing-0.3.2}/tests/test_trace.py +0 -0
|
@@ -388,27 +388,45 @@ happens only when nothing runs in the domain, so nothing else's memory is at ris
|
|
|
388
388
|
Pressure feedback moves only the CPU limit.
|
|
389
389
|
|
|
390
390
|
Demand comes from history, never from declarations. `Estimates.from_run` keeps, per
|
|
391
|
-
test, the largest
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
several).
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
391
|
+
test, the largest need any attempt showed (the worst case is what an out-of-memory
|
|
392
|
+
kill depends on), and per shared fixture key the memory the attempt that paid its
|
|
393
|
+
set-up kept resident (`after - base`, split evenly when one attempt paid for
|
|
394
|
+
several). A payer's rise spans its set-ups, so its own need is `peak - after`: what
|
|
395
|
+
it used beyond what stayed, which is the fixtures'. Session- and package-scoped
|
|
396
|
+
fixtures are the exception: what they keep is every worker's baseline
|
|
397
|
+
(`is_baseline`), never attributed, charged or projected. Every worker sets them up
|
|
398
|
+
once and never lets go, so gating on them cannot spare the host; it can only hold
|
|
399
|
+
tests back, or park a worker until stall resolution shuts it down, which is what a
|
|
400
|
+
large session fixture did before this rule. The first attempt on each worker is a
|
|
401
|
+
warm-up window (lazy imports, caches, the allocator's first growth), so its rise and
|
|
402
|
+
residual count only for a test or fixture with no other attempt.
|
|
403
|
+
|
|
404
|
+
`Costs.charge` adds a `memory` to each `Charge`: the test's need plus what the
|
|
405
|
+
module- and class-scoped fixtures alive around it keep, projected through the lane's
|
|
406
|
+
fixture state exactly as CPU holds are, since workers report nothing about memory at
|
|
407
|
+
run time. A fixture with recorded memory belongs to its tests' families even when its
|
|
408
|
+
set-up was too quick to matter for time. `_need_memory` is the twin of `_need`; an
|
|
409
|
+
idle worker reserves what its fixtures keep, and a worker whose next test could never
|
|
410
|
+
fit next to what the other workers' fixtures keep is parked like one blocked by CPU
|
|
411
|
+
holds. A fixture reached at run time adds its recorded memory to the queued charges
|
|
412
|
+
when its request is granted.
|
|
403
413
|
|
|
404
414
|
The budget is `--timing-memory`: a size, or `auto`, which takes 80% of the smallest
|
|
405
415
|
memory the domain's workers report (physical memory capped by the cgroup limit,
|
|
406
416
|
`memory.max` on v2 and `memory.limit_in_bytes` on v1, read like the CPU quota). The
|
|
407
417
|
share is deliberate: estimates are rises above each worker's footprint, and the
|
|
408
|
-
footprints, the controller and the rest of the host are
|
|
409
|
-
recorded run every estimate is zero and the gate admits
|
|
410
|
-
so. The planner ignores memory: with tests kept apart by
|
|
411
|
-
memory would add little, and the lane plan can still move.
|
|
418
|
+
footprints (session fixtures included), the controller and the rest of the host are
|
|
419
|
+
not in them. Without a recorded run every estimate is zero and the gate admits
|
|
420
|
+
everything; the summary says so. The planner ignores memory: with tests kept apart by
|
|
421
|
+
the gate, balancing lanes by memory would add little, and the lane plan can still move.
|
|
422
|
+
|
|
423
|
+
Every `Admission` has a `kind` (`cpu` or `memory`), and a refusal records which kinds
|
|
424
|
+
refused (`refused`), as does passing a test over for what other workers keep. When
|
|
425
|
+
the worker is admitted, the wait it records (`AdmissionWait.gates`) says which gates
|
|
426
|
+
held it, and the collector puts it on the test's `cpu.wait` or `memory.wait`
|
|
427
|
+
accordingly. A worker that leaves while parked, having run nothing it waited for,
|
|
428
|
+
has no test to carry its wait: `_release_runtime` records it in `parked`, and each
|
|
429
|
+
gate's summary reports the parked time and worker count beside the tests' waits.
|
|
412
430
|
|
|
413
431
|
## Measuring CPU work
|
|
414
432
|
|
|
@@ -1,5 +1,36 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.3.2
|
|
4
|
+
|
|
5
|
+
- The HTML report shows CPU and memory usage. The table gains "CPU time", "CPUs"
|
|
6
|
+
(the CPUs a test kept busy on average, beside the slots it declared) and "Memory"
|
|
7
|
+
(what it needed on top of its worker's footprint) columns, the hover details say
|
|
8
|
+
the same with what the measurement covered, the header totals the run and names
|
|
9
|
+
each host's budget, and two graphs show CPUs busy and resident memory over time,
|
|
10
|
+
estimated from the per-test records. All of it appears only where there was a
|
|
11
|
+
reading.
|
|
12
|
+
|
|
13
|
+
## 0.3.1
|
|
14
|
+
|
|
15
|
+
- Memory admission no longer charges what session- and package-scoped fixtures keep
|
|
16
|
+
resident. Every worker sets them up once and never lets go, so counting them once
|
|
17
|
+
per lane, and again against every other lane, could only hold tests back or park
|
|
18
|
+
a worker until it was shut down: on a suite with a large session fixture the gate
|
|
19
|
+
left two of eight workers nearly idle and made the run almost three times longer
|
|
20
|
+
while the host had tens of gigabytes free. Their memory is now the worker's
|
|
21
|
+
baseline, which the budget's headroom covers (#7).
|
|
22
|
+
- Memory estimates are what a test needs of its own. An attempt that set up shared
|
|
23
|
+
fixtures was charged its whole rise and, again, what the fixtures kept, twice the
|
|
24
|
+
residual; its own need is now the rise beyond what stayed. The first attempt on
|
|
25
|
+
each worker, whose window also covers the worker's warm-up, counts only for a
|
|
26
|
+
test or fixture with no other attempt (#7).
|
|
27
|
+
- Waits say which gate held the test. A test the memory gate held back carries the
|
|
28
|
+
seconds in `memory.wait`, beside `cpu.wait` for CPU slots, and the summary line
|
|
29
|
+
reports memory waits even when a CPU budget is set. Time a worker sat parked and
|
|
30
|
+
then left without running what it waited for is reported per gate as well, instead
|
|
31
|
+
of vanishing. The HTML report's table gains a "Held" column and its hover details a
|
|
32
|
+
"held" line, and the terminal's slowest-tests rows say how long each was held (#7).
|
|
33
|
+
|
|
3
34
|
## 0.3.0
|
|
4
35
|
|
|
5
36
|
- Estimate a test's cost from the median of its best attempts instead of its
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: pytest-timing
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.2
|
|
4
4
|
Summary: Record test timings under pytest-xdist and render cargo-style timing reports (ASCII Gantt and HTML).
|
|
5
5
|
Project-URL: Homepage, https://github.com/messense/pytest-timing
|
|
6
6
|
Author-email: messense <messense@icloud.com>
|
|
@@ -128,14 +128,23 @@ of the worker's process tree around each test, in bytes: `base` before its set-u
|
|
|
128
128
|
`peak` during it, and `after` at its teardown. The difference between `peak` and
|
|
129
129
|
`base` is what the test needed on top of the worker's existing footprint; `after`
|
|
130
130
|
minus `base` is what stayed, such as a shared fixture it set up. With
|
|
131
|
-
`--timing-memory`, the run's `memory` describes the budget and how admission went
|
|
131
|
+
`--timing-memory`, the run's `memory` describes the budget and how admission went,
|
|
132
|
+
and a test the memory gate held back carries the seconds it waited in `memory.wait`,
|
|
133
|
+
as `cpu.wait` does for the CPU gate; the HTML report's table and hover details and
|
|
134
|
+
the terminal's slowest-tests rows show them.
|
|
132
135
|
It also records worker
|
|
133
136
|
lifecycles, detected host CPU environments, and how the session ended (`finished`,
|
|
134
137
|
`collect_only`, `interrupted`, `aborted`, `internal_error`, with pytest's reason). It is
|
|
135
138
|
the input to the CLI below and to `--timing-schedule`.
|
|
136
139
|
|
|
137
140
|
**HTML** is a single file with no external dependencies, so it can be attached to a CI
|
|
138
|
-
job as an artifact and opened anywhere.
|
|
141
|
+
job as an artifact and opened anywhere. Where CPU and memory were measured it shows
|
|
142
|
+
them: each test's CPU time, the CPUs it kept busy on average (beside the slots it
|
|
143
|
+
declared) and the memory it needed on top of its worker's footprint, in the table and
|
|
144
|
+
the hover details; the run's totals and budgets in the header; and CPU and memory
|
|
145
|
+
usage over time. The two graphs are drawn from the per-test records, not sampled: a
|
|
146
|
+
test's CPU time is spread evenly over its run, and a worker counts at its running
|
|
147
|
+
test's peak, so they show where the load was rather than its exact shape.
|
|
139
148
|
|
|
140
149
|
**Trace** is a Chrome Trace Event file. Open it in
|
|
141
150
|
[Perfetto UI](https://ui.perfetto.dev) with "Open trace file", or in `chrome://tracing`.
|
|
@@ -268,8 +277,8 @@ What to expect:
|
|
|
268
277
|
forces the oldest blocked request if needed. Such forced reservations can exceed
|
|
269
278
|
the limit and are counted in the summary.
|
|
270
279
|
- The summary reports the budget, the number of tests over one slot, and how long
|
|
271
|
-
tests waited for slots in total. Each test's JSON record carries its own wait
|
|
272
|
-
`cpu.runtime_wait` records admission waits inside a running test. These waits
|
|
280
|
+
tests waited for slots in total. Each test's JSON record carries its own wait in
|
|
281
|
+
`cpu.wait`. `cpu.runtime_wait` records admission waits inside a running test. These waits
|
|
273
282
|
are excluded from duration estimates, fixture setup costs and measured CPU rate.
|
|
274
283
|
A runtime request that gets no grant within five minutes cancels the fixture
|
|
275
284
|
setup. It waits to recover the test's original reservation before failing, so
|
|
@@ -305,18 +314,26 @@ pytest -n 4 --timing-json --timing-schedule pytest-timing.json --timing-memory 1
|
|
|
305
314
|
takes 80% of physical memory, capped by the cgroup memory limit. It reads the memory
|
|
306
315
|
each test needed from the run at `--timing-schedule`, so the first run only records;
|
|
307
316
|
the summary says how many tests had recorded memory. A test with no record weighs
|
|
308
|
-
nothing until it has run once. A
|
|
309
|
-
(a loaded model, say) reserved for as long as it is alive, on that
|
|
317
|
+
nothing until it has run once. A module- or class-scoped fixture keeps what its first
|
|
318
|
+
test left resident (a loaded model, say) reserved for as long as it is alive, on that
|
|
319
|
+
worker. What session- and package-scoped fixtures keep is treated as part of every
|
|
320
|
+
worker's footprint instead: each worker sets them up once and never lets go, so
|
|
321
|
+
gating on them could only delay the run, never spare the host.
|
|
310
322
|
|
|
311
323
|
What to expect:
|
|
312
324
|
|
|
313
325
|
- The gate is the same fair waiting line as for CPU: a worker whose next test does
|
|
314
326
|
not fit waits with its fixtures alive, and the oldest request goes first. Memory
|
|
315
327
|
and CPU budgets are checked together; a test starts only when both fit.
|
|
316
|
-
- Estimates are the largest
|
|
328
|
+
- Estimates are the largest need any recorded attempt showed. For an attempt that
|
|
329
|
+
set up shared fixtures, what stayed resident afterwards is the fixtures' (or the
|
|
330
|
+
worker's footprint), and the test's own need is what it used beyond that. The
|
|
331
|
+
first test on each worker also pays for the worker's warm-up, so its record is
|
|
332
|
+
used only for a test or fixture with no other attempt. Estimates are relative to
|
|
317
333
|
the worker's footprint, so `auto` leaves a fifth of the host for the workers
|
|
318
|
-
themselves, the controller and everything else; set a
|
|
319
|
-
machine
|
|
334
|
+
themselves, their session fixtures, the controller and everything else; set a
|
|
335
|
+
smaller budget on a shared machine, and a larger share of headroom when session
|
|
336
|
+
fixtures are big.
|
|
320
337
|
- A test recorded above the budget runs alone, and the summary counts it. Allocator
|
|
321
338
|
behaviour can make an estimate low: a test that reuses heap an earlier test freed
|
|
322
339
|
shows a smaller rise than it needs. Large buffers and subprocesses, the usual
|
|
@@ -324,6 +341,9 @@ What to expect:
|
|
|
324
341
|
- Memory does not enter the planner: lanes are still balanced by duration and fixture
|
|
325
342
|
cost, and the gate only delays starts. Under `--dist load` and `worksteal`, like
|
|
326
343
|
CPU admission.
|
|
344
|
+
- The summary says how many tests waited for memory and for how long, and how long
|
|
345
|
+
workers sat parked at the gate without running what they waited for. Each held
|
|
346
|
+
test's JSON record carries its wait in `memory.wait`.
|
|
327
347
|
|
|
328
348
|
## Re-render or merge saved runs
|
|
329
349
|
|
|
@@ -102,14 +102,23 @@ of the worker's process tree around each test, in bytes: `base` before its set-u
|
|
|
102
102
|
`peak` during it, and `after` at its teardown. The difference between `peak` and
|
|
103
103
|
`base` is what the test needed on top of the worker's existing footprint; `after`
|
|
104
104
|
minus `base` is what stayed, such as a shared fixture it set up. With
|
|
105
|
-
`--timing-memory`, the run's `memory` describes the budget and how admission went
|
|
105
|
+
`--timing-memory`, the run's `memory` describes the budget and how admission went,
|
|
106
|
+
and a test the memory gate held back carries the seconds it waited in `memory.wait`,
|
|
107
|
+
as `cpu.wait` does for the CPU gate; the HTML report's table and hover details and
|
|
108
|
+
the terminal's slowest-tests rows show them.
|
|
106
109
|
It also records worker
|
|
107
110
|
lifecycles, detected host CPU environments, and how the session ended (`finished`,
|
|
108
111
|
`collect_only`, `interrupted`, `aborted`, `internal_error`, with pytest's reason). It is
|
|
109
112
|
the input to the CLI below and to `--timing-schedule`.
|
|
110
113
|
|
|
111
114
|
**HTML** is a single file with no external dependencies, so it can be attached to a CI
|
|
112
|
-
job as an artifact and opened anywhere.
|
|
115
|
+
job as an artifact and opened anywhere. Where CPU and memory were measured it shows
|
|
116
|
+
them: each test's CPU time, the CPUs it kept busy on average (beside the slots it
|
|
117
|
+
declared) and the memory it needed on top of its worker's footprint, in the table and
|
|
118
|
+
the hover details; the run's totals and budgets in the header; and CPU and memory
|
|
119
|
+
usage over time. The two graphs are drawn from the per-test records, not sampled: a
|
|
120
|
+
test's CPU time is spread evenly over its run, and a worker counts at its running
|
|
121
|
+
test's peak, so they show where the load was rather than its exact shape.
|
|
113
122
|
|
|
114
123
|
**Trace** is a Chrome Trace Event file. Open it in
|
|
115
124
|
[Perfetto UI](https://ui.perfetto.dev) with "Open trace file", or in `chrome://tracing`.
|
|
@@ -242,8 +251,8 @@ What to expect:
|
|
|
242
251
|
forces the oldest blocked request if needed. Such forced reservations can exceed
|
|
243
252
|
the limit and are counted in the summary.
|
|
244
253
|
- The summary reports the budget, the number of tests over one slot, and how long
|
|
245
|
-
tests waited for slots in total. Each test's JSON record carries its own wait
|
|
246
|
-
`cpu.runtime_wait` records admission waits inside a running test. These waits
|
|
254
|
+
tests waited for slots in total. Each test's JSON record carries its own wait in
|
|
255
|
+
`cpu.wait`. `cpu.runtime_wait` records admission waits inside a running test. These waits
|
|
247
256
|
are excluded from duration estimates, fixture setup costs and measured CPU rate.
|
|
248
257
|
A runtime request that gets no grant within five minutes cancels the fixture
|
|
249
258
|
setup. It waits to recover the test's original reservation before failing, so
|
|
@@ -279,18 +288,26 @@ pytest -n 4 --timing-json --timing-schedule pytest-timing.json --timing-memory 1
|
|
|
279
288
|
takes 80% of physical memory, capped by the cgroup memory limit. It reads the memory
|
|
280
289
|
each test needed from the run at `--timing-schedule`, so the first run only records;
|
|
281
290
|
the summary says how many tests had recorded memory. A test with no record weighs
|
|
282
|
-
nothing until it has run once. A
|
|
283
|
-
(a loaded model, say) reserved for as long as it is alive, on that
|
|
291
|
+
nothing until it has run once. A module- or class-scoped fixture keeps what its first
|
|
292
|
+
test left resident (a loaded model, say) reserved for as long as it is alive, on that
|
|
293
|
+
worker. What session- and package-scoped fixtures keep is treated as part of every
|
|
294
|
+
worker's footprint instead: each worker sets them up once and never lets go, so
|
|
295
|
+
gating on them could only delay the run, never spare the host.
|
|
284
296
|
|
|
285
297
|
What to expect:
|
|
286
298
|
|
|
287
299
|
- The gate is the same fair waiting line as for CPU: a worker whose next test does
|
|
288
300
|
not fit waits with its fixtures alive, and the oldest request goes first. Memory
|
|
289
301
|
and CPU budgets are checked together; a test starts only when both fit.
|
|
290
|
-
- Estimates are the largest
|
|
302
|
+
- Estimates are the largest need any recorded attempt showed. For an attempt that
|
|
303
|
+
set up shared fixtures, what stayed resident afterwards is the fixtures' (or the
|
|
304
|
+
worker's footprint), and the test's own need is what it used beyond that. The
|
|
305
|
+
first test on each worker also pays for the worker's warm-up, so its record is
|
|
306
|
+
used only for a test or fixture with no other attempt. Estimates are relative to
|
|
291
307
|
the worker's footprint, so `auto` leaves a fifth of the host for the workers
|
|
292
|
-
themselves, the controller and everything else; set a
|
|
293
|
-
machine
|
|
308
|
+
themselves, their session fixtures, the controller and everything else; set a
|
|
309
|
+
smaller budget on a shared machine, and a larger share of headroom when session
|
|
310
|
+
fixtures are big.
|
|
294
311
|
- A test recorded above the budget runs alone, and the summary counts it. Allocator
|
|
295
312
|
behaviour can make an estimate low: a test that reuses heap an earlier test freed
|
|
296
313
|
shows a smaller rise than it needs. Large buffers and subprocesses, the usual
|
|
@@ -298,6 +315,9 @@ What to expect:
|
|
|
298
315
|
- Memory does not enter the planner: lanes are still balanced by duration and fixture
|
|
299
316
|
cost, and the gate only delays starts. Under `--dist load` and `worksteal`, like
|
|
300
317
|
CPU admission.
|
|
318
|
+
- The summary says how many tests waited for memory and for how long, and how long
|
|
319
|
+
workers sat parked at the gate without running what they waited for. Each held
|
|
320
|
+
test's JSON record carries its wait in `memory.wait`.
|
|
301
321
|
|
|
302
322
|
## Re-render or merge saved runs
|
|
303
323
|
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "pytest-timing"
|
|
7
|
-
version = "0.3.
|
|
7
|
+
version = "0.3.2"
|
|
8
8
|
description = "Record test timings under pytest-xdist and render cargo-style timing reports (ASCII Gantt and HTML)."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -40,8 +40,9 @@ class Admission:
|
|
|
40
40
|
RECOVER_AFTER = 8
|
|
41
41
|
COOLDOWN = 10.0
|
|
42
42
|
|
|
43
|
-
def __init__(self, budget: int, name: str = "local") -> None:
|
|
43
|
+
def __init__(self, budget: int, name: str = "local", kind: str = "cpu") -> None:
|
|
44
44
|
self.name = name
|
|
45
|
+
self.kind = kind # what the slots are: ``cpu`` or ``memory`` (bytes)
|
|
45
46
|
self.budget = max(1, int(budget))
|
|
46
47
|
self.limit = self.budget
|
|
47
48
|
self.reserved: dict[Hashable, Reservation] = {}
|
|
@@ -14,6 +14,7 @@ correct span; they do not change that public occurrence/attempt numbering.
|
|
|
14
14
|
|
|
15
15
|
from __future__ import annotations
|
|
16
16
|
|
|
17
|
+
from collections.abc import Iterable
|
|
17
18
|
from dataclasses import dataclass
|
|
18
19
|
from typing import Any
|
|
19
20
|
|
|
@@ -83,12 +84,19 @@ class Collector:
|
|
|
83
84
|
worker.items = items
|
|
84
85
|
|
|
85
86
|
def add_wait(
|
|
86
|
-
self,
|
|
87
|
+
self,
|
|
88
|
+
worker_id: str,
|
|
89
|
+
nodeid: str,
|
|
90
|
+
index: int,
|
|
91
|
+
attempt: int,
|
|
92
|
+
seconds: float,
|
|
93
|
+
gates: Iterable[str] = (),
|
|
87
94
|
) -> None:
|
|
88
95
|
"""Add an admission delay to the exact execution that waited.
|
|
89
96
|
|
|
90
97
|
A nodeid can have several selections and retries. Looking up its last span
|
|
91
|
-
at session finish would put every delay on the final one instead.
|
|
98
|
+
at session finish would put every delay on the final one instead. ``gates``
|
|
99
|
+
names what held the test (``cpu``, ``memory``); unnamed, the CPU gate did.
|
|
92
100
|
"""
|
|
93
101
|
span = self._executions.get((worker_id, index, attempt))
|
|
94
102
|
if span is None:
|
|
@@ -98,9 +106,15 @@ class Collector:
|
|
|
98
106
|
span = self._last.get((worker_id, nodeid))
|
|
99
107
|
if span is None or span.outcome != "crashed" or span.phases or span.attempt != attempt:
|
|
100
108
|
return
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
109
|
+
gates = set(gates)
|
|
110
|
+
if "cpu" in gates or not gates:
|
|
111
|
+
if span.cpu is None:
|
|
112
|
+
span.cpu = CpuRecord(elapsed=span.duration)
|
|
113
|
+
span.cpu.wait += seconds
|
|
114
|
+
if "memory" in gates:
|
|
115
|
+
if span.memory is None:
|
|
116
|
+
span.memory = MemoryRecord() # no reading, but the wait is a fact
|
|
117
|
+
span.memory.wait += seconds
|
|
104
118
|
|
|
105
119
|
def worker_down(self, worker_id: str, epoch: float, error: str | None) -> None:
|
|
106
120
|
worker = self._worker(worker_id)
|
|
@@ -212,14 +212,17 @@ class MemoryRecord:
|
|
|
212
212
|
``peak`` the highest reading until the teardown report, ``after`` the reading at
|
|
213
213
|
that report. ``coverage`` says whether descendants were counted (``tree``) or
|
|
214
214
|
only the worker (``self``). A worker's heap rarely shrinks, so ``rise`` is
|
|
215
|
-
what the attempt needed on top of what was already there,
|
|
216
|
-
stayed resident afterwards
|
|
215
|
+
what the attempt needed on top of what was already there, ``retained`` what
|
|
216
|
+
stayed resident afterwards (a shared fixture it set up, or a leak), and
|
|
217
|
+
``transient`` the rest: what it needed beyond what stayed. ``wait`` is how long
|
|
218
|
+
memory admission held the attempt back before it started, in seconds.
|
|
217
219
|
"""
|
|
218
220
|
|
|
219
221
|
base: int = 0
|
|
220
222
|
peak: int = 0
|
|
221
223
|
after: int = 0
|
|
222
224
|
coverage: str = "none"
|
|
225
|
+
wait: float = 0.0
|
|
223
226
|
|
|
224
227
|
@property
|
|
225
228
|
def rise(self) -> int:
|
|
@@ -229,13 +232,25 @@ class MemoryRecord:
|
|
|
229
232
|
def retained(self) -> int:
|
|
230
233
|
return max(0, self.after - self.base)
|
|
231
234
|
|
|
235
|
+
@property
|
|
236
|
+
def transient(self) -> int:
|
|
237
|
+
return max(0, self.peak - self.after)
|
|
238
|
+
|
|
239
|
+
@property
|
|
240
|
+
def measured(self) -> bool:
|
|
241
|
+
"""Did the platform give a reading? A live process is never resident at zero."""
|
|
242
|
+
return self.peak > 0
|
|
243
|
+
|
|
232
244
|
def to_dict(self) -> dict[str, Any]:
|
|
233
|
-
|
|
245
|
+
doc: dict[str, Any] = {
|
|
234
246
|
"base": self.base,
|
|
235
247
|
"peak": self.peak,
|
|
236
248
|
"after": self.after,
|
|
237
249
|
"coverage": self.coverage,
|
|
238
250
|
}
|
|
251
|
+
if self.wait:
|
|
252
|
+
doc["wait"] = _round(self.wait)
|
|
253
|
+
return doc
|
|
239
254
|
|
|
240
255
|
@classmethod
|
|
241
256
|
def from_dict(cls, data: dict[str, Any]) -> MemoryRecord:
|
|
@@ -244,6 +259,7 @@ class MemoryRecord:
|
|
|
244
259
|
peak=int(data.get("peak", 0)),
|
|
245
260
|
after=int(data.get("after", 0)),
|
|
246
261
|
coverage=str(data.get("coverage", "none")),
|
|
262
|
+
wait=float(data.get("wait", 0.0)),
|
|
247
263
|
)
|
|
248
264
|
|
|
249
265
|
|
|
@@ -176,6 +176,18 @@ def format_bytes(size: float) -> str:
|
|
|
176
176
|
return f"{size:.1f} TiB"
|
|
177
177
|
|
|
178
178
|
|
|
179
|
+
def _waited(summary: dict[str, Any], what: str) -> str:
|
|
180
|
+
"""The waiting part of a gate's summary line: tests held back, workers parked."""
|
|
181
|
+
text = ""
|
|
182
|
+
if summary.get("waited_tests"):
|
|
183
|
+
n = summary["waited_tests"]
|
|
184
|
+
text += f"; {n} test{'s' if n != 1 else ''} waited {summary['waited']:.2f}s {what}"
|
|
185
|
+
if summary.get("parked_workers"):
|
|
186
|
+
n = summary["parked_workers"]
|
|
187
|
+
text += f"; {n} worker{'s' if n != 1 else ''} sat {summary['parked']:.2f}s parked {what}"
|
|
188
|
+
return text
|
|
189
|
+
|
|
190
|
+
|
|
179
191
|
class Settings:
|
|
180
192
|
"""Resolve values by CLI, environment, then ini; enable timing if any source asks."""
|
|
181
193
|
|
|
@@ -573,9 +585,7 @@ class TimingPlugin:
|
|
|
573
585
|
line += f", largest {format_bytes(summary['largest'])}"
|
|
574
586
|
else:
|
|
575
587
|
line += "; no recorded memory in the schedule history yet"
|
|
576
|
-
|
|
577
|
-
n = summary["waited_tests"]
|
|
578
|
-
line += f"; {n} test{'s' if n != 1 else ''} waited {summary['waited']:.2f}s for memory"
|
|
588
|
+
line += _waited(summary, "for memory")
|
|
579
589
|
return line
|
|
580
590
|
|
|
581
591
|
def _cpu_summary(self) -> str | None:
|
|
@@ -609,9 +619,7 @@ class TimingPlugin:
|
|
|
609
619
|
return None # no CPU budget: a memory budget alone has its own line
|
|
610
620
|
line = "cpu: " + "; ".join(parts)
|
|
611
621
|
line += f"; {heavy} test{'s' if heavy != 1 else ''} over one slot"
|
|
612
|
-
|
|
613
|
-
n = summary["waited_tests"]
|
|
614
|
-
line += f"; {n} test{'s' if n != 1 else ''} waited {summary['waited']:.2f}s for slots"
|
|
622
|
+
line += _waited(summary, "for slots")
|
|
615
623
|
if summary.get("cancelled"):
|
|
616
624
|
n = summary["cancelled"]
|
|
617
625
|
line += f"; {n} unadmitted test{'s' if n != 1 else ''} withdrawn at shutdown"
|
|
@@ -672,6 +680,7 @@ class TimingPlugin:
|
|
|
672
680
|
wait.index,
|
|
673
681
|
wait.attempt,
|
|
674
682
|
wait.seconds,
|
|
683
|
+
wait.gates,
|
|
675
684
|
)
|
|
676
685
|
self.collector.run.cpu = scheduler.cpu_summary()
|
|
677
686
|
self.collector.run.memory = scheduler.memory_summary()
|
|
@@ -122,8 +122,10 @@ def render_ascii(
|
|
|
122
122
|
bar = f"{_RED}{bar}{_RESET}"
|
|
123
123
|
duration = f"{format_seconds(test.duration):>7}"
|
|
124
124
|
lines.append(f"{test.worker:<{label_width}} {bar} {duration}")
|
|
125
|
-
|
|
126
|
-
|
|
125
|
+
held = _held(test)
|
|
126
|
+
room = width - label_width - 1 - (len(held) + 2 if held else 0)
|
|
127
|
+
nodeid = _fit_nodeid(test.nodeid, room, glyphs.ellipsis)
|
|
128
|
+
lines.append(f"{'':<{label_width}} {nodeid}{' ' + held if held else ''}")
|
|
127
129
|
return "\n".join(lines)
|
|
128
130
|
|
|
129
131
|
|
|
@@ -265,6 +267,16 @@ def _test_bar(test: TestSpan, scale: Scale, glyphs: Glyphs) -> str:
|
|
|
265
267
|
return "".join(row)
|
|
266
268
|
|
|
267
269
|
|
|
270
|
+
def _held(test: TestSpan) -> str:
|
|
271
|
+
"""How long admission held the test back before it started, and at which gate."""
|
|
272
|
+
parts = []
|
|
273
|
+
if test.cpu is not None and test.cpu.wait:
|
|
274
|
+
parts.append(f"waited {format_seconds(test.cpu.wait)} for cpu")
|
|
275
|
+
if test.memory is not None and test.memory.wait:
|
|
276
|
+
parts.append(f"waited {format_seconds(test.memory.wait)} for memory")
|
|
277
|
+
return ", ".join(parts)
|
|
278
|
+
|
|
279
|
+
|
|
268
280
|
def _fit_nodeid(nodeid: str, room: int, ellipsis: str) -> str:
|
|
269
281
|
if room >= len(nodeid) or room < 8 + len(ellipsis):
|
|
270
282
|
return nodeid
|
|
@@ -13,7 +13,8 @@ from pathlib import Path
|
|
|
13
13
|
from statistics import fmean, median
|
|
14
14
|
|
|
15
15
|
from pytest_timing.demand import Declarations, key_base
|
|
16
|
-
from pytest_timing.
|
|
16
|
+
from pytest_timing.fixtures import key_scope
|
|
17
|
+
from pytest_timing.model import Run, TestSpan
|
|
17
18
|
|
|
18
19
|
Family = frozenset[str]
|
|
19
20
|
NO_FIXTURES: Family = frozenset()
|
|
@@ -23,6 +24,29 @@ EPSILON = 1e-9
|
|
|
23
24
|
SPLIT_GAIN = 0.01
|
|
24
25
|
"""A family is split over more workers only for at least this much predicted gain:
|
|
25
26
|
duplicated set-ups are real cost, and gains this small are within estimate noise."""
|
|
27
|
+
BASELINE_SCOPES = frozenset({"session", "package"})
|
|
28
|
+
"""Scopes whose fixtures' memory is part of the worker's footprint, not a test's need:
|
|
29
|
+
every worker sets them up once and keeps them, so gating on them could only delay
|
|
30
|
+
the run, never spare the host. The budget's headroom is what covers them."""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def is_baseline(key: str) -> bool:
|
|
34
|
+
"""Does the fixture behind ``key`` keep memory as a per-worker baseline?"""
|
|
35
|
+
return key_scope(key) in BASELINE_SCOPES
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _warm_ups(run: Run) -> set[int]:
|
|
39
|
+
"""Ids of the first attempt each worker ran.
|
|
40
|
+
|
|
41
|
+
Its memory window also covers the worker's warm-up (lazy imports, caches, the
|
|
42
|
+
allocator's first growth), which is neither the test's need nor a fixture's.
|
|
43
|
+
"""
|
|
44
|
+
first: dict[str, TestSpan] = {}
|
|
45
|
+
for test in run.tests:
|
|
46
|
+
current = first.get(test.worker)
|
|
47
|
+
if current is None or test.start < current.start:
|
|
48
|
+
first[test.worker] = test
|
|
49
|
+
return {id(test) for test in first.values()}
|
|
26
50
|
|
|
27
51
|
|
|
28
52
|
@cache
|
|
@@ -89,16 +113,21 @@ class Estimates:
|
|
|
89
113
|
does not grow with the number of attempts the way the longest one does, so a
|
|
90
114
|
history merged from many runs stays comparable to a single run.
|
|
91
115
|
|
|
92
|
-
Memory is the opposite: the largest
|
|
93
|
-
case is what an out-of-memory kill depends on.
|
|
94
|
-
|
|
95
|
-
when it paid for several at once,
|
|
116
|
+
Memory is the opposite: the largest need any attempt showed, since the worst
|
|
117
|
+
case is what an out-of-memory kill depends on. An attempt that paid for shared
|
|
118
|
+
set-ups needed its whole rise, but what stayed resident afterwards is the
|
|
119
|
+
fixtures' (split evenly when it paid for several at once) or, for session and
|
|
120
|
+
package scope, the worker's baseline; the test's own need is the rest
|
|
121
|
+
(``transient``). The first attempt on each worker also covers the worker's
|
|
122
|
+
warm-up, so its memory counts only for a test or fixture that has no other
|
|
123
|
+
attempt.
|
|
96
124
|
"""
|
|
97
125
|
attempts: dict[str, dict[int, list[float]]] = {} # nodeid -> rank -> own seconds
|
|
98
126
|
families: dict[str, set[str]] = {}
|
|
99
127
|
setups: dict[str, list[float]] = {}
|
|
100
|
-
memory: dict[str, int] = {}
|
|
101
|
-
retained: dict[str, int] = {}
|
|
128
|
+
memory: dict[bool, dict[str, int]] = {False: {}, True: {}} # warm-up? -> bytes
|
|
129
|
+
retained: dict[bool, dict[str, int]] = {False: {}, True: {}}
|
|
130
|
+
warm_ups = _warm_ups(run)
|
|
102
131
|
for test in run.tests:
|
|
103
132
|
paid = []
|
|
104
133
|
if test.fixtures:
|
|
@@ -107,12 +136,16 @@ class Estimates:
|
|
|
107
136
|
if seconds:
|
|
108
137
|
setups.setdefault(key, []).append(seconds)
|
|
109
138
|
paid.append(key)
|
|
110
|
-
if test.memory is not None:
|
|
111
|
-
|
|
139
|
+
if test.memory is not None and test.memory.measured:
|
|
140
|
+
warm = id(test) in warm_ups
|
|
141
|
+
rises, keeps = memory[warm], retained[warm]
|
|
142
|
+
need = test.memory.transient if paid else test.memory.rise
|
|
143
|
+
rises[test.nodeid] = max(rises.get(test.nodeid, 0), need)
|
|
112
144
|
if paid and test.memory.retained:
|
|
113
145
|
share = test.memory.retained // len(paid)
|
|
114
146
|
for key in paid:
|
|
115
|
-
|
|
147
|
+
if not is_baseline(key):
|
|
148
|
+
keeps[key] = max(keeps.get(key, 0), share)
|
|
116
149
|
if test.outcome == "crashed":
|
|
117
150
|
continue # the recorded stop is the worker's death, not the test's
|
|
118
151
|
own = test.duration - test.shared_setup
|
|
@@ -132,8 +165,8 @@ class Estimates:
|
|
|
132
165
|
source,
|
|
133
166
|
{nodeid: frozenset(keys) for nodeid, keys in families.items()},
|
|
134
167
|
{key: median(seconds) for key, seconds in setups.items()},
|
|
135
|
-
memory,
|
|
136
|
-
retained,
|
|
168
|
+
{**memory[True], **memory[False]}, # a clean attempt beats a warm-up one
|
|
169
|
+
{**retained[True], **retained[False]},
|
|
137
170
|
)
|
|
138
171
|
|
|
139
172
|
@classmethod
|
|
@@ -168,7 +201,10 @@ class Estimates:
|
|
|
168
201
|
return self.memory.get(nodeid, 0)
|
|
169
202
|
|
|
170
203
|
def kept(self, key: str) -> int:
|
|
171
|
-
"""Bytes a shared fixture instance keeps resident while alive; unknown is zero
|
|
204
|
+
"""Bytes a shared fixture instance keeps resident while alive; unknown is zero,
|
|
205
|
+
and so is a session or package fixture, whose memory is the worker's baseline."""
|
|
206
|
+
if is_baseline(key):
|
|
207
|
+
return 0
|
|
172
208
|
return self.retained.get(key, 0)
|
|
173
209
|
|
|
174
210
|
|
|
@@ -182,10 +218,11 @@ class Costs:
|
|
|
182
218
|
tests' families even when no run has timed it yet; one a recorded run reports a
|
|
183
219
|
test using without naming it (``getfixturevalue``) is matched by definition.
|
|
184
220
|
|
|
185
|
-
Memory comes from history alone: the bytes each test
|
|
186
|
-
|
|
187
|
-
its tests' families like a declared one, so the planner
|
|
188
|
-
lane that pays for it and the gate knows it is there.
|
|
221
|
+
Memory comes from history alone: the bytes each test needs of its own, and the
|
|
222
|
+
bytes each module- or class-scoped fixture instance keeps resident. A fixture kept
|
|
223
|
+
for its memory belongs to its tests' families like a declared one, so the planner
|
|
224
|
+
keeps it alive on the lane that pays for it and the gate knows it is there. What
|
|
225
|
+
session and package fixtures keep is every worker's baseline and is not charged.
|
|
189
226
|
"""
|
|
190
227
|
|
|
191
228
|
def __init__(
|