pytest-timing 0.3.1__tar.gz → 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/ARCHITECTURE.md +21 -1
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/CHANGELOG.md +33 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/PKG-INFO +76 -3
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/README.md +75 -2
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/pyproject.toml +1 -1
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/__init__.py +1 -1
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/cli.py +122 -3
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/collector.py +7 -0
- pytest_timing-0.4.0/src/pytest_timing/compare.py +200 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/demand.py +20 -10
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/model.py +24 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/outputs.py +4 -3
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/plugin.py +26 -4
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/schedule.py +54 -4
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/static/report.html +416 -73
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/xdist_scheduler.py +58 -33
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/conftest.py +11 -2
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_cli.py +20 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_collector.py +29 -0
- pytest_timing-0.4.0/tests/test_compare.py +207 -0
- pytest_timing-0.4.0/tests/test_html.py +461 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_memory_gate.py +28 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_plugin.py +42 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_runtime_admission.py +62 -2
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_schedule.py +128 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/uv.lock +1 -1
- pytest_timing-0.3.1/tests/test_html.py +0 -101
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/.github/workflows/ci.yml +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/.github/workflows/release.yml +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/.gitignore +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/LICENSE +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/benchmarks/cpu_bench.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/docs/report.png +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/examples/test_demo.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/admission.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/fixtures.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/render/__init__.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/render/ascii.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/render/html.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/render/trace.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/static/__init__.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/telemetry.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/src/pytest_timing/xdist_compat.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/fixtures/legacy_complete_false.json +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/fixtures/legacy_complete_true.json +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/fixtures/legacy_no_flag.json +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_admission.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_ascii.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_benchmarks.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_cpu.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_fixtures.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_memory.py +0 -0
- {pytest_timing-0.3.1 → pytest_timing-0.4.0}/tests/test_trace.py +0 -0
|
@@ -11,9 +11,10 @@ schedules a run from the previous one. For usage, see the [README](README.md).
|
|
|
11
11
|
| `collector.py` | Folds phase reports and worker events into a `Run`. Knows nothing about pytest objects. |
|
|
12
12
|
| `model.py` | The recorded run: `Run`, `Worker`, `TestSpan`, `Phase`, and the JSON representation. |
|
|
13
13
|
| `outputs.py` | The ASCII, HTML and trace renderers, in one table shared by the plugin and the CLI. |
|
|
14
|
-
| `cli.py` | `pytest-timing render` and `
|
|
14
|
+
| `cli.py` | `pytest-timing render`, `merge` and `compare`. |
|
|
15
15
|
| `fixtures.py` | Times shared fixture set-up where tests run and lists dependencies on reports. |
|
|
16
16
|
| `schedule.py` | Duration estimates, the fixture-aware cost model and the planner. Free of xdist. |
|
|
17
|
+
| `compare.py` | Saved-run comparison, final-attempt aggregation and per-test CI budgets. Free of pytest hooks and output I/O. |
|
|
17
18
|
| `demand.py` | Declared CPU demand: the `timing_cpu` marker, the fixture decorator, the declarations a worker sends, and the meter that records each test's CPU work and resident memory. |
|
|
18
19
|
| `admission.py` | One host's CPU budget, reservations, fair waiting line and pressure feedback. Free of xdist and platform reads. |
|
|
19
20
|
| `telemetry.py` | Platform reads: affinity and cgroup quota, CPU time and resident memory of a process tree, pressure and throttling. |
|
|
@@ -97,6 +98,14 @@ A missing or unreadable history file leaves xdist's scheduler in place unless an
|
|
|
97
98
|
explicit CPU budget was requested. With that budget the custom scheduler still runs,
|
|
98
99
|
using equal 1 ms test estimates. It supports only `load` and `worksteal`.
|
|
99
100
|
|
|
101
|
+
## Saved-run comparisons
|
|
102
|
+
|
|
103
|
+
Saved-run comparisons use a separate policy from scheduling estimates: compare
|
|
104
|
+
the median of final attempts, preserving missing metrics and rejecting incomplete
|
|
105
|
+
budget checks. Scheduling may choose uncontended historical observations instead.
|
|
106
|
+
Keeping these policies separate avoids treating a failed or missing observation as
|
|
107
|
+
a successful CI check merely because it was usable for queue planning.
|
|
108
|
+
|
|
100
109
|
## Cost model
|
|
101
110
|
|
|
102
111
|
A shared fixture makes the cost of a test depend on the worker. Every test has a
|
|
@@ -118,10 +127,16 @@ CPU work, peak demand and fixture transition. `Costs.project` folds these into a
|
|
|
118
127
|
|
|
119
128
|
The scheduler keeps one `Charge` per dispatched item. It records the holds already
|
|
120
129
|
included in the peak, so actual live holds supplement it without double counting.
|
|
130
|
+
It also records each fixture's contribution to CPU holds and retained memory.
|
|
131
|
+
Runtime admission previews and commits the same reconciliation: a fixture already
|
|
132
|
+
covered by history adds only a missing contribution, never its full cost again.
|
|
121
133
|
Charged duration, CPU work and the pre-dispatch fixture checkpoint remain historical
|
|
122
134
|
facts even when a runtime request changes the reservation. Stealing refunds the same
|
|
123
135
|
record and restores that checkpoint.
|
|
124
136
|
Ordinary transfers and steals compare finish times with the same CPU-work floor.
|
|
137
|
+
After placement, a stable module/class grouping is also tried within each worker's
|
|
138
|
+
assignment. It replaces exact-family ordering only when `Costs.project` predicts
|
|
139
|
+
less lane time and no more CPU work, so parameter locality can still win.
|
|
125
140
|
|
|
126
141
|
## Planning
|
|
127
142
|
|
|
@@ -364,6 +379,11 @@ as `runtime_wait`. All open fixture clocks pause during that wait, and test esti
|
|
|
364
379
|
subtract it along with shared setup. CPU `elapsed` and `work` exclude the wait and
|
|
365
380
|
its process-tree CPU work, so the measured rate covers execution. Older records
|
|
366
381
|
without `runtime_wait` default to zero; pre-start waits are not subtracted again.
|
|
382
|
+
The controller also attaches each wait's end epoch and monotonic duration as an
|
|
383
|
+
optional `TestSpan.admission_waits` interval on the run axis. Model rebasing moves
|
|
384
|
+
these intervals with phases and test spans. The HTML report displays them once
|
|
385
|
+
even when both gates refused the same interval. Older totals without intervals stay
|
|
386
|
+
unlocated; parked workers without an executed item are only in gate summaries.
|
|
367
387
|
Event handlers are registered before each worker starts, even when another plugin
|
|
368
388
|
selects the scheduler: without an admission gate a request is granted immediately.
|
|
369
389
|
The cgroup quota behind `auto` is read from the process's own group, walking up to
|
|
@@ -1,5 +1,38 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.4.0
|
|
4
|
+
|
|
5
|
+
- Keep slowest-test labels visible at the end of the timeline and wrap long labels
|
|
6
|
+
below their bars when neither side has room.
|
|
7
|
+
- Add `pytest-timing compare` with per-test duration, phase, CPU and memory changes,
|
|
8
|
+
optional CI regression budgets, machine-readable JSON and explicit unavailable checks.
|
|
9
|
+
- Add shared fixture costs by worker and linked time-range selection in HTML reports.
|
|
10
|
+
Zooming and scrolling resample the visible interval for finer detail.
|
|
11
|
+
- Record admission wait intervals on their test attempts, preserve them through
|
|
12
|
+
merging and show them in lanes, a wait graph and range statistics.
|
|
13
|
+
- Add `--timing-capture=light` to skip optional CPU and memory measurements while
|
|
14
|
+
retaining durations and resource admission; the default remains full capture.
|
|
15
|
+
- Bound HTML chart samples and ticks independently of zoom, retaining peak values
|
|
16
|
+
over each bucket; cache merged-test totals and top-ten hover details.
|
|
17
|
+
- Try module/class-contiguous ordering across overlapping fixture families without
|
|
18
|
+
changing worker assignments or increasing modeled CPU work.
|
|
19
|
+
- Trace-only output no longer builds an unused JSON report document.
|
|
20
|
+
- Runtime fixture admission no longer reserves a dynamic fixture's held CPU slots
|
|
21
|
+
or retained memory twice when it was already included in scheduling history.
|
|
22
|
+
- The HTML CPU graph spreads recorded CPU time over the full test span, including
|
|
23
|
+
runtime admission waits, so the graph preserves the recorded total. Test details
|
|
24
|
+
continue to show average CPU use during execution, with those waits excluded.
|
|
25
|
+
|
|
26
|
+
## 0.3.2
|
|
27
|
+
|
|
28
|
+
- The HTML report shows CPU and memory usage. The table gains "CPU time", "CPUs"
|
|
29
|
+
(the CPUs a test kept busy on average, beside the slots it declared) and "Memory"
|
|
30
|
+
(what it needed on top of its worker's footprint) columns, the hover details say
|
|
31
|
+
the same with what the measurement covered, the header totals the run and names
|
|
32
|
+
each host's budget, and two graphs show CPUs busy and resident memory over time,
|
|
33
|
+
estimated from the per-test records. All of it appears only where there was a
|
|
34
|
+
reading.
|
|
35
|
+
|
|
3
36
|
## 0.3.1
|
|
4
37
|
|
|
5
38
|
- Memory admission no longer charges what session- and package-scoped fixtures keep
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: pytest-timing
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.0
|
|
4
4
|
Summary: Record test timings under pytest-xdist and render cargo-style timing reports (ASCII Gantt and HTML).
|
|
5
5
|
Project-URL: Homepage, https://github.com/messense/pytest-timing
|
|
6
6
|
Author-email: messense <messense@icloud.com>
|
|
@@ -89,6 +89,7 @@ waited for. Each record identifies its measurement coverage.
|
|
|
89
89
|
| Option | Effect |
|
|
90
90
|
|---|---|
|
|
91
91
|
| `--timing` | Record timings and print the ASCII chart in the terminal summary. |
|
|
92
|
+
| `--timing-capture=full\|light` | Full metrics (default), or timings without CPU/memory measurement. Implies timing. |
|
|
92
93
|
| `--timing-html` | Write a self-contained HTML report to `pytest-timing.html`. |
|
|
93
94
|
| `--timing-json` | Write the recorded run to `pytest-timing.json`. |
|
|
94
95
|
| `--timing-trace` | Write a Chrome trace file for [Perfetto](https://ui.perfetto.dev) to `pytest-timing.trace.json`. |
|
|
@@ -109,7 +110,7 @@ and schedule paths are resolved from pytest's root directory.
|
|
|
109
110
|
|
|
110
111
|
The ini key `timing` is a boolean; `timing_html`, `timing_json` and `timing_trace`
|
|
111
112
|
accept paths or `true` for the default filenames. Other ini keys are
|
|
112
|
-
`timing_schedule`, `timing_cpus`, `timing_memory`, `timing_top`, `timing_min` and
|
|
113
|
+
`timing_capture`, `timing_schedule`, `timing_cpus`, `timing_memory`, `timing_top`, `timing_min` and
|
|
113
114
|
`timing_ascii_style`.
|
|
114
115
|
Environment variables use the uppercase key prefixed by `PYTEST_`, for example
|
|
115
116
|
`PYTEST_TIMING=1`, `PYTEST_TIMING_HTML=path` or `PYTEST_TIMING_CPUS=auto`.
|
|
@@ -119,6 +120,16 @@ For output paths, scheduling and formatting, the command line wins over the
|
|
|
119
120
|
environment, which wins over ini. Timing is enabled if any source requests it:
|
|
120
121
|
`PYTEST_TIMING=0` does not disable `timing=true` in ini or an enabled output.
|
|
121
122
|
|
|
123
|
+
Use `--timing-capture=light` (or `PYTEST_TIMING_CAPTURE=light`, or
|
|
124
|
+
`timing_capture = light` in ini) for timing-only collection. It keeps phase and
|
|
125
|
+
shared-fixture durations, worker lifecycle and admission waits, but does not create
|
|
126
|
+
a process-tree CPU reader or memory sampler. CPU work/rate, pressure and resident
|
|
127
|
+
memory measurements are unavailable; wait-only records have unknown measurement
|
|
128
|
+
coverage, rather than a measured zero. Declared CPU admission and memory admission
|
|
129
|
+
from an existing full report still apply. Capture stays `full` unless explicitly
|
|
130
|
+
changed; use full capture when refreshing resource estimates for future scheduling.
|
|
131
|
+
Capture mode names are case-insensitive in CLI, environment and ini settings.
|
|
132
|
+
|
|
122
133
|
## Outputs
|
|
123
134
|
|
|
124
135
|
**JSON** is the plugin's own record of the run: tests with their worker, phases and
|
|
@@ -138,7 +149,17 @@ lifecycles, detected host CPU environments, and how the session ended (`finished
|
|
|
138
149
|
the input to the CLI below and to `--timing-schedule`.
|
|
139
150
|
|
|
140
151
|
**HTML** is a single file with no external dependencies, so it can be attached to a CI
|
|
141
|
-
job as an artifact and opened anywhere.
|
|
152
|
+
job as an artifact and opened anywhere. Where CPU and memory were measured it shows
|
|
153
|
+
them: each test's CPU time, the CPUs it kept busy on average (beside the slots it
|
|
154
|
+
declared) and the memory it needed on top of its worker's footprint, in the table and
|
|
155
|
+
the hover details; the run's totals and budgets in the header; and CPU and memory
|
|
156
|
+
usage over time. The two graphs are drawn from the per-test records, not sampled: a
|
|
157
|
+
test's CPU time is spread evenly over its run, and a worker counts at its running
|
|
158
|
+
test's peak, so they show where the load was rather than its exact shape.
|
|
159
|
+
Each curve uses at most 4096 display samples, even at deep zoom. A sample keeps
|
|
160
|
+
the maximum over its time bucket; hover reports that bucket's time range. This
|
|
161
|
+
bounds display memory without dropping short peaks. Merged tiny-test bars cache
|
|
162
|
+
their total duration and ten longest tests for repeated hover.
|
|
142
163
|
|
|
143
164
|
**Trace** is a Chrome Trace Event file. Open it in
|
|
144
165
|
[Perfetto UI](https://ui.perfetto.dev) with "Open trace file", or in `chrome://tracing`.
|
|
@@ -349,6 +370,58 @@ pytest-timing merge shard1.json shard2.json -o all.json
|
|
|
349
370
|
`render` produces any of the outputs from a saved JSON run. `merge` places several runs
|
|
350
371
|
(for example CI shards) on one shared time axis using their absolute start times.
|
|
351
372
|
|
|
373
|
+
## Compare runs and enforce CI budgets
|
|
374
|
+
|
|
375
|
+
```
|
|
376
|
+
pytest-timing compare before.json after.json
|
|
377
|
+
pytest-timing compare before.json after.json --budget duration=10% --budget memory=64MiB --json comparison.json
|
|
378
|
+
```
|
|
379
|
+
|
|
380
|
+
Comparisons match tests by node ID and show total, setup, call and teardown time,
|
|
381
|
+
CPU time, and peak memory rise above the worker's baseline. For each run, only the
|
|
382
|
+
final retry of each worker/occurrence is used, then repeated observations are
|
|
383
|
+
combined with a median. The output includes retry counts and added/removed tests.
|
|
384
|
+
Missing measurements stay unavailable, including partial coverage across repeats.
|
|
385
|
+
|
|
386
|
+
A budget sets the maximum increase **per common test**. Percentages are relative
|
|
387
|
+
to the baseline; a zero baseline permits no increase under a percentage budget.
|
|
388
|
+
Time limits use seconds or `ms`; memory limits use bytes, `KiB`, `MiB` or `GiB`.
|
|
389
|
+
Each metric accepts one budget. Added and removed tests are listed but not budgeted.
|
|
390
|
+
Exit codes are `0` for success (or a comparison without budgets), `1` for exceeded
|
|
391
|
+
budgets, and `2` for invalid input or unavailable checks. Budget checks need complete
|
|
392
|
+
runs, at least one common test, successful final attempts and every requested metric.
|
|
393
|
+
Differences in Python, pytest, worker count or distribution mode produce warnings;
|
|
394
|
+
use comparable environments and repeat noisy workloads before choosing thresholds.
|
|
395
|
+
|
|
396
|
+
## Explore the HTML report
|
|
397
|
+
|
|
398
|
+
The scale control zooms all timelines together. Scrolling any timeline moves the
|
|
399
|
+
others to the same time. Curves are resampled from the visible window, so a narrow
|
|
400
|
+
peak gains detail as you zoom in, with at most 4,096 samples per curve. Hover values
|
|
401
|
+
are the maximum within the displayed bucket.
|
|
402
|
+
|
|
403
|
+
Enter start and end times in seconds and select **Select range** to fit that interval
|
|
404
|
+
across lanes, concurrency, admission waits, CPU and memory. The test and fixture
|
|
405
|
+
tables follow the selection and existing filters. **Clear range** returns to the
|
|
406
|
+
whole run. The range summary covers all tests, independently of text/outcome/fixture
|
|
407
|
+
filters; CPU totals integrate the original estimated curve, never its peak buckets.
|
|
408
|
+
CPU time is spread evenly over each test span, while the memory curve places the
|
|
409
|
+
recorded peak over that span; neither is a sampled execution trace.
|
|
410
|
+
|
|
411
|
+
The shared fixture table groups recorded setup costs by fixture key and worker,
|
|
412
|
+
showing repetition across tests and workers. Select a fixture to see its associated
|
|
413
|
+
tests and use **Clear fixture filter** to restore them. Costs belong to overlapping
|
|
414
|
+
tests: fixture setup timestamps are not recorded, so costs cannot be clipped to a
|
|
415
|
+
selected interval. A `None` fixture value means reuse and contributes no setup.
|
|
416
|
+
|
|
417
|
+
New reports include controller-observed `admission_waits` intervals on the exact
|
|
418
|
+
test attempt that waited. Lane strips and the wait graph display those intervals;
|
|
419
|
+
one interval refused by both CPU and memory counts once in the graph, summary and
|
|
420
|
+
Held column (its hover still lists both gates).
|
|
421
|
+
Older JSON files remain readable, but their wait totals cannot be placed on a
|
|
422
|
+
timeline. Workers parked without ever running a test remain in the run-level gate
|
|
423
|
+
summary rather than being assigned to an unrelated test.
|
|
424
|
+
|
|
352
425
|
## Overhead
|
|
353
426
|
|
|
354
427
|
Timing collection also measures fixtures and CPU work in each worker. Its overhead
|
|
@@ -63,6 +63,7 @@ waited for. Each record identifies its measurement coverage.
|
|
|
63
63
|
| Option | Effect |
|
|
64
64
|
|---|---|
|
|
65
65
|
| `--timing` | Record timings and print the ASCII chart in the terminal summary. |
|
|
66
|
+
| `--timing-capture=full\|light` | Full metrics (default), or timings without CPU/memory measurement. Implies timing. |
|
|
66
67
|
| `--timing-html` | Write a self-contained HTML report to `pytest-timing.html`. |
|
|
67
68
|
| `--timing-json` | Write the recorded run to `pytest-timing.json`. |
|
|
68
69
|
| `--timing-trace` | Write a Chrome trace file for [Perfetto](https://ui.perfetto.dev) to `pytest-timing.trace.json`. |
|
|
@@ -83,7 +84,7 @@ and schedule paths are resolved from pytest's root directory.
|
|
|
83
84
|
|
|
84
85
|
The ini key `timing` is a boolean; `timing_html`, `timing_json` and `timing_trace`
|
|
85
86
|
accept paths or `true` for the default filenames. Other ini keys are
|
|
86
|
-
`timing_schedule`, `timing_cpus`, `timing_memory`, `timing_top`, `timing_min` and
|
|
87
|
+
`timing_capture`, `timing_schedule`, `timing_cpus`, `timing_memory`, `timing_top`, `timing_min` and
|
|
87
88
|
`timing_ascii_style`.
|
|
88
89
|
Environment variables use the uppercase key prefixed by `PYTEST_`, for example
|
|
89
90
|
`PYTEST_TIMING=1`, `PYTEST_TIMING_HTML=path` or `PYTEST_TIMING_CPUS=auto`.
|
|
@@ -93,6 +94,16 @@ For output paths, scheduling and formatting, the command line wins over the
|
|
|
93
94
|
environment, which wins over ini. Timing is enabled if any source requests it:
|
|
94
95
|
`PYTEST_TIMING=0` does not disable `timing=true` in ini or an enabled output.
|
|
95
96
|
|
|
97
|
+
Use `--timing-capture=light` (or `PYTEST_TIMING_CAPTURE=light`, or
|
|
98
|
+
`timing_capture = light` in ini) for timing-only collection. It keeps phase and
|
|
99
|
+
shared-fixture durations, worker lifecycle and admission waits, but does not create
|
|
100
|
+
a process-tree CPU reader or memory sampler. CPU work/rate, pressure and resident
|
|
101
|
+
memory measurements are unavailable; wait-only records have unknown measurement
|
|
102
|
+
coverage, rather than a measured zero. Declared CPU admission and memory admission
|
|
103
|
+
from an existing full report still apply. Capture stays `full` unless explicitly
|
|
104
|
+
changed; use full capture when refreshing resource estimates for future scheduling.
|
|
105
|
+
Capture mode names are case-insensitive in CLI, environment and ini settings.
|
|
106
|
+
|
|
96
107
|
## Outputs
|
|
97
108
|
|
|
98
109
|
**JSON** is the plugin's own record of the run: tests with their worker, phases and
|
|
@@ -112,7 +123,17 @@ lifecycles, detected host CPU environments, and how the session ended (`finished
|
|
|
112
123
|
the input to the CLI below and to `--timing-schedule`.
|
|
113
124
|
|
|
114
125
|
**HTML** is a single file with no external dependencies, so it can be attached to a CI
|
|
115
|
-
job as an artifact and opened anywhere.
|
|
126
|
+
job as an artifact and opened anywhere. Where CPU and memory were measured it shows
|
|
127
|
+
them: each test's CPU time, the CPUs it kept busy on average (beside the slots it
|
|
128
|
+
declared) and the memory it needed on top of its worker's footprint, in the table and
|
|
129
|
+
the hover details; the run's totals and budgets in the header; and CPU and memory
|
|
130
|
+
usage over time. The two graphs are drawn from the per-test records, not sampled: a
|
|
131
|
+
test's CPU time is spread evenly over its run, and a worker counts at its running
|
|
132
|
+
test's peak, so they show where the load was rather than its exact shape.
|
|
133
|
+
Each curve uses at most 4096 display samples, even at deep zoom. A sample keeps
|
|
134
|
+
the maximum over its time bucket; hover reports that bucket's time range. This
|
|
135
|
+
bounds display memory without dropping short peaks. Merged tiny-test bars cache
|
|
136
|
+
their total duration and ten longest tests for repeated hover.
|
|
116
137
|
|
|
117
138
|
**Trace** is a Chrome Trace Event file. Open it in
|
|
118
139
|
[Perfetto UI](https://ui.perfetto.dev) with "Open trace file", or in `chrome://tracing`.
|
|
@@ -323,6 +344,58 @@ pytest-timing merge shard1.json shard2.json -o all.json
|
|
|
323
344
|
`render` produces any of the outputs from a saved JSON run. `merge` places several runs
|
|
324
345
|
(for example CI shards) on one shared time axis using their absolute start times.
|
|
325
346
|
|
|
347
|
+
## Compare runs and enforce CI budgets
|
|
348
|
+
|
|
349
|
+
```
|
|
350
|
+
pytest-timing compare before.json after.json
|
|
351
|
+
pytest-timing compare before.json after.json --budget duration=10% --budget memory=64MiB --json comparison.json
|
|
352
|
+
```
|
|
353
|
+
|
|
354
|
+
Comparisons match tests by node ID and show total, setup, call and teardown time,
|
|
355
|
+
CPU time, and peak memory rise above the worker's baseline. For each run, only the
|
|
356
|
+
final retry of each worker/occurrence is used, then repeated observations are
|
|
357
|
+
combined with a median. The output includes retry counts and added/removed tests.
|
|
358
|
+
Missing measurements stay unavailable, including partial coverage across repeats.
|
|
359
|
+
|
|
360
|
+
A budget sets the maximum increase **per common test**. Percentages are relative
|
|
361
|
+
to the baseline; a zero baseline permits no increase under a percentage budget.
|
|
362
|
+
Time limits use seconds or `ms`; memory limits use bytes, `KiB`, `MiB` or `GiB`.
|
|
363
|
+
Each metric accepts one budget. Added and removed tests are listed but not budgeted.
|
|
364
|
+
Exit codes are `0` for success (or a comparison without budgets), `1` for exceeded
|
|
365
|
+
budgets, and `2` for invalid input or unavailable checks. Budget checks need complete
|
|
366
|
+
runs, at least one common test, successful final attempts and every requested metric.
|
|
367
|
+
Differences in Python, pytest, worker count or distribution mode produce warnings;
|
|
368
|
+
use comparable environments and repeat noisy workloads before choosing thresholds.
|
|
369
|
+
|
|
370
|
+
## Explore the HTML report
|
|
371
|
+
|
|
372
|
+
The scale control zooms all timelines together. Scrolling any timeline moves the
|
|
373
|
+
others to the same time. Curves are resampled from the visible window, so a narrow
|
|
374
|
+
peak gains detail as you zoom in, with at most 4,096 samples per curve. Hover values
|
|
375
|
+
are the maximum within the displayed bucket.
|
|
376
|
+
|
|
377
|
+
Enter start and end times in seconds and select **Select range** to fit that interval
|
|
378
|
+
across lanes, concurrency, admission waits, CPU and memory. The test and fixture
|
|
379
|
+
tables follow the selection and existing filters. **Clear range** returns to the
|
|
380
|
+
whole run. The range summary covers all tests, independently of text/outcome/fixture
|
|
381
|
+
filters; CPU totals integrate the original estimated curve, never its peak buckets.
|
|
382
|
+
CPU time is spread evenly over each test span, while the memory curve places the
|
|
383
|
+
recorded peak over that span; neither is a sampled execution trace.
|
|
384
|
+
|
|
385
|
+
The shared fixture table groups recorded setup costs by fixture key and worker,
|
|
386
|
+
showing repetition across tests and workers. Select a fixture to see its associated
|
|
387
|
+
tests and use **Clear fixture filter** to restore them. Costs belong to overlapping
|
|
388
|
+
tests: fixture setup timestamps are not recorded, so costs cannot be clipped to a
|
|
389
|
+
selected interval. A `None` fixture value means reuse and contributes no setup.
|
|
390
|
+
|
|
391
|
+
New reports include controller-observed `admission_waits` intervals on the exact
|
|
392
|
+
test attempt that waited. Lane strips and the wait graph display those intervals;
|
|
393
|
+
one interval refused by both CPU and memory counts once in the graph, summary and
|
|
394
|
+
Held column (its hover still lists both gates).
|
|
395
|
+
Older JSON files remain readable, but their wait totals cannot be placed on a
|
|
396
|
+
timeline. Workers parked without ever running a test remain in the run-level gate
|
|
397
|
+
summary rather than being assigned to an unrelated test.
|
|
398
|
+
|
|
326
399
|
## Overhead
|
|
327
400
|
|
|
328
401
|
Timing collection also measures fixtures and CPU work in each worker. Its overhead
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "pytest-timing"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.4.0"
|
|
8
8
|
description = "Record test timings under pytest-xdist and render cargo-style timing reports (ASCII Gantt and HTML)."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -1,18 +1,106 @@
|
|
|
1
|
-
"""``pytest-timing`` command line:
|
|
1
|
+
"""``pytest-timing`` command line: render, merge or compare saved JSON runs."""
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import math
|
|
6
8
|
import sys
|
|
7
9
|
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
8
11
|
|
|
12
|
+
from pytest_timing.compare import compare_runs, format_comparison, parse_budget
|
|
9
13
|
from pytest_timing.model import Run, RunInfo, TestSpan, Worker
|
|
10
14
|
from pytest_timing.outputs import OUTPUTS, write_output
|
|
11
15
|
from pytest_timing.render.ascii import render_ascii
|
|
12
16
|
|
|
13
17
|
|
|
18
|
+
def _number(value: Any, where: str) -> None:
|
|
19
|
+
if not isinstance(value, (int, float, str)):
|
|
20
|
+
raise ValueError(f"{where} must be a number")
|
|
21
|
+
try:
|
|
22
|
+
finite = math.isfinite(float(value))
|
|
23
|
+
except OverflowError:
|
|
24
|
+
finite = False
|
|
25
|
+
if not finite:
|
|
26
|
+
raise ValueError(f"{where} must be finite")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _record(
|
|
30
|
+
value: Any, where: str, *, required: str = "", numbers: str = "", optional: str = ""
|
|
31
|
+
) -> dict[str, Any]:
|
|
32
|
+
if not isinstance(value, dict):
|
|
33
|
+
raise ValueError(f"{where} must be an object")
|
|
34
|
+
for key in required.split():
|
|
35
|
+
if key not in value:
|
|
36
|
+
raise ValueError(f"{where}.{key} is required")
|
|
37
|
+
for key in numbers.split():
|
|
38
|
+
if key in value:
|
|
39
|
+
_number(value[key], f"{where}.{key}")
|
|
40
|
+
for key in optional.split():
|
|
41
|
+
if value.get(key) is not None:
|
|
42
|
+
_number(value[key], f"{where}.{key}")
|
|
43
|
+
return value
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _array(value: Any, where: str) -> list[Any]:
|
|
47
|
+
if not isinstance(value, list):
|
|
48
|
+
raise ValueError(f"{where} must be an array")
|
|
49
|
+
return value
|
|
50
|
+
|
|
51
|
+
|
|
14
52
|
def _load(path: str) -> Run:
|
|
15
|
-
|
|
53
|
+
# Check external input before model conversion. Programming errors in the
|
|
54
|
+
# model, comparison or formatter must propagate rather than becoming exit 2.
|
|
55
|
+
doc = _record(
|
|
56
|
+
json.loads(Path(path).read_text("utf-8")), "report", required="run", numbers="schema"
|
|
57
|
+
)
|
|
58
|
+
info = _record(
|
|
59
|
+
doc["run"],
|
|
60
|
+
"run",
|
|
61
|
+
required="start stop",
|
|
62
|
+
numbers="start stop",
|
|
63
|
+
optional="numprocesses exit_status",
|
|
64
|
+
)
|
|
65
|
+
if info.get("termination") is not None and not isinstance(info["termination"], str):
|
|
66
|
+
raise ValueError("run.termination must be a string")
|
|
67
|
+
_array(info.get("argv", []), "run.argv")
|
|
68
|
+
for i, value in enumerate(_array(doc.get("workers", []), "workers")):
|
|
69
|
+
_record(value, f"workers[{i}]", required="id", optional="start ready collected down items")
|
|
70
|
+
for i, value in enumerate(_array(doc.get("tests", []), "tests")):
|
|
71
|
+
where = f"tests[{i}]"
|
|
72
|
+
test = _record(
|
|
73
|
+
value,
|
|
74
|
+
where,
|
|
75
|
+
required="nodeid worker outcome start stop",
|
|
76
|
+
numbers="start stop attempt occurrence",
|
|
77
|
+
)
|
|
78
|
+
for name, phase in _record(test.get("phases", {}), f"{where}.phases").items():
|
|
79
|
+
at = f"{where}.phases.{name}"
|
|
80
|
+
if len(_array(phase, at)) != 3:
|
|
81
|
+
raise ValueError(f"{at} must contain start, stop and duration")
|
|
82
|
+
for number in phase:
|
|
83
|
+
_number(number, at)
|
|
84
|
+
for key, seconds in _record(test.get("fixtures") or {}, f"{where}.fixtures").items():
|
|
85
|
+
if seconds is not None:
|
|
86
|
+
_number(seconds, f"{where}.fixtures.{key}")
|
|
87
|
+
if test.get("cpu") is not None:
|
|
88
|
+
_record(
|
|
89
|
+
test["cpu"],
|
|
90
|
+
f"{where}.cpu",
|
|
91
|
+
numbers="elapsed work setup_work demand wait runtime_wait",
|
|
92
|
+
optional="pressure",
|
|
93
|
+
)
|
|
94
|
+
if test.get("memory") is not None:
|
|
95
|
+
_record(test["memory"], f"{where}.memory", numbers="base peak after wait")
|
|
96
|
+
for j, value in enumerate(
|
|
97
|
+
_array(test.get("admission_waits", []), f"{where}.admission_waits")
|
|
98
|
+
):
|
|
99
|
+
at = f"{where}.admission_waits[{j}]"
|
|
100
|
+
wait = _record(value, at, required="start stop", numbers="start stop")
|
|
101
|
+
if not all(isinstance(g, str) for g in _array(wait.get("gates", []), f"{at}.gates")):
|
|
102
|
+
raise ValueError(f"{at}.gates must contain strings")
|
|
103
|
+
return Run.from_dict(doc)
|
|
16
104
|
|
|
17
105
|
|
|
18
106
|
def cmd_render(args: argparse.Namespace) -> int:
|
|
@@ -22,7 +110,8 @@ def cmd_render(args: argparse.Namespace) -> int:
|
|
|
22
110
|
for kind, output in OUTPUTS.items():
|
|
23
111
|
path = getattr(args, kind)
|
|
24
112
|
if path:
|
|
25
|
-
doc
|
|
113
|
+
if doc is None and output.needs_doc:
|
|
114
|
+
doc = run.to_dict()
|
|
26
115
|
print(write_output(output, Path(path), run, doc))
|
|
27
116
|
wrote = True
|
|
28
117
|
if args.ascii or not wrote:
|
|
@@ -39,6 +128,20 @@ def cmd_render(args: argparse.Namespace) -> int:
|
|
|
39
128
|
return 0
|
|
40
129
|
|
|
41
130
|
|
|
131
|
+
def cmd_compare(args: argparse.Namespace) -> int:
|
|
132
|
+
try:
|
|
133
|
+
result = compare_runs(_load(args.before), _load(args.after), args.budget)
|
|
134
|
+
if args.json:
|
|
135
|
+
Path(args.json).write_text(
|
|
136
|
+
json.dumps(result, indent=2, allow_nan=False), encoding="utf-8"
|
|
137
|
+
)
|
|
138
|
+
print(format_comparison(result, args.top))
|
|
139
|
+
return int(result["exit_code"])
|
|
140
|
+
except (OSError, ValueError) as exc:
|
|
141
|
+
print(f"pytest-timing compare: {exc}", file=sys.stderr)
|
|
142
|
+
return 2
|
|
143
|
+
|
|
144
|
+
|
|
42
145
|
def _uniquify(candidate: str, used: set[str]) -> str:
|
|
43
146
|
"""``candidate``, or ``candidate#2``, ``#3``... until it is not in ``used``."""
|
|
44
147
|
name, n = candidate, 1
|
|
@@ -137,6 +240,22 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
137
240
|
merge.add_argument("runs", nargs="+", help="pytest-timing JSON files")
|
|
138
241
|
merge.add_argument("-o", "--output", required=True, metavar="PATH")
|
|
139
242
|
merge.set_defaults(func=cmd_merge)
|
|
243
|
+
compare = sub.add_parser(
|
|
244
|
+
"compare", help="compare two saved runs and check per-test regression budgets"
|
|
245
|
+
)
|
|
246
|
+
compare.add_argument("before", help="baseline JSON run")
|
|
247
|
+
compare.add_argument("after", help="current JSON run")
|
|
248
|
+
compare.add_argument(
|
|
249
|
+
"--budget",
|
|
250
|
+
action="append",
|
|
251
|
+
type=parse_budget,
|
|
252
|
+
default=[],
|
|
253
|
+
metavar="METRIC=LIMIT",
|
|
254
|
+
help="maximum increase per common test, e.g. duration=10%%, cpu=50ms, memory=64MiB",
|
|
255
|
+
)
|
|
256
|
+
compare.add_argument("--json", metavar="PATH", help="write all differences and budget results")
|
|
257
|
+
compare.add_argument("--top", type=int, default=20, help="test details to print (default: 20)")
|
|
258
|
+
compare.set_defaults(func=cmd_compare)
|
|
140
259
|
return parser
|
|
141
260
|
|
|
142
261
|
|
|
@@ -26,6 +26,7 @@ from pytest_timing.model import (
|
|
|
26
26
|
Run,
|
|
27
27
|
RunInfo,
|
|
28
28
|
TestSpan,
|
|
29
|
+
WaitInterval,
|
|
29
30
|
Worker,
|
|
30
31
|
)
|
|
31
32
|
|
|
@@ -91,6 +92,7 @@ class Collector:
|
|
|
91
92
|
attempt: int,
|
|
92
93
|
seconds: float,
|
|
93
94
|
gates: Iterable[str] = (),
|
|
95
|
+
ended: float | None = None,
|
|
94
96
|
) -> None:
|
|
95
97
|
"""Add an admission delay to the exact execution that waited.
|
|
96
98
|
|
|
@@ -107,6 +109,11 @@ class Collector:
|
|
|
107
109
|
if span is None or span.outcome != "crashed" or span.phases or span.attempt != attempt:
|
|
108
110
|
return
|
|
109
111
|
gates = set(gates)
|
|
112
|
+
if ended is not None:
|
|
113
|
+
stop = self._rel(ended)
|
|
114
|
+
span.admission_waits.append(
|
|
115
|
+
WaitInterval(stop - seconds, stop, tuple(sorted(gates or {"cpu"})))
|
|
116
|
+
)
|
|
110
117
|
if "cpu" in gates or not gates:
|
|
111
118
|
if span.cpu is None:
|
|
112
119
|
span.cpu = CpuRecord(elapsed=span.duration)
|