scope-profiler 0.2.2__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/PKG-INFO +22 -4
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/README.md +16 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/pyproject.toml +7 -3
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/post_processing.py +88 -48
- scope_profiler-0.2.3/src/scope_profiler/prof_export.py +208 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/profile_manager.py +13 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/region_profiler.py +45 -6
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/test_app.py +2 -0
- scope_profiler-0.2.3/src/scope_profiler/tests/test_prof_export.py +225 -0
- scope_profiler-0.2.3/src/scope_profiler/tests/test_repeated_finalize.py +99 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler.egg-info/PKG-INFO +22 -4
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler.egg-info/SOURCES.txt +4 -1
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler.egg-info/requires.txt +6 -3
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/setup.cfg +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/__init__.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/__main__.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/h5reader.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/inspection.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/metadata.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/mpi_region.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/plotting_scripts.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/profile_config.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/region.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/summary.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/__init__.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/examples.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/examples_pylikwid.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/pylikwid_readme.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/test_buffer_growth.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/test_inspection.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/test_metadata.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/test_mpi.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/test_overhead.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/test_post_processing.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/test_reader_api.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/test_readme.py +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler.egg-info/dependency_links.txt +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler.egg-info/entry_points.txt +0 -0
- {scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scope-profiler
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: Profile code regions in python, optionally with LIKWID markers.
|
|
5
5
|
Author: Max
|
|
6
6
|
Project-URL: Source, https://github.com/max-models/scope-profiler
|
|
@@ -16,7 +16,8 @@ Requires-Python: >=3.10
|
|
|
16
16
|
Description-Content-Type: text/markdown
|
|
17
17
|
Requires-Dist: h5py
|
|
18
18
|
Requires-Dist: numpy
|
|
19
|
-
|
|
19
|
+
Provides-Extra: line-profiler
|
|
20
|
+
Requires-Dist: line-profiler; extra == "line-profiler"
|
|
20
21
|
Provides-Extra: mpi
|
|
21
22
|
Requires-Dist: mpi4py; extra == "mpi"
|
|
22
23
|
Provides-Extra: pproc
|
|
@@ -25,11 +26,12 @@ Requires-Dist: jupyterlab; extra == "pproc"
|
|
|
25
26
|
Requires-Dist: matplotlib; extra == "pproc"
|
|
26
27
|
Requires-Dist: maxplotlibx>=0.1.5; extra == "pproc"
|
|
27
28
|
Requires-Dist: pandas; extra == "pproc"
|
|
29
|
+
Requires-Dist: snakeviz; extra == "pproc"
|
|
28
30
|
Provides-Extra: dev
|
|
29
31
|
Requires-Dist: black[jupyter]; extra == "dev"
|
|
30
32
|
Requires-Dist: isort; extra == "dev"
|
|
31
33
|
Requires-Dist: ruff; extra == "dev"
|
|
32
|
-
Requires-Dist: scope-profiler[docs,mpi,pproc,test]; extra == "dev"
|
|
34
|
+
Requires-Dist: scope-profiler[docs,line-profiler,mpi,pproc,test]; extra == "dev"
|
|
33
35
|
Provides-Extra: docs
|
|
34
36
|
Requires-Dist: myst-parser; extra == "docs"
|
|
35
37
|
Requires-Dist: nbconvert; extra == "docs"
|
|
@@ -38,7 +40,7 @@ Requires-Dist: pre-commit; extra == "docs"
|
|
|
38
40
|
Requires-Dist: pyproject-fmt; extra == "docs"
|
|
39
41
|
Requires-Dist: sphinx; extra == "docs"
|
|
40
42
|
Requires-Dist: sphinx-book-theme; extra == "docs"
|
|
41
|
-
Requires-Dist: scope-profiler[pproc]; extra == "docs"
|
|
43
|
+
Requires-Dist: scope-profiler[line-profiler,pproc]; extra == "docs"
|
|
42
44
|
Provides-Extra: test
|
|
43
45
|
Requires-Dist: coverage; extra == "test"
|
|
44
46
|
Requires-Dist: pytest; extra == "test"
|
|
@@ -427,3 +429,19 @@ straight from the JSON:
|
|
|
427
429
|
scope-profiler pproc profiling_data.h5 -o figures \
|
|
428
430
|
--export-data --export-data-format json --skip-plot-images
|
|
429
431
|
```
|
|
432
|
+
|
|
433
|
+
### Viewing a run in snakeviz
|
|
434
|
+
|
|
435
|
+
`--export-prof` writes the profile in the `.prof` format of the standard
|
|
436
|
+
library's `cProfile`, so a run can be explored with
|
|
437
|
+
[snakeviz](https://jiffyclub.github.io/snakeviz/) or `python -m pstats`:
|
|
438
|
+
|
|
439
|
+
```bash
|
|
440
|
+
scope-profiler pproc profiling_data.h5 -o figures --export-prof --skip-plot-images
|
|
441
|
+
snakeviz figures/profile_rank0.prof
|
|
442
|
+
```
|
|
443
|
+
|
|
444
|
+
Regions become "functions": `cumtime` is a region's total wall time and
|
|
445
|
+
`tottime` is that minus the time spent in its nested regions. One file is
|
|
446
|
+
written per exported rank, since `.prof` has no notion of ranks — see
|
|
447
|
+
[the CLI docs](docs/source/cli.md) for the caveats of the reconstruction.
|
|
@@ -382,3 +382,19 @@ straight from the JSON:
|
|
|
382
382
|
scope-profiler pproc profiling_data.h5 -o figures \
|
|
383
383
|
--export-data --export-data-format json --skip-plot-images
|
|
384
384
|
```
|
|
385
|
+
|
|
386
|
+
### Viewing a run in snakeviz
|
|
387
|
+
|
|
388
|
+
`--export-prof` writes the profile in the `.prof` format of the standard
|
|
389
|
+
library's `cProfile`, so a run can be explored with
|
|
390
|
+
[snakeviz](https://jiffyclub.github.io/snakeviz/) or `python -m pstats`:
|
|
391
|
+
|
|
392
|
+
```bash
|
|
393
|
+
scope-profiler pproc profiling_data.h5 -o figures --export-prof --skip-plot-images
|
|
394
|
+
snakeviz figures/profile_rank0.prof
|
|
395
|
+
```
|
|
396
|
+
|
|
397
|
+
Regions become "functions": `cumtime` is a region's total wall time and
|
|
398
|
+
`tottime` is that minus the time spent in its nested regions. One file is
|
|
399
|
+
written per exported rank, since `.prof` has no notion of ranks — see
|
|
400
|
+
[the CLI docs](docs/source/cli.md) for the caveats of the reconstruction.
|
|
@@ -5,7 +5,7 @@ requires = [ "setuptools", "wheel" ]
|
|
|
5
5
|
|
|
6
6
|
[project]
|
|
7
7
|
name = "scope-profiler"
|
|
8
|
-
version = "0.2.
|
|
8
|
+
version = "0.2.3"
|
|
9
9
|
description = "Profile code regions in python, optionally with LIKWID markers."
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
keywords = [ "python" ]
|
|
@@ -24,6 +24,9 @@ classifiers = [
|
|
|
24
24
|
dependencies = [
|
|
25
25
|
"h5py",
|
|
26
26
|
"numpy",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
optional-dependencies.line-profiler = [
|
|
27
30
|
"line-profiler",
|
|
28
31
|
]
|
|
29
32
|
|
|
@@ -37,13 +40,14 @@ optional-dependencies.pproc = [
|
|
|
37
40
|
"matplotlib",
|
|
38
41
|
"maxplotlibx >= 0.1.5",
|
|
39
42
|
"pandas",
|
|
43
|
+
"snakeviz",
|
|
40
44
|
]
|
|
41
45
|
|
|
42
46
|
optional-dependencies.dev = [
|
|
43
47
|
"black[jupyter]",
|
|
44
48
|
"isort",
|
|
45
49
|
"ruff",
|
|
46
|
-
"scope-profiler[docs,
|
|
50
|
+
"scope-profiler[docs,line-profiler,mpi,pproc,test]",
|
|
47
51
|
]
|
|
48
52
|
# https://medium.com/@pratikdomadiya123/build-project-documentation-quickly-with-the-sphinx-python-2a9732b66594
|
|
49
53
|
optional-dependencies.docs = [
|
|
@@ -54,7 +58,7 @@ optional-dependencies.docs = [
|
|
|
54
58
|
"pyproject-fmt",
|
|
55
59
|
"sphinx",
|
|
56
60
|
"sphinx-book-theme",
|
|
57
|
-
"scope-profiler[pproc]",
|
|
61
|
+
"scope-profiler[line-profiler,pproc]",
|
|
58
62
|
]
|
|
59
63
|
optional-dependencies.test = [
|
|
60
64
|
"coverage",
|
|
@@ -13,6 +13,7 @@ from scope_profiler.plotting_scripts import (
|
|
|
13
13
|
plot_speedup,
|
|
14
14
|
write_region_statistics_json,
|
|
15
15
|
)
|
|
16
|
+
from scope_profiler.prof_export import export_prof
|
|
16
17
|
|
|
17
18
|
|
|
18
19
|
def parse_ranks(spec: str, verbose: bool = False) -> list[int]:
|
|
@@ -155,14 +156,26 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
155
156
|
"consistent colors."
|
|
156
157
|
),
|
|
157
158
|
)
|
|
159
|
+
parser.add_argument(
|
|
160
|
+
"--export-prof",
|
|
161
|
+
action="store_true",
|
|
162
|
+
help=(
|
|
163
|
+
"Also write one profile_rank<N>.prof file per exported rank in the "
|
|
164
|
+
"cProfile/pstats format, so the run can be browsed with external "
|
|
165
|
+
"tools (`snakeviz profile_rank0.prof`, `python -m pstats ...`). "
|
|
166
|
+
"The call graph is reconstructed from region nesting; only ranks "
|
|
167
|
+
"selected with --ranks are exported (default: rank 0). Requires "
|
|
168
|
+
"-o/--output."
|
|
169
|
+
),
|
|
170
|
+
)
|
|
158
171
|
parser.add_argument(
|
|
159
172
|
"--skip-plot-images",
|
|
160
173
|
action="store_true",
|
|
161
174
|
help=(
|
|
162
175
|
"Do not render/save the PNG plot images, only the outputs from "
|
|
163
|
-
"--export-data. Useful when charts are rendered
|
|
164
|
-
"side (e.g. with Plotly) from the exported data.
|
|
165
|
-
"--export-data."
|
|
176
|
+
"--export-data/--export-prof. Useful when charts are rendered "
|
|
177
|
+
"entirely client-side (e.g. with Plotly) from the exported data. "
|
|
178
|
+
"Requires --export-data or --export-prof."
|
|
166
179
|
),
|
|
167
180
|
)
|
|
168
181
|
return parser
|
|
@@ -203,8 +216,11 @@ def main(argv: list[str] | None = None):
|
|
|
203
216
|
if args.export_data and not args.output:
|
|
204
217
|
parser.error("--export-data requires -o/--output.")
|
|
205
218
|
|
|
206
|
-
if args.
|
|
207
|
-
parser.error("--
|
|
219
|
+
if args.export_prof and not args.output:
|
|
220
|
+
parser.error("--export-prof requires -o/--output.")
|
|
221
|
+
|
|
222
|
+
if args.skip_plot_images and not (args.export_data or args.export_prof):
|
|
223
|
+
parser.error("--skip-plot-images requires --export-data or --export-prof.")
|
|
208
224
|
|
|
209
225
|
if args.ranks:
|
|
210
226
|
ranks = []
|
|
@@ -243,6 +259,9 @@ def main(argv: list[str] | None = None):
|
|
|
243
259
|
flame_data_path = None
|
|
244
260
|
durations_data_path = None
|
|
245
261
|
speedup_data_path = None
|
|
262
|
+
prof_path = None
|
|
263
|
+
prof_paths: list = []
|
|
264
|
+
durations_paths: list = []
|
|
246
265
|
if args.output:
|
|
247
266
|
os.makedirs(args.output, exist_ok=True)
|
|
248
267
|
if not args.skip_plot_images:
|
|
@@ -266,62 +285,80 @@ def main(argv: list[str] | None = None):
|
|
|
266
285
|
speedup_data_path = os.path.join(
|
|
267
286
|
args.output, f"speedup_data.{data_ext}"
|
|
268
287
|
)
|
|
288
|
+
if args.export_prof:
|
|
289
|
+
prof_path = os.path.join(args.output, "profile.prof")
|
|
269
290
|
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
include=args.include,
|
|
275
|
-
exclude=args.exclude,
|
|
276
|
-
ranks=args.ranks,
|
|
277
|
-
cmap=args.cmap,
|
|
278
|
-
data_filepath=gantt_data_path,
|
|
279
|
-
data_format=args.export_data_format,
|
|
280
|
-
backend=args.backend,
|
|
281
|
-
)
|
|
291
|
+
# --skip-plot-images still needs the plotting functions to produce the
|
|
292
|
+
# --export-data files, but a prof-only run should not touch the plotting
|
|
293
|
+
# backend at all.
|
|
294
|
+
render_plots = not args.skip_plot_images or args.export_data
|
|
282
295
|
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
data_format=args.export_data_format,
|
|
293
|
-
backend=args.backend,
|
|
294
|
-
)
|
|
296
|
+
if args.export_prof:
|
|
297
|
+
prof_paths = export_prof(
|
|
298
|
+
profiling_data=readers,
|
|
299
|
+
filepath=prof_path,
|
|
300
|
+
ranks=args.ranks,
|
|
301
|
+
include=args.include,
|
|
302
|
+
exclude=args.exclude,
|
|
303
|
+
verbose=False,
|
|
304
|
+
)
|
|
295
305
|
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
306
|
+
if render_plots:
|
|
307
|
+
plot_gantt(
|
|
308
|
+
profiling_data=readers,
|
|
309
|
+
filepath=gantt_path,
|
|
310
|
+
show=args.show,
|
|
311
|
+
include=args.include,
|
|
312
|
+
exclude=args.exclude,
|
|
313
|
+
ranks=args.ranks,
|
|
314
|
+
cmap=args.cmap,
|
|
315
|
+
data_filepath=gantt_data_path,
|
|
316
|
+
data_format=args.export_data_format,
|
|
317
|
+
backend=args.backend,
|
|
318
|
+
)
|
|
309
319
|
|
|
310
|
-
|
|
311
|
-
plot_speedup(
|
|
320
|
+
plot_flame(
|
|
312
321
|
profiling_data=readers,
|
|
313
|
-
|
|
322
|
+
filepath=flame_path,
|
|
323
|
+
show=args.show,
|
|
324
|
+
include=args.include,
|
|
325
|
+
exclude=args.exclude,
|
|
314
326
|
ranks=args.ranks,
|
|
315
|
-
|
|
327
|
+
cmap=args.cmap,
|
|
328
|
+
data_filepath=flame_data_path,
|
|
329
|
+
data_format=args.export_data_format,
|
|
330
|
+
backend=args.backend,
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
durations_paths = plot_durations(
|
|
334
|
+
profiling_data=readers,
|
|
335
|
+
filepath=durations_path,
|
|
316
336
|
show=args.show,
|
|
317
337
|
include=args.include,
|
|
318
338
|
exclude=args.exclude,
|
|
339
|
+
ranks=args.ranks,
|
|
340
|
+
metrics=args.metrics,
|
|
319
341
|
cmap=args.cmap,
|
|
320
|
-
data_filepath=
|
|
342
|
+
data_filepath=durations_data_path,
|
|
321
343
|
data_format=args.export_data_format,
|
|
322
344
|
backend=args.backend,
|
|
323
345
|
)
|
|
324
346
|
|
|
347
|
+
if len(readers) > 1:
|
|
348
|
+
plot_speedup(
|
|
349
|
+
profiling_data=readers,
|
|
350
|
+
x_field=args.x_field,
|
|
351
|
+
ranks=args.ranks,
|
|
352
|
+
filepath=speedup_path,
|
|
353
|
+
show=args.show,
|
|
354
|
+
include=args.include,
|
|
355
|
+
exclude=args.exclude,
|
|
356
|
+
cmap=args.cmap,
|
|
357
|
+
data_filepath=speedup_data_path,
|
|
358
|
+
data_format=args.export_data_format,
|
|
359
|
+
backend=args.backend,
|
|
360
|
+
)
|
|
361
|
+
|
|
325
362
|
if statistics_path:
|
|
326
363
|
write_region_statistics_json(
|
|
327
364
|
profiling_data=readers,
|
|
@@ -333,7 +370,7 @@ def main(argv: list[str] | None = None):
|
|
|
333
370
|
|
|
334
371
|
if args.output and not args.show:
|
|
335
372
|
saved = [
|
|
336
|
-
path
|
|
373
|
+
str(path)
|
|
337
374
|
for path in (
|
|
338
375
|
gantt_path,
|
|
339
376
|
flame_path,
|
|
@@ -344,10 +381,13 @@ def main(argv: list[str] | None = None):
|
|
|
344
381
|
flame_data_path,
|
|
345
382
|
durations_data_path,
|
|
346
383
|
speedup_data_path,
|
|
384
|
+
*prof_paths,
|
|
347
385
|
)
|
|
348
386
|
if path
|
|
349
387
|
]
|
|
350
388
|
print("Outputs saved to:\n " + "\n ".join(saved))
|
|
389
|
+
if prof_paths:
|
|
390
|
+
print(f"\nView a .prof file with: snakeviz {prof_paths[0]}")
|
|
351
391
|
|
|
352
392
|
|
|
353
393
|
if __name__ == "__main__":
|
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
"""Export merged HDF5 profiling data to the cProfile ``.prof`` (pstats) format.
|
|
2
|
+
|
|
3
|
+
A ``.prof`` file is nothing more than a :mod:`marshal` dump of the dict that
|
|
4
|
+
``cProfile.Profile.dump_stats`` writes: it maps a ``(filename, lineno,
|
|
5
|
+
funcname)`` key to ``(cc, nc, tt, ct, callers)``, where ``callers`` maps a
|
|
6
|
+
caller key to its own ``(cc, nc, tt, ct)`` sub-tuple. Writing that dict is
|
|
7
|
+
enough for :mod:`pstats`, ``snakeviz`` and friends to read the data.
|
|
8
|
+
|
|
9
|
+
Regions carry no call graph of their own, so the caller/callee relations are
|
|
10
|
+
reconstructed from timestamp containment - the same reconstruction the flame
|
|
11
|
+
chart uses (:func:`~scope_profiler.plotting_scripts._build_call_stack_intervals`).
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import marshal
|
|
17
|
+
from collections import defaultdict
|
|
18
|
+
from collections.abc import Sequence
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from scope_profiler.h5reader import ProfilingH5Reader
|
|
22
|
+
from scope_profiler.plotting_scripts import (
|
|
23
|
+
_as_readers,
|
|
24
|
+
_build_call_stack_intervals,
|
|
25
|
+
_normalize_ranks,
|
|
26
|
+
_unique_labels,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
# pstats keys are (filename, lineno, funcname) triples, and
|
|
30
|
+
# ``pstats.func_std_string`` renders a key starting with ("~", 0) as the bare
|
|
31
|
+
# function name - the convention cProfile uses for builtins. Regions have no
|
|
32
|
+
# source location, so borrowing it keeps them labelled "solve" rather than
|
|
33
|
+
# "profiling_data.h5:0(solve)" in snakeviz.
|
|
34
|
+
PSEUDO_FILENAME = "~"
|
|
35
|
+
PSEUDO_LINENO = 0
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _key(name: str) -> tuple[str, int, str]:
|
|
39
|
+
"""Build the pstats key identifying a region by name."""
|
|
40
|
+
return (PSEUDO_FILENAME, PSEUDO_LINENO, name)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _ancestor_names(calls: list[dict], index: int) -> set[str]:
|
|
44
|
+
"""Names of all calls enclosing ``calls[index]``."""
|
|
45
|
+
names = set()
|
|
46
|
+
parent = calls[index]["parent"]
|
|
47
|
+
while parent is not None:
|
|
48
|
+
names.add(calls[parent]["name"])
|
|
49
|
+
parent = calls[parent]["parent"]
|
|
50
|
+
return names
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def build_pstats_dict(
|
|
54
|
+
calls: list[dict], root_name: str | None = None
|
|
55
|
+
) -> dict[tuple[str, int, str], tuple]:
|
|
56
|
+
"""Turn reconstructed calls into a pstats-format statistics dict.
|
|
57
|
+
|
|
58
|
+
Parameters
|
|
59
|
+
----------
|
|
60
|
+
calls : list[dict]
|
|
61
|
+
Calls as returned by
|
|
62
|
+
:func:`~scope_profiler.plotting_scripts._build_call_stack_intervals`:
|
|
63
|
+
each entry has ``name``, ``start`` and ``end`` in seconds, and
|
|
64
|
+
``parent`` (an index into this list, or ``None`` for a top-level call).
|
|
65
|
+
root_name : str, optional
|
|
66
|
+
When given, a synthetic frame of this name is added as the caller of
|
|
67
|
+
every top-level call, so viewers that draw a single tree (snakeviz)
|
|
68
|
+
show the whole run instead of only its largest region.
|
|
69
|
+
|
|
70
|
+
Returns
|
|
71
|
+
-------
|
|
72
|
+
dict
|
|
73
|
+
Maps ``(filename, lineno, funcname)`` to
|
|
74
|
+
``(cc, nc, tt, ct, callers)`` with times in seconds.
|
|
75
|
+
"""
|
|
76
|
+
child_time: dict[int, float] = defaultdict(float)
|
|
77
|
+
for call in calls:
|
|
78
|
+
if call["parent"] is not None:
|
|
79
|
+
child_time[call["parent"]] += call["end"] - call["start"]
|
|
80
|
+
|
|
81
|
+
root_key = _key(root_name) if root_name is not None else None
|
|
82
|
+
# [primitive_calls, total_calls, self_time, cumulative_time, callers]
|
|
83
|
+
stats: dict[tuple[str, int, str], list] = {}
|
|
84
|
+
root_cumulative = 0.0
|
|
85
|
+
|
|
86
|
+
for index, call in enumerate(calls):
|
|
87
|
+
duration = call["end"] - call["start"]
|
|
88
|
+
# Regions that only partially overlap their enclosing region are
|
|
89
|
+
# reconstructed as children of it, which can push the parent's self
|
|
90
|
+
# time below zero; pstats has no meaning for a negative tt.
|
|
91
|
+
self_time = max(duration - child_time[index], 0.0)
|
|
92
|
+
# pstats counts cumulative time for the outermost call of a recursive
|
|
93
|
+
# chain only, otherwise the same seconds are counted at every level.
|
|
94
|
+
primitive = call["name"] not in _ancestor_names(calls, index)
|
|
95
|
+
cumulative = duration if primitive else 0.0
|
|
96
|
+
|
|
97
|
+
entry = stats.setdefault(_key(call["name"]), [0, 0, 0.0, 0.0, {}])
|
|
98
|
+
entry[0] += int(primitive)
|
|
99
|
+
entry[1] += 1
|
|
100
|
+
entry[2] += self_time
|
|
101
|
+
entry[3] += cumulative
|
|
102
|
+
|
|
103
|
+
if call["parent"] is not None:
|
|
104
|
+
caller_key = _key(calls[call["parent"]]["name"])
|
|
105
|
+
else:
|
|
106
|
+
caller_key = root_key
|
|
107
|
+
root_cumulative += duration
|
|
108
|
+
if caller_key is not None:
|
|
109
|
+
caller = entry[4].setdefault(caller_key, [0, 0, 0.0, 0.0])
|
|
110
|
+
caller[0] += int(primitive)
|
|
111
|
+
caller[1] += 1
|
|
112
|
+
caller[2] += self_time
|
|
113
|
+
caller[3] += cumulative
|
|
114
|
+
|
|
115
|
+
if root_key is not None and calls:
|
|
116
|
+
stats[root_key] = [1, 1, 0.0, root_cumulative, {}]
|
|
117
|
+
|
|
118
|
+
return {
|
|
119
|
+
key: (
|
|
120
|
+
value[0],
|
|
121
|
+
value[1],
|
|
122
|
+
value[2],
|
|
123
|
+
value[3],
|
|
124
|
+
{caller: tuple(times) for caller, times in value[4].items()},
|
|
125
|
+
)
|
|
126
|
+
for key, value in stats.items()
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def write_prof_file(
|
|
131
|
+
filepath: str | Path, stats: dict[tuple[str, int, str], tuple]
|
|
132
|
+
) -> Path:
|
|
133
|
+
"""Marshal a pstats dict to ``filepath``, as ``cProfile`` would."""
|
|
134
|
+
output_path = Path(filepath)
|
|
135
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
136
|
+
with open(output_path, "wb") as f:
|
|
137
|
+
marshal.dump(stats, f)
|
|
138
|
+
return output_path
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def export_prof(
|
|
142
|
+
profiling_data: ProfilingH5Reader | Sequence[ProfilingH5Reader],
|
|
143
|
+
filepath: str | Path,
|
|
144
|
+
ranks: list[int] | int | None = None,
|
|
145
|
+
include: list[str] | str | None = None,
|
|
146
|
+
exclude: list[str] | str | None = None,
|
|
147
|
+
verbose: bool = True,
|
|
148
|
+
) -> list[Path]:
|
|
149
|
+
"""Write per-rank ``.prof`` files readable by ``pstats``/``snakeviz``.
|
|
150
|
+
|
|
151
|
+
Parameters
|
|
152
|
+
----------
|
|
153
|
+
profiling_data : ProfilingH5Reader | Sequence[ProfilingH5Reader]
|
|
154
|
+
Reader(s) for the merged HDF5 file(s) to export.
|
|
155
|
+
filepath : str | Path
|
|
156
|
+
Base output path, e.g. ``figures/profile.prof``. A ``_rank<N>`` suffix
|
|
157
|
+
is appended per rank (and the input file's stem too, when more than one
|
|
158
|
+
file is exported), since ``.prof`` has no notion of ranks or runs.
|
|
159
|
+
ranks : list[int] | int, optional
|
|
160
|
+
Ranks to export (default: rank 0 only).
|
|
161
|
+
include, exclude : list[str] | str, optional
|
|
162
|
+
Region name filters, as for the plotting functions.
|
|
163
|
+
|
|
164
|
+
Returns
|
|
165
|
+
-------
|
|
166
|
+
list[Path]
|
|
167
|
+
The files written, in the order they were written.
|
|
168
|
+
"""
|
|
169
|
+
readers = _as_readers(profiling_data)
|
|
170
|
+
if not readers:
|
|
171
|
+
raise ValueError("No profiling data provided.")
|
|
172
|
+
|
|
173
|
+
normalized_ranks = _normalize_ranks(ranks) if ranks is not None else [0]
|
|
174
|
+
|
|
175
|
+
labels = _unique_labels([reader.file_path.stem for reader in readers])
|
|
176
|
+
|
|
177
|
+
prepared = []
|
|
178
|
+
for label, reader in zip(labels, readers):
|
|
179
|
+
regions = reader.get_regions(include=include, exclude=exclude)
|
|
180
|
+
if not regions:
|
|
181
|
+
raise ValueError("No regions matched the selected filters.")
|
|
182
|
+
for rank in normalized_ranks:
|
|
183
|
+
if rank < 0 or rank >= reader.num_ranks:
|
|
184
|
+
raise ValueError(f"Invalid rank requested: {rank}")
|
|
185
|
+
calls = _build_call_stack_intervals(regions, rank)
|
|
186
|
+
if calls:
|
|
187
|
+
prepared.append((label, rank, calls))
|
|
188
|
+
|
|
189
|
+
if not prepared:
|
|
190
|
+
raise ValueError("No calls recorded for the requested ranks.")
|
|
191
|
+
|
|
192
|
+
base_path = Path(filepath)
|
|
193
|
+
suffix = base_path.suffix or ".prof"
|
|
194
|
+
multiple_files = len(readers) > 1
|
|
195
|
+
|
|
196
|
+
written = []
|
|
197
|
+
for label, rank, calls in prepared:
|
|
198
|
+
parts = [base_path.stem]
|
|
199
|
+
if multiple_files:
|
|
200
|
+
parts.append(label)
|
|
201
|
+
parts.append(f"rank{rank}")
|
|
202
|
+
out_path = base_path.with_name("_".join(parts) + suffix)
|
|
203
|
+
stats = build_pstats_dict(calls, root_name=f"<{label} rank {rank}>")
|
|
204
|
+
written.append(write_prof_file(out_path, stats))
|
|
205
|
+
if verbose:
|
|
206
|
+
print(f"Wrote {out_path} (view with: snakeviz {out_path})")
|
|
207
|
+
|
|
208
|
+
return written
|
|
@@ -376,13 +376,26 @@ class ProfileManager:
|
|
|
376
376
|
rank = config._rank
|
|
377
377
|
size = config._size
|
|
378
378
|
|
|
379
|
+
# 0. Discard any per-rank file left by an earlier finalize() on this
|
|
380
|
+
# config. Nothing else writes it, so its presence means a previous run
|
|
381
|
+
# in this process already finalized, and its regions would otherwise be
|
|
382
|
+
# merged into this run's output alongside the current ones.
|
|
383
|
+
stale_file = config._local_file_path
|
|
384
|
+
if os.path.exists(stale_file):
|
|
385
|
+
os.remove(stale_file)
|
|
386
|
+
|
|
379
387
|
# 1. Write every region's buffered timestamps to its per-rank file.
|
|
380
388
|
# Regions that record no timestamps write only their call count, which
|
|
381
389
|
# is cheap and has nothing to do with timing buffers, so they write
|
|
382
390
|
# even when flush_to_disk is off. Otherwise their counts would be lost.
|
|
391
|
+
# Once written, a region is marked so finalize() acts as a run
|
|
392
|
+
# boundary: a second run in the same process (e.g. a restart) writes
|
|
393
|
+
# only its own events. Regions that were not written keep their
|
|
394
|
+
# buffers, because with flush_to_disk off those are the only copy.
|
|
383
395
|
for region in cls.get_all_regions().values():
|
|
384
396
|
if config.flush_to_disk or not region._records_time:
|
|
385
397
|
region.write_to_disk()
|
|
398
|
+
region.mark_written()
|
|
386
399
|
|
|
387
400
|
# 2. Barrier to ensure all ranks finished writing
|
|
388
401
|
if comm is not None:
|
|
@@ -27,10 +27,17 @@ def _import_pylikwid():
|
|
|
27
27
|
def _import_line_profiler():
|
|
28
28
|
"""Import and return the LineProfiler class from line_profiler.
|
|
29
29
|
|
|
30
|
-
|
|
31
|
-
|
|
30
|
+
Imported lazily: line_profiler is an optional dependency, needed only when
|
|
31
|
+
``use_line_profiler=True``.
|
|
32
32
|
"""
|
|
33
|
-
|
|
33
|
+
try:
|
|
34
|
+
from line_profiler import LineProfiler
|
|
35
|
+
except ImportError as exc:
|
|
36
|
+
raise ImportError(
|
|
37
|
+
"Line-by-line profiling requested but line_profiler is not "
|
|
38
|
+
"installed. Install scope-profiler[line-profiler], or "
|
|
39
|
+
"line_profiler directly."
|
|
40
|
+
) from exc
|
|
34
41
|
|
|
35
42
|
return LineProfiler
|
|
36
43
|
|
|
@@ -57,6 +64,7 @@ class BaseProfileRegion:
|
|
|
57
64
|
"start_times",
|
|
58
65
|
"end_times",
|
|
59
66
|
"num_calls",
|
|
67
|
+
"_num_calls_written",
|
|
60
68
|
"ptr",
|
|
61
69
|
"buffer_limit",
|
|
62
70
|
"capacity",
|
|
@@ -83,6 +91,8 @@ class BaseProfileRegion:
|
|
|
83
91
|
self.region_name = region_name
|
|
84
92
|
self.config = config
|
|
85
93
|
self.num_calls = 0
|
|
94
|
+
# Calls already persisted by an earlier finalize(); see mark_written.
|
|
95
|
+
self._num_calls_written = 0
|
|
86
96
|
|
|
87
97
|
# Preallocate buffers (skipped entirely when no timing is recorded).
|
|
88
98
|
# `buffer_limit` is the *initial* capacity: `_grow` doubles it as
|
|
@@ -163,6 +173,12 @@ class BaseProfileRegion:
|
|
|
163
173
|
|
|
164
174
|
with h5py.File(self.config._local_file_path, "a") as f:
|
|
165
175
|
grp = f.require_group(self.group_path)
|
|
176
|
+
# Never fail on a dataset that is already there: writing twice into
|
|
177
|
+
# the same per-rank file (a second write_to_disk() outside the
|
|
178
|
+
# finalize() path) should replace the data, not raise from h5py.
|
|
179
|
+
for name in ("start_times", "end_times"):
|
|
180
|
+
if name in grp:
|
|
181
|
+
del grp[name]
|
|
166
182
|
grp.create_dataset("start_times", data=self.start_times[: self.ptr])
|
|
167
183
|
grp.create_dataset("end_times", data=self.end_times[: self.ptr])
|
|
168
184
|
|
|
@@ -171,15 +187,38 @@ class BaseProfileRegion:
|
|
|
171
187
|
|
|
172
188
|
Used by regions that record no timestamps, so their call counts survive
|
|
173
189
|
into the merged output file instead of being lost with the process.
|
|
190
|
+
Only the calls made since the last finalize() are written, matching the
|
|
191
|
+
per-run slice of timestamps the timing regions write.
|
|
174
192
|
"""
|
|
175
193
|
with h5py.File(self.config._local_file_path, "a") as f:
|
|
176
194
|
grp = f.require_group(self.group_path)
|
|
177
|
-
grp.attrs["num_calls"] = self.num_calls
|
|
195
|
+
grp.attrs["num_calls"] = self.num_calls - self._num_calls_written
|
|
178
196
|
|
|
179
197
|
def get_durations_numpy(self) -> np.ndarray:
|
|
180
198
|
"""Return durations (end - start) for buffered entries as a NumPy array."""
|
|
181
199
|
return self.end_times[: self.ptr] - self.start_times[: self.ptr]
|
|
182
200
|
|
|
201
|
+
def mark_written(self) -> None:
|
|
202
|
+
"""Record that everything buffered so far has reached the disk.
|
|
203
|
+
|
|
204
|
+
Called by ``finalize()`` once the data is safely written, so that a
|
|
205
|
+
second run in the same process reports only its own events instead of
|
|
206
|
+
re-reporting the first run's. The timestamp buffer rewinds (the arrays
|
|
207
|
+
are reused; anything past ``ptr`` is unread scratch), while
|
|
208
|
+
``num_calls`` keeps counting for the lifetime of the process — it is
|
|
209
|
+
the in-memory view of the region, which callers inspect after
|
|
210
|
+
``finalize()`` — so the watermark below is what makes the *written*
|
|
211
|
+
count per-run.
|
|
212
|
+
|
|
213
|
+
A region that is currently open has a slot reserved in the buffer and
|
|
214
|
+
an index waiting to be popped on exit, so rewinding under it would let
|
|
215
|
+
the next call overwrite a live slot. Such a region is left untouched.
|
|
216
|
+
"""
|
|
217
|
+
if self._scope_ptr_stack:
|
|
218
|
+
return
|
|
219
|
+
self.ptr = 0
|
|
220
|
+
self._num_calls_written = self.num_calls
|
|
221
|
+
|
|
183
222
|
def get_end_times_numpy(self) -> np.ndarray:
|
|
184
223
|
"""Return end times offset by config creation time."""
|
|
185
224
|
return self.end_times[: self.ptr] - self.config.config_creation_time
|
|
@@ -249,7 +288,7 @@ class NCallsOnlyProfileRegion(BaseProfileRegion):
|
|
|
249
288
|
|
|
250
289
|
def write_to_disk(self):
|
|
251
290
|
"""Persist the call count — the only thing this region records."""
|
|
252
|
-
if self.num_calls:
|
|
291
|
+
if self.num_calls > self._num_calls_written:
|
|
253
292
|
self._write_num_calls()
|
|
254
293
|
|
|
255
294
|
def get_durations_numpy(self):
|
|
@@ -330,7 +369,7 @@ class LikwidOnlyProfileRegion(BaseProfileRegion):
|
|
|
330
369
|
|
|
331
370
|
def write_to_disk(self):
|
|
332
371
|
"""Persist the call count; LIKWID counters go to LIKWID's own output."""
|
|
333
|
-
if self.num_calls:
|
|
372
|
+
if self.num_calls > self._num_calls_written:
|
|
334
373
|
self._write_num_calls()
|
|
335
374
|
|
|
336
375
|
def wrap(self, func):
|
|
@@ -155,6 +155,7 @@ def test_all_region_types():
|
|
|
155
155
|
|
|
156
156
|
|
|
157
157
|
def test_line_profiler_decorator():
|
|
158
|
+
pytest.importorskip("line_profiler")
|
|
158
159
|
ProfileManager.setup(
|
|
159
160
|
use_line_profiler=True,
|
|
160
161
|
time_trace=True,
|
|
@@ -187,6 +188,7 @@ def test_line_profiler_decorator():
|
|
|
187
188
|
|
|
188
189
|
|
|
189
190
|
def test_line_profiler_context_manager():
|
|
191
|
+
pytest.importorskip("line_profiler")
|
|
190
192
|
ProfileManager.setup(
|
|
191
193
|
use_line_profiler=True,
|
|
192
194
|
time_trace=True,
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
import marshal
|
|
2
|
+
import pstats
|
|
3
|
+
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
6
|
+
from scope_profiler.h5reader import ProfilingH5Reader
|
|
7
|
+
from scope_profiler.post_processing import main
|
|
8
|
+
from scope_profiler.prof_export import build_pstats_dict, export_prof
|
|
9
|
+
from scope_profiler.tests.test_post_processing import _write_sample_h5
|
|
10
|
+
|
|
11
|
+
MS = 1_000_000 # nanoseconds per millisecond, the unit stored in the HDF5 files
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _nested_file_data():
|
|
15
|
+
"""One rank whose regions nest: main > (setup, solve > assemble)."""
|
|
16
|
+
return {
|
|
17
|
+
0: {
|
|
18
|
+
"main": ([0], [100 * MS]),
|
|
19
|
+
"setup": ([0], [20 * MS]),
|
|
20
|
+
"solve": ([20 * MS], [90 * MS]),
|
|
21
|
+
"assemble": ([30 * MS], [60 * MS]),
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _calls(*specs):
|
|
27
|
+
"""Build the call list ``build_pstats_dict`` expects, in seconds."""
|
|
28
|
+
return [
|
|
29
|
+
{"name": name, "start": start, "end": end, "parent": parent}
|
|
30
|
+
for name, start, end, parent in specs
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _stats_of(path):
|
|
35
|
+
with open(path, "rb") as f:
|
|
36
|
+
return marshal.load(f)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def test_build_pstats_dict_nesting_and_self_time():
|
|
40
|
+
calls = _calls(
|
|
41
|
+
("main", 0.0, 1.0, None),
|
|
42
|
+
("child", 0.1, 0.4, 0),
|
|
43
|
+
("child", 0.5, 0.7, 0),
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
stats = build_pstats_dict(calls)
|
|
47
|
+
|
|
48
|
+
main = stats[("~", 0, "main")]
|
|
49
|
+
child = stats[("~", 0, "child")]
|
|
50
|
+
|
|
51
|
+
assert main[:2] == (1, 1)
|
|
52
|
+
assert main[2] == pytest.approx(0.5) # 1.0 total minus 0.3 + 0.2 of children
|
|
53
|
+
assert main[3] == pytest.approx(1.0)
|
|
54
|
+
assert main[4] == {}
|
|
55
|
+
|
|
56
|
+
assert child[:2] == (2, 2)
|
|
57
|
+
assert child[2] == pytest.approx(0.5)
|
|
58
|
+
assert child[3] == pytest.approx(0.5)
|
|
59
|
+
assert child[4] == {
|
|
60
|
+
("~", 0, "main"): (2, 2, pytest.approx(0.5), pytest.approx(0.5))
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def test_build_pstats_dict_counts_recursion_like_cprofile():
|
|
65
|
+
calls = _calls(
|
|
66
|
+
("recurse", 0.0, 1.0, None),
|
|
67
|
+
("recurse", 0.2, 0.6, 0),
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
entry = build_pstats_dict(calls)[("~", 0, "recurse")]
|
|
71
|
+
|
|
72
|
+
# One primitive call, two total, and the shared seconds counted once.
|
|
73
|
+
assert entry[0] == 1
|
|
74
|
+
assert entry[1] == 2
|
|
75
|
+
assert entry[2] == pytest.approx(1.0)
|
|
76
|
+
assert entry[3] == pytest.approx(1.0)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def test_build_pstats_dict_clamps_partial_overlap():
|
|
80
|
+
# "long" is reconstructed as a child of "short" because it starts inside
|
|
81
|
+
# it, even though it runs past its end - self time must not go negative.
|
|
82
|
+
calls = _calls(
|
|
83
|
+
("short", 0.0, 0.5, None),
|
|
84
|
+
("long", 0.1, 2.0, 0),
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
stats = build_pstats_dict(calls)
|
|
88
|
+
|
|
89
|
+
assert stats[("~", 0, "short")][2] == 0.0
|
|
90
|
+
assert all(entry[2] >= 0.0 for entry in stats.values())
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def test_build_pstats_dict_synthetic_root():
|
|
94
|
+
calls = _calls(("a", 0.0, 1.0, None), ("b", 2.0, 2.5, None))
|
|
95
|
+
|
|
96
|
+
stats = build_pstats_dict(calls, root_name="<run>")
|
|
97
|
+
|
|
98
|
+
root_key = ("~", 0, "<run>")
|
|
99
|
+
assert stats[root_key][:4] == (1, 1, 0.0, pytest.approx(1.5))
|
|
100
|
+
for name in ("a", "b"):
|
|
101
|
+
assert root_key in stats[("~", 0, name)][4]
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def test_export_prof_readable_by_pstats(tmp_path):
|
|
105
|
+
h5_file = tmp_path / "profiling_data.h5"
|
|
106
|
+
_write_sample_h5(h5_file, _nested_file_data())
|
|
107
|
+
|
|
108
|
+
written = export_prof(
|
|
109
|
+
ProfilingH5Reader(h5_file), tmp_path / "profile.prof", verbose=False
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
assert written == [tmp_path / "profile_rank0.prof"]
|
|
113
|
+
|
|
114
|
+
stats = pstats.Stats(str(written[0]))
|
|
115
|
+
names = {key[2] for key in stats.stats}
|
|
116
|
+
assert {"main", "setup", "solve", "assemble"} <= names
|
|
117
|
+
|
|
118
|
+
# main: 0.1s wall, minus setup (0.02) and solve (0.07); solve minus assemble.
|
|
119
|
+
assert stats.stats[("~", 0, "main")][2] == pytest.approx(0.01)
|
|
120
|
+
assert stats.stats[("~", 0, "solve")][2] == pytest.approx(0.04)
|
|
121
|
+
assert stats.stats[("~", 0, "assemble")][3] == pytest.approx(0.03)
|
|
122
|
+
assert stats.stats[("~", 0, "assemble")][4].keys() == {("~", 0, "solve")}
|
|
123
|
+
assert stats.total_calls == 5 # four regions plus the synthetic root
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def test_export_prof_per_rank_and_per_file(tmp_path):
|
|
127
|
+
file_one = tmp_path / "run_one.h5"
|
|
128
|
+
file_two = tmp_path / "run_two.h5"
|
|
129
|
+
_write_sample_h5(file_one, {rank: _nested_file_data()[0] for rank in (0, 1)})
|
|
130
|
+
_write_sample_h5(file_two, {rank: _nested_file_data()[0] for rank in (0, 1)})
|
|
131
|
+
|
|
132
|
+
readers = [ProfilingH5Reader(file_one), ProfilingH5Reader(file_two)]
|
|
133
|
+
written = export_prof(
|
|
134
|
+
readers, tmp_path / "profile.prof", ranks=[0, 1], verbose=False
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
assert [path.name for path in written] == [
|
|
138
|
+
"profile_run_one_rank0.prof",
|
|
139
|
+
"profile_run_one_rank1.prof",
|
|
140
|
+
"profile_run_two_rank0.prof",
|
|
141
|
+
"profile_run_two_rank1.prof",
|
|
142
|
+
]
|
|
143
|
+
for path in written:
|
|
144
|
+
assert pstats.Stats(str(path)).total_calls == 5
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def test_export_prof_rejects_unknown_rank(tmp_path):
|
|
148
|
+
h5_file = tmp_path / "profiling_data.h5"
|
|
149
|
+
_write_sample_h5(h5_file, _nested_file_data())
|
|
150
|
+
|
|
151
|
+
with pytest.raises(ValueError, match="Invalid rank"):
|
|
152
|
+
export_prof(
|
|
153
|
+
ProfilingH5Reader(h5_file),
|
|
154
|
+
tmp_path / "profile.prof",
|
|
155
|
+
ranks=[3],
|
|
156
|
+
verbose=False,
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def test_cli_export_prof_without_plots(tmp_path, capsys):
|
|
161
|
+
h5_file = tmp_path / "profiling_data.h5"
|
|
162
|
+
_write_sample_h5(h5_file, _nested_file_data())
|
|
163
|
+
out_dir = tmp_path / "figures"
|
|
164
|
+
|
|
165
|
+
main(
|
|
166
|
+
[
|
|
167
|
+
str(h5_file),
|
|
168
|
+
"-o",
|
|
169
|
+
str(out_dir),
|
|
170
|
+
"--export-prof",
|
|
171
|
+
"--skip-plot-images",
|
|
172
|
+
]
|
|
173
|
+
)
|
|
174
|
+
|
|
175
|
+
prof_file = out_dir / "profile_rank0.prof"
|
|
176
|
+
assert prof_file.exists()
|
|
177
|
+
assert not list(out_dir.glob("*.png"))
|
|
178
|
+
assert "snakeviz" in capsys.readouterr().out
|
|
179
|
+
|
|
180
|
+
stats = _stats_of(prof_file)
|
|
181
|
+
assert {key[2] for key in stats} >= {"main", "setup", "solve", "assemble"}
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def test_cli_export_prof_alongside_plots(tmp_path):
|
|
185
|
+
h5_file = tmp_path / "profiling_data.h5"
|
|
186
|
+
_write_sample_h5(h5_file, _nested_file_data())
|
|
187
|
+
out_dir = tmp_path / "figures"
|
|
188
|
+
|
|
189
|
+
main([str(h5_file), "-o", str(out_dir), "--export-prof", "--ranks", "0"])
|
|
190
|
+
|
|
191
|
+
assert (out_dir / "profile_rank0.prof").exists()
|
|
192
|
+
assert (out_dir / "flame_plot.png").exists()
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def test_cli_export_prof_requires_output(tmp_path):
|
|
196
|
+
h5_file = tmp_path / "profiling_data.h5"
|
|
197
|
+
_write_sample_h5(h5_file, _nested_file_data())
|
|
198
|
+
|
|
199
|
+
with pytest.raises(SystemExit):
|
|
200
|
+
main([str(h5_file), "--export-prof"])
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def test_exported_prof_loads_in_snakeviz(tmp_path):
|
|
204
|
+
snakeviz_stats = pytest.importorskip("snakeviz.stats")
|
|
205
|
+
|
|
206
|
+
h5_file = tmp_path / "profiling_data.h5"
|
|
207
|
+
_write_sample_h5(h5_file, _nested_file_data())
|
|
208
|
+
written = export_prof(
|
|
209
|
+
ProfilingH5Reader(h5_file), tmp_path / "profile.prof", verbose=False
|
|
210
|
+
)
|
|
211
|
+
|
|
212
|
+
stats = pstats.Stats(str(written[0]))
|
|
213
|
+
assert len(snakeviz_stats.table_rows(stats)) == 5
|
|
214
|
+
|
|
215
|
+
# snakeviz drops entries that neither call nor are called; every region has
|
|
216
|
+
# to survive that, otherwise it cannot be drawn.
|
|
217
|
+
tree = snakeviz_stats.json_stats(stats)
|
|
218
|
+
assert set(tree) == {
|
|
219
|
+
"~:0(<profiling_data rank 0>)",
|
|
220
|
+
"~:0(main)",
|
|
221
|
+
"~:0(setup)",
|
|
222
|
+
"~:0(solve)",
|
|
223
|
+
"~:0(assemble)",
|
|
224
|
+
}
|
|
225
|
+
assert set(tree["~:0(solve)"]["children"]) == {"~:0(assemble)"}
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""Calling finalize() more than once in a process, as a restarted run does."""
|
|
2
|
+
|
|
3
|
+
from time import sleep
|
|
4
|
+
|
|
5
|
+
from scope_profiler import ProfileManager
|
|
6
|
+
from scope_profiler.h5reader import ProfilingH5Reader
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def _run(label: str, num_calls: int) -> None:
|
|
10
|
+
for _ in range(num_calls):
|
|
11
|
+
with ProfileManager.profile_region(label):
|
|
12
|
+
sleep(0.0001)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_second_finalize_without_setup(tmp_path):
|
|
16
|
+
"""A second run must finalize cleanly and report only its own events."""
|
|
17
|
+
out = tmp_path / "profiling_data.h5"
|
|
18
|
+
ProfileManager.setup(file_path=str(out), flush_to_disk=True)
|
|
19
|
+
|
|
20
|
+
_run("step", 3)
|
|
21
|
+
ProfileManager.finalize(verbose=False)
|
|
22
|
+
|
|
23
|
+
region = ProfilingH5Reader(str(out)).get_region("step")
|
|
24
|
+
assert region.num_calls == 3
|
|
25
|
+
|
|
26
|
+
# Second run, reusing the config and the region objects of the first.
|
|
27
|
+
_run("step", 5)
|
|
28
|
+
ProfileManager.finalize(verbose=False)
|
|
29
|
+
|
|
30
|
+
region = ProfilingH5Reader(str(out)).get_region("step")
|
|
31
|
+
assert region.num_calls == 5, "second run inherited the first run's events"
|
|
32
|
+
assert len(region.durations) == 5
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def test_stale_regions_do_not_leak_into_the_second_run(tmp_path):
|
|
36
|
+
"""A region used only in the first run must not reappear in the second."""
|
|
37
|
+
out = tmp_path / "profiling_data.h5"
|
|
38
|
+
ProfileManager.setup(file_path=str(out), flush_to_disk=True)
|
|
39
|
+
|
|
40
|
+
_run("first_only", 2)
|
|
41
|
+
ProfileManager.finalize(verbose=False)
|
|
42
|
+
assert "first_only" in ProfilingH5Reader(str(out)).region_names
|
|
43
|
+
|
|
44
|
+
_run("second_only", 2)
|
|
45
|
+
ProfileManager.finalize(verbose=False)
|
|
46
|
+
|
|
47
|
+
names = ProfilingH5Reader(str(out)).region_names
|
|
48
|
+
assert "second_only" in names
|
|
49
|
+
assert "first_only" not in names
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def test_call_counts_are_per_run_without_timing(tmp_path):
|
|
53
|
+
"""Call-count-only regions also report per-run numbers, not running totals."""
|
|
54
|
+
out = tmp_path / "profiling_data.h5"
|
|
55
|
+
ProfileManager.setup(file_path=str(out), time_trace=False)
|
|
56
|
+
|
|
57
|
+
_run("step", 3)
|
|
58
|
+
ProfileManager.finalize(verbose=False)
|
|
59
|
+
assert ProfilingH5Reader(str(out)).get_region("step").num_calls == 3
|
|
60
|
+
|
|
61
|
+
_run("step", 5)
|
|
62
|
+
ProfileManager.finalize(verbose=False)
|
|
63
|
+
assert ProfilingH5Reader(str(out)).get_region("step").num_calls == 5
|
|
64
|
+
|
|
65
|
+
# In memory the counter keeps running for the lifetime of the process.
|
|
66
|
+
assert ProfileManager.get_region("step").num_calls == 8
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_finalize_keeps_buffers_when_not_flushing(tmp_path):
|
|
70
|
+
"""With flush_to_disk off the buffers are the only copy, so they survive."""
|
|
71
|
+
ProfileManager.setup(
|
|
72
|
+
file_path=str(tmp_path / "profiling_data.h5"),
|
|
73
|
+
flush_to_disk=False,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
_run("step", 4)
|
|
77
|
+
ProfileManager.finalize(verbose=False)
|
|
78
|
+
|
|
79
|
+
region = ProfileManager.get_region("step")
|
|
80
|
+
assert region.num_calls == 4
|
|
81
|
+
assert region.ptr == 4
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def test_finalize_inside_an_open_region(tmp_path):
|
|
85
|
+
"""A region still open at finalize() keeps its reserved slot."""
|
|
86
|
+
ProfileManager.setup(
|
|
87
|
+
file_path=str(tmp_path / "profiling_data.h5"),
|
|
88
|
+
flush_to_disk=True,
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
with ProfileManager.profile_region("outer"):
|
|
92
|
+
ProfileManager.finalize(verbose=False)
|
|
93
|
+
|
|
94
|
+
region = ProfileManager.get_region("outer")
|
|
95
|
+
# Rewinding under the open scope would have let the exit write into a slot
|
|
96
|
+
# the next call could reuse; the region is left alone instead.
|
|
97
|
+
assert region.num_calls == 1
|
|
98
|
+
assert region.ptr == 1
|
|
99
|
+
assert region.get_durations_numpy()[0] > 0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scope-profiler
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: Profile code regions in python, optionally with LIKWID markers.
|
|
5
5
|
Author: Max
|
|
6
6
|
Project-URL: Source, https://github.com/max-models/scope-profiler
|
|
@@ -16,7 +16,8 @@ Requires-Python: >=3.10
|
|
|
16
16
|
Description-Content-Type: text/markdown
|
|
17
17
|
Requires-Dist: h5py
|
|
18
18
|
Requires-Dist: numpy
|
|
19
|
-
|
|
19
|
+
Provides-Extra: line-profiler
|
|
20
|
+
Requires-Dist: line-profiler; extra == "line-profiler"
|
|
20
21
|
Provides-Extra: mpi
|
|
21
22
|
Requires-Dist: mpi4py; extra == "mpi"
|
|
22
23
|
Provides-Extra: pproc
|
|
@@ -25,11 +26,12 @@ Requires-Dist: jupyterlab; extra == "pproc"
|
|
|
25
26
|
Requires-Dist: matplotlib; extra == "pproc"
|
|
26
27
|
Requires-Dist: maxplotlibx>=0.1.5; extra == "pproc"
|
|
27
28
|
Requires-Dist: pandas; extra == "pproc"
|
|
29
|
+
Requires-Dist: snakeviz; extra == "pproc"
|
|
28
30
|
Provides-Extra: dev
|
|
29
31
|
Requires-Dist: black[jupyter]; extra == "dev"
|
|
30
32
|
Requires-Dist: isort; extra == "dev"
|
|
31
33
|
Requires-Dist: ruff; extra == "dev"
|
|
32
|
-
Requires-Dist: scope-profiler[docs,mpi,pproc,test]; extra == "dev"
|
|
34
|
+
Requires-Dist: scope-profiler[docs,line-profiler,mpi,pproc,test]; extra == "dev"
|
|
33
35
|
Provides-Extra: docs
|
|
34
36
|
Requires-Dist: myst-parser; extra == "docs"
|
|
35
37
|
Requires-Dist: nbconvert; extra == "docs"
|
|
@@ -38,7 +40,7 @@ Requires-Dist: pre-commit; extra == "docs"
|
|
|
38
40
|
Requires-Dist: pyproject-fmt; extra == "docs"
|
|
39
41
|
Requires-Dist: sphinx; extra == "docs"
|
|
40
42
|
Requires-Dist: sphinx-book-theme; extra == "docs"
|
|
41
|
-
Requires-Dist: scope-profiler[pproc]; extra == "docs"
|
|
43
|
+
Requires-Dist: scope-profiler[line-profiler,pproc]; extra == "docs"
|
|
42
44
|
Provides-Extra: test
|
|
43
45
|
Requires-Dist: coverage; extra == "test"
|
|
44
46
|
Requires-Dist: pytest; extra == "test"
|
|
@@ -427,3 +429,19 @@ straight from the JSON:
|
|
|
427
429
|
scope-profiler pproc profiling_data.h5 -o figures \
|
|
428
430
|
--export-data --export-data-format json --skip-plot-images
|
|
429
431
|
```
|
|
432
|
+
|
|
433
|
+
### Viewing a run in snakeviz
|
|
434
|
+
|
|
435
|
+
`--export-prof` writes the profile in the `.prof` format of the standard
|
|
436
|
+
library's `cProfile`, so a run can be explored with
|
|
437
|
+
[snakeviz](https://jiffyclub.github.io/snakeviz/) or `python -m pstats`:
|
|
438
|
+
|
|
439
|
+
```bash
|
|
440
|
+
scope-profiler pproc profiling_data.h5 -o figures --export-prof --skip-plot-images
|
|
441
|
+
snakeviz figures/profile_rank0.prof
|
|
442
|
+
```
|
|
443
|
+
|
|
444
|
+
Regions become "functions": `cumtime` is a region's total wall time and
|
|
445
|
+
`tottime` is that minus the time spent in its nested regions. One file is
|
|
446
|
+
written per exported rank, since `.prof` has no notion of ranks — see
|
|
447
|
+
[the CLI docs](docs/source/cli.md) for the caveats of the reconstruction.
|
|
@@ -8,6 +8,7 @@ src/scope_profiler/metadata.py
|
|
|
8
8
|
src/scope_profiler/mpi_region.py
|
|
9
9
|
src/scope_profiler/plotting_scripts.py
|
|
10
10
|
src/scope_profiler/post_processing.py
|
|
11
|
+
src/scope_profiler/prof_export.py
|
|
11
12
|
src/scope_profiler/profile_config.py
|
|
12
13
|
src/scope_profiler/profile_manager.py
|
|
13
14
|
src/scope_profiler/region.py
|
|
@@ -30,5 +31,7 @@ src/scope_profiler/tests/test_metadata.py
|
|
|
30
31
|
src/scope_profiler/tests/test_mpi.py
|
|
31
32
|
src/scope_profiler/tests/test_overhead.py
|
|
32
33
|
src/scope_profiler/tests/test_post_processing.py
|
|
34
|
+
src/scope_profiler/tests/test_prof_export.py
|
|
33
35
|
src/scope_profiler/tests/test_reader_api.py
|
|
34
|
-
src/scope_profiler/tests/test_readme.py
|
|
36
|
+
src/scope_profiler/tests/test_readme.py
|
|
37
|
+
src/scope_profiler/tests/test_repeated_finalize.py
|
|
@@ -1,12 +1,11 @@
|
|
|
1
1
|
h5py
|
|
2
2
|
numpy
|
|
3
|
-
line-profiler
|
|
4
3
|
|
|
5
4
|
[dev]
|
|
6
5
|
black[jupyter]
|
|
7
6
|
isort
|
|
8
7
|
ruff
|
|
9
|
-
scope-profiler[docs,mpi,pproc,test]
|
|
8
|
+
scope-profiler[docs,line-profiler,mpi,pproc,test]
|
|
10
9
|
|
|
11
10
|
[docs]
|
|
12
11
|
myst-parser
|
|
@@ -16,7 +15,10 @@ pre-commit
|
|
|
16
15
|
pyproject-fmt
|
|
17
16
|
sphinx
|
|
18
17
|
sphinx-book-theme
|
|
19
|
-
scope-profiler[pproc]
|
|
18
|
+
scope-profiler[line-profiler,pproc]
|
|
19
|
+
|
|
20
|
+
[line-profiler]
|
|
21
|
+
line-profiler
|
|
20
22
|
|
|
21
23
|
[mpi]
|
|
22
24
|
mpi4py
|
|
@@ -27,6 +29,7 @@ jupyterlab
|
|
|
27
29
|
matplotlib
|
|
28
30
|
maxplotlibx>=0.1.5
|
|
29
31
|
pandas
|
|
32
|
+
snakeviz
|
|
30
33
|
|
|
31
34
|
[test]
|
|
32
35
|
coverage
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/test_buffer_growth.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler/tests/test_post_processing.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{scope_profiler-0.2.2 → scope_profiler-0.2.3}/src/scope_profiler.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|