geoswe 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- geoswe-1.0.0/CITATION.cff +29 -0
- geoswe-1.0.0/LICENSE +29 -0
- geoswe-1.0.0/MANIFEST.in +13 -0
- geoswe-1.0.0/PKG-INFO +217 -0
- geoswe-1.0.0/README.md +164 -0
- geoswe-1.0.0/docs/amd_gpus.md +165 -0
- geoswe-1.0.0/docs/api/compressed.rst +13 -0
- geoswe-1.0.0/docs/api/forcing.rst +21 -0
- geoswe-1.0.0/docs/api/index.md +18 -0
- geoswe-1.0.0/docs/api/mesh.rst +22 -0
- geoswe-1.0.0/docs/api/runlib.rst +37 -0
- geoswe-1.0.0/docs/api/solver.rst +16 -0
- geoswe-1.0.0/docs/benchmarks.md +81 -0
- geoswe-1.0.0/docs/citing.md +23 -0
- geoswe-1.0.0/docs/compressed_mesh.md +96 -0
- geoswe-1.0.0/docs/conf.py +95 -0
- geoswe-1.0.0/docs/configuration.md +194 -0
- geoswe-1.0.0/docs/examples.md +96 -0
- geoswe-1.0.0/docs/flood_tutorial.md +162 -0
- geoswe-1.0.0/docs/images/helene_cascade.gif +0 -0
- geoswe-1.0.0/docs/images/helene_cascade.mp4 +0 -0
- geoswe-1.0.0/docs/images/intro_bench.png +0 -0
- geoswe-1.0.0/docs/images/scaling.png +0 -0
- geoswe-1.0.0/docs/images/scaling_frontier.png +0 -0
- geoswe-1.0.0/docs/index.md +84 -0
- geoswe-1.0.0/docs/installation.md +127 -0
- geoswe-1.0.0/docs/multigpu_mpi.md +72 -0
- geoswe-1.0.0/docs/quickstart.md +98 -0
- geoswe-1.0.0/docs/requirements.txt +9 -0
- geoswe-1.0.0/docs/userguide/boundary_conditions.md +52 -0
- geoswe-1.0.0/docs/userguide/forcings.md +102 -0
- geoswe-1.0.0/docs/userguide/friction.md +69 -0
- geoswe-1.0.0/docs/userguide/governing_equations.md +65 -0
- geoswe-1.0.0/docs/userguide/numerical_methods.md +78 -0
- geoswe-1.0.0/docs/userguide/well_balanced.md +44 -0
- geoswe-1.0.0/environment.yml +25 -0
- geoswe-1.0.0/examples/README.md +41 -0
- geoswe-1.0.0/examples/data/cookcounty_mini.npz +0 -0
- geoswe-1.0.0/examples/ex01_dam_break_1d.py +72 -0
- geoswe-1.0.0/examples/ex02_circular_dam_break_2d.py +65 -0
- geoswe-1.0.0/examples/ex03_lake_at_rest_2d.py +60 -0
- geoswe-1.0.0/examples/ex04_rain_on_slope_2d.py +77 -0
- geoswe-1.0.0/examples/ex05_scaling_bench.py +97 -0
- geoswe-1.0.0/examples/ex06_pluvial_flood_realcase.py +114 -0
- geoswe-1.0.0/examples/ex07_convergence_order.py +125 -0
- geoswe-1.0.0/examples/ex08_compressed_mesh.py +102 -0
- geoswe-1.0.0/pyproject.toml +88 -0
- geoswe-1.0.0/setup.cfg +4 -0
- geoswe-1.0.0/src/geoswe/__init__.py +102 -0
- geoswe-1.0.0/src/geoswe/backend.py +335 -0
- geoswe-1.0.0/src/geoswe/bc.py +235 -0
- geoswe-1.0.0/src/geoswe/compressed_mesh.py +191 -0
- geoswe-1.0.0/src/geoswe/compressed_rhs.py +1082 -0
- geoswe-1.0.0/src/geoswe/compressed_solver.py +3085 -0
- geoswe-1.0.0/src/geoswe/data_prep.py +399 -0
- geoswe-1.0.0/src/geoswe/elliptic.py +154 -0
- geoswe-1.0.0/src/geoswe/elliptic_cuda.py +258 -0
- geoswe-1.0.0/src/geoswe/flux.py +151 -0
- geoswe-1.0.0/src/geoswe/forcing.py +329 -0
- geoswe-1.0.0/src/geoswe/friction.py +108 -0
- geoswe-1.0.0/src/geoswe/gauges.py +109 -0
- geoswe-1.0.0/src/geoswe/io_geotiff.py +220 -0
- geoswe-1.0.0/src/geoswe/mesh.py +87 -0
- geoswe-1.0.0/src/geoswe/mpi_halo.py +495 -0
- geoswe-1.0.0/src/geoswe/reconstruction.py +195 -0
- geoswe-1.0.0/src/geoswe/rhs_cuda.py +2487 -0
- geoswe-1.0.0/src/geoswe/runlib/__init__.py +27 -0
- geoswe-1.0.0/src/geoswe/runlib/case.py +117 -0
- geoswe-1.0.0/src/geoswe/runlib/cli.py +126 -0
- geoswe-1.0.0/src/geoswe/runlib/driver.py +1546 -0
- geoswe-1.0.0/src/geoswe/runlib/replay.py +162 -0
- geoswe-1.0.0/src/geoswe/solver.py +2721 -0
- geoswe-1.0.0/src/geoswe/swe.py +74 -0
- geoswe-1.0.0/src/geoswe/well_balanced.py +415 -0
- geoswe-1.0.0/src/geoswe.egg-info/PKG-INFO +217 -0
- geoswe-1.0.0/src/geoswe.egg-info/SOURCES.txt +106 -0
- geoswe-1.0.0/src/geoswe.egg-info/dependency_links.txt +1 -0
- geoswe-1.0.0/src/geoswe.egg-info/requires.txt +37 -0
- geoswe-1.0.0/src/geoswe.egg-info/top_level.txt +1 -0
- geoswe-1.0.0/tests/conftest.py +111 -0
- geoswe-1.0.0/tests/mpi_bitcheck.py +102 -0
- geoswe-1.0.0/tests/parity_gate.py +72 -0
- geoswe-1.0.0/tests/test_api_smoke.py +60 -0
- geoswe-1.0.0/tests/test_backend_cupy_builds.py +37 -0
- geoswe-1.0.0/tests/test_bc.py +93 -0
- geoswe-1.0.0/tests/test_cfl_guards.py +48 -0
- geoswe-1.0.0/tests/test_config_validation.py +60 -0
- geoswe-1.0.0/tests/test_dam_break_1d.py +28 -0
- geoswe-1.0.0/tests/test_friction_dry.py +68 -0
- geoswe-1.0.0/tests/test_gpu_checkpoint.py +201 -0
- geoswe-1.0.0/tests/test_gpu_compressed_equiv.py +134 -0
- geoswe-1.0.0/tests/test_gpu_compressed_from_dense.py +144 -0
- geoswe-1.0.0/tests/test_gpu_dense_fused_forcings.py +97 -0
- geoswe-1.0.0/tests/test_gpu_platform.py +210 -0
- geoswe-1.0.0/tests/test_gpu_portability.py +203 -0
- geoswe-1.0.0/tests/test_gpu_smoke.py +57 -0
- geoswe-1.0.0/tests/test_gpu_storage_fused.py +138 -0
- geoswe-1.0.0/tests/test_hllc_dry_front.py +55 -0
- geoswe-1.0.0/tests/test_inflow_bc.py +95 -0
- geoswe-1.0.0/tests/test_inflow_compressed.py +71 -0
- geoswe-1.0.0/tests/test_kernel_sources_ascii.py +35 -0
- geoswe-1.0.0/tests/test_rain_row_window.py +93 -0
- geoswe-1.0.0/tests/test_reconstruction_order.py +63 -0
- geoswe-1.0.0/tests/test_sigma_storage_guard.py +47 -0
- geoswe-1.0.0/tests/test_solver_convergence_order.py +63 -0
- geoswe-1.0.0/tests/test_srm_cpu_matches_gpu.py +59 -0
- geoswe-1.0.0/tests/test_user_conveniences.py +162 -0
- geoswe-1.0.0/tests/test_well_balanced.py +29 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
cff-version: 1.2.0
|
|
2
|
+
message: "If you use GeoSWE in your research, please cite it as below."
|
|
3
|
+
title: "GeoSWE: Geophysical Shallow-Water Engine"
|
|
4
|
+
abstract: >-
|
|
5
|
+
GeoSWE is a GPU-accelerated finite-volume solver for the 2D nonlinear
|
|
6
|
+
shallow-water equations, designed for flood modeling from coastal county to
|
|
7
|
+
continental scale. It combines HLLC/Lax-Friedrichs fluxes, well-balanced
|
|
8
|
+
surface reconstruction, implicit Manning friction, rainfall and coastal-stage
|
|
9
|
+
forcings, and a compressed active-cell mesh that reaches continental scale on
|
|
10
|
+
a single GPU node and scales across nodes with MPI.
|
|
11
|
+
type: software
|
|
12
|
+
authors:
|
|
13
|
+
- family-names: Chen
|
|
14
|
+
given-names: Peng
|
|
15
|
+
email: pchen402@gatech.edu
|
|
16
|
+
affiliation: "Georgia Institute of Technology"
|
|
17
|
+
version: "1.0.0"
|
|
18
|
+
date-released: "2026-10-06"
|
|
19
|
+
license: BSD-3-Clause
|
|
20
|
+
repository-code: "https://github.com/GeoSWE/geoswe"
|
|
21
|
+
url: "https://geoswe.github.io/geoswe"
|
|
22
|
+
keywords:
|
|
23
|
+
- shallow-water equations
|
|
24
|
+
- flood modeling
|
|
25
|
+
- GPU computing
|
|
26
|
+
- finite-volume method
|
|
27
|
+
- CuPy
|
|
28
|
+
# Add a `preferred-citation:` block for the article once it is accepted
|
|
29
|
+
# (title, authors, journal, year, doi), so citation tools prefer the paper.
|
geoswe-1.0.0/LICENSE
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
BSD 3-Clause License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026, Peng Chen and the GeoSWE contributors.
|
|
4
|
+
All rights reserved.
|
|
5
|
+
|
|
6
|
+
Redistribution and use in source and binary forms, with or without
|
|
7
|
+
modification, are permitted provided that the following conditions are met:
|
|
8
|
+
|
|
9
|
+
1. Redistributions of source code must retain the above copyright notice, this
|
|
10
|
+
list of conditions and the following disclaimer.
|
|
11
|
+
|
|
12
|
+
2. Redistributions in binary form must reproduce the above copyright notice,
|
|
13
|
+
this list of conditions and the following disclaimer in the documentation
|
|
14
|
+
and/or other materials provided with the distribution.
|
|
15
|
+
|
|
16
|
+
3. Neither the name of the copyright holder nor the names of its
|
|
17
|
+
contributors may be used to endorse or promote products derived from
|
|
18
|
+
this software without specific prior written permission.
|
|
19
|
+
|
|
20
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
21
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
22
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
23
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
24
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
25
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
26
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
27
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
28
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
29
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
geoswe-1.0.0/MANIFEST.in
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
include LICENSE
|
|
2
|
+
include README.md
|
|
3
|
+
include CITATION.cff
|
|
4
|
+
include environment.yml
|
|
5
|
+
recursive-include src/geoswe *.py
|
|
6
|
+
graft examples
|
|
7
|
+
graft tests
|
|
8
|
+
graft docs
|
|
9
|
+
prune docs/_build
|
|
10
|
+
# internal review notes and release planning stay in the repo, not the sdist
|
|
11
|
+
prune docs/dev
|
|
12
|
+
global-exclude __pycache__/*
|
|
13
|
+
global-exclude *.py[cod]
|
geoswe-1.0.0/PKG-INFO
ADDED
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: geoswe
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: GeoSWE (Geophysical Shallow-Water Engine): a GPU flood solver with a NumPy CPU fallback.
|
|
5
|
+
Author-email: Peng Chen <pchen402@gatech.edu>
|
|
6
|
+
License-Expression: BSD-3-Clause
|
|
7
|
+
Project-URL: Homepage, https://github.com/GeoSWE/geoswe
|
|
8
|
+
Project-URL: Documentation, https://geoswe.github.io/geoswe
|
|
9
|
+
Project-URL: Repository, https://github.com/GeoSWE/geoswe
|
|
10
|
+
Project-URL: Issues, https://github.com/GeoSWE/geoswe/issues
|
|
11
|
+
Keywords: shallow-water-equations,flood-modeling,computational-fluid-dynamics,GPU,CuPy,finite-volume,HLLC,well-balanced
|
|
12
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Hydrology
|
|
20
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
21
|
+
Requires-Python: >=3.10
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Requires-Dist: numpy>=1.24
|
|
25
|
+
Provides-Extra: gpu
|
|
26
|
+
Requires-Dist: cupy-cuda12x[ctk]>=14.0; extra == "gpu"
|
|
27
|
+
Requires-Dist: scipy>=1.10; extra == "gpu"
|
|
28
|
+
Provides-Extra: gpu-cuda13
|
|
29
|
+
Requires-Dist: cupy-cuda13x[ctk]>=14.0; extra == "gpu-cuda13"
|
|
30
|
+
Requires-Dist: scipy>=1.10; extra == "gpu-cuda13"
|
|
31
|
+
Provides-Extra: gpu-rocm
|
|
32
|
+
Requires-Dist: cupy-rocm-7-0>=14.0; extra == "gpu-rocm"
|
|
33
|
+
Requires-Dist: scipy>=1.10; extra == "gpu-rocm"
|
|
34
|
+
Provides-Extra: mpi
|
|
35
|
+
Requires-Dist: mpi4py>=3.1; extra == "mpi"
|
|
36
|
+
Provides-Extra: io
|
|
37
|
+
Requires-Dist: rasterio>=1.3; extra == "io"
|
|
38
|
+
Requires-Dist: pyproj>=3.5; extra == "io"
|
|
39
|
+
Provides-Extra: forcings
|
|
40
|
+
Requires-Dist: pandas>=2.0; extra == "forcings"
|
|
41
|
+
Requires-Dist: scipy>=1.10; extra == "forcings"
|
|
42
|
+
Provides-Extra: docs
|
|
43
|
+
Requires-Dist: sphinx>=7; extra == "docs"
|
|
44
|
+
Requires-Dist: myst-parser>=2; extra == "docs"
|
|
45
|
+
Requires-Dist: furo; extra == "docs"
|
|
46
|
+
Requires-Dist: sphinx-copybutton; extra == "docs"
|
|
47
|
+
Requires-Dist: sphinx-design; extra == "docs"
|
|
48
|
+
Provides-Extra: test
|
|
49
|
+
Requires-Dist: pytest>=7; extra == "test"
|
|
50
|
+
Provides-Extra: all
|
|
51
|
+
Requires-Dist: geoswe[forcings,gpu,io,mpi]; extra == "all"
|
|
52
|
+
Dynamic: license-file
|
|
53
|
+
|
|
54
|
+
<p align="center">
|
|
55
|
+
<img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/helene_cascade.gif" alt="Animation of Hurricane Helene rainfall and flood depth on three nested GeoSWE domains: CONUS at 30 m, Florida at 10 m, Pinellas County at 3 m" width="720">
|
|
56
|
+
</p>
|
|
57
|
+
|
|
58
|
+
<h1 align="center">GeoSWE</h1>
|
|
59
|
+
<p align="center"><b>Geophysical Shallow-Water Engine</b><br>
|
|
60
|
+
A GPU finite-volume solver for the full 2D shallow-water equations,<br>
|
|
61
|
+
from a 3 m coastal county to the conterminous United States on one multi-GPU node.</p>
|
|
62
|
+
|
|
63
|
+
<p align="center">
|
|
64
|
+
<a href="https://github.com/GeoSWE/geoswe/actions/workflows/test.yml"><img src="https://github.com/GeoSWE/geoswe/actions/workflows/test.yml/badge.svg" alt="tests"></a>
|
|
65
|
+
<a href="https://github.com/GeoSWE/geoswe/actions/workflows/lint.yml"><img src="https://github.com/GeoSWE/geoswe/actions/workflows/lint.yml/badge.svg" alt="lint"></a>
|
|
66
|
+
<a href="https://github.com/GeoSWE/geoswe/actions/workflows/docs.yml"><img src="https://github.com/GeoSWE/geoswe/actions/workflows/docs.yml/badge.svg" alt="docs"></a>
|
|
67
|
+
<a href="https://github.com/GeoSWE/geoswe/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-BSD--3--Clause-blue.svg" alt="BSD 3-Clause"></a>
|
|
68
|
+
<img src="https://img.shields.io/badge/python-3.10%2B-blue.svg" alt="Python 3.10+">
|
|
69
|
+
</p>
|
|
70
|
+
|
|
71
|
+
*Above: 72 hours of Hurricane Helene (September 2024) simulated by the same code on three nested domains: MRMS rainfall in purple, flood depth in color, and the NOAA CO-OPS tide gauges that drive the coastal boundary as stars. [Full-resolution video (MP4)](https://github.com/GeoSWE/geoswe/blob/main/docs/images/helene_cascade.mp4).*
|
|
72
|
+
|
|
73
|
+
---
|
|
74
|
+
|
|
75
|
+
GeoSWE solves the nonlinear shallow-water equations with a well-balanced HLLC
|
|
76
|
+
finite-volume scheme, Manning friction, wetting and drying, rainfall, and an
|
|
77
|
+
observation-driven coastal stage boundary. Its distinguishing feature is a
|
|
78
|
+
**compressed active-cell mesh**: the cells that matter (land plus a nearshore
|
|
79
|
+
band) are packed into flat arrays before the run, so memory and work scale
|
|
80
|
+
with the flooded landscape rather than its bounding rectangle. The same
|
|
81
|
+
finite-volume kernel runs on the dense grid and on the compressed mesh, bit for
|
|
82
|
+
bit. It runs on NVIDIA and AMD GPUs through [CuPy](https://cupy.dev), scales across
|
|
83
|
+
GPUs with `mpi4py`, and falls back to NumPy on the CPU for prototyping and CI.
|
|
84
|
+
|
|
85
|
+
## Highlights
|
|
86
|
+
|
|
87
|
+
| | |
|
|
88
|
+
|---|---|
|
|
89
|
+
| **Continental domains on one node** | A 72-hour Hurricane Helene scenario over CONUS at 30 m (8.88 billion active cells, 207-gauge coastal boundary) runs on eight H100 GPUs in 10.8 h; Florida at 10 m (1.78 billion active cells) on four GPUs in 9.5 h. |
|
|
90
|
+
| **Fastest and leanest in a four-code benchmark** | On a 214-million-cell county case against TRITON, SynxFlow, and SERGHEI, GeoSWE's active-cell configuration has the lowest wall time and GPU memory at every GPU count: 3.5 to 3.9 times faster and 1.5 to 1.9 times leaner than the nearest peer, with pairwise CSI near 0.99. |
|
|
91
|
+
| **Compression without a numerical penalty** | Dense and compressed runs take identical step counts and agree to 0.05 cm RMSE (CSI 1.000) on the same cells. |
|
|
92
|
+
| **Near-ideal scaling** | 99.5 % weak-scaling efficiency at 16 H100 GPUs (10.24 billion cells) and 98.8 % at 32 Blackwell MIG slices across two nodes (20.48 billion cells). |
|
|
93
|
+
| **Thin-film accuracy** | On a steady rained slope with a high-accuracy reference solution, GeoSWE stays within 1 % of the reference film depth where other codes depart by tens of percent. |
|
|
94
|
+
|
|
95
|
+
The application runs are computational demonstrations under stated, uncalibrated settings, not validated flood hindcasts.
|
|
96
|
+
|
|
97
|
+
## How it compares
|
|
98
|
+
|
|
99
|
+
<p align="center"><img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/intro_bench.png" alt="Benchmark: per-step wall time versus peak GPU memory per code, and thin-film depth error on a rained slope" width="900"></p>
|
|
100
|
+
|
|
101
|
+
*(a) Per-step wall time against peak GPU memory per rank for each code's fastest configuration on the 3 m Pinellas County benchmark (1, 2, and 4 GPUs; large marker = 4 GPUs). (b) Depth error against the steady shallow-water reference on a uniformly rained slope.*
|
|
102
|
+
|
|
103
|
+
Comparison codes as benchmarked: TRITON (commit `ec35bc4`), SERGHEI (commit `39a10f2`), and SynxFlow 1.0.2; all four in fp32 at CFL 0.5, first-order well-balanced schemes, same H100 node, identical inputs. Builds, decks, and patches are in the paper's reproducibility appendix.
|
|
104
|
+
|
|
105
|
+
## How it scales
|
|
106
|
+
|
|
107
|
+
<p align="center"><img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/scaling.png" alt="Strong scaling, weak scaling, and memory per rank on 1 to 16 H100 GPUs" width="900"></p>
|
|
108
|
+
|
|
109
|
+
*Dense and flat storage layouts on an everywhere-wet synthetic domain, 640 million cells per GPU, 1 to 16 H100 GPUs across two nodes. Strong scaling reaches 15.5x on 16 GPUs for the flat layout; weak scaling stays above 99 %.*
|
|
110
|
+
|
|
111
|
+
<p align="center"><img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/scaling_frontier.png" alt="Weak scaling on OLCF Frontier at one billion cells per GPU: 1 to 32 GPUs, and 1 to 128 nodes with 1.024 trillion cells" width="900"></p>
|
|
112
|
+
|
|
113
|
+
*The flat layout on AMD GPUs at OLCF Frontier, one billion cells per GPU (one GCD, half of an MI250X card; eight per node): time per step divided by that of the smallest size, so ideal weak scaling is the dashed line. On 128 nodes, 1.024 trillion cells advance at 162 ms per step: 98.0 % weak-scaling efficiency against one node and 96.7 % against one GPU. A marker is the mean of up to three launches, each timed over one 40-second window; see [`benchmark/frontier_amd`](https://github.com/GeoSWE/geoswe/blob/main/benchmark/frontier_amd/README.md).*
|
|
114
|
+
|
|
115
|
+
## What is inside
|
|
116
|
+
|
|
117
|
+
- **Numerics:** HLLC and local Lax-Friedrichs fluxes; the Xia et al. (2017) surface-reconstruction method and Audusse hydrostatic reconstruction for exact lake-at-rest balance over arbitrary bathymetry; first-order, MUSCL, and fifth-order reconstruction; forward Euler and SSP-RK3.
|
|
118
|
+
- **Physics:** point-implicit Manning friction, wetting and drying, gridded rainfall, Green-Ampt infiltration, depth sinks, and an inverse-distance-weighted coastal stage ring driven by NOAA CO-OPS gauge records.
|
|
119
|
+
- **Compressed active-cell mesh:** static active set chosen from terrain criteria before the run, `int16` neighbor offsets, a two-cell ghost halo, build-once caching, active-cell-balanced multi-GPU partitions, and checkpoint/restart.
|
|
120
|
+
- **Backends:** CuPy on NVIDIA GPUs (CUDA) and AMD GPUs (ROCm), with a fused single-kernel time step; `mpi4py` for multi-GPU runs with GPU-aware halo exchange; and a NumPy CPU fallback that runs the same scheme in float64.
|
|
121
|
+
|
|
122
|
+
## Installation
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
# CPU only (NumPy backend): enough for the examples, tests, and docs
|
|
126
|
+
pip install geoswe
|
|
127
|
+
|
|
128
|
+
# GPU on CUDA 12 or CUDA 13 drivers, multi-GPU, GeoTIFF I/O, and forcing readers
|
|
129
|
+
# (installs one CuPy build with its CUDA headers; do not add a second CuPy build)
|
|
130
|
+
pip install "geoswe[gpu,mpi,io,forcings]"
|
|
131
|
+
|
|
132
|
+
# The same on an AMD GPU with ROCm 7
|
|
133
|
+
pip install "geoswe[gpu-rocm,mpi,io,forcings]"
|
|
134
|
+
|
|
135
|
+
# From a source checkout (needed for the examples and the bundled terrain)
|
|
136
|
+
git clone https://github.com/GeoSWE/geoswe.git && cd geoswe
|
|
137
|
+
pip install -e ".[all]"
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
The development version installs straight from the repository:
|
|
141
|
+
`pip install "geoswe @ git+https://github.com/GeoSWE/geoswe.git"`.
|
|
142
|
+
|
|
143
|
+
See the [installation page](https://geoswe.github.io/geoswe/installation.html) for CUDA and CuPy version notes and the conda environment, and the [AMD GPUs page](https://geoswe.github.io/geoswe/amd_gpus.html) for ROCm.
|
|
144
|
+
|
|
145
|
+
## Quick start
|
|
146
|
+
|
|
147
|
+
A storm on real terrain, from the repository root. The same script runs on the
|
|
148
|
+
CPU (about half a minute) and on a GPU (two seconds):
|
|
149
|
+
|
|
150
|
+
```python
|
|
151
|
+
import numpy as np
|
|
152
|
+
from geoswe import Mesh2D, Config, Solver2D, RainfallForcing
|
|
153
|
+
|
|
154
|
+
# terrain and roughness: arrays indexed [x, y], elevations in metres
|
|
155
|
+
case = np.load("examples/data/cookcounty_mini.npz")
|
|
156
|
+
win = np.s_[336:464, 336:464] # a 1.3 km window; use np.s_[:, :] on a GPU
|
|
157
|
+
bed, manning = case["bed"][win], case["manning"][win]
|
|
158
|
+
nx, ny = bed.shape
|
|
159
|
+
mesh = Mesh2D(nx=nx, ny=ny, dx=float(case["dx"]), dy=float(case["dy"]))
|
|
160
|
+
|
|
161
|
+
# 75 mm/h for one hour on a dry bed; water leaves freely at the edges
|
|
162
|
+
rain = RainfallForcing(time_s=[0, 3600], rate_mm_h=[75, 0])
|
|
163
|
+
cfg = Config(friction="manning", bc_x="fall", bc_y="fall", rainfall_forcing=rain)
|
|
164
|
+
solver = Solver2D(mesh, cfg, np.zeros((3, nx, ny)), bed)
|
|
165
|
+
solver.set_manning(manning)
|
|
166
|
+
|
|
167
|
+
solver.run(t_end=2 * 3600)
|
|
168
|
+
peak = solver.max_depth() # NumPy array (nx, ny), metres
|
|
169
|
+
print(f"deepest water {peak.max():.2f} m; {100 * (peak > 0.05).mean():.0f}% of the area reached 5 cm")
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
`Config()` with no arguments is the production scheme of the paper (first-order
|
|
173
|
+
HLLC, SRM well balancing, forward Euler, CFL 0.5). The
|
|
174
|
+
[flood tutorial](https://geoswe.github.io/geoswe/flood_tutorial.html) explains each line and adds your own
|
|
175
|
+
GeoTIFF terrain, a tide or surge, and the compressed active-cell mesh.
|
|
176
|
+
|
|
177
|
+
[`examples/`](https://github.com/GeoSWE/geoswe/tree/main/examples) has runnable scripts for 1D and 2D dam breaks, a
|
|
178
|
+
lake-at-rest check, rain on a slope, convergence order, multi-GPU scaling, a
|
|
179
|
+
rain-driven flood on real terrain with bundled data, and the compressed mesh on
|
|
180
|
+
that terrain.
|
|
181
|
+
|
|
182
|
+
## Reproducing the paper
|
|
183
|
+
|
|
184
|
+
[`benchmark/`](https://github.com/GeoSWE/geoswe/tree/main/benchmark) holds the GeoSWE-side pipeline for the paper's
|
|
185
|
+
benchmark cases: the four-code Pinellas County comparison, the steady sheet-flow
|
|
186
|
+
plane, and the synthetic scaling sweep. Each case ships the scripts that
|
|
187
|
+
prepare the inputs, run the simulations, and produce the reported numbers; see
|
|
188
|
+
[`benchmark/README.md`](https://github.com/GeoSWE/geoswe/blob/main/benchmark/README.md) for data requirements and command
|
|
189
|
+
sequences. The Florida and CONUS application pipelines are not part of this
|
|
190
|
+
release; they exercise the same solver paths as the Pinellas case.
|
|
191
|
+
|
|
192
|
+
## Documentation
|
|
193
|
+
|
|
194
|
+
User guide, configuration reference, the compressed mesh, multi-GPU runs, and
|
|
195
|
+
the API reference: **https://geoswe.github.io/geoswe** (or build locally with
|
|
196
|
+
`pip install ".[docs]" && sphinx-build -b html docs docs/_build`).
|
|
197
|
+
|
|
198
|
+
## Citing
|
|
199
|
+
|
|
200
|
+
If you use GeoSWE, please cite it; see [CITATION.cff](https://github.com/GeoSWE/geoswe/blob/main/CITATION.cff). A paper
|
|
201
|
+
describing the method, the cross-code benchmark, and the county-to-continent
|
|
202
|
+
applications is in preparation.
|
|
203
|
+
|
|
204
|
+
## Acknowledgments
|
|
205
|
+
|
|
206
|
+
This material is based upon work supported by the National Science Foundation
|
|
207
|
+
under Grant No. 2325631, by a 2025 IDEaS + Cloud Hub award with support from
|
|
208
|
+
Microsoft at the Georgia Institute of Technology, and by the U.S. Department of
|
|
209
|
+
Energy under Contract No. DE-AC05-00OR22725 through the Genesis Mission project,
|
|
210
|
+
in collaboration with Oak Ridge National Laboratory, the Tennessee Valley
|
|
211
|
+
Authority, and AMD. Computing resources were provided in part by the Partnership
|
|
212
|
+
for an Advanced Computing Environment (PACE) at Georgia Tech and by the Frontier
|
|
213
|
+
supercomputer at the Oak Ridge Leadership Computing Facility.
|
|
214
|
+
|
|
215
|
+
## License
|
|
216
|
+
|
|
217
|
+
BSD 3-Clause; see [LICENSE](https://github.com/GeoSWE/geoswe/blob/main/LICENSE).
|
geoswe-1.0.0/README.md
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
<p align="center">
|
|
2
|
+
<img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/helene_cascade.gif" alt="Animation of Hurricane Helene rainfall and flood depth on three nested GeoSWE domains: CONUS at 30 m, Florida at 10 m, Pinellas County at 3 m" width="720">
|
|
3
|
+
</p>
|
|
4
|
+
|
|
5
|
+
<h1 align="center">GeoSWE</h1>
|
|
6
|
+
<p align="center"><b>Geophysical Shallow-Water Engine</b><br>
|
|
7
|
+
A GPU finite-volume solver for the full 2D shallow-water equations,<br>
|
|
8
|
+
from a 3 m coastal county to the conterminous United States on one multi-GPU node.</p>
|
|
9
|
+
|
|
10
|
+
<p align="center">
|
|
11
|
+
<a href="https://github.com/GeoSWE/geoswe/actions/workflows/test.yml"><img src="https://github.com/GeoSWE/geoswe/actions/workflows/test.yml/badge.svg" alt="tests"></a>
|
|
12
|
+
<a href="https://github.com/GeoSWE/geoswe/actions/workflows/lint.yml"><img src="https://github.com/GeoSWE/geoswe/actions/workflows/lint.yml/badge.svg" alt="lint"></a>
|
|
13
|
+
<a href="https://github.com/GeoSWE/geoswe/actions/workflows/docs.yml"><img src="https://github.com/GeoSWE/geoswe/actions/workflows/docs.yml/badge.svg" alt="docs"></a>
|
|
14
|
+
<a href="https://github.com/GeoSWE/geoswe/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-BSD--3--Clause-blue.svg" alt="BSD 3-Clause"></a>
|
|
15
|
+
<img src="https://img.shields.io/badge/python-3.10%2B-blue.svg" alt="Python 3.10+">
|
|
16
|
+
</p>
|
|
17
|
+
|
|
18
|
+
*Above: 72 hours of Hurricane Helene (September 2024) simulated by the same code on three nested domains: MRMS rainfall in purple, flood depth in color, and the NOAA CO-OPS tide gauges that drive the coastal boundary as stars. [Full-resolution video (MP4)](https://github.com/GeoSWE/geoswe/blob/main/docs/images/helene_cascade.mp4).*
|
|
19
|
+
|
|
20
|
+
---
|
|
21
|
+
|
|
22
|
+
GeoSWE solves the nonlinear shallow-water equations with a well-balanced HLLC
|
|
23
|
+
finite-volume scheme, Manning friction, wetting and drying, rainfall, and an
|
|
24
|
+
observation-driven coastal stage boundary. Its distinguishing feature is a
|
|
25
|
+
**compressed active-cell mesh**: the cells that matter (land plus a nearshore
|
|
26
|
+
band) are packed into flat arrays before the run, so memory and work scale
|
|
27
|
+
with the flooded landscape rather than its bounding rectangle. The same
|
|
28
|
+
finite-volume kernel runs on the dense grid and on the compressed mesh, bit for
|
|
29
|
+
bit. It runs on NVIDIA and AMD GPUs through [CuPy](https://cupy.dev), scales across
|
|
30
|
+
GPUs with `mpi4py`, and falls back to NumPy on the CPU for prototyping and CI.
|
|
31
|
+
|
|
32
|
+
## Highlights
|
|
33
|
+
|
|
34
|
+
| | |
|
|
35
|
+
|---|---|
|
|
36
|
+
| **Continental domains on one node** | A 72-hour Hurricane Helene scenario over CONUS at 30 m (8.88 billion active cells, 207-gauge coastal boundary) runs on eight H100 GPUs in 10.8 h; Florida at 10 m (1.78 billion active cells) on four GPUs in 9.5 h. |
|
|
37
|
+
| **Fastest and leanest in a four-code benchmark** | On a 214-million-cell county case against TRITON, SynxFlow, and SERGHEI, GeoSWE's active-cell configuration has the lowest wall time and GPU memory at every GPU count: 3.5 to 3.9 times faster and 1.5 to 1.9 times leaner than the nearest peer, with pairwise CSI near 0.99. |
|
|
38
|
+
| **Compression without a numerical penalty** | Dense and compressed runs take identical step counts and agree to 0.05 cm RMSE (CSI 1.000) on the same cells. |
|
|
39
|
+
| **Near-ideal scaling** | 99.5 % weak-scaling efficiency at 16 H100 GPUs (10.24 billion cells) and 98.8 % at 32 Blackwell MIG slices across two nodes (20.48 billion cells). |
|
|
40
|
+
| **Thin-film accuracy** | On a steady rained slope with a high-accuracy reference solution, GeoSWE stays within 1 % of the reference film depth where other codes depart by tens of percent. |
|
|
41
|
+
|
|
42
|
+
The application runs are computational demonstrations under stated, uncalibrated settings, not validated flood hindcasts.
|
|
43
|
+
|
|
44
|
+
## How it compares
|
|
45
|
+
|
|
46
|
+
<p align="center"><img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/intro_bench.png" alt="Benchmark: per-step wall time versus peak GPU memory per code, and thin-film depth error on a rained slope" width="900"></p>
|
|
47
|
+
|
|
48
|
+
*(a) Per-step wall time against peak GPU memory per rank for each code's fastest configuration on the 3 m Pinellas County benchmark (1, 2, and 4 GPUs; large marker = 4 GPUs). (b) Depth error against the steady shallow-water reference on a uniformly rained slope.*
|
|
49
|
+
|
|
50
|
+
Comparison codes as benchmarked: TRITON (commit `ec35bc4`), SERGHEI (commit `39a10f2`), and SynxFlow 1.0.2; all four in fp32 at CFL 0.5, first-order well-balanced schemes, same H100 node, identical inputs. Builds, decks, and patches are in the paper's reproducibility appendix.
|
|
51
|
+
|
|
52
|
+
## How it scales
|
|
53
|
+
|
|
54
|
+
<p align="center"><img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/scaling.png" alt="Strong scaling, weak scaling, and memory per rank on 1 to 16 H100 GPUs" width="900"></p>
|
|
55
|
+
|
|
56
|
+
*Dense and flat storage layouts on an everywhere-wet synthetic domain, 640 million cells per GPU, 1 to 16 H100 GPUs across two nodes. Strong scaling reaches 15.5x on 16 GPUs for the flat layout; weak scaling stays above 99 %.*
|
|
57
|
+
|
|
58
|
+
<p align="center"><img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/scaling_frontier.png" alt="Weak scaling on OLCF Frontier at one billion cells per GPU: 1 to 32 GPUs, and 1 to 128 nodes with 1.024 trillion cells" width="900"></p>
|
|
59
|
+
|
|
60
|
+
*The flat layout on AMD GPUs at OLCF Frontier, one billion cells per GPU (one GCD, half of an MI250X card; eight per node): time per step divided by that of the smallest size, so ideal weak scaling is the dashed line. On 128 nodes, 1.024 trillion cells advance at 162 ms per step: 98.0 % weak-scaling efficiency against one node and 96.7 % against one GPU. A marker is the mean of up to three launches, each timed over one 40-second window; see [`benchmark/frontier_amd`](https://github.com/GeoSWE/geoswe/blob/main/benchmark/frontier_amd/README.md).*
|
|
61
|
+
|
|
62
|
+
## What is inside
|
|
63
|
+
|
|
64
|
+
- **Numerics:** HLLC and local Lax-Friedrichs fluxes; the Xia et al. (2017) surface-reconstruction method and Audusse hydrostatic reconstruction for exact lake-at-rest balance over arbitrary bathymetry; first-order, MUSCL, and fifth-order reconstruction; forward Euler and SSP-RK3.
|
|
65
|
+
- **Physics:** point-implicit Manning friction, wetting and drying, gridded rainfall, Green-Ampt infiltration, depth sinks, and an inverse-distance-weighted coastal stage ring driven by NOAA CO-OPS gauge records.
|
|
66
|
+
- **Compressed active-cell mesh:** static active set chosen from terrain criteria before the run, `int16` neighbor offsets, a two-cell ghost halo, build-once caching, active-cell-balanced multi-GPU partitions, and checkpoint/restart.
|
|
67
|
+
- **Backends:** CuPy on NVIDIA GPUs (CUDA) and AMD GPUs (ROCm), with a fused single-kernel time step; `mpi4py` for multi-GPU runs with GPU-aware halo exchange; and a NumPy CPU fallback that runs the same scheme in float64.
|
|
68
|
+
|
|
69
|
+
## Installation
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
# CPU only (NumPy backend): enough for the examples, tests, and docs
|
|
73
|
+
pip install geoswe
|
|
74
|
+
|
|
75
|
+
# GPU on CUDA 12 or CUDA 13 drivers, multi-GPU, GeoTIFF I/O, and forcing readers
|
|
76
|
+
# (installs one CuPy build with its CUDA headers; do not add a second CuPy build)
|
|
77
|
+
pip install "geoswe[gpu,mpi,io,forcings]"
|
|
78
|
+
|
|
79
|
+
# The same on an AMD GPU with ROCm 7
|
|
80
|
+
pip install "geoswe[gpu-rocm,mpi,io,forcings]"
|
|
81
|
+
|
|
82
|
+
# From a source checkout (needed for the examples and the bundled terrain)
|
|
83
|
+
git clone https://github.com/GeoSWE/geoswe.git && cd geoswe
|
|
84
|
+
pip install -e ".[all]"
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
The development version installs straight from the repository:
|
|
88
|
+
`pip install "geoswe @ git+https://github.com/GeoSWE/geoswe.git"`.
|
|
89
|
+
|
|
90
|
+
See the [installation page](https://geoswe.github.io/geoswe/installation.html) for CUDA and CuPy version notes and the conda environment, and the [AMD GPUs page](https://geoswe.github.io/geoswe/amd_gpus.html) for ROCm.
|
|
91
|
+
|
|
92
|
+
## Quick start
|
|
93
|
+
|
|
94
|
+
A storm on real terrain, from the repository root. The same script runs on the
|
|
95
|
+
CPU (about half a minute) and on a GPU (two seconds):
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
import numpy as np
|
|
99
|
+
from geoswe import Mesh2D, Config, Solver2D, RainfallForcing
|
|
100
|
+
|
|
101
|
+
# terrain and roughness: arrays indexed [x, y], elevations in metres
|
|
102
|
+
case = np.load("examples/data/cookcounty_mini.npz")
|
|
103
|
+
win = np.s_[336:464, 336:464] # a 1.3 km window; use np.s_[:, :] on a GPU
|
|
104
|
+
bed, manning = case["bed"][win], case["manning"][win]
|
|
105
|
+
nx, ny = bed.shape
|
|
106
|
+
mesh = Mesh2D(nx=nx, ny=ny, dx=float(case["dx"]), dy=float(case["dy"]))
|
|
107
|
+
|
|
108
|
+
# 75 mm/h for one hour on a dry bed; water leaves freely at the edges
|
|
109
|
+
rain = RainfallForcing(time_s=[0, 3600], rate_mm_h=[75, 0])
|
|
110
|
+
cfg = Config(friction="manning", bc_x="fall", bc_y="fall", rainfall_forcing=rain)
|
|
111
|
+
solver = Solver2D(mesh, cfg, np.zeros((3, nx, ny)), bed)
|
|
112
|
+
solver.set_manning(manning)
|
|
113
|
+
|
|
114
|
+
solver.run(t_end=2 * 3600)
|
|
115
|
+
peak = solver.max_depth() # NumPy array (nx, ny), metres
|
|
116
|
+
print(f"deepest water {peak.max():.2f} m; {100 * (peak > 0.05).mean():.0f}% of the area reached 5 cm")
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
`Config()` with no arguments is the production scheme of the paper (first-order
|
|
120
|
+
HLLC, SRM well balancing, forward Euler, CFL 0.5). The
|
|
121
|
+
[flood tutorial](https://geoswe.github.io/geoswe/flood_tutorial.html) explains each line and adds your own
|
|
122
|
+
GeoTIFF terrain, a tide or surge, and the compressed active-cell mesh.
|
|
123
|
+
|
|
124
|
+
[`examples/`](https://github.com/GeoSWE/geoswe/tree/main/examples) has runnable scripts for 1D and 2D dam breaks, a
|
|
125
|
+
lake-at-rest check, rain on a slope, convergence order, multi-GPU scaling, a
|
|
126
|
+
rain-driven flood on real terrain with bundled data, and the compressed mesh on
|
|
127
|
+
that terrain.
|
|
128
|
+
|
|
129
|
+
## Reproducing the paper
|
|
130
|
+
|
|
131
|
+
[`benchmark/`](https://github.com/GeoSWE/geoswe/tree/main/benchmark) holds the GeoSWE-side pipeline for the paper's
|
|
132
|
+
benchmark cases: the four-code Pinellas County comparison, the steady sheet-flow
|
|
133
|
+
plane, and the synthetic scaling sweep. Each case ships the scripts that
|
|
134
|
+
prepare the inputs, run the simulations, and produce the reported numbers; see
|
|
135
|
+
[`benchmark/README.md`](https://github.com/GeoSWE/geoswe/blob/main/benchmark/README.md) for data requirements and command
|
|
136
|
+
sequences. The Florida and CONUS application pipelines are not part of this
|
|
137
|
+
release; they exercise the same solver paths as the Pinellas case.
|
|
138
|
+
|
|
139
|
+
## Documentation
|
|
140
|
+
|
|
141
|
+
User guide, configuration reference, the compressed mesh, multi-GPU runs, and
|
|
142
|
+
the API reference: **https://geoswe.github.io/geoswe** (or build locally with
|
|
143
|
+
`pip install ".[docs]" && sphinx-build -b html docs docs/_build`).
|
|
144
|
+
|
|
145
|
+
## Citing
|
|
146
|
+
|
|
147
|
+
If you use GeoSWE, please cite it; see [CITATION.cff](https://github.com/GeoSWE/geoswe/blob/main/CITATION.cff). A paper
|
|
148
|
+
describing the method, the cross-code benchmark, and the county-to-continent
|
|
149
|
+
applications is in preparation.
|
|
150
|
+
|
|
151
|
+
## Acknowledgments
|
|
152
|
+
|
|
153
|
+
This material is based upon work supported by the National Science Foundation
|
|
154
|
+
under Grant No. 2325631, by a 2025 IDEaS + Cloud Hub award with support from
|
|
155
|
+
Microsoft at the Georgia Institute of Technology, and by the U.S. Department of
|
|
156
|
+
Energy under Contract No. DE-AC05-00OR22725 through the Genesis Mission project,
|
|
157
|
+
in collaboration with Oak Ridge National Laboratory, the Tennessee Valley
|
|
158
|
+
Authority, and AMD. Computing resources were provided in part by the Partnership
|
|
159
|
+
for an Advanced Computing Environment (PACE) at Georgia Tech and by the Frontier
|
|
160
|
+
supercomputer at the Oak Ridge Leadership Computing Facility.
|
|
161
|
+
|
|
162
|
+
## License
|
|
163
|
+
|
|
164
|
+
BSD 3-Clause; see [LICENSE](https://github.com/GeoSWE/geoswe/blob/main/LICENSE).
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
# AMD GPUs (ROCm)
|
|
2
|
+
|
|
3
|
+
GeoSWE runs on AMD GPUs through CuPy's ROCm build. The API, the kernels and the
|
|
4
|
+
`SWE_*` settings are the same as on NVIDIA; `geoswe.gpu_platform()` returns
|
|
5
|
+
`"hip"` instead of `"cuda"`.
|
|
6
|
+
|
|
7
|
+
```{admonition} What has been tested
|
|
8
|
+
:class: note
|
|
9
|
+
|
|
10
|
+
AMD Instinct MI210 and MI250X (`gfx90a`) on OLCF Frontier, with ROCm 7.0.2 and
|
|
11
|
+
7.2.0 and two CuPy builds: `cupy-rocm-7-0` 14.2.0 from PyPI and AMD's `amd-cupy`
|
|
12
|
+
13.5.1. Both pass the whole GPU test suite. Other AMD GPUs and ROCm versions are
|
|
13
|
+
untested. CuPy itself still labels its ROCm support experimental.
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
## Installing
|
|
17
|
+
|
|
18
|
+
CuPy's ROCm build compiles kernels with the ROCm installation on the machine, so
|
|
19
|
+
ROCm 7 must be installed and visible at run time:
|
|
20
|
+
|
|
21
|
+
- `hipcc` on `PATH` (CuPy runs it to find the include directories), and
|
|
22
|
+
- `ROCM_HOME` pointing at the ROCm tree, for example `/opt/rocm-7.2.0`.
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pip install "geoswe[gpu-rocm]" # CuPy for ROCm 7 from PyPI
|
|
26
|
+
pip install "geoswe[gpu-rocm,mpi,io,forcings]" # with MPI, GeoTIFF I/O and forcings
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
AMD also publishes its own build, which OLCF documents for Frontier. Install it
|
|
30
|
+
instead of the extra, never next to it:
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
pip install amd-cupy --extra-index-url https://pypi.amd.com/rocm-7.2.0/simple
|
|
34
|
+
pip install "geoswe[mpi,io,forcings]"
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Check the install:
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
python -c "import geoswe; print(geoswe.get_backend(), geoswe.gpu_platform())" # cupy hip
|
|
41
|
+
pytest -m gpu -q
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
The first run compiles every kernel, which takes a minute or so; later runs load
|
|
45
|
+
them from CuPy's cache. On a cluster, put that cache on a filesystem the compute
|
|
46
|
+
nodes share, and keep AMD's own compiler cache off NFS inside jobs:
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
export CUPY_CACHE_DIR=/path/on/scratch/cupy_cache # default: ~/.cupy/kernel_cache
|
|
50
|
+
export AMD_COMGR_CACHE_DIR=/tmp/$USER-comgr # default: ~/.cache/comgr
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
## What differs from NVIDIA
|
|
54
|
+
|
|
55
|
+
The kernels are CUDA C. NVIDIA's compiler (NVRTC) and AMD's (clang, through HIP)
|
|
56
|
+
disagree about three things in them, and `geoswe.backend` handles each one. On
|
|
57
|
+
NVIDIA nothing changes: kernel sources and options reach CuPy exactly as before.
|
|
58
|
+
|
|
59
|
+
| | NVIDIA | AMD | What GeoSWE does on AMD |
|
|
60
|
+
|---|---|---|---|
|
|
61
|
+
| Register cap `-maxrregcount` (`SWE_DENSE_MAXRREG`, `SWE_FLAT_MAXRREG`) | applied on H100 | not an option of the compiler | never requested; ignored with a warning if set by hand |
|
|
62
|
+
| `__fmul_rn` / `__fadd_rn` | never fused into a multiply-add | plain `*` and `+`, which clang fuses | redefined with contraction switched off |
|
|
63
|
+
| Warp-level compaction | 32 lanes, 32-bit masks | 64 lanes, 64-bit masks | a kernel without warp intrinsics |
|
|
64
|
+
|
|
65
|
+
The first one also needed a fix to device detection: CuPy reports the compute
|
|
66
|
+
capability of a `gfx90a` card as `"90"`, the same string as an H100.
|
|
67
|
+
|
|
68
|
+
### Floating-point contraction
|
|
69
|
+
|
|
70
|
+
`GEOSWE_HIP_FP_CONTRACT` sets how the ROCm compiler may fuse `a*b + c` into one
|
|
71
|
+
multiply-add:
|
|
72
|
+
|
|
73
|
+
| Value | Meaning |
|
|
74
|
+
|---|---|
|
|
75
|
+
| `off` (default) | never; every operation rounds once, in source order |
|
|
76
|
+
| `on` | only inside a single source expression |
|
|
77
|
+
| `fast` | at the optimizer's discretion, across statements (clang's own default for HIP) |
|
|
78
|
+
|
|
79
|
+
GeoSWE relies on several code paths giving identical bits: the fused and the
|
|
80
|
+
split time step, the dense and the compressed mesh, a run and its restart. With
|
|
81
|
+
`fast` the same expression can be fused in one kernel and not in another, and the
|
|
82
|
+
fused and split steps with sub-grid storage then differ in the last bit
|
|
83
|
+
(`tests/test_gpu_storage_fused.py` fails). With `off` or `on` the arithmetic is
|
|
84
|
+
fixed by the source text.
|
|
85
|
+
|
|
86
|
+
`off` is the default because it makes those identities hold by construction: two
|
|
87
|
+
kernels agree whenever their statements do. It is the only mode in which the
|
|
88
|
+
dense and the compressed solver agree bit for bit on the bowl-and-dam problem of
|
|
89
|
+
`tests/test_gpu_compressed_equiv.py`; `on` and `fast` leave one unit in the last
|
|
90
|
+
place between them. `on` is about 3 % faster (25.0 against 25.8 ms/step on an
|
|
91
|
+
MI250X at 147 M cells; `fast` is in between) and keeps the fused and split steps
|
|
92
|
+
identical in the test suite. All three modes keep a lake at rest to round-off.
|
|
93
|
+
|
|
94
|
+
Results on AMD and NVIDIA are not bit-identical to each other: the compilers and
|
|
95
|
+
their math libraries differ, at round-off level.
|
|
96
|
+
|
|
97
|
+
## Multiple GPUs
|
|
98
|
+
|
|
99
|
+
One MPI rank drives one GPU, as on NVIDIA. Under Slurm, let the scheduler bind
|
|
100
|
+
the devices:
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
srun -n 8 --gpus-per-task=1 --gpu-bind=closest python my_run.py
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
The halo travels through host memory unless `SWE_HALO_CUDA_AWARE=1` asks for
|
|
107
|
+
GPU-aware MPI. Despite the name, that setting is not specific to CUDA: CuPy's
|
|
108
|
+
ROCm build exposes device arrays through the same interface, and `mpi4py` hands
|
|
109
|
+
the pointers to MPI. With HPE Cray MPICH this needs two more things:
|
|
110
|
+
|
|
111
|
+
- `mpi4py` built with the `craype-accel-amd-*` module loaded, so that it links the
|
|
112
|
+
GPU transport library, and
|
|
113
|
+
- `MPICH_GPU_SUPPORT_ENABLED=1` in the job.
|
|
114
|
+
|
|
115
|
+
If `SWE_HALO_CUDA_AWARE=1` is set while Cray MPICH runs without its GPU support,
|
|
116
|
+
GeoSWE warns and stages through the host instead of passing device pointers to a
|
|
117
|
+
library that would treat them as host memory.
|
|
118
|
+
|
|
119
|
+
## OLCF Frontier
|
|
120
|
+
|
|
121
|
+
`benchmark/frontier_amd/` in the repository has the module set, the build steps
|
|
122
|
+
for a GPU-aware `mpi4py`, and a one-node job that runs the GPU test suite, a
|
|
123
|
+
partition-invariance check and a weak-scaling sweep. Measured there on one node
|
|
124
|
+
(an MI250X card is two devices, so eight per node; 147 M cells per device,
|
|
125
|
+
float32, host-staged halo):
|
|
126
|
+
|
|
127
|
+
| Tier | 1 device | 8 devices | Weak efficiency |
|
|
128
|
+
|---|---|---|---|
|
|
129
|
+
| Dense | 25.8 ms/step | 26.1 ms/step | 99.0 % |
|
|
130
|
+
| Compressed | 23.7 ms/step | 23.9 ms/step | 99.2 % |
|
|
131
|
+
|
|
132
|
+
Repeat launches differ by up to 2 %. The global solution is bit-identical on 1,
|
|
133
|
+
2, 4 and 8 devices, with the host-staged and the GPU-aware halo, with and without
|
|
134
|
+
the halo/compute overlap.
|
|
135
|
+
|
|
136
|
+
On two nodes, the scaling harness of `benchmark/scaling_640m` (640 M cells per
|
|
137
|
+
device, in the configuration its launchers pin) gives:
|
|
138
|
+
|
|
139
|
+
| Layout | Weak scaling, 16 devices, 10.24 B cells | Strong scaling, 16 devices |
|
|
140
|
+
|---|---|---|
|
|
141
|
+
| Flat | 118.7 ms/step, 99.2 % efficiency | 15.0x |
|
|
142
|
+
| Dense | 190.2 ms/step, 99.6 % efficiency | 15.4x |
|
|
143
|
+
|
|
144
|
+
With the solver's defaults the same 10.24 billion cells take 102 ms/step (flat)
|
|
145
|
+
and 112 ms/step (dense). The solution on sixteen devices across the two nodes has
|
|
146
|
+
the digest of the one-device run.
|
|
147
|
+
|
|
148
|
+
```{note}
|
|
149
|
+
A device here is one GCD, half an MI250X card. For the same 640 M cells with the
|
|
150
|
+
solver's defaults, the H100 of the scaling figure in the
|
|
151
|
+
[repository README](https://github.com/GeoSWE/geoswe#how-it-scales) takes about
|
|
152
|
+
42 ms/step (flat) and 46 ms/step (dense), where one GCD takes 101 and 111. That
|
|
153
|
+
is 2.4 times faster, the ratio of their FP32 lanes (16,896 CUDA cores to 7,040
|
|
154
|
+
stream processors), so a whole MI250X card delivers about 0.83 of an H100. The
|
|
155
|
+
machines differ in more than the GPU: read this as arithmetic on published
|
|
156
|
+
numbers, not as a controlled comparison.
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
## Not done yet
|
|
160
|
+
|
|
161
|
+
- Of the paper's benchmark cases, only the synthetic scaling harness has been run
|
|
162
|
+
on AMD hardware.
|
|
163
|
+
- The register cap that speeds up the residual kernel on H100 has no ROCm
|
|
164
|
+
counterpart, and no AMD-specific tuning of block sizes has been tried.
|
|
165
|
+
- Two nodes (16 devices) are the most that has been used.
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
Compressed active-cell solver
|
|
2
|
+
=============================
|
|
3
|
+
|
|
4
|
+
GPU-only. Imported lazily as ``geoswe.CompressedSolver``. See
|
|
5
|
+
:doc:`../compressed_mesh` for the concepts.
|
|
6
|
+
|
|
7
|
+
.. currentmodule:: geoswe.compressed_solver
|
|
8
|
+
|
|
9
|
+
.. autoclass:: CompressedSolver
|
|
10
|
+
:members:
|
|
11
|
+
:member-order: bysource
|
|
12
|
+
|
|
13
|
+
.. autofunction:: run_cached
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
Forcings and I/O
|
|
2
|
+
================
|
|
3
|
+
|
|
4
|
+
.. currentmodule:: geoswe.forcing
|
|
5
|
+
|
|
6
|
+
.. autoclass:: RainfallForcing
|
|
7
|
+
:members:
|
|
8
|
+
|
|
9
|
+
.. autoclass:: StageBoundary
|
|
10
|
+
:members:
|
|
11
|
+
|
|
12
|
+
GeoTIFF I/O
|
|
13
|
+
-----------
|
|
14
|
+
|
|
15
|
+
.. currentmodule:: geoswe.io_geotiff
|
|
16
|
+
|
|
17
|
+
.. autoclass:: GeoArray
|
|
18
|
+
:members:
|
|
19
|
+
|
|
20
|
+
.. autofunction:: read_geotiff
|
|
21
|
+
.. autofunction:: write_geotiff
|