geoswe 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. geoswe-1.0.0/CITATION.cff +29 -0
  2. geoswe-1.0.0/LICENSE +29 -0
  3. geoswe-1.0.0/MANIFEST.in +13 -0
  4. geoswe-1.0.0/PKG-INFO +217 -0
  5. geoswe-1.0.0/README.md +164 -0
  6. geoswe-1.0.0/docs/amd_gpus.md +165 -0
  7. geoswe-1.0.0/docs/api/compressed.rst +13 -0
  8. geoswe-1.0.0/docs/api/forcing.rst +21 -0
  9. geoswe-1.0.0/docs/api/index.md +18 -0
  10. geoswe-1.0.0/docs/api/mesh.rst +22 -0
  11. geoswe-1.0.0/docs/api/runlib.rst +37 -0
  12. geoswe-1.0.0/docs/api/solver.rst +16 -0
  13. geoswe-1.0.0/docs/benchmarks.md +81 -0
  14. geoswe-1.0.0/docs/citing.md +23 -0
  15. geoswe-1.0.0/docs/compressed_mesh.md +96 -0
  16. geoswe-1.0.0/docs/conf.py +95 -0
  17. geoswe-1.0.0/docs/configuration.md +194 -0
  18. geoswe-1.0.0/docs/examples.md +96 -0
  19. geoswe-1.0.0/docs/flood_tutorial.md +162 -0
  20. geoswe-1.0.0/docs/images/helene_cascade.gif +0 -0
  21. geoswe-1.0.0/docs/images/helene_cascade.mp4 +0 -0
  22. geoswe-1.0.0/docs/images/intro_bench.png +0 -0
  23. geoswe-1.0.0/docs/images/scaling.png +0 -0
  24. geoswe-1.0.0/docs/images/scaling_frontier.png +0 -0
  25. geoswe-1.0.0/docs/index.md +84 -0
  26. geoswe-1.0.0/docs/installation.md +127 -0
  27. geoswe-1.0.0/docs/multigpu_mpi.md +72 -0
  28. geoswe-1.0.0/docs/quickstart.md +98 -0
  29. geoswe-1.0.0/docs/requirements.txt +9 -0
  30. geoswe-1.0.0/docs/userguide/boundary_conditions.md +52 -0
  31. geoswe-1.0.0/docs/userguide/forcings.md +102 -0
  32. geoswe-1.0.0/docs/userguide/friction.md +69 -0
  33. geoswe-1.0.0/docs/userguide/governing_equations.md +65 -0
  34. geoswe-1.0.0/docs/userguide/numerical_methods.md +78 -0
  35. geoswe-1.0.0/docs/userguide/well_balanced.md +44 -0
  36. geoswe-1.0.0/environment.yml +25 -0
  37. geoswe-1.0.0/examples/README.md +41 -0
  38. geoswe-1.0.0/examples/data/cookcounty_mini.npz +0 -0
  39. geoswe-1.0.0/examples/ex01_dam_break_1d.py +72 -0
  40. geoswe-1.0.0/examples/ex02_circular_dam_break_2d.py +65 -0
  41. geoswe-1.0.0/examples/ex03_lake_at_rest_2d.py +60 -0
  42. geoswe-1.0.0/examples/ex04_rain_on_slope_2d.py +77 -0
  43. geoswe-1.0.0/examples/ex05_scaling_bench.py +97 -0
  44. geoswe-1.0.0/examples/ex06_pluvial_flood_realcase.py +114 -0
  45. geoswe-1.0.0/examples/ex07_convergence_order.py +125 -0
  46. geoswe-1.0.0/examples/ex08_compressed_mesh.py +102 -0
  47. geoswe-1.0.0/pyproject.toml +88 -0
  48. geoswe-1.0.0/setup.cfg +4 -0
  49. geoswe-1.0.0/src/geoswe/__init__.py +102 -0
  50. geoswe-1.0.0/src/geoswe/backend.py +335 -0
  51. geoswe-1.0.0/src/geoswe/bc.py +235 -0
  52. geoswe-1.0.0/src/geoswe/compressed_mesh.py +191 -0
  53. geoswe-1.0.0/src/geoswe/compressed_rhs.py +1082 -0
  54. geoswe-1.0.0/src/geoswe/compressed_solver.py +3085 -0
  55. geoswe-1.0.0/src/geoswe/data_prep.py +399 -0
  56. geoswe-1.0.0/src/geoswe/elliptic.py +154 -0
  57. geoswe-1.0.0/src/geoswe/elliptic_cuda.py +258 -0
  58. geoswe-1.0.0/src/geoswe/flux.py +151 -0
  59. geoswe-1.0.0/src/geoswe/forcing.py +329 -0
  60. geoswe-1.0.0/src/geoswe/friction.py +108 -0
  61. geoswe-1.0.0/src/geoswe/gauges.py +109 -0
  62. geoswe-1.0.0/src/geoswe/io_geotiff.py +220 -0
  63. geoswe-1.0.0/src/geoswe/mesh.py +87 -0
  64. geoswe-1.0.0/src/geoswe/mpi_halo.py +495 -0
  65. geoswe-1.0.0/src/geoswe/reconstruction.py +195 -0
  66. geoswe-1.0.0/src/geoswe/rhs_cuda.py +2487 -0
  67. geoswe-1.0.0/src/geoswe/runlib/__init__.py +27 -0
  68. geoswe-1.0.0/src/geoswe/runlib/case.py +117 -0
  69. geoswe-1.0.0/src/geoswe/runlib/cli.py +126 -0
  70. geoswe-1.0.0/src/geoswe/runlib/driver.py +1546 -0
  71. geoswe-1.0.0/src/geoswe/runlib/replay.py +162 -0
  72. geoswe-1.0.0/src/geoswe/solver.py +2721 -0
  73. geoswe-1.0.0/src/geoswe/swe.py +74 -0
  74. geoswe-1.0.0/src/geoswe/well_balanced.py +415 -0
  75. geoswe-1.0.0/src/geoswe.egg-info/PKG-INFO +217 -0
  76. geoswe-1.0.0/src/geoswe.egg-info/SOURCES.txt +106 -0
  77. geoswe-1.0.0/src/geoswe.egg-info/dependency_links.txt +1 -0
  78. geoswe-1.0.0/src/geoswe.egg-info/requires.txt +37 -0
  79. geoswe-1.0.0/src/geoswe.egg-info/top_level.txt +1 -0
  80. geoswe-1.0.0/tests/conftest.py +111 -0
  81. geoswe-1.0.0/tests/mpi_bitcheck.py +102 -0
  82. geoswe-1.0.0/tests/parity_gate.py +72 -0
  83. geoswe-1.0.0/tests/test_api_smoke.py +60 -0
  84. geoswe-1.0.0/tests/test_backend_cupy_builds.py +37 -0
  85. geoswe-1.0.0/tests/test_bc.py +93 -0
  86. geoswe-1.0.0/tests/test_cfl_guards.py +48 -0
  87. geoswe-1.0.0/tests/test_config_validation.py +60 -0
  88. geoswe-1.0.0/tests/test_dam_break_1d.py +28 -0
  89. geoswe-1.0.0/tests/test_friction_dry.py +68 -0
  90. geoswe-1.0.0/tests/test_gpu_checkpoint.py +201 -0
  91. geoswe-1.0.0/tests/test_gpu_compressed_equiv.py +134 -0
  92. geoswe-1.0.0/tests/test_gpu_compressed_from_dense.py +144 -0
  93. geoswe-1.0.0/tests/test_gpu_dense_fused_forcings.py +97 -0
  94. geoswe-1.0.0/tests/test_gpu_platform.py +210 -0
  95. geoswe-1.0.0/tests/test_gpu_portability.py +203 -0
  96. geoswe-1.0.0/tests/test_gpu_smoke.py +57 -0
  97. geoswe-1.0.0/tests/test_gpu_storage_fused.py +138 -0
  98. geoswe-1.0.0/tests/test_hllc_dry_front.py +55 -0
  99. geoswe-1.0.0/tests/test_inflow_bc.py +95 -0
  100. geoswe-1.0.0/tests/test_inflow_compressed.py +71 -0
  101. geoswe-1.0.0/tests/test_kernel_sources_ascii.py +35 -0
  102. geoswe-1.0.0/tests/test_rain_row_window.py +93 -0
  103. geoswe-1.0.0/tests/test_reconstruction_order.py +63 -0
  104. geoswe-1.0.0/tests/test_sigma_storage_guard.py +47 -0
  105. geoswe-1.0.0/tests/test_solver_convergence_order.py +63 -0
  106. geoswe-1.0.0/tests/test_srm_cpu_matches_gpu.py +59 -0
  107. geoswe-1.0.0/tests/test_user_conveniences.py +162 -0
  108. geoswe-1.0.0/tests/test_well_balanced.py +29 -0
@@ -0,0 +1,29 @@
1
+ cff-version: 1.2.0
2
+ message: "If you use GeoSWE in your research, please cite it as below."
3
+ title: "GeoSWE: Geophysical Shallow-Water Engine"
4
+ abstract: >-
5
+ GeoSWE is a GPU-accelerated finite-volume solver for the 2D nonlinear
6
+ shallow-water equations, designed for flood modeling from coastal county to
7
+ continental scale. It combines HLLC/Lax-Friedrichs fluxes, well-balanced
8
+ surface reconstruction, implicit Manning friction, rainfall and coastal-stage
9
+ forcings, and a compressed active-cell mesh that reaches continental scale on
10
+ a single GPU node and scales across nodes with MPI.
11
+ type: software
12
+ authors:
13
+ - family-names: Chen
14
+ given-names: Peng
15
+ email: pchen402@gatech.edu
16
+ affiliation: "Georgia Institute of Technology"
17
+ version: "1.0.0"
18
+ date-released: "2026-10-06"
19
+ license: BSD-3-Clause
20
+ repository-code: "https://github.com/GeoSWE/geoswe"
21
+ url: "https://geoswe.github.io/geoswe"
22
+ keywords:
23
+ - shallow-water equations
24
+ - flood modeling
25
+ - GPU computing
26
+ - finite-volume method
27
+ - CuPy
28
+ # Add a `preferred-citation:` block for the article once it is accepted
29
+ # (title, authors, journal, year, doi), so citation tools prefer the paper.
geoswe-1.0.0/LICENSE ADDED
@@ -0,0 +1,29 @@
1
+ BSD 3-Clause License
2
+
3
+ Copyright (c) 2026, Peng Chen and the GeoSWE contributors.
4
+ All rights reserved.
5
+
6
+ Redistribution and use in source and binary forms, with or without
7
+ modification, are permitted provided that the following conditions are met:
8
+
9
+ 1. Redistributions of source code must retain the above copyright notice, this
10
+ list of conditions and the following disclaimer.
11
+
12
+ 2. Redistributions in binary form must reproduce the above copyright notice,
13
+ this list of conditions and the following disclaimer in the documentation
14
+ and/or other materials provided with the distribution.
15
+
16
+ 3. Neither the name of the copyright holder nor the names of its
17
+ contributors may be used to endorse or promote products derived from
18
+ this software without specific prior written permission.
19
+
20
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
21
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
22
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
23
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
24
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
25
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
26
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
27
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
28
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
29
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,13 @@
1
+ include LICENSE
2
+ include README.md
3
+ include CITATION.cff
4
+ include environment.yml
5
+ recursive-include src/geoswe *.py
6
+ graft examples
7
+ graft tests
8
+ graft docs
9
+ prune docs/_build
10
+ # internal review notes and release planning stay in the repo, not the sdist
11
+ prune docs/dev
12
+ global-exclude __pycache__/*
13
+ global-exclude *.py[cod]
geoswe-1.0.0/PKG-INFO ADDED
@@ -0,0 +1,217 @@
1
+ Metadata-Version: 2.4
2
+ Name: geoswe
3
+ Version: 1.0.0
4
+ Summary: GeoSWE (Geophysical Shallow-Water Engine): a GPU flood solver with a NumPy CPU fallback.
5
+ Author-email: Peng Chen <pchen402@gatech.edu>
6
+ License-Expression: BSD-3-Clause
7
+ Project-URL: Homepage, https://github.com/GeoSWE/geoswe
8
+ Project-URL: Documentation, https://geoswe.github.io/geoswe
9
+ Project-URL: Repository, https://github.com/GeoSWE/geoswe
10
+ Project-URL: Issues, https://github.com/GeoSWE/geoswe/issues
11
+ Keywords: shallow-water-equations,flood-modeling,computational-fluid-dynamics,GPU,CuPy,finite-volume,HLLC,well-balanced
12
+ Classifier: Development Status :: 5 - Production/Stable
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.10
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Topic :: Scientific/Engineering
19
+ Classifier: Topic :: Scientific/Engineering :: Hydrology
20
+ Classifier: Operating System :: POSIX :: Linux
21
+ Requires-Python: >=3.10
22
+ Description-Content-Type: text/markdown
23
+ License-File: LICENSE
24
+ Requires-Dist: numpy>=1.24
25
+ Provides-Extra: gpu
26
+ Requires-Dist: cupy-cuda12x[ctk]>=14.0; extra == "gpu"
27
+ Requires-Dist: scipy>=1.10; extra == "gpu"
28
+ Provides-Extra: gpu-cuda13
29
+ Requires-Dist: cupy-cuda13x[ctk]>=14.0; extra == "gpu-cuda13"
30
+ Requires-Dist: scipy>=1.10; extra == "gpu-cuda13"
31
+ Provides-Extra: gpu-rocm
32
+ Requires-Dist: cupy-rocm-7-0>=14.0; extra == "gpu-rocm"
33
+ Requires-Dist: scipy>=1.10; extra == "gpu-rocm"
34
+ Provides-Extra: mpi
35
+ Requires-Dist: mpi4py>=3.1; extra == "mpi"
36
+ Provides-Extra: io
37
+ Requires-Dist: rasterio>=1.3; extra == "io"
38
+ Requires-Dist: pyproj>=3.5; extra == "io"
39
+ Provides-Extra: forcings
40
+ Requires-Dist: pandas>=2.0; extra == "forcings"
41
+ Requires-Dist: scipy>=1.10; extra == "forcings"
42
+ Provides-Extra: docs
43
+ Requires-Dist: sphinx>=7; extra == "docs"
44
+ Requires-Dist: myst-parser>=2; extra == "docs"
45
+ Requires-Dist: furo; extra == "docs"
46
+ Requires-Dist: sphinx-copybutton; extra == "docs"
47
+ Requires-Dist: sphinx-design; extra == "docs"
48
+ Provides-Extra: test
49
+ Requires-Dist: pytest>=7; extra == "test"
50
+ Provides-Extra: all
51
+ Requires-Dist: geoswe[forcings,gpu,io,mpi]; extra == "all"
52
+ Dynamic: license-file
53
+
54
+ <p align="center">
55
+ <img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/helene_cascade.gif" alt="Animation of Hurricane Helene rainfall and flood depth on three nested GeoSWE domains: CONUS at 30 m, Florida at 10 m, Pinellas County at 3 m" width="720">
56
+ </p>
57
+
58
+ <h1 align="center">GeoSWE</h1>
59
+ <p align="center"><b>Geophysical Shallow-Water Engine</b><br>
60
+ A GPU finite-volume solver for the full 2D shallow-water equations,<br>
61
+ from a 3 m coastal county to the conterminous United States on one multi-GPU node.</p>
62
+
63
+ <p align="center">
64
+ <a href="https://github.com/GeoSWE/geoswe/actions/workflows/test.yml"><img src="https://github.com/GeoSWE/geoswe/actions/workflows/test.yml/badge.svg" alt="tests"></a>
65
+ <a href="https://github.com/GeoSWE/geoswe/actions/workflows/lint.yml"><img src="https://github.com/GeoSWE/geoswe/actions/workflows/lint.yml/badge.svg" alt="lint"></a>
66
+ <a href="https://github.com/GeoSWE/geoswe/actions/workflows/docs.yml"><img src="https://github.com/GeoSWE/geoswe/actions/workflows/docs.yml/badge.svg" alt="docs"></a>
67
+ <a href="https://github.com/GeoSWE/geoswe/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-BSD--3--Clause-blue.svg" alt="BSD 3-Clause"></a>
68
+ <img src="https://img.shields.io/badge/python-3.10%2B-blue.svg" alt="Python 3.10+">
69
+ </p>
70
+
71
+ *Above: 72 hours of Hurricane Helene (September 2024) simulated by the same code on three nested domains: MRMS rainfall in purple, flood depth in color, and the NOAA CO-OPS tide gauges that drive the coastal boundary as stars. [Full-resolution video (MP4)](https://github.com/GeoSWE/geoswe/blob/main/docs/images/helene_cascade.mp4).*
72
+
73
+ ---
74
+
75
+ GeoSWE solves the nonlinear shallow-water equations with a well-balanced HLLC
76
+ finite-volume scheme, Manning friction, wetting and drying, rainfall, and an
77
+ observation-driven coastal stage boundary. Its distinguishing feature is a
78
+ **compressed active-cell mesh**: the cells that matter (land plus a nearshore
79
+ band) are packed into flat arrays before the run, so memory and work scale
80
+ with the flooded landscape rather than its bounding rectangle. The same
81
+ finite-volume kernel runs on the dense grid and on the compressed mesh, bit for
82
+ bit. It runs on NVIDIA and AMD GPUs through [CuPy](https://cupy.dev), scales across
83
+ GPUs with `mpi4py`, and falls back to NumPy on the CPU for prototyping and CI.
84
+
85
+ ## Highlights
86
+
87
+ | | |
88
+ |---|---|
89
+ | **Continental domains on one node** | A 72-hour Hurricane Helene scenario over CONUS at 30 m (8.88 billion active cells, 207-gauge coastal boundary) runs on eight H100 GPUs in 10.8 h; Florida at 10 m (1.78 billion active cells) on four GPUs in 9.5 h. |
90
+ | **Fastest and leanest in a four-code benchmark** | On a 214-million-cell county case against TRITON, SynxFlow, and SERGHEI, GeoSWE's active-cell configuration has the lowest wall time and GPU memory at every GPU count: 3.5 to 3.9 times faster and 1.5 to 1.9 times leaner than the nearest peer, with pairwise CSI near 0.99. |
91
+ | **Compression without a numerical penalty** | Dense and compressed runs take identical step counts and agree to 0.05 cm RMSE (CSI 1.000) on the same cells. |
92
+ | **Near-ideal scaling** | 99.5 % weak-scaling efficiency at 16 H100 GPUs (10.24 billion cells) and 98.8 % at 32 Blackwell MIG slices across two nodes (20.48 billion cells). |
93
+ | **Thin-film accuracy** | On a steady rained slope with a high-accuracy reference solution, GeoSWE stays within 1 % of the reference film depth where other codes depart by tens of percent. |
94
+
95
+ The application runs are computational demonstrations under stated, uncalibrated settings, not validated flood hindcasts.
96
+
97
+ ## How it compares
98
+
99
+ <p align="center"><img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/intro_bench.png" alt="Benchmark: per-step wall time versus peak GPU memory per code, and thin-film depth error on a rained slope" width="900"></p>
100
+
101
+ *(a) Per-step wall time against peak GPU memory per rank for each code's fastest configuration on the 3 m Pinellas County benchmark (1, 2, and 4 GPUs; large marker = 4 GPUs). (b) Depth error against the steady shallow-water reference on a uniformly rained slope.*
102
+
103
+ Comparison codes as benchmarked: TRITON (commit `ec35bc4`), SERGHEI (commit `39a10f2`), and SynxFlow 1.0.2; all four in fp32 at CFL 0.5, first-order well-balanced schemes, same H100 node, identical inputs. Builds, decks, and patches are in the paper's reproducibility appendix.
104
+
105
+ ## How it scales
106
+
107
+ <p align="center"><img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/scaling.png" alt="Strong scaling, weak scaling, and memory per rank on 1 to 16 H100 GPUs" width="900"></p>
108
+
109
+ *Dense and flat storage layouts on an everywhere-wet synthetic domain, 640 million cells per GPU, 1 to 16 H100 GPUs across two nodes. Strong scaling reaches 15.5x on 16 GPUs for the flat layout; weak scaling stays above 99 %.*
110
+
111
+ <p align="center"><img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/scaling_frontier.png" alt="Weak scaling on OLCF Frontier at one billion cells per GPU: 1 to 32 GPUs, and 1 to 128 nodes with 1.024 trillion cells" width="900"></p>
112
+
113
+ *The flat layout on AMD GPUs at OLCF Frontier, one billion cells per GPU (one GCD, half of an MI250X card; eight per node): time per step divided by that of the smallest size, so ideal weak scaling is the dashed line. On 128 nodes, 1.024 trillion cells advance at 162 ms per step: 98.0 % weak-scaling efficiency against one node and 96.7 % against one GPU. A marker is the mean of up to three launches, each timed over one 40-second window; see [`benchmark/frontier_amd`](https://github.com/GeoSWE/geoswe/blob/main/benchmark/frontier_amd/README.md).*
114
+
115
+ ## What is inside
116
+
117
+ - **Numerics:** HLLC and local Lax-Friedrichs fluxes; the Xia et al. (2017) surface-reconstruction method and Audusse hydrostatic reconstruction for exact lake-at-rest balance over arbitrary bathymetry; first-order, MUSCL, and fifth-order reconstruction; forward Euler and SSP-RK3.
118
+ - **Physics:** point-implicit Manning friction, wetting and drying, gridded rainfall, Green-Ampt infiltration, depth sinks, and an inverse-distance-weighted coastal stage ring driven by NOAA CO-OPS gauge records.
119
+ - **Compressed active-cell mesh:** static active set chosen from terrain criteria before the run, `int16` neighbor offsets, a two-cell ghost halo, build-once caching, active-cell-balanced multi-GPU partitions, and checkpoint/restart.
120
+ - **Backends:** CuPy on NVIDIA GPUs (CUDA) and AMD GPUs (ROCm), with a fused single-kernel time step; `mpi4py` for multi-GPU runs with GPU-aware halo exchange; and a NumPy CPU fallback that runs the same scheme in float64.
121
+
122
+ ## Installation
123
+
124
+ ```bash
125
+ # CPU only (NumPy backend): enough for the examples, tests, and docs
126
+ pip install geoswe
127
+
128
+ # GPU on CUDA 12 or CUDA 13 drivers, multi-GPU, GeoTIFF I/O, and forcing readers
129
+ # (installs one CuPy build with its CUDA headers; do not add a second CuPy build)
130
+ pip install "geoswe[gpu,mpi,io,forcings]"
131
+
132
+ # The same on an AMD GPU with ROCm 7
133
+ pip install "geoswe[gpu-rocm,mpi,io,forcings]"
134
+
135
+ # From a source checkout (needed for the examples and the bundled terrain)
136
+ git clone https://github.com/GeoSWE/geoswe.git && cd geoswe
137
+ pip install -e ".[all]"
138
+ ```
139
+
140
+ The development version installs straight from the repository:
141
+ `pip install "geoswe @ git+https://github.com/GeoSWE/geoswe.git"`.
142
+
143
+ See the [installation page](https://geoswe.github.io/geoswe/installation.html) for CUDA and CuPy version notes and the conda environment, and the [AMD GPUs page](https://geoswe.github.io/geoswe/amd_gpus.html) for ROCm.
144
+
145
+ ## Quick start
146
+
147
+ A storm on real terrain, from the repository root. The same script runs on the
148
+ CPU (about half a minute) and on a GPU (two seconds):
149
+
150
+ ```python
151
+ import numpy as np
152
+ from geoswe import Mesh2D, Config, Solver2D, RainfallForcing
153
+
154
+ # terrain and roughness: arrays indexed [x, y], elevations in metres
155
+ case = np.load("examples/data/cookcounty_mini.npz")
156
+ win = np.s_[336:464, 336:464] # a 1.3 km window; use np.s_[:, :] on a GPU
157
+ bed, manning = case["bed"][win], case["manning"][win]
158
+ nx, ny = bed.shape
159
+ mesh = Mesh2D(nx=nx, ny=ny, dx=float(case["dx"]), dy=float(case["dy"]))
160
+
161
+ # 75 mm/h for one hour on a dry bed; water leaves freely at the edges
162
+ rain = RainfallForcing(time_s=[0, 3600], rate_mm_h=[75, 0])
163
+ cfg = Config(friction="manning", bc_x="fall", bc_y="fall", rainfall_forcing=rain)
164
+ solver = Solver2D(mesh, cfg, np.zeros((3, nx, ny)), bed)
165
+ solver.set_manning(manning)
166
+
167
+ solver.run(t_end=2 * 3600)
168
+ peak = solver.max_depth() # NumPy array (nx, ny), metres
169
+ print(f"deepest water {peak.max():.2f} m; {100 * (peak > 0.05).mean():.0f}% of the area reached 5 cm")
170
+ ```
171
+
172
+ `Config()` with no arguments is the production scheme of the paper (first-order
173
+ HLLC, SRM well balancing, forward Euler, CFL 0.5). The
174
+ [flood tutorial](https://geoswe.github.io/geoswe/flood_tutorial.html) explains each line and adds your own
175
+ GeoTIFF terrain, a tide or surge, and the compressed active-cell mesh.
176
+
177
+ [`examples/`](https://github.com/GeoSWE/geoswe/tree/main/examples) has runnable scripts for 1D and 2D dam breaks, a
178
+ lake-at-rest check, rain on a slope, convergence order, multi-GPU scaling, a
179
+ rain-driven flood on real terrain with bundled data, and the compressed mesh on
180
+ that terrain.
181
+
182
+ ## Reproducing the paper
183
+
184
+ [`benchmark/`](https://github.com/GeoSWE/geoswe/tree/main/benchmark) holds the GeoSWE-side pipeline for the paper's
185
+ benchmark cases: the four-code Pinellas County comparison, the steady sheet-flow
186
+ plane, and the synthetic scaling sweep. Each case ships the scripts that
187
+ prepare the inputs, run the simulations, and produce the reported numbers; see
188
+ [`benchmark/README.md`](https://github.com/GeoSWE/geoswe/blob/main/benchmark/README.md) for data requirements and command
189
+ sequences. The Florida and CONUS application pipelines are not part of this
190
+ release; they exercise the same solver paths as the Pinellas case.
191
+
192
+ ## Documentation
193
+
194
+ User guide, configuration reference, the compressed mesh, multi-GPU runs, and
195
+ the API reference: **https://geoswe.github.io/geoswe** (or build locally with
196
+ `pip install ".[docs]" && sphinx-build -b html docs docs/_build`).
197
+
198
+ ## Citing
199
+
200
+ If you use GeoSWE, please cite it; see [CITATION.cff](https://github.com/GeoSWE/geoswe/blob/main/CITATION.cff). A paper
201
+ describing the method, the cross-code benchmark, and the county-to-continent
202
+ applications is in preparation.
203
+
204
+ ## Acknowledgments
205
+
206
+ This material is based upon work supported by the National Science Foundation
207
+ under Grant No. 2325631, by a 2025 IDEaS + Cloud Hub award with support from
208
+ Microsoft at the Georgia Institute of Technology, and by the U.S. Department of
209
+ Energy under Contract No. DE-AC05-00OR22725 through the Genesis Mission project,
210
+ in collaboration with Oak Ridge National Laboratory, the Tennessee Valley
211
+ Authority, and AMD. Computing resources were provided in part by the Partnership
212
+ for an Advanced Computing Environment (PACE) at Georgia Tech and by the Frontier
213
+ supercomputer at the Oak Ridge Leadership Computing Facility.
214
+
215
+ ## License
216
+
217
+ BSD 3-Clause; see [LICENSE](https://github.com/GeoSWE/geoswe/blob/main/LICENSE).
geoswe-1.0.0/README.md ADDED
@@ -0,0 +1,164 @@
1
+ <p align="center">
2
+ <img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/helene_cascade.gif" alt="Animation of Hurricane Helene rainfall and flood depth on three nested GeoSWE domains: CONUS at 30 m, Florida at 10 m, Pinellas County at 3 m" width="720">
3
+ </p>
4
+
5
+ <h1 align="center">GeoSWE</h1>
6
+ <p align="center"><b>Geophysical Shallow-Water Engine</b><br>
7
+ A GPU finite-volume solver for the full 2D shallow-water equations,<br>
8
+ from a 3 m coastal county to the conterminous United States on one multi-GPU node.</p>
9
+
10
+ <p align="center">
11
+ <a href="https://github.com/GeoSWE/geoswe/actions/workflows/test.yml"><img src="https://github.com/GeoSWE/geoswe/actions/workflows/test.yml/badge.svg" alt="tests"></a>
12
+ <a href="https://github.com/GeoSWE/geoswe/actions/workflows/lint.yml"><img src="https://github.com/GeoSWE/geoswe/actions/workflows/lint.yml/badge.svg" alt="lint"></a>
13
+ <a href="https://github.com/GeoSWE/geoswe/actions/workflows/docs.yml"><img src="https://github.com/GeoSWE/geoswe/actions/workflows/docs.yml/badge.svg" alt="docs"></a>
14
+ <a href="https://github.com/GeoSWE/geoswe/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-BSD--3--Clause-blue.svg" alt="BSD 3-Clause"></a>
15
+ <img src="https://img.shields.io/badge/python-3.10%2B-blue.svg" alt="Python 3.10+">
16
+ </p>
17
+
18
+ *Above: 72 hours of Hurricane Helene (September 2024) simulated by the same code on three nested domains: MRMS rainfall in purple, flood depth in color, and the NOAA CO-OPS tide gauges that drive the coastal boundary as stars. [Full-resolution video (MP4)](https://github.com/GeoSWE/geoswe/blob/main/docs/images/helene_cascade.mp4).*
19
+
20
+ ---
21
+
22
+ GeoSWE solves the nonlinear shallow-water equations with a well-balanced HLLC
23
+ finite-volume scheme, Manning friction, wetting and drying, rainfall, and an
24
+ observation-driven coastal stage boundary. Its distinguishing feature is a
25
+ **compressed active-cell mesh**: the cells that matter (land plus a nearshore
26
+ band) are packed into flat arrays before the run, so memory and work scale
27
+ with the flooded landscape rather than its bounding rectangle. The same
28
+ finite-volume kernel runs on the dense grid and on the compressed mesh, bit for
29
+ bit. It runs on NVIDIA and AMD GPUs through [CuPy](https://cupy.dev), scales across
30
+ GPUs with `mpi4py`, and falls back to NumPy on the CPU for prototyping and CI.
31
+
32
+ ## Highlights
33
+
34
+ | | |
35
+ |---|---|
36
+ | **Continental domains on one node** | A 72-hour Hurricane Helene scenario over CONUS at 30 m (8.88 billion active cells, 207-gauge coastal boundary) runs on eight H100 GPUs in 10.8 h; Florida at 10 m (1.78 billion active cells) on four GPUs in 9.5 h. |
37
+ | **Fastest and leanest in a four-code benchmark** | On a 214-million-cell county case against TRITON, SynxFlow, and SERGHEI, GeoSWE's active-cell configuration has the lowest wall time and GPU memory at every GPU count: 3.5 to 3.9 times faster and 1.5 to 1.9 times leaner than the nearest peer, with pairwise CSI near 0.99. |
38
+ | **Compression without a numerical penalty** | Dense and compressed runs take identical step counts and agree to 0.05 cm RMSE (CSI 1.000) on the same cells. |
39
+ | **Near-ideal scaling** | 99.5 % weak-scaling efficiency at 16 H100 GPUs (10.24 billion cells) and 98.8 % at 32 Blackwell MIG slices across two nodes (20.48 billion cells). |
40
+ | **Thin-film accuracy** | On a steady rained slope with a high-accuracy reference solution, GeoSWE stays within 1 % of the reference film depth where other codes depart by tens of percent. |
41
+
42
+ The application runs are computational demonstrations under stated, uncalibrated settings, not validated flood hindcasts.
43
+
44
+ ## How it compares
45
+
46
+ <p align="center"><img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/intro_bench.png" alt="Benchmark: per-step wall time versus peak GPU memory per code, and thin-film depth error on a rained slope" width="900"></p>
47
+
48
+ *(a) Per-step wall time against peak GPU memory per rank for each code's fastest configuration on the 3 m Pinellas County benchmark (1, 2, and 4 GPUs; large marker = 4 GPUs). (b) Depth error against the steady shallow-water reference on a uniformly rained slope.*
49
+
50
+ Comparison codes as benchmarked: TRITON (commit `ec35bc4`), SERGHEI (commit `39a10f2`), and SynxFlow 1.0.2; all four in fp32 at CFL 0.5, first-order well-balanced schemes, same H100 node, identical inputs. Builds, decks, and patches are in the paper's reproducibility appendix.
51
+
52
+ ## How it scales
53
+
54
+ <p align="center"><img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/scaling.png" alt="Strong scaling, weak scaling, and memory per rank on 1 to 16 H100 GPUs" width="900"></p>
55
+
56
+ *Dense and flat storage layouts on an everywhere-wet synthetic domain, 640 million cells per GPU, 1 to 16 H100 GPUs across two nodes. Strong scaling reaches 15.5x on 16 GPUs for the flat layout; weak scaling stays above 99 %.*
57
+
58
+ <p align="center"><img src="https://raw.githubusercontent.com/GeoSWE/geoswe/main/docs/images/scaling_frontier.png" alt="Weak scaling on OLCF Frontier at one billion cells per GPU: 1 to 32 GPUs, and 1 to 128 nodes with 1.024 trillion cells" width="900"></p>
59
+
60
+ *The flat layout on AMD GPUs at OLCF Frontier, one billion cells per GPU (one GCD, half of an MI250X card; eight per node): time per step divided by that of the smallest size, so ideal weak scaling is the dashed line. On 128 nodes, 1.024 trillion cells advance at 162 ms per step: 98.0 % weak-scaling efficiency against one node and 96.7 % against one GPU. A marker is the mean of up to three launches, each timed over one 40-second window; see [`benchmark/frontier_amd`](https://github.com/GeoSWE/geoswe/blob/main/benchmark/frontier_amd/README.md).*
61
+
62
+ ## What is inside
63
+
64
+ - **Numerics:** HLLC and local Lax-Friedrichs fluxes; the Xia et al. (2017) surface-reconstruction method and Audusse hydrostatic reconstruction for exact lake-at-rest balance over arbitrary bathymetry; first-order, MUSCL, and fifth-order reconstruction; forward Euler and SSP-RK3.
65
+ - **Physics:** point-implicit Manning friction, wetting and drying, gridded rainfall, Green-Ampt infiltration, depth sinks, and an inverse-distance-weighted coastal stage ring driven by NOAA CO-OPS gauge records.
66
+ - **Compressed active-cell mesh:** static active set chosen from terrain criteria before the run, `int16` neighbor offsets, a two-cell ghost halo, build-once caching, active-cell-balanced multi-GPU partitions, and checkpoint/restart.
67
+ - **Backends:** CuPy on NVIDIA GPUs (CUDA) and AMD GPUs (ROCm), with a fused single-kernel time step; `mpi4py` for multi-GPU runs with GPU-aware halo exchange; and a NumPy CPU fallback that runs the same scheme in float64.
68
+
69
+ ## Installation
70
+
71
+ ```bash
72
+ # CPU only (NumPy backend): enough for the examples, tests, and docs
73
+ pip install geoswe
74
+
75
+ # GPU on CUDA 12 or CUDA 13 drivers, multi-GPU, GeoTIFF I/O, and forcing readers
76
+ # (installs one CuPy build with its CUDA headers; do not add a second CuPy build)
77
+ pip install "geoswe[gpu,mpi,io,forcings]"
78
+
79
+ # The same on an AMD GPU with ROCm 7
80
+ pip install "geoswe[gpu-rocm,mpi,io,forcings]"
81
+
82
+ # From a source checkout (needed for the examples and the bundled terrain)
83
+ git clone https://github.com/GeoSWE/geoswe.git && cd geoswe
84
+ pip install -e ".[all]"
85
+ ```
86
+
87
+ The development version installs straight from the repository:
88
+ `pip install "geoswe @ git+https://github.com/GeoSWE/geoswe.git"`.
89
+
90
+ See the [installation page](https://geoswe.github.io/geoswe/installation.html) for CUDA and CuPy version notes and the conda environment, and the [AMD GPUs page](https://geoswe.github.io/geoswe/amd_gpus.html) for ROCm.
91
+
92
+ ## Quick start
93
+
94
+ A storm on real terrain, from the repository root. The same script runs on the
95
+ CPU (about half a minute) and on a GPU (two seconds):
96
+
97
+ ```python
98
+ import numpy as np
99
+ from geoswe import Mesh2D, Config, Solver2D, RainfallForcing
100
+
101
+ # terrain and roughness: arrays indexed [x, y], elevations in metres
102
+ case = np.load("examples/data/cookcounty_mini.npz")
103
+ win = np.s_[336:464, 336:464] # a 1.3 km window; use np.s_[:, :] on a GPU
104
+ bed, manning = case["bed"][win], case["manning"][win]
105
+ nx, ny = bed.shape
106
+ mesh = Mesh2D(nx=nx, ny=ny, dx=float(case["dx"]), dy=float(case["dy"]))
107
+
108
+ # 75 mm/h for one hour on a dry bed; water leaves freely at the edges
109
+ rain = RainfallForcing(time_s=[0, 3600], rate_mm_h=[75, 0])
110
+ cfg = Config(friction="manning", bc_x="fall", bc_y="fall", rainfall_forcing=rain)
111
+ solver = Solver2D(mesh, cfg, np.zeros((3, nx, ny)), bed)
112
+ solver.set_manning(manning)
113
+
114
+ solver.run(t_end=2 * 3600)
115
+ peak = solver.max_depth() # NumPy array (nx, ny), metres
116
+ print(f"deepest water {peak.max():.2f} m; {100 * (peak > 0.05).mean():.0f}% of the area reached 5 cm")
117
+ ```
118
+
119
+ `Config()` with no arguments is the production scheme of the paper (first-order
120
+ HLLC, SRM well balancing, forward Euler, CFL 0.5). The
121
+ [flood tutorial](https://geoswe.github.io/geoswe/flood_tutorial.html) explains each line and adds your own
122
+ GeoTIFF terrain, a tide or surge, and the compressed active-cell mesh.
123
+
124
+ [`examples/`](https://github.com/GeoSWE/geoswe/tree/main/examples) has runnable scripts for 1D and 2D dam breaks, a
125
+ lake-at-rest check, rain on a slope, convergence order, multi-GPU scaling, a
126
+ rain-driven flood on real terrain with bundled data, and the compressed mesh on
127
+ that terrain.
128
+
129
+ ## Reproducing the paper
130
+
131
+ [`benchmark/`](https://github.com/GeoSWE/geoswe/tree/main/benchmark) holds the GeoSWE-side pipeline for the paper's
132
+ benchmark cases: the four-code Pinellas County comparison, the steady sheet-flow
133
+ plane, and the synthetic scaling sweep. Each case ships the scripts that
134
+ prepare the inputs, run the simulations, and produce the reported numbers; see
135
+ [`benchmark/README.md`](https://github.com/GeoSWE/geoswe/blob/main/benchmark/README.md) for data requirements and command
136
+ sequences. The Florida and CONUS application pipelines are not part of this
137
+ release; they exercise the same solver paths as the Pinellas case.
138
+
139
+ ## Documentation
140
+
141
+ User guide, configuration reference, the compressed mesh, multi-GPU runs, and
142
+ the API reference: **https://geoswe.github.io/geoswe** (or build locally with
143
+ `pip install ".[docs]" && sphinx-build -b html docs docs/_build`).
144
+
145
+ ## Citing
146
+
147
+ If you use GeoSWE, please cite it; see [CITATION.cff](https://github.com/GeoSWE/geoswe/blob/main/CITATION.cff). A paper
148
+ describing the method, the cross-code benchmark, and the county-to-continent
149
+ applications is in preparation.
150
+
151
+ ## Acknowledgments
152
+
153
+ This material is based upon work supported by the National Science Foundation
154
+ under Grant No. 2325631, by a 2025 IDEaS + Cloud Hub award with support from
155
+ Microsoft at the Georgia Institute of Technology, and by the U.S. Department of
156
+ Energy under Contract No. DE-AC05-00OR22725 through the Genesis Mission project,
157
+ in collaboration with Oak Ridge National Laboratory, the Tennessee Valley
158
+ Authority, and AMD. Computing resources were provided in part by the Partnership
159
+ for an Advanced Computing Environment (PACE) at Georgia Tech and by the Frontier
160
+ supercomputer at the Oak Ridge Leadership Computing Facility.
161
+
162
+ ## License
163
+
164
+ BSD 3-Clause; see [LICENSE](https://github.com/GeoSWE/geoswe/blob/main/LICENSE).
@@ -0,0 +1,165 @@
1
+ # AMD GPUs (ROCm)
2
+
3
+ GeoSWE runs on AMD GPUs through CuPy's ROCm build. The API, the kernels and the
4
+ `SWE_*` settings are the same as on NVIDIA; `geoswe.gpu_platform()` returns
5
+ `"hip"` instead of `"cuda"`.
6
+
7
+ ```{admonition} What has been tested
8
+ :class: note
9
+
10
+ AMD Instinct MI210 and MI250X (`gfx90a`) on OLCF Frontier, with ROCm 7.0.2 and
11
+ 7.2.0 and two CuPy builds: `cupy-rocm-7-0` 14.2.0 from PyPI and AMD's `amd-cupy`
12
+ 13.5.1. Both pass the whole GPU test suite. Other AMD GPUs and ROCm versions are
13
+ untested. CuPy itself still labels its ROCm support experimental.
14
+ ```
15
+
16
+ ## Installing
17
+
18
+ CuPy's ROCm build compiles kernels with the ROCm installation on the machine, so
19
+ ROCm 7 must be installed and visible at run time:
20
+
21
+ - `hipcc` on `PATH` (CuPy runs it to find the include directories), and
22
+ - `ROCM_HOME` pointing at the ROCm tree, for example `/opt/rocm-7.2.0`.
23
+
24
+ ```bash
25
+ pip install "geoswe[gpu-rocm]" # CuPy for ROCm 7 from PyPI
26
+ pip install "geoswe[gpu-rocm,mpi,io,forcings]" # with MPI, GeoTIFF I/O and forcings
27
+ ```
28
+
29
+ AMD also publishes its own build, which OLCF documents for Frontier. Install it
30
+ instead of the extra, never next to it:
31
+
32
+ ```bash
33
+ pip install amd-cupy --extra-index-url https://pypi.amd.com/rocm-7.2.0/simple
34
+ pip install "geoswe[mpi,io,forcings]"
35
+ ```
36
+
37
+ Check the install:
38
+
39
+ ```bash
40
+ python -c "import geoswe; print(geoswe.get_backend(), geoswe.gpu_platform())" # cupy hip
41
+ pytest -m gpu -q
42
+ ```
43
+
44
+ The first run compiles every kernel, which takes a minute or so; later runs load
45
+ them from CuPy's cache. On a cluster, put that cache on a filesystem the compute
46
+ nodes share, and keep AMD's own compiler cache off NFS inside jobs:
47
+
48
+ ```bash
49
+ export CUPY_CACHE_DIR=/path/on/scratch/cupy_cache # default: ~/.cupy/kernel_cache
50
+ export AMD_COMGR_CACHE_DIR=/tmp/$USER-comgr # default: ~/.cache/comgr
51
+ ```
52
+
53
+ ## What differs from NVIDIA
54
+
55
+ The kernels are CUDA C. NVIDIA's compiler (NVRTC) and AMD's (clang, through HIP)
56
+ disagree about three things in them, and `geoswe.backend` handles each one. On
57
+ NVIDIA nothing changes: kernel sources and options reach CuPy exactly as before.
58
+
59
+ | | NVIDIA | AMD | What GeoSWE does on AMD |
60
+ |---|---|---|---|
61
+ | Register cap `-maxrregcount` (`SWE_DENSE_MAXRREG`, `SWE_FLAT_MAXRREG`) | applied on H100 | not an option of the compiler | never requested; ignored with a warning if set by hand |
62
+ | `__fmul_rn` / `__fadd_rn` | never fused into a multiply-add | plain `*` and `+`, which clang fuses | redefined with contraction switched off |
63
+ | Warp-level compaction | 32 lanes, 32-bit masks | 64 lanes, 64-bit masks | a kernel without warp intrinsics |
64
+
65
+ The first one also needed a fix to device detection: CuPy reports the compute
66
+ capability of a `gfx90a` card as `"90"`, the same string as an H100.
67
+
68
+ ### Floating-point contraction
69
+
70
+ `GEOSWE_HIP_FP_CONTRACT` sets how the ROCm compiler may fuse `a*b + c` into one
71
+ multiply-add:
72
+
73
+ | Value | Meaning |
74
+ |---|---|
75
+ | `off` (default) | never; every operation rounds once, in source order |
76
+ | `on` | only inside a single source expression |
77
+ | `fast` | at the optimizer's discretion, across statements (clang's own default for HIP) |
78
+
79
+ GeoSWE relies on several code paths giving identical bits: the fused and the
80
+ split time step, the dense and the compressed mesh, a run and its restart. With
81
+ `fast` the same expression can be fused in one kernel and not in another, and the
82
+ fused and split steps with sub-grid storage then differ in the last bit
83
+ (`tests/test_gpu_storage_fused.py` fails). With `off` or `on` the arithmetic is
84
+ fixed by the source text.
85
+
86
+ `off` is the default because it makes those identities hold by construction: two
87
+ kernels agree whenever their statements do. It is the only mode in which the
88
+ dense and the compressed solver agree bit for bit on the bowl-and-dam problem of
89
+ `tests/test_gpu_compressed_equiv.py`; `on` and `fast` leave one unit in the last
90
+ place between them. `on` is about 3 % faster (25.0 against 25.8 ms/step on an
91
+ MI250X at 147 M cells; `fast` is in between) and keeps the fused and split steps
92
+ identical in the test suite. All three modes keep a lake at rest to round-off.
93
+
94
+ Results on AMD and NVIDIA are not bit-identical to each other: the compilers and
95
+ their math libraries differ, at round-off level.
96
+
97
+ ## Multiple GPUs
98
+
99
+ One MPI rank drives one GPU, as on NVIDIA. Under Slurm, let the scheduler bind
100
+ the devices:
101
+
102
+ ```bash
103
+ srun -n 8 --gpus-per-task=1 --gpu-bind=closest python my_run.py
104
+ ```
105
+
106
+ The halo travels through host memory unless `SWE_HALO_CUDA_AWARE=1` asks for
107
+ GPU-aware MPI. Despite the name, that setting is not specific to CUDA: CuPy's
108
+ ROCm build exposes device arrays through the same interface, and `mpi4py` hands
109
+ the pointers to MPI. With HPE Cray MPICH this needs two more things:
110
+
111
+ - `mpi4py` built with the `craype-accel-amd-*` module loaded, so that it links the
112
+ GPU transport library, and
113
+ - `MPICH_GPU_SUPPORT_ENABLED=1` in the job.
114
+
115
+ If `SWE_HALO_CUDA_AWARE=1` is set while Cray MPICH runs without its GPU support,
116
+ GeoSWE warns and stages through the host instead of passing device pointers to a
117
+ library that would treat them as host memory.
118
+
119
+ ## OLCF Frontier
120
+
121
+ `benchmark/frontier_amd/` in the repository has the module set, the build steps
122
+ for a GPU-aware `mpi4py`, and a one-node job that runs the GPU test suite, a
123
+ partition-invariance check and a weak-scaling sweep. Measured there on one node
124
+ (an MI250X card is two devices, so eight per node; 147 M cells per device,
125
+ float32, host-staged halo):
126
+
127
+ | Tier | 1 device | 8 devices | Weak efficiency |
128
+ |---|---|---|---|
129
+ | Dense | 25.8 ms/step | 26.1 ms/step | 99.0 % |
130
+ | Compressed | 23.7 ms/step | 23.9 ms/step | 99.2 % |
131
+
132
+ Repeat launches differ by up to 2 %. The global solution is bit-identical on 1,
133
+ 2, 4 and 8 devices, with the host-staged and the GPU-aware halo, with and without
134
+ the halo/compute overlap.
135
+
136
+ On two nodes, the scaling harness of `benchmark/scaling_640m` (640 M cells per
137
+ device, in the configuration its launchers pin) gives:
138
+
139
+ | Layout | Weak scaling, 16 devices, 10.24 B cells | Strong scaling, 16 devices |
140
+ |---|---|---|
141
+ | Flat | 118.7 ms/step, 99.2 % efficiency | 15.0x |
142
+ | Dense | 190.2 ms/step, 99.6 % efficiency | 15.4x |
143
+
144
+ With the solver's defaults the same 10.24 billion cells take 102 ms/step (flat)
145
+ and 112 ms/step (dense). The solution on sixteen devices across the two nodes has
146
+ the digest of the one-device run.
147
+
148
+ ```{note}
149
+ A device here is one GCD, half an MI250X card. For the same 640 M cells with the
150
+ solver's defaults, the H100 of the scaling figure in the
151
+ [repository README](https://github.com/GeoSWE/geoswe#how-it-scales) takes about
152
+ 42 ms/step (flat) and 46 ms/step (dense), where one GCD takes 101 and 111. That
153
+ is 2.4 times faster, the ratio of their FP32 lanes (16,896 CUDA cores to 7,040
154
+ stream processors), so a whole MI250X card delivers about 0.83 of an H100. The
155
+ machines differ in more than the GPU: read this as arithmetic on published
156
+ numbers, not as a controlled comparison.
157
+ ```
158
+
159
+ ## Not done yet
160
+
161
+ - Of the paper's benchmark cases, only the synthetic scaling harness has been run
162
+ on AMD hardware.
163
+ - The register cap that speeds up the residual kernel on H100 has no ROCm
164
+ counterpart, and no AMD-specific tuning of block sizes has been tried.
165
+ - Two nodes (16 devices) are the most that has been used.
@@ -0,0 +1,13 @@
1
+ Compressed active-cell solver
2
+ =============================
3
+
4
+ GPU-only. Imported lazily as ``geoswe.CompressedSolver``. See
5
+ :doc:`../compressed_mesh` for the concepts.
6
+
7
+ .. currentmodule:: geoswe.compressed_solver
8
+
9
+ .. autoclass:: CompressedSolver
10
+ :members:
11
+ :member-order: bysource
12
+
13
+ .. autofunction:: run_cached
@@ -0,0 +1,21 @@
1
+ Forcings and I/O
2
+ ================
3
+
4
+ .. currentmodule:: geoswe.forcing
5
+
6
+ .. autoclass:: RainfallForcing
7
+ :members:
8
+
9
+ .. autoclass:: StageBoundary
10
+ :members:
11
+
12
+ GeoTIFF I/O
13
+ -----------
14
+
15
+ .. currentmodule:: geoswe.io_geotiff
16
+
17
+ .. autoclass:: GeoArray
18
+ :members:
19
+
20
+ .. autofunction:: read_geotiff
21
+ .. autofunction:: write_geotiff