peclet-coupling 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (22) hide show
  1. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/PKG-INFO +10 -4
  2. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/README.md +9 -3
  3. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/cmake/PecletDeps.cmake +2 -2
  4. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/pyproject.toml +1 -1
  5. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/python/peclet_coupling/__init__.py +3 -1
  6. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/python/peclet_coupling/driver.py +164 -26
  7. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/src/coupling_bindings.cpp +44 -18
  8. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/src/coupling_kernels.hpp +172 -18
  9. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/src/drag.hpp +36 -1
  10. peclet_coupling-0.3.0/tests/smoothing_ref_eps.npy +0 -0
  11. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/tests/test_fixed_bed_ergun.py +1 -1
  12. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/tests/test_mpi_fixed_bed_ergun.py +1 -1
  13. peclet_coupling-0.3.0/tests/test_mpi_smoothing.py +114 -0
  14. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/tests/test_terminal_velocity.py +2 -1
  15. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/.github/workflows/release.yml +0 -0
  16. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/.gitignore +0 -0
  17. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/CMakeLists.txt +0 -0
  18. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/LICENSE +0 -0
  19. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/examples/fluidized_bed.py +0 -0
  20. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/tests/CMakeLists.txt +0 -0
  21. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/tests/test_fixed_bed_ergun_porous.py +0 -0
  22. {peclet_coupling-0.2.0 → peclet_coupling-0.3.0}/tests/test_mpi_moving_suspension.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: peclet-coupling
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: peclet.coupling — unresolved point-particle CFD-DEM coupling of peclet.flow + peclet.dem
5
5
  Author-Email: Frank Peters <e.a.j.f.peters@gmail.com>
6
6
  License-Expression: MIT
@@ -32,9 +32,15 @@ architecture (Python is the composition layer).
32
32
  Per fluid step (`CfdDem.step()`):
33
33
  1. **Void fraction** — scatter each particle's volume onto the grid (trilinear, **wall-aware**: near
34
34
  an immersed solid the weights re-normalise over the fluid corners so no hold-up leaks into walls),
35
- periodic-fold the ghost deposits, and `ε = clamp(1 − Vsolid/Vcell, eps_min, 1)`. The floor
36
- `eps_min` defaults to 0.4 ≈ the random-close-packing voidage (the drag correlations are invalid,
37
- and Ergun's `1/ε` powers explosive, below a physical packing).
35
+ fold the ghost deposits (periodic wrap on periodic axes; **same-side fold onto the boundary cell
36
+ at a non-periodic domain face** — a grain resting on the distributor scatters part of its volume
37
+ below z=0, and that hold-up belongs to the bottom cell, not to a ghost the fluid never owns), and
38
+ `ε = clamp(1 − Vsolid/Vcell, eps_min, 1)`. The floor `eps_min` defaults to 0.4 ≈ the
39
+ random-close-packing voidage (the drag correlations are invalid, and Ergun's `1/ε` powers
40
+ explosive, below a physical packing). A particle whose trilinear stencil falls **outside the
41
+ domain by more than one ghost layer** (e.g. pushed through a DEM wall by a violent contact solve)
42
+ is dropped from the exchange entirely — no deposit, zero drag — so a runaway escapee can never
43
+ feed a diverging `β·u_p` source into the boundary row.
38
44
  2. **Drag + feedback** — gather the fluid velocity and ε at each particle, evaluate the drag law
39
45
  (Stokes / Schiller–Naumann / Ergun / Di Felice / Wen & Yu / Gidaspow), write the drag force to the
40
46
  particles and deposit the reaction onto the fluid momentum source.
@@ -16,9 +16,15 @@ architecture (Python is the composition layer).
16
16
  Per fluid step (`CfdDem.step()`):
17
17
  1. **Void fraction** — scatter each particle's volume onto the grid (trilinear, **wall-aware**: near
18
18
  an immersed solid the weights re-normalise over the fluid corners so no hold-up leaks into walls),
19
- periodic-fold the ghost deposits, and `ε = clamp(1 − Vsolid/Vcell, eps_min, 1)`. The floor
20
- `eps_min` defaults to 0.4 ≈ the random-close-packing voidage (the drag correlations are invalid,
21
- and Ergun's `1/ε` powers explosive, below a physical packing).
19
+ fold the ghost deposits (periodic wrap on periodic axes; **same-side fold onto the boundary cell
20
+ at a non-periodic domain face** — a grain resting on the distributor scatters part of its volume
21
+ below z=0, and that hold-up belongs to the bottom cell, not to a ghost the fluid never owns), and
22
+ `ε = clamp(1 − Vsolid/Vcell, eps_min, 1)`. The floor `eps_min` defaults to 0.4 ≈ the
23
+ random-close-packing voidage (the drag correlations are invalid, and Ergun's `1/ε` powers
24
+ explosive, below a physical packing). A particle whose trilinear stencil falls **outside the
25
+ domain by more than one ghost layer** (e.g. pushed through a DEM wall by a violent contact solve)
26
+ is dropped from the exchange entirely — no deposit, zero drag — so a runaway escapee can never
27
+ feed a diverging `β·u_p` source into the boundary row.
22
28
  2. **Drag + feedback** — gather the fluid velocity and ε at each particle, evaluate the drag law
23
29
  (Stokes / Schiller–Naumann / Ergun / Di Felice / Wen & Yu / Gidaspow), write the drag force to the
24
30
  particles and deposit the reaction onto the fluid momentum source.
@@ -16,8 +16,8 @@ include(FetchContent)
16
16
 
17
17
  set(PECLET_KOKKOS_TAG "5.1.1" CACHE STRING "Vendored Kokkos git tag")
18
18
  set(PECLET_ARBORX_TAG "v2.1" CACHE STRING "Vendored ArborX git tag")
19
- set(PECLET_TPX_TAG "v0.4.0" CACHE STRING "Vendored core git tag (headers)")
20
- set(PECLET_MORTON_TAG "v0.2.0" CACHE STRING "Vendored morton git tag (headers)")
19
+ set(PECLET_TPX_TAG "v0.5.0" CACHE STRING "Vendored core git tag (headers)")
20
+ set(PECLET_MORTON_TAG "v0.2.1" CACHE STRING "Vendored morton git tag (headers)")
21
21
  option(PECLET_VENDOR_DEPS "Force FetchContent-build of Kokkos/ArborX/siblings (self-contained wheel)" OFF)
22
22
 
23
23
  # nanobind — found via the active interpreter (scikit-build-core supplies it as a build requirement),
@@ -16,7 +16,7 @@ build-backend = "scikit_build_core.build"
16
16
 
17
17
  [project]
18
18
  name = "peclet-coupling"
19
- version = "0.2.0"
19
+ version = "0.3.0"
20
20
  description = "peclet.coupling — unresolved point-particle CFD-DEM coupling of peclet.flow + peclet.dem"
21
21
  readme = "README.md"
22
22
  requires-python = ">=3.10"
@@ -14,8 +14,10 @@ DRAG_ERGUN = 2
14
14
  DRAG_DI_FELICE = 3
15
15
  DRAG_WEN_YU = 4
16
16
  DRAG_GIDASPOW = 5 # Ergun (dense) + Wen & Yu (dilute), switched at eps = 0.8
17
+ DRAG_BEETSTRA = 6 # Beetstra-van der Hoef-Kuipers (2007) DNS drag — the published "BVK2"
18
+ DRAG_TANG = 7 # Tang et al. (2015) DNS drag — what MFIX-Exa's "BVK2" option actually executes
17
19
 
18
20
  __version__ = "0.2.0"
19
21
 
20
22
  __all__ = ["CfdDem", "_coupling", "DRAG_STOKES", "DRAG_SCHILLER_NAUMANN", "DRAG_ERGUN",
21
- "DRAG_DI_FELICE", "DRAG_WEN_YU", "DRAG_GIDASPOW"]
23
+ "DRAG_DI_FELICE", "DRAG_WEN_YU", "DRAG_GIDASPOW", "DRAG_BEETSTRA", "DRAG_TANG"]
@@ -25,10 +25,11 @@ def _sl(axis, idx):
25
25
 
26
26
  class CfdDem:
27
27
  def __init__(self, flow, dem, *, fluid_dt, mu, rho, radius, drag="schiller_naumann",
28
- dem_substeps=20, eps_min=0.4, periodic=(True, True, True), h=1.0,
29
- move_particles=True, implicit_drag=True, porous=False, advection=True):
28
+ dem_substeps=20, eps_min=0.25, smooth_width=0.0, periodic=(True, True, True), h=1.0,
29
+ move_particles=True, implicit_drag=True, porous=True, advection=True,
30
+ gravity=(0.0, 0.0, 0.0)):
30
31
  from . import (_coupling, DRAG_STOKES, DRAG_SCHILLER_NAUMANN, DRAG_ERGUN, DRAG_DI_FELICE,
31
- DRAG_WEN_YU, DRAG_GIDASPOW)
32
+ DRAG_WEN_YU, DRAG_GIDASPOW, DRAG_BEETSTRA, DRAG_TANG)
32
33
  self._c = _coupling
33
34
  self.flow = flow
34
35
  self.dem = dem
@@ -48,22 +49,53 @@ class CfdDem:
48
49
  self.fluid_dt = float(fluid_dt)
49
50
  self.dem_substeps = int(dem_substeps)
50
51
  self.dt_dem = self.fluid_dt / self.dem_substeps
51
- # Void-fraction floor. Default 0.4 ~ the random-close-packing voidage: a physical packing
52
- # cannot go below ~0.36, so deposits under the floor are local over-concentration of the
53
- # trilinear/wall-aware scatter, and the Ergun-branch drag (beta ~ (1-eps)^2/eps...) explodes
54
- # outside its validity range (a porous Model-B bed at eps_min=0.3 diverges; 0.4 is stable).
52
+ # Void-fraction floor (default 0.25): a PHYSICAL regularisation, not just a guard. Real
53
+ # voidage bottoms out near random close packing (~0.36 monodisperse, ~0.25 for wide bidisperse
54
+ # mixes); anything lower can only come from numerically interpenetrated particles or deposit
55
+ # artifacts, and must not reach the volume-averaged fluid — the eps-conservative projection
56
+ # legitimately amplifies the interstitial velocity by 1/eps, so junk eps -> junk gas. 0.25
57
+ # keeps the Ergun/drag fidelity over the physical range (the old 0.4 clamp under-predicted
58
+ # dense-bed drag ~3x; the interim 0.05 guard let interpenetration artifacts detonate a bed).
55
59
  self.eps_min = float(eps_min)
60
+ # Porosity smoothing length (grid cells), decoupled from the CFD cell size — MFIX's
61
+ # DES_DIFFUSE_WIDTH. 0 = off (plain trilinear deposit). For a coarse cell/dp bed set it ~1 cell
62
+ # (a few particle diameters) so the void fraction the drag sees is smooth and grid-independent;
63
+ # converted to `nsweeps` explicit diffusion sweeps (sigma = sqrt(2*alpha*nsweeps), alpha=1/6).
64
+ self.smooth_width = float(smooth_width)
65
+ # Volume-averaging validity: eps must be smooth over >~ the particle scale. With cells not
66
+ # much larger than d_p the raw trilinear deposit is not a proper volume filter — smooth it.
67
+ if np.isscalar(radius) and float(h) < 3.0 * (2.0 * float(radius)) \
68
+ and self.smooth_width * float(h) < 1.5 * (2.0 * float(radius)):
69
+ import warnings
70
+ warnings.warn(
71
+ f"CfdDem: cell size h={float(h):g} is < 3 particle diameters and smooth_width is "
72
+ f"below ~1.5 d_p — the deposited void fraction is not a proper volume average at "
73
+ f"this resolution. Set smooth_width so the smoothing length exceeds the particle "
74
+ f"diameter (e.g. smooth_width={1.5 * 2.0 * float(radius) / float(h):.1f}).")
75
+ self._smooth_alpha = 1.0 / 6.0
76
+ self._smooth_sweeps = (max(1, int(round(self.smooth_width ** 2 / (2.0 * self._smooth_alpha))))
77
+ if self.smooth_width > 0.0 else 0)
56
78
  self.move_particles = bool(move_particles) # False: fixed bed — skip DEM dynamics entirely
57
79
  self.implicit_drag = bool(implicit_drag) # beta on the fluid diagonal (stable for stiff beds)
58
- # Volume-averaged continuity d(eps)/dt+div(eps u)=0 (proper unresolved CFD-DEM). When off the
59
- # fluid is solved incompressible (eps only in the drag) — cheaper, fine for dilute/steady beds.
80
+ # Volume-averaged continuity d(eps)/dt+div(eps u)=0 (proper unresolved CFD-DEM) — the DEFAULT:
81
+ # porosity must enter the volume-averaged Navier-Stokes equations, not just the drag closure.
82
+ # (MFIX-Exa likewise advects with the superficial velocity and projects div(eps u)=0; its
83
+ # d(eps)/dt constraint term is optional/"under development" — ours keeps it.) porous=False
84
+ # solves plain incompressible NS (eps only in the drag): a cheaper approximation for
85
+ # dilute/steady beds, not a faithful CFD-DEM.
60
86
  self.porous = bool(porous)
61
87
  self.h = float(h)
62
88
  self.inv_vcell = 1.0 / (self.h ** 3)
89
+ # Constant external acceleration dem applies per substep (its set_gravity vector; dem has
90
+ # no getter, so pass it here too). Feeds the stiff-safe drag cap's gravity-exact correction
91
+ # F -= m g (1 - beta_eff/beta), which restores the physical steady-state slip m g / beta.
92
+ self.gravity = tuple(float(c) for c in gravity)
63
93
  self.periodic = tuple(bool(p) for p in periodic)
64
94
  self.drag_kind = {"stokes": DRAG_STOKES, "schiller_naumann": DRAG_SCHILLER_NAUMANN,
65
95
  "ergun": DRAG_ERGUN, "di_felice": DRAG_DI_FELICE,
66
- "wen_yu": DRAG_WEN_YU, "gidaspow": DRAG_GIDASPOW}[drag]
96
+ "wen_yu": DRAG_WEN_YU, "gidaspow": DRAG_GIDASPOW,
97
+ "beetstra": DRAG_BEETSTRA, "bvk": DRAG_BEETSTRA,
98
+ "tang": DRAG_TANG, "bvk2": DRAG_TANG}[drag]
67
99
 
68
100
  nx, ny, nz = flow.get_resolution() # LOCAL block dims under MPI
69
101
  self.g = flow.ghost_width()
@@ -87,6 +119,22 @@ class CfdDem:
87
119
  self._ox, self._oy, self._oz = bo[0] * self.h, bo[1] * self.h, bo[2] * self.h
88
120
  gnx, gny, gnz = flow.global_resolution() if self.mpi else (nx, ny, nz)
89
121
  self.gnx, self.gny, self.gnz = gnx, gny, gnz
122
+ # Smoothing under MPI: a local block face that is an INTERIOR rank boundary is not a wall —
123
+ # the diffusion sweep must read the halo ghost there (bit set), while faces on the GLOBAL
124
+ # domain boundary keep the validated single-rank zero-flux closed-box behaviour (bit clear,
125
+ # also on periodic axes, matching the single-rank path byte-for-byte). The sweep loop in
126
+ # update_void_fraction halo-refreshes solidvol before every sweep, so multi-rank smoothing
127
+ # reproduces the single-rank arithmetic exactly.
128
+ self._smooth_open = 0
129
+ if self.mpi:
130
+ lo = (bo[0], bo[1], bo[2])
131
+ dims = (nx, ny, nz)
132
+ gdims = (gnx, gny, gnz)
133
+ for a in range(3):
134
+ if lo[a] > 0:
135
+ self._smooth_open |= 1 << (2 * a)
136
+ if lo[a] + dims[a] < gdims[a]:
137
+ self._smooth_open |= 1 << (2 * a + 1)
90
138
  # The CURRENT shared decomposition, as an x-fastest per-cell weight field. Uniform => the
91
139
  # default equal-cell ORB flow's init_mpi built; rebalance() overwrites it. dem is migrated onto
92
140
  # this each moving step so its ownership tracks flow's grid partition (the deposit stays
@@ -106,6 +154,15 @@ class CfdDem:
106
154
  self._solidvol = xp.zeros((self.ex, self.ey, self.ez), dtype=xp.float64, order="F")
107
155
  self._eps = xp.ones((self.ex, self.ey, self.ez), dtype=xp.float64, order="F")
108
156
  if self.porous:
157
+ # The porous projection lives on the cut-cell pressure operator (eps-weighted
158
+ # coefficients ride the openness rails). A domain-BC-only box (no set_solid /
159
+ # set_pressure_geometry) has no such operator, and flow would silently solve plain
160
+ # div(u)=0 — the gas never accelerates to the interstitial velocity in the bed and the
161
+ # drag is far too weak to fluidize. Auto-install an all-fluid geometry in that case
162
+ # (flow's project() now also throws rather than silently degrade).
163
+ if hasattr(flow, "has_cutcell_pressure") and not flow.has_cutcell_pressure():
164
+ allfluid = np.full((self.nx, self.ny, self.nz), 1e6, dtype=np.float64)
165
+ flow.set_pressure_geometry(allfluid.flatten(order="F"))
109
166
  flow.set_porous_continuity(True) # projection enforces d(eps)/dt + div(eps u) = 0
110
167
  # Gas convection ON by default: fully-implicit FOU operator + explicit deferred-correction
111
168
  # TVD (unconditionally stable at the large coupled dt on both the periodic and domain-BC
@@ -163,24 +220,68 @@ class CfdDem:
163
220
  vel = np.ascontiguousarray(self.dem.get_velocities(), dtype=np.float32)
164
221
  return pos, vel
165
222
 
166
- # --- periodic ghost handling on a padded (ex,ey,ez) buffer -------------------------------
167
- def _fold(self, f): # deposits that landed one cell into the ghost wrap back to the inner edge
223
+ # --- ghost handling on a padded (ex,ey,ez) buffer ----------------------------------------
224
+ def _fold(self, f):
225
+ """Fold ghost-layer deposits back onto owned cells. Periodic axis: wrap to the opposite
226
+ inner edge. Non-periodic axis: fold into the SAME-side boundary cell — a grain resting on
227
+ the distributor scatters part of its volume (and drag beta / feedback) one layer below
228
+ z=0; dropping it would lose hold-up exactly where the bed is densest and leave the
229
+ deposit's leakage in ghosts the fluid reads."""
168
230
  g = self.g
169
231
  for a, (n, per) in enumerate(zip((self.nx, self.ny, self.nz), self.periodic)):
170
- if not per:
171
- continue
172
- f[_sl(a, n + g - 1)] += f[_sl(a, g - 1)]
173
- f[_sl(a, g)] += f[_sl(a, n + g)]
232
+ if per:
233
+ f[_sl(a, n + g - 1)] += f[_sl(a, g - 1)]
234
+ f[_sl(a, g)] += f[_sl(a, n + g)]
235
+ else:
236
+ f[_sl(a, g)] += f[_sl(a, g - 1)]
237
+ f[_sl(a, n + g - 1)] += f[_sl(a, n + g)]
174
238
  f[_sl(a, g - 1)] = 0.0
175
239
  f[_sl(a, n + g)] = 0.0
176
240
 
177
- def _fill(self, f): # fill the one ghost layer the gather stencil reads (periodic wrap)
241
+ def _fold_domain(self, f):
242
+ """MPI: same-side fold of the non-periodic GLOBAL-domain-boundary ghosts of the local
243
+ block (the reverse halo never touches them). Call BEFORE exchange_field_add."""
244
+ g = self.g
245
+ lo = (int(self._ox / self.h), int(self._oy / self.h), int(self._oz / self.h))
246
+ dims = (self.nx, self.ny, self.nz)
247
+ gdims = (self.gnx, self.gny, self.gnz)
248
+ for a in range(3):
249
+ if self.periodic[a]:
250
+ continue
251
+ if lo[a] == 0:
252
+ f[_sl(a, g)] += f[_sl(a, g - 1)]
253
+ f[_sl(a, g - 1)] = 0.0
254
+ if lo[a] + dims[a] == gdims[a]:
255
+ f[_sl(a, dims[a] + g - 1)] += f[_sl(a, dims[a] + g)]
256
+ f[_sl(a, dims[a] + g)] = 0.0
257
+
258
+ def _fill(self, f):
259
+ """Fill the one ghost layer the gather stencil reads: periodic wrap, or zero-gradient at a
260
+ non-periodic domain face (a grain near the floor must see the LOCAL bed eps/velocity, not
261
+ the deposit's stale ghost values)."""
178
262
  g = self.g
179
263
  for a, (n, per) in enumerate(zip((self.nx, self.ny, self.nz), self.periodic)):
180
- if not per:
264
+ if per:
265
+ f[_sl(a, g - 1)] = f[_sl(a, n + g - 1)]
266
+ f[_sl(a, n + g)] = f[_sl(a, g)]
267
+ else:
268
+ f[_sl(a, g - 1)] = f[_sl(a, g)]
269
+ f[_sl(a, n + g)] = f[_sl(a, n + g - 1)]
270
+
271
+ def _fill_domain(self, f):
272
+ """MPI: zero-gradient fill of the non-periodic global-domain-boundary ghosts (the halo
273
+ fill never touches them). Call AFTER exchange_field."""
274
+ g = self.g
275
+ lo = (int(self._ox / self.h), int(self._oy / self.h), int(self._oz / self.h))
276
+ dims = (self.nx, self.ny, self.nz)
277
+ gdims = (self.gnx, self.gny, self.gnz)
278
+ for a in range(3):
279
+ if self.periodic[a]:
181
280
  continue
182
- f[_sl(a, g - 1)] = f[_sl(a, n + g - 1)]
183
- f[_sl(a, n + g)] = f[_sl(a, g)]
281
+ if lo[a] == 0:
282
+ f[_sl(a, g - 1)] = f[_sl(a, g)]
283
+ if lo[a] + dims[a] == gdims[a]:
284
+ f[_sl(a, dims[a] + g)] = f[_sl(a, dims[a] + g - 1)]
184
285
 
185
286
  def update_void_fraction(self, pos):
186
287
  # deposit target: a registered flow field under MPI (so the halo folds ghost deposits + fills
@@ -194,14 +295,35 @@ class CfdDem:
194
295
  # With no geometry set sdf is all-zero -> every corner fluid -> plain trilinear deposit.
195
296
  self._c.deposit_solid_volume(pos, self._rad, sv, self._fv("sdf"), *self._gm())
196
297
  if self.mpi:
298
+ self._fold_domain(sv) # non-periodic domain-boundary ghosts (halo never folds them)
197
299
  self.flow.exchange_field_add("solidvol") # fold cross-rank + periodic ghost deposits
198
300
  else:
199
301
  self._fold(sv)
302
+ if self._smooth_sweeps:
303
+ # Diffusive smoothing of the deposited solid volume (MFIX DES_DIFFUSE_WIDTH): decouple the
304
+ # porosity smoothing length from the CFD cell so a coarse cell/dp bed sees a smooth,
305
+ # grid-independent void fraction. Volume-conserving (zero-flux at the global walls).
306
+ if self.mpi:
307
+ # Interior rank faces diffuse across the block boundary: refresh the halo before
308
+ # every Jacobi sweep so open faces read the neighbour's pre-sweep values — this
309
+ # reproduces the single-rank closed-box arithmetic exactly (global faces stay
310
+ # zero-flux, flux across rank faces is antisymmetric => volume conserved).
311
+ for _ in range(self._smooth_sweeps):
312
+ self.flow.exchange_field("solidvol")
313
+ self._c.smooth_solid_volume(sv, *self._gm(), 1, self._smooth_alpha,
314
+ self._smooth_open)
315
+ else:
316
+ self._c.smooth_solid_volume(sv, *self._gm(), self._smooth_sweeps, self._smooth_alpha)
200
317
  self._c.compute_void_fraction(sv, ep, self.inv_vcell, self.eps_min)
201
318
  if self.mpi:
202
319
  self.flow.exchange_field("eps") # fill the ghosts the gather stencil reads
320
+ self._fill_domain(ep)
203
321
  else:
204
322
  self._fill(ep)
323
+ # Non-periodic ghost fills extrapolate and can leave eps outside [eps_min, 1] (observed
324
+ # 1.95 at a freeboard boundary); the porous coefficients and the gather stencil read those
325
+ # ghosts, so clamp the whole padded block to the physical range.
326
+ self.xp.clip(ep, self.eps_min, 1.0, out=ep)
205
327
  self._eps = ep # compute_forces reads this at the particles
206
328
 
207
329
  def compute_forces(self, pos, vel):
@@ -213,19 +335,34 @@ class CfdDem:
213
335
  gm = self._gm()
214
336
  has_p = pos.shape[0] > 0 # a rank may own no particles under MPI (skip the per-particle
215
337
  db = self._fv("drag_beta") if self.implicit_drag else None # kernels, keep the collectives)
338
+ # per-particle inverse mass (zero-copy dem view) for the stiff-safe exponential-integrator
339
+ # drag cap: beta_eff = (m/dt)(1 - exp(-beta dt/m)) — exact for linear drag over the coupling
340
+ # interval, unconditionally stable for any drag stiffness (the explicit per-substep force
341
+ # application otherwise blows up once beta*dt/m ~ 1).
342
+ im = self.xp.from_dlpack(self.dem.get_inv_mass_view()) if self.device \
343
+ else np.asarray(self.dem.get_inv_mass_view())
344
+ # The cap models the PARTICLE momentum update over the coupling interval; a fixed bed
345
+ # (move_particles=False) integrates no particles, so the cap must be off — dt_exch=0 makes
346
+ # effectiveBeta return the raw beta on both sides of the exchange. (This also guards against
347
+ # dem's set_positions (N,4) convention, which remaps w==0 to inv_mass=1: "fixed" bed
348
+ # particles otherwise look like unit-mass movers and the cap floors the dense-bed drag.)
349
+ dt_exch = self.fluid_dt if self.move_particles else 0.0
216
350
  # porous (volume-averaged, Model B: the fluid carries the full -grad p) converts the drag
217
351
  # closures beta_B = beta_A/eps inside the kernel (model_b flag); the incompressible mode
218
352
  # keeps the literature Model-A forms unchanged.
219
353
  if has_p and self.implicit_drag:
220
- self._c.compute_drag_implicit(pos, vel, self._rad, uf, vf, wf, self._eps, sd, self._fdrag,
221
- db, fx, fy, fz, *gm, self.mu, self.rho, self.inv_vcell,
222
- self.drag_kind, self.porous)
354
+ self._c.compute_drag_implicit(pos, vel, self._rad, im, uf, vf, wf, self._eps, sd,
355
+ self._fdrag, db, fx, fy, fz, *gm, self.mu, self.rho,
356
+ self.inv_vcell, self.drag_kind, self.porous,
357
+ dt_exch, *self.gravity)
223
358
  elif has_p:
224
- self._c.compute_drag_feedback(pos, vel, self._rad, uf, vf, wf, self._eps, sd, self._fdrag,
225
- fx, fy, fz, *gm, self.mu, self.rho, self.inv_vcell,
226
- self.drag_kind, self.porous)
359
+ self._c.compute_drag_feedback(pos, vel, self._rad, im, uf, vf, wf, self._eps, sd,
360
+ self._fdrag, fx, fy, fz, *gm, self.mu, self.rho,
361
+ self.inv_vcell, self.drag_kind, self.porous,
362
+ dt_exch, *self.gravity)
227
363
  if self.implicit_drag:
228
364
  if self.mpi:
365
+ self._fold_domain(db)
229
366
  self.flow.exchange_field_add("drag_beta")
230
367
  else:
231
368
  self._fold(db)
@@ -235,6 +372,7 @@ class CfdDem:
235
372
  # fold the reaction feedback (force_*) onto owners: reverse halo under MPI, periodic wrap else.
236
373
  for nm, f in (("force_x", fx), ("force_y", fy), ("force_z", fz)):
237
374
  if self.mpi:
375
+ self._fold_domain(f)
238
376
  self.flow.exchange_field_add(nm)
239
377
  else:
240
378
  self._fold(f)
@@ -84,6 +84,26 @@ NB_MODULE(_coupling, m) {
84
84
  "wall-aware: the hold-up is distributed over the fluid corners only (sdf>=0), reweighted to a "
85
85
  "partition of unity so no volume leaks into the solid. Fold ghosts before compute_void_fraction.");
86
86
 
87
+ m.def(
88
+ "smooth_solid_volume",
89
+ [](nb::ndarray<> solidvol, double ox, double oy, double oz, double h, int ex, int ey, int ez,
90
+ int g, int nsweeps, double alpha, int open_faces) {
91
+ auto sv = flatField(solidvol, "smooth_solid_volume(solidvol)");
92
+ Kokkos::View<double*, MemSpace> owner("peclet::coupling::smooth_tmp", sv.extent(0));
93
+ FlatV tmp(owner.data(), owner.extent(0)); // unmanaged alias: same View type as `sv`
94
+ peclet::coupling::smoothField(sv, tmp, gmap(ox, oy, oz, h, ex, ey, ez, g), nsweeps, alpha,
95
+ open_faces);
96
+ },
97
+ nb::arg("solidvol"), nb::arg("ox"), nb::arg("oy"), nb::arg("oz"), nb::arg("h"), nb::arg("ex"),
98
+ nb::arg("ey"), nb::arg("ez"), nb::arg("g"), nb::arg("nsweeps"), nb::arg("alpha") = 1.0 / 6.0,
99
+ nb::arg("open_faces") = 0,
100
+ "Volume-conserving diffusive smoothing of the deposited `solidvol` (MFIX DES_DIFFUSE_WIDTH "
101
+ "analog): nsweeps explicit diffusion sweeps => Gaussian sigma=sqrt(2*alpha*nsweeps) cells, "
102
+ "zero-flux at the domain boundary (conserves total solid volume). Call after folding ghosts, "
103
+ "before compute_void_fraction. `open_faces` (bit 2*axis / 2*axis+1 = minus/plus face): local "
104
+ "faces that are interior MPI rank boundaries read the halo ghost instead of zero-flux — "
105
+ "halo-fill before every sweep (nsweeps=1 per refresh).");
106
+
87
107
  m.def(
88
108
  "compute_void_fraction",
89
109
  [](nb::ndarray<> solidvol, nb::ndarray<> eps, double inv_vcell, double eps_min) {
@@ -113,10 +133,11 @@ NB_MODULE(_coupling, m) {
113
133
 
114
134
  m.def(
115
135
  "compute_drag_feedback",
116
- [](nb::ndarray<> pos, nb::ndarray<> vel, nb::ndarray<> rad, nb::ndarray<> uf, nb::ndarray<> vf,
117
- nb::ndarray<> wf, nb::ndarray<> eps, nb::ndarray<> sdf, nb::ndarray<> fdrag, nb::ndarray<> fx,
118
- nb::ndarray<> fy, nb::ndarray<> fz, double ox, double oy, double oz, double h, int ex,
119
- int ey, int ez, int g, double mu, double rho, double inv_vcell, int drag_kind, bool model_b) {
136
+ [](nb::ndarray<> pos, nb::ndarray<> vel, nb::ndarray<> rad, nb::ndarray<> inv_mass,
137
+ nb::ndarray<> uf, nb::ndarray<> vf, nb::ndarray<> wf, nb::ndarray<> eps, nb::ndarray<> sdf,
138
+ nb::ndarray<> fdrag, nb::ndarray<> fx, nb::ndarray<> fy, nb::ndarray<> fz, double ox,
139
+ double oy, double oz, double h, int ex, int ey, int ez, int g, double mu, double rho,
140
+ double inv_vcell, int drag_kind, bool model_b, double dt_exch, double gx, double gy, double gz) {
120
141
  const GridMap mp = gmap(ox, oy, oz, h, ex, ey, ez, g);
121
142
  auto Fx = flatField(fx, "fx"), Fy = flatField(fy, "fy"), Fz = flatField(fz, "fz");
122
143
  Kokkos::deep_copy(Fx, 0.0);
@@ -124,27 +145,29 @@ NB_MODULE(_coupling, m) {
124
145
  Kokkos::deep_copy(Fz, 0.0);
125
146
  peclet::coupling::computeDragFeedback(
126
147
  (int)pos.shape(0), vec3(pos, "pos"), vec3(vel, "vel"), vecf(rad, "rad"),
127
- flatField(uf, "uf"), flatField(vf, "vf"), flatField(wf, "wf"), flatField(eps, "eps"),
128
- flatField(sdf, "sdf"), vec3(fdrag, "fdrag"), Fx, Fy, Fz, mp, mu, rho, inv_vcell,
129
- drag_kind, model_b);
148
+ vecf(inv_mass, "inv_mass"), flatField(uf, "uf"), flatField(vf, "vf"),
149
+ flatField(wf, "wf"), flatField(eps, "eps"), flatField(sdf, "sdf"), vec3(fdrag, "fdrag"),
150
+ Fx, Fy, Fz, mp, mu, rho, inv_vcell, drag_kind, model_b, dt_exch, gx, gy, gz);
130
151
  },
131
- nb::arg("pos"), nb::arg("vel"), nb::arg("rad"), nb::arg("uf"), nb::arg("vf"), nb::arg("wf"),
152
+ nb::arg("pos"), nb::arg("vel"), nb::arg("rad"), nb::arg("inv_mass"), nb::arg("uf"),
153
+ nb::arg("vf"), nb::arg("wf"),
132
154
  nb::arg("eps"), nb::arg("sdf"), nb::arg("fdrag"), nb::arg("fx"), nb::arg("fy"), nb::arg("fz"),
133
155
  nb::arg("ox"), nb::arg("oy"), nb::arg("oz"), nb::arg("h"), nb::arg("ex"), nb::arg("ey"),
134
156
  nb::arg("ez"), nb::arg("g"), nb::arg("mu"), nb::arg("rho"), nb::arg("inv_vcell"),
135
- nb::arg("drag_kind"), nb::arg("model_b") = false,
157
+ nb::arg("drag_kind"), nb::arg("model_b") = false, nb::arg("dt_exch") = 0.0, nb::arg("gx") = 0.0, nb::arg("gy") = 0.0, nb::arg("gz") = 0.0,
136
158
  "Gather (uf,vf,wf,eps) at each particle, evaluate the drag law (0 Stokes, 1 Schiller-Naumann, "
137
- "2 Ergun, 3 Di Felice), write the drag force to `fdrag` (N,3) and the reaction force density "
159
+ "2 Ergun, 3 Di Felice, 4 Wen-Yu, 5 Gidaspow, 6 Beetstra/BVK), write the drag force to `fdrag` "
160
+ "(N,3) and the reaction force density "
138
161
  "-F/Vcell onto (fx,fy,fz) (zeroed here). Momentum-conserving. EXPLICIT feedback — use "
139
162
  "compute_drag_implicit for stiff (dense-bed) drag.");
140
163
 
141
164
  m.def(
142
165
  "compute_drag_implicit",
143
- [](nb::ndarray<> pos, nb::ndarray<> vel, nb::ndarray<> rad, nb::ndarray<> uf, nb::ndarray<> vf,
144
- nb::ndarray<> wf, nb::ndarray<> eps, nb::ndarray<> sdf, nb::ndarray<> fdrag,
145
- nb::ndarray<> dragbeta, nb::ndarray<> fx, nb::ndarray<> fy, nb::ndarray<> fz, double ox,
146
- double oy, double oz, double h, int ex, int ey, int ez, int g, double mu, double rho,
147
- double inv_vcell, int drag_kind, bool model_b) {
166
+ [](nb::ndarray<> pos, nb::ndarray<> vel, nb::ndarray<> rad, nb::ndarray<> inv_mass,
167
+ nb::ndarray<> uf, nb::ndarray<> vf, nb::ndarray<> wf, nb::ndarray<> eps, nb::ndarray<> sdf,
168
+ nb::ndarray<> fdrag, nb::ndarray<> dragbeta, nb::ndarray<> fx, nb::ndarray<> fy,
169
+ nb::ndarray<> fz, double ox, double oy, double oz, double h, int ex, int ey, int ez, int g,
170
+ double mu, double rho, double inv_vcell, int drag_kind, bool model_b, double dt_exch, double gx, double gy, double gz) {
148
171
  const GridMap mp = gmap(ox, oy, oz, h, ex, ey, ez, g);
149
172
  auto Db = flatField(dragbeta, "drag_beta"), Fx = flatField(fx, "fx"),
150
173
  Fy = flatField(fy, "fy"), Fz = flatField(fz, "fz");
@@ -154,15 +177,18 @@ NB_MODULE(_coupling, m) {
154
177
  Kokkos::deep_copy(Fz, 0.0);
155
178
  peclet::coupling::computeDragImplicit(
156
179
  (int)pos.shape(0), vec3(pos, "pos"), vec3(vel, "vel"), vecf(rad, "rad"),
157
- flatField(uf, "uf"), flatField(vf, "vf"), flatField(wf, "wf"), flatField(eps, "eps"),
180
+ vecf(inv_mass, "inv_mass"), flatField(uf, "uf"), flatField(vf, "vf"),
181
+ flatField(wf, "wf"), flatField(eps, "eps"),
158
182
  flatField(sdf, "sdf"), vec3(fdrag, "fdrag"), Db, Fx, Fy, Fz, mp, mu, rho, inv_vcell,
159
- drag_kind, model_b);
183
+ drag_kind, model_b, dt_exch, gx, gy, gz);
160
184
  },
161
- nb::arg("pos"), nb::arg("vel"), nb::arg("rad"), nb::arg("uf"), nb::arg("vf"), nb::arg("wf"),
185
+ nb::arg("pos"), nb::arg("vel"), nb::arg("rad"), nb::arg("inv_mass"), nb::arg("uf"),
186
+ nb::arg("vf"), nb::arg("wf"),
162
187
  nb::arg("eps"), nb::arg("sdf"), nb::arg("fdrag"), nb::arg("drag_beta"), nb::arg("fx"),
163
188
  nb::arg("fy"), nb::arg("fz"), nb::arg("ox"), nb::arg("oy"), nb::arg("oz"), nb::arg("h"),
164
189
  nb::arg("ex"), nb::arg("ey"), nb::arg("ez"), nb::arg("g"), nb::arg("mu"), nb::arg("rho"),
165
190
  nb::arg("inv_vcell"), nb::arg("drag_kind"), nb::arg("model_b") = false,
191
+ nb::arg("dt_exch") = 0.0, nb::arg("gx") = 0.0, nb::arg("gy") = 0.0, nb::arg("gz") = 0.0,
166
192
  "Implicit (semi-implicit) drag: writes `fdrag` (particle force) and deposits the linear-drag "
167
193
  "coefficient density onto `drag_beta` and the target beta*u_p onto (fx,fy,fz) (all zeroed "
168
194
  "here) for flow.enable_drag() to treat -beta*(u-u_p) implicitly. Stable for stiff beds.");
@@ -21,6 +21,24 @@ namespace peclet::coupling {
21
21
 
22
22
  using peclet::core::interp::GridMap;
23
23
 
24
+ // Is the particle's trilinear stencil within coupling range of the local block? axisStencil CLAMPS
25
+ // out-of-range particles into the boundary cell, so a grain that has escaped the domain (e.g. pushed
26
+ // through a DEM wall by a violent contact solve and now in free fall below the distributor) would
27
+ // keep depositing its full volume and drag target (with a runaway u_p) into the boundary row forever
28
+ // — a guaranteed gas blow-up. A particle is coupled iff its stencil overlaps [-1, n] (one ghost
29
+ // layer, whose deposits the driver folds back onto the boundary cell); anything further out is
30
+ // dropped from the exchange (no deposit, zero drag — ballistic until the DEM recovers it).
31
+ KOKKOS_INLINE_FUNCTION bool axisInRange(double p, double origin, double inv, int nInner) {
32
+ const double si = (p - origin) * inv - 0.5; // continuous cell-centre coordinate
33
+ return si > -2.0 && si < (double)nInner + 1.0;
34
+ }
35
+ template <class GM>
36
+ KOKKOS_INLINE_FUNCTION bool stencilInRange(double px, double py, double pz, const GM& m, int nx,
37
+ int ny, int nz) {
38
+ return axisInRange(px, m.ox, m.idx, nx) && axisInRange(py, m.oy, m.idy, ny) &&
39
+ axisInRange(pz, m.oz, m.idz, nz);
40
+ }
41
+
24
42
  // A stencil corner is a FLUID cell iff the (cell-centred) SDF there is >= 0 (SDF < 0 inside the
25
43
  // solid, per docs/CONVENTIONS.md). With no geometry set, flow's sdf field is all-zero -> every corner
26
44
  // reads as fluid -> the wall-aware paths below reduce EXACTLY to plain trilinear (the periodic no-wall
@@ -30,6 +48,25 @@ KOKKOS_INLINE_FUNCTION bool cornerIsFluid(const MaskV& sdf, long o) {
30
48
  return (double)sdf(o) >= 0.0;
31
49
  }
32
50
 
51
+ // Stiff-safe (exponential-integrator) effective drag coefficient. The particle-side exchange applies
52
+ // a CONSTANT force over the coupling interval dt (dem substeps accumulate it linearly), so the raw
53
+ // linear coefficient beta is explicit there and blows up for beta*dt/m >~ 1 (measured: a coupled
54
+ // fluidized bed doubling |v| per step once the eps-conservative projection raised the interstitial
55
+ // velocities). The exact solution of m dv/dt = beta (u - v) over dt is reproduced by the constant
56
+ // force F = beta_eff (u - v0) with
57
+ // beta_eff = (m/dt) (1 - exp(-beta dt / m))
58
+ // — equal to beta for beta*dt/m << 1 and saturating at m/dt (v lands exactly ON u, never beyond):
59
+ // unconditionally stable for any drag stiffness. Used for the particle force AND the fluid-side
60
+ // deposits (beta / feedback), so the exchange stays momentum-conserving. invM <= 0 (static) => raw.
61
+ KOKKOS_INLINE_FUNCTION double effectiveBeta(double beta, double invM, double dt) {
62
+ if (invM <= 0.0 || beta <= 0.0 || dt <= 0.0)
63
+ return beta;
64
+ const double x = beta * invM * dt; // beta*dt/m
65
+ if (x < 1e-4)
66
+ return beta; // expm1 not needed; avoids 0/0 noise
67
+ return (1.0 - Kokkos::exp(-x)) / (invM * dt);
68
+ }
69
+
33
70
  // Wall-aware trilinear gather: interpolate a cell-centred field at the particle using ONLY the fluid
34
71
  // corners, reweighted to a partition of unity (a solid corner carries no data — its velocity is the
35
72
  // no-slip 0 and its eps is meaningless — so including it biases the interpolant). Reduces to plain
@@ -96,6 +133,8 @@ void depositSolidVolume(int np, PosV pos, RadV rad, FieldV solidvol, FieldV sdf,
96
133
  const int nx = m.ex - 2 * m.g, ny = m.ey - 2 * m.g, nz = m.ez - 2 * m.g;
97
134
  Kokkos::parallel_for(
98
135
  "peclet::coupling::deposit_vol", Kokkos::RangePolicy<Exec>(0, np), KOKKOS_LAMBDA(int p) {
136
+ if (!stencilInRange((double)pos(p, 0), (double)pos(p, 1), (double)pos(p, 2), m, nx, ny, nz))
137
+ return; // escaped the domain: no deposit (axisStencil would clamp it into the boundary)
99
138
  int i0, j0, k0;
100
139
  double wx, wy, wz;
101
140
  peclet::core::interp::detail::axisStencil((double)pos(p, 0), m.ox, m.idx, nx, i0, wx);
@@ -110,7 +149,62 @@ void depositSolidVolume(int np, PosV pos, RadV rad, FieldV solidvol, FieldV sdf,
110
149
  });
111
150
  }
112
151
 
113
- // eps = clamp(1 - solidvol/Vcell, epsMin, 1) over the whole (padded) field.
152
+ // Volume-conserving diffusive smoothing of a deposited field (the MFIX DES_DIFFUSE_WIDTH analog).
153
+ // `nsweeps` explicit Jacobi diffusion sweeps with per-neighbour coefficient `alpha` (<= 1/6 for 3D
154
+ // stability) spread each particle's deposited volume over a Gaussian of length sigma = sqrt(2*alpha*
155
+ // nsweeps) cells, DECOUPLING the porosity smoothing length from the CFD cell size — the fix a coarse
156
+ // cell/dp bed needs for a smooth, grid-independent void fraction (MFIX-Exa: "smoothing of the fields
157
+ // is crucial to achieve grid-independent results"). Zero-flux (Neumann) at the inner-domain boundary
158
+ // so nothing diffuses out through the distributor/walls and the total solid volume is conserved
159
+ // exactly (the discrete zero-flux Laplacian sums to zero over the inner cells). Ping-pongs between
160
+ // `f` and scratch `tmp` (same padded size); the result is left in `f`. With `openFaces == 0` ghost
161
+ // cells are never read (neighbours outside the inner domain contribute the cell's own value => zero
162
+ // flux), so the caller's ghost fill after this is unaffected. Under MPI a local block face that is an
163
+ // INTERIOR rank boundary is not a wall: setting its bit in `openFaces` (bit 2*axis = minus face,
164
+ // bit 2*axis+1 = plus face) makes the sweep read the first ghost layer there instead — the caller
165
+ // must halo-fill `f` before EVERY sweep (call with nsweeps=1 per refresh; the sweep is Jacobi, so a
166
+ // per-sweep refresh reproduces the single-rank closed-box arithmetic bit-for-bit; flux out of one
167
+ // block equals flux into its neighbour, so global volume conservation is preserved). Faces on the
168
+ // GLOBAL domain boundary stay closed (bit clear) to match the validated single-rank behaviour.
169
+ // NOTE: smooths across immersed-solid cells too — fine for the walled beds here (no inner SDF); an
170
+ // SDF-masked variant is a follow-up.
171
+ template <class FieldV>
172
+ void smoothField(FieldV f, FieldV tmp, GridMap m, int nsweeps, double alpha, int openFaces = 0) {
173
+ using Exec = Kokkos::DefaultExecutionSpace;
174
+ const int nx = m.ex - 2 * m.g, ny = m.ey - 2 * m.g, nz = m.ez - 2 * m.g, g = m.g;
175
+ const long sx = 1, sy = m.ex, sz = (long)m.ex * m.ey;
176
+ const bool oxm = openFaces & 1, oxp = openFaces & 2, oym = openFaces & 4, oyp = openFaces & 8,
177
+ ozm = openFaces & 16, ozp = openFaces & 32;
178
+ Kokkos::deep_copy(tmp, f); // seed ghosts so an odd-count final copy-back preserves f's ghost
179
+ // cells (and so open faces read current ghosts from either buffer)
180
+ for (int s = 0; s < nsweeps; ++s) {
181
+ FieldV src = (s % 2 == 0) ? f : tmp;
182
+ FieldV dst = (s % 2 == 0) ? tmp : f;
183
+ Kokkos::parallel_for(
184
+ "peclet::coupling::smooth",
185
+ Kokkos::MDRangePolicy<Exec, Kokkos::Rank<3>>({0, 0, 0}, {nx, ny, nz}),
186
+ KOKKOS_LAMBDA(int ix, int iy, int iz) {
187
+ const long c = (long)(ix + g) + (long)(iy + g) * sy + (long)(iz + g) * sz;
188
+ const double v = (double)src(c);
189
+ double lap = 0.0; // zero-flux: an out-of-domain neighbour contributes v (no gradient),
190
+ // unless that face is an open (rank-boundary) face — then read the ghost
191
+ lap += (ix > 0 || oxm ? (double)src(c - sx) : v) - v;
192
+ lap += (ix < nx - 1 || oxp ? (double)src(c + sx) : v) - v;
193
+ lap += (iy > 0 || oym ? (double)src(c - sy) : v) - v;
194
+ lap += (iy < ny - 1 || oyp ? (double)src(c + sy) : v) - v;
195
+ lap += (iz > 0 || ozm ? (double)src(c - sz) : v) - v;
196
+ lap += (iz < nz - 1 || ozp ? (double)src(c + sz) : v) - v;
197
+ dst(c) = (typename FieldV::value_type)(v + alpha * lap);
198
+ });
199
+ }
200
+ if (nsweeps % 2 == 1)
201
+ Kokkos::deep_copy(f, tmp); // odd sweep count left the result in tmp
202
+ }
203
+
204
+ // eps = clamp(1 - solidvol/Vcell, epsMin, 1) over the whole (padded) field. With smoothing on and a
205
+ // small epsMin, epsMin is only a divide-by-zero guard; the physical void fraction (which can fall
206
+ // well below the random-close-packing 0.4 in a dense cell) is preserved instead of being clamped up —
207
+ // clamping ε up to 0.4 under-predicts the Ergun 1/ε^3 drag ~3x in a dense bed and it never fluidizes.
114
208
  template <class FieldV>
115
209
  void voidFraction(FieldV solidvol, FieldV eps, double invVcell, double epsMin) {
116
210
  using Exec = Kokkos::DefaultExecutionSpace;
@@ -129,15 +223,19 @@ void voidFraction(FieldV solidvol, FieldV eps, double invVcell, double epsMin) {
129
223
  // Fused drag + feedback. Gathers (uf,vf,wf,eps) at each particle, evaluates the drag law, writes the
130
224
  // drag force to fdrag(p,:) (dem external force), and scatters the reaction -F*invVcell onto the
131
225
  // (pre-zeroed) grid force-density fields fx,fy,fz. rho/mu physical; dragKind per drag.hpp.
132
- template <class PosV, class VelV, class RadV, class FieldV, class OutV>
133
- void computeDragFeedback(int np, PosV pos, VelV vel, RadV rad, FieldV uf, FieldV vf, FieldV wf,
134
- FieldV eps, FieldV sdf, OutV fdrag, FieldV fx, FieldV fy, FieldV fz,
135
- GridMap m, double mu, double rhof, double invVcell, int dragKind,
136
- bool modelB) {
226
+ template <class PosV, class VelV, class RadV, class InvMV, class FieldV, class OutV>
227
+ void computeDragFeedback(int np, PosV pos, VelV vel, RadV rad, InvMV invMass, FieldV uf, FieldV vf,
228
+ FieldV wf, FieldV eps, FieldV sdf, OutV fdrag, FieldV fx, FieldV fy,
229
+ FieldV fz, GridMap m, double mu, double rhof, double invVcell, int dragKind,
230
+ bool modelB, double dtExch, double gx, double gy, double gz) {
137
231
  using Exec = Kokkos::DefaultExecutionSpace;
138
232
  const int nx = m.ex - 2 * m.g, ny = m.ey - 2 * m.g, nz = m.ez - 2 * m.g;
139
233
  Kokkos::parallel_for(
140
234
  "peclet::coupling::drag_feedback", Kokkos::RangePolicy<Exec>(0, np), KOKKOS_LAMBDA(int p) {
235
+ if (!stencilInRange((double)pos(p, 0), (double)pos(p, 1), (double)pos(p, 2), m, nx, ny, nz)) {
236
+ fdrag(p, 0) = fdrag(p, 1) = fdrag(p, 2) = 0; // escaped: ballistic, no exchange
237
+ return;
238
+ }
141
239
  int i0, j0, k0;
142
240
  double wx, wy, wz;
143
241
  peclet::core::interp::detail::axisStencil((double)pos(p, 0), m.ox, m.idx, nx, i0, wx);
@@ -160,6 +258,35 @@ void computeDragFeedback(int np, PosV pos, VelV vel, RadV rad, FieldV uf, FieldV
160
258
  Fy /= eP;
161
259
  Fz /= eP;
162
260
  }
261
+ // Stiff-safe cap + gravity-exact correction. The constant force reproducing the exact
262
+ // endpoint of m dv/dt = beta (u - v) + m g over dtExch is
263
+ // F = beta_eff (u - v0) - m g (1 - beta_eff/beta)
264
+ // — without the g term the discrete steady state balances at slip = m g / beta_eff (biased
265
+ // by the cap factor, measured 26% on the Stokes terminal-velocity test); with it the fixed
266
+ // point is the physical slip = m g / beta. F*dt stays the exact drag impulse, so the fluid
267
+ // reaction (-F) remains exactly momentum-conserving. g is the constant external
268
+ // acceleration dem applies per substep (gravity); invM<=0 or dtExch<=0 => raw beta, no
269
+ // correction.
270
+ {
271
+ const double vmag2 = vrx * vrx + vry * vry + vrz * vrz;
272
+ if (vmag2 > 1e-60) {
273
+ const double vmag = Kokkos::sqrt(vmag2);
274
+ const double bon = Kokkos::sqrt(Fx * Fx + Fy * Fy + Fz * Fz) / vmag;
275
+ const double invM = (double)invMass(p);
276
+ if (bon > 0.0) {
277
+ const double sc = effectiveBeta(bon, invM, dtExch) / bon;
278
+ Fx *= sc;
279
+ Fy *= sc;
280
+ Fz *= sc;
281
+ if (invM > 0.0 && sc < 1.0) {
282
+ const double corr = (1.0 - sc) / invM; // m (1 - beta_eff/beta)
283
+ Fx -= corr * gx;
284
+ Fy -= corr * gy;
285
+ Fz -= corr * gz;
286
+ }
287
+ }
288
+ }
289
+ }
163
290
  fdrag(p, 0) = (typename OutV::value_type)Fx;
164
291
  fdrag(p, 1) = (typename OutV::value_type)Fy;
165
292
  fdrag(p, 2) = (typename OutV::value_type)Fz;
@@ -175,15 +302,20 @@ void computeDragFeedback(int np, PosV pos, VelV vel, RadV rad, FieldV uf, FieldV
175
302
  // treats -beta*(u - u_p) implicitly (unconditionally stable for a stiff bed). The particle drag
176
303
  // force (evaluated at the current slip) still goes to `fdrag` (explicit on the particle side).
177
304
  // beta_over_n = |F_p| / |vrel|, recovered from the drag law by a unit-slip evaluation.
178
- template <class PosV, class VelV, class RadV, class FieldV, class OutV>
179
- void computeDragImplicit(int np, PosV pos, VelV vel, RadV rad, FieldV uf, FieldV vf, FieldV wf,
180
- FieldV eps, FieldV sdf, OutV fdrag, FieldV dragBeta, FieldV fx, FieldV fy,
181
- FieldV fz, GridMap m, double mu, double rhof, double invVcell, int dragKind,
182
- bool modelB) {
305
+ template <class PosV, class VelV, class RadV, class InvMV, class FieldV, class OutV>
306
+ void computeDragImplicit(int np, PosV pos, VelV vel, RadV rad, InvMV invMass, FieldV uf, FieldV vf,
307
+ FieldV wf, FieldV eps, FieldV sdf, OutV fdrag, FieldV dragBeta, FieldV fx,
308
+ FieldV fy, FieldV fz, GridMap m, double mu, double rhof, double invVcell,
309
+ int dragKind, bool modelB, double dtExch, double gx, double gy,
310
+ double gz) {
183
311
  using Exec = Kokkos::DefaultExecutionSpace;
184
312
  const int nx = m.ex - 2 * m.g, ny = m.ey - 2 * m.g, nz = m.ez - 2 * m.g;
185
313
  Kokkos::parallel_for(
186
314
  "peclet::coupling::drag_implicit", Kokkos::RangePolicy<Exec>(0, np), KOKKOS_LAMBDA(int p) {
315
+ if (!stencilInRange((double)pos(p, 0), (double)pos(p, 1), (double)pos(p, 2), m, nx, ny, nz)) {
316
+ fdrag(p, 0) = fdrag(p, 1) = fdrag(p, 2) = 0; // escaped: ballistic, no exchange
317
+ return;
318
+ }
187
319
  int i0, j0, k0;
188
320
  double wx, wy, wz;
189
321
  peclet::core::interp::detail::axisStencil((double)pos(p, 0), m.ox, m.idx, nx, i0, wx);
@@ -205,17 +337,39 @@ void computeDragImplicit(int np, PosV pos, VelV vel, RadV rad, FieldV uf, FieldV
205
337
  Fy /= eP;
206
338
  Fz /= eP;
207
339
  }
340
+ // beta_over_n = |F|/|vrel| (isotropic linear coefficient at the frozen slip), then the
341
+ // stiff-safe exponential-integrator cap (effectiveBeta) — used consistently for the
342
+ // particle force AND the fluid-side deposits so the exchange conserves momentum. The
343
+ // gravity-exact correction -m g (1 - beta_eff/beta) (see computeDragFeedback) rides the
344
+ // particle force, and its reaction is deposited as a constant term with the drag target.
345
+ const double vmag = Kokkos::sqrt(vrx * vrx + vry * vry + vrz * vrz);
346
+ const double Fmag = Kokkos::sqrt(Fx * Fx + Fy * Fy + Fz * Fz);
347
+ const double bonRaw = (vmag > 1e-30) ? Fmag / vmag : 0.0;
348
+ const double bon = effectiveBeta(bonRaw, (double)invMass(p), dtExch);
349
+ double cgx = 0.0, cgy = 0.0, cgz = 0.0; // m (1 - beta_eff/beta) * g
350
+ if (bonRaw > 0.0) {
351
+ const double sc = bon / bonRaw;
352
+ Fx *= sc;
353
+ Fy *= sc;
354
+ Fz *= sc;
355
+ const double invM = (double)invMass(p);
356
+ if (invM > 0.0 && sc < 1.0) {
357
+ const double corr = (1.0 - sc) / invM;
358
+ cgx = corr * gx;
359
+ cgy = corr * gy;
360
+ cgz = corr * gz;
361
+ Fx -= cgx;
362
+ Fy -= cgy;
363
+ Fz -= cgz;
364
+ }
365
+ }
208
366
  fdrag(p, 0) = (typename OutV::value_type)Fx;
209
367
  fdrag(p, 1) = (typename OutV::value_type)Fy;
210
368
  fdrag(p, 2) = (typename OutV::value_type)Fz;
211
- // beta_over_n = |F|/|vrel| (isotropic linear coefficient at the frozen slip)
212
- const double vmag = Kokkos::sqrt(vrx * vrx + vry * vry + vrz * vrz);
213
- const double Fmag = Kokkos::sqrt(Fx * Fx + Fy * Fy + Fz * Fz);
214
- const double bon = (vmag > 1e-30) ? Fmag / vmag : 0.0;
215
369
  scatterAtMasked(dragBeta, sdf, b, sx, sy, sz, wx, wy, wz, bon * invVcell);
216
- scatterAtMasked(fx, sdf, b, sx, sy, sz, wx, wy, wz, bon * upx * invVcell);
217
- scatterAtMasked(fy, sdf, b, sx, sy, sz, wx, wy, wz, bon * upy * invVcell);
218
- scatterAtMasked(fz, sdf, b, sx, sy, sz, wx, wy, wz, bon * upz * invVcell);
370
+ scatterAtMasked(fx, sdf, b, sx, sy, sz, wx, wy, wz, (bon * upx + cgx) * invVcell);
371
+ scatterAtMasked(fy, sdf, b, sx, sy, sz, wx, wy, wz, (bon * upy + cgy) * invVcell);
372
+ scatterAtMasked(fz, sdf, b, sx, sy, sz, wx, wy, wz, (bon * upz + cgz) * invVcell);
219
373
  });
220
374
  }
221
375
 
@@ -25,7 +25,10 @@ enum DragKind {
25
25
  ERGUN = 2,
26
26
  DI_FELICE = 3,
27
27
  WEN_YU = 4,
28
- GIDASPOW = 5
28
+ GIDASPOW = 5,
29
+ BEETSTRA = 6, // Beetstra-van der Hoef-Kuipers DNS drag (the published "BVK2")
30
+ TANG = 7 // Tang-Peters-Kuipers-Kriebitzsch-van der Hoef (2015) — what MFIX-Exa's
31
+ // "BVK2" option ACTUALLY executes (the Beetstra branch is #if 0'd out)
29
32
  };
30
33
 
31
34
  // Wen & Yu interphase drag coefficient per particle, F = beta_over_n * vrel. The dilute branch of
@@ -75,6 +78,38 @@ KOKKOS_INLINE_FUNCTION void dragForce(int kind, double vx, double vy, double vz,
75
78
  beta_over_n = 6.0 * M_PI * mu * r * corr * Kokkos::pow(eps, -(chi - 1.0));
76
79
  } else if (kind == WEN_YU) {
77
80
  beta_over_n = wenYuBetaOverN(vmag, d, mu, rhof, eps, Vp);
81
+ } else if (kind == BEETSTRA) {
82
+ // Beetstra, van der Hoef & Kuipers (AIChE J 53, 489, 2007) monodisperse DNS drag — the "BVK2"
83
+ // law of MFIX(-Exa). Dimensionless drag F(phi,Re) normalized by the Stokes force 3 pi mu d u on
84
+ // an isolated sphere; per-particle F = 3 pi mu d eps F(phi,Re) vrel (the extra eps is the
85
+ // MFIX/TFM interstitial-slip convention: beta = 18 mu eps phi F / d^2, beta_over_n = beta Vp/phi).
86
+ // Re is the voidage (superficial) particle Reynolds number. F -> 1 as phi -> 0, Re -> 0.
87
+ const double phi = 1.0 - eps;
88
+ const double Re = eps * rhof * d * vmag / mu;
89
+ const double e2 = eps * eps;
90
+ double F = 10.0 * phi / e2 + e2 * (1.0 + 1.5 * Kokkos::sqrt(phi));
91
+ if (Re > 1e-12) {
92
+ F += 0.413 * Re / (24.0 * e2) *
93
+ (1.0 / eps + 3.0 * eps * phi + 8.4 * Kokkos::pow(Re, -0.343)) /
94
+ (1.0 + Kokkos::pow(10.0, 3.0 * phi) * Kokkos::pow(Re, -0.5 * (1.0 + 4.0 * phi)));
95
+ }
96
+ beta_over_n = 3.0 * M_PI * mu * d * eps * F;
97
+ } else if (kind == TANG) {
98
+ // Tang, Peters, Kuipers, Kriebitzsch & van der Hoef (AIChE J 61, 688, 2015) monodisperse DNS
99
+ // drag — the correlation MFIX-Exa's "BVK2" drag option actually executes (verified against
100
+ // mfix_des_drag_K.H: the Beetstra 2007 branch there is compiled out). Same static part as
101
+ // Beetstra; the inertial part is ~15% stronger for Re ~ 2-20. Same normalization + Re as BVK:
102
+ // per-particle F = 3 pi mu d eps F(phi,Re) vrel, Re the voidage particle Reynolds number.
103
+ const double phi = 1.0 - eps;
104
+ const double Re = eps * rhof * d * vmag / mu;
105
+ const double e2 = eps * eps;
106
+ const double inv_e4 = 1.0 / (e2 * e2);
107
+ double F = 10.0 * phi / e2 + e2 * (1.0 + 1.5 * Kokkos::sqrt(phi));
108
+ if (Re > 1e-12) {
109
+ F += Re * (0.11 * phi * (1.0 + phi) - 4.56e-3 * inv_e4 +
110
+ Kokkos::pow(Re, -0.343) * (0.169 * eps + 6.44e-2 * inv_e4));
111
+ }
112
+ beta_over_n = 3.0 * M_PI * mu * d * eps * F;
78
113
  } else if (kind == GIDASPOW) {
79
114
  // Gidaspow (1994): Ergun for the dense regime, Wen & Yu for the dilute, switched at eps = 0.8.
80
115
  beta_over_n = (eps < 0.8) ? ergunBetaOverN(vmag, d, mu, rhof, eps, Vp)
@@ -48,7 +48,7 @@ def run_bed(f_drive, eps_target=0.6, N=16, mu=1.0, rho=1.0, dt=0.5, steps=120):
48
48
  d.enable_periodicity(True, True, True)
49
49
  d.set_positions(posw)
50
50
  d.set_velocities(np.zeros((Np, 3), dtype=np.float32))
51
- cpl = CfdDem(s, d, fluid_dt=dt, mu=mu, rho=rho, radius=r, drag="ergun", eps_min=0.05,
51
+ cpl = CfdDem(s, d, fluid_dt=dt, mu=mu, rho=rho, radius=r, drag="ergun", eps_min=0.05, porous=False,
52
52
  move_particles=False) # fixed bed: no DEM dynamics
53
53
  for _ in range(steps):
54
54
  cpl.step()
@@ -56,7 +56,7 @@ def run_bed(f_drive, comm, eps_target=0.6, N=16, mu=1.0, rho=1.0, dt=0.5, steps=
56
56
  d.set_positions(posw)
57
57
  d.set_velocities(np.zeros((Np, 3), dtype=np.float32))
58
58
 
59
- cpl = CfdDem(s, d, fluid_dt=dt, mu=mu, rho=rho, radius=r, drag="ergun", eps_min=0.05,
59
+ cpl = CfdDem(s, d, fluid_dt=dt, mu=mu, rho=rho, radius=r, drag="ergun", eps_min=0.05, porous=False,
60
60
  move_particles=False)
61
61
  for _ in range(steps):
62
62
  cpl.step()
@@ -0,0 +1,114 @@
1
+ """Multi-rank void-fraction SMOOTHING: distributed diffusive smoothing must reproduce the
2
+ single-rank result exactly.
3
+
4
+ The MFIX-style diffusive smoothing (smooth_width) was single-rank-only: the sweep treated every
5
+ local block face as a zero-flux wall, so a rank boundary acted as a spurious internal wall. Now
6
+ interior rank faces read the halo ghost (open_faces mask) and the driver refreshes the solidvol
7
+ halo before every Jacobi sweep, which makes the multi-rank sweep arithmetic identical to the
8
+ single-rank closed-box sweep (global faces stay zero-flux, matching the validated single-rank
9
+ path byte-for-byte).
10
+
11
+ This test deposits a deterministic particle cloud, smooths, and checks:
12
+ 1. global solid-volume conservation (smoothing must not create/destroy hold-up), and
13
+ 2. the gathered global eps field matches the np=1 reference to ~machine precision.
14
+
15
+ Run: mpirun -np 1 python test_mpi_smoothing.py (writes the reference)
16
+ mpirun -np {2,4} python test_mpi_smoothing.py
17
+ """
18
+ import os
19
+ import numpy as np
20
+ import peclet.flow
21
+ import peclet.dem
22
+ from peclet.coupling import CfdDem
23
+ from mpi4py import MPI
24
+
25
+ REF_FILE = os.path.join(os.path.dirname(os.path.abspath(__file__)), "smoothing_ref_eps.npy")
26
+ N = 16 # global grid N^3, h = 1
27
+ R = 0.3 # particle radius
28
+ SMOOTH_W = 2.0 # smoothing length in cells
29
+ EPS_MIN = 0.05
30
+
31
+
32
+ def global_particles():
33
+ rng = np.random.default_rng(42)
34
+ # keep a margin off the domain faces so the trilinear deposit never lands in a non-periodic
35
+ # corner case; the box is fully periodic anyway.
36
+ return (rng.uniform(0.5, N - 0.5, size=(200, 3))).astype(np.float32)
37
+
38
+
39
+ def run(comm):
40
+ rank, size = comm.Get_rank(), comm.Get_size()
41
+ gpos = global_particles()
42
+
43
+ if size > 1:
44
+ (ox, oy, oz), (lnx, lny, lnz) = peclet.flow.mpi_block(N, N, N)
45
+ else:
46
+ (ox, oy, oz), (lnx, lny, lnz) = (0, 0, 0), (N, N, N)
47
+ cell = np.floor(gpos).astype(int)
48
+ keep = ((cell[:, 0] >= ox) & (cell[:, 0] < ox + lnx) &
49
+ (cell[:, 1] >= oy) & (cell[:, 1] < oy + lny) &
50
+ (cell[:, 2] >= oz) & (cell[:, 2] < oz + lnz))
51
+ mine = gpos[keep]
52
+ Np = mine.shape[0]
53
+
54
+ s = peclet.flow.Solver(lnx, lny, lnz)
55
+ s.set_rho(1.0); s.set_mu(1.0); s.set_dt(0.1)
56
+ if size > 1:
57
+ s.init_mpi(N, N, N)
58
+ s.set_pressure_geometry(np.asfortranarray(np.full((lnx, lny, lnz), 10.0)))
59
+
60
+ d = peclet.dem.Simulation(max(Np, 1))
61
+ d.initialize(shape_type=1, radius=R)
62
+ d.set_domain((0, 0, 0), (N, N, N))
63
+ d.enable_periodicity(True, True, True)
64
+ posw = np.concatenate([mine, np.zeros((Np, 1), dtype=np.float32)], axis=1) # invMass 0: fixed
65
+ d.set_positions(posw)
66
+ d.set_velocities(np.zeros((Np, 3), dtype=np.float32))
67
+
68
+ cpl = CfdDem(s, d, fluid_dt=0.1, mu=1.0, rho=1.0, radius=R, drag="stokes",
69
+ eps_min=EPS_MIN, smooth_width=SMOOTH_W, move_particles=False)
70
+ assert cpl._smooth_sweeps > 0, "smoothing must be active under MPI now"
71
+ cpl._resize_particles(Np)
72
+ cpl.update_void_fraction(cpl.xp.asarray(mine))
73
+
74
+ g = cpl.g
75
+ sv = cpl._fv("solidvol") if cpl._eps_is_field else cpl._solidvol
76
+ ep = cpl._eps
77
+ if cpl.device:
78
+ sv, ep = sv.get(), ep.get()
79
+ sv_in = np.asarray(sv)[g:g + lnx, g:g + lny, g:g + lnz]
80
+ ep_in = np.asarray(ep)[g:g + lnx, g:g + lny, g:g + lnz]
81
+
82
+ # 1) conservation: smoothing must preserve the total deposited solid volume.
83
+ vol = comm.allreduce(float(sv_in.sum()), op=MPI.SUM)
84
+ r32 = float(np.float32(R)) # the kernel deposits with the float32-cast radius
85
+ vol_exact = gpos.shape[0] * (4.0 / 3.0) * np.pi * r32 ** 3
86
+ cons_err = abs(vol - vol_exact) / vol_exact
87
+
88
+ # 2) gather the global eps and compare with the np=1 reference.
89
+ blocks = comm.gather(((ox, oy, oz), np.ascontiguousarray(ep_in)), root=0)
90
+ ok = cons_err < 1e-12
91
+ if rank == 0:
92
+ geps = np.zeros((N, N, N))
93
+ for (bx, by, bz), b in blocks:
94
+ geps[bx:bx + b.shape[0], by:by + b.shape[1], bz:bz + b.shape[2]] = b
95
+ if size == 1:
96
+ np.save(REF_FILE, geps)
97
+ tag = "reference written"
98
+ elif os.path.exists(REF_FILE):
99
+ ref = np.load(REF_FILE)
100
+ err = float(np.max(np.abs(geps - ref)))
101
+ ok = ok and err < 1e-12
102
+ tag = f"vs np=1 max|deps|={err:.3e}"
103
+ else:
104
+ tag = "NO REFERENCE (run np=1 first)"
105
+ ok = False
106
+ print(f"[np={size}] conservation rel-err={cons_err:.3e} {tag}")
107
+ print(f"MPI SMOOTHING (np={size}): {'PASS' if ok else 'FAIL'}")
108
+ ok = comm.bcast(ok if rank == 0 else None, root=0)
109
+ if not ok:
110
+ raise SystemExit(1)
111
+
112
+
113
+ if __name__ == "__main__":
114
+ run(MPI.COMM_WORLD)
@@ -27,7 +27,8 @@ def terminal(drag, g=1e-3, r=1.0, mu=1.0, rho_p=1.0, rho_f=1.0, N=32, steps=120)
27
27
  d.set_gravity(0, 0, -g) # acceleration
28
28
  d.set_positions(np.array([[N / 2, N / 2, N / 2, 1.0 / m_p]], dtype=np.float32)) # w = invMass
29
29
  d.set_velocities(np.zeros((1, 3), dtype=np.float32))
30
- cpl = CfdDem(s, d, fluid_dt=0.1, mu=mu, rho=rho_f, radius=r, drag=drag, dem_substeps=10)
30
+ cpl = CfdDem(s, d, fluid_dt=0.1, mu=mu, rho=rho_f, radius=r, drag=drag, dem_substeps=10,
31
+ porous=False, gravity=(0, 0, -g))
31
32
  slip_hist = []
32
33
  for _ in range(steps):
33
34
  cpl.step()
File without changes