simpok 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
simpok/__init__.py ADDED
@@ -0,0 +1,6 @@
1
+ from .device import Device, get_device
2
+ from ._result import Result
3
+ from .tdgl import TDGL
4
+ from .chc import CHC
5
+
6
+ __all__ = ["Device", "get_device", "TDGL", "CHC", "Result"]
@@ -0,0 +1,3 @@
1
+ from ._utils import get_device
2
+ from ._tdgl import run_tdgl
3
+ from ._chc import run_chc
@@ -0,0 +1,336 @@
1
+ from std.math import ceildiv, sqrt
2
+ from std.memory import unsafe_memcpy
3
+ from std.memory.alloc import alloc, Layout
4
+ from std.sys import has_accelerator, size_of
5
+ from std.sys.info import has_apple_gpu_accelerator
6
+ from std.gpu import global_idx
7
+ from max.gpu.host import DeviceContext
8
+ from layout import TileTensor, TensorLayout, row_major
9
+
10
+ from ._utils import gaussian, TPB, DEVICE_ACCEL, DEVICE_CPU
11
+
12
+
13
+ def _mu[dt: DType](
14
+ c: Scalar[dt],
15
+ xp: Scalar[dt],
16
+ xm: Scalar[dt],
17
+ yp: Scalar[dt],
18
+ ym: Scalar[dt],
19
+ inv_dx2: Scalar[dt],
20
+ ) -> Scalar[dt]:
21
+ return -c + c * c * c - (xp + xm + yp + ym - 4.0 * c) * inv_dx2
22
+
23
+
24
+ def _update[dt: DType](
25
+ c: Scalar[dt],
26
+ xp: Scalar[dt],
27
+ xm: Scalar[dt],
28
+ yp: Scalar[dt],
29
+ ym: Scalar[dt],
30
+ xpp: Scalar[dt],
31
+ xmm: Scalar[dt],
32
+ ypp: Scalar[dt],
33
+ ymm: Scalar[dt],
34
+ pp: Scalar[dt],
35
+ pm: Scalar[dt],
36
+ mp: Scalar[dt],
37
+ mm: Scalar[dt],
38
+ step_dt: Scalar[dt],
39
+ inv_dx2: Scalar[dt],
40
+ ) -> Scalar[dt]:
41
+ var mc = _mu[dt](c, xp, xm, yp, ym, inv_dx2)
42
+ var mxp = _mu[dt](xp, xpp, c, pp, pm, inv_dx2)
43
+ var mxm = _mu[dt](xm, c, xmm, mp, mm, inv_dx2)
44
+ var myp = _mu[dt](yp, pp, mp, ypp, c, inv_dx2)
45
+ var mym = _mu[dt](ym, pm, mm, c, ymm, inv_dx2)
46
+ return c + step_dt * (mxp + mxm + myp + mym - 4.0 * mc) * inv_dx2
47
+
48
+
49
+ def _divnoise[dt: DType](
50
+ n: Int, i: Int, j: Int, im: Int, jm: Int, seed: UInt64, step: UInt64
51
+ ) -> Scalar[dt]:
52
+ var here = UInt64(2 * (i * n + j))
53
+ return (
54
+ gaussian[dt](seed, step, here)
55
+ - gaussian[dt](seed, step, UInt64(2 * (i * n + jm)))
56
+ + gaussian[dt](seed, step, here + 1)
57
+ - gaussian[dt](seed, step, UInt64(2 * (im * n + j) + 1))
58
+ )
59
+
60
+
61
+ def _step_kernel[dt: DType, LT: TensorLayout](
62
+ psi: TileTensor[dt, LT, MutAnyOrigin],
63
+ nxt: TileTensor[dt, LT, MutAnyOrigin],
64
+ n: Int32,
65
+ step_dt: Scalar[dt],
66
+ inv_dx2: Scalar[dt],
67
+ noise_amp: Scalar[dt],
68
+ seed: UInt64,
69
+ step: UInt64,
70
+ ):
71
+ comptime assert psi.flat_rank == 2, "field must be 2d"
72
+ comptime assert nxt.flat_rank == 2, "field must be 2d"
73
+ var nn = Int(n)
74
+ var j = Int(global_idx.x)
75
+ var i = Int(global_idx.y)
76
+ if i >= nn or j >= nn:
77
+ return
78
+
79
+ var ip = i + 1 if i + 1 < nn else i + 1 - nn
80
+ var im = i - 1 if i >= 1 else i - 1 + nn
81
+ var jp = j + 1 if j + 1 < nn else j + 1 - nn
82
+ var jm = j - 1 if j >= 1 else j - 1 + nn
83
+ var ipp = i + 2 if i + 2 < nn else i + 2 - nn
84
+ var imm = i - 2 if i >= 2 else i - 2 + nn
85
+ var jpp = j + 2 if j + 2 < nn else j + 2 - nn
86
+ var jmm = j - 2 if j >= 2 else j - 2 + nn
87
+
88
+ var v = _update[dt](
89
+ rebind[Scalar[dt]](psi[i, j]),
90
+ rebind[Scalar[dt]](psi[ip, j]),
91
+ rebind[Scalar[dt]](psi[im, j]),
92
+ rebind[Scalar[dt]](psi[i, jp]),
93
+ rebind[Scalar[dt]](psi[i, jm]),
94
+ rebind[Scalar[dt]](psi[ipp, j]),
95
+ rebind[Scalar[dt]](psi[imm, j]),
96
+ rebind[Scalar[dt]](psi[i, jpp]),
97
+ rebind[Scalar[dt]](psi[i, jmm]),
98
+ rebind[Scalar[dt]](psi[ip, jp]),
99
+ rebind[Scalar[dt]](psi[ip, jm]),
100
+ rebind[Scalar[dt]](psi[im, jp]),
101
+ rebind[Scalar[dt]](psi[im, jm]),
102
+ step_dt,
103
+ inv_dx2,
104
+ )
105
+ if noise_amp != 0.0:
106
+ v += noise_amp * _divnoise[dt](nn, i, j, im, jm, seed, step)
107
+ nxt[i, j] = rebind[nxt.ElementType](v)
108
+
109
+
110
+ def _step_cpu[dt: DType](
111
+ psi: Pointer[Scalar[dt], MutAnyOrigin],
112
+ nxt: Pointer[Scalar[dt], MutAnyOrigin],
113
+ n: Int,
114
+ step_dt: Scalar[dt],
115
+ inv_dx2: Scalar[dt],
116
+ noise_amp: Scalar[dt],
117
+ seed: UInt64,
118
+ step: UInt64,
119
+ ):
120
+ for i in range(n):
121
+ var ip = i + 1 if i + 1 < n else i + 1 - n
122
+ var im = i - 1 if i >= 1 else i - 1 + n
123
+ var ipp = i + 2 if i + 2 < n else i + 2 - n
124
+ var imm = i - 2 if i >= 2 else i - 2 + n
125
+ for j in range(n):
126
+ var jp = j + 1 if j + 1 < n else j + 1 - n
127
+ var jm = j - 1 if j >= 1 else j - 1 + n
128
+ var jpp = j + 2 if j + 2 < n else j + 2 - n
129
+ var jmm = j - 2 if j >= 2 else j - 2 + n
130
+ var v = _update[dt](
131
+ psi[unsafe_offset=i * n + j],
132
+ psi[unsafe_offset=ip * n + j],
133
+ psi[unsafe_offset=im * n + j],
134
+ psi[unsafe_offset=i * n + jp],
135
+ psi[unsafe_offset=i * n + jm],
136
+ psi[unsafe_offset=ipp * n + j],
137
+ psi[unsafe_offset=imm * n + j],
138
+ psi[unsafe_offset=i * n + jpp],
139
+ psi[unsafe_offset=i * n + jmm],
140
+ psi[unsafe_offset=ip * n + jp],
141
+ psi[unsafe_offset=ip * n + jm],
142
+ psi[unsafe_offset=im * n + jp],
143
+ psi[unsafe_offset=im * n + jm],
144
+ step_dt,
145
+ inv_dx2,
146
+ )
147
+ if noise_amp != 0.0:
148
+ v += noise_amp * _divnoise[dt](n, i, j, im, jm, seed, step)
149
+ nxt[unsafe_offset=i * n + j] = v
150
+
151
+
152
+ def _run_cpu[dt: DType](
153
+ field: Pointer[Scalar[dt], MutAnyOrigin],
154
+ snaps_addr: Int,
155
+ n: Int,
156
+ nsteps: Int,
157
+ nevery: Int,
158
+ step_dt: Scalar[dt],
159
+ inv_dx2: Scalar[dt],
160
+ noise_amp: Scalar[dt],
161
+ seed: UInt64,
162
+ step0: Int,
163
+ ) raises -> String:
164
+ comptime esize = size_of[Scalar[dt]]()
165
+ var size = n * n
166
+ var owned_a = alloc(Layout[Scalar[dt]](count=size)).into_managed()
167
+ var owned_b = alloc(Layout[Scalar[dt]](count=size)).into_managed()
168
+ var a = Pointer[Scalar[dt], MutAnyOrigin](
169
+ unsafe_from_address=Int(owned_a.unsafe_ptr())
170
+ )
171
+ var b = Pointer[Scalar[dt], MutAnyOrigin](
172
+ unsafe_from_address=Int(owned_b.unsafe_ptr())
173
+ )
174
+ unsafe_memcpy(dest=a, src=field, count=size)
175
+
176
+ var k = 0
177
+ for s in range(1, nsteps + 1):
178
+ _step_cpu[dt](
179
+ a, b, n, step_dt, inv_dx2, noise_amp, seed, UInt64(step0 + s - 1),
180
+ )
181
+ swap(a, b)
182
+ if s % nevery == 0:
183
+ var dst = Pointer[Scalar[dt], MutAnyOrigin](
184
+ unsafe_from_address=snaps_addr + k * size * esize
185
+ )
186
+ unsafe_memcpy(dest=dst, src=a, count=size)
187
+ k += 1
188
+
189
+ unsafe_memcpy(dest=field, src=a, count=size)
190
+ _ = owned_a^
191
+ _ = owned_b^
192
+ return String("cpu")
193
+
194
+
195
+ def _run_accel[dt: DType](
196
+ field: Pointer[Scalar[dt], MutAnyOrigin],
197
+ snaps_addr: Int,
198
+ n: Int,
199
+ nsteps: Int,
200
+ nevery: Int,
201
+ step_dt: Scalar[dt],
202
+ inv_dx2: Scalar[dt],
203
+ noise_amp: Scalar[dt],
204
+ seed: UInt64,
205
+ step0: Int,
206
+ ) raises -> String:
207
+ comptime esize = size_of[Scalar[dt]]()
208
+ var size = n * n
209
+ var ctx = DeviceContext()
210
+ var a = ctx.enqueue_create_buffer[dt](size)
211
+ var b = ctx.enqueue_create_buffer[dt](size)
212
+ ctx.enqueue_copy(dst_buf=a, src_ptr=field)
213
+
214
+ var layout = row_major(n, n)
215
+ comptime kern = _step_kernel[dt, type_of(layout)]
216
+ var grid = (ceildiv(n, TPB), ceildiv(n, TPB))
217
+
218
+ var k = 0
219
+ for s in range(1, nsteps + 1):
220
+ ctx.enqueue_function[kern](
221
+ TileTensor(a, layout),
222
+ TileTensor(b, layout),
223
+ Int32(n),
224
+ step_dt,
225
+ inv_dx2,
226
+ noise_amp,
227
+ seed,
228
+ UInt64(step0 + s - 1),
229
+ grid_dim=grid,
230
+ block_dim=(TPB, TPB),
231
+ )
232
+ swap(a, b)
233
+ if s % nevery == 0:
234
+ var dst = Pointer[Scalar[dt], MutAnyOrigin](
235
+ unsafe_from_address=snaps_addr + k * size * esize
236
+ )
237
+ ctx.enqueue_copy(dst_ptr=dst, src_buf=a)
238
+ k += 1
239
+
240
+ ctx.enqueue_copy(dst_ptr=field, src_buf=a)
241
+ ctx.synchronize()
242
+ return String("accelerator")
243
+
244
+
245
+ def _dispatch[dt: DType](
246
+ field_addr: Int,
247
+ snaps_addr: Int,
248
+ n: Int,
249
+ nsteps: Int,
250
+ nevery: Int,
251
+ step_dt: Float64,
252
+ dx: Float64,
253
+ eps: Float64,
254
+ seed: Int,
255
+ step0: Int,
256
+ device: Int,
257
+ ) raises -> String:
258
+ var field = Pointer[Scalar[dt], MutAnyOrigin](
259
+ unsafe_from_address=field_addr
260
+ )
261
+ var inv_dx2 = 1.0 / (dx * dx)
262
+ var amp = sqrt(2.0 * eps * step_dt) * inv_dx2 if eps > 0.0 else 0.0
263
+ var sdt = Scalar[dt](step_dt)
264
+ var sidx = Scalar[dt](inv_dx2)
265
+ var samp = Scalar[dt](amp)
266
+ var useed = UInt64(seed)
267
+
268
+ comptime no_f64_on_accel = (
269
+ has_apple_gpu_accelerator() and dt == DType.float64
270
+ )
271
+
272
+ comptime if not has_accelerator():
273
+ if device == DEVICE_ACCEL:
274
+ raise Error(
275
+ "accelerator requested but this build has no supported"
276
+ " accelerator"
277
+ )
278
+ return _run_cpu[dt](
279
+ field, snaps_addr, n, nsteps, nevery, sdt, sidx, samp, useed, step0,
280
+ )
281
+ else:
282
+ comptime if no_f64_on_accel:
283
+ if device == DEVICE_ACCEL:
284
+ raise Error(
285
+ "accelerator requested but it has no float64"
286
+ " support; use dtype=float32"
287
+ )
288
+ return _run_cpu[dt](
289
+ field, snaps_addr, n, nsteps, nevery, sdt, sidx, samp, useed,
290
+ step0,
291
+ )
292
+ else:
293
+ if device == DEVICE_CPU:
294
+ return _run_cpu[dt](
295
+ field, snaps_addr, n, nsteps, nevery, sdt, sidx, samp,
296
+ useed, step0,
297
+ )
298
+ if DeviceContext.number_of_devices() == 0:
299
+ if device == DEVICE_ACCEL:
300
+ raise Error(
301
+ "accelerator requested but none was detected"
302
+ " at runtime"
303
+ )
304
+ return _run_cpu[dt](
305
+ field, snaps_addr, n, nsteps, nevery, sdt, sidx, samp,
306
+ useed, step0,
307
+ )
308
+ return _run_accel[dt](
309
+ field, snaps_addr, n, nsteps, nevery, sdt, sidx, samp, useed,
310
+ step0,
311
+ )
312
+
313
+
314
+ def run_chc(
315
+ field_addr: Int,
316
+ snaps_addr: Int,
317
+ n: Int,
318
+ nsteps: Int,
319
+ nevery: Int,
320
+ step_dt: Float64,
321
+ dx: Float64,
322
+ eps: Float64,
323
+ seed: Int,
324
+ step0: Int,
325
+ single: Bool,
326
+ device: Int,
327
+ ) raises -> String:
328
+ if single:
329
+ return _dispatch[DType.float32](
330
+ field_addr, snaps_addr, n, nsteps, nevery, step_dt, dx, eps, seed,
331
+ step0, device,
332
+ )
333
+ return _dispatch[DType.float64](
334
+ field_addr, snaps_addr, n, nsteps, nevery, step_dt, dx, eps, seed,
335
+ step0, device,
336
+ )
@@ -0,0 +1,290 @@
1
+ from std.math import ceildiv, sqrt
2
+ from std.memory import unsafe_memcpy
3
+ from std.memory.alloc import alloc, Layout
4
+ from std.sys import has_accelerator, size_of
5
+ from std.sys.info import has_apple_gpu_accelerator
6
+ from std.gpu import global_idx
7
+ from max.gpu.host import DeviceContext
8
+ from layout import TileTensor, TensorLayout, row_major
9
+
10
+ from ._utils import gaussian, TPB, DEVICE_ACCEL, DEVICE_CPU
11
+
12
+
13
+ def _update[dt: DType](
14
+ c: Scalar[dt],
15
+ up: Scalar[dt],
16
+ dn: Scalar[dt],
17
+ lf: Scalar[dt],
18
+ rt: Scalar[dt],
19
+ step_dt: Scalar[dt],
20
+ inv_dx2: Scalar[dt],
21
+ h: Scalar[dt],
22
+ ) -> Scalar[dt]:
23
+ var lap = (up + dn + lf + rt - 4.0 * c) * inv_dx2
24
+ return c + step_dt * (c - c * c * c + h + lap)
25
+
26
+
27
+ def _step_kernel[dt: DType, LT: TensorLayout](
28
+ psi: TileTensor[dt, LT, MutAnyOrigin],
29
+ nxt: TileTensor[dt, LT, MutAnyOrigin],
30
+ n: Int32,
31
+ step_dt: Scalar[dt],
32
+ inv_dx2: Scalar[dt],
33
+ h: Scalar[dt],
34
+ noise_amp: Scalar[dt],
35
+ seed: UInt64,
36
+ step: UInt64,
37
+ ):
38
+ comptime assert psi.flat_rank == 2, "field must be 2d"
39
+ comptime assert nxt.flat_rank == 2, "field must be 2d"
40
+ var nn = Int(n)
41
+ var j = global_idx.x
42
+ var i = global_idx.y
43
+ if i >= nn or j >= nn:
44
+ return
45
+
46
+ var ip = i + 1 if i + 1 < nn else 0
47
+ var im = i - 1 if i > 0 else nn - 1
48
+ var jp = j + 1 if j + 1 < nn else 0
49
+ var jm = j - 1 if j > 0 else nn - 1
50
+
51
+ var v = _update[dt](
52
+ rebind[Scalar[dt]](psi[i, j]),
53
+ rebind[Scalar[dt]](psi[ip, j]),
54
+ rebind[Scalar[dt]](psi[im, j]),
55
+ rebind[Scalar[dt]](psi[i, jp]),
56
+ rebind[Scalar[dt]](psi[i, jm]),
57
+ step_dt,
58
+ inv_dx2,
59
+ h,
60
+ )
61
+ if noise_amp != 0.0:
62
+ v += noise_amp * gaussian[dt](seed, step, UInt64(i * nn + j))
63
+ nxt[i, j] = rebind[nxt.ElementType](v)
64
+
65
+
66
+ def _step_cpu[dt: DType](
67
+ psi: Pointer[Scalar[dt], MutAnyOrigin],
68
+ nxt: Pointer[Scalar[dt], MutAnyOrigin],
69
+ n: Int,
70
+ step_dt: Scalar[dt],
71
+ inv_dx2: Scalar[dt],
72
+ h: Scalar[dt],
73
+ noise_amp: Scalar[dt],
74
+ seed: UInt64,
75
+ step: UInt64,
76
+ ):
77
+ for i in range(n):
78
+ var ip = i + 1 if i + 1 < n else 0
79
+ var im = i - 1 if i > 0 else n - 1
80
+ for j in range(n):
81
+ var jp = j + 1 if j + 1 < n else 0
82
+ var jm = j - 1 if j > 0 else n - 1
83
+ var v = _update[dt](
84
+ psi[unsafe_offset=i * n + j],
85
+ psi[unsafe_offset=ip * n + j],
86
+ psi[unsafe_offset=im * n + j],
87
+ psi[unsafe_offset=i * n + jp],
88
+ psi[unsafe_offset=i * n + jm],
89
+ step_dt,
90
+ inv_dx2,
91
+ h,
92
+ )
93
+ if noise_amp != 0.0:
94
+ v += noise_amp * gaussian[dt](seed, step, UInt64(i * n + j))
95
+ nxt[unsafe_offset=i * n + j] = v
96
+
97
+
98
+ def _run_cpu[dt: DType](
99
+ field: Pointer[Scalar[dt], MutAnyOrigin],
100
+ snaps_addr: Int,
101
+ n: Int,
102
+ nsteps: Int,
103
+ nevery: Int,
104
+ step_dt: Scalar[dt],
105
+ inv_dx2: Scalar[dt],
106
+ h: Scalar[dt],
107
+ noise_amp: Scalar[dt],
108
+ seed: UInt64,
109
+ step0: Int,
110
+ ) raises -> String:
111
+ comptime esize = size_of[Scalar[dt]]()
112
+ var size = n * n
113
+ var owned_a = alloc(Layout[Scalar[dt]](count=size)).into_managed()
114
+ var owned_b = alloc(Layout[Scalar[dt]](count=size)).into_managed()
115
+ var a = Pointer[Scalar[dt], MutAnyOrigin](
116
+ unsafe_from_address=Int(owned_a.unsafe_ptr())
117
+ )
118
+ var b = Pointer[Scalar[dt], MutAnyOrigin](
119
+ unsafe_from_address=Int(owned_b.unsafe_ptr())
120
+ )
121
+ unsafe_memcpy(dest=a, src=field, count=size)
122
+
123
+ var k = 0
124
+ for s in range(1, nsteps + 1):
125
+ _step_cpu[dt](
126
+ a, b, n, step_dt, inv_dx2, h, noise_amp, seed,
127
+ UInt64(step0 + s - 1),
128
+ )
129
+ swap(a, b)
130
+ if s % nevery == 0:
131
+ var dst = Pointer[Scalar[dt], MutAnyOrigin](
132
+ unsafe_from_address=snaps_addr + k * size * esize
133
+ )
134
+ unsafe_memcpy(dest=dst, src=a, count=size)
135
+ k += 1
136
+
137
+ unsafe_memcpy(dest=field, src=a, count=size)
138
+ _ = owned_a^
139
+ _ = owned_b^
140
+ return String("cpu")
141
+
142
+
143
+ def _run_accel[dt: DType](
144
+ field: Pointer[Scalar[dt], MutAnyOrigin],
145
+ snaps_addr: Int,
146
+ n: Int,
147
+ nsteps: Int,
148
+ nevery: Int,
149
+ step_dt: Scalar[dt],
150
+ inv_dx2: Scalar[dt],
151
+ h: Scalar[dt],
152
+ noise_amp: Scalar[dt],
153
+ seed: UInt64,
154
+ step0: Int,
155
+ ) raises -> String:
156
+ comptime esize = size_of[Scalar[dt]]()
157
+ var size = n * n
158
+ var ctx = DeviceContext()
159
+ var a = ctx.enqueue_create_buffer[dt](size)
160
+ var b = ctx.enqueue_create_buffer[dt](size)
161
+ ctx.enqueue_copy(dst_buf=a, src_ptr=field)
162
+
163
+ var layout = row_major(n, n)
164
+ comptime kern = _step_kernel[dt, type_of(layout)]
165
+ var grid = (ceildiv(n, TPB), ceildiv(n, TPB))
166
+
167
+ var k = 0
168
+ for s in range(1, nsteps + 1):
169
+ ctx.enqueue_function[kern](
170
+ TileTensor(a, layout),
171
+ TileTensor(b, layout),
172
+ Int32(n),
173
+ step_dt,
174
+ inv_dx2,
175
+ h,
176
+ noise_amp,
177
+ seed,
178
+ UInt64(step0 + s - 1),
179
+ grid_dim=grid,
180
+ block_dim=(TPB, TPB),
181
+ )
182
+ swap(a, b)
183
+ if s % nevery == 0:
184
+ var dst = Pointer[Scalar[dt], MutAnyOrigin](
185
+ unsafe_from_address=snaps_addr + k * size * esize
186
+ )
187
+ ctx.enqueue_copy(dst_ptr=dst, src_buf=a)
188
+ k += 1
189
+
190
+ ctx.enqueue_copy(dst_ptr=field, src_buf=a)
191
+ ctx.synchronize()
192
+ return String("accelerator")
193
+
194
+
195
+ def _dispatch[dt: DType](
196
+ field_addr: Int,
197
+ snaps_addr: Int,
198
+ n: Int,
199
+ nsteps: Int,
200
+ nevery: Int,
201
+ step_dt: Float64,
202
+ dx: Float64,
203
+ h: Float64,
204
+ eps: Float64,
205
+ seed: Int,
206
+ step0: Int,
207
+ device: Int,
208
+ ) raises -> String:
209
+ var field = Pointer[Scalar[dt], MutAnyOrigin](
210
+ unsafe_from_address=field_addr
211
+ )
212
+ var inv_dx2 = 1.0 / (dx * dx)
213
+ var amp = sqrt(2.0 * eps * step_dt * inv_dx2) if eps > 0.0 else 0.0
214
+ var sdt = Scalar[dt](step_dt)
215
+ var sidx = Scalar[dt](inv_dx2)
216
+ var sh = Scalar[dt](h)
217
+ var samp = Scalar[dt](amp)
218
+ var useed = UInt64(seed)
219
+
220
+ comptime no_f64_on_accel = (
221
+ has_apple_gpu_accelerator() and dt == DType.float64
222
+ )
223
+
224
+ comptime if not has_accelerator():
225
+ if device == DEVICE_ACCEL:
226
+ raise Error(
227
+ "accelerator requested but this build has no supported"
228
+ " accelerator"
229
+ )
230
+ return _run_cpu[dt](
231
+ field, snaps_addr, n, nsteps, nevery, sdt, sidx, sh,
232
+ samp, useed, step0,
233
+ )
234
+ else:
235
+ comptime if no_f64_on_accel:
236
+ if device == DEVICE_ACCEL:
237
+ raise Error(
238
+ "accelerator requested but it has no float64"
239
+ " support; use dtype=float32"
240
+ )
241
+ return _run_cpu[dt](
242
+ field, snaps_addr, n, nsteps, nevery, sdt, sidx, sh,
243
+ samp, useed, step0,
244
+ )
245
+ else:
246
+ if device == DEVICE_CPU:
247
+ return _run_cpu[dt](
248
+ field, snaps_addr, n, nsteps, nevery, sdt, sidx, sh,
249
+ samp, useed, step0,
250
+ )
251
+ if DeviceContext.number_of_devices() == 0:
252
+ if device == DEVICE_ACCEL:
253
+ raise Error(
254
+ "accelerator requested but none was detected"
255
+ " at runtime"
256
+ )
257
+ return _run_cpu[dt](
258
+ field, snaps_addr, n, nsteps, nevery, sdt, sidx, sh,
259
+ samp, useed, step0,
260
+ )
261
+ return _run_accel[dt](
262
+ field, snaps_addr, n, nsteps, nevery, sdt, sidx, sh,
263
+ samp, useed, step0,
264
+ )
265
+
266
+
267
+ def run_tdgl(
268
+ field_addr: Int,
269
+ snaps_addr: Int,
270
+ n: Int,
271
+ nsteps: Int,
272
+ nevery: Int,
273
+ step_dt: Float64,
274
+ dx: Float64,
275
+ h: Float64,
276
+ eps: Float64,
277
+ seed: Int,
278
+ step0: Int,
279
+ single: Bool,
280
+ device: Int,
281
+ ) raises -> String:
282
+ if single:
283
+ return _dispatch[DType.float32](
284
+ field_addr, snaps_addr, n, nsteps, nevery, step_dt, dx, h,
285
+ eps, seed, step0, device,
286
+ )
287
+ return _dispatch[DType.float64](
288
+ field_addr, snaps_addr, n, nsteps, nevery, step_dt, dx, h,
289
+ eps, seed, step0, device,
290
+ )
@@ -0,0 +1,49 @@
1
+ from std.math import sqrt, log, cos
2
+ from std.sys import has_accelerator
3
+ from max.gpu.host import DeviceContext
4
+
5
+ comptime DEVICE_AUTO = 0
6
+ comptime DEVICE_ACCEL = 1
7
+ comptime DEVICE_CPU = 2
8
+
9
+ comptime TPB = 16
10
+ comptime GOLDEN = 0x9E3779B97F4A7C15
11
+ comptime TWO_PI = 6.283185307179586
12
+
13
+
14
+ def get_device() -> Tuple[Bool, String, String]:
15
+ comptime if not has_accelerator():
16
+ return (
17
+ False, String(""), String("no supported accelerator at build time")
18
+ )
19
+ else:
20
+ try:
21
+ if DeviceContext.number_of_devices() == 0:
22
+ return (False, String(""), String("no accelerator detected"))
23
+ var ctx = DeviceContext()
24
+ return (True, ctx.name(), String(""))
25
+ except e:
26
+ return (False, String(""), String(e))
27
+
28
+
29
+ def mix(x: UInt64) -> UInt64:
30
+ var z = x + GOLDEN
31
+ z = (z ^ (z >> 30)) * 0xBF58476D1CE4E5B9
32
+ z = (z ^ (z >> 27)) * 0x94D049BB133111EB
33
+ return z ^ (z >> 31)
34
+
35
+
36
+ def gaussian[dt: DType](
37
+ seed: UInt64, step: UInt64, site: UInt64
38
+ ) -> Scalar[dt]:
39
+ comptime assert dt.is_floating_point(), "dt must be a float type"
40
+ comptime shift: UInt64 = 40 if dt == DType.float32 else 11
41
+ comptime scale = (
42
+ 1.0 / 16777216.0 if dt == DType.float32 else 1.1102230246251565e-16
43
+ )
44
+ var h1 = mix(seed ^ mix(step * GOLDEN ^ site))
45
+ var h2 = mix(h1)
46
+ var u1 = (Scalar[dt](h1 >> shift) + 1.0) * Scalar[dt](scale)
47
+ var u2 = Scalar[dt](h2 >> shift) * Scalar[dt](scale)
48
+ var phase = cos(Float32(TWO_PI) * Float32(u2))
49
+ return sqrt(-2.0 * log(u1)) * Scalar[dt](phase)
simpok/_result.py ADDED
@@ -0,0 +1,55 @@
1
+ from typing import NamedTuple
2
+
3
+ import numpy as np
4
+
5
+ _DEVICE_NAME = None
6
+ _DEVICE_CODES = {"auto": 0, "accelerator": 1, "gpu": 1, "cpu": 2}
7
+
8
+
9
+ class Result(NamedTuple):
10
+ snaps: np.ndarray
11
+ meta: dict
12
+
13
+
14
+ def device_name():
15
+ global _DEVICE_NAME
16
+
17
+ if _DEVICE_NAME is None:
18
+ from .device import get_device
19
+
20
+ _DEVICE_NAME = get_device().name or "cpu"
21
+ return _DEVICE_NAME
22
+
23
+
24
+ def device_code(device):
25
+ try:
26
+ return _DEVICE_CODES[device]
27
+ except (KeyError, TypeError):
28
+ raise ValueError(
29
+ f"device must be one of {sorted(_DEVICE_CODES)}, got {device!r}"
30
+ ) from None
31
+
32
+
33
+ def make_meta(solver, *, n, dx, dt, dtype, seed, amplitude, steps, nevery,
34
+ step_start, step_end, backend, device, elapsed, **physics):
35
+ nsnap = steps // nevery
36
+ meta = {"solver": solver, "n": n, "dx": dx, "dt": dt}
37
+ meta.update(physics)
38
+ meta.update(
39
+ {
40
+ "dtype": np.dtype(dtype).name,
41
+ "seed": seed,
42
+ "amplitude": amplitude,
43
+ "steps": steps,
44
+ "nevery": nevery,
45
+ "nsnap": nsnap,
46
+ "step_start": step_start,
47
+ "step_end": step_end,
48
+ "times": (step_start + np.arange(1, nsnap + 1) * nevery) * dt,
49
+ "backend": backend,
50
+ "device": device_name(),
51
+ "device_request": device,
52
+ "elapsed": elapsed,
53
+ }
54
+ )
55
+ return meta
simpok/chc.py ADDED
@@ -0,0 +1,131 @@
1
+ import time
2
+ import numpy as np
3
+ from ._result import Result, device_code, make_meta
4
+
5
+ class CHC:
6
+ def __init__(self, n=256, dx=1.0, dt=0.01, eps=0.0,
7
+ dtype=np.float64, device="auto"):
8
+ dtype = np.dtype(dtype)
9
+ self._device_code = device_code(device)
10
+ self.device = device
11
+ if dtype not in (np.dtype(np.float32), np.dtype(np.float64)):
12
+ raise ValueError(
13
+ f"dtype must be float32 or float64, got {dtype.name}"
14
+ )
15
+ if int(n) != n or n < 5:
16
+ raise ValueError(f"n must be an integer >= 5, got {n!r}")
17
+ if dx <= 0.0:
18
+ raise ValueError(f"dx must be positive, got {dx!r}")
19
+ if dt <= 0.0:
20
+ raise ValueError(f"dt must be positive, got {dt!r}")
21
+ if eps < 0.0:
22
+ raise ValueError(f"eps must be non-negative, got {eps!r}")
23
+
24
+ q = 8.0 / (dx * dx)
25
+ growth = q * q - q
26
+ if growth > 0.0:
27
+ limit = 2.0 / growth
28
+ if dt > limit:
29
+ raise ValueError(
30
+ f"dt={dt} exceeds the explicit-Euler stability limit "
31
+ f"{limit} for the 2-d 5-point biharmonic at dx={dx}"
32
+ )
33
+
34
+ self.n = int(n)
35
+ self.dx = float(dx)
36
+ self.dt = float(dt)
37
+ self.eps = float(eps)
38
+ self.dtype = dtype
39
+
40
+ self.psi = None
41
+ self.seed = 0
42
+ self.amplitude = 0.0
43
+ self.psi0 = 0.0
44
+ self.step = 0
45
+ self.backend = None
46
+
47
+ def ic(self, seed=0, amplitude=0.01, psi0=0.0):
48
+ if amplitude <= 0.0:
49
+ raise ValueError(f"amplitude must be positive, got {amplitude!r}")
50
+
51
+ rng = np.random.default_rng(seed)
52
+ self.psi = (
53
+ psi0 + rng.uniform(-amplitude, amplitude, size=(self.n, self.n))
54
+ ).astype(self.dtype)
55
+ self.seed = int(seed)
56
+ self.amplitude = float(amplitude)
57
+ self.psi0 = float(psi0)
58
+ self.step = 0
59
+ return self
60
+
61
+ @property
62
+ def t(self):
63
+ return self.step * self.dt
64
+
65
+ def run(self, steps, nevery):
66
+ if self.psi is None:
67
+ raise RuntimeError("call ic() before run()")
68
+ if np.shape(self.psi) != (self.n, self.n):
69
+ raise ValueError(
70
+ f"psi has shape {np.shape(self.psi)}, expected "
71
+ f"{(self.n, self.n)}"
72
+ )
73
+ if int(steps) != steps or steps < 1:
74
+ raise ValueError(f"steps must be a positive integer, got {steps!r}")
75
+ if int(nevery) != nevery or nevery < 1:
76
+ raise ValueError(
77
+ f"nevery must be a positive integer, got {nevery!r}"
78
+ )
79
+ if nevery > steps:
80
+ raise ValueError(f"nevery={nevery} exceeds steps={steps}")
81
+
82
+ steps = int(steps)
83
+ nevery = int(nevery)
84
+ nsnap = steps // nevery
85
+ step_start = self.step
86
+
87
+ import mojo.importer # noqa: F401
88
+
89
+ from .mojo_module import _chc_run
90
+
91
+ field = np.ascontiguousarray(self.psi, dtype=self.dtype)
92
+ snaps = np.empty((nsnap, self.n, self.n), dtype=self.dtype)
93
+ params = (
94
+ self.n,
95
+ steps,
96
+ nevery,
97
+ self.dt,
98
+ self.dx,
99
+ self.eps,
100
+ self.seed,
101
+ self.step,
102
+ self.dtype == np.dtype(np.float32),
103
+ self._device_code,
104
+ )
105
+
106
+ t0 = time.perf_counter()
107
+ self.backend = _chc_run(field, snaps, params)
108
+ elapsed = time.perf_counter() - t0
109
+
110
+ self.psi = field
111
+ self.step += steps
112
+
113
+ meta = make_meta(
114
+ "chc",
115
+ n=self.n,
116
+ dx=self.dx,
117
+ dt=self.dt,
118
+ eps=self.eps,
119
+ psi0=self.psi0,
120
+ dtype=self.dtype,
121
+ seed=self.seed,
122
+ amplitude=self.amplitude,
123
+ steps=steps,
124
+ nevery=nevery,
125
+ step_start=step_start,
126
+ step_end=self.step,
127
+ backend=self.backend,
128
+ device=self.device,
129
+ elapsed=elapsed,
130
+ )
131
+ return Result(snaps, meta)
simpok/device.py ADDED
@@ -0,0 +1,16 @@
1
+ from typing import NamedTuple
2
+
3
+
4
+ class Device(NamedTuple):
5
+ available: bool
6
+ name: str | None
7
+ reason: str | None
8
+
9
+
10
+ def get_device() -> Device:
11
+ import mojo.importer # noqa: F401
12
+
13
+ from .mojo_module import _get_device
14
+
15
+ available, name, reason = _get_device()
16
+ return Device(available, name or None, reason or None)
@@ -0,0 +1,71 @@
1
+ from std.python import PythonObject
2
+ from std.python import Python
3
+ from std.python.bindings import PythonModuleBuilder
4
+ from _core_mojo import get_device, run_tdgl, run_chc
5
+ from std.os import abort
6
+
7
+
8
+ @export
9
+ def PyInit_mojo_module() abi("C") -> PythonObject:
10
+ try:
11
+ var m = PythonModuleBuilder("mojo_module")
12
+ m.def_function[_get_device](
13
+ "_get_device", docstring="Probe the default accelerator"
14
+ )
15
+ m.def_function[_tdgl_run](
16
+ "_tdgl_run", docstring="Evolve a 2d TDGL field in place"
17
+ )
18
+ m.def_function[_chc_run](
19
+ "_chc_run", docstring="Evolve a 2d CHC field in place"
20
+ )
21
+ return m.finalize()
22
+ except e:
23
+ abort(String("error creating Python Mojo module: ", e))
24
+
25
+
26
+ def _get_device() raises -> PythonObject:
27
+ var available, name, reason = get_device()
28
+ return Python.tuple(
29
+ PythonObject(available), PythonObject(name), PythonObject(reason)
30
+ )
31
+
32
+
33
+ def _tdgl_run(
34
+ field: PythonObject, snaps: PythonObject, params: PythonObject
35
+ ) raises -> PythonObject:
36
+ var backend = run_tdgl(
37
+ Int(py=field.ctypes.data),
38
+ Int(py=snaps.ctypes.data),
39
+ Int(py=params[0]),
40
+ Int(py=params[1]),
41
+ Int(py=params[2]),
42
+ Float64(py=params[3]),
43
+ Float64(py=params[4]),
44
+ Float64(py=params[5]),
45
+ Float64(py=params[6]),
46
+ Int(py=params[7]),
47
+ Int(py=params[8]),
48
+ Bool(py=params[9]),
49
+ Int(py=params[10]),
50
+ )
51
+ return PythonObject(backend)
52
+
53
+
54
+ def _chc_run(
55
+ field: PythonObject, snaps: PythonObject, params: PythonObject
56
+ ) raises -> PythonObject:
57
+ var backend = run_chc(
58
+ Int(py=field.ctypes.data),
59
+ Int(py=snaps.ctypes.data),
60
+ Int(py=params[0]),
61
+ Int(py=params[1]),
62
+ Int(py=params[2]),
63
+ Float64(py=params[3]),
64
+ Float64(py=params[4]),
65
+ Float64(py=params[5]),
66
+ Int(py=params[6]),
67
+ Int(py=params[7]),
68
+ Bool(py=params[8]),
69
+ Int(py=params[9]),
70
+ )
71
+ return PythonObject(backend)
simpok/tdgl.py ADDED
@@ -0,0 +1,128 @@
1
+ import time
2
+ import numpy as np
3
+ from ._result import Result, device_code, make_meta
4
+
5
+ class TDGL:
6
+ def __init__(self, n=256, dx=1.0, dt=0.1, h=0.0, eps=0.0,
7
+ dtype=np.float64, device="auto"):
8
+ dtype = np.dtype(dtype)
9
+ self._device_code = device_code(device)
10
+ self.device = device
11
+ if dtype not in (np.dtype(np.float32), np.dtype(np.float64)):
12
+ raise ValueError(
13
+ f"dtype must be float32 or float64, got {dtype.name}"
14
+ )
15
+ if int(n) != n or n < 3:
16
+ raise ValueError(f"n must be an integer >= 3, got {n!r}")
17
+ if dx <= 0.0:
18
+ raise ValueError(f"dx must be positive, got {dx!r}")
19
+ if dt <= 0.0:
20
+ raise ValueError(f"dt must be positive, got {dt!r}")
21
+ if eps < 0.0:
22
+ raise ValueError(f"eps must be non-negative, got {eps!r}")
23
+
24
+ limit = 0.25 * dx * dx
25
+ if dt > limit:
26
+ raise ValueError(
27
+ f"dt={dt} exceeds the explicit-Euler stability limit "
28
+ f"dx^2/4={limit} for the 2-d 5-point Laplacian"
29
+ )
30
+
31
+ self.n = int(n)
32
+ self.dx = float(dx)
33
+ self.dt = float(dt)
34
+ self.h = float(h)
35
+ self.eps = float(eps)
36
+ self.dtype = dtype
37
+
38
+ self.psi = None
39
+ self.seed = 0
40
+ self.amplitude = 0.0
41
+ self.step = 0
42
+ self.backend = None
43
+
44
+ def ic(self, seed=0, amplitude=0.01):
45
+ if amplitude <= 0.0:
46
+ raise ValueError(f"amplitude must be positive, got {amplitude!r}")
47
+
48
+ rng = np.random.default_rng(seed)
49
+ self.psi = rng.uniform(
50
+ -amplitude, amplitude, size=(self.n, self.n)
51
+ ).astype(self.dtype)
52
+ self.seed = int(seed)
53
+ self.amplitude = float(amplitude)
54
+ self.step = 0
55
+ return self
56
+
57
+ @property
58
+ def t(self):
59
+ return self.step * self.dt
60
+
61
+ def run(self, steps, nevery):
62
+ if self.psi is None:
63
+ raise RuntimeError("call ic() before run()")
64
+ if np.shape(self.psi) != (self.n, self.n):
65
+ raise ValueError(
66
+ f"psi has shape {np.shape(self.psi)}, expected "
67
+ f"{(self.n, self.n)}"
68
+ )
69
+ if int(steps) != steps or steps < 1:
70
+ raise ValueError(f"steps must be a positive integer, got {steps!r}")
71
+ if int(nevery) != nevery or nevery < 1:
72
+ raise ValueError(
73
+ f"nevery must be a positive integer, got {nevery!r}"
74
+ )
75
+ if nevery > steps:
76
+ raise ValueError(f"nevery={nevery} exceeds steps={steps}")
77
+
78
+ steps = int(steps)
79
+ nevery = int(nevery)
80
+ nsnap = steps // nevery
81
+ step_start = self.step
82
+
83
+ import mojo.importer # noqa: F401
84
+
85
+ from .mojo_module import _tdgl_run
86
+
87
+ field = np.ascontiguousarray(self.psi, dtype=self.dtype)
88
+ snaps = np.empty((nsnap, self.n, self.n), dtype=self.dtype)
89
+ params = (
90
+ self.n,
91
+ steps,
92
+ nevery,
93
+ self.dt,
94
+ self.dx,
95
+ self.h,
96
+ self.eps,
97
+ self.seed,
98
+ self.step,
99
+ self.dtype == np.dtype(np.float32),
100
+ self._device_code,
101
+ )
102
+
103
+ t0 = time.perf_counter()
104
+ self.backend = _tdgl_run(field, snaps, params)
105
+ elapsed = time.perf_counter() - t0
106
+
107
+ self.psi = field
108
+ self.step += steps
109
+
110
+ meta = make_meta(
111
+ "tdgl",
112
+ n=self.n,
113
+ dx=self.dx,
114
+ dt=self.dt,
115
+ h=self.h,
116
+ eps=self.eps,
117
+ dtype=self.dtype,
118
+ seed=self.seed,
119
+ amplitude=self.amplitude,
120
+ steps=steps,
121
+ nevery=nevery,
122
+ step_start=step_start,
123
+ step_end=self.step,
124
+ backend=self.backend,
125
+ device=self.device,
126
+ elapsed=elapsed,
127
+ )
128
+ return Result(snaps, meta)
@@ -0,0 +1,173 @@
1
+ Metadata-Version: 2.5
2
+ Name: simpok
3
+ Version: 0.1.0
4
+ Summary: High-level Python API for phase-ordering kinetics, with the computation in Mojo
5
+ Project-URL: Homepage, https://github.com/ivijayyadav/simpok
6
+ Project-URL: Repository, https://github.com/ivijayyadav/simpok
7
+ Project-URL: Issues, https://github.com/ivijayyadav/simpok/issues
8
+ Author-email: vijay <vijpiml@gmail.com>
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: cahn-hilliard,ginzburg-landau,gpu,mojo,phase-field,phase-ordering,spinodal-decomposition
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: Operating System :: MacOS :: MacOS X
15
+ Classifier: Operating System :: POSIX :: Linux
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Scientific/Engineering :: Physics
22
+ Requires-Python: >=3.10
23
+ Requires-Dist: max<27,>=26.5
24
+ Requires-Dist: mojo<2,>=1.0
25
+ Requires-Dist: numpy>=1.24
26
+ Provides-Extra: examples
27
+ Requires-Dist: matplotlib>=3.8; extra == 'examples'
28
+ Description-Content-Type: text/markdown
29
+
30
+ # simpok
31
+
32
+ Phase-ordering kinetics in Python, with the computation in [Mojo](https://www.modular.com/mojo).
33
+
34
+ You write ordinary Python; the solvers run on your GPU if you have one, and on the CPU
35
+ otherwise, from the same code. The intent is to keep the modelling in a high-level API and push
36
+ the per-site arithmetic down to Mojo.
37
+
38
+ > **Status: alpha.** Two solvers, 2-d only, 5-point Laplacian, periodic boundaries. The API may
39
+ > still change.
40
+
41
+ ## Install
42
+
43
+ ```bash
44
+ pip install simpok
45
+ ```
46
+
47
+ Python 3.10+. The Mojo toolchain arrives as a dependency — there is nothing else to install, and
48
+ no compiler to set up by hand.
49
+
50
+ **The first `import simpok` takes about 20 seconds.** It is compiling the Mojo sources for your
51
+ machine; the result is cached and every later import is instant. It recompiles only when the
52
+ package is upgraded. This is deliberate rather than a packaging shortcut: whether an accelerator
53
+ is targeted is decided at Mojo compile time, so building on your machine is what lets one
54
+ universal wheel use whatever hardware you actually have.
55
+
56
+ ## Quick start
57
+
58
+ ```python
59
+ from simpok import TDGL
60
+
61
+ sim = TDGL().ic(seed=0)
62
+ snaps, meta = sim.run(steps=90000, nevery=1000)
63
+
64
+ print(snaps.shape) # (90, 256, 256)
65
+ print(meta["backend"]) # 'accelerator' or 'cpu'
66
+ print(meta["times"][-1]) # 9000.0
67
+ ```
68
+
69
+ `run` returns a `Result` — a named tuple of `snaps` (a `(nsnap, n, n)` NumPy array) and `meta`
70
+ (a dict of every parameter used, plus `times`, `backend`, `device` and `elapsed`).
71
+
72
+ Conserved dynamics works the same way:
73
+
74
+ ```python
75
+ from simpok import CHC
76
+
77
+ snaps, meta = CHC().ic(seed=0, psi0=-0.4).run(steps=50000, nevery=500)
78
+ ```
79
+
80
+ Check what you're running on:
81
+
82
+ ```python
83
+ from simpok import get_device
84
+ print(get_device()) # Device(available=True, name='NVIDIA RTX A5000', reason=None)
85
+ ```
86
+
87
+ ## The models
88
+
89
+ **`TDGL`** — nonconserved scalar order parameter (Model A). Domain coarsening in a quenched
90
+ ferromagnet; the Allen–Cahn growth law `L(t) ~ t^(1/2)`.
91
+
92
+ ```
93
+ ∂ψ/∂t = ψ − ψ³ + h + ∇²ψ + θ
94
+ ```
95
+
96
+ **`CHC`** — conserved scalar order parameter (Model B). Spinodal decomposition in a binary
97
+ mixture; the Lifshitz–Slyozov growth law `L(t) ~ t^(1/3)`. The mean of ψ is conserved exactly,
98
+ so `psi0` sets the composition — `0.0` gives the bicontinuous critical quench, `-0.4` gives
99
+ minority droplets.
100
+
101
+ ```
102
+ ∂ψ/∂t = −∇²(ψ − ψ³ + ∇²ψ) + ∇·θ
103
+ ```
104
+
105
+ Thermal noise is off by default (`eps=0`) — it is asymptotically irrelevant to
106
+ the growth laws. Set `eps > 0` to switch it on; for `CHC` it is applied as a bond-centred current
107
+ so that conservation stays exact to round-off.
108
+
109
+ ## API
110
+
111
+ ```python
112
+ TDGL(n=256, dx=1.0, dt=0.1, h=0.0, eps=0.0, dtype=np.float64, device="auto")
113
+ CHC (n=256, dx=1.0, dt=0.01, eps=0.0, dtype=np.float64, device="auto")
114
+ ```
115
+
116
+ | Method | |
117
+ |---|---|
118
+ | `.ic(seed=0, amplitude=0.01)` | random initial condition; `CHC` also takes `psi0=0.0`. Returns `self`, so it chains. |
119
+ | `.run(steps, nevery)` | evolve `steps` steps, saving every `nevery`. Returns `Result(snaps, meta)`. |
120
+ | `.t` | current simulation time |
121
+ | `.psi` | current field, carried across calls — successive `run` calls continue the trajectory |
122
+
123
+ `device` is `"auto"` (default), `"accelerator"`, or `"cpu"`. `"accelerator"` raises if none is
124
+ usable rather than silently falling back. `dtype` is `float64` (default) or `float32`.
125
+
126
+ Timesteps are checked against the explicit-Euler stability limit at construction — `dt <= dx²/4`
127
+ for `TDGL`, and the much tighter fourth-order bound for `CHC`, which is why its default `dt` is
128
+ ten times smaller.
129
+
130
+ ## Hardware notes
131
+
132
+ - **NVIDIA** — both dtypes work.
133
+ - **Apple Silicon** — Metal has no float64 at all, so use `dtype=np.float32` to run on the GPU.
134
+ A float64 run falls back to the CPU automatically; asking for `device="accelerator"` with
135
+ float64 raises with an explanation.
136
+ - **No GPU** — everything runs on the CPU with identical results. Deterministic (`eps=0`) runs
137
+ are bit-identical between CPU and accelerator.
138
+
139
+ On float32, `CHC` is several times faster than float64 on consumer NVIDIA cards, whose
140
+ double-precision throughput is heavily reduced; the domain length scale agrees with float64 to
141
+ better than 0.01%. `TDGL` at small grids is limited by kernel-launch overhead rather than
142
+ arithmetic, so larger lattices use the GPU far more efficiently than small ones.
143
+
144
+ ## Correctness
145
+
146
+ Both solvers are checked against analytic results, not just for plausibility:
147
+
148
+ - equilibrium interface `tanh(z/√2)`, second-order convergent in `dx`
149
+ - `TDGL`: droplet collapse `dR²/dt = −2(d−1)`; the `t^(1/2)` growth law
150
+ - `CHC`: the Cahn dispersion relation `σ(q) = q − q²` reproduced to 1 part in 10¹⁰; the mean
151
+ conserved to 1 part in 10¹⁷; the `t^(1/3)` growth law
152
+ - conserved noise verified against the discrete fluctuation–dissipation relation
153
+ - seed reproducibility, resumability, and CPU/accelerator agreement
154
+
155
+ ## Examples
156
+
157
+ `examples/` contains runnable scripts that produce snapshot figures:
158
+
159
+ ```bash
160
+ python -m examples.tdgl_quickstart # writes tdgl.png
161
+ python -m examples.chc_quickstart # writes chc.png
162
+ ```
163
+
164
+ They need matplotlib: `pip install simpok[examples]`.
165
+
166
+ ## Reference
167
+
168
+ Sanjay Puri, "Kinetics of Phase Transitions", Ch. 1 in *Kinetics of Phase Transitions*,
169
+ S. Puri and V. Wadhawan (eds.), CRC Press (2009).
170
+
171
+ ## License
172
+
173
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,14 @@
1
+ simpok/__init__.py,sha256=qAusnQnv2qQUBzwP3Y_Hdi0OMHiIB8aj55Qg0ih1wuI,172
2
+ simpok/_result.py,sha256=kauuiZy77ug4z21S5BA8PFZEE8W0jCsMk2CyeO4ee1w,1419
3
+ simpok/chc.py,sha256=4rrpw8ODPvQKXmjHdHEImkROoAxh9kWPYrxSmNgxBps,4069
4
+ simpok/device.py,sha256=O2rfjc0QCTyN8Tfc1kjqmQGsKapqv88hz9ou0_SgyIQ,337
5
+ simpok/mojo_module.mojo,sha256=8l4WqOmYdEzVSw_pm0BTkuBdSbY_jtJZt6S3vgWSNew,2050
6
+ simpok/tdgl.py,sha256=BaJh5q3tKe0oQeRMNOXbuMqxOuiiN87b8NOduxZpX18,3937
7
+ simpok/_core_mojo/__init__.mojo,sha256=fa4nrlLtv-mIAF0fJlLI8rggolZWONAFD0VcXZoPsX8,85
8
+ simpok/_core_mojo/_chc.mojo,sha256=6OiZbsTtpD40TxuW8bsPjkUZHdTZH1rejXY4MRi9UuM,10286
9
+ simpok/_core_mojo/_tdgl.mojo,sha256=7AJQlT-I4SKnvg5Ffn-WJ2COr6ZdEhzs1zn0VOUVD8M,8364
10
+ simpok/_core_mojo/_utils.mojo,sha256=Yejb5W7UpBDq79SAdGnYtWoCUpMcUupKsvCou8d7KoU,1582
11
+ simpok-0.1.0.dist-info/METADATA,sha256=BxS5WLovjFjSa0prVEJJdqF-NLuHVvOTPviW27ndl-k,6620
12
+ simpok-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
13
+ simpok-0.1.0.dist-info/licenses/LICENSE,sha256=XunFJgLLY9giZlL2BsiFAlkkhEOA2BFwkvYpuo1k9GU,1068
14
+ simpok-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Vijay Yadav
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.