simpok 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- simpok/__init__.py +6 -0
- simpok/_core_mojo/__init__.mojo +3 -0
- simpok/_core_mojo/_chc.mojo +336 -0
- simpok/_core_mojo/_tdgl.mojo +290 -0
- simpok/_core_mojo/_utils.mojo +49 -0
- simpok/_result.py +55 -0
- simpok/chc.py +131 -0
- simpok/device.py +16 -0
- simpok/mojo_module.mojo +71 -0
- simpok/tdgl.py +128 -0
- simpok-0.1.0.dist-info/METADATA +173 -0
- simpok-0.1.0.dist-info/RECORD +14 -0
- simpok-0.1.0.dist-info/WHEEL +4 -0
- simpok-0.1.0.dist-info/licenses/LICENSE +21 -0
simpok/__init__.py
ADDED
|
@@ -0,0 +1,336 @@
|
|
|
1
|
+
from std.math import ceildiv, sqrt
|
|
2
|
+
from std.memory import unsafe_memcpy
|
|
3
|
+
from std.memory.alloc import alloc, Layout
|
|
4
|
+
from std.sys import has_accelerator, size_of
|
|
5
|
+
from std.sys.info import has_apple_gpu_accelerator
|
|
6
|
+
from std.gpu import global_idx
|
|
7
|
+
from max.gpu.host import DeviceContext
|
|
8
|
+
from layout import TileTensor, TensorLayout, row_major
|
|
9
|
+
|
|
10
|
+
from ._utils import gaussian, TPB, DEVICE_ACCEL, DEVICE_CPU
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _mu[dt: DType](
|
|
14
|
+
c: Scalar[dt],
|
|
15
|
+
xp: Scalar[dt],
|
|
16
|
+
xm: Scalar[dt],
|
|
17
|
+
yp: Scalar[dt],
|
|
18
|
+
ym: Scalar[dt],
|
|
19
|
+
inv_dx2: Scalar[dt],
|
|
20
|
+
) -> Scalar[dt]:
|
|
21
|
+
return -c + c * c * c - (xp + xm + yp + ym - 4.0 * c) * inv_dx2
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _update[dt: DType](
|
|
25
|
+
c: Scalar[dt],
|
|
26
|
+
xp: Scalar[dt],
|
|
27
|
+
xm: Scalar[dt],
|
|
28
|
+
yp: Scalar[dt],
|
|
29
|
+
ym: Scalar[dt],
|
|
30
|
+
xpp: Scalar[dt],
|
|
31
|
+
xmm: Scalar[dt],
|
|
32
|
+
ypp: Scalar[dt],
|
|
33
|
+
ymm: Scalar[dt],
|
|
34
|
+
pp: Scalar[dt],
|
|
35
|
+
pm: Scalar[dt],
|
|
36
|
+
mp: Scalar[dt],
|
|
37
|
+
mm: Scalar[dt],
|
|
38
|
+
step_dt: Scalar[dt],
|
|
39
|
+
inv_dx2: Scalar[dt],
|
|
40
|
+
) -> Scalar[dt]:
|
|
41
|
+
var mc = _mu[dt](c, xp, xm, yp, ym, inv_dx2)
|
|
42
|
+
var mxp = _mu[dt](xp, xpp, c, pp, pm, inv_dx2)
|
|
43
|
+
var mxm = _mu[dt](xm, c, xmm, mp, mm, inv_dx2)
|
|
44
|
+
var myp = _mu[dt](yp, pp, mp, ypp, c, inv_dx2)
|
|
45
|
+
var mym = _mu[dt](ym, pm, mm, c, ymm, inv_dx2)
|
|
46
|
+
return c + step_dt * (mxp + mxm + myp + mym - 4.0 * mc) * inv_dx2
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _divnoise[dt: DType](
|
|
50
|
+
n: Int, i: Int, j: Int, im: Int, jm: Int, seed: UInt64, step: UInt64
|
|
51
|
+
) -> Scalar[dt]:
|
|
52
|
+
var here = UInt64(2 * (i * n + j))
|
|
53
|
+
return (
|
|
54
|
+
gaussian[dt](seed, step, here)
|
|
55
|
+
- gaussian[dt](seed, step, UInt64(2 * (i * n + jm)))
|
|
56
|
+
+ gaussian[dt](seed, step, here + 1)
|
|
57
|
+
- gaussian[dt](seed, step, UInt64(2 * (im * n + j) + 1))
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _step_kernel[dt: DType, LT: TensorLayout](
|
|
62
|
+
psi: TileTensor[dt, LT, MutAnyOrigin],
|
|
63
|
+
nxt: TileTensor[dt, LT, MutAnyOrigin],
|
|
64
|
+
n: Int32,
|
|
65
|
+
step_dt: Scalar[dt],
|
|
66
|
+
inv_dx2: Scalar[dt],
|
|
67
|
+
noise_amp: Scalar[dt],
|
|
68
|
+
seed: UInt64,
|
|
69
|
+
step: UInt64,
|
|
70
|
+
):
|
|
71
|
+
comptime assert psi.flat_rank == 2, "field must be 2d"
|
|
72
|
+
comptime assert nxt.flat_rank == 2, "field must be 2d"
|
|
73
|
+
var nn = Int(n)
|
|
74
|
+
var j = Int(global_idx.x)
|
|
75
|
+
var i = Int(global_idx.y)
|
|
76
|
+
if i >= nn or j >= nn:
|
|
77
|
+
return
|
|
78
|
+
|
|
79
|
+
var ip = i + 1 if i + 1 < nn else i + 1 - nn
|
|
80
|
+
var im = i - 1 if i >= 1 else i - 1 + nn
|
|
81
|
+
var jp = j + 1 if j + 1 < nn else j + 1 - nn
|
|
82
|
+
var jm = j - 1 if j >= 1 else j - 1 + nn
|
|
83
|
+
var ipp = i + 2 if i + 2 < nn else i + 2 - nn
|
|
84
|
+
var imm = i - 2 if i >= 2 else i - 2 + nn
|
|
85
|
+
var jpp = j + 2 if j + 2 < nn else j + 2 - nn
|
|
86
|
+
var jmm = j - 2 if j >= 2 else j - 2 + nn
|
|
87
|
+
|
|
88
|
+
var v = _update[dt](
|
|
89
|
+
rebind[Scalar[dt]](psi[i, j]),
|
|
90
|
+
rebind[Scalar[dt]](psi[ip, j]),
|
|
91
|
+
rebind[Scalar[dt]](psi[im, j]),
|
|
92
|
+
rebind[Scalar[dt]](psi[i, jp]),
|
|
93
|
+
rebind[Scalar[dt]](psi[i, jm]),
|
|
94
|
+
rebind[Scalar[dt]](psi[ipp, j]),
|
|
95
|
+
rebind[Scalar[dt]](psi[imm, j]),
|
|
96
|
+
rebind[Scalar[dt]](psi[i, jpp]),
|
|
97
|
+
rebind[Scalar[dt]](psi[i, jmm]),
|
|
98
|
+
rebind[Scalar[dt]](psi[ip, jp]),
|
|
99
|
+
rebind[Scalar[dt]](psi[ip, jm]),
|
|
100
|
+
rebind[Scalar[dt]](psi[im, jp]),
|
|
101
|
+
rebind[Scalar[dt]](psi[im, jm]),
|
|
102
|
+
step_dt,
|
|
103
|
+
inv_dx2,
|
|
104
|
+
)
|
|
105
|
+
if noise_amp != 0.0:
|
|
106
|
+
v += noise_amp * _divnoise[dt](nn, i, j, im, jm, seed, step)
|
|
107
|
+
nxt[i, j] = rebind[nxt.ElementType](v)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _step_cpu[dt: DType](
|
|
111
|
+
psi: Pointer[Scalar[dt], MutAnyOrigin],
|
|
112
|
+
nxt: Pointer[Scalar[dt], MutAnyOrigin],
|
|
113
|
+
n: Int,
|
|
114
|
+
step_dt: Scalar[dt],
|
|
115
|
+
inv_dx2: Scalar[dt],
|
|
116
|
+
noise_amp: Scalar[dt],
|
|
117
|
+
seed: UInt64,
|
|
118
|
+
step: UInt64,
|
|
119
|
+
):
|
|
120
|
+
for i in range(n):
|
|
121
|
+
var ip = i + 1 if i + 1 < n else i + 1 - n
|
|
122
|
+
var im = i - 1 if i >= 1 else i - 1 + n
|
|
123
|
+
var ipp = i + 2 if i + 2 < n else i + 2 - n
|
|
124
|
+
var imm = i - 2 if i >= 2 else i - 2 + n
|
|
125
|
+
for j in range(n):
|
|
126
|
+
var jp = j + 1 if j + 1 < n else j + 1 - n
|
|
127
|
+
var jm = j - 1 if j >= 1 else j - 1 + n
|
|
128
|
+
var jpp = j + 2 if j + 2 < n else j + 2 - n
|
|
129
|
+
var jmm = j - 2 if j >= 2 else j - 2 + n
|
|
130
|
+
var v = _update[dt](
|
|
131
|
+
psi[unsafe_offset=i * n + j],
|
|
132
|
+
psi[unsafe_offset=ip * n + j],
|
|
133
|
+
psi[unsafe_offset=im * n + j],
|
|
134
|
+
psi[unsafe_offset=i * n + jp],
|
|
135
|
+
psi[unsafe_offset=i * n + jm],
|
|
136
|
+
psi[unsafe_offset=ipp * n + j],
|
|
137
|
+
psi[unsafe_offset=imm * n + j],
|
|
138
|
+
psi[unsafe_offset=i * n + jpp],
|
|
139
|
+
psi[unsafe_offset=i * n + jmm],
|
|
140
|
+
psi[unsafe_offset=ip * n + jp],
|
|
141
|
+
psi[unsafe_offset=ip * n + jm],
|
|
142
|
+
psi[unsafe_offset=im * n + jp],
|
|
143
|
+
psi[unsafe_offset=im * n + jm],
|
|
144
|
+
step_dt,
|
|
145
|
+
inv_dx2,
|
|
146
|
+
)
|
|
147
|
+
if noise_amp != 0.0:
|
|
148
|
+
v += noise_amp * _divnoise[dt](n, i, j, im, jm, seed, step)
|
|
149
|
+
nxt[unsafe_offset=i * n + j] = v
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _run_cpu[dt: DType](
|
|
153
|
+
field: Pointer[Scalar[dt], MutAnyOrigin],
|
|
154
|
+
snaps_addr: Int,
|
|
155
|
+
n: Int,
|
|
156
|
+
nsteps: Int,
|
|
157
|
+
nevery: Int,
|
|
158
|
+
step_dt: Scalar[dt],
|
|
159
|
+
inv_dx2: Scalar[dt],
|
|
160
|
+
noise_amp: Scalar[dt],
|
|
161
|
+
seed: UInt64,
|
|
162
|
+
step0: Int,
|
|
163
|
+
) raises -> String:
|
|
164
|
+
comptime esize = size_of[Scalar[dt]]()
|
|
165
|
+
var size = n * n
|
|
166
|
+
var owned_a = alloc(Layout[Scalar[dt]](count=size)).into_managed()
|
|
167
|
+
var owned_b = alloc(Layout[Scalar[dt]](count=size)).into_managed()
|
|
168
|
+
var a = Pointer[Scalar[dt], MutAnyOrigin](
|
|
169
|
+
unsafe_from_address=Int(owned_a.unsafe_ptr())
|
|
170
|
+
)
|
|
171
|
+
var b = Pointer[Scalar[dt], MutAnyOrigin](
|
|
172
|
+
unsafe_from_address=Int(owned_b.unsafe_ptr())
|
|
173
|
+
)
|
|
174
|
+
unsafe_memcpy(dest=a, src=field, count=size)
|
|
175
|
+
|
|
176
|
+
var k = 0
|
|
177
|
+
for s in range(1, nsteps + 1):
|
|
178
|
+
_step_cpu[dt](
|
|
179
|
+
a, b, n, step_dt, inv_dx2, noise_amp, seed, UInt64(step0 + s - 1),
|
|
180
|
+
)
|
|
181
|
+
swap(a, b)
|
|
182
|
+
if s % nevery == 0:
|
|
183
|
+
var dst = Pointer[Scalar[dt], MutAnyOrigin](
|
|
184
|
+
unsafe_from_address=snaps_addr + k * size * esize
|
|
185
|
+
)
|
|
186
|
+
unsafe_memcpy(dest=dst, src=a, count=size)
|
|
187
|
+
k += 1
|
|
188
|
+
|
|
189
|
+
unsafe_memcpy(dest=field, src=a, count=size)
|
|
190
|
+
_ = owned_a^
|
|
191
|
+
_ = owned_b^
|
|
192
|
+
return String("cpu")
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _run_accel[dt: DType](
|
|
196
|
+
field: Pointer[Scalar[dt], MutAnyOrigin],
|
|
197
|
+
snaps_addr: Int,
|
|
198
|
+
n: Int,
|
|
199
|
+
nsteps: Int,
|
|
200
|
+
nevery: Int,
|
|
201
|
+
step_dt: Scalar[dt],
|
|
202
|
+
inv_dx2: Scalar[dt],
|
|
203
|
+
noise_amp: Scalar[dt],
|
|
204
|
+
seed: UInt64,
|
|
205
|
+
step0: Int,
|
|
206
|
+
) raises -> String:
|
|
207
|
+
comptime esize = size_of[Scalar[dt]]()
|
|
208
|
+
var size = n * n
|
|
209
|
+
var ctx = DeviceContext()
|
|
210
|
+
var a = ctx.enqueue_create_buffer[dt](size)
|
|
211
|
+
var b = ctx.enqueue_create_buffer[dt](size)
|
|
212
|
+
ctx.enqueue_copy(dst_buf=a, src_ptr=field)
|
|
213
|
+
|
|
214
|
+
var layout = row_major(n, n)
|
|
215
|
+
comptime kern = _step_kernel[dt, type_of(layout)]
|
|
216
|
+
var grid = (ceildiv(n, TPB), ceildiv(n, TPB))
|
|
217
|
+
|
|
218
|
+
var k = 0
|
|
219
|
+
for s in range(1, nsteps + 1):
|
|
220
|
+
ctx.enqueue_function[kern](
|
|
221
|
+
TileTensor(a, layout),
|
|
222
|
+
TileTensor(b, layout),
|
|
223
|
+
Int32(n),
|
|
224
|
+
step_dt,
|
|
225
|
+
inv_dx2,
|
|
226
|
+
noise_amp,
|
|
227
|
+
seed,
|
|
228
|
+
UInt64(step0 + s - 1),
|
|
229
|
+
grid_dim=grid,
|
|
230
|
+
block_dim=(TPB, TPB),
|
|
231
|
+
)
|
|
232
|
+
swap(a, b)
|
|
233
|
+
if s % nevery == 0:
|
|
234
|
+
var dst = Pointer[Scalar[dt], MutAnyOrigin](
|
|
235
|
+
unsafe_from_address=snaps_addr + k * size * esize
|
|
236
|
+
)
|
|
237
|
+
ctx.enqueue_copy(dst_ptr=dst, src_buf=a)
|
|
238
|
+
k += 1
|
|
239
|
+
|
|
240
|
+
ctx.enqueue_copy(dst_ptr=field, src_buf=a)
|
|
241
|
+
ctx.synchronize()
|
|
242
|
+
return String("accelerator")
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _dispatch[dt: DType](
|
|
246
|
+
field_addr: Int,
|
|
247
|
+
snaps_addr: Int,
|
|
248
|
+
n: Int,
|
|
249
|
+
nsteps: Int,
|
|
250
|
+
nevery: Int,
|
|
251
|
+
step_dt: Float64,
|
|
252
|
+
dx: Float64,
|
|
253
|
+
eps: Float64,
|
|
254
|
+
seed: Int,
|
|
255
|
+
step0: Int,
|
|
256
|
+
device: Int,
|
|
257
|
+
) raises -> String:
|
|
258
|
+
var field = Pointer[Scalar[dt], MutAnyOrigin](
|
|
259
|
+
unsafe_from_address=field_addr
|
|
260
|
+
)
|
|
261
|
+
var inv_dx2 = 1.0 / (dx * dx)
|
|
262
|
+
var amp = sqrt(2.0 * eps * step_dt) * inv_dx2 if eps > 0.0 else 0.0
|
|
263
|
+
var sdt = Scalar[dt](step_dt)
|
|
264
|
+
var sidx = Scalar[dt](inv_dx2)
|
|
265
|
+
var samp = Scalar[dt](amp)
|
|
266
|
+
var useed = UInt64(seed)
|
|
267
|
+
|
|
268
|
+
comptime no_f64_on_accel = (
|
|
269
|
+
has_apple_gpu_accelerator() and dt == DType.float64
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
comptime if not has_accelerator():
|
|
273
|
+
if device == DEVICE_ACCEL:
|
|
274
|
+
raise Error(
|
|
275
|
+
"accelerator requested but this build has no supported"
|
|
276
|
+
" accelerator"
|
|
277
|
+
)
|
|
278
|
+
return _run_cpu[dt](
|
|
279
|
+
field, snaps_addr, n, nsteps, nevery, sdt, sidx, samp, useed, step0,
|
|
280
|
+
)
|
|
281
|
+
else:
|
|
282
|
+
comptime if no_f64_on_accel:
|
|
283
|
+
if device == DEVICE_ACCEL:
|
|
284
|
+
raise Error(
|
|
285
|
+
"accelerator requested but it has no float64"
|
|
286
|
+
" support; use dtype=float32"
|
|
287
|
+
)
|
|
288
|
+
return _run_cpu[dt](
|
|
289
|
+
field, snaps_addr, n, nsteps, nevery, sdt, sidx, samp, useed,
|
|
290
|
+
step0,
|
|
291
|
+
)
|
|
292
|
+
else:
|
|
293
|
+
if device == DEVICE_CPU:
|
|
294
|
+
return _run_cpu[dt](
|
|
295
|
+
field, snaps_addr, n, nsteps, nevery, sdt, sidx, samp,
|
|
296
|
+
useed, step0,
|
|
297
|
+
)
|
|
298
|
+
if DeviceContext.number_of_devices() == 0:
|
|
299
|
+
if device == DEVICE_ACCEL:
|
|
300
|
+
raise Error(
|
|
301
|
+
"accelerator requested but none was detected"
|
|
302
|
+
" at runtime"
|
|
303
|
+
)
|
|
304
|
+
return _run_cpu[dt](
|
|
305
|
+
field, snaps_addr, n, nsteps, nevery, sdt, sidx, samp,
|
|
306
|
+
useed, step0,
|
|
307
|
+
)
|
|
308
|
+
return _run_accel[dt](
|
|
309
|
+
field, snaps_addr, n, nsteps, nevery, sdt, sidx, samp, useed,
|
|
310
|
+
step0,
|
|
311
|
+
)
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def run_chc(
|
|
315
|
+
field_addr: Int,
|
|
316
|
+
snaps_addr: Int,
|
|
317
|
+
n: Int,
|
|
318
|
+
nsteps: Int,
|
|
319
|
+
nevery: Int,
|
|
320
|
+
step_dt: Float64,
|
|
321
|
+
dx: Float64,
|
|
322
|
+
eps: Float64,
|
|
323
|
+
seed: Int,
|
|
324
|
+
step0: Int,
|
|
325
|
+
single: Bool,
|
|
326
|
+
device: Int,
|
|
327
|
+
) raises -> String:
|
|
328
|
+
if single:
|
|
329
|
+
return _dispatch[DType.float32](
|
|
330
|
+
field_addr, snaps_addr, n, nsteps, nevery, step_dt, dx, eps, seed,
|
|
331
|
+
step0, device,
|
|
332
|
+
)
|
|
333
|
+
return _dispatch[DType.float64](
|
|
334
|
+
field_addr, snaps_addr, n, nsteps, nevery, step_dt, dx, eps, seed,
|
|
335
|
+
step0, device,
|
|
336
|
+
)
|
|
@@ -0,0 +1,290 @@
|
|
|
1
|
+
from std.math import ceildiv, sqrt
|
|
2
|
+
from std.memory import unsafe_memcpy
|
|
3
|
+
from std.memory.alloc import alloc, Layout
|
|
4
|
+
from std.sys import has_accelerator, size_of
|
|
5
|
+
from std.sys.info import has_apple_gpu_accelerator
|
|
6
|
+
from std.gpu import global_idx
|
|
7
|
+
from max.gpu.host import DeviceContext
|
|
8
|
+
from layout import TileTensor, TensorLayout, row_major
|
|
9
|
+
|
|
10
|
+
from ._utils import gaussian, TPB, DEVICE_ACCEL, DEVICE_CPU
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _update[dt: DType](
|
|
14
|
+
c: Scalar[dt],
|
|
15
|
+
up: Scalar[dt],
|
|
16
|
+
dn: Scalar[dt],
|
|
17
|
+
lf: Scalar[dt],
|
|
18
|
+
rt: Scalar[dt],
|
|
19
|
+
step_dt: Scalar[dt],
|
|
20
|
+
inv_dx2: Scalar[dt],
|
|
21
|
+
h: Scalar[dt],
|
|
22
|
+
) -> Scalar[dt]:
|
|
23
|
+
var lap = (up + dn + lf + rt - 4.0 * c) * inv_dx2
|
|
24
|
+
return c + step_dt * (c - c * c * c + h + lap)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _step_kernel[dt: DType, LT: TensorLayout](
|
|
28
|
+
psi: TileTensor[dt, LT, MutAnyOrigin],
|
|
29
|
+
nxt: TileTensor[dt, LT, MutAnyOrigin],
|
|
30
|
+
n: Int32,
|
|
31
|
+
step_dt: Scalar[dt],
|
|
32
|
+
inv_dx2: Scalar[dt],
|
|
33
|
+
h: Scalar[dt],
|
|
34
|
+
noise_amp: Scalar[dt],
|
|
35
|
+
seed: UInt64,
|
|
36
|
+
step: UInt64,
|
|
37
|
+
):
|
|
38
|
+
comptime assert psi.flat_rank == 2, "field must be 2d"
|
|
39
|
+
comptime assert nxt.flat_rank == 2, "field must be 2d"
|
|
40
|
+
var nn = Int(n)
|
|
41
|
+
var j = global_idx.x
|
|
42
|
+
var i = global_idx.y
|
|
43
|
+
if i >= nn or j >= nn:
|
|
44
|
+
return
|
|
45
|
+
|
|
46
|
+
var ip = i + 1 if i + 1 < nn else 0
|
|
47
|
+
var im = i - 1 if i > 0 else nn - 1
|
|
48
|
+
var jp = j + 1 if j + 1 < nn else 0
|
|
49
|
+
var jm = j - 1 if j > 0 else nn - 1
|
|
50
|
+
|
|
51
|
+
var v = _update[dt](
|
|
52
|
+
rebind[Scalar[dt]](psi[i, j]),
|
|
53
|
+
rebind[Scalar[dt]](psi[ip, j]),
|
|
54
|
+
rebind[Scalar[dt]](psi[im, j]),
|
|
55
|
+
rebind[Scalar[dt]](psi[i, jp]),
|
|
56
|
+
rebind[Scalar[dt]](psi[i, jm]),
|
|
57
|
+
step_dt,
|
|
58
|
+
inv_dx2,
|
|
59
|
+
h,
|
|
60
|
+
)
|
|
61
|
+
if noise_amp != 0.0:
|
|
62
|
+
v += noise_amp * gaussian[dt](seed, step, UInt64(i * nn + j))
|
|
63
|
+
nxt[i, j] = rebind[nxt.ElementType](v)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _step_cpu[dt: DType](
|
|
67
|
+
psi: Pointer[Scalar[dt], MutAnyOrigin],
|
|
68
|
+
nxt: Pointer[Scalar[dt], MutAnyOrigin],
|
|
69
|
+
n: Int,
|
|
70
|
+
step_dt: Scalar[dt],
|
|
71
|
+
inv_dx2: Scalar[dt],
|
|
72
|
+
h: Scalar[dt],
|
|
73
|
+
noise_amp: Scalar[dt],
|
|
74
|
+
seed: UInt64,
|
|
75
|
+
step: UInt64,
|
|
76
|
+
):
|
|
77
|
+
for i in range(n):
|
|
78
|
+
var ip = i + 1 if i + 1 < n else 0
|
|
79
|
+
var im = i - 1 if i > 0 else n - 1
|
|
80
|
+
for j in range(n):
|
|
81
|
+
var jp = j + 1 if j + 1 < n else 0
|
|
82
|
+
var jm = j - 1 if j > 0 else n - 1
|
|
83
|
+
var v = _update[dt](
|
|
84
|
+
psi[unsafe_offset=i * n + j],
|
|
85
|
+
psi[unsafe_offset=ip * n + j],
|
|
86
|
+
psi[unsafe_offset=im * n + j],
|
|
87
|
+
psi[unsafe_offset=i * n + jp],
|
|
88
|
+
psi[unsafe_offset=i * n + jm],
|
|
89
|
+
step_dt,
|
|
90
|
+
inv_dx2,
|
|
91
|
+
h,
|
|
92
|
+
)
|
|
93
|
+
if noise_amp != 0.0:
|
|
94
|
+
v += noise_amp * gaussian[dt](seed, step, UInt64(i * n + j))
|
|
95
|
+
nxt[unsafe_offset=i * n + j] = v
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _run_cpu[dt: DType](
|
|
99
|
+
field: Pointer[Scalar[dt], MutAnyOrigin],
|
|
100
|
+
snaps_addr: Int,
|
|
101
|
+
n: Int,
|
|
102
|
+
nsteps: Int,
|
|
103
|
+
nevery: Int,
|
|
104
|
+
step_dt: Scalar[dt],
|
|
105
|
+
inv_dx2: Scalar[dt],
|
|
106
|
+
h: Scalar[dt],
|
|
107
|
+
noise_amp: Scalar[dt],
|
|
108
|
+
seed: UInt64,
|
|
109
|
+
step0: Int,
|
|
110
|
+
) raises -> String:
|
|
111
|
+
comptime esize = size_of[Scalar[dt]]()
|
|
112
|
+
var size = n * n
|
|
113
|
+
var owned_a = alloc(Layout[Scalar[dt]](count=size)).into_managed()
|
|
114
|
+
var owned_b = alloc(Layout[Scalar[dt]](count=size)).into_managed()
|
|
115
|
+
var a = Pointer[Scalar[dt], MutAnyOrigin](
|
|
116
|
+
unsafe_from_address=Int(owned_a.unsafe_ptr())
|
|
117
|
+
)
|
|
118
|
+
var b = Pointer[Scalar[dt], MutAnyOrigin](
|
|
119
|
+
unsafe_from_address=Int(owned_b.unsafe_ptr())
|
|
120
|
+
)
|
|
121
|
+
unsafe_memcpy(dest=a, src=field, count=size)
|
|
122
|
+
|
|
123
|
+
var k = 0
|
|
124
|
+
for s in range(1, nsteps + 1):
|
|
125
|
+
_step_cpu[dt](
|
|
126
|
+
a, b, n, step_dt, inv_dx2, h, noise_amp, seed,
|
|
127
|
+
UInt64(step0 + s - 1),
|
|
128
|
+
)
|
|
129
|
+
swap(a, b)
|
|
130
|
+
if s % nevery == 0:
|
|
131
|
+
var dst = Pointer[Scalar[dt], MutAnyOrigin](
|
|
132
|
+
unsafe_from_address=snaps_addr + k * size * esize
|
|
133
|
+
)
|
|
134
|
+
unsafe_memcpy(dest=dst, src=a, count=size)
|
|
135
|
+
k += 1
|
|
136
|
+
|
|
137
|
+
unsafe_memcpy(dest=field, src=a, count=size)
|
|
138
|
+
_ = owned_a^
|
|
139
|
+
_ = owned_b^
|
|
140
|
+
return String("cpu")
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _run_accel[dt: DType](
|
|
144
|
+
field: Pointer[Scalar[dt], MutAnyOrigin],
|
|
145
|
+
snaps_addr: Int,
|
|
146
|
+
n: Int,
|
|
147
|
+
nsteps: Int,
|
|
148
|
+
nevery: Int,
|
|
149
|
+
step_dt: Scalar[dt],
|
|
150
|
+
inv_dx2: Scalar[dt],
|
|
151
|
+
h: Scalar[dt],
|
|
152
|
+
noise_amp: Scalar[dt],
|
|
153
|
+
seed: UInt64,
|
|
154
|
+
step0: Int,
|
|
155
|
+
) raises -> String:
|
|
156
|
+
comptime esize = size_of[Scalar[dt]]()
|
|
157
|
+
var size = n * n
|
|
158
|
+
var ctx = DeviceContext()
|
|
159
|
+
var a = ctx.enqueue_create_buffer[dt](size)
|
|
160
|
+
var b = ctx.enqueue_create_buffer[dt](size)
|
|
161
|
+
ctx.enqueue_copy(dst_buf=a, src_ptr=field)
|
|
162
|
+
|
|
163
|
+
var layout = row_major(n, n)
|
|
164
|
+
comptime kern = _step_kernel[dt, type_of(layout)]
|
|
165
|
+
var grid = (ceildiv(n, TPB), ceildiv(n, TPB))
|
|
166
|
+
|
|
167
|
+
var k = 0
|
|
168
|
+
for s in range(1, nsteps + 1):
|
|
169
|
+
ctx.enqueue_function[kern](
|
|
170
|
+
TileTensor(a, layout),
|
|
171
|
+
TileTensor(b, layout),
|
|
172
|
+
Int32(n),
|
|
173
|
+
step_dt,
|
|
174
|
+
inv_dx2,
|
|
175
|
+
h,
|
|
176
|
+
noise_amp,
|
|
177
|
+
seed,
|
|
178
|
+
UInt64(step0 + s - 1),
|
|
179
|
+
grid_dim=grid,
|
|
180
|
+
block_dim=(TPB, TPB),
|
|
181
|
+
)
|
|
182
|
+
swap(a, b)
|
|
183
|
+
if s % nevery == 0:
|
|
184
|
+
var dst = Pointer[Scalar[dt], MutAnyOrigin](
|
|
185
|
+
unsafe_from_address=snaps_addr + k * size * esize
|
|
186
|
+
)
|
|
187
|
+
ctx.enqueue_copy(dst_ptr=dst, src_buf=a)
|
|
188
|
+
k += 1
|
|
189
|
+
|
|
190
|
+
ctx.enqueue_copy(dst_ptr=field, src_buf=a)
|
|
191
|
+
ctx.synchronize()
|
|
192
|
+
return String("accelerator")
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _dispatch[dt: DType](
|
|
196
|
+
field_addr: Int,
|
|
197
|
+
snaps_addr: Int,
|
|
198
|
+
n: Int,
|
|
199
|
+
nsteps: Int,
|
|
200
|
+
nevery: Int,
|
|
201
|
+
step_dt: Float64,
|
|
202
|
+
dx: Float64,
|
|
203
|
+
h: Float64,
|
|
204
|
+
eps: Float64,
|
|
205
|
+
seed: Int,
|
|
206
|
+
step0: Int,
|
|
207
|
+
device: Int,
|
|
208
|
+
) raises -> String:
|
|
209
|
+
var field = Pointer[Scalar[dt], MutAnyOrigin](
|
|
210
|
+
unsafe_from_address=field_addr
|
|
211
|
+
)
|
|
212
|
+
var inv_dx2 = 1.0 / (dx * dx)
|
|
213
|
+
var amp = sqrt(2.0 * eps * step_dt * inv_dx2) if eps > 0.0 else 0.0
|
|
214
|
+
var sdt = Scalar[dt](step_dt)
|
|
215
|
+
var sidx = Scalar[dt](inv_dx2)
|
|
216
|
+
var sh = Scalar[dt](h)
|
|
217
|
+
var samp = Scalar[dt](amp)
|
|
218
|
+
var useed = UInt64(seed)
|
|
219
|
+
|
|
220
|
+
comptime no_f64_on_accel = (
|
|
221
|
+
has_apple_gpu_accelerator() and dt == DType.float64
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
comptime if not has_accelerator():
|
|
225
|
+
if device == DEVICE_ACCEL:
|
|
226
|
+
raise Error(
|
|
227
|
+
"accelerator requested but this build has no supported"
|
|
228
|
+
" accelerator"
|
|
229
|
+
)
|
|
230
|
+
return _run_cpu[dt](
|
|
231
|
+
field, snaps_addr, n, nsteps, nevery, sdt, sidx, sh,
|
|
232
|
+
samp, useed, step0,
|
|
233
|
+
)
|
|
234
|
+
else:
|
|
235
|
+
comptime if no_f64_on_accel:
|
|
236
|
+
if device == DEVICE_ACCEL:
|
|
237
|
+
raise Error(
|
|
238
|
+
"accelerator requested but it has no float64"
|
|
239
|
+
" support; use dtype=float32"
|
|
240
|
+
)
|
|
241
|
+
return _run_cpu[dt](
|
|
242
|
+
field, snaps_addr, n, nsteps, nevery, sdt, sidx, sh,
|
|
243
|
+
samp, useed, step0,
|
|
244
|
+
)
|
|
245
|
+
else:
|
|
246
|
+
if device == DEVICE_CPU:
|
|
247
|
+
return _run_cpu[dt](
|
|
248
|
+
field, snaps_addr, n, nsteps, nevery, sdt, sidx, sh,
|
|
249
|
+
samp, useed, step0,
|
|
250
|
+
)
|
|
251
|
+
if DeviceContext.number_of_devices() == 0:
|
|
252
|
+
if device == DEVICE_ACCEL:
|
|
253
|
+
raise Error(
|
|
254
|
+
"accelerator requested but none was detected"
|
|
255
|
+
" at runtime"
|
|
256
|
+
)
|
|
257
|
+
return _run_cpu[dt](
|
|
258
|
+
field, snaps_addr, n, nsteps, nevery, sdt, sidx, sh,
|
|
259
|
+
samp, useed, step0,
|
|
260
|
+
)
|
|
261
|
+
return _run_accel[dt](
|
|
262
|
+
field, snaps_addr, n, nsteps, nevery, sdt, sidx, sh,
|
|
263
|
+
samp, useed, step0,
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def run_tdgl(
|
|
268
|
+
field_addr: Int,
|
|
269
|
+
snaps_addr: Int,
|
|
270
|
+
n: Int,
|
|
271
|
+
nsteps: Int,
|
|
272
|
+
nevery: Int,
|
|
273
|
+
step_dt: Float64,
|
|
274
|
+
dx: Float64,
|
|
275
|
+
h: Float64,
|
|
276
|
+
eps: Float64,
|
|
277
|
+
seed: Int,
|
|
278
|
+
step0: Int,
|
|
279
|
+
single: Bool,
|
|
280
|
+
device: Int,
|
|
281
|
+
) raises -> String:
|
|
282
|
+
if single:
|
|
283
|
+
return _dispatch[DType.float32](
|
|
284
|
+
field_addr, snaps_addr, n, nsteps, nevery, step_dt, dx, h,
|
|
285
|
+
eps, seed, step0, device,
|
|
286
|
+
)
|
|
287
|
+
return _dispatch[DType.float64](
|
|
288
|
+
field_addr, snaps_addr, n, nsteps, nevery, step_dt, dx, h,
|
|
289
|
+
eps, seed, step0, device,
|
|
290
|
+
)
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
from std.math import sqrt, log, cos
|
|
2
|
+
from std.sys import has_accelerator
|
|
3
|
+
from max.gpu.host import DeviceContext
|
|
4
|
+
|
|
5
|
+
comptime DEVICE_AUTO = 0
|
|
6
|
+
comptime DEVICE_ACCEL = 1
|
|
7
|
+
comptime DEVICE_CPU = 2
|
|
8
|
+
|
|
9
|
+
comptime TPB = 16
|
|
10
|
+
comptime GOLDEN = 0x9E3779B97F4A7C15
|
|
11
|
+
comptime TWO_PI = 6.283185307179586
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def get_device() -> Tuple[Bool, String, String]:
|
|
15
|
+
comptime if not has_accelerator():
|
|
16
|
+
return (
|
|
17
|
+
False, String(""), String("no supported accelerator at build time")
|
|
18
|
+
)
|
|
19
|
+
else:
|
|
20
|
+
try:
|
|
21
|
+
if DeviceContext.number_of_devices() == 0:
|
|
22
|
+
return (False, String(""), String("no accelerator detected"))
|
|
23
|
+
var ctx = DeviceContext()
|
|
24
|
+
return (True, ctx.name(), String(""))
|
|
25
|
+
except e:
|
|
26
|
+
return (False, String(""), String(e))
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def mix(x: UInt64) -> UInt64:
|
|
30
|
+
var z = x + GOLDEN
|
|
31
|
+
z = (z ^ (z >> 30)) * 0xBF58476D1CE4E5B9
|
|
32
|
+
z = (z ^ (z >> 27)) * 0x94D049BB133111EB
|
|
33
|
+
return z ^ (z >> 31)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def gaussian[dt: DType](
|
|
37
|
+
seed: UInt64, step: UInt64, site: UInt64
|
|
38
|
+
) -> Scalar[dt]:
|
|
39
|
+
comptime assert dt.is_floating_point(), "dt must be a float type"
|
|
40
|
+
comptime shift: UInt64 = 40 if dt == DType.float32 else 11
|
|
41
|
+
comptime scale = (
|
|
42
|
+
1.0 / 16777216.0 if dt == DType.float32 else 1.1102230246251565e-16
|
|
43
|
+
)
|
|
44
|
+
var h1 = mix(seed ^ mix(step * GOLDEN ^ site))
|
|
45
|
+
var h2 = mix(h1)
|
|
46
|
+
var u1 = (Scalar[dt](h1 >> shift) + 1.0) * Scalar[dt](scale)
|
|
47
|
+
var u2 = Scalar[dt](h2 >> shift) * Scalar[dt](scale)
|
|
48
|
+
var phase = cos(Float32(TWO_PI) * Float32(u2))
|
|
49
|
+
return sqrt(-2.0 * log(u1)) * Scalar[dt](phase)
|
simpok/_result.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
from typing import NamedTuple
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
|
|
5
|
+
_DEVICE_NAME = None
|
|
6
|
+
_DEVICE_CODES = {"auto": 0, "accelerator": 1, "gpu": 1, "cpu": 2}
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class Result(NamedTuple):
|
|
10
|
+
snaps: np.ndarray
|
|
11
|
+
meta: dict
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def device_name():
|
|
15
|
+
global _DEVICE_NAME
|
|
16
|
+
|
|
17
|
+
if _DEVICE_NAME is None:
|
|
18
|
+
from .device import get_device
|
|
19
|
+
|
|
20
|
+
_DEVICE_NAME = get_device().name or "cpu"
|
|
21
|
+
return _DEVICE_NAME
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def device_code(device):
|
|
25
|
+
try:
|
|
26
|
+
return _DEVICE_CODES[device]
|
|
27
|
+
except (KeyError, TypeError):
|
|
28
|
+
raise ValueError(
|
|
29
|
+
f"device must be one of {sorted(_DEVICE_CODES)}, got {device!r}"
|
|
30
|
+
) from None
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def make_meta(solver, *, n, dx, dt, dtype, seed, amplitude, steps, nevery,
|
|
34
|
+
step_start, step_end, backend, device, elapsed, **physics):
|
|
35
|
+
nsnap = steps // nevery
|
|
36
|
+
meta = {"solver": solver, "n": n, "dx": dx, "dt": dt}
|
|
37
|
+
meta.update(physics)
|
|
38
|
+
meta.update(
|
|
39
|
+
{
|
|
40
|
+
"dtype": np.dtype(dtype).name,
|
|
41
|
+
"seed": seed,
|
|
42
|
+
"amplitude": amplitude,
|
|
43
|
+
"steps": steps,
|
|
44
|
+
"nevery": nevery,
|
|
45
|
+
"nsnap": nsnap,
|
|
46
|
+
"step_start": step_start,
|
|
47
|
+
"step_end": step_end,
|
|
48
|
+
"times": (step_start + np.arange(1, nsnap + 1) * nevery) * dt,
|
|
49
|
+
"backend": backend,
|
|
50
|
+
"device": device_name(),
|
|
51
|
+
"device_request": device,
|
|
52
|
+
"elapsed": elapsed,
|
|
53
|
+
}
|
|
54
|
+
)
|
|
55
|
+
return meta
|
simpok/chc.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
import time
|
|
2
|
+
import numpy as np
|
|
3
|
+
from ._result import Result, device_code, make_meta
|
|
4
|
+
|
|
5
|
+
class CHC:
|
|
6
|
+
def __init__(self, n=256, dx=1.0, dt=0.01, eps=0.0,
|
|
7
|
+
dtype=np.float64, device="auto"):
|
|
8
|
+
dtype = np.dtype(dtype)
|
|
9
|
+
self._device_code = device_code(device)
|
|
10
|
+
self.device = device
|
|
11
|
+
if dtype not in (np.dtype(np.float32), np.dtype(np.float64)):
|
|
12
|
+
raise ValueError(
|
|
13
|
+
f"dtype must be float32 or float64, got {dtype.name}"
|
|
14
|
+
)
|
|
15
|
+
if int(n) != n or n < 5:
|
|
16
|
+
raise ValueError(f"n must be an integer >= 5, got {n!r}")
|
|
17
|
+
if dx <= 0.0:
|
|
18
|
+
raise ValueError(f"dx must be positive, got {dx!r}")
|
|
19
|
+
if dt <= 0.0:
|
|
20
|
+
raise ValueError(f"dt must be positive, got {dt!r}")
|
|
21
|
+
if eps < 0.0:
|
|
22
|
+
raise ValueError(f"eps must be non-negative, got {eps!r}")
|
|
23
|
+
|
|
24
|
+
q = 8.0 / (dx * dx)
|
|
25
|
+
growth = q * q - q
|
|
26
|
+
if growth > 0.0:
|
|
27
|
+
limit = 2.0 / growth
|
|
28
|
+
if dt > limit:
|
|
29
|
+
raise ValueError(
|
|
30
|
+
f"dt={dt} exceeds the explicit-Euler stability limit "
|
|
31
|
+
f"{limit} for the 2-d 5-point biharmonic at dx={dx}"
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
self.n = int(n)
|
|
35
|
+
self.dx = float(dx)
|
|
36
|
+
self.dt = float(dt)
|
|
37
|
+
self.eps = float(eps)
|
|
38
|
+
self.dtype = dtype
|
|
39
|
+
|
|
40
|
+
self.psi = None
|
|
41
|
+
self.seed = 0
|
|
42
|
+
self.amplitude = 0.0
|
|
43
|
+
self.psi0 = 0.0
|
|
44
|
+
self.step = 0
|
|
45
|
+
self.backend = None
|
|
46
|
+
|
|
47
|
+
def ic(self, seed=0, amplitude=0.01, psi0=0.0):
|
|
48
|
+
if amplitude <= 0.0:
|
|
49
|
+
raise ValueError(f"amplitude must be positive, got {amplitude!r}")
|
|
50
|
+
|
|
51
|
+
rng = np.random.default_rng(seed)
|
|
52
|
+
self.psi = (
|
|
53
|
+
psi0 + rng.uniform(-amplitude, amplitude, size=(self.n, self.n))
|
|
54
|
+
).astype(self.dtype)
|
|
55
|
+
self.seed = int(seed)
|
|
56
|
+
self.amplitude = float(amplitude)
|
|
57
|
+
self.psi0 = float(psi0)
|
|
58
|
+
self.step = 0
|
|
59
|
+
return self
|
|
60
|
+
|
|
61
|
+
@property
|
|
62
|
+
def t(self):
|
|
63
|
+
return self.step * self.dt
|
|
64
|
+
|
|
65
|
+
def run(self, steps, nevery):
|
|
66
|
+
if self.psi is None:
|
|
67
|
+
raise RuntimeError("call ic() before run()")
|
|
68
|
+
if np.shape(self.psi) != (self.n, self.n):
|
|
69
|
+
raise ValueError(
|
|
70
|
+
f"psi has shape {np.shape(self.psi)}, expected "
|
|
71
|
+
f"{(self.n, self.n)}"
|
|
72
|
+
)
|
|
73
|
+
if int(steps) != steps or steps < 1:
|
|
74
|
+
raise ValueError(f"steps must be a positive integer, got {steps!r}")
|
|
75
|
+
if int(nevery) != nevery or nevery < 1:
|
|
76
|
+
raise ValueError(
|
|
77
|
+
f"nevery must be a positive integer, got {nevery!r}"
|
|
78
|
+
)
|
|
79
|
+
if nevery > steps:
|
|
80
|
+
raise ValueError(f"nevery={nevery} exceeds steps={steps}")
|
|
81
|
+
|
|
82
|
+
steps = int(steps)
|
|
83
|
+
nevery = int(nevery)
|
|
84
|
+
nsnap = steps // nevery
|
|
85
|
+
step_start = self.step
|
|
86
|
+
|
|
87
|
+
import mojo.importer # noqa: F401
|
|
88
|
+
|
|
89
|
+
from .mojo_module import _chc_run
|
|
90
|
+
|
|
91
|
+
field = np.ascontiguousarray(self.psi, dtype=self.dtype)
|
|
92
|
+
snaps = np.empty((nsnap, self.n, self.n), dtype=self.dtype)
|
|
93
|
+
params = (
|
|
94
|
+
self.n,
|
|
95
|
+
steps,
|
|
96
|
+
nevery,
|
|
97
|
+
self.dt,
|
|
98
|
+
self.dx,
|
|
99
|
+
self.eps,
|
|
100
|
+
self.seed,
|
|
101
|
+
self.step,
|
|
102
|
+
self.dtype == np.dtype(np.float32),
|
|
103
|
+
self._device_code,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
t0 = time.perf_counter()
|
|
107
|
+
self.backend = _chc_run(field, snaps, params)
|
|
108
|
+
elapsed = time.perf_counter() - t0
|
|
109
|
+
|
|
110
|
+
self.psi = field
|
|
111
|
+
self.step += steps
|
|
112
|
+
|
|
113
|
+
meta = make_meta(
|
|
114
|
+
"chc",
|
|
115
|
+
n=self.n,
|
|
116
|
+
dx=self.dx,
|
|
117
|
+
dt=self.dt,
|
|
118
|
+
eps=self.eps,
|
|
119
|
+
psi0=self.psi0,
|
|
120
|
+
dtype=self.dtype,
|
|
121
|
+
seed=self.seed,
|
|
122
|
+
amplitude=self.amplitude,
|
|
123
|
+
steps=steps,
|
|
124
|
+
nevery=nevery,
|
|
125
|
+
step_start=step_start,
|
|
126
|
+
step_end=self.step,
|
|
127
|
+
backend=self.backend,
|
|
128
|
+
device=self.device,
|
|
129
|
+
elapsed=elapsed,
|
|
130
|
+
)
|
|
131
|
+
return Result(snaps, meta)
|
simpok/device.py
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
from typing import NamedTuple
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class Device(NamedTuple):
|
|
5
|
+
available: bool
|
|
6
|
+
name: str | None
|
|
7
|
+
reason: str | None
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def get_device() -> Device:
|
|
11
|
+
import mojo.importer # noqa: F401
|
|
12
|
+
|
|
13
|
+
from .mojo_module import _get_device
|
|
14
|
+
|
|
15
|
+
available, name, reason = _get_device()
|
|
16
|
+
return Device(available, name or None, reason or None)
|
simpok/mojo_module.mojo
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
from std.python import PythonObject
|
|
2
|
+
from std.python import Python
|
|
3
|
+
from std.python.bindings import PythonModuleBuilder
|
|
4
|
+
from _core_mojo import get_device, run_tdgl, run_chc
|
|
5
|
+
from std.os import abort
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@export
|
|
9
|
+
def PyInit_mojo_module() abi("C") -> PythonObject:
|
|
10
|
+
try:
|
|
11
|
+
var m = PythonModuleBuilder("mojo_module")
|
|
12
|
+
m.def_function[_get_device](
|
|
13
|
+
"_get_device", docstring="Probe the default accelerator"
|
|
14
|
+
)
|
|
15
|
+
m.def_function[_tdgl_run](
|
|
16
|
+
"_tdgl_run", docstring="Evolve a 2d TDGL field in place"
|
|
17
|
+
)
|
|
18
|
+
m.def_function[_chc_run](
|
|
19
|
+
"_chc_run", docstring="Evolve a 2d CHC field in place"
|
|
20
|
+
)
|
|
21
|
+
return m.finalize()
|
|
22
|
+
except e:
|
|
23
|
+
abort(String("error creating Python Mojo module: ", e))
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _get_device() raises -> PythonObject:
|
|
27
|
+
var available, name, reason = get_device()
|
|
28
|
+
return Python.tuple(
|
|
29
|
+
PythonObject(available), PythonObject(name), PythonObject(reason)
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _tdgl_run(
|
|
34
|
+
field: PythonObject, snaps: PythonObject, params: PythonObject
|
|
35
|
+
) raises -> PythonObject:
|
|
36
|
+
var backend = run_tdgl(
|
|
37
|
+
Int(py=field.ctypes.data),
|
|
38
|
+
Int(py=snaps.ctypes.data),
|
|
39
|
+
Int(py=params[0]),
|
|
40
|
+
Int(py=params[1]),
|
|
41
|
+
Int(py=params[2]),
|
|
42
|
+
Float64(py=params[3]),
|
|
43
|
+
Float64(py=params[4]),
|
|
44
|
+
Float64(py=params[5]),
|
|
45
|
+
Float64(py=params[6]),
|
|
46
|
+
Int(py=params[7]),
|
|
47
|
+
Int(py=params[8]),
|
|
48
|
+
Bool(py=params[9]),
|
|
49
|
+
Int(py=params[10]),
|
|
50
|
+
)
|
|
51
|
+
return PythonObject(backend)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _chc_run(
|
|
55
|
+
field: PythonObject, snaps: PythonObject, params: PythonObject
|
|
56
|
+
) raises -> PythonObject:
|
|
57
|
+
var backend = run_chc(
|
|
58
|
+
Int(py=field.ctypes.data),
|
|
59
|
+
Int(py=snaps.ctypes.data),
|
|
60
|
+
Int(py=params[0]),
|
|
61
|
+
Int(py=params[1]),
|
|
62
|
+
Int(py=params[2]),
|
|
63
|
+
Float64(py=params[3]),
|
|
64
|
+
Float64(py=params[4]),
|
|
65
|
+
Float64(py=params[5]),
|
|
66
|
+
Int(py=params[6]),
|
|
67
|
+
Int(py=params[7]),
|
|
68
|
+
Bool(py=params[8]),
|
|
69
|
+
Int(py=params[9]),
|
|
70
|
+
)
|
|
71
|
+
return PythonObject(backend)
|
simpok/tdgl.py
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
import time
|
|
2
|
+
import numpy as np
|
|
3
|
+
from ._result import Result, device_code, make_meta
|
|
4
|
+
|
|
5
|
+
class TDGL:
|
|
6
|
+
def __init__(self, n=256, dx=1.0, dt=0.1, h=0.0, eps=0.0,
|
|
7
|
+
dtype=np.float64, device="auto"):
|
|
8
|
+
dtype = np.dtype(dtype)
|
|
9
|
+
self._device_code = device_code(device)
|
|
10
|
+
self.device = device
|
|
11
|
+
if dtype not in (np.dtype(np.float32), np.dtype(np.float64)):
|
|
12
|
+
raise ValueError(
|
|
13
|
+
f"dtype must be float32 or float64, got {dtype.name}"
|
|
14
|
+
)
|
|
15
|
+
if int(n) != n or n < 3:
|
|
16
|
+
raise ValueError(f"n must be an integer >= 3, got {n!r}")
|
|
17
|
+
if dx <= 0.0:
|
|
18
|
+
raise ValueError(f"dx must be positive, got {dx!r}")
|
|
19
|
+
if dt <= 0.0:
|
|
20
|
+
raise ValueError(f"dt must be positive, got {dt!r}")
|
|
21
|
+
if eps < 0.0:
|
|
22
|
+
raise ValueError(f"eps must be non-negative, got {eps!r}")
|
|
23
|
+
|
|
24
|
+
limit = 0.25 * dx * dx
|
|
25
|
+
if dt > limit:
|
|
26
|
+
raise ValueError(
|
|
27
|
+
f"dt={dt} exceeds the explicit-Euler stability limit "
|
|
28
|
+
f"dx^2/4={limit} for the 2-d 5-point Laplacian"
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
self.n = int(n)
|
|
32
|
+
self.dx = float(dx)
|
|
33
|
+
self.dt = float(dt)
|
|
34
|
+
self.h = float(h)
|
|
35
|
+
self.eps = float(eps)
|
|
36
|
+
self.dtype = dtype
|
|
37
|
+
|
|
38
|
+
self.psi = None
|
|
39
|
+
self.seed = 0
|
|
40
|
+
self.amplitude = 0.0
|
|
41
|
+
self.step = 0
|
|
42
|
+
self.backend = None
|
|
43
|
+
|
|
44
|
+
def ic(self, seed=0, amplitude=0.01):
|
|
45
|
+
if amplitude <= 0.0:
|
|
46
|
+
raise ValueError(f"amplitude must be positive, got {amplitude!r}")
|
|
47
|
+
|
|
48
|
+
rng = np.random.default_rng(seed)
|
|
49
|
+
self.psi = rng.uniform(
|
|
50
|
+
-amplitude, amplitude, size=(self.n, self.n)
|
|
51
|
+
).astype(self.dtype)
|
|
52
|
+
self.seed = int(seed)
|
|
53
|
+
self.amplitude = float(amplitude)
|
|
54
|
+
self.step = 0
|
|
55
|
+
return self
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def t(self):
|
|
59
|
+
return self.step * self.dt
|
|
60
|
+
|
|
61
|
+
def run(self, steps, nevery):
|
|
62
|
+
if self.psi is None:
|
|
63
|
+
raise RuntimeError("call ic() before run()")
|
|
64
|
+
if np.shape(self.psi) != (self.n, self.n):
|
|
65
|
+
raise ValueError(
|
|
66
|
+
f"psi has shape {np.shape(self.psi)}, expected "
|
|
67
|
+
f"{(self.n, self.n)}"
|
|
68
|
+
)
|
|
69
|
+
if int(steps) != steps or steps < 1:
|
|
70
|
+
raise ValueError(f"steps must be a positive integer, got {steps!r}")
|
|
71
|
+
if int(nevery) != nevery or nevery < 1:
|
|
72
|
+
raise ValueError(
|
|
73
|
+
f"nevery must be a positive integer, got {nevery!r}"
|
|
74
|
+
)
|
|
75
|
+
if nevery > steps:
|
|
76
|
+
raise ValueError(f"nevery={nevery} exceeds steps={steps}")
|
|
77
|
+
|
|
78
|
+
steps = int(steps)
|
|
79
|
+
nevery = int(nevery)
|
|
80
|
+
nsnap = steps // nevery
|
|
81
|
+
step_start = self.step
|
|
82
|
+
|
|
83
|
+
import mojo.importer # noqa: F401
|
|
84
|
+
|
|
85
|
+
from .mojo_module import _tdgl_run
|
|
86
|
+
|
|
87
|
+
field = np.ascontiguousarray(self.psi, dtype=self.dtype)
|
|
88
|
+
snaps = np.empty((nsnap, self.n, self.n), dtype=self.dtype)
|
|
89
|
+
params = (
|
|
90
|
+
self.n,
|
|
91
|
+
steps,
|
|
92
|
+
nevery,
|
|
93
|
+
self.dt,
|
|
94
|
+
self.dx,
|
|
95
|
+
self.h,
|
|
96
|
+
self.eps,
|
|
97
|
+
self.seed,
|
|
98
|
+
self.step,
|
|
99
|
+
self.dtype == np.dtype(np.float32),
|
|
100
|
+
self._device_code,
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
t0 = time.perf_counter()
|
|
104
|
+
self.backend = _tdgl_run(field, snaps, params)
|
|
105
|
+
elapsed = time.perf_counter() - t0
|
|
106
|
+
|
|
107
|
+
self.psi = field
|
|
108
|
+
self.step += steps
|
|
109
|
+
|
|
110
|
+
meta = make_meta(
|
|
111
|
+
"tdgl",
|
|
112
|
+
n=self.n,
|
|
113
|
+
dx=self.dx,
|
|
114
|
+
dt=self.dt,
|
|
115
|
+
h=self.h,
|
|
116
|
+
eps=self.eps,
|
|
117
|
+
dtype=self.dtype,
|
|
118
|
+
seed=self.seed,
|
|
119
|
+
amplitude=self.amplitude,
|
|
120
|
+
steps=steps,
|
|
121
|
+
nevery=nevery,
|
|
122
|
+
step_start=step_start,
|
|
123
|
+
step_end=self.step,
|
|
124
|
+
backend=self.backend,
|
|
125
|
+
device=self.device,
|
|
126
|
+
elapsed=elapsed,
|
|
127
|
+
)
|
|
128
|
+
return Result(snaps, meta)
|
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: simpok
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: High-level Python API for phase-ordering kinetics, with the computation in Mojo
|
|
5
|
+
Project-URL: Homepage, https://github.com/ivijayyadav/simpok
|
|
6
|
+
Project-URL: Repository, https://github.com/ivijayyadav/simpok
|
|
7
|
+
Project-URL: Issues, https://github.com/ivijayyadav/simpok/issues
|
|
8
|
+
Author-email: vijay <vijpiml@gmail.com>
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: cahn-hilliard,ginzburg-landau,gpu,mojo,phase-field,phase-ordering,spinodal-decomposition
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Operating System :: MacOS :: MacOS X
|
|
15
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Physics
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Requires-Dist: max<27,>=26.5
|
|
24
|
+
Requires-Dist: mojo<2,>=1.0
|
|
25
|
+
Requires-Dist: numpy>=1.24
|
|
26
|
+
Provides-Extra: examples
|
|
27
|
+
Requires-Dist: matplotlib>=3.8; extra == 'examples'
|
|
28
|
+
Description-Content-Type: text/markdown
|
|
29
|
+
|
|
30
|
+
# simpok
|
|
31
|
+
|
|
32
|
+
Phase-ordering kinetics in Python, with the computation in [Mojo](https://www.modular.com/mojo).
|
|
33
|
+
|
|
34
|
+
You write ordinary Python; the solvers run on your GPU if you have one, and on the CPU
|
|
35
|
+
otherwise, from the same code. The intent is to keep the modelling in a high-level API and push
|
|
36
|
+
the per-site arithmetic down to Mojo.
|
|
37
|
+
|
|
38
|
+
> **Status: alpha.** Two solvers, 2-d only, 5-point Laplacian, periodic boundaries. The API may
|
|
39
|
+
> still change.
|
|
40
|
+
|
|
41
|
+
## Install
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
pip install simpok
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
Python 3.10+. The Mojo toolchain arrives as a dependency — there is nothing else to install, and
|
|
48
|
+
no compiler to set up by hand.
|
|
49
|
+
|
|
50
|
+
**The first `import simpok` takes about 20 seconds.** It is compiling the Mojo sources for your
|
|
51
|
+
machine; the result is cached and every later import is instant. It recompiles only when the
|
|
52
|
+
package is upgraded. This is deliberate rather than a packaging shortcut: whether an accelerator
|
|
53
|
+
is targeted is decided at Mojo compile time, so building on your machine is what lets one
|
|
54
|
+
universal wheel use whatever hardware you actually have.
|
|
55
|
+
|
|
56
|
+
## Quick start
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
from simpok import TDGL
|
|
60
|
+
|
|
61
|
+
sim = TDGL().ic(seed=0)
|
|
62
|
+
snaps, meta = sim.run(steps=90000, nevery=1000)
|
|
63
|
+
|
|
64
|
+
print(snaps.shape) # (90, 256, 256)
|
|
65
|
+
print(meta["backend"]) # 'accelerator' or 'cpu'
|
|
66
|
+
print(meta["times"][-1]) # 9000.0
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
`run` returns a `Result` — a named tuple of `snaps` (a `(nsnap, n, n)` NumPy array) and `meta`
|
|
70
|
+
(a dict of every parameter used, plus `times`, `backend`, `device` and `elapsed`).
|
|
71
|
+
|
|
72
|
+
Conserved dynamics works the same way:
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
from simpok import CHC
|
|
76
|
+
|
|
77
|
+
snaps, meta = CHC().ic(seed=0, psi0=-0.4).run(steps=50000, nevery=500)
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Check what you're running on:
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
from simpok import get_device
|
|
84
|
+
print(get_device()) # Device(available=True, name='NVIDIA RTX A5000', reason=None)
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## The models
|
|
88
|
+
|
|
89
|
+
**`TDGL`** — nonconserved scalar order parameter (Model A). Domain coarsening in a quenched
|
|
90
|
+
ferromagnet; the Allen–Cahn growth law `L(t) ~ t^(1/2)`.
|
|
91
|
+
|
|
92
|
+
```
|
|
93
|
+
∂ψ/∂t = ψ − ψ³ + h + ∇²ψ + θ
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
**`CHC`** — conserved scalar order parameter (Model B). Spinodal decomposition in a binary
|
|
97
|
+
mixture; the Lifshitz–Slyozov growth law `L(t) ~ t^(1/3)`. The mean of ψ is conserved exactly,
|
|
98
|
+
so `psi0` sets the composition — `0.0` gives the bicontinuous critical quench, `-0.4` gives
|
|
99
|
+
minority droplets.
|
|
100
|
+
|
|
101
|
+
```
|
|
102
|
+
∂ψ/∂t = −∇²(ψ − ψ³ + ∇²ψ) + ∇·θ
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Thermal noise is off by default (`eps=0`) — it is asymptotically irrelevant to
|
|
106
|
+
the growth laws. Set `eps > 0` to switch it on; for `CHC` it is applied as a bond-centred current
|
|
107
|
+
so that conservation stays exact to round-off.
|
|
108
|
+
|
|
109
|
+
## API
|
|
110
|
+
|
|
111
|
+
```python
|
|
112
|
+
TDGL(n=256, dx=1.0, dt=0.1, h=0.0, eps=0.0, dtype=np.float64, device="auto")
|
|
113
|
+
CHC (n=256, dx=1.0, dt=0.01, eps=0.0, dtype=np.float64, device="auto")
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
| Method | |
|
|
117
|
+
|---|---|
|
|
118
|
+
| `.ic(seed=0, amplitude=0.01)` | random initial condition; `CHC` also takes `psi0=0.0`. Returns `self`, so it chains. |
|
|
119
|
+
| `.run(steps, nevery)` | evolve `steps` steps, saving every `nevery`. Returns `Result(snaps, meta)`. |
|
|
120
|
+
| `.t` | current simulation time |
|
|
121
|
+
| `.psi` | current field, carried across calls — successive `run` calls continue the trajectory |
|
|
122
|
+
|
|
123
|
+
`device` is `"auto"` (default), `"accelerator"`, or `"cpu"`. `"accelerator"` raises if none is
|
|
124
|
+
usable rather than silently falling back. `dtype` is `float64` (default) or `float32`.
|
|
125
|
+
|
|
126
|
+
Timesteps are checked against the explicit-Euler stability limit at construction — `dt <= dx²/4`
|
|
127
|
+
for `TDGL`, and the much tighter fourth-order bound for `CHC`, which is why its default `dt` is
|
|
128
|
+
ten times smaller.
|
|
129
|
+
|
|
130
|
+
## Hardware notes
|
|
131
|
+
|
|
132
|
+
- **NVIDIA** — both dtypes work.
|
|
133
|
+
- **Apple Silicon** — Metal has no float64 at all, so use `dtype=np.float32` to run on the GPU.
|
|
134
|
+
A float64 run falls back to the CPU automatically; asking for `device="accelerator"` with
|
|
135
|
+
float64 raises with an explanation.
|
|
136
|
+
- **No GPU** — everything runs on the CPU with identical results. Deterministic (`eps=0`) runs
|
|
137
|
+
are bit-identical between CPU and accelerator.
|
|
138
|
+
|
|
139
|
+
On float32, `CHC` is several times faster than float64 on consumer NVIDIA cards, whose
|
|
140
|
+
double-precision throughput is heavily reduced; the domain length scale agrees with float64 to
|
|
141
|
+
better than 0.01%. `TDGL` at small grids is limited by kernel-launch overhead rather than
|
|
142
|
+
arithmetic, so larger lattices use the GPU far more efficiently than small ones.
|
|
143
|
+
|
|
144
|
+
## Correctness
|
|
145
|
+
|
|
146
|
+
Both solvers are checked against analytic results, not just for plausibility:
|
|
147
|
+
|
|
148
|
+
- equilibrium interface `tanh(z/√2)`, second-order convergent in `dx`
|
|
149
|
+
- `TDGL`: droplet collapse `dR²/dt = −2(d−1)`; the `t^(1/2)` growth law
|
|
150
|
+
- `CHC`: the Cahn dispersion relation `σ(q) = q − q²` reproduced to 1 part in 10¹⁰; the mean
|
|
151
|
+
conserved to 1 part in 10¹⁷; the `t^(1/3)` growth law
|
|
152
|
+
- conserved noise verified against the discrete fluctuation–dissipation relation
|
|
153
|
+
- seed reproducibility, resumability, and CPU/accelerator agreement
|
|
154
|
+
|
|
155
|
+
## Examples
|
|
156
|
+
|
|
157
|
+
`examples/` contains runnable scripts that produce snapshot figures:
|
|
158
|
+
|
|
159
|
+
```bash
|
|
160
|
+
python -m examples.tdgl_quickstart # writes tdgl.png
|
|
161
|
+
python -m examples.chc_quickstart # writes chc.png
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
They need matplotlib: `pip install simpok[examples]`.
|
|
165
|
+
|
|
166
|
+
## Reference
|
|
167
|
+
|
|
168
|
+
Sanjay Puri, "Kinetics of Phase Transitions", Ch. 1 in *Kinetics of Phase Transitions*,
|
|
169
|
+
S. Puri and V. Wadhawan (eds.), CRC Press (2009).
|
|
170
|
+
|
|
171
|
+
## License
|
|
172
|
+
|
|
173
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
simpok/__init__.py,sha256=qAusnQnv2qQUBzwP3Y_Hdi0OMHiIB8aj55Qg0ih1wuI,172
|
|
2
|
+
simpok/_result.py,sha256=kauuiZy77ug4z21S5BA8PFZEE8W0jCsMk2CyeO4ee1w,1419
|
|
3
|
+
simpok/chc.py,sha256=4rrpw8ODPvQKXmjHdHEImkROoAxh9kWPYrxSmNgxBps,4069
|
|
4
|
+
simpok/device.py,sha256=O2rfjc0QCTyN8Tfc1kjqmQGsKapqv88hz9ou0_SgyIQ,337
|
|
5
|
+
simpok/mojo_module.mojo,sha256=8l4WqOmYdEzVSw_pm0BTkuBdSbY_jtJZt6S3vgWSNew,2050
|
|
6
|
+
simpok/tdgl.py,sha256=BaJh5q3tKe0oQeRMNOXbuMqxOuiiN87b8NOduxZpX18,3937
|
|
7
|
+
simpok/_core_mojo/__init__.mojo,sha256=fa4nrlLtv-mIAF0fJlLI8rggolZWONAFD0VcXZoPsX8,85
|
|
8
|
+
simpok/_core_mojo/_chc.mojo,sha256=6OiZbsTtpD40TxuW8bsPjkUZHdTZH1rejXY4MRi9UuM,10286
|
|
9
|
+
simpok/_core_mojo/_tdgl.mojo,sha256=7AJQlT-I4SKnvg5Ffn-WJ2COr6ZdEhzs1zn0VOUVD8M,8364
|
|
10
|
+
simpok/_core_mojo/_utils.mojo,sha256=Yejb5W7UpBDq79SAdGnYtWoCUpMcUupKsvCou8d7KoU,1582
|
|
11
|
+
simpok-0.1.0.dist-info/METADATA,sha256=BxS5WLovjFjSa0prVEJJdqF-NLuHVvOTPviW27ndl-k,6620
|
|
12
|
+
simpok-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
13
|
+
simpok-0.1.0.dist-info/licenses/LICENSE,sha256=XunFJgLLY9giZlL2BsiFAlkkhEOA2BFwkvYpuo1k9GU,1068
|
|
14
|
+
simpok-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Vijay Yadav
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|