stanli 0.3.0__py3-none-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stanli/LICENSE +29 -0
- stanli/THIRD_PARTY_LICENSES.md +31 -0
- stanli/__init__.py +225 -0
- stanli/_bin/stanc.exe +0 -0
- stanli/_bin/stanli.dll +0 -0
- stanli-0.3.0.dist-info/METADATA +243 -0
- stanli-0.3.0.dist-info/RECORD +10 -0
- stanli-0.3.0.dist-info/WHEEL +5 -0
- stanli-0.3.0.dist-info/licenses/stanli/LICENSE +29 -0
- stanli-0.3.0.dist-info/top_level.txt +1 -0
stanli/LICENSE
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
BSD 3-Clause License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026, Sean Talts
|
|
4
|
+
|
|
5
|
+
Redistribution and use in source and binary forms, with or without
|
|
6
|
+
modification, are permitted provided that the following conditions are met:
|
|
7
|
+
|
|
8
|
+
1. Redistributions of source code must retain the above copyright notice,
|
|
9
|
+
this list of conditions and the following disclaimer.
|
|
10
|
+
|
|
11
|
+
2. Redistributions in binary form must reproduce the above copyright
|
|
12
|
+
notice, this list of conditions and the following disclaimer in the
|
|
13
|
+
documentation and/or other materials provided with the distribution.
|
|
14
|
+
|
|
15
|
+
3. Neither the name of the copyright holder nor the names of its
|
|
16
|
+
contributors may be used to endorse or promote products derived from
|
|
17
|
+
this software without specific prior written permission.
|
|
18
|
+
|
|
19
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
20
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
21
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
|
22
|
+
ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE
|
|
23
|
+
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
|
24
|
+
CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
|
25
|
+
SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
|
26
|
+
INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
|
27
|
+
CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
|
28
|
+
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
|
29
|
+
POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# Third-party components in the stanli binary
|
|
2
|
+
|
|
3
|
+
The stanli shared library is self-contained: it links only the system C
|
|
4
|
+
and C++ runtimes, and everything else is compiled in. This file lists what
|
|
5
|
+
is compiled in and under what terms, because distributing the binary
|
|
6
|
+
distributes those components too.
|
|
7
|
+
|
|
8
|
+
| Component | Role in the binary | License |
|
|
9
|
+
| --- | --- | --- |
|
|
10
|
+
| [Stan math library](https://github.com/stan-dev/math) | every density, constraint transform, and reverse-mode derivative | BSD 3-Clause |
|
|
11
|
+
| [Stan](https://github.com/stan-dev/stan) | NUTS sampler and adaptation | BSD 3-Clause |
|
|
12
|
+
| [stanc3](https://github.com/stan-dev/stanc3) | the Stan compiler, compiled to a self-contained object and linked in | BSD 3-Clause |
|
|
13
|
+
| OCaml runtime | required by the compiled stanc3 object | LGPL 2.1 with the OCaml linking exception, which explicitly permits linking into a binary under other terms |
|
|
14
|
+
| [Eigen](https://eigen.tuxfamily.org) | dense linear algebra behind stan-math | MPL 2.0 |
|
|
15
|
+
| [Boost](https://www.boost.org) | math special functions and utilities used by stan-math | Boost Software License 1.0 |
|
|
16
|
+
| [SUNDIALS / CVODES](https://computing.llnl.gov/projects/sundials) | ODE integration for `integrate_ode_rk45` and `integrate_ode_bdf` | BSD 3-Clause |
|
|
17
|
+
| [nlohmann/json](https://github.com/nlohmann/json) | reading CmdStan-format JSON data | MIT |
|
|
18
|
+
|
|
19
|
+
Full license texts ship with the vendored sources fetched by
|
|
20
|
+
`deps/fetch.sh`; see `deps/math/LICENSE.md`, `deps/stan/LICENSE.md`, and
|
|
21
|
+
the license files under `deps/math/lib/`.
|
|
22
|
+
|
|
23
|
+
Notes on the two that carry conditions beyond attribution:
|
|
24
|
+
|
|
25
|
+
- **Eigen (MPL 2.0)** is a file-level copyleft: distributing the binary is
|
|
26
|
+
fine, and modifications to Eigen's own files would have to be published.
|
|
27
|
+
stanli does not modify Eigen.
|
|
28
|
+
- **OCaml runtime (LGPL 2.1)** ships with a linking exception written for
|
|
29
|
+
exactly this case ("you may link this library into an executable and
|
|
30
|
+
distribute that executable under terms of your choice"), so no relinking
|
|
31
|
+
obligation attaches to the stanli binary.
|
stanli/__init__.py
ADDED
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
"""stanli: the Stan Language Interpreter.
|
|
2
|
+
|
|
3
|
+
Compiles a .stan model with stanc3 (linked into the bundled shared library,
|
|
4
|
+
or a bundled stanc binary as a subprocess where it is not), lowers it to an
|
|
5
|
+
op graph in-process, and samples with NUTS. No C++ toolchain, no model
|
|
6
|
+
compilation on this machine.
|
|
7
|
+
"""
|
|
8
|
+
import ctypes
|
|
9
|
+
import json
|
|
10
|
+
import pathlib
|
|
11
|
+
import subprocess
|
|
12
|
+
import sys
|
|
13
|
+
|
|
14
|
+
import numpy as np
|
|
15
|
+
|
|
16
|
+
__all__ = ["Model", "__version__"]
|
|
17
|
+
# The one place the version lives. setup.py and the release workflow both
|
|
18
|
+
# read it from here.
|
|
19
|
+
__version__ = "0.3.0"
|
|
20
|
+
|
|
21
|
+
_BIN = pathlib.Path(__file__).parent / "_bin"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _load_lib():
|
|
25
|
+
names = {"darwin": "libstanli.dylib", "linux": "libstanli.so"}
|
|
26
|
+
lib = ctypes.CDLL(str(_BIN / names.get(sys.platform, "stanli.dll")))
|
|
27
|
+
lib.stanli_model_new.restype = ctypes.c_void_p
|
|
28
|
+
lib.stanli_model_new.argtypes = [ctypes.c_char_p, ctypes.c_char_p,
|
|
29
|
+
ctypes.c_char_p, ctypes.c_size_t]
|
|
30
|
+
lib.stanli_model_free.argtypes = [ctypes.c_void_p]
|
|
31
|
+
lib.stanli_n_unconstrained.restype = ctypes.c_int64
|
|
32
|
+
lib.stanli_n_unconstrained.argtypes = [ctypes.c_void_p]
|
|
33
|
+
lib.stanli_grad.restype = ctypes.c_int
|
|
34
|
+
lib.stanli_grad.argtypes = [ctypes.c_void_p,
|
|
35
|
+
ctypes.POINTER(ctypes.c_double),
|
|
36
|
+
ctypes.POINTER(ctypes.c_double),
|
|
37
|
+
ctypes.POINTER(ctypes.c_double)]
|
|
38
|
+
lib.stanli_sample.restype = ctypes.c_int
|
|
39
|
+
lib.stanli_sample.argtypes = [ctypes.c_void_p, ctypes.c_uint32,
|
|
40
|
+
ctypes.c_int, ctypes.c_int, ctypes.c_double,
|
|
41
|
+
ctypes.POINTER(ctypes.c_double),
|
|
42
|
+
ctypes.c_char_p, ctypes.c_size_t]
|
|
43
|
+
lib.stanli_n_constrained.restype = ctypes.c_int64
|
|
44
|
+
lib.stanli_n_constrained.argtypes = [ctypes.c_void_p]
|
|
45
|
+
lib.stanli_constrained_name.restype = ctypes.c_char_p
|
|
46
|
+
lib.stanli_constrained_name.argtypes = [ctypes.c_void_p, ctypes.c_int64]
|
|
47
|
+
lib.stanli_constrain.restype = ctypes.c_int
|
|
48
|
+
lib.stanli_constrain.argtypes = [ctypes.c_void_p,
|
|
49
|
+
ctypes.POINTER(ctypes.c_double),
|
|
50
|
+
ctypes.POINTER(ctypes.c_double)]
|
|
51
|
+
lib.stanli_has_embedded_stanc.restype = ctypes.c_int
|
|
52
|
+
lib.stanli_exact_lp.restype = ctypes.c_int
|
|
53
|
+
lib.stanli_model_new_from_stan.restype = ctypes.c_void_p
|
|
54
|
+
lib.stanli_model_new_from_stan.argtypes = [ctypes.c_char_p,
|
|
55
|
+
ctypes.c_char_p,
|
|
56
|
+
ctypes.c_char_p,
|
|
57
|
+
ctypes.c_size_t]
|
|
58
|
+
lib.stanli_wa_n_columns.restype = ctypes.c_int64
|
|
59
|
+
lib.stanli_wa_n_columns.argtypes = [ctypes.c_void_p]
|
|
60
|
+
lib.stanli_wa_column_name.restype = ctypes.c_char_p
|
|
61
|
+
lib.stanli_wa_column_name.argtypes = [ctypes.c_void_p, ctypes.c_int64]
|
|
62
|
+
lib.stanli_wa_seed.argtypes = [ctypes.c_void_p, ctypes.c_uint32]
|
|
63
|
+
lib.stanli_wa_row.restype = ctypes.c_int
|
|
64
|
+
lib.stanli_wa_row.argtypes = [ctypes.c_void_p,
|
|
65
|
+
ctypes.POINTER(ctypes.c_double),
|
|
66
|
+
ctypes.POINTER(ctypes.c_double)]
|
|
67
|
+
return lib
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
_lib = _load_lib()
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _stanc_mir(model_path: pathlib.Path) -> str:
|
|
74
|
+
stanc = _BIN / ("stanc.exe" if sys.platform == "win32" else "stanc")
|
|
75
|
+
r = subprocess.run([str(stanc), "--debug-transformed-mir",
|
|
76
|
+
str(model_path)],
|
|
77
|
+
capture_output=True, text=True)
|
|
78
|
+
if r.returncode != 0 or not r.stdout:
|
|
79
|
+
raise RuntimeError(f"stanc failed:\n{r.stderr}")
|
|
80
|
+
return r.stdout
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _data_to_json(data) -> str:
|
|
84
|
+
"""Accept what callers naturally have: a dict, a path, or JSON text.
|
|
85
|
+
|
|
86
|
+
numpy arrays are converted, since data almost always arrives as one.
|
|
87
|
+
"""
|
|
88
|
+
if data is None:
|
|
89
|
+
return "{}"
|
|
90
|
+
if isinstance(data, pathlib.Path):
|
|
91
|
+
return data.read_text()
|
|
92
|
+
if isinstance(data, str):
|
|
93
|
+
stripped = data.lstrip()
|
|
94
|
+
if stripped.startswith("{"):
|
|
95
|
+
return data
|
|
96
|
+
return pathlib.Path(data).read_text()
|
|
97
|
+
|
|
98
|
+
def encode(value):
|
|
99
|
+
if hasattr(value, "tolist"):
|
|
100
|
+
return value.tolist()
|
|
101
|
+
raise TypeError(f"cannot serialise {type(value).__name__} as Stan data")
|
|
102
|
+
|
|
103
|
+
return json.dumps(data, default=encode)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def exact_lp() -> bool:
|
|
107
|
+
"""True if lp__ reproduces CmdStan's exactly.
|
|
108
|
+
|
|
109
|
+
The wheel is built this way and always has been. A STANLI_LITE_LP
|
|
110
|
+
build -- which is what ships to the browser -- drops stan-math's
|
|
111
|
+
propto instantiations to halve the library, leaving every gradient
|
|
112
|
+
bitwise identical and lp__ a per-model constant higher. A pinned seed
|
|
113
|
+
still gives a different chain there: lp is added to the kinetic
|
|
114
|
+
energy, so shifting it changes the rounding, and NUTS amplifies that
|
|
115
|
+
into a different (equally valid) trajectory.
|
|
116
|
+
"""
|
|
117
|
+
return bool(_lib.stanli_exact_lp())
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
class Model:
|
|
121
|
+
"""A compiled (model, data) pair."""
|
|
122
|
+
|
|
123
|
+
def __init__(self, stan_file=None, data=None, stan_code=None):
|
|
124
|
+
if stan_code is None:
|
|
125
|
+
if stan_file is None:
|
|
126
|
+
raise ValueError("provide stan_file or stan_code")
|
|
127
|
+
stan_code = pathlib.Path(stan_file).read_text()
|
|
128
|
+
data_json = _data_to_json(data)
|
|
129
|
+
|
|
130
|
+
err = ctypes.create_string_buffer(8192)
|
|
131
|
+
if _lib.stanli_has_embedded_stanc():
|
|
132
|
+
# Fully in-process: embedded stanc3 compiles the model.
|
|
133
|
+
self._m = _lib.stanli_model_new_from_stan(
|
|
134
|
+
stan_code.encode(), data_json.encode(), err, len(err))
|
|
135
|
+
else:
|
|
136
|
+
# Fallback: bundled stanc binary as a subprocess.
|
|
137
|
+
import tempfile
|
|
138
|
+
tmp = pathlib.Path(tempfile.mkdtemp()) / "model.stan"
|
|
139
|
+
tmp.write_text(stan_code)
|
|
140
|
+
mir = _stanc_mir(tmp)
|
|
141
|
+
self._m = _lib.stanli_model_new(mir.encode(), data_json.encode(),
|
|
142
|
+
err, len(err))
|
|
143
|
+
if not self._m:
|
|
144
|
+
raise RuntimeError(err.value.decode())
|
|
145
|
+
self.n_unconstrained = _lib.stanli_n_unconstrained(self._m)
|
|
146
|
+
n_con = _lib.stanli_n_constrained(self._m)
|
|
147
|
+
self.constrained_names = [
|
|
148
|
+
_lib.stanli_constrained_name(self._m, i).decode()
|
|
149
|
+
for i in range(n_con)
|
|
150
|
+
]
|
|
151
|
+
|
|
152
|
+
def __del__(self):
|
|
153
|
+
if getattr(self, "_m", None):
|
|
154
|
+
_lib.stanli_model_free(self._m)
|
|
155
|
+
self._m = None
|
|
156
|
+
|
|
157
|
+
def log_prob_grad(self, q):
|
|
158
|
+
"""log density (jacobian included) and gradient at unconstrained q."""
|
|
159
|
+
q = np.ascontiguousarray(q, dtype=np.float64)
|
|
160
|
+
if q.size != self.n_unconstrained:
|
|
161
|
+
raise ValueError(f"q has {q.size} elements, model has "
|
|
162
|
+
f"{self.n_unconstrained} unconstrained parameters")
|
|
163
|
+
lp = ctypes.c_double()
|
|
164
|
+
grad = np.empty(self.n_unconstrained)
|
|
165
|
+
rc = _lib.stanli_grad(
|
|
166
|
+
self._m,
|
|
167
|
+
q.ctypes.data_as(ctypes.POINTER(ctypes.c_double)),
|
|
168
|
+
ctypes.byref(lp),
|
|
169
|
+
grad.ctypes.data_as(ctypes.POINTER(ctypes.c_double)))
|
|
170
|
+
if rc != 0:
|
|
171
|
+
# The gradient buffer is uninitialized on failure; never let a
|
|
172
|
+
# caller see it.
|
|
173
|
+
raise RuntimeError("log density evaluation failed at this point "
|
|
174
|
+
"(domain error in a distribution or function)")
|
|
175
|
+
return lp.value, grad
|
|
176
|
+
|
|
177
|
+
def sample(self, *, seed=1, warmup=1000, samples=1000, delta=0.8):
|
|
178
|
+
"""NUTS draws as {name: array} of CSV columns.
|
|
179
|
+
|
|
180
|
+
Models with a generate_quantities section return every column
|
|
181
|
+
CmdStan's CSV would carry: constrained parameters, transformed
|
|
182
|
+
parameters, and generated quantities, with RNG draws streamed
|
|
183
|
+
from `seed`. Models without one return the constrained
|
|
184
|
+
parameters.
|
|
185
|
+
"""
|
|
186
|
+
n = self.n_unconstrained
|
|
187
|
+
draws = np.empty((samples, n))
|
|
188
|
+
err = ctypes.create_string_buffer(4096)
|
|
189
|
+
rc = _lib.stanli_sample(
|
|
190
|
+
self._m, seed, warmup, samples, delta,
|
|
191
|
+
draws.ctypes.data_as(ctypes.POINTER(ctypes.c_double)),
|
|
192
|
+
err, len(err))
|
|
193
|
+
if rc != 0:
|
|
194
|
+
raise RuntimeError(err.value.decode())
|
|
195
|
+
|
|
196
|
+
n_wa = _lib.stanli_wa_n_columns(self._m)
|
|
197
|
+
if n_wa > 0:
|
|
198
|
+
names = [_lib.stanli_wa_column_name(self._m, i).decode()
|
|
199
|
+
for i in range(n_wa)]
|
|
200
|
+
_lib.stanli_wa_seed(self._m, seed)
|
|
201
|
+
out = np.empty((samples, n_wa))
|
|
202
|
+
row = np.empty(n_wa)
|
|
203
|
+
for s in range(samples):
|
|
204
|
+
if _lib.stanli_wa_row(
|
|
205
|
+
self._m,
|
|
206
|
+
draws[s].ctypes.data_as(
|
|
207
|
+
ctypes.POINTER(ctypes.c_double)),
|
|
208
|
+
row.ctypes.data_as(
|
|
209
|
+
ctypes.POINTER(ctypes.c_double))) != 0:
|
|
210
|
+
raise RuntimeError(
|
|
211
|
+
f"write_array failed on draw {s}")
|
|
212
|
+
out[s] = row
|
|
213
|
+
return {name: out[:, i] for i, name in enumerate(names)}
|
|
214
|
+
|
|
215
|
+
n_con = len(self.constrained_names)
|
|
216
|
+
con = np.empty((samples, n_con))
|
|
217
|
+
row = np.empty(n_con)
|
|
218
|
+
for s in range(samples):
|
|
219
|
+
_lib.stanli_constrain(
|
|
220
|
+
self._m,
|
|
221
|
+
draws[s].ctypes.data_as(ctypes.POINTER(ctypes.c_double)),
|
|
222
|
+
row.ctypes.data_as(ctypes.POINTER(ctypes.c_double)))
|
|
223
|
+
con[s] = row
|
|
224
|
+
return {name: con[:, i]
|
|
225
|
+
for i, name in enumerate(self.constrained_names)}
|
stanli/_bin/stanc.exe
ADDED
|
Binary file
|
stanli/_bin/stanli.dll
ADDED
|
Binary file
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: stanli
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Stan Language Interpreter: compile and sample Stan models with no C++ toolchain
|
|
5
|
+
Home-page: https://github.com/seantalts/stanli
|
|
6
|
+
Author: Sean Talts
|
|
7
|
+
License: BSD-3-Clause
|
|
8
|
+
Project-URL: Source, https://github.com/seantalts/stanli
|
|
9
|
+
Project-URL: Issues, https://github.com/seantalts/stanli/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/seantalts/stanli/blob/main/CHANGELOG.md
|
|
11
|
+
Project-URL: Benchmarks, https://github.com/seantalts/stanli/blob/main/docs/benchmarks.md
|
|
12
|
+
Project-URL: Model coverage, https://github.com/seantalts/stanli/blob/main/docs/corpus-status.md
|
|
13
|
+
Keywords: stan,bayesian,mcmc,nuts,hmc,statistics,probabilistic-programming,inference,autodiff
|
|
14
|
+
Classifier: Development Status :: 3 - Alpha
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: License :: OSI Approved :: BSD License
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
24
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
25
|
+
Classifier: Programming Language :: C++
|
|
26
|
+
Classifier: Topic :: Scientific/Engineering :: Mathematics
|
|
27
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
28
|
+
Classifier: Operating System :: MacOS :: MacOS X
|
|
29
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
30
|
+
Requires-Python: >=3.9
|
|
31
|
+
Description-Content-Type: text/markdown
|
|
32
|
+
License-File: stanli/LICENSE
|
|
33
|
+
Requires-Dist: numpy>=1.22
|
|
34
|
+
Dynamic: author
|
|
35
|
+
Dynamic: classifier
|
|
36
|
+
Dynamic: description
|
|
37
|
+
Dynamic: description-content-type
|
|
38
|
+
Dynamic: home-page
|
|
39
|
+
Dynamic: keywords
|
|
40
|
+
Dynamic: license
|
|
41
|
+
Dynamic: license-file
|
|
42
|
+
Dynamic: project-url
|
|
43
|
+
Dynamic: requires-dist
|
|
44
|
+
Dynamic: requires-python
|
|
45
|
+
Dynamic: summary
|
|
46
|
+
|
|
47
|
+
# stanli
|
|
48
|
+
|
|
49
|
+
**The Stan Language Interpreter.** Compile and sample Stan models with no
|
|
50
|
+
C++ toolchain on the machine.
|
|
51
|
+
|
|
52
|
+
[](https://pypi.org/project/stanli/)
|
|
53
|
+
[](https://pypi.org/project/stanli/)
|
|
54
|
+
[](https://github.com/seantalts/stanli/blob/main/LICENSE)
|
|
55
|
+
[](https://github.com/seantalts/stanli/actions/workflows/wheels.yml)
|
|
56
|
+
|
|
57
|
+
```console
|
|
58
|
+
pip install stanli
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
That is the whole install. No compiler, no `make`, no CmdStan checkout, no
|
|
62
|
+
multi-minute first-run build. One wheel, one shared library, under seven
|
|
63
|
+
megabytes.
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
import stanli
|
|
67
|
+
|
|
68
|
+
model = stanli.Model(stan_file="eight_schools.stan", data="data.json")
|
|
69
|
+
draws = model.sample(seed=1, warmup=1000, samples=1000)
|
|
70
|
+
|
|
71
|
+
draws["mu"].mean() # one numpy array of draws per constrained parameter
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
Model preparation takes milliseconds, so the first draw arrives about 20x
|
|
75
|
+
sooner than a toolchain that compiles C++ per model.
|
|
76
|
+
|
|
77
|
+
## How it works
|
|
78
|
+
|
|
79
|
+
Every Stan model is a composition of a fixed vocabulary of operations:
|
|
80
|
+
densities, constraint transforms, linear algebra, elementwise math. stanli
|
|
81
|
+
ships those precompiled and turns each model into *data*, a static graph of
|
|
82
|
+
ops over flat preallocated buffers, instead of generating and compiling C++
|
|
83
|
+
per model.
|
|
84
|
+
|
|
85
|
+
```
|
|
86
|
+
model.stan + data.json
|
|
87
|
+
| stanc3, the official OCaml compiler, linked into the library
|
|
88
|
+
v
|
|
89
|
+
transformed MIR
|
|
90
|
+
| lowering: transformed data evaluated eagerly, data-bound loops unrolled,
|
|
91
|
+
| then periodic regions re-rolled back into vectorized ops
|
|
92
|
+
v
|
|
93
|
+
op graph over preallocated value/adjoint arenas
|
|
94
|
+
| forward sweep = log density, reverse sweep = gradient
|
|
95
|
+
v
|
|
96
|
+
NUTS with diagonal-metric adaptation -> draws
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
The graph doubles as the autodiff tape, so a reverse sweep is a backwards
|
|
100
|
+
loop over an array rather than a walk through a pointer-chasing tape, and
|
|
101
|
+
steady-state gradient evaluation allocates nothing.
|
|
102
|
+
|
|
103
|
+
Two things are not reimplemented, which is what makes the results
|
|
104
|
+
trustworthy: the compiler is the real stanc3, linked in-process, so the
|
|
105
|
+
Stan language behaves as the official toolchain makes it behave; and the
|
|
106
|
+
math is unmodified stan-math, the same code CmdStan runs.
|
|
107
|
+
|
|
108
|
+
## Correctness
|
|
109
|
+
|
|
110
|
+
Nothing here ships on "looks close".
|
|
111
|
+
|
|
112
|
+
**<!--gen:corpus_verified_of-->118 of 120<!--/gen--> posteriordb models**
|
|
113
|
+
are differentially verified against CmdStan: same model, same data, same
|
|
114
|
+
evaluation point, comparing the log density and every single gradient
|
|
115
|
+
component. **<!--gen:corpus_bitwise-->45<!--/gen--> agree bitwise.** The
|
|
116
|
+
worst deviation across the entire corpus is
|
|
117
|
+
**<!--gen:corpus_worst-->2.6e-12<!--/gen--> relative**.
|
|
118
|
+
|
|
119
|
+
The two exceptions are documented rather than hidden. `sir`'s ODE solution
|
|
120
|
+
dips about 1e-9 below a declared lower bound at the shared evaluation point,
|
|
121
|
+
where CmdStan rejects it too; `kronecker_gp` matches on the log density and
|
|
122
|
+
436 of 438 gradients, differing on the two that flow through eigenvectors of
|
|
123
|
+
a nearly degenerate covariance matrix.
|
|
124
|
+
|
|
125
|
+
Full per-model accuracy table:
|
|
126
|
+
[docs/corpus-status.md](https://github.com/seantalts/stanli/blob/main/docs/corpus-status.md)
|
|
127
|
+
|
|
128
|
+
## Performance
|
|
129
|
+
|
|
130
|
+
Per-gradient latency against CmdStan, same models, same evaluation point,
|
|
131
|
+
both sides `-O3` with FP contraction pinned off:
|
|
132
|
+
|
|
133
|
+
<!--gen:bench_table_us-->
|
|
134
|
+
| model | params | stanli | CmdStan | speedup |
|
|
135
|
+
| --- | ---: | ---: | ---: | ---: |
|
|
136
|
+
| `radon_pooled` | 3 | 52.9 us | 320.9 us | **6.1x** |
|
|
137
|
+
| `arK` | 7 | 2.4 us | 12.5 us | **5.2x** |
|
|
138
|
+
| `radon_hierarchical_intercept_centered` | 391 | 111.6 us | 569.1 us | **5.1x** |
|
|
139
|
+
| `radon_county_intercept` | 388 | 89.7 us | 431.6 us | **4.8x** |
|
|
140
|
+
| `nes` | 10 | 19.7 us | 69.3 us | **3.5x** |
|
|
141
|
+
| `eight_schools_noncentered` | 10 | 0.23 us | 0.74 us | **3.3x** |
|
|
142
|
+
| `election88_full` | 90 | 295.3 us | 902.0 us | **3.0x** |
|
|
143
|
+
| `bym2_offset_only` | 3845 | 39.6 us | 114.6 us | **2.9x** |
|
|
144
|
+
| `dogs` | 3 | 22.0 us | 63.7 us | **2.9x** |
|
|
145
|
+
| `kidscore_momiq` | 3 | 1.9 us | 4.9 us | **2.6x** |
|
|
146
|
+
| `lsat_model` | 1006 | 45.5 us | 91.2 us | **2.0x** |
|
|
147
|
+
| `state_space_stochastic_level_stochastic_seasonal` | 389 | 17.2 us | 26.3 us | **1.5x** |
|
|
148
|
+
| `normal_mixture` | 3 | 79.0 us | 88.2 us | **1.1x** |
|
|
149
|
+
| `low_dim_gauss_mix` | 5 | 88.9 us | 98.3 us | **1.1x** |
|
|
150
|
+
| `wells_dist100ars_model` | 3 | 17.4 us | 19.0 us | **1.1x** |
|
|
151
|
+
| `radon_county` | 389 | 83.2 us | 82.1 us | **1.0x** |
|
|
152
|
+
| `arma11` | 4 | 6.7 us | 6.2 us | 0.93x |
|
|
153
|
+
| `diamonds` | 26 | 35.4 us | 31.5 us | 0.89x |
|
|
154
|
+
| `garch11` | 4 | 11.2 us | 9.7 us | 0.86x |
|
|
155
|
+
| `hmm_drive_0` | 6 | 173.0 us | 132.8 us | 0.77x |
|
|
156
|
+
| `hmm_example` | 4 | 36.3 us | 27.1 us | 0.75x |
|
|
157
|
+
| `ldaK2` | 7 | 145.9 us | 104.1 us | 0.71x |
|
|
158
|
+
| `iohmm_reg` | 29 | 545.2 us | 320.3 us | 0.59x |
|
|
159
|
+
<!--/gen-->
|
|
160
|
+
|
|
161
|
+
The wins come from op granularity. CmdStan's var tape allocates, walks, and
|
|
162
|
+
frees one node per scalar operation per leapfrog step; stanli pays a fixed
|
|
163
|
+
cost per *op*, and a vectorized statement over N elements amortizes that to
|
|
164
|
+
nothing. Across the whole posteriordb corpus the median is
|
|
165
|
+
<!--gen:corpus_median-->2.07x<!--/gen--> and
|
|
166
|
+
<!--gen:corpus_at_par-->93<!--/gen--> of
|
|
167
|
+
<!--gen:corpus_n_grad-->119<!--/gen--> models are at or above CmdStan.
|
|
168
|
+
|
|
169
|
+
The losses are honest and understood, and they are all one shape: a
|
|
170
|
+
recurrence. `hmm_*`, `garch11` and `arma11` step through time with each
|
|
171
|
+
step reading the last one's parameter-dependent result, which nothing can
|
|
172
|
+
vectorize, so the work is scalar on both sides and CmdStan's generated C++
|
|
173
|
+
is the faster way to run scalar work. `ldaK2` is a mixture over more than
|
|
174
|
+
two components, which the fusion pass does not yet widen.
|
|
175
|
+
|
|
176
|
+
ODE models are the other place stanli is still behind. An ODE right-hand
|
|
177
|
+
side is the one user function that cannot be inlined at lowering time,
|
|
178
|
+
since the integrator picks the times; it now compiles into a flat register
|
|
179
|
+
machine instead of being tree-walked, and the forward sweep keeps the
|
|
180
|
+
sensitivities it was already computing instead of solving twice. Together
|
|
181
|
+
that is 29x to 39x faster than the tree-walking interpreter it replaces,
|
|
182
|
+
which puts `lotka_volterra` and `soil_incubation` at 0.58x and 0.63x of
|
|
183
|
+
CmdStan rather than 0.015x.
|
|
184
|
+
|
|
185
|
+
Method and full table:
|
|
186
|
+
[docs/benchmarks.md](https://github.com/seantalts/stanli/blob/main/docs/benchmarks.md)
|
|
187
|
+
|
|
188
|
+
## API
|
|
189
|
+
|
|
190
|
+
The surface is small on purpose.
|
|
191
|
+
|
|
192
|
+
```python
|
|
193
|
+
import stanli
|
|
194
|
+
|
|
195
|
+
# A path to a .stan file, or the model source directly.
|
|
196
|
+
model = stanli.Model(stan_file="model.stan", data="data.json")
|
|
197
|
+
model = stanli.Model(stan_code=src, data={"J": 8, "y": y, "sigma": sigma})
|
|
198
|
+
|
|
199
|
+
model.n_unconstrained # length of the unconstrained vector
|
|
200
|
+
model.constrained_names # ['mu', 'tau', 'theta.1', ...]
|
|
201
|
+
|
|
202
|
+
lp, grad = model.log_prob_grad(q) # sampling log density and its gradient
|
|
203
|
+
|
|
204
|
+
draws = model.sample(seed=1, warmup=1000, samples=1000, delta=0.8)
|
|
205
|
+
draws["mu"] # ndarray of length `samples`
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
`data` accepts a path to a JSON file or a dict of Python scalars, lists, and
|
|
209
|
+
numpy arrays. `sample` returns one array of constrained draws per scalar
|
|
210
|
+
parameter, named the way CmdStan names them, so `theta` declared as
|
|
211
|
+
`vector[8]` arrives as `theta.1` through `theta.8`.
|
|
212
|
+
|
|
213
|
+
## Platforms
|
|
214
|
+
|
|
215
|
+
Wheels for macOS (arm64 and x86_64) and Linux (x86_64 and aarch64,
|
|
216
|
+
manylinux_2_28). Windows is not built yet; it needs a mingw-w64 toolchain,
|
|
217
|
+
because stan-math does not build under MSVC.
|
|
218
|
+
|
|
219
|
+
The installed library is 21.3 MB, which is the trade this design makes:
|
|
220
|
+
ship the compiler and every kernel once, so that nothing is ever built on
|
|
221
|
+
the user's machine. Roughly half of that is the embedded stanc3 and
|
|
222
|
+
somewhat under half is stan-math. The interpreter and NUTS together are
|
|
223
|
+
about 410 KB.
|
|
224
|
+
|
|
225
|
+
## Status
|
|
226
|
+
|
|
227
|
+
Early, and deliberately narrow. The sampler is Stan's own NUTS with
|
|
228
|
+
diagonal-metric adaptation. Known limits, stated plainly:
|
|
229
|
+
|
|
230
|
+
- `sample()` returns declared parameters only. Transformed parameters and
|
|
231
|
+
generated quantities are computed by the runtime and written by the
|
|
232
|
+
command line tool, but are not exposed through the Python API yet, so
|
|
233
|
+
the non-centered eight schools gives you `mu`, `tau`, and
|
|
234
|
+
`theta_tilde`, not `theta`.
|
|
235
|
+
- No variational inference, no optimization, no multi-chain threading.
|
|
236
|
+
- No convergence diagnostics. Pair it with ArviZ or similar for now.
|
|
237
|
+
|
|
238
|
+
What is here is verified against CmdStan model by model, and every number
|
|
239
|
+
on this page is reproducible from the repository.
|
|
240
|
+
|
|
241
|
+
- Source, issues, and roadmap:
|
|
242
|
+
[github.com/seantalts/stanli](https://github.com/seantalts/stanli)
|
|
243
|
+
- License: BSD-3-Clause, matching Stan's own.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
stanli/LICENSE,sha256=1rSQPamG3JBHPWryYJwX8eNWFCQpIESVpgnjdgBgXeY,1526
|
|
2
|
+
stanli/THIRD_PARTY_LICENSES.md,sha256=KR6DxDOiUX_CAd-JCKAL-5LXivN8C72H-MZfAADC-ic,2071
|
|
3
|
+
stanli/__init__.py,sha256=6lFYuf2lXiCIVPqIQKPhdfVu_mHioS43SC5zNZjmev0,9692
|
|
4
|
+
stanli/_bin/stanc.exe,sha256=5TCCyJXWfWVFWpETWNApeM1bqcH7hkXCcLQ-bNjlcjQ,12322816
|
|
5
|
+
stanli/_bin/stanli.dll,sha256=tVFEK5GzeyBraK7zjSmPdp3yv3XxJH7RlSg2GWyACpc,16671329
|
|
6
|
+
stanli-0.3.0.dist-info/licenses/stanli/LICENSE,sha256=1rSQPamG3JBHPWryYJwX8eNWFCQpIESVpgnjdgBgXeY,1526
|
|
7
|
+
stanli-0.3.0.dist-info/METADATA,sha256=w6sDUwsNuNnmNMQAuGj7w-VAggk7b9n_PGq4ea2HhGc,10739
|
|
8
|
+
stanli-0.3.0.dist-info/WHEEL,sha256=D6hG7Lx54YqYkSprY2f2Fnm_KmrXgxuCq8q_2w9SpLg,98
|
|
9
|
+
stanli-0.3.0.dist-info/top_level.txt,sha256=jDP3ch9Mp2Ug4AO8eePaYqwHMVhcnPKmifI5FZ95a4A,7
|
|
10
|
+
stanli-0.3.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
BSD 3-Clause License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026, Sean Talts
|
|
4
|
+
|
|
5
|
+
Redistribution and use in source and binary forms, with or without
|
|
6
|
+
modification, are permitted provided that the following conditions are met:
|
|
7
|
+
|
|
8
|
+
1. Redistributions of source code must retain the above copyright notice,
|
|
9
|
+
this list of conditions and the following disclaimer.
|
|
10
|
+
|
|
11
|
+
2. Redistributions in binary form must reproduce the above copyright
|
|
12
|
+
notice, this list of conditions and the following disclaimer in the
|
|
13
|
+
documentation and/or other materials provided with the distribution.
|
|
14
|
+
|
|
15
|
+
3. Neither the name of the copyright holder nor the names of its
|
|
16
|
+
contributors may be used to endorse or promote products derived from
|
|
17
|
+
this software without specific prior written permission.
|
|
18
|
+
|
|
19
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
20
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
21
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
|
|
22
|
+
ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE
|
|
23
|
+
LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
|
|
24
|
+
CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
|
|
25
|
+
SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
|
|
26
|
+
INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
|
|
27
|
+
CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
|
|
28
|
+
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
|
|
29
|
+
POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
stanli
|