funcyflows 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/samplers/latent_pcn.py +90 -94
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/transports/continuous/grid_fields.py +34 -15
- {funcyflows-0.2.0 → funcyflows-0.2.2}/PKG-INFO +11 -3
- {funcyflows-0.2.0 → funcyflows-0.2.2}/README.md +7 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/funcyflows.egg-info/PKG-INFO +11 -3
- {funcyflows-0.2.0 → funcyflows-0.2.2}/funcyflows.egg-info/requires.txt +0 -1
- {funcyflows-0.2.0 → funcyflows-0.2.2}/pyproject.toml +5 -3
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/__init__.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/base_measures/__init__.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/base_measures/abstract_reference_measure.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/base_measures/bases.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/base_measures/gaussian_reference_measure.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/diagnostics/__init__.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/diagnostics/coverage.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/diagnostics/importance.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/examples/__init__.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/examples/__main__.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/examples/_common.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/examples/bimodal_posterior.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/examples/cloud_inpainting.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/examples/nonstationary.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/examples/phase_inpainting.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/examples/positivity.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/examples/ring_inpainting.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/objectives/__init__.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/objectives/alpha_divergence.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/objectives/flow_matching.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/objectives/negative_logl.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/objectives/reverse_kl.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/samplers/__init__.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/transports/__init__.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/transports/abstract_transformation.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/transports/continuous/__init__.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/transports/continuous/base_continuous.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/transports/continuous/conditioners.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/transports/continuous/vector_fields.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/transports/layers/__init__.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/transports/layers/base_discrete.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/transports/layers/layer_classes.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/utils/__init__.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/utils/gaussian_misfit.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/utils/train.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/LICENSE +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/funcyflows.egg-info/SOURCES.txt +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/funcyflows.egg-info/dependency_links.txt +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/funcyflows.egg-info/top_level.txt +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/setup.cfg +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/tests/test_basis.py +0 -0
- {funcyflows-0.2.0 → funcyflows-0.2.2}/tests/test_measures.py +0 -0
|
@@ -1,86 +1,5 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
What problem this solves
|
|
4
|
-
------------------------
|
|
5
|
-
You have a posterior on coefficients,
|
|
6
|
-
|
|
7
|
-
pi(dv) proportional to exp(-Phi(v)) rho(dv),
|
|
8
|
-
|
|
9
|
-
with Phi the misfit (minus log likelihood) and rho a prior. Sampling it directly with a random-walk
|
|
10
|
-
Metropolis proposal fails as the number of modes M grows: to keep the acceptance rate away from
|
|
11
|
-
zero the step size has to shrink like M^(-1/2), so the chain needs O(M) steps to move anywhere.
|
|
12
|
-
|
|
13
|
-
pCN (Cotter, Roberts, Stuart & White, Statistical Science 2013, arXiv:1202.0709) fixes that for a
|
|
14
|
-
GAUSSIAN reference measure mu0. Its proposal is
|
|
15
|
-
|
|
16
|
-
z' = sqrt(1 - beta^2) z + beta xi, xi ~ mu0,
|
|
17
|
-
|
|
18
|
-
which leaves mu0 exactly invariant -- if z ~ mu0 then (z, z') is jointly Gaussian and exchangeable,
|
|
19
|
-
so the pair is reversible. Because the proposal already carries the reference measure, the
|
|
20
|
-
Metropolis ratio keeps only the likelihood:
|
|
21
|
-
|
|
22
|
-
accept with probability min(1, exp(Phi(z) - Phi(z'))).
|
|
23
|
-
|
|
24
|
-
No prior density appears, nothing scales with M, and beta can stay O(1) at any truncation.
|
|
25
|
-
|
|
26
|
-
What the "latent" part adds
|
|
27
|
-
---------------------------
|
|
28
|
-
pCN needs the reference measure to be Gaussian, which is a real restriction: it means the geometry
|
|
29
|
-
the sampler assumes is the geometry of mu0. If the posterior is curved, multimodal or far from the
|
|
30
|
-
prior, a Gaussian-geometry sampler crawls.
|
|
31
|
-
|
|
32
|
-
So run pCN not on v but on the flow's latent variable z, where v = T(z) and T is the trained
|
|
33
|
-
transport. The reference measure there IS Gaussian by construction -- it is the flow's base measure
|
|
34
|
-
-- and the flow has already absorbed the awkward geometry. This is transport-map preconditioning
|
|
35
|
-
(Parno & Marzouk, arXiv:1412.5492; the neural version, with HMC rather than pCN, is NeuTra,
|
|
36
|
-
arXiv:1903.03704).
|
|
37
|
-
|
|
38
|
-
The potential to hand it
|
|
39
|
-
------------------------
|
|
40
|
-
The chain targets nu(dz) proportional to exp(-Psi(z)) mu0(dz), and v = T(z) is then distributed as
|
|
41
|
-
|
|
42
|
-
exp(-Psi(T^-1 v)) q(dv), q = T_# mu0 (the flow's pushforward).
|
|
43
|
-
|
|
44
|
-
Setting that equal to pi gives the rule, once and for all:
|
|
45
|
-
|
|
46
|
-
Psi(z) = Phi(T z) + log (dq/dmu0)(T z) - log (drho/dmu0)(T z).
|
|
47
|
-
|
|
48
|
-
Two cases cover everything in this package, and they are NOT the same call:
|
|
49
|
-
|
|
50
|
-
* T is a learned PRIOR, so rho = q. The two log terms cancel:
|
|
51
|
-
|
|
52
|
-
Psi = Phi . T potential = misfit
|
|
53
|
-
|
|
54
|
-
The flow's Jacobian never enters, so ANY trained prior flow works and the trace need not even
|
|
55
|
-
be available. This is what the inpainting examples used.
|
|
56
|
-
|
|
57
|
-
* T approximates the POSTERIOR and the prior is the base measure, rho = mu0. Then:
|
|
58
|
-
|
|
59
|
-
Psi = Phi . T + log_rn_at . T potential = lambda v: misfit(v) + flow.log_rn_at(v)
|
|
60
|
-
|
|
61
|
-
Here the trace does enter, exactly. An estimated (Hutchinson) trace makes the chain converge to
|
|
62
|
-
the wrong measure, not to a noisy version of the right one. See bimodal_posterior.py.
|
|
63
|
-
|
|
64
|
-
`potential` is called on COEFFICIENTS v, never on z; this function does the transport for you.
|
|
65
|
-
|
|
66
|
-
When it works and when it does not
|
|
67
|
-
----------------------------------
|
|
68
|
-
If the flow is good, q is close to pi, Psi is nearly constant, and almost every proposal is accepted
|
|
69
|
-
even at beta close to 1 -- the chain returns near-independent samples. If the flow is poor, or the
|
|
70
|
-
likelihood is much sharper than anything the flow has seen, the adapted beta collapses and the
|
|
71
|
-
chain barely moves. That failure is quiet: the chains sit on their initialisation and the output
|
|
72
|
-
looks like a posterior. `info["moved"]` is reported for exactly this reason -- see its description
|
|
73
|
-
in the Returns section, and prefer an amortised conditional flow when it is small.
|
|
74
|
-
"""
|
|
75
|
-
import math
|
|
76
|
-
|
|
77
|
-
import torch
|
|
78
|
-
|
|
79
|
-
try: # progress bar is optional
|
|
80
|
-
from tqdm.auto import trange
|
|
81
|
-
except ImportError: # pragma: no cover
|
|
82
|
-
def trange(n, **kwargs):
|
|
83
|
-
return range(n)
|
|
1
|
+
import math, torch
|
|
2
|
+
from tqdm.auto import trange
|
|
84
3
|
|
|
85
4
|
|
|
86
5
|
def _as_coefficients(result):
|
|
@@ -93,7 +12,10 @@ def _as_coefficients(result):
|
|
|
93
12
|
@torch.no_grad()
|
|
94
13
|
def latent_pcn(flow, potential, num_chains=64, num_steps=2000, beta=0.2, init=None, burn=None,
|
|
95
14
|
thin=1, adapt_to=None, context=None, progress=True, generator=None):
|
|
96
|
-
r"""Sample a posterior by running pCN in the latent space of `flow`.
|
|
15
|
+
r"""Sample a posterior by running pCN in the latent space of `flow`. If unfamiliar with pCN,
|
|
16
|
+
it's an MCMC method that makes proposals by scaling the current MCMC sample and adding
|
|
17
|
+
an adjusted sample from the prior. Scales better with dimension than standard MCMC because
|
|
18
|
+
of the implicit geometry added along with the prior samples.
|
|
97
19
|
|
|
98
20
|
Parameters
|
|
99
21
|
----------
|
|
@@ -104,8 +26,8 @@ def latent_pcn(flow, potential, num_chains=64, num_steps=2000, beta=0.2, init=No
|
|
|
104
26
|
training, and every chain step costs one solve.
|
|
105
27
|
potential : callable
|
|
106
28
|
``potential(v) -> [batch]``, the NEGATIVE log of the target's density relative to the
|
|
107
|
-
flow's pushforward, as a function of COEFFICIENTS. See
|
|
108
|
-
the two forms you want
|
|
29
|
+
flow's pushforward, as a function of COEFFICIENTS. See explanation further down
|
|
30
|
+
in this docstring for which of the two forms you want.
|
|
109
31
|
num_chains : int
|
|
110
32
|
Chains advanced in lockstep. They share one ``transport`` call per step, so more chains are
|
|
111
33
|
nearly free up to the batch the ODE can hold.
|
|
@@ -140,14 +62,88 @@ def latent_pcn(flow, potential, num_chains=64, num_steps=2000, beta=0.2, init=No
|
|
|
140
62
|
draws : Tensor [num_kept * num_chains, M]
|
|
141
63
|
Post-burn-in states in COEFFICIENT space, chains concatenated.
|
|
142
64
|
info : dict
|
|
143
|
-
``acceptance``
|
|
144
|
-
``acceptance_burn``
|
|
145
|
-
|
|
146
|
-
``
|
|
147
|
-
``
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
65
|
+
- ``acceptance`` -- mean acceptance after burn-in (the number that matters).
|
|
66
|
+
- ``acceptance_burn`` -- mean acceptance during burn-in, for checking the adaptation
|
|
67
|
+
worked.
|
|
68
|
+
- ``beta`` -- final step size.
|
|
69
|
+
- ``potential`` -- mean potential over the kept states; a sanity check that it plateaued.
|
|
70
|
+
- ``moved`` -- mean distance travelled from the starting states, relative to their own
|
|
71
|
+
norm. Below ~0.1 the chain never left its initialisation and the "posterior" you are
|
|
72
|
+
looking at is whatever you passed as ``init``. Reported only when ``init`` is given.
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
Preconditioned Crank-Nicolson MCMC in the latent space of a trained flow.
|
|
76
|
+
|
|
77
|
+
What we're doing
|
|
78
|
+
------------------------
|
|
79
|
+
You have a posterior on coefficients,
|
|
80
|
+
|
|
81
|
+
pi(dv) proportional to exp(-Phi(v)) rho(dv),
|
|
82
|
+
|
|
83
|
+
with Phi the misfit (minus log likelihood) and rho a prior. Sampling it directly with a random-walk
|
|
84
|
+
Metropolis proposal fails as the number of modes M grows: to keep the acceptance rate away from
|
|
85
|
+
zero the step size has to shrink like M^(-1/2), so the chain needs O(M) steps to move anywhere.
|
|
86
|
+
|
|
87
|
+
pCN (Cotter, Roberts, Stuart & White, Statistical Science 2013, arXiv:1202.0709) fixes that for a
|
|
88
|
+
GAUSSIAN reference measure mu0. Its proposal is
|
|
89
|
+
|
|
90
|
+
z' = sqrt(1 - beta^2) z + beta xi, xi ~ mu0,
|
|
91
|
+
|
|
92
|
+
which leaves mu0 exactly invariant -- if z ~ mu0 then (z, z') is jointly Gaussian and exchangeable,
|
|
93
|
+
so the pair is reversible. Because the proposal already carries the reference measure, the
|
|
94
|
+
Metropolis ratio keeps only the likelihood:
|
|
95
|
+
|
|
96
|
+
accept with probability min(1, exp(Phi(z) - Phi(z'))).
|
|
97
|
+
|
|
98
|
+
No prior density appears, nothing scales with M, and beta can stay O(1) at any truncation.
|
|
99
|
+
|
|
100
|
+
What the "latent" part adds
|
|
101
|
+
---------------------------
|
|
102
|
+
pCN needs the reference measure to be Gaussian, because the Gaussian family is closed under this operation
|
|
103
|
+
and almost nothing else is: with v and xi independent N(0,C), the combination has covariance
|
|
104
|
+
|
|
105
|
+
(1-beta^2)C + beta^2 C = C,
|
|
106
|
+
|
|
107
|
+
and a mean-zero Gaussian is determined by its covariance, so v' is N(0,C) again.
|
|
108
|
+
|
|
109
|
+
If the posterior is curved, multimodal or far from the prior, a Gaussian-geometry sampler does not do very well.
|
|
110
|
+
|
|
111
|
+
The constraint is on the REFERENCE measure, not the prior. An arbitrary prior rho enters through -log(drho/dmu0)
|
|
112
|
+
in the potential (formula above) with no change to the algorithm, or gets absorbed into the transport, which is what a learned prior flow is.
|
|
113
|
+
What must be closed under the proposal is the reference: Gaussian is the only finite-variance choice,
|
|
114
|
+
though alpha-stable references work with rebalanced coefficients.
|
|
115
|
+
|
|
116
|
+
But, we run pCN not on v but on the flow's latent variable z, where v = T(z) and T is the trained
|
|
117
|
+
transport. The reference measure there *IS* Gaussian by construction -- it is the flow's base measure
|
|
118
|
+
-- and the flow has already absorbed the awkward geometry. This is transport-map preconditioning
|
|
119
|
+
(Parno & Marzouk, arXiv:1412.5492; the neural version, with HMC rather than pCN, is NeuTra,arXiv:1903.03704).
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
Which potential to pass
|
|
124
|
+
-----------------------
|
|
125
|
+
The chain targets exp(-Psi(z)) mu0(dz), so v = T(z) comes out distributed as
|
|
126
|
+
|
|
127
|
+
Psi(z) = Phi(Tz) + log (dq/dmu0)(Tz) - log (drho/dmu0)(Tz), q = T_# mu0
|
|
128
|
+
|
|
129
|
+
Two cases cover the below, they are not the same call :grimace: :
|
|
130
|
+
|
|
131
|
+
* T is a learned PRIOR, so rho = q and the log terms cancel::
|
|
132
|
+
|
|
133
|
+
potential = misfit
|
|
134
|
+
|
|
135
|
+
The flow's Jacobian never enters, so any trained prior flow works and the trace
|
|
136
|
+
need not even be computable.
|
|
137
|
+
|
|
138
|
+
* T approximates the POSTERIOR and the prior is the base measure, rho = mu0::
|
|
139
|
+
|
|
140
|
+
potential = lambda v: misfit(v) + flow.log_rn_at(v)
|
|
141
|
+
|
|
142
|
+
Here the trace DOES enter, and it unfortunately has to be exact.
|
|
143
|
+
A Hutchinson estimate doesn't give a noisy version of the 'correct' chain:
|
|
144
|
+
|
|
145
|
+
- the randomness lands in the accept test
|
|
146
|
+
- so the chain has a different invariant measure.
|
|
151
147
|
"""
|
|
152
148
|
if not 0 < beta <= 1:
|
|
153
149
|
raise ValueError(f"beta must be in (0, 1]; got {beta}")
|
|
@@ -52,15 +52,18 @@ class GridTransform(torch.nn.Module):
|
|
|
52
52
|
Notes
|
|
53
53
|
-----
|
|
54
54
|
1. The grid is **cell-edge**: xₚ = p/G for p = 0 … G-1.
|
|
55
|
+
|
|
55
56
|
- Not cell-centred, so the FFT needs no half-sample phase factor.
|
|
56
57
|
- On a periodic uniform grid the rectangle rule is spectrally accurate for band-limited
|
|
57
58
|
functions, so nothing fancier (trapezium, Gauss) buys anything for a Fourier basis.
|
|
58
59
|
|
|
59
60
|
2. Dense path. Store Φ as a [Gᵈ, M] buffer and matrix-multiply.
|
|
61
|
+
|
|
60
62
|
- Works for any basis, any grid size.
|
|
61
63
|
- Cost O(B·M·Gᵈ) per direction, and O(Gᵈ·M) of memory.
|
|
62
64
|
|
|
63
65
|
3. FFT path. Separable Fourier bases only.
|
|
66
|
+
|
|
64
67
|
- One rfft/irfft per axis, the d axes done in turn with ``movedim``.
|
|
65
68
|
- Cost O(d·B·Gᵈ·log G), no dense matrix touched.
|
|
66
69
|
- The packed [M] coefficient vector is scattered into a [max_axis_mode]ᵈ tensor-product
|
|
@@ -70,6 +73,7 @@ class GridTransform(torch.nn.Module):
|
|
|
70
73
|
1/G and √2/G coming back.
|
|
71
74
|
|
|
72
75
|
4. Which path, via ``mode``:
|
|
76
|
+
|
|
73
77
|
- ``"auto"`` (default): build the FFT tables if the basis exposes ``wavenumbers``, keep the
|
|
74
78
|
FFT only if ``_agrees`` passes, otherwise fall back to dense.
|
|
75
79
|
- ``"dense"``: never try the FFT.
|
|
@@ -303,6 +307,8 @@ class PointwiseField(VectorField):
|
|
|
303
307
|
|
|
304
308
|
**Layer Diagram**
|
|
305
309
|
|
|
310
|
+
::
|
|
311
|
+
|
|
306
312
|
|
|
307
313
|
g(x,t) ū = Φ(κ(t) ⊙ ṽ) c(x,t) a(x,t)
|
|
308
314
|
└──────────────┬──────────────────┘ │
|
|
@@ -340,33 +346,40 @@ class PointwiseField(VectorField):
|
|
|
340
346
|
order one before anything nonlinear sees it. (Undone at the end; the divergence is unchanged.)
|
|
341
347
|
This is done with the `mode_scale` attributes.
|
|
342
348
|
|
|
343
|
-
2. Go to the grid. Evaluate the whitened function at G uniformly spaced points.
|
|
349
|
+
2. Go to the grid. Evaluate the whitened function at G uniformly spaced points.
|
|
350
|
+
|
|
344
351
|
- Also build a low-pass copy: keep only the first few modes, each scaled by a learned number, and evaluate
|
|
345
|
-
|
|
346
|
-
|
|
352
|
+
that too. This gives each grid point a local average to compare itself against.
|
|
353
|
+
Giving more context on the 'neighbourhoods' of the points.
|
|
347
354
|
|
|
348
355
|
- The grid transformations are handled by, you guessed it, `GridTransform`.
|
|
349
356
|
- Do you ever worry that you make variable names __too__ obvious? Just me? Neat
|
|
350
357
|
|
|
351
358
|
3. Form the pre-activation at every grid point
|
|
359
|
+
|
|
352
360
|
- The operation in english is
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
361
|
+
|
|
362
|
+
- a *learned* gain times the function value,
|
|
363
|
+
- plus the low-pass copy,
|
|
364
|
+
- plus a learned offset.
|
|
365
|
+
|
|
356
366
|
- Gain and offset are smooth functions of position (a handful of low spatial modes each), so neighbouring points are treated alike.
|
|
357
367
|
- They don't depend on the functions/function coefficients being transformed. Keeping the derivative tractable.
|
|
358
368
|
|
|
359
|
-
4. Apply `tanh`, point by point.
|
|
369
|
+
4. Apply `tanh`, point by point.
|
|
370
|
+
|
|
360
371
|
- This is the only place a non-linearity comes in for this class by itself.
|
|
361
372
|
- This simplicity keeps the divergence closed-form.
|
|
362
373
|
- Use `SumField` would stack this, but not compose the non-linearity. There should be enough seeing as the transformation
|
|
363
|
-
|
|
374
|
+
is done at every integration step though.
|
|
375
|
+
|
|
376
|
+
5. Multiply by a learned amplitude, also a smooth function of position.
|
|
364
377
|
|
|
365
|
-
5. Multiply by a learned amplitude, also a smooth function of position.
|
|
366
378
|
- This decides how much velocity the layer is allowed to produce in each region.
|
|
367
379
|
- Starts small so the flow starts near the identity (controlled by init_scale) so the flow starts near the identity
|
|
368
380
|
|
|
369
381
|
6. Transform back to coefficients
|
|
382
|
+
|
|
370
383
|
- integrate against each basis function over the grid (a sum over grid points weighted by the cell area), then un-whiten.
|
|
371
384
|
|
|
372
385
|
Every learned quantity (gain, offset, amplitude, low-pass scalings) is a short cosine series in 'flow time',
|
|
@@ -635,6 +648,8 @@ class OperatorField(VectorField):
|
|
|
635
648
|
|
|
636
649
|
**Layer Diagram**
|
|
637
650
|
|
|
651
|
+
::
|
|
652
|
+
|
|
638
653
|
|
|
639
654
|
b_l(t,c)
|
|
640
655
|
│
|
|
@@ -684,7 +699,7 @@ class OperatorField(VectorField):
|
|
|
684
699
|
ℓ `lift` [C] fixed channel scale; buffer, not learned
|
|
685
700
|
z_l `state` [B, Gᵈ, C] layer state on the grid
|
|
686
701
|
W_l(t) `pointwise_stack[l]` [T_m, C, C] local channel mixing
|
|
687
|
-
κ_l(t) `spectral_stack[l]` [T_m,
|
|
702
|
+
κ_l(t) `spectral_stack[l]` [T_m,C,C,N_s] per-mode multiplier, dense in channels
|
|
688
703
|
κ_l(t)_k `multipliers` [C, C, M] the same, after both contractions
|
|
689
704
|
b_l(t,c) `bias_stack[l]` [T_m, C] per-channel offset; the context entry point
|
|
690
705
|
π(t) `project_stack` [T_m, C] channel readout
|
|
@@ -759,12 +774,14 @@ class OperatorField(VectorField):
|
|
|
759
774
|
|
|
760
775
|
1. Whiten and lift. Divide by the base-measure scales, evaluate on the grid, and copy the
|
|
761
776
|
single field into ``num_channels`` identical channels scaled by ``lift``.
|
|
777
|
+
|
|
762
778
|
- ``lift`` is a **buffer, not a parameter**: a whitened field is O(√num_active) on the grid,
|
|
763
779
|
so without the ``num_active^-1/2`` factor every ``tanh`` starts saturated and nothing --
|
|
764
780
|
neither signal nor gradient -- reaches layer 2. Same trap as ``PointwiseField``'s ``κ``.
|
|
765
781
|
- All channels start identical; they only differentiate through ``pointwise_stack``.
|
|
766
782
|
|
|
767
783
|
2. Each layer does three things and then a ``tanh``:
|
|
784
|
+
|
|
768
785
|
- ``W_l z``: mixes channels *at each grid point*, local.
|
|
769
786
|
- ``K_l z``: projects to coefficients, multiplies each mode by a learned number, and comes
|
|
770
787
|
back. Global, and the only way information crosses the grid inside the stack.
|
|
@@ -799,6 +816,7 @@ class OperatorField(VectorField):
|
|
|
799
816
|
***DANGER DANGER***
|
|
800
817
|
|
|
801
818
|
1. **The estimated trace is unbiased in log, but biased in density.**
|
|
819
|
+
|
|
802
820
|
- Hutchinson gives an unbiased :math:`\mathrm{Tr}\,J`, so ``log_rn_at`` is unbiased,
|
|
803
821
|
- BUT importance weights exponentiate it:
|
|
804
822
|
:math:`\mathbb{E}[e^{\epsilon}] \neq e^{\mathbb{E}[\epsilon]}`.
|
|
@@ -806,12 +824,13 @@ class OperatorField(VectorField):
|
|
|
806
824
|
noisy.
|
|
807
825
|
|
|
808
826
|
2. **``latent_pcn`` is invalid with this field.**
|
|
809
|
-
|
|
810
|
-
|
|
811
|
-
|
|
827
|
+
|
|
828
|
+
- Its acceptance ratio needs the exact ``log_rn_at``: a noisy one makes the chain target
|
|
829
|
+
the wrong measure rather than a noisy version of the right one.
|
|
812
830
|
|
|
813
831
|
3. **Aliasing.** ``PointwiseField`` wants ``grid_size >= 6 * k_max`` because one ``tanh``
|
|
814
832
|
reaches 3·k_max.
|
|
833
|
+
|
|
815
834
|
- Composing L of them reaches further; the harmonic amplitudes decay, so ``2 * 3^L * k_max``
|
|
816
835
|
is pessimistic.
|
|
817
836
|
- Treat ``6 * k_max`` as a floor here, not a rule -- and note that unlike ``PointwiseField``
|
|
@@ -819,6 +838,7 @@ class OperatorField(VectorField):
|
|
|
819
838
|
|
|
820
839
|
4. **No residual connections.** ``z_l = tanh(...)`` overwrites rather than adds, so the signal
|
|
821
840
|
passes through L saturating nonlinearities in series.
|
|
841
|
+
|
|
822
842
|
- Combined with the zero-init of ``spectral_stack``, layers 2…L start as near-copies of a
|
|
823
843
|
single ``tanh``
|
|
824
844
|
- depth only appears once training has *moved* the weights.
|
|
@@ -845,11 +865,10 @@ class OperatorField(VectorField):
|
|
|
845
865
|
Time-mode-first stacks contracted with ``cosines(t)``; ``spectral_stack`` starts at zero.
|
|
846
866
|
trace_samples : int
|
|
847
867
|
Hutchinson probes per evaluation.
|
|
848
|
-
supports_batched_time : bool
|
|
849
|
-
True -- ``time_value`` may carry a leading batch dimension.
|
|
850
868
|
"""
|
|
851
869
|
|
|
852
870
|
supports_batched_time = True
|
|
871
|
+
"""``time_value`` may carry a leading batch dimension."""
|
|
853
872
|
|
|
854
873
|
def __init__(self, basis, num_active, grid_size, num_channels=4, num_layers=3,
|
|
855
874
|
num_time_modes=4, num_spectral_modes=8, context_dim=0, total_time=1.0,
|
|
@@ -1,10 +1,12 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: funcyflows
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: Normalizing flows and flow matching on function spaces, with exact-trace vector fields
|
|
5
5
|
Author: Liam Pinchbeck
|
|
6
6
|
License: MIT
|
|
7
|
-
Project-URL: Homepage, https://
|
|
7
|
+
Project-URL: Homepage, https://github.com/LiamCPinchbeck/FuncFlows
|
|
8
|
+
Project-URL: Documentation, https://funcyflows.readthedocs.io/en/latest/
|
|
9
|
+
Project-URL: Source, https://github.com/LiamCPinchbeck/FuncFlows
|
|
8
10
|
Classifier: Development Status :: 3 - Alpha
|
|
9
11
|
Classifier: Intended Audience :: Science/Research
|
|
10
12
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -14,7 +16,6 @@ Requires-Python: >=3.10
|
|
|
14
16
|
Description-Content-Type: text/markdown
|
|
15
17
|
License-File: LICENSE
|
|
16
18
|
Requires-Dist: torch
|
|
17
|
-
Requires-Dist: torchvision
|
|
18
19
|
Requires-Dist: tqdm
|
|
19
20
|
Requires-Dist: scipy
|
|
20
21
|
Requires-Dist: matplotlib
|
|
@@ -26,6 +27,11 @@ Dynamic: license-file
|
|
|
26
27
|
|
|
27
28
|
Normalizing flows and flow matching on function spaces.
|
|
28
29
|
|
|
30
|
+
<p align="center">
|
|
31
|
+
<img src="https://raw.githubusercontent.com/LiamCPinchbeck/FuncFlows/main/docs/logo_flow.gif"
|
|
32
|
+
alt="a flow turning Gaussian noise into the FuncyFlows logo" width="340">
|
|
33
|
+
</p>
|
|
34
|
+
|
|
29
35
|
A function is represented by its coefficients on a Laplacian eigenbasis (`CosineBasis` or
|
|
30
36
|
`FourierBasis`), the reference measure is a Gaussian on those coefficients, and a transport is a
|
|
31
37
|
neural ODE in coefficient space. The vector fields (`LinearField`, `MatrixField`) have closed-form
|
|
@@ -37,6 +43,8 @@ likelihood training and importance reweighting usable at hundreds of modes.
|
|
|
37
43
|
pip install funcyflows # torch + tqdm
|
|
38
44
|
```
|
|
39
45
|
|
|
46
|
+
Full documentation can be found [here](https://funcyflows.readthedocs.io/en/latest/).
|
|
47
|
+
|
|
40
48
|
|
|
41
49
|
## Examples
|
|
42
50
|
|
|
@@ -2,6 +2,11 @@
|
|
|
2
2
|
|
|
3
3
|
Normalizing flows and flow matching on function spaces.
|
|
4
4
|
|
|
5
|
+
<p align="center">
|
|
6
|
+
<img src="https://raw.githubusercontent.com/LiamCPinchbeck/FuncFlows/main/docs/logo_flow.gif"
|
|
7
|
+
alt="a flow turning Gaussian noise into the FuncyFlows logo" width="340">
|
|
8
|
+
</p>
|
|
9
|
+
|
|
5
10
|
A function is represented by its coefficients on a Laplacian eigenbasis (`CosineBasis` or
|
|
6
11
|
`FourierBasis`), the reference measure is a Gaussian on those coefficients, and a transport is a
|
|
7
12
|
neural ODE in coefficient space. The vector fields (`LinearField`, `MatrixField`) have closed-form
|
|
@@ -13,6 +18,8 @@ likelihood training and importance reweighting usable at hundreds of modes.
|
|
|
13
18
|
pip install funcyflows # torch + tqdm
|
|
14
19
|
```
|
|
15
20
|
|
|
21
|
+
Full documentation can be found [here](https://funcyflows.readthedocs.io/en/latest/).
|
|
22
|
+
|
|
16
23
|
|
|
17
24
|
## Examples
|
|
18
25
|
|
|
@@ -1,10 +1,12 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: funcyflows
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: Normalizing flows and flow matching on function spaces, with exact-trace vector fields
|
|
5
5
|
Author: Liam Pinchbeck
|
|
6
6
|
License: MIT
|
|
7
|
-
Project-URL: Homepage, https://
|
|
7
|
+
Project-URL: Homepage, https://github.com/LiamCPinchbeck/FuncFlows
|
|
8
|
+
Project-URL: Documentation, https://funcyflows.readthedocs.io/en/latest/
|
|
9
|
+
Project-URL: Source, https://github.com/LiamCPinchbeck/FuncFlows
|
|
8
10
|
Classifier: Development Status :: 3 - Alpha
|
|
9
11
|
Classifier: Intended Audience :: Science/Research
|
|
10
12
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -14,7 +16,6 @@ Requires-Python: >=3.10
|
|
|
14
16
|
Description-Content-Type: text/markdown
|
|
15
17
|
License-File: LICENSE
|
|
16
18
|
Requires-Dist: torch
|
|
17
|
-
Requires-Dist: torchvision
|
|
18
19
|
Requires-Dist: tqdm
|
|
19
20
|
Requires-Dist: scipy
|
|
20
21
|
Requires-Dist: matplotlib
|
|
@@ -26,6 +27,11 @@ Dynamic: license-file
|
|
|
26
27
|
|
|
27
28
|
Normalizing flows and flow matching on function spaces.
|
|
28
29
|
|
|
30
|
+
<p align="center">
|
|
31
|
+
<img src="https://raw.githubusercontent.com/LiamCPinchbeck/FuncFlows/main/docs/logo_flow.gif"
|
|
32
|
+
alt="a flow turning Gaussian noise into the FuncyFlows logo" width="340">
|
|
33
|
+
</p>
|
|
34
|
+
|
|
29
35
|
A function is represented by its coefficients on a Laplacian eigenbasis (`CosineBasis` or
|
|
30
36
|
`FourierBasis`), the reference measure is a Gaussian on those coefficients, and a transport is a
|
|
31
37
|
neural ODE in coefficient space. The vector fields (`LinearField`, `MatrixField`) have closed-form
|
|
@@ -37,6 +43,8 @@ likelihood training and importance reweighting usable at hundreds of modes.
|
|
|
37
43
|
pip install funcyflows # torch + tqdm
|
|
38
44
|
```
|
|
39
45
|
|
|
46
|
+
Full documentation can be found [here](https://funcyflows.readthedocs.io/en/latest/).
|
|
47
|
+
|
|
40
48
|
|
|
41
49
|
## Examples
|
|
42
50
|
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "funcyflows"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.2"
|
|
8
8
|
description = "Normalizing flows and flow matching on function spaces, with exact-trace vector fields"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { text = "MIT" }
|
|
@@ -18,13 +18,15 @@ classifiers = [
|
|
|
18
18
|
"Topic :: Scientific/Engineering",
|
|
19
19
|
]
|
|
20
20
|
# only what FuncyFlows/ itself imports: torch everywhere, tqdm in utils/train.py
|
|
21
|
-
dependencies = ["torch", "
|
|
21
|
+
dependencies = ["torch", "tqdm", "scipy", "matplotlib"]
|
|
22
22
|
|
|
23
23
|
[project.optional-dependencies]
|
|
24
24
|
test = ["pytest"]
|
|
25
25
|
|
|
26
26
|
[project.urls]
|
|
27
|
-
Homepage = "https://
|
|
27
|
+
Homepage = "https://github.com/LiamCPinchbeck/FuncFlows"
|
|
28
|
+
Documentation = "https://funcyflows.readthedocs.io/en/latest/"
|
|
29
|
+
Source = "https://github.com/LiamCPinchbeck/FuncFlows"
|
|
28
30
|
|
|
29
31
|
[tool.setuptools.packages.find]
|
|
30
32
|
include = ["FuncyFlows*"]
|
|
File without changes
|
|
File without changes
|
{funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/base_measures/abstract_reference_measure.py
RENAMED
|
File without changes
|
|
File without changes
|
{funcyflows-0.2.0 → funcyflows-0.2.2}/FuncyFlows/base_measures/gaussian_reference_measure.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|