PyAntiGen 1.0.9__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- framework/AntimonyGen.py +48 -0
- framework/RxnDict_to_antimony.py +594 -0
- framework/TelluriumGen.py +16 -0
- framework/__init__.py +0 -0
- framework/antimony_utils.py +294 -0
- framework/cli.py +229 -0
- framework/data_interpolation.py +340 -0
- framework/isotopomer_tools.py +41 -0
- framework/model_generation.py +46 -0
- framework/models.py +189 -0
- framework/module_base.py +42 -0
- framework/pyantigen.py +51 -0
- framework/rate_laws.py +101 -0
- framework/reaction_creation.py +43 -0
- framework/template/Example/AntiGen_paths.py +23 -0
- framework/template/Example/Engine/Anchor_cache.py +193 -0
- framework/template/Example/Engine/Deadline.py +535 -0
- framework/template/Example/Engine/Evaluator.py +1176 -0
- framework/template/Example/Engine/Event_times.py +491 -0
- framework/template/Example/Engine/Fast_profile.py +701 -0
- framework/template/Example/Engine/Fit_cache.py +329 -0
- framework/template/Example/Engine/Identifiability.py +698 -0
- framework/template/Example/Engine/Model_optimize.py +1483 -0
- framework/template/Example/Engine/Model_simulate.py +124 -0
- framework/template/Example/Engine/Nuisance_sensitivity.py +298 -0
- framework/template/Example/Engine/Optimize.py +6862 -0
- framework/template/Example/Engine/Petab_export.py +398 -0
- framework/template/Example/Engine/Preequil_cache.py +361 -0
- framework/template/Example/Engine/Profile_checkpoint.py +399 -0
- framework/template/Example/Engine/Results.py +395 -0
- framework/template/Example/Engine/Sensitivity_analysis.py +320 -0
- framework/template/Example/Engine/Simulate.py +617 -0
- framework/template/Example/Flipflop_reference.py +401 -0
- framework/template/Example/Model_generate.py +37 -0
- framework/template/Example/Model_run.py +261 -0
- framework/template/Example/Modules/Data.py +63 -0
- framework/template/Example/Modules/Events.py +14 -0
- framework/template/Example/Modules/Experiment.py +194 -0
- framework/template/Example/Modules/Loss_config.py +61 -0
- framework/template/Example/Modules/Observed_species.py +3 -0
- framework/template/Example/Modules/Optimizer_settings.py +258 -0
- framework/template/Example/Modules/Plots.py +89 -0
- framework/template/Example/Modules/Solver_settings.py +16 -0
- framework/template/Example/Modules/Update_opt_parameters.py +24 -0
- framework/template/Example/Modules/Update_parameters.py +49 -0
- framework/template/data/ADneg.csv +27 -0
- framework/template/data/ADpos.csv +27 -0
- framework/template/data/Flipflop.csv +29 -0
- framework/template/data/make_flipflop_data.py +174 -0
- pyantigen-1.0.9.dist-info/METADATA +129 -0
- pyantigen-1.0.9.dist-info/RECORD +55 -0
- pyantigen-1.0.9.dist-info/WHEEL +5 -0
- pyantigen-1.0.9.dist-info/entry_points.txt +2 -0
- pyantigen-1.0.9.dist-info/licenses/LICENSE +21 -0
- pyantigen-1.0.9.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,698 @@
|
|
|
1
|
+
"""The slice screen: proving a parameter unbounded before profiling it.
|
|
2
|
+
|
|
3
|
+
A *slice* holds the nuisance parameters at their fitted values and varies one
|
|
4
|
+
parameter alone. A *profile* re-minimizes over the nuisance parameters at every
|
|
5
|
+
fixed value. The fitted nuisance vector is one feasible point of that
|
|
6
|
+
minimization, so for every value of every parameter
|
|
7
|
+
|
|
8
|
+
dNLL_profile(x) <= dNLL_slice(x)
|
|
9
|
+
|
|
10
|
+
and the slice is an upper bound on the profile. That inequality is the whole of
|
|
11
|
+
this module. The codebase already relies on it elsewhere -- ``_profile_task``
|
|
12
|
+
keeps a half-finished point because "a half-finished point is a real point that
|
|
13
|
+
happens to sit too high" -- and here it is used in the one direction that turns
|
|
14
|
+
a cheap evaluation into a proof:
|
|
15
|
+
|
|
16
|
+
**If the slice far from the fitted value sits below the 1.9207 threshold, the
|
|
17
|
+
profile there does too, so the data do not exclude that value and this side of
|
|
18
|
+
the confidence interval is open.** One evaluation per side settles it, with no
|
|
19
|
+
nuisance optimization at all.
|
|
20
|
+
|
|
21
|
+
How far is "far", and why not the declared bound
|
|
22
|
+
------------------------------------------------
|
|
23
|
+
An earlier version of this screen read the verdict at the parameter's declared
|
|
24
|
+
bound. That is the wrong ruler. On this spec every bound is ``x0/10`` to
|
|
25
|
+
``x0*10`` -- a uniform one-decade search box around the starting value, written
|
|
26
|
+
to keep the optimizer in a sensible region, carrying no claim about physics. A
|
|
27
|
+
verdict read there says "the slice did not cross inside the box someone drew",
|
|
28
|
+
which halts the analysis on a convention when the crossing may sit just outside
|
|
29
|
+
it.
|
|
30
|
+
|
|
31
|
+
So the screen walks to whichever is *further*, the declared bound or
|
|
32
|
+
``span_decades`` from the fitted value, and states its reach in decades. Going
|
|
33
|
+
past a bound is safe in both directions and is the point of doing it. A bound is
|
|
34
|
+
a prior, not data: a slice that stays flat beyond it makes the finding stronger,
|
|
35
|
+
not weaker, and a slice that crosses just beyond it prevents exactly the false
|
|
36
|
+
halt an arbitrary box would have caused. The verdict then rests on a scale-free
|
|
37
|
+
statement -- "a thousandfold change in this parameter costs nothing" -- rather
|
|
38
|
+
than on where the box was drawn.
|
|
39
|
+
|
|
40
|
+
Where the bound still matters it is reported rather than assumed: a side whose
|
|
41
|
+
slice does not cross inside the declared bound but does cross outside it means
|
|
42
|
+
the fit's own search box is narrower than the confidence interval, which is
|
|
43
|
+
worth knowing and is not a reason to stop.
|
|
44
|
+
|
|
45
|
+
The screen is one-directional and must be read that way. It can prove a
|
|
46
|
+
parameter unbounded; it can never prove one identifiable. A slice that crosses
|
|
47
|
+
the threshold says nothing about whether the profile ever will -- the profile
|
|
48
|
+
may flatten out beyond the slice crossing and never reach it. So a pass here is
|
|
49
|
+
permission to spend profile evaluations, not an answer.
|
|
50
|
+
|
|
51
|
+
Why it is worth the evaluations
|
|
52
|
+
-------------------------------
|
|
53
|
+
An unbounded parameter is the most expensive object in the profile run. Every
|
|
54
|
+
one of ``max_extend`` extension steps fires on it, each a full nuisance
|
|
55
|
+
minimization at increasingly extreme values -- where the integrator is slowest
|
|
56
|
+
and ``safe_simulate``'s retry ladder is most likely to fire -- and the answer at
|
|
57
|
+
the end is still "open interval". At the measured cost of this spec the screen
|
|
58
|
+
is about ten minutes across the pool and the run it can avoid is days.
|
|
59
|
+
|
|
60
|
+
Why it halts rather than skipping
|
|
61
|
+
---------------------------------
|
|
62
|
+
A parameter the data cannot bound has to be fixed to a defensible value, from
|
|
63
|
+
the literature or to zero, and removed from the fit. That is a modelling
|
|
64
|
+
decision resting on evidence outside this run, and it is not one an automated
|
|
65
|
+
pass may take on its own or defer. So the screen raises
|
|
66
|
+
:class:`UnidentifiableParameters` and the analysis stops before spending any
|
|
67
|
+
profile compute. There is deliberately no override flag: an override is exactly
|
|
68
|
+
the on-the-fly decision this exists to prevent.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
import json
|
|
72
|
+
import os
|
|
73
|
+
|
|
74
|
+
import numpy as np
|
|
75
|
+
|
|
76
|
+
from Engine.Evaluator import FAILURE_VALUE
|
|
77
|
+
|
|
78
|
+
# chi2(df=1, p=0.95) / 2 -- the same threshold the profile crosses, and it has
|
|
79
|
+
# to stay the same or the screen would rule out intervals the profile would
|
|
80
|
+
# have drawn.
|
|
81
|
+
THRESHOLD = 1.9207
|
|
82
|
+
|
|
83
|
+
# Written into the checkpoint directory, which is already keyed by model hash,
|
|
84
|
+
# spec hash and the optimum, so a file found there belongs to this run.
|
|
85
|
+
SCREEN_FILENAME = "slice_screen.json"
|
|
86
|
+
|
|
87
|
+
# How far from the fitted value the screen walks, when the declared bound does
|
|
88
|
+
# not already reach further. Three decades either way: if the data cannot tell a
|
|
89
|
+
# parameter from a thousandth or a thousand times itself, nothing a profile does
|
|
90
|
+
# will bound it. Wider costs nothing when the model survives out there and
|
|
91
|
+
# yields no verdict when it does not, which is the safe direction.
|
|
92
|
+
SPAN_DECADES = 3.0
|
|
93
|
+
|
|
94
|
+
# The reach a side must actually achieve before "the slice never crossed" is
|
|
95
|
+
# allowed to mean anything. A slice that stays flat over a factor of ten is a
|
|
96
|
+
# statement; one that stays flat over a factor of 1.1 is not, and would halt the
|
|
97
|
+
# run on a parameter nobody had looked at properly. This bites when the model
|
|
98
|
+
# stops evaluating close in and the walk cannot get out to the span.
|
|
99
|
+
MIN_REACH_DECADES = 1.0
|
|
100
|
+
|
|
101
|
+
# One side's verdict. Only "open" halts the run.
|
|
102
|
+
# crossed the slice rose above the threshold: profiling may proceed
|
|
103
|
+
# open the slice is still below the threshold decades out: PROVEN open
|
|
104
|
+
# blocked nothing on this side could be evaluated: no verdict
|
|
105
|
+
# short-reach the furthest evaluable point is too close in to conclude from
|
|
106
|
+
# empty there is no side to walk
|
|
107
|
+
_FAILING_STATES = ("open",)
|
|
108
|
+
_INCONCLUSIVE_STATES = ("blocked", "short-reach")
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class UnidentifiableParameters(RuntimeError):
|
|
112
|
+
"""The slice screen proved one or more sides cannot be bounded.
|
|
113
|
+
|
|
114
|
+
Carries the full report so the caller can print it rather than reconstruct
|
|
115
|
+
it, and so the message is self-contained wherever it surfaces -- on a
|
|
116
|
+
cluster this is read out of a Slurm log hours later, with nothing else to
|
|
117
|
+
hand.
|
|
118
|
+
"""
|
|
119
|
+
|
|
120
|
+
def __init__(self, failures, path=None, report=None):
|
|
121
|
+
self.failures = list(failures)
|
|
122
|
+
self.path = path
|
|
123
|
+
self.report = report
|
|
124
|
+
named = ", ".join(f"{f['name']} ({f['side']})" for f in self.failures[:6])
|
|
125
|
+
more = (f" and {len(self.failures) - 6} more"
|
|
126
|
+
if len(self.failures) > 6 else "")
|
|
127
|
+
where = f"; see {path}" if path else ""
|
|
128
|
+
super().__init__(
|
|
129
|
+
f"the slice screen proved {len(self.failures)} side(s) unbounded: "
|
|
130
|
+
f"{named}{more}. Fix each parameter to a literature-supported "
|
|
131
|
+
f"value, or to zero, and drop it from the fit before "
|
|
132
|
+
f"profiling{where}"
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _se_for(wald_se, param_idx):
|
|
137
|
+
"""The Wald SE for one parameter in optimizer space, or None.
|
|
138
|
+
|
|
139
|
+
None is a real answer, not a missing one: it means the Hessian was singular
|
|
140
|
+
in this direction. The grid below falls back to a multiplicative offset,
|
|
141
|
+
the same fallback ``_profile_grid_for`` uses.
|
|
142
|
+
"""
|
|
143
|
+
if wald_se is None:
|
|
144
|
+
return None
|
|
145
|
+
try:
|
|
146
|
+
cand = float(np.atleast_1d(wald_se)[param_idx])
|
|
147
|
+
except (IndexError, TypeError, ValueError):
|
|
148
|
+
return None
|
|
149
|
+
return cand if np.isfinite(cand) and cand > 0 else None
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def decades_from(p_opt, x, is_log):
|
|
153
|
+
"""How many decades *x* sits from the fitted value, or nan.
|
|
154
|
+
|
|
155
|
+
The screen's own ruler, and the one the report quotes. A log-scaled
|
|
156
|
+
parameter is stored as its own logarithm, so the distance is already in
|
|
157
|
+
decades; a positive linear one is taken into logarithms for the same
|
|
158
|
+
reason. A linear parameter at or below zero has no meaningful ratio and
|
|
159
|
+
returns nan, which the verdict falls back from rather than guesses at.
|
|
160
|
+
"""
|
|
161
|
+
if is_log:
|
|
162
|
+
return abs(float(x) - float(p_opt))
|
|
163
|
+
if p_opt > 0 and x > 0:
|
|
164
|
+
return abs(float(np.log10(x)) - float(np.log10(p_opt)))
|
|
165
|
+
return float("nan")
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _span_target(p_opt, sign, is_log, span_decades):
|
|
169
|
+
"""The value *span_decades* out from the optimum, or None if undefined."""
|
|
170
|
+
if is_log:
|
|
171
|
+
return float(p_opt) + sign * float(span_decades)
|
|
172
|
+
if p_opt > 0:
|
|
173
|
+
return float(p_opt) * float(10.0 ** (sign * span_decades))
|
|
174
|
+
return None
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def screen_target(p_opt, lb, ub, sign, is_log, span_decades=SPAN_DECADES):
|
|
178
|
+
"""How far this side walks: the declared bound or the span, whichever is further.
|
|
179
|
+
|
|
180
|
+
Deliberately past the bound when the span reaches further. The bounds on
|
|
181
|
+
this spec are a one-decade search box around the starting value, and a
|
|
182
|
+
verdict read at a box edge is a verdict about the box. Evaluating outside it
|
|
183
|
+
is safe -- there is no optimizer here, only an evaluation, and a value the
|
|
184
|
+
model cannot take comes back as the failure sentinel and yields no verdict
|
|
185
|
+
-- and it is what keeps an arbitrary bound from either manufacturing a halt
|
|
186
|
+
or hiding one.
|
|
187
|
+
"""
|
|
188
|
+
bound = lb if sign < 0 else ub
|
|
189
|
+
span = _span_target(p_opt, sign, is_log, span_decades)
|
|
190
|
+
if bound is None or not np.isfinite(bound):
|
|
191
|
+
return span
|
|
192
|
+
if span is None:
|
|
193
|
+
return float(bound)
|
|
194
|
+
# Whichever lies further out in the direction of travel.
|
|
195
|
+
return float(max(bound, span) if sign > 0 else min(bound, span))
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def screen_values(p_opt, lb, ub, se, sign, is_log, max_points=6, growth=2.0,
|
|
199
|
+
range_factor=2.0, span_decades=SPAN_DECADES):
|
|
200
|
+
"""Values to evaluate on one side, innermost first, ending at the target.
|
|
201
|
+
|
|
202
|
+
Two jobs, and the second is what makes this a screen rather than a scan.
|
|
203
|
+
|
|
204
|
+
Resolution near the crossing: the ladder starts at half a Wald SE and
|
|
205
|
+
doubles its distance from the optimum, so it straddles the 1.96 SE where a
|
|
206
|
+
crossing is expected. The slice crossing sits *inside* the profile crossing
|
|
207
|
+
-- the slice is the steeper curve -- so a ladder placed for the profile
|
|
208
|
+
brackets the slice comfortably.
|
|
209
|
+
|
|
210
|
+
Half an SE rather than a whole one, because the innermost point is the one
|
|
211
|
+
that certifies the inner bracket. A ladder starting at 1 SE puts its first
|
|
212
|
+
point outside the slice crossing of any parameter tighter than the Hessian
|
|
213
|
+
predicted, and then every evaluated point is above the threshold and
|
|
214
|
+
nothing is certified. Starting inside costs one evaluation per side and is
|
|
215
|
+
what makes the bracket usable.
|
|
216
|
+
|
|
217
|
+
Reach: the last value is always :func:`screen_target` -- the declared bound
|
|
218
|
+
or ``span_decades`` out, whichever is further -- even when the ladder would
|
|
219
|
+
have stopped short of it. Without that point the screen concludes nothing. A
|
|
220
|
+
slice that fails to cross within four standard errors is not evidence of
|
|
221
|
+
anything; only one that fails to cross across decades is.
|
|
222
|
+
|
|
223
|
+
The declared bound is included as a point of its own when the walk goes past
|
|
224
|
+
it, because "does this cross inside the search box" is a separate and useful
|
|
225
|
+
question from "does this cross at all" -- it is how a fit running against
|
|
226
|
+
its own walls is detected.
|
|
227
|
+
|
|
228
|
+
Distances are multiplied rather than added, via the same
|
|
229
|
+
``_next_extension_value`` the profile's own extension pass uses, so a
|
|
230
|
+
log-scaled parameter walks in decades and a positive linear one walks in
|
|
231
|
+
ratios. Adding a fixed offset instead collapses the interesting range on
|
|
232
|
+
exactly the parameters that span decades.
|
|
233
|
+
"""
|
|
234
|
+
from Engine.Optimize import _at_bound, _next_extension_value
|
|
235
|
+
|
|
236
|
+
bound = lb if sign < 0 else ub
|
|
237
|
+
has_bound = bound is not None and np.isfinite(bound)
|
|
238
|
+
target = screen_target(p_opt, lb, ub, sign, is_log, span_decades)
|
|
239
|
+
if target is None or not np.isfinite(target):
|
|
240
|
+
return []
|
|
241
|
+
|
|
242
|
+
# Nowhere to walk: the optimum already sits at the furthest point this side
|
|
243
|
+
# would reach. Reporting "unbounded" from here would be nonsense.
|
|
244
|
+
if _at_bound(p_opt, target, sign):
|
|
245
|
+
return []
|
|
246
|
+
|
|
247
|
+
# The walk is clipped to the target rather than to the bound, so a span
|
|
248
|
+
# reaching past the bound is followed rather than truncated.
|
|
249
|
+
walk_lb = target if sign < 0 else -np.inf
|
|
250
|
+
walk_ub = target if sign > 0 else np.inf
|
|
251
|
+
|
|
252
|
+
if se is not None:
|
|
253
|
+
first = p_opt + sign * 0.5 * float(se)
|
|
254
|
+
elif is_log:
|
|
255
|
+
first = p_opt + sign * float(np.log10(range_factor))
|
|
256
|
+
elif p_opt > 0:
|
|
257
|
+
first = p_opt * range_factor if sign > 0 else p_opt / range_factor
|
|
258
|
+
else:
|
|
259
|
+
# A linear parameter at or below zero has no meaningful ratio, so the
|
|
260
|
+
# step is absolute and scaled to the parameter's own size.
|
|
261
|
+
first = p_opt + sign * max(abs(p_opt), 1.0)
|
|
262
|
+
|
|
263
|
+
first = max(first, target) if sign < 0 else min(first, target)
|
|
264
|
+
if not np.isfinite(first) or first == p_opt:
|
|
265
|
+
return [float(target)]
|
|
266
|
+
|
|
267
|
+
# Two slots are reserved: the target, and the declared bound when the walk
|
|
268
|
+
# passes it. Both are appended below if the ladder has not landed on them.
|
|
269
|
+
# ``_at_bound`` asks "has this reached or passed the bound", which every
|
|
270
|
+
# point beyond it also satisfies, so the test here is a strict comparison:
|
|
271
|
+
# what matters is whether the bound lies *inside* the walk.
|
|
272
|
+
tol = 1e-12 * max(abs(bound), 1.0) if has_bound else 0.0
|
|
273
|
+
beyond = has_bound and (target < bound - tol if sign < 0
|
|
274
|
+
else target > bound + tol)
|
|
275
|
+
n_ladder = max(1, int(max_points) - (2 if beyond else 1))
|
|
276
|
+
vals = [float(first)]
|
|
277
|
+
while len(vals) < n_ladder:
|
|
278
|
+
nxt = _next_extension_value(p_opt, vals[-1], walk_lb, walk_ub, sign,
|
|
279
|
+
is_log, growth)
|
|
280
|
+
if nxt is None:
|
|
281
|
+
break
|
|
282
|
+
vals.append(float(nxt))
|
|
283
|
+
|
|
284
|
+
if beyond and not any(abs(v - bound) <= tol for v in vals):
|
|
285
|
+
vals.append(float(bound))
|
|
286
|
+
if not _at_bound(vals[-1], target, sign):
|
|
287
|
+
vals.append(float(target))
|
|
288
|
+
|
|
289
|
+
vals = sorted(set(vals), reverse=(sign < 0))
|
|
290
|
+
return [v for v in vals if (v < p_opt if sign < 0 else v > p_opt)]
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def _verdict(points, bound, threshold, p_opt, sign, is_log,
|
|
294
|
+
min_reach_decades=MIN_REACH_DECADES):
|
|
295
|
+
"""One side's state, its reach, and the certified inner bracket.
|
|
296
|
+
|
|
297
|
+
The verdict is read at the furthest point the model could actually be
|
|
298
|
+
evaluated at, and it counts only if that point is at least
|
|
299
|
+
``min_reach_decades`` from the fitted value. Two things follow, and both are
|
|
300
|
+
the point of reading it this way rather than at the declared bound.
|
|
301
|
+
|
|
302
|
+
A bound narrower than the span no longer decides anything: the walk goes
|
|
303
|
+
past it, so a slice that crosses just outside an arbitrary box is seen to
|
|
304
|
+
cross and the side is cleared. And a model that stops evaluating part way
|
|
305
|
+
out no longer yields a verdict by default: the reach shrinks to wherever the
|
|
306
|
+
last finite value was, and if that is too close in the side is reported as
|
|
307
|
+
unscreened rather than declared open.
|
|
308
|
+
|
|
309
|
+
``inner_bracket`` is the outermost value whose slice sits at or below the
|
|
310
|
+
threshold with nothing above it in between -- so every point from the
|
|
311
|
+
optimum out to it is certified to lie *inside* the confidence interval, by
|
|
312
|
+
the same inequality the module relies on. No profile evaluation ever needs
|
|
313
|
+
to be spent there. It is recorded rather than consumed: wiring it into the
|
|
314
|
+
profile's opening grid is a separate change.
|
|
315
|
+
"""
|
|
316
|
+
def _usable(p):
|
|
317
|
+
return np.isfinite(p["dnll"]) and p["nll"] < FAILURE_VALUE
|
|
318
|
+
|
|
319
|
+
finite = [p for p in points if _usable(p)]
|
|
320
|
+
|
|
321
|
+
inner = None
|
|
322
|
+
for p in points: # points are ordered outward
|
|
323
|
+
if not _usable(p) or p["dnll"] > threshold:
|
|
324
|
+
break
|
|
325
|
+
inner = p["x_linear"]
|
|
326
|
+
|
|
327
|
+
crossed_any = any(p["dnll"] > threshold for p in finite)
|
|
328
|
+
has_bound = bound is not None and np.isfinite(bound)
|
|
329
|
+
|
|
330
|
+
# Inside the declared search box, treated separately from the verdict: a
|
|
331
|
+
# side that does not cross in the box but does cross outside it says the
|
|
332
|
+
# fit's own walls are narrower than the interval.
|
|
333
|
+
in_box = [p for p in finite
|
|
334
|
+
if not has_bound
|
|
335
|
+
or (p["x"] >= bound - 1e-12 if sign < 0
|
|
336
|
+
else p["x"] <= bound + 1e-12)]
|
|
337
|
+
crossed_in_box = any(p["dnll"] > threshold for p in in_box)
|
|
338
|
+
|
|
339
|
+
outer = finite[-1] if finite else None
|
|
340
|
+
reach = (decades_from(p_opt, outer["x"], is_log)
|
|
341
|
+
if outer is not None else float("nan"))
|
|
342
|
+
# A linear parameter straddling zero has no decades. Fall back to the older
|
|
343
|
+
# question -- did the walk get all the way out -- rather than guessing.
|
|
344
|
+
reached_far = (reach >= min_reach_decades if np.isfinite(reach)
|
|
345
|
+
else (outer is not None and points
|
|
346
|
+
and outer["x"] == points[-1]["x"]))
|
|
347
|
+
|
|
348
|
+
if not points:
|
|
349
|
+
state = "empty"
|
|
350
|
+
elif outer is None:
|
|
351
|
+
# Nothing on this side could be evaluated at all. Not proof of
|
|
352
|
+
# anything: a model may legitimately break away from its fitted region.
|
|
353
|
+
state = "blocked"
|
|
354
|
+
elif outer["dnll"] <= threshold:
|
|
355
|
+
state = "open" if reached_far else "short-reach"
|
|
356
|
+
elif crossed_any:
|
|
357
|
+
state = "crossed"
|
|
358
|
+
else:
|
|
359
|
+
state = "short-reach"
|
|
360
|
+
|
|
361
|
+
return {
|
|
362
|
+
"state": state,
|
|
363
|
+
"inner_bracket": inner,
|
|
364
|
+
"max_dnll": max((p["dnll"] for p in finite), default=None),
|
|
365
|
+
"reach": (float(outer["x_linear"]) if outer is not None else None),
|
|
366
|
+
"reach_decades": (float(reach) if np.isfinite(reach) else None),
|
|
367
|
+
"dnll_at_reach": (float(outer["dnll"]) if outer is not None else None),
|
|
368
|
+
"bound": float(bound) if has_bound else None,
|
|
369
|
+
"walked_past_bound": bool(
|
|
370
|
+
has_bound and outer is not None
|
|
371
|
+
and (outer["x"] < bound - 1e-12 if sign < 0
|
|
372
|
+
else outer["x"] > bound + 1e-12)),
|
|
373
|
+
# The search box is narrower than the interval: nothing inside the
|
|
374
|
+
# declared bound crossed, but something outside it did. Not a reason to
|
|
375
|
+
# stop, and worth saying -- it means the fit was working against its
|
|
376
|
+
# own walls.
|
|
377
|
+
"box_too_narrow": bool(crossed_any and not crossed_in_box and has_bound),
|
|
378
|
+
# The slice crossed on the way out and fell back below the threshold
|
|
379
|
+
# further on. The proof stands -- the data do not exclude the far value
|
|
380
|
+
# -- but the curve is not monotone and the report has to say so,
|
|
381
|
+
# because "open" alone would misdescribe it.
|
|
382
|
+
"non_monotone": bool(crossed_any and state == "open"),
|
|
383
|
+
"points": points,
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def run_slice_screen(nll_batch, res_x, nll_at_optimum, param_names, bounds,
|
|
388
|
+
scales=None, wald_se=None, threshold=THRESHOLD,
|
|
389
|
+
max_points=6, growth=2.0, range_factor=2.0,
|
|
390
|
+
span_decades=SPAN_DECADES,
|
|
391
|
+
min_reach_decades=MIN_REACH_DECADES, verbose=True):
|
|
392
|
+
"""Evaluate every parameter's slice out across decades, as one batch.
|
|
393
|
+
|
|
394
|
+
A slice point has no dependency on any other, so all of them go out
|
|
395
|
+
together: this is the cheapest possible use of the pool, one evaluation per
|
|
396
|
+
point with no optimizer wrapped around it.
|
|
397
|
+
|
|
398
|
+
That shape has a second use. If evaluations far from the fitted values are
|
|
399
|
+
pathologically slow -- the region where the integrator struggles and
|
|
400
|
+
``safe_simulate`` enters its retry ladder -- this finds out in minutes, and
|
|
401
|
+
says so plainly, instead of the run discovering it hours into a profile
|
|
402
|
+
batch where every stuck evaluation is buried inside a nuisance
|
|
403
|
+
minimization.
|
|
404
|
+
"""
|
|
405
|
+
from Engine.Optimize import _param_bounds
|
|
406
|
+
|
|
407
|
+
res_x = np.asarray(res_x, dtype=float)
|
|
408
|
+
scales = list(scales) if scales is not None else ["lin"] * len(param_names)
|
|
409
|
+
|
|
410
|
+
plan, xs = [], []
|
|
411
|
+
for i, name in enumerate(param_names):
|
|
412
|
+
is_log = scales[i] == "log10"
|
|
413
|
+
lb, ub = _param_bounds(bounds, i)
|
|
414
|
+
se = _se_for(wald_se, i)
|
|
415
|
+
for sign, side in ((-1, "lower"), (1, "upper")):
|
|
416
|
+
vals = screen_values(res_x[i], lb, ub, se, sign, is_log,
|
|
417
|
+
max_points=max_points, growth=growth,
|
|
418
|
+
range_factor=range_factor,
|
|
419
|
+
span_decades=span_decades)
|
|
420
|
+
plan.append({"index": i, "name": name, "side": side, "sign": sign,
|
|
421
|
+
"is_log": is_log, "values": vals,
|
|
422
|
+
"p_opt": float(res_x[i]),
|
|
423
|
+
"bound": (lb if sign < 0 else ub)})
|
|
424
|
+
for v in vals:
|
|
425
|
+
x = res_x.copy()
|
|
426
|
+
x[i] = v
|
|
427
|
+
xs.append(x)
|
|
428
|
+
|
|
429
|
+
if verbose:
|
|
430
|
+
print(f"\n[screen] slice screen: {len(param_names)} parameter(s) x 2 "
|
|
431
|
+
f"side(s) = {len(xs)} evaluation(s), submitted as one batch. "
|
|
432
|
+
f"No nuisance optimization: each point is an upper bound on the "
|
|
433
|
+
f"profile, which is all the screen needs.", flush=True)
|
|
434
|
+
|
|
435
|
+
values = list(nll_batch(xs, label="slice-screen")) if xs else []
|
|
436
|
+
|
|
437
|
+
report = {"threshold": float(threshold),
|
|
438
|
+
"anchor": float(nll_at_optimum),
|
|
439
|
+
"res_x": [float(v) for v in res_x],
|
|
440
|
+
"param_names": list(param_names),
|
|
441
|
+
"span_decades": float(span_decades),
|
|
442
|
+
"min_reach_decades": float(min_reach_decades),
|
|
443
|
+
"n_evaluations": len(xs),
|
|
444
|
+
"parameters": {}}
|
|
445
|
+
|
|
446
|
+
pos = 0
|
|
447
|
+
for entry in plan:
|
|
448
|
+
pts = []
|
|
449
|
+
for v in entry["values"]:
|
|
450
|
+
nll = float(values[pos])
|
|
451
|
+
pos += 1
|
|
452
|
+
pts.append({
|
|
453
|
+
"x": float(v),
|
|
454
|
+
"x_linear": float(10.0 ** v if entry["is_log"] else v),
|
|
455
|
+
"nll": nll,
|
|
456
|
+
"dnll": float(nll - nll_at_optimum),
|
|
457
|
+
})
|
|
458
|
+
side = _verdict(pts, entry["bound"], threshold, entry["p_opt"],
|
|
459
|
+
entry["sign"], entry["is_log"],
|
|
460
|
+
min_reach_decades=min_reach_decades)
|
|
461
|
+
side["is_log"] = entry["is_log"]
|
|
462
|
+
report["parameters"].setdefault(entry["name"], {})[entry["side"]] = side
|
|
463
|
+
|
|
464
|
+
states = [s["state"] for sides in report["parameters"].values()
|
|
465
|
+
for s in sides.values()]
|
|
466
|
+
report["n_open"] = states.count("open")
|
|
467
|
+
report["n_inconclusive"] = sum(states.count(s) for s in _INCONCLUSIVE_STATES)
|
|
468
|
+
report["n_crossed"] = states.count("crossed")
|
|
469
|
+
report["n_box_too_narrow"] = sum(
|
|
470
|
+
1 for sides in report["parameters"].values()
|
|
471
|
+
for s in sides.values() if s["box_too_narrow"])
|
|
472
|
+
return report
|
|
473
|
+
|
|
474
|
+
|
|
475
|
+
def failing_sides(report):
|
|
476
|
+
"""Every side the screen proved unbounded, flattened for reporting."""
|
|
477
|
+
out = []
|
|
478
|
+
for name, sides in report.get("parameters", {}).items():
|
|
479
|
+
for side, rec in sides.items():
|
|
480
|
+
if rec["state"] in _FAILING_STATES:
|
|
481
|
+
out.append(dict(rec, name=name, side=side))
|
|
482
|
+
return out
|
|
483
|
+
|
|
484
|
+
|
|
485
|
+
def screen_summary(report):
|
|
486
|
+
"""The screen without its raw points, small enough to ride in the results.
|
|
487
|
+
|
|
488
|
+
The evaluated points stay in ``slice_screen.json``; what a consumer of the
|
|
489
|
+
results snapshot needs is the verdict per side and the certified inner
|
|
490
|
+
bracket, so that a reader months later can tell a profile that was allowed
|
|
491
|
+
to run from one that was never screened.
|
|
492
|
+
"""
|
|
493
|
+
if not report:
|
|
494
|
+
return None
|
|
495
|
+
return {
|
|
496
|
+
"threshold": report.get("threshold"),
|
|
497
|
+
"n_evaluations": report.get("n_evaluations"),
|
|
498
|
+
"n_open": report.get("n_open"),
|
|
499
|
+
"n_inconclusive": report.get("n_inconclusive"),
|
|
500
|
+
"n_crossed": report.get("n_crossed"),
|
|
501
|
+
"n_box_too_narrow": report.get("n_box_too_narrow"),
|
|
502
|
+
"parameters": {
|
|
503
|
+
name: {side: {"state": rec["state"],
|
|
504
|
+
"inner_bracket": rec["inner_bracket"],
|
|
505
|
+
"reach": rec["reach"],
|
|
506
|
+
"reach_decades": rec["reach_decades"],
|
|
507
|
+
"dnll_at_reach": rec["dnll_at_reach"],
|
|
508
|
+
"walked_past_bound": rec["walked_past_bound"],
|
|
509
|
+
"box_too_narrow": rec["box_too_narrow"],
|
|
510
|
+
"bound": rec["bound"]}
|
|
511
|
+
for side, rec in sides.items()}
|
|
512
|
+
for name, sides in report.get("parameters", {}).items()
|
|
513
|
+
},
|
|
514
|
+
}
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
def _decades_text(rec):
|
|
518
|
+
d = rec.get("reach_decades")
|
|
519
|
+
return f"{d:.2g} decade(s) out" if d is not None else "at its bound"
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
def print_screen_report(report):
|
|
523
|
+
"""The screen's findings, in the order a reader needs them."""
|
|
524
|
+
failures = failing_sides(report)
|
|
525
|
+
thr = report["threshold"]
|
|
526
|
+
|
|
527
|
+
if failures:
|
|
528
|
+
print(f"\n[screen] {len(failures)} side(s) are PROVEN unbounded:",
|
|
529
|
+
flush=True)
|
|
530
|
+
for f in failures:
|
|
531
|
+
notes = []
|
|
532
|
+
if f["walked_past_bound"]:
|
|
533
|
+
notes.append("the walk went past the declared bound of "
|
|
534
|
+
f"{f['bound']:.6g}, so this verdict does not rest "
|
|
535
|
+
f"on it")
|
|
536
|
+
if f["non_monotone"]:
|
|
537
|
+
notes.append("the slice rose above the threshold further in "
|
|
538
|
+
"and came back down, so this curve is not "
|
|
539
|
+
"monotone and is worth looking at directly")
|
|
540
|
+
tail = ("; " + "; ".join(notes)) if notes else ""
|
|
541
|
+
print(f" {f['name']} ({f['side']}): slice dNLL "
|
|
542
|
+
f"{f['dnll_at_reach']:.4g} at {f['reach']:.6g}, "
|
|
543
|
+
f"{_decades_text(f)}, still below {thr}. The profile there "
|
|
544
|
+
f"is no higher, so the data do not exclude it{tail}.")
|
|
545
|
+
|
|
546
|
+
narrow = [(name, side, rec)
|
|
547
|
+
for name, sides in report["parameters"].items()
|
|
548
|
+
for side, rec in sides.items() if rec["box_too_narrow"]]
|
|
549
|
+
if narrow:
|
|
550
|
+
print(f"\n[screen] {len(narrow)} side(s) cross the threshold only "
|
|
551
|
+
f"*outside* the declared bound, so the fit's own search box is "
|
|
552
|
+
f"narrower than the interval (this does not halt the run):",
|
|
553
|
+
flush=True)
|
|
554
|
+
for name, side, rec in narrow[:20]:
|
|
555
|
+
print(f" {name} ({side}): nothing inside the bound "
|
|
556
|
+
f"{rec['bound']:.6g} reached {thr}; widen it if wider values "
|
|
557
|
+
f"are physical, or the fit is working against its own walls")
|
|
558
|
+
if len(narrow) > 20:
|
|
559
|
+
print(f" ... and {len(narrow) - 20} more")
|
|
560
|
+
|
|
561
|
+
inconclusive = [(name, side, rec)
|
|
562
|
+
for name, sides in report["parameters"].items()
|
|
563
|
+
for side, rec in sides.items()
|
|
564
|
+
if rec["state"] in _INCONCLUSIVE_STATES]
|
|
565
|
+
if inconclusive:
|
|
566
|
+
print(f"\n[screen] {len(inconclusive)} side(s) could not be screened "
|
|
567
|
+
f"(this does not halt the run, and does not clear them either):",
|
|
568
|
+
flush=True)
|
|
569
|
+
for name, side, rec in inconclusive[:20]:
|
|
570
|
+
if rec["state"] == "blocked":
|
|
571
|
+
why = "nothing on this side could be evaluated at all"
|
|
572
|
+
else:
|
|
573
|
+
why = (f"the furthest evaluable point is only "
|
|
574
|
+
f"{_decades_text(rec)}, too close in to conclude from")
|
|
575
|
+
print(f" {name} ({side}): {why}")
|
|
576
|
+
if len(inconclusive) > 20:
|
|
577
|
+
print(f" ... and {len(inconclusive) - 20} more")
|
|
578
|
+
|
|
579
|
+
if not failures:
|
|
580
|
+
print(f"\n[screen] every screened side rose above {thr}. That is "
|
|
581
|
+
f"permission to profile, not a finding: the slice is an upper "
|
|
582
|
+
f"bound, so it can prove a parameter unbounded but never prove "
|
|
583
|
+
f"one identifiable.", flush=True)
|
|
584
|
+
|
|
585
|
+
|
|
586
|
+
def save_screen(report, ckpt_dir):
|
|
587
|
+
"""Write the screen beside the points it will govern. Returns the path.
|
|
588
|
+
|
|
589
|
+
Atomically, and never fatally: several links may share the directory, and a
|
|
590
|
+
screen that cannot be cached is a re-run of ten minutes, not a reason to
|
|
591
|
+
lose the run.
|
|
592
|
+
"""
|
|
593
|
+
if not ckpt_dir:
|
|
594
|
+
return None
|
|
595
|
+
path = os.path.join(ckpt_dir, SCREEN_FILENAME)
|
|
596
|
+
tmp = f"{path}.{os.getpid()}.tmp"
|
|
597
|
+
try:
|
|
598
|
+
os.makedirs(ckpt_dir, exist_ok=True)
|
|
599
|
+
with open(tmp, "w", encoding="utf-8") as fh:
|
|
600
|
+
json.dump(report, fh, indent=1)
|
|
601
|
+
os.replace(tmp, path)
|
|
602
|
+
except OSError:
|
|
603
|
+
try:
|
|
604
|
+
os.unlink(tmp)
|
|
605
|
+
except OSError:
|
|
606
|
+
pass
|
|
607
|
+
return None
|
|
608
|
+
return path
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
def load_screen(ckpt_dir, param_names, res_x, threshold=THRESHOLD,
|
|
612
|
+
span_decades=SPAN_DECADES,
|
|
613
|
+
min_reach_decades=MIN_REACH_DECADES):
|
|
614
|
+
"""A screen already run for this exact fit, or None.
|
|
615
|
+
|
|
616
|
+
The checkpoint directory is keyed by model hash, spec hash and the optimum,
|
|
617
|
+
so a file found in it already belongs to this run; the fields are checked
|
|
618
|
+
anyway because the cost of re-running the screen is ten minutes and the
|
|
619
|
+
cost of honouring someone else's is a wrong verdict on identifiability.
|
|
620
|
+
"""
|
|
621
|
+
if not ckpt_dir:
|
|
622
|
+
return None
|
|
623
|
+
path = os.path.join(ckpt_dir, SCREEN_FILENAME)
|
|
624
|
+
if not os.path.exists(path):
|
|
625
|
+
return None
|
|
626
|
+
try:
|
|
627
|
+
with open(path, "r", encoding="utf-8") as fh:
|
|
628
|
+
report = json.load(fh)
|
|
629
|
+
except (OSError, ValueError):
|
|
630
|
+
return None
|
|
631
|
+
|
|
632
|
+
if list(report.get("param_names") or []) != list(param_names):
|
|
633
|
+
return None
|
|
634
|
+
# Every number that can change a verdict is part of the key. A screen run
|
|
635
|
+
# over one decade must not answer for a run asked to look over three.
|
|
636
|
+
for key, want in (("threshold", threshold),
|
|
637
|
+
("span_decades", span_decades),
|
|
638
|
+
("min_reach_decades", min_reach_decades)):
|
|
639
|
+
try:
|
|
640
|
+
if abs(float(report.get(key)) - float(want)) > 1e-12:
|
|
641
|
+
return None
|
|
642
|
+
except (TypeError, ValueError):
|
|
643
|
+
return None
|
|
644
|
+
stored = np.asarray(report.get("res_x") or [], dtype=float)
|
|
645
|
+
current = np.asarray(res_x, dtype=float)
|
|
646
|
+
if stored.shape != current.shape or not np.allclose(stored, current,
|
|
647
|
+
rtol=0, atol=1e-12):
|
|
648
|
+
return None
|
|
649
|
+
return report
|
|
650
|
+
|
|
651
|
+
|
|
652
|
+
def screen_or_raise(nll_batch, res_x, nll_at_optimum, param_names, bounds,
|
|
653
|
+
scales=None, wald_se=None, ckpt_dir=None,
|
|
654
|
+
threshold=THRESHOLD, max_points=6, growth=2.0,
|
|
655
|
+
range_factor=2.0, span_decades=SPAN_DECADES,
|
|
656
|
+
min_reach_decades=MIN_REACH_DECADES, verbose=True):
|
|
657
|
+
"""Run the screen (or reuse one), report it, and stop the run if it failed.
|
|
658
|
+
|
|
659
|
+
Raises :class:`UnidentifiableParameters` when any side is proven unbounded.
|
|
660
|
+
Returns the report otherwise, whose ``inner_bracket`` values are certified
|
|
661
|
+
to lie inside the confidence interval.
|
|
662
|
+
"""
|
|
663
|
+
report = load_screen(ckpt_dir, param_names, res_x, threshold,
|
|
664
|
+
span_decades, min_reach_decades)
|
|
665
|
+
if report is not None:
|
|
666
|
+
if verbose:
|
|
667
|
+
print(f"\n[screen] reusing the slice screen already run for this "
|
|
668
|
+
f"fit ({report.get('n_evaluations', 0)} evaluation(s)); "
|
|
669
|
+
f"nothing is recomputed.", flush=True)
|
|
670
|
+
path = os.path.join(ckpt_dir, SCREEN_FILENAME)
|
|
671
|
+
else:
|
|
672
|
+
report = run_slice_screen(
|
|
673
|
+
nll_batch, res_x, nll_at_optimum, param_names, bounds,
|
|
674
|
+
scales=scales, wald_se=wald_se, threshold=threshold,
|
|
675
|
+
max_points=max_points, growth=growth, range_factor=range_factor,
|
|
676
|
+
span_decades=span_decades, min_reach_decades=min_reach_decades,
|
|
677
|
+
verbose=verbose,
|
|
678
|
+
)
|
|
679
|
+
path = save_screen(report, ckpt_dir)
|
|
680
|
+
|
|
681
|
+
if verbose:
|
|
682
|
+
print_screen_report(report)
|
|
683
|
+
|
|
684
|
+
failures = failing_sides(report)
|
|
685
|
+
if failures:
|
|
686
|
+
if verbose:
|
|
687
|
+
print(f"\n[screen] HALTED before any profile point was started. "
|
|
688
|
+
f"A parameter the data cannot bound has to be fixed to a "
|
|
689
|
+
f"defensible value -- from the literature, or to zero -- and "
|
|
690
|
+
f"dropped from the fit. That rests on evidence outside this "
|
|
691
|
+
f"run, so it is not a decision to take here or to defer, and "
|
|
692
|
+
f"there is deliberately no flag to skip this check. Profiling "
|
|
693
|
+
f"these parameters would spend days to report the open "
|
|
694
|
+
f"intervals the screen has just proved in minutes.",
|
|
695
|
+
flush=True)
|
|
696
|
+
raise UnidentifiableParameters(failures, path=path, report=report)
|
|
697
|
+
|
|
698
|
+
return report
|