PyAntiGen 1.0.9__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- framework/AntimonyGen.py +48 -0
- framework/RxnDict_to_antimony.py +594 -0
- framework/TelluriumGen.py +16 -0
- framework/__init__.py +0 -0
- framework/antimony_utils.py +294 -0
- framework/cli.py +229 -0
- framework/data_interpolation.py +340 -0
- framework/isotopomer_tools.py +41 -0
- framework/model_generation.py +46 -0
- framework/models.py +189 -0
- framework/module_base.py +42 -0
- framework/pyantigen.py +51 -0
- framework/rate_laws.py +101 -0
- framework/reaction_creation.py +43 -0
- framework/template/Example/AntiGen_paths.py +23 -0
- framework/template/Example/Engine/Anchor_cache.py +193 -0
- framework/template/Example/Engine/Deadline.py +535 -0
- framework/template/Example/Engine/Evaluator.py +1176 -0
- framework/template/Example/Engine/Event_times.py +491 -0
- framework/template/Example/Engine/Fast_profile.py +701 -0
- framework/template/Example/Engine/Fit_cache.py +329 -0
- framework/template/Example/Engine/Identifiability.py +698 -0
- framework/template/Example/Engine/Model_optimize.py +1483 -0
- framework/template/Example/Engine/Model_simulate.py +124 -0
- framework/template/Example/Engine/Nuisance_sensitivity.py +298 -0
- framework/template/Example/Engine/Optimize.py +6862 -0
- framework/template/Example/Engine/Petab_export.py +398 -0
- framework/template/Example/Engine/Preequil_cache.py +361 -0
- framework/template/Example/Engine/Profile_checkpoint.py +399 -0
- framework/template/Example/Engine/Results.py +395 -0
- framework/template/Example/Engine/Sensitivity_analysis.py +320 -0
- framework/template/Example/Engine/Simulate.py +617 -0
- framework/template/Example/Flipflop_reference.py +401 -0
- framework/template/Example/Model_generate.py +37 -0
- framework/template/Example/Model_run.py +261 -0
- framework/template/Example/Modules/Data.py +63 -0
- framework/template/Example/Modules/Events.py +14 -0
- framework/template/Example/Modules/Experiment.py +194 -0
- framework/template/Example/Modules/Loss_config.py +61 -0
- framework/template/Example/Modules/Observed_species.py +3 -0
- framework/template/Example/Modules/Optimizer_settings.py +258 -0
- framework/template/Example/Modules/Plots.py +89 -0
- framework/template/Example/Modules/Solver_settings.py +16 -0
- framework/template/Example/Modules/Update_opt_parameters.py +24 -0
- framework/template/Example/Modules/Update_parameters.py +49 -0
- framework/template/data/ADneg.csv +27 -0
- framework/template/data/ADpos.csv +27 -0
- framework/template/data/Flipflop.csv +29 -0
- framework/template/data/make_flipflop_data.py +174 -0
- pyantigen-1.0.9.dist-info/METADATA +129 -0
- pyantigen-1.0.9.dist-info/RECORD +55 -0
- pyantigen-1.0.9.dist-info/WHEEL +5 -0
- pyantigen-1.0.9.dist-info/entry_points.txt +2 -0
- pyantigen-1.0.9.dist-info/licenses/LICENSE +21 -0
- pyantigen-1.0.9.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,535 @@
|
|
|
1
|
+
"""Wall-clock budget for a run that expects to be killed.
|
|
2
|
+
|
|
3
|
+
On a preemptible partition the job does not end when the work ends; it ends when
|
|
4
|
+
the scheduler says so. The four-hour cap on Hyak's ``ckpt`` partition is not by
|
|
5
|
+
itself the problem -- the profile store is per-point and a relaunch resumes --
|
|
6
|
+
but a run that does not know its own deadline keeps handing out new profile
|
|
7
|
+
points right up to the moment it is shot, and every point in flight at that
|
|
8
|
+
moment is lost. With forty workers each holding a point that has been running
|
|
9
|
+
for hours, an eviction at 3h55m throws away most of a link's compute. That, and
|
|
10
|
+
not the length of the work, is why four hours was not enough.
|
|
11
|
+
|
|
12
|
+
The fix is admission control: before starting another point, ask whether it can
|
|
13
|
+
plausibly finish. A point that cannot is simply not started, the link ends early
|
|
14
|
+
and cleanly, and everything already computed is on disk. What made this
|
|
15
|
+
affordable is that the profile store is append-only and a resume is a no-op, so
|
|
16
|
+
"stop early" costs nothing but the points not yet begun.
|
|
17
|
+
|
|
18
|
+
How long a point takes is measured rather than guessed. The first link of a
|
|
19
|
+
chain has nothing to go on and admits work until its own first result lands;
|
|
20
|
+
every link after it starts from the durations the earlier ones wrote down.
|
|
21
|
+
|
|
22
|
+
Nothing here is Slurm-specific at the call site. Off-cluster
|
|
23
|
+
:func:`resolve_deadline` returns ``None``, every check passes, and the run
|
|
24
|
+
behaves exactly as it did before this module existed.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
import json
|
|
28
|
+
import math
|
|
29
|
+
import os
|
|
30
|
+
import subprocess
|
|
31
|
+
import time
|
|
32
|
+
|
|
33
|
+
# How close to the deadline a link is willing to start nothing new. It has to
|
|
34
|
+
# cover what happens *after* the last point lands: the pool shutdown, trace
|
|
35
|
+
# assembly, plotting and the results write, none of which are instant on a
|
|
36
|
+
# large spec.
|
|
37
|
+
DEFAULT_MARGIN_S = 600.0
|
|
38
|
+
|
|
39
|
+
# Set by Model_run.py from --wall-time. An environment variable rather than a
|
|
40
|
+
# threaded argument because the only consumer sits at the bottom of a call chain
|
|
41
|
+
# that would otherwise need the flag added to a dozen signatures for it, and
|
|
42
|
+
# because on a cluster the deadline genuinely does arrive through the
|
|
43
|
+
# environment.
|
|
44
|
+
WALL_TIME_ENV = "PROFILE_WALL_TIME"
|
|
45
|
+
MARGIN_ENV = "PROFILE_DEADLINE_MARGIN"
|
|
46
|
+
MIN_SLICE_ENV = "PROFILE_MIN_SLICE"
|
|
47
|
+
SLICE_CAP_ENV = "PROFILE_SLICE_CAP"
|
|
48
|
+
|
|
49
|
+
# How long a single job may run before it must hand its state back, whether or
|
|
50
|
+
# not the link is anywhere near its deadline.
|
|
51
|
+
#
|
|
52
|
+
# The deadline protects against the *time limit*. On a preemptible partition
|
|
53
|
+
# that is not what usually stops the job: preemption arrives unannounced, and
|
|
54
|
+
# `SLURM_JOB_END_TIME` reports the limit rather than the eviction. A worker only
|
|
55
|
+
# returns its result -- and the parent only checkpoints it -- when its job ends,
|
|
56
|
+
# so without this cap a preemption three hours into a four-hour point discards
|
|
57
|
+
# all three hours.
|
|
58
|
+
#
|
|
59
|
+
# Capping the job turns that unbounded loss into a bounded one: at most one
|
|
60
|
+
# slice per point in flight. The cost is the n+1 evaluations each new job spends
|
|
61
|
+
# re-establishing its simplex, which at 30 minutes is a few percent.
|
|
62
|
+
DEFAULT_SLICE_CAP_S = 1800.0
|
|
63
|
+
|
|
64
|
+
# The shortest slice of wall clock worth starting a point in. A resumed point
|
|
65
|
+
# re-evaluates its whole simplex before it improves on anything, so a slice
|
|
66
|
+
# below this is spent on overhead and the work is better left to the next link.
|
|
67
|
+
DEFAULT_MIN_SLICE_S = 900.0
|
|
68
|
+
|
|
69
|
+
# Durations older than this many points are dropped. Keeps the file small and
|
|
70
|
+
# lets the estimate track a spec whose cost has changed rather than averaging
|
|
71
|
+
# over every run the directory has ever seen.
|
|
72
|
+
_KEEP_DURATIONS = 200
|
|
73
|
+
|
|
74
|
+
# Writing timing.json on every result would be one fsync per point for data
|
|
75
|
+
# that is only advisory; writing it only at the end would lose it to the very
|
|
76
|
+
# eviction it exists to survive.
|
|
77
|
+
_SAVE_EVERY = 10
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class DeadlineReached(RuntimeError):
|
|
81
|
+
"""Raised when the wall budget stopped a batch before its jobs were done.
|
|
82
|
+
|
|
83
|
+
Carries how much work was left so the caller can say so plainly. This is a
|
|
84
|
+
normal, successful outcome of a link on a preemptible queue -- not a
|
|
85
|
+
failure -- and callers are expected to catch it, finish reporting on what
|
|
86
|
+
they have, and exit 0.
|
|
87
|
+
"""
|
|
88
|
+
|
|
89
|
+
def __init__(self, n_remaining, label=None):
|
|
90
|
+
self.n_remaining = int(n_remaining)
|
|
91
|
+
self.label = label
|
|
92
|
+
where = f" in {label}" if label else ""
|
|
93
|
+
super().__init__(
|
|
94
|
+
f"wall-clock budget reached with {self.n_remaining} point(s) "
|
|
95
|
+
f"not started{where}"
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
# ---------------------------------------------------------------------------
|
|
100
|
+
# Parsing and discovery
|
|
101
|
+
# ---------------------------------------------------------------------------
|
|
102
|
+
|
|
103
|
+
def parse_duration(text):
|
|
104
|
+
"""Seconds from ``4h``, ``3.5h``, ``90m``, ``45s``, ``4:00:00`` or ``1-12:00:00``.
|
|
105
|
+
|
|
106
|
+
A bare number is seconds. That differs from ``sbatch --time``, where a bare
|
|
107
|
+
number is minutes, so prefer an explicit suffix when writing one by hand.
|
|
108
|
+
Returns None for anything unparseable, which the callers treat as "no
|
|
109
|
+
budget" rather than as an error -- a malformed value must not take down a
|
|
110
|
+
run that would otherwise have completed.
|
|
111
|
+
"""
|
|
112
|
+
if text is None:
|
|
113
|
+
return None
|
|
114
|
+
if isinstance(text, (int, float)):
|
|
115
|
+
return float(text) if text > 0 else None
|
|
116
|
+
|
|
117
|
+
s = str(text).strip().lower()
|
|
118
|
+
if not s:
|
|
119
|
+
return None
|
|
120
|
+
|
|
121
|
+
days = 0.0
|
|
122
|
+
if "-" in s:
|
|
123
|
+
head, _, s = s.partition("-")
|
|
124
|
+
try:
|
|
125
|
+
days = float(head)
|
|
126
|
+
except ValueError:
|
|
127
|
+
return None
|
|
128
|
+
|
|
129
|
+
try:
|
|
130
|
+
if ":" in s:
|
|
131
|
+
parts = [float(p) for p in s.split(":")]
|
|
132
|
+
if len(parts) > 3:
|
|
133
|
+
return None
|
|
134
|
+
while len(parts) < 3:
|
|
135
|
+
parts.insert(0, 0.0)
|
|
136
|
+
h, m, sec = parts
|
|
137
|
+
total = h * 3600.0 + m * 60.0 + sec
|
|
138
|
+
elif s[-1] in "smhd":
|
|
139
|
+
mult = {"s": 1.0, "m": 60.0, "h": 3600.0, "d": 86400.0}[s[-1]]
|
|
140
|
+
total = float(s[:-1]) * mult
|
|
141
|
+
else:
|
|
142
|
+
total = float(s)
|
|
143
|
+
except (ValueError, IndexError):
|
|
144
|
+
return None
|
|
145
|
+
|
|
146
|
+
total += days * 86400.0
|
|
147
|
+
return total if total > 0 else None
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _deadline_from_slurm_env():
|
|
151
|
+
"""``SLURM_JOB_END_TIME``: the scheduler's own answer, in epoch seconds."""
|
|
152
|
+
raw = os.environ.get("SLURM_JOB_END_TIME")
|
|
153
|
+
if not raw:
|
|
154
|
+
return None
|
|
155
|
+
try:
|
|
156
|
+
end = float(raw)
|
|
157
|
+
except ValueError:
|
|
158
|
+
return None
|
|
159
|
+
return end if end > 0 else None
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _deadline_from_scontrol():
|
|
163
|
+
"""The same answer via ``scontrol``, for builds that export no end time.
|
|
164
|
+
|
|
165
|
+
Deliberately best-effort: a missing binary, a slow controller or an
|
|
166
|
+
unparseable field all return None and leave the run unlimited, which is the
|
|
167
|
+
behaviour that existed before any of this.
|
|
168
|
+
"""
|
|
169
|
+
job_id = os.environ.get("SLURM_JOB_ID") or os.environ.get("SLURM_JOBID")
|
|
170
|
+
if not job_id:
|
|
171
|
+
return None
|
|
172
|
+
try:
|
|
173
|
+
out = subprocess.run(
|
|
174
|
+
["scontrol", "show", "job", "-o", str(job_id)],
|
|
175
|
+
capture_output=True, text=True, timeout=10, check=False,
|
|
176
|
+
).stdout
|
|
177
|
+
except (OSError, subprocess.SubprocessError):
|
|
178
|
+
return None
|
|
179
|
+
|
|
180
|
+
for field in out.split():
|
|
181
|
+
if not field.startswith("EndTime="):
|
|
182
|
+
continue
|
|
183
|
+
value = field[len("EndTime="):]
|
|
184
|
+
if value in ("Unknown", "None", ""):
|
|
185
|
+
return None
|
|
186
|
+
try:
|
|
187
|
+
# Slurm renders local time as YYYY-MM-DDTHH:MM:SS.
|
|
188
|
+
return time.mktime(time.strptime(value, "%Y-%m-%dT%H:%M:%S"))
|
|
189
|
+
except ValueError:
|
|
190
|
+
return None
|
|
191
|
+
return None
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def resolve_deadline(wall_time=None, now=None):
|
|
195
|
+
"""Absolute epoch time this process should be finished by, or None.
|
|
196
|
+
|
|
197
|
+
Sources, in order of trust:
|
|
198
|
+
|
|
199
|
+
1. *wall_time*, else ``PROFILE_WALL_TIME`` -- an explicit budget measured
|
|
200
|
+
from now. It wins over the scheduler because it is how someone asks for
|
|
201
|
+
a shorter link than the allocation allows, and because it is the only
|
|
202
|
+
source that exists on a laptop.
|
|
203
|
+
2. ``SLURM_JOB_END_TIME``.
|
|
204
|
+
3. ``scontrol show job``.
|
|
205
|
+
|
|
206
|
+
None means no deadline, which turns every downstream check into a no-op.
|
|
207
|
+
"""
|
|
208
|
+
now = time.time() if now is None else now
|
|
209
|
+
|
|
210
|
+
budget = parse_duration(wall_time)
|
|
211
|
+
if budget is None:
|
|
212
|
+
budget = parse_duration(os.environ.get(WALL_TIME_ENV))
|
|
213
|
+
if budget is not None:
|
|
214
|
+
return now + budget
|
|
215
|
+
|
|
216
|
+
for source in (_deadline_from_slurm_env, _deadline_from_scontrol):
|
|
217
|
+
end = source()
|
|
218
|
+
if end is not None and end > now:
|
|
219
|
+
return end
|
|
220
|
+
return None
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _default_margin():
|
|
224
|
+
return parse_duration(os.environ.get(MARGIN_ENV)) or DEFAULT_MARGIN_S
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _default_min_slice():
|
|
228
|
+
return parse_duration(os.environ.get(MIN_SLICE_ENV)) or DEFAULT_MIN_SLICE_S
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def partition_is_preemptible():
|
|
232
|
+
"""Whether this job can be evicted before its time limit.
|
|
233
|
+
|
|
234
|
+
On Hyak the checkpoint partitions -- ``ckpt``, ``ckpt-g2``, ``ckpt-all`` --
|
|
235
|
+
run on other groups' idle nodes and are stopped without notice when an
|
|
236
|
+
owner reclaims them. Every other partition runs to its ``--time`` and is
|
|
237
|
+
not preempted.
|
|
238
|
+
|
|
239
|
+
Read from the partition name because that is the only thing Slurm exposes
|
|
240
|
+
that actually distinguishes the two. Off a scheduler entirely there is
|
|
241
|
+
nothing to be preempted by, so the answer is False.
|
|
242
|
+
"""
|
|
243
|
+
partition = (os.environ.get("SLURM_JOB_PARTITION") or "").lower()
|
|
244
|
+
return "ckpt" in partition
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _default_slice_cap():
|
|
248
|
+
"""How often a job must hand its optimizer state back.
|
|
249
|
+
|
|
250
|
+
Slicing exists to bound what an eviction destroys, and it is not free: a
|
|
251
|
+
job boundary empties the evaluation cache, so the next job spends n+1 real
|
|
252
|
+
evaluations re-establishing its simplex before it can improve on anything.
|
|
253
|
+
On the SILK spec that is 16 evaluations at 113 s -- half an hour per
|
|
254
|
+
resume. Paying that against a preemption that cannot happen is a large,
|
|
255
|
+
silent waste, which is why this is not simply always on.
|
|
256
|
+
|
|
257
|
+
So: cap on a preemptible partition, and run to the link's own deadline
|
|
258
|
+
anywhere else. ``PROFILE_SLICE_CAP`` overrides in both directions, with
|
|
259
|
+
``off`` (or ``0``) meaning "never slice".
|
|
260
|
+
"""
|
|
261
|
+
raw = os.environ.get(SLICE_CAP_ENV)
|
|
262
|
+
if raw is not None and str(raw).strip():
|
|
263
|
+
text = str(raw).strip().lower()
|
|
264
|
+
if text in ("0", "off", "no", "none", "never", "inf", "infinite"):
|
|
265
|
+
return float("inf")
|
|
266
|
+
parsed = parse_duration(raw)
|
|
267
|
+
if parsed:
|
|
268
|
+
return parsed
|
|
269
|
+
return DEFAULT_SLICE_CAP_S if partition_is_preemptible() else float("inf")
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
# ---------------------------------------------------------------------------
|
|
273
|
+
# Observed point durations
|
|
274
|
+
# ---------------------------------------------------------------------------
|
|
275
|
+
|
|
276
|
+
def _quantile(values, q):
|
|
277
|
+
"""Linear-interpolated quantile of an unsorted list, or None if empty."""
|
|
278
|
+
vals = sorted(v for v in values if v is not None and math.isfinite(v))
|
|
279
|
+
if not vals:
|
|
280
|
+
return None
|
|
281
|
+
if len(vals) == 1:
|
|
282
|
+
return vals[0]
|
|
283
|
+
pos = q * (len(vals) - 1)
|
|
284
|
+
lo, hi = math.floor(pos), math.ceil(pos)
|
|
285
|
+
if lo == hi:
|
|
286
|
+
return vals[int(lo)]
|
|
287
|
+
return vals[int(lo)] + (vals[int(hi)] - vals[int(lo)]) * (pos - lo)
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
class RunBudget:
|
|
291
|
+
"""What is left of this launch's wall clock, and what a point costs.
|
|
292
|
+
|
|
293
|
+
Both halves are needed for the one question worth asking -- "is there time
|
|
294
|
+
to start another point?" -- and neither is useful alone: a deadline with no
|
|
295
|
+
cost model cannot tell a five-second point from a five-hour one, and a cost
|
|
296
|
+
model with no deadline has nothing to compare against.
|
|
297
|
+
|
|
298
|
+
An unset deadline makes every method permissive, so the same object can be
|
|
299
|
+
constructed unconditionally and passed everywhere.
|
|
300
|
+
"""
|
|
301
|
+
|
|
302
|
+
def __init__(self, deadline=None, margin_s=None, timing_path=None,
|
|
303
|
+
quantile=0.9, min_slice_s=None, slice_cap_s=None):
|
|
304
|
+
self.deadline = deadline
|
|
305
|
+
self.margin_s = _default_margin() if margin_s is None else float(margin_s)
|
|
306
|
+
self.min_slice_s = (_default_min_slice() if min_slice_s is None
|
|
307
|
+
else float(min_slice_s))
|
|
308
|
+
self.slice_cap_s = (_default_slice_cap() if slice_cap_s is None
|
|
309
|
+
else float(slice_cap_s))
|
|
310
|
+
self.timing_path = timing_path
|
|
311
|
+
self.quantile = float(quantile)
|
|
312
|
+
self.durations, self.per_eval = self._load()
|
|
313
|
+
self._n_since_save = 0
|
|
314
|
+
# Set once a batch has been cut short, so the run can report honestly
|
|
315
|
+
# that it stopped for time rather than because the work was done.
|
|
316
|
+
self.stopped_early = False
|
|
317
|
+
|
|
318
|
+
# -- persistence -------------------------------------------------------
|
|
319
|
+
|
|
320
|
+
def _load(self):
|
|
321
|
+
if not self.timing_path or not os.path.exists(self.timing_path):
|
|
322
|
+
return [], []
|
|
323
|
+
try:
|
|
324
|
+
with open(self.timing_path, "r", encoding="utf-8") as fh:
|
|
325
|
+
data = json.load(fh)
|
|
326
|
+
vals = [float(v) for v in data.get("durations", [])]
|
|
327
|
+
per = [float(v) for v in data.get("per_eval", [])]
|
|
328
|
+
except (OSError, ValueError, TypeError, AttributeError):
|
|
329
|
+
# A truncated or hand-edited timing file is not worth failing a
|
|
330
|
+
# multi-hour run over; an empty history just means this link
|
|
331
|
+
# calibrates itself the way the first one did.
|
|
332
|
+
return [], []
|
|
333
|
+
|
|
334
|
+
def _clean(xs):
|
|
335
|
+
return [v for v in xs
|
|
336
|
+
if math.isfinite(v) and v > 0][-_KEEP_DURATIONS:]
|
|
337
|
+
|
|
338
|
+
return _clean(vals), _clean(per)
|
|
339
|
+
|
|
340
|
+
def save(self):
|
|
341
|
+
"""Write the duration history, atomically.
|
|
342
|
+
|
|
343
|
+
Written via a temporary file and ``os.replace`` because several array
|
|
344
|
+
tasks may share the directory: last writer wins, which is fine for
|
|
345
|
+
advisory data, but a half-written file read by the next link is not.
|
|
346
|
+
"""
|
|
347
|
+
if not self.timing_path:
|
|
348
|
+
return
|
|
349
|
+
payload = {
|
|
350
|
+
"durations": self.durations[-_KEEP_DURATIONS:],
|
|
351
|
+
"per_eval": self.per_eval[-_KEEP_DURATIONS:],
|
|
352
|
+
"quantile": self.quantile,
|
|
353
|
+
"updated": time.strftime("%Y-%m-%dT%H:%M:%S"),
|
|
354
|
+
}
|
|
355
|
+
tmp = f"{self.timing_path}.{os.getpid()}.tmp"
|
|
356
|
+
try:
|
|
357
|
+
os.makedirs(os.path.dirname(self.timing_path), exist_ok=True)
|
|
358
|
+
with open(tmp, "w", encoding="utf-8") as fh:
|
|
359
|
+
json.dump(payload, fh)
|
|
360
|
+
os.replace(tmp, self.timing_path)
|
|
361
|
+
except OSError:
|
|
362
|
+
try:
|
|
363
|
+
os.unlink(tmp)
|
|
364
|
+
except OSError:
|
|
365
|
+
pass
|
|
366
|
+
else:
|
|
367
|
+
self._n_since_save = 0
|
|
368
|
+
|
|
369
|
+
def record(self, seconds, n_evals=None):
|
|
370
|
+
"""Note how long one point took, and how many evaluations it bought."""
|
|
371
|
+
try:
|
|
372
|
+
v = float(seconds)
|
|
373
|
+
except (TypeError, ValueError):
|
|
374
|
+
return
|
|
375
|
+
if not math.isfinite(v) or v <= 0:
|
|
376
|
+
return
|
|
377
|
+
self.durations.append(v)
|
|
378
|
+
del self.durations[:-_KEEP_DURATIONS]
|
|
379
|
+
try:
|
|
380
|
+
n = int(n_evals)
|
|
381
|
+
except (TypeError, ValueError):
|
|
382
|
+
n = 0
|
|
383
|
+
if n > 0:
|
|
384
|
+
self.per_eval.append(v / n)
|
|
385
|
+
del self.per_eval[:-_KEEP_DURATIONS]
|
|
386
|
+
self._n_since_save += 1
|
|
387
|
+
if self._n_since_save >= _SAVE_EVERY:
|
|
388
|
+
self.save()
|
|
389
|
+
|
|
390
|
+
def seconds_per_eval(self):
|
|
391
|
+
"""Measured cost of one objective evaluation, or None.
|
|
392
|
+
|
|
393
|
+
Sent to the worker so the very first slice of a point can be sized to
|
|
394
|
+
the time available instead of guessed at. It is the median rather than
|
|
395
|
+
an upper quantile: this one is used to *size* work, not to decide
|
|
396
|
+
whether to start it, and over-estimating here would cut every slice
|
|
397
|
+
short and pay the restart overhead more often than necessary.
|
|
398
|
+
"""
|
|
399
|
+
vals = sorted(v for v in self.per_eval if math.isfinite(v) and v > 0)
|
|
400
|
+
if not vals:
|
|
401
|
+
return None
|
|
402
|
+
return vals[len(vals) // 2]
|
|
403
|
+
|
|
404
|
+
# -- the actual question -----------------------------------------------
|
|
405
|
+
|
|
406
|
+
@property
|
|
407
|
+
def is_limited(self):
|
|
408
|
+
return self.deadline is not None
|
|
409
|
+
|
|
410
|
+
def remaining(self, now=None):
|
|
411
|
+
"""Seconds left before the deadline; ``inf`` when there is none."""
|
|
412
|
+
if self.deadline is None:
|
|
413
|
+
return float("inf")
|
|
414
|
+
return self.deadline - (time.time() if now is None else now)
|
|
415
|
+
|
|
416
|
+
def work_deadline(self):
|
|
417
|
+
"""The last moment this link may still be computing.
|
|
418
|
+
|
|
419
|
+
Leaves the margin for trace assembly, plotting and the results write.
|
|
420
|
+
None means no limit, and points run to convergence.
|
|
421
|
+
"""
|
|
422
|
+
if self.deadline is None:
|
|
423
|
+
return None
|
|
424
|
+
return self.deadline - self.margin_s
|
|
425
|
+
|
|
426
|
+
def job_deadline(self, min_needed_s=None, now=None):
|
|
427
|
+
"""When the job being started now must hand its state back.
|
|
428
|
+
|
|
429
|
+
The earlier of "the link is ending" and "this job has had its slice".
|
|
430
|
+
The second bound is what makes preemption survivable: a job that ends
|
|
431
|
+
every ``slice_cap_s`` has been checkpointed that recently, so an
|
|
432
|
+
eviction costs at most one slice per point in flight instead of
|
|
433
|
+
everything since the point began.
|
|
434
|
+
|
|
435
|
+
*min_needed_s* is how long the optimizer needs to reach a state it can
|
|
436
|
+
hand back at all. The cap is raised to meet it, because a cap below it
|
|
437
|
+
is worse than no cap: the job is stopped before any state exists, so it
|
|
438
|
+
resumes from scratch and the slice is spent for nothing. A cap only
|
|
439
|
+
helps if the work inside it can finish. The observed case was a
|
|
440
|
+
30-minute default against a 16-parameter model at 116 s per evaluation,
|
|
441
|
+
which needs about half an hour just to establish its simplex.
|
|
442
|
+
|
|
443
|
+
The cap does not depend on the link's own deadline being known, and
|
|
444
|
+
that is deliberate. ``SLURM_JOB_END_TIME`` is occasionally absent and
|
|
445
|
+
``scontrol`` occasionally does not answer, and when both fail
|
|
446
|
+
:func:`resolve_deadline` returns None. Tying the cap to the deadline
|
|
447
|
+
meant a job that could not read its end time also stopped slicing: no
|
|
448
|
+
point ever handed its state back, nothing was checkpointed, and the
|
|
449
|
+
first eviction discarded every hour of it. The two facts are separate
|
|
450
|
+
-- "how long until the allocation ends" and "how often must state be
|
|
451
|
+
saved" -- so they are now read separately.
|
|
452
|
+
|
|
453
|
+
Off a scheduler ``slice_cap_s`` is already infinite, so a laptop run
|
|
454
|
+
with no deadline still returns None and points run to convergence
|
|
455
|
+
exactly as they always did.
|
|
456
|
+
"""
|
|
457
|
+
end = self.work_deadline()
|
|
458
|
+
cap = self.slice_cap_s
|
|
459
|
+
if min_needed_s and min_needed_s > cap:
|
|
460
|
+
cap = float(min_needed_s)
|
|
461
|
+
if math.isinf(cap):
|
|
462
|
+
return end
|
|
463
|
+
now = time.time() if now is None else now
|
|
464
|
+
sliced = now + cap
|
|
465
|
+
return sliced if end is None else min(end, sliced)
|
|
466
|
+
|
|
467
|
+
def estimate(self):
|
|
468
|
+
"""How long a point has been taking, or None.
|
|
469
|
+
|
|
470
|
+
The upper quantile rather than the mean, because it is read as "how
|
|
471
|
+
much longer might this run", not "what is typical". Used for the log
|
|
472
|
+
and the progress estimate; since points can now be interrupted and
|
|
473
|
+
resumed it no longer gates admission.
|
|
474
|
+
"""
|
|
475
|
+
return _quantile(self.durations, self.quantile)
|
|
476
|
+
|
|
477
|
+
def admits(self, now=None):
|
|
478
|
+
"""Whether there is room to start one more point.
|
|
479
|
+
|
|
480
|
+
The rule here changed when points became interruptible, and the old one
|
|
481
|
+
would now be actively harmful. It compared the time left against how
|
|
482
|
+
long a point takes, and refused anything that would not fit -- correct
|
|
483
|
+
when an unfinished point was a total loss, but catastrophic once a
|
|
484
|
+
point can take forty hours: every four-hour link would look at a
|
|
485
|
+
forty-hour estimate, admit nothing at all, and the profile would never
|
|
486
|
+
advance.
|
|
487
|
+
|
|
488
|
+
What matters instead is whether a slice is long enough to be worth
|
|
489
|
+
starting. A resumed point re-evaluates its simplex before it makes any
|
|
490
|
+
progress, so a slice shorter than that is spent entirely on overhead;
|
|
491
|
+
``min_slice_s`` is the floor below which the link stops handing out
|
|
492
|
+
work and lets the margin do its job.
|
|
493
|
+
"""
|
|
494
|
+
if self.deadline is None:
|
|
495
|
+
return True
|
|
496
|
+
return self.remaining(now) >= self.margin_s + self.min_slice_s
|
|
497
|
+
|
|
498
|
+
def describe(self, effective_slice_s=None):
|
|
499
|
+
"""One line for the log, saying what the budget will actually do.
|
|
500
|
+
|
|
501
|
+
Three separate numbers govern three separate things, and an earlier
|
|
502
|
+
version of this line conflated them -- it reported ``min_slice_s`` as
|
|
503
|
+
"points interrupted at N min remaining", which is not what that
|
|
504
|
+
threshold does and left the slice cap, the mechanism that actually
|
|
505
|
+
bounds preemption damage, unmentioned.
|
|
506
|
+
|
|
507
|
+
*effective_slice_s* is the cap after being raised to fit a simplex,
|
|
508
|
+
known only to the caller that has the jobs in hand.
|
|
509
|
+
"""
|
|
510
|
+
if self.deadline is None:
|
|
511
|
+
return "no wall-clock deadline; running until the work is done"
|
|
512
|
+
|
|
513
|
+
est = self.estimate()
|
|
514
|
+
est_txt = (f"points have been taking up to {est / 60.0:.0f} min"
|
|
515
|
+
if est is not None else "no timing history yet")
|
|
516
|
+
|
|
517
|
+
if math.isinf(self.slice_cap_s):
|
|
518
|
+
# Not a preemptible partition, or slicing switched off. Points run
|
|
519
|
+
# to convergence and are interrupted only by the link's own
|
|
520
|
+
# deadline, which saves the n+1 evaluations a resume would spend
|
|
521
|
+
# rebuilding its simplex.
|
|
522
|
+
hand_back = ("running points are interrupted only at the deadline "
|
|
523
|
+
"(no slice cap: nothing here preempts)")
|
|
524
|
+
else:
|
|
525
|
+
cap = max(self.slice_cap_s, effective_slice_s or 0.0)
|
|
526
|
+
raised = (" (raised to fit a simplex)"
|
|
527
|
+
if cap > self.slice_cap_s else "")
|
|
528
|
+
hand_back = (f"running points hand their state back every "
|
|
529
|
+
f"{cap / 60.0:.0f} min{raised}")
|
|
530
|
+
|
|
531
|
+
return (f"{self.remaining() / 60.0:.0f} min left, "
|
|
532
|
+
f"{self.margin_s / 60.0:.0f} min of it reserved for the "
|
|
533
|
+
f"write-up; no new point starts with under "
|
|
534
|
+
f"{self.min_slice_s / 60.0:.0f} min to go; {hand_back}; "
|
|
535
|
+
f"{est_txt}")
|