PyAntiGen 1.0.9__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. framework/AntimonyGen.py +48 -0
  2. framework/RxnDict_to_antimony.py +594 -0
  3. framework/TelluriumGen.py +16 -0
  4. framework/__init__.py +0 -0
  5. framework/antimony_utils.py +294 -0
  6. framework/cli.py +229 -0
  7. framework/data_interpolation.py +340 -0
  8. framework/isotopomer_tools.py +41 -0
  9. framework/model_generation.py +46 -0
  10. framework/models.py +189 -0
  11. framework/module_base.py +42 -0
  12. framework/pyantigen.py +51 -0
  13. framework/rate_laws.py +101 -0
  14. framework/reaction_creation.py +43 -0
  15. framework/template/Example/AntiGen_paths.py +23 -0
  16. framework/template/Example/Engine/Anchor_cache.py +193 -0
  17. framework/template/Example/Engine/Deadline.py +535 -0
  18. framework/template/Example/Engine/Evaluator.py +1176 -0
  19. framework/template/Example/Engine/Event_times.py +491 -0
  20. framework/template/Example/Engine/Fast_profile.py +701 -0
  21. framework/template/Example/Engine/Fit_cache.py +329 -0
  22. framework/template/Example/Engine/Identifiability.py +698 -0
  23. framework/template/Example/Engine/Model_optimize.py +1483 -0
  24. framework/template/Example/Engine/Model_simulate.py +124 -0
  25. framework/template/Example/Engine/Nuisance_sensitivity.py +298 -0
  26. framework/template/Example/Engine/Optimize.py +6862 -0
  27. framework/template/Example/Engine/Petab_export.py +398 -0
  28. framework/template/Example/Engine/Preequil_cache.py +361 -0
  29. framework/template/Example/Engine/Profile_checkpoint.py +399 -0
  30. framework/template/Example/Engine/Results.py +395 -0
  31. framework/template/Example/Engine/Sensitivity_analysis.py +320 -0
  32. framework/template/Example/Engine/Simulate.py +617 -0
  33. framework/template/Example/Flipflop_reference.py +401 -0
  34. framework/template/Example/Model_generate.py +37 -0
  35. framework/template/Example/Model_run.py +261 -0
  36. framework/template/Example/Modules/Data.py +63 -0
  37. framework/template/Example/Modules/Events.py +14 -0
  38. framework/template/Example/Modules/Experiment.py +194 -0
  39. framework/template/Example/Modules/Loss_config.py +61 -0
  40. framework/template/Example/Modules/Observed_species.py +3 -0
  41. framework/template/Example/Modules/Optimizer_settings.py +258 -0
  42. framework/template/Example/Modules/Plots.py +89 -0
  43. framework/template/Example/Modules/Solver_settings.py +16 -0
  44. framework/template/Example/Modules/Update_opt_parameters.py +24 -0
  45. framework/template/Example/Modules/Update_parameters.py +49 -0
  46. framework/template/data/ADneg.csv +27 -0
  47. framework/template/data/ADpos.csv +27 -0
  48. framework/template/data/Flipflop.csv +29 -0
  49. framework/template/data/make_flipflop_data.py +174 -0
  50. pyantigen-1.0.9.dist-info/METADATA +129 -0
  51. pyantigen-1.0.9.dist-info/RECORD +55 -0
  52. pyantigen-1.0.9.dist-info/WHEEL +5 -0
  53. pyantigen-1.0.9.dist-info/entry_points.txt +2 -0
  54. pyantigen-1.0.9.dist-info/licenses/LICENSE +21 -0
  55. pyantigen-1.0.9.dist-info/top_level.txt +1 -0
@@ -0,0 +1,399 @@
1
+ """Append-only checkpoint store for profile-likelihood runs.
2
+
3
+ A full profile on a QSP model is hours of compute. Without checkpointing, a
4
+ crash, a reboot, or a decision to move the job to a bigger machine throws all of
5
+ it away -- which is why "massively parallel" and "checkpointed" are the same
6
+ requirement, not two.
7
+
8
+ Format: one JSON object per line, one file per parameter, under
9
+
10
+ results/<MODEL>/profiles/<run_id>/<param>.jsonl
11
+
12
+ Append-only JSONL is chosen deliberately over a single JSON document or a
13
+ database:
14
+
15
+ * A half-written line is detectable and discardable; a half-written JSON
16
+ document is not, so a kill during a write cannot corrupt earlier results.
17
+ * One file per parameter means the 2k concurrent writers never contend.
18
+ * Concatenating directories from two machines merges two partial runs, so a job
19
+ can be split across boxes and reassembled.
20
+
21
+ Each record carries the hashes of the spec and the model, so a resume against a
22
+ changed model or a changed optimization spec is detected rather than silently
23
+ mixing incompatible points.
24
+ """
25
+
26
+ import hashlib
27
+ import json
28
+ import os
29
+ import re
30
+ import time
31
+ from datetime import datetime
32
+
33
+ import numpy as np
34
+
35
+
36
+ def _safe_name(s):
37
+ """Filesystem-safe version of a parameter name."""
38
+ out = re.sub(r"[^A-Za-z0-9_.-]", "_", str(s))
39
+ return out or "param"
40
+
41
+
42
+ # Files written as "<name>.<pid>.tmp" and renamed into place. A process killed
43
+ # between the write and the rename leaves one behind, and on a preemptible
44
+ # queue that happens routinely.
45
+ _TEMP_SUFFIX = ".tmp"
46
+
47
+ # Only temp files older than this are swept. A younger one may belong to a
48
+ # process that is still running -- another array task, or this one -- and
49
+ # deleting it would make that process's rename fail for no reason.
50
+ _TEMP_MAX_AGE_S = 3600.0
51
+
52
+
53
+ def sweep_stale_temp_files(directory, max_age_s=_TEMP_MAX_AGE_S):
54
+ """Delete abandoned temp files in *directory*; return how many went.
55
+
56
+ These are litter rather than corruption: the atomic-rename pattern that
57
+ creates them is exactly what stops a half-written file from ever being
58
+ read. But one accumulates per kill, and a run that is preempted a hundred
59
+ times leaves a hundred of them beside the results.
60
+
61
+ Never raises. A sweep that cannot run is not a reason to fail a run that
62
+ would otherwise work.
63
+ """
64
+ removed = 0
65
+ try:
66
+ names = os.listdir(directory)
67
+ except OSError:
68
+ return 0
69
+
70
+ now = time.time()
71
+ for name in names:
72
+ if not name.endswith(_TEMP_SUFFIX):
73
+ continue
74
+ path = os.path.join(directory, name)
75
+ try:
76
+ if now - os.path.getmtime(path) < max_age_s:
77
+ continue
78
+ os.unlink(path)
79
+ removed += 1
80
+ except OSError:
81
+ continue
82
+ return removed
83
+
84
+
85
+ def record_is_better(rec, prev):
86
+ """Whether *rec* should replace *prev* as the value stored at a grid point.
87
+
88
+ Lower NLL wins. Every profile point is an *upper bound* on the true profile
89
+ -- it is a real evaluation of the nuisance minimum, just not necessarily a
90
+ converged one -- so when two records exist for the same fixed value both are
91
+ valid and the lower one is strictly closer to the truth. Taking the minimum
92
+ is what makes the warm-started continuation pass safe to run on top of an
93
+ existing checkpoint: it can only lower the curve, never raise it.
94
+
95
+ Ties go to the later record so that a re-run which only updates bookkeeping
96
+ (a warm pass that did not improve a point, but must still mark it as visited
97
+ so a resume does not redo it) is not discarded.
98
+
99
+ A record whose NLL is missing or non-finite -- a failed simulation, a
100
+ sentinel -- never displaces a usable one. Getting this backwards would let a
101
+ single failed re-evaluation erase a good point that cost minutes to compute.
102
+ """
103
+ def _nll(r):
104
+ try:
105
+ v = float(r.get("nll"))
106
+ except (TypeError, ValueError):
107
+ return None
108
+ return v if np.isfinite(v) else None
109
+
110
+ a, b = _nll(rec), _nll(prev)
111
+ if a is None:
112
+ # Junk never wins over a real value; two junk records tie to the later.
113
+ return b is None
114
+ if b is None:
115
+ return True
116
+ return a <= b
117
+
118
+
119
+ def _round_sig(x, sig=6):
120
+ """Round to *sig* significant figures (0 and non-finite pass through)."""
121
+ x = float(x)
122
+ if x == 0.0 or not np.isfinite(x):
123
+ return 0.0 if x == 0.0 else None
124
+ return round(x, sig - 1 - int(np.floor(np.log10(abs(x)))))
125
+
126
+
127
+ def solver_fingerprint(replicates):
128
+ """A hash of how every replicate will actually be integrated, or None.
129
+
130
+ The solver settings are part of the objective, not decoration. Output
131
+ density feeds ``np.interp`` onto the data times, and the tolerances and step
132
+ budget decide what the integrator returns, so changing any of them moves the
133
+ NLL -- on the SILK spec, dropping the labelling window from 200,000 output
134
+ points to 10,000 shifts it by about 1.5e-3 nats.
135
+
136
+ That is small against the 1.9207 threshold and large against nothing at all,
137
+ which is exactly the situation a fingerprint is for: points computed under
138
+ two different settings are two different curves, and a checkpoint directory
139
+ that mixes them is quietly wrong. Nothing else in the fingerprint sees these
140
+ numbers -- they live in functions on the replicates, not in the model text
141
+ or the spec -- so before this they could change under a resumed run with no
142
+ signal at all.
143
+
144
+ Best-effort by design. A settings function that will not run here would
145
+ otherwise take down a profile that was going to work, so the failure is
146
+ recorded as an unknown marker for that replicate rather than raised. The
147
+ marker still participates in the hash, so "we could not read this" is itself
148
+ a stable, distinguishable state.
149
+ """
150
+ if not replicates:
151
+ return None
152
+
153
+ def _norm(value):
154
+ if isinstance(value, dict):
155
+ return {str(k): _norm(v) for k, v in sorted(value.items())}
156
+ if isinstance(value, (list, tuple)):
157
+ return [_norm(v) for v in value]
158
+ if isinstance(value, (int, float, np.floating, np.integer)):
159
+ # Rounded so float noise in a computed block boundary -- an age in
160
+ # hours, say -- cannot invalidate a directory on its own.
161
+ return _round_sig(float(value), 9)
162
+ if isinstance(value, (str, bool)) or value is None:
163
+ return value
164
+ return f"<{type(value).__name__}>"
165
+
166
+ entries = {}
167
+
168
+ # Engine-level constants that decide what the integrator returns. They are
169
+ # not in any settings dict -- they live in Engine.Simulate -- but a run with
170
+ # a different tolerance floor or dust threshold is integrating a different
171
+ # problem, and its points do not belong in the same directory. Read lazily
172
+ # so this module stays importable without the simulation stack.
173
+ try:
174
+ from Engine.Simulate import _DUST_THRESHOLD, _MIN_ABSOLUTE_TOLERANCE
175
+ entries["__engine__"] = {
176
+ "min_absolute_tolerance": float(_MIN_ABSOLUTE_TOLERANCE),
177
+ "dust_threshold": float(_DUST_THRESHOLD),
178
+ }
179
+ except Exception:
180
+ entries["__engine__"] = "<unreadable>"
181
+
182
+ for name, rep in sorted(replicates.items()):
183
+ fn = rep.get("Solver_settings") if hasattr(rep, "get") else None
184
+ if fn is None:
185
+ entries[str(name)] = "<no-solver-settings>"
186
+ continue
187
+ try:
188
+ settings = fn(rep)
189
+ entries[str(name)] = _norm(
190
+ {k: v for k, v in settings.items() if k != "event_times"}
191
+ )
192
+ except Exception as exc: # noqa: BLE001 - see the docstring
193
+ entries[str(name)] = f"<unreadable:{type(exc).__name__}>"
194
+
195
+ blob = json.dumps(entries, sort_keys=True, default=str)
196
+ return hashlib.sha256(blob.encode("utf-8")).hexdigest()[:16]
197
+
198
+
199
+ def spec_fingerprint(param_names, x_opt, groups, scales, model_text,
200
+ fixed_sigmas=None, solver_hash=None):
201
+ """Stable hashes identifying what a set of profile points belongs to.
202
+
203
+ Points computed against a different model, parameter set, optimum, or set of
204
+ solver settings must not be silently reused, so all of them are hashed and
205
+ stored on every record.
206
+
207
+ The optimum is rounded to six significant figures first. Profile points are
208
+ centred on it and their dNLL is measured relative to it, so a genuinely
209
+ different optimum must invalidate them -- but a re-fit that lands 1e-9 away
210
+ is the *same* optimum, and letting float noise discard hours of completed
211
+ work would make resuming useless in practice.
212
+
213
+ *solver_hash* comes from :func:`solver_fingerprint` and is passed by the
214
+ profile driver, which is the only caller with the replicates in hand. It is
215
+ optional so that callers without them -- tests, and anything profiling a
216
+ bare function -- keep working; when it is absent the hash is the same as it
217
+ was before solver settings were tracked.
218
+ """
219
+ model_hash = hashlib.sha256((model_text or "").encode("utf-8")).hexdigest()[:16]
220
+ spec_blob = json.dumps(
221
+ {
222
+ "param_names": list(param_names),
223
+ "x_opt": [_round_sig(v) for v in np.atleast_1d(x_opt)],
224
+ "scales": list(scales),
225
+ "groups": sorted(groups.keys()) if hasattr(groups, "keys") else None,
226
+ # Bumped when the meaning of a stored dNLL changes.
227
+ # v1 carried the objective's group averaging and weights.
228
+ # v2 was the summed unweighted NLL with sigmas frozen at the fit
229
+ # optimum -- a different function from the one the fit
230
+ # minimized, which is what let dNLL go negative.
231
+ # v3 is the concentrated Gaussian likelihood, with each block's
232
+ # sigma profiled out analytically. It is the same function the
233
+ # fit minimizes, so its dNLL is on a different scale again and
234
+ # v1/v2 points must not be resumed into a v3 run.
235
+ "likelihood_convention": "v3-concentrated-gaussian",
236
+ # Under v3 sigma is profiled out per evaluation rather than frozen,
237
+ # so these no longer enter dNLL. They are still hashed because they
238
+ # are a compact fingerprint of the residuals at the optimum, which
239
+ # does change whenever the fit lands somewhere else.
240
+ "sigmas": sorted(
241
+ (str(k), round(float(v), 12))
242
+ for k, v in (fixed_sigmas or {}).items()
243
+ ),
244
+ # Absent for callers with no replicates to read, which keeps their
245
+ # hash identical to what it was before solver settings were
246
+ # tracked; present, it makes any change to how the model is
247
+ # integrated start a new directory. See solver_fingerprint.
248
+ **({"solver": str(solver_hash)} if solver_hash else {}),
249
+ },
250
+ sort_keys=True,
251
+ )
252
+ spec_hash = hashlib.sha256(spec_blob.encode("utf-8")).hexdigest()[:16]
253
+ return model_hash, spec_hash
254
+
255
+
256
+ def default_run_id(tag, model_hash, spec_hash):
257
+ """Directory name for a run's checkpoints.
258
+
259
+ Deliberately excludes any timestamp: a run_id that changes every launch can
260
+ never resume, which defeats the entire mechanism. Identity comes from *what
261
+ is being profiled*, so relaunching the same problem finds its own results
262
+ and a different problem gets a different directory.
263
+ """
264
+ return f"{_safe_name(tag)}_{model_hash[:8]}_{spec_hash[:8]}"
265
+
266
+
267
+ class ProfileCheckpoint:
268
+ """Reads and appends profile points for one run."""
269
+
270
+ def __init__(self, root, run_id, model_hash, spec_hash, enabled=True):
271
+ self.enabled = bool(enabled)
272
+ self.run_id = run_id
273
+ self.model_hash = model_hash
274
+ self.spec_hash = spec_hash
275
+ self.dir = os.path.join(root, "profiles", run_id) if root else None
276
+ self._handles = {}
277
+ self.n_loaded = 0
278
+ self.n_skipped_stale = 0
279
+ self.n_temp_swept = 0
280
+ if self.enabled and self.dir:
281
+ os.makedirs(self.dir, exist_ok=True)
282
+ self.n_temp_swept = sweep_stale_temp_files(self.dir)
283
+
284
+ # -- paths -------------------------------------------------------------
285
+
286
+ def path_for(self, param_name):
287
+ return os.path.join(self.dir, f"{_safe_name(param_name)}.jsonl")
288
+
289
+ # -- reading -----------------------------------------------------------
290
+
291
+ def load(self, param_names):
292
+ """Return {param_name: {rounded_x_fixed: record}} for completed points.
293
+
294
+ Records whose model or spec hash does not match the current run are
295
+ counted and ignored -- resuming onto a changed model must not silently
296
+ blend old points with new ones.
297
+
298
+ Where a fixed value appears more than once -- which is what the
299
+ warm-started pass produces -- the *lowest* NLL is kept rather than the
300
+ last one written. Relying on write order would mean a later, worse
301
+ evaluation silently replaced a better earlier one on the next resume.
302
+ """
303
+ found = {name: {} for name in param_names}
304
+ if not (self.enabled and self.dir and os.path.isdir(self.dir)):
305
+ return found
306
+
307
+ for name in param_names:
308
+ path = self.path_for(name)
309
+ if not os.path.exists(path):
310
+ continue
311
+ with open(path, "r", encoding="utf-8") as fh:
312
+ for line in fh:
313
+ line = line.strip()
314
+ if not line:
315
+ continue
316
+ try:
317
+ rec = json.loads(line)
318
+ except json.JSONDecodeError:
319
+ # Truncated final line from a killed run: expected, skip.
320
+ continue
321
+ if (rec.get("model_hash") != self.model_hash
322
+ or rec.get("spec_hash") != self.spec_hash):
323
+ self.n_skipped_stale += 1
324
+ continue
325
+ if rec.get("status") != "ok":
326
+ continue
327
+ key = self._key(rec.get("x_fixed"))
328
+ if key is None:
329
+ continue
330
+ prev = found[name].get(key)
331
+ if prev is not None and not record_is_better(rec, prev):
332
+ continue
333
+ found[name][key] = rec
334
+
335
+ self.n_loaded = sum(len(v) for v in found.values())
336
+ return found
337
+
338
+ @staticmethod
339
+ def _key(x):
340
+ """Grid points are matched on value, rounded so float noise cannot
341
+ create a near-duplicate that gets recomputed every resume."""
342
+ try:
343
+ return round(float(x), 12)
344
+ except (TypeError, ValueError):
345
+ return None
346
+
347
+ # -- writing -----------------------------------------------------------
348
+
349
+ def append(self, record):
350
+ if not (self.enabled and self.dir):
351
+ return
352
+ name = record.get("param_name", "param")
353
+ rec = dict(record)
354
+ rec.setdefault("timestamp", datetime.now().isoformat(timespec="seconds"))
355
+ rec["model_hash"] = self.model_hash
356
+ rec["spec_hash"] = self.spec_hash
357
+ fh = self._handles.get(name)
358
+ if fh is None:
359
+ fh = open(self.path_for(name), "a", encoding="utf-8")
360
+ self._handles[name] = fh
361
+ json.dump(_jsonable(rec), fh)
362
+ fh.write("\n")
363
+ # Flush per record: the value of a checkpoint is entirely in surviving
364
+ # an abrupt kill, which buffering would defeat.
365
+ fh.flush()
366
+ os.fsync(fh.fileno())
367
+
368
+ def close(self):
369
+ for fh in self._handles.values():
370
+ try:
371
+ fh.close()
372
+ except Exception:
373
+ pass
374
+ self._handles.clear()
375
+
376
+ def __enter__(self):
377
+ return self
378
+
379
+ def __exit__(self, exc_type, exc, tb):
380
+ self.close()
381
+ return False
382
+
383
+
384
+ def _jsonable(obj):
385
+ """Convert numpy scalars/arrays so json.dump accepts them."""
386
+ if isinstance(obj, dict):
387
+ return {k: _jsonable(v) for k, v in obj.items()}
388
+ if isinstance(obj, (list, tuple)):
389
+ return [_jsonable(v) for v in obj]
390
+ if isinstance(obj, np.ndarray):
391
+ return obj.tolist()
392
+ if isinstance(obj, (np.integer,)):
393
+ return int(obj)
394
+ if isinstance(obj, (np.floating,)):
395
+ v = float(obj)
396
+ return v if np.isfinite(v) else None
397
+ if isinstance(obj, float):
398
+ return obj if np.isfinite(obj) else None
399
+ return obj