rafkit 0.5.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rafkit-0.6.0/PKG-INFO +499 -0
- rafkit-0.6.0/README.md +468 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/pyproject.toml +6 -1
- rafkit-0.6.0/src/rafkit/__init__.py +83 -0
- rafkit-0.6.0/src/rafkit/autocatalysis.py +223 -0
- rafkit-0.6.0/src/rafkit/dilution.py +194 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit/gillespie.py +79 -9
- rafkit-0.6.0/src/rafkit/permeation.py +135 -0
- rafkit-0.6.0/src/rafkit/thermo.py +858 -0
- rafkit-0.6.0/src/rafkit.egg-info/PKG-INFO +499 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit.egg-info/SOURCES.txt +10 -1
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit.egg-info/requires.txt +4 -0
- rafkit-0.6.0/tests/test_autocatalysis.py +309 -0
- rafkit-0.6.0/tests/test_dilution.py +118 -0
- rafkit-0.6.0/tests/test_permeation.py +281 -0
- rafkit-0.6.0/tests/test_thermo.py +772 -0
- rafkit-0.6.0/tests/test_thermo_kinetics.py +437 -0
- rafkit-0.5.0/PKG-INFO +0 -250
- rafkit-0.5.0/README.md +0 -222
- rafkit-0.5.0/src/rafkit/__init__.py +0 -47
- rafkit-0.5.0/src/rafkit.egg-info/PKG-INFO +0 -250
- {rafkit-0.5.0 → rafkit-0.6.0}/LICENSE +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/setup.cfg +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit/binary_polymer.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit/catalysis.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit/complementary_polymer.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit/crs.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit/firing_disk.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit/inhibition.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit/network.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit/pnml.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit/raf.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit.egg-info/dependency_links.txt +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/src/rafkit.egg-info/top_level.txt +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/tests/test_complementary_polymer.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/tests/test_crs.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/tests/test_firing_disk.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/tests/test_gillespie.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/tests/test_inhibition.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/tests/test_pnml.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/tests/test_published_examples.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/tests/test_raf.py +0 -0
- {rafkit-0.5.0 → rafkit-0.6.0}/tests/test_seeding.py +0 -0
rafkit-0.6.0/PKG-INFO
ADDED
|
@@ -0,0 +1,499 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: rafkit
|
|
3
|
+
Version: 0.6.0
|
|
4
|
+
Summary: Autocatalytic (RAF) sets in catalytic reaction networks: maximal RAFs, irreducible cores, and Kauffman binary polymer models.
|
|
5
|
+
Author: James P. Galasyn, Claude Théodore
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/jimgalasyn/rafkit
|
|
8
|
+
Project-URL: Issues, https://github.com/jimgalasyn/rafkit/issues
|
|
9
|
+
Keywords: autocatalytic-sets,RAF,origin-of-life,chemical-reaction-networks,systems-chemistry,binary-polymer-model,catalysis,abiogenesis
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Chemistry
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Life
|
|
19
|
+
Requires-Python: >=3.11
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Requires-Dist: numpy>=1.24
|
|
23
|
+
Provides-Extra: cac
|
|
24
|
+
Requires-Dist: scipy>=1.10; extra == "cac"
|
|
25
|
+
Provides-Extra: test
|
|
26
|
+
Requires-Dist: pytest>=8; extra == "test"
|
|
27
|
+
Requires-Dist: pytest-cov>=4; extra == "test"
|
|
28
|
+
Requires-Dist: pytest-xdist>=3; extra == "test"
|
|
29
|
+
Requires-Dist: scipy>=1.10; extra == "test"
|
|
30
|
+
Dynamic: license-file
|
|
31
|
+
|
|
32
|
+
# rafkit
|
|
33
|
+
|
|
34
|
+
[](https://github.com/JimGalasyn/rafkit/actions/workflows/ci.yml)
|
|
35
|
+
[](https://codecov.io/gh/JimGalasyn/rafkit)
|
|
36
|
+
[](https://pypi.org/project/rafkit/)
|
|
37
|
+
[](https://pypi.org/project/rafkit/)
|
|
38
|
+
[](LICENSE)
|
|
39
|
+
[](https://doi.org/10.5281/zenodo.21954795)
|
|
40
|
+
|
|
41
|
+
Autocatalytic (RAF) sets in catalytic reaction networks — maximal RAFs, irreducible
|
|
42
|
+
cores, Kauffman binary polymer models, and interoperability with
|
|
43
|
+
[CatReNet](https://github.com/husonlab/catrenet).
|
|
44
|
+
|
|
45
|
+
Pure Python and NumPy. No Java, no GUI, no install beyond `pip`.
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
from rafkit import binary_polymer, max_raf, sample_irrraf
|
|
49
|
+
import numpy as np
|
|
50
|
+
|
|
51
|
+
net = binary_polymer(max_len=8, food_len=2, p=1.5e-3, cleavage=True)
|
|
52
|
+
raf = max_raf(net)
|
|
53
|
+
print(raf.size, "reactions in the maximal RAF")
|
|
54
|
+
|
|
55
|
+
core = sample_irrraf(net, raf.reactions, np.random.default_rng(0))
|
|
56
|
+
print(len(core), "reactions in one irreducible core")
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## Why this exists
|
|
60
|
+
|
|
61
|
+
RAF theory (Hordijk & Steel 2004) formalises collectively autocatalytic sets: a set of
|
|
62
|
+
reactions is a RAF over a food set when every reaction is catalysed by something the
|
|
63
|
+
set can make, and every reactant can be built up from food using the set. The
|
|
64
|
+
reference implementation, **CatReNet**, is an excellent Java/JavaFX desktop
|
|
65
|
+
application. This is a small library for people who want the same algorithms inside a
|
|
66
|
+
Python analysis pipeline.
|
|
67
|
+
|
|
68
|
+
## Validated against the reference implementation
|
|
69
|
+
|
|
70
|
+
`tests/data/catrenet_polymer_n6.crs` was generated by CatReNet's own `polymer-tool`,
|
|
71
|
+
and the expected counts in `tests/test_crs.py` are what CatReNet's `catrenet-tool`
|
|
72
|
+
reports on it. The test suite therefore checks this implementation against an
|
|
73
|
+
independent one on every run, with no Java required.
|
|
74
|
+
|
|
75
|
+
| algorithm | rafkit | CatReNet 1.1.0 |
|
|
76
|
+
|---|---|---|
|
|
77
|
+
| `max_raf` | 183 | 183 |
|
|
78
|
+
| `catrenet_strictly_autocatalytic` | 175 | 175 |
|
|
79
|
+
| `max_raf_strict` | 161 | *(different object — see below)* |
|
|
80
|
+
|
|
81
|
+
**A documented divergence.** CatReNet's `strictlyAutocatalyticMaxRaf` *filters* the
|
|
82
|
+
maximal RAF for reactions having a non-food catalyst, without re-refining, so its
|
|
83
|
+
result need not itself be a RAF. `max_raf_strict` imposes the same condition inside
|
|
84
|
+
the fixpoint, so its result is a RAF, and is correspondingly smaller. Both are
|
|
85
|
+
available; they answer different questions. CatReNet's behaviour was reproduced by
|
|
86
|
+
black-box inference from its output — no CatReNet source was read or used.
|
|
87
|
+
|
|
88
|
+
## Calibration
|
|
89
|
+
|
|
90
|
+
The RAF phase transition in Kauffman's binary polymer model, measured here against
|
|
91
|
+
the published value of *f* ≈ 1.20 (Steel, Hordijk & Smith 2012, n=10, t=2), where
|
|
92
|
+
*f* = p|R| is the mean number of catalysed reactions per molecule:
|
|
93
|
+
|
|
94
|
+
| model | transition |
|
|
95
|
+
|---|---|
|
|
96
|
+
| ligation only | *f* ≈ 4.7 |
|
|
97
|
+
| ligation + cleavage | *f* ≈ 3.1 |
|
|
98
|
+
| ligation + cleavage, **catalysis paired per reversible reaction** | 0 seeds at *f* ≤ 1.22, all seeds by *f* ≈ 1.59 |
|
|
99
|
+
|
|
100
|
+
Two conventions have to match before any comparison to the literature means anything:
|
|
101
|
+
the model must include **cleavage**, and a reversible cleavage–ligation pair must be
|
|
102
|
+
counted as **one** catalysed reaction, not two. Use `net.catalysis_level` — not
|
|
103
|
+
`mean_catalysed_per_molecule` — whenever a number is placed beside a published *f*.
|
|
104
|
+
|
|
105
|
+
## What's implemented
|
|
106
|
+
|
|
107
|
+
| | |
|
|
108
|
+
|---|---|
|
|
109
|
+
| `max_raf` | maximal RAF, by fixpoint (Hordijk & Steel 2004) |
|
|
110
|
+
| `max_raf_strict` | maximal RAF whose catalysts must be non-food products |
|
|
111
|
+
| `catrenet_strictly_autocatalytic` | CatReNet's similarly-named filter, for interop |
|
|
112
|
+
| `sample_irrraf` | one irreducible RAF, by randomised shrinking (Steel, Hordijk & Smith 2012) |
|
|
113
|
+
| `irrraf_census` | how many *distinct* irreducible cores a network carries |
|
|
114
|
+
| `exploitability` | share of RAF products contributing no catalysis back |
|
|
115
|
+
| `is_food_catalysed` | whether a core runs on food catalysis alone, and so carries no heredity |
|
|
116
|
+
| `core_raf` / `has_unique_irraf` | Huson, Xavier & Steel's polynomial test for a *unique* irreducible RAF |
|
|
117
|
+
| `catalytically_reachable` | what can be made without any spontaneous reaction |
|
|
118
|
+
| `binary_polymer` | Kauffman binary polymer generator, with optional cleavage |
|
|
119
|
+
| `ReactionNetwork` | arbitrary catalytic reaction systems, same protocol |
|
|
120
|
+
| `read_crs` / `write_crs` | CatReNet's CRS interchange format |
|
|
121
|
+
| `to_pnml` / `write_pnml` | PNML export (ISO/IEC 15909-2) for the Petri net ecosystem |
|
|
122
|
+
| `simulate` | Gillespie direct method — watch subRAFs seed themselves into existence |
|
|
123
|
+
| `max_urafs` | uninhibited RAFs, when a molecule can prevent a reaction |
|
|
124
|
+
| `run_serial_dilution` / `run_cstr` | dilution protocols for growing–dividing compartments — **not a RAF algorithm**, see below |
|
|
125
|
+
| `permeation_flux` / `permeable_by_length` | size-selective transport across a compartment membrane — **not a RAF algorithm**, see below |
|
|
126
|
+
| `is_thermodynamically_consistent` | can these reactions all run forward at once? — **not a RAF algorithm**, see below |
|
|
127
|
+
| `BondEnergies` / `rate_constants` | free energy of a polymer chemistry, and the rate constants it forces — **not a RAF algorithm**, see below |
|
|
128
|
+
| `detailed_balance_residual` | whether a set of rate constants is consistent with the free energies |
|
|
129
|
+
| `elongation_ratio` / `mean_length` / `sequence_correlation_length` | the equilibrium ensemble, closed form |
|
|
130
|
+
| `Kinetics` / `kinetics_from_energies` | rate constants a simulator can run on — `simulate(..., kinetics=...)` |
|
|
131
|
+
| `unpaired_catalysis` | reversible pairs whose two directions have different catalysts, which is impossible |
|
|
132
|
+
|
|
133
|
+
Every algorithm carries hand-computed known-answer tests, because a RAF algorithm that
|
|
134
|
+
is subtly wrong produces plausible numbers rather than errors.
|
|
135
|
+
|
|
136
|
+
## Four modules are deliberately off-theme: `dilution`, `permeation`, `thermo` and `autocatalysis`
|
|
137
|
+
|
|
138
|
+
Everything above takes a `ReactionNetwork` and asks a RAF question of it. `rafkit.dilution`
|
|
139
|
+
takes no network at all — it is a **two-species ordinary differential equation with no RAF
|
|
140
|
+
structure**, reproducing the minimal model of Matsubara, Ameta, Thutupalli, Nghe & Krishna
|
|
141
|
+
([arXiv:2211.03155](https://arxiv.org/abs/2211.03155)).
|
|
142
|
+
|
|
143
|
+
It earns its place for one reason: **it is this library's only *analytic* calibration.**
|
|
144
|
+
Every other check here is against a reference implementation (CatReNet) or a published
|
|
145
|
+
figure (Steel, Hordijk & Smith) — matched to a count, or to a shape. Matsubara et al. derive
|
|
146
|
+
closed-form conditions, which can be hit or missed to nine significant figures:
|
|
147
|
+
|
|
148
|
+
| their claim | status |
|
|
149
|
+
|---|---|
|
|
150
|
+
| `r(x)x` linear ⇒ only the symmetric trajectory is stable, **no bistability** | reproduced |
|
|
151
|
+
| `r(x)x = ε + κx²` ⇒ **bistability** at their `Δt`=1, `κ`=8, `ε`=0.5, `φ`=1 | reproduced |
|
|
152
|
+
| bistability **lost above a critical cycle interval** | reproduced |
|
|
153
|
+
| their equation (2), **parameter-free** | reproduced to 2×10⁻⁹ (asserted at 1e-8) |
|
|
154
|
+
|
|
155
|
+
The two parameterisations also agree on the *sign*: their equation (3) makes the
|
|
156
|
+
amplification factor crossing 1 the sufficient condition for bistability, and it measures
|
|
157
|
+
0.900 for the linear flux against 1.243 for the quadratic — the numerical test and the
|
|
158
|
+
analytic criterion picking out the same case from independent computations.
|
|
159
|
+
|
|
160
|
+
The wider use is that RAF work increasingly runs networks inside growing, dividing
|
|
161
|
+
compartments, where the dilution protocol is a modelling choice that changes the answer.
|
|
162
|
+
This gives that choice a validated implementation and a benchmark, independent of any RAF
|
|
163
|
+
structure. If you only want RAF algorithms, ignore this module; nothing else imports it.
|
|
164
|
+
|
|
165
|
+
### `permeation` — one line of transport physics, and a trap worth a module
|
|
166
|
+
|
|
167
|
+
`rafkit.permeation` is smaller and earns its place differently. RAF work increasingly runs
|
|
168
|
+
networks inside compartments embedded in a shared medium, and every such model needs a rule
|
|
169
|
+
for what crosses the boundary. The rule implemented is Hordijk, Naylor, Krasnogor &
|
|
170
|
+
Fellermann's ([*Life* **8**(3), 33, 2018](https://doi.org/10.3390/life8030033)) — *"molecules
|
|
171
|
+
are allowed to permeate compartment membranes if their lengths do not exceed a certain
|
|
172
|
+
threshold. Permeation is proportional to the concentration difference."*
|
|
173
|
+
|
|
174
|
+
**"Proportional to the concentration difference" is not "proportional to the count
|
|
175
|
+
difference",** and the two coincide only when compartment and medium have the same volume. In
|
|
176
|
+
a spatial model they generally do not: in the paper above a compartment of radius 0.5 sits in
|
|
177
|
+
a diffusion voxel of 2.5 × 2.5 — and since that world is two-dimensional ("if the world type
|
|
178
|
+
is set to 2D then Y is forced to 1"), the voxel is a slab of unit thickness. **Both are then
|
|
179
|
+
volumes**: a sphere of 0.524 against a slab of 6.25, a ratio of **11.9**. Quoting a voxel
|
|
180
|
+
*area* against a sphere *volume* would be dimensionally meaningless. Writing the flux as
|
|
181
|
+
`P · (n_out − n_in)` silently asserts they are the same size, and produces plausible numbers
|
|
182
|
+
rather than an error — which is the failure mode this whole library is written against.
|
|
183
|
+
|
|
184
|
+
How much it matters: reproducing that paper's own induction experiment with every printed
|
|
185
|
+
parameter taken from [the authors' published input files](http://ico2s.org/data/extras/compartments/),
|
|
186
|
+
the count-difference form **cannot match both published arms at any permeability** — swept, it
|
|
187
|
+
reaches the control value at an effect ratio of 1.15 on one branch or 2.44 on the other,
|
|
188
|
+
bracketing the published 1.60 without hitting it. The concentration form reproduces both arms
|
|
189
|
+
(16.5 ± 5.9 against their 16.3; 27.8 ± 6.0 against their 26.0) with a single free parameter.
|
|
190
|
+
|
|
191
|
+
⚠ Note the calibration tier: that is a published *figure* matched with one fitted parameter, so
|
|
192
|
+
it sits with this library's reference-implementation and figure checks — **not** with
|
|
193
|
+
`dilution`, which remains the only analytic anchor here. The module's own tests are
|
|
194
|
+
deterministic properties (equal concentrations give zero flux; equal *counts* do not; flux
|
|
195
|
+
bounded by what is present), because a library gate should be fast and exact; the stochastic
|
|
196
|
+
reproduction lives in the downstream research client.
|
|
197
|
+
|
|
198
|
+
### `thermo` — the free energy the rest of the library leaves implicit
|
|
199
|
+
|
|
200
|
+
`rafkit.thermo` is the third off-theme module, and it is here because **the rest of this
|
|
201
|
+
library already has a thermodynamics — an unwritten one, and it is the wrong one.**
|
|
202
|
+
|
|
203
|
+
`gillespie` gives every reaction a unit rate constant. In a cleavage–ligation chemistry that
|
|
204
|
+
means `k_f = k_r` for every reaction, so `K_eq = 1` and `ΔG° = 0`: **every polymer is
|
|
205
|
+
isoenergetic with the parts it is made of.** Nothing is more stable than anything else, no
|
|
206
|
+
sequence is preferred over any other, and the polymer growth those runs show is driven
|
|
207
|
+
entirely by the food boundary condition. That is not a badly chosen parameter; it is free
|
|
208
|
+
energy being *absent* while the model talks as though it were present. The check is one call —
|
|
209
|
+
against a bond energy of −1 to −3 with an association cost of 0.5, unit rate constants score a
|
|
210
|
+
`detailed_balance_residual` of 2.5, and the rates this module builds score 2×10⁻¹⁶.
|
|
211
|
+
|
|
212
|
+
**Catalysis has the same problem one level down.** Where the uncatalysed rate is zero,
|
|
213
|
+
"catalysis" is not acceleration but *enablement*: the catalyst decides whether the reaction
|
|
214
|
+
exists at all, so "which species catalyses what" becomes a choice of which reactions there
|
|
215
|
+
are, wearing a kinetic name. A catalyst that enables also moves the equilibrium — from
|
|
216
|
+
unreachable to reachable — and no catalyst does that. Here a catalyst is a ratio applied to
|
|
217
|
+
**both directions at once**, which is the only form leaving `K_eq` alone; the residual for
|
|
218
|
+
enablement is `inf`, not a large number. For scale, 100× is a barrier drop of 11.4 kJ/mol at
|
|
219
|
+
298 K, about one hydrogen bond.
|
|
220
|
+
|
|
221
|
+
Detailed balance is structural rather than imposed: the barrier is split between the two
|
|
222
|
+
directions in the Brønsted way, `ΔG‡_f = barrier + β·ΔG` and `ΔG‡_r = barrier − (1−β)·ΔG`, so
|
|
223
|
+
`k_f/k_r = exp(−ΔG/RT)` identically for **every** barrier and every `β`. There is no parameter
|
|
224
|
+
setting that violates it, and `k_uncat` stops being a switch: fix the barrier and both rates
|
|
225
|
+
follow, neither of them zero.
|
|
226
|
+
|
|
227
|
+
**Why three bond energies and not one.** A uniform bond energy gives every equal-length
|
|
228
|
+
sequence identical free energy, so no thermodynamic sequence preference can exist — in a unary
|
|
229
|
+
alphabet the question cannot arise, in a binary one it is the whole point. But "uniform" is not
|
|
230
|
+
the tight condition. The ensemble these energies induce is exactly a **one-dimensional Ising
|
|
231
|
+
chain**, whose transfer matrix loses its second eigenvalue whenever
|
|
232
|
+
|
|
233
|
+
```
|
|
234
|
+
ε = E₀₀ + E₁₁ − E₀₁ − E₁₀ = 0
|
|
235
|
+
```
|
|
236
|
+
|
|
237
|
+
— the *additive* case `E[a][b] = h(a) + g(b)`, where the bond energy says something about the
|
|
238
|
+
left residue and something about the right one and nothing about the pair. (Under
|
|
239
|
+
`BondEnergies.symmetric`, where `E₀₁ = E₁₀`, that is the familiar `E₀₀ + E₁₁ − 2·E₀₁`. The
|
|
240
|
+
matrix is not required to be symmetric — a directional backbone need not be — and the doubled
|
|
241
|
+
form reports a preference that is not there when it is not.) Every additive assignment has a sequence correlation length
|
|
242
|
+
of exactly zero, uniform or not, so three energies satisfying `E₀₁ = (E₀₀+E₁₁)/2` buy nothing
|
|
243
|
+
over one. **It is the non-additivity `ε` alone that makes ordering thermodynamically visible**:
|
|
244
|
+
`ε > 0` favours alternation, `ε < 0` favours blocks. That is where a thermodynamic basis for
|
|
245
|
+
templating would have to come from, as opposed to an imposed rule.
|
|
246
|
+
|
|
247
|
+
Two exact anchors, both in `tests/test_thermo.py`:
|
|
248
|
+
|
|
249
|
+
| claim | status |
|
|
250
|
+
|---|---|
|
|
251
|
+
| ΔG of a ligation is the **junction bond alone**, and every split of every sequence agrees | exact — this is Wegscheider consistency |
|
|
252
|
+
| equilibrium length distribution is geometric with ratio `ρ = λ_max(T)`, mean `1/(1−ρ)` | exact for an additive assignment, from the first **bond** |
|
|
253
|
+
|
|
254
|
+
⚠ That last row was wrong in its first form, and the correction is worth stating: the geometric
|
|
255
|
+
law starts at the first *bond*, not the first *molecule*. A monomer has no bonds, so the step
|
|
256
|
+
from length 1 to length 2 is a boundary term and need not equal `ρ`. A uniform assignment hides
|
|
257
|
+
this — its first step happens to equal `ρ` — so the null case is the one case where the effect
|
|
258
|
+
is invisible. On an additive-but-not-uniform assignment at `ρ = 0.4` the first step is 0.377 and
|
|
259
|
+
`mean_length` overstates the true number-average by 1.4%.
|
|
260
|
+
|
|
261
|
+
⚠ And a third, found only because the binary case is the easy one: **Perron–Frobenius
|
|
262
|
+
constrains the leading eigenvalue and nothing else.** A positive matrix of size 3 or more may
|
|
263
|
+
have complex subdominant eigenvalues — a randomly drawn 4×4 bond-energy matrix does on the
|
|
264
|
+
first try — so rejecting them as impossible refused an ordinary nucleotide chemistry. They are
|
|
265
|
+
legitimate: a complex pair is a correlation that *oscillates* as it decays, and the decay
|
|
266
|
+
length is set by the modulus either way.
|
|
267
|
+
|
|
268
|
+
**Reaching the simulator.** `kinetics_from_energies` turns bond energies into a `Kinetics` —
|
|
269
|
+
per-reaction *uncatalysed* rate constants plus the factor a **present** catalyst applies — and
|
|
270
|
+
`propensities(..., kinetics=...)` runs on it. The split is deliberate: `thermo` says what the rate
|
|
271
|
+
constants are, and whether a catalyst is present at any instant is state, so the simulator never
|
|
272
|
+
holds a "catalysed rate constant" it could apply to one direction.
|
|
273
|
+
|
|
274
|
+
⚠ **The existing model was already in this form and did not say so.** `gillespie`'s uncatalysed
|
|
275
|
+
factor of 20 *is* `Kinetics.uniform(net.n_reactions, 1/20, 20)`, and reproduces its propensities to the
|
|
276
|
+
last ulp — so catalysis was never the missing piece. What was missing is the equilibrium: unit
|
|
277
|
+
constants make `k_f = k_r` whether or not a catalyst is present, so `K_eq = 1` regardless.
|
|
278
|
+
|
|
279
|
+
The consequence that can change a trajectory: a chemistry built this way has its **stationary point
|
|
280
|
+
at the thermodynamic equilibrium.** Balance is on the *combinatorial factors* —
|
|
281
|
+
`k_f · combos_forward = k_r · combos_reverse` — which for `a + b → ab` with `a ≠ b` is the familiar
|
|
282
|
+
`n_ab/(n_a·n_b) = K`. A present catalyst raises both directions by the enhancement **without moving
|
|
283
|
+
that point**. Under unit constants the balance sits at `K = 1` for every reaction, whatever the
|
|
284
|
+
molecules are.
|
|
285
|
+
|
|
286
|
+
⚠ **Not so for a self-ligation.** `a + a → aa` takes the pair count `n_a(n_a−1)/2`, so it balances at
|
|
287
|
+
`n_aa = K·n_a(n_a−1)/2` and the count ratio `n_aa/n_a²` tends to **`K/2`**. Measured on `0 + 0 → 00`
|
|
288
|
+
under unit constants at `n_a = 20`, balance is at `n_aa = 190`, not 400. The factor is the standard
|
|
289
|
+
stochastic symmetry number and `_pair_count` is correct — but **the map from ΔG to a count ratio is
|
|
290
|
+
not uniform across the chemistry**, and anyone reading equilibrium constants off a trajectory needs
|
|
291
|
+
the qualification.
|
|
292
|
+
|
|
293
|
+
⚠ `binary_polymer(paired_catalysis=False)` is not a variant chemistry — it is a **thermodynamically
|
|
294
|
+
impossible** one. Drawing the two directions' catalysts separately gives molecules that accelerate a
|
|
295
|
+
ligation but not its cleavage, and whenever such a molecule is present the reaction is a free-energy
|
|
296
|
+
source; measured on a hand-built case, 100× net flux from nothing. `unpaired_catalysis` finds them
|
|
297
|
+
and `kinetics_from_energies` refuses them.
|
|
298
|
+
|
|
299
|
+
⚠ A `simulate` run with `kinetics` is still **driven**, not closed: the food floor holds a chemical
|
|
300
|
+
potential at the boundary. It does not relax to the equilibrium ensemble computed above, and
|
|
301
|
+
comparing the two directly would be comparing a driven steady state to an equilibrium.
|
|
302
|
+
|
|
303
|
+
**`dg_assoc` is required, with no default.** The association cost — the standard-state price of
|
|
304
|
+
turning two molecules into one — is the only sequence- and length-independent term in a ligation,
|
|
305
|
+
and `0.0` is not a neutral absence but the claim *"joining is free"*. For scale, Ross & Deamer
|
|
306
|
+
(*Life* **6**(3):28) put phosphodiester formation at **+3.3 kcal/mol at 85 °C (≈ +4.7 RT,
|
|
307
|
+
K₁ ≈ 1e-3)**, so zero is not a small value of this quantity — it is a different claim. It also
|
|
308
|
+
decides whether an equilibrium exists at all: at monomer 0.5 with `E = −1`, `dg_assoc = 0` puts the
|
|
309
|
+
elongation ratio **above 1** (runaway, no equilibrium) and `+4.7` brings it back below.
|
|
310
|
+
|
|
311
|
+
Same rule as `permeation_flux` requiring both volumes: where a value that looks like an absence is
|
|
312
|
+
really an assertion, there is no default that would let it happen silently. ⚠ The single exemption
|
|
313
|
+
is `sequence_correlation_length`, where the argument **provably cancels** — requiring a value that
|
|
314
|
+
cannot change the answer only trains the reflex that makes the requirement worthless everywhere
|
|
315
|
+
else.
|
|
316
|
+
|
|
317
|
+
⚠ And it is where **water activity** enters, since a ligation releases water: mass action for
|
|
318
|
+
`N_m + N_n ⇌ N_{m+n} + H₂O` puts `+RT·ln(a_W)` in exactly this term, negative when dry. So the old
|
|
319
|
+
default silently asserted `a_W = 1` — permanently wet. **Not implemented**: that is the form, not a
|
|
320
|
+
number.
|
|
321
|
+
|
|
322
|
+
⚠ Calibration tier: **algebraic**, not empirical. Everything above is an identity that holds or
|
|
323
|
+
does not, so it sits beside `dilution` rather than beside the figure reproductions — but it
|
|
324
|
+
reproduces no experiment and calibrates against no published number. It says the model is
|
|
325
|
+
*consistent*, not that it is *right*.
|
|
326
|
+
|
|
327
|
+
### `autocatalysis` — the one place this library can be told it is wrong by someone else
|
|
328
|
+
|
|
329
|
+
`rafkit.autocatalysis` decides a question RAF theory cannot pose: **given that these reactions must
|
|
330
|
+
run in these directions, does any assignment of chemical potentials make that happen?** That is the
|
|
331
|
+
CAC question of Kosc, Kuperberg, Rajon & Charlat, [*PNAS* **122**(18) e2421274122
|
|
332
|
+
(2025)](https://doi.org/10.1073/pnas.2421274122).
|
|
333
|
+
|
|
334
|
+
**The whole module is one reduction.** With `x = e^μ` and barrier factor `b_i = e^{−G‡_i}`, local
|
|
335
|
+
detailed balance gives `v_i = b_i(∏x^{S⁻} − ∏x^{S⁺})`. Since `b_i > 0`, **the barrier scales the flow
|
|
336
|
+
but cannot flip its sign**, so with `y = ln x = μ` reaction `i` runs forward exactly when
|
|
337
|
+
`(Sᵀy)_i < 0`. The question becomes **strict linear feasibility of `Sᵀ y < 0`** — a linear program in
|
|
338
|
+
the chemical potentials, and nothing else.
|
|
339
|
+
|
|
340
|
+
⚠⚠ **The verdict therefore depends on no rate constant and no barrier**, which the paper states
|
|
341
|
+
outright. The API accepts none, and a test asserts that it accepts none.
|
|
342
|
+
|
|
343
|
+
**Two independent methods, required to agree.** Gordan's theorem says exactly one of `Sᵀy < 0` (a
|
|
344
|
+
witness) and `S w = 0, w ≥ 0, w ≠ 0` (a certificate) can hold. Both are computed and disagreement
|
|
345
|
+
**raises** rather than picking a winner.
|
|
346
|
+
|
|
347
|
+
**Checked against the published answers** — Kosc's Fig. 4, two cores sharing `{R2, R3}`:
|
|
348
|
+
|
|
349
|
+
| | verdict | source |
|
|
350
|
+
|---|---|---|
|
|
351
|
+
| `{R1,R2,R3,R4}` alone | consistent | **Theorem 2** — a single PAC always is |
|
|
352
|
+
| `{R1',R2,R3,R4'}` alone | consistent | **Theorem 2** |
|
|
353
|
+
| both together | **inconsistent** | **Box 2** — a multiPAC that is not a multiCAC |
|
|
354
|
+
|
|
355
|
+
⚠ That network is *reconstructed* from Box 2's flow equations (the paper draws it as a figure) and
|
|
356
|
+
cross-checked against the figure's composition glyphs: every reaction mass-balances, both cores are
|
|
357
|
+
autocatalytic in `e4`, and they share exactly two reactions as the caption says.
|
|
358
|
+
|
|
359
|
+
**And the certificate explains itself.** It comes back with every weight equal — the six reactions at
|
|
360
|
+
unit flux return the system to its starting composition. Then `Σ w_i A_i = −(S w)·y = 0`, so the
|
|
361
|
+
affinities cannot all be positive: **a cycle that returns to its starting composition cannot be
|
|
362
|
+
downhill all the way round.** The second law, as a linear-algebra identity.
|
|
363
|
+
|
|
364
|
+
**Why it matters more than a bigger test suite.** Theorem 2 is an *external* guarantee — a single PAC
|
|
365
|
+
is always consistent — so **every network this library can generate is a pass/fail case adjudicated
|
|
366
|
+
by someone else.** It is the one place rafkit can be shown wrong without anyone here noticing first.
|
|
367
|
+
|
|
368
|
+
⚠ Needs `scipy` for the LP: `pip install rafkit[cac]`. Imported lazily, so `import rafkit` and every
|
|
369
|
+
other module stay numpy-only.
|
|
370
|
+
|
|
371
|
+
## A reaction network is a Petri net
|
|
372
|
+
|
|
373
|
+
Species are places, reactions are transitions, molecule counts are tokens. `write_pnml`
|
|
374
|
+
exports to PNML (ISO/IEC 15909-2), so these networks open in Petri net editors, model
|
|
375
|
+
checkers, and the unfolding tools that compute the causal structure of a run.
|
|
376
|
+
|
|
377
|
+
Three things need care, and each is explicit rather than silent:
|
|
378
|
+
|
|
379
|
+
- **Catalysis becomes a self-loop.** P/T nets have no read arc, so a catalyst is a pair
|
|
380
|
+
of arcs, consuming the token and returning it.
|
|
381
|
+
- **Alternative catalyst sets become separate transitions**, named `r1`, `r1#2`, …, since
|
|
382
|
+
a transition's preset is a conjunction and cannot express "either set".
|
|
383
|
+
- **Food gets source transitions**, because RAF food is inexhaustible and no initial
|
|
384
|
+
marking expresses that — a marking of *n* deadlocks after *n* uses.
|
|
385
|
+
|
|
386
|
+
Inhibition has no `ptnet` representation and is written as a `toolspecific` annotation,
|
|
387
|
+
with a warning in the file: a reader that ignores it gets a *different system*.
|
|
388
|
+
Reactions requiring a catalyst that nothing provides are omitted and counted, since
|
|
389
|
+
emitting them unconstrained would make them freely fireable — the opposite of the intent.
|
|
390
|
+
|
|
391
|
+
## Catalysis is a relation, not a list
|
|
392
|
+
|
|
393
|
+
`catalysts[r]` is a set of **alternative catalyst sets**, following Huson, Xavier &
|
|
394
|
+
Steel (2024). Any one set being fully present suffices, and each set is a conjunctive
|
|
395
|
+
requirement:
|
|
396
|
+
|
|
397
|
+
| `catalysts[r]` | meaning |
|
|
398
|
+
|---|---|
|
|
399
|
+
| `{{a}, {b}}` | *a* **or** *b* — the simple case, and what a flat list of catalysts meant |
|
|
400
|
+
| `{{a, d}, {e}}` | (*a* **and** *d*) **or** *e* |
|
|
401
|
+
| `{}` | **must** be catalysed, and nothing does: never in a RAF |
|
|
402
|
+
| `{frozenset()}` | **may proceed uncatalysed**; always satisfied |
|
|
403
|
+
|
|
404
|
+
The last two rows are a real distinction rather than a technicality — in the §2.4 system
|
|
405
|
+
of that paper it decides which reactions can join an RAF — and a flat list collapses
|
|
406
|
+
them. In CRS, a braced group is conjunctive: `[{a,d}, e]`, with `[]` and `[{}]` for the
|
|
407
|
+
last two rows.
|
|
408
|
+
|
|
409
|
+
Constructors still accept a plain iterable of molecules and normalise it, so simple
|
|
410
|
+
systems stay simple to write.
|
|
411
|
+
|
|
412
|
+
## Inhibition
|
|
413
|
+
|
|
414
|
+
A molecule can prevent a reaction. `max_urafs` returns the **uninhibited RAFs** of
|
|
415
|
+
Hordijk & Steel (2012), and returns a *collection* rather than one set, because
|
|
416
|
+
inhibition destroys the monotonicity that makes a maximal RAF unique — there is no
|
|
417
|
+
"the" maximal u-RAF.
|
|
418
|
+
|
|
419
|
+
```
|
|
420
|
+
Food: a, b
|
|
421
|
+
r1 : a + b [a] {d} => c # inhibited by d
|
|
422
|
+
r2 : a + b [b] {c} => d # inhibited by c
|
|
423
|
+
```
|
|
424
|
+
|
|
425
|
+
Two maximal u-RAFs, `{r1}` and `{r2}`: each is an RAF whose support avoids its own
|
|
426
|
+
inhibitor, and their union is an RAF that fails the uninhibited condition.
|
|
427
|
+
|
|
428
|
+
`simulate` respects inhibition too — an inhibited reaction has propensity zero, so a
|
|
429
|
+
running network can **lose** a subRAF, not merely gain one. See
|
|
430
|
+
`examples/inhibition_dissolution.py`.
|
|
431
|
+
|
|
432
|
+
The set-theoretic tools need no special handling: `sample_irrraf`, `irrraf_census`,
|
|
433
|
+
`core_raf` and `catalytically_reachable` all take a reaction set, and passing a u-RAF
|
|
434
|
+
is correct because the uninhibited property is inherited downward — every sub-RAF of a
|
|
435
|
+
u-RAF is a u-RAF.
|
|
436
|
+
|
|
437
|
+
Deciding whether a u-RAF exists is NP-complete, but the problem is fixed-parameter
|
|
438
|
+
tractable in *k*, the number of inhibition classes — and **k is a property of how you
|
|
439
|
+
encode inhibition, not of the chemistry.** `classes_from_inhibitors` groups by
|
|
440
|
+
inhibiting *molecule*, so *k* is the number of distinct inhibitors rather than the
|
|
441
|
+
number of inhibited reactions, which is the difference between `2^k` being feasible
|
|
442
|
+
and not.
|
|
443
|
+
|
|
444
|
+
## Notes on irreducible cores
|
|
445
|
+
|
|
446
|
+
There may be **exponentially many** irreducible RAFs inside one maximal RAF, and
|
|
447
|
+
finding the smallest is NP-hard (Steel, Hordijk & Smith 2012). `sample_irrraf` returns
|
|
448
|
+
*one*, chosen by the random order it walks; `irrraf_census` samples repeatedly and
|
|
449
|
+
reports how many distinct ones it saw. That count is always a **lower bound**, never
|
|
450
|
+
an upper one.
|
|
451
|
+
|
|
452
|
+
`is_food_catalysed` exists because a core whose every reaction has a food catalyst
|
|
453
|
+
satisfies the letter of the RAF definition while being in no sense self-referential —
|
|
454
|
+
it runs wherever the food runs. Split those out before reading a count of cores as a
|
|
455
|
+
count of anything biological.
|
|
456
|
+
|
|
457
|
+
## Install
|
|
458
|
+
|
|
459
|
+
```bash
|
|
460
|
+
pip install rafkit
|
|
461
|
+
```
|
|
462
|
+
|
|
463
|
+
Development:
|
|
464
|
+
|
|
465
|
+
```bash
|
|
466
|
+
pip install -e ".[test]"
|
|
467
|
+
pytest -q
|
|
468
|
+
```
|
|
469
|
+
|
|
470
|
+
Releases are documented in [CHANGELOG.md](CHANGELOG.md); the release procedure is
|
|
471
|
+
[docs/RELEASING.md](docs/RELEASING.md).
|
|
472
|
+
|
|
473
|
+
## Citing
|
|
474
|
+
|
|
475
|
+
Cite the concept DOI [10.5281/zenodo.21954795](https://doi.org/10.5281/zenodo.21954795),
|
|
476
|
+
which always resolves to the latest version; `CITATION.cff` also lists the per-version
|
|
477
|
+
DOI. If you use the CatReNet interoperability or the validation fixture, please cite
|
|
478
|
+
CatReNet too.
|
|
479
|
+
|
|
480
|
+
## References
|
|
481
|
+
|
|
482
|
+
- Hordijk & Steel, "Detecting autocatalytic, self-sustaining sets in chemical reaction
|
|
483
|
+
systems," *J. Theor. Biol.* 227, 451 (2004).
|
|
484
|
+
- Steel, Hordijk & Smith, "Minimal autocatalytic networks," *J. Theor. Biol.* 332, 96
|
|
485
|
+
(2013); arXiv:1212.4450.
|
|
486
|
+
- Hordijk & Steel, "Autocatalytic sets extended: dynamics, inhibition, and a
|
|
487
|
+
generalization," *J. Syst. Chem.* 3, 5 (2012); arXiv:1206.1017.
|
|
488
|
+
- Huson, Xavier & Steel, "CatReNet: interactive analysis of (auto-)catalytic reaction
|
|
489
|
+
networks," *Bioinformatics* 40(8), btae515 (2024).
|
|
490
|
+
- Serra & Villani, "Template-Based Catalysis and the Emergence of Collectively
|
|
491
|
+
Autocatalytic Systems," *Entropy* 28(2), 184 (2026).
|
|
492
|
+
- Matsubara, Ameta, Thutupalli, Nghe & Krishna, "Conditions for Darwinian evolution in
|
|
493
|
+
compartmentalized autocatalytic reaction networks," arXiv:2211.03155 — the analytic
|
|
494
|
+
benchmark behind `rafkit.dilution`.
|
|
495
|
+
|
|
496
|
+
## License
|
|
497
|
+
|
|
498
|
+
MIT. CatReNet is GPL v3 and is **not** a dependency — this library interoperates with
|
|
499
|
+
it only through files, and contains no code derived from it.
|