@dsh-bio/dsh-bio-gem 0.1.3 → 0.1.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +60 -8
- package/docs/ARCHITECTURE.md +145 -116
- package/docs/DECISIONS-2026-09-21.md +76 -0
- package/docs/releases/v0.1.12.md +43 -0
- package/package.json +5 -3
- package/python/benchmark.py +3 -3
- package/python/biomass_tools.py +2 -2
- package/python/bootstrap_carveme.py +259 -0
- package/python/build.py +10 -2
- package/python/coherence.py +159 -0
- package/python/double_knockout.py +1 -1
- package/python/essential_scan.py +1 -1
- package/python/gapfind.py +413 -396
- package/python/gem_ops.py +64 -1
- package/python/l3_fix.py +4 -4
- package/python/ledger.py +1 -1
- package/python/model_card.py +3 -2
- package/python/precursor_scan.py +127 -0
- package/python/quality.py +577 -0
- package/python/sampling.py +269 -0
- package/python/sensitivity.py +2 -2
- package/python/validate.py +435 -409
- package/skills/gem-expert.md +3 -1
- package/src/capabilities.js +152 -0
- package/src/index.js +21 -2
- package/src/integration.js +507 -0
- package/src/jobs.js +5 -21
- package/src/python.js +113 -17
- package/src/tools.js +98 -22
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
"""Flux-space sampling primitives for the ``sample`` gem operation."""
|
|
2
|
+
import contextlib
|
|
3
|
+
import hashlib
|
|
4
|
+
import io
|
|
5
|
+
import os
|
|
6
|
+
import time
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
from cobra.sampling import ACHRSampler
|
|
10
|
+
from cobra.util.solver import linear_reaction_coefficients
|
|
11
|
+
|
|
12
|
+
from gapfind import expand_medium, resolve_medium
|
|
13
|
+
from silentio import silent_read_sbml
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
EX_PREFIXES = ("EX_", "DM_", "SK_")
|
|
17
|
+
FLUX_TOLERANCE = 1e-9
|
|
18
|
+
MIN_SAMPLES = 10
|
|
19
|
+
MAX_SAMPLES = 20000
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _model_hash(path):
|
|
23
|
+
digest = hashlib.sha256()
|
|
24
|
+
with open(path, "rb") as handle:
|
|
25
|
+
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
|
|
26
|
+
digest.update(chunk)
|
|
27
|
+
return digest.hexdigest()[:16]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _is_boundary(reaction):
|
|
31
|
+
return reaction.boundary or reaction.id.startswith(EX_PREFIXES)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _configure_medium(model, medium):
|
|
35
|
+
"""Apply only an explicitly supplied medium; preserve SBML defaults otherwise."""
|
|
36
|
+
if medium is not None and not isinstance(medium, dict):
|
|
37
|
+
raise ValueError("medium must be an object when provided")
|
|
38
|
+
|
|
39
|
+
expanded, preset = expand_medium(medium)
|
|
40
|
+
resolved, unresolved = resolve_medium(model, expanded) if expanded else ({}, [])
|
|
41
|
+
if medium is not None:
|
|
42
|
+
for reaction in model.reactions:
|
|
43
|
+
if _is_boundary(reaction):
|
|
44
|
+
reaction.lower_bound = 0.0
|
|
45
|
+
for reaction_id, lower_bound in resolved.items():
|
|
46
|
+
model.reactions.get_by_id(reaction_id).lower_bound = lower_bound
|
|
47
|
+
return {"preset": preset, "resolved_exchanges": len(resolved)}, unresolved
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _require_int(name, value, minimum=None, maximum=None):
|
|
51
|
+
if isinstance(value, bool) or not isinstance(value, int):
|
|
52
|
+
raise ValueError(f"{name} must be an integer")
|
|
53
|
+
if minimum is not None and value < minimum:
|
|
54
|
+
raise ValueError(f"{name} must be >= {minimum}")
|
|
55
|
+
if maximum is not None and value > maximum:
|
|
56
|
+
raise ValueError(f"{name} must be <= {maximum}")
|
|
57
|
+
return value
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _growth_reaction(model):
|
|
61
|
+
"""Select the strongest positive linear objective reaction deterministically."""
|
|
62
|
+
coefficients = linear_reaction_coefficients(model)
|
|
63
|
+
if not coefficients:
|
|
64
|
+
raise ValueError("model needs a linear objective to identify the growth reaction")
|
|
65
|
+
|
|
66
|
+
candidates = [(reaction, float(coefficient)) for reaction, coefficient in coefficients.items()]
|
|
67
|
+
positive = [(reaction, coefficient) for reaction, coefficient in candidates if coefficient > 0]
|
|
68
|
+
pool = positive or candidates
|
|
69
|
+
reaction, _coefficient = sorted(pool, key=lambda item: (-abs(item[1]), item[0].id))[0]
|
|
70
|
+
return reaction
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _describe(values):
|
|
74
|
+
values = np.asarray(values, dtype=float)
|
|
75
|
+
return {
|
|
76
|
+
"median": float(np.quantile(values, 0.5)),
|
|
77
|
+
"mean": float(np.mean(values)),
|
|
78
|
+
"min": float(np.min(values)),
|
|
79
|
+
"max": float(np.max(values)),
|
|
80
|
+
"q05": float(np.quantile(values, 0.05)),
|
|
81
|
+
"q25": float(np.quantile(values, 0.25)),
|
|
82
|
+
"q75": float(np.quantile(values, 0.75)),
|
|
83
|
+
"q95": float(np.quantile(values, 0.95)),
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _reaction_stats(values):
|
|
88
|
+
values = np.asarray(values, dtype=float)
|
|
89
|
+
return {
|
|
90
|
+
"median": float(np.quantile(values, 0.5)),
|
|
91
|
+
"q05": float(np.quantile(values, 0.05)),
|
|
92
|
+
"q95": float(np.quantile(values, 0.95)),
|
|
93
|
+
"sign_probability": float(np.mean(values > FLUX_TOLERANCE)),
|
|
94
|
+
"near_zero_fraction": float(np.mean(np.abs(values) <= FLUX_TOLERANCE)),
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _selected_reactions(samples, growth_reaction_id, reactions):
|
|
99
|
+
if reactions is not None:
|
|
100
|
+
if not isinstance(reactions, list) or any(not isinstance(item, str) for item in reactions):
|
|
101
|
+
raise ValueError("reactions must be an array of reaction IDs")
|
|
102
|
+
selected = []
|
|
103
|
+
for reaction_id in reactions:
|
|
104
|
+
if reaction_id not in samples.columns:
|
|
105
|
+
raise ValueError(f"reaction not found in model: {reaction_id}")
|
|
106
|
+
if reaction_id not in selected:
|
|
107
|
+
selected.append(reaction_id)
|
|
108
|
+
return selected
|
|
109
|
+
|
|
110
|
+
iqr = samples.quantile(0.75) - samples.quantile(0.25)
|
|
111
|
+
ranked = sorted(
|
|
112
|
+
(reaction_id for reaction_id in samples.columns if reaction_id != growth_reaction_id),
|
|
113
|
+
key=lambda reaction_id: (-float(iqr[reaction_id]), reaction_id),
|
|
114
|
+
)
|
|
115
|
+
return [growth_reaction_id] + ranked[:20]
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _validate_requested_reactions(model, reactions):
|
|
119
|
+
"""Reject malformed target lists before expensive ACHR warmup starts."""
|
|
120
|
+
if reactions is None:
|
|
121
|
+
return
|
|
122
|
+
if not isinstance(reactions, list) or any(not isinstance(item, str) for item in reactions):
|
|
123
|
+
raise ValueError("reactions must be an array of reaction IDs")
|
|
124
|
+
for reaction_id in reactions:
|
|
125
|
+
if reaction_id not in model.reactions:
|
|
126
|
+
raise ValueError(f"reaction not found in model: {reaction_id}")
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _method(method):
|
|
130
|
+
if method is None:
|
|
131
|
+
method = "auto"
|
|
132
|
+
if not isinstance(method, str):
|
|
133
|
+
raise ValueError("method must be one of auto, achr, optgp")
|
|
134
|
+
normalized = method.lower()
|
|
135
|
+
if normalized not in {"auto", "achr", "optgp"}:
|
|
136
|
+
raise ValueError("method must be one of auto, achr, optgp")
|
|
137
|
+
if normalized == "auto":
|
|
138
|
+
return "achr"
|
|
139
|
+
if normalized == "optgp" and os.name == "nt":
|
|
140
|
+
# OptGP creates a worker pool. gem_ops is deliberately import-safe,
|
|
141
|
+
# but this operation must never expose Windows callers to an accidental
|
|
142
|
+
# spawn loop until that execution route has its own verified harness.
|
|
143
|
+
raise ValueError("method 'optgp' is not supported on Windows; use 'achr' or 'auto'")
|
|
144
|
+
return normalized
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def sample_fluxes(
|
|
148
|
+
model_path,
|
|
149
|
+
medium=None,
|
|
150
|
+
n=1000,
|
|
151
|
+
method="auto",
|
|
152
|
+
thinning=100,
|
|
153
|
+
growth_floor_fraction=None,
|
|
154
|
+
reactions=None,
|
|
155
|
+
seed=42,
|
|
156
|
+
export_csv=None,
|
|
157
|
+
):
|
|
158
|
+
"""Sample a model's feasible flux space with a deterministic ACHR default."""
|
|
159
|
+
if not model_path or not os.path.isfile(model_path):
|
|
160
|
+
raise ValueError(f"model file not found: {model_path}")
|
|
161
|
+
n = _require_int("n", n, MIN_SAMPLES, MAX_SAMPLES)
|
|
162
|
+
thinning = _require_int("thinning", thinning, 1)
|
|
163
|
+
seed = _require_int("seed", seed)
|
|
164
|
+
if export_csv is not None and not isinstance(export_csv, str):
|
|
165
|
+
raise ValueError("export_csv must be a path string when provided")
|
|
166
|
+
if growth_floor_fraction is not None:
|
|
167
|
+
if isinstance(growth_floor_fraction, bool) or not isinstance(growth_floor_fraction, (int, float)):
|
|
168
|
+
raise ValueError("growth_floor_fraction must be a number strictly between 0 and 1")
|
|
169
|
+
growth_floor_fraction = float(growth_floor_fraction)
|
|
170
|
+
if not 0.0 < growth_floor_fraction < 1.0:
|
|
171
|
+
raise ValueError("growth_floor_fraction must be strictly between 0 and 1")
|
|
172
|
+
|
|
173
|
+
method_used = _method(method)
|
|
174
|
+
started = time.perf_counter()
|
|
175
|
+
base_model = silent_read_sbml(model_path)
|
|
176
|
+
configured_model = base_model.copy()
|
|
177
|
+
medium_summary, unresolved_medium = _configure_medium(configured_model, medium)
|
|
178
|
+
|
|
179
|
+
growth_reaction = _growth_reaction(configured_model)
|
|
180
|
+
fba_solution = configured_model.optimize()
|
|
181
|
+
if fba_solution.status != "optimal":
|
|
182
|
+
raise ValueError(f"FBA under the selected medium is not optimal: {fba_solution.status}")
|
|
183
|
+
max_growth = float(fba_solution.fluxes[growth_reaction.id])
|
|
184
|
+
|
|
185
|
+
# Every sampling run gets an independent model copy. The growth-floor
|
|
186
|
+
# branch therefore cannot leak modified bounds into the caller or a retry.
|
|
187
|
+
sampling_model = configured_model.copy()
|
|
188
|
+
if growth_floor_fraction is not None:
|
|
189
|
+
if max_growth <= FLUX_TOLERANCE:
|
|
190
|
+
raise ValueError("cannot apply growth_floor_fraction because maximum growth is not positive")
|
|
191
|
+
sampling_growth = sampling_model.reactions.get_by_id(growth_reaction.id)
|
|
192
|
+
sampling_growth.lower_bound = growth_floor_fraction * max_growth
|
|
193
|
+
|
|
194
|
+
_validate_requested_reactions(sampling_model, reactions)
|
|
195
|
+
|
|
196
|
+
# The only fully supported execution route for the Windows target is ACHR.
|
|
197
|
+
# A non-Windows future caller can still ask for OptGP explicitly.
|
|
198
|
+
# Sampling libraries can emit backend diagnostics. Keep them off stdout so
|
|
199
|
+
# gem_ops retains its one-JSON-object protocol.
|
|
200
|
+
with contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr(io.StringIO()):
|
|
201
|
+
if method_used == "achr":
|
|
202
|
+
sampler = ACHRSampler(sampling_model, thinning=thinning, seed=seed)
|
|
203
|
+
else:
|
|
204
|
+
from cobra.sampling import OptGPSampler
|
|
205
|
+
|
|
206
|
+
sampler = OptGPSampler(sampling_model, thinning=thinning, processes=1, seed=seed)
|
|
207
|
+
samples = sampler.sample(n, fluxes=True)
|
|
208
|
+
runtime_s = time.perf_counter() - started
|
|
209
|
+
if len(samples) != n:
|
|
210
|
+
raise RuntimeError(f"sampler returned {len(samples)} rows for n={n}")
|
|
211
|
+
|
|
212
|
+
validation_codes = sampler.validate(samples.to_numpy())
|
|
213
|
+
n_valid = int(np.sum(validation_codes == "v"))
|
|
214
|
+
selected = _selected_reactions(samples, growth_reaction.id, reactions)
|
|
215
|
+
reaction_summary = {
|
|
216
|
+
reaction_id: _reaction_stats(samples[reaction_id].to_numpy())
|
|
217
|
+
for reaction_id in selected
|
|
218
|
+
}
|
|
219
|
+
growth_summary = _describe(samples[growth_reaction.id].to_numpy())
|
|
220
|
+
|
|
221
|
+
if export_csv:
|
|
222
|
+
samples.to_csv(export_csv, index=False)
|
|
223
|
+
|
|
224
|
+
if growth_floor_fraction is None:
|
|
225
|
+
boundary_space = "full_feasible_space"
|
|
226
|
+
floor_note = "未施加 growth floor。"
|
|
227
|
+
else:
|
|
228
|
+
boundary_space = "growth_floor_constrained_space"
|
|
229
|
+
floor_note = (
|
|
230
|
+
f"已在独立 model.copy() 上将 {growth_reaction.id} 的下界设为 "
|
|
231
|
+
f"{growth_floor_fraction:.6g} × 当前 FBA 最大生长。"
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
boundary_note = (
|
|
235
|
+
"全空间均匀采样 ≠ 生物学上有意义的活跃状态;关心近最优生长态请传 "
|
|
236
|
+
"growth_floor_fraction(如 0.9)。"
|
|
237
|
+
f"当前介质下 {growth_reaction.id} 的 FBA 最大生长为 {max_growth:.9g}。{floor_note}"
|
|
238
|
+
)
|
|
239
|
+
feasibility_note = (
|
|
240
|
+
"n_valid 为 cobra sampler.validate 返回 'v'(同时满足稳态、上下界)的样本数;"
|
|
241
|
+
"validation_failures 为其余 l/u/e 代码的样本数。"
|
|
242
|
+
)
|
|
243
|
+
if unresolved_medium:
|
|
244
|
+
feasibility_note += " 未解析的介质成分未施加:" + ", ".join(sorted(unresolved_medium)) + "。"
|
|
245
|
+
|
|
246
|
+
return {
|
|
247
|
+
"model": model_path,
|
|
248
|
+
"model_hash": _model_hash(model_path),
|
|
249
|
+
"method_used": method_used,
|
|
250
|
+
"n_requested": n,
|
|
251
|
+
"n_samples": int(len(samples)),
|
|
252
|
+
"thinning": thinning,
|
|
253
|
+
"seed": seed,
|
|
254
|
+
"runtime_s": runtime_s,
|
|
255
|
+
"growth_reaction": growth_reaction.id,
|
|
256
|
+
"growth": growth_summary,
|
|
257
|
+
"reactions": reaction_summary,
|
|
258
|
+
"feasibility": {
|
|
259
|
+
"n_valid": n_valid,
|
|
260
|
+
"validation_failures": int(len(samples) - n_valid),
|
|
261
|
+
"note": feasibility_note,
|
|
262
|
+
},
|
|
263
|
+
"boundary": {
|
|
264
|
+
"space": boundary_space,
|
|
265
|
+
"growth_floor_fraction": growth_floor_fraction,
|
|
266
|
+
"note": boundary_note,
|
|
267
|
+
},
|
|
268
|
+
"medium": medium_summary,
|
|
269
|
+
}
|
package/python/sensitivity.py
CHANGED
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
# 产物(目标汇连接组分),GAM 网格只动 stub(等比缩放 X/GAM_ORIG)。
|
|
8
8
|
# 锚点: 基准组合与 essential_scan 完全同参数 -> 必须精确复现 155;且 155 全部在
|
|
9
9
|
# always_essential ∪ conditionally_essential(基准在网格内故断言必成立)。
|
|
10
|
-
#
|
|
10
|
+
# 生长数值口径: 单点 FBA objective_value = 比生长速率(1/h,biomass 归一化口径);区间制对比请用 gem_fluxscan。
|
|
11
11
|
import os
|
|
12
12
|
import sys
|
|
13
13
|
import csv
|
|
@@ -391,7 +391,7 @@ def sensitivity(model_path, medium=None, biomass_scales=None, gam_grid=None,
|
|
|
391
391
|
"component_sensitivity": {"top_sensitive": top_sensitive, "rows": comp_rows},
|
|
392
392
|
"component_essentiality_drift": drift,
|
|
393
393
|
"card_robustness_written": card_written,
|
|
394
|
-
"units": "
|
|
394
|
+
"units": "1/h",
|
|
395
395
|
"timing_seconds": round(time.time() - t_start, 1),
|
|
396
396
|
}
|
|
397
397
|
try:
|