@dsh-bio/dsh-bio-gem 0.1.3 → 0.1.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +60 -8
- package/docs/ARCHITECTURE.md +145 -116
- package/docs/DECISIONS-2026-09-21.md +76 -0
- package/docs/releases/v0.1.12.md +43 -0
- package/package.json +5 -3
- package/python/benchmark.py +3 -3
- package/python/biomass_tools.py +2 -2
- package/python/bootstrap_carveme.py +259 -0
- package/python/build.py +10 -2
- package/python/coherence.py +159 -0
- package/python/double_knockout.py +1 -1
- package/python/essential_scan.py +1 -1
- package/python/gapfind.py +413 -396
- package/python/gem_ops.py +64 -1
- package/python/l3_fix.py +4 -4
- package/python/ledger.py +1 -1
- package/python/model_card.py +3 -2
- package/python/precursor_scan.py +127 -0
- package/python/quality.py +577 -0
- package/python/sampling.py +269 -0
- package/python/sensitivity.py +2 -2
- package/python/validate.py +435 -409
- package/skills/gem-expert.md +3 -1
- package/src/capabilities.js +152 -0
- package/src/index.js +21 -2
- package/src/integration.js +507 -0
- package/src/jobs.js +5 -21
- package/src/python.js +113 -17
- package/src/tools.js +98 -22
package/python/validate.py
CHANGED
|
@@ -1,410 +1,436 @@
|
|
|
1
|
-
# validate.py — dsh-bio-gem 五道验证关卡(M1)
|
|
2
|
-
# G1 加载统计 / G2 内部反应元素平衡 / G3 生长真实性 / G4 底物表型(条件) / G5 必需基因抽检(条件)
|
|
3
|
-
# 规格: docs/ARCHITECTURE.md §5;判据口径 = FBA objective_value(
|
|
4
|
-
# 实现从 HANDOFF-03 五道关卡协议产品化(农杆菌项目验证过的逻辑)
|
|
5
|
-
import re
|
|
6
|
-
import os
|
|
7
|
-
import json
|
|
8
|
-
import sys
|
|
9
|
-
import cobra
|
|
10
|
-
|
|
11
|
-
# Python -I isolated 模式下脚本目录不进 sys.path——显式插入以导入同目录模块
|
|
12
|
-
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
13
|
-
|
|
14
|
-
EX_PREFIX = ("EX_", "DM_", "SK_")
|
|
15
|
-
CORE_ELEMS = ("C", "N", "P", "S") # 硬核:不平衡必须 = 0
|
|
16
|
-
REPORT_ELEMS = ("H", "O") # 报告不阻塞
|
|
17
|
-
ELM_RE = re.compile(r"([A-Z][a-z]?)(\d*)")
|
|
18
|
-
|
|
19
|
-
# 关卡注册器(GLM 建议 + 2026-08-29 采纳):未来加 G7 不改主流程
|
|
20
|
-
GATE_REGISTRY = {}
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
def register_gate(name):
|
|
24
|
-
def deco(fn):
|
|
25
|
-
GATE_REGISTRY[name] = fn
|
|
26
|
-
return fn
|
|
27
|
-
return deco
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
def parse_formula(f):
|
|
31
|
-
"""C10H13N5O13P3 -> {"C":10,...}; 忽略 R/X 等通用占位。"""
|
|
32
|
-
d = {}
|
|
33
|
-
if not f:
|
|
34
|
-
return d
|
|
35
|
-
for m in ELM_RE.finditer(f):
|
|
36
|
-
el = m.group(1)
|
|
37
|
-
n = int(m.group(2) or 1)
|
|
38
|
-
d[el] = d.get(el, 0) + n
|
|
39
|
-
return d
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
def _rxn_elem_balance(rxn):
|
|
43
|
-
"""内部反应元素平衡:返回 {elem: delta}(delta=产物-底物,应接近 0)。"""
|
|
44
|
-
bal = {}
|
|
45
|
-
for met, coeff in rxn.metabolites.items():
|
|
46
|
-
if not met.formula:
|
|
47
|
-
continue
|
|
48
|
-
d = parse_formula(met.formula)
|
|
49
|
-
for el, n in d.items():
|
|
50
|
-
bal[el] = bal.get(el, 0) + coeff * n
|
|
51
|
-
return bal
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
class Validator:
|
|
55
|
-
def __init__(self, model_path):
|
|
56
|
-
from silentio import silent_read_sbml
|
|
57
|
-
self.path = model_path
|
|
58
|
-
self.m = silent_read_sbml(model_path)
|
|
59
|
-
|
|
60
|
-
# ------------------------------------------------------------------ G1
|
|
61
|
-
def g1_load(self):
|
|
62
|
-
m = self.m
|
|
63
|
-
from collections import Counter
|
|
64
|
-
repl = Counter()
|
|
65
|
-
bad_ids = []
|
|
66
|
-
for g in m.genes:
|
|
67
|
-
parts = g.id.split("_")
|
|
68
|
-
if len(parts) >= 3 and parts[0] == "NC":
|
|
69
|
-
repl["_".join(parts[:2])] += 1
|
|
70
|
-
else:
|
|
71
|
-
repl["other"] += 1
|
|
72
|
-
bad_ids.append(g.id)
|
|
73
|
-
n_genes_with_rxn = sum(1 for g in m.genes if len(g.reactions) > 0)
|
|
74
|
-
rep = {
|
|
75
|
-
"status": "PASS" if not bad_ids else "WARN",
|
|
76
|
-
"genes": len(m.genes), "reactions": len(m.reactions),
|
|
77
|
-
"metabolites": len(m.metabolites),
|
|
78
|
-
"replicons": dict(repl),
|
|
79
|
-
"non_nc_gene_ids": bad_ids[:10],
|
|
80
|
-
"gpr_gene_coverage": round(n_genes_with_rxn / len(m.genes), 4) if m.genes else 0,
|
|
81
|
-
}
|
|
82
|
-
return rep
|
|
83
|
-
|
|
84
|
-
# ------------------------------------------------------------------ G2
|
|
85
|
-
def g2_balance(self):
|
|
86
|
-
m = self.m
|
|
87
|
-
internal = [r for r in m.reactions
|
|
88
|
-
if not (r.id.startswith(EX_PREFIX) or r.boundary)]
|
|
89
|
-
formula_coverage = sum(1 for x in m.metabolites if x.formula) / len(m.metabolites)
|
|
90
|
-
bad_core = {} # elem -> [rxn ids]
|
|
91
|
-
bad_report = {}
|
|
92
|
-
checked = 0
|
|
93
|
-
for r in internal:
|
|
94
|
-
if not all(x.formula for x in r.metabolites):
|
|
95
|
-
continue # 公式缺失不计入不平衡(先报覆盖率)
|
|
96
|
-
checked += 1
|
|
97
|
-
bal = _rxn_elem_balance(r)
|
|
98
|
-
for el in CORE_ELEMS:
|
|
99
|
-
v = bal.get(el, 0)
|
|
100
|
-
if abs(v) > 1e-6:
|
|
101
|
-
bad_core.setdefault(el, []).append(r.id)
|
|
102
|
-
for el in REPORT_ELEMS:
|
|
103
|
-
v = bal.get(el, 0)
|
|
104
|
-
if abs(v) > 2: # charged 公式惯例噪声容忍 ±2
|
|
105
|
-
bad_report.setdefault(el, []).append(r.id)
|
|
106
|
-
n_bad = sum(len(v) for v in bad_core.values())
|
|
107
|
-
frac = 1.0 - n_bad / checked if checked else 0.0
|
|
108
|
-
status = "PASS" if n_bad == 0 else ("WARN" if frac >= 0.85 else "FAIL")
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
if
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
if
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
if
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
"status": "
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
report
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
1
|
+
# validate.py — dsh-bio-gem 五道验证关卡(M1)
|
|
2
|
+
# G1 加载统计 / G2 内部反应元素平衡 / G3 生长真实性 / G4 底物表型(条件) / G5 必需基因抽检(条件)
|
|
3
|
+
# 规格: docs/ARCHITECTURE.md §5;判据口径 = FBA objective_value(biomass 归一化 → 比生长速率 μ,单位 1/h)
|
|
4
|
+
# 实现从 HANDOFF-03 五道关卡协议产品化(农杆菌项目验证过的逻辑)
|
|
5
|
+
import re
|
|
6
|
+
import os
|
|
7
|
+
import json
|
|
8
|
+
import sys
|
|
9
|
+
import cobra
|
|
10
|
+
|
|
11
|
+
# Python -I isolated 模式下脚本目录不进 sys.path——显式插入以导入同目录模块
|
|
12
|
+
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
13
|
+
|
|
14
|
+
EX_PREFIX = ("EX_", "DM_", "SK_")
|
|
15
|
+
CORE_ELEMS = ("C", "N", "P", "S") # 硬核:不平衡必须 = 0
|
|
16
|
+
REPORT_ELEMS = ("H", "O") # 报告不阻塞
|
|
17
|
+
ELM_RE = re.compile(r"([A-Z][a-z]?)(\d*)")
|
|
18
|
+
|
|
19
|
+
# 关卡注册器(GLM 建议 + 2026-08-29 采纳):未来加 G7 不改主流程
|
|
20
|
+
GATE_REGISTRY = {}
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def register_gate(name):
|
|
24
|
+
def deco(fn):
|
|
25
|
+
GATE_REGISTRY[name] = fn
|
|
26
|
+
return fn
|
|
27
|
+
return deco
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def parse_formula(f):
|
|
31
|
+
"""C10H13N5O13P3 -> {"C":10,...}; 忽略 R/X 等通用占位。"""
|
|
32
|
+
d = {}
|
|
33
|
+
if not f:
|
|
34
|
+
return d
|
|
35
|
+
for m in ELM_RE.finditer(f):
|
|
36
|
+
el = m.group(1)
|
|
37
|
+
n = int(m.group(2) or 1)
|
|
38
|
+
d[el] = d.get(el, 0) + n
|
|
39
|
+
return d
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _rxn_elem_balance(rxn):
|
|
43
|
+
"""内部反应元素平衡:返回 {elem: delta}(delta=产物-底物,应接近 0)。"""
|
|
44
|
+
bal = {}
|
|
45
|
+
for met, coeff in rxn.metabolites.items():
|
|
46
|
+
if not met.formula:
|
|
47
|
+
continue
|
|
48
|
+
d = parse_formula(met.formula)
|
|
49
|
+
for el, n in d.items():
|
|
50
|
+
bal[el] = bal.get(el, 0) + coeff * n
|
|
51
|
+
return bal
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class Validator:
|
|
55
|
+
def __init__(self, model_path):
|
|
56
|
+
from silentio import silent_read_sbml
|
|
57
|
+
self.path = model_path
|
|
58
|
+
self.m = silent_read_sbml(model_path)
|
|
59
|
+
|
|
60
|
+
# ------------------------------------------------------------------ G1
|
|
61
|
+
def g1_load(self):
|
|
62
|
+
m = self.m
|
|
63
|
+
from collections import Counter
|
|
64
|
+
repl = Counter()
|
|
65
|
+
bad_ids = []
|
|
66
|
+
for g in m.genes:
|
|
67
|
+
parts = g.id.split("_")
|
|
68
|
+
if len(parts) >= 3 and parts[0] == "NC":
|
|
69
|
+
repl["_".join(parts[:2])] += 1
|
|
70
|
+
else:
|
|
71
|
+
repl["other"] += 1
|
|
72
|
+
bad_ids.append(g.id)
|
|
73
|
+
n_genes_with_rxn = sum(1 for g in m.genes if len(g.reactions) > 0)
|
|
74
|
+
rep = {
|
|
75
|
+
"status": "PASS" if not bad_ids else "WARN",
|
|
76
|
+
"genes": len(m.genes), "reactions": len(m.reactions),
|
|
77
|
+
"metabolites": len(m.metabolites),
|
|
78
|
+
"replicons": dict(repl),
|
|
79
|
+
"non_nc_gene_ids": bad_ids[:10],
|
|
80
|
+
"gpr_gene_coverage": round(n_genes_with_rxn / len(m.genes), 4) if m.genes else 0,
|
|
81
|
+
}
|
|
82
|
+
return rep
|
|
83
|
+
|
|
84
|
+
# ------------------------------------------------------------------ G2
|
|
85
|
+
def g2_balance(self):
|
|
86
|
+
m = self.m
|
|
87
|
+
internal = [r for r in m.reactions
|
|
88
|
+
if not (r.id.startswith(EX_PREFIX) or r.boundary)]
|
|
89
|
+
formula_coverage = sum(1 for x in m.metabolites if x.formula) / len(m.metabolites)
|
|
90
|
+
bad_core = {} # elem -> [rxn ids]
|
|
91
|
+
bad_report = {}
|
|
92
|
+
checked = 0
|
|
93
|
+
for r in internal:
|
|
94
|
+
if not all(x.formula for x in r.metabolites):
|
|
95
|
+
continue # 公式缺失不计入不平衡(先报覆盖率)
|
|
96
|
+
checked += 1
|
|
97
|
+
bal = _rxn_elem_balance(r)
|
|
98
|
+
for el in CORE_ELEMS:
|
|
99
|
+
v = bal.get(el, 0)
|
|
100
|
+
if abs(v) > 1e-6:
|
|
101
|
+
bad_core.setdefault(el, []).append(r.id)
|
|
102
|
+
for el in REPORT_ELEMS:
|
|
103
|
+
v = bal.get(el, 0)
|
|
104
|
+
if abs(v) > 2: # charged 公式惯例噪声容忍 ±2
|
|
105
|
+
bad_report.setdefault(el, []).append(r.id)
|
|
106
|
+
n_bad = sum(len(v) for v in bad_core.values())
|
|
107
|
+
frac = 1.0 - n_bad / checked if checked else 0.0
|
|
108
|
+
status = "PASS" if n_bad == 0 else ("WARN" if frac >= 0.85 else "FAIL")
|
|
109
|
+
|
|
110
|
+
# 2026-09-11 修复(agent 在真实 E2E 中发现并指出):**公式覆盖率是 PASS 结论的
|
|
111
|
+
# 作用域上界** —— 覆盖率低时「无反应不平衡」只说明被检查的那部分没问题,不能
|
|
112
|
+
# 外推为整体 PASS。实测 iNX1344_v3:覆盖率 68.35% 却判 PASS,agent 据此指出
|
|
113
|
+
# 「g2 的 PASS 是假阳性,因为它只检查了有 formula 的 68.35% 代谢物」。
|
|
114
|
+
# 该结论会误导后续判断(例如「配平没问题」),故低覆盖时降级为 WARN。
|
|
115
|
+
G2_COVERAGE_GATE = 0.90
|
|
116
|
+
unchecked = len(internal) - checked
|
|
117
|
+
coverage_scope_note = None
|
|
118
|
+
if status == "PASS" and formula_coverage < G2_COVERAGE_GATE:
|
|
119
|
+
status = "WARN"
|
|
120
|
+
coverage_scope_note = (
|
|
121
|
+
f"元素平衡结论只覆盖 {formula_coverage * 100:.2f}% 代谢物、"
|
|
122
|
+
f"{checked}/{len(internal)} 条内部反应可判定(跳过 {unchecked} 条)"
|
|
123
|
+
f"→ 不足以判 PASS。补全代谢物 formula 后必须重跑本关;"
|
|
124
|
+
f"在此之前不要据本关结论断言「配平没问题」"
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
rep = {
|
|
128
|
+
"status": status,
|
|
129
|
+
"internal_reactions": len(internal), "formula_checked": checked,
|
|
130
|
+
"formula_unchecked_reactions": unchecked,
|
|
131
|
+
"metabolite_formula_coverage": round(formula_coverage, 4),
|
|
132
|
+
"coverage_gate": G2_COVERAGE_GATE,
|
|
133
|
+
"core_unbalanced": {k: len(v) for k, v in bad_core.items()},
|
|
134
|
+
"core_unbalanced_examples": {k: v[:5] for k, v in bad_core.items()},
|
|
135
|
+
"h_o_report": {k: len(v) for k, v in bad_report.items()},
|
|
136
|
+
"core_balance_frac": round(frac, 4),
|
|
137
|
+
# P0-2(2026-08-31 LBA9402 会话实测):agent 会把 0.9985 心算换算成百分比而被防火墙拦——直接给原始百分数字段
|
|
138
|
+
"core_balance_frac_pct": round(frac * 100, 2),
|
|
139
|
+
}
|
|
140
|
+
if coverage_scope_note:
|
|
141
|
+
rep["coverage_scope_note"] = coverage_scope_note
|
|
142
|
+
return rep
|
|
143
|
+
|
|
144
|
+
# ------------------------------------------------------------------ G3
|
|
145
|
+
def g3_growth(self, medium, reference_growth=None):
|
|
146
|
+
"""medium: {EX_id: lower_bound}; 三态:medium / no-carbon / all-closed。"""
|
|
147
|
+
m = self.m
|
|
148
|
+
|
|
149
|
+
def _setup(medium_dict):
|
|
150
|
+
for r in m.reactions:
|
|
151
|
+
if r.id.startswith(EX_PREFIX) or r.boundary:
|
|
152
|
+
r.lower_bound = 0.0
|
|
153
|
+
for rid, lb in (medium_dict or {}).items():
|
|
154
|
+
if rid in m.reactions:
|
|
155
|
+
m.reactions.get_by_id(rid).lower_bound = lb
|
|
156
|
+
else:
|
|
157
|
+
return rid
|
|
158
|
+
return None
|
|
159
|
+
|
|
160
|
+
miss = _setup(medium or {})
|
|
161
|
+
if miss:
|
|
162
|
+
return {"status": "FAIL", "reason": f"medium exchange not in model: {miss}",
|
|
163
|
+
"medium_provided": bool(medium)}
|
|
164
|
+
with m:
|
|
165
|
+
wt = m.optimize().objective_value
|
|
166
|
+
# no-carbon: 去掉 formula 含 C 的交换
|
|
167
|
+
no_c_medium = dict(medium or {})
|
|
168
|
+
for rid in list(no_c_medium):
|
|
169
|
+
if rid.startswith("EX_") and rid in m.reactions:
|
|
170
|
+
met = list(m.reactions.get_by_id(rid).metabolites)[0]
|
|
171
|
+
if met.formula and "C" in parse_formula(met.formula):
|
|
172
|
+
del no_c_medium[rid]
|
|
173
|
+
miss = _setup(no_c_medium)
|
|
174
|
+
with m:
|
|
175
|
+
g_no_c = m.optimize().objective_value
|
|
176
|
+
miss = _setup({}) # all closed
|
|
177
|
+
with m:
|
|
178
|
+
g_closed = m.optimize().objective_value
|
|
179
|
+
ok_grow = wt > 1e-6
|
|
180
|
+
ok_noc = abs(g_no_c) < 1e-6
|
|
181
|
+
ok_closed = abs(g_closed) < 1e-6
|
|
182
|
+
ratio = None
|
|
183
|
+
if reference_growth and reference_growth > 0:
|
|
184
|
+
ratio = wt / reference_growth
|
|
185
|
+
if not medium:
|
|
186
|
+
status = "WARN" # 无声明培养基 -> 无法验证
|
|
187
|
+
elif ok_grow and ok_noc and ok_closed and (ratio is None or ratio >= 0.99):
|
|
188
|
+
status = "PASS"
|
|
189
|
+
else:
|
|
190
|
+
status = "FAIL" # 有培养基但生长不达标(或对照泄漏)——构建侧必须走补洞闭环
|
|
191
|
+
rep = {
|
|
192
|
+
"status": status,
|
|
193
|
+
"medium_provided": bool(medium),
|
|
194
|
+
"growth_medium": round(wt, 6),
|
|
195
|
+
"growth_no_carbon": round(g_no_c, 6),
|
|
196
|
+
"growth_all_closed": round(g_closed, 6),
|
|
197
|
+
"ratio_vs_reference": round(ratio, 4) if ratio is not None else None,
|
|
198
|
+
"checks": {"medium>0": ok_grow, "no_carbon==0": ok_noc, "closed==0": ok_closed},
|
|
199
|
+
# 阶段A-M4 口径声明(只增):单点 FBA 值非硬结论
|
|
200
|
+
"units": "1/h",
|
|
201
|
+
"point_value_note": "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定",
|
|
202
|
+
}
|
|
203
|
+
return rep
|
|
204
|
+
|
|
205
|
+
# ------------------------------------------------------------------ G6
|
|
206
|
+
@register_gate("G6")
|
|
207
|
+
def g6_atp_leak(self, context=None):
|
|
208
|
+
"""ATP 泄漏测试(MEMOTE 核心测试;G3 all-closed 的必要不充分检查):
|
|
209
|
+
全关交换后最大化 ATP 代谢物的净消耗(demand),通量 > 0.01 → WARN。
|
|
210
|
+
补洞后必跑(context.post_gapfill 时不再跳过)。
|
|
211
|
+
P1-5 修复(2026-08-31 LBA9402 会话实测):CarveMe 模型 ATP id 为 M_atp_c,
|
|
212
|
+
旧匹配只看 atp_c/cpd00002_c0 -> 误 SKIP「未找到 ATP」——扩展命名模式 + SKIP 时列出尝试模式与模型内候选。
|
|
213
|
+
P2-9 修复(2026-09-10 iNX1344 E2E 实测):MetaCyc/BioCyc 导出模型 ATP 为 M00002_c
|
|
214
|
+
(name='ATP',formula 为去质子化变体),仍不在模式表内 → 二次误 SKIP(G6 直接失效)。
|
|
215
|
+
改为三级通用解析:id 模式 → name 匹配 → formula 匹配,跨 ID 体系自适应。"""
|
|
216
|
+
m = self.m
|
|
217
|
+
atp_patterns = ("atp_c", "cpd00002_c0", "m_atp_c", "atp_c0", "cpd00002", "atp",
|
|
218
|
+
"m00002", "m00002_c", "m00002_c0")
|
|
219
|
+
cands = [x for x in m.metabolites if (x.id or "").lower() in atp_patterns]
|
|
220
|
+
atp_source = "id_pattern"
|
|
221
|
+
if not cands:
|
|
222
|
+
# 回退 1:name 匹配(跨 ID 体系最稳的信号;排除 dATP 等衍生物)
|
|
223
|
+
cands = [x for x in m.metabolites
|
|
224
|
+
if (x.name or "").strip().upper() == "ATP"
|
|
225
|
+
or ("atp" in (x.name or "").lower() and "datp" not in (x.name or "").lower())]
|
|
226
|
+
atp_source = "name_match"
|
|
227
|
+
if not cands:
|
|
228
|
+
# 回退 2:分子式(含去质子化变体——部分模型 formula 非标准形式)
|
|
229
|
+
cands = [x for x in m.metabolites
|
|
230
|
+
if (x.formula or "").replace(" ", "") in ("C10H16N5O13P3", "C10H12N5O13P3")]
|
|
231
|
+
atp_source = "formula_match"
|
|
232
|
+
cyto = [x for x in cands if x.compartment in ("c0", "c")]
|
|
233
|
+
atp_c = (cyto or cands or [None])[0]
|
|
234
|
+
if atp_c is None:
|
|
235
|
+
atp_like = sorted({x.id for x in m.metabolites
|
|
236
|
+
if "atp" in (x.id or "").lower() or "atp" in (x.name or "").lower()})[:10]
|
|
237
|
+
return {"status": "SKIP",
|
|
238
|
+
"reason": "未找到 ATP 代谢物(id 模式 / name 匹配 / formula 匹配三级回退均未命中)",
|
|
239
|
+
"tried_patterns": list(atp_patterns),
|
|
240
|
+
"atp_like_ids_in_model": atp_like,
|
|
241
|
+
"note": "SKIP 系命名口径未命中(非模型缺陷证明);若模型含 ATP 但三级回退均未命中,"
|
|
242
|
+
"请在模型内显式标注 atp 角色后重跑"}
|
|
243
|
+
dm = cobra.Reaction("DM_gem_atp_leak", name="G6 ATP 泄漏检测 demand",
|
|
244
|
+
lower_bound=0.0, upper_bound=1000.0)
|
|
245
|
+
dm.add_metabolites({atp_c: -1})
|
|
246
|
+
m.add_reactions([dm])
|
|
247
|
+
try:
|
|
248
|
+
with m:
|
|
249
|
+
for r in m.reactions:
|
|
250
|
+
if r.id.startswith(EX_PREFIX) or r.boundary:
|
|
251
|
+
r.lower_bound = 0.0
|
|
252
|
+
v = m.optimize().objective_value
|
|
253
|
+
finally:
|
|
254
|
+
m.remove_reactions([dm])
|
|
255
|
+
leak = abs(v)
|
|
256
|
+
status = "PASS" if leak <= 0.01 else "WARN"
|
|
257
|
+
return {"status": status, "atp_leak_flux": round(leak, 6),
|
|
258
|
+
"atp_metabolite_found": atp_c.id, "atp_resolved_by": atp_source,
|
|
259
|
+
"threshold": 0.01, "post_gapfill": bool((context or {}).get("post_gapfill")),
|
|
260
|
+
"note": "全关交换后 ATP demand 通量应≈0;>0.01 提示能量循环泄漏(L3 MILP 补洞最可能引入)"}
|
|
261
|
+
|
|
262
|
+
# ------------------------------------------------------------------ G4
|
|
263
|
+
def g4_phenotype(self, table_path=None, substrates=None, medium=None, carbon_mode="supplement"):
|
|
264
|
+
"""条件执行:需参照表(TSV: substrate<TAB>published 0/1)或 substrates+published。
|
|
265
|
+
carbon_mode: supplement=基准培养基不变+底物-10(对齐 HANDOFF-03 关卡4 基线 16/19→17/19);
|
|
266
|
+
sole=去含碳交换后底物-10(唯一碳源严格语义,氮源类测试会误判)。"""
|
|
267
|
+
if table_path and os.path.exists(table_path):
|
|
268
|
+
rows = []
|
|
269
|
+
with open(table_path, encoding="utf-8") as f:
|
|
270
|
+
for line in f:
|
|
271
|
+
line = line.rstrip("\r\n")
|
|
272
|
+
if not line or line.startswith("#") or line.startswith("substrate"):
|
|
273
|
+
continue
|
|
274
|
+
p = line.split("\t")
|
|
275
|
+
if len(p) >= 2:
|
|
276
|
+
rows.append((p[0].strip(), int(float(p[1]))))
|
|
277
|
+
elif substrates:
|
|
278
|
+
rows = substrates
|
|
279
|
+
else:
|
|
280
|
+
return {"status": "SKIP", "reason": "no phenotype reference provided"}
|
|
281
|
+
if not medium:
|
|
282
|
+
return {"status": "SKIP", "reason": "G4 needs medium to define base"}
|
|
283
|
+
from gapfind import build_ex_index, match_ex, SYN
|
|
284
|
+
m = self.m
|
|
285
|
+
# 统一走 gapfind 的 build_ex_index + match_ex(修复过子串误配规则;勿再各自实现)
|
|
286
|
+
ex_idx = build_ex_index(m)
|
|
287
|
+
results = []
|
|
288
|
+
matched = 0
|
|
289
|
+
for sub, pub in rows:
|
|
290
|
+
exid = match_ex(sub, ex_idx)
|
|
291
|
+
# 基底:medium(supplement)或去碳后加底物(sole)
|
|
292
|
+
med2 = dict(medium)
|
|
293
|
+
if carbon_mode == "sole":
|
|
294
|
+
for rid in list(med2):
|
|
295
|
+
if rid.startswith("EX_") and rid in m.reactions:
|
|
296
|
+
met = list(m.reactions.get_by_id(rid).metabolites)[0]
|
|
297
|
+
if met.formula and "C" in parse_formula(met.formula):
|
|
298
|
+
del med2[rid]
|
|
299
|
+
if exid and exid in m.reactions:
|
|
300
|
+
med2[exid] = -10.0
|
|
301
|
+
for r in m.reactions:
|
|
302
|
+
if r.id.startswith(EX_PREFIX) or r.boundary:
|
|
303
|
+
r.lower_bound = 0.0
|
|
304
|
+
for rid, lb in med2.items():
|
|
305
|
+
if rid in m.reactions:
|
|
306
|
+
m.reactions.get_by_id(rid).lower_bound = lb
|
|
307
|
+
with m:
|
|
308
|
+
g = m.optimize().objective_value
|
|
309
|
+
pred = g > 1e-6
|
|
310
|
+
ok = (pred == bool(pub))
|
|
311
|
+
if ok:
|
|
312
|
+
matched += 1
|
|
313
|
+
results.append({"substrate": sub, "published": int(pub), "predicted": int(pred),
|
|
314
|
+
"growth": round(g, 6), "exchange": exid or None, "match": bool(ok),
|
|
315
|
+
# 阶段A-M4 口径声明(只增):每底物 growth 为单点 FBA 值
|
|
316
|
+
"units": "1/h",
|
|
317
|
+
"point_value_note": "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定"})
|
|
318
|
+
rep = {
|
|
319
|
+
"status": "PASS" if rows and matched / len(rows) >= 0.8 else ("WARN" if rows else "SKIP"),
|
|
320
|
+
"carbon_mode": carbon_mode,
|
|
321
|
+
"matched": matched, "total": len(rows),
|
|
322
|
+
"rate": round(matched / len(rows), 4) if rows else None,
|
|
323
|
+
"results": results,
|
|
324
|
+
}
|
|
325
|
+
return rep
|
|
326
|
+
|
|
327
|
+
# ------------------------------------------------------------------ G5
|
|
328
|
+
def g5_essentiality(self, essential_test, medium, reference_essential=None):
|
|
329
|
+
"""条件执行:对给定基因列表逐一手工敲除(with m: 循环),输出每个基因的必要性。
|
|
330
|
+
若给 reference_essential(已知必需基因集)→ 对交集算召回。"""
|
|
331
|
+
if not essential_test:
|
|
332
|
+
return {"status": "SKIP", "reason": "no essential_test gene list provided"}
|
|
333
|
+
if not medium:
|
|
334
|
+
return {"status": "SKIP", "reason": "G5 needs medium"}
|
|
335
|
+
m = self.m
|
|
336
|
+
present = [g for g in essential_test if g in m.genes]
|
|
337
|
+
if len(present) / len(essential_test) < 0.8:
|
|
338
|
+
return {"status": "SKIP", "reason": "gene mapping coverage < 80%",
|
|
339
|
+
"present": len(present), "total": len(essential_test)}
|
|
340
|
+
results = []
|
|
341
|
+
for gid in present:
|
|
342
|
+
with m:
|
|
343
|
+
for r in m.reactions:
|
|
344
|
+
if r.id.startswith(EX_PREFIX) or r.boundary:
|
|
345
|
+
r.lower_bound = 0.0
|
|
346
|
+
for rid, lb in medium.items():
|
|
347
|
+
if rid in m.reactions:
|
|
348
|
+
m.reactions.get_by_id(rid).lower_bound = lb
|
|
349
|
+
m.genes.get_by_id(gid).knock_out()
|
|
350
|
+
g = m.optimize().objective_value
|
|
351
|
+
results.append({"gene": gid, "growth": round(g, 6),
|
|
352
|
+
"essential": bool(g < 1e-6)})
|
|
353
|
+
n_ess = sum(1 for r_ in results if r_["essential"])
|
|
354
|
+
recall = None
|
|
355
|
+
if reference_essential:
|
|
356
|
+
ref = set(reference_essential)
|
|
357
|
+
tp = sum(1 for r_ in results if r_["essential"] and r_["gene"] in ref)
|
|
358
|
+
recall = round(tp / len(ref), 4) if ref else None
|
|
359
|
+
rep = {
|
|
360
|
+
"status": "PASS" if (recall is None or recall >= 0.4) else "WARN",
|
|
361
|
+
"tested": len(results), "essential_found": n_ess,
|
|
362
|
+
"recall_vs_reference": recall,
|
|
363
|
+
"details": results,
|
|
364
|
+
}
|
|
365
|
+
return rep
|
|
366
|
+
|
|
367
|
+
# ------------------------------------------------------------------ run
|
|
368
|
+
def run(self, medium=None, phenotype_table=None, essential_test=None,
|
|
369
|
+
reference_growth=None, reference_essential=None, carbon_mode="supplement",
|
|
370
|
+
context=None):
|
|
371
|
+
from gapfind import resolve_medium, expand_medium
|
|
372
|
+
from coherence import model_coherence
|
|
373
|
+
medium, _preset = expand_medium(medium)
|
|
374
|
+
resolved_med, unresolved = resolve_medium(self.m, medium) if medium else ({}, [])
|
|
375
|
+
report = {"model": self.path,
|
|
376
|
+
"units": {"growth": "1/h", "note": "growth 为比生长速率 μ:biomass 反应归一化到 1 gDW 时其通量数值等于 μ(标准 GEM 约定);非归一化模型的 growth 应以 mmol/gDW/h 解读"},
|
|
377
|
+
# G0 模型数据质量前置诊断:未映射前体等数据问题会让 G2/G3 与 gapfind 的
|
|
378
|
+
# 结论失真(实测 iNX1344_v3:gapfind 报的 5 个 L3 缺口实为未映射前体所致)
|
|
379
|
+
"g0": model_coherence(self.m),
|
|
380
|
+
"g1": self.g1_load(),
|
|
381
|
+
"g2": self.g2_balance(),
|
|
382
|
+
"g6": self.g6_atp_leak(context=context or {})}
|
|
383
|
+
g3 = self.g3_growth(resolved_med, reference_growth)
|
|
384
|
+
if unresolved:
|
|
385
|
+
g3["medium_unresolved"] = unresolved
|
|
386
|
+
report["g3"] = g3
|
|
387
|
+
if phenotype_table:
|
|
388
|
+
report["g4"] = self.g4_phenotype(table_path=phenotype_table, medium=resolved_med,
|
|
389
|
+
carbon_mode=carbon_mode)
|
|
390
|
+
else:
|
|
391
|
+
report["g4"] = {"status": "SKIP", "reason": "no phenotype reference provided"}
|
|
392
|
+
if essential_test:
|
|
393
|
+
report["g5"] = self.g5_essentiality(essential_test, resolved_med, reference_essential)
|
|
394
|
+
else:
|
|
395
|
+
report["g5"] = {"status": "SKIP", "reason": "no essential_test provided"}
|
|
396
|
+
# 总判定:G1/G2/G3/G6(FATAL 才 FAIL;G6 WARN 级不阻塞但补洞后必查)
|
|
397
|
+
blocked = ["g1", "g2", "g3"]
|
|
398
|
+
fails = [k for k in blocked if report[k]["status"] == "FAIL"]
|
|
399
|
+
warns = [k for k in blocked if report[k]["status"] == "WARN"]
|
|
400
|
+
report["overall"] = "FAIL" if fails else ("WARN" if warns else "PASS")
|
|
401
|
+
report["blocking"] = blocked
|
|
402
|
+
return report
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def validate_model(model_path, medium=None, phenotype_table=None,
|
|
406
|
+
essential_test=None, reference_growth=None, reference_essential=None,
|
|
407
|
+
carbon_mode="supplement"):
|
|
408
|
+
v = Validator(model_path)
|
|
409
|
+
return v.run(medium=medium, phenotype_table=phenotype_table,
|
|
410
|
+
essential_test=essential_test, reference_growth=reference_growth,
|
|
411
|
+
reference_essential=reference_essential, carbon_mode=carbon_mode)
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
if __name__ == "__main__":
|
|
415
|
+
# 命令行直跑(开发/测试用):python validate.py <model.xml> [--medium-json x] [--table t] [--g5 g1,g2]
|
|
416
|
+
import sys
|
|
417
|
+
path = sys.argv[1]
|
|
418
|
+
med = None
|
|
419
|
+
table = None
|
|
420
|
+
g5 = None
|
|
421
|
+
refg = None
|
|
422
|
+
i = 2
|
|
423
|
+
while i < len(sys.argv):
|
|
424
|
+
if sys.argv[i] == "--medium-json" and i + 1 < len(sys.argv):
|
|
425
|
+
med = json.loads(sys.argv[i + 1]); i += 2
|
|
426
|
+
elif sys.argv[i] == "--table" and i + 1 < len(sys.argv):
|
|
427
|
+
table = sys.argv[i + 1]; i += 2
|
|
428
|
+
elif sys.argv[i] == "--g5" and i + 1 < len(sys.argv):
|
|
429
|
+
g5 = sys.argv[i + 1].split(","); i += 2
|
|
430
|
+
elif sys.argv[i] == "--ref-growth" and i + 1 < len(sys.argv):
|
|
431
|
+
refg = float(sys.argv[i + 1]); i += 2
|
|
432
|
+
else:
|
|
433
|
+
i += 1
|
|
434
|
+
rep = validate_model(path, medium=med, phenotype_table=table, essential_test=g5,
|
|
435
|
+
reference_growth=refg)
|
|
410
436
|
print(json.dumps(rep, ensure_ascii=False, indent=2))
|