@dsh-bio/dsh-bio-gem 0.1.2 → 0.1.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,393 +1,436 @@
1
- # validate.py — dsh-bio-gem 五道验证关卡(M1)
2
- # G1 加载统计 / G2 内部反应元素平衡 / G3 生长真实性 / G4 底物表型(条件) / G5 必需基因抽检(条件)
3
- # 规格: docs/ARCHITECTURE.md §5;判据口径 = FBA objective_value(mmol/gDW/h,不用 μ)
4
- # 实现从 HANDOFF-03 五道关卡协议产品化(农杆菌项目验证过的逻辑)
5
- import re
6
- import os
7
- import json
8
- import sys
9
- import cobra
10
-
11
- # Python -I isolated 模式下脚本目录不进 sys.path——显式插入以导入同目录模块
12
- sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
13
-
14
- EX_PREFIX = ("EX_", "DM_", "SK_")
15
- CORE_ELEMS = ("C", "N", "P", "S") # 硬核:不平衡必须 = 0
16
- REPORT_ELEMS = ("H", "O") # 报告不阻塞
17
- ELM_RE = re.compile(r"([A-Z][a-z]?)(\d*)")
18
-
19
- # 关卡注册器(GLM 建议 + 2026-08-29 采纳):未来加 G7 不改主流程
20
- GATE_REGISTRY = {}
21
-
22
-
23
- def register_gate(name):
24
- def deco(fn):
25
- GATE_REGISTRY[name] = fn
26
- return fn
27
- return deco
28
-
29
-
30
- def parse_formula(f):
31
- """C10H13N5O13P3 -> {"C":10,...}; 忽略 R/X 等通用占位。"""
32
- d = {}
33
- if not f:
34
- return d
35
- for m in ELM_RE.finditer(f):
36
- el = m.group(1)
37
- n = int(m.group(2) or 1)
38
- d[el] = d.get(el, 0) + n
39
- return d
40
-
41
-
42
- def _rxn_elem_balance(rxn):
43
- """内部反应元素平衡:返回 {elem: delta}(delta=产物-底物,应接近 0)。"""
44
- bal = {}
45
- for met, coeff in rxn.metabolites.items():
46
- if not met.formula:
47
- continue
48
- d = parse_formula(met.formula)
49
- for el, n in d.items():
50
- bal[el] = bal.get(el, 0) + coeff * n
51
- return bal
52
-
53
-
54
- class Validator:
55
- def __init__(self, model_path):
56
- from silentio import silent_read_sbml
57
- self.path = model_path
58
- self.m = silent_read_sbml(model_path)
59
-
60
- # ------------------------------------------------------------------ G1
61
- def g1_load(self):
62
- m = self.m
63
- from collections import Counter
64
- repl = Counter()
65
- bad_ids = []
66
- for g in m.genes:
67
- parts = g.id.split("_")
68
- if len(parts) >= 3 and parts[0] == "NC":
69
- repl["_".join(parts[:2])] += 1
70
- else:
71
- repl["other"] += 1
72
- bad_ids.append(g.id)
73
- n_genes_with_rxn = sum(1 for g in m.genes if len(g.reactions) > 0)
74
- rep = {
75
- "status": "PASS" if not bad_ids else "WARN",
76
- "genes": len(m.genes), "reactions": len(m.reactions),
77
- "metabolites": len(m.metabolites),
78
- "replicons": dict(repl),
79
- "non_nc_gene_ids": bad_ids[:10],
80
- "gpr_gene_coverage": round(n_genes_with_rxn / len(m.genes), 4) if m.genes else 0,
81
- }
82
- return rep
83
-
84
- # ------------------------------------------------------------------ G2
85
- def g2_balance(self):
86
- m = self.m
87
- internal = [r for r in m.reactions
88
- if not (r.id.startswith(EX_PREFIX) or r.boundary)]
89
- formula_coverage = sum(1 for x in m.metabolites if x.formula) / len(m.metabolites)
90
- bad_core = {} # elem -> [rxn ids]
91
- bad_report = {}
92
- checked = 0
93
- for r in internal:
94
- if not all(x.formula for x in r.metabolites):
95
- continue # 公式缺失不计入不平衡(先报覆盖率)
96
- checked += 1
97
- bal = _rxn_elem_balance(r)
98
- for el in CORE_ELEMS:
99
- v = bal.get(el, 0)
100
- if abs(v) > 1e-6:
101
- bad_core.setdefault(el, []).append(r.id)
102
- for el in REPORT_ELEMS:
103
- v = bal.get(el, 0)
104
- if abs(v) > 2: # charged 公式惯例噪声容忍 ±2
105
- bad_report.setdefault(el, []).append(r.id)
106
- n_bad = sum(len(v) for v in bad_core.values())
107
- frac = 1.0 - n_bad / checked if checked else 0.0
108
- status = "PASS" if n_bad == 0 else ("WARN" if frac >= 0.85 else "FAIL")
109
- rep = {
110
- "status": status,
111
- "internal_reactions": len(internal), "formula_checked": checked,
112
- "metabolite_formula_coverage": round(formula_coverage, 4),
113
- "core_unbalanced": {k: len(v) for k, v in bad_core.items()},
114
- "core_unbalanced_examples": {k: v[:5] for k, v in bad_core.items()},
115
- "h_o_report": {k: len(v) for k, v in bad_report.items()},
116
- "core_balance_frac": round(frac, 4),
117
- # P0-2(2026-08-31 LBA9402 会话实测):agent 会把 0.9985 心算换算成百分比而被防火墙拦——直接给原始百分数字段
118
- "core_balance_frac_pct": round(frac * 100, 2),
119
- }
120
- return rep
121
-
122
- # ------------------------------------------------------------------ G3
123
- def g3_growth(self, medium, reference_growth=None):
124
- """medium: {EX_id: lower_bound}; 三态:medium / no-carbon / all-closed。"""
125
- m = self.m
126
-
127
- def _setup(medium_dict):
128
- for r in m.reactions:
129
- if r.id.startswith(EX_PREFIX) or r.boundary:
130
- r.lower_bound = 0.0
131
- for rid, lb in (medium_dict or {}).items():
132
- if rid in m.reactions:
133
- m.reactions.get_by_id(rid).lower_bound = lb
134
- else:
135
- return rid
136
- return None
137
-
138
- miss = _setup(medium or {})
139
- if miss:
140
- return {"status": "FAIL", "reason": f"medium exchange not in model: {miss}",
141
- "medium_provided": bool(medium)}
142
- with m:
143
- wt = m.optimize().objective_value
144
- # no-carbon: 去掉 formula 含 C 的交换
145
- no_c_medium = dict(medium or {})
146
- for rid in list(no_c_medium):
147
- if rid.startswith("EX_") and rid in m.reactions:
148
- met = list(m.reactions.get_by_id(rid).metabolites)[0]
149
- if met.formula and "C" in parse_formula(met.formula):
150
- del no_c_medium[rid]
151
- miss = _setup(no_c_medium)
152
- with m:
153
- g_no_c = m.optimize().objective_value
154
- miss = _setup({}) # all closed
155
- with m:
156
- g_closed = m.optimize().objective_value
157
- ok_grow = wt > 1e-6
158
- ok_noc = abs(g_no_c) < 1e-6
159
- ok_closed = abs(g_closed) < 1e-6
160
- ratio = None
161
- if reference_growth and reference_growth > 0:
162
- ratio = wt / reference_growth
163
- if not medium:
164
- status = "WARN" # 无声明培养基 -> 无法验证
165
- elif ok_grow and ok_noc and ok_closed and (ratio is None or ratio >= 0.99):
166
- status = "PASS"
167
- else:
168
- status = "FAIL" # 有培养基但生长不达标(或对照泄漏)——构建侧必须走补洞闭环
169
- rep = {
170
- "status": status,
171
- "medium_provided": bool(medium),
172
- "growth_medium": round(wt, 6),
173
- "growth_no_carbon": round(g_no_c, 6),
174
- "growth_all_closed": round(g_closed, 6),
175
- "ratio_vs_reference": round(ratio, 4) if ratio is not None else None,
176
- "checks": {"medium>0": ok_grow, "no_carbon==0": ok_noc, "closed==0": ok_closed},
177
- # 阶段A-M4 口径声明(只增):单点 FBA 值非硬结论
178
- "units": "mmol/gDW/h",
179
- "point_value_note": "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定",
180
- }
181
- return rep
182
-
183
- # ------------------------------------------------------------------ G6
184
- @register_gate("G6")
185
- def g6_atp_leak(self, context=None):
186
- """ATP 泄漏测试(MEMOTE 核心测试;G3 all-closed 的必要不充分检查):
187
- 全关交换后最大化 ATP 代谢物的净消耗(demand),通量 > 0.01 → WARN。
188
- 补洞后必跑(context.post_gapfill 时不再跳过)。
189
- P1-5 修复(2026-08-31 LBA9402 会话实测):CarveMe 模型 ATP id 为 M_atp_c,
190
- 旧匹配只看 atp_c/cpd00002_c0 -> 误 SKIP「未找到 ATP」——扩展命名模式 + SKIP 时列出尝试模式与模型内候选。"""
191
- m = self.m
192
- atp_patterns = ("atp_c", "cpd00002_c0", "m_atp_c", "atp_c0", "cpd00002", "atp")
193
- cands = [x for x in m.metabolites if (x.id or "").lower() in atp_patterns]
194
- cyto = [x for x in cands if x.compartment in ("c0", "c")]
195
- atp_c = (cyto or cands or [None])[0]
196
- if atp_c is None:
197
- atp_like = sorted({x.id for x in m.metabolites if "atp" in (x.id or "").lower()})[:10]
198
- return {"status": "SKIP",
199
- "reason": "未找到 ATP 代谢物(已尝试命名模式: " + ", ".join(atp_patterns) + ")",
200
- "tried_patterns": list(atp_patterns),
201
- "atp_like_ids_in_model": atp_like,
202
- "note": "SKIP 系命名口径未命中(非模型缺陷证明);若模型含 ATP 但 id 不在尝试模式中,"
203
- "补充模式或标注 atp 角色后重跑"}
204
- dm = cobra.Reaction("DM_gem_atp_leak", name="G6 ATP 泄漏检测 demand",
205
- lower_bound=0.0, upper_bound=1000.0)
206
- dm.add_metabolites({atp_c: -1})
207
- m.add_reactions([dm])
208
- try:
209
- with m:
210
- for r in m.reactions:
211
- if r.id.startswith(EX_PREFIX) or r.boundary:
212
- r.lower_bound = 0.0
213
- v = m.optimize().objective_value
214
- finally:
215
- m.remove_reactions([dm])
216
- leak = abs(v)
217
- status = "PASS" if leak <= 0.01 else "WARN"
218
- return {"status": status, "atp_leak_flux": round(leak, 6),
219
- "atp_metabolite_found": atp_c.id,
220
- "threshold": 0.01, "post_gapfill": bool((context or {}).get("post_gapfill")),
221
- "note": "全关交换后 ATP demand 通量应≈0;>0.01 提示能量循环泄漏(L3 MILP 补洞最可能引入)"}
222
-
223
- # ------------------------------------------------------------------ G4
224
- def g4_phenotype(self, table_path=None, substrates=None, medium=None, carbon_mode="supplement"):
225
- """条件执行:需参照表(TSV: substrate<TAB>published 0/1)或 substrates+published。
226
- carbon_mode: supplement=基准培养基不变+底物-10(对齐 HANDOFF-03 关卡4 基线 16/19→17/19);
227
- sole=去含碳交换后底物-10(唯一碳源严格语义,氮源类测试会误判)。"""
228
- if table_path and os.path.exists(table_path):
229
- rows = []
230
- with open(table_path, encoding="utf-8") as f:
231
- for line in f:
232
- line = line.rstrip("\r\n")
233
- if not line or line.startswith("#") or line.startswith("substrate"):
234
- continue
235
- p = line.split("\t")
236
- if len(p) >= 2:
237
- rows.append((p[0].strip(), int(float(p[1]))))
238
- elif substrates:
239
- rows = substrates
240
- else:
241
- return {"status": "SKIP", "reason": "no phenotype reference provided"}
242
- if not medium:
243
- return {"status": "SKIP", "reason": "G4 needs medium to define base"}
244
- from gapfind import build_ex_index, match_ex, SYN
245
- m = self.m
246
- # 统一走 gapfind 的 build_ex_index + match_ex(修复过子串误配规则;勿再各自实现)
247
- ex_idx = build_ex_index(m)
248
- results = []
249
- matched = 0
250
- for sub, pub in rows:
251
- exid = match_ex(sub, ex_idx)
252
- # 基底:medium(supplement)或去碳后加底物(sole)
253
- med2 = dict(medium)
254
- if carbon_mode == "sole":
255
- for rid in list(med2):
256
- if rid.startswith("EX_") and rid in m.reactions:
257
- met = list(m.reactions.get_by_id(rid).metabolites)[0]
258
- if met.formula and "C" in parse_formula(met.formula):
259
- del med2[rid]
260
- if exid and exid in m.reactions:
261
- med2[exid] = -10.0
262
- for r in m.reactions:
263
- if r.id.startswith(EX_PREFIX) or r.boundary:
264
- r.lower_bound = 0.0
265
- for rid, lb in med2.items():
266
- if rid in m.reactions:
267
- m.reactions.get_by_id(rid).lower_bound = lb
268
- with m:
269
- g = m.optimize().objective_value
270
- pred = g > 1e-6
271
- ok = (pred == bool(pub))
272
- if ok:
273
- matched += 1
274
- results.append({"substrate": sub, "published": int(pub), "predicted": int(pred),
275
- "growth": round(g, 6), "exchange": exid or None, "match": bool(ok),
276
- # 阶段A-M4 口径声明(只增):每底物 growth 为单点 FBA 值
277
- "units": "mmol/gDW/h",
278
- "point_value_note": "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定"})
279
- rep = {
280
- "status": "PASS" if rows and matched / len(rows) >= 0.8 else ("WARN" if rows else "SKIP"),
281
- "carbon_mode": carbon_mode,
282
- "matched": matched, "total": len(rows),
283
- "rate": round(matched / len(rows), 4) if rows else None,
284
- "results": results,
285
- }
286
- return rep
287
-
288
- # ------------------------------------------------------------------ G5
289
- def g5_essentiality(self, essential_test, medium, reference_essential=None):
290
- """条件执行:对给定基因列表逐一手工敲除(with m: 循环),输出每个基因的必要性。
291
- 若给 reference_essential(已知必需基因集)→ 对交集算召回。"""
292
- if not essential_test:
293
- return {"status": "SKIP", "reason": "no essential_test gene list provided"}
294
- if not medium:
295
- return {"status": "SKIP", "reason": "G5 needs medium"}
296
- m = self.m
297
- present = [g for g in essential_test if g in m.genes]
298
- if len(present) / len(essential_test) < 0.8:
299
- return {"status": "SKIP", "reason": "gene mapping coverage < 80%",
300
- "present": len(present), "total": len(essential_test)}
301
- results = []
302
- for gid in present:
303
- with m:
304
- for r in m.reactions:
305
- if r.id.startswith(EX_PREFIX) or r.boundary:
306
- r.lower_bound = 0.0
307
- for rid, lb in medium.items():
308
- if rid in m.reactions:
309
- m.reactions.get_by_id(rid).lower_bound = lb
310
- m.genes.get_by_id(gid).knock_out()
311
- g = m.optimize().objective_value
312
- results.append({"gene": gid, "growth": round(g, 6),
313
- "essential": bool(g < 1e-6)})
314
- n_ess = sum(1 for r_ in results if r_["essential"])
315
- recall = None
316
- if reference_essential:
317
- ref = set(reference_essential)
318
- tp = sum(1 for r_ in results if r_["essential"] and r_["gene"] in ref)
319
- recall = round(tp / len(ref), 4) if ref else None
320
- rep = {
321
- "status": "PASS" if (recall is None or recall >= 0.4) else "WARN",
322
- "tested": len(results), "essential_found": n_ess,
323
- "recall_vs_reference": recall,
324
- "details": results,
325
- }
326
- return rep
327
-
328
- # ------------------------------------------------------------------ run
329
- def run(self, medium=None, phenotype_table=None, essential_test=None,
330
- reference_growth=None, reference_essential=None, carbon_mode="supplement",
331
- context=None):
332
- from gapfind import resolve_medium, expand_medium
333
- medium, _preset = expand_medium(medium)
334
- resolved_med, unresolved = resolve_medium(self.m, medium) if medium else ({}, [])
335
- report = {"model": self.path,
336
- "units": {"growth": "mmol/gDW/h", "note": "objective_value 是 FBA 通量(mmol/gDW/h),不是比生长速率 μ(h⁻¹)"},
337
- "g1": self.g1_load(),
338
- "g2": self.g2_balance(),
339
- "g6": self.g6_atp_leak(context=context or {})}
340
- g3 = self.g3_growth(resolved_med, reference_growth)
341
- if unresolved:
342
- g3["medium_unresolved"] = unresolved
343
- report["g3"] = g3
344
- if phenotype_table:
345
- report["g4"] = self.g4_phenotype(table_path=phenotype_table, medium=resolved_med,
346
- carbon_mode=carbon_mode)
347
- else:
348
- report["g4"] = {"status": "SKIP", "reason": "no phenotype reference provided"}
349
- if essential_test:
350
- report["g5"] = self.g5_essentiality(essential_test, resolved_med, reference_essential)
351
- else:
352
- report["g5"] = {"status": "SKIP", "reason": "no essential_test provided"}
353
- # 总判定:G1/G2/G3/G6(FATAL 才 FAIL;G6 WARN 级不阻塞但补洞后必查)
354
- blocked = ["g1", "g2", "g3"]
355
- fails = [k for k in blocked if report[k]["status"] == "FAIL"]
356
- warns = [k for k in blocked if report[k]["status"] == "WARN"]
357
- report["overall"] = "FAIL" if fails else ("WARN" if warns else "PASS")
358
- report["blocking"] = blocked
359
- return report
360
-
361
-
362
- def validate_model(model_path, medium=None, phenotype_table=None,
363
- essential_test=None, reference_growth=None, reference_essential=None,
364
- carbon_mode="supplement"):
365
- v = Validator(model_path)
366
- return v.run(medium=medium, phenotype_table=phenotype_table,
367
- essential_test=essential_test, reference_growth=reference_growth,
368
- reference_essential=reference_essential, carbon_mode=carbon_mode)
369
-
370
-
371
- if __name__ == "__main__":
372
- # 命令行直跑(开发/测试用):python validate.py <model.xml> [--medium-json x] [--table t] [--g5 g1,g2]
373
- import sys
374
- path = sys.argv[1]
375
- med = None
376
- table = None
377
- g5 = None
378
- refg = None
379
- i = 2
380
- while i < len(sys.argv):
381
- if sys.argv[i] == "--medium-json" and i + 1 < len(sys.argv):
382
- med = json.loads(sys.argv[i + 1]); i += 2
383
- elif sys.argv[i] == "--table" and i + 1 < len(sys.argv):
384
- table = sys.argv[i + 1]; i += 2
385
- elif sys.argv[i] == "--g5" and i + 1 < len(sys.argv):
386
- g5 = sys.argv[i + 1].split(","); i += 2
387
- elif sys.argv[i] == "--ref-growth" and i + 1 < len(sys.argv):
388
- refg = float(sys.argv[i + 1]); i += 2
389
- else:
390
- i += 1
391
- rep = validate_model(path, medium=med, phenotype_table=table, essential_test=g5,
392
- reference_growth=refg)
1
+ # validate.py — dsh-bio-gem 五道验证关卡(M1)
2
+ # G1 加载统计 / G2 内部反应元素平衡 / G3 生长真实性 / G4 底物表型(条件) / G5 必需基因抽检(条件)
3
+ # 规格: docs/ARCHITECTURE.md §5;判据口径 = FBA objective_value(mmol/gDW/h,不用 μ)
4
+ # 实现从 HANDOFF-03 五道关卡协议产品化(农杆菌项目验证过的逻辑)
5
+ import re
6
+ import os
7
+ import json
8
+ import sys
9
+ import cobra
10
+
11
+ # Python -I isolated 模式下脚本目录不进 sys.path——显式插入以导入同目录模块
12
+ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
13
+
14
+ EX_PREFIX = ("EX_", "DM_", "SK_")
15
+ CORE_ELEMS = ("C", "N", "P", "S") # 硬核:不平衡必须 = 0
16
+ REPORT_ELEMS = ("H", "O") # 报告不阻塞
17
+ ELM_RE = re.compile(r"([A-Z][a-z]?)(\d*)")
18
+
19
+ # 关卡注册器(GLM 建议 + 2026-08-29 采纳):未来加 G7 不改主流程
20
+ GATE_REGISTRY = {}
21
+
22
+
23
+ def register_gate(name):
24
+ def deco(fn):
25
+ GATE_REGISTRY[name] = fn
26
+ return fn
27
+ return deco
28
+
29
+
30
+ def parse_formula(f):
31
+ """C10H13N5O13P3 -> {"C":10,...}; 忽略 R/X 等通用占位。"""
32
+ d = {}
33
+ if not f:
34
+ return d
35
+ for m in ELM_RE.finditer(f):
36
+ el = m.group(1)
37
+ n = int(m.group(2) or 1)
38
+ d[el] = d.get(el, 0) + n
39
+ return d
40
+
41
+
42
+ def _rxn_elem_balance(rxn):
43
+ """内部反应元素平衡:返回 {elem: delta}(delta=产物-底物,应接近 0)。"""
44
+ bal = {}
45
+ for met, coeff in rxn.metabolites.items():
46
+ if not met.formula:
47
+ continue
48
+ d = parse_formula(met.formula)
49
+ for el, n in d.items():
50
+ bal[el] = bal.get(el, 0) + coeff * n
51
+ return bal
52
+
53
+
54
+ class Validator:
55
+ def __init__(self, model_path):
56
+ from silentio import silent_read_sbml
57
+ self.path = model_path
58
+ self.m = silent_read_sbml(model_path)
59
+
60
+ # ------------------------------------------------------------------ G1
61
+ def g1_load(self):
62
+ m = self.m
63
+ from collections import Counter
64
+ repl = Counter()
65
+ bad_ids = []
66
+ for g in m.genes:
67
+ parts = g.id.split("_")
68
+ if len(parts) >= 3 and parts[0] == "NC":
69
+ repl["_".join(parts[:2])] += 1
70
+ else:
71
+ repl["other"] += 1
72
+ bad_ids.append(g.id)
73
+ n_genes_with_rxn = sum(1 for g in m.genes if len(g.reactions) > 0)
74
+ rep = {
75
+ "status": "PASS" if not bad_ids else "WARN",
76
+ "genes": len(m.genes), "reactions": len(m.reactions),
77
+ "metabolites": len(m.metabolites),
78
+ "replicons": dict(repl),
79
+ "non_nc_gene_ids": bad_ids[:10],
80
+ "gpr_gene_coverage": round(n_genes_with_rxn / len(m.genes), 4) if m.genes else 0,
81
+ }
82
+ return rep
83
+
84
+ # ------------------------------------------------------------------ G2
85
+ def g2_balance(self):
86
+ m = self.m
87
+ internal = [r for r in m.reactions
88
+ if not (r.id.startswith(EX_PREFIX) or r.boundary)]
89
+ formula_coverage = sum(1 for x in m.metabolites if x.formula) / len(m.metabolites)
90
+ bad_core = {} # elem -> [rxn ids]
91
+ bad_report = {}
92
+ checked = 0
93
+ for r in internal:
94
+ if not all(x.formula for x in r.metabolites):
95
+ continue # 公式缺失不计入不平衡(先报覆盖率)
96
+ checked += 1
97
+ bal = _rxn_elem_balance(r)
98
+ for el in CORE_ELEMS:
99
+ v = bal.get(el, 0)
100
+ if abs(v) > 1e-6:
101
+ bad_core.setdefault(el, []).append(r.id)
102
+ for el in REPORT_ELEMS:
103
+ v = bal.get(el, 0)
104
+ if abs(v) > 2: # charged 公式惯例噪声容忍 ±2
105
+ bad_report.setdefault(el, []).append(r.id)
106
+ n_bad = sum(len(v) for v in bad_core.values())
107
+ frac = 1.0 - n_bad / checked if checked else 0.0
108
+ status = "PASS" if n_bad == 0 else ("WARN" if frac >= 0.85 else "FAIL")
109
+
110
+ # 2026-09-11 修复(agent 在真实 E2E 中发现并指出):**公式覆盖率是 PASS 结论的
111
+ # 作用域上界** —— 覆盖率低时「无反应不平衡」只说明被检查的那部分没问题,不能
112
+ # 外推为整体 PASS。实测 iNX1344_v3:覆盖率 68.35% 却判 PASS,agent 据此指出
113
+ # 「g2 的 PASS 是假阳性,因为它只检查了有 formula 的 68.35% 代谢物」。
114
+ # 该结论会误导后续判断(例如「配平没问题」),故低覆盖时降级为 WARN。
115
+ G2_COVERAGE_GATE = 0.90
116
+ unchecked = len(internal) - checked
117
+ coverage_scope_note = None
118
+ if status == "PASS" and formula_coverage < G2_COVERAGE_GATE:
119
+ status = "WARN"
120
+ coverage_scope_note = (
121
+ f"元素平衡结论只覆盖 {formula_coverage * 100:.2f}% 代谢物、"
122
+ f"{checked}/{len(internal)} 条内部反应可判定(跳过 {unchecked} 条)"
123
+ f"→ 不足以判 PASS。补全代谢物 formula 后必须重跑本关;"
124
+ f"在此之前不要据本关结论断言「配平没问题」"
125
+ )
126
+
127
+ rep = {
128
+ "status": status,
129
+ "internal_reactions": len(internal), "formula_checked": checked,
130
+ "formula_unchecked_reactions": unchecked,
131
+ "metabolite_formula_coverage": round(formula_coverage, 4),
132
+ "coverage_gate": G2_COVERAGE_GATE,
133
+ "core_unbalanced": {k: len(v) for k, v in bad_core.items()},
134
+ "core_unbalanced_examples": {k: v[:5] for k, v in bad_core.items()},
135
+ "h_o_report": {k: len(v) for k, v in bad_report.items()},
136
+ "core_balance_frac": round(frac, 4),
137
+ # P0-2(2026-08-31 LBA9402 会话实测):agent 会把 0.9985 心算换算成百分比而被防火墙拦——直接给原始百分数字段
138
+ "core_balance_frac_pct": round(frac * 100, 2),
139
+ }
140
+ if coverage_scope_note:
141
+ rep["coverage_scope_note"] = coverage_scope_note
142
+ return rep
143
+
144
+ # ------------------------------------------------------------------ G3
145
+ def g3_growth(self, medium, reference_growth=None):
146
+ """medium: {EX_id: lower_bound}; 三态:medium / no-carbon / all-closed。"""
147
+ m = self.m
148
+
149
+ def _setup(medium_dict):
150
+ for r in m.reactions:
151
+ if r.id.startswith(EX_PREFIX) or r.boundary:
152
+ r.lower_bound = 0.0
153
+ for rid, lb in (medium_dict or {}).items():
154
+ if rid in m.reactions:
155
+ m.reactions.get_by_id(rid).lower_bound = lb
156
+ else:
157
+ return rid
158
+ return None
159
+
160
+ miss = _setup(medium or {})
161
+ if miss:
162
+ return {"status": "FAIL", "reason": f"medium exchange not in model: {miss}",
163
+ "medium_provided": bool(medium)}
164
+ with m:
165
+ wt = m.optimize().objective_value
166
+ # no-carbon: 去掉 formula 含 C 的交换
167
+ no_c_medium = dict(medium or {})
168
+ for rid in list(no_c_medium):
169
+ if rid.startswith("EX_") and rid in m.reactions:
170
+ met = list(m.reactions.get_by_id(rid).metabolites)[0]
171
+ if met.formula and "C" in parse_formula(met.formula):
172
+ del no_c_medium[rid]
173
+ miss = _setup(no_c_medium)
174
+ with m:
175
+ g_no_c = m.optimize().objective_value
176
+ miss = _setup({}) # all closed
177
+ with m:
178
+ g_closed = m.optimize().objective_value
179
+ ok_grow = wt > 1e-6
180
+ ok_noc = abs(g_no_c) < 1e-6
181
+ ok_closed = abs(g_closed) < 1e-6
182
+ ratio = None
183
+ if reference_growth and reference_growth > 0:
184
+ ratio = wt / reference_growth
185
+ if not medium:
186
+ status = "WARN" # 无声明培养基 -> 无法验证
187
+ elif ok_grow and ok_noc and ok_closed and (ratio is None or ratio >= 0.99):
188
+ status = "PASS"
189
+ else:
190
+ status = "FAIL" # 有培养基但生长不达标(或对照泄漏)——构建侧必须走补洞闭环
191
+ rep = {
192
+ "status": status,
193
+ "medium_provided": bool(medium),
194
+ "growth_medium": round(wt, 6),
195
+ "growth_no_carbon": round(g_no_c, 6),
196
+ "growth_all_closed": round(g_closed, 6),
197
+ "ratio_vs_reference": round(ratio, 4) if ratio is not None else None,
198
+ "checks": {"medium>0": ok_grow, "no_carbon==0": ok_noc, "closed==0": ok_closed},
199
+ # 阶段A-M4 口径声明(只增):单点 FBA 值非硬结论
200
+ "units": "mmol/gDW/h",
201
+ "point_value_note": "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定",
202
+ }
203
+ return rep
204
+
205
+ # ------------------------------------------------------------------ G6
206
+ @register_gate("G6")
207
+ def g6_atp_leak(self, context=None):
208
+ """ATP 泄漏测试(MEMOTE 核心测试;G3 all-closed 的必要不充分检查):
209
+ 全关交换后最大化 ATP 代谢物的净消耗(demand),通量 > 0.01 → WARN。
210
+ 补洞后必跑(context.post_gapfill 时不再跳过)。
211
+ P1-5 修复(2026-08-31 LBA9402 会话实测):CarveMe 模型 ATP id 为 M_atp_c,
212
+ 旧匹配只看 atp_c/cpd00002_c0 -> 误 SKIP「未找到 ATP」——扩展命名模式 + SKIP 时列出尝试模式与模型内候选。
213
+ P2-9 修复(2026-09-10 iNX1344 E2E 实测):MetaCyc/BioCyc 导出模型 ATP 为 M00002_c
214
+ (name='ATP',formula 为去质子化变体),仍不在模式表内 → 二次误 SKIP(G6 直接失效)。
215
+ 改为三级通用解析:id 模式 → name 匹配 → formula 匹配,跨 ID 体系自适应。"""
216
+ m = self.m
217
+ atp_patterns = ("atp_c", "cpd00002_c0", "m_atp_c", "atp_c0", "cpd00002", "atp",
218
+ "m00002", "m00002_c", "m00002_c0")
219
+ cands = [x for x in m.metabolites if (x.id or "").lower() in atp_patterns]
220
+ atp_source = "id_pattern"
221
+ if not cands:
222
+ # 回退 1:name 匹配(跨 ID 体系最稳的信号;排除 dATP 等衍生物)
223
+ cands = [x for x in m.metabolites
224
+ if (x.name or "").strip().upper() == "ATP"
225
+ or ("atp" in (x.name or "").lower() and "datp" not in (x.name or "").lower())]
226
+ atp_source = "name_match"
227
+ if not cands:
228
+ # 回退 2:分子式(含去质子化变体——部分模型 formula 非标准形式)
229
+ cands = [x for x in m.metabolites
230
+ if (x.formula or "").replace(" ", "") in ("C10H16N5O13P3", "C10H12N5O13P3")]
231
+ atp_source = "formula_match"
232
+ cyto = [x for x in cands if x.compartment in ("c0", "c")]
233
+ atp_c = (cyto or cands or [None])[0]
234
+ if atp_c is None:
235
+ atp_like = sorted({x.id for x in m.metabolites
236
+ if "atp" in (x.id or "").lower() or "atp" in (x.name or "").lower()})[:10]
237
+ return {"status": "SKIP",
238
+ "reason": "未找到 ATP 代谢物(id 模式 / name 匹配 / formula 匹配三级回退均未命中)",
239
+ "tried_patterns": list(atp_patterns),
240
+ "atp_like_ids_in_model": atp_like,
241
+ "note": "SKIP 系命名口径未命中(非模型缺陷证明);若模型含 ATP 但三级回退均未命中,"
242
+ "请在模型内显式标注 atp 角色后重跑"}
243
+ dm = cobra.Reaction("DM_gem_atp_leak", name="G6 ATP 泄漏检测 demand",
244
+ lower_bound=0.0, upper_bound=1000.0)
245
+ dm.add_metabolites({atp_c: -1})
246
+ m.add_reactions([dm])
247
+ try:
248
+ with m:
249
+ for r in m.reactions:
250
+ if r.id.startswith(EX_PREFIX) or r.boundary:
251
+ r.lower_bound = 0.0
252
+ v = m.optimize().objective_value
253
+ finally:
254
+ m.remove_reactions([dm])
255
+ leak = abs(v)
256
+ status = "PASS" if leak <= 0.01 else "WARN"
257
+ return {"status": status, "atp_leak_flux": round(leak, 6),
258
+ "atp_metabolite_found": atp_c.id, "atp_resolved_by": atp_source,
259
+ "threshold": 0.01, "post_gapfill": bool((context or {}).get("post_gapfill")),
260
+ "note": "全关交换后 ATP demand 通量应≈0;>0.01 提示能量循环泄漏(L3 MILP 补洞最可能引入)"}
261
+
262
+ # ------------------------------------------------------------------ G4
263
+ def g4_phenotype(self, table_path=None, substrates=None, medium=None, carbon_mode="supplement"):
264
+ """条件执行:需参照表(TSV: substrate<TAB>published 0/1)或 substrates+published。
265
+ carbon_mode: supplement=基准培养基不变+底物-10(对齐 HANDOFF-03 关卡4 基线 16/19→17/19);
266
+ sole=去含碳交换后底物-10(唯一碳源严格语义,氮源类测试会误判)。"""
267
+ if table_path and os.path.exists(table_path):
268
+ rows = []
269
+ with open(table_path, encoding="utf-8") as f:
270
+ for line in f:
271
+ line = line.rstrip("\r\n")
272
+ if not line or line.startswith("#") or line.startswith("substrate"):
273
+ continue
274
+ p = line.split("\t")
275
+ if len(p) >= 2:
276
+ rows.append((p[0].strip(), int(float(p[1]))))
277
+ elif substrates:
278
+ rows = substrates
279
+ else:
280
+ return {"status": "SKIP", "reason": "no phenotype reference provided"}
281
+ if not medium:
282
+ return {"status": "SKIP", "reason": "G4 needs medium to define base"}
283
+ from gapfind import build_ex_index, match_ex, SYN
284
+ m = self.m
285
+ # 统一走 gapfind 的 build_ex_index + match_ex(修复过子串误配规则;勿再各自实现)
286
+ ex_idx = build_ex_index(m)
287
+ results = []
288
+ matched = 0
289
+ for sub, pub in rows:
290
+ exid = match_ex(sub, ex_idx)
291
+ # 基底:medium(supplement)或去碳后加底物(sole)
292
+ med2 = dict(medium)
293
+ if carbon_mode == "sole":
294
+ for rid in list(med2):
295
+ if rid.startswith("EX_") and rid in m.reactions:
296
+ met = list(m.reactions.get_by_id(rid).metabolites)[0]
297
+ if met.formula and "C" in parse_formula(met.formula):
298
+ del med2[rid]
299
+ if exid and exid in m.reactions:
300
+ med2[exid] = -10.0
301
+ for r in m.reactions:
302
+ if r.id.startswith(EX_PREFIX) or r.boundary:
303
+ r.lower_bound = 0.0
304
+ for rid, lb in med2.items():
305
+ if rid in m.reactions:
306
+ m.reactions.get_by_id(rid).lower_bound = lb
307
+ with m:
308
+ g = m.optimize().objective_value
309
+ pred = g > 1e-6
310
+ ok = (pred == bool(pub))
311
+ if ok:
312
+ matched += 1
313
+ results.append({"substrate": sub, "published": int(pub), "predicted": int(pred),
314
+ "growth": round(g, 6), "exchange": exid or None, "match": bool(ok),
315
+ # 阶段A-M4 口径声明(只增):每底物 growth 为单点 FBA 值
316
+ "units": "mmol/gDW/h",
317
+ "point_value_note": "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定"})
318
+ rep = {
319
+ "status": "PASS" if rows and matched / len(rows) >= 0.8 else ("WARN" if rows else "SKIP"),
320
+ "carbon_mode": carbon_mode,
321
+ "matched": matched, "total": len(rows),
322
+ "rate": round(matched / len(rows), 4) if rows else None,
323
+ "results": results,
324
+ }
325
+ return rep
326
+
327
+ # ------------------------------------------------------------------ G5
328
+ def g5_essentiality(self, essential_test, medium, reference_essential=None):
329
+ """条件执行:对给定基因列表逐一手工敲除(with m: 循环),输出每个基因的必要性。
330
+ 若给 reference_essential(已知必需基因集)→ 对交集算召回。"""
331
+ if not essential_test:
332
+ return {"status": "SKIP", "reason": "no essential_test gene list provided"}
333
+ if not medium:
334
+ return {"status": "SKIP", "reason": "G5 needs medium"}
335
+ m = self.m
336
+ present = [g for g in essential_test if g in m.genes]
337
+ if len(present) / len(essential_test) < 0.8:
338
+ return {"status": "SKIP", "reason": "gene mapping coverage < 80%",
339
+ "present": len(present), "total": len(essential_test)}
340
+ results = []
341
+ for gid in present:
342
+ with m:
343
+ for r in m.reactions:
344
+ if r.id.startswith(EX_PREFIX) or r.boundary:
345
+ r.lower_bound = 0.0
346
+ for rid, lb in medium.items():
347
+ if rid in m.reactions:
348
+ m.reactions.get_by_id(rid).lower_bound = lb
349
+ m.genes.get_by_id(gid).knock_out()
350
+ g = m.optimize().objective_value
351
+ results.append({"gene": gid, "growth": round(g, 6),
352
+ "essential": bool(g < 1e-6)})
353
+ n_ess = sum(1 for r_ in results if r_["essential"])
354
+ recall = None
355
+ if reference_essential:
356
+ ref = set(reference_essential)
357
+ tp = sum(1 for r_ in results if r_["essential"] and r_["gene"] in ref)
358
+ recall = round(tp / len(ref), 4) if ref else None
359
+ rep = {
360
+ "status": "PASS" if (recall is None or recall >= 0.4) else "WARN",
361
+ "tested": len(results), "essential_found": n_ess,
362
+ "recall_vs_reference": recall,
363
+ "details": results,
364
+ }
365
+ return rep
366
+
367
+ # ------------------------------------------------------------------ run
368
+ def run(self, medium=None, phenotype_table=None, essential_test=None,
369
+ reference_growth=None, reference_essential=None, carbon_mode="supplement",
370
+ context=None):
371
+ from gapfind import resolve_medium, expand_medium
372
+ from coherence import model_coherence
373
+ medium, _preset = expand_medium(medium)
374
+ resolved_med, unresolved = resolve_medium(self.m, medium) if medium else ({}, [])
375
+ report = {"model": self.path,
376
+ "units": {"growth": "mmol/gDW/h", "note": "objective_value 是 FBA 通量(mmol/gDW/h),不是比生长速率 μ(h⁻¹)"},
377
+ # G0 模型数据质量前置诊断:未映射前体等数据问题会让 G2/G3 与 gapfind 的
378
+ # 结论失真(实测 iNX1344_v3:gapfind 报的 5 个 L3 缺口实为未映射前体所致)
379
+ "g0": model_coherence(self.m),
380
+ "g1": self.g1_load(),
381
+ "g2": self.g2_balance(),
382
+ "g6": self.g6_atp_leak(context=context or {})}
383
+ g3 = self.g3_growth(resolved_med, reference_growth)
384
+ if unresolved:
385
+ g3["medium_unresolved"] = unresolved
386
+ report["g3"] = g3
387
+ if phenotype_table:
388
+ report["g4"] = self.g4_phenotype(table_path=phenotype_table, medium=resolved_med,
389
+ carbon_mode=carbon_mode)
390
+ else:
391
+ report["g4"] = {"status": "SKIP", "reason": "no phenotype reference provided"}
392
+ if essential_test:
393
+ report["g5"] = self.g5_essentiality(essential_test, resolved_med, reference_essential)
394
+ else:
395
+ report["g5"] = {"status": "SKIP", "reason": "no essential_test provided"}
396
+ # 总判定:G1/G2/G3/G6(FATAL 才 FAIL;G6 WARN 级不阻塞但补洞后必查)
397
+ blocked = ["g1", "g2", "g3"]
398
+ fails = [k for k in blocked if report[k]["status"] == "FAIL"]
399
+ warns = [k for k in blocked if report[k]["status"] == "WARN"]
400
+ report["overall"] = "FAIL" if fails else ("WARN" if warns else "PASS")
401
+ report["blocking"] = blocked
402
+ return report
403
+
404
+
405
+ def validate_model(model_path, medium=None, phenotype_table=None,
406
+ essential_test=None, reference_growth=None, reference_essential=None,
407
+ carbon_mode="supplement"):
408
+ v = Validator(model_path)
409
+ return v.run(medium=medium, phenotype_table=phenotype_table,
410
+ essential_test=essential_test, reference_growth=reference_growth,
411
+ reference_essential=reference_essential, carbon_mode=carbon_mode)
412
+
413
+
414
+ if __name__ == "__main__":
415
+ # 命令行直跑(开发/测试用):python validate.py <model.xml> [--medium-json x] [--table t] [--g5 g1,g2]
416
+ import sys
417
+ path = sys.argv[1]
418
+ med = None
419
+ table = None
420
+ g5 = None
421
+ refg = None
422
+ i = 2
423
+ while i < len(sys.argv):
424
+ if sys.argv[i] == "--medium-json" and i + 1 < len(sys.argv):
425
+ med = json.loads(sys.argv[i + 1]); i += 2
426
+ elif sys.argv[i] == "--table" and i + 1 < len(sys.argv):
427
+ table = sys.argv[i + 1]; i += 2
428
+ elif sys.argv[i] == "--g5" and i + 1 < len(sys.argv):
429
+ g5 = sys.argv[i + 1].split(","); i += 2
430
+ elif sys.argv[i] == "--ref-growth" and i + 1 < len(sys.argv):
431
+ refg = float(sys.argv[i + 1]); i += 2
432
+ else:
433
+ i += 1
434
+ rep = validate_model(path, medium=med, phenotype_table=table, essential_test=g5,
435
+ reference_growth=refg)
393
436
  print(json.dumps(rep, ensure_ascii=False, indent=2))