@dsh-bio/dsh-bio-gem 0.1.3 → 0.1.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/python/gem_ops.py CHANGED
@@ -309,6 +309,48 @@ def op_sensitivity(args):
309
309
  export_csv=args.get("export_csv"), baseline_check=baseline_check)}
310
310
 
311
311
 
312
+ # ---------------------------------------------------------------------------
313
+ # op: quality — gem-qi-v1 模型质量摘要(只读)
314
+ # ---------------------------------------------------------------------------
315
+ def op_quality(args):
316
+ from quality import quality_report
317
+ model = args.get("model")
318
+ if not model or not os.path.exists(model):
319
+ return {"ok": False, "error": f"model file not found: {model}"}
320
+ try:
321
+ result = quality_report(
322
+ model, medium=args.get("medium"), checks=args.get("checks"),
323
+ export_csv=args.get("export_csv"))
324
+ except (ValueError, OSError) as e:
325
+ return {"ok": False, "error": str(e)}
326
+ return {"ok": True, "result": result}
327
+
328
+
329
+ # ---------------------------------------------------------------------------
330
+ # op: sample — COBRA 通量空间采样(默认 Windows-safe ACHR)
331
+ # ---------------------------------------------------------------------------
332
+ def op_sample(args):
333
+ from sampling import sample_fluxes
334
+ model = args.get("model")
335
+ if not model or not os.path.exists(model):
336
+ return {"ok": False, "error": f"model file not found: {model}"}
337
+ try:
338
+ result = sample_fluxes(
339
+ model,
340
+ medium=args.get("medium"),
341
+ n=args.get("n", 1000),
342
+ method=args.get("method", "auto"),
343
+ thinning=args.get("thinning", 100),
344
+ growth_floor_fraction=args.get("growth_floor_fraction"),
345
+ reactions=args.get("reactions"),
346
+ seed=args.get("seed", 42),
347
+ export_csv=args.get("export_csv"),
348
+ )
349
+ except (ValueError, OSError) as e:
350
+ return {"ok": False, "error": str(e)}
351
+ return {"ok": True, "result": result}
352
+
353
+
312
354
  # ---------------------------------------------------------------------------
313
355
  # 分发器
314
356
  # ---------------------------------------------------------------------------
@@ -323,6 +365,8 @@ OPS = {
323
365
  "annotate": op_annotate,
324
366
  "media_resolve": op_media_resolve,
325
367
  "l3_fix": op_l3_fix,
368
+ "quality": op_quality,
369
+ "sample": op_sample,
326
370
  }
327
371
 
328
372
 
@@ -486,6 +530,25 @@ def op_targets(args):
486
530
  OPS["targets"] = op_targets
487
531
 
488
532
 
533
+ # ---------------------------------------------------------------------------
534
+ # op: precursor_scan — 阻塞前体分析
535
+ # 「模型为什么不长」的结构级定位:逐前体做移除测试,找出卡住生长的前体。
536
+ # 来源:2026-09-11 E2E 绕道归因(agent 手写该逻辑 6+ 次,无工具可用)。
537
+ # 判据刻意用「相对判断」而非绝对可达性 —— 对可生长模型天然零误报。
538
+ # ---------------------------------------------------------------------------
539
+ def op_precursor_scan(args):
540
+ from precursor_scan import scan_precursors
541
+ model = args.get("model")
542
+ if not model or not os.path.exists(model):
543
+ return {"ok": False, "error": f"model file not found: {model}"}
544
+ return {"ok": True, "result": scan_precursors(
545
+ model, medium=args.get("medium"),
546
+ max_precursors=args.get("max_precursors", 200))}
547
+
548
+
549
+ OPS["precursor_scan"] = op_precursor_scan
550
+
551
+
489
552
 
490
553
  def main():
491
554
  line = sys.stdin.read()
@@ -517,4 +580,4 @@ def main():
517
580
 
518
581
 
519
582
  if __name__ == "__main__":
520
- main()
583
+ main()
package/python/l3_fix.py CHANGED
@@ -417,7 +417,7 @@ def l3_fix(model_path, medium=None, substrates=None, out=None,
417
417
  if exid and exid in m.reactions and g < 1e-6:
418
418
  l3.append({"substrate": sub, "exchange": exid, "growth_sole_before": round(g, 6),
419
419
  # 阶段A-M4 口径声明(只增)
420
- "units": "mmol/gDW/h",
420
+ "units": "1/h",
421
421
  "point_value_note": "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定"})
422
422
 
423
423
  # 第五闸门(入口预检: 每底物至少 1 条新增预估)
@@ -493,7 +493,7 @@ def l3_fix(model_path, medium=None, substrates=None, out=None,
493
493
  "evidence": "EVIDENCE_math"})
494
494
  growth_a = _growth_sole(cur, resolved_med, exid)
495
495
  l3a["growth_sole_after"] = round(growth_a, 6)
496
- l3a["units"] = "mmol/gDW/h"
496
+ l3a["units"] = "1/h"
497
497
  l3a["point_value_note"] = "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定"
498
498
  entry["l3a"] = l3a
499
499
 
@@ -556,12 +556,12 @@ def l3_fix(model_path, medium=None, substrates=None, out=None,
556
556
  _note(f"[l3b] {sub}: picked {len(picked)}, added {len(added_here)}")
557
557
  growth_b = _growth_sole(cur, resolved_med, exid)
558
558
  l3b["growth_sole_after"] = round(growth_b, 6)
559
- l3b["units"] = "mmol/gDW/h"
559
+ l3b["units"] = "1/h"
560
560
  l3b["point_value_note"] = "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定"
561
561
  l3b["added"] = added_here
562
562
  entry["l3b"] = l3b
563
563
  entry["growth_sole_after"] = round(max(growth_a, growth_b), 6)
564
- entry["units"] = "mmol/gDW/h"
564
+ entry["units"] = "1/h"
565
565
  entry["point_value_note"] = "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定"
566
566
  entry["verdict"] = "fixed" if growth_b > 1e-6 else "not_fixable"
567
567
  if entry["verdict"] == "not_fixable":
package/python/ledger.py CHANGED
@@ -401,7 +401,7 @@ def register_phenotype(model_path, g4_results, condition=None, lineage_version=N
401
401
  "type": "phenotype",
402
402
  "content": (f"底物 {sub} 预测{'生长' if r.get('predicted') else '不生长'}"
403
403
  f"(文献={r.get('published')},匹配={r.get('match')},"
404
- f"growth={r.get('growth')} mmol/gDW/h)"),
404
+ f"growth={r.get('growth')} 1/h)"),
405
405
  "model": model_path,
406
406
  "model_lineage_version": lineage_version,
407
407
  "condition": condition,
@@ -3,14 +3,15 @@
3
3
  # / set_verified_phenotypes(phenotype 结果)/ set_essential_genes(必需基因 + 证据分级)
4
4
  # 纪律: 各工具完成后**仅当产物模型旁已有 card** 才向后追加;无卡不动(不凭空造卡)。
5
5
  # 兼容: build.py 旧卡(无 schema 字段)读取时即时迁移到 v2(新增字段缺失不报错)。
6
- # units: growth_rate 一律 mmol/gDW/h(schema v2 规定,勿用 1/h)。
6
+ # units: growth_rate 一律 1/h(比生长速率;biomass 反应 gDW 归一化口径,数值 = μ)。
7
+ # 2026-09-21 由 mmol/gDW/h 演进为 1/h:数值不变、生物语义更准(外部评审 P0;schema 向后兼容,旧卡照读)。
7
8
  import os
8
9
  import json
9
10
  import time
10
11
 
11
12
  CARD_SUFFIX = ".card.json"
12
13
  SCHEMA = "v2"
13
- GROWTH_UNITS = "mmol/gDW/h"
14
+ GROWTH_UNITS = "1/h"
14
15
 
15
16
 
16
17
  def _now():
@@ -0,0 +1,127 @@
1
+ # precursor_scan.py — dsh-bio-gem 阻塞前体分析(「模型为什么不长」的结构级定位)
2
+ #
3
+ # 来源(2026-09-11 E2E 绕道归因):agent 在不生长模型上反复手写「逐前体探测」逻辑
4
+ # (6+ 次调用),本模块把它固化为工具。
5
+ #
6
+ # ⚠️ 判据设计——**刻意不用「绝对可达性」**:
7
+ # 初版曾用「全开交换下逐前体 demand 能否净生产」,在教科书模型 e_coli_core 上
8
+ # 把 atp_c / accoa_c / nad_c / nadph_c 误报为「既不能合成也不能摄取」。原因是
9
+ # 辅因子有循环补给路径,稳态下**不净生产 ≠ 网络不能供给**。该判据已否决。
10
+ #
11
+ # 现用「**移除测试**」这一相对判据:
12
+ # 基线 biomass 通量 > 0 → 直接返回「无阻塞」(对健康模型天然零误报)
13
+ # 基线 biomass 通量 = 0 → 逐一移除某个前体的需求,看 biomass 是否恢复通量
14
+ # 恢复了 → 该前体即阻塞点(结论由「恢复与否」直接定义,
15
+ # 不依赖对网络的任何绝对判断)
16
+ # 两个验证锚:iNX1344_v3(不生长,应报出阻塞前体)+ e_coli_core(能生长,应报无阻塞)。
17
+ import os
18
+ import sys
19
+
20
+ import cobra # noqa: E402
21
+
22
+ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
23
+
24
+ EPS = 1e-6
25
+
26
+
27
+ def _find_biomass(model):
28
+ """定位 biomass 反应:先按 id/name 命中,再退回 objective 变量。"""
29
+ for r in model.reactions:
30
+ if "biomass" in f"{r.id} {r.name or ''}".lower():
31
+ return r
32
+ try:
33
+ syms = [s.name for s in model.objective.expression.free_symbols]
34
+ except Exception: # noqa: BLE001
35
+ syms = []
36
+ live = [r for r in model.reactions if r.id in syms]
37
+ return live[0] if live else None
38
+
39
+
40
+ def scan_precursors(model_path, medium=None, max_precursors=200):
41
+ """阻塞前体分析。
42
+
43
+ medium: 可选,{EX_id: lower_bound} 或 {"medium_name": "AB"/"M9"}(走 gapfind 的解析器)。
44
+ 返回 {verdict, baseline_flux, blocking_precursors[], note}。
45
+ """
46
+ from silentio import silent_read_sbml
47
+ from gapfind import expand_medium, resolve_medium
48
+
49
+ m = silent_read_sbml(model_path)
50
+ bio = _find_biomass(m)
51
+ if bio is None:
52
+ return {"error": "未定位到 biomass 反应(id/name 与 objective 均未命中)"}
53
+
54
+ medium_applied = 0
55
+ unresolved = []
56
+ if medium:
57
+ med, _preset = expand_medium(medium)
58
+ resolved, unresolved = resolve_medium(m, med)
59
+ with m:
60
+ for rid, lb in resolved.items():
61
+ if rid in m.reactions:
62
+ m.reactions.get_by_id(rid).lower_bound = float(lb)
63
+ medium_applied += 1
64
+ baseline = m.slim_optimize()
65
+ else:
66
+ baseline = m.slim_optimize()
67
+ baseline = float(baseline) if baseline is not None else 0.0
68
+
69
+ out = {
70
+ "model": model_path,
71
+ "biomass_reaction": bio.id,
72
+ "biomass_name": bio.name or "",
73
+ "medium_applied_exchanges": medium_applied,
74
+ "medium_unresolved": unresolved,
75
+ "baseline_flux": round(baseline, 6),
76
+ "units": "1/h",
77
+ "point_value_note": "单点 FBA 值,非解空间硬结论",
78
+ "blocking_precursors": [],
79
+ "n_blocking": 0,
80
+ "verdict": "",
81
+ "note": "",
82
+ }
83
+
84
+ # 健康路径:能生长 → 不做逐前体测试(这是零误报的关键)
85
+ if baseline > EPS:
86
+ out["verdict"] = "growable"
87
+ out["note"] = (
88
+ "该条件下 biomass 可携带通量 → **无阻塞前体**。已刻意跳过逐前体测试,"
89
+ "避免对可生长模型产生假阳性。扰动/区间分析请用 gem_fluxscan、gem_sensitivity。"
90
+ )
91
+ return out
92
+
93
+ precursors = [k for k, v in bio.metabolites.items() if v < 0][:max_precursors]
94
+ blocking = []
95
+ for met in precursors:
96
+ coeff = bio.metabolites[met]
97
+ with m:
98
+ bio.add_metabolites({met: -coeff}) # 系数清零 = 移除该前体需求
99
+ val = m.slim_optimize()
100
+ val = float(val) if val is not None else 0.0
101
+ if val > EPS:
102
+ blocking.append({
103
+ "metabolite": met.id,
104
+ "name": met.name or "",
105
+ "formula": met.formula or "",
106
+ "coefficient": round(coeff, 4),
107
+ "flux_without_it": round(val, 6),
108
+ })
109
+
110
+ out["blocking_precursors"] = blocking
111
+ out["n_blocking"] = len(blocking)
112
+ out["n_precursors_tested"] = len(precursors)
113
+ if blocking:
114
+ out["verdict"] = "blocked"
115
+ out["note"] = (
116
+ "移除下列前体后 biomass 恢复携带通量 → 它们是**阻塞点**。"
117
+ "排查顺序:① 该前体在模型里是否有合成路径(路径缺失=需补反应);"
118
+ "② 是否被边界/约束卡住;③ 计量或命名是否有问题"
119
+ "(先用 gem_validate 的 g0 数据质量诊断与 g2 配平结论交叉读取)。"
120
+ )
121
+ else:
122
+ out["verdict"] = "infeasible_or_constrained"
123
+ out["note"] = (
124
+ "逐前体移除均未恢复通量 → 阻塞不在单个前体上。可能是 biomass 方程整体计量/方向问题、"
125
+ "约束冲突或能量项缺失;建议先看 gem_validate 的 g0(数据质量)与 g2(配平)。"
126
+ )
127
+ return out