@dsh-bio/dsh-bio-gem 0.1.3 → 0.1.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +60 -8
- package/docs/ARCHITECTURE.md +145 -116
- package/docs/DECISIONS-2026-09-21.md +76 -0
- package/docs/releases/v0.1.12.md +43 -0
- package/package.json +5 -3
- package/python/benchmark.py +3 -3
- package/python/biomass_tools.py +2 -2
- package/python/bootstrap_carveme.py +259 -0
- package/python/build.py +10 -2
- package/python/coherence.py +159 -0
- package/python/double_knockout.py +1 -1
- package/python/essential_scan.py +1 -1
- package/python/gapfind.py +413 -396
- package/python/gem_ops.py +64 -1
- package/python/l3_fix.py +4 -4
- package/python/ledger.py +1 -1
- package/python/model_card.py +3 -2
- package/python/precursor_scan.py +127 -0
- package/python/quality.py +577 -0
- package/python/sampling.py +269 -0
- package/python/sensitivity.py +2 -2
- package/python/validate.py +435 -409
- package/skills/gem-expert.md +3 -1
- package/src/capabilities.js +152 -0
- package/src/index.js +21 -2
- package/src/integration.js +507 -0
- package/src/jobs.js +5 -21
- package/src/python.js +113 -17
- package/src/tools.js +98 -22
package/python/gem_ops.py
CHANGED
|
@@ -309,6 +309,48 @@ def op_sensitivity(args):
|
|
|
309
309
|
export_csv=args.get("export_csv"), baseline_check=baseline_check)}
|
|
310
310
|
|
|
311
311
|
|
|
312
|
+
# ---------------------------------------------------------------------------
|
|
313
|
+
# op: quality — gem-qi-v1 模型质量摘要(只读)
|
|
314
|
+
# ---------------------------------------------------------------------------
|
|
315
|
+
def op_quality(args):
|
|
316
|
+
from quality import quality_report
|
|
317
|
+
model = args.get("model")
|
|
318
|
+
if not model or not os.path.exists(model):
|
|
319
|
+
return {"ok": False, "error": f"model file not found: {model}"}
|
|
320
|
+
try:
|
|
321
|
+
result = quality_report(
|
|
322
|
+
model, medium=args.get("medium"), checks=args.get("checks"),
|
|
323
|
+
export_csv=args.get("export_csv"))
|
|
324
|
+
except (ValueError, OSError) as e:
|
|
325
|
+
return {"ok": False, "error": str(e)}
|
|
326
|
+
return {"ok": True, "result": result}
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
# ---------------------------------------------------------------------------
|
|
330
|
+
# op: sample — COBRA 通量空间采样(默认 Windows-safe ACHR)
|
|
331
|
+
# ---------------------------------------------------------------------------
|
|
332
|
+
def op_sample(args):
|
|
333
|
+
from sampling import sample_fluxes
|
|
334
|
+
model = args.get("model")
|
|
335
|
+
if not model or not os.path.exists(model):
|
|
336
|
+
return {"ok": False, "error": f"model file not found: {model}"}
|
|
337
|
+
try:
|
|
338
|
+
result = sample_fluxes(
|
|
339
|
+
model,
|
|
340
|
+
medium=args.get("medium"),
|
|
341
|
+
n=args.get("n", 1000),
|
|
342
|
+
method=args.get("method", "auto"),
|
|
343
|
+
thinning=args.get("thinning", 100),
|
|
344
|
+
growth_floor_fraction=args.get("growth_floor_fraction"),
|
|
345
|
+
reactions=args.get("reactions"),
|
|
346
|
+
seed=args.get("seed", 42),
|
|
347
|
+
export_csv=args.get("export_csv"),
|
|
348
|
+
)
|
|
349
|
+
except (ValueError, OSError) as e:
|
|
350
|
+
return {"ok": False, "error": str(e)}
|
|
351
|
+
return {"ok": True, "result": result}
|
|
352
|
+
|
|
353
|
+
|
|
312
354
|
# ---------------------------------------------------------------------------
|
|
313
355
|
# 分发器
|
|
314
356
|
# ---------------------------------------------------------------------------
|
|
@@ -323,6 +365,8 @@ OPS = {
|
|
|
323
365
|
"annotate": op_annotate,
|
|
324
366
|
"media_resolve": op_media_resolve,
|
|
325
367
|
"l3_fix": op_l3_fix,
|
|
368
|
+
"quality": op_quality,
|
|
369
|
+
"sample": op_sample,
|
|
326
370
|
}
|
|
327
371
|
|
|
328
372
|
|
|
@@ -486,6 +530,25 @@ def op_targets(args):
|
|
|
486
530
|
OPS["targets"] = op_targets
|
|
487
531
|
|
|
488
532
|
|
|
533
|
+
# ---------------------------------------------------------------------------
|
|
534
|
+
# op: precursor_scan — 阻塞前体分析
|
|
535
|
+
# 「模型为什么不长」的结构级定位:逐前体做移除测试,找出卡住生长的前体。
|
|
536
|
+
# 来源:2026-09-11 E2E 绕道归因(agent 手写该逻辑 6+ 次,无工具可用)。
|
|
537
|
+
# 判据刻意用「相对判断」而非绝对可达性 —— 对可生长模型天然零误报。
|
|
538
|
+
# ---------------------------------------------------------------------------
|
|
539
|
+
def op_precursor_scan(args):
|
|
540
|
+
from precursor_scan import scan_precursors
|
|
541
|
+
model = args.get("model")
|
|
542
|
+
if not model or not os.path.exists(model):
|
|
543
|
+
return {"ok": False, "error": f"model file not found: {model}"}
|
|
544
|
+
return {"ok": True, "result": scan_precursors(
|
|
545
|
+
model, medium=args.get("medium"),
|
|
546
|
+
max_precursors=args.get("max_precursors", 200))}
|
|
547
|
+
|
|
548
|
+
|
|
549
|
+
OPS["precursor_scan"] = op_precursor_scan
|
|
550
|
+
|
|
551
|
+
|
|
489
552
|
|
|
490
553
|
def main():
|
|
491
554
|
line = sys.stdin.read()
|
|
@@ -517,4 +580,4 @@ def main():
|
|
|
517
580
|
|
|
518
581
|
|
|
519
582
|
if __name__ == "__main__":
|
|
520
|
-
main()
|
|
583
|
+
main()
|
package/python/l3_fix.py
CHANGED
|
@@ -417,7 +417,7 @@ def l3_fix(model_path, medium=None, substrates=None, out=None,
|
|
|
417
417
|
if exid and exid in m.reactions and g < 1e-6:
|
|
418
418
|
l3.append({"substrate": sub, "exchange": exid, "growth_sole_before": round(g, 6),
|
|
419
419
|
# 阶段A-M4 口径声明(只增)
|
|
420
|
-
"units": "
|
|
420
|
+
"units": "1/h",
|
|
421
421
|
"point_value_note": "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定"})
|
|
422
422
|
|
|
423
423
|
# 第五闸门(入口预检: 每底物至少 1 条新增预估)
|
|
@@ -493,7 +493,7 @@ def l3_fix(model_path, medium=None, substrates=None, out=None,
|
|
|
493
493
|
"evidence": "EVIDENCE_math"})
|
|
494
494
|
growth_a = _growth_sole(cur, resolved_med, exid)
|
|
495
495
|
l3a["growth_sole_after"] = round(growth_a, 6)
|
|
496
|
-
l3a["units"] = "
|
|
496
|
+
l3a["units"] = "1/h"
|
|
497
497
|
l3a["point_value_note"] = "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定"
|
|
498
498
|
entry["l3a"] = l3a
|
|
499
499
|
|
|
@@ -556,12 +556,12 @@ def l3_fix(model_path, medium=None, substrates=None, out=None,
|
|
|
556
556
|
_note(f"[l3b] {sub}: picked {len(picked)}, added {len(added_here)}")
|
|
557
557
|
growth_b = _growth_sole(cur, resolved_med, exid)
|
|
558
558
|
l3b["growth_sole_after"] = round(growth_b, 6)
|
|
559
|
-
l3b["units"] = "
|
|
559
|
+
l3b["units"] = "1/h"
|
|
560
560
|
l3b["point_value_note"] = "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定"
|
|
561
561
|
l3b["added"] = added_here
|
|
562
562
|
entry["l3b"] = l3b
|
|
563
563
|
entry["growth_sole_after"] = round(max(growth_a, growth_b), 6)
|
|
564
|
-
entry["units"] = "
|
|
564
|
+
entry["units"] = "1/h"
|
|
565
565
|
entry["point_value_note"] = "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定"
|
|
566
566
|
entry["verdict"] = "fixed" if growth_b > 1e-6 else "not_fixable"
|
|
567
567
|
if entry["verdict"] == "not_fixable":
|
package/python/ledger.py
CHANGED
|
@@ -401,7 +401,7 @@ def register_phenotype(model_path, g4_results, condition=None, lineage_version=N
|
|
|
401
401
|
"type": "phenotype",
|
|
402
402
|
"content": (f"底物 {sub} 预测{'生长' if r.get('predicted') else '不生长'}"
|
|
403
403
|
f"(文献={r.get('published')},匹配={r.get('match')},"
|
|
404
|
-
f"growth={r.get('growth')}
|
|
404
|
+
f"growth={r.get('growth')} 1/h)"),
|
|
405
405
|
"model": model_path,
|
|
406
406
|
"model_lineage_version": lineage_version,
|
|
407
407
|
"condition": condition,
|
package/python/model_card.py
CHANGED
|
@@ -3,14 +3,15 @@
|
|
|
3
3
|
# / set_verified_phenotypes(phenotype 结果)/ set_essential_genes(必需基因 + 证据分级)
|
|
4
4
|
# 纪律: 各工具完成后**仅当产物模型旁已有 card** 才向后追加;无卡不动(不凭空造卡)。
|
|
5
5
|
# 兼容: build.py 旧卡(无 schema 字段)读取时即时迁移到 v2(新增字段缺失不报错)。
|
|
6
|
-
# units: growth_rate 一律
|
|
6
|
+
# units: growth_rate 一律 1/h(比生长速率;biomass 反应 gDW 归一化口径,数值 = μ)。
|
|
7
|
+
# 2026-09-21 由 mmol/gDW/h 演进为 1/h:数值不变、生物语义更准(外部评审 P0;schema 向后兼容,旧卡照读)。
|
|
7
8
|
import os
|
|
8
9
|
import json
|
|
9
10
|
import time
|
|
10
11
|
|
|
11
12
|
CARD_SUFFIX = ".card.json"
|
|
12
13
|
SCHEMA = "v2"
|
|
13
|
-
GROWTH_UNITS = "
|
|
14
|
+
GROWTH_UNITS = "1/h"
|
|
14
15
|
|
|
15
16
|
|
|
16
17
|
def _now():
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
# precursor_scan.py — dsh-bio-gem 阻塞前体分析(「模型为什么不长」的结构级定位)
|
|
2
|
+
#
|
|
3
|
+
# 来源(2026-09-11 E2E 绕道归因):agent 在不生长模型上反复手写「逐前体探测」逻辑
|
|
4
|
+
# (6+ 次调用),本模块把它固化为工具。
|
|
5
|
+
#
|
|
6
|
+
# ⚠️ 判据设计——**刻意不用「绝对可达性」**:
|
|
7
|
+
# 初版曾用「全开交换下逐前体 demand 能否净生产」,在教科书模型 e_coli_core 上
|
|
8
|
+
# 把 atp_c / accoa_c / nad_c / nadph_c 误报为「既不能合成也不能摄取」。原因是
|
|
9
|
+
# 辅因子有循环补给路径,稳态下**不净生产 ≠ 网络不能供给**。该判据已否决。
|
|
10
|
+
#
|
|
11
|
+
# 现用「**移除测试**」这一相对判据:
|
|
12
|
+
# 基线 biomass 通量 > 0 → 直接返回「无阻塞」(对健康模型天然零误报)
|
|
13
|
+
# 基线 biomass 通量 = 0 → 逐一移除某个前体的需求,看 biomass 是否恢复通量
|
|
14
|
+
# 恢复了 → 该前体即阻塞点(结论由「恢复与否」直接定义,
|
|
15
|
+
# 不依赖对网络的任何绝对判断)
|
|
16
|
+
# 两个验证锚:iNX1344_v3(不生长,应报出阻塞前体)+ e_coli_core(能生长,应报无阻塞)。
|
|
17
|
+
import os
|
|
18
|
+
import sys
|
|
19
|
+
|
|
20
|
+
import cobra # noqa: E402
|
|
21
|
+
|
|
22
|
+
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
|
23
|
+
|
|
24
|
+
EPS = 1e-6
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _find_biomass(model):
|
|
28
|
+
"""定位 biomass 反应:先按 id/name 命中,再退回 objective 变量。"""
|
|
29
|
+
for r in model.reactions:
|
|
30
|
+
if "biomass" in f"{r.id} {r.name or ''}".lower():
|
|
31
|
+
return r
|
|
32
|
+
try:
|
|
33
|
+
syms = [s.name for s in model.objective.expression.free_symbols]
|
|
34
|
+
except Exception: # noqa: BLE001
|
|
35
|
+
syms = []
|
|
36
|
+
live = [r for r in model.reactions if r.id in syms]
|
|
37
|
+
return live[0] if live else None
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def scan_precursors(model_path, medium=None, max_precursors=200):
|
|
41
|
+
"""阻塞前体分析。
|
|
42
|
+
|
|
43
|
+
medium: 可选,{EX_id: lower_bound} 或 {"medium_name": "AB"/"M9"}(走 gapfind 的解析器)。
|
|
44
|
+
返回 {verdict, baseline_flux, blocking_precursors[], note}。
|
|
45
|
+
"""
|
|
46
|
+
from silentio import silent_read_sbml
|
|
47
|
+
from gapfind import expand_medium, resolve_medium
|
|
48
|
+
|
|
49
|
+
m = silent_read_sbml(model_path)
|
|
50
|
+
bio = _find_biomass(m)
|
|
51
|
+
if bio is None:
|
|
52
|
+
return {"error": "未定位到 biomass 反应(id/name 与 objective 均未命中)"}
|
|
53
|
+
|
|
54
|
+
medium_applied = 0
|
|
55
|
+
unresolved = []
|
|
56
|
+
if medium:
|
|
57
|
+
med, _preset = expand_medium(medium)
|
|
58
|
+
resolved, unresolved = resolve_medium(m, med)
|
|
59
|
+
with m:
|
|
60
|
+
for rid, lb in resolved.items():
|
|
61
|
+
if rid in m.reactions:
|
|
62
|
+
m.reactions.get_by_id(rid).lower_bound = float(lb)
|
|
63
|
+
medium_applied += 1
|
|
64
|
+
baseline = m.slim_optimize()
|
|
65
|
+
else:
|
|
66
|
+
baseline = m.slim_optimize()
|
|
67
|
+
baseline = float(baseline) if baseline is not None else 0.0
|
|
68
|
+
|
|
69
|
+
out = {
|
|
70
|
+
"model": model_path,
|
|
71
|
+
"biomass_reaction": bio.id,
|
|
72
|
+
"biomass_name": bio.name or "",
|
|
73
|
+
"medium_applied_exchanges": medium_applied,
|
|
74
|
+
"medium_unresolved": unresolved,
|
|
75
|
+
"baseline_flux": round(baseline, 6),
|
|
76
|
+
"units": "1/h",
|
|
77
|
+
"point_value_note": "单点 FBA 值,非解空间硬结论",
|
|
78
|
+
"blocking_precursors": [],
|
|
79
|
+
"n_blocking": 0,
|
|
80
|
+
"verdict": "",
|
|
81
|
+
"note": "",
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
# 健康路径:能生长 → 不做逐前体测试(这是零误报的关键)
|
|
85
|
+
if baseline > EPS:
|
|
86
|
+
out["verdict"] = "growable"
|
|
87
|
+
out["note"] = (
|
|
88
|
+
"该条件下 biomass 可携带通量 → **无阻塞前体**。已刻意跳过逐前体测试,"
|
|
89
|
+
"避免对可生长模型产生假阳性。扰动/区间分析请用 gem_fluxscan、gem_sensitivity。"
|
|
90
|
+
)
|
|
91
|
+
return out
|
|
92
|
+
|
|
93
|
+
precursors = [k for k, v in bio.metabolites.items() if v < 0][:max_precursors]
|
|
94
|
+
blocking = []
|
|
95
|
+
for met in precursors:
|
|
96
|
+
coeff = bio.metabolites[met]
|
|
97
|
+
with m:
|
|
98
|
+
bio.add_metabolites({met: -coeff}) # 系数清零 = 移除该前体需求
|
|
99
|
+
val = m.slim_optimize()
|
|
100
|
+
val = float(val) if val is not None else 0.0
|
|
101
|
+
if val > EPS:
|
|
102
|
+
blocking.append({
|
|
103
|
+
"metabolite": met.id,
|
|
104
|
+
"name": met.name or "",
|
|
105
|
+
"formula": met.formula or "",
|
|
106
|
+
"coefficient": round(coeff, 4),
|
|
107
|
+
"flux_without_it": round(val, 6),
|
|
108
|
+
})
|
|
109
|
+
|
|
110
|
+
out["blocking_precursors"] = blocking
|
|
111
|
+
out["n_blocking"] = len(blocking)
|
|
112
|
+
out["n_precursors_tested"] = len(precursors)
|
|
113
|
+
if blocking:
|
|
114
|
+
out["verdict"] = "blocked"
|
|
115
|
+
out["note"] = (
|
|
116
|
+
"移除下列前体后 biomass 恢复携带通量 → 它们是**阻塞点**。"
|
|
117
|
+
"排查顺序:① 该前体在模型里是否有合成路径(路径缺失=需补反应);"
|
|
118
|
+
"② 是否被边界/约束卡住;③ 计量或命名是否有问题"
|
|
119
|
+
"(先用 gem_validate 的 g0 数据质量诊断与 g2 配平结论交叉读取)。"
|
|
120
|
+
)
|
|
121
|
+
else:
|
|
122
|
+
out["verdict"] = "infeasible_or_constrained"
|
|
123
|
+
out["note"] = (
|
|
124
|
+
"逐前体移除均未恢复通量 → 阻塞不在单个前体上。可能是 biomass 方程整体计量/方向问题、"
|
|
125
|
+
"约束冲突或能量项缺失;建议先看 gem_validate 的 g0(数据质量)与 g2(配平)。"
|
|
126
|
+
)
|
|
127
|
+
return out
|