@dsh-bio/dsh-bio-gem 0.1.3 → 0.1.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,7 +4,7 @@
4
4
  # apply(显式): biomass_profile 覆盖表(op=set|add|remove)→ 副本替换 biomass → 强制 G1-G6 重验
5
5
  # + 三联对照(生长/表型/必需基因 delta)→ model_lineage 追加(有 card 时)
6
6
  # 原则: 默认不应用任何 profile;生长变差 WARN 不阻塞;C58 CarveMe AB 0.624 锚点保护(delta 如实报告)。
7
- # units: growth 一律 mmol/gDW/h。
7
+ # units: growth 一律 1/h(比生长速率;biomass 反应 gDW 归一化口径,数值 = μ)。
8
8
  import os
9
9
  import re
10
10
  import sys
@@ -18,7 +18,7 @@ import cobra
18
18
  EX_PREFIX = ("EX_", "DM_", "SK_")
19
19
  DEFAULT_UNIVERSAL = r"D:\Program\hermes\temp\gem_universal\iML1515.xml"
20
20
  DEFAULT_INX = r"F:\A_NGJ plan\Zcode\models\iNX1344_v4.xml"
21
- GROWTH_UNITS = "mmol/gDW/h"
21
+ GROWTH_UNITS = "1/h"
22
22
  EPS = 1e-6
23
23
 
24
24
  # 分类词表(name 规约匹配为主,id 规约为辅;覆盖 BiGG/ModelSEED/MetaCyc 常见命名)
@@ -0,0 +1,259 @@
1
+ # bootstrap_carveme.py — CarveMe 运行时「零手动部署」(探测 → uv venv → carveme → diamond → 冒烟 → manifest)
2
+ #
3
+ # 产品原则:用户零手动安装。gem_build(engine=carveme) 首次调用时自动补齐运行时:
4
+ # 1. 探测现有 venv(carve.exe + diamond.exe + 冒烟通过)→ 就绪则秒回
5
+ # 2. 缺则部署:uv venv(uv 探测链见下)→ uv pip install carveme → 下载 diamond 官方二进制 → 冒烟
6
+ # 3. 全过程写进度事件(progress.jsonl),失败给「可执行指引」而不是裸异常
7
+ #
8
+ # uv 探测链:GEM_UV env → genie 自举 uv(~/.dsh/dsh-bio-genie/bin/uv.exe)→ PATH uv
9
+ # diamond 来源:GitHub release 固定版本(v2.2.8, diamond-windows.zip, 3.4MB 纯二进制)
10
+ # 下载策略:直连 → 环境变量代理(HTTPS_PROXY/HTTP_PROXY)→ 失败给指引
11
+ #
12
+ # 幂等性:重复调用不重复部署;diamond 单独缺失时只补 diamond;manifest.json 记录版本与来源
13
+ import json
14
+ import os
15
+ import shutil
16
+ import subprocess
17
+ import sys
18
+ import time
19
+ import urllib.request
20
+ import zipfile
21
+
22
+ DEFAULT_CARVE_VENV = os.path.join(os.path.expanduser("~"), ".dsh", "dsh-bio-gem", "venv-carveme")
23
+ GEM_ROOT = os.path.join(os.path.expanduser("~"), ".dsh", "dsh-bio-gem")
24
+ DIAMOND_VERSION = "v2.2.8"
25
+ DIAMOND_URL = f"https://github.com/bbuchfink/diamond/releases/download/{DIAMOND_VERSION}/diamond-windows.zip"
26
+ GENIE_UV = os.path.join(os.path.expanduser("~"), ".dsh", "dsh-bio-genie", "bin", "uv.exe")
27
+
28
+ VERIFY_TIMEOUT = 60
29
+ DOWNLOAD_TIMEOUT = 300
30
+
31
+
32
+ def _log(progress_path, event):
33
+ """进度事件(与 build.py 同格式;progress_path 为 None 时静默)。"""
34
+ if not progress_path:
35
+ return
36
+ try:
37
+ ev = {"ts": time.time(), **event}
38
+ with open(progress_path, "a", encoding="utf-8") as f:
39
+ f.write(json.dumps(ev, ensure_ascii=False) + "\n")
40
+ except Exception:
41
+ pass
42
+
43
+
44
+ def _run(cmd, timeout=VERIFY_TIMEOUT, env=None):
45
+ return subprocess.run(cmd, capture_output=True, text=True, timeout=timeout,
46
+ env=env, encoding="utf-8", errors="replace")
47
+
48
+
49
+ def find_uv():
50
+ """uv 探测链:GEM_UV → genie 自举 → PATH。返回 (uv_path, source) 或 (None, None)。"""
51
+ env_uv = os.environ.get("GEM_UV")
52
+ if env_uv and os.path.exists(env_uv):
53
+ return env_uv, "env:GEM_UV"
54
+ if os.path.exists(GENIE_UV):
55
+ return GENIE_UV, "genie-bootstrap"
56
+ which = shutil.which("uv")
57
+ if which:
58
+ return which, "PATH"
59
+ return None, None
60
+
61
+
62
+ def _carve_exe(venv=None):
63
+ return os.path.join(venv or DEFAULT_CARVE_VENV, "Scripts", "carve.exe")
64
+
65
+
66
+ def _diamond_exe(venv=None):
67
+ return os.path.join(venv or DEFAULT_CARVE_VENV, "Scripts", "diamond.exe")
68
+
69
+
70
+ def _probe(venv=None, deep=False):
71
+ """就绪探测。
72
+ deep=False(默认):文件存在 + manifest 信任 —— 秒级(gem_build 每次调用的日常路径);
73
+ deep=True:额外跑 diamond/carve 冒烟(部署完成后复验一次用)。
74
+ 返回 (ok, detail)。"""
75
+ carve, diamond = _carve_exe(venv), _diamond_exe(venv)
76
+ if not os.path.exists(carve):
77
+ return False, {"missing": "carve.exe"}
78
+ if not os.path.exists(diamond):
79
+ return False, {"missing": "diamond.exe"}
80
+ if not deep and _read_manifest(venv):
81
+ return True, {"mode": "fast-manifest"}
82
+ try:
83
+ r1 = _run([diamond, "--version"])
84
+ if r1.returncode != 0:
85
+ return False, {"diamond_broken": (r1.stderr or r1.stdout or "")[:200]}
86
+ # carve 无 --version:用 -h 验证可启动(exit 0)
87
+ r2 = _run([carve, "-h"])
88
+ if r2.returncode not in (0, 1): # argparse -h 正常为 0;某些版本非 0
89
+ return False, {"carve_broken": (r2.stderr or r2.stdout or "")[:200]}
90
+ return True, {"diamond": (r1.stdout or "").strip()[:80], "mode": "deep-smoke"}
91
+ except Exception as e:
92
+ return False, {"probe_error": f"{type(e).__name__}: {e}"}
93
+
94
+
95
+ def _download(url, dest, progress_path=None):
96
+ """下载(直连 → 环境代理 → 抛错含指引)。返回使用的通道。"""
97
+ headers = {"User-Agent": "dsh-bio-gem-bootstrap/1.0"}
98
+
99
+ def _get(opener=None):
100
+ req = urllib.request.Request(url, headers=headers)
101
+ op = opener.open(req, timeout=DOWNLOAD_TIMEOUT) if opener else urllib.request.urlopen(req, timeout=DOWNLOAD_TIMEOUT)
102
+ with op as resp, open(dest, "wb") as f:
103
+ shutil.copyfileobj(resp, f)
104
+
105
+ try:
106
+ _get()
107
+ return "direct"
108
+ except Exception as e1:
109
+ _log(progress_path, {"event": "diamond_direct_failed", "err": str(e1)[:200]})
110
+ proxy = (os.environ.get("HTTPS_PROXY") or os.environ.get("https_proxy")
111
+ or os.environ.get("HTTP_PROXY") or os.environ.get("http_proxy"))
112
+ if proxy:
113
+ try:
114
+ opener = urllib.request.build_opener(
115
+ urllib.request.ProxyHandler({"http": proxy, "https": proxy}))
116
+ _get(opener)
117
+ return f"proxy:{proxy}"
118
+ except Exception as e2:
119
+ _log(progress_path, {"event": "diamond_proxy_failed", "err": str(e2)[:200]})
120
+ raise RuntimeError(
121
+ f"无法下载 diamond({url})。可执行的手动方案:\n"
122
+ f" 1) 设置代理后重试:set HTTPS_PROXY=http://127.0.0.1:端口\n"
123
+ f" 2) 或手动下载 diamond-windows.zip,把解压出的 diamond.exe 放到:\n"
124
+ f" {os.path.join(DEFAULT_CARVE_VENV, 'Scripts')}")
125
+
126
+
127
+ def ensure_carveme(venv=None, progress_path=None, force_diamond=False):
128
+ """确保 CarveMe 运行时可用(幂等)。返回 dict:
129
+ {ready, action: already|deployed|repaired|failed, carve, diamond, version, note?, error?}
130
+ """
131
+ venv = venv or DEFAULT_CARVE_VENV
132
+ ok, detail = _probe(venv)
133
+ if ok and not force_diamond:
134
+ return {"ready": True, "action": "already", "carve": _carve_exe(venv),
135
+ "diamond": _diamond_exe(venv), "note": detail.get("diamond", "")}
136
+
137
+ _log(progress_path, {"event": "bootstrap_start", "venv": venv, "probe": detail})
138
+ os.makedirs(os.path.dirname(venv), exist_ok=True)
139
+
140
+ uv, uv_src = find_uv()
141
+ if not uv:
142
+ return {"ready": False, "action": "failed", "error":
143
+ "未找到 uv(探测链:GEM_UV env → ~/.dsh/dsh-bio-genie/bin/uv.exe → PATH)。"
144
+ "请安装 uv(https://docs.astral.sh/uv/)或安装宿主插件 dsh-bio-genie 以复用其自举 uv。"}
145
+
146
+ # ---- 1) venv + carveme(carve.exe 缺失或损坏时) ----
147
+ need_env = not os.path.exists(_carve_exe(venv))
148
+ if need_env:
149
+ _log(progress_path, {"event": "uv_venv", "uv": uv, "uv_source": uv_src})
150
+ try:
151
+ r = _run([uv, "venv", venv, "--python", "3.13"], timeout=300)
152
+ if r.returncode != 0 or not os.path.exists(os.path.join(venv, "Scripts", "python.exe")):
153
+ return {"ready": False, "action": "failed",
154
+ "error": f"uv venv 创建失败 rc={r.returncode}: {(r.stderr or '')[-300:]}"}
155
+ except Exception as e:
156
+ return {"ready": False, "action": "failed", "error": f"uv venv 异常: {e}"}
157
+
158
+ py = os.path.join(venv, "Scripts", "python.exe")
159
+ _log(progress_path, {"event": "uv_pip_carveme", "detail": "uv pip install carveme"})
160
+ try:
161
+ r = _run([uv, "pip", "install", "--python", py, "carveme"], timeout=900)
162
+ if r.returncode != 0:
163
+ return {"ready": False, "action": "failed",
164
+ "error": f"carveme 安装失败 rc={r.returncode}: {(r.stderr or '')[-400:]}"}
165
+ except Exception as e:
166
+ return {"ready": False, "action": "failed", "error": f"carveme 安装异常: {e}"}
167
+
168
+ # ---- 2) diamond(缺失/损坏时补) ----
169
+ if not os.path.exists(_diamond_exe(venv)) or force_diamond:
170
+ scripts_dir = os.path.join(venv, "Scripts")
171
+ os.makedirs(scripts_dir, exist_ok=True)
172
+ tmp_zip = os.path.join(venv, "diamond-download.zip")
173
+ _log(progress_path, {"event": "diamond_download", "url": DIAMOND_URL})
174
+ try:
175
+ channel = _download(DIAMOND_URL, tmp_zip, progress_path)
176
+ except RuntimeError as e:
177
+ return {"ready": False, "action": "failed", "error": str(e)}
178
+ try:
179
+ with zipfile.ZipFile(tmp_zip) as z:
180
+ names = z.namelist()
181
+ if "diamond.exe" not in names:
182
+ return {"ready": False, "action": "failed",
183
+ "error": f"diamond zip 内容异常: {names}"}
184
+ z.extract("diamond.exe", scripts_dir)
185
+ except Exception as e:
186
+ return {"ready": False, "action": "failed", "error": f"diamond 解压失败: {e}"}
187
+ finally:
188
+ try:
189
+ os.remove(tmp_zip)
190
+ except OSError:
191
+ pass
192
+ _log(progress_path, {"event": "diamond_ready", "channel": channel})
193
+
194
+ # ---- 3) 冒烟复验(deep:真跑 diamond/carve 一次) ----
195
+ ok2, detail2 = _probe(venv, deep=True)
196
+ version = _read_carve_version(venv)
197
+ manifest = {
198
+ "carve_version": version,
199
+ "diamond_version": DIAMOND_VERSION,
200
+ "diamond_url": DIAMOND_URL,
201
+ "uv_source": uv_src,
202
+ "installed_at": time.strftime("%Y-%m-%d %H:%M:%S"),
203
+ }
204
+ try:
205
+ with open(os.path.join(venv, "manifest.json"), "w", encoding="utf-8") as f:
206
+ json.dump(manifest, f, ensure_ascii=False, indent=1)
207
+ except OSError:
208
+ pass
209
+
210
+ if not ok2:
211
+ return {"ready": False, "action": "failed",
212
+ "error": f"部署后冒烟未通过: {detail2}"}
213
+ action = "deployed" if need_env else "repaired"
214
+ _log(progress_path, {"event": "bootstrap_done", "action": action, "manifest": manifest})
215
+ return {"ready": True, "action": action, "carve": _carve_exe(venv),
216
+ "diamond": _diamond_exe(venv), "manifest": manifest}
217
+
218
+
219
+ def _read_carve_version(venv=None):
220
+ py = os.path.join(venv or DEFAULT_CARVE_VENV, "Scripts", "python.exe")
221
+ if not os.path.exists(py):
222
+ return None
223
+ try:
224
+ r = _run([py, "-c",
225
+ "import importlib.metadata as im; print(im.version('carveme'))"])
226
+ if r.returncode == 0:
227
+ return (r.stdout or "").strip()
228
+ except Exception:
229
+ pass
230
+ return None
231
+
232
+
233
+ def carveme_status(venv=None):
234
+ """只读状态(供诊断/工具面板)。"""
235
+ ok, detail = _probe(venv)
236
+ return {"ready": ok, "venv": venv or DEFAULT_CARVE_VENV,
237
+ "carve": _carve_exe(venv), "diamond": _diamond_exe(venv),
238
+ "detail": detail, "manifest": _read_manifest(venv),
239
+ "uv": find_uv()[1]}
240
+
241
+
242
+ def _read_manifest(venv=None):
243
+ p = os.path.join(venv or DEFAULT_CARVE_VENV, "manifest.json")
244
+ try:
245
+ with open(p, encoding="utf-8") as f:
246
+ return json.load(f)
247
+ except Exception:
248
+ return None
249
+
250
+
251
+ if __name__ == "__main__":
252
+ # CLI:python bootstrap_carveme.py [status|ensure]
253
+ action = sys.argv[1] if len(sys.argv) > 1 else "ensure"
254
+ if action == "status":
255
+ print(json.dumps(carveme_status(), ensure_ascii=False, indent=1))
256
+ else:
257
+ out = ensure_carveme()
258
+ print(json.dumps(out, ensure_ascii=False, indent=1))
259
+ sys.exit(0 if out.get("ready") else 1)
package/python/build.py CHANGED
@@ -138,6 +138,14 @@ def build(input_spec, name=None, medium=None, venv=None, out_dir=None, progress_
138
138
  os.makedirs(out_dir, exist_ok=True)
139
139
  out_xml = os.path.join(out_dir, name + ".xml")
140
140
 
141
+ # 0) 运行时自举(零手动安装):carve + diamond 缺失时自动部署(幂等,就绪时秒过)
142
+ from bootstrap_carveme import ensure_carveme
143
+ boot = ensure_carveme(venv=venv, progress_path=progress_path)
144
+ if not boot.get("ready"):
145
+ _log(progress_path, {"event": "bootstrap_failed", "error": boot.get("error")})
146
+ raise RuntimeError("CarveMe 运行时不可用:" + str(boot.get("error")))
147
+ _log(progress_path, {"event": "bootstrap_ok", "action": boot.get("action")})
148
+
141
149
  # 1) carve(自带 M9 gapfill,CarveMe 原生最小培养基)
142
150
  st = time.time()
143
151
  if not (os.path.exists(out_xml) and os.path.getmtime(out_xml) > os.path.getmtime(proteins)):
@@ -195,7 +203,7 @@ def build(input_spec, name=None, medium=None, venv=None, out_dir=None, progress_
195
203
  # 4) 模型卡(schema v2 起步:supported_mediums 由验证结果得出)
196
204
  supported = [
197
205
  {"medium_name": "M9", "ex_reactions": sorted(med_m9),
198
- "growth_rate": g3_m9.get("growth_medium"), "units": "mmol/gDW/h",
206
+ "growth_rate": g3_m9.get("growth_medium"), "units": "1/h",
199
207
  "validation_status": "verified_G3" if g3_m9.get("status") == "PASS" else "unverified"},
200
208
  ]
201
209
  if target and target.get("g3") == "PASS":
@@ -204,7 +212,7 @@ def build(input_spec, name=None, medium=None, venv=None, out_dir=None, progress_
204
212
  tname = medium.get("medium_name") or "custom"
205
213
  supported.append({
206
214
  "medium_name": tname, "ex_reactions": target.get("resolved_exchanges"),
207
- "growth_rate": target.get("growth"), "units": "mmol/gDW/h",
215
+ "growth_rate": target.get("growth"), "units": "1/h",
208
216
  "validation_status": "verified_G3_G4" if target.get("gapfixes_applied", 0) == 0 else "verified_G3_only",
209
217
  })
210
218
  # 模型卡(schema v2:init_card 统一基座 —— lineage v0.1.0 起始 + changelog=[build])
@@ -0,0 +1,159 @@
1
+ # coherence.py — dsh-bio-gem 模型自洽性前置诊断(G0)
2
+ #
3
+ # 目的:在 G3/G4/G5 与 gapfind 之前,先判断**模型自身数据质量是否允许下结论**,
4
+ # 避免把「未映射代谢物」这类数据问题,误报成「通路缺口 / 必需基因异常」。
5
+ #
6
+ # 实测来源(2026-09-10 E2E,iNX1344_v3 —— MetaCyc 风格 id 的公开模型):
7
+ # - gem_gapfind 报 5 个 L3「内部通路缺口」,根因实为 biomass 前体未映射;
8
+ # - agent 为证伪这些假阳性,手写 cobra 代码 18 次(占该轮调用的一半)。
9
+ #
10
+ # ⚠️⚠️ 判据设计原则:**零误报优先**。以下两类判据在设计中被实测否决,切勿加回:
11
+ #
12
+ # 1. 「biomass 元素配平」——对 biomass 方程**不适用**。标准 biomass 方程代表大分子
13
+ # 聚合,产物侧用占位代谢物表示生物量(无独立化学式),元素净不平衡是**预期**
14
+ # 行为而非缺陷。对照实验:教科书模型 e_coli_core 的 Biomass_Ecoli_core 净不平衡
15
+ # C -42.56 / N -5.45 / P -3.68,用它判据会把公认良好的模型判成 FAIL。
16
+ # 2. 「前体可达性(demand 逐前体 FBA)」——初版实现同样在 e_coli_core 上把
17
+ # atp_c / accoa_c / nad_c / nadph_c 误报为「既不能合成也不能摄取」。全开交换下
18
+ # 的 demand 语义与胞内辅因子/能量货币的循环补给路径纠缠,判据未成熟。
19
+ # 正确方法(agent 在 E2E 中手工探索过)待重新设计后引入。
20
+ #
21
+ # 保留的判据都经过「问题模型报出真问题 + 标准模型零误报」双向验证:
22
+ # - id 体系识别(信息性,决定下游名称映射口径)
23
+ # - biomass 未映射前体(无 name / 无 formula)→ iNX1344_v3 报 8 个,e_coli_core 报 0 个
24
+ # - 产物侧出现 ATP(生长方向疑似写反)→ 两个模型均不报
25
+ import os
26
+ import re
27
+ import sys
28
+
29
+ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
30
+
31
+ # 代谢物 id 命名体系 → 正则(命中率 < ID_FRACTION 记为 mixed/unknown)
32
+ ID_SYSTEM_PATTERNS = (
33
+ ("bigg", re.compile(r"^[a-z][a-z0-9]{1,}_[a-z]\d?$")), # atp_c / h2o_c / g6p_c
34
+ ("metacyc", re.compile(r"^[Mm]?\d{5}(?:_[a-z]\d?)?$")), # M00002_c / cpd00002_c0
35
+ ("carveme", re.compile(r"^M_[a-z0-9]+_[a-z]\d?$")), # M_atp_c
36
+ )
37
+ ID_FRACTION = 0.5
38
+
39
+
40
+ # ---------------------------------------------------------------- ID 体系
41
+ def classify_id_system(model):
42
+ """按代谢物 id 命名习惯**归类** ID 体系(决定下游关卡的名称映射口径)。
43
+
44
+ 与 `benchmark.detect_id_system` 的区别(刻意不同名,勿合并):后者列举
45
+ 基因/反应/代谢物的 ID **风格样本**(返回基因、反应、代谢物三组标签与计数);
46
+ 本函数按正则归类出**体系名**(bigg / metacyc / carveme / mixed / unknown)
47
+ 并给出各体系命中比例,供路由与降级逻辑判断。
48
+ """
49
+ mets = [x for x in model.metabolites if x.id]
50
+ if not mets:
51
+ return {"system": "unknown", "fractions": {}, "sampled": 0}
52
+ hits = {name: 0 for name, _ in ID_SYSTEM_PATTERNS}
53
+ for met in mets:
54
+ for name, pat in ID_SYSTEM_PATTERNS:
55
+ if pat.match(met.id):
56
+ hits[name] += 1
57
+ break
58
+ n = len(mets)
59
+ fracs = {k: round(v / n, 4) for k, v in hits.items()}
60
+ best = max(fracs, key=fracs.get)
61
+ system = best if fracs[best] >= ID_FRACTION else ("mixed" if any(fracs.values()) else "unknown")
62
+ return {"system": system, "fractions": fracs, "sampled": n}
63
+
64
+
65
+ # ---------------------------------------------------------------- biomass
66
+ def locate_biomass(model):
67
+ """定位 biomass 反应:先按 id/name 命中,再退回 objective 变量。
68
+
69
+ 与 `biomass_tools.find_biomass` 的区别(刻意不同名,勿合并):后者按
70
+ `objective_coefficient != 0` 找 FBA 目标反应、多个时取组分最多者,服务于
71
+ biomass 精修;本函数按**名称**优先,服务于「这个模型的生长目标长什么样」
72
+ 的数据质量诊断,返回 (reaction, 命中方式)。
73
+ """
74
+ for r in model.reactions:
75
+ if "biomass" in f"{r.id} {r.name or ''}".lower():
76
+ return r, "id_or_name"
77
+ try:
78
+ syms = [s.name for s in model.objective.expression.free_symbols]
79
+ except Exception: # noqa: BLE001
80
+ syms = []
81
+ live = [r for r in model.reactions if r.id in syms]
82
+ if len(live) == 1:
83
+ return live[0], "objective"
84
+ if live:
85
+ return live[0], "objective_multi"
86
+ return None, "not_found"
87
+
88
+
89
+ def check_biomass(model):
90
+ """biomass 可用性:未映射前体(主判据)+ 产物侧 ATP(方向异常)。
91
+
92
+ 不做元素配平判定——标准 biomass 方程本就不配平(见模块头注释)。
93
+ """
94
+ bio, source = locate_biomass(model)
95
+ if bio is None:
96
+ return {"status": "WARN", "found_by": source,
97
+ "notes": ["未定位到 biomass 反应(id/name 与 objective 均未命中)→ 无法评估生长目标"]}
98
+
99
+ prods = [k for k, v in bio.metabolites.items() if v > 0]
100
+ subs = [k for k, v in bio.metabolites.items() if v < 0]
101
+ unmapped = [k.id for k in bio.metabolites if not (k.formula or "").strip()]
102
+ unnamed = [k.id for k in bio.metabolites if not (k.name or "").strip()]
103
+ # 方向异常:产物侧出现 ATP(生长应消耗 ATP、产出 ADP)。ADP/Pi 在产物侧属正常。
104
+ atp_in_products = [k.id for k in prods if (k.name or "").strip().upper() == "ATP"]
105
+
106
+ notes, status = [], "PASS"
107
+ if unmapped:
108
+ status = "WARN"
109
+ notes.append(f"biomass 含 {len(unmapped)} 个未映射代谢物(无 formula)→ "
110
+ "其质量未定义,配平/缺口类结论均不可靠;先补映射再解读下游结果")
111
+ if unnamed:
112
+ notes.append(f"另有 {len(unnamed)} 个无名称代谢物 → 报告可读性受限")
113
+ if atp_in_products:
114
+ status = "FAIL"
115
+ notes.append(f"产物侧出现 ATP({atp_in_products[:3]})→ 生长方向疑似写反")
116
+ if not notes:
117
+ notes.append("biomass 组成部分映射完整,未发现方向异常")
118
+
119
+ return {
120
+ "status": status, "found_by": source,
121
+ "reaction": bio.id, "reaction_name": bio.name or "",
122
+ "bounds": [bio.lower_bound, bio.upper_bound],
123
+ "n_substrates": len(subs), "n_products": len(prods),
124
+ "unmapped_metabolites": unmapped, "unnamed_metabolites": unnamed[:10],
125
+ "atp_in_products": atp_in_products,
126
+ "notes": notes,
127
+ }
128
+
129
+
130
+ # ---------------------------------------------------------------- 汇总
131
+ def model_coherence(model):
132
+ """模型数据质量诊断:ID 体系 + biomass 可用性。"""
133
+ idrep = classify_id_system(model)
134
+ biorep = check_biomass(model)
135
+
136
+ rank = {"PASS": 0, "WARN": 1, "FAIL": 2}
137
+ status = max((biorep.get("status", "PASS"),), key=lambda s: rank.get(s, 0))
138
+
139
+ hints = []
140
+ if biorep.get("unmapped_metabolites"):
141
+ hints.append(f"biomass 含未映射代谢物 {biorep['unmapped_metabolites'][:3]} → "
142
+ "先补 name/formula,再信任何配平 / 缺口 / 必需性结论")
143
+ if biorep.get("atp_in_products"):
144
+ hints.append("biomass 产物侧出现 ATP → 生长方向疑似写反,先修方程")
145
+ if idrep["system"] in ("metacyc", "mixed", "unknown"):
146
+ hints.append(f"id 体系为 {idrep['system']}(非 BiGG)→ 培养基/代谢物名称需跨体系匹配,"
147
+ "天然名解析失败时先判为命名口径问题,而非模型缺陷")
148
+
149
+ return {
150
+ "status": status,
151
+ "id_system": idrep,
152
+ "biomass": biorep,
153
+ "downstream_hint": ";".join(hints),
154
+ }
155
+
156
+
157
+ def coherence_from_path(model_path):
158
+ from silentio import silent_read_sbml
159
+ return model_coherence(silent_read_sbml(model_path))
@@ -63,7 +63,7 @@ def double_knockout(model_path, medium=None, max_pairs=5000, export_csv=None,
63
63
  log(f"[dk] {model_path} medium={medium} wt={wt} max_pairs={max_pairs}")
64
64
 
65
65
  out = {"model": model_path, "medium": medium, "medium_preset": preset,
66
- "wt_growth": wt, "units": "mmol/gDW/h", "assumption_note": ASSUMPTION_NOTE,
66
+ "wt_growth": wt, "units": "1/h", "assumption_note": ASSUMPTION_NOTE,
67
67
  "max_pairs": max_pairs, "eps": EPS,
68
68
  "degenerate": wt <= EPS}
69
69
  if out["degenerate"]:
@@ -151,7 +151,7 @@ def essential_scan(model_path, medium=None, gene_subset=None, progress=None, led
151
151
  "medium_unresolved": unresolved,
152
152
  "note": "必需判定=A 培养基下敲除生长<1e-6;evidence 分级按基因支撑反应是否含 EVIDENCE_math(Q2)",
153
153
  # 阶段A-M4 口径声明(只增):wt_growth 为单点 FBA 值
154
- "units": "mmol/gDW/h",
154
+ "units": "1/h",
155
155
  "point_value_note": "单点 FBA 值,非解空间硬结论;条件对比请用 gem_fluxscan 区间分离判定",
156
156
  })
157
157
  # 模型卡 schema v2 回写(产物旁已有 card 才写;无卡不凭空造卡)