@dsh-bio/dsh-bio-gem 0.1.3 → 0.1.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +60 -8
- package/docs/ARCHITECTURE.md +145 -116
- package/docs/DECISIONS-2026-09-21.md +76 -0
- package/docs/releases/v0.1.12.md +43 -0
- package/package.json +5 -3
- package/python/benchmark.py +3 -3
- package/python/biomass_tools.py +2 -2
- package/python/bootstrap_carveme.py +259 -0
- package/python/build.py +10 -2
- package/python/coherence.py +159 -0
- package/python/double_knockout.py +1 -1
- package/python/essential_scan.py +1 -1
- package/python/gapfind.py +413 -396
- package/python/gem_ops.py +64 -1
- package/python/l3_fix.py +4 -4
- package/python/ledger.py +1 -1
- package/python/model_card.py +3 -2
- package/python/precursor_scan.py +127 -0
- package/python/quality.py +577 -0
- package/python/sampling.py +269 -0
- package/python/sensitivity.py +2 -2
- package/python/validate.py +435 -409
- package/skills/gem-expert.md +3 -1
- package/src/capabilities.js +152 -0
- package/src/index.js +21 -2
- package/src/integration.js +507 -0
- package/src/jobs.js +5 -21
- package/src/python.js +113 -17
- package/src/tools.js +98 -22
|
@@ -0,0 +1,577 @@
|
|
|
1
|
+
"""Model-quality report primitives for the ``quality`` gem operation.
|
|
2
|
+
|
|
3
|
+
The report intentionally stays at the model/constraint level. It is a quick
|
|
4
|
+
comparison aid, not a replacement for model curation or biological validation.
|
|
5
|
+
"""
|
|
6
|
+
import contextlib
|
|
7
|
+
import csv
|
|
8
|
+
import hashlib
|
|
9
|
+
import io
|
|
10
|
+
import os
|
|
11
|
+
from collections import defaultdict, deque
|
|
12
|
+
|
|
13
|
+
from cobra.flux_analysis import fastcc, find_blocked_reactions
|
|
14
|
+
|
|
15
|
+
from gapfind import expand_medium, resolve_medium
|
|
16
|
+
from silentio import silent_read_sbml
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
EX_PREFIXES = ("EX_", "DM_", "SK_")
|
|
20
|
+
ID_SAMPLE_LIMIT = 25
|
|
21
|
+
BALANCE_TOLERANCE = 1e-6
|
|
22
|
+
CYCLE_DIRECTION_BOUND = 1e4
|
|
23
|
+
|
|
24
|
+
CHECK_NAMES = (
|
|
25
|
+
"blocked_reactions",
|
|
26
|
+
"cyclic_reactions",
|
|
27
|
+
"elemental_balance",
|
|
28
|
+
"orphan_metabolites",
|
|
29
|
+
"dead_end_metabolites",
|
|
30
|
+
"gpr_coverage",
|
|
31
|
+
"annotation_coverage",
|
|
32
|
+
"connectivity",
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
WEIGHTS = {
|
|
36
|
+
"blocked": 0.2,
|
|
37
|
+
"balance": 0.2,
|
|
38
|
+
"cyclic": 0.1,
|
|
39
|
+
"gpr": 0.25,
|
|
40
|
+
"reaction_annotation": 0.125,
|
|
41
|
+
"metabolite_annotation": 0.125,
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
QUALITY_INDEX_NOTE = (
|
|
45
|
+
"quality_index 是启发式聚合(gem-qi-v1),仅用于快速比较;不得作为单一质量结论引用——"
|
|
46
|
+
"分项指标与 failed_checks 才是判断依据"
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _model_hash(path):
|
|
51
|
+
"""Return the stable, compact content hash required by the operation API."""
|
|
52
|
+
digest = hashlib.sha256()
|
|
53
|
+
with open(path, "rb") as handle:
|
|
54
|
+
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
|
|
55
|
+
digest.update(chunk)
|
|
56
|
+
return digest.hexdigest()[:16]
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _is_boundary(reaction):
|
|
60
|
+
"""Use the repository's boundary/exchange convention consistently."""
|
|
61
|
+
return reaction.boundary or reaction.id.startswith(EX_PREFIXES)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _internal_reactions(model):
|
|
65
|
+
return [reaction for reaction in model.reactions if not _is_boundary(reaction)]
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _selected_checks(checks):
|
|
69
|
+
if checks is None:
|
|
70
|
+
return list(CHECK_NAMES)
|
|
71
|
+
if not isinstance(checks, list) or any(not isinstance(item, str) for item in checks):
|
|
72
|
+
raise ValueError("checks must be an array of quality check names")
|
|
73
|
+
|
|
74
|
+
selected = []
|
|
75
|
+
for item in checks:
|
|
76
|
+
if item not in CHECK_NAMES:
|
|
77
|
+
allowed = ", ".join(CHECK_NAMES)
|
|
78
|
+
raise ValueError(f"unknown quality check: {item} (allowed: {allowed})")
|
|
79
|
+
if item not in selected:
|
|
80
|
+
selected.append(item)
|
|
81
|
+
return selected
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _configure_medium(model, medium):
|
|
85
|
+
"""Apply an explicit medium, or preserve the SBML's bounds when omitted."""
|
|
86
|
+
if medium is not None and not isinstance(medium, dict):
|
|
87
|
+
raise ValueError("medium must be an object when provided")
|
|
88
|
+
|
|
89
|
+
expanded, preset = expand_medium(medium)
|
|
90
|
+
resolved, unresolved = resolve_medium(model, expanded) if expanded else ({}, [])
|
|
91
|
+
|
|
92
|
+
# An omitted medium is explicitly different from an empty supplied medium:
|
|
93
|
+
# the former uses the model's own defaults, while the latter closes uptake.
|
|
94
|
+
if medium is not None:
|
|
95
|
+
for reaction in model.reactions:
|
|
96
|
+
if _is_boundary(reaction):
|
|
97
|
+
reaction.lower_bound = 0.0
|
|
98
|
+
for reaction_id, lower_bound in resolved.items():
|
|
99
|
+
model.reactions.get_by_id(reaction_id).lower_bound = lower_bound
|
|
100
|
+
|
|
101
|
+
return {
|
|
102
|
+
"preset": preset,
|
|
103
|
+
"resolved_exchanges": len(resolved),
|
|
104
|
+
}, unresolved
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _blocked_reaction_ids(model):
|
|
108
|
+
"""Find blocked internal reactions with the Windows-safe FVA setting."""
|
|
109
|
+
internal = _internal_reactions(model)
|
|
110
|
+
if not internal:
|
|
111
|
+
return [], 0
|
|
112
|
+
|
|
113
|
+
# Cobra itself is normally quiet here, but redirecting protects the JSON
|
|
114
|
+
# stdin/stdout protocol from a solver/backend informational message.
|
|
115
|
+
with contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr(io.StringIO()):
|
|
116
|
+
blocked = find_blocked_reactions(
|
|
117
|
+
model,
|
|
118
|
+
reaction_list=internal,
|
|
119
|
+
processes=1,
|
|
120
|
+
)
|
|
121
|
+
return sorted(str(reaction_id) for reaction_id in blocked), len(internal)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _cyclic_direction_cone(model):
|
|
125
|
+
"""Build cobra's documented cyclic-reaction S+direction problem as a Model.
|
|
126
|
+
|
|
127
|
+
``find_cyclic_reactions`` intentionally ignores medium and other model
|
|
128
|
+
constraints. A fresh model is therefore required here: copying the source
|
|
129
|
+
model would accidentally retain those constraints. Reverse-only reactions
|
|
130
|
+
are stoichiometrically flipped so FASTCC can evaluate a non-negative flux
|
|
131
|
+
cone without losing their original identifiers.
|
|
132
|
+
"""
|
|
133
|
+
import cobra
|
|
134
|
+
|
|
135
|
+
cone = cobra.Model("gem_cyclic_direction_cone")
|
|
136
|
+
metabolites = {}
|
|
137
|
+
|
|
138
|
+
def cone_metabolite(metabolite):
|
|
139
|
+
if metabolite.id not in metabolites:
|
|
140
|
+
metabolites[metabolite.id] = cobra.Metabolite(
|
|
141
|
+
metabolite.id,
|
|
142
|
+
name=metabolite.name,
|
|
143
|
+
compartment=metabolite.compartment,
|
|
144
|
+
)
|
|
145
|
+
return metabolites[metabolite.id]
|
|
146
|
+
|
|
147
|
+
# Keep every non-boundary column while constructing S. The final report
|
|
148
|
+
# later applies the repository's EX/DM/SK filtering, but those unusual
|
|
149
|
+
# non-boundary columns must still be allowed to complete an internal cycle.
|
|
150
|
+
for reaction in model.reactions:
|
|
151
|
+
if reaction.boundary:
|
|
152
|
+
continue
|
|
153
|
+
lower_bound, upper_bound = reaction.bounds
|
|
154
|
+
if lower_bound == 0.0 and upper_bound == 0.0:
|
|
155
|
+
continue
|
|
156
|
+
|
|
157
|
+
if lower_bound < 0.0 and upper_bound > 0.0:
|
|
158
|
+
sign, bounds = 1.0, (-CYCLE_DIRECTION_BOUND, CYCLE_DIRECTION_BOUND)
|
|
159
|
+
elif upper_bound > 0.0:
|
|
160
|
+
sign, bounds = 1.0, (0.0, CYCLE_DIRECTION_BOUND)
|
|
161
|
+
elif lower_bound < 0.0:
|
|
162
|
+
sign, bounds = -1.0, (0.0, CYCLE_DIRECTION_BOUND)
|
|
163
|
+
else:
|
|
164
|
+
# An invalid or fixed-zero bound cannot contribute to a nonzero
|
|
165
|
+
# direction cone. The normal cobra model validator rejects it.
|
|
166
|
+
continue
|
|
167
|
+
|
|
168
|
+
cone_reaction = cobra.Reaction(reaction.id, name=reaction.name)
|
|
169
|
+
cone_reaction.lower_bound, cone_reaction.upper_bound = bounds
|
|
170
|
+
cone_reaction.add_metabolites({
|
|
171
|
+
cone_metabolite(metabolite): sign * coefficient
|
|
172
|
+
for metabolite, coefficient in reaction.metabolites.items()
|
|
173
|
+
})
|
|
174
|
+
cone.add_reactions([cone_reaction])
|
|
175
|
+
return cone
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _cyclic_reaction_ids(model):
|
|
179
|
+
"""Find reactions that can carry flux in the structural steady-state cone.
|
|
180
|
+
|
|
181
|
+
On cobra 0.32.1/GLPK, the native randomized optimized loop detector can run
|
|
182
|
+
for a very long time and fail during per-direction verification on large
|
|
183
|
+
models. FASTCC is a deterministic, cobra-bundled LP consistency algorithm.
|
|
184
|
+
Applied to the fresh S+direction cone above, a retained reaction is exactly
|
|
185
|
+
a nonzero steady-state direction and hence a member of the cyclic-reaction
|
|
186
|
+
union reported by ``find_cyclic_reactions``; it does not use medium bounds
|
|
187
|
+
or source-model solver constraints.
|
|
188
|
+
"""
|
|
189
|
+
cone = _cyclic_direction_cone(model)
|
|
190
|
+
if not cone.reactions:
|
|
191
|
+
return []
|
|
192
|
+
with contextlib.redirect_stdout(io.StringIO()), contextlib.redirect_stderr(io.StringIO()):
|
|
193
|
+
consistent_cone = fastcc(
|
|
194
|
+
cone,
|
|
195
|
+
flux_threshold=1.0,
|
|
196
|
+
zero_cutoff=model.tolerance,
|
|
197
|
+
)
|
|
198
|
+
reportable_ids = {reaction.id for reaction in _internal_reactions(model)}
|
|
199
|
+
return sorted(
|
|
200
|
+
reaction.id for reaction in consistent_cone.reactions if reaction.id in reportable_ids
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _elemental_balance(model):
|
|
205
|
+
"""Measure C/N/P/S balance using the established validate.py convention."""
|
|
206
|
+
from validate import CORE_ELEMS, parse_formula
|
|
207
|
+
|
|
208
|
+
unbalanced = []
|
|
209
|
+
details = {}
|
|
210
|
+
checked = 0
|
|
211
|
+
skipped_missing_formula = 0
|
|
212
|
+
skipped_objective_formula = 0
|
|
213
|
+
|
|
214
|
+
# A single-objective biomass equation often contains a pseudo-metabolite
|
|
215
|
+
# without a formula. It is counted as a skipped, not unbalanced, reaction.
|
|
216
|
+
try:
|
|
217
|
+
from cobra.util.solver import linear_reaction_coefficients
|
|
218
|
+
|
|
219
|
+
objective_ids = {reaction.id for reaction in linear_reaction_coefficients(model)}
|
|
220
|
+
except Exception:
|
|
221
|
+
objective_ids = set()
|
|
222
|
+
|
|
223
|
+
for reaction in _internal_reactions(model):
|
|
224
|
+
if not reaction.metabolites or not all(met.formula for met in reaction.metabolites):
|
|
225
|
+
skipped_missing_formula += 1
|
|
226
|
+
if reaction.id in objective_ids:
|
|
227
|
+
skipped_objective_formula += 1
|
|
228
|
+
continue
|
|
229
|
+
|
|
230
|
+
checked += 1
|
|
231
|
+
deltas = defaultdict(float)
|
|
232
|
+
for metabolite, coefficient in reaction.metabolites.items():
|
|
233
|
+
for element, atom_count in parse_formula(metabolite.formula).items():
|
|
234
|
+
if element in CORE_ELEMS:
|
|
235
|
+
deltas[element] += coefficient * atom_count
|
|
236
|
+
|
|
237
|
+
imbalance = {
|
|
238
|
+
element: value
|
|
239
|
+
for element, value in deltas.items()
|
|
240
|
+
if abs(value) > BALANCE_TOLERANCE
|
|
241
|
+
}
|
|
242
|
+
if imbalance:
|
|
243
|
+
unbalanced.append(reaction.id)
|
|
244
|
+
details[reaction.id] = "; ".join(
|
|
245
|
+
f"{element}:{value:+g}" for element, value in sorted(imbalance.items())
|
|
246
|
+
)
|
|
247
|
+
|
|
248
|
+
skipped_note = (
|
|
249
|
+
"口径:仅检查非 boundary/EX/DM/SK 内部反应,且反应中每个代谢物都有 formula;"
|
|
250
|
+
"按现有 gem_validate G2 的 C/N/P/S 严格配平口径计数(H/O 的质子化约定不计入"
|
|
251
|
+
"unbalanced_count)。目标/生物质方程如化学式完整则照常检查;仅当其包含无 formula 的"
|
|
252
|
+
"伪代谢物时跳过并单列说明。"
|
|
253
|
+
f"跳过 {skipped_missing_formula} 条缺公式反应"
|
|
254
|
+
+ (
|
|
255
|
+
f"(其中 {skipped_objective_formula} 条为目标/生物质式且含伪代谢物)"
|
|
256
|
+
if skipped_objective_formula
|
|
257
|
+
else ""
|
|
258
|
+
)
|
|
259
|
+
+ "。"
|
|
260
|
+
)
|
|
261
|
+
return checked, sorted(unbalanced), details, skipped_note
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
def _orphan_and_dead_end_metabolites(model):
|
|
265
|
+
"""Return internal-only metabolites that have only one stoichiometric role."""
|
|
266
|
+
orphan, dead_end = [], []
|
|
267
|
+
skipped_boundary_connected = 0
|
|
268
|
+
|
|
269
|
+
for metabolite in model.metabolites:
|
|
270
|
+
reactions = list(metabolite.reactions)
|
|
271
|
+
# Boundary reactions represent environment/source/sink. A metabolite
|
|
272
|
+
# attached to one is intentionally not labelled a network dead end.
|
|
273
|
+
if not reactions or any(_is_boundary(reaction) for reaction in reactions):
|
|
274
|
+
if reactions:
|
|
275
|
+
skipped_boundary_connected += 1
|
|
276
|
+
continue
|
|
277
|
+
|
|
278
|
+
produces = any(reaction.metabolites[metabolite] > 0 for reaction in reactions)
|
|
279
|
+
consumes = any(reaction.metabolites[metabolite] < 0 for reaction in reactions)
|
|
280
|
+
if consumes and not produces:
|
|
281
|
+
orphan.append(metabolite.id)
|
|
282
|
+
elif produces and not consumes:
|
|
283
|
+
dead_end.append(metabolite.id)
|
|
284
|
+
|
|
285
|
+
note = (
|
|
286
|
+
"孤儿=在内部反应中仅被消耗、无生成反应;死端=仅被生成、无消耗反应。"
|
|
287
|
+
"任何连接 boundary/EX/DM/SK 反应的代谢物均排除,避免把环境供给或排出误判为网络断点;"
|
|
288
|
+
f"本模型因此排除 {skipped_boundary_connected} 个代谢物。"
|
|
289
|
+
)
|
|
290
|
+
return sorted(orphan), sorted(dead_end), note
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def _gpr_coverage(model):
|
|
294
|
+
reactions = list(model.reactions)
|
|
295
|
+
total = len(reactions)
|
|
296
|
+
with_gpr = sum(bool((reaction.gene_reaction_rule or "").strip()) for reaction in reactions)
|
|
297
|
+
return with_gpr, total, (with_gpr / total if total else 0.0)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _annotation_coverage(model):
|
|
301
|
+
reactions = list(model.reactions)
|
|
302
|
+
metabolites = list(model.metabolites)
|
|
303
|
+
annotated_reactions = [reaction.id for reaction in reactions if bool(reaction.annotation)]
|
|
304
|
+
annotated_metabolites = [metabolite.id for metabolite in metabolites if bool(metabolite.annotation)]
|
|
305
|
+
reaction_fraction = len(annotated_reactions) / len(reactions) if reactions else 0.0
|
|
306
|
+
metabolite_fraction = len(annotated_metabolites) / len(metabolites) if metabolites else 0.0
|
|
307
|
+
note = (
|
|
308
|
+
"SBML annotation 统计口径:cobra Reaction.annotation / Metabolite.annotation 非空即计为已注释;"
|
|
309
|
+
"不将 id、name、formula 或 gene_reaction_rule 视为 annotation。"
|
|
310
|
+
)
|
|
311
|
+
return (
|
|
312
|
+
reaction_fraction,
|
|
313
|
+
metabolite_fraction,
|
|
314
|
+
annotated_reactions,
|
|
315
|
+
annotated_metabolites,
|
|
316
|
+
note,
|
|
317
|
+
)
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def _connectivity(model):
|
|
321
|
+
"""Count connected components in the internal metabolite-reaction bipartite graph."""
|
|
322
|
+
adjacency = defaultdict(set)
|
|
323
|
+
nodes = set()
|
|
324
|
+
for reaction in _internal_reactions(model):
|
|
325
|
+
reaction_node = f"reaction:{reaction.id}"
|
|
326
|
+
nodes.add(reaction_node)
|
|
327
|
+
for metabolite in reaction.metabolites:
|
|
328
|
+
metabolite_node = f"metabolite:{metabolite.id}"
|
|
329
|
+
nodes.add(metabolite_node)
|
|
330
|
+
adjacency[reaction_node].add(metabolite_node)
|
|
331
|
+
adjacency[metabolite_node].add(reaction_node)
|
|
332
|
+
|
|
333
|
+
components = 0
|
|
334
|
+
largest = 0
|
|
335
|
+
unseen = set(nodes)
|
|
336
|
+
while unseen:
|
|
337
|
+
components += 1
|
|
338
|
+
start = unseen.pop()
|
|
339
|
+
queue = deque([start])
|
|
340
|
+
component_size = 1
|
|
341
|
+
while queue:
|
|
342
|
+
node = queue.popleft()
|
|
343
|
+
for neighbor in adjacency[node]:
|
|
344
|
+
if neighbor in unseen:
|
|
345
|
+
unseen.remove(neighbor)
|
|
346
|
+
queue.append(neighbor)
|
|
347
|
+
component_size += 1
|
|
348
|
+
largest = max(largest, component_size)
|
|
349
|
+
|
|
350
|
+
fraction = largest / len(nodes) if nodes else 0.0
|
|
351
|
+
note = (
|
|
352
|
+
"connectivity 以非 boundary/EX/DM/SK 反应和其代谢物构成二部图;"
|
|
353
|
+
"components 与 largest_component_fraction 按该图的全部节点数(反应节点+代谢物节点)计算。"
|
|
354
|
+
)
|
|
355
|
+
return components, fraction, note
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def _write_export_csv(path, records):
|
|
359
|
+
"""Write long-form complete lists; JSON samples remain deliberately capped."""
|
|
360
|
+
with open(path, "w", encoding="utf-8", newline="") as handle:
|
|
361
|
+
writer = csv.DictWriter(handle, fieldnames=("check", "item_id", "detail"))
|
|
362
|
+
writer.writeheader()
|
|
363
|
+
writer.writerows(records)
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def quality_report(model_path, medium=None, checks=None, export_csv=None):
|
|
367
|
+
"""Build a gem-qi-v1 report for one SBML model.
|
|
368
|
+
|
|
369
|
+
When ``checks`` is a subset, only requested metric objects are emitted and
|
|
370
|
+
only their assessable scoring components enter the quality-index denominator.
|
|
371
|
+
This prevents an omitted expensive check from being mistaken for missing data.
|
|
372
|
+
"""
|
|
373
|
+
if not model_path or not os.path.isfile(model_path):
|
|
374
|
+
raise ValueError(f"model file not found: {model_path}")
|
|
375
|
+
if export_csv is not None and not isinstance(export_csv, str):
|
|
376
|
+
raise ValueError("export_csv must be a path string when provided")
|
|
377
|
+
|
|
378
|
+
selected = _selected_checks(checks)
|
|
379
|
+
model = silent_read_sbml(model_path)
|
|
380
|
+
medium_summary, unresolved_medium = _configure_medium(model, medium)
|
|
381
|
+
|
|
382
|
+
metrics = {}
|
|
383
|
+
failed_checks = []
|
|
384
|
+
not_assessable = []
|
|
385
|
+
scores = {}
|
|
386
|
+
export_records = []
|
|
387
|
+
notes = [QUALITY_INDEX_NOTE]
|
|
388
|
+
|
|
389
|
+
if medium is None:
|
|
390
|
+
notes.append("介质口径:未提供 medium,blocked_reactions 按 SBML 模型自带的当前 exchange bounds 计算。")
|
|
391
|
+
else:
|
|
392
|
+
notes.append(
|
|
393
|
+
"介质口径:已关闭全部 boundary/EX/DM/SK 摄取下界,再应用请求介质;"
|
|
394
|
+
f"preset={medium_summary['preset']!r},resolved_exchanges={medium_summary['resolved_exchanges']}。"
|
|
395
|
+
)
|
|
396
|
+
if unresolved_medium:
|
|
397
|
+
notes.append("未解析的介质成分未施加:" + ", ".join(sorted(unresolved_medium)) + "。")
|
|
398
|
+
|
|
399
|
+
if "blocked_reactions" in selected:
|
|
400
|
+
blocked_ids, total_checked = _blocked_reaction_ids(model)
|
|
401
|
+
fraction = len(blocked_ids) / total_checked if total_checked else 0.0
|
|
402
|
+
metrics["blocked_reactions"] = {
|
|
403
|
+
"count": len(blocked_ids),
|
|
404
|
+
"total_checked": total_checked,
|
|
405
|
+
"fraction": fraction,
|
|
406
|
+
"ids_sample": blocked_ids[:ID_SAMPLE_LIMIT],
|
|
407
|
+
}
|
|
408
|
+
export_records.extend(
|
|
409
|
+
{"check": "blocked_reactions", "item_id": reaction_id, "detail": "find_blocked_reactions"}
|
|
410
|
+
for reaction_id in blocked_ids
|
|
411
|
+
)
|
|
412
|
+
if total_checked:
|
|
413
|
+
scores["blocked"] = max(0.0, 1.0 - fraction / 0.5)
|
|
414
|
+
if fraction > 0.2:
|
|
415
|
+
failed_checks.append("blocked_reactions")
|
|
416
|
+
else:
|
|
417
|
+
not_assessable.append("blocked_reactions")
|
|
418
|
+
notes.append(
|
|
419
|
+
"blocked_reactions 使用 cobra.find_blocked_reactions;仅计非 boundary/EX/DM/SK 反应,"
|
|
420
|
+
"并显式 processes=1 以兼容 Windows FVA。"
|
|
421
|
+
)
|
|
422
|
+
|
|
423
|
+
if "cyclic_reactions" in selected:
|
|
424
|
+
cyclic_ids = _cyclic_reaction_ids(model)
|
|
425
|
+
metrics["cyclic_reactions"] = {
|
|
426
|
+
"count": len(cyclic_ids),
|
|
427
|
+
"ids_sample": cyclic_ids[:ID_SAMPLE_LIMIT],
|
|
428
|
+
}
|
|
429
|
+
export_records.extend(
|
|
430
|
+
{"check": "cyclic_reactions", "item_id": reaction_id, "detail": "cobra.fastcc_direction_cone"}
|
|
431
|
+
for reaction_id in cyclic_ids
|
|
432
|
+
)
|
|
433
|
+
if _internal_reactions(model):
|
|
434
|
+
scores["cyclic"] = max(0.0, 1.0 - len(cyclic_ids) / 50.0)
|
|
435
|
+
if cyclic_ids:
|
|
436
|
+
failed_checks.append("cyclic_reactions")
|
|
437
|
+
else:
|
|
438
|
+
not_assessable.append("cyclic_reactions")
|
|
439
|
+
notes.append(
|
|
440
|
+
"cyclic_reactions 在独立的化学计量+反应方向锥上以 cobra.fastcc 计算:"
|
|
441
|
+
"非 boundary 反应按可用方向正规化,反向专用反应翻转化学计量,fixed-zero 反应排除。"
|
|
442
|
+
"该口径等价于 cobra.find_cyclic_reactions 所声明的潜在稳态环并集,且刻意忽略介质、"
|
|
443
|
+
"数值 bounds 大小与源模型额外约束;因此它不是当前培养基下每个环都必然可行的通量证明。"
|
|
444
|
+
)
|
|
445
|
+
|
|
446
|
+
if "elemental_balance" in selected:
|
|
447
|
+
checked, unbalanced_ids, imbalance_details, skipped_note = _elemental_balance(model)
|
|
448
|
+
metrics["elemental_balance"] = {
|
|
449
|
+
"checked": checked,
|
|
450
|
+
"balanced": checked - len(unbalanced_ids),
|
|
451
|
+
"unbalanced_count": len(unbalanced_ids),
|
|
452
|
+
"unbalanced_ids_sample": unbalanced_ids[:ID_SAMPLE_LIMIT],
|
|
453
|
+
"skipped_note": skipped_note,
|
|
454
|
+
}
|
|
455
|
+
export_records.extend(
|
|
456
|
+
{
|
|
457
|
+
"check": "elemental_balance",
|
|
458
|
+
"item_id": reaction_id,
|
|
459
|
+
"detail": imbalance_details[reaction_id],
|
|
460
|
+
}
|
|
461
|
+
for reaction_id in unbalanced_ids
|
|
462
|
+
)
|
|
463
|
+
if checked:
|
|
464
|
+
scores["balance"] = (checked - len(unbalanced_ids)) / checked
|
|
465
|
+
if unbalanced_ids:
|
|
466
|
+
failed_checks.append("elemental_balance")
|
|
467
|
+
else:
|
|
468
|
+
not_assessable.append("elemental_balance")
|
|
469
|
+
|
|
470
|
+
if "orphan_metabolites" in selected or "dead_end_metabolites" in selected:
|
|
471
|
+
orphan_ids, dead_end_ids, topology_note = _orphan_and_dead_end_metabolites(model)
|
|
472
|
+
if "orphan_metabolites" in selected:
|
|
473
|
+
metrics["orphan_metabolites"] = {
|
|
474
|
+
"count": len(orphan_ids),
|
|
475
|
+
"ids_sample": orphan_ids[:ID_SAMPLE_LIMIT],
|
|
476
|
+
}
|
|
477
|
+
export_records.extend(
|
|
478
|
+
{"check": "orphan_metabolites", "item_id": metabolite_id, "detail": "only_consumed"}
|
|
479
|
+
for metabolite_id in orphan_ids
|
|
480
|
+
)
|
|
481
|
+
if "dead_end_metabolites" in selected:
|
|
482
|
+
metrics["dead_end_metabolites"] = {
|
|
483
|
+
"count": len(dead_end_ids),
|
|
484
|
+
"ids_sample": dead_end_ids[:ID_SAMPLE_LIMIT],
|
|
485
|
+
}
|
|
486
|
+
export_records.extend(
|
|
487
|
+
{"check": "dead_end_metabolites", "item_id": metabolite_id, "detail": "only_produced"}
|
|
488
|
+
for metabolite_id in dead_end_ids
|
|
489
|
+
)
|
|
490
|
+
notes.append(topology_note)
|
|
491
|
+
|
|
492
|
+
if "gpr_coverage" in selected:
|
|
493
|
+
with_gpr, total, fraction = _gpr_coverage(model)
|
|
494
|
+
metrics["gpr_coverage"] = {
|
|
495
|
+
"reactions_with_gpr": with_gpr,
|
|
496
|
+
"reactions_total": total,
|
|
497
|
+
"fraction": fraction,
|
|
498
|
+
}
|
|
499
|
+
missing_gpr = sorted(
|
|
500
|
+
reaction.id for reaction in model.reactions if not (reaction.gene_reaction_rule or "").strip()
|
|
501
|
+
)
|
|
502
|
+
export_records.extend(
|
|
503
|
+
{"check": "gpr_coverage", "item_id": reaction_id, "detail": "missing_gene_reaction_rule"}
|
|
504
|
+
for reaction_id in missing_gpr
|
|
505
|
+
)
|
|
506
|
+
if total:
|
|
507
|
+
scores["gpr"] = fraction
|
|
508
|
+
if fraction < 0.5:
|
|
509
|
+
failed_checks.append("gpr_coverage")
|
|
510
|
+
else:
|
|
511
|
+
not_assessable.append("gpr_coverage")
|
|
512
|
+
notes.append("gpr_coverage 按全部 SBML reactions 计;非空 gene_reaction_rule 视为已有 GPR。")
|
|
513
|
+
|
|
514
|
+
if "annotation_coverage" in selected:
|
|
515
|
+
(
|
|
516
|
+
reaction_fraction,
|
|
517
|
+
metabolite_fraction,
|
|
518
|
+
annotated_reactions,
|
|
519
|
+
annotated_metabolites,
|
|
520
|
+
annotation_note,
|
|
521
|
+
) = _annotation_coverage(model)
|
|
522
|
+
metrics["annotation_coverage"] = {
|
|
523
|
+
"reaction_annotation_fraction": reaction_fraction,
|
|
524
|
+
"metabolite_annotation_fraction": metabolite_fraction,
|
|
525
|
+
"note": annotation_note,
|
|
526
|
+
}
|
|
527
|
+
export_records.extend(
|
|
528
|
+
{"check": "annotation_coverage", "item_id": reaction_id, "detail": "reaction_annotation_present"}
|
|
529
|
+
for reaction_id in annotated_reactions
|
|
530
|
+
)
|
|
531
|
+
export_records.extend(
|
|
532
|
+
{"check": "annotation_coverage", "item_id": metabolite_id, "detail": "metabolite_annotation_present"}
|
|
533
|
+
for metabolite_id in annotated_metabolites
|
|
534
|
+
)
|
|
535
|
+
if not annotated_reactions and not annotated_metabolites:
|
|
536
|
+
not_assessable.append("annotation_coverage")
|
|
537
|
+
elif model.reactions and model.metabolites:
|
|
538
|
+
scores["reaction_annotation"] = reaction_fraction
|
|
539
|
+
scores["metabolite_annotation"] = metabolite_fraction
|
|
540
|
+
if reaction_fraction < 0.5 or metabolite_fraction < 0.5:
|
|
541
|
+
failed_checks.append("annotation_coverage")
|
|
542
|
+
else:
|
|
543
|
+
not_assessable.append("annotation_coverage")
|
|
544
|
+
notes.append(annotation_note)
|
|
545
|
+
|
|
546
|
+
if "connectivity" in selected:
|
|
547
|
+
components, largest_fraction, connectivity_note = _connectivity(model)
|
|
548
|
+
metrics["connectivity"] = {
|
|
549
|
+
"components": components,
|
|
550
|
+
"largest_component_fraction": largest_fraction,
|
|
551
|
+
}
|
|
552
|
+
if components == 0:
|
|
553
|
+
not_assessable.append("connectivity")
|
|
554
|
+
notes.append(connectivity_note)
|
|
555
|
+
|
|
556
|
+
numerator = sum(WEIGHTS[name] * score for name, score in scores.items())
|
|
557
|
+
denominator = sum(WEIGHTS[name] for name in scores)
|
|
558
|
+
quality_index = 100.0 * numerator / denominator if denominator else 0.0
|
|
559
|
+
if not denominator:
|
|
560
|
+
notes.append("没有可评估的计分项,因此 quality_index 返回 0.0 且不可作比较。")
|
|
561
|
+
|
|
562
|
+
if export_csv:
|
|
563
|
+
_write_export_csv(export_csv, export_records)
|
|
564
|
+
notes.append(f"完整清单已写入 export_csv:{export_csv}。")
|
|
565
|
+
|
|
566
|
+
return {
|
|
567
|
+
"model": model_path,
|
|
568
|
+
"model_hash": _model_hash(model_path),
|
|
569
|
+
"medium": medium_summary,
|
|
570
|
+
"metrics": metrics,
|
|
571
|
+
"quality_index": quality_index,
|
|
572
|
+
"score_profile_version": "gem-qi-v1",
|
|
573
|
+
"weights": dict(WEIGHTS),
|
|
574
|
+
"failed_checks": failed_checks,
|
|
575
|
+
"not_assessable_checks": not_assessable,
|
|
576
|
+
"notes": notes,
|
|
577
|
+
}
|