msp-sc 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- msp/__init__.py +35 -0
- msp/__main__.py +122 -0
- msp/agent_util.py +70 -0
- msp/annotate.py +632 -0
- msp/inspect.py +601 -0
- msp/integrate.py +890 -0
- msp/plots.py +138 -0
- msp/report.py +751 -0
- msp_sc-0.2.0.dist-info/METADATA +223 -0
- msp_sc-0.2.0.dist-info/RECORD +13 -0
- msp_sc-0.2.0.dist-info/WHEEL +5 -0
- msp_sc-0.2.0.dist-info/licenses/LICENSE +21 -0
- msp_sc-0.2.0.dist-info/top_level.txt +1 -0
msp/__init__.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""msp (multi-sample-pipeline): integrate osp per-sample outputs (harmony)
|
|
2
|
+
→ multi-resolution leiden + UMAP → cluster QC / DEG tables → self-contained
|
|
3
|
+
HTML report, with two optional Claude-agent steps that run afterwards:
|
|
4
|
+
|
|
5
|
+
integrate (msp.integrate) propose-only: nothing deleted, nothing named
|
|
6
|
+
inspect (msp.inspect) per-cluster five-test QC verdicts, proposals only
|
|
7
|
+
annotate (msp.annotate) coarse/fine cell identity on msp_leiden_r2.0,
|
|
8
|
+
explicit merges, REAL removal → annotated.h5ad
|
|
9
|
+
|
|
10
|
+
Entry points:
|
|
11
|
+
from msp import run_multi_sample_pipeline, generate_report
|
|
12
|
+
|
|
13
|
+
run_multi_sample_pipeline(["A/clustered.h5ad", "B/clustered.h5ad"],
|
|
14
|
+
batch_col="project", outdir="msp_out")
|
|
15
|
+
generate_report("msp_out")
|
|
16
|
+
|
|
17
|
+
Command line:
|
|
18
|
+
python -m msp A/clustered.h5ad B/clustered.h5ad --batch-col project --outdir msp_out
|
|
19
|
+
python -m msp ... --inspect --annotate --model claude-sonnet-5 # full chain
|
|
20
|
+
python -m msp.inspect msp_out # QC inspection agent only
|
|
21
|
+
python -m msp.annotate msp_out # annotation agent only (after inspect)
|
|
22
|
+
python -m msp.report msp_out # rebuild the report only
|
|
23
|
+
|
|
24
|
+
msp.inspect / msp.annotate are intentionally not imported here — they depend
|
|
25
|
+
on the optional claude-agent-sdk (`pip install "msp[agent]"`); use
|
|
26
|
+
`from msp.inspect import inspect_clusters` / `from msp.annotate import
|
|
27
|
+
annotate_clusters` when needed.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from .integrate import integrate_adata, load_and_merge, run_multi_sample_pipeline
|
|
31
|
+
from .plots import save_single_umap
|
|
32
|
+
from .report import generate_report
|
|
33
|
+
|
|
34
|
+
__all__ = ["integrate_adata", "load_and_merge", "run_multi_sample_pipeline", "generate_report",
|
|
35
|
+
"save_single_umap"]
|
msp/__main__.py
ADDED
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
"""python -m msp: integrate osp per-sample outputs (concat → harmony → leiden →
|
|
2
|
+
UMAP → QC/DEG tables → HTML report), end to end.
|
|
3
|
+
|
|
4
|
+
With --inspect, the per-cluster QC inspection agent (msp.inspect) runs
|
|
5
|
+
afterwards; with --annotate, the cell-type annotation agent (msp.annotate)
|
|
6
|
+
runs after that (both need the optional claude-agent-sdk). Each step is
|
|
7
|
+
skipped when its contract file already exists, so re-running the same
|
|
8
|
+
command resumes where it stopped; --force redoes everything.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import argparse
|
|
12
|
+
import os
|
|
13
|
+
import sys
|
|
14
|
+
|
|
15
|
+
from .integrate import integrate_adata, run_multi_sample_pipeline
|
|
16
|
+
from .report import generate_report, write_report_context
|
|
17
|
+
|
|
18
|
+
parser = argparse.ArgumentParser(prog="msp", description=__doc__,
|
|
19
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
20
|
+
parser.add_argument("inputs", nargs="*", help="per-sample clustered.h5ad files (osp outputs)")
|
|
21
|
+
parser.add_argument("--from-h5ad", default=None, metavar="H5AD",
|
|
22
|
+
help="instead of per-sample inputs: one already-merged h5ad with layers['counts'] "
|
|
23
|
+
"(e.g. a previous round's annotated_zmip.h5ad) — re-integrated from scratch via "
|
|
24
|
+
"integrate_adata; prior obs columns ride along as annotation evidence")
|
|
25
|
+
parser.add_argument("--batch-col", required=True, help="obs column naming the sample/batch")
|
|
26
|
+
parser.add_argument("--outdir", required=True)
|
|
27
|
+
parser.add_argument("--species", default=None, help="stored in uns['msp']; context for the agents")
|
|
28
|
+
parser.add_argument("--resolutions", type=float, nargs="+", default=[0.3, 1.0, 2.0],
|
|
29
|
+
help="leiden resolutions; 1.0 and 2.0 must be present for inspect/annotate")
|
|
30
|
+
parser.add_argument("--n-top-genes", type=int, default=2000)
|
|
31
|
+
parser.add_argument("--n-pcs", type=int, default=50)
|
|
32
|
+
parser.add_argument("--n-neighbors", type=int, default=15)
|
|
33
|
+
parser.add_argument("--harmony", action="append", default=[], metavar="KEY=VALUE",
|
|
34
|
+
help="harmonypy.run_harmony override, repeatable: e.g. --harmony theta=1 "
|
|
35
|
+
"--harmony lamb=-1 --harmony max_iter_harmony=20 --harmony sigma=0.2 "
|
|
36
|
+
"(defaults: theta=2, lamb=1, sigma=0.1, nclust=min(N/30,100), "
|
|
37
|
+
"max_iter_harmony=10, max_iter_kmeans=20)")
|
|
38
|
+
parser.add_argument("--inspect", action="store_true",
|
|
39
|
+
help="after integration, run the per-cluster QC inspection agent (msp.inspect)")
|
|
40
|
+
parser.add_argument("--annotate", action="store_true",
|
|
41
|
+
help="after inspection, run the cell-type annotation agent (msp.annotate); "
|
|
42
|
+
"implies --inspect")
|
|
43
|
+
parser.add_argument("--language", default="English", help='agent prose language (default "English")')
|
|
44
|
+
parser.add_argument("--model", default=None, help='model for the agents, e.g. "claude-sonnet-5"')
|
|
45
|
+
parser.add_argument("--effort", default=None, choices=["low", "medium", "high", "xhigh", "max"],
|
|
46
|
+
help="reasoning effort for the agents (models that support it)")
|
|
47
|
+
parser.add_argument("--max-turns", type=int, default=None,
|
|
48
|
+
help="agent turn budget (defaults: inspect 100, annotate 200)")
|
|
49
|
+
parser.add_argument("--report-context", default=None, metavar="TEXT",
|
|
50
|
+
help='where this run sits, for report titles (e.g. "round 2 · fu2022-meniscus"); '
|
|
51
|
+
"persisted in <outdir>/report_context.txt so later report refreshes keep it")
|
|
52
|
+
parser.add_argument("--force", action="store_true", help="redo steps whose outputs already exist")
|
|
53
|
+
args = parser.parse_args()
|
|
54
|
+
|
|
55
|
+
if bool(args.inputs) == bool(args.from_h5ad):
|
|
56
|
+
sys.exit("give either per-sample inputs or --from-h5ad, not both / neither")
|
|
57
|
+
if args.annotate:
|
|
58
|
+
args.inspect = True
|
|
59
|
+
if args.inspect and not {1.0, 2.0} <= set(args.resolutions):
|
|
60
|
+
sys.exit("--inspect/--annotate need leiden resolutions 1.0 and 2.0 (see --resolutions)")
|
|
61
|
+
|
|
62
|
+
out = args.outdir
|
|
63
|
+
write_report_context(out, args.report_context)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _parse_kv(items):
|
|
67
|
+
"""KEY=VALUE → {key: number|list|str}; comma-separated values become lists."""
|
|
68
|
+
def conv(v):
|
|
69
|
+
for cast in (int, float):
|
|
70
|
+
try:
|
|
71
|
+
return cast(v)
|
|
72
|
+
except ValueError:
|
|
73
|
+
pass
|
|
74
|
+
return v
|
|
75
|
+
out = {}
|
|
76
|
+
for it in items:
|
|
77
|
+
if "=" not in it:
|
|
78
|
+
sys.exit(f"--harmony expects KEY=VALUE, got {it!r}")
|
|
79
|
+
k, v = it.split("=", 1)
|
|
80
|
+
out[k.strip()] = [conv(x) for x in v.split(",")] if "," in v else conv(v)
|
|
81
|
+
return out
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
harmony_kwargs = _parse_kv(args.harmony)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _done(*names):
|
|
88
|
+
return all(os.path.exists(os.path.join(out, n)) for n in names)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
if args.force or not _done("integrated.h5ad", "report.html"):
|
|
92
|
+
kw = dict(species=args.species, resolutions=tuple(args.resolutions), n_top_genes=args.n_top_genes,
|
|
93
|
+
n_pcs=args.n_pcs, n_neighbors=args.n_neighbors, harmony_kwargs=harmony_kwargs)
|
|
94
|
+
if args.from_h5ad:
|
|
95
|
+
import scanpy as sc
|
|
96
|
+
|
|
97
|
+
ad = sc.read_h5ad(args.from_h5ad)
|
|
98
|
+
_, summary = integrate_adata(ad, args.batch_col, out, inputs=[args.from_h5ad], **kw)
|
|
99
|
+
else:
|
|
100
|
+
_, summary = run_multi_sample_pipeline(args.inputs, batch_col=args.batch_col, outdir=out, **kw)
|
|
101
|
+
print(summary)
|
|
102
|
+
print(f"report: {generate_report(out)}")
|
|
103
|
+
else:
|
|
104
|
+
print(f"[resume] integration already done in {out} (integrated.h5ad + report.html) — skipping")
|
|
105
|
+
|
|
106
|
+
agent_kw = dict(species=args.species, language=args.language, model=args.model, effort=args.effort)
|
|
107
|
+
|
|
108
|
+
if args.inspect:
|
|
109
|
+
if args.force or not _done("inspection_proposal.json"):
|
|
110
|
+
from .inspect import inspect_clusters
|
|
111
|
+
|
|
112
|
+
inspect_clusters(out, max_turns=args.max_turns or 100, **agent_kw)
|
|
113
|
+
else:
|
|
114
|
+
print(f"[resume] inspection_proposal.json exists in {out} — skipping inspect")
|
|
115
|
+
|
|
116
|
+
if args.annotate:
|
|
117
|
+
if args.force or not _done("annotation_proposal.json", "annotated.h5ad"):
|
|
118
|
+
from .annotate import annotate_clusters
|
|
119
|
+
|
|
120
|
+
annotate_clusters(out, max_turns=args.max_turns or 200, **agent_kw)
|
|
121
|
+
else:
|
|
122
|
+
print(f"[resume] annotation_proposal.json + annotated.h5ad exist in {out} — skipping annotate")
|
msp/agent_util.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""Shared agent-call helper for every claude-agent-sdk session in msp / zmip.
|
|
2
|
+
|
|
3
|
+
The CLI itself retries transient API errors (429 / overloaded / 5xx) with
|
|
4
|
+
backoff, but a subscription usage window that is used up ("Claude usage
|
|
5
|
+
limit reached …") ends the session with an error result and the CLI does
|
|
6
|
+
NOT wait for the reset. A self-driving loop must not stop for that: this
|
|
7
|
+
wrapper recognises limit-type failures and re-runs the whole query after a
|
|
8
|
+
wait (session state is in-memory on the host — submitted entries persist,
|
|
9
|
+
the agent simply starts its investigation again), bounded by a total wait
|
|
10
|
+
budget. Any other failure is raised immediately.
|
|
11
|
+
|
|
12
|
+
async for message in run_query(prompt, options, label="inspect"):
|
|
13
|
+
...
|
|
14
|
+
|
|
15
|
+
Env: AGENT_LIMIT_WAIT_MIN (minutes between retries, default 10),
|
|
16
|
+
AGENT_LIMIT_WAIT_MAX_H (total hours to keep waiting, default 12).
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import asyncio
|
|
20
|
+
import os
|
|
21
|
+
import re
|
|
22
|
+
import time
|
|
23
|
+
|
|
24
|
+
LIMIT_PATTERN = re.compile(
|
|
25
|
+
r"usage limit|rate[ _-]?limit|limit will reset|resets at|too many requests|overloaded|"
|
|
26
|
+
r"quota|429|capacity|out of extra usage|spend limit",
|
|
27
|
+
re.IGNORECASE,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def is_limit_error(text) -> bool:
|
|
32
|
+
return bool(text) and bool(LIMIT_PATTERN.search(str(text)))
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class AgentLimitExhausted(RuntimeError):
|
|
36
|
+
pass
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
async def run_query(prompt, options, label="agent"):
|
|
40
|
+
"""Yield the SDK's messages exactly like query(); if the run ends in a
|
|
41
|
+
limit-type error, wait and start over (bounded)."""
|
|
42
|
+
from claude_agent_sdk import ResultMessage, query
|
|
43
|
+
|
|
44
|
+
wait_min = float(os.environ.get("AGENT_LIMIT_WAIT_MIN", "10"))
|
|
45
|
+
max_h = float(os.environ.get("AGENT_LIMIT_WAIT_MAX_H", "12"))
|
|
46
|
+
waited = 0.0
|
|
47
|
+
attempt = 0
|
|
48
|
+
while True:
|
|
49
|
+
attempt += 1
|
|
50
|
+
limit_hit = None
|
|
51
|
+
try:
|
|
52
|
+
async for message in query(prompt=prompt, options=options):
|
|
53
|
+
if isinstance(message, ResultMessage) and getattr(message, "is_error", False) \
|
|
54
|
+
and is_limit_error(getattr(message, "result", "")):
|
|
55
|
+
limit_hit = message.result
|
|
56
|
+
continue # swallow: the retry below replaces this result
|
|
57
|
+
yield message
|
|
58
|
+
except Exception as e: # transport-level failure carrying a limit message
|
|
59
|
+
if not is_limit_error(str(e)):
|
|
60
|
+
raise
|
|
61
|
+
limit_hit = str(e)
|
|
62
|
+
if limit_hit is None:
|
|
63
|
+
return
|
|
64
|
+
if waited / 3600 >= max_h:
|
|
65
|
+
raise AgentLimitExhausted(f"[{label}] usage limit still in force after {waited / 3600:.1f} h: {limit_hit}")
|
|
66
|
+
print(f"== [{label}] usage/rate limit (attempt {attempt}): {str(limit_hit)[:160]!r} — "
|
|
67
|
+
f"waiting {wait_min:.0f} min, {max_h - waited / 3600:.1f} h of wait budget left", flush=True)
|
|
68
|
+
t0 = time.time()
|
|
69
|
+
await asyncio.sleep(wait_min * 60)
|
|
70
|
+
waited += time.time() - t0
|