ecarsi 0.2.8__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ecarsi/__init__.py ADDED
@@ -0,0 +1,72 @@
1
+ """ecarsi — pluggable-harness tooling for the eca-rsi curation loop.
2
+
3
+ HARNESS env var selects the agent execution backend for every call in this
4
+ package (see ecarsi.harness): 'openai' (default — OpenAI Agents SDK driving
5
+ Doubao through Ark), 'deepseek' (DeepSeek Harness / dsh driving Doubao), or
6
+ 'claude' (claude_agent_sdk, spends Claude Code quota)."""
7
+
8
+ from __future__ import annotations
9
+
10
+ import os
11
+
12
+
13
+ def model() -> str:
14
+ """Model for every agent call in this package: MODEL env, else the
15
+ HARNESS-appropriate default — a bare model name is never portable
16
+ across backends. Same rule as osp/msp/zmip's harness.default_model()."""
17
+ from .harness import default_model
18
+
19
+ return default_model()
20
+
21
+
22
+ def agent_config() -> dict[str, str]:
23
+ """The two independent choices that must stay fixed within a run."""
24
+ from .harness import backend_name
25
+
26
+ return {"harness": backend_name(), "model": model()}
27
+
28
+
29
+ def effective_or_requested(fn) -> dict[str, str]:
30
+ """{"harness", "model"} for a manifest field recorded *after* an agent
31
+ call: the AgentConfig that actually produced the result if `fn` stashed
32
+ one (the `.last_effective_config` convention, same idea as `.last_cost` --
33
+ set by the caller of run_agent(), read here), else the plain env
34
+ snapshot from agent_config(). With a fallback pool (AGENT_MODEL_POOL) in
35
+ play, these can differ; without one they are always the same value.
36
+
37
+ Only usable where the manifest field is a *record* of what happened, not
38
+ an input to a pre-call resume/identity decision -- persample's own
39
+ config is built and hashed before its agent call can run at all, so it
40
+ correctly keeps using agent_config() unconditionally.
41
+ """
42
+ cfg = getattr(fn, "last_effective_config", None)
43
+ return cfg.as_manifest() if cfg is not None else agent_config()
44
+
45
+
46
+ def check_agent_config(recorded: dict, where: str) -> str | None:
47
+ """Compare a resumed stage's recorded {harness, model} against the
48
+ current selection; never blocks the resume (2026-09-10: switching
49
+ backend mid-run is a legitimate operator move -- round 3 shows the
50
+ current model's QC judgement is too lax, so the remaining rounds
51
+ continue with a stronger one -- dropped the --allow-agent-change guard
52
+ that used to require asking permission for it every time).
53
+
54
+ Returns the human-readable mismatch message (also printed) so a caller
55
+ that has a durable per-unit log can additionally record it there; None
56
+ when the config matches or predates harness/model recording (older
57
+ manifests remain resumable with a warning because their original choice
58
+ cannot be proved either way).
59
+ """
60
+ if "harness" not in recorded or "model" not in recorded:
61
+ print(f"[agent] {where} predates harness/model recording — resume cannot verify the old choice")
62
+ return None
63
+ want = agent_config()
64
+ got = {"harness": str(recorded["harness"]), "model": str(recorded["model"])}
65
+ if got == want:
66
+ return None
67
+ message = (
68
+ f"{where} used harness={got['harness']} model={got['model']}; current selection is "
69
+ f"harness={want['harness']} model={want['model']}"
70
+ )
71
+ print(f"[agent] {message} — continuing (mixed-model run)")
72
+ return message
ecarsi/__main__.py ADDED
@@ -0,0 +1,154 @@
1
+ """eca-rsi — the one entry point of the main line.
2
+
3
+ eca-rsi [--harness BACKEND] [--model MODEL] run <eca-pp-dir> <root> [--rounds N] [--cap 10] [--mirror DIR] [--serve [PORT]]
4
+ eca-rsi organize <eca-pp-dir> <root> [--mirror DIR]
5
+ eca-rsi persample <unit> [...] eca-rsi loop <unit> [...] (both also take --mirror DIR)
6
+ eca-rsi crosssample <unit> [round_dir] eca-rsi zoomin <unit> [round_dir]
7
+ eca-rsi ledger <unit> [round dirs] eca-rsi index <root|unit>
8
+ eca-rsi prune <root|unit> [--dry-run] (runs by itself after every release unless --no-prune)
9
+ eca-rsi serve [dir...] [--registry F] [--port] [--ngrok --domain D] [--auth U:P]
10
+ eca-rsi serve scan-add|remove|list|dump|reload ... (edit the registry file)
11
+ eca-rsi umapdata <h5ad> <out.json>
12
+
13
+ `run` chains everything for one dataset: organize the eca-pp products into
14
+ <root>, then for every analysis unit persample (osp) → loop (msp + zmip
15
+ rounds until the cell count converges) → release; with --serve it finally
16
+ adds <root> to the serve registry and serves it (and everything else in the
17
+ registry) in the foreground at http://127.0.0.1:PORT/<root-name>/ (add
18
+ --ngrok to publish; Ctrl-C to stop). --mirror DIR keeps a copy of <root>
19
+ on long-term storage: light files (pages, logs, manifests, reports, figures)
20
+ after every step, everything at release (ecarsi.mirror); DIR is remembered
21
+ in <root>/mirror.json, so resumes and single steps keep mirroring.
22
+ Every step resumes, so re-running the same command after an interruption
23
+ continues where it stopped. `python -m ecarsi ...` is the same thing.
24
+ The global --harness and --model options may appear before or after the
25
+ subcommand; explicit CLI values override HARNESS / MODEL environment values.
26
+ A resume with a different harness/model than an earlier stage used is
27
+ allowed (switching mid-run to a stronger model is a legitimate operator
28
+ move) and reported in progress.log / needs_review, never blocked.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ import argparse
34
+ import os
35
+ import sys
36
+ from pathlib import Path
37
+
38
+ from . import layout as L
39
+
40
+ STEPS = ("organize", "persample", "crosssample", "zoomin", "loop", "ledger", "index", "serve", "umapdata", "prune")
41
+
42
+
43
+ def _module(name: str):
44
+ import importlib
45
+
46
+ return importlib.import_module(f"ecarsi.{name}")
47
+
48
+
49
+ def run(argv: list[str]) -> int:
50
+ ap = argparse.ArgumentParser(prog="eca-rsi run", description="organize → persample → loop for every unit (→ serve)")
51
+ ap.add_argument("input", help="eca-pp output directory (standardize/standardized.h5ad + result.json per sample)")
52
+ ap.add_argument("root", help="run root; everything lands under <root>/units/<unit>/ (ecarsi.layout)")
53
+ ap.add_argument("--stop-after", choices=["organize", "persample"], help="validate the front pipeline without entering the loop")
54
+ ap.add_argument("--plan-json", help="explicit organize plan")
55
+ ap.add_argument("--rounds", type=int, default=None, help="fixed number of loop rounds (default: converge on cell count)")
56
+ ap.add_argument("--cap", type=int, default=None, help="loop safety cap (default 10)")
57
+ ap.add_argument("--force-reopen", action="store_true", help="continue past an existing release")
58
+ ap.add_argument("--no-prune", action="store_true", help="keep intermediate round h5ads after release (default: prune them)")
59
+ ap.add_argument("--mirror", metavar="DIR", help="keep a copy of <root> here: light files after every step, all of it at release")
60
+ ap.add_argument("--serve", nargs="?", const=8899, type=int, default=None, metavar="PORT",
61
+ help="after the run, add <root> to the serve registry and serve (foreground) on this port (default 8899)")
62
+ ap.add_argument("--ngrok", action="store_true", help="with --serve: also open an ngrok tunnel")
63
+ ap.add_argument("--domain", default=None, help="with --serve: reserved ngrok domain")
64
+ ap.add_argument("--auth", default=None, help="with --serve: web-level password USER:PASS")
65
+ a = ap.parse_args(argv)
66
+ root = Path(a.root).resolve()
67
+ if a.mirror:
68
+ _module("mirror").configure(root, a.mirror) # remembered in <root>/mirror.json; every step reads it there
69
+
70
+ organize_args = [a.input, str(root)]
71
+ if a.plan_json:
72
+ organize_args += ["--plan-json", a.plan_json]
73
+ rc = _module("organize").main(organize_args)
74
+ if rc or a.stop_after == "organize":
75
+ return rc
76
+ units = L.units(root)
77
+ if not units:
78
+ print(f"[eca-rsi] no analysis units under {L.units_root(root)}")
79
+ return 3
80
+ print(f"[eca-rsi] {len(units)} unit(s): " + ", ".join(u.name for u in units))
81
+ loop_args = []
82
+ if a.rounds is not None:
83
+ loop_args += ["--rounds", str(a.rounds)]
84
+ if a.cap is not None:
85
+ loop_args += ["--cap", str(a.cap)]
86
+ if a.force_reopen:
87
+ loop_args.append("--force-reopen")
88
+ if a.no_prune:
89
+ loop_args.append("--no-prune")
90
+ failed = []
91
+ for u in units:
92
+ print(f"\n[eca-rsi] ===== unit {u.name}: persample =====", flush=True)
93
+ rc = _module("persample").main([str(u)])
94
+ if rc:
95
+ failed.append((u.name, "persample", rc))
96
+ continue
97
+ if a.stop_after == "persample":
98
+ continue
99
+ print(f"\n[eca-rsi] ===== unit {u.name}: loop =====", flush=True)
100
+ rc = _module("loop").main([str(u), *loop_args])
101
+ if rc:
102
+ failed.append((u.name, "loop", rc))
103
+ _module("index").write_all(root)
104
+ for name, step, rc in failed:
105
+ print(f"[eca-rsi] FAILED {name} at {step} (rc={rc}) — re-run the same command to resume")
106
+ if a.serve is not None:
107
+ # record this root in the registry file (so any later `eca-rsi serve`
108
+ # shows it too), then serve everything in the registry in the
109
+ # foreground until Ctrl-C — the server itself keeps no state
110
+ serve = _module("serve")
111
+ try:
112
+ serve.Registry(serve.default_registry()).bind(root.name, root)
113
+ except ValueError as e:
114
+ print(f"[eca-rsi] not added to the registry ({e}); serving it for this process only")
115
+ serve_args = [str(root), "--port", str(a.serve)]
116
+ if a.ngrok:
117
+ serve_args.append("--ngrok")
118
+ if a.domain:
119
+ serve_args += ["--domain", a.domain]
120
+ if a.auth:
121
+ serve_args += ["--auth", a.auth]
122
+ print(f"[eca-rsi] serving {root.name} at http://127.0.0.1:{a.serve}/{root.name}/", flush=True)
123
+ rc = serve.main(serve_args)
124
+ return rc or (1 if failed else 0)
125
+ print(f"\n[eca-rsi] done — landing page: {root / L.INDEX} (eca-rsi serve scan-add {root}; eca-rsi serve)")
126
+ return 1 if failed else 0
127
+
128
+
129
+ def main(argv: list[str] | None = None) -> int:
130
+ from harness_bridge import configure_logging
131
+ configure_logging("ecarsi", stream=sys.stderr)
132
+ argv = sys.argv[1:] if argv is None else argv
133
+ runtime = argparse.ArgumentParser(add_help=False)
134
+ runtime.add_argument("--harness", choices=["deepseek", "openai", "claude"])
135
+ runtime.add_argument("--model")
136
+ selected, argv = runtime.parse_known_args(argv)
137
+ if selected.harness:
138
+ os.environ["HARNESS"] = selected.harness
139
+ if selected.model:
140
+ os.environ["MODEL"] = selected.model
141
+ if not argv or argv[0] in ("-h", "--help"):
142
+ print(__doc__)
143
+ return 0 if argv else 2
144
+ cmd, rest = argv[0], argv[1:]
145
+ if cmd == "run":
146
+ return run(rest)
147
+ if cmd in STEPS:
148
+ return _module(cmd).main(rest)
149
+ print(f"unknown command {cmd!r}; expected run or one of {', '.join(STEPS)}")
150
+ return 2
151
+
152
+
153
+ if __name__ == "__main__":
154
+ raise SystemExit(main())
ecarsi/agent_retry.py ADDED
@@ -0,0 +1,41 @@
1
+ """Retry wrapper for the cheap, side-effect-free agent-only calls (plan,
2
+ sample-column identification): under concurrent Slurm job start, many
3
+ `claude` CLI subprocesses initializing at once can blow the SDK's control
4
+ handshake ("Control request timeout: initialize") or die with a transient
5
+ connection error. These calls do nothing but read + return structured
6
+ output, so a bare retry is safe — no partial state to clean up.
7
+
8
+ Heavier steps (persample driving, crosssample, zoomin) are not wrapped here:
9
+ they write real files and are already resumable at the eca-rsi step level
10
+ (the sbatch wrapper retries `eca-rsi run` itself, which reuses that resume).
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import time
16
+ from typing import Callable, Coroutine, TypeVar
17
+
18
+ T = TypeVar("T")
19
+
20
+ MAX_ATTEMPTS = 6
21
+ BACKOFF_SECONDS = 20 # linear: 20s, 40s, 60s, 80s, 100s
22
+
23
+
24
+ def run_with_retry(coro_fn: Callable[[], Coroutine[object, object, T]], label: str) -> T:
25
+ """asyncio.run(coro_fn()) with retries on transient SDK/agent failures."""
26
+ import asyncio
27
+
28
+ last_exc: Exception | None = None
29
+ for attempt in range(1, MAX_ATTEMPTS + 1):
30
+ try:
31
+ return asyncio.run(coro_fn())
32
+ except Exception as e: # noqa: BLE001 - deliberately broad, see module docstring
33
+ last_exc = e
34
+ if attempt == MAX_ATTEMPTS:
35
+ break
36
+ wait = BACKOFF_SECONDS * attempt
37
+ print(f"[retry] {label} attempt {attempt}/{MAX_ATTEMPTS} failed "
38
+ f"({type(e).__name__}: {e}); retrying in {wait}s")
39
+ time.sleep(wait)
40
+ assert last_exc is not None
41
+ raise last_exc
ecarsi/cost.py ADDED
@@ -0,0 +1,184 @@
1
+ """ecarsi.cost — agent spend and token usage, recorded per step in progress.log
2
+ and summed at release.
3
+
4
+ The harness prints one of two per-run lines to stdout (both in this process
5
+ and inside the osp / msp / zmip kernels, which are subprocesses):
6
+ `== [label] agent cost: $X` (claude, real dollars) or
7
+ `== [label] ... N input / M output tokens ...` (openai/Doubao, no dollar figure
8
+ but real token counts). Nothing persisted that; it scrolled by in the Slurm
9
+ log. Now:
10
+
11
+ - kernel subprocesses are run through `run_streamed()`, which echoes their
12
+ output unchanged and, on every cost/token line, appends a `cost ...` event
13
+ to the unit's progress.log;
14
+ - ecarsi's own agent calls call `record()` with the harness result's
15
+ cost_usd/tokens_in/tokens_out;
16
+ - `summarize()` reads those events back (progress.log is the audit trail
17
+ anyway) and release/summary.md gets an "Agent cost" section from it.
18
+
19
+ A backend that reports neither (deepseek) simply produces no events — the
20
+ section then says so instead of pretending zero.
21
+
22
+ Separately, the bridge (agent-harness-bridge >= 0.2.8) prints one more line
23
+ for every backend regardless of whether it reports cost: `== [label]
24
+ resolved backend: harness=H model=M`. `run_streamed()` / persample's `_pump`
25
+ also catch that line and record it (`backend_events` / `round_backends`) --
26
+ this is how ecarsi finds out which model actually answered inside a kernel
27
+ subprocess (osp per-sample worker, msp/zmip round subprocess) when a
28
+ fallback pool is in play (eca-rsi#6); release/summary.md's "Backends per
29
+ round" section (eca-rsi#5) is built from it.
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import re
35
+ import subprocess
36
+ from collections import OrderedDict
37
+ from pathlib import Path
38
+
39
+ from . import layout as L
40
+
41
+ COST_RE = re.compile(r"(?:\[(?P<label>[^\]]*)\]\s*)?(?P<pre>[\w -]*?)\s*agent cost: \$(?P<usd>[0-9]+(?:\.[0-9]+)?)")
42
+ TOKEN_RE = re.compile(r"\[(?P<label>[^\]]*)\].*?(?P<tin>\d+) input / (?P<tout>\d+) output tokens")
43
+ EVENT_RE = re.compile(
44
+ r"^cost step=(?P<step>\S+)(?: usd=(?P<usd>[0-9.]+))?"
45
+ r"(?: tokens_in=(?P<tin>\d+))?(?: tokens_out=(?P<tout>\d+))?(?: label=(?P<label>.*))?$"
46
+ )
47
+ BACKEND_RE = re.compile(r"\[(?P<label>[^\]]*)\] resolved backend: harness=(?P<harness>\S+) model=(?P<model>\S+)")
48
+ BACKEND_EVENT_RE = re.compile(
49
+ r"^agent step=(?P<step>\S+) harness=(?P<harness>\S+) model=(?P<model>\S+)(?: label=(?P<label>.*))?$"
50
+ )
51
+ ROUND_RE = re.compile(r"^round(\d+)")
52
+
53
+
54
+ def record(unit: Path, step: str, usd: float | None, label: str = "",
55
+ tokens_in: int | None = None, tokens_out: int | None = None) -> None:
56
+ """One agent run's spend/usage -> progress.log (no-op when the backend gave neither)."""
57
+ if usd is None and tokens_in is None and tokens_out is None:
58
+ return
59
+ parts = [f"step={step}"]
60
+ if usd is not None:
61
+ parts.append(f"usd={usd:.4f}")
62
+ if tokens_in is not None:
63
+ parts.append(f"tokens_in={tokens_in}")
64
+ if tokens_out is not None:
65
+ parts.append(f"tokens_out={tokens_out}")
66
+ if label:
67
+ parts.append(f"label={label}")
68
+ L.log_event(unit, "cost " + " ".join(parts), echo=False)
69
+
70
+
71
+ def record_backend(unit: Path, step: str, harness: str, model: str, label: str = "") -> None:
72
+ """One agent run's {harness, model} -> progress.log, independent of the
73
+ cost/token event above (a single call prints both a cost-or-token line
74
+ and this line; recording them separately avoids double-counting `n` in
75
+ summarize())."""
76
+ line = f"agent step={step} harness={harness} model={model}" + (f" label={label}" if label else "")
77
+ L.log_event(unit, line, echo=False)
78
+
79
+
80
+ def _scan_line(unit: Path, step: str, line: str) -> None:
81
+ """Check one line of kernel subprocess stdout against every pattern this
82
+ module knows how to record; shared by run_streamed() and persample's
83
+ per-sample pump so both capture cost/token/backend lines the same way."""
84
+ m = COST_RE.search(line)
85
+ if m:
86
+ label = (m.group("label") or m.group("pre") or "").strip()
87
+ record(unit, step, float(m.group("usd")), label)
88
+ m = TOKEN_RE.search(line)
89
+ if m:
90
+ record(unit, step, None, m.group("label").strip(), int(m.group("tin")), int(m.group("tout")))
91
+ m = BACKEND_RE.search(line)
92
+ if m:
93
+ record_backend(unit, step, m.group("harness"), m.group("model"), m.group("label").strip())
94
+
95
+
96
+ def run_streamed(cmd: str, unit: Path, step: str) -> int:
97
+ """subprocess.run(cmd, shell=True) with the output passed through line by
98
+ line and every harness cost/token/backend line also recorded against `step`."""
99
+ proc = subprocess.Popen(cmd, shell=True, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, bufsize=1)
100
+ assert proc.stdout is not None
101
+ for line in proc.stdout:
102
+ print(line, end="", flush=True)
103
+ _scan_line(unit, step, line)
104
+ return proc.wait()
105
+
106
+
107
+ def events(unit: Path) -> list[dict]:
108
+ out = []
109
+ for ts, ev in L.read_log(unit):
110
+ m = EVENT_RE.match(ev)
111
+ if m:
112
+ out.append({
113
+ "time": ts, "step": m.group("step"),
114
+ "usd": float(m.group("usd")) if m.group("usd") else None,
115
+ "tokens_in": int(m.group("tin")) if m.group("tin") else None,
116
+ "tokens_out": int(m.group("tout")) if m.group("tout") else None,
117
+ "label": m.group("label") or "",
118
+ })
119
+ return out
120
+
121
+
122
+ def summarize(unit: Path) -> dict:
123
+ """{"total": float, "tokens_in": int, "tokens_out": int, "n": int,
124
+ "by_step": {step: {"usd", "tokens_in", "tokens_out", "n"}}} in first-seen step order."""
125
+ by: "OrderedDict[str, dict]" = OrderedDict()
126
+ total, tin_total, tout_total, n = 0.0, 0, 0, 0
127
+ for e in events(unit):
128
+ d = by.setdefault(e["step"], {"usd": 0.0, "tokens_in": 0, "tokens_out": 0, "n": 0})
129
+ d["usd"] += e["usd"] or 0.0
130
+ d["tokens_in"] += e["tokens_in"] or 0
131
+ d["tokens_out"] += e["tokens_out"] or 0
132
+ d["n"] += 1
133
+ total += e["usd"] or 0.0
134
+ tin_total += e["tokens_in"] or 0
135
+ tout_total += e["tokens_out"] or 0
136
+ n += 1
137
+ return {
138
+ "total": round(total, 2), "tokens_in": tin_total, "tokens_out": tout_total, "n": n,
139
+ "by_step": {k: {"usd": round(v["usd"], 2), "tokens_in": v["tokens_in"], "tokens_out": v["tokens_out"], "n": v["n"]}
140
+ for k, v in by.items()},
141
+ }
142
+
143
+
144
+ def backend_events(unit: Path) -> list[dict]:
145
+ out = []
146
+ for ts, ev in L.read_log(unit):
147
+ m = BACKEND_EVENT_RE.match(ev)
148
+ if m:
149
+ out.append({"time": ts, "step": m.group("step"), "harness": m.group("harness"),
150
+ "model": m.group("model"), "label": m.group("label") or ""})
151
+ return out
152
+
153
+
154
+ def round_backends(unit: Path) -> "OrderedDict[str, list[str]]":
155
+ """{round label ("round03" or "front" for pre-round steps): sorted
156
+ ["harness:model", ...] seen there}, in first-seen order. Distinct
157
+ configs within one round are normal (osp/msp/zmip can each resolve a
158
+ fallback pool differently) -- eca-rsi#5 wants this listed as fact, not
159
+ flagged as a mismatch the way check_agent_config's resume check is."""
160
+ by: "OrderedDict[str, set]" = OrderedDict()
161
+ for e in backend_events(unit):
162
+ m = ROUND_RE.match(e["step"])
163
+ key = m.group(0) if m else "front"
164
+ by.setdefault(key, set()).add(f"{e['harness']}:{e['model']}")
165
+ return OrderedDict((k, sorted(v)) for k, v in by.items())
166
+
167
+
168
+ def backend_summary_md(unit: Path) -> list[str]:
169
+ by_round = round_backends(unit)
170
+ if not by_round:
171
+ return ["## Backends per round", "", "no backend reported (predates backend logging, or no fallback pool was in play)"]
172
+ rows = [f"| {k} | {', '.join(v)} |" for k, v in by_round.items()]
173
+ return ["## Backends per round", "", "| round | backend(s) used |", "|---|---|", *rows]
174
+
175
+
176
+ def summary_md(unit: Path) -> list[str]:
177
+ s = summarize(unit)
178
+ if not s["n"]:
179
+ return ["## Agent cost", "", "no cost/token usage reported (backend does not report it, or the run predates cost logging)"]
180
+ rows = [f"| {step} | {v['n']} | ${v['usd']:.2f} | {v['tokens_in']} | {v['tokens_out']} |" for step, v in s["by_step"].items()]
181
+ return ["## Agent cost", "",
182
+ f"Total: **${s['total']:.2f}**, {s['tokens_in']} input / {s['tokens_out']} output tokens over "
183
+ f"{s['n']} agent run(s) — from `cost` events in progress.log", "",
184
+ "| step | agent runs | USD | tokens in | tokens out |", "|---|---|---|---|---|", *rows]