ecarsi 0.2.8__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ecarsi/__init__.py +72 -0
- ecarsi/__main__.py +154 -0
- ecarsi/agent_retry.py +41 -0
- ecarsi/cost.py +184 -0
- ecarsi/crosssample.py +439 -0
- ecarsi/design.py +107 -0
- ecarsi/downstream.py +368 -0
- ecarsi/execute.py +259 -0
- ecarsi/harness.py +59 -0
- ecarsi/index.py +981 -0
- ecarsi/layout.py +249 -0
- ecarsi/ledger.py +474 -0
- ecarsi/loop.py +470 -0
- ecarsi/mirror.py +164 -0
- ecarsi/organize.py +163 -0
- ecarsi/osp_contract.py +158 -0
- ecarsi/osp_worker.py +120 -0
- ecarsi/persample.py +580 -0
- ecarsi/plan.py +160 -0
- ecarsi/policies.py +263 -0
- ecarsi/prompts/batch_key.md +17 -0
- ecarsi/prompts/plan.md +52 -0
- ecarsi/prompts/sample_column.md +62 -0
- ecarsi/prompts/sample_inclusion.md +38 -0
- ecarsi/prune.py +158 -0
- ecarsi/release_state.py +132 -0
- ecarsi/resources.py +122 -0
- ecarsi/review.py +403 -0
- ecarsi/run_state.py +103 -0
- ecarsi/sample_mapping.py +172 -0
- ecarsi/serve.py +1339 -0
- ecarsi/umapdata.py +141 -0
- ecarsi/upstream.py +190 -0
- ecarsi/zoomin.py +125 -0
- ecarsi-0.2.8.dist-info/METADATA +411 -0
- ecarsi-0.2.8.dist-info/RECORD +39 -0
- ecarsi-0.2.8.dist-info/WHEEL +5 -0
- ecarsi-0.2.8.dist-info/entry_points.txt +2 -0
- ecarsi-0.2.8.dist-info/top_level.txt +1 -0
ecarsi/__init__.py
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""ecarsi — pluggable-harness tooling for the eca-rsi curation loop.
|
|
2
|
+
|
|
3
|
+
HARNESS env var selects the agent execution backend for every call in this
|
|
4
|
+
package (see ecarsi.harness): 'openai' (default — OpenAI Agents SDK driving
|
|
5
|
+
Doubao through Ark), 'deepseek' (DeepSeek Harness / dsh driving Doubao), or
|
|
6
|
+
'claude' (claude_agent_sdk, spends Claude Code quota)."""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import os
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def model() -> str:
|
|
14
|
+
"""Model for every agent call in this package: MODEL env, else the
|
|
15
|
+
HARNESS-appropriate default — a bare model name is never portable
|
|
16
|
+
across backends. Same rule as osp/msp/zmip's harness.default_model()."""
|
|
17
|
+
from .harness import default_model
|
|
18
|
+
|
|
19
|
+
return default_model()
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def agent_config() -> dict[str, str]:
|
|
23
|
+
"""The two independent choices that must stay fixed within a run."""
|
|
24
|
+
from .harness import backend_name
|
|
25
|
+
|
|
26
|
+
return {"harness": backend_name(), "model": model()}
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def effective_or_requested(fn) -> dict[str, str]:
|
|
30
|
+
"""{"harness", "model"} for a manifest field recorded *after* an agent
|
|
31
|
+
call: the AgentConfig that actually produced the result if `fn` stashed
|
|
32
|
+
one (the `.last_effective_config` convention, same idea as `.last_cost` --
|
|
33
|
+
set by the caller of run_agent(), read here), else the plain env
|
|
34
|
+
snapshot from agent_config(). With a fallback pool (AGENT_MODEL_POOL) in
|
|
35
|
+
play, these can differ; without one they are always the same value.
|
|
36
|
+
|
|
37
|
+
Only usable where the manifest field is a *record* of what happened, not
|
|
38
|
+
an input to a pre-call resume/identity decision -- persample's own
|
|
39
|
+
config is built and hashed before its agent call can run at all, so it
|
|
40
|
+
correctly keeps using agent_config() unconditionally.
|
|
41
|
+
"""
|
|
42
|
+
cfg = getattr(fn, "last_effective_config", None)
|
|
43
|
+
return cfg.as_manifest() if cfg is not None else agent_config()
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def check_agent_config(recorded: dict, where: str) -> str | None:
|
|
47
|
+
"""Compare a resumed stage's recorded {harness, model} against the
|
|
48
|
+
current selection; never blocks the resume (2026-09-10: switching
|
|
49
|
+
backend mid-run is a legitimate operator move -- round 3 shows the
|
|
50
|
+
current model's QC judgement is too lax, so the remaining rounds
|
|
51
|
+
continue with a stronger one -- dropped the --allow-agent-change guard
|
|
52
|
+
that used to require asking permission for it every time).
|
|
53
|
+
|
|
54
|
+
Returns the human-readable mismatch message (also printed) so a caller
|
|
55
|
+
that has a durable per-unit log can additionally record it there; None
|
|
56
|
+
when the config matches or predates harness/model recording (older
|
|
57
|
+
manifests remain resumable with a warning because their original choice
|
|
58
|
+
cannot be proved either way).
|
|
59
|
+
"""
|
|
60
|
+
if "harness" not in recorded or "model" not in recorded:
|
|
61
|
+
print(f"[agent] {where} predates harness/model recording — resume cannot verify the old choice")
|
|
62
|
+
return None
|
|
63
|
+
want = agent_config()
|
|
64
|
+
got = {"harness": str(recorded["harness"]), "model": str(recorded["model"])}
|
|
65
|
+
if got == want:
|
|
66
|
+
return None
|
|
67
|
+
message = (
|
|
68
|
+
f"{where} used harness={got['harness']} model={got['model']}; current selection is "
|
|
69
|
+
f"harness={want['harness']} model={want['model']}"
|
|
70
|
+
)
|
|
71
|
+
print(f"[agent] {message} — continuing (mixed-model run)")
|
|
72
|
+
return message
|
ecarsi/__main__.py
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"""eca-rsi — the one entry point of the main line.
|
|
2
|
+
|
|
3
|
+
eca-rsi [--harness BACKEND] [--model MODEL] run <eca-pp-dir> <root> [--rounds N] [--cap 10] [--mirror DIR] [--serve [PORT]]
|
|
4
|
+
eca-rsi organize <eca-pp-dir> <root> [--mirror DIR]
|
|
5
|
+
eca-rsi persample <unit> [...] eca-rsi loop <unit> [...] (both also take --mirror DIR)
|
|
6
|
+
eca-rsi crosssample <unit> [round_dir] eca-rsi zoomin <unit> [round_dir]
|
|
7
|
+
eca-rsi ledger <unit> [round dirs] eca-rsi index <root|unit>
|
|
8
|
+
eca-rsi prune <root|unit> [--dry-run] (runs by itself after every release unless --no-prune)
|
|
9
|
+
eca-rsi serve [dir...] [--registry F] [--port] [--ngrok --domain D] [--auth U:P]
|
|
10
|
+
eca-rsi serve scan-add|remove|list|dump|reload ... (edit the registry file)
|
|
11
|
+
eca-rsi umapdata <h5ad> <out.json>
|
|
12
|
+
|
|
13
|
+
`run` chains everything for one dataset: organize the eca-pp products into
|
|
14
|
+
<root>, then for every analysis unit persample (osp) → loop (msp + zmip
|
|
15
|
+
rounds until the cell count converges) → release; with --serve it finally
|
|
16
|
+
adds <root> to the serve registry and serves it (and everything else in the
|
|
17
|
+
registry) in the foreground at http://127.0.0.1:PORT/<root-name>/ (add
|
|
18
|
+
--ngrok to publish; Ctrl-C to stop). --mirror DIR keeps a copy of <root>
|
|
19
|
+
on long-term storage: light files (pages, logs, manifests, reports, figures)
|
|
20
|
+
after every step, everything at release (ecarsi.mirror); DIR is remembered
|
|
21
|
+
in <root>/mirror.json, so resumes and single steps keep mirroring.
|
|
22
|
+
Every step resumes, so re-running the same command after an interruption
|
|
23
|
+
continues where it stopped. `python -m ecarsi ...` is the same thing.
|
|
24
|
+
The global --harness and --model options may appear before or after the
|
|
25
|
+
subcommand; explicit CLI values override HARNESS / MODEL environment values.
|
|
26
|
+
A resume with a different harness/model than an earlier stage used is
|
|
27
|
+
allowed (switching mid-run to a stronger model is a legitimate operator
|
|
28
|
+
move) and reported in progress.log / needs_review, never blocked.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import argparse
|
|
34
|
+
import os
|
|
35
|
+
import sys
|
|
36
|
+
from pathlib import Path
|
|
37
|
+
|
|
38
|
+
from . import layout as L
|
|
39
|
+
|
|
40
|
+
STEPS = ("organize", "persample", "crosssample", "zoomin", "loop", "ledger", "index", "serve", "umapdata", "prune")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _module(name: str):
|
|
44
|
+
import importlib
|
|
45
|
+
|
|
46
|
+
return importlib.import_module(f"ecarsi.{name}")
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def run(argv: list[str]) -> int:
|
|
50
|
+
ap = argparse.ArgumentParser(prog="eca-rsi run", description="organize → persample → loop for every unit (→ serve)")
|
|
51
|
+
ap.add_argument("input", help="eca-pp output directory (standardize/standardized.h5ad + result.json per sample)")
|
|
52
|
+
ap.add_argument("root", help="run root; everything lands under <root>/units/<unit>/ (ecarsi.layout)")
|
|
53
|
+
ap.add_argument("--stop-after", choices=["organize", "persample"], help="validate the front pipeline without entering the loop")
|
|
54
|
+
ap.add_argument("--plan-json", help="explicit organize plan")
|
|
55
|
+
ap.add_argument("--rounds", type=int, default=None, help="fixed number of loop rounds (default: converge on cell count)")
|
|
56
|
+
ap.add_argument("--cap", type=int, default=None, help="loop safety cap (default 10)")
|
|
57
|
+
ap.add_argument("--force-reopen", action="store_true", help="continue past an existing release")
|
|
58
|
+
ap.add_argument("--no-prune", action="store_true", help="keep intermediate round h5ads after release (default: prune them)")
|
|
59
|
+
ap.add_argument("--mirror", metavar="DIR", help="keep a copy of <root> here: light files after every step, all of it at release")
|
|
60
|
+
ap.add_argument("--serve", nargs="?", const=8899, type=int, default=None, metavar="PORT",
|
|
61
|
+
help="after the run, add <root> to the serve registry and serve (foreground) on this port (default 8899)")
|
|
62
|
+
ap.add_argument("--ngrok", action="store_true", help="with --serve: also open an ngrok tunnel")
|
|
63
|
+
ap.add_argument("--domain", default=None, help="with --serve: reserved ngrok domain")
|
|
64
|
+
ap.add_argument("--auth", default=None, help="with --serve: web-level password USER:PASS")
|
|
65
|
+
a = ap.parse_args(argv)
|
|
66
|
+
root = Path(a.root).resolve()
|
|
67
|
+
if a.mirror:
|
|
68
|
+
_module("mirror").configure(root, a.mirror) # remembered in <root>/mirror.json; every step reads it there
|
|
69
|
+
|
|
70
|
+
organize_args = [a.input, str(root)]
|
|
71
|
+
if a.plan_json:
|
|
72
|
+
organize_args += ["--plan-json", a.plan_json]
|
|
73
|
+
rc = _module("organize").main(organize_args)
|
|
74
|
+
if rc or a.stop_after == "organize":
|
|
75
|
+
return rc
|
|
76
|
+
units = L.units(root)
|
|
77
|
+
if not units:
|
|
78
|
+
print(f"[eca-rsi] no analysis units under {L.units_root(root)}")
|
|
79
|
+
return 3
|
|
80
|
+
print(f"[eca-rsi] {len(units)} unit(s): " + ", ".join(u.name for u in units))
|
|
81
|
+
loop_args = []
|
|
82
|
+
if a.rounds is not None:
|
|
83
|
+
loop_args += ["--rounds", str(a.rounds)]
|
|
84
|
+
if a.cap is not None:
|
|
85
|
+
loop_args += ["--cap", str(a.cap)]
|
|
86
|
+
if a.force_reopen:
|
|
87
|
+
loop_args.append("--force-reopen")
|
|
88
|
+
if a.no_prune:
|
|
89
|
+
loop_args.append("--no-prune")
|
|
90
|
+
failed = []
|
|
91
|
+
for u in units:
|
|
92
|
+
print(f"\n[eca-rsi] ===== unit {u.name}: persample =====", flush=True)
|
|
93
|
+
rc = _module("persample").main([str(u)])
|
|
94
|
+
if rc:
|
|
95
|
+
failed.append((u.name, "persample", rc))
|
|
96
|
+
continue
|
|
97
|
+
if a.stop_after == "persample":
|
|
98
|
+
continue
|
|
99
|
+
print(f"\n[eca-rsi] ===== unit {u.name}: loop =====", flush=True)
|
|
100
|
+
rc = _module("loop").main([str(u), *loop_args])
|
|
101
|
+
if rc:
|
|
102
|
+
failed.append((u.name, "loop", rc))
|
|
103
|
+
_module("index").write_all(root)
|
|
104
|
+
for name, step, rc in failed:
|
|
105
|
+
print(f"[eca-rsi] FAILED {name} at {step} (rc={rc}) — re-run the same command to resume")
|
|
106
|
+
if a.serve is not None:
|
|
107
|
+
# record this root in the registry file (so any later `eca-rsi serve`
|
|
108
|
+
# shows it too), then serve everything in the registry in the
|
|
109
|
+
# foreground until Ctrl-C — the server itself keeps no state
|
|
110
|
+
serve = _module("serve")
|
|
111
|
+
try:
|
|
112
|
+
serve.Registry(serve.default_registry()).bind(root.name, root)
|
|
113
|
+
except ValueError as e:
|
|
114
|
+
print(f"[eca-rsi] not added to the registry ({e}); serving it for this process only")
|
|
115
|
+
serve_args = [str(root), "--port", str(a.serve)]
|
|
116
|
+
if a.ngrok:
|
|
117
|
+
serve_args.append("--ngrok")
|
|
118
|
+
if a.domain:
|
|
119
|
+
serve_args += ["--domain", a.domain]
|
|
120
|
+
if a.auth:
|
|
121
|
+
serve_args += ["--auth", a.auth]
|
|
122
|
+
print(f"[eca-rsi] serving {root.name} at http://127.0.0.1:{a.serve}/{root.name}/", flush=True)
|
|
123
|
+
rc = serve.main(serve_args)
|
|
124
|
+
return rc or (1 if failed else 0)
|
|
125
|
+
print(f"\n[eca-rsi] done — landing page: {root / L.INDEX} (eca-rsi serve scan-add {root}; eca-rsi serve)")
|
|
126
|
+
return 1 if failed else 0
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def main(argv: list[str] | None = None) -> int:
|
|
130
|
+
from harness_bridge import configure_logging
|
|
131
|
+
configure_logging("ecarsi", stream=sys.stderr)
|
|
132
|
+
argv = sys.argv[1:] if argv is None else argv
|
|
133
|
+
runtime = argparse.ArgumentParser(add_help=False)
|
|
134
|
+
runtime.add_argument("--harness", choices=["deepseek", "openai", "claude"])
|
|
135
|
+
runtime.add_argument("--model")
|
|
136
|
+
selected, argv = runtime.parse_known_args(argv)
|
|
137
|
+
if selected.harness:
|
|
138
|
+
os.environ["HARNESS"] = selected.harness
|
|
139
|
+
if selected.model:
|
|
140
|
+
os.environ["MODEL"] = selected.model
|
|
141
|
+
if not argv or argv[0] in ("-h", "--help"):
|
|
142
|
+
print(__doc__)
|
|
143
|
+
return 0 if argv else 2
|
|
144
|
+
cmd, rest = argv[0], argv[1:]
|
|
145
|
+
if cmd == "run":
|
|
146
|
+
return run(rest)
|
|
147
|
+
if cmd in STEPS:
|
|
148
|
+
return _module(cmd).main(rest)
|
|
149
|
+
print(f"unknown command {cmd!r}; expected run or one of {', '.join(STEPS)}")
|
|
150
|
+
return 2
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
if __name__ == "__main__":
|
|
154
|
+
raise SystemExit(main())
|
ecarsi/agent_retry.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Retry wrapper for the cheap, side-effect-free agent-only calls (plan,
|
|
2
|
+
sample-column identification): under concurrent Slurm job start, many
|
|
3
|
+
`claude` CLI subprocesses initializing at once can blow the SDK's control
|
|
4
|
+
handshake ("Control request timeout: initialize") or die with a transient
|
|
5
|
+
connection error. These calls do nothing but read + return structured
|
|
6
|
+
output, so a bare retry is safe — no partial state to clean up.
|
|
7
|
+
|
|
8
|
+
Heavier steps (persample driving, crosssample, zoomin) are not wrapped here:
|
|
9
|
+
they write real files and are already resumable at the eca-rsi step level
|
|
10
|
+
(the sbatch wrapper retries `eca-rsi run` itself, which reuses that resume).
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import time
|
|
16
|
+
from typing import Callable, Coroutine, TypeVar
|
|
17
|
+
|
|
18
|
+
T = TypeVar("T")
|
|
19
|
+
|
|
20
|
+
MAX_ATTEMPTS = 6
|
|
21
|
+
BACKOFF_SECONDS = 20 # linear: 20s, 40s, 60s, 80s, 100s
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def run_with_retry(coro_fn: Callable[[], Coroutine[object, object, T]], label: str) -> T:
|
|
25
|
+
"""asyncio.run(coro_fn()) with retries on transient SDK/agent failures."""
|
|
26
|
+
import asyncio
|
|
27
|
+
|
|
28
|
+
last_exc: Exception | None = None
|
|
29
|
+
for attempt in range(1, MAX_ATTEMPTS + 1):
|
|
30
|
+
try:
|
|
31
|
+
return asyncio.run(coro_fn())
|
|
32
|
+
except Exception as e: # noqa: BLE001 - deliberately broad, see module docstring
|
|
33
|
+
last_exc = e
|
|
34
|
+
if attempt == MAX_ATTEMPTS:
|
|
35
|
+
break
|
|
36
|
+
wait = BACKOFF_SECONDS * attempt
|
|
37
|
+
print(f"[retry] {label} attempt {attempt}/{MAX_ATTEMPTS} failed "
|
|
38
|
+
f"({type(e).__name__}: {e}); retrying in {wait}s")
|
|
39
|
+
time.sleep(wait)
|
|
40
|
+
assert last_exc is not None
|
|
41
|
+
raise last_exc
|
ecarsi/cost.py
ADDED
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
"""ecarsi.cost — agent spend and token usage, recorded per step in progress.log
|
|
2
|
+
and summed at release.
|
|
3
|
+
|
|
4
|
+
The harness prints one of two per-run lines to stdout (both in this process
|
|
5
|
+
and inside the osp / msp / zmip kernels, which are subprocesses):
|
|
6
|
+
`== [label] agent cost: $X` (claude, real dollars) or
|
|
7
|
+
`== [label] ... N input / M output tokens ...` (openai/Doubao, no dollar figure
|
|
8
|
+
but real token counts). Nothing persisted that; it scrolled by in the Slurm
|
|
9
|
+
log. Now:
|
|
10
|
+
|
|
11
|
+
- kernel subprocesses are run through `run_streamed()`, which echoes their
|
|
12
|
+
output unchanged and, on every cost/token line, appends a `cost ...` event
|
|
13
|
+
to the unit's progress.log;
|
|
14
|
+
- ecarsi's own agent calls call `record()` with the harness result's
|
|
15
|
+
cost_usd/tokens_in/tokens_out;
|
|
16
|
+
- `summarize()` reads those events back (progress.log is the audit trail
|
|
17
|
+
anyway) and release/summary.md gets an "Agent cost" section from it.
|
|
18
|
+
|
|
19
|
+
A backend that reports neither (deepseek) simply produces no events — the
|
|
20
|
+
section then says so instead of pretending zero.
|
|
21
|
+
|
|
22
|
+
Separately, the bridge (agent-harness-bridge >= 0.2.8) prints one more line
|
|
23
|
+
for every backend regardless of whether it reports cost: `== [label]
|
|
24
|
+
resolved backend: harness=H model=M`. `run_streamed()` / persample's `_pump`
|
|
25
|
+
also catch that line and record it (`backend_events` / `round_backends`) --
|
|
26
|
+
this is how ecarsi finds out which model actually answered inside a kernel
|
|
27
|
+
subprocess (osp per-sample worker, msp/zmip round subprocess) when a
|
|
28
|
+
fallback pool is in play (eca-rsi#6); release/summary.md's "Backends per
|
|
29
|
+
round" section (eca-rsi#5) is built from it.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import re
|
|
35
|
+
import subprocess
|
|
36
|
+
from collections import OrderedDict
|
|
37
|
+
from pathlib import Path
|
|
38
|
+
|
|
39
|
+
from . import layout as L
|
|
40
|
+
|
|
41
|
+
COST_RE = re.compile(r"(?:\[(?P<label>[^\]]*)\]\s*)?(?P<pre>[\w -]*?)\s*agent cost: \$(?P<usd>[0-9]+(?:\.[0-9]+)?)")
|
|
42
|
+
TOKEN_RE = re.compile(r"\[(?P<label>[^\]]*)\].*?(?P<tin>\d+) input / (?P<tout>\d+) output tokens")
|
|
43
|
+
EVENT_RE = re.compile(
|
|
44
|
+
r"^cost step=(?P<step>\S+)(?: usd=(?P<usd>[0-9.]+))?"
|
|
45
|
+
r"(?: tokens_in=(?P<tin>\d+))?(?: tokens_out=(?P<tout>\d+))?(?: label=(?P<label>.*))?$"
|
|
46
|
+
)
|
|
47
|
+
BACKEND_RE = re.compile(r"\[(?P<label>[^\]]*)\] resolved backend: harness=(?P<harness>\S+) model=(?P<model>\S+)")
|
|
48
|
+
BACKEND_EVENT_RE = re.compile(
|
|
49
|
+
r"^agent step=(?P<step>\S+) harness=(?P<harness>\S+) model=(?P<model>\S+)(?: label=(?P<label>.*))?$"
|
|
50
|
+
)
|
|
51
|
+
ROUND_RE = re.compile(r"^round(\d+)")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def record(unit: Path, step: str, usd: float | None, label: str = "",
|
|
55
|
+
tokens_in: int | None = None, tokens_out: int | None = None) -> None:
|
|
56
|
+
"""One agent run's spend/usage -> progress.log (no-op when the backend gave neither)."""
|
|
57
|
+
if usd is None and tokens_in is None and tokens_out is None:
|
|
58
|
+
return
|
|
59
|
+
parts = [f"step={step}"]
|
|
60
|
+
if usd is not None:
|
|
61
|
+
parts.append(f"usd={usd:.4f}")
|
|
62
|
+
if tokens_in is not None:
|
|
63
|
+
parts.append(f"tokens_in={tokens_in}")
|
|
64
|
+
if tokens_out is not None:
|
|
65
|
+
parts.append(f"tokens_out={tokens_out}")
|
|
66
|
+
if label:
|
|
67
|
+
parts.append(f"label={label}")
|
|
68
|
+
L.log_event(unit, "cost " + " ".join(parts), echo=False)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def record_backend(unit: Path, step: str, harness: str, model: str, label: str = "") -> None:
|
|
72
|
+
"""One agent run's {harness, model} -> progress.log, independent of the
|
|
73
|
+
cost/token event above (a single call prints both a cost-or-token line
|
|
74
|
+
and this line; recording them separately avoids double-counting `n` in
|
|
75
|
+
summarize())."""
|
|
76
|
+
line = f"agent step={step} harness={harness} model={model}" + (f" label={label}" if label else "")
|
|
77
|
+
L.log_event(unit, line, echo=False)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _scan_line(unit: Path, step: str, line: str) -> None:
|
|
81
|
+
"""Check one line of kernel subprocess stdout against every pattern this
|
|
82
|
+
module knows how to record; shared by run_streamed() and persample's
|
|
83
|
+
per-sample pump so both capture cost/token/backend lines the same way."""
|
|
84
|
+
m = COST_RE.search(line)
|
|
85
|
+
if m:
|
|
86
|
+
label = (m.group("label") or m.group("pre") or "").strip()
|
|
87
|
+
record(unit, step, float(m.group("usd")), label)
|
|
88
|
+
m = TOKEN_RE.search(line)
|
|
89
|
+
if m:
|
|
90
|
+
record(unit, step, None, m.group("label").strip(), int(m.group("tin")), int(m.group("tout")))
|
|
91
|
+
m = BACKEND_RE.search(line)
|
|
92
|
+
if m:
|
|
93
|
+
record_backend(unit, step, m.group("harness"), m.group("model"), m.group("label").strip())
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def run_streamed(cmd: str, unit: Path, step: str) -> int:
|
|
97
|
+
"""subprocess.run(cmd, shell=True) with the output passed through line by
|
|
98
|
+
line and every harness cost/token/backend line also recorded against `step`."""
|
|
99
|
+
proc = subprocess.Popen(cmd, shell=True, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, bufsize=1)
|
|
100
|
+
assert proc.stdout is not None
|
|
101
|
+
for line in proc.stdout:
|
|
102
|
+
print(line, end="", flush=True)
|
|
103
|
+
_scan_line(unit, step, line)
|
|
104
|
+
return proc.wait()
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def events(unit: Path) -> list[dict]:
|
|
108
|
+
out = []
|
|
109
|
+
for ts, ev in L.read_log(unit):
|
|
110
|
+
m = EVENT_RE.match(ev)
|
|
111
|
+
if m:
|
|
112
|
+
out.append({
|
|
113
|
+
"time": ts, "step": m.group("step"),
|
|
114
|
+
"usd": float(m.group("usd")) if m.group("usd") else None,
|
|
115
|
+
"tokens_in": int(m.group("tin")) if m.group("tin") else None,
|
|
116
|
+
"tokens_out": int(m.group("tout")) if m.group("tout") else None,
|
|
117
|
+
"label": m.group("label") or "",
|
|
118
|
+
})
|
|
119
|
+
return out
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def summarize(unit: Path) -> dict:
|
|
123
|
+
"""{"total": float, "tokens_in": int, "tokens_out": int, "n": int,
|
|
124
|
+
"by_step": {step: {"usd", "tokens_in", "tokens_out", "n"}}} in first-seen step order."""
|
|
125
|
+
by: "OrderedDict[str, dict]" = OrderedDict()
|
|
126
|
+
total, tin_total, tout_total, n = 0.0, 0, 0, 0
|
|
127
|
+
for e in events(unit):
|
|
128
|
+
d = by.setdefault(e["step"], {"usd": 0.0, "tokens_in": 0, "tokens_out": 0, "n": 0})
|
|
129
|
+
d["usd"] += e["usd"] or 0.0
|
|
130
|
+
d["tokens_in"] += e["tokens_in"] or 0
|
|
131
|
+
d["tokens_out"] += e["tokens_out"] or 0
|
|
132
|
+
d["n"] += 1
|
|
133
|
+
total += e["usd"] or 0.0
|
|
134
|
+
tin_total += e["tokens_in"] or 0
|
|
135
|
+
tout_total += e["tokens_out"] or 0
|
|
136
|
+
n += 1
|
|
137
|
+
return {
|
|
138
|
+
"total": round(total, 2), "tokens_in": tin_total, "tokens_out": tout_total, "n": n,
|
|
139
|
+
"by_step": {k: {"usd": round(v["usd"], 2), "tokens_in": v["tokens_in"], "tokens_out": v["tokens_out"], "n": v["n"]}
|
|
140
|
+
for k, v in by.items()},
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def backend_events(unit: Path) -> list[dict]:
|
|
145
|
+
out = []
|
|
146
|
+
for ts, ev in L.read_log(unit):
|
|
147
|
+
m = BACKEND_EVENT_RE.match(ev)
|
|
148
|
+
if m:
|
|
149
|
+
out.append({"time": ts, "step": m.group("step"), "harness": m.group("harness"),
|
|
150
|
+
"model": m.group("model"), "label": m.group("label") or ""})
|
|
151
|
+
return out
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def round_backends(unit: Path) -> "OrderedDict[str, list[str]]":
|
|
155
|
+
"""{round label ("round03" or "front" for pre-round steps): sorted
|
|
156
|
+
["harness:model", ...] seen there}, in first-seen order. Distinct
|
|
157
|
+
configs within one round are normal (osp/msp/zmip can each resolve a
|
|
158
|
+
fallback pool differently) -- eca-rsi#5 wants this listed as fact, not
|
|
159
|
+
flagged as a mismatch the way check_agent_config's resume check is."""
|
|
160
|
+
by: "OrderedDict[str, set]" = OrderedDict()
|
|
161
|
+
for e in backend_events(unit):
|
|
162
|
+
m = ROUND_RE.match(e["step"])
|
|
163
|
+
key = m.group(0) if m else "front"
|
|
164
|
+
by.setdefault(key, set()).add(f"{e['harness']}:{e['model']}")
|
|
165
|
+
return OrderedDict((k, sorted(v)) for k, v in by.items())
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def backend_summary_md(unit: Path) -> list[str]:
|
|
169
|
+
by_round = round_backends(unit)
|
|
170
|
+
if not by_round:
|
|
171
|
+
return ["## Backends per round", "", "no backend reported (predates backend logging, or no fallback pool was in play)"]
|
|
172
|
+
rows = [f"| {k} | {', '.join(v)} |" for k, v in by_round.items()]
|
|
173
|
+
return ["## Backends per round", "", "| round | backend(s) used |", "|---|---|", *rows]
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def summary_md(unit: Path) -> list[str]:
|
|
177
|
+
s = summarize(unit)
|
|
178
|
+
if not s["n"]:
|
|
179
|
+
return ["## Agent cost", "", "no cost/token usage reported (backend does not report it, or the run predates cost logging)"]
|
|
180
|
+
rows = [f"| {step} | {v['n']} | ${v['usd']:.2f} | {v['tokens_in']} | {v['tokens_out']} |" for step, v in s["by_step"].items()]
|
|
181
|
+
return ["## Agent cost", "",
|
|
182
|
+
f"Total: **${s['total']:.2f}**, {s['tokens_in']} input / {s['tokens_out']} output tokens over "
|
|
183
|
+
f"{s['n']} agent run(s) — from `cost` events in progress.log", "",
|
|
184
|
+
"| step | agent runs | USD | tokens in | tokens out |", "|---|---|---|---|---|", *rows]
|