eval-builder 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
eval_builder/select.py ADDED
@@ -0,0 +1,412 @@
1
+ """Pick a small, diverse, representative set of traces without calling any model.
2
+
3
+ Pipeline: exact dedupe (normalized text hash) -> near-duplicate merge (character
4
+ n-gram cosine) -> k-means topic clusters -> failure oversampling ->
5
+ stratum coverage -> one central example per cluster -> proportional fill with
6
+ farthest-point sampling. Every pick records the reasons it was picked.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import hashlib
12
+ import math
13
+ import re
14
+ from collections import Counter, defaultdict
15
+ from pathlib import Path
16
+ from typing import Any
17
+
18
+ import numpy as np
19
+ from scipy import sparse
20
+ from sklearn.cluster import KMeans
21
+ from sklearn.feature_extraction.text import TfidfVectorizer
22
+
23
+ from .ingest import load_traces
24
+ from .io import write_json
25
+ from .schema import Trace
26
+ from .workspace import Workspace
27
+
28
+ DEFAULT_STRATA = ("route", "tools", "error", "feedback")
29
+
30
+ _WS = re.compile(r"\s+")
31
+
32
+
33
+ def normalize_text(s: str) -> str:
34
+ return _WS.sub(" ", s.lower()).strip()
35
+
36
+
37
+ def text_hash(s: str) -> str:
38
+ return hashlib.sha256(normalize_text(s).encode("utf-8")).hexdigest()
39
+
40
+
41
+ def is_failure(t: Trace) -> bool:
42
+ return bool(t.error) or t.feedback == "negative"
43
+
44
+
45
+ def failure_kind(t: Trace) -> str:
46
+ kinds = []
47
+ if t.error:
48
+ kinds.append("error flag")
49
+ if t.feedback == "negative":
50
+ kinds.append("negative user feedback")
51
+ return " + ".join(kinds)
52
+
53
+
54
+ def user_text(t: Trace) -> str:
55
+ """Everything the user said in the conversation, in order.
56
+
57
+ Using only the last user message would merge unrelated conversations that end
58
+ with the same generic follow-up ("make it shorter").
59
+ """
60
+ turns = [m["content"] for m in t.messages if m.get("role") == "user"]
61
+ return "\n".join(turns) if turns else t.input
62
+
63
+
64
+ def _key_text(t: Trace, dedupe_on: str) -> str:
65
+ return user_text(t) if dedupe_on == "input" else f"{user_text(t)}\n{t.output}"
66
+
67
+
68
+ class _UnionFind:
69
+ def __init__(self, n: int) -> None:
70
+ self.p = list(range(n))
71
+
72
+ def find(self, x: int) -> int:
73
+ while self.p[x] != x:
74
+ self.p[x] = self.p[self.p[x]]
75
+ x = self.p[x]
76
+ return x
77
+
78
+ def union(self, a: int, b: int) -> None:
79
+ ra, rb = self.find(a), self.find(b)
80
+ if ra != rb:
81
+ self.p[max(ra, rb)] = min(ra, rb)
82
+
83
+
84
+ def near_duplicate_groups(texts: list[str], threshold: float, chunk: int = 1000) -> list[list[int]]:
85
+ """Group texts whose character 3-5 gram cosine similarity is >= threshold."""
86
+ n = len(texts)
87
+ if n < 2:
88
+ return [[i] for i in range(n)]
89
+ # Plain term-frequency vectors (no IDF) so the similarity of two texts does not
90
+ # depend on what else is in the corpus.
91
+ vec = TfidfVectorizer(analyzer="char_wb", ngram_range=(3, 5), lowercase=True, use_idf=False)
92
+ try:
93
+ x = vec.fit_transform(texts)
94
+ except ValueError: # all empty
95
+ return [[i] for i in range(n)]
96
+ uf = _UnionFind(n)
97
+ for start in range(0, n, chunk):
98
+ coo = sparse.coo_matrix(x[start : start + chunk] @ x.T)
99
+ for r, c, v in zip(coo.row, coo.col, coo.data, strict=True):
100
+ i = start + int(r)
101
+ if int(c) > i and v >= threshold:
102
+ uf.union(i, int(c))
103
+ groups: dict[int, list[int]] = defaultdict(list)
104
+ for i in range(n):
105
+ groups[uf.find(i)].append(i)
106
+ return sorted(groups.values(), key=lambda g: g[0])
107
+
108
+
109
+ def _pick_rep(members: list[int], traces: list[Trace]) -> int:
110
+ for i in members:
111
+ if is_failure(traces[i]):
112
+ return i
113
+ return members[0]
114
+
115
+
116
+ def _strata_values(t: Trace, field: str) -> list[str]:
117
+ if field == "tools":
118
+ return [f"{tool}" for tool in t.tools] or ["(none)"]
119
+ if field == "error":
120
+ return [str(bool(t.error)).lower()]
121
+ if field == "feedback":
122
+ return [t.feedback or "(none)"]
123
+ if field == "route":
124
+ return [t.route] if t.route else []
125
+ if field == "model":
126
+ return [t.model] if t.model else []
127
+ cur: Any = t.metadata
128
+ for part in field.split("."):
129
+ if isinstance(cur, dict) and part in cur:
130
+ cur = cur[part]
131
+ else:
132
+ return []
133
+ if isinstance(cur, list):
134
+ return [str(x) for x in cur]
135
+ return [str(cur)] if cur is not None else []
136
+
137
+
138
+ def select(
139
+ workspace: str | Path,
140
+ n: int = 50,
141
+ seed: int = 0,
142
+ clusters: int | None = None,
143
+ failure_share: float = 0.3,
144
+ near_dup_threshold: float = 0.9,
145
+ dedupe_on: str = "input",
146
+ stratify: list[str] | None = None,
147
+ ) -> dict[str, Any]:
148
+ if n < 1:
149
+ raise ValueError("n must be >= 1")
150
+ if dedupe_on not in ("input", "input+output"):
151
+ raise ValueError("dedupe_on must be 'input' or 'input+output'")
152
+ ws = Workspace.at(workspace)
153
+ traces = load_traces(ws.root)
154
+ if not traces:
155
+ raise ValueError("no traces in workspace; run ingest first")
156
+ result = select_traces(
157
+ traces,
158
+ n=n,
159
+ seed=seed,
160
+ clusters=clusters,
161
+ failure_share=failure_share,
162
+ near_dup_threshold=near_dup_threshold,
163
+ dedupe_on=dedupe_on,
164
+ stratify=stratify,
165
+ )
166
+ write_json(ws.selection, result)
167
+ return result
168
+
169
+
170
+ def select_traces(
171
+ traces: list[Trace],
172
+ n: int = 50,
173
+ seed: int = 0,
174
+ clusters: int | None = None,
175
+ failure_share: float = 0.3,
176
+ near_dup_threshold: float = 0.9,
177
+ dedupe_on: str = "input",
178
+ stratify: list[str] | None = None,
179
+ ) -> dict[str, Any]:
180
+ # 1. exact dedupe
181
+ exact: dict[str, list[int]] = defaultdict(list)
182
+ for i, t in enumerate(traces):
183
+ exact[text_hash(_key_text(t, dedupe_on))].append(i)
184
+ exact_groups = sorted(exact.values(), key=lambda g: g[0])
185
+ reps = [_pick_rep(g, traces) for g in exact_groups]
186
+ members_of: dict[int, list[int]] = {r: g for r, g in zip(reps, exact_groups, strict=True)}
187
+
188
+ # 2. near-duplicate merge among exact representatives
189
+ near = near_duplicate_groups(
190
+ [_key_text(traces[r], dedupe_on) for r in reps], near_dup_threshold
191
+ )
192
+ unique: list[int] = []
193
+ group_members: dict[int, list[int]] = {}
194
+ near_merged = 0
195
+ for g in near:
196
+ all_members = [m for gi in g for m in members_of[reps[gi]]]
197
+ rep = _pick_rep(sorted(all_members), traces)
198
+ unique.append(rep)
199
+ group_members[rep] = sorted(all_members)
200
+ if len(g) > 1:
201
+ near_merged += len(g) - 1
202
+ unique.sort()
203
+ u_traces = [traces[i] for i in unique]
204
+ nu = len(unique)
205
+
206
+ # 3. topic clusters over the user side of each conversation
207
+ vec = TfidfVectorizer(
208
+ ngram_range=(1, 2), sublinear_tf=True, stop_words="english", max_features=20000, min_df=1
209
+ )
210
+ try:
211
+ x = vec.fit_transform([user_text(t) for t in u_traces])
212
+ terms = np.array(vec.get_feature_names_out())
213
+ except ValueError:
214
+ x = sparse.csr_matrix(np.ones((nu, 1)))
215
+ terms = np.array(["(empty)"])
216
+ k = clusters or max(2, round(math.sqrt(nu)))
217
+ k = max(1, min(k, nu, n))
218
+ if k >= 2:
219
+ km = KMeans(n_clusters=k, n_init=10, random_state=seed)
220
+ labels = km.fit_predict(x)
221
+ centers = km.cluster_centers_
222
+ else:
223
+ labels = np.zeros(nu, dtype=int)
224
+ centers = np.asarray(x.mean(axis=0))
225
+ dense_dist = np.asarray(
226
+ [np.linalg.norm(x[i].toarray().ravel() - centers[labels[i]]) for i in range(nu)]
227
+ )
228
+ cluster_info: list[dict[str, Any]] = []
229
+ for c in range(k):
230
+ idx = np.where(labels == c)[0]
231
+ top = [str(terms[j]) for j in np.argsort(-centers[c])[:5] if centers[c][j] > 0]
232
+ cluster_info.append(
233
+ {"id": c, "size": int(len(idx)), "share": round(len(idx) / nu, 4), "terms": top}
234
+ )
235
+
236
+ # helpers
237
+ selected: dict[int, list[str]] = {} # position in u_traces -> reasons
238
+
239
+ def add(pos: int, reason: str) -> None:
240
+ selected.setdefault(pos, []).append(reason)
241
+
242
+ def central_order(positions: list[int]) -> list[int]:
243
+ return sorted(positions, key=lambda p: (dense_dist[p], p))
244
+
245
+ def cluster_note(c: int) -> str:
246
+ info = cluster_info[c]
247
+ terms = ", ".join(info["terms"][:3])
248
+ return f"cluster {c} ({info['size']} traces, {info['share']:.0%}; {terms})"
249
+
250
+ # 4. failure oversampling
251
+ fail_pos = [p for p, t in enumerate(u_traces) if is_failure(t)]
252
+ pop_fail_share = len(fail_pos) / nu if nu else 0.0
253
+ fail_budget = min(
254
+ len(fail_pos), n, max(math.ceil(failure_share * n), round(pop_fail_share * n))
255
+ )
256
+ if fail_budget:
257
+ by_cluster: dict[int, list[int]] = defaultdict(list)
258
+ for p in central_order(fail_pos):
259
+ by_cluster[int(labels[p])].append(p)
260
+ order = sorted(by_cluster, key=lambda c: (-cluster_info[c]["size"], c))
261
+ picked = 0
262
+ while picked < fail_budget:
263
+ progressed = False
264
+ for c in order:
265
+ if by_cluster[c] and picked < fail_budget:
266
+ p = by_cluster[c].pop(0)
267
+ add(
268
+ p,
269
+ f"failure ({failure_kind(u_traces[p])}); {pop_fail_share:.0%} of unique "
270
+ f"traces are failures and at least {failure_share:.0%} of picks are "
271
+ f"reserved for them",
272
+ )
273
+ picked += 1
274
+ progressed = True
275
+ if not progressed:
276
+ break
277
+
278
+ # 5. stratum coverage
279
+ fields = list(DEFAULT_STRATA) + [f for f in (stratify or []) if f not in DEFAULT_STRATA]
280
+ strata_pop: dict[str, Counter[str]] = {}
281
+ skipped_fields: dict[str, str] = {}
282
+ max_card = max(10, n // 2)
283
+ for f in fields:
284
+ counts: Counter[str] = Counter()
285
+ for t in u_traces:
286
+ counts.update(_strata_values(t, f))
287
+ if not counts or (len(counts) == 1 and f in DEFAULT_STRATA):
288
+ continue
289
+ if len(counts) > max_card:
290
+ skipped_fields[f] = f"{len(counts)} distinct values (limit {max_card})"
291
+ continue
292
+ strata_pop[f] = counts
293
+ for f, counts in strata_pop.items():
294
+ for value, cnt in sorted(counts.items(), key=lambda kv: (kv[1], kv[0])):
295
+ if len(selected) >= n:
296
+ break
297
+ members = [p for p, t in enumerate(u_traces) if value in _strata_values(t, f)]
298
+ if any(p in selected for p in members):
299
+ continue
300
+ p = central_order(members)[0]
301
+ add(p, f"covers {f}={value} ({cnt} unique traces, {cnt / nu:.0%})")
302
+
303
+ # 6. one central example per uncovered cluster
304
+ covered = {int(labels[p]) for p in selected}
305
+ for c in sorted(range(k), key=lambda c: (-cluster_info[c]["size"], c)):
306
+ if len(selected) >= n:
307
+ break
308
+ if c in covered:
309
+ continue
310
+ members = [p for p in range(nu) if labels[p] == c]
311
+ add(central_order(members)[0], f"central example of {cluster_note(c)}")
312
+
313
+ # 7. proportional fill with farthest-point sampling inside each cluster
314
+ remaining = n - len(selected)
315
+ if remaining > 0:
316
+ quotas = {c: n * cluster_info[c]["size"] / nu for c in range(k)}
317
+ have = Counter(int(labels[p]) for p in selected)
318
+ need = {c: max(0.0, quotas[c] - have[c]) for c in range(k)}
319
+ floor = {c: int(need[c]) for c in range(k)}
320
+ left = remaining - sum(floor.values())
321
+ for c in sorted(range(k), key=lambda c: (-(need[c] - floor[c]), c)):
322
+ if left <= 0:
323
+ break
324
+ floor[c] += 1
325
+ left -= 1
326
+ for c in range(k):
327
+ for _ in range(floor[c]):
328
+ if len(selected) >= n:
329
+ break
330
+ far = _farthest(
331
+ x,
332
+ [q for q in range(nu) if labels[q] == c and q not in selected],
333
+ [q for q in selected if labels[q] == c],
334
+ )
335
+ if far is None:
336
+ break
337
+ add(
338
+ far,
339
+ f"adds variety within {cluster_note(c)}; least similar to cases already "
340
+ f"picked there",
341
+ )
342
+ while len(selected) < min(n, nu):
343
+ far = _farthest(x, [q for q in range(nu) if q not in selected], list(selected))
344
+ if far is None:
345
+ break
346
+ add(far, "adds variety: least similar to every case already picked")
347
+
348
+ # 8. assemble
349
+ picks = []
350
+ for p in sorted(selected, key=lambda p: (int(labels[p]), p)):
351
+ t = u_traces[p]
352
+ grp = group_members[unique[p]]
353
+ picks.append(
354
+ {
355
+ "trace_id": t.id,
356
+ "reasons": selected[p],
357
+ "cluster": int(labels[p]),
358
+ "represents": len(grp),
359
+ "duplicates": [traces[i].id for i in grp if i != unique[p]],
360
+ "failure": is_failure(t),
361
+ "failing_in_group": sum(1 for i in grp if is_failure(traces[i])),
362
+ "strata": {f: _strata_values(t, f) for f in strata_pop},
363
+ }
364
+ )
365
+ sel_counts = Counter(int(labels[p]) for p in selected)
366
+ for info in cluster_info:
367
+ info["selected"] = sel_counts.get(info["id"], 0)
368
+ strata_report = {
369
+ f: {
370
+ v: {
371
+ "population": cnt,
372
+ "selected": sum(1 for p in selected if v in _strata_values(u_traces[p], f)),
373
+ }
374
+ for v, cnt in sorted(counts.items())
375
+ }
376
+ for f, counts in strata_pop.items()
377
+ }
378
+ return {
379
+ "params": {
380
+ "n": n,
381
+ "seed": seed,
382
+ "clusters": k,
383
+ "failure_share": failure_share,
384
+ "near_dup_threshold": near_dup_threshold,
385
+ "dedupe_on": dedupe_on,
386
+ "stratify": fields,
387
+ },
388
+ "population": {
389
+ "traces": len(traces),
390
+ "exact_duplicates_removed": len(traces) - len(exact_groups),
391
+ "near_duplicates_merged": near_merged,
392
+ "unique": nu,
393
+ "failures_unique": len(fail_pos),
394
+ },
395
+ "clusters": cluster_info,
396
+ "strata": strata_report,
397
+ "strata_skipped": skipped_fields,
398
+ "selected_count": len(picks),
399
+ "selected_failures": sum(1 for p in picks if p["failure"]),
400
+ "selected": picks,
401
+ }
402
+
403
+
404
+ def _farthest(x: sparse.csr_matrix, candidates: list[int], chosen: list[int]) -> int | None:
405
+ if not candidates:
406
+ return None
407
+ if not chosen:
408
+ return candidates[0]
409
+ sims = (x[candidates] @ x[chosen].T).toarray()
410
+ max_sim = sims.max(axis=1)
411
+ best = int(np.argmin(max_sim)) # first index wins ties: deterministic
412
+ return candidates[best]
@@ -0,0 +1,190 @@
1
+ """`eval-builder setup`: register the MCP server with Claude Code, Codex and Cursor.
2
+
3
+ Shows every change first. Applies only with --yes. Backs up any file it edits.
4
+ Running it twice changes nothing the second time.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import shutil
11
+ import subprocess
12
+ import tomllib
13
+ from dataclasses import dataclass, field
14
+ from datetime import datetime
15
+ from importlib import resources
16
+ from pathlib import Path
17
+ from typing import Any
18
+
19
+ NAME = "eval-builder"
20
+ DEFAULT_SERVER = ["uvx", NAME, "mcp"]
21
+
22
+
23
+ @dataclass
24
+ class Action:
25
+ agent: str
26
+ kind: str # "command" | "edit" | "copy" | "skip"
27
+ target: str
28
+ detail: str
29
+ apply_fn: Any = field(default=None, repr=False)
30
+
31
+ def as_dict(self) -> dict[str, str]:
32
+ return {
33
+ "agent": self.agent,
34
+ "kind": self.kind,
35
+ "target": self.target,
36
+ "detail": self.detail,
37
+ }
38
+
39
+
40
+ def skill_text() -> str:
41
+ try:
42
+ return resources.files("eval_builder").joinpath("data/SKILL.md").read_text("utf-8")
43
+ except (FileNotFoundError, ModuleNotFoundError):
44
+ repo = Path(__file__).resolve().parents[2] / "skills" / NAME / "SKILL.md"
45
+ return repo.read_text("utf-8")
46
+
47
+
48
+ def _backup(path: Path) -> Path | None:
49
+ if not path.exists():
50
+ return None
51
+ stamp = datetime.now().strftime("%Y%m%d%H%M%S")
52
+ dest = path.with_name(f"{path.name}.bak-{NAME}-{stamp}")
53
+ shutil.copy2(path, dest)
54
+ return dest
55
+
56
+
57
+ def _claude_action(server: list[str], path_env: str | None) -> Action:
58
+ exe = shutil.which("claude", path=path_env)
59
+ if not exe:
60
+ return Action(
61
+ "Claude Code",
62
+ "skip",
63
+ "claude CLI",
64
+ "claude CLI not found on PATH; to add later run: claude mcp add --scope "
65
+ f"user {NAME} -- {' '.join(server)}",
66
+ )
67
+ probe = subprocess.run([exe, "mcp", "get", NAME], capture_output=True, text=True, timeout=60)
68
+ if probe.returncode == 0:
69
+ return Action("Claude Code", "skip", "claude mcp", f"{NAME} already registered")
70
+ argv = [exe, "mcp", "add", "--scope", "user", NAME, "--", *server]
71
+
72
+ def run() -> str:
73
+ res = subprocess.run(argv, capture_output=True, text=True, timeout=60)
74
+ if res.returncode != 0:
75
+ raise RuntimeError(res.stderr.strip() or res.stdout.strip())
76
+ return res.stdout.strip()
77
+
78
+ return Action("Claude Code", "command", "claude mcp", "run: claude " + " ".join(argv[1:]), run)
79
+
80
+
81
+ def _codex_action(home: Path, server: list[str], path_env: str | None) -> Action:
82
+ codex_dir = home / ".codex"
83
+ if not codex_dir.exists() and not shutil.which("codex", path=path_env):
84
+ return Action("Codex", "skip", str(codex_dir), "Codex not detected")
85
+ cfg = codex_dir / "config.toml"
86
+ existing = cfg.read_text("utf-8") if cfg.exists() else ""
87
+ try:
88
+ parsed = tomllib.loads(existing) if existing else {}
89
+ except tomllib.TOMLDecodeError as e:
90
+ return Action(
91
+ "Codex", "skip", str(cfg), f"config.toml does not parse ({e}); not touching it"
92
+ )
93
+ if NAME in (parsed.get("mcp_servers") or {}):
94
+ return Action("Codex", "skip", str(cfg), f"[mcp_servers.{NAME}] already present")
95
+ block = f'\n[mcp_servers.{NAME}]\ncommand = "{server[0]}"\nargs = {json.dumps(server[1:])}\n'
96
+
97
+ def run() -> str:
98
+ codex_dir.mkdir(parents=True, exist_ok=True)
99
+ b = _backup(cfg)
100
+ with open(cfg, "a", encoding="utf-8") as f:
101
+ if existing and not existing.endswith("\n"):
102
+ f.write("\n")
103
+ f.write(block)
104
+ return f"appended block; backup: {b}" if b else "created config.toml"
105
+
106
+ return Action("Codex", "edit", str(cfg), f"append:{block}", run)
107
+
108
+
109
+ def _cursor_action(home: Path, server: list[str], path_env: str | None) -> Action:
110
+ cursor_dir = home / ".cursor"
111
+ if not cursor_dir.exists() and not shutil.which("cursor", path=path_env):
112
+ return Action("Cursor", "skip", str(cursor_dir), "Cursor not detected")
113
+ cfg = cursor_dir / "mcp.json"
114
+ data: dict[str, Any] = {}
115
+ if cfg.exists():
116
+ try:
117
+ data = json.loads(cfg.read_text("utf-8") or "{}")
118
+ except json.JSONDecodeError as e:
119
+ return Action(
120
+ "Cursor", "skip", str(cfg), f"mcp.json does not parse ({e}); not touching it"
121
+ )
122
+ entry = {"command": server[0], "args": server[1:]}
123
+ if (data.get("mcpServers") or {}).get(NAME) == entry:
124
+ return Action("Cursor", "skip", str(cfg), f"{NAME} already present")
125
+
126
+ def run() -> str:
127
+ cursor_dir.mkdir(parents=True, exist_ok=True)
128
+ b = _backup(cfg)
129
+ data.setdefault("mcpServers", {})[NAME] = entry
130
+ cfg.write_text(json.dumps(data, indent=2) + "\n", encoding="utf-8")
131
+ return f"wrote mcpServers.{NAME}; backup: {b}" if b else f"created {cfg}"
132
+
133
+ return Action("Cursor", "edit", str(cfg), f"set mcpServers.{NAME} = {json.dumps(entry)}", run)
134
+
135
+
136
+ def _skill_actions(home: Path) -> list[Action]:
137
+ text = skill_text()
138
+ out = []
139
+ for agent, base in (("Claude Code", home / ".claude"), ("Codex", home / ".codex")):
140
+ dest = base / "skills" / NAME / "SKILL.md"
141
+ if not base.exists():
142
+ out.append(Action(agent, "skip", str(dest), f"{base} not found"))
143
+ continue
144
+ if dest.exists() and dest.read_text("utf-8") == text:
145
+ out.append(Action(agent, "skip", str(dest), "skill already up to date"))
146
+ continue
147
+
148
+ def run(dest: Path = dest) -> str:
149
+ dest.parent.mkdir(parents=True, exist_ok=True)
150
+ b = _backup(dest)
151
+ dest.write_text(text, encoding="utf-8")
152
+ return f"wrote skill; backup: {b}" if b else "wrote skill"
153
+
154
+ out.append(Action(agent, "copy", str(dest), "install agent instructions (SKILL.md)", run))
155
+ return out
156
+
157
+
158
+ def plan(
159
+ home: Path | None = None, server: list[str] | None = None, path_env: str | None = None
160
+ ) -> list[Action]:
161
+ home = home or Path.home()
162
+ server = server or DEFAULT_SERVER
163
+ return [
164
+ _claude_action(server, path_env),
165
+ _codex_action(home, server, path_env),
166
+ _cursor_action(home, server, path_env),
167
+ *_skill_actions(home),
168
+ ]
169
+
170
+
171
+ def setup(
172
+ yes: bool = False,
173
+ home: Path | None = None,
174
+ server: list[str] | None = None,
175
+ path_env: str | None = None,
176
+ ) -> dict[str, Any]:
177
+ actions = plan(home, server, path_env)
178
+ result: dict[str, Any] = {"applied": False, "actions": [a.as_dict() for a in actions]}
179
+ if not yes:
180
+ result["next"] = "re-run with --yes to apply the changes above"
181
+ return result
182
+ for a, d in zip(actions, result["actions"], strict=True):
183
+ if a.apply_fn is None:
184
+ continue
185
+ try:
186
+ d["result"] = a.apply_fn()
187
+ except Exception as e: # report and continue with the other agents
188
+ d["result"] = f"failed: {e}"
189
+ result["applied"] = True
190
+ return result
eval_builder/status.py ADDED
@@ -0,0 +1,32 @@
1
+ """Summarize which steps have run in a workspace and suggest the next one."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+ from .workspace import Workspace
9
+
10
+
11
+ def status(workspace: str | Path) -> dict[str, Any]:
12
+ ws = Workspace.at(workspace)
13
+ steps = {
14
+ "ingest": ws.traces.exists(),
15
+ "select": ws.selection.exists(),
16
+ "draft": ws.cases.exists(),
17
+ "judge_plan": ws.judge_requests.exists(),
18
+ "judgments": ws.judgments.exists(),
19
+ "labels": ws.labels.exists(),
20
+ "judge_check": ws.judge_check.exists(),
21
+ "export": (ws.exports / "manifest.json").exists(),
22
+ "report": ws.report_md.exists(),
23
+ }
24
+ order = [
25
+ ("ingest", "eval-builder ingest <logs>"),
26
+ ("select", "eval-builder select"),
27
+ ("draft", "eval-builder draft"),
28
+ ("export", "fill cases, then eval-builder export"),
29
+ ("report", "eval-builder report"),
30
+ ]
31
+ nxt = next((cmd for step, cmd in order if not steps[step]), "done")
32
+ return {"workspace": str(ws.root), "steps": steps, "next": nxt}
@@ -0,0 +1,67 @@
1
+ """Standard file layout inside a workspace directory. Every step reads and writes here."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from pathlib import Path
7
+
8
+
9
+ @dataclass(frozen=True)
10
+ class Workspace:
11
+ root: Path
12
+
13
+ @classmethod
14
+ def at(cls, path: str | Path) -> Workspace:
15
+ return cls(Path(path))
16
+
17
+ def ensure(self) -> Workspace:
18
+ self.root.mkdir(parents=True, exist_ok=True)
19
+ return self
20
+
21
+ @property
22
+ def traces(self) -> Path:
23
+ return self.root / "traces.jsonl"
24
+
25
+ @property
26
+ def ingest_report(self) -> Path:
27
+ return self.root / "ingest.json"
28
+
29
+ @property
30
+ def selection(self) -> Path:
31
+ return self.root / "selection.json"
32
+
33
+ @property
34
+ def cases(self) -> Path:
35
+ return self.root / "cases.yaml"
36
+
37
+ @property
38
+ def rubric(self) -> Path:
39
+ return self.root / "rubric.yaml"
40
+
41
+ @property
42
+ def judge_requests(self) -> Path:
43
+ return self.root / "judge_requests.jsonl"
44
+
45
+ @property
46
+ def judgments(self) -> Path:
47
+ return self.root / "judgments.jsonl"
48
+
49
+ @property
50
+ def labels(self) -> Path:
51
+ return self.root / "labels.jsonl"
52
+
53
+ @property
54
+ def judge_check(self) -> Path:
55
+ return self.root / "judge_check.json"
56
+
57
+ @property
58
+ def exports(self) -> Path:
59
+ return self.root / "exports"
60
+
61
+ @property
62
+ def report_md(self) -> Path:
63
+ return self.root / "report.md"
64
+
65
+ @property
66
+ def report_json(self) -> Path:
67
+ return self.root / "report.json"