eval-builder 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eval_builder/__init__.py +3 -0
- eval_builder/cli.py +354 -0
- eval_builder/data/SKILL.md +78 -0
- eval_builder/draft.py +271 -0
- eval_builder/export.py +284 -0
- eval_builder/ingest/__init__.py +181 -0
- eval_builder/ingest/formats.py +645 -0
- eval_builder/io.py +82 -0
- eval_builder/judge/__init__.py +1 -0
- eval_builder/judge/check.py +420 -0
- eval_builder/judge/plan.py +151 -0
- eval_builder/judge/run.py +149 -0
- eval_builder/judge/stats.py +76 -0
- eval_builder/mcp_server.py +175 -0
- eval_builder/redact.py +101 -0
- eval_builder/report.py +263 -0
- eval_builder/schema.py +242 -0
- eval_builder/select.py +412 -0
- eval_builder/setup_agents.py +190 -0
- eval_builder/status.py +32 -0
- eval_builder/workspace.py +67 -0
- eval_builder-0.1.0.dist-info/METADATA +154 -0
- eval_builder-0.1.0.dist-info/RECORD +26 -0
- eval_builder-0.1.0.dist-info/WHEEL +4 -0
- eval_builder-0.1.0.dist-info/entry_points.txt +2 -0
- eval_builder-0.1.0.dist-info/licenses/LICENSE +21 -0
eval_builder/select.py
ADDED
|
@@ -0,0 +1,412 @@
|
|
|
1
|
+
"""Pick a small, diverse, representative set of traces without calling any model.
|
|
2
|
+
|
|
3
|
+
Pipeline: exact dedupe (normalized text hash) -> near-duplicate merge (character
|
|
4
|
+
n-gram cosine) -> k-means topic clusters -> failure oversampling ->
|
|
5
|
+
stratum coverage -> one central example per cluster -> proportional fill with
|
|
6
|
+
farthest-point sampling. Every pick records the reasons it was picked.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hashlib
|
|
12
|
+
import math
|
|
13
|
+
import re
|
|
14
|
+
from collections import Counter, defaultdict
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
import numpy as np
|
|
19
|
+
from scipy import sparse
|
|
20
|
+
from sklearn.cluster import KMeans
|
|
21
|
+
from sklearn.feature_extraction.text import TfidfVectorizer
|
|
22
|
+
|
|
23
|
+
from .ingest import load_traces
|
|
24
|
+
from .io import write_json
|
|
25
|
+
from .schema import Trace
|
|
26
|
+
from .workspace import Workspace
|
|
27
|
+
|
|
28
|
+
DEFAULT_STRATA = ("route", "tools", "error", "feedback")
|
|
29
|
+
|
|
30
|
+
_WS = re.compile(r"\s+")
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def normalize_text(s: str) -> str:
|
|
34
|
+
return _WS.sub(" ", s.lower()).strip()
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def text_hash(s: str) -> str:
|
|
38
|
+
return hashlib.sha256(normalize_text(s).encode("utf-8")).hexdigest()
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def is_failure(t: Trace) -> bool:
|
|
42
|
+
return bool(t.error) or t.feedback == "negative"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def failure_kind(t: Trace) -> str:
|
|
46
|
+
kinds = []
|
|
47
|
+
if t.error:
|
|
48
|
+
kinds.append("error flag")
|
|
49
|
+
if t.feedback == "negative":
|
|
50
|
+
kinds.append("negative user feedback")
|
|
51
|
+
return " + ".join(kinds)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def user_text(t: Trace) -> str:
|
|
55
|
+
"""Everything the user said in the conversation, in order.
|
|
56
|
+
|
|
57
|
+
Using only the last user message would merge unrelated conversations that end
|
|
58
|
+
with the same generic follow-up ("make it shorter").
|
|
59
|
+
"""
|
|
60
|
+
turns = [m["content"] for m in t.messages if m.get("role") == "user"]
|
|
61
|
+
return "\n".join(turns) if turns else t.input
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _key_text(t: Trace, dedupe_on: str) -> str:
|
|
65
|
+
return user_text(t) if dedupe_on == "input" else f"{user_text(t)}\n{t.output}"
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class _UnionFind:
|
|
69
|
+
def __init__(self, n: int) -> None:
|
|
70
|
+
self.p = list(range(n))
|
|
71
|
+
|
|
72
|
+
def find(self, x: int) -> int:
|
|
73
|
+
while self.p[x] != x:
|
|
74
|
+
self.p[x] = self.p[self.p[x]]
|
|
75
|
+
x = self.p[x]
|
|
76
|
+
return x
|
|
77
|
+
|
|
78
|
+
def union(self, a: int, b: int) -> None:
|
|
79
|
+
ra, rb = self.find(a), self.find(b)
|
|
80
|
+
if ra != rb:
|
|
81
|
+
self.p[max(ra, rb)] = min(ra, rb)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def near_duplicate_groups(texts: list[str], threshold: float, chunk: int = 1000) -> list[list[int]]:
|
|
85
|
+
"""Group texts whose character 3-5 gram cosine similarity is >= threshold."""
|
|
86
|
+
n = len(texts)
|
|
87
|
+
if n < 2:
|
|
88
|
+
return [[i] for i in range(n)]
|
|
89
|
+
# Plain term-frequency vectors (no IDF) so the similarity of two texts does not
|
|
90
|
+
# depend on what else is in the corpus.
|
|
91
|
+
vec = TfidfVectorizer(analyzer="char_wb", ngram_range=(3, 5), lowercase=True, use_idf=False)
|
|
92
|
+
try:
|
|
93
|
+
x = vec.fit_transform(texts)
|
|
94
|
+
except ValueError: # all empty
|
|
95
|
+
return [[i] for i in range(n)]
|
|
96
|
+
uf = _UnionFind(n)
|
|
97
|
+
for start in range(0, n, chunk):
|
|
98
|
+
coo = sparse.coo_matrix(x[start : start + chunk] @ x.T)
|
|
99
|
+
for r, c, v in zip(coo.row, coo.col, coo.data, strict=True):
|
|
100
|
+
i = start + int(r)
|
|
101
|
+
if int(c) > i and v >= threshold:
|
|
102
|
+
uf.union(i, int(c))
|
|
103
|
+
groups: dict[int, list[int]] = defaultdict(list)
|
|
104
|
+
for i in range(n):
|
|
105
|
+
groups[uf.find(i)].append(i)
|
|
106
|
+
return sorted(groups.values(), key=lambda g: g[0])
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _pick_rep(members: list[int], traces: list[Trace]) -> int:
|
|
110
|
+
for i in members:
|
|
111
|
+
if is_failure(traces[i]):
|
|
112
|
+
return i
|
|
113
|
+
return members[0]
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _strata_values(t: Trace, field: str) -> list[str]:
|
|
117
|
+
if field == "tools":
|
|
118
|
+
return [f"{tool}" for tool in t.tools] or ["(none)"]
|
|
119
|
+
if field == "error":
|
|
120
|
+
return [str(bool(t.error)).lower()]
|
|
121
|
+
if field == "feedback":
|
|
122
|
+
return [t.feedback or "(none)"]
|
|
123
|
+
if field == "route":
|
|
124
|
+
return [t.route] if t.route else []
|
|
125
|
+
if field == "model":
|
|
126
|
+
return [t.model] if t.model else []
|
|
127
|
+
cur: Any = t.metadata
|
|
128
|
+
for part in field.split("."):
|
|
129
|
+
if isinstance(cur, dict) and part in cur:
|
|
130
|
+
cur = cur[part]
|
|
131
|
+
else:
|
|
132
|
+
return []
|
|
133
|
+
if isinstance(cur, list):
|
|
134
|
+
return [str(x) for x in cur]
|
|
135
|
+
return [str(cur)] if cur is not None else []
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def select(
|
|
139
|
+
workspace: str | Path,
|
|
140
|
+
n: int = 50,
|
|
141
|
+
seed: int = 0,
|
|
142
|
+
clusters: int | None = None,
|
|
143
|
+
failure_share: float = 0.3,
|
|
144
|
+
near_dup_threshold: float = 0.9,
|
|
145
|
+
dedupe_on: str = "input",
|
|
146
|
+
stratify: list[str] | None = None,
|
|
147
|
+
) -> dict[str, Any]:
|
|
148
|
+
if n < 1:
|
|
149
|
+
raise ValueError("n must be >= 1")
|
|
150
|
+
if dedupe_on not in ("input", "input+output"):
|
|
151
|
+
raise ValueError("dedupe_on must be 'input' or 'input+output'")
|
|
152
|
+
ws = Workspace.at(workspace)
|
|
153
|
+
traces = load_traces(ws.root)
|
|
154
|
+
if not traces:
|
|
155
|
+
raise ValueError("no traces in workspace; run ingest first")
|
|
156
|
+
result = select_traces(
|
|
157
|
+
traces,
|
|
158
|
+
n=n,
|
|
159
|
+
seed=seed,
|
|
160
|
+
clusters=clusters,
|
|
161
|
+
failure_share=failure_share,
|
|
162
|
+
near_dup_threshold=near_dup_threshold,
|
|
163
|
+
dedupe_on=dedupe_on,
|
|
164
|
+
stratify=stratify,
|
|
165
|
+
)
|
|
166
|
+
write_json(ws.selection, result)
|
|
167
|
+
return result
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def select_traces(
|
|
171
|
+
traces: list[Trace],
|
|
172
|
+
n: int = 50,
|
|
173
|
+
seed: int = 0,
|
|
174
|
+
clusters: int | None = None,
|
|
175
|
+
failure_share: float = 0.3,
|
|
176
|
+
near_dup_threshold: float = 0.9,
|
|
177
|
+
dedupe_on: str = "input",
|
|
178
|
+
stratify: list[str] | None = None,
|
|
179
|
+
) -> dict[str, Any]:
|
|
180
|
+
# 1. exact dedupe
|
|
181
|
+
exact: dict[str, list[int]] = defaultdict(list)
|
|
182
|
+
for i, t in enumerate(traces):
|
|
183
|
+
exact[text_hash(_key_text(t, dedupe_on))].append(i)
|
|
184
|
+
exact_groups = sorted(exact.values(), key=lambda g: g[0])
|
|
185
|
+
reps = [_pick_rep(g, traces) for g in exact_groups]
|
|
186
|
+
members_of: dict[int, list[int]] = {r: g for r, g in zip(reps, exact_groups, strict=True)}
|
|
187
|
+
|
|
188
|
+
# 2. near-duplicate merge among exact representatives
|
|
189
|
+
near = near_duplicate_groups(
|
|
190
|
+
[_key_text(traces[r], dedupe_on) for r in reps], near_dup_threshold
|
|
191
|
+
)
|
|
192
|
+
unique: list[int] = []
|
|
193
|
+
group_members: dict[int, list[int]] = {}
|
|
194
|
+
near_merged = 0
|
|
195
|
+
for g in near:
|
|
196
|
+
all_members = [m for gi in g for m in members_of[reps[gi]]]
|
|
197
|
+
rep = _pick_rep(sorted(all_members), traces)
|
|
198
|
+
unique.append(rep)
|
|
199
|
+
group_members[rep] = sorted(all_members)
|
|
200
|
+
if len(g) > 1:
|
|
201
|
+
near_merged += len(g) - 1
|
|
202
|
+
unique.sort()
|
|
203
|
+
u_traces = [traces[i] for i in unique]
|
|
204
|
+
nu = len(unique)
|
|
205
|
+
|
|
206
|
+
# 3. topic clusters over the user side of each conversation
|
|
207
|
+
vec = TfidfVectorizer(
|
|
208
|
+
ngram_range=(1, 2), sublinear_tf=True, stop_words="english", max_features=20000, min_df=1
|
|
209
|
+
)
|
|
210
|
+
try:
|
|
211
|
+
x = vec.fit_transform([user_text(t) for t in u_traces])
|
|
212
|
+
terms = np.array(vec.get_feature_names_out())
|
|
213
|
+
except ValueError:
|
|
214
|
+
x = sparse.csr_matrix(np.ones((nu, 1)))
|
|
215
|
+
terms = np.array(["(empty)"])
|
|
216
|
+
k = clusters or max(2, round(math.sqrt(nu)))
|
|
217
|
+
k = max(1, min(k, nu, n))
|
|
218
|
+
if k >= 2:
|
|
219
|
+
km = KMeans(n_clusters=k, n_init=10, random_state=seed)
|
|
220
|
+
labels = km.fit_predict(x)
|
|
221
|
+
centers = km.cluster_centers_
|
|
222
|
+
else:
|
|
223
|
+
labels = np.zeros(nu, dtype=int)
|
|
224
|
+
centers = np.asarray(x.mean(axis=0))
|
|
225
|
+
dense_dist = np.asarray(
|
|
226
|
+
[np.linalg.norm(x[i].toarray().ravel() - centers[labels[i]]) for i in range(nu)]
|
|
227
|
+
)
|
|
228
|
+
cluster_info: list[dict[str, Any]] = []
|
|
229
|
+
for c in range(k):
|
|
230
|
+
idx = np.where(labels == c)[0]
|
|
231
|
+
top = [str(terms[j]) for j in np.argsort(-centers[c])[:5] if centers[c][j] > 0]
|
|
232
|
+
cluster_info.append(
|
|
233
|
+
{"id": c, "size": int(len(idx)), "share": round(len(idx) / nu, 4), "terms": top}
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
# helpers
|
|
237
|
+
selected: dict[int, list[str]] = {} # position in u_traces -> reasons
|
|
238
|
+
|
|
239
|
+
def add(pos: int, reason: str) -> None:
|
|
240
|
+
selected.setdefault(pos, []).append(reason)
|
|
241
|
+
|
|
242
|
+
def central_order(positions: list[int]) -> list[int]:
|
|
243
|
+
return sorted(positions, key=lambda p: (dense_dist[p], p))
|
|
244
|
+
|
|
245
|
+
def cluster_note(c: int) -> str:
|
|
246
|
+
info = cluster_info[c]
|
|
247
|
+
terms = ", ".join(info["terms"][:3])
|
|
248
|
+
return f"cluster {c} ({info['size']} traces, {info['share']:.0%}; {terms})"
|
|
249
|
+
|
|
250
|
+
# 4. failure oversampling
|
|
251
|
+
fail_pos = [p for p, t in enumerate(u_traces) if is_failure(t)]
|
|
252
|
+
pop_fail_share = len(fail_pos) / nu if nu else 0.0
|
|
253
|
+
fail_budget = min(
|
|
254
|
+
len(fail_pos), n, max(math.ceil(failure_share * n), round(pop_fail_share * n))
|
|
255
|
+
)
|
|
256
|
+
if fail_budget:
|
|
257
|
+
by_cluster: dict[int, list[int]] = defaultdict(list)
|
|
258
|
+
for p in central_order(fail_pos):
|
|
259
|
+
by_cluster[int(labels[p])].append(p)
|
|
260
|
+
order = sorted(by_cluster, key=lambda c: (-cluster_info[c]["size"], c))
|
|
261
|
+
picked = 0
|
|
262
|
+
while picked < fail_budget:
|
|
263
|
+
progressed = False
|
|
264
|
+
for c in order:
|
|
265
|
+
if by_cluster[c] and picked < fail_budget:
|
|
266
|
+
p = by_cluster[c].pop(0)
|
|
267
|
+
add(
|
|
268
|
+
p,
|
|
269
|
+
f"failure ({failure_kind(u_traces[p])}); {pop_fail_share:.0%} of unique "
|
|
270
|
+
f"traces are failures and at least {failure_share:.0%} of picks are "
|
|
271
|
+
f"reserved for them",
|
|
272
|
+
)
|
|
273
|
+
picked += 1
|
|
274
|
+
progressed = True
|
|
275
|
+
if not progressed:
|
|
276
|
+
break
|
|
277
|
+
|
|
278
|
+
# 5. stratum coverage
|
|
279
|
+
fields = list(DEFAULT_STRATA) + [f for f in (stratify or []) if f not in DEFAULT_STRATA]
|
|
280
|
+
strata_pop: dict[str, Counter[str]] = {}
|
|
281
|
+
skipped_fields: dict[str, str] = {}
|
|
282
|
+
max_card = max(10, n // 2)
|
|
283
|
+
for f in fields:
|
|
284
|
+
counts: Counter[str] = Counter()
|
|
285
|
+
for t in u_traces:
|
|
286
|
+
counts.update(_strata_values(t, f))
|
|
287
|
+
if not counts or (len(counts) == 1 and f in DEFAULT_STRATA):
|
|
288
|
+
continue
|
|
289
|
+
if len(counts) > max_card:
|
|
290
|
+
skipped_fields[f] = f"{len(counts)} distinct values (limit {max_card})"
|
|
291
|
+
continue
|
|
292
|
+
strata_pop[f] = counts
|
|
293
|
+
for f, counts in strata_pop.items():
|
|
294
|
+
for value, cnt in sorted(counts.items(), key=lambda kv: (kv[1], kv[0])):
|
|
295
|
+
if len(selected) >= n:
|
|
296
|
+
break
|
|
297
|
+
members = [p for p, t in enumerate(u_traces) if value in _strata_values(t, f)]
|
|
298
|
+
if any(p in selected for p in members):
|
|
299
|
+
continue
|
|
300
|
+
p = central_order(members)[0]
|
|
301
|
+
add(p, f"covers {f}={value} ({cnt} unique traces, {cnt / nu:.0%})")
|
|
302
|
+
|
|
303
|
+
# 6. one central example per uncovered cluster
|
|
304
|
+
covered = {int(labels[p]) for p in selected}
|
|
305
|
+
for c in sorted(range(k), key=lambda c: (-cluster_info[c]["size"], c)):
|
|
306
|
+
if len(selected) >= n:
|
|
307
|
+
break
|
|
308
|
+
if c in covered:
|
|
309
|
+
continue
|
|
310
|
+
members = [p for p in range(nu) if labels[p] == c]
|
|
311
|
+
add(central_order(members)[0], f"central example of {cluster_note(c)}")
|
|
312
|
+
|
|
313
|
+
# 7. proportional fill with farthest-point sampling inside each cluster
|
|
314
|
+
remaining = n - len(selected)
|
|
315
|
+
if remaining > 0:
|
|
316
|
+
quotas = {c: n * cluster_info[c]["size"] / nu for c in range(k)}
|
|
317
|
+
have = Counter(int(labels[p]) for p in selected)
|
|
318
|
+
need = {c: max(0.0, quotas[c] - have[c]) for c in range(k)}
|
|
319
|
+
floor = {c: int(need[c]) for c in range(k)}
|
|
320
|
+
left = remaining - sum(floor.values())
|
|
321
|
+
for c in sorted(range(k), key=lambda c: (-(need[c] - floor[c]), c)):
|
|
322
|
+
if left <= 0:
|
|
323
|
+
break
|
|
324
|
+
floor[c] += 1
|
|
325
|
+
left -= 1
|
|
326
|
+
for c in range(k):
|
|
327
|
+
for _ in range(floor[c]):
|
|
328
|
+
if len(selected) >= n:
|
|
329
|
+
break
|
|
330
|
+
far = _farthest(
|
|
331
|
+
x,
|
|
332
|
+
[q for q in range(nu) if labels[q] == c and q not in selected],
|
|
333
|
+
[q for q in selected if labels[q] == c],
|
|
334
|
+
)
|
|
335
|
+
if far is None:
|
|
336
|
+
break
|
|
337
|
+
add(
|
|
338
|
+
far,
|
|
339
|
+
f"adds variety within {cluster_note(c)}; least similar to cases already "
|
|
340
|
+
f"picked there",
|
|
341
|
+
)
|
|
342
|
+
while len(selected) < min(n, nu):
|
|
343
|
+
far = _farthest(x, [q for q in range(nu) if q not in selected], list(selected))
|
|
344
|
+
if far is None:
|
|
345
|
+
break
|
|
346
|
+
add(far, "adds variety: least similar to every case already picked")
|
|
347
|
+
|
|
348
|
+
# 8. assemble
|
|
349
|
+
picks = []
|
|
350
|
+
for p in sorted(selected, key=lambda p: (int(labels[p]), p)):
|
|
351
|
+
t = u_traces[p]
|
|
352
|
+
grp = group_members[unique[p]]
|
|
353
|
+
picks.append(
|
|
354
|
+
{
|
|
355
|
+
"trace_id": t.id,
|
|
356
|
+
"reasons": selected[p],
|
|
357
|
+
"cluster": int(labels[p]),
|
|
358
|
+
"represents": len(grp),
|
|
359
|
+
"duplicates": [traces[i].id for i in grp if i != unique[p]],
|
|
360
|
+
"failure": is_failure(t),
|
|
361
|
+
"failing_in_group": sum(1 for i in grp if is_failure(traces[i])),
|
|
362
|
+
"strata": {f: _strata_values(t, f) for f in strata_pop},
|
|
363
|
+
}
|
|
364
|
+
)
|
|
365
|
+
sel_counts = Counter(int(labels[p]) for p in selected)
|
|
366
|
+
for info in cluster_info:
|
|
367
|
+
info["selected"] = sel_counts.get(info["id"], 0)
|
|
368
|
+
strata_report = {
|
|
369
|
+
f: {
|
|
370
|
+
v: {
|
|
371
|
+
"population": cnt,
|
|
372
|
+
"selected": sum(1 for p in selected if v in _strata_values(u_traces[p], f)),
|
|
373
|
+
}
|
|
374
|
+
for v, cnt in sorted(counts.items())
|
|
375
|
+
}
|
|
376
|
+
for f, counts in strata_pop.items()
|
|
377
|
+
}
|
|
378
|
+
return {
|
|
379
|
+
"params": {
|
|
380
|
+
"n": n,
|
|
381
|
+
"seed": seed,
|
|
382
|
+
"clusters": k,
|
|
383
|
+
"failure_share": failure_share,
|
|
384
|
+
"near_dup_threshold": near_dup_threshold,
|
|
385
|
+
"dedupe_on": dedupe_on,
|
|
386
|
+
"stratify": fields,
|
|
387
|
+
},
|
|
388
|
+
"population": {
|
|
389
|
+
"traces": len(traces),
|
|
390
|
+
"exact_duplicates_removed": len(traces) - len(exact_groups),
|
|
391
|
+
"near_duplicates_merged": near_merged,
|
|
392
|
+
"unique": nu,
|
|
393
|
+
"failures_unique": len(fail_pos),
|
|
394
|
+
},
|
|
395
|
+
"clusters": cluster_info,
|
|
396
|
+
"strata": strata_report,
|
|
397
|
+
"strata_skipped": skipped_fields,
|
|
398
|
+
"selected_count": len(picks),
|
|
399
|
+
"selected_failures": sum(1 for p in picks if p["failure"]),
|
|
400
|
+
"selected": picks,
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
def _farthest(x: sparse.csr_matrix, candidates: list[int], chosen: list[int]) -> int | None:
|
|
405
|
+
if not candidates:
|
|
406
|
+
return None
|
|
407
|
+
if not chosen:
|
|
408
|
+
return candidates[0]
|
|
409
|
+
sims = (x[candidates] @ x[chosen].T).toarray()
|
|
410
|
+
max_sim = sims.max(axis=1)
|
|
411
|
+
best = int(np.argmin(max_sim)) # first index wins ties: deterministic
|
|
412
|
+
return candidates[best]
|
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
"""`eval-builder setup`: register the MCP server with Claude Code, Codex and Cursor.
|
|
2
|
+
|
|
3
|
+
Shows every change first. Applies only with --yes. Backs up any file it edits.
|
|
4
|
+
Running it twice changes nothing the second time.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import shutil
|
|
11
|
+
import subprocess
|
|
12
|
+
import tomllib
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from datetime import datetime
|
|
15
|
+
from importlib import resources
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
NAME = "eval-builder"
|
|
20
|
+
DEFAULT_SERVER = ["uvx", NAME, "mcp"]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class Action:
|
|
25
|
+
agent: str
|
|
26
|
+
kind: str # "command" | "edit" | "copy" | "skip"
|
|
27
|
+
target: str
|
|
28
|
+
detail: str
|
|
29
|
+
apply_fn: Any = field(default=None, repr=False)
|
|
30
|
+
|
|
31
|
+
def as_dict(self) -> dict[str, str]:
|
|
32
|
+
return {
|
|
33
|
+
"agent": self.agent,
|
|
34
|
+
"kind": self.kind,
|
|
35
|
+
"target": self.target,
|
|
36
|
+
"detail": self.detail,
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def skill_text() -> str:
|
|
41
|
+
try:
|
|
42
|
+
return resources.files("eval_builder").joinpath("data/SKILL.md").read_text("utf-8")
|
|
43
|
+
except (FileNotFoundError, ModuleNotFoundError):
|
|
44
|
+
repo = Path(__file__).resolve().parents[2] / "skills" / NAME / "SKILL.md"
|
|
45
|
+
return repo.read_text("utf-8")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _backup(path: Path) -> Path | None:
|
|
49
|
+
if not path.exists():
|
|
50
|
+
return None
|
|
51
|
+
stamp = datetime.now().strftime("%Y%m%d%H%M%S")
|
|
52
|
+
dest = path.with_name(f"{path.name}.bak-{NAME}-{stamp}")
|
|
53
|
+
shutil.copy2(path, dest)
|
|
54
|
+
return dest
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _claude_action(server: list[str], path_env: str | None) -> Action:
|
|
58
|
+
exe = shutil.which("claude", path=path_env)
|
|
59
|
+
if not exe:
|
|
60
|
+
return Action(
|
|
61
|
+
"Claude Code",
|
|
62
|
+
"skip",
|
|
63
|
+
"claude CLI",
|
|
64
|
+
"claude CLI not found on PATH; to add later run: claude mcp add --scope "
|
|
65
|
+
f"user {NAME} -- {' '.join(server)}",
|
|
66
|
+
)
|
|
67
|
+
probe = subprocess.run([exe, "mcp", "get", NAME], capture_output=True, text=True, timeout=60)
|
|
68
|
+
if probe.returncode == 0:
|
|
69
|
+
return Action("Claude Code", "skip", "claude mcp", f"{NAME} already registered")
|
|
70
|
+
argv = [exe, "mcp", "add", "--scope", "user", NAME, "--", *server]
|
|
71
|
+
|
|
72
|
+
def run() -> str:
|
|
73
|
+
res = subprocess.run(argv, capture_output=True, text=True, timeout=60)
|
|
74
|
+
if res.returncode != 0:
|
|
75
|
+
raise RuntimeError(res.stderr.strip() or res.stdout.strip())
|
|
76
|
+
return res.stdout.strip()
|
|
77
|
+
|
|
78
|
+
return Action("Claude Code", "command", "claude mcp", "run: claude " + " ".join(argv[1:]), run)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _codex_action(home: Path, server: list[str], path_env: str | None) -> Action:
|
|
82
|
+
codex_dir = home / ".codex"
|
|
83
|
+
if not codex_dir.exists() and not shutil.which("codex", path=path_env):
|
|
84
|
+
return Action("Codex", "skip", str(codex_dir), "Codex not detected")
|
|
85
|
+
cfg = codex_dir / "config.toml"
|
|
86
|
+
existing = cfg.read_text("utf-8") if cfg.exists() else ""
|
|
87
|
+
try:
|
|
88
|
+
parsed = tomllib.loads(existing) if existing else {}
|
|
89
|
+
except tomllib.TOMLDecodeError as e:
|
|
90
|
+
return Action(
|
|
91
|
+
"Codex", "skip", str(cfg), f"config.toml does not parse ({e}); not touching it"
|
|
92
|
+
)
|
|
93
|
+
if NAME in (parsed.get("mcp_servers") or {}):
|
|
94
|
+
return Action("Codex", "skip", str(cfg), f"[mcp_servers.{NAME}] already present")
|
|
95
|
+
block = f'\n[mcp_servers.{NAME}]\ncommand = "{server[0]}"\nargs = {json.dumps(server[1:])}\n'
|
|
96
|
+
|
|
97
|
+
def run() -> str:
|
|
98
|
+
codex_dir.mkdir(parents=True, exist_ok=True)
|
|
99
|
+
b = _backup(cfg)
|
|
100
|
+
with open(cfg, "a", encoding="utf-8") as f:
|
|
101
|
+
if existing and not existing.endswith("\n"):
|
|
102
|
+
f.write("\n")
|
|
103
|
+
f.write(block)
|
|
104
|
+
return f"appended block; backup: {b}" if b else "created config.toml"
|
|
105
|
+
|
|
106
|
+
return Action("Codex", "edit", str(cfg), f"append:{block}", run)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _cursor_action(home: Path, server: list[str], path_env: str | None) -> Action:
|
|
110
|
+
cursor_dir = home / ".cursor"
|
|
111
|
+
if not cursor_dir.exists() and not shutil.which("cursor", path=path_env):
|
|
112
|
+
return Action("Cursor", "skip", str(cursor_dir), "Cursor not detected")
|
|
113
|
+
cfg = cursor_dir / "mcp.json"
|
|
114
|
+
data: dict[str, Any] = {}
|
|
115
|
+
if cfg.exists():
|
|
116
|
+
try:
|
|
117
|
+
data = json.loads(cfg.read_text("utf-8") or "{}")
|
|
118
|
+
except json.JSONDecodeError as e:
|
|
119
|
+
return Action(
|
|
120
|
+
"Cursor", "skip", str(cfg), f"mcp.json does not parse ({e}); not touching it"
|
|
121
|
+
)
|
|
122
|
+
entry = {"command": server[0], "args": server[1:]}
|
|
123
|
+
if (data.get("mcpServers") or {}).get(NAME) == entry:
|
|
124
|
+
return Action("Cursor", "skip", str(cfg), f"{NAME} already present")
|
|
125
|
+
|
|
126
|
+
def run() -> str:
|
|
127
|
+
cursor_dir.mkdir(parents=True, exist_ok=True)
|
|
128
|
+
b = _backup(cfg)
|
|
129
|
+
data.setdefault("mcpServers", {})[NAME] = entry
|
|
130
|
+
cfg.write_text(json.dumps(data, indent=2) + "\n", encoding="utf-8")
|
|
131
|
+
return f"wrote mcpServers.{NAME}; backup: {b}" if b else f"created {cfg}"
|
|
132
|
+
|
|
133
|
+
return Action("Cursor", "edit", str(cfg), f"set mcpServers.{NAME} = {json.dumps(entry)}", run)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _skill_actions(home: Path) -> list[Action]:
|
|
137
|
+
text = skill_text()
|
|
138
|
+
out = []
|
|
139
|
+
for agent, base in (("Claude Code", home / ".claude"), ("Codex", home / ".codex")):
|
|
140
|
+
dest = base / "skills" / NAME / "SKILL.md"
|
|
141
|
+
if not base.exists():
|
|
142
|
+
out.append(Action(agent, "skip", str(dest), f"{base} not found"))
|
|
143
|
+
continue
|
|
144
|
+
if dest.exists() and dest.read_text("utf-8") == text:
|
|
145
|
+
out.append(Action(agent, "skip", str(dest), "skill already up to date"))
|
|
146
|
+
continue
|
|
147
|
+
|
|
148
|
+
def run(dest: Path = dest) -> str:
|
|
149
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
150
|
+
b = _backup(dest)
|
|
151
|
+
dest.write_text(text, encoding="utf-8")
|
|
152
|
+
return f"wrote skill; backup: {b}" if b else "wrote skill"
|
|
153
|
+
|
|
154
|
+
out.append(Action(agent, "copy", str(dest), "install agent instructions (SKILL.md)", run))
|
|
155
|
+
return out
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def plan(
|
|
159
|
+
home: Path | None = None, server: list[str] | None = None, path_env: str | None = None
|
|
160
|
+
) -> list[Action]:
|
|
161
|
+
home = home or Path.home()
|
|
162
|
+
server = server or DEFAULT_SERVER
|
|
163
|
+
return [
|
|
164
|
+
_claude_action(server, path_env),
|
|
165
|
+
_codex_action(home, server, path_env),
|
|
166
|
+
_cursor_action(home, server, path_env),
|
|
167
|
+
*_skill_actions(home),
|
|
168
|
+
]
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def setup(
|
|
172
|
+
yes: bool = False,
|
|
173
|
+
home: Path | None = None,
|
|
174
|
+
server: list[str] | None = None,
|
|
175
|
+
path_env: str | None = None,
|
|
176
|
+
) -> dict[str, Any]:
|
|
177
|
+
actions = plan(home, server, path_env)
|
|
178
|
+
result: dict[str, Any] = {"applied": False, "actions": [a.as_dict() for a in actions]}
|
|
179
|
+
if not yes:
|
|
180
|
+
result["next"] = "re-run with --yes to apply the changes above"
|
|
181
|
+
return result
|
|
182
|
+
for a, d in zip(actions, result["actions"], strict=True):
|
|
183
|
+
if a.apply_fn is None:
|
|
184
|
+
continue
|
|
185
|
+
try:
|
|
186
|
+
d["result"] = a.apply_fn()
|
|
187
|
+
except Exception as e: # report and continue with the other agents
|
|
188
|
+
d["result"] = f"failed: {e}"
|
|
189
|
+
result["applied"] = True
|
|
190
|
+
return result
|
eval_builder/status.py
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""Summarize which steps have run in a workspace and suggest the next one."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from .workspace import Workspace
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def status(workspace: str | Path) -> dict[str, Any]:
|
|
12
|
+
ws = Workspace.at(workspace)
|
|
13
|
+
steps = {
|
|
14
|
+
"ingest": ws.traces.exists(),
|
|
15
|
+
"select": ws.selection.exists(),
|
|
16
|
+
"draft": ws.cases.exists(),
|
|
17
|
+
"judge_plan": ws.judge_requests.exists(),
|
|
18
|
+
"judgments": ws.judgments.exists(),
|
|
19
|
+
"labels": ws.labels.exists(),
|
|
20
|
+
"judge_check": ws.judge_check.exists(),
|
|
21
|
+
"export": (ws.exports / "manifest.json").exists(),
|
|
22
|
+
"report": ws.report_md.exists(),
|
|
23
|
+
}
|
|
24
|
+
order = [
|
|
25
|
+
("ingest", "eval-builder ingest <logs>"),
|
|
26
|
+
("select", "eval-builder select"),
|
|
27
|
+
("draft", "eval-builder draft"),
|
|
28
|
+
("export", "fill cases, then eval-builder export"),
|
|
29
|
+
("report", "eval-builder report"),
|
|
30
|
+
]
|
|
31
|
+
nxt = next((cmd for step, cmd in order if not steps[step]), "done")
|
|
32
|
+
return {"workspace": str(ws.root), "steps": steps, "next": nxt}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Standard file layout inside a workspace directory. Every step reads and writes here."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass(frozen=True)
|
|
10
|
+
class Workspace:
|
|
11
|
+
root: Path
|
|
12
|
+
|
|
13
|
+
@classmethod
|
|
14
|
+
def at(cls, path: str | Path) -> Workspace:
|
|
15
|
+
return cls(Path(path))
|
|
16
|
+
|
|
17
|
+
def ensure(self) -> Workspace:
|
|
18
|
+
self.root.mkdir(parents=True, exist_ok=True)
|
|
19
|
+
return self
|
|
20
|
+
|
|
21
|
+
@property
|
|
22
|
+
def traces(self) -> Path:
|
|
23
|
+
return self.root / "traces.jsonl"
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def ingest_report(self) -> Path:
|
|
27
|
+
return self.root / "ingest.json"
|
|
28
|
+
|
|
29
|
+
@property
|
|
30
|
+
def selection(self) -> Path:
|
|
31
|
+
return self.root / "selection.json"
|
|
32
|
+
|
|
33
|
+
@property
|
|
34
|
+
def cases(self) -> Path:
|
|
35
|
+
return self.root / "cases.yaml"
|
|
36
|
+
|
|
37
|
+
@property
|
|
38
|
+
def rubric(self) -> Path:
|
|
39
|
+
return self.root / "rubric.yaml"
|
|
40
|
+
|
|
41
|
+
@property
|
|
42
|
+
def judge_requests(self) -> Path:
|
|
43
|
+
return self.root / "judge_requests.jsonl"
|
|
44
|
+
|
|
45
|
+
@property
|
|
46
|
+
def judgments(self) -> Path:
|
|
47
|
+
return self.root / "judgments.jsonl"
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def labels(self) -> Path:
|
|
51
|
+
return self.root / "labels.jsonl"
|
|
52
|
+
|
|
53
|
+
@property
|
|
54
|
+
def judge_check(self) -> Path:
|
|
55
|
+
return self.root / "judge_check.json"
|
|
56
|
+
|
|
57
|
+
@property
|
|
58
|
+
def exports(self) -> Path:
|
|
59
|
+
return self.root / "exports"
|
|
60
|
+
|
|
61
|
+
@property
|
|
62
|
+
def report_md(self) -> Path:
|
|
63
|
+
return self.root / "report.md"
|
|
64
|
+
|
|
65
|
+
@property
|
|
66
|
+
def report_json(self) -> Path:
|
|
67
|
+
return self.root / "report.json"
|