eval-builder 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
eval_builder/report.py ADDED
@@ -0,0 +1,263 @@
1
+ """Markdown and JSON report over whatever steps have run in a workspace."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from datetime import UTC, datetime
6
+ from pathlib import Path
7
+ from typing import Any
8
+
9
+ from . import __version__
10
+ from .draft import load_cases
11
+ from .io import read_json, write_json
12
+ from .workspace import Workspace
13
+
14
+ STATIC_LIMITS = [
15
+ "Selection is lexical (TF-IDF over the user input). Two inputs that mean the same thing "
16
+ "in different words can land in different clusters, and near-duplicate detection only "
17
+ "catches close textual matches.",
18
+ "Redaction is pattern-based (emails, common API key formats, Luhn-valid card numbers, "
19
+ "US-style phone numbers, SSN-shaped numbers). Names, addresses and free-form secrets are "
20
+ "not detected. Review traces before sharing them.",
21
+ "eval-builder does not write expected behavior. Cases are only as good as what the agent "
22
+ "and user put in cases.yaml.",
23
+ "Judge verdicts depend on the thresholds shown and on how many cases, trials and human "
24
+ "labels were provided. Small samples give wide intervals; read the intervals, not just "
25
+ "the point estimates.",
26
+ "The bias probes cover answer order and irrelevant padding only. They do not detect "
27
+ "self-preference, style bias or rubric misreadings.",
28
+ ]
29
+
30
+
31
+ _DEDUPE_LABEL = {"input": "the user turns", "input+output": "the user turns plus the output"}
32
+
33
+
34
+ def _pct(x: float | None) -> str:
35
+ return "n/a" if x is None else f"{x:.0%}"
36
+
37
+
38
+ def _ci(r: dict[str, Any] | None) -> str:
39
+ if not r or r.get("rate") is None:
40
+ return "n/a"
41
+ lo, hi = r["ci95"]
42
+ return f"{r['rate']:.0%} [{lo:.0%}, {hi:.0%}] (n={r['n']})"
43
+
44
+
45
+ def _md_escape(s: Any) -> str:
46
+ return str(s).replace("|", "\\|").replace("\n", " ")
47
+
48
+
49
+ def build_report(workspace: str | Path, title: str | None = None) -> dict[str, Any]:
50
+ ws = Workspace.at(workspace)
51
+ data: dict[str, Any] = {
52
+ "generated_at": datetime.now(UTC).isoformat(timespec="seconds"),
53
+ "eval_builder_version": __version__,
54
+ "workspace": str(ws.root),
55
+ }
56
+ if ws.ingest_report.exists():
57
+ data["ingest"] = read_json(ws.ingest_report)
58
+ if ws.selection.exists():
59
+ data["selection"] = read_json(ws.selection)
60
+ if ws.cases.exists():
61
+ cases = load_cases(ws.root).get("cases") or []
62
+ data["cases"] = {
63
+ "total": len(cases),
64
+ "by_status": {
65
+ s: sum(1 for c in cases if c.get("status") == s)
66
+ for s in ("draft", "ready", "dropped")
67
+ },
68
+ }
69
+ if ws.judge_check.exists():
70
+ data["judge_check"] = read_json(ws.judge_check)
71
+ manifest = ws.exports / "manifest.json"
72
+ if manifest.exists():
73
+ data["exports"] = read_json(manifest)
74
+ run_log = ws.root / "judge_run_log.json"
75
+ if run_log.exists():
76
+ data["judge_run"] = read_json(run_log)
77
+ data["limits"] = list(STATIC_LIMITS) + _dynamic_limits(data)
78
+ write_json(ws.report_json, data)
79
+ ws.report_md.write_text(render_markdown(data, title), encoding="utf-8")
80
+ return {
81
+ "report_md": str(ws.report_md),
82
+ "report_json": str(ws.report_json),
83
+ "sections": [
84
+ k for k in ("ingest", "selection", "cases", "judge_check", "exports") if k in data
85
+ ],
86
+ }
87
+
88
+
89
+ def _dynamic_limits(data: dict[str, Any]) -> list[str]:
90
+ out = []
91
+ jc = data.get("judge_check")
92
+ if jc and not jc.get("labels_file"):
93
+ out.append("No human labels were provided, so no judge can be marked trustworthy.")
94
+ sel = data.get("selection")
95
+ if sel and sel["population"]["failures_unique"] == 0:
96
+ out.append(
97
+ "No traces carried an error flag or negative feedback, so failure "
98
+ "oversampling had nothing to work with."
99
+ )
100
+ return out
101
+
102
+
103
+ def render_markdown(d: dict[str, Any], title: str | None = None) -> str:
104
+ L: list[str] = [f"# {title or 'eval-builder report'}", ""]
105
+ L.append(f"Generated {d['generated_at']} by eval-builder {d['eval_builder_version']}.")
106
+ L.append("")
107
+ ing = d.get("ingest")
108
+ if ing:
109
+ L += [
110
+ "## Sources",
111
+ "",
112
+ "| file | format | sha256 | records | traces | skipped |",
113
+ "|---|---|---|---|---|---|",
114
+ ]
115
+ for s in ing["sources"]:
116
+ skipped = ", ".join(f"{k}: {v}" for k, v in s["skipped"].items()) or "0"
117
+ L.append(
118
+ f"| {_md_escape(Path(s['path']).name)} | {s['format']} | "
119
+ f"`{s['sha256'][:16]}` | {s['records']} | {s['traces']} | "
120
+ f"{_md_escape(skipped)} |"
121
+ )
122
+ L += [
123
+ "",
124
+ f"Traces ingested: {ing['traces']} ({ing['multi_turn']} multi-turn, "
125
+ f"{ing['with_error']} with an error flag). Feedback: "
126
+ + ", ".join(f"{k} {v}" for k, v in ing["feedback"].items())
127
+ + ".",
128
+ "",
129
+ ]
130
+ r = ing["redactions"]
131
+ L += ["## Redactions", ""]
132
+ if not r["enabled"]:
133
+ L.append("Redaction was turned off for this run.")
134
+ elif r["total"] == 0:
135
+ L.append("Redaction was on. No matches.")
136
+ else:
137
+ L.append(
138
+ f"Redaction was on. {r['total']} value(s) replaced in "
139
+ f"{r['traces_affected']} trace(s): "
140
+ + ", ".join(f"{k} {v}" for k, v in r["by_kind"].items())
141
+ + "."
142
+ )
143
+ L.append("")
144
+ sel = d.get("selection")
145
+ if sel:
146
+ p = sel["population"]
147
+ L += [
148
+ "## Selection",
149
+ "",
150
+ f"{p['traces']} traces, {p['exact_duplicates_removed']} exact duplicates removed, "
151
+ f"{p['near_duplicates_merged']} near-duplicates merged (cosine >= "
152
+ f"{sel['params']['near_dup_threshold']} on "
153
+ f"{_DEDUPE_LABEL.get(sel['params']['dedupe_on'], sel['params']['dedupe_on'])}), "
154
+ f"{p['unique']} unique. Selected {sel['selected_count']} "
155
+ f"({sel['selected_failures']} failures; failures are {p['failures_unique']} of "
156
+ f"{p['unique']} unique traces). Seed {sel['params']['seed']}, "
157
+ f"{sel['params']['clusters']} k-means clusters.",
158
+ "",
159
+ "| cluster | traces | share | selected | top terms |",
160
+ "|---|---|---|---|---|",
161
+ ]
162
+ for c in sel["clusters"]:
163
+ L.append(
164
+ f"| {c['id']} | {c['size']} | {c['share']:.0%} | {c['selected']} | "
165
+ f"{_md_escape(', '.join(c['terms']))} |"
166
+ )
167
+ if sel.get("strata"):
168
+ L += ["", "| stratum | value | unique traces | selected |", "|---|---|---|---|"]
169
+ for f, vals in sel["strata"].items():
170
+ for v, cnt in vals.items():
171
+ L.append(f"| {f} | {_md_escape(v)} | {cnt['population']} | {cnt['selected']} |")
172
+ if sel.get("strata_skipped"):
173
+ L.append("")
174
+ L.append(
175
+ "Not stratified: "
176
+ + ", ".join(f"{k} ({v})" for k, v in sel["strata_skipped"].items())
177
+ )
178
+ L += ["", "| trace | cluster | represents | why it was picked |", "|---|---|---|---|"]
179
+ for s in sel["selected"]:
180
+ L.append(
181
+ f"| `{s['trace_id']}` | {s['cluster']} | {s['represents']} | "
182
+ f"{_md_escape('; '.join(s['reasons']))} |"
183
+ )
184
+ L.append("")
185
+ cs = d.get("cases")
186
+ if cs:
187
+ b = cs["by_status"]
188
+ L += [
189
+ "## Cases",
190
+ "",
191
+ f"{cs['total']} cases: {b['ready']} ready, {b['draft']} draft, {b['dropped']} dropped.",
192
+ "",
193
+ ]
194
+ jc = d.get("judge_check")
195
+ if jc:
196
+ L += ["## Judge reliability", ""]
197
+ th = jc["thresholds"]
198
+ L.append(
199
+ f"Thresholds: flip rate <= {th['max_flip_rate']:.0%} (cases with >= "
200
+ f"{th['min_trials']} trials, >= {th['min_cases']} cases), position "
201
+ f"consistency >= {th['min_position_consistency']:.0%}, padding moves verdict "
202
+ f"toward padded answer <= {th['max_toward_padded_rate']:.0%}, Cohen's kappa "
203
+ f">= {th['min_kappa']} on >= {th['min_labeled']} human-labeled cases. Intervals "
204
+ f"are 95% (Wilson for rates, Cohen's large-sample SE for kappa). Self-agreement is "
205
+ "the chance one call matches the judge's own majority; majority-of-3 stable is "
206
+ "the chance a 3-call majority vote matches it (exact, needs >= 5 trials)."
207
+ )
208
+ L += [
209
+ "",
210
+ "| judge | mode | verdict | flip rate | self-agreement | majority-of-3 stable | "
211
+ "accuracy vs humans | kappa | position consistency | first-shown picked | "
212
+ "padding helped |",
213
+ "|---|---|---|---|---|---|---|---|---|---|---|",
214
+ ]
215
+ for jid, r in jc["judges"].items():
216
+ st, ag = r["stability"], r["human_agreement"] or {}
217
+ pos, verb = r["position_probe"] or {}, r["verbosity_probe"] or {}
218
+ kappa = "n/a"
219
+ if ag.get("kappa") is not None:
220
+ kappa = f"{ag['kappa']:.2f}"
221
+ if ag.get("kappa_ci95"):
222
+ kappa += f" [{ag['kappa_ci95'][0]:.2f}, {ag['kappa_ci95'][1]:.2f}]"
223
+ L.append(
224
+ f"| {_md_escape(jid)} | {r['mode']} | **{r['verdict']}** | "
225
+ f"{_ci(st['flip_rate'])} | {_pct(st['mean_self_agreement'])} | "
226
+ f"{_pct(st.get('majority_of_3_stability'))} | "
227
+ f"{_ci(ag.get('accuracy'))} | {kappa} | {_ci(pos.get('consistency'))} | "
228
+ f"{_ci(pos.get('first_position_rate'))} | {_ci(verb.get('toward_padded'))} |"
229
+ )
230
+ L.append("")
231
+ for jid, r in jc["judges"].items():
232
+ L.append(
233
+ f"- **{jid}** ({r['verdict']}): "
234
+ + " ".join(s.rstrip(".") + "." for s in r["reasons"])
235
+ )
236
+ L.append("")
237
+ run = d.get("judge_run")
238
+ if run:
239
+ L += ["Judge calls made through the opt-in judge plugin (`judge-run`):", ""]
240
+ for i, rr in enumerate(run.get("runs", [run]), 1):
241
+ for jid, e in rr["judges"].items():
242
+ if not e["requests"]:
243
+ continue
244
+ L.append(
245
+ f"- run {i}, {jid}: `{_md_escape(e['command'])}`, {e['ok']} ok, "
246
+ f"{e['errors']} errors, {e['seconds']} s"
247
+ )
248
+ L.append("")
249
+ ex = d.get("exports")
250
+ if ex:
251
+ L += [
252
+ "## Exports",
253
+ "",
254
+ f"{ex['cases']} ready cases exported.",
255
+ "",
256
+ "| file | sha256 |",
257
+ "|---|---|",
258
+ ]
259
+ for f in ex["files"]:
260
+ L.append(f"| {f['path']} | `{f['sha256'][:16]}` |")
261
+ L.append("")
262
+ L += ["## Limits", ""] + [f"- {x}" for x in d["limits"]] + [""]
263
+ return "\n".join(L)
eval_builder/schema.py ADDED
@@ -0,0 +1,242 @@
1
+ """The normalized trace schema every ingest format maps into."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ import re
8
+ from dataclasses import asdict, dataclass, field
9
+ from typing import Any
10
+
11
+
12
+ @dataclass
13
+ class Trace:
14
+ """One logged interaction, normalized.
15
+
16
+ `input` is the last user message before the final assistant reply and `output`
17
+ is that reply's text. `messages` keeps the whole conversation (including the
18
+ final reply) so multi-turn context is never lost.
19
+ """
20
+
21
+ id: str
22
+ input: str
23
+ output: str
24
+ messages: list[dict[str, str]] = field(default_factory=list)
25
+ system: str | None = None
26
+ tools: list[str] = field(default_factory=list)
27
+ error: bool = False
28
+ feedback: str | None = None # "positive" | "negative" | "neutral" | None
29
+ route: str | None = None
30
+ model: str | None = None
31
+ metadata: dict[str, Any] = field(default_factory=dict)
32
+ source: dict[str, Any] = field(default_factory=dict)
33
+
34
+ def to_dict(self) -> dict[str, Any]:
35
+ return asdict(self)
36
+
37
+ @classmethod
38
+ def from_dict(cls, d: dict[str, Any]) -> Trace:
39
+ known = {f for f in cls.__dataclass_fields__}
40
+ return cls(**{k: v for k, v in d.items() if k in known})
41
+
42
+ def prior_turns(self) -> list[dict[str, str]]:
43
+ """Conversation before the last user message (the case input).
44
+
45
+ Tool calls and intermediate assistant steps after that message are part of the
46
+ app's behavior being evaluated, so they are not included.
47
+ """
48
+ last_user = max(
49
+ (i for i, m in enumerate(self.messages) if m.get("role") == "user"), default=0
50
+ )
51
+ return list(self.messages[:last_user])
52
+
53
+
54
+ def text_of(content: Any) -> str:
55
+ """Extract plain text from the many shapes providers use for message content."""
56
+ if content is None:
57
+ return ""
58
+ if isinstance(content, str):
59
+ return content
60
+ if isinstance(content, (int, float, bool)):
61
+ return str(content)
62
+ if isinstance(content, list):
63
+ parts = [text_of(p) for p in content]
64
+ return "\n".join(p for p in parts if p)
65
+ if isinstance(content, dict):
66
+ ptype = content.get("type")
67
+ if ptype in ("tool_use", "tool_call", "function_call", "image", "image_url", "input_image"):
68
+ return ""
69
+ for key in ("text", "content", "value", "output_text", "input_text"):
70
+ if key in content:
71
+ return text_of(content[key])
72
+ return ""
73
+ return str(content)
74
+
75
+
76
+ _FEEDBACK_POS = {
77
+ "up",
78
+ "thumbs_up",
79
+ "thumbsup",
80
+ "positive",
81
+ "good",
82
+ "like",
83
+ "liked",
84
+ "yes",
85
+ "true",
86
+ "+1",
87
+ "1",
88
+ "helpful",
89
+ "pass",
90
+ "correct",
91
+ }
92
+ _FEEDBACK_NEG = {
93
+ "down",
94
+ "thumbs_down",
95
+ "thumbsdown",
96
+ "negative",
97
+ "bad",
98
+ "dislike",
99
+ "disliked",
100
+ "no",
101
+ "false",
102
+ "-1",
103
+ "0",
104
+ "unhelpful",
105
+ "fail",
106
+ "incorrect",
107
+ "wrong",
108
+ }
109
+
110
+
111
+ def normalize_feedback(value: Any) -> str | None:
112
+ """Map thumbs, booleans, 0/1 and 1-5 ratings onto positive/negative/neutral."""
113
+ if value is None:
114
+ return None
115
+ if isinstance(value, bool):
116
+ return "positive" if value else "negative"
117
+ if isinstance(value, (int, float)):
118
+ v = float(value)
119
+ if v < 0:
120
+ return "negative"
121
+ if v <= 1:
122
+ return "negative" if v < 0.5 else "positive"
123
+ if v <= 5:
124
+ if v <= 2:
125
+ return "negative"
126
+ return "positive" if v >= 4 else "neutral"
127
+ return None
128
+ if isinstance(value, str):
129
+ s = value.strip().lower().replace(" ", "_")
130
+ if s in _FEEDBACK_POS:
131
+ return "positive"
132
+ if s in _FEEDBACK_NEG:
133
+ return "negative"
134
+ if s in ("neutral", "meh", "mixed"):
135
+ return "neutral"
136
+ try:
137
+ return normalize_feedback(float(s))
138
+ except ValueError:
139
+ return None
140
+ if isinstance(value, dict):
141
+ for key in ("value", "score", "rating", "label"):
142
+ if key in value:
143
+ return normalize_feedback(value[key])
144
+ return None
145
+
146
+
147
+ FEEDBACK_KEYS = ("feedback", "user_feedback", "thumbs", "rating", "user_rating", "vote")
148
+ ERROR_KEYS = ("error", "is_error", "failed", "exception")
149
+ ROUTE_KEYS = ("route", "endpoint", "path", "feature", "workflow", "task", "intent")
150
+
151
+
152
+ def fill_common_fields(trace: Trace) -> Trace:
153
+ """Derive error, feedback and route from metadata when a format did not set them."""
154
+ md = trace.metadata
155
+ if trace.feedback is None:
156
+ for key in FEEDBACK_KEYS:
157
+ if key in md:
158
+ trace.feedback = normalize_feedback(md[key])
159
+ if trace.feedback:
160
+ break
161
+ if not trace.error:
162
+ for key in ERROR_KEYS:
163
+ v = md.get(key)
164
+ if v not in (None, False, "", 0, "false", "False", "none", "None"):
165
+ trace.error = True
166
+ break
167
+ status = str(md.get("status", "")).lower()
168
+ if status in ("error", "failed", "failure"):
169
+ trace.error = True
170
+ if trace.route is None:
171
+ for key in ROUTE_KEYS:
172
+ v = md.get(key)
173
+ if isinstance(v, (str, int)) and str(v):
174
+ trace.route = str(v)
175
+ break
176
+ if trace.model is None and isinstance(md.get("model"), str):
177
+ trace.model = md["model"]
178
+ return trace
179
+
180
+
181
+ _ID_SAFE = re.compile(r"[^A-Za-z0-9_.:-]+")
182
+
183
+
184
+ def make_id(*parts: Any, given: Any = None) -> str:
185
+ if given not in (None, ""):
186
+ cleaned = _ID_SAFE.sub("-", str(given)).strip("-")
187
+ if cleaned:
188
+ return cleaned[:80]
189
+ blob = json.dumps(parts, ensure_ascii=False, sort_keys=True, default=str)
190
+ return "t-" + hashlib.sha1(blob.encode("utf-8")).hexdigest()[:12]
191
+
192
+
193
+ def build_trace(
194
+ messages: list[dict[str, Any]],
195
+ *,
196
+ fmt: str,
197
+ file: str,
198
+ index: int,
199
+ given_id: Any = None,
200
+ system: str | None = None,
201
+ tools: list[str] | None = None,
202
+ metadata: dict[str, Any] | None = None,
203
+ error: bool = False,
204
+ feedback: str | None = None,
205
+ route: str | None = None,
206
+ model: str | None = None,
207
+ ) -> Trace:
208
+ """Build a Trace from role/content messages (content already flattened to text)."""
209
+ norm: list[dict[str, str]] = []
210
+ sys_parts: list[str] = [system] if system else []
211
+ for m in messages:
212
+ role = str(m.get("role", "user")).lower()
213
+ if role in ("system", "developer"):
214
+ t = text_of(m.get("content"))
215
+ if t:
216
+ sys_parts.append(t)
217
+ continue
218
+ if role in ("human",):
219
+ role = "user"
220
+ if role in ("ai", "model", "bot"):
221
+ role = "assistant"
222
+ norm.append({"role": role, "content": text_of(m.get("content"))})
223
+ output = ""
224
+ if norm and norm[-1]["role"] == "assistant":
225
+ output = norm[-1]["content"]
226
+ user_msgs = [m["content"] for m in norm if m["role"] == "user"]
227
+ inp = user_msgs[-1] if user_msgs else ""
228
+ trace = Trace(
229
+ id=make_id(fmt, file, index, inp, output, given=given_id),
230
+ input=inp,
231
+ output=output,
232
+ messages=norm,
233
+ system="\n\n".join(sys_parts) or None,
234
+ tools=sorted(set(tools or [])),
235
+ error=error,
236
+ feedback=feedback,
237
+ route=route,
238
+ model=model,
239
+ metadata=dict(metadata or {}),
240
+ source={"format": fmt, "file": file, "index": index},
241
+ )
242
+ return fill_common_fields(trace)