eval-builder 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- eval_builder/__init__.py +3 -0
- eval_builder/cli.py +354 -0
- eval_builder/data/SKILL.md +78 -0
- eval_builder/draft.py +271 -0
- eval_builder/export.py +284 -0
- eval_builder/ingest/__init__.py +181 -0
- eval_builder/ingest/formats.py +645 -0
- eval_builder/io.py +82 -0
- eval_builder/judge/__init__.py +1 -0
- eval_builder/judge/check.py +420 -0
- eval_builder/judge/plan.py +151 -0
- eval_builder/judge/run.py +149 -0
- eval_builder/judge/stats.py +76 -0
- eval_builder/mcp_server.py +175 -0
- eval_builder/redact.py +101 -0
- eval_builder/report.py +263 -0
- eval_builder/schema.py +242 -0
- eval_builder/select.py +412 -0
- eval_builder/setup_agents.py +190 -0
- eval_builder/status.py +32 -0
- eval_builder/workspace.py +67 -0
- eval_builder-0.1.0.dist-info/METADATA +154 -0
- eval_builder-0.1.0.dist-info/RECORD +26 -0
- eval_builder-0.1.0.dist-info/WHEEL +4 -0
- eval_builder-0.1.0.dist-info/entry_points.txt +2 -0
- eval_builder-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
"""Read traces from supported formats, normalize, redact, and write traces.jsonl."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from collections import Counter
|
|
7
|
+
from collections.abc import Sequence
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from ..io import sha256_file, write_json, write_jsonl
|
|
12
|
+
from ..redact import Redactor
|
|
13
|
+
from ..schema import Trace
|
|
14
|
+
from ..workspace import Workspace
|
|
15
|
+
from .formats import FORMATS, PARSERS, SkipRecord, detect_format, iter_otel_spans, parse_otel_trace
|
|
16
|
+
|
|
17
|
+
__all__ = ["FORMATS", "ingest", "load_records", "parse_file"]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def load_records(path: Path) -> list[Any]:
|
|
21
|
+
"""Load a .json file (array, object, or {"data": [...]}) or a .jsonl file."""
|
|
22
|
+
text = path.read_text(encoding="utf-8")
|
|
23
|
+
stripped = text.lstrip()
|
|
24
|
+
if stripped.startswith("[") or stripped.startswith("{"):
|
|
25
|
+
try:
|
|
26
|
+
doc = json.loads(text)
|
|
27
|
+
except json.JSONDecodeError:
|
|
28
|
+
doc = None
|
|
29
|
+
if isinstance(doc, list):
|
|
30
|
+
return doc
|
|
31
|
+
if isinstance(doc, dict):
|
|
32
|
+
if isinstance(doc.get("data"), list) and "resourceSpans" not in doc:
|
|
33
|
+
return doc["data"]
|
|
34
|
+
if isinstance(doc.get("traces"), list):
|
|
35
|
+
return doc["traces"]
|
|
36
|
+
return [doc]
|
|
37
|
+
records = []
|
|
38
|
+
for lineno, line in enumerate(text.splitlines(), 1):
|
|
39
|
+
line = line.strip()
|
|
40
|
+
if not line:
|
|
41
|
+
continue
|
|
42
|
+
try:
|
|
43
|
+
records.append(json.loads(line))
|
|
44
|
+
except json.JSONDecodeError as e:
|
|
45
|
+
raise ValueError(f"{path}:{lineno}: not valid JSON or JSONL ({e.msg})") from e
|
|
46
|
+
return records
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def parse_file(path: Path, fmt: str | None = None) -> tuple[str, list[Trace], Counter[str], int]:
|
|
50
|
+
"""Return (format, traces, skip reasons, records read)."""
|
|
51
|
+
records = load_records(path)
|
|
52
|
+
fmt = fmt or detect_format(records)
|
|
53
|
+
if fmt not in FORMATS:
|
|
54
|
+
raise ValueError(f"unknown format {fmt!r}; expected one of {', '.join(FORMATS)}")
|
|
55
|
+
skipped: Counter[str] = Counter()
|
|
56
|
+
traces: list[Trace] = []
|
|
57
|
+
name = path.name
|
|
58
|
+
if fmt == "otel":
|
|
59
|
+
groups: dict[str, list[dict[str, Any]]] = {}
|
|
60
|
+
order: list[str] = []
|
|
61
|
+
for rec in records:
|
|
62
|
+
if not isinstance(rec, dict):
|
|
63
|
+
skipped["not a JSON object"] += 1
|
|
64
|
+
continue
|
|
65
|
+
for span in iter_otel_spans(rec):
|
|
66
|
+
tid = str(
|
|
67
|
+
span.get("traceId")
|
|
68
|
+
or span.get("trace_id")
|
|
69
|
+
or (span.get("context") or {}).get("trace_id")
|
|
70
|
+
or ""
|
|
71
|
+
)
|
|
72
|
+
if tid not in groups:
|
|
73
|
+
order.append(tid)
|
|
74
|
+
groups.setdefault(tid, []).append(span)
|
|
75
|
+
for i, tid in enumerate(order):
|
|
76
|
+
try:
|
|
77
|
+
traces.append(parse_otel_trace(tid, groups[tid], name, i))
|
|
78
|
+
except SkipRecord as e:
|
|
79
|
+
skipped[str(e)] += 1
|
|
80
|
+
return fmt, traces, skipped, len(records)
|
|
81
|
+
parser = PARSERS[fmt]
|
|
82
|
+
for i, rec in enumerate(records):
|
|
83
|
+
if not isinstance(rec, dict):
|
|
84
|
+
skipped["not a JSON object"] += 1
|
|
85
|
+
continue
|
|
86
|
+
try:
|
|
87
|
+
t = parser(rec, name, i)
|
|
88
|
+
except SkipRecord as e:
|
|
89
|
+
skipped[str(e)] += 1
|
|
90
|
+
continue
|
|
91
|
+
except (TypeError, AttributeError, KeyError) as e:
|
|
92
|
+
skipped[f"malformed record ({type(e).__name__})"] += 1
|
|
93
|
+
continue
|
|
94
|
+
if not t.input.strip() and not t.output.strip():
|
|
95
|
+
skipped["empty input and output"] += 1
|
|
96
|
+
continue
|
|
97
|
+
if not t.input.strip():
|
|
98
|
+
skipped["no user message"] += 1
|
|
99
|
+
continue
|
|
100
|
+
traces.append(t)
|
|
101
|
+
return fmt, traces, skipped, len(records)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _expand(paths: Sequence[str | Path]) -> list[Path]:
|
|
105
|
+
out: list[Path] = []
|
|
106
|
+
for p in map(Path, paths):
|
|
107
|
+
if p.is_dir():
|
|
108
|
+
out.extend(sorted(x for x in p.rglob("*") if x.suffix in (".json", ".jsonl")))
|
|
109
|
+
elif p.exists():
|
|
110
|
+
out.append(p)
|
|
111
|
+
else:
|
|
112
|
+
raise FileNotFoundError(f"no such file or directory: {p}")
|
|
113
|
+
return out
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _redact_trace(t: Trace, r: Redactor) -> Trace:
|
|
117
|
+
r.start_trace(t.id)
|
|
118
|
+
contents = {m["content"] for m in t.messages}
|
|
119
|
+
t.messages = [{"role": m["role"], "content": r.text(m["content"])} for m in t.messages]
|
|
120
|
+
# input and output are copies of message text; do not count their matches twice
|
|
121
|
+
t.input = r.text(t.input, count=t.input not in contents)
|
|
122
|
+
t.output = r.text(t.output, count=t.output not in contents)
|
|
123
|
+
t.system = r.text(t.system) if t.system else t.system
|
|
124
|
+
t.metadata = r.value(t.metadata)
|
|
125
|
+
return t
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def ingest(
|
|
129
|
+
paths: Sequence[str | Path],
|
|
130
|
+
workspace: str | Path,
|
|
131
|
+
fmt: str | None = None,
|
|
132
|
+
redact: bool = True,
|
|
133
|
+
) -> dict[str, Any]:
|
|
134
|
+
"""Ingest one or more files or directories into the workspace. Overwrites traces.jsonl."""
|
|
135
|
+
ws = Workspace.at(workspace).ensure()
|
|
136
|
+
redactor = Redactor(enabled=redact)
|
|
137
|
+
all_traces: list[Trace] = []
|
|
138
|
+
sources = []
|
|
139
|
+
seen_ids: Counter[str] = Counter()
|
|
140
|
+
for path in _expand(paths):
|
|
141
|
+
used_fmt, traces, skipped, n_records = parse_file(path, fmt)
|
|
142
|
+
for t in traces:
|
|
143
|
+
seen_ids[t.id] += 1
|
|
144
|
+
if seen_ids[t.id] > 1:
|
|
145
|
+
t.id = f"{t.id}~{seen_ids[t.id]}"
|
|
146
|
+
all_traces.append(_redact_trace(t, redactor))
|
|
147
|
+
sources.append(
|
|
148
|
+
{
|
|
149
|
+
"path": str(path),
|
|
150
|
+
"sha256": sha256_file(path),
|
|
151
|
+
"format": used_fmt,
|
|
152
|
+
"records": n_records,
|
|
153
|
+
"traces": len(traces),
|
|
154
|
+
"skipped": dict(skipped),
|
|
155
|
+
}
|
|
156
|
+
)
|
|
157
|
+
write_jsonl(ws.traces, (t.to_dict() for t in all_traces))
|
|
158
|
+
summary = {
|
|
159
|
+
"traces": len(all_traces),
|
|
160
|
+
"sources": sources,
|
|
161
|
+
"redactions": redactor.summary(),
|
|
162
|
+
"with_error": sum(t.error for t in all_traces),
|
|
163
|
+
"feedback": dict(Counter(t.feedback or "none" for t in all_traces)),
|
|
164
|
+
"routes": dict(Counter(t.route or "none" for t in all_traces).most_common(20)),
|
|
165
|
+
"tools": dict(Counter(tool for t in all_traces for tool in t.tools).most_common(20)),
|
|
166
|
+
"multi_turn": sum(
|
|
167
|
+
1 for t in all_traces if sum(m["role"] == "user" for m in t.messages) > 1
|
|
168
|
+
),
|
|
169
|
+
"output_file": str(ws.traces),
|
|
170
|
+
}
|
|
171
|
+
write_json(ws.ingest_report, summary)
|
|
172
|
+
return summary
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def load_traces(workspace: str | Path) -> list[Trace]:
|
|
176
|
+
from ..io import read_jsonl
|
|
177
|
+
|
|
178
|
+
ws = Workspace.at(workspace)
|
|
179
|
+
if not ws.traces.exists():
|
|
180
|
+
raise FileNotFoundError(f"{ws.traces} not found; run `eval-builder ingest` first")
|
|
181
|
+
return [Trace.from_dict(d) for d in read_jsonl(ws.traces)]
|