evalkeep 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalkeep/__init__.py +12 -0
- evalkeep/__main__.py +6 -0
- evalkeep/adapters/__init__.py +45 -0
- evalkeep/adapters/base.py +92 -0
- evalkeep/adapters/jsonl.py +164 -0
- evalkeep/adapters/langsmith.py +436 -0
- evalkeep/adapters/otlp.py +442 -0
- evalkeep/adapters/semconv.py +208 -0
- evalkeep/analysis.py +174 -0
- evalkeep/analysis_run.py +160 -0
- evalkeep/analyzers/__init__.py +52 -0
- evalkeep/analyzers/anthropic.py +145 -0
- evalkeep/analyzers/stub.py +34 -0
- evalkeep/cache.py +122 -0
- evalkeep/cli.py +1933 -0
- evalkeep/clustering.py +383 -0
- evalkeep/clusters.py +101 -0
- evalkeep/commands/__init__.py +1 -0
- evalkeep/commands/analyze_cmd.py +100 -0
- evalkeep/commands/compare_cmd.py +169 -0
- evalkeep/commands/dataset_cmd.py +182 -0
- evalkeep/commands/detect_cmd.py +154 -0
- evalkeep/commands/discover_cmd.py +274 -0
- evalkeep/commands/ingest_cmd.py +50 -0
- evalkeep/commands/init_cmd.py +151 -0
- evalkeep/commands/pipeline_cmd.py +156 -0
- evalkeep/commands/review_cmd.py +141 -0
- evalkeep/commands/run_cmd.py +131 -0
- evalkeep/commands/target_cmd.py +109 -0
- evalkeep/commands/trace_cmd.py +58 -0
- evalkeep/comparison.py +432 -0
- evalkeep/config.py +209 -0
- evalkeep/detection.py +94 -0
- evalkeep/detectors.py +182 -0
- evalkeep/discovery.py +208 -0
- evalkeep/embeddings/__init__.py +31 -0
- evalkeep/embeddings/base.py +32 -0
- evalkeep/embeddings/hashing.py +98 -0
- evalkeep/errors.py +42 -0
- evalkeep/examples/__init__.py +37 -0
- evalkeep/examples/langsmith/runs.jsonl +18 -0
- evalkeep/examples/opentelemetry/spans.json +898 -0
- evalkeep/examples/refund-agent/agents/baseline.py +66 -0
- evalkeep/examples/refund-agent/agents/candidate.py +66 -0
- evalkeep/examples/refund-agent/traces.jsonl +5 -0
- evalkeep/examples/tau-bench/prepare.py +230 -0
- evalkeep/exporters/__init__.py +45 -0
- evalkeep/exporters/generic.py +31 -0
- evalkeep/exporters/promptfoo.py +219 -0
- evalkeep/failures.py +95 -0
- evalkeep/generation.py +303 -0
- evalkeep/hashing.py +56 -0
- evalkeep/ingest.py +257 -0
- evalkeep/prompts.py +127 -0
- evalkeep/pseudonyms.py +82 -0
- evalkeep/py.typed +0 -0
- evalkeep/redaction.py +333 -0
- evalkeep/regression.py +409 -0
- evalkeep/review.py +309 -0
- evalkeep/runner.py +302 -0
- evalkeep/runs.py +185 -0
- evalkeep/storage/__init__.py +37 -0
- evalkeep/storage/clusters.py +163 -0
- evalkeep/storage/failures.py +254 -0
- evalkeep/storage/migrations.py +370 -0
- evalkeep/storage/regression.py +136 -0
- evalkeep/storage/runs.py +223 -0
- evalkeep/storage/store.py +429 -0
- evalkeep/targets.py +205 -0
- evalkeep/trace.py +238 -0
- evalkeep-0.1.0.dist-info/METADATA +221 -0
- evalkeep-0.1.0.dist-info/RECORD +75 -0
- evalkeep-0.1.0.dist-info/WHEEL +4 -0
- evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
- evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/review.py
ADDED
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
"""Human review: the gate between a generated draft and a committed test.
|
|
2
|
+
|
|
3
|
+
Everything here is a pure function over a test. The interactive terminal loop is
|
|
4
|
+
a thin shell in the CLI that calls these, which is what keeps the guide's
|
|
5
|
+
"non-interactive CI review format" possible later: a different front end -- a
|
|
6
|
+
web form, a PR check, a batch file of decisions -- needs no new logic, only a
|
|
7
|
+
different way of collecting the same decisions.
|
|
8
|
+
|
|
9
|
+
Two rules the review gate enforces rather than suggests:
|
|
10
|
+
|
|
11
|
+
* **A contradictory test cannot be approved.** It would fail on a correct agent
|
|
12
|
+
too, reporting a regression that is really a bug in the suite.
|
|
13
|
+
* **Editing cannot change what a test *is*.** A reviewer edits the input and the
|
|
14
|
+
expectations -- the parts that encode intent. The test ID, provenance and
|
|
15
|
+
fixtures are facts about where the test came from, and letting a review rewrite
|
|
16
|
+
them would make the audit trail describe something that never happened.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from dataclasses import dataclass, field
|
|
22
|
+
from datetime import UTC, datetime
|
|
23
|
+
from enum import StrEnum
|
|
24
|
+
from typing import Any
|
|
25
|
+
|
|
26
|
+
import yaml
|
|
27
|
+
|
|
28
|
+
from evalkeep.generation import NO_POSITIVE_EXPECTATION
|
|
29
|
+
from evalkeep.regression import (
|
|
30
|
+
CaseInput,
|
|
31
|
+
Expectation,
|
|
32
|
+
ExpectationType,
|
|
33
|
+
RegressionTest,
|
|
34
|
+
ReviewStatus,
|
|
35
|
+
find_contradictions,
|
|
36
|
+
validate_expectation,
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
#: The only fields a reviewer may change.
|
|
40
|
+
EDITABLE_FIELDS = ("input", "expectations")
|
|
41
|
+
|
|
42
|
+
EDIT_HELP = """\
|
|
43
|
+
# Editing {test_id}
|
|
44
|
+
#
|
|
45
|
+
# Change the input and the expectations. Everything else -- the test ID, the
|
|
46
|
+
# provenance, the recorded fixtures -- describes where this test came from and
|
|
47
|
+
# is not editable.
|
|
48
|
+
#
|
|
49
|
+
# Expectation types:
|
|
50
|
+
# output_contains value: text the answer must contain
|
|
51
|
+
# output_not_contains value: text the answer must not contain
|
|
52
|
+
# output_matches value: a regular expression the answer must match
|
|
53
|
+
# tool_called tool: a tool that must be called
|
|
54
|
+
# tool_not_called tool: a tool that must never be called
|
|
55
|
+
# tool_argument_equals tool, path, value
|
|
56
|
+
# tool_argument_not_equals tool, path, value
|
|
57
|
+
# max_tool_calls value: a whole number; tool: optional
|
|
58
|
+
# human_rubric value: what should happen, judged by a model at run time
|
|
59
|
+
#
|
|
60
|
+
# Delete every expectation to abandon the edit.
|
|
61
|
+
"""
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class ReviewDecision(StrEnum):
|
|
65
|
+
APPROVE = "approve"
|
|
66
|
+
EDIT = "edit"
|
|
67
|
+
REJECT = "reject"
|
|
68
|
+
SKIP = "skip"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@dataclass
|
|
72
|
+
class ReviewOutcome:
|
|
73
|
+
"""What one review session did."""
|
|
74
|
+
|
|
75
|
+
reviewed: int = 0
|
|
76
|
+
approved: int = 0
|
|
77
|
+
rejected: int = 0
|
|
78
|
+
edited: int = 0
|
|
79
|
+
skipped: int = 0
|
|
80
|
+
remaining: int = 0
|
|
81
|
+
|
|
82
|
+
@property
|
|
83
|
+
def changed(self) -> bool:
|
|
84
|
+
return bool(self.approved or self.rejected or self.edited)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
@dataclass
|
|
88
|
+
class EditResult:
|
|
89
|
+
"""A parsed edit, or the reasons it could not be applied."""
|
|
90
|
+
|
|
91
|
+
test: RegressionTest | None = None
|
|
92
|
+
errors: list[str] = field(default_factory=list)
|
|
93
|
+
|
|
94
|
+
@property
|
|
95
|
+
def ok(self) -> bool:
|
|
96
|
+
return self.test is not None and not self.errors
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class ReviewError(Exception):
|
|
100
|
+
"""A decision that cannot be recorded as asked."""
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def recompute_warnings(test: RegressionTest) -> list[str]:
|
|
104
|
+
"""What still needs a reviewer's attention, for the test as it stands now.
|
|
105
|
+
|
|
106
|
+
Recomputed rather than carried forward: warnings describe current content,
|
|
107
|
+
and a note about how the draft was generated stops being true the moment a
|
|
108
|
+
person edits it.
|
|
109
|
+
"""
|
|
110
|
+
warnings: list[str] = []
|
|
111
|
+
for expectation in test.expectations:
|
|
112
|
+
problem = validate_expectation(expectation)
|
|
113
|
+
if problem is not None:
|
|
114
|
+
warnings.append(f"Invalid expectation ({expectation.describe()}): {problem}")
|
|
115
|
+
warnings.extend(
|
|
116
|
+
f"Contradictory expectations: {contradiction.describe()}"
|
|
117
|
+
for contradiction in test.contradictions
|
|
118
|
+
)
|
|
119
|
+
if not test.has_positive_expectation:
|
|
120
|
+
warnings.append(NO_POSITIVE_EXPECTATION)
|
|
121
|
+
return warnings
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def blocking_problems(test: RegressionTest) -> list[str]:
|
|
125
|
+
"""Reasons a test must not be approved as it stands."""
|
|
126
|
+
problems = [
|
|
127
|
+
f"Invalid expectation ({expectation.describe()}): {problem}"
|
|
128
|
+
for expectation in test.expectations
|
|
129
|
+
if (problem := validate_expectation(expectation)) is not None
|
|
130
|
+
]
|
|
131
|
+
problems.extend(
|
|
132
|
+
f"Contradictory expectations: {contradiction.describe()}"
|
|
133
|
+
for contradiction in test.contradictions
|
|
134
|
+
)
|
|
135
|
+
if not test.expectations:
|
|
136
|
+
problems.append("A test with no expectations checks nothing.")
|
|
137
|
+
return problems
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def approve(test: RegressionTest, *, reviewer: str, reason: str | None = None) -> RegressionTest:
|
|
141
|
+
"""Approve a test, refusing if it could never pass on a correct agent."""
|
|
142
|
+
problems = blocking_problems(test)
|
|
143
|
+
if problems:
|
|
144
|
+
raise ReviewError(
|
|
145
|
+
"This test cannot be approved as it stands:\n - " + "\n - ".join(problems)
|
|
146
|
+
)
|
|
147
|
+
return _record(test, ReviewStatus.APPROVED, reviewer=reviewer, reason=reason)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def reject(test: RegressionTest, *, reviewer: str, reason: str | None = None) -> RegressionTest:
|
|
151
|
+
"""Reject a test. The record is kept, not deleted: a rejection is evidence."""
|
|
152
|
+
return _record(test, ReviewStatus.REJECTED, reviewer=reviewer, reason=reason)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _record(
|
|
156
|
+
test: RegressionTest, status: ReviewStatus, *, reviewer: str, reason: str | None
|
|
157
|
+
) -> RegressionTest:
|
|
158
|
+
test.status = status
|
|
159
|
+
test.reviewer = reviewer
|
|
160
|
+
test.review_reason = reason
|
|
161
|
+
test.reviewed_at = datetime.now(UTC)
|
|
162
|
+
test.updated_at = test.reviewed_at
|
|
163
|
+
test.warnings = recompute_warnings(test)
|
|
164
|
+
return test
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def render_editable(test: RegressionTest) -> str:
|
|
168
|
+
"""The YAML a reviewer edits, with the guidance they need above it."""
|
|
169
|
+
body = {
|
|
170
|
+
"input": {
|
|
171
|
+
key: value for key, value in test.input.to_dict().items() if value not in (None, [])
|
|
172
|
+
},
|
|
173
|
+
"expectations": [expectation.to_dict() for expectation in test.expectations],
|
|
174
|
+
}
|
|
175
|
+
header = EDIT_HELP.format(test_id=test.test_id)
|
|
176
|
+
return header + yaml.safe_dump(body, sort_keys=False, default_flow_style=False)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def apply_edits(test: RegressionTest, text: str, *, editor: str) -> EditResult:
|
|
180
|
+
"""Parse an edited document and return an updated copy, or the errors.
|
|
181
|
+
|
|
182
|
+
Nothing is written until the result is valid: an unparseable or
|
|
183
|
+
self-contradictory edit leaves the stored draft exactly as it was.
|
|
184
|
+
"""
|
|
185
|
+
try:
|
|
186
|
+
# safe_load, never load: this document is arbitrary text from an editor,
|
|
187
|
+
# and full YAML can construct objects.
|
|
188
|
+
raw: Any = yaml.safe_load(text)
|
|
189
|
+
except yaml.YAMLError as exc:
|
|
190
|
+
return EditResult(errors=[f"Could not parse YAML: {exc}"])
|
|
191
|
+
|
|
192
|
+
if raw is None:
|
|
193
|
+
return EditResult(errors=["The document is empty."])
|
|
194
|
+
if not isinstance(raw, dict):
|
|
195
|
+
return EditResult(errors=[f"Expected a mapping, got {type(raw).__name__}."])
|
|
196
|
+
|
|
197
|
+
unknown = set(raw) - set(EDITABLE_FIELDS)
|
|
198
|
+
if unknown:
|
|
199
|
+
return EditResult(
|
|
200
|
+
errors=[
|
|
201
|
+
f"Only {' and '.join(EDITABLE_FIELDS)} can be edited; "
|
|
202
|
+
f"remove: {', '.join(sorted(unknown))}."
|
|
203
|
+
]
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
errors: list[str] = []
|
|
207
|
+
case_input, input_errors = _parse_input(raw.get("input"))
|
|
208
|
+
errors.extend(input_errors)
|
|
209
|
+
expectations, expectation_errors = _parse_expectations(raw.get("expectations"))
|
|
210
|
+
errors.extend(expectation_errors)
|
|
211
|
+
|
|
212
|
+
if errors:
|
|
213
|
+
return EditResult(errors=errors)
|
|
214
|
+
|
|
215
|
+
updated = _copy_with_edits(test, case_input, expectations, editor=editor)
|
|
216
|
+
contradictions = [
|
|
217
|
+
f"Contradictory expectations: {contradiction.describe()}"
|
|
218
|
+
for contradiction in find_contradictions(expectations)
|
|
219
|
+
]
|
|
220
|
+
if contradictions:
|
|
221
|
+
return EditResult(errors=contradictions)
|
|
222
|
+
return EditResult(test=updated)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _copy_with_edits(
|
|
226
|
+
test: RegressionTest,
|
|
227
|
+
case_input: CaseInput,
|
|
228
|
+
expectations: list[Expectation],
|
|
229
|
+
*,
|
|
230
|
+
editor: str,
|
|
231
|
+
) -> RegressionTest:
|
|
232
|
+
now = datetime.now(UTC)
|
|
233
|
+
updated = RegressionTest(
|
|
234
|
+
test_id=test.test_id,
|
|
235
|
+
failure_id=test.failure_id,
|
|
236
|
+
input=case_input,
|
|
237
|
+
provenance=test.provenance,
|
|
238
|
+
status=test.status,
|
|
239
|
+
fixtures=test.fixtures,
|
|
240
|
+
expectations=expectations,
|
|
241
|
+
reviewer=test.reviewer,
|
|
242
|
+
review_reason=test.review_reason,
|
|
243
|
+
reviewed_at=test.reviewed_at,
|
|
244
|
+
edited=True,
|
|
245
|
+
edited_by=editor,
|
|
246
|
+
created_at=test.created_at,
|
|
247
|
+
updated_at=now,
|
|
248
|
+
)
|
|
249
|
+
updated.warnings = recompute_warnings(updated)
|
|
250
|
+
return updated
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _parse_input(raw: Any) -> tuple[CaseInput, list[str]]:
|
|
254
|
+
if raw is None:
|
|
255
|
+
return CaseInput(), ["input is required."]
|
|
256
|
+
if not isinstance(raw, dict):
|
|
257
|
+
return CaseInput(), [f"input must be a mapping, got {type(raw).__name__}."]
|
|
258
|
+
|
|
259
|
+
text = raw.get("text")
|
|
260
|
+
messages = raw.get("messages") or []
|
|
261
|
+
if text is not None and not isinstance(text, str):
|
|
262
|
+
return CaseInput(), ["input.text must be text."]
|
|
263
|
+
if not isinstance(messages, list):
|
|
264
|
+
return CaseInput(), ["input.messages must be a list."]
|
|
265
|
+
|
|
266
|
+
parsed: list[dict[str, str]] = []
|
|
267
|
+
for index, message in enumerate(messages):
|
|
268
|
+
if not isinstance(message, dict) or "role" not in message or "content" not in message:
|
|
269
|
+
return CaseInput(), [f"input.messages[{index}] needs a role and content."]
|
|
270
|
+
parsed.append({"role": str(message["role"]), "content": str(message["content"])})
|
|
271
|
+
|
|
272
|
+
if not (text or "").strip() and not parsed:
|
|
273
|
+
return CaseInput(), ["input needs non-empty text or at least one message."]
|
|
274
|
+
return CaseInput(text=text, messages=parsed), []
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _parse_expectations(raw: Any) -> tuple[list[Expectation], list[str]]:
|
|
278
|
+
if raw is None or raw == []:
|
|
279
|
+
return [], ["A test needs at least one expectation."]
|
|
280
|
+
if not isinstance(raw, list):
|
|
281
|
+
return [], [f"expectations must be a list, got {type(raw).__name__}."]
|
|
282
|
+
|
|
283
|
+
expectations: list[Expectation] = []
|
|
284
|
+
errors: list[str] = []
|
|
285
|
+
for index, item in enumerate(raw):
|
|
286
|
+
if not isinstance(item, dict):
|
|
287
|
+
errors.append(f"expectations[{index}] must be a mapping.")
|
|
288
|
+
continue
|
|
289
|
+
kind = item.get("type")
|
|
290
|
+
try:
|
|
291
|
+
expectation_type = ExpectationType(str(kind))
|
|
292
|
+
except ValueError:
|
|
293
|
+
known = ", ".join(member.value for member in ExpectationType)
|
|
294
|
+
errors.append(f"expectations[{index}]: unknown type {kind!r}. Known types: {known}.")
|
|
295
|
+
continue
|
|
296
|
+
|
|
297
|
+
expectation = Expectation(
|
|
298
|
+
type=expectation_type,
|
|
299
|
+
value=item.get("value"),
|
|
300
|
+
tool=item.get("tool"),
|
|
301
|
+
path=item.get("path"),
|
|
302
|
+
)
|
|
303
|
+
problem = validate_expectation(expectation)
|
|
304
|
+
if problem is not None:
|
|
305
|
+
errors.append(f"expectations[{index}]: {problem}")
|
|
306
|
+
continue
|
|
307
|
+
expectations.append(expectation)
|
|
308
|
+
|
|
309
|
+
return expectations, errors
|
evalkeep/runner.py
ADDED
|
@@ -0,0 +1,302 @@
|
|
|
1
|
+
"""Delegating execution to Promptfoo, and reading the results back.
|
|
2
|
+
|
|
3
|
+
Two things this module is careful about:
|
|
4
|
+
|
|
5
|
+
* **The runner is invoked as an argument list, never through a shell.** Test
|
|
6
|
+
inputs, tool names and file paths all come from recorded traces. Building a
|
|
7
|
+
command string out of them would make a trace containing ``; rm -rf`` a
|
|
8
|
+
remote-code-execution bug, so ``subprocess.run`` is called with a list and
|
|
9
|
+
``shell=False``, and nothing is ever passed through a shell.
|
|
10
|
+
* **A test that never ran is not a test that failed.** Promptfoo distinguishes
|
|
11
|
+
an assertion failure from a provider error, and so does the import: an error
|
|
12
|
+
is recorded as :class:`~evalkeep.runs.Outcome.ERROR`, never as a failure.
|
|
13
|
+
Letting a crashed provider look like a regression is precisely the wrong
|
|
14
|
+
answer for a tool whose job is deciding whether a release got worse.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import json
|
|
20
|
+
import platform
|
|
21
|
+
import subprocess
|
|
22
|
+
import sys
|
|
23
|
+
import uuid
|
|
24
|
+
from dataclasses import dataclass, field
|
|
25
|
+
from datetime import UTC, datetime
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
from typing import Any
|
|
28
|
+
|
|
29
|
+
import yaml
|
|
30
|
+
|
|
31
|
+
from evalkeep.errors import CommandError
|
|
32
|
+
from evalkeep.exporters.promptfoo import build_config
|
|
33
|
+
from evalkeep.redaction import RedactionSummary, Redactor
|
|
34
|
+
from evalkeep.regression import RegressionTest
|
|
35
|
+
from evalkeep.runs import (
|
|
36
|
+
CaseResult,
|
|
37
|
+
CaseSummary,
|
|
38
|
+
ErrorKind,
|
|
39
|
+
EvaluationRun,
|
|
40
|
+
Outcome,
|
|
41
|
+
RunStatus,
|
|
42
|
+
suite_hash,
|
|
43
|
+
summarize,
|
|
44
|
+
)
|
|
45
|
+
from evalkeep.targets import Target, referenced_environment
|
|
46
|
+
|
|
47
|
+
CONFIG_FILENAME = "promptfooconfig.yaml"
|
|
48
|
+
RESULTS_FILENAME = "results.json"
|
|
49
|
+
|
|
50
|
+
#: Promptfoo's own failure taxonomy, which the import preserves rather than
|
|
51
|
+
#: flattening: 0 none, 1 assertion, 2 provider/execution error.
|
|
52
|
+
_ASSERTION_FAILURE = 1
|
|
53
|
+
_EXECUTION_ERROR = 2
|
|
54
|
+
|
|
55
|
+
_TIMEOUT_MARKERS = ("timeout", "timed out", "etimedout", "esockettimedout")
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass
|
|
59
|
+
class RunOutcome:
|
|
60
|
+
run: EvaluationRun
|
|
61
|
+
results: list[CaseResult] = field(default_factory=list)
|
|
62
|
+
#: Anything the runner said that a person should see.
|
|
63
|
+
messages: list[str] = field(default_factory=list)
|
|
64
|
+
|
|
65
|
+
@property
|
|
66
|
+
def counts(self) -> dict[Outcome, int]:
|
|
67
|
+
"""Per-execution counts. With repetitions these exceed the test count."""
|
|
68
|
+
tally: dict[Outcome, int] = {}
|
|
69
|
+
for result in self.results:
|
|
70
|
+
tally[result.outcome] = tally.get(result.outcome, 0) + 1
|
|
71
|
+
return tally
|
|
72
|
+
|
|
73
|
+
@property
|
|
74
|
+
def summaries(self) -> dict[str, CaseSummary]:
|
|
75
|
+
"""Per-case verdicts, which is what a repeated run is actually for."""
|
|
76
|
+
return summarize(self.results)
|
|
77
|
+
|
|
78
|
+
@property
|
|
79
|
+
def flaky(self) -> list[CaseSummary]:
|
|
80
|
+
return [s for s in self.summaries.values() if s.flaky]
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def write_suite(
|
|
84
|
+
tests: list[RegressionTest],
|
|
85
|
+
target: Target,
|
|
86
|
+
directory: Path,
|
|
87
|
+
*,
|
|
88
|
+
project_root: Path | None = None,
|
|
89
|
+
) -> Path:
|
|
90
|
+
"""Write a Promptfoo configuration for ``tests`` into ``directory``."""
|
|
91
|
+
directory.mkdir(parents=True, exist_ok=True)
|
|
92
|
+
config = build_config(tests, target, project_root=project_root, config_dir=directory)
|
|
93
|
+
path = directory / CONFIG_FILENAME
|
|
94
|
+
path.write_text(
|
|
95
|
+
yaml.safe_dump(config, sort_keys=False, default_flow_style=False, allow_unicode=True),
|
|
96
|
+
encoding="utf-8",
|
|
97
|
+
)
|
|
98
|
+
return path
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def execute(
|
|
102
|
+
tests: list[RegressionTest],
|
|
103
|
+
target: Target,
|
|
104
|
+
*,
|
|
105
|
+
directory: Path,
|
|
106
|
+
command: list[str],
|
|
107
|
+
timeout_seconds: int,
|
|
108
|
+
working_directory: Path,
|
|
109
|
+
redactor: Redactor | None = None,
|
|
110
|
+
repetitions: int = 1,
|
|
111
|
+
) -> RunOutcome:
|
|
112
|
+
"""Run the suite against ``target`` and import what came back."""
|
|
113
|
+
missing = [name for name, present in referenced_environment(target).items() if not present]
|
|
114
|
+
if missing:
|
|
115
|
+
raise CommandError(
|
|
116
|
+
f"Target {target.target_id!r} needs environment variables that are not "
|
|
117
|
+
f"set: {', '.join(sorted(missing))}.",
|
|
118
|
+
hint="Export them, or put them in a .env file that is not committed.",
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
config_path = write_suite(tests, target, directory, project_root=working_directory)
|
|
122
|
+
results_path = directory / RESULTS_FILENAME
|
|
123
|
+
|
|
124
|
+
run = EvaluationRun(
|
|
125
|
+
run_id=uuid.uuid4().hex,
|
|
126
|
+
target_id=target.target_id,
|
|
127
|
+
suite_hash=suite_hash([test.test_id for test in tests]),
|
|
128
|
+
tests=len(tests),
|
|
129
|
+
repetitions=repetitions,
|
|
130
|
+
environment=_environment(),
|
|
131
|
+
output_dir=str(directory),
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
argv = [
|
|
135
|
+
*command,
|
|
136
|
+
"eval",
|
|
137
|
+
"--config",
|
|
138
|
+
str(config_path),
|
|
139
|
+
"--output",
|
|
140
|
+
str(results_path),
|
|
141
|
+
# Caching would defeat the purpose: repeating a call and getting the
|
|
142
|
+
# cached answer back measures the cache, not the agent.
|
|
143
|
+
"--no-cache",
|
|
144
|
+
]
|
|
145
|
+
if repetitions > 1:
|
|
146
|
+
argv += ["--repeat", str(repetitions)]
|
|
147
|
+
try:
|
|
148
|
+
# shell=False is the default and is relied upon: every element here can
|
|
149
|
+
# contain text that came out of a recorded trace.
|
|
150
|
+
completed = subprocess.run(
|
|
151
|
+
argv,
|
|
152
|
+
cwd=working_directory,
|
|
153
|
+
capture_output=True,
|
|
154
|
+
text=True,
|
|
155
|
+
timeout=timeout_seconds,
|
|
156
|
+
check=False,
|
|
157
|
+
shell=False,
|
|
158
|
+
)
|
|
159
|
+
except FileNotFoundError as exc:
|
|
160
|
+
raise CommandError(
|
|
161
|
+
f"Could not run {command[0]!r}: {exc}.",
|
|
162
|
+
hint="Promptfoo runs on Node. Install Node.js, or set runner.command.",
|
|
163
|
+
) from exc
|
|
164
|
+
except subprocess.TimeoutExpired as exc:
|
|
165
|
+
run.status = RunStatus.FAILED
|
|
166
|
+
run.finished_at = datetime.now(UTC)
|
|
167
|
+
raise CommandError(
|
|
168
|
+
f"The runner did not finish within {timeout_seconds}s.",
|
|
169
|
+
hint="Raise runner.timeout_seconds, or run fewer tests with --limit.",
|
|
170
|
+
) from exc
|
|
171
|
+
|
|
172
|
+
messages: list[str] = []
|
|
173
|
+
if not results_path.is_file():
|
|
174
|
+
raise CommandError(
|
|
175
|
+
"The runner produced no results file.",
|
|
176
|
+
hint=_tail(completed.stderr or completed.stdout),
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
results = import_results(results_path, redactor=redactor or Redactor())
|
|
180
|
+
run.finished_at = datetime.now(UTC)
|
|
181
|
+
run.runner = _runner_version(results_path)
|
|
182
|
+
# A non-zero exit is expected when tests fail; it only matters when nothing
|
|
183
|
+
# came back at all, which the missing-file check above already covers.
|
|
184
|
+
if completed.returncode != 0 and not results:
|
|
185
|
+
run.status = RunStatus.FAILED
|
|
186
|
+
messages.append(_tail(completed.stderr or completed.stdout))
|
|
187
|
+
|
|
188
|
+
return RunOutcome(run=run, results=results, messages=messages)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def import_results(path: Path, *, redactor: Redactor | None = None) -> list[CaseResult]:
|
|
192
|
+
"""Read a Promptfoo results file into per-test outcomes."""
|
|
193
|
+
redactor = redactor or Redactor()
|
|
194
|
+
try:
|
|
195
|
+
payload: Any = json.loads(path.read_text(encoding="utf-8"))
|
|
196
|
+
except (OSError, json.JSONDecodeError) as exc:
|
|
197
|
+
raise CommandError(f"Could not read the runner's results at {path}: {exc}") from exc
|
|
198
|
+
|
|
199
|
+
records = (payload.get("results") or {}).get("results") or []
|
|
200
|
+
|
|
201
|
+
# The runner returns one row per execution, in order, with repeated runs of
|
|
202
|
+
# the same case sharing a test ID. Numbering them here is what lets a case
|
|
203
|
+
# be judged on all of its attempts rather than its last one.
|
|
204
|
+
seen: dict[str, int] = {}
|
|
205
|
+
results: list[CaseResult] = []
|
|
206
|
+
for record in records:
|
|
207
|
+
test_id = _test_id(record)
|
|
208
|
+
if not test_id:
|
|
209
|
+
continue
|
|
210
|
+
repetition = seen.get(test_id, 0)
|
|
211
|
+
seen[test_id] = repetition + 1
|
|
212
|
+
results.append(_result(record, redactor, repetition))
|
|
213
|
+
return results
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def _result(record: dict[str, Any], redactor: Redactor, repetition: int = 0) -> CaseResult:
|
|
217
|
+
test_id = _test_id(record)
|
|
218
|
+
failure_reason = record.get("failureReason") or 0
|
|
219
|
+
error = record.get("error") or None
|
|
220
|
+
|
|
221
|
+
if record.get("success"):
|
|
222
|
+
outcome, error_kind = Outcome.PASS, None
|
|
223
|
+
elif failure_reason == _EXECUTION_ERROR:
|
|
224
|
+
outcome = Outcome.ERROR
|
|
225
|
+
error_kind = (
|
|
226
|
+
ErrorKind.TIMEOUT
|
|
227
|
+
if any(marker in (error or "").lower() for marker in _TIMEOUT_MARKERS)
|
|
228
|
+
else ErrorKind.EXECUTION_ERROR
|
|
229
|
+
)
|
|
230
|
+
elif failure_reason == _ASSERTION_FAILURE:
|
|
231
|
+
outcome, error_kind = Outcome.FAIL, None
|
|
232
|
+
else:
|
|
233
|
+
# A failure Promptfoo did not classify is not assumed to be the agent's
|
|
234
|
+
# fault; it is reported as an error so it cannot masquerade as one.
|
|
235
|
+
outcome = Outcome.ERROR
|
|
236
|
+
error_kind = ErrorKind.EXECUTION_ERROR
|
|
237
|
+
|
|
238
|
+
summary = RedactionSummary()
|
|
239
|
+
return CaseResult(
|
|
240
|
+
test_id=test_id,
|
|
241
|
+
outcome=outcome,
|
|
242
|
+
error_kind=error_kind,
|
|
243
|
+
error=redactor.redact_text(error, summary) if error else None,
|
|
244
|
+
latency_ms=record.get("latencyMs"),
|
|
245
|
+
observation=redactor.redact_text(_observation(record), summary) or None,
|
|
246
|
+
failed_assertions=[
|
|
247
|
+
redactor.redact_text(reason, summary) for reason in _failed_assertions(record)
|
|
248
|
+
],
|
|
249
|
+
repetition=repetition,
|
|
250
|
+
)
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def _test_id(record: dict[str, Any]) -> str:
|
|
254
|
+
case = record.get("testCase") or {}
|
|
255
|
+
metadata = case.get("metadata") or {}
|
|
256
|
+
return str(metadata.get("test_id") or case.get("description") or "")
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _observation(record: dict[str, Any]) -> str:
|
|
260
|
+
response = record.get("response") or {}
|
|
261
|
+
output = response.get("output")
|
|
262
|
+
if output is None:
|
|
263
|
+
return ""
|
|
264
|
+
if isinstance(output, str):
|
|
265
|
+
return output
|
|
266
|
+
return json.dumps(output, sort_keys=True, default=str)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _failed_assertions(record: dict[str, Any]) -> list[str]:
|
|
270
|
+
grading = record.get("gradingResult") or {}
|
|
271
|
+
components = grading.get("componentResults") or []
|
|
272
|
+
reasons = [
|
|
273
|
+
str(component.get("reason", "")).strip()
|
|
274
|
+
for component in components
|
|
275
|
+
if not component.get("pass", True)
|
|
276
|
+
]
|
|
277
|
+
if reasons:
|
|
278
|
+
return reasons
|
|
279
|
+
reason = str(grading.get("reason", "")).strip()
|
|
280
|
+
return [reason] if reason and not grading.get("pass", True) else []
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _runner_version(results_path: Path) -> str | None:
|
|
284
|
+
try:
|
|
285
|
+
payload = json.loads(results_path.read_text(encoding="utf-8"))
|
|
286
|
+
except (OSError, json.JSONDecodeError): # pragma: no cover - already parsed once
|
|
287
|
+
return None
|
|
288
|
+
version = (payload.get("results") or {}).get("version")
|
|
289
|
+
return f"promptfoo:{version}" if version else None
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def _environment() -> dict[str, Any]:
|
|
293
|
+
"""What the run happened on, so a surprising result can be placed."""
|
|
294
|
+
return {
|
|
295
|
+
"python": sys.version.split()[0],
|
|
296
|
+
"platform": platform.platform(),
|
|
297
|
+
"machine": platform.machine(),
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _tail(text: str, lines: int = 12) -> str:
|
|
302
|
+
return "\n".join((text or "").strip().splitlines()[-lines:])
|