evalkeep 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. evalkeep/__init__.py +12 -0
  2. evalkeep/__main__.py +6 -0
  3. evalkeep/adapters/__init__.py +45 -0
  4. evalkeep/adapters/base.py +92 -0
  5. evalkeep/adapters/jsonl.py +164 -0
  6. evalkeep/adapters/langsmith.py +436 -0
  7. evalkeep/adapters/otlp.py +442 -0
  8. evalkeep/adapters/semconv.py +208 -0
  9. evalkeep/analysis.py +174 -0
  10. evalkeep/analysis_run.py +160 -0
  11. evalkeep/analyzers/__init__.py +52 -0
  12. evalkeep/analyzers/anthropic.py +145 -0
  13. evalkeep/analyzers/stub.py +34 -0
  14. evalkeep/cache.py +122 -0
  15. evalkeep/cli.py +1933 -0
  16. evalkeep/clustering.py +383 -0
  17. evalkeep/clusters.py +101 -0
  18. evalkeep/commands/__init__.py +1 -0
  19. evalkeep/commands/analyze_cmd.py +100 -0
  20. evalkeep/commands/compare_cmd.py +169 -0
  21. evalkeep/commands/dataset_cmd.py +182 -0
  22. evalkeep/commands/detect_cmd.py +154 -0
  23. evalkeep/commands/discover_cmd.py +274 -0
  24. evalkeep/commands/ingest_cmd.py +50 -0
  25. evalkeep/commands/init_cmd.py +151 -0
  26. evalkeep/commands/pipeline_cmd.py +156 -0
  27. evalkeep/commands/review_cmd.py +141 -0
  28. evalkeep/commands/run_cmd.py +131 -0
  29. evalkeep/commands/target_cmd.py +109 -0
  30. evalkeep/commands/trace_cmd.py +58 -0
  31. evalkeep/comparison.py +432 -0
  32. evalkeep/config.py +209 -0
  33. evalkeep/detection.py +94 -0
  34. evalkeep/detectors.py +182 -0
  35. evalkeep/discovery.py +208 -0
  36. evalkeep/embeddings/__init__.py +31 -0
  37. evalkeep/embeddings/base.py +32 -0
  38. evalkeep/embeddings/hashing.py +98 -0
  39. evalkeep/errors.py +42 -0
  40. evalkeep/examples/__init__.py +37 -0
  41. evalkeep/examples/langsmith/runs.jsonl +18 -0
  42. evalkeep/examples/opentelemetry/spans.json +898 -0
  43. evalkeep/examples/refund-agent/agents/baseline.py +66 -0
  44. evalkeep/examples/refund-agent/agents/candidate.py +66 -0
  45. evalkeep/examples/refund-agent/traces.jsonl +5 -0
  46. evalkeep/examples/tau-bench/prepare.py +230 -0
  47. evalkeep/exporters/__init__.py +45 -0
  48. evalkeep/exporters/generic.py +31 -0
  49. evalkeep/exporters/promptfoo.py +219 -0
  50. evalkeep/failures.py +95 -0
  51. evalkeep/generation.py +303 -0
  52. evalkeep/hashing.py +56 -0
  53. evalkeep/ingest.py +257 -0
  54. evalkeep/prompts.py +127 -0
  55. evalkeep/pseudonyms.py +82 -0
  56. evalkeep/py.typed +0 -0
  57. evalkeep/redaction.py +333 -0
  58. evalkeep/regression.py +409 -0
  59. evalkeep/review.py +309 -0
  60. evalkeep/runner.py +302 -0
  61. evalkeep/runs.py +185 -0
  62. evalkeep/storage/__init__.py +37 -0
  63. evalkeep/storage/clusters.py +163 -0
  64. evalkeep/storage/failures.py +254 -0
  65. evalkeep/storage/migrations.py +370 -0
  66. evalkeep/storage/regression.py +136 -0
  67. evalkeep/storage/runs.py +223 -0
  68. evalkeep/storage/store.py +429 -0
  69. evalkeep/targets.py +205 -0
  70. evalkeep/trace.py +238 -0
  71. evalkeep-0.1.0.dist-info/METADATA +221 -0
  72. evalkeep-0.1.0.dist-info/RECORD +75 -0
  73. evalkeep-0.1.0.dist-info/WHEEL +4 -0
  74. evalkeep-0.1.0.dist-info/entry_points.txt +3 -0
  75. evalkeep-0.1.0.dist-info/licenses/LICENSE +202 -0
evalkeep/review.py ADDED
@@ -0,0 +1,309 @@
1
+ """Human review: the gate between a generated draft and a committed test.
2
+
3
+ Everything here is a pure function over a test. The interactive terminal loop is
4
+ a thin shell in the CLI that calls these, which is what keeps the guide's
5
+ "non-interactive CI review format" possible later: a different front end -- a
6
+ web form, a PR check, a batch file of decisions -- needs no new logic, only a
7
+ different way of collecting the same decisions.
8
+
9
+ Two rules the review gate enforces rather than suggests:
10
+
11
+ * **A contradictory test cannot be approved.** It would fail on a correct agent
12
+ too, reporting a regression that is really a bug in the suite.
13
+ * **Editing cannot change what a test *is*.** A reviewer edits the input and the
14
+ expectations -- the parts that encode intent. The test ID, provenance and
15
+ fixtures are facts about where the test came from, and letting a review rewrite
16
+ them would make the audit trail describe something that never happened.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ from dataclasses import dataclass, field
22
+ from datetime import UTC, datetime
23
+ from enum import StrEnum
24
+ from typing import Any
25
+
26
+ import yaml
27
+
28
+ from evalkeep.generation import NO_POSITIVE_EXPECTATION
29
+ from evalkeep.regression import (
30
+ CaseInput,
31
+ Expectation,
32
+ ExpectationType,
33
+ RegressionTest,
34
+ ReviewStatus,
35
+ find_contradictions,
36
+ validate_expectation,
37
+ )
38
+
39
+ #: The only fields a reviewer may change.
40
+ EDITABLE_FIELDS = ("input", "expectations")
41
+
42
+ EDIT_HELP = """\
43
+ # Editing {test_id}
44
+ #
45
+ # Change the input and the expectations. Everything else -- the test ID, the
46
+ # provenance, the recorded fixtures -- describes where this test came from and
47
+ # is not editable.
48
+ #
49
+ # Expectation types:
50
+ # output_contains value: text the answer must contain
51
+ # output_not_contains value: text the answer must not contain
52
+ # output_matches value: a regular expression the answer must match
53
+ # tool_called tool: a tool that must be called
54
+ # tool_not_called tool: a tool that must never be called
55
+ # tool_argument_equals tool, path, value
56
+ # tool_argument_not_equals tool, path, value
57
+ # max_tool_calls value: a whole number; tool: optional
58
+ # human_rubric value: what should happen, judged by a model at run time
59
+ #
60
+ # Delete every expectation to abandon the edit.
61
+ """
62
+
63
+
64
+ class ReviewDecision(StrEnum):
65
+ APPROVE = "approve"
66
+ EDIT = "edit"
67
+ REJECT = "reject"
68
+ SKIP = "skip"
69
+
70
+
71
+ @dataclass
72
+ class ReviewOutcome:
73
+ """What one review session did."""
74
+
75
+ reviewed: int = 0
76
+ approved: int = 0
77
+ rejected: int = 0
78
+ edited: int = 0
79
+ skipped: int = 0
80
+ remaining: int = 0
81
+
82
+ @property
83
+ def changed(self) -> bool:
84
+ return bool(self.approved or self.rejected or self.edited)
85
+
86
+
87
+ @dataclass
88
+ class EditResult:
89
+ """A parsed edit, or the reasons it could not be applied."""
90
+
91
+ test: RegressionTest | None = None
92
+ errors: list[str] = field(default_factory=list)
93
+
94
+ @property
95
+ def ok(self) -> bool:
96
+ return self.test is not None and not self.errors
97
+
98
+
99
+ class ReviewError(Exception):
100
+ """A decision that cannot be recorded as asked."""
101
+
102
+
103
+ def recompute_warnings(test: RegressionTest) -> list[str]:
104
+ """What still needs a reviewer's attention, for the test as it stands now.
105
+
106
+ Recomputed rather than carried forward: warnings describe current content,
107
+ and a note about how the draft was generated stops being true the moment a
108
+ person edits it.
109
+ """
110
+ warnings: list[str] = []
111
+ for expectation in test.expectations:
112
+ problem = validate_expectation(expectation)
113
+ if problem is not None:
114
+ warnings.append(f"Invalid expectation ({expectation.describe()}): {problem}")
115
+ warnings.extend(
116
+ f"Contradictory expectations: {contradiction.describe()}"
117
+ for contradiction in test.contradictions
118
+ )
119
+ if not test.has_positive_expectation:
120
+ warnings.append(NO_POSITIVE_EXPECTATION)
121
+ return warnings
122
+
123
+
124
+ def blocking_problems(test: RegressionTest) -> list[str]:
125
+ """Reasons a test must not be approved as it stands."""
126
+ problems = [
127
+ f"Invalid expectation ({expectation.describe()}): {problem}"
128
+ for expectation in test.expectations
129
+ if (problem := validate_expectation(expectation)) is not None
130
+ ]
131
+ problems.extend(
132
+ f"Contradictory expectations: {contradiction.describe()}"
133
+ for contradiction in test.contradictions
134
+ )
135
+ if not test.expectations:
136
+ problems.append("A test with no expectations checks nothing.")
137
+ return problems
138
+
139
+
140
+ def approve(test: RegressionTest, *, reviewer: str, reason: str | None = None) -> RegressionTest:
141
+ """Approve a test, refusing if it could never pass on a correct agent."""
142
+ problems = blocking_problems(test)
143
+ if problems:
144
+ raise ReviewError(
145
+ "This test cannot be approved as it stands:\n - " + "\n - ".join(problems)
146
+ )
147
+ return _record(test, ReviewStatus.APPROVED, reviewer=reviewer, reason=reason)
148
+
149
+
150
+ def reject(test: RegressionTest, *, reviewer: str, reason: str | None = None) -> RegressionTest:
151
+ """Reject a test. The record is kept, not deleted: a rejection is evidence."""
152
+ return _record(test, ReviewStatus.REJECTED, reviewer=reviewer, reason=reason)
153
+
154
+
155
+ def _record(
156
+ test: RegressionTest, status: ReviewStatus, *, reviewer: str, reason: str | None
157
+ ) -> RegressionTest:
158
+ test.status = status
159
+ test.reviewer = reviewer
160
+ test.review_reason = reason
161
+ test.reviewed_at = datetime.now(UTC)
162
+ test.updated_at = test.reviewed_at
163
+ test.warnings = recompute_warnings(test)
164
+ return test
165
+
166
+
167
+ def render_editable(test: RegressionTest) -> str:
168
+ """The YAML a reviewer edits, with the guidance they need above it."""
169
+ body = {
170
+ "input": {
171
+ key: value for key, value in test.input.to_dict().items() if value not in (None, [])
172
+ },
173
+ "expectations": [expectation.to_dict() for expectation in test.expectations],
174
+ }
175
+ header = EDIT_HELP.format(test_id=test.test_id)
176
+ return header + yaml.safe_dump(body, sort_keys=False, default_flow_style=False)
177
+
178
+
179
+ def apply_edits(test: RegressionTest, text: str, *, editor: str) -> EditResult:
180
+ """Parse an edited document and return an updated copy, or the errors.
181
+
182
+ Nothing is written until the result is valid: an unparseable or
183
+ self-contradictory edit leaves the stored draft exactly as it was.
184
+ """
185
+ try:
186
+ # safe_load, never load: this document is arbitrary text from an editor,
187
+ # and full YAML can construct objects.
188
+ raw: Any = yaml.safe_load(text)
189
+ except yaml.YAMLError as exc:
190
+ return EditResult(errors=[f"Could not parse YAML: {exc}"])
191
+
192
+ if raw is None:
193
+ return EditResult(errors=["The document is empty."])
194
+ if not isinstance(raw, dict):
195
+ return EditResult(errors=[f"Expected a mapping, got {type(raw).__name__}."])
196
+
197
+ unknown = set(raw) - set(EDITABLE_FIELDS)
198
+ if unknown:
199
+ return EditResult(
200
+ errors=[
201
+ f"Only {' and '.join(EDITABLE_FIELDS)} can be edited; "
202
+ f"remove: {', '.join(sorted(unknown))}."
203
+ ]
204
+ )
205
+
206
+ errors: list[str] = []
207
+ case_input, input_errors = _parse_input(raw.get("input"))
208
+ errors.extend(input_errors)
209
+ expectations, expectation_errors = _parse_expectations(raw.get("expectations"))
210
+ errors.extend(expectation_errors)
211
+
212
+ if errors:
213
+ return EditResult(errors=errors)
214
+
215
+ updated = _copy_with_edits(test, case_input, expectations, editor=editor)
216
+ contradictions = [
217
+ f"Contradictory expectations: {contradiction.describe()}"
218
+ for contradiction in find_contradictions(expectations)
219
+ ]
220
+ if contradictions:
221
+ return EditResult(errors=contradictions)
222
+ return EditResult(test=updated)
223
+
224
+
225
+ def _copy_with_edits(
226
+ test: RegressionTest,
227
+ case_input: CaseInput,
228
+ expectations: list[Expectation],
229
+ *,
230
+ editor: str,
231
+ ) -> RegressionTest:
232
+ now = datetime.now(UTC)
233
+ updated = RegressionTest(
234
+ test_id=test.test_id,
235
+ failure_id=test.failure_id,
236
+ input=case_input,
237
+ provenance=test.provenance,
238
+ status=test.status,
239
+ fixtures=test.fixtures,
240
+ expectations=expectations,
241
+ reviewer=test.reviewer,
242
+ review_reason=test.review_reason,
243
+ reviewed_at=test.reviewed_at,
244
+ edited=True,
245
+ edited_by=editor,
246
+ created_at=test.created_at,
247
+ updated_at=now,
248
+ )
249
+ updated.warnings = recompute_warnings(updated)
250
+ return updated
251
+
252
+
253
+ def _parse_input(raw: Any) -> tuple[CaseInput, list[str]]:
254
+ if raw is None:
255
+ return CaseInput(), ["input is required."]
256
+ if not isinstance(raw, dict):
257
+ return CaseInput(), [f"input must be a mapping, got {type(raw).__name__}."]
258
+
259
+ text = raw.get("text")
260
+ messages = raw.get("messages") or []
261
+ if text is not None and not isinstance(text, str):
262
+ return CaseInput(), ["input.text must be text."]
263
+ if not isinstance(messages, list):
264
+ return CaseInput(), ["input.messages must be a list."]
265
+
266
+ parsed: list[dict[str, str]] = []
267
+ for index, message in enumerate(messages):
268
+ if not isinstance(message, dict) or "role" not in message or "content" not in message:
269
+ return CaseInput(), [f"input.messages[{index}] needs a role and content."]
270
+ parsed.append({"role": str(message["role"]), "content": str(message["content"])})
271
+
272
+ if not (text or "").strip() and not parsed:
273
+ return CaseInput(), ["input needs non-empty text or at least one message."]
274
+ return CaseInput(text=text, messages=parsed), []
275
+
276
+
277
+ def _parse_expectations(raw: Any) -> tuple[list[Expectation], list[str]]:
278
+ if raw is None or raw == []:
279
+ return [], ["A test needs at least one expectation."]
280
+ if not isinstance(raw, list):
281
+ return [], [f"expectations must be a list, got {type(raw).__name__}."]
282
+
283
+ expectations: list[Expectation] = []
284
+ errors: list[str] = []
285
+ for index, item in enumerate(raw):
286
+ if not isinstance(item, dict):
287
+ errors.append(f"expectations[{index}] must be a mapping.")
288
+ continue
289
+ kind = item.get("type")
290
+ try:
291
+ expectation_type = ExpectationType(str(kind))
292
+ except ValueError:
293
+ known = ", ".join(member.value for member in ExpectationType)
294
+ errors.append(f"expectations[{index}]: unknown type {kind!r}. Known types: {known}.")
295
+ continue
296
+
297
+ expectation = Expectation(
298
+ type=expectation_type,
299
+ value=item.get("value"),
300
+ tool=item.get("tool"),
301
+ path=item.get("path"),
302
+ )
303
+ problem = validate_expectation(expectation)
304
+ if problem is not None:
305
+ errors.append(f"expectations[{index}]: {problem}")
306
+ continue
307
+ expectations.append(expectation)
308
+
309
+ return expectations, errors
evalkeep/runner.py ADDED
@@ -0,0 +1,302 @@
1
+ """Delegating execution to Promptfoo, and reading the results back.
2
+
3
+ Two things this module is careful about:
4
+
5
+ * **The runner is invoked as an argument list, never through a shell.** Test
6
+ inputs, tool names and file paths all come from recorded traces. Building a
7
+ command string out of them would make a trace containing ``; rm -rf`` a
8
+ remote-code-execution bug, so ``subprocess.run`` is called with a list and
9
+ ``shell=False``, and nothing is ever passed through a shell.
10
+ * **A test that never ran is not a test that failed.** Promptfoo distinguishes
11
+ an assertion failure from a provider error, and so does the import: an error
12
+ is recorded as :class:`~evalkeep.runs.Outcome.ERROR`, never as a failure.
13
+ Letting a crashed provider look like a regression is precisely the wrong
14
+ answer for a tool whose job is deciding whether a release got worse.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import json
20
+ import platform
21
+ import subprocess
22
+ import sys
23
+ import uuid
24
+ from dataclasses import dataclass, field
25
+ from datetime import UTC, datetime
26
+ from pathlib import Path
27
+ from typing import Any
28
+
29
+ import yaml
30
+
31
+ from evalkeep.errors import CommandError
32
+ from evalkeep.exporters.promptfoo import build_config
33
+ from evalkeep.redaction import RedactionSummary, Redactor
34
+ from evalkeep.regression import RegressionTest
35
+ from evalkeep.runs import (
36
+ CaseResult,
37
+ CaseSummary,
38
+ ErrorKind,
39
+ EvaluationRun,
40
+ Outcome,
41
+ RunStatus,
42
+ suite_hash,
43
+ summarize,
44
+ )
45
+ from evalkeep.targets import Target, referenced_environment
46
+
47
+ CONFIG_FILENAME = "promptfooconfig.yaml"
48
+ RESULTS_FILENAME = "results.json"
49
+
50
+ #: Promptfoo's own failure taxonomy, which the import preserves rather than
51
+ #: flattening: 0 none, 1 assertion, 2 provider/execution error.
52
+ _ASSERTION_FAILURE = 1
53
+ _EXECUTION_ERROR = 2
54
+
55
+ _TIMEOUT_MARKERS = ("timeout", "timed out", "etimedout", "esockettimedout")
56
+
57
+
58
+ @dataclass
59
+ class RunOutcome:
60
+ run: EvaluationRun
61
+ results: list[CaseResult] = field(default_factory=list)
62
+ #: Anything the runner said that a person should see.
63
+ messages: list[str] = field(default_factory=list)
64
+
65
+ @property
66
+ def counts(self) -> dict[Outcome, int]:
67
+ """Per-execution counts. With repetitions these exceed the test count."""
68
+ tally: dict[Outcome, int] = {}
69
+ for result in self.results:
70
+ tally[result.outcome] = tally.get(result.outcome, 0) + 1
71
+ return tally
72
+
73
+ @property
74
+ def summaries(self) -> dict[str, CaseSummary]:
75
+ """Per-case verdicts, which is what a repeated run is actually for."""
76
+ return summarize(self.results)
77
+
78
+ @property
79
+ def flaky(self) -> list[CaseSummary]:
80
+ return [s for s in self.summaries.values() if s.flaky]
81
+
82
+
83
+ def write_suite(
84
+ tests: list[RegressionTest],
85
+ target: Target,
86
+ directory: Path,
87
+ *,
88
+ project_root: Path | None = None,
89
+ ) -> Path:
90
+ """Write a Promptfoo configuration for ``tests`` into ``directory``."""
91
+ directory.mkdir(parents=True, exist_ok=True)
92
+ config = build_config(tests, target, project_root=project_root, config_dir=directory)
93
+ path = directory / CONFIG_FILENAME
94
+ path.write_text(
95
+ yaml.safe_dump(config, sort_keys=False, default_flow_style=False, allow_unicode=True),
96
+ encoding="utf-8",
97
+ )
98
+ return path
99
+
100
+
101
+ def execute(
102
+ tests: list[RegressionTest],
103
+ target: Target,
104
+ *,
105
+ directory: Path,
106
+ command: list[str],
107
+ timeout_seconds: int,
108
+ working_directory: Path,
109
+ redactor: Redactor | None = None,
110
+ repetitions: int = 1,
111
+ ) -> RunOutcome:
112
+ """Run the suite against ``target`` and import what came back."""
113
+ missing = [name for name, present in referenced_environment(target).items() if not present]
114
+ if missing:
115
+ raise CommandError(
116
+ f"Target {target.target_id!r} needs environment variables that are not "
117
+ f"set: {', '.join(sorted(missing))}.",
118
+ hint="Export them, or put them in a .env file that is not committed.",
119
+ )
120
+
121
+ config_path = write_suite(tests, target, directory, project_root=working_directory)
122
+ results_path = directory / RESULTS_FILENAME
123
+
124
+ run = EvaluationRun(
125
+ run_id=uuid.uuid4().hex,
126
+ target_id=target.target_id,
127
+ suite_hash=suite_hash([test.test_id for test in tests]),
128
+ tests=len(tests),
129
+ repetitions=repetitions,
130
+ environment=_environment(),
131
+ output_dir=str(directory),
132
+ )
133
+
134
+ argv = [
135
+ *command,
136
+ "eval",
137
+ "--config",
138
+ str(config_path),
139
+ "--output",
140
+ str(results_path),
141
+ # Caching would defeat the purpose: repeating a call and getting the
142
+ # cached answer back measures the cache, not the agent.
143
+ "--no-cache",
144
+ ]
145
+ if repetitions > 1:
146
+ argv += ["--repeat", str(repetitions)]
147
+ try:
148
+ # shell=False is the default and is relied upon: every element here can
149
+ # contain text that came out of a recorded trace.
150
+ completed = subprocess.run(
151
+ argv,
152
+ cwd=working_directory,
153
+ capture_output=True,
154
+ text=True,
155
+ timeout=timeout_seconds,
156
+ check=False,
157
+ shell=False,
158
+ )
159
+ except FileNotFoundError as exc:
160
+ raise CommandError(
161
+ f"Could not run {command[0]!r}: {exc}.",
162
+ hint="Promptfoo runs on Node. Install Node.js, or set runner.command.",
163
+ ) from exc
164
+ except subprocess.TimeoutExpired as exc:
165
+ run.status = RunStatus.FAILED
166
+ run.finished_at = datetime.now(UTC)
167
+ raise CommandError(
168
+ f"The runner did not finish within {timeout_seconds}s.",
169
+ hint="Raise runner.timeout_seconds, or run fewer tests with --limit.",
170
+ ) from exc
171
+
172
+ messages: list[str] = []
173
+ if not results_path.is_file():
174
+ raise CommandError(
175
+ "The runner produced no results file.",
176
+ hint=_tail(completed.stderr or completed.stdout),
177
+ )
178
+
179
+ results = import_results(results_path, redactor=redactor or Redactor())
180
+ run.finished_at = datetime.now(UTC)
181
+ run.runner = _runner_version(results_path)
182
+ # A non-zero exit is expected when tests fail; it only matters when nothing
183
+ # came back at all, which the missing-file check above already covers.
184
+ if completed.returncode != 0 and not results:
185
+ run.status = RunStatus.FAILED
186
+ messages.append(_tail(completed.stderr or completed.stdout))
187
+
188
+ return RunOutcome(run=run, results=results, messages=messages)
189
+
190
+
191
+ def import_results(path: Path, *, redactor: Redactor | None = None) -> list[CaseResult]:
192
+ """Read a Promptfoo results file into per-test outcomes."""
193
+ redactor = redactor or Redactor()
194
+ try:
195
+ payload: Any = json.loads(path.read_text(encoding="utf-8"))
196
+ except (OSError, json.JSONDecodeError) as exc:
197
+ raise CommandError(f"Could not read the runner's results at {path}: {exc}") from exc
198
+
199
+ records = (payload.get("results") or {}).get("results") or []
200
+
201
+ # The runner returns one row per execution, in order, with repeated runs of
202
+ # the same case sharing a test ID. Numbering them here is what lets a case
203
+ # be judged on all of its attempts rather than its last one.
204
+ seen: dict[str, int] = {}
205
+ results: list[CaseResult] = []
206
+ for record in records:
207
+ test_id = _test_id(record)
208
+ if not test_id:
209
+ continue
210
+ repetition = seen.get(test_id, 0)
211
+ seen[test_id] = repetition + 1
212
+ results.append(_result(record, redactor, repetition))
213
+ return results
214
+
215
+
216
+ def _result(record: dict[str, Any], redactor: Redactor, repetition: int = 0) -> CaseResult:
217
+ test_id = _test_id(record)
218
+ failure_reason = record.get("failureReason") or 0
219
+ error = record.get("error") or None
220
+
221
+ if record.get("success"):
222
+ outcome, error_kind = Outcome.PASS, None
223
+ elif failure_reason == _EXECUTION_ERROR:
224
+ outcome = Outcome.ERROR
225
+ error_kind = (
226
+ ErrorKind.TIMEOUT
227
+ if any(marker in (error or "").lower() for marker in _TIMEOUT_MARKERS)
228
+ else ErrorKind.EXECUTION_ERROR
229
+ )
230
+ elif failure_reason == _ASSERTION_FAILURE:
231
+ outcome, error_kind = Outcome.FAIL, None
232
+ else:
233
+ # A failure Promptfoo did not classify is not assumed to be the agent's
234
+ # fault; it is reported as an error so it cannot masquerade as one.
235
+ outcome = Outcome.ERROR
236
+ error_kind = ErrorKind.EXECUTION_ERROR
237
+
238
+ summary = RedactionSummary()
239
+ return CaseResult(
240
+ test_id=test_id,
241
+ outcome=outcome,
242
+ error_kind=error_kind,
243
+ error=redactor.redact_text(error, summary) if error else None,
244
+ latency_ms=record.get("latencyMs"),
245
+ observation=redactor.redact_text(_observation(record), summary) or None,
246
+ failed_assertions=[
247
+ redactor.redact_text(reason, summary) for reason in _failed_assertions(record)
248
+ ],
249
+ repetition=repetition,
250
+ )
251
+
252
+
253
+ def _test_id(record: dict[str, Any]) -> str:
254
+ case = record.get("testCase") or {}
255
+ metadata = case.get("metadata") or {}
256
+ return str(metadata.get("test_id") or case.get("description") or "")
257
+
258
+
259
+ def _observation(record: dict[str, Any]) -> str:
260
+ response = record.get("response") or {}
261
+ output = response.get("output")
262
+ if output is None:
263
+ return ""
264
+ if isinstance(output, str):
265
+ return output
266
+ return json.dumps(output, sort_keys=True, default=str)
267
+
268
+
269
+ def _failed_assertions(record: dict[str, Any]) -> list[str]:
270
+ grading = record.get("gradingResult") or {}
271
+ components = grading.get("componentResults") or []
272
+ reasons = [
273
+ str(component.get("reason", "")).strip()
274
+ for component in components
275
+ if not component.get("pass", True)
276
+ ]
277
+ if reasons:
278
+ return reasons
279
+ reason = str(grading.get("reason", "")).strip()
280
+ return [reason] if reason and not grading.get("pass", True) else []
281
+
282
+
283
+ def _runner_version(results_path: Path) -> str | None:
284
+ try:
285
+ payload = json.loads(results_path.read_text(encoding="utf-8"))
286
+ except (OSError, json.JSONDecodeError): # pragma: no cover - already parsed once
287
+ return None
288
+ version = (payload.get("results") or {}).get("version")
289
+ return f"promptfoo:{version}" if version else None
290
+
291
+
292
+ def _environment() -> dict[str, Any]:
293
+ """What the run happened on, so a surprising result can be placed."""
294
+ return {
295
+ "python": sys.version.split()[0],
296
+ "platform": platform.platform(),
297
+ "machine": platform.machine(),
298
+ }
299
+
300
+
301
+ def _tail(text: str, lines: int = 12) -> str:
302
+ return "\n".join((text or "").strip().splitlines()[-lines:])