@arizeai/phoenix-client 6.10.1 → 6.11.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +62 -0
- package/dist/esm/__generated__/api/v1.d.ts +254 -2
- package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
- package/dist/esm/jest/index.d.ts +5 -0
- package/dist/esm/jest/index.d.ts.map +1 -0
- package/dist/esm/jest/index.js +49 -0
- package/dist/esm/jest/index.js.map +1 -0
- package/dist/esm/jest/reporter.d.ts +13 -0
- package/dist/esm/jest/reporter.d.ts.map +1 -0
- package/dist/esm/jest/reporter.js +19 -0
- package/dist/esm/jest/reporter.js.map +1 -0
- package/dist/esm/prompts/sdks/toAI.d.ts +2 -2
- package/dist/esm/prompts/sdks/toAI.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toAI.js.map +1 -1
- package/dist/esm/prompts/sdks/toAnthropic.d.ts +2 -2
- package/dist/esm/prompts/sdks/toAnthropic.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toAnthropic.js.map +1 -1
- package/dist/esm/prompts/sdks/toOpenAI.d.ts +2 -2
- package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
- package/dist/esm/prompts/sdks/toSDK.d.ts +8 -8
- package/dist/esm/prompts/sdks/toSDK.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toSDK.js.map +1 -1
- package/dist/esm/prompts/sdks/types.d.ts +2 -2
- package/dist/esm/prompts/sdks/types.d.ts.map +1 -1
- package/dist/esm/schemas/llm/anthropic/converters.d.ts +8 -8
- package/dist/esm/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
- package/dist/esm/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
- package/dist/esm/schemas/llm/constants.d.ts +3 -3
- package/dist/esm/schemas/llm/converters.d.ts +12 -12
- package/dist/esm/schemas/llm/openai/converters.d.ts +3 -3
- package/dist/esm/schemas/llm/schemas.d.ts +2 -2
- package/dist/esm/testing/acceptance.d.ts +20 -0
- package/dist/esm/testing/acceptance.d.ts.map +1 -0
- package/dist/esm/testing/acceptance.js +129 -0
- package/dist/esm/testing/acceptance.js.map +1 -0
- package/dist/esm/testing/define-api.d.ts +157 -0
- package/dist/esm/testing/define-api.d.ts.map +1 -0
- package/dist/esm/testing/define-api.js +78 -0
- package/dist/esm/testing/define-api.js.map +1 -0
- package/dist/esm/testing/helpers.d.ts +55 -0
- package/dist/esm/testing/helpers.d.ts.map +1 -0
- package/dist/esm/testing/helpers.js +179 -0
- package/dist/esm/testing/helpers.js.map +1 -0
- package/dist/esm/testing/phoenix-test-tracking.d.ts +68 -0
- package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -0
- package/dist/esm/testing/phoenix-test-tracking.js +521 -0
- package/dist/esm/testing/phoenix-test-tracking.js.map +1 -0
- package/dist/esm/testing/report-artifacts.d.ts +45 -0
- package/dist/esm/testing/report-artifacts.d.ts.map +1 -0
- package/dist/esm/testing/report-artifacts.js +218 -0
- package/dist/esm/testing/report-artifacts.js.map +1 -0
- package/dist/esm/testing/report-run.d.ts +22 -0
- package/dist/esm/testing/report-run.d.ts.map +1 -0
- package/dist/esm/testing/report-run.js +41 -0
- package/dist/esm/testing/report-run.js.map +1 -0
- package/dist/esm/testing/reporter-format.d.ts +83 -0
- package/dist/esm/testing/reporter-format.d.ts.map +1 -0
- package/dist/esm/testing/reporter-format.js +852 -0
- package/dist/esm/testing/reporter-format.js.map +1 -0
- package/dist/esm/testing/runner.d.ts +31 -0
- package/dist/esm/testing/runner.d.ts.map +1 -0
- package/dist/esm/testing/runner.js +238 -0
- package/dist/esm/testing/runner.js.map +1 -0
- package/dist/esm/testing/state.d.ts +138 -0
- package/dist/esm/testing/state.d.ts.map +1 -0
- package/dist/esm/testing/state.js +31 -0
- package/dist/esm/testing/state.js.map +1 -0
- package/dist/esm/testing/types.d.ts +319 -0
- package/dist/esm/testing/types.d.ts.map +1 -0
- package/dist/esm/testing/types.js +9 -0
- package/dist/esm/testing/types.js.map +1 -0
- package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
- package/dist/esm/utils/channel.d.ts +7 -7
- package/dist/esm/utils/channel.d.ts.map +1 -1
- package/dist/esm/utils/channel.js +1 -1
- package/dist/esm/utils/channel.js.map +1 -1
- package/dist/esm/utils/formatPromptMessages.d.ts.map +1 -1
- package/dist/esm/utils/getPromptBySelector.d.ts.map +1 -1
- package/dist/esm/utils/promisifyResult.d.ts +1 -1
- package/dist/esm/utils/promisifyResult.d.ts.map +1 -1
- package/dist/esm/utils/promisifyResult.js.map +1 -1
- package/dist/esm/utils/schemaMatches.d.ts +5 -5
- package/dist/esm/utils/schemaMatches.d.ts.map +1 -1
- package/dist/esm/utils/schemaMatches.js.map +1 -1
- package/dist/esm/vitest/index.d.ts +5 -0
- package/dist/esm/vitest/index.d.ts.map +1 -0
- package/dist/esm/vitest/index.js +15 -0
- package/dist/esm/vitest/index.js.map +1 -0
- package/dist/esm/vitest/reporter.d.ts +19 -0
- package/dist/esm/vitest/reporter.d.ts.map +1 -0
- package/dist/esm/vitest/reporter.js +27 -0
- package/dist/esm/vitest/reporter.js.map +1 -0
- package/dist/src/__generated__/api/v1.d.ts +254 -2
- package/dist/src/__generated__/api/v1.d.ts.map +1 -1
- package/dist/src/jest/index.d.ts +5 -0
- package/dist/src/jest/index.d.ts.map +1 -0
- package/dist/src/jest/index.js +58 -0
- package/dist/src/jest/index.js.map +1 -0
- package/dist/src/jest/reporter.d.ts +13 -0
- package/dist/src/jest/reporter.d.ts.map +1 -0
- package/dist/src/jest/reporter.js +23 -0
- package/dist/src/jest/reporter.js.map +1 -0
- package/dist/src/prompts/sdks/toAI.d.ts +2 -2
- package/dist/src/prompts/sdks/toAI.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toAI.js.map +1 -1
- package/dist/src/prompts/sdks/toAnthropic.d.ts +2 -2
- package/dist/src/prompts/sdks/toAnthropic.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toAnthropic.js.map +1 -1
- package/dist/src/prompts/sdks/toOpenAI.d.ts +2 -2
- package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
- package/dist/src/prompts/sdks/toSDK.d.ts +8 -8
- package/dist/src/prompts/sdks/toSDK.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toSDK.js.map +1 -1
- package/dist/src/prompts/sdks/types.d.ts +2 -2
- package/dist/src/prompts/sdks/types.d.ts.map +1 -1
- package/dist/src/schemas/llm/anthropic/converters.d.ts +8 -8
- package/dist/src/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
- package/dist/src/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
- package/dist/src/schemas/llm/constants.d.ts +3 -3
- package/dist/src/schemas/llm/converters.d.ts +12 -12
- package/dist/src/schemas/llm/openai/converters.d.ts +3 -3
- package/dist/src/schemas/llm/schemas.d.ts +2 -2
- package/dist/src/testing/acceptance.d.ts +20 -0
- package/dist/src/testing/acceptance.d.ts.map +1 -0
- package/dist/src/testing/acceptance.js +114 -0
- package/dist/src/testing/acceptance.js.map +1 -0
- package/dist/src/testing/define-api.d.ts +157 -0
- package/dist/src/testing/define-api.d.ts.map +1 -0
- package/dist/src/testing/define-api.js +81 -0
- package/dist/src/testing/define-api.js.map +1 -0
- package/dist/src/testing/helpers.d.ts +55 -0
- package/dist/src/testing/helpers.d.ts.map +1 -0
- package/dist/src/testing/helpers.js +182 -0
- package/dist/src/testing/helpers.js.map +1 -0
- package/dist/src/testing/phoenix-test-tracking.d.ts +68 -0
- package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -0
- package/dist/src/testing/phoenix-test-tracking.js +530 -0
- package/dist/src/testing/phoenix-test-tracking.js.map +1 -0
- package/dist/src/testing/report-artifacts.d.ts +45 -0
- package/dist/src/testing/report-artifacts.d.ts.map +1 -0
- package/dist/src/testing/report-artifacts.js +225 -0
- package/dist/src/testing/report-artifacts.js.map +1 -0
- package/dist/src/testing/report-run.d.ts +22 -0
- package/dist/src/testing/report-run.d.ts.map +1 -0
- package/dist/src/testing/report-run.js +47 -0
- package/dist/src/testing/report-run.js.map +1 -0
- package/dist/src/testing/reporter-format.d.ts +83 -0
- package/dist/src/testing/reporter-format.d.ts.map +1 -0
- package/dist/src/testing/reporter-format.js +870 -0
- package/dist/src/testing/reporter-format.js.map +1 -0
- package/dist/src/testing/runner.d.ts +31 -0
- package/dist/src/testing/runner.d.ts.map +1 -0
- package/dist/src/testing/runner.js +258 -0
- package/dist/src/testing/runner.js.map +1 -0
- package/dist/src/testing/state.d.ts +138 -0
- package/dist/src/testing/state.d.ts.map +1 -0
- package/dist/src/testing/state.js +38 -0
- package/dist/src/testing/state.js.map +1 -0
- package/dist/src/testing/types.d.ts +319 -0
- package/dist/src/testing/types.d.ts.map +1 -0
- package/dist/src/testing/types.js +13 -0
- package/dist/src/testing/types.js.map +1 -0
- package/dist/src/utils/channel.d.ts +7 -7
- package/dist/src/utils/channel.d.ts.map +1 -1
- package/dist/src/utils/channel.js +1 -1
- package/dist/src/utils/channel.js.map +1 -1
- package/dist/src/utils/formatPromptMessages.d.ts.map +1 -1
- package/dist/src/utils/getPromptBySelector.d.ts.map +1 -1
- package/dist/src/utils/promisifyResult.d.ts +1 -1
- package/dist/src/utils/promisifyResult.d.ts.map +1 -1
- package/dist/src/utils/promisifyResult.js.map +1 -1
- package/dist/src/utils/schemaMatches.d.ts +5 -5
- package/dist/src/utils/schemaMatches.d.ts.map +1 -1
- package/dist/src/utils/schemaMatches.js.map +1 -1
- package/dist/src/vitest/index.d.ts +5 -0
- package/dist/src/vitest/index.d.ts.map +1 -0
- package/dist/src/vitest/index.js +23 -0
- package/dist/src/vitest/index.js.map +1 -0
- package/dist/src/vitest/reporter.d.ts +19 -0
- package/dist/src/vitest/reporter.d.ts.map +1 -0
- package/dist/src/vitest/reporter.js +34 -0
- package/dist/src/vitest/reporter.js.map +1 -0
- package/dist/tsconfig.tsbuildinfo +1 -1
- package/docs/ci-evals-annotations.mdx +190 -0
- package/docs/ci-evals-jest.mdx +78 -0
- package/docs/ci-evals-vitest.mdx +240 -0
- package/docs/ci-evals.mdx +263 -0
- package/docs/overview.mdx +9 -1
- package/package.json +49 -17
- package/src/__generated__/api/v1.ts +254 -2
- package/src/jest/index.ts +124 -0
- package/src/jest/reporter.ts +22 -0
- package/src/prompts/sdks/toAI.ts +4 -3
- package/src/prompts/sdks/toAnthropic.ts +4 -3
- package/src/prompts/sdks/toOpenAI.ts +4 -3
- package/src/prompts/sdks/toSDK.ts +16 -11
- package/src/prompts/sdks/types.ts +2 -2
- package/src/testing/acceptance.ts +190 -0
- package/src/testing/define-api.ts +279 -0
- package/src/testing/helpers.ts +251 -0
- package/src/testing/phoenix-test-tracking.ts +637 -0
- package/src/testing/report-artifacts.ts +272 -0
- package/src/testing/report-run.ts +44 -0
- package/src/testing/reporter-format.ts +1072 -0
- package/src/testing/runner.ts +350 -0
- package/src/testing/state.ts +165 -0
- package/src/testing/types.ts +366 -0
- package/src/utils/channel.ts +17 -15
- package/src/utils/promisifyResult.ts +6 -4
- package/src/utils/schemaMatches.ts +12 -10
- package/src/vitest/index.ts +57 -0
- package/src/vitest/reporter.ts +32 -0
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
import type { TestResult } from "./state";
|
|
2
|
+
import type {
|
|
3
|
+
AcceptanceCriterion,
|
|
4
|
+
AcceptanceResult,
|
|
5
|
+
Annotation,
|
|
6
|
+
OptimizationDirection,
|
|
7
|
+
} from "./types";
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Evaluate all configured aggregate acceptance rules against completed runs.
|
|
11
|
+
* @param params - Evaluation parameters.
|
|
12
|
+
* @param params.criteria - Aggregate rules configured on the suite.
|
|
13
|
+
* @param params.results - Completed test results in the suite.
|
|
14
|
+
*/
|
|
15
|
+
export function evaluateAcceptanceCriteria({
|
|
16
|
+
criteria,
|
|
17
|
+
results,
|
|
18
|
+
}: {
|
|
19
|
+
criteria: readonly AcceptanceCriterion[] | undefined;
|
|
20
|
+
results: readonly TestResult[];
|
|
21
|
+
}): AcceptanceResult[] {
|
|
22
|
+
if (!criteria || criteria.length === 0) {
|
|
23
|
+
return [];
|
|
24
|
+
}
|
|
25
|
+
return criteria.map((criterion) =>
|
|
26
|
+
evaluateAcceptanceCriterion({ criterion, results })
|
|
27
|
+
);
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Build one error containing every failed aggregate criterion.
|
|
32
|
+
* @param results - Computed acceptance results.
|
|
33
|
+
*/
|
|
34
|
+
export function createAcceptanceFailureError(
|
|
35
|
+
results: readonly AcceptanceResult[]
|
|
36
|
+
): Error | undefined {
|
|
37
|
+
const failedResults = results.filter((result) => !result.passed);
|
|
38
|
+
if (failedResults.length === 0) {
|
|
39
|
+
return undefined;
|
|
40
|
+
}
|
|
41
|
+
return new Error(
|
|
42
|
+
[
|
|
43
|
+
"Acceptance criteria failed:",
|
|
44
|
+
...failedResults.map((result) => ` ${formatAcceptanceResult(result)}`),
|
|
45
|
+
].join("\n")
|
|
46
|
+
);
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/** Format an acceptance result for reporters and thrown errors. */
|
|
50
|
+
export function formatAcceptanceResult(result: AcceptanceResult): string {
|
|
51
|
+
const status = result.passed ? "PASS" : "FAIL";
|
|
52
|
+
const value = result.value === null ? "n/a" : result.value.toFixed(3);
|
|
53
|
+
const sampleLabel = result.sampleCount === 1 ? "sample" : "samples";
|
|
54
|
+
let requirement: string;
|
|
55
|
+
if (result.metric === "average") {
|
|
56
|
+
const cmp = (result.direction ?? "maximize") === "minimize" ? "<=" : ">=";
|
|
57
|
+
requirement = `mean ${cmp} ${result.threshold.toFixed(3)}`;
|
|
58
|
+
} else {
|
|
59
|
+
requirement = `pass rate >= ${result.minPassRate.toFixed(3)}`;
|
|
60
|
+
}
|
|
61
|
+
const reason = result.failureReason ? ` - ${result.failureReason}` : "";
|
|
62
|
+
return `${status} ${result.annotationName} ${result.metric} ${value} (need ${requirement}; ${result.sampleCount} ${sampleLabel})${reason}`;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
function evaluateAcceptanceCriterion({
|
|
66
|
+
criterion,
|
|
67
|
+
results,
|
|
68
|
+
}: {
|
|
69
|
+
criterion: AcceptanceCriterion;
|
|
70
|
+
results: readonly TestResult[];
|
|
71
|
+
}): AcceptanceResult {
|
|
72
|
+
const annotations = collectAnnotations({ criterion, results });
|
|
73
|
+
|
|
74
|
+
if (criterion.metric === "average") {
|
|
75
|
+
// Only numeric / boolean scores can be averaged.
|
|
76
|
+
const scores = annotations
|
|
77
|
+
.map((annotation) => annotation.score)
|
|
78
|
+
.filter(isValidScore);
|
|
79
|
+
if (scores.length === 0) {
|
|
80
|
+
return {
|
|
81
|
+
...criterion,
|
|
82
|
+
value: null,
|
|
83
|
+
sampleCount: 0,
|
|
84
|
+
passed: false,
|
|
85
|
+
failureReason: "no numeric or boolean scores found",
|
|
86
|
+
};
|
|
87
|
+
}
|
|
88
|
+
const direction = criterion.direction ?? "maximize";
|
|
89
|
+
const value = calculateAverage(scores);
|
|
90
|
+
return {
|
|
91
|
+
...criterion,
|
|
92
|
+
value,
|
|
93
|
+
sampleCount: scores.length,
|
|
94
|
+
passed: meetsBar(value, criterion.threshold, direction),
|
|
95
|
+
};
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
// passRate: each run passes when `passFn` returns true for its annotation;
|
|
99
|
+
// the suite passes when the fraction of passing runs is at least
|
|
100
|
+
// `minPassRate`. The reported value is that fraction.
|
|
101
|
+
if (annotations.length === 0) {
|
|
102
|
+
return {
|
|
103
|
+
...criterion,
|
|
104
|
+
value: null,
|
|
105
|
+
sampleCount: 0,
|
|
106
|
+
passed: false,
|
|
107
|
+
failureReason: "no matching annotations found",
|
|
108
|
+
};
|
|
109
|
+
}
|
|
110
|
+
const passed = annotations.filter((annotation) =>
|
|
111
|
+
criterion.passFn(annotation)
|
|
112
|
+
).length;
|
|
113
|
+
const value = passed / annotations.length;
|
|
114
|
+
return {
|
|
115
|
+
...criterion,
|
|
116
|
+
value,
|
|
117
|
+
sampleCount: annotations.length,
|
|
118
|
+
passed: value >= criterion.minPassRate,
|
|
119
|
+
};
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/** Whether `value` clears `bar` in the given optimization direction. */
|
|
123
|
+
function meetsBar(
|
|
124
|
+
value: number,
|
|
125
|
+
bar: number,
|
|
126
|
+
direction: OptimizationDirection
|
|
127
|
+
): boolean {
|
|
128
|
+
return direction === "minimize" ? value <= bar : value >= bar;
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* The last annotation matching `annotationName` from each non-skipped run that
|
|
133
|
+
* logged it. One entry per run; runs that never logged the annotation are
|
|
134
|
+
* omitted.
|
|
135
|
+
*/
|
|
136
|
+
function collectAnnotations({
|
|
137
|
+
criterion,
|
|
138
|
+
results,
|
|
139
|
+
}: {
|
|
140
|
+
criterion: AcceptanceCriterion;
|
|
141
|
+
results: readonly TestResult[];
|
|
142
|
+
}): Annotation[] {
|
|
143
|
+
return results
|
|
144
|
+
.filter((result) => result.status !== "skipped")
|
|
145
|
+
.map((result) =>
|
|
146
|
+
findLastAnnotation({
|
|
147
|
+
annotations: result.annotations,
|
|
148
|
+
annotationName: criterion.annotationName,
|
|
149
|
+
})
|
|
150
|
+
)
|
|
151
|
+
.filter((annotation): annotation is Annotation => annotation !== undefined);
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
function findLastAnnotation({
|
|
155
|
+
annotations,
|
|
156
|
+
annotationName,
|
|
157
|
+
}: {
|
|
158
|
+
annotations: readonly Annotation[];
|
|
159
|
+
annotationName: string;
|
|
160
|
+
}): Annotation | undefined {
|
|
161
|
+
for (
|
|
162
|
+
let annotationIndex = annotations.length - 1;
|
|
163
|
+
annotationIndex >= 0;
|
|
164
|
+
annotationIndex--
|
|
165
|
+
) {
|
|
166
|
+
const annotation = annotations[annotationIndex];
|
|
167
|
+
if (annotation?.name === annotationName) {
|
|
168
|
+
return annotation;
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
return undefined;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
function isValidScore(score: Annotation["score"]): score is number | boolean {
|
|
175
|
+
return (
|
|
176
|
+
typeof score === "boolean" ||
|
|
177
|
+
(typeof score === "number" && Number.isFinite(score))
|
|
178
|
+
);
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
function calculateAverage(scores: readonly (number | boolean)[]): number {
|
|
182
|
+
const total = scores
|
|
183
|
+
.map(scoreToNumber)
|
|
184
|
+
.reduce((sum, score) => sum + score, 0);
|
|
185
|
+
return total / scores.length;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
function scoreToNumber(score: number | boolean): number {
|
|
189
|
+
return typeof score === "boolean" ? (score ? 1 : 0) : score;
|
|
190
|
+
}
|
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
import { declareDescribe, declareTest, type RunnerHooks } from "./runner";
|
|
2
|
+
import {
|
|
3
|
+
type KVMap,
|
|
4
|
+
resolveReference,
|
|
5
|
+
type SuiteConfig,
|
|
6
|
+
type TestEachRow,
|
|
7
|
+
type TestFn,
|
|
8
|
+
type TestParams,
|
|
9
|
+
} from "./types";
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Declare Phoenix eval test suites.
|
|
13
|
+
*
|
|
14
|
+
* Drop-in replacement for the test runner's own `describe`. The suite name
|
|
15
|
+
* doubles as the dataset and experiment name on the Phoenix server, and the
|
|
16
|
+
* optional {@link SuiteConfig} controls dataset naming, repetitions, dry-run
|
|
17
|
+
* mode, and CI acceptance criteria.
|
|
18
|
+
*
|
|
19
|
+
* @example
|
|
20
|
+
* ```ts
|
|
21
|
+
* import * as px from "@arizeai/phoenix-client/vitest";
|
|
22
|
+
*
|
|
23
|
+
* px.describe("generate sql demo", () => {
|
|
24
|
+
* px.test("offtopic input", { input: { question: "hi" } }, async ({ input }) => {
|
|
25
|
+
* // ...
|
|
26
|
+
* });
|
|
27
|
+
* }, { metadata: { model: "gpt-4o-mini" } });
|
|
28
|
+
* ```
|
|
29
|
+
*/
|
|
30
|
+
export interface PhoenixDescribe {
|
|
31
|
+
/**
|
|
32
|
+
* Declare a Phoenix eval test suite.
|
|
33
|
+
*
|
|
34
|
+
* @param name - Suite name; doubles as the dataset / experiment name on Phoenix.
|
|
35
|
+
* @param fn - Suite body that declares its `test` / `it` cases.
|
|
36
|
+
* @param config - Optional suite-level config (dataset name, repetitions, dry-run, acceptance criteria).
|
|
37
|
+
*/
|
|
38
|
+
(name: string, fn: () => void, config?: SuiteConfig): void;
|
|
39
|
+
/**
|
|
40
|
+
* Run only this suite, skipping all sibling suites
|
|
41
|
+
* (matches the runner's `describe.only`).
|
|
42
|
+
*
|
|
43
|
+
* @param name - Suite name; doubles as the dataset / experiment name on Phoenix.
|
|
44
|
+
* @param fn - Suite body that declares its `test` / `it` cases.
|
|
45
|
+
* @param config - Optional suite-level config.
|
|
46
|
+
*/
|
|
47
|
+
only(name: string, fn: () => void, config?: SuiteConfig): void;
|
|
48
|
+
/**
|
|
49
|
+
* Skip this suite entirely (matches the runner's `describe.skip`). No dataset
|
|
50
|
+
* or experiment is created on Phoenix.
|
|
51
|
+
*
|
|
52
|
+
* @param name - Suite name; doubles as the dataset / experiment name on Phoenix.
|
|
53
|
+
* @param fn - Suite body (not executed).
|
|
54
|
+
* @param config - Optional suite-level config.
|
|
55
|
+
*/
|
|
56
|
+
skip(name: string, fn: () => void, config?: SuiteConfig): void;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* The test body returned by {@link PhoenixTest.each} after a table is bound.
|
|
61
|
+
*
|
|
62
|
+
* @param name - Test name, or a template (`%i` / `%s` / `%j`), or a function
|
|
63
|
+
* that derives the name from the row and its index.
|
|
64
|
+
* @param fn - The test handler, run once per row in the bound table.
|
|
65
|
+
* @param timeout - Optional per-test timeout in milliseconds.
|
|
66
|
+
*/
|
|
67
|
+
export type PhoenixTestEach<
|
|
68
|
+
Input extends KVMap = KVMap,
|
|
69
|
+
Expected extends KVMap = KVMap,
|
|
70
|
+
> = (
|
|
71
|
+
name: string | ((row: TestEachRow<Input, Expected>, index: number) => string),
|
|
72
|
+
fn: TestFn<Input, Expected>,
|
|
73
|
+
timeout?: number
|
|
74
|
+
) => void;
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Declare a single Phoenix eval test case.
|
|
78
|
+
*
|
|
79
|
+
* Drop-in replacement for the test runner's own `test` / `it`. The `params`
|
|
80
|
+
* argument carries the `input` and the reference output (`expected` /
|
|
81
|
+
* `reference` / `output`) that become the dataset example; whatever the handler
|
|
82
|
+
* returns (or passes to `logOutput()`) is recorded as the experiment run's
|
|
83
|
+
* output and made available to evaluators.
|
|
84
|
+
*
|
|
85
|
+
* `it` is the canonical alias for `test`; the two are identical.
|
|
86
|
+
*
|
|
87
|
+
* @example
|
|
88
|
+
* ```ts
|
|
89
|
+
* px.test(
|
|
90
|
+
* "summarizes the article",
|
|
91
|
+
* { input: { article }, expected: { summary } },
|
|
92
|
+
* async ({ input, expected }) => {
|
|
93
|
+
* const output = await summarize(input.article);
|
|
94
|
+
* px.logOutput(output);
|
|
95
|
+
* await px.evaluate({ name: "matches", evaluate: () => output === expected.summary });
|
|
96
|
+
* }
|
|
97
|
+
* );
|
|
98
|
+
* ```
|
|
99
|
+
*/
|
|
100
|
+
export interface PhoenixTest {
|
|
101
|
+
/**
|
|
102
|
+
* Declare a single Phoenix eval test case.
|
|
103
|
+
*
|
|
104
|
+
* @param name - Test case name; doubles as the dataset example label.
|
|
105
|
+
* @param params - Inline `input` and reference output that become the dataset example.
|
|
106
|
+
* @param fn - Test handler; receives `{ input, expected, metadata }`.
|
|
107
|
+
* @param timeout - Optional per-test timeout in milliseconds.
|
|
108
|
+
*/
|
|
109
|
+
<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
|
|
110
|
+
name: string,
|
|
111
|
+
params: TestParams<Input, Expected>,
|
|
112
|
+
fn: TestFn<Input, Expected>,
|
|
113
|
+
timeout?: number
|
|
114
|
+
): void;
|
|
115
|
+
/**
|
|
116
|
+
* Run only this test case, skipping its siblings
|
|
117
|
+
* (matches the runner's `test.only`).
|
|
118
|
+
*
|
|
119
|
+
* @param name - Test case name; doubles as the dataset example label.
|
|
120
|
+
* @param params - Inline `input` and reference output that become the dataset example.
|
|
121
|
+
* @param fn - Test handler; receives `{ input, expected, metadata }`.
|
|
122
|
+
* @param timeout - Optional per-test timeout in milliseconds.
|
|
123
|
+
*/
|
|
124
|
+
only<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
|
|
125
|
+
name: string,
|
|
126
|
+
params: TestParams<Input, Expected>,
|
|
127
|
+
fn: TestFn<Input, Expected>,
|
|
128
|
+
timeout?: number
|
|
129
|
+
): void;
|
|
130
|
+
/**
|
|
131
|
+
* Skip this test case (matches the runner's `test.skip`). No dataset example
|
|
132
|
+
* or experiment run is created on Phoenix.
|
|
133
|
+
*
|
|
134
|
+
* @param name - Test case name; doubles as the dataset example label.
|
|
135
|
+
* @param params - Inline `input` and reference output (not used while skipped).
|
|
136
|
+
* @param fn - Test handler (not executed).
|
|
137
|
+
* @param timeout - Optional per-test timeout in milliseconds.
|
|
138
|
+
*/
|
|
139
|
+
skip<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
|
|
140
|
+
name: string,
|
|
141
|
+
params: TestParams<Input, Expected>,
|
|
142
|
+
fn: TestFn<Input, Expected>,
|
|
143
|
+
timeout?: number
|
|
144
|
+
): void;
|
|
145
|
+
/**
|
|
146
|
+
* Run the same test handler across many examples. Returns a function that
|
|
147
|
+
* takes a name (or template / name-builder) and the shared test body; each
|
|
148
|
+
* row in `table` becomes its own dataset example and experiment run.
|
|
149
|
+
*
|
|
150
|
+
* @param table - Rows of `{ input, expected?, metadata?, ... }` to fan out over.
|
|
151
|
+
* @returns A {@link PhoenixTestEach} that binds the name and shared handler.
|
|
152
|
+
*
|
|
153
|
+
* @example
|
|
154
|
+
* ```ts
|
|
155
|
+
* px.test.each([
|
|
156
|
+
* { input: { a: 1, b: 2 }, expected: { sum: 3 } },
|
|
157
|
+
* { input: { a: 2, b: 2 }, expected: { sum: 4 } },
|
|
158
|
+
* ])("adds %j", async ({ input, expected }) => {
|
|
159
|
+
* // ...
|
|
160
|
+
* });
|
|
161
|
+
* ```
|
|
162
|
+
*/
|
|
163
|
+
each<Input extends KVMap, Expected extends KVMap>(
|
|
164
|
+
table: TestEachRow<Input, Expected>[]
|
|
165
|
+
): PhoenixTestEach<Input, Expected>;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/** The public testing surface returned by {@link createTestApi}. */
|
|
169
|
+
export interface PhoenixTestApi {
|
|
170
|
+
/** Declare a Phoenix eval test suite. See {@link PhoenixDescribe}. */
|
|
171
|
+
describe: PhoenixDescribe;
|
|
172
|
+
/** Declare a Phoenix eval test case. See {@link PhoenixTest}. */
|
|
173
|
+
test: PhoenixTest;
|
|
174
|
+
/** Canonical alias for {@link PhoenixTestApi.test}. */
|
|
175
|
+
it: PhoenixTest;
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* Build the public `describe`/`test`/`it` API for a runner adapter.
|
|
180
|
+
*
|
|
181
|
+
* Both the jest and vitest entrypoints expose the identical surface; the only
|
|
182
|
+
* thing that differs between them is how the {@link RunnerHooks} are obtained
|
|
183
|
+
* (vitest imports them statically, jest resolves them lazily from globals).
|
|
184
|
+
* That difference is captured by `getHooks`, which is invoked once per
|
|
185
|
+
* declaration so adapters are free to resolve hooks lazily.
|
|
186
|
+
*
|
|
187
|
+
* The JSDoc that surfaces in editors lives on the {@link PhoenixDescribe} and
|
|
188
|
+
* {@link PhoenixTest} interfaces rather than the implementations below, so the
|
|
189
|
+
* docs survive the `export const { describe, test, it } = createTestApi(...)`
|
|
190
|
+
* destructuring in each adapter.
|
|
191
|
+
*/
|
|
192
|
+
export function createTestApi(getHooks: () => RunnerHooks): PhoenixTestApi {
|
|
193
|
+
const describe = ((
|
|
194
|
+
name: string,
|
|
195
|
+
fn: () => void,
|
|
196
|
+
config?: SuiteConfig
|
|
197
|
+
): void => {
|
|
198
|
+
declareDescribe(getHooks(), name, fn, config ?? {});
|
|
199
|
+
}) as PhoenixDescribe;
|
|
200
|
+
describe.only = (name, fn, config) => {
|
|
201
|
+
declareDescribe(getHooks(), name, fn, config ?? {}, "only");
|
|
202
|
+
};
|
|
203
|
+
describe.skip = (name, fn, config) => {
|
|
204
|
+
declareDescribe(getHooks(), name, fn, config ?? {}, "skip");
|
|
205
|
+
};
|
|
206
|
+
|
|
207
|
+
const test = (<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
|
|
208
|
+
name: string,
|
|
209
|
+
params: TestParams<Input, Expected>,
|
|
210
|
+
fn: TestFn<Input, Expected>,
|
|
211
|
+
timeout?: number
|
|
212
|
+
): void => {
|
|
213
|
+
declareTest(getHooks(), name, params, fn, "default", timeout);
|
|
214
|
+
}) as PhoenixTest;
|
|
215
|
+
test.only = (name, params, fn, timeout) => {
|
|
216
|
+
declareTest(getHooks(), name, params, fn, "only", timeout);
|
|
217
|
+
};
|
|
218
|
+
test.skip = (name, params, fn, timeout) => {
|
|
219
|
+
declareTest(getHooks(), name, params, fn, "skip", timeout);
|
|
220
|
+
};
|
|
221
|
+
test.each = <Input extends KVMap, Expected extends KVMap>(
|
|
222
|
+
table: TestEachRow<Input, Expected>[]
|
|
223
|
+
): PhoenixTestEach<Input, Expected> => {
|
|
224
|
+
return (name, fn, timeout) => {
|
|
225
|
+
table.forEach((row, i) => {
|
|
226
|
+
const testName =
|
|
227
|
+
typeof name === "function"
|
|
228
|
+
? name(row, i)
|
|
229
|
+
: interpolateName(name, row, i);
|
|
230
|
+
declareTest(
|
|
231
|
+
getHooks(),
|
|
232
|
+
testName,
|
|
233
|
+
{
|
|
234
|
+
id: row.id,
|
|
235
|
+
input: row.input,
|
|
236
|
+
expected: resolveReference(row),
|
|
237
|
+
metadata: row.metadata,
|
|
238
|
+
splits: row.splits,
|
|
239
|
+
repetitions: row.repetitions,
|
|
240
|
+
dryRun: row.dryRun,
|
|
241
|
+
},
|
|
242
|
+
fn,
|
|
243
|
+
"default",
|
|
244
|
+
timeout
|
|
245
|
+
);
|
|
246
|
+
});
|
|
247
|
+
};
|
|
248
|
+
};
|
|
249
|
+
|
|
250
|
+
// `it` is the canonical alias for `test`.
|
|
251
|
+
const it = test;
|
|
252
|
+
|
|
253
|
+
return { describe, test, it };
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
/**
|
|
257
|
+
* Interpolate a `test.each` name template for a single row. Supports the
|
|
258
|
+
* common `%i`/`%s`/`%j` placeholders for surface parity with the underlying
|
|
259
|
+
* runners; when no placeholder is present the 1-based row index is appended.
|
|
260
|
+
*/
|
|
261
|
+
function interpolateName(
|
|
262
|
+
name: string,
|
|
263
|
+
row: TestEachRow,
|
|
264
|
+
index: number
|
|
265
|
+
): string {
|
|
266
|
+
if (!name.includes("%")) {
|
|
267
|
+
return `${name} #${index + 1}`;
|
|
268
|
+
}
|
|
269
|
+
const replacements: Array<[RegExp, string]> = [
|
|
270
|
+
[/%i/g, String(index)],
|
|
271
|
+
[/%s/g, JSON.stringify(row.input)],
|
|
272
|
+
[/%j/g, JSON.stringify(row)],
|
|
273
|
+
];
|
|
274
|
+
let out = name;
|
|
275
|
+
for (const [pattern, value] of replacements) {
|
|
276
|
+
out = out.replace(pattern, value);
|
|
277
|
+
}
|
|
278
|
+
return out;
|
|
279
|
+
}
|