@arizeai/phoenix-client 6.10.1 → 6.11.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (214) hide show
  1. package/README.md +62 -0
  2. package/dist/esm/__generated__/api/v1.d.ts +254 -2
  3. package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
  4. package/dist/esm/jest/index.d.ts +5 -0
  5. package/dist/esm/jest/index.d.ts.map +1 -0
  6. package/dist/esm/jest/index.js +49 -0
  7. package/dist/esm/jest/index.js.map +1 -0
  8. package/dist/esm/jest/reporter.d.ts +13 -0
  9. package/dist/esm/jest/reporter.d.ts.map +1 -0
  10. package/dist/esm/jest/reporter.js +19 -0
  11. package/dist/esm/jest/reporter.js.map +1 -0
  12. package/dist/esm/prompts/sdks/toAI.d.ts +2 -2
  13. package/dist/esm/prompts/sdks/toAI.d.ts.map +1 -1
  14. package/dist/esm/prompts/sdks/toAI.js.map +1 -1
  15. package/dist/esm/prompts/sdks/toAnthropic.d.ts +2 -2
  16. package/dist/esm/prompts/sdks/toAnthropic.d.ts.map +1 -1
  17. package/dist/esm/prompts/sdks/toAnthropic.js.map +1 -1
  18. package/dist/esm/prompts/sdks/toOpenAI.d.ts +2 -2
  19. package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
  20. package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
  21. package/dist/esm/prompts/sdks/toSDK.d.ts +8 -8
  22. package/dist/esm/prompts/sdks/toSDK.d.ts.map +1 -1
  23. package/dist/esm/prompts/sdks/toSDK.js.map +1 -1
  24. package/dist/esm/prompts/sdks/types.d.ts +2 -2
  25. package/dist/esm/prompts/sdks/types.d.ts.map +1 -1
  26. package/dist/esm/schemas/llm/anthropic/converters.d.ts +8 -8
  27. package/dist/esm/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  28. package/dist/esm/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  29. package/dist/esm/schemas/llm/constants.d.ts +3 -3
  30. package/dist/esm/schemas/llm/converters.d.ts +12 -12
  31. package/dist/esm/schemas/llm/openai/converters.d.ts +3 -3
  32. package/dist/esm/schemas/llm/schemas.d.ts +2 -2
  33. package/dist/esm/testing/acceptance.d.ts +20 -0
  34. package/dist/esm/testing/acceptance.d.ts.map +1 -0
  35. package/dist/esm/testing/acceptance.js +129 -0
  36. package/dist/esm/testing/acceptance.js.map +1 -0
  37. package/dist/esm/testing/define-api.d.ts +157 -0
  38. package/dist/esm/testing/define-api.d.ts.map +1 -0
  39. package/dist/esm/testing/define-api.js +78 -0
  40. package/dist/esm/testing/define-api.js.map +1 -0
  41. package/dist/esm/testing/helpers.d.ts +55 -0
  42. package/dist/esm/testing/helpers.d.ts.map +1 -0
  43. package/dist/esm/testing/helpers.js +179 -0
  44. package/dist/esm/testing/helpers.js.map +1 -0
  45. package/dist/esm/testing/phoenix-test-tracking.d.ts +68 -0
  46. package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -0
  47. package/dist/esm/testing/phoenix-test-tracking.js +521 -0
  48. package/dist/esm/testing/phoenix-test-tracking.js.map +1 -0
  49. package/dist/esm/testing/report-artifacts.d.ts +45 -0
  50. package/dist/esm/testing/report-artifacts.d.ts.map +1 -0
  51. package/dist/esm/testing/report-artifacts.js +218 -0
  52. package/dist/esm/testing/report-artifacts.js.map +1 -0
  53. package/dist/esm/testing/report-run.d.ts +22 -0
  54. package/dist/esm/testing/report-run.d.ts.map +1 -0
  55. package/dist/esm/testing/report-run.js +41 -0
  56. package/dist/esm/testing/report-run.js.map +1 -0
  57. package/dist/esm/testing/reporter-format.d.ts +83 -0
  58. package/dist/esm/testing/reporter-format.d.ts.map +1 -0
  59. package/dist/esm/testing/reporter-format.js +852 -0
  60. package/dist/esm/testing/reporter-format.js.map +1 -0
  61. package/dist/esm/testing/runner.d.ts +31 -0
  62. package/dist/esm/testing/runner.d.ts.map +1 -0
  63. package/dist/esm/testing/runner.js +238 -0
  64. package/dist/esm/testing/runner.js.map +1 -0
  65. package/dist/esm/testing/state.d.ts +138 -0
  66. package/dist/esm/testing/state.d.ts.map +1 -0
  67. package/dist/esm/testing/state.js +31 -0
  68. package/dist/esm/testing/state.js.map +1 -0
  69. package/dist/esm/testing/types.d.ts +319 -0
  70. package/dist/esm/testing/types.d.ts.map +1 -0
  71. package/dist/esm/testing/types.js +9 -0
  72. package/dist/esm/testing/types.js.map +1 -0
  73. package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
  74. package/dist/esm/utils/channel.d.ts +7 -7
  75. package/dist/esm/utils/channel.d.ts.map +1 -1
  76. package/dist/esm/utils/channel.js +1 -1
  77. package/dist/esm/utils/channel.js.map +1 -1
  78. package/dist/esm/utils/formatPromptMessages.d.ts.map +1 -1
  79. package/dist/esm/utils/getPromptBySelector.d.ts.map +1 -1
  80. package/dist/esm/utils/promisifyResult.d.ts +1 -1
  81. package/dist/esm/utils/promisifyResult.d.ts.map +1 -1
  82. package/dist/esm/utils/promisifyResult.js.map +1 -1
  83. package/dist/esm/utils/schemaMatches.d.ts +5 -5
  84. package/dist/esm/utils/schemaMatches.d.ts.map +1 -1
  85. package/dist/esm/utils/schemaMatches.js.map +1 -1
  86. package/dist/esm/vitest/index.d.ts +5 -0
  87. package/dist/esm/vitest/index.d.ts.map +1 -0
  88. package/dist/esm/vitest/index.js +15 -0
  89. package/dist/esm/vitest/index.js.map +1 -0
  90. package/dist/esm/vitest/reporter.d.ts +19 -0
  91. package/dist/esm/vitest/reporter.d.ts.map +1 -0
  92. package/dist/esm/vitest/reporter.js +27 -0
  93. package/dist/esm/vitest/reporter.js.map +1 -0
  94. package/dist/src/__generated__/api/v1.d.ts +254 -2
  95. package/dist/src/__generated__/api/v1.d.ts.map +1 -1
  96. package/dist/src/jest/index.d.ts +5 -0
  97. package/dist/src/jest/index.d.ts.map +1 -0
  98. package/dist/src/jest/index.js +58 -0
  99. package/dist/src/jest/index.js.map +1 -0
  100. package/dist/src/jest/reporter.d.ts +13 -0
  101. package/dist/src/jest/reporter.d.ts.map +1 -0
  102. package/dist/src/jest/reporter.js +23 -0
  103. package/dist/src/jest/reporter.js.map +1 -0
  104. package/dist/src/prompts/sdks/toAI.d.ts +2 -2
  105. package/dist/src/prompts/sdks/toAI.d.ts.map +1 -1
  106. package/dist/src/prompts/sdks/toAI.js.map +1 -1
  107. package/dist/src/prompts/sdks/toAnthropic.d.ts +2 -2
  108. package/dist/src/prompts/sdks/toAnthropic.d.ts.map +1 -1
  109. package/dist/src/prompts/sdks/toAnthropic.js.map +1 -1
  110. package/dist/src/prompts/sdks/toOpenAI.d.ts +2 -2
  111. package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
  112. package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
  113. package/dist/src/prompts/sdks/toSDK.d.ts +8 -8
  114. package/dist/src/prompts/sdks/toSDK.d.ts.map +1 -1
  115. package/dist/src/prompts/sdks/toSDK.js.map +1 -1
  116. package/dist/src/prompts/sdks/types.d.ts +2 -2
  117. package/dist/src/prompts/sdks/types.d.ts.map +1 -1
  118. package/dist/src/schemas/llm/anthropic/converters.d.ts +8 -8
  119. package/dist/src/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  120. package/dist/src/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  121. package/dist/src/schemas/llm/constants.d.ts +3 -3
  122. package/dist/src/schemas/llm/converters.d.ts +12 -12
  123. package/dist/src/schemas/llm/openai/converters.d.ts +3 -3
  124. package/dist/src/schemas/llm/schemas.d.ts +2 -2
  125. package/dist/src/testing/acceptance.d.ts +20 -0
  126. package/dist/src/testing/acceptance.d.ts.map +1 -0
  127. package/dist/src/testing/acceptance.js +114 -0
  128. package/dist/src/testing/acceptance.js.map +1 -0
  129. package/dist/src/testing/define-api.d.ts +157 -0
  130. package/dist/src/testing/define-api.d.ts.map +1 -0
  131. package/dist/src/testing/define-api.js +81 -0
  132. package/dist/src/testing/define-api.js.map +1 -0
  133. package/dist/src/testing/helpers.d.ts +55 -0
  134. package/dist/src/testing/helpers.d.ts.map +1 -0
  135. package/dist/src/testing/helpers.js +182 -0
  136. package/dist/src/testing/helpers.js.map +1 -0
  137. package/dist/src/testing/phoenix-test-tracking.d.ts +68 -0
  138. package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -0
  139. package/dist/src/testing/phoenix-test-tracking.js +530 -0
  140. package/dist/src/testing/phoenix-test-tracking.js.map +1 -0
  141. package/dist/src/testing/report-artifacts.d.ts +45 -0
  142. package/dist/src/testing/report-artifacts.d.ts.map +1 -0
  143. package/dist/src/testing/report-artifacts.js +225 -0
  144. package/dist/src/testing/report-artifacts.js.map +1 -0
  145. package/dist/src/testing/report-run.d.ts +22 -0
  146. package/dist/src/testing/report-run.d.ts.map +1 -0
  147. package/dist/src/testing/report-run.js +47 -0
  148. package/dist/src/testing/report-run.js.map +1 -0
  149. package/dist/src/testing/reporter-format.d.ts +83 -0
  150. package/dist/src/testing/reporter-format.d.ts.map +1 -0
  151. package/dist/src/testing/reporter-format.js +870 -0
  152. package/dist/src/testing/reporter-format.js.map +1 -0
  153. package/dist/src/testing/runner.d.ts +31 -0
  154. package/dist/src/testing/runner.d.ts.map +1 -0
  155. package/dist/src/testing/runner.js +258 -0
  156. package/dist/src/testing/runner.js.map +1 -0
  157. package/dist/src/testing/state.d.ts +138 -0
  158. package/dist/src/testing/state.d.ts.map +1 -0
  159. package/dist/src/testing/state.js +38 -0
  160. package/dist/src/testing/state.js.map +1 -0
  161. package/dist/src/testing/types.d.ts +319 -0
  162. package/dist/src/testing/types.d.ts.map +1 -0
  163. package/dist/src/testing/types.js +13 -0
  164. package/dist/src/testing/types.js.map +1 -0
  165. package/dist/src/utils/channel.d.ts +7 -7
  166. package/dist/src/utils/channel.d.ts.map +1 -1
  167. package/dist/src/utils/channel.js +1 -1
  168. package/dist/src/utils/channel.js.map +1 -1
  169. package/dist/src/utils/formatPromptMessages.d.ts.map +1 -1
  170. package/dist/src/utils/getPromptBySelector.d.ts.map +1 -1
  171. package/dist/src/utils/promisifyResult.d.ts +1 -1
  172. package/dist/src/utils/promisifyResult.d.ts.map +1 -1
  173. package/dist/src/utils/promisifyResult.js.map +1 -1
  174. package/dist/src/utils/schemaMatches.d.ts +5 -5
  175. package/dist/src/utils/schemaMatches.d.ts.map +1 -1
  176. package/dist/src/utils/schemaMatches.js.map +1 -1
  177. package/dist/src/vitest/index.d.ts +5 -0
  178. package/dist/src/vitest/index.d.ts.map +1 -0
  179. package/dist/src/vitest/index.js +23 -0
  180. package/dist/src/vitest/index.js.map +1 -0
  181. package/dist/src/vitest/reporter.d.ts +19 -0
  182. package/dist/src/vitest/reporter.d.ts.map +1 -0
  183. package/dist/src/vitest/reporter.js +34 -0
  184. package/dist/src/vitest/reporter.js.map +1 -0
  185. package/dist/tsconfig.tsbuildinfo +1 -1
  186. package/docs/ci-evals-annotations.mdx +190 -0
  187. package/docs/ci-evals-jest.mdx +78 -0
  188. package/docs/ci-evals-vitest.mdx +240 -0
  189. package/docs/ci-evals.mdx +263 -0
  190. package/docs/overview.mdx +9 -1
  191. package/package.json +49 -17
  192. package/src/__generated__/api/v1.ts +254 -2
  193. package/src/jest/index.ts +124 -0
  194. package/src/jest/reporter.ts +22 -0
  195. package/src/prompts/sdks/toAI.ts +4 -3
  196. package/src/prompts/sdks/toAnthropic.ts +4 -3
  197. package/src/prompts/sdks/toOpenAI.ts +4 -3
  198. package/src/prompts/sdks/toSDK.ts +16 -11
  199. package/src/prompts/sdks/types.ts +2 -2
  200. package/src/testing/acceptance.ts +190 -0
  201. package/src/testing/define-api.ts +279 -0
  202. package/src/testing/helpers.ts +251 -0
  203. package/src/testing/phoenix-test-tracking.ts +637 -0
  204. package/src/testing/report-artifacts.ts +272 -0
  205. package/src/testing/report-run.ts +44 -0
  206. package/src/testing/reporter-format.ts +1072 -0
  207. package/src/testing/runner.ts +350 -0
  208. package/src/testing/state.ts +165 -0
  209. package/src/testing/types.ts +366 -0
  210. package/src/utils/channel.ts +17 -15
  211. package/src/utils/promisifyResult.ts +6 -4
  212. package/src/utils/schemaMatches.ts +12 -10
  213. package/src/vitest/index.ts +57 -0
  214. package/src/vitest/reporter.ts +32 -0
@@ -0,0 +1,190 @@
1
+ import type { TestResult } from "./state";
2
+ import type {
3
+ AcceptanceCriterion,
4
+ AcceptanceResult,
5
+ Annotation,
6
+ OptimizationDirection,
7
+ } from "./types";
8
+
9
+ /**
10
+ * Evaluate all configured aggregate acceptance rules against completed runs.
11
+ * @param params - Evaluation parameters.
12
+ * @param params.criteria - Aggregate rules configured on the suite.
13
+ * @param params.results - Completed test results in the suite.
14
+ */
15
+ export function evaluateAcceptanceCriteria({
16
+ criteria,
17
+ results,
18
+ }: {
19
+ criteria: readonly AcceptanceCriterion[] | undefined;
20
+ results: readonly TestResult[];
21
+ }): AcceptanceResult[] {
22
+ if (!criteria || criteria.length === 0) {
23
+ return [];
24
+ }
25
+ return criteria.map((criterion) =>
26
+ evaluateAcceptanceCriterion({ criterion, results })
27
+ );
28
+ }
29
+
30
+ /**
31
+ * Build one error containing every failed aggregate criterion.
32
+ * @param results - Computed acceptance results.
33
+ */
34
+ export function createAcceptanceFailureError(
35
+ results: readonly AcceptanceResult[]
36
+ ): Error | undefined {
37
+ const failedResults = results.filter((result) => !result.passed);
38
+ if (failedResults.length === 0) {
39
+ return undefined;
40
+ }
41
+ return new Error(
42
+ [
43
+ "Acceptance criteria failed:",
44
+ ...failedResults.map((result) => ` ${formatAcceptanceResult(result)}`),
45
+ ].join("\n")
46
+ );
47
+ }
48
+
49
+ /** Format an acceptance result for reporters and thrown errors. */
50
+ export function formatAcceptanceResult(result: AcceptanceResult): string {
51
+ const status = result.passed ? "PASS" : "FAIL";
52
+ const value = result.value === null ? "n/a" : result.value.toFixed(3);
53
+ const sampleLabel = result.sampleCount === 1 ? "sample" : "samples";
54
+ let requirement: string;
55
+ if (result.metric === "average") {
56
+ const cmp = (result.direction ?? "maximize") === "minimize" ? "<=" : ">=";
57
+ requirement = `mean ${cmp} ${result.threshold.toFixed(3)}`;
58
+ } else {
59
+ requirement = `pass rate >= ${result.minPassRate.toFixed(3)}`;
60
+ }
61
+ const reason = result.failureReason ? ` - ${result.failureReason}` : "";
62
+ return `${status} ${result.annotationName} ${result.metric} ${value} (need ${requirement}; ${result.sampleCount} ${sampleLabel})${reason}`;
63
+ }
64
+
65
+ function evaluateAcceptanceCriterion({
66
+ criterion,
67
+ results,
68
+ }: {
69
+ criterion: AcceptanceCriterion;
70
+ results: readonly TestResult[];
71
+ }): AcceptanceResult {
72
+ const annotations = collectAnnotations({ criterion, results });
73
+
74
+ if (criterion.metric === "average") {
75
+ // Only numeric / boolean scores can be averaged.
76
+ const scores = annotations
77
+ .map((annotation) => annotation.score)
78
+ .filter(isValidScore);
79
+ if (scores.length === 0) {
80
+ return {
81
+ ...criterion,
82
+ value: null,
83
+ sampleCount: 0,
84
+ passed: false,
85
+ failureReason: "no numeric or boolean scores found",
86
+ };
87
+ }
88
+ const direction = criterion.direction ?? "maximize";
89
+ const value = calculateAverage(scores);
90
+ return {
91
+ ...criterion,
92
+ value,
93
+ sampleCount: scores.length,
94
+ passed: meetsBar(value, criterion.threshold, direction),
95
+ };
96
+ }
97
+
98
+ // passRate: each run passes when `passFn` returns true for its annotation;
99
+ // the suite passes when the fraction of passing runs is at least
100
+ // `minPassRate`. The reported value is that fraction.
101
+ if (annotations.length === 0) {
102
+ return {
103
+ ...criterion,
104
+ value: null,
105
+ sampleCount: 0,
106
+ passed: false,
107
+ failureReason: "no matching annotations found",
108
+ };
109
+ }
110
+ const passed = annotations.filter((annotation) =>
111
+ criterion.passFn(annotation)
112
+ ).length;
113
+ const value = passed / annotations.length;
114
+ return {
115
+ ...criterion,
116
+ value,
117
+ sampleCount: annotations.length,
118
+ passed: value >= criterion.minPassRate,
119
+ };
120
+ }
121
+
122
+ /** Whether `value` clears `bar` in the given optimization direction. */
123
+ function meetsBar(
124
+ value: number,
125
+ bar: number,
126
+ direction: OptimizationDirection
127
+ ): boolean {
128
+ return direction === "minimize" ? value <= bar : value >= bar;
129
+ }
130
+
131
+ /**
132
+ * The last annotation matching `annotationName` from each non-skipped run that
133
+ * logged it. One entry per run; runs that never logged the annotation are
134
+ * omitted.
135
+ */
136
+ function collectAnnotations({
137
+ criterion,
138
+ results,
139
+ }: {
140
+ criterion: AcceptanceCriterion;
141
+ results: readonly TestResult[];
142
+ }): Annotation[] {
143
+ return results
144
+ .filter((result) => result.status !== "skipped")
145
+ .map((result) =>
146
+ findLastAnnotation({
147
+ annotations: result.annotations,
148
+ annotationName: criterion.annotationName,
149
+ })
150
+ )
151
+ .filter((annotation): annotation is Annotation => annotation !== undefined);
152
+ }
153
+
154
+ function findLastAnnotation({
155
+ annotations,
156
+ annotationName,
157
+ }: {
158
+ annotations: readonly Annotation[];
159
+ annotationName: string;
160
+ }): Annotation | undefined {
161
+ for (
162
+ let annotationIndex = annotations.length - 1;
163
+ annotationIndex >= 0;
164
+ annotationIndex--
165
+ ) {
166
+ const annotation = annotations[annotationIndex];
167
+ if (annotation?.name === annotationName) {
168
+ return annotation;
169
+ }
170
+ }
171
+ return undefined;
172
+ }
173
+
174
+ function isValidScore(score: Annotation["score"]): score is number | boolean {
175
+ return (
176
+ typeof score === "boolean" ||
177
+ (typeof score === "number" && Number.isFinite(score))
178
+ );
179
+ }
180
+
181
+ function calculateAverage(scores: readonly (number | boolean)[]): number {
182
+ const total = scores
183
+ .map(scoreToNumber)
184
+ .reduce((sum, score) => sum + score, 0);
185
+ return total / scores.length;
186
+ }
187
+
188
+ function scoreToNumber(score: number | boolean): number {
189
+ return typeof score === "boolean" ? (score ? 1 : 0) : score;
190
+ }
@@ -0,0 +1,279 @@
1
+ import { declareDescribe, declareTest, type RunnerHooks } from "./runner";
2
+ import {
3
+ type KVMap,
4
+ resolveReference,
5
+ type SuiteConfig,
6
+ type TestEachRow,
7
+ type TestFn,
8
+ type TestParams,
9
+ } from "./types";
10
+
11
+ /**
12
+ * Declare Phoenix eval test suites.
13
+ *
14
+ * Drop-in replacement for the test runner's own `describe`. The suite name
15
+ * doubles as the dataset and experiment name on the Phoenix server, and the
16
+ * optional {@link SuiteConfig} controls dataset naming, repetitions, dry-run
17
+ * mode, and CI acceptance criteria.
18
+ *
19
+ * @example
20
+ * ```ts
21
+ * import * as px from "@arizeai/phoenix-client/vitest";
22
+ *
23
+ * px.describe("generate sql demo", () => {
24
+ * px.test("offtopic input", { input: { question: "hi" } }, async ({ input }) => {
25
+ * // ...
26
+ * });
27
+ * }, { metadata: { model: "gpt-4o-mini" } });
28
+ * ```
29
+ */
30
+ export interface PhoenixDescribe {
31
+ /**
32
+ * Declare a Phoenix eval test suite.
33
+ *
34
+ * @param name - Suite name; doubles as the dataset / experiment name on Phoenix.
35
+ * @param fn - Suite body that declares its `test` / `it` cases.
36
+ * @param config - Optional suite-level config (dataset name, repetitions, dry-run, acceptance criteria).
37
+ */
38
+ (name: string, fn: () => void, config?: SuiteConfig): void;
39
+ /**
40
+ * Run only this suite, skipping all sibling suites
41
+ * (matches the runner's `describe.only`).
42
+ *
43
+ * @param name - Suite name; doubles as the dataset / experiment name on Phoenix.
44
+ * @param fn - Suite body that declares its `test` / `it` cases.
45
+ * @param config - Optional suite-level config.
46
+ */
47
+ only(name: string, fn: () => void, config?: SuiteConfig): void;
48
+ /**
49
+ * Skip this suite entirely (matches the runner's `describe.skip`). No dataset
50
+ * or experiment is created on Phoenix.
51
+ *
52
+ * @param name - Suite name; doubles as the dataset / experiment name on Phoenix.
53
+ * @param fn - Suite body (not executed).
54
+ * @param config - Optional suite-level config.
55
+ */
56
+ skip(name: string, fn: () => void, config?: SuiteConfig): void;
57
+ }
58
+
59
+ /**
60
+ * The test body returned by {@link PhoenixTest.each} after a table is bound.
61
+ *
62
+ * @param name - Test name, or a template (`%i` / `%s` / `%j`), or a function
63
+ * that derives the name from the row and its index.
64
+ * @param fn - The test handler, run once per row in the bound table.
65
+ * @param timeout - Optional per-test timeout in milliseconds.
66
+ */
67
+ export type PhoenixTestEach<
68
+ Input extends KVMap = KVMap,
69
+ Expected extends KVMap = KVMap,
70
+ > = (
71
+ name: string | ((row: TestEachRow<Input, Expected>, index: number) => string),
72
+ fn: TestFn<Input, Expected>,
73
+ timeout?: number
74
+ ) => void;
75
+
76
+ /**
77
+ * Declare a single Phoenix eval test case.
78
+ *
79
+ * Drop-in replacement for the test runner's own `test` / `it`. The `params`
80
+ * argument carries the `input` and the reference output (`expected` /
81
+ * `reference` / `output`) that become the dataset example; whatever the handler
82
+ * returns (or passes to `logOutput()`) is recorded as the experiment run's
83
+ * output and made available to evaluators.
84
+ *
85
+ * `it` is the canonical alias for `test`; the two are identical.
86
+ *
87
+ * @example
88
+ * ```ts
89
+ * px.test(
90
+ * "summarizes the article",
91
+ * { input: { article }, expected: { summary } },
92
+ * async ({ input, expected }) => {
93
+ * const output = await summarize(input.article);
94
+ * px.logOutput(output);
95
+ * await px.evaluate({ name: "matches", evaluate: () => output === expected.summary });
96
+ * }
97
+ * );
98
+ * ```
99
+ */
100
+ export interface PhoenixTest {
101
+ /**
102
+ * Declare a single Phoenix eval test case.
103
+ *
104
+ * @param name - Test case name; doubles as the dataset example label.
105
+ * @param params - Inline `input` and reference output that become the dataset example.
106
+ * @param fn - Test handler; receives `{ input, expected, metadata }`.
107
+ * @param timeout - Optional per-test timeout in milliseconds.
108
+ */
109
+ <Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
110
+ name: string,
111
+ params: TestParams<Input, Expected>,
112
+ fn: TestFn<Input, Expected>,
113
+ timeout?: number
114
+ ): void;
115
+ /**
116
+ * Run only this test case, skipping its siblings
117
+ * (matches the runner's `test.only`).
118
+ *
119
+ * @param name - Test case name; doubles as the dataset example label.
120
+ * @param params - Inline `input` and reference output that become the dataset example.
121
+ * @param fn - Test handler; receives `{ input, expected, metadata }`.
122
+ * @param timeout - Optional per-test timeout in milliseconds.
123
+ */
124
+ only<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
125
+ name: string,
126
+ params: TestParams<Input, Expected>,
127
+ fn: TestFn<Input, Expected>,
128
+ timeout?: number
129
+ ): void;
130
+ /**
131
+ * Skip this test case (matches the runner's `test.skip`). No dataset example
132
+ * or experiment run is created on Phoenix.
133
+ *
134
+ * @param name - Test case name; doubles as the dataset example label.
135
+ * @param params - Inline `input` and reference output (not used while skipped).
136
+ * @param fn - Test handler (not executed).
137
+ * @param timeout - Optional per-test timeout in milliseconds.
138
+ */
139
+ skip<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
140
+ name: string,
141
+ params: TestParams<Input, Expected>,
142
+ fn: TestFn<Input, Expected>,
143
+ timeout?: number
144
+ ): void;
145
+ /**
146
+ * Run the same test handler across many examples. Returns a function that
147
+ * takes a name (or template / name-builder) and the shared test body; each
148
+ * row in `table` becomes its own dataset example and experiment run.
149
+ *
150
+ * @param table - Rows of `{ input, expected?, metadata?, ... }` to fan out over.
151
+ * @returns A {@link PhoenixTestEach} that binds the name and shared handler.
152
+ *
153
+ * @example
154
+ * ```ts
155
+ * px.test.each([
156
+ * { input: { a: 1, b: 2 }, expected: { sum: 3 } },
157
+ * { input: { a: 2, b: 2 }, expected: { sum: 4 } },
158
+ * ])("adds %j", async ({ input, expected }) => {
159
+ * // ...
160
+ * });
161
+ * ```
162
+ */
163
+ each<Input extends KVMap, Expected extends KVMap>(
164
+ table: TestEachRow<Input, Expected>[]
165
+ ): PhoenixTestEach<Input, Expected>;
166
+ }
167
+
168
+ /** The public testing surface returned by {@link createTestApi}. */
169
+ export interface PhoenixTestApi {
170
+ /** Declare a Phoenix eval test suite. See {@link PhoenixDescribe}. */
171
+ describe: PhoenixDescribe;
172
+ /** Declare a Phoenix eval test case. See {@link PhoenixTest}. */
173
+ test: PhoenixTest;
174
+ /** Canonical alias for {@link PhoenixTestApi.test}. */
175
+ it: PhoenixTest;
176
+ }
177
+
178
+ /**
179
+ * Build the public `describe`/`test`/`it` API for a runner adapter.
180
+ *
181
+ * Both the jest and vitest entrypoints expose the identical surface; the only
182
+ * thing that differs between them is how the {@link RunnerHooks} are obtained
183
+ * (vitest imports them statically, jest resolves them lazily from globals).
184
+ * That difference is captured by `getHooks`, which is invoked once per
185
+ * declaration so adapters are free to resolve hooks lazily.
186
+ *
187
+ * The JSDoc that surfaces in editors lives on the {@link PhoenixDescribe} and
188
+ * {@link PhoenixTest} interfaces rather than the implementations below, so the
189
+ * docs survive the `export const { describe, test, it } = createTestApi(...)`
190
+ * destructuring in each adapter.
191
+ */
192
+ export function createTestApi(getHooks: () => RunnerHooks): PhoenixTestApi {
193
+ const describe = ((
194
+ name: string,
195
+ fn: () => void,
196
+ config?: SuiteConfig
197
+ ): void => {
198
+ declareDescribe(getHooks(), name, fn, config ?? {});
199
+ }) as PhoenixDescribe;
200
+ describe.only = (name, fn, config) => {
201
+ declareDescribe(getHooks(), name, fn, config ?? {}, "only");
202
+ };
203
+ describe.skip = (name, fn, config) => {
204
+ declareDescribe(getHooks(), name, fn, config ?? {}, "skip");
205
+ };
206
+
207
+ const test = (<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
208
+ name: string,
209
+ params: TestParams<Input, Expected>,
210
+ fn: TestFn<Input, Expected>,
211
+ timeout?: number
212
+ ): void => {
213
+ declareTest(getHooks(), name, params, fn, "default", timeout);
214
+ }) as PhoenixTest;
215
+ test.only = (name, params, fn, timeout) => {
216
+ declareTest(getHooks(), name, params, fn, "only", timeout);
217
+ };
218
+ test.skip = (name, params, fn, timeout) => {
219
+ declareTest(getHooks(), name, params, fn, "skip", timeout);
220
+ };
221
+ test.each = <Input extends KVMap, Expected extends KVMap>(
222
+ table: TestEachRow<Input, Expected>[]
223
+ ): PhoenixTestEach<Input, Expected> => {
224
+ return (name, fn, timeout) => {
225
+ table.forEach((row, i) => {
226
+ const testName =
227
+ typeof name === "function"
228
+ ? name(row, i)
229
+ : interpolateName(name, row, i);
230
+ declareTest(
231
+ getHooks(),
232
+ testName,
233
+ {
234
+ id: row.id,
235
+ input: row.input,
236
+ expected: resolveReference(row),
237
+ metadata: row.metadata,
238
+ splits: row.splits,
239
+ repetitions: row.repetitions,
240
+ dryRun: row.dryRun,
241
+ },
242
+ fn,
243
+ "default",
244
+ timeout
245
+ );
246
+ });
247
+ };
248
+ };
249
+
250
+ // `it` is the canonical alias for `test`.
251
+ const it = test;
252
+
253
+ return { describe, test, it };
254
+ }
255
+
256
+ /**
257
+ * Interpolate a `test.each` name template for a single row. Supports the
258
+ * common `%i`/`%s`/`%j` placeholders for surface parity with the underlying
259
+ * runners; when no placeholder is present the 1-based row index is appended.
260
+ */
261
+ function interpolateName(
262
+ name: string,
263
+ row: TestEachRow,
264
+ index: number
265
+ ): string {
266
+ if (!name.includes("%")) {
267
+ return `${name} #${index + 1}`;
268
+ }
269
+ const replacements: Array<[RegExp, string]> = [
270
+ [/%i/g, String(index)],
271
+ [/%s/g, JSON.stringify(row.input)],
272
+ [/%j/g, JSON.stringify(row)],
273
+ ];
274
+ let out = name;
275
+ for (const [pattern, value] of replacements) {
276
+ out = out.replace(pattern, value);
277
+ }
278
+ return out;
279
+ }