@arizeai/phoenix-client 6.10.0 → 6.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (244) hide show
  1. package/README.md +62 -0
  2. package/dist/esm/__generated__/api/v1.d.ts +452 -28
  3. package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
  4. package/dist/esm/experiments/helpers/getExampleGlobalId.d.ts +8 -0
  5. package/dist/esm/experiments/helpers/getExampleGlobalId.d.ts.map +1 -0
  6. package/dist/esm/experiments/helpers/getExampleGlobalId.js +9 -0
  7. package/dist/esm/experiments/helpers/getExampleGlobalId.js.map +1 -0
  8. package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
  9. package/dist/esm/experiments/resumeEvaluation.js +2 -1
  10. package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
  11. package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
  12. package/dist/esm/experiments/resumeExperiment.js +3 -2
  13. package/dist/esm/experiments/resumeExperiment.js.map +1 -1
  14. package/dist/esm/experiments/runExperiment.d.ts.map +1 -1
  15. package/dist/esm/experiments/runExperiment.js +6 -3
  16. package/dist/esm/experiments/runExperiment.js.map +1 -1
  17. package/dist/esm/jest/index.d.ts +5 -0
  18. package/dist/esm/jest/index.d.ts.map +1 -0
  19. package/dist/esm/jest/index.js +49 -0
  20. package/dist/esm/jest/index.js.map +1 -0
  21. package/dist/esm/jest/reporter.d.ts +13 -0
  22. package/dist/esm/jest/reporter.d.ts.map +1 -0
  23. package/dist/esm/jest/reporter.js +19 -0
  24. package/dist/esm/jest/reporter.js.map +1 -0
  25. package/dist/esm/prompts/sdks/toAI.d.ts +2 -2
  26. package/dist/esm/prompts/sdks/toAI.d.ts.map +1 -1
  27. package/dist/esm/prompts/sdks/toAI.js.map +1 -1
  28. package/dist/esm/prompts/sdks/toAnthropic.d.ts +2 -2
  29. package/dist/esm/prompts/sdks/toAnthropic.d.ts.map +1 -1
  30. package/dist/esm/prompts/sdks/toAnthropic.js.map +1 -1
  31. package/dist/esm/prompts/sdks/toOpenAI.d.ts +2 -2
  32. package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
  33. package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
  34. package/dist/esm/prompts/sdks/toSDK.d.ts +8 -8
  35. package/dist/esm/prompts/sdks/toSDK.d.ts.map +1 -1
  36. package/dist/esm/prompts/sdks/toSDK.js.map +1 -1
  37. package/dist/esm/prompts/sdks/types.d.ts +2 -2
  38. package/dist/esm/prompts/sdks/types.d.ts.map +1 -1
  39. package/dist/esm/schemas/llm/anthropic/converters.d.ts +8 -8
  40. package/dist/esm/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  41. package/dist/esm/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  42. package/dist/esm/schemas/llm/constants.d.ts +3 -3
  43. package/dist/esm/schemas/llm/converters.d.ts +12 -12
  44. package/dist/esm/schemas/llm/openai/converters.d.ts +3 -3
  45. package/dist/esm/schemas/llm/schemas.d.ts +2 -2
  46. package/dist/esm/testing/acceptance.d.ts +20 -0
  47. package/dist/esm/testing/acceptance.d.ts.map +1 -0
  48. package/dist/esm/testing/acceptance.js +129 -0
  49. package/dist/esm/testing/acceptance.js.map +1 -0
  50. package/dist/esm/testing/define-api.d.ts +157 -0
  51. package/dist/esm/testing/define-api.d.ts.map +1 -0
  52. package/dist/esm/testing/define-api.js +78 -0
  53. package/dist/esm/testing/define-api.js.map +1 -0
  54. package/dist/esm/testing/helpers.d.ts +55 -0
  55. package/dist/esm/testing/helpers.d.ts.map +1 -0
  56. package/dist/esm/testing/helpers.js +179 -0
  57. package/dist/esm/testing/helpers.js.map +1 -0
  58. package/dist/esm/testing/phoenix-test-tracking.d.ts +68 -0
  59. package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -0
  60. package/dist/esm/testing/phoenix-test-tracking.js +521 -0
  61. package/dist/esm/testing/phoenix-test-tracking.js.map +1 -0
  62. package/dist/esm/testing/report-artifacts.d.ts +45 -0
  63. package/dist/esm/testing/report-artifacts.d.ts.map +1 -0
  64. package/dist/esm/testing/report-artifacts.js +218 -0
  65. package/dist/esm/testing/report-artifacts.js.map +1 -0
  66. package/dist/esm/testing/report-run.d.ts +22 -0
  67. package/dist/esm/testing/report-run.d.ts.map +1 -0
  68. package/dist/esm/testing/report-run.js +41 -0
  69. package/dist/esm/testing/report-run.js.map +1 -0
  70. package/dist/esm/testing/reporter-format.d.ts +83 -0
  71. package/dist/esm/testing/reporter-format.d.ts.map +1 -0
  72. package/dist/esm/testing/reporter-format.js +852 -0
  73. package/dist/esm/testing/reporter-format.js.map +1 -0
  74. package/dist/esm/testing/runner.d.ts +31 -0
  75. package/dist/esm/testing/runner.d.ts.map +1 -0
  76. package/dist/esm/testing/runner.js +238 -0
  77. package/dist/esm/testing/runner.js.map +1 -0
  78. package/dist/esm/testing/state.d.ts +138 -0
  79. package/dist/esm/testing/state.d.ts.map +1 -0
  80. package/dist/esm/testing/state.js +31 -0
  81. package/dist/esm/testing/state.js.map +1 -0
  82. package/dist/esm/testing/types.d.ts +319 -0
  83. package/dist/esm/testing/types.d.ts.map +1 -0
  84. package/dist/esm/testing/types.js +9 -0
  85. package/dist/esm/testing/types.js.map +1 -0
  86. package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
  87. package/dist/esm/utils/channel.d.ts +7 -7
  88. package/dist/esm/utils/channel.d.ts.map +1 -1
  89. package/dist/esm/utils/channel.js +1 -1
  90. package/dist/esm/utils/channel.js.map +1 -1
  91. package/dist/esm/utils/formatPromptMessages.d.ts.map +1 -1
  92. package/dist/esm/utils/getPromptBySelector.d.ts.map +1 -1
  93. package/dist/esm/utils/promisifyResult.d.ts +1 -1
  94. package/dist/esm/utils/promisifyResult.d.ts.map +1 -1
  95. package/dist/esm/utils/promisifyResult.js.map +1 -1
  96. package/dist/esm/utils/schemaMatches.d.ts +5 -5
  97. package/dist/esm/utils/schemaMatches.d.ts.map +1 -1
  98. package/dist/esm/utils/schemaMatches.js.map +1 -1
  99. package/dist/esm/vitest/index.d.ts +5 -0
  100. package/dist/esm/vitest/index.d.ts.map +1 -0
  101. package/dist/esm/vitest/index.js +15 -0
  102. package/dist/esm/vitest/index.js.map +1 -0
  103. package/dist/esm/vitest/reporter.d.ts +19 -0
  104. package/dist/esm/vitest/reporter.d.ts.map +1 -0
  105. package/dist/esm/vitest/reporter.js +27 -0
  106. package/dist/esm/vitest/reporter.js.map +1 -0
  107. package/dist/src/__generated__/api/v1.d.ts +452 -28
  108. package/dist/src/__generated__/api/v1.d.ts.map +1 -1
  109. package/dist/src/experiments/helpers/getExampleGlobalId.d.ts +8 -0
  110. package/dist/src/experiments/helpers/getExampleGlobalId.d.ts.map +1 -0
  111. package/dist/src/experiments/helpers/getExampleGlobalId.js +13 -0
  112. package/dist/src/experiments/helpers/getExampleGlobalId.js.map +1 -0
  113. package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
  114. package/dist/src/experiments/resumeEvaluation.js +2 -1
  115. package/dist/src/experiments/resumeEvaluation.js.map +1 -1
  116. package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
  117. package/dist/src/experiments/resumeExperiment.js +3 -2
  118. package/dist/src/experiments/resumeExperiment.js.map +1 -1
  119. package/dist/src/experiments/runExperiment.d.ts.map +1 -1
  120. package/dist/src/experiments/runExperiment.js +6 -3
  121. package/dist/src/experiments/runExperiment.js.map +1 -1
  122. package/dist/src/jest/index.d.ts +5 -0
  123. package/dist/src/jest/index.d.ts.map +1 -0
  124. package/dist/src/jest/index.js +58 -0
  125. package/dist/src/jest/index.js.map +1 -0
  126. package/dist/src/jest/reporter.d.ts +13 -0
  127. package/dist/src/jest/reporter.d.ts.map +1 -0
  128. package/dist/src/jest/reporter.js +23 -0
  129. package/dist/src/jest/reporter.js.map +1 -0
  130. package/dist/src/prompts/sdks/toAI.d.ts +2 -2
  131. package/dist/src/prompts/sdks/toAI.d.ts.map +1 -1
  132. package/dist/src/prompts/sdks/toAI.js.map +1 -1
  133. package/dist/src/prompts/sdks/toAnthropic.d.ts +2 -2
  134. package/dist/src/prompts/sdks/toAnthropic.d.ts.map +1 -1
  135. package/dist/src/prompts/sdks/toAnthropic.js.map +1 -1
  136. package/dist/src/prompts/sdks/toOpenAI.d.ts +2 -2
  137. package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
  138. package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
  139. package/dist/src/prompts/sdks/toSDK.d.ts +8 -8
  140. package/dist/src/prompts/sdks/toSDK.d.ts.map +1 -1
  141. package/dist/src/prompts/sdks/toSDK.js.map +1 -1
  142. package/dist/src/prompts/sdks/types.d.ts +2 -2
  143. package/dist/src/prompts/sdks/types.d.ts.map +1 -1
  144. package/dist/src/schemas/llm/anthropic/converters.d.ts +8 -8
  145. package/dist/src/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  146. package/dist/src/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  147. package/dist/src/schemas/llm/constants.d.ts +3 -3
  148. package/dist/src/schemas/llm/converters.d.ts +12 -12
  149. package/dist/src/schemas/llm/openai/converters.d.ts +3 -3
  150. package/dist/src/schemas/llm/schemas.d.ts +2 -2
  151. package/dist/src/testing/acceptance.d.ts +20 -0
  152. package/dist/src/testing/acceptance.d.ts.map +1 -0
  153. package/dist/src/testing/acceptance.js +114 -0
  154. package/dist/src/testing/acceptance.js.map +1 -0
  155. package/dist/src/testing/define-api.d.ts +157 -0
  156. package/dist/src/testing/define-api.d.ts.map +1 -0
  157. package/dist/src/testing/define-api.js +81 -0
  158. package/dist/src/testing/define-api.js.map +1 -0
  159. package/dist/src/testing/helpers.d.ts +55 -0
  160. package/dist/src/testing/helpers.d.ts.map +1 -0
  161. package/dist/src/testing/helpers.js +182 -0
  162. package/dist/src/testing/helpers.js.map +1 -0
  163. package/dist/src/testing/phoenix-test-tracking.d.ts +68 -0
  164. package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -0
  165. package/dist/src/testing/phoenix-test-tracking.js +530 -0
  166. package/dist/src/testing/phoenix-test-tracking.js.map +1 -0
  167. package/dist/src/testing/report-artifacts.d.ts +45 -0
  168. package/dist/src/testing/report-artifacts.d.ts.map +1 -0
  169. package/dist/src/testing/report-artifacts.js +225 -0
  170. package/dist/src/testing/report-artifacts.js.map +1 -0
  171. package/dist/src/testing/report-run.d.ts +22 -0
  172. package/dist/src/testing/report-run.d.ts.map +1 -0
  173. package/dist/src/testing/report-run.js +47 -0
  174. package/dist/src/testing/report-run.js.map +1 -0
  175. package/dist/src/testing/reporter-format.d.ts +83 -0
  176. package/dist/src/testing/reporter-format.d.ts.map +1 -0
  177. package/dist/src/testing/reporter-format.js +870 -0
  178. package/dist/src/testing/reporter-format.js.map +1 -0
  179. package/dist/src/testing/runner.d.ts +31 -0
  180. package/dist/src/testing/runner.d.ts.map +1 -0
  181. package/dist/src/testing/runner.js +258 -0
  182. package/dist/src/testing/runner.js.map +1 -0
  183. package/dist/src/testing/state.d.ts +138 -0
  184. package/dist/src/testing/state.d.ts.map +1 -0
  185. package/dist/src/testing/state.js +38 -0
  186. package/dist/src/testing/state.js.map +1 -0
  187. package/dist/src/testing/types.d.ts +319 -0
  188. package/dist/src/testing/types.d.ts.map +1 -0
  189. package/dist/src/testing/types.js +13 -0
  190. package/dist/src/testing/types.js.map +1 -0
  191. package/dist/src/utils/channel.d.ts +7 -7
  192. package/dist/src/utils/channel.d.ts.map +1 -1
  193. package/dist/src/utils/channel.js +1 -1
  194. package/dist/src/utils/channel.js.map +1 -1
  195. package/dist/src/utils/formatPromptMessages.d.ts.map +1 -1
  196. package/dist/src/utils/getPromptBySelector.d.ts.map +1 -1
  197. package/dist/src/utils/promisifyResult.d.ts +1 -1
  198. package/dist/src/utils/promisifyResult.d.ts.map +1 -1
  199. package/dist/src/utils/promisifyResult.js.map +1 -1
  200. package/dist/src/utils/schemaMatches.d.ts +5 -5
  201. package/dist/src/utils/schemaMatches.d.ts.map +1 -1
  202. package/dist/src/utils/schemaMatches.js.map +1 -1
  203. package/dist/src/vitest/index.d.ts +5 -0
  204. package/dist/src/vitest/index.d.ts.map +1 -0
  205. package/dist/src/vitest/index.js +23 -0
  206. package/dist/src/vitest/index.js.map +1 -0
  207. package/dist/src/vitest/reporter.d.ts +19 -0
  208. package/dist/src/vitest/reporter.d.ts.map +1 -0
  209. package/dist/src/vitest/reporter.js +34 -0
  210. package/dist/src/vitest/reporter.js.map +1 -0
  211. package/dist/tsconfig.tsbuildinfo +1 -1
  212. package/docs/ci-evals-annotations.mdx +190 -0
  213. package/docs/ci-evals-jest.mdx +78 -0
  214. package/docs/ci-evals-vitest.mdx +240 -0
  215. package/docs/ci-evals.mdx +263 -0
  216. package/docs/overview.mdx +9 -1
  217. package/package.json +49 -17
  218. package/src/__generated__/api/v1.ts +452 -28
  219. package/src/experiments/helpers/getExampleGlobalId.ts +12 -0
  220. package/src/experiments/resumeEvaluation.ts +2 -1
  221. package/src/experiments/resumeExperiment.ts +3 -2
  222. package/src/experiments/runExperiment.ts +6 -3
  223. package/src/jest/index.ts +124 -0
  224. package/src/jest/reporter.ts +22 -0
  225. package/src/prompts/sdks/toAI.ts +4 -3
  226. package/src/prompts/sdks/toAnthropic.ts +4 -3
  227. package/src/prompts/sdks/toOpenAI.ts +4 -3
  228. package/src/prompts/sdks/toSDK.ts +16 -11
  229. package/src/prompts/sdks/types.ts +2 -2
  230. package/src/testing/acceptance.ts +190 -0
  231. package/src/testing/define-api.ts +279 -0
  232. package/src/testing/helpers.ts +251 -0
  233. package/src/testing/phoenix-test-tracking.ts +637 -0
  234. package/src/testing/report-artifacts.ts +272 -0
  235. package/src/testing/report-run.ts +44 -0
  236. package/src/testing/reporter-format.ts +1072 -0
  237. package/src/testing/runner.ts +350 -0
  238. package/src/testing/state.ts +165 -0
  239. package/src/testing/types.ts +366 -0
  240. package/src/utils/channel.ts +17 -15
  241. package/src/utils/promisifyResult.ts +6 -4
  242. package/src/utils/schemaMatches.ts +12 -10
  243. package/src/vitest/index.ts +57 -0
  244. package/src/vitest/reporter.ts +32 -0
@@ -0,0 +1,190 @@
1
+ ---
2
+ title: "CI Eval Test Annotations"
3
+ description: "Record annotations and evaluator results on Phoenix experiment runs"
4
+ ---
5
+
6
+ CI eval test annotations are how the `@arizeai/phoenix-client/vitest` and
7
+ `@arizeai/phoenix-client/jest` submodules record scored, labeled, or
8
+ explained results on an experiment run. They map directly to Phoenix's
9
+ `experiment_evaluations` REST surface — each annotation becomes one
10
+ `ExperimentEvaluation` with an `ExperimentEvaluationResult` body.
11
+
12
+ ## The `Annotation` Shape
13
+
14
+ ```ts
15
+ interface Annotation {
16
+ /** Phoenix evaluation name. Required. */
17
+ name: string;
18
+ /** Numeric or boolean score. Booleans are stored as 0 / 1. */
19
+ score?: number | boolean | null;
20
+ /** Categorical label. */
21
+ label?: string | null;
22
+ /** Free-form explanation, shown in the Phoenix UI. */
23
+ explanation?: string | null;
24
+ /** Custom metadata. */
25
+ metadata?: Record<string, unknown>;
26
+ /** Source of the annotation. Defaults to "CODE". */
27
+ annotatorKind?: "LLM" | "CODE" | "HUMAN";
28
+ }
29
+ ```
30
+
31
+ The fields line up with Phoenix's
32
+ `ExperimentEvaluationResult` plus the `name` and `annotator_kind` carried
33
+ on the surrounding evaluation body.
34
+
35
+ ## `logAnnotation(annotation)`
36
+
37
+ Records a single annotation against the current run. Must be called
38
+ inside a `test()` body.
39
+
40
+ ```ts
41
+ import * as px from "@arizeai/phoenix-client/vitest";
42
+
43
+ px.describe("demo", () => {
44
+ px.test(
45
+ "manual annotation",
46
+ { input: { x: 1 }, expected: { y: 2 } },
47
+ async ({ input, expected }) => {
48
+ const result = myApp(input.x);
49
+ px.logOutput({ y: result });
50
+
51
+ px.logAnnotation({
52
+ name: "harmfulness",
53
+ score: 0.2,
54
+ explanation: "no PII detected",
55
+ annotatorKind: "CODE",
56
+ });
57
+ },
58
+ );
59
+ });
60
+ ```
61
+
62
+ ## `evaluate(evaluator, params?)`
63
+
64
+ Runs an evaluator object and records its result as an annotation on the
65
+ current run. An evaluator is any object with a `name` and an `evaluate`
66
+ function, including evaluators created with
67
+ `@arizeai/phoenix-evals.createEvaluator()` and
68
+ `@arizeai/phoenix-client/experiments.asExperimentEvaluator()`.
69
+ The evaluator call is traced as an OpenInference `EVALUATOR` span, and the
70
+ annotation is linked back to that evaluator trace.
71
+
72
+ If `params` is omitted, Phoenix supplies the current test's `input`,
73
+ recorded `output`, `expected`, `metadata`, and task `traceId`. If `params`
74
+ is supplied, it is merged on top of those defaults.
75
+
76
+ ```ts
77
+ import * as px from "@arizeai/phoenix-client/vitest";
78
+ import { createEvaluator } from "@arizeai/phoenix-evals";
79
+
80
+ const correctness = createEvaluator(
81
+ async ({ output, expected }: {
82
+ output: { sql: string };
83
+ expected: { sql: string };
84
+ }) => {
85
+ const grade = await llmAsJudge(output.sql, expected.sql);
86
+ return {
87
+ score: grade.score,
88
+ label: grade.passed ? "correct" : "incorrect",
89
+ explanation: grade.rationale,
90
+ };
91
+ },
92
+ { name: "correctness", kind: "LLM" },
93
+ );
94
+
95
+ px.describe("generate sql", () => {
96
+ px.test(
97
+ "select all",
98
+ {
99
+ input: { userQuery: "Get all users from the customers table" },
100
+ expected: { sql: "SELECT * FROM customers;" },
101
+ },
102
+ async ({ input, expected }) => {
103
+ const sql = await myApp(input.userQuery);
104
+ px.logOutput({ sql });
105
+ await px.evaluate(correctness, {
106
+ output: { sql },
107
+ expected: expected ?? { sql: "" },
108
+ });
109
+ },
110
+ );
111
+ });
112
+ ```
113
+
114
+ The annotation name comes from `evaluator.name`. The evaluator result can be
115
+ a number, boolean, string label, `null`, or an object with `score`, `label`,
116
+ `explanation`, and `metadata`.
117
+
118
+ ## Using `@arizeai/phoenix-evals`
119
+
120
+ `createEvaluator()` gives you a reusable evaluator object. `px.evaluate()`
121
+ runs that evaluator in the test context, traces the call, and records its
122
+ result on the experiment run. If the evaluator itself uses OpenInference
123
+ telemetry, those implementation spans appear under the evaluator trace:
124
+
125
+ ```ts
126
+ import * as px from "@arizeai/phoenix-client/vitest";
127
+ import { createEvaluator } from "@arizeai/phoenix-evals";
128
+
129
+ async function judgeHallucination(answer: string) {
130
+ return llmAsJudge({ answer });
131
+ }
132
+
133
+ const hallucination = createEvaluator(
134
+ async ({ output }: { output: { answer: string } }) => {
135
+ const result = await judgeHallucination(output.answer);
136
+ return {
137
+ score: result.score,
138
+ label: result.label,
139
+ explanation: result.explanation,
140
+ };
141
+ },
142
+ { name: "hallucination", kind: "LLM" },
143
+ );
144
+ ```
145
+
146
+ If you prefer to hand-write a plain evaluator, use the same shape:
147
+
148
+ ```ts
149
+ const exactMatch = {
150
+ name: "exact_match",
151
+ kind: "CODE" as const,
152
+ evaluate: ({ output, expected }: {
153
+ output: { sql: string };
154
+ expected: { sql: string };
155
+ }) => ({
156
+ score: output.sql === expected.sql,
157
+ }),
158
+ };
159
+ ```
160
+
161
+ Use `createEvaluator()` or OpenInference decorators from
162
+ `@arizeai/phoenix-otel` when the evaluator implementation should emit child
163
+ spans of its own. The older `traceEvaluator(fn)` helper remains available for
164
+ raw function wrapping, but evaluator objects are the preferred interface.
165
+
166
+ ## Built-In `pass` Annotation
167
+
168
+ Every test automatically records a `pass` boolean annotation based on
169
+ whether the test body threw. You don't have to log it yourself; it's
170
+ included in the reporter summary and on the run in Phoenix.
171
+
172
+ ## Aggregating Annotations In CI
173
+
174
+ Suite `acceptanceCriteria` aggregate annotation scores after all cases run.
175
+ Use them when a metric should clear a threshold across the dataset — either an
176
+ `average` bar (e.g. mean `correctness >= 0.8`) or a `passRate` rule requiring a
177
+ minimum fraction of runs to satisfy a per-run `passFn` predicate (e.g. 100% of
178
+ runs must have `valid_sql === true`). See
179
+ [CI Eval Tests: Vitest](./ci-evals-vitest#acceptance-criteria) for the full
180
+ configuration shape.
181
+
182
+ <section className="hidden" data-agent-context="source-map" aria-label="Source map">
183
+ <h2>Source Map</h2>
184
+ <ul>
185
+ <li><code>src/testing/helpers.ts</code></li>
186
+ <li><code>src/testing/acceptance.ts</code></li>
187
+ <li><code>src/testing/phoenix-test-tracking.ts</code></li>
188
+ <li><code>src/testing/types.ts</code></li>
189
+ </ul>
190
+ </section>
@@ -0,0 +1,78 @@
1
+ ---
2
+ title: "CI Eval Tests: Jest"
3
+ description: "Wire @arizeai/phoenix-client/jest into a Jest project"
4
+ ---
5
+
6
+ `@arizeai/phoenix-client/jest` exposes the same API as the Vitest
7
+ entrypoint but binds to Jest's globals (`describe`, `test`, `it`,
8
+ `beforeAll`, `afterAll`).
9
+
10
+ ## Setup
11
+
12
+ Create a separate `phoenix.jest.config.cjs`:
13
+
14
+ ```js
15
+ module.exports = {
16
+ testMatch: ["**/*.eval.?(c|m)[jt]s"],
17
+ reporters: ["default", "@arizeai/phoenix-client/jest/reporter"],
18
+ setupFiles: ["dotenv/config"],
19
+ testTimeout: 30000,
20
+ };
21
+ ```
22
+
23
+ - `testMatch` keeps eval suites separate from regular tests.
24
+ - `reporters` keeps Jest's default reporter and adds the Phoenix
25
+ summary block at the end of the run.
26
+ - `setupFiles: ["dotenv/config"]` loads `PHOENIX_HOST`, `PHOENIX_API_KEY`,
27
+ and other env vars from `.env`.
28
+ - `testTimeout` is bumped because LLM calls can be slow.
29
+
30
+ <Note>
31
+ The `jsdom` test environment is not supported. Either omit
32
+ `testEnvironment` or set it to `"node"`. For TypeScript or ESM use
33
+ `ts-jest` / `@swc/jest` per Jest's docs.
34
+ </Note>
35
+
36
+ Add a script to `package.json`:
37
+
38
+ ```json
39
+ {
40
+ "scripts": {
41
+ "eval": "jest --config phoenix.jest.config.cjs"
42
+ }
43
+ }
44
+ ```
45
+
46
+ ## API
47
+
48
+ ```ts
49
+ import * as px from "@arizeai/phoenix-client/jest";
50
+ ```
51
+
52
+ The exported names match the Vitest entrypoint exactly:
53
+
54
+ - `describe`, `describe.only`, `describe.skip`
55
+ - `test`, `test.only`, `test.skip`, `test.each`
56
+ - `it` (alias for `test`)
57
+ - `logOutput`, `logAnnotation`, `evaluate`
58
+
59
+ See [CI Eval Tests: Vitest](./ci-evals-vitest) for the full API reference — the surface is
60
+ identical. The only difference is the import path and the underlying
61
+ runner.
62
+
63
+ ## Notes
64
+
65
+ - This module reads Jest globals off `globalThis` at suite-declaration
66
+ time, so importing it outside of a Jest test run will throw with a
67
+ helpful error message.
68
+ - Jest's `--bail` flag will short-circuit the experiment; the Phoenix
69
+ reporter still prints a summary for the suites that ran.
70
+
71
+ <section className="hidden" data-agent-context="source-map" aria-label="Source map">
72
+ <h2>Source Map</h2>
73
+ <ul>
74
+ <li><code>src/jest/index.ts</code></li>
75
+ <li><code>src/jest/reporter.ts</code></li>
76
+ <li><code>src/testing/runner.ts</code></li>
77
+ </ul>
78
+ </section>
@@ -0,0 +1,240 @@
1
+ ---
2
+ title: "CI Eval Tests: Vitest"
3
+ description: "Wire @arizeai/phoenix-client/vitest into a Vitest project"
4
+ ---
5
+
6
+ `@arizeai/phoenix-client/vitest` ships a Vitest entrypoint plus an optional
7
+ reporter that prints a Phoenix-flavored summary at the end of the run.
8
+
9
+ ## Setup
10
+
11
+ Create a separate `phoenix.vitest.config.ts` so eval files don't get
12
+ swept into your normal unit-test config:
13
+
14
+ ```ts
15
+ import { defineConfig } from "vitest/config";
16
+
17
+ export default defineConfig({
18
+ test: {
19
+ include: ["**/*.eval.?(c|m)[jt]s"],
20
+ reporters: ["default", "@arizeai/phoenix-client/vitest/reporter"],
21
+ setupFiles: ["dotenv/config"],
22
+ testTimeout: 30000,
23
+ },
24
+ });
25
+ ```
26
+
27
+ - `include` keeps eval suites separate from unit tests by matching the
28
+ `*.eval.ts` convention.
29
+ - `reporters` keeps Vitest's default diagnostics and enables the Phoenix
30
+ summary block.
31
+ - `setupFiles: ["dotenv/config"]` loads `PHOENIX_HOST`, `PHOENIX_API_KEY`,
32
+ and any other env vars from `.env`.
33
+ - `testTimeout` is bumped because LLM calls can be slow.
34
+
35
+ <Note>
36
+ The `jsdom` test environment is not supported. Either omit `environment`
37
+ or set it to `"node"`.
38
+ </Note>
39
+
40
+ Add a script to `package.json`:
41
+
42
+ ```json
43
+ {
44
+ "scripts": {
45
+ "eval": "vitest run --config phoenix.vitest.config.ts"
46
+ }
47
+ }
48
+ ```
49
+
50
+ The script intentionally uses `vitest run` rather than watch mode —
51
+ many evaluators include longer-running LLM calls.
52
+
53
+ ## API
54
+
55
+ ```ts
56
+ import * as px from "@arizeai/phoenix-client/vitest";
57
+ ```
58
+
59
+ ### `describe(name, fn, config?)`
60
+
61
+ Declares a Phoenix test suite. The suite name is the dataset and
62
+ experiment name on the Phoenix server. `describe.only` and
63
+ `describe.skip` work like Vitest's variants.
64
+
65
+ ```ts
66
+ px.describe("my suite", () => { ... }, {
67
+ datasetName: "override-dataset-and-experiment-name",
68
+ description: "what this suite is for",
69
+ metadata: { model: "gpt-4o-mini" },
70
+ client: myCustomPhoenixClient, // overrides createClient()
71
+ repetitions: 3, // run each test in this suite 3x
72
+ acceptanceCriteria: [
73
+ { annotationName: "token_f1", metric: "average", threshold: 0.8 },
74
+ {
75
+ annotationName: "token_f1",
76
+ metric: "passRate",
77
+ passFn: (a) => typeof a.score === "number" && a.score >= 0.5,
78
+ minPassRate: 0.9,
79
+ },
80
+ ],
81
+ dryRun: true, // run this suite locally; sync nothing
82
+ });
83
+ ```
84
+
85
+ | Field | Type | Description |
86
+ |--------|------|-------------|
87
+ | `datasetName` | `string` | Override the dataset / experiment name (defaults to the suite name). |
88
+ | `description` | `string` | Description recorded on the dataset and experiment. |
89
+ | `metadata` | `Record<string, unknown>` | Suite-level metadata applied to the experiment. |
90
+ | `client` | `PhoenixClient` | Pre-configured `@arizeai/phoenix-client` instance. |
91
+ | `repetitions` | `number` | Run each test in the suite this many times (default `1`; `PHOENIX_TEST_REPETITIONS` overrides the default). Per-test `repetitions` wins. |
92
+ | `acceptanceCriteria` | `AcceptanceCriterion[]` | Aggregate annotation thresholds that fail the suite after all tests run. |
93
+ | `dryRun` | `boolean` | Run the whole suite locally — no dataset, experiment, runs, or annotations are created in Phoenix. Same effect as `PHOENIX_TEST_TRACING=false`, scoped to this suite. |
94
+
95
+ ### `test(name, params, fn, timeout?)`
96
+
97
+ Declares a single Phoenix test case. The `params` object carries the
98
+ Phoenix `Example` fields. `test.only`, `test.skip`, and `test.each`
99
+ mirror Vitest semantics. `it` is a re-export of `test`.
100
+
101
+ ```ts
102
+ px.test(
103
+ "a case",
104
+ {
105
+ input: { userQuery: "..." },
106
+ expected: { sql: "..." },
107
+ metadata: { hard: true },
108
+ id: "stable-example-id",
109
+ },
110
+ async ({ input, expected, metadata }) => {
111
+ const sql = await myApp(input.userQuery);
112
+ px.logOutput({ sql });
113
+ expect(sql).toEqual(expected?.sql);
114
+ },
115
+ );
116
+ ```
117
+
118
+ | Param field | Maps to |
119
+ |--------|---------|
120
+ | `input` | `Example.input` |
121
+ | `expected` | `Example.output` (reference) |
122
+ | `metadata` | `Example.metadata` |
123
+ | `id` | `Example.id` (stable upsert id) |
124
+ | `repetitions` | Number of runs against this example (overrides the suite value). Reported as `"<name> [rep i/N]"`. |
125
+ | `dryRun` | When `true`, this case runs as an ordinary local test — no dataset example, no run, nothing uploaded — even if the suite syncs. |
126
+
127
+ ### `test.each(table)(name, fn, timeout?)`
128
+
129
+ Run the same test body across many examples.
130
+
131
+ ```ts
132
+ const DATASET = [
133
+ { input: { userQuery: "whats up" }, expected: { sql: "n/a" } },
134
+ { input: { userQuery: "how are you?" }, expected: { sql: "n/a" } },
135
+ ];
136
+
137
+ px.describe("offtopic inputs", () => {
138
+ px.test.each(DATASET)("offtopic input", async ({ input, expected }) => {
139
+ const sql = await myApp(input.userQuery);
140
+ px.logOutput({ sql });
141
+ });
142
+ });
143
+ ```
144
+
145
+ The name template supports `%i`, `%s`, and `%j` for parity with Vitest's
146
+ `test.each`. Without a placeholder the row index is appended.
147
+
148
+ ### Logging
149
+
150
+ - `px.logOutput(value)` records the actual `output` for the run.
151
+ - `px.logAnnotation({ name, score, ... })` records an annotation.
152
+ - `px.evaluate(evaluator, params?)` runs an evaluator object and records its
153
+ result as an annotation linked to the evaluator trace. Evaluators can come
154
+ from `@arizeai/phoenix-evals.createEvaluator()`, `asExperimentEvaluator()`,
155
+ or any plain `{ name, evaluate }` object.
156
+
157
+ See [CI Eval Test Annotations](./ci-evals-annotations) for the full annotation shape.
158
+
159
+ ### Acceptance Criteria
160
+
161
+ Use `acceptanceCriteria` to gate the suite on aggregate annotation scores in
162
+ CI. Criteria run *after* the suite finishes, so all cases still execute and the
163
+ reporter shows the full scorecard before failing. Each criterion aggregates one
164
+ annotation (by `annotationName`) with one `metric`:
165
+
166
+ - `metric: "average"` — gate on overall quality: the mean score across all runs
167
+ must clear `threshold` (compared in `direction`).
168
+ - `metric: "passRate"` — gate on consistency: each run passes when its `passFn`
169
+ predicate returns `true`, and the suite passes when the fraction of passing
170
+ runs is at least `minPassRate` (e.g. `minPassRate: 0.9` ⇒ 90% must pass).
171
+
172
+ `passFn` receives the run's annotation and returns a boolean, so it can express
173
+ any pass rule — a score bar, a range, a label match, a metadata check, etc.
174
+
175
+ ```ts
176
+ px.describe("text-to-sql scorecard", () => {
177
+ // tests log token_f1 (0–1), valid_sql (boolean), and latency_ms annotations
178
+ }, {
179
+ acceptanceCriteria: [
180
+ // overall quality: the mean token_f1 must be >= 0.8
181
+ { annotationName: "token_f1", metric: "average", threshold: 0.8 },
182
+ // consistency: at least 90% of runs must score >= 0.7 on token_f1
183
+ {
184
+ annotationName: "token_f1",
185
+ metric: "passRate",
186
+ passFn: (a) => typeof a.score === "number" && a.score >= 0.7,
187
+ minPassRate: 0.9,
188
+ },
189
+ // hard floor: every run must produce valid SQL (boolean must be true)
190
+ {
191
+ annotationName: "valid_sql",
192
+ metric: "passRate",
193
+ passFn: (a) => a.score === true,
194
+ minPassRate: 1,
195
+ },
196
+ // budget: lower is better, so the mean latency must stay <= 800ms
197
+ {
198
+ annotationName: "latency_ms",
199
+ metric: "average",
200
+ threshold: 800,
201
+ direction: "minimize",
202
+ },
203
+ ],
204
+ });
205
+ ```
206
+
207
+ | Field | Description |
208
+ |-------|-------------|
209
+ | `annotationName` | Annotation to aggregate. If a run logs the same annotation more than once, the last one counts. |
210
+ | `metric` | `"average"` checks the mean of all numeric/boolean scores against `threshold` (tolerates weak runs if the mean holds); `"passRate"` counts runs whose `passFn` returns `true` and requires that **fraction** to reach `minPassRate`. |
211
+ | `threshold` | **`average` only.** The bar the mean must clear (in `direction`). |
212
+ | `direction` | **`average` only.** `"maximize"` (default; higher mean is better, clears `threshold` when `>=`) or `"minimize"` (lower is better, clears when `<=`). Use `"minimize"` for cost, latency, or error rates. |
213
+ | `passFn` | **`passRate` only.** `(annotation) => boolean` predicate deciding whether a single run passes, given its last annotation for `annotationName` (`score`, `label`, `explanation`, `metadata`, …). |
214
+ | `minPassRate` | **`passRate` only.** Minimum fraction of runs (`0`–`1`) that must pass for the suite to pass (`1` = all). The suite passes when `passRate >= minPassRate`. |
215
+
216
+ **Edge cases.** An `average` criterion with no numeric/boolean scores — or a
217
+ `passRate` criterion whose annotation was never logged — fails rather than
218
+ passing vacuously. Skipped tests are excluded from the aggregate; dry-run tests
219
+ are included because they still execute locally.
220
+
221
+ In the reporter's `Acceptance Criteria` block the reported value is the **mean**
222
+ for `average` and the **fraction of runs that passed** for `passRate` (so a
223
+ fully-passing `passRate` criterion reads `1.000`).
224
+
225
+ ## Reporter Output
226
+
227
+ When `@arizeai/phoenix-client/vitest/reporter` is loaded, the runner prints
228
+ a per-suite block at the end of the run with pass/fail counts,
229
+ annotation aggregates, acceptance criteria, and links to the Phoenix dataset and experiment.
230
+ The default Vitest reporter still runs alongside it.
231
+
232
+ <section className="hidden" data-agent-context="source-map" aria-label="Source map">
233
+ <h2>Source Map</h2>
234
+ <ul>
235
+ <li><code>src/vitest/index.ts</code></li>
236
+ <li><code>src/vitest/reporter.ts</code></li>
237
+ <li><code>src/testing/runner.ts</code></li>
238
+ <li><code>src/testing/acceptance.ts</code></li>
239
+ </ul>
240
+ </section>