@arizeai/phoenix-client 6.10.1 → 6.11.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (214) hide show
  1. package/README.md +62 -0
  2. package/dist/esm/__generated__/api/v1.d.ts +254 -2
  3. package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
  4. package/dist/esm/jest/index.d.ts +5 -0
  5. package/dist/esm/jest/index.d.ts.map +1 -0
  6. package/dist/esm/jest/index.js +49 -0
  7. package/dist/esm/jest/index.js.map +1 -0
  8. package/dist/esm/jest/reporter.d.ts +13 -0
  9. package/dist/esm/jest/reporter.d.ts.map +1 -0
  10. package/dist/esm/jest/reporter.js +19 -0
  11. package/dist/esm/jest/reporter.js.map +1 -0
  12. package/dist/esm/prompts/sdks/toAI.d.ts +2 -2
  13. package/dist/esm/prompts/sdks/toAI.d.ts.map +1 -1
  14. package/dist/esm/prompts/sdks/toAI.js.map +1 -1
  15. package/dist/esm/prompts/sdks/toAnthropic.d.ts +2 -2
  16. package/dist/esm/prompts/sdks/toAnthropic.d.ts.map +1 -1
  17. package/dist/esm/prompts/sdks/toAnthropic.js.map +1 -1
  18. package/dist/esm/prompts/sdks/toOpenAI.d.ts +2 -2
  19. package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
  20. package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
  21. package/dist/esm/prompts/sdks/toSDK.d.ts +8 -8
  22. package/dist/esm/prompts/sdks/toSDK.d.ts.map +1 -1
  23. package/dist/esm/prompts/sdks/toSDK.js.map +1 -1
  24. package/dist/esm/prompts/sdks/types.d.ts +2 -2
  25. package/dist/esm/prompts/sdks/types.d.ts.map +1 -1
  26. package/dist/esm/schemas/llm/anthropic/converters.d.ts +8 -8
  27. package/dist/esm/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  28. package/dist/esm/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  29. package/dist/esm/schemas/llm/constants.d.ts +3 -3
  30. package/dist/esm/schemas/llm/converters.d.ts +12 -12
  31. package/dist/esm/schemas/llm/openai/converters.d.ts +3 -3
  32. package/dist/esm/schemas/llm/schemas.d.ts +2 -2
  33. package/dist/esm/testing/acceptance.d.ts +20 -0
  34. package/dist/esm/testing/acceptance.d.ts.map +1 -0
  35. package/dist/esm/testing/acceptance.js +129 -0
  36. package/dist/esm/testing/acceptance.js.map +1 -0
  37. package/dist/esm/testing/define-api.d.ts +157 -0
  38. package/dist/esm/testing/define-api.d.ts.map +1 -0
  39. package/dist/esm/testing/define-api.js +78 -0
  40. package/dist/esm/testing/define-api.js.map +1 -0
  41. package/dist/esm/testing/helpers.d.ts +55 -0
  42. package/dist/esm/testing/helpers.d.ts.map +1 -0
  43. package/dist/esm/testing/helpers.js +179 -0
  44. package/dist/esm/testing/helpers.js.map +1 -0
  45. package/dist/esm/testing/phoenix-test-tracking.d.ts +68 -0
  46. package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -0
  47. package/dist/esm/testing/phoenix-test-tracking.js +521 -0
  48. package/dist/esm/testing/phoenix-test-tracking.js.map +1 -0
  49. package/dist/esm/testing/report-artifacts.d.ts +45 -0
  50. package/dist/esm/testing/report-artifacts.d.ts.map +1 -0
  51. package/dist/esm/testing/report-artifacts.js +218 -0
  52. package/dist/esm/testing/report-artifacts.js.map +1 -0
  53. package/dist/esm/testing/report-run.d.ts +22 -0
  54. package/dist/esm/testing/report-run.d.ts.map +1 -0
  55. package/dist/esm/testing/report-run.js +41 -0
  56. package/dist/esm/testing/report-run.js.map +1 -0
  57. package/dist/esm/testing/reporter-format.d.ts +83 -0
  58. package/dist/esm/testing/reporter-format.d.ts.map +1 -0
  59. package/dist/esm/testing/reporter-format.js +852 -0
  60. package/dist/esm/testing/reporter-format.js.map +1 -0
  61. package/dist/esm/testing/runner.d.ts +31 -0
  62. package/dist/esm/testing/runner.d.ts.map +1 -0
  63. package/dist/esm/testing/runner.js +238 -0
  64. package/dist/esm/testing/runner.js.map +1 -0
  65. package/dist/esm/testing/state.d.ts +138 -0
  66. package/dist/esm/testing/state.d.ts.map +1 -0
  67. package/dist/esm/testing/state.js +31 -0
  68. package/dist/esm/testing/state.js.map +1 -0
  69. package/dist/esm/testing/types.d.ts +319 -0
  70. package/dist/esm/testing/types.d.ts.map +1 -0
  71. package/dist/esm/testing/types.js +9 -0
  72. package/dist/esm/testing/types.js.map +1 -0
  73. package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
  74. package/dist/esm/utils/channel.d.ts +7 -7
  75. package/dist/esm/utils/channel.d.ts.map +1 -1
  76. package/dist/esm/utils/channel.js +1 -1
  77. package/dist/esm/utils/channel.js.map +1 -1
  78. package/dist/esm/utils/formatPromptMessages.d.ts.map +1 -1
  79. package/dist/esm/utils/getPromptBySelector.d.ts.map +1 -1
  80. package/dist/esm/utils/promisifyResult.d.ts +1 -1
  81. package/dist/esm/utils/promisifyResult.d.ts.map +1 -1
  82. package/dist/esm/utils/promisifyResult.js.map +1 -1
  83. package/dist/esm/utils/schemaMatches.d.ts +5 -5
  84. package/dist/esm/utils/schemaMatches.d.ts.map +1 -1
  85. package/dist/esm/utils/schemaMatches.js.map +1 -1
  86. package/dist/esm/vitest/index.d.ts +5 -0
  87. package/dist/esm/vitest/index.d.ts.map +1 -0
  88. package/dist/esm/vitest/index.js +15 -0
  89. package/dist/esm/vitest/index.js.map +1 -0
  90. package/dist/esm/vitest/reporter.d.ts +19 -0
  91. package/dist/esm/vitest/reporter.d.ts.map +1 -0
  92. package/dist/esm/vitest/reporter.js +27 -0
  93. package/dist/esm/vitest/reporter.js.map +1 -0
  94. package/dist/src/__generated__/api/v1.d.ts +254 -2
  95. package/dist/src/__generated__/api/v1.d.ts.map +1 -1
  96. package/dist/src/jest/index.d.ts +5 -0
  97. package/dist/src/jest/index.d.ts.map +1 -0
  98. package/dist/src/jest/index.js +58 -0
  99. package/dist/src/jest/index.js.map +1 -0
  100. package/dist/src/jest/reporter.d.ts +13 -0
  101. package/dist/src/jest/reporter.d.ts.map +1 -0
  102. package/dist/src/jest/reporter.js +23 -0
  103. package/dist/src/jest/reporter.js.map +1 -0
  104. package/dist/src/prompts/sdks/toAI.d.ts +2 -2
  105. package/dist/src/prompts/sdks/toAI.d.ts.map +1 -1
  106. package/dist/src/prompts/sdks/toAI.js.map +1 -1
  107. package/dist/src/prompts/sdks/toAnthropic.d.ts +2 -2
  108. package/dist/src/prompts/sdks/toAnthropic.d.ts.map +1 -1
  109. package/dist/src/prompts/sdks/toAnthropic.js.map +1 -1
  110. package/dist/src/prompts/sdks/toOpenAI.d.ts +2 -2
  111. package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
  112. package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
  113. package/dist/src/prompts/sdks/toSDK.d.ts +8 -8
  114. package/dist/src/prompts/sdks/toSDK.d.ts.map +1 -1
  115. package/dist/src/prompts/sdks/toSDK.js.map +1 -1
  116. package/dist/src/prompts/sdks/types.d.ts +2 -2
  117. package/dist/src/prompts/sdks/types.d.ts.map +1 -1
  118. package/dist/src/schemas/llm/anthropic/converters.d.ts +8 -8
  119. package/dist/src/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  120. package/dist/src/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  121. package/dist/src/schemas/llm/constants.d.ts +3 -3
  122. package/dist/src/schemas/llm/converters.d.ts +12 -12
  123. package/dist/src/schemas/llm/openai/converters.d.ts +3 -3
  124. package/dist/src/schemas/llm/schemas.d.ts +2 -2
  125. package/dist/src/testing/acceptance.d.ts +20 -0
  126. package/dist/src/testing/acceptance.d.ts.map +1 -0
  127. package/dist/src/testing/acceptance.js +114 -0
  128. package/dist/src/testing/acceptance.js.map +1 -0
  129. package/dist/src/testing/define-api.d.ts +157 -0
  130. package/dist/src/testing/define-api.d.ts.map +1 -0
  131. package/dist/src/testing/define-api.js +81 -0
  132. package/dist/src/testing/define-api.js.map +1 -0
  133. package/dist/src/testing/helpers.d.ts +55 -0
  134. package/dist/src/testing/helpers.d.ts.map +1 -0
  135. package/dist/src/testing/helpers.js +182 -0
  136. package/dist/src/testing/helpers.js.map +1 -0
  137. package/dist/src/testing/phoenix-test-tracking.d.ts +68 -0
  138. package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -0
  139. package/dist/src/testing/phoenix-test-tracking.js +530 -0
  140. package/dist/src/testing/phoenix-test-tracking.js.map +1 -0
  141. package/dist/src/testing/report-artifacts.d.ts +45 -0
  142. package/dist/src/testing/report-artifacts.d.ts.map +1 -0
  143. package/dist/src/testing/report-artifacts.js +225 -0
  144. package/dist/src/testing/report-artifacts.js.map +1 -0
  145. package/dist/src/testing/report-run.d.ts +22 -0
  146. package/dist/src/testing/report-run.d.ts.map +1 -0
  147. package/dist/src/testing/report-run.js +47 -0
  148. package/dist/src/testing/report-run.js.map +1 -0
  149. package/dist/src/testing/reporter-format.d.ts +83 -0
  150. package/dist/src/testing/reporter-format.d.ts.map +1 -0
  151. package/dist/src/testing/reporter-format.js +870 -0
  152. package/dist/src/testing/reporter-format.js.map +1 -0
  153. package/dist/src/testing/runner.d.ts +31 -0
  154. package/dist/src/testing/runner.d.ts.map +1 -0
  155. package/dist/src/testing/runner.js +258 -0
  156. package/dist/src/testing/runner.js.map +1 -0
  157. package/dist/src/testing/state.d.ts +138 -0
  158. package/dist/src/testing/state.d.ts.map +1 -0
  159. package/dist/src/testing/state.js +38 -0
  160. package/dist/src/testing/state.js.map +1 -0
  161. package/dist/src/testing/types.d.ts +319 -0
  162. package/dist/src/testing/types.d.ts.map +1 -0
  163. package/dist/src/testing/types.js +13 -0
  164. package/dist/src/testing/types.js.map +1 -0
  165. package/dist/src/utils/channel.d.ts +7 -7
  166. package/dist/src/utils/channel.d.ts.map +1 -1
  167. package/dist/src/utils/channel.js +1 -1
  168. package/dist/src/utils/channel.js.map +1 -1
  169. package/dist/src/utils/formatPromptMessages.d.ts.map +1 -1
  170. package/dist/src/utils/getPromptBySelector.d.ts.map +1 -1
  171. package/dist/src/utils/promisifyResult.d.ts +1 -1
  172. package/dist/src/utils/promisifyResult.d.ts.map +1 -1
  173. package/dist/src/utils/promisifyResult.js.map +1 -1
  174. package/dist/src/utils/schemaMatches.d.ts +5 -5
  175. package/dist/src/utils/schemaMatches.d.ts.map +1 -1
  176. package/dist/src/utils/schemaMatches.js.map +1 -1
  177. package/dist/src/vitest/index.d.ts +5 -0
  178. package/dist/src/vitest/index.d.ts.map +1 -0
  179. package/dist/src/vitest/index.js +23 -0
  180. package/dist/src/vitest/index.js.map +1 -0
  181. package/dist/src/vitest/reporter.d.ts +19 -0
  182. package/dist/src/vitest/reporter.d.ts.map +1 -0
  183. package/dist/src/vitest/reporter.js +34 -0
  184. package/dist/src/vitest/reporter.js.map +1 -0
  185. package/dist/tsconfig.tsbuildinfo +1 -1
  186. package/docs/ci-evals-annotations.mdx +190 -0
  187. package/docs/ci-evals-jest.mdx +78 -0
  188. package/docs/ci-evals-vitest.mdx +240 -0
  189. package/docs/ci-evals.mdx +263 -0
  190. package/docs/overview.mdx +9 -1
  191. package/package.json +49 -17
  192. package/src/__generated__/api/v1.ts +254 -2
  193. package/src/jest/index.ts +124 -0
  194. package/src/jest/reporter.ts +22 -0
  195. package/src/prompts/sdks/toAI.ts +4 -3
  196. package/src/prompts/sdks/toAnthropic.ts +4 -3
  197. package/src/prompts/sdks/toOpenAI.ts +4 -3
  198. package/src/prompts/sdks/toSDK.ts +16 -11
  199. package/src/prompts/sdks/types.ts +2 -2
  200. package/src/testing/acceptance.ts +190 -0
  201. package/src/testing/define-api.ts +279 -0
  202. package/src/testing/helpers.ts +251 -0
  203. package/src/testing/phoenix-test-tracking.ts +637 -0
  204. package/src/testing/report-artifacts.ts +272 -0
  205. package/src/testing/report-run.ts +44 -0
  206. package/src/testing/reporter-format.ts +1072 -0
  207. package/src/testing/runner.ts +350 -0
  208. package/src/testing/state.ts +165 -0
  209. package/src/testing/types.ts +366 -0
  210. package/src/utils/channel.ts +17 -15
  211. package/src/utils/promisifyResult.ts +6 -4
  212. package/src/utils/schemaMatches.ts +12 -10
  213. package/src/vitest/index.ts +57 -0
  214. package/src/vitest/reporter.ts +32 -0
@@ -0,0 +1,263 @@
1
+ ---
2
+ title: "CI Eval Tests"
3
+ description: "Run dataset-backed Phoenix evaluations as Vitest or Jest tests"
4
+ ---
5
+
6
+ `@arizeai/phoenix-client/vitest` and `@arizeai/phoenix-client/jest` let
7
+ you write evaluations as ordinary Vitest or Jest tests that fit cleanly
8
+ into local development and CI. Each `describe()`
9
+ block becomes a Phoenix dataset and a new experiment; each `test()`
10
+ becomes a dataset example plus a recorded experiment run; the assertion
11
+ outcome is captured as a `pass` boolean annotation. Anything you log via
12
+ `logOutput()` / `logAnnotation()` / `evaluate()` lands on the run.
13
+ Suite-level `acceptanceCriteria` can fail CI on aggregate annotation metrics,
14
+ for example when average quality drops below `0.8`.
15
+
16
+ Tracing is provided by `@arizeai/phoenix-otel` (OpenInference). LLM and
17
+ agent calls instrumented with OpenInference appear as child spans of
18
+ each test's task span.
19
+
20
+ ## Install
21
+
22
+ ```bash
23
+ npm install -D @arizeai/phoenix-client @arizeai/phoenix-evals vitest dotenv
24
+ # or, for jest:
25
+ npm install -D @arizeai/phoenix-client @arizeai/phoenix-evals jest dotenv
26
+ ```
27
+
28
+ ## Minimal Example
29
+
30
+ ```ts
31
+ import * as px from "@arizeai/phoenix-client/vitest";
32
+ import { expect } from "vitest";
33
+
34
+ px.describe("generate sql demo", () => {
35
+ px.test(
36
+ "generates select all",
37
+ {
38
+ input: { userQuery: "Get all users from the customers table" },
39
+ expected: { sql: "SELECT * FROM customers;" },
40
+ },
41
+ async ({ input, expected }) => {
42
+ const sql = await myApp(input.userQuery);
43
+ px.logOutput({ sql });
44
+ expect(sql).toEqual(expected?.sql);
45
+ },
46
+ );
47
+ });
48
+ ```
49
+
50
+ ## Docs And Source In `node_modules`
51
+
52
+ After install, a coding agent can inspect the installed package directly:
53
+
54
+ ```text
55
+ node_modules/@arizeai/phoenix-client/docs/
56
+ node_modules/@arizeai/phoenix-client/src/
57
+ ```
58
+
59
+ That gives the agent version-matched docs plus the exact implementation
60
+ that shipped with your project.
61
+
62
+ ## Module Map
63
+
64
+ | Import | Purpose |
65
+ |--------|---------|
66
+ | `@arizeai/phoenix-client/vitest` | Vitest entrypoint (`describe`, `test`, `it`, `logOutput`, `logAnnotation`, `evaluate`) |
67
+ | `@arizeai/phoenix-client/vitest/reporter` | Vitest reporter that prints a Phoenix-flavored summary at the end of the run |
68
+ | `@arizeai/phoenix-client/jest` | Same API surface, wired to Jest globals |
69
+ | `@arizeai/phoenix-client/jest/reporter` | Jest reporter |
70
+
71
+ ## Phoenix Terminology
72
+
73
+ The public API uses Phoenix terms end-to-end — what you read in this package
74
+ matches what shows up in the Phoenix UI and the REST API.
75
+
76
+ | Test field | Phoenix concept |
77
+ | --------------------------- | ---------------------------------------- |
78
+ | `input` | `Example.input` |
79
+ | `expected` | `Example.output` (reference) |
80
+ | `metadata` | `Example.metadata` |
81
+ | `id` | `Example.id` (stable upsert id) |
82
+ | `logOutput(value)` | `ExperimentRun.output` |
83
+ | `Annotation.name` | `ExperimentEvaluation.name` |
84
+ | `Annotation.score` | `ExperimentEvaluationResult.score` |
85
+ | `Annotation.label` | `ExperimentEvaluationResult.label` |
86
+ | `Annotation.explanation` | `ExperimentEvaluationResult.explanation` |
87
+ | `Annotation.annotatorKind` | `annotator_kind` (`LLM` / `CODE` / `HUMAN`) |
88
+
89
+ ## Configuration
90
+
91
+ The Vitest and Jest submodules reuse the standard `@arizeai/phoenix-client`
92
+ and `@arizeai/phoenix-otel` configuration, so setup goes through the
93
+ standard Phoenix env vars.
94
+
95
+ | Variable | Purpose |
96
+ | --- | --- |
97
+ | `PHOENIX_HOST` | Phoenix base URL |
98
+ | `PHOENIX_API_KEY` | Bearer token for Phoenix |
99
+ | `PHOENIX_CLIENT_HEADERS` | Optional JSON headers forwarded to the Phoenix client and tracer |
100
+ | `PHOENIX_TEST_TRACKING=false` | Disable sync to Phoenix for the current run (tracing is on by default) |
101
+ | `PHOENIX_TEST_REPETITIONS` | Default number of times to run each test |
102
+ | `PHOENIX_TEST_REPORTER=verbose` | Show every test row plus per-test `output:` detail (default is the compact view) |
103
+ | `PHOENIX_TEST_REPORTER_MAX_ROWS` | Max test rows shown per suite in compact mode (default `10`; failures are never hidden) |
104
+ | `PHOENIX_TEST_COLOR` | Force ANSI color on/off (otherwise auto: on for a TTY, off in CI / `NO_COLOR`) |
105
+
106
+ `describe()` also accepts `repetitions` and `dryRun` on its config object,
107
+ plus `acceptanceCriteria` for aggregate score thresholds. `test()` accepts
108
+ `repetitions` and `dryRun` on its params — see
109
+ [CI Eval Tests: Vitest](./ci-evals-vitest) /
110
+ [CI Eval Tests: Jest](./ci-evals-jest).
111
+
112
+ ## Repetitions
113
+
114
+ Run a test (or a whole suite) more than once to measure non-determinism.
115
+ Each repetition is a separate experiment run against the same dataset
116
+ example, carrying its own `repetition_number`, so the Phoenix compare view
117
+ lines them up. Resolution order: per-test `repetitions` → suite
118
+ `repetitions` → `PHOENIX_TEST_REPETITIONS` → `1`.
119
+
120
+ ## Dry-Run Mode
121
+
122
+ Dry-run executes test bodies (and tracing, when a tracer is attached) but
123
+ creates no dataset, experiment, run, or annotations in Phoenix. The
124
+ reporter still prints a local summary.
125
+
126
+ - **Whole process** — `PHOENIX_TEST_TRACKING=false`.
127
+ - **One suite** — `describe(name, fn, { dryRun: true })`.
128
+ - **One test** — `test(name, { input, dryRun: true }, fn)`; that case runs
129
+ as an ordinary local test, with no dataset example and nothing uploaded,
130
+ even when the rest of the suite syncs.
131
+
132
+ ## Reporter Output
133
+
134
+ The Vitest and Jest reporters print a Phoenix-flavored summary at the end of a
135
+ run. By default the output is **compact** and scales to large suites:
136
+
137
+ - A **scoreboard** across all suites — passed count, the gated metric average,
138
+ the acceptance verdict, and the experiment link, one row per suite.
139
+ - A per-suite **results table** showing only failures and evaluator *misses*
140
+ (a run whose annotation score falls below its acceptance bar), with an
141
+ `AGGREGATE` row over the whole suite. Passing rows are hidden behind a
142
+ `… N passing rows hidden` footer — their full detail (input, output,
143
+ annotations) is always written to the JSON artifacts (see
144
+ `PHOENIX_TEST_REPORT_DIR`).
145
+
146
+ Set `PHOENIX_TEST_REPORTER=verbose` to expand every test row and restore the
147
+ per-test `output:` detail block. `PHOENIX_TEST_REPORTER_MAX_ROWS` caps the rows
148
+ shown per suite in compact mode (failures are never hidden).
149
+
150
+ ## Acceptance Criteria
151
+
152
+ Acceptance criteria turn a suite into a CI gate. After every test runs, Phoenix
153
+ aggregates the annotation scores you logged and fails the suite if any criterion
154
+ misses its bar. Because they run *after* all tests, every case still executes
155
+ and the reporter prints the full scorecard before failing — you see every
156
+ regression in one run, not just the first.
157
+
158
+ Each criterion aggregates one annotation (by `annotationName`) with one
159
+ `metric`:
160
+
161
+ - **`average`** — gate on overall quality. The mean score across all runs must
162
+ clear `threshold` (compared in `direction`). A few weak runs are tolerated as
163
+ long as the mean holds.
164
+ - **`passRate`** — gate on consistency. Each run *passes* when its `passFn`
165
+ predicate returns `true`, and the suite passes when the **fraction** of
166
+ passing runs is at least `minPassRate` (e.g. `minPassRate: 0.9` ⇒ 90% must
167
+ pass; `minPassRate: 1` ⇒ all of them).
168
+
169
+ `passFn` receives the run's annotation (`score`, `label`, `explanation`,
170
+ `metadata`, …) and returns a boolean, so it can express any pass rule — a score
171
+ bar, a score range, a label match, a metadata check, etc.
172
+
173
+ ```ts
174
+ px.describe("text-to-sql", () => {
175
+ // each test logs `token_f1` (0–1), `valid_sql` (boolean), and `latency_ms`
176
+ }, {
177
+ acceptanceCriteria: [
178
+ // overall quality: the mean token_f1 across the suite must be >= 0.8
179
+ { annotationName: "token_f1", metric: "average", threshold: 0.8 },
180
+ // consistency: at least 90% of runs must score >= 0.7 on token_f1
181
+ {
182
+ annotationName: "token_f1",
183
+ metric: "passRate",
184
+ passFn: (a) => typeof a.score === "number" && a.score >= 0.7,
185
+ minPassRate: 0.9,
186
+ },
187
+ // hard floor: every run must produce valid SQL (boolean must be true)
188
+ {
189
+ annotationName: "valid_sql",
190
+ metric: "passRate",
191
+ passFn: (a) => a.score === true,
192
+ minPassRate: 1,
193
+ },
194
+ // budget: lower is better, so the mean latency must stay <= 800ms
195
+ {
196
+ annotationName: "latency_ms",
197
+ metric: "average",
198
+ threshold: 800,
199
+ direction: "minimize",
200
+ },
201
+ ],
202
+ });
203
+ ```
204
+
205
+ ### Direction
206
+
207
+ `direction` applies to the `average` metric only — `passRate` encodes its own
208
+ comparison inside `passFn`:
209
+
210
+ - `"maximize"` (default) — higher is better; the mean clears `threshold` when it
211
+ is `>=` it.
212
+ - `"minimize"` — lower is better; the mean clears `threshold` when it is `<=`
213
+ it. Use it for cost, latency, or error-rate annotations.
214
+
215
+ ### Scoring details
216
+
217
+ - **`passFn` flexibility** — because `passFn` receives the whole annotation, a
218
+ `passRate` criterion can gate on a score bar (`a.score >= 0.7`), a range
219
+ (`a.score >= 0.5 && a.score <= 0.9`), a label (`a.label === "correct"`), or
220
+ any combination. Booleans arrive as `a.score === true` / `false`.
221
+ - **Booleans in `average`** count as `1` (`true`) / `0` (`false`) when computing
222
+ the mean.
223
+ - **Duplicate annotations** — if a run logs the same `annotationName` more than
224
+ once, the last one counts.
225
+ - **Missing annotations** — an `average` criterion with no numeric/boolean
226
+ scores, or a `passRate` criterion whose annotation was never logged, fails
227
+ with a "no … found" reason rather than passing vacuously.
228
+ - **Skipped vs dry-run** — skipped tests are excluded from the aggregate;
229
+ dry-run tests are included because they still execute locally.
230
+
231
+ ### Reporter output
232
+
233
+ Criteria are evaluated once, after all tests finish. If any fail, the suite
234
+ throws a single aggregated error and the reporter prints an `Acceptance
235
+ Criteria` block listing each criterion's observed value, the bar it needed to
236
+ clear, and its sample count. The reported value is the **mean** for `average`,
237
+ and the **fraction of runs that passed** for `passRate` (so a fully-passing
238
+ `passRate` criterion reads `1.000`).
239
+
240
+ ## Where To Start
241
+
242
+ - [CI Eval Tests: Vitest](./ci-evals-vitest) — config, reporter, and the full `describe` /
243
+ `test` / `test.each` API as it surfaces in Vitest
244
+ - [CI Eval Tests: Jest](./ci-evals-jest) — the same API surface in a Jest project
245
+ - [CI Eval Test Annotations](./ci-evals-annotations) — `logAnnotation`, `evaluate`, and
246
+ the `Annotation` shape
247
+
248
+ <section className="hidden" data-agent-context="source-map" aria-label="Source map">
249
+ <h2>Source Map</h2>
250
+ <ul>
251
+ <li><code>src/vitest/index.ts</code></li>
252
+ <li><code>src/vitest/reporter.ts</code></li>
253
+ <li><code>src/jest/index.ts</code></li>
254
+ <li><code>src/jest/reporter.ts</code></li>
255
+ <li><code>src/testing/runner.ts</code></li>
256
+ <li><code>src/testing/acceptance.ts</code></li>
257
+ <li><code>src/testing/helpers.ts</code></li>
258
+ <li><code>src/testing/phoenix-test-tracking.ts</code></li>
259
+ <li><code>src/testing/state.ts</code></li>
260
+ <li><code>src/testing/types.ts</code></li>
261
+ <li><code>src/testing/reporter-format.ts</code></li>
262
+ </ul>
263
+ </section>
package/docs/overview.mdx CHANGED
@@ -3,7 +3,7 @@ title: "Overview"
3
3
  description: "Typed TypeScript client for Phoenix platform APIs"
4
4
  ---
5
5
 
6
- `@arizeai/phoenix-client` is the typed TypeScript client for Phoenix platform APIs. It ships a small root REST client plus focused module entrypoints for prompts, datasets, experiments, spans, sessions, and traces.
6
+ `@arizeai/phoenix-client` is the typed TypeScript client for Phoenix platform APIs. It ships a small root REST client plus focused module entrypoints for prompts, datasets, experiments, spans, sessions, traces, and CI-friendly dataset-backed eval tests.
7
7
 
8
8
  ## Install
9
9
 
@@ -43,6 +43,10 @@ That gives the agent version-matched docs plus the exact implementation and gene
43
43
  | `@arizeai/phoenix-client/spans` | Span search, notes, and span/document annotations |
44
44
  | `@arizeai/phoenix-client/sessions` | Session listing, retrieval, and session annotations |
45
45
  | `@arizeai/phoenix-client/traces` | Project trace retrieval and trace annotations |
46
+ | `@arizeai/phoenix-client/vitest` | Vitest entrypoint for dataset-backed eval tests |
47
+ | `@arizeai/phoenix-client/vitest/reporter` | Vitest reporter for Phoenix eval summaries |
48
+ | `@arizeai/phoenix-client/jest` | Jest entrypoint for dataset-backed eval tests |
49
+ | `@arizeai/phoenix-client/jest/reporter` | Jest reporter for Phoenix eval summaries |
46
50
 
47
51
  ## Configuration
48
52
 
@@ -156,6 +160,7 @@ Prefer this layer when:
156
160
 
157
161
  - [Prompts](./prompts), [Datasets](./datasets), [Experiments](./experiments) — higher-level workflows
158
162
  - [Annotations](./annotations) — annotation concepts, then [Span](./span-annotations), [Document](./document-annotations), and [Session](./session-annotations) annotations for detailed usage
163
+ - [CI Eval Tests](./ci-evals) — Vitest/Jest eval suites backed by Phoenix datasets and experiments
159
164
  - [Spans](./spans), [Sessions](./sessions), [Traces](./traces) — retrieval and maintenance
160
165
 
161
166
  <section className="hidden" data-agent-context="source-map" aria-label="Source map">
@@ -171,6 +176,9 @@ Prefer this layer when:
171
176
  <li><code>src/spans/</code></li>
172
177
  <li><code>src/sessions/</code></li>
173
178
  <li><code>src/traces/</code></li>
179
+ <li><code>src/vitest/</code></li>
180
+ <li><code>src/jest/</code></li>
181
+ <li><code>src/testing/</code></li>
174
182
  <li><code>src/types/</code></li>
175
183
  </ul>
176
184
  </section>
package/package.json CHANGED
@@ -1,16 +1,18 @@
1
1
  {
2
2
  "name": "@arizeai/phoenix-client",
3
- "version": "6.10.1",
3
+ "version": "6.11.1",
4
4
  "description": "A client for the Phoenix API",
5
5
  "keywords": [
6
6
  "arize",
7
7
  "datasets",
8
8
  "evaluation",
9
9
  "experiments",
10
+ "jest",
10
11
  "llm",
11
12
  "phoenix",
12
13
  "prompts",
13
- "tracing"
14
+ "tracing",
15
+ "vitest"
14
16
  ],
15
17
  "homepage": "https://github.com/Arize-ai/phoenix/tree/main/js/packages/phoenix-client",
16
18
  "bugs": {
@@ -63,6 +65,26 @@
63
65
  "import": "./dist/esm/datasets/index.js",
64
66
  "require": "./dist/src/datasets/index.js"
65
67
  },
68
+ "./vitest": {
69
+ "types": "./dist/src/vitest/index.d.ts",
70
+ "import": "./dist/esm/vitest/index.js",
71
+ "require": "./dist/src/vitest/index.js"
72
+ },
73
+ "./vitest/reporter": {
74
+ "types": "./dist/src/vitest/reporter.d.ts",
75
+ "import": "./dist/esm/vitest/reporter.js",
76
+ "require": "./dist/src/vitest/reporter.js"
77
+ },
78
+ "./jest": {
79
+ "types": "./dist/src/jest/index.d.ts",
80
+ "import": "./dist/esm/jest/index.js",
81
+ "require": "./dist/src/jest/index.js"
82
+ },
83
+ "./jest/reporter": {
84
+ "types": "./dist/src/jest/reporter.d.ts",
85
+ "import": "./dist/esm/jest/reporter.js",
86
+ "require": "./dist/src/jest/reporter.js"
87
+ },
66
88
  "./utils/*": {
67
89
  "import": "./dist/esm/utils/*.js",
68
90
  "require": "./dist/src/utils/*.js"
@@ -73,33 +95,37 @@
73
95
  }
74
96
  },
75
97
  "dependencies": {
76
- "@arizeai/openinference-semantic-conventions": "^2.1.7",
77
- "@arizeai/openinference-vercel": "^2.7.0",
98
+ "@arizeai/openinference-semantic-conventions": "^2.5.0",
99
+ "@arizeai/openinference-vercel": "^2.7.8",
78
100
  "async": "^3.2.6",
79
101
  "openapi-fetch": "^0.17.0",
80
102
  "tiny-invariant": "^1.3.3",
81
- "zod": "^4.0.14",
103
+ "zod": "^4.4.3",
82
104
  "@arizeai/phoenix-config": "0.1.4",
83
105
  "@arizeai/phoenix-otel": "1.0.2"
84
106
  },
85
107
  "devDependencies": {
86
- "@ai-sdk/openai": "^3.0.29",
87
- "@anthropic-ai/sdk": "^0.35.0",
88
- "@opentelemetry/api": "^1.9.0",
89
- "@opentelemetry/sdk-trace-node": "^2.5.1",
90
- "@types/async": "^3.2.24",
91
- "@types/node": "^20.17.22",
92
- "ai": "^6.0.90",
93
- "openai": "^6.10.0",
94
- "openapi-typescript": "^7.6.1",
95
- "tsx": "^4.19.3",
96
- "vitest": "^4.1.0",
108
+ "@ai-sdk/openai": "^3.0.73",
109
+ "@anthropic-ai/sdk": "^0.102.0",
110
+ "@opentelemetry/api": "^1.9.1",
111
+ "@opentelemetry/sdk-trace-node": "^2.8.0",
112
+ "@types/async": "^3.2.25",
113
+ "@types/node": "^25.9.4",
114
+ "ai": "^6.0.208",
115
+ "dotenv": "^16.4.7",
116
+ "jest": "^29.7.0",
117
+ "openai": "^6.44.0",
118
+ "openapi-typescript": "^7.13.0",
119
+ "tsx": "^4.22.4",
120
+ "vitest": "^4.1.9",
97
121
  "@arizeai/phoenix-evals": "1.0.3"
98
122
  },
99
123
  "peerDependencies": {
100
124
  "@anthropic-ai/sdk": "^0.35.0",
101
125
  "ai": "^6.0.90",
102
- "openai": "^6.10.0"
126
+ "jest": ">=27",
127
+ "openai": "^6.10.0",
128
+ "vitest": ">=1"
103
129
  },
104
130
  "peerDependenciesMeta": {
105
131
  "@anthropic-ai/sdk": {
@@ -108,8 +134,14 @@
108
134
  "ai": {
109
135
  "optional": true
110
136
  },
137
+ "jest": {
138
+ "optional": true
139
+ },
111
140
  "openai": {
112
141
  "optional": true
142
+ },
143
+ "vitest": {
144
+ "optional": true
113
145
  }
114
146
  },
115
147
  "engines": {