@arizeai/phoenix-client 6.10.0 → 6.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (244) hide show
  1. package/README.md +62 -0
  2. package/dist/esm/__generated__/api/v1.d.ts +452 -28
  3. package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
  4. package/dist/esm/experiments/helpers/getExampleGlobalId.d.ts +8 -0
  5. package/dist/esm/experiments/helpers/getExampleGlobalId.d.ts.map +1 -0
  6. package/dist/esm/experiments/helpers/getExampleGlobalId.js +9 -0
  7. package/dist/esm/experiments/helpers/getExampleGlobalId.js.map +1 -0
  8. package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
  9. package/dist/esm/experiments/resumeEvaluation.js +2 -1
  10. package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
  11. package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
  12. package/dist/esm/experiments/resumeExperiment.js +3 -2
  13. package/dist/esm/experiments/resumeExperiment.js.map +1 -1
  14. package/dist/esm/experiments/runExperiment.d.ts.map +1 -1
  15. package/dist/esm/experiments/runExperiment.js +6 -3
  16. package/dist/esm/experiments/runExperiment.js.map +1 -1
  17. package/dist/esm/jest/index.d.ts +5 -0
  18. package/dist/esm/jest/index.d.ts.map +1 -0
  19. package/dist/esm/jest/index.js +49 -0
  20. package/dist/esm/jest/index.js.map +1 -0
  21. package/dist/esm/jest/reporter.d.ts +13 -0
  22. package/dist/esm/jest/reporter.d.ts.map +1 -0
  23. package/dist/esm/jest/reporter.js +19 -0
  24. package/dist/esm/jest/reporter.js.map +1 -0
  25. package/dist/esm/prompts/sdks/toAI.d.ts +2 -2
  26. package/dist/esm/prompts/sdks/toAI.d.ts.map +1 -1
  27. package/dist/esm/prompts/sdks/toAI.js.map +1 -1
  28. package/dist/esm/prompts/sdks/toAnthropic.d.ts +2 -2
  29. package/dist/esm/prompts/sdks/toAnthropic.d.ts.map +1 -1
  30. package/dist/esm/prompts/sdks/toAnthropic.js.map +1 -1
  31. package/dist/esm/prompts/sdks/toOpenAI.d.ts +2 -2
  32. package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
  33. package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
  34. package/dist/esm/prompts/sdks/toSDK.d.ts +8 -8
  35. package/dist/esm/prompts/sdks/toSDK.d.ts.map +1 -1
  36. package/dist/esm/prompts/sdks/toSDK.js.map +1 -1
  37. package/dist/esm/prompts/sdks/types.d.ts +2 -2
  38. package/dist/esm/prompts/sdks/types.d.ts.map +1 -1
  39. package/dist/esm/schemas/llm/anthropic/converters.d.ts +8 -8
  40. package/dist/esm/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  41. package/dist/esm/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  42. package/dist/esm/schemas/llm/constants.d.ts +3 -3
  43. package/dist/esm/schemas/llm/converters.d.ts +12 -12
  44. package/dist/esm/schemas/llm/openai/converters.d.ts +3 -3
  45. package/dist/esm/schemas/llm/schemas.d.ts +2 -2
  46. package/dist/esm/testing/acceptance.d.ts +20 -0
  47. package/dist/esm/testing/acceptance.d.ts.map +1 -0
  48. package/dist/esm/testing/acceptance.js +129 -0
  49. package/dist/esm/testing/acceptance.js.map +1 -0
  50. package/dist/esm/testing/define-api.d.ts +157 -0
  51. package/dist/esm/testing/define-api.d.ts.map +1 -0
  52. package/dist/esm/testing/define-api.js +78 -0
  53. package/dist/esm/testing/define-api.js.map +1 -0
  54. package/dist/esm/testing/helpers.d.ts +55 -0
  55. package/dist/esm/testing/helpers.d.ts.map +1 -0
  56. package/dist/esm/testing/helpers.js +179 -0
  57. package/dist/esm/testing/helpers.js.map +1 -0
  58. package/dist/esm/testing/phoenix-test-tracking.d.ts +68 -0
  59. package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -0
  60. package/dist/esm/testing/phoenix-test-tracking.js +521 -0
  61. package/dist/esm/testing/phoenix-test-tracking.js.map +1 -0
  62. package/dist/esm/testing/report-artifacts.d.ts +45 -0
  63. package/dist/esm/testing/report-artifacts.d.ts.map +1 -0
  64. package/dist/esm/testing/report-artifacts.js +218 -0
  65. package/dist/esm/testing/report-artifacts.js.map +1 -0
  66. package/dist/esm/testing/report-run.d.ts +22 -0
  67. package/dist/esm/testing/report-run.d.ts.map +1 -0
  68. package/dist/esm/testing/report-run.js +41 -0
  69. package/dist/esm/testing/report-run.js.map +1 -0
  70. package/dist/esm/testing/reporter-format.d.ts +83 -0
  71. package/dist/esm/testing/reporter-format.d.ts.map +1 -0
  72. package/dist/esm/testing/reporter-format.js +852 -0
  73. package/dist/esm/testing/reporter-format.js.map +1 -0
  74. package/dist/esm/testing/runner.d.ts +31 -0
  75. package/dist/esm/testing/runner.d.ts.map +1 -0
  76. package/dist/esm/testing/runner.js +238 -0
  77. package/dist/esm/testing/runner.js.map +1 -0
  78. package/dist/esm/testing/state.d.ts +138 -0
  79. package/dist/esm/testing/state.d.ts.map +1 -0
  80. package/dist/esm/testing/state.js +31 -0
  81. package/dist/esm/testing/state.js.map +1 -0
  82. package/dist/esm/testing/types.d.ts +319 -0
  83. package/dist/esm/testing/types.d.ts.map +1 -0
  84. package/dist/esm/testing/types.js +9 -0
  85. package/dist/esm/testing/types.js.map +1 -0
  86. package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
  87. package/dist/esm/utils/channel.d.ts +7 -7
  88. package/dist/esm/utils/channel.d.ts.map +1 -1
  89. package/dist/esm/utils/channel.js +1 -1
  90. package/dist/esm/utils/channel.js.map +1 -1
  91. package/dist/esm/utils/formatPromptMessages.d.ts.map +1 -1
  92. package/dist/esm/utils/getPromptBySelector.d.ts.map +1 -1
  93. package/dist/esm/utils/promisifyResult.d.ts +1 -1
  94. package/dist/esm/utils/promisifyResult.d.ts.map +1 -1
  95. package/dist/esm/utils/promisifyResult.js.map +1 -1
  96. package/dist/esm/utils/schemaMatches.d.ts +5 -5
  97. package/dist/esm/utils/schemaMatches.d.ts.map +1 -1
  98. package/dist/esm/utils/schemaMatches.js.map +1 -1
  99. package/dist/esm/vitest/index.d.ts +5 -0
  100. package/dist/esm/vitest/index.d.ts.map +1 -0
  101. package/dist/esm/vitest/index.js +15 -0
  102. package/dist/esm/vitest/index.js.map +1 -0
  103. package/dist/esm/vitest/reporter.d.ts +19 -0
  104. package/dist/esm/vitest/reporter.d.ts.map +1 -0
  105. package/dist/esm/vitest/reporter.js +27 -0
  106. package/dist/esm/vitest/reporter.js.map +1 -0
  107. package/dist/src/__generated__/api/v1.d.ts +452 -28
  108. package/dist/src/__generated__/api/v1.d.ts.map +1 -1
  109. package/dist/src/experiments/helpers/getExampleGlobalId.d.ts +8 -0
  110. package/dist/src/experiments/helpers/getExampleGlobalId.d.ts.map +1 -0
  111. package/dist/src/experiments/helpers/getExampleGlobalId.js +13 -0
  112. package/dist/src/experiments/helpers/getExampleGlobalId.js.map +1 -0
  113. package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
  114. package/dist/src/experiments/resumeEvaluation.js +2 -1
  115. package/dist/src/experiments/resumeEvaluation.js.map +1 -1
  116. package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
  117. package/dist/src/experiments/resumeExperiment.js +3 -2
  118. package/dist/src/experiments/resumeExperiment.js.map +1 -1
  119. package/dist/src/experiments/runExperiment.d.ts.map +1 -1
  120. package/dist/src/experiments/runExperiment.js +6 -3
  121. package/dist/src/experiments/runExperiment.js.map +1 -1
  122. package/dist/src/jest/index.d.ts +5 -0
  123. package/dist/src/jest/index.d.ts.map +1 -0
  124. package/dist/src/jest/index.js +58 -0
  125. package/dist/src/jest/index.js.map +1 -0
  126. package/dist/src/jest/reporter.d.ts +13 -0
  127. package/dist/src/jest/reporter.d.ts.map +1 -0
  128. package/dist/src/jest/reporter.js +23 -0
  129. package/dist/src/jest/reporter.js.map +1 -0
  130. package/dist/src/prompts/sdks/toAI.d.ts +2 -2
  131. package/dist/src/prompts/sdks/toAI.d.ts.map +1 -1
  132. package/dist/src/prompts/sdks/toAI.js.map +1 -1
  133. package/dist/src/prompts/sdks/toAnthropic.d.ts +2 -2
  134. package/dist/src/prompts/sdks/toAnthropic.d.ts.map +1 -1
  135. package/dist/src/prompts/sdks/toAnthropic.js.map +1 -1
  136. package/dist/src/prompts/sdks/toOpenAI.d.ts +2 -2
  137. package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
  138. package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
  139. package/dist/src/prompts/sdks/toSDK.d.ts +8 -8
  140. package/dist/src/prompts/sdks/toSDK.d.ts.map +1 -1
  141. package/dist/src/prompts/sdks/toSDK.js.map +1 -1
  142. package/dist/src/prompts/sdks/types.d.ts +2 -2
  143. package/dist/src/prompts/sdks/types.d.ts.map +1 -1
  144. package/dist/src/schemas/llm/anthropic/converters.d.ts +8 -8
  145. package/dist/src/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  146. package/dist/src/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  147. package/dist/src/schemas/llm/constants.d.ts +3 -3
  148. package/dist/src/schemas/llm/converters.d.ts +12 -12
  149. package/dist/src/schemas/llm/openai/converters.d.ts +3 -3
  150. package/dist/src/schemas/llm/schemas.d.ts +2 -2
  151. package/dist/src/testing/acceptance.d.ts +20 -0
  152. package/dist/src/testing/acceptance.d.ts.map +1 -0
  153. package/dist/src/testing/acceptance.js +114 -0
  154. package/dist/src/testing/acceptance.js.map +1 -0
  155. package/dist/src/testing/define-api.d.ts +157 -0
  156. package/dist/src/testing/define-api.d.ts.map +1 -0
  157. package/dist/src/testing/define-api.js +81 -0
  158. package/dist/src/testing/define-api.js.map +1 -0
  159. package/dist/src/testing/helpers.d.ts +55 -0
  160. package/dist/src/testing/helpers.d.ts.map +1 -0
  161. package/dist/src/testing/helpers.js +182 -0
  162. package/dist/src/testing/helpers.js.map +1 -0
  163. package/dist/src/testing/phoenix-test-tracking.d.ts +68 -0
  164. package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -0
  165. package/dist/src/testing/phoenix-test-tracking.js +530 -0
  166. package/dist/src/testing/phoenix-test-tracking.js.map +1 -0
  167. package/dist/src/testing/report-artifacts.d.ts +45 -0
  168. package/dist/src/testing/report-artifacts.d.ts.map +1 -0
  169. package/dist/src/testing/report-artifacts.js +225 -0
  170. package/dist/src/testing/report-artifacts.js.map +1 -0
  171. package/dist/src/testing/report-run.d.ts +22 -0
  172. package/dist/src/testing/report-run.d.ts.map +1 -0
  173. package/dist/src/testing/report-run.js +47 -0
  174. package/dist/src/testing/report-run.js.map +1 -0
  175. package/dist/src/testing/reporter-format.d.ts +83 -0
  176. package/dist/src/testing/reporter-format.d.ts.map +1 -0
  177. package/dist/src/testing/reporter-format.js +870 -0
  178. package/dist/src/testing/reporter-format.js.map +1 -0
  179. package/dist/src/testing/runner.d.ts +31 -0
  180. package/dist/src/testing/runner.d.ts.map +1 -0
  181. package/dist/src/testing/runner.js +258 -0
  182. package/dist/src/testing/runner.js.map +1 -0
  183. package/dist/src/testing/state.d.ts +138 -0
  184. package/dist/src/testing/state.d.ts.map +1 -0
  185. package/dist/src/testing/state.js +38 -0
  186. package/dist/src/testing/state.js.map +1 -0
  187. package/dist/src/testing/types.d.ts +319 -0
  188. package/dist/src/testing/types.d.ts.map +1 -0
  189. package/dist/src/testing/types.js +13 -0
  190. package/dist/src/testing/types.js.map +1 -0
  191. package/dist/src/utils/channel.d.ts +7 -7
  192. package/dist/src/utils/channel.d.ts.map +1 -1
  193. package/dist/src/utils/channel.js +1 -1
  194. package/dist/src/utils/channel.js.map +1 -1
  195. package/dist/src/utils/formatPromptMessages.d.ts.map +1 -1
  196. package/dist/src/utils/getPromptBySelector.d.ts.map +1 -1
  197. package/dist/src/utils/promisifyResult.d.ts +1 -1
  198. package/dist/src/utils/promisifyResult.d.ts.map +1 -1
  199. package/dist/src/utils/promisifyResult.js.map +1 -1
  200. package/dist/src/utils/schemaMatches.d.ts +5 -5
  201. package/dist/src/utils/schemaMatches.d.ts.map +1 -1
  202. package/dist/src/utils/schemaMatches.js.map +1 -1
  203. package/dist/src/vitest/index.d.ts +5 -0
  204. package/dist/src/vitest/index.d.ts.map +1 -0
  205. package/dist/src/vitest/index.js +23 -0
  206. package/dist/src/vitest/index.js.map +1 -0
  207. package/dist/src/vitest/reporter.d.ts +19 -0
  208. package/dist/src/vitest/reporter.d.ts.map +1 -0
  209. package/dist/src/vitest/reporter.js +34 -0
  210. package/dist/src/vitest/reporter.js.map +1 -0
  211. package/dist/tsconfig.tsbuildinfo +1 -1
  212. package/docs/ci-evals-annotations.mdx +190 -0
  213. package/docs/ci-evals-jest.mdx +78 -0
  214. package/docs/ci-evals-vitest.mdx +240 -0
  215. package/docs/ci-evals.mdx +263 -0
  216. package/docs/overview.mdx +9 -1
  217. package/package.json +49 -17
  218. package/src/__generated__/api/v1.ts +452 -28
  219. package/src/experiments/helpers/getExampleGlobalId.ts +12 -0
  220. package/src/experiments/resumeEvaluation.ts +2 -1
  221. package/src/experiments/resumeExperiment.ts +3 -2
  222. package/src/experiments/runExperiment.ts +6 -3
  223. package/src/jest/index.ts +124 -0
  224. package/src/jest/reporter.ts +22 -0
  225. package/src/prompts/sdks/toAI.ts +4 -3
  226. package/src/prompts/sdks/toAnthropic.ts +4 -3
  227. package/src/prompts/sdks/toOpenAI.ts +4 -3
  228. package/src/prompts/sdks/toSDK.ts +16 -11
  229. package/src/prompts/sdks/types.ts +2 -2
  230. package/src/testing/acceptance.ts +190 -0
  231. package/src/testing/define-api.ts +279 -0
  232. package/src/testing/helpers.ts +251 -0
  233. package/src/testing/phoenix-test-tracking.ts +637 -0
  234. package/src/testing/report-artifacts.ts +272 -0
  235. package/src/testing/report-run.ts +44 -0
  236. package/src/testing/reporter-format.ts +1072 -0
  237. package/src/testing/runner.ts +350 -0
  238. package/src/testing/state.ts +165 -0
  239. package/src/testing/types.ts +366 -0
  240. package/src/utils/channel.ts +17 -15
  241. package/src/utils/promisifyResult.ts +6 -4
  242. package/src/utils/schemaMatches.ts +12 -10
  243. package/src/vitest/index.ts +57 -0
  244. package/src/vitest/reporter.ts +32 -0
@@ -0,0 +1,366 @@
1
+ import type { PhoenixClient } from "../index";
2
+ import type { AnnotatorKind } from "../types/annotations";
3
+ import type {
4
+ EvaluatorParams,
5
+ EvaluationResult as ExperimentEvaluationResult,
6
+ } from "../types/experiments";
7
+
8
+ /**
9
+ * Phoenix annotator kind, re-exported from the shared client types so the
10
+ * testing module and the rest of the client agree on a single definition.
11
+ */
12
+ export type { AnnotatorKind };
13
+
14
+ /** A JSON-serializable map. */
15
+ export type KVMap = Record<string, unknown>;
16
+
17
+ /**
18
+ * Domain language
19
+ * ---------------
20
+ * The unit these tests evaluate over is an **Example**: a single AI example —
21
+ * an `input`, its `expected` output, and optional `metadata` / `splits` — over
22
+ * which the task under test is run and then scored. This is the same notion as
23
+ * the dataset `Example` (`../types/datasets`): each test case _is_ one example.
24
+ * When tracked, a case is recorded to Phoenix as a dataset example and
25
+ * evaluated as one experiment run.
26
+ *
27
+ * These field names form the shared vocabulary across this module:
28
+ * - `input` — the example's input, passed to the task under evaluation.
29
+ * - `expected` — the example's expected (reference / ground-truth) output.
30
+ * - `metadata` — extra fields carried on the example.
31
+ * - `splits` — slice labels for the example.
32
+ * - `id` — stable example id, used to upsert the example across runs.
33
+ */
34
+
35
+ /**
36
+ * The expected output of an `Example`, accepted under any one of three
37
+ * interchangeable keys. All three normalize to the same slot: when recorded to
38
+ * Phoenix the value becomes the dataset example's `output`, and it is exposed
39
+ * to evaluators as `expected` on `EvaluatorParams`. At most one key may be set.
40
+ *
41
+ * - `expected` — the canonical name (the ground-truth / reference output).
42
+ * - `reference` — alias preferred by frameworks that name the slot "reference".
43
+ * - `output` — alias for callers who think in terms of the example's `output`.
44
+ *
45
+ * Modeled as a union so supplying more than one key at a time is a type error.
46
+ */
47
+ export type ReferenceOutput<Expected extends KVMap = KVMap> =
48
+ | { expected?: Expected; reference?: never; output?: never }
49
+ | { reference?: Expected; expected?: never; output?: never }
50
+ | { output?: Expected; expected?: never; reference?: never };
51
+
52
+ /**
53
+ * The `Example` fields that define a single test case, excluding its
54
+ * expected output (which is supplied separately via {@link ReferenceOutput}).
55
+ *
56
+ * `input` is the example's input — the value fed to the task under evaluation.
57
+ * When the case is tracked, this becomes the dataset example's `input`.
58
+ */
59
+ export interface TestParamsBase<Input extends KVMap = KVMap> {
60
+ /** Optional stable example id; used to upsert the example between runs. */
61
+ id?: string;
62
+ /** The example's input — fed to the task under evaluation. Required. */
63
+ input: Input;
64
+ /** Additional metadata stored on the example and its run. */
65
+ metadata?: KVMap;
66
+ /**
67
+ * Split assignment(s) for the example, used to slice the dataset and
68
+ * experiment in the Phoenix UI (e.g. `["factual_accuracy", "correct"]`).
69
+ */
70
+ splits?: string[];
71
+ /** Per-test config (tags + metadata recorded on the run). */
72
+ config?: TestConfig;
73
+ /**
74
+ * Number of times to run this test case. Each repetition becomes a
75
+ * separate experiment run against the same dataset example (carrying a
76
+ * distinct `repetition_number`). Overrides the suite-level `repetitions`.
77
+ * Defaults to the suite value, then `PHOENIX_TEST_REPETITIONS`, then `1`.
78
+ */
79
+ repetitions?: number;
80
+ /**
81
+ * When `true`, this test runs as an ordinary local test only — no dataset
82
+ * example is created and no experiment run or annotations are uploaded to
83
+ * Phoenix. Useful for scaffolding a case before it's ready to track.
84
+ */
85
+ dryRun?: boolean;
86
+ }
87
+
88
+ /**
89
+ * The full inline definition of a single `Example` under test.
90
+ *
91
+ * Combines {@link TestParamsBase} with a {@link ReferenceOutput}, so the
92
+ * example's expected output may be given under `expected`, `reference`, or
93
+ * `output` (at most one). All three resolve to the same canonical `expected`
94
+ * slot.
95
+ */
96
+ export type TestParams<
97
+ Input extends KVMap = KVMap,
98
+ Expected extends KVMap = KVMap,
99
+ > = TestParamsBase<Input> & ReferenceOutput<Expected>;
100
+
101
+ /**
102
+ * Resolve an `Example`'s expected output from a value that may carry it
103
+ * under any of the `expected` / `reference` / `output` aliases (see
104
+ * {@link ReferenceOutput}). Returns the first one set, or `undefined` if none.
105
+ */
106
+ export function resolveReference<Expected extends KVMap = KVMap>(
107
+ params: ReferenceOutput<Expected>
108
+ ): Expected | undefined {
109
+ return params.expected ?? params.reference ?? params.output;
110
+ }
111
+
112
+ /** Per-test runtime configuration. */
113
+ export interface TestConfig {
114
+ /** Tags recorded on the experiment run for filtering in the Phoenix UI. */
115
+ tags?: string[];
116
+ /** Extra metadata recorded on the experiment run. */
117
+ metadata?: KVMap;
118
+ }
119
+
120
+ /**
121
+ * How a criterion aggregates an annotation's scores to gate the suite:
122
+ *
123
+ * - `"average"` — gate on overall quality: the **mean** score across all runs
124
+ * must clear the criterion's `threshold`. A few weak runs are tolerated as
125
+ * long as the mean holds.
126
+ * - `"passRate"` — gate on consistency: each run **passes** when the
127
+ * criterion's `passFn` predicate returns `true` for its annotation, and the
128
+ * suite passes when the **fraction** of runs that pass is at least
129
+ * `minPassRate` (e.g. `minPassRate: 0.9` ⇒ 90% must pass; `1` ⇒ all).
130
+ */
131
+ export type AcceptanceMetric = "average" | "passRate";
132
+
133
+ /**
134
+ * Optimization direction for a criterion's scores: `"maximize"` (higher is
135
+ * better, the default) or `"minimize"` (lower is better). Controls every
136
+ * score comparison the criterion makes.
137
+ */
138
+ export type OptimizationDirection = "maximize" | "minimize";
139
+
140
+ /** Fields shared by every {@link AcceptanceCriterion} variant. */
141
+ export interface AcceptanceCriterionBase {
142
+ /** Annotation name to aggregate across completed test runs. */
143
+ annotationName: string;
144
+ }
145
+
146
+ /**
147
+ * Gate the suite on the **mean** score: the average across all runs must clear
148
+ * `threshold` (compared in `direction`).
149
+ */
150
+ export interface AverageAcceptanceCriterion extends AcceptanceCriterionBase {
151
+ metric: "average";
152
+ /**
153
+ * The bar the mean score must clear, compared in `direction`. Boolean scores
154
+ * average as `1` (`true`) / `0` (`false`).
155
+ */
156
+ threshold: number;
157
+ /**
158
+ * Optimization direction; defaults to `"maximize"`. `"maximize"` treats a
159
+ * higher mean as better (clears when `>= threshold`); `"minimize"` treats a
160
+ * lower mean as better (clears when `<= threshold`) — use it for cost,
161
+ * latency, or error-rate annotations.
162
+ */
163
+ direction?: OptimizationDirection;
164
+ }
165
+
166
+ /**
167
+ * Gate the suite on the **pass rate**: each run passes when `passFn` returns
168
+ * `true` for its annotation, and the suite passes when at least `minPassRate`
169
+ * of runs do. `passFn` decides what "passing" means, so any logic works — a
170
+ * score bar, a score range, a label match, a metadata check, etc.
171
+ */
172
+ export interface PassRateAcceptanceCriterion extends AcceptanceCriterionBase {
173
+ metric: "passRate";
174
+ /**
175
+ * Predicate deciding whether a single run passes, given the run's last
176
+ * {@link Annotation} for `annotationName` (its `score`, `label`,
177
+ * `explanation`, `metadata`, …). Runs whose predicate returns `true` count
178
+ * toward the pass rate.
179
+ */
180
+ passFn: (annotation: Annotation) => boolean;
181
+ /**
182
+ * Minimum fraction of runs (`0`–`1`) that must pass for the suite to pass —
183
+ * e.g. `0.9` requires 90% of runs to satisfy `passFn`, `1` requires all of
184
+ * them. The suite passes when `passRate >= minPassRate`.
185
+ */
186
+ minPassRate: number;
187
+ }
188
+
189
+ /**
190
+ * One aggregate acceptance rule, evaluated once after every test in the suite
191
+ * has run. Each criterion aggregates a single annotation's scores with one
192
+ * {@link AcceptanceMetric} and fails the suite when the result misses its bar.
193
+ *
194
+ * Scoring notes shared by every metric:
195
+ * - Boolean scores count as `1` (`true`) / `0` (`false`).
196
+ * - If a run logs the same annotation more than once, the last one counts.
197
+ * - Skipped tests are excluded; dry-run tests are included (they still run).
198
+ * - A criterion whose annotation was never logged on any run fails (rather
199
+ * than passing vacuously) — see {@link AcceptanceResultFields.failureReason}.
200
+ */
201
+ export type AcceptanceCriterion =
202
+ | AverageAcceptanceCriterion
203
+ | PassRateAcceptanceCriterion;
204
+
205
+ /** The computed fields added to an {@link AcceptanceCriterion} once evaluated. */
206
+ export interface AcceptanceResultFields {
207
+ /**
208
+ * The aggregate the criterion gated on, or `null` when there were no runs to
209
+ * aggregate. For `"average"` this is the mean score; for `"passRate"` it is
210
+ * the fraction of runs that passed (so a fully-passing `"passRate"` criterion
211
+ * reports `1`).
212
+ */
213
+ value: number | null;
214
+ /** Number of runs included in the aggregate. */
215
+ sampleCount: number;
216
+ /** Whether the aggregate cleared the criterion. */
217
+ passed: boolean;
218
+ /** Human-readable failure reason for invalid or empty aggregates. */
219
+ failureReason?: string;
220
+ }
221
+
222
+ /** Computed result for one aggregate acceptance rule. */
223
+ export type AcceptanceResult = AcceptanceCriterion & AcceptanceResultFields;
224
+
225
+ /** Suite-level configuration accepted by `describe()`. */
226
+ export interface SuiteConfig {
227
+ /** Override the dataset / experiment name used for the suite. */
228
+ datasetName?: string;
229
+ /** Description for the dataset and experiment. */
230
+ description?: string;
231
+ /** Suite-level metadata applied to every run in this experiment. */
232
+ metadata?: KVMap;
233
+ /** Override the Phoenix client used for syncing this suite. */
234
+ client?: PhoenixClient;
235
+ /**
236
+ * Number of times to run each test case in this suite. Individual tests
237
+ * may override this via `TestParams.repetitions`. Defaults to the
238
+ * `PHOENIX_TEST_REPETITIONS` env var, then `1`.
239
+ */
240
+ repetitions?: number;
241
+ /**
242
+ * When `true`, the whole suite runs as ordinary local tests — no dataset
243
+ * is uploaded and no experiment, runs, or annotations are created in
244
+ * Phoenix. Equivalent to `PHOENIX_TEST_TRACING=false` scoped to this
245
+ * suite. The reporter still prints a local summary.
246
+ */
247
+ dryRun?: boolean;
248
+ /**
249
+ * Aggregate annotation criteria that gate the suite after all tests run.
250
+ * Each criterion fails the suite when its scores miss the configured bar
251
+ * (see {@link AcceptanceCriterion}).
252
+ */
253
+ acceptanceCriteria?: AcceptanceCriterion[];
254
+ }
255
+
256
+ /**
257
+ * Arguments passed to a `test()` body: the `Example` under test, exposed
258
+ * as its `input`, `expected` output, and `metadata`. Read straight from the
259
+ * test's {@link TestParams} — the runner does not transform them.
260
+ */
261
+ export interface TestArgs<
262
+ Input extends KVMap = KVMap,
263
+ Expected extends KVMap = KVMap,
264
+ > {
265
+ /** The example's input under test. */
266
+ input: Input;
267
+ /** The example's expected (reference) output, when one was supplied. */
268
+ expected?: Expected;
269
+ /** Any metadata attached to the example. */
270
+ metadata?: KVMap;
271
+ }
272
+
273
+ /**
274
+ * Object form of an evaluator result. Reuses the shared experiment
275
+ * {@link ExperimentEvaluationResult} shape (label / explanation / metadata)
276
+ * but widens `score` to also accept booleans, which the testing API stores as
277
+ * `1` / `0`.
278
+ */
279
+ export interface EvaluationResultObject extends Omit<
280
+ ExperimentEvaluationResult,
281
+ "score"
282
+ > {
283
+ /** Numeric or boolean score; booleans are stored as `1` / `0`. */
284
+ score?: number | boolean | null;
285
+ }
286
+
287
+ /**
288
+ * One annotation recorded against a run. Extends the evaluator
289
+ * {@link EvaluationResultObject} with the `name` and `annotatorKind` carried
290
+ * on the evaluation body, plus an optional originating trace id.
291
+ */
292
+ export interface Annotation extends EvaluationResultObject {
293
+ /** Phoenix evaluation name. Required, and unique per run (last write wins). */
294
+ name: string;
295
+ /** Who or what produced the annotation. Defaults to `"CODE"`. */
296
+ annotatorKind?: AnnotatorKind;
297
+ /** Trace id for this evaluation, when the annotation was produced by a traced evaluator. */
298
+ traceId?: string | null;
299
+ }
300
+
301
+ /** Result returned by `traceEvaluator` for any evaluator-shaped value. */
302
+ export type EvaluatorResult = Annotation | (KVMap & { name: string });
303
+
304
+ /** Result shape produced by evaluator objects used in eval tests. */
305
+ export type EvaluationResult =
306
+ | number
307
+ | boolean
308
+ | string
309
+ | null
310
+ | EvaluationResultObject;
311
+
312
+ /**
313
+ * Parameters passed to an evaluator when it runs inside a test. A relaxation of
314
+ * the shared {@link EvaluatorParams}: `input` is always present, while `output`
315
+ * (an evaluator may run before `logOutput()`) and the remaining fields are
316
+ * optional. Deriving from `EvaluatorParams` keeps this aligned with the
317
+ * experiment evaluator contract as that shape evolves.
318
+ */
319
+ export type EvaluationParams = Partial<EvaluatorParams> & {
320
+ /** The example's input under test. */
321
+ input: KVMap;
322
+ };
323
+
324
+ /** Structural evaluator interface accepted by `evaluate()`. */
325
+ export interface Evaluator<
326
+ Params extends KVMap = EvaluationParams & KVMap,
327
+ Result = EvaluationResult,
328
+ > {
329
+ /** Annotation/evaluation name. */
330
+ name: string;
331
+ /** Who or what produced the result. Defaults to `"CODE"`. */
332
+ kind?: AnnotatorKind;
333
+ /** Compute the evaluation result. */
334
+ evaluate: (params: Params) => Result | Promise<Result>;
335
+ }
336
+
337
+ /** Test handler signature. */
338
+ export type TestFn<
339
+ Input extends KVMap = KVMap,
340
+ Expected extends KVMap = KVMap,
341
+ > = (args: TestArgs<Input, Expected>) => unknown | Promise<unknown>;
342
+
343
+ /**
344
+ * Each-row shape accepted by `test.each(table)(name, fn)`; each row defines one
345
+ * `Example`.
346
+ *
347
+ * Like {@link TestParams}, the example's expected output is supplied via
348
+ * {@link ReferenceOutput} (`expected` / `reference` / `output`, at most one).
349
+ * The trailing index signature still permits arbitrary extra columns on a row
350
+ * (e.g. for `%j` name interpolation) without weakening that constraint.
351
+ */
352
+ export type TestEachRow<
353
+ Input extends KVMap = KVMap,
354
+ Expected extends KVMap = KVMap,
355
+ > = {
356
+ id?: string;
357
+ input: Input;
358
+ metadata?: KVMap;
359
+ /** Per-row split assignment(s); see `TestParams.splits`. */
360
+ splits?: string[];
361
+ /** Per-row repetition count; see `TestParams.repetitions`. */
362
+ repetitions?: number;
363
+ /** Per-row dry-run flag; see `TestParams.dryRun`. */
364
+ dryRun?: boolean;
365
+ } & ReferenceOutput<Expected> &
366
+ Record<string, unknown>;
@@ -74,7 +74,7 @@
74
74
  *
75
75
  * @see https://en.wikipedia.org/wiki/Communicating_sequential_processes
76
76
  *
77
- * @template T The type of values sent through the channel
77
+ * @template Message The type of values sent through the channel
78
78
  *
79
79
  * @example Safe Producer-Consumer Pattern
80
80
  * ```typescript
@@ -116,8 +116,8 @@
116
116
  /**
117
117
  * Internal type for blocked senders waiting to deliver values
118
118
  */
119
- interface Sender<T> {
120
- readonly value: T;
119
+ interface Sender<Message> {
120
+ readonly value: Message;
121
121
  readonly resolve: () => void;
122
122
  readonly reject: (error: Error) => void;
123
123
  }
@@ -125,8 +125,8 @@ interface Sender<T> {
125
125
  /**
126
126
  * Internal type for blocked receivers waiting for values
127
127
  */
128
- interface Receiver<T> {
129
- readonly resolve: (value: T | typeof CLOSED) => void;
128
+ interface Receiver<Message> {
129
+ readonly resolve: (value: Message | typeof CLOSED) => void;
130
130
  readonly reject: (error: Error) => void;
131
131
  }
132
132
 
@@ -149,10 +149,10 @@ const ERRORS = {
149
149
  NEGATIVE_CAPACITY: "Channel capacity must be non-negative",
150
150
  } as const satisfies Record<string, string>;
151
151
 
152
- export class Channel<T> {
153
- #buffer: T[] = [];
154
- #sendQueue: Sender<T>[] = [];
155
- #receiveQueue: Receiver<T>[] = [];
152
+ export class Channel<Message> {
153
+ #buffer: Message[] = [];
154
+ #sendQueue: Sender<Message>[] = [];
155
+ #receiveQueue: Receiver<Message>[] = [];
156
156
  #closed = false;
157
157
  readonly #capacity: number;
158
158
 
@@ -191,7 +191,7 @@ export class Channel<T> {
191
191
  * @param value - The value to send
192
192
  * @throws {ChannelError} If channel is closed
193
193
  */
194
- async send(value: T): Promise<void> {
194
+ async send(value: Message): Promise<void> {
195
195
  if (this.#closed) {
196
196
  throw new ChannelError(ERRORS.SEND_TO_CLOSED);
197
197
  }
@@ -221,7 +221,7 @@ export class Channel<T> {
221
221
  *
222
222
  * @returns The received value, or CLOSED symbol if channel is closed and empty
223
223
  */
224
- async receive(): Promise<T | typeof CLOSED> {
224
+ async receive(): Promise<Message | typeof CLOSED> {
225
225
  // Drain buffer first
226
226
  if (this.#buffer.length > 0) {
227
227
  const value = this.#buffer.shift()!;
@@ -249,7 +249,7 @@ export class Channel<T> {
249
249
  }
250
250
 
251
251
  // Block until value available
252
- return new Promise<T | typeof CLOSED>((resolve, reject) => {
252
+ return new Promise<Message | typeof CLOSED>((resolve, reject) => {
253
253
  this.#receiveQueue.push({ resolve, reject });
254
254
  });
255
255
  }
@@ -271,7 +271,7 @@ export class Channel<T> {
271
271
  * }
272
272
  * ```
273
273
  */
274
- tryReceive(): T | typeof CLOSED | undefined {
274
+ tryReceive(): Message | typeof CLOSED | undefined {
275
275
  // Drain buffer first
276
276
  if (this.#buffer.length > 0) {
277
277
  const value = this.#buffer.shift()!;
@@ -362,7 +362,7 @@ export class Channel<T> {
362
362
  /**
363
363
  * Async iterator support for for-await-of loops
364
364
  */
365
- async *[Symbol.asyncIterator](): AsyncIterableIterator<T> {
365
+ async *[Symbol.asyncIterator](): AsyncIterableIterator<Message> {
366
366
  while (true) {
367
367
  const value = await this.receive();
368
368
  if (value === CLOSED) return;
@@ -392,6 +392,8 @@ export const CLOSED = Symbol("CLOSED");
392
392
  * }
393
393
  * ```
394
394
  */
395
- export function isClosed<T>(value: T | typeof CLOSED): value is typeof CLOSED {
395
+ export function isClosed<Message>(
396
+ value: Message | typeof CLOSED
397
+ ): value is typeof CLOSED {
396
398
  return value === CLOSED;
397
399
  }
@@ -2,12 +2,14 @@
2
2
  * If the incoming function returns a promise, return the promise.
3
3
  * Otherwise, return a promise that resolves to the incoming function's return value.
4
4
  */
5
- export function promisifyResult<T>(result: T) {
5
+ export function promisifyResult<Result>(result: Result) {
6
6
  if (result instanceof Promise) {
7
- return result as T extends Promise<infer U> ? Promise<U> : never;
7
+ return result as Result extends Promise<infer ResolvedValue>
8
+ ? Promise<ResolvedValue>
9
+ : never;
8
10
  }
9
11
 
10
- return Promise.resolve(result) as T extends Promise<unknown>
12
+ return Promise.resolve(result) as Result extends Promise<unknown>
11
13
  ? never
12
- : Promise<T>;
14
+ : Promise<Result>;
13
15
  }
@@ -3,8 +3,10 @@ import type { z } from "zod";
3
3
  /**
4
4
  * Simple utility to check if two types are exactly equivalent
5
5
  */
6
- export type AssertEqual<T, U> =
7
- (<V>() => V extends T ? 1 : 2) extends <V>() => V extends U ? 1 : 2
6
+ export type AssertEqual<Actual, Expected> =
7
+ (<Argument>() => Argument extends Actual ? 1 : 2) extends <
8
+ Argument,
9
+ >() => Argument extends Expected ? 1 : 2
8
10
  ? true
9
11
  : false;
10
12
 
@@ -14,16 +16,16 @@ export type AssertEqual<T, U> =
14
16
  * @see https://github.com/colinhacks/zod/issues/372#issuecomment-2445439772
15
17
  */
16
18
  export const schemaMatches =
17
- <T>() =>
18
- <S extends z.ZodType<T, unknown>>(
19
- schema: AssertEqual<S["_output"], T> extends true
20
- ? S
21
- : S & {
19
+ <Value>() =>
20
+ <Schema extends z.ZodType<Value, unknown>>(
21
+ schema: AssertEqual<Schema["_output"], Value> extends true
22
+ ? Schema
23
+ : Schema & {
22
24
  "types do not match": {
23
- expected: T;
24
- received: S["_output"];
25
+ expected: Value;
26
+ received: Schema["_output"];
25
27
  };
26
28
  }
27
- ): S => {
29
+ ): Schema => {
28
30
  return schema;
29
31
  };
@@ -0,0 +1,57 @@
1
+ import {
2
+ afterAll as vitestAfterAll,
3
+ beforeAll as vitestBeforeAll,
4
+ describe as vitestDescribe,
5
+ test as vitestTest,
6
+ } from "vitest";
7
+
8
+ import { createTestApi } from "../testing/define-api";
9
+ import type { RunnerHooks } from "../testing/runner";
10
+
11
+ export type {
12
+ PhoenixDescribe,
13
+ PhoenixTest,
14
+ PhoenixTestApi,
15
+ PhoenixTestEach,
16
+ } from "../testing/define-api";
17
+
18
+ export type {
19
+ AcceptanceCriterion,
20
+ AcceptanceMetric,
21
+ AcceptanceResult,
22
+ Annotation,
23
+ AnnotatorKind,
24
+ EvaluationParams,
25
+ EvaluationResult,
26
+ Evaluator,
27
+ EvaluatorResult,
28
+ KVMap,
29
+ ReferenceOutput,
30
+ SuiteConfig,
31
+ TestArgs,
32
+ TestConfig,
33
+ TestEachRow,
34
+ TestFn,
35
+ TestParams,
36
+ TestParamsBase,
37
+ } from "../testing/types";
38
+
39
+ export {
40
+ evaluate,
41
+ logAnnotation,
42
+ logOutput,
43
+ traceEvaluator,
44
+ } from "../testing/helpers";
45
+
46
+ const hooks: RunnerHooks = {
47
+ describe: vitestDescribe,
48
+ describeOnly: vitestDescribe.only,
49
+ describeSkip: vitestDescribe.skip,
50
+ test: vitestTest,
51
+ testOnly: vitestTest.only,
52
+ testSkip: vitestTest.skip,
53
+ beforeAll: vitestBeforeAll,
54
+ afterAll: vitestAfterAll,
55
+ };
56
+
57
+ export const { describe, test, it } = createTestApi(() => hooks);
@@ -0,0 +1,32 @@
1
+ import { SuiteSummaryReportRun } from "../testing/report-run";
2
+ import {
3
+ formatSuiteSummary,
4
+ printSuiteSummaries,
5
+ } from "../testing/reporter-format";
6
+
7
+ /**
8
+ * Vitest reporter for `@arizeai/phoenix-client/vitest`.
9
+ *
10
+ * The reporter does not replace Vitest's default test output. Instead it
11
+ * appends a Phoenix-flavored summary at the end of the run that lists
12
+ * outputs, annotations, and dataset/experiment links for each suite.
13
+ *
14
+ * The class implements the Vitest `Reporter` interface structurally rather
15
+ * than nominally so we don't have to import a CJS type from `vitest/reporters`.
16
+ */
17
+ export default class PhoenixVitestReporter {
18
+ private readonly report = new SuiteSummaryReportRun();
19
+
20
+ onTestRunStart(): void {
21
+ this.report.begin();
22
+ }
23
+ onTestRunEnd(): void {
24
+ this.report.finish();
25
+ }
26
+ // Vitest also calls `onFinished` on legacy reporters; alias for safety.
27
+ onFinished(): void {
28
+ this.report.finish();
29
+ }
30
+ }
31
+
32
+ export { formatSuiteSummary, printSuiteSummaries };