@arizeai/phoenix-client 6.10.0 → 6.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (244) hide show
  1. package/README.md +62 -0
  2. package/dist/esm/__generated__/api/v1.d.ts +452 -28
  3. package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
  4. package/dist/esm/experiments/helpers/getExampleGlobalId.d.ts +8 -0
  5. package/dist/esm/experiments/helpers/getExampleGlobalId.d.ts.map +1 -0
  6. package/dist/esm/experiments/helpers/getExampleGlobalId.js +9 -0
  7. package/dist/esm/experiments/helpers/getExampleGlobalId.js.map +1 -0
  8. package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
  9. package/dist/esm/experiments/resumeEvaluation.js +2 -1
  10. package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
  11. package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
  12. package/dist/esm/experiments/resumeExperiment.js +3 -2
  13. package/dist/esm/experiments/resumeExperiment.js.map +1 -1
  14. package/dist/esm/experiments/runExperiment.d.ts.map +1 -1
  15. package/dist/esm/experiments/runExperiment.js +6 -3
  16. package/dist/esm/experiments/runExperiment.js.map +1 -1
  17. package/dist/esm/jest/index.d.ts +5 -0
  18. package/dist/esm/jest/index.d.ts.map +1 -0
  19. package/dist/esm/jest/index.js +49 -0
  20. package/dist/esm/jest/index.js.map +1 -0
  21. package/dist/esm/jest/reporter.d.ts +13 -0
  22. package/dist/esm/jest/reporter.d.ts.map +1 -0
  23. package/dist/esm/jest/reporter.js +19 -0
  24. package/dist/esm/jest/reporter.js.map +1 -0
  25. package/dist/esm/prompts/sdks/toAI.d.ts +2 -2
  26. package/dist/esm/prompts/sdks/toAI.d.ts.map +1 -1
  27. package/dist/esm/prompts/sdks/toAI.js.map +1 -1
  28. package/dist/esm/prompts/sdks/toAnthropic.d.ts +2 -2
  29. package/dist/esm/prompts/sdks/toAnthropic.d.ts.map +1 -1
  30. package/dist/esm/prompts/sdks/toAnthropic.js.map +1 -1
  31. package/dist/esm/prompts/sdks/toOpenAI.d.ts +2 -2
  32. package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
  33. package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
  34. package/dist/esm/prompts/sdks/toSDK.d.ts +8 -8
  35. package/dist/esm/prompts/sdks/toSDK.d.ts.map +1 -1
  36. package/dist/esm/prompts/sdks/toSDK.js.map +1 -1
  37. package/dist/esm/prompts/sdks/types.d.ts +2 -2
  38. package/dist/esm/prompts/sdks/types.d.ts.map +1 -1
  39. package/dist/esm/schemas/llm/anthropic/converters.d.ts +8 -8
  40. package/dist/esm/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  41. package/dist/esm/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  42. package/dist/esm/schemas/llm/constants.d.ts +3 -3
  43. package/dist/esm/schemas/llm/converters.d.ts +12 -12
  44. package/dist/esm/schemas/llm/openai/converters.d.ts +3 -3
  45. package/dist/esm/schemas/llm/schemas.d.ts +2 -2
  46. package/dist/esm/testing/acceptance.d.ts +20 -0
  47. package/dist/esm/testing/acceptance.d.ts.map +1 -0
  48. package/dist/esm/testing/acceptance.js +129 -0
  49. package/dist/esm/testing/acceptance.js.map +1 -0
  50. package/dist/esm/testing/define-api.d.ts +157 -0
  51. package/dist/esm/testing/define-api.d.ts.map +1 -0
  52. package/dist/esm/testing/define-api.js +78 -0
  53. package/dist/esm/testing/define-api.js.map +1 -0
  54. package/dist/esm/testing/helpers.d.ts +55 -0
  55. package/dist/esm/testing/helpers.d.ts.map +1 -0
  56. package/dist/esm/testing/helpers.js +179 -0
  57. package/dist/esm/testing/helpers.js.map +1 -0
  58. package/dist/esm/testing/phoenix-test-tracking.d.ts +68 -0
  59. package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -0
  60. package/dist/esm/testing/phoenix-test-tracking.js +521 -0
  61. package/dist/esm/testing/phoenix-test-tracking.js.map +1 -0
  62. package/dist/esm/testing/report-artifacts.d.ts +45 -0
  63. package/dist/esm/testing/report-artifacts.d.ts.map +1 -0
  64. package/dist/esm/testing/report-artifacts.js +218 -0
  65. package/dist/esm/testing/report-artifacts.js.map +1 -0
  66. package/dist/esm/testing/report-run.d.ts +22 -0
  67. package/dist/esm/testing/report-run.d.ts.map +1 -0
  68. package/dist/esm/testing/report-run.js +41 -0
  69. package/dist/esm/testing/report-run.js.map +1 -0
  70. package/dist/esm/testing/reporter-format.d.ts +83 -0
  71. package/dist/esm/testing/reporter-format.d.ts.map +1 -0
  72. package/dist/esm/testing/reporter-format.js +852 -0
  73. package/dist/esm/testing/reporter-format.js.map +1 -0
  74. package/dist/esm/testing/runner.d.ts +31 -0
  75. package/dist/esm/testing/runner.d.ts.map +1 -0
  76. package/dist/esm/testing/runner.js +238 -0
  77. package/dist/esm/testing/runner.js.map +1 -0
  78. package/dist/esm/testing/state.d.ts +138 -0
  79. package/dist/esm/testing/state.d.ts.map +1 -0
  80. package/dist/esm/testing/state.js +31 -0
  81. package/dist/esm/testing/state.js.map +1 -0
  82. package/dist/esm/testing/types.d.ts +319 -0
  83. package/dist/esm/testing/types.d.ts.map +1 -0
  84. package/dist/esm/testing/types.js +9 -0
  85. package/dist/esm/testing/types.js.map +1 -0
  86. package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
  87. package/dist/esm/utils/channel.d.ts +7 -7
  88. package/dist/esm/utils/channel.d.ts.map +1 -1
  89. package/dist/esm/utils/channel.js +1 -1
  90. package/dist/esm/utils/channel.js.map +1 -1
  91. package/dist/esm/utils/formatPromptMessages.d.ts.map +1 -1
  92. package/dist/esm/utils/getPromptBySelector.d.ts.map +1 -1
  93. package/dist/esm/utils/promisifyResult.d.ts +1 -1
  94. package/dist/esm/utils/promisifyResult.d.ts.map +1 -1
  95. package/dist/esm/utils/promisifyResult.js.map +1 -1
  96. package/dist/esm/utils/schemaMatches.d.ts +5 -5
  97. package/dist/esm/utils/schemaMatches.d.ts.map +1 -1
  98. package/dist/esm/utils/schemaMatches.js.map +1 -1
  99. package/dist/esm/vitest/index.d.ts +5 -0
  100. package/dist/esm/vitest/index.d.ts.map +1 -0
  101. package/dist/esm/vitest/index.js +15 -0
  102. package/dist/esm/vitest/index.js.map +1 -0
  103. package/dist/esm/vitest/reporter.d.ts +19 -0
  104. package/dist/esm/vitest/reporter.d.ts.map +1 -0
  105. package/dist/esm/vitest/reporter.js +27 -0
  106. package/dist/esm/vitest/reporter.js.map +1 -0
  107. package/dist/src/__generated__/api/v1.d.ts +452 -28
  108. package/dist/src/__generated__/api/v1.d.ts.map +1 -1
  109. package/dist/src/experiments/helpers/getExampleGlobalId.d.ts +8 -0
  110. package/dist/src/experiments/helpers/getExampleGlobalId.d.ts.map +1 -0
  111. package/dist/src/experiments/helpers/getExampleGlobalId.js +13 -0
  112. package/dist/src/experiments/helpers/getExampleGlobalId.js.map +1 -0
  113. package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
  114. package/dist/src/experiments/resumeEvaluation.js +2 -1
  115. package/dist/src/experiments/resumeEvaluation.js.map +1 -1
  116. package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
  117. package/dist/src/experiments/resumeExperiment.js +3 -2
  118. package/dist/src/experiments/resumeExperiment.js.map +1 -1
  119. package/dist/src/experiments/runExperiment.d.ts.map +1 -1
  120. package/dist/src/experiments/runExperiment.js +6 -3
  121. package/dist/src/experiments/runExperiment.js.map +1 -1
  122. package/dist/src/jest/index.d.ts +5 -0
  123. package/dist/src/jest/index.d.ts.map +1 -0
  124. package/dist/src/jest/index.js +58 -0
  125. package/dist/src/jest/index.js.map +1 -0
  126. package/dist/src/jest/reporter.d.ts +13 -0
  127. package/dist/src/jest/reporter.d.ts.map +1 -0
  128. package/dist/src/jest/reporter.js +23 -0
  129. package/dist/src/jest/reporter.js.map +1 -0
  130. package/dist/src/prompts/sdks/toAI.d.ts +2 -2
  131. package/dist/src/prompts/sdks/toAI.d.ts.map +1 -1
  132. package/dist/src/prompts/sdks/toAI.js.map +1 -1
  133. package/dist/src/prompts/sdks/toAnthropic.d.ts +2 -2
  134. package/dist/src/prompts/sdks/toAnthropic.d.ts.map +1 -1
  135. package/dist/src/prompts/sdks/toAnthropic.js.map +1 -1
  136. package/dist/src/prompts/sdks/toOpenAI.d.ts +2 -2
  137. package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
  138. package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
  139. package/dist/src/prompts/sdks/toSDK.d.ts +8 -8
  140. package/dist/src/prompts/sdks/toSDK.d.ts.map +1 -1
  141. package/dist/src/prompts/sdks/toSDK.js.map +1 -1
  142. package/dist/src/prompts/sdks/types.d.ts +2 -2
  143. package/dist/src/prompts/sdks/types.d.ts.map +1 -1
  144. package/dist/src/schemas/llm/anthropic/converters.d.ts +8 -8
  145. package/dist/src/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
  146. package/dist/src/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
  147. package/dist/src/schemas/llm/constants.d.ts +3 -3
  148. package/dist/src/schemas/llm/converters.d.ts +12 -12
  149. package/dist/src/schemas/llm/openai/converters.d.ts +3 -3
  150. package/dist/src/schemas/llm/schemas.d.ts +2 -2
  151. package/dist/src/testing/acceptance.d.ts +20 -0
  152. package/dist/src/testing/acceptance.d.ts.map +1 -0
  153. package/dist/src/testing/acceptance.js +114 -0
  154. package/dist/src/testing/acceptance.js.map +1 -0
  155. package/dist/src/testing/define-api.d.ts +157 -0
  156. package/dist/src/testing/define-api.d.ts.map +1 -0
  157. package/dist/src/testing/define-api.js +81 -0
  158. package/dist/src/testing/define-api.js.map +1 -0
  159. package/dist/src/testing/helpers.d.ts +55 -0
  160. package/dist/src/testing/helpers.d.ts.map +1 -0
  161. package/dist/src/testing/helpers.js +182 -0
  162. package/dist/src/testing/helpers.js.map +1 -0
  163. package/dist/src/testing/phoenix-test-tracking.d.ts +68 -0
  164. package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -0
  165. package/dist/src/testing/phoenix-test-tracking.js +530 -0
  166. package/dist/src/testing/phoenix-test-tracking.js.map +1 -0
  167. package/dist/src/testing/report-artifacts.d.ts +45 -0
  168. package/dist/src/testing/report-artifacts.d.ts.map +1 -0
  169. package/dist/src/testing/report-artifacts.js +225 -0
  170. package/dist/src/testing/report-artifacts.js.map +1 -0
  171. package/dist/src/testing/report-run.d.ts +22 -0
  172. package/dist/src/testing/report-run.d.ts.map +1 -0
  173. package/dist/src/testing/report-run.js +47 -0
  174. package/dist/src/testing/report-run.js.map +1 -0
  175. package/dist/src/testing/reporter-format.d.ts +83 -0
  176. package/dist/src/testing/reporter-format.d.ts.map +1 -0
  177. package/dist/src/testing/reporter-format.js +870 -0
  178. package/dist/src/testing/reporter-format.js.map +1 -0
  179. package/dist/src/testing/runner.d.ts +31 -0
  180. package/dist/src/testing/runner.d.ts.map +1 -0
  181. package/dist/src/testing/runner.js +258 -0
  182. package/dist/src/testing/runner.js.map +1 -0
  183. package/dist/src/testing/state.d.ts +138 -0
  184. package/dist/src/testing/state.d.ts.map +1 -0
  185. package/dist/src/testing/state.js +38 -0
  186. package/dist/src/testing/state.js.map +1 -0
  187. package/dist/src/testing/types.d.ts +319 -0
  188. package/dist/src/testing/types.d.ts.map +1 -0
  189. package/dist/src/testing/types.js +13 -0
  190. package/dist/src/testing/types.js.map +1 -0
  191. package/dist/src/utils/channel.d.ts +7 -7
  192. package/dist/src/utils/channel.d.ts.map +1 -1
  193. package/dist/src/utils/channel.js +1 -1
  194. package/dist/src/utils/channel.js.map +1 -1
  195. package/dist/src/utils/formatPromptMessages.d.ts.map +1 -1
  196. package/dist/src/utils/getPromptBySelector.d.ts.map +1 -1
  197. package/dist/src/utils/promisifyResult.d.ts +1 -1
  198. package/dist/src/utils/promisifyResult.d.ts.map +1 -1
  199. package/dist/src/utils/promisifyResult.js.map +1 -1
  200. package/dist/src/utils/schemaMatches.d.ts +5 -5
  201. package/dist/src/utils/schemaMatches.d.ts.map +1 -1
  202. package/dist/src/utils/schemaMatches.js.map +1 -1
  203. package/dist/src/vitest/index.d.ts +5 -0
  204. package/dist/src/vitest/index.d.ts.map +1 -0
  205. package/dist/src/vitest/index.js +23 -0
  206. package/dist/src/vitest/index.js.map +1 -0
  207. package/dist/src/vitest/reporter.d.ts +19 -0
  208. package/dist/src/vitest/reporter.d.ts.map +1 -0
  209. package/dist/src/vitest/reporter.js +34 -0
  210. package/dist/src/vitest/reporter.js.map +1 -0
  211. package/dist/tsconfig.tsbuildinfo +1 -1
  212. package/docs/ci-evals-annotations.mdx +190 -0
  213. package/docs/ci-evals-jest.mdx +78 -0
  214. package/docs/ci-evals-vitest.mdx +240 -0
  215. package/docs/ci-evals.mdx +263 -0
  216. package/docs/overview.mdx +9 -1
  217. package/package.json +49 -17
  218. package/src/__generated__/api/v1.ts +452 -28
  219. package/src/experiments/helpers/getExampleGlobalId.ts +12 -0
  220. package/src/experiments/resumeEvaluation.ts +2 -1
  221. package/src/experiments/resumeExperiment.ts +3 -2
  222. package/src/experiments/runExperiment.ts +6 -3
  223. package/src/jest/index.ts +124 -0
  224. package/src/jest/reporter.ts +22 -0
  225. package/src/prompts/sdks/toAI.ts +4 -3
  226. package/src/prompts/sdks/toAnthropic.ts +4 -3
  227. package/src/prompts/sdks/toOpenAI.ts +4 -3
  228. package/src/prompts/sdks/toSDK.ts +16 -11
  229. package/src/prompts/sdks/types.ts +2 -2
  230. package/src/testing/acceptance.ts +190 -0
  231. package/src/testing/define-api.ts +279 -0
  232. package/src/testing/helpers.ts +251 -0
  233. package/src/testing/phoenix-test-tracking.ts +637 -0
  234. package/src/testing/report-artifacts.ts +272 -0
  235. package/src/testing/report-run.ts +44 -0
  236. package/src/testing/reporter-format.ts +1072 -0
  237. package/src/testing/runner.ts +350 -0
  238. package/src/testing/state.ts +165 -0
  239. package/src/testing/types.ts +366 -0
  240. package/src/utils/channel.ts +17 -15
  241. package/src/utils/promisifyResult.ts +6 -4
  242. package/src/utils/schemaMatches.ts +12 -10
  243. package/src/vitest/index.ts +57 -0
  244. package/src/vitest/reporter.ts +32 -0
@@ -0,0 +1,279 @@
1
+ import { declareDescribe, declareTest, type RunnerHooks } from "./runner";
2
+ import {
3
+ type KVMap,
4
+ resolveReference,
5
+ type SuiteConfig,
6
+ type TestEachRow,
7
+ type TestFn,
8
+ type TestParams,
9
+ } from "./types";
10
+
11
+ /**
12
+ * Declare Phoenix eval test suites.
13
+ *
14
+ * Drop-in replacement for the test runner's own `describe`. The suite name
15
+ * doubles as the dataset and experiment name on the Phoenix server, and the
16
+ * optional {@link SuiteConfig} controls dataset naming, repetitions, dry-run
17
+ * mode, and CI acceptance criteria.
18
+ *
19
+ * @example
20
+ * ```ts
21
+ * import * as px from "@arizeai/phoenix-client/vitest";
22
+ *
23
+ * px.describe("generate sql demo", () => {
24
+ * px.test("offtopic input", { input: { question: "hi" } }, async ({ input }) => {
25
+ * // ...
26
+ * });
27
+ * }, { metadata: { model: "gpt-4o-mini" } });
28
+ * ```
29
+ */
30
+ export interface PhoenixDescribe {
31
+ /**
32
+ * Declare a Phoenix eval test suite.
33
+ *
34
+ * @param name - Suite name; doubles as the dataset / experiment name on Phoenix.
35
+ * @param fn - Suite body that declares its `test` / `it` cases.
36
+ * @param config - Optional suite-level config (dataset name, repetitions, dry-run, acceptance criteria).
37
+ */
38
+ (name: string, fn: () => void, config?: SuiteConfig): void;
39
+ /**
40
+ * Run only this suite, skipping all sibling suites
41
+ * (matches the runner's `describe.only`).
42
+ *
43
+ * @param name - Suite name; doubles as the dataset / experiment name on Phoenix.
44
+ * @param fn - Suite body that declares its `test` / `it` cases.
45
+ * @param config - Optional suite-level config.
46
+ */
47
+ only(name: string, fn: () => void, config?: SuiteConfig): void;
48
+ /**
49
+ * Skip this suite entirely (matches the runner's `describe.skip`). No dataset
50
+ * or experiment is created on Phoenix.
51
+ *
52
+ * @param name - Suite name; doubles as the dataset / experiment name on Phoenix.
53
+ * @param fn - Suite body (not executed).
54
+ * @param config - Optional suite-level config.
55
+ */
56
+ skip(name: string, fn: () => void, config?: SuiteConfig): void;
57
+ }
58
+
59
+ /**
60
+ * The test body returned by {@link PhoenixTest.each} after a table is bound.
61
+ *
62
+ * @param name - Test name, or a template (`%i` / `%s` / `%j`), or a function
63
+ * that derives the name from the row and its index.
64
+ * @param fn - The test handler, run once per row in the bound table.
65
+ * @param timeout - Optional per-test timeout in milliseconds.
66
+ */
67
+ export type PhoenixTestEach<
68
+ Input extends KVMap = KVMap,
69
+ Expected extends KVMap = KVMap,
70
+ > = (
71
+ name: string | ((row: TestEachRow<Input, Expected>, index: number) => string),
72
+ fn: TestFn<Input, Expected>,
73
+ timeout?: number
74
+ ) => void;
75
+
76
+ /**
77
+ * Declare a single Phoenix eval test case.
78
+ *
79
+ * Drop-in replacement for the test runner's own `test` / `it`. The `params`
80
+ * argument carries the `input` and the reference output (`expected` /
81
+ * `reference` / `output`) that become the dataset example; whatever the handler
82
+ * returns (or passes to `logOutput()`) is recorded as the experiment run's
83
+ * output and made available to evaluators.
84
+ *
85
+ * `it` is the canonical alias for `test`; the two are identical.
86
+ *
87
+ * @example
88
+ * ```ts
89
+ * px.test(
90
+ * "summarizes the article",
91
+ * { input: { article }, expected: { summary } },
92
+ * async ({ input, expected }) => {
93
+ * const output = await summarize(input.article);
94
+ * px.logOutput(output);
95
+ * await px.evaluate({ name: "matches", evaluate: () => output === expected.summary });
96
+ * }
97
+ * );
98
+ * ```
99
+ */
100
+ export interface PhoenixTest {
101
+ /**
102
+ * Declare a single Phoenix eval test case.
103
+ *
104
+ * @param name - Test case name; doubles as the dataset example label.
105
+ * @param params - Inline `input` and reference output that become the dataset example.
106
+ * @param fn - Test handler; receives `{ input, expected, metadata }`.
107
+ * @param timeout - Optional per-test timeout in milliseconds.
108
+ */
109
+ <Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
110
+ name: string,
111
+ params: TestParams<Input, Expected>,
112
+ fn: TestFn<Input, Expected>,
113
+ timeout?: number
114
+ ): void;
115
+ /**
116
+ * Run only this test case, skipping its siblings
117
+ * (matches the runner's `test.only`).
118
+ *
119
+ * @param name - Test case name; doubles as the dataset example label.
120
+ * @param params - Inline `input` and reference output that become the dataset example.
121
+ * @param fn - Test handler; receives `{ input, expected, metadata }`.
122
+ * @param timeout - Optional per-test timeout in milliseconds.
123
+ */
124
+ only<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
125
+ name: string,
126
+ params: TestParams<Input, Expected>,
127
+ fn: TestFn<Input, Expected>,
128
+ timeout?: number
129
+ ): void;
130
+ /**
131
+ * Skip this test case (matches the runner's `test.skip`). No dataset example
132
+ * or experiment run is created on Phoenix.
133
+ *
134
+ * @param name - Test case name; doubles as the dataset example label.
135
+ * @param params - Inline `input` and reference output (not used while skipped).
136
+ * @param fn - Test handler (not executed).
137
+ * @param timeout - Optional per-test timeout in milliseconds.
138
+ */
139
+ skip<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
140
+ name: string,
141
+ params: TestParams<Input, Expected>,
142
+ fn: TestFn<Input, Expected>,
143
+ timeout?: number
144
+ ): void;
145
+ /**
146
+ * Run the same test handler across many examples. Returns a function that
147
+ * takes a name (or template / name-builder) and the shared test body; each
148
+ * row in `table` becomes its own dataset example and experiment run.
149
+ *
150
+ * @param table - Rows of `{ input, expected?, metadata?, ... }` to fan out over.
151
+ * @returns A {@link PhoenixTestEach} that binds the name and shared handler.
152
+ *
153
+ * @example
154
+ * ```ts
155
+ * px.test.each([
156
+ * { input: { a: 1, b: 2 }, expected: { sum: 3 } },
157
+ * { input: { a: 2, b: 2 }, expected: { sum: 4 } },
158
+ * ])("adds %j", async ({ input, expected }) => {
159
+ * // ...
160
+ * });
161
+ * ```
162
+ */
163
+ each<Input extends KVMap, Expected extends KVMap>(
164
+ table: TestEachRow<Input, Expected>[]
165
+ ): PhoenixTestEach<Input, Expected>;
166
+ }
167
+
168
+ /** The public testing surface returned by {@link createTestApi}. */
169
+ export interface PhoenixTestApi {
170
+ /** Declare a Phoenix eval test suite. See {@link PhoenixDescribe}. */
171
+ describe: PhoenixDescribe;
172
+ /** Declare a Phoenix eval test case. See {@link PhoenixTest}. */
173
+ test: PhoenixTest;
174
+ /** Canonical alias for {@link PhoenixTestApi.test}. */
175
+ it: PhoenixTest;
176
+ }
177
+
178
+ /**
179
+ * Build the public `describe`/`test`/`it` API for a runner adapter.
180
+ *
181
+ * Both the jest and vitest entrypoints expose the identical surface; the only
182
+ * thing that differs between them is how the {@link RunnerHooks} are obtained
183
+ * (vitest imports them statically, jest resolves them lazily from globals).
184
+ * That difference is captured by `getHooks`, which is invoked once per
185
+ * declaration so adapters are free to resolve hooks lazily.
186
+ *
187
+ * The JSDoc that surfaces in editors lives on the {@link PhoenixDescribe} and
188
+ * {@link PhoenixTest} interfaces rather than the implementations below, so the
189
+ * docs survive the `export const { describe, test, it } = createTestApi(...)`
190
+ * destructuring in each adapter.
191
+ */
192
+ export function createTestApi(getHooks: () => RunnerHooks): PhoenixTestApi {
193
+ const describe = ((
194
+ name: string,
195
+ fn: () => void,
196
+ config?: SuiteConfig
197
+ ): void => {
198
+ declareDescribe(getHooks(), name, fn, config ?? {});
199
+ }) as PhoenixDescribe;
200
+ describe.only = (name, fn, config) => {
201
+ declareDescribe(getHooks(), name, fn, config ?? {}, "only");
202
+ };
203
+ describe.skip = (name, fn, config) => {
204
+ declareDescribe(getHooks(), name, fn, config ?? {}, "skip");
205
+ };
206
+
207
+ const test = (<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
208
+ name: string,
209
+ params: TestParams<Input, Expected>,
210
+ fn: TestFn<Input, Expected>,
211
+ timeout?: number
212
+ ): void => {
213
+ declareTest(getHooks(), name, params, fn, "default", timeout);
214
+ }) as PhoenixTest;
215
+ test.only = (name, params, fn, timeout) => {
216
+ declareTest(getHooks(), name, params, fn, "only", timeout);
217
+ };
218
+ test.skip = (name, params, fn, timeout) => {
219
+ declareTest(getHooks(), name, params, fn, "skip", timeout);
220
+ };
221
+ test.each = <Input extends KVMap, Expected extends KVMap>(
222
+ table: TestEachRow<Input, Expected>[]
223
+ ): PhoenixTestEach<Input, Expected> => {
224
+ return (name, fn, timeout) => {
225
+ table.forEach((row, i) => {
226
+ const testName =
227
+ typeof name === "function"
228
+ ? name(row, i)
229
+ : interpolateName(name, row, i);
230
+ declareTest(
231
+ getHooks(),
232
+ testName,
233
+ {
234
+ id: row.id,
235
+ input: row.input,
236
+ expected: resolveReference(row),
237
+ metadata: row.metadata,
238
+ splits: row.splits,
239
+ repetitions: row.repetitions,
240
+ dryRun: row.dryRun,
241
+ },
242
+ fn,
243
+ "default",
244
+ timeout
245
+ );
246
+ });
247
+ };
248
+ };
249
+
250
+ // `it` is the canonical alias for `test`.
251
+ const it = test;
252
+
253
+ return { describe, test, it };
254
+ }
255
+
256
+ /**
257
+ * Interpolate a `test.each` name template for a single row. Supports the
258
+ * common `%i`/`%s`/`%j` placeholders for surface parity with the underlying
259
+ * runners; when no placeholder is present the 1-based row index is appended.
260
+ */
261
+ function interpolateName(
262
+ name: string,
263
+ row: TestEachRow,
264
+ index: number
265
+ ): string {
266
+ if (!name.includes("%")) {
267
+ return `${name} #${index + 1}`;
268
+ }
269
+ const replacements: Array<[RegExp, string]> = [
270
+ [/%i/g, String(index)],
271
+ [/%s/g, JSON.stringify(row.input)],
272
+ [/%j/g, JSON.stringify(row)],
273
+ ];
274
+ let out = name;
275
+ for (const [pattern, value] of replacements) {
276
+ out = out.replace(pattern, value);
277
+ }
278
+ return out;
279
+ }
@@ -0,0 +1,251 @@
1
+ import {
2
+ endTaskSpanForRun,
3
+ postAnnotation,
4
+ runEvaluatorWithTracing,
5
+ } from "./phoenix-test-tracking";
6
+ import { currentRun, type SuiteState } from "./state";
7
+ import type {
8
+ Annotation,
9
+ EvaluationParams,
10
+ EvaluationResult,
11
+ EvaluationResultObject,
12
+ Evaluator,
13
+ KVMap,
14
+ } from "./types";
15
+
16
+ /**
17
+ * Log the output produced by the test for the current run.
18
+ *
19
+ * Calling this multiple times overwrites the previously recorded value.
20
+ * The argument can be any JSON-serializable value — typically an object
21
+ * matching the shape of the example's `expected` field.
22
+ */
23
+ export function logOutput(output: unknown): void {
24
+ const run = currentRun();
25
+ if (!run) {
26
+ throw new Error(
27
+ "logOutput() must be called inside a Phoenix eval test body"
28
+ );
29
+ }
30
+ run.output = output;
31
+ run.outputSet = true;
32
+ endTaskSpanForRun(run);
33
+ }
34
+
35
+ /**
36
+ * Record an annotation on the current run.
37
+ *
38
+ * Annotations are collected during the test and posted to Phoenix as
39
+ * experiment evaluations after the test completes. The `name` is the
40
+ * Phoenix evaluation name; `score`, `label`, and `explanation` map to
41
+ * the standard Phoenix `EvaluationResult` fields.
42
+ *
43
+ * The annotation name `"pass"` is reserved — Phoenix eval tests always write
44
+ * a `pass` annotation derived from the test's assertion outcome, so a
45
+ * user-supplied annotation with that name would race / overwrite the
46
+ * built-in one. Such calls are silently ignored.
47
+ */
48
+ export function logAnnotation(annotation: Annotation): void {
49
+ const run = currentRun();
50
+ if (!run) {
51
+ throw new Error(
52
+ "logAnnotation() must be called inside a Phoenix eval test body"
53
+ );
54
+ }
55
+ if (annotation.name === "pass") return;
56
+ run.annotations.push(annotation);
57
+ }
58
+
59
+ /**
60
+ * Run an evaluator object against the current test run and record the result.
61
+ *
62
+ * The evaluator may come from `@arizeai/phoenix-evals.createEvaluator`,
63
+ * `asExperimentEvaluator`, or any plain object with `{ name, evaluate }`.
64
+ * When `params` is omitted, the current test's `input`, recorded `output`,
65
+ * `expected`, `metadata`, and task `traceId` are supplied.
66
+ */
67
+ export async function evaluate<
68
+ Params extends KVMap = EvaluationParams & KVMap,
69
+ Result = EvaluationResult,
70
+ >(
71
+ evaluator: Evaluator<Params, Result>,
72
+ params?: Partial<Params> & KVMap
73
+ ): Promise<Result> {
74
+ const run = currentRun();
75
+ if (!run) {
76
+ return await evaluator.evaluate((params ?? {}) as Params);
77
+ }
78
+
79
+ if (!run.outputSet && !(params && "output" in params)) {
80
+ warnEvaluateBeforeOutput(run.suite, evaluator.name, run.testName);
81
+ }
82
+
83
+ const evaluatorParams = {
84
+ input: run.params.input,
85
+ // `run.output` is only ever set together with `outputSet`, so it is already
86
+ // `undefined` until a value is recorded.
87
+ output: run.output,
88
+ expected: run.params.expected,
89
+ metadata: run.params.metadata,
90
+ traceId: run.traceId ?? null,
91
+ ...(params ?? {}),
92
+ } as unknown as Params;
93
+
94
+ const { result, traceId } = await runEvaluatorWithTracing(
95
+ run.suite,
96
+ evaluator.name,
97
+ evaluatorParams,
98
+ (paramsToEvaluate) => evaluator.evaluate(paramsToEvaluate)
99
+ );
100
+ logAnnotation(
101
+ toAnnotation({
102
+ name: evaluator.name,
103
+ kind: evaluator.kind,
104
+ result,
105
+ traceId,
106
+ })
107
+ );
108
+ return result;
109
+ }
110
+
111
+ /**
112
+ * Trace an evaluator function so its execution shows up as a separate
113
+ * `EVALUATOR` span in Phoenix and any `{ name, score }`-shaped return
114
+ * value is automatically captured as an annotation on the current run.
115
+ *
116
+ * The annotation name defaults to the traced function's name, falling
117
+ * back to `"evaluator"`.
118
+ */
119
+ export function traceEvaluator<EvaluatorParams extends KVMap, EvaluatorResult>(
120
+ fn: (params: EvaluatorParams) => EvaluatorResult | Promise<EvaluatorResult>,
121
+ options?: { name?: string }
122
+ ): (params: EvaluatorParams) => Promise<EvaluatorResult> {
123
+ const evaluatorName =
124
+ options?.name ?? (fn.name && fn.name !== "" ? fn.name : "evaluator");
125
+ return async (params: EvaluatorParams) => {
126
+ const run = currentRun();
127
+ if (!run) {
128
+ // outside a test context, just call the function plainly
129
+ return await fn(params);
130
+ }
131
+ const { result, traceId } = await runEvaluatorWithTracing(
132
+ run.suite,
133
+ evaluatorName,
134
+ params,
135
+ fn
136
+ );
137
+ if (isAnnotationShaped(result)) {
138
+ logAnnotation({ ...result, traceId });
139
+ }
140
+ return result;
141
+ };
142
+ }
143
+
144
+ /**
145
+ * Warn (at most once per suite) when an evaluator runs before any output was
146
+ * recorded and none was passed explicitly. Such an evaluator receives
147
+ * `output: undefined`, which silently scores against nothing — almost always a
148
+ * forgotten `logOutput()`. Harmless for evaluators that only read `input`.
149
+ */
150
+ const warnedOutputSuites = new WeakSet<SuiteState>();
151
+ function warnEvaluateBeforeOutput(
152
+ suite: SuiteState,
153
+ evaluatorName: string,
154
+ testName: string
155
+ ): void {
156
+ if (warnedOutputSuites.has(suite)) return;
157
+ warnedOutputSuites.add(suite);
158
+ // eslint-disable-next-line no-console
159
+ console.warn(
160
+ `[@arizeai/phoenix-client] evaluate("${evaluatorName}") ran before ` +
161
+ `logOutput() on test "${testName}", so the evaluator received ` +
162
+ `output=undefined. Call logOutput(...) first, or pass { output } ` +
163
+ `explicitly. (Ignore if this evaluator only needs input.)`
164
+ );
165
+ }
166
+
167
+ /**
168
+ * Normalize an evaluator's return value into an {@link Annotation}. The value
169
+ * is already typed as an {@link EvaluationResult}, so we only dispatch on its
170
+ * runtime shape: a string becomes a `label`, a number/boolean/null becomes a
171
+ * `score`, and an object contributes its `score`/`label`/`explanation`/
172
+ * `metadata` directly.
173
+ */
174
+ function toAnnotation({
175
+ name,
176
+ kind,
177
+ result,
178
+ traceId,
179
+ }: {
180
+ name: string;
181
+ kind?: Annotation["annotatorKind"];
182
+ result: unknown;
183
+ traceId?: string | null;
184
+ }): Annotation {
185
+ const annotatorKind = kind ?? "CODE";
186
+ if (typeof result === "string") {
187
+ return { name, label: result, annotatorKind, traceId };
188
+ }
189
+ if (
190
+ typeof result === "number" ||
191
+ typeof result === "boolean" ||
192
+ result === null
193
+ ) {
194
+ return { name, score: result, annotatorKind, traceId };
195
+ }
196
+ if (typeof result === "object" && !Array.isArray(result)) {
197
+ const { score, label, explanation, metadata } =
198
+ result as EvaluationResultObject;
199
+ return {
200
+ name,
201
+ score,
202
+ label,
203
+ explanation,
204
+ metadata,
205
+ annotatorKind,
206
+ traceId,
207
+ };
208
+ }
209
+ return { name, annotatorKind, traceId };
210
+ }
211
+
212
+ function isAnnotationShaped(value: unknown): value is Annotation {
213
+ if (!value || typeof value !== "object") return false;
214
+ const v = value as { name?: unknown; score?: unknown };
215
+ if (typeof v.name !== "string") return false;
216
+ if (
217
+ v.score !== undefined &&
218
+ typeof v.score !== "number" &&
219
+ typeof v.score !== "boolean" &&
220
+ v.score !== null
221
+ ) {
222
+ return false;
223
+ }
224
+ return true;
225
+ }
226
+
227
+ /**
228
+ * Internal: persist all collected annotations for the run.
229
+ *
230
+ * Phoenix's `experiment_evaluations` endpoint is keyed by
231
+ * `(experiment_run_id, name)` so two annotations with the same name on
232
+ * the same run race each other. We collapse duplicates by name (last
233
+ * wins) up front, which makes the final state deterministic; the
234
+ * remaining writes target distinct names, so they post in parallel.
235
+ */
236
+ export async function flushAnnotations(
237
+ runId: string | undefined,
238
+ annotations: Annotation[],
239
+ suite: SuiteState
240
+ ): Promise<void> {
241
+ if (!annotations.length) return;
242
+ const byName = new Map<string, Annotation>();
243
+ for (const annotation of annotations) {
244
+ byName.set(annotation.name, annotation);
245
+ }
246
+ await Promise.all(
247
+ Array.from(byName.values(), (annotation) =>
248
+ postAnnotation(suite, runId, annotation)
249
+ )
250
+ );
251
+ }