@arizeai/phoenix-client 6.10.0 → 6.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +62 -0
- package/dist/esm/__generated__/api/v1.d.ts +452 -28
- package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
- package/dist/esm/experiments/helpers/getExampleGlobalId.d.ts +8 -0
- package/dist/esm/experiments/helpers/getExampleGlobalId.d.ts.map +1 -0
- package/dist/esm/experiments/helpers/getExampleGlobalId.js +9 -0
- package/dist/esm/experiments/helpers/getExampleGlobalId.js.map +1 -0
- package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
- package/dist/esm/experiments/resumeEvaluation.js +2 -1
- package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
- package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
- package/dist/esm/experiments/resumeExperiment.js +3 -2
- package/dist/esm/experiments/resumeExperiment.js.map +1 -1
- package/dist/esm/experiments/runExperiment.d.ts.map +1 -1
- package/dist/esm/experiments/runExperiment.js +6 -3
- package/dist/esm/experiments/runExperiment.js.map +1 -1
- package/dist/esm/jest/index.d.ts +5 -0
- package/dist/esm/jest/index.d.ts.map +1 -0
- package/dist/esm/jest/index.js +49 -0
- package/dist/esm/jest/index.js.map +1 -0
- package/dist/esm/jest/reporter.d.ts +13 -0
- package/dist/esm/jest/reporter.d.ts.map +1 -0
- package/dist/esm/jest/reporter.js +19 -0
- package/dist/esm/jest/reporter.js.map +1 -0
- package/dist/esm/prompts/sdks/toAI.d.ts +2 -2
- package/dist/esm/prompts/sdks/toAI.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toAI.js.map +1 -1
- package/dist/esm/prompts/sdks/toAnthropic.d.ts +2 -2
- package/dist/esm/prompts/sdks/toAnthropic.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toAnthropic.js.map +1 -1
- package/dist/esm/prompts/sdks/toOpenAI.d.ts +2 -2
- package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
- package/dist/esm/prompts/sdks/toSDK.d.ts +8 -8
- package/dist/esm/prompts/sdks/toSDK.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toSDK.js.map +1 -1
- package/dist/esm/prompts/sdks/types.d.ts +2 -2
- package/dist/esm/prompts/sdks/types.d.ts.map +1 -1
- package/dist/esm/schemas/llm/anthropic/converters.d.ts +8 -8
- package/dist/esm/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
- package/dist/esm/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
- package/dist/esm/schemas/llm/constants.d.ts +3 -3
- package/dist/esm/schemas/llm/converters.d.ts +12 -12
- package/dist/esm/schemas/llm/openai/converters.d.ts +3 -3
- package/dist/esm/schemas/llm/schemas.d.ts +2 -2
- package/dist/esm/testing/acceptance.d.ts +20 -0
- package/dist/esm/testing/acceptance.d.ts.map +1 -0
- package/dist/esm/testing/acceptance.js +129 -0
- package/dist/esm/testing/acceptance.js.map +1 -0
- package/dist/esm/testing/define-api.d.ts +157 -0
- package/dist/esm/testing/define-api.d.ts.map +1 -0
- package/dist/esm/testing/define-api.js +78 -0
- package/dist/esm/testing/define-api.js.map +1 -0
- package/dist/esm/testing/helpers.d.ts +55 -0
- package/dist/esm/testing/helpers.d.ts.map +1 -0
- package/dist/esm/testing/helpers.js +179 -0
- package/dist/esm/testing/helpers.js.map +1 -0
- package/dist/esm/testing/phoenix-test-tracking.d.ts +68 -0
- package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -0
- package/dist/esm/testing/phoenix-test-tracking.js +521 -0
- package/dist/esm/testing/phoenix-test-tracking.js.map +1 -0
- package/dist/esm/testing/report-artifacts.d.ts +45 -0
- package/dist/esm/testing/report-artifacts.d.ts.map +1 -0
- package/dist/esm/testing/report-artifacts.js +218 -0
- package/dist/esm/testing/report-artifacts.js.map +1 -0
- package/dist/esm/testing/report-run.d.ts +22 -0
- package/dist/esm/testing/report-run.d.ts.map +1 -0
- package/dist/esm/testing/report-run.js +41 -0
- package/dist/esm/testing/report-run.js.map +1 -0
- package/dist/esm/testing/reporter-format.d.ts +83 -0
- package/dist/esm/testing/reporter-format.d.ts.map +1 -0
- package/dist/esm/testing/reporter-format.js +852 -0
- package/dist/esm/testing/reporter-format.js.map +1 -0
- package/dist/esm/testing/runner.d.ts +31 -0
- package/dist/esm/testing/runner.d.ts.map +1 -0
- package/dist/esm/testing/runner.js +238 -0
- package/dist/esm/testing/runner.js.map +1 -0
- package/dist/esm/testing/state.d.ts +138 -0
- package/dist/esm/testing/state.d.ts.map +1 -0
- package/dist/esm/testing/state.js +31 -0
- package/dist/esm/testing/state.js.map +1 -0
- package/dist/esm/testing/types.d.ts +319 -0
- package/dist/esm/testing/types.d.ts.map +1 -0
- package/dist/esm/testing/types.js +9 -0
- package/dist/esm/testing/types.js.map +1 -0
- package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
- package/dist/esm/utils/channel.d.ts +7 -7
- package/dist/esm/utils/channel.d.ts.map +1 -1
- package/dist/esm/utils/channel.js +1 -1
- package/dist/esm/utils/channel.js.map +1 -1
- package/dist/esm/utils/formatPromptMessages.d.ts.map +1 -1
- package/dist/esm/utils/getPromptBySelector.d.ts.map +1 -1
- package/dist/esm/utils/promisifyResult.d.ts +1 -1
- package/dist/esm/utils/promisifyResult.d.ts.map +1 -1
- package/dist/esm/utils/promisifyResult.js.map +1 -1
- package/dist/esm/utils/schemaMatches.d.ts +5 -5
- package/dist/esm/utils/schemaMatches.d.ts.map +1 -1
- package/dist/esm/utils/schemaMatches.js.map +1 -1
- package/dist/esm/vitest/index.d.ts +5 -0
- package/dist/esm/vitest/index.d.ts.map +1 -0
- package/dist/esm/vitest/index.js +15 -0
- package/dist/esm/vitest/index.js.map +1 -0
- package/dist/esm/vitest/reporter.d.ts +19 -0
- package/dist/esm/vitest/reporter.d.ts.map +1 -0
- package/dist/esm/vitest/reporter.js +27 -0
- package/dist/esm/vitest/reporter.js.map +1 -0
- package/dist/src/__generated__/api/v1.d.ts +452 -28
- package/dist/src/__generated__/api/v1.d.ts.map +1 -1
- package/dist/src/experiments/helpers/getExampleGlobalId.d.ts +8 -0
- package/dist/src/experiments/helpers/getExampleGlobalId.d.ts.map +1 -0
- package/dist/src/experiments/helpers/getExampleGlobalId.js +13 -0
- package/dist/src/experiments/helpers/getExampleGlobalId.js.map +1 -0
- package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
- package/dist/src/experiments/resumeEvaluation.js +2 -1
- package/dist/src/experiments/resumeEvaluation.js.map +1 -1
- package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
- package/dist/src/experiments/resumeExperiment.js +3 -2
- package/dist/src/experiments/resumeExperiment.js.map +1 -1
- package/dist/src/experiments/runExperiment.d.ts.map +1 -1
- package/dist/src/experiments/runExperiment.js +6 -3
- package/dist/src/experiments/runExperiment.js.map +1 -1
- package/dist/src/jest/index.d.ts +5 -0
- package/dist/src/jest/index.d.ts.map +1 -0
- package/dist/src/jest/index.js +58 -0
- package/dist/src/jest/index.js.map +1 -0
- package/dist/src/jest/reporter.d.ts +13 -0
- package/dist/src/jest/reporter.d.ts.map +1 -0
- package/dist/src/jest/reporter.js +23 -0
- package/dist/src/jest/reporter.js.map +1 -0
- package/dist/src/prompts/sdks/toAI.d.ts +2 -2
- package/dist/src/prompts/sdks/toAI.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toAI.js.map +1 -1
- package/dist/src/prompts/sdks/toAnthropic.d.ts +2 -2
- package/dist/src/prompts/sdks/toAnthropic.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toAnthropic.js.map +1 -1
- package/dist/src/prompts/sdks/toOpenAI.d.ts +2 -2
- package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
- package/dist/src/prompts/sdks/toSDK.d.ts +8 -8
- package/dist/src/prompts/sdks/toSDK.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toSDK.js.map +1 -1
- package/dist/src/prompts/sdks/types.d.ts +2 -2
- package/dist/src/prompts/sdks/types.d.ts.map +1 -1
- package/dist/src/schemas/llm/anthropic/converters.d.ts +8 -8
- package/dist/src/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
- package/dist/src/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
- package/dist/src/schemas/llm/constants.d.ts +3 -3
- package/dist/src/schemas/llm/converters.d.ts +12 -12
- package/dist/src/schemas/llm/openai/converters.d.ts +3 -3
- package/dist/src/schemas/llm/schemas.d.ts +2 -2
- package/dist/src/testing/acceptance.d.ts +20 -0
- package/dist/src/testing/acceptance.d.ts.map +1 -0
- package/dist/src/testing/acceptance.js +114 -0
- package/dist/src/testing/acceptance.js.map +1 -0
- package/dist/src/testing/define-api.d.ts +157 -0
- package/dist/src/testing/define-api.d.ts.map +1 -0
- package/dist/src/testing/define-api.js +81 -0
- package/dist/src/testing/define-api.js.map +1 -0
- package/dist/src/testing/helpers.d.ts +55 -0
- package/dist/src/testing/helpers.d.ts.map +1 -0
- package/dist/src/testing/helpers.js +182 -0
- package/dist/src/testing/helpers.js.map +1 -0
- package/dist/src/testing/phoenix-test-tracking.d.ts +68 -0
- package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -0
- package/dist/src/testing/phoenix-test-tracking.js +530 -0
- package/dist/src/testing/phoenix-test-tracking.js.map +1 -0
- package/dist/src/testing/report-artifacts.d.ts +45 -0
- package/dist/src/testing/report-artifacts.d.ts.map +1 -0
- package/dist/src/testing/report-artifacts.js +225 -0
- package/dist/src/testing/report-artifacts.js.map +1 -0
- package/dist/src/testing/report-run.d.ts +22 -0
- package/dist/src/testing/report-run.d.ts.map +1 -0
- package/dist/src/testing/report-run.js +47 -0
- package/dist/src/testing/report-run.js.map +1 -0
- package/dist/src/testing/reporter-format.d.ts +83 -0
- package/dist/src/testing/reporter-format.d.ts.map +1 -0
- package/dist/src/testing/reporter-format.js +870 -0
- package/dist/src/testing/reporter-format.js.map +1 -0
- package/dist/src/testing/runner.d.ts +31 -0
- package/dist/src/testing/runner.d.ts.map +1 -0
- package/dist/src/testing/runner.js +258 -0
- package/dist/src/testing/runner.js.map +1 -0
- package/dist/src/testing/state.d.ts +138 -0
- package/dist/src/testing/state.d.ts.map +1 -0
- package/dist/src/testing/state.js +38 -0
- package/dist/src/testing/state.js.map +1 -0
- package/dist/src/testing/types.d.ts +319 -0
- package/dist/src/testing/types.d.ts.map +1 -0
- package/dist/src/testing/types.js +13 -0
- package/dist/src/testing/types.js.map +1 -0
- package/dist/src/utils/channel.d.ts +7 -7
- package/dist/src/utils/channel.d.ts.map +1 -1
- package/dist/src/utils/channel.js +1 -1
- package/dist/src/utils/channel.js.map +1 -1
- package/dist/src/utils/formatPromptMessages.d.ts.map +1 -1
- package/dist/src/utils/getPromptBySelector.d.ts.map +1 -1
- package/dist/src/utils/promisifyResult.d.ts +1 -1
- package/dist/src/utils/promisifyResult.d.ts.map +1 -1
- package/dist/src/utils/promisifyResult.js.map +1 -1
- package/dist/src/utils/schemaMatches.d.ts +5 -5
- package/dist/src/utils/schemaMatches.d.ts.map +1 -1
- package/dist/src/utils/schemaMatches.js.map +1 -1
- package/dist/src/vitest/index.d.ts +5 -0
- package/dist/src/vitest/index.d.ts.map +1 -0
- package/dist/src/vitest/index.js +23 -0
- package/dist/src/vitest/index.js.map +1 -0
- package/dist/src/vitest/reporter.d.ts +19 -0
- package/dist/src/vitest/reporter.d.ts.map +1 -0
- package/dist/src/vitest/reporter.js +34 -0
- package/dist/src/vitest/reporter.js.map +1 -0
- package/dist/tsconfig.tsbuildinfo +1 -1
- package/docs/ci-evals-annotations.mdx +190 -0
- package/docs/ci-evals-jest.mdx +78 -0
- package/docs/ci-evals-vitest.mdx +240 -0
- package/docs/ci-evals.mdx +263 -0
- package/docs/overview.mdx +9 -1
- package/package.json +49 -17
- package/src/__generated__/api/v1.ts +452 -28
- package/src/experiments/helpers/getExampleGlobalId.ts +12 -0
- package/src/experiments/resumeEvaluation.ts +2 -1
- package/src/experiments/resumeExperiment.ts +3 -2
- package/src/experiments/runExperiment.ts +6 -3
- package/src/jest/index.ts +124 -0
- package/src/jest/reporter.ts +22 -0
- package/src/prompts/sdks/toAI.ts +4 -3
- package/src/prompts/sdks/toAnthropic.ts +4 -3
- package/src/prompts/sdks/toOpenAI.ts +4 -3
- package/src/prompts/sdks/toSDK.ts +16 -11
- package/src/prompts/sdks/types.ts +2 -2
- package/src/testing/acceptance.ts +190 -0
- package/src/testing/define-api.ts +279 -0
- package/src/testing/helpers.ts +251 -0
- package/src/testing/phoenix-test-tracking.ts +637 -0
- package/src/testing/report-artifacts.ts +272 -0
- package/src/testing/report-run.ts +44 -0
- package/src/testing/reporter-format.ts +1072 -0
- package/src/testing/runner.ts +350 -0
- package/src/testing/state.ts +165 -0
- package/src/testing/types.ts +366 -0
- package/src/utils/channel.ts +17 -15
- package/src/utils/promisifyResult.ts +6 -4
- package/src/utils/schemaMatches.ts +12 -10
- package/src/vitest/index.ts +57 -0
- package/src/vitest/reporter.ts +32 -0
|
@@ -0,0 +1,279 @@
|
|
|
1
|
+
import { declareDescribe, declareTest, type RunnerHooks } from "./runner";
|
|
2
|
+
import {
|
|
3
|
+
type KVMap,
|
|
4
|
+
resolveReference,
|
|
5
|
+
type SuiteConfig,
|
|
6
|
+
type TestEachRow,
|
|
7
|
+
type TestFn,
|
|
8
|
+
type TestParams,
|
|
9
|
+
} from "./types";
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Declare Phoenix eval test suites.
|
|
13
|
+
*
|
|
14
|
+
* Drop-in replacement for the test runner's own `describe`. The suite name
|
|
15
|
+
* doubles as the dataset and experiment name on the Phoenix server, and the
|
|
16
|
+
* optional {@link SuiteConfig} controls dataset naming, repetitions, dry-run
|
|
17
|
+
* mode, and CI acceptance criteria.
|
|
18
|
+
*
|
|
19
|
+
* @example
|
|
20
|
+
* ```ts
|
|
21
|
+
* import * as px from "@arizeai/phoenix-client/vitest";
|
|
22
|
+
*
|
|
23
|
+
* px.describe("generate sql demo", () => {
|
|
24
|
+
* px.test("offtopic input", { input: { question: "hi" } }, async ({ input }) => {
|
|
25
|
+
* // ...
|
|
26
|
+
* });
|
|
27
|
+
* }, { metadata: { model: "gpt-4o-mini" } });
|
|
28
|
+
* ```
|
|
29
|
+
*/
|
|
30
|
+
export interface PhoenixDescribe {
|
|
31
|
+
/**
|
|
32
|
+
* Declare a Phoenix eval test suite.
|
|
33
|
+
*
|
|
34
|
+
* @param name - Suite name; doubles as the dataset / experiment name on Phoenix.
|
|
35
|
+
* @param fn - Suite body that declares its `test` / `it` cases.
|
|
36
|
+
* @param config - Optional suite-level config (dataset name, repetitions, dry-run, acceptance criteria).
|
|
37
|
+
*/
|
|
38
|
+
(name: string, fn: () => void, config?: SuiteConfig): void;
|
|
39
|
+
/**
|
|
40
|
+
* Run only this suite, skipping all sibling suites
|
|
41
|
+
* (matches the runner's `describe.only`).
|
|
42
|
+
*
|
|
43
|
+
* @param name - Suite name; doubles as the dataset / experiment name on Phoenix.
|
|
44
|
+
* @param fn - Suite body that declares its `test` / `it` cases.
|
|
45
|
+
* @param config - Optional suite-level config.
|
|
46
|
+
*/
|
|
47
|
+
only(name: string, fn: () => void, config?: SuiteConfig): void;
|
|
48
|
+
/**
|
|
49
|
+
* Skip this suite entirely (matches the runner's `describe.skip`). No dataset
|
|
50
|
+
* or experiment is created on Phoenix.
|
|
51
|
+
*
|
|
52
|
+
* @param name - Suite name; doubles as the dataset / experiment name on Phoenix.
|
|
53
|
+
* @param fn - Suite body (not executed).
|
|
54
|
+
* @param config - Optional suite-level config.
|
|
55
|
+
*/
|
|
56
|
+
skip(name: string, fn: () => void, config?: SuiteConfig): void;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* The test body returned by {@link PhoenixTest.each} after a table is bound.
|
|
61
|
+
*
|
|
62
|
+
* @param name - Test name, or a template (`%i` / `%s` / `%j`), or a function
|
|
63
|
+
* that derives the name from the row and its index.
|
|
64
|
+
* @param fn - The test handler, run once per row in the bound table.
|
|
65
|
+
* @param timeout - Optional per-test timeout in milliseconds.
|
|
66
|
+
*/
|
|
67
|
+
export type PhoenixTestEach<
|
|
68
|
+
Input extends KVMap = KVMap,
|
|
69
|
+
Expected extends KVMap = KVMap,
|
|
70
|
+
> = (
|
|
71
|
+
name: string | ((row: TestEachRow<Input, Expected>, index: number) => string),
|
|
72
|
+
fn: TestFn<Input, Expected>,
|
|
73
|
+
timeout?: number
|
|
74
|
+
) => void;
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Declare a single Phoenix eval test case.
|
|
78
|
+
*
|
|
79
|
+
* Drop-in replacement for the test runner's own `test` / `it`. The `params`
|
|
80
|
+
* argument carries the `input` and the reference output (`expected` /
|
|
81
|
+
* `reference` / `output`) that become the dataset example; whatever the handler
|
|
82
|
+
* returns (or passes to `logOutput()`) is recorded as the experiment run's
|
|
83
|
+
* output and made available to evaluators.
|
|
84
|
+
*
|
|
85
|
+
* `it` is the canonical alias for `test`; the two are identical.
|
|
86
|
+
*
|
|
87
|
+
* @example
|
|
88
|
+
* ```ts
|
|
89
|
+
* px.test(
|
|
90
|
+
* "summarizes the article",
|
|
91
|
+
* { input: { article }, expected: { summary } },
|
|
92
|
+
* async ({ input, expected }) => {
|
|
93
|
+
* const output = await summarize(input.article);
|
|
94
|
+
* px.logOutput(output);
|
|
95
|
+
* await px.evaluate({ name: "matches", evaluate: () => output === expected.summary });
|
|
96
|
+
* }
|
|
97
|
+
* );
|
|
98
|
+
* ```
|
|
99
|
+
*/
|
|
100
|
+
export interface PhoenixTest {
|
|
101
|
+
/**
|
|
102
|
+
* Declare a single Phoenix eval test case.
|
|
103
|
+
*
|
|
104
|
+
* @param name - Test case name; doubles as the dataset example label.
|
|
105
|
+
* @param params - Inline `input` and reference output that become the dataset example.
|
|
106
|
+
* @param fn - Test handler; receives `{ input, expected, metadata }`.
|
|
107
|
+
* @param timeout - Optional per-test timeout in milliseconds.
|
|
108
|
+
*/
|
|
109
|
+
<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
|
|
110
|
+
name: string,
|
|
111
|
+
params: TestParams<Input, Expected>,
|
|
112
|
+
fn: TestFn<Input, Expected>,
|
|
113
|
+
timeout?: number
|
|
114
|
+
): void;
|
|
115
|
+
/**
|
|
116
|
+
* Run only this test case, skipping its siblings
|
|
117
|
+
* (matches the runner's `test.only`).
|
|
118
|
+
*
|
|
119
|
+
* @param name - Test case name; doubles as the dataset example label.
|
|
120
|
+
* @param params - Inline `input` and reference output that become the dataset example.
|
|
121
|
+
* @param fn - Test handler; receives `{ input, expected, metadata }`.
|
|
122
|
+
* @param timeout - Optional per-test timeout in milliseconds.
|
|
123
|
+
*/
|
|
124
|
+
only<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
|
|
125
|
+
name: string,
|
|
126
|
+
params: TestParams<Input, Expected>,
|
|
127
|
+
fn: TestFn<Input, Expected>,
|
|
128
|
+
timeout?: number
|
|
129
|
+
): void;
|
|
130
|
+
/**
|
|
131
|
+
* Skip this test case (matches the runner's `test.skip`). No dataset example
|
|
132
|
+
* or experiment run is created on Phoenix.
|
|
133
|
+
*
|
|
134
|
+
* @param name - Test case name; doubles as the dataset example label.
|
|
135
|
+
* @param params - Inline `input` and reference output (not used while skipped).
|
|
136
|
+
* @param fn - Test handler (not executed).
|
|
137
|
+
* @param timeout - Optional per-test timeout in milliseconds.
|
|
138
|
+
*/
|
|
139
|
+
skip<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
|
|
140
|
+
name: string,
|
|
141
|
+
params: TestParams<Input, Expected>,
|
|
142
|
+
fn: TestFn<Input, Expected>,
|
|
143
|
+
timeout?: number
|
|
144
|
+
): void;
|
|
145
|
+
/**
|
|
146
|
+
* Run the same test handler across many examples. Returns a function that
|
|
147
|
+
* takes a name (or template / name-builder) and the shared test body; each
|
|
148
|
+
* row in `table` becomes its own dataset example and experiment run.
|
|
149
|
+
*
|
|
150
|
+
* @param table - Rows of `{ input, expected?, metadata?, ... }` to fan out over.
|
|
151
|
+
* @returns A {@link PhoenixTestEach} that binds the name and shared handler.
|
|
152
|
+
*
|
|
153
|
+
* @example
|
|
154
|
+
* ```ts
|
|
155
|
+
* px.test.each([
|
|
156
|
+
* { input: { a: 1, b: 2 }, expected: { sum: 3 } },
|
|
157
|
+
* { input: { a: 2, b: 2 }, expected: { sum: 4 } },
|
|
158
|
+
* ])("adds %j", async ({ input, expected }) => {
|
|
159
|
+
* // ...
|
|
160
|
+
* });
|
|
161
|
+
* ```
|
|
162
|
+
*/
|
|
163
|
+
each<Input extends KVMap, Expected extends KVMap>(
|
|
164
|
+
table: TestEachRow<Input, Expected>[]
|
|
165
|
+
): PhoenixTestEach<Input, Expected>;
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/** The public testing surface returned by {@link createTestApi}. */
|
|
169
|
+
export interface PhoenixTestApi {
|
|
170
|
+
/** Declare a Phoenix eval test suite. See {@link PhoenixDescribe}. */
|
|
171
|
+
describe: PhoenixDescribe;
|
|
172
|
+
/** Declare a Phoenix eval test case. See {@link PhoenixTest}. */
|
|
173
|
+
test: PhoenixTest;
|
|
174
|
+
/** Canonical alias for {@link PhoenixTestApi.test}. */
|
|
175
|
+
it: PhoenixTest;
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* Build the public `describe`/`test`/`it` API for a runner adapter.
|
|
180
|
+
*
|
|
181
|
+
* Both the jest and vitest entrypoints expose the identical surface; the only
|
|
182
|
+
* thing that differs between them is how the {@link RunnerHooks} are obtained
|
|
183
|
+
* (vitest imports them statically, jest resolves them lazily from globals).
|
|
184
|
+
* That difference is captured by `getHooks`, which is invoked once per
|
|
185
|
+
* declaration so adapters are free to resolve hooks lazily.
|
|
186
|
+
*
|
|
187
|
+
* The JSDoc that surfaces in editors lives on the {@link PhoenixDescribe} and
|
|
188
|
+
* {@link PhoenixTest} interfaces rather than the implementations below, so the
|
|
189
|
+
* docs survive the `export const { describe, test, it } = createTestApi(...)`
|
|
190
|
+
* destructuring in each adapter.
|
|
191
|
+
*/
|
|
192
|
+
export function createTestApi(getHooks: () => RunnerHooks): PhoenixTestApi {
|
|
193
|
+
const describe = ((
|
|
194
|
+
name: string,
|
|
195
|
+
fn: () => void,
|
|
196
|
+
config?: SuiteConfig
|
|
197
|
+
): void => {
|
|
198
|
+
declareDescribe(getHooks(), name, fn, config ?? {});
|
|
199
|
+
}) as PhoenixDescribe;
|
|
200
|
+
describe.only = (name, fn, config) => {
|
|
201
|
+
declareDescribe(getHooks(), name, fn, config ?? {}, "only");
|
|
202
|
+
};
|
|
203
|
+
describe.skip = (name, fn, config) => {
|
|
204
|
+
declareDescribe(getHooks(), name, fn, config ?? {}, "skip");
|
|
205
|
+
};
|
|
206
|
+
|
|
207
|
+
const test = (<Input extends KVMap = KVMap, Expected extends KVMap = KVMap>(
|
|
208
|
+
name: string,
|
|
209
|
+
params: TestParams<Input, Expected>,
|
|
210
|
+
fn: TestFn<Input, Expected>,
|
|
211
|
+
timeout?: number
|
|
212
|
+
): void => {
|
|
213
|
+
declareTest(getHooks(), name, params, fn, "default", timeout);
|
|
214
|
+
}) as PhoenixTest;
|
|
215
|
+
test.only = (name, params, fn, timeout) => {
|
|
216
|
+
declareTest(getHooks(), name, params, fn, "only", timeout);
|
|
217
|
+
};
|
|
218
|
+
test.skip = (name, params, fn, timeout) => {
|
|
219
|
+
declareTest(getHooks(), name, params, fn, "skip", timeout);
|
|
220
|
+
};
|
|
221
|
+
test.each = <Input extends KVMap, Expected extends KVMap>(
|
|
222
|
+
table: TestEachRow<Input, Expected>[]
|
|
223
|
+
): PhoenixTestEach<Input, Expected> => {
|
|
224
|
+
return (name, fn, timeout) => {
|
|
225
|
+
table.forEach((row, i) => {
|
|
226
|
+
const testName =
|
|
227
|
+
typeof name === "function"
|
|
228
|
+
? name(row, i)
|
|
229
|
+
: interpolateName(name, row, i);
|
|
230
|
+
declareTest(
|
|
231
|
+
getHooks(),
|
|
232
|
+
testName,
|
|
233
|
+
{
|
|
234
|
+
id: row.id,
|
|
235
|
+
input: row.input,
|
|
236
|
+
expected: resolveReference(row),
|
|
237
|
+
metadata: row.metadata,
|
|
238
|
+
splits: row.splits,
|
|
239
|
+
repetitions: row.repetitions,
|
|
240
|
+
dryRun: row.dryRun,
|
|
241
|
+
},
|
|
242
|
+
fn,
|
|
243
|
+
"default",
|
|
244
|
+
timeout
|
|
245
|
+
);
|
|
246
|
+
});
|
|
247
|
+
};
|
|
248
|
+
};
|
|
249
|
+
|
|
250
|
+
// `it` is the canonical alias for `test`.
|
|
251
|
+
const it = test;
|
|
252
|
+
|
|
253
|
+
return { describe, test, it };
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
/**
|
|
257
|
+
* Interpolate a `test.each` name template for a single row. Supports the
|
|
258
|
+
* common `%i`/`%s`/`%j` placeholders for surface parity with the underlying
|
|
259
|
+
* runners; when no placeholder is present the 1-based row index is appended.
|
|
260
|
+
*/
|
|
261
|
+
function interpolateName(
|
|
262
|
+
name: string,
|
|
263
|
+
row: TestEachRow,
|
|
264
|
+
index: number
|
|
265
|
+
): string {
|
|
266
|
+
if (!name.includes("%")) {
|
|
267
|
+
return `${name} #${index + 1}`;
|
|
268
|
+
}
|
|
269
|
+
const replacements: Array<[RegExp, string]> = [
|
|
270
|
+
[/%i/g, String(index)],
|
|
271
|
+
[/%s/g, JSON.stringify(row.input)],
|
|
272
|
+
[/%j/g, JSON.stringify(row)],
|
|
273
|
+
];
|
|
274
|
+
let out = name;
|
|
275
|
+
for (const [pattern, value] of replacements) {
|
|
276
|
+
out = out.replace(pattern, value);
|
|
277
|
+
}
|
|
278
|
+
return out;
|
|
279
|
+
}
|
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
import {
|
|
2
|
+
endTaskSpanForRun,
|
|
3
|
+
postAnnotation,
|
|
4
|
+
runEvaluatorWithTracing,
|
|
5
|
+
} from "./phoenix-test-tracking";
|
|
6
|
+
import { currentRun, type SuiteState } from "./state";
|
|
7
|
+
import type {
|
|
8
|
+
Annotation,
|
|
9
|
+
EvaluationParams,
|
|
10
|
+
EvaluationResult,
|
|
11
|
+
EvaluationResultObject,
|
|
12
|
+
Evaluator,
|
|
13
|
+
KVMap,
|
|
14
|
+
} from "./types";
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* Log the output produced by the test for the current run.
|
|
18
|
+
*
|
|
19
|
+
* Calling this multiple times overwrites the previously recorded value.
|
|
20
|
+
* The argument can be any JSON-serializable value — typically an object
|
|
21
|
+
* matching the shape of the example's `expected` field.
|
|
22
|
+
*/
|
|
23
|
+
export function logOutput(output: unknown): void {
|
|
24
|
+
const run = currentRun();
|
|
25
|
+
if (!run) {
|
|
26
|
+
throw new Error(
|
|
27
|
+
"logOutput() must be called inside a Phoenix eval test body"
|
|
28
|
+
);
|
|
29
|
+
}
|
|
30
|
+
run.output = output;
|
|
31
|
+
run.outputSet = true;
|
|
32
|
+
endTaskSpanForRun(run);
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Record an annotation on the current run.
|
|
37
|
+
*
|
|
38
|
+
* Annotations are collected during the test and posted to Phoenix as
|
|
39
|
+
* experiment evaluations after the test completes. The `name` is the
|
|
40
|
+
* Phoenix evaluation name; `score`, `label`, and `explanation` map to
|
|
41
|
+
* the standard Phoenix `EvaluationResult` fields.
|
|
42
|
+
*
|
|
43
|
+
* The annotation name `"pass"` is reserved — Phoenix eval tests always write
|
|
44
|
+
* a `pass` annotation derived from the test's assertion outcome, so a
|
|
45
|
+
* user-supplied annotation with that name would race / overwrite the
|
|
46
|
+
* built-in one. Such calls are silently ignored.
|
|
47
|
+
*/
|
|
48
|
+
export function logAnnotation(annotation: Annotation): void {
|
|
49
|
+
const run = currentRun();
|
|
50
|
+
if (!run) {
|
|
51
|
+
throw new Error(
|
|
52
|
+
"logAnnotation() must be called inside a Phoenix eval test body"
|
|
53
|
+
);
|
|
54
|
+
}
|
|
55
|
+
if (annotation.name === "pass") return;
|
|
56
|
+
run.annotations.push(annotation);
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Run an evaluator object against the current test run and record the result.
|
|
61
|
+
*
|
|
62
|
+
* The evaluator may come from `@arizeai/phoenix-evals.createEvaluator`,
|
|
63
|
+
* `asExperimentEvaluator`, or any plain object with `{ name, evaluate }`.
|
|
64
|
+
* When `params` is omitted, the current test's `input`, recorded `output`,
|
|
65
|
+
* `expected`, `metadata`, and task `traceId` are supplied.
|
|
66
|
+
*/
|
|
67
|
+
export async function evaluate<
|
|
68
|
+
Params extends KVMap = EvaluationParams & KVMap,
|
|
69
|
+
Result = EvaluationResult,
|
|
70
|
+
>(
|
|
71
|
+
evaluator: Evaluator<Params, Result>,
|
|
72
|
+
params?: Partial<Params> & KVMap
|
|
73
|
+
): Promise<Result> {
|
|
74
|
+
const run = currentRun();
|
|
75
|
+
if (!run) {
|
|
76
|
+
return await evaluator.evaluate((params ?? {}) as Params);
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
if (!run.outputSet && !(params && "output" in params)) {
|
|
80
|
+
warnEvaluateBeforeOutput(run.suite, evaluator.name, run.testName);
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
const evaluatorParams = {
|
|
84
|
+
input: run.params.input,
|
|
85
|
+
// `run.output` is only ever set together with `outputSet`, so it is already
|
|
86
|
+
// `undefined` until a value is recorded.
|
|
87
|
+
output: run.output,
|
|
88
|
+
expected: run.params.expected,
|
|
89
|
+
metadata: run.params.metadata,
|
|
90
|
+
traceId: run.traceId ?? null,
|
|
91
|
+
...(params ?? {}),
|
|
92
|
+
} as unknown as Params;
|
|
93
|
+
|
|
94
|
+
const { result, traceId } = await runEvaluatorWithTracing(
|
|
95
|
+
run.suite,
|
|
96
|
+
evaluator.name,
|
|
97
|
+
evaluatorParams,
|
|
98
|
+
(paramsToEvaluate) => evaluator.evaluate(paramsToEvaluate)
|
|
99
|
+
);
|
|
100
|
+
logAnnotation(
|
|
101
|
+
toAnnotation({
|
|
102
|
+
name: evaluator.name,
|
|
103
|
+
kind: evaluator.kind,
|
|
104
|
+
result,
|
|
105
|
+
traceId,
|
|
106
|
+
})
|
|
107
|
+
);
|
|
108
|
+
return result;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/**
|
|
112
|
+
* Trace an evaluator function so its execution shows up as a separate
|
|
113
|
+
* `EVALUATOR` span in Phoenix and any `{ name, score }`-shaped return
|
|
114
|
+
* value is automatically captured as an annotation on the current run.
|
|
115
|
+
*
|
|
116
|
+
* The annotation name defaults to the traced function's name, falling
|
|
117
|
+
* back to `"evaluator"`.
|
|
118
|
+
*/
|
|
119
|
+
export function traceEvaluator<EvaluatorParams extends KVMap, EvaluatorResult>(
|
|
120
|
+
fn: (params: EvaluatorParams) => EvaluatorResult | Promise<EvaluatorResult>,
|
|
121
|
+
options?: { name?: string }
|
|
122
|
+
): (params: EvaluatorParams) => Promise<EvaluatorResult> {
|
|
123
|
+
const evaluatorName =
|
|
124
|
+
options?.name ?? (fn.name && fn.name !== "" ? fn.name : "evaluator");
|
|
125
|
+
return async (params: EvaluatorParams) => {
|
|
126
|
+
const run = currentRun();
|
|
127
|
+
if (!run) {
|
|
128
|
+
// outside a test context, just call the function plainly
|
|
129
|
+
return await fn(params);
|
|
130
|
+
}
|
|
131
|
+
const { result, traceId } = await runEvaluatorWithTracing(
|
|
132
|
+
run.suite,
|
|
133
|
+
evaluatorName,
|
|
134
|
+
params,
|
|
135
|
+
fn
|
|
136
|
+
);
|
|
137
|
+
if (isAnnotationShaped(result)) {
|
|
138
|
+
logAnnotation({ ...result, traceId });
|
|
139
|
+
}
|
|
140
|
+
return result;
|
|
141
|
+
};
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Warn (at most once per suite) when an evaluator runs before any output was
|
|
146
|
+
* recorded and none was passed explicitly. Such an evaluator receives
|
|
147
|
+
* `output: undefined`, which silently scores against nothing — almost always a
|
|
148
|
+
* forgotten `logOutput()`. Harmless for evaluators that only read `input`.
|
|
149
|
+
*/
|
|
150
|
+
const warnedOutputSuites = new WeakSet<SuiteState>();
|
|
151
|
+
function warnEvaluateBeforeOutput(
|
|
152
|
+
suite: SuiteState,
|
|
153
|
+
evaluatorName: string,
|
|
154
|
+
testName: string
|
|
155
|
+
): void {
|
|
156
|
+
if (warnedOutputSuites.has(suite)) return;
|
|
157
|
+
warnedOutputSuites.add(suite);
|
|
158
|
+
// eslint-disable-next-line no-console
|
|
159
|
+
console.warn(
|
|
160
|
+
`[@arizeai/phoenix-client] evaluate("${evaluatorName}") ran before ` +
|
|
161
|
+
`logOutput() on test "${testName}", so the evaluator received ` +
|
|
162
|
+
`output=undefined. Call logOutput(...) first, or pass { output } ` +
|
|
163
|
+
`explicitly. (Ignore if this evaluator only needs input.)`
|
|
164
|
+
);
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* Normalize an evaluator's return value into an {@link Annotation}. The value
|
|
169
|
+
* is already typed as an {@link EvaluationResult}, so we only dispatch on its
|
|
170
|
+
* runtime shape: a string becomes a `label`, a number/boolean/null becomes a
|
|
171
|
+
* `score`, and an object contributes its `score`/`label`/`explanation`/
|
|
172
|
+
* `metadata` directly.
|
|
173
|
+
*/
|
|
174
|
+
function toAnnotation({
|
|
175
|
+
name,
|
|
176
|
+
kind,
|
|
177
|
+
result,
|
|
178
|
+
traceId,
|
|
179
|
+
}: {
|
|
180
|
+
name: string;
|
|
181
|
+
kind?: Annotation["annotatorKind"];
|
|
182
|
+
result: unknown;
|
|
183
|
+
traceId?: string | null;
|
|
184
|
+
}): Annotation {
|
|
185
|
+
const annotatorKind = kind ?? "CODE";
|
|
186
|
+
if (typeof result === "string") {
|
|
187
|
+
return { name, label: result, annotatorKind, traceId };
|
|
188
|
+
}
|
|
189
|
+
if (
|
|
190
|
+
typeof result === "number" ||
|
|
191
|
+
typeof result === "boolean" ||
|
|
192
|
+
result === null
|
|
193
|
+
) {
|
|
194
|
+
return { name, score: result, annotatorKind, traceId };
|
|
195
|
+
}
|
|
196
|
+
if (typeof result === "object" && !Array.isArray(result)) {
|
|
197
|
+
const { score, label, explanation, metadata } =
|
|
198
|
+
result as EvaluationResultObject;
|
|
199
|
+
return {
|
|
200
|
+
name,
|
|
201
|
+
score,
|
|
202
|
+
label,
|
|
203
|
+
explanation,
|
|
204
|
+
metadata,
|
|
205
|
+
annotatorKind,
|
|
206
|
+
traceId,
|
|
207
|
+
};
|
|
208
|
+
}
|
|
209
|
+
return { name, annotatorKind, traceId };
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
function isAnnotationShaped(value: unknown): value is Annotation {
|
|
213
|
+
if (!value || typeof value !== "object") return false;
|
|
214
|
+
const v = value as { name?: unknown; score?: unknown };
|
|
215
|
+
if (typeof v.name !== "string") return false;
|
|
216
|
+
if (
|
|
217
|
+
v.score !== undefined &&
|
|
218
|
+
typeof v.score !== "number" &&
|
|
219
|
+
typeof v.score !== "boolean" &&
|
|
220
|
+
v.score !== null
|
|
221
|
+
) {
|
|
222
|
+
return false;
|
|
223
|
+
}
|
|
224
|
+
return true;
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
/**
|
|
228
|
+
* Internal: persist all collected annotations for the run.
|
|
229
|
+
*
|
|
230
|
+
* Phoenix's `experiment_evaluations` endpoint is keyed by
|
|
231
|
+
* `(experiment_run_id, name)` so two annotations with the same name on
|
|
232
|
+
* the same run race each other. We collapse duplicates by name (last
|
|
233
|
+
* wins) up front, which makes the final state deterministic; the
|
|
234
|
+
* remaining writes target distinct names, so they post in parallel.
|
|
235
|
+
*/
|
|
236
|
+
export async function flushAnnotations(
|
|
237
|
+
runId: string | undefined,
|
|
238
|
+
annotations: Annotation[],
|
|
239
|
+
suite: SuiteState
|
|
240
|
+
): Promise<void> {
|
|
241
|
+
if (!annotations.length) return;
|
|
242
|
+
const byName = new Map<string, Annotation>();
|
|
243
|
+
for (const annotation of annotations) {
|
|
244
|
+
byName.set(annotation.name, annotation);
|
|
245
|
+
}
|
|
246
|
+
await Promise.all(
|
|
247
|
+
Array.from(byName.values(), (annotation) =>
|
|
248
|
+
postAnnotation(suite, runId, annotation)
|
|
249
|
+
)
|
|
250
|
+
);
|
|
251
|
+
}
|