@arizeai/phoenix-client 6.10.1 → 6.11.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +62 -0
- package/dist/esm/__generated__/api/v1.d.ts +254 -2
- package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
- package/dist/esm/jest/index.d.ts +5 -0
- package/dist/esm/jest/index.d.ts.map +1 -0
- package/dist/esm/jest/index.js +49 -0
- package/dist/esm/jest/index.js.map +1 -0
- package/dist/esm/jest/reporter.d.ts +13 -0
- package/dist/esm/jest/reporter.d.ts.map +1 -0
- package/dist/esm/jest/reporter.js +19 -0
- package/dist/esm/jest/reporter.js.map +1 -0
- package/dist/esm/prompts/sdks/toAI.d.ts +2 -2
- package/dist/esm/prompts/sdks/toAI.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toAI.js.map +1 -1
- package/dist/esm/prompts/sdks/toAnthropic.d.ts +2 -2
- package/dist/esm/prompts/sdks/toAnthropic.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toAnthropic.js.map +1 -1
- package/dist/esm/prompts/sdks/toOpenAI.d.ts +2 -2
- package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
- package/dist/esm/prompts/sdks/toSDK.d.ts +8 -8
- package/dist/esm/prompts/sdks/toSDK.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toSDK.js.map +1 -1
- package/dist/esm/prompts/sdks/types.d.ts +2 -2
- package/dist/esm/prompts/sdks/types.d.ts.map +1 -1
- package/dist/esm/schemas/llm/anthropic/converters.d.ts +8 -8
- package/dist/esm/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
- package/dist/esm/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
- package/dist/esm/schemas/llm/constants.d.ts +3 -3
- package/dist/esm/schemas/llm/converters.d.ts +12 -12
- package/dist/esm/schemas/llm/openai/converters.d.ts +3 -3
- package/dist/esm/schemas/llm/schemas.d.ts +2 -2
- package/dist/esm/testing/acceptance.d.ts +20 -0
- package/dist/esm/testing/acceptance.d.ts.map +1 -0
- package/dist/esm/testing/acceptance.js +129 -0
- package/dist/esm/testing/acceptance.js.map +1 -0
- package/dist/esm/testing/define-api.d.ts +157 -0
- package/dist/esm/testing/define-api.d.ts.map +1 -0
- package/dist/esm/testing/define-api.js +78 -0
- package/dist/esm/testing/define-api.js.map +1 -0
- package/dist/esm/testing/helpers.d.ts +55 -0
- package/dist/esm/testing/helpers.d.ts.map +1 -0
- package/dist/esm/testing/helpers.js +179 -0
- package/dist/esm/testing/helpers.js.map +1 -0
- package/dist/esm/testing/phoenix-test-tracking.d.ts +68 -0
- package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -0
- package/dist/esm/testing/phoenix-test-tracking.js +521 -0
- package/dist/esm/testing/phoenix-test-tracking.js.map +1 -0
- package/dist/esm/testing/report-artifacts.d.ts +45 -0
- package/dist/esm/testing/report-artifacts.d.ts.map +1 -0
- package/dist/esm/testing/report-artifacts.js +218 -0
- package/dist/esm/testing/report-artifacts.js.map +1 -0
- package/dist/esm/testing/report-run.d.ts +22 -0
- package/dist/esm/testing/report-run.d.ts.map +1 -0
- package/dist/esm/testing/report-run.js +41 -0
- package/dist/esm/testing/report-run.js.map +1 -0
- package/dist/esm/testing/reporter-format.d.ts +83 -0
- package/dist/esm/testing/reporter-format.d.ts.map +1 -0
- package/dist/esm/testing/reporter-format.js +852 -0
- package/dist/esm/testing/reporter-format.js.map +1 -0
- package/dist/esm/testing/runner.d.ts +31 -0
- package/dist/esm/testing/runner.d.ts.map +1 -0
- package/dist/esm/testing/runner.js +238 -0
- package/dist/esm/testing/runner.js.map +1 -0
- package/dist/esm/testing/state.d.ts +138 -0
- package/dist/esm/testing/state.d.ts.map +1 -0
- package/dist/esm/testing/state.js +31 -0
- package/dist/esm/testing/state.js.map +1 -0
- package/dist/esm/testing/types.d.ts +319 -0
- package/dist/esm/testing/types.d.ts.map +1 -0
- package/dist/esm/testing/types.js +9 -0
- package/dist/esm/testing/types.js.map +1 -0
- package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
- package/dist/esm/utils/channel.d.ts +7 -7
- package/dist/esm/utils/channel.d.ts.map +1 -1
- package/dist/esm/utils/channel.js +1 -1
- package/dist/esm/utils/channel.js.map +1 -1
- package/dist/esm/utils/formatPromptMessages.d.ts.map +1 -1
- package/dist/esm/utils/getPromptBySelector.d.ts.map +1 -1
- package/dist/esm/utils/promisifyResult.d.ts +1 -1
- package/dist/esm/utils/promisifyResult.d.ts.map +1 -1
- package/dist/esm/utils/promisifyResult.js.map +1 -1
- package/dist/esm/utils/schemaMatches.d.ts +5 -5
- package/dist/esm/utils/schemaMatches.d.ts.map +1 -1
- package/dist/esm/utils/schemaMatches.js.map +1 -1
- package/dist/esm/vitest/index.d.ts +5 -0
- package/dist/esm/vitest/index.d.ts.map +1 -0
- package/dist/esm/vitest/index.js +15 -0
- package/dist/esm/vitest/index.js.map +1 -0
- package/dist/esm/vitest/reporter.d.ts +19 -0
- package/dist/esm/vitest/reporter.d.ts.map +1 -0
- package/dist/esm/vitest/reporter.js +27 -0
- package/dist/esm/vitest/reporter.js.map +1 -0
- package/dist/src/__generated__/api/v1.d.ts +254 -2
- package/dist/src/__generated__/api/v1.d.ts.map +1 -1
- package/dist/src/jest/index.d.ts +5 -0
- package/dist/src/jest/index.d.ts.map +1 -0
- package/dist/src/jest/index.js +58 -0
- package/dist/src/jest/index.js.map +1 -0
- package/dist/src/jest/reporter.d.ts +13 -0
- package/dist/src/jest/reporter.d.ts.map +1 -0
- package/dist/src/jest/reporter.js +23 -0
- package/dist/src/jest/reporter.js.map +1 -0
- package/dist/src/prompts/sdks/toAI.d.ts +2 -2
- package/dist/src/prompts/sdks/toAI.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toAI.js.map +1 -1
- package/dist/src/prompts/sdks/toAnthropic.d.ts +2 -2
- package/dist/src/prompts/sdks/toAnthropic.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toAnthropic.js.map +1 -1
- package/dist/src/prompts/sdks/toOpenAI.d.ts +2 -2
- package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
- package/dist/src/prompts/sdks/toSDK.d.ts +8 -8
- package/dist/src/prompts/sdks/toSDK.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toSDK.js.map +1 -1
- package/dist/src/prompts/sdks/types.d.ts +2 -2
- package/dist/src/prompts/sdks/types.d.ts.map +1 -1
- package/dist/src/schemas/llm/anthropic/converters.d.ts +8 -8
- package/dist/src/schemas/llm/anthropic/messagePartSchemas.d.ts +4 -4
- package/dist/src/schemas/llm/anthropic/messageSchemas.d.ts +6 -6
- package/dist/src/schemas/llm/constants.d.ts +3 -3
- package/dist/src/schemas/llm/converters.d.ts +12 -12
- package/dist/src/schemas/llm/openai/converters.d.ts +3 -3
- package/dist/src/schemas/llm/schemas.d.ts +2 -2
- package/dist/src/testing/acceptance.d.ts +20 -0
- package/dist/src/testing/acceptance.d.ts.map +1 -0
- package/dist/src/testing/acceptance.js +114 -0
- package/dist/src/testing/acceptance.js.map +1 -0
- package/dist/src/testing/define-api.d.ts +157 -0
- package/dist/src/testing/define-api.d.ts.map +1 -0
- package/dist/src/testing/define-api.js +81 -0
- package/dist/src/testing/define-api.js.map +1 -0
- package/dist/src/testing/helpers.d.ts +55 -0
- package/dist/src/testing/helpers.d.ts.map +1 -0
- package/dist/src/testing/helpers.js +182 -0
- package/dist/src/testing/helpers.js.map +1 -0
- package/dist/src/testing/phoenix-test-tracking.d.ts +68 -0
- package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -0
- package/dist/src/testing/phoenix-test-tracking.js +530 -0
- package/dist/src/testing/phoenix-test-tracking.js.map +1 -0
- package/dist/src/testing/report-artifacts.d.ts +45 -0
- package/dist/src/testing/report-artifacts.d.ts.map +1 -0
- package/dist/src/testing/report-artifacts.js +225 -0
- package/dist/src/testing/report-artifacts.js.map +1 -0
- package/dist/src/testing/report-run.d.ts +22 -0
- package/dist/src/testing/report-run.d.ts.map +1 -0
- package/dist/src/testing/report-run.js +47 -0
- package/dist/src/testing/report-run.js.map +1 -0
- package/dist/src/testing/reporter-format.d.ts +83 -0
- package/dist/src/testing/reporter-format.d.ts.map +1 -0
- package/dist/src/testing/reporter-format.js +870 -0
- package/dist/src/testing/reporter-format.js.map +1 -0
- package/dist/src/testing/runner.d.ts +31 -0
- package/dist/src/testing/runner.d.ts.map +1 -0
- package/dist/src/testing/runner.js +258 -0
- package/dist/src/testing/runner.js.map +1 -0
- package/dist/src/testing/state.d.ts +138 -0
- package/dist/src/testing/state.d.ts.map +1 -0
- package/dist/src/testing/state.js +38 -0
- package/dist/src/testing/state.js.map +1 -0
- package/dist/src/testing/types.d.ts +319 -0
- package/dist/src/testing/types.d.ts.map +1 -0
- package/dist/src/testing/types.js +13 -0
- package/dist/src/testing/types.js.map +1 -0
- package/dist/src/utils/channel.d.ts +7 -7
- package/dist/src/utils/channel.d.ts.map +1 -1
- package/dist/src/utils/channel.js +1 -1
- package/dist/src/utils/channel.js.map +1 -1
- package/dist/src/utils/formatPromptMessages.d.ts.map +1 -1
- package/dist/src/utils/getPromptBySelector.d.ts.map +1 -1
- package/dist/src/utils/promisifyResult.d.ts +1 -1
- package/dist/src/utils/promisifyResult.d.ts.map +1 -1
- package/dist/src/utils/promisifyResult.js.map +1 -1
- package/dist/src/utils/schemaMatches.d.ts +5 -5
- package/dist/src/utils/schemaMatches.d.ts.map +1 -1
- package/dist/src/utils/schemaMatches.js.map +1 -1
- package/dist/src/vitest/index.d.ts +5 -0
- package/dist/src/vitest/index.d.ts.map +1 -0
- package/dist/src/vitest/index.js +23 -0
- package/dist/src/vitest/index.js.map +1 -0
- package/dist/src/vitest/reporter.d.ts +19 -0
- package/dist/src/vitest/reporter.d.ts.map +1 -0
- package/dist/src/vitest/reporter.js +34 -0
- package/dist/src/vitest/reporter.js.map +1 -0
- package/dist/tsconfig.tsbuildinfo +1 -1
- package/docs/ci-evals-annotations.mdx +190 -0
- package/docs/ci-evals-jest.mdx +78 -0
- package/docs/ci-evals-vitest.mdx +240 -0
- package/docs/ci-evals.mdx +263 -0
- package/docs/overview.mdx +9 -1
- package/package.json +49 -17
- package/src/__generated__/api/v1.ts +254 -2
- package/src/jest/index.ts +124 -0
- package/src/jest/reporter.ts +22 -0
- package/src/prompts/sdks/toAI.ts +4 -3
- package/src/prompts/sdks/toAnthropic.ts +4 -3
- package/src/prompts/sdks/toOpenAI.ts +4 -3
- package/src/prompts/sdks/toSDK.ts +16 -11
- package/src/prompts/sdks/types.ts +2 -2
- package/src/testing/acceptance.ts +190 -0
- package/src/testing/define-api.ts +279 -0
- package/src/testing/helpers.ts +251 -0
- package/src/testing/phoenix-test-tracking.ts +637 -0
- package/src/testing/report-artifacts.ts +272 -0
- package/src/testing/report-run.ts +44 -0
- package/src/testing/reporter-format.ts +1072 -0
- package/src/testing/runner.ts +350 -0
- package/src/testing/state.ts +165 -0
- package/src/testing/types.ts +366 -0
- package/src/utils/channel.ts +17 -15
- package/src/utils/promisifyResult.ts +6 -4
- package/src/utils/schemaMatches.ts +12 -10
- package/src/vitest/index.ts +57 -0
- package/src/vitest/reporter.ts +32 -0
|
@@ -0,0 +1,319 @@
|
|
|
1
|
+
import type { PhoenixClient } from "../index.js";
|
|
2
|
+
import type { AnnotatorKind } from "../types/annotations.js";
|
|
3
|
+
import type { EvaluatorParams, EvaluationResult as ExperimentEvaluationResult } from "../types/experiments.js";
|
|
4
|
+
/**
|
|
5
|
+
* Phoenix annotator kind, re-exported from the shared client types so the
|
|
6
|
+
* testing module and the rest of the client agree on a single definition.
|
|
7
|
+
*/
|
|
8
|
+
export type { AnnotatorKind };
|
|
9
|
+
/** A JSON-serializable map. */
|
|
10
|
+
export type KVMap = Record<string, unknown>;
|
|
11
|
+
/**
|
|
12
|
+
* Domain language
|
|
13
|
+
* ---------------
|
|
14
|
+
* The unit these tests evaluate over is an **Example**: a single AI example —
|
|
15
|
+
* an `input`, its `expected` output, and optional `metadata` / `splits` — over
|
|
16
|
+
* which the task under test is run and then scored. This is the same notion as
|
|
17
|
+
* the dataset `Example` (`../types/datasets`): each test case _is_ one example.
|
|
18
|
+
* When tracked, a case is recorded to Phoenix as a dataset example and
|
|
19
|
+
* evaluated as one experiment run.
|
|
20
|
+
*
|
|
21
|
+
* These field names form the shared vocabulary across this module:
|
|
22
|
+
* - `input` — the example's input, passed to the task under evaluation.
|
|
23
|
+
* - `expected` — the example's expected (reference / ground-truth) output.
|
|
24
|
+
* - `metadata` — extra fields carried on the example.
|
|
25
|
+
* - `splits` — slice labels for the example.
|
|
26
|
+
* - `id` — stable example id, used to upsert the example across runs.
|
|
27
|
+
*/
|
|
28
|
+
/**
|
|
29
|
+
* The expected output of an `Example`, accepted under any one of three
|
|
30
|
+
* interchangeable keys. All three normalize to the same slot: when recorded to
|
|
31
|
+
* Phoenix the value becomes the dataset example's `output`, and it is exposed
|
|
32
|
+
* to evaluators as `expected` on `EvaluatorParams`. At most one key may be set.
|
|
33
|
+
*
|
|
34
|
+
* - `expected` — the canonical name (the ground-truth / reference output).
|
|
35
|
+
* - `reference` — alias preferred by frameworks that name the slot "reference".
|
|
36
|
+
* - `output` — alias for callers who think in terms of the example's `output`.
|
|
37
|
+
*
|
|
38
|
+
* Modeled as a union so supplying more than one key at a time is a type error.
|
|
39
|
+
*/
|
|
40
|
+
export type ReferenceOutput<Expected extends KVMap = KVMap> = {
|
|
41
|
+
expected?: Expected;
|
|
42
|
+
reference?: never;
|
|
43
|
+
output?: never;
|
|
44
|
+
} | {
|
|
45
|
+
reference?: Expected;
|
|
46
|
+
expected?: never;
|
|
47
|
+
output?: never;
|
|
48
|
+
} | {
|
|
49
|
+
output?: Expected;
|
|
50
|
+
expected?: never;
|
|
51
|
+
reference?: never;
|
|
52
|
+
};
|
|
53
|
+
/**
|
|
54
|
+
* The `Example` fields that define a single test case, excluding its
|
|
55
|
+
* expected output (which is supplied separately via {@link ReferenceOutput}).
|
|
56
|
+
*
|
|
57
|
+
* `input` is the example's input — the value fed to the task under evaluation.
|
|
58
|
+
* When the case is tracked, this becomes the dataset example's `input`.
|
|
59
|
+
*/
|
|
60
|
+
export interface TestParamsBase<Input extends KVMap = KVMap> {
|
|
61
|
+
/** Optional stable example id; used to upsert the example between runs. */
|
|
62
|
+
id?: string;
|
|
63
|
+
/** The example's input — fed to the task under evaluation. Required. */
|
|
64
|
+
input: Input;
|
|
65
|
+
/** Additional metadata stored on the example and its run. */
|
|
66
|
+
metadata?: KVMap;
|
|
67
|
+
/**
|
|
68
|
+
* Split assignment(s) for the example, used to slice the dataset and
|
|
69
|
+
* experiment in the Phoenix UI (e.g. `["factual_accuracy", "correct"]`).
|
|
70
|
+
*/
|
|
71
|
+
splits?: string[];
|
|
72
|
+
/** Per-test config (tags + metadata recorded on the run). */
|
|
73
|
+
config?: TestConfig;
|
|
74
|
+
/**
|
|
75
|
+
* Number of times to run this test case. Each repetition becomes a
|
|
76
|
+
* separate experiment run against the same dataset example (carrying a
|
|
77
|
+
* distinct `repetition_number`). Overrides the suite-level `repetitions`.
|
|
78
|
+
* Defaults to the suite value, then `PHOENIX_TEST_REPETITIONS`, then `1`.
|
|
79
|
+
*/
|
|
80
|
+
repetitions?: number;
|
|
81
|
+
/**
|
|
82
|
+
* When `true`, this test runs as an ordinary local test only — no dataset
|
|
83
|
+
* example is created and no experiment run or annotations are uploaded to
|
|
84
|
+
* Phoenix. Useful for scaffolding a case before it's ready to track.
|
|
85
|
+
*/
|
|
86
|
+
dryRun?: boolean;
|
|
87
|
+
}
|
|
88
|
+
/**
|
|
89
|
+
* The full inline definition of a single `Example` under test.
|
|
90
|
+
*
|
|
91
|
+
* Combines {@link TestParamsBase} with a {@link ReferenceOutput}, so the
|
|
92
|
+
* example's expected output may be given under `expected`, `reference`, or
|
|
93
|
+
* `output` (at most one). All three resolve to the same canonical `expected`
|
|
94
|
+
* slot.
|
|
95
|
+
*/
|
|
96
|
+
export type TestParams<Input extends KVMap = KVMap, Expected extends KVMap = KVMap> = TestParamsBase<Input> & ReferenceOutput<Expected>;
|
|
97
|
+
/**
|
|
98
|
+
* Resolve an `Example`'s expected output from a value that may carry it
|
|
99
|
+
* under any of the `expected` / `reference` / `output` aliases (see
|
|
100
|
+
* {@link ReferenceOutput}). Returns the first one set, or `undefined` if none.
|
|
101
|
+
*/
|
|
102
|
+
export declare function resolveReference<Expected extends KVMap = KVMap>(params: ReferenceOutput<Expected>): Expected | undefined;
|
|
103
|
+
/** Per-test runtime configuration. */
|
|
104
|
+
export interface TestConfig {
|
|
105
|
+
/** Tags recorded on the experiment run for filtering in the Phoenix UI. */
|
|
106
|
+
tags?: string[];
|
|
107
|
+
/** Extra metadata recorded on the experiment run. */
|
|
108
|
+
metadata?: KVMap;
|
|
109
|
+
}
|
|
110
|
+
/**
|
|
111
|
+
* How a criterion aggregates an annotation's scores to gate the suite:
|
|
112
|
+
*
|
|
113
|
+
* - `"average"` — gate on overall quality: the **mean** score across all runs
|
|
114
|
+
* must clear the criterion's `threshold`. A few weak runs are tolerated as
|
|
115
|
+
* long as the mean holds.
|
|
116
|
+
* - `"passRate"` — gate on consistency: each run **passes** when the
|
|
117
|
+
* criterion's `passFn` predicate returns `true` for its annotation, and the
|
|
118
|
+
* suite passes when the **fraction** of runs that pass is at least
|
|
119
|
+
* `minPassRate` (e.g. `minPassRate: 0.9` ⇒ 90% must pass; `1` ⇒ all).
|
|
120
|
+
*/
|
|
121
|
+
export type AcceptanceMetric = "average" | "passRate";
|
|
122
|
+
/**
|
|
123
|
+
* Optimization direction for a criterion's scores: `"maximize"` (higher is
|
|
124
|
+
* better, the default) or `"minimize"` (lower is better). Controls every
|
|
125
|
+
* score comparison the criterion makes.
|
|
126
|
+
*/
|
|
127
|
+
export type OptimizationDirection = "maximize" | "minimize";
|
|
128
|
+
/** Fields shared by every {@link AcceptanceCriterion} variant. */
|
|
129
|
+
export interface AcceptanceCriterionBase {
|
|
130
|
+
/** Annotation name to aggregate across completed test runs. */
|
|
131
|
+
annotationName: string;
|
|
132
|
+
}
|
|
133
|
+
/**
|
|
134
|
+
* Gate the suite on the **mean** score: the average across all runs must clear
|
|
135
|
+
* `threshold` (compared in `direction`).
|
|
136
|
+
*/
|
|
137
|
+
export interface AverageAcceptanceCriterion extends AcceptanceCriterionBase {
|
|
138
|
+
metric: "average";
|
|
139
|
+
/**
|
|
140
|
+
* The bar the mean score must clear, compared in `direction`. Boolean scores
|
|
141
|
+
* average as `1` (`true`) / `0` (`false`).
|
|
142
|
+
*/
|
|
143
|
+
threshold: number;
|
|
144
|
+
/**
|
|
145
|
+
* Optimization direction; defaults to `"maximize"`. `"maximize"` treats a
|
|
146
|
+
* higher mean as better (clears when `>= threshold`); `"minimize"` treats a
|
|
147
|
+
* lower mean as better (clears when `<= threshold`) — use it for cost,
|
|
148
|
+
* latency, or error-rate annotations.
|
|
149
|
+
*/
|
|
150
|
+
direction?: OptimizationDirection;
|
|
151
|
+
}
|
|
152
|
+
/**
|
|
153
|
+
* Gate the suite on the **pass rate**: each run passes when `passFn` returns
|
|
154
|
+
* `true` for its annotation, and the suite passes when at least `minPassRate`
|
|
155
|
+
* of runs do. `passFn` decides what "passing" means, so any logic works — a
|
|
156
|
+
* score bar, a score range, a label match, a metadata check, etc.
|
|
157
|
+
*/
|
|
158
|
+
export interface PassRateAcceptanceCriterion extends AcceptanceCriterionBase {
|
|
159
|
+
metric: "passRate";
|
|
160
|
+
/**
|
|
161
|
+
* Predicate deciding whether a single run passes, given the run's last
|
|
162
|
+
* {@link Annotation} for `annotationName` (its `score`, `label`,
|
|
163
|
+
* `explanation`, `metadata`, …). Runs whose predicate returns `true` count
|
|
164
|
+
* toward the pass rate.
|
|
165
|
+
*/
|
|
166
|
+
passFn: (annotation: Annotation) => boolean;
|
|
167
|
+
/**
|
|
168
|
+
* Minimum fraction of runs (`0`–`1`) that must pass for the suite to pass —
|
|
169
|
+
* e.g. `0.9` requires 90% of runs to satisfy `passFn`, `1` requires all of
|
|
170
|
+
* them. The suite passes when `passRate >= minPassRate`.
|
|
171
|
+
*/
|
|
172
|
+
minPassRate: number;
|
|
173
|
+
}
|
|
174
|
+
/**
|
|
175
|
+
* One aggregate acceptance rule, evaluated once after every test in the suite
|
|
176
|
+
* has run. Each criterion aggregates a single annotation's scores with one
|
|
177
|
+
* {@link AcceptanceMetric} and fails the suite when the result misses its bar.
|
|
178
|
+
*
|
|
179
|
+
* Scoring notes shared by every metric:
|
|
180
|
+
* - Boolean scores count as `1` (`true`) / `0` (`false`).
|
|
181
|
+
* - If a run logs the same annotation more than once, the last one counts.
|
|
182
|
+
* - Skipped tests are excluded; dry-run tests are included (they still run).
|
|
183
|
+
* - A criterion whose annotation was never logged on any run fails (rather
|
|
184
|
+
* than passing vacuously) — see {@link AcceptanceResultFields.failureReason}.
|
|
185
|
+
*/
|
|
186
|
+
export type AcceptanceCriterion = AverageAcceptanceCriterion | PassRateAcceptanceCriterion;
|
|
187
|
+
/** The computed fields added to an {@link AcceptanceCriterion} once evaluated. */
|
|
188
|
+
export interface AcceptanceResultFields {
|
|
189
|
+
/**
|
|
190
|
+
* The aggregate the criterion gated on, or `null` when there were no runs to
|
|
191
|
+
* aggregate. For `"average"` this is the mean score; for `"passRate"` it is
|
|
192
|
+
* the fraction of runs that passed (so a fully-passing `"passRate"` criterion
|
|
193
|
+
* reports `1`).
|
|
194
|
+
*/
|
|
195
|
+
value: number | null;
|
|
196
|
+
/** Number of runs included in the aggregate. */
|
|
197
|
+
sampleCount: number;
|
|
198
|
+
/** Whether the aggregate cleared the criterion. */
|
|
199
|
+
passed: boolean;
|
|
200
|
+
/** Human-readable failure reason for invalid or empty aggregates. */
|
|
201
|
+
failureReason?: string;
|
|
202
|
+
}
|
|
203
|
+
/** Computed result for one aggregate acceptance rule. */
|
|
204
|
+
export type AcceptanceResult = AcceptanceCriterion & AcceptanceResultFields;
|
|
205
|
+
/** Suite-level configuration accepted by `describe()`. */
|
|
206
|
+
export interface SuiteConfig {
|
|
207
|
+
/** Override the dataset / experiment name used for the suite. */
|
|
208
|
+
datasetName?: string;
|
|
209
|
+
/** Description for the dataset and experiment. */
|
|
210
|
+
description?: string;
|
|
211
|
+
/** Suite-level metadata applied to every run in this experiment. */
|
|
212
|
+
metadata?: KVMap;
|
|
213
|
+
/** Override the Phoenix client used for syncing this suite. */
|
|
214
|
+
client?: PhoenixClient;
|
|
215
|
+
/**
|
|
216
|
+
* Number of times to run each test case in this suite. Individual tests
|
|
217
|
+
* may override this via `TestParams.repetitions`. Defaults to the
|
|
218
|
+
* `PHOENIX_TEST_REPETITIONS` env var, then `1`.
|
|
219
|
+
*/
|
|
220
|
+
repetitions?: number;
|
|
221
|
+
/**
|
|
222
|
+
* When `true`, the whole suite runs as ordinary local tests — no dataset
|
|
223
|
+
* is uploaded and no experiment, runs, or annotations are created in
|
|
224
|
+
* Phoenix. Equivalent to `PHOENIX_TEST_TRACKING=false` scoped to this
|
|
225
|
+
* suite. The reporter still prints a local summary.
|
|
226
|
+
*/
|
|
227
|
+
dryRun?: boolean;
|
|
228
|
+
/**
|
|
229
|
+
* Aggregate annotation criteria that gate the suite after all tests run.
|
|
230
|
+
* Each criterion fails the suite when its scores miss the configured bar
|
|
231
|
+
* (see {@link AcceptanceCriterion}).
|
|
232
|
+
*/
|
|
233
|
+
acceptanceCriteria?: AcceptanceCriterion[];
|
|
234
|
+
}
|
|
235
|
+
/**
|
|
236
|
+
* Arguments passed to a `test()` body: the `Example` under test, exposed
|
|
237
|
+
* as its `input`, `expected` output, and `metadata`. Read straight from the
|
|
238
|
+
* test's {@link TestParams} — the runner does not transform them.
|
|
239
|
+
*/
|
|
240
|
+
export interface TestArgs<Input extends KVMap = KVMap, Expected extends KVMap = KVMap> {
|
|
241
|
+
/** The example's input under test. */
|
|
242
|
+
input: Input;
|
|
243
|
+
/** The example's expected (reference) output, when one was supplied. */
|
|
244
|
+
expected?: Expected;
|
|
245
|
+
/** Any metadata attached to the example. */
|
|
246
|
+
metadata?: KVMap;
|
|
247
|
+
}
|
|
248
|
+
/**
|
|
249
|
+
* Object form of an evaluator result. Reuses the shared experiment
|
|
250
|
+
* {@link ExperimentEvaluationResult} shape (label / explanation / metadata)
|
|
251
|
+
* but widens `score` to also accept booleans, which the testing API stores as
|
|
252
|
+
* `1` / `0`.
|
|
253
|
+
*/
|
|
254
|
+
export interface EvaluationResultObject extends Omit<ExperimentEvaluationResult, "score"> {
|
|
255
|
+
/** Numeric or boolean score; booleans are stored as `1` / `0`. */
|
|
256
|
+
score?: number | boolean | null;
|
|
257
|
+
}
|
|
258
|
+
/**
|
|
259
|
+
* One annotation recorded against a run. Extends the evaluator
|
|
260
|
+
* {@link EvaluationResultObject} with the `name` and `annotatorKind` carried
|
|
261
|
+
* on the evaluation body, plus an optional originating trace id.
|
|
262
|
+
*/
|
|
263
|
+
export interface Annotation extends EvaluationResultObject {
|
|
264
|
+
/** Phoenix evaluation name. Required, and unique per run (last write wins). */
|
|
265
|
+
name: string;
|
|
266
|
+
/** Who or what produced the annotation. Defaults to `"CODE"`. */
|
|
267
|
+
annotatorKind?: AnnotatorKind;
|
|
268
|
+
/** Trace id for this evaluation, when the annotation was produced by a traced evaluator. */
|
|
269
|
+
traceId?: string | null;
|
|
270
|
+
}
|
|
271
|
+
/** Result returned by `traceEvaluator` for any evaluator-shaped value. */
|
|
272
|
+
export type EvaluatorResult = Annotation | (KVMap & {
|
|
273
|
+
name: string;
|
|
274
|
+
});
|
|
275
|
+
/** Result shape produced by evaluator objects used in eval tests. */
|
|
276
|
+
export type EvaluationResult = number | boolean | string | null | EvaluationResultObject;
|
|
277
|
+
/**
|
|
278
|
+
* Parameters passed to an evaluator when it runs inside a test. A relaxation of
|
|
279
|
+
* the shared {@link EvaluatorParams}: `input` is always present, while `output`
|
|
280
|
+
* (an evaluator may run before `logOutput()`) and the remaining fields are
|
|
281
|
+
* optional. Deriving from `EvaluatorParams` keeps this aligned with the
|
|
282
|
+
* experiment evaluator contract as that shape evolves.
|
|
283
|
+
*/
|
|
284
|
+
export type EvaluationParams = Partial<EvaluatorParams> & {
|
|
285
|
+
/** The example's input under test. */
|
|
286
|
+
input: KVMap;
|
|
287
|
+
};
|
|
288
|
+
/** Structural evaluator interface accepted by `evaluate()`. */
|
|
289
|
+
export interface Evaluator<Params extends KVMap = EvaluationParams & KVMap, Result = EvaluationResult> {
|
|
290
|
+
/** Annotation/evaluation name. */
|
|
291
|
+
name: string;
|
|
292
|
+
/** Who or what produced the result. Defaults to `"CODE"`. */
|
|
293
|
+
kind?: AnnotatorKind;
|
|
294
|
+
/** Compute the evaluation result. */
|
|
295
|
+
evaluate: (params: Params) => Result | Promise<Result>;
|
|
296
|
+
}
|
|
297
|
+
/** Test handler signature. */
|
|
298
|
+
export type TestFn<Input extends KVMap = KVMap, Expected extends KVMap = KVMap> = (args: TestArgs<Input, Expected>) => unknown | Promise<unknown>;
|
|
299
|
+
/**
|
|
300
|
+
* Each-row shape accepted by `test.each(table)(name, fn)`; each row defines one
|
|
301
|
+
* `Example`.
|
|
302
|
+
*
|
|
303
|
+
* Like {@link TestParams}, the example's expected output is supplied via
|
|
304
|
+
* {@link ReferenceOutput} (`expected` / `reference` / `output`, at most one).
|
|
305
|
+
* The trailing index signature still permits arbitrary extra columns on a row
|
|
306
|
+
* (e.g. for `%j` name interpolation) without weakening that constraint.
|
|
307
|
+
*/
|
|
308
|
+
export type TestEachRow<Input extends KVMap = KVMap, Expected extends KVMap = KVMap> = {
|
|
309
|
+
id?: string;
|
|
310
|
+
input: Input;
|
|
311
|
+
metadata?: KVMap;
|
|
312
|
+
/** Per-row split assignment(s); see `TestParams.splits`. */
|
|
313
|
+
splits?: string[];
|
|
314
|
+
/** Per-row repetition count; see `TestParams.repetitions`. */
|
|
315
|
+
repetitions?: number;
|
|
316
|
+
/** Per-row dry-run flag; see `TestParams.dryRun`. */
|
|
317
|
+
dryRun?: boolean;
|
|
318
|
+
} & ReferenceOutput<Expected> & Record<string, unknown>;
|
|
319
|
+
//# sourceMappingURL=types.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types.d.ts","sourceRoot":"","sources":["../../../src/testing/types.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,UAAU,CAAC;AAC9C,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,sBAAsB,CAAC;AAC1D,OAAO,KAAK,EACV,eAAe,EACf,gBAAgB,IAAI,0BAA0B,EAC/C,MAAM,sBAAsB,CAAC;AAE9B;;;GAGG;AACH,YAAY,EAAE,aAAa,EAAE,CAAC;AAE9B,+BAA+B;AAC/B,MAAM,MAAM,KAAK,GAAG,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;AAE5C;;;;;;;;;;;;;;;;GAgBG;AAEH;;;;;;;;;;;GAWG;AACH,MAAM,MAAM,eAAe,CAAC,QAAQ,SAAS,KAAK,GAAG,KAAK,IACtD;IAAE,QAAQ,CAAC,EAAE,QAAQ,CAAC;IAAC,SAAS,CAAC,EAAE,KAAK,CAAC;IAAC,MAAM,CAAC,EAAE,KAAK,CAAA;CAAE,GAC1D;IAAE,SAAS,CAAC,EAAE,QAAQ,CAAC;IAAC,QAAQ,CAAC,EAAE,KAAK,CAAC;IAAC,MAAM,CAAC,EAAE,KAAK,CAAA;CAAE,GAC1D;IAAE,MAAM,CAAC,EAAE,QAAQ,CAAC;IAAC,QAAQ,CAAC,EAAE,KAAK,CAAC;IAAC,SAAS,CAAC,EAAE,KAAK,CAAA;CAAE,CAAC;AAE/D;;;;;;GAMG;AACH,MAAM,WAAW,cAAc,CAAC,KAAK,SAAS,KAAK,GAAG,KAAK;IACzD,2EAA2E;IAC3E,EAAE,CAAC,EAAE,MAAM,CAAC;IACZ,wEAAwE;IACxE,KAAK,EAAE,KAAK,CAAC;IACb,6DAA6D;IAC7D,QAAQ,CAAC,EAAE,KAAK,CAAC;IACjB;;;OAGG;IACH,MAAM,CAAC,EAAE,MAAM,EAAE,CAAC;IAClB,6DAA6D;IAC7D,MAAM,CAAC,EAAE,UAAU,CAAC;IACpB;;;;;OAKG;IACH,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB;;;;OAIG;IACH,MAAM,CAAC,EAAE,OAAO,CAAC;CAClB;AAED;;;;;;;GAOG;AACH,MAAM,MAAM,UAAU,CACpB,KAAK,SAAS,KAAK,GAAG,KAAK,EAC3B,QAAQ,SAAS,KAAK,GAAG,KAAK,IAC5B,cAAc,CAAC,KAAK,CAAC,GAAG,eAAe,CAAC,QAAQ,CAAC,CAAC;AAEtD;;;;GAIG;AACH,wBAAgB,gBAAgB,CAAC,QAAQ,SAAS,KAAK,GAAG,KAAK,EAC7D,MAAM,EAAE,eAAe,CAAC,QAAQ,CAAC,GAChC,QAAQ,GAAG,SAAS,CAEtB;AAED,sCAAsC;AACtC,MAAM,WAAW,UAAU;IACzB,2EAA2E;IAC3E,IAAI,CAAC,EAAE,MAAM,EAAE,CAAC;IAChB,qDAAqD;IACrD,QAAQ,CAAC,EAAE,KAAK,CAAC;CAClB;AAED;;;;;;;;;;GAUG;AACH,MAAM,MAAM,gBAAgB,GAAG,SAAS,GAAG,UAAU,CAAC;AAEtD;;;;GAIG;AACH,MAAM,MAAM,qBAAqB,GAAG,UAAU,GAAG,UAAU,CAAC;AAE5D,kEAAkE;AAClE,MAAM,WAAW,uBAAuB;IACtC,+DAA+D;IAC/D,cAAc,EAAE,MAAM,CAAC;CACxB;AAED;;;GAGG;AACH,MAAM,WAAW,0BAA2B,SAAQ,uBAAuB;IACzE,MAAM,EAAE,SAAS,CAAC;IAClB;;;OAGG;IACH,SAAS,EAAE,MAAM,CAAC;IAClB;;;;;OAKG;IACH,SAAS,CAAC,EAAE,qBAAqB,CAAC;CACnC;AAED;;;;;GAKG;AACH,MAAM,WAAW,2BAA4B,SAAQ,uBAAuB;IAC1E,MAAM,EAAE,UAAU,CAAC;IACnB;;;;;OAKG;IACH,MAAM,EAAE,CAAC,UAAU,EAAE,UAAU,KAAK,OAAO,CAAC;IAC5C;;;;OAIG;IACH,WAAW,EAAE,MAAM,CAAC;CACrB;AAED;;;;;;;;;;;GAWG;AACH,MAAM,MAAM,mBAAmB,GAC3B,0BAA0B,GAC1B,2BAA2B,CAAC;AAEhC,kFAAkF;AAClF,MAAM,WAAW,sBAAsB;IACrC;;;;;OAKG;IACH,KAAK,EAAE,MAAM,GAAG,IAAI,CAAC;IACrB,gDAAgD;IAChD,WAAW,EAAE,MAAM,CAAC;IACpB,mDAAmD;IACnD,MAAM,EAAE,OAAO,CAAC;IAChB,qEAAqE;IACrE,aAAa,CAAC,EAAE,MAAM,CAAC;CACxB;AAED,yDAAyD;AACzD,MAAM,MAAM,gBAAgB,GAAG,mBAAmB,GAAG,sBAAsB,CAAC;AAE5E,0DAA0D;AAC1D,MAAM,WAAW,WAAW;IAC1B,iEAAiE;IACjE,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,kDAAkD;IAClD,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,oEAAoE;IACpE,QAAQ,CAAC,EAAE,KAAK,CAAC;IACjB,+DAA+D;IAC/D,MAAM,CAAC,EAAE,aAAa,CAAC;IACvB;;;;OAIG;IACH,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB;;;;;OAKG;IACH,MAAM,CAAC,EAAE,OAAO,CAAC;IACjB;;;;OAIG;IACH,kBAAkB,CAAC,EAAE,mBAAmB,EAAE,CAAC;CAC5C;AAED;;;;GAIG;AACH,MAAM,WAAW,QAAQ,CACvB,KAAK,SAAS,KAAK,GAAG,KAAK,EAC3B,QAAQ,SAAS,KAAK,GAAG,KAAK;IAE9B,sCAAsC;IACtC,KAAK,EAAE,KAAK,CAAC;IACb,wEAAwE;IACxE,QAAQ,CAAC,EAAE,QAAQ,CAAC;IACpB,4CAA4C;IAC5C,QAAQ,CAAC,EAAE,KAAK,CAAC;CAClB;AAED;;;;;GAKG;AACH,MAAM,WAAW,sBAAuB,SAAQ,IAAI,CAClD,0BAA0B,EAC1B,OAAO,CACR;IACC,kEAAkE;IAClE,KAAK,CAAC,EAAE,MAAM,GAAG,OAAO,GAAG,IAAI,CAAC;CACjC;AAED;;;;GAIG;AACH,MAAM,WAAW,UAAW,SAAQ,sBAAsB;IACxD,+EAA+E;IAC/E,IAAI,EAAE,MAAM,CAAC;IACb,iEAAiE;IACjE,aAAa,CAAC,EAAE,aAAa,CAAC;IAC9B,4FAA4F;IAC5F,OAAO,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;CACzB;AAED,0EAA0E;AAC1E,MAAM,MAAM,eAAe,GAAG,UAAU,GAAG,CAAC,KAAK,GAAG;IAAE,IAAI,EAAE,MAAM,CAAA;CAAE,CAAC,CAAC;AAEtE,qEAAqE;AACrE,MAAM,MAAM,gBAAgB,GACxB,MAAM,GACN,OAAO,GACP,MAAM,GACN,IAAI,GACJ,sBAAsB,CAAC;AAE3B;;;;;;GAMG;AACH,MAAM,MAAM,gBAAgB,GAAG,OAAO,CAAC,eAAe,CAAC,GAAG;IACxD,sCAAsC;IACtC,KAAK,EAAE,KAAK,CAAC;CACd,CAAC;AAEF,+DAA+D;AAC/D,MAAM,WAAW,SAAS,CACxB,MAAM,SAAS,KAAK,GAAG,gBAAgB,GAAG,KAAK,EAC/C,MAAM,GAAG,gBAAgB;IAEzB,kCAAkC;IAClC,IAAI,EAAE,MAAM,CAAC;IACb,6DAA6D;IAC7D,IAAI,CAAC,EAAE,aAAa,CAAC;IACrB,qCAAqC;IACrC,QAAQ,EAAE,CAAC,MAAM,EAAE,MAAM,KAAK,MAAM,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC;CACxD;AAED,8BAA8B;AAC9B,MAAM,MAAM,MAAM,CAChB,KAAK,SAAS,KAAK,GAAG,KAAK,EAC3B,QAAQ,SAAS,KAAK,GAAG,KAAK,IAC5B,CAAC,IAAI,EAAE,QAAQ,CAAC,KAAK,EAAE,QAAQ,CAAC,KAAK,OAAO,GAAG,OAAO,CAAC,OAAO,CAAC,CAAC;AAEpE;;;;;;;;GAQG;AACH,MAAM,MAAM,WAAW,CACrB,KAAK,SAAS,KAAK,GAAG,KAAK,EAC3B,QAAQ,SAAS,KAAK,GAAG,KAAK,IAC5B;IACF,EAAE,CAAC,EAAE,MAAM,CAAC;IACZ,KAAK,EAAE,KAAK,CAAC;IACb,QAAQ,CAAC,EAAE,KAAK,CAAC;IACjB,4DAA4D;IAC5D,MAAM,CAAC,EAAE,MAAM,EAAE,CAAC;IAClB,8DAA8D;IAC9D,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,qDAAqD;IACrD,MAAM,CAAC,EAAE,OAAO,CAAC;CAClB,GAAG,eAAe,CAAC,QAAQ,CAAC,GAC3B,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC"}
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Resolve an `Example`'s expected output from a value that may carry it
|
|
3
|
+
* under any of the `expected` / `reference` / `output` aliases (see
|
|
4
|
+
* {@link ReferenceOutput}). Returns the first one set, or `undefined` if none.
|
|
5
|
+
*/
|
|
6
|
+
export function resolveReference(params) {
|
|
7
|
+
return params.expected ?? params.reference ?? params.output;
|
|
8
|
+
}
|
|
9
|
+
//# sourceMappingURL=types.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types.js","sourceRoot":"","sources":["../../../src/testing/types.ts"],"names":[],"mappings":"AAoGA;;;;GAIG;AACH,MAAM,UAAU,gBAAgB,CAC9B,MAAiC;IAEjC,OAAO,MAAM,CAAC,QAAQ,IAAI,MAAM,CAAC,SAAS,IAAI,MAAM,CAAC,MAAM,CAAC;AAC9D,CAAC"}
|