@arizeai/phoenix-client 6.14.2 → 7.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/__generated__/api/v1.d.ts +8 -0
- package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
- package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
- package/dist/esm/experiments/resumeEvaluation.js +43 -26
- package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
- package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
- package/dist/esm/experiments/resumeExperiment.js +43 -26
- package/dist/esm/experiments/resumeExperiment.js.map +1 -1
- package/dist/esm/prompts/sdks/toAI.d.ts +2 -1
- package/dist/esm/prompts/sdks/toAI.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toAI.js.map +1 -1
- package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
- package/dist/src/__generated__/api/v1.d.ts +8 -0
- package/dist/src/__generated__/api/v1.d.ts.map +1 -1
- package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
- package/dist/src/experiments/resumeEvaluation.js +43 -26
- package/dist/src/experiments/resumeEvaluation.js.map +1 -1
- package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
- package/dist/src/experiments/resumeExperiment.js +43 -26
- package/dist/src/experiments/resumeExperiment.js.map +1 -1
- package/dist/src/prompts/sdks/toAI.d.ts +2 -1
- package/dist/src/prompts/sdks/toAI.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toAI.js.map +1 -1
- package/dist/tsconfig.tsbuildinfo +1 -1
- package/docs/datasets.mdx +15 -9
- package/docs/experiments.mdx +18 -19
- package/package.json +10 -10
- package/src/__generated__/api/v1.ts +8 -0
- package/src/experiments/resumeEvaluation.ts +55 -25
- package/src/experiments/resumeExperiment.ts +57 -25
- package/src/prompts/sdks/toAI.ts +3 -1
package/docs/datasets.mdx
CHANGED
|
@@ -3,14 +3,14 @@ title: "Datasets"
|
|
|
3
3
|
description: "Create and inspect datasets with @arizeai/phoenix-client"
|
|
4
4
|
---
|
|
5
5
|
|
|
6
|
-
Datasets are the foundation for experiment runs. The dataset helpers cover creation
|
|
6
|
+
Datasets are the foundation for experiment runs. The dataset helpers cover creation (which upserts by name), record inspection, and example appends.
|
|
7
7
|
|
|
8
8
|
<section className="hidden" data-agent-context="relevant-source-files" aria-label="Relevant source files">
|
|
9
9
|
<h2>Relevant Source Files</h2>
|
|
10
10
|
<ul>
|
|
11
11
|
<li>
|
|
12
|
-
<code>src/datasets/
|
|
13
|
-
|
|
12
|
+
<code>src/datasets/createDataset.ts</code> for the exact return shape and
|
|
13
|
+
upsert-by-name behavior
|
|
14
14
|
</li>
|
|
15
15
|
</ul>
|
|
16
16
|
</section>
|
|
@@ -33,18 +33,25 @@ const { datasetId } = await createDataset({
|
|
|
33
33
|
});
|
|
34
34
|
```
|
|
35
35
|
|
|
36
|
-
##
|
|
36
|
+
## Upsert Or Append
|
|
37
|
+
|
|
38
|
+
`createDataset()` upserts by name: re-running it with the same name updates the existing dataset to match the examples you pass, and an unchanged upload is a no-op. To keep existing examples and add more, use `appendDatasetExamples()` instead of re-running `createDataset()` with the extra examples.
|
|
37
39
|
|
|
38
40
|
```ts
|
|
39
41
|
import {
|
|
40
42
|
appendDatasetExamples,
|
|
41
|
-
|
|
43
|
+
createDataset,
|
|
42
44
|
} from "@arizeai/phoenix-client/datasets";
|
|
43
45
|
|
|
44
|
-
const dataset = await
|
|
46
|
+
const dataset = await createDataset({
|
|
45
47
|
name: "support-eval",
|
|
46
48
|
description: "Support questions with expected answers",
|
|
47
|
-
examples: [
|
|
49
|
+
examples: [
|
|
50
|
+
{
|
|
51
|
+
input: { question: "Where is my order?" },
|
|
52
|
+
output: { answer: "Use the tracking page in your account." },
|
|
53
|
+
},
|
|
54
|
+
],
|
|
48
55
|
});
|
|
49
56
|
|
|
50
57
|
await appendDatasetExamples({
|
|
@@ -58,7 +65,7 @@ await appendDatasetExamples({
|
|
|
58
65
|
});
|
|
59
66
|
```
|
|
60
67
|
|
|
61
|
-
`
|
|
68
|
+
`createDataset()` returns `{ datasetId }`, so you can pass that object directly as the dataset selector for append or experiment calls.
|
|
62
69
|
|
|
63
70
|
## Read Back Dataset State
|
|
64
71
|
|
|
@@ -68,7 +75,6 @@ Use `getDataset`, `getDatasetExamples`, and `getDatasetInfo` to inspect datasets
|
|
|
68
75
|
<h2>Source Map</h2>
|
|
69
76
|
<ul>
|
|
70
77
|
<li><code>src/datasets/createDataset.ts</code></li>
|
|
71
|
-
<li><code>src/datasets/createOrGetDataset.ts</code></li>
|
|
72
78
|
<li><code>src/datasets/appendDatasetExamples.ts</code></li>
|
|
73
79
|
<li><code>src/datasets/getDataset.ts</code></li>
|
|
74
80
|
<li><code>src/datasets/getDatasetExamples.ts</code></li>
|
package/docs/experiments.mdx
CHANGED
|
@@ -111,11 +111,18 @@ This pattern is useful when:
|
|
|
111
111
|
|
|
112
112
|
## Model-Backed Example
|
|
113
113
|
|
|
114
|
-
If you want a model-backed experiment with automatic tracing and an LLM-as-a-judge evaluator, this is the core pattern
|
|
114
|
+
If you want a model-backed experiment with automatic tracing and an LLM-as-a-judge evaluator, this is the core pattern.
|
|
115
|
+
|
|
116
|
+
Install the AI SDK packages alongside the Phoenix packages — `@ai-sdk/otel` provides the telemetry integration that sends AI SDK traces to Phoenix:
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
npm install @arizeai/phoenix-client @arizeai/phoenix-evals ai @ai-sdk/openai @ai-sdk/otel
|
|
120
|
+
```
|
|
115
121
|
|
|
116
122
|
```ts
|
|
117
123
|
import { openai } from "@ai-sdk/openai";
|
|
118
|
-
import {
|
|
124
|
+
import { OpenTelemetry } from "@ai-sdk/otel";
|
|
125
|
+
import { createDataset } from "@arizeai/phoenix-client/datasets";
|
|
119
126
|
import { runExperiment } from "@arizeai/phoenix-client/experiments";
|
|
120
127
|
import type { ExperimentTask } from "@arizeai/phoenix-client/types/experiments";
|
|
121
128
|
import { createClassificationEvaluator } from "@arizeai/phoenix-evals";
|
|
@@ -135,7 +142,7 @@ const main = async () => {
|
|
|
135
142
|
},
|
|
136
143
|
});
|
|
137
144
|
|
|
138
|
-
const dataset = await
|
|
145
|
+
const dataset = await createDataset({
|
|
139
146
|
name: "correctness-eval",
|
|
140
147
|
description: "Evaluate the correctness of the model",
|
|
141
148
|
examples: [
|
|
@@ -157,23 +164,15 @@ const main = async () => {
|
|
|
157
164
|
throw new Error("Invalid input: context must be a string");
|
|
158
165
|
}
|
|
159
166
|
|
|
167
|
+
// The per-call `@ai-sdk/otel` integration traces this call through the
|
|
168
|
+
// experiment's tracer provider that runExperiment mounts while tasks run.
|
|
160
169
|
return generateText({
|
|
161
170
|
model,
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
},
|
|
165
|
-
prompt: [
|
|
166
|
-
{
|
|
167
|
-
role: "system",
|
|
168
|
-
content: `You answer questions based on this context: ${example.input.context}`,
|
|
169
|
-
},
|
|
170
|
-
{
|
|
171
|
-
role: "user",
|
|
172
|
-
content: example.input.question,
|
|
173
|
-
},
|
|
174
|
-
],
|
|
171
|
+
instructions: `You answer questions based on this context: ${example.input.context}`,
|
|
172
|
+
prompt: example.input.question,
|
|
173
|
+
telemetry: { integrations: [new OpenTelemetry()] },
|
|
175
174
|
}).then((response) => {
|
|
176
|
-
if (response.text) {
|
|
175
|
+
if (typeof response.text === "string") {
|
|
177
176
|
return response.text;
|
|
178
177
|
}
|
|
179
178
|
throw new Error("Invalid response: text is required");
|
|
@@ -200,9 +199,9 @@ main().catch(console.error);
|
|
|
200
199
|
|
|
201
200
|
## What This Example Shows
|
|
202
201
|
|
|
203
|
-
- `
|
|
202
|
+
- `createDataset()` creates or reuses the dataset the experiment will run against (it upserts by name)
|
|
204
203
|
- `task` receives the full dataset example object
|
|
205
|
-
- `generateText()` emits traces that Phoenix
|
|
204
|
+
- `generateText()` emits traces that Phoenix attaches to the experiment because the call passes the `@ai-sdk/otel` integration via `telemetry.integrations` — without it, no AI SDK spans reach Phoenix
|
|
206
205
|
- `createClassificationEvaluator()` from `@arizeai/phoenix-evals` can be passed directly to `runExperiment()`
|
|
207
206
|
- `runExperiment()` records both task runs and evaluation runs in Phoenix
|
|
208
207
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@arizeai/phoenix-client",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "7.0.1",
|
|
4
4
|
"description": "A client for the Phoenix API",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"arize",
|
|
@@ -96,34 +96,34 @@
|
|
|
96
96
|
},
|
|
97
97
|
"dependencies": {
|
|
98
98
|
"@arizeai/openinference-semantic-conventions": "^2.5.0",
|
|
99
|
-
"@arizeai/openinference-vercel": "^2.8.1",
|
|
100
99
|
"async": "^3.2.6",
|
|
101
100
|
"openapi-fetch": "^0.17.0",
|
|
102
101
|
"tiny-invariant": "^1.3.3",
|
|
103
102
|
"zod": "^4.4.3",
|
|
104
|
-
"@arizeai/phoenix-
|
|
105
|
-
"@arizeai/phoenix-
|
|
103
|
+
"@arizeai/phoenix-config": "0.4.0",
|
|
104
|
+
"@arizeai/phoenix-otel": "2.1.0"
|
|
106
105
|
},
|
|
107
106
|
"devDependencies": {
|
|
108
|
-
"@ai-sdk/openai": "^
|
|
107
|
+
"@ai-sdk/openai": "^4.0.0",
|
|
108
|
+
"@ai-sdk/otel": "^1.0.0",
|
|
109
109
|
"@anthropic-ai/sdk": "^0.111.0",
|
|
110
110
|
"@opentelemetry/api": "^1.9.1",
|
|
111
111
|
"@opentelemetry/sdk-trace-node": "^2.9.0",
|
|
112
112
|
"@types/async": "^3.2.25",
|
|
113
113
|
"@types/node": "^26.1.1",
|
|
114
|
-
"ai": "^
|
|
114
|
+
"ai": "^7.0.0",
|
|
115
115
|
"dotenv": "^17.4.2",
|
|
116
116
|
"jest": "^30.4.2",
|
|
117
117
|
"openai": "^6.48.0",
|
|
118
118
|
"openapi-typescript": "^7.13.0",
|
|
119
119
|
"tsx": "^4.23.1",
|
|
120
120
|
"vitest": "^4.1.10",
|
|
121
|
-
"@arizeai/phoenix-
|
|
122
|
-
"@arizeai/phoenix-
|
|
121
|
+
"@arizeai/phoenix-evals": "2.1.0",
|
|
122
|
+
"@arizeai/phoenix-testing": "0.0.0"
|
|
123
123
|
},
|
|
124
124
|
"peerDependencies": {
|
|
125
125
|
"@anthropic-ai/sdk": "^0.35.0",
|
|
126
|
-
"ai": "^
|
|
126
|
+
"ai": "^7.0.0",
|
|
127
127
|
"jest": ">=27",
|
|
128
128
|
"openai": "^6.10.0",
|
|
129
129
|
"vitest": ">=1"
|
|
@@ -156,6 +156,6 @@
|
|
|
156
156
|
"prebuild": "pnpm run clean && pnpm run generate",
|
|
157
157
|
"test": "vitest run",
|
|
158
158
|
"test:watch": "vitest watch",
|
|
159
|
-
"typecheck": "tsc --noEmit"
|
|
159
|
+
"typecheck": "tsc --noEmit && tsc -p examples/tsconfig.examples.json"
|
|
160
160
|
}
|
|
161
161
|
}
|
|
@@ -2417,6 +2417,14 @@ export interface components {
|
|
|
2417
2417
|
* Format: date-time
|
|
2418
2418
|
*/
|
|
2419
2419
|
updated_at: string;
|
|
2420
|
+
source?: components["schemas"]["DatasetExampleSource"] | null;
|
|
2421
|
+
};
|
|
2422
|
+
/** DatasetExampleSource */
|
|
2423
|
+
DatasetExampleSource: {
|
|
2424
|
+
/** Span Id */
|
|
2425
|
+
span_id: string;
|
|
2426
|
+
/** Span Node Id */
|
|
2427
|
+
span_node_id: string;
|
|
2420
2428
|
};
|
|
2421
2429
|
/** DatasetLabel */
|
|
2422
2430
|
DatasetLabel: {
|
|
@@ -522,7 +522,7 @@ export async function resumeEvaluation({
|
|
|
522
522
|
}
|
|
523
523
|
|
|
524
524
|
// Start concurrent execution
|
|
525
|
-
// Wrap in try-finally to ensure channel is always closed, even if
|
|
525
|
+
// Wrap in try-finally to ensure channel is always closed, even if a task throws
|
|
526
526
|
let executionError: Error | null = null;
|
|
527
527
|
try {
|
|
528
528
|
const producerTask = fetchIncompleteEvaluations();
|
|
@@ -530,31 +530,61 @@ export async function resumeEvaluation({
|
|
|
530
530
|
processEvaluationsFromChannel()
|
|
531
531
|
);
|
|
532
532
|
|
|
533
|
-
// Wait for producer
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
//
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
//
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
545
|
-
|
|
546
|
-
|
|
547
|
-
|
|
548
|
-
|
|
533
|
+
// Wait for the producer AND every worker to settle before continuing.
|
|
534
|
+
// Using allSettled (rather than Promise.all) is important: on the first
|
|
535
|
+
// worker error, Promise.all rejects immediately while the remaining
|
|
536
|
+
// workers keep running detached, logging and hitting the API after this
|
|
537
|
+
// function has already returned/thrown. Draining all tasks guarantees no
|
|
538
|
+
// background work outlives the call (and avoids teardown races in tests
|
|
539
|
+
// where late console output is flushed after the test completes).
|
|
540
|
+
const settled = await Promise.allSettled([producerTask, ...workerTasks]);
|
|
541
|
+
const rejections = settled
|
|
542
|
+
.filter(
|
|
543
|
+
(result): result is PromiseRejectedResult =>
|
|
544
|
+
result.status === "rejected"
|
|
545
|
+
)
|
|
546
|
+
.map((result) => result.reason);
|
|
547
|
+
|
|
548
|
+
if (rejections.length > 0) {
|
|
549
|
+
// Classify and handle errors based on their nature. When multiple tasks
|
|
550
|
+
// reject, prefer the most meaningful error over incidental fallout
|
|
551
|
+
// (e.g. a ChannelError raised in a blocked worker when the channel
|
|
552
|
+
// closes on abort).
|
|
553
|
+
const fetchError = rejections.find(
|
|
554
|
+
(reason) => reason instanceof EvaluationFetchError
|
|
549
555
|
);
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
556
|
+
const workerError = rejections.find(
|
|
557
|
+
(reason) =>
|
|
558
|
+
reason instanceof Error &&
|
|
559
|
+
!(reason instanceof EvaluationFetchError) &&
|
|
560
|
+
!(reason instanceof ChannelError)
|
|
561
|
+
);
|
|
562
|
+
const channelError = rejections.find(
|
|
563
|
+
(reason) => reason instanceof ChannelError
|
|
564
|
+
);
|
|
565
|
+
|
|
566
|
+
if (fetchError) {
|
|
567
|
+
// Producer failed - this is ALWAYS critical regardless of stopOnFirstError
|
|
568
|
+
logger.error(`Critical: Failed to fetch evaluations from server`);
|
|
569
|
+
executionError = fetchError;
|
|
570
|
+
} else if (workerError) {
|
|
571
|
+
// Worker error in stopOnFirstError mode - already logged by worker
|
|
572
|
+
executionError = workerError;
|
|
573
|
+
} else if (channelError && signal.aborted) {
|
|
574
|
+
// Channel closed due to intentional abort - wrap in semantic error
|
|
575
|
+
executionError = new EvaluationAbortedError(
|
|
576
|
+
"Evaluation stopped due to error in concurrent evaluator",
|
|
577
|
+
channelError
|
|
578
|
+
);
|
|
579
|
+
} else {
|
|
580
|
+
// Unexpected error (not from worker, not from producer fetch)
|
|
581
|
+
// This could be a bug in our code or infrastructure failure
|
|
582
|
+
const reason = rejections[0];
|
|
583
|
+
const err =
|
|
584
|
+
reason instanceof Error ? reason : new Error(String(reason));
|
|
585
|
+
logger.error(`Unexpected error during evaluation: ${err.message}`);
|
|
586
|
+
executionError = err;
|
|
587
|
+
}
|
|
558
588
|
}
|
|
559
589
|
} finally {
|
|
560
590
|
// Ensure channel is closed even if there are unexpected errors
|
|
@@ -489,7 +489,7 @@ export async function resumeExperiment({
|
|
|
489
489
|
}
|
|
490
490
|
|
|
491
491
|
// Start concurrent execution
|
|
492
|
-
// Wrap in try-finally to ensure channel is always closed, even if
|
|
492
|
+
// Wrap in try-finally to ensure channel is always closed, even if a task throws
|
|
493
493
|
let executionError: Error | null = null;
|
|
494
494
|
try {
|
|
495
495
|
const producerTask = fetchIncompleteRuns();
|
|
@@ -497,31 +497,63 @@ export async function resumeExperiment({
|
|
|
497
497
|
processTasksFromChannel()
|
|
498
498
|
);
|
|
499
499
|
|
|
500
|
-
// Wait for producer
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
//
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
//
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
500
|
+
// Wait for the producer AND every worker to settle before continuing.
|
|
501
|
+
// Using allSettled (rather than Promise.all) is important: on the first
|
|
502
|
+
// worker error, Promise.all rejects immediately while the remaining
|
|
503
|
+
// workers keep running detached, logging and hitting the API after this
|
|
504
|
+
// function has already returned/thrown. Draining all tasks guarantees no
|
|
505
|
+
// background work outlives the call (and avoids teardown races in tests
|
|
506
|
+
// where late console output is flushed after the test completes).
|
|
507
|
+
const settled = await Promise.allSettled([producerTask, ...workerTasks]);
|
|
508
|
+
const rejections = settled
|
|
509
|
+
.filter(
|
|
510
|
+
(result): result is PromiseRejectedResult =>
|
|
511
|
+
result.status === "rejected"
|
|
512
|
+
)
|
|
513
|
+
.map((result) => result.reason);
|
|
514
|
+
|
|
515
|
+
if (rejections.length > 0) {
|
|
516
|
+
// Classify and handle errors based on their nature. When multiple tasks
|
|
517
|
+
// reject, prefer the most meaningful error over incidental fallout
|
|
518
|
+
// (e.g. a ChannelError raised in a blocked worker when the channel
|
|
519
|
+
// closes on abort).
|
|
520
|
+
const fetchError = rejections.find(
|
|
521
|
+
(reason) => reason instanceof TaskFetchError
|
|
516
522
|
);
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
523
|
+
const taskError = rejections.find(
|
|
524
|
+
(reason) =>
|
|
525
|
+
reason instanceof Error &&
|
|
526
|
+
!(reason instanceof TaskFetchError) &&
|
|
527
|
+
!(reason instanceof ChannelError)
|
|
528
|
+
);
|
|
529
|
+
const channelError = rejections.find(
|
|
530
|
+
(reason) => reason instanceof ChannelError
|
|
531
|
+
);
|
|
532
|
+
|
|
533
|
+
if (fetchError) {
|
|
534
|
+
// Producer failed - this is ALWAYS critical regardless of stopOnFirstError
|
|
535
|
+
logger.error(`Critical: Failed to fetch incomplete runs from server`);
|
|
536
|
+
executionError = fetchError;
|
|
537
|
+
} else if (taskError) {
|
|
538
|
+
// Worker error in stopOnFirstError mode - already logged by worker
|
|
539
|
+
executionError = taskError;
|
|
540
|
+
} else if (channelError && signal.aborted) {
|
|
541
|
+
// Channel closed due to intentional abort - wrap in semantic error
|
|
542
|
+
executionError = new TaskAbortedError(
|
|
543
|
+
"Task execution stopped due to error in concurrent worker",
|
|
544
|
+
channelError
|
|
545
|
+
);
|
|
546
|
+
} else {
|
|
547
|
+
// Unexpected error (not from worker, not from producer fetch)
|
|
548
|
+
// This could be a bug in our code or infrastructure failure
|
|
549
|
+
const reason = rejections[0];
|
|
550
|
+
const err =
|
|
551
|
+
reason instanceof Error ? reason : new Error(String(reason));
|
|
552
|
+
logger.error(
|
|
553
|
+
`Unexpected error during task execution: ${err.message}`
|
|
554
|
+
);
|
|
555
|
+
executionError = err;
|
|
556
|
+
}
|
|
525
557
|
}
|
|
526
558
|
} finally {
|
|
527
559
|
// Ensure channel is closed even if there are unexpected errors
|