@arizeai/phoenix-client 6.14.2 → 7.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/dist/esm/__generated__/api/v1.d.ts +8 -0
  2. package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
  3. package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
  4. package/dist/esm/experiments/resumeEvaluation.js +43 -26
  5. package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
  6. package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
  7. package/dist/esm/experiments/resumeExperiment.js +43 -26
  8. package/dist/esm/experiments/resumeExperiment.js.map +1 -1
  9. package/dist/esm/prompts/sdks/toAI.d.ts +2 -1
  10. package/dist/esm/prompts/sdks/toAI.d.ts.map +1 -1
  11. package/dist/esm/prompts/sdks/toAI.js.map +1 -1
  12. package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
  13. package/dist/src/__generated__/api/v1.d.ts +8 -0
  14. package/dist/src/__generated__/api/v1.d.ts.map +1 -1
  15. package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
  16. package/dist/src/experiments/resumeEvaluation.js +43 -26
  17. package/dist/src/experiments/resumeEvaluation.js.map +1 -1
  18. package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
  19. package/dist/src/experiments/resumeExperiment.js +43 -26
  20. package/dist/src/experiments/resumeExperiment.js.map +1 -1
  21. package/dist/src/prompts/sdks/toAI.d.ts +2 -1
  22. package/dist/src/prompts/sdks/toAI.d.ts.map +1 -1
  23. package/dist/src/prompts/sdks/toAI.js.map +1 -1
  24. package/dist/tsconfig.tsbuildinfo +1 -1
  25. package/docs/datasets.mdx +15 -9
  26. package/docs/experiments.mdx +18 -19
  27. package/package.json +10 -10
  28. package/src/__generated__/api/v1.ts +8 -0
  29. package/src/experiments/resumeEvaluation.ts +55 -25
  30. package/src/experiments/resumeExperiment.ts +57 -25
  31. package/src/prompts/sdks/toAI.ts +3 -1
package/docs/datasets.mdx CHANGED
@@ -3,14 +3,14 @@ title: "Datasets"
3
3
  description: "Create and inspect datasets with @arizeai/phoenix-client"
4
4
  ---
5
5
 
6
- Datasets are the foundation for experiment runs. The dataset helpers cover creation, idempotent creation, record inspection, and example appends.
6
+ Datasets are the foundation for experiment runs. The dataset helpers cover creation (which upserts by name), record inspection, and example appends.
7
7
 
8
8
  <section className="hidden" data-agent-context="relevant-source-files" aria-label="Relevant source files">
9
9
  <h2>Relevant Source Files</h2>
10
10
  <ul>
11
11
  <li>
12
- <code>src/datasets/createOrGetDataset.ts</code> for the exact return
13
- shape of the idempotent helper
12
+ <code>src/datasets/createDataset.ts</code> for the exact return shape and
13
+ upsert-by-name behavior
14
14
  </li>
15
15
  </ul>
16
16
  </section>
@@ -33,18 +33,25 @@ const { datasetId } = await createDataset({
33
33
  });
34
34
  ```
35
35
 
36
- ## Reuse Or Append
36
+ ## Upsert Or Append
37
+
38
+ `createDataset()` upserts by name: re-running it with the same name updates the existing dataset to match the examples you pass, and an unchanged upload is a no-op. To keep existing examples and add more, use `appendDatasetExamples()` instead of re-running `createDataset()` with the extra examples.
37
39
 
38
40
  ```ts
39
41
  import {
40
42
  appendDatasetExamples,
41
- createOrGetDataset,
43
+ createDataset,
42
44
  } from "@arizeai/phoenix-client/datasets";
43
45
 
44
- const dataset = await createOrGetDataset({
46
+ const dataset = await createDataset({
45
47
  name: "support-eval",
46
48
  description: "Support questions with expected answers",
47
- examples: [],
49
+ examples: [
50
+ {
51
+ input: { question: "Where is my order?" },
52
+ output: { answer: "Use the tracking page in your account." },
53
+ },
54
+ ],
48
55
  });
49
56
 
50
57
  await appendDatasetExamples({
@@ -58,7 +65,7 @@ await appendDatasetExamples({
58
65
  });
59
66
  ```
60
67
 
61
- `createOrGetDataset()` returns `{ datasetId }`, so you can pass that object directly as the dataset selector for append or experiment calls.
68
+ `createDataset()` returns `{ datasetId }`, so you can pass that object directly as the dataset selector for append or experiment calls.
62
69
 
63
70
  ## Read Back Dataset State
64
71
 
@@ -68,7 +75,6 @@ Use `getDataset`, `getDatasetExamples`, and `getDatasetInfo` to inspect datasets
68
75
  <h2>Source Map</h2>
69
76
  <ul>
70
77
  <li><code>src/datasets/createDataset.ts</code></li>
71
- <li><code>src/datasets/createOrGetDataset.ts</code></li>
72
78
  <li><code>src/datasets/appendDatasetExamples.ts</code></li>
73
79
  <li><code>src/datasets/getDataset.ts</code></li>
74
80
  <li><code>src/datasets/getDatasetExamples.ts</code></li>
@@ -111,11 +111,18 @@ This pattern is useful when:
111
111
 
112
112
  ## Model-Backed Example
113
113
 
114
- If you want a model-backed experiment with automatic tracing and an LLM-as-a-judge evaluator, this is the core pattern:
114
+ If you want a model-backed experiment with automatic tracing and an LLM-as-a-judge evaluator, this is the core pattern.
115
+
116
+ Install the AI SDK packages alongside the Phoenix packages — `@ai-sdk/otel` provides the telemetry integration that sends AI SDK traces to Phoenix:
117
+
118
+ ```bash
119
+ npm install @arizeai/phoenix-client @arizeai/phoenix-evals ai @ai-sdk/openai @ai-sdk/otel
120
+ ```
115
121
 
116
122
  ```ts
117
123
  import { openai } from "@ai-sdk/openai";
118
- import { createOrGetDataset } from "@arizeai/phoenix-client/datasets";
124
+ import { OpenTelemetry } from "@ai-sdk/otel";
125
+ import { createDataset } from "@arizeai/phoenix-client/datasets";
119
126
  import { runExperiment } from "@arizeai/phoenix-client/experiments";
120
127
  import type { ExperimentTask } from "@arizeai/phoenix-client/types/experiments";
121
128
  import { createClassificationEvaluator } from "@arizeai/phoenix-evals";
@@ -135,7 +142,7 @@ const main = async () => {
135
142
  },
136
143
  });
137
144
 
138
- const dataset = await createOrGetDataset({
145
+ const dataset = await createDataset({
139
146
  name: "correctness-eval",
140
147
  description: "Evaluate the correctness of the model",
141
148
  examples: [
@@ -157,23 +164,15 @@ const main = async () => {
157
164
  throw new Error("Invalid input: context must be a string");
158
165
  }
159
166
 
167
+ // The per-call `@ai-sdk/otel` integration traces this call through the
168
+ // experiment's tracer provider that runExperiment mounts while tasks run.
160
169
  return generateText({
161
170
  model,
162
- experimental_telemetry: {
163
- isEnabled: true,
164
- },
165
- prompt: [
166
- {
167
- role: "system",
168
- content: `You answer questions based on this context: ${example.input.context}`,
169
- },
170
- {
171
- role: "user",
172
- content: example.input.question,
173
- },
174
- ],
171
+ instructions: `You answer questions based on this context: ${example.input.context}`,
172
+ prompt: example.input.question,
173
+ telemetry: { integrations: [new OpenTelemetry()] },
175
174
  }).then((response) => {
176
- if (response.text) {
175
+ if (typeof response.text === "string") {
177
176
  return response.text;
178
177
  }
179
178
  throw new Error("Invalid response: text is required");
@@ -200,9 +199,9 @@ main().catch(console.error);
200
199
 
201
200
  ## What This Example Shows
202
201
 
203
- - `createOrGetDataset()` creates or reuses the dataset the experiment will run against
202
+ - `createDataset()` creates or reuses the dataset the experiment will run against (it upserts by name)
204
203
  - `task` receives the full dataset example object
205
- - `generateText()` emits traces that Phoenix can attach to the experiment when telemetry is enabled
204
+ - `generateText()` emits traces that Phoenix attaches to the experiment because the call passes the `@ai-sdk/otel` integration via `telemetry.integrations` — without it, no AI SDK spans reach Phoenix
206
205
  - `createClassificationEvaluator()` from `@arizeai/phoenix-evals` can be passed directly to `runExperiment()`
207
206
  - `runExperiment()` records both task runs and evaluation runs in Phoenix
208
207
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@arizeai/phoenix-client",
3
- "version": "6.14.2",
3
+ "version": "7.0.1",
4
4
  "description": "A client for the Phoenix API",
5
5
  "keywords": [
6
6
  "arize",
@@ -96,34 +96,34 @@
96
96
  },
97
97
  "dependencies": {
98
98
  "@arizeai/openinference-semantic-conventions": "^2.5.0",
99
- "@arizeai/openinference-vercel": "^2.8.1",
100
99
  "async": "^3.2.6",
101
100
  "openapi-fetch": "^0.17.0",
102
101
  "tiny-invariant": "^1.3.3",
103
102
  "zod": "^4.4.3",
104
- "@arizeai/phoenix-otel": "2.1.0",
105
- "@arizeai/phoenix-config": "0.4.0"
103
+ "@arizeai/phoenix-config": "0.4.0",
104
+ "@arizeai/phoenix-otel": "2.1.0"
106
105
  },
107
106
  "devDependencies": {
108
- "@ai-sdk/openai": "^3.0.86",
107
+ "@ai-sdk/openai": "^4.0.0",
108
+ "@ai-sdk/otel": "^1.0.0",
109
109
  "@anthropic-ai/sdk": "^0.111.0",
110
110
  "@opentelemetry/api": "^1.9.1",
111
111
  "@opentelemetry/sdk-trace-node": "^2.9.0",
112
112
  "@types/async": "^3.2.25",
113
113
  "@types/node": "^26.1.1",
114
- "ai": "^6.0.230",
114
+ "ai": "^7.0.0",
115
115
  "dotenv": "^17.4.2",
116
116
  "jest": "^30.4.2",
117
117
  "openai": "^6.48.0",
118
118
  "openapi-typescript": "^7.13.0",
119
119
  "tsx": "^4.23.1",
120
120
  "vitest": "^4.1.10",
121
- "@arizeai/phoenix-testing": "0.0.0",
122
- "@arizeai/phoenix-evals": "1.2.0"
121
+ "@arizeai/phoenix-evals": "2.1.0",
122
+ "@arizeai/phoenix-testing": "0.0.0"
123
123
  },
124
124
  "peerDependencies": {
125
125
  "@anthropic-ai/sdk": "^0.35.0",
126
- "ai": "^6.0.90",
126
+ "ai": "^7.0.0",
127
127
  "jest": ">=27",
128
128
  "openai": "^6.10.0",
129
129
  "vitest": ">=1"
@@ -156,6 +156,6 @@
156
156
  "prebuild": "pnpm run clean && pnpm run generate",
157
157
  "test": "vitest run",
158
158
  "test:watch": "vitest watch",
159
- "typecheck": "tsc --noEmit"
159
+ "typecheck": "tsc --noEmit && tsc -p examples/tsconfig.examples.json"
160
160
  }
161
161
  }
@@ -2417,6 +2417,14 @@ export interface components {
2417
2417
  * Format: date-time
2418
2418
  */
2419
2419
  updated_at: string;
2420
+ source?: components["schemas"]["DatasetExampleSource"] | null;
2421
+ };
2422
+ /** DatasetExampleSource */
2423
+ DatasetExampleSource: {
2424
+ /** Span Id */
2425
+ span_id: string;
2426
+ /** Span Node Id */
2427
+ span_node_id: string;
2420
2428
  };
2421
2429
  /** DatasetLabel */
2422
2430
  DatasetLabel: {
@@ -522,7 +522,7 @@ export async function resumeEvaluation({
522
522
  }
523
523
 
524
524
  // Start concurrent execution
525
- // Wrap in try-finally to ensure channel is always closed, even if Promise.all throws
525
+ // Wrap in try-finally to ensure channel is always closed, even if a task throws
526
526
  let executionError: Error | null = null;
527
527
  try {
528
528
  const producerTask = fetchIncompleteEvaluations();
@@ -530,31 +530,61 @@ export async function resumeEvaluation({
530
530
  processEvaluationsFromChannel()
531
531
  );
532
532
 
533
- // Wait for producer and all workers to finish
534
- await Promise.all([producerTask, ...workerTasks]);
535
- } catch (error) {
536
- // Classify and handle errors based on their nature
537
- const err = error instanceof Error ? error : new Error(String(error));
538
-
539
- // Always surface producer/infrastructure errors
540
- if (error instanceof EvaluationFetchError) {
541
- // Producer failed - this is ALWAYS critical regardless of stopOnFirstError
542
- logger.error(`Critical: Failed to fetch evaluations from server`);
543
- executionError = err;
544
- } else if (error instanceof ChannelError && signal.aborted) {
545
- // Channel closed due to intentional abort - wrap in semantic error
546
- executionError = new EvaluationAbortedError(
547
- "Evaluation stopped due to error in concurrent evaluator",
548
- err
533
+ // Wait for the producer AND every worker to settle before continuing.
534
+ // Using allSettled (rather than Promise.all) is important: on the first
535
+ // worker error, Promise.all rejects immediately while the remaining
536
+ // workers keep running detached, logging and hitting the API after this
537
+ // function has already returned/thrown. Draining all tasks guarantees no
538
+ // background work outlives the call (and avoids teardown races in tests
539
+ // where late console output is flushed after the test completes).
540
+ const settled = await Promise.allSettled([producerTask, ...workerTasks]);
541
+ const rejections = settled
542
+ .filter(
543
+ (result): result is PromiseRejectedResult =>
544
+ result.status === "rejected"
545
+ )
546
+ .map((result) => result.reason);
547
+
548
+ if (rejections.length > 0) {
549
+ // Classify and handle errors based on their nature. When multiple tasks
550
+ // reject, prefer the most meaningful error over incidental fallout
551
+ // (e.g. a ChannelError raised in a blocked worker when the channel
552
+ // closes on abort).
553
+ const fetchError = rejections.find(
554
+ (reason) => reason instanceof EvaluationFetchError
549
555
  );
550
- } else if (stopOnFirstError) {
551
- // Worker error in stopOnFirstError mode - already logged by worker
552
- executionError = err;
553
- } else {
554
- // Unexpected error (not from worker, not from producer fetch)
555
- // This could be a bug in our code or infrastructure failure
556
- logger.error(`Unexpected error during evaluation: ${err.message}`);
557
- executionError = err;
556
+ const workerError = rejections.find(
557
+ (reason) =>
558
+ reason instanceof Error &&
559
+ !(reason instanceof EvaluationFetchError) &&
560
+ !(reason instanceof ChannelError)
561
+ );
562
+ const channelError = rejections.find(
563
+ (reason) => reason instanceof ChannelError
564
+ );
565
+
566
+ if (fetchError) {
567
+ // Producer failed - this is ALWAYS critical regardless of stopOnFirstError
568
+ logger.error(`Critical: Failed to fetch evaluations from server`);
569
+ executionError = fetchError;
570
+ } else if (workerError) {
571
+ // Worker error in stopOnFirstError mode - already logged by worker
572
+ executionError = workerError;
573
+ } else if (channelError && signal.aborted) {
574
+ // Channel closed due to intentional abort - wrap in semantic error
575
+ executionError = new EvaluationAbortedError(
576
+ "Evaluation stopped due to error in concurrent evaluator",
577
+ channelError
578
+ );
579
+ } else {
580
+ // Unexpected error (not from worker, not from producer fetch)
581
+ // This could be a bug in our code or infrastructure failure
582
+ const reason = rejections[0];
583
+ const err =
584
+ reason instanceof Error ? reason : new Error(String(reason));
585
+ logger.error(`Unexpected error during evaluation: ${err.message}`);
586
+ executionError = err;
587
+ }
558
588
  }
559
589
  } finally {
560
590
  // Ensure channel is closed even if there are unexpected errors
@@ -489,7 +489,7 @@ export async function resumeExperiment({
489
489
  }
490
490
 
491
491
  // Start concurrent execution
492
- // Wrap in try-finally to ensure channel is always closed, even if Promise.all throws
492
+ // Wrap in try-finally to ensure channel is always closed, even if a task throws
493
493
  let executionError: Error | null = null;
494
494
  try {
495
495
  const producerTask = fetchIncompleteRuns();
@@ -497,31 +497,63 @@ export async function resumeExperiment({
497
497
  processTasksFromChannel()
498
498
  );
499
499
 
500
- // Wait for producer and all workers to finish
501
- await Promise.all([producerTask, ...workerTasks]);
502
- } catch (error) {
503
- // Classify and handle errors based on their nature
504
- const err = error instanceof Error ? error : new Error(String(error));
505
-
506
- // Always surface producer/infrastructure errors
507
- if (error instanceof TaskFetchError) {
508
- // Producer failed - this is ALWAYS critical regardless of stopOnFirstError
509
- logger.error(`Critical: Failed to fetch incomplete runs from server`);
510
- executionError = err;
511
- } else if (error instanceof ChannelError && signal.aborted) {
512
- // Channel closed due to intentional abort - wrap in semantic error
513
- executionError = new TaskAbortedError(
514
- "Task execution stopped due to error in concurrent worker",
515
- err
500
+ // Wait for the producer AND every worker to settle before continuing.
501
+ // Using allSettled (rather than Promise.all) is important: on the first
502
+ // worker error, Promise.all rejects immediately while the remaining
503
+ // workers keep running detached, logging and hitting the API after this
504
+ // function has already returned/thrown. Draining all tasks guarantees no
505
+ // background work outlives the call (and avoids teardown races in tests
506
+ // where late console output is flushed after the test completes).
507
+ const settled = await Promise.allSettled([producerTask, ...workerTasks]);
508
+ const rejections = settled
509
+ .filter(
510
+ (result): result is PromiseRejectedResult =>
511
+ result.status === "rejected"
512
+ )
513
+ .map((result) => result.reason);
514
+
515
+ if (rejections.length > 0) {
516
+ // Classify and handle errors based on their nature. When multiple tasks
517
+ // reject, prefer the most meaningful error over incidental fallout
518
+ // (e.g. a ChannelError raised in a blocked worker when the channel
519
+ // closes on abort).
520
+ const fetchError = rejections.find(
521
+ (reason) => reason instanceof TaskFetchError
516
522
  );
517
- } else if (stopOnFirstError) {
518
- // Worker error in stopOnFirstError mode - already logged by worker
519
- executionError = err;
520
- } else {
521
- // Unexpected error (not from worker, not from producer fetch)
522
- // This could be a bug in our code or infrastructure failure
523
- logger.error(`Unexpected error during task execution: ${err.message}`);
524
- executionError = err;
523
+ const taskError = rejections.find(
524
+ (reason) =>
525
+ reason instanceof Error &&
526
+ !(reason instanceof TaskFetchError) &&
527
+ !(reason instanceof ChannelError)
528
+ );
529
+ const channelError = rejections.find(
530
+ (reason) => reason instanceof ChannelError
531
+ );
532
+
533
+ if (fetchError) {
534
+ // Producer failed - this is ALWAYS critical regardless of stopOnFirstError
535
+ logger.error(`Critical: Failed to fetch incomplete runs from server`);
536
+ executionError = fetchError;
537
+ } else if (taskError) {
538
+ // Worker error in stopOnFirstError mode - already logged by worker
539
+ executionError = taskError;
540
+ } else if (channelError && signal.aborted) {
541
+ // Channel closed due to intentional abort - wrap in semantic error
542
+ executionError = new TaskAbortedError(
543
+ "Task execution stopped due to error in concurrent worker",
544
+ channelError
545
+ );
546
+ } else {
547
+ // Unexpected error (not from worker, not from producer fetch)
548
+ // This could be a bug in our code or infrastructure failure
549
+ const reason = rejections[0];
550
+ const err =
551
+ reason instanceof Error ? reason : new Error(String(reason));
552
+ logger.error(
553
+ `Unexpected error during task execution: ${err.message}`
554
+ );
555
+ executionError = err;
556
+ }
525
557
  }
526
558
  } finally {
527
559
  // Ensure channel is closed even if there are unexpected errors
@@ -1,4 +1,6 @@
1
- import type { ModelMessage, ToolChoice, ToolSet } from "ai";
1
+ import type { ModelMessage, ToolChoice, ToolSet } from "ai" with {
2
+ "resolution-mode": "import",
3
+ };
2
4
  import invariant from "tiny-invariant";
3
5
 
4
6
  import {