@arizeai/phoenix-client 7.0.0 → 7.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/dist/esm/__generated__/api/v1.d.ts +12 -0
  2. package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
  3. package/dist/esm/constants/serverRequirements.d.ts +1 -0
  4. package/dist/esm/constants/serverRequirements.d.ts.map +1 -1
  5. package/dist/esm/constants/serverRequirements.js +8 -0
  6. package/dist/esm/constants/serverRequirements.js.map +1 -1
  7. package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
  8. package/dist/esm/experiments/resumeEvaluation.js +43 -26
  9. package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
  10. package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
  11. package/dist/esm/experiments/resumeExperiment.js +43 -26
  12. package/dist/esm/experiments/resumeExperiment.js.map +1 -1
  13. package/dist/esm/spans/getSpans.d.ts +11 -1
  14. package/dist/esm/spans/getSpans.d.ts.map +1 -1
  15. package/dist/esm/spans/getSpans.js +16 -2
  16. package/dist/esm/spans/getSpans.js.map +1 -1
  17. package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
  18. package/dist/src/__generated__/api/v1.d.ts +12 -0
  19. package/dist/src/__generated__/api/v1.d.ts.map +1 -1
  20. package/dist/src/constants/serverRequirements.d.ts +1 -0
  21. package/dist/src/constants/serverRequirements.d.ts.map +1 -1
  22. package/dist/src/constants/serverRequirements.js +9 -1
  23. package/dist/src/constants/serverRequirements.js.map +1 -1
  24. package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
  25. package/dist/src/experiments/resumeEvaluation.js +43 -26
  26. package/dist/src/experiments/resumeEvaluation.js.map +1 -1
  27. package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
  28. package/dist/src/experiments/resumeExperiment.js +43 -26
  29. package/dist/src/experiments/resumeExperiment.js.map +1 -1
  30. package/dist/src/spans/getSpans.d.ts +11 -1
  31. package/dist/src/spans/getSpans.d.ts.map +1 -1
  32. package/dist/src/spans/getSpans.js +15 -1
  33. package/dist/src/spans/getSpans.js.map +1 -1
  34. package/dist/tsconfig.tsbuildinfo +1 -1
  35. package/docs/datasets.mdx +15 -9
  36. package/docs/overview.mdx +1 -1
  37. package/package.json +3 -3
  38. package/src/__generated__/api/v1.ts +12 -0
  39. package/src/constants/serverRequirements.ts +9 -0
  40. package/src/experiments/resumeEvaluation.ts +55 -25
  41. package/src/experiments/resumeExperiment.ts +57 -25
  42. package/src/spans/getSpans.ts +19 -0
package/docs/datasets.mdx CHANGED
@@ -3,14 +3,14 @@ title: "Datasets"
3
3
  description: "Create and inspect datasets with @arizeai/phoenix-client"
4
4
  ---
5
5
 
6
- Datasets are the foundation for experiment runs. The dataset helpers cover creation, idempotent creation, record inspection, and example appends.
6
+ Datasets are the foundation for experiment runs. The dataset helpers cover creation (which upserts by name), record inspection, and example appends.
7
7
 
8
8
  <section className="hidden" data-agent-context="relevant-source-files" aria-label="Relevant source files">
9
9
  <h2>Relevant Source Files</h2>
10
10
  <ul>
11
11
  <li>
12
- <code>src/datasets/createOrGetDataset.ts</code> for the exact return
13
- shape of the idempotent helper
12
+ <code>src/datasets/createDataset.ts</code> for the exact return shape and
13
+ upsert-by-name behavior
14
14
  </li>
15
15
  </ul>
16
16
  </section>
@@ -33,18 +33,25 @@ const { datasetId } = await createDataset({
33
33
  });
34
34
  ```
35
35
 
36
- ## Reuse Or Append
36
+ ## Upsert Or Append
37
+
38
+ `createDataset()` upserts by name: re-running it with the same name updates the existing dataset to match the examples you pass, and an unchanged upload is a no-op. To keep existing examples and add more, use `appendDatasetExamples()` instead of re-running `createDataset()` with the extra examples.
37
39
 
38
40
  ```ts
39
41
  import {
40
42
  appendDatasetExamples,
41
- createOrGetDataset,
43
+ createDataset,
42
44
  } from "@arizeai/phoenix-client/datasets";
43
45
 
44
- const dataset = await createOrGetDataset({
46
+ const dataset = await createDataset({
45
47
  name: "support-eval",
46
48
  description: "Support questions with expected answers",
47
- examples: [],
49
+ examples: [
50
+ {
51
+ input: { question: "Where is my order?" },
52
+ output: { answer: "Use the tracking page in your account." },
53
+ },
54
+ ],
48
55
  });
49
56
 
50
57
  await appendDatasetExamples({
@@ -58,7 +65,7 @@ await appendDatasetExamples({
58
65
  });
59
66
  ```
60
67
 
61
- `createOrGetDataset()` returns `{ datasetId }`, so you can pass that object directly as the dataset selector for append or experiment calls.
68
+ `createDataset()` returns `{ datasetId }`, so you can pass that object directly as the dataset selector for append or experiment calls.
62
69
 
63
70
  ## Read Back Dataset State
64
71
 
@@ -68,7 +75,6 @@ Use `getDataset`, `getDatasetExamples`, and `getDatasetInfo` to inspect datasets
68
75
  <h2>Source Map</h2>
69
76
  <ul>
70
77
  <li><code>src/datasets/createDataset.ts</code></li>
71
- <li><code>src/datasets/createOrGetDataset.ts</code></li>
72
78
  <li><code>src/datasets/appendDatasetExamples.ts</code></li>
73
79
  <li><code>src/datasets/getDataset.ts</code></li>
74
80
  <li><code>src/datasets/getDatasetExamples.ts</code></li>
package/docs/overview.mdx CHANGED
@@ -61,7 +61,7 @@ export PHOENIX_HOST=http://localhost:6006
61
61
  export PHOENIX_API_KEY=<your-api-key>
62
62
  ```
63
63
 
64
- If you're using Phoenix Cloud, `PHOENIX_HOST` may look like `https://app.phoenix.arize.com/s/my-space`.
64
+ For a remote deployment, `PHOENIX_HOST` is that instance's base URL, e.g. `https://your-phoenix.example.com`.
65
65
 
66
66
  ```ts
67
67
  import { createClient } from "@arizeai/phoenix-client";
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@arizeai/phoenix-client",
3
- "version": "7.0.0",
3
+ "version": "7.1.0",
4
4
  "description": "A client for the Phoenix API",
5
5
  "keywords": [
6
6
  "arize",
@@ -118,8 +118,8 @@
118
118
  "openapi-typescript": "^7.13.0",
119
119
  "tsx": "^4.23.1",
120
120
  "vitest": "^4.1.10",
121
- "@arizeai/phoenix-evals": "2.0.0",
122
- "@arizeai/phoenix-testing": "0.0.0"
121
+ "@arizeai/phoenix-testing": "0.0.0",
122
+ "@arizeai/phoenix-evals": "2.1.0"
123
123
  },
124
124
  "peerDependencies": {
125
125
  "@anthropic-ai/sdk": "^0.35.0",
@@ -2417,6 +2417,14 @@ export interface components {
2417
2417
  * Format: date-time
2418
2418
  */
2419
2419
  updated_at: string;
2420
+ source?: components["schemas"]["DatasetExampleSource"] | null;
2421
+ };
2422
+ /** DatasetExampleSource */
2423
+ DatasetExampleSource: {
2424
+ /** Span Id */
2425
+ span_id: string;
2426
+ /** Span Node Id */
2427
+ span_node_id: string;
2420
2428
  };
2421
2429
  /** DatasetLabel */
2422
2430
  DatasetLabel: {
@@ -8628,6 +8636,8 @@ export interface operations {
8628
8636
  end_time?: string | null;
8629
8637
  /** @description Filter by one or more trace IDs */
8630
8638
  trace_id?: string[] | null;
8639
+ /** @description Filter by one or more span IDs */
8640
+ span_id?: string[] | null;
8631
8641
  /** @description Filter by parent span ID. Use "null" to get root spans only. */
8632
8642
  parent_id?: string | null;
8633
8643
  /** @description Filter by span name(s) */
@@ -8697,6 +8707,8 @@ export interface operations {
8697
8707
  end_time?: string | null;
8698
8708
  /** @description Filter by one or more trace IDs */
8699
8709
  trace_id?: string[] | null;
8710
+ /** @description Filter by one or more span IDs */
8711
+ span_id?: string[] | null;
8700
8712
  /** @description Filter by parent span ID. Use "null" to get root spans only. */
8701
8713
  parent_id?: string | null;
8702
8714
  /** @description Filter by span name(s) */
@@ -75,6 +75,14 @@ export const GET_SPANS_TRACE_IDS: ParameterRequirement = {
75
75
  minServerVersion: [13, 9, 0],
76
76
  };
77
77
 
78
+ export const GET_SPANS_SPAN_IDS: ParameterRequirement = {
79
+ kind: "parameter",
80
+ parameterName: "span_id",
81
+ parameterLocation: "query",
82
+ route: "GET /v1/projects/{id}/spans",
83
+ minServerVersion: [19, 6, 0],
84
+ };
85
+
78
86
  export const GET_SPANS_FILTERS: ParameterRequirement = {
79
87
  kind: "parameter",
80
88
  parameterName: "span_kind",
@@ -145,6 +153,7 @@ export const ALL_REQUIREMENTS: readonly CapabilityRequirement[] = [
145
153
  ADD_TRACE_NOTE,
146
154
  ADD_SESSION_NOTE,
147
155
  GET_SPANS_TRACE_IDS,
156
+ GET_SPANS_SPAN_IDS,
148
157
  GET_SPANS_FILTERS,
149
158
  GET_SPANS_BY_ATTRIBUTE,
150
159
  LIST_PROJECT_TRACES,
@@ -522,7 +522,7 @@ export async function resumeEvaluation({
522
522
  }
523
523
 
524
524
  // Start concurrent execution
525
- // Wrap in try-finally to ensure channel is always closed, even if Promise.all throws
525
+ // Wrap in try-finally to ensure channel is always closed, even if a task throws
526
526
  let executionError: Error | null = null;
527
527
  try {
528
528
  const producerTask = fetchIncompleteEvaluations();
@@ -530,31 +530,61 @@ export async function resumeEvaluation({
530
530
  processEvaluationsFromChannel()
531
531
  );
532
532
 
533
- // Wait for producer and all workers to finish
534
- await Promise.all([producerTask, ...workerTasks]);
535
- } catch (error) {
536
- // Classify and handle errors based on their nature
537
- const err = error instanceof Error ? error : new Error(String(error));
538
-
539
- // Always surface producer/infrastructure errors
540
- if (error instanceof EvaluationFetchError) {
541
- // Producer failed - this is ALWAYS critical regardless of stopOnFirstError
542
- logger.error(`Critical: Failed to fetch evaluations from server`);
543
- executionError = err;
544
- } else if (error instanceof ChannelError && signal.aborted) {
545
- // Channel closed due to intentional abort - wrap in semantic error
546
- executionError = new EvaluationAbortedError(
547
- "Evaluation stopped due to error in concurrent evaluator",
548
- err
533
+ // Wait for the producer AND every worker to settle before continuing.
534
+ // Using allSettled (rather than Promise.all) is important: on the first
535
+ // worker error, Promise.all rejects immediately while the remaining
536
+ // workers keep running detached, logging and hitting the API after this
537
+ // function has already returned/thrown. Draining all tasks guarantees no
538
+ // background work outlives the call (and avoids teardown races in tests
539
+ // where late console output is flushed after the test completes).
540
+ const settled = await Promise.allSettled([producerTask, ...workerTasks]);
541
+ const rejections = settled
542
+ .filter(
543
+ (result): result is PromiseRejectedResult =>
544
+ result.status === "rejected"
545
+ )
546
+ .map((result) => result.reason);
547
+
548
+ if (rejections.length > 0) {
549
+ // Classify and handle errors based on their nature. When multiple tasks
550
+ // reject, prefer the most meaningful error over incidental fallout
551
+ // (e.g. a ChannelError raised in a blocked worker when the channel
552
+ // closes on abort).
553
+ const fetchError = rejections.find(
554
+ (reason) => reason instanceof EvaluationFetchError
549
555
  );
550
- } else if (stopOnFirstError) {
551
- // Worker error in stopOnFirstError mode - already logged by worker
552
- executionError = err;
553
- } else {
554
- // Unexpected error (not from worker, not from producer fetch)
555
- // This could be a bug in our code or infrastructure failure
556
- logger.error(`Unexpected error during evaluation: ${err.message}`);
557
- executionError = err;
556
+ const workerError = rejections.find(
557
+ (reason) =>
558
+ reason instanceof Error &&
559
+ !(reason instanceof EvaluationFetchError) &&
560
+ !(reason instanceof ChannelError)
561
+ );
562
+ const channelError = rejections.find(
563
+ (reason) => reason instanceof ChannelError
564
+ );
565
+
566
+ if (fetchError) {
567
+ // Producer failed - this is ALWAYS critical regardless of stopOnFirstError
568
+ logger.error(`Critical: Failed to fetch evaluations from server`);
569
+ executionError = fetchError;
570
+ } else if (workerError) {
571
+ // Worker error in stopOnFirstError mode - already logged by worker
572
+ executionError = workerError;
573
+ } else if (channelError && signal.aborted) {
574
+ // Channel closed due to intentional abort - wrap in semantic error
575
+ executionError = new EvaluationAbortedError(
576
+ "Evaluation stopped due to error in concurrent evaluator",
577
+ channelError
578
+ );
579
+ } else {
580
+ // Unexpected error (not from worker, not from producer fetch)
581
+ // This could be a bug in our code or infrastructure failure
582
+ const reason = rejections[0];
583
+ const err =
584
+ reason instanceof Error ? reason : new Error(String(reason));
585
+ logger.error(`Unexpected error during evaluation: ${err.message}`);
586
+ executionError = err;
587
+ }
558
588
  }
559
589
  } finally {
560
590
  // Ensure channel is closed even if there are unexpected errors
@@ -489,7 +489,7 @@ export async function resumeExperiment({
489
489
  }
490
490
 
491
491
  // Start concurrent execution
492
- // Wrap in try-finally to ensure channel is always closed, even if Promise.all throws
492
+ // Wrap in try-finally to ensure channel is always closed, even if a task throws
493
493
  let executionError: Error | null = null;
494
494
  try {
495
495
  const producerTask = fetchIncompleteRuns();
@@ -497,31 +497,63 @@ export async function resumeExperiment({
497
497
  processTasksFromChannel()
498
498
  );
499
499
 
500
- // Wait for producer and all workers to finish
501
- await Promise.all([producerTask, ...workerTasks]);
502
- } catch (error) {
503
- // Classify and handle errors based on their nature
504
- const err = error instanceof Error ? error : new Error(String(error));
505
-
506
- // Always surface producer/infrastructure errors
507
- if (error instanceof TaskFetchError) {
508
- // Producer failed - this is ALWAYS critical regardless of stopOnFirstError
509
- logger.error(`Critical: Failed to fetch incomplete runs from server`);
510
- executionError = err;
511
- } else if (error instanceof ChannelError && signal.aborted) {
512
- // Channel closed due to intentional abort - wrap in semantic error
513
- executionError = new TaskAbortedError(
514
- "Task execution stopped due to error in concurrent worker",
515
- err
500
+ // Wait for the producer AND every worker to settle before continuing.
501
+ // Using allSettled (rather than Promise.all) is important: on the first
502
+ // worker error, Promise.all rejects immediately while the remaining
503
+ // workers keep running detached, logging and hitting the API after this
504
+ // function has already returned/thrown. Draining all tasks guarantees no
505
+ // background work outlives the call (and avoids teardown races in tests
506
+ // where late console output is flushed after the test completes).
507
+ const settled = await Promise.allSettled([producerTask, ...workerTasks]);
508
+ const rejections = settled
509
+ .filter(
510
+ (result): result is PromiseRejectedResult =>
511
+ result.status === "rejected"
512
+ )
513
+ .map((result) => result.reason);
514
+
515
+ if (rejections.length > 0) {
516
+ // Classify and handle errors based on their nature. When multiple tasks
517
+ // reject, prefer the most meaningful error over incidental fallout
518
+ // (e.g. a ChannelError raised in a blocked worker when the channel
519
+ // closes on abort).
520
+ const fetchError = rejections.find(
521
+ (reason) => reason instanceof TaskFetchError
516
522
  );
517
- } else if (stopOnFirstError) {
518
- // Worker error in stopOnFirstError mode - already logged by worker
519
- executionError = err;
520
- } else {
521
- // Unexpected error (not from worker, not from producer fetch)
522
- // This could be a bug in our code or infrastructure failure
523
- logger.error(`Unexpected error during task execution: ${err.message}`);
524
- executionError = err;
523
+ const taskError = rejections.find(
524
+ (reason) =>
525
+ reason instanceof Error &&
526
+ !(reason instanceof TaskFetchError) &&
527
+ !(reason instanceof ChannelError)
528
+ );
529
+ const channelError = rejections.find(
530
+ (reason) => reason instanceof ChannelError
531
+ );
532
+
533
+ if (fetchError) {
534
+ // Producer failed - this is ALWAYS critical regardless of stopOnFirstError
535
+ logger.error(`Critical: Failed to fetch incomplete runs from server`);
536
+ executionError = fetchError;
537
+ } else if (taskError) {
538
+ // Worker error in stopOnFirstError mode - already logged by worker
539
+ executionError = taskError;
540
+ } else if (channelError && signal.aborted) {
541
+ // Channel closed due to intentional abort - wrap in semantic error
542
+ executionError = new TaskAbortedError(
543
+ "Task execution stopped due to error in concurrent worker",
544
+ channelError
545
+ );
546
+ } else {
547
+ // Unexpected error (not from worker, not from producer fetch)
548
+ // This could be a bug in our code or infrastructure failure
549
+ const reason = rejections[0];
550
+ const err =
551
+ reason instanceof Error ? reason : new Error(String(reason));
552
+ logger.error(
553
+ `Unexpected error during task execution: ${err.message}`
554
+ );
555
+ executionError = err;
556
+ }
525
557
  }
526
558
  } finally {
527
559
  // Ensure channel is closed even if there are unexpected errors
@@ -3,6 +3,7 @@ import { createClient } from "../client";
3
3
  import {
4
4
  GET_SPANS_BY_ATTRIBUTE,
5
5
  GET_SPANS_FILTERS,
6
+ GET_SPANS_SPAN_IDS,
6
7
  GET_SPANS_TRACE_IDS,
7
8
  } from "../constants/serverRequirements";
8
9
  import type { ClientFn } from "../types/core";
@@ -59,6 +60,8 @@ export interface GetSpansParams extends ClientFn {
59
60
  limit?: number;
60
61
  /** Filter spans by one or more trace IDs */
61
62
  traceIds?: string[] | null;
63
+ /** Filter spans by one or more span IDs */
64
+ spanIds?: string[] | null;
62
65
  /** Filter by parent span ID. Use `null` or the string `"null"` to get root spans only. */
63
66
  parentId?: string | null;
64
67
  /** Filter by span name(s) */
@@ -96,6 +99,7 @@ export type GetSpansResult = {
96
99
  * @returns A paginated response containing spans and optional next cursor
97
100
  *
98
101
  * @requires Phoenix server >= 13.9.0 when filtering by `traceIds`
102
+ * @requires Phoenix server >= 19.6.0 when filtering by `spanIds`
99
103
  * @requires Phoenix server >= 14.9.0 when filtering by `attributes`
100
104
  *
101
105
  * @example
@@ -125,6 +129,13 @@ export type GetSpansResult = {
125
129
  * traceIds: ["trace-abc-123", "trace-def-456"],
126
130
  * });
127
131
  *
132
+ * // Get specific spans by span ID (requires Phoenix server >= 19.6.0)
133
+ * const result = await getSpans({
134
+ * client,
135
+ * project: { projectName: "my-project" },
136
+ * spanIds: ["span-abc-123", "span-def-456"],
137
+ * });
138
+ *
128
139
  * // Paginate through results
129
140
  * let cursor: string | undefined;
130
141
  * do {
@@ -152,6 +163,7 @@ export async function getSpans({
152
163
  startTime,
153
164
  endTime,
154
165
  traceIds,
166
+ spanIds,
155
167
  parentId,
156
168
  name,
157
169
  spanKind,
@@ -168,6 +180,9 @@ export async function getSpans({
168
180
  if (traceIds) {
169
181
  await ensureServerCapability({ client, requirement: GET_SPANS_TRACE_IDS });
170
182
  }
183
+ if (spanIds) {
184
+ await ensureServerCapability({ client, requirement: GET_SPANS_SPAN_IDS });
185
+ }
171
186
  if (name != null || spanKind != null || statusCode != null) {
172
187
  await ensureServerCapability({ client, requirement: GET_SPANS_FILTERS });
173
188
  }
@@ -200,6 +215,10 @@ export async function getSpans({
200
215
  params.trace_id = traceIds;
201
216
  }
202
217
 
218
+ if (spanIds) {
219
+ params.span_id = spanIds;
220
+ }
221
+
203
222
  if (parentId !== undefined) {
204
223
  params.parent_id = parentId === null ? "null" : parentId;
205
224
  }