@arizeai/phoenix-client 6.5.5 → 6.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/dist/esm/__generated__/api/v1.d.ts +38 -124
  2. package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
  3. package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
  4. package/dist/esm/experiments/resumeEvaluation.js +3 -2
  5. package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
  6. package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
  7. package/dist/esm/experiments/resumeExperiment.js +1 -2
  8. package/dist/esm/experiments/resumeExperiment.js.map +1 -1
  9. package/dist/esm/experiments/runExperiment.d.ts +1 -2
  10. package/dist/esm/experiments/runExperiment.d.ts.map +1 -1
  11. package/dist/esm/experiments/runExperiment.js +2 -2
  12. package/dist/esm/experiments/runExperiment.js.map +1 -1
  13. package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
  14. package/dist/esm/types/experiments.d.ts +6 -0
  15. package/dist/esm/types/experiments.d.ts.map +1 -1
  16. package/dist/esm/types/spans.d.ts +1 -1
  17. package/dist/esm/types/spans.d.ts.map +1 -1
  18. package/dist/esm/utils/formatPromptMessages.d.ts.map +1 -1
  19. package/dist/esm/utils/getPromptBySelector.d.ts.map +1 -1
  20. package/dist/src/__generated__/api/v1.d.ts +38 -124
  21. package/dist/src/__generated__/api/v1.d.ts.map +1 -1
  22. package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
  23. package/dist/src/experiments/resumeEvaluation.js +5 -4
  24. package/dist/src/experiments/resumeEvaluation.js.map +1 -1
  25. package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
  26. package/dist/src/experiments/resumeExperiment.js +3 -4
  27. package/dist/src/experiments/resumeExperiment.js.map +1 -1
  28. package/dist/src/experiments/runExperiment.d.ts +1 -2
  29. package/dist/src/experiments/runExperiment.d.ts.map +1 -1
  30. package/dist/src/experiments/runExperiment.js +13 -13
  31. package/dist/src/experiments/runExperiment.js.map +1 -1
  32. package/dist/src/types/experiments.d.ts +6 -0
  33. package/dist/src/types/experiments.d.ts.map +1 -1
  34. package/dist/src/types/spans.d.ts +1 -1
  35. package/dist/src/types/spans.d.ts.map +1 -1
  36. package/dist/src/utils/formatPromptMessages.d.ts.map +1 -1
  37. package/dist/src/utils/getPromptBySelector.d.ts.map +1 -1
  38. package/dist/tsconfig.tsbuildinfo +1 -1
  39. package/docs/experiments.mdx +112 -1
  40. package/package.json +4 -4
  41. package/src/__generated__/api/v1.ts +38 -124
  42. package/src/experiments/resumeEvaluation.ts +8 -10
  43. package/src/experiments/resumeExperiment.ts +6 -10
  44. package/src/experiments/runExperiment.ts +7 -10
  45. package/src/types/experiments.ts +6 -0
  46. package/src/types/spans.ts +1 -1
@@ -12,6 +12,8 @@ The experiments module runs tasks over dataset examples, records experiment runs
12
12
  <li><code>src/experiments/helpers/getExperimentEvaluators.ts</code> for evaluator normalization</li>
13
13
  <li><code>src/experiments/helpers/fromPhoenixLLMEvaluator.ts</code> for the phoenix-evals bridge</li>
14
14
  <li><code>src/experiments/getExperimentRuns.ts</code> for reading runs back after execution</li>
15
+ <li><code>src/types/experiments.ts</code> for <code>EvaluatorParams</code> including <code>traceId</code></li>
16
+ <li><code>src/spans/getSpans.ts</code> for fetching spans by trace ID and span kind</li>
15
17
  </ul>
16
18
  </section>
17
19
 
@@ -226,10 +228,119 @@ When an evaluator runs, it receives a normalized object with these fields:
226
228
  | `output` | The task output for that run |
227
229
  | `expected` | The dataset example's `output` object |
228
230
  | `metadata` | The dataset example's `metadata` object |
231
+ | `traceId` | The OpenTelemetry trace ID of the task run (optional, `string \| null`) |
229
232
 
230
233
  This is why the `createClassificationEvaluator()` prompt can reference `{{input.question}}` and `{{output}}`.
231
234
 
232
- For code-based evaluators created with `asExperimentEvaluator()`, those same fields are available inside `evaluate({ input, output, expected, metadata })`.
235
+ For code-based evaluators created with `asExperimentEvaluator()`, those same fields are available inside `evaluate({ input, output, expected, metadata, traceId })`.
236
+
237
+ ## Trace-Based Evaluation
238
+
239
+ Each task run captures an OpenTelemetry trace ID. Evaluators can use `traceId` to fetch the task's spans from Phoenix and evaluate the execution trajectory — for example, verifying that specific tool calls were made or inspecting intermediate steps.
240
+
241
+ This pattern works best with `evaluateExperiment()` as a separate step after `runExperiment()`, so that all task spans are ingested into Phoenix before the evaluator queries them.
242
+
243
+ If you want to trace task code with helpers like `traceTool`, install `@arizeai/phoenix-otel` alongside `@arizeai/phoenix-client`:
244
+
245
+ ```bash
246
+ npm install @arizeai/phoenix-client @arizeai/phoenix-otel
247
+ ```
248
+
249
+ ```ts
250
+ import { createClient } from "@arizeai/phoenix-client";
251
+ import { createDataset } from "@arizeai/phoenix-client/datasets";
252
+ import {
253
+ asExperimentEvaluator,
254
+ evaluateExperiment,
255
+ runExperiment,
256
+ } from "@arizeai/phoenix-client/experiments";
257
+ import { getSpans } from "@arizeai/phoenix-client/spans";
258
+ import { traceTool } from "@arizeai/phoenix-otel";
259
+
260
+ const client = createClient();
261
+
262
+ const { datasetId } = await createDataset({
263
+ client,
264
+ name: "tool-call-dataset",
265
+ description: "Questions that require tool use",
266
+ examples: [
267
+ {
268
+ input: { question: "What is the weather in San Francisco?" },
269
+ output: { expectedTool: "getWeather" },
270
+ metadata: {},
271
+ },
272
+ ],
273
+ });
274
+
275
+ // Step 1: Run the experiment with traced tool calls
276
+ const experiment = await runExperiment({
277
+ client,
278
+ dataset: { datasetId },
279
+ setGlobalTracerProvider: true,
280
+ task: async (example) => {
281
+ // traceTool wraps a function with a TOOL span
282
+ const getWeather = traceTool(
283
+ ({ location }: { location: string }) => ({
284
+ location,
285
+ temperature: 72,
286
+ condition: "sunny",
287
+ }),
288
+ { name: "getWeather" }
289
+ );
290
+
291
+ const city = (example.input.question as string).match(/in (.+)\?/)?.[1];
292
+ const result = getWeather({ location: city ?? "Unknown" });
293
+ return `The weather in ${result.location} is ${result.temperature}F.`;
294
+ },
295
+ });
296
+
297
+ const projectName = experiment.projectName!;
298
+
299
+ // Step 2: Evaluate using traceId to inspect the task's spans
300
+ const evaluated = await evaluateExperiment({
301
+ client,
302
+ experiment,
303
+ evaluators: [
304
+ asExperimentEvaluator({
305
+ name: "has-expected-tool-call",
306
+ kind: "CODE",
307
+ evaluate: async ({ traceId, expected }) => {
308
+ if (!traceId) {
309
+ return { label: "no trace", score: 0 };
310
+ }
311
+
312
+ // Fetch TOOL spans from this task's trace
313
+ const { spans: toolSpans } = await getSpans({
314
+ client,
315
+ project: { projectName },
316
+ traceIds: [traceId],
317
+ spanKind: "TOOL",
318
+ });
319
+
320
+ const expectedTool = (expected as { expectedTool?: string })
321
+ ?.expectedTool;
322
+ const toolNames = toolSpans.map((s) => s.name);
323
+ const found = toolNames.some((name) => name.includes(expectedTool!));
324
+
325
+ return {
326
+ label: found ? "tool called" : "no tool call",
327
+ score: found ? 1 : 0,
328
+ explanation: found
329
+ ? `Found: ${toolNames.join(", ")}`
330
+ : `Expected "${expectedTool}" but found none`,
331
+ };
332
+ },
333
+ }),
334
+ ],
335
+ });
336
+ ```
337
+
338
+ Key points:
339
+
340
+ - Use `setGlobalTracerProvider: true` on `runExperiment()` so that child spans from `traceTool` or other OTel instrumentation land in the same trace as the task
341
+ - Use `evaluateExperiment()` as a separate step so spans are ingested before querying
342
+ - Use `getSpans()` with `traceIds` and `spanKind` filters to fetch specific spans from the task trace
343
+ - `traceId` is `null` in dry-run mode since no real traces are recorded
233
344
 
234
345
  ## What `runExperiment()` Returns
235
346
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@arizeai/phoenix-client",
3
- "version": "6.5.5",
3
+ "version": "6.6.1",
4
4
  "description": "A client for the Phoenix API",
5
5
  "keywords": [
6
6
  "arize",
@@ -73,14 +73,14 @@
73
73
  }
74
74
  },
75
75
  "dependencies": {
76
- "@arizeai/openinference-semantic-conventions": "^1.1.0",
76
+ "@arizeai/openinference-semantic-conventions": "^2.1.7",
77
77
  "@arizeai/openinference-vercel": "^2.7.0",
78
78
  "async": "^3.2.6",
79
79
  "openapi-fetch": "^0.12.5",
80
80
  "tiny-invariant": "^1.3.3",
81
81
  "zod": "^4.0.14",
82
- "@arizeai/phoenix-otel": "0.4.3",
83
- "@arizeai/phoenix-config": "0.1.3"
82
+ "@arizeai/phoenix-config": "0.1.3",
83
+ "@arizeai/phoenix-otel": "1.0.0"
84
84
  },
85
85
  "devDependencies": {
86
86
  "@ai-sdk/openai": "^3.0.29",
@@ -595,24 +595,6 @@ export interface paths {
595
595
  patch?: never;
596
596
  trace?: never;
597
597
  };
598
- "/v1/evaluations": {
599
- parameters: {
600
- query?: never;
601
- header?: never;
602
- path?: never;
603
- cookie?: never;
604
- };
605
- /** Get span, trace, or document evaluations from a project */
606
- get: operations["getEvaluations"];
607
- put?: never;
608
- /** Add span, trace, or document evaluations */
609
- post: operations["addEvaluations"];
610
- delete?: never;
611
- options?: never;
612
- head?: never;
613
- patch?: never;
614
- trace?: never;
615
- };
616
598
  "/v1/prompts": {
617
599
  parameters: {
618
600
  query?: never;
@@ -2954,7 +2936,27 @@ export interface components {
2954
2936
  */
2955
2937
  end_time: string;
2956
2938
  };
2957
- /** Span */
2939
+ /**
2940
+ * Span
2941
+ * @example {
2942
+ * "attributes": {
2943
+ * "llm.model_name": "gpt-4",
2944
+ * "llm.token_count.completion": 50,
2945
+ * "llm.token_count.prompt": 100
2946
+ * },
2947
+ * "context": {
2948
+ * "span_id": "1a2b3c4d5e6f7a8b",
2949
+ * "trace_id": "a1b2c3d4e5f6a1b2c3d4e5f6a1b2c3d4"
2950
+ * },
2951
+ * "end_time": "2024-01-01T12:00:01Z",
2952
+ * "events": [],
2953
+ * "name": "llm_call",
2954
+ * "span_kind": "LLM",
2955
+ * "start_time": "2024-01-01T12:00:00Z",
2956
+ * "status_code": "OK",
2957
+ * "status_message": ""
2958
+ * }
2959
+ */
2958
2960
  Span: {
2959
2961
  /**
2960
2962
  * Id
@@ -3109,7 +3111,13 @@ export interface components {
3109
3111
  /** Next Cursor */
3110
3112
  next_cursor: string | null;
3111
3113
  };
3112
- /** SpanContext */
3114
+ /**
3115
+ * SpanContext
3116
+ * @example {
3117
+ * "span_id": "1a2b3c4d5e6f7a8b",
3118
+ * "trace_id": "a1b2c3d4e5f6a1b2c3d4e5f6a1b2c3d4"
3119
+ * }
3120
+ */
3113
3121
  SpanContext: {
3114
3122
  /**
3115
3123
  * Trace Id
@@ -3161,7 +3169,16 @@ export interface components {
3161
3169
  */
3162
3170
  document_position: number;
3163
3171
  };
3164
- /** SpanEvent */
3172
+ /**
3173
+ * SpanEvent
3174
+ * @example {
3175
+ * "attributes": {
3176
+ * "exception.message": "Connection refused"
3177
+ * },
3178
+ * "name": "exception",
3179
+ * "timestamp": "2024-01-01T12:00:00Z"
3180
+ * }
3181
+ */
3165
3182
  SpanEvent: {
3166
3183
  /**
3167
3184
  * Name
@@ -5465,109 +5482,6 @@ export interface operations {
5465
5482
  };
5466
5483
  };
5467
5484
  };
5468
- getEvaluations: {
5469
- parameters: {
5470
- query?: {
5471
- /** @description The name of the project to get evaluations from (if omitted, evaluations will be drawn from the `default` project) */
5472
- project_name?: string | null;
5473
- };
5474
- header?: never;
5475
- path?: never;
5476
- cookie?: never;
5477
- };
5478
- requestBody?: never;
5479
- responses: {
5480
- /** @description Successful Response */
5481
- 200: {
5482
- headers: {
5483
- [name: string]: unknown;
5484
- };
5485
- content: {
5486
- "application/json": unknown;
5487
- };
5488
- };
5489
- /** @description Forbidden */
5490
- 403: {
5491
- headers: {
5492
- [name: string]: unknown;
5493
- };
5494
- content: {
5495
- "text/plain": string;
5496
- };
5497
- };
5498
- /** @description Not Found */
5499
- 404: {
5500
- headers: {
5501
- [name: string]: unknown;
5502
- };
5503
- content: {
5504
- "text/plain": string;
5505
- };
5506
- };
5507
- /** @description Validation Error */
5508
- 422: {
5509
- headers: {
5510
- [name: string]: unknown;
5511
- };
5512
- content: {
5513
- "application/json": components["schemas"]["HTTPValidationError"];
5514
- };
5515
- };
5516
- };
5517
- };
5518
- addEvaluations: {
5519
- parameters: {
5520
- query?: never;
5521
- header?: {
5522
- "content-type"?: string | null;
5523
- "content-encoding"?: string | null;
5524
- };
5525
- path?: never;
5526
- cookie?: never;
5527
- };
5528
- requestBody: {
5529
- content: {
5530
- "application/x-protobuf": string;
5531
- "application/x-pandas-arrow": string;
5532
- };
5533
- };
5534
- responses: {
5535
- /** @description Successful Response */
5536
- 204: {
5537
- headers: {
5538
- [name: string]: unknown;
5539
- };
5540
- content?: never;
5541
- };
5542
- /** @description Forbidden */
5543
- 403: {
5544
- headers: {
5545
- [name: string]: unknown;
5546
- };
5547
- content: {
5548
- "text/plain": string;
5549
- };
5550
- };
5551
- /** @description Unsupported content type, only gzipped protobuf and pandas-arrow are supported */
5552
- 415: {
5553
- headers: {
5554
- [name: string]: unknown;
5555
- };
5556
- content: {
5557
- "text/plain": string;
5558
- };
5559
- };
5560
- /** @description Unprocessable Entity */
5561
- 422: {
5562
- headers: {
5563
- [name: string]: unknown;
5564
- };
5565
- content: {
5566
- "text/plain": string;
5567
- };
5568
- };
5569
- };
5570
- };
5571
5485
  getPrompts: {
5572
5486
  parameters: {
5573
5487
  query?: {
@@ -1,18 +1,14 @@
1
- import {
2
- MimeType,
3
- OpenInferenceSpanKind,
4
- SemanticConventions,
5
- } from "@arizeai/openinference-semantic-conventions";
6
- import type {
7
- GlobalTracerProviderRegistration,
8
- NodeTracerProvider,
9
- Tracer,
10
- } from "@arizeai/phoenix-otel";
11
1
  import {
12
2
  attachGlobalTracerProvider,
13
3
  type DiagLogLevel,
4
+ type GlobalTracerProviderRegistration,
5
+ MimeType,
6
+ type NodeTracerProvider,
14
7
  objectAsAttributes,
8
+ OpenInferenceSpanKind,
15
9
  register,
10
+ SemanticConventions,
11
+ type Tracer,
16
12
  SpanStatusCode,
17
13
  } from "@arizeai/phoenix-otel";
18
14
  import invariant from "tiny-invariant";
@@ -692,6 +688,7 @@ async function runSingleEvaluation({
692
688
  output: taskOutput,
693
689
  expected: expectedOutput,
694
690
  metadata: datasetExample.metadata,
691
+ traceId: experimentRun.traceId,
695
692
  })
696
693
  );
697
694
  results = Array.isArray(result) ? result : [result];
@@ -746,6 +743,7 @@ async function runSingleEvaluation({
746
743
  output: taskOutput,
747
744
  expected: expectedOutput,
748
745
  metadata: datasetExample.metadata,
746
+ traceId: experimentRun.traceId,
749
747
  })
750
748
  );
751
749
 
@@ -1,18 +1,14 @@
1
- import {
2
- MimeType,
3
- OpenInferenceSpanKind,
4
- SemanticConventions,
5
- } from "@arizeai/openinference-semantic-conventions";
6
- import type {
7
- GlobalTracerProviderRegistration,
8
- NodeTracerProvider,
9
- Tracer,
10
- } from "@arizeai/phoenix-otel";
11
1
  import {
12
2
  attachGlobalTracerProvider,
13
3
  type DiagLogLevel,
4
+ type GlobalTracerProviderRegistration,
5
+ MimeType,
6
+ type NodeTracerProvider,
14
7
  objectAsAttributes,
8
+ OpenInferenceSpanKind,
15
9
  register,
10
+ SemanticConventions,
11
+ type Tracer,
16
12
  SpanStatusCode,
17
13
  } from "@arizeai/phoenix-otel";
18
14
  import invariant from "tiny-invariant";
@@ -1,18 +1,14 @@
1
- import {
2
- MimeType,
3
- OpenInferenceSpanKind,
4
- SemanticConventions,
5
- } from "@arizeai/openinference-semantic-conventions";
6
- import type {
7
- GlobalTracerProviderRegistration,
8
- NodeTracerProvider,
9
- Tracer,
10
- } from "@arizeai/phoenix-otel";
11
1
  import {
12
2
  attachGlobalTracerProvider,
13
3
  createNoOpProvider,
14
4
  type DiagLogLevel,
5
+ type GlobalTracerProviderRegistration,
6
+ MimeType,
7
+ type NodeTracerProvider,
15
8
  objectAsAttributes,
9
+ OpenInferenceSpanKind,
10
+ SemanticConventions,
11
+ type Tracer,
16
12
  register,
17
13
  SpanStatusCode,
18
14
  } from "@arizeai/phoenix-otel";
@@ -853,6 +849,7 @@ async function runEvaluator({
853
849
  output: run.output ?? null,
854
850
  expected: example.output,
855
851
  metadata: example?.metadata,
852
+ traceId: run.traceId,
856
853
  });
857
854
  thisEval.result = result;
858
855
  } catch (error) {
@@ -131,6 +131,12 @@ export type EvaluatorParams<TaskOutputType = TaskOutput> = {
131
131
  * Metadata associated with the Dataset Example
132
132
  */
133
133
  metadata?: Example["metadata"];
134
+ /**
135
+ * The trace ID of the task run, if available.
136
+ * Can be used to fetch and analyze the task's trace
137
+ * (e.g., for trajectory evaluation or action verification).
138
+ */
139
+ traceId?: string | null;
134
140
  };
135
141
 
136
142
  export type Evaluator = {
@@ -1,4 +1,4 @@
1
- import type { OpenInferenceSpanKind } from "@arizeai/openinference-semantic-conventions";
1
+ import type { OpenInferenceSpanKind } from "@arizeai/phoenix-otel";
2
2
 
3
3
  /**
4
4
  * Status codes for spans.