@arizeai/phoenix-client 6.5.5 → 6.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/esm/__generated__/api/v1.d.ts +38 -124
- package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
- package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
- package/dist/esm/experiments/resumeEvaluation.js +3 -2
- package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
- package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
- package/dist/esm/experiments/resumeExperiment.js +1 -2
- package/dist/esm/experiments/resumeExperiment.js.map +1 -1
- package/dist/esm/experiments/runExperiment.d.ts +1 -2
- package/dist/esm/experiments/runExperiment.d.ts.map +1 -1
- package/dist/esm/experiments/runExperiment.js +2 -2
- package/dist/esm/experiments/runExperiment.js.map +1 -1
- package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
- package/dist/esm/types/experiments.d.ts +6 -0
- package/dist/esm/types/experiments.d.ts.map +1 -1
- package/dist/esm/types/spans.d.ts +1 -1
- package/dist/esm/types/spans.d.ts.map +1 -1
- package/dist/esm/utils/formatPromptMessages.d.ts.map +1 -1
- package/dist/esm/utils/getPromptBySelector.d.ts.map +1 -1
- package/dist/src/__generated__/api/v1.d.ts +38 -124
- package/dist/src/__generated__/api/v1.d.ts.map +1 -1
- package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
- package/dist/src/experiments/resumeEvaluation.js +5 -4
- package/dist/src/experiments/resumeEvaluation.js.map +1 -1
- package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
- package/dist/src/experiments/resumeExperiment.js +3 -4
- package/dist/src/experiments/resumeExperiment.js.map +1 -1
- package/dist/src/experiments/runExperiment.d.ts +1 -2
- package/dist/src/experiments/runExperiment.d.ts.map +1 -1
- package/dist/src/experiments/runExperiment.js +13 -13
- package/dist/src/experiments/runExperiment.js.map +1 -1
- package/dist/src/types/experiments.d.ts +6 -0
- package/dist/src/types/experiments.d.ts.map +1 -1
- package/dist/src/types/spans.d.ts +1 -1
- package/dist/src/types/spans.d.ts.map +1 -1
- package/dist/src/utils/formatPromptMessages.d.ts.map +1 -1
- package/dist/src/utils/getPromptBySelector.d.ts.map +1 -1
- package/dist/tsconfig.tsbuildinfo +1 -1
- package/docs/experiments.mdx +112 -1
- package/package.json +4 -4
- package/src/__generated__/api/v1.ts +38 -124
- package/src/experiments/resumeEvaluation.ts +8 -10
- package/src/experiments/resumeExperiment.ts +6 -10
- package/src/experiments/runExperiment.ts +7 -10
- package/src/types/experiments.ts +6 -0
- package/src/types/spans.ts +1 -1
package/docs/experiments.mdx
CHANGED
|
@@ -12,6 +12,8 @@ The experiments module runs tasks over dataset examples, records experiment runs
|
|
|
12
12
|
<li><code>src/experiments/helpers/getExperimentEvaluators.ts</code> for evaluator normalization</li>
|
|
13
13
|
<li><code>src/experiments/helpers/fromPhoenixLLMEvaluator.ts</code> for the phoenix-evals bridge</li>
|
|
14
14
|
<li><code>src/experiments/getExperimentRuns.ts</code> for reading runs back after execution</li>
|
|
15
|
+
<li><code>src/types/experiments.ts</code> for <code>EvaluatorParams</code> including <code>traceId</code></li>
|
|
16
|
+
<li><code>src/spans/getSpans.ts</code> for fetching spans by trace ID and span kind</li>
|
|
15
17
|
</ul>
|
|
16
18
|
</section>
|
|
17
19
|
|
|
@@ -226,10 +228,119 @@ When an evaluator runs, it receives a normalized object with these fields:
|
|
|
226
228
|
| `output` | The task output for that run |
|
|
227
229
|
| `expected` | The dataset example's `output` object |
|
|
228
230
|
| `metadata` | The dataset example's `metadata` object |
|
|
231
|
+
| `traceId` | The OpenTelemetry trace ID of the task run (optional, `string \| null`) |
|
|
229
232
|
|
|
230
233
|
This is why the `createClassificationEvaluator()` prompt can reference `{{input.question}}` and `{{output}}`.
|
|
231
234
|
|
|
232
|
-
For code-based evaluators created with `asExperimentEvaluator()`, those same fields are available inside `evaluate({ input, output, expected, metadata })`.
|
|
235
|
+
For code-based evaluators created with `asExperimentEvaluator()`, those same fields are available inside `evaluate({ input, output, expected, metadata, traceId })`.
|
|
236
|
+
|
|
237
|
+
## Trace-Based Evaluation
|
|
238
|
+
|
|
239
|
+
Each task run captures an OpenTelemetry trace ID. Evaluators can use `traceId` to fetch the task's spans from Phoenix and evaluate the execution trajectory — for example, verifying that specific tool calls were made or inspecting intermediate steps.
|
|
240
|
+
|
|
241
|
+
This pattern works best with `evaluateExperiment()` as a separate step after `runExperiment()`, so that all task spans are ingested into Phoenix before the evaluator queries them.
|
|
242
|
+
|
|
243
|
+
If you want to trace task code with helpers like `traceTool`, install `@arizeai/phoenix-otel` alongside `@arizeai/phoenix-client`:
|
|
244
|
+
|
|
245
|
+
```bash
|
|
246
|
+
npm install @arizeai/phoenix-client @arizeai/phoenix-otel
|
|
247
|
+
```
|
|
248
|
+
|
|
249
|
+
```ts
|
|
250
|
+
import { createClient } from "@arizeai/phoenix-client";
|
|
251
|
+
import { createDataset } from "@arizeai/phoenix-client/datasets";
|
|
252
|
+
import {
|
|
253
|
+
asExperimentEvaluator,
|
|
254
|
+
evaluateExperiment,
|
|
255
|
+
runExperiment,
|
|
256
|
+
} from "@arizeai/phoenix-client/experiments";
|
|
257
|
+
import { getSpans } from "@arizeai/phoenix-client/spans";
|
|
258
|
+
import { traceTool } from "@arizeai/phoenix-otel";
|
|
259
|
+
|
|
260
|
+
const client = createClient();
|
|
261
|
+
|
|
262
|
+
const { datasetId } = await createDataset({
|
|
263
|
+
client,
|
|
264
|
+
name: "tool-call-dataset",
|
|
265
|
+
description: "Questions that require tool use",
|
|
266
|
+
examples: [
|
|
267
|
+
{
|
|
268
|
+
input: { question: "What is the weather in San Francisco?" },
|
|
269
|
+
output: { expectedTool: "getWeather" },
|
|
270
|
+
metadata: {},
|
|
271
|
+
},
|
|
272
|
+
],
|
|
273
|
+
});
|
|
274
|
+
|
|
275
|
+
// Step 1: Run the experiment with traced tool calls
|
|
276
|
+
const experiment = await runExperiment({
|
|
277
|
+
client,
|
|
278
|
+
dataset: { datasetId },
|
|
279
|
+
setGlobalTracerProvider: true,
|
|
280
|
+
task: async (example) => {
|
|
281
|
+
// traceTool wraps a function with a TOOL span
|
|
282
|
+
const getWeather = traceTool(
|
|
283
|
+
({ location }: { location: string }) => ({
|
|
284
|
+
location,
|
|
285
|
+
temperature: 72,
|
|
286
|
+
condition: "sunny",
|
|
287
|
+
}),
|
|
288
|
+
{ name: "getWeather" }
|
|
289
|
+
);
|
|
290
|
+
|
|
291
|
+
const city = (example.input.question as string).match(/in (.+)\?/)?.[1];
|
|
292
|
+
const result = getWeather({ location: city ?? "Unknown" });
|
|
293
|
+
return `The weather in ${result.location} is ${result.temperature}F.`;
|
|
294
|
+
},
|
|
295
|
+
});
|
|
296
|
+
|
|
297
|
+
const projectName = experiment.projectName!;
|
|
298
|
+
|
|
299
|
+
// Step 2: Evaluate using traceId to inspect the task's spans
|
|
300
|
+
const evaluated = await evaluateExperiment({
|
|
301
|
+
client,
|
|
302
|
+
experiment,
|
|
303
|
+
evaluators: [
|
|
304
|
+
asExperimentEvaluator({
|
|
305
|
+
name: "has-expected-tool-call",
|
|
306
|
+
kind: "CODE",
|
|
307
|
+
evaluate: async ({ traceId, expected }) => {
|
|
308
|
+
if (!traceId) {
|
|
309
|
+
return { label: "no trace", score: 0 };
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
// Fetch TOOL spans from this task's trace
|
|
313
|
+
const { spans: toolSpans } = await getSpans({
|
|
314
|
+
client,
|
|
315
|
+
project: { projectName },
|
|
316
|
+
traceIds: [traceId],
|
|
317
|
+
spanKind: "TOOL",
|
|
318
|
+
});
|
|
319
|
+
|
|
320
|
+
const expectedTool = (expected as { expectedTool?: string })
|
|
321
|
+
?.expectedTool;
|
|
322
|
+
const toolNames = toolSpans.map((s) => s.name);
|
|
323
|
+
const found = toolNames.some((name) => name.includes(expectedTool!));
|
|
324
|
+
|
|
325
|
+
return {
|
|
326
|
+
label: found ? "tool called" : "no tool call",
|
|
327
|
+
score: found ? 1 : 0,
|
|
328
|
+
explanation: found
|
|
329
|
+
? `Found: ${toolNames.join(", ")}`
|
|
330
|
+
: `Expected "${expectedTool}" but found none`,
|
|
331
|
+
};
|
|
332
|
+
},
|
|
333
|
+
}),
|
|
334
|
+
],
|
|
335
|
+
});
|
|
336
|
+
```
|
|
337
|
+
|
|
338
|
+
Key points:
|
|
339
|
+
|
|
340
|
+
- Use `setGlobalTracerProvider: true` on `runExperiment()` so that child spans from `traceTool` or other OTel instrumentation land in the same trace as the task
|
|
341
|
+
- Use `evaluateExperiment()` as a separate step so spans are ingested before querying
|
|
342
|
+
- Use `getSpans()` with `traceIds` and `spanKind` filters to fetch specific spans from the task trace
|
|
343
|
+
- `traceId` is `null` in dry-run mode since no real traces are recorded
|
|
233
344
|
|
|
234
345
|
## What `runExperiment()` Returns
|
|
235
346
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@arizeai/phoenix-client",
|
|
3
|
-
"version": "6.
|
|
3
|
+
"version": "6.6.1",
|
|
4
4
|
"description": "A client for the Phoenix API",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"arize",
|
|
@@ -73,14 +73,14 @@
|
|
|
73
73
|
}
|
|
74
74
|
},
|
|
75
75
|
"dependencies": {
|
|
76
|
-
"@arizeai/openinference-semantic-conventions": "^
|
|
76
|
+
"@arizeai/openinference-semantic-conventions": "^2.1.7",
|
|
77
77
|
"@arizeai/openinference-vercel": "^2.7.0",
|
|
78
78
|
"async": "^3.2.6",
|
|
79
79
|
"openapi-fetch": "^0.12.5",
|
|
80
80
|
"tiny-invariant": "^1.3.3",
|
|
81
81
|
"zod": "^4.0.14",
|
|
82
|
-
"@arizeai/phoenix-
|
|
83
|
-
"@arizeai/phoenix-
|
|
82
|
+
"@arizeai/phoenix-config": "0.1.3",
|
|
83
|
+
"@arizeai/phoenix-otel": "1.0.0"
|
|
84
84
|
},
|
|
85
85
|
"devDependencies": {
|
|
86
86
|
"@ai-sdk/openai": "^3.0.29",
|
|
@@ -595,24 +595,6 @@ export interface paths {
|
|
|
595
595
|
patch?: never;
|
|
596
596
|
trace?: never;
|
|
597
597
|
};
|
|
598
|
-
"/v1/evaluations": {
|
|
599
|
-
parameters: {
|
|
600
|
-
query?: never;
|
|
601
|
-
header?: never;
|
|
602
|
-
path?: never;
|
|
603
|
-
cookie?: never;
|
|
604
|
-
};
|
|
605
|
-
/** Get span, trace, or document evaluations from a project */
|
|
606
|
-
get: operations["getEvaluations"];
|
|
607
|
-
put?: never;
|
|
608
|
-
/** Add span, trace, or document evaluations */
|
|
609
|
-
post: operations["addEvaluations"];
|
|
610
|
-
delete?: never;
|
|
611
|
-
options?: never;
|
|
612
|
-
head?: never;
|
|
613
|
-
patch?: never;
|
|
614
|
-
trace?: never;
|
|
615
|
-
};
|
|
616
598
|
"/v1/prompts": {
|
|
617
599
|
parameters: {
|
|
618
600
|
query?: never;
|
|
@@ -2954,7 +2936,27 @@ export interface components {
|
|
|
2954
2936
|
*/
|
|
2955
2937
|
end_time: string;
|
|
2956
2938
|
};
|
|
2957
|
-
/**
|
|
2939
|
+
/**
|
|
2940
|
+
* Span
|
|
2941
|
+
* @example {
|
|
2942
|
+
* "attributes": {
|
|
2943
|
+
* "llm.model_name": "gpt-4",
|
|
2944
|
+
* "llm.token_count.completion": 50,
|
|
2945
|
+
* "llm.token_count.prompt": 100
|
|
2946
|
+
* },
|
|
2947
|
+
* "context": {
|
|
2948
|
+
* "span_id": "1a2b3c4d5e6f7a8b",
|
|
2949
|
+
* "trace_id": "a1b2c3d4e5f6a1b2c3d4e5f6a1b2c3d4"
|
|
2950
|
+
* },
|
|
2951
|
+
* "end_time": "2024-01-01T12:00:01Z",
|
|
2952
|
+
* "events": [],
|
|
2953
|
+
* "name": "llm_call",
|
|
2954
|
+
* "span_kind": "LLM",
|
|
2955
|
+
* "start_time": "2024-01-01T12:00:00Z",
|
|
2956
|
+
* "status_code": "OK",
|
|
2957
|
+
* "status_message": ""
|
|
2958
|
+
* }
|
|
2959
|
+
*/
|
|
2958
2960
|
Span: {
|
|
2959
2961
|
/**
|
|
2960
2962
|
* Id
|
|
@@ -3109,7 +3111,13 @@ export interface components {
|
|
|
3109
3111
|
/** Next Cursor */
|
|
3110
3112
|
next_cursor: string | null;
|
|
3111
3113
|
};
|
|
3112
|
-
/**
|
|
3114
|
+
/**
|
|
3115
|
+
* SpanContext
|
|
3116
|
+
* @example {
|
|
3117
|
+
* "span_id": "1a2b3c4d5e6f7a8b",
|
|
3118
|
+
* "trace_id": "a1b2c3d4e5f6a1b2c3d4e5f6a1b2c3d4"
|
|
3119
|
+
* }
|
|
3120
|
+
*/
|
|
3113
3121
|
SpanContext: {
|
|
3114
3122
|
/**
|
|
3115
3123
|
* Trace Id
|
|
@@ -3161,7 +3169,16 @@ export interface components {
|
|
|
3161
3169
|
*/
|
|
3162
3170
|
document_position: number;
|
|
3163
3171
|
};
|
|
3164
|
-
/**
|
|
3172
|
+
/**
|
|
3173
|
+
* SpanEvent
|
|
3174
|
+
* @example {
|
|
3175
|
+
* "attributes": {
|
|
3176
|
+
* "exception.message": "Connection refused"
|
|
3177
|
+
* },
|
|
3178
|
+
* "name": "exception",
|
|
3179
|
+
* "timestamp": "2024-01-01T12:00:00Z"
|
|
3180
|
+
* }
|
|
3181
|
+
*/
|
|
3165
3182
|
SpanEvent: {
|
|
3166
3183
|
/**
|
|
3167
3184
|
* Name
|
|
@@ -5465,109 +5482,6 @@ export interface operations {
|
|
|
5465
5482
|
};
|
|
5466
5483
|
};
|
|
5467
5484
|
};
|
|
5468
|
-
getEvaluations: {
|
|
5469
|
-
parameters: {
|
|
5470
|
-
query?: {
|
|
5471
|
-
/** @description The name of the project to get evaluations from (if omitted, evaluations will be drawn from the `default` project) */
|
|
5472
|
-
project_name?: string | null;
|
|
5473
|
-
};
|
|
5474
|
-
header?: never;
|
|
5475
|
-
path?: never;
|
|
5476
|
-
cookie?: never;
|
|
5477
|
-
};
|
|
5478
|
-
requestBody?: never;
|
|
5479
|
-
responses: {
|
|
5480
|
-
/** @description Successful Response */
|
|
5481
|
-
200: {
|
|
5482
|
-
headers: {
|
|
5483
|
-
[name: string]: unknown;
|
|
5484
|
-
};
|
|
5485
|
-
content: {
|
|
5486
|
-
"application/json": unknown;
|
|
5487
|
-
};
|
|
5488
|
-
};
|
|
5489
|
-
/** @description Forbidden */
|
|
5490
|
-
403: {
|
|
5491
|
-
headers: {
|
|
5492
|
-
[name: string]: unknown;
|
|
5493
|
-
};
|
|
5494
|
-
content: {
|
|
5495
|
-
"text/plain": string;
|
|
5496
|
-
};
|
|
5497
|
-
};
|
|
5498
|
-
/** @description Not Found */
|
|
5499
|
-
404: {
|
|
5500
|
-
headers: {
|
|
5501
|
-
[name: string]: unknown;
|
|
5502
|
-
};
|
|
5503
|
-
content: {
|
|
5504
|
-
"text/plain": string;
|
|
5505
|
-
};
|
|
5506
|
-
};
|
|
5507
|
-
/** @description Validation Error */
|
|
5508
|
-
422: {
|
|
5509
|
-
headers: {
|
|
5510
|
-
[name: string]: unknown;
|
|
5511
|
-
};
|
|
5512
|
-
content: {
|
|
5513
|
-
"application/json": components["schemas"]["HTTPValidationError"];
|
|
5514
|
-
};
|
|
5515
|
-
};
|
|
5516
|
-
};
|
|
5517
|
-
};
|
|
5518
|
-
addEvaluations: {
|
|
5519
|
-
parameters: {
|
|
5520
|
-
query?: never;
|
|
5521
|
-
header?: {
|
|
5522
|
-
"content-type"?: string | null;
|
|
5523
|
-
"content-encoding"?: string | null;
|
|
5524
|
-
};
|
|
5525
|
-
path?: never;
|
|
5526
|
-
cookie?: never;
|
|
5527
|
-
};
|
|
5528
|
-
requestBody: {
|
|
5529
|
-
content: {
|
|
5530
|
-
"application/x-protobuf": string;
|
|
5531
|
-
"application/x-pandas-arrow": string;
|
|
5532
|
-
};
|
|
5533
|
-
};
|
|
5534
|
-
responses: {
|
|
5535
|
-
/** @description Successful Response */
|
|
5536
|
-
204: {
|
|
5537
|
-
headers: {
|
|
5538
|
-
[name: string]: unknown;
|
|
5539
|
-
};
|
|
5540
|
-
content?: never;
|
|
5541
|
-
};
|
|
5542
|
-
/** @description Forbidden */
|
|
5543
|
-
403: {
|
|
5544
|
-
headers: {
|
|
5545
|
-
[name: string]: unknown;
|
|
5546
|
-
};
|
|
5547
|
-
content: {
|
|
5548
|
-
"text/plain": string;
|
|
5549
|
-
};
|
|
5550
|
-
};
|
|
5551
|
-
/** @description Unsupported content type, only gzipped protobuf and pandas-arrow are supported */
|
|
5552
|
-
415: {
|
|
5553
|
-
headers: {
|
|
5554
|
-
[name: string]: unknown;
|
|
5555
|
-
};
|
|
5556
|
-
content: {
|
|
5557
|
-
"text/plain": string;
|
|
5558
|
-
};
|
|
5559
|
-
};
|
|
5560
|
-
/** @description Unprocessable Entity */
|
|
5561
|
-
422: {
|
|
5562
|
-
headers: {
|
|
5563
|
-
[name: string]: unknown;
|
|
5564
|
-
};
|
|
5565
|
-
content: {
|
|
5566
|
-
"text/plain": string;
|
|
5567
|
-
};
|
|
5568
|
-
};
|
|
5569
|
-
};
|
|
5570
|
-
};
|
|
5571
5485
|
getPrompts: {
|
|
5572
5486
|
parameters: {
|
|
5573
5487
|
query?: {
|
|
@@ -1,18 +1,14 @@
|
|
|
1
|
-
import {
|
|
2
|
-
MimeType,
|
|
3
|
-
OpenInferenceSpanKind,
|
|
4
|
-
SemanticConventions,
|
|
5
|
-
} from "@arizeai/openinference-semantic-conventions";
|
|
6
|
-
import type {
|
|
7
|
-
GlobalTracerProviderRegistration,
|
|
8
|
-
NodeTracerProvider,
|
|
9
|
-
Tracer,
|
|
10
|
-
} from "@arizeai/phoenix-otel";
|
|
11
1
|
import {
|
|
12
2
|
attachGlobalTracerProvider,
|
|
13
3
|
type DiagLogLevel,
|
|
4
|
+
type GlobalTracerProviderRegistration,
|
|
5
|
+
MimeType,
|
|
6
|
+
type NodeTracerProvider,
|
|
14
7
|
objectAsAttributes,
|
|
8
|
+
OpenInferenceSpanKind,
|
|
15
9
|
register,
|
|
10
|
+
SemanticConventions,
|
|
11
|
+
type Tracer,
|
|
16
12
|
SpanStatusCode,
|
|
17
13
|
} from "@arizeai/phoenix-otel";
|
|
18
14
|
import invariant from "tiny-invariant";
|
|
@@ -692,6 +688,7 @@ async function runSingleEvaluation({
|
|
|
692
688
|
output: taskOutput,
|
|
693
689
|
expected: expectedOutput,
|
|
694
690
|
metadata: datasetExample.metadata,
|
|
691
|
+
traceId: experimentRun.traceId,
|
|
695
692
|
})
|
|
696
693
|
);
|
|
697
694
|
results = Array.isArray(result) ? result : [result];
|
|
@@ -746,6 +743,7 @@ async function runSingleEvaluation({
|
|
|
746
743
|
output: taskOutput,
|
|
747
744
|
expected: expectedOutput,
|
|
748
745
|
metadata: datasetExample.metadata,
|
|
746
|
+
traceId: experimentRun.traceId,
|
|
749
747
|
})
|
|
750
748
|
);
|
|
751
749
|
|
|
@@ -1,18 +1,14 @@
|
|
|
1
|
-
import {
|
|
2
|
-
MimeType,
|
|
3
|
-
OpenInferenceSpanKind,
|
|
4
|
-
SemanticConventions,
|
|
5
|
-
} from "@arizeai/openinference-semantic-conventions";
|
|
6
|
-
import type {
|
|
7
|
-
GlobalTracerProviderRegistration,
|
|
8
|
-
NodeTracerProvider,
|
|
9
|
-
Tracer,
|
|
10
|
-
} from "@arizeai/phoenix-otel";
|
|
11
1
|
import {
|
|
12
2
|
attachGlobalTracerProvider,
|
|
13
3
|
type DiagLogLevel,
|
|
4
|
+
type GlobalTracerProviderRegistration,
|
|
5
|
+
MimeType,
|
|
6
|
+
type NodeTracerProvider,
|
|
14
7
|
objectAsAttributes,
|
|
8
|
+
OpenInferenceSpanKind,
|
|
15
9
|
register,
|
|
10
|
+
SemanticConventions,
|
|
11
|
+
type Tracer,
|
|
16
12
|
SpanStatusCode,
|
|
17
13
|
} from "@arizeai/phoenix-otel";
|
|
18
14
|
import invariant from "tiny-invariant";
|
|
@@ -1,18 +1,14 @@
|
|
|
1
|
-
import {
|
|
2
|
-
MimeType,
|
|
3
|
-
OpenInferenceSpanKind,
|
|
4
|
-
SemanticConventions,
|
|
5
|
-
} from "@arizeai/openinference-semantic-conventions";
|
|
6
|
-
import type {
|
|
7
|
-
GlobalTracerProviderRegistration,
|
|
8
|
-
NodeTracerProvider,
|
|
9
|
-
Tracer,
|
|
10
|
-
} from "@arizeai/phoenix-otel";
|
|
11
1
|
import {
|
|
12
2
|
attachGlobalTracerProvider,
|
|
13
3
|
createNoOpProvider,
|
|
14
4
|
type DiagLogLevel,
|
|
5
|
+
type GlobalTracerProviderRegistration,
|
|
6
|
+
MimeType,
|
|
7
|
+
type NodeTracerProvider,
|
|
15
8
|
objectAsAttributes,
|
|
9
|
+
OpenInferenceSpanKind,
|
|
10
|
+
SemanticConventions,
|
|
11
|
+
type Tracer,
|
|
16
12
|
register,
|
|
17
13
|
SpanStatusCode,
|
|
18
14
|
} from "@arizeai/phoenix-otel";
|
|
@@ -853,6 +849,7 @@ async function runEvaluator({
|
|
|
853
849
|
output: run.output ?? null,
|
|
854
850
|
expected: example.output,
|
|
855
851
|
metadata: example?.metadata,
|
|
852
|
+
traceId: run.traceId,
|
|
856
853
|
});
|
|
857
854
|
thisEval.result = result;
|
|
858
855
|
} catch (error) {
|
package/src/types/experiments.ts
CHANGED
|
@@ -131,6 +131,12 @@ export type EvaluatorParams<TaskOutputType = TaskOutput> = {
|
|
|
131
131
|
* Metadata associated with the Dataset Example
|
|
132
132
|
*/
|
|
133
133
|
metadata?: Example["metadata"];
|
|
134
|
+
/**
|
|
135
|
+
* The trace ID of the task run, if available.
|
|
136
|
+
* Can be used to fetch and analyze the task's trace
|
|
137
|
+
* (e.g., for trajectory evaluation or action verification).
|
|
138
|
+
*/
|
|
139
|
+
traceId?: string | null;
|
|
134
140
|
};
|
|
135
141
|
|
|
136
142
|
export type Evaluator = {
|
package/src/types/spans.ts
CHANGED