@arizeai/phoenix-client 7.7.0 → 7.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. package/CHANGELOG.md +6 -0
  2. package/dist/esm/experiments/resumeEvaluation.d.ts +0 -60
  3. package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
  4. package/dist/esm/experiments/resumeEvaluation.js +44 -37
  5. package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
  6. package/dist/esm/experiments/resumeExperiment.d.ts +0 -52
  7. package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
  8. package/dist/esm/experiments/resumeExperiment.js +42 -35
  9. package/dist/esm/experiments/resumeExperiment.js.map +1 -1
  10. package/dist/esm/experiments/runExperiment.d.ts.map +1 -1
  11. package/dist/esm/experiments/runExperiment.js +152 -123
  12. package/dist/esm/experiments/runExperiment.js.map +1 -1
  13. package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
  14. package/dist/esm/prompts/sdks/toOpenAI.js +49 -58
  15. package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
  16. package/dist/esm/sessions/sessionUtils.d.ts.map +1 -1
  17. package/dist/esm/sessions/sessionUtils.js +3 -0
  18. package/dist/esm/sessions/sessionUtils.js.map +1 -1
  19. package/dist/esm/spans/getSpans.d.ts +0 -70
  20. package/dist/esm/spans/getSpans.d.ts.map +1 -1
  21. package/dist/esm/spans/getSpans.js +42 -93
  22. package/dist/esm/spans/getSpans.js.map +1 -1
  23. package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -1
  24. package/dist/esm/testing/phoenix-test-tracking.js +109 -78
  25. package/dist/esm/testing/phoenix-test-tracking.js.map +1 -1
  26. package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
  27. package/dist/esm/types/sessions.d.ts +6 -0
  28. package/dist/esm/types/sessions.d.ts.map +1 -1
  29. package/dist/src/experiments/resumeEvaluation.d.ts +0 -60
  30. package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
  31. package/dist/src/experiments/resumeEvaluation.js +44 -37
  32. package/dist/src/experiments/resumeEvaluation.js.map +1 -1
  33. package/dist/src/experiments/resumeExperiment.d.ts +0 -52
  34. package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
  35. package/dist/src/experiments/resumeExperiment.js +42 -35
  36. package/dist/src/experiments/resumeExperiment.js.map +1 -1
  37. package/dist/src/experiments/runExperiment.d.ts.map +1 -1
  38. package/dist/src/experiments/runExperiment.js +148 -116
  39. package/dist/src/experiments/runExperiment.js.map +1 -1
  40. package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
  41. package/dist/src/prompts/sdks/toOpenAI.js +56 -63
  42. package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
  43. package/dist/src/sessions/sessionUtils.d.ts.map +1 -1
  44. package/dist/src/sessions/sessionUtils.js +3 -0
  45. package/dist/src/sessions/sessionUtils.js.map +1 -1
  46. package/dist/src/spans/getSpans.d.ts +0 -70
  47. package/dist/src/spans/getSpans.d.ts.map +1 -1
  48. package/dist/src/spans/getSpans.js +43 -94
  49. package/dist/src/spans/getSpans.js.map +1 -1
  50. package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -1
  51. package/dist/src/testing/phoenix-test-tracking.js +117 -84
  52. package/dist/src/testing/phoenix-test-tracking.js.map +1 -1
  53. package/dist/src/types/sessions.d.ts +6 -0
  54. package/dist/src/types/sessions.d.ts.map +1 -1
  55. package/dist/tsconfig.tsbuildinfo +1 -1
  56. package/docs/sessions.mdx +10 -1
  57. package/package.json +1 -1
  58. package/src/experiments/resumeEvaluation.ts +78 -48
  59. package/src/experiments/resumeExperiment.ts +73 -46
  60. package/src/experiments/runExperiment.ts +235 -129
  61. package/src/prompts/sdks/toOpenAI.ts +58 -61
  62. package/src/sessions/sessionUtils.ts +3 -0
  63. package/src/spans/getSpans.ts +84 -48
  64. package/src/testing/phoenix-test-tracking.ts +154 -90
  65. package/src/types/sessions.ts +6 -0
package/docs/sessions.mdx CHANGED
@@ -29,10 +29,19 @@ const sessions = await listSessions({
29
29
  });
30
30
 
31
31
  for (const session of sessions) {
32
- console.log(session.sessionId);
32
+ console.log({
33
+ sessionId: session.sessionId,
34
+ promptTokens: session.tokenCountPrompt,
35
+ completionTokens: session.tokenCountCompletion,
36
+ totalTokens: session.tokenCountTotal,
37
+ });
33
38
  }
34
39
  ```
35
40
 
41
+ Each session includes cumulative prompt, completion, and total token counts across
42
+ all of its spans. The fields may be `undefined` when using a Phoenix server that
43
+ does not return session token usage.
44
+
36
45
  ## Retrieve A Session And Its Turns
37
46
 
38
47
  ```ts
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@arizeai/phoenix-client",
3
- "version": "7.7.0",
3
+ "version": "7.7.1",
4
4
  "description": "A client for the Phoenix API",
5
5
  "keywords": [
6
6
  "arize",
@@ -301,6 +301,71 @@ function setupEvaluationTracer({
301
301
  * });
302
302
  * ```
303
303
  */
304
+ function isEmptyEvaluationBatch({
305
+ batchLength,
306
+ totalProcessed,
307
+ logger,
308
+ }: {
309
+ batchLength: number;
310
+ totalProcessed: number;
311
+ logger: Logger;
312
+ }): boolean {
313
+ if (batchLength > 0) return false;
314
+ if (totalProcessed === 0) {
315
+ logger.info(`${PROGRESS_PREFIX.completed}No incomplete evaluations found.`);
316
+ }
317
+ return true;
318
+ }
319
+
320
+ function shouldContinueEvaluationFetch({
321
+ cursor,
322
+ signal,
323
+ }: {
324
+ cursor: string | null;
325
+ signal: AbortSignal;
326
+ }): boolean {
327
+ return cursor !== null && !signal.aborted;
328
+ }
329
+
330
+ function getEvaluationExecutionError({
331
+ rejections,
332
+ isAborted,
333
+ logger,
334
+ }: {
335
+ rejections: unknown[];
336
+ isAborted: boolean;
337
+ logger: Logger;
338
+ }): Error | null {
339
+ if (rejections.length === 0) return null;
340
+ const fetchError = rejections.find(
341
+ (reason) => reason instanceof EvaluationFetchError
342
+ );
343
+ if (fetchError instanceof Error) {
344
+ logger.error(`Critical: Failed to fetch evaluations from server`);
345
+ return fetchError;
346
+ }
347
+ const workerError = rejections.find(
348
+ (reason) =>
349
+ reason instanceof Error &&
350
+ !(reason instanceof EvaluationFetchError) &&
351
+ !(reason instanceof ChannelError)
352
+ );
353
+ if (workerError instanceof Error) return workerError;
354
+ const channelError = rejections.find(
355
+ (reason) => reason instanceof ChannelError
356
+ );
357
+ if (channelError instanceof Error && isAborted) {
358
+ return new EvaluationAbortedError(
359
+ "Evaluation stopped due to error in concurrent evaluator",
360
+ channelError
361
+ );
362
+ }
363
+ const reason = rejections[0];
364
+ const error = reason instanceof Error ? reason : new Error(String(reason));
365
+ logger.error(`Unexpected error during evaluation: ${error.message}`);
366
+ return error;
367
+ }
368
+
304
369
  export async function resumeEvaluation({
305
370
  client: _client,
306
371
  experimentId,
@@ -425,12 +490,13 @@ export async function resumeEvaluation({
425
490
  const batchIncomplete = res.data?.data;
426
491
  invariant(batchIncomplete, "Failed to fetch incomplete evaluations");
427
492
 
428
- if (batchIncomplete.length === 0) {
429
- if (totalProcessed === 0) {
430
- logger.info(
431
- `${PROGRESS_PREFIX.completed}No incomplete evaluations found.`
432
- );
433
- }
493
+ if (
494
+ isEmptyEvaluationBatch({
495
+ batchLength: batchIncomplete.length,
496
+ totalProcessed,
497
+ logger,
498
+ })
499
+ ) {
434
500
  break;
435
501
  }
436
502
 
@@ -468,7 +534,7 @@ export async function resumeEvaluation({
468
534
  logger.debug(
469
535
  `${PROGRESS_PREFIX.progress}Fetched batch of ${batchCount} evaluation tasks.`
470
536
  );
471
- } while (cursor !== null && !signal.aborted);
537
+ } while (shouldContinueEvaluationFetch({ cursor, signal }));
472
538
  } catch (error) {
473
539
  // Re-throw with context preservation
474
540
  if (error instanceof EvaluationFetchError) {
@@ -547,47 +613,11 @@ export async function resumeEvaluation({
547
613
  )
548
614
  .map((result) => result.reason);
549
615
 
550
- if (rejections.length > 0) {
551
- // Classify and handle errors based on their nature. When multiple tasks
552
- // reject, prefer the most meaningful error over incidental fallout
553
- // (e.g. a ChannelError raised in a blocked worker when the channel
554
- // closes on abort).
555
- const fetchError = rejections.find(
556
- (reason) => reason instanceof EvaluationFetchError
557
- );
558
- const workerError = rejections.find(
559
- (reason) =>
560
- reason instanceof Error &&
561
- !(reason instanceof EvaluationFetchError) &&
562
- !(reason instanceof ChannelError)
563
- );
564
- const channelError = rejections.find(
565
- (reason) => reason instanceof ChannelError
566
- );
567
-
568
- if (fetchError) {
569
- // Producer failed - this is ALWAYS critical regardless of stopOnFirstError
570
- logger.error(`Critical: Failed to fetch evaluations from server`);
571
- executionError = fetchError;
572
- } else if (workerError) {
573
- // Worker error in stopOnFirstError mode - already logged by worker
574
- executionError = workerError;
575
- } else if (channelError && signal.aborted) {
576
- // Channel closed due to intentional abort - wrap in semantic error
577
- executionError = new EvaluationAbortedError(
578
- "Evaluation stopped due to error in concurrent evaluator",
579
- channelError
580
- );
581
- } else {
582
- // Unexpected error (not from worker, not from producer fetch)
583
- // This could be a bug in our code or infrastructure failure
584
- const reason = rejections[0];
585
- const err =
586
- reason instanceof Error ? reason : new Error(String(reason));
587
- logger.error(`Unexpected error during evaluation: ${err.message}`);
588
- executionError = err;
589
- }
590
- }
616
+ executionError = getEvaluationExecutionError({
617
+ rejections,
618
+ isAborted: signal.aborted,
619
+ logger,
620
+ });
591
621
  } finally {
592
622
  // Ensure channel is closed even if there are unexpected errors
593
623
  // This is a safety net in case producer's finally block didn't execute
@@ -276,6 +276,67 @@ function setupTracer({
276
276
  * });
277
277
  * ```
278
278
  */
279
+ function getResumeEvaluators({
280
+ evaluators,
281
+ executionError,
282
+ }: {
283
+ evaluators: readonly ExperimentEvaluatorLike[] | undefined;
284
+ executionError: Error | null;
285
+ }): readonly ExperimentEvaluatorLike[] | null {
286
+ return evaluators && evaluators.length > 0 && !executionError
287
+ ? evaluators
288
+ : null;
289
+ }
290
+
291
+ function shouldWarnAboutFailedRuns({
292
+ totalFailed,
293
+ executionError,
294
+ }: {
295
+ totalFailed: number;
296
+ executionError: Error | null;
297
+ }): boolean {
298
+ return totalFailed > 0 && !executionError;
299
+ }
300
+
301
+ function getTaskExecutionError({
302
+ rejections,
303
+ isAborted,
304
+ logger,
305
+ }: {
306
+ rejections: unknown[];
307
+ isAborted: boolean;
308
+ logger: Logger;
309
+ }): Error | null {
310
+ if (rejections.length === 0) return null;
311
+ const fetchError = rejections.find(
312
+ (reason) => reason instanceof TaskFetchError
313
+ );
314
+ if (fetchError instanceof Error) {
315
+ logger.error(`Critical: Failed to fetch incomplete runs from server`);
316
+ return fetchError;
317
+ }
318
+ const taskError = rejections.find(
319
+ (reason) =>
320
+ reason instanceof Error &&
321
+ !(reason instanceof TaskFetchError) &&
322
+ !(reason instanceof ChannelError)
323
+ );
324
+ if (taskError instanceof Error) return taskError;
325
+ const channelError = rejections.find(
326
+ (reason) => reason instanceof ChannelError
327
+ );
328
+ if (channelError instanceof Error && isAborted) {
329
+ return new TaskAbortedError(
330
+ "Task execution stopped due to error in concurrent worker",
331
+ channelError
332
+ );
333
+ }
334
+ const reason = rejections[0];
335
+ const error = reason instanceof Error ? reason : new Error(String(reason));
336
+ logger.error(`Unexpected error during task execution: ${error.message}`);
337
+ return error;
338
+ }
339
+
279
340
  export async function resumeExperiment({
280
341
  client: _client,
281
342
  experimentId,
@@ -514,49 +575,11 @@ export async function resumeExperiment({
514
575
  )
515
576
  .map((result) => result.reason);
516
577
 
517
- if (rejections.length > 0) {
518
- // Classify and handle errors based on their nature. When multiple tasks
519
- // reject, prefer the most meaningful error over incidental fallout
520
- // (e.g. a ChannelError raised in a blocked worker when the channel
521
- // closes on abort).
522
- const fetchError = rejections.find(
523
- (reason) => reason instanceof TaskFetchError
524
- );
525
- const taskError = rejections.find(
526
- (reason) =>
527
- reason instanceof Error &&
528
- !(reason instanceof TaskFetchError) &&
529
- !(reason instanceof ChannelError)
530
- );
531
- const channelError = rejections.find(
532
- (reason) => reason instanceof ChannelError
533
- );
534
-
535
- if (fetchError) {
536
- // Producer failed - this is ALWAYS critical regardless of stopOnFirstError
537
- logger.error(`Critical: Failed to fetch incomplete runs from server`);
538
- executionError = fetchError;
539
- } else if (taskError) {
540
- // Worker error in stopOnFirstError mode - already logged by worker
541
- executionError = taskError;
542
- } else if (channelError && signal.aborted) {
543
- // Channel closed due to intentional abort - wrap in semantic error
544
- executionError = new TaskAbortedError(
545
- "Task execution stopped due to error in concurrent worker",
546
- channelError
547
- );
548
- } else {
549
- // Unexpected error (not from worker, not from producer fetch)
550
- // This could be a bug in our code or infrastructure failure
551
- const reason = rejections[0];
552
- const err =
553
- reason instanceof Error ? reason : new Error(String(reason));
554
- logger.error(
555
- `Unexpected error during task execution: ${err.message}`
556
- );
557
- executionError = err;
558
- }
559
- }
578
+ executionError = getTaskExecutionError({
579
+ rejections,
580
+ isAborted: signal.aborted,
581
+ logger,
582
+ });
560
583
  } finally {
561
584
  // Ensure channel is closed even if there are unexpected errors
562
585
  // This is a safety net in case producer's finally block didn't execute
@@ -570,11 +593,15 @@ export async function resumeExperiment({
570
593
  logger.info(`${PROGRESS_PREFIX.completed}Task runs completed.`);
571
594
  }
572
595
 
573
- if (totalFailed > 0 && !executionError) {
596
+ if (shouldWarnAboutFailedRuns({ totalFailed, executionError })) {
574
597
  logger.warn(`${totalFailed} out of ${totalProcessed} runs failed.`);
575
598
  }
576
599
 
577
- if (evaluators && evaluators.length > 0 && !executionError) {
600
+ const resumeEvaluators = getResumeEvaluators({
601
+ evaluators,
602
+ executionError,
603
+ });
604
+ if (resumeEvaluators) {
578
605
  await cleanupOwnedTracerProvider({
579
606
  provider,
580
607
  globalRegistration,
@@ -585,7 +612,7 @@ export async function resumeExperiment({
585
612
  logger.info(`${PROGRESS_PREFIX.start}Running evaluators.`);
586
613
  await resumeEvaluation({
587
614
  experimentId,
588
- evaluators: [...evaluators],
615
+ evaluators: [...resumeEvaluators],
589
616
  client,
590
617
  logger,
591
618
  concurrency,