@arizeai/phoenix-client 7.7.0 → 7.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/CHANGELOG.md +12 -0
  2. package/dist/esm/__generated__/api/v1.d.ts +127 -6
  3. package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
  4. package/dist/esm/experiments/resumeEvaluation.d.ts +0 -60
  5. package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
  6. package/dist/esm/experiments/resumeEvaluation.js +44 -37
  7. package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
  8. package/dist/esm/experiments/resumeExperiment.d.ts +0 -52
  9. package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
  10. package/dist/esm/experiments/resumeExperiment.js +42 -35
  11. package/dist/esm/experiments/resumeExperiment.js.map +1 -1
  12. package/dist/esm/experiments/runExperiment.d.ts.map +1 -1
  13. package/dist/esm/experiments/runExperiment.js +152 -123
  14. package/dist/esm/experiments/runExperiment.js.map +1 -1
  15. package/dist/esm/prompts/constants.d.ts.map +1 -1
  16. package/dist/esm/prompts/constants.js +1 -0
  17. package/dist/esm/prompts/constants.js.map +1 -1
  18. package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
  19. package/dist/esm/prompts/sdks/toOpenAI.js +51 -58
  20. package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
  21. package/dist/esm/sessions/sessionUtils.d.ts.map +1 -1
  22. package/dist/esm/sessions/sessionUtils.js +3 -0
  23. package/dist/esm/sessions/sessionUtils.js.map +1 -1
  24. package/dist/esm/spans/getSpans.d.ts +0 -70
  25. package/dist/esm/spans/getSpans.d.ts.map +1 -1
  26. package/dist/esm/spans/getSpans.js +42 -93
  27. package/dist/esm/spans/getSpans.js.map +1 -1
  28. package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -1
  29. package/dist/esm/testing/phoenix-test-tracking.js +109 -78
  30. package/dist/esm/testing/phoenix-test-tracking.js.map +1 -1
  31. package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
  32. package/dist/esm/types/prompts.d.ts +1 -1
  33. package/dist/esm/types/prompts.d.ts.map +1 -1
  34. package/dist/esm/types/prompts.js.map +1 -1
  35. package/dist/esm/types/sessions.d.ts +6 -0
  36. package/dist/esm/types/sessions.d.ts.map +1 -1
  37. package/dist/esm/utils/getPromptBySelector.d.ts +1 -1
  38. package/dist/src/__generated__/api/v1.d.ts +127 -6
  39. package/dist/src/__generated__/api/v1.d.ts.map +1 -1
  40. package/dist/src/experiments/resumeEvaluation.d.ts +0 -60
  41. package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
  42. package/dist/src/experiments/resumeEvaluation.js +44 -37
  43. package/dist/src/experiments/resumeEvaluation.js.map +1 -1
  44. package/dist/src/experiments/resumeExperiment.d.ts +0 -52
  45. package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
  46. package/dist/src/experiments/resumeExperiment.js +42 -35
  47. package/dist/src/experiments/resumeExperiment.js.map +1 -1
  48. package/dist/src/experiments/runExperiment.d.ts.map +1 -1
  49. package/dist/src/experiments/runExperiment.js +148 -116
  50. package/dist/src/experiments/runExperiment.js.map +1 -1
  51. package/dist/src/prompts/constants.d.ts.map +1 -1
  52. package/dist/src/prompts/constants.js +1 -0
  53. package/dist/src/prompts/constants.js.map +1 -1
  54. package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
  55. package/dist/src/prompts/sdks/toOpenAI.js +58 -63
  56. package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
  57. package/dist/src/sessions/sessionUtils.d.ts.map +1 -1
  58. package/dist/src/sessions/sessionUtils.js +3 -0
  59. package/dist/src/sessions/sessionUtils.js.map +1 -1
  60. package/dist/src/spans/getSpans.d.ts +0 -70
  61. package/dist/src/spans/getSpans.d.ts.map +1 -1
  62. package/dist/src/spans/getSpans.js +43 -94
  63. package/dist/src/spans/getSpans.js.map +1 -1
  64. package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -1
  65. package/dist/src/testing/phoenix-test-tracking.js +117 -84
  66. package/dist/src/testing/phoenix-test-tracking.js.map +1 -1
  67. package/dist/src/types/prompts.d.ts +1 -1
  68. package/dist/src/types/prompts.d.ts.map +1 -1
  69. package/dist/src/types/prompts.js.map +1 -1
  70. package/dist/src/types/sessions.d.ts +6 -0
  71. package/dist/src/types/sessions.d.ts.map +1 -1
  72. package/dist/src/utils/getPromptBySelector.d.ts +1 -1
  73. package/dist/tsconfig.tsbuildinfo +1 -1
  74. package/docs/sessions.mdx +10 -1
  75. package/package.json +1 -1
  76. package/src/__generated__/api/v1.ts +127 -6
  77. package/src/experiments/resumeEvaluation.ts +78 -48
  78. package/src/experiments/resumeExperiment.ts +73 -46
  79. package/src/experiments/runExperiment.ts +235 -129
  80. package/src/prompts/constants.ts +1 -0
  81. package/src/prompts/sdks/toOpenAI.ts +60 -61
  82. package/src/sessions/sessionUtils.ts +3 -0
  83. package/src/spans/getSpans.ts +84 -48
  84. package/src/testing/phoenix-test-tracking.ts +154 -90
  85. package/src/types/prompts.ts +2 -1
  86. package/src/types/sessions.ts +6 -0
package/docs/sessions.mdx CHANGED
@@ -29,10 +29,19 @@ const sessions = await listSessions({
29
29
  });
30
30
 
31
31
  for (const session of sessions) {
32
- console.log(session.sessionId);
32
+ console.log({
33
+ sessionId: session.sessionId,
34
+ promptTokens: session.tokenCountPrompt,
35
+ completionTokens: session.tokenCountCompletion,
36
+ totalTokens: session.tokenCountTotal,
37
+ });
33
38
  }
34
39
  ```
35
40
 
41
+ Each session includes cumulative prompt, completion, and total token counts across
42
+ all of its spans. The fields may be `undefined` when using a Phoenix server that
43
+ does not return session token usage.
44
+
36
45
  ## Retrieve A Session And Its Turns
37
46
 
38
47
  ```ts
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@arizeai/phoenix-client",
3
- "version": "7.7.0",
3
+ "version": "7.8.0",
4
4
  "description": "A client for the Phoenix API",
5
5
  "keywords": [
6
6
  "arize",
@@ -761,7 +761,11 @@ export interface paths {
761
761
  get: operations["listProjectTraces"];
762
762
  put?: never;
763
763
  post?: never;
764
- delete?: never;
764
+ /**
765
+ * Delete traces from a project
766
+ * @description Delete traces from a project without deleting the project or its configuration. Only traces whose start time is within the required `[start_time, end_time)` interval are deleted. Associated spans are cascade deleted, and project sessions left with no remaining traces are also deleted. Naive datetimes are interpreted as UTC.
767
+ */
768
+ delete: operations["deleteProjectTraces"];
765
769
  options?: never;
766
770
  head?: never;
767
771
  patch?: never;
@@ -1568,7 +1572,7 @@ export interface paths {
1568
1572
  put?: never;
1569
1573
  /**
1570
1574
  * OpenAI-compatible chat completions
1571
- * @description Creates a chat completion using the OpenAI wire format, proxying to the selected provider with credentials resolved on the server (secret store first, environment second) — callers never handle provider API keys. Model must be '{provider}:{model_name}' for a built-in provider (one of anthropic, aws, azure_openai, cerebras, deepseek, fireworks, google, groq, moonshot, ollama, openai, perplexity, together, xai) or 'custom:{provider_id}:{model_name}' for a stored custom provider, e.g. 'openai:gpt-4o' or 'anthropic:claude-sonnet-4-5'. Set `stream: true` for server-sent events of `chat.completion.chunk` payloads terminated by `data: [DONE]`. Tool calling is not supported.
1575
+ * @description Creates a chat completion using the OpenAI wire format, proxying to the selected provider with credentials resolved on the server (secret store first, environment second) — callers never handle provider API keys. Model must be '{provider}:{model_name}' for a built-in provider (one of anthropic, aws, azure_openai, cerebras, deepseek, fireworks, google, groq, moonshot, ollama, openai, perplexity, together, xai, zai) or 'custom:{provider_id}:{model_name}' for a stored custom provider, e.g. 'openai:gpt-4o' or 'anthropic:claude-sonnet-4-5'. Set `stream: true` for server-sent events of `chat.completion.chunk` payloads terminated by `data: [DONE]`. Tool calling is not supported.
1572
1576
  *
1573
1577
  * **Phoenix is not an AI gateway.** The same server also takes on trace ingestion traffic, so routing production LLM calls through it competes with ingestion. Use this endpoint only to quickly try out different models in non-production environments.
1574
1578
  */
@@ -2424,6 +2428,11 @@ export interface components {
2424
2428
  * @description The id of the last transcript message the client has rendered, used for optimistic concurrency. Omit when the session has no messages; required (and validated against the persisted transcript) once it does. On mismatch the server rejects the send with HTTP 409 and code ``agent_session_messages_stale`` — the client should refetch the session before retrying.
2425
2429
  */
2426
2430
  lastMessageId?: string | null;
2431
+ /**
2432
+ * Credentials
2433
+ * @description Client-held credentials for optional integrations (e.g. the user's own GitHub personal access token under the key ``GITHUB_PERSONAL_ACCESS_TOKEN``), used only for the duration of the turn and never persisted. Unknown keys are rejected.
2434
+ */
2435
+ credentials?: components["schemas"]["ChatRequestCredential"][];
2427
2436
  /**
2428
2437
  * Recordlocaltraces
2429
2438
  * @default false
@@ -2441,6 +2450,28 @@ export interface components {
2441
2450
  */
2442
2451
  instrumentUserId?: boolean;
2443
2452
  };
2453
+ /**
2454
+ * ChatRequestCredential
2455
+ * @description One client-held credential riding the request for the duration of a turn.
2456
+ *
2457
+ * The value is ephemeral: it is injected server-side as transport auth for
2458
+ * the matching integration and is never persisted, traced, or echoed. It is
2459
+ * top-level on the request body — never part of the message — so it cannot
2460
+ * reach the session transcript.
2461
+ */
2462
+ ChatRequestCredential: {
2463
+ /**
2464
+ * Key
2465
+ * @description The credential's secret-key name.
2466
+ * @constant
2467
+ */
2468
+ key: "GITHUB_PERSONAL_ACCESS_TOKEN";
2469
+ /**
2470
+ * Value
2471
+ * Format: password
2472
+ */
2473
+ value: string;
2474
+ };
2444
2475
  /** CodeEvaluatorUIContext */
2445
2476
  CodeEvaluatorUIContext: {
2446
2477
  /**
@@ -2561,7 +2592,7 @@ export interface components {
2561
2592
  CreateChatCompletionRequestBody: {
2562
2593
  /**
2563
2594
  * Model
2564
- * @description Model must be '{provider}:{model_name}' for a built-in provider (one of anthropic, aws, azure_openai, cerebras, deepseek, fireworks, google, groq, moonshot, ollama, openai, perplexity, together, xai) or 'custom:{provider_id}:{model_name}' for a stored custom provider, e.g. 'openai:gpt-4o' or 'anthropic:claude-sonnet-4-5'.
2595
+ * @description Model must be '{provider}:{model_name}' for a built-in provider (one of anthropic, aws, azure_openai, cerebras, deepseek, fireworks, google, groq, moonshot, ollama, openai, perplexity, together, xai, zai) or 'custom:{provider_id}:{model_name}' for a stored custom provider, e.g. 'openai:gpt-4o' or 'anthropic:claude-sonnet-4-5'.
2565
2596
  */
2566
2597
  model: string;
2567
2598
  /** Messages */
@@ -4057,7 +4088,7 @@ export interface components {
4057
4088
  * ModelProvider
4058
4089
  * @enum {string}
4059
4090
  */
4060
- ModelProvider: "OPENAI" | "AZURE_OPENAI" | "ANTHROPIC" | "GOOGLE" | "DEEPSEEK" | "XAI" | "OLLAMA" | "AWS" | "CEREBRAS" | "FIREWORKS" | "GROQ" | "MOONSHOT" | "PERPLEXITY" | "TOGETHER";
4091
+ ModelProvider: "OPENAI" | "AZURE_OPENAI" | "ANTHROPIC" | "GOOGLE" | "DEEPSEEK" | "XAI" | "OLLAMA" | "AWS" | "CEREBRAS" | "FIREWORKS" | "GROQ" | "MOONSHOT" | "PERPLEXITY" | "TOGETHER" | "ZAI";
4061
4092
  /** OAuth2User */
4062
4093
  OAuth2User: {
4063
4094
  /** Id */
@@ -5236,7 +5267,7 @@ export interface components {
5236
5267
  template_type: components["schemas"]["PromptTemplateType"];
5237
5268
  template_format: components["schemas"]["PromptTemplateFormat"];
5238
5269
  /** Invocation Parameters */
5239
- invocation_parameters: components["schemas"]["PromptOpenAIInvocationParameters"] | components["schemas"]["PromptAzureOpenAIInvocationParameters"] | components["schemas"]["PromptAnthropicInvocationParameters"] | components["schemas"]["PromptGoogleInvocationParameters"] | components["schemas"]["PromptDeepSeekInvocationParameters"] | components["schemas"]["PromptXAIInvocationParameters"] | components["schemas"]["PromptOllamaInvocationParameters"] | components["schemas"]["PromptAwsInvocationParameters"] | components["schemas"]["PromptCerebrasInvocationParameters"] | components["schemas"]["PromptFireworksInvocationParameters"] | components["schemas"]["PromptGroqInvocationParameters"] | components["schemas"]["PromptMoonshotInvocationParameters"] | components["schemas"]["PromptPerplexityInvocationParameters"] | components["schemas"]["PromptTogetherInvocationParameters"];
5270
+ invocation_parameters: components["schemas"]["PromptOpenAIInvocationParameters"] | components["schemas"]["PromptAzureOpenAIInvocationParameters"] | components["schemas"]["PromptAnthropicInvocationParameters"] | components["schemas"]["PromptGoogleInvocationParameters"] | components["schemas"]["PromptDeepSeekInvocationParameters"] | components["schemas"]["PromptXAIInvocationParameters"] | components["schemas"]["PromptOllamaInvocationParameters"] | components["schemas"]["PromptAwsInvocationParameters"] | components["schemas"]["PromptCerebrasInvocationParameters"] | components["schemas"]["PromptFireworksInvocationParameters"] | components["schemas"]["PromptGroqInvocationParameters"] | components["schemas"]["PromptMoonshotInvocationParameters"] | components["schemas"]["PromptPerplexityInvocationParameters"] | components["schemas"]["PromptTogetherInvocationParameters"] | components["schemas"]["PromptZAIInvocationParameters"];
5240
5271
  tools?: components["schemas"]["PromptTools"] | null;
5241
5272
  /** Response Format */
5242
5273
  response_format?: components["schemas"]["PromptResponseFormatJSONSchema"] | null;
@@ -5255,7 +5286,7 @@ export interface components {
5255
5286
  template_type: components["schemas"]["PromptTemplateType"];
5256
5287
  template_format: components["schemas"]["PromptTemplateFormat"];
5257
5288
  /** Invocation Parameters */
5258
- invocation_parameters: components["schemas"]["PromptOpenAIInvocationParameters"] | components["schemas"]["PromptAzureOpenAIInvocationParameters"] | components["schemas"]["PromptAnthropicInvocationParameters"] | components["schemas"]["PromptGoogleInvocationParameters"] | components["schemas"]["PromptDeepSeekInvocationParameters"] | components["schemas"]["PromptXAIInvocationParameters"] | components["schemas"]["PromptOllamaInvocationParameters"] | components["schemas"]["PromptAwsInvocationParameters"] | components["schemas"]["PromptCerebrasInvocationParameters"] | components["schemas"]["PromptFireworksInvocationParameters"] | components["schemas"]["PromptGroqInvocationParameters"] | components["schemas"]["PromptMoonshotInvocationParameters"] | components["schemas"]["PromptPerplexityInvocationParameters"] | components["schemas"]["PromptTogetherInvocationParameters"];
5289
+ invocation_parameters: components["schemas"]["PromptOpenAIInvocationParameters"] | components["schemas"]["PromptAzureOpenAIInvocationParameters"] | components["schemas"]["PromptAnthropicInvocationParameters"] | components["schemas"]["PromptGoogleInvocationParameters"] | components["schemas"]["PromptDeepSeekInvocationParameters"] | components["schemas"]["PromptXAIInvocationParameters"] | components["schemas"]["PromptOllamaInvocationParameters"] | components["schemas"]["PromptAwsInvocationParameters"] | components["schemas"]["PromptCerebrasInvocationParameters"] | components["schemas"]["PromptFireworksInvocationParameters"] | components["schemas"]["PromptGroqInvocationParameters"] | components["schemas"]["PromptMoonshotInvocationParameters"] | components["schemas"]["PromptPerplexityInvocationParameters"] | components["schemas"]["PromptTogetherInvocationParameters"] | components["schemas"]["PromptZAIInvocationParameters"];
5259
5290
  tools?: components["schemas"]["PromptTools"] | null;
5260
5291
  /** Response Format */
5261
5292
  response_format?: components["schemas"]["PromptResponseFormatJSONSchema"] | null;
@@ -5323,6 +5354,43 @@ export interface components {
5323
5354
  [key: string]: unknown;
5324
5355
  };
5325
5356
  };
5357
+ /** PromptZAIInvocationParameters */
5358
+ PromptZAIInvocationParameters: {
5359
+ /**
5360
+ * @description discriminator enum property added by openapi-typescript
5361
+ * @enum {string}
5362
+ */
5363
+ type: "zai";
5364
+ zai: components["schemas"]["PromptZAIInvocationParametersContent"];
5365
+ };
5366
+ /** PromptZAIInvocationParametersContent */
5367
+ PromptZAIInvocationParametersContent: {
5368
+ /** Temperature */
5369
+ temperature?: number;
5370
+ /** Max Tokens */
5371
+ max_tokens?: number;
5372
+ /** Max Completion Tokens */
5373
+ max_completion_tokens?: number;
5374
+ /** Frequency Penalty */
5375
+ frequency_penalty?: number;
5376
+ /** Presence Penalty */
5377
+ presence_penalty?: number;
5378
+ /** Top P */
5379
+ top_p?: number;
5380
+ /** Seed */
5381
+ seed?: number;
5382
+ /** Stop */
5383
+ stop?: string[];
5384
+ /**
5385
+ * Reasoning Effort
5386
+ * @enum {string}
5387
+ */
5388
+ reasoning_effort?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh";
5389
+ /** Extra Body */
5390
+ extra_body?: {
5391
+ [key: string]: unknown;
5392
+ };
5393
+ };
5326
5394
  /**
5327
5395
  * PydanticAIMessageMetadata
5328
5396
  * @description Local pin of pydantic-ai's message-level ``pydantic_ai`` metadata
@@ -10041,6 +10109,59 @@ export interface operations {
10041
10109
  };
10042
10110
  };
10043
10111
  };
10112
+ deleteProjectTraces: {
10113
+ parameters: {
10114
+ query: {
10115
+ /** @description Required inclusive lower bound on trace start time (ISO 8601). */
10116
+ start_time: string;
10117
+ /** @description Required exclusive upper bound on trace start time (ISO 8601). */
10118
+ end_time: string;
10119
+ };
10120
+ header?: never;
10121
+ path: {
10122
+ /** @description The project identifier: either project ID or project name. */
10123
+ project_identifier: string;
10124
+ };
10125
+ cookie?: never;
10126
+ };
10127
+ requestBody?: never;
10128
+ responses: {
10129
+ /** @description No content returned after the matching traces are deleted */
10130
+ 204: {
10131
+ headers: {
10132
+ [name: string]: unknown;
10133
+ };
10134
+ content?: never;
10135
+ };
10136
+ /** @description Forbidden */
10137
+ 403: {
10138
+ headers: {
10139
+ [name: string]: unknown;
10140
+ };
10141
+ content: {
10142
+ "text/plain": string;
10143
+ };
10144
+ };
10145
+ /** @description Not Found */
10146
+ 404: {
10147
+ headers: {
10148
+ [name: string]: unknown;
10149
+ };
10150
+ content: {
10151
+ "text/plain": string;
10152
+ };
10153
+ };
10154
+ /** @description Unprocessable Entity */
10155
+ 422: {
10156
+ headers: {
10157
+ [name: string]: unknown;
10158
+ };
10159
+ content: {
10160
+ "text/plain": string;
10161
+ };
10162
+ };
10163
+ };
10164
+ };
10044
10165
  annotateTraces: {
10045
10166
  parameters: {
10046
10167
  query?: {
@@ -301,6 +301,71 @@ function setupEvaluationTracer({
301
301
  * });
302
302
  * ```
303
303
  */
304
+ function isEmptyEvaluationBatch({
305
+ batchLength,
306
+ totalProcessed,
307
+ logger,
308
+ }: {
309
+ batchLength: number;
310
+ totalProcessed: number;
311
+ logger: Logger;
312
+ }): boolean {
313
+ if (batchLength > 0) return false;
314
+ if (totalProcessed === 0) {
315
+ logger.info(`${PROGRESS_PREFIX.completed}No incomplete evaluations found.`);
316
+ }
317
+ return true;
318
+ }
319
+
320
+ function shouldContinueEvaluationFetch({
321
+ cursor,
322
+ signal,
323
+ }: {
324
+ cursor: string | null;
325
+ signal: AbortSignal;
326
+ }): boolean {
327
+ return cursor !== null && !signal.aborted;
328
+ }
329
+
330
+ function getEvaluationExecutionError({
331
+ rejections,
332
+ isAborted,
333
+ logger,
334
+ }: {
335
+ rejections: unknown[];
336
+ isAborted: boolean;
337
+ logger: Logger;
338
+ }): Error | null {
339
+ if (rejections.length === 0) return null;
340
+ const fetchError = rejections.find(
341
+ (reason) => reason instanceof EvaluationFetchError
342
+ );
343
+ if (fetchError instanceof Error) {
344
+ logger.error(`Critical: Failed to fetch evaluations from server`);
345
+ return fetchError;
346
+ }
347
+ const workerError = rejections.find(
348
+ (reason) =>
349
+ reason instanceof Error &&
350
+ !(reason instanceof EvaluationFetchError) &&
351
+ !(reason instanceof ChannelError)
352
+ );
353
+ if (workerError instanceof Error) return workerError;
354
+ const channelError = rejections.find(
355
+ (reason) => reason instanceof ChannelError
356
+ );
357
+ if (channelError instanceof Error && isAborted) {
358
+ return new EvaluationAbortedError(
359
+ "Evaluation stopped due to error in concurrent evaluator",
360
+ channelError
361
+ );
362
+ }
363
+ const reason = rejections[0];
364
+ const error = reason instanceof Error ? reason : new Error(String(reason));
365
+ logger.error(`Unexpected error during evaluation: ${error.message}`);
366
+ return error;
367
+ }
368
+
304
369
  export async function resumeEvaluation({
305
370
  client: _client,
306
371
  experimentId,
@@ -425,12 +490,13 @@ export async function resumeEvaluation({
425
490
  const batchIncomplete = res.data?.data;
426
491
  invariant(batchIncomplete, "Failed to fetch incomplete evaluations");
427
492
 
428
- if (batchIncomplete.length === 0) {
429
- if (totalProcessed === 0) {
430
- logger.info(
431
- `${PROGRESS_PREFIX.completed}No incomplete evaluations found.`
432
- );
433
- }
493
+ if (
494
+ isEmptyEvaluationBatch({
495
+ batchLength: batchIncomplete.length,
496
+ totalProcessed,
497
+ logger,
498
+ })
499
+ ) {
434
500
  break;
435
501
  }
436
502
 
@@ -468,7 +534,7 @@ export async function resumeEvaluation({
468
534
  logger.debug(
469
535
  `${PROGRESS_PREFIX.progress}Fetched batch of ${batchCount} evaluation tasks.`
470
536
  );
471
- } while (cursor !== null && !signal.aborted);
537
+ } while (shouldContinueEvaluationFetch({ cursor, signal }));
472
538
  } catch (error) {
473
539
  // Re-throw with context preservation
474
540
  if (error instanceof EvaluationFetchError) {
@@ -547,47 +613,11 @@ export async function resumeEvaluation({
547
613
  )
548
614
  .map((result) => result.reason);
549
615
 
550
- if (rejections.length > 0) {
551
- // Classify and handle errors based on their nature. When multiple tasks
552
- // reject, prefer the most meaningful error over incidental fallout
553
- // (e.g. a ChannelError raised in a blocked worker when the channel
554
- // closes on abort).
555
- const fetchError = rejections.find(
556
- (reason) => reason instanceof EvaluationFetchError
557
- );
558
- const workerError = rejections.find(
559
- (reason) =>
560
- reason instanceof Error &&
561
- !(reason instanceof EvaluationFetchError) &&
562
- !(reason instanceof ChannelError)
563
- );
564
- const channelError = rejections.find(
565
- (reason) => reason instanceof ChannelError
566
- );
567
-
568
- if (fetchError) {
569
- // Producer failed - this is ALWAYS critical regardless of stopOnFirstError
570
- logger.error(`Critical: Failed to fetch evaluations from server`);
571
- executionError = fetchError;
572
- } else if (workerError) {
573
- // Worker error in stopOnFirstError mode - already logged by worker
574
- executionError = workerError;
575
- } else if (channelError && signal.aborted) {
576
- // Channel closed due to intentional abort - wrap in semantic error
577
- executionError = new EvaluationAbortedError(
578
- "Evaluation stopped due to error in concurrent evaluator",
579
- channelError
580
- );
581
- } else {
582
- // Unexpected error (not from worker, not from producer fetch)
583
- // This could be a bug in our code or infrastructure failure
584
- const reason = rejections[0];
585
- const err =
586
- reason instanceof Error ? reason : new Error(String(reason));
587
- logger.error(`Unexpected error during evaluation: ${err.message}`);
588
- executionError = err;
589
- }
590
- }
616
+ executionError = getEvaluationExecutionError({
617
+ rejections,
618
+ isAborted: signal.aborted,
619
+ logger,
620
+ });
591
621
  } finally {
592
622
  // Ensure channel is closed even if there are unexpected errors
593
623
  // This is a safety net in case producer's finally block didn't execute
@@ -276,6 +276,67 @@ function setupTracer({
276
276
  * });
277
277
  * ```
278
278
  */
279
+ function getResumeEvaluators({
280
+ evaluators,
281
+ executionError,
282
+ }: {
283
+ evaluators: readonly ExperimentEvaluatorLike[] | undefined;
284
+ executionError: Error | null;
285
+ }): readonly ExperimentEvaluatorLike[] | null {
286
+ return evaluators && evaluators.length > 0 && !executionError
287
+ ? evaluators
288
+ : null;
289
+ }
290
+
291
+ function shouldWarnAboutFailedRuns({
292
+ totalFailed,
293
+ executionError,
294
+ }: {
295
+ totalFailed: number;
296
+ executionError: Error | null;
297
+ }): boolean {
298
+ return totalFailed > 0 && !executionError;
299
+ }
300
+
301
+ function getTaskExecutionError({
302
+ rejections,
303
+ isAborted,
304
+ logger,
305
+ }: {
306
+ rejections: unknown[];
307
+ isAborted: boolean;
308
+ logger: Logger;
309
+ }): Error | null {
310
+ if (rejections.length === 0) return null;
311
+ const fetchError = rejections.find(
312
+ (reason) => reason instanceof TaskFetchError
313
+ );
314
+ if (fetchError instanceof Error) {
315
+ logger.error(`Critical: Failed to fetch incomplete runs from server`);
316
+ return fetchError;
317
+ }
318
+ const taskError = rejections.find(
319
+ (reason) =>
320
+ reason instanceof Error &&
321
+ !(reason instanceof TaskFetchError) &&
322
+ !(reason instanceof ChannelError)
323
+ );
324
+ if (taskError instanceof Error) return taskError;
325
+ const channelError = rejections.find(
326
+ (reason) => reason instanceof ChannelError
327
+ );
328
+ if (channelError instanceof Error && isAborted) {
329
+ return new TaskAbortedError(
330
+ "Task execution stopped due to error in concurrent worker",
331
+ channelError
332
+ );
333
+ }
334
+ const reason = rejections[0];
335
+ const error = reason instanceof Error ? reason : new Error(String(reason));
336
+ logger.error(`Unexpected error during task execution: ${error.message}`);
337
+ return error;
338
+ }
339
+
279
340
  export async function resumeExperiment({
280
341
  client: _client,
281
342
  experimentId,
@@ -514,49 +575,11 @@ export async function resumeExperiment({
514
575
  )
515
576
  .map((result) => result.reason);
516
577
 
517
- if (rejections.length > 0) {
518
- // Classify and handle errors based on their nature. When multiple tasks
519
- // reject, prefer the most meaningful error over incidental fallout
520
- // (e.g. a ChannelError raised in a blocked worker when the channel
521
- // closes on abort).
522
- const fetchError = rejections.find(
523
- (reason) => reason instanceof TaskFetchError
524
- );
525
- const taskError = rejections.find(
526
- (reason) =>
527
- reason instanceof Error &&
528
- !(reason instanceof TaskFetchError) &&
529
- !(reason instanceof ChannelError)
530
- );
531
- const channelError = rejections.find(
532
- (reason) => reason instanceof ChannelError
533
- );
534
-
535
- if (fetchError) {
536
- // Producer failed - this is ALWAYS critical regardless of stopOnFirstError
537
- logger.error(`Critical: Failed to fetch incomplete runs from server`);
538
- executionError = fetchError;
539
- } else if (taskError) {
540
- // Worker error in stopOnFirstError mode - already logged by worker
541
- executionError = taskError;
542
- } else if (channelError && signal.aborted) {
543
- // Channel closed due to intentional abort - wrap in semantic error
544
- executionError = new TaskAbortedError(
545
- "Task execution stopped due to error in concurrent worker",
546
- channelError
547
- );
548
- } else {
549
- // Unexpected error (not from worker, not from producer fetch)
550
- // This could be a bug in our code or infrastructure failure
551
- const reason = rejections[0];
552
- const err =
553
- reason instanceof Error ? reason : new Error(String(reason));
554
- logger.error(
555
- `Unexpected error during task execution: ${err.message}`
556
- );
557
- executionError = err;
558
- }
559
- }
578
+ executionError = getTaskExecutionError({
579
+ rejections,
580
+ isAborted: signal.aborted,
581
+ logger,
582
+ });
560
583
  } finally {
561
584
  // Ensure channel is closed even if there are unexpected errors
562
585
  // This is a safety net in case producer's finally block didn't execute
@@ -570,11 +593,15 @@ export async function resumeExperiment({
570
593
  logger.info(`${PROGRESS_PREFIX.completed}Task runs completed.`);
571
594
  }
572
595
 
573
- if (totalFailed > 0 && !executionError) {
596
+ if (shouldWarnAboutFailedRuns({ totalFailed, executionError })) {
574
597
  logger.warn(`${totalFailed} out of ${totalProcessed} runs failed.`);
575
598
  }
576
599
 
577
- if (evaluators && evaluators.length > 0 && !executionError) {
600
+ const resumeEvaluators = getResumeEvaluators({
601
+ evaluators,
602
+ executionError,
603
+ });
604
+ if (resumeEvaluators) {
578
605
  await cleanupOwnedTracerProvider({
579
606
  provider,
580
607
  globalRegistration,
@@ -585,7 +612,7 @@ export async function resumeExperiment({
585
612
  logger.info(`${PROGRESS_PREFIX.start}Running evaluators.`);
586
613
  await resumeEvaluation({
587
614
  experimentId,
588
- evaluators: [...evaluators],
615
+ evaluators: [...resumeEvaluators],
589
616
  client,
590
617
  logger,
591
618
  concurrency,