@arizeai/phoenix-client 7.7.0 → 7.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/dist/esm/__generated__/api/v1.d.ts +127 -6
- package/dist/esm/__generated__/api/v1.d.ts.map +1 -1
- package/dist/esm/experiments/resumeEvaluation.d.ts +0 -60
- package/dist/esm/experiments/resumeEvaluation.d.ts.map +1 -1
- package/dist/esm/experiments/resumeEvaluation.js +44 -37
- package/dist/esm/experiments/resumeEvaluation.js.map +1 -1
- package/dist/esm/experiments/resumeExperiment.d.ts +0 -52
- package/dist/esm/experiments/resumeExperiment.d.ts.map +1 -1
- package/dist/esm/experiments/resumeExperiment.js +42 -35
- package/dist/esm/experiments/resumeExperiment.js.map +1 -1
- package/dist/esm/experiments/runExperiment.d.ts.map +1 -1
- package/dist/esm/experiments/runExperiment.js +152 -123
- package/dist/esm/experiments/runExperiment.js.map +1 -1
- package/dist/esm/prompts/constants.d.ts.map +1 -1
- package/dist/esm/prompts/constants.js +1 -0
- package/dist/esm/prompts/constants.js.map +1 -1
- package/dist/esm/prompts/sdks/toOpenAI.d.ts.map +1 -1
- package/dist/esm/prompts/sdks/toOpenAI.js +51 -58
- package/dist/esm/prompts/sdks/toOpenAI.js.map +1 -1
- package/dist/esm/sessions/sessionUtils.d.ts.map +1 -1
- package/dist/esm/sessions/sessionUtils.js +3 -0
- package/dist/esm/sessions/sessionUtils.js.map +1 -1
- package/dist/esm/spans/getSpans.d.ts +0 -70
- package/dist/esm/spans/getSpans.d.ts.map +1 -1
- package/dist/esm/spans/getSpans.js +42 -93
- package/dist/esm/spans/getSpans.js.map +1 -1
- package/dist/esm/testing/phoenix-test-tracking.d.ts.map +1 -1
- package/dist/esm/testing/phoenix-test-tracking.js +109 -78
- package/dist/esm/testing/phoenix-test-tracking.js.map +1 -1
- package/dist/esm/tsconfig.esm.tsbuildinfo +1 -1
- package/dist/esm/types/prompts.d.ts +1 -1
- package/dist/esm/types/prompts.d.ts.map +1 -1
- package/dist/esm/types/prompts.js.map +1 -1
- package/dist/esm/types/sessions.d.ts +6 -0
- package/dist/esm/types/sessions.d.ts.map +1 -1
- package/dist/esm/utils/getPromptBySelector.d.ts +1 -1
- package/dist/src/__generated__/api/v1.d.ts +127 -6
- package/dist/src/__generated__/api/v1.d.ts.map +1 -1
- package/dist/src/experiments/resumeEvaluation.d.ts +0 -60
- package/dist/src/experiments/resumeEvaluation.d.ts.map +1 -1
- package/dist/src/experiments/resumeEvaluation.js +44 -37
- package/dist/src/experiments/resumeEvaluation.js.map +1 -1
- package/dist/src/experiments/resumeExperiment.d.ts +0 -52
- package/dist/src/experiments/resumeExperiment.d.ts.map +1 -1
- package/dist/src/experiments/resumeExperiment.js +42 -35
- package/dist/src/experiments/resumeExperiment.js.map +1 -1
- package/dist/src/experiments/runExperiment.d.ts.map +1 -1
- package/dist/src/experiments/runExperiment.js +148 -116
- package/dist/src/experiments/runExperiment.js.map +1 -1
- package/dist/src/prompts/constants.d.ts.map +1 -1
- package/dist/src/prompts/constants.js +1 -0
- package/dist/src/prompts/constants.js.map +1 -1
- package/dist/src/prompts/sdks/toOpenAI.d.ts.map +1 -1
- package/dist/src/prompts/sdks/toOpenAI.js +58 -63
- package/dist/src/prompts/sdks/toOpenAI.js.map +1 -1
- package/dist/src/sessions/sessionUtils.d.ts.map +1 -1
- package/dist/src/sessions/sessionUtils.js +3 -0
- package/dist/src/sessions/sessionUtils.js.map +1 -1
- package/dist/src/spans/getSpans.d.ts +0 -70
- package/dist/src/spans/getSpans.d.ts.map +1 -1
- package/dist/src/spans/getSpans.js +43 -94
- package/dist/src/spans/getSpans.js.map +1 -1
- package/dist/src/testing/phoenix-test-tracking.d.ts.map +1 -1
- package/dist/src/testing/phoenix-test-tracking.js +117 -84
- package/dist/src/testing/phoenix-test-tracking.js.map +1 -1
- package/dist/src/types/prompts.d.ts +1 -1
- package/dist/src/types/prompts.d.ts.map +1 -1
- package/dist/src/types/prompts.js.map +1 -1
- package/dist/src/types/sessions.d.ts +6 -0
- package/dist/src/types/sessions.d.ts.map +1 -1
- package/dist/src/utils/getPromptBySelector.d.ts +1 -1
- package/dist/tsconfig.tsbuildinfo +1 -1
- package/docs/sessions.mdx +10 -1
- package/package.json +1 -1
- package/src/__generated__/api/v1.ts +127 -6
- package/src/experiments/resumeEvaluation.ts +78 -48
- package/src/experiments/resumeExperiment.ts +73 -46
- package/src/experiments/runExperiment.ts +235 -129
- package/src/prompts/constants.ts +1 -0
- package/src/prompts/sdks/toOpenAI.ts +60 -61
- package/src/sessions/sessionUtils.ts +3 -0
- package/src/spans/getSpans.ts +84 -48
- package/src/testing/phoenix-test-tracking.ts +154 -90
- package/src/types/prompts.ts +2 -1
- package/src/types/sessions.ts +6 -0
package/docs/sessions.mdx
CHANGED
|
@@ -29,10 +29,19 @@ const sessions = await listSessions({
|
|
|
29
29
|
});
|
|
30
30
|
|
|
31
31
|
for (const session of sessions) {
|
|
32
|
-
console.log(
|
|
32
|
+
console.log({
|
|
33
|
+
sessionId: session.sessionId,
|
|
34
|
+
promptTokens: session.tokenCountPrompt,
|
|
35
|
+
completionTokens: session.tokenCountCompletion,
|
|
36
|
+
totalTokens: session.tokenCountTotal,
|
|
37
|
+
});
|
|
33
38
|
}
|
|
34
39
|
```
|
|
35
40
|
|
|
41
|
+
Each session includes cumulative prompt, completion, and total token counts across
|
|
42
|
+
all of its spans. The fields may be `undefined` when using a Phoenix server that
|
|
43
|
+
does not return session token usage.
|
|
44
|
+
|
|
36
45
|
## Retrieve A Session And Its Turns
|
|
37
46
|
|
|
38
47
|
```ts
|
package/package.json
CHANGED
|
@@ -761,7 +761,11 @@ export interface paths {
|
|
|
761
761
|
get: operations["listProjectTraces"];
|
|
762
762
|
put?: never;
|
|
763
763
|
post?: never;
|
|
764
|
-
|
|
764
|
+
/**
|
|
765
|
+
* Delete traces from a project
|
|
766
|
+
* @description Delete traces from a project without deleting the project or its configuration. Only traces whose start time is within the required `[start_time, end_time)` interval are deleted. Associated spans are cascade deleted, and project sessions left with no remaining traces are also deleted. Naive datetimes are interpreted as UTC.
|
|
767
|
+
*/
|
|
768
|
+
delete: operations["deleteProjectTraces"];
|
|
765
769
|
options?: never;
|
|
766
770
|
head?: never;
|
|
767
771
|
patch?: never;
|
|
@@ -1568,7 +1572,7 @@ export interface paths {
|
|
|
1568
1572
|
put?: never;
|
|
1569
1573
|
/**
|
|
1570
1574
|
* OpenAI-compatible chat completions
|
|
1571
|
-
* @description Creates a chat completion using the OpenAI wire format, proxying to the selected provider with credentials resolved on the server (secret store first, environment second) — callers never handle provider API keys. Model must be '{provider}:{model_name}' for a built-in provider (one of anthropic, aws, azure_openai, cerebras, deepseek, fireworks, google, groq, moonshot, ollama, openai, perplexity, together, xai) or 'custom:{provider_id}:{model_name}' for a stored custom provider, e.g. 'openai:gpt-4o' or 'anthropic:claude-sonnet-4-5'. Set `stream: true` for server-sent events of `chat.completion.chunk` payloads terminated by `data: [DONE]`. Tool calling is not supported.
|
|
1575
|
+
* @description Creates a chat completion using the OpenAI wire format, proxying to the selected provider with credentials resolved on the server (secret store first, environment second) — callers never handle provider API keys. Model must be '{provider}:{model_name}' for a built-in provider (one of anthropic, aws, azure_openai, cerebras, deepseek, fireworks, google, groq, moonshot, ollama, openai, perplexity, together, xai, zai) or 'custom:{provider_id}:{model_name}' for a stored custom provider, e.g. 'openai:gpt-4o' or 'anthropic:claude-sonnet-4-5'. Set `stream: true` for server-sent events of `chat.completion.chunk` payloads terminated by `data: [DONE]`. Tool calling is not supported.
|
|
1572
1576
|
*
|
|
1573
1577
|
* **Phoenix is not an AI gateway.** The same server also takes on trace ingestion traffic, so routing production LLM calls through it competes with ingestion. Use this endpoint only to quickly try out different models in non-production environments.
|
|
1574
1578
|
*/
|
|
@@ -2424,6 +2428,11 @@ export interface components {
|
|
|
2424
2428
|
* @description The id of the last transcript message the client has rendered, used for optimistic concurrency. Omit when the session has no messages; required (and validated against the persisted transcript) once it does. On mismatch the server rejects the send with HTTP 409 and code ``agent_session_messages_stale`` — the client should refetch the session before retrying.
|
|
2425
2429
|
*/
|
|
2426
2430
|
lastMessageId?: string | null;
|
|
2431
|
+
/**
|
|
2432
|
+
* Credentials
|
|
2433
|
+
* @description Client-held credentials for optional integrations (e.g. the user's own GitHub personal access token under the key ``GITHUB_PERSONAL_ACCESS_TOKEN``), used only for the duration of the turn and never persisted. Unknown keys are rejected.
|
|
2434
|
+
*/
|
|
2435
|
+
credentials?: components["schemas"]["ChatRequestCredential"][];
|
|
2427
2436
|
/**
|
|
2428
2437
|
* Recordlocaltraces
|
|
2429
2438
|
* @default false
|
|
@@ -2441,6 +2450,28 @@ export interface components {
|
|
|
2441
2450
|
*/
|
|
2442
2451
|
instrumentUserId?: boolean;
|
|
2443
2452
|
};
|
|
2453
|
+
/**
|
|
2454
|
+
* ChatRequestCredential
|
|
2455
|
+
* @description One client-held credential riding the request for the duration of a turn.
|
|
2456
|
+
*
|
|
2457
|
+
* The value is ephemeral: it is injected server-side as transport auth for
|
|
2458
|
+
* the matching integration and is never persisted, traced, or echoed. It is
|
|
2459
|
+
* top-level on the request body — never part of the message — so it cannot
|
|
2460
|
+
* reach the session transcript.
|
|
2461
|
+
*/
|
|
2462
|
+
ChatRequestCredential: {
|
|
2463
|
+
/**
|
|
2464
|
+
* Key
|
|
2465
|
+
* @description The credential's secret-key name.
|
|
2466
|
+
* @constant
|
|
2467
|
+
*/
|
|
2468
|
+
key: "GITHUB_PERSONAL_ACCESS_TOKEN";
|
|
2469
|
+
/**
|
|
2470
|
+
* Value
|
|
2471
|
+
* Format: password
|
|
2472
|
+
*/
|
|
2473
|
+
value: string;
|
|
2474
|
+
};
|
|
2444
2475
|
/** CodeEvaluatorUIContext */
|
|
2445
2476
|
CodeEvaluatorUIContext: {
|
|
2446
2477
|
/**
|
|
@@ -2561,7 +2592,7 @@ export interface components {
|
|
|
2561
2592
|
CreateChatCompletionRequestBody: {
|
|
2562
2593
|
/**
|
|
2563
2594
|
* Model
|
|
2564
|
-
* @description Model must be '{provider}:{model_name}' for a built-in provider (one of anthropic, aws, azure_openai, cerebras, deepseek, fireworks, google, groq, moonshot, ollama, openai, perplexity, together, xai) or 'custom:{provider_id}:{model_name}' for a stored custom provider, e.g. 'openai:gpt-4o' or 'anthropic:claude-sonnet-4-5'.
|
|
2595
|
+
* @description Model must be '{provider}:{model_name}' for a built-in provider (one of anthropic, aws, azure_openai, cerebras, deepseek, fireworks, google, groq, moonshot, ollama, openai, perplexity, together, xai, zai) or 'custom:{provider_id}:{model_name}' for a stored custom provider, e.g. 'openai:gpt-4o' or 'anthropic:claude-sonnet-4-5'.
|
|
2565
2596
|
*/
|
|
2566
2597
|
model: string;
|
|
2567
2598
|
/** Messages */
|
|
@@ -4057,7 +4088,7 @@ export interface components {
|
|
|
4057
4088
|
* ModelProvider
|
|
4058
4089
|
* @enum {string}
|
|
4059
4090
|
*/
|
|
4060
|
-
ModelProvider: "OPENAI" | "AZURE_OPENAI" | "ANTHROPIC" | "GOOGLE" | "DEEPSEEK" | "XAI" | "OLLAMA" | "AWS" | "CEREBRAS" | "FIREWORKS" | "GROQ" | "MOONSHOT" | "PERPLEXITY" | "TOGETHER";
|
|
4091
|
+
ModelProvider: "OPENAI" | "AZURE_OPENAI" | "ANTHROPIC" | "GOOGLE" | "DEEPSEEK" | "XAI" | "OLLAMA" | "AWS" | "CEREBRAS" | "FIREWORKS" | "GROQ" | "MOONSHOT" | "PERPLEXITY" | "TOGETHER" | "ZAI";
|
|
4061
4092
|
/** OAuth2User */
|
|
4062
4093
|
OAuth2User: {
|
|
4063
4094
|
/** Id */
|
|
@@ -5236,7 +5267,7 @@ export interface components {
|
|
|
5236
5267
|
template_type: components["schemas"]["PromptTemplateType"];
|
|
5237
5268
|
template_format: components["schemas"]["PromptTemplateFormat"];
|
|
5238
5269
|
/** Invocation Parameters */
|
|
5239
|
-
invocation_parameters: components["schemas"]["PromptOpenAIInvocationParameters"] | components["schemas"]["PromptAzureOpenAIInvocationParameters"] | components["schemas"]["PromptAnthropicInvocationParameters"] | components["schemas"]["PromptGoogleInvocationParameters"] | components["schemas"]["PromptDeepSeekInvocationParameters"] | components["schemas"]["PromptXAIInvocationParameters"] | components["schemas"]["PromptOllamaInvocationParameters"] | components["schemas"]["PromptAwsInvocationParameters"] | components["schemas"]["PromptCerebrasInvocationParameters"] | components["schemas"]["PromptFireworksInvocationParameters"] | components["schemas"]["PromptGroqInvocationParameters"] | components["schemas"]["PromptMoonshotInvocationParameters"] | components["schemas"]["PromptPerplexityInvocationParameters"] | components["schemas"]["PromptTogetherInvocationParameters"];
|
|
5270
|
+
invocation_parameters: components["schemas"]["PromptOpenAIInvocationParameters"] | components["schemas"]["PromptAzureOpenAIInvocationParameters"] | components["schemas"]["PromptAnthropicInvocationParameters"] | components["schemas"]["PromptGoogleInvocationParameters"] | components["schemas"]["PromptDeepSeekInvocationParameters"] | components["schemas"]["PromptXAIInvocationParameters"] | components["schemas"]["PromptOllamaInvocationParameters"] | components["schemas"]["PromptAwsInvocationParameters"] | components["schemas"]["PromptCerebrasInvocationParameters"] | components["schemas"]["PromptFireworksInvocationParameters"] | components["schemas"]["PromptGroqInvocationParameters"] | components["schemas"]["PromptMoonshotInvocationParameters"] | components["schemas"]["PromptPerplexityInvocationParameters"] | components["schemas"]["PromptTogetherInvocationParameters"] | components["schemas"]["PromptZAIInvocationParameters"];
|
|
5240
5271
|
tools?: components["schemas"]["PromptTools"] | null;
|
|
5241
5272
|
/** Response Format */
|
|
5242
5273
|
response_format?: components["schemas"]["PromptResponseFormatJSONSchema"] | null;
|
|
@@ -5255,7 +5286,7 @@ export interface components {
|
|
|
5255
5286
|
template_type: components["schemas"]["PromptTemplateType"];
|
|
5256
5287
|
template_format: components["schemas"]["PromptTemplateFormat"];
|
|
5257
5288
|
/** Invocation Parameters */
|
|
5258
|
-
invocation_parameters: components["schemas"]["PromptOpenAIInvocationParameters"] | components["schemas"]["PromptAzureOpenAIInvocationParameters"] | components["schemas"]["PromptAnthropicInvocationParameters"] | components["schemas"]["PromptGoogleInvocationParameters"] | components["schemas"]["PromptDeepSeekInvocationParameters"] | components["schemas"]["PromptXAIInvocationParameters"] | components["schemas"]["PromptOllamaInvocationParameters"] | components["schemas"]["PromptAwsInvocationParameters"] | components["schemas"]["PromptCerebrasInvocationParameters"] | components["schemas"]["PromptFireworksInvocationParameters"] | components["schemas"]["PromptGroqInvocationParameters"] | components["schemas"]["PromptMoonshotInvocationParameters"] | components["schemas"]["PromptPerplexityInvocationParameters"] | components["schemas"]["PromptTogetherInvocationParameters"];
|
|
5289
|
+
invocation_parameters: components["schemas"]["PromptOpenAIInvocationParameters"] | components["schemas"]["PromptAzureOpenAIInvocationParameters"] | components["schemas"]["PromptAnthropicInvocationParameters"] | components["schemas"]["PromptGoogleInvocationParameters"] | components["schemas"]["PromptDeepSeekInvocationParameters"] | components["schemas"]["PromptXAIInvocationParameters"] | components["schemas"]["PromptOllamaInvocationParameters"] | components["schemas"]["PromptAwsInvocationParameters"] | components["schemas"]["PromptCerebrasInvocationParameters"] | components["schemas"]["PromptFireworksInvocationParameters"] | components["schemas"]["PromptGroqInvocationParameters"] | components["schemas"]["PromptMoonshotInvocationParameters"] | components["schemas"]["PromptPerplexityInvocationParameters"] | components["schemas"]["PromptTogetherInvocationParameters"] | components["schemas"]["PromptZAIInvocationParameters"];
|
|
5259
5290
|
tools?: components["schemas"]["PromptTools"] | null;
|
|
5260
5291
|
/** Response Format */
|
|
5261
5292
|
response_format?: components["schemas"]["PromptResponseFormatJSONSchema"] | null;
|
|
@@ -5323,6 +5354,43 @@ export interface components {
|
|
|
5323
5354
|
[key: string]: unknown;
|
|
5324
5355
|
};
|
|
5325
5356
|
};
|
|
5357
|
+
/** PromptZAIInvocationParameters */
|
|
5358
|
+
PromptZAIInvocationParameters: {
|
|
5359
|
+
/**
|
|
5360
|
+
* @description discriminator enum property added by openapi-typescript
|
|
5361
|
+
* @enum {string}
|
|
5362
|
+
*/
|
|
5363
|
+
type: "zai";
|
|
5364
|
+
zai: components["schemas"]["PromptZAIInvocationParametersContent"];
|
|
5365
|
+
};
|
|
5366
|
+
/** PromptZAIInvocationParametersContent */
|
|
5367
|
+
PromptZAIInvocationParametersContent: {
|
|
5368
|
+
/** Temperature */
|
|
5369
|
+
temperature?: number;
|
|
5370
|
+
/** Max Tokens */
|
|
5371
|
+
max_tokens?: number;
|
|
5372
|
+
/** Max Completion Tokens */
|
|
5373
|
+
max_completion_tokens?: number;
|
|
5374
|
+
/** Frequency Penalty */
|
|
5375
|
+
frequency_penalty?: number;
|
|
5376
|
+
/** Presence Penalty */
|
|
5377
|
+
presence_penalty?: number;
|
|
5378
|
+
/** Top P */
|
|
5379
|
+
top_p?: number;
|
|
5380
|
+
/** Seed */
|
|
5381
|
+
seed?: number;
|
|
5382
|
+
/** Stop */
|
|
5383
|
+
stop?: string[];
|
|
5384
|
+
/**
|
|
5385
|
+
* Reasoning Effort
|
|
5386
|
+
* @enum {string}
|
|
5387
|
+
*/
|
|
5388
|
+
reasoning_effort?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh";
|
|
5389
|
+
/** Extra Body */
|
|
5390
|
+
extra_body?: {
|
|
5391
|
+
[key: string]: unknown;
|
|
5392
|
+
};
|
|
5393
|
+
};
|
|
5326
5394
|
/**
|
|
5327
5395
|
* PydanticAIMessageMetadata
|
|
5328
5396
|
* @description Local pin of pydantic-ai's message-level ``pydantic_ai`` metadata
|
|
@@ -10041,6 +10109,59 @@ export interface operations {
|
|
|
10041
10109
|
};
|
|
10042
10110
|
};
|
|
10043
10111
|
};
|
|
10112
|
+
deleteProjectTraces: {
|
|
10113
|
+
parameters: {
|
|
10114
|
+
query: {
|
|
10115
|
+
/** @description Required inclusive lower bound on trace start time (ISO 8601). */
|
|
10116
|
+
start_time: string;
|
|
10117
|
+
/** @description Required exclusive upper bound on trace start time (ISO 8601). */
|
|
10118
|
+
end_time: string;
|
|
10119
|
+
};
|
|
10120
|
+
header?: never;
|
|
10121
|
+
path: {
|
|
10122
|
+
/** @description The project identifier: either project ID or project name. */
|
|
10123
|
+
project_identifier: string;
|
|
10124
|
+
};
|
|
10125
|
+
cookie?: never;
|
|
10126
|
+
};
|
|
10127
|
+
requestBody?: never;
|
|
10128
|
+
responses: {
|
|
10129
|
+
/** @description No content returned after the matching traces are deleted */
|
|
10130
|
+
204: {
|
|
10131
|
+
headers: {
|
|
10132
|
+
[name: string]: unknown;
|
|
10133
|
+
};
|
|
10134
|
+
content?: never;
|
|
10135
|
+
};
|
|
10136
|
+
/** @description Forbidden */
|
|
10137
|
+
403: {
|
|
10138
|
+
headers: {
|
|
10139
|
+
[name: string]: unknown;
|
|
10140
|
+
};
|
|
10141
|
+
content: {
|
|
10142
|
+
"text/plain": string;
|
|
10143
|
+
};
|
|
10144
|
+
};
|
|
10145
|
+
/** @description Not Found */
|
|
10146
|
+
404: {
|
|
10147
|
+
headers: {
|
|
10148
|
+
[name: string]: unknown;
|
|
10149
|
+
};
|
|
10150
|
+
content: {
|
|
10151
|
+
"text/plain": string;
|
|
10152
|
+
};
|
|
10153
|
+
};
|
|
10154
|
+
/** @description Unprocessable Entity */
|
|
10155
|
+
422: {
|
|
10156
|
+
headers: {
|
|
10157
|
+
[name: string]: unknown;
|
|
10158
|
+
};
|
|
10159
|
+
content: {
|
|
10160
|
+
"text/plain": string;
|
|
10161
|
+
};
|
|
10162
|
+
};
|
|
10163
|
+
};
|
|
10164
|
+
};
|
|
10044
10165
|
annotateTraces: {
|
|
10045
10166
|
parameters: {
|
|
10046
10167
|
query?: {
|
|
@@ -301,6 +301,71 @@ function setupEvaluationTracer({
|
|
|
301
301
|
* });
|
|
302
302
|
* ```
|
|
303
303
|
*/
|
|
304
|
+
function isEmptyEvaluationBatch({
|
|
305
|
+
batchLength,
|
|
306
|
+
totalProcessed,
|
|
307
|
+
logger,
|
|
308
|
+
}: {
|
|
309
|
+
batchLength: number;
|
|
310
|
+
totalProcessed: number;
|
|
311
|
+
logger: Logger;
|
|
312
|
+
}): boolean {
|
|
313
|
+
if (batchLength > 0) return false;
|
|
314
|
+
if (totalProcessed === 0) {
|
|
315
|
+
logger.info(`${PROGRESS_PREFIX.completed}No incomplete evaluations found.`);
|
|
316
|
+
}
|
|
317
|
+
return true;
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
function shouldContinueEvaluationFetch({
|
|
321
|
+
cursor,
|
|
322
|
+
signal,
|
|
323
|
+
}: {
|
|
324
|
+
cursor: string | null;
|
|
325
|
+
signal: AbortSignal;
|
|
326
|
+
}): boolean {
|
|
327
|
+
return cursor !== null && !signal.aborted;
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
function getEvaluationExecutionError({
|
|
331
|
+
rejections,
|
|
332
|
+
isAborted,
|
|
333
|
+
logger,
|
|
334
|
+
}: {
|
|
335
|
+
rejections: unknown[];
|
|
336
|
+
isAborted: boolean;
|
|
337
|
+
logger: Logger;
|
|
338
|
+
}): Error | null {
|
|
339
|
+
if (rejections.length === 0) return null;
|
|
340
|
+
const fetchError = rejections.find(
|
|
341
|
+
(reason) => reason instanceof EvaluationFetchError
|
|
342
|
+
);
|
|
343
|
+
if (fetchError instanceof Error) {
|
|
344
|
+
logger.error(`Critical: Failed to fetch evaluations from server`);
|
|
345
|
+
return fetchError;
|
|
346
|
+
}
|
|
347
|
+
const workerError = rejections.find(
|
|
348
|
+
(reason) =>
|
|
349
|
+
reason instanceof Error &&
|
|
350
|
+
!(reason instanceof EvaluationFetchError) &&
|
|
351
|
+
!(reason instanceof ChannelError)
|
|
352
|
+
);
|
|
353
|
+
if (workerError instanceof Error) return workerError;
|
|
354
|
+
const channelError = rejections.find(
|
|
355
|
+
(reason) => reason instanceof ChannelError
|
|
356
|
+
);
|
|
357
|
+
if (channelError instanceof Error && isAborted) {
|
|
358
|
+
return new EvaluationAbortedError(
|
|
359
|
+
"Evaluation stopped due to error in concurrent evaluator",
|
|
360
|
+
channelError
|
|
361
|
+
);
|
|
362
|
+
}
|
|
363
|
+
const reason = rejections[0];
|
|
364
|
+
const error = reason instanceof Error ? reason : new Error(String(reason));
|
|
365
|
+
logger.error(`Unexpected error during evaluation: ${error.message}`);
|
|
366
|
+
return error;
|
|
367
|
+
}
|
|
368
|
+
|
|
304
369
|
export async function resumeEvaluation({
|
|
305
370
|
client: _client,
|
|
306
371
|
experimentId,
|
|
@@ -425,12 +490,13 @@ export async function resumeEvaluation({
|
|
|
425
490
|
const batchIncomplete = res.data?.data;
|
|
426
491
|
invariant(batchIncomplete, "Failed to fetch incomplete evaluations");
|
|
427
492
|
|
|
428
|
-
if (
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
}
|
|
493
|
+
if (
|
|
494
|
+
isEmptyEvaluationBatch({
|
|
495
|
+
batchLength: batchIncomplete.length,
|
|
496
|
+
totalProcessed,
|
|
497
|
+
logger,
|
|
498
|
+
})
|
|
499
|
+
) {
|
|
434
500
|
break;
|
|
435
501
|
}
|
|
436
502
|
|
|
@@ -468,7 +534,7 @@ export async function resumeEvaluation({
|
|
|
468
534
|
logger.debug(
|
|
469
535
|
`${PROGRESS_PREFIX.progress}Fetched batch of ${batchCount} evaluation tasks.`
|
|
470
536
|
);
|
|
471
|
-
} while (cursor
|
|
537
|
+
} while (shouldContinueEvaluationFetch({ cursor, signal }));
|
|
472
538
|
} catch (error) {
|
|
473
539
|
// Re-throw with context preservation
|
|
474
540
|
if (error instanceof EvaluationFetchError) {
|
|
@@ -547,47 +613,11 @@ export async function resumeEvaluation({
|
|
|
547
613
|
)
|
|
548
614
|
.map((result) => result.reason);
|
|
549
615
|
|
|
550
|
-
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
const fetchError = rejections.find(
|
|
556
|
-
(reason) => reason instanceof EvaluationFetchError
|
|
557
|
-
);
|
|
558
|
-
const workerError = rejections.find(
|
|
559
|
-
(reason) =>
|
|
560
|
-
reason instanceof Error &&
|
|
561
|
-
!(reason instanceof EvaluationFetchError) &&
|
|
562
|
-
!(reason instanceof ChannelError)
|
|
563
|
-
);
|
|
564
|
-
const channelError = rejections.find(
|
|
565
|
-
(reason) => reason instanceof ChannelError
|
|
566
|
-
);
|
|
567
|
-
|
|
568
|
-
if (fetchError) {
|
|
569
|
-
// Producer failed - this is ALWAYS critical regardless of stopOnFirstError
|
|
570
|
-
logger.error(`Critical: Failed to fetch evaluations from server`);
|
|
571
|
-
executionError = fetchError;
|
|
572
|
-
} else if (workerError) {
|
|
573
|
-
// Worker error in stopOnFirstError mode - already logged by worker
|
|
574
|
-
executionError = workerError;
|
|
575
|
-
} else if (channelError && signal.aborted) {
|
|
576
|
-
// Channel closed due to intentional abort - wrap in semantic error
|
|
577
|
-
executionError = new EvaluationAbortedError(
|
|
578
|
-
"Evaluation stopped due to error in concurrent evaluator",
|
|
579
|
-
channelError
|
|
580
|
-
);
|
|
581
|
-
} else {
|
|
582
|
-
// Unexpected error (not from worker, not from producer fetch)
|
|
583
|
-
// This could be a bug in our code or infrastructure failure
|
|
584
|
-
const reason = rejections[0];
|
|
585
|
-
const err =
|
|
586
|
-
reason instanceof Error ? reason : new Error(String(reason));
|
|
587
|
-
logger.error(`Unexpected error during evaluation: ${err.message}`);
|
|
588
|
-
executionError = err;
|
|
589
|
-
}
|
|
590
|
-
}
|
|
616
|
+
executionError = getEvaluationExecutionError({
|
|
617
|
+
rejections,
|
|
618
|
+
isAborted: signal.aborted,
|
|
619
|
+
logger,
|
|
620
|
+
});
|
|
591
621
|
} finally {
|
|
592
622
|
// Ensure channel is closed even if there are unexpected errors
|
|
593
623
|
// This is a safety net in case producer's finally block didn't execute
|
|
@@ -276,6 +276,67 @@ function setupTracer({
|
|
|
276
276
|
* });
|
|
277
277
|
* ```
|
|
278
278
|
*/
|
|
279
|
+
function getResumeEvaluators({
|
|
280
|
+
evaluators,
|
|
281
|
+
executionError,
|
|
282
|
+
}: {
|
|
283
|
+
evaluators: readonly ExperimentEvaluatorLike[] | undefined;
|
|
284
|
+
executionError: Error | null;
|
|
285
|
+
}): readonly ExperimentEvaluatorLike[] | null {
|
|
286
|
+
return evaluators && evaluators.length > 0 && !executionError
|
|
287
|
+
? evaluators
|
|
288
|
+
: null;
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
function shouldWarnAboutFailedRuns({
|
|
292
|
+
totalFailed,
|
|
293
|
+
executionError,
|
|
294
|
+
}: {
|
|
295
|
+
totalFailed: number;
|
|
296
|
+
executionError: Error | null;
|
|
297
|
+
}): boolean {
|
|
298
|
+
return totalFailed > 0 && !executionError;
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
function getTaskExecutionError({
|
|
302
|
+
rejections,
|
|
303
|
+
isAborted,
|
|
304
|
+
logger,
|
|
305
|
+
}: {
|
|
306
|
+
rejections: unknown[];
|
|
307
|
+
isAborted: boolean;
|
|
308
|
+
logger: Logger;
|
|
309
|
+
}): Error | null {
|
|
310
|
+
if (rejections.length === 0) return null;
|
|
311
|
+
const fetchError = rejections.find(
|
|
312
|
+
(reason) => reason instanceof TaskFetchError
|
|
313
|
+
);
|
|
314
|
+
if (fetchError instanceof Error) {
|
|
315
|
+
logger.error(`Critical: Failed to fetch incomplete runs from server`);
|
|
316
|
+
return fetchError;
|
|
317
|
+
}
|
|
318
|
+
const taskError = rejections.find(
|
|
319
|
+
(reason) =>
|
|
320
|
+
reason instanceof Error &&
|
|
321
|
+
!(reason instanceof TaskFetchError) &&
|
|
322
|
+
!(reason instanceof ChannelError)
|
|
323
|
+
);
|
|
324
|
+
if (taskError instanceof Error) return taskError;
|
|
325
|
+
const channelError = rejections.find(
|
|
326
|
+
(reason) => reason instanceof ChannelError
|
|
327
|
+
);
|
|
328
|
+
if (channelError instanceof Error && isAborted) {
|
|
329
|
+
return new TaskAbortedError(
|
|
330
|
+
"Task execution stopped due to error in concurrent worker",
|
|
331
|
+
channelError
|
|
332
|
+
);
|
|
333
|
+
}
|
|
334
|
+
const reason = rejections[0];
|
|
335
|
+
const error = reason instanceof Error ? reason : new Error(String(reason));
|
|
336
|
+
logger.error(`Unexpected error during task execution: ${error.message}`);
|
|
337
|
+
return error;
|
|
338
|
+
}
|
|
339
|
+
|
|
279
340
|
export async function resumeExperiment({
|
|
280
341
|
client: _client,
|
|
281
342
|
experimentId,
|
|
@@ -514,49 +575,11 @@ export async function resumeExperiment({
|
|
|
514
575
|
)
|
|
515
576
|
.map((result) => result.reason);
|
|
516
577
|
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
const fetchError = rejections.find(
|
|
523
|
-
(reason) => reason instanceof TaskFetchError
|
|
524
|
-
);
|
|
525
|
-
const taskError = rejections.find(
|
|
526
|
-
(reason) =>
|
|
527
|
-
reason instanceof Error &&
|
|
528
|
-
!(reason instanceof TaskFetchError) &&
|
|
529
|
-
!(reason instanceof ChannelError)
|
|
530
|
-
);
|
|
531
|
-
const channelError = rejections.find(
|
|
532
|
-
(reason) => reason instanceof ChannelError
|
|
533
|
-
);
|
|
534
|
-
|
|
535
|
-
if (fetchError) {
|
|
536
|
-
// Producer failed - this is ALWAYS critical regardless of stopOnFirstError
|
|
537
|
-
logger.error(`Critical: Failed to fetch incomplete runs from server`);
|
|
538
|
-
executionError = fetchError;
|
|
539
|
-
} else if (taskError) {
|
|
540
|
-
// Worker error in stopOnFirstError mode - already logged by worker
|
|
541
|
-
executionError = taskError;
|
|
542
|
-
} else if (channelError && signal.aborted) {
|
|
543
|
-
// Channel closed due to intentional abort - wrap in semantic error
|
|
544
|
-
executionError = new TaskAbortedError(
|
|
545
|
-
"Task execution stopped due to error in concurrent worker",
|
|
546
|
-
channelError
|
|
547
|
-
);
|
|
548
|
-
} else {
|
|
549
|
-
// Unexpected error (not from worker, not from producer fetch)
|
|
550
|
-
// This could be a bug in our code or infrastructure failure
|
|
551
|
-
const reason = rejections[0];
|
|
552
|
-
const err =
|
|
553
|
-
reason instanceof Error ? reason : new Error(String(reason));
|
|
554
|
-
logger.error(
|
|
555
|
-
`Unexpected error during task execution: ${err.message}`
|
|
556
|
-
);
|
|
557
|
-
executionError = err;
|
|
558
|
-
}
|
|
559
|
-
}
|
|
578
|
+
executionError = getTaskExecutionError({
|
|
579
|
+
rejections,
|
|
580
|
+
isAborted: signal.aborted,
|
|
581
|
+
logger,
|
|
582
|
+
});
|
|
560
583
|
} finally {
|
|
561
584
|
// Ensure channel is closed even if there are unexpected errors
|
|
562
585
|
// This is a safety net in case producer's finally block didn't execute
|
|
@@ -570,11 +593,15 @@ export async function resumeExperiment({
|
|
|
570
593
|
logger.info(`${PROGRESS_PREFIX.completed}Task runs completed.`);
|
|
571
594
|
}
|
|
572
595
|
|
|
573
|
-
if (totalFailed
|
|
596
|
+
if (shouldWarnAboutFailedRuns({ totalFailed, executionError })) {
|
|
574
597
|
logger.warn(`${totalFailed} out of ${totalProcessed} runs failed.`);
|
|
575
598
|
}
|
|
576
599
|
|
|
577
|
-
|
|
600
|
+
const resumeEvaluators = getResumeEvaluators({
|
|
601
|
+
evaluators,
|
|
602
|
+
executionError,
|
|
603
|
+
});
|
|
604
|
+
if (resumeEvaluators) {
|
|
578
605
|
await cleanupOwnedTracerProvider({
|
|
579
606
|
provider,
|
|
580
607
|
globalRegistration,
|
|
@@ -585,7 +612,7 @@ export async function resumeExperiment({
|
|
|
585
612
|
logger.info(`${PROGRESS_PREFIX.start}Running evaluators.`);
|
|
586
613
|
await resumeEvaluation({
|
|
587
614
|
experimentId,
|
|
588
|
-
evaluators: [...
|
|
615
|
+
evaluators: [...resumeEvaluators],
|
|
589
616
|
client,
|
|
590
617
|
logger,
|
|
591
618
|
concurrency,
|