@anvia/core 1.0.0-rc.1 → 1.0.0-rc.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (151) hide show
  1. package/README.md +344 -85
  2. package/dist/agent/index.d.ts +33 -18
  3. package/dist/agent/index.js +42 -17
  4. package/dist/agent-CXScUOpX.d.ts +150 -0
  5. package/dist/{chunk-R2LJSQUW.js → chunk-3CMWKK32.js} +260 -104
  6. package/dist/chunk-3CMWKK32.js.map +1 -0
  7. package/dist/chunk-3RM57ZT2.js +31 -0
  8. package/dist/chunk-3RM57ZT2.js.map +1 -0
  9. package/dist/chunk-5MK3BISB.js +704 -0
  10. package/dist/chunk-5MK3BISB.js.map +1 -0
  11. package/dist/chunk-AI5JMUW7.js +45 -0
  12. package/dist/chunk-AI5JMUW7.js.map +1 -0
  13. package/dist/{chunk-FNQB2JEH.js → chunk-AZB6N7P4.js} +94 -65
  14. package/dist/chunk-AZB6N7P4.js.map +1 -0
  15. package/dist/chunk-BEYBU7VJ.js +53 -0
  16. package/dist/chunk-BEYBU7VJ.js.map +1 -0
  17. package/dist/chunk-BY4OMNIU.js +4699 -0
  18. package/dist/chunk-BY4OMNIU.js.map +1 -0
  19. package/dist/chunk-D2ECOWN5.js +515 -0
  20. package/dist/chunk-D2ECOWN5.js.map +1 -0
  21. package/dist/{chunk-A3UBYTJT.js → chunk-FN7HLLAZ.js} +60 -6
  22. package/dist/chunk-FN7HLLAZ.js.map +1 -0
  23. package/dist/chunk-GKDERULJ.js +50 -0
  24. package/dist/chunk-GKDERULJ.js.map +1 -0
  25. package/dist/chunk-GQL5KRAH.js +331 -0
  26. package/dist/chunk-GQL5KRAH.js.map +1 -0
  27. package/dist/{chunk-5XKDW36Z.js → chunk-KFR4CDGK.js} +70 -75
  28. package/dist/chunk-KFR4CDGK.js.map +1 -0
  29. package/dist/chunk-KV6QCRQ4.js +171 -0
  30. package/dist/chunk-KV6QCRQ4.js.map +1 -0
  31. package/dist/{chunk-DMPUP4R3.js → chunk-LXMAFD23.js} +5 -5
  32. package/dist/{chunk-I6G7BC42.js → chunk-MGNTXGTX.js} +2 -6
  33. package/dist/chunk-MGNTXGTX.js.map +1 -0
  34. package/dist/chunk-PK5LKOFL.js +239 -0
  35. package/dist/chunk-PK5LKOFL.js.map +1 -0
  36. package/dist/chunk-QOUPWOGW.js +29 -0
  37. package/dist/chunk-QOUPWOGW.js.map +1 -0
  38. package/dist/chunk-TRQ3XRFD.js +54 -0
  39. package/dist/chunk-TRQ3XRFD.js.map +1 -0
  40. package/dist/chunk-VKYZZXP5.js +323 -0
  41. package/dist/chunk-VKYZZXP5.js.map +1 -0
  42. package/dist/chunk-ZCPOIDAQ.js +127 -0
  43. package/dist/chunk-ZCPOIDAQ.js.map +1 -0
  44. package/dist/client-DvUatElp.d.ts +84 -0
  45. package/dist/completion/index.d.ts +5 -4
  46. package/dist/completion/index.js +25 -28
  47. package/dist/documents/index.d.ts +34 -0
  48. package/dist/documents/index.js +315 -0
  49. package/dist/documents/index.js.map +1 -0
  50. package/dist/{dynamic-tools-i9woysFa.d.ts → dynamic-tools-AEBOTgu1.d.ts} +21 -9
  51. package/dist/embeddings/index.d.ts +11 -9
  52. package/dist/embeddings/index.js +3 -3
  53. package/dist/evals/index.d.ts +53 -25
  54. package/dist/evals/index.js +153 -103
  55. package/dist/evals/index.js.map +1 -1
  56. package/dist/extractor/index.d.ts +23 -26
  57. package/dist/extractor/index.js +7 -8
  58. package/dist/guardrails/index.d.ts +7 -129
  59. package/dist/guardrails/index.js +1 -1
  60. package/dist/image-generation/index.d.ts +18 -16
  61. package/dist/image-generation/index.js +2 -2
  62. package/dist/index.d.ts +26 -23
  63. package/dist/index.js +71 -48
  64. package/dist/interactions-csssHYcN.d.ts +174 -0
  65. package/dist/internal/agent.d.ts +36 -29
  66. package/dist/internal/agent.js +48 -15
  67. package/dist/internal/agent.js.map +1 -1
  68. package/dist/mcp/index.d.ts +19 -13
  69. package/dist/mcp/index.js +14 -348
  70. package/dist/mcp/index.js.map +1 -1
  71. package/dist/memory/index.d.ts +20 -6
  72. package/dist/memory/index.js +10 -10
  73. package/dist/message-schema-B2M7AYzR.d.ts +52 -0
  74. package/dist/model-call-options-CZkSw_xN.d.ts +6 -0
  75. package/dist/model-listing/index.d.ts +3 -1
  76. package/dist/observability/index.d.ts +21 -6
  77. package/dist/observability/index.js +8 -5
  78. package/dist/observability/index.js.map +1 -1
  79. package/dist/pipeline/index.d.ts +106 -57
  80. package/dist/pipeline/index.js +419 -236
  81. package/dist/pipeline/index.js.map +1 -1
  82. package/dist/{retry-D3Ruy-ba.d.ts → retry-CjvSlKGW.d.ts} +2 -1
  83. package/dist/skills/index.d.ts +5 -4
  84. package/dist/skills/index.js +8 -7
  85. package/dist/speech-generation/index.d.ts +37 -0
  86. package/dist/speech-generation/index.js +8 -0
  87. package/dist/text-C_6eKrbC.d.ts +32 -0
  88. package/dist/{think-tool-DZqZj2LV.d.ts → think-tool-D8frXxwV.d.ts} +16 -2
  89. package/dist/tool/index.d.ts +9 -7
  90. package/dist/tool/index.js +12 -10
  91. package/dist/{tool-C_8GudXn.d.ts → tool-r62gf3QY.d.ts} +17 -8
  92. package/dist/transcription/index.d.ts +23 -14
  93. package/dist/transcription/index.js +2 -2
  94. package/dist/types-C3gv9EX6.d.ts +130 -0
  95. package/dist/{index-DxB74ybJ.d.ts → types-CElwurHZ.d.ts} +139 -88
  96. package/dist/types-CIPaIbZo.d.ts +105 -0
  97. package/dist/types-Cr4uiYo5.d.ts +99 -0
  98. package/dist/types-D9rKQ_as.d.ts +341 -0
  99. package/dist/types-DbolvLtZ.d.ts +155 -0
  100. package/dist/{types-DhQfwiZn.d.ts → types-RT3cuyRk.d.ts} +1 -1
  101. package/dist/vector-store/index.d.ts +33 -34
  102. package/dist/vector-store/index.js +7 -6
  103. package/package.json +10 -14
  104. package/dist/agent-DGYY_onS.d.ts +0 -145
  105. package/dist/audio-generation/index.d.ts +0 -31
  106. package/dist/audio-generation/index.js +0 -8
  107. package/dist/chunk-5XKDW36Z.js.map +0 -1
  108. package/dist/chunk-6WGGTJE6.js +0 -139
  109. package/dist/chunk-6WGGTJE6.js.map +0 -1
  110. package/dist/chunk-A3UBYTJT.js.map +0 -1
  111. package/dist/chunk-ADH7NNCS.js +0 -512
  112. package/dist/chunk-ADH7NNCS.js.map +0 -1
  113. package/dist/chunk-DZ3CKYTH.js +0 -267
  114. package/dist/chunk-DZ3CKYTH.js.map +0 -1
  115. package/dist/chunk-FNQB2JEH.js.map +0 -1
  116. package/dist/chunk-I6G7BC42.js.map +0 -1
  117. package/dist/chunk-IL3ZSNJI.js +0 -3748
  118. package/dist/chunk-IL3ZSNJI.js.map +0 -1
  119. package/dist/chunk-J6GNQART.js +0 -122
  120. package/dist/chunk-J6GNQART.js.map +0 -1
  121. package/dist/chunk-M3IOWB4Y.js +0 -280
  122. package/dist/chunk-M3IOWB4Y.js.map +0 -1
  123. package/dist/chunk-QY5GZ7HR.js +0 -39
  124. package/dist/chunk-QY5GZ7HR.js.map +0 -1
  125. package/dist/chunk-R2LJSQUW.js.map +0 -1
  126. package/dist/chunk-RFDVKCIE.js +0 -12
  127. package/dist/chunk-RFDVKCIE.js.map +0 -1
  128. package/dist/chunk-RMUBRRSK.js +0 -34
  129. package/dist/chunk-RMUBRRSK.js.map +0 -1
  130. package/dist/chunk-RXKHOYEH.js +0 -466
  131. package/dist/chunk-RXKHOYEH.js.map +0 -1
  132. package/dist/chunk-XDRSPSR7.js +0 -38
  133. package/dist/chunk-XDRSPSR7.js.map +0 -1
  134. package/dist/chunk-XWUC7CIT.js +0 -1
  135. package/dist/chunk-XWUC7CIT.js.map +0 -1
  136. package/dist/errors-CYKUcU89.d.ts +0 -17
  137. package/dist/json-Zo7K3xmt.d.ts +0 -40
  138. package/dist/loaders/index.d.ts +0 -86
  139. package/dist/loaders/index.js +0 -299
  140. package/dist/loaders/index.js.map +0 -1
  141. package/dist/middleware-KynLg_RW.d.ts +0 -56
  142. package/dist/types-B3A4YxaA.d.ts +0 -354
  143. package/dist/types-BBEGKB9m.d.ts +0 -119
  144. package/dist/types-DT3nEemY.d.ts +0 -79
  145. package/dist/types-DulRCYdB.d.ts +0 -70
  146. package/dist/types-p55Jc-sr.d.ts +0 -47
  147. package/dist/ui/index.d.ts +0 -107
  148. package/dist/ui/index.js +0 -10
  149. package/dist/ui/index.js.map +0 -1
  150. /package/dist/{chunk-DMPUP4R3.js.map → chunk-LXMAFD23.js.map} +0 -0
  151. /package/dist/{audio-generation → speech-generation}/index.js.map +0 -0
@@ -1,22 +1,15 @@
1
- import { U as Usage, l as JsonValue, C as CompletionModel, M as Message } from '../types-B3A4YxaA.js';
1
+ import { U as Usage, J as JsonObject, C as CompletionModel, M as Message } from '../types-D9rKQ_as.js';
2
2
  import { Z as ZodSchema } from '../zod-schema-C7F4clpm.js';
3
- import { A as Agent } from '../agent-DGYY_onS.js';
4
- import { l as AgentResponse } from '../index-DxB74ybJ.js';
5
- import { E as EmbeddingModel } from '../types-p55Jc-sr.js';
3
+ import { a as AgentInteractionRequest, b as AgentInteractionResponse } from '../interactions-csssHYcN.js';
4
+ import { t as AgentSuspendedResult, j as AgentResponse, l as AgentRunOptions, k as AgentResult } from '../types-CElwurHZ.js';
5
+ import { E as EmbeddingModel } from '../types-Cr4uiYo5.js';
6
+ import '../model-call-options-CZkSw_xN.js';
6
7
  import 'zod';
7
- import '../guardrails/index.js';
8
8
  import '../type-utils-CtHVDRn_.js';
9
- import '../types-BBEGKB9m.js';
10
- import '../middleware-KynLg_RW.js';
11
- import '../tool-C_8GudXn.js';
12
- import '../types-DulRCYdB.js';
13
- import '@modelcontextprotocol/sdk/client/sse.js';
14
- import '@modelcontextprotocol/sdk/client/stdio.js';
15
- import '@modelcontextprotocol/sdk/client/streamableHttp.js';
16
- import '../types-DhQfwiZn.js';
17
- import '../dynamic-tools-i9woysFa.js';
18
- import '../types-DT3nEemY.js';
19
- import '../retry-D3Ruy-ba.js';
9
+ import '../retry-CjvSlKGW.js';
10
+ import '../types-CIPaIbZo.js';
11
+ import '../types-DbolvLtZ.js';
12
+ import '../tool-r62gf3QY.js';
20
13
 
21
14
  type EvalOutcome<Score = unknown> = {
22
15
  outcome: "pass";
@@ -57,7 +50,8 @@ declare const EvalOutcome: {
57
50
  }): EvalOutcome<Score>;
58
51
  };
59
52
 
60
- type EvalMetadata = Record<string, JsonValue | undefined>;
53
+ type EvalMetadata = JsonObject;
54
+ type EvalReporterErrorPolicy = "collect" | "throw";
61
55
  type EvalRunOptions = {
62
56
  id?: string | undefined;
63
57
  datasetName?: string | undefined;
@@ -85,6 +79,7 @@ type EvalTurn = {
85
79
  metadata?: EvalMetadata | undefined;
86
80
  };
87
81
  type EvalTraceRef = {
82
+ observer?: string | undefined;
88
83
  traceId: string;
89
84
  observationId?: string | undefined;
90
85
  responseId?: string | undefined;
@@ -255,7 +250,7 @@ type RunEvalSuiteOptions<Input, Output, Expected = unknown, Metrics extends read
255
250
  concurrency?: number | undefined;
256
251
  trace?: EvalTraceSelector<NoInfer<Input>, NoInfer<Output>, NoInfer<Expected>> | undefined;
257
252
  reporters?: readonly EvalReporter<NoInfer<Input>, NoInfer<Output>, NoInfer<Expected>>[] | undefined;
258
- failOnReporterError?: boolean | undefined;
253
+ reporterErrorPolicy?: EvalReporterErrorPolicy | undefined;
259
254
  targetUsage?: EvalTargetUsageSelector<NoInfer<Input>, NoInfer<Output>, NoInfer<Expected>> | undefined;
260
255
  cost?: EvalCostOptions<NoInfer<Input>, NoInfer<Output>, NoInfer<Expected>> | undefined;
261
256
  };
@@ -388,12 +383,41 @@ declare function knowledgeRetention<Input, Output, Expected = unknown, const Nam
388
383
  }): EvalMetric<Input, Output, number, Expected, Name>;
389
384
  type ConversationSource = EvalTurn[] | Message[];
390
385
 
391
- type AgentEvalTargetOptions<Input, Output = AgentResponse, Expected = unknown> = {
392
- prompt?: ((input: Input, testCase: EvalCase<Input, Expected>) => string | Message) | undefined;
393
- output?: ((response: AgentResponse, testCase: EvalCase<Input, Expected>) => Output) | undefined;
394
- };
395
- declare function agentEvalTarget<Input, Expected = unknown>(agent: Agent, options?: AgentEvalTargetOptions<Input, AgentResponse, Expected>): EvalTarget<Input, AgentResponse, Expected>;
396
- declare function agentEvalTarget<Input, Output, Expected = unknown>(agent: Agent, options: AgentEvalTargetOptions<Input, Output, Expected>): EvalTarget<Input, Output, Expected>;
386
+ type EvaluableAgent<Output> = {
387
+ generate(input: AgentRunOptions<Output>): Promise<AgentResult<Output>>;
388
+ };
389
+ type IsAny<Value> = 0 extends 1 & Value ? true : false;
390
+ type IsExactly<Left, Right> = IsAny<Left> extends true ? false : [Left] extends [Right] ? [Right] extends [Left] ? true : false : false;
391
+ type AgentEvalTargetOptions<Input, AgentOutput = string, Output = AgentResponse<AgentOutput>, Expected = unknown> = {
392
+ agent: EvaluableAgent<AgentOutput>;
393
+ request(args: {
394
+ input: Input;
395
+ testCase: EvalCase<Input, Expected>;
396
+ }): AgentRunOptions<AgentOutput> | Promise<AgentRunOptions<AgentOutput>>;
397
+ interactions?: {
398
+ maxResponses?: number | undefined;
399
+ respond(args: {
400
+ interaction: AgentInteractionRequest;
401
+ testCase: EvalCase<Input, Expected>;
402
+ phase: number;
403
+ }): AgentInteractionResponse | Promise<AgentInteractionResponse>;
404
+ } | undefined;
405
+ } & (IsExactly<Output, AgentResponse<AgentOutput>> extends true ? {
406
+ output?(args: {
407
+ response: AgentResponse<AgentOutput>;
408
+ testCase: EvalCase<Input, Expected>;
409
+ }): Output | Promise<Output>;
410
+ } : {
411
+ output(args: {
412
+ response: AgentResponse<AgentOutput>;
413
+ testCase: EvalCase<Input, Expected>;
414
+ }): Output | Promise<Output>;
415
+ });
416
+ declare class AgentEvalSuspensionError extends Error {
417
+ readonly result: AgentSuspendedResult;
418
+ constructor(result: AgentSuspendedResult, message?: string);
419
+ }
420
+ declare function agentEvalTarget<Input, AgentOutput = string, Output = AgentResponse<AgentOutput>, Expected = unknown>(options: AgentEvalTargetOptions<Input, AgentOutput, Output, Expected>): EvalTarget<Input, Output, Expected>;
397
421
 
398
422
  type EvalOutputFormat = "pretty" | "json" | "quiet";
399
423
  type EvalExpectedTotals = Partial<EvalTotals> & {
@@ -569,6 +593,10 @@ declare function resolveEvalTraceRef(args: {
569
593
  }): EvalTraceRef | undefined;
570
594
  declare function defaultEvalTraceSelector<Input, Output, Expected>(args: EvalTraceSelectorArgs<Input, Output, Expected>): EvalTraceRef | undefined;
571
595
 
596
+ declare class EvalReporterDispatchError extends AggregateError {
597
+ readonly phase: string;
598
+ constructor(phase: string, errors: readonly unknown[]);
599
+ }
572
600
  declare function runEvalSuite<Input, Output, Expected = unknown, const Metrics extends readonly EvalMetric<NoInfer<Input>, NoInfer<Output>, unknown, NoInfer<Expected>, string>[] = readonly EvalMetric<NoInfer<Input>, NoInfer<Output>, unknown, NoInfer<Expected>, string>[]>(options: RunEvalSuiteOptions<Input, Output, Expected, Metrics>): Promise<EvalSuiteResult<Input, Output, Expected, Metrics>>;
573
601
 
574
602
  declare function selectPromptOutput(args: EvalMetricArgs<unknown, unknown, unknown>): string;
@@ -640,4 +668,4 @@ declare function defineEvalSuite<const Cases extends readonly EvalCaseLike[], co
640
668
  target: Target;
641
669
  };
642
670
 
643
- export { type AbstentionCategory, type AbstentionOptions, type AgentEvalTargetOptions, type AnswerRelevancyOptions, type AnyEvalMetric, type ContainsAllOptions, type ContainsAnyOptions, type ContainsListOptions, type ContainsOptions, type DefaultEvalActual, type DefinedEvalSuite, type DoesNotMatchOptions, EvalAssertionError, type EvalCase, type EvalCaseRequirements, type EvalCaseResult, type EvalCasesExpected, type EvalCasesForMetrics, type EvalCasesInput, type EvalCostCalculatorArgs, type EvalCostOptions, type EvalCostSummary, type EvalDataType, type EvalExpectations, type EvalExpectedOutcomes, type EvalExpectedTotals, type EvalMetadata, type EvalMetric, type EvalMetricArgs, type EvalMetricDescriptor, type EvalMetricResult, type EvalMetricResultFor, type EvalMetricScore, EvalOutcome, type EvalOutcomeStatus, type EvalOutputFormat, type EvalOutputWriters, type EvalReportArgs, type EvalReporter, type EvalRunContext, type EvalRunEndArgs, type EvalRunOptions, type EvalRunStartArgs, type EvalScoreDirection, type EvalScoreMap, type EvalScoreProjection, type EvalSuiteResult, type EvalTarget, type EvalTargetUsageSelector, type EvalTotals, type EvalTraceCarrier, type EvalTraceRef, type EvalTraceSelector, type EvalTraceSelectorArgs, type EvalTurn, type EvalUsageSummary, type ExactMatchOptions, type FaithfulnessOptions, type GEvalOptions, type GEvalParameter, type GEvalRubric, type HallucinationOptions, type JsonCorrectnessOptions, type KnowledgeRetentionOptions, type LlmJudgeOptions, type LlmScoreMetricScore, type LlmScoreOptions, type MatchesOptions, type MaxLengthOptions, type NotContainsOptions, type PrintEvalResultOptions, type PromptAlignmentOptions, type RequiredFieldsOptions, type RunEvalCliOptions, type RunEvalSuiteOptions, type SelectorOrValue, type SemanticSimilarityOptions, type SummarizationOptions, type TurnRelevancyOptions, type ValueSelector, abstention, agentEvalTarget, answerRelevancy, assertEvalOutcomes, assertEvalTotals, contains, containsAll, containsAny, defaultEvalTraceSelector, defineEvalCases, defineEvalSuite, defineMetric, doesNotMatch, evalExitCode, exactMatch, faithfulness, gEval, hallucination, jsonCorrectness, knowledgeRetention, llmJudge, llmScore, matches, maxLength, notContains, printEvalResult, projectEvalOutcome, promptAlignment, requiredFields, resolveEvalTraceRef, runEvalCli, runEvalSuite, selectPromptOutput, semanticSimilarity, summarization, turnRelevancy };
671
+ export { type AbstentionCategory, type AbstentionOptions, AgentEvalSuspensionError, type AgentEvalTargetOptions, type AnswerRelevancyOptions, type AnyEvalMetric, type ContainsAllOptions, type ContainsAnyOptions, type ContainsListOptions, type ContainsOptions, type DefaultEvalActual, type DefinedEvalSuite, type DoesNotMatchOptions, EvalAssertionError, type EvalCase, type EvalCaseRequirements, type EvalCaseResult, type EvalCasesExpected, type EvalCasesForMetrics, type EvalCasesInput, type EvalCostCalculatorArgs, type EvalCostOptions, type EvalCostSummary, type EvalDataType, type EvalExpectations, type EvalExpectedOutcomes, type EvalExpectedTotals, type EvalMetadata, type EvalMetric, type EvalMetricArgs, type EvalMetricDescriptor, type EvalMetricResult, type EvalMetricResultFor, type EvalMetricScore, EvalOutcome, type EvalOutcomeStatus, type EvalOutputFormat, type EvalOutputWriters, type EvalReportArgs, type EvalReporter, EvalReporterDispatchError, type EvalReporterErrorPolicy, type EvalRunContext, type EvalRunEndArgs, type EvalRunOptions, type EvalRunStartArgs, type EvalScoreDirection, type EvalScoreMap, type EvalScoreProjection, type EvalSuiteResult, type EvalTarget, type EvalTargetUsageSelector, type EvalTotals, type EvalTraceCarrier, type EvalTraceRef, type EvalTraceSelector, type EvalTraceSelectorArgs, type EvalTurn, type EvalUsageSummary, type ExactMatchOptions, type FaithfulnessOptions, type GEvalOptions, type GEvalParameter, type GEvalRubric, type HallucinationOptions, type JsonCorrectnessOptions, type KnowledgeRetentionOptions, type LlmJudgeOptions, type LlmScoreMetricScore, type LlmScoreOptions, type MatchesOptions, type MaxLengthOptions, type NotContainsOptions, type PrintEvalResultOptions, type PromptAlignmentOptions, type RequiredFieldsOptions, type RunEvalCliOptions, type RunEvalSuiteOptions, type SelectorOrValue, type SemanticSimilarityOptions, type SummarizationOptions, type TurnRelevancyOptions, type ValueSelector, abstention, agentEvalTarget, answerRelevancy, assertEvalOutcomes, assertEvalTotals, contains, containsAll, containsAny, defaultEvalTraceSelector, defineEvalCases, defineEvalSuite, defineMetric, doesNotMatch, evalExitCode, exactMatch, faithfulness, gEval, hallucination, jsonCorrectness, knowledgeRetention, llmJudge, llmScore, matches, maxLength, notContains, printEvalResult, projectEvalOutcome, promptAlignment, requiredFields, resolveEvalTraceRef, runEvalCli, runEvalSuite, selectPromptOutput, semanticSimilarity, summarization, turnRelevancy };
@@ -1,33 +1,29 @@
1
1
  import {
2
- cancelAgentApproval
3
- } from "../chunk-IL3ZSNJI.js";
4
- import "../chunk-YK4WAAS4.js";
5
- import "../chunk-6WGGTJE6.js";
6
- import "../chunk-R2LJSQUW.js";
7
- import "../chunk-5XKDW36Z.js";
2
+ AgentRunBlockedError
3
+ } from "../chunk-TRQ3XRFD.js";
8
4
  import {
9
5
  cosineSimilarity,
10
- embedText,
6
+ embedText
7
+ } from "../chunk-GQL5KRAH.js";
8
+ import {
11
9
  mapWithConcurrency
12
- } from "../chunk-DZ3CKYTH.js";
10
+ } from "../chunk-3RM57ZT2.js";
13
11
  import {
14
- Extractor
15
- } from "../chunk-J6GNQART.js";
16
- import "../chunk-I6G7BC42.js";
17
- import "../chunk-M3IOWB4Y.js";
18
- import "../chunk-RFDVKCIE.js";
12
+ extract
13
+ } from "../chunk-ZCPOIDAQ.js";
14
+ import "../chunk-MGNTXGTX.js";
19
15
  import {
20
16
  Usage
21
- } from "../chunk-ADH7NNCS.js";
22
- import "../chunk-A3UBYTJT.js";
23
- import "../chunk-FNQB2JEH.js";
17
+ } from "../chunk-D2ECOWN5.js";
18
+ import "../chunk-PK5LKOFL.js";
19
+ import "../chunk-FN7HLLAZ.js";
24
20
 
25
21
  // src/evals/advanced-metrics.ts
26
22
  import { z } from "zod";
27
23
 
28
24
  // src/evals/format.ts
29
25
  function defaultOutputValue(output) {
30
- if (typeof output === "object" && output !== null && "output" in output && typeof output.output === "string") {
26
+ if (typeof output === "object" && output !== null && "output" in output) {
31
27
  return output.output;
32
28
  }
33
29
  return output;
@@ -54,16 +50,15 @@ function errorMessage(error) {
54
50
 
55
51
  // src/evals/judge.ts
56
52
  async function runJudge(args) {
57
- const extractor = new Extractor({
53
+ const result = await extract({
58
54
  model: args.model,
59
55
  outputSchema: args.schema,
60
- instructions: args.instructions
61
- });
62
- const result = await extractor.extractResult(args.prompt, {
56
+ instructions: args.instructions,
57
+ text: args.prompt,
63
58
  temperature: 0,
64
59
  retries: args.retries <= 0 ? void 0 : { maxAttempts: Math.trunc(args.retries) + 1 }
65
60
  });
66
- return { data: result.data, usage: result.usage };
61
+ return { data: result.output, usage: result.usage };
67
62
  }
68
63
  function addUsage(...values) {
69
64
  return values.reduce((total, usage) => Usage.add(total, usage), Usage.empty());
@@ -1120,17 +1115,61 @@ function contentText(content) {
1120
1115
  }
1121
1116
 
1122
1117
  // src/evals/agent-target.ts
1123
- function agentEvalTarget(agent, options = {}) {
1118
+ var AgentEvalSuspensionError = class extends Error {
1119
+ constructor(result, message = "Agent eval target suspended without an interaction responder.") {
1120
+ super(message);
1121
+ this.result = result;
1122
+ this.name = "AgentEvalSuspensionError";
1123
+ }
1124
+ result;
1125
+ };
1126
+ function agentEvalTarget(options) {
1124
1127
  return async (input, testCase) => {
1125
- const prompt = options.prompt?.(input, testCase) ?? String(input);
1126
- const response = await agent.generate(prompt);
1127
- if (response.status === "approval_required") {
1128
- await cancelAgentApproval(response, "Agent eval targets cannot suspend for tool approval.");
1129
- throw new Error("Agent eval targets cannot suspend for tool approval.");
1128
+ const maxResponses = options.interactions?.maxResponses ?? 10;
1129
+ if (!Number.isSafeInteger(maxResponses) || maxResponses < 1) {
1130
+ throw new TypeError("Agent eval interactions.maxResponses must be a positive integer.");
1131
+ }
1132
+ const request = await options.request({ input, testCase });
1133
+ const runSettings = agentRunSettings(request);
1134
+ let response = await options.agent.generate(request);
1135
+ let phase = 0;
1136
+ while (response.status === "suspended") {
1137
+ if (options.interactions === void 0) {
1138
+ throw new AgentEvalSuspensionError(response);
1139
+ }
1140
+ if (phase >= maxResponses) {
1141
+ throw new AgentEvalSuspensionError(
1142
+ response,
1143
+ `Agent eval target exceeded the interaction response limit of ${maxResponses}.`
1144
+ );
1145
+ }
1146
+ phase += 1;
1147
+ const interactionResponse = await options.interactions.respond({
1148
+ interaction: response.interaction,
1149
+ testCase,
1150
+ phase
1151
+ });
1152
+ response = await options.agent.generate({
1153
+ continuation: response.continuation,
1154
+ response: interactionResponse,
1155
+ ...runSettings
1156
+ });
1130
1157
  }
1131
- return options.output === void 0 ? response : options.output(response, testCase);
1158
+ if (response.status === "blocked") throw new AgentRunBlockedError(response);
1159
+ return options.output === void 0 ? response : await options.output({ response, testCase });
1132
1160
  };
1133
1161
  }
1162
+ function agentRunSettings(request) {
1163
+ const {
1164
+ prompt: _prompt,
1165
+ messages: _messages,
1166
+ session: _session,
1167
+ continuation: _continuation,
1168
+ response: _response,
1169
+ ...settings
1170
+ } = request;
1171
+ return settings;
1172
+ }
1134
1173
 
1135
1174
  // src/evals/reporting.ts
1136
1175
  function projectEvalOutcome(outcome, dataType, projectScore) {
@@ -1193,6 +1232,7 @@ function traceFromCarrier(value) {
1193
1232
  function traceFromMetadata(metadata) {
1194
1233
  if (metadata === void 0) return void 0;
1195
1234
  return readTraceRef({
1235
+ observer: metadata.traceObserver,
1196
1236
  traceId: metadata.traceId,
1197
1237
  observationId: metadata.observationId,
1198
1238
  responseId: metadata.responseId
@@ -1202,9 +1242,11 @@ function readTraceRef(value) {
1202
1242
  if (typeof value !== "object" || value === null) return void 0;
1203
1243
  const traceId = value.traceId;
1204
1244
  if (typeof traceId !== "string" || traceId.length === 0) return void 0;
1245
+ const observer = value.observer;
1205
1246
  const observationId = value.observationId;
1206
1247
  const responseId = value.responseId;
1207
1248
  const trace = { traceId };
1249
+ if (typeof observer === "string" && observer.length > 0) trace.observer = observer;
1208
1250
  if (typeof observationId === "string" && observationId.length > 0) {
1209
1251
  trace.observationId = observationId;
1210
1252
  }
@@ -1213,6 +1255,14 @@ function readTraceRef(value) {
1213
1255
  }
1214
1256
 
1215
1257
  // src/evals/runner.ts
1258
+ var EvalReporterDispatchError = class extends AggregateError {
1259
+ phase;
1260
+ constructor(phase, errors) {
1261
+ super(errors, `Evaluation reporter ${phase} failed ${errors.length} time(s).`);
1262
+ this.name = "EvalReporterDispatchError";
1263
+ this.phase = phase;
1264
+ }
1265
+ };
1216
1266
  async function runEvalSuite(options) {
1217
1267
  validateSuiteOptions(options);
1218
1268
  const startedAtMs = Date.now();
@@ -1229,7 +1279,7 @@ async function runEvalSuite(options) {
1229
1279
  reporterErrors = await notifyRunStart(
1230
1280
  reporters,
1231
1281
  lifecycle,
1232
- options.failOnReporterError === true
1282
+ options.reporterErrorPolicy ?? "collect"
1233
1283
  );
1234
1284
  } catch (error) {
1235
1285
  await notifyRunEnd(reporters, {
@@ -1264,25 +1314,26 @@ async function runEvalSuite(options) {
1264
1314
  metrics: aggregates.metrics,
1265
1315
  cases: aggregates.cases,
1266
1316
  usage: aggregates.usage,
1267
- ...aggregates.cost === void 0 ? {} : { cost: aggregates.cost },
1268
1317
  durationMs: Date.now() - startedAtMs,
1269
1318
  reporterErrors
1270
1319
  };
1320
+ if (aggregates.cost !== void 0) {
1321
+ result.cost = aggregates.cost;
1322
+ }
1323
+ const runEndArgs = {
1324
+ ...lifecycle,
1325
+ status: "completed",
1326
+ completedAt,
1327
+ durationMs: result.durationMs,
1328
+ metrics: result.metrics,
1329
+ cases: result.cases,
1330
+ usage: result.usage
1331
+ };
1332
+ if (result.cost !== void 0) {
1333
+ runEndArgs.cost = result.cost;
1334
+ }
1271
1335
  result.reporterErrors.push(
1272
- ...await notifyRunEnd(
1273
- reporters,
1274
- {
1275
- ...lifecycle,
1276
- status: "completed",
1277
- completedAt,
1278
- durationMs: result.durationMs,
1279
- metrics: result.metrics,
1280
- cases: result.cases,
1281
- usage: result.usage,
1282
- ...result.cost === void 0 ? {} : { cost: result.cost }
1283
- },
1284
- options.failOnReporterError === true
1285
- )
1336
+ ...await notifyRunEnd(reporters, runEndArgs, options.reporterErrorPolicy ?? "collect")
1286
1337
  );
1287
1338
  return result;
1288
1339
  }
@@ -1335,7 +1386,7 @@ async function runEvalCase(options, testCase, run) {
1335
1386
  trace: traceResult.trace,
1336
1387
  traceError: traceResult.error,
1337
1388
  reporters: options.reporters ?? [],
1338
- failOnReporterError: options.failOnReporterError === true
1389
+ reporterErrorPolicy: options.reporterErrorPolicy ?? "collect"
1339
1390
  });
1340
1391
  const metricResult = {
1341
1392
  metricName: metric.name,
@@ -1388,9 +1439,7 @@ async function safeEvaluate(suiteName, testCase, output, metric) {
1388
1439
  async function reportOutcome(args) {
1389
1440
  const errors = [];
1390
1441
  if (args.traceError !== void 0) {
1391
- if (args.failOnReporterError) throw args.traceError;
1392
1442
  errors.push(args.traceError);
1393
- return errors;
1394
1443
  }
1395
1444
  for (const reporter of args.reporters) {
1396
1445
  try {
@@ -1405,12 +1454,10 @@ async function reportOutcome(args) {
1405
1454
  outcome: args.outcome
1406
1455
  });
1407
1456
  } catch (error) {
1408
- if (args.failOnReporterError) {
1409
- throw error;
1410
- }
1411
1457
  errors.push(error);
1412
1458
  }
1413
1459
  }
1460
+ throwReporterErrors("report", errors, args.reporterErrorPolicy);
1414
1461
  return errors;
1415
1462
  }
1416
1463
  function resolveRun(options, startedAtMs) {
@@ -1426,40 +1473,46 @@ function resolveRun(options, startedAtMs) {
1426
1473
  throw new TypeError(`Evaluation run ${label} must contain 1 to 256 characters`);
1427
1474
  }
1428
1475
  }
1429
- return {
1476
+ const run = {
1430
1477
  id,
1431
- startedAt: new Date(startedAtMs).toISOString(),
1432
- ...options.run?.datasetName === void 0 ? {} : { datasetName: options.run.datasetName },
1433
- ...options.run?.datasetVersion === void 0 ? {} : { datasetVersion: options.run.datasetVersion },
1434
- ...options.run?.metadata === void 0 ? {} : { metadata: options.run.metadata }
1478
+ startedAt: new Date(startedAtMs).toISOString()
1435
1479
  };
1480
+ if (options.run?.datasetName !== void 0) run.datasetName = options.run.datasetName;
1481
+ if (options.run?.datasetVersion !== void 0) run.datasetVersion = options.run.datasetVersion;
1482
+ if (options.run?.metadata !== void 0) run.metadata = options.run.metadata;
1483
+ return run;
1436
1484
  }
1437
- async function notifyRunStart(reporters, args, failOnReporterError) {
1485
+ async function notifyRunStart(reporters, args, errorPolicy) {
1438
1486
  const errors = [];
1439
1487
  for (const reporter of reporters) {
1440
1488
  if (reporter.onRunStart === void 0) continue;
1441
1489
  try {
1442
1490
  await reporter.onRunStart(args);
1443
1491
  } catch (error) {
1444
- if (failOnReporterError) throw error;
1445
1492
  errors.push(error);
1446
1493
  }
1447
1494
  }
1495
+ throwReporterErrors("onRunStart", errors, errorPolicy);
1448
1496
  return errors;
1449
1497
  }
1450
- async function notifyRunEnd(reporters, args, failOnReporterError = false) {
1498
+ async function notifyRunEnd(reporters, args, errorPolicy = "collect") {
1451
1499
  const errors = [];
1452
1500
  for (const reporter of reporters) {
1453
1501
  if (reporter.onRunEnd === void 0) continue;
1454
1502
  try {
1455
1503
  await reporter.onRunEnd(args);
1456
1504
  } catch (error) {
1457
- if (failOnReporterError) throw error;
1458
1505
  errors.push(error);
1459
1506
  }
1460
1507
  }
1508
+ throwReporterErrors("onRunEnd", errors, errorPolicy);
1461
1509
  return errors;
1462
1510
  }
1511
+ function throwReporterErrors(phase, errors, errorPolicy) {
1512
+ if (errorPolicy === "throw" && errors.length > 0) {
1513
+ throw new EvalReporterDispatchError(phase, errors);
1514
+ }
1515
+ }
1463
1516
  function countMetricOutcomes(results) {
1464
1517
  const totals = emptyTotals();
1465
1518
  for (const result of results) {
@@ -1714,18 +1767,21 @@ function jsonResult(result) {
1714
1767
  }
1715
1768
  function expectationMismatches(result, expectations) {
1716
1769
  if (expectations === void 0) return [];
1717
- return [
1718
- ...expectations.totals === void 0 ? [] : totalMismatches(result, expectations.totals),
1719
- ...expectations.outcomes === void 0 ? [] : outcomeMismatches(result, expectations.outcomes)
1720
- ];
1770
+ const mismatches = [];
1771
+ if (expectations.totals !== void 0) {
1772
+ mismatches.push(...totalMismatches(result, expectations.totals));
1773
+ }
1774
+ if (expectations.outcomes !== void 0) {
1775
+ mismatches.push(...outcomeMismatches(result, expectations.outcomes));
1776
+ }
1777
+ return mismatches;
1721
1778
  }
1722
1779
  function totalMismatches(result, expected) {
1723
- const directMetrics = {
1724
- ...expected.total === void 0 ? {} : { total: expected.total },
1725
- ...expected.passed === void 0 ? {} : { passed: expected.passed },
1726
- ...expected.failed === void 0 ? {} : { failed: expected.failed },
1727
- ...expected.invalid === void 0 ? {} : { invalid: expected.invalid }
1728
- };
1780
+ const directMetrics = {};
1781
+ if (expected.total !== void 0) directMetrics.total = expected.total;
1782
+ if (expected.passed !== void 0) directMetrics.passed = expected.passed;
1783
+ if (expected.failed !== void 0) directMetrics.failed = expected.failed;
1784
+ if (expected.invalid !== void 0) directMetrics.invalid = expected.invalid;
1729
1785
  return [
1730
1786
  ...totalsGroupMismatches("metrics", result.metrics, {
1731
1787
  ...directMetrics,
@@ -2004,9 +2060,9 @@ function semanticSimilarity(options) {
2004
2060
  if (typeof expected !== "string") {
2005
2061
  return EvalOutcome.invalid("Semantic similarity expected value must be a string.");
2006
2062
  }
2007
- const [actualEmbedding, expectedEmbedding] = await Promise.all([
2008
- embedText(options.model, actual),
2009
- embedText(options.model, expected)
2063
+ const [{ embedding: actualEmbedding }, { embedding: expectedEmbedding }] = await Promise.all([
2064
+ embedText({ model: options.model, text: actual }),
2065
+ embedText({ model: options.model, text: expected })
2010
2066
  ]);
2011
2067
  const score = cosineSimilarity(actualEmbedding.vector, expectedEmbedding.vector);
2012
2068
  return score >= options.threshold ? EvalOutcome.pass(score) : EvalOutcome.fail(score, { comment: `Similarity below threshold ${options.threshold}.` });
@@ -2014,23 +2070,19 @@ function semanticSimilarity(options) {
2014
2070
  };
2015
2071
  }
2016
2072
  function llmJudge(options) {
2017
- const extractor = new Extractor({
2018
- model: options.model,
2019
- outputSchema: options.schema,
2020
- instructions: options.instructions ?? "Judge the eval case by the requested schema. Submit the judgment using the schema."
2021
- });
2022
2073
  return {
2023
2074
  name: options.name ?? "llm_judge",
2024
2075
  required: options.required ?? true,
2025
2076
  async evaluate(args) {
2026
2077
  try {
2027
- const result = await extractor.extractResult(
2028
- await resolveJudgePrompt(options.prompt, args),
2029
- {
2030
- retries: evalExtractionRetries(options.retries)
2031
- }
2032
- );
2033
- return options.passes(result.data) ? EvalOutcome.pass(result.data, { usage: result.usage }) : EvalOutcome.fail(result.data, { usage: result.usage });
2078
+ const result = await extract({
2079
+ model: options.model,
2080
+ outputSchema: options.schema,
2081
+ instructions: options.instructions ?? "Judge the eval case by the requested schema. Submit the judgment using the schema.",
2082
+ text: await resolveJudgePrompt(options.prompt, args),
2083
+ retries: evalExtractionRetries(options.retries)
2084
+ });
2085
+ return options.passes(result.output) ? EvalOutcome.pass(result.output, { usage: result.usage }) : EvalOutcome.fail(result.output, { usage: result.usage });
2034
2086
  } catch (error) {
2035
2087
  return EvalOutcome.invalid(errorMessage(error));
2036
2088
  }
@@ -2039,17 +2091,6 @@ function llmJudge(options) {
2039
2091
  }
2040
2092
  function llmScore(options) {
2041
2093
  const criteria = Array.isArray(options.criteria) ? options.criteria.join("\n") : options.criteria;
2042
- const extractor = new Extractor({
2043
- model: options.model,
2044
- outputSchema: z2.object({
2045
- score: z2.number(),
2046
- feedback: z2.string()
2047
- }),
2048
- instructions: options.instructions ?? `Score the eval case against these criteria:
2049
- ${criteria}
2050
-
2051
- Return a score between 0 and 1 and brief feedback.`
2052
- });
2053
2094
  return {
2054
2095
  name: options.name ?? "llm_score",
2055
2096
  required: options.required ?? true,
@@ -2059,13 +2100,20 @@ Return a score between 0 and 1 and brief feedback.`
2059
2100
  threshold: options.threshold,
2060
2101
  async evaluate(args) {
2061
2102
  try {
2062
- const result = await extractor.extractResult(
2063
- await resolveJudgePrompt(options.prompt, args),
2064
- {
2065
- retries: evalExtractionRetries(options.retries)
2066
- }
2067
- );
2068
- const score = result.data;
2103
+ const result = await extract({
2104
+ model: options.model,
2105
+ outputSchema: z2.object({
2106
+ score: z2.number(),
2107
+ feedback: z2.string()
2108
+ }),
2109
+ instructions: options.instructions ?? `Score the eval case against these criteria:
2110
+ ${criteria}
2111
+
2112
+ Return a score between 0 and 1 and brief feedback.`,
2113
+ text: await resolveJudgePrompt(options.prompt, args),
2114
+ retries: evalExtractionRetries(options.retries)
2115
+ });
2116
+ const score = result.output;
2069
2117
  if (score.score < 0 || score.score > 1) {
2070
2118
  return EvalOutcome.invalid(`Score ${score.score} outside valid range [0, 1].`, {
2071
2119
  score,
@@ -2100,8 +2148,10 @@ function defineEvalSuite(options) {
2100
2148
  };
2101
2149
  }
2102
2150
  export {
2151
+ AgentEvalSuspensionError,
2103
2152
  EvalAssertionError,
2104
2153
  EvalOutcome,
2154
+ EvalReporterDispatchError,
2105
2155
  abstention,
2106
2156
  agentEvalTarget,
2107
2157
  answerRelevancy,