@anvia/core 1.0.0-rc.1 → 1.0.0-rc.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +283 -84
- package/dist/agent/index.d.ts +26 -18
- package/dist/agent/index.js +19 -15
- package/dist/agent-yBG8e9Jj.d.ts +153 -0
- package/dist/chunk-4WI3VGZK.js +127 -0
- package/dist/chunk-4WI3VGZK.js.map +1 -0
- package/dist/{chunk-R2LJSQUW.js → chunk-77AIYARB.js} +251 -104
- package/dist/chunk-77AIYARB.js.map +1 -0
- package/dist/chunk-AI5JMUW7.js +45 -0
- package/dist/chunk-AI5JMUW7.js.map +1 -0
- package/dist/chunk-BEYBU7VJ.js +53 -0
- package/dist/chunk-BEYBU7VJ.js.map +1 -0
- package/dist/chunk-CWM7B2OO.js +315 -0
- package/dist/chunk-CWM7B2OO.js.map +1 -0
- package/dist/{chunk-6WGGTJE6.js → chunk-D37WMSP6.js} +52 -60
- package/dist/chunk-D37WMSP6.js.map +1 -0
- package/dist/{chunk-IL3ZSNJI.js → chunk-DBSCKTUL.js} +845 -800
- package/dist/chunk-DBSCKTUL.js.map +1 -0
- package/dist/chunk-EINTGEGY.js +629 -0
- package/dist/chunk-EINTGEGY.js.map +1 -0
- package/dist/chunk-FMHLROPQ.js +351 -0
- package/dist/chunk-FMHLROPQ.js.map +1 -0
- package/dist/{chunk-A3UBYTJT.js → chunk-FN7HLLAZ.js} +60 -6
- package/dist/chunk-FN7HLLAZ.js.map +1 -0
- package/dist/chunk-GKDERULJ.js +50 -0
- package/dist/chunk-GKDERULJ.js.map +1 -0
- package/dist/{chunk-FNQB2JEH.js → chunk-KIDGIJM7.js} +26 -5
- package/dist/chunk-KIDGIJM7.js.map +1 -0
- package/dist/chunk-KR3RCCHJ.js +500 -0
- package/dist/chunk-KR3RCCHJ.js.map +1 -0
- package/dist/{chunk-I6G7BC42.js → chunk-MGNTXGTX.js} +2 -6
- package/dist/chunk-MGNTXGTX.js.map +1 -0
- package/dist/{chunk-5XKDW36Z.js → chunk-OJH6HWHX.js} +64 -75
- package/dist/chunk-OJH6HWHX.js.map +1 -0
- package/dist/chunk-QOUPWOGW.js +29 -0
- package/dist/chunk-QOUPWOGW.js.map +1 -0
- package/dist/chunk-TIZYTPJO.js +208 -0
- package/dist/chunk-TIZYTPJO.js.map +1 -0
- package/dist/{chunk-DMPUP4R3.js → chunk-XKYYBABH.js} +5 -5
- package/dist/client-B8DT1vUG.d.ts +83 -0
- package/dist/completion/index.d.ts +5 -4
- package/dist/completion/index.js +25 -28
- package/dist/{dynamic-tools-i9woysFa.d.ts → dynamic-tools-KwEGRbHp.d.ts} +21 -9
- package/dist/embeddings/index.d.ts +11 -9
- package/dist/embeddings/index.js +2 -3
- package/dist/evals/index.d.ts +44 -24
- package/dist/evals/index.js +86 -69
- package/dist/evals/index.js.map +1 -1
- package/dist/extractor/index.d.ts +23 -26
- package/dist/extractor/index.js +7 -8
- package/dist/guardrails/index.d.ts +2 -1
- package/dist/guardrails/index.js +1 -1
- package/dist/image-generation/index.d.ts +18 -16
- package/dist/image-generation/index.js +2 -2
- package/dist/index.d.ts +24 -22
- package/dist/index.js +50 -48
- package/dist/internal/agent.d.ts +30 -19
- package/dist/internal/agent.js +13 -11
- package/dist/loaders/index.d.ts +2 -1
- package/dist/mcp/index.d.ts +19 -13
- package/dist/mcp/index.js +13 -348
- package/dist/mcp/index.js.map +1 -1
- package/dist/memory/index.d.ts +20 -6
- package/dist/memory/index.js +10 -10
- package/dist/message-schema-Cn7ezRfJ.d.ts +52 -0
- package/dist/{middleware-KynLg_RW.d.ts → middleware-C7PdmKF7.d.ts} +4 -4
- package/dist/model-call-options-CZkSw_xN.d.ts +6 -0
- package/dist/model-listing/index.d.ts +3 -1
- package/dist/observability/index.d.ts +19 -5
- package/dist/observability/index.js +8 -5
- package/dist/observability/index.js.map +1 -1
- package/dist/pipeline/index.d.ts +105 -56
- package/dist/pipeline/index.js +422 -230
- package/dist/pipeline/index.js.map +1 -1
- package/dist/{retry-D3Ruy-ba.d.ts → retry-CjvSlKGW.d.ts} +2 -1
- package/dist/skills/index.d.ts +5 -4
- package/dist/skills/index.js +7 -7
- package/dist/speech-generation/index.d.ts +37 -0
- package/dist/speech-generation/index.js +8 -0
- package/dist/{think-tool-DZqZj2LV.d.ts → think-tool-6H2QNQzQ.d.ts} +1 -1
- package/dist/tool/index.d.ts +9 -7
- package/dist/tool/index.js +7 -10
- package/dist/{tool-C_8GudXn.d.ts → tool--Mz4v1eL.d.ts} +15 -7
- package/dist/transcription/index.d.ts +23 -14
- package/dist/transcription/index.js +2 -2
- package/dist/{index-DxB74ybJ.d.ts → types-3x3mOLHU.d.ts} +94 -64
- package/dist/types-BVzM4RsC.d.ts +155 -0
- package/dist/types-CQ_ioXWK.d.ts +319 -0
- package/dist/types-Cr4uiYo5.d.ts +99 -0
- package/dist/{types-DhQfwiZn.d.ts → types-DiJqJejp.d.ts} +1 -1
- package/dist/types-Dld1TpWj.d.ts +130 -0
- package/dist/vector-store/index.d.ts +33 -34
- package/dist/vector-store/index.js +6 -6
- package/package.json +6 -10
- package/dist/agent-DGYY_onS.d.ts +0 -145
- package/dist/audio-generation/index.d.ts +0 -31
- package/dist/audio-generation/index.js +0 -8
- package/dist/chunk-5XKDW36Z.js.map +0 -1
- package/dist/chunk-6WGGTJE6.js.map +0 -1
- package/dist/chunk-A3UBYTJT.js.map +0 -1
- package/dist/chunk-ADH7NNCS.js +0 -512
- package/dist/chunk-ADH7NNCS.js.map +0 -1
- package/dist/chunk-DZ3CKYTH.js +0 -267
- package/dist/chunk-DZ3CKYTH.js.map +0 -1
- package/dist/chunk-FNQB2JEH.js.map +0 -1
- package/dist/chunk-I6G7BC42.js.map +0 -1
- package/dist/chunk-IL3ZSNJI.js.map +0 -1
- package/dist/chunk-J6GNQART.js +0 -122
- package/dist/chunk-J6GNQART.js.map +0 -1
- package/dist/chunk-M3IOWB4Y.js +0 -280
- package/dist/chunk-M3IOWB4Y.js.map +0 -1
- package/dist/chunk-QY5GZ7HR.js +0 -39
- package/dist/chunk-QY5GZ7HR.js.map +0 -1
- package/dist/chunk-R2LJSQUW.js.map +0 -1
- package/dist/chunk-RFDVKCIE.js +0 -12
- package/dist/chunk-RFDVKCIE.js.map +0 -1
- package/dist/chunk-RMUBRRSK.js +0 -34
- package/dist/chunk-RMUBRRSK.js.map +0 -1
- package/dist/chunk-RXKHOYEH.js +0 -466
- package/dist/chunk-RXKHOYEH.js.map +0 -1
- package/dist/chunk-XDRSPSR7.js +0 -38
- package/dist/chunk-XDRSPSR7.js.map +0 -1
- package/dist/chunk-XWUC7CIT.js +0 -1
- package/dist/chunk-XWUC7CIT.js.map +0 -1
- package/dist/errors-CYKUcU89.d.ts +0 -17
- package/dist/json-Zo7K3xmt.d.ts +0 -40
- package/dist/types-B3A4YxaA.d.ts +0 -354
- package/dist/types-BBEGKB9m.d.ts +0 -119
- package/dist/types-DT3nEemY.d.ts +0 -79
- package/dist/types-DulRCYdB.d.ts +0 -70
- package/dist/types-p55Jc-sr.d.ts +0 -47
- package/dist/ui/index.d.ts +0 -107
- package/dist/ui/index.js +0 -10
- package/dist/ui/index.js.map +0 -1
- /package/dist/{chunk-DMPUP4R3.js.map → chunk-XKYYBABH.js.map} +0 -0
- /package/dist/{audio-generation → speech-generation}/index.js.map +0 -0
package/dist/evals/index.d.ts
CHANGED
|
@@ -1,22 +1,15 @@
|
|
|
1
|
-
import { U as Usage,
|
|
1
|
+
import { U as Usage, J as JsonObject, C as CompletionModel, M as Message } from '../types-CQ_ioXWK.js';
|
|
2
2
|
import { Z as ZodSchema } from '../zod-schema-C7F4clpm.js';
|
|
3
|
-
import {
|
|
4
|
-
import {
|
|
5
|
-
import
|
|
3
|
+
import { b as AgentApprovalRequiredResult, m as AgentResponse, o as AgentRunOptions, n as AgentResult } from '../types-3x3mOLHU.js';
|
|
4
|
+
import { E as EmbeddingModel } from '../types-Cr4uiYo5.js';
|
|
5
|
+
import '../model-call-options-CZkSw_xN.js';
|
|
6
6
|
import 'zod';
|
|
7
|
+
import '../retry-CjvSlKGW.js';
|
|
7
8
|
import '../guardrails/index.js';
|
|
8
9
|
import '../type-utils-CtHVDRn_.js';
|
|
9
|
-
import '../types-
|
|
10
|
-
import '../
|
|
11
|
-
import '../
|
|
12
|
-
import '../types-DulRCYdB.js';
|
|
13
|
-
import '@modelcontextprotocol/sdk/client/sse.js';
|
|
14
|
-
import '@modelcontextprotocol/sdk/client/stdio.js';
|
|
15
|
-
import '@modelcontextprotocol/sdk/client/streamableHttp.js';
|
|
16
|
-
import '../types-DhQfwiZn.js';
|
|
17
|
-
import '../dynamic-tools-i9woysFa.js';
|
|
18
|
-
import '../types-DT3nEemY.js';
|
|
19
|
-
import '../retry-D3Ruy-ba.js';
|
|
10
|
+
import '../types-BVzM4RsC.js';
|
|
11
|
+
import '../tool--Mz4v1eL.js';
|
|
12
|
+
import '../middleware-C7PdmKF7.js';
|
|
20
13
|
|
|
21
14
|
type EvalOutcome<Score = unknown> = {
|
|
22
15
|
outcome: "pass";
|
|
@@ -57,7 +50,8 @@ declare const EvalOutcome: {
|
|
|
57
50
|
}): EvalOutcome<Score>;
|
|
58
51
|
};
|
|
59
52
|
|
|
60
|
-
type EvalMetadata =
|
|
53
|
+
type EvalMetadata = JsonObject;
|
|
54
|
+
type EvalReporterErrorPolicy = "collect" | "throw";
|
|
61
55
|
type EvalRunOptions = {
|
|
62
56
|
id?: string | undefined;
|
|
63
57
|
datasetName?: string | undefined;
|
|
@@ -85,6 +79,7 @@ type EvalTurn = {
|
|
|
85
79
|
metadata?: EvalMetadata | undefined;
|
|
86
80
|
};
|
|
87
81
|
type EvalTraceRef = {
|
|
82
|
+
observer?: string | undefined;
|
|
88
83
|
traceId: string;
|
|
89
84
|
observationId?: string | undefined;
|
|
90
85
|
responseId?: string | undefined;
|
|
@@ -255,7 +250,7 @@ type RunEvalSuiteOptions<Input, Output, Expected = unknown, Metrics extends read
|
|
|
255
250
|
concurrency?: number | undefined;
|
|
256
251
|
trace?: EvalTraceSelector<NoInfer<Input>, NoInfer<Output>, NoInfer<Expected>> | undefined;
|
|
257
252
|
reporters?: readonly EvalReporter<NoInfer<Input>, NoInfer<Output>, NoInfer<Expected>>[] | undefined;
|
|
258
|
-
|
|
253
|
+
reporterErrorPolicy?: EvalReporterErrorPolicy | undefined;
|
|
259
254
|
targetUsage?: EvalTargetUsageSelector<NoInfer<Input>, NoInfer<Output>, NoInfer<Expected>> | undefined;
|
|
260
255
|
cost?: EvalCostOptions<NoInfer<Input>, NoInfer<Output>, NoInfer<Expected>> | undefined;
|
|
261
256
|
};
|
|
@@ -388,12 +383,33 @@ declare function knowledgeRetention<Input, Output, Expected = unknown, const Nam
|
|
|
388
383
|
}): EvalMetric<Input, Output, number, Expected, Name>;
|
|
389
384
|
type ConversationSource = EvalTurn[] | Message[];
|
|
390
385
|
|
|
391
|
-
type
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
386
|
+
type EvaluableAgent<Output> = {
|
|
387
|
+
generate(input: AgentRunOptions<Output>): Promise<AgentResult<Output>>;
|
|
388
|
+
};
|
|
389
|
+
type IsAny<Value> = 0 extends 1 & Value ? true : false;
|
|
390
|
+
type IsExactly<Left, Right> = IsAny<Left> extends true ? false : [Left] extends [Right] ? [Right] extends [Left] ? true : false : false;
|
|
391
|
+
type AgentEvalTargetOptions<Input, AgentOutput = string, Output = AgentResponse<AgentOutput>, Expected = unknown> = {
|
|
392
|
+
agent: EvaluableAgent<AgentOutput>;
|
|
393
|
+
request(args: {
|
|
394
|
+
input: Input;
|
|
395
|
+
testCase: EvalCase<Input, Expected>;
|
|
396
|
+
}): AgentRunOptions<AgentOutput> | Promise<AgentRunOptions<AgentOutput>>;
|
|
397
|
+
} & (IsExactly<Output, AgentResponse<AgentOutput>> extends true ? {
|
|
398
|
+
output?(args: {
|
|
399
|
+
response: AgentResponse<AgentOutput>;
|
|
400
|
+
testCase: EvalCase<Input, Expected>;
|
|
401
|
+
}): Output | Promise<Output>;
|
|
402
|
+
} : {
|
|
403
|
+
output(args: {
|
|
404
|
+
response: AgentResponse<AgentOutput>;
|
|
405
|
+
testCase: EvalCase<Input, Expected>;
|
|
406
|
+
}): Output | Promise<Output>;
|
|
407
|
+
});
|
|
408
|
+
declare class AgentEvalApprovalError extends Error {
|
|
409
|
+
readonly result: AgentApprovalRequiredResult;
|
|
410
|
+
constructor(result: AgentApprovalRequiredResult);
|
|
411
|
+
}
|
|
412
|
+
declare function agentEvalTarget<Input, AgentOutput = string, Output = AgentResponse<AgentOutput>, Expected = unknown>(options: AgentEvalTargetOptions<Input, AgentOutput, Output, Expected>): EvalTarget<Input, Output, Expected>;
|
|
397
413
|
|
|
398
414
|
type EvalOutputFormat = "pretty" | "json" | "quiet";
|
|
399
415
|
type EvalExpectedTotals = Partial<EvalTotals> & {
|
|
@@ -569,6 +585,10 @@ declare function resolveEvalTraceRef(args: {
|
|
|
569
585
|
}): EvalTraceRef | undefined;
|
|
570
586
|
declare function defaultEvalTraceSelector<Input, Output, Expected>(args: EvalTraceSelectorArgs<Input, Output, Expected>): EvalTraceRef | undefined;
|
|
571
587
|
|
|
588
|
+
declare class EvalReporterDispatchError extends AggregateError {
|
|
589
|
+
readonly phase: string;
|
|
590
|
+
constructor(phase: string, errors: readonly unknown[]);
|
|
591
|
+
}
|
|
572
592
|
declare function runEvalSuite<Input, Output, Expected = unknown, const Metrics extends readonly EvalMetric<NoInfer<Input>, NoInfer<Output>, unknown, NoInfer<Expected>, string>[] = readonly EvalMetric<NoInfer<Input>, NoInfer<Output>, unknown, NoInfer<Expected>, string>[]>(options: RunEvalSuiteOptions<Input, Output, Expected, Metrics>): Promise<EvalSuiteResult<Input, Output, Expected, Metrics>>;
|
|
573
593
|
|
|
574
594
|
declare function selectPromptOutput(args: EvalMetricArgs<unknown, unknown, unknown>): string;
|
|
@@ -640,4 +660,4 @@ declare function defineEvalSuite<const Cases extends readonly EvalCaseLike[], co
|
|
|
640
660
|
target: Target;
|
|
641
661
|
};
|
|
642
662
|
|
|
643
|
-
export { type AbstentionCategory, type AbstentionOptions, type AgentEvalTargetOptions, type AnswerRelevancyOptions, type AnyEvalMetric, type ContainsAllOptions, type ContainsAnyOptions, type ContainsListOptions, type ContainsOptions, type DefaultEvalActual, type DefinedEvalSuite, type DoesNotMatchOptions, EvalAssertionError, type EvalCase, type EvalCaseRequirements, type EvalCaseResult, type EvalCasesExpected, type EvalCasesForMetrics, type EvalCasesInput, type EvalCostCalculatorArgs, type EvalCostOptions, type EvalCostSummary, type EvalDataType, type EvalExpectations, type EvalExpectedOutcomes, type EvalExpectedTotals, type EvalMetadata, type EvalMetric, type EvalMetricArgs, type EvalMetricDescriptor, type EvalMetricResult, type EvalMetricResultFor, type EvalMetricScore, EvalOutcome, type EvalOutcomeStatus, type EvalOutputFormat, type EvalOutputWriters, type EvalReportArgs, type EvalReporter, type EvalRunContext, type EvalRunEndArgs, type EvalRunOptions, type EvalRunStartArgs, type EvalScoreDirection, type EvalScoreMap, type EvalScoreProjection, type EvalSuiteResult, type EvalTarget, type EvalTargetUsageSelector, type EvalTotals, type EvalTraceCarrier, type EvalTraceRef, type EvalTraceSelector, type EvalTraceSelectorArgs, type EvalTurn, type EvalUsageSummary, type ExactMatchOptions, type FaithfulnessOptions, type GEvalOptions, type GEvalParameter, type GEvalRubric, type HallucinationOptions, type JsonCorrectnessOptions, type KnowledgeRetentionOptions, type LlmJudgeOptions, type LlmScoreMetricScore, type LlmScoreOptions, type MatchesOptions, type MaxLengthOptions, type NotContainsOptions, type PrintEvalResultOptions, type PromptAlignmentOptions, type RequiredFieldsOptions, type RunEvalCliOptions, type RunEvalSuiteOptions, type SelectorOrValue, type SemanticSimilarityOptions, type SummarizationOptions, type TurnRelevancyOptions, type ValueSelector, abstention, agentEvalTarget, answerRelevancy, assertEvalOutcomes, assertEvalTotals, contains, containsAll, containsAny, defaultEvalTraceSelector, defineEvalCases, defineEvalSuite, defineMetric, doesNotMatch, evalExitCode, exactMatch, faithfulness, gEval, hallucination, jsonCorrectness, knowledgeRetention, llmJudge, llmScore, matches, maxLength, notContains, printEvalResult, projectEvalOutcome, promptAlignment, requiredFields, resolveEvalTraceRef, runEvalCli, runEvalSuite, selectPromptOutput, semanticSimilarity, summarization, turnRelevancy };
|
|
663
|
+
export { type AbstentionCategory, type AbstentionOptions, AgentEvalApprovalError, type AgentEvalTargetOptions, type AnswerRelevancyOptions, type AnyEvalMetric, type ContainsAllOptions, type ContainsAnyOptions, type ContainsListOptions, type ContainsOptions, type DefaultEvalActual, type DefinedEvalSuite, type DoesNotMatchOptions, EvalAssertionError, type EvalCase, type EvalCaseRequirements, type EvalCaseResult, type EvalCasesExpected, type EvalCasesForMetrics, type EvalCasesInput, type EvalCostCalculatorArgs, type EvalCostOptions, type EvalCostSummary, type EvalDataType, type EvalExpectations, type EvalExpectedOutcomes, type EvalExpectedTotals, type EvalMetadata, type EvalMetric, type EvalMetricArgs, type EvalMetricDescriptor, type EvalMetricResult, type EvalMetricResultFor, type EvalMetricScore, EvalOutcome, type EvalOutcomeStatus, type EvalOutputFormat, type EvalOutputWriters, type EvalReportArgs, type EvalReporter, EvalReporterDispatchError, type EvalReporterErrorPolicy, type EvalRunContext, type EvalRunEndArgs, type EvalRunOptions, type EvalRunStartArgs, type EvalScoreDirection, type EvalScoreMap, type EvalScoreProjection, type EvalSuiteResult, type EvalTarget, type EvalTargetUsageSelector, type EvalTotals, type EvalTraceCarrier, type EvalTraceRef, type EvalTraceSelector, type EvalTraceSelectorArgs, type EvalTurn, type EvalUsageSummary, type ExactMatchOptions, type FaithfulnessOptions, type GEvalOptions, type GEvalParameter, type GEvalRubric, type HallucinationOptions, type JsonCorrectnessOptions, type KnowledgeRetentionOptions, type LlmJudgeOptions, type LlmScoreMetricScore, type LlmScoreOptions, type MatchesOptions, type MaxLengthOptions, type NotContainsOptions, type PrintEvalResultOptions, type PromptAlignmentOptions, type RequiredFieldsOptions, type RunEvalCliOptions, type RunEvalSuiteOptions, type SelectorOrValue, type SemanticSimilarityOptions, type SummarizationOptions, type TurnRelevancyOptions, type ValueSelector, abstention, agentEvalTarget, answerRelevancy, assertEvalOutcomes, assertEvalTotals, contains, containsAll, containsAny, defaultEvalTraceSelector, defineEvalCases, defineEvalSuite, defineMetric, doesNotMatch, evalExitCode, exactMatch, faithfulness, gEval, hallucination, jsonCorrectness, knowledgeRetention, llmJudge, llmScore, matches, maxLength, notContains, printEvalResult, projectEvalOutcome, promptAlignment, requiredFields, resolveEvalTraceRef, runEvalCli, runEvalSuite, selectPromptOutput, semanticSimilarity, summarization, turnRelevancy };
|
package/dist/evals/index.js
CHANGED
|
@@ -1,33 +1,36 @@
|
|
|
1
1
|
import {
|
|
2
|
+
AgentRunBlockedError,
|
|
2
3
|
cancelAgentApproval
|
|
3
|
-
} from "../chunk-
|
|
4
|
+
} from "../chunk-DBSCKTUL.js";
|
|
4
5
|
import "../chunk-YK4WAAS4.js";
|
|
5
|
-
import "../chunk-
|
|
6
|
-
import "../chunk-
|
|
7
|
-
import "../chunk-
|
|
6
|
+
import "../chunk-EINTGEGY.js";
|
|
7
|
+
import "../chunk-CQNNSZPG.js";
|
|
8
|
+
import "../chunk-D37WMSP6.js";
|
|
9
|
+
import "../chunk-77AIYARB.js";
|
|
10
|
+
import "../chunk-OJH6HWHX.js";
|
|
11
|
+
import "../chunk-CWM7B2OO.js";
|
|
8
12
|
import {
|
|
9
13
|
cosineSimilarity,
|
|
10
14
|
embedText,
|
|
11
15
|
mapWithConcurrency
|
|
12
|
-
} from "../chunk-
|
|
16
|
+
} from "../chunk-FMHLROPQ.js";
|
|
13
17
|
import {
|
|
14
|
-
|
|
15
|
-
} from "../chunk-
|
|
16
|
-
import "../chunk-
|
|
17
|
-
import "../chunk-M3IOWB4Y.js";
|
|
18
|
-
import "../chunk-RFDVKCIE.js";
|
|
18
|
+
extract
|
|
19
|
+
} from "../chunk-4WI3VGZK.js";
|
|
20
|
+
import "../chunk-MGNTXGTX.js";
|
|
19
21
|
import {
|
|
20
22
|
Usage
|
|
21
|
-
} from "../chunk-
|
|
22
|
-
import "../chunk-
|
|
23
|
-
import "../chunk-
|
|
23
|
+
} from "../chunk-KR3RCCHJ.js";
|
|
24
|
+
import "../chunk-TIZYTPJO.js";
|
|
25
|
+
import "../chunk-KIDGIJM7.js";
|
|
26
|
+
import "../chunk-FN7HLLAZ.js";
|
|
24
27
|
|
|
25
28
|
// src/evals/advanced-metrics.ts
|
|
26
29
|
import { z } from "zod";
|
|
27
30
|
|
|
28
31
|
// src/evals/format.ts
|
|
29
32
|
function defaultOutputValue(output) {
|
|
30
|
-
if (typeof output === "object" && output !== null && "output" in output
|
|
33
|
+
if (typeof output === "object" && output !== null && "output" in output) {
|
|
31
34
|
return output.output;
|
|
32
35
|
}
|
|
33
36
|
return output;
|
|
@@ -54,16 +57,15 @@ function errorMessage(error) {
|
|
|
54
57
|
|
|
55
58
|
// src/evals/judge.ts
|
|
56
59
|
async function runJudge(args) {
|
|
57
|
-
const
|
|
60
|
+
const result = await extract({
|
|
58
61
|
model: args.model,
|
|
59
62
|
outputSchema: args.schema,
|
|
60
|
-
instructions: args.instructions
|
|
61
|
-
|
|
62
|
-
const result = await extractor.extractResult(args.prompt, {
|
|
63
|
+
instructions: args.instructions,
|
|
64
|
+
text: args.prompt,
|
|
63
65
|
temperature: 0,
|
|
64
66
|
retries: args.retries <= 0 ? void 0 : { maxAttempts: Math.trunc(args.retries) + 1 }
|
|
65
67
|
});
|
|
66
|
-
return { data: result.
|
|
68
|
+
return { data: result.output, usage: result.usage };
|
|
67
69
|
}
|
|
68
70
|
function addUsage(...values) {
|
|
69
71
|
return values.reduce((total, usage) => Usage.add(total, usage), Usage.empty());
|
|
@@ -1120,15 +1122,24 @@ function contentText(content) {
|
|
|
1120
1122
|
}
|
|
1121
1123
|
|
|
1122
1124
|
// src/evals/agent-target.ts
|
|
1123
|
-
|
|
1125
|
+
var AgentEvalApprovalError = class extends Error {
|
|
1126
|
+
constructor(result) {
|
|
1127
|
+
super("Agent eval targets cannot suspend for tool approval.");
|
|
1128
|
+
this.result = result;
|
|
1129
|
+
this.name = "AgentEvalApprovalError";
|
|
1130
|
+
}
|
|
1131
|
+
result;
|
|
1132
|
+
};
|
|
1133
|
+
function agentEvalTarget(options) {
|
|
1124
1134
|
return async (input, testCase) => {
|
|
1125
|
-
const
|
|
1126
|
-
const response = await agent.generate(
|
|
1135
|
+
const request = await options.request({ input, testCase });
|
|
1136
|
+
const response = await options.agent.generate(request);
|
|
1127
1137
|
if (response.status === "approval_required") {
|
|
1128
1138
|
await cancelAgentApproval(response, "Agent eval targets cannot suspend for tool approval.");
|
|
1129
|
-
throw new
|
|
1139
|
+
throw new AgentEvalApprovalError(response);
|
|
1130
1140
|
}
|
|
1131
|
-
|
|
1141
|
+
if (response.status === "blocked") throw new AgentRunBlockedError(response);
|
|
1142
|
+
return options.output === void 0 ? response : await options.output({ response, testCase });
|
|
1132
1143
|
};
|
|
1133
1144
|
}
|
|
1134
1145
|
|
|
@@ -1193,6 +1204,7 @@ function traceFromCarrier(value) {
|
|
|
1193
1204
|
function traceFromMetadata(metadata) {
|
|
1194
1205
|
if (metadata === void 0) return void 0;
|
|
1195
1206
|
return readTraceRef({
|
|
1207
|
+
observer: metadata.traceObserver,
|
|
1196
1208
|
traceId: metadata.traceId,
|
|
1197
1209
|
observationId: metadata.observationId,
|
|
1198
1210
|
responseId: metadata.responseId
|
|
@@ -1202,9 +1214,11 @@ function readTraceRef(value) {
|
|
|
1202
1214
|
if (typeof value !== "object" || value === null) return void 0;
|
|
1203
1215
|
const traceId = value.traceId;
|
|
1204
1216
|
if (typeof traceId !== "string" || traceId.length === 0) return void 0;
|
|
1217
|
+
const observer = value.observer;
|
|
1205
1218
|
const observationId = value.observationId;
|
|
1206
1219
|
const responseId = value.responseId;
|
|
1207
1220
|
const trace = { traceId };
|
|
1221
|
+
if (typeof observer === "string" && observer.length > 0) trace.observer = observer;
|
|
1208
1222
|
if (typeof observationId === "string" && observationId.length > 0) {
|
|
1209
1223
|
trace.observationId = observationId;
|
|
1210
1224
|
}
|
|
@@ -1213,6 +1227,14 @@ function readTraceRef(value) {
|
|
|
1213
1227
|
}
|
|
1214
1228
|
|
|
1215
1229
|
// src/evals/runner.ts
|
|
1230
|
+
var EvalReporterDispatchError = class extends AggregateError {
|
|
1231
|
+
phase;
|
|
1232
|
+
constructor(phase, errors) {
|
|
1233
|
+
super(errors, `Evaluation reporter ${phase} failed ${errors.length} time(s).`);
|
|
1234
|
+
this.name = "EvalReporterDispatchError";
|
|
1235
|
+
this.phase = phase;
|
|
1236
|
+
}
|
|
1237
|
+
};
|
|
1216
1238
|
async function runEvalSuite(options) {
|
|
1217
1239
|
validateSuiteOptions(options);
|
|
1218
1240
|
const startedAtMs = Date.now();
|
|
@@ -1229,7 +1251,7 @@ async function runEvalSuite(options) {
|
|
|
1229
1251
|
reporterErrors = await notifyRunStart(
|
|
1230
1252
|
reporters,
|
|
1231
1253
|
lifecycle,
|
|
1232
|
-
options.
|
|
1254
|
+
options.reporterErrorPolicy ?? "collect"
|
|
1233
1255
|
);
|
|
1234
1256
|
} catch (error) {
|
|
1235
1257
|
await notifyRunEnd(reporters, {
|
|
@@ -1281,7 +1303,7 @@ async function runEvalSuite(options) {
|
|
|
1281
1303
|
usage: result.usage,
|
|
1282
1304
|
...result.cost === void 0 ? {} : { cost: result.cost }
|
|
1283
1305
|
},
|
|
1284
|
-
options.
|
|
1306
|
+
options.reporterErrorPolicy ?? "collect"
|
|
1285
1307
|
)
|
|
1286
1308
|
);
|
|
1287
1309
|
return result;
|
|
@@ -1335,7 +1357,7 @@ async function runEvalCase(options, testCase, run) {
|
|
|
1335
1357
|
trace: traceResult.trace,
|
|
1336
1358
|
traceError: traceResult.error,
|
|
1337
1359
|
reporters: options.reporters ?? [],
|
|
1338
|
-
|
|
1360
|
+
reporterErrorPolicy: options.reporterErrorPolicy ?? "collect"
|
|
1339
1361
|
});
|
|
1340
1362
|
const metricResult = {
|
|
1341
1363
|
metricName: metric.name,
|
|
@@ -1388,9 +1410,7 @@ async function safeEvaluate(suiteName, testCase, output, metric) {
|
|
|
1388
1410
|
async function reportOutcome(args) {
|
|
1389
1411
|
const errors = [];
|
|
1390
1412
|
if (args.traceError !== void 0) {
|
|
1391
|
-
if (args.failOnReporterError) throw args.traceError;
|
|
1392
1413
|
errors.push(args.traceError);
|
|
1393
|
-
return errors;
|
|
1394
1414
|
}
|
|
1395
1415
|
for (const reporter of args.reporters) {
|
|
1396
1416
|
try {
|
|
@@ -1405,12 +1425,10 @@ async function reportOutcome(args) {
|
|
|
1405
1425
|
outcome: args.outcome
|
|
1406
1426
|
});
|
|
1407
1427
|
} catch (error) {
|
|
1408
|
-
if (args.failOnReporterError) {
|
|
1409
|
-
throw error;
|
|
1410
|
-
}
|
|
1411
1428
|
errors.push(error);
|
|
1412
1429
|
}
|
|
1413
1430
|
}
|
|
1431
|
+
throwReporterErrors("report", errors, args.reporterErrorPolicy);
|
|
1414
1432
|
return errors;
|
|
1415
1433
|
}
|
|
1416
1434
|
function resolveRun(options, startedAtMs) {
|
|
@@ -1434,32 +1452,37 @@ function resolveRun(options, startedAtMs) {
|
|
|
1434
1452
|
...options.run?.metadata === void 0 ? {} : { metadata: options.run.metadata }
|
|
1435
1453
|
};
|
|
1436
1454
|
}
|
|
1437
|
-
async function notifyRunStart(reporters, args,
|
|
1455
|
+
async function notifyRunStart(reporters, args, errorPolicy) {
|
|
1438
1456
|
const errors = [];
|
|
1439
1457
|
for (const reporter of reporters) {
|
|
1440
1458
|
if (reporter.onRunStart === void 0) continue;
|
|
1441
1459
|
try {
|
|
1442
1460
|
await reporter.onRunStart(args);
|
|
1443
1461
|
} catch (error) {
|
|
1444
|
-
if (failOnReporterError) throw error;
|
|
1445
1462
|
errors.push(error);
|
|
1446
1463
|
}
|
|
1447
1464
|
}
|
|
1465
|
+
throwReporterErrors("onRunStart", errors, errorPolicy);
|
|
1448
1466
|
return errors;
|
|
1449
1467
|
}
|
|
1450
|
-
async function notifyRunEnd(reporters, args,
|
|
1468
|
+
async function notifyRunEnd(reporters, args, errorPolicy = "collect") {
|
|
1451
1469
|
const errors = [];
|
|
1452
1470
|
for (const reporter of reporters) {
|
|
1453
1471
|
if (reporter.onRunEnd === void 0) continue;
|
|
1454
1472
|
try {
|
|
1455
1473
|
await reporter.onRunEnd(args);
|
|
1456
1474
|
} catch (error) {
|
|
1457
|
-
if (failOnReporterError) throw error;
|
|
1458
1475
|
errors.push(error);
|
|
1459
1476
|
}
|
|
1460
1477
|
}
|
|
1478
|
+
throwReporterErrors("onRunEnd", errors, errorPolicy);
|
|
1461
1479
|
return errors;
|
|
1462
1480
|
}
|
|
1481
|
+
function throwReporterErrors(phase, errors, errorPolicy) {
|
|
1482
|
+
if (errorPolicy === "throw" && errors.length > 0) {
|
|
1483
|
+
throw new EvalReporterDispatchError(phase, errors);
|
|
1484
|
+
}
|
|
1485
|
+
}
|
|
1463
1486
|
function countMetricOutcomes(results) {
|
|
1464
1487
|
const totals = emptyTotals();
|
|
1465
1488
|
for (const result of results) {
|
|
@@ -2004,9 +2027,9 @@ function semanticSimilarity(options) {
|
|
|
2004
2027
|
if (typeof expected !== "string") {
|
|
2005
2028
|
return EvalOutcome.invalid("Semantic similarity expected value must be a string.");
|
|
2006
2029
|
}
|
|
2007
|
-
const [actualEmbedding, expectedEmbedding] = await Promise.all([
|
|
2008
|
-
embedText(options.model, actual),
|
|
2009
|
-
embedText(options.model, expected)
|
|
2030
|
+
const [{ embedding: actualEmbedding }, { embedding: expectedEmbedding }] = await Promise.all([
|
|
2031
|
+
embedText({ model: options.model, text: actual }),
|
|
2032
|
+
embedText({ model: options.model, text: expected })
|
|
2010
2033
|
]);
|
|
2011
2034
|
const score = cosineSimilarity(actualEmbedding.vector, expectedEmbedding.vector);
|
|
2012
2035
|
return score >= options.threshold ? EvalOutcome.pass(score) : EvalOutcome.fail(score, { comment: `Similarity below threshold ${options.threshold}.` });
|
|
@@ -2014,23 +2037,19 @@ function semanticSimilarity(options) {
|
|
|
2014
2037
|
};
|
|
2015
2038
|
}
|
|
2016
2039
|
function llmJudge(options) {
|
|
2017
|
-
const extractor = new Extractor({
|
|
2018
|
-
model: options.model,
|
|
2019
|
-
outputSchema: options.schema,
|
|
2020
|
-
instructions: options.instructions ?? "Judge the eval case by the requested schema. Submit the judgment using the schema."
|
|
2021
|
-
});
|
|
2022
2040
|
return {
|
|
2023
2041
|
name: options.name ?? "llm_judge",
|
|
2024
2042
|
required: options.required ?? true,
|
|
2025
2043
|
async evaluate(args) {
|
|
2026
2044
|
try {
|
|
2027
|
-
const result = await
|
|
2028
|
-
|
|
2029
|
-
|
|
2030
|
-
|
|
2031
|
-
|
|
2032
|
-
|
|
2033
|
-
|
|
2045
|
+
const result = await extract({
|
|
2046
|
+
model: options.model,
|
|
2047
|
+
outputSchema: options.schema,
|
|
2048
|
+
instructions: options.instructions ?? "Judge the eval case by the requested schema. Submit the judgment using the schema.",
|
|
2049
|
+
text: await resolveJudgePrompt(options.prompt, args),
|
|
2050
|
+
retries: evalExtractionRetries(options.retries)
|
|
2051
|
+
});
|
|
2052
|
+
return options.passes(result.output) ? EvalOutcome.pass(result.output, { usage: result.usage }) : EvalOutcome.fail(result.output, { usage: result.usage });
|
|
2034
2053
|
} catch (error) {
|
|
2035
2054
|
return EvalOutcome.invalid(errorMessage(error));
|
|
2036
2055
|
}
|
|
@@ -2039,17 +2058,6 @@ function llmJudge(options) {
|
|
|
2039
2058
|
}
|
|
2040
2059
|
function llmScore(options) {
|
|
2041
2060
|
const criteria = Array.isArray(options.criteria) ? options.criteria.join("\n") : options.criteria;
|
|
2042
|
-
const extractor = new Extractor({
|
|
2043
|
-
model: options.model,
|
|
2044
|
-
outputSchema: z2.object({
|
|
2045
|
-
score: z2.number(),
|
|
2046
|
-
feedback: z2.string()
|
|
2047
|
-
}),
|
|
2048
|
-
instructions: options.instructions ?? `Score the eval case against these criteria:
|
|
2049
|
-
${criteria}
|
|
2050
|
-
|
|
2051
|
-
Return a score between 0 and 1 and brief feedback.`
|
|
2052
|
-
});
|
|
2053
2061
|
return {
|
|
2054
2062
|
name: options.name ?? "llm_score",
|
|
2055
2063
|
required: options.required ?? true,
|
|
@@ -2059,13 +2067,20 @@ Return a score between 0 and 1 and brief feedback.`
|
|
|
2059
2067
|
threshold: options.threshold,
|
|
2060
2068
|
async evaluate(args) {
|
|
2061
2069
|
try {
|
|
2062
|
-
const result = await
|
|
2063
|
-
|
|
2064
|
-
{
|
|
2065
|
-
|
|
2066
|
-
|
|
2067
|
-
|
|
2068
|
-
|
|
2070
|
+
const result = await extract({
|
|
2071
|
+
model: options.model,
|
|
2072
|
+
outputSchema: z2.object({
|
|
2073
|
+
score: z2.number(),
|
|
2074
|
+
feedback: z2.string()
|
|
2075
|
+
}),
|
|
2076
|
+
instructions: options.instructions ?? `Score the eval case against these criteria:
|
|
2077
|
+
${criteria}
|
|
2078
|
+
|
|
2079
|
+
Return a score between 0 and 1 and brief feedback.`,
|
|
2080
|
+
text: await resolveJudgePrompt(options.prompt, args),
|
|
2081
|
+
retries: evalExtractionRetries(options.retries)
|
|
2082
|
+
});
|
|
2083
|
+
const score = result.output;
|
|
2069
2084
|
if (score.score < 0 || score.score > 1) {
|
|
2070
2085
|
return EvalOutcome.invalid(`Score ${score.score} outside valid range [0, 1].`, {
|
|
2071
2086
|
score,
|
|
@@ -2100,8 +2115,10 @@ function defineEvalSuite(options) {
|
|
|
2100
2115
|
};
|
|
2101
2116
|
}
|
|
2102
2117
|
export {
|
|
2118
|
+
AgentEvalApprovalError,
|
|
2103
2119
|
EvalAssertionError,
|
|
2104
2120
|
EvalOutcome,
|
|
2121
|
+
EvalReporterDispatchError,
|
|
2105
2122
|
abstention,
|
|
2106
2123
|
agentEvalTarget,
|
|
2107
2124
|
answerRelevancy,
|